{
 "built": "2026-09-17",
 "facts": {
  "ci_anchor.advisor": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": "#advisor-strategy-escalate-hard-decisions",
   "verified": "2026-09-16"
  },
  "ci_anchor.batch": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#batch-work-that-can-wait",
   "value": "#batch-work-that-can-wait",
   "verified": "2026-09-16"
  },
  "ci_anchor.benchmarks": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": "#benchmarks-referenced",
   "verified": "2026-09-16"
  },
  "ci_anchor.budgets": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": "#set-budgets-and-output-caps",
   "verified": "2026-09-16"
  },
  "ci_anchor.cache": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#cache-repeated-context",
   "value": "#cache-repeated-context",
   "verified": "2026-09-16"
  },
  "ci_anchor.cache_breaks": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#what-breaks-the-cache",
   "value": "#what-breaks-the-cache",
   "verified": "2026-09-16"
  },
  "ci_anchor.cache_duration": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": "#pick-the-cache-duration",
   "verified": "2026-09-16"
  },
  "ci_anchor.compare_models": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": "#compare-models-on-cost-per-task",
   "verified": "2026-09-16"
  },
  "ci_anchor.context_lifecycle": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#manage-the-context-lifecycle",
   "value": "#manage-the-context-lifecycle",
   "verified": "2026-09-16"
  },
  "ci_anchor.data_files": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#keep-data-files-out-of-the-prompt",
   "value": "#keep-data-files-out-of-the-prompt",
   "verified": "2026-09-16"
  },
  "ci_anchor.effort": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": "#tune-effort",
   "verified": "2026-09-16"
  },
  "ci_anchor.measure": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#measure-on-your-own-workload",
   "value": "#measure-on-your-own-workload",
   "verified": "2026-09-16"
  },
  "ci_anchor.orchestrator": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": "#orchestrator-strategy-delegate-bulk-work",
   "verified": "2026-09-16"
  },
  "ci_anchor.page": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence",
   "value": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence",
   "verified": "2026-09-16"
  },
  "ci_anchor.pricing_page": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": "https://platform.claude.com/docs/en/about-claude/pricing",
   "verified": "2026-09-16"
  },
  "ci_anchor.prompt_audit": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#audit-prompts-against-the-current-model",
   "value": "#audit-prompts-against-the-current-model",
   "verified": "2026-09-16"
  },
  "ci_anchor.rerun": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#re-run-failures-at-higher-effort",
   "value": "#re-run-failures-at-higher-effort",
   "verified": "2026-09-16"
  },
  "ci_anchor.start_here": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#start-here",
   "value": "#start-here",
   "verified": "2026-09-16"
  },
  "ci_anchor.tool_search": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#defer-unused-tool-definitions",
   "value": "#defer-unused-tool-definitions",
   "verified": "2026-09-16"
  },
  "ci_anchor.trim": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#trim-input-and-context-tokens",
   "value": "#trim-input-and-context-tokens",
   "verified": "2026-09-16"
  },
  "ci_anchor.upgrade": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": "#upgrade-the-model",
   "verified": "2026-09-16"
  },
  "ci_assume.advisor_answer_length": {
   "note": "Assumed: the page does not state this number. It is the engine's modeling constant, fitted so the calibration anchors at this source hold; change it and rerun the engine tests.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": 800,
   "verified": "2026-09-16"
  },
  "ci_assume.compaction_summary_size": {
   "note": "Assumed: the page does not state this number. It is the engine's modeling constant, fitted so the calibration anchors at this source hold; change it and rerun the engine tests.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#manage-the-context-lifecycle",
   "value": 2000,
   "verified": "2026-09-16"
  },
  "ci_assume.compaction_threshold": {
   "note": "Assumed: the page does not state this number. It is the engine's modeling constant, fitted so the calibration anchors at this source hold; change it and rerun the engine tests.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#manage-the-context-lifecycle",
   "value": 80000,
   "verified": "2026-09-16"
  },
  "ci_assume.context_editing_threshold": {
   "note": "Assumed: the page does not state this number. It is the engine's modeling constant, fitted so the calibration anchors at this source hold; change it and rerun the engine tests.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#manage-the-context-lifecycle",
   "value": 6000,
   "verified": "2026-09-16"
  },
  "ci_assume.data_file_pointer_size": {
   "note": "Assumed: the page does not state this number. It is the engine's modeling constant, fitted so the calibration anchors at this source hold; change it and rerun the engine tests.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#keep-data-files-out-of-the-prompt",
   "value": 500,
   "verified": "2026-09-16"
  },
  "ci_assume.effort_output_exponent": {
   "note": "Assumed: see effort_turns_exponent; the two sum to 1.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.3,
   "verified": "2026-09-16"
  },
  "ci_assume.effort_turns_exponent": {
   "note": "Assumed split of an effort level's measured cost fraction: this exponent goes to fewer turns, the rest to shorter outputs. The two exponents sum to 1 so the per-task fraction stays what the page measured.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.7,
   "verified": "2026-09-16"
  },
  "ci_assume.orchestrator_overhead_share": {
   "note": "Assumed: the page does not state this number. It is the engine's modeling constant, fitted so the calibration anchors at this source hold; change it and rerun the engine tests.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 0.1,
   "verified": "2026-09-16"
  },
  "ci_assume.pruned_tool_result_size": {
   "note": "Assumed: the page does not state this number. It is the engine's modeling constant, fitted so the calibration anchors at this source hold; change it and rerun the engine tests.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#manage-the-context-lifecycle",
   "value": 40,
   "verified": "2026-09-16"
  },
  "ci_assume.tool_search_overhead": {
   "note": "Assumed: the page does not state this number. It is the engine's modeling constant, fitted so the calibration anchors at this source hold; change it and rerun the engine tests.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#defer-unused-tool-definitions",
   "value": 300,
   "verified": "2026-09-16"
  },
  "ci_bench.chart_fable_low_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.15,
   "verified": "2026-09-16"
  },
  "ci_bench.chart_fable_low_score": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 62.5,
   "verified": "2026-09-16"
  },
  "ci_bench.chart_opus_low_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.38,
   "verified": "2026-09-16"
  },
  "ci_bench.chart_opus_low_score": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 49,
   "verified": "2026-09-16"
  },
  "ci_bench.chartography_questions_phrase": {
   "number": 100.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": "100-question set",
   "verified": "2026-09-16"
  },
  "ci_bench.coding_fable_medium_cost_per_attempt": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 2.91,
   "verified": "2026-09-16"
  },
  "ci_bench.coding_opus_default_cost_per_attempt": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 8.5,
   "verified": "2026-09-16"
  },
  "ci_bench.corpus_solo_cost_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 552,
   "verified": "2026-09-16"
  },
  "ci_bench.corpus_solo_cost_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 468,
   "verified": "2026-09-16"
  },
  "ci_bench.corpus_tokens_phrase": {
   "number": 21.6,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": "21.6-million-token corpus",
   "verified": "2026-09-16"
  },
  "ci_bench.corpus_workers_phrase": {
   "number": 25.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": "25 concurrent Claude Sonnet 5 workers",
   "verified": "2026-09-16"
  },
  "ci_bench.deepresearch_tasks_phrase": {
   "number": 132.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": "132 research tasks across 22 domains",
   "verified": "2026-09-16"
  },
  "ci_bench.fable_5_vs_5_1_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.43,
   "verified": "2026-09-16"
  },
  "ci_bench.gpqa_haiku_accuracy": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.63,
   "verified": "2026-09-16"
  },
  "ci_bench.gpqa_haiku_cost_fraction": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.1,
   "verified": "2026-09-16"
  },
  "ci_bench.gpqa_opus_accuracy": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.92,
   "verified": "2026-09-16"
  },
  "ci_bench.gpqa_questions_phrase": {
   "number": 198.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": "198-question Diamond subset",
   "verified": "2026-09-16"
  },
  "ci_bench.measurement_month": {
   "number": 2026.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": "August 2026",
   "verified": "2026-09-16"
  },
  "ci_bench.research_fable_low_score": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.66,
   "verified": "2026-09-16"
  },
  "ci_bench.research_sonnet_cost_per_task": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 1.2,
   "verified": "2026-09-16"
  },
  "ci_bench.research_sonnet_score": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.56,
   "verified": "2026-09-16"
  },
  "ci_bench.support_desk_tickets_phrase": {
   "number": 44.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": "44 support tickets",
   "verified": "2026-09-16"
  },
  "ci_bench.swe_fable_default_cost_per_solved": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 1.19,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_fable_default_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.921,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_fable_low_cost_per_solved": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.54,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_fable_low_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.886,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_opus_4_8_vs_5_cost_per_solved_extra": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.21,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_opus_4_8_vs_5_pass_delta": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.12,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_opus_5_low_vs_4_8_cost_fraction": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.3,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_opus_default_cost_per_solved": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 1.01,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_opus_default_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.917,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_opus_low_cost_per_solved": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.25,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_opus_low_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.84,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_sonnet_4_6_vs_5_pass_delta": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.05,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_sonnet_4_6_vs_5_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.15,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_sonnet_default_cost_per_solved": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.84,
   "verified": "2026-09-16"
  },
  "ci_bench.swe_sonnet_default_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.774,
   "verified": "2026-09-16"
  },
  "ci_bench.terminal_cost_per_solved_4_7": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 183,
   "verified": "2026-09-16"
  },
  "ci_bench.terminal_cost_per_solved_4_8": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 63,
   "verified": "2026-09-16"
  },
  "ci_bench.terminal_cost_per_solved_5": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 28,
   "verified": "2026-09-16"
  },
  "ci_bench.terminal_opus_4_7_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.07,
   "verified": "2026-09-16"
  },
  "ci_bench.terminal_opus_4_8_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.15,
   "verified": "2026-09-16"
  },
  "ci_bench.terminal_opus_5_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.41,
   "verified": "2026-09-16"
  },
  "ci_cache.agent_loop_factor_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#cache-repeated-context",
   "value": 5.3,
   "verified": "2026-09-16"
  },
  "ci_cache.agent_loop_factor_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#cache-repeated-context",
   "value": 2.7,
   "verified": "2026-09-16"
  },
  "ci_cache.change_after_compaction_session_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#what-breaks-the-cache",
   "value": 0.75,
   "verified": "2026-09-16"
  },
  "ci_cache.fable_keep_warm_saving_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": 0.2,
   "verified": "2026-09-16"
  },
  "ci_cache.fable_keep_warm_saving_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": 0.13,
   "verified": "2026-09-16"
  },
  "ci_cache.keep_warm_interval_phrase": {
   "number": 4.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": "keep-alive every 4 minutes",
   "verified": "2026-09-16"
  },
  "ci_cache.lifetime_1h_minutes": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": 60,
   "verified": "2026-09-16"
  },
  "ci_cache.lifetime_1h_name": {
   "number": 1.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": "1-hour cache lifetime",
   "verified": "2026-09-16"
  },
  "ci_cache.lifetime_5m_name": {
   "number": 5.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#cache-repeated-context",
   "value": "5-minute cache lifetime",
   "verified": "2026-09-16"
  },
  "ci_cache.mid_session_change_session_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#what-breaks-the-cache",
   "value": 0.95,
   "verified": "2026-09-16"
  },
  "ci_cache.mid_session_tokens_rewritten_effort": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#what-breaks-the-cache",
   "value": 39000,
   "verified": "2026-09-16"
  },
  "ci_cache.mid_session_tokens_rewritten_tool": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#what-breaks-the-cache",
   "value": 60000,
   "verified": "2026-09-16"
  },
  "ci_cache.no_change_session_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": 0.81,
   "verified": "2026-09-16"
  },
  "ci_cache.one_hour_crossover_measured": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": 0.033,
   "verified": "2026-09-16"
  },
  "ci_cache.one_hour_no_pause_extra_opus": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": 0.11,
   "verified": "2026-09-16"
  },
  "ci_cache.one_hour_no_pause_extra_sonnet": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": 0.15,
   "verified": "2026-09-16"
  },
  "ci_cache.one_hour_pause_share_rule": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#pick-the-cache-duration",
   "value": 0.05,
   "verified": "2026-09-16"
  },
  "ci_cache.production_read_share_median": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#cache-repeated-context",
   "value": 0.842,
   "verified": "2026-09-16"
  },
  "ci_cache.production_read_share_top_decile": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#cache-repeated-context",
   "value": 0.94,
   "verified": "2026-09-16"
  },
  "ci_cache.read_share_investigate_below": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#cache-repeated-context",
   "value": 0.8,
   "verified": "2026-09-16"
  },
  "ci_cache.status_line_run_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#what-breaks-the-cache",
   "value": 4.24,
   "verified": "2026-09-16"
  },
  "ci_cache.triage_issues_per_run_phrase": {
   "number": 20.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#trim-input-and-context-tokens",
   "value": "20 issues",
   "verified": "2026-09-16"
  },
  "ci_cache.triage_run_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#what-breaks-the-cache",
   "value": 0.59,
   "verified": "2026-09-16"
  },
  "ci_cache.triage_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#start-here",
   "value": 0.83,
   "verified": "2026-09-16"
  },
  "ci_cache.triage_saving_with_trimming": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#start-here",
   "value": 0.88,
   "verified": "2026-09-16"
  },
  "ci_cache.write_1h_multiplier": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#measure-on-your-own-workload",
   "value": 2.0,
   "verified": "2026-09-16"
  },
  "ci_cache.write_5m_multiplier": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#measure-on-your-own-workload",
   "value": 1.25,
   "verified": "2026-09-16"
  },
  "ci_effect.advisor_chart_fable_medium_score": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": 67.5,
   "verified": "2026-09-16"
  },
  "ci_effect.advisor_chart_pairing_score": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": 65.0,
   "verified": "2026-09-16"
  },
  "ci_effect.advisor_chart_price_multiple": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": 2.6,
   "verified": "2026-09-16"
  },
  "ci_effect.advisor_coding_cost_per_attempt": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": 7.69,
   "verified": "2026-09-16"
  },
  "ci_effect.advisor_coding_pass_delta_over_opus": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": 0.035,
   "verified": "2026-09-16"
  },
  "ci_effect.advisor_consults_per_attempt_phrase": {
   "number": 2.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": "about 2",
   "verified": "2026-09-16"
  },
  "ci_effect.advisor_deepswe_sonnet_low_gain": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": 0.23,
   "verified": "2026-09-16"
  },
  "ci_effect.advisor_gap_closed_at_least": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#advisor-strategy-escalate-hard-decisions",
   "value": 0.5,
   "verified": "2026-09-16"
  },
  "ci_effect.answer_memo_cost_multiple": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 2.8,
   "verified": "2026-09-16"
  },
  "ci_effect.answer_memo_output_multiple_phrase": {
   "number": 6.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": "6 times",
   "verified": "2026-09-16"
  },
  "ci_effect.answer_one_line_cost_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.14,
   "verified": "2026-09-16"
  },
  "ci_effect.answer_one_line_output_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.39,
   "verified": "2026-09-16"
  },
  "ci_effect.batch_discount": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#batch-work-that-can-wait",
   "value": 0.5,
   "verified": "2026-09-16"
  },
  "ci_effect.batch_window_phrase": {
   "number": 24.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#batch-work-that-can-wait",
   "value": "batch results within 24 hours",
   "verified": "2026-09-16"
  },
  "ci_effect.budget_beta_header": {
   "number": 2026.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": "task-budgets-2026-03-13",
   "verified": "2026-09-16"
  },
  "ci_effect.budget_floor_tokens": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 20000,
   "verified": "2026-09-16"
  },
  "ci_effect.budget_generous_pass_delta": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.03,
   "verified": "2026-09-16"
  },
  "ci_effect.budget_generous_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.44,
   "verified": "2026-09-16"
  },
  "ci_effect.budget_tight_pass_delta": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.06,
   "verified": "2026-09-16"
  },
  "ci_effect.budget_tight_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.58,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_128k_output_tokens": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 128000,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_16k_capped_still_passed_phrase": {
   "number": 9.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": "9 of 117",
   "verified": "2026-09-16"
  },
  "ci_effect.cap_16k_cost_per_solved_phrase": {
   "number": 21.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": "$21",
   "verified": "2026-09-16"
  },
  "ci_effect.cap_16k_ended_fable": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.43,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_16k_ended_haiku_estimate": {
   "note": "Estimate, not a measurement: the page capped Opus 5 (0.15) and Fable 5.1 (0.43); this value is assumed between them for a model the page did not cap. The engine reports it as estimated.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.3,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_16k_ended_opus": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.15,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_16k_ended_sonnet_estimate": {
   "note": "Estimate, not a measurement: the page capped Opus 5 (0.15) and Fable 5.1 (0.43); this value is assumed between them for a model the page did not cap. The engine reports it as estimated.",
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.25,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_16k_output_tokens": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 16000,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_64k_cost_per_solved_phrase": {
   "number": 22.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": "$22",
   "verified": "2026-09-16"
  },
  "ci_effect.cap_64k_cut_turns_phrase": {
   "number": 2.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": "2 of about 14,000",
   "verified": "2026-09-16"
  },
  "ci_effect.cap_64k_output_tokens": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 64000,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_fable_128k_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#benchmarks-referenced",
   "value": 0.6,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_fable_16k_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.363,
   "verified": "2026-09-16"
  },
  "ci_effect.cap_fable_64k_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#set-budgets-and-output-caps",
   "value": 0.585,
   "verified": "2026-09-16"
  },
  "ci_effect.compaction_saving_long": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#manage-the-context-lifecycle",
   "value": 0.32,
   "verified": "2026-09-16"
  },
  "ci_effect.context_editing_extra_short": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#manage-the-context-lifecycle",
   "value": 0.74,
   "verified": "2026-09-16"
  },
  "ci_effect.data_file_cost_fraction": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#keep-data-files-out-of-the-prompt",
   "value": 0.083,
   "verified": "2026-09-16"
  },
  "ci_effect.data_file_pasted_correct_phrase": {
   "number": 6.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#keep-data-files-out-of-the-prompt",
   "value": "6 of 25",
   "verified": "2026-09-16"
  },
  "ci_effect.data_file_pasted_tokens": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#keep-data-files-out-of-the-prompt",
   "value": 91000,
   "verified": "2026-09-16"
  },
  "ci_effect.data_file_uploaded_correct_phrase": {
   "number": 25.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#keep-data-files-out-of-the-prompt",
   "value": "25 of 25",
   "verified": "2026-09-16"
  },
  "ci_effect.effort_coding_low_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.25,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_coding_low_pass_delta": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.08,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_coding_medium_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.5,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_coding_medium_pass_delta": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.02,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_knowledge_low_cost_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.67,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_knowledge_low_cost_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.5,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_knowledge_low_pass_delta_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.03,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_knowledge_low_pass_delta_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.01,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_knowledge_medium_cost_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.87,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_knowledge_medium_cost_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 0.69,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_research_high_cost_per_task": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 7.12,
   "verified": "2026-09-16"
  },
  "ci_effect.effort_research_low_cost_per_task": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#tune-effort",
   "value": 4.66,
   "verified": "2026-09-16"
  },
  "ci_effect.long_run_token_multiple": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#trim-input-and-context-tokens",
   "value": 2.6,
   "verified": "2026-09-16"
  },
  "ci_effect.newer_tokenizer_extra": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#upgrade-the-model",
   "value": 0.3,
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_corpus_hours_phrase": {
   "number": 2.3,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": "about 2.3 hours against 15 to 20",
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_corpus_saving_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 0.55,
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_corpus_saving_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 0.47,
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_cost_fraction": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 0.5,
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_easy_slice_p90_phrase": {
   "number": 12.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": "$12 compared with $33",
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_full_set_solo_cheaper_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 0.3,
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_full_set_solo_cheaper_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 0.22,
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_pass_delta_below_high": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 0.12,
   "verified": "2026-09-16"
  },
  "ci_effect.orchestrator_pass_delta_below_low": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#orchestrator-strategy-delegate-bulk-work",
   "value": 0.1,
   "verified": "2026-09-16"
  },
  "ci_effect.prompt_audit_accuracy_after": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#audit-prompts-against-the-current-model",
   "value": 0.97,
   "verified": "2026-09-16"
  },
  "ci_effect.prompt_audit_accuracy_before": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#audit-prompts-against-the-current-model",
   "value": 0.92,
   "verified": "2026-09-16"
  },
  "ci_effect.prompt_audit_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#audit-prompts-against-the-current-model",
   "value": 0.14,
   "verified": "2026-09-16"
  },
  "ci_effect.prompt_stale_overspend": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#audit-prompts-against-the-current-model",
   "value": 0.36,
   "verified": "2026-09-16"
  },
  "ci_effect.prune_saving_long": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#manage-the-context-lifecycle",
   "value": 0.39,
   "verified": "2026-09-16"
  },
  "ci_effect.rerun_all_default_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#re-run-failures-at-higher-effort",
   "value": 0.93,
   "verified": "2026-09-16"
  },
  "ci_effect.rerun_all_default_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#re-run-failures-at-higher-effort",
   "value": 0.917,
   "verified": "2026-09-16"
  },
  "ci_effect.rerun_low_fail_share": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#re-run-failures-at-higher-effort",
   "value": 0.16,
   "verified": "2026-09-16"
  },
  "ci_effect.rerun_low_then_default_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#re-run-failures-at-higher-effort",
   "value": 0.45,
   "verified": "2026-09-16"
  },
  "ci_effect.rerun_low_then_default_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#re-run-failures-at-higher-effort",
   "value": 0.93,
   "verified": "2026-09-16"
  },
  "ci_effect.rerun_medium_then_default_cost": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#re-run-failures-at-higher-effort",
   "value": 0.61,
   "verified": "2026-09-16"
  },
  "ci_effect.rerun_medium_then_default_pass": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#re-run-failures-at-higher-effort",
   "value": 0.94,
   "verified": "2026-09-16"
  },
  "ci_effect.tail_problems_phrase": {
   "number": 2.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": "2 of 20",
   "verified": "2026-09-16"
  },
  "ci_effect.tail_spend_share": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#compare-models-on-cost-per-task",
   "value": 0.43,
   "verified": "2026-09-16"
  },
  "ci_effect.tool_catalog_accuracy_phrase": {
   "number": 15.0,
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#defer-unused-tool-definitions",
   "value": "15 to 18 of 20",
   "verified": "2026-09-16"
  },
  "ci_effect.tool_search_catalog_size": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#defer-unused-tool-definitions",
   "value": 502,
   "verified": "2026-09-16"
  },
  "ci_effect.tool_search_saving_502": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#defer-unused-tool-definitions",
   "value": 0.45,
   "verified": "2026-09-16"
  },
  "ci_effect.tool_search_saving_github_mcp": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#defer-unused-tool-definitions",
   "value": 0.2,
   "verified": "2026-09-16"
  },
  "ci_effect.trimming_further_points": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#start-here",
   "value": 0.05,
   "verified": "2026-09-16"
  },
  "ci_effect.trimming_long_run": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#trim-input-and-context-tokens",
   "value": 0.21,
   "verified": "2026-09-16"
  },
  "ci_effect.trimming_short_run": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#trim-input-and-context-tokens",
   "value": 0.26,
   "verified": "2026-09-16"
  },
  "ci_effect.verify_twice_saving": {
   "source": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence#audit-prompts-against-the-current-model",
   "value": 0.33,
   "verified": "2026-09-16"
  },
  "ci_price.fable_5_1_cache_read": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 0.25,
   "verified": "2026-09-16"
  },
  "ci_price.fable_5_1_input": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 10.0,
   "verified": "2026-09-16"
  },
  "ci_price.fable_5_1_output": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 50.0,
   "verified": "2026-09-16"
  },
  "ci_price.haiku_4_5_cache_read": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 0.1,
   "verified": "2026-09-16"
  },
  "ci_price.haiku_4_5_input": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 1.0,
   "verified": "2026-09-16"
  },
  "ci_price.haiku_4_5_output": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 5.0,
   "verified": "2026-09-16"
  },
  "ci_price.opus_5_cache_read": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 0.5,
   "verified": "2026-09-16"
  },
  "ci_price.opus_5_input": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 5.0,
   "verified": "2026-09-16"
  },
  "ci_price.opus_5_output": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 25.0,
   "verified": "2026-09-16"
  },
  "ci_price.per_tokens": {
   "note": "Every token price above is per this many tokens; the engine divides by it, so the unit lives beside the prices.",
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 1000000,
   "verified": "2026-09-16"
  },
  "ci_price.sonnet_5_cache_read": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 0.2,
   "verified": "2026-09-16"
  },
  "ci_price.sonnet_5_input": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 2.0,
   "verified": "2026-09-16"
  },
  "ci_price.sonnet_5_output": {
   "source": "https://platform.claude.com/docs/en/about-claude/pricing",
   "value": 10.0,
   "verified": "2026-09-16"
  }
 },
 "glossary": [
  {
   "anchor": "orchestrator",
   "definition": "How much text the model can read in one request",
   "everything so far included.": null,
   "term": "context window"
  },
  {
   "anchor": "cache_breaks",
   "definition": "The part of a request that is identical to the last one",
   "from the first byte. Only a prefix can be cached.": null,
   "term": "prefix"
  },
  {
   "anchor": "cache",
   "definition": "The stored copy of a request's prefix that later requests reuse at a fraction of the price.",
   "term": "cache"
  },
  {
   "anchor": "cache",
   "definition": "Tokens billed at the cheap rate because they matched the cached prefix.",
   "term": "cache read"
  },
  {
   "anchor": "cache",
   "definition": "Tokens billed at a premium because they were new and stored for later reuse.",
   "term": "cache write"
  },
  {
   "anchor": "cache_duration",
   "definition": "How long the cache keeps a prefix before forgetting it. The default is short; a longer one costs more to write.",
   "term": "cache lifetime"
  },
  {
   "anchor": "cache_duration",
   "definition": "A request that re-sends the unchanged prefix so the cache does not expire. Generates nothing.",
   "term": "keep warm"
  },
  {
   "anchor": "effort",
   "definition": "How much thinking",
   "term": "effort",
   "tool calling and self-checking the model does. The default is the highest.": null
  },
  {
   "anchor": "cache",
   "definition": "One trip around the loop. Each turn re-sends the whole conversation so far.",
   "term": "turn"
  },
  {
   "anchor": "measure",
   "definition": "The unit text is billed in. Roughly three quarters of a word.",
   "term": "token"
  },
  {
   "anchor": "budgets",
   "definition": "The most one response may contain",
   "set by max_tokens. The model cannot see it.": null,
   "term": "output cap"
  },
  {
   "anchor": "budgets",
   "definition": "A token allowance the model can see for the whole task",
   "so it wraps up instead of spiraling. Advisory.": null,
   "term": "task budget"
  },
  {
   "anchor": "batch",
   "definition": "Requests nobody is waiting on",
   "run at half price with results within a day.": null,
   "term": "batch"
  },
  {
   "anchor": "tool_search",
   "definition": "The description of a tool the model may call. Every definition attached is input on every turn.",
   "term": "tool definition"
  },
  {
   "anchor": "tool_search",
   "definition": "Loading a tool definition only when the model asks for it.",
   "term": "tool search"
  },
  {
   "anchor": "orchestrator",
   "definition": "A model that splits work across cheaper workers and combines the results.",
   "term": "orchestrator"
  },
  {
   "anchor": "advisor",
   "definition": "A frontier model a cheaper one consults on hard decisions.",
   "term": "advisor"
  },
  {
   "anchor": "measure",
   "definition": "The share of tasks the check says were done right.",
   "term": "pass rate"
  },
  {
   "anchor": "compare_models",
   "definition": "Total spend divided by the tasks that passed. Failed tasks still bill.",
   "term": "cost per solved task"
  },
  {
   "anchor": "compare_models",
   "definition": "The hardest tenth of tasks",
   "term": "tail",
   "which carries most of the bill.": null
  },
  {
   "anchor": "page",
   "definition": "The curve of the best score available at each cost. Free levers move toward it; tradeoffs move along it.",
   "term": "frontier"
  },
  {
   "a date and a noise band.": null,
   "anchor": "benchmarks",
   "definition": "A configuration the page actually ran",
   "term": "measured",
   "with a benchmark": null
  },
  {
   "anchor": "measure",
   "definition": "A configuration the page did not run. The engine interpolated from the nearest measurements.",
   "term": "estimated"
  },
  {
   "anchor": "context_lifecycle",
   "definition": "Summarizing the conversation so far when it grows past a threshold.",
   "term": "compaction"
  },
  {
   "anchor": "context_lifecycle",
   "definition": "Replacing stale tool results with a one-line extract at each task boundary.",
   "term": "prune"
  },
  {
   "anchor": "context_lifecycle",
   "definition": "Clearing old tool results mid-task. Invalidates the cache from that point.",
   "term": "context editing"
  }
 ],
 "levers": [
  {
   "anchor": "cache",
   "control": {
    "type": "switch"
   },
   "data_keys": [
    "ci_cache.triage_saving",
    "ci_cache.agent_loop_factor_low",
    "ci_cache.agent_loop_factor_high",
    "ci_cache.production_read_share_median"
   ],
   "effort": 1,
   "group": "free",
   "headline_off": "Every turn re-sends everything at full price.",
   "headline_on": "Repeated prefix now billed at a tenth.",
   "icon": "database-zap",
   "id": "caching",
   "label": "Remember the repeated part of every request",
   "measured_on": [
    "knowledge",
    "coding"
   ],
   "param": {
    "name": "cache_control",
    "snippet": "cache_control: {\"type\": \"ephemeral\"}"
   },
   "sentence": "On the measured triage run caching cut the bill by 83% of its size, and on agent-loop benchmarks by a factor of 2.7 to 5.3. Production agent loops read a median 84.2% of their input from the cache.",
   "short": "Caching",
   "status": "measured",
   "zippy": {
    "example": "\"We turned it on and the bill dropped by more than half the same afternoon.\"",
    "what": "Every step of an agent re-sends the whole conversation so far. Caching bills the part it has already seen at about a tenth of the price.",
    "why": "It is the largest single lever the page measured, on every model and benchmark, and it costs one field."
   }
  },
  {
   "anchor": "cache_duration",
   "control": {
    "options": [
     {
      "label": "5 minutes",
      "value": "five_min"
     },
     {
      "label": "1 hour",
      "value": "one_hour"
     },
     {
      "gate": "keep_warm",
      "label": "Keep warm",
      "value": "keep_warm"
     }
    ],
    "type": "select"
   },
   "data_keys": [
    "ci_cache.one_hour_pause_share_rule",
    "ci_cache.one_hour_no_pause_extra_sonnet",
    "ci_cache.one_hour_no_pause_extra_opus",
    "ci_cache.fable_keep_warm_saving_low",
    "ci_cache.fable_keep_warm_saving_high",
    "ci_cache.lifetime_5m_name",
    "ci_cache.lifetime_1h_name",
    "ci_cache.keep_warm_interval_phrase"
   ],
   "effort": 1,
   "group": "advanced",
   "headline_off": "Cheapest when turns are seconds apart.",
   "headline_on": "Cache survives the pause; writes cost more.",
   "icon": "timer",
   "id": "cache_lifetime",
   "label": "How long should the cache stay warm?",
   "measured_on": [
    "knowledge"
   ],
   "param": {
    "name": "cache_control.ttl",
    "snippet": "cache_control: {\"type\": \"ephemeral\", \"ttl\": \"1h\"}"
   },
   "requires": [
    "pause_pattern"
   ],
   "sentence": "The 1-hour setting pays once about 5% of turns follow a pause between 5 minutes and an hour; with no pauses it cost 15% more on Sonnet 5 and 11% more on Opus 5. On Fable 5.1, keeping the 5-minute cache warm cost 13% to 20% less than the 1-hour cache for minute-scale pauses.",
   "short": "Cache lifetime",
   "status": "measured",
   "zippy": {
    "example": "\"The rep reads the draft and comes back after a meeting.\"",
    "what": "The cache forgets after the 5-minute cache lifetime unless you buy the 1-hour cache lifetime, which costs more to write. Keep-warm re-sends the unchanged prefix with a keep-alive every 4 minutes so it never expires.",
    "why": "A person pausing between steps decides which one pays. With no pauses the 1-hour setting simply costs more."
   }
  },
  {
   "anchor": "cache_breaks",
   "control": {
    "inverted": true,
    "type": "switch"
   },
   "data_keys": [
    "ci_cache.status_line_run_cost",
    "ci_cache.triage_run_cost"
   ],
   "effort": 2,
   "group": "free",
   "headline_off": "Per-request text moved to the newest turn.",
   "headline_on": "Every request now a full cache write.",
   "icon": "flag",
   "id": "timestamp_prefix",
   "label": "Anything that changes every request at the top of the prompt?",
   "measured_on": [
    "knowledge"
   ],
   "param": {
    "name": "system",
    "snippet": "system: \"Now: 2026-09-16T10:41Z ...\"  # per-request text ahead of the stable prefix"
   },
   "sentence": "On the triage run a 25-token status line at the front of the system prompt cost $4.24 per run instead of $0.59, more than running with caching off.",
   "short": "Changing prefix",
   "status": "measured",
   "zippy": {
    "example": "\"We print the date and the ticket count at the top of the system prompt.\"",
    "what": "The cache matches the request byte for byte from the start. A timestamp or queue position at the top means nothing after it ever matches.",
    "why": "It turns every request into a full cache write, which costs more than caching turned off."
   }
  },
  {
   "anchor": "tool_search",
   "control": {
    "type": "switch"
   },
   "data_keys": [
    "ci_effect.tool_search_saving_502",
    "ci_effect.tool_search_catalog_size",
    "ci_effect.tool_search_saving_github_mcp",
    "ci_effect.tool_catalog_accuracy_phrase"
   ],
   "effort": 1,
   "group": "free",
   "headline_off": "Every tool definition billed on every turn.",
   "headline_on": "Catalog no longer re-sent every turn.",
   "icon": "search",
   "id": "tool_search",
   "label": "Load tools only when needed",
   "measured_on": [
    "knowledge"
   ],
   "param": {
    "name": "defer_loading",
    "snippet": "tools: [{\"type\": \"tool_search_tool_regex_20251119\"}, {..., \"defer_loading\": true}]"
   },
   "sentence": "With every definition loaded the run nearly doubled as the catalog grew; with tool search it stayed flat, 45% less at 502 tools, and 20% less with a GitHub MCP server. Accuracy was 15 to 18 of 20 either way.",
   "short": "Tool search",
   "status": "measured",
   "zippy": {
    "example": "\"We attached the whole GitHub server. Four hundred tools, it uses three.\"",
    "what": "Every tool definition attached to a request is input on every turn. Tool search loads a definition only when the model asks for it.",
    "why": "A few MCP servers add hundreds of definitions. The catalog costs money, not correctness."
   }
  },
  {
   "anchor": "tool_search",
   "control": {
    "max": 502,
    "min": 2,
    "step": 25,
    "type": "stepper"
   },
   "data_keys": [
    "ci_effect.tool_search_catalog_size"
   ],
   "effort": 1,
   "group": "free",
   "headline_off": "Fewer definitions on every turn.",
   "headline_on": "More definitions on every turn.",
   "icon": "wrench",
   "id": "tool_catalog",
   "label": "How many tools are attached?",
   "measured_on": [
    "knowledge"
   ],
   "param": {
    "name": "tools",
    "snippet": "tools: [...]  # N definitions"
   },
   "sentence": "The measured catalog ran up to 502 tools; loaded, the run cost tracked the schema tokens on each request.",
   "short": "Tools attached",
   "status": "measured",
   "zippy": {
    "example": "\"We connected three MCP servers to be safe.\"",
    "what": "The number of tool definitions the request carries. The agent's own tools plus whatever servers are attached.",
    "why": "Without tool search, each one is paid for on every turn."
   }
  },
  {
   "anchor": "data_files",
   "control": {
    "type": "switch"
   },
   "data_keys": [
    "ci_effect.data_file_cost_fraction",
    "ci_effect.data_file_pasted_correct_phrase",
    "ci_effect.data_file_uploaded_correct_phrase",
    "ci_effect.data_file_pasted_tokens"
   ],
   "effort": 3,
   "group": "free",
   "headline_off": "Whole table re-sent on every request.",
   "headline_on": "Table queried by code, not re-read.",
   "icon": "file-up",
   "id": "data_file",
   "label": "Upload the data file instead of pasting it",
   "measured_on": [
    "knowledge"
   ],
   "param": {
    "name": "files",
    "snippet": "container: {\"files\": [file_id]}, tools: [{\"type\": \"code_execution_20260120\"}]"
   },
   "scenario_flag": "has_table",
   "sentence": "Pasted, the table was about 91,000 input tokens on every request and Sonnet 5 answered 6 of 25 questions. Uploaded with code execution it answered 25 of 25 for about 8.3% of the cost.",
   "short": "Data file",
   "status": "measured",
   "zippy": {
    "example": "\"We paste the whole CSV into the prompt so it can see the data.\"",
    "what": "A table pasted into the prompt is re-read on every request. Uploaded through the Files API, the model queries it with code instead.",
    "why": "Cheaper by an order of magnitude, and the answers get right instead of wrong."
   }
  },
  {
   "anchor": "context_lifecycle",
   "control": {
    "options": [
     {
      "label": "Keep everything",
      "value": "none"
     },
     {
      "label": "Prune at task boundaries",
      "value": "prune"
     },
     {
      "label": "Compaction",
      "value": "compaction"
     },
     {
      "label": "Context editing",
      "value": "context_editing"
     }
    ],
    "type": "select"
   },
   "data_keys": [
    "ci_effect.prune_saving_long",
    "ci_effect.compaction_saving_long",
    "ci_effect.context_editing_extra_short",
    "ci_effect.long_run_token_multiple"
   ],
   "effort": 3,
   "group": "advanced",
   "headline_off": "Every old result rides along.",
   "headline_on": "Stale results dropped at the boundary.",
   "icon": "scissors",
   "id": "context_lifecycle",
   "label": "What happens to stale results in a long session?",
   "measured_on": [
    "knowledge"
   ],
   "param": {
    "name": "context_management",
    "snippet": "context_management: {\"edits\": [{\"type\": \"clear_tool_uses_20250919\"}]}  # or a prune at each task boundary"
   },
   "sentence": "On the long run the prune saved 39% and compaction 32%; on the short run they saved nothing and context editing cost 74% more.",
   "short": "Stale results",
   "status": "measured",
   "zippy": {
    "example": "\"The agent works through the whole queue in one session.\"",
    "what": "In a long session the old tool results ride along on every request. A prune replaces them with a line at each task boundary; compaction summarizes; context editing clears them mid-task.",
    "why": "They only pay on a session long enough to need them, and one of them cost more than it saved."
   }
  },
  {
   "anchor": "batch",
   "control": {
    "type": "switch"
   },
   "data_keys": [
    "ci_effect.batch_discount",
    "ci_effect.batch_window_phrase"
   ],
   "effort": 3,
   "group": "free",
   "headline_off": "Interactive path, full price.",
   "headline_on": "Every token half price, results tomorrow.",
   "icon": "package",
   "id": "batch",
   "label": "Can results wait until tomorrow?",
   "measured_on": [
    "knowledge",
    "coding"
   ],
   "param": {
    "name": "batches",
    "snippet": "POST /v1/messages/batches  {\"requests\": [{\"custom_id\": \"...\", \"params\": {...}}]}"
   },
   "sentence": "The Batch API takes 50% off every token of a request, including cached ones, with batch results within 24 hours.",
   "short": "Batch",
   "status": "measured",
   "zippy": {
    "example": "\"It runs overnight anyway.\"",
    "what": "The Batch API takes half off every token, cached ones included, in exchange for results arriving any time within a day.",
    "why": "The second-largest free lever for work nobody is waiting on. Backfills, evaluations, scheduled jobs."
   }
  },
  {
   "anchor": "prompt_audit",
   "control": {
    "statuses": [
     "current",
     "previous"
    ],
    "type": "model_select"
   },
   "data_keys": [
    "ci_effect.prompt_stale_overspend",
    "ci_effect.prompt_audit_saving",
    "ci_effect.prompt_audit_accuracy_before",
    "ci_effect.prompt_audit_accuracy_after",
    "ci_effect.verify_twice_saving"
   ],
   "effort": 2,
   "group": "free",
   "headline_off": "Prompt matches the model it runs on.",
   "headline_on": "Old-model instructions cost extra rounds.",
   "icon": "file-search",
   "id": "prompt_written_for",
   "label": "Prompt was written for",
   "measured_on": [
    "knowledge"
   ],
   "param": {
    "name": "system",
    "snippet": "system: \"...verify twice... be maximally thorough...\"  # written for an older model"
   },
   "sentence": "Prompts written for Opus 4.8 cost 36% more per ticket on Opus 5 for no change in accuracy. The audit made Opus 5 14.0% cheaper and lifted accuracy from 92% to 97%; removing 'verify twice' alone cut 33%.",
   "short": "Prompt audit",
   "status": "measured",
   "zippy": {
    "example": "\"We tell it to verify twice and be maximally thorough. That was for the old model.\"",
    "what": "A prompt accumulates instructions written to steer an older model. A newer model follows them to the letter and does extra work for nothing.",
    "why": "Auditing the prompt against the model you run now is free and makes it cheaper and more accurate."
   }
  },
  {
   "anchor": "budgets",
   "control": {
    "options": [
     {
      "label": "One line",
      "value": "one_line"
     },
     {
      "label": "Two lines",
      "value": "two_lines"
     },
     {
      "label": "Memo",
      "value": "memo"
     }
    ],
    "type": "select"
   },
   "data_keys": [
    "ci_effect.answer_one_line_output_saving",
    "ci_effect.answer_one_line_cost_saving",
    "ci_effect.answer_memo_output_multiple_phrase",
    "ci_effect.answer_memo_cost_multiple"
   ],
   "effort": 2,
   "group": "free",
   "headline_off": "Long answers echo through every later turn.",
   "headline_on": "Answer sized to what gets read.",
   "icon": "align-left",
   "id": "answer_length",
   "label": "One line, two lines, or a memo?",
   "measured_on": [
    "knowledge"
   ],
   "param": {
    "name": "final_instruction",
    "snippet": "4. Finish with exactly one line in this form: DECISION | LABEL | REASON"
   },
   "sentence": "The one-line answer used 39% fewer output tokens than the two-line original and cost 14.0% less per run. The memo used 6 times the output tokens and cost 2.8x the one-line answer, and all three scored within noise.",
   "short": "Answer length",
   "status": "measured",
   "zippy": {
    "example": "\"We ask for a five-section memo so it looks thorough.\"",
    "what": "The shape of the final answer you ask for. Output tokens cost several times input, and in a loop every answer is read back on every later turn.",
    "why": "All three formats scored the same. They differ in what you pay."
   }
  },
  {
   "anchor": "compare_models",
   "control": {
    "statuses": [
     "current"
    ],
    "type": "model_select"
   },
   "data_keys": [
    "ci_bench.swe_fable_low_pass",
    "ci_bench.swe_fable_low_cost_per_solved",
    "ci_bench.swe_sonnet_default_pass",
    "ci_bench.swe_sonnet_default_cost_per_solved",
    "ci_bench.swe_opus_default_pass",
    "ci_bench.swe_opus_default_cost_per_solved",
    "ci_bench.gpqa_haiku_accuracy",
    "ci_bench.gpqa_opus_accuracy"
   ],
   "effort": 1,
   "group": "tradeoff",
   "headline_off": "Compare on cost per solved task.",
   "headline_on": "Compare on cost per solved task.",
   "icon": "cpu",
   "id": "model",
   "label": "Which model?",
   "measured_on": [
    "coding"
   ],
   "param": {
    "name": "model",
    "snippet": "model: \"claude-opus-5\""
   },
   "sentence": "On the coding subset Fable 5.1 at low effort solved 88.6% of tasks for $0.54 per solved task against 77.4% for $0.84 from Sonnet 5, despite a per-token price five times higher. Opus 5 at its default solved 91.7% for $1.01.",
   "short": "Model",
   "status": "measured",
   "zippy": {
    "example": "\"The frontier model is five times the price per token, so it is off the table.\"",
    "what": "Price lists are per token. You pay per completed task, and a more capable model finishes with fewer turns and less backtracking.",
    "why": "The ranking flips by workload and no price list tells you which way. Compare on cost per solved task, on your hardest tenth."
   }
  },
  {
   "anchor": "effort",
   "control": {
    "options": [
     {
      "label": "Low",
      "value": "low"
     },
     {
      "label": "Medium",
      "value": "medium"
     },
     {
      "label": "High",
      "value": "high"
     }
    ],
    "type": "select"
   },
   "data_keys": [
    "ci_effect.effort_knowledge_low_cost_low",
    "ci_effect.effort_knowledge_low_cost_high",
    "ci_effect.effort_knowledge_low_pass_delta_low",
    "ci_effect.effort_knowledge_low_pass_delta_high",
    "ci_effect.effort_coding_medium_cost",
    "ci_effect.effort_coding_medium_pass_delta",
    "ci_effect.effort_coding_low_cost",
    "ci_effect.effort_coding_low_pass_delta",
    "ci_effect.effort_research_low_cost_per_task",
    "ci_effect.effort_research_high_cost_per_task"
   ],
   "effort": 1,
   "group": "tradeoff",
   "headline_off": "Default effort may buy unused depth.",
   "headline_on": "Less thinking, same score on flat work.",
   "icon": "gauge",
   "id": "effort",
   "label": "How hard should it think?",
   "measured_on": [
    "knowledge",
    "coding"
   ],
   "param": {
    "name": "effort",
    "snippet": "output_config: {\"effort\": \"low\"}"
   },
   "sentence": "On research and knowledge-work benchmarks low effort gave up 1% to 3% of pass rate for about half the cost. On long-horizon coding Opus 5 gave up 2% at medium for half the cost and 8% at low for a quarter of it. On deep research, low to high moved cost from $4.66 to $7.12 per task with nearly the same score.",
   "short": "Effort",
   "status": "measured",
   "zippy": {
    "example": "\"We run everything at the top setting to be safe.\"",
    "what": "How much thinking, tool calling and self-checking the model does. Cost scales with all of it; accuracy scales only with the part the task needs.",
    "why": "On knowledge work the curve is nearly flat. On long coding, effort buys accuracy. Sweep it before touching anything else."
   }
  },
  {
   "anchor": "rerun",
   "control": {
    "type": "switch"
   },
   "data_keys": [
    "ci_effect.rerun_low_fail_share",
    "ci_effect.rerun_low_then_default_pass",
    "ci_effect.rerun_low_then_default_cost",
    "ci_effect.rerun_all_default_pass",
    "ci_effect.rerun_all_default_cost"
   ],
   "effort": 3,
   "group": "tradeoff",
   "headline_off": "Every task pays for full effort.",
   "headline_on": "Same pass rate, about half the cost.",
   "icon": "refresh-cw",
   "id": "rerun_failures",
   "label": "Run cheap, re-run what fails",
   "measured_on": [
    "coding"
   ],
   "param": {
    "name": "retry_loop",
    "snippet": "for task in tasks: r = run(task, effort=\"low\"); if not passes(r): r = run(task, effort=\"high\")"
   },
   "requires": [
    "check"
   ],
   "sentence": "With Opus 5 at low, 16% of tasks failed; re-run at the default, about 93% passed for about $0.45 each, against 91.7% for $0.93 running everything at the default.",
   "short": "Re-run failures",
   "status": "measured",
   "zippy": {
    "example": "\"If the tests pass we merge it; if not, a person looks.\"",
    "what": "Run every task at a low setting and re-run only the failures at a higher one. You need a check that says which ones failed.",
    "why": "Same pass rate for about half the cost, counting the failed cheap attempts. Paid for in latency on the failures."
   }
  },
  {
   "anchor": "budgets",
   "control": {
    "options": [
     {
      "label": "Off",
      "value": "off"
     },
     {
      "label": "Generous",
      "value": "generous"
     },
     {
      "label": "Tight",
      "value": "tight"
     }
    ],
    "type": "select"
   },
   "data_keys": [
    "ci_effect.budget_generous_saving",
    "ci_effect.budget_generous_pass_delta",
    "ci_effect.budget_tight_saving",
    "ci_effect.budget_tight_pass_delta",
    "ci_effect.budget_floor_tokens",
    "ci_effect.budget_beta_header"
   ],
   "effort": 3,
   "group": "tradeoff",
   "headline_off": "No countdown; the tail runs free.",
   "headline_on": "Tail trimmed; a few points of pass rate.",
   "icon": "wallet",
   "id": "task_budget",
   "label": "Give it a token budget it can see",
   "measured_on": [
    "coding"
   ],
   "param": {
    "name": "task_budget",
    "snippet": "headers: {\"anthropic-beta\": \"task-budgets-2026-03-13\"}, task_budget: {\"tokens\": 200000}"
   },
   "requires": [
    "check"
   ],
   "sentence": "A generous budget cut cost per task 44% for about 3% of pass rate; the tightest allowed cut it 58.0% for 6%. Budgets below 20,000 tokens are rejected.",
   "short": "Token budget",
   "status": "beta",
   "zippy": {
    "example": "\"A handful of runs cost fifty times the median.\"",
    "what": "The model sees a live token countdown for the whole task and wraps up instead of spiraling. It is advisory, and it targets the expensive tail.",
    "why": "Buys efficiency at a price in pass rate that grows as the budget tightens. Set it once, on the first request."
   }
  },
  {
   "anchor": "budgets",
   "control": {
    "options": [
     {
      "label": "16k tokens",
      "value": "cap_16k"
     },
     {
      "label": "64k tokens",
      "value": "cap_64k"
     },
     {
      "label": "128k tokens",
      "value": "cap_128k"
     }
    ],
    "type": "select"
   },
   "data_keys": [
    "ci_effect.cap_16k_ended_opus",
    "ci_effect.cap_16k_ended_fable",
    "ci_effect.cap_16k_capped_still_passed_phrase",
    "ci_effect.cap_16k_cost_per_solved_phrase",
    "ci_effect.cap_64k_cost_per_solved_phrase",
    "ci_effect.cap_64k_cut_turns_phrase",
    "ci_effect.cap_fable_16k_pass",
    "ci_effect.cap_fable_64k_pass"
   ],
   "effort": 1,
   "group": "tradeoff",
   "headline_off": "Room for the rare long turn.",
   "headline_on": "Cheaper attempts, same cost per solved task.",
   "icon": "ruler",
   "id": "max_tokens",
   "label": "Cap a single answer",
   "measured_on": [
    "coding"
   ],
   "param": {
    "name": "max_tokens",
    "snippet": "max_tokens: 64000"
   },
   "sentence": "A 16k cap ended 15% of Opus 5's attempts and 43% of Fable 5.1's, and only 9 of 117 capped attempts still passed: cost per solved task was $21 against $22 at 64k, where 2 of about 14,000 turns were still cut off and Fable 5.1 solved 58.5% instead of 36.3%.",
   "short": "Answer cap",
   "status": "measured",
   "zippy": {
    "example": "\"We lowered max_tokens to keep the bill down.\"",
    "what": "The most one response may be. The model cannot see it, so lowering it does not make the model economize; the turns that needed the room are discarded and still billed.",
    "why": "It is a safety cap, not a saving. Cost per solved task barely moves."
   }
  },
  {
   "anchor": "advisor",
   "control": {
    "type": "switch"
   },
   "data_keys": [
    "ci_effect.advisor_coding_pass_delta_over_opus",
    "ci_effect.advisor_consults_per_attempt_phrase",
    "ci_effect.advisor_chart_price_multiple"
   ],
   "effort": 4,
   "group": "advanced",
   "headline_off": "One model, one curve to beat.",
   "headline_on": "Every hard call pays the frontier price.",
   "icon": "user-check",
   "id": "advisor",
   "label": "Let a frontier model answer the hard calls",
   "measured_on": [
    "coding"
   ],
   "param": {
    "name": "advisor",
    "snippet": "tools: [{\"type\": \"advisor_20260301\", \"model\": \"claude-fable-5-1\"}]"
   },
   "requires": [
    "effort_sweep"
   ],
   "sentence": "The coding pairing scored 3.5% over Opus 5 alone with about 2 consultations per attempt; the chart-reading pairing matched the advisor's model alone at medium for about 2.6x the price.",
   "short": "Advisor",
   "status": "measured",
   "zippy": {
    "example": "\"The cheap model stalls on a few decisions per ticket.\"",
    "what": "A cheaper executor runs the loop and consults a frontier model on hard decisions.",
    "why": "It pays only when the advisor is priced well above the executor and actually consulted. Draw the single-model effort curve first; a second model has to beat the whole curve."
   }
  },
  {
   "anchor": "orchestrator",
   "control": {
    "type": "switch"
   },
   "data_keys": [
    "ci_effect.orchestrator_cost_fraction",
    "ci_effect.orchestrator_pass_delta_below_low",
    "ci_effect.orchestrator_pass_delta_below_high"
   ],
   "effort": 4,
   "group": "advanced",
   "headline_off": "One model reads it all.",
   "headline_on": "Half the frontier cost beyond one window.",
   "icon": "network",
   "id": "orchestrator",
   "label": "Split the work across cheaper workers",
   "measured_on": [
    "coding"
   ],
   "param": {
    "name": "workers",
    "snippet": "coordinator: \"claude-fable-5-1\", workers: [{\"model\": \"claude-sonnet-5\"} * N]"
   },
   "requires": [
    "effort_sweep"
   ],
   "sentence": "Delegation cost about 50% of the frontier model, both beyond one context window and on routine tails, at 10% to 12% below it.",
   "short": "Workers",
   "status": "measured",
   "zippy": {
    "example": "\"It has to read four hundred contracts.\"",
    "what": "A coordinator partitions work that exceeds one context window and delegates the parts to cheaper workers.",
    "why": "About half the cost of the frontier model on work too big for one window, at some points of accuracy. On work that fits, lowering effort beat it."
   }
  },
  {
   "anchor": "orchestrator",
   "control": {
    "max": 25,
    "min": 1,
    "step": 1,
    "type": "stepper"
   },
   "data_keys": [
    "ci_effect.orchestrator_cost_fraction"
   ],
   "effort": 4,
   "group": "advanced",
   "headline_off": "Fewer workers, slower, same bill.",
   "headline_on": "More workers, faster, same bill.",
   "icon": "split",
   "id": "workers",
   "label": "How many workers?",
   "measured_on": [
    "coding"
   ],
   "param": {
    "name": "workers",
    "snippet": "workers: N"
   },
   "requires": [
    "orchestrator"
   ],
   "sentence": "The charted team ran at the platform's documented limit of 25 concurrent Sonnet 5 workers.",
   "short": "Worker count",
   "status": "measured",
   "zippy": {
    "example": "\"Can it go faster?\"",
    "what": "The number of parallel workers the coordinator fans out to. The platform's documented limit is 25.",
    "why": "More workers finish faster; the bill is the work, not the worker count."
   }
  }
 ],
 "models": [
  {
   "api_id": "claude-haiku-4-5-20251001",
   "family": "haiku",
   "id": "haiku_4_5",
   "name": "Claude Haiku 4.5",
   "price_keys": {
    "cache_read": "ci_price.haiku_4_5_cache_read",
    "input": "ci_price.haiku_4_5_input",
    "output": "ci_price.haiku_4_5_output"
   },
   "prices": {
    "cache_read": 0.1,
    "input": 1.0,
    "output": 5.0
   },
   "read_multiplier_note": "standard",
   "status": "current",
   "tier": 1
  },
  {
   "api_id": "claude-sonnet-5",
   "family": "sonnet",
   "id": "sonnet_5",
   "name": "Claude Sonnet 5",
   "price_keys": {
    "cache_read": "ci_price.sonnet_5_cache_read",
    "input": "ci_price.sonnet_5_input",
    "output": "ci_price.sonnet_5_output"
   },
   "prices": {
    "cache_read": 0.2,
    "input": 2.0,
    "output": 10.0
   },
   "read_multiplier_note": "standard",
   "status": "current",
   "tier": 2
  },
  {
   "api_id": "claude-opus-5",
   "family": "opus",
   "id": "opus_5",
   "name": "Claude Opus 5",
   "price_keys": {
    "cache_read": "ci_price.opus_5_cache_read",
    "input": "ci_price.opus_5_input",
    "output": "ci_price.opus_5_output"
   },
   "prices": {
    "cache_read": 0.5,
    "input": 5.0,
    "output": 25.0
   },
   "read_multiplier_note": "standard",
   "status": "current",
   "tier": 3
  },
  {
   "api_id": "claude-fable-5-1",
   "family": "fable",
   "gates": [
    "keep_warm"
   ],
   "id": "fable_5_1",
   "name": "Claude Fable 5.1",
   "price_keys": {
    "cache_read": "ci_price.fable_5_1_cache_read",
    "input": "ci_price.fable_5_1_input",
    "output": "ci_price.fable_5_1_output"
   },
   "prices": {
    "cache_read": 0.25,
    "input": 10.0,
    "output": 50.0
   },
   "read_multiplier_note": "reduced",
   "status": "current",
   "tier": 4
  },
  {
   "family": "sonnet",
   "id": "sonnet_4_6",
   "name": "Claude Sonnet 4.6",
   "prices": {},
   "status": "previous",
   "successor": "sonnet_5",
   "tier": 2
  },
  {
   "family": "opus",
   "id": "opus_4_8",
   "name": "Claude Opus 4.8",
   "prices": {},
   "status": "previous",
   "successor": "opus_5",
   "tier": 3
  },
  {
   "family": "opus",
   "id": "opus_4_7",
   "name": "Claude Opus 4.7",
   "prices": {},
   "status": "previous",
   "successor": "opus_5",
   "tier": 3
  },
  {
   "family": "fable",
   "id": "fable_5",
   "name": "Claude Fable 5",
   "prices": {},
   "status": "previous",
   "successor": "fable_5_1",
   "tier": 4
  }
 ],
 "page_url": "https://platform.claude.com/docs/en/about-claude/models/optimizing-for-cost-and-intelligence",
 "reference": {
  "questions": [
   {
    "icon": "shield-check",
    "id": "quality_ok",
    "label": "Is the quality good enough today?",
    "options": [
     {
      "label": "Yes",
      "value": "yes"
     },
     {
      "label": "No",
      "value": "no"
     }
    ],
    "zippy": {
     "example": "\"The labels are right, it just costs too much.\"",
     "what": "Whether the customer is happy with the answers and only the bill is the problem.",
     "why": "If quality is fine, the cheapest move is lowering effort on the current model; if not, the order is different."
    }
   },
   {
    "icon": "user-round",
    "id": "pause",
    "label": "Does a person wait between steps?",
    "options": [
     {
      "label": "No",
      "value": "none"
     },
     {
      "label": "Minutes",
      "value": "minutes"
     },
     {
      "label": "Toward an hour",
      "value": "toward_hour"
     },
     {
      "label": "Over an hour",
      "value": "over_hour"
     }
    ],
    "zippy": {
     "example": "\"The rep reads the draft, then comes back after lunch.\"",
     "what": "Whether the agent runs on its own or pauses for a human reply.",
     "why": "The cache forgets after the 5-minute cache lifetime by default, and a pause decides which cache lifetime pays."
    }
   },
   {
    "icon": "check-circle",
    "id": "check",
    "label": "How would you know it worked?",
    "options": [
     {
      "label": "Tests pass",
      "value": "tests"
     },
     {
      "label": "Ticket closed",
      "value": "ticket"
     },
     {
      "label": "Number matches",
      "value": "number"
     },
     {
      "label": "A person reviews",
      "value": "person"
     },
     {
      "label": "Can't tell",
      "value": "none"
     }
    ],
    "zippy": {
     "example": "\"If the tests pass, we ship it.\"",
     "what": "Whether there is a check that says right or wrong without a human reading it.",
     "why": "With a check you can run cheap first and re-run only failures for about half the cost."
    }
   },
   {
    "icon": "maximize",
    "id": "size",
    "label": "How big is the work compared with what the model can hold at once?",
    "options": [
     {
      "label": "Fits",
      "value": "fits"
     },
     {
      "label": "Bigger",
      "value": "bigger"
     }
    ],
    "zippy": {
     "example": "\"It has to read the whole contract set, 400 documents.\"",
     "what": "A context window is how much the model can read in one go.",
     "why": "Work bigger than one window is where splitting across workers pays."
    }
   },
   {
    "icon": "list-checks",
    "id": "extras",
    "label": "Anything else true here?",
    "multi": true,
    "options": [
     {
      "label": "Answers get cut off before they finish",
      "value": "stopped_at_cap"
     },
     {
      "label": "Not on the latest model",
      "value": "not_latest"
     },
     {
      "label": "A few runs cost most of the bill",
      "value": "tail_heavy"
     }
    ],
    "zippy": {
     "example": "\"Half the spend is a handful of runs.\"",
     "what": "Three situations the page routes separately.",
     "why": "Each has its own lever and its own number."
    }
   }
  ],
  "rows": [
   {
    "anchor": "cache",
    "figure": "Caching cut agent-loop cost by a factor of 2.7 to 5.3 and the triage bill by 83%, or 88% with input trimming.",
    "first": true,
    "id": "any_workload",
    "lever": "caching",
    "match": {
     "always": true
    },
    "scenario": "triage",
    "strip": {
     "cost": "free",
     "figure": "it cut the triage bill 83%"
    },
    "title": "Turn on prompt caching and trim unneeded tokens; both are free"
   },
   {
    "anchor": "cache_duration",
    "figure": "On Fable 5.1, keep the 5-minute cache warm while pauses run minutes and buy the 1-hour duration when pauses run toward an hour; keep-warm cost 13% to 20% less than the 1-hour cache for minute-scale pauses.",
    "id": "person_waits",
    "lever": "cache_lifetime",
    "match": {
     "pause": [
      "minutes",
      "toward_hour"
     ]
    },
    "scenario": "sales_calls",
    "title": "Use the 1-hour cache duration once about 1 turn in 20 follows a pause between 5 minutes and an hour"
   },
   {
    "anchor": "cache_duration",
    "figure": "The 1-hour duration pays only when at least about 40% of long pauses end within the hour; with no pauses the default cost 15% less on Sonnet 5.",
    "id": "over_hour",
    "lever": "cache_lifetime",
    "match": {
     "pause": [
      "over_hour"
     ]
    },
    "scenario": "triage",
    "title": "Stay on the 5-minute default; a gap over an hour expires both durations"
   },
   {
    "anchor": "effort",
    "figure": "On knowledge work low effort gave up 1% to 3% for a third to a half off; on long coding, 2% at medium for half the cost.",
    "id": "costs_high_quality_fine",
    "lever": "effort",
    "match": {
     "quality_ok": [
      "yes"
     ]
    },
    "scenario": "triage",
    "title": "Sweep effort down on your current model"
   },
   {
    "anchor": "compare_models",
    "figure": "Opus 5 at low effort beat Opus 4.8's default for about 30% of its cost per solved task; Fable 5.1 at low solved 88.6% for $0.54 per solved task.",
    "id": "quality_not_enough",
    "lever": "model",
    "match": {
     "quality_ok": [
      "no"
     ]
    },
    "scenario": "coding",
    "title": "If you lowered effort, restore it; otherwise try the next tier up at low effort"
   },
   {
    "anchor": "upgrade",
    "figure": "Opus 4.8 to Opus 5: 12% more of tasks at 21% more per solved task; Sonnet 4.6 to Sonnet 5: 15% less per solved task; Fable 5 to 5.1: 43% less.",
    "id": "not_latest",
    "lever": "model",
    "match": {
     "extras": [
      "not_latest"
     ]
    },
    "scenario": "migration",
    "title": "Upgrade; the current model solves more tasks at a cost per solved task from about 40% lower to about 20% higher"
   },
   {
    "anchor": "rerun",
    "figure": "On the coding benchmark the pass rate held, about 93%, at about half the cost: $0.45 against $0.93 per task.",
    "id": "checkable",
    "lever": "rerun_failures",
    "match": {
     "check": [
      "tests",
      "ticket",
      "number"
     ]
    },
    "scenario": "coding",
    "title": "Run everything at low effort and re-run failures at the default"
   },
   {
    "anchor": "budgets",
    "figure": "At 64k, 2 of about 14,000 turns were cut off; Fable 5.1 solved 58.5% instead of 36.3% at 16k, at about the same cost per solved task.",
    "id": "stopped_at_cap",
    "lever": "max_tokens",
    "match": {
     "extras": [
      "stopped_at_cap"
     ]
    },
    "scenario": "coding",
    "title": "Raise max_tokens; 64,000 covered all but 2 of 14,000 turns and 128,000 cost nothing extra per solved task"
   },
   {
    "anchor": "budgets",
    "figure": "A generous task budget cut cost per task 44% for about 3% of pass rate; the tightest cut it 58.0% for 6%.",
    "id": "tail_heavy",
    "lever": "task_budget",
    "match": {
     "extras": [
      "tail_heavy"
     ]
    },
    "scenario": "triage",
    "title": "Set a task budget, a session budget and a workspace spend limit"
   },
   {
    "anchor": "advisor",
    "figure": "The coding pairing scored 3.5% over Opus 5 alone; the chart-reading pairing matched the advisor's model alone for about 2.6x the price.",
    "id": "stalls_on_hard",
    "lever": "advisor",
    "match": {
     "check": [
      "person",
      "none"
     ],
     "quality_ok": [
      "no"
     ]
    },
    "scenario": "coding",
    "title": "Add a frontier advisor, after pricing the advisor's model alone at low effort and measuring the consult rate"
   },
   {
    "anchor": "orchestrator",
    "figure": "About 50% of the frontier model's cost beyond one context window, at 10% to 12% below it.",
    "id": "bigger_than_window",
    "lever": "orchestrator",
    "match": {
     "size": [
      "bigger"
     ]
    },
    "scenario": "contracts",
    "title": "Delegate partitions to cheaper workers"
   }
  ]
 },
 "scenarios": [
  {
   "calibration": {
    "anchors": [
     {
      "config": {
       "model": "fable_5_1"
      },
      "cost_per_task": 0.15,
      "cost_per_task_key": "ci_bench.chart_fable_low_cost",
      "label": "Fable 5.1 low, cost per chart"
     },
     {
      "config": {
       "model": "opus_5"
      },
      "cost_per_task": 0.38,
      "cost_per_task_key": "ci_bench.chart_opus_low_cost",
      "label": "Opus 5 low, cost per chart"
     }
    ],
    "base": {
     "effort": "low"
    },
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "It reads the numbers off the weekly dashboards and answers questions about them. We picked the mid-tier model because the frontier one is twice the price per token.",
    "quality": "Ops needs six readings in ten right; the rest a person checks.",
    "target": "Ops wants it under 450 a month."
   },
   "domain": "Operations",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 500,
     "one_line": 40,
     "two_lines": 120
    },
    "base_turns": 4,
    "difficulty_weights": [
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     2.1,
     2.5
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": true,
      "pass_high": 0.645,
      "turn_multiplier": 1.25
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.32,
      "turn_multiplier": 1.2
     },
     "opus_5": {
      "measured": true,
      "pass_high": 0.51,
      "turn_multiplier": 4.6
     },
     "sonnet_5": {
      "measured": false,
      "pass_high": 0.45,
      "turn_multiplier": 1.1
     }
    },
    "own_tools": 2,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 1200,
    "task_input_tokens": 2600,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 180,
    "tool_result_tokens": 1500
   },
   "has_table": false,
   "icon": "bar-chart",
   "id": "charts",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "Chartography, 100-question set"
    ],
    "models": [
     "fable_5_1",
     "opus_5"
    ],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "images"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "low",
    "max_tokens": "cap_64k",
    "model": "opus_5",
    "orchestrator": false,
    "prompt_written_for": "opus_5",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "operations",
    "images",
    "model-choice-flips",
    "unattended",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 450,
    "quality_floor": 0.6
   },
   "title": "Chart reader",
   "traps_featured": [
    "per_token",
    "second_model_first"
   ],
   "workflow": {
    "check": "number",
    "pause_pattern": "none",
    "size_in_windows": 0.2,
    "steps_per_task": 4,
    "tasks_per_month": 3000
   },
   "zippy": {
    "case": "The team compared the models per token and picked the cheaper one.",
    "what": "One chart image per question, a few turns to zoom and compute, an answer graded against the true value.",
    "why": "On the chart-reading benchmark the frontier model at low effort scored higher for less than half the cost per chart of the tier below it."
   }
  },
  {
   "calibration": {
    "anchors": [],
    "base": {},
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "New hires ask it policy questions in a chat that runs all week. Answers are long and friendly and the bill per conversation keeps growing the longer people talk.",
    "quality": "HR needs the answer to match the handbook nine times in ten.",
    "target": "HR wants it under 700 a month."
   },
   "domain": "HR",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 900,
     "one_line": 80,
     "two_lines": 220
    },
    "base_turns": 2,
    "difficulty_weights": [
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.9,
     0.95,
     0.95,
     1.0,
     1.0,
     1.0,
     1.0,
     1.05,
     1.05,
     1.1,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     1.6,
     1.9,
     2.2
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": false,
      "pass_high": 0.96,
      "turn_multiplier": 1.0
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.82,
      "turn_multiplier": 1.0
     },
     "opus_5": {
      "measured": false,
      "pass_high": 0.95,
      "turn_multiplier": 1.0
     },
     "sonnet_5": {
      "measured": false,
      "pass_high": 0.92,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 1,
    "pause_every": 2,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 3000,
    "task_input_tokens": 150,
    "tasks_per_session": 12,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 40,
    "tool_result_tokens": 800
   },
   "has_table": false,
   "icon": "user-check",
   "id": "chatbot",
   "interaction": "person_in_loop",
   "measured_on": {
    "benchmarks": [
     "Answer-length measurement on the triage job; this workload itself is not measured"
    ],
    "models": [],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "chat"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "memo",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "sonnet_5",
    "orchestrator": false,
    "pause_pattern": "minutes",
    "prompt_written_for": "sonnet_5",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "hr",
    "chat",
    "answer-length",
    "person-waits",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 700,
    "quality_floor": 0.9
   },
   "title": "Onboarding chatbot",
   "traps_featured": [
    "memo",
    "one_hour_everywhere"
   ],
   "workflow": {
    "check": "person",
    "pause_pattern": "minutes",
    "size_in_windows": 0.3,
    "steps_per_task": 2,
    "tasks_per_month": 30000
   },
   "zippy": {
    "case": "The team asks for a warm, thorough, multi-paragraph answer every time, on the 5-minute cache while people type.",
    "what": "A conversation of a dozen questions, each with a policy lookup, with the whole conversation re-sent on every turn.",
    "why": "Output costs several times input, and in a conversation every answer is read back as input on every later turn. The memo-length answer is the page's echo."
   }
  },
  {
   "calibration": {
    "anchors": [
     {
      "config": {},
      "cost_per_task": 0.93,
      "cost_per_task_key": "ci_effect.rerun_all_default_cost",
      "label": "Opus 5 default cost per task"
     },
     {
      "config": {},
      "label": "Opus 5 default pass rate",
      "pass": 0.917,
      "pass_key": "ci_bench.swe_opus_default_pass"
     },
     {
      "config": {
       "model": "sonnet_5"
      },
      "cost_per_solved": 0.84,
      "cost_per_solved_key": "ci_bench.swe_sonnet_default_cost_per_solved",
      "label": "Sonnet 5 default cost per solved"
     },
     {
      "config": {
       "model": "fable_5_1"
      },
      "cost_per_solved": 1.19,
      "cost_per_solved_key": "ci_bench.swe_fable_default_cost_per_solved",
      "label": "Fable 5.1 default cost per solved"
     },
     {
      "config": {
       "effort": "low",
       "model": "fable_5_1"
      },
      "cost_per_solved": 0.54,
      "cost_per_solved_key": "ci_bench.swe_fable_low_cost_per_solved",
      "label": "Fable 5.1 low cost per solved"
     },
     {
      "config": {
       "effort": "low"
      },
      "cost_per_solved": 0.25,
      "cost_per_solved_key": "ci_bench.swe_opus_low_cost_per_solved",
      "label": "Opus 5 low cost per solved",
      "tolerance": 0.25
     }
    ],
    "base": {
     "max_tokens": "cap_64k",
     "model": "opus_5"
    },
    "cost_per_task_default": 0.93,
    "cost_per_task_default_key": "ci_effect.rerun_all_default_cost",
    "fable_cost_per_solved": 1.19,
    "fable_cost_per_solved_key": "ci_bench.swe_fable_default_cost_per_solved",
    "model": "opus_5",
    "pass_default": 0.917,
    "pass_default_key": "ci_bench.swe_opus_default_pass",
    "sonnet_cost_per_solved": 0.84,
    "sonnet_cost_per_solved_key": "ci_bench.swe_sonnet_default_cost_per_solved",
    "tolerance": 0.15
   },
   "customer_lines": {
    "objection": "It picks up tickets in our repo and opens pull requests. Half the monthly spend is a handful of runs, and the team says the frontier model is off the table at that price per token.",
    "quality": "Below 88 of 100 passing the test suite and the reviewers stop trusting the queue.",
    "target": "Engineering wants it under 2,000 a month, and the frontier model is off the table at that price per token."
   },
   "domain": "Engineering",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 600,
     "one_line": 60,
     "two_lines": 120
    },
    "base_turns": 28,
    "difficulty_weights": [
     0.5,
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     2.3,
     2.6
    ],
    "effort_overrides": {
     "fable_5_1": {
      "low": {
       "cost_fraction": 0.54,
       "cost_fraction_key": "ci_bench.swe_fable_low_cost_per_solved",
       "pass": 0.886,
       "pass_key": "ci_bench.swe_fable_low_pass"
      }
     }
    },
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": true,
      "pass_high": 0.921,
      "turn_multiplier": 0.66
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.55,
      "turn_multiplier": 1.5
     },
     "opus_5": {
      "measured": true,
      "pass_high": 0.917,
      "turn_multiplier": 0.75
     },
     "sonnet_5": {
      "measured": true,
      "pass_high": 0.774,
      "turn_multiplier": 1.1
     }
    },
    "own_tools": 4,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 3000,
    "task_input_tokens": 2500,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 320,
    "tool_result_tokens": 1800
   },
   "has_table": false,
   "icon": "code",
   "id": "coding",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "SWE-bench Pro subset",
     "482 problems"
    ],
    "models": [
     "sonnet_5",
     "opus_5",
     "fable_5_1"
    ],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "code"
   ],
   "shape": "coding",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_16k",
    "model": "sonnet_5",
    "orchestrator": false,
    "prompt_written_for": "sonnet_5",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 4,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "engineering",
    "code",
    "checkable",
    "effort-steep",
    "coding"
   ],
   "thresholds": {
    "cost_target": 2000,
    "quality_floor": 0.88
   },
   "title": "Coding agent on a repo",
   "traps_featured": [
    "per_token",
    "lower_max_tokens",
    "model_before_effort",
    "second_model_first",
    "rerun_without_check",
    "judge_median",
    "high_effort_hard"
   ],
   "workflow": {
    "check": "tests",
    "pause_pattern": "none",
    "size_in_windows": 0.6,
    "steps_per_task": 24,
    "tasks_per_month": 5000
   },
   "zippy": {
    "case": "The team runs everything at the default effort on the mid-tier model, compares models per token, and lowered max_tokens to save money.",
    "what": "One long session per ticket. Reads the code, edits, runs tests, edits again. Dozens of steps, and a test suite that says pass or fail.",
    "why": "This is the shape where effort genuinely buys accuracy, and where a checkable outcome makes run-cheap-then-re-run the cheapest policy on the curve."
   }
  },
  {
   "calibration": {
    "anchors": [],
    "base": {},
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "It has to read the whole contract set, four hundred documents, for every question. The frontier model reads it serially and the bill per question is enormous.",
    "quality": "Legal needs eight clauses in ten found, and the misses flagged.",
    "target": "Legal wants it under 1,500 a month."
   },
   "domain": "Legal",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 1800,
     "one_line": 150,
     "two_lines": 600
    },
    "base_turns": 30,
    "difficulty_weights": [
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     2.1,
     2.5
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": false,
      "pass_high": 0.92,
      "turn_multiplier": 0.85
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.5,
      "turn_multiplier": 1.3
     },
     "opus_5": {
      "measured": false,
      "pass_high": 0.85,
      "turn_multiplier": 0.9
     },
     "sonnet_5": {
      "measured": false,
      "pass_high": 0.7,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 2,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 2000,
    "task_input_tokens": 800,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 200,
    "tool_result_tokens": 4000
   },
   "has_table": false,
   "icon": "layers",
   "id": "contracts",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "Corpus defect sweep, 21.6-million-token corpus; this workload itself is not measured"
    ],
    "models": [],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "pdf",
    "text"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "fable_5_1",
    "orchestrator": false,
    "prompt_written_for": "fable_5_1",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "legal",
    "long-documents",
    "bigger-than-a-window",
    "split",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 1500,
    "quality_floor": 0.8
   },
   "title": "Contract clause finder",
   "traps_featured": [
    "second_model_first",
    "model_before_effort"
   ],
   "workflow": {
    "check": "person",
    "pause_pattern": "none",
    "size_in_windows": 6,
    "steps_per_task": 30,
    "tasks_per_month": 200
   },
   "zippy": {
    "case": "The team runs the frontier model alone and lowered effort to save money, which changed the accuracy and not the bill.",
    "what": "One question, a corpus too big for one context window, a serial read that re-reads its own state on every pass.",
    "why": "Work larger than one window is the one case where delegation to workers paid on the page; lowering effort cannot help because the bill is the read itself."
   }
  },
  {
   "calibration": {
    "anchors": [
     {
      "config": {
       "data_file": false
      },
      "cost_fraction": 0.083,
      "cost_fraction_key": "ci_effect.data_file_cost_fraction",
      "inverse": true,
      "label": "pasted table costs the page's multiple of uploaded",
      "tolerance": 0.3,
      "versus": {}
     },
     {
      "config": {
       "data_file": false
      },
      "label": "pasted table answers 6 of 25",
      "pass_value": 0.24
     },
     {
      "config": {},
      "label": "uploaded answers all",
      "pass_value": 0.99
     }
    ],
    "base": {
     "caching": false,
     "data_file": true,
     "effort": "low"
    },
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "People ask it sums and counts over the sales export. It gets most of them wrong and each question costs more than we expected.",
    "quality": "Ops needs nine answers in ten to match the spreadsheet.",
    "target": "Ops wants it under 200 a month."
   },
   "domain": "Analytics",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 600,
     "one_line": 60,
     "two_lines": 150
    },
    "base_turns": 3,
    "difficulty_weights": [
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     2.1,
     2.5
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": false,
      "pass_high": 0.99,
      "turn_multiplier": 0.9
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.9,
      "turn_multiplier": 1.1
     },
     "opus_5": {
      "measured": true,
      "pass_high": 0.99,
      "turn_multiplier": 0.95
     },
     "sonnet_5": {
      "measured": true,
      "pass_high": 0.99,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 2,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 4000,
    "task_input_tokens": 300,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 220,
    "tool_result_tokens": 2600
   },
   "has_table": true,
   "icon": "table",
   "id": "data_desk",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "Data-file question set, 25 aggregate questions over a 1,862-row CSV"
    ],
    "models": [
     "sonnet_5",
     "opus_5"
    ],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "tables"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": false,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "low",
    "max_tokens": "cap_64k",
    "model": "sonnet_5",
    "orchestrator": false,
    "prompt_written_for": "sonnet_5",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "analytics",
    "tables",
    "files-api",
    "unattended",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 200,
    "quality_floor": 0.9
   },
   "title": "Data question desk",
   "traps_featured": [
    "paste_table"
   ],
   "workflow": {
    "check": "number",
    "pause_pattern": "none",
    "size_in_windows": 0.5,
    "steps_per_task": 3,
    "tasks_per_month": 2000
   },
   "zippy": {
    "case": "The team pastes the CSV into the prompt so the model can see the data.",
    "what": "A question, a table of a couple of thousand rows, an answer. The table is either pasted into every request or uploaded once and queried with code.",
    "why": "Pasted, the model reads the whole table on every request and still gets most answers wrong. Uploaded with code execution it gets them right for a twelfth of the cost."
   }
  },
  {
   "calibration": {
    "anchors": [],
    "base": {},
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "It pulls the fields off every invoice PDF and matches the vendor. It runs all day and nobody reads the output until the weekly close.",
    "quality": "Finance needs nine in ten invoices matched without a person touching them.",
    "target": "Finance wants it under 600 a month."
   },
   "domain": "Finance",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 900,
     "one_line": 120,
     "two_lines": 250
    },
    "base_turns": 3,
    "difficulty_weights": [
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     2.1,
     2.5
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": false,
      "pass_high": 0.96,
      "turn_multiplier": 0.85
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.8,
      "turn_multiplier": 1.2
     },
     "opus_5": {
      "measured": false,
      "pass_high": 0.95,
      "turn_multiplier": 0.9
     },
     "sonnet_5": {
      "measured": false,
      "pass_high": 0.92,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 2,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 1500,
    "task_input_tokens": 1800,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 80,
    "tool_result_tokens": 500
   },
   "has_table": true,
   "icon": "file-up",
   "id": "invoices",
   "interaction": "batch",
   "measured_on": {
    "benchmarks": [
     "No page measurement on this shape; effects use the page's ratios and are labeled estimated"
    ],
    "models": [],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "pdf",
    "tables"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": false,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "sonnet_5",
    "orchestrator": false,
    "prompt_written_for": "sonnet_5",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "finance",
    "pdf",
    "batch",
    "unattended",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 600,
    "quality_floor": 0.9
   },
   "title": "Invoice extraction",
   "traps_featured": [
    "paste_table",
    "caching_always"
   ],
   "workflow": {
    "check": "number",
    "pause_pattern": "none",
    "size_in_windows": 0.5,
    "steps_per_task": 3,
    "tasks_per_month": 20000
   },
   "zippy": {
    "case": "The team pastes the vendor table into every request and runs on the interactive path.",
    "what": "One short loop per invoice: read the PDF text, look up the vendor, emit the fields. The vendor master table rides along in the prompt.",
    "why": "Nothing here is interactive, so the batch discount applies to every token, and the pasted table is the page's file-in-the-prompt case."
   }
  },
  {
   "calibration": {
    "anchors": [
     {
      "config": {
       "tool_search": true
      },
      "label": "tool search saves the measured share at 502 tools",
      "saving": 0.45,
      "saving_key": "ci_effect.tool_search_saving_502",
      "versus": {
       "tool_search": false
      }
     },
     {
      "config": {
       "tool_search": true
      },
      "label": "with tool search the session costs what the plain triage run cost",
      "session_cost": 0.59,
      "session_cost_key": "ci_cache.triage_run_cost"
     }
    ],
    "base": {
     "tool_catalog": 502
    },
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "We connected three MCP servers so the agent could do anything. It uses three tools out of five hundred and the bill nearly doubled.",
    "quality": "Engineering needs the same label accuracy as before, 17 in 20.",
    "target": "Engineering wants it back under 200 a month."
   },
   "domain": "Platform",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 1600,
     "one_line": 60,
     "two_lines": 300
    },
    "base_turns": 4,
    "difficulty_weights": [
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.25,
     1.35,
     1.5,
     2.0,
     2.4
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.3,
    "models": {
     "fable_5_1": {
      "measured": true,
      "pass_high": 0.9,
      "turn_multiplier": 0.85
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.7,
      "turn_multiplier": 1.4
     },
     "opus_5": {
      "measured": true,
      "pass_high": 0.9,
      "turn_multiplier": 0.9
     },
     "sonnet_5": {
      "measured": true,
      "pass_high": 0.85,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 2,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 1800,
    "task_input_tokens": 480,
    "tasks_per_session": 20,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 120,
    "tool_result_tokens": 170
   },
   "has_table": false,
   "icon": "wrench",
   "id": "mcp_agent",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "Issue-triage agent with up to 502 tool definitions"
    ],
    "models": [
     "sonnet_5"
    ],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "text",
    "screenshots"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "sonnet_5",
    "orchestrator": false,
    "prompt_written_for": "sonnet_5",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 502,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "platform",
    "tool-catalog",
    "tool-search",
    "unattended",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 200,
    "quality_floor": 0.85
   },
   "title": "MCP-heavy agent",
   "traps_featured": [
    "caching_always",
    "timestamp"
   ],
   "workflow": {
    "check": "ticket",
    "pause_pattern": "none",
    "size_in_windows": 0.4,
    "steps_per_task": 4,
    "tasks_per_month": 5000
   },
   "zippy": {
    "case": "The team attached every server and loads every definition on every turn.",
    "what": "The support triage loop with a catalog of hundreds of tool definitions attached to every request.",
    "why": "Every definition is input on every turn. With tool search the run stayed flat at every catalog size; the catalog costs money, not correctness."
   }
  },
  {
   "calibration": {
    "anchors": [
     {
      "config": {},
      "label": "audited prompt costs the measured fraction of the stale one",
      "saving": 0.14,
      "saving_key": "ci_effect.prompt_audit_saving",
      "tolerance": 0.2,
      "versus": {
       "prompt_written_for": "opus_4_8"
      }
     },
     {
      "config": {
       "prompt_written_for": "opus_4_8"
      },
      "label": "stale prompt accuracy",
      "pass": 0.92,
      "pass_key": "ci_effect.prompt_audit_accuracy_before"
     },
     {
      "config": {},
      "label": "audited accuracy",
      "pass": 0.97,
      "pass_key": "ci_effect.prompt_audit_accuracy_after"
     }
    ],
    "base": {
     "model": "opus_5",
     "prompt_written_for": "opus_5"
    },
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "We moved the support agent to the new model and the bill per ticket went up, not down. The prompt has not changed in a year.",
    "quality": "Support needs 95 tickets in 100 handled without escalation.",
    "target": "Support wants it under 600 a month."
   },
   "domain": "Support",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 900,
     "one_line": 120,
     "two_lines": 300
    },
    "base_turns": 5,
    "difficulty_weights": [
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     2.1,
     2.5
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": false,
      "pass_high": 0.97,
      "turn_multiplier": 0.85
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.85,
      "turn_multiplier": 1.2
     },
     "opus_5": {
      "measured": true,
      "pass_high": 0.97,
      "turn_multiplier": 0.9
     },
     "sonnet_5": {
      "measured": true,
      "pass_high": 0.95,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 3,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 2200,
    "task_input_tokens": 700,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 150,
    "tool_result_tokens": 600
   },
   "has_table": false,
   "icon": "file-search",
   "id": "migration",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "Support-desk prompt-audit evaluation, 44 support tickets"
    ],
    "models": [
     "opus_5",
     "sonnet_5"
    ],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "text"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "opus_5",
    "orchestrator": false,
    "prompt_written_for": "opus_4_8",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "support",
    "prompt-text",
    "prompt-audit",
    "unattended",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 600,
    "quality_floor": 0.95
   },
   "title": "Migration from an older model",
   "traps_featured": [
    "verify_twice",
    "high_effort_hard"
   ],
   "workflow": {
    "check": "ticket",
    "pause_pattern": "none",
    "size_in_windows": 0.3,
    "steps_per_task": 5,
    "tasks_per_month": 8000
   },
   "zippy": {
    "case": "The team changed the model string and nothing else.",
    "what": "A ticket, an order lookup, a policy check, a reply. The system prompt still says verify twice and be maximally thorough, written for the previous model.",
    "why": "A newer model follows old-model instructions to the letter and does extra rounds for nothing. The audit is one command and it made the newer model both cheaper and more accurate."
   }
  },
  {
   "calibration": {
    "anchors": [
     {
      "config": {},
      "cost_per_task": 7.12,
      "cost_per_task_key": "ci_effect.effort_research_high_cost_per_task",
      "label": "Fable 5.1 high, cost per task"
     },
     {
      "config": {
       "effort": "low"
      },
      "cost_per_task": 4.66,
      "cost_per_task_key": "ci_effect.effort_research_low_cost_per_task",
      "label": "Fable 5.1 low, cost per task"
     },
     {
      "config": {
       "effort": "low",
       "model": "sonnet_5"
      },
      "cost_per_task": 1.2,
      "cost_per_task_key": "ci_bench.research_sonnet_cost_per_task",
      "label": "Sonnet 5 low, cost per task",
      "tolerance": 0.3
     }
    ],
    "base": {
     "effort": "high",
     "model": "fable_5_1"
    },
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "It writes market briefs from the web. The team runs it at the top setting because the briefs matter, and each one costs more than a contractor hour.",
    "quality": "The briefs have to pass the analysts' rubric two times in three.",
    "target": "Research wants it under 3,000 a month."
   },
   "domain": "Research",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 4000,
     "one_line": 300,
     "two_lines": 1500
    },
    "base_turns": 60,
    "difficulty_weights": [
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     2.1,
     2.5
    ],
    "effort_overrides": {
     "fable_5_1": {
      "low": {
       "cost_per_task": 4.66,
       "cost_per_task_key": "ci_effect.effort_research_low_cost_per_task",
       "high_cost_per_task": 7.12,
       "high_cost_per_task_key": "ci_effect.effort_research_high_cost_per_task"
      }
     }
    },
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": true,
      "pass_high": 0.66,
      "turn_multiplier": 1.0
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.4,
      "turn_multiplier": 1.3
     },
     "opus_5": {
      "measured": false,
      "pass_high": 0.62,
      "turn_multiplier": 0.95
     },
     "sonnet_5": {
      "measured": true,
      "pass_high": 0.56,
      "turn_multiplier": 0.78
     }
    },
    "own_tools": 3,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 2500,
    "task_input_tokens": 600,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 380,
    "tool_result_tokens": 3800
   },
   "has_table": false,
   "icon": "newspaper",
   "id": "research",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "DeepResearch Bench II",
     "132 research tasks across 22 domains"
    ],
    "models": [
     "fable_5_1",
     "sonnet_5"
    ],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "text",
    "web"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "fable_5_1",
    "orchestrator": false,
    "prompt_written_for": "fable_5_1",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "research",
    "web",
    "flat-effort",
    "unattended",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 3000,
    "quality_floor": 0.6
   },
   "title": "Market brief writer",
   "traps_featured": [
    "high_effort_hard",
    "second_model_first",
    "model_before_effort"
   ],
   "workflow": {
    "check": "person",
    "pause_pattern": "none",
    "size_in_windows": 0.9,
    "steps_per_task": 60,
    "tasks_per_month": 500
   },
   "zippy": {
    "case": "The team assumed hard work needs high effort and never swept it.",
    "what": "One long research loop per brief: search, fetch, read, search again, then write. Dozens of turns over a large context.",
    "why": "On deep research the score barely moved across effort settings while the cost did; the frontier model does more work here, not less."
   }
  },
  {
   "calibration": {
    "anchors": [],
    "base": {},
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "It drafts the call summary from the transcript, the rep reads it, asks for changes, and comes back to it between meetings. The bill is fine per call and large in total.",
    "quality": "Reps have to accept the draft without a rewrite four times in five.",
    "target": "Sales ops wants it under 500 a month."
   },
   "domain": "Sales",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 1200,
     "one_line": 120,
     "two_lines": 400
    },
    "base_turns": 8,
    "difficulty_weights": [
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     2.1,
     2.5
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": false,
      "pass_high": 0.92,
      "turn_multiplier": 0.85
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.7,
      "turn_multiplier": 1.1
     },
     "opus_5": {
      "measured": false,
      "pass_high": 0.9,
      "turn_multiplier": 0.9
     },
     "sonnet_5": {
      "measured": false,
      "pass_high": 0.85,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 2,
    "pause_every": 2,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 1800,
    "task_input_tokens": 6000,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 150,
    "tool_result_tokens": 300
   },
   "has_table": false,
   "icon": "user-round",
   "id": "sales_calls",
   "interaction": "person_in_loop",
   "measured_on": {
    "benchmarks": [
     "Cache duration on the triage job with inserted pauses; this workload itself is not measured"
    ],
    "models": [],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "audio_transcript"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "sonnet_5",
    "orchestrator": false,
    "pause_pattern": "toward_hour",
    "prompt_written_for": "sonnet_5",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "sales",
    "audio-transcript",
    "person-waits",
    "cache-lifetime",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 500,
    "quality_floor": 0.8
   },
   "title": "Sales call summarizer",
   "traps_featured": [
    "one_hour_everywhere",
    "memo"
   ],
   "workflow": {
    "check": "person",
    "pause_pattern": "toward_hour",
    "size_in_windows": 0.4,
    "steps_per_task": 8,
    "tasks_per_month": 4000
   },
   "zippy": {
    "case": "The team runs the 5-minute default and pays a full re-write after every meeting.",
    "what": "One long transcript, a draft, then a conversation with pauses while the rep is in the next meeting. Minutes to an hour between turns.",
    "why": "The cache forgets after 5 minutes by default. When a person waits, the lifetime you buy decides whether each return re-writes the whole transcript."
   }
  },
  {
   "calibration": {
    "anchors": [
     {
      "config": {
       "model": "haiku_4_5"
      },
      "cost_fraction": 0.1,
      "cost_fraction_key": "ci_bench.gpqa_haiku_cost_fraction",
      "label": "Haiku costs a tenth of Opus per question",
      "tolerance": 0.25,
      "versus": {
       "model": "opus_5"
      }
     },
     {
      "config": {
       "model": "haiku_4_5"
      },
      "label": "Haiku accuracy",
      "pass": 0.63,
      "pass_key": "ci_bench.gpqa_haiku_accuracy"
     },
     {
      "config": {
       "model": "opus_5"
      },
      "label": "Opus accuracy",
      "pass": 0.92,
      "pass_key": "ci_bench.gpqa_opus_accuracy"
     }
    ],
    "base": {},
    "tolerance": 0.2
   },
   "customer_lines": {
    "objection": "Researchers ask it hard science questions all day. The small model is cheap and wrong a third of the time; the big one is right and ten times the price.",
    "quality": "R&D needs nine answers in ten right.",
    "target": "R&D wants it under 1,600 a month."
   },
   "domain": "R&D",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 700,
     "one_line": 60,
     "two_lines": 200
    },
    "base_turns": 2,
    "difficulty_weights": [
     0.55,
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.3,
     1.4,
     1.5,
     2.1,
     2.5
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.0,
    "models": {
     "fable_5_1": {
      "measured": false,
      "pass_high": 0.94,
      "turn_multiplier": 1.0
     },
     "haiku_4_5": {
      "measured": true,
      "output_multiplier": 0.42,
      "pass_high": 0.63,
      "turn_multiplier": 1.0
     },
     "opus_5": {
      "measured": true,
      "pass_high": 0.92,
      "turn_multiplier": 1.0
     },
     "sonnet_5": {
      "measured": false,
      "output_multiplier": 0.8,
      "pass_high": 0.8,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 0,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 1200,
    "task_input_tokens": 400,
    "tasks_per_session": 1,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 1400,
    "tool_result_tokens": 0
   },
   "has_table": false,
   "icon": "flask-conical",
   "id": "science_qa",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "GPQA Diamond, 198-question Diamond subset"
    ],
    "models": [
     "haiku_4_5",
     "opus_5"
    ],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "text"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": true,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "haiku_4_5",
    "orchestrator": false,
    "prompt_written_for": "haiku_4_5",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": false,
    "tool_catalog": 0,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "rnd",
    "text",
    "advisor",
    "single-turn",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 1600,
    "quality_floor": 0.9
   },
   "title": "Hard science Q&A",
   "traps_featured": [
    "second_model_first",
    "per_token"
   ],
   "workflow": {
    "check": "number",
    "pause_pattern": "none",
    "size_in_windows": 0.05,
    "steps_per_task": 2,
    "tasks_per_month": 20000
   },
   "zippy": {
    "case": "The team wants the small model with a frontier advisor bolted on.",
    "what": "A question, a reasoned answer, no tools. Single turn.",
    "why": "On the graduate-level benchmark the small model answered at a tenth of the cost with much lower accuracy. An advisor fits poorly when there is nothing to plan."
   }
  },
  {
   "calibration": {
    "anchors": [
     {
      "config": {},
      "label": "session cost with caching on",
      "session_cost": 0.59,
      "session_cost_key": "ci_cache.triage_run_cost"
     },
     {
      "config": {},
      "label": "caching saves the measured share",
      "saving": 0.83,
      "saving_key": "ci_cache.triage_saving",
      "versus": {
       "caching": false
      }
     },
     {
      "config": {
       "timestamp_prefix": true
      },
      "label": "a status line ahead of the prefix",
      "session_cost": 4.24,
      "session_cost_key": "ci_cache.status_line_run_cost"
     }
    ],
    "base": {
     "caching": true,
     "prompt_written_for": "sonnet_5",
     "timestamp_prefix": false
    },
    "caching_saving": 0.83,
    "caching_saving_key": "ci_cache.triage_saving",
    "model": "sonnet_5",
    "session_cost_cached": 0.59,
    "session_cost_cached_key": "ci_cache.triage_run_cost",
    "status_line_session_cost": 4.24,
    "status_line_session_cost_key": "ci_cache.status_line_run_cost",
    "tolerance": 0.15
   },
   "customer_lines": {
    "objection": "It reads every bug report with the screenshots and labels it for engineering. It works, and the bill is climbing every month.",
    "quality": "We cannot go below 85 of 100 labels right, or engineering stops trusting it.",
    "target": "Finance wants it under 400 a month."
   },
   "domain": "Support",
   "effort_scale_overrides": {},
   "engine": {
    "answer_tokens": {
     "memo": 1600,
     "one_line": 60,
     "two_lines": 300
    },
    "base_turns": 4,
    "difficulty_weights": [
     0.6,
     0.65,
     0.7,
     0.75,
     0.8,
     0.85,
     0.9,
     0.9,
     0.95,
     1.0,
     1.0,
     1.05,
     1.1,
     1.15,
     1.2,
     1.25,
     1.35,
     1.5,
     2.0,
     2.4
    ],
    "long_session_threshold_turns": 120,
    "memo_turn_multiplier": 1.3,
    "models": {
     "fable_5_1": {
      "measured": true,
      "pass_high": 0.9,
      "turn_multiplier": 0.85
     },
     "haiku_4_5": {
      "measured": false,
      "pass_high": 0.7,
      "turn_multiplier": 1.4
     },
     "opus_5": {
      "measured": true,
      "pass_high": 0.9,
      "turn_multiplier": 0.9
     },
     "sonnet_5": {
      "measured": true,
      "pass_high": 0.85,
      "turn_multiplier": 1.0
     }
    },
    "own_tools": 2,
    "pause_minutes": {
     "minutes": 6,
     "none": 0,
     "over_hour": 90,
     "toward_hour": 45
    },
    "pause_share": 0.05,
    "pause_share_key": "ci_cache.one_hour_pause_share_rule",
    "sample_tasks": 100,
    "system_tokens": 1800,
    "task_input_tokens": 480,
    "tasks_per_session": 20,
    "tokens_per_tool": 60,
    "tool_call_output_tokens": 120,
    "tool_result_tokens": 170
   },
   "has_table": false,
   "icon": "life-buoy",
   "id": "triage",
   "interaction": "unattended",
   "measured_on": {
    "benchmarks": [
     "Issue-triage agent",
     "20 real bug reports with screenshots"
    ],
    "models": [
     "sonnet_5",
     "opus_5",
     "fable_5_1"
    ],
    "month": "August 2026",
    "month_key": "ci_bench.measurement_month"
   },
   "media": [
    "text",
    "screenshots"
   ],
   "shape": "knowledge",
   "starting_config": {
    "advisor": false,
    "answer_length": "two_lines",
    "batch": false,
    "cache_lifetime": "five_min",
    "caching": false,
    "context_lifecycle": "none",
    "data_file": false,
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "sonnet_5",
    "orchestrator": false,
    "prompt_written_for": "sonnet_4_6",
    "rerun_failures": false,
    "task_budget": "off",
    "timestamp_prefix": true,
    "tool_catalog": 2,
    "tool_search": false,
    "workers": 1
   },
   "tags": [
    "support",
    "screenshots",
    "cache-dominated",
    "unattended",
    "knowledge"
   ],
   "thresholds": {
    "cost_target": 400,
    "quality_floor": 0.85
   },
   "title": "Support triage agent",
   "traps_featured": [
    "timestamp",
    "caching_always",
    "memo",
    "verify_twice",
    "paste_table",
    "one_hour_everywhere",
    "context_editing_short",
    "mid_session_change",
    "high_effort_hard"
   ],
   "workflow": {
    "check": "ticket",
    "pause_pattern": "none",
    "size_in_windows": 0.35,
    "steps_per_task": 4,
    "tasks_per_month": 5000
   },
   "zippy": {
    "case": "The team turned caching on late, printed the date at the top of the prompt, asked for a two-line summary, and never audited the prompt after moving to the newer model.",
    "what": "A queue of bug reports with screenshots, worked through in one long session by an agent with a couple of tools. Short loops, a few steps per ticket.",
    "why": "Every free lever on the page was measured on this job, so nearly every move you make here has a number behind it."
   }
  }
 ],
 "shape": [
  {
   "control": {
    "max": 60,
    "min": 2,
    "step": 1,
    "type": "stepper"
   },
   "icon": "list-ordered",
   "id": "steps_per_task",
   "label": "Steps per task",
   "short": "Steps",
   "zippy": {
    "example": "\"It reads the ticket, looks at two screenshots, checks the repo, then decides.\"",
    "what": "How many times the agent goes around the loop for one task. Each step re-sends the whole conversation.",
    "why": "Task cost grows with roughly the square of the step count without caching."
   }
  },
  {
   "control": {
    "options": [
     {
      "label": "No",
      "value": "none"
     },
     {
      "label": "Minutes",
      "value": "minutes"
     },
     {
      "label": "Toward an hour",
      "value": "toward_hour"
     },
     {
      "label": "Over an hour",
      "value": "over_hour"
     }
    ],
    "type": "select"
   },
   "icon": "user-round",
   "id": "pause_pattern",
   "label": "Does a person wait between steps?",
   "short": "Pauses",
   "zippy": {
    "example": "\"The rep reads the draft, then comes back after lunch.\"",
    "what": "Whether the agent runs on its own or pauses for a human between steps, and for how long.",
    "why": "The cache forgets after 5 minutes by default. The pause pattern decides the cache lifetime."
   }
  },
  {
   "control": {
    "options": [
     {
      "label": "Tests pass",
      "value": "tests"
     },
     {
      "label": "Ticket closed",
      "value": "ticket"
     },
     {
      "label": "Number matches",
      "value": "number"
     },
     {
      "label": "A person reviews",
      "value": "person"
     },
     {
      "label": "Can't tell",
      "value": "none"
     }
    ],
    "type": "select"
   },
   "icon": "check-circle",
   "id": "check",
   "label": "How would you know it worked?",
   "short": "Check",
   "zippy": {
    "example": "\"If the tests pass, we ship it.\"",
    "what": "Whether something says right or wrong without a human reading it.",
    "why": "With a check you can run cheap first and re-run only failures. Without one, the re-run and budget levers are guesses."
   }
  },
  {
   "control": {
    "options": [
     {
      "label": "A tenth of a window",
      "value": 0.1
     },
     {
      "label": "A third of a window",
      "value": 0.33
     },
     {
      "label": "Half a window",
      "value": 0.5
     },
     {
      "label": "One window",
      "value": 1
     },
     {
      "label": "2 windows",
      "value": 2
     },
     {
      "label": "3 windows",
      "value": 3
     },
     {
      "label": "6 windows",
      "value": 6
     },
     {
      "label": "12 windows",
      "value": 12
     }
    ],
    "type": "select"
   },
   "icon": "maximize",
   "id": "size_in_windows",
   "label": "How big is the work?",
   "short": "Size",
   "zippy": {
    "example": "\"It has to read the whole contract set.\"",
    "what": "The size of one task compared with what the model can hold at once, a context window. Above 1x it no longer fits.",
    "why": "Work bigger than one window is where splitting across workers pays; below it, splitting costs more."
   }
  },
  {
   "control": {
    "max": 1000000,
    "min": 1,
    "step": 1,
    "type": "number"
   },
   "icon": "calendar",
   "id": "tasks_per_month",
   "label": "Tasks per month",
   "short": "Tasks",
   "zippy": {
    "example": "\"About three thousand tickets a month.\"",
    "what": "How many tasks the customer runs in a month. It scales the bill, not the shape.",
    "why": "The monthly number is what the CFO sees."
   }
  },
  {
   "control": {
    "max": 1,
    "min": 0,
    "step": 0.01,
    "type": "number"
   },
   "icon": "shield-check",
   "id": "quality_floor",
   "label": "Quality the customer will not go below",
   "short": "Floor",
   "zippy": {
    "example": "\"We cannot drop below nine in ten correct labels.\"",
    "what": "The pass rate the customer says they need. Drawn as a line on the charts.",
    "why": "Every customer's line is different, and the right tradeoff depends on where it is."
   }
  },
  {
   "control": {
    "max": 10000000,
    "min": 0,
    "step": 100,
    "type": "number"
   },
   "icon": "target",
   "id": "cost_target",
   "label": "Monthly bill the customer wants",
   "short": "Target",
   "zippy": {
    "example": "\"The CFO wants it under four thousand a month.\"",
    "what": "The monthly number the customer has in mind. Drawn as a line on the charts.",
    "why": "Some targets are reachable with free levers alone. Some need a tradeoff, and the honest answer says which."
   }
  }
 ],
 "staleness_days": 90,
 "traps": [
  {
   "anchor": "compare_models",
   "belief": "Two bars: the frontier model tall and thin, the cheap model short.",
   "card": {
    "correct": "model",
    "correct_changes": {
     "effort": "low"
    },
    "correct_value": "fable_5_1",
    "kind": "pick_lever",
    "options": [
     "model",
     "effort",
     "batch",
     "max_tokens"
    ],
    "wrong_pick_headline": "Per-token price is not the bill."
   },
   "claim": "A model with a higher per-token price can cost less per solved task.",
   "experiment": "Will the cheaper per-token model cost less per solved task?",
   "feel": "A thin tall glass holds less than a wide short one.",
   "focus": "frontier",
   "headline": "Pay per solved task, not per token.",
   "id": "per_token",
   "lever": "model",
   "measured": "Price tall by tokens wide. The cheap model's rectangle is wider and its area larger, and its failed tasks stack beside it as hollow rectangles that still have height.",
   "scenario": "coding",
   "sentence": "Fable 5.1 at low effort solved 88.6% of the coding subset for $0.54 per solved task; Sonnet 5 at its default solved 77.4% for $0.84, despite a per-token price five times lower.",
   "setup": {
    "effort": "high",
    "max_tokens": "cap_64k",
    "model": "sonnet_5"
   },
   "title": "Per token cheaper is not always cheaper"
  },
  {
   "anchor": "budgets",
   "belief": "The bar gets shorter.",
   "card": {
    "after": {
     "max_tokens": "cap_16k"
    },
    "before": {
     "max_tokens": "cap_64k"
    },
    "correct": "max_tokens",
    "kind": "what_changed",
    "options": [
     "max_tokens",
     "effort",
     "rerun_failures",
     "task_budget"
    ],
    "wrong_pick_headline": "The cap is a safety net, not a saving."
   },
   "claim": "Lowering max_tokens cuts cost per attempt but not cost per solved task.",
   "experiment": "Will capping the answer length save money?",
   "feel": "Cutting the rope shorter does not make the climb cheaper.",
   "focus": "outcome",
   "headline": "Cheaper attempts, same cost per solved task.",
   "id": "lower_max_tokens",
   "lever": "max_tokens",
   "measured": "The spend bar and the solved-count bar shrink by the same amount. The per-solved bar holds. Capped cells show the cut mark.",
   "scenario": "coding",
   "sentence": "A 16k cap ended 15% of Opus 5's attempts; only 9 of 117 capped attempts still passed, so cost per solved task was $21 against $22 at 64k.",
   "setup": {
    "effort": "high",
    "model": "opus_5"
   },
   "title": "Capping max_tokens saves per attempt, not per solved task"
  },
  {
   "anchor": "effort",
   "belief": "A second model's point sits below the current one.",
   "card": {
    "correct": "effort",
    "correct_value": "medium",
    "kind": "pick_lever",
    "options": [
     "effort",
     "advisor",
     "orchestrator",
     "model"
    ],
    "wrong_pick_headline": "Draw the single-model curve first."
   },
   "claim": "A cheaper-looking multi-model setup can cost more than the single model at lower effort.",
   "experiment": "Will switching models save more than turning effort down?",
   "feel": "You reached for a new tool before turning the dial you had.",
   "focus": "frontier",
   "headline": "Turn the dial you have first.",
   "id": "model_before_effort",
   "lever": "effort",
   "measured": "The current model's curve at medium sits below the second model's point.",
   "scenario": "coding",
   "sentence": "In the internal measurements a multi-model configuration that looked cheaper than the default single model cost more than that same model at lower effort. On coding, Opus 5 gave up 2% at medium for half the cost.",
   "setup": {
    "effort": "high",
    "model": "opus_5"
   },
   "title": "Sweep effort before switching models"
  },
  {
   "anchor": "advisor",
   "belief": "The paired point sits above the single model's whole curve.",
   "card": {
    "correct": "effort",
    "correct_changes": {
     "advisor": false
    },
    "correct_value": "medium",
    "kind": "pick_lever",
    "options": [
     "advisor",
     "effort",
     "model",
     "orchestrator"
    ],
    "wrong_pick_headline": "Sweep effort before adding a model."
   },
   "claim": "A multi-model configuration must beat the single model's whole effort curve.",
   "experiment": "Does a frontier advisor make the cheap model as good?",
   "feel": "The bar it must clear is drawn before the jump.",
   "focus": "frontier",
   "headline": "A second model must clear the whole curve.",
   "id": "second_model_first",
   "lever": "advisor",
   "measured": "The single-model curve is drawn first; the multi-model point has to clear it, and here it does not.",
   "scenario": "coding",
   "sentence": "An advisor pays off when priced well above the executor and actually consulted. The coding pairing scored 3.5% over Opus 5 alone with about 2 consultations per attempt; the chart-reading pairing matched the advisor's model alone for about 2.6x the price.",
   "setup": {
    "advisor": true,
    "effort": "high",
    "model": "sonnet_5"
   },
   "title": "An advisor must clear the whole effort curve first"
  },
  {
   "anchor": "effort",
   "belief": "A rising line from low to high.",
   "card": {
    "correct": "effort",
    "correct_value": "medium",
    "kind": "pick_lever",
    "options": [
     "effort",
     "model",
     "advisor",
     "task_budget"
    ],
    "wrong_pick_headline": "The curve is flat here."
   },
   "claim": "On research and knowledge work, effort above medium buys nothing measurable.",
   "experiment": "Does hard work need high effort?",
   "feel": "Pushing harder on a door that is already open.",
   "focus": "frontier",
   "headline": "Score holds while cost climbs.",
   "id": "high_effort_hard",
   "lever": "effort",
   "measured": "The score line holds (the held animation) while the cost bars climb underneath.",
   "scenario": "triage",
   "sentence": "On the research and knowledge-work benchmarks medium matched the default's accuracy at about 69% to 87% of its cost, and the default bought nothing measurable over medium. On deep research Fable 5.1 scored nearly the same at low, medium and high while cost per task rose from $4.66 to $7.12.",
   "setup": {
    "caching": true,
    "effort": "high",
    "prompt_written_for": "sonnet_5",
    "timestamp_prefix": false
   },
   "title": "Hard work does not always need high effort"
  },
  {
   "anchor": "cache_breaks",
   "belief": "One extra sliver per bar.",
   "card": {
    "after": {
     "caching": true,
     "timestamp_prefix": true
    },
    "before": {
     "caching": true,
     "timestamp_prefix": false
    },
    "correct": "timestamp_prefix",
    "kind": "what_changed",
    "options": [
     "timestamp_prefix",
     "tool_catalog",
     "answer_length",
     "cache_lifetime"
    ],
    "wrong_pick_headline": "The cache matches from the first byte."
   },
   "claim": "Per-request text ahead of the stable prefix turns every request into a full cache write.",
   "experiment": "Is a timestamp at the top of the prompt harmless?",
   "feel": "One drip that resets the whole bucket every time.",
   "focus": "pipeline",
   "headline": "Every request now a full cache write.",
   "id": "timestamp",
   "lever": "timestamp_prefix",
   "measured": "In the pipeline the flagged block at the top breaks the cache shading on every turn. Every bar turns entirely cache-write, taller than no caching at all.",
   "scenario": "triage",
   "sentence": "On the triage run a 25-token status line at the front of the system prompt cost $4.24 per run instead of $0.59, more than running with caching off.",
   "setup": {
    "caching": true,
    "timestamp_prefix": true
   },
   "title": "A timestamp at the top breaks every cache read"
  },
  {
   "anchor": "cache_breaks",
   "belief": "A blip.",
   "card": {
    "after": {
     "caching": true,
     "mid_session_change": true
    },
    "before": {
     "caching": true,
     "mid_session_change": false
    },
    "correct": "mid_session_change",
    "kind": "what_changed",
    "options": [
     "mid_session_change",
     "tool_catalog",
     "answer_length",
     "batch"
    ],
    "wrong_pick_headline": "Make cache-breaking changes at natural breaks."
   },
   "claim": "Changing effort or the tool list mid-session invalidates the cached prefix.",
   "experiment": "Is changing effort or adding a tool mid-session just a blip?",
   "feel": "Knocking the bottom out of a stack.",
   "focus": "pipeline",
   "headline": "The cached prefix collapses and rebuilds.",
   "id": "mid_session_change",
   "lever": "effort",
   "measured": "The cached prefix shading collapses to zero at that turn and rebuilds as one tall write bar.",
   "scenario": "triage",
   "sentence": "An effort change and an added tool made mid-session rewrote 39,000 and 60,000 cached tokens, and those sessions cost $0.95 against $0.81 with no change; the same changes on the first request after compaction cost $0.75.",
   "setup": {
    "caching": true,
    "timestamp_prefix": false
   },
   "title": "Change effort or tools at a break, not mid-session"
  },
  {
   "anchor": "prompt_audit",
   "belief": "Same bars, more confidence.",
   "card": {
    "correct": "prompt_written_for",
    "correct_value": "sonnet_5",
    "kind": "pick_lever",
    "options": [
     "prompt_written_for",
     "effort",
     "model",
     "answer_length"
    ],
    "wrong_pick_headline": "The prompt was written for a model you no longer run."
   },
   "claim": "Old-model instructions in a prompt cost extra rounds on a newer model with no accuracy gain.",
   "experiment": "Is asking the model to verify twice safer?",
   "feel": "Walking the corridor twice to arrive at the same door.",
   "focus": "pipeline",
   "headline": "Extra rounds, same answers.",
   "id": "verify_twice",
   "lever": "prompt_written_for",
   "measured": "More turns per task appear in the pipeline (extra tool rounds); the outcome strip holds.",
   "scenario": "triage",
   "sentence": "Prompts written for Opus 4.8 cost 36% more per ticket on Opus 5 for no change in accuracy; removing 'verify twice' alone cut a third. The audit made the newer model both 14.0% cheaper and more accurate, 97% of tickets up from 92%.",
   "setup": {
    "caching": true,
    "prompt_written_for": "sonnet_4_6",
    "timestamp_prefix": false
   },
   "title": "'Verify twice' costs extra rounds, not extra accuracy"
  },
  {
   "anchor": "data_files",
   "belief": "A big first bar.",
   "card": {
    "correct": "data_file",
    "kind": "pick_lever",
    "options": [
     "data_file",
     "caching",
     "effort",
     "model"
    ],
    "wrong_pick_headline": "Let the model query the file with code."
   },
   "claim": "A table pasted into the prompt costs an order of magnitude more and answers worse than a file queried by code.",
   "experiment": "Is pasting the table into the prompt good enough?",
   "feel": "Carrying the filing cabinet to every meeting.",
   "focus": "pipeline",
   "headline": "Table re-read every turn, answers wrong.",
   "id": "paste_table",
   "lever": "data_file",
   "measured": "A huge base band on every pipeline block, and most outcome cells hollow.",
   "scenario": "triage",
   "sentence": "Pasted, the table was about 91,000 input tokens on every request and Sonnet 5 answered 6 of 25 questions. Uploaded with code execution it answered 25 of 25 for about 8.3% of the cost.",
   "setup": {
    "caching": true,
    "data_file": false,
    "has_table": true,
    "timestamp_prefix": false
   },
   "title": "Upload the table; pasting it bills every turn"
  },
  {
   "anchor": "compare_models",
   "belief": "Twenty similar bars.",
   "card": {
    "correct": "model",
    "correct_value": "opus_5",
    "kind": "pick_lever",
    "options": [
     "model",
     "task_budget",
     "effort",
     "batch"
    ],
    "wrong_pick_headline": "Price the tail, not the median."
   },
   "claim": "The bill is decided by the hardest tenth of tasks, and a failed task still bills its tokens.",
   "experiment": "Does judging on the typical task predict the bill?",
   "feel": "The bill is decided by the two heavy boxes.",
   "focus": "tail",
   "headline": "Two heavy boxes decide the bill.",
   "id": "judge_median",
   "lever": "model",
   "measured": "Sorted bars, two tall ones carry most of the spend; hollow bars still have height.",
   "scenario": "coding",
   "sentence": "On a 20-problem research run, 2 of 20 problems carried 43% of the spend. On the typical task every model looks similar and the cheapest looks best; the bill is decided by the tasks the cheaper model fails, because a failed task still bills its tokens, then the retry, then whatever the failure costs downstream.",
   "setup": {
    "model": "sonnet_5"
   },
   "title": "Judge on the hardest tenth, not the typical task"
  },
  {
   "anchor": "budgets",
   "belief": "A thicker output band once.",
   "card": {
    "correct": "answer_length",
    "correct_value": "one_line",
    "kind": "pick_lever",
    "options": [
     "answer_length",
     "effort",
     "model",
     "context_lifecycle"
    ],
    "wrong_pick_headline": "Ask for the answer you will read."
   },
   "claim": "A longer final answer costs more on every later turn and scores the same.",
   "experiment": "Is the long, thorough-looking memo a one-time cost?",
   "feel": "Your long answer is read back to you every turn, and billed.",
   "focus": "pipeline",
   "headline": "Long answers echo through every later turn.",
   "id": "memo",
   "lever": "answer_length",
   "measured": "The output block grows, then flows into the next turn's context, and the next; each bar taller than the last. The outcome strip holds.",
   "scenario": "triage",
   "sentence": "The memo used 6 times the output tokens and cost 2.8x the one-line answer; all three formats scored within run-to-run noise. Ask for the answer you will read, not the one that looks thorough.",
   "setup": {
    "answer_length": "memo",
    "caching": true,
    "timestamp_prefix": false
   },
   "title": "The memo looks thorough but bills every later turn"
  },
  {
   "anchor": "context_lifecycle",
   "belief": "Bars shrink.",
   "card": {
    "correct": "context_lifecycle",
    "correct_value": "none",
    "kind": "pick_lever",
    "options": [
     "context_lifecycle",
     "answer_length",
     "effort",
     "batch"
    ],
    "wrong_pick_headline": "Only a long session needs the lifecycle levers."
   },
   "claim": "Context lifecycle levers only pay on sessions long enough to need them; context editing cost more on the short run.",
   "experiment": "Does context editing help every loop?",
   "feel": "Cleaning the desk more often than you use it.",
   "focus": "pipeline",
   "headline": "Cleaning the desk more than you use it.",
   "id": "context_editing_short",
   "lever": "context_lifecycle",
   "measured": "On a short loop, extra write bands appear after each clear; the total is higher.",
   "scenario": "triage",
   "sentence": "On the 20-issue run the context levers saved nothing and context editing cost 74% more. On the long run the prune saved 39% and compaction 32%.",
   "setup": {
    "caching": true,
    "context_lifecycle": "context_editing",
    "timestamp_prefix": false
   },
   "title": "Context editing only pays on a long session"
  },
  {
   "anchor": "cache_duration",
   "belief": "Same bars, safer.",
   "card": {
    "correct": "cache_lifetime",
    "correct_value": "five_min",
    "kind": "pick_lever",
    "options": [
     "cache_lifetime",
     "caching",
     "effort",
     "batch"
    ],
    "wrong_pick_headline": "Count the pauses first."
   },
   "claim": "With no pauses the 1-hour cache costs more than the 5-minute default.",
   "experiment": "Is the 1-hour cache worth turning on everywhere?",
   "feel": "Paying for a longer parking ticket you never use.",
   "focus": "pipeline",
   "headline": "Longer ticket, never used.",
   "id": "one_hour_everywhere",
   "lever": "cache_lifetime",
   "measured": "Every write sliver is thicker; with no pauses the total is higher.",
   "scenario": "triage",
   "sentence": "With no pauses the 5-minute default cost 15% less than the 1-hour setting on Sonnet 5 and 11% less on Opus 5. The 1-hour setting pays once about 5% of turns follow a pause between 5 minutes and an hour.",
   "setup": {
    "cache_lifetime": "one_hour",
    "caching": true,
    "pause_pattern": "none",
    "timestamp_prefix": false
   },
   "title": "The 1-hour cache pays only when turns follow pauses"
  },
  {
   "anchor": "rerun",
   "belief": "Every hollow cell fills on the second pass.",
   "card": {
    "correct": "check",
    "correct_value": "tests",
    "kind": "pick_lever",
    "options": [
     "check",
     "rerun_failures",
     "effort",
     "task_budget"
    ],
    "wrong_pick_headline": "Name the check before the tradeoff."
   },
   "claim": "Run-cheap-then-re-run needs a failure signal; a checker that passes bad work lets failures through.",
   "experiment": "Will re-running failures fix them without a check?",
   "feel": "Retrying what you cannot see.",
   "focus": "outcome",
   "headline": "Retrying what you cannot see.",
   "id": "rerun_without_check",
   "lever": "rerun_failures",
   "measured": "The second pass runs on hollow cells a bad checker marked solid; failures pass through.",
   "scenario": "coding",
   "sentence": "Two conditions apply: you need a failure signal (on the benchmark, its own tests), and every first-pass failure takes two runs' worth of wall-clock time. With a check, Opus 5 at low then the default solved about 93% for about $0.45 each.",
   "setup": {
    "check": "none",
    "effort": "low",
    "model": "opus_5",
    "rerun_failures": true
   },
   "title": "Re-run failures only when a check can see them"
  },
  {
   "anchor": "cache",
   "belief": "Every bar shrinks once the switch is on.",
   "card": {
    "correct": "effort",
    "correct_value": "medium",
    "kind": "pick_lever",
    "options": [
     "caching",
     "effort",
     "cache_lifetime",
     "tool_search"
    ],
    "wrong_pick_headline": "Reads are already at the ceiling."
   },
   "claim": "When cache reads are already high, the caching switch changes nothing and the tail is where the cost is.",
   "experiment": "Is caching always the answer?",
   "feel": "The lever is already pulled; look elsewhere.",
   "focus": "tail",
   "headline": "The lever is already pulled.",
   "id": "caching_always",
   "lever": "task_budget",
   "measured": "The cache meter is already high; the caching switch produces the held animation everywhere; the tail chart is where the height is.",
   "scenario": "triage",
   "sentence": "Production agent loops read a median 84.2% of their input from the cache and the top harnesses 94% or more. Below about 80%, look for something breaking the cache; above it, the money is in the tail and the tradeoff levers.",
   "setup": {
    "answer_length": "one_line",
    "caching": true,
    "effort": "high",
    "prompt_written_for": "sonnet_5",
    "timestamp_prefix": false,
    "tool_search": true
   },
   "title": "Caching is not always the answer"
  }
 ],
 "verified_range": [
  "2026-09-16",
  "2026-09-16"
 ],
 "version": "bb46498f19a2"
}
