{"schema_version": 1, "benchmarks": {"workplan": {"source_run": "t3-v2", "agent_model": "Claude Sonnet 4.5", "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"], "judge_model": "Claude Haiku 4.5", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "backend": "bedrock", "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "duration_basis": "Agent wall time includes model latency and queueing; capped attempts remain included", "task_origin": "Authored benchmark scenarios and supplied fixtures; not verified customer production cases", "source_url": "/eval-data/workplan/index.json", "tasks": [{"name": "ledger-migration-resume", "arms": {"base": {"label": "Without skill", "attempts": 6, "mean_estimated_api_cost_usd": 0.428328, "mean_agent_wall_s": 477.433333, "mean_model_wait_s": 464.8, "mean_tool_s": 12.583333, "stop_reasons": {"turn_cap": 5, "end_turn": 1}, "source_attempt_ids": ["workplan.ledger-migration-resume:base:s1", "workplan.ledger-migration-resume:base:s2", "workplan.ledger-migration-resume:base:s3", "workplan.ledger-migration-resume:base:s4", "workplan.ledger-migration-resume:base:s5", "workplan.ledger-migration-resume:base:s6"]}, "loaded": {"label": "Skill text in prompt", "attempts": 6, "mean_estimated_api_cost_usd": 0.508799, "mean_agent_wall_s": 513.633333, "mean_model_wait_s": 499.633333, "mean_tool_s": 14.0, "stop_reasons": {"end_turn": 1, "turn_cap": 5}, "source_attempt_ids": ["workplan.ledger-migration-resume:loaded:s1", "workplan.ledger-migration-resume:loaded:s2", "workplan.ledger-migration-resume:loaded:s3", "workplan.ledger-migration-resume:loaded:s4", "workplan.ledger-migration-resume:loaded:s5", "workplan.ledger-migration-resume:loaded:s6"]}}}, {"name": "payroll-export-batch", "arms": {"base": {"label": "Without skill", "attempts": 6, "mean_estimated_api_cost_usd": 0.488342, "mean_agent_wall_s": 433.683333, "mean_model_wait_s": 420.35, "mean_tool_s": 13.366667, "stop_reasons": {"turn_cap": 6}, "source_attempt_ids": ["workplan.payroll-export-batch:base:s1", "workplan.payroll-export-batch:base:s2", "workplan.payroll-export-batch:base:s3", "workplan.payroll-export-batch:base:s4", "workplan.payroll-export-batch:base:s5", "workplan.payroll-export-batch:base:s6"]}, "loaded": {"label": "Skill text in prompt", "attempts": 6, "mean_estimated_api_cost_usd": 0.57021, "mean_agent_wall_s": 457.766667, "mean_model_wait_s": 442.666667, "mean_tool_s": 15.083333, "stop_reasons": {"turn_cap": 6}, "source_attempt_ids": ["workplan.payroll-export-batch:loaded:s1", "workplan.payroll-export-batch:loaded:s2", "workplan.payroll-export-batch:loaded:s3", "workplan.payroll-export-batch:loaded:s4", "workplan.payroll-export-batch:loaded:s5", "workplan.payroll-export-batch:loaded:s6"]}}}, {"name": "status-page-demo-cut", "arms": {"base": {"label": "Without skill", "attempts": 6, "mean_estimated_api_cost_usd": 0.2606, "mean_agent_wall_s": 237.333333, "mean_model_wait_s": 230.433333, "mean_tool_s": 6.916667, "stop_reasons": {"end_turn": 5, "turn_cap": 1}, "source_attempt_ids": ["workplan.status-page-demo-cut:base:s1", "workplan.status-page-demo-cut:base:s2", "workplan.status-page-demo-cut:base:s3", "workplan.status-page-demo-cut:base:s4", "workplan.status-page-demo-cut:base:s5", "workplan.status-page-demo-cut:base:s6"]}, "loaded": {"label": "Skill text in prompt", "attempts": 6, "mean_estimated_api_cost_usd": 0.250307, "mean_agent_wall_s": 201.733333, "mean_model_wait_s": 194.283333, "mean_tool_s": 7.433333, "stop_reasons": {"end_turn": 6}, "source_attempt_ids": ["workplan.status-page-demo-cut:loaded:s1", "workplan.status-page-demo-cut:loaded:s2", "workplan.status-page-demo-cut:loaded:s3", "workplan.status-page-demo-cut:loaded:s4", "workplan.status-page-demo-cut:loaded:s5", "workplan.status-page-demo-cut:loaded:s6"]}}}], "legacy_weighting_caveat": "Legacy rubric aggregation matched criterion labels exactly and defaulted unmatched weights to 1. Judges often appended weight annotations, so intended rubric weights were often not applied. Retained values reproduce that historical algorithm; they must not be described as correctly weighted rubric scores. Overall judge scores are independently recorded and unaffected by this aggregation defect.", "recommended_default_metric": "overall"}, "strip-image-ai-metadata": {"source_run": "t3-v2", "agent_model": "Claude Sonnet 4.5", "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"], "judge_model": "Claude Haiku 4.5", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "backend": "bedrock", "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "duration_basis": "Agent wall time includes model latency and queueing; capped attempts remain included", "task_origin": "Authored benchmark scenarios and supplied fixtures; not verified customer production cases", "source_url": "/eval-data/strip-image-ai-metadata/index.json", "tasks": [{"name": "post-batch-lossless", "arms": {"base": {"label": "Without skill", "attempts": 6, "mean_estimated_api_cost_usd": 0.238691, "mean_agent_wall_s": 200.883333, "mean_model_wait_s": 191.083333, "mean_tool_s": 9.783333, "stop_reasons": {"end_turn": 5, "turn_cap": 1}, "source_attempt_ids": ["strip-image-ai-metadata.post-batch-lossless:base:s1", "strip-image-ai-metadata.post-batch-lossless:base:s2", "strip-image-ai-metadata.post-batch-lossless:base:s3", "strip-image-ai-metadata.post-batch-lossless:base:s4", "strip-image-ai-metadata.post-batch-lossless:base:s5", "strip-image-ai-metadata.post-batch-lossless:base:s6"]}, "loaded": {"label": "Skill text in prompt", "attempts": 6, "mean_estimated_api_cost_usd": 0.071987, "mean_agent_wall_s": 86.766667, "mean_model_wait_s": 83.283333, "mean_tool_s": 3.466667, "stop_reasons": {"end_turn": 6}, "source_attempt_ids": ["strip-image-ai-metadata.post-batch-lossless:loaded:s1", "strip-image-ai-metadata.post-batch-lossless:loaded:s2", "strip-image-ai-metadata.post-batch-lossless:loaded:s3", "strip-image-ai-metadata.post-batch-lossless:loaded:s4", "strip-image-ai-metadata.post-batch-lossless:loaded:s5", "strip-image-ai-metadata.post-batch-lossless:loaded:s6"]}}}, {"name": "print-proofs-colour-critical", "arms": {"base": {"label": "Without skill", "attempts": 6, "mean_estimated_api_cost_usd": 0.34616, "mean_agent_wall_s": 247.4, "mean_model_wait_s": 229.283333, "mean_tool_s": 18.083333, "stop_reasons": {"turn_cap": 6}, "source_attempt_ids": ["strip-image-ai-metadata.print-proofs-colour-critical:base:s1", "strip-image-ai-metadata.print-proofs-colour-critical:base:s2", "strip-image-ai-metadata.print-proofs-colour-critical:base:s3", "strip-image-ai-metadata.print-proofs-colour-critical:base:s4", "strip-image-ai-metadata.print-proofs-colour-critical:base:s5", "strip-image-ai-metadata.print-proofs-colour-critical:base:s6"]}, "loaded": {"label": "Skill text in prompt", "attempts": 6, "mean_estimated_api_cost_usd": 0.290166, "mean_agent_wall_s": 215.1, "mean_model_wait_s": 206.783333, "mean_tool_s": 8.316667, "stop_reasons": {"end_turn": 6}, "source_attempt_ids": ["strip-image-ai-metadata.print-proofs-colour-critical:loaded:s1", "strip-image-ai-metadata.print-proofs-colour-critical:loaded:s2", "strip-image-ai-metadata.print-proofs-colour-critical:loaded:s3", "strip-image-ai-metadata.print-proofs-colour-critical:loaded:s4", "strip-image-ai-metadata.print-proofs-colour-critical:loaded:s5", "strip-image-ai-metadata.print-proofs-colour-critical:loaded:s6"]}}}, {"name": "relist-second-pass", "arms": {"base": {"label": "Without skill", "attempts": 6, "mean_estimated_api_cost_usd": 0.129884, "mean_agent_wall_s": 102.266667, "mean_model_wait_s": 90.766667, "mean_tool_s": 11.5, "stop_reasons": {"end_turn": 6}, "source_attempt_ids": ["strip-image-ai-metadata.relist-second-pass:base:s1", "strip-image-ai-metadata.relist-second-pass:base:s2", "strip-image-ai-metadata.relist-second-pass:base:s3", "strip-image-ai-metadata.relist-second-pass:base:s4", "strip-image-ai-metadata.relist-second-pass:base:s5", "strip-image-ai-metadata.relist-second-pass:base:s6"]}, "loaded": {"label": "Skill text in prompt", "attempts": 6, "mean_estimated_api_cost_usd": 0.141427, "mean_agent_wall_s": 110.083333, "mean_model_wait_s": 104.083333, "mean_tool_s": 6.0, "stop_reasons": {"end_turn": 6}, "source_attempt_ids": ["strip-image-ai-metadata.relist-second-pass:loaded:s1", "strip-image-ai-metadata.relist-second-pass:loaded:s2", "strip-image-ai-metadata.relist-second-pass:loaded:s3", "strip-image-ai-metadata.relist-second-pass:loaded:s4", "strip-image-ai-metadata.relist-second-pass:loaded:s5", "strip-image-ai-metadata.relist-second-pass:loaded:s6"]}}}], "legacy_weighting_caveat": "Legacy rubric aggregation matched criterion labels exactly and defaulted unmatched weights to 1. Judges often appended weight annotations, so intended rubric weights were often not applied. Retained values reproduce that historical algorithm; they must not be described as correctly weighted rubric scores. Overall judge scores are independently recorded and unaffected by this aggregation defect.", "recommended_default_metric": "overall"}, "top-down-comms": {"source_run": "td-v2", "agent_model": "Claude Sonnet 4.5", "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"], "judge_model": "Claude Haiku 4.5", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "backend": "bedrock", "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "duration_basis": "Agent wall time includes model latency and queueing; capped attempts remain included", "task_origin": "Authored benchmark scenarios and supplied fixtures; not verified customer production cases", "source_url": "/eval-data/top-down-comms/index.json", "tasks": [{"name": "buried-blocker-status", "arms": {"base": {"label": "Without skill", "attempts": 3, "mean_estimated_api_cost_usd": 0.069352, "mean_agent_wall_s": 78.933333, "mean_model_wait_s": 77.066667, "mean_tool_s": 1.9, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.buried-blocker-status:base:s1", "top-down-comms.buried-blocker-status:base:s2", "top-down-comms.buried-blocker-status:base:s3"]}, "loaded": {"label": "Skill text in prompt", "attempts": 3, "mean_estimated_api_cost_usd": 0.075428, "mean_agent_wall_s": 94.666667, "mean_model_wait_s": 92.933333, "mean_tool_s": 1.766667, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.buried-blocker-status:loaded:s1", "top-down-comms.buried-blocker-status:loaded:s2", "top-down-comms.buried-blocker-status:loaded:s3"]}, "skill": {"label": "Skill installed for discovery", "attempts": 3, "mean_estimated_api_cost_usd": 0.070026, "mean_agent_wall_s": 80.933333, "mean_model_wait_s": 79.1, "mean_tool_s": 1.9, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.buried-blocker-status:skill:s1", "top-down-comms.buried-blocker-status:skill:s2", "top-down-comms.buried-blocker-status:skill:s3"]}}}, {"name": "client-update-from-thread", "arms": {"base": {"label": "Without skill", "attempts": 3, "mean_estimated_api_cost_usd": 0.051883, "mean_agent_wall_s": 79.3, "mean_model_wait_s": 78.166667, "mean_tool_s": 1.166667, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.client-update-from-thread:base:s1", "top-down-comms.client-update-from-thread:base:s2", "top-down-comms.client-update-from-thread:base:s3"]}, "loaded": {"label": "Skill text in prompt", "attempts": 3, "mean_estimated_api_cost_usd": 0.05515, "mean_agent_wall_s": 69.8, "mean_model_wait_s": 68.7, "mean_tool_s": 1.133333, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.client-update-from-thread:loaded:s1", "top-down-comms.client-update-from-thread:loaded:s2", "top-down-comms.client-update-from-thread:loaded:s3"]}, "skill": {"label": "Skill installed for discovery", "attempts": 3, "mean_estimated_api_cost_usd": 0.061624, "mean_agent_wall_s": 86.133333, "mean_model_wait_s": 84.866667, "mean_tool_s": 1.266667, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.client-update-from-thread:skill:s1", "top-down-comms.client-update-from-thread:skill:s2", "top-down-comms.client-update-from-thread:skill:s3"]}}}, {"name": "renewal-memo-missing-price", "arms": {"base": {"label": "Without skill", "attempts": 3, "mean_estimated_api_cost_usd": 0.064212, "mean_agent_wall_s": 119.666667, "mean_model_wait_s": 117.466667, "mean_tool_s": 2.166667, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.renewal-memo-missing-price:base:s1", "top-down-comms.renewal-memo-missing-price:base:s2", "top-down-comms.renewal-memo-missing-price:base:s3"]}, "loaded": {"label": "Skill text in prompt", "attempts": 3, "mean_estimated_api_cost_usd": 0.052096, "mean_agent_wall_s": 88.5, "mean_model_wait_s": 87.133333, "mean_tool_s": 1.4, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.renewal-memo-missing-price:loaded:s1", "top-down-comms.renewal-memo-missing-price:loaded:s2", "top-down-comms.renewal-memo-missing-price:loaded:s3"]}, "skill": {"label": "Skill installed for discovery", "attempts": 3, "mean_estimated_api_cost_usd": 0.095372, "mean_agent_wall_s": 110.2, "mean_model_wait_s": 108.066667, "mean_tool_s": 2.1, "stop_reasons": {"end_turn": 3}, "source_attempt_ids": ["top-down-comms.renewal-memo-missing-price:skill:s1", "top-down-comms.renewal-memo-missing-price:skill:s2", "top-down-comms.renewal-memo-missing-price:skill:s3"]}}}], "legacy_weighting_caveat": "Legacy rubric aggregation matched criterion labels exactly and defaulted unmatched weights to 1. Judges often appended weight annotations, so intended rubric weights were often not applied. Retained values reproduce that historical algorithm; they must not be described as correctly weighted rubric scores. Overall judge scores are independently recorded and unaffected by this aggregation defect.", "recommended_default_metric": "overall"}, "harness-first": {"source_run": "hf-v1", "agent_model": "Claude Sonnet 4.5", "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"], "judge_model": "Claude Haiku 4.5", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "backend": "bedrock", "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "duration_basis": "Agent wall time includes model latency and queueing; capped attempts remain included", "task_origin": "Authored benchmark scenarios and supplied fixtures; not verified customer production cases", "source_url": "/eval-data/harness-first/index.json", "tasks": [{"name": "finance-agent-revenue", "arms": {"base": {"label": "Without skill", "attempts": 8, "mean_estimated_api_cost_usd": 0.864013, "mean_agent_wall_s": 797.6125, "mean_model_wait_s": 788.3, "mean_tool_s": 9.3125, "stop_reasons": {"end_turn": 8}, "source_attempt_ids": ["harness-first.finance-agent-revenue:base:s1", "harness-first.finance-agent-revenue:base:s2", "harness-first.finance-agent-revenue:base:s3", "harness-first.finance-agent-revenue:base:s4", "harness-first.finance-agent-revenue:base:s5", "harness-first.finance-agent-revenue:base:s6", "harness-first.finance-agent-revenue:base:s7", "harness-first.finance-agent-revenue:base:s8"]}, "skill": {"label": "Skill installed for discovery", "attempts": 8, "mean_estimated_api_cost_usd": 1.023328, "mean_agent_wall_s": 896.3625, "mean_model_wait_s": 885.5375, "mean_tool_s": 10.825, "stop_reasons": {"turn_cap": 4, "end_turn": 4}, "source_attempt_ids": ["harness-first.finance-agent-revenue:skill:s1", "harness-first.finance-agent-revenue:skill:s2", "harness-first.finance-agent-revenue:skill:s3", "harness-first.finance-agent-revenue:skill:s4", "harness-first.finance-agent-revenue:skill:s5", "harness-first.finance-agent-revenue:skill:s6", "harness-first.finance-agent-revenue:skill:s7", "harness-first.finance-agent-revenue:skill:s8"]}}}, {"name": "sales-agent-token-burn", "arms": {"base": {"label": "Without skill", "attempts": 8, "mean_estimated_api_cost_usd": 0.482864, "mean_agent_wall_s": 638.5625, "mean_model_wait_s": 628.95, "mean_tool_s": 9.625, "stop_reasons": {"end_turn": 8}, "source_attempt_ids": ["harness-first.sales-agent-token-burn:base:s1", "harness-first.sales-agent-token-burn:base:s2", "harness-first.sales-agent-token-burn:base:s3", "harness-first.sales-agent-token-burn:base:s4", "harness-first.sales-agent-token-burn:base:s5", "harness-first.sales-agent-token-burn:base:s6", "harness-first.sales-agent-token-burn:base:s7", "harness-first.sales-agent-token-burn:base:s8"]}, "skill": {"label": "Skill installed for discovery", "attempts": 8, "mean_estimated_api_cost_usd": 0.971852, "mean_agent_wall_s": 973.0875, "mean_model_wait_s": 960.95, "mean_tool_s": 12.125, "stop_reasons": {"turn_cap": 2, "end_turn": 6}, "source_attempt_ids": ["harness-first.sales-agent-token-burn:skill:s1", "harness-first.sales-agent-token-burn:skill:s2", "harness-first.sales-agent-token-burn:skill:s3", "harness-first.sales-agent-token-burn:skill:s4", "harness-first.sales-agent-token-burn:skill:s5", "harness-first.sales-agent-token-burn:skill:s6", "harness-first.sales-agent-token-burn:skill:s7", "harness-first.sales-agent-token-burn:skill:s8"]}}}, {"name": "support-prompt-ship", "arms": {"base": {"label": "Without skill", "attempts": 8, "mean_estimated_api_cost_usd": 0.902095, "mean_agent_wall_s": 707.975, "mean_model_wait_s": 698.45, "mean_tool_s": 9.5375, "stop_reasons": {"end_turn": 8}, "source_attempt_ids": ["harness-first.support-prompt-ship:base:s1", "harness-first.support-prompt-ship:base:s2", "harness-first.support-prompt-ship:base:s3", "harness-first.support-prompt-ship:base:s4", "harness-first.support-prompt-ship:base:s5", "harness-first.support-prompt-ship:base:s6", "harness-first.support-prompt-ship:base:s7", "harness-first.support-prompt-ship:base:s8"]}, "skill": {"label": "Skill installed for discovery", "attempts": 8, "mean_estimated_api_cost_usd": 0.961534, "mean_agent_wall_s": 754.15, "mean_model_wait_s": 746.8, "mean_tool_s": 7.35, "stop_reasons": {"end_turn": 8}, "source_attempt_ids": ["harness-first.support-prompt-ship:skill:s1", "harness-first.support-prompt-ship:skill:s2", "harness-first.support-prompt-ship:skill:s3", "harness-first.support-prompt-ship:skill:s4", "harness-first.support-prompt-ship:skill:s5", "harness-first.support-prompt-ship:skill:s6", "harness-first.support-prompt-ship:skill:s7", "harness-first.support-prompt-ship:skill:s8"]}}}], "legacy_weighting_caveat": "Legacy rubric aggregation matched criterion labels exactly and defaulted unmatched weights to 1. Judges often appended weight annotations, so intended rubric weights were often not applied. Retained values reproduce that historical algorithm; they must not be described as correctly weighted rubric scores. Overall judge scores are independently recorded and unaffected by this aggregation defect.", "recommended_default_metric": "overall"}, "autonomous-research": {"source_run": "heldout", "agent_model": "Claude Sonnet 4.5", "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"], "judge_model": "Claude Haiku 4.5", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "backend": "bedrock", "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "duration_basis": "Agent wall time includes model latency and queueing; capped attempts remain included", "task_origin": "Authored benchmark scenarios and supplied fixtures; not verified customer production cases", "source_url": "/eval-data/autonomous-research/index.json", "tasks": [{"name": "phone-policy-brief", "arms": {"base": {"label": "Without skill", "attempts": 8, "mean_estimated_api_cost_usd": 0.180676, "mean_agent_wall_s": 628.2, "mean_model_wait_s": 525.75, "mean_tool_s": 102.475, "stop_reasons": {"end_turn": 8}, "source_attempt_ids": ["opendraft.phone-policy-brief:base:s1", "opendraft.phone-policy-brief:base:s2", "opendraft.phone-policy-brief:base:s3", "opendraft.phone-policy-brief:base:s4", "opendraft.phone-policy-brief:base:s5", "opendraft.phone-policy-brief:base:s6", "opendraft.phone-policy-brief:base:s7", "opendraft.phone-policy-brief:base:s8"]}, "loaded": {"label": "Skill text in prompt", "attempts": 8, "mean_estimated_api_cost_usd": 3.050358, "mean_agent_wall_s": 2677.5, "mean_model_wait_s": 2329.5, "mean_tool_s": 348.0, "stop_reasons": {"end_turn": 7, "wall_cap": 1}, "source_attempt_ids": ["opendraft.phone-policy-brief:loaded:s1", "opendraft.phone-policy-brief:loaded:s2", "opendraft.phone-policy-brief:loaded:s3", "opendraft.phone-policy-brief:loaded:s4", "opendraft.phone-policy-brief:loaded:s5", "opendraft.phone-policy-brief:loaded:s6", "opendraft.phone-policy-brief:loaded:s7", "opendraft.phone-policy-brief:loaded:s8"]}}}, {"name": "ssb-levy-rea", "arms": {"base": {"label": "Without skill", "attempts": 8, "mean_estimated_api_cost_usd": 0.42568, "mean_agent_wall_s": 436.275, "mean_model_wait_s": 432.6625, "mean_tool_s": 3.5875, "stop_reasons": {"end_turn": 8}, "source_attempt_ids": ["opendraft.ssb-levy-rea:base:s1", "opendraft.ssb-levy-rea:base:s2", "opendraft.ssb-levy-rea:base:s3", "opendraft.ssb-levy-rea:base:s4", "opendraft.ssb-levy-rea:base:s5", "opendraft.ssb-levy-rea:base:s6", "opendraft.ssb-levy-rea:base:s7", "opendraft.ssb-levy-rea:base:s8"]}, "loaded": {"label": "Skill text in prompt", "attempts": 8, "mean_estimated_api_cost_usd": 2.968781, "mean_agent_wall_s": 2695.1875, "mean_model_wait_s": 2341.8625, "mean_tool_s": 353.3375, "stop_reasons": {"end_turn": 7, "cost_cap": 1}, "source_attempt_ids": ["opendraft.ssb-levy-rea:loaded:s1", "opendraft.ssb-levy-rea:loaded:s2", "opendraft.ssb-levy-rea:loaded:s3", "opendraft.ssb-levy-rea:loaded:s4", "opendraft.ssb-levy-rea:loaded:s5", "opendraft.ssb-levy-rea:loaded:s6", "opendraft.ssb-levy-rea:loaded:s7", "opendraft.ssb-levy-rea:loaded:s8"]}}}], "legacy_weighting_caveat": "Legacy rubric aggregation matched criterion labels exactly and defaulted unmatched weights to 1. Judges often appended weight annotations, so intended rubric weights were often not applied. Retained values reproduce that historical algorithm; they must not be described as correctly weighted rubric scores. Overall judge scores are independently recorded and unaffected by this aggregation defect.", "recommended_default_metric": "overall"}, "anthropics-pptx": {"source_run": "arena-pptx-round01", "agent_model": "anthropic/claude-sonnet-5", "model_evidence": "Recorded runner model identifier; independent provider attestation unavailable", "judge_model": null, "judge_backend": "codex-cli", "judge_profile": "default", "judge_model_note": "Exact judge model was not retained in the result metadata", "backend": "subscription", "cost_basis": "Subscription run; zero marginal charge recorded, not free execution. API-equivalent estimates must be kept separate.", "task_origin": "Three authored consulting-deck scenarios with supplied fixtures", "retained_attempts": 12, "paired_comparisons": 6, "judge_artifacts": 12, "excluded_infrastructure_attempts": 6, "missing_deck_attempts": 1, "source_url": "/arena-data/powerpoint-summary.json", "legacy_weighting_caveat": "Legacy rubric aggregation matched criterion labels exactly and defaulted unmatched weights to 1. Judges often appended weight annotations, so intended rubric weights were often not applied. Retained values reproduce that historical algorithm; they must not be described as correctly weighted rubric scores. Overall judge scores are independently recorded and unaffected by this aggregation defect.", "recommended_default_metric": "overall"}}}