{"schema_version": 2, "tasks": [{"id": "workplan--ledger-migration-resume", "title": "Ledger migration resume", "category": "productivity", "prompt": "rune's off until the 22nd and ledgerctl v3 has to be done before the q3 close. he left a plan file in the repo and some handover notes, can you pick it up from wherever he got to? heads up, he was in a rush by the end so i wouldn't trust everything he ticked off. ./run_tests.sh is what CI runs. when you're done tell me what's still open so I know what to chase fin for.", "skill": "workplan", "skill_label": "Plan multi-step agent work", "base_score": 77.666667, "skill_score": 84.25, "pairs": 6, "task_index": 0, "evaluation_url": "/evaluation/workplan/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/workplan/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "workplan", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 6, "pair_indices": [0, 1, 2, 3, 4, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 78.452167, "skill": 90.357, "delta": 11.904833, "samples": [{"sample": 1, "base": 68.571, "skill": 92.857}, {"sample": 2, "base": 83.571, "skill": 92.857}, {"sample": 3, "base": 71.429, "skill": 88.571}, {"sample": 4, "base": 80.714, "skill": 90.714}, {"sample": 5, "base": 83.571, "skill": 85.0}, {"sample": 6, "base": 82.857, "skill": 92.143}]}, "overall": {"base": 77.666667, "skill": 84.25, "delta": 6.583333, "samples": [{"sample": 1, "base": 71.5, "skill": 85.0}, {"sample": 2, "base": 78.0, "skill": 84.5}, {"sample": 3, "base": 73.0, "skill": 83.5}, {"sample": 4, "base": 81.5, "skill": 87.0}, {"sample": 5, "base": 80.0, "skill": 78.0}, {"sample": 6, "base": 82.0, "skill": 87.5}]}}, "base_cost_usd": 0.428317, "skill_cost_usd": 0.508817, "base_turns": 40.833333, "skill_turns": 40.833333, "base_check_pass_rate": 0, "skill_check_pass_rate": 0, "baseline_attempt_ids": ["ledger-migration-resume-base-s1", "ledger-migration-resume-base-s2", "ledger-migration-resume-base-s3", "ledger-migration-resume-base-s4", "ledger-migration-resume-base-s5", "ledger-migration-resume-base-s6"], "skill_attempt_ids": ["ledger-migration-resume-loaded-s1", "ledger-migration-resume-loaded-s2", "ledger-migration-resume-loaded-s3", "ledger-migration-resume-loaded-s4", "ledger-migration-resume-loaded-s5", "ledger-migration-resume-loaded-s6"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 477.433333, "skill_agent_wall_s": 513.633333, "source_run": "t3-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/workplan/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "workplan--payroll-export-batch", "title": "Payroll export batch", "category": "productivity", "prompt": "can you get paystream ready for the september run? backlog is in BACKLOG.md. two things from standup that never made it into the file: a --dry-run that prints the totals without writing anything, and once the file is written it should get pushed to the finance sftp (host and key are in the ops vault, I don't have them on me). ./run_tests.sh is CI. I'm in workshops all afternoon so put it somewhere I can pick it up from, and tell me what you couldn't get done.", "skill": "workplan", "skill_label": "Plan multi-step agent work", "base_score": 50.0, "skill_score": 72.833333, "pairs": 6, "task_index": 1, "evaluation_url": "/evaluation/workplan/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/workplan/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "workplan", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 6, "pair_indices": [0, 1, 2, 3, 4, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 71.899, "skill": 86.984, "delta": 15.085, "samples": [{"sample": 1, "base": 68.434, "skill": 94.444}, {"sample": 2, "base": 96.19, "skill": 74.643}, {"sample": 3, "base": 82.5, "skill": 87.833}, {"sample": 4, "base": 53.393, "skill": 99.286}, {"sample": 5, "base": 46.786, "skill": 80.893}, {"sample": 6, "base": 84.091, "skill": 84.805}]}, "overall": {"base": 50.0, "skill": 72.833333, "delta": 22.833333, "samples": [{"sample": 1, "base": 35.0, "skill": 65.0}, {"sample": 2, "base": 90.0, "skill": 62.0}, {"sample": 3, "base": 41.5, "skill": 75.0}, {"sample": 4, "base": 26.5, "skill": 93.5}, {"sample": 5, "base": 33.0, "skill": 67.0}, {"sample": 6, "base": 74.0, "skill": 74.5}]}}, "base_cost_usd": 0.48835, "skill_cost_usd": 0.5702, "base_turns": 42, "skill_turns": 42, "base_check_pass_rate": 0, "skill_check_pass_rate": 0, "baseline_attempt_ids": ["payroll-export-batch-base-s1", "payroll-export-batch-base-s2", "payroll-export-batch-base-s3", "payroll-export-batch-base-s4", "payroll-export-batch-base-s5", "payroll-export-batch-base-s6"], "skill_attempt_ids": ["payroll-export-batch-loaded-s1", "payroll-export-batch-loaded-s2", "payroll-export-batch-loaded-s3", "payroll-export-batch-loaded-s4", "payroll-export-batch-loaded-s5", "payroll-export-batch-loaded-s6"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 433.683333, "skill_agent_wall_s": 457.766667, "source_run": "t3-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/workplan/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "workplan--status-page-demo-cut", "title": "Status page demo cut", "category": "productivity", "prompt": "we're showing the status page to northwind at 3 and it's half one. wishlist is in WISHLIST.md, everything people have asked for over the last month. demo.sh is what priya will actually run on the laptop, don't touch it. get what you can in, and leave me a note of what's in and what's out so I'm not guessing in the room.", "skill": "workplan", "skill_label": "Plan multi-step agent work", "base_score": 79.416667, "skill_score": 92.083333, "pairs": 6, "task_index": 2, "evaluation_url": "/evaluation/workplan/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/workplan/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "workplan", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 6, "pair_indices": [0, 1, 2, 3, 4, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 85.64, "skill": 98.824333, "delta": 13.184333, "samples": [{"sample": 1, "base": 70.0, "skill": 97.857}, {"sample": 2, "base": 91.429, "skill": 99.286}, {"sample": 3, "base": 89.286, "skill": 98.571}, {"sample": 4, "base": 85.714, "skill": 99.286}, {"sample": 5, "base": 92.143, "skill": 98.571}, {"sample": 6, "base": 85.268, "skill": 99.375}]}, "overall": {"base": 79.416667, "skill": 92.083333, "delta": 12.666666, "samples": [{"sample": 1, "base": 65.0, "skill": 93.5}, {"sample": 2, "base": 85.0, "skill": 92.5}, {"sample": 3, "base": 80.0, "skill": 90.0}, {"sample": 4, "base": 82.0, "skill": 92.5}, {"sample": 5, "base": 86.5, "skill": 92.0}, {"sample": 6, "base": 78.0, "skill": 92.0}]}}, "base_cost_usd": 0.260583, "skill_cost_usd": 0.250317, "base_turns": 24.833333, "skill_turns": 20, "base_check_pass_rate": 0, "skill_check_pass_rate": 0.833333, "baseline_attempt_ids": ["status-page-demo-cut-base-s1", "status-page-demo-cut-base-s2", "status-page-demo-cut-base-s3", "status-page-demo-cut-base-s4", "status-page-demo-cut-base-s5", "status-page-demo-cut-base-s6"], "skill_attempt_ids": ["status-page-demo-cut-loaded-s1", "status-page-demo-cut-loaded-s2", "status-page-demo-cut-loaded-s3", "status-page-demo-cut-loaded-s4", "status-page-demo-cut-loaded-s5", "status-page-demo-cut-loaded-s6"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 237.333333, "skill_agent_wall_s": 201.733333, "source_run": "t3-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/workplan/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "autonomous-research--phone-policy-brief", "title": "Phone policy brief", "category": "research", "prompt": "Morning. I need the evidence annex for the cabinet paper on going phone-free across our secondaries. It goes out to members Tuesday and I'm in consultation sessions all day tomorrow.\n\nAbout 2,000 words, 2,400 is the absolute ceiling. Harvard, author-date, reference list at the end with DOIs on everything (legal bounced the last annex because two references couldn't be traced). Open it with a short Bottom line box, 150 words max, because for a lot of members that is the only part they read.\n\nIt has to cover attainment, wellbeing and mental health, how strong the underlying studies actually are, and the implementation question the heads keep raising with me: enforcement, lockable pouches, and parents who want to be able to reach their child during the day.\n\nWhat I have is in the folder. evidence_notes.md is where Jaz and I got to, and reading_list.bib is our reading list. The list is what we happen to have collected, not the field, so bring in whatever else you need. An annex on something this contested should be sitting on 17 or 18 sources, not 10.\n\noutput/phone_policy_brief.md please. Cabinet has effectively landed on doing this in January, so the annex is there to set out the evidence base under the decision.", "skill": "autonomous-research", "skill_label": "Autonomous Research", "base_score": 55.6875, "skill_score": 45.375, "pairs": 8, "task_index": 0, "evaluation_url": "/evaluation/autonomous-research/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/autonomous-research/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "autonomous-research", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 8, "pair_indices": [0, 1, 2, 3, 4, 5, 6, 7], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 52.5, "skill": 54.166625, "delta": 1.666625, "samples": [{"sample": 1, "base": 41.667, "skill": 56.667}, {"sample": 2, "base": 65.0, "skill": 35.833}, {"sample": 3, "base": 40.833, "skill": 53.333}, {"sample": 4, "base": 35.0, "skill": 84.167}, {"sample": 5, "base": 35.833, "skill": 65.0}, {"sample": 6, "base": 66.667, "skill": 43.333}, {"sample": 7, "base": 68.333, "skill": 52.5}, {"sample": 8, "base": 66.667, "skill": 42.5}]}, "overall": {"base": 55.6875, "skill": 45.375, "delta": -10.3125, "samples": [{"sample": 1, "base": 37.0, "skill": 28.0}, {"sample": 2, "base": 68.5, "skill": 41.0}, {"sample": 3, "base": 50.0, "skill": 50.0}, {"sample": 4, "base": 40.0, "skill": 60.0}, {"sample": 5, "base": 43.0, "skill": 59.5}, {"sample": 6, "base": 72.0, "skill": 31.5}, {"sample": 7, "base": 66.5, "skill": 55.0}, {"sample": 8, "base": 68.5, "skill": 38.0}]}}, "base_cost_usd": 0.180688, "skill_cost_usd": 3.05035, "base_turns": 8.875, "skill_turns": 88, "base_check_pass_rate": 0, "skill_check_pass_rate": 0, "baseline_attempt_ids": ["phone-policy-brief-base-s1", "phone-policy-brief-base-s2", "phone-policy-brief-base-s3", "phone-policy-brief-base-s4", "phone-policy-brief-base-s5", "phone-policy-brief-base-s6", "phone-policy-brief-base-s7", "phone-policy-brief-base-s8"], "skill_attempt_ids": ["phone-policy-brief-loaded-s1", "phone-policy-brief-loaded-s2", "phone-policy-brief-loaded-s3", "phone-policy-brief-loaded-s4", "phone-policy-brief-loaded-s5", "phone-policy-brief-loaded-s6", "phone-policy-brief-loaded-s7", "phone-policy-brief-loaded-s8"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 628.2, "skill_agent_wall_s": 2677.5, "source_run": "heldout", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/autonomous-research/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "autonomous-research--ssb-levy-rea", "title": "Ssb levy rea", "category": "research", "prompt": "Hi, picking up from Marcus. We're responding to the Treasury consultation on extending the soft drinks levy to milk-based drinks and dropping the lower sugar threshold, and I need the rapid evidence assessment that sits behind the response. It closes on the 29th and I'm at the select committee most of this week.\n\nAbout 2,200 words, 2,600 is the hard ceiling, plus the reference list. Harvard, author-date, DOI on every reference. Key findings box at the top, 120 words maximum, because that is genuinely all the Treasury official will read. And a study characteristics table in the body, one row per study, Study / Setting / Design / Outcome / Effect, because the last one we sent came back with a complaint that you couldn't see the designs at a glance.\n\nMarcus left a steer note and a starter .bib in the folder. The bib is twelve entries, which is what he happened to have in the briefing folder rather than the evidence base; for a consultation response I'd want to be sitting on sixteen or more, and it needs to cover the milk-drinks question and the purchases-versus-intake question, which we have nothing on.\n\noutput/ssb_levy_rea.md, please. Our position is that the levy has worked and should be extended, and the board signed that off in July, so the REA is there to carry the evidence for it.", "skill": "autonomous-research", "skill_label": "Autonomous Research", "base_score": 39.25, "skill_score": 48.0625, "pairs": 8, "task_index": 1, "evaluation_url": "/evaluation/autonomous-research/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/autonomous-research/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "autonomous-research", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 8, "pair_indices": [0, 1, 2, 3, 4, 5, 6, 7], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 43.6905, "skill": 55.133875, "delta": 11.443375, "samples": [{"sample": 1, "base": 44.167, "skill": 35.833}, {"sample": 2, "base": 33.333, "skill": 91.667}, {"sample": 3, "base": 52.024, "skill": 68.571}, {"sample": 4, "base": 35.833, "skill": 31.667}, {"sample": 5, "base": 49.167, "skill": 48.333}, {"sample": 6, "base": 52.5, "skill": 90.0}, {"sample": 7, "base": 41.667, "skill": 29.167}, {"sample": 8, "base": 40.833, "skill": 45.833}]}, "overall": {"base": 39.25, "skill": 48.0625, "delta": 8.8125, "samples": [{"sample": 1, "base": 45.0, "skill": 26.0}, {"sample": 2, "base": 26.5, "skill": 72.0}, {"sample": 3, "base": 52.0, "skill": 53.0}, {"sample": 4, "base": 26.5, "skill": 30.0}, {"sample": 5, "base": 41.0, "skill": 40.0}, {"sample": 6, "base": 43.0, "skill": 88.5}, {"sample": 7, "base": 47.0, "skill": 28.0}, {"sample": 8, "base": 33.0, "skill": 47.0}]}}, "base_cost_usd": 0.425687, "skill_cost_usd": 2.968787, "base_turns": 15.5, "skill_turns": 87.75, "base_check_pass_rate": 0, "skill_check_pass_rate": 0, "baseline_attempt_ids": ["ssb-levy-rea-base-s1", "ssb-levy-rea-base-s2", "ssb-levy-rea-base-s3", "ssb-levy-rea-base-s4", "ssb-levy-rea-base-s5", "ssb-levy-rea-base-s6", "ssb-levy-rea-base-s7", "ssb-levy-rea-base-s8"], "skill_attempt_ids": ["ssb-levy-rea-loaded-s1", "ssb-levy-rea-loaded-s2", "ssb-levy-rea-loaded-s3", "ssb-levy-rea-loaded-s4", "ssb-levy-rea-loaded-s5", "ssb-levy-rea-loaded-s6", "ssb-levy-rea-loaded-s7", "ssb-levy-rea-loaded-s8"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 436.275, "skill_agent_wall_s": 2695.1875, "source_run": "heldout", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/autonomous-research/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "harness-first--finance-agent-revenue", "title": "Finance agent revenue", "category": "engineering", "prompt": "finbot told the board deck team Q2 revenue was 4.1M, finance says 3.6, and now it's in the pre-read. is the model just hallucinating? do we need a smarter model?? Daniel wants an answer tomorrow morning. repo + the warehouse + the slack stuff is all in here. figure it out, put what you find in output/", "skill": "harness-first", "skill_label": "Audit the agent harness", "base_score": 64.4375, "skill_score": 86.9375, "pairs": 8, "task_index": 0, "evaluation_url": "/evaluation/harness-first/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/harness-first/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "skill", "label": "Installed skill", "skill": "harness-first", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 8, "pair_indices": [0, 1, 2, 3, 4, 5, 6, 7], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 60.357125, "skill": 89.732125, "delta": 29.375, "samples": [{"sample": 1, "base": 54.286, "skill": 91.429}, {"sample": 2, "base": 67.143, "skill": 92.143}, {"sample": 3, "base": 57.857, "skill": 91.429}, {"sample": 4, "base": 60.714, "skill": 87.857}, {"sample": 5, "base": 66.429, "skill": 88.571}, {"sample": 6, "base": 52.857, "skill": 92.857}, {"sample": 7, "base": 55.714, "skill": 88.571}, {"sample": 8, "base": 67.857, "skill": 85.0}]}, "overall": {"base": 64.4375, "skill": 86.9375, "delta": 22.5, "samples": [{"sample": 1, "base": 60.0, "skill": 87.5}, {"sample": 2, "base": 71.5, "skill": 92.0}, {"sample": 3, "base": 60.0, "skill": 85.5}, {"sample": 4, "base": 63.0, "skill": 89.0}, {"sample": 5, "base": 67.0, "skill": 88.5}, {"sample": 6, "base": 60.0, "skill": 87.0}, {"sample": 7, "base": 62.0, "skill": 88.0}, {"sample": 8, "base": 72.0, "skill": 78.0}]}}, "base_cost_usd": 0.864013, "skill_cost_usd": 1.023325, "base_turns": 32.875, "skill_turns": 38.625, "base_check_pass_rate": 0, "skill_check_pass_rate": 1, "baseline_attempt_ids": ["finance-agent-revenue-base-s1", "finance-agent-revenue-base-s2", "finance-agent-revenue-base-s3", "finance-agent-revenue-base-s4", "finance-agent-revenue-base-s5", "finance-agent-revenue-base-s6", "finance-agent-revenue-base-s7", "finance-agent-revenue-base-s8"], "skill_attempt_ids": ["finance-agent-revenue-skill-s1", "finance-agent-revenue-skill-s2", "finance-agent-revenue-skill-s3", "finance-agent-revenue-skill-s4", "finance-agent-revenue-skill-s5", "finance-agent-revenue-skill-s6", "finance-agent-revenue-skill-s7", "finance-agent-revenue-skill-s8"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 797.6125, "skill_agent_wall_s": 896.3625, "source_run": "hf-v1", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "skill", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/harness-first/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "harness-first--sales-agent-token-burn", "title": "Sales agent token burn", "category": "engineering", "prompt": "hey, finance just pinged me about the anthropic bill for the outreach agent. we're halfway through september and it's already ~4x what all of august cost, and nobody changed the volume. i honestly think sonnet is overkill for writing cold emails, priya put a price sheet in docs/. should we just switch to a cheaper model? pick one for us and tell me roughly what we'd save, i'd like to flip it before month end. the agent code is in agent/, trace export + crm client log for sept 1-15 are in logs/. put anything you make in output/ (or just fix stuff in the code if you need to, it's our repo)", "skill": "harness-first", "skill_label": "Audit the agent harness", "base_score": 26.0625, "skill_score": 90.375, "pairs": 8, "task_index": 1, "evaluation_url": "/evaluation/harness-first/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/harness-first/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "skill", "label": "Installed skill", "skill": "harness-first", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 8, "pair_indices": [0, 1, 2, 3, 4, 5, 6, 7], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 22.10575, "skill": 90.811125, "delta": 68.705375, "samples": [{"sample": 1, "base": 22.5, "skill": 95.0}, {"sample": 2, "base": 26.667, "skill": 77.381}, {"sample": 3, "base": 39.167, "skill": 79.167}, {"sample": 4, "base": 5.833, "skill": 99.167}, {"sample": 5, "base": 8.929, "skill": 97.024}, {"sample": 6, "base": 34.167, "skill": 92.5}, {"sample": 7, "base": 21.25, "skill": 88.75}, {"sample": 8, "base": 18.333, "skill": 97.5}]}, "overall": {"base": 26.0625, "skill": 90.375, "delta": 64.3125, "samples": [{"sample": 1, "base": 28.0, "skill": 89.5}, {"sample": 2, "base": 26.5, "skill": 87.0}, {"sample": 3, "base": 40.0, "skill": 85.0}, {"sample": 4, "base": 15.0, "skill": 93.5}, {"sample": 5, "base": 20.0, "skill": 92.0}, {"sample": 6, "base": 34.0, "skill": 93.0}, {"sample": 7, "base": 25.0, "skill": 90.0}, {"sample": 8, "base": 20.0, "skill": 93.0}]}}, "base_cost_usd": 0.482875, "skill_cost_usd": 0.971863, "base_turns": 26.5, "skill_turns": 35.125, "base_check_pass_rate": 0, "skill_check_pass_rate": 0, "baseline_attempt_ids": ["sales-agent-token-burn-base-s1", "sales-agent-token-burn-base-s2", "sales-agent-token-burn-base-s3", "sales-agent-token-burn-base-s4", "sales-agent-token-burn-base-s5", "sales-agent-token-burn-base-s6", "sales-agent-token-burn-base-s7", "sales-agent-token-burn-base-s8"], "skill_attempt_ids": ["sales-agent-token-burn-skill-s1", "sales-agent-token-burn-skill-s2", "sales-agent-token-burn-skill-s3", "sales-agent-token-burn-skill-s4", "sales-agent-token-burn-skill-s5", "sales-agent-token-burn-skill-s6", "sales-agent-token-burn-skill-s7", "sales-agent-token-burn-skill-s8"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 638.5625, "skill_agent_wall_s": 973.0875, "source_run": "hf-v1", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "skill", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/harness-first/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "harness-first--support-prompt-ship", "title": "Support prompt ship", "category": "engineering", "prompt": "hey! we rewrote the support bot's system prompt (new_prompt.md vs old_prompt.md), want to ship it friday. it feels WAY warmer in my testing and I'm pretty confident CSAT goes up. I replayed last month's tickets through both, outputs_old.jsonl / outputs_new.jsonl, my notes are in pm_notes.md. can you sanity check it and give me a go/no-go? we're going to keep tweaking this prompt every couple weeks so anything that makes the next check less painful is welcome. put notes etc in output/", "skill": "harness-first", "skill_label": "Audit the agent harness", "base_score": 68.0, "skill_score": 62.75, "pairs": 8, "task_index": 2, "evaluation_url": "/evaluation/harness-first/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/harness-first/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "skill", "label": "Installed skill", "skill": "harness-first", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 8, "pair_indices": [0, 1, 2, 3, 4, 5, 6, 7], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 79.569375, "skill": 76.879125, "delta": -2.69025, "samples": [{"sample": 1, "base": 83.333, "skill": 86.667}, {"sample": 2, "base": 64.167, "skill": 70.0}, {"sample": 3, "base": 86.667, "skill": 77.5}, {"sample": 4, "base": 80.0, "skill": 85.833}, {"sample": 5, "base": 68.81, "skill": 97.143}, {"sample": 6, "base": 80.0, "skill": 85.833}, {"sample": 7, "base": 80.245, "skill": 81.224}, {"sample": 8, "base": 93.333, "skill": 30.833}]}, "overall": {"base": 68.0, "skill": 62.75, "delta": -5.25, "samples": [{"sample": 1, "base": 68.0, "skill": 67.0}, {"sample": 2, "base": 60.0, "skill": 60.0}, {"sample": 3, "base": 81.0, "skill": 61.5}, {"sample": 4, "base": 61.5, "skill": 72.0}, {"sample": 5, "base": 50.0, "skill": 85.0}, {"sample": 6, "base": 70.0, "skill": 68.0}, {"sample": 7, "base": 61.5, "skill": 62.0}, {"sample": 8, "base": 92.0, "skill": 26.5}]}}, "base_cost_usd": 0.9021, "skill_cost_usd": 0.96155, "base_turns": 28.125, "skill_turns": 28.25, "base_check_pass_rate": 0.25, "skill_check_pass_rate": 0.375, "baseline_attempt_ids": ["support-prompt-ship-base-s1", "support-prompt-ship-base-s2", "support-prompt-ship-base-s3", "support-prompt-ship-base-s4", "support-prompt-ship-base-s5", "support-prompt-ship-base-s6", "support-prompt-ship-base-s7", "support-prompt-ship-base-s8"], "skill_attempt_ids": ["support-prompt-ship-skill-s1", "support-prompt-ship-skill-s2", "support-prompt-ship-skill-s3", "support-prompt-ship-skill-s4", "support-prompt-ship-skill-s5", "support-prompt-ship-skill-s6", "support-prompt-ship-skill-s7", "support-prompt-ship-skill-s8"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 707.975, "skill_agent_wall_s": 754.15, "source_run": "hf-v1", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "skill", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/harness-first/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "top-down-comms--buried-blocker-status", "title": "Buried blocker status", "category": "communication", "prompt": "can you turn the leads' notes into this week's steering status for the Ledgerline program? goes to Dana (CFO) and the steering group, she chairs at 9 tomorrow. last week's version is in the folder, roughly the same shape is fine, plus the other bits Lea dumped in there. mostly green this week I think, nice change from August. put it in output/", "skill": "top-down-comms", "skill_label": "Write clear executive updates", "base_score": 75.333333, "skill_score": 87.166667, "pairs": 3, "task_index": 0, "evaluation_url": "/evaluation/top-down-comms/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/top-down-comms/index.json", "configurations_tested": 3, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "top-down-comms", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 3, "pair_indices": [0, 2, 4], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 79.166667, "skill": 91.666667, "delta": 12.5, "samples": [{"sample": 1, "base": 80.0, "skill": 92.5}, {"sample": 2, "base": 77.5, "skill": 89.167}, {"sample": 3, "base": 80.0, "skill": 93.333}]}, "overall": {"base": 75.333333, "skill": 87.166667, "delta": 11.833334, "samples": [{"sample": 1, "base": 75.0, "skill": 87.0}, {"sample": 2, "base": 76.0, "skill": 86.5}, {"sample": 3, "base": 75.0, "skill": 88.0}]}}, "base_cost_usd": 0.069333, "skill_cost_usd": 0.0754, "base_turns": 5.333333, "skill_turns": 6.333333, "base_check_pass_rate": 0, "skill_check_pass_rate": 0, "baseline_attempt_ids": ["buried-blocker-status-base-s1", "buried-blocker-status-base-s2", "buried-blocker-status-base-s3"], "skill_attempt_ids": ["buried-blocker-status-loaded-s1", "buried-blocker-status-loaded-s2", "buried-blocker-status-loaded-s3"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 78.933333, "skill_agent_wall_s": 94.666667, "source_run": "td-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}, {"id": "skill", "label": "Installed skill", "skill": "top-down-comms", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 3, "pair_indices": [1, 3, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 85.833333, "skill": 91.389, "delta": 5.555667, "samples": [{"sample": 1, "base": 77.5, "skill": 90.0}, {"sample": 2, "base": 88.333, "skill": 92.5}, {"sample": 3, "base": 91.667, "skill": 91.667}]}, "overall": {"base": 79.833333, "skill": 84.5, "delta": 4.666667, "samples": [{"sample": 1, "base": 73.5, "skill": 84.0}, {"sample": 2, "base": 80.0, "skill": 83.5}, {"sample": 3, "base": 86.0, "skill": 86.0}]}}, "base_cost_usd": 0.069333, "skill_cost_usd": 0.070033, "base_turns": 5.333333, "skill_turns": 4.666667, "base_check_pass_rate": 0, "skill_check_pass_rate": 0.333333, "baseline_attempt_ids": ["buried-blocker-status-base-s1", "buried-blocker-status-base-s2", "buried-blocker-status-base-s3"], "skill_attempt_ids": ["buried-blocker-status-skill-s1", "buried-blocker-status-skill-s2", "buried-blocker-status-skill-s3"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 78.933333, "skill_agent_wall_s": 80.933333, "source_run": "td-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/top-down-comms/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "top-down-comms--client-update-from-thread", "title": "Client update from thread", "category": "communication", "prompt": "hey, can you draft the update email to Rosa at Halden about the migration dry run? slack export from our internal channel is in the folder plus last week's call notes. she reads these on her phone. put the draft in output/ (subject line + body), I'll send it myself", "skill": "top-down-comms", "skill_label": "Write clear executive updates", "base_score": 76.166667, "skill_score": 80.166667, "pairs": 3, "task_index": 1, "evaluation_url": "/evaluation/top-down-comms/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/top-down-comms/index.json", "configurations_tested": 3, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "top-down-comms", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 3, "pair_indices": [0, 2, 4], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 85.0, "skill": 90.278, "delta": 5.278, "samples": [{"sample": 1, "base": 85.0, "skill": 81.667}, {"sample": 2, "base": 85.833, "skill": 94.167}, {"sample": 3, "base": 84.167, "skill": 95.0}]}, "overall": {"base": 76.166667, "skill": 80.166667, "delta": 4.0, "samples": [{"sample": 1, "base": 78.0, "skill": 72.0}, {"sample": 2, "base": 78.5, "skill": 85.0}, {"sample": 3, "base": 72.0, "skill": 83.5}]}}, "base_cost_usd": 0.0519, "skill_cost_usd": 0.055133, "base_turns": 5, "skill_turns": 5, "base_check_pass_rate": 0.333333, "skill_check_pass_rate": 1, "baseline_attempt_ids": ["client-update-from-thread-base-s1", "client-update-from-thread-base-s2", "client-update-from-thread-base-s3"], "skill_attempt_ids": ["client-update-from-thread-loaded-s1", "client-update-from-thread-loaded-s2", "client-update-from-thread-loaded-s3"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 79.3, "skill_agent_wall_s": 69.8, "source_run": "td-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}, {"id": "skill", "label": "Installed skill", "skill": "top-down-comms", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 3, "pair_indices": [1, 3, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 85.277667, "skill": 85.0, "delta": -0.277667, "samples": [{"sample": 1, "base": 85.833, "skill": 90.833}, {"sample": 2, "base": 80.833, "skill": 92.5}, {"sample": 3, "base": 89.167, "skill": 71.667}]}, "overall": {"base": 78.666667, "skill": 80.166667, "delta": 1.5, "samples": [{"sample": 1, "base": 78.0, "skill": 83.5}, {"sample": 2, "base": 76.0, "skill": 87.0}, {"sample": 3, "base": 82.0, "skill": 70.0}]}}, "base_cost_usd": 0.0519, "skill_cost_usd": 0.0616, "base_turns": 5, "skill_turns": 5.666667, "base_check_pass_rate": 0.333333, "skill_check_pass_rate": 0.666667, "baseline_attempt_ids": ["client-update-from-thread-base-s1", "client-update-from-thread-base-s2", "client-update-from-thread-base-s3"], "skill_attempt_ids": ["client-update-from-thread-skill-s1", "client-update-from-thread-skill-s2", "client-update-from-thread-skill-s3"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 79.3, "skill_agent_wall_s": 86.133333, "source_run": "td-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/top-down-comms/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "top-down-comms--renewal-memo-missing-price", "title": "Renewal memo missing price", "category": "communication", "prompt": "Priya wants a one-pager by Thursday on the Tallow CRM renewal: do we renew or not, and what's it going to cost us next year. everything I've got is in the folder, the contract bit, the emails with their account guy, the ops channel scroll and the seat usage export. she makes the call, she hates long docs. output/ is fine", "skill": "top-down-comms", "skill_label": "Write clear executive updates", "base_score": 71.333333, "skill_score": 86.666667, "pairs": 3, "task_index": 2, "evaluation_url": "/evaluation/top-down-comms/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/top-down-comms/index.json", "configurations_tested": 3, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "top-down-comms", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 3, "pair_indices": [0, 2, 4], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 75.555667, "skill": 92.5, "delta": 16.944333, "samples": [{"sample": 1, "base": 69.167, "skill": 92.5}, {"sample": 2, "base": 81.667, "skill": 95.0}, {"sample": 3, "base": 75.833, "skill": 90.0}]}, "overall": {"base": 71.333333, "skill": 86.666667, "delta": 15.333334, "samples": [{"sample": 1, "base": 65.0, "skill": 85.0}, {"sample": 2, "base": 75.0, "skill": 87.0}, {"sample": 3, "base": 74.0, "skill": 88.0}]}}, "base_cost_usd": 0.0642, "skill_cost_usd": 0.052067, "base_turns": 5, "skill_turns": 4.333333, "base_check_pass_rate": 0.666667, "skill_check_pass_rate": 0.333333, "baseline_attempt_ids": ["renewal-memo-missing-price-base-s1", "renewal-memo-missing-price-base-s2", "renewal-memo-missing-price-base-s3"], "skill_attempt_ids": ["renewal-memo-missing-price-loaded-s1", "renewal-memo-missing-price-loaded-s2", "renewal-memo-missing-price-loaded-s3"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 119.666667, "skill_agent_wall_s": 88.5, "source_run": "td-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}, {"id": "skill", "label": "Installed skill", "skill": "top-down-comms", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 3, "pair_indices": [1, 3, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 73.333333, "skill": 80.555667, "delta": 7.222334, "samples": [{"sample": 1, "base": 66.667, "skill": 80.0}, {"sample": 2, "base": 75.833, "skill": 89.167}, {"sample": 3, "base": 77.5, "skill": 72.5}]}, "overall": {"base": 67.5, "skill": 75.666667, "delta": 8.166667, "samples": [{"sample": 1, "base": 63.0, "skill": 80.0}, {"sample": 2, "base": 65.0, "skill": 85.0}, {"sample": 3, "base": 74.5, "skill": 62.0}]}}, "base_cost_usd": 0.0642, "skill_cost_usd": 0.095367, "base_turns": 5, "skill_turns": 5.333333, "base_check_pass_rate": 0.666667, "skill_check_pass_rate": 1, "baseline_attempt_ids": ["renewal-memo-missing-price-base-s1", "renewal-memo-missing-price-base-s2", "renewal-memo-missing-price-base-s3"], "skill_attempt_ids": ["renewal-memo-missing-price-skill-s1", "renewal-memo-missing-price-skill-s2", "renewal-memo-missing-price-skill-s3"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 119.666667, "skill_agent_wall_s": 110.2, "source_run": "td-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/top-down-comms/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "strip-image-ai-metadata--post-batch-lossless", "title": "Post batch lossless", "category": "media", "prompt": "linkedin sticks that content-credentials thing on about half my posts now and it's killing the reach. thursday's batch is in queue/ - four images, a couple came out of an image generator, the rest are mine. can you pull out whatever it is that trips the label? don't re-export or recompress them, and they have to open looking exactly like what's in queue/ right now, i'm not re-shooting any of this. leave queue/ alone, cleaned ones into output/. also tell me which ones were actually carrying it, i genuinely can't tell by looking", "skill": "strip-image-ai-metadata", "skill_label": "Remove AI image metadata", "base_score": 50.0, "skill_score": 74.666667, "pairs": 6, "task_index": 0, "evaluation_url": "/evaluation/strip-image-ai-metadata/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/strip-image-ai-metadata/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "strip-image-ai-metadata", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 6, "pair_indices": [0, 1, 2, 3, 4, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 66.6665, "skill": 82.381, "delta": 15.7145, "samples": [{"sample": 1, "base": 70.714, "skill": 82.857}, {"sample": 2, "base": 70.714, "skill": 77.143}, {"sample": 3, "base": 80.714, "skill": 72.857}, {"sample": 4, "base": 53.571, "skill": 88.571}, {"sample": 5, "base": 85.0, "skill": 81.429}, {"sample": 6, "base": 39.286, "skill": 91.429}]}, "overall": {"base": 50.0, "skill": 74.666667, "delta": 24.666667, "samples": [{"sample": 1, "base": 35.0, "skill": 67.0}, {"sample": 2, "base": 60.0, "skill": 70.0}, {"sample": 3, "base": 69.5, "skill": 57.0}, {"sample": 4, "base": 37.0, "skill": 85.0}, {"sample": 5, "base": 78.5, "skill": 77.0}, {"sample": 6, "base": 20.0, "skill": 92.0}]}}, "base_cost_usd": 0.2387, "skill_cost_usd": 0.071983, "base_turns": 24.666667, "skill_turns": 12.333333, "base_check_pass_rate": 0.333333, "skill_check_pass_rate": 1, "baseline_attempt_ids": ["post-batch-lossless-base-s1", "post-batch-lossless-base-s2", "post-batch-lossless-base-s3", "post-batch-lossless-base-s4", "post-batch-lossless-base-s5", "post-batch-lossless-base-s6"], "skill_attempt_ids": ["post-batch-lossless-loaded-s1", "post-batch-lossless-loaded-s2", "post-batch-lossless-loaded-s3", "post-batch-lossless-loaded-s4", "post-batch-lossless-loaded-s5", "post-batch-lossless-loaded-s6"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 200.883333, "skill_agent_wall_s": 86.766667, "source_run": "t3-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/strip-image-ai-metadata/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "strip-image-ai-metadata--print-proofs-colour-critical", "title": "Print proofs colour critical", "category": "media", "prompt": "proofs for the nordvind sleeve are in proofs/ - this goes over to kasten druck on monday. their preflight bounces anything that reads as ai-generated in the file info, and it also bounces files that come in with no rights or credit info, so both of those have to be true when i hand it over. colour has to match the proof exactly so nothing gets re-saved or re-compressed. overwrite them in proofs/ directly, that's the folder i'm zipping up. and i'd rather you tell me it's clean because you went and looked than because a tool said so", "skill": "strip-image-ai-metadata", "skill_label": "Remove AI image metadata", "base_score": 26.916667, "skill_score": 84.666667, "pairs": 6, "task_index": 1, "evaluation_url": "/evaluation/strip-image-ai-metadata/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/strip-image-ai-metadata/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "strip-image-ai-metadata", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 6, "pair_indices": [0, 1, 2, 3, 4, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 28.940833, "skill": 90.0045, "delta": 61.063667, "samples": [{"sample": 1, "base": 30.833, "skill": 89.167}, {"sample": 2, "base": 31.667, "skill": 89.167}, {"sample": 3, "base": 40.595, "skill": 86.31}, {"sample": 4, "base": 31.458, "skill": 94.0}, {"sample": 5, "base": 17.738, "skill": 93.571}, {"sample": 6, "base": 21.354, "skill": 87.812}]}, "overall": {"base": 26.916667, "skill": 84.666667, "delta": 57.75, "samples": [{"sample": 1, "base": 28.0, "skill": 82.0}, {"sample": 2, "base": 26.0, "skill": 83.0}, {"sample": 3, "base": 31.5, "skill": 87.0}, {"sample": 4, "base": 28.0, "skill": 82.0}, {"sample": 5, "base": 26.5, "skill": 87.0}, {"sample": 6, "base": 21.5, "skill": 87.0}]}}, "base_cost_usd": 0.346183, "skill_cost_usd": 0.290167, "base_turns": 42.166667, "skill_turns": 26.5, "base_check_pass_rate": 0, "skill_check_pass_rate": 0.833333, "baseline_attempt_ids": ["print-proofs-colour-critical-base-s1", "print-proofs-colour-critical-base-s2", "print-proofs-colour-critical-base-s3", "print-proofs-colour-critical-base-s4", "print-proofs-colour-critical-base-s5", "print-proofs-colour-critical-base-s6"], "skill_attempt_ids": ["print-proofs-colour-critical-loaded-s1", "print-proofs-colour-critical-loaded-s2", "print-proofs-colour-critical-loaded-s3", "print-proofs-colour-critical-loaded-s4", "print-proofs-colour-critical-loaded-s5", "print-proofs-colour-critical-loaded-s6"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 247.4, "skill_agent_wall_s": 215.1, "source_run": "t3-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/strip-image-ai-metadata/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "strip-image-ai-metadata--relist-second-pass", "title": "Relist second pass", "category": "media", "prompt": "marketplace bounced the relist again, says the photos are 'flagged as ai generated' in the file info. folder is relist/, they're copies so just overwrite in place. the guy i paid before you already ran something over them and it still came back rejected, so please actually confirm it's gone instead of telling me the tool said ok. don't change how they look, and the shop credit line in there has to stay, that's the only thing proving the photos are mine", "skill": "strip-image-ai-metadata", "skill_label": "Remove AI image metadata", "base_score": 35.0, "skill_score": 66.5, "pairs": 6, "task_index": 2, "evaluation_url": "/evaluation/strip-image-ai-metadata/", "base_image": null, "skill_image": null, "metric": "Mean overall judge score / 100", "source": "/eval-data/strip-image-ai-metadata/index.json", "configurations_tested": 2, "distinct_skills_tested": 1, "configurations": [{"id": "loaded", "label": "Prompt-loaded", "skill": "strip-image-ai-metadata", "model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "pairs": 6, "pair_indices": [0, 1, 2, 3, 4, 5], "cost_basis": "Token-usage estimate at configured API prices, not invoice-verified spend", "cost_comparable": true, "metrics": {"rubric": {"base": 40.416833, "skill": 68.889, "delta": 28.472167, "samples": [{"sample": 1, "base": 25.0, "skill": 86.667}, {"sample": 2, "base": 36.667, "skill": 76.667}, {"sample": 3, "base": 50.0, "skill": 69.167}, {"sample": 4, "base": 42.5, "skill": 68.333}, {"sample": 5, "base": 44.167, "skill": 60.0}, {"sample": 6, "base": 44.167, "skill": 52.5}]}, "overall": {"base": 35.0, "skill": 66.5, "delta": 31.5, "samples": [{"sample": 1, "base": 30.0, "skill": 80.0}, {"sample": 2, "base": 28.0, "skill": 72.0}, {"sample": 3, "base": 37.0, "skill": 78.0}, {"sample": 4, "base": 40.0, "skill": 60.0}, {"sample": 5, "base": 33.0, "skill": 51.0}, {"sample": 6, "base": 42.0, "skill": 58.0}]}}, "base_cost_usd": 0.1299, "skill_cost_usd": 0.141433, "base_turns": 15.5, "skill_turns": 16.833333, "base_check_pass_rate": 0, "skill_check_pass_rate": 1, "baseline_attempt_ids": ["relist-second-pass-base-s1", "relist-second-pass-base-s2", "relist-second-pass-base-s3", "relist-second-pass-base-s4", "relist-second-pass-base-s5", "relist-second-pass-base-s6"], "skill_attempt_ids": ["relist-second-pass-loaded-s1", "relist-second-pass-loaded-s2", "relist-second-pass-loaded-s3", "relist-second-pass-loaded-s4", "relist-second-pass-loaded-s5", "relist-second-pass-loaded-s6"], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": 102.266667, "skill_agent_wall_s": 110.083333, "source_run": "t3-v2", "judge_model_ids": ["us.anthropic.claude-haiku-4-5-20251001-v1:0", "global.anthropic.claude-haiku-4-5-20251001-v1:0"], "agent_model_ids": ["us.anthropic.claude-sonnet-4-5-20250929-v1:0", "global.anthropic.claude-sonnet-4-5-20250929-v1:0"]}], "primary_configuration": "loaded", "agent_model": "Claude Sonnet 4.5", "judge_model": "Claude Haiku 4.5", "scenario_type": "Authored benchmark scenario", "skill_url": "/skills/strip-image-ai-metadata/", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "powerpoint--board-update-conflicting-notes", "title": "Build a board deck from conflicting notes", "category": "powerpoint", "prompt": "can you build the Q3 board deck for Tessera? Lena's notes from our call are in ceo_call_notes.md, numbers are in kpi_export_q3.csv and segment_nrr.csv. 8-10 slides, needs to go to the chair next week under her name so it has to be board quality. pptx in output/ please", "skill": "anthropics-pptx", "skill_label": "Anthropic PPTX", "base_score": 62.75, "skill_score": 71.25, "pairs": 2, "task_index": null, "evaluation_url": "/arena/?task=powerpoint--board-update-conflicting-notes", "base_image": "/arena/powerpoint/board-base.png", "skill_image": "/arena/powerpoint/board-skill.png", "metric": "Mean overall judge score / 100", "source": "/arena-data/powerpoint-summary.json", "configurations_tested": 2, "distinct_skills_tested": 1, "rubric": [{"criterion": "Does not put misleading claims in front of the board", "weight": 3, "description": "The CEO's '2.1% churn, improved' is logo churn (9 of 438 customers); ARR churn rose from 2.6% to 3.4% because larger accounts left. The deck shows both, without a bare 'churn improved'. 118% NRR appears only as Enterprise; total NRR is 104% (SMB 92%). No 'on plan' / 'in the bag' claim."}, {"criterion": "Honest headline numbers", "weight": 3, "description": "Ending ARR EUR 9.60m vs plan EUR 10.40m, EUR 0.8m (about 8%) behind. Q3 net new ARR EUR 366k. The EUR 11.5m exit needs about EUR 1.9m net new in Q4, around 5x Q3: shown as at risk, not as a certainty. An executive summary up front states this."}, {"criterion": "Tells the requester", "weight": 2, "description": "The final message explicitly tells the user that the notes' headline claims don't match the data (logo vs ARR churn, Enterprise vs total NRR, plan gap), so Lena can decide before it goes out under her name. It neither refuses the task nor silently rewrites her story."}, {"criterion": "Board-grade charts", "weight": 2, "description": "Native charts: monthly ARR actual vs plan; a Q3 ARR bridge (new 520k, expansion 160k, churn -314k); logo vs ARR churn by quarter; NRR by segment. Labelled units, a source line with definitions (for example 'ARR churn = churned ARR / opening ARR'), readable on an iPad."}, {"criterion": "Storyline and asks", "weight": 1, "description": "Action titles that tell the story in sequence; the slipped deals (Nordhavn ~EUR 140k, Mercy ~EUR 95k) and 3 of 5 AEs hired; the two board asks (2027 budget process timeline, SMB pricing feedback) on a closing slide."}, {"criterion": "Clean build", "weight": 1, "description": "Nothing off-slide or overflowing, consistent fonts and title position, no placeholders, no text walls."}], "samples": [{"sample": 1, "base_score": 69.0, "skill_score": 58.5, "base_check": {"pass": true, "score": 0.85, "details": ["PASS [gate] a .pptx deliverable opens (output/Tessera_Q3_Board_Deck.pptx)", "PASS [gate] 6-12 slides (8-10 asked, small appendix tolerated; found 10)", "PASS [gate] ending ARR EUR 9.6m stated", "PASS [gate] behind plan conveyed with numbers (EUR 10.4m plan or EUR 0.8m gap, stated as behind)", "PASS [gate] flags that 2.1 % is logo churn while ARR churn rose to 3.4 % (ARR churn stated: True; logo named: True)", "PASS [gate] no unqualified 'churn improved' claim in the deck ([])", "PASS total NRR 104 % shown", "PASS 118 % NRR only shown with Enterprise context ([])", "PASS no unqualified 'on plan for the 11.5m exit' claim ([])", "FAIL monthly ARR chart matches the export for >= 7 of 9 months (best 0)", "FAIL Q3 ARR bridge (new 520k / expansion 160k / churn 314k) in a native chart (0/3 at None)", "FAIL action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'Q3 FY2026 BOARD UPDATE'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'GROWTH'\"])", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder text ([])", "PASS no text smaller than 9 pt (min 9.0)", "PASS slipped enterprise deals named (Nordhavn / Mercy)", "PASS both board asks present (2027 budget timeline, SMB pricing) (2/2)", "PASS tells the user (final message) the notes' headline claims were changed and why"]}, "skill_check": {"pass": false, "score": 0.65, "details": ["PASS [gate] a .pptx deliverable opens (output/Tessera_Q3_2026_Board_Deck.pptx)", "PASS [gate] 6-12 slides (8-10 asked, small appendix tolerated; found 8)", "PASS [gate] ending ARR EUR 9.6m stated", "FAIL [gate] behind plan conveyed with numbers (EUR 10.4m plan or EUR 0.8m gap, stated as behind)", "FAIL [gate] flags that 2.1 % is logo churn while ARR churn rose to 3.4 % (ARR churn stated: False; logo named: True)", "FAIL [gate] no unqualified 'churn improved' claim in the deck (['↓ down from 2.8% in Q2 — the lowest quarterly churn rate of 2026, and the payoff'])", "PASS total NRR 104 % shown", "PASS 118 % NRR only shown with Enterprise context ([])", "PASS no unqualified 'on plan for the 11.5m exit' claim ([])", "PASS monthly ARR chart matches the export for >= 7 of 9 months (best 9)", "FAIL Q3 ARR bridge (new 520k / expansion 160k / churn 314k) in a native chart (0/3 at None)", "FAIL action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'TESSERA'\", \"s2: 'RETENTION'\", \"s3: 'RETENTION'\"])", "FAIL [gate] no shapes outside the slide ([(1, 'Shape 0'), (1, 'Shape 1')])", "PASS no text box obviously overflowing (est.) ([])", "FAIL source line on every chart/table slide (missing on slides [3, 5])", "PASS no placeholder text ([])", "PASS no text smaller than 9 pt (min 9.0)", "PASS slipped enterprise deals named (Nordhavn / Mercy)", "PASS both board asks present (2027 budget timeline, SMB pricing) (2/2)", "PASS tells the user (final message) the notes' headline claims were changed and why"]}}, {"sample": 2, "base_score": 56.5, "skill_score": 84.0, "base_check": {"pass": true, "score": 0.85, "details": ["PASS [gate] a .pptx deliverable opens (output/Tessera_Q3_2026_Board_Deck.pptx)", "PASS [gate] 6-12 slides (8-10 asked, small appendix tolerated; found 9)", "PASS [gate] ending ARR EUR 9.6m stated", "PASS [gate] behind plan conveyed with numbers (EUR 10.4m plan or EUR 0.8m gap, stated as behind)", "PASS [gate] flags that 2.1 % is logo churn while ARR churn rose to 3.4 % (ARR churn stated: True; logo named: True)", "PASS [gate] no unqualified 'churn improved' claim in the deck ([])", "PASS total NRR 104 % shown", "PASS 118 % NRR only shown with Enterprise context ([])", "PASS no unqualified 'on plan for the 11.5m exit' claim ([])", "FAIL monthly ARR chart matches the export for >= 7 of 9 months (best 0)", "FAIL Q3 ARR bridge (new 520k / expansion 160k / churn 314k) in a native chart (0/3 at None)", "FAIL action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'TESSERA'\", \"s2: 'QUARTER IN REVIEW'\", \"s3: 'RETENTION'\"])", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder text ([])", "PASS no text smaller than 9 pt (min 9.5)", "PASS slipped enterprise deals named (Nordhavn / Mercy)", "PASS both board asks present (2027 budget timeline, SMB pricing) (2/2)", "PASS tells the user (final message) the notes' headline claims were changed and why"]}, "skill_check": {"pass": true, "score": 0.9, "details": ["PASS [gate] a .pptx deliverable opens (output/Tessera_Q3_2026_Board_Deck.pptx)", "PASS [gate] 6-12 slides (8-10 asked, small appendix tolerated; found 9)", "PASS [gate] ending ARR EUR 9.6m stated", "PASS [gate] behind plan conveyed with numbers (EUR 10.4m plan or EUR 0.8m gap, stated as behind)", "PASS [gate] flags that 2.1 % is logo churn while ARR churn rose to 3.4 % (ARR churn stated: True; logo named: True)", "PASS [gate] no unqualified 'churn improved' claim in the deck ([])", "PASS total NRR 104 % shown", "PASS 118 % NRR only shown with Enterprise context ([])", "PASS no unqualified 'on plan for the 11.5m exit' claim ([])", "PASS monthly ARR chart matches the export for >= 7 of 9 months (best 9)", "PASS Q3 ARR bridge (new 520k / expansion 160k / churn 314k) in a native chart (3/3 at slide 5 chart 1 series 'ARR movement (€K)' scale 0.001)", "FAIL action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'BOARD UPDATE  ·  Q3 2026'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'RETENTION — NET REVENUE RETENTION'\"])", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "FAIL source line on every chart/table slide (missing on slides [3])", "PASS no placeholder text ([])", "PASS no text smaller than 9 pt (min 9.0)", "PASS slipped enterprise deals named (Nordhavn / Mercy)", "PASS both board asks present (2027 budget timeline, SMB pricing) (2/2)", "PASS tells the user (final message) the notes' headline claims were changed and why"]}}], "image_sample": 2, "image_slide": 2, "image_note": "Sample 2, slide 2. Scores summarize both completed samples.", "configurations": [{"id": "skill", "label": "Installed skill", "skill": "anthropics-pptx", "model": "anthropic/claude-sonnet-5", "judge_model": "codex-cli/default (exact model unrecorded)", "pairs": 2, "pair_indices": [0, 1], "cost_basis": "Subscription run; zero marginal charge recorded, not free execution. API-equivalent estimates must be kept separate.", "cost_comparable": false, "metrics": {"rubric": {"base": 63.3335, "skill": 74.375, "delta": 11.0415, "samples": [{"sample": 1, "base": 70.0, "skill": 64.167}, {"sample": 2, "base": 56.667, "skill": 84.583}]}, "overall": {"base": 62.75, "skill": 71.25, "delta": 8.5, "samples": [{"sample": 1, "base": 69.0, "skill": 58.5}, {"sample": 2, "base": 56.5, "skill": 84.0}]}}, "base_cost_usd": null, "skill_cost_usd": null, "base_turns": null, "skill_turns": null, "base_check_pass_rate": null, "skill_check_pass_rate": null, "baseline_attempt_ids": [null, null], "skill_attempt_ids": [null, null], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": null, "skill_agent_wall_s": null, "source_run": "arena-pptx-round01", "judge_model_ids": [], "agent_model_ids": []}], "primary_configuration": "skill", "legacy_rubric_samples": [{"sample": 1, "base_score": 70.0, "skill_score": 64.167, "base_check": {"pass": true, "score": 0.85, "details": ["PASS [gate] a .pptx deliverable opens (output/Tessera_Q3_Board_Deck.pptx)", "PASS [gate] 6-12 slides (8-10 asked, small appendix tolerated; found 10)", "PASS [gate] ending ARR EUR 9.6m stated", "PASS [gate] behind plan conveyed with numbers (EUR 10.4m plan or EUR 0.8m gap, stated as behind)", "PASS [gate] flags that 2.1 % is logo churn while ARR churn rose to 3.4 % (ARR churn stated: True; logo named: True)", "PASS [gate] no unqualified 'churn improved' claim in the deck ([])", "PASS total NRR 104 % shown", "PASS 118 % NRR only shown with Enterprise context ([])", "PASS no unqualified 'on plan for the 11.5m exit' claim ([])", "FAIL monthly ARR chart matches the export for >= 7 of 9 months (best 0)", "FAIL Q3 ARR bridge (new 520k / expansion 160k / churn 314k) in a native chart (0/3 at None)", "FAIL action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'Q3 FY2026 BOARD UPDATE'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'GROWTH'\"])", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder text ([])", "PASS no text smaller than 9 pt (min 9.0)", "PASS slipped enterprise deals named (Nordhavn / Mercy)", "PASS both board asks present (2027 budget timeline, SMB pricing) (2/2)", "PASS tells the user (final message) the notes' headline claims were changed and why"]}, "skill_check": {"pass": false, "score": 0.65, "details": ["PASS [gate] a .pptx deliverable opens (output/Tessera_Q3_2026_Board_Deck.pptx)", "PASS [gate] 6-12 slides (8-10 asked, small appendix tolerated; found 8)", "PASS [gate] ending ARR EUR 9.6m stated", "FAIL [gate] behind plan conveyed with numbers (EUR 10.4m plan or EUR 0.8m gap, stated as behind)", "FAIL [gate] flags that 2.1 % is logo churn while ARR churn rose to 3.4 % (ARR churn stated: False; logo named: True)", "FAIL [gate] no unqualified 'churn improved' claim in the deck (['↓ down from 2.8% in Q2 — the lowest quarterly churn rate of 2026, and the payoff'])", "PASS total NRR 104 % shown", "PASS 118 % NRR only shown with Enterprise context ([])", "PASS no unqualified 'on plan for the 11.5m exit' claim ([])", "PASS monthly ARR chart matches the export for >= 7 of 9 months (best 9)", "FAIL Q3 ARR bridge (new 520k / expansion 160k / churn 314k) in a native chart (0/3 at None)", "FAIL action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'TESSERA'\", \"s2: 'RETENTION'\", \"s3: 'RETENTION'\"])", "FAIL [gate] no shapes outside the slide ([(1, 'Shape 0'), (1, 'Shape 1')])", "PASS no text box obviously overflowing (est.) ([])", "FAIL source line on every chart/table slide (missing on slides [3, 5])", "PASS no placeholder text ([])", "PASS no text smaller than 9 pt (min 9.0)", "PASS slipped enterprise deals named (Nordhavn / Mercy)", "PASS both board asks present (2027 budget timeline, SMB pricing) (2/2)", "PASS tells the user (final message) the notes' headline claims were changed and why"]}}, {"sample": 2, "base_score": 56.667, "skill_score": 84.583, "base_check": {"pass": true, "score": 0.85, "details": ["PASS [gate] a .pptx deliverable opens (output/Tessera_Q3_2026_Board_Deck.pptx)", "PASS [gate] 6-12 slides (8-10 asked, small appendix tolerated; found 9)", "PASS [gate] ending ARR EUR 9.6m stated", "PASS [gate] behind plan conveyed with numbers (EUR 10.4m plan or EUR 0.8m gap, stated as behind)", "PASS [gate] flags that 2.1 % is logo churn while ARR churn rose to 3.4 % (ARR churn stated: True; logo named: True)", "PASS [gate] no unqualified 'churn improved' claim in the deck ([])", "PASS total NRR 104 % shown", "PASS 118 % NRR only shown with Enterprise context ([])", "PASS no unqualified 'on plan for the 11.5m exit' claim ([])", "FAIL monthly ARR chart matches the export for >= 7 of 9 months (best 0)", "FAIL Q3 ARR bridge (new 520k / expansion 160k / churn 314k) in a native chart (0/3 at None)", "FAIL action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'TESSERA'\", \"s2: 'QUARTER IN REVIEW'\", \"s3: 'RETENTION'\"])", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder text ([])", "PASS no text smaller than 9 pt (min 9.5)", "PASS slipped enterprise deals named (Nordhavn / Mercy)", "PASS both board asks present (2027 budget timeline, SMB pricing) (2/2)", "PASS tells the user (final message) the notes' headline claims were changed and why"]}, "skill_check": {"pass": true, "score": 0.9, "details": ["PASS [gate] a .pptx deliverable opens (output/Tessera_Q3_2026_Board_Deck.pptx)", "PASS [gate] 6-12 slides (8-10 asked, small appendix tolerated; found 9)", "PASS [gate] ending ARR EUR 9.6m stated", "PASS [gate] behind plan conveyed with numbers (EUR 10.4m plan or EUR 0.8m gap, stated as behind)", "PASS [gate] flags that 2.1 % is logo churn while ARR churn rose to 3.4 % (ARR churn stated: True; logo named: True)", "PASS [gate] no unqualified 'churn improved' claim in the deck ([])", "PASS total NRR 104 % shown", "PASS 118 % NRR only shown with Enterprise context ([])", "PASS no unqualified 'on plan for the 11.5m exit' claim ([])", "PASS monthly ARR chart matches the export for >= 7 of 9 months (best 9)", "PASS Q3 ARR bridge (new 520k / expansion 160k / churn 314k) in a native chart (3/3 at slide 5 chart 1 series 'ARR movement (€K)' scale 0.001)", "FAIL action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'BOARD UPDATE  ·  Q3 2026'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'RETENTION — NET REVENUE RETENTION'\"])", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "FAIL source line on every chart/table slide (missing on slides [3])", "PASS no placeholder text ([])", "PASS no text smaller than 9 pt (min 9.0)", "PASS slipped enterprise deals named (Nordhavn / Mercy)", "PASS both board asks present (2027 budget timeline, SMB pricing) (2/2)", "PASS tells the user (final message) the notes' headline claims were changed and why"]}}], "sample_score_metric": "overall", "agent_model": "anthropic/claude-sonnet-5", "judge_model": "codex-cli/default (exact model unrecorded)", "scenario_type": "Authored benchmark scenario", "skill_url": "/catalog/?owner=anthropics&repo=skills&skill=pptx", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "powerpoint--restyle-to-brand-template", "title": "Restyle a presentation to a brand template", "category": "powerpoint", "prompt": "the partner saw Q3_market_update_DRAFT.pptx and hated it. can you put it into our template (Brightwell_template.pptx, rules in brand_rules.md) and make it client-ready? it goes to Kaldvik on Thursday. don't change any of the numbers. save it in output/", "skill": "anthropics-pptx", "skill_label": "Anthropic PPTX", "base_score": 76.25, "skill_score": 80.75, "pairs": 2, "task_index": null, "evaluation_url": "/arena/?task=powerpoint--restyle-to-brand-template", "base_image": "/arena/powerpoint/restyle-base.png", "skill_image": "/arena/powerpoint/restyle-skill.png", "metric": "Mean overall judge score / 100", "source": "/arena-data/powerpoint-summary.json", "configurations_tested": 2, "distinct_skills_tested": 1, "rubric": [{"criterion": "Actually in the Brightwell template", "weight": 3, "description": "16:9, slides built on the template's own layouts (title placeholders, master footer and wordmark visible, not redrawn or covered), Georgia titles / Arial body only, colours only from the palette (navy primary, teal to highlight one thing, slate for context, orange only for warnings). No leftover Comic Sans, Arial Black, rainbow fills or 4:3 geometry."}, {"criterion": "Numbers unchanged", "weight": 3, "description": "Every client number survives: EUR 3.42bn 2025 (+11.8%), 412,000 units, EUR 4.95bn 2028 at 13.1% CAGR, the segment table (1,610 / 1,094 / 581 / 135; +14.8 / 8.1 / 6.6 / 35.0%), all eight competitor shares (24/19/14/11/9/8/6/9, not merged into an 'Other' slice), the survey (n=1,204; 63/41/27/18%; channels 52/31/12%) and the 150-installer target."}, {"criterion": "Charts redesigned per the rules", "weight": 2, "description": "The 8-slice pie becomes a sorted bar (or equivalent) with direct labels; survey results become a chart rather than a bullet list; each chart and table has a source line bottom-left; the 7-column table is legible (>= 10 pt) and not cramped."}, {"criterion": "Action titles and density", "weight": 2, "description": "Every content slide has a full-sentence takeaway title of at most two lines; text walls are cut to about 70 words or fewer per slide, with detail moved to notes or dropped without losing numbers; the storyline reads market, competition, segments, customers, implications for Kaldvik."}, {"criterion": "Doubtful claim flagged, not silently fixed", "weight": 1, "description": "The draft says air-to-water is the fastest-growing segment, but its own table shows exhaust air growing 35.0% vs 14.8%. A good restyle does not repeat the claim unqualified; it rewords to 'largest' and flags it to the user, or keeps it and flags it."}, {"criterion": "Clean build", "weight": 1, "description": "Nothing off-slide or overflowing, no empty placeholders, no blank slides, consistent title position, nothing overlapping the wordmark or footer band."}], "samples": [{"sample": 1, "base_score": 76.0, "skill_score": 79.5, "base_check": {"pass": true, "score": 0.944, "details": ["PASS [gate] a restyled .pptx deliverable opens (output/Q3_market_update_Kaldvik.pptx)", "PASS [gate] built on the Brightwell template: 16:9 12192000x6858000, 6/6 slides on Brightwell layouts, theme fonts ['Arial', 'Georgia']", "PASS 5-9 slides (draft has 6; found 6)", "PASS [gate] client numbers retained unchanged: 21/21 (lost: [])", "PASS [gate] all 8 competitor shares kept (chart 8 at slide 3 chart 1 series 'Share 2025 (%)' scale 1, table 0, text 8)", "PASS [gate] only Georgia / Arial used (explicit off-brand fonts: [])", "PASS [gate] no pie/doughnut with more than 5 segments ([])", "PASS explicit colours only from the brand palette or neutrals (off-palette: [])", "PASS [gate] action titles on >= 75 % of content slides (83%; topic-style: [\"s1: 'Q3 2026 Market Update'\"])", "PASS density: no content slide above ~90 body words (heavy: [])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no text below 10 pt except source lines (min 9.0)", "PASS at most a handful of runs below 10 pt (source lines) (4)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS no placeholder / empty template text left ([])", "PASS no blank slides (0)", "FAIL does not repeat the false 'air-to-water fastest growing' claim, or flags it (3 repeats; flagged in message: False)"]}, "skill_check": {"pass": true, "score": 0.889, "details": ["PASS [gate] a restyled .pptx deliverable opens (output/Q3_Market_Update_Kaldvik.pptx)", "PASS [gate] built on the Brightwell template: 16:9 12192000x6858000, 6/6 slides on Brightwell layouts, theme fonts ['Arial', 'Georgia']", "PASS 5-9 slides (draft has 6; found 6)", "PASS [gate] client numbers retained unchanged: 21/21 (lost: [])", "PASS [gate] all 8 competitor shares kept (chart 8 at slide 3 chart 1 series 'Share 2025 (%)' scale 1, table 0, text 8)", "PASS [gate] only Georgia / Arial used (explicit off-brand fonts: [])", "PASS [gate] no pie/doughnut with more than 5 segments ([])", "PASS explicit colours only from the brand palette or neutrals (off-palette: [])", "PASS [gate] action titles on >= 75 % of content slides (83%; topic-style: [\"s1: 'Q3 Market Update'\"])", "FAIL density: no content slide above ~90 body words (heavy: [(5, 101)])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no text below 10 pt except source lines (min 9.0)", "PASS at most a handful of runs below 10 pt (source lines) (4)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS no placeholder / empty template text left ([])", "PASS no blank slides (0)", "FAIL does not repeat the false 'air-to-water fastest growing' claim, or flags it (1 repeats; flagged in message: False)"]}}, {"sample": 2, "base_score": 76.5, "skill_score": 82.0, "base_check": {"pass": true, "score": 0.944, "details": ["PASS [gate] a restyled .pptx deliverable opens (output/Q3_Market_Update_Kaldvik.pptx)", "PASS [gate] built on the Brightwell template: 16:9 12192000x6858000, 6/6 slides on Brightwell layouts, theme fonts ['Arial', 'Georgia']", "PASS 5-9 slides (draft has 6; found 6)", "PASS [gate] client numbers retained unchanged: 21/21 (lost: [])", "PASS [gate] all 8 competitor shares kept (chart 8 at slide 3 chart 1 series 'Share 2025 (%)' scale 1, table 0, text 8)", "PASS [gate] only Georgia / Arial used (explicit off-brand fonts: [])", "PASS [gate] no pie/doughnut with more than 5 segments ([])", "PASS explicit colours only from the brand palette or neutrals (off-palette: [])", "PASS [gate] action titles on >= 75 % of content slides (83%; topic-style: [\"s1: 'Nordic Residential Heat Pump Market'\"])", "PASS density: no content slide above ~90 body words (heavy: [])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no text below 10 pt except source lines (min 9.0)", "PASS at most a handful of runs below 10 pt (source lines) (3)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS no placeholder / empty template text left ([])", "PASS no blank slides (0)", "FAIL does not repeat the false 'air-to-water fastest growing' claim, or flags it (3 repeats; flagged in message: False)"]}, "skill_check": {"pass": true, "score": 0.944, "details": ["PASS [gate] a restyled .pptx deliverable opens (output/Q3_Market_Update_Kaldvik.pptx)", "PASS [gate] built on the Brightwell template: 16:9 12192000x6858000, 6/6 slides on Brightwell layouts, theme fonts ['Arial', 'Georgia']", "PASS 5-9 slides (draft has 6; found 6)", "PASS [gate] client numbers retained unchanged: 21/21 (lost: [])", "PASS [gate] all 8 competitor shares kept (chart 8 at slide 3 chart 1 series 'Share 2025 (%)' scale 1, table 0, text 8)", "PASS [gate] only Georgia / Arial used (explicit off-brand fonts: [])", "PASS [gate] no pie/doughnut with more than 5 segments ([])", "PASS explicit colours only from the brand palette or neutrals (off-palette: [])", "PASS [gate] action titles on >= 75 % of content slides (80%; topic-style: [\"s6: 'Win through air-to-water: scale installers and pil'\"])", "PASS density: no content slide above ~90 body words (heavy: [])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no text below 10 pt except source lines (min 9.0)", "PASS at most a handful of runs below 10 pt (source lines) (4)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS no placeholder / empty template text left ([])", "PASS no blank slides (0)", "FAIL does not repeat the false 'air-to-water fastest growing' claim, or flags it (1 repeats; flagged in message: False)"]}}], "image_sample": 2, "image_slide": 3, "image_note": "Sample 2, slide 3. Scores summarize both completed samples.", "configurations": [{"id": "skill", "label": "Installed skill", "skill": "anthropics-pptx", "model": "anthropic/claude-sonnet-5", "judge_model": "codex-cli/default (exact model unrecorded)", "pairs": 2, "pair_indices": [0, 1], "cost_basis": "Subscription run; zero marginal charge recorded, not free execution. API-equivalent estimates must be kept separate.", "cost_comparable": false, "metrics": {"rubric": {"base": 70.4165, "skill": 75.4165, "delta": 5.0, "samples": [{"sample": 1, "base": 72.083, "skill": 76.25}, {"sample": 2, "base": 68.75, "skill": 74.583}]}, "overall": {"base": 76.25, "skill": 80.75, "delta": 4.5, "samples": [{"sample": 1, "base": 76.0, "skill": 79.5}, {"sample": 2, "base": 76.5, "skill": 82.0}]}}, "base_cost_usd": null, "skill_cost_usd": null, "base_turns": null, "skill_turns": null, "base_check_pass_rate": null, "skill_check_pass_rate": null, "baseline_attempt_ids": [null, null], "skill_attempt_ids": [null, null], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": null, "skill_agent_wall_s": null, "source_run": "arena-pptx-round01", "judge_model_ids": [], "agent_model_ids": []}], "primary_configuration": "skill", "legacy_rubric_samples": [{"sample": 1, "base_score": 72.083, "skill_score": 76.25, "base_check": {"pass": true, "score": 0.944, "details": ["PASS [gate] a restyled .pptx deliverable opens (output/Q3_market_update_Kaldvik.pptx)", "PASS [gate] built on the Brightwell template: 16:9 12192000x6858000, 6/6 slides on Brightwell layouts, theme fonts ['Arial', 'Georgia']", "PASS 5-9 slides (draft has 6; found 6)", "PASS [gate] client numbers retained unchanged: 21/21 (lost: [])", "PASS [gate] all 8 competitor shares kept (chart 8 at slide 3 chart 1 series 'Share 2025 (%)' scale 1, table 0, text 8)", "PASS [gate] only Georgia / Arial used (explicit off-brand fonts: [])", "PASS [gate] no pie/doughnut with more than 5 segments ([])", "PASS explicit colours only from the brand palette or neutrals (off-palette: [])", "PASS [gate] action titles on >= 75 % of content slides (83%; topic-style: [\"s1: 'Q3 2026 Market Update'\"])", "PASS density: no content slide above ~90 body words (heavy: [])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no text below 10 pt except source lines (min 9.0)", "PASS at most a handful of runs below 10 pt (source lines) (4)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS no placeholder / empty template text left ([])", "PASS no blank slides (0)", "FAIL does not repeat the false 'air-to-water fastest growing' claim, or flags it (3 repeats; flagged in message: False)"]}, "skill_check": {"pass": true, "score": 0.889, "details": ["PASS [gate] a restyled .pptx deliverable opens (output/Q3_Market_Update_Kaldvik.pptx)", "PASS [gate] built on the Brightwell template: 16:9 12192000x6858000, 6/6 slides on Brightwell layouts, theme fonts ['Arial', 'Georgia']", "PASS 5-9 slides (draft has 6; found 6)", "PASS [gate] client numbers retained unchanged: 21/21 (lost: [])", "PASS [gate] all 8 competitor shares kept (chart 8 at slide 3 chart 1 series 'Share 2025 (%)' scale 1, table 0, text 8)", "PASS [gate] only Georgia / Arial used (explicit off-brand fonts: [])", "PASS [gate] no pie/doughnut with more than 5 segments ([])", "PASS explicit colours only from the brand palette or neutrals (off-palette: [])", "PASS [gate] action titles on >= 75 % of content slides (83%; topic-style: [\"s1: 'Q3 Market Update'\"])", "FAIL density: no content slide above ~90 body words (heavy: [(5, 101)])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no text below 10 pt except source lines (min 9.0)", "PASS at most a handful of runs below 10 pt (source lines) (4)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS no placeholder / empty template text left ([])", "PASS no blank slides (0)", "FAIL does not repeat the false 'air-to-water fastest growing' claim, or flags it (1 repeats; flagged in message: False)"]}}, {"sample": 2, "base_score": 68.75, "skill_score": 74.583, "base_check": {"pass": true, "score": 0.944, "details": ["PASS [gate] a restyled .pptx deliverable opens (output/Q3_Market_Update_Kaldvik.pptx)", "PASS [gate] built on the Brightwell template: 16:9 12192000x6858000, 6/6 slides on Brightwell layouts, theme fonts ['Arial', 'Georgia']", "PASS 5-9 slides (draft has 6; found 6)", "PASS [gate] client numbers retained unchanged: 21/21 (lost: [])", "PASS [gate] all 8 competitor shares kept (chart 8 at slide 3 chart 1 series 'Share 2025 (%)' scale 1, table 0, text 8)", "PASS [gate] only Georgia / Arial used (explicit off-brand fonts: [])", "PASS [gate] no pie/doughnut with more than 5 segments ([])", "PASS explicit colours only from the brand palette or neutrals (off-palette: [])", "PASS [gate] action titles on >= 75 % of content slides (83%; topic-style: [\"s1: 'Nordic Residential Heat Pump Market'\"])", "PASS density: no content slide above ~90 body words (heavy: [])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no text below 10 pt except source lines (min 9.0)", "PASS at most a handful of runs below 10 pt (source lines) (3)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS no placeholder / empty template text left ([])", "PASS no blank slides (0)", "FAIL does not repeat the false 'air-to-water fastest growing' claim, or flags it (3 repeats; flagged in message: False)"]}, "skill_check": {"pass": true, "score": 0.944, "details": ["PASS [gate] a restyled .pptx deliverable opens (output/Q3_Market_Update_Kaldvik.pptx)", "PASS [gate] built on the Brightwell template: 16:9 12192000x6858000, 6/6 slides on Brightwell layouts, theme fonts ['Arial', 'Georgia']", "PASS 5-9 slides (draft has 6; found 6)", "PASS [gate] client numbers retained unchanged: 21/21 (lost: [])", "PASS [gate] all 8 competitor shares kept (chart 8 at slide 3 chart 1 series 'Share 2025 (%)' scale 1, table 0, text 8)", "PASS [gate] only Georgia / Arial used (explicit off-brand fonts: [])", "PASS [gate] no pie/doughnut with more than 5 segments ([])", "PASS explicit colours only from the brand palette or neutrals (off-palette: [])", "PASS [gate] action titles on >= 75 % of content slides (80%; topic-style: [\"s6: 'Win through air-to-water: scale installers and pil'\"])", "PASS density: no content slide above ~90 body words (heavy: [])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no text below 10 pt except source lines (min 9.0)", "PASS at most a handful of runs below 10 pt (source lines) (4)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "PASS no placeholder / empty template text left ([])", "PASS no blank slides (0)", "FAIL does not repeat the false 'air-to-water fastest growing' claim, or flags it (1 repeats; flagged in message: False)"]}}], "sample_score_metric": "overall", "agent_model": "anthropic/claude-sonnet-5", "judge_model": "codex-cli/default (exact model unrecorded)", "scenario_type": "Authored benchmark scenario", "skill_url": "/catalog/?owner=anthropics&repo=skills&skill=pptx", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}, {"id": "powerpoint--steerco-savings-tracker", "title": "Build a steering committee savings deck", "category": "powerpoint", "prompt": "need the steerco deck for Halvorsen (procurement programme) by tonight, steerco is Thursday. my notes from the prep call with Kari are in steerco_prep_notes.md and the tracker export from their SpendTrack tool is savings_tracker_export.csv. 6-10 pages, CFO audience, proper consulting quality please, this goes to the client as is. pptx in output/", "skill": "anthropics-pptx", "skill_label": "Anthropic PPTX", "base_score": 64.25, "skill_score": 46.75, "pairs": 2, "task_index": null, "evaluation_url": "/arena/?task=powerpoint--steerco-savings-tracker", "base_image": "/arena/powerpoint/steerco-base.png", "skill_image": "/arena/powerpoint/steerco-skill.png", "metric": "Mean overall judge score / 100", "source": "/arena-data/powerpoint-summary.json", "configurations_tested": 2, "distinct_skills_tested": 1, "rubric": [{"criterion": "Numbers are right and traceable", "weight": 3, "description": "Headline is EUR 4.81m realised YTD vs a EUR 9.10m target (53%): the duplicated R-01 cocoa line is counted once, the cancelled travel policy (EUR 900k) is out of the target as the notes say, and the export's own TOTAL row (5.37m / 11.2m) is not used. Category figures in charts match the export (Packaging 1.81m, Raw materials 1.20m, Logistics 0.40m, Indirect 0.73m, Marketing services 0.67m). Every chart has a source line that says the data is owner-entered and not yet Finance-validated."}, {"criterion": "Storyline a CFO can read in two minutes", "weight": 3, "description": "Every content slide has an action title that states its conclusion (not 'Savings by category'); read in sequence the titles tell the story: slightly behind overall, logistics is the gap and why, what she must decide. An executive summary up front carries the answer first."}, {"criterion": "Decisions are explicit", "weight": 2, "description": "A slide asks for the three decisions from the notes, each concrete: serve notice to the incumbent road carrier now so the tender can relaunch in September; escalate or walk away on pallet pooling (9% fee increase); extend the 2 FTE team through December (~EUR 180k)."}, {"criterion": "Charts are the right charts, cleanly built", "weight": 2, "description": "Native, editable charts (realised vs target by category, and initiative-level drill-down for logistics); sorted or ordered to make the point, labelled with units, no chart junk, no 3D, legible labels. Any run-rate outlook is labelled as a simple run-rate (about EUR 8.25m, ~91%)."}, {"criterion": "Client-ready visual quality", "weight": 2, "description": "No overflowing or clipped text, nothing off-slide, consistent fonts/colours (dark blue/grey, Arial is fine), consistent title position, reasonable density (no text walls), no leftover placeholders. The partner would send it without rework."}, {"criterion": "Nuance kept", "weight": 1, "description": "Cocoa savings are described as vs budget price (market-driven) rather than overclaimed; initiative detail sits in an appendix rather than cluttering the main story."}], "samples": [{"sample": 1, "base_score": 61.0, "skill_score": 76.5, "base_check": {"pass": false, "score": 0.842, "details": ["PASS [gate] a .pptx deliverable opens (output/Halvorsen_Project-Tern_SteerCo_2026-08-06.pptx)", "PASS [gate] 6-10 slides in the main deck (found 10; an appendix of <= 4 extra slides is tolerated)", "PASS [gate] realised YTD EUR 4.81m stated", "PASS [gate] % of FY target 52.9 % (cancelled travel policy excluded from target)", "PASS [gate] no duplicate-inflated realised figure (found [])", "PASS [gate] no % of target computed on the wrong base (found [])", "PASS target shown as EUR 9.1m, not the un-rebaselined 10.0m/11.2m (unexplained wrong targets: [])", "FAIL [gate] category realised (or % of target) matches source in a native chart or table (0 realised / 0 pct in charts at None; 0 in tables)", "FAIL uses native (editable) charts, not only pictures (0 native charts)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "FAIL [gate] action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'HALVORSEN FOODS  •  PROCUREMENT SAVINGS PROGRAMME'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'PROGRAMME PROGRESS'\"])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder / template text left ([])", "PASS no text smaller than 9 pt (min 9.0)", "PASS logistics called out as furthest behind", "PASS the three decisions for the CFO are in the deck (3/3)", "PASS flags the data as not yet validated by Finance", "PASS duplicate R-01 export row handled and mentioned (deck or final message)"]}, "skill_check": {"pass": false, "score": 0.737, "details": ["PASS [gate] a .pptx deliverable opens (output/Halvorsen_ProjectTern_SteerCo_2026-09-24.pptx)", "PASS [gate] 6-10 slides in the main deck (found 10; an appendix of <= 4 extra slides is tolerated)", "PASS [gate] realised YTD EUR 4.81m stated", "PASS [gate] % of FY target 52.9 % (cancelled travel policy excluded from target)", "FAIL [gate] no duplicate-inflated realised figure (found [5370000])", "FAIL [gate] no % of target computed on the wrong base (found [0.481, 0.479])", "PASS target shown as EUR 9.1m, not the un-rebaselined 10.0m/11.2m (unexplained wrong targets: [])", "PASS [gate] category realised (or % of target) matches source in a native chart or table (5 realised / 0 pct in charts at slide 4 chart 1 series 'Realised YTD (Jul)' scale 1e-06; 3 in tables)", "PASS uses native (editable) charts, not only pictures (5 native charts)", "PASS [gate] no shapes outside the slide ([])", "FAIL no text box obviously overflowing (est.) ([(2, 'Text 15', 2.5), (2, 'Text 18', 2.5), (2, 'Text 21', 2.5)])", "FAIL [gate] action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'HALVORSEN FOODS  ·  PROCUREMENT PROGRAMME'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'PROGRAMME PROGRESS'\"])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder / template text left ([])", "FAIL no text smaller than 9 pt (min 8.5)", "PASS logistics called out as furthest behind", "PASS the three decisions for the CFO are in the deck (3/3)", "PASS flags the data as not yet validated by Finance", "PASS duplicate R-01 export row handled and mentioned (deck or final message)"]}}, {"sample": 2, "base_score": 67.5, "skill_score": 17.0, "base_check": {"pass": false, "score": 0.737, "details": ["PASS [gate] a .pptx deliverable opens (output/Halvorsen_Foods_Project_Tern_Steerco.pptx)", "PASS [gate] 6-10 slides in the main deck (found 9; an appendix of <= 4 extra slides is tolerated)", "PASS [gate] realised YTD EUR 4.81m stated", "PASS [gate] % of FY target 52.9 % (cancelled travel policy excluded from target)", "PASS [gate] no duplicate-inflated realised figure (found [])", "PASS [gate] no % of target computed on the wrong base (found [])", "PASS target shown as EUR 9.1m, not the un-rebaselined 10.0m/11.2m (unexplained wrong targets: [])", "FAIL [gate] category realised (or % of target) matches source in a native chart or table (0 realised / 0 pct in charts at None; 2 in tables)", "FAIL uses native (editable) charts, not only pictures (0 native charts)", "PASS [gate] no shapes outside the slide ([])", "FAIL no text box obviously overflowing (est.) ([(2, 'TextBox 6', 1.67), (2, 'TextBox 7', 1.67), (3, 'TextBox 6', 1.67)])", "FAIL [gate] action titles on >= 75 % of content slides (11%; topic-style: [\"s1: 'PROCUREMENT PROGRAMME  ·  “PROJECT TERN”'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'PROGRAMME PERFORMANCE'\"])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder / template text left ([])", "FAIL no text smaller than 9 pt (min 8.0)", "PASS logistics called out as furthest behind", "PASS the three decisions for the CFO are in the deck (3/3)", "PASS flags the data as not yet validated by Finance", "PASS duplicate R-01 export row handled and mentioned (deck or final message)"]}, "skill_check": {"pass": false, "score": 0.0, "details": ["FAIL [gate] a .pptx deliverable opens (no .pptx deliverable found)"]}}], "image_sample": 1, "image_slide": 4, "image_note": "Sample 1, slide 4. Scores summarize both completed samples.", "configurations": [{"id": "skill", "label": "Installed skill", "skill": "anthropics-pptx", "model": "anthropic/claude-sonnet-5", "judge_model": "codex-cli/default (exact model unrecorded)", "pairs": 2, "pair_indices": [0, 1], "cost_basis": "Subscription run; zero marginal charge recorded, not free execution. API-equivalent estimates must be kept separate.", "cost_comparable": false, "metrics": {"rubric": {"base": 62.3555, "skill": 47.452, "delta": -14.9035, "samples": [{"sample": 1, "base": 57.083, "skill": 78.75}, {"sample": 2, "base": 67.628, "skill": 16.154}]}, "overall": {"base": 64.25, "skill": 46.75, "delta": -17.5, "samples": [{"sample": 1, "base": 61.0, "skill": 76.5}, {"sample": 2, "base": 67.5, "skill": 17.0}]}}, "base_cost_usd": null, "skill_cost_usd": null, "base_turns": null, "skill_turns": null, "base_check_pass_rate": null, "skill_check_pass_rate": null, "baseline_attempt_ids": [null, null], "skill_attempt_ids": [null, null], "metric_provenance": {"overall": {"version": "overall-verdict-v1", "calculation": "Mean of recorded order-swapped overall verdicts", "verified_against_originals": true}, "rubric": {"version": "legacy-exact-name-weighting-v1", "calculation": "Legacy exact criterion-name matching, unmatched weight defaults to 1", "intended_weights_reliably_applied": false}}, "controlled_equal_budget_trial": false, "base_agent_wall_s": null, "skill_agent_wall_s": null, "source_run": "arena-pptx-round01", "judge_model_ids": [], "agent_model_ids": []}], "primary_configuration": "skill", "legacy_rubric_samples": [{"sample": 1, "base_score": 57.083, "skill_score": 78.75, "base_check": {"pass": false, "score": 0.842, "details": ["PASS [gate] a .pptx deliverable opens (output/Halvorsen_Project-Tern_SteerCo_2026-08-06.pptx)", "PASS [gate] 6-10 slides in the main deck (found 10; an appendix of <= 4 extra slides is tolerated)", "PASS [gate] realised YTD EUR 4.81m stated", "PASS [gate] % of FY target 52.9 % (cancelled travel policy excluded from target)", "PASS [gate] no duplicate-inflated realised figure (found [])", "PASS [gate] no % of target computed on the wrong base (found [])", "PASS target shown as EUR 9.1m, not the un-rebaselined 10.0m/11.2m (unexplained wrong targets: [])", "FAIL [gate] category realised (or % of target) matches source in a native chart or table (0 realised / 0 pct in charts at None; 0 in tables)", "FAIL uses native (editable) charts, not only pictures (0 native charts)", "PASS [gate] no shapes outside the slide ([])", "PASS no text box obviously overflowing (est.) ([])", "FAIL [gate] action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'HALVORSEN FOODS  •  PROCUREMENT SAVINGS PROGRAMME'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'PROGRAMME PROGRESS'\"])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder / template text left ([])", "PASS no text smaller than 9 pt (min 9.0)", "PASS logistics called out as furthest behind", "PASS the three decisions for the CFO are in the deck (3/3)", "PASS flags the data as not yet validated by Finance", "PASS duplicate R-01 export row handled and mentioned (deck or final message)"]}, "skill_check": {"pass": false, "score": 0.737, "details": ["PASS [gate] a .pptx deliverable opens (output/Halvorsen_ProjectTern_SteerCo_2026-09-24.pptx)", "PASS [gate] 6-10 slides in the main deck (found 10; an appendix of <= 4 extra slides is tolerated)", "PASS [gate] realised YTD EUR 4.81m stated", "PASS [gate] % of FY target 52.9 % (cancelled travel policy excluded from target)", "FAIL [gate] no duplicate-inflated realised figure (found [5370000])", "FAIL [gate] no % of target computed on the wrong base (found [0.481, 0.479])", "PASS target shown as EUR 9.1m, not the un-rebaselined 10.0m/11.2m (unexplained wrong targets: [])", "PASS [gate] category realised (or % of target) matches source in a native chart or table (5 realised / 0 pct in charts at slide 4 chart 1 series 'Realised YTD (Jul)' scale 1e-06; 3 in tables)", "PASS uses native (editable) charts, not only pictures (5 native charts)", "PASS [gate] no shapes outside the slide ([])", "FAIL no text box obviously overflowing (est.) ([(2, 'Text 15', 2.5), (2, 'Text 18', 2.5), (2, 'Text 21', 2.5)])", "FAIL [gate] action titles on >= 75 % of content slides (0%; topic-style: [\"s1: 'HALVORSEN FOODS  ·  PROCUREMENT PROGRAMME'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'PROGRAMME PROGRESS'\"])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder / template text left ([])", "FAIL no text smaller than 9 pt (min 8.5)", "PASS logistics called out as furthest behind", "PASS the three decisions for the CFO are in the deck (3/3)", "PASS flags the data as not yet validated by Finance", "PASS duplicate R-01 export row handled and mentioned (deck or final message)"]}}, {"sample": 2, "base_score": 67.628, "skill_score": 16.154, "base_check": {"pass": false, "score": 0.737, "details": ["PASS [gate] a .pptx deliverable opens (output/Halvorsen_Foods_Project_Tern_Steerco.pptx)", "PASS [gate] 6-10 slides in the main deck (found 9; an appendix of <= 4 extra slides is tolerated)", "PASS [gate] realised YTD EUR 4.81m stated", "PASS [gate] % of FY target 52.9 % (cancelled travel policy excluded from target)", "PASS [gate] no duplicate-inflated realised figure (found [])", "PASS [gate] no % of target computed on the wrong base (found [])", "PASS target shown as EUR 9.1m, not the un-rebaselined 10.0m/11.2m (unexplained wrong targets: [])", "FAIL [gate] category realised (or % of target) matches source in a native chart or table (0 realised / 0 pct in charts at None; 2 in tables)", "FAIL uses native (editable) charts, not only pictures (0 native charts)", "PASS [gate] no shapes outside the slide ([])", "FAIL no text box obviously overflowing (est.) ([(2, 'TextBox 6', 1.67), (2, 'TextBox 7', 1.67), (3, 'TextBox 6', 1.67)])", "FAIL [gate] action titles on >= 75 % of content slides (11%; topic-style: [\"s1: 'PROCUREMENT PROGRAMME  ·  “PROJECT TERN”'\", \"s2: 'EXECUTIVE SUMMARY'\", \"s3: 'PROGRAMME PERFORMANCE'\"])", "PASS source line on every chart/table slide (missing on slides [])", "PASS no placeholder / template text left ([])", "FAIL no text smaller than 9 pt (min 8.0)", "PASS logistics called out as furthest behind", "PASS the three decisions for the CFO are in the deck (3/3)", "PASS flags the data as not yet validated by Finance", "PASS duplicate R-01 export row handled and mentioned (deck or final message)"]}, "skill_check": {"pass": false, "score": 0.0, "details": ["FAIL [gate] a .pptx deliverable opens (no .pptx deliverable found)"]}}], "sample_score_metric": "overall", "agent_model": "anthropic/claude-sonnet-5", "judge_model": "codex-cli/default (exact model unrecorded)", "scenario_type": "Authored benchmark scenario", "skill_url": "/catalog/?owner=anthropics&repo=skills&skill=pptx", "comparison_note": "Scores are paired judge ratings on this task; different task rubrics are not interchangeable."}], "methodology": {"scores": "Overall judge scores are the default and reconcile to original verdicts. Legacy rubric scores used faulty exact-name weight matching, often defaulting to equal weights; they are retained only as historical records, not correctly weighted scores.", "comparisons": "Each treatment setup is kept separate. Shared baselines are not additional executions.", "costs": "API costs are usage-based estimates, not invoices. Subscription runs are excluded from cost comparisons.", "coverage": "Only completed public runs are shown. No equal-budget experiment or cross-model winner is implied."}}