{
  "slug": "agent-receipt",
  "title": "Agent Receipt",
  "description": "Both arms passed the corrected evidence-binding checks, and all three blind comparisons tied.",
  "status": "inconclusive",
  "status_label": "Inconclusive",
  "status_explanation": "The corrected comparison demonstrated no lift over the empty baseline.",
  "claim": "Produced public-safe receipts without demonstrating a measurable advantage.",
  "method": "Three blind paired Harbor runs against the same agent without the skill.",
  "model": "openai/gpt-5.6-sol",
  "package_hash": "fa93aa2dc0b9b4bc32c421ef972762c3f62c968a37a1b5497b73dfa81d2e0139",
  "stats": [
    {
      "value": "0/3",
      "label": "Treatment wins"
    },
    {
      "value": "3/3",
      "label": "Treatment verifier passes"
    },
    {
      "value": "3/3",
      "label": "Baseline verifier passes"
    },
    {
      "value": "0",
      "label": "Tool errors"
    }
  ],
  "limitations": [
    "One representative task",
    "One model",
    "Three valid pairs",
    "Earlier undisclosed-schema comparison was invalidated"
  ]
}
