{
  "slug": "system-prompt-doctor",
  "title": "System Prompt Doctor",
  "description": "The tested skill reduced performance on a conflicting-prompt repair task.",
  "status": "negative",
  "status_label": "Negative in this pilot",
  "status_explanation": "The baseline won two of three valid pairs. The package carries no efficacy claim.",
  "claim": "Exposed a concrete regression that must be fixed before efficacy can be claimed.",
  "method": "Three blind paired Harbor runs against the same agent without the skill.",
  "model": "openai/gpt-5.6-sol",
  "package_hash": "a4764d030b749eb4043d4b26ed248b6ef630c9155afcbace8696e4a66ea4c25e",
  "stats": [
    {
      "value": "1/3",
      "label": "Treatment wins"
    },
    {
      "value": "1/3",
      "label": "Treatment verifier passes"
    },
    {
      "value": "2/3",
      "label": "Baseline verifier passes"
    },
    {
      "value": "0",
      "label": "Tool errors"
    }
  ],
  "limitations": [
    "One representative prompt-repair task",
    "One model",
    "Three paired samples"
  ]
}
