{
  "slug": "last30days",
  "title": "/last30days",
  "description": "A controlled three-pair evaluation of the pinned /last30days skill on a date-bounded browser-agent landscape report.",
  "status": "REJECTED",
  "status_label": "No measured lift \u00b7 rejected",
  "package_hash": "bdde67d5df7fd7f73dbfff4ee7f9fea919babedd5a5b17690e6d216fc1769bc6",
  "repo_url": "https://github.com/mvanhorn/last30days-skill/tree/349ca444b4fda466e74d471dffa2aff36bb997f1/skills/last30days",
  "get_label": "Get the pinned skill",
  "model": "OpenAI GPT-5.6 Sol",
  "method": "Three blind baseline-versus-skill pairs in isolated Harbor containers. Same model, prompt, date window, and objective checker; only the skill installation changed.",
  "claim": "On this date-bounded research task, all three blind pairs tied and both arms passed every objective check; this run did not measure an advantage from installing the skill.",
  "status_explanation": "All three valid pairs tied. Under Edge's preregistered gate, a tie without an approved probation case is rejected rather than called inconclusive.",
  "stats": [
    {
      "value": "0 / 3 / 0",
      "label": "wins / ties / losses"
    },
    {
      "value": "3 of 3",
      "label": "skill outputs passed"
    },
    {
      "value": "3 of 3",
      "label": "baseline outputs passed"
    },
    {
      "value": "6",
      "label": "recorded agent runs"
    }
  ],
  "limitations": [
    "One task family with three paired samples; no confidence interval can support a catalog-wide efficacy claim.",
    "The harness confirmed installation but could not determine whether the agent loaded the skill instructions.",
    "The blind judge could verify artifact and run metadata but could not compare the full report contents, so substantive differences may be under-detected."
  ]
}
