{
  "title": "Same model, different tools and feedback",
  "period": "Study published 2026-05-07",
  "compiled_on": "2026-10-07",
  "scope": "Selected figures compiled for an unpublished report preview; not a full underlying dataset",
  "methodology": "GLM-5.1 was held fixed. The table reports mean pass@1 over two runs on a 100-task SWE-bench Verified subset. The increase from the minimal to full setup was 13 percentage points. This is a controlled coding result, not an estimate for every task an agent might attempt.",
  "rows": [
    {
      "Agent setup": "Minimal",
      "Tasks resolved": "52.5%",
      "source_url": "https://arxiv.org/html/2605.23950v1",
      "source_period": "2026-05-07",
      "verification": "Primary source checked 2026-10-07"
    },
    {
      "Agent setup": "Improved",
      "Tasks resolved": "56.5%",
      "source_url": "https://arxiv.org/html/2605.23950v1",
      "source_period": "2026-05-07",
      "verification": "Primary source checked 2026-10-07"
    },
    {
      "Agent setup": "Full",
      "Tasks resolved": "65.5%",
      "source_url": "https://arxiv.org/html/2605.23950v1",
      "source_period": "2026-05-07",
      "verification": "Primary source checked 2026-10-07"
    }
  ]
}