{
  "title": "UI-TARS-2 computer-use results",
  "report_year": 2025,
  "period": "Results reported in the 2025 edition",
  "methodology": "These are separate benchmarks with different tasks and environments. Their percentages should not be averaged or interpreted as a universal probability of an agent completing office work. The report identifies continuing long-horizon failures.",
  "source_deck": "https://docs.google.com/presentation/d/1xiLl0VdrlNMAei8pmaX4ojIOfej6lhvZbOIK7Z6C-Go/edit",
  "web_edition_date": "2026-10-10",
  "rows": [
    {
      "measure": "OSWorld",
      "value": "47.5%",
      "definition": "Benchmark score",
      "printed_slide": 81,
      "pdf_page": 82,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2025.pdf#page=82"
    },
    {
      "measure": "WindowsAgentArena",
      "value": "50.6%",
      "definition": "Benchmark score",
      "printed_slide": 81,
      "pdf_page": 82,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2025.pdf#page=82"
    },
    {
      "measure": "AndroidWorld",
      "value": "73.3%",
      "definition": "Benchmark score",
      "printed_slide": 81,
      "pdf_page": 82,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2025.pdf#page=82"
    },
    {
      "measure": "Online-Mind2Web",
      "value": "88.2%",
      "definition": "Benchmark score",
      "printed_slide": 81,
      "pdf_page": 82,
      "source_url": "https://www.stateof.ai/State-of-AI-Report-2025.pdf#page=82"
    }
  ]
}
