{
  "id": 9081603,
  "title": "OSWorld-Pro: Process-based Evaluation for Computer Use Agents",
  "url": "https://urgent.news/2026/09/21/osworld-pro-process-based-evaluation-for-computer-use-agents",
  "topic": "ai",
  "section": "AI",
  "published": "2026-09-21T16:55:24.000Z",
  "source": {
    "name": "arXiv cs.AI",
    "slug": "arxiv-cs-ai",
    "url": "https://arxiv.org/abs/2609.24890v1"
  },
  "original_language": "en",
  "account": null,
  "summary": "Evaluation of Computer-Use Agents (CUAs) is often limited to the final deliverables they create (at the end of hundreds of steps) and assessed with functional verifiers, as seen in OSWorld. However, such evaluation of end-state performance lacks transparency into how and why agents fail in various tasks, obfuscating critical insight for subsequent improvement. For instance, agents that err during…",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}