{
  "id": 11793177,
  "title": "The Agent Said It Was Done. The Database Disagreed.",
  "url": "https://urgent.news/2026/10/03/the-agent-said-it-was-done-the-database-disagreed",
  "topic": "ai",
  "section": "AI",
  "published": "2026-10-03T22:56:48.000Z",
  "source": {
    "name": "Hugging Face",
    "slug": "hugging-face",
    "url": "https://huggingface.co/blog/microsoft/thinkingbox"
  },
  "original_language": "en",
  "account": "Microsoft's ThinkingBox evaluates AI agents based on the records they generate and the changes they leave in databases, rather than solely on the sentences they produce. This benchmark assesses the consistency of AI agents across stateful business workflows when interacting with large language models (LLMs). The findings show that an agent can perform well in a single trial but still leave incorrect data in the database, leading to inconsistencies in overall performance. Among the 12 LLM models tested, Claude Opus 5.5 performed the best at 67.16% in consistent performance, while Kimi-K3 had the broadest coverage, solving 93.89% of tasks at least once. However, Kimi-K3 had the lowest consistency, as only 13.41% of its tasks succeeded in all 20 attempts. The study concludes that pass@1 is an inadequate measure for deployment, as it does not account for the consistency of an agent's actions in real-world scenarios.",
  "summary": null,
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}