{
  "id": 246277,
  "title": "The Illusion of Visual Tool-Use: A Causal Audit of Thinking with Images",
  "url": "https://urgent.news/2026/08/06/the-illusion-of-visual-tool-use-a-causal-audit-of-thinking-with-images",
  "topic": "ai",
  "section": "AI",
  "published": "2026-08-06T17:01:08.000Z",
  "source": {
    "name": "arXiv cs.AI",
    "slug": "arxiv-cs-ai",
    "url": "https://arxiv.org/abs/2608.06270v1"
  },
  "original_language": "en",
  "account": null,
  "summary": "The \"thinking-with-images\" paradigm equips multimodal LLMs with active visual operations such as crop-and-zoom. However, models using these operations often achieve only marginal or negative gains over direct inference at substantially higher token cost. They may also repeatedly crop irrelevant regions and fail on questions that direct inference answers correctly. We ask whether the returned…",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}