{
  "id": 9256660,
  "title": "Beyond Repeated Sampling: Learning Search Policies for LLM Reasoning",
  "url": "https://urgent.news/2026/09/22/beyond-repeated-sampling-learning-search-policies-for-llm-reasoning",
  "topic": "ai",
  "section": "AI",
  "published": "2026-09-22T16:56:53.000Z",
  "source": {
    "name": "arXiv cs.AI",
    "slug": "arxiv-cs-ai",
    "url": "https://arxiv.org/abs/2609.26704v1"
  },
  "original_language": "en",
  "account": null,
  "summary": "Large language models increasingly tackle hard reasoning problems by spending more test-time compute, yet the dominant strategy remains naive repeated sampling: draw many independent solutions and hope one is correct. Because such sampling explores only through local decoding noise, it tends to produce many near duplicate attempts rather than genuinely different ideas. We ask whether exploration…",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}