{
  "id": 11083808,
  "title": "PivotOPD: Learning to Recover from Pivotal Mistakes in Multi-Turn Agents",
  "url": "https://urgent.news/2026/09/30/pivotopd-learning-to-recover-from-pivotal-mistakes-in-multi-turn",
  "topic": "ai",
  "section": "AI",
  "published": "2026-09-30T17:48:11.000Z",
  "source": {
    "name": "arXiv cs.AI",
    "slug": "arxiv-cs-ai",
    "url": "https://arxiv.org/abs/2609.40285v1"
  },
  "original_language": "en",
  "account": null,
  "summary": "On-policy distillation (OPD) is a promising approach for training language agents, providing dense teacher supervision on student-generated trajectories. However, in multi-turn interaction, an incorrect action changes the states the student encounters later, so errors compound across turns. In preliminary experiments across three Qwen3 models (8B to 235B), we find that more than half of the…",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}