{
  "id": 158343,
  "title": "ReflectRL: Learning from Golden Negative Trajectories via Reflective-to-Direct Reasoning",
  "url": "https://urgent.news/2026/08/04/reflectrl-learning-from-golden-negative-trajectories-via-reflective",
  "topic": "ai",
  "section": "AI",
  "published": "2026-08-04T17:40:08.000Z",
  "source": {
    "name": "arXiv cs.AI",
    "slug": "arxiv-cs-ai",
    "url": "https://arxiv.org/abs/2608.03972v1"
  },
  "original_language": "en",
  "account": null,
  "summary": "On-policy training has emerged as a powerful post-training paradigm for improving the reasoning capabilities of large language models, and is often enhanced by golden trajectories from stronger expert models. However, when the expert fails on harder problems, existing trajectory-guided methods lose their main source of supervision, and these failed trajectories are typically discarded as negative…",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}