{
  "id": 11339934,
  "title": "TRACE: Trajectory Return Attribution and Contrastive Erasure for Multi-Turn Safety",
  "url": "https://urgent.news/2026/10/01/trace-trajectory-return-attribution-and-contrastive-erasure-for-multi",
  "topic": "ai",
  "section": "AI",
  "published": "2026-10-01T08:49:41.000Z",
  "source": {
    "name": "arXiv cs.AI",
    "slug": "arxiv-cs-ai",
    "url": "https://arxiv.org/abs/2610.01323v1"
  },
  "original_language": "en",
  "account": null,
  "summary": "Safety-aligned large language models (LLMs) often refuse a harmful request but comply once the same goal is spread over several turns. Preference objectives score whole responses to single prompts, so their training loss alone cannot control risk on unseen histories. Our analysis gives sufficient conditions under which suppression at supervised single-turn contexts yields a bound on multi-turn…",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}