{
  "id": 7924106,
  "title": "A Zeroth-Order Paradigm for LLM Preference Alignment",
  "url": "https://urgent.news/2026/09/16/a-zeroth-order-paradigm-for-llm-preference-alignment",
  "topic": "ai",
  "section": "AI",
  "published": "2026-09-16T17:59:35.000Z",
  "source": {
    "name": "arXiv cs.AI",
    "slug": "arxiv-cs-ai",
    "url": "https://arxiv.org/abs/2609.19144v1"
  },
  "original_language": "en",
  "account": null,
  "summary": "Direct preference alignment methods are widely used to align large language models (LLMs) with human preferences because of their computational and memory efficiency. However, likelihood displacement motivates alternative ways to extract information from preference pairs with small likelihood margins. In this paper, we propose and analyze Comparison-based Preference Optimization (ComPO), a…",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}