{
  "id": 3868803,
  "title": "PACE: A Unified Condense-and-Extract Paradigm for Fast VLM Inference",
  "url": "https://urgent.news/2026/08/27/pace-a-unified-condense-and-extract-paradigm-for-fast-vlm-inference",
  "topic": "ai",
  "section": "AI",
  "published": "2026-08-27T14:52:09.000Z",
  "source": {
    "name": "arXiv cs.AI",
    "slug": "arxiv-cs-ai",
    "url": "https://arxiv.org/abs/2608.27206v1"
  },
  "original_language": "en",
  "account": null,
  "summary": "Vision-Language Models (VLMs) demonstrate exceptional visual reasoning capabilities, yet their inference costs escalate rapidly with the proliferation of visual tokens. Existing visual token pruning methods exhibit two fundamental limitations. First, most approaches operate exclusively post-vision encoder, leaving the substantial latency of the visual encoding phase unoptimized. Second, under…",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 1,
    "also_reported_by": []
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}