{
  "id": 3944067,
  "title": "Gemini-3.5-Transcribe",
  "url": "https://urgent.news/2026/08/27/gemini-3-5-transcribe-3944067",
  "topic": "ai",
  "section": "AI",
  "published": "2026-08-27T18:03:42.000Z",
  "source": {
    "name": "Hacker News Best",
    "slug": "hacker-news-best",
    "url": "https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-5-transcribe/"
  },
  "original_language": "en",
  "account": "Gemini Audio Team introduces Gemini 3.5 Transcribe, a highly accurate speech-to-text model tailored for intelligent voice interactions. This model excels in converting raw audio into polished, formatted text, even in challenging environments with background noise, complex jargon, and disfluencies. Gemini 3.5 Transcribe enhances user experience across products like the Gemini app and Android devices, enabling new voice capabilities such as Rambler on Android and the Gemini app on macOS. Developers can now incorporate similar features into their applications using the Gemini API within Google AI Studio and the Gemini Enterprise Agent Platform. The model integrates smoothly into developer workflows, making it ideal for creating voice agents, real-time captioning tools, or post-call analytics pipelines. Available through two separate APIs, Gemini 3.5 Transcribe is designed to adapt to natural speaking styles and custom vocabulary, improving task execution via voice commands. Its superior performance marks a significant leap from the previous Chirp 3 model, with notable improvements in word error rates (WER) and reduced latency. Testing by Artificial Analysis indicates a 70% improvement in time to final transcription. Moreover, FLEURS benchmark evaluations across multiple languages and locales demonstrate its precision and effectiveness, achieving a 5.50% WER in streaming mode and a 5.04% WER in non-streaming use-cases. Beyond the Gemini API, 3.5 Transcribe enriches user experience by integrating context-aware understanding into Google's ecosystem, including Gboard, Antigravity, the Gemini app, and Chrome. Developers can leverage the Gemini Live API in platforms like Agora, Fishjam, LangChain, LiveKit, Pipecat, Vercel, and Vision Agents to build high-performance voice-driven interfaces without managing complex media streaming infrastructure. Industry partners such as vivo, Intellitek Health, and Lingopal have praised 3.5 Transcribe for its impressive latency, accuracy, and extensive language coverage.",
  "summary": "Article URL: https://blog.google/innovation-and-ai/models-and-research/gemini-models/gemini-3-5-transcribe/ Comments URL: https://news.ycombinator.com/item?id=49468818 Points: 291 # Comments: 92",
  "key_points": [],
  "editors_take": null,
  "illustration": null,
  "coverage": {
    "outlets": 2,
    "also_reported_by": [
      {
        "outlet": "Hacker News",
        "title": "Gemini-3.5-Transcribe",
        "url": "https://urgent.news/2026/08/27/gemini-3-5-transcribe",
        "published": "2026-08-27T18:03:42.000Z"
      }
    ]
  },
  "ai_generated": true,
  "disclaimer": "Summaries, key points and the editor’s take are written by software from other outlets’ reporting and may contain errors — always check the linked original."
}