<?xml version="1.0" encoding="UTF-8"?>
<?xml-stylesheet type="text/xsl" href="https://media.rss.com/style.xsl"?>
<rss xmlns:podcast="https://podcastindex.org/namespace/1.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd" xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:psc="http://podlove.org/simple-chapters" xmlns:atom="http://www.w3.org/2005/Atom" xml:lang="en" version="2.0">
  <channel>
    <title><![CDATA[AI Google Scholar Podcast]]></title>
    <link>https://rss.com/podcasts/ai-google-scholar-talking-turns-benchmarking-audio-foundation-models</link>
    <atom:link href="https://media.rss.com/ai-google-scholar-talking-turns-benchmarking-audio-foundation-models/feed.xml" rel="self" type="application/rss+xml"/>
    <atom:link rel="hub" href="https://pubsubhubbub.appspot.com/"/>
    <description><![CDATA[<p>Decoding AI Research paper daily on. my podcast - making the wild world of AI simple. I will address the key challenges in understanding conversational AI, multimodal AI and explore how Large Language Models are the key to unlocking the true power of machines understanding each other and humans.</p><p></p><p>We will go to the limits on exploration into the abyss of AI and beyond...</p>]]></description>
    <generator>RSS.com 2026.401.141116</generator>
    <lastBuildDate>Fri, 17 Apr 2026 13:11:23 GMT</lastBuildDate>
    <language>en</language>
    <copyright><![CDATA[Zhiyun Lu, Chung-Cheng Chiu, Ruoming Pang, Shinji Watanabe]]></copyright>
    <itunes:image href="https://media.rss.com/ai-google-scholar-talking-turns-benchmarking-audio-foundation-models/20251002_031000_df7570081f187bfb92157945d85216b0.png"/>
    <podcast:guid>84f63a74-68cd-5ce2-939d-3b41c32ee7b7</podcast:guid>
    <image>
      <url>https://media.rss.com/ai-google-scholar-talking-turns-benchmarking-audio-foundation-models/20251002_031000_df7570081f187bfb92157945d85216b0.png</url>
      <title>AI Google Scholar Podcast</title>
      <link>https://rss.com/podcasts/ai-google-scholar-talking-turns-benchmarking-audio-foundation-models</link>
    </image>
    <podcast:locked>yes</podcast:locked>
    <podcast:license>Zhiyun Lu, Chung-Cheng Chiu, Ruoming Pang, Shinji Watanabe</podcast:license>
    <itunes:author>Logan Sloan</itunes:author>
    <itunes:owner>
      <itunes:name>Logan Sloan</itunes:name>
    </itunes:owner>
    <itunes:explicit>false</itunes:explicit>
    <itunes:type>episodic</itunes:type>
    <itunes:category text="Technology"/>
    <podcast:medium>podcast</podcast:medium>
    <item>
      <title><![CDATA[Talking Turns: Benchmarking Audio Foundation Models]]></title>
      <itunes:title><![CDATA[Talking Turns: Benchmarking Audio Foundation Models]]></itunes:title>
      <description><![CDATA[<p>The paper addresses a key challenge in conversational AI: making interactions with voice assistants feel natural and interactive. Current systems often use simple methods, like waiting for a period of silence, to decide when to speak. However, human conversation is much more complex, involving a fluent succession of turns, subtle cues, interruptions, and minimal long silences or overlapping speech....</p><p>The authors propose a <strong>novel evaluation protocol to assess an AI's turn-taking capabilities</strong>. Their goal is to measure if an AI understands <em>when</em> to listen, speak, interrupt, or provide feedback (like "uh-huh") in a way that mimics natural human-human conversation. They use this protocol to test existing spoken dialogue systems and other audio FMs, revealing significant room for improvement</p>]]></description>
      <link>https://rss.com/podcasts/ai-google-scholar-talking-turns-benchmarking-audio-foundation-models/2249026</link>
      <enclosure url="https://content.rss.com/episodes/349703/2249026/ai-google-scholar-talking-turns-benchmarking-audio-foundation-models/2025_10_02_03_28_14_d2dcd420-c1e6-44cf-9c42-bb12cd7dadb6.mp3" length="17575320" type="audio/mpeg"/>
      <guid isPermaLink="false">27cc2b60-120d-49ed-9491-7a1cf87ca2a3</guid>
      <itunes:duration>1098</itunes:duration>
      <itunes:episodeType>full</itunes:episodeType>
      <itunes:season>1</itunes:season>
      <podcast:season>1</podcast:season>
      <itunes:episode>1</itunes:episode>
      <podcast:episode>1</podcast:episode>
      <itunes:explicit>false</itunes:explicit>
      <pubDate>Thu, 02 Oct 2025 03:36:36 GMT</pubDate>
      <itunes:image href="https://media.rss.com/ai-google-scholar-talking-turns-benchmarking-audio-foundation-models/ep_cover_20251002_031054_11397cef6c5a5de5db901f4160c6cf8d.png"/>
      <podcast:location rel="subject" geo="geo:30.2711286,-97.7436995" osm="R113314" country="us">Austin, Austin, Travis County, Texas, 78701, USA</podcast:location>
    </item>
  </channel>
</rss>