<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:content="http://purl.org/rss/1.0/modules/content/">
  <channel>
    <title>Tensorrt on Juntak Noh — AI Notes</title>
    <link>https://ai.klavierhye.cc/tags/tensorrt/</link>
    <description>Recent content in Tensorrt on Juntak Noh — AI Notes</description>
    <generator>Hugo -- 0.147.7</generator>
    <language>en</language>
    <lastBuildDate>Tue, 30 Jun 2026 00:00:00 +0000</lastBuildDate>
    <atom:link href="https://ai.klavierhye.cc/tags/tensorrt/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>The Bigger Model Answered Faster: Latency Lessons from a Real-Time Voice Agent</title>
      <link>https://ai.klavierhye.cc/posts/bigger-model-faster-voice-latency/</link>
      <pubDate>Tue, 30 Jun 2026 00:00:00 +0000</pubDate>
      <guid>https://ai.klavierhye.cc/posts/bigger-model-faster-voice-latency/</guid>
      <description>&lt;p&gt;&lt;em&gt;This is &lt;strong&gt;Part 3&lt;/strong&gt; of Field Notes from a Korean Phone Voice Agent, a seven-part series about a project I led: a real-time Korean phone voice agent (STT → LLM → TTS) for a public-service call line. &lt;a href=&#34;https://ai.klavierhye.cc/posts/stt-eval-real-calls/&#34;&gt;Part 1&lt;/a&gt; covered how my STT evaluation misled me, and &lt;a href=&#34;https://ai.klavierhye.cc/posts/dont-let-the-llm-read-numbers/&#34;&gt;Part 2&lt;/a&gt; where rules beat models. This part is about latency: where the seconds went, and why the fixes that mattered were almost never &amp;ldquo;use a smaller model.&amp;rdquo;&lt;/em&gt;&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
