<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>VLLM on </title>
    <link>https://hiren.me/tags/vllm/</link>
    <description>Recent content in VLLM on </description>
    <generator>Hugo</generator>
    <language>en</language>
    <lastBuildDate>Wed, 30 Sep 2026 00:40:00 -0700</lastBuildDate>
    <atom:link href="https://hiren.me/tags/vllm/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Inference in production (part 2): the gateway, the router and the autoscaler</title>
      <link>https://hiren.me/posts/inference-in-production-part-2/</link>
      <pubDate>Wed, 30 Sep 2026 00:40:00 -0700</pubDate>
      <guid>https://hiren.me/posts/inference-in-production-part-2/</guid>
      <description>&lt;p&gt;&lt;a href=&#34;https://hiren.me/posts/inference-in-production-part-1/&#34; &gt;Part 1&lt;/a&gt; followed one request through prefill and decode pools on B200 and GB200 NVL72. This part covers what sits in front of those pools: the gateway that accepts the request, the router that picks the GPUs for it, and the autoscaler that decides how many GPUs there are. The last section measures why a new replica takes minutes to become useful, with vLLM cold starts I timed on H100s. The rest comes from project docs and published benchmarks, linked where they&amp;rsquo;re used.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
