<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Reliability on </title>
    <link>https://hiren.me/tags/reliability/</link>
    <description>Recent content in Reliability on </description>
    <generator>Hugo</generator>
    <language>en</language>
    <lastBuildDate>Wed, 30 Sep 2026 11:00:00 -0700</lastBuildDate>
    <atom:link href="https://hiren.me/tags/reliability/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Inference in production (part 3): when hardware goes bad</title>
      <link>https://hiren.me/posts/inference-in-production-part-3/</link>
      <pubDate>Wed, 30 Sep 2026 11:00:00 -0700</pubDate>
      <guid>https://hiren.me/posts/inference-in-production-part-3/</guid>
      <description>&lt;p&gt;&lt;a href=&#34;https://hiren.me/posts/inference-in-production-part-1/&#34; &gt;Part 1&lt;/a&gt; followed a request through prefill and decode pools, and &lt;a href=&#34;https://hiren.me/posts/inference-in-production-part-2/&#34; &gt;part 2&lt;/a&gt; covered the gateway, router and autoscaler in front of them. This part is about what happens when a GPU, a node or an NVL72 tray fails under them: how often that happens, how it shows up, how much of the deployment it takes out, and what becomes of the requests that were running on it. The failure data comes from published papers and NVIDIA&amp;rsquo;s docs; the part about requests in flight I measured by killing vLLM replicas on Modal. The script is in the &lt;a href=&#34;https://github.com/hirenp/kv-cache-lab&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;lab repo&lt;/a&gt;.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
