<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
	<channel>
		<title>Capacity Planning on </title>
		<link>https://hiren.me/tags/capacity-planning/</link>
		<description>Recent content in Capacity Planning on </description>
		<generator>Hugo</generator>
		<language>en</language>
		
		
		
		
			<lastBuildDate>Tue, 06 Oct 2026 18:29:36 -0700</lastBuildDate>
		
			<atom:link href="https://hiren.me/tags/capacity-planning/index.xml" rel="self" type="application/rss+xml" />
			<item>
				<title>Sizing inference: from traffic to a GPU count</title>
				<link>https://hiren.me/posts/sizing-inference/</link>
				<pubDate>Tue, 06 Oct 2026 18:29:36 -0700</pubDate>
				<guid>https://hiren.me/posts/sizing-inference/</guid>
				<description>&lt;p&gt;&amp;ldquo;How many GPUs does it take to serve this model to these users?&amp;rdquo; comes up in every capacity plan. A handful of formulas get you to an estimate. This post works one problem from start to finish and introduces each formula where it&amp;rsquo;s needed. Each one is checked against numbers measured in &lt;a href=&#34;https://hiren.me/posts/weight-precision-part-1/&#34; &gt;Weight precision part 1&lt;/a&gt; and &lt;a href=&#34;https://hiren.me/posts/weight-precision-part-2/&#34; &gt;part 2&lt;/a&gt;: Qwen3-8B in BF16 on vLLM 0.30, one run on one H100 and one B200.&lt;/p&gt;</description>
			</item>
	</channel>
</rss>
