<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Together on </title>
    <link>https://hiren.me/tags/together/</link>
    <description>Recent content in Together on </description>
    <generator>Hugo</generator>
    <language>en</language>
    <lastBuildDate>Mon, 05 Oct 2026 16:20:00 -0700</lastBuildDate>
    <atom:link href="https://hiren.me/tags/together/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Inference engines (part 3): the stacks you can&#39;t see</title>
      <link>https://hiren.me/posts/inference-engines-part-3/</link>
      <pubDate>Mon, 05 Oct 2026 16:20:00 -0700</pubDate>
      <guid>https://hiren.me/posts/inference-engines-part-3/</guid>
      <description>&lt;p&gt;&lt;a href=&#34;https://hiren.me/posts/inference-engines-part-1/&#34; &gt;Part 1&lt;/a&gt; covered what an engine is, and &lt;a href=&#34;https://hiren.me/posts/inference-engines-part-2/&#34; &gt;part 2&lt;/a&gt; how to deploy one. The alternative is calling an API, where the provider picks the engine, the GPUs and the number formats. This post maps those options, what each provider says about its stack, and how to evaluate one you can&amp;rsquo;t inspect.&lt;/p&gt;&#xA;&lt;figure class=&#34;espec&#34;&gt;&#xA;  &lt;style&gt;&#xA;    .espec { --ep-box: rgba(128,128,128,.10); --ep-edge: rgba(128,128,128,.55);&#xA;             border: 1px solid rgba(128,128,128,.35); border-radius: 8px; padding: 14px; margin: 1.5em 0; }&#xA;    .espec svg { width: 100%; height: auto; display: block; }&#xA;    .espec svg text { fill: currentColor; font-size: 12.5px; }&#xA;    .espec svg .t { font-weight: 700; font-size: 13px; }&#xA;    .espec svg .s { font-size: 10.5px; opacity: .8; }&#xA;    .espec svg .h { font-size: 10.5px; font-weight: 700; opacity: .7; }&#xA;    .espec svg rect { fill: var(--ep-box); stroke: var(--ep-edge); }&#xA;    .espec svg line { stroke: var(--ep-edge); stroke-width: 1.5; }&#xA;    .espec figcaption { font-size: 12.5px; opacity: .8; margin-top: 8px; }&#xA;  &lt;/style&gt;&#xA;  &lt;svg viewBox=&#34;0 0 760 270&#34; role=&#34;img&#34; aria-label=&#34;Ways to serve a model, from running an engine yourself to custom chips&#34;&gt;&#xA;    &lt;defs&gt;&lt;marker id=&#34;ep-a&#34; viewBox=&#34;0 0 10 10&#34; refX=&#34;9&#34; refY=&#34;5&#34; markerWidth=&#34;7&#34; markerHeight=&#34;7&#34; orient=&#34;auto&#34;&gt;&lt;path d=&#34;M0,0 L10,5 L0,10 z&#34; style=&#34;fill:rgba(128,128,128,.7)&#34;/&gt;&lt;/marker&gt;&lt;/defs&gt;&#xA;    &lt;line x1=&#34;20&#34; y1=&#34;18&#34; x2=&#34;740&#34; y2=&#34;18&#34; marker-end=&#34;url(#ep-a)&#34;/&gt;&#xA;    &lt;line x1=&#34;740&#34; y1=&#34;18&#34; x2=&#34;20&#34; y2=&#34;18&#34; marker-end=&#34;url(#ep-a)&#34;/&gt;&#xA;    &lt;text class=&#34;s&#34; x=&#34;20&#34; y=&#34;38&#34;&gt;you control more, and operate more&lt;/text&gt;&#xA;    &lt;text class=&#34;s&#34; x=&#34;740&#34; y=&#34;38&#34; text-anchor=&#34;end&#34;&gt;you operate less, and see less&lt;/text&gt;&#xA;&#xA;    &lt;g&gt;&#xA;      &lt;rect x=&#34;10&#34; y=&#34;52&#34; width=&#34;176&#34; height=&#34;208&#34; rx=&#34;6&#34;/&gt;&#xA;      &lt;text class=&#34;t&#34; x=&#34;98&#34; y=&#34;76&#34; text-anchor=&#34;middle&#34;&gt;Run it yourself&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;98&#34; y=&#34;94&#34; text-anchor=&#34;middle&#34;&gt;open engine on your GPUs&lt;/text&gt;&#xA;      &lt;text class=&#34;h&#34; x=&#34;22&#34; y=&#34;124&#34;&gt;YOU PICK&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;22&#34; y=&#34;140&#34;&gt;engine, version, flags,&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;22&#34; y=&#34;154&#34;&gt;GPUs, formats&lt;/text&gt;&#xA;      &lt;text class=&#34;h&#34; x=&#34;22&#34; y=&#34;184&#34;&gt;EXAMPLES&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;22&#34; y=&#34;200&#34;&gt;vLLM, SGLang or&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;22&#34; y=&#34;214&#34;&gt;TensorRT-LLM on your&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;22&#34; y=&#34;228&#34;&gt;own cluster or cloud&lt;/text&gt;&#xA;    &lt;/g&gt;&#xA;    &lt;g&gt;&#xA;      &lt;rect x=&#34;196&#34; y=&#34;52&#34; width=&#34;176&#34; height=&#34;208&#34; rx=&#34;6&#34;/&gt;&#xA;      &lt;text class=&#34;t&#34; x=&#34;284&#34; y=&#34;76&#34; text-anchor=&#34;middle&#34;&gt;Managed open engine&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;284&#34; y=&#34;94&#34; text-anchor=&#34;middle&#34;&gt;a platform runs it for you&lt;/text&gt;&#xA;      &lt;text class=&#34;h&#34; x=&#34;208&#34; y=&#34;124&#34;&gt;YOU PICK&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;208&#34; y=&#34;140&#34;&gt;model, GPU type; often&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;208&#34; y=&#34;154&#34;&gt;the engine&lt;/text&gt;&#xA;      &lt;text class=&#34;h&#34; x=&#34;208&#34; y=&#34;184&#34;&gt;EXAMPLES&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;208&#34; y=&#34;200&#34;&gt;Baseten (picks among&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;208&#34; y=&#34;214&#34;&gt;TRT-LLM, SGLang, vLLM),&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;208&#34; y=&#34;228&#34;&gt;Anyscale (Ray + vLLM),&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;208&#34; y=&#34;242&#34;&gt;Modal&lt;/text&gt;&#xA;    &lt;/g&gt;&#xA;    &lt;g&gt;&#xA;      &lt;rect x=&#34;382&#34; y=&#34;52&#34; width=&#34;176&#34; height=&#34;208&#34; rx=&#34;6&#34;/&gt;&#xA;      &lt;text class=&#34;t&#34; x=&#34;470&#34; y=&#34;76&#34; text-anchor=&#34;middle&#34;&gt;Proprietary stack&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;470&#34; y=&#34;94&#34; text-anchor=&#34;middle&#34;&gt;their own engine on GPUs&lt;/text&gt;&#xA;      &lt;text class=&#34;h&#34; x=&#34;394&#34; y=&#34;124&#34;&gt;YOU PICK&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;394&#34; y=&#34;140&#34;&gt;model, and a dedicated&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;394&#34; y=&#34;154&#34;&gt;or shared deployment&lt;/text&gt;&#xA;      &lt;text class=&#34;h&#34; x=&#34;394&#34; y=&#34;184&#34;&gt;EXAMPLES&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;394&#34; y=&#34;200&#34;&gt;Fireworks,&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;394&#34; y=&#34;214&#34;&gt;Together&lt;/text&gt;&#xA;    &lt;/g&gt;&#xA;    &lt;g&gt;&#xA;      &lt;rect x=&#34;568&#34; y=&#34;52&#34; width=&#34;182&#34; height=&#34;208&#34; rx=&#34;6&#34;/&gt;&#xA;      &lt;text class=&#34;t&#34; x=&#34;659&#34; y=&#34;76&#34; text-anchor=&#34;middle&#34;&gt;Custom silicon&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;659&#34; y=&#34;94&#34; text-anchor=&#34;middle&#34;&gt;their own chips, not GPUs&lt;/text&gt;&#xA;      &lt;text class=&#34;h&#34; x=&#34;580&#34; y=&#34;124&#34;&gt;YOU PICK&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;580&#34; y=&#34;140&#34;&gt;model, from their list&lt;/text&gt;&#xA;      &lt;text class=&#34;h&#34; x=&#34;580&#34; y=&#34;184&#34;&gt;EXAMPLES&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;580&#34; y=&#34;200&#34;&gt;Groq (LPU),&lt;/text&gt;&#xA;      &lt;text class=&#34;s&#34; x=&#34;580&#34; y=&#34;214&#34;&gt;Cerebras (wafer-scale)&lt;/text&gt;&#xA;    &lt;/g&gt;&#xA;  &lt;/svg&gt;&#xA;  &lt;figcaption&gt;Ways to serve a model. From left to right, you run less of the stack yourself, and you can check less of what runs it.&lt;/figcaption&gt;&#xA;&lt;/figure&gt;&#xA;&#xA;&lt;h2 id=&#34;what-providers-say-about-their-stacks&#34;&gt;&#xA;  What providers say about their stacks&#xA;  &lt;a class=&#34;heading-link&#34; href=&#34;#what-providers-say-about-their-stacks&#34;&gt;&#xA;    &lt;i class=&#34;fa-solid fa-link&#34; aria-hidden=&#34;true&#34; title=&#34;Link to heading&#34;&gt;&lt;/i&gt;&#xA;    &lt;span class=&#34;sr-only&#34;&gt;Link to heading&lt;/span&gt;&#xA;  &lt;/a&gt;&#xA;&lt;/h2&gt;&#xA;&lt;table&gt;&#xA;  &lt;thead&gt;&#xA;      &lt;tr&gt;&#xA;          &lt;th&gt;Provider&lt;/th&gt;&#xA;          &lt;th&gt;What they say&lt;/th&gt;&#xA;          &lt;th&gt;Source&lt;/th&gt;&#xA;      &lt;/tr&gt;&#xA;  &lt;/thead&gt;&#xA;  &lt;tbody&gt;&#xA;      &lt;tr&gt;&#xA;          &lt;td&gt;Baseten&lt;/td&gt;&#xA;          &lt;td&gt;&amp;ldquo;we benchmark frameworks like TensorRT-LLM, SGLang, and vLLM to select the best-performing framework&amp;rdquo;&lt;/td&gt;&#xA;          &lt;td&gt;&lt;a href=&#34;https://www.baseten.co/resources/guide/the-baseten-inference-stack/&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;guide&lt;/a&gt;, Jan 2026&lt;/td&gt;&#xA;      &lt;/tr&gt;&#xA;      &lt;tr&gt;&#xA;          &lt;td&gt;Anyscale&lt;/td&gt;&#xA;          &lt;td&gt;&amp;ldquo;Ray Serve for orchestration and scaling. vLLM for inference.&amp;rdquo;&lt;/td&gt;&#xA;          &lt;td&gt;&lt;a href=&#34;https://docs.anyscale.com/llm/serving&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;docs&lt;/a&gt;&lt;/td&gt;&#xA;      &lt;/tr&gt;&#xA;      &lt;tr&gt;&#xA;          &lt;td&gt;DeepInfra&lt;/td&gt;&#xA;          &lt;td&gt;&amp;ldquo;our inference stack is built on TensorRT-LLM and NVIDIA Dynamo&amp;rdquo;&lt;/td&gt;&#xA;          &lt;td&gt;&lt;a href=&#34;https://deepinfra.com/blog/deepinfra-nvidia-inference-stack&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;blog&lt;/a&gt;, Jun 2026&lt;/td&gt;&#xA;      &lt;/tr&gt;&#xA;      &lt;tr&gt;&#xA;          &lt;td&gt;Fireworks&lt;/td&gt;&#xA;          &lt;td&gt;&amp;ldquo;Fireworks proprietary LLM serving stack, which consists of CUDA kernels, optimized for both FP16 and FP8&amp;rdquo;&lt;/td&gt;&#xA;          &lt;td&gt;&lt;a href=&#34;https://fireworks.ai/blog/fire-attention-serving-open-source-models-4x-faster-than-vllm-by-quantizing-with-no-tradeoffs&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;blog&lt;/a&gt;, Jan 2024&lt;/td&gt;&#xA;      &lt;/tr&gt;&#xA;      &lt;tr&gt;&#xA;          &lt;td&gt;Together&lt;/td&gt;&#xA;          &lt;td&gt;the Together Inference Engine builds on &amp;ldquo;FlashAttention-3, faster GEMM &amp;amp; MHA kernels, innovations in quality-preserving quantization, and speculative decoding&amp;rdquo;&lt;/td&gt;&#xA;          &lt;td&gt;&lt;a href=&#34;https://www.together.ai/blog/together-inference-engine-2&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;blog&lt;/a&gt;, Jul 2024&lt;/td&gt;&#xA;      &lt;/tr&gt;&#xA;      &lt;tr&gt;&#xA;          &lt;td&gt;Groq&lt;/td&gt;&#xA;          &lt;td&gt;&amp;ldquo;Data flow is statically scheduled by the software during compilation&amp;rdquo;&lt;/td&gt;&#xA;          &lt;td&gt;&lt;a href=&#34;https://groq.com/blog/the-groq-lpu-explained&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;blog&lt;/a&gt;, Mar 2025&lt;/td&gt;&#xA;      &lt;/tr&gt;&#xA;      &lt;tr&gt;&#xA;          &lt;td&gt;Cerebras&lt;/td&gt;&#xA;          &lt;td&gt;&amp;ldquo;we are able to integrate 44GB of SRAM on a single chip&amp;rdquo;&lt;/td&gt;&#xA;          &lt;td&gt;&lt;a href=&#34;https://www.cerebras.ai/blog/introducing-cerebras-inference-ai-at-instant-speed&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;blog&lt;/a&gt;, Aug 2024&lt;/td&gt;&#xA;      &lt;/tr&gt;&#xA;  &lt;/tbody&gt;&#xA;&lt;/table&gt;&#xA;&lt;p&gt;Fireworks&amp;rsquo; current &lt;a href=&#34;https://fireworks.ai/inference&#34;  class=&#34;external-link&#34; target=&#34;_blank&#34; rel=&#34;noopener&#34;&gt;inference page&lt;/a&gt; says &amp;ldquo;We built it from the ground up, optimized every layer we control, from GPU memory layout to the runtime&amp;rdquo;. Whether any of it started from an open engine, neither Fireworks nor Together says.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
