<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:dc="http://purl.org/dc/elements/1.1/" version="2.0">
  <channel>
    <title>Hyperstack - Tutorials</title>
    <link>https://www.hyperstack.cloud/technical-resources/tutorials</link>
    <description>Hyperstack - Tutorials</description>
    <language>en</language>
    <pubDate>Wed, 26 Aug 2026 13:23:03 GMT</pubDate>
    <dc:date>2026-08-26T13:23:03Z</dc:date>
    <dc:language>en</dc:language>
    <item>
      <title>Deploy Qwen3.8 Max on GPU Cloud</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-qwen3.8-max-on-gpu-cloud-for-multi-node-2.4t-inference</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-qwen3.8-max-on-gpu-cloud-for-multi-node-2.4t-inference" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600%20(3).png" alt="Deploy Qwen3.8 Max on GPU Cloud " class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p&gt;&lt;a href="https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Qwen3.8 Max&lt;/a&gt;&lt;span style="color: #000000;"&gt; is a 2.4 trillion parameter mixture-of-experts model from the Qwen team, and 95 billion of those parameters are awake for any one token. Alibaba put the hosted version in front of developers on 3 August 2026 and published the weights shortly afterwards, which is the first time a Qwen Max class model has been released for anyone to run on their own hardware. The FP8 checkpoint is 2.50 TB across 213 safetensors shards, so a Qwen3.8 Max deployment is a multi-node exercise before it is anything else.&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-qwen3.8-max-on-gpu-cloud-for-multi-node-2.4t-inference" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600%20(3).png" alt="Deploy Qwen3.8 Max on GPU Cloud " class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p&gt;&lt;a href="https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Qwen3.8 Max&lt;/a&gt;&lt;span style="color: #000000;"&gt; is a 2.4 trillion parameter mixture-of-experts model from the Qwen team, and 95 billion of those parameters are awake for any one token. Alibaba put the hosted version in front of developers on 3 August 2026 and published the weights shortly afterwards, which is the first time a Qwen Max class model has been released for anyone to run on their own hardware. The FP8 checkpoint is 2.50 TB across 213 safetensors shards, so a Qwen3.8 Max deployment is a multi-node exercise before it is anything else.&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fdeploy-qwen3.8-max-on-gpu-cloud-for-multi-node-2.4t-inference&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Innovation</category>
      <category>AI</category>
      <category>LLM</category>
      <category>Gen AI</category>
      <category>H100</category>
      <category>Guides</category>
      <pubDate>Thu, 13 Aug 2026 10:46:36 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-qwen3.8-max-on-gpu-cloud-for-multi-node-2.4t-inference</guid>
      <dc:date>2026-08-13T10:46:36Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>Deploy MiniMax H3 on GPU Cloud</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-minimax-h3-on-gpu-cloud-for-video-and-audio-in-one-pass</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-minimax-h3-on-gpu-cloud-for-video-and-audio-in-one-pass" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600%20(2)-1.png" alt="Deploy MiniMax H3 on GPU Cloud " class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p&gt;&lt;a href="https://huggingface.co/MiniMaxAI/MiniMax-H3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;MiniMax H3&lt;/a&gt;&lt;span style="color: #000000;"&gt; is a 33 billion parameter omni-modal generative system, open sourced on 3 August 2026, and it does something the text-to-video models before it did not: it writes the soundtrack at the same time as the picture. Video latents and audio latents are predicted by the same transformer, in the same forward pass, from one packed multimodal sequence. What lands on disk is a single MP4 carrying H.264 video at 24 frames per second and AAC stereo at 32 kHz, already in sync, with no separate audio model and no post-production step.&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-minimax-h3-on-gpu-cloud-for-video-and-audio-in-one-pass" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600%20(2)-1.png" alt="Deploy MiniMax H3 on GPU Cloud " class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p&gt;&lt;a href="https://huggingface.co/MiniMaxAI/MiniMax-H3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;MiniMax H3&lt;/a&gt;&lt;span style="color: #000000;"&gt; is a 33 billion parameter omni-modal generative system, open sourced on 3 August 2026, and it does something the text-to-video models before it did not: it writes the soundtrack at the same time as the picture. Video latents and audio latents are predicted by the same transformer, in the same forward pass, from one packed multimodal sequence. What lands on disk is a single MP4 carrying H.264 video at 24 frames per second and AAC stereo at 32 kHz, already in sync, with no separate audio model and no post-production step.&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fdeploy-minimax-h3-on-gpu-cloud-for-video-and-audio-in-one-pass&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Innovation</category>
      <category>AI</category>
      <category>LLM</category>
      <category>Gen AI</category>
      <category>H100</category>
      <category>Guides</category>
      <pubDate>Tue, 11 Aug 2026 11:17:46 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-minimax-h3-on-gpu-cloud-for-video-and-audio-in-one-pass</guid>
      <dc:date>2026-08-11T11:17:46Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>Deploy Kimi K3 on GPU Cloud for Multi-Node 2.8T Inference</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-kimi-k3-on-gpu-cloud-for-multi-node-2.8t-inference</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-kimi-k3-on-gpu-cloud-for-multi-node-2.8t-inference" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/Blog%20thumbnail%20-%201000x600%20(3).png" alt="Deploy Kimi K3 on GPU Cloud for Multi-Node 2.8T Inference" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p&gt;&lt;a href="https://huggingface.co/moonshotai/Kimi-K3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Kimi K3&lt;/a&gt;&lt;span style="color: #000000;"&gt; is Moonshot AI's 2.8 trillion parameter flagship, and since the weights were published on 27 July 2026 anyone can serve it themselves. It is a sparse Mixture-of-Experts model built on Kimi Delta Attention and Attention Residuals, activating 104 billion parameters through 16 of 896 experts per token, with native vision through MoonViT-V2 and a one million token context window. The checkpoint ships in native MXFP4, and when we pulled it the 96 safetensors shards measured &lt;strong&gt;1,560.94 GB&lt;/strong&gt;. That one number decides everything about serving it. The &lt;/span&gt;&lt;a href="https://recipes.vllm.ai/moonshotai/Kimi-K3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;official vLLM recipe&lt;/a&gt;&lt;span style="color: #000000;"&gt; and the &lt;/span&gt;&lt;a href="https://vllm.ai/blog/2026-07-27-k3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;vLLM launch blog&lt;/a&gt;&lt;span style="color: #000000;"&gt; both put the floor at Blackwell: at least one 8x NVIDIA B300 node, or an NVIDIA GB300 NVL72, with 16x NVIDIA B200 also supported. NVIDIA H100 is not on that list.&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-kimi-k3-on-gpu-cloud-for-multi-node-2.8t-inference" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/Blog%20thumbnail%20-%201000x600%20(3).png" alt="Deploy Kimi K3 on GPU Cloud for Multi-Node 2.8T Inference" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p&gt;&lt;a href="https://huggingface.co/moonshotai/Kimi-K3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Kimi K3&lt;/a&gt;&lt;span style="color: #000000;"&gt; is Moonshot AI's 2.8 trillion parameter flagship, and since the weights were published on 27 July 2026 anyone can serve it themselves. It is a sparse Mixture-of-Experts model built on Kimi Delta Attention and Attention Residuals, activating 104 billion parameters through 16 of 896 experts per token, with native vision through MoonViT-V2 and a one million token context window. The checkpoint ships in native MXFP4, and when we pulled it the 96 safetensors shards measured &lt;strong&gt;1,560.94 GB&lt;/strong&gt;. That one number decides everything about serving it. The &lt;/span&gt;&lt;a href="https://recipes.vllm.ai/moonshotai/Kimi-K3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;official vLLM recipe&lt;/a&gt;&lt;span style="color: #000000;"&gt; and the &lt;/span&gt;&lt;a href="https://vllm.ai/blog/2026-07-27-k3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;vLLM launch blog&lt;/a&gt;&lt;span style="color: #000000;"&gt; both put the floor at Blackwell: at least one 8x NVIDIA B300 node, or an NVIDIA GB300 NVL72, with 16x NVIDIA B200 also supported. NVIDIA H100 is not on that list.&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fdeploy-kimi-k3-on-gpu-cloud-for-multi-node-2.8t-inference&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Innovation</category>
      <category>AI</category>
      <category>LLM</category>
      <category>Gen AI</category>
      <category>H100</category>
      <category>Guides</category>
      <pubDate>Tue, 28 Jul 2026 14:38:54 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-kimi-k3-on-gpu-cloud-for-multi-node-2.8t-inference</guid>
      <dc:date>2026-07-28T14:38:54Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>Kimi K3: A Technical Deep Dive into the First Open 3T-Class Model</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/kimi-k3-a-technical-deep-dive-into-the-first-open-3t-class-model</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/kimi-k3-a-technical-deep-dive-into-the-first-open-3t-class-model" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600%20(1).png" alt="Kimi K3: A Technical Deep Dive into the First Open 3T-Class Model" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;&lt;strong style="color: #9233e9;"&gt;Kimi K3&lt;/strong&gt; is the most capable model Moonshot AI has released, a &lt;strong style="color: #9233e9;"&gt;2.8 trillion parameter&lt;/strong&gt; system built on two architectural updates, Kimi Delta Attention and Attention Residuals, with &lt;strong style="color: #9233e9;"&gt;native vision&lt;/strong&gt; and a &lt;strong style="color: #9233e9;"&gt;one million token context window&lt;/strong&gt;. It is the world's first open 3T-class model, designed for frontier intelligence across long-horizon coding, knowledge work and reasoning, and Moonshot AI has committed to releasing the full model weights by &lt;strong style="color: #9233e9;"&gt;27 July 2026&lt;/strong&gt;. In this deep dive, we walk through the architecture, the training and inference infrastructure, the benchmark results and the case studies from the &lt;a href="https://www.kimi.com/blog/kimi-k3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;official Kimi K3 release blog&lt;/a&gt;, then look at what it takes to be ready to run the model on your own infrastructure the day the weights land.&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/kimi-k3-a-technical-deep-dive-into-the-first-open-3t-class-model" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600%20(1).png" alt="Kimi K3: A Technical Deep Dive into the First Open 3T-Class Model" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;&lt;strong style="color: #9233e9;"&gt;Kimi K3&lt;/strong&gt; is the most capable model Moonshot AI has released, a &lt;strong style="color: #9233e9;"&gt;2.8 trillion parameter&lt;/strong&gt; system built on two architectural updates, Kimi Delta Attention and Attention Residuals, with &lt;strong style="color: #9233e9;"&gt;native vision&lt;/strong&gt; and a &lt;strong style="color: #9233e9;"&gt;one million token context window&lt;/strong&gt;. It is the world's first open 3T-class model, designed for frontier intelligence across long-horizon coding, knowledge work and reasoning, and Moonshot AI has committed to releasing the full model weights by &lt;strong style="color: #9233e9;"&gt;27 July 2026&lt;/strong&gt;. In this deep dive, we walk through the architecture, the training and inference infrastructure, the benchmark results and the case studies from the &lt;a href="https://www.kimi.com/blog/kimi-k3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;official Kimi K3 release blog&lt;/a&gt;, then look at what it takes to be ready to run the model on your own infrastructure the day the weights land.&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fkimi-k3-a-technical-deep-dive-into-the-first-open-3t-class-model&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Innovation</category>
      <category>AI</category>
      <category>LLM</category>
      <category>Gen AI</category>
      <category>H100</category>
      <category>Guides</category>
      <pubDate>Tue, 21 Jul 2026 13:39:36 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/kimi-k3-a-technical-deep-dive-into-the-first-open-3t-class-model</guid>
      <dc:date>2026-07-21T13:39:36Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>Deploy Hy3 on GPU Cloud for Multi-Node 295B Inference</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-hy3-on-gpu-cloud-for-multi-node-295b-inference</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-hy3-on-gpu-cloud-for-multi-node-295b-inference" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600-1.png" alt="Deploy Hy3 on GPU Cloud for Multi-Node 295B Inference" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p&gt;&lt;a href="https://huggingface.co/tencent/Hy3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Tencent Hy3&lt;/a&gt;&lt;span style="color: #000000;"&gt; is an open-weight, hybrid fast-and-slow-thinking model built on a sparse Mixture-of-Experts architecture with &lt;/span&gt;&lt;strong style="color: #000000;"&gt;295 billion total parameters&lt;/strong&gt;&lt;span style="color: #000000;"&gt;, of which only &lt;/span&gt;&lt;strong style="color: #000000;"&gt;21 billion activate&lt;/strong&gt;&lt;span style="color: #000000;"&gt; per token. It routes across 192 experts, carries a native 256K-token context window, ships its weights in BF16 under the Apache 2.0 licence, and scores 90.4 on GPQA Diamond and 72.0 on USAMO 2026. There is one catch that shapes everything about deploying it: the &lt;/span&gt;&lt;a href="https://recipes.vllm.ai/tencent/Hy3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;official vLLM recipe&lt;/a&gt;&lt;span style="color: #000000;"&gt; states plainly that a single 8x NVIDIA H100-80G node cannot fit it. The BF16 weights alone measure 557 GiB against roughly 637 GiB of VRAM on an 8-GPU NVIDIA H100 node, which leaves about 10 GiB per card to cover the KV cache, the activations and the communication buffers. Hy3 needs two nodes.&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-hy3-on-gpu-cloud-for-multi-node-295b-inference" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600-1.png" alt="Deploy Hy3 on GPU Cloud for Multi-Node 295B Inference" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p&gt;&lt;a href="https://huggingface.co/tencent/Hy3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Tencent Hy3&lt;/a&gt;&lt;span style="color: #000000;"&gt; is an open-weight, hybrid fast-and-slow-thinking model built on a sparse Mixture-of-Experts architecture with &lt;/span&gt;&lt;strong style="color: #000000;"&gt;295 billion total parameters&lt;/strong&gt;&lt;span style="color: #000000;"&gt;, of which only &lt;/span&gt;&lt;strong style="color: #000000;"&gt;21 billion activate&lt;/strong&gt;&lt;span style="color: #000000;"&gt; per token. It routes across 192 experts, carries a native 256K-token context window, ships its weights in BF16 under the Apache 2.0 licence, and scores 90.4 on GPQA Diamond and 72.0 on USAMO 2026. There is one catch that shapes everything about deploying it: the &lt;/span&gt;&lt;a href="https://recipes.vllm.ai/tencent/Hy3" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;official vLLM recipe&lt;/a&gt;&lt;span style="color: #000000;"&gt; states plainly that a single 8x NVIDIA H100-80G node cannot fit it. The BF16 weights alone measure 557 GiB against roughly 637 GiB of VRAM on an 8-GPU NVIDIA H100 node, which leaves about 10 GiB per card to cover the KV cache, the activations and the communication buffers. Hy3 needs two nodes.&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fdeploy-hy3-on-gpu-cloud-for-multi-node-295b-inference&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Innovation</category>
      <category>AI</category>
      <category>LLM</category>
      <category>Gen AI</category>
      <category>H100</category>
      <category>Guides</category>
      <pubDate>Fri, 17 Jul 2026 10:52:19 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-hy3-on-gpu-cloud-for-multi-node-295b-inference</guid>
      <dc:date>2026-07-17T10:52:19Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>How to Use GLM-5.2 on AI Studio</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/how-to-use-glm-5.2-on-hyperstack-ai-studio-from-chat-to-agents</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/how-to-use-glm-5.2-on-hyperstack-ai-studio-from-chat-to-agents" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/Blog%20thumbnail%20-%201000x600%20(2).png" alt="How to Use GLM-5.2 on AI Studio" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;&lt;a href="https://www.hyperstack.cloud/ai-studio" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Hyperstack AI Studio&lt;/a&gt; now serves &lt;strong style="color: #9233e9;"&gt;GLM-5.2&lt;/strong&gt;, the latest open-weight model from &lt;strong style="color: #9233e9;"&gt;Z.ai&lt;/strong&gt; (formerly Zhipu AI). It is a reasoning and coding model with a &lt;strong style="color: #9233e9;"&gt;one million token context window&lt;/strong&gt;, and it is available through the same serverless API and point-and-click Playground you already use for every other model on the platform. There is no GPU to provision, no weights to download and no server to keep warm. You send a list of messages, and you get back a reply.&lt;/span&gt;&lt;/p&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;There are two ways to use it. The &lt;strong style="color: #9233e9;"&gt;Playground&lt;/strong&gt; is the fastest way to try it by hand, and the &lt;strong style="color: #9233e9;"&gt;API&lt;/strong&gt; is how you put GLM-5.2 into a product, a coding assistant or an automated pipeline. This guide covers both, with a heavy focus on the API. Every code block in the API section was run against the live endpoint, and the output shown beneath it is the real response.&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/how-to-use-glm-5.2-on-hyperstack-ai-studio-from-chat-to-agents" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/Blog%20thumbnail%20-%201000x600%20(2).png" alt="How to Use GLM-5.2 on AI Studio" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;&lt;a href="https://www.hyperstack.cloud/ai-studio" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Hyperstack AI Studio&lt;/a&gt; now serves &lt;strong style="color: #9233e9;"&gt;GLM-5.2&lt;/strong&gt;, the latest open-weight model from &lt;strong style="color: #9233e9;"&gt;Z.ai&lt;/strong&gt; (formerly Zhipu AI). It is a reasoning and coding model with a &lt;strong style="color: #9233e9;"&gt;one million token context window&lt;/strong&gt;, and it is available through the same serverless API and point-and-click Playground you already use for every other model on the platform. There is no GPU to provision, no weights to download and no server to keep warm. You send a list of messages, and you get back a reply.&lt;/span&gt;&lt;/p&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;There are two ways to use it. The &lt;strong style="color: #9233e9;"&gt;Playground&lt;/strong&gt; is the fastest way to try it by hand, and the &lt;strong style="color: #9233e9;"&gt;API&lt;/strong&gt; is how you put GLM-5.2 into a product, a coding assistant or an automated pipeline. This guide covers both, with a heavy focus on the API. Every code block in the API section was run against the live endpoint, and the output shown beneath it is the real response.&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fhow-to-use-glm-5.2-on-hyperstack-ai-studio-from-chat-to-agents&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Innovation</category>
      <category>AI</category>
      <category>LLM</category>
      <category>Gen AI</category>
      <category>AI Studio</category>
      <category>Guides</category>
      <pubDate>Tue, 30 Jun 2026 08:45:34 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/how-to-use-glm-5.2-on-hyperstack-ai-studio-from-chat-to-agents</guid>
      <dc:date>2026-06-30T08:45:34Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>Generate Stunning AI Images with AI Studio</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/generate-stunning-ai-images-with-hyperstack-ai-studio-a-complete-guide</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/generate-stunning-ai-images-with-hyperstack-ai-studio-a-complete-guide" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/WEEKLY%2056%20-%20Blog%20thumbnail%20-%201000x600%20(2).png" alt="Generate Stunning AI Images with AI Studio" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;&lt;a href="https://www.hyperstack.cloud/ai-studio" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Hyperstack AI Studio&lt;/a&gt; can now generate images. Alongside its language models, the platform serves a growing line-up of &lt;strong style="color: #9233e9;"&gt;text-to-image&lt;/strong&gt; and &lt;strong style="color: #9233e9;"&gt;image-to-image&lt;/strong&gt; models, including FLUX.1, FLUX.2, Qwen-Image, Stable Diffusion 3.5 and more, behind a single serverless API and a point-and-click Playground. There is no GPU to rent, no container to build and no model to download. You send a prompt, and you get back an image.&lt;/span&gt;&lt;/p&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;There are two ways to use it. The &lt;strong style="color: #9233e9;"&gt;Playground&lt;/strong&gt; is the fastest way to experiment by hand, and the &lt;strong style="color: #9233e9;"&gt;API&lt;/strong&gt; is how you put image generation into a product or an automated pipeline. This guide covers both, with a heavy focus on the API. Each section pairs a runnable code block with the image it produces, so you can follow along and generate the same results on Hyperstack AI Studio.&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/generate-stunning-ai-images-with-hyperstack-ai-studio-a-complete-guide" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/WEEKLY%2056%20-%20Blog%20thumbnail%20-%201000x600%20(2).png" alt="Generate Stunning AI Images with AI Studio" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;&lt;a href="https://www.hyperstack.cloud/ai-studio" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;Hyperstack AI Studio&lt;/a&gt; can now generate images. Alongside its language models, the platform serves a growing line-up of &lt;strong style="color: #9233e9;"&gt;text-to-image&lt;/strong&gt; and &lt;strong style="color: #9233e9;"&gt;image-to-image&lt;/strong&gt; models, including FLUX.1, FLUX.2, Qwen-Image, Stable Diffusion 3.5 and more, behind a single serverless API and a point-and-click Playground. There is no GPU to rent, no container to build and no model to download. You send a prompt, and you get back an image.&lt;/span&gt;&lt;/p&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;There are two ways to use it. The &lt;strong style="color: #9233e9;"&gt;Playground&lt;/strong&gt; is the fastest way to experiment by hand, and the &lt;strong style="color: #9233e9;"&gt;API&lt;/strong&gt; is how you put image generation into a product or an automated pipeline. This guide covers both, with a heavy focus on the API. Each section pairs a runnable code block with the image it produces, so you can follow along and generate the same results on Hyperstack AI Studio.&lt;/span&gt;&lt;/p&gt;   
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fgenerate-stunning-ai-images-with-hyperstack-ai-studio-a-complete-guide&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Innovation</category>
      <category>AI</category>
      <category>LLM</category>
      <category>stable diffusion</category>
      <category>AI Studio</category>
      <category>Guides</category>
      <pubDate>Tue, 23 Jun 2026 08:37:59 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/generate-stunning-ai-images-with-hyperstack-ai-studio-a-complete-guide</guid>
      <dc:date>2026-06-23T08:37:59Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>Deploy DiffusionGemma on a Cloud GPU</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-diffusiongemma-on-cloud-gpus-for-fast-high-throughput-text-generation</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-diffusiongemma-on-cloud-gpus-for-fast-high-throughput-text-generation" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600.png" alt="Deploy DiffusionGemma on a Cloud GPU" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;h2 style="line-height: 1.25; color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;What is DiffusionGemma?&lt;/span&gt;&lt;/h2&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;&lt;strong style="color: #9233e9;"&gt;DiffusionGemma&lt;/strong&gt; is an open-weights, diffusion-based language model built by Google DeepMind on the &lt;a href="https://huggingface.co/google/diffusiongemma-26B-A4B-it" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;26B-A4B Mixture-of-Experts Gemma 4 architecture&lt;/a&gt;. Instead of generating text one token at a time, DiffusionGemma generates a whole block of tokens in parallel using discrete diffusion. It carries &lt;strong style="color: #9233e9;"&gt;25.2B total parameters&lt;/strong&gt; while activating only &lt;strong style="color: #9233e9;"&gt;3.8B parameters&lt;/strong&gt; during inference. It accepts interleaved text, image, and video input to produce text output, and it ships under the Apache 2.0 license. The headline result is speed. By denoising a 256-token canvas in parallel, DiffusionGemma reaches &lt;strong style="color: #9233e9;"&gt;over 1,000 tokens per second on a single NVIDIA H100&lt;/strong&gt;, which is roughly 4x the throughput of a comparable autoregressive model.&lt;/span&gt;&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/deploy-diffusiongemma-on-cloud-gpus-for-fast-high-throughput-text-generation" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/DG-%20Blog%20thumbnail%20-%201000x600.png" alt="Deploy DiffusionGemma on a Cloud GPU" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;h2 style="line-height: 1.25; color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;What is DiffusionGemma?&lt;/span&gt;&lt;/h2&gt; 
&lt;p style="color: #d4d4d4;"&gt;&lt;span style="color: #000000;"&gt;&lt;strong style="color: #9233e9;"&gt;DiffusionGemma&lt;/strong&gt; is an open-weights, diffusion-based language model built by Google DeepMind on the &lt;a href="https://huggingface.co/google/diffusiongemma-26B-A4B-it" style="color: #9233e9; font-weight: bold; text-decoration: underline;"&gt;26B-A4B Mixture-of-Experts Gemma 4 architecture&lt;/a&gt;. Instead of generating text one token at a time, DiffusionGemma generates a whole block of tokens in parallel using discrete diffusion. It carries &lt;strong style="color: #9233e9;"&gt;25.2B total parameters&lt;/strong&gt; while activating only &lt;strong style="color: #9233e9;"&gt;3.8B parameters&lt;/strong&gt; during inference. It accepts interleaved text, image, and video input to produce text output, and it ships under the Apache 2.0 license. The headline result is speed. By denoising a 256-token canvas in parallel, DiffusionGemma reaches &lt;strong style="color: #9233e9;"&gt;over 1,000 tokens per second on a single NVIDIA H100&lt;/strong&gt;, which is roughly 4x the throughput of a comparable autoregressive model.&lt;/span&gt;&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fdeploy-diffusiongemma-on-cloud-gpus-for-fast-high-throughput-text-generation&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>Innovation</category>
      <category>AI</category>
      <category>Machine Learning</category>
      <category>LLM</category>
      <category>High-Performance Computing (HPC)</category>
      <category>H100</category>
      <pubDate>Fri, 12 Jun 2026 08:44:49 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/deploy-diffusiongemma-on-cloud-gpus-for-fast-high-throughput-text-generation</guid>
      <dc:date>2026-06-12T08:44:49Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>How to Optimise the KV Cache: A Guide to Faster, Cheaper LLM Inference</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/how-to-optimise-the-kv-cache-a-guide-to-faster-cheaper-llm-inference</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/how-to-optimise-the-kv-cache-a-guide-to-faster-cheaper-llm-inference" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/KV%20Cache-%20Blog%20thumbnail%20-%201000x600.png" alt="How to Optimise the KV Cache: A Guide to Faster, Cheaper LLM Inference" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #1a1a1a; font-size: 17px; line-height: 1.75; margin: 0 0 20px 0;"&gt;Serving LLMs efficiently comes down, again and again, to one component: the &lt;strong&gt;KV cache&lt;/strong&gt;. After the model weights themselves, it is the single biggest consumer of NVIDIA GPU memory during inference, and it is the reason long contexts and large batches get expensive fast.&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/how-to-optimise-the-kv-cache-a-guide-to-faster-cheaper-llm-inference" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/KV%20Cache-%20Blog%20thumbnail%20-%201000x600.png" alt="How to Optimise the KV Cache: A Guide to Faster, Cheaper LLM Inference" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #1a1a1a; font-size: 17px; line-height: 1.75; margin: 0 0 20px 0;"&gt;Serving LLMs efficiently comes down, again and again, to one component: the &lt;strong&gt;KV cache&lt;/strong&gt;. After the model weights themselves, it is the single biggest consumer of NVIDIA GPU memory during inference, and it is the reason long contexts and large batches get expensive fast.&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fhow-to-optimise-the-kv-cache-a-guide-to-faster-cheaper-llm-inference&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>AI</category>
      <category>LLM</category>
      <category>AI Ethics &amp; Regulation</category>
      <category>GPU Cloud</category>
      <category>H100</category>
      <category>GPU Clusters</category>
      <category>Secure Private Cloud</category>
      <category>Inference</category>
      <pubDate>Wed, 10 Jun 2026 04:15:01 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/how-to-optimise-the-kv-cache-a-guide-to-faster-cheaper-llm-inference</guid>
      <dc:date>2026-06-10T04:15:01Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
    <item>
      <title>How to Run Distributed Inference with vLLM</title>
      <link>https://www.hyperstack.cloud/technical-resources/tutorials/how-to-run-distributed-inference-with-vllm-tensor-and-pipeline-parallelism-on-nvidia-h100-gpus</link>
      <description>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/how-to-run-distributed-inference-with-vllm-tensor-and-pipeline-parallelism-on-nvidia-h100-gpus" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/vllm%20inferene%20h100%20-%20Blog%20thumbnail%20-%201000x600%20(2).png" alt="How to Run Distributed Inference with vLLM" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #1a1a1a; font-size: 17px; line-height: 1.75; margin: 0 0 20px 0;"&gt;LLMs have outgrown single GPUs. A 70B model in 16-bit precision needs roughly 140 GB just for weights, more than fits on any single 80 GB card, and that is before you reserve a single byte for the KV cache that serves requests. The moment you try, vLLM greets you with the most familiar error in the field: &lt;code style="background: #f3eaff; color: #7a1fce; padding: 2px 7px; border-radius: 5px; font-family: 'SF Mono', Consolas, monospace; font-size: 0.9em;"&gt;CUDA out of memory&lt;/code&gt;.&lt;/p&gt;</description>
      <content:encoded>&lt;div class="hs-featured-image-wrapper"&gt; 
 &lt;a href="https://www.hyperstack.cloud/technical-resources/tutorials/how-to-run-distributed-inference-with-vllm-tensor-and-pipeline-parallelism-on-nvidia-h100-gpus" title="" class="hs-featured-image-link"&gt; &lt;img src="https://www.hyperstack.cloud/hubfs/vllm%20inferene%20h100%20-%20Blog%20thumbnail%20-%201000x600%20(2).png" alt="How to Run Distributed Inference with vLLM" class="hs-featured-image" style="width:auto !important; max-width:50%; float:left; margin:0 15px 15px 0;"&gt; &lt;/a&gt; 
&lt;/div&gt; 
&lt;p style="color: #1a1a1a; font-size: 17px; line-height: 1.75; margin: 0 0 20px 0;"&gt;LLMs have outgrown single GPUs. A 70B model in 16-bit precision needs roughly 140 GB just for weights, more than fits on any single 80 GB card, and that is before you reserve a single byte for the KV cache that serves requests. The moment you try, vLLM greets you with the most familiar error in the field: &lt;code style="background: #f3eaff; color: #7a1fce; padding: 2px 7px; border-radius: 5px; font-family: 'SF Mono', Consolas, monospace; font-size: 0.9em;"&gt;CUDA out of memory&lt;/code&gt;.&lt;/p&gt;  
&lt;img src="https://track-eu1.hubspot.com/__ptq.gif?a=26282475&amp;amp;k=14&amp;amp;r=https%3A%2F%2Fwww.hyperstack.cloud%2Ftechnical-resources%2Ftutorials%2Fhow-to-run-distributed-inference-with-vllm-tensor-and-pipeline-parallelism-on-nvidia-h100-gpus&amp;amp;bu=https%253A%252F%252Fwww.hyperstack.cloud%252Ftechnical-resources%252Ftutorials&amp;amp;bvt=rss" alt="" width="1" height="1" style="min-height:1px!important;width:1px!important;border-width:0!important;margin-top:0!important;margin-bottom:0!important;margin-right:0!important;margin-left:0!important;padding-top:0!important;padding-bottom:0!important;padding-right:0!important;padding-left:0!important; "&gt;</content:encoded>
      <category>AI</category>
      <category>Cloud Computing</category>
      <category>GPU Cloud</category>
      <category>H100</category>
      <category>Inference</category>
      <category>Guides</category>
      <pubDate>Tue, 02 Jun 2026 09:47:19 GMT</pubDate>
      <guid>https://www.hyperstack.cloud/technical-resources/tutorials/how-to-run-distributed-inference-with-vllm-tensor-and-pipeline-parallelism-on-nvidia-h100-gpus</guid>
      <dc:date>2026-06-02T09:47:19Z</dc:date>
      <dc:creator>Fareed Khan</dc:creator>
    </item>
  </channel>
</rss>
