<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Speculative Decoding on 67AI Lab</title>
    <link>https://67ailab.com/tags/speculative-decoding/</link>
    <description>Recent content in Speculative Decoding on 67AI Lab</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Tue, 18 Aug 2026 23:11:08 +0000</lastBuildDate>
    <atom:link href="https://67ailab.com/tags/speculative-decoding/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>One Config Line Made My 27B Model 2.7× Faster</title>
      <link>https://67ailab.com/posts/qwen38-27b-dgx-spark-mtp-speedup/</link>
      <pubDate>Tue, 18 Aug 2026 23:11:08 +0000</pubDate>
      <guid>https://67ailab.com/posts/qwen38-27b-dgx-spark-mtp-speedup/</guid>
      <description>&lt;h2 id=&#34;deploying-qwen38-27b-on-a-dgx-spark-with-and-without-mtp&#34;&gt;Deploying Qwen3.8-27B on a DGX Spark, with and without MTP&lt;/h2&gt;&#xA;&lt;p&gt;A dense 27B model on a DGX Spark generates about 11 tokens per second. That is&#xA;not a bug, a bad build, or a thermal problem — it is arithmetic, and you can&#xA;predict it before you download the weights.&lt;/p&gt;&#xA;&lt;p&gt;Then you turn on one setting and get 29 t/s, with identical output quality.&lt;/p&gt;&#xA;&lt;p&gt;Here is the whole story, measured on the box.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
