<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Serving on ZML - Model to Metal</title>
    <link>https://zml.ai/topics/serving/</link>
    <description>Recent content in Serving on ZML - Model to Metal</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Tue, 29 Sep 2026 09:00:00 +0200</lastBuildDate>
    <atom:link href="https://zml.ai/topics/serving/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>ZML/LLMD 20260929.0</title>
      <link>https://zml.ai/posts/llmd-20260929.0/</link>
      <pubDate>Tue, 29 Sep 2026 09:00:00 +0200</pubDate>
      <guid>https://zml.ai/posts/llmd-20260929.0/</guid>
      <description>&lt;p&gt;We&amp;rsquo;re happy to announce the release of &lt;code&gt;zml/llmd 20260929.0&lt;/code&gt;, which notably brings&#xA;&lt;a href=&#34;https://hf.co/deepseek-ai/DeepSeek-V4.1-Flash&#34;&gt;deepseek-ai/DeepSeek-V4.1-Flash&lt;/a&gt; support on NVIDIA and AMD GPUs.&lt;/p&gt;&#xA;&lt;h1 id=&#34;models&#34;&gt;Models&lt;/h1&gt;&#xA;&lt;p&gt;This release adds 2 model families:&lt;/p&gt;&#xA;&lt;ul&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://hf.co/deepseek-ai/DeepSeek-V4.1-Flash&#34;&gt;deepseek-ai/DeepSeek-V4.1-Flash&lt;/a&gt; (with native DSpark support)&lt;/li&gt;&#xA;&lt;li&gt;&lt;a href=&#34;https://hf.co/meta-models/Muse-Glimmer-30B&#34;&gt;meta-models/Muse-Glimmer-30B&lt;/a&gt;&lt;/li&gt;&#xA;&lt;/ul&gt;&#xA;&lt;p&gt;As stated, DeepSeek is only available on platforms with enough VRAM to host it natively. We do not plan to support custom quants&#xA;in the near future.&lt;/p&gt;&#xA;&lt;h1 id=&#34;deepseek-v41-flash&#34;&gt;DeepSeek-V4.1-Flash&lt;/h1&gt;&#xA;&lt;p&gt;Obviously the highlight, this is our first release supporting a frontier model. ZML is now mature enough to support complex&#xA;frontier models, and we wanted to bring it to you. Bear in mind this is a 763B model, weighing 510GB, so for now we only&#xA;support it on platforms LLMD supports with enough VRAM, namely NVIDIA and AMD GPUs.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
