<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Workers AI - Developers Digest</title>
    <link>https://www.developersdigest.tech/blog/tags/workers-ai</link>
    <description>Articles about Workers AI on Developers Digest</description>
    <language>en</language>
    <lastBuildDate>Tue, 04 Aug 2026 09:36:28 GMT</lastBuildDate>
    <atom:link href="https://www.developersdigest.tech/blog/tags/workers-ai/feed.xml" rel="self" type="application/rss+xml" />
    <item>
      <title><![CDATA[Cloudflare Runs Kimi and GLM at Scale: FP8 KV Caches, INT4 Weights, and a Cache Safety Net]]></title>
      <link>https://www.developersdigest.tech/blog/cloudflare-kimi-glm-at-scale-2026</link>
      <guid isPermaLink="true">https://www.developersdigest.tech/blog/cloudflare-kimi-glm-at-scale-2026</guid>
      <description><![CDATA[Cloudflare published the serving playbook behind Workers AI running Moonshot Kimi K2.6 and Zhipu GLM 5.2: FP8 KV caches double Kimi's resident context to 1.37M tokens, INT4 weights shrink GLM 5.2's checkpoint 40%, and a page-tagging integrity check protects the shared cache at under 1% overhead. The numbers show what actually matters when open frontier models run on GPU fleets.]]></description>
      <pubDate>Mon, 03 Aug 2026 00:00:00 GMT</pubDate>
      <category>News</category>
      <category>Cloudflare</category>
      <category>Workers AI</category>
      <category>Inference</category>
      <category>Open Weights</category>
      <category>AI Infrastructure</category>
      <enclosure url="https://www.developersdigest.tech/images/blog/colibri-glm-52-slow-computer-local-inference/hero.webp" type="image/webp" />
    </item>
  </channel>
</rss>