<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>quantized.uk — updates</title>
    <link>https://quantized.uk/</link>
    <description>New models, data cadence, and site updates for LLM quantization intelligence.</description>
    <language>en</language>
    <lastBuildDate>Sat, 08 Aug 2026 12:00:00 GMT</lastBuildDate>
    <atom:link href="https://quantized.uk/feed.xml" rel="self" type="application/rss+xml"/>
    
    <item>
      <title>New cookbook: running GPT-OSS 20B/120B locally without re-quantizing (23 guides). VRAM calculator now offers MXFP4 — picking Q4_K_M for GPT-OSS overstated weights by ~14%</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-08-08-New cookbook: running GPT-OSS 20B/120B l</guid>
      <pubDate>Sat, 08 Aug 2026 12:00:00 GMT</pubDate>
      <description>New cookbook: running GPT-OSS 20B/120B locally without re-quantizing (23 guides). VRAM calculator now offers MXFP4 — picking Q4_K_M for GPT-OSS overstated weights by ~14%</description>
    </item>

    <item>
      <title>Model index +4 → 75: GPT-OSS 20B/120B (native MXFP4), GLM-4.5-Air 106B-A12B, Devstral Small 1.1 — MoE-heavy batch for 16GB cards and unified-memory Macs</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-08-07-Model index +4 → 75: GPT-OSS 20B/120B (n</guid>
      <pubDate>Fri, 07 Aug 2026 12:00:00 GMT</pubDate>
      <description>Model index +4 → 75: GPT-OSS 20B/120B (native MXFP4), GLM-4.5-Air 106B-A12B, Devstral Small 1.1 — MoE-heavy batch for 16GB cards and unified-memory Macs</description>
    </item>

    <item>
      <title>Polish: re-rendered og.png (71+ models), all 22 cookbook guides have verified stack banners</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-07-22-Polish: re-rendered og.png (71+ models),</guid>
      <pubDate>Wed, 22 Jul 2026 12:00:00 GMT</pubDate>
      <description>Polish: re-rendered og.png (71+ models), all 22 cookbook guides have verified stack banners</description>
    </item>

    <item>
      <title>Cadence pack: +4 models (Gemma 3 27B, R1-Llama-8B, Phi-4, Qwen3 1.7B), superseded tags, measured/estimated labels, Hub “recent”, weekly updates, RSS, cookbook verified stack (71 models)</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-07-22-Cadence pack: +4 models (Gemma 3 27B, R1</guid>
      <pubDate>Wed, 22 Jul 2026 12:00:00 GMT</pubDate>
      <description>Cadence pack: +4 models (Gemma 3 27B, R1-Llama-8B, Phi-4, Qwen3 1.7B), superseded tags, measured/estimated labels, Hub “recent”, weekly updates, RSS, cookbook verified stack (71 models)</description>
    </item>

    <item>
      <title>UX for real traffic: job paths, mobile GPU profile, OG/favicon, honest format heat, feedback email</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-06-26-UX for real traffic: job paths, mobile G</guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>UX for real traffic: job paths, mobile GPU profile, OG/favicon, honest format heat, feedback email</description>
    </item>

    <item>
      <title>Model index +4: Qwen3 4B, Qwen3-Coder 30B-A3B, Mistral Large 3, GLM-4-9B (67 total)</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-06-26-Model index +4: Qwen3 4B, Qwen3-Coder 30</guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>Model index +4: Qwen3 4B, Qwen3-Coder 30B-A3B, Mistral Large 3, GLM-4-9B (67 total)</description>
    </item>

    <item>
      <title>QA fixes: HF stats merge on failure, ≤3B filter, CLI/VRAM tool bugs, i18n polish</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-06-26-QA fixes: HF stats merge on failure, ≤3B</guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>QA fixes: HF stats merge on failure, ≤3B filter, CLI/VRAM tool bugs, i18n polish</description>
    </item>

    <item>
      <title>Model index +5: Qwen3 32B, 30B-A3B MoE, 235B-A22B, DeepSeek-V3, DeepSeek-R1 (63 total)</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-06-26-Model index +5: Qwen3 32B, 30B-A3B MoE, </guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>Model index +5: Qwen3 32B, 30B-A3B MoE, 235B-A22B, DeepSeek-V3, DeepSeek-R1 (63 total)</description>
    </item>

    <item>
      <title>Model index +7: Qwen3 8B/14B, Gemma 3 4B/12B, Llama 4 Scout/Maverick, Llama 3.1 405B (58 total)</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-06-26-Model index +7: Qwen3 8B/14B, Gemma 3 4B</guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>Model index +7: Qwen3 8B/14B, Gemma 3 4B/12B, Llama 4 Scout/Maverick, Llama 3.1 405B (58 total)</description>
    </item>

    <item>
      <title>Cookbook TOC scroll highlight, code block copy, Quant Hub Markdown export</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-06-25-Cookbook TOC scroll highlight, code bloc</guid>
      <pubDate>Thu, 25 Jun 2026 12:00:00 GMT</pubDate>
      <description>Cookbook TOC scroll highlight, code block copy, Quant Hub Markdown export</description>
    </item>

    <item>
      <title>Cookbook reading progress bar, model HF link copy, Quant Hub shareable filter URLs</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-06-25-Cookbook reading progress bar, model HF </guid>
      <pubDate>Thu, 25 Jun 2026 12:00:00 GMT</pubDate>
      <description>Cookbook reading progress bar, model HF link copy, Quant Hub shareable filter URLs</description>
    </item>

    <item>
      <title>Breadcrumb nav + JSON-LD, cookbook article TOC, Quant Hub GPU quick-filter chips</title>
      <link>https://quantized.uk/#changelog</link>
      <guid isPermaLink="false">changelog-2026-06-25-Breadcrumb nav + JSON-LD, cookbook artic</guid>
      <pubDate>Thu, 25 Jun 2026 12:00:00 GMT</pubDate>
      <description>Breadcrumb nav + JSON-LD, cookbook article TOC, Quant Hub GPU quick-filter chips</description>
    </item>

    <item>
      <title>Model: Qwen3 4B Instruct</title>
      <link>https://quantized.uk/quant-hub/qwen3-4b/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/qwen3-4b/</guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>Smallest Qwen3 dense with thinking mode. Q4 ~3.2GB — ideal for 8GB GPUs and edge devices.</description>
    </item>

    <item>
      <title>Model: Qwen3-Coder 30B-A3B Instruct</title>
      <link>https://quantized.uk/quant-hub/qwen3-coder-30b-a3b/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/qwen3-coder-30b-a3b/</guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>Agentic coding MoE with 3.3B active params and 256K native context. Top open coder for 16–24GB cards.</description>
    </item>

    <item>
      <title>Model: Mistral Large 3 675B Instruct</title>
      <link>https://quantized.uk/quant-hub/mistral-large-3/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/mistral-large-3/</guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>Mistral 3 flagship MoE (41B active / 675B total) with vision encoder. FP8 on 8×H200; GGUF quant for research clusters only.</description>
    </item>

    <item>
      <title>Model: GLM-4-9B-Chat</title>
      <link>https://quantized.uk/quant-hub/glm-4-9b-chat/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/glm-4-9b-chat/</guid>
      <pubDate>Fri, 26 Jun 2026 12:00:00 GMT</pubDate>
      <description>Zhipu GLM-4 open 9B with 128K context, tool calling, and strong bilingual (EN/ZH) performance.</description>
    </item>

    <item>
      <title>Model: Gemma 3 27B IT</title>
      <link>https://quantized.uk/quant-hub/gemma-3-27b-it/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/gemma-3-27b-it/</guid>
      <pubDate>Wed, 22 Jul 2026 12:00:00 GMT</pubDate>
      <description>Gemma 3 large instruct with long context and multimodal support. Q4 ~16GB — dual-GPU or 24GB card with short ctx.</description>
    </item>

    <item>
      <title>Model: DeepSeek-R1-Distill-Llama-8B</title>
      <link>https://quantized.uk/quant-hub/deepseek-r1-distill-llama-8b/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/deepseek-r1-distill-llama-8b/</guid>
      <pubDate>Wed, 22 Jul 2026 12:00:00 GMT</pubDate>
      <description>R1 reasoning distilled into Llama 3.1 8B. Best chain-of-thought for 8–12GB cards; huge community GGUF support.</description>
    </item>

    <item>
      <title>Model: Phi-4 14B</title>
      <link>https://quantized.uk/quant-hub/phi-4/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/phi-4/</guid>
      <pubDate>Wed, 22 Jul 2026 12:00:00 GMT</pubDate>
      <description>Microsoft Phi-4 dense 14B — strong reasoning for size. Q4 ~9GB fits 12GB cards with moderate context.</description>
    </item>

    <item>
      <title>Model: Qwen3 1.7B Instruct</title>
      <link>https://quantized.uk/quant-hub/qwen3-1.7b/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/qwen3-1.7b/</guid>
      <pubDate>Wed, 22 Jul 2026 12:00:00 GMT</pubDate>
      <description>Tiny Qwen3 with thinking mode. Q4 ~1.4GB — phones, NUC, and always-on local agents.</description>
    </item>

    <item>
      <title>Model: GPT-OSS 20B</title>
      <link>https://quantized.uk/quant-hub/gpt-oss-20b/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/gpt-oss-20b/</guid>
      <pubDate>Fri, 07 Aug 2026 12:00:00 GMT</pubDate>
      <description>OpenAI open-weight MoE (21B total / 3.6B active), shipped natively in MXFP4 — ~12.8GB runs on a 16GB card with no quality tax. Only 3.6B active params means CPU-offload stays usable.</description>
    </item>

    <item>
      <title>Model: GPT-OSS 120B</title>
      <link>https://quantized.uk/quant-hub/gpt-oss-120b/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/gpt-oss-120b/</guid>
      <pubDate>Fri, 07 Aug 2026 12:00:00 GMT</pubDate>
      <description>The big GPT-OSS (117B total / 5.1B active). Native MXFP4 checkpoint is ~61GB — fits one 80GB card or a 128GB unified-memory Mac. Partial offload on 24GB consumer cards is slow but works.</description>
    </item>

    <item>
      <title>Model: GLM-4.5-Air</title>
      <link>https://quantized.uk/quant-hub/glm-4.5-air/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/glm-4.5-air/</guid>
      <pubDate>Fri, 07 Aug 2026 12:00:00 GMT</pubDate>
      <description>Zhipu's agentic/reasoning MoE (106B total / 12B active). Q4 ~64GB — the sweet spot is a 96GB+ unified Mac or 2× 48GB cards. Strong tool-calling for its class.</description>
    </item>

    <item>
      <title>Model: Devstral Small 1.1 24B</title>
      <link>https://quantized.uk/quant-hub/devstral-small-2507/</link>
      <guid isPermaLink="true">https://quantized.uk/quant-hub/devstral-small-2507/</guid>
      <pubDate>Fri, 07 Aug 2026 12:00:00 GMT</pubDate>
      <description>Mistral + All Hands agentic coding model on a Mistral Small 3.1 base. Built for repo-scale tool use rather than single-file completion. Q4 ~14GB fits a 16GB card.</description>
    </item>
  </channel>
</rss>