<?xml version="1.0" encoding="utf-8"?>
<rss version="2.0">
    <channel>
        <title>ai-muninn</title>
        <link>https://ai-muninn.com/en/blog</link>
        <description>Notes on AI inference infrastructure: DGX Spark, vLLM, local AI agents.</description>
        <lastBuildDate>Sun, 06 Sep 2026 17:04:26 GMT</lastBuildDate>
        <docs>https://validator.w3.org/feed/docs/rss2.html</docs>
        <generator>https://github.com/jpmonette/feed</generator>
        <language>en</language>
        <copyright>2026 coolthor</copyright>
        <item>
            <title><![CDATA[[Benchmark] Qwen3.8-Flash-Next NVFP4 on a DGX Spark: 41.7 tok/s, RAM for Traffic, Disk for the Dictionary]]></title>
            <link>https://ai-muninn.com/en/blog/qwen38-flash-next-nvfp4-dgx-spark-vllm-recipe</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/qwen38-flash-next-nvfp4-dgx-spark-vllm-recipe</guid>
            <pubDate>Sun, 06 Sep 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[NVIDIA's NVFP4 checkpoint at 41.7 tok/s on one DGX Spark via nine bind-mounted vLLM files, plus six figures on why the 47.68 GiB n-gram table lives on NVMe.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>vLLM</category>
            <category>NVFP4</category>
            <category>MTP</category>
            <category>Qwen3.8</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Two characters in one shot: MiniMax-H3 Ref2VA takes multiple reference images]]></title>
            <link>https://ai-muninn.com/en/blog/minimax-h3-ref2va-two-characters</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/minimax-h3-ref2va-two-characters</guid>
            <pubDate>Sun, 06 Sep 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Ref2VA is the only MiniMax-H3 mode that takes several reference images. Two characters, 243 frames, 437 s on one RTX 5090, and CER 6.8% on the dialogue.]]></description>
            <category>MiniMax-H3</category>
            <category>Ref2VA</category>
            <category>ComfyUI</category>
            <category>RTX 5090</category>
            <category>AI video</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] From OOM to 8.6 minutes: a MiniMax-H3 config stack for one RTX 5090]]></title>
            <link>https://ai-muninn.com/en/blog/minimax-h3-rtx5090-config-stack</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/minimax-h3-rtx5090-config-stack</guid>
            <pubDate>Thu, 03 Sep 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[MiniMax-H3 at native 1344x768 on a single RTX 5090: 15 seconds of video in 518 s. What KJNodes chunking, torch cu130 and the NVFP4 kernels are each worth.]]></description>
            <category>MiniMax-H3</category>
            <category>RTX 5090</category>
            <category>ComfyUI</category>
            <category>NVFP4</category>
            <category>video generation</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] AI Slop Is Structural: 93.2% Detectable With Every Style Tell Stripped]]></title>
            <link>https://ai-muninn.com/en/blog/ai-slop-is-structural</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-slop-is-structural</guid>
            <pubDate>Mon, 31 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A COLM 2026 paper strips every style signal from 61,608 stories and still tells human from AI at 93.2% macro-F1. The tell is structure, not em-dashes.]]></description>
            <category>AI writing</category>
            <category>blog workflow</category>
            <category>Claude Code</category>
            <category>skill</category>
            <category>StoryScope</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Ten effect embeddings for MiniMax-H3 — 10 MB that buys you bullet time, fire breath and a full year of seasons]]></title>
            <link>https://ai-muninn.com/en/blog/minimax-h3-effect-embeddings-rtx5090</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/minimax-h3-effect-embeddings-rtx5090</guid>
            <pubDate>Mon, 31 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Ten community effect embeddings for MiniMax-H3, tested at native 1344x768 on one RTX 5090. What each one actually does, the prompt shape that lets them work, and the placement rule that decides whether they fire at all.]]></description>
            <category>MiniMax-H3</category>
            <category>ComfyUI</category>
            <category>RTX 5090</category>
            <category>video generation</category>
            <category>embeddings</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] GLM-5.3-Flash 320B on One DGX Spark: 15.4 tok/s with Vision Working]]></title>
            <link>https://ai-muninn.com/en/blog/glm53-flash-one-dgx-spark</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/glm53-flash-one-dgx-spark</guid>
            <pubDate>Sun, 30 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A 320B-A18B MoE at UD-Q2_K_XL fits in 121 GB. 15.40 tok/s on English prose, 16.23 on Chinese, 23.64 on Three.js code. Vision works only if you drop --spec-type.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>GLM-5.3</category>
            <category>llama.cpp</category>
            <category>multimodal</category>
            <category>speculative decoding</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] A 177B MoE at Full 262K Context on a DGX Spark: 76.65 tok/s on a File Edit]]></title>
            <link>https://ai-muninn.com/en/blog/qwen4exp-dgx-spark-262k-no-tradeoff</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/qwen4exp-dgx-spark-262k-no-tradeoff</guid>
            <pubDate>Fri, 28 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Qwen3.8-Flash-Next UD-Q4_K_XL is 104 GiB. GB10 holds all of it plus a 262,144 context in one 121 GB pool. 76.65 tok/s on a file edit, 20.51 on prose.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>llama.cpp</category>
            <category>MoE</category>
            <category>long context</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] A 177B MoE on Three Modded 2080 Tis: 23 tok/s at 128K, 77 on a File Edit]]></title>
            <link>https://ai-muninn.com/en/blog/qwen4exp-177b-three-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/qwen4exp-177b-three-2080ti</guid>
            <pubDate>Fri, 28 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Qwen3.8-Flash-Next (176.94B, arch qwen4exp) runs on 3× modded 2080 Ti 22G at 23.14 tok/s at 128K context — and 3.51× that on a file edit, with no draft model.]]></description>
            <category>llama.cpp</category>
            <category>2080 Ti</category>
            <category>MoE</category>
            <category>speculative decoding</category>
            <category>quantization</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Qwen3.8-27B hits 65 tok/s on one DGX Spark — 12.3 without speculative decoding]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-bandwidth-ceiling-85-percent</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-bandwidth-ceiling-85-percent</guid>
            <pubDate>Sun, 23 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[GB10 gives you 273 GB/s and the model reads 18.77 GB per token, so the ceiling is 14.54 tok/s. Measured 12.3 — 85% of it. Kernel swaps can't recover the rest. Speculative decoding gets 5.28x, and the win came from one flag.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>NVFP4</category>
            <category>Qwen3.8-27B</category>
            <category>SGLang</category>
            <category>speculative decoding</category>
            <category>DFlash2</category>
            <category>DSpark</category>
            <category>MTP</category>
            <category>memory bandwidth</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Fixing NCCL's Stub-Library Error Cut PCIe Traffic 99% and Barely Moved Speed]]></title>
            <link>https://ai-muninn.com/en/blog/nccl-2080ti-stub-library-fix</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/nccl-2080ti-stub-library-fix</guid>
            <pubDate>Sun, 23 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[NCCL died on two modded 2080 Tis with 'CUDA driver is a stub library'. Building v2.31.2 from source fixed it: PCIe traffic down 99%, code generation unchanged.]]></description>
            <category>NCCL</category>
            <category>llama.cpp</category>
            <category>tensor parallel</category>
            <category>2080 Ti</category>
            <category>CUDA</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Dual-GPU AllReduce Only Uses 8% of PCIe Gen3 x16 — It Moves Little, Very Often]]></title>
            <link>https://ai-muninn.com/en/blog/pcie-allreduce-dual-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/pcie-allreduce-dual-2080ti</guid>
            <pubDate>Sun, 23 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Measured with nvidia-smi dmon on two modded 2080 Tis: AllReduce costs 67 MB/s at decode and 973 MB/s at prefill, and the bus peaks at 8.2% of Gen3 x16.]]></description>
            <category>llama.cpp</category>
            <category>tensor parallel</category>
            <category>PCIe</category>
            <category>NVLink</category>
            <category>2080 Ti</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] MiniMax-H3 1080p Isn't Unsupported, It's Three and a Half Hours]]></title>
            <link>https://ai-muninn.com/en/blog/minimax-h3-rtx5090-character-lock</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/minimax-h3-rtx5090-character-lock</guid>
            <pubDate>Sun, 23 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Three mistakes shooting a wuxia scene in MiniMax-H3: a 502 that wasn't a rejection, blur upscaling can't fix, and a prompt that described a face instead of naming one.]]></description>
            <category>MiniMax-H3</category>
            <category>RTX 5090</category>
            <category>z-image</category>
            <category>character consistency</category>
            <category>ComfyUI</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Two Modded 2080 Tis Reach 59.6 tok/s on Qwen3.8-27B With llama.cpp Tensor Parallel]]></title>
            <link>https://ai-muninn.com/en/blog/qwen38-dual-2080ti-tensor-parallel</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/qwen38-dual-2080ti-tensor-parallel</guid>
            <pubDate>Sat, 22 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Two modded 22GB RTX 2080 Tis hit 59.6 tok/s on Qwen3.8-27B using llama.cpp's -sm tensor split mode plus MTP, after -sm row was deleted upstream. Config, gotchas and the failed routes included.]]></description>
            <category>llama.cpp</category>
            <category>tensor parallel</category>
            <category>2080 Ti</category>
            <category>Qwen3.8</category>
            <category>MTP</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] From 41 Minutes to 73 Seconds: Why My Coding Agent's Small Tickets Were So Slow]]></title>
            <link>https://ai-muninn.com/en/blog/agent-ticket-41min-to-73s</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/agent-ticket-41min-to-73s</guid>
            <pubDate>Fri, 21 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Dissecting 41 Codex sessions turned up a wall-clock formula (tool calls x 14.3s) and cut a stuck ticket from 41 minutes down to 73 seconds.]]></description>
            <category>AI Agent</category>
            <category>Claude Code</category>
            <category>Codex</category>
            <category>agent orchestration</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Free 15% Speedup on a 2080 Ti: One Broken Chat Template, One Starving MTP Head]]></title>
            <link>https://ai-muninn.com/en/blog/chat-template-kv-mtp-free-speedup</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/chat-template-kv-mtp-free-speedup</guid>
            <pubDate>Fri, 21 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A frozen GGUF chat template was killing my KV cache; a community fix plus MTP draft depth 4 took Qwen3.8-27B from 41.6 to 47.6 tok/s on a 2080 Ti 22GB.]]></description>
            <category>llama.cpp</category>
            <category>chat template</category>
            <category>KV cache</category>
            <category>MTP</category>
            <category>speculative decoding</category>
            <category>2080 Ti</category>
            <category>Qwen3.8</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] Why Isn't Your 4-Bit Quant Faster on a 2080 Ti? I Tore Open the CUDA Backend to Find Out]]></title>
            <link>https://ai-muninn.com/en/blog/why-4bit-isnt-faster-on-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/why-4bit-isnt-faster-on-2080ti</guid>
            <pubDate>Thu, 20 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Quantized weights are half the file size on a 2080 Ti but the speed doesn't move. Dumping the CUDA backend's .so with nm -D shows why, and which quant format is actually accelerated.]]></description>
            <category>2080 Ti</category>
            <category>Turing</category>
            <category>quantization</category>
            <category>GGUF</category>
            <category>AWQ</category>
            <category>Marlin</category>
            <category>INT8</category>
            <category>W4A8</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[Junk-Tier Big Models #5] Qwen3.8-27B on a 2018 22GB card: 30 tok/s, and it one-shot a 3D scene]]></title>
            <link>https://ai-muninn.com/en/blog/qwen38-27b-ud-on-one-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/qwen38-27b-ud-on-one-2080ti</guid>
            <pubDate>Tue, 18 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A modded 22GB RTX 2080 Ti runs Qwen3.8-27B at 30-31 tok/s with 128K context. The full llama.cpp command, the MTP flags, and the thinking-budget trap.]]></description>
            <category>Qwen3.8</category>
            <category>unsloth</category>
            <category>2080 Ti</category>
            <category>llama.cpp</category>
            <category>GGUF</category>
            <category>local LLM</category>
            <category>MTP</category>
        </item>
        <item>
            <title><![CDATA[[Announcement] On hiatus until August 21]]></title>
            <link>https://ai-muninn.com/en/blog/back-after-acute-cholecystitis</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/back-after-acute-cholecystitis</guid>
            <pubDate>Sat, 15 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[The open models are landing thick and fast right now, and my gallbladder picked exactly this moment to give out. Nothing new here for a couple of weeks.]]></description>
            <category>Announcement</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Running MiniMax-H3 on a DGX Spark — and why NVIDIA VSR is off the table for now]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-minimax-h3-span-upscaler</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-minimax-h3-span-upscaler</guid>
            <pubDate>Fri, 07 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Fifteen seconds of 1080p video with audio in 741s on a GB10 DGX Spark. Swapping Real-ESRGAN for SPAN saved 282s, and nvidia-vfx ships x86_64 wheels only — nothing for ARM.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>MiniMax-H3</category>
            <category>ComfyUI</category>
            <category>SPAN</category>
            <category>aarch64</category>
            <category>SageAttention</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] Wait, a 2080 Ti Can Run MiniMax-H3? 1080p With Audio on a 2018 Card]]></title>
            <link>https://ai-muninn.com/en/blog/minimax-h3-on-modded-2080ti-22gb</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/minimax-h3-on-modded-2080ti-22gb</guid>
            <pubDate>Thu, 06 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[The four files are 38 GiB on disk; the card has 22. A modded 2080 Ti 22G still renders 15s of 1080p with audio in 23 minutes. Full config, measured speed and quality, then how it got there.]]></description>
            <category>2080 Ti</category>
            <category>Turing</category>
            <category>MiniMax-H3</category>
            <category>ComfyUI</category>
            <category>SageAttention</category>
            <category>quantization</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Twice as Fast: MiniMax-H3 on an RTX 5090, 625s Down to 314s]]></title>
            <link>https://ai-muninn.com/en/blog/minimax-h3-rtx5090-speedup-vsr</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/minimax-h3-rtx5090-speedup-vsr</guid>
            <pubDate>Thu, 06 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Three stacked changes took a 15-second 1080p MiniMax-H3 render from 625s to 314s on one RTX 5090: 14 steps, SageAttention 2.2.0, and RTX VSR replacing Real-ESRGAN.]]></description>
            <category>MiniMax-H3</category>
            <category>RTX 5090</category>
            <category>ComfyUI</category>
            <category>RTX Video Super Resolution</category>
            <category>SageAttention</category>
            <category>Real-ESRGAN</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Running MiniMax-H3 on one RTX 5090: four files, 31.7 GB, 175s per talking clip]]></title>
            <link>https://ai-muninn.com/en/blog/minimax-h3-nvfp4-rtx5090</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/minimax-h3-nvfp4-rtx5090</guid>
            <pubDate>Tue, 04 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Beginner walkthrough for MiniMax-H3, the 33B model that generates video and stereo audio in one pass. Full precision is 115 GB; quantized it fits one RTX 5090. What to download, where it goes, how to prompt it.]]></description>
            <category>MiniMax-H3</category>
            <category>NVFP4</category>
            <category>RTX 5090</category>
            <category>ComfyUI</category>
            <category>text-to-video</category>
            <category>quantization</category>
        </item>
        <item>
            <title><![CDATA[[DeepSeek-V4-Flash] A frontier-class open model on hardware you own: running DeepSeek-V4-Flash-0731 on a DGX Spark]]></title>
            <link>https://ai-muninn.com/en/blog/ds4-0731-on-dgx-spark-howto</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ds4-0731-on-dgx-spark-howto</guid>
            <pubDate>Sun, 02 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[DeepSeek-V4-Flash-0731 on a DGX Spark: 17.5-17.9 tok/s at 256K context. Unified memory removes the offload decision entirely — and it is only 7% faster than a used 2080 Ti.]]></description>
            <category>DeepSeek-V4-Flash</category>
            <category>0731</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>llama.cpp</category>
            <category>CUDA</category>
            <category>unified memory</category>
            <category>local LLM</category>
            <category>tutorial</category>
        </item>
        <item>
            <title><![CDATA[[Junk-Tier Big Models #4] A frontier-class open model on hardware you already own: DeepSeek-V4-Flash-0731 on one 22GB 2080 Ti]]></title>
            <link>https://ai-muninn.com/en/blog/ds4-0731-on-one-2080ti-howto</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ds4-0731-on-one-2080ti-howto</guid>
            <pubDate>Sun, 02 Aug 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[How to run DeepSeek-V4-Flash-0731 (91GB, 284B MoE) on a single modded 22GB 2080 Ti at 16.5 tok/s and 1M context — including a formula for picking -ncmoe.]]></description>
            <category>DeepSeek-V4-Flash</category>
            <category>0731</category>
            <category>MoE</category>
            <category>2080 Ti</category>
            <category>llama.cpp</category>
            <category>expert offload</category>
            <category>local LLM</category>
            <category>tutorial</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] My Skill Had the Check Written Perfectly — It Just Never Ran]]></title>
            <link>https://ai-muninn.com/en/blog/ai-tic-check-that-never-ran</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-tic-check-that-never-ran</guid>
            <pubDate>Wed, 29 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A draft cleared fact-check and two rounds of native-speaker review, then one reader caught it in a sentence. The check that should have caught it was labelled manual, so it had never run. Here's the script that replaced it, and how I calibrated the thresholds.]]></description>
            <category>AI writing</category>
            <category>de-AI</category>
            <category>writing workflow</category>
            <category>Claude Code</category>
            <category>editing</category>
            <category>quality gates</category>
            <category>LLM</category>
        </item>
        <item>
            <title><![CDATA[[Junk-Tier Big Models #3] A 284B MoE on ONE 2080 Ti — and it beats my DGX Spark]]></title>
            <link>https://ai-muninn.com/en/blog/deepseek-v4-flash-284b-on-one-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/deepseek-v4-flash-284b-on-one-2080ti</guid>
            <pubDate>Wed, 29 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[DeepSeek-V4-Flash is 284B. It decodes at 17.4 tok/s on a single modded 22GB 2080 Ti — faster than the DGX Spark I serve it on. Card count barely matters, and MTP speculative decoding dies on unimplemented runtime, not a missing GPU.]]></description>
            <category>DeepSeek-V4-Flash</category>
            <category>MoE</category>
            <category>2080 Ti</category>
            <category>llama.cpp</category>
            <category>expert offload</category>
            <category>MTP</category>
            <category>speculative decoding</category>
            <category>DGX Spark</category>
            <category>local LLM</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] Agent Memory Self-Poisoning: When an AI Agent Trusts Its Own Wrong Answers]]></title>
            <link>https://ai-muninn.com/en/blog/agent-memory-self-poisoning</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/agent-memory-self-poisoning</guid>
            <pubDate>Mon, 27 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[My AI agent's durable memory auto-saved a wrong answer, then cited it back as fact — outranking the corrected truth. The three-part pathology, and why the fix is ranking memory, not adding more of it.]]></description>
            <category>AI agent</category>
            <category>agent memory</category>
            <category>Hermes</category>
            <category>RAG</category>
            <category>LLM</category>
        </item>
        <item>
            <title><![CDATA[[DeepSeek-V4-Flash] The Engine Upgrade That OOM'd My DGX Spark — and Why I Rolled It Back]]></title>
            <link>https://ai-muninn.com/en/blog/hikari-ds4-oom-rollback</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hikari-ds4-oom-rollback</guid>
            <pubDate>Sat, 25 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Part 9's Entrpi engine hit 20 tok/s. Days later it OOM-crashed under real agent traffic. Root cause: two copies of the weights in memory. I rolled back.]]></description>
            <category>DeepSeek-V4-Flash</category>
            <category>DGX Spark</category>
            <category>ds4</category>
            <category>OOM</category>
            <category>Entrpi</category>
            <category>KV cache</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[LLM Deep Dive] Surgical GGUF Quantization: Quantize Only the Tensors You Choose]]></title>
            <link>https://ai-muninn.com/en/blog/layer-aware-gguf-quantization</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/layer-aware-gguf-quantization</guid>
            <pubDate>Thu, 23 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A GGUF isn't uniform precision. Inspect per-tensor types, then quantize only the tensors you choose with llama-quantize — pin every other family back to its own type so it copies through untouched, and never stack requant error.]]></description>
            <category>quantization</category>
            <category>GGUF</category>
            <category>llama.cpp</category>
            <category>llama-quantize</category>
            <category>tensor-type</category>
            <category>mixed precision</category>
            <category>Laguna</category>
            <category>LLM Deep Dive</category>
        </item>
        <item>
            <title><![CDATA[[LLM Deep Dive] The Best Free Open-Source Model for a Single 24GB GPU? My Pick Is ThinkingCap-Qwen3.6-27B]]></title>
            <link>https://ai-muninn.com/en/blog/thinkingcap-qwen36-27b-local-brain</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/thinkingcap-qwen36-27b-local-brain</guid>
            <pubDate>Wed, 22 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Got one 24GB consumer GPU (or a modded 2080 Ti 22G)? My current top free open-source pick is Huihui-ThinkingCap-Qwen3.6-27B-abliterated: Q4_K_S ~16GB, half the thinking tokens, ~38 tok/s with MTP, almost never refuses, all Apache-2.0.]]></description>
            <category>Qwen3.6</category>
            <category>ThinkingCap</category>
            <category>abliterated</category>
            <category>local LLM</category>
            <category>llama.cpp</category>
            <category>2080 Ti</category>
            <category>efficient thinking</category>
            <category>MTP</category>
            <category>reasoning model</category>
        </item>
        <item>
            <title><![CDATA[[Junk-Tier Big Models #2] Running Poolside Laguna S 2.1, a 118B Coding MoE, on ONE 22GB 2080 Ti]]></title>
            <link>https://ai-muninn.com/en/blog/laguna-118b-moe-on-one-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/laguna-118b-moe-on-one-2080ti</guid>
            <pubDate>Wed, 22 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Poolside Laguna S 2.1, a 118B-A8B coding MoE, on one 22GB 2080 Ti via CPU/GPU hybrid offload plus a companion DFlash speculative-decoding draft at ~29 tok/s; attention-Q8 saves ~2.45 GiB, +7% decode.]]></description>
            <category>Laguna</category>
            <category>Poolside</category>
            <category>MoE</category>
            <category>2080 Ti</category>
            <category>llama.cpp</category>
            <category>DFlash</category>
            <category>speculative decoding</category>
            <category>local LLM</category>
            <category>quantization</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #16] Hermes config health check: 5 silent gotchas that make your assistant act weird]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-config-gotchas</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-config-gotchas</guid>
            <pubDate>Tue, 21 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Part 11 said a haywire assistant is usually a broken car (tools, config, memory), not a dumb engine (the model). This is that checklist: a context_length set at the wrong level silently compresses early, Qwen thinking left on runs 10x slower, an MCP tool that connects but every call fails, and a sib running a different model than you think. Five real config gotchas, each with a check you can hand to your agent to run on itself, plus the fix.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>config</category>
            <category>debugging</category>
            <category>self-hosted</category>
        </item>
        <item>
            <title><![CDATA[[DeepSeek-V4-Flash] Swapping the ds4 Engine for a Free Half-Generation Speedup on One DGX Spark]]></title>
            <link>https://ai-muninn.com/en/blog/ds4-entrpi-engine-swap</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ds4-entrpi-engine-swap</guid>
            <pubDate>Tue, 21 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Asked Codex if my ds4 repo had a single-DGX-Spark optimization; found Entrpi. Swapped the engine, not the model — decode 14-16 to about 20 tok/s, prefill about 2×.]]></description>
            <category>DeepSeek-V4-Flash</category>
            <category>DGX Spark</category>
            <category>ds4</category>
            <category>speculative decoding</category>
            <category>KV cache</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[Junk-Tier Big Models #1] Running a 119B MoE at 74 tok/s on Three 2080 Tis]]></title>
            <link>https://ai-muninn.com/en/blog/run-119b-moe-on-3x-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/run-119b-moe-on-3x-2080ti</guid>
            <pubDate>Tue, 21 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A US$1.6k junk EPYC + 3× 2080 Ti 22G box (66G VRAM) runs a quantized 119B MoE at 74 tok/s. Expert offload to RAM costs about 2.4× decode — plus a multi-card OOM gotcha.]]></description>
            <category>llama.cpp</category>
            <category>MoE</category>
            <category>expert offload</category>
            <category>2080 Ti</category>
            <category>EPYC</category>
            <category>quantization</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] Two Identical 2080 Tis, One 3.4× Slower — the Culprit Was a Single dtype Log Line]]></title>
            <link>https://ai-muninn.com/en/blog/2080ti-turing-fp32-fallback</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/2080ti-turing-fp32-fallback</guid>
            <pubDate>Sat, 18 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[The same modded 2080 Ti 22G ran Z-Image 3.4× slower on one machine than another. Not the hardware, not the OS, not a missing package — a Turing bf16→fp32 fallback hiding in one log line. One flag fixed it.]]></description>
            <category>2080 Ti</category>
            <category>Turing</category>
            <category>ComfyUI</category>
            <category>Z-Image</category>
            <category>quantization</category>
            <category>GPU</category>
            <category>benchmark</category>
            <category>dev workflow</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] Your AI Agent's Skills Are a Context Budget: Cutting 193 to 7]]></title>
            <link>https://ai-muninn.com/en/blog/slimming-the-agent-skill-budget</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/slimming-the-agent-skill-budget</guid>
            <pubDate>Fri, 17 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[One of my AI agents was auto-loading 193 skills into a 2% context budget, silently truncating every description. The fix was visibility governance, not deletion — an allowlist, thin-shell skills, and three layers that stop it re-bloating.]]></description>
            <category>Dev Workflow</category>
            <category>AI agent</category>
            <category>Claude Code</category>
            <category>Codex</category>
            <category>skills</category>
            <category>context engineering</category>
            <category>governance</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] The Two Axes That Let a Fleet of AIs Collaborate Without Re-Explaining]]></title>
            <link>https://ai-muninn.com/en/blog/how-ai-agents-collaborate-without-re-explaining</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/how-ai-agents-collaborate-without-re-explaining</guid>
            <pubDate>Thu, 16 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Six posts in, my AI setup is really two axes of one system: durable knowledge and live task state, both in plain files. Here's how they converge so different AIs hand off work without re-explaining it or losing it.]]></description>
            <category>Dev Workflow</category>
            <category>AI agent</category>
            <category>multi-agent</category>
            <category>handoff</category>
            <category>knowledge base</category>
            <category>Claude Code</category>
            <category>Codex</category>
            <category>governance</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] Why an AI Agent's Memory Needs a Distilled Layer Above Search]]></title>
            <link>https://ai-muninn.com/en/blog/why-agent-memory-needs-a-distilled-layer</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/why-agent-memory-needs-a-distilled-layer</guid>
            <pubDate>Wed, 15 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Search finds an AI agent's notes but hands back raw material to re-derive each session. I distill ~600 files into canonical claims — the goal is ending re-explanation, not enforcing agreement.]]></description>
            <category>Dev Workflow</category>
            <category>AI agent</category>
            <category>memory</category>
            <category>knowledge base</category>
            <category>distillation</category>
            <category>RAG</category>
            <category>Claude Code</category>
            <category>governance</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] From Markdown Search to a Knowledge Graph: How My AI's Memory Grew a Second Layer]]></title>
            <link>https://ai-muninn.com/en/blog/from-markdown-search-to-knowledge-graph</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/from-markdown-search-to-knowledge-graph</guid>
            <pubDate>Tue, 14 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[My AI's long-term memory is ~600 markdown files in three layers: the files are the source of truth, a search engine makes them findable, and a knowledge graph links them by concept. Here's the design, why each layer exists, and the wrong turns I took building it.]]></description>
            <category>Dev Workflow</category>
            <category>knowledge graph</category>
            <category>AI agent</category>
            <category>memory</category>
            <category>qmd</category>
            <category>musubi</category>
            <category>Claude Code</category>
            <category>RAG</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] Retiring an AI Agent's Memory System: Three Traps When Doctrine Drifts From Runtime]]></title>
            <link>https://ai-muninn.com/en/blog/retiring-ai-agent-memory-system</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/retiring-ai-agent-memory-system</guid>
            <pubDate>Sun, 12 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I set out to retire a 'dead' shared-memory folder my knowledge base had replaced. It was writing to itself at 2:30 AM. Three traps in doctrine-vs-runtime drift, and why you map a memory system's topology from the hub, not the spoke.]]></description>
            <category>AI agent</category>
            <category>knowledge base</category>
            <category>memory</category>
            <category>dev workflow</category>
            <category>Syncthing</category>
            <category>governance</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] Delegating to an AI Coding Agent: The Unit of Work Is a Ticket File, Not a Conversation]]></title>
            <link>https://ai-muninn.com/en/blog/dispatch-tickets-not-conversations</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dispatch-tickets-not-conversations</guid>
            <pubDate>Sun, 12 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[How I hand engineering tasks to Codex: self-contained ticket files with machine-checkable acceptance, background execution, three flags I paid for in dead time, and a report-is-not-a-result gate.]]></description>
            <category>Codex</category>
            <category>Claude Code</category>
            <category>AI agent</category>
            <category>dev workflow</category>
            <category>delegation</category>
            <category>orchestration</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] When Your Quota Runs Out Mid-Task: A Live-State Handoff Protocol for Claude Code and Codex]]></title>
            <link>https://ai-muninn.com/en/blog/claude-code-codex-handoff-protocol</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/claude-code-codex-handoff-protocol</guid>
            <pubDate>Sat, 11 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[When an AI coding CLI runs out of quota mid-task, what dies isn't your knowledge base — it's the live state trapped in the session. A file-based handoff protocol lets Claude Code and Codex resume each other's half-finished work.]]></description>
            <category>Claude Code</category>
            <category>Codex</category>
            <category>AI agent</category>
            <category>dev workflow</category>
            <category>handoff</category>
            <category>hooks</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] 0xc0000409: When My AI Service Died Silently and the Log Ate the Evidence]]></title>
            <link>https://ai-muninn.com/en/blog/0xc0000409-crash-detective-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/0xc0000409-crash-detective-2080ti</guid>
            <pubDate>Fri, 10 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A local brain on a headless Windows box kept dying under real load: the client caught a brief 503, the service quietly restarted itself, and the application log was blank. The worst part wasn't the crash — it was that my own log truncated the one line that explained it on restart. This is the hunt to dig that reason back out from under its own log: why a self-restarting service is the best at burying the reason for its own crash, why you go to the OS-layer Event Log first, and what 0xc0000409 actually means. Full disclosure: I haven't 100% pinned the root cause — this is an open investigation, not a closed case.]]></description>
            <category>local LLM</category>
            <category>llama.cpp</category>
            <category>Windows</category>
            <category>crash forensics</category>
            <category>observability</category>
            <category>0xc0000409</category>
        </item>
        <item>
            <title><![CDATA[[Local LLM] How to tell if a hyped LLM optimization is real on your hardware: read the source, find the ceiling, run one experiment]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-evaluating-optimization-claims</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-evaluating-optimization-claims</guid>
            <pubDate>Fri, 10 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Most hyped LLM optimizations don't survive contact with your own model and hardware. Three cheap checks — read the source, find the true ceiling, run one discriminating experiment — that tell you which are real, anchored in the FlashMemory investigation on DeepSeek-V4-Flash.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DeepSeek-V4-Flash</category>
            <category>methodology</category>
            <category>local LLM</category>
            <category>verification</category>
        </item>
        <item>
            <title><![CDATA[Hermes Agent: The Complete Self-Hosted Guide — Desktop Install to a Local-Model Fleet]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-agent-complete-guide</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-agent-complete-guide</guid>
            <pubDate>Fri, 10 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Hermes agent, self-hosted from a desktop install to a local-model fleet: ChatGPT OAuth (no API key), Telegram/LINE control, local models, on Mac and Windows.]]></description>
            <category>Hermes</category>
            <category>AI agent</category>
            <category>self-hosted</category>
            <category>ChatGPT OAuth</category>
            <category>Telegram</category>
            <category>local LLM</category>
            <category>MCP</category>
        </item>
        <item>
            <title><![CDATA[[Local LLM] FlashMemory can't improve DeepSeek-V4-Flash's own lightning indexer — I retrained it on my exact Q2 and it still lost]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-flashmemory-vs-native-indexer</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-flashmemory-vs-native-indexer</guid>
            <pubDate>Thu, 09 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[V4-Flash already ships a native lightning indexer tracking true attention at 93–96%. FlashMemory pre-filters candidate chunks, but it makes near-random selections on my Q2 and reaches only 89–92% when retrained — still a NO-GO on GB10.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DeepSeek-V4-Flash</category>
            <category>KV cache</category>
            <category>sparse attention</category>
            <category>ds4</category>
            <category>local LLM</category>
        </item>
        <item>
            <title><![CDATA[DGX Spark in 2026: What Still Works, What Broke, and What I'd Run Today]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-2026-current-guide</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-2026-current-guide</guid>
            <pubDate>Tue, 07 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A current 2026 guide to running local AI on DGX Spark: vLLM, official Gemma 4 NVFP4 weights, MTP, long-context multimodal options, and the traps still worth avoiding.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>Gemma 4</category>
            <category>vLLM</category>
            <category>NVFP4</category>
            <category>MTP</category>
            <category>Ollama</category>
            <category>local LLM</category>
            <category>2026 guide</category>
        </item>
        <item>
            <title><![CDATA[[Local LLM] Why a 284B fits a 128GB GB10 at long context: DeepSeek-V4-Flash attacks the KV cache, not the parameter count]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-architecture</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-architecture</guid>
            <pubDate>Tue, 07 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[DeepSeek-V4-Flash is a 284B MoE that stays fast at long context on a 128GB GB10 because its hybrid CSA/HCA attention and lightning indexer shrink the KV cache to ~871MiB at 64K and read only a few hundred compressed rows per step. What I found reading ds4's code and DeepSeek's V4 paper.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DeepSeek-V4-Flash</category>
            <category>KV cache</category>
            <category>sparse attention</category>
            <category>ds4</category>
            <category>local LLM</category>
        </item>
        <item>
            <title><![CDATA[How to Run an AI Agent from Your Own Desktop: ChatGPT OAuth, Telegram, LINE, and Local Models]]></title>
            <link>https://ai-muninn.com/en/blog/run-ai-agent-from-your-desktop</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/run-ai-agent-from-your-desktop</guid>
            <pubDate>Tue, 07 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A map for running a self-hosted AI agent: desktop body, ChatGPT OAuth brain, Telegram or LINE channels, tools, and local LLMs when you want autonomy.]]></description>
            <category>AI agent</category>
            <category>self-hosted</category>
            <category>ChatGPT OAuth</category>
            <category>openclaw</category>
            <category>Hermes</category>
            <category>Telegram</category>
            <category>LINE</category>
            <category>local LLM</category>
        </item>
        <item>
            <title><![CDATA[[Local LLM] Depth-1 MTP on V4-Flash: +9% on agent turns, −4% on prose — route speculative decode by workload]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-mtp-workload-asymmetry</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-mtp-workload-asymmetry</guid>
            <pubDate>Mon, 06 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Depth-1 MTP speculative decode on DeepSeek-V4-Flash is lossless but workload-dependent on a GB10: +9.4% on agent turns and +6.5% on code, −3.6% on prose and −3.7% on Chinese chat. The sign follows the acceptance rate because decode here is verify-bound (108ms verify vs 4ms draft). It's not a global faster switch — route it by workload.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DeepSeek-V4-Flash</category>
            <category>speculative decoding</category>
            <category>MTP</category>
            <category>ds4</category>
            <category>local LLM</category>
        </item>
        <item>
            <title><![CDATA[[Local LLM] My 284B agent quietly stopped reusing its KV cache — the ds4 evict storm that re-paid prefill every turn]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-kv-evict-storm</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-kv-evict-storm</guid>
            <pubDate>Sun, 05 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A month after wiring DeepSeek-V4-Flash into a daily agent, it felt slow again. Two log lines explained it: the disk-KV cache was evicting live prefixes (hits=0), and tool-call turns never saved a checkpoint — so common=268 out of 14209 and every turn re-paid full prefill. The fix: 256G KV budget + PR #489.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DeepSeek-V4-Flash</category>
            <category>KV cache</category>
            <category>ds4</category>
            <category>local LLM</category>
            <category>prefill</category>
            <category>agent</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] I Doubled My Agent's Decode Speed and It Got Slower: TTFT Is the Number You Actually Feel]]></title>
            <link>https://ai-muninn.com/en/blog/why-30-toks-feels-slower-than-14-ttft</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/why-30-toks-feels-slower-than-14-ttft</guid>
            <pubDate>Fri, 03 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I swapped my home agent's brain for one that decodes 30-40 tok/s instead of 14, and it felt slower. The number I'd stared at for a year — tok/s — only measures how fast tokens come out, not how long before they start. On a hybrid model, a single cache miss re-prefills the entire prompt: same box, same brain, 2.6s warm vs 216s cold. Here's the live log.]]></description>
            <category>local LLM</category>
            <category>AI agent</category>
            <category>TTFT</category>
            <category>llama.cpp</category>
            <category>Qwen3</category>
            <category>KV cache</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] Progressive Streaming on a Slow Model Got My Bot Rate-Limited by Telegram]]></title>
            <link>https://ai-muninn.com/en/blog/telegram-streaming-flood-control-slow-llm</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/telegram-streaming-flood-control-slow-llm</guid>
            <pubDate>Thu, 02 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[To ease the wait on a pokey local agent, I turned on Telegram streaming — which, the way this bot did it, means rewriting the same message every fraction of a second. On a 14 tok/s brain, a single 175-second reply works out to an estimated couple hundred edit requests, which slammed into Telegram's flood control and got the whole bot benched for four minutes — final answer included. The short, ugly lesson: slow models should not fake streaming with edits. Send the finished answer once. Live logs inside.]]></description>
            <category>local LLM</category>
            <category>AI agent</category>
            <category>Telegram</category>
            <category>streaming</category>
            <category>rate limit</category>
            <category>llama.cpp</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #15] Hermes /learn: I had a local 27B write its own reusable skill]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-learn-self-authored-skills</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-learn-self-authored-skills</guid>
            <pubDate>Wed, 01 Jul 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Hermes has a /learn command that turns 'something you just did' into a reusable skill — a SKILL.md. I wired it into my own fleet: one Kanban card, a local 27B running on a modded 2080 Ti, and about 3 minutes later it handed back a clean, spec-compliant skill — plus two implementation details the docs don't spell out (slash command vs. dispatch, and where skills actually live). A plain-language walkthrough of what /learn does, how to use it, and where its limits are.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>skills</category>
            <category>/learn</category>
            <category>local model</category>
            <category>Kanban</category>
            <category>advanced</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #14] One spec, three assistants, three Tetris games: a Hermes Kanban dispatch test]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-project-three-sibs-tetris</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-project-three-sibs-tetris</guid>
            <pubDate>Mon, 29 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[After raising a fleet of assistants, I gave them the same one-line 'make a Tetris game' spec — no details at all — one card each, and let them each write a web Tetris in a single shot. I touched zero lines of game code; I only published the result. The surprise: from that one line, the Hermes harness plus a local model I tuned myself (on a modded 2080 Ti) filled in things I never asked for — a ghost piece and wall-kick — in one shot. You can play all three.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>dispatch</category>
            <category>Kanban</category>
            <category>local model</category>
            <category>Tetris</category>
        </item>
        <item>
            <title><![CDATA[[Troubleshooting] HuggingFace download stuck at 0 bytes on Windows — Xet, Python 3.13, ai-toolkit]]></title>
            <link>https://ai-muninn.com/en/blog/huggingface-download-stuck-zero-bytes-windows-ai-toolkit</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/huggingface-download-stuck-zero-bytes-windows-ai-toolkit</guid>
            <pubDate>Mon, 29 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Training with ai-toolkit on Windows + RTX 5090 hit three walls before it even started: Python 3.13 dependency hell, a HuggingFace download frozen at 0 bytes, and ssh killing the process. Each one's error pointed the wrong way — diagnosis and fix for all three.]]></description>
            <category>Troubleshooting</category>
            <category>HuggingFace</category>
            <category>Windows</category>
            <category>ai-toolkit</category>
            <category>RTX 5090</category>
        </item>
        <item>
            <title><![CDATA[[LoRA] The character-LoRA control panel: dialing in style, realism, and identity]]></title>
            <link>https://ai-muninn.com/en/blog/character-lora-control-panel-wan22</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/character-lora-control-panel-wan22</guid>
            <pubDate>Sun, 28 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Once your character LoRA is trained, how do you control it? Why lightning flattens style, when to spend full steps, how to stack a style LoRA, and why the trigger word alone won't hold the look.]]></description>
            <category>LoRA</category>
            <category>Wan 2.2</category>
            <category>ComfyUI</category>
            <category>AI character</category>
            <category>local generation</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] The Tool-Definition Tax: 17K Tokens Before I Say a Word, Re-Billed on Every Cache Miss]]></title>
            <link>https://ai-muninn.com/en/blog/tool-definition-tax-17k-context-economics</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/tool-definition-tax-17k-context-economics</guid>
            <pubDate>Sat, 27 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I added up what my home agent pays before it reads a single word from me: ~23K tokens of overhead, and 17K of that is just the instruction manuals for its tools. Worse, it runs a hybrid model — on a cache miss it re-processes all 17K from scratch, and a single user turn can do that a dozen-plus times. This is context economics, badly underestimated. The fix isn't cutting tools; it's loading them on demand, the way skills already do.]]></description>
            <category>local LLM</category>
            <category>AI agent</category>
            <category>context</category>
            <category>llama.cpp</category>
            <category>Qwen3</category>
            <category>tool calling</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] llama.cpp won't persist KV cache to disk — so I put a 60-line proxy in front of it (7× faster restore)]]></title>
            <link>https://ai-muninn.com/en/blog/kv-cache-disk-restore-7x</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/kv-cache-disk-restore-7x</guid>
            <pubDate>Fri, 26 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[On a long conversation, every message makes the model re-read the whole thing (re-prefill) before it answers — worst right after a restart or a cache eviction. Stock llama.cpp can save the KV cache to disk (--slot-save-path) but won't do it on its own — the auto-persist feature request is closed as not planned. A tiny stdlib reverse-proxy restores instead of re-prefilling: 9.9s → 1.4s on a 5K chat (7×). Mechanism, proxy design, and why I haven't shipped it yet.]]></description>
            <category>local LLM</category>
            <category>llama.cpp</category>
            <category>KV cache</category>
            <category>TTFT</category>
            <category>Qwen3</category>
            <category>prefill</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] Quantizing the Draft Cache Backfired — A Counterintuitive Look at Qwen MTP (f16 ran 34% faster than q4)]]></title>
            <link>https://ai-muninn.com/en/blog/mtp-quantized-draft-cache-backfires</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/mtp-quantized-draft-cache-backfires</guid>
            <pubDate>Thu, 25 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Quantizing the main KV cache to q4 to save memory is fine. So I quantized the MTP draft cache too — it's just a little draft, surely a free win. It wasn't: q4 draft cache ran 29.6 tok/s, the un-quantized f16 ran 39.7, and f16 used less VRAM on top of that. The draft cache is one of the few places where quantizing is a net loss — here's the triple penalty.]]></description>
            <category>MTP</category>
            <category>speculative decoding</category>
            <category>local LLM</category>
            <category>Qwen3</category>
            <category>llama.cpp</category>
            <category>KV cache</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #13] See what your fleet of AI agents is doing — from your phone: Muninn adds a Kanban board]]></title>
            <link>https://ai-muninn.com/en/blog/muninn-kanban-board</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/muninn-kanban-board</guid>
            <pubDate>Wed, 24 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Hermes has a built-in Kanban, but on your phone all you get is Telegram's plain text. Muninn now pulls that board onto the phone: Running / Blocked / Done columns — who's working on what, which card got blocked — at a glance. Zero backend, pure P2P.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>Muninn</category>
            <category>Kanban</category>
            <category>iOS</category>
            <category>phone</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] I Maxed Context to 256K, It Loaded Fine — Then Crashed in Real Use: A VRAM Detective Story on a 22GB Frankencard]]></title>
            <link>https://ai-muninn.com/en/blog/context-vs-vram-256k-oom-2080ti</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/context-vs-vram-256k-oom-2080ti</guid>
            <pubDate>Wed, 24 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[The model card says n_ctx_train=262144. The card has 22GB. The 27B's Q4 weights are only 15.7GB. The math looks obvious: max it to 256K, plenty to spare. -c 262144, launch — loads fine, no error. A few turns of real conversation later: 503, the service restarts itself. No tidy out-of-memory in the log, just a lone 0xc0000409. nvidia-smi: free VRAM down to ~170 MiB. Where did the gigabytes go? This is the hunt: I first blamed context checkpoints, but the llama.cpp source says they live in host RAM — the real VRAM eater is the KV cache; free-VRAM-vs-context is nonlinear, and the one stable sweet spot isn't 256K — it's 128K.]]></description>
            <category>local LLM</category>
            <category>llama.cpp</category>
            <category>Qwen3</category>
            <category>VRAM</category>
            <category>context window</category>
            <category>KV cache</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #12] Reach your home AI agent from anywhere: Muninn, a private iOS app over iroh P2P]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-muninn-phone-bridge</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-muninn-phone-bridge</guid>
            <pubDate>Tue, 23 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Hermes runs at home, but you lose it the second you walk out. Bridging through Telegram works but it's fiddly and routes every message through someone else's server. Muninn is an iOS app built for Hermes: give your agent one command, scan a QR, and your phone connects straight home over an encrypted iroh tunnel — no cloud in the path.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>iroh</category>
            <category>P2P</category>
            <category>iOS</category>
            <category>phone</category>
            <category>bridge</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] I Gave Up 100 tok/s for 30 — Fast Isn't the Same as Useful]]></title>
            <link>https://ai-muninn.com/en/blog/gemma-12b-vs-qwen-27b-agentic-discipline</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/gemma-12b-vs-qwen-27b-agentic-discipline</guid>
            <pubDate>Tue, 23 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Picking a local model, I looked at tok/s first too. Gemma 12B does 90-100 and it's great — until you put it on a kanban board, where it finishes the work and just walks away, never marking the card done. A Qwen 27B that's three times slower actually closes the loop. Why throughput is the wrong number for an agent — plus how grep almost lied to me about it.]]></description>
            <category>local LLM</category>
            <category>AI agent</category>
            <category>Qwen3</category>
            <category>Gemma</category>
            <category>llama.cpp</category>
            <category>kanban</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #11] Assistant gone haywire? Don't blame the engine — usually it's the car that broke, not the engine]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-harness-debugging</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-harness-debugging</guid>
            <pubDate>Mon, 22 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[When an AI assistant loops, wanders, freezes, or answers the wrong question, your first instinct is 'this model is dumb.' But from my own debugging, eight times out of ten it's not the model — it's the ring around it (tools, config, memory). The model is the engine; that ring is the car. A car that won't move usually doesn't have a broken engine — it has a flat tire or a clogged fuel line.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>debugging</category>
            <category>harness</category>
            <category>advanced</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun — Advanced] I Scored a 22GB-Modded 2080 Ti for ~$340 All-In — Just Enough to Keep a 27B Agent Running at Home]]></title>
            <link>https://ai-muninn.com/en/blog/modded-2080ti-22gb-local-agent</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/modded-2080ti-22gb-local-agent</guid>
            <pubDate>Mon, 22 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I dug up a 22GB-modded RTX 2080 Ti for ~$340 all-in (¥2079 sticker + shipping) — just enough to keep a resident 27B agent brain running on the same cheap old desktop. What the mod changes, and the gotchas.]]></description>
            <category>RTX 2080 Ti</category>
            <category>GPU mod</category>
            <category>local LLM</category>
            <category>AI agent</category>
            <category>llama.cpp</category>
            <category>Qwen3</category>
        </item>
        <item>
            <title><![CDATA[Directional Steering on an Abliterated DeepSeek-V4 (DGX Spark): the same scalpel as abliteration, and why the second cut fights back]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-directional-steering</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-directional-steering</guid>
            <pubDate>Sun, 21 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[ds4 ships directional steering — a runtime activation edit that nudges the model along a chosen direction, and the math is literally abliteration with a continuous, signed scale. I got it running on GB10/CUDA (the tooling looks Metal-only, but the activation dump fires on CUDA too) and pulled a verbosity vector from our abliterated Q2 model. The dial works, but it ignores the textbook: the sweep is non-monotonic and positive scales collapse the output to a four-word fragment. Two cuts from the same scalpel, fighting each other.]]></description>
            <category>DeepSeek V4</category>
            <category>directional steering</category>
            <category>activation steering</category>
            <category>abliteration</category>
            <category>representation engineering</category>
            <category>ds4</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>CUDA</category>
            <category>control vectors</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #10] Installed it, now what? Give your assistant hands — connect your own tools]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-connect-your-tools</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-connect-your-tools</guid>
            <pubDate>Sat, 20 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Your assistant is installed, but right now it only talks — it's all mouth. This post gives it hands: connect tools so it actually checks your folders, runs your commands, calls services you wrote yourself. The key idea is MCP, the 'universal outlet' standard for tools — plug one in and the assistant can use it. All running on your side, connected to your own stuff.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>MCP</category>
            <category>tools</category>
            <category>automation</category>
            <category>getting started</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #9] Swap your assistant's brain for one on your own machine: from cloud ChatGPT to a local model]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-local-brain</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-local-brain</guid>
            <pubDate>Fri, 19 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[We used ChatGPT as the assistant's brain. This post does something bolder — swaps that brain from the cloud to a local model running on your own machine (e.g. ds4). The payoff is an autonomous brain: no cloud model provider, your conversations stay on your machine, no usage caps. The honest cost: local brains are usually slower (~10 tok/s on my ds4) and need a capable machine. Swap the brain, keep the body — Hermes doesn't change at all.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>local model</category>
            <category>privacy</category>
            <category>autonomy</category>
            <category>ds4</category>
        </item>
        <item>
            <title><![CDATA[[LoRA] Train your own AI character on an RTX 5090 — one image to a usable character]]></title>
            <link>https://ai-muninn.com/en/blog/train-character-lora-wan22-rtx5090</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/train-character-lora-wan22-rtx5090</guid>
            <pubDate>Thu, 18 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Train a Wan 2.2 character LoRA on your own RTX 5090 from a single reference image. Then generate the same person from text — new outfits, scenes, art styles, even video. No cloud, no bill.]]></description>
            <category>LoRA</category>
            <category>Wan 2.2</category>
            <category>RTX 5090</category>
            <category>AI character</category>
            <category>local generation</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #8] One person, a whole team of assistants: each with its own desk, brain, and memory]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-multiple-sibs</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-multiple-sibs</guid>
            <pubDate>Wed, 17 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Comfortable with one assistant and want a second and third? Hermes gives each one its own home (config, memory, personality), each able to run a different model and handle different tasks. Plain-language: why split them, how, and the three I actually run. Honest: most people only need one — this is for when you want to tinker.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>profiles</category>
            <category>advanced</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #7] Give your AI assistant eyes and ears: vision + voice for a text-only brain]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-agent-vision-voice</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-agent-vision-voice</guid>
            <pubDate>Tue, 16 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Your AI assistant only reads text? Give it eyes and ears — send a photo it understands, send a voice clip it understands. Not by swapping in a pricier model, but by bolting on a small vision model as a perception side-car. Hermes's built-in auxiliary.vision + faster-whisper, measured end to end.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>multimodal</category>
            <category>vision</category>
            <category>voice</category>
            <category>faster-whisper</category>
            <category>local model</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #6] Let your assistant run on its own: daily research that pings your Telegram]]></title>
            <link>https://ai-muninn.com/en/blog/hermes-autonomous-daily</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/hermes-autonomous-daily</guid>
            <pubDate>Tue, 16 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[The last and most satisfying step: set up a task that runs itself. Tell it in plain words, and every day it researches what you care about, sums it up, and messages your Telegram. Set it once, close the laptop, and it pings you the next morning.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>Telegram</category>
            <category>automation</category>
            <category>scheduling</category>
            <category>getting started</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #5] Use your AI assistant from your phone: connect Hermes to Telegram]]></title>
            <link>https://ai-muninn.com/en/blog/connect-hermes-to-telegram</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/connect-hermes-to-telegram</guid>
            <pubDate>Tue, 16 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Order your assistant around from your phone. Chat with one official Telegram bot, get a key (token), hand it to Hermes — done. No public URL, no webhook, no tunnel, because Hermes fetches messages from Telegram itself.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>Telegram</category>
            <category>getting started</category>
            <category>tutorial</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #4] How to install Hermes Agent Desktop: your first AI assistant, no terminal]]></title>
            <link>https://ai-muninn.com/en/blog/install-hermes-desktop</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/install-hermes-desktop</guid>
            <pubDate>Tue, 16 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Install the Hermes Agent desktop app — no terminal. Download it, let it auto-install dependencies, sign in with your ChatGPT account, and your first AI assistant is running in about 15 minutes.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>ChatGPT</category>
            <category>getting started</category>
            <category>tutorial</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #3] The fixed combo we'll use: ChatGPT as the brain, Hermes as the body]]></title>
            <link>https://ai-muninn.com/en/blog/ai-brain-hermes-body</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-brain-hermes-body</guid>
            <pubDate>Tue, 16 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[An AI assistant = a brain + a body. Use your ChatGPT account as the brain and Hermes as the body — one fixed combo, nothing to choose. Here's why it's set up this way, and what to have ready before you install.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>ChatGPT</category>
            <category>getting started</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #2] What is an agent framework? Why you shouldn't roll your own — just use one that exists]]></title>
            <link>https://ai-muninn.com/en/blog/what-is-an-agent-framework</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/what-is-an-agent-framework</guid>
            <pubDate>Tue, 16 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[You don't need to write code to have your own AI assistant. An agent framework already packages the hard parts so you just install and go. Here's why you shouldn't wire it yourself — and why this series uses Hermes.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>Hermes</category>
            <category>getting started</category>
        </item>
        <item>
            <title><![CDATA[[Agent 101 #1] AI assistant vs ChatGPT: one answers you, one uses your tools to get things done]]></title>
            <link>https://ai-muninn.com/en/blog/ai-agent-vs-chatbot</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-agent-vs-chatbot</guid>
            <pubDate>Tue, 16 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[You mostly use ChatGPT one question at a time. A self-hosted AI assistant (agent) finishes the job with your own tools, runs on your side, and plugs into the apps you use daily. Lesson one of building your own assistant from zero.]]></description>
            <category>AI assistant</category>
            <category>AI agent</category>
            <category>ChatGPT</category>
            <category>getting started</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun] A blog RAG support bot on a GTX 970: no torch, no vector DB, no LangChain]]></title>
            <link>https://ai-muninn.com/en/blog/gtx-970-blog-rag-bot</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/gtx-970-blog-rag-bot</guid>
            <pubDate>Sun, 14 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A retrieval-augmented support bot for my blog, running on a 2014 GTX 970 and a ~600MB embedding model. llama.cpp embeddings on CPU, numpy brute-force cosine over 3,475 chunks, an embedding-score guardrail, and Cloudflare Tunnel.]]></description>
            <category>Gemma 4</category>
            <category>GTX 970</category>
            <category>RAG</category>
            <category>llama.cpp</category>
            <category>Cloudflare Tunnel</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun] On a GTX 970, Flash Attention nearly doubles long-context decode (24.3 → 42.5 tok/s)]]></title>
            <link>https://ai-muninn.com/en/blog/gtx-970-gemma4-e2b-kv-cache-flash-attention</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/gtx-970-gemma4-e2b-kv-cache-flash-attention</guid>
            <pubDate>Sun, 14 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[On a tensor-core-less Maxwell GTX 970 running Gemma 4 E2B, Flash Attention nearly doubles long-context decode (24.3 → 42.5 tok/s) and saves ~430MB VRAM — while q8 KV cache barely saves memory and slows decode. The usual KV-cache advice flips.]]></description>
            <category>Gemma 4</category>
            <category>GTX 970</category>
            <category>Flash Attention</category>
            <category>KV cache</category>
            <category>llama.cpp</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] DiffusionGemma 26B NVFP4 on a DGX Spark: 158 tok/s, and why diffusion tok/s lies]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-diffusiongemma-nvfp4-vllm</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-diffusiongemma-nvfp4-vllm</guid>
            <pubDate>Sat, 13 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[DiffusionGemma 26B-A4B runs on vLLM on a 128GB DGX Spark via an official prebuilt image — no PR-waiting, no cherry-picking. NVFP4 hits 158 tok/s single-stream and 257 aggregate. But a single tok/s number lies: diffusion speed is decided by whether the 256-token canvas fills.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DiffusionGemma</category>
            <category>diffusion LLM</category>
            <category>NVFP4</category>
            <category>vLLM</category>
            <category>local LLM</category>
        </item>
        <item>
            <title><![CDATA[[Local LLM] Weights win: a 284B crushed to 2-bit still beats the small model that fits]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-284b-q2-quality</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-284b-q2-quality</guid>
            <pubDate>Fri, 12 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[DeepSeek-V4-Flash (284B) only fits a 128GB box at asymmetric Q2 (~80GB). Sounds like suicide quantization — but it's surgical: only the layers that barely affect quality get cut. As a daily agent it ran 280 turns with zero degradation. Big enough weights survive 2-bit.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DeepSeek-V4-Flash</category>
            <category>quantization</category>
            <category>Q2</category>
            <category>asymmetric quantization</category>
            <category>local LLM</category>
            <category>ds4</category>
        </item>
        <item>
            <title><![CDATA[[Local LLM] Running a 15 tok/s 284B as your daily agent brain — the settings that make it bearable]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-context-memory-engineering</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-context-memory-engineering</guid>
            <pubDate>Fri, 12 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A 284B model at 15 tok/s, wired into a daily agent. Two sets of settings make it comfortable — server-side and agent-framework-side. --no-mmap cuts cold start to 57s, the KV disk cache halves prefill, and one missing context_length will crash the whole session.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DeepSeek-V4-Flash</category>
            <category>KV cache</category>
            <category>context</category>
            <category>memory</category>
            <category>ds4</category>
            <category>local LLM</category>
        </item>
        <item>
            <title><![CDATA[[Local LLM] My first Q2 model looked broken on a 128GB box — the real culprit was a parser that couldn't read DSML, not the quantization]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-284b-ds4-engine</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deepseek-v4-flash-284b-ds4-engine</guid>
            <pubDate>Fri, 12 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[DeepSeek-V4-Flash is 284B. I got it onto a single 128GB GB10 with antirez's ds4 engine and an asymmetric Q2 GGUF at 15.6 tok/s. The fun part: the broken tool calls weren't the 2-bit quant's fault. The runtime just couldn't parse DSML.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>DeepSeek-V4-Flash</category>
            <category>ds4</category>
            <category>quantization</category>
            <category>Q2</category>
            <category>tool-call</category>
            <category>DSML</category>
            <category>local LLM</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Qwen3.5-122B on DGX Spark — 2× faster]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-qwen3-122b-vllm-to-atlas-2x</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-qwen3-122b-vllm-to-atlas-2x</guid>
            <pubDate>Thu, 11 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Qwen3.5-122B-A10B tops out at 17 tok/s on a 128GB DGX Spark — the GDN wall in vLLM won't budge, not even with a merged perf PR. I swapped vLLM for the Atlas engine on the same abliterated NVFP4 weights and the throughput doubled to 33.9 tok/s (36.5 with MTP, ~2×), uncensored behavior intact. The real lever was outside the quant toolbox.]]></description>
            <category>Qwen3.5</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>GDN</category>
            <category>vLLM</category>
            <category>Atlas</category>
            <category>NVFP4</category>
            <category>Benchmark</category>
            <category>SM121</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun] A GTX 970 as an offline voice assistant: Gemma 4 E2B + Piper TTS (2.8s end-to-end)]]></title>
            <link>https://ai-muninn.com/en/blog/gtx-970-gemma4-e2b-voice-assistant</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/gtx-970-gemma4-e2b-voice-assistant</guid>
            <pubDate>Tue, 09 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A 2014 GTX 970 running Gemma 4 E2B (vision + audio) plus Piper TTS — a full offline voice assistant that sees, listens, talks back, and writes code. ~2.8s end-to-end, ~$15 of hardware.]]></description>
            <category>Gemma 4</category>
            <category>GTX 970</category>
            <category>multimodal</category>
            <category>Piper TTS</category>
            <category>voice assistant</category>
        </item>
        <item>
            <title><![CDATA[[Just for Fun] Gemma 4 E2B on a GTX 970: the biggest quant runs fastest (47.6 tok/s)]]></title>
            <link>https://ai-muninn.com/en/blog/gtx-970-gemma4-e2b-quantization-benchmark</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/gtx-970-gemma4-e2b-quantization-benchmark</guid>
            <pubDate>Tue, 09 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Four Gemma 4 E2B quants on a 2014 GTX 970. The bigger 3.2GB QAT Q4_0 beats the 2.9GB Q2_K — 47.6 vs 32.8 tok/s — because a tensor-core-less Maxwell card is dequant-bound, not bandwidth-bound.]]></description>
            <category>Gemma 4</category>
            <category>quantization</category>
            <category>GTX 970</category>
            <category>llama.cpp</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] NVFP4 Weight-Only Quantization Taxes Chinese ~2x Harder Than English (gemma-4-12B)]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-nvfp4-quant-tax-chinese-vs-english</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-nvfp4-quant-tax-chinese-vs-english</guid>
            <pubDate>Fri, 05 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I benchmarked BF16 vs FP8 vs NVFP4 weight-only on gemma-4-12B across English (MMLU) and Traditional Chinese (TMMLU+) on a DGX Spark. FP8 is near-lossless on both; NVFP4 drops Chinese ~6pp but English only ~3pp.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>Gemma 4</category>
            <category>NVFP4</category>
            <category>FP8</category>
            <category>quantization</category>
            <category>MMLU</category>
            <category>TMMLU+</category>
            <category>Traditional Chinese</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Gemma 4 12B Omni on DGX Spark: Weight-Only NVFP4 Beats W4A4 (and Keeps Multimodal)]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-12b-omni-nvfp4-weight-only</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-12b-omni-nvfp4-weight-only</guid>
            <pubDate>Thu, 04 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I quantized Google's new omni Gemma 4 12B on a DGX Spark GB10. Weight-only NVFP4 hits 24.9 tok/s in 7.7 GB and keeps image/audio/video working — full W4A4 is slower AND breaks multimodal.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>Gemma 4</category>
            <category>NVFP4</category>
            <category>FP8</category>
            <category>quantization</category>
            <category>vLLM</category>
            <category>omni</category>
            <category>multimodal</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] My Local Agent Flailed at Image Gen — It Was the Harness, Not the Weights]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-agent-harness-not-weights</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-agent-harness-not-weights</guid>
            <pubDate>Tue, 02 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[My local 35B agent went haywire generating images until I read its tool-call logs: 0% malformed calls. The model was fine — a broken ComfyUI tool was making it improvise. The fix was a clean ACI skill, not fine-tuning.]]></description>
            <category>AI Agent</category>
            <category>ACI</category>
            <category>harness</category>
            <category>ComfyUI</category>
            <category>DGX Spark</category>
            <category>Qwen3.6</category>
            <category>tool design</category>
            <category>Hermes</category>
            <category>local LLM</category>
            <category>agent infrastructure</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] NVFP4 shrinks a video model 33% on a DGX Spark — with zero speed gain]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-sulphur-nvfp4-video</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-sulphur-nvfp4-video</guid>
            <pubDate>Mon, 01 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[NVFP4 took a distilled Sulphur 2 (LTX-2.3) video model from 29 to 19.5 GB on a GB10 DGX Spark with no quality loss and — since video is compute-bound — no speed gain (if anything a hair slower).]]></description>
            <category>NVFP4</category>
            <category>Sulphur 2</category>
            <category>LTX-2.3</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>video generation</category>
            <category>ComfyUI</category>
            <category>quantization</category>
            <category>diffusion</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] NVFP4 W4A4 beats FP8 on a DGX Spark MoE: 67 vs 52 tok/s once CUDA graphs fire]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-nvfp4-w4a4-moe-cudagraph</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-nvfp4-w4a4-moe-cudagraph</guid>
            <pubDate>Mon, 01 Jun 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[On a GB10 DGX Spark, NVFP4 W4A4 went from 23 to 67 tok/s the moment I dropped --enforce-eager — beating FP8 by 29% and saving 16GB. The catch from Part 32 was real, just dense-only.]]></description>
            <category>NVFP4</category>
            <category>W4A4</category>
            <category>FP8</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>vLLM</category>
            <category>MoE</category>
            <category>CUDA graph</category>
            <category>MTP</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[NVFP4 is 1.5× FP8 on a DGX Spark — but it's compression, not the FP4 cores]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-nvfp4-compression-not-compute</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-nvfp4-compression-not-compute</guid>
            <pubDate>Sat, 30 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[On a GB10 DGX Spark, NVFP4 beats FP8 by ~1.5× for single-stream decode on a dense model. But the win is bandwidth (smaller weights), not the FP4 tensor cores — the fastest path never touches them.]]></description>
            <category>NVFP4</category>
            <category>FP8</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>vLLM</category>
            <category>benchmark</category>
            <category>quantization</category>
            <category>bandwidth</category>
        </item>
        <item>
            <title><![CDATA[[LLM 101 #7] How to spot AI hallucinations — three red flags before you verify]]></title>
            <link>https://ai-muninn.com/en/blog/llm-101-spot-ai-hallucination</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/llm-101-spot-ai-hallucination</guid>
            <pubDate>Sat, 23 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[AI delivers wrong answers in the same confident tone as right ones. Three red flags to catch it early — impossible numbers, suspiciously specific details, answers that shift on a re-ask — plus a case where ChatGPT gave me a +205% P&L that can't exist.]]></description>
            <category>LLM</category>
            <category>hallucination</category>
            <category>beginner</category>
            <category>verification</category>
            <category>ChatGPT</category>
        </item>
        <item>
            <title><![CDATA[Round 2 EAGLE-3 retrain didn't break the ceiling — a 60-hour null-result writeup]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-eagle3-round2-null-result</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-eagle3-round2-null-result</guid>
            <pubDate>Thu, 21 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[After Part 30's endpoint correction showed Round 1 didn't actually 2x chat throughput, Round 2 added 30k regenerated Chinese instruction samples and trained for 41 hours. Result: Round 2 B drafter delivers chat EN 45 tok/s / ZH 29 tok/s — essentially the same as v1 (EN 46 / ZH 27), and well below vanilla MTP n=4's EN 53 / ZH 45. The EAGLE-3 small head hits an architectural ceiling against the abliterated body; more data doesn't fix it. Plus we found a scheduler deadlock in the vLLM Gemma 4 preview image (`gemma4-0505-arm64-cu130`, internal build `0.20.2rc1.dev49+g9b4e83934`) under long-running extract_hidden_states use (hit three times, mitigated with a watchdog).]]></description>
            <category>Gemma 4</category>
            <category>abliteration</category>
            <category>EAGLE-3</category>
            <category>speculative decoding</category>
            <category>vLLM</category>
            <category>speculators</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>FP8</category>
            <category>huihui-ai</category>
            <category>fine-tune</category>
            <category>benchmark</category>
            <category>null result</category>
        </item>
        <item>
            <title><![CDATA[[Claude Code] Rules I'd Skip, Hooks I Can't — I Wrote a Hook That Blocks My Own Blog Commits]]></title>
            <link>https://ai-muninn.com/en/blog/claude-code-publish-gate-hook</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/claude-code-publish-gate-hook</guid>
            <pubDate>Tue, 19 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I had a rule called 'fact-check before publishing.' I still shipped three fabrications. The problem wasn't the rule — it was where I put it. This is how I promoted it from skill to hook: a small script guarding the moment I press 'send,' so I can't even try without verification.]]></description>
            <category>Claude Code</category>
            <category>hooks</category>
            <category>skills</category>
            <category>verification debt</category>
            <category>AI Workflow</category>
            <category>git commit</category>
            <category>publish gate</category>
            <category>self-modification</category>
            <category>agent self-improvement</category>
        </item>
        <item>
            <title><![CDATA[EAGLE-3 fine-tune against an abliterated Gemma 4 body — Round 1 flattens the acceptance curve (plus a measurement lesson)]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-eagle3-finetune-abliterated-round1</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-eagle3-finetune-abliterated-round1</guid>
            <pubDate>Sat, 16 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[RedHatAI's EAGLE-3 drafter fine-tuned to realign with huihui Gemma 4 26B-A4B abliterated FP8 on a single DGX Spark GB10 — 1 epoch / 50k Magpie samples / 11h. Inference bench on raw `/v1/completions`: pos 3 acceptance climbs from vanilla's 20.5% to 72.7%; n=4 throughput goes from ~50 to 100.36 tok/s aggregate. **A later paired bench revealed the throughput comparison used different endpoints for baseline (chat) and retrain (raw) — on production chat workloads the real uplift is far smaller than 2×; see the endpoint correction at the top of the post**. Part 28's mechanism observation (deep speculation acceptance scatters on abliterated distributions) still holds. Includes a Speculators upstream create_empty_sample dtype bug + patch and a Phase 0 catalog of 6 community prior-art repos.]]></description>
            <category>Gemma 4</category>
            <category>abliteration</category>
            <category>EAGLE-3</category>
            <category>speculative decoding</category>
            <category>vLLM</category>
            <category>speculators</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>FP8</category>
            <category>huihui-ai</category>
            <category>fine-tune</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[30 lines of docker for +34% on DGX Spark: huihui Gemma 4 FP8 + vanilla MTP n=1 deployment recipe]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-huihui-gemma4-mtp-n1-recipe</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-huihui-gemma4-mtp-n1-recipe</guid>
            <pubDate>Thu, 14 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Part 28 explained why deep speculation breaks on an abliterated body; this post is the recipe for the part that already works. huihui Gemma 4 26B-A4B FP8 + Google's vanilla MTP draft at num_speculative_tokens=1 takes baseline 39.3 tok/s to 52.6 tok/s (+34%) on GB10, no retraining required. ~30 lines of docker plus a bind-mount of PR #41745's gemma4_mtp.py. Includes a 3-step sanity check and a clear list of when n=1 stops being enough.]]></description>
            <category>Gemma 4</category>
            <category>abliteration</category>
            <category>MTP</category>
            <category>speculative decoding</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>FP8</category>
            <category>huihui-ai</category>
            <category>deployment</category>
            <category>recipe</category>
        </item>
        <item>
            <title><![CDATA[Want MTP speedup on abliterated Gemma 4? Vanilla draft can't track the modified body]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-huihui-gemma4-fp8-mtp-34pct</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-huihui-gemma4-fp8-mtp-34pct</guid>
            <pubDate>Sat, 09 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I self-quantized huihui's abliterated Gemma 4 26B-A4B to FP8-Dynamic and shipped it to HF. After sweeping num_speculative_tokens 1→4, the abliterated body is exactly as fast as vanilla on the same stack (39.4 vs 39.3 tok/s baseline) and the MTP boost at n=1 is equivalent — but per-position acceptance decays so steeply that deeper speculation is wasted. Three drafts of this article each smuggled in a different fabrication that Codex caught; this is the corrected version.]]></description>
            <category>Gemma 4</category>
            <category>abliteration</category>
            <category>MTP</category>
            <category>speculative decoding</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>FP8</category>
            <category>huihui-ai</category>
            <category>benchmark</category>
        </item>
        <item>
            <title><![CDATA[Liftoff: Gemma 4 hits 670 tok/s aggregate on DGX Spark (108 tok/s single-stream)]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-mtp-108-toks</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-mtp-108-toks</guid>
            <pubDate>Wed, 06 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Google announced Multi-Token Prediction drafters for Gemma 4 on 2026-05-05. The vLLM PR was opened and approved the same day; a preview Docker image shipped hours later. I tested it on DGX Spark: Gemma 4 26B-A4B-it FP8 + MTP γ=4 hits 108.78 tok/s single-stream (2.66× baseline), 674.28 tok/s aggregate at concurrency=8. One undocumented trap: the drafter pairs with -it, not base.]]></description>
            <category>Gemma 4</category>
            <category>MTP</category>
            <category>speculative decoding</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>FP8</category>
            <category>NVFP4</category>
            <category>benchmark</category>
            <category>Gemma 4 26B-A4B</category>
        </item>
        <item>
            <title><![CDATA[How a zh-TW Linter Found 128 Mainland-China Drift in My Own Writing]]></title>
            <link>https://ai-muninn.com/en/blog/zhtw-mcp-calque-blindspot-sweep</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/zhtw-mcp-calque-blindspot-sweep</guid>
            <pubDate>Tue, 05 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I ran sysprog21/zhtw-mcp across 72 of my Traditional Chinese articles. Three sweeps, 128 cross-strait substitutions across 42 files. The real takeaway wasn't the count — it was discovering my blindspot isn't 'I don't know the right Taiwanese term,' it's 'when a Mainland term shows up I don't auto-doubt it.']]></description>
            <category>zh-TW</category>
            <category>AI Workflow</category>
            <category>linter</category>
            <category>skills</category>
            <category>calque</category>
            <category>localization</category>
            <category>zhtw-mcp</category>
            <category>sysprog21</category>
            <category>writing pipeline</category>
            <category>Traditional Chinese</category>
            <category>LLM bias</category>
            <category>Claude Code</category>
        </item>
        <item>
            <title><![CDATA[[Field Guide] Z-Image Turbo — does choosing a faster config hurt quality? LPIPS + CLIPScore answer]]></title>
            <link>https://ai-muninn.com/en/blog/zimage-turbo-quality-lpips-clipscore</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/zimage-turbo-quality-lpips-clipscore</guid>
            <pubDate>Mon, 04 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Does Z-Image Turbo quantization break image quality? Two-axis benchmark — LPIPS (perceptual distance vs BF16) + CLIPScore (image-text alignment) — across 6 prompts × 4 configs × 3 seeds = 72 samples. Result: NVFP4 produces images that look different from BF16, but no measured regression in this sample — all 4 configs land within ±0.04 std on CLIPScore, smaller than the noise floor. Production users should re-verify with their own prompt set.]]></description>
            <category>Z-Image</category>
            <category>ComfyUI</category>
            <category>NVFP4</category>
            <category>FP8</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>image gen</category>
            <category>benchmark</category>
            <category>quantization</category>
            <category>LPIPS</category>
            <category>CLIPScore</category>
            <category>quality</category>
            <category>Z-Image Turbo</category>
        </item>
        <item>
            <title><![CDATA[[Field Guide] Z-Image Turbo — choosing the right config (1.37× faster, 44% less RAM)]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-zimage-turbo-nvfp4-bench</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-zimage-turbo-nvfp4-bench</guid>
            <pubDate>Mon, 04 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[I ran six Z-Image Turbo quantization configs on DGX Spark GB10 — BF16 baseline, FP8 cast standard, FP8 cast fast, FP8 scaled (Kijai), NVFP4, NVFP4+FP8 encoder. With N=10 isolated GPU, NVFP4 transformer hits 5.50s warm versus BF16 7.55s (1.37× faster). All three FP8 paths are slower than BF16. Model working set drops from 20.6 GB (BF16) to 11.5 GB (NVFP4+FP8 encoder) — 44% smaller.]]></description>
            <category>Z-Image</category>
            <category>ComfyUI</category>
            <category>NVFP4</category>
            <category>FP8</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>image gen</category>
            <category>benchmark</category>
            <category>quantization</category>
            <category>Z-Image Turbo</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Watching English Videos with DGX Spark: Nemotron Omni Multimodal on GB10]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-nemotron-omni-multimodal-video</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-nemotron-omni-multimodal-video</guid>
            <pubDate>Fri, 01 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Same DGX Spark, different goal: watch a 3-minute Andrej Karpathy talk and output the spoken content + visual scene. 89 seconds wall, 53,842 prompt tokens, factually correct. The use_audio_in_video flag, the upstream-image gotcha, and the long-video knob math.]]></description>
            <category>Nemotron Omni</category>
            <category>multimodal</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>video understanding</category>
            <category>Parakeet</category>
            <category>audio transcription</category>
            <category>Nemotron 3 Nano Omni</category>
            <category>Nemotron 3 Nano Omni 30B-A3B</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Nemotron 3 Nano on DGX Spark: 74.75 tok/s NVFP4 — 11.5% Past the Public Baseline]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-nemotron-3-nano-w4a16-74-toks</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-nemotron-3-nano-w4a16-74-toks</guid>
            <pubDate>Fri, 01 May 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Ten days ago I called NVFP4 a trap on DGX Spark GB10. Today the same hardware hits 74.75 tok/s on Nemotron 3 Nano W4A16, beating my own FP8 ceiling and the public 67 tok/s forum number. The 4-layer patch stack, the quant variant choice, and the bandwidth math behind it.]]></description>
            <category>Nemotron 3</category>
            <category>NVFP4</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>MoE</category>
            <category>W4A16</category>
            <category>b12x</category>
            <category>benchmark</category>
            <category>Nemotron 3 Nano</category>
            <category>Nemotron 3 Nano 30B-A3B</category>
        </item>
        <item>
            <title><![CDATA[Vercel Hobby hit 1M/1M Edge Requests. The bug was a Cache-Control header.]]></title>
            <link>https://ai-muninn.com/en/blog/vercel-edge-requests-must-revalidate-trap</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/vercel-edge-requests-must-revalidate-trap</guid>
            <pubDate>Thu, 30 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[ai-muninn.com burned through Vercel Hobby's 1M Edge Requests quota this month. It wasn't traffic, wasn't bots, wasn't large images. Next.js defaults /public/* to must-revalidate, which makes every conditional GET (even 304s) count as an edge request. Three lines of next.config.ts to fix. Three rounds of fact-check rewrites to publish.]]></description>
            <category>Vercel</category>
            <category>Next.js</category>
            <category>Cache-Control</category>
            <category>Edge Network</category>
            <category>Performance</category>
            <category>Postmortem</category>
        </item>
        <item>
            <title><![CDATA[[SWE-bench] Where Qwen 3.6 35B Loses on SWE-bench Lite: Anatomy of 155 Unresolved Tasks]]></title>
            <link>https://ai-muninn.com/en/blog/swe-bench-qwen36-failure-modes</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/swe-bench-qwen36-failure-modes</guid>
            <pubDate>Tue, 28 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Qwen 3.6 35B-A3B FP8 hits 48.33% (145/300) on SWE-bench Lite with the same scaffold that gets Gemma 4 26B to 38.67%. The 9.66-point gap deserves an explanation. This is a deep dive on Qwen 3.6's 155 failures: 76% are wrong-logic patches, 14% are incomplete fixes, 10% never submit. The categorization is asymmetric — Gemma 4's failures haven't been classified the same way yet — so the cross-model comparison is part hypothesis, part data.]]></description>
            <category>SWE-bench</category>
            <category>Qwen 3.6</category>
            <category>Gemma 4</category>
            <category>failure analysis</category>
            <category>open-source models</category>
            <category>scaffold</category>
            <category>DGX Spark</category>
            <category>wrong_logic</category>
            <category>Qwen 3.6 35B-A3B</category>
        </item>
        <item>
            <title><![CDATA[[llm-compressor] Self-Quantizing a 35B Abliterated MoE to FP8 on DGX Spark: 4 OOMs, 3 Prefix Bugs, and Why the First Success Wasn't Actually FP8]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-abliterated-fp8-uma-quantization</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-abliterated-fp8-uma-quantization</guid>
            <pubDate>Tue, 28 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Quantizing huihui-ai's Qwen3.6-35B-A3B abliterated to FP8 for vLLM on a 128 GB UMA box. Seven attempts, two distinct OOM modes, a model class that silently breaks vLLM's loader, and why streaming save_pretrained returns BF16 not FP8. Final result: 51.72 tok/s, 1.68× BF16.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>llm-compressor</category>
            <category>FP8</category>
            <category>FP8_DYNAMIC</category>
            <category>abliteration</category>
            <category>Qwen 3.6</category>
            <category>MoE</category>
            <category>UMA</category>
            <category>OOM</category>
            <category>MTP</category>
            <category>speculative-decoding</category>
            <category>quantization</category>
            <category>Qwen 3.6 35B-A3B</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Abliteration Costs 1.85pp on Traditional Chinese — and 7.7pp on Trust Law]]></title>
            <link>https://ai-muninn.com/en/blog/tmmluplus-qwen-abliterated-cost</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/tmmluplus-qwen-abliterated-cost</guid>
            <pubDate>Sun, 26 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Ran huihui-ai's abliterated Qwen 3.6 35B through the same TMMLU+ harness as Part 21. Aggregate dropped 75.07% → 73.22%. The cost isn't uniform: regulatory subjects (信託 −7.7, 行政法 −7.1) lose the most, while pure logic and math actually improve. Hokkien also got worse — abliteration doesn't fix data scarcity.]]></description>
            <category>TMMLU+</category>
            <category>abliteration</category>
            <category>Traditional Chinese</category>
            <category>Qwen 3.6</category>
            <category>uncensored</category>
            <category>huihui-ai</category>
            <category>DGX Spark</category>
            <category>vLLM</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] TMMLU+ Paired Eval: Qwen 3.6 35B Sweeps Gemma 4 26B 51-of-51 on Traditional Chinese]]></title>
            <link>https://ai-muninn.com/en/blog/tmmluplus-qwen-vs-gemma-traditional-chinese</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/tmmluplus-qwen-vs-gemma-traditional-chinese</guid>
            <pubDate>Sat, 25 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Two MoE models on the same DGX Spark, same harness, same 22,690 questions. Qwen 3.6 35B-A3B scored 75.07%, Gemma 4 26B-A4B scored 46.30%. Qwen won every single one of the 51 subjects — including Taiwan-specific topics where I expected Gemma to win.]]></description>
            <category>TMMLU+</category>
            <category>Traditional Chinese</category>
            <category>Qwen 3.6</category>
            <category>Gemma 4</category>
            <category>benchmark</category>
            <category>繁體中文</category>
            <category>DGX Spark</category>
            <category>vLLM</category>
            <category>lm-evaluation-harness</category>
        </item>
        <item>
            <title><![CDATA[[Hands-On] Making NVFP4 17% Faster on GB10 with a Triton FP8 Bypass]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-nvfp4-fp8-triton-patch</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-nvfp4-fp8-triton-patch</guid>
            <pubDate>Wed, 22 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Part 19 proved NVFP4 is a trap on DGX Spark. This time we fight back: a Triton kernel that dequants NVFP4 to FP8 and feeds the FP8 tensor cores. 40.8 → 47.6 tok/s, with full code.]]></description>
            <category>NVFP4</category>
            <category>FP8</category>
            <category>Triton</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>vLLM</category>
            <category>Qwen 3.6</category>
            <category>Qwen 3.6 35B-A3B</category>
            <category>kernel</category>
            <category>quantization</category>
            <category>tensor core</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] NVFP4 Is a Trap on GB10: FP8 Wins by 32% (vLLM + SGLang Tested)]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-nvfp4-trap-gb10-fp8-wins</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-nvfp4-trap-gb10-fp8-wins</guid>
            <pubDate>Tue, 21 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[NVFP4 should be faster than FP8 — fewer bits, less bandwidth. On DGX Spark's GB10 (SM121), it's 32% slower. Root cause: missing hardware instruction. Dual-engine proof with vLLM and SGLang.]]></description>
            <category>NVFP4</category>
            <category>FP8</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>vLLM</category>
            <category>SGLang</category>
            <category>Qwen 3.6</category>
            <category>Qwen 3.6 35B-A3B</category>
            <category>benchmark</category>
            <category>quantization</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Same Scaffold, Three Models: 16% → 38% → 48% on SWE-bench Lite]]></title>
            <link>https://ai-muninn.com/en/blog/swe-bench-scaffold-transfers-three-models</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/swe-bench-scaffold-transfers-three-models</guid>
            <pubDate>Mon, 20 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[One scaffold (backticks + edit-tool + budget prompt), three models (Gemma 4 E4B, Gemma 4 26B, Qwen 3.6 35B), zero code changes between runs. Qwen 3.6 hit 48.33% — beating SWE-agent + Claude 3.7 Sonnet. The scaffold is the fixed cost; the model is the variable.]]></description>
            <category>SWE-bench</category>
            <category>Gemma 4</category>
            <category>Qwen 3.6</category>
            <category>scaffold</category>
            <category>mini-swe-agent</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>benchmark</category>
            <category>open-source</category>
            <category>Gemma 4 26B-A4B</category>
            <category>Qwen 3.6 35B-A3B</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] SWE-bench Lite 38.67% with a 26B Local Model — 0.33% from Claude 3.5 Sonnet Scaffolds]]></title>
            <link>https://ai-muninn.com/en/blog/swe-bench-lite-gemma4-26b-38-percent</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/swe-bench-lite-gemma4-26b-38-percent</guid>
            <pubDate>Fri, 17 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Gemma 4 26B-A4B FP8 scored 116/300 on SWE-bench Lite, ranking #16 globally. Zero API cost on a DGX Spark. The scaffold — not the model — was the differentiator.]]></description>
            <category>SWE-bench</category>
            <category>Gemma 4</category>
            <category>mini-swe-agent</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>benchmark</category>
            <category>open-source</category>
            <category>scaffold</category>
            <category>edit-tool</category>
            <category>Gemma 4 26B-A4B</category>
        </item>
        <item>
            <title><![CDATA[[LLM 101 #6] Why Run AI on Your Own Computer? It's Not a Cheaper ChatGPT — It's a Different Tool]]></title>
            <link>https://ai-muninn.com/en/blog/llm-101-why-run-ai-locally</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/llm-101-why-run-ai-locally</guid>
            <pubDate>Fri, 17 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Local AI isn't a budget ChatGPT. It's a knowledge extractor, private code assistant, and offline tool. Monthly power cost ~$1.20 vs ChatGPT Plus $20. This guide has a decision table for when to use which.]]></description>
            <category>LLM</category>
            <category>Local AI</category>
            <category>Ollama</category>
            <category>Beginner</category>
            <category>Privacy</category>
            <category>Cost</category>
        </item>
        <item>
            <title><![CDATA[[Ask AI Right #7] What AI Does Poorly — Four Landmines to Know Before Using ChatGPT or Claude in 2026]]></title>
            <link>https://ai-muninn.com/en/blog/ai-ask-right-what-ai-does-poorly</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-ask-right-what-ai-does-poorly</guid>
            <pubDate>Thu, 16 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[AI is strong, but four things still trip it up in 2026: hallucinations, stale knowledge, short memory, and privacy defaults. Even Anthropic's own lawyers got caught by the first one.]]></description>
            <category>AI</category>
            <category>Hallucination</category>
            <category>ChatGPT</category>
            <category>Claude</category>
            <category>Privacy</category>
            <category>Beginner</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] Gemma 4 26B Cleared a SWE-bench Lite Instance — After 28 Tries Across Two Days]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-swe-bench-scaffold-engineering</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-swe-bench-scaffold-engineering</guid>
            <pubDate>Wed, 15 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Two days running mini-swe-agent + vLLM on a GB10. From wrong doc conclusions to Gemma 4 self-submitting a clean patch in 38 steps — what actually unlocked it.]]></description>
            <category>SWE-bench</category>
            <category>mini-swe-agent</category>
            <category>Gemma 4</category>
            <category>vLLM</category>
            <category>agent scaffold</category>
            <category>GX10</category>
            <category>Gemma 4 26B-A4B</category>
        </item>
        <item>
            <title><![CDATA[[LLM Deep Dive] What Quantization Algorithms Actually Do: From Q4_K_M to TurboQuant]]></title>
            <link>https://ai-muninn.com/en/blog/llm-deep-dive-quantization-algorithms</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/llm-deep-dive-quantization-algorithms</guid>
            <pubDate>Wed, 15 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[How does Q4_K_M fit a 14B model into 4 bits without ruining it? Not by 'cutting off 75%' — but through three layers: K-quant super-blocks, TurboQuant random rotation, and a 1-bit JL sign sketch. A mechanism walkthrough without the equations.]]></description>
            <category>LLM</category>
            <category>Quantization</category>
            <category>K-quant</category>
            <category>TurboQuant</category>
            <category>PolarQuant</category>
            <category>QJL</category>
            <category>Algorithms</category>
        </item>
        <item>
            <title><![CDATA[[Ask AI Right #6] The Art of Follow-Up Questions — What to Do When the First Answer Is Too Shallow]]></title>
            <link>https://ai-muninn.com/en/blog/ai-ask-right-follow-up-questions</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-ask-right-follow-up-questions</guid>
            <pubDate>Tue, 14 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[The first answer AI gives you is a rough draft, not the final answer. Learn 5 follow-up techniques — adding constraints, asking for comparisons, and letting AI ask YOU questions — to get dramatically better results.]]></description>
            <category>AI</category>
            <category>Conversation</category>
            <category>Follow-up</category>
            <category>Beginner</category>
            <category>Productivity</category>
        </item>
        <item>
            <title><![CDATA[[LLM 101 #5] Context Window — How Much Can AI Read at Once?]]></title>
            <link>https://ai-muninn.com/en/blog/llm-101-context-window</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/llm-101-context-window</guid>
            <pubDate>Tue, 14 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[AI forgets what you said 20 messages ago. It's not broken — its desk is full. This guide explains context windows, why conversations go stale, and how to work around the limit.]]></description>
            <category>LLM</category>
            <category>Context Window</category>
            <category>Beginner</category>
            <category>Tokens</category>
            <category>AI Memory</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] Gemma 4 Went from 40 Errors to a 9-Step Bug Fix — by Switching One Thing]]></title>
            <link>https://ai-muninn.com/en/blog/swe-bench-local-models-framework-matters</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/swe-bench-local-models-framework-matters</guid>
            <pubDate>Mon, 13 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A feasibility test: can open-source models run SWE-Bench locally for free? Gemma 4 26B failed on OpenHands (40+ errors) but fixed a test bug in 9 steps on SWE-agent. Same model — the action format was the difference.]]></description>
            <category>SWE-Bench</category>
            <category>Gemma 4</category>
            <category>Qwen 3.5</category>
            <category>OpenHands</category>
            <category>SWE-agent</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>Tool Calling</category>
            <category>AI Agent</category>
            <category>Gemma 4 26B-A4B</category>
            <category>Qwen 3.5 35B</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Gemma 4 on DGX Spark — Which Model Should You Pick?]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-complete-guide</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-complete-guide</guid>
            <pubDate>Mon, 13 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Gemma 4 E2B / E4B / 26B MoE / 31B Dense benchmarked on DGX Spark, RTX 5090, and MacBook Pro. One table with speed, memory, quantization format. Selection guide included.]]></description>
            <category>Gemma 4</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>benchmark</category>
            <category>MoE</category>
            <category>NVFP4</category>
            <category>vLLM</category>
            <category>Ollama</category>
            <category>DGX Spark worth it</category>
            <category>DGX Spark vs RTX 5090</category>
            <category>best model DGX Spark 2026</category>
        </item>
        <item>
            <title><![CDATA[Claude Code Burning Through Tokens? 8 Fixes to Make Sessions Last 10x Longer]]></title>
            <link>https://ai-muninn.com/en/blog/dev-workflow-token-burn-rate</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dev-workflow-token-burn-rate</guid>
            <pubDate>Mon, 13 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[You just started using Claude Code and the context window keeps filling up. Here's where the tokens actually go, what you can do about it, and how to make Claude remember things without re-reading everything.]]></description>
            <category>Claude Code</category>
            <category>tokens</category>
            <category>context window</category>
            <category>beginner</category>
            <category>tips</category>
            <category>MCP</category>
            <category>CLAUDE.md</category>
            <category>knowledge graph</category>
            <category>musubi</category>
            <category>QMD</category>
            <category>AI cost</category>
            <category>save tokens</category>
        </item>
        <item>
            <title><![CDATA[[Ask AI Right #5] Before You Build It, Ask: Does This Already Exist?]]></title>
            <link>https://ai-muninn.com/en/blog/ai-ask-right-does-it-exist</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-ask-right-does-it-exist</guid>
            <pubDate>Mon, 13 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Your first question to AI shouldn't be 'help me do X.' It should be 'is there something that already does X?' This article teaches you how to use AI as a research assistant — finding tools, comparing alternatives, and verifying they're still alive.]]></description>
            <category>AI</category>
            <category>Tools</category>
            <category>Research</category>
            <category>Beginner</category>
            <category>Productivity</category>
        </item>
        <item>
            <title><![CDATA[[Claude Code] Build a Self-Auditing Skill That Keeps Your Config Lean]]></title>
            <link>https://ai-muninn.com/en/blog/claude-code-slim-self-audit-skill</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/claude-code-slim-self-audit-skill</guid>
            <pubDate>Mon, 13 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Your CLAUDE.md and MEMORY.md grow silently until they eat 10K+ tokens per turn. I built a /slim skill that lets Claude diagnose and fix its own bloat — here's how.]]></description>
            <category>Claude Code</category>
            <category>tokens</category>
            <category>context window</category>
            <category>skills</category>
            <category>CLAUDE.md</category>
            <category>MEMORY.md</category>
            <category>config optimization</category>
            <category>self-audit</category>
            <category>QMD</category>
            <category>AI Workflow</category>
        </item>
        <item>
            <title><![CDATA[[DGX Spark] From Unboxing to Running: Complete Deployment Guide]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-deployment-guide</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-deployment-guide</guid>
            <pubDate>Mon, 13 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Everything you need to go from a sealed DGX Spark box to serving your first local LLM. Hardware check, Ollama quickstart, vLLM production setup, model selection, and the 5 gotchas that cost hours.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>GX10</category>
            <category>vLLM</category>
            <category>Ollama</category>
            <category>deployment</category>
            <category>Blackwell</category>
            <category>SM121</category>
            <category>DGX Spark worth it</category>
            <category>DGX Spark setup 2026</category>
            <category>DGX Spark vs RTX 5090</category>
            <category>Gemma 4</category>
            <category>Gemma 4 26B-A4B</category>
        </item>
        <item>
            <title><![CDATA[[Ask AI Right #4] Why AI Feels Useless to You — Answer Machine vs Collaboration Tool]]></title>
            <link>https://ai-muninn.com/en/blog/ai-ask-right-why-ai-feels-useless</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-ask-right-why-ai-feels-useless</guid>
            <pubDate>Sat, 11 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Same AI, same question, different results. The people who find ChatGPT life-changing and the people who think it's useless are doing completely different things — and the difference is a single mindset shift.]]></description>
            <category>AI</category>
            <category>ChatGPT</category>
            <category>Prompting</category>
            <category>Beginner</category>
            <category>Workflow</category>
            <category>Mindset</category>
        </item>
        <item>
            <title><![CDATA[[LLM 101 #4] What Is Quantization? Q4, Q8, FP16 Explained]]></title>
            <link>https://ai-muninn.com/en/blog/llm-101-what-is-quantization</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/llm-101-what-is-quantization</guid>
            <pubDate>Fri, 10 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Q4_K_M, Q8_0, FP16 — the same model comes in a dozen versions and the names look like hieroglyphs. This guide explains what quantization actually does, why it doesn't ruin the model, and which level to pick.]]></description>
            <category>LLM</category>
            <category>Quantization</category>
            <category>Ollama</category>
            <category>Beginner</category>
            <category>Q4_K_M</category>
            <category>GGUF</category>
        </item>
        <item>
            <title><![CDATA[[Ask AI Right #3] You Don't Know What You Need — Let AI Find It]]></title>
            <link>https://ai-muninn.com/en/blog/ai-ask-right-find-your-needs</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-ask-right-find-your-needs</guid>
            <pubDate>Fri, 10 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Most people don't struggle with using AI — they struggle with knowing what to use it for. This article teaches you a simple method to let AI identify the repetitive parts of your workday you've stopped noticing.]]></description>
            <category>AI</category>
            <category>Productivity</category>
            <category>Beginner</category>
            <category>Workflow</category>
        </item>
        <item>
            <title><![CDATA[[LLM 101 #3] How to Choose an AI Model: Gemma vs Llama vs Qwen vs Mistral (2026)]]></title>
            <link>https://ai-muninn.com/en/blog/llm-101-how-to-choose-a-model</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/llm-101-how-to-choose-a-model</guid>
            <pubDate>Fri, 10 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Which local AI model should you download? Gemma, Llama, Qwen, Mistral compared by size, speed, and quality. Simple formula: parameters × 0.6 = GB needed. Beginner-friendly guide.]]></description>
            <category>LLM</category>
            <category>Model Selection</category>
            <category>Ollama</category>
            <category>Beginner</category>
            <category>Gemma</category>
            <category>Llama</category>
            <category>Qwen</category>
        </item>
        <item>
            <title><![CDATA[[Ask AI Right #2] You Opened AI — Now What Do You Say?]]></title>
            <link>https://ai-muninn.com/en/blog/ai-ask-right-first-message</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-ask-right-first-message</guid>
            <pubDate>Thu, 09 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[AI isn't Google — you're not searching, you're having a conversation. This article teaches you what to say when you first open ChatGPT, five things you can try right now, and how to adjust when the answer isn't quite right.]]></description>
            <category>AI</category>
            <category>ChatGPT</category>
            <category>Beginner</category>
            <category>Conversation</category>
        </item>
        <item>
            <title><![CDATA[[Ask AI Right #1] Which AI Should You Use in 2026?]]></title>
            <link>https://ai-muninn.com/en/blog/ai-ask-right-which-ai-to-use-2026</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/ai-ask-right-which-ai-to-use-2026</guid>
            <pubDate>Thu, 09 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[ChatGPT, Claude, and Gemini — the three AI assistants you can start using right now. A no-jargon guide to what each one does best, how much they cost, and how to get started.]]></description>
            <category>AI</category>
            <category>ChatGPT</category>
            <category>Claude</category>
            <category>Gemini</category>
            <category>Beginner</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Rescuing Gemma 4 31B on a 32GB MacBook Pro: From 1.5 to 12.8 tok/s]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-31b-rescue-mbp-32gb</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-31b-rescue-mbp-32gb</guid>
            <pubDate>Wed, 08 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Gemma 4 31B runs at 1.5 tok/s on MBP M1 Max with Ollama due to swap. The fix: reduce context window (9 tok/s) or switch to oMLX (12.8 tok/s). The real culprit is KV cache allocation, not model size.]]></description>
            <category>Gemma 4</category>
            <category>31B</category>
            <category>M1 Max</category>
            <category>Ollama</category>
            <category>oMLX</category>
            <category>swap</category>
            <category>KV cache</category>
            <category>TurboQuant</category>
            <category>Apple Silicon</category>
            <category>memory management</category>
            <category>Gemma 4 31B</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] 4 Machines, 4 Models, 1 Answer: Memory Decides Everything]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-4-machines-4-models-bandwidth</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-4-machines-4-models-bandwidth</guid>
            <pubDate>Wed, 08 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Gemma 4 E2B through 31B benchmarked on RTX 5090, M1 Max, DGX Spark, and M4 with Ollama. E2B hits 310 tok/s on 5090. 31B hits 1.5 tok/s on MBP — swap kills faster hardware. Memory capacity > bandwidth.]]></description>
            <category>Gemma 4</category>
            <category>RTX 5090</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>M1 Max</category>
            <category>M4</category>
            <category>Ollama</category>
            <category>benchmark</category>
            <category>memory bandwidth</category>
            <category>swap</category>
            <category>E2B</category>
            <category>E4B</category>
            <category>26B</category>
            <category>31B</category>
        </item>
        <item>
            <title><![CDATA[[LLM 101 #2] Dense, MoE, PLE, SSM — Four AI Model Architectures Explained Simply]]></title>
            <link>https://ai-muninn.com/en/blog/llm-101-dense-moe-ple-ssm-architectures</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/llm-101-dense-moe-ple-ssm-architectures</guid>
            <pubDate>Wed, 08 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Dense is everyone working. MoE is expert rotation. PLE is a dictionary on every floor. SSM is a speed reader. A zero-jargon guide to the four main AI model architectures and how to pick between them.]]></description>
            <category>Dense</category>
            <category>MoE</category>
            <category>PLE</category>
            <category>SSM</category>
            <category>Mamba</category>
            <category>LLM</category>
            <category>model architecture</category>
            <category>beginner</category>
            <category>explainer</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Gemma 4 E2B vs E4B: 81 tok/s vs 52 on Three Machines — Bandwidth Is Everything]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-e2b-vs-e4b-ollama-3-machines</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-e2b-vs-e4b-ollama-3-machines</guid>
            <pubDate>Tue, 07 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Gemma 4 E2B is 44-82% faster than E4B across M1 Max, GB10, and M4. We benchmarked both on Ollama with 3 runs per scenario, unique prompts, and proper warm-up. Memory bandwidth predicts generation speed better than anything else.]]></description>
            <category>Gemma 4</category>
            <category>E2B</category>
            <category>E4B</category>
            <category>Ollama</category>
            <category>benchmark</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>M1 Max</category>
            <category>M4</category>
            <category>Apple Silicon</category>
            <category>memory bandwidth</category>
            <category>Gemma 4 E2B</category>
            <category>Gemma 4 E4B</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] From 19 to 50 tok/s: We Quantized Gemma 4 E4B to NVFP4 Before Anyone Else]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-e4b-nvfp4-50-toks</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-e4b-nvfp4-50-toks</guid>
            <pubDate>Tue, 07 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Gemma 4 E4B NVFP4A16 hits 49.9 tok/s on DGX Spark — 2.6x faster than BF16. First NVFP4 checkpoint on HuggingFace. PLE architecture, FP8 vs NVFP4, and the llm-compressor version hell that almost stopped us.]]></description>
            <category>Gemma 4</category>
            <category>E4B</category>
            <category>NVFP4</category>
            <category>FP8</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>quantization</category>
            <category>llm-compressor</category>
            <category>PLE</category>
            <category>HuggingFace</category>
            <category>Gemma 4 E4B</category>
        </item>
        <item>
            <title><![CDATA[[LLM 101 #1] Ollama vs vLLM: Two Ways to Run AI on Your Own Computer]]></title>
            <link>https://ai-muninn.com/en/blog/llm-101-ollama-vs-vllm</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/llm-101-ollama-vs-vllm</guid>
            <pubDate>Tue, 07 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Ollama is a microwave — one command and you're chatting with AI. vLLM is a professional oven — 30% faster, handles multiple users, but takes real setup. A zero-jargon guide to choosing between them.]]></description>
            <category>Ollama</category>
            <category>vLLM</category>
            <category>LLM</category>
            <category>local AI</category>
            <category>beginner</category>
            <category>explainer</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Gemma 4 31B Dense on DGX Spark: 7 tok/s and the Bandwidth Wall]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-31b-dense-7-toks</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-31b-dense-7-toks</guid>
            <pubDate>Sun, 05 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Gemma 4 31B-IT NVFP4 on GB10 maxes out at 7.0 tok/s — bandwidth-bound at 273 GB/s. The math predicted 4.4 tok/s theoretical; NVFP4 compression buys 60% but can't escape the wall. Choose MoE.]]></description>
            <category>Gemma 4</category>
            <category>NVFP4</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>dense</category>
            <category>benchmark</category>
            <category>bandwidth</category>
            <category>Gemma 4 31B</category>
            <category>Gemma 4 31B-IT</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] vLLM vs Ollama on the Same Model: Why 30% Faster on GB10]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-vllm-vs-ollama-same-model</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-vllm-vs-ollama-same-model</guid>
            <pubDate>Sun, 05 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Same Gemma 4 26B-A4B, same GPU, 30% speed gap. vLLM NVFP4 hits 52 tok/s while Ollama Q4_K_M tops at 40. Root cause: Marlin kernels, CUDA graphs, and an Ollama CPU/GPU split trap.]]></description>
            <category>vLLM</category>
            <category>Ollama</category>
            <category>benchmark</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>Gemma 4</category>
            <category>NVFP4</category>
            <category>inference</category>
            <category>Gemma 4 26B-A4B</category>
        </item>
        <item>
            <title><![CDATA[Gemma 4 26B-A4B on DGX Spark: 52 tok/s with NVFP4, skip the 31B]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-gemma4-26b-nvfp4-52-toks</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-gemma4-26b-nvfp4-52-toks</guid>
            <pubDate>Sun, 05 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Gemma 4 26B-A4B + NVFP4 hits 52 tok/s on DGX Spark (GB10) — 7.5× faster than the 31B dense, in 16.5 GB with 82 GB free for KV cache. Plus the vLLM 0.19 marlin/patch gotchas that make it work.]]></description>
            <category>Gemma 4</category>
            <category>NVFP4</category>
            <category>vLLM</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>MoE</category>
            <category>benchmark</category>
            <category>Gemma 4 26B-A4B</category>
        </item>
        <item>
            <title><![CDATA[[DGX Spark] Overheating, 100W Power Cap, 30W Safety Mode — Complete Diagnostic Guide]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-30w-power-safety-mode</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-30w-power-safety-mode</guid>
            <pubDate>Thu, 02 Apr 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[DGX Spark power and thermal issues blew up after Carmack's criticism. This guide covers three distinct symptoms: 30W PD controller defect (needs RMA), 100W thermal throttling, and 5W driver bug (fixable). One command, 30 seconds to diagnose.]]></description>
            <category>GX10</category>
            <category>GB10</category>
            <category>DGX Spark</category>
            <category>power delivery</category>
            <category>vLLM</category>
            <category>hardware</category>
            <category>overheating</category>
            <category>100W</category>
            <category>Carmack</category>
            <category>Gemma 4</category>
            <category>Gemma 4 26B-A4B</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] TurboQuant on GX10: Is 3-bit KV Cache Compression Actually Lossless?]]></title>
            <link>https://ai-muninn.com/en/blog/turboquant-kv-cache-benchmark-gx10</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/turboquant-kv-cache-benchmark-gx10</guid>
            <pubDate>Mon, 30 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Real benchmark numbers for Google's TurboQuant on a GB10/SM121 (DGX Spark) — actual compression ratios, Qwen2.5-3B accuracy validation, and why Qwen3.5-35B's hybrid attention architecture makes things complicated.]]></description>
            <category>TurboQuant</category>
            <category>KV Cache</category>
            <category>Quantization</category>
            <category>vLLM</category>
            <category>Benchmark</category>
            <category>Qwen3.5</category>
            <category>GX10</category>
            <category>SM121</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] openclaw + ChatGPT OAuth: Run GPT-5.4 Agents Without API Credits]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-chatgpt-oauth-gpt54-no-api-key</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-chatgpt-oauth-gpt54-no-api-key</guid>
            <pubDate>Tue, 24 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Your ChatGPT Plus subscription already includes GPT-5.4 with 1M context. openclaw's OAuth flow lets you use it for AI agents — zero API credits, one command. Full setup guide.]]></description>
            <category>openclaw</category>
            <category>GPT-5.4</category>
            <category>ChatGPT</category>
            <category>OAuth</category>
            <category>AI agent</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] NemoClaw Without the Cloud: Swapping Nemotron for a Local Ollama Model]]></title>
            <link>https://ai-muninn.com/en/blog/nemoclaw-local-inference-ollama</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/nemoclaw-local-inference-ollama</guid>
            <pubDate>Tue, 24 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[How to point NemoClaw's inference backend to a local Ollama or vLLM endpoint. Config location, model swap, and what OpenShell still enforces when the cloud is gone.]]></description>
            <category>NemoClaw</category>
            <category>OpenClaw</category>
            <category>OpenShell</category>
            <category>Ollama</category>
            <category>vLLM</category>
            <category>AI Agent</category>
            <category>NVIDIA</category>
            <category>GX10</category>
            <category>Local Inference</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] How to Install NemoClaw on DGX Spark (4 Undocumented Fixes)]]></title>
            <link>https://ai-muninn.com/en/blog/nemoclaw-install-gx10-from-scratch</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/nemoclaw-install-gx10-from-scratch</guid>
            <pubDate>Mon, 23 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[NemoClaw's official installer fails on DGX Spark out of the box. This guide covers the 4 fixes — Node upgrade, npm link, OpenShell tar.gz, cgroupns — to get your first AI agent running in 30 minutes.]]></description>
            <category>NemoClaw</category>
            <category>OpenClaw</category>
            <category>OpenShell</category>
            <category>AI Agent</category>
            <category>NVIDIA</category>
            <category>DGX Spark</category>
            <category>GX10</category>
            <category>GB10</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] What Is NemoClaw? NVIDIA's AI Agent Framework for DGX Spark Explained]]></title>
            <link>https://ai-muninn.com/en/blog/nemoclaw-what-it-is-why-it-exists</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/nemoclaw-what-it-is-why-it-exists</guid>
            <pubDate>Mon, 23 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[NemoClaw bundles OpenClaw + OpenShell + NVIDIA Agent Toolkit into one installer for DGX Spark. Architecture breakdown, what it does, and whether it's worth your time.]]></description>
            <category>NemoClaw</category>
            <category>OpenClaw</category>
            <category>OpenShell</category>
            <category>AI Agent</category>
            <category>NVIDIA</category>
            <category>DGX Spark</category>
            <category>GX10</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] openclaw Real-Time Streaming via Telegram Bot API 9.5 sendMessageDraft]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-telegram-sendmessagedraft-streaming</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-telegram-sendmessagedraft-streaming</guid>
            <pubDate>Sat, 21 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Replacing choppy editMessageText polling with Telegram's sendMessageDraft for live animated output. The patch, the think-block filter, and the optional chaining trap in DM chats.]]></description>
            <category>openclaw</category>
            <category>Telegram</category>
            <category>streaming</category>
            <category>Bot API</category>
            <category>undici</category>
            <category>GLM</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] openclaw + 131K Context: When max_tokens Goes Negative]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-context-budget-negative-maxtokens</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-context-budget-negative-maxtokens</guid>
            <pubDate>Sat, 21 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Connecting openclaw to a 131K context model and hitting 400 max_tokens must be at least 1, got -1292. The context budget math, the config key trap, and the fix.]]></description>
            <category>openclaw</category>
            <category>context window</category>
            <category>vLLM</category>
            <category>gpt-oss</category>
            <category>configuration</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] FP8 KV Cache on GB10: Why Outputs Collapse into Repetition Loops]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-fp8-kvcache-repetition</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-fp8-kvcache-repetition</guid>
            <pubDate>Sat, 21 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Adding --kv-cache-dtype fp8 to a vLLM serve script on GB10 causes outputs to degrade into repetition after ~500 tokens. Root cause: missing calibration data, q_scale defaults to 1.0.]]></description>
            <category>vLLM</category>
            <category>FP8</category>
            <category>KV cache</category>
            <category>GB10</category>
            <category>DGX Spark</category>
            <category>quantization</category>
            <category>SM121</category>
            <category>Qwen 3.5</category>
            <category>Qwen 3.5 35B</category>
        </item>
        <item>
            <title><![CDATA[[Claude Code] claude-agent-sdk vs subprocess: Why Intermediate Turns Disappear]]></title>
            <link>https://ai-muninn.com/en/blog/claude-code-agent-sdk-orchestrator</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/claude-code-agent-sdk-orchestrator</guid>
            <pubDate>Sat, 21 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Building a multi-agent orchestrator with `claude -p` subprocess reveals a silent data loss problem. The SDK fix, session resume, parallel execution, and why setting_sources matters.]]></description>
            <category>Claude Code</category>
            <category>claude-agent-sdk</category>
            <category>multi-agent</category>
            <category>orchestrator</category>
            <category>Python</category>
            <category>asyncio</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] openclaw: Why the Bot Went Silent — Tailscale, IPv6, and a Node.js Happy Eyeballs Trap]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-telegram-ipv6-tailscale-silent-bot</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-telegram-ipv6-tailscale-silent-bot</guid>
            <pubDate>Thu, 19 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[The bot process is running. The token is valid. Messages are being consumed. Nobody is home. A systematic takedown of every wrong hypothesis — and the hidden causal chain that connects Tailscale routing tables to silent sendMessage failures in Node.js.]]></description>
            <category>Node.js</category>
            <category>Tailscale</category>
            <category>IPv6</category>
            <category>undici</category>
            <category>Happy Eyeballs</category>
            <category>Telegram</category>
            <category>Debugging</category>
            <category>Networking</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Running a 120B Model on DGX Spark at 60 tok/s — Zero API Cost, Six Bugs]]></title>
            <link>https://ai-muninn.com/en/blog/part2-gpt-oss-120b-serve-script</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/part2-gpt-oss-120b-serve-script</guid>
            <pubDate>Thu, 19 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[How to get gpt-oss-120B running on a DGX Spark (GB10, SM121) with vLLM. The goal: a 120B model serving a local AI agent at zero API cost. The path: six bugs, one silent env var, and a startup log that tells you everything.]]></description>
            <category>DGX Spark</category>
            <category>SM121</category>
            <category>vLLM</category>
            <category>gpt-oss</category>
            <category>MXFP4</category>
            <category>Blackwell</category>
            <category>LLM Serving</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Qwen3.5-122B Runs. But at 14 tok/s.]]></title>
            <link>https://ai-muninn.com/en/blog/part2-qwen-122b-14-toks-gdn-kernel-gap</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/part2-qwen-122b-14-toks-gdn-kernel-gap</guid>
            <pubDate>Thu, 19 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[After fixing the four SM121 NVFP4 bugs, Qwen3.5-122B boots cleanly and generates correct output. Then you check the speed. 14 tok/s. No flags to fix it. Here's why — and what to wait for.]]></description>
            <category>DGX Spark</category>
            <category>SM121</category>
            <category>Qwen3.5-122B</category>
            <category>vLLM</category>
            <category>NVFP4</category>
            <category>Marlin</category>
            <category>GDN</category>
            <category>LLM Serving</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] openclaw: When the Agent Calls for Help]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-callhelp-spawning-cli-from-agent-loop</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-callhelp-spawning-cli-from-agent-loop</guid>
            <pubDate>Wed, 18 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[How to wire a callhelp tool into a local agent loop so it can spawn Codex CLI mid-reasoning. One permission flag you must set, and why Claude's quota stays mine.]]></description>
            <category>AI Agent</category>
            <category>openclaw</category>
            <category>Codex</category>
            <category>LLM</category>
            <category>Agent Tools</category>
            <category>Local AI</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Why Your DGX Spark Only Says "!!!!!": Debugging NVFP4 on SM121]]></title>
            <link>https://ai-muninn.com/en/blog/part1-why-your-dgx-spark-says-exclamation-marks</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/part1-why-your-dgx-spark-says-exclamation-marks</guid>
            <pubDate>Tue, 17 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[CUTLASS FP4 kernels target SM120 (GB200). On SM121 (GB10, DGX Spark) they run silently and produce garbage. Here's the full diagnostic story — 4 bugs, the row-identical failure signature, and the working fix.]]></description>
            <category>DGX Spark</category>
            <category>SM121</category>
            <category>vLLM</category>
            <category>NVFP4</category>
            <category>MXFP4</category>
            <category>Blackwell</category>
            <category>CUDA</category>
            <category>LLM Serving</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] The Codex-Executor Pattern: Keeping Agent Sessions Small]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-codex-executor-agent-architecture</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-codex-executor-agent-architecture</guid>
            <pubDate>Mon, 16 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Why we stopped having the OpenClaw agent orchestrate multi-step tasks directly, and started spawning Codex subprocesses instead. The pattern that keeps agent context minimal and tasks reliable.]]></description>
            <category>AI Agent</category>
            <category>Claude Code</category>
            <category>Codex</category>
            <category>Agent Architecture</category>
            <category>OpenClaw</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Nemotron-3-Super-120B on a Single GB10: Full Day Debug Log]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-nemotron-120b-vllm</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-nemotron-120b-vllm</guid>
            <pubDate>Fri, 13 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Getting NVIDIA's Nemotron-3-Super-120B-NVFP4 running on an ASUS GX10 (SM121, 128GB). Four SM121-specific pitfalls, the env-var-that-does-nothing, and a working docker command.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>SM121</category>
            <category>Nemotron</category>
            <category>vLLM</category>
            <category>NVFP4</category>
            <category>Blackwell</category>
            <category>LLM Serving</category>
            <category>Nemotron 3 Super</category>
            <category>Nemotron 3 Super 120B</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Ollama's KEEP_ALIVE Is Silently Eating Your vLLM Headroom]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-ollama-vllm-gpu-conflict</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-ollama-vllm-gpu-conflict</guid>
            <pubDate>Sat, 07 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[vLLM OOMed on restart despite 128GB unified memory. Cause: Ollama's KEEP_ALIVE=2h was holding 19-51GB in GPU. Diagnosis command, manual unload fix, and why to set KEEP_ALIVE=0 once vLLM is your primary stack.]]></description>
            <category>vLLM</category>
            <category>Ollama</category>
            <category>GPU Memory</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>LLM Serving</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Don't Add --enable-chunked-prefill to SSM Models]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-chunked-prefill-ssm-trap</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-chunked-prefill-ssm-trap</guid>
            <pubDate>Fri, 06 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Adding --enable-chunked-prefill to a Qwen3.5-35B (SSM+MoE hybrid) dropped throughput from 47 tok/s to 5.7 tok/s. Why SSM recurrence and chunked prefill are fundamentally incompatible.]]></description>
            <category>vLLM</category>
            <category>SSM</category>
            <category>Qwen</category>
            <category>DGX Spark</category>
            <category>LLM Serving</category>
            <category>Performance</category>
        </item>
        <item>
            <title><![CDATA[[vLLM] Qwen3.5-35B at 47 tok/s on DGX Spark: Ollama to vLLM Migration Guide]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-vllm-qwen35-setup</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-vllm-qwen35-setup</guid>
            <pubDate>Thu, 05 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Step-by-step guide: Ollama to vLLM on DGX Spark GB10. Qwen3.5-35B hits 47 tok/s with TTFT dropping from 3s to 0.12s. Covers 6 real gotchas including SSM + chunked prefill trap and GPU memory conflicts.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>vLLM</category>
            <category>Ollama</category>
            <category>Qwen 3.5</category>
            <category>Docker</category>
            <category>Blackwell</category>
            <category>AI Agent</category>
            <category>Qwen 3.5 35B</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] Zero API Cost: Running OpenClaw on DGX Spark + Mac Mini]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-dgx-spark-local-ai-agent</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-dgx-spark-local-ai-agent</guid>
            <pubDate>Thu, 05 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Full stack local AI agent: Mac Mini M4 as the always-on gateway, GX10 for inference, Telegram as the UI. No subscriptions, no cloud APIs. Six deployment lessons from the trenches.]]></description>
            <category>OpenClaw</category>
            <category>AI Agent</category>
            <category>DGX Spark</category>
            <category>Mac Mini</category>
            <category>Self-Hosted</category>
            <category>Ollama</category>
            <category>SearXNG</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] Pure MoE vs SSM Hybrid: Context Decay and Why It Matters for Agents]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-moe-ssm-context-decay</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-moe-ssm-context-decay</guid>
            <pubDate>Sun, 01 Mar 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[GLM-4.7-Flash hits 57.8 tok/s on short context but drops to 42 tok/s at 8K. Qwen3.5-35B SSM hybrid: 56 tok/s at short, 56 tok/s at 8K. Why agents with long system prompts should care about this difference.]]></description>
            <category>Benchmark</category>
            <category>SSM</category>
            <category>MoE</category>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>LLM Serving</category>
            <category>AI Agents</category>
        </item>
        <item>
            <title><![CDATA[[Dev Workflow] I Made Two AIs Argue. The Disagreements Are the Point.]]></title>
            <link>https://ai-muninn.com/en/blog/claude-code-debate-system</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/claude-code-debate-system</guid>
            <pubDate>Thu, 26 Feb 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A custom /debate command that pits Codex CLI against Gemini CLI on architecture, code, and decisions. Different training data, different blind spots — and the disagreements between them are usually the most useful output.]]></description>
            <category>Dev Workflow</category>
            <category>Claude Code</category>
            <category>Gemini</category>
            <category>Codex</category>
            <category>Multi-AI</category>
            <category>Code Review</category>
        </item>
        <item>
            <title><![CDATA[[Claude Code] Testing iOS Apps with Claude Code: 81% Context Reduction]]></title>
            <link>https://ai-muninn.com/en/blog/claude-code-ios-testing-bpstracker</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/claude-code-ios-testing-bpstracker</guid>
            <pubDate>Thu, 26 Feb 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[How I replaced screenshot-heavy iOS test runs with ui_describe_all-first testing in Claude Code, cutting context usage by 81% for BPS Tracker. Plus Fastlane integration for App Store automation.]]></description>
            <category>Claude Code</category>
            <category>iOS</category>
            <category>Swift</category>
            <category>Testing</category>
            <category>Fastlane</category>
            <category>BPS Tracker</category>
        </item>
        <item>
            <title><![CDATA[[AI Agent] OpenClaw Config Hot-Reload: No Restart Needed]]></title>
            <link>https://ai-muninn.com/en/blog/openclaw-config-hot-reload</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/openclaw-config-hot-reload</guid>
            <pubDate>Wed, 25 Feb 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Spent weeks restarting the OpenClaw gateway for every config change. Then discovered the file watcher. What hot-reloads instantly, what still needs a restart, and how to tell auth failures from transient network errors.]]></description>
            <category>AI Agent</category>
            <category>OpenClaw</category>
            <category>Configuration</category>
            <category>Developer Workflow</category>
        </item>
        <item>
            <title><![CDATA[[Claude Code] I Wrote MANDATORY. The AI Ignored It.]]></title>
            <link>https://ai-muninn.com/en/blog/claude-code-mandatory-instructions</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/claude-code-mandatory-instructions</guid>
            <pubDate>Thu, 19 Feb 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[A Claude Code config rule marked MANDATORY was skipped twice in one session. Here's the root cause — three architectural reasons why emphasis doesn't work — and three system-level solutions that do.]]></description>
            <category>Claude Code</category>
            <category>AI Agents</category>
            <category>Prompt Engineering</category>
            <category>Systems Design</category>
            <category>Developer Workflow</category>
        </item>
        <item>
            <title><![CDATA[[Benchmark] 8 Models on DGX Spark: Finding the Best Stack for AI Agents]]></title>
            <link>https://ai-muninn.com/en/blog/dgx-spark-ollama-benchmark-8-models</link>
            <guid isPermaLink="false">https://ai-muninn.com/en/blog/dgx-spark-ollama-benchmark-8-models</guid>
            <pubDate>Thu, 19 Feb 2026 00:00:00 GMT</pubDate>
            <description><![CDATA[Benchmarking 8 local LLMs on NVIDIA GB10 (128GB unified memory) across 7 task categories. Quantization surprises, a 120B model that fails at JSON, and thinking models that spend their entire budget thinking.]]></description>
            <category>DGX Spark</category>
            <category>GB10</category>
            <category>Ollama</category>
            <category>Benchmark</category>
            <category>LLM</category>
            <category>AI Agent</category>
            <category>Blackwell</category>
        </item>
    </channel>
</rss>