<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Mcp on doug.sh</title>
    <link>https://doug.sh/tags/mcp/</link>
    <description>Recent content in Mcp on doug.sh</description>
    <generator>Hugo</generator>
    <language>en-us</language>
    <lastBuildDate>Tue, 15 Sep 2026 00:00:00 +0000</lastBuildDate>
    <atom:link href="https://doug.sh/tags/mcp/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Keeping vLLM&#39;s Prefix Cache Warm Between Agent Turns</title>
      <link>https://doug.sh/posts/vllm-kv-cache-agents/</link>
      <pubDate>Tue, 15 Sep 2026 00:00:00 +0000</pubDate>
      <guid>https://doug.sh/posts/vllm-kv-cache-agents/</guid>
      <description>&lt;h2 id=&#34;from-55-to-95-cached&#34;&gt;From 55% to 95% cached &lt;a href=&#34;#from-55-to-95-cached&#34; class=&#34;anchor&#34;&gt;🔗&lt;/a&gt;&lt;/h2&gt;&lt;p&gt;I&amp;rsquo;ve been playing with a few different ways to host Qwen3.8 locally. I&amp;rsquo;m aiming for something that&#xA;can replace Claude Code for most of my tasks. Right now the 27B runs on two RTX 3090s under vLLM, and&#xA;it&amp;rsquo;s usable, but the first day was rough. In the morning the wait before the first word averaged&#xA;about half a minute, and some replies took several minutes. The time went into rereading large parts&#xA;of every prompt from scratch, because the cached copy from the previous turn had been thrown away or&#xA;no longer matched. By the evening the server was keeping almost everything from one turn to the next.&lt;/p&gt;</description>
    </item>
  </channel>
</rss>
