<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>Harshit Malik</title>
    <link>https://harshitmalik.dev</link>
    <description>Harshit Malik is an AI engineer working on LLM inference and multi-agent systems in production: KV cache arithmetic, serving economics, and agent orchestration under real traffic.</description>
    <language>en</language>
    <atom:link href="https://harshitmalik.dev/feed.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Inside the KV Cache: The Life of a Gigabyte</title>
      <link>https://harshitmalik.dev/blog/inside-the-kv-cache</link>
      <guid isPermaLink="true">https://harshitmalik.dev/blog/inside-the-kv-cache</guid>
      <pubDate>Tue, 29 Sep 2026 00:00:00 GMT</pubDate>
      <description>From gpu_memory_utilization and max_model_len to weights, activations, CUDA graphs and the KV pool: where every gigabyte on the card actually goes.</description>
    </item>
  </channel>
</rss>
