<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:image="http://www.google.com/schemas/sitemap-image/1.1">
<url>
<loc>https://inferenceengineering.co</loc>
<lastmod>2026-09-18T12:58:36.172Z</lastmod>
<changefreq>daily</changefreq>
<priority>1</priority>
</url>
<url>
<loc>https://inferenceengineering.co/sections/features</loc>
<lastmod>2026-09-18T06:51:30.742Z</lastmod>
<changefreq>weekly</changefreq>
<priority>0.5</priority>
</url>
<url>
<loc>https://inferenceengineering.co/sections/kv-cache-systems</loc>
<lastmod>2026-09-18T12:58:36.172Z</lastmod>
<changefreq>weekly</changefreq>
<priority>0.5</priority>
</url>
<url>
<loc>https://inferenceengineering.co/authors/omar-ford</loc>
<lastmod>2026-09-18T12:58:36.172Z</lastmod>
<changefreq>weekly</changefreq>
<priority>0.4</priority>
</url>
<url>
<loc>https://inferenceengineering.co/authors/rohan-reyes</loc>
<lastmod>2026-09-18T12:58:36.172Z</lastmod>
<changefreq>weekly</changefreq>
<priority>0.4</priority>
</url>
<url>
<loc>https://inferenceengineering.co/posts/lru-vs-gdsf-vs-recency-weighted-eviction-for-long-context-kv-cache</loc>
<image:image>
<image:loc>https://mcxtywzsmeezbwapdahq.supabase.co/storage/v1/object/public/canvas-images/1ea5b494-0bbf-4b2b-b155-3eecc768ddf1/canvas/f21f0e25-0dd0-49e2-b6aa-ab4ef4acca9d/generated/219447c9-a215-4cb7-b148-5166c56ae5dd.jpeg</image:loc>
</image:image>
<lastmod>2026-09-18T12:58:36.172Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://inferenceengineering.co/posts/prefill-and-decode-disaggregation-in-llm-serving</loc>
<image:image>
<image:loc>https://mcxtywzsmeezbwapdahq.supabase.co/storage/v1/object/public/canvas-images/1ea5b494-0bbf-4b2b-b155-3eecc768ddf1/canvas/f21f0e25-0dd0-49e2-b6aa-ab4ef4acca9d/generated/572be82f-2b4c-4f93-a481-e198d658563d.jpeg</image:loc>
</image:image>
<lastmod>2026-09-18T06:51:30.742Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
</urlset>
