<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
<url>
  <loc>https://www.ligne8.studio/blog/modeles-plateformes/amazon-sagemaker-hyperpod-optimise-l-inference-llm-avec-un-cache-kv-hierarchise-part-270439</loc>
  <news:news>
    <news:publication>
      <news:name>ligne8 Studio</news:name>
      <news:language>fr</news:language>
    </news:publication>
    <news:publication_date>2026-08-15T15:02:18.117Z</news:publication_date>
    <news:title>Amazon SageMaker HyperPod optimise l’inférence LLM avec un cache KV hiérarchisé partagé</news:title>
    <news:keywords>LLM, inférence, cache KV, Amazon SageMaker, GPU, NVMe, Curvine, cloud</news:keywords>
  </news:news>
</url>
</urlset>