{"componentChunkName":"component---src-templates-tag-page-js","path":"/tags/v-llm/","result":{"data":{"site":{"siteMetadata":{"title":"M.Hassan Ahmed","author":"Hassan11196"}},"allMarkdownRemark":{"totalCount":1,"edges":[{"node":{"excerpt":"How PagedAttention Powers vLLM’s KV Cache Here is a result that surprises people the first time they hit it: you load a 13B model onto a GPU…","fields":{"slug":"/2026-08-16-pagedattention-vllm-kv-cache/"},"frontmatter":{"date":"2026-08-16T00:00:00.000Z","title":"How PagedAttention Powers vLLM's KV Cache","description":"A self-hosted LLM server wastes most of its GPU memory to KV cache fragmentation. Here is how PagedAttention in vLLM pages the cache like an OS.","tags":["AI","LLM","GPU","Inference","vLLM"],"thumbnail":null}}}]}},"pageContext":{"tag":"vLLM"}},"staticQueryHashes":["32046230"]}