{"componentChunkName":"component---src-templates-tag-page-js","path":"/tags/transformers/","result":{"data":{"site":{"siteMetadata":{"title":"M.Hassan Ahmed","author":"Hassan11196"}},"allMarkdownRemark":{"totalCount":1,"edges":[{"node":{"excerpt":"How Grouped-Query Attention Shrinks the KV Cache In an earlier post on the KV cache I listed grouped-query attention as the first lever to…","fields":{"slug":"/2026-08-15-grouped-query-attention-kv-cache/"},"frontmatter":{"date":"2026-08-15T00:00:00.000Z","title":"How Grouped-Query Attention Shrinks the KV Cache","description":"Grouped-query attention shares key/value heads across query heads to cut the KV cache. How GQA sits between MHA and MQA, with the memory math and code.","tags":["AI","LLM","GPU","Inference","Transformers"],"thumbnail":null}}}]}},"pageContext":{"tag":"Transformers"}},"staticQueryHashes":["32046230"]}