{"componentChunkName":"component---src-templates-tag-page-js","path":"/tags/serving/","result":{"data":{"site":{"siteMetadata":{"title":"M.Hassan Ahmed","author":"Hassan11196"}},"allMarkdownRemark":{"totalCount":1,"edges":[{"node":{"excerpt":"Continuous Batching for LLM Inference Here is a number that looks wrong the first time you see it. You put a GPU behind an LLM, send it one…","fields":{"slug":"/2026-07-21-continuous-batching-llm-inference/"},"frontmatter":{"date":"2026-07-21T00:00:00.000Z","title":"Continuous Batching for LLM Inference","description":"Static batching leaves the GPU idle when requests finish at different steps. How continuous batching schedules LLM inference per token to raise throughput.","tags":["AI","LLM","GPU","Inference","Serving"],"thumbnail":null}}}]}},"pageContext":{"tag":"Serving"}},"staticQueryHashes":["32046230"]}