{"componentChunkName":"component---src-templates-tag-page-js","path":"/tags/performance/","result":{"data":{"site":{"siteMetadata":{"title":"M.Hassan Ahmed","author":"Hassan11196"}},"allMarkdownRemark":{"totalCount":2,"edges":[{"node":{"excerpt":"Speculative Decoding: Faster LLM Inference Text generation from a transformer is stubbornly sequential. To produce the tenth token the model…","fields":{"slug":"/2026-08-25-speculative-decoding-faster-llm-inference/"},"frontmatter":{"date":"2026-08-25T00:00:00.000Z","title":"Speculative Decoding: Faster LLM Inference","description":"Speculative decoding uses a small draft model to guess tokens a big model verifies in one parallel pass, cutting LLM latency with no change to output.","tags":["AI","LLM","GPU","Inference","Performance"],"thumbnail":null}}},{"node":{"excerpt":"How Speculative Decoding Speeds Up LLM Inference Watch a large language model generate text on a GPU and something looks wrong. The card is…","fields":{"slug":"/2026-07-24-speculative-decoding-llm-inference/"},"frontmatter":{"date":"2026-07-24T00:00:00.000Z","title":"How Speculative Decoding Speeds Up LLM Inference","description":"Speculative decoding uses a small draft model to guess tokens a big model verifies in one pass, cutting LLM latency 2-3x without changing the output.","tags":["AI","LLM","GPU","Inference","Performance"],"thumbnail":null}}}]}},"pageContext":{"tag":"Performance"}},"staticQueryHashes":["32046230"]}