{"componentChunkName":"component---src-templates-tag-page-js","path":"/tags/quantization/","result":{"data":{"site":{"siteMetadata":{"title":"M.Hassan Ahmed","author":"Hassan11196"}},"allMarkdownRemark":{"totalCount":1,"edges":[{"node":{"excerpt":"LLM Quantization: INT8, GPTQ, and AWQ Explained In an earlier post about the KV cache I made a point of separating two kinds of GPU memory…","fields":{"slug":"/2026-07-26-llm-quantization-int8-gptq-awq/"},"frontmatter":{"date":"2026-07-26T00:00:00.000Z","title":"LLM Quantization: INT8, GPTQ, and AWQ Explained","description":"The weights don't fit on the GPU you have. How LLM quantization shrinks them to INT8 or INT4, why GPTQ and AWQ beat naive rounding, and where it breaks.","tags":["AI","LLM","GPU","Quantization","Inference"],"thumbnail":null}}}]}},"pageContext":{"tag":"Quantization"}},"staticQueryHashes":["32046230"]}