{
    "archive_path": "archive/1748840674.884257",
    "base_url": "www.seangoedecke.com/inference-batching-and-deepseek",
    "basename": "",
    "bookmarked_date": "2025-06-02 05:04",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/www.seangoedecke.com/inference-batching-and-deepseek",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=www.seangoedecke.com",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "www.seangoedecke.com",
    "downloaded_at": "2025-06-02T05:04:40.347007+00:00",
    "downloaded_datestr": "2025-06-02 05:04",
    "extension": "",
    "hash": "1WH9G3H2Z3Q93APJNXD4",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://www.seangoedecke.com/inference-batching-and-deepseek/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-06-02T05:05:31.897210+00:00",
                "index_texts": null,
                "output": "ArchiveError: Failed to find \"content-location\" URL header in Archive.org response.",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:05:30.000409+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://www.seangoedecke.com/inference-batching-and-deepseek/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2025-06-02T05:05:05.910229+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:04:48.972790+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=www.seangoedecke.com"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-06-02T05:04:44.185269+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:04:40.645540+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.seangoedecke.com/inference-batching-and-deepseek/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-06-02T05:04:44.273048+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:04:44.220144+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2025-06-02T05:05:23.611117+00:00",
                "index_texts": [
                    "(/favicon-32x32.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/manifest.webmanifest) (/icons/icon-48x48.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-72x72.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-96x96.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-144x144.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-192x192.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-256x256.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-384x384.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-512x512.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) Why DeepSeek is cheap at scale but expensive to run locally | sean goedecke (seangoedecke.com RSS feed) (/rss.xml) (seangoedecke.com RSS feed) (/feed.xml) (seangoedecke.com RSS feed) (/atom.xml) (/webpack-runtime-1c490e24812c971f677c.js) (/framework-4ce5382688c7c66a6d68.js) (/styles-806f4b82980a743e7980.js) (/app-5ea5ab43ea6709b618e0.js) (/commons-8bba03b02b475bd1322b.js) (/component---src-templates-blog-post-js-1d517f24415df614196c.js) (/page-data/inference-batching-and-deepseek/page-data.json) (/page-data/sq/d/1146911855.json) (/page-data/sq/d/3000541721.json) (/page-data/app-data.json)  (/) sean goedecke   Why DeepSeek is cheap at scale but expensive to run locally  Why is DeepSeek-V3 supposedly fast and cheap to serve at scale, but too slow and expensive to run locally? Why are some AI models slow to respond but fast once they get going? AI inference providers often talk about a fundamental tradeoff between throughput and latency : for any given model, you can either serve it at high-throughput high-latency, or low-throughput low-latency. In fact, some models are so naturally GPU-inefficient that in practice they must be served at high-latency to have any workable throughput at all (for instance, DeepSeek-V3). This tradeoff comes from the batch size the inference provider chooses for the model: not batching inference inside an individual request1  , but batching inference across tens or hundreds of concurrent user requests. It\u2019s a peculiar feature of transformer-based LLMs that computing a batch of completions at the same time is almost as fast as computing a single completion. Why is that? What is batch inference? GPUs are good at doing big matrix multiplications (GEMMs, or \u201cgeneral matrix multiplications\u201d). Say you have a single token that you want to pass through a model (i.e. by multiplying against all its weights - other architecture details aren\u2019t relevant). You express that as a vector that matches the dimension (or hidden size) of the model (i.e. 1 x the width of its big weights matrices) and multiply it through. That\u2019s 1 GEMM. But if you want to pass ten tokens through in a batch, that\u2019s still only one GEMM, because you can stack the tokens into one matrix (10 x the model dimension). That\u2019s a lot faster than doing ten slightly smaller GEMMs. So an inference server implementation might look something like this: A request comes in with a prompt That prompt is pre-filled (passed through attention - we\u2019ll see later how that can be batched as well2  ), forming a KV cache and a token-sized matrix (1 x model-size) that will eventually become the predicted token3   That token-sized matrix goes into a queue A GPU server pulls batches (e.g. of 128) off that queue, stacks them up into a 128 x model-size matrix, and multiplies them through the feed-forward model weights The end result is then split into 128 separate tokens The one for the original request is streamed back to the user Assuming that token isn\u2019t an end-of-sequence token, return to step 2 to continue generating the next token in the response  Note that the server decides how big a batch size to pull. It\u2019s a tradeoff between throughput and latency. If you do no batching and just process tokens one by one, no user ever waits in a queue (step 3 above), so latency is low (assuming you have enough GPUs). However, if you do a lot of batching, latency is high because users will be waiting until the batch size fills up, but throughput will be much higher because the GPUs are being used more efficiently. Why are GPUs faster at multiplying large matrices once than small matrices many times? Two reasons. First, there\u2019s some overhead involved in issuing each command to the GPU, and one big multiplication can be launched with a single command. Second, each new GPU command involves fetching weights from memory, which can be expensive for large weights. If you run lots of small GEMMs, you can end up spending most of your time shipping weights in and out of memory instead of computing. Why are some models tuned for high batch sizes? Typically an inference server will have a \u201ccollection window\u201d where user requests come in and are queued. Chat servers typically aim for 5-10ms, but very high-batch backends might go as wide as 200ms. If a new request comes in at the start of the window, it might wait the entire window duration before being processed4  . When the window closes, all the queued requests are batched up (i.e. all the 1xmodel-size matrices are concatenated into a single 128xmodel-size matrix) and that batch is sent through the pipeline. Running a batch like this is sometimes called a \u201ctick\u201d. As the explanation above suggests, you can run any model at any batch size. There\u2019s nothing inherently about the batching process that would rule out some types of model. However, it is possible to build a model so GPU-inefficiently that it effectively needs batching in order to be practical. Why mixture of experts requires higher batch sizes For instance, take a mixture-of-experts model (like DeepSeek-V3 or supposedly the original GPT-4). You can get a strong model by training it to have hundreds and hundreds of \u201cexperts\u201d: separate blocks of feed-forward weights, from which a routing layer picks a subset that\u2019s used on each token. But a model like this is really GPU-inefficient. We can see why: GPUs want to do a small number of really big matrix multiplications, but if you have many experts you\u2019re forced into many small multiplications. Unless you do your inference in batches, that\u2019s going to mean low throughput. Let\u2019s think through how a \u201ccollection window\u201d of 5ms and 200ms would perform for a large mixture-of-experts model. Suppose you pick up ten user requests in that 5ms window. If you have many experts, some experts might end up only running against one or two tokens (i.e. the batch size for each expert will be much lower than the total set of requests you\u2019ve picked up in your window). If, however, you wait for 200ms and pick up 4000 user requests, you are much more likely to saturate all your experts. At the cost of some latency, you\u2019re making sure that your GEMMs are large and your GPUs are constantly utilized at their maximum capacity. Why large pipelines require high batch sizes to avoid pipeline bubbles For large models, it can be a challenge to keep the GPUs active at all. Large models typically have many transformer layers: i.e. hundreds of matrices of weights that make up the feed-forward network. The only way to do fast inference here is to pipeline those layers by having one GPU handle the first ten layers, another handle the next ten, and so on. Otherwise you just won\u2019t be able to fit all the weights in a single GPU\u2019s memory, so you\u2019ll spend a ton of time swapping weights in and out of memory and it\u2019ll end up being really slow. During inference, each token (typically in a \u201cmicro batch\u201d of a few tens of tokens each) passes sequentially through that pipeline of GPUs. How efficient your pipeline is depends on the number of layers you have and the size of your collection window. When you\u2019re processing the tokens in a window during a \u201ctick\u201d, you\u2019ll get some idle GPUs at the start (because GPUs in later layers won\u2019t have anything to work on yet) and some more idle GPUs at the end (when there\u2019s no more tokens in the queue, GPUs in early layers will have to wait for the next \u201ctick\u201d). These periods of idleness are sometimes called \u201cwarmup\u201d and \u201cdrain\u201d. If you have many small windows, you\u2019re going to spend more GPU time in warmup and drain than if you have fewer large windows. By picking your window size, you\u2019re thus directly trading off between throughput and latency. If you have a ton of layers and your collection window is really short, you might sometimes end up with fewer tokens to process than layers. This is called a \u201cpipeline bubble\u201d - in effect the \u201cdrain\u201d stage starts earlier than usual. You can\u2019t eliminate warmup and drain (for reasons discussed below, inference has to operate in sequential \u201cticks\u201d), but you can eliminate pipeline bubbles by making your collection window long enough. Pipeline bubbles can be absolutely brutal for model throughput, so inference providers always set their windows wide enough to avoid them. That adds noticeable latency for models with many layers. Can\u2019t you just keep the queue full? Why couldn\u2019t inference providers eliminate warmup and drain entirely by keeping the GPU queue full of tokens? In other words, couldn\u2019t you do away with ticks altogether and just keep the token micro-batches flowing? Of course each user\u2019s inference has to be sequential (since you can\u2019t start generating the next token until the current token is done), but large inference providers should have enough concurrent traffic to keep the queue full of separate user requests. I\u2019ll confess I struggle to see why this shouldn\u2019t be possible in theory. As far as I can tell the practical barrier is how the attention step is batched: if you want to batch up attention GEMMs, they need to all be the same shape (i.e. the same number of prior tokens in the sequence). So you have to run groups of the same shape at the same time, instead of being able to just maintain a single queue. There\u2019s at least (https://arxiv.org/abs/2403.02310) some public research on this front, but I wouldn\u2019t be surprised if there were more clever tricks for doing this that I haven\u2019t seen. Another idea: if you need ticks for the attention step, why not just have a tick-based attention inference system and a more efficient continuous system for the FFN? As I understand it, the reason is memory overhead : Since the attention output is needed for the FFN, you\u2019d need to have some place in-memory to park it while it waits for its slot in the FFN queue, which would quickly become too expensive. Modern inference stacks are able to combine the attention and FFN step into a couple of large GEMMs in a single \u201coperation\u201d. If you\u2019re doing these on different GPUs, you have to run different operations and shuttle the weights in and out of memory.  Summary GPUs are most efficient on large GEMMs, so stacking many tokens into a single matrix multiply gives far higher token throughput than processing them one-by-one During decoding, attention can only be batched for tokens at the same step , forcing schedulers to run in short \u201cticks\u201d. How many tokens you pack into a single \u201ctick\u201d (i.e. how long you wait to collect tokens) is your batch size These are tokens from different users . You can\u2019t batch tokens from the same user because you need previous tokens to generate the next one, so batching requires a high volume of traffic from different users   Bigger batches raise latency because user tokens might be waiting up to 200ms before the batch is full enough to run, but they boost throughput by allowing larger (and thus more efficient) GEMMs in the feed-forward step Models with many layers (e.g. long pipelines) need larger batches to avoid pipeline bubbles (by ensuring each tick contains more batches than pipeline steps)  Mixture-of-Experts models need to be served with high-latency to be efficient: each expert sees only the tokens routed to it, so you need larger global batches to keep every expert busy.  Inference providers pick a batch size/window that clears pipeline bubbles and saturates experts. High batch sizes buy you more throughput at the cost of higher latency as tokens wait to fill up the tick Some models (like DeepSeek\u2019s) that are mixture-of-experts with many layers thus require large batch sizes and high latency, otherwise throughput drops off a cliff. That\u2019s why it\u2019s commonly said that you can\u2019t easily run DeepSeek for personal use: because with a single user running one inference at a time, it runs at very low efficiency/throughput The fact that OpenAI and Anthropic\u2019s models are quick to respond suggests that either: Their models have a more efficient architecture (non-MoE, fewer layers), or OpenAI/Anthropic have some very clever tricks for serving inference, or they\u2019re paying through the nose for way more GPUs than they strictly need    edit: This was posted on (https://news.ycombinator.com/item?id=44149238) Hacker News with a bunch of comments. I kind of wish I\u2019d titled this post differently - it\u2019s not really about running models on your own computer. It\u2019s about running the models for personal use, assuming you have all the GPUs (i.e. the batching/throughput tradeoff). 1  One commonly-observed strength of transformers is that they can batch prefill within a single user request. When you pass them a long prompt, they can process that prompt all at once because of how the attention mechanism works. Previous recurrent models had to go token-by-token, which was much slower (because it involved many more GEMMs). This has nothing to do with the kind of batching I\u2019m talking about in this post . I\u2019m talking about how you can efficiently batch inference across many different user requests once the prefilling is complete. 2  This can also be batched, so long as you\u2019re only batching attention operations with the same number of tokens in the sequence (i.e. every sequence predicting the fourth token can be batched together). Otherwise the size of the KV cache matrices are different, so you can\u2019t easily combine them into a single batch. More on that later. 3  Technically it\u2019s not a token being generated, but the \u201clogits\u201d (a probability distribution across all possible tokens). I\u2019ll say \u201ctoken\u201d here and later on to keep it simpler. 4  Note that in practice modern inference stacks will use \u201ccontinuous batching\u201d, where a batch is sent off as soon as it\u2019s full instead of waiting for the entire length of the fixed time window. However, the inference is still done in batches, to the core tradeoff between throughput and latency is the same.  If you liked this post, consider (https://buttondown.com/seangoedecke) subscribing to email updates about my new posts. June 1, 2025\u2502 Tags: (/tags/ai/) ai , (/tags/explainers/) explainers , (/tags/deepseek/) deepseek  (/) posts \u2502 (https://buttondown.com/seangoedecke) subscribe \u2502 (https://www.linkedin.com/in/sean-goedecke-5495a7137/) linkedin \u2502 (/rss.xml) rss \u2502 (/book) read my book about software engineering     (/not-your-codebase/) \u2190 It's not your codebase            "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:05:23.593130+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://www.seangoedecke.com/inference-batching-and-deepseek/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2025-06-02T05:05:29.945080+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:05:25.634647+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://www.seangoedecke.com/inference-batching-and-deepseek/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2025-06-02T05:05:23.564775+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:05:21.325574+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmp31exoj8n",
                    "https://www.seangoedecke.com/inference-batching-and-deepseek/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2025-06-02T05:05:09.950533+00:00",
                "index_texts": [
                    "Why is DeepSeek-V3 supposedly fast and cheap to serve at scale, but too slow and expensive to run locally? Why are some AI models slow to respond but fast once they get going?\nAI inference providers often talk about a fundamental tradeoff between throughput and latency: for any given model, you can either serve it at high-throughput high-latency, or low-throughput low-latency. In fact, some models are so naturally GPU-inefficient that in practice they must be served at high-latency to have any workable throughput at all (for instance, DeepSeek-V3).\nThis tradeoff comes from the batch size the inference provider chooses for the model: not batching inference inside an individual request1, but batching inference across tens or hundreds of concurrent user requests. It\u2019s a peculiar feature of transformer-based LLMs that computing a batch of completions at the same time is almost as fast as computing a single completion. Why is that?\nWhat is batch inference?\nGPUs are good at doing big matrix multiplications (GEMMs, or \u201cgeneral matrix multiplications\u201d). Say you have a single token that you want to pass through a model (i.e. by multiplying against all its weights - other architecture details aren\u2019t relevant). You express that as a vector that matches the dimension (or hidden size) of the model (i.e. 1 x the width of its big weights matrices) and multiply it through. That\u2019s 1 GEMM. But if you want to pass ten tokens through in a batch, that\u2019s still only one GEMM, because you can stack the tokens into one matrix (10 x the model dimension). That\u2019s a lot faster than doing ten slightly smaller GEMMs. So an inference server implementation might look something like this:\n\nA request comes in with a prompt\nThat prompt is pre-filled (passed through attention - we\u2019ll see later how that can be batched as well2), forming a KV cache and a token-sized matrix (1 x model-size) that will eventually become the predicted token3\nThat token-sized matrix goes into a queue\nA GPU server pulls batches (e.g. of 128) off that queue, stacks them up into a 128 x model-size matrix, and multiplies them through the feed-forward model weights\nThe end result is then split into 128 separate tokens\nThe one for the original request is streamed back to the user\nAssuming that token isn\u2019t an end-of-sequence token, return to step 2 to continue generating the next token in the response\n\nNote that the server decides how big a batch size to pull. It\u2019s a tradeoff between throughput and latency. If you do no batching and just process tokens one by one, no user ever waits in a queue (step 3 above), so latency is low (assuming you have enough GPUs). However, if you do a lot of batching, latency is high because users will be waiting until the batch size fills up, but throughput will be much higher because the GPUs are being used more efficiently.\nWhy are GPUs faster at multiplying large matrices once than small matrices many times? Two reasons. First, there\u2019s some overhead involved in issuing each command to the GPU, and one big multiplication can be launched with a single command. Second, each new GPU command involves fetching weights from memory, which can be expensive for large weights. If you run lots of small GEMMs, you can end up spending most of your time shipping weights in and out of memory instead of computing.\nWhy are some models tuned for high batch sizes?\nTypically an inference server will have a \u201ccollection window\u201d where user requests come in and are queued. Chat servers typically aim for 5-10ms, but very high-batch backends might go as wide as 200ms. If a new request comes in at the start of the window, it might wait the entire window duration before being processed4. When the window closes, all the queued requests are batched up (i.e. all the 1xmodel-size matrices are concatenated into a single 128xmodel-size matrix) and that batch is sent through the pipeline. Running a batch like this is sometimes called a \u201ctick\u201d.\nAs the explanation above suggests, you can run any model at any batch size. There\u2019s nothing inherently about the batching process that would rule out some types of model. However, it is possible to build a model so GPU-inefficiently that it effectively needs batching in order to be practical.\nWhy mixture of experts requires higher batch sizes\nFor instance, take a mixture-of-experts model (like DeepSeek-V3 or supposedly the original GPT-4). You can get a strong model by training it to have hundreds and hundreds of \u201cexperts\u201d: separate blocks of feed-forward weights, from which a routing layer picks a subset that\u2019s used on each token. But a model like this is really GPU-inefficient. We can see why: GPUs want to do a small number of really big matrix multiplications, but if you have many experts you\u2019re forced into many small multiplications. Unless you do your inference in batches, that\u2019s going to mean low throughput.\nLet\u2019s think through how a \u201ccollection window\u201d of 5ms and 200ms would perform for a large mixture-of-experts model. Suppose you pick up ten user requests in that 5ms window. If you have many experts, some experts might end up only running against one or two tokens (i.e. the batch size for each expert will be much lower than the total set of requests you\u2019ve picked up in your window). If, however, you wait for 200ms and pick up 4000 user requests, you are much more likely to saturate all your experts. At the cost of some latency, you\u2019re making sure that your GEMMs are large and your GPUs are constantly utilized at their maximum capacity.\nWhy large pipelines require high batch sizes to avoid pipeline bubbles\nFor large models, it can be a challenge to keep the GPUs active at all. Large models typically have many transformer layers: i.e. hundreds of matrices of weights that make up the feed-forward network. The only way to do fast inference here is to pipeline those layers by having one GPU handle the first ten layers, another handle the next ten, and so on. Otherwise you just won\u2019t be able to fit all the weights in a single GPU\u2019s memory, so you\u2019ll spend a ton of time swapping weights in and out of memory and it\u2019ll end up being really slow. During inference, each token (typically in a \u201cmicro batch\u201d of a few tens of tokens each) passes sequentially through that pipeline of GPUs.\nHow efficient your pipeline is depends on the number of layers you have and the size of your collection window. When you\u2019re processing the tokens in a window during a \u201ctick\u201d, you\u2019ll get some idle GPUs at the start (because GPUs in later layers won\u2019t have anything to work on yet) and some more idle GPUs at the end (when there\u2019s no more tokens in the queue, GPUs in early layers will have to wait for the next \u201ctick\u201d). These periods of idleness are sometimes called \u201cwarmup\u201d and \u201cdrain\u201d. If you have many small windows, you\u2019re going to spend more GPU time in warmup and drain than if you have fewer large windows. By picking your window size, you\u2019re thus directly trading off between throughput and latency.\nIf you have a ton of layers and your collection window is really short, you might sometimes end up with fewer tokens to process than layers. This is called a \u201cpipeline bubble\u201d - in effect the \u201cdrain\u201d stage starts earlier than usual. You can\u2019t eliminate warmup and drain (for reasons discussed below, inference has to operate in sequential \u201cticks\u201d), but you can eliminate pipeline bubbles by making your collection window long enough. Pipeline bubbles can be absolutely brutal for model throughput, so inference providers always set their windows wide enough to avoid them. That adds noticeable latency for models with many layers.\nCan\u2019t you just keep the queue full?\nWhy couldn\u2019t inference providers eliminate warmup and drain entirely by keeping the GPU queue full of tokens? In other words, couldn\u2019t you do away with ticks altogether and just keep the token micro-batches flowing? Of course each user\u2019s inference has to be sequential (since you can\u2019t start generating the next token until the current token is done), but large inference providers should have enough concurrent traffic to keep the queue full of separate user requests.\nI\u2019ll confess I struggle to see why this shouldn\u2019t be possible in theory. As far as I can tell the practical barrier is how the attention step is batched: if you want to batch up attention GEMMs, they need to all be the same shape (i.e. the same number of prior tokens in the sequence). So you have to run groups of the same shape at the same time, instead of being able to just maintain a single queue. There\u2019s at least some public research on this front, but I wouldn\u2019t be surprised if there were more clever tricks for doing this that I haven\u2019t seen.\nAnother idea: if you need ticks for the attention step, why not just have a tick-based attention inference system and a more efficient continuous system for the FFN? As I understand it, the reason is memory overhead:\n\nSince the attention output is needed for the FFN, you\u2019d need to have some place in-memory to park it while it waits for its slot in the FFN queue, which would quickly become too expensive.\nModern inference stacks are able to combine the attention and FFN step into a couple of large GEMMs in a single \u201coperation\u201d. If you\u2019re doing these on different GPUs, you have to run different operations and shuttle the weights in and out of memory.\n\nSummary\n\nGPUs are most efficient on large GEMMs, so stacking many tokens into a single matrix multiply gives far higher token throughput than processing them one-by-one\n\nDuring decoding, attention can only be batched for tokens at the same step, forcing schedulers to run in short \u201cticks\u201d. How many tokens you pack into a single \u201ctick\u201d (i.e. how long you wait to collect tokens) is your batch size\n\nThese are tokens from different users. You can\u2019t batch tokens from the same user because you need previous tokens to generate the next one, so batching requires a high volume of traffic from different users\n\n\nBigger batches raise latency because user tokens might be waiting up to 200ms before the batch is full enough to run, but they boost throughput by allowing larger (and thus more efficient) GEMMs in the feed-forward step\nModels with many layers (e.g. long pipelines) need larger batches to avoid pipeline bubbles (by ensuring each tick contains more batches than pipeline steps) \nMixture-of-Experts models need to be served with high-latency to be efficient: each expert sees only the tokens routed to it, so you need larger global batches to keep every expert busy.  \nInference providers pick a batch size/window that clears pipeline bubbles and saturates experts. High batch sizes buy you more throughput at the cost of higher latency as tokens wait to fill up the tick\nSome models (like DeepSeek\u2019s) that are mixture-of-experts with many layers thus require large batch sizes and high latency, otherwise throughput drops off a cliff. That\u2019s why it\u2019s commonly said that you can\u2019t easily run DeepSeek for personal use: because with a single user running one inference at a time, it runs at very low efficiency/throughput\n\nThe fact that OpenAI and Anthropic\u2019s models are quick to respond suggests that either:\n\nTheir models have a more efficient architecture (non-MoE, fewer layers), or\nOpenAI/Anthropic have some very clever tricks for serving inference, or\nthey\u2019re paying through the nose for way more GPUs than they strictly need\n\n\n\nedit: This was posted on Hacker News with a bunch of comments. I kind of wish I\u2019d titled this post differently - it\u2019s not really about running models on your own computer. It\u2019s about running the models for personal use, assuming you have all the GPUs (i.e. the batching/throughput tradeoff).\n1 One commonly-observed strength of transformers is that they can batch prefill within a single user request. When you pass them a long prompt, they can process that prompt all at once because of how the attention mechanism works. Previous recurrent models had to go token-by-token, which was much slower (because it involved many more GEMMs). This has nothing to do with the kind of batching I\u2019m talking about in this post. I\u2019m talking about how you can efficiently batch inference across many different user requests once the prefilling is complete.\n2 This can also be batched, so long as you\u2019re only batching attention operations with the same number of tokens in the sequence (i.e. every sequence predicting the fourth token can be batched together). Otherwise the size of the KV cache matrices are different, so you can\u2019t easily combine them into a single batch. More on that later.\n3 Technically it\u2019s not a token being generated, but the \u201clogits\u201d (a probability distribution across all possible tokens). I\u2019ll say \u201ctoken\u201d here and later on to keep it simpler.\n4 Note that in practice modern inference stacks will use \u201ccontinuous batching\u201d, where a batch is sent off as soon as it\u2019s full instead of waiting for the entire length of the fixed time window. However, the inference is still done in batches, to the core tradeoff between throughput and latency is the same.If you liked this post, consider subscribing to email updates about my new posts.June 1, 2025\u00a0\u2502 Tags: ai, explainers, deepseek"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:05:07.399164+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.seangoedecke.com/inference-batching-and-deepseek/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-06-02T05:05:06.390247+00:00",
                "index_texts": null,
                "output": "Why DeepSeek is cheap at scale but expensive to run locally | sean goedecke",
                "pwd": "/data/archive/1748840674.884257",
                "schema": "ArchiveResult",
                "start_ts": "2025-06-02T05:05:06.334297+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "ArchiveError: Failed to find \"content-location\" URL header in Archive.org response.",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "Why DeepSeek is cheap at scale but expensive to run locally | sean goedecke",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1748840674.884257",
    "newest_archive_date": "2025-06-02T05:05:30.000409+00:00",
    "num_failures": 1,
    "num_outputs": 8,
    "oldest_archive_date": "2025-06-02T05:04:40.645540+00:00",
    "path": "/inference-batching-and-deepseek/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01JWQGDXKJD08341A601NFQ0Q5",
    "snapshot_id": "a04d36a1-ba04-40bd-b1f9-b755aafb82e5",
    "sources": [
        "/data/sources/1748840674-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1748840674.884257",
    "title": "Why DeepSeek is cheap at scale but expensive to run locally | sean goedecke",
    "url": "https://www.seangoedecke.com/inference-batching-and-deepseek/"
}