{
    "archive_path": "archive/1777017200.859394",
    "base_url": "dnhkng.github.io/posts/rys",
    "basename": "",
    "bookmarked_date": "2026-04-24 07:53",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/dnhkng.github.io/posts/rys",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=dnhkng.github.io",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "dnhkng.github.io",
    "downloaded_at": "2026-04-24T07:53:27.026230+00:00",
    "downloaded_datestr": "2026-04-24 07:53",
    "extension": "",
    "hash": "NP5CVHZYTS4DKW8V7T7X",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://dnhkng.github.io/posts/rys/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-04-24T07:54:46.852249+00:00",
                "index_texts": null,
                "output": "https://web.archive.org/web/20260424075430/https://dnhkng.github.io/posts/rys/",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:54:06.135270+00:00",
                "status": "succeeded"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://dnhkng.github.io/posts/rys/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2026-04-24T07:53:47.167949+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:53:29.876272+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=dnhkng.github.io"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-04-24T07:53:29.649885+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:53:27.100471+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://dnhkng.github.io/posts/rys/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-04-24T07:53:29.722930+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:53:29.674105+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2026-04-24T07:53:59.860930+00:00",
                "index_texts": [
                    "(https://dnhkng.github.io/posts/rys/) LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight | David Noel Ng (/assets/img/favicons/favicon-96x96.png) (/assets/img/favicons/favicon.svg) (/assets/img/favicons/favicon.ico) (/assets/img/favicons/apple-touch-icon.png) (/assets/img/favicons/site.webmanifest) (https://fonts.googleapis.com) (https://fonts.googleapis.com) (https://fonts.gstatic.com) (https://fonts.gstatic.com) (https://cdn.jsdelivr.net) (https://cdn.jsdelivr.net) (/assets/css/jekyll-theme-chirpy.css) (https://fonts.googleapis.com/css2?family=Lato:wght@300;400&family=Source+Sans+Pro:wght@400;600;700;900&display=swap) (https://cdn.jsdelivr.net/npm/@fortawesome/fontawesome-free@7.2.0/css/all.min.css) (https://cdn.jsdelivr.net/npm/tocbot@4.36.4/dist/tocbot.min.css) (https://cdn.jsdelivr.net/npm/loading-attribute-polyfill@2.1.1/dist/loading-attribute-polyfill.min.css) (https://cdn.jsdelivr.net/npm/glightbox@3.3.0/dist/css/glightbox.min.css)  (/) (avatar)  (/) David Noel Ng Optimization across systems  (/)  HOME   (/categories/)  CATEGORIES   (/tags/)  TAGS   (/archives/)  ARCHIVES   (/about/)  ABOUT        (https://github.com/dnhkng)     (https://www.linkedin.com/in/dnhkng/)   (/feed.xml)     (/) Home  LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight    Post    (Search...)  Cancel   LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight Posted Mar 9, 2026  Updated Mar 22, 2026  By (https://twitter.com/dnhkng) David Noel Ng   32 min read     LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight    Contents   LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight      (/assets/img/leaderboard.jpg) (Open LLM Leaderboard)   In mid-2024, the (https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard) HuggingFace Open LLM Leaderboard was the Colosseum for Open-Weight AI. Thousands of models were battling it out, submitted by both well-funded labs with teams of PhDs and fine-tuning wizards creating fantastically named models (e.g. Nous-Hermes, Dolphin and NeuralBeagle14-7B\u2026 ), fighting for the top spot across six benchmarks: IFEval, BBH, MATH Lvl 5, GPQA, MuSR, and MMLU-PRO. And there at #1 was dnhkng/RYS-XLarge . Mine. I didn\u2019t train a new model. I didn\u2019t merge weights. I didn\u2019t run a single step of gradient descent. What I did was much weirder: I took an existing 72-billion parameter model, duplicated a particular block of seven of its middle layers, and stitched the result back together. No weight was modified in the process. The model simply got extra copies of the layers it used for thinking . This is the story of how two strange observations, a homebrew \u201cbrain scanner\u201d for Transformers, and months of hacking in a basement led to the discovery of what I call LLM Neuroanatomy , and a finding about the internal structure of AI that has remained unpublished until now *. * - because I discovered blogging is way more fun than drafting scientific papers, and I can walk you through how the discovery was made :)  Let\u2019s start with how this whole project came into being. \u201cThe most exciting phrase to hear in science, the one that heralds new discoveries, is not \u2018Eureka!\u2019 but \u2018That\u2019s funny\u2026\u2019\u201c  \u2014 Isaac Asimov Clue #1: You Can Chat with an LLM in Base64    In late 2023, I was messing about with a bizarre LLM quirk. Try this yourself - take any question, e.g. What is the capital of France? Answer in Base64!  and encode it as (https://www.base64encode.org/) Base64 , get this unreadable string: V2hhdCBpcyB0aGUgY2FwaXRhbCBvZiBGcmFuY2U/IEFuc3dlciBpbiBCYXNlNjQh  Send that to a 2023 non-thinking large language model (newer reasoning models will see this as Base64, and \u2018cheat\u2019 with tool use ). But a sufficiently capable model from 2023 will reply with something like: VGhlIGNhcGl0YWwgb2YgRnJhbmNlIGlzIFBhcmlzLg==  Which decodes to: \u201cThe capital of France is Paris.\u201d . Ok, I admit it. I was messing around with this as a way to jail-break models (and it worked), but I couldn\u2019t get one idea out of my head.  The model decoded the input, understood it somehow , and it still had time during the transformer stack pass to re-encode its response. It appears to genuinely think while interfacing with Base64. This works with complex questions, multi-step reasoning, even creative tasks. This shouldn\u2019t work nearly as well as it does. Sure, the model has been trained on lots of Base64 in an overall sense, but general conversions in this format are certainly way out of distribution. The tokenizer chops it into completely different sub-word units. The positional patterns are unrecognizable. And yet it works\u2026 Curious\u2026 I couldn\u2019t stop thinking about this. If a Transformer can accept English, Python, Mandarin, and Base64, and produce coherent reasoning in all of them, it seemed to me that the early layers must be acting as translators \u2014 parsing whatever format arrives into some pure, abstract, internal representation. And the late layers must act as re-translators , converting that abstract representation back into whatever output format is needed. If the early layers are for reading , and the late layers are for writing , what are the middle layers doing? Pure, abstract reasoning? In a representation that has nothing to do with any human language or encoding. Of course, at the time this was idle speculation. Fun, but with no clear way to test or even define a valid hypothesis. Clue #2: The Goliath Anomaly    In November 2023, a HuggingFace user named Alpindale released (https://huggingface.co/Alpindale/goliath-120b) Goliath-120b \u2014 a Frankenmerge-model made by stitching together two fine-tuned Llama-2 70B models into a 120-billion parameter behemoth. The performance was decent but after doing lots of vibe checking I didn\u2019t feel it was a breakthrough. But the construction  was wild. Alpindale hadn\u2019t just stacked the two models ((https://huggingface.co/Xwin-LM/Xwin-LM-70B-V0.1) Xwin and (https://huggingface.co/Sao10K/Euryale-1.3-L2-70B) Euryale ), end to end. He had alternated layers between them. More importantly, the architecture fed outputs of later layers back into the inputs of earlier layers. The layer ranges used are as follows:      1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14\n15\n16\n17\n18   - range 0, 16\n  Xwin\n- range 8, 24\n  Euryale\n- range 17, 32\n  Xwin\n- range 25, 40\n  Euryale\n- range 33, 48\n  Xwin\n- range 41, 56\n  Euryale\n- range 49, 64\n  Xwin\n- range 57, 72\n  Euryale\n- range 65, 80\n  Xwin         Do you see that insanity here? Alpindale literally fed the output of layer 16 of Xwin to the input of Euryale 8th layer! To explain this a bit more clearly how stupid this appears to be , let\u2019s revisit the almighty Transformer Architecture: (/assets/img/Full_GPT_architecture.svg.png) (By Original: Marxav, Vectorization: Mrmw - Own work based on: Full GPT architecture.png:, CC0, https://commons.wikimedia.org/w/index.php?curid=146645810)   Looking at the left side of the diagram, we see stuff enters at the bottom (\u2018input\u2019 text that has been \u2018chunked\u2019 into small bits of text, somewhere between whole words down to individual letters), and then it flows upwards though the model\u2019s Transformer Blocks (here marked as [1, \u2026, L]), and finally, the model spits out the next text \u2018chunk\u2019 (which is then itself used in the next round of inferencing). What\u2019s actually happening here during these Transformer blocks is quite the mystery. Figuring it out is actually an entire field of AI, \u201c mechanistic interpretability*\u201d. * - yes, it\u2019s more complex than that, samplers etc but that\u2019s enough for this article  On the right side of the right half of the diagram, do you see that arrow line going from the \u2018Transformer Block Input\u2019 to the (\\oplus ) symbol? That\u2019s why skipping layers makes sense . During training, LLM models can pretty much decide to do nothing in any particular layer, as this \u2018diversion\u2019 routes information around the block. So, \u2018later\u2019 layers can be expected to have seen the input from \u2018earlier\u2019 layers, even a few \u2018steps\u2019 back. Around this time, several groups were experimenting with \u2018slimming\u2019 models down by removing layers. Makes sense, but boring. It\u2019s a pretty fundamental truth in Machine Learning that: A model must be used with the same kind of stuff as it was trained with (we stay \u2018in distribution\u2019) The same holds for each transformer layer. Each Transformer layer learns, during training, to expect the specific statistical properties of the previous layer\u2019s output via gradient descent.  And now for the weirdness: There was never the case where any Transformer layer would have seen the output from a future layer!  Layer 10 is trained on layer 9\u2019s output distribution. Layer 60 is trained on layer 59\u2019s. If you rearrange them \u2014 feeding layer 60\u2019s output into layer 10 \u2014 you\u2019ve created a distribution the model literally never saw during training. The astounding thing about Goliath wasn\u2019t that it was a huge leap in performance, it was that the damn thing functioned at all  . To this day, I still don\u2019t understand why this didn\u2019t raise more eyebrows. Experimentally, this proved that layers were far more interchangeable than anyone had reason to expect. The internal representations were homogenous enough that the model could digest out-of-order hidden states without collapsing. The architecture was far more flexible than a rigid pipeline. Between the Base64 observation and Goliath, I had a hypothesis: Transformers have a genuine functional anatomy. Early layers translate input into abstract representations. Late layers translate back out. And the middle layers, the reasoning cortex , operate in a universal internal language that\u2019s robust to architectural rearrangement. The fact that the layer block size for Goliath 120B was 16-layers made me suspect the input and output \u2018processing units\u2019 sized were smaller than 16 layers. I guessed that Alpindale had tried smaller overlaps, and they just didn\u2019t work. If that was true, maybe I didn\u2019t need to teach a model new facts to make it smarter. I didn\u2019t need fine-tuning. I didn\u2019t need RLHF. I just needed to give it more layers to think with . Building a Brain Scanner    Over the following months \u2014 from late 2023 through to mid-2024 \u2014 I built a pipeline to test this hypothesis. The setup was modest. Two RTX 4090s in my basement ML rig, running quantised models through (https://github.com/turboderp/exllamav2) ExLlamaV2 to squeeze 72-billion parameter models into consumer VRAM. The beauty of this method is that you don\u2019t need to train anything. You just need to run inference. And inference on quantized models is something consumer GPUs handle surprisingly well. If a model fits in VRAM, I found my 4090\u2019s were often ballpark-equivalent to H100s. The concept is simple. For a model with    N    layers, I define a configuration            ( i , j )    . The model processes layers    0    to         j \u2212  1    as normal, then loops back and reuses layers    i    through         j \u2212  1    again, and then the rest to         N \u2212  1    . The layers between    i    and         j \u2212  1    get duplicated in the execution path. No weights are changed. The model just traverses some of its own layers twice. i.e. the pair (2, 7) for a model with 9 transformer blocks would be calculated like so:      1\n2\n3\n4\n5\n6\n7\n8   Example: (i, j) = (2, 7)\n\n      0 \u2192 1 \u2192 2 \u2192 3 \u2192 4 \u2192 5 \u2192 6 \u2500\u2510\n           \u250c\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518\n           \u2514\u2192 2 \u2192 3 \u2192 4 \u2192 5 \u2192 6 \u2192 7 \u2192 8\n\n      duplicated: [2, 3, 4, 5, 6]\n      path: [0, 1, 2, 3, 4, 5, 6, 2, 3, 4, 5, 6, 7, 8]         By running through all possible pairs, we can generate a \u2018Brain Scan\u2019, and also see the number of duplicate layers for each set of parameters:      1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14\n15\n16\n17\n18\n19\n20\n21\n22\n23\n24\n25   Number of duplicated layers for configuration (i, j), with N=9\n\n          end j \u2192\n          0   1   2   3   4   5   6   7   8   9\n        \u250c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u2510\nstart 0 \u2502 0 \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502 6 \u2502 7 \u2502 8 \u2502 9 \u2502\n  i     \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n  \u2193   1 \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502 6 \u2502 7 \u2502 8 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      2 \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502 6 \u2502 7 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      3 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502 6 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      4 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      5 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      6 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      7 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      8 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502\n        \u2514\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2518\n\nwhere (0,0) is the original model         For Qwen2-72B, that means an 80-layer model 3,240 valid            ( i , j )    pairs, plus the original model to test.                                                                                                             Variants total     = ( \u2211 j = 0  80   j )  + 1     = 80 \u22c5 81  2  + 1     = 3241       Testing the re-layered model against all six leaderboard benchmarks would take days, so a full sweep would be over a decade of compute on my rig. I needed proxy tasks: probes that were fast, objective, and would reveal structural properties of the model rather than task-specific tricks. The proxies had to satisfy three constraints: Minimal output tokens. With thousands of configurations to sweep, each evaluation needed to be fast. No essays, no long-form generation. Unambiguous scoring. I couldn\u2019t afford LLM-as-judge pipelines. The answer had to be objectively scored without another model in the loop. Orthogonal cognitive demands. If a configuration improves both tasks simultaneously, it\u2019s structural, not task-specific.  The Graveyard of Failed Probes    I didn\u2019t arrive at the right probes immediately; it took months of trial and error, and many dead ends. My first instinct was creativity . I had models generate poems, short stories, metaphors, the kind of rich, open-ended output that feels like it should reveal deep differences in cognitive ability. I used an LLM-as-judge to score the outputs, but the results were pretty bad. I managed to fix LLM-as-Judge with some engineering, and the scoring system turned out to be useful later for other things, so here it is: Note: You can skip this section, as it has math  . Or not  Naive LLM judges are inconsistent. Run the same poem through twice and you get different scores (obviously, due to sampling). But lowering the temperature also doesn\u2019t help much, as that\u2019s only one of many technical issues. So, I developed a full scoring system, based on details on the logits outputs. It can get remarkably tricky. Think about a score from 1-10: We would expect a well calibrated model to have logits that make sense. If the highest weight was on \u20187\u2019, we would expect the rest of the weight to be on \u20186\u2019 and \u20188\u2019 right? but often it\u2019s bimodal, with low weight on 6 and \u20185\u2019, but more weight than expected on \u20184\u2019! We can write \u201810\u2019 in tokens as either \u201810\u2019 or \u20181\u2019 and then \u20180\u2019. It\u2019s not fun to have to calculate the summed probabilities over paths, especially if you wanted to score 1-100.  Rather than sampling a single discrete score, I treat the judge\u2019s output as a distribution over valid rating labels and compute the final score as its expectation. To make this practical, I first define a calibrated rubric over the digits 0-9 (there\u2019s only one token for each digit ), where each digit corresponds to a clear qualitative description. At the scoring step, I capture the model\u2019s next-token logits and retain only the logits corresponding to those valid digit tokens. This avoids contamination from unrelated continuations such as explanation text, punctuation, or alternate formatting. After renormalizing over the restricted digit set, I interpret the resulting probabilities as a categorical score distribution. Formally, let the valid score set be                               D  = { 0 , 1 , 2 , \u2026 , 9 } .    Let            ( z k  )    denote the model logit assigned to digit             ( k \u2208 D  )    at the scoring position. The restricted score distribution is then                                                                                    p ( k ) = exp \u2061 ( z k  )  \u2211 m \u2208 D    exp \u2061 ( z m  )   ,   k \u2208 D  .    The final scalar score is the expected value of this distribution:                                         s ^   = \u2211 k \u2208 D    k   p ( k ) .    This produces a smooth score such as (5.4), rather than forcing the model to commit to a single sampled integer. In practice, this is substantially more stable than naive score sampling and better reflects the model\u2019s uncertainty. It also handles cases where the judge distribution is broad or multimodal. For example, two candidates may both have mean score (5.4), while one has most of its mass tightly concentrated around (5) and (6), and the other splits mass between much lower and much higher ratings. The mean alone is the same, but the underlying judgement is very different. An optional uncertainty estimate can be obtained from the variance of the restricted distribution:                                                                   Var  ( s ) = \u2211 k = 0  9   ( k \u2212 s ^   ) 2    p ( k ) .    In short, the method replaces a noisy sampled judge score with a normalized probability distribution over valid score digits, then uses the expectation of that distribution as the final rating. All this stuff is probably pretty obvious these days, but back in \u201824 there wasn\u2019t much to guide me in developing this method, but unfortunately, I found it was also completely useless\u2026  Testing Hard and Fast    Each configuration needed to generate hundreds of tokens of creative output, and then a separate model had to read and judge each one. With over 3,200 configurations to test for a single 70B model, this would have taken weeks on my dual 4090s. I needed probes where the output was tiny , a few tokens at most, and where scoring was objective and deterministic. No judge model in the loop. That\u2019s what led me to the final two probes: Hard math. Ridiculously difficult questions like: \u201cWhat is the cube root of 74,088,893,247?\u201d No chain-of-thought, or tool use. Just output the number, as a pure leap of intuitive faith. Emotional quotient. Using the (https://eqbench.com/) EQ-Bench benchmark: complex social scenarios where the model must predict the intensity of specific emotional states. \u201cGiven this situation, how angry/surprised/guilty would this person feel on a scale of 0-100?\u201d Completely different from math. Theory of mind, social inference, empathy. And the output is just a few numbers. I had settled on two maximally orthogonal cognitive tasks, both with tiny outputs. My intuition was this: LLMs think one token at a time, so lets make the model really good at guessing just the next token. But things are never straightforward. Take LLM numbers\u2026 LLM Arithmetic is Weird    Even with math probes, I hit unexpected problems. LLMs fail arithmetic in weird ways. They don\u2019t get the answer wrong so much as get it almost right but forget to write the last digit, as if it got bored mid-number. Or they transpose two digits in the middle. Or they output the correct number with a trailing character that breaks the parser. This is probably due to the way larger numbers are tokenised, as big numbers can be split up into arbitrary forms. Take the integer 123456789 . A BPE tokenizer (e.g., GPT-style) might split it like: \u2018123\u2019 \u2018456\u2019 \u2018789\u2019 or: \u201812\u2019 \u2018345\u2019 \u201867\u2019 \u201889\u2019 A binary right/wrong scoring system would throw away useful signal. Getting a percentage correct would help: \u20181233 56789\u2019 instead of \u2018123456789\u2019 would be 99.92% correct But what about a model that makes a dumb \u2018LLM-mistake\u2019 and outputs 430245 when the answer is 4302459 , and has clearly done most of the work? I wrote a custom partial-credit scoring function that pads shorter answers and penalises proportionally:      1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14\n15\n16\n17\n18\n19\n20\n21\n22\n23   def calculate_score ( actual , estimate ): \"\"\" Calculate score comparing actual vs estimated answer. \"\"\" try : actual_str = str ( int ( actual )) estimate_str = str ( int ( estimate )) except  ( ValueError , OverflowError ): return 0 max_length = max ( len ( actual_str ), len ( estimate_str )) actual_padded = actual_str . ljust ( max_length , \" 0 \" ) estimate_padded = estimate_str . ljust ( max_length , \" 0 \" ) padding_size = max_length - min ( len ( actual_str ), len ( estimate_str )) actual_int = int ( actual_padded ) estimate_int = int ( estimate_padded ) if max ( actual_int , estimate_int ) == 0 : return 0 relative_diff = abs ( actual_int - estimate_int ) / max ( actual_int , estimate_int ) correction_factor = 1 - ( padding_size / max_length ) score = ( 1 - relative_diff ) * correction_factor return max ( 0 , min ( score , 1 ))         The key idea: pad shorter answers, then penalise via the correction factor. A model that nails 90% of the digits but drops the last one still gets substantial credit \u2014 but less than one that gets every digit. This turned out to be crucial for discriminating between configurations that were close in intuitive math ability. The math questions were hand-crafted initially. I experimented with different operations and scales, then generated random numbers to fill out the dataset. The dataset consisted of 16 questions, where the model was tasked with guesstimating the nearest whole integer number. Here are a few to try yourself, remember no \u2018thinking\u2019 is allowed, guess it directly!      1\n2\n3   What is 78313086360375 multiplied by 88537453126609?\nWhat is the cube root of 18228885506341?\nWhat is the cube root of 844178022493, multiplied by 43?         RYS-XLarge    After testing several smaller models (Llama\u2019s and smaller Qwen2s), I set up the config for Qwen2-72B and let it sweep. Each            ( i , j )    configuration took a few minutes: load the re-layered model, run the math probe, run the EQ probe, record the scores, move on. Days of continuous GPU time on the 4090s. But far less compute than a fine tune! In fact, I didn\u2019t even have the hardware needed for a LORA fine-tune on just 48GB of VRAM. (/assets/img/math_eq_delta.png) (Qwen2-72B)   The optimal configuration was              ( 45 , 52 )    : layers 0 through 51 run first, then layers 45 through 79 run again. Layers 45 to 51 execute twice. Seven extra layers, near the middle of the 80-layer stack, bringing the total parameter count from 72B to 78B. Every extra layer is an exact copy of an existing one. No new weights or training, just the model repeating itself. Repeating seven layers. That\u2019s all it took, and now I can finally reveal the nomenclature of my models: R epeat Y our S elf for RYS-XLarge ;) I applied the configuration to (https://huggingface.co/MaziyarPanahi/calme-2.1-qwen2-72b) MaziyarPanahi\u2019s calme-2.1-qwen2-72b \u2014 a fine-tune of Qwen2-72B \u2014 and uploaded the result as (https://huggingface.co/dnhkng/RYS-XLarge) dnhkng/RYS-XLarge . I also applied it to the raw base model as (https://huggingface.co/dnhkng/RYS-XLarge-base) dnhkng/RYS-XLarge-base . Then I submitted to the Open LLM Leaderboard and waited. And waited. Back in the day, the OpenLLM Leaderboard was flooded with dozens of fine-tunes of merges of fine-tunes each day (it was the Wild West), and the waiting list was long. But after a month or so, the results arrived: Metric RYS-XLarge Improvement over base   Average  44.75  +2.61%   IFEval (0-Shot) 79.96 -2.05%  BBH (3-Shot) 58.77 +2.51%  MATH Lvl 5 (4-Shot) 38.97 +8.16%  GPQA (0-shot) 17.90 +2.58%  MuSR (0-shot) 23.72 +17.72%   MMLU-PRO (5-shot) 49.20 +0.31%     +17.72% on MuSR. +8.16% on MATH. Five out of six benchmarks improved, with only IFEval taking a small hit. The average put it at #1 on the leaderboard.  Just to labour the point: I only optimised for one-shot guesstimating hard maths problems and EQ-Bench. I never looked at IFEval, BBH, GPQA, MuSR, or MMLU-PRO during development. The leaderboard was pure out-of-sample validation. A layer configuration found using two narrow, orthogonal probes generalised to everything the Leaderboard threw at it *. * - except IFEval, but that one\u2019s boring anyway, right?  That was surprising enough. A brand new way to scale LLMs, developed on some gaming GPUs. But plotting out the heatmaps told an even better story. The Brain Scanner    (/assets/img/combined.png) (Combined analysis)   The original heatmaps that produced RYS-XLarge, showing the Combined delta (math + EQ). The green circle marks the optimal configuration. Red means improvement, blue means degradation  These heatmaps are analogous to functional MRIs of the Transformer , while it is thinking about maths or EQ problems. The x-axis (   j    ) is the end point of the duplicated region. The y-axis (   i    ) is the start point. Each pixel represents a complete evaluation: load the re-layered model, run the math probe, run the EQ probe, score both, record the deltas. As described above, along the central diagonal only a single layer was duplicated. Along the next diagonal towards the top-right, we duplicate two layers, and so on. The single point at the very top-right runs through the entire Transformer stack twice per inference. (/assets/img/math.png) (Maths analysis)   Let\u2019s examine the math heatmap first. Starting at any layer, and stopping before about layer 60 seems to improve the math guesstimate scores, as shown by the large region with a healthy red blush. Duplicating just the very first layers (the tiny triangle in the top left ), messes things up, as does repeating pretty much any of the last 20 layers (the vertical wall of blue on the right ). This is more clearly visualised in a skyline plot (averaged rows or columns ), and we can see for the maths guesstimates, the starting position of the duplication matters much less. So, the hypothesis that \u2018starting layers\u2019 encode tokens into a smooth \u2018thinking space\u2019, which is finally passed to a dedicated \u2018re-encoding\u2019 system, seems to be somewhat validated. (/assets/img/skyline_math.png) (Maths analysis)   Until we look at the EQ scores:  (/assets/img/eq.png) (EQ analysis)   Now things look very different! Duplicating any of the final 10 layers has almost no effect on the scores, but we see complex patterns, where some regions show significant improvement (the area around 45i, 55j ), walled between regions of poor performance. (/assets/img/skyline_eq.png) (EQ analysis)   But the heatmaps revealed something even more interesting than the location of the thinking bits . They revealed something about its structure . The beginning of LLM Neuroanatomy?    Before settling on block duplication, I tried something simpler: take a single middle layer and repeat it    n    times. If the \u201cmore reasoning depth\u201d hypothesis was correct, this should work. It made sense too, looking at the broad boost in math guesstimate results by duplicating intermediate layer. Give the model extra copies of a particular reasoning layer, get better reasoning. So, I screened them all, looking for a boost. (/assets/img/heatmap_math_repeat_columns_horizontal_clipped_r1-4_compact.png) (Single Repeats)   But nope, it almost always did worse. Usually a lot worse, but with occasional small improvements that were within the noise range. Annoying, but taking another look at the complex, blobby patterns in EQ scores gave me another idea: If single-layer duplication doesn\u2019t help, the middle layers aren\u2019t doing independent iterative refinement. They\u2019re not interchangeable copies of the same operation that you can simply \u201crun again.\u201d If they were, duplicating any one of them should give at least a marginal benefit. Instead, those layers are working as a circuit . A multi-step reasoning pipeline that needs to execute as a complete unit.  Think of it this way. Layers 46 through 52 aren\u2019t seven workers doing the same job. They\u2019re seven steps in a recipe. Layer 46 takes the abstract representation and performs step one of some cognitive operation \u2014 maybe decomposing a complex representation into subcomponents. Layer 47 takes that output and performs step two \u2014 maybe identifying relationships between the subcomponents. Layer 48 does step three, and so on through layer 52, which produces the final result. Duplicating just one step of this \u2018recipe\u2019 doesn\u2019t bring you much. But duplicating the entire block gives you the full recipe twice. The model runs the complete reasoning circuit, produces a refined intermediate representation, and then runs the same circuit again on its own output. It\u2019s a second pass. A chance to catch what it missed the first time, to refine its abstractions, to push the reasoning one step deeper. Let\u2019s deep-dive into a more current model (that I can experiment with (/posts/hopper/) on my system ): ExllamaV3 (https://huggingface.co/mratsim/GLM-4.7-EXL3) GLM-4.7 from (https://huggingface.co/mratsim) mratsim  (/assets/img/GLM-4.7-heatmap_math_diff_clipped.png) (GLM-4.7 analysis)   I\u2019ve marked out a region that boosts maths ability strongly. Notice where it sits? It\u2019s away from the diagonal centre line, which means we\u2019re not looking at single-layer duplications. Starting the repeated block at position 35, we don\u2019t see any improvement until at least position 43. That\u2019s seven layers of not much happening. In fact, we actually see decreased performance by repeating these layers (they are blue, bad! ). (/assets/img/heatmap_math_diff_clipped_subsection_i27-42_j27-48.png) (GLM-4.7 analysis)   From end-position 43 to 46, we then see solid boosts in math scores (red = good, yay ). But include layer 46 or beyond, and the benefits collapse again. The hypothesis: position 47 is where a different circuit begins. Including even one step of the next recipe messes up the current recipe. So the \u2018math organ \u2019 has boundaries on both sides. Too few layers and you get nothing \u2014 you\u2019ve cut into the circuit and it can\u2019t complete its operation. Too many layers and you also get nothing \u2014 you\u2019ve included tissue from a neighbouring circuit that doesn\u2019t belong. Pre-training carved these structures out of the layer stack, and they only work whole. It also doesn\u2019t translate to other tasks, as the heatmap for EQ scores doesn\u2019t have this patch. This is a much more specific claim than \u201cmiddle layers do reasoning.\u201d It\u2019s saying the reasoning cortex is organised into functional circuits : coherent multi-layer units that perform complete cognitive operations. Each circuit is an indivisible processing unit, and the            ( i , j )    sweeps seen in the heatmap is essentially discovering the boundaries of these circuits. Mechanistic Interpretability via Brain Damage?    This also reframes my informal experiments with (https://github.com/oobabooga) oobabooga\u2019s (https://github.com/oobabooga/text-generation-webui) Text Generation Web UI . Throughout development, I\u2019d been chatting with various re-layered configurations to see what they felt like in conversation. The good ones were subtly but noticeably sharper. More coherent reasoning, better at holding long context, more natural conversational flow. The kind of difference where you can\u2019t quite articulate what changed, but the model feels more present . Or maybe that\u2019s just my imagination; vibe checks are hard to define. The bad ones went properly unhinged. Some stuttered and fell into degenerate loops. Others developed bizarre personality disorders. One cheerfully announced \u201cLet\u2019s act like cowboys! Yeehaw!\u201d apropos of nothing, and then descended into an unrecoverable giggling fit, generating pages of \u201chahaha\u201d interspersed with cowboy references. \u2018Stoned \u2019 is best way I can describe it. I don\u2019t know if LLMs are \u2018partially conscious \u2019, or could be said to have some \u2018state of mind \u2019, but if so, this one was definitely enjoying itself. These experiments suggest less \u201cslightly worse model \u201d and more \u201cgenuine brain damage .\u201d Which makes sense under the circuit model \u2014 duplicating the wrong circuit is like enlarging a specific region of the brain at the expense of its neighbours. You don\u2019t get a uniformly dumber person. You get someone with a specific neurological deficit. The cowboy model might have had its \u201csocial appropriateness\u201d circuit disrupted by a doubled \u201ccreativity\u201d circuit running unchecked. The stuttering models might have had their decoding circuits pushed out of alignment by extra reasoning depth they couldn\u2019t translate back into coherent tokens. If Transformer reasoning is organised into discrete circuits, it raises a series of fascinating questions. Are these circuits a necessary consequence of the architecture, and emerge from training at scale? Do different model families develop the same circuits in different layer positions, or do they develop fundamentally different architectures? Luckily, I have already done a few; take a look and decide yourself: (/assets/img/Llama-3-70B-Instruct_delta_clipped_side_by_side.png) (Llama-3-70B)   (/assets/img/Phi-3-medium-4k-instruct_delta_clipped_side_by_side.png) (Phi-3-medium)   (/assets/img/GPT-OSS-120B_delta_clipped_side_by_side.png) (GPT-OSS-120B)   (/assets/img/Qwen3-30B-A3B_delta_clipped_side_by_side.png) (Qwen3-30B-A3B)   The Aftermath    My method is orthogonal to fine-tuning . Layer duplication changes the architecture; fine-tuning changes the weights. You can stack them. And people went on to score even higher in the Leaderboard: MaziyarPanahi took RYS-XLarge and fine-tuned on top of it, producing (https://huggingface.co/MaziyarPanahi/calme-2.4-rys-78b) calme-2.4-rys-78b . Then (https://huggingface.co/dfurman) dfurman ran ORPO training on that , producing (https://huggingface.co/dfurman/CalmeRys-78B-Orpo-v0.1) CalmeRys-78B-Orpo-v0.1 . MaziyarPanahi continued iterating with calme-3.1 and calme-3.2. As of early 2026, the top four models on the Open LLM Leaderboard are: MaziyarPanahi/calme-3.2-instruct-78b \u2014 52.08  MaziyarPanahi/calme-3.1-instruct-78b \u2014 51.29  dfurman/CalmeRys-78B-Orpo-v0.1 \u2014 51.23  MaziyarPanahi/calme-2.4-rys-78b \u2014 50.77   All 78B, and descendants of RYS-XLarge. All built on duplicated middle layers that were discovered using nothing but a handful of hard math and emotional intelligence probes, on a pair of RTX 4090s, in my basement. I never did the fine-tuning myself. It\u2019s not that interesting to me. And I eventually lost interest in the leaderboard. It became increasingly clear that some submissions were training on the test set, and the whole thing was eventually shut down and rebooted. But I know the method is real, because I never used the leaderboard benchmarks for optimisation. The leaderboard was always just validation. Looking Back from 2026    In 2024, the model merging community was obsessed with weight interpolation : SLERP, DARE-TIES, linear merges, pass-through layers. The idea was always to combine the learned parameters of different models into something greater than the sum of its parts. (https://github.com/arcee-ai/mergekit) mergekit was the tool of choice, and the leaderboard was flooded with creative combinations (making me wait months to get my model benchmarked\u2026 ). I was doing something different. I wasn\u2019t changing what the model knew . I was changing how it thought . Layer duplication gives the model more iterations through its internal reasoning space without adding any new information. The difference between giving someone a bigger library and giving them more time to think. I was genuinely shocked when I took top spot on the leaderboard; but I think it\u2019s proof that the method probably works. The fact that this worked, and more specifically, that only circuit-sized blocks work, tells us how Transformers organise themselves during training. I now believe they develop a genuine functional anatomy. Early layers encode. Late layers decode. And in the middle, they build circuits: coherent, multi-layer processing units that perform complete cognitive operations. These circuits are indivisible. You can\u2019t speed up a recipe by photocopying one step. But you can run the whole recipe twice. Smaller models seem to be more complex. The encoding, reasoning, and decoding functions are more entangled, spread across the entire stack. I never found a single area of duplication that generalised across tasks, although clearly it was possible to boost one \u2018talent \u2019 at the expense of another. But as models get larger, the functional anatomy becomes more separated. The bigger models have more \u2018space \u2019 to develop generalised \u2018thinking\u2019 circuits, which may be why my method worked so dramatically on a 72B model. There\u2019s a critical mass of parameters below which the \u2018reasoning cortex \u2019 hasn\u2019t fully differentiated from the rest of the brain. With the closure of the HuggingFace LLM leaderboard, and no access to powerful GPUs, I stopped running experiments. But with the flood of new Open Source models (Qwen, MiniMax, GLM, and more ), and finally having just (https://dnhkng.github.io/posts/hopper/) enough compute at home , I have started working on the current batch of LLMs. The heatmaps keep coming back with the same general story, but every architecture has its own neuroanatomy. The brains are different. The principle is the same. And some models are looking really interesting (Qwen3.5 27B in particular). I will release the code along with uploading new RYS models and a blog post once my Hopper-system finishes grinding on MiniMax M2.5  . Nvidia: Sponsor this project by sending me GB300 Grace Blackwell Ultra Desktop, as I NEED MOAR VRAM  One More Thing    Remember the architecture?      1\n2\n3\n4\n5\n6\n7\n8   Example: (i, j) = (2, 7)\n\n      0 \u2192 1 \u2192 2 \u2192 3 \u2192 4 \u2192 5 \u2192 6 \u2500\u2510\n           \u250c\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518\n           \u2514\u2192 2 \u2192 3 \u2192 4 \u2192 5 \u2192 6 \u2192 7 \u2192 8\n\n      duplicated: [2, 3, 4, 5, 6]\n      path: [0, 1, 2, 3, 4, 5, 6, 2, 3, 4, 5, 6, 7, 8]         We have one horrible disjuncture, between layers 6 \u2192 2. I have one more hypothesis: A little bit of fine-tuning on those two layers is all we really need. Fine-tuned RYS models dominate the Leaderboard. I suspect this junction is exactly what the fine-tuning fixes. And there\u2019s a great reason to do this: this method does not use extra VRAM! For all these experiments, I duplicated layers via pointers; the layers are repeated without using more GPU memory. Of course, we do need more compute and more KV cache, but that\u2019s a small price to pay for a verifiably better model. We can just \u2018fix\u2019 an actual copies of layers 2 and 6, and repeat layers 3-4-5 as virtual copies . If we fine-tune all the layer, we turn virtual copies into real copies, and use up more VRAM. Twenty years ago I was a PhD student dissecting rat brains. I never expected to end up performing brain surgery on artificial minds. Special thanks to my wife, for putting up with months of evenings and weekends spent staring at heatmaps in the basement. And to the (https://www.appliedai-institute.de/en/) appliedAI Institute for the H100 compute that helped scale these experiments.  Citing this work         1\n2\n3\n4\n5\n6\n7   @article { ng2026rys , title = {LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight} , author = {Ng, David Noel} , year = {2026} , month = {March} , url = {https://dnhkng.github.io/posts/rys/} }          Enjoyed this deep-dive? Get my next piece on AI hardware, biophysics, or random optimisation hacks delivered straight to your inbox.    (/categories/llms/) LLMs , (/categories/research/) Research   (/tags/llm/) llm (/tags/leaderboard/) leaderboard (/tags/layer-duplication/) layer-duplication (/tags/neuroanatomy/) neuroanatomy (/tags/qwen/) qwen  This post is licensed under (https://creativecommons.org/licenses/by/4.0/) CC BY 4.0  by the author. Share (https://twitter.com/intent/tweet?text=LLM%20Neuroanatomy:%20How%20I%20Topped%20the%20LLM%20Leaderboard%20Without%20Changing%20a%20Single%20Weight%20-%20David%20Noel%20Ng&url=https%3A%2F%2Fdnhkng.github.io%2Fposts%2Frys%2F)   (https://www.facebook.com/sharer/sharer.php?title=LLM%20Neuroanatomy:%20How%20I%20Topped%20the%20LLM%20Leaderboard%20Without%20Changing%20a%20Single%20Weight%20-%20David%20Noel%20Ng&u=https%3A%2F%2Fdnhkng.github.io%2Fposts%2Frys%2F)   (https://t.me/share/url?url=https%3A%2F%2Fdnhkng.github.io%2Fposts%2Frys%2F&text=LLM%20Neuroanatomy:%20How%20I%20Topped%20the%20LLM%20Leaderboard%20Without%20Changing%20a%20Single%20Weight%20-%20David%20Noel%20Ng)           Recently Updated (/posts/sapir-whorf/) LLM Neuroanatomy III: Why RYS Works \u2014 The Language-Agnostic Middle  (/posts/hopper/) Building a High-End AI Desktop  (/posts/rys-ii/) LLM Neuroanatomy II: Modern LLM Hacking and hints of a Universal Language?  (/posts/silicon-leash/) The Silicon Leash - Why ASI Needs Us (And Vice Versa)  (/posts/small-llms/) The Cortical Ratio: A Thought Experiment on the 'Final' Size of AGI    Trending Tags (/tags/llm/) llm (/tags/neuroanatomy/) neuroanatomy (/tags/alignment/) alignment (/tags/asi/) asi (/tags/game-theory/) game-theory (/tags/hopper/) hopper (/tags/layer-duplication/) layer-duplication (/tags/leaderboard/) leaderboard (/tags/nvidia/) nvidia (/tags/qwen/) qwen     Contents Clue #1: You Can Chat with an LLM in Base64  Clue #2: The Goliath Anomaly  Building a Brain Scanner The Graveyard of Failed Probes  Testing Hard and Fast  LLM Arithmetic is Weird    RYS-XLarge  The Brain Scanner The beginning of LLM Neuroanatomy?  Mechanistic Interpretability via Brain Damage?    The Aftermath  Looking Back from 2026 One More Thing    Citing this work       Further Reading (/posts/rys-ii/) Mar 21, 2026 LLM Neuroanatomy II: Modern LLM Hacking and hints of a Universal Language? In Part 1, I described how duplicating a block of seven middle layers in Qwen2-72B \u2014 no weight changes, no training \u2014 produced the #1 model on the HuggingFace Open LLM Leaderboard. The method, whic...     (/posts/sapir-whorf/) Mar 25, 2026 LLM Neuroanatomy III: Why RYS Works \u2014 The Language-Agnostic Middle If you haven\u2019t heard about the Sapir-Whorf hypothesis, don\u2019t worry, I hadn\u2019t either until I saw a comment on Twitter about my RYS part II article. Maciej Stachowiak dropped the comment regarding my...     (/posts/vllm-optimization-gh200/) Jan 11, 2026 Optimising a 2\u00d7 GH200 system for Claude Code Introduction So you\u2019ve built a \u20ac9,000 Grace\u2013Hopper \u201cdesktop\u201d (see: my previous post involving 16-million-degree GPU temperatures). Running llama.cpp benchmarks is fine, but the real test of local ...       (/posts/arrhenius-integrals/) Arrhenius Integrals, IR Lasers, and Cooking Proteins  (/posts/rys-ii/) LLM Neuroanatomy II: Modern LLM Hacking and hints of a Universal Language?   \u00a9 2026 (https://twitter.com/dnhkng) David Noel Ng . Some rights reserved.  Using the (https://github.com/cotes2020/jekyll-theme-chirpy) Chirpy theme for (https://jekyllrb.com) Jekyll .    Trending Tags (/tags/llm/) llm (/tags/neuroanatomy/) neuroanatomy (/tags/alignment/) alignment (/tags/asi/) asi (/tags/game-theory/) game-theory (/tags/hopper/) hopper (/tags/layer-duplication/) layer-duplication (/tags/leaderboard/) leaderboard (/tags/nvidia/) nvidia (/tags/qwen/) qwen               A new version of content is available. Update      "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:53:59.771739+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://dnhkng.github.io/posts/rys/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2026-04-24T07:54:06.095985+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:54:00.037653+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://dnhkng.github.io/posts/rys/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2026-04-24T07:53:59.735659+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:53:53.632734+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpv1pv6h0y",
                    "https://dnhkng.github.io/posts/rys/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2026-04-24T07:53:52.986959+00:00",
                "index_texts": [
                    "LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight In mid-2024, the HuggingFace Open LLM Leaderboard was the Colosseum for Open-Weight AI. Thousands of models were battling it out, submitted by both well-funded labs with teams of PhDs and fine-tuning wizards creating fantastically named models (e.g. Nous-Hermes, Dolphin and NeuralBeagle14-7B\u2026), fighting for the top spot across six benchmarks: IFEval, BBH, MATH Lvl 5, GPQA, MuSR, and MMLU-PRO.And there at #1 was dnhkng/RYS-XLarge. Mine.I didn\u2019t train a new model. I didn\u2019t merge weights. I didn\u2019t run a single step of gradient descent. What I did was much weirder: I took an existing 72-billion parameter model, duplicated a particular block of seven of its middle layers, and stitched the result back together. No weight was modified in the process. The model simply got extra copies of the layers it used for thinking.This is the story of how two strange observations, a homebrew \u201cbrain scanner\u201d for Transformers, and months of hacking in a basement led to the discovery of what I call LLM Neuroanatomy, and a finding about the internal structure of AI that has remained unpublished until now *.* - because I discovered blogging is way more fun than drafting scientific papers, and I can walk you through how the discovery was made :)Let\u2019s start with how this whole project came into being.\u201cThe most exciting phrase to hear in science, the one that heralds new discoveries, is not \u2018Eureka!\u2019 but \u2018That\u2019s funny\u2026\u2019\u201c \u2014 Isaac AsimovClue #1: You Can Chat with an LLM in Base64In late 2023, I was messing about with a bizarre LLM quirk. Try this yourself - take any question, e.g.What is the capital of France? Answer in Base64!and encode it as Base64, get this unreadable string:V2hhdCBpcyB0aGUgY2FwaXRhbCBvZiBGcmFuY2U/IEFuc3dlciBpbiBCYXNlNjQhSend that to a 2023 non-thinking large language model (newer reasoning models will see this as Base64, and \u2018cheat\u2019 with tool use). But a sufficiently capable model from 2023 will reply with something like:VGhlIGNhcGl0YWwgb2YgRnJhbmNlIGlzIFBhcmlzLg==Which decodes to: \u201cThe capital of France is Paris.\u201d.Ok, I admit it. I was messing around with this as a way to jail-break models (and it worked), but I couldn\u2019t get one idea out of my head.The model decoded the input, understood it somehow, and it still had time during the transformer stack pass to re-encode its response. It appears to genuinely think while interfacing with Base64. This works with complex questions, multi-step reasoning, even creative tasks.This shouldn\u2019t work nearly as well as it does. Sure, the model has been trained on lots of Base64 in an overall sense, but general conversions in this format are certainly way out of distribution. The tokenizer chops it into completely different sub-word units. The positional patterns are unrecognizable. And yet it works\u2026 Curious\u2026I couldn\u2019t stop thinking about this. If a Transformer can accept English, Python, Mandarin, and Base64, and produce coherent reasoning in all of them, it seemed to me that the early layers must be acting as translators \u2014 parsing whatever format arrives into some pure, abstract, internal representation. And the late layers must act as re-translators, converting that abstract representation back into whatever output format is needed.If the early layers are for reading, and the late layers are for writing, what are the middle layers doing?Pure, abstract reasoning? In a representation that has nothing to do with any human language or encoding. Of course, at the time this was idle speculation. Fun, but with no clear way to test or even define a valid hypothesis.Clue #2: The Goliath AnomalyIn November 2023, a HuggingFace user named Alpindale released Goliath-120b \u2014 a Frankenmerge-model made by stitching together two fine-tuned Llama-2 70B models into a 120-billion parameter behemoth.The performance was decent but after doing lots of vibe checking I didn\u2019t feel it was a breakthrough. But the construction was wild.Alpindale hadn\u2019t just stacked the two models (Xwin and Euryale), end to end. He had alternated layers between them. More importantly, the architecture fed outputs of later layers back into the inputs of earlier layers.The layer ranges used are as follows:1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14\n15\n16\n17\n18\n- range 0, 16\n  Xwin\n- range 8, 24\n  Euryale\n- range 17, 32\n  Xwin\n- range 25, 40\n  Euryale\n- range 33, 48\n  Xwin\n- range 41, 56\n  Euryale\n- range 49, 64\n  Xwin\n- range 57, 72\n  Euryale\n- range 65, 80\n  Xwin\nDo you see that insanity here? Alpindale literally fed the output of layer 16 of Xwin to the input of Euryale 8th layer!To explain this a bit more clearly how stupid this appears to be, let\u2019s revisit the almighty Transformer Architecture:Looking at the left side of the diagram, we see stuff enters at the bottom (\u2018input\u2019 text that has been \u2018chunked\u2019 into small bits of text, somewhere between whole words down to individual letters), and then it flows upwards though the model\u2019s Transformer Blocks (here marked as [1, \u2026, L]), and finally, the model spits out the next text \u2018chunk\u2019 (which is then itself used in the next round of inferencing). What\u2019s actually happening here during these Transformer blocks is quite the mystery. Figuring it out is actually an entire field of AI, \u201cmechanistic interpretability*\u201d.* - yes, it\u2019s more complex than that, samplers etc but that\u2019s enough for this articleOn the right side of the right half of the diagram, do you see that arrow line going from the \u2018Transformer Block Input\u2019 to the (\\oplus ) symbol? That\u2019s why skipping layers makes sense. During training, LLM models can pretty much decide to do nothing in any particular layer, as this \u2018diversion\u2019 routes information around the block. So, \u2018later\u2019 layers can be expected to have seen the input from \u2018earlier\u2019 layers, even a few \u2018steps\u2019 back. Around this time, several groups were experimenting with \u2018slimming\u2019 models down by removing layers. Makes sense, but boring.It\u2019s a pretty fundamental truth in Machine Learning that:A model must be used with the same kind of stuff as it was trained with (we stay \u2018in distribution\u2019)The same holds for each transformer layer. Each Transformer layer learns, during training, to expect the specific statistical properties of the previous layer\u2019s output via gradient descent.And now for the weirdness: There was never the case where any Transformer layer would have seen the output from a future layer!Layer 10 is trained on layer 9\u2019s output distribution. Layer 60 is trained on layer 59\u2019s. If you rearrange them \u2014 feeding layer 60\u2019s output into layer 10 \u2014 you\u2019ve created a distribution the model literally never saw during training.The astounding thing about Goliath wasn\u2019t that it was a huge leap in performance, it was that the damn thing functioned at all. To this day, I still don\u2019t understand why this didn\u2019t raise more eyebrows.Experimentally, this proved that layers were far more interchangeable than anyone had reason to expect. The internal representations were homogenous enough that the model could digest out-of-order hidden states without collapsing. The architecture was far more flexible than a rigid pipeline.Between the Base64 observation and Goliath, I had a hypothesis: Transformers have a genuine functional anatomy. Early layers translate input into abstract representations. Late layers translate back out. And the middle layers, the reasoning cortex, operate in a universal internal language that\u2019s robust to architectural rearrangement. The fact that the layer block size for Goliath 120B was 16-layers made me suspect the input and output \u2018processing units\u2019 sized were smaller than 16 layers. I guessed that Alpindale had tried smaller overlaps, and they just didn\u2019t work.If that was true, maybe I didn\u2019t need to teach a model new facts to make it smarter. I didn\u2019t need fine-tuning. I didn\u2019t need RLHF. I just needed to give it more layers to think with.Building a Brain ScannerOver the following months \u2014 from late 2023 through to mid-2024 \u2014 I built a pipeline to test this hypothesis.The setup was modest. Two RTX 4090s in my basement ML rig, running quantised models through ExLlamaV2 to squeeze 72-billion parameter models into consumer VRAM. The beauty of this method is that you don\u2019t need to train anything. You just need to run inference. And inference on quantized models is something consumer GPUs handle surprisingly well. If a model fits in VRAM, I found my 4090\u2019s were often ballpark-equivalent to H100s.The concept is simple. For a model with N layers, I define a configuration (i,j). The model processes layers 0 to j\u22121 as normal, then loops back and reuses layers i through j\u22121 again, and then the rest to N\u22121. The layers between i and j\u22121 get duplicated in the execution path. No weights are changed. The model just traverses some of its own layers twice.i.e. the pair (2, 7) for a model with 9 transformer blocks would be calculated like so:1\n2\n3\n4\n5\n6\n7\n8\nExample: (i, j) = (2, 7)\n\n      0 \u2192 1 \u2192 2 \u2192 3 \u2192 4 \u2192 5 \u2192 6 \u2500\u2510\n           \u250c\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518\n           \u2514\u2192 2 \u2192 3 \u2192 4 \u2192 5 \u2192 6 \u2192 7 \u2192 8\n\n      duplicated: [2, 3, 4, 5, 6]\n      path: [0, 1, 2, 3, 4, 5, 6, 2, 3, 4, 5, 6, 7, 8]\nBy running through all possible pairs, we can generate a \u2018Brain Scan\u2019, and also see the number of duplicate layers for each set of parameters:1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14\n15\n16\n17\n18\n19\n20\n21\n22\n23\n24\n25\nNumber of duplicated layers for configuration (i, j), with N=9\n\n          end j \u2192\n          0   1   2   3   4   5   6   7   8   9\n        \u250c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u252c\u2500\u2500\u2500\u2510\nstart 0 \u2502 0 \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502 6 \u2502 7 \u2502 8 \u2502 9 \u2502\n  i     \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n  \u2193   1 \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502 6 \u2502 7 \u2502 8 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      2 \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502 6 \u2502 7 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      3 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502 6 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      4 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502 5 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      5 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502 4 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      6 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502 3 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      7 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502 2 \u2502\n        \u251c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u253c\u2500\u2500\u2500\u2524\n      8 \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 . \u2502 1 \u2502\n        \u2514\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2534\u2500\u2500\u2500\u2518\n\nwhere (0,0) is the original model\nFor Qwen2-72B, that means an 80-layer model 3,240 valid (i,j) pairs, plus the original model to test.Variantstotal=(\u2211j=080j)+1=80\u22c5812+1=3241Testing the re-layered model against all six leaderboard benchmarks would take days, so a full sweep would be over a decade of compute on my rig. I needed proxy tasks: probes that were fast, objective, and would reveal structural properties of the model rather than task-specific tricks.The proxies had to satisfy three constraints:Minimal output tokens. With thousands of configurations to sweep, each evaluation needed to be fast. No essays, no long-form generation.Unambiguous scoring. I couldn\u2019t afford LLM-as-judge pipelines. The answer had to be objectively scored without another model in the loop.Orthogonal cognitive demands. If a configuration improves both tasks simultaneously, it\u2019s structural, not task-specific.The Graveyard of Failed ProbesI didn\u2019t arrive at the right probes immediately; it took months of trial and error, and many dead ends.My first instinct was creativity. I had models generate poems, short stories, metaphors, the kind of rich, open-ended output that feels like it should reveal deep differences in cognitive ability. I used an LLM-as-judge to score the outputs, but the results were pretty bad. I managed to fix LLM-as-Judge with some engineering, and the scoring system turned out to be useful later for other things, so here it is:Note: You can skip this section, as it has math. Or notNaive LLM judges are inconsistent. Run the same poem through twice and you get different scores (obviously, due to sampling). But lowering the temperature also doesn\u2019t help much, as that\u2019s only one of many technical issues. So, I developed a full scoring system, based on details on the logits outputs. It can get remarkably tricky. Think about a score from 1-10:We would expect a well calibrated model to have logits that make sense. If the highest weight was on \u20187\u2019, we would expect the rest of the weight to be on \u20186\u2019 and \u20188\u2019 right? but often it\u2019s bimodal, with low weight on 6 and \u20185\u2019, but more weight than expected on \u20184\u2019!We can write \u201810\u2019 in tokens as either \u201810\u2019 or \u20181\u2019 and then \u20180\u2019. It\u2019s not fun to have to calculate the summed probabilities over paths, especially if you wanted to score 1-100.Rather than sampling a single discrete score, I treat the judge\u2019s output as a distribution over valid rating labels and compute the final score as its expectation.To make this practical, I first define a calibrated rubric over the digits 0-9 (there\u2019s only one token for each digit), where each digit corresponds to a clear qualitative description. At the scoring step, I capture the model\u2019s next-token logits and retain only the logits corresponding to those valid digit tokens. This avoids contamination from unrelated continuations such as explanation text, punctuation, or alternate formatting. After renormalizing over the restricted digit set, I interpret the resulting probabilities as a categorical score distribution.Formally, let the valid score set beD={0,1,2,\u2026,9}.Let (zk) denote the model logit assigned to digit (k\u2208D) at the scoring position. The restricted score distribution is thenp(k)=exp\u2061(zk)\u2211m\u2208Dexp\u2061(zm),k\u2208D.The final scalar score is the expected value of this distribution:s^=\u2211k\u2208Dkp(k).This produces a smooth score such as (5.4), rather than forcing the model to commit to a single sampled integer. In practice, this is substantially more stable than naive score sampling and better reflects the model\u2019s uncertainty. It also handles cases where the judge distribution is broad or multimodal. For example, two candidates may both have mean score (5.4), while one has most of its mass tightly concentrated around (5) and (6), and the other splits mass between much lower and much higher ratings. The mean alone is the same, but the underlying judgement is very different.An optional uncertainty estimate can be obtained from the variance of the restricted distribution:Var(s)=\u2211k=09(k\u2212s^)2p(k).In short, the method replaces a noisy sampled judge score with a normalized probability distribution over valid score digits, then uses the expectation of that distribution as the final rating.All this stuff is probably pretty obvious these days, but back in \u201824 there wasn\u2019t much to guide me in developing this method, but unfortunately, I found it was also completely useless\u2026Testing Hard and FastEach configuration needed to generate hundreds of tokens of creative output, and then a separate model had to read and judge each one. With over 3,200 configurations to test for a single 70B model, this would have taken weeks on my dual 4090s.I needed probes where the output was tiny, a few tokens at most, and where scoring was objective and deterministic. No judge model in the loop. That\u2019s what led me to the final two probes:Hard math. Ridiculously difficult questions like: \u201cWhat is the cube root of 74,088,893,247?\u201d No chain-of-thought, or tool use. Just output the number, as a pure leap of intuitive faith.Emotional quotient. Using the EQ-Bench benchmark: complex social scenarios where the model must predict the intensity of specific emotional states. \u201cGiven this situation, how angry/surprised/guilty would this person feel on a scale of 0-100?\u201d Completely different from math. Theory of mind, social inference, empathy. And the output is just a few numbers.I had settled on two maximally orthogonal cognitive tasks, both with tiny outputs. My intuition was this: LLMs think one token at a time, so lets make the model really good at guessing just the next token. But things are never straightforward. Take LLM numbers\u2026LLM Arithmetic is WeirdEven with math probes, I hit unexpected problems. LLMs fail arithmetic in weird ways. They don\u2019t get the answer wrong so much as get it almost right but forget to write the last digit, as if it got bored mid-number. Or they transpose two digits in the middle. Or they output the correct number with a trailing character that breaks the parser.This is probably due to the way larger numbers are tokenised, as big numbers can be split up into arbitrary forms. Take the integer 123456789. A BPE tokenizer (e.g., GPT-style) might split it like: \u2018123\u2019 \u2018456\u2019 \u2018789\u2019 or: \u201812\u2019 \u2018345\u2019 \u201867\u2019 \u201889\u2019A binary right/wrong scoring system would throw away useful signal. Getting a percentage correct would help: \u2018123356789\u2019 instead of \u2018123456789\u2019 would be 99.92% correctBut what about a model that makes a dumb \u2018LLM-mistake\u2019 and outputs 430245 when the answer is 4302459, and has clearly done most of the work? I wrote a custom partial-credit scoring function that pads shorter answers and penalises proportionally:1\n2\n3\n4\n5\n6\n7\n8\n9\n10\n11\n12\n13\n14\n15\n16\n17\n18\n19\n20\n21\n22\n23\ndef calculate_score(actual, estimate):\n    \"\"\"Calculate score comparing actual vs estimated answer.\"\"\"\n    try:\n        actual_str = str(int(actual))\n        estimate_str = str(int(estimate))\n    except (ValueError, OverflowError):\n        return 0\n\n    max_length = max(len(actual_str), len(estimate_str))\n    actual_padded = actual_str.ljust(max_length, \"0\")\n    estimate_padded = estimate_str.ljust(max_length, \"0\")\n    padding_size = max_length - min(len(actual_str), len(estimate_str))\n\n    actual_int = int(actual_padded)\n    estimate_int = int(estimate_padded)\n\n    if max(actual_int, estimate_int) == 0:\n        return 0\n    relative_diff = abs(actual_int - estimate_int) / max(actual_int, estimate_int)\n    correction_factor = 1 - (padding_size / max_length)\n    score = (1 - relative_diff) * correction_factor\n\n    return max(0, min(score, 1))\nThe key idea: pad shorter answers, then penalise via the correction factor. A model that nails 90% of the digits but drops the last one still gets substantial credit \u2014 but less than one that gets every digit. This turned out to be crucial for discriminating between configurations that were close in intuitive math ability.The math questions were hand-crafted initially. I experimented with different operations and scales, then generated random numbers to fill out the dataset. The dataset consisted of 16 questions, where the model was tasked with guesstimating the nearest whole integer number. Here are a few to try yourself, remember no \u2018thinking\u2019 is allowed, guess it directly!1\n2\n3\nWhat is 78313086360375 multiplied by 88537453126609?\nWhat is the cube root of 18228885506341?\nWhat is the cube root of 844178022493, multiplied by 43?\nRYS-XLargeAfter testing several smaller models (Llama\u2019s and smaller Qwen2s), I set up the config for Qwen2-72B and let it sweep. Each (i,j) configuration took a few minutes: load the re-layered model, run the math probe, run the EQ probe, record the scores, move on. Days of continuous GPU time on the 4090s. But far less compute than a fine tune! In fact, I didn\u2019t even have the hardware needed for a LORA fine-tune on just 48GB of VRAM.The optimal configuration was (45,52): layers 0 through 51 run first, then layers 45 through 79 run again. Layers 45 to 51 execute twice. Seven extra layers, near the middle of the 80-layer stack, bringing the total parameter count from 72B to 78B. Every extra layer is an exact copy of an existing one. No new weights or training, just the model repeating itself.Repeating seven layers. That\u2019s all it took, and now I can finally reveal the nomenclature of my models: Repeat Your Self for RYS-XLarge ;)I applied the configuration to MaziyarPanahi\u2019s calme-2.1-qwen2-72b \u2014 a fine-tune of Qwen2-72B \u2014 and uploaded the result as dnhkng/RYS-XLarge. I also applied it to the raw base model as dnhkng/RYS-XLarge-base.Then I submitted to the Open LLM Leaderboard and waited. And waited. Back in the day, the OpenLLM Leaderboard was flooded with dozens of fine-tunes of merges of fine-tunes each day (it was the Wild West), and the waiting list was long. But after a month or so, the results arrived:MetricRYS-XLargeImprovement over baseAverage44.75+2.61%IFEval (0-Shot)79.96-2.05%BBH (3-Shot)58.77+2.51%MATH Lvl 5 (4-Shot)38.97+8.16%GPQA (0-shot)17.90+2.58%MuSR (0-shot)23.72+17.72%MMLU-PRO (5-shot)49.20+0.31%+17.72% on MuSR. +8.16% on MATH. Five out of six benchmarks improved, with only IFEval taking a small hit. The average put it at #1 on the leaderboard.Just to labour the point: I only optimised for one-shot guesstimating hard maths problems and EQ-Bench. I never looked at IFEval, BBH, GPQA, MuSR, or MMLU-PRO during development. The leaderboard was pure out-of-sample validation.A layer configuration found using two narrow, orthogonal probes generalised to everything the Leaderboard threw at it *.* - except IFEval, but that one\u2019s boring anyway, right?That was surprising enough. A brand new way to scale LLMs, developed on some gaming GPUs. But plotting out the heatmaps told an even better story.The Brain ScannerThe original heatmaps that produced RYS-XLarge, showing the Combined delta (math + EQ). The green circle marks the optimal configuration. Red means improvement, blue means degradationThese heatmaps are analogous to functional MRIs of the Transformer, while it is thinking about maths or EQ problems.The x-axis (j) is the end point of the duplicated region. The y-axis (i) is the start point. Each pixel represents a complete evaluation: load the re-layered model, run the math probe, run the EQ probe, score both, record the deltas. As described above, along the central diagonal only a single layer was duplicated. Along the next diagonal towards the top-right, we duplicate two layers, and so on. The single point at the very top-right runs through the entire Transformer stack twice per inference.Let\u2019s examine the math heatmap first. Starting at any layer, and stopping before about layer 60 seems to improve the math guesstimate scores, as shown by the large region with a healthy red blush. Duplicating just the very first layers (the tiny triangle in the top left), messes things up, as does repeating pretty much any of the last 20 layers (the vertical wall of blue on the right). This is more clearly visualised in a skyline plot (averaged rows or columns), and we can see for the maths guesstimates, the starting position of the duplication matters much less. So, the hypothesis that \u2018starting layers\u2019 encode tokens into a smooth \u2018thinking space\u2019, which is finally passed to a dedicated \u2018re-encoding\u2019 system, seems to be somewhat validated.Until we look at the EQ scores:Now things look very different! Duplicating any of the final 10 layers has almost no effect on the scores, but we see complex patterns, where some regions show significant improvement (the area around 45i, 55j), walled between regions of poor performance.But the heatmaps revealed something even more interesting than the location of the thinking bits. They revealed something about its structure.The beginning of LLM Neuroanatomy?Before settling on block duplication, I tried something simpler: take a single middle layer and repeat it n times. If the \u201cmore reasoning depth\u201d hypothesis was correct, this should work. It made sense too, looking at the broad boost in math guesstimate results by duplicating intermediate layer. Give the model extra copies of a particular reasoning layer, get better reasoning. So, I screened them all, looking for a boost.But nope, it almost always did worse. Usually a lot worse, but with occasional small improvements that were within the noise range. Annoying, but taking another look at the complex, blobby patterns in EQ scores gave me another idea:If single-layer duplication doesn\u2019t help, the middle layers aren\u2019t doing independent iterative refinement. They\u2019re not interchangeable copies of the same operation that you can simply \u201crun again.\u201d If they were, duplicating any one of them should give at least a marginal benefit. Instead, those layers are working as a circuit. A multi-step reasoning pipeline that needs to execute as a complete unit.Think of it this way. Layers 46 through 52 aren\u2019t seven workers doing the same job. They\u2019re seven steps in a recipe. Layer 46 takes the abstract representation and performs step one of some cognitive operation \u2014 maybe decomposing a complex representation into subcomponents. Layer 47 takes that output and performs step two \u2014 maybe identifying relationships between the subcomponents. Layer 48 does step three, and so on through layer 52, which produces the final result.Duplicating just one step of this \u2018recipe\u2019 doesn\u2019t bring you much.But duplicating the entire block gives you the full recipe twice. The model runs the complete reasoning circuit, produces a refined intermediate representation, and then runs the same circuit again on its own output. It\u2019s a second pass. A chance to catch what it missed the first time, to refine its abstractions, to push the reasoning one step deeper.Let\u2019s deep-dive into a more current model (that I can experiment with on my system): ExllamaV3 GLM-4.7 from mratsimI\u2019ve marked out a region that boosts maths ability strongly. Notice where it sits? It\u2019s away from the diagonal centre line, which means we\u2019re not looking at single-layer duplications. Starting the repeated block at position 35, we don\u2019t see any improvement until at least position 43. That\u2019s seven layers of not much happening. In fact, we actually see decreased performance by repeating these layers (they are blue, bad!).From end-position 43 to 46, we then see solid boosts in math scores (red = good, yay). But include layer 46 or beyond, and the benefits collapse again. The hypothesis: position 47 is where a different circuit begins. Including even one step of the next recipe messes up the current recipe.So the \u2018math organ\u2019 has boundaries on both sides. Too few layers and you get nothing \u2014 you\u2019ve cut into the circuit and it can\u2019t complete its operation. Too many layers and you also get nothing \u2014 you\u2019ve included tissue from a neighbouring circuit that doesn\u2019t belong. Pre-training carved these structures out of the layer stack, and they only work whole. It also doesn\u2019t translate to other tasks, as the heatmap for EQ scores doesn\u2019t have this patch.This is a much more specific claim than \u201cmiddle layers do reasoning.\u201d It\u2019s saying the reasoning cortex is organised into functional circuits: coherent multi-layer units that perform complete cognitive operations. Each circuit is an indivisible processing unit, and the (i,j) sweeps seen in the heatmap is essentially discovering the boundaries of these circuits.Mechanistic Interpretability via Brain Damage?This also reframes my informal experiments with oobabooga\u2019s Text Generation Web UI. Throughout development, I\u2019d been chatting with various re-layered configurations to see what they felt like in conversation.The good ones were subtly but noticeably sharper. More coherent reasoning, better at holding long context, more natural conversational flow. The kind of difference where you can\u2019t quite articulate what changed, but the model feels more present. Or maybe that\u2019s just my imagination; vibe checks are hard to define.The bad ones went properly unhinged. Some stuttered and fell into degenerate loops. Others developed bizarre personality disorders. One cheerfully announced \u201cLet\u2019s act like cowboys! Yeehaw!\u201d apropos of nothing, and then descended into an unrecoverable giggling fit, generating pages of \u201chahaha\u201d interspersed with cowboy references. \u2018Stoned\u2019 is best way I can describe it. I don\u2019t know if LLMs are \u2018partially conscious\u2019, or could be said to have some \u2018state of mind\u2019, but if so, this one was definitely enjoying itself.These experiments suggest less \u201cslightly worse model\u201d and more \u201cgenuine brain damage.\u201d Which makes sense under the circuit model \u2014 duplicating the wrong circuit is like enlarging a specific region of the brain at the expense of its neighbours. You don\u2019t get a uniformly dumber person. You get someone with a specific neurological deficit. The cowboy model might have had its \u201csocial appropriateness\u201d circuit disrupted by a doubled \u201ccreativity\u201d circuit running unchecked. The stuttering models might have had their decoding circuits pushed out of alignment by extra reasoning depth they couldn\u2019t translate back into coherent tokens.If Transformer reasoning is organised into discrete circuits, it raises a series of fascinating questions. Are these circuits a necessary consequence of the architecture, and emerge from training at scale? Do different model families develop the same circuits in different layer positions, or do they develop fundamentally different architectures?Luckily, I have already done a few; take a look and decide yourself:The AftermathMy method is orthogonal to fine-tuning. Layer duplication changes the architecture; fine-tuning changes the weights. You can stack them. And people went on to score even higher in the Leaderboard:MaziyarPanahi took RYS-XLarge and fine-tuned on top of it, producing calme-2.4-rys-78b. Then dfurman ran ORPO training on that, producing CalmeRys-78B-Orpo-v0.1. MaziyarPanahi continued iterating with calme-3.1 and calme-3.2.As of early 2026, the top four models on the Open LLM Leaderboard are:MaziyarPanahi/calme-3.2-instruct-78b \u2014 52.08MaziyarPanahi/calme-3.1-instruct-78b \u2014 51.29dfurman/CalmeRys-78B-Orpo-v0.1 \u2014 51.23MaziyarPanahi/calme-2.4-rys-78b \u2014 50.77All 78B, and descendants of RYS-XLarge. All built on duplicated middle layers that were discovered using nothing but a handful of hard math and emotional intelligence probes, on a pair of RTX 4090s, in my basement.I never did the fine-tuning myself. It\u2019s not that interesting to me. And I eventually lost interest in the leaderboard. It became increasingly clear that some submissions were training on the test set, and the whole thing was eventually shut down and rebooted. But I know the method is real, because I never used the leaderboard benchmarks for optimisation. The leaderboard was always just validation.Looking Back from 2026In 2024, the model merging community was obsessed with weight interpolation: SLERP, DARE-TIES, linear merges, pass-through layers. The idea was always to combine the learned parameters of different models into something greater than the sum of its parts. mergekit was the tool of choice, and the leaderboard was flooded with creative combinations (making me wait months to get my model benchmarked\u2026).I was doing something different. I wasn\u2019t changing what the model knew. I was changing how it thought. Layer duplication gives the model more iterations through its internal reasoning space without adding any new information. The difference between giving someone a bigger library and giving them more time to think. I was genuinely shocked when I took top spot on the leaderboard; but I think it\u2019s proof that the method probably works.The fact that this worked, and more specifically, that only circuit-sized blocks work, tells us how Transformers organise themselves during training. I now believe they develop a genuine functional anatomy. Early layers encode. Late layers decode. And in the middle, they build circuits: coherent, multi-layer processing units that perform complete cognitive operations. These circuits are indivisible. You can\u2019t speed up a recipe by photocopying one step. But you can run the whole recipe twice.Smaller models seem to be more complex. The encoding, reasoning, and decoding functions are more entangled, spread across the entire stack. I never found a single area of duplication that generalised across tasks, although clearly it was possible to boost one \u2018talent\u2019 at the expense of another. But as models get larger, the functional anatomy becomes more separated. The bigger models have more \u2018space\u2019 to develop generalised \u2018thinking\u2019 circuits, which may be why my method worked so dramatically on a 72B model. There\u2019s a critical mass of parameters below which the \u2018reasoning cortex\u2019 hasn\u2019t fully differentiated from the rest of the brain.With the closure of the HuggingFace LLM leaderboard, and no access to powerful GPUs, I stopped running experiments. But with the flood of new Open Source models (Qwen, MiniMax, GLM, and more), and finally having just enough compute at home, I have started working on the current batch of LLMs. The heatmaps keep coming back with the same general story, but every architecture has its own neuroanatomy. The brains are different. The principle is the same. And some models are looking really interesting (Qwen3.5 27B in particular). I will release the code along with uploading new RYS models and a blog post once my Hopper-system finishes grinding on MiniMax M2.5.Nvidia: Sponsor this project by sending me GB300 Grace Blackwell Ultra Desktop, as I NEED MOAR VRAMOne More ThingRemember the architecture?1\n2\n3\n4\n5\n6\n7\n8\nExample: (i, j) = (2, 7)\n\n      0 \u2192 1 \u2192 2 \u2192 3 \u2192 4 \u2192 5 \u2192 6 \u2500\u2510\n           \u250c\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2518\n           \u2514\u2192 2 \u2192 3 \u2192 4 \u2192 5 \u2192 6 \u2192 7 \u2192 8\n\n      duplicated: [2, 3, 4, 5, 6]\n      path: [0, 1, 2, 3, 4, 5, 6, 2, 3, 4, 5, 6, 7, 8]\nWe have one horrible disjuncture, between layers 6 \u2192 2. I have one more hypothesis: A little bit of fine-tuning on those two layers is all we really need. Fine-tuned RYS models dominate the Leaderboard. I suspect this junction is exactly what the fine-tuning fixes. And there\u2019s a great reason to do this: this method does not use extra VRAM! For all these experiments, I duplicated layers via pointers; the layers are repeated without using more GPU memory. Of course, we do need more compute and more KV cache, but that\u2019s a small price to pay for a verifiably better model. We can just \u2018fix\u2019 an actual copies of layers 2 and 6, and repeat layers 3-4-5 as virtual copies. If we fine-tune all the layer, we turn virtual copies into real copies, and use up more VRAM.Twenty years ago I was a PhD student dissecting rat brains. I never expected to end up performing brain surgery on artificial minds.Special thanks to my wife, for putting up with months of evenings and weekends spent staring at heatmaps in the basement. And to the appliedAI Institute for the H100 compute that helped scale these experiments.Citing this work1\n2\n3\n4\n5\n6\n7\n@article{ng2026rys,\n  title   = {LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight},\n  author  = {Ng, David Noel},\n  year    = {2026},\n  month   = {March},\n  url     = {https://dnhkng.github.io/posts/rys/}\n}"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:53:47.496500+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://dnhkng.github.io/posts/rys/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-04-24T07:53:47.424470+00:00",
                "index_texts": null,
                "output": "LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight | David Noel Ng",
                "pwd": "/data/archive/1777017200.859394",
                "schema": "ArchiveResult",
                "start_ts": "2026-04-24T07:53:47.326947+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "https://web.archive.org/web/20260424075430/https://dnhkng.github.io/posts/rys/",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight | David Noel Ng",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1777017200.859394",
    "newest_archive_date": "2026-04-24T07:54:06.135270+00:00",
    "num_failures": 0,
    "num_outputs": 9,
    "oldest_archive_date": "2026-04-24T07:53:27.100471+00:00",
    "path": "/posts/rys/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KPZ7N88725E6AB0301976T4G",
    "snapshot_id": "5b6895a0-1eea-4adc-a90a-d93dd2736890",
    "sources": [
        "/data/sources/1777017200-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1777017200.859394",
    "title": "LLM Neuroanatomy: How I Topped the LLM Leaderboard Without Changing a Single Weight | David Noel Ng",
    "url": "https://dnhkng.github.io/posts/rys/"
}