{
    "archive_path": "archive/1779035615.091695",
    "base_url": "www.seangoedecke.com/steering-vectors",
    "basename": "",
    "bookmarked_date": "2026-05-17 16:33",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/www.seangoedecke.com/steering-vectors",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=www.seangoedecke.com",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "www.seangoedecke.com",
    "downloaded_at": "2026-05-17T16:33:43.473256+00:00",
    "downloaded_datestr": "2026-05-17 16:33",
    "extension": "",
    "hash": "B7KC5A2E62GYVXDVM0GB",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://www.seangoedecke.com/steering-vectors/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-05-17T16:34:45.049146+00:00",
                "index_texts": null,
                "output": "ArchiveError: Failed to find \"content-location\" URL header in Archive.org response.",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:34:33.381312+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://www.seangoedecke.com/steering-vectors/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2026-05-17T16:34:03.217607+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:33:54.806965+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=www.seangoedecke.com"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-05-17T16:33:46.821514+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:33:43.887449+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.seangoedecke.com/steering-vectors/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-05-17T16:33:46.981003+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:33:46.879454+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2026-05-17T16:34:24.654313+00:00",
                "index_texts": [
                    "(/favicon-32x32.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/manifest.webmanifest) (/icons/icon-48x48.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-72x72.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-96x96.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-144x144.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-192x192.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-256x256.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-384x384.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) (/icons/icon-512x512.png?v=ac7bb3aa286bd21c42741d9c9aa60cb7) DeepSeek-V4-Flash means LLM steering is interesting again (seangoedecke.com RSS feed) (/rss.xml) (seangoedecke.com RSS feed) (/feed.xml) (seangoedecke.com RSS feed) (/atom.xml) (/webpack-runtime-b8d0f2e0df3c60c0b15d.js) (/framework-7752ef71ff976174a5db.js) (/styles-72fb21565c5bfe3db587.js) (/app-d15dace2eb1904b9d918.js) (/commons-814c64de73b8ea4567bc.js) (/component---src-templates-blog-post-js-70d3284f99b18947a6db.js) (/page-data/steering-vectors/page-data.json) (/page-data/sq/d/1146911855.json) (/page-data/sq/d/3764592887.json) (/page-data/app-data.json) (/page-data/tags/ai/page-data.json) (/page-data/tags/steering/page-data.json) (/page-data/index/page-data.json) (/component---src-templates-tag-page-js-0fea5171caee3f4c4080.js) (/component---src-templates-tag-page-js-0fea5171caee3f4c4080.js) (/component---src-templates-blog-list-js-baf1178fec4006be3549.js)  (/) sean goedecke  May 16, 2026\u2502(/tags/ai/) ai , (/tags/steering/) steering   DeepSeek-V4-Flash means LLM steering is interesting again  Ever since (https://www.anthropic.com/news/golden-gate-claude) Golden Gate Claude I\u2019ve been fascinated with \u201csteering\u201d: the idea that you can guide LLM outputs by directly manipulating the activations of the model mid-flight. DeepSeek V4 Flash I was inspired to write this post by antirez\u2019s recent project (https://github.com/antirez/ds4/tree/main) DwarfStar 4 , which is a version of (https://github.com/ggml-org/llama.cpp) llama.cpp that\u2019s been stripped down to run only DeepSeek-V4-Flash. What\u2019s so special about this model? It might be what many engineers have been waiting for: a local model good enough to compete with at least the low end of frontier model agentic coding. Since steering requires a local model, it\u2019s now practical for many engineers to try it out for the first time. And indeed, antirez has baked (https://github.com/antirez/ds4/tree/main/dir-steering) steering into DwarfStar 4 as a first-class citizen. Right now it\u2019s very rudimentary (basically just the toy \u201cverbosity\u201d example you can replicate via prompting), but the initial release was only (https://github.com/antirez/ds4/commit/d997b56c151184bcff469dd8302ed97f23481024) eight days ago . I plan to follow this project closely. How steering works The basic idea behind steering is extracting a concept (like \u201crespond tersely\u201d) from the model\u2019s internal brain state, then reaching in during inference and boosting the numerical activations that form that concept. One way you might do this is to feed your model the same set of a hundred prompts twice, once with the normal prompts and once with the words \u201crespond tersely\u201d appended. Then measure the difference in the model\u2019s activations1  for each prompt pair (by subtracting one activation matrix from the other). That\u2019s your \u201csteering vector\u201d. In theory, you can go and add that to the same activation layer for any prompt and get the same effect (of the model responding tersely). Another, more sophisticated way you might do this is to train a second model to extract \u201cfeatures\u201d from your model\u2019s activations: patterns of behavior that seem to show up together. Then you can try to map those features back to individual concepts, and boost them in the same way. This is more or less what Anthropic is doing with (https://transformer-circuits.pub/2024/scaling-monosemanticity/index.html) sparse autoencoders 2  . It\u2019s the same principle as the naive approach, but it lets you capture deeper patterns (at the cost of being much more expensive in time, compute and expertise). Why steering is interesting Steering sounds like a cheat code. Instead of painstakingly assembling a training set that tries to push the model towards the \u201csmart\u201d end of the distribution in its training data, why not simply go uncover the \u201csmart\u201d dial in the model\u2019s brain and turn it all the way to the right? It also seems like a more elegant way to adjust the way models talk. Instead of fiddling with the prompt (adding or removing qualifiers like \u201cyou MUST\u201d), couldn\u2019t we just have a control panel of sliders like \u201csuccinctness/verbosity\u201d or \u201cconscientiousness/speed\u201d and move them around directly? Finally, it\u2019s just cool . Watching Golden Gate Claude unwillingly (https://www.anthropic.com/news/golden-gate-claude) drag every sentence back to the Golden Gate Bridge is as fascinating and unsettling as Oliver Sacks\u2019 neurological (https://en.wikipedia.org/wiki/The_Man_Who_Mistook_His_Wife_for_a_Hat) anecdotes . What if your own mind was tweaked in a similar way? Would it still be you? Why steering hasn\u2019t been used Why don\u2019t we steer more, then? Why don\u2019t ChatGPT and Claude Code already have a steering panel where you can adjust the model\u2019s brain in real time? One reason is that steering is kind of an unfortunately \u201cmiddle class\u201d idea in AI research. It\u2019s beneath the big AI labs, who can manipulate their models directly without having to do awkward brain surgery mid-inference. Anthropic is working on this stuff, but largely from an interpretability and safety perspective (as far as I know). When they want a model to behave in a certain way, they don\u2019t mess around with steering, they just train the model. Steering is also out of reach for regular AI users like you and me3  , who use LLMs via an API and thus don\u2019t have access to the model weights or activations needed to steer the model. Only OpenAI can identify or expose steering vectors for GPT-5.5, for instance. We could do this for open-weights models, but until very recently (more on that later) there haven\u2019t been any open models strong enough to be worth doing this for. On top of that, most basic applications of steering are outcompeted by just prompting the model. It sounds pretty impressive to be able to manipulate the model\u2019s brain directly. But you know what else manipulates the model\u2019s brain directly? Prompt tokens. You can exercise fairly fine-grained control over activations with steering, but you can already exercise extremely fine-grained control by tweaking the language of your prompt. In other words, there\u2019s not much point going to the trouble to steer a model to be more verbose when you could simply ask . Steering the unpromptable One way for steering to be really useful is if we could identify a concept that can\u2019t be prompted for. What about \u201cintelligence\u201d? You used to be able to prompt for intelligence - this is why 4o-era prompting always began with \u201cyou are an expert\u201d - but current-generation models have that baked into their personalities, so prompting for it does nothing. Maybe steering for it would still work? Ultimately this is an empirical question, but I\u2019m skeptical that we\u2019ll be able to find an \u201cintelligence\u201d steering vector. Put another way, the steering vector that makes up a concept as difficult as \u201cintelligence\u201d might be almost coextensive with the entire set of weights of the model, and thus identifying it reduces to the problem of \u201ctraining a smart model\u201d. A sufficiently sophisticated steering approach ends up just replacing the actual model. If I take GPT-2, and at each layer I swap out the activations with the activations from a much stronger model with the same architecture, I will get a much better result. But at that point you\u2019re not making GPT-2 more intelligent, you\u2019re just talking to the stronger model instead. The intelligence is in the steering, not in the model. For much more on this, see my post (/philosophy-and-ai-interpretability/) AI interpretability has the same problems as philosophy of mind .  Steering as data compression Another way for steering to be useful is if we could somehow steer for a concept that requires a ton of tokens to express. Steering would thus save us a big chunk of the model\u2019s context window. Intuitively, we might think of this as a way to shift a concept from the model\u2019s working memory into its implicit memory. For instance, what if we could identify a \u201cknowledge of my particular codebase\u201d concept? When GPT-5.5 speed-reads my codebase, some of that knowledge it gains has to be buried in the activations, right? Maybe we could drag that out into a very large steering vector. I would be surprised if this could work. I think we\u2019ll run into the same problem as with extracting \u201cintelligence\u201d: the \u201cknows my codebase\u201d concept is probably sophisticated enough to require a full fine-tune of the model4  . But it at least seems possible. Conclusion I\u2019m fascinated with steering, but I\u2019m not particularly optimistic about it. I think most of the gains can be more efficiently reproduced with prompts, and that the truly ambitious steering goals can be more efficiently reproduced by training or fine-tuning the model. However, the open-source community hasn\u2019t done a lot of work on steering yet, and that might be just starting to change now. If I\u2019m wrong and it does have practical applications, we should find that out in the next six months. It\u2019ll be interesting to see if bespoke per-model tools like DwarfStar 4 end up including a \u201clibrary\u201d of boostable features. When a popular open-weights model is released, the community always rushes to release a suite of wrappers and quantized versions. Could we also see a rush to extract boostable features from the model? edit: this post got some comments on (https://news.ycombinator.com/item?id=48160807) Hacker News . Several commenters (including antirez himself) (https://news.ycombinator.com/item?id=48161688) pointed out that steering can change some \u201ctrained in\u201d behavior in ways that prompting can\u2019t: most notably to remove refusal from the model. Another commenter (https://news.ycombinator.com/item?id=48161488) says that this is how uncensoring/abliteration is already done for open models. I didn\u2019t know that - I thought the uncensored models were typically LoRA fine-tunes. On this point, antirez (https://news.ycombinator.com/item?id=48161688) noted that modifying the weights can damage model capabilities more than the more lightweight runtime-steering approach (which can only be applied when needed). Makes sense to me. Models have lots of different activations you might measure (after attention, between each layer, etc). You can basically pick any one you want, or try multiple and see what works best. \u21a9  I recently read a really good (https://huggingface.co/spaces/dlouapre/eiffel-tower-llama) deep dive into doing this with an open LLaMA model (and I (https://github.com/sgoedecke/skills/blob/main/skills/extract-features-clamp-inference/SKILL.md) tried it myself a few months ago, with mixed results.) \u21a9  Apologies to my readers from the big AI labs. Please email me if you have tried steering internally to boost capabilities and it hasn\u2019t worked. I promise I won\u2019t tell anyone. \u21a9  And even then, the results of \u201cfine tune a model on your codebase\u201d in the industry have largely been unsuccessful. \u21a9     If you liked this post, consider(https://buttondown.com/seangoedecke) subscribing to email updates about my new posts, or(https://news.ycombinator.com/submitlink?u=https://www.seangoedecke.com/steering-vectors/&t=DeepSeek-V4-Flash means LLM steering is interesting again) sharing it on Hacker News . Here's a preview of a related post that shares tags with this one. LLM-generated skills work, if you generate them afterwards LLM (https://github.com/anthropics/skills) \u201cskills\u201d are a short explanatory prompt for a particular task, typically bundled with helper scripts. A recent (https://arxiv.org/abs/2602.12670) paper showed that while skills are useful to LLMs, LLM-authored skills are not. From the abstract: Self-generated skills provide no benefit on average, showing that models cannot reliably author the procedural knowledge they benefit from consuming For the moment, I don\u2019t really want to dive into the paper. I just want to note that the way the paper uses LLMs to generate skills is bad, and you shouldn\u2019t do this. Here\u2019s how the paper prompts a LLM to produce skills:(/generate-skills-afterwards/) Continue reading...    (https://buttondown.com/seangoedecke) subscribe \u2502 (/about) about \u2502 (/podcasts) podcasts \u2502 (/popular) popular \u2502 (/tags) tags \u2502 (/rss.xml) rss      Search seangoedecke.com () (Search posts) Search         "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:34:24.618790+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://www.seangoedecke.com/steering-vectors/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2026-05-17T16:34:33.290905+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:34:27.281066+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://www.seangoedecke.com/steering-vectors/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2026-05-17T16:34:24.457188+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:34:18.960430+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpoeaxd72z",
                    "https://www.seangoedecke.com/steering-vectors/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2026-05-17T16:34:07.453694+00:00",
                "index_texts": [
                    "Ever since Golden Gate Claude I\u2019ve been fascinated with \u201csteering\u201d: the idea that you can guide LLM outputs by directly manipulating the activations of the model mid-flight.\nDeepSeek V4 Flash\nI was inspired to write this post by antirez\u2019s recent project DwarfStar 4, which is a version of llama.cpp that\u2019s been stripped down to run only DeepSeek-V4-Flash. What\u2019s so special about this model? It might be what many engineers have been waiting for: a local model good enough to compete with at least the low end of frontier model agentic coding.\nSince steering requires a local model, it\u2019s now practical for many engineers to try it out for the first time. And indeed, antirez has baked steering into DwarfStar 4 as a first-class citizen. Right now it\u2019s very rudimentary (basically just the toy \u201cverbosity\u201d example you can replicate via prompting), but the initial release was only eight days ago. I plan to follow this project closely.\nHow steering works\nThe basic idea behind steering is extracting a concept (like \u201crespond tersely\u201d) from the model\u2019s internal brain state, then reaching in during inference and boosting the numerical activations that form that concept.\nOne way you might do this is to feed your model the same set of a hundred prompts twice, once with the normal prompts and once with the words \u201crespond tersely\u201d appended. Then measure the difference in the model\u2019s activations1 for each prompt pair (by subtracting one activation matrix from the other). That\u2019s your \u201csteering vector\u201d. In theory, you can go and add that to the same activation layer for any prompt and get the same effect (of the model responding tersely).\nAnother, more sophisticated way you might do this is to train a second model to extract \u201cfeatures\u201d from your model\u2019s activations: patterns of behavior that seem to show up together. Then you can try to map those features back to individual concepts, and boost them in the same way. This is more or less what Anthropic is doing with sparse autoencoders2. It\u2019s the same principle as the naive approach, but it lets you capture deeper patterns (at the cost of being much more expensive in time, compute and expertise).\nWhy steering is interesting\nSteering sounds like a cheat code. Instead of painstakingly assembling a training set that tries to push the model towards the \u201csmart\u201d end of the distribution in its training data, why not simply go uncover the \u201csmart\u201d dial in the model\u2019s brain and turn it all the way to the right?\nIt also seems like a more elegant way to adjust the way models talk. Instead of fiddling with the prompt (adding or removing qualifiers like \u201cyou MUST\u201d), couldn\u2019t we just have a control panel of sliders like \u201csuccinctness/verbosity\u201d or \u201cconscientiousness/speed\u201d and move them around directly?\nFinally, it\u2019s just cool. Watching Golden Gate Claude unwillingly drag every sentence back to the Golden Gate Bridge is as fascinating and unsettling as Oliver Sacks\u2019 neurological anecdotes. What if your own mind was tweaked in a similar way? Would it still be you?\nWhy steering hasn\u2019t been used\nWhy don\u2019t we steer more, then? Why don\u2019t ChatGPT and Claude Code already have a steering panel where you can adjust the model\u2019s brain in real time? One reason is that steering is kind of an unfortunately \u201cmiddle class\u201d idea in AI research.\nIt\u2019s beneath the big AI labs, who can manipulate their models directly without having to do awkward brain surgery mid-inference. Anthropic is working on this stuff, but largely from an interpretability and safety perspective (as far as I know). When they want a model to behave in a certain way, they don\u2019t mess around with steering, they just train the model.\nSteering is also out of reach for regular AI users like you and me3, who use LLMs via an API and thus don\u2019t have access to the model weights or activations needed to steer the model. Only OpenAI can identify or expose steering vectors for GPT-5.5, for instance. We could do this for open-weights models, but until very recently (more on that later) there haven\u2019t been any open models strong enough to be worth doing this for.\nOn top of that, most basic applications of steering are outcompeted by just prompting the model. It sounds pretty impressive to be able to manipulate the model\u2019s brain directly. But you know what else manipulates the model\u2019s brain directly? Prompt tokens. You can exercise fairly fine-grained control over activations with steering, but you can already exercise extremely fine-grained control by tweaking the language of your prompt. In other words, there\u2019s not much point going to the trouble to steer a model to be more verbose when you could simply ask.\nSteering the unpromptable\nOne way for steering to be really useful is if we could identify a concept that can\u2019t be prompted for. What about \u201cintelligence\u201d? You used to be able to prompt for intelligence - this is why 4o-era prompting always began with \u201cyou are an expert\u201d - but current-generation models have that baked into their personalities, so prompting for it does nothing. Maybe steering for it would still work?\nUltimately this is an empirical question, but I\u2019m skeptical that we\u2019ll be able to find an \u201cintelligence\u201d steering vector. Put another way, the steering vector that makes up a concept as difficult as \u201cintelligence\u201d might be almost coextensive with the entire set of weights of the model, and thus identifying it reduces to the problem of \u201ctraining a smart model\u201d.\nA sufficiently sophisticated steering approach ends up just replacing the actual model. If I take GPT-2, and at each layer I swap out the activations with the activations from a much stronger model with the same architecture, I will get a much better result. But at that point you\u2019re not making GPT-2 more intelligent, you\u2019re just talking to the stronger model instead. The intelligence is in the steering, not in the model. For much more on this, see my post AI interpretability has the same problems as philosophy of mind.\nSteering as data compression\nAnother way for steering to be useful is if we could somehow steer for a concept that requires a ton of tokens to express. Steering would thus save us a big chunk of the model\u2019s context window. Intuitively, we might think of this as a way to shift a concept from the model\u2019s working memory into its implicit memory.\nFor instance, what if we could identify a \u201cknowledge of my particular codebase\u201d concept? When GPT-5.5 speed-reads my codebase, some of that knowledge it gains has to be buried in the activations, right? Maybe we could drag that out into a very large steering vector.\nI would be surprised if this could work. I think we\u2019ll run into the same problem as with extracting \u201cintelligence\u201d: the \u201cknows my codebase\u201d concept is probably sophisticated enough to require a full fine-tune of the model4. But it at least seems possible.\nConclusion\nI\u2019m fascinated with steering, but I\u2019m not particularly optimistic about it. I think most of the gains can be more efficiently reproduced with prompts, and that the truly ambitious steering goals can be more efficiently reproduced by training or fine-tuning the model.\nHowever, the open-source community hasn\u2019t done a lot of work on steering yet, and that might be just starting to change now. If I\u2019m wrong and it does have practical applications, we should find that out in the next six months.\nIt\u2019ll be interesting to see if bespoke per-model tools like DwarfStar 4 end up including a \u201clibrary\u201d of boostable features. When a popular open-weights model is released, the community always rushes to release a suite of wrappers and quantized versions. Could we also see a rush to extract boostable features from the model?\nedit: this post got some comments on Hacker News. Several commenters (including antirez himself) pointed out that steering can change some \u201ctrained in\u201d behavior in ways that prompting can\u2019t: most notably to remove refusal from the model. Another commenter says that this is how uncensoring/abliteration is already done for open models. I didn\u2019t know that - I thought the uncensored models were typically LoRA fine-tunes. On this point, antirez noted that modifying the weights can damage model capabilities more than the more lightweight runtime-steering approach (which can only be applied when needed). Makes sense to me.\nHere's a preview of a related post that shares tags with this one."
                ],
                "output": "readability/",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:34:04.391789+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.seangoedecke.com/steering-vectors/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-05-17T16:34:03.291860+00:00",
                "index_texts": null,
                "output": "DeepSeek-V4-Flash means LLM steering is interesting again",
                "pwd": "/data/archive/1779035615.091695",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T16:34:03.271538+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "ArchiveError: Failed to find \"content-location\" URL header in Archive.org response.",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "DeepSeek-V4-Flash means LLM steering is interesting again",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1779035615.091695",
    "newest_archive_date": "2026-05-17T16:34:33.381312+00:00",
    "num_failures": 1,
    "num_outputs": 8,
    "oldest_archive_date": "2026-05-17T16:33:43.887449+00:00",
    "path": "/steering-vectors/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KRVCJCEJD08341A6018MMDB8",
    "snapshot_id": "80b96fc1-bf39-48a8-bf5f-12dbd14a3568",
    "sources": [
        "/data/sources/1779035614-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1779035615.091695",
    "title": "DeepSeek-V4-Flash means LLM steering is interesting again",
    "url": "https://www.seangoedecke.com/steering-vectors/"
}