{
    "archive_path": "archive/1769152883.742596",
    "base_url": "rijnard.com/blog/the-code-only-agent",
    "basename": "the-code-only-agent",
    "bookmarked_date": "2026-01-23 07:21",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/rijnard.com/blog/the-code-only-agent",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=rijnard.com",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "rijnard.com",
    "downloaded_at": "2026-01-23T07:21:29.588317+00:00",
    "downloaded_datestr": "2026-01-23 07:21",
    "extension": "",
    "hash": "1BMQ6EEA97MA58C2AQGH",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://rijnard.com/blog/the-code-only-agent"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-01-23T07:22:32.486700+00:00",
                "index_texts": null,
                "output": "https://web.archive.org/web/20260123072215/https://rijnard.com/blog/the-code-only-agent",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:22:11.090023+00:00",
                "status": "succeeded"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://rijnard.com/blog/the-code-only-agent"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2026-01-23T07:21:48.185250+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:21:39.284892+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=rijnard.com"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-01-23T07:21:33.369948+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:21:29.916597+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://rijnard.com/blog/the-code-only-agent"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-01-23T07:21:33.535760+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:21:33.391776+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2026-01-23T07:22:03.865621+00:00",
                "index_texts": [
                    "The Code-Only Agent \u2022 Rijnard van Tonder (../assets/normalize.css) (../assets/main.css) (//maxcdn.bootstrapcdn.com/font-awesome/4.2.0/css/font-awesome.min.css) (https://fonts.googleapis.com/css?family=Vollkorn) (https://rijnard.com/blog/the-code-only-agent)  (../index.html) Rijnardvan Tonder  (https://x.com/rvtond) \ud835\udd4f @rvtond   (../index.html) Home (index.html) \u2190 Back to all posts   The Code-Only Agent When Code Execution Really is All You Need  (Code-Only Agent)  If you're building an agent, you're probably overwhelmed. Tools.\n            MCP. Subagents. Skills. The ecosystem pushes you toward complexity,\n            toward \"the right way\" to do things. You should know: Concepts like\n            \"Skills\" and \"MCP\" are actually outcomes of an ongoing learning process of humans figuring stuff out. The\n            space is wide open for exploration. With this mindset I\n            wanted to try something different. Simplify the assumptions.  What if the agent only had one tool   ? Not just any tool, but the most powerful one. The Turing-complete  one: execute code  .  Truly one tool means: no `bash`, no `ls`, no `grep`. Only execute_code  . And you enforce it.  When you watch an agent run, you might think: \"I wonder what tools\n            it'll use to figure this out. Oh look, it ran `ls`. That makes\n            sense. Next, `grep`. Cool.\"  The simpler Code-Only paradigm makes that question irrelevant. The\n            question shifts from \"what tools?\" to \"what code will it produce?\"\n            And that's when things get interesting.  execute_code  : One Tool to Rule Them All Traditional prompting works like this: > Agent, do thing  > Agent responds   with thing   Contrast with: > Agent, do thing  > Agent creates and runs code   to do thing   It does this every time. No, really, every   time. Pick a runtime for our Code-Only agent, say Python. It needs\n            to find a file? It writes Python code to find the file and executes\n            the code. Maybe it runs rglob  . Maybe it does os.walk  .  It needs to create a script that crawls a website? It doesn't write\n            the script to your filesystem (reminder: there's no create_file  tool to do that!). It writes code to output a script that crawls a website  .1   We make it so that there is literally no way for the agent to do  anything productive without writing code  .  So what? Why do this? You're probably thinking, how is this useful?\n            Just give it `bash` tool already man.  Let's think a bit more deeply what's happening. Traditional agents\n            respond with something. Tell it to find some DNA pattern across 100\n            files. It might `ls` and `grep`, it might do that in some\n            nondeterministic order, it'll figure out an answer  and\n            maybe you continue interacting because it missed a directory or you\n            added more files. After some time, you end up with a conversation of\n            tool calls, responses, and an answer.  At some point the agent might even write a Python script to do this\n            DNA pattern finding. That would be a lucky happy path, because we\n            could rerun that script or update it later... Wait, that's handy...\n            actually, more than handy... isn't that ideal   ? Wouldn't it be better if we told it to write a script at the\n            start? You see, the Code-Only agent doesn't need to be told to write\n            a script. It has   to, because that's literally the only way for it to do anything of\n            substance.  The Code-Only agent produces something more precise than an answer\n            in natural language. It produces a code witness  of an\n            answer. The answer is the output from running the code. The agent\n            can interpret that output in natural language (or by writing code),\n            but the \"work\" is codified in a very literal sense. The Code-Only\n            agent doesn't respond with something. It produces a code witness\n            that outputs something.  Try \u276f\u276f (https://github.com/rvantonder/execute_code_py) Code-Only plugin for Claude Code   Code witnesses are semantic guarantees Let's follow the consequences. The code witness must abide by\n            certain rules: The rules imposed by the language runtime semantics\n            (e.g., of Python). That's not a \"next token\" process. That's not a\n            \"LLM figures out sequence of tool calls, no that's not what I\n            wanted\". It's piece of code. A piece of code! Our one-tool agent has\n            a wonderful property: It went through latent space to produce\n            something that has a defined semantics, repeatably runnable, and\n            imminently comprehensible (for humans or agents alike to reason\n            about). This is nondeterministic LLM token-generation projected into\n            the space of Turing-complete code, an executable description of\n            behavior as we best understand it.  Is a Code-Only agent really enough, or too extreme? I'll be frank: I\n            pursued this extreme after two things (1) inspiration from articles in Further Reading below (2) being annoyed at agents for not comprehensively and\n            exhaustively analyzing 1000s of files on my laptop. They would skip,\n            take shortcuts, hallucinate. I knew how to solve part of that\n            problem: create a programmatic   loop and try have fresh instances/prompts to do the work\n            comprehensively. I can rely on the semantics of a loop written in\n            Python. Take this idea further, and you realize that for anything\n            long-running and computable (e.g., bash or some tool), you actually\n            want the real McCoy: the full witness of code, a trace of why things\n            work or don't work. The Code-Only agent enforces   that principle.  Code-Only agents are not too extreme. I think they're the only way\n            forward for computable things. If you're writing travel blog posts,\n            you accept the LLMs answer (and you don't need to run tools for\n            that). When something is computable though, Code-Only is the only\n            path to a fully trustworthy   way to make progress where you need guarantees (subject to\n            the semantics that your language of choice guarantees, of course). When I say\n            guarantees, I mean that in the looser sense, and also in a (https://en.wikipedia.org/wiki/Formal_verification) Formal  sense. Which beckons: What happens when we use a language like (https://lean-lang.org/) Lean  with some of the\n            strongest guarantees? Did we not observe that (https://en.wikipedia.org/wiki/Curry%E2%80%93Howard_correspondence) programs are proofs  ?  This lens says the Code-Only agent is a producer of proofs,\n            witnesses of computational behavior in the world of\n            proofs-as-programs. An LLM in a loop forced to produce proofs, run\n            proofs, interpret proof results. That's all.  Going Code-Only So you want to go Code-Only. What happens? The paradigm is simple,\n            but the design choices are surprising.  First, the harness. The LLM's output is code, and you execute that\n            code. What should be communicated back? Exit code makes sense. What\n            about output? What if the output is very large? Since you're running\n            code, you can specify the result type that running the code should\n            return.  I've personally, e.g., had the tool return results directly if under\n            a certain threshold (1K bytes). This would go into the session context.\n\t\t\t\t\t\tAlternatively, write the results to a JSON\n            file on disk if it exceeds the threshold. This avoids context blowup and the result tells the\n            agent about the output file path written to disk. How best to pass\n            results, persist them, and optimize for size and context fill are\n            open questions. You also want to define a way to deal with `stdout`\n            and `stderr`: Do you expose these to the agent? Do you summarize\n            before exposing?  Next, enforcement. Let's say you're using Claude Code. It's not\n            enough to persuade it to always create and run code. It turns out\n            it's surprisingly twisty to force Claude Code into a single tool\n            (maybe support for this will improve). The best plugin-based\n            solution I found is a tool PreHook that catches banned tool uses.\n            This wastes some iterations when Claude Code tries to use a tool\n            that's not allowed, but it learns to stop attempting filesystem\n            reads/writes. An initial prompt helps direct.  Next, the language runtime. Python, TypeScript, Rust, Bash. Any\n            language capable of being executed is fair game, but you'll need to\n            think through whether it works for your domain. Dynamic languages\n            like Python are interesting because you can run code natively in the\n            agent's own runtime, rather than through subprocess calls. Likewise\n            TypeScript/JS can be injected into TypeScript-based agents (see Further Reading ).  Once you get into the Code-Only mindset, you'll see the potential\n            for composition and reuse. Claude Skills define reusable processes\n            in natural language. What's the equivalent for a Code-Only agent?\n            I'm not sure a Skills equivalent exists yet, but I anticipate it\n            will take shape soon: code as building blocks for specific domains\n            where Code-Only agents compose programmatic patterns. How is that\n            different from calling APIs? APIs form part of the reusable blocks,\n            but their composition (loops, parallelism, asynchrony) is what a\n            Code-Only agent generates.  What about heterogeneous languages and runtimes for our `execute_tool`? I don't think we've thought that far yet.  Further Reading The agent landscape is quickly evolving. My thoughts on how the\n            Code-Only paradigm fits into inspiring articles and trends, from\n            most recent and going back:  (https://prose.md) prose.md   (Jan 2026) \u2014 Code-Only reduces prompts to executable code (with\n              loops and statement sequences). Prose expands prompts into natural\n              language with program-like constructs (also loops, sequences,\n              parallelism). The interplay of natural language for agent\n              orchestration and rigid semantics for agent execution could be\n              extremely powerful.  (https://steve-yegge.medium.com/welcome-to-gas-town-4f25ee16dd04) Welcome to Gas Town   (Jan 2026) \u2014 Agent orchestration gone berserk. Tool running is the low-level\n              operation at the bottom of the agent stack. Code-Only fits as the\n              primitive: no matter how many agents you orchestrate, each one\n              reduces to generating and executing code.  (https://www.anthropic.com/engineering/code-execution-with-mcp) Anthropic Code Execution with MCP article   (Nov 2025) \u2014 MCP-centric view of exposing MCP servers as code API\n              and not tool calls. Code-Only is simpler and more general. It\n              doesn't care about MCP, and casting the MCP interface as an API is\n              a mechanical necessity that acknowledges the power of going\n              Code-Only.  (https://www.anthropic.com/engineering/equipping-agents-for-the-real-world-with-agent-skills) Anthropic Agent Skills article   (Oct 2025) \u2014 Skills embody reusable processes framed in natural\n              language. They can generate and run code, but that's not their\n              only purpose. Code-Only is narrower (but computationally\n              all-powerful): the reusable unit is always executable. The analog\n              to Skills manifests as pluggable executable pieces: functions,\n              loops, composable routines over APIs.  (https://blog.cloudflare.com/code-mode/) Cloudflare Code Mode article   (Sep 2025) \u2014 Possibly the earliest concrete single-code-tool\n              implementation. Code Mode converts MCP tools into a TypeScript API\n              and gives the agent one tool: execute TypeScript. Their insight is\n              pragmatic: LLMs write better code than tool calls because of\n              training data. In its most general sense, going Code-Only doesn't\n              need to rely on MCP or APIs, and encapsulates all code execution\n              concerns.  (https://ghuntley.com/ralph/) Ralph Wiggum as a \"software engineer\"   (Jul 2025) \u2014 A programmatic loop over agents (agent\n              orchestration). Huntley describes it as \"deterministically bad in\n              a nondeterministic world\". Code-Only inverts this a bit:\n              projection of a nondeterministic model into deterministic\n              execution. Agent orchestration on top of an agent's Code-Only\n              inner-loop could be a powerful combination.  (https://lucumr.pocoo.org/2025/7/3/tools/) Tools: Code is All You Need   (Jul 2025) \u2014 Raises code as a first-order concern for agents.\n              Ronacher's observation: asking an LLM to write a script to\n              transform markdown makes it possible to reason about and trust the\n              process. The script is reviewable, repeatable, composable.\n              Code-Only takes this further where every action becomes a script\n              you can reason about.  (https://ampcode.com/how-to-build-an-agent) How to Build an Agent   (Apr 2025) \u2014 The cleanest way to achieve a Code-Only agent today\n              may be to build it from scratch. Tweaking current agents like\n              Claude Code to enforce a single tool means friction. Thorsten's\n              article is a lucid account for building an agent loop with tool\n              calls. If you want to enforce Code-Only, this makes it easy to do\n              it yourself.   What's Next Two directions feel inevitable. First, agent orchestration. Tools\n            like (https://prose.md) prose.md  let you compose\n            agents in natural language with program-like constructs. What\n            happens when those agents are Code-Only in their inner loop? You get\n            natural language for coordination, rigid semantics for execution.\n            The best of both.  Second, hybrid tooling. Skills work well for processes that live in\n            natural language. Code-Only works well for processes that need\n            guarantees. We'll see agents that fluidly mix both: Skills for\n            orchestration and intent, Code-Only for computation and precision.\n            The line between \"prompting an agent\" and \"programming an agent\"\n            will blur until it disappears.  Try \u276f\u276f (https://github.com/rvantonder/execute_code_py) Code-Only plugin for Claude Code   1 There is something beautifully (https://en.wikipedia.org/wiki/Quine_(computing)) quine-like about this agent. I've always (https://github.com/rvantonder/pentaquine) loved quines .  Timestamped  9 Jan 2026       "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:22:03.841342+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://rijnard.com/blog/the-code-only-agent"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2026-01-23T07:22:11.058946+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:22:05.581474+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://rijnard.com/blog/the-code-only-agent"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2026-01-23T07:22:03.811555+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:22:00.062159+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpvfel11p7",
                    "https://rijnard.com/blog/the-code-only-agent"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2026-01-23T07:21:51.178211+00:00",
                "index_texts": [
                    "The Code-Only Agent\n          \n            When Code Execution Really is All You Need\n          \n\n          \n            \n          \n\n          \n            If you're building an agent, you're probably overwhelmed. Tools.\n            MCP. Subagents. Skills. The ecosystem pushes you toward complexity,\n            toward \"the right way\" to do things. You should know: Concepts like\n            \"Skills\" and \"MCP\" are actually outcomes of an\n            ongoing learning process of humans figuring stuff out. The\n            space is wide open for exploration. With this mindset I\n            wanted to try something different. Simplify the assumptions.\n          \n\n          \n            What if the agent only had\n            one tool? Not just any tool, but the most powerful one. The\n            Turing-complete one: execute code.\n          \n\n          \n            Truly one tool means: no `bash`, no `ls`, no `grep`. Only\n            execute_code. And you enforce it.\n          \n\n          \n            When you watch an agent run, you might think: \"I wonder what tools\n            it'll use to figure this out. Oh look, it ran `ls`. That makes\n            sense. Next, `grep`. Cool.\"\n          \n          \n            The simpler Code-Only paradigm makes that question irrelevant. The\n            question shifts from \"what tools?\" to \"what code will it produce?\"\n            And that's when things get interesting.\n          \n\n\t\t\t\t\texecute_code: One Tool to Rule Them All\n\n          Traditional prompting works like this:\n\n          \n            > Agent, do thing\n            \n            > Agent\n            responds\n            with thing\n          \n\n          Contrast with:\n\n          \n            > Agent, do thing\n            \n            > Agent\n            creates and runs code\n            to do thing\n          \n\n          \n            It does this every time. No, really,\n            every\n            time. Pick a runtime for our Code-Only agent, say Python. It needs\n            to find a file? It writes Python code to find the file and executes\n            the code. Maybe it runs rglob. Maybe it does os.walk.\n          \n\n          \n            It needs to create a script that crawls a website? It doesn't write\n            the script to your filesystem (reminder: there's no\n            create_file tool to do that!). It\n            writes code to output a script that crawls a website.1\n          \n\n          \n            We make it so that there is literally no way for the agent to\n            do anything productive without\n            writing code.\n          \n\n          \n            So what? Why do this? You're probably thinking, how is this useful?\n            Just give it `bash` tool already man.\n          \n\n          \n            Let's think a bit more deeply what's happening. Traditional agents\n            respond with something. Tell it to find some DNA pattern across 100\n            files. It might `ls` and `grep`, it might do that in some\n            nondeterministic order, it'll figure out an answer and\n            maybe you continue interacting because it missed a directory or you\n            added more files. After some time, you end up with a conversation of\n            tool calls, responses, and an answer.\n          \n\n          \n            At some point the agent might even write a Python script to do this\n            DNA pattern finding. That would be a lucky happy path, because we\n            could rerun that script or update it later... Wait, that's handy...\n            actually, more than handy... isn't that\n            ideal? Wouldn't it be better if we told it to write a script at the\n            start? You see, the Code-Only agent doesn't need to be told to write\n            a script. It\n            has\n            to, because that's literally the only way for it to do anything of\n            substance.\n          \n\n          \n            The Code-Only agent produces something more precise than an answer\n            in natural language. It produces a code witness of an\n            answer. The answer is the output from running the code. The agent\n            can interpret that output in natural language (or by writing code),\n            but the \"work\" is codified in a very literal sense. The Code-Only\n            agent doesn't respond with something. It produces a code witness\n            that outputs something.\n          \n\n          \n           Try \u276f\u276f Code-Only plugin for Claude Code\n            \n          \n\n          Code witnesses are semantic guarantees\n\n          \n            Let's follow the consequences. The code witness must abide by\n            certain rules: The rules imposed by the language runtime semantics\n            (e.g., of Python). That's not a \"next token\" process. That's not a\n            \"LLM figures out sequence of tool calls, no that's not what I\n            wanted\". It's piece of code. A piece of code! Our one-tool agent has\n            a wonderful property: It went through latent space to produce\n            something that has a defined semantics, repeatably runnable, and\n            imminently comprehensible (for humans or agents alike to reason\n            about). This is nondeterministic LLM token-generation projected into\n            the space of Turing-complete code, an executable description of\n            behavior as we best understand it.\n          \n\n          \n            Is a Code-Only agent really enough, or too extreme? I'll be frank: I\n            pursued this extreme after two things (1) inspiration from articles in Further Reading below (2) being annoyed at agents for not comprehensively and\n            exhaustively analyzing 1000s of files on my laptop. They would skip,\n            take shortcuts, hallucinate. I knew how to solve part of that\n            problem: create a\n            programmatic\n            loop and try have fresh instances/prompts to do the work\n            comprehensively. I can rely on the semantics of a loop written in\n            Python. Take this idea further, and you realize that for anything\n            long-running and computable (e.g., bash or some tool), you actually\n            want the real McCoy: the full witness of code, a trace of why things\n            work or don't work. The Code-Only agent\n            enforces\n            that principle.\n          \n\n          \n            Code-Only agents are not too extreme. I think they're the only way\n            forward for computable things. If you're writing travel blog posts,\n            you accept the LLMs answer (and you don't need to run tools for\n            that). When something is computable though, Code-Only is the only\n            path to a\n            fully trustworthy\n            way to make progress where you need guarantees (subject to\n            the semantics that your language of choice guarantees, of course). When I say\n            guarantees, I mean that in the looser sense, and also in a\n            Formal\n            sense. Which beckons: What happens when we use a language like\n            Lean with some of the\n            strongest guarantees? Did we not observe that\n            programs are proofs?\n          \n\n          \n            This lens says the Code-Only agent is a producer of proofs,\n            witnesses of computational behavior in the world of\n            proofs-as-programs. An LLM in a loop forced to produce proofs, run\n            proofs, interpret proof results. That's all.\n          \n\n          Going Code-Only\n\n          \n            So you want to go Code-Only. What happens? The paradigm is simple,\n            but the design choices are surprising.\n          \n\n          \n            First, the harness. The LLM's output is code, and you execute that\n            code. What should be communicated back? Exit code makes sense. What\n            about output? What if the output is very large? Since you're running\n            code, you can specify the result type that running the code should\n            return.\n          \n\n          \n            I've personally, e.g., had the tool return results directly if under\n            a certain threshold (1K bytes). This would go into the session context.\n\t\t\t\t\t\tAlternatively, write the results to a JSON\n            file on disk if it exceeds the threshold. This avoids context blowup and the result tells the\n            agent about the output file path written to disk. How best to pass\n            results, persist them, and optimize for size and context fill are\n            open questions. You also want to define a way to deal with `stdout`\n            and `stderr`: Do you expose these to the agent? Do you summarize\n            before exposing?\n          \n\n          \n            Next, enforcement. Let's say you're using Claude Code. It's not\n            enough to persuade it to always create and run code. It turns out\n            it's surprisingly twisty to force Claude Code into a single tool\n            (maybe support for this will improve). The best plugin-based\n            solution I found is a tool PreHook that catches banned tool uses.\n            This wastes some iterations when Claude Code tries to use a tool\n            that's not allowed, but it learns to stop attempting filesystem\n            reads/writes. An initial prompt helps direct.\n          \n\n          \n            Next, the language runtime. Python, TypeScript, Rust, Bash. Any\n            language capable of being executed is fair game, but you'll need to\n            think through whether it works for your domain. Dynamic languages\n            like Python are interesting because you can run code natively in the\n            agent's own runtime, rather than through subprocess calls. Likewise\n            TypeScript/JS can be injected into TypeScript-based agents (see\n            Further Reading).\n          \n\n          \n            Once you get into the Code-Only mindset, you'll see the potential\n            for composition and reuse. Claude Skills define reusable processes\n            in natural language. What's the equivalent for a Code-Only agent?\n            I'm not sure a Skills equivalent exists yet, but I anticipate it\n            will take shape soon: code as building blocks for specific domains\n            where Code-Only agents compose programmatic patterns. How is that\n            different from calling APIs? APIs form part of the reusable blocks,\n            but their composition (loops, parallelism, asynchrony) is what a\n            Code-Only agent generates.\n          \n\n          \n            What about heterogeneous languages and runtimes for our `execute_tool`? I don't think we've thought that far yet.\n          \n\n          Further Reading\n\n          \n            The agent landscape is quickly evolving. My thoughts on how the\n            Code-Only paradigm fits into inspiring articles and trends, from\n            most recent and going back:\n          \n\n          \n            \n              prose.md\n              (Jan 2026) \u2014 Code-Only reduces prompts to executable code (with\n              loops and statement sequences). Prose expands prompts into natural\n              language with program-like constructs (also loops, sequences,\n              parallelism). The interplay of natural language for agent\n              orchestration and rigid semantics for agent execution could be\n              extremely powerful.\n            \n            \n              Welcome to Gas Town\n              (Jan 2026) \u2014 Agent orchestration gone berserk. Tool running is the low-level\n              operation at the bottom of the agent stack. Code-Only fits as the\n              primitive: no matter how many agents you orchestrate, each one\n              reduces to generating and executing code.\n            \n            \n              Anthropic Code Execution with MCP article\n              (Nov 2025) \u2014 MCP-centric view of exposing MCP servers as code API\n              and not tool calls. Code-Only is simpler and more general. It\n              doesn't care about MCP, and casting the MCP interface as an API is\n              a mechanical necessity that acknowledges the power of going\n              Code-Only.\n            \n            \n              Anthropic Agent Skills article\n              (Oct 2025) \u2014 Skills embody reusable processes framed in natural\n              language. They can generate and run code, but that's not their\n              only purpose. Code-Only is narrower (but computationally\n              all-powerful): the reusable unit is always executable. The analog\n              to Skills manifests as pluggable executable pieces: functions,\n              loops, composable routines over APIs.\n            \n            \n              Cloudflare Code Mode article\n              (Sep 2025) \u2014 Possibly the earliest concrete single-code-tool\n              implementation. Code Mode converts MCP tools into a TypeScript API\n              and gives the agent one tool: execute TypeScript. Their insight is\n              pragmatic: LLMs write better code than tool calls because of\n              training data. In its most general sense, going Code-Only doesn't\n              need to rely on MCP or APIs, and encapsulates all code execution\n              concerns.\n            \n            \n              Ralph Wiggum as a \"software engineer\"\n              (Jul 2025) \u2014 A programmatic loop over agents (agent\n              orchestration). Huntley describes it as \"deterministically bad in\n              a nondeterministic world\". Code-Only inverts this a bit:\n              projection of a nondeterministic model into deterministic\n              execution. Agent orchestration on top of an agent's Code-Only\n              inner-loop could be a powerful combination.\n            \n            \n              Tools: Code is All You Need\n              (Jul 2025) \u2014 Raises code as a first-order concern for agents.\n              Ronacher's observation: asking an LLM to write a script to\n              transform markdown makes it possible to reason about and trust the\n              process. The script is reviewable, repeatable, composable.\n              Code-Only takes this further where every action becomes a script\n              you can reason about.\n            \n            \n              How to Build an Agent\n              (Apr 2025) \u2014 The cleanest way to achieve a Code-Only agent today\n              may be to build it from scratch. Tweaking current agents like\n              Claude Code to enforce a single tool means friction. Thorsten's\n              article is a lucid account for building an agent loop with tool\n              calls. If you want to enforce Code-Only, this makes it easy to do\n              it yourself.\n            \n          \n\n          What's Next\n\n          \n            Two directions feel inevitable. First, agent orchestration. Tools\n            like prose.md let you compose\n            agents in natural language with program-like constructs. What\n            happens when those agents are Code-Only in their inner loop? You get\n            natural language for coordination, rigid semantics for execution.\n            The best of both.\n          \n\n          \n            Second, hybrid tooling. Skills work well for processes that live in\n            natural language. Code-Only works well for processes that need\n            guarantees. We'll see agents that fluidly mix both: Skills for\n            orchestration and intent, Code-Only for computation and precision.\n            The line between \"prompting an agent\" and \"programming an agent\"\n            will blur until it disappears.\n          \n\n          \n            Try \u276f\u276f Code-Only plugin for Claude Code\n            \n          \n\n          \n          \n            1There is something beautifully\n            quine-like\n            about this agent. I've always\n            loved quines.\n          \n\t\t\t\t\t\n\t\t\t\t\t\t\t\t\tTimestamped  9 Jan 2026"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:21:49.199220+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://rijnard.com/blog/the-code-only-agent"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-01-23T07:21:48.249256+00:00",
                "index_texts": null,
                "output": "The Code-Only Agent \u2022 Rijnard van Tonder",
                "pwd": "/data/archive/1769152883.742596",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-23T07:21:48.232325+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "https://web.archive.org/web/20260123072215/https://rijnard.com/blog/the-code-only-agent",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "The Code-Only Agent \u2022 Rijnard van Tonder",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1769152883.742596",
    "newest_archive_date": "2026-01-23T07:22:11.090023+00:00",
    "num_failures": 0,
    "num_outputs": 9,
    "oldest_archive_date": "2026-01-23T07:21:29.916597+00:00",
    "path": "/blog/the-code-only-agent",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KFMVNB2EC87C4A4701CTGZ9B",
    "snapshot_id": "a46655a1-f191-4b44-9303-03ad59a87d2b",
    "sources": [
        "/data/sources/1769152883-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1769152883.742596",
    "title": "The Code-Only Agent \u2022 Rijnard van Tonder",
    "url": "https://rijnard.com/blog/the-code-only-agent"
}