{
    "archive_path": "archive/1767503101.628046",
    "base_url": "blog.bryanl.dev/posts/agent-framework-vision",
    "basename": "",
    "bookmarked_date": "2026-01-04 05:05",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/blog.bryanl.dev/posts/agent-framework-vision",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=blog.bryanl.dev",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "blog.bryanl.dev",
    "downloaded_at": "2026-01-04T05:05:08.760380+00:00",
    "downloaded_datestr": "2026-01-04 05:05",
    "extension": "",
    "hash": "9B6M0G6C64JPPJDA5A5M",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://blog.bryanl.dev/posts/agent-framework-vision/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-01-04T05:06:43.008124+00:00",
                "index_texts": null,
                "output": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://blog.bryanl.dev/posts/agent-framework-vision/']' timed out after 60 seconds",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:42.967129+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://blog.bryanl.dev/posts/agent-framework-vision/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2026-01-04T05:05:20.423371+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:16.321228+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=blog.bryanl.dev"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-01-04T05:05:12.122495+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:09.124713+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://blog.bryanl.dev/posts/agent-framework-vision/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-01-04T05:05:12.198383+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:12.150355+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2026-01-04T05:05:37.971000+00:00",
                "index_texts": [
                    "(https://blog.bryanl.dev/posts/agent-framework-vision/) (/sitemap-index.xml) (RSS Feed) (/rss.xml) Agents Done Right: A Framework Vision for 2026 (/_astro/index.BAieSGK2.css)  (/) \u2190 blog.bryanl.dev        Agents Done Right: A Framework Vision for 2026 December 28, 2025  (/tags/agents/) agents  (/tags/ai/) ai  (/tags/frameworks/) frameworks  (/tags/typescript/) typescript    The Problem with Today\u2019s Agent Frameworks I\u2019ve been building with LLM-based agents for a while now, and I\u2019ve been using a lot of other people\u2019s tools too. The space is growing fast, but something feels off. It\u2019s getting complex for the sake of being complex. New abstractions pile on top of old ones. Configuration options multiply. (https://sensible.com/) Steve Krug\u2019s first law of usability is \u201cdon\u2019t make me think.\u201d Current agent frameworks violate this constantly. And yet I keep running into the same walls. The agent starts strong, but as the task gets complex, its context window fills up. The context window is the LLM\u2019s working memory, a limit on how much text it can hold at once, measured in tokens (roughly, chunks of words). Fill it up, and the model loses track. It forgets what it was doing. It repeats itself. It starts hallucinating, making things up with confidence. Eventually it fails. Not because the model isn\u2019t capable, but because the architecture around it can\u2019t manage the complexity. We\u2019re in the \u201cRuby on Rails 1.0\u201d era of agent development. Everyone is building agents, but we\u2019re all solving the same problems from scratch: context exhaustion, doom loops, model selection paralysis, and the cognitive overload of reviewing agent output. I once watched an agent run npm run build over and over for five minutes while I was on a phone call. It just kept going, burning tokens, stuck in a loop it couldn\u2019t escape. That\u2019s a doom loop. The agent repeats the same failed action, hoping for a different result. The teams that have shipped production agents have independently discovered similar architectural patterns. But these insights are locked inside proprietary systems. The open source ecosystem is still dominated by thin wrappers around LLM APIs that avoid the hard problems. It\u2019s 2026. We should have a framework that makes the right architecture the easy architecture. This post is my attempt to think through what that framework should look like. Core Principles 1. Convention Over Configuration Ruby on Rails succeeded because it made decisions for you. Database table names, file locations, URL structures were all derived from conventions. You could override them, but you didn\u2019t have to think about them. Agent frameworks today are the opposite. Every project starts with: Which model? Which embedding provider? How should I structure tools? What\u2019s my context strategy? How do subagents communicate? The framework should have strong defaults for all of this. Here are the conventions I\u2019d choose: Model selection by task complexity. Simple edits (adding a log statement, fixing a typo) use a fast model. Multi-file refactors or debugging sessions use a reasoning model. The framework infers complexity from the task description and codebase scope. You don\u2019t specify a model. Context budgets with inheritance. Each agent gets a context budget. Subagents inherit a portion of their parent\u2019s remaining budget. When an agent approaches its limit, the framework triggers automatic summarization. This forces the architecture toward delegation: if a task doesn\u2019t fit in your budget, spawn a subagent. Mandatory checkpoints for risky operations. File deletion, multi-file modifications, and detected uncertainty (phrases like \u201cI\u2019m not sure\u201d or \u201cthis might break\u201d) require human approval. Everything else proceeds automatically. You opt out of safety, not into it. Curated tool profiles by archetype. Searchers get search tools. Writers get write tools. Researchers get web access. Archetypes don\u2019t get tools outside their profile unless explicitly granted. This prevents the \u201ctoo many tools\u201d problem where agents waste context evaluating irrelevant options. These conventions encode lessons from watching agents fail. Context exhaustion, review fatigue, tool confusion: the failure modes are predictable. The conventions exist to make the right architecture the easy architecture. // This should just work - all conventions applied  const agent = new Agent ({ codebase:  \"./my-project\" });  await agent. run ( \"Add input validation to the signup form\" );    Copy  Override when you need to. But start productive. Configuration complexity Minimal config  (0) Simple Verbose   Without Conventions(5lines)  With Conventions(5lines)     Drag the slider to add config sections, then compare the tabs   2. Tasks, Not Models The current generation of tools asks: \u201cWhich model do you want to use?\u201d This is the wrong question. For most tasks, users shouldn\u2019t be thinking about models at all. When you ask an agent to \u201cadd input validation to the signup form,\u201d you don\u2019t care whether it uses a fast model or a reasoning model. You care that it works. Model selection is an optimization detail. You describe what you want; the framework figures out how to deliver it efficiently. A quick code edit optimizes for speed. Complex debugging allocates thinking time. Large refactors balance cost and capability. You shouldn\u2019t have to care which model runs under the hood. Framework implication : No model parameter in the default API. The framework infers requirements from the task. Power users can override when needed, but the default path requires zero model knowledge. Traditional approach Try task-first \u2192   Select a model: Choose model...     7 options, each with tradeoffs. Which one is right for your task?   Model selection puts cognitive load on the user   3. Subagents as the Primary Scaling Mechanism Watch an agent tackle a complex task. It reads files, searches the codebase, backtracks, tries a different approach. The context window fills up. By the time it\u2019s ready to write code, it\u2019s forgotten the original requirements. It loops. It hallucinates. It fails. I hit this wall constantly when researching new topics. One PDF can consume your entire context window. A large codebase is even worse. You end up working in awkward, inconvenient ways just to get access to the knowledge you need. An obvious fix is to \u201cmake context windows bigger.\u201d But bigger windows just delay the problem. They\u2019re slower and more expensive. A better approach is factoring : break the work into isolated subtasks, each with its own context. A subagent searches the codebase and returns only the relevant file paths. Another analyzes a specific function and returns only its findings. The parent agent stays focused, receiving distilled results instead of accumulating everything. Subagents are to agents what functions are to programs. They isolate context for subtasks and return only relevant results to the parent. They can be optimized for different task types, whether speed or depth, and they enable parallelism without context collision. The difference is dramatic. A single-agent approach might use 90% of its context window stumbling through a task. The same task with subagents? The parent uses 25%, staying sharp throughout. Spawning subagents should feel as natural as calling a function. Not an advanced feature you learn later, but the default pattern from day one. Framework implication : // This should feel as natural as a function call  const result = await agent. delegate ( \"searcher\" , {  task:  \"locate authentication logic\" ,  returnFormat:  \"file_paths_with_snippets\"  });    Copy  The framework also needs primitives that make context management automatic: const agent = new Agent ({  contextBudget:  128_000 ,  // Explicit budget, inherited by subagents  overflowStrategy:  \"summarize\" ,  // What to do when approaching limit  loopDetection:  true // Circuit-break on repetitive patterns  });    Copy  With these primitives, agents can focus on the task while the framework handles the resource management. Just like garbage collection lets programmers focus on logic instead of memory. Single Agent With Subagents  Run Task   Agent Context Window 0%     Click \"Run Task\" to see the difference  Single agent accumulates all context, filling the window   4. Opinionated Subagent Archetypes Watch enough production agents and you\u2019ll see the same patterns emerge. Teams independently discover that certain subtask shapes keep recurring, and that specialized subagents for those shapes dramatically outperform general-purpose ones. Here are the archetypes that have proven themselves: Archetype What it does Why it\u2019s specialized   Searcher  Finds relevant code, files, symbols Optimized for speed; returns locations, not content  Thinker  Reasons through complex problems Allocates thinking time; returns analysis, not action  Researcher  Gathers external knowledge (docs, APIs) Web/retrieval access; returns synthesized context  Writer  Makes targeted code changes Diff-focused; returns patches, not explanations  Planner  Breaks down complex tasks Strategic focus; returns task breakdown  Checker  Validates and critiques work Independent perspective; adversarial stance    Notice the pattern: each archetype has a constrained output format . A Searcher returns locations. A Thinker returns analysis. A Writer returns patches. This constraint is the key. It forces the subagent to distill its work rather than dump everything into the parent\u2019s context. These archetypes should ship as built-in, customizable components. You shouldn\u2019t have to reinvent the Searcher pattern from scratch. It should just be there, ready to use, with sensible defaults you can override when needed. Subagent Archetypes \u2014 tap to explore  \ud83d\udd0d Searcher  Finds relevant code fast  \ud83e\udde0 Thinker  Reasons through complexity  \ud83d\udcda Researcher  Gathers external knowledge  \u270d\ufe0f Writer  Makes targeted code changes  \ud83d\udccb Planner  Breaks down complex tasks  \u2713 Checker  Validates and critiques    Each archetype is optimized for a specific type of subtask   5. Human-Agent Workflow is Part of the Framework The bottleneck has shifted. It\u2019s no longer \u201ccan the agent write code?\u201d It\u2019s \u201ccan the human review and trust the code fast enough?\u201d Today\u2019s workflow: agent dumps a wall of changes, human squints at diffs, tries to understand the intent, hopes nothing broke. This doesn\u2019t scale. As agents get more capable, the review burden grows faster than human attention. When you run an agent, you should be able to specify where you want control. Checkpoints give you natural stopping points for review. Change proposals explain what will change and why before changes are made. Validation hooks let you plug in secondary verification (another model, static analysis, or tests). Diff streaming gives real-time visibility into changes as they happen. And approval gates let you require sign-off for high-risk operations. Here\u2019s what that looks like: const result = await agent. run ({  task:  \"Add rate limiting to /api/users\" ,  checkpoints: [ \"before_write\" ,  \"after_plan\" ],  requireApproval: [ \"delete_file\" ,  \"modify_config\" ],  onProposal : ( plan )  => showPlanToUser (plan)  });    Copy  This isn\u2019t UI. It\u2019s protocol. The framework defines the contract between agent and human; UIs implement it. A CLI might show a simple approve/reject prompt. An IDE might render an interactive diff viewer. Same protocol, different presentations. CLI IDE  Run Agent        $ agent run \"Add rate limiting\"   Click \"Run Agent\" to see the checkpoint flow  Same protocol, different presentations \u2014 CLI shows inline prompts   6. Tools are Curated, Not Collected The promise of universal tool interoperability sounds great in theory. In practice: more tools = more confusion. Here\u2019s why. Every tool you give an agent is a decision it has to make: \u201cShould I use this?\u201d With 5 well-chosen tools, that decision is easy. With 50 tools, many overlapping, some poorly documented, others rarely useful, the agent wastes context evaluating options, makes false starts with wrong tools, and loses focus on the actual task. I ran into this recently with (https://modelcontextprotocol.io/) MCP (Model Context Protocol) servers. I built one for Outlook to read my mail and calendar. But with multiple MCP servers installed, the agent kept trying to look up information on the web instead of using the mail tool sitting right there. Too many options, not enough guidance about which one to pick. What you want is a curated core toolset optimized for coding tasks, not a grab-bag of integrations. Different subagents should get different tool subsets: a Searcher doesn\u2019t need web access. Destructive tools should require human approval. And tool usage patterns should be tracked to surface which tools help vs. which cause thrashing. Framework implication : const searcher = new Searcher ({  tools: [ \"grep\" ,  \"ast_search\" ,  \"file_read\" ],  // Curated subset  restricted: [ \"web_fetch\" ,  \"file_write\" ]  // Explicitly excluded  });   // Tools are first-class objects with metadata  const grepTool = {  name:  \"grep\" ,  description:  \"Search file contents with regex\" ,  whenToUse:  \"Looking for specific strings or patterns in code\" ,  costEstimate:  \"low\" ,  requiresApproval:  false  };    Copy  The goal: agents that pick the right tool immediately, not agents that waste time evaluating irrelevant options. Tools available: 5 10 20 50   Run Task   Available Tools (5) grep file_read file_write ast_search run_test    Agent Execution Click \"Run Task\" to see agent behavior     Curated toolset: agent stays focused, minimal wasted effort   7. When to Use a Subagent vs. a Tool There\u2019s a simple heuristic for deciding whether something should be a tool or a subagent: Tool : Stateless transformation. Input goes in, output comes out, no iteration required. Format a date. Parse JSON. Run a regex. The operation doesn\u2019t accumulate context or require judgment. Subagent : Iteration, judgment, or context accumulation. The operation explores, backtracks, tries alternatives, or builds up intermediate state. Search, research, analysis, planning: these need their own context window. The architectural reason to care: if you implement an iterative operation as a tool, its entire execution trace pollutes the parent context. Every search result, every intermediate step, every dead end. All of it crowds out space for the actual task. // Bad: \"tool\" that leaks context  const searchTool = {  name:  \"search_codebase\" ,  execute :  async ( query )  => {  // All of this ends up in parent context  const files = await grep (query);  const ranked = await rerankResults (files);  const snippets = await extractSnippets (ranked);  return snippets;  }  };   // Good: subagent with isolated context  const searcher = new Searcher ({  // Runs in own context window  // Only `snippets` returned to parent  returns:  \"snippets_with_locations\"  });    Copy  This matters because many things we call tools are actually subagents in disguise. Web search, documentation lookup, code analysis: these involve iteration and judgment. The name \u201ctool\u201d makes them sound simple, but they\u2019re not. Implementing them as proper subagents with isolated context is what makes the archetype pattern (Searcher, Thinker, Researcher) work. Tool vs Subagent: Context Impact Run Search   Tool (leaks context) +0tokens  Parent Context    Waiting to run...   Subagent (isolated) +0tokens  Parent Context    Waiting to run...     Same search task \u2014 subagent isolates working context from parent   8. Subagent Context is Ephemeral by Default A subagent\u2019s working context is like local variables in a function. It exists during execution and is discarded when done. Consider a Researcher subagent investigating a library. It fetches 10 documentation pages, reads API references and examples, follows links to GitHub issues, and synthesizes findings into a summary. All of that is working context. But it returns only the summary to the parent. The parent doesn\u2019t need the 10 pages, the API refs, or the GitHub issues. It needs the answer. All that intermediate work is temporary scaffolding. const researcher = new Researcher ({  // Everything inside runs in ephemeral context  // Only the return value persists  returns:  \"synthesis\" ,   // Optional: persist specific artifacts  persist: [ \"key_code_snippets\" ,  \"api_signatures\" ]  });   // Parent receives ~500 tokens, not 50,000  const findings = await agent. delegate (researcher, {  task:  \"How does auth work in this library?\"  });    Copy  The principle : Subagents accumulate context to do their job, then compress before returning. The parent receives a distilled result, not the full execution trace. I built something like this to explore a large codebase. A custom agent walked through the source files, read them, and wrote summarized markdown documentation. The agent consumed thousands of lines of code, but what I got back was a concise capture of the service\u2019s flow. The essence, not the exhaustive detail. That\u2019s the pattern: do the heavy lifting in isolated context, return only what matters. This is how human experts work too. When you ask a colleague to research something, you want their conclusion, not everything they read along the way. Ephemeral Context: Research Journey Start Research   Researcher Working Context 0tokens     Click \"Start Research\" to begin   Parent Context +0tokens     Waiting for results...     Subagent accumulates context, then compresses before returning   The Developer Experience What Building an Agent Should Feel Like All of the principles above collapse into a simple question: what does it feel like to build with this framework? If the conventions are right, the code should be obvious. You declare what you want, not how to manage it. The framework handles orchestration, context, and human checkpoints. You focus on the task. Here\u2019s what that looks like: import { Agent, Searcher, Thinker, Writer }  from \"agentkit\" ;   // Declare a coding agent with specialized subagents  const agent = new Agent ({  name:  \"code-assistant\" ,  mode:  \"smart\" ,  // vs \"fast\" for quick iteration  subagents: [  new Searcher ({ tools: [ \"grep\" ,  \"ast_search\" ,  \"embeddings\" ] }),  new Thinker ({ timeout:  120 }),  new Writer ({ requireApproval:  false }),  ],  contextBudget:  128_000 ,  humanCheckpoints: [ \"before_multi_file_edit\" ,  \"on_uncertainty\" ],  });   // Run a task - framework handles subagent orchestration  const result = await agent. run ({  task:  \"Add rate limiting to the /api/users endpoint\" ,  codebase:  \"/path/to/repo\" ,  });   // Result includes structured output for UI integration  result.changes;  // List of file changes  result.proposal;  // What changed and why  result.contextUsed;  // Debugging/optimization info    Copy  Why TypeScript? TypeScript offers patterns that make agent code safer in ways that Python\u2019s type system can\u2019t match. Branded types for resource budgets. Context tokens aren\u2019t just numbers. They\u2019re a distinct unit. Branded types prevent you from accidentally passing a line count where a token count is expected: type ContextTokens = number & {  readonly brand : unique symbol };  const budget : ContextTokens = 64_000 as ContextTokens ;  // Can't accidentally pass a plain number where ContextTokens is required    Copy  Discriminated unions for checkpoint states. When a checkpoint can be pending, approved, or rejected, discriminated unions force exhaustive handling. The compiler catches missing cases at build time, not runtime: type CheckpointState =  | {  status : \"pending\" }  | {  status : \"approved\" ;  by : string ;  at : Date }  | {  status : \"rejected\" ;  reason : string };   function handleCheckpoint ( state : CheckpointState ) {  switch (state.status) {  case \"pending\" :  return showWaiting ();  case \"approved\" :  return proceed (state.by);  case \"rejected\" :  return showError (state.reason);  // TypeScript errors if you miss a case  }  }    Copy  Generic constraints for subagent return types. A Searcher<FileLocation[]> and a Thinker<Analysis> are different types. The parent agent knows exactly what shape to expect from each delegation: const locations = await agent. delegate < FileLocation []>(searcher, { task:  \"find auth\" });  // locations is typed as FileLocation[], not unknown    Copy  Beyond type safety, the ecosystem fits: VS Code extensions, language servers, and most developer tooling already run on TypeScript. Using (https://bun.sh) Bun , agents compile to standalone executables with no runtime dependencies. What This Enables Think about what web development looked like before Rails. Every project started with the same decisions: How do I structure my code? How do I talk to the database? How do I handle routing? Teams spent months on plumbing before writing a single line of business logic. Rails changed that by encoding the answers into conventions. Suddenly, developers could go from idea to working application in hours instead of weeks. Agent development is in that pre-Rails moment right now. Every team building production agents is solving the same problems: context management, subagent orchestration, human review workflows, tool selection. The solutions exist, but they\u2019re locked inside proprietary systems or tribal knowledge. With the right framework primitives, agent builders can focus on what makes their agent unique instead of reinventing context management for the hundredth time. They can experiment with novel interaction patterns instead of debugging the same doom loops. They can invest in evaluation and domain expertise instead of infrastructure. And users get agents that actually work. Not agents that succeed on simple tasks and fall apart on complex ones. Not agents that require babysitting to avoid going off the rails. Agents with predictable behavior, transparent operation, and graceful degradation when things get hard. Open Questions There\u2019s still work to figure out. These are the problems I\u2019m thinking through, along with my current hunches. How do subagents share learned context? There\u2019s probably a \u201cmemory\u201d layer that persists across tasks. Something like a project-level knowledge base that subagents can read from and write to. The Researcher finds something important, it goes into shared memory. The Writer pulls from it later. This doesn\u2019t have to be LLM-backed memory. It could be a graph database, a structured knowledge store, or something we haven\u2019t invented yet. But the interface has to be simple enough that it doesn\u2019t become another configuration burden. How do peer agents collaborate? Most agent architectures are hierarchical. Parent spawns child, child returns result. But some tasks need peers working in parallel, sharing findings as they go. I suspect the answer is message-passing with typed channels, similar to how concurrent systems handle coordination. The framework handles the plumbing; agents just send and receive. How do you evaluate agents systematically? This might be the hardest problem, and I don\u2019t think anyone has solved it yet. The core challenge is ground truth. To evaluate whether an agent did the right thing, you need to know what \u201cright\u201d means for that task. For coding tasks, you can check some things automatically (does the code compile? do tests pass?) but these only catch obvious failures. An agent can produce working code that\u2019s architecturally wrong, or secure code that\u2019s unmaintainable. I think evaluation has to happen at multiple layers. Task success is the baseline: did the change work at all? Behavioral consistency asks whether the same task on the same codebase produces similar results across runs. If an agent gives wildly different answers each time, something\u2019s wrong. Human preference captures whether humans actually accept the changes in practice. The infrastructure for this is expensive to build. You need captured scenarios (codebase + task + expected behavior), replay with controlled randomness, and scoring that\u2019s more nuanced than pass/fail. It\u2019s similar to how you\u2019d evaluate a human candidate through realistic work samples rather than abstract tests. The framework should probably include hooks for scenario capture and replay, even if the evaluation logic is user-defined. There\u2019s also a meta-problem: evaluating the evaluations. A scenario that tests whether the agent can add two numbers tells you nothing useful. The scenarios themselves need to be validated for difficulty and discriminative power. Do they distinguish good agents from bad ones? Do they catch regressions? This is the same challenge test engineers face with code coverage: 100% coverage means nothing if the tests are trivial. My intuition is that evaluation requires three primitives: Scenario (a frozen codebase + task), Assertion (did the output compile, pass tests, match intent), and Consistency (same scenario, N runs, what\u2019s the variance). This is closer to integration testing than unit testing. How do you budget spend across subagent hierarchies? Context tokens are one cost, but API calls add up fast when you\u2019re spawning subagents. The framework probably needs a cost model that propagates budgets down the hierarchy. Parent allocates a budget, subagents inherit portions of it, and the framework enforces limits before you get a surprise bill. What happens when a subagent fails? The boring answer is probably the right one: retry with exponential backoff, then escalate to the parent with an error. The parent decides whether to substitute a different approach or surface the failure to the user. Automatic substitution sounds clever but might hide problems that humans should see. What\u2019s Next The patterns are emerging. Production agent systems have proven what works. Now we need to turn these lessons into infrastructure that benefits everyone. I started this post frustrated with complexity. Every agent framework I tried added abstraction layers without solving the hard problems. Configuration options multiplied while agents still choked on context. New features shipped while doom loops went unfixed. I kept trying new tools and spending more time learning their paradigms than actually getting work done. That\u2019s a red flag. It means we haven\u2019t found the right way to look at this space yet. Ruby on Rails became popular not because it introduced a lot of new ideas, but because it made easy things trivial and hard things possible in a reasonable amount of time. That\u2019s why everyone learned Ruby. We need the same thing for agents. The answer isn\u2019t more complexity. It\u2019s the right complexity: conventions that encode hard-won lessons, primitives that make the correct architecture easy, and defaults that work out of the box. I don\u2019t think there\u2019s one framework to rule them all. But the patterns matter. Whether you\u2019re building agents, evaluating frameworks, or just trying to understand why your agent keeps running npm run build in a loop, these architectural ideas can help you reason about what\u2019s going wrong and what to try next. If more people internalized these patterns, we\u2019d all waste less time on plumbing and more time on problems that actually matter. And that\u2019s the point.   On this page        The Problem with Today\u2019s Agent Frameworks   Core Principles   1. Convention Over Configuration   2. Tasks, Not Models   3. Subagents as the Primary Scaling Mechanism   4. Opinionated Subagent Archetypes   5. Human-Agent Workflow is Part of the Framework   6. Tools are Curated, Not Collected   7. When to Use a Subagent vs. a Tool   8. Subagent Context is Ephemeral by Default   The Developer Experience   What Building an Agent Should Feel Like   Why TypeScript?   What This Enables   Open Questions   What\u2019s Next        \u00a9 2006\u20132025 blog.bryanl.dev     "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:37.893181+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://blog.bryanl.dev/posts/agent-framework-vision/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2026-01-04T05:05:42.930313+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:39.293361+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://blog.bryanl.dev/posts/agent-framework-vision/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2026-01-04T05:05:37.844483+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:34.468739+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmp1tx3cz4z",
                    "https://blog.bryanl.dev/posts/agent-framework-vision/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2026-01-04T05:05:26.267120+00:00",
                "index_texts": [
                    "The Problem with Today\u2019s Agent Frameworks\nI\u2019ve been building with LLM-based agents for a while now, and I\u2019ve been using a lot of other people\u2019s tools too. The space is growing fast, but something feels off. It\u2019s getting complex for the sake of being complex. New abstractions pile on top of old ones. Configuration options multiply. Steve Krug\u2019s first law of usability is \u201cdon\u2019t make me think.\u201d Current agent frameworks violate this constantly. And yet I keep running into the same walls.\nThe agent starts strong, but as the task gets complex, its context window fills up. The context window is the LLM\u2019s working memory, a limit on how much text it can hold at once, measured in tokens (roughly, chunks of words). Fill it up, and the model loses track. It forgets what it was doing. It repeats itself. It starts hallucinating, making things up with confidence. Eventually it fails. Not because the model isn\u2019t capable, but because the architecture around it can\u2019t manage the complexity.\nWe\u2019re in the \u201cRuby on Rails 1.0\u201d era of agent development. Everyone is building agents, but we\u2019re all solving the same problems from scratch: context exhaustion, doom loops, model selection paralysis, and the cognitive overload of reviewing agent output. I once watched an agent run npm run build over and over for five minutes while I was on a phone call. It just kept going, burning tokens, stuck in a loop it couldn\u2019t escape. That\u2019s a doom loop. The agent repeats the same failed action, hoping for a different result.\nThe teams that have shipped production agents have independently discovered similar architectural patterns. But these insights are locked inside proprietary systems. The open source ecosystem is still dominated by thin wrappers around LLM APIs that avoid the hard problems.\nIt\u2019s 2026. We should have a framework that makes the right architecture the easy architecture. This post is my attempt to think through what that framework should look like.\n\nCore Principles\n1. Convention Over Configuration\nRuby on Rails succeeded because it made decisions for you. Database table names, file locations, URL structures were all derived from conventions. You could override them, but you didn\u2019t have to think about them.\nAgent frameworks today are the opposite. Every project starts with: Which model? Which embedding provider? How should I structure tools? What\u2019s my context strategy? How do subagents communicate?\nThe framework should have strong defaults for all of this. Here are the conventions I\u2019d choose:\nModel selection by task complexity. Simple edits (adding a log statement, fixing a typo) use a fast model. Multi-file refactors or debugging sessions use a reasoning model. The framework infers complexity from the task description and codebase scope. You don\u2019t specify a model.\nContext budgets with inheritance. Each agent gets a context budget. Subagents inherit a portion of their parent\u2019s remaining budget. When an agent approaches its limit, the framework triggers automatic summarization. This forces the architecture toward delegation: if a task doesn\u2019t fit in your budget, spawn a subagent.\nMandatory checkpoints for risky operations. File deletion, multi-file modifications, and detected uncertainty (phrases like \u201cI\u2019m not sure\u201d or \u201cthis might break\u201d) require human approval. Everything else proceeds automatically. You opt out of safety, not into it.\nCurated tool profiles by archetype. Searchers get search tools. Writers get write tools. Researchers get web access. Archetypes don\u2019t get tools outside their profile unless explicitly granted. This prevents the \u201ctoo many tools\u201d problem where agents waste context evaluating irrelevant options.\nThese conventions encode lessons from watching agents fail. Context exhaustion, review fatigue, tool confusion: the failure modes are predictable. The conventions exist to make the right architecture the easy architecture.\n// This should just work - all conventions applied\nconst agent = new Agent({ codebase: \"./my-project\" });\nawait agent.run(\"Add input validation to the signup form\");\nOverride when you need to. But start productive.\n\n\n2. Tasks, Not Models\nThe current generation of tools asks: \u201cWhich model do you want to use?\u201d\nThis is the wrong question. For most tasks, users shouldn\u2019t be thinking about models at all. When you ask an agent to \u201cadd input validation to the signup form,\u201d you don\u2019t care whether it uses a fast model or a reasoning model. You care that it works.\nModel selection is an optimization detail. You describe what you want; the framework figures out how to deliver it efficiently. A quick code edit optimizes for speed. Complex debugging allocates thinking time. Large refactors balance cost and capability. You shouldn\u2019t have to care which model runs under the hood.\nFramework implication: No model parameter in the default API. The framework infers requirements from the task. Power users can override when needed, but the default path requires zero model knowledge.\n\n\n3. Subagents as the Primary Scaling Mechanism\nWatch an agent tackle a complex task. It reads files, searches the codebase, backtracks, tries a different approach. The context window fills up. By the time it\u2019s ready to write code, it\u2019s forgotten the original requirements. It loops. It hallucinates. It fails.\nI hit this wall constantly when researching new topics. One PDF can consume your entire context window. A large codebase is even worse. You end up working in awkward, inconvenient ways just to get access to the knowledge you need.\nAn obvious fix is to \u201cmake context windows bigger.\u201d But bigger windows just delay the problem. They\u2019re slower and more expensive.\nA better approach is factoring: break the work into isolated subtasks, each with its own context. A subagent searches the codebase and returns only the relevant file paths. Another analyzes a specific function and returns only its findings. The parent agent stays focused, receiving distilled results instead of accumulating everything.\nSubagents are to agents what functions are to programs. They isolate context for subtasks and return only relevant results to the parent. They can be optimized for different task types, whether speed or depth, and they enable parallelism without context collision.\nThe difference is dramatic. A single-agent approach might use 90% of its context window stumbling through a task. The same task with subagents? The parent uses 25%, staying sharp throughout.\nSpawning subagents should feel as natural as calling a function. Not an advanced feature you learn later, but the default pattern from day one.\nFramework implication:\n// This should feel as natural as a function call\nconst result = await agent.delegate(\"searcher\", {\n  task: \"locate authentication logic\",\n  returnFormat: \"file_paths_with_snippets\"\n});\nThe framework also needs primitives that make context management automatic:\nconst agent = new Agent({\n  contextBudget: 128_000,           // Explicit budget, inherited by subagents\n  overflowStrategy: \"summarize\",    // What to do when approaching limit\n  loopDetection: true               // Circuit-break on repetitive patterns\n});\nWith these primitives, agents can focus on the task while the framework handles the resource management. Just like garbage collection lets programmers focus on logic instead of memory.\n\n\n4. Opinionated Subagent Archetypes\nWatch enough production agents and you\u2019ll see the same patterns emerge. Teams independently discover that certain subtask shapes keep recurring, and that specialized subagents for those shapes dramatically outperform general-purpose ones.\nHere are the archetypes that have proven themselves:\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\nArchetypeWhat it doesWhy it\u2019s specializedSearcherFinds relevant code, files, symbolsOptimized for speed; returns locations, not contentThinkerReasons through complex problemsAllocates thinking time; returns analysis, not actionResearcherGathers external knowledge (docs, APIs)Web/retrieval access; returns synthesized contextWriterMakes targeted code changesDiff-focused; returns patches, not explanationsPlannerBreaks down complex tasksStrategic focus; returns task breakdownCheckerValidates and critiques workIndependent perspective; adversarial stance\nNotice the pattern: each archetype has a constrained output format. A Searcher returns locations. A Thinker returns analysis. A Writer returns patches. This constraint is the key. It forces the subagent to distill its work rather than dump everything into the parent\u2019s context.\nThese archetypes should ship as built-in, customizable components. You shouldn\u2019t have to reinvent the Searcher pattern from scratch. It should just be there, ready to use, with sensible defaults you can override when needed.\n\n\n5. Human-Agent Workflow is Part of the Framework\nThe bottleneck has shifted. It\u2019s no longer \u201ccan the agent write code?\u201d It\u2019s \u201ccan the human review and trust the code fast enough?\u201d\nToday\u2019s workflow: agent dumps a wall of changes, human squints at diffs, tries to understand the intent, hopes nothing broke. This doesn\u2019t scale. As agents get more capable, the review burden grows faster than human attention.\nWhen you run an agent, you should be able to specify where you want control. Checkpoints give you natural stopping points for review. Change proposals explain what will change and why before changes are made. Validation hooks let you plug in secondary verification (another model, static analysis, or tests). Diff streaming gives real-time visibility into changes as they happen. And approval gates let you require sign-off for high-risk operations.\nHere\u2019s what that looks like:\nconst result = await agent.run({\n  task: \"Add rate limiting to /api/users\",\n  checkpoints: [\"before_write\", \"after_plan\"],\n  requireApproval: [\"delete_file\", \"modify_config\"],\n  onProposal: (plan) => showPlanToUser(plan)\n});\nThis isn\u2019t UI. It\u2019s protocol. The framework defines the contract between agent and human; UIs implement it. A CLI might show a simple approve/reject prompt. An IDE might render an interactive diff viewer. Same protocol, different presentations.\n\n\n6. Tools are Curated, Not Collected\nThe promise of universal tool interoperability sounds great in theory. In practice: more tools = more confusion.\nHere\u2019s why. Every tool you give an agent is a decision it has to make: \u201cShould I use this?\u201d With 5 well-chosen tools, that decision is easy. With 50 tools, many overlapping, some poorly documented, others rarely useful, the agent wastes context evaluating options, makes false starts with wrong tools, and loses focus on the actual task.\nI ran into this recently with MCP (Model Context Protocol) servers. I built one for Outlook to read my mail and calendar. But with multiple MCP servers installed, the agent kept trying to look up information on the web instead of using the mail tool sitting right there. Too many options, not enough guidance about which one to pick.\nWhat you want is a curated core toolset optimized for coding tasks, not a grab-bag of integrations. Different subagents should get different tool subsets: a Searcher doesn\u2019t need web access. Destructive tools should require human approval. And tool usage patterns should be tracked to surface which tools help vs. which cause thrashing.\nFramework implication:\nconst searcher = new Searcher({\n  tools: [\"grep\", \"ast_search\", \"file_read\"],  // Curated subset\n  restricted: [\"web_fetch\", \"file_write\"]      // Explicitly excluded\n});\n\n// Tools are first-class objects with metadata\nconst grepTool = {\n  name: \"grep\",\n  description: \"Search file contents with regex\",\n  whenToUse: \"Looking for specific strings or patterns in code\",\n  costEstimate: \"low\",\n  requiresApproval: false\n};\nThe goal: agents that pick the right tool immediately, not agents that waste time evaluating irrelevant options.\n\n\n7. When to Use a Subagent vs. a Tool\nThere\u2019s a simple heuristic for deciding whether something should be a tool or a subagent:\nTool: Stateless transformation. Input goes in, output comes out, no iteration required. Format a date. Parse JSON. Run a regex. The operation doesn\u2019t accumulate context or require judgment.\nSubagent: Iteration, judgment, or context accumulation. The operation explores, backtracks, tries alternatives, or builds up intermediate state. Search, research, analysis, planning: these need their own context window.\nThe architectural reason to care: if you implement an iterative operation as a tool, its entire execution trace pollutes the parent context. Every search result, every intermediate step, every dead end. All of it crowds out space for the actual task.\n// Bad: \"tool\" that leaks context\nconst searchTool = {\n  name: \"search_codebase\",\n  execute: async (query) => {\n    // All of this ends up in parent context\n    const files = await grep(query);\n    const ranked = await rerankResults(files);\n    const snippets = await extractSnippets(ranked);\n    return snippets;\n  }\n};\n\n// Good: subagent with isolated context\nconst searcher = new Searcher({\n  // Runs in own context window\n  // Only `snippets` returned to parent\n  returns: \"snippets_with_locations\"\n});\nThis matters because many things we call tools are actually subagents in disguise. Web search, documentation lookup, code analysis: these involve iteration and judgment. The name \u201ctool\u201d makes them sound simple, but they\u2019re not. Implementing them as proper subagents with isolated context is what makes the archetype pattern (Searcher, Thinker, Researcher) work.\n\n\n8. Subagent Context is Ephemeral by Default\nA subagent\u2019s working context is like local variables in a function. It exists during execution and is discarded when done.\nConsider a Researcher subagent investigating a library. It fetches 10 documentation pages, reads API references and examples, follows links to GitHub issues, and synthesizes findings into a summary. All of that is working context. But it returns only the summary to the parent.\nThe parent doesn\u2019t need the 10 pages, the API refs, or the GitHub issues. It needs the answer. All that intermediate work is temporary scaffolding.\nconst researcher = new Researcher({\n  // Everything inside runs in ephemeral context\n  // Only the return value persists\n  returns: \"synthesis\",\n\n  // Optional: persist specific artifacts\n  persist: [\"key_code_snippets\", \"api_signatures\"]\n});\n\n// Parent receives ~500 tokens, not 50,000\nconst findings = await agent.delegate(researcher, {\n  task: \"How does auth work in this library?\"\n});\nThe principle: Subagents accumulate context to do their job, then compress before returning. The parent receives a distilled result, not the full execution trace.\nI built something like this to explore a large codebase. A custom agent walked through the source files, read them, and wrote summarized markdown documentation. The agent consumed thousands of lines of code, but what I got back was a concise capture of the service\u2019s flow. The essence, not the exhaustive detail. That\u2019s the pattern: do the heavy lifting in isolated context, return only what matters.\nThis is how human experts work too. When you ask a colleague to research something, you want their conclusion, not everything they read along the way.\n\n\nThe Developer Experience\nWhat Building an Agent Should Feel Like\nAll of the principles above collapse into a simple question: what does it feel like to build with this framework? If the conventions are right, the code should be obvious. You declare what you want, not how to manage it. The framework handles orchestration, context, and human checkpoints. You focus on the task.\nHere\u2019s what that looks like:\nimport { Agent, Searcher, Thinker, Writer } from \"agentkit\";\n\n// Declare a coding agent with specialized subagents\nconst agent = new Agent({\n  name: \"code-assistant\",\n  mode: \"smart\", // vs \"fast\" for quick iteration\n  subagents: [\n    new Searcher({ tools: [\"grep\", \"ast_search\", \"embeddings\"] }),\n    new Thinker({ timeout: 120 }),\n    new Writer({ requireApproval: false }),\n  ],\n  contextBudget: 128_000,\n  humanCheckpoints: [\"before_multi_file_edit\", \"on_uncertainty\"],\n});\n\n// Run a task - framework handles subagent orchestration\nconst result = await agent.run({\n  task: \"Add rate limiting to the /api/users endpoint\",\n  codebase: \"/path/to/repo\",\n});\n\n// Result includes structured output for UI integration\nresult.changes;      // List of file changes\nresult.proposal;     // What changed and why\nresult.contextUsed;  // Debugging/optimization info\nWhy TypeScript?\nTypeScript offers patterns that make agent code safer in ways that Python\u2019s type system can\u2019t match.\nBranded types for resource budgets. Context tokens aren\u2019t just numbers. They\u2019re a distinct unit. Branded types prevent you from accidentally passing a line count where a token count is expected:\ntype ContextTokens = number & { readonly brand: unique symbol };\nconst budget: ContextTokens = 64_000 as ContextTokens;\n// Can't accidentally pass a plain number where ContextTokens is required\nDiscriminated unions for checkpoint states. When a checkpoint can be pending, approved, or rejected, discriminated unions force exhaustive handling. The compiler catches missing cases at build time, not runtime:\ntype CheckpointState =\n  | { status: \"pending\" }\n  | { status: \"approved\"; by: string; at: Date }\n  | { status: \"rejected\"; reason: string };\n\nfunction handleCheckpoint(state: CheckpointState) {\n  switch (state.status) {\n    case \"pending\": return showWaiting();\n    case \"approved\": return proceed(state.by);\n    case \"rejected\": return showError(state.reason);\n    // TypeScript errors if you miss a case\n  }\n}\nGeneric constraints for subagent return types. A Searcher<FileLocation[]> and a Thinker<Analysis> are different types. The parent agent knows exactly what shape to expect from each delegation:\nconst locations = await agent.delegate<FileLocation[]>(searcher, { task: \"find auth\" });\n// locations is typed as FileLocation[], not unknown\nBeyond type safety, the ecosystem fits: VS Code extensions, language servers, and most developer tooling already run on TypeScript. Using Bun, agents compile to standalone executables with no runtime dependencies.\n\nWhat This Enables\nThink about what web development looked like before Rails. Every project started with the same decisions: How do I structure my code? How do I talk to the database? How do I handle routing? Teams spent months on plumbing before writing a single line of business logic. Rails changed that by encoding the answers into conventions. Suddenly, developers could go from idea to working application in hours instead of weeks.\nAgent development is in that pre-Rails moment right now. Every team building production agents is solving the same problems: context management, subagent orchestration, human review workflows, tool selection. The solutions exist, but they\u2019re locked inside proprietary systems or tribal knowledge.\nWith the right framework primitives, agent builders can focus on what makes their agent unique instead of reinventing context management for the hundredth time. They can experiment with novel interaction patterns instead of debugging the same doom loops. They can invest in evaluation and domain expertise instead of infrastructure.\nAnd users get agents that actually work. Not agents that succeed on simple tasks and fall apart on complex ones. Not agents that require babysitting to avoid going off the rails. Agents with predictable behavior, transparent operation, and graceful degradation when things get hard.\n\nOpen Questions\nThere\u2019s still work to figure out. These are the problems I\u2019m thinking through, along with my current hunches.\nHow do subagents share learned context? There\u2019s probably a \u201cmemory\u201d layer that persists across tasks. Something like a project-level knowledge base that subagents can read from and write to. The Researcher finds something important, it goes into shared memory. The Writer pulls from it later. This doesn\u2019t have to be LLM-backed memory. It could be a graph database, a structured knowledge store, or something we haven\u2019t invented yet. But the interface has to be simple enough that it doesn\u2019t become another configuration burden.\nHow do peer agents collaborate? Most agent architectures are hierarchical. Parent spawns child, child returns result. But some tasks need peers working in parallel, sharing findings as they go. I suspect the answer is message-passing with typed channels, similar to how concurrent systems handle coordination. The framework handles the plumbing; agents just send and receive.\nHow do you evaluate agents systematically? This might be the hardest problem, and I don\u2019t think anyone has solved it yet.\nThe core challenge is ground truth. To evaluate whether an agent did the right thing, you need to know what \u201cright\u201d means for that task. For coding tasks, you can check some things automatically (does the code compile? do tests pass?) but these only catch obvious failures. An agent can produce working code that\u2019s architecturally wrong, or secure code that\u2019s unmaintainable.\nI think evaluation has to happen at multiple layers. Task success is the baseline: did the change work at all? Behavioral consistency asks whether the same task on the same codebase produces similar results across runs. If an agent gives wildly different answers each time, something\u2019s wrong. Human preference captures whether humans actually accept the changes in practice.\nThe infrastructure for this is expensive to build. You need captured scenarios (codebase + task + expected behavior), replay with controlled randomness, and scoring that\u2019s more nuanced than pass/fail. It\u2019s similar to how you\u2019d evaluate a human candidate through realistic work samples rather than abstract tests. The framework should probably include hooks for scenario capture and replay, even if the evaluation logic is user-defined.\nThere\u2019s also a meta-problem: evaluating the evaluations. A scenario that tests whether the agent can add two numbers tells you nothing useful. The scenarios themselves need to be validated for difficulty and discriminative power. Do they distinguish good agents from bad ones? Do they catch regressions? This is the same challenge test engineers face with code coverage: 100% coverage means nothing if the tests are trivial.\nMy intuition is that evaluation requires three primitives: Scenario (a frozen codebase + task), Assertion (did the output compile, pass tests, match intent), and Consistency (same scenario, N runs, what\u2019s the variance). This is closer to integration testing than unit testing.\nHow do you budget spend across subagent hierarchies? Context tokens are one cost, but API calls add up fast when you\u2019re spawning subagents. The framework probably needs a cost model that propagates budgets down the hierarchy. Parent allocates a budget, subagents inherit portions of it, and the framework enforces limits before you get a surprise bill.\nWhat happens when a subagent fails? The boring answer is probably the right one: retry with exponential backoff, then escalate to the parent with an error. The parent decides whether to substitute a different approach or surface the failure to the user. Automatic substitution sounds clever but might hide problems that humans should see.\n\nWhat\u2019s Next\nThe patterns are emerging. Production agent systems have proven what works. Now we need to turn these lessons into infrastructure that benefits everyone.\nI started this post frustrated with complexity. Every agent framework I tried added abstraction layers without solving the hard problems. Configuration options multiplied while agents still choked on context. New features shipped while doom loops went unfixed. I kept trying new tools and spending more time learning their paradigms than actually getting work done. That\u2019s a red flag. It means we haven\u2019t found the right way to look at this space yet.\nRuby on Rails became popular not because it introduced a lot of new ideas, but because it made easy things trivial and hard things possible in a reasonable amount of time. That\u2019s why everyone learned Ruby. We need the same thing for agents.\nThe answer isn\u2019t more complexity. It\u2019s the right complexity: conventions that encode hard-won lessons, primitives that make the correct architecture easy, and defaults that work out of the box.\nI don\u2019t think there\u2019s one framework to rule them all. But the patterns matter. Whether you\u2019re building agents, evaluating frameworks, or just trying to understand why your agent keeps running npm run build in a loop, these architectural ideas can help you reason about what\u2019s going wrong and what to try next.\nIf more people internalized these patterns, we\u2019d all waste less time on plumbing and more time on problems that actually matter. And that\u2019s the point."
                ],
                "output": "readability/",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:21.102721+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://blog.bryanl.dev/posts/agent-framework-vision/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-01-04T05:05:20.507360+00:00",
                "index_texts": null,
                "output": "Agents Done Right: A Framework Vision for 2026",
                "pwd": "/data/archive/1767503101.628046",
                "schema": "ArchiveResult",
                "start_ts": "2026-01-04T05:05:20.464665+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://blog.bryanl.dev/posts/agent-framework-vision/']' timed out after 60 seconds",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "Agents Done Right: A Framework Vision for 2026",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1767503101.628046",
    "newest_archive_date": "2026-01-04T05:05:42.967129+00:00",
    "num_failures": 1,
    "num_outputs": 8,
    "oldest_archive_date": "2026-01-04T05:05:09.124713+00:00",
    "path": "/posts/agent-framework-vision/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KE3P9ZR1CBB3A51C01SH3QPJ",
    "snapshot_id": "62f3ed43-0876-4e7c-b93d-0b663311ded2",
    "sources": [
        "/data/sources/1767503100-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1767503101.628046",
    "title": "Agents Done Right: A Framework Vision for 2026",
    "url": "https://blog.bryanl.dev/posts/agent-framework-vision/"
}