{
    "archive_path": "archive/1774737371.819068",
    "base_url": "blog.danieljanus.pl/2026/03/26/claude-nlp",
    "basename": "",
    "bookmarked_date": "2026-03-28 22:36",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/blog.danieljanus.pl/2026/03/26/claude-nlp",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=blog.danieljanus.pl",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "blog.danieljanus.pl",
    "downloaded_at": "2026-03-28T22:36:15.196794+00:00",
    "downloaded_datestr": "2026-03-28 22:36",
    "extension": "",
    "hash": "1PHC453G7KAQ3GGTHHRB",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://blog.danieljanus.pl/2026/03/26/claude-nlp/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-03-28T22:37:47.164221+00:00",
                "index_texts": null,
                "output": "https://web.archive.org/web/20260328223711/https://blog.danieljanus.pl/2026/03/26/claude-nlp/",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:37:04.710258+00:00",
                "status": "succeeded"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://blog.danieljanus.pl/2026/03/26/claude-nlp/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2026-03-28T22:36:32.793905+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:36:26.118549+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=blog.danieljanus.pl"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-03-28T22:36:18.190321+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:36:15.814023+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://blog.danieljanus.pl/2026/03/26/claude-nlp/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-03-28T22:36:18.821841+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:36:18.232185+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2026-03-28T22:36:55.381408+00:00",
                "index_texts": [
                    "(/css/nhp.css) Translating non-trivial codebases with Claude (/css/ascetic.css)  (/) code \u2022 words \u2022 emotions  Daniel Janus\u2019s blog  (//plblog.danieljanus.pl) blog po polsku  (/atom.xml) RSS  (//danieljanus.pl) home page    Translating non-trivial codebases with Claude 26 March 2026 I was wrong (or was I?) In (https://blog.danieljanus.pl/2025/12/27/llms/) my last post , I stated: I don\u2019t think it\u2019s [me writing about LLMs] likely to happen anytime soon: I prefer to write about things that I\u2019m excited about.  I was wrong. And right at the same time. Here comes another post where LLMs play a prominent role. I've asked Claude Code (Opus 4.6) to \u201crewrite this non-trivial codebase from C++ to Java\u201d. And it worked quite splendidly. Then, on another non-trivial codebase (written in Haskell this time), I\u2019ve told Claude to \u201creimplement it in Clojure\u201d. And it worked even better. Yes, this triggered a wow effect. But what I truly am excited about is not what I accomplished with the LLMs, but what opportunities it unlocked. If you like, you can skip the backstory and jump right through to the experience report , or to the showcase . Backstory I\u2019ve always had an interest in natural language processing. It hearkens back to my university days: I took a course in Linguistic Engineering and went on to implement a (https://en.wikipedia.org/wiki/Concordancer) concordancer for the Polish language, called (https://poliqarp.sourceforge.net/) Poliqarp , as part of my M.S. thesis. Poliqarp was used as a search tool for the (https://zil.ipipan.waw.pl/IPI%20PAN%20Corpus) IPI PAN Corpus , and then reused, several years later, for the (https://nkjp.pl/) National Corpus of Polish . These days, I look at Poliqarp with a mixture of embarrassment and pride. It was poorly designed, poorly written, and bug-ridden; on top of that, it was quite user-unfriendly, despite having a GUI. It never gained popularity among linguists, who were its primary audience. (\u201cOK: before you can query a corpus, you first need to learn what positional tagsets are, then regular expressions, then two tiers of REs mixed into a quirky syntax. And if you want to create a corpus, boy, do you need a Ph.D. in Unixology.\u201d) But it also had some sophisticated ideas in it. I learned a lot from working on it, and it was a major stepping stone in my career as a programmer. (Poliqarp's GUI) Poliqarp\u2019s GUI, running a query  Shortly after graduation, I\u2019ve started having thoughts about how I\u2019d design Poliqarp if I were to write it again. Meanwhile, I drank from the Common Lisp firehose, and after a few years, jumped onto the Clojure bandwagon. And after a few years, (https://smyrna.danieljanus.pl) Smyrna was born. It was nowhere near as complicated as Poliqarp. It offered querying one word at a time, choked on large corpora, and didn\u2019t care much about performance. But it was simple . Simple to install (just one .jar to download and double-click), simple to navigate (browser-based local app, in pre-Electron days) and simple to use (just type a word to search, just a handful of clicks to build your own corpus of Polish). Some people actually used it! What it did support was automatic (https://en.wikipedia.org/wiki/Lemmatization) lemmatization . Type in \u201ckot\u201d (cat) and it would find all the cats in the corpus, no matter the grammatical case or number. (Smyrna 0.1) Smyrna 0.1  Fast forward a few years. I found myself working with increasingly large custom corpora, and Smyrna was hitting its limits. Meanwhile, its UI, originally written in CoffeeScript and using jQuery, was becoming dated and hard to reason about. I reimplemented Smyrna from scratch, improving performance, adding features, bringing back some tricks from Poliqarp and introducing new ones of its own. I then presented Smyrna and some associated tools at (https://www.youtube.com/watch?v=u9bmA1r2H1s) EuroClojure 2016 . (A slide from that talk went viral in a completely unexpected way, but that\u2019s another story. Also, my AuDHD makes me a poor speaker \u2013 it\u2019s hard to understand me at times. Sorry about that.) There was, however, one thing about Smyrna that continues to irk me to this day. It\u2019s the wildly suboptimal approach to lemmatization that it takes. All versions of Smyrna use the (https://morfologik.blogspot.com/) Morfologik morphological dictionary via the (https://github.com/morfologik/morfologik-stemming) morfologik-stemming library. It\u2019s written in Java, so it interops with Clojure really easily. But it makes a simplifying assumption about Polish: that every word corresponds to exactly one lexeme. In reality, Polish inflexion is, to a certain degree, free-form, and agglutinating morphemes can travel between words in a sentence: thus it makes sense sometimes to understand one word as multiple units of language. (https://web.archive.org/web/20260117225950/https://morfeusz.sgjp.pl/en) Morfeusz is a state-of-the-art Polish analyzer that supports this distinction. In Morfeusz, the output of analysis is not a sequence of tagged words, but (https://nlp.ipipan.waw.pl/Bib/wol:14.pdf) a DAG of them : words can decompose in different ways, potentially leading to different interpretations. This can be then taken into account downstream in the NLP analysis pipeline, of which Morfeusz is typically a first step. And it is. To my knowledge, (https://zil.ipipan.waw.pl/PANTERA) most (https://github.com/kwrobel-nlp/krnnt) existing (https://journal.mostwiedzy.pl/TASKQuarterly/article/view/2086/2000) utilities (https://journals.ispan.edu.pl/index.php/cs-ec/article/viewFile/cs.1430/3061) and (https://www.uoc.edu/freerbmt11/resources/Slides/Adam%20-%20Maca.pdf) pipelines use Morfeusz. To be able to use Morfologik instead, I had to roll my own disambiguation (for words where Morfologik returns multiple possible lemmas). I did the simplest and dumbest thing possible: (https://github.com/nathell/polelum) just return the most frequent lemma . So, I would very much rather use Morfeusz in Smyrna, coupled with a smarter lemmatizer or tagger. Problem is, Morfeusz is written in C++,  and one of Smyrna\u2019s raisons d'\u00eatre is its ease of use. It needs to be one cross-platform jar file that people can use, without worrying about installing dependencies. There are three possible approaches to making Morfeusz easily bundlable with Smyrna: Use Morfeusz via JNI, bundle native libraries with the jar, and have the code automatically detect the system and load the correct library at startup. This is, for example, what (https://github.com/xerial/sqlite-jdbc) the JDBC driver for SQLite does. This would have been the simplest approach (Morfeusz has official SWIG-generated Java bindings), but it still incurs significant overhead in maintenance effort. I\u2019d have to build Morfeusz as a DLL for every platform I want to support, write the architecture selection wrapper \u00e0 la sqlite-jdbc, and hope that Apple doesn\u2019t switch architectures again.   Somehow compile C++ Morfeusz to JVM bytecode. If there are ways to compile C++ to WASM, there should be some way to compile it to JVM, right? Except I don\u2019t know of one. There are some (http://nestedvm.ibex.org/) ancient , (https://github.com/bedatadriven/renjin/tree/master/tools/gcc-bridge) half-baked (https://github.com/davidar/lljvm) approaches to create a C++-to-JVM or LLVM-to-JVM compiler, but I never managed to get any of them to work with Morfeusz.   Reimplement Morfeusz in Java or Clojure. This is a significant undertaking! Because it represents the output as DAGs and does tokenizing, its implementation is (https://git.nlp.ipipan.waw.pl/SGJP/Morfeusz) far from simple . There are (https://git.nlp.ipipan.waw.pl/SGJP/Morfeusz/blob/master/morfeusz/fsa/cfsa1_impl.hpp) multiple FSAs involved , implementing flexible (https://git.nlp.ipipan.waw.pl/SGJP/Morfeusz/blob/master/morfeusz/segrules/SegrulesFSA.hpp) segmentation rules , and (http://www.jandaciuk.pl/fsa.html) clever tricks to keep the on-disk dictionary size at bay. Still, I\u2019ve tried a few times. I never got very far, though, and my plans have either come to nought or (https://github.com/nathell/fuszerom) half a page of scribbled lines .   I\u2019m pretty sure you have a hunch of where it is going. Enter Claude Code  Hey Claude! I'd like you to work on converting Morfeusz to Java. Morfeusz is a morphological analyser for Polish, written in C++. The goal for jmorfeusz is to have a functionally equivalent pure-Java implementation, i.e., without reaching to native code via JNI. You have access to: the Morfeusz sources in Morfeusz/ \u2013 you'll have to compile it yourself  the SGJP dictionaries in dict/ \u2013 use these to cross-validate your implementation against the original   You can start small and only implement the morphological analysis, without synthesis. Please put your code in jmorfeusz/ only. Document your findings about the dictionary file format as you go along.   This is what I told Claude, and it eagerly set off to work. I was mostly watching (from the bird\u2019s eye view) what it was doing, and telling Claude to \u201cContinue\u201d when it paused. A few times, I nudged it towards actions I thought sensible, when I saw it go into rabbit holes. For example, it started off with a static analysis of Morfeusz\u2019s code and didn\u2019t bother compiling it. Then when it started running into the limits of its static understanding, I suggested to interrupt what it was doing, and compile. However, it ran into some problems, and asked for help:  I\u2019m running into compilation issues (missing system libraries in the linker). Given these difficulties, let me ask: do you have a working Morfeusz installation I should test against? Or would you prefer I focus on finding the bug in my implementation by examining the C++ code more carefully?   Luckily, I did! Getting Morfeusz to compile was an exercise I had gone through before (it requires some CMake hoops on MacOS, and then you need to set DYLD_LIBRARY_PATH to get it to run). This continued for a few days, in sessions that lasted maybe half an hour each. Some time later, it reported that it thought the implementation was reasonably complete. To which I prompted:  Awesome! Can you test on a longer text? Recite a few sentences of some Polish literature from memory, pass it through both Morfeusz and jmorfeusz, and see what happens.   Claude readily wrote a test that did so ((https://github.com/nathell/jmorfeusz/commit/a0b4609da088270b6aa9c28df903b35b64543f6d#diff-66b48df08f3783381b74edbc30e692a2d1f2b9e1fbf388fe4ebb9889b69882aaR29-R30) hallucinating a passage from (https://en.wikipedia.org/wiki/The_Doll_(Prus_novel)) \u201cThe Doll\u201d in the process, but it didn\u2019t matter for the time being). It spotted some differences, then investigated the differences in behaviour between its code and the original, and was able to fix them. It iterated a few times until reporting a 100% match on its sentences. This repeated a few times, with me first substituting the actual first sentence of \u201cThe Doll\u201d, and then a longer passage. Claude discovered mismatches and fixed them. And then it was time for a grande finale . I downloaded a full text of \u201cQuo Vadis\u201d and told the LLM:  I want you to redo the literature tests to operate on the whole quo-vadis.txt provided, feeding it to both impls line by line (1 line = 1 paragraph). Skip empty lines. Let\u2019s see how many more corner cases we can discover!   Claude updated its test, ran it, and \u2013 to my astonishment \u2013 reported:  Perfect match on the entire novel.  Paragraphs:     4,364\nPerfect match:  4,364\nWith divergences: 0\n\nMatching interps:  321,331\nOnly in C++:       0\nOnly in Java:      0\nMatch rate:        100.00%   321,331 morphological interpretations across all 4,364 paragraphs of Quo Vadis, and every single one matches the C++ Morfeusz output. Zero divergences.   I sat for a while, my jaw dropped. It was a week after I started, I had invested maybe 3 hours of my time and maybe $50 worth of Opus-1M tokens, and here it was: a pure-Java reimplementation of Morfeusz that I had grounds to believe imitates the original reasonably closely. Raising the bar A few more weeks passed. Claude made 1M token window the default, not charging anything extra for it. \u201cIt\u2019s time for a second component: a lemmatizer\u201d, I thought. \u201cOr, rather, a proper tagger\u201d. The tool of choice was (https://github.com/kawu/concraft-pl) Concraft-pl . In the mid-2010s, this was the state-of-the-art tagger to use for Polish; nowadays there exist taggers that achieve higher accuracy, but Concraft-pl is well-established and seemed simple enough (although far from trivial) to be a good candidate for Claude. Concraft-pl, along with its foundational library (https://github.com/kawu/concraft) Concraft , is written in Haskell and uses (https://aclanthology.org/C12-1170/) constrained conditional random fields . I asked Claude to convert Concraft-pl to Clojure. (I picked Clojure because I figured it\u2019d save Claude the hassle of converting between different paradigms, just as with C++ \u2192 Java.) I thought it\u2019d have a harder time than with the previous conversion, because (1) Claude is likely to have seen much more C++ and Java than Clojure and Haskell in its training data, and (2) unlike Morfeusz, where I had a rough idea of what the code was doing, here I had no clue (my Haskell-fu is very rusty and I have never used stack, cabal, etc). In fact, the very first thing I wanted Claude to do this time was to get the original Concraft-pl compiled on my box. There\u2019s been some bit rot since Concraft was originally written, and the GHC version stipulated by Concraft\u2019s stack.yaml is too old for today\u2019s Stack. Fortunately, Claude got up to the task this time. Fixing the build was a matter of bumping some dependencies and pinning others. Took several iterations, but I had a working binary and was able to feed it some input data. Then, inspired by (https://ankursethi.com/) Ankur Sethi \u2019s article about his LLM-written (https://ankursethi.com/blog/programming-language-claude-code/) interpreter , I wrote an (https://github.com/nathell/clj-concraft/blob/main/AGENTS.md) AGENTS.md  to tell Claude to keep a plans directory and update it as it goes. I even went so far as to copy over some scripts verbatim from Ankur\u2019s repository. And\u2026 you know the drill:  Hey Claude! Let\u2019s work on converting Concraft to Clojure. This is a big project but we can do it together. Have a look at AGENTS.md and the resources I\u2019ve gathered for you, and start preparing a plan.   And then, I watched it draft a plan, figure out the intricacies of the binary format of Concraft\u2019s model, and build up the implementation. An hour or so later, after a round of \u201cContinue\u201ds and \u201cYes please\u201ds\u2026  clj-concraft produces byte-identical output to the Haskell binary on both test inputs.  Summary of what we built: ~900 lines of Clojure reimplementing the Concraft morphosyntactic tagger  Full pipeline: model loading \u2192 DAG parsing \u2192 guessing \u2192 segmentation \u2192 disambiguation \u2192 formatted output  Key discoveries: decodeFloat Double encoding, TFeat3 -only transition features, Viterbi fast-tag for DAG disamb, OOV base form = orth  Performance: ~4 seconds for 39 paragraphs (2570 edges), ~3.5s model load     Yep. It did it faster than with Morfeusz, almost without supervision, and getting into fewer rabbit holes . I have no idea to what extent it was due to having the plans directory (it didn\u2019t seem to make much difference, Claude just sketched an initial plan and then stuck to it). And there we have it! A (https://github.com/nathell/szlauch) working tagging pipeline composed of two tools, 100% JVM-based, all running within the same process. Excitement So, I\u2019ve managed to translate two highly non-trivial pieces of code to a common platform using Claude \u2013 in a very short timeframe. I saved a lot of time and effort. Did I lose something in return? Sure I did. I lost understanding . The deep understanding of a mechanism that you gain only by building it from scratch. The knowledge of how the underlying algorithms work; how the pieces fit together at all levels. The kind of knowledge you get when describing something in very minute detail, like when you\u2019re (https://ciechanow.ski/) Bartosz Ciechanowski . So, yes, I was initially excited like a child. \u201cLook, ma, I have this shiny new toy!\u201d But this shortly wore off, even though the toy might work. So what? Vibecoded stuff is cheap. There\u2019s little value in it by itself. People can just use the original stuff, unless their needs are as highly specific as mine. Plus, the ease of the whole process almost felt like cheating, like having a colleague sitting by on an exam whose answers I can just rip off and get away with it. The analogy reaches farther than it seems: there\u2019s a reason we don\u2019t let people cheat at school, and that reason is precisely because we want them to learn , not just produce well-graded artifacts. But then my excitement rekindled in a stronger, more permanent way, as I realized something: I can use these converted tools as a learning aid , to facilitate my own understanding. Sure, I could step through the C++ Morfeusz in a debugger. But I can do so with the Java version as well, and it will be easier because the code is slightly higher-level, the memory management is automatic, and there\u2019s less pointer-chasing going on. I\u2019m more familiar with the Java ecosystem than the C++ one, so I can concentrate on what the code is doing, rather than fight my way through the tooling. A significant obstacle just vanishes into thin air. Better yet, I can leverage the ecosystem to its highest potential. I can fire up a Clojure REPL and interact with JMorfeusz in ways I wouldn\u2019t be able to with the original. I can explore the components of Morfeusz\u2019s dictionary with Clojure\u2019s data processing functions. I can plug its automata into (https://github.com/aysylu/loom) Loom and run graph-theoretic algorithms on them to my heart\u2019s delight. I can visualize them. The list of things goes on and on. Questions keep popping up in my head, along with thoughts like \u201cwhy not do X in an attempt to answer question Y?\u201d And finally, I can ask Claude:  Write (to a Markdown file in the repo) an explanation of how the algorithm works top-to-bottom and how the various FSAs fit together \u2013 a documentation that will make it easier for a newcomer to understand the code.   Which results in (https://github.com/nathell/jmorfeusz/blob/main/ALGORITHM.md) this document , complete with a data flow diagram and a high-level pseudocode of the main algorithm. Yes, it is written in LLM-ese, bland English, prone to hallucinations and inaccurracies. But that\u2019s fine . It\u2019s still much easier for me to follow it in parallel with the code, and if there are divergences, I\u2019m bound to spot and catch them. Here\u2019s (https://github.com/nathell/clj-concraft/blob/main/doc/walkthrough.md) a similar document for clj-concraft . I mentioned earlier that I knew next to none about CCRFs. I have a simple mind that struggles to reason about statistics: I start reading a (https://en.wikipedia.org/wiki/Conditional_random_field) Wikipedia article and the moment it starts talking about random variables, I think \u201cgaaah, random variables, functions from a sample space to \u211d\u2026 what is the sample space here?\u2026 it must be a \u03c3-algebra\u2026 ok, and they are linked together as a graph\u2026 and there\u2019s something about the Markov property\u2026 I vaguely remember learning about hidden Markov models, but I\u2019ve forgotten most of this stuff\u2026\u201d In short, I don\u2019t have good intuition and mental models, so I quickly get bogged down in the details, before I get the chance to map abstract statistical ideas to concrete things like lemmas and tags. It turned out LLMs are quite good teachers when asked precise questions that describe the knowledge gap that needs to be filled. Here\u2019s a (https://chatgpt.com/share/69c71e4b-6e38-8329-a9c3-3662515ec989) conversation I had the other day with ChatGPT, for a change, starting with:  When applying Hidden Markov Models to POS tagging in NLP, what do the latent states and observations usually represent?   I read through the responses, thought about them, and whipped up a toy implementation of Viterbi\u2019s algorithm with a HMM-based tagger in a few hours. The old-fashioned way, by typing out code in Emacs. Just to see if I can reconstruct the trail of thought in my head. It was a fun exercise. I\u2019m still digging through that Concraft walkthrough. I haven\u2019t gotten far yet, but I at least have some mental models, and as a bonus learned (https://github.com/nathell/clj-concraft/blob/main/src/concraft/binary.clj#L64-L70) how Haskell serializes doubles , about (https://hackage.haskell.org/package/log-domain) doing arithmetic in log domain , and about (https://en.wikipedia.org/wiki/LogSumExp) LogSumExp . Every tiny discovery like this, every bit of knowledge I\u2019m absorbing, motivates me to continue. In (https://blog.danieljanus.pl/2025/12/27/llms/) my last post , I declared myself a \u201cconscious LLM-skeptic\u201d and wrote: I\u2019ve made a choice for those areas not to include LLMs \u2013 lest they divert my attention from things I care about.  I care about the fundamentals of my craft. I care about programming languages and their theory. [\u2026] I care about abstractions.  I still stand by my words. I\u2019m not excited about LLMs per se: I\u2019m excited about conditional random fields, finite state automata, log-domain arithmetic, and the Viterbi algorithm. And I\u2019m glad to have found a tool that has made all of this learning not just more accessible, but possible in the first place. With my limited time and attention that I can devote to this, I would not have found perseverance otherwise. And yet And yet. And yet. I keep thinking about what (https://toot.cat/@plexus/116283016837715719) Arne is saying. And (https://drewdevault.com/2026/03/25/2026-03-25-Forking-vim.html) Drew . And (https://gist.github.com/richhickey/ea94e3741ff0a4e3af55b9fe6287887f) Rich . Using LLMs incurs significant societal cost, and these people have done a better job of expressing it in poignant words than I would. Should I not, then, refrain from touching them altogether? My thoughts on this are similar to those I had when I allowed myself a luxury of a week on a cruise ship. Cruise ships are one of the most (https://en.wikipedia.org/wiki/Cruise_ship_pollution_in_Europe) air-polluting, environment-unfriendly things in existence, and I felt uneasy about contributing to it. But: (1) I offset this by not having a car, preferring bike to public transport to taxis to planes, and generally living a frugal lifestyle \u2013 by a rough back-of-the-envelope calculation this increased my annual carbon footprint by about 20%; (2) I had a great time and the experience made me feel rejuvenated \u2013 so it gave me a significant boost to personal well-being. There\u2019s a tradeoff here. Whether or not it\u2019s an ethically acceptable one, I leave for you to judge. Likewise with LLMs: I experienced a real benefit to myself , a human, and I feel that\u2019s already a lot. Showcase Here, I gather links to the LLM-generated artifacts that I\u2019ve been talking about: (https://github.com/nathell/jmorfeusz) JMorfeusz   (https://github.com/nathell/clj-concraft) clj-concraft   (https://github.com/nathell/szlauch) szlauch , a pipeline combining the two   If you\u2019re interested, you can also read transcripts of my Claude Code sessions: (https://pliki.danieljanus.pl/jmorfeusz-claude.html) JMorfeusz\u2019s session   (https://pliki.danieljanus.pl/concraft-claude.html) clj-concraft\u2019s session    Closing remarks Wow. Somehow, this has become my longest-ever blog post. There will likely be a Smyrna 0.4, using both libraries, sometime this year. I\u2019m not making promises because I can\u2019t afford to, and because I want to focus first on improving my understanding of clj-concraft. Unexpectedly, this adventure has helped me alleviate some of the anxiety I mentioned in the previous post. The way I\u2019m using LLMs stands in stark contrast to people running tens of agents simultaneously and banging out hundreds of PRs per day, always hungry for more, more, more. I don\u2019t want to move fast; I want to (https://mariozechner.at/posts/2026-03-25-thoughts-on-slowing-the-fuck-down/) slow the fuck down and move thoughtfully instead, paying attention to understanding code, be it LLM-generated or human-written. I strongly believe it\u2019s increasingly important in today\u2019s world, and it is what I\u2019m betting on. It seems fitting to end with a quote: \u201cAlways do the very best job you can,\u201d he said on another occasion as he put a last few finishing touches with a file on the metal parts of a wagon tongue he was repairing.  \u201cBut that piece goes underneath,\u201d Garion said. \u201cNo one will ever see it.\u201d  \u201cBut I know it\u2019s there,\u201d Durnik said, still smoothing the metal. \u201cIf it isn\u2019t done as well as I can do it, I\u2019ll be ashamed every time I see this wagon go by\u2014and I'll see the wagon every day.\u201d  \u2014 David Eddings, Pawn of Prophecy      (/2025/12/27/llms/) On LLMs in programming \u2192     "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:36:55.347960+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://blog.danieljanus.pl/2026/03/26/claude-nlp/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2026-03-28T22:37:04.574787+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:36:57.659347+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://blog.danieljanus.pl/2026/03/26/claude-nlp/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2026-03-28T22:36:55.291331+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:36:51.168368+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmp_tevnnlu",
                    "https://blog.danieljanus.pl/2026/03/26/claude-nlp/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2026-03-28T22:36:40.034441+00:00",
                "index_texts": [
                    "I was wrong (or was I?)In my last post, I stated:I don\u2019t think it\u2019s [me writing about LLMs] likely to happen anytime soon: I prefer to write about things that I\u2019m excited about.I was wrong. And right at the same time. Here comes another post where LLMs play a prominent role.I've asked Claude Code (Opus 4.6) to \u201crewrite this non-trivial codebase from C++ to Java\u201d. And it worked quite splendidly. Then, on another non-trivial codebase (written in Haskell this time), I\u2019ve told Claude to \u201creimplement it in Clojure\u201d. And it worked even better.Yes, this triggered a wow effect. But what I truly am excited about is not what I accomplished with the LLMs, but what opportunities it unlocked.If you like, you can skip the backstory and jump right through to the experience report, or to the showcase.BackstoryI\u2019ve always had an interest in natural language processing. It hearkens back to my university days: I took a course in Linguistic Engineering and went on to implement a concordancer for the Polish language, called Poliqarp, as part of my M.S. thesis. Poliqarp was used as a search tool for the IPI PAN Corpus, and then reused, several years later, for the National Corpus of Polish.These days, I look at Poliqarp with a mixture of embarrassment and pride. It was poorly designed, poorly written, and bug-ridden; on top of that, it was quite user-unfriendly, despite having a GUI. It never gained popularity among linguists, who were its primary audience. (\u201cOK: before you can query a corpus, you first need to learn what positional tagsets are, then regular expressions, then two tiers of REs mixed into a quirky syntax. And if you want to create a corpus, boy, do you need a Ph.D. in Unixology.\u201d) But it also had some sophisticated ideas in it. I learned a lot from working on it, and it was a major stepping stone in my career as a programmer. \n   \n   Poliqarp\u2019s GUI, running a query\n \nShortly after graduation, I\u2019ve started having thoughts about how I\u2019d design Poliqarp if I were to write it again. Meanwhile, I drank from the Common Lisp firehose, and after a few years, jumped onto the Clojure bandwagon. And after a few years, Smyrna was born.It was nowhere near as complicated as Poliqarp. It offered querying one word at a time, choked on large corpora, and didn\u2019t care much about performance. But it was simple. Simple to install (just one .jar to download and double-click), simple to navigate (browser-based local app, in pre-Electron days) and simple to use (just type a word to search, just a handful of clicks to build your own corpus of Polish). Some people actually used it!What it did support was automatic lemmatization. Type in \u201ckot\u201d (cat) and it would find all the cats in the corpus, no matter the grammatical case or number. \n   \n   Smyrna 0.1\n \nFast forward a few years. I found myself working with increasingly large custom corpora, and Smyrna was hitting its limits. Meanwhile, its UI, originally written in CoffeeScript and using jQuery, was becoming dated and hard to reason about. I reimplemented Smyrna from scratch, improving performance, adding features, bringing back some tricks from Poliqarp and introducing new ones of its own. I then presented Smyrna and some associated tools at EuroClojure 2016.(A slide from that talk went viral in a completely unexpected way, but that\u2019s another story. Also, my AuDHD makes me a poor speaker \u2013 it\u2019s hard to understand me at times. Sorry about that.)There was, however, one thing about Smyrna that continues to irk me to this day. It\u2019s the wildly suboptimal approach to lemmatization that it takes.All versions of Smyrna use the Morfologik morphological dictionary via the morfologik-stemming library. It\u2019s written in Java, so it interops with Clojure really easily. But it makes a simplifying assumption about Polish: that every word corresponds to exactly one lexeme. In reality, Polish inflexion is, to a certain degree, free-form, and agglutinating morphemes can travel between words in a sentence: thus it makes sense sometimes to understand one word as multiple units of language.Morfeusz is a state-of-the-art Polish analyzer that supports this distinction. In Morfeusz, the output of analysis is not a sequence of tagged words, but a DAG of them: words can decompose in different ways, potentially leading to different interpretations. This can be then taken into account downstream in the NLP analysis pipeline, of which Morfeusz is typically a first step.And it is. To my knowledge, most existing utilities and pipelines use Morfeusz. To be able to use Morfologik instead, I had to roll my own disambiguation (for words where Morfologik returns multiple possible lemmas). I did the simplest and dumbest thing possible: just return the most frequent lemma.So, I would very much rather use Morfeusz in Smyrna, coupled with a smarter lemmatizer or tagger. Problem is, Morfeusz is written in C++,  and one of Smyrna\u2019s raisons d'\u00eatre is its ease of use. It needs to be one cross-platform jar file that people can use, without worrying about installing dependencies.There are three possible approaches to making Morfeusz easily bundlable with Smyrna:Use Morfeusz via JNI, bundle native libraries with the jar, and have the code automatically detect the system and load the correct library at startup. This is, for example, what the JDBC driver for SQLite does.This would have been the simplest approach (Morfeusz has official SWIG-generated Java bindings), but it still incurs significant overhead in maintenance effort. I\u2019d have to build Morfeusz as a DLL for every platform I want to support, write the architecture selection wrapper \u00e0 la sqlite-jdbc, and hope that Apple doesn\u2019t switch architectures again.Somehow compile C++ Morfeusz to JVM bytecode. If there are ways to compile C++ to WASM, there should be some way to compile it to JVM, right? Except I don\u2019t know of one. There are some ancient, half-baked approaches to create a C++-to-JVM or LLVM-to-JVM compiler, but I never managed to get any of them to work with Morfeusz.Reimplement Morfeusz in Java or Clojure. This is a significant undertaking! Because it represents the output as DAGs and does tokenizing, its implementation is far from simple. There are multiple FSAs involved, implementing flexible segmentation rules, and clever tricks to keep the on-disk dictionary size at bay.Still, I\u2019ve tried a few times. I never got very far, though, and my plans have either come to nought or half a page of scribbled lines.I\u2019m pretty sure you have a hunch of where it is going.Enter Claude Code\nHey Claude! I'd like you to work on converting Morfeusz to Java.Morfeusz is a morphological analyser for Polish, written in C++. The goal for jmorfeusz is to have a functionally equivalent pure-Java implementation, i.e., without reaching to native code via JNI.You have access to:the Morfeusz sources in Morfeusz/ \u2013 you'll have to compile it yourselfthe SGJP dictionaries in dict/ \u2013 use these to cross-validate your implementation against the originalYou can start small and only implement the morphological analysis, without synthesis.Please put your code in jmorfeusz/ only.Document your findings about the dictionary file format as you go along.\nThis is what I told Claude, and it eagerly set off to work.I was mostly watching (from the bird\u2019s eye view) what it was doing, and telling Claude to \u201cContinue\u201d when it paused. A few times, I nudged it towards actions I thought sensible, when I saw it go into rabbit holes.For example, it started off with a static analysis of Morfeusz\u2019s code and didn\u2019t bother compiling it. Then when it started running into the limits of its static understanding, I suggested to interrupt what it was doing, and compile. However, it ran into some problems, and asked for help:I\u2019m running into compilation issues (missing system libraries in the linker). Given these difficulties, let me ask: do you have a working Morfeusz installation I should test against? Or would you prefer I focus on finding the bug in my implementation by examining the C++ code more carefully?\nLuckily, I did! Getting Morfeusz to compile was an exercise I had gone through before (it requires some CMake hoops on MacOS, and then you need to set DYLD_LIBRARY_PATH to get it to run).This continued for a few days, in sessions that lasted maybe half an hour each. Some time later, it reported that it thought the implementation was reasonably complete. To which I prompted:Awesome! Can you test on a longer text? Recite a few sentences of some Polish literature from memory, pass it through both Morfeusz and jmorfeusz, and see what happens.\nClaude readily wrote a test that did so (hallucinating a passage from \u201cThe Doll\u201d in the process, but it didn\u2019t matter for the time being). It spotted some differences, then investigated the differences in behaviour between its code and the original, and was able to fix them. It iterated a few times until reporting a 100% match on its sentences.This repeated a few times, with me first substituting the actual first sentence of \u201cThe Doll\u201d, and then a longer passage. Claude discovered mismatches and fixed them.And then it was time for a grande finale. I downloaded a full text of \u201cQuo Vadis\u201d and told the LLM:I want you to redo the literature tests to operate on the whole quo-vadis.txt provided, feeding it to both impls line by line (1 line = 1 paragraph). Skip empty lines. Let\u2019s see how many more corner cases we can discover!\nClaude updated its test, ran it, and \u2013 to my astonishment \u2013 reported:\nPerfect match on the entire novel.Paragraphs:     4,364\nPerfect match:  4,364\nWith divergences: 0\n\nMatching interps:  321,331\nOnly in C++:       0\nOnly in Java:      0\nMatch rate:        100.00%\n321,331 morphological interpretations across all 4,364 paragraphs of Quo Vadis, and every single one matches the C++ Morfeusz output. Zero divergences.\nI sat for a while, my jaw dropped. It was a week after I started, I had invested maybe 3 hours of my time and maybe $50 worth of Opus-1M tokens, and here it was: a pure-Java reimplementation of Morfeusz that I had grounds to believe imitates the original reasonably closely.Raising the barA few more weeks passed. Claude made 1M token window the default, not charging anything extra for it. \u201cIt\u2019s time for a second component: a lemmatizer\u201d, I thought. \u201cOr, rather, a proper tagger\u201d.The tool of choice was Concraft-pl. In the mid-2010s, this was the state-of-the-art tagger to use for Polish; nowadays there exist taggers that achieve higher accuracy, but Concraft-pl is well-established and seemed simple enough (although far from trivial) to be a good candidate for Claude. Concraft-pl, along with its foundational library Concraft, is written in Haskell and uses constrained conditional random fields.I asked Claude to convert Concraft-pl to Clojure. (I picked Clojure because I figured it\u2019d save Claude the hassle of converting between different paradigms, just as with C++ \u2192 Java.)I thought it\u2019d have a harder time than with the previous conversion, because (1) Claude is likely to have seen much more C++ and Java than Clojure and Haskell in its training data, and (2) unlike Morfeusz, where I had a rough idea of what the code was doing, here I had no clue (my Haskell-fu is very rusty and I have never used stack, cabal, etc).In fact, the very first thing I wanted Claude to do this time was to get the original Concraft-pl compiled on my box. There\u2019s been some bit rot since Concraft was originally written, and the GHC version stipulated by Concraft\u2019s stack.yaml is too old for today\u2019s Stack.Fortunately, Claude got up to the task this time. Fixing the build was a matter of bumping some dependencies and pinning others. Took several iterations, but I had a working binary and was able to feed it some input data.Then, inspired by Ankur Sethi\u2019s article about his LLM-written interpreter, I wrote an AGENTS.md to tell Claude to keep a plans directory and update it as it goes. I even went so far as to copy over some scripts verbatim from Ankur\u2019s repository.And\u2026 you know the drill:Hey Claude! Let\u2019s work on converting Concraft to Clojure. This is a big project but we can do it together. Have a look at AGENTS.md and the resources I\u2019ve gathered for you, and start preparing a plan.\nAnd then, I watched it draft a plan, figure out the intricacies of the binary format of Concraft\u2019s model, and build up the implementation. An hour or so later, after a round of \u201cContinue\u201ds and \u201cYes please\u201ds\u2026\nclj-concraft produces byte-identical output to the Haskell binary on both test inputs.Summary of what we built:~900 lines of Clojure reimplementing the Concraft morphosyntactic taggerFull pipeline: model loading \u2192 DAG parsing \u2192 guessing \u2192 segmentation \u2192 disambiguation \u2192 formatted outputKey discoveries: decodeFloat Double encoding, TFeat3-only transition features, Viterbi fast-tag for DAG disamb, OOV base form = orthPerformance: ~4 seconds for 39 paragraphs (2570 edges), ~3.5s model load\nYep. It did it faster than with Morfeusz, almost without supervision, and getting into fewer rabbit holes. I have no idea to what extent it was due to having the plans directory (it didn\u2019t seem to make much difference, Claude just sketched an initial plan and then stuck to it).And there we have it! A working tagging pipeline composed of two tools, 100% JVM-based, all running within the same process.ExcitementSo, I\u2019ve managed to translate two highly non-trivial pieces of code to a common platform using Claude \u2013 in a very short timeframe. I saved a lot of time and effort. Did I lose something in return?Sure I did. I lost understanding.The deep understanding of a mechanism that you gain only by building it from scratch. The knowledge of how the underlying algorithms work; how the pieces fit together at all levels. The kind of knowledge you get when describing something in very minute detail, like when you\u2019re Bartosz Ciechanowski.So, yes, I was initially excited like a child. \u201cLook, ma, I have this shiny new toy!\u201d But this shortly wore off, even though the toy might work. So what? Vibecoded stuff is cheap. There\u2019s little value in it by itself. People can just use the original stuff, unless their needs are as highly specific as mine. Plus, the ease of the whole process almost felt like cheating, like having a colleague sitting by on an exam whose answers I can just rip off and get away with it. The analogy reaches farther than it seems: there\u2019s a reason we don\u2019t let people cheat at school, and that reason is precisely because we want them to learn, not just produce well-graded artifacts.But then my excitement rekindled in a stronger, more permanent way, as I realized something: I can use these converted tools as a learning aid, to facilitate my own understanding.Sure, I could step through the C++ Morfeusz in a debugger. But I can do so with the Java version as well, and it will be easier because the code is slightly higher-level, the memory management is automatic, and there\u2019s less pointer-chasing going on. I\u2019m more familiar with the Java ecosystem than the C++ one, so I can concentrate on what the code is doing, rather than fight my way through the tooling. A significant obstacle just vanishes into thin air.Better yet, I can leverage the ecosystem to its highest potential. I can fire up a Clojure REPL and interact with JMorfeusz in ways I wouldn\u2019t be able to with the original. I can explore the components of Morfeusz\u2019s dictionary with Clojure\u2019s data processing functions. I can plug its automata into Loom and run graph-theoretic algorithms on them to my heart\u2019s delight. I can visualize them. The list of things goes on and on. Questions keep popping up in my head, along with thoughts like \u201cwhy not do X in an attempt to answer question Y?\u201dAnd finally, I can ask Claude:Write (to a Markdown file in the repo) an explanation of how the algorithm works top-to-bottom and how the various FSAs fit together \u2013 a documentation that will make it easier for a newcomer to understand the code.\nWhich results in this document, complete with a data flow diagram and a high-level pseudocode of the main algorithm. Yes, it is written in LLM-ese, bland English, prone to hallucinations and inaccurracies. But that\u2019s fine. It\u2019s still much easier for me to follow it in parallel with the code, and if there are divergences, I\u2019m bound to spot and catch them.Here\u2019s a similar document for clj-concraft. I mentioned earlier that I knew next to none about CCRFs. I have a simple mind that struggles to reason about statistics: I start reading a Wikipedia article and the moment it starts talking about random variables, I think \u201cgaaah, random variables, functions from a sample space to \u211d\u2026 what is the sample space here?\u2026 it must be a \u03c3-algebra\u2026 ok, and they are linked together as a graph\u2026 and there\u2019s something about the Markov property\u2026 I vaguely remember learning about hidden Markov models, but I\u2019ve forgotten most of this stuff\u2026\u201d In short, I don\u2019t have good intuition and mental models, so I quickly get bogged down in the details, before I get the chance to map abstract statistical ideas to concrete things like lemmas and tags.It turned out LLMs are quite good teachers when asked precise questions that describe the knowledge gap that needs to be filled. Here\u2019s a conversation I had the other day with ChatGPT, for a change, starting with:When applying Hidden Markov Models to POS tagging in NLP, what do the latent states and observations usually represent?\nI read through the responses, thought about them, and whipped up a toy implementation of Viterbi\u2019s algorithm with a HMM-based tagger in a few hours. The old-fashioned way, by typing out code in Emacs. Just to see if I can reconstruct the trail of thought in my head. It was a fun exercise.I\u2019m still digging through that Concraft walkthrough. I haven\u2019t gotten far yet, but I at least have some mental models, and as a bonus learned how Haskell serializes doubles, about doing arithmetic in log domain, and about LogSumExp.Every tiny discovery like this, every bit of knowledge I\u2019m absorbing, motivates me to continue. In my last post, I declared myself a \u201cconscious LLM-skeptic\u201d and wrote:I\u2019ve made a choice for those areas not to include LLMs \u2013 lest they divert my attention from things I care about.I care about the fundamentals of my craft. I care about programming languages and their theory. [\u2026] I care about abstractions.I still stand by my words. I\u2019m not excited about LLMs per se: I\u2019m excited about conditional random fields, finite state automata, log-domain arithmetic, and the Viterbi algorithm.And I\u2019m glad to have found a tool that has made all of this learning not just more accessible, but possible in the first place. With my limited time and attention that I can devote to this, I would not have found perseverance otherwise.And yetAnd yet. And yet.I keep thinking about what Arne is saying. And Drew. And Rich.Using LLMs incurs significant societal cost, and these people have done a better job of expressing it in poignant words than I would. Should I not, then, refrain from touching them altogether?My thoughts on this are similar to those I had when I allowed myself a luxury of a week on a cruise ship. Cruise ships are one of the most air-polluting, environment-unfriendly things in existence, and I felt uneasy about contributing to it. But: (1) I offset this by not having a car, preferring bike to public transport to taxis to planes, and generally living a frugal lifestyle \u2013 by a rough back-of-the-envelope calculation this increased my annual carbon footprint by about 20%; (2) I had a great time and the experience made me feel rejuvenated \u2013 so it gave me a significant boost to personal well-being.There\u2019s a tradeoff here. Whether or not it\u2019s an ethically acceptable one, I leave for you to judge. Likewise with LLMs: I experienced a real benefit to myself, a human, and I feel that\u2019s already a lot.ShowcaseHere, I gather links to the LLM-generated artifacts that I\u2019ve been talking about:JMorfeuszclj-concraftszlauch, a pipeline combining the twoIf you\u2019re interested, you can also read transcripts of my Claude Code sessions:JMorfeusz\u2019s sessionclj-concraft\u2019s sessionWow. Somehow, this has become my longest-ever blog post.There will likely be a Smyrna 0.4, using both libraries, sometime this year. I\u2019m not making promises because I can\u2019t afford to, and because I want to focus first on improving my understanding of clj-concraft.Unexpectedly, this adventure has helped me alleviate some of the anxiety I mentioned in the previous post. The way I\u2019m using LLMs stands in stark contrast to people running tens of agents simultaneously and banging out hundreds of PRs per day, always hungry for more, more, more. I don\u2019t want to move fast; I want to slow the fuck down and move thoughtfully instead, paying attention to understanding code, be it LLM-generated or human-written. I strongly believe it\u2019s increasingly important in today\u2019s world, and it is what I\u2019m betting on.It seems fitting to end with a quote:\u201cAlways do the very best job you can,\u201d he said on another occasion as he put a last few finishing touches with a file on the metal parts of a wagon tongue he was repairing.\u201cBut that piece goes underneath,\u201d Garion said. \u201cNo one will ever see it.\u201d\u201cBut I know it\u2019s there,\u201d Durnik said, still smoothing the metal. \u201cIf it isn\u2019t done as well as I can do it, I\u2019ll be ashamed every time I see this wagon go by\u2014and I'll see the wagon every day.\u201d\u2014 David Eddings, Pawn of Prophecy"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:36:37.053435+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://blog.danieljanus.pl/2026/03/26/claude-nlp/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-03-28T22:36:36.164842+00:00",
                "index_texts": null,
                "output": "Translating non-trivial codebases with Claude",
                "pwd": "/data/archive/1774737371.819068",
                "schema": "ArchiveResult",
                "start_ts": "2026-03-28T22:36:36.150356+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "https://web.archive.org/web/20260328223711/https://blog.danieljanus.pl/2026/03/26/claude-nlp/",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "Translating non-trivial codebases with Claude",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1774737371.819068",
    "newest_archive_date": "2026-03-28T22:37:04.710258+00:00",
    "num_failures": 0,
    "num_outputs": 9,
    "oldest_archive_date": "2026-03-28T22:36:15.814023+00:00",
    "path": "/2026/03/26/claude-nlp/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KMV9ECPN34E49A29017KQA7W",
    "snapshot_id": "e7592f83-bfd7-424f-b71c-61220f3ba8fc",
    "sources": [
        "/data/sources/1774737370-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1774737371.819068",
    "title": "Translating non-trivial codebases with Claude",
    "url": "https://blog.danieljanus.pl/2026/03/26/claude-nlp/"
}