{
    "archive_path": "archive/1754256232.237508",
    "base_url": "lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate",
    "basename": "",
    "bookmarked_date": "2025-08-03 21:23",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=lexi-lambda.github.io",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "lexi-lambda.github.io",
    "downloaded_at": "2025-08-03T21:23:54.770216+00:00",
    "downloaded_datestr": "2025-08-03 21:23",
    "extension": "",
    "hash": "8CE4V1CA0X8FWM2A5M73",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-08-03T21:25:06.561621+00:00",
                "index_texts": null,
                "output": "https://web.archive.org/web/20250803212449/https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:24:45.068764+00:00",
                "status": "succeeded"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2025-08-03T21:24:15.692924+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:24:06.266905+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=lexi-lambda.github.io"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-08-03T21:23:58.433781+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:23:55.462463+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-08-03T21:23:58.799319+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:23:58.532054+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2025-08-03T21:24:36.543995+00:00",
                "index_texts": [
                    "Parse, don\u2019t validate (https://fonts.googleapis.com/css?family=Merriweather+Sans:400,300,300italic,400italic,700,700italic,800,800italic|Merriweather:400,300,300italic,400italic,700,700italic,900,900italic|Fira+Code:300,400,500,600,700) (/css/application.min.css) (/css/pygments.min.css) (Atom Feed) (/feeds/all.atom.xml) (RSS Feed) (/feeds/all.rss.xml)  (/) Alexis King    (/) Home  (/about.html) About Me     Parse, don\u2019t validate 2019-11-05 \u29bf (/tags/functional-programming.html) functional programming , (/tags/haskell.html) haskell , (/tags/types.html) types   Historically, I\u2019ve struggled to find a concise, simple way to explain what it means to practice type-driven design. Too often, when someone asks me \u201cHow did you come up with this approach?\u201d I find I can\u2019t give them a satisfying answer. I know it didn\u2019t just come to me in a vision\u2014I have an iterative design process that doesn\u2019t require plucking the \u201cright\u201d approach out of thin air\u2014yet I haven\u2019t been very successful in communicating that process to others. However, about a month ago, (https://twitter.com/lexi_lambda/status/1182242561655746560) I was reflecting on Twitter about the differences I experienced parsing JSON in statically- and dynamically-typed languages, and finally, I realized what I was looking for. Now I have a single, snappy slogan that encapsulates what type-driven design means to me, and better yet, it\u2019s only three words long: Parse, don\u2019t validate.   The essence of type-driven design Alright, I\u2019ll confess: unless you already know what type-driven design is, my catchy slogan probably doesn\u2019t mean all that much to you. Fortunately, that\u2019s what the remainder of this blog post is for. I\u2019m going to explain precisely what I mean in gory detail\u2014but first, we need to practice a little wishful thinking.  The realm of possibility One of the wonderful things about static type systems is that they can make it possible, and sometimes even easy, to answer questions like \u201cis it possible to write this function?\u201d For an extreme example, consider the following Haskell type signature: foo  ::  Integer  ->  Void   Is it possible to implement foo ? Trivially, the answer is no , as Void is a type that contains no values, so it\u2019s impossible for any function to produce a value of type Void .1  That example is pretty boring, but the question gets much more interesting if we choose a more realistic example: head  ::  [ a ]  ->  a   This function returns the first element from a list. Is it possible to implement? It certainly doesn\u2019t sound like it does anything very complicated, but if we attempt to implement it, the compiler won\u2019t be satisfied: head  ::  [ a ]  ->  a head  ( x : _ )  =  x   warning: [-Wincomplete-patterns]\n    Pattern match(es) are non-exhaustive\n    In an equation for \u2018head\u2019: Patterns not matched: []   This message is helpfully pointing out that our function is partial , which is to say it is not defined for all possible inputs. Specifically, it is not defined when the input is [] , the empty list. This makes sense, as it isn\u2019t possible to return the first element of a list if the list is empty\u2014there\u2019s no element to return! So, remarkably, we learn this function isn\u2019t possible to implement, either.  Turning partial functions total To someone coming from a dynamically-typed background, this might seem perplexing. If we have a list, we might very well want to get the first element in it. And indeed, the operation of \u201cgetting the first element of a list\u201d isn\u2019t impossible in Haskell, it just requires a little extra ceremony. There are two different ways to fix the head function, and we\u2019ll start with the simplest one.  Managing expectations As established, head is partial because there is no element to return if the list is empty: we\u2019ve made a promise we cannot possibly fulfill. Fortunately, there\u2019s an easy solution to that dilemma: we can weaken our promise. Since we cannot guarantee the caller an element of the list, we\u2019ll have to practice a little expectation management: we\u2019ll do our best return an element if we can, but we reserve the right to return nothing at all. In Haskell, we express this possibility using the Maybe type: head  ::  [ a ]  ->  Maybe  a   This buys us the freedom we need to implement head \u2014it allows us to return Nothing when we discover we can\u2019t produce a value of type a after all: head  ::  [ a ]  ->  Maybe  a head  ( x : _ )  =  Just  x head  []  =  Nothing   Problem solved, right? For the moment, yes\u2026 but this solution has a hidden cost. Returning Maybe is undoubtably convenient when we\u2019re implementing head . However, it becomes significantly less convenient when we want to actually use it! Since head always has the potential to return Nothing , the burden falls upon its callers to handle that possibility, and sometimes that passing of the buck can be incredibly frustrating. To see why, consider the following code: getConfigurationDirectories  ::  IO  [ FilePath ] getConfigurationDirectories  =  do  configDirsString  <-  getEnv  \"CONFIG_DIRS\"  let  configDirsList  =  split  ','  configDirsString  when  ( null  configDirsList )  $  throwIO  $  userError  \"CONFIG_DIRS cannot be empty\"  pure  configDirsList main  ::  IO  () main  =  do  configDirs  <-  getConfigurationDirectories  case  head  configDirs  of  Just  cacheDir  ->  initializeCache  cacheDir  Nothing  ->  error  \"should never happen; already checked configDirs is non-empty\"   When getConfigurationDirectories retrieves a list of file paths from the environment, it proactively checks that the list is non-empty. However, when we use head in main to get the first element of the list, the Maybe FilePath result still requires us to handle a Nothing case that we know will never happen! This is terribly bad for several reasons: First, it\u2019s just annoying. We already checked that the list is non-empty, why do we have to clutter our code with another redundant check?  Second, it has a potential performance cost. Although the cost of the redundant check is trivial in this particular example, one could imagine a more complex scenario where the redundant checks could add up, such as if they were happening in a tight loop.  Finally, and worst of all, this code is a bug waiting to happen! What if getConfigurationDirectories were modified to stop checking that the list is empty, intentionally or unintentionally? The programmer might not remember to update main , and suddenly the \u201cimpossible\u201d error becomes not only possible, but probable.   The need for this redundant check has essentially forced us to punch a hole in our type system. If we could statically prove the Nothing case impossible, then a modification to getConfigurationDirectories that stopped checking if the list was empty would invalidate the proof and trigger a compile-time failure. However, as-written, we\u2019re forced to rely on a test suite or manual inspection to catch the bug.  Paying it forward Clearly, our modified version of head leaves some things to be desired. Somehow, we\u2019d like it to be smarter: if we already checked that the list was non-empty, head should unconditionally return the first element without forcing us to handle the case we know is impossible. How can we do that? Let\u2019s look at the original (partial) type signature for head again: head  ::  [ a ]  ->  a   The previous section illustrated that we can turn that partial type signature into a total one by weakening the promise made in the return type. However, since we don\u2019t want to do that, there\u2019s only one thing left that can be changed: the argument type (in this case, [a] ). Instead of weakening the return type, we can strengthen the argument type, eliminating the possibility of head ever being called on an empty list in the first place. To do this, we need a type that represents non-empty lists. Fortunately, the existing NonEmpty type from Data.List.NonEmpty is exactly that. It has the following definition: data  NonEmpty  a  =  a  :|  [ a ]   Note that NonEmpty a is really just a tuple of an a and an ordinary, possibly-empty [a] . This conveniently models a non-empty list by storing the first element of the list separately from the list\u2019s tail: even if the [a] component is [] , the a component must always be present. This makes head completely trivial to implement:2   head  ::  NonEmpty  a  ->  a head  ( x :| _ )  =  x   Unlike before, GHC accepts this definition without complaint\u2014this definition is total , not partial. We can update our program to use the new implementation: getConfigurationDirectories  ::  IO  ( NonEmpty  FilePath ) getConfigurationDirectories  =  do  configDirsString  <-  getEnv  \"CONFIG_DIRS\"  let  configDirsList  =  split  ','  configDirsString  case  nonEmpty  configDirsList  of  Just  nonEmptyConfigDirsList  ->  pure  nonEmptyConfigDirsList  Nothing  ->  throwIO  $  userError  \"CONFIG_DIRS cannot be empty\" main  ::  IO  () main  =  do  configDirs  <-  getConfigurationDirectories  initializeCache  ( head  configDirs )   Note that the redundant check in main is now completely gone! Instead, we perform the check exactly once, in getConfigurationDirectories . It constructs a NonEmpty a from a [a] using the nonEmpty function from Data.List.NonEmpty , which has the following type: nonEmpty  ::  [ a ]  ->  Maybe  ( NonEmpty  a )   The Maybe is still there, but this time, we handle the Nothing case very early in our program: right in the same place we were already doing the input validation. Once that check has passed, we now have a NonEmpty FilePath value, which preserves (in the type system!) the knowledge that the list really is non-empty. Put another way, you can think of a value of type NonEmpty a as being like a value of type [a] , plus a proof that the list is non-empty. By strengthening the type of the argument to head instead of weakening the type of its result, we\u2019ve completely eliminated all the problems from the previous section: The code has no redundant checks, so there can\u2019t be any performance overhead.  Furthermore, if getConfigurationDirectories changes to stop checking that the list is non-empty, its return type must change, too. Consequently, main will fail to typecheck, alerting us to the problem before we even run the program!   What\u2019s more, it\u2019s trivial to recover the old behavior of head from the new one by composing head with nonEmpty : head'  ::  [ a ]  ->  Maybe  a head'  =  fmap  head  .  nonEmpty   Note that the inverse is not true: there is no way to obtain the new version of head from the old one. All in all, the second approach is superior on all axes.  The power of parsing You may be wondering what the above example has to do with the title of this blog post. After all, we only examined two different ways to validate that a list was non-empty\u2014no parsing in sight. That interpretation isn\u2019t wrong, but I\u2019d like to propose another perspective: in my mind, the difference between validation and parsing lies almost entirely in how information is preserved. Consider the following pair of functions: validateNonEmpty  ::  [ a ]  ->  IO  () validateNonEmpty  ( _ : _ )  =  pure  () validateNonEmpty  []  =  throwIO  $  userError  \"list cannot be empty\" parseNonEmpty  ::  [ a ]  ->  IO  ( NonEmpty  a ) parseNonEmpty  ( x : xs )  =  pure  ( x :| xs ) parseNonEmpty  []  =  throwIO  $  userError  \"list cannot be empty\"   These two functions are nearly identical: they check if the provided list is empty, and if it is, they abort the program with an error message. The difference lies entirely in the return type: validateNonEmpty always returns () , the type that contains no information, but parseNonEmpty returns NonEmpty a , a refinement of the input type that preserves the knowledge gained in the type system. Both of these functions check the same thing, but parseNonEmpty gives the caller access to the information it learned, while validateNonEmpty just throws it away. These two functions elegantly illustrate two different perspectives on the role of a static type system: validateNonEmpty obeys the typechecker well enough, but only parseNonEmpty takes full advantage of it. If you see why parseNonEmpty is preferable, you understand what I mean by the mantra \u201cparse, don\u2019t validate.\u201d Still, perhaps you are skeptical of parseNonEmpty \u2019s name. Is it really parsing anything, or is it merely validating its input and returning a result? While the precise definition of what it means to parse or validate something is debatable, I believe parseNonEmpty is a bona-fide parser (albeit a particularly simple one). Consider: what is a parser? Really, a parser is just a function that consumes less-structured input and produces more-structured output. By its very nature, a parser is a partial function\u2014some values in the domain do not correspond to any value in the range\u2014so all parsers must have some notion of failure. Often, the input to a parser is text, but this is by no means a requirement, and parseNonEmpty is a perfectly cromulent parser: it parses lists into non-empty lists, signaling failure by terminating the program with an error message. Under this flexible definition, parsers are an incredibly powerful tool: they allow discharging checks on input up-front, right on the boundary between a program and the outside world, and once those checks have been performed, they never need to be checked again! Haskellers are well-aware of this power, and they use many different types of parsers on a regular basis: The (https://hackage.haskell.org/package/aeson) aeson library provides a Parser type that can be used to parse JSON data into domain types.  Likewise, (https://hackage.haskell.org/package/optparse-applicative) optparse-applicative provides a set of parser combinators for parsing command-line arguments.  Database libraries like (https://hackage.haskell.org/package/persistent) persistent and (https://hackage.haskell.org/package/postgresql-simple) postgresql-simple have a mechanism for parsing values held in an external data store.  The (https://hackage.haskell.org/package/servant) servant ecosystem is built around parsing Haskell datatypes from path components, query parameters, HTTP headers, and more.   The common theme between all these libraries is that they sit on the boundary between your Haskell application and the external world. That world doesn\u2019t speak in product and sum types, but in streams of bytes, so there\u2019s no getting around a need to do some parsing. Doing that parsing up front, before acting on the data, can go a long way toward avoiding many classes of bugs, some of which might even be security vulnerabilities. One drawback to this approach of parsing everything up front is that it sometimes requires values be parsed long before they are actually used. In a dynamically-typed language, this can make keeping the parsing and processing logic in sync a little tricky without extensive test coverage, much of which can be laborious to maintain. However, with a static type system, the problem becomes marvelously simple, as demonstrated by the NonEmpty example above: if the parsing and processing logic go out of sync, the program will fail to even compile.  The danger of validation Hopefully, by this point, you are at least somewhat sold on the idea that parsing is preferable to validation, but you may have lingering doubts. Is validation really so bad if the type system is going to force you to do the necessary checks eventually anyway? Maybe the error reporting will be a little bit worse, but a bit of redundant checking can\u2019t hurt, right? Unfortunately, it isn\u2019t so simple. Ad-hoc validation leads to a phenomenon that the (http://langsec.org) language-theoretic security field calls shotgun parsing . In the 2016 paper, (http://langsec.org/papers/langsec-cwes-secdev2016.pdf) The Seven Turrets of Babel: A Taxonomy of LangSec Errors and How to Expunge Them , its authors provide the following definition: Shotgun parsing is a programming antipattern whereby parsing and input-validating code is mixed with and spread across processing code\u2014throwing a cloud of checks at the input, and hoping, without any systematic justification, that one or another would catch all the \u201cbad\u201d cases.  They go on to explain the problems inherent to such validation techniques: Shotgun parsing necessarily deprives the program of the ability to reject invalid input instead of processing it. Late-discovered errors in an input stream will result in some portion of invalid input having been processed, with the consequence that program state is difficult to accurately predict.  In other words, a program that does not parse all of its input up front runs the risk of acting upon a valid portion of the input, discovering a different portion is invalid, and suddenly needing to roll back whatever modifications it already executed in order to maintain consistency. Sometimes this is possible\u2014such as rolling back a transaction in an RDBMS\u2014but in general it may not be. It may not be immediately apparent what shotgun parsing has to do with validation\u2014after all, if you do all your validation up front, you mitigate the risk of shotgun parsing. The problem is that validation-based approaches make it extremely difficult or impossible to determine if everything was actually validated up front or if some of those so-called \u201cimpossible\u201d cases might actually happen. The entire program must assume that raising an exception anywhere is not only possible, it\u2019s regularly necessary. Parsing avoids this problem by stratifying the program into two phases\u2014parsing and execution\u2014where failure due to invalid input can only happen in the first phase. The set of remaining failure modes during execution is minimal by comparison, and they can be handled with the tender care they require.  Parsing, not validating, in practice So far, this blog post has been something of a sales pitch. \u201cYou, dear reader, ought to be parsing!\u201d it says, and if I\u2019ve done my job properly, at least some of you are sold. However, even if you understand the \u201cwhat\u201d and the \u201cwhy,\u201d you might not feel especially confident about the \u201chow.\u201d My advice: focus on the datatypes. Suppose you are writing a function that accepts a list of tuples representing key-value pairs, and you suddenly realize you aren\u2019t sure what to do if the list has duplicate keys. One solution would be to write a function that asserts there aren\u2019t any duplicates in the list: checkNoDuplicateKeys  ::  ( MonadError  AppError  m ,  Eq  k )  =>  [( k ,  v )]  ->  m  ()   However, this check is fragile: it\u2019s extremely easy to forget. Because its return value is unused, it can always be omitted, and the code that needs it would still typecheck. A better solution is to choose a data structure that disallows duplicate keys by construction, such as a Map . Adjust your function\u2019s type signature to accept a Map instead of a list of tuples, and implement it as you normally would. Once you\u2019ve done that, the call site of your new function will likely fail to typecheck, since it is still being passed a list of tuples. If the caller was given the value via one of its arguments, or if it received it from the result of some other function, you can continue updating the type from list to Map , all the way up the call chain. Eventually, you will either reach the location the value is created, or you\u2019ll find a place where duplicates actually ought to be allowed. At that point, you can insert a call to a modified version of checkNoDuplicateKeys : checkNoDuplicateKeys  ::  ( MonadError  AppError  m ,  Eq  k )  =>  [( k ,  v )]  ->  m  ( Map  k  v )   Now the check cannot be omitted, since its result is actually necessary for the program to proceed! This hypothetical scenario highlights two simple ideas: Use a data structure that makes illegal states unrepresentable. Model your data using the most precise data structure you reasonably can. If ruling out a particular possibility is too hard using the encoding you are currently using, consider alternate encodings that can express the property you care about more easily. Don\u2019t be afraid to refactor.  Push the burden of proof upward as far as possible, but no further. Get your data into the most precise representation you need as quickly as you can. Ideally, this should happen at the boundary of your system, before any of the data is acted upon.3   If one particular code branch eventually requires a more precise representation of a piece of data, parse the data into the more precise representation as soon as the branch is selected. Use sum types judiciously to allow your datatypes to reflect and adapt to control flow.   In other words, write functions on the data representation you wish you had, not the data representation you are given. The design process then becomes an exercise in bridging the gap, often by working from both ends until they meet somewhere in the middle. Don\u2019t be afraid to iteratively adjust parts of the design as you go, since you may learn something new during the refactoring process! Here are a handful of additional points of advice, arranged in no particular order: Let your datatypes inform your code, don\u2019t let your code control your datatypes. Avoid the temptation to just stick a Bool in a record somewhere because it\u2019s needed by the function you\u2019re currently writing. Don\u2019t be afraid to refactor code to use the right data representation\u2014the type system will ensure you\u2019ve covered all the places that need changing, and it will likely save you a headache later.  Treat functions that return m () with deep suspicion. Sometimes these are genuinely necessary, as they may perform an imperative effect with no meaningful result, but if the primary purpose of that effect is raising an error, it\u2019s likely there\u2019s a better way.  Don\u2019t be afraid to parse data in multiple passes. Avoiding shotgun parsing just means you shouldn\u2019t act on the input data before it\u2019s fully parsed, not that you can\u2019t use some of the input data to decide how to parse other input data. Plenty of useful parsers are context-sensitive.  Avoid denormalized representations of data, especially if it\u2019s mutable. Duplicating the same data in multiple places introduces a trivially representable illegal state: the places getting out of sync. Strive for a single source of truth. Keep denormalized representations of data behind abstraction boundaries. If denormalization is absolutely necessary, use encapsulation to ensure a small, trusted module holds sole responsibility for keeping the representations in sync.    Use abstract datatypes to make validators \u201clook like\u201d parsers. Sometimes, making an illegal state truly unrepresentable is just plain impractical given the tools Haskell provides, such as ensuring an integer is in a particular range. In that case, use an abstract newtype with a smart constructor to \u201cfake\u201d a parser from a validator.   As always, use your best judgement. It probably isn\u2019t worth breaking out (https://hackage.haskell.org/package/singletons) singletons and refactoring your entire application just to get rid of a single error \"impossible\" call somewhere\u2014just make sure to treat those situations like the radioactive substance they are, and handle them with the appropriate care. If all else fails, at least leave a comment to document the invariant for whoever needs to modify the code next.  Recap, reflection, and related reading That\u2019s all, really. Hopefully this blog post proves that taking advantage of the Haskell type system doesn\u2019t require a PhD, and it doesn\u2019t even require using the latest and greatest of GHC\u2019s shiny new language extensions\u2014though they can certainly sometimes help! Sometimes the biggest obstacle to using Haskell to its fullest is simply being aware what options are available, and unfortunately, one downside of Haskell\u2019s small community is a relative dearth of resources that document design patterns and techniques that have become tribal knowledge. None of the ideas in this blog post are new. In fact, the core idea\u2014\u201cwrite total functions\u201d\u2014is conceptually quite simple. Despite that, I find it remarkably challenging to communicate actionable, practicable details about the way I write Haskell code. It\u2019s easy to spend lots of time talking about abstract concepts\u2014many of which are quite valuable!\u2014without communicating anything useful about process . My hope is that this is a small step in that direction. Sadly, I don\u2019t know very many other resources on this particular topic, but I do know of one: I never hesitate to recommend Matt Parson\u2019s fantastic blog post (https://www.parsonsmatt.org/2017/10/11/type_safety_back_and_forth.html) Type Safety Back and Forth . If you want another accessible perspective on these ideas, including another worked example, I\u2019d highly encourage giving it a read. For a significantly more advanced take on many of these ideas, I can also recommend Matt Noonan\u2019s 2018 paper (https://kataskeue.com/gdp.pdf) Ghosts of Departed Proofs , which outlines a handful of techniques for capturing more complex invariants in the type system than I have described here. As a closing note, I want to say that doing the kind of refactoring described in this blog post is not always easy. The examples I\u2019ve given are simple, but real life is often much less straightforward. Even for those experienced in type-driven design, it can be genuinely difficult to capture certain invariants in the type system, so do not consider it a personal failing if you cannot solve something the way you\u2019d like! Consider the principles in this blog post ideals to strive for, not strict requirements to meet. All that matters is to try. Technically, in Haskell, this ignores \u201cbottoms,\u201d constructions that can inhabit any value. These aren\u2019t \u201creal\u201d values (unlike null in some other languages)\u2014they\u2019re things like infinite loops or computations that raise exceptions\u2014and in idiomatic Haskell, we usually try to avoid them, so reasoning that pretends they don\u2019t exist still has value. But don\u2019t take my word for it\u2014I\u2019ll let Danielsson et al. convince you that (https://www.cs.ox.ac.uk/jeremy.gibbons/publications/fast+loose.pdf) Fast and Loose Reasoning is Morally Correct . \u21a9   In fact, Data.List.NonEmpty already provides a head function with this type, but just for the sake of illustration, we\u2019ll reimplement it ourselves. \u21a9   Sometimes it is necessary to perform some kind of authorization before parsing user input to avoid denial of service attacks, but that\u2019s okay: authorization should have a relatively small surface area, and it shouldn\u2019t cause any significant modifications to the state of your system. \u21a9    (/blog/2020/01/19/no-dynamic-type-systems-are-not-inherently-more-open/) \u2190 No, dynamic type systems are not inherently more open   (/blog/2019/10/19/empathy-and-subjective-experience-in-programming-languages/) Empathy and subjective experience in programming languages \u2192      \u00a9 2025, Alexis King Built with (https://docs.racket-lang.org/scribble/index.html) Scribble  , the Racket document preparation system. Feeds are available via (/feeds/all.atom.xml) Atom or (/feeds/all.rss.xml) RSS .    "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:24:36.438099+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2025-08-03T21:24:45.024233+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:24:40.123233+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2025-08-03T21:24:36.364674+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:24:29.668765+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpgioy7vks",
                    "https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2025-08-03T21:24:20.503842+00:00",
                "index_texts": [
                    "Parse, don\u2019t validateHistorically, I\u2019ve struggled to find a concise, simple way to explain what it means to practice type-driven design. Too often, when someone asks me \u201cHow did you come up with this approach?\u201d I find I can\u2019t give them a satisfying answer. I know it didn\u2019t just come to me in a vision\u2014I have an iterative design process that doesn\u2019t require plucking the \u201cright\u201d approach out of thin air\u2014yet I haven\u2019t been very successful in communicating that process to others.However, about a month ago, I was reflecting on Twitter about the differences I experienced parsing JSON in statically- and dynamically-typed languages, and finally, I realized what I was looking for. Now I have a single, snappy slogan that encapsulates what type-driven design means to me, and better yet, it\u2019s only three words long:Parse, don\u2019t validate.The essence of type-driven designAlright, I\u2019ll confess: unless you already know what type-driven design is, my catchy slogan probably doesn\u2019t mean all that much to you. Fortunately, that\u2019s what the remainder of this blog post is for. I\u2019m going to explain precisely what I mean in gory detail\u2014but first, we need to practice a little wishful thinking.The realm of possibilityOne of the wonderful things about static type systems is that they can make it possible, and sometimes even easy, to answer questions like \u201cis it possible to write this function?\u201d For an extreme example, consider the following Haskell type signature:foo :: Integer -> VoidIs it possible to implement foo? Trivially, the answer is no, as Void is a type that contains no values, so it\u2019s impossible for any function to produce a value of type Void.1 That example is pretty boring, but the question gets much more interesting if we choose a more realistic example:head :: [a] -> aThis function returns the first element from a list. Is it possible to implement? It certainly doesn\u2019t sound like it does anything very complicated, but if we attempt to implement it, the compiler won\u2019t be satisfied:head :: [a] -> a\nhead (x:_) = xwarning: [-Wincomplete-patterns]\n    Pattern match(es) are non-exhaustive\n    In an equation for \u2018head\u2019: Patterns not matched: []\nThis message is helpfully pointing out that our function is partial, which is to say it is not defined for all possible inputs. Specifically, it is not defined when the input is [], the empty list. This makes sense, as it isn\u2019t possible to return the first element of a list if the list is empty\u2014there\u2019s no element to return! So, remarkably, we learn this function isn\u2019t possible to implement, either.Turning partial functions totalTo someone coming from a dynamically-typed background, this might seem perplexing. If we have a list, we might very well want to get the first element in it. And indeed, the operation of \u201cgetting the first element of a list\u201d isn\u2019t impossible in Haskell, it just requires a little extra ceremony. There are two different ways to fix the head function, and we\u2019ll start with the simplest one.Managing expectationsAs established, head is partial because there is no element to return if the list is empty: we\u2019ve made a promise we cannot possibly fulfill. Fortunately, there\u2019s an easy solution to that dilemma: we can weaken our promise. Since we cannot guarantee the caller an element of the list, we\u2019ll have to practice a little expectation management: we\u2019ll do our best return an element if we can, but we reserve the right to return nothing at all. In Haskell, we express this possibility using the Maybe type:head :: [a] -> Maybe aThis buys us the freedom we need to implement head\u2014it allows us to return Nothing when we discover we can\u2019t produce a value of type a after all:head :: [a] -> Maybe a\nhead (x:_) = Just x\nhead []    = NothingProblem solved, right? For the moment, yes\u2026 but this solution has a hidden cost.Returning Maybe is undoubtably convenient when we\u2019re implementing  head. However, it becomes significantly less convenient when we want to actually use it! Since head always has the potential to return Nothing, the burden falls upon its callers to handle that possibility, and sometimes that passing of the buck can be incredibly frustrating. To see why, consider the following code:getConfigurationDirectories :: IO [FilePath]\ngetConfigurationDirectories = do\n  configDirsString <- getEnv \"CONFIG_DIRS\"\n  let configDirsList = split ',' configDirsString\n  when (null configDirsList) $\n    throwIO $ userError \"CONFIG_DIRS cannot be empty\"\n  pure configDirsList\n\nmain :: IO ()\nmain = do\n  configDirs <- getConfigurationDirectories\n  case head configDirs of\n    Just cacheDir -> initializeCache cacheDir\n    Nothing -> error \"should never happen; already checked configDirs is non-empty\"When getConfigurationDirectories retrieves a list of file paths from the environment, it proactively checks that the list is non-empty. However, when we use head in main to get the first element of the list, the Maybe FilePath result still requires us to handle a Nothing case that we know will never happen! This is terribly bad for several reasons:First, it\u2019s just annoying. We already checked that the list is non-empty, why do we have to clutter our code with another redundant check?Second, it has a potential performance cost. Although the cost of the redundant check is trivial in this particular example, one could imagine a more complex scenario where the redundant checks could add up, such as if they were happening in a tight loop.Finally, and worst of all, this code is a bug waiting to happen! What if getConfigurationDirectories were modified to stop checking that the list is empty, intentionally or unintentionally? The programmer might not remember to update main, and suddenly the \u201cimpossible\u201d error becomes not only possible, but probable.The need for this redundant check has essentially forced us to punch a hole in our type system. If we could statically prove the Nothing case impossible, then a modification to getConfigurationDirectories that stopped checking if the list was empty would invalidate the proof and trigger a compile-time failure. However, as-written, we\u2019re forced to rely on a test suite or manual inspection to catch the bug.Paying it forwardClearly, our modified version of head leaves some things to be desired. Somehow, we\u2019d like it to be smarter: if we already checked that the list was non-empty, head should unconditionally return the first element without forcing us to handle the case we know is impossible. How can we do that?Let\u2019s look at the original (partial) type signature for head again:head :: [a] -> aThe previous section illustrated that we can turn that partial type signature into a total one by weakening the promise made in the return type. However, since we don\u2019t want to do that, there\u2019s only one thing left that can be changed: the argument type (in this case, [a]). Instead of weakening the return type, we can strengthen the argument type, eliminating the possibility of head ever being called on an empty list in the first place.To do this, we need a type that represents non-empty lists. Fortunately, the existing NonEmpty type from Data.List.NonEmpty is exactly that. It has the following definition:data NonEmpty a = a :| [a]Note that NonEmpty a is really just a tuple of an a and an ordinary, possibly-empty [a]. This conveniently models a non-empty list by storing the first element of the list separately from the list\u2019s tail: even if the [a] component is [], the a component must always be present. This makes head completely trivial to implement:2head :: NonEmpty a -> a\nhead (x:|_) = xUnlike before, GHC accepts this definition without complaint\u2014this definition is total, not partial. We can update our program to use the new implementation:getConfigurationDirectories :: IO (NonEmpty FilePath)\ngetConfigurationDirectories = do\n  configDirsString <- getEnv \"CONFIG_DIRS\"\n  let configDirsList = split ',' configDirsString\n  case nonEmpty configDirsList of\n    Just nonEmptyConfigDirsList -> pure nonEmptyConfigDirsList\n    Nothing -> throwIO $ userError \"CONFIG_DIRS cannot be empty\"\n\nmain :: IO ()\nmain = do\n  configDirs <- getConfigurationDirectories\n  initializeCache (head configDirs)Note that the redundant check in main is now completely gone! Instead, we perform the check exactly once, in getConfigurationDirectories. It constructs a NonEmpty a from a [a] using the nonEmpty function from Data.List.NonEmpty, which has the following type:nonEmpty :: [a] -> Maybe (NonEmpty a)The Maybe is still there, but this time, we handle the Nothing case very early in our program: right in the same place we were already doing the input validation. Once that check has passed, we now have a NonEmpty FilePath value, which preserves (in the type system!) the knowledge that the list really is non-empty. Put another way, you can think of a value of type NonEmpty a as being like a value of type [a], plus a proof that the list is non-empty.By strengthening the type of the argument to head instead of weakening the type of its result, we\u2019ve completely eliminated all the problems from the previous section:The code has no redundant checks, so there can\u2019t be any performance overhead.Furthermore, if getConfigurationDirectories changes to stop checking that the list is non-empty, its return type must change, too. Consequently, main will fail to typecheck, alerting us to the problem before we even run the program!What\u2019s more, it\u2019s trivial to recover the old behavior of head from the new one by composing head with nonEmpty:head' :: [a] -> Maybe a\nhead' = fmap head . nonEmptyNote that the inverse is not true: there is no way to obtain the new version of head from the old one. All in all, the second approach is superior on all axes.The power of parsingYou may be wondering what the above example has to do with the title of this blog post. After all, we only examined two different ways to validate that a list was non-empty\u2014no parsing in sight. That interpretation isn\u2019t wrong, but I\u2019d like to propose another perspective: in my mind, the difference between validation and parsing lies almost entirely in how information is preserved. Consider the following pair of functions:validateNonEmpty :: [a] -> IO ()\nvalidateNonEmpty (_:_) = pure ()\nvalidateNonEmpty [] = throwIO $ userError \"list cannot be empty\"\n\nparseNonEmpty :: [a] -> IO (NonEmpty a)\nparseNonEmpty (x:xs) = pure (x:|xs)\nparseNonEmpty [] = throwIO $ userError \"list cannot be empty\"These two functions are nearly identical: they check if the provided list is empty, and if it is, they abort the program with an error message. The difference lies entirely in the return type: validateNonEmpty always returns (), the type that contains no information, but parseNonEmpty returns NonEmpty a, a refinement of the input type that preserves the knowledge gained in the type system. Both of these functions check the same thing, but parseNonEmpty gives the caller access to the information it learned, while validateNonEmpty just throws it away.These two functions elegantly illustrate two different perspectives on the role of a static type system: validateNonEmpty obeys the typechecker well enough, but only parseNonEmpty takes full advantage of it. If you see why parseNonEmpty is preferable, you understand what I mean by the mantra \u201cparse, don\u2019t validate.\u201d Still, perhaps you are skeptical of parseNonEmpty\u2019s name. Is it really parsing anything, or is it merely validating its input and returning a result? While the precise definition of what it means to parse or validate something is debatable, I believe parseNonEmpty is a bona-fide parser (albeit a particularly simple one).Consider: what is a parser? Really, a parser is just a function that consumes less-structured input and produces more-structured output. By its very nature, a parser is a partial function\u2014some values in the domain do not correspond to any value in the range\u2014so all parsers must have some notion of failure. Often, the input to a parser is text, but this is by no means a requirement, and parseNonEmpty is a perfectly cromulent parser: it parses lists into non-empty lists, signaling failure by terminating the program with an error message.Under this flexible definition, parsers are an incredibly powerful tool: they allow discharging checks on input up-front, right on the boundary between a program and the outside world, and once those checks have been performed, they never need to be checked again! Haskellers are well-aware of this power, and they use many different types of parsers on a regular basis:The aeson library provides a Parser type that can be used to parse JSON data into domain types.Likewise, optparse-applicative provides a set of parser combinators for parsing command-line arguments.Database libraries like persistent and postgresql-simple have a mechanism for parsing values held in an external data store.The servant ecosystem is built around parsing Haskell datatypes from path components, query parameters, HTTP headers, and more.The common theme between all these libraries is that they sit on the boundary between your Haskell application and the external world. That world doesn\u2019t speak in product and sum types, but in streams of bytes, so there\u2019s no getting around a need to do some parsing. Doing that parsing up front, before acting on the data, can go a long way toward avoiding many classes of bugs, some of which might even be security vulnerabilities.One drawback to this approach of parsing everything up front is that it sometimes requires values be parsed long before they are actually used. In a dynamically-typed language, this can make keeping the parsing and processing logic in sync a little tricky without extensive test coverage, much of which can be laborious to maintain. However, with a static type system, the problem becomes marvelously simple, as demonstrated by the NonEmpty example above: if the parsing and processing logic go out of sync, the program will fail to even compile.The danger of validationHopefully, by this point, you are at least somewhat sold on the idea that parsing is preferable to validation, but you may have lingering doubts. Is validation really so bad if the type system is going to force you to do the necessary checks eventually anyway? Maybe the error reporting will be a little bit worse, but a bit of redundant checking can\u2019t hurt, right?Unfortunately, it isn\u2019t so simple. Ad-hoc validation leads to a phenomenon that the language-theoretic security field calls shotgun parsing. In the 2016 paper, The Seven Turrets of Babel: A Taxonomy of LangSec Errors and How to Expunge Them, its authors provide the following definition:Shotgun parsing is a programming antipattern whereby parsing and input-validating code is mixed with and spread across processing code\u2014throwing a cloud of checks at the input, and hoping, without any systematic justification, that one or another would catch all the \u201cbad\u201d cases.They go on to explain the problems inherent to such validation techniques:Shotgun parsing necessarily deprives the program of the ability to reject invalid input instead of processing it. Late-discovered errors in an input stream will result in some portion of invalid input having been processed, with the consequence that program state is difficult to accurately predict.In other words, a program that does not parse all of its input up front runs the risk of acting upon a valid portion of the input, discovering a different portion is invalid, and suddenly needing to roll back whatever modifications it already executed in order to maintain consistency. Sometimes this is possible\u2014such as rolling back a transaction in an RDBMS\u2014but in general it may not be.It may not be immediately apparent what shotgun parsing has to do with validation\u2014after all, if you do all your validation up front, you mitigate the risk of shotgun parsing. The problem is that validation-based approaches make it extremely difficult or impossible to determine if everything was actually validated up front or if some of those so-called \u201cimpossible\u201d cases might actually happen. The entire program must assume that raising an exception anywhere is not only possible, it\u2019s regularly necessary.Parsing avoids this problem by stratifying the program into two phases\u2014parsing and execution\u2014where failure due to invalid input can only happen in the first phase. The set of remaining failure modes during execution is minimal by comparison, and they can be handled with the tender care they require.Parsing, not validating, in practiceSo far, this blog post has been something of a sales pitch. \u201cYou, dear reader, ought to be parsing!\u201d it says, and if I\u2019ve done my job properly, at least some of you are sold. However, even if you understand the \u201cwhat\u201d and the \u201cwhy,\u201d you might not feel especially confident about the \u201chow.\u201dMy advice: focus on the datatypes.Suppose you are writing a function that accepts a list of tuples representing key-value pairs, and you suddenly realize you aren\u2019t sure what to do if the list has duplicate keys. One solution would be to write a function that asserts there aren\u2019t any duplicates in the list:checkNoDuplicateKeys :: (MonadError AppError m, Eq k) => [(k, v)] -> m ()However, this check is fragile: it\u2019s extremely easy to forget. Because its return value is unused, it can always be omitted, and the code that needs it would still typecheck. A better solution is to choose a data structure that disallows duplicate keys by construction, such as a Map. Adjust your function\u2019s type signature to accept a Map instead of a list of tuples, and implement it as you normally would.Once you\u2019ve done that, the call site of your new function will likely fail to typecheck, since it is still being passed a list of tuples. If the caller was given the value via one of its arguments, or if it received it from the result of some other function, you can continue updating the type from list to Map, all the way up the call chain. Eventually, you will either reach the location the value is created, or you\u2019ll find a place where duplicates actually ought to be allowed. At that point, you can insert a call to a modified version of checkNoDuplicateKeys:checkNoDuplicateKeys :: (MonadError AppError m, Eq k) => [(k, v)] -> m (Map k v)Now the check cannot be omitted, since its result is actually necessary for the program to proceed!This hypothetical scenario highlights two simple ideas:Use a data structure that makes illegal states unrepresentable. Model your data using the most precise data structure you reasonably can. If ruling out a particular possibility is too hard using the encoding you are currently using, consider alternate encodings that can express the property you care about more easily. Don\u2019t be afraid to refactor.Push the burden of proof upward as far as possible, but no further. Get your data into the most precise representation you need as quickly as you can. Ideally, this should happen at the boundary of your system, before any of the data is acted upon.3If one particular code branch eventually requires a more precise representation of a piece of data, parse the data into the more precise representation as soon as the branch is selected. Use sum types judiciously to allow your datatypes to reflect and adapt to control flow.In other words, write functions on the data representation you wish you had, not the data representation you are given. The design process then becomes an exercise in bridging the gap, often by working from both ends until they meet somewhere in the middle. Don\u2019t be afraid to iteratively adjust parts of the design as you go, since you may learn something new during the refactoring process!Here are a handful of additional points of advice, arranged in no particular order:Let your datatypes inform your code, don\u2019t let your code control your datatypes. Avoid the temptation to just stick a Bool in a record somewhere because it\u2019s needed by the function you\u2019re currently writing. Don\u2019t be afraid to refactor code to use the right data representation\u2014the type system will ensure you\u2019ve covered all the places that need changing, and it will likely save you a headache later.Treat functions that return m () with deep suspicion. Sometimes these are genuinely necessary, as they may perform an imperative effect with no meaningful result, but if the primary purpose of that effect is raising an error, it\u2019s likely there\u2019s a better way.Don\u2019t be afraid to parse data in multiple passes. Avoiding shotgun parsing just means you shouldn\u2019t act on the input data before it\u2019s fully parsed, not that you can\u2019t use some of the input data to decide how to parse other input data. Plenty of useful parsers are context-sensitive.Avoid denormalized representations of data, especially if it\u2019s mutable. Duplicating the same data in multiple places introduces a trivially representable illegal state: the places getting out of sync. Strive for a single source of truth.Keep denormalized representations of data behind abstraction boundaries. If denormalization is absolutely necessary, use encapsulation to ensure a small, trusted module holds sole responsibility for keeping the representations in sync.Use abstract datatypes to make validators \u201clook like\u201d parsers. Sometimes, making an illegal state truly unrepresentable is just plain impractical given the tools Haskell provides, such as ensuring an integer is in a particular range. In that case, use an abstract newtype with a smart constructor to \u201cfake\u201d a parser from a validator.As always, use your best judgement. It probably isn\u2019t worth breaking out singletons and refactoring your entire application just to get rid of a single error \"impossible\" call somewhere\u2014just make sure to treat those situations like the radioactive substance they are, and handle them with the appropriate care. If all else fails, at least leave a comment to document the invariant for whoever needs to modify the code next.Recap, reflection, and related readingThat\u2019s all, really. Hopefully this blog post proves that taking advantage of the Haskell type system doesn\u2019t require a PhD, and it doesn\u2019t even require using the latest and greatest of GHC\u2019s shiny new language extensions\u2014though they can certainly sometimes help! Sometimes the biggest obstacle to using Haskell to its fullest is simply being aware what options are available, and unfortunately, one downside of Haskell\u2019s small community is a relative dearth of resources that document design patterns and techniques that have become tribal knowledge.None of the ideas in this blog post are new. In fact, the core idea\u2014\u201cwrite total functions\u201d\u2014is conceptually quite simple. Despite that, I find it remarkably challenging to communicate actionable, practicable details about the way I write Haskell code. It\u2019s easy to spend lots of time talking about abstract concepts\u2014many of which are quite valuable!\u2014without communicating anything useful about process. My hope is that this is a small step in that direction.Sadly, I don\u2019t know very many other resources on this particular topic, but I do know of one: I never hesitate to recommend Matt Parson\u2019s fantastic blog post Type Safety Back and Forth. If you want another accessible perspective on these ideas, including another worked example, I\u2019d highly encourage giving it a read. For a significantly more advanced take on many of these ideas, I can also recommend Matt Noonan\u2019s 2018 paper Ghosts of Departed Proofs, which outlines a handful of techniques for capturing more complex invariants in the type system than I have described here.As a closing note, I want to say that doing the kind of refactoring described in this blog post is not always easy. The examples I\u2019ve given are simple, but real life is often much less straightforward. Even for those experienced in type-driven design, it can be genuinely difficult to capture certain invariants in the type system, so do not consider it a personal failing if you cannot solve something the way you\u2019d like! Consider the principles in this blog post ideals to strive for, not strict requirements to meet. All that matters is to try.Technically, in Haskell, this ignores \u201cbottoms,\u201d constructions that can inhabit any value. These aren\u2019t \u201creal\u201d values (unlike null in some other languages)\u2014they\u2019re things like infinite loops or computations that raise exceptions\u2014and in idiomatic Haskell, we usually try to avoid them, so reasoning that pretends they don\u2019t exist still has value. But don\u2019t take my word for it\u2014I\u2019ll let Danielsson et al. convince you that Fast and Loose Reasoning is Morally Correct. \u21a9In fact, Data.List.NonEmpty already provides a head function with this type, but just for the sake of illustration, we\u2019ll reimplement it ourselves. \u21a9Sometimes it is necessary to perform some kind of authorization before parsing user input to avoid denial of service attacks, but that\u2019s okay: authorization should have a relatively small surface area, and it shouldn\u2019t cause any significant modifications to the state of your system. \u21a9"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:24:17.471476+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-08-03T21:24:15.955875+00:00",
                "index_texts": null,
                "output": "Parse, don\u2019t validate",
                "pwd": "/data/archive/1754256232.237508",
                "schema": "ArchiveResult",
                "start_ts": "2025-08-03T21:24:15.789916+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "https://web.archive.org/web/20250803212449/https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "Parse, don\u2019t validate",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1754256232.237508",
    "newest_archive_date": "2025-08-03T21:24:45.068764+00:00",
    "num_failures": 0,
    "num_outputs": 9,
    "oldest_archive_date": "2025-08-03T21:23:55.462463+00:00",
    "path": "/blog/2019/11/05/parse-don-t-validate/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01K1RX3KTY59CB22B9011NRY39",
    "snapshot_id": "28a72cff-b292-4086-81f9-d1b8435c7869",
    "sources": [
        "/data/sources/1754256230-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1754256232.237508",
    "title": "Parse, don\u2019t validate",
    "url": "https://lexi-lambda.github.io/blog/2019/11/05/parse-don-t-validate/"
}