{
    "archive_path": "archive/1779046538.567373",
    "base_url": "matklad.github.io/2021/05/31/how-to-test.html",
    "basename": "how-to-test.html",
    "bookmarked_date": "2026-05-17 19:35",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/matklad.github.io/2021/05/31/how-to-test.html",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=matklad.github.io",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "matklad.github.io",
    "downloaded_at": "2026-05-17T19:35:44.247757+00:00",
    "downloaded_datestr": "2026-05-17 19:35",
    "extension": "html",
    "hash": "1W49KXJDEZJ6SD1079K5",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://matklad.github.io/2021/05/31/how-to-test.html"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-05-17T19:36:43.636203+00:00",
                "index_texts": null,
                "output": "ArchiveError: Failed to find \"content-location\" URL header in Archive.org response.",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:36:41.902323+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://matklad.github.io/2021/05/31/how-to-test.html"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2026-05-17T19:36:09.126456+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:35:56.900445+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=matklad.github.io"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-05-17T19:35:47.669263+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:35:44.766274+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://matklad.github.io/2021/05/31/how-to-test.html"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-05-17T19:35:47.759284+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:35:47.710825+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2026-05-17T19:36:32.878001+00:00",
                "index_texts": [
                    "How to Test (/favicon.png) (/favicon.svg) (https://matklad.github.io/2021/05/31/how-to-test.html) (matklad) (https://matklad.github.io/feed.xml) (/css/main.css)  (/) matklad (/about.html) About (/links.html) Links (/blogroll.html) Blogroll   How to Test May 31, 2021  Alternative titles:Unit Tests are a Scam Test Features, Not Code Data Driven Integrated\n            Tests  This post describes my current approach to testing. When I started\n          programming professionally, I knew how to write good code, but good\n          tests remained a mystery for a long time. This is not due to the lack\n          of advice \u2014 on the contrary, there\u2019s abundance of information &\n          terminology about testing. This celestial emporium of benevolent\n          knowledge includes TDD, BDD, unit tests, integrated tests, integration\n          tests, end-to-end tests, functional tests, non-functional tests,\n          blackbox tests, glassbox tests, \u2026  Knowing all this didn\u2019t help me to create better software. What did\n          help was trying out different testing approaches myself, and looking\n          at how other people write tests. Keep in mind that my background is\n          mostly in writing (https://github.com/intellij-rust/intellij-rust) compiler (https://github.com/rust-analyzer/rust-analyzer/) front-ends for IDEs. This is a rather niche area, which is\n          especially amendable to testing. Compilers are pure self-contained\n          functions. I don\u2019t know how to best test modern HTTP applications\n          built around inter-process communication.  Without further ado, let\u2019s see what I have learned! Further ado(2024-05-21):  while\n          writing this post, I was missing a key piece of terminology for\n          crisply describing various kinds of tests. If you like this post, you\n          might want to read (https://matklad.github.io/2022/07/04/unit-and-integration-tests.html) Unit and Integration Tests  . That post supplies better vocabulary for talking about phenomena\n          described in the present article.  Test Driven Design Ossification  This is something I inflicted upon myself early in my career, and\n            something I routinely observe. You want to refactor some code, say\n            add a new function parameter. Turns out, there are a dozen of tests\n            calling this function, so now a simple refactor also involves fixing\n            all the tests.  There is a simple, mechanical fix to this problem: introduce the check function which encapsulates API under test. It\u2019s\n            easier to explain using a toy example. Let\u2019s look at testing\n            something simple, like a binary search, just to illustrate the\n            technique.  We start with direct testing: /// Given a *sorted* `haystack`, returns `true`  /// if it contains the `needle`.  fn binary_search (haystack: &[T], needle: &T) -> bool { ... }  #[test]  fn binary_search_empty () { let res = binary_search (&[], &0 ); assert_eq! (res, false ); }  #[test]  fn binary_search_singleton () { let res = binary_search (&[92 ], &0 ); assert_eq! (res, false );  let res = binary_search (&[92 ], &92 ); assert_eq! (res, true );  let res = binary_search (&[92 ], &100 ); assert_eq! (res, false ); }  // And a dozen more of other similar tests...     Some time passes, and we realize that -> bool is not\n            the best signature for binary search. It\u2019s better if it returned an\n            insertion point (an index where element should be inserted to\n            maintain sortedness). That is, we want to change the signature to  fn binary_search (haystack: &[T], needle: &T) -> Result <usize , usize >;    Now we have to change every test, because the tests are tightly\n            coupled to the specific API.  My solution to this problem is making the tests data driven. Instead\n            of every test interacting with the API directly, I like to define a\n            single check function which calls the API. This\n            function takes a pair of input and expected result. For binary\n            search example, it will look like this:  #[track_caller]  fn check ( input_haystack: &[i32 ], input_needle: i32 , expected_result: bool , ) { let actual_result = binary_search (input_haystack, &input_needle); assert_eq! (expected_result, actual_result); }  #[test]  fn binary_search_empty () { check (&[], 0 , false ); }  #[test]  fn binary_search_singleton () { check (&[92 ], 0 , false ); check (&[92 ], 92 , true ); check (&[92 ], 100 , false ); }    Now, when the API of the binary_search function\n            changes, we only need to adjust the single place \u2014 check function:  #[track_caller]  fn check ( input_haystack: &[i32 ], input_needle: i32 , expected_result: bool , ) { let actual_result = binary_search (input_haystack, &input_needle).is_ok (); assert_eq! (expected_result, actual_result); }    To be clear, after you\u2019ve done the refactor, you\u2019ll need to adjust\n            the tests to check the index as well, but this can be done\n            separately. Existing test suite does not impede changes.  (/assets/icons.svg#info)   Key point:  keep an eye on\n                tests standing in a way of refactors. Use the check idiom to make tests resilient to changes.    Keep in mind that the binary search example is artificially simple.\n            The main danger here is that this is a (https://en.wikipedia.org/wiki/Boiling_frog) boiling frog type of situation. While the project is small and\n            the tests are few, you don\u2019t notice that refactors are ever so\n            slightly longer than necessary. Then, several tens of thousands\n            lines of code later, you realize that to make a simple change you\n            need to fix a hundred tests.   Test Friction  Almost no one likes to write tests. I\u2019ve noticed many times how,\n            upon fixing a trivial bug, I am prone to skipping the testing work.\n            Specifically, if writing a test is more effort than the fix itself,\n            testing tends to go out of the window. Hence,  (/assets/icons.svg#info)   Key point:  work hard on making\n                adding new tests trivial.    Coming back to the binary search example, note how check function reduces the amount of typing to add a new\n            test. For tests, this is a significant saving, not because typing is\n            hard, but because it lowers the cognitive barrier to actually do the\n            work.   Test Features, Not Code  The over-simplified binary search example can be stretched further.\n            What if you replace the sorted array with a hash map inside your\n            application? Or what if the calling code no longer needs to search\n            at all, and wants to process all of the elements instead?  Good code (https://programmingisterrible.com/post/139222674273/how-to-write-disposable-code-in-large-systems) is easy to delete . Tests represent an investment into existing\n            code, and make it costlier to delete (or change).  The solution is to write tests for features in such a way that they\n            are independent of the code. I like to use the neural network test\n            for this:  Neural Network Test Can you re-use the test suite if your entire software is\n                replaced with an opaque neural network?    To give a real-life example this time, suppose that you are writing\n            that part of code-completion engine which sorts potential\n            completions according to relevance. (something I should probably be\n            doing right now, instead of writing this article :-) )  Internally, you have a bunch of functions that compute relevance\n            facts, like:  Is there direct type match (.foo has the desired\n              type)?  Is there an indirect type match (.foo.bar has the\n              right type)?  How frequently is this completion used in the current module?   Then, there\u2019s the final ranking function that takes these facts and\n            comes up with an overall rank.  The classical unit-test approach here would be to write a bunch of\n            isolated tests for each of the relevance functions, and a separate\n            bunch of tests which feeds the ranking function a list of relevance\n            facts and checks the final score.  This approach obviously fails the neural network test. An alternative approach is to write a test to check that at a given\n            position a specific ordered list of entries is returned. That suite\n            could work as a cross-validation for an ML-based implementation.  In practice, it\u2019s unlikely (but not impossible), that we use actual\n            ML here. But it\u2019s highly probably that the naive independent weights\n            model isn\u2019t the end of the story. At some point there will be\n            special cases which would necessitate change of the interface.  (/assets/icons.svg#info)   Key point:  duh, test features,\n                not code! (https://www.tedinski.com/2019/03/19/testing-at-the-boundaries.html) Test at the boundaries .  If you build a library, the boundary is the public API. If you\n                are building an application, you are not building the library.\n                The boundary is what a human in front of a display sees.    Note that this advice goes directly against one common understanding\n            of unit-testing. I am fairly confident that it results in better\n            software over the long run.   Make Tests Fast  There\u2019s one talk about software engineering, which stands out for\n            me, and which is my favorite. It is (https://www.destroyallsoftware.com/talks/boundaries) Boundaries by Gary Bernhardt. There\u2019s a point there though,\n            which I strongly disagree with:  Integration Tests are Superlinear? When you use integration tests, any new feature is accompanied\n                by a bit of new code and a new test. However, new code slows\n                down all other tests, so the the overall test suite becomes\n                slow, as the total time grows super-linearly.    I don\u2019t think more code under test translates to slower test suite.\n            Merge sort spends more lines of code than bubble sort, but it is way\n            faster.  In the abstract, yes, more code generally means more execution time,\n            but I doubt this is the defining factor in tests execution time.\n            What actually happens is usually:  Input/Output \u2014 reading just a bit from a disk, network or another\n              process slows down the tests significantly.  Outliers \u2014 very often, testing time is dominated by only a couple\n              of slow tests.  Overly large input \u2014 throwing enough data at any software makes it\n              slow.   The problem with integrated tests is not code volume per se, but the\n            fact that they typically mean doing a lot of IO. But this\n            doesn\u2019t need to be the case  (/assets/icons.svg#info)   Key point:  architecture the\n                software to keep as much as possible (https://sans-io.readthedocs.io) sans io . Let the caller do input and output, and let the\n                callee do compute. It doesn\u2019t matter if the callee is large and\n                complex. Even if it is the whole compiler, testing is fast and\n                easy as long as no IO is involved.    Nonetheless, some tests are going to be slow. It pays off to\n            introduce the concept of slow tests early on, arrange the skipping\n            of such tests by default and only exercise them on CI. You don\u2019t\n            need to be fancy, just checking an environment variable at the start\n            of the test is perfectly fine:  #[test]  fn completion_works_with_real_standard_library () { if std::env::var (\"RUN_SLOW_TESTS\" ).is_err () { return ; } ... }    Definitely do not use conditional compilation to hide slow\n            tests \u2014 it\u2019s an obvious solution which makes your life harder ((https://peter.bourgon.org/blog/2021/04/02/dont-use-build-tags-for-integration-tests.html) similar observation from the Go ecosystem).  To deal with outliers, print each test\u2019s execution time by default.\n            Having the numbers fly by gives you immediate feedback and incentive\n            to improve.   Data Driven Testing  All these together lead to a particular style of architecture and\n            tests, which I call data driven testing. The bulk of the software is\n            a pure function, where the state is passed in explicitly. Removing\n            IO from the picture necessitates that the interface of software is\n            specified in terms of data. Value in, value out.  One property of data is that it can be serialized and deserialized.\n            That means that the check style tests can easily accept\n            arbitrary complex input, which is specified in a structured format\n            (JSON), ad-hoc plain text format, or via embedded DSL (builder-style\n            interface for data objects).  Similarly, The \u201cexpected\u201d argument of check is data. It\n            is a result which is more-or-less directly displayed to the user.  A convincing example of a data driven test would be a \u201cGoto\n            Definition\u201d tests from rust-analyzer ((https://github.com/rust-analyzer/rust-analyzer/blob/92b9e5ef3c03d51713ff5fa32cd58bdf97701b5e/crates/ide/src/goto_definition.rs#L168-L185) source ):  ()  In this case, the check function has only a single\n            argument \u2014 a string which specifies both the input and the expected\n            result. The input is a rust project with three files (//-\n              /file.rs syntax shows the boundary between the files). The\n            current cursor position is also a part of the input and is specified\n            with the $0 syntax. The result is the //^^^ comment which marks the target of the \u201cGoto\n            Definition\u201d call. The check function creates an\n            in-memory Rust project, invokes \u201cGoto Definition\u201d at the position\n            signified by $0 , and checks that the result is the\n            position marked with ^^^ .  Note that this is decidedly not a unit test. Nothing is stubbed or\n            mocked. This test invokes the whole compilation pipeline: virtual\n            file system, parser, macro expander, name resolution. It runs on top\n            of our incremental computation engine. It touches a significant\n            fraction of the IDE APIs. Yet, it takes 4ms in debug mode (and 500\u00b5s\n            in release mode). And note that it absolutely does not depend on any\n            internal API \u2014 if we replace our dumb compiler with sufficiently\n            smart neural net, nothing needs to be adjusted in the tests.  There\u2019s one question though: why on earth am I using a png image to\n            display a bit of code? Only to show that the raw string literal\n            (r#\"\"# ) which contains Rust code is highlighted as\n            such. This is possible because we re-use the same input format (with //- , $0 and couple of other markup\n            elements) for almost every test in rust-analyzer. As such, we can\n            invest effort into building cool things on top of this format, which\n            subsequently benefit all our tests.   Expect Tests  Previous example had a complex data input, but a relatively simple\n            data output \u2014 a position in the file. Often, the output is messy and\n            has a complicated structure as well (a symptom of (https://buttondown.email/hillelwayne/archive/cross-branch-testing/) rho problem ). Worse, sometimes the output is a part that is\n            changed frequently. This often necessitates updating a lot of tests.\n            Going back to the binary search example, the change from ->\n              bool to -> Result<usize, usize> was\n            an example of this effect.  There is a technique to make such simultaneous changes to all gold\n            outputs easy \u2014 testing with expectations. You specify the expected\n            result as a bit of data inline with the test. There\u2019s a special mode\n            of running the test suite for updating this data. Instead of failing\n            the test, a mismatch between expected and actual causes the gold\n            value to be updated in-place. That is, the test framework edits the\n            code of the test itself.  Here\u2019s an example of this workflow in rust-analyzer, used for\n            testing code completion:    Often, just Debug representation of the type works well\n            for expect tests, but you can do something more fun. See this post\n            from Jane Street for a great example: (https://blog.janestreet.com/using-ascii-waveforms-to-test-hardware-designs/) Using ASCII waveforms to test hardware designs .  There are several libraries for this in Rust: (https://github.com/mitsuhiko/insta) insta , (https://github.com/aaronabramov/k9) k9 , (https://github.com/rust-analyzer/expect-test) expect-test .   Fluent Assertions  An extremely popular genre for a testing library is a collection of\n            fluent assertions:  // Built-in assertion:  assert! (x > y);  // Fluent assertion:  assert_that (x).is_greater_than (y);    The benefit of this style are better error messages. Instead of just\n            \u201cfalse is not true\u201d, the testing framework can print values for x and y .  I don\u2019t find this useful. Using the check style\n            testing, there are very few assertions actually written in code.\n            Usually, I start with plain asserts without messages. The first time\n            I debug an actual test failure for a particular function, I spend\n            some time to write a detailed assertion message. To me, fluent\n            assertions are not an attractive point on the curve that includes\n            plain asserts and hand-written, context aware explanations of\n            failures. A notable exception here is pytest approach \u2014 this testing\n            framework overrides the standard assert to provide a\n            rich diff without ceremony.  (/assets/icons.svg#info)   Key Point:  invest into testing\n                infrastructure in a scalable way. Write a single check function with artisanally crafted error message,\n                define a universal fixture format for the input, use expectation\n                testing for output.     Peeking Inside  One apparent limitation of the style of integrated testing I am\n            describing is checking for properties which are not part of\n            the output. For example, if some kind of caching is involved, you\n            might want to check that the cache is actually being hit, and is not\n            just sitting there. But, by definition, cache is not something that\n            an outside client can observe.  The solution to this problem is to make this extra data a part of\n            the system\u2019s output by adding extra observability points. A good\n            example here is Cargo\u2019s test suite. It is set-up in an integrated,\n            data-driven fashion. Each tests starts with a succinct DSL for\n            setting up a tree of files on disk. Then, a full cargo command is\n            invoked. Finally, the test looks at the command\u2019s output and the\n            resulting state of the file system, and asserts the relevant facts.  Tests for caching additionally enable verbose internal logging. In\n            this mode, Cargo prints information about cache hits and misses.\n            These messages are then used (https://github.com/rust-lang/cargo/blob/57b75970e022e8519fe82cc38a7aed4862f67089/tests/testsuite/rustc_info_cache.rs#L68-L70) in assertions .  A close idea is (https://ferrous-systems.com/blog/coverage-marks/) coverage marks . Some times, you want to check that something does not  happen. Tests for this tend to be fragile\n            \u2014 often the thing does not happen, but for the wrong reason. You can\n            add a side channel which explains the reasoning behind particular\n            behavior, and additionally assert this as well.   Externalized Tests  In the ultimate stage of data driven tests the definitions of test\n            cases are moved out of test functions and into external files. That\n            is, you don\u2019t do this:  #[test]  fn test_foo () { check (\"foo\" , \"oof\" ) }  #[test]  fn test_bar () { check (\"bar\" , \"rab\" ) }    Rather, there is a single test that looks like this: #[test]  fn test_all () { for file in read_dir (\"./test_data/in\" ) { let input = read_to_string ( &format! (\"./test_data/in/{}\" , file), ); let output = read_to_string ( &format! (\"./test_data/out/{}\" , file), ); check (input, output) } }    I have a love-hate relationship with this approach. It has at least\n            two attractive properties. First, it forces data driven approach without any cheating. Second, it makes the test suite more re-usable. An\n            alternative implementation in a different programming language can\n            use the same tests.  But there\u2019s a drawback as well \u2014 without literal #[test] attributes, integration with tooling suffers. For\n            example, you don\u2019t automatically get \u201cX out of Y tests passed\u201d at\n            the end of test run. You can\u2019t conveniently debug just a single\n            test, there isn\u2019t a helpful \u201cRun\u201d icon/shortcut you can use in an\n            IDE.  When I do externalized test cases, I like to leave a trivial smoke\n            test behind:  #[test]  fn smoke () { check (\"\" , \"\" ); }    If I need to debug a failing external test, I first paste the input\n            into this smoke test, and then get my IDE tooling back.   Beyond Example Based Testing  Reading from a file is not the most fun way to come up with a data\n            input for a check function.  Here are a few other popular ones: Property Based Testing Generate the input at random and verify that the output makes\n                sense. For a binary search, check that the needle indeed lies between the two elements at the insertion point.   Full Coverage Better still, instead of generating some random inputs, just\n                check that the answer is correct for all inputs. This\n                is how you should be testing binary search \u2014 generate every\n                sorted list of length at most 7 with numbers in the 0..=6 range. Then, for each list and for each\n                number, check that the binary search gives the same result as a\n                naive linear search.   Coverage Guided Fuzzing Just throw random bytes at the check function. Random bytes\n                probably don\u2019t make much sense, but it\u2019s good to verify that the\n                program returns an error instead of summoning nasal demons.\n                Instead of piling bytes completely at random, observe which\n                branches are taken, and try to invent byte sequences which cover\n                more branches. Note that this test is polymorphic in the system\n                under test.   Structured Fuzzing / Coverage Guided Property Testing Use random bytes as a seed to generate \u201csyntactically valid\u201d\n                inputs, then see you software crash and burn when the most\n                hideous edge cases are uncovered. If you use Rust, check out (https://github.com/bytecodealliance/wasm-tools/tree/f632261627a0ea758762e431d8be32740111e33c/crates/wasm-smith) wasm-smith and (https://lib.rs/crates/arbitrary) arbitrary crates.    (/assets/icons.svg#info)   Key Point:  once you formulated\n                the tests in terms of data, you no longer need to write code to\n                add your tests. If code is not required, you can generate test\n                cases easily.     The External World  What if isolating IO is not possible, and the application is\n            fundamentally build around interacting with external systems? In\n            this case, my advice is to just accept that the tests are going to\n            be slow, and might need extra effort to avoid flakiness.  Cargo is the perfect case study here. Its raison d\u2019\u00eatre is\n            orchestrating a herd of external processes. Let\u2019s look at the basic\n            test:  #[test]  fn cargo_compile_simple () { let p = project () .file (\"Cargo.toml\" , &basic_bin_manifest (\"foo\" )) .file (\"src/foo.rs\" , &main_file (r#\"\"i am foo\"\"# , &[])) .build ();  p.cargo (\"build\" ).run ();  assert! (p.bin (\"foo\" ).is_file ()); p.process (&p.bin (\"foo\" )).with_stdout (\"i am foo\\n\" ).run (); }    The project() part is a builder, which describes the\n            state of the a system. First, .build() writes the specified files to\n            a disk in a temporary directory. Then, p.cargo(\"build\").run() executes the real cargo build command. Finally, a bunch of assertions are made about the end state\n            of the file system.  Neural network test: this is completely independent of internal\n            Cargo APIs, by virtue of interacting with a cargo process via IPC.  To give an order-of-magnitude feeling for the cost of IO, Cargo\u2019s\n            test suite takes around seven minutes (-j 1 ), while\n            rust-analyzer finishes in less than half a minute.  An interesting case is the middle ground, when the IO-ing part is\n            just big and important enough to be annoying. That is the case for\n            rust-analyzer \u2014 although almost all code is pure, there\u2019s a part\n            which interacts with a specific editor. What makes this especially\n            finicky is that, in the case of Cargo, it\u2019s Cargo who calls external\n            processes. With rust-analyzer, it\u2019s something which we don\u2019t\n            control, the editor, which schedules the IO. This often results in\n            hard-to-imagine bugs which are caused by particularly weird\n            environments.  I don\u2019t have good answers here, but here are the tricks I use: Accept that something will break during integration. Even\n              if you always create perfect code and never make bugs,\n              your upstream integration point will be buggy sometimes.  Make integration bugs less costly: use release trains,  make patch release process non-exceptional and easy,  have a checklist for manual QA before the release.    Separate the tricky to test bits into a separate project. This\n              allows you to write slow and not 100% reliable tests for\n              integration parts, while keeping the core test suite fast and\n              dependable.   (/assets/icons.svg#info)   Key Point:  if you can\u2019t avoid\n                IO, embrace it. Even if a data driven test suite is slow, it\n                gives you a lot of confidence that features work, without\n                intervening with refactors.     The Concurrent World  Consider the following API: fn do_stuff_in_background (p: Param) { std::thread::spawn (move || { // Stuff  }) }    This API is fundamentally untestable. Can you see why? It spawns a\n            concurrent computation, but it doesn\u2019t allow waiting for this\n            computation to be finished. So, any test that calls do_stuff_in_background can\u2019t check that the \u201cStuff\u201d is done.\n            Worse, even tests which do not call this function might start to\n            fail \u2014 they now can get interference from other tests. The\n            concurrent computation can outlive the test that originally spawned\n            it.  This problem plagues almost every concurrent application I see. A\n            common symptom is adding timeouts and sleeps to test, to increase\n            the probability of stuff getting done. Such timeouts are another\n            common cause of slow test suites.  What makes this problem truly insidious is that there\u2019s no\n            work-around. Broken once, causality link is not reforgable by a\n            layer above.  The solution is simple: don\u2019t do this. (/assets/icons.svg#info)   Key Point:  grab a (large) cup\n                of coffee and go read (https://vorpus.org/blog/notes-on-structured-concurrency-or-go-statement-considered-harmful/) Go statement considered harmful . I will wait until you are\n                done, and then proceed with my article.     Layers  Another common problem I see in complex projects is a beautifully\n            layered architecture, which is \u201cinverted\u201d in tests.  Let\u2019s say you have something fabulous, like L1 <- L2 <-\n              L3 <- L4 . To test L1 , the path of least\n            resistance is often to write tests which exercise L4 .\n            You might even think that this is the setup I am advocating for. Not\n            exactly.  The problem with L1 <- L2 <- L3 <- L4 <-\n              Tests is that working on L1 becomes slower,\n            especially in compiled languages. If you make a change to L1 , then, before you get to the tests, you need to recompile\n            the whole chain of reverse dependencies. My \u201cfavorite\u201d example here\n            is rustc \u2014 when I worked on the lexer (T1 ), I spent a lot of time waiting for the rest of the\n            compiler to be rebuild to check my small change.  The right setup here is to write integrated tests for each layer:  L1 <- Tests L1 <- L2 <- Tests L1 <- L2 <- L3 <- Tests L1 <- L2 <- L3 <- L4 <- Tests    Note that testing L4 involves testing L1 , L2 an L3 . This is not a problem. Due to\n            layering, only L4 needs to be recompiled .\n            Other layers don\u2019t affect run time meaningfully. Remember \u2014\n            it\u2019s IO (and sleep-based synchronization) that kills performance,\n            not just code volume.   Test Everything  In a nutshell, a #[test] is just a bit of code which is\n            plugged into the build system to be executed automatically. Use this\n            to your advantage, simplify the automation by moving as much as\n            possible into tests.  Here\u2019s some things in rust-analyzer which are just\n            tests:  Code formatting (most common one \u2014 you don\u2019t need an extra pile of\n              YAML in CI, you can shell out to the formatter from the test).  Checking that the history does not contain merge commits and\n              teaching new contributors git survival skills.  Collecting the manual from specially-formatted doc comments across\n              the code base.  Checking that the code base is, in fact, reasonably\n              well-documented.  Ensuring that the licenses of dependencies are compatible.  Ensuring that high-level operations are linear in the size of the\n              input. Syntax-highlight a synthetic file of 1, 2, 4, 8, 16\n              kilobytes, run linear regression, check that result looks like a\n              line rather than a parabola.    Use Bors  This essay already mentioned a couple of cognitive tricks for better\n            testing: reducing the fixed costs for adding new tests, and\n            plotting/printing test times. The best trick in a similar vein is\n            the (https://graydon2.dreamwidth.org/1597.html) \u201cnot rocket science\u201d rule of software engineering.  The idea is to have a robot which checks that the merge\n                commit  passes all the tests, before advancing the\n            state of the main branch.  Besides the evergreen master, such bot adds pressure to keep the\n            test suite fast and non-flaky. This is another boiling frog,\n            something you need to constantly keep an eye on. If you have any a\n            single flaky test, it\u2019s very easy to miss when the second one is\n            added.  (/assets/icons.svg#info)   Key point:  use (https://bors.tech) https://bors.tech , a no-nonsense implementation of \u201cnot\n                rocket science\u201d rule.     Recap  This was a long essay. Let\u2019s look back at some of the key points:  There is a lot of information about testing, but it is not always\n              helpful. At least, it was not helpful for me.  The core characteristic of the test suite is how easy it makes\n              changing the software under test.  To that end, a good strategy is to focus on testing the features\n              of the application, rather than on testing the code used to\n              implement those features.  A good test suite passes the neural network test \u2014 it is still\n              useful if the entire application is replaced by an ML model which\n              just comes up with the right answer.  Corollary: good tests are not helpful for design in the small \u2014 a\n              good test won\u2019t tell you the best signatures for functions.  Testing time is something worth optimizing for. Tests are\n              sensitive to IO and IPC. Tests are relatively insensitive to the\n              amount of code under tests.  There are useful techniques which are underused \u2014 expectation\n              tests, coverage marks, externalized tests.  There are not so useful techniques which are over-represented in\n              the discourse: fluent assertions, mocks, BDD.  The key for unlocking many of the above techniques is thinking in\n              terms of data, rather than interfaces or objects.  Corollary: good tests are helpful for design in the large. They\n              help to crystalize the data model your application is built\n              around.    Links  (https://www.destroyallsoftware.com/talks/boundaries) https://www.destroyallsoftware.com/talks/boundaries  (https://www.tedinski.com/2019/03/19/testing-at-the-boundaries.html) https://www.tedinski.com/2019/03/19/testing-at-the-boundaries.html  (https://programmingisterrible.com/post/139222674273/how-to-write-disposable-code-in-large-systems) https://programmingisterrible.com/post/139222674273/how-to-write-disposable-code-in-large-systems  (https://sans-io.readthedocs.io) https://sans-io.readthedocs.io  (https://peter.bourgon.org/blog/2021/04/02/dont-use-build-tags-for-integration-tests.html) https://peter.bourgon.org/blog/2021/04/02/dont-use-build-tags-for-integration-tests.html  (https://buttondown.email/hillelwayne/archive/cross-branch-testing/) https://buttondown.email/hillelwayne/archive/cross-branch-testing/  (https://blog.janestreet.com/testing-with-expectations/) https://blog.janestreet.com/testing-with-expectations/  (https://blog.janestreet.com/using-ascii-waveforms-to-test-hardware-designs/) https://blog.janestreet.com/using-ascii-waveforms-to-test-hardware-designs/  (https://ferrous-systems.com/blog/coverage-marks/) https://ferrous-systems.com/blog/coverage-marks/  (https://vorpus.org/blog/notes-on-structured-concurrency-or-go-statement-considered-harmful/) https://vorpus.org/blog/notes-on-structured-concurrency-or-go-statement-considered-harmful/  (https://graydon2.dreamwidth.org/1597.html) https://graydon2.dreamwidth.org/1597.html  (https://bors.tech) https://bors.tech  (https://fsharpforfunandprofit.com/posts/property-based-testing/) https://fsharpforfunandprofit.com/posts/property-based-testing/  (https://fsharpforfunandprofit.com/posts/property-based-testing-1/) https://fsharpforfunandprofit.com/posts/property-based-testing-1/  (https://fsharpforfunandprofit.com/posts/property-based-testing-2/) https://fsharpforfunandprofit.com/posts/property-based-testing-2/  (https://www.sqlite.org/testing.html) https://www.sqlite.org/testing.html   Somewhat amusingly, after writing this article I\u2019ve learned about an\n            excellent post by Tim Bray which argues for the opposite point:  (https://www.tbray.org/ongoing/When/202x/2021/05/15/Testing-in-2021) https://www.tbray.org/ongoing/When/202x/2021/05/15/Testing-in-2021  (/assets/icons.svg#info)   This post is a part of (https://matklad.github.io/2021/09/05/Rust100k.html) \u201cOne Hundred Thousand Lines of Rust\u201d .       (https://github.com/matklad/matklad.github.io/edit/master/content/posts/2021-05-31-how-to-test.dj) Fix typo (/feed.xml) (/assets/icons.svg#rss)   Subscribe (/) All posts (mailto:aleksey.kladov+blog@gmail.com) (/assets/icons.svg#email)   Get in touch (https://github.com/matklad) (/assets/icons.svg#github)   matklad     "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:36:32.803575+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://matklad.github.io/2021/05/31/how-to-test.html"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2026-05-17T19:36:41.746737+00:00",
                "index_texts": [],
                "output": "ArchiveError: Failed to save media",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:36:35.474908+00:00",
                "status": "failed"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://matklad.github.io/2021/05/31/how-to-test.html"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2026-05-17T19:36:32.746857+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:36:28.793683+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpa8abahpb",
                    "https://matklad.github.io/2021/05/31/how-to-test.html"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2026-05-17T19:36:16.295335+00:00",
                "index_texts": [
                    "How to Test\n          May 31, 2021\n        \n        \n          Alternative titles:\n          \u00a0\u00a0\u00a0\u00a0 Unit Tests are a Scam\n          \u00a0\u00a0\u00a0\u00a0 Test Features, Not Code\n          \u00a0\u00a0\u00a0\u00a0 Data Driven Integrated\n            Tests\n        \n        \n          This post describes my current approach to testing. When I started\n          programming professionally, I knew how to write good code, but good\n          tests remained a mystery for a long time. This is not due to the lack\n          of advice \u2014 on the contrary, there\u2019s abundance of information &\n          terminology about testing. This celestial emporium of benevolent\n          knowledge includes TDD, BDD, unit tests, integrated tests, integration\n          tests, end-to-end tests, functional tests, non-functional tests,\n          blackbox tests, glassbox tests, \u2026\n        \n        \n          Knowing all this didn\u2019t help me to create better software. What did\n          help was trying out different testing approaches myself, and looking\n          at how other people write tests. Keep in mind that my background is\n          mostly in writing compiler front-ends for IDEs. This is a rather niche area, which is\n          especially amendable to testing. Compilers are pure self-contained\n          functions. I don\u2019t know how to best test modern HTTP applications\n          built around inter-process communication.\n        \n        Without further ado, let\u2019s see what I have learned!\n        \n          Further ado(2024-05-21): while\n          writing this post, I was missing a key piece of terminology for\n          crisply describing various kinds of tests. If you like this post, you\n          might want to read\n          Unit and Integration Tests\n          . That post supplies better vocabulary for talking about phenomena\n          described in the present article.\n        \n        \n          \n            Test Driven Design Ossification\n          \n          \n            This is something I inflicted upon myself early in my career, and\n            something I routinely observe. You want to refactor some code, say\n            add a new function parameter. Turns out, there are a dozen of tests\n            calling this function, so now a simple refactor also involves fixing\n            all the tests.\n          \n          \n            There is a simple, mechanical fix to this problem: introduce the\n            check function which encapsulates API under test. It\u2019s\n            easier to explain using a toy example. Let\u2019s look at testing\n            something simple, like a binary search, just to illustrate the\n            technique.\n          \n          We start with direct testing:\n\n          \n            /// Given a *sorted* `haystack`, returns `true`\n/// if it contains the `needle`.\nfn binary_search(haystack: &[T], needle: &T) -> bool {\n    ...\n}\n\n#[test]\nfn binary_search_empty() {\n  let res = binary_search(&[], &0);\n  assert_eq!(res, false);\n}\n\n#[test]\nfn binary_search_singleton() {\n  let res = binary_search(&[92], &0);\n  assert_eq!(res, false);\n\n  let res = binary_search(&[92], &92);\n  assert_eq!(res, true);\n\n  let res = binary_search(&[92], &100);\n  assert_eq!(res, false);\n}\n\n// And a dozen more of other similar tests...\n          \n          \n            Some time passes, and we realize that -> bool is not\n            the best signature for binary search. It\u2019s better if it returned an\n            insertion point (an index where element should be inserted to\n            maintain sortedness). That is, we want to change the signature to\n          \n\n          \n            fn binary_search(haystack: &[T], needle: &T) -> Result<usize, usize>;\n          \n          \n            Now we have to change every test, because the tests are tightly\n            coupled to the specific API.\n          \n          \n            My solution to this problem is making the tests data driven. Instead\n            of every test interacting with the API directly, I like to define a\n            single check function which calls the API. This\n            function takes a pair of input and expected result. For binary\n            search example, it will look like this:\n          \n\n          \n            #[track_caller]\nfn check(\n  input_haystack: &[i32],\n  input_needle: i32,\n  expected_result: bool,\n) {\n  let actual_result =\n    binary_search(input_haystack, &input_needle);\n  assert_eq!(expected_result, actual_result);\n}\n\n#[test]\nfn binary_search_empty() {\n  check(&[], 0, false);\n}\n\n#[test]\nfn binary_search_singleton() {\n  check(&[92], 0, false);\n  check(&[92], 92, true);\n  check(&[92], 100, false);\n}\n          \n          \n            Now, when the API of the binary_search function\n            changes, we only need to adjust the single place \u2014 check function:\n          \n\n          \n            #[track_caller]\nfn check(\n  input_haystack: &[i32],\n  input_needle: i32,\n  expected_result: bool,\n) {\n  let actual_result =\n    binary_search(input_haystack, &input_needle).is_ok();\n  assert_eq!(expected_result, actual_result);\n}\n          \n          \n            To be clear, after you\u2019ve done the refactor, you\u2019ll need to adjust\n            the tests to check the index as well, but this can be done\n            separately. Existing test suite does not impede changes.\n          \n\n          \n          \n            Keep in mind that the binary search example is artificially simple.\n            The main danger here is that this is a boiling frog type of situation. While the project is small and\n            the tests are few, you don\u2019t notice that refactors are ever so\n            slightly longer than necessary. Then, several tens of thousands\n            lines of code later, you realize that to make a simple change you\n            need to fix a hundred tests.\n          \n        \n        \n          Test Friction\n          \n            Almost no one likes to write tests. I\u2019ve noticed many times how,\n            upon fixing a trivial bug, I am prone to skipping the testing work.\n            Specifically, if writing a test is more effort than the fix itself,\n            testing tends to go out of the window. Hence,\n          \n\n          \n          \n            Coming back to the binary search example, note how check function reduces the amount of typing to add a new\n            test. For tests, this is a significant saving, not because typing is\n            hard, but because it lowers the cognitive barrier to actually do the\n            work.\n          \n        \n        \n          Test Features, Not Code\n          \n            The over-simplified binary search example can be stretched further.\n            What if you replace the sorted array with a hash map inside your\n            application? Or what if the calling code no longer needs to search\n            at all, and wants to process all of the elements instead?\n          \n          \n            Good code is easy to delete. Tests represent an investment into existing\n            code, and make it costlier to delete (or change).\n          \n          \n            The solution is to write tests for features in such a way that they\n            are independent of the code. I like to use the neural network test\n            for this:\n          \n          \n            Neural Network Test\n            \n              \n                Can you re-use the test suite if your entire software is\n                replaced with an opaque neural network?\n              \n            \n          \n          \n            To give a real-life example this time, suppose that you are writing\n            that part of code-completion engine which sorts potential\n            completions according to relevance. (something I should probably be\n            doing right now, instead of writing this article :-) )\n          \n          \n            Internally, you have a bunch of functions that compute relevance\n            facts, like:\n          \n          \n            \n              Is there direct type match (.foo has the desired\n              type)?\n            \n            \n              Is there an indirect type match (.foo.bar has the\n              right type)?\n            \n            \n              How frequently is this completion used in the current module?\n            \n          \n          \n            Then, there\u2019s the final ranking function that takes these facts and\n            comes up with an overall rank.\n          \n          \n            The classical unit-test approach here would be to write a bunch of\n            isolated tests for each of the relevance functions, and a separate\n            bunch of tests which feeds the ranking function a list of relevance\n            facts and checks the final score.\n          \n          This approach obviously fails the neural network test.\n          \n            An alternative approach is to write a test to check that at a given\n            position a specific ordered list of entries is returned. That suite\n            could work as a cross-validation for an ML-based implementation.\n          \n          \n            In practice, it\u2019s unlikely (but not impossible), that we use actual\n            ML here. But it\u2019s highly probably that the naive independent weights\n            model isn\u2019t the end of the story. At some point there will be\n            special cases which would necessitate change of the interface.\n          \n\n          \n          \n            Note that this advice goes directly against one common understanding\n            of unit-testing. I am fairly confident that it results in better\n            software over the long run.\n          \n        \n        \n          Make Tests Fast\n          \n            There\u2019s one talk about software engineering, which stands out for\n            me, and which is my favorite. It is Boundaries by Gary Bernhardt. There\u2019s a point there though,\n            which I strongly disagree with:\n          \n          \n            Integration Tests are Superlinear?\n            \n              \n                When you use integration tests, any new feature is accompanied\n                by a bit of new code and a new test. However, new code slows\n                down all other tests, so the the overall test suite becomes\n                slow, as the total time grows super-linearly.\n              \n            \n          \n          \n            I don\u2019t think more code under test translates to slower test suite.\n            Merge sort spends more lines of code than bubble sort, but it is way\n            faster.\n          \n          \n            In the abstract, yes, more code generally means more execution time,\n            but I doubt this is the defining factor in tests execution time.\n            What actually happens is usually:\n          \n          \n            \n              Input/Output \u2014 reading just a bit from a disk, network or another\n              process slows down the tests significantly.\n            \n            \n              Outliers \u2014 very often, testing time is dominated by only a couple\n              of slow tests.\n            \n            \n              Overly large input \u2014 throwing enough data at any software makes it\n              slow.\n            \n          \n          \n            The problem with integrated tests is not code volume per se, but the\n            fact that they typically mean doing a lot of IO. But this\n            doesn\u2019t need to be the case\n          \n\n          \n          \n            Nonetheless, some tests are going to be slow. It pays off to\n            introduce the concept of slow tests early on, arrange the skipping\n            of such tests by default and only exercise them on CI. You don\u2019t\n            need to be fancy, just checking an environment variable at the start\n            of the test is perfectly fine:\n          \n\n          \n            #[test]\nfn completion_works_with_real_standard_library() {\n  if std::env::var(\"RUN_SLOW_TESTS\").is_err() {\n    return;\n  }\n  ...\n}\n          \n          \n            Definitely do not use conditional compilation to hide slow\n            tests \u2014 it\u2019s an obvious solution which makes your life harder (similar observation from the Go ecosystem).\n          \n          \n            To deal with outliers, print each test\u2019s execution time by default.\n            Having the numbers fly by gives you immediate feedback and incentive\n            to improve.\n          \n        \n        \n          Data Driven Testing\n          \n            All these together lead to a particular style of architecture and\n            tests, which I call data driven testing. The bulk of the software is\n            a pure function, where the state is passed in explicitly. Removing\n            IO from the picture necessitates that the interface of software is\n            specified in terms of data. Value in, value out.\n          \n          \n            One property of data is that it can be serialized and deserialized.\n            That means that the check style tests can easily accept\n            arbitrary complex input, which is specified in a structured format\n            (JSON), ad-hoc plain text format, or via embedded DSL (builder-style\n            interface for data objects).\n          \n          \n            Similarly, The \u201cexpected\u201d argument of check is data. It\n            is a result which is more-or-less directly displayed to the user.\n          \n          \n            A convincing example of a data driven test would be a \u201cGoto\n            Definition\u201d tests from rust-analyzer (source):\n          \n\n          \n            \n          \n          \n            In this case, the check function has only a single\n            argument \u2014 a string which specifies both the input and the expected\n            result. The input is a rust project with three files (//-\n              /file.rs syntax shows the boundary between the files). The\n            current cursor position is also a part of the input and is specified\n            with the $0 syntax. The result is the //^^^ comment which marks the target of the \u201cGoto\n            Definition\u201d call. The check function creates an\n            in-memory Rust project, invokes \u201cGoto Definition\u201d at the position\n            signified by $0, and checks that the result is the\n            position marked with ^^^.\n          \n          \n            Note that this is decidedly not a unit test. Nothing is stubbed or\n            mocked. This test invokes the whole compilation pipeline: virtual\n            file system, parser, macro expander, name resolution. It runs on top\n            of our incremental computation engine. It touches a significant\n            fraction of the IDE APIs. Yet, it takes 4ms in debug mode (and 500\u00b5s\n            in release mode). And note that it absolutely does not depend on any\n            internal API \u2014 if we replace our dumb compiler with sufficiently\n            smart neural net, nothing needs to be adjusted in the tests.\n          \n          \n            There\u2019s one question though: why on earth am I using a png image to\n            display a bit of code? Only to show that the raw string literal\n            (r#\"\"#) which contains Rust code is highlighted as\n            such. This is possible because we re-use the same input format (with\n            //-, $0 and couple of other markup\n            elements) for almost every test in rust-analyzer. As such, we can\n            invest effort into building cool things on top of this format, which\n            subsequently benefit all our tests.\n          \n        \n        \n          Expect Tests\n          \n            Previous example had a complex data input, but a relatively simple\n            data output \u2014 a position in the file. Often, the output is messy and\n            has a complicated structure as well (a symptom of rho problem). Worse, sometimes the output is a part that is\n            changed frequently. This often necessitates updating a lot of tests.\n            Going back to the binary search example, the change from ->\n              bool to -> Result<usize, usize> was\n            an example of this effect.\n          \n          \n            There is a technique to make such simultaneous changes to all gold\n            outputs easy \u2014 testing with expectations. You specify the expected\n            result as a bit of data inline with the test. There\u2019s a special mode\n            of running the test suite for updating this data. Instead of failing\n            the test, a mismatch between expected and actual causes the gold\n            value to be updated in-place. That is, the test framework edits the\n            code of the test itself.\n          \n          \n            Here\u2019s an example of this workflow in rust-analyzer, used for\n            testing code completion:\n          \n\n          \n            \n            \n          \n          \n            Often, just Debug representation of the type works well\n            for expect tests, but you can do something more fun. See this post\n            from Jane Street for a great example:\n            Using ASCII waveforms to test hardware designs.\n          \n          \n            There are several libraries for this in Rust: insta, k9, expect-test.\n          \n        \n        \n          Fluent Assertions\n          \n            An extremely popular genre for a testing library is a collection of\n            fluent assertions:\n          \n\n          \n            // Built-in assertion:\nassert!(x > y);\n\n// Fluent assertion:\nassert_that(x).is_greater_than(y);\n          \n          \n            The benefit of this style are better error messages. Instead of just\n            \u201cfalse is not true\u201d, the testing framework can print values for\n            x and y.\n          \n          \n            I don\u2019t find this useful. Using the check style\n            testing, there are very few assertions actually written in code.\n            Usually, I start with plain asserts without messages. The first time\n            I debug an actual test failure for a particular function, I spend\n            some time to write a detailed assertion message. To me, fluent\n            assertions are not an attractive point on the curve that includes\n            plain asserts and hand-written, context aware explanations of\n            failures. A notable exception here is pytest approach \u2014 this testing\n            framework overrides the standard assert to provide a\n            rich diff without ceremony.\n          \n\n          \n        \n        \n          Peeking Inside\n          \n            One apparent limitation of the style of integrated testing I am\n            describing is checking for properties which are not part of\n            the output. For example, if some kind of caching is involved, you\n            might want to check that the cache is actually being hit, and is not\n            just sitting there. But, by definition, cache is not something that\n            an outside client can observe.\n          \n          \n            The solution to this problem is to make this extra data a part of\n            the system\u2019s output by adding extra observability points. A good\n            example here is Cargo\u2019s test suite. It is set-up in an integrated,\n            data-driven fashion. Each tests starts with a succinct DSL for\n            setting up a tree of files on disk. Then, a full cargo command is\n            invoked. Finally, the test looks at the command\u2019s output and the\n            resulting state of the file system, and asserts the relevant facts.\n          \n          \n            Tests for caching additionally enable verbose internal logging. In\n            this mode, Cargo prints information about cache hits and misses.\n            These messages are then used in assertions.\n          \n          \n            A close idea is coverage marks. Some times, you want to check that something\n            does not happen. Tests for this tend to be fragile\n            \u2014 often the thing does not happen, but for the wrong reason. You can\n            add a side channel which explains the reasoning behind particular\n            behavior, and additionally assert this as well.\n          \n        \n        \n          Externalized Tests\n          \n            In the ultimate stage of data driven tests the definitions of test\n            cases are moved out of test functions and into external files. That\n            is, you don\u2019t do this:\n          \n\n          \n            #[test]\nfn test_foo() {\n  check(\"foo\", \"oof\")\n}\n\n#[test]\nfn test_bar() {\n  check(\"bar\", \"rab\")\n}\n          \n          Rather, there is a single test that looks like this:\n\n          \n            #[test]\nfn test_all() {\n  for file in read_dir(\"./test_data/in\") {\n    let input = read_to_string(\n      &format!(\"./test_data/in/{}\", file),\n    );\n    let output = read_to_string(\n      &format!(\"./test_data/out/{}\", file),\n    );\n    check(input, output)\n  }\n}\n          \n          \n            I have a love-hate relationship with this approach. It has at least\n            two attractive properties.\n            First, it forces data driven approach without any cheating.\n            Second, it makes the test suite more re-usable. An\n            alternative implementation in a different programming language can\n            use the same tests.\n          \n          \n            But there\u2019s a drawback as well \u2014 without literal #[test] attributes, integration with tooling suffers. For\n            example, you don\u2019t automatically get \u201cX out of Y tests passed\u201d at\n            the end of test run. You can\u2019t conveniently debug just a single\n            test, there isn\u2019t a helpful \u201cRun\u201d icon/shortcut you can use in an\n            IDE.\n          \n          \n            When I do externalized test cases, I like to leave a trivial smoke\n            test behind:\n          \n\n          \n            #[test]\nfn smoke() {\n  check(\"\", \"\");\n}\n          \n          \n            If I need to debug a failing external test, I first paste the input\n            into this smoke test, and then get my IDE tooling back.\n          \n        \n        \n          \n            Beyond Example Based Testing\n          \n          \n            Reading from a file is not the most fun way to come up with a data\n            input for a check function.\n          \n          Here are a few other popular ones:\n          \n            Property Based Testing\n            \n              \n                Generate the input at random and verify that the output makes\n                sense. For a binary search, check that the needle\n                indeed lies between the two elements at the insertion point.\n              \n            \n            Full Coverage\n            \n              \n                Better still, instead of generating some random inputs, just\n                check that the answer is correct for all inputs. This\n                is how you should be testing binary search \u2014 generate every\n                sorted list of length at most 7 with numbers in the\n                0..=6 range. Then, for each list and for each\n                number, check that the binary search gives the same result as a\n                naive linear search.\n              \n            \n            Coverage Guided Fuzzing\n            \n              \n                Just throw random bytes at the check function. Random bytes\n                probably don\u2019t make much sense, but it\u2019s good to verify that the\n                program returns an error instead of summoning nasal demons.\n                Instead of piling bytes completely at random, observe which\n                branches are taken, and try to invent byte sequences which cover\n                more branches. Note that this test is polymorphic in the system\n                under test.\n              \n            \n            Structured Fuzzing / Coverage Guided Property Testing\n            \n              \n                Use random bytes as a seed to generate \u201csyntactically valid\u201d\n                inputs, then see you software crash and burn when the most\n                hideous edge cases are uncovered. If you use Rust, check out wasm-smith and arbitrary crates.\n              \n            \n          \n\n          \n        \n        \n          The External World\n          \n            What if isolating IO is not possible, and the application is\n            fundamentally build around interacting with external systems? In\n            this case, my advice is to just accept that the tests are going to\n            be slow, and might need extra effort to avoid flakiness.\n          \n          \n            Cargo is the perfect case study here. Its raison d\u2019\u00eatre is\n            orchestrating a herd of external processes. Let\u2019s look at the basic\n            test:\n          \n\n          \n            #[test]\nfn cargo_compile_simple() {\n  let p = project()\n    .file(\"Cargo.toml\", &basic_bin_manifest(\"foo\"))\n    .file(\"src/foo.rs\", &main_file(r#\"\"i am foo\"\"#, &[]))\n    .build();\n\n  p.cargo(\"build\").run();\n\n  assert!(p.bin(\"foo\").is_file());\n  p.process(&p.bin(\"foo\")).with_stdout(\"i am foo\\n\").run();\n}\n          \n          \n            The project() part is a builder, which describes the\n            state of the a system.\n            First, .build() writes the specified files to\n            a disk in a temporary directory.\n            Then, p.cargo(\"build\").run() executes the real\n            cargo build command.\n            Finally, a bunch of assertions are made about the end state\n            of the file system.\n          \n          \n            Neural network test: this is completely independent of internal\n            Cargo APIs, by virtue of interacting with a cargo\n            process via IPC.\n          \n          \n            To give an order-of-magnitude feeling for the cost of IO, Cargo\u2019s\n            test suite takes around seven minutes (-j 1), while\n            rust-analyzer finishes in less than half a minute.\n          \n          \n            An interesting case is the middle ground, when the IO-ing part is\n            just big and important enough to be annoying. That is the case for\n            rust-analyzer \u2014 although almost all code is pure, there\u2019s a part\n            which interacts with a specific editor. What makes this especially\n            finicky is that, in the case of Cargo, it\u2019s Cargo who calls external\n            processes. With rust-analyzer, it\u2019s something which we don\u2019t\n            control, the editor, which schedules the IO. This often results in\n            hard-to-imagine bugs which are caused by particularly weird\n            environments.\n          \n          I don\u2019t have good answers here, but here are the tricks I use:\n          \n            \n              Accept that something will break during integration. Even\n              if you always create perfect code and never make bugs,\n              your upstream integration point will be buggy sometimes.\n            \n            \n              Make integration bugs less costly:\n              \n                \n                  use release trains,\n                \n                \n                  make patch release process non-exceptional and easy,\n                \n                \n                  have a checklist for manual QA before the release.\n                \n              \n            \n            \n              Separate the tricky to test bits into a separate project. This\n              allows you to write slow and not 100% reliable tests for\n              integration parts, while keeping the core test suite fast and\n              dependable.\n            \n          \n\n          \n        \n        \n          The Concurrent World\n          Consider the following API:\n\n          \n            fn do_stuff_in_background(p: Param) {\n  std::thread::spawn(move || {\n    // Stuff\n  })\n}\n          \n          \n            This API is fundamentally untestable. Can you see why? It spawns a\n            concurrent computation, but it doesn\u2019t allow waiting for this\n            computation to be finished. So, any test that calls do_stuff_in_background can\u2019t check that the \u201cStuff\u201d is done.\n            Worse, even tests which do not call this function might start to\n            fail \u2014 they now can get interference from other tests. The\n            concurrent computation can outlive the test that originally spawned\n            it.\n          \n          \n            This problem plagues almost every concurrent application I see. A\n            common symptom is adding timeouts and sleeps to test, to increase\n            the probability of stuff getting done. Such timeouts are another\n            common cause of slow test suites.\n          \n          \n            What makes this problem truly insidious is that there\u2019s no\n            work-around. Broken once, causality link is not reforgable by a\n            layer above.\n          \n          The solution is simple: don\u2019t do this.\n\n          \n        \n        \n          Layers\n          \n            Another common problem I see in complex projects is a beautifully\n            layered architecture, which is \u201cinverted\u201d in tests.\n          \n          \n            Let\u2019s say you have something fabulous, like L1 <- L2 <-\n              L3 <- L4. To test L1, the path of least\n            resistance is often to write tests which exercise L4.\n            You might even think that this is the setup I am advocating for. Not\n            exactly.\n          \n          \n            The problem with L1 <- L2 <- L3 <- L4 <-\n              Tests is that working on L1 becomes slower,\n            especially in compiled languages. If you make a change to L1, then, before you get to the tests, you need to recompile\n            the whole chain of reverse dependencies. My \u201cfavorite\u201d example here\n            is rustc \u2014 when I worked on the lexer (T1), I spent a lot of time waiting for the rest of the\n            compiler to be rebuild to check my small change.\n          \n          \n            The right setup here is to write integrated tests for each layer:\n          \n\n          \n            L1 <- Tests\nL1 <- L2 <- Tests\nL1 <- L2 <- L3 <- Tests\nL1 <- L2 <- L3 <- L4 <- Tests\n          \n          \n            Note that testing L4 involves testing L1,\n            L2 an L3. This is not a problem. Due to\n            layering, only L4 needs to be recompiled.\n            Other layers don\u2019t affect run time meaningfully. Remember \u2014\n            it\u2019s IO (and sleep-based synchronization) that kills performance,\n            not just code volume.\n          \n        \n        \n          Test Everything\n          \n            In a nutshell, a #[test] is just a bit of code which is\n            plugged into the build system to be executed automatically. Use this\n            to your advantage, simplify the automation by moving as much as\n            possible into tests.\n          \n          \n            Here\u2019s some things in rust-analyzer which are just\n            tests:\n          \n          \n            \n              Code formatting (most common one \u2014 you don\u2019t need an extra pile of\n              YAML in CI, you can shell out to the formatter from the test).\n            \n            \n              Checking that the history does not contain merge commits and\n              teaching new contributors git survival skills.\n            \n            \n              Collecting the manual from specially-formatted doc comments across\n              the code base.\n            \n            \n              Checking that the code base is, in fact, reasonably\n              well-documented.\n            \n            \n              Ensuring that the licenses of dependencies are compatible.\n            \n            \n              Ensuring that high-level operations are linear in the size of the\n              input. Syntax-highlight a synthetic file of 1, 2, 4, 8, 16\n              kilobytes, run linear regression, check that result looks like a\n              line rather than a parabola.\n            \n          \n        \n        \n          Use Bors\n          \n            This essay already mentioned a couple of cognitive tricks for better\n            testing: reducing the fixed costs for adding new tests, and\n            plotting/printing test times. The best trick in a similar vein is\n            the \u201cnot rocket science\u201d rule of software engineering.\n          \n          \n            The idea is to have a robot which checks that the merge\n                commit passes all the tests, before advancing the\n            state of the main branch.\n          \n          \n            Besides the evergreen master, such bot adds pressure to keep the\n            test suite fast and non-flaky. This is another boiling frog,\n            something you need to constantly keep an eye on. If you have any a\n            single flaky test, it\u2019s very easy to miss when the second one is\n            added.\n          \n\n          \n        \n        \n          Recap\n          \n            This was a long essay. Let\u2019s look back at some of the key points:\n          \n          \n            \n              There is a lot of information about testing, but it is not always\n              helpful. At least, it was not helpful for me.\n            \n            \n              The core characteristic of the test suite is how easy it makes\n              changing the software under test.\n            \n            \n              To that end, a good strategy is to focus on testing the features\n              of the application, rather than on testing the code used to\n              implement those features.\n            \n            \n              A good test suite passes the neural network test \u2014 it is still\n              useful if the entire application is replaced by an ML model which\n              just comes up with the right answer.\n            \n            \n              Corollary: good tests are not helpful for design in the small \u2014 a\n              good test won\u2019t tell you the best signatures for functions.\n            \n            \n              Testing time is something worth optimizing for. Tests are\n              sensitive to IO and IPC. Tests are relatively insensitive to the\n              amount of code under tests.\n            \n            \n              There are useful techniques which are underused \u2014 expectation\n              tests, coverage marks, externalized tests.\n            \n            \n              There are not so useful techniques which are over-represented in\n              the discourse: fluent assertions, mocks, BDD.\n            \n            \n              The key for unlocking many of the above techniques is thinking in\n              terms of data, rather than interfaces or objects.\n            \n            \n              Corollary: good tests are helpful for design in the large. They\n              help to crystalize the data model your application is built\n              around.\n            \n          \n        \n        \n          Links\n          \n            \n              https://www.destroyallsoftware.com/talks/boundaries\n            \n            \n              https://www.tedinski.com/2019/03/19/testing-at-the-boundaries.html\n            \n            \n              https://programmingisterrible.com/post/139222674273/how-to-write-disposable-code-in-large-systems\n            \n            \n              https://sans-io.readthedocs.io\n            \n            \n              https://peter.bourgon.org/blog/2021/04/02/dont-use-build-tags-for-integration-tests.html\n            \n            \n              https://buttondown.email/hillelwayne/archive/cross-branch-testing/\n            \n            \n              https://blog.janestreet.com/testing-with-expectations/\n            \n            \n              https://blog.janestreet.com/using-ascii-waveforms-to-test-hardware-designs/\n            \n            \n              https://ferrous-systems.com/blog/coverage-marks/\n            \n            \n              https://vorpus.org/blog/notes-on-structured-concurrency-or-go-statement-considered-harmful/\n            \n            \n              https://graydon2.dreamwidth.org/1597.html\n            \n            \n              https://bors.tech\n            \n            \n              https://fsharpforfunandprofit.com/posts/property-based-testing/\n            \n            \n              https://fsharpforfunandprofit.com/posts/property-based-testing-1/\n            \n            \n              https://fsharpforfunandprofit.com/posts/property-based-testing-2/\n            \n            \n              https://www.sqlite.org/testing.html\n            \n          \n          \n            Somewhat amusingly, after writing this article I\u2019ve learned about an\n            excellent post by Tim Bray which argues for the opposite point:\n          \n          \n            https://www.tbray.org/ongoing/When/202x/2021/05/15/Testing-in-2021"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:36:10.353400+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://matklad.github.io/2021/05/31/how-to-test.html"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-05-17T19:36:09.224308+00:00",
                "index_texts": null,
                "output": "How to Test",
                "pwd": "/data/archive/1779046538.567373",
                "schema": "ArchiveResult",
                "start_ts": "2026-05-17T19:36:09.188861+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "ArchiveError: Failed to find \"content-location\" URL header in Archive.org response.",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "ArchiveError: Failed to save media",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "How to Test",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1779046538.567373",
    "newest_archive_date": "2026-05-17T19:36:41.902323+00:00",
    "num_failures": 2,
    "num_outputs": 7,
    "oldest_archive_date": "2026-05-17T19:35:44.766274+00:00",
    "path": "/2021/05/31/how-to-test.html",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KRVPZQBZ347CC436011QK4DV",
    "snapshot_id": "c4ec8a39-451f-45dd-ac73-140e837991bb",
    "sources": [
        "/data/sources/1779046537-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1779046538.567373",
    "title": "How to Test",
    "url": "https://matklad.github.io/2021/05/31/how-to-test.html"
}