{
    "archive_path": "archive/1766192981.241534",
    "base_url": "davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html",
    "basename": "rustforge-wild-performance-tricks.html",
    "bookmarked_date": "2025-12-20 01:09",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=davidlattimore.github.io",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "davidlattimore.github.io",
    "downloaded_at": "2025-12-20T01:09:46.872989+00:00",
    "downloaded_datestr": "2025-12-20 01:09",
    "extension": "html",
    "hash": "CMEC47B1Z1HX45E25S57",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-20T01:11:16.176250+00:00",
                "index_texts": null,
                "output": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html']' timed out after 60 seconds",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:10:16.079910+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2025-12-20T01:09:57.425789+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:09:51.124126+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=davidlattimore.github.io"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-20T01:09:50.923675+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:09:46.912625+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-20T01:09:50.992404+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:09:50.944632+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2025-12-20T01:10:11.130400+00:00",
                "index_texts": [
                    "Wild Performance Tricks | David Lattimore (/css/common.css) (/css/monokai.sublime.css) (David Lattimore's Blog) (/feed.xml)  Wild Performance Tricks   (/) Home  (/about) About    David Lattimore - 2025-09-02  Last week I had the pleasure of attending RustForge in Wellington, New Zealand. I gave a talk titled\n\u201cWild performance tricks\u201d. You can watch a (https://www.youtube.com/live/6Scgq9fBZQM?t=9246s) recording of my\ntalk . If you\u2019d prefer to read rather than watch,\nthe rest of this post will cover more or less the same material. The talk shows some linker\nbenchmarks, which I\u2019ll skip here and focus instead on the optimisations, which I think are the more\ninteresting part of the talk. The tricks here are a few of my favourites that I\u2019ve used in the (https://github.com/davidlattimore/wild) Wild\nlinker . Mutable slicing for sharing between threads In the linker, we have a type SymbolId defined as: struct SymbolId ( u32 );     We need a way to store resolutions, where one SymbolId resolves (maps) to another SymbolId . If\nwe need to look up which symbol SymbolId(5) maps to, we then look at index 5 in the Vec .\nBecause every symbol maps to some other symbol (possibly itself), this means that we make use of the\nentire Vec . i.e. it\u2019s dense, not sparse. For a sparse mapping, a HashMap might be preferable. The Wild linker is very multi-threaded, so we want to be able to process symbols for our input\nobjects in parallel. To achieve this, we make sure that all symbols for a given object get allocated\nadjacent to each other. i.e. each object has SymbolId s in a contiguous range. This is good for\ncache locality because when a thread is working with an object, all its symbols will be nearby in\nmemory, so more likely to be in cache. It also lets us do things like this: fn parallel_process_resolutions ( mut resolutions : & mut [ SymbolId ], objects : & [ Object ]) { objects .iter () .map (| obj | ( obj , resolutions .split_off_mut ( .. obj .num_symbols ) .unwrap ())) .par_bridge () .for_each (|( obj , object_resolutions )| { obj .process_resolutions ( object_resolutions ); }); }     Here, we\u2019re using the Rayon crate to process the resolutions for all our objects in parallel from\nmultiple threads. We start by iterating over our objects, then for each object, we use split_off_mut to split off a mutable slice of resolutions that contains the resolutions for that\nobject. par_bridge converts this regular Rust iterator into a Rayon parallel iterator. The closure\npassed to for_each then runs in parallel on multiple threads, with each thread getting access to\nthe object and a mutable slice of that object\u2019s resolutions. Parallel initialisation of the Vec The previous technique of using split_off_mut to get multiple non-overlapping mutable slices of\nour Vec relies on the Vec having already been initialised. We\u2019d like to initialise our Vec in\nparallel, otherwise we\u2019d have to wait for the main thread to fill the entire Vec with a placeholder\nvalue only to then have our threads overwrite those placeholder values. To do this, we can use the sharded-vec-writer crate, which was created for use in Wild, but which can be used for similar\npurposes elsewhere. First, we create a Vec with sufficient capacity to store the resolutions for all our symbols: let mut resolutions : Vec < SymbolId > = Vec :: with_capacity ( total_num_symbols );     At this point, we\u2019ve allocated space on the heap for the Vec, but that space is still uninitialised.\ni.e. the length is still zero. Next, we create a VecWriter , which mutably borrows the Vec, then split that writer into shards,\nwith each shard having a size equal to the number of symbols in the corresponding object. let mut writer = VecWriter :: new ( & mut resolutions ); let mut shards = writer .take_shards ( objects .iter () .map (| o | o .num_symbols ));     We can now, in parallel, iterate through our objects and their corresponding shards and initialise\nthe shards. objects .par_iter () .zip_eq ( & mut shards ) .for_each (|( obj , shard )| { for symbol in obj .symbols () { shard .push ( ... ); } });     Lastly, we return the shards to the writer, which verifies that all the shards were fully\ninitialised, thus resizing the Vec, after which it can be used normally. writer .return_shards ( shards );     Atomic - non-atomic in-place conversion Most parts of the linker can make do with either exclusive access to part of the resolutions Vec,\nor shared access to the entire Vec. However, there\u2019s one part of the linker where we need to perform\nrandom writes to the resolutions Vec. This is done when we have multiple symbol definitions with\nthe same name. Originally, I just did this work from the main thread, since I figured most of the\ntime there would only be a small number of symbols that had the same name. This was mostly true,\nhowever for large C++ binaries like Chromium, it turns out that there are actually a lot of symbols\nwith the same names, presumably due to C++\u2019s use of header files, which create lots of identical\ndefinitions. To allow random writes to resolutions , we introduce a new type: struct AtomicSymbolId ( AtomicU32 );     Being an atomic, we can write to an AtomicSymbolId using only a shared (non-exclusive) reference.\nHowever, we need a way to temporarily view our Vec<SymbolId> as a &[AtomicSymbolId] . The standard library has something that might help - AtomicU32::from_mut_slice : fn from_mut_slice ( v : & mut [ u32 ]) -> & mut [ AtomicU32 ]     However, it\u2019s unstable (nightly only). Even if it were stable, it only works with slices of\nprimitive types, so we\u2019d have to lose our newtypes (SymbolId etc). Another option would be to always use atomics, however that would quite possibly hurt performance of\nthe rest of the linker, which doesn\u2019t need atomics. It\u2019d also hurt ergonomics, since currently our SymbolId s implement Copy , but if they wrapped an AtomicU32 , then they wouldn\u2019t be able to. A reasonable option at this point would be to resort to unsafe and use something like core::mem::transmute . We\u2019d need to check all the rules and make sure that we were meeting all the\nrequirements. This is not a bad option, but I personally like the challenge of doing things without\nunsafe if I can, especially if I can do so without loss of performance. Indeed, it turns out that we can, as follows: fn into_atomic ( symbols : Vec < SymbolId > ) -> Vec < AtomicSymbolId > { symbols .into_iter () .map (| s | AtomicSymbolId ( AtomicU32 :: new ( s .0 ))) .collect () }     It\u2019d be reasonable to think that this will have a runtime cost, however it doesn\u2019t. The reason is\nthat the Rust standard library has a nice optimisation in it that when we consume a Vec and collect\nthe result into a new Vec, in many circumstances, the heap allocation of the original Vec can be\nreused. This applies in this case. But what even with the heap allocation being reused, we\u2019re still\nlooping over all the elements to transform them right? Because the in-memory representation of an AtomicSymbolId is identical to that of a SymbolId , our loop becomes a no-op and is optimised\naway. We can verify this by looking at the assembly produced for this function: movups xmm0 , xmmword , ptr , [ rsi ] mov rax , qword , ptr , [ rsi , + , 16 ] movups xmmword , ptr , [ rdi ], xmm0 mov qword , ptr , [ rdi , + , 16 ], rax ret     The main takeaway from this assembly is that there\u2019s no branching, no looping, just a few moves and\na return. If we allowed this function to be inlined into the caller, it would likely vanish to\nnothing. For conversion back to the non-atomic form, we can do much the same: fn into_non_atomic ( atomic_symbols : Vec < AtomicSymbolId > ) -> Vec < SymbolId > { atomic_symbols .into_iter () .map (| s | SymbolId ( s .0 .into_inner ())) .collect () }     The main thing to note here is that we avoid doing an atomic load from the atomic and instead\nconsume the atomic with into_inner . This is easier for the compiler to optimise and if we look at\nthe assembly produced it\u2019s identical to what we got for into_atomic . To actually use these functions, we first need to get ownership of our Vec using core::mem::take .\nThis puts an empty Vec in its place. Empty Vecs don\u2019t heap allocate, so this is very cheap. We then\ncall into_atomic to convert the Vec into the form we need. let atomic_resolutions = into_atomic ( core :: mem :: take ( & mut self .resolutions ));     We can then do whatever parallel processing we need with the Vec in its atomic form. process_resolutions_in_parallel ( & atomic_resolutions );     Finally, we convert back to the original non-atomic form and store back where we got it from,\noverwriting the empty Vec that we temporarily put in its place. self .resolutions = into_non_atomic ( atomic_resolutions );     One thing worth noting here is that if we panic (or do an early return), we might leave self.resolutions as the empty Vec. This isn\u2019t a problem in the linker, since if we\u2019re returning an\nerror or have hit a panic, then we don\u2019t care at that point about resolutions. It would be possible\nto ensure that the proper Vec was restored for use-cases where that was important, however it would\nadd extra complexity and might be enough to convince me that it\u2019d be better to just use transmute. Buffer reuse Doing too much heap allocation tends to hurt performance. A common trick is to move heap allocations\noutside of loops. For example, rather than this: loop { let mut buffer = Vec :: new (); // Do work with `buffer`. }     We might prefer to allocate buffer before the loop, then just clear it inside the loop: let mut buffer = Vec :: new (); loop { buffer .clear (); // Do work with `buffer`. }     However, if we\u2019re storing something into a Vec that has a non-static lifetime, then we can run into\nproblems. Here, we have a variable text , which holds a String . We then split that string and\nstore the resulting string-slices into buffer . Even though we clear buffer at the end of the\nloop, the compiler is unhappy. It wants text to outlive buffer because we\u2019re storing references\nto text into buffer . let mut buffer = Vec :: new (); loop { let text = get_text (); buffer .extend ( text .split ( \",\" )); // Do work with `buffer`. buffer .clear (); }     We could at this point give up and just move our Vec creation back inside the loop. However, it\nturns out that there\u2019s another solution. fn reuse_vec < T , U > ( mut v : Vec < T > ) -> Vec < U > { v .clear (); v .into_iter () .map (| x | unreachable! ()) .collect () }     The idea of this function is to convert from a Vec of some time to an empty Vec of another type,\nreusing the heap allocation. This works in a very similar way to how we converted between atomic and\nnon-atomic SymbolId s, except this time because we first clear the Vec, the body of our map function is unreachable. The optimisation in the Rust standard library that allows reuse of the heap allocation will only\nactually work if the size and alignment of T and U are the same, so let\u2019s verify that that\u2019s the\ncase. We can do the check at compile time, so if we accidentally call this function with\nincompatible T and U , we\u2019ll get a compilation error at the call site. fn reuse_vec < T , U > ( mut v : Vec < T > ) -> Vec < U > { const { assert! ( size_of :: < T > () == size_of :: < U > ()); assert! ( align_of :: < T > () == align_of :: < U > ()); } v .clear (); v .into_iter () .map (| _ | unreachable! ()) .collect () }     Let\u2019s verify that this optimises as we expect: mov qword , ptr , [ rsi , + , 16 ], 0 movups xmm0 , xmmword , ptr , [ rsi ] movups xmmword , ptr , [ rdi ], xmm0 mov qword , ptr , [ rdi , + , 16 ], 0 ret     More or less the same assembly as before, except that we\u2019re now setting the length of the Vec to 0.\nNote, that the loop and the panic from the use of unreachable! are gone. We can now integrate this into our previous code as follows: let mut buffer_store : Vec <& str > = Vec :: new (); loop { let mut buffer = reuse_vec ( buffer_store ); let text = get_text (); buffer .extend ( text .split ( \",\" )); // Do work with `buffer`. buffer_store = reuse_vec ( buffer ); }     Effectively, each time around the loop we move out of buffer_store , converting the type of the Vec , use it for a bit, then convert it back and store it again in buffer_store . The only time\nwe\u2019ll need a new heap allocation is when our Vec needs to grow. The types of buffer_store and buffer are both Vec<&str> , however the lifetime of the references is different. Deallocation on a separate thread Freeing memory is generally a lot slower than allocating it. If we\u2019ve done a very large allocation,\nit can sometimes be worthwhile passing it to another thread to free it, so that we can get on with\nother work. For example, if using rayon, we might use rayon::spawn to spawn a task that drops our buffer: fn process_buffer ( buffer : Vec < u8 > ) { // Do some work with `buffer`. rayon :: spawn (|| drop ( buffer )); }     Note, that rayon::spawn itself does a heap allocation, so this would only be worthwhile if buffer was potentially very large. This is definitely something you\u2019d want to benchmark to see if\nit actually improves the runtime for your use-case. There is at least one place in the Wild linker\nwhere we did this and it did give a measurable reduction in runtime. Similar to buffer reuse, if our heap allocation has non-static lifetimes associated with it, we can\nget rid of them using reuse_vec . fn process_buffer ( names : Vec <& [ u8 ] > ) { // Do some work with `names`. let names : Vec <& [ u8 ] > = reuse_vec ( names ); rayon :: spawn (|| drop ( names )); }     In this case, we\u2019re converting the Vec from a Vec<&[u8]> to Vec<&'static [u8]> . Bonus: Strip lifetime with non-trivial Drop This is a bonus tip that wasn\u2019t included in the talk and builds on the previous tip and is in\nresponse to a question by VorpalWay on Reddit. If you want to drop a Vec<T> and T has both a\nnon-static lifetime and a non-trivial Drop , then things get slightly more tricky. The trick here\nis to convert to a struct that is the same as T , but has non-static references replaced with MaybeUninit . For example, suppose we have the following struct: pub struct Foo < 'a > { owned : String , borrowed : & 'a str , }     We can define a new struct: struct StaticFoo { owned : String , borrowed : MaybeUninit <& 'static str > , }     We can then convert our Vec to the new type with zero cost and no unsafe: fn without_lifetime ( foos : Vec < Foo > ) -> Vec < StaticFoo > { foos .into_iter () .map (| f | StaticFoo { owned : f .owned , borrowed : MaybeUninit :: uninit (), }) .collect () }     The presence of MaybeUnit::uninit() tells the compiler that it\u2019s OK to have anything there, so it\ncan choose to leave whatever &str was in the original Foo struct. This means that it\u2019s valid to\nproduce a StaticFoo with the same in-memory representation as the Foo that it replaces, allowing\nit to eliminate the loop. The asm for this function is: movups xmm0 , xmmword , ptr , [ rsi ] mov rax , qword , ptr , [ rsi , + , 16 ] movups xmmword , ptr , [ rdi ], xmm0 mov qword , ptr , [ rdi , + , 16 ], rax ret     i.e. the loop was indeed eliminated. Now that we have a Vec with no non-static lifetimes, we can safely move it to another thread. Thanks Thanks to my (https://github.com/sponsors/davidlattimore) github sponsors . Your contributions help\nto make it possible for me to continue to work on this kind of stuff rather than going and getting a\n\u201creal job\u201d. CodeursenLiberte Urgau pmarks repi embark-studios mati865 bes joshtriplett mstange bcmyers Rafferty97 acshi Kobzol flba-eb jonhoo marxin tommythorn binarybana teburd bearcove yerke teh twilco Shnatsel coastalwhite wezm davidcornu gendx rrbutani nazar-pc willstott101 tatsuya6502 teohhanhui jkendall327 EdorianDark drmason13 HadrienG2 jplatte rukai ymgyt dream-dasher alexkirsz Pratyush Tudyx coreyja dralley irfanghat mvolfik simtheverse  Discussion (https://www.reddit.com/r/rust/comments/1n7814i/wild_performance_tricks/) Reddit     (https://mas.to/@davidlattimore)  (https://github.com/davidlattimore)  (mailto:dvdlttmr@gmail.com)  (/feed.xml)     "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:10:11.050468+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2025-12-20T01:10:16.040724+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:10:11.263703+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2025-12-20T01:10:11.015546+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:10:02.168653+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpillppma2",
                    "https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2025-12-20T01:10:01.568097+00:00",
                "index_texts": [
                    "David Lattimore - 2025-09-02\n  \n  \n\n      Last week I had the pleasure of attending RustForge in Wellington, New Zealand. I gave a talk titled\n\u201cWild performance tricks\u201d. You can watch a recording of my\ntalk. If you\u2019d prefer to read rather than watch,\nthe rest of this post will cover more or less the same material. The talk shows some linker\nbenchmarks, which I\u2019ll skip here and focus instead on the optimisations, which I think are the more\ninteresting part of the talk.\n\nThe tricks here are a few of my favourites that I\u2019ve used in the Wild\nlinker.\n\nMutable slicing for sharing between threads\n\nIn the linker, we have a type SymbolId defined as:\n\n\n\nWe need a way to store resolutions, where one SymbolId resolves (maps) to another SymbolId. If\nwe need to look up which symbol SymbolId(5) maps to, we then look at index 5 in the Vec.\nBecause every symbol maps to some other symbol (possibly itself), this means that we make use of the\nentire Vec. i.e. it\u2019s dense, not sparse. For a sparse mapping, a HashMap might be preferable.\n\nThe Wild linker is very multi-threaded, so we want to be able to process symbols for our input\nobjects in parallel. To achieve this, we make sure that all symbols for a given object get allocated\nadjacent to each other. i.e. each object has SymbolIds in a contiguous range. This is good for\ncache locality because when a thread is working with an object, all its symbols will be nearby in\nmemory, so more likely to be in cache. It also lets us do things like this:\n\nfn parallel_process_resolutions(mut resolutions: &mut [SymbolId], objects: &[Object]) {\n   objects\n       .iter()\n       .map(|obj| (obj, resolutions.split_off_mut(..obj.num_symbols).unwrap()))\n       .par_bridge()\n       .for_each(|(obj, object_resolutions)| {\n           obj.process_resolutions(object_resolutions);\n       });\n}\n\n\nHere, we\u2019re using the Rayon crate to process the resolutions for all our objects in parallel from\nmultiple threads. We start by iterating over our objects, then for each object, we use\nsplit_off_mut to split off a mutable slice of resolutions that contains the resolutions for that\nobject. par_bridge converts this regular Rust iterator into a Rayon parallel iterator. The closure\npassed to for_each then runs in parallel on multiple threads, with each thread getting access to\nthe object and a mutable slice of that object\u2019s resolutions.\n\nParallel initialisation of the Vec\n\nThe previous technique of using split_off_mut to get multiple non-overlapping mutable slices of\nour Vec relies on the Vec having already been initialised. We\u2019d like to initialise our Vec in\nparallel, otherwise we\u2019d have to wait for the main thread to fill the entire Vec with a placeholder\nvalue only to then have our threads overwrite those placeholder values. To do this, we can use the\nsharded-vec-writer crate, which was created for use in Wild, but which can be used for similar\npurposes elsewhere.\n\nFirst, we create a Vec with sufficient capacity to store the resolutions for all our symbols:\n\nlet mut resolutions: Vec<SymbolId> = Vec::with_capacity(total_num_symbols);\n\n\nAt this point, we\u2019ve allocated space on the heap for the Vec, but that space is still uninitialised.\ni.e. the length is still zero.\n\nNext, we create a VecWriter, which mutably borrows the Vec, then split that writer into shards,\nwith each shard having a size equal to the number of symbols in the corresponding object.\n\nlet mut writer = VecWriter::new(&mut resolutions);\nlet mut shards = writer.take_shards(objects.iter().map(|o| o.num_symbols));\n\n\nWe can now, in parallel, iterate through our objects and their corresponding shards and initialise\nthe shards.\n\nobjects\n   .par_iter()\n   .zip_eq(&mut shards)\n   .for_each(|(obj, shard)| {\n      for symbol in obj.symbols() {\n         shard.push(...);\n      }\n   });\n\n\nLastly, we return the shards to the writer, which verifies that all the shards were fully\ninitialised, thus resizing the Vec, after which it can be used normally.\n\nwriter.return_shards(shards);\n\n\nAtomic - non-atomic in-place conversion\n\nMost parts of the linker can make do with either exclusive access to part of the resolutions Vec,\nor shared access to the entire Vec. However, there\u2019s one part of the linker where we need to perform\nrandom writes to the resolutions Vec. This is done when we have multiple symbol definitions with\nthe same name. Originally, I just did this work from the main thread, since I figured most of the\ntime there would only be a small number of symbols that had the same name. This was mostly true,\nhowever for large C++ binaries like Chromium, it turns out that there are actually a lot of symbols\nwith the same names, presumably due to C++\u2019s use of header files, which create lots of identical\ndefinitions.\n\nTo allow random writes to resolutions, we introduce a new type:\n\nstruct AtomicSymbolId(AtomicU32);\n\n\nBeing an atomic, we can write to an AtomicSymbolId using only a shared (non-exclusive) reference.\nHowever, we need a way to temporarily view our Vec<SymbolId> as a &[AtomicSymbolId].\n\nThe standard library has something that might help - AtomicU32::from_mut_slice:\n\nfn from_mut_slice(v: &mut [u32]) -> &mut [AtomicU32]\n\n\nHowever, it\u2019s unstable (nightly only). Even if it were stable, it only works with slices of\nprimitive types, so we\u2019d have to lose our newtypes (SymbolId etc).\n\nAnother option would be to always use atomics, however that would quite possibly hurt performance of\nthe rest of the linker, which doesn\u2019t need atomics. It\u2019d also hurt ergonomics, since currently our\nSymbolIds implement Copy, but if they wrapped an AtomicU32, then they wouldn\u2019t be able to.\n\nA reasonable option at this point would be to resort to unsafe and use something like\ncore::mem::transmute. We\u2019d need to check all the rules and make sure that we were meeting all the\nrequirements. This is not a bad option, but I personally like the challenge of doing things without\nunsafe if I can, especially if I can do so without loss of performance.\n\nIndeed, it turns out that we can, as follows:\n\nfn into_atomic(symbols: Vec<SymbolId>) -> Vec<AtomicSymbolId> {\n   symbols\n       .into_iter()\n       .map(|s| AtomicSymbolId(AtomicU32::new(s.0)))\n       .collect()\n}\n\n\nIt\u2019d be reasonable to think that this will have a runtime cost, however it doesn\u2019t. The reason is\nthat the Rust standard library has a nice optimisation in it that when we consume a Vec and collect\nthe result into a new Vec, in many circumstances, the heap allocation of the original Vec can be\nreused. This applies in this case. But what even with the heap allocation being reused, we\u2019re still\nlooping over all the elements to transform them right? Because the in-memory representation of an\nAtomicSymbolId is identical to that of a SymbolId, our loop becomes a no-op and is optimised\naway.\n\nWe can verify this by looking at the assembly produced for this function:\n\nmovups  xmm0, xmmword, ptr, [rsi]\nmov     rax, qword, ptr, [rsi, +, 16]\nmovups  xmmword, ptr, [rdi], xmm0\nmov     qword, ptr, [rdi, +, 16], rax\nret\n\n\nThe main takeaway from this assembly is that there\u2019s no branching, no looping, just a few moves and\na return. If we allowed this function to be inlined into the caller, it would likely vanish to\nnothing.\n\nFor conversion back to the non-atomic form, we can do much the same:\n\nfn into_non_atomic(atomic_symbols: Vec<AtomicSymbolId>) -> Vec<SymbolId> {\n   atomic_symbols\n       .into_iter()\n       .map(|s| SymbolId(s.0.into_inner()))\n       .collect()\n}\n\n\nThe main thing to note here is that we avoid doing an atomic load from the atomic and instead\nconsume the atomic with into_inner. This is easier for the compiler to optimise and if we look at\nthe assembly produced it\u2019s identical to what we got for into_atomic.\n\nTo actually use these functions, we first need to get ownership of our Vec using core::mem::take.\nThis puts an empty Vec in its place. Empty Vecs don\u2019t heap allocate, so this is very cheap. We then\ncall into_atomic to convert the Vec into the form we need.\n\nlet atomic_resolutions = into_atomic(core::mem::take(&mut self.resolutions));\n\n\nWe can then do whatever parallel processing we need with the Vec in its atomic form.\n\nprocess_resolutions_in_parallel(&atomic_resolutions);\n\n\nFinally, we convert back to the original non-atomic form and store back where we got it from,\noverwriting the empty Vec that we temporarily put in its place.\n\nself.resolutions = into_non_atomic(atomic_resolutions);\n\n\nOne thing worth noting here is that if we panic (or do an early return), we might leave\nself.resolutions as the empty Vec. This isn\u2019t a problem in the linker, since if we\u2019re returning an\nerror or have hit a panic, then we don\u2019t care at that point about resolutions. It would be possible\nto ensure that the proper Vec was restored for use-cases where that was important, however it would\nadd extra complexity and might be enough to convince me that it\u2019d be better to just use transmute.\n\nBuffer reuse\n\nDoing too much heap allocation tends to hurt performance. A common trick is to move heap allocations\noutside of loops. For example, rather than this:\n\nloop {\n    let mut buffer = Vec::new();\n    // Do work with `buffer`.\n}\n\n\nWe might prefer to allocate buffer before the loop, then just clear it inside the loop:\n\nlet mut buffer = Vec::new();\nloop {\n    buffer.clear();\n    // Do work with `buffer`.\n}\n\n\nHowever, if we\u2019re storing something into a Vec that has a non-static lifetime, then we can run into\nproblems. Here, we have a variable text, which holds a String. We then split that string and\nstore the resulting string-slices into buffer. Even though we clear buffer at the end of the\nloop, the compiler is unhappy. It wants text to outlive buffer because we\u2019re storing references\nto text into buffer.\n\nlet mut buffer = Vec::new();\nloop {\n    let text = get_text();\n    buffer.extend(text.split(\",\"));\n    // Do work with `buffer`.\n    buffer.clear();\n}\n\n\nWe could at this point give up and just move our Vec creation back inside the loop. However, it\nturns out that there\u2019s another solution.\n\nfn reuse_vec<T, U>(mut v: Vec<T>) -> Vec<U> {\n   v.clear();\n   v.into_iter().map(|x| unreachable!()).collect()\n}\n\n\nThe idea of this function is to convert from a Vec of some time to an empty Vec of another type,\nreusing the heap allocation. This works in a very similar way to how we converted between atomic and\nnon-atomic SymbolIds, except this time because we first clear the Vec, the body of our map\nfunction is unreachable.\n\nThe optimisation in the Rust standard library that allows reuse of the heap allocation will only\nactually work if the size and alignment of T and U are the same, so let\u2019s verify that that\u2019s the\ncase. We can do the check at compile time, so if we accidentally call this function with\nincompatible T and U, we\u2019ll get a compilation error at the call site.\n\nfn reuse_vec<T, U>(mut v: Vec<T>) -> Vec<U> {\n   const {\n       assert!(size_of::<T>() == size_of::<U>());\n       assert!(align_of::<T>() == align_of::<U>());\n   }\n   v.clear();\n   v.into_iter().map(|_| unreachable!()).collect()\n}\n\n\nLet\u2019s verify that this optimises as we expect:\n\nmov     qword, ptr, [rsi, +, 16], 0\nmovups  xmm0, xmmword, ptr, [rsi]\nmovups  xmmword, ptr, [rdi], xmm0\nmov     qword, ptr, [rdi, +, 16], 0\nret\n\n\nMore or less the same assembly as before, except that we\u2019re now setting the length of the Vec to 0.\nNote, that the loop and the panic from the use of unreachable! are gone.\n\nWe can now integrate this into our previous code as follows:\n\nlet mut buffer_store: Vec<&str> = Vec::new();\nloop {\n    let mut buffer = reuse_vec(buffer_store);\n    let text = get_text();\n    buffer.extend(text.split(\",\"));\n    // Do work with `buffer`.\n    buffer_store = reuse_vec(buffer);\n}\n\n\nEffectively, each time around the loop we move out of buffer_store, converting the type of the\nVec, use it for a bit, then convert it back and store it again in buffer_store. The only time\nwe\u2019ll need a new heap allocation is when our Vec needs to grow. The types of buffer_store and\nbuffer are both Vec<&str>, however the lifetime of the references is different.\n\nDeallocation on a separate thread\n\nFreeing memory is generally a lot slower than allocating it. If we\u2019ve done a very large allocation,\nit can sometimes be worthwhile passing it to another thread to free it, so that we can get on with\nother work.\n\nFor example, if using rayon, we might use rayon::spawn to spawn a task that drops our buffer:\n\nfn process_buffer(buffer: Vec<u8>) {\n   // Do some work with `buffer`.\n\n   rayon::spawn(|| drop(buffer));\n}\n\n\nNote, that rayon::spawn itself does a heap allocation, so this would only be worthwhile if\nbuffer was potentially very large. This is definitely something you\u2019d want to benchmark to see if\nit actually improves the runtime for your use-case. There is at least one place in the Wild linker\nwhere we did this and it did give a measurable reduction in runtime.\n\nSimilar to buffer reuse, if our heap allocation has non-static lifetimes associated with it, we can\nget rid of them using reuse_vec.\n\nfn process_buffer(names: Vec<&[u8]>) {\n   // Do some work with `names`.\n\n   let names: Vec<&[u8]> = reuse_vec(names);\n   rayon::spawn(|| drop(names));\n}\n\n\nIn this case, we\u2019re converting the Vec from a Vec<&[u8]> to  Vec<&'static [u8]>.\n\nBonus: Strip lifetime with non-trivial Drop\n\nThis is a bonus tip that wasn\u2019t included in the talk and builds on the previous tip and is in\nresponse to a question by VorpalWay on Reddit. If you want to drop a Vec<T> and T has both a\nnon-static lifetime and a non-trivial Drop, then things get slightly more tricky. The trick here\nis to convert to a struct that is the same as T, but has non-static references replaced with\nMaybeUninit.\n\nFor example, suppose we have the following struct:\n\npub struct Foo<'a> {\n    owned: String,\n    borrowed: &'a str,\n}\n\n\nWe can define a new struct:\n\nstruct StaticFoo {\n    owned: String,\n    borrowed: MaybeUninit<&'static str>,\n}\n\n\nWe can then convert our Vec to the new type with zero cost and no unsafe:\n\nfn without_lifetime(foos: Vec<Foo>) -> Vec<StaticFoo> {\n    foos.into_iter()\n        .map(|f| StaticFoo {\n            owned: f.owned,\n            borrowed: MaybeUninit::uninit(),\n        })\n        .collect()\n}\n\n\nThe presence of MaybeUnit::uninit() tells the compiler that it\u2019s OK to have anything there, so it\ncan choose to leave whatever &str was in the original Foo struct. This means that it\u2019s valid to\nproduce a StaticFoo with the same in-memory representation as the Foo that it replaces, allowing\nit to eliminate the loop. The asm for this function is:\n\n movups  xmm0, xmmword, ptr, [rsi]\n mov     rax, qword, ptr, [rsi, +, 16]\n movups  xmmword, ptr, [rdi], xmm0\n mov     qword, ptr, [rdi, +, 16], rax\n ret\n\n\ni.e. the loop was indeed eliminated.\n\nNow that we have a Vec with no non-static lifetimes, we can safely move it to another thread.\n\nThanks\n\nThanks to my github sponsors. Your contributions help\nto make it possible for me to continue to work on this kind of stuff rather than going and getting a\n\u201creal job\u201d.\n\n\n  CodeursenLiberte\n  Urgau\n  pmarks\n  repi\n  embark-studios\n  mati865\n  bes\n  joshtriplett\n  mstange\n  bcmyers\n  Rafferty97\n  acshi\n  Kobzol\n  flba-eb\n  jonhoo\n  marxin\n  tommythorn\n  binarybana\n  teburd\n  bearcove\n  yerke\n  teh\n  twilco\n  Shnatsel\n  coastalwhite\n  wezm\n  davidcornu\n  gendx\n  rrbutani\n  nazar-pc\n  willstott101\n  tatsuya6502\n  teohhanhui\n  jkendall327\n  EdorianDark\n  drmason13\n  HadrienG2\n  jplatte\n  rukai\n  ymgyt\n  dream-dasher\n  alexkirsz\n  Pratyush\n  Tudyx\n  coreyja\n  dralley\n  irfanghat\n  mvolfik\n  simtheverse\n\n\nDiscussion\n\n\n  Reddit"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:09:57.568219+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-20T01:09:57.502731+00:00",
                "index_texts": null,
                "output": "Wild Performance Tricks | David Lattimore",
                "pwd": "/data/archive/1766192981.241534",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-20T01:09:57.464912+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html']' timed out after 60 seconds",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "Wild Performance Tricks | David Lattimore",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1766192981.241534",
    "newest_archive_date": "2025-12-20T01:10:16.079910+00:00",
    "num_failures": 1,
    "num_outputs": 8,
    "oldest_archive_date": "2025-12-20T01:09:46.912625+00:00",
    "path": "/posts/2025/09/02/rustforge-wild-performance-tricks.html",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KCWMW991D8178EE5017MX735",
    "snapshot_id": "073e1c76-daef-4f15-8c86-c3764f4e9c65",
    "sources": [
        "/data/sources/1766192979-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1766192981.241534",
    "title": "Wild Performance Tricks | David Lattimore",
    "url": "https://davidlattimore.github.io/posts/2025/09/02/rustforge-wild-performance-tricks.html"
}