{
    "archive_path": "archive/1780723852.876703",
    "base_url": "alexispurslane.github.io/rsync-analysis",
    "basename": "",
    "bookmarked_date": "2026-06-06 05:30",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/alexispurslane.github.io/rsync-analysis",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=alexispurslane.github.io",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "alexispurslane.github.io",
    "downloaded_at": "2026-06-06T05:30:57.824388+00:00",
    "downloaded_datestr": "2026-06-06 05:30",
    "extension": "",
    "hash": "1FXX3JZCFV4X3CDGEE31",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://alexispurslane.github.io/rsync-analysis/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-06-06T05:32:47.132095+00:00",
                "index_texts": null,
                "output": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://alexispurslane.github.io/rsync-analysis/']' timed out after 60 seconds",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:31:47.097428+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://alexispurslane.github.io/rsync-analysis/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2026-06-06T05:31:19.005462+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:31:09.957296+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=alexispurslane.github.io"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-06-06T05:31:00.835955+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:30:58.341796+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://alexispurslane.github.io/rsync-analysis/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-06-06T05:31:00.906519+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:31:00.873505+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2026-06-06T05:31:40.001413+00:00",
                "index_texts": [
                    "Did Claude Increase Bugs in rsync? (bug_rate_report.css)  Data Analysis \u00b7 June 2026 Did Claude Increase Bugs in rsync? A simple distributional analysis of every rsync release with bug data. Nothing\n            complicated, answers only one question: are the Claude-assisted releases unusually buggy? Repository: (https://github.com/RsyncProject/rsync) RsyncProject/rsync Method: severity-weighted bugs per 10 commits, exact permutation test   0 \u00b7 Disclaimer: How AI Assistance Was Used In order to avoid accuastions of this \"just being Claude defending Claude,\" \"AI slop,\" \"probably all\n            hallucinations,\" etc., I've decided it's probably worth explaining a few key points about how this report\n            was\n            created: All metrics, methodology, and data sources were exclusively chosen by me, in consultation with my wife,\n                who has a Master's Degree in Statistics from Penn State University. The methodology is directly based on my wife's input: she was the one that pointed out that trying to\n                just compare bugs per ten lines of code before and after would likely be too effected by noise because\n                of the low number of post-Claude samples, and that, for similar reasons, trying to build some kind of\n                linear regression model to ascertain the relative effects of different variables would probably also not\n                work. She specifically told me that looking at where the post-Claude releases fall into the historical\n                distribution, and how likely from the historical distribution we would be to get releases as \"bad\" or\n                worse than the post-Claude releases, was probably the best that could be done. I spent several days on this, two before even creating the GitHub repo and had at least one major total\n                rewrite of the report to use a better methodology (given the feedback from my wife mentioned above).\n                This was a lot of manual, cognitive effort on my end. The scripts used to fetch the data, collate it into a DuckDB database file, construct the views on that\n                DB, and then do the statistical analysis on that data, were indeed written by GLM 5.1 , as was the\n                HTML and much of the original prose for the final report webpage you're looking at right now. Crucially, however, all numbers, statistics, cards, and graphs in this report are automatically\n                    templated in directly by the Python script that ran the statistical analysis , thus avoiding any\n                possibility of hallucinations or inconsistencies in the numbers. After posting this (https://news.ycombinator.com/item?id=48411635) on Hacker News and\n                recieving almost no substantive input, discussion, or response on the actual content of the\n                article, I decided to rewrite all of the prose in my own voice. If anyone complains about my verbosity\n                or sentence structure \u2014 as they usually do, which is the reason I originally let the AI write the prose,\n                among other reasons obsoleted by templating \u2014 they can go fuck themselves. If you want to replicate the data and results here, and inspect exactly how they were calculated, you\n                can find the repository (https://github.com/alexispurslane/rsync-analysis/) here . I have\n                purposefully made it so that the pipeline can be run end to end completely from scratch, so you can see\n                the entire pipeline end-to end, with no mysterious DB blobs forcing you to trust that I didn't doctor\n                or screw up the data. If you want to be mad about the numbers, look there first.    1 \u00b7 Background: The rsync Outrage In late May 2026, rsync blew up. First, an (https://mastodon.gamedev.place/@JeremiahFieldhaven/116654345332213390) evidence-free Mastodon post\n                was made pointing to a spurious correlation between a regression that particular user experienced\n            upon upgrading to a release, and that release having Claude commits in it. It was viewed an unknown number\n            of times, but even likes and boosts passed the thousands mark handily, and it gained significant traction \u2014\n            as all spurious\n            anti-AI hate does \u2014, seeing 58 replies from 32 unique users. Someone rages about \"cognitive surrender\" with\n            no evidence; another suggests adding rsync to the famous (https://codeberg.org/small-hack/open-slopware) open-slopware (https://en.wikipedia.org/wiki/Index_Librorum_Prohibitorum) blacklist . From\n            there, it spread to (https://news.ycombinator.com/item?id=48334021) Hacker News , with 81\n            comments,\n            full of mixed dread, anger, and crowing about how this finally proves once and for all no one can use LLMs\n            safely. Among all that was (https://news.ycombinator.com/item?id=48334270) one particular\n                comment which spurred\n            further the view that the regressions and bugs were caused by Claude. This On May 30, 2026, this burgeoning outrage emergently coalesced into a single focal point: a GitHub issue\n            titled (https://github.com/RsyncProject/rsync/issues/929) \"Please Do\n                Not Vibe Fuck Up This Software\" , opened against the rsync repository. It attached a screenshot of\n            the Mastodon post criticizing the project's use of Claude. That's it. No bug report, no technical content,\n            no\n            attempt to actually ascertain if the concern was real or justified; just 350+ comments ranging from thoughtful concern to outright\n            harassment (most of the most egregious, unreasonable, and outright violent comments have since been deleted;\n            few thought to preserve them). (GitHub issue screenshot) The GitHub issue that started it all. The original post was a screenshot of a Mastodon critique,\n                no bug report, no technical content. It has since accumulated 329 comments .   (Hacker News thread screenshot) The thread quickly escalated, from \"the software is free, if you don't like it then fork it or\n                fuck off\" to: \"just because you are giving free soup to the homeless, does not mean you can piss in\n                    it\" .  The thread did not stop at words. It eventually escalated to, at one point, visual depictions of\n            fantasies of violence, when one user posted a now deleted comment including My Little Pony drawings of themselves\n            strangling the\n            \"project janitor that pushed vibecoded commits\": (Threatening drawing) A user posting drawings depicting violence against the rsync maintainer, one of several threats\n                that escalated the issue from heated debate to harassment.  Completing the internet outrage cycle, this issue in turn spread to (https://news.ycombinator.com/item?id=48342705) Hacker News , generating hundreds more comments.\n            Some (https://news.ycombinator.com/item?id=48346124) attempted to point at the number of\n            regressions after the introduction of Claude \u2014 \"The Linux Mint Timeshift tool has an issue open documenting a number of regressions that are currently\n                open on the rsync issues page, that were only introduced post-vibecoding\" \u2014 as evidence that it was\n            worse. Others (https://news.ycombinator.com/item?id=48348708) pointed out that those regressions\n            were not caused by Claude, and in response, the goalposts were moved again. Over and over, the core theme\n            was one\n            central claim, repeated everywhere: Claude-assisted development introduced bugs\n                into a previously stable tool. AI is cognitive surrender, is cocaine, is loss of craft, and the\n            users are right to be angry as a result:  People are very justifiably angry that a very stable, well trusted tool , has started to\n                immediately go downhill\u2026 all because the main dev is vibecoding that software.  \u2014 fao_ on Hacker News  However, this isn't doesn't have to be a question solved only on the basis of \u2014 ironically \u2014 vibes. This is\n            something that could be, at least to a degree, empirically tested. Some even pointed that out: On Lobste.rs, in response to (https://medium.com/@tridge60/rsync-and-outrage-d9849599e5a0) the Medium\n                essay Tridge himself posted in response , finally some users like boramalper begin to\n            actually ask for evidence one way or another: It'd be interesting if someone actually did a timechart of regressions after each release (if at all\n                possible) to see if the number actually went up recently or not. \u2014 boramalper on Lobsters  User bitshift replied: \"I would also love to see such a chart. It wouldn't be completely\n                informative\u2026 But at least it would be something objective we could measure.\"  This analysis is that chart. Or, well, as best as it can be made, given the limitations of\n            the data (see the previous section).   2 \u00b7 Executive Summary 36 releases with bug data, spanning v2.4.6 to\n                v3.4.3 2 releases have Claude commits: v3.4.2 (9 Claude, 0.00 sev/10c) and v3.4.3 (28 Claude, 3.29 sev/10c) The Claude releases bracket the IQR in opposite directions: v3.4.2 is below the IQR, v3.4.3 is above it. Neither is an outlier. Exact permutation test p-value = 46%: pick any\n                2 releases at random, you'd score as bad or worse 46% of the time.\n                This is the strongest available test and it finds nothing. Fisher's exact test p-value = 74%: Claude releases are\n                no more likely to fall above the historical median than any other releases\n                (odds ratio 1.06). The historical mean is 1.8\u00d7 the Claude mean (2.95 vs\n                1.65 sev/10c) v3.4.1 (59 bugs / 9 commits, no Claude) is an outlier\n                but belongs in the baseline \u2014 it is a release, and the distribution already captures it    3 \u00b7 The Metric The analysis uses a single metric: severity-weighted bugs per 10 commits (sev/10c).\n            Each bug is normalized to a 0\u20131 severity score (its LLM-assigned severity divided by 100), and those\n            scores are summed per release instead of simply counting bugs. The raw bug count is also shown in the\n            table for reference, but sev/10c drives all statistical tests. sev/10c = (\u03a3 severity/100 \u00f7 total_commits) \u00d7 10  How commits are assigned to releases Every commit on the default branch was ordered by committer date to produce a sequential timeline. Each git\n            tag points to a specific commit in this timeline. A release's range is all commits between the previous tag\n            and its own tag. Pre-release tags (\"pre\", \"rc\") are skipped as boundaries and absorbed into their final\n            release. Every commit belongs to exactly one release. How bugs are found and assigned to releases Bug reports come from three sources: GitHub issues in the rsync repository (collated via the GitHub REST API), the rsync Bugzilla instance (collected via the API), and the rsync mailing list.  GitHub issues and mailing-list bugs are attributed to the most recent\n            release that shipped before the bug was reported. For Bugzilla, each entry has a \"Version\" field that\n            explicitly states which release the bug was reported against, and bugs\n            are attributed to that release. Severity scoring To control for bug severity \u2014 so that, as someone on HN said, a typo in a button and a CVE aren't rated\n            equally \u2014 every bug report was scored for severity on a\n            0\u2013100 scale. The scorer is (https://huggingface.co/Qwen/Qwen3-35B) Qwen\u00a03 35B ,\n            a small open-weight language model, prompted as a senior reliability engineer assessing real-world impact.\n            Each bug report was given to the model as its title and body text (truncated to 3,000 characters),\n            along with the following rubric: Score Category Description   90\u2013100 Data loss / corruption Silent data corruption or data loss. The user's files or backups are wrong and they may not notice until it's too late. Security vulnerabilities that allow remote code execution or unauthorized access.  70\u201389 Crash / hang / broken backups rsync crashes, hangs, or fails in a way that breaks automated backups or cron jobs. Data is not corrupted but backups are missed. High CPU or memory usage that makes rsync unusable in production. Build or compilation failures \u2014 if rsync cannot be built from source, users cannot install it at all. This is a blocking problem, not a minor inconvenience. Score these at least 70. Security vulnerabilities that expose sensitive data.  50\u201369 Feature regression Feature regressed \u2014 something that used to work no longer works, but there's a workaround. Performance regressions large enough to disrupt production workflows. Incorrect output that is visible (errors, wrong filenames) but doesn't corrupt data.  30\u201349 Minor regression Minor feature regression with easy workaround. Error messages are confusing but the operation still succeeds. Intermittent test failures. Portability issues on uncommon platforms.  10\u201329 Cosmetic / low impact Cosmetic issues, documentation errors, minor UX annoyances. Test-only issues that don't affect users.  0 Feature request If the issue is asking for a new feature, a change in default behavior, or a packaging suggestion \u2014 no matter how reasonable \u2014 it is NOT a bug. Score it 0.  0\u20139 Not a real bug Spam, off-topic, duplicate. Issues that are clearly not about rsync or are empty/meaningless.    All three bug sources \u2014 GitHub issues, Bugzilla, and the rsync mailing list \u2014 were scored. Bugzilla and\n            mailing list reports had only a title (no body), so the model scored those from the title alone. The model\n            was instructed to fall back on the title and lean toward the middle of the range (40\u201360) when the body\n            didn't provide enough information. The model was also told to output only a severity integer via (https://openai.com/index/introducing-structured-outputs-in-the-api/) structured output (JSON schema), so there were no free-text responses to parse. Scoring was done at temperature\u00a00\n            for determinism \u2014 the same input always produces the same score.  Issues scored severity\u00a00 \u2014 feature requests, spam, off-topic rants about AI, empty submissions \u2014 are\n            excluded from bug counts by default. This matters because some releases attracted a lot of noise on\n            GitHub. v3.4.2 had four issues filed; the model scored all four at severity\u00a00 (a feature-request\n            option, a missing tarball question, and two more feature requests).  Example scores from the database, one per tier: Score Release Title   95 (https://github.com/RsyncProject/rsync/issues/750) v3.4.1  Destination --chmod and --fake-super: chmod applied to fake-super, backup permissions lost  75 (https://github.com/RsyncProject/rsync/issues/951) v3.4.3  error in rsync protocol data stream (code 12) at token.c(490) [sender=3.4.3]  55 (https://github.com/RsyncProject/rsync/issues/880) v3.4.1  --dry-run doesn't work with --mkpath when copying files  35 (https://github.com/RsyncProject/rsync/issues/846) v3.4.1  manpages not installed in out-of-tree builds  15 (https://github.com/RsyncProject/rsync/issues/866) v3.4.1  Minor inconsistencies in options in manpage  0 (https://github.com/RsyncProject/rsync/issues/968) v3.4.3  PAM Support - Open for Discussion    Why the release is the unit of analysis Why group commits by release, bugs by release, and then ascertain the correlation \u2014 or lack\n            thereof \u2014 between Claude commits and bugs through the intermediary of releases? This is for two\n            reasons. First, because the claim that the critics are making is also, itself, made in terms of releases: that\n            having any Claude commits in a release makes the whole release more buggy as a whole in a noticeable way,\n            not just that Claude-authored commits may introduce more bugs; the latter is a different metric, because later Claude- or human-authored commits could correct for those bugs within the same release , and\n            nobody would then notice as part of the release, and overall it wouldn't matter to users; additionally,\n            it's simply important, as stated elsewhere, to meet the claim of the critics where it's at. If this forces\n            them to make their claims more nuanced \u2014 or otherwise move the goalposts \u2014 then mission accomplished .  Second, it's a problem of attribution: the vast, vast majority of bugs do not state exactly which commit\n            caused them, because doing so would require extensive research and analysis that is often not worth it in\n            favor of simply fixing-forward, and even if that analysis was done \u2014 via something like git bisect \u2014 it wouldn't necessarily result in anything useful, or anything at all.\n            Many bugs can result from a combination of multiple commits, often separated significantly over time,\n            where it's unclear whether one commit or the other really introduced the bug. Or, one commit can reveal\n            several latent bugs introduced by other commits at once, and so on.  Why bugs and commits? The critics' claim is simplistic, absolute, and universalistic: the rate of bugs in the Claude-exposed\n            releases went up. Therefore, the simplest honest response is to analyize precisely what is being claimed:\n            bugs, commits, releases, and Claude-exposed commits. If the Claude releases sit in the middle of the\n            historical distribution, the burden shifts to the critics to explain why this particular middle is somehow\n            worse than all the other middles that came before it. Even by weighting by severity, I feel that I am giving extensive generosity to the anti-AI point in all this, but enough of the more intelligent critics brought it up that I found it worth it. Even if that results in is shifting the conversation\n            toward a more nuanced discussion of the quality and type and user impact of the bugs in\n            the releases, it will already have been a major win for the pro-AI crowd, and a shifting of the goalposts\n            for the anti-AI crowd, and then we can do further analysis based on that. And the ball's in the anti-AI\n            court for that game. What this approach does not do I'm aware that this metric does not control for commit complexity or security intensity. It is a blunt\n            instrument. But the critics' accusation is also\n            blunt: \"Claude is making things worse.\" A blunt instrument is what is required in response. Blood begets\n            blood.   4 \u00b7 Results Claude Releases Before we jump into deeper analysis, let's just look at the two Claude releases themselves, to get a sense\n            for them: v3.4.2 0.00 sev/10c  0 bugs \u00b7 50 commits \u00b7 9 Claude 0th percentile (rank 0 of 35)  v3.4.3 3.29 sev/10c  17 bugs \u00b7 34 commits \u00b7 28 Claude 77th percentile (rank 27 of 35)   If that doesn't look like a red flag to you, you'd be right. Exact Permutation Test So the question is: are the Claude releases unusually buggy, or could you easily pull a group just as bad\n            out of the historical distribution by dumb luck? The way you answer that question statistically is an (https://en.wikipedia.org/wiki/Permutation_test) exact permutation test  , which\n            just enumerates all pairs of two releases and asks: what fraction have a\n            mean bug rate as bad or worse than the one we actually observed? That fraction is the p-value of the\n            hypothesis under test.  46% exact permutation test p-value (one-sided, H\u2081: Claude mean > historical)  272 of 595 possible groups of 2 historical releases have mean\n                sev/10c \u2265 1.65. Nearly half. The Claude releases sit right in the middle of the\n                permutation distribution \u2014 there is nothing extreme about them. Test statistic: mean sev/10c per group \u00b7 Claude group mean: 1.65 \u00b7 Historical mean:\n                    2.95    What this p-value tells us is that the hypothesis that Claude makes releases worse has, at least so far,\n            about as much predictive power as a coin flip: if you closed your eyes and picked 2\n            releases at random, you'd do as bad or\n            worse\n            nearly half the time. There's nothing unusual about the Claude group. Fisher's Exact Test The permutation test asks: how likely is it that a random group of releases scores as badly as the\n            Claude group? But there's another way to pose the question:\n            are Claude releases more likely than non-Claude releases to fall above the historical\n            median? That's a textbook 2\u00d72 contingency table, and the standard test for it is (https://en.wikipedia.org/wiki/Fisher%27s_exact_test) Fisher's exact test .   \u2264 median > median   Non-Claude 18 17  Claude 1 1    74% one-sided p-value (H\u2081: Claude more likely above median) Fisher's exact test asks: if we split all releases at the historical median\n                (0.74 sev/10c), are these Claude releases significantly buggy than previous releases (more likely to land above the median)? With a p-value of 74%, the answer is a decisive no.\n                The odds ratio is 1.06 \u2014 essentially 1:1. Claude releases are\n                no more likely to be above the median than any other releases. Odds ratio: 1.06 \u00b7 Median: 0.74 sev/10c    To emphasize, this does not mean that all Claude releases in the future will not be more buggy. We don't have nearly enough data to build a model and extrapolate out like that, and that's not what a Fisher's exact test is for. The point that's being made here is that these specific releases are not at all notable; if no one had known they were AI, no one would have cared or noticed anything out of the ordinary, and there is no evidence with which to conclude that Claude made anything worse yet, unlike the objective, absolutist, universal claims made by critics. The Distribution In case you're not convinced, here's a visual aid, showing where these releases fall in the distribution of\n            all prior releases:    middle 50%  (v2.4.6: 0.92 sev/10c)  (v2.5.0: 0.18 sev/10c)  (v2.5.1: 0.26 sev/10c)  (v2.5.2: 0.25 sev/10c)  (v2.5.4: 1.64 sev/10c)  (v2.5.5: 1.32 sev/10c)  (v2.5.6: 0.28 sev/10c)  (v2.6.0: 0.18 sev/10c)  (v2.6.1: 0.08 sev/10c)  (v2.6.2: 9.88 sev/10c)  (v2.6.3: 0.59 sev/10c)  (v2.6.4: 0.12 sev/10c)  (v2.6.5: 0.49 sev/10c)  (v2.6.7: 0.09 sev/10c)  (v2.6.8: 0.74 sev/10c)  (v2.6.9: 0.72 sev/10c)  (v3.0.0: 0.31 sev/10c)  (v3.0.1: 0.40 sev/10c)  (v3.0.2: 4.56 sev/10c)  (v3.0.3: 1.96 sev/10c)  (v3.1.0: 0.92 sev/10c)  (v3.1.1: 4.86 sev/10c)  (v3.1.2: 3.22 sev/10c)  (v3.1.3: 5.07 sev/10c)  (v3.2.0: 0.25 sev/10c)  (v3.2.1: 0.72 sev/10c)  (v3.2.2: 1.51 sev/10c)  (v3.2.3: 3.52 sev/10c)  (v3.2.4: 0.67 sev/10c)  (v3.2.5: 1.04 sev/10c)  (v3.2.6: 1.13 sev/10c)  (v3.2.7: 8.73 sev/10c)  (v3.3.0: 6.66 sev/10c)  (v3.4.0: 0.66 sev/10c)  (v3.4.1: 39.39 sev/10c)  (v3.4.2: 0.00 sev/10c)  v3.4.2 (v3.4.3: 3.29 sev/10c)  v3.4.3  0.01 0.1 1 10 100   Historical  Claude  Middle 50% (IQR)   Outside IQR   How to read this graph: Each dot is a release. The shaded green band is the (https://en.wikipedia.org/wiki/Interquartile_range) interquartile range \u2014\n            the middle 50% of historical releases, from 0.29 to 2.59 sev/10c.\n            The darker regions on either side are the lower and upper quarters.  This is another way of saying the same thing the previous two tests said, but more\n            intuitively: the Claude releases (green dots) bracket the IQR in opposite directions . v3.4.2, with zero real\n            bugs, sits just below the IQR; v3.4.3 sits just above it. They bracket the middle of the distribution\n            in opposite directions. Neither is a negative outlier, and since they're on either side of the IQR, there's\n            no evidence Claude usage bends releases in either direction.  Commit Rate One possible objection I've seen is that while perhaps the defect rate of Claude-authored commits is not\n            worse than human-authored ones, Claude sped up developmend so much \u2014 due to \"vibe coding\" perhaps \u2014 that the\n            total number of bugs in each release got too high for comfort anyway, in which case it doesn't matter that\n            the defect rate per commit isn't so bad, because that's not what downstream users experience. We can check\n            this, though: p=88% exact permutation test: do Claude releases have more commits? Claude releases averaged 42 commits; non-Claude releases averaged\n                185. If you pick any 2 releases at random, you'd see as many\n                or more commits 88% of the time.   So it seems that Claude releases did not have meaningfully\n            more commits. If anything, they had a lot fewer , but of course, the data is underpowered to say that\n            for sure; just like the rest of this article, I'm only trying to show a startling lack of evidence\n            for the supposed massive harms of using Claude, not trying to prove there must have been none, or that it\n            was even \"beneficial\" in terms of focused releases. But commit count is a blunt measure. A commit can be a one-line typo fix or a rewrite of the entire\n            protocol handler. What about lines of code changed? GitHub's compare API gives us additions and deletions\n            between each pair of consecutive tags: p=5% exact permutation test: do Claude releases have more lines changed? Claude releases averaged 3756 lines changed; non-Claude releases averaged\n                696. If you pick any 2 releases at random, you'd see as many\n                or more lines changed only 5% of the time. Claude releases were larger\n                by this measure.   Lines changed went up by a lot. But did absolute bug numbers? p=77% exact permutation test: do Claude releases have more severity-weighted bugs? Claude releases averaged 5.6 severity-weighted bugs; non-Claude releases averaged\n                14.9. If you pick any 2 releases at random, you'd see as many\n                or more severity-weighted bugs 77% of the time. More lines changed, but not more bugs.   So: the Claude releases changed way more lines of code than historical ones, but didn't have more bugs.\n            More code, same bugs. That's not what you'd expect if Claude were making things worse. Regime Check The obvious counterargument is that maybe earlier rsync releases were less in maintenence-mode, and so\n            had more bugs, but recent rsync releases have been more stable, so comparing the two Claude-exposed\n            releases to the full historical distribution is masking the fact that they're actually outliers for their\n            regime. Luckily, there's a way to test this statistically. Yes, the historical mean (2.95 sev/10c) was driven by a bimodal distribution: v2.x releases\n            average 1.11 sev/10c; v3.x releases average 4.23. But even within the v3.x regime,\n            the Claude releases sit in the middle of the pack or better:    middle 50%  (v3.0.0: 0.31 sev/10c)  (v3.0.1: 0.40 sev/10c)  (v3.0.2: 4.56 sev/10c)  (v3.0.3: 1.96 sev/10c)  (v3.1.0: 0.92 sev/10c)  (v3.1.1: 4.86 sev/10c)  (v3.1.2: 3.22 sev/10c)  (v3.1.3: 5.07 sev/10c)  (v3.2.0: 0.25 sev/10c)  (v3.2.1: 0.72 sev/10c)  (v3.2.2: 1.51 sev/10c)  (v3.2.3: 3.52 sev/10c)  (v3.2.4: 0.67 sev/10c)  (v3.2.5: 1.04 sev/10c)  (v3.2.6: 1.13 sev/10c)  (v3.2.7: 8.73 sev/10c)  (v3.3.0: 6.66 sev/10c)  (v3.4.0: 0.66 sev/10c)  (v3.4.1: 39.39 sev/10c)  (v3.4.2: 0.00 sev/10c)  v3.4.2 (v3.4.3: 3.29 sev/10c)  v3.4.3inside middle 50% \u2713   0.01 0.1 1 10 100   v3.x historical  Claude  Middle 50% (IQR)   So the regime-shift argument doesn't just fail \u2014 it fails backwards . The v3.x era has a\n            much higher mean sev/10c than v2.x. If you restrict the comparison to v3.x only, Claude releases\n            don't stand out at all \u2014 and one of them is better than most. The only way to make Claude look\n            like an outlier is to compare it against a quieter era and then blame the shift on Claude, when\n            the data says the shift predates Claude entirely. We can further test whether there are meaningfully different regimes in the version history, and thus\n            whether using the full historical data is valid, by doing a (https://en.wikipedia.org/wiki/Wald%E2%80%93Wolfowitz_runs_test) runs test .\n            If such regimes existed, the runs test would detect non-random clustering:  p=0.060 Wald\u2013Wolfowitz runs test on 35 non-Claude releases 13 runs observed (expected 18.5 under randomness, z=-1.88).\n                At p=0.060 this is marginal \u2014 not strong enough to reject randomness at the conventional\n                0.05 level, but close enough that it warrants the v3.x-only comparison above, which\n                already accounts for the v2.x/v3.x difference. Using the full historical baseline\n                is a conservative choice: it includes a noisier era, which makes it harder, not easier,\n                to show that Claude releases are normal. Observed runs: 13 \u00b7 Expected: 18.5 \u00b7 z=-1.88    The Pre-Claude Outlier Here's my favorite part, though. Digging into the data, one of the first things that jumped out at me with\n            blinding clarity was that the worst release, by far, in rsync history was entirely prior to the\n                introduction of Claude : 39.39 bugs per 10 commits \u2014 v3.4.1, no Claude The highest bug rate in the entire dataset. 59 bugs in 9 commits, a hotfix\n                release\n                the day after v3.4.0. It exceeds every other release by an order of magnitude.   And yet nobody noticed. There was no AI to blame so there was no GitHub\n            issue\n            with\n            300 comments, no death threats, no threats to fork or move to openrsync. A maintainer\n            shipped a broken release and fixed it, just like normal. The only thing that made v3.4.3\n            special was the availability of an enemy everyone had already decided to hate . All Releases (chronological) Release Bugs Sev Commits Claude Bugs/10c Sev/10c \u2195 Percentile   v2.4.6 2 1.2 13 0 1.54 0.92 54th percentile  v2.5.0 4 1.3 73 0 0.55 0.18 11th percentile  v2.5.1 4 1.8 69 0 0.58 0.26 20th percentile  v2.5.2 6 2.9 117 0 0.51 0.25 14th percentile  v2.5.4 5 3.5 21 0 2.38 1.64 69th percentile  v2.5.5 22 11.6 88 0 2.50 1.32 63rd percentile  v2.5.6 14 6.6 239 0 0.59 0.28 23rd percentile  v2.6.0 8 4.7 267 0 0.30 0.18 9th percentile  v2.6.1 5 3.3 444 0 0.11 0.08 0th percentile  v2.6.2 29 16.8 17 0 17.06 9.88 94th percentile  v2.6.3 49 22.7 381 0 1.29 0.59 34th percentile  v2.6.4 22 9.4 760 0 0.29 0.12 6th percentile  v2.6.5 16 7.1 146 0 1.10 0.49 31st percentile  v2.6.7 15 5.8 649 0 0.23 0.09 3rd percentile  v2.6.8 12 5.4 72 0 1.67 0.74 49th percentile  v2.6.9 53 18.8 261 0 2.03 0.72 43rd percentile  v3.0.0 64 27.9 909 0 0.70 0.31 26th percentile  v3.0.1 6 4.0 102 0 0.59 0.40 29th percentile  v3.0.2 10 4.1 9 0 11.11 4.56 80th percentile  v3.0.3 22 10.8 55 0 4.00 1.96 71st percentile  v3.1.0 170 52.4 571 0 2.98 0.92 51st percentile  v3.1.1 68 32.1 66 0 10.30 4.86 83rd percentile  v3.1.2 55 18.4 57 0 9.65 3.22 74th percentile  v3.1.3 85 30.9 61 0 13.93 5.07 86th percentile  v3.2.0 22 7.8 304 0 0.72 0.25 17th percentile  v3.2.1 7 4.5 63 0 1.11 0.72 46th percentile  v3.2.2 13 8.8 58 0 2.24 1.51 66th percentile  v3.2.3 95 55.3 157 0 6.05 3.52 77th percentile  v3.2.4 20 14.3 213 0 0.94 0.67 40th percentile  v3.2.5 9 5.5 53 0 1.70 1.04 57th percentile  v3.2.6 6 3.2 28 0 2.14 1.13 60th percentile  v3.2.7 88 52.4 60 0 14.67 8.73 91st percentile  v3.3.0 42 25.3 38 0 11.05 6.66 89th percentile  v3.4.0 6 4.0 60 0 1.00 0.66 37th percentile  v3.4.1 59 35.5 9 0 65.56 39.39 97th percentile  v3.4.2 0 0.0 50 9 0.00 0.00 0th percentile  v3.4.3 17 11.2 34 28 5.00 3.29 77th percentile       5 \u00b7 What the Data Is Consistent And Inconsistent With \u2713 \"The Claude releases are statistically indistinguishable from historical\n                        releases\" One Claude release sits just below the IQR (v3.4.2, with zero real\n                        bugs), the other just above it.\n                        They bracket the middle of the distribution in opposite directions \u2014 neither is an outlier.\n                        The exact permutation test yields a p-value of 46% \u2014 pick any\n                        2 releases at random and you'd do as bad or worse nearly half the time.\n                        There is no signal of abnormality.   \u2713 \"The outrage selected on a single tail event and narrativized it\" A Mastodon user noticed a regression in v3.4.3, saw Claude\n                        commits, and concluded causation. But v3.4.3 at 3.29 sev/10c is at\n                        the 77th percentile \u2014 elevated but not extreme. 8\n                        historical releases scored higher. The correlation is noise.   \u2717 \"Claude clearly made things worse\" &emdash; the main claim The Claude releases bracket the IQR \u2014  one below, one above. Neither is\n                        an outlier.\n                        There is no distributional evidence of harm. The claim rests entirely on a post-hoc correlation\n                        observed by a social media user.   \u2717 \"Claude commits (in general) do not and will not make things worse\" This is a common misrepresentation of my claims here. I am not trying to extrapolate out into the future, and say something like \"in general, Claude won't make things worse\" or \"Claude will never make things worse.\" Instead, the point is this: there is no evidence of it having done so, and the two Claude releases we have currently are thoroughly unremarkable, so the outrage is totally unjustified.   \u2717 \"The regressions speak for themselves\" v3.4.1 \u2014  a pre-Claude release \u2014  has the highest bug rate in the\n                        dataset (39.39 sev/10c). Nobody noticed, because there was no AI to be angry at.\n                        The regressions only \"speak\" when you ignore the historical distribution.   \u2717 \"Just wait, more bugs will surface\" v3.4.3 has been out long enough that its rate\n                        (3.29) is already comparable to historical releases. The \"wait and see\"\n                        argument is an appeal to an unknowable future that shifts the burden of proof away from the\n                        critics. If more bugs surface, they will enter the distribution like every other release. There\n                        is no reason to expect a regime break.     Discussion, and Tridge's Response So, why do people feel like they've been betrayed, and feel so sure that things have \"clearly\" gotten worse &emdash; that Claude \"broke rsync\" &emdash; when there is no evidence for this, and no data except two thoroughly unremarkable releases? A lot of it is just sheer, blind outrage at the use of\n                LLMs. However, there are some confounders that might have caused people to feel that way: On the HN thread, user zos_kia (https://news.ycombinator.com/item?id=48347728) pointed at the confound directly: From a cursory look, it looks like a security fix in response to a CVE surfaced a coding error which\n                    has\n                    been present in the code since 2007. This is so banal that it's actually hilarious to see people lose their shit over it. \u2014 zos_kia on Hacker News  On Lobsters, user jbert (https://lobste.rs/s/k1b0za/rsync_outrage#c_2iowov) spelled\n                    out the causal chain: The trigger for the increased volume of changes (and hence increased number of regressions) was the\n                    influx of (mostly) LLM-enabled security issues. i.e. the causal chain was: LLMs \u2192 more known security issues \u2192 more changes needed than usual \u2192 more\n                        regressions than usual.  \u2014 jbert on Lobsters  Essentially, this isn't a \"Claude\" problem, it's a \"more security work\" problem, something that (https://medium.com/@tridge60/rsync-and-outrage-d9849599e5a0) Tridge himself confirmed in\n                his response, describing how a flood of AI-generated CVE reports forced rapid,\n                extensive\n                changes to rsync's attack surface.  But, as with all things AI, it doesn't matter. In the end, the outrage isn't about whether rsync is worse\n                or better now, it's about people not liking AI, and arguing from a priori definitions, not\n                empirical results, to the desired conclusion: that AI is bad: Like I said, the author \"tried to balance security against feature regression.\" I don't dispute that\n                    he tried. I merely dispute that the chatbots are good at writing code; in fact, they are bad at\n                    writing code. If the author had approached these security bugs by hand with a mental model (a\n                        Naur theory!) which preserves their desired features and functionality then they would have\n                        caused fewer regressions ... \u2014 Corbin (https://lobste.rs/s/k1b0za/rsync_outrage#c_ccwvw4) on\n                        Lobste.rs   In response to this sweeping, absolute, causal claim made with no evidence \u2014 and in fact, counter to the\n                evidence \u2014 based on an old philosophical claim about the epistemology of programming, it is perhaps best\n                to leave the victim of this outrage himself with the final word: \u2026for the people saying things like \"I'm a PhD from xyz uni and I'm telling you LLMs are just\n                    stochastic tools that make everything up and the world will fall apart if you use them\", I'm here to\n                    tell you that you are out of date. The world of software engineering has changed dramatically in the\n                    last few months. The world of IT security and maintaining software in the face of the flood of\n                    reports has completely and utterly changed just in the last few weeks. Anything you learned about\n                    this stuff last year might as well be from another planet\u2026 Bottom line is I do know (well, roughly!)\n                    how LLMs work, but that doesn't make them not useful. It does mean you have to be cautious, but I am\n                    being cautious, or as cautious as I can be given my desire to be sailing and not dealing with a\n                    flood of gunk from so-called internet experts. \u2014 (https://medium.com/@tridge60/rsync-and-outrage-d9849599e5a0) Andrew\n                        Tridgell     Sev/10c Distribution \u2014 rsync \u00b7 June 2026 \u00b7 All data from GitHub REST API, Bugzilla, and mailing lists    "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:31:39.972649+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://alexispurslane.github.io/rsync-analysis/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2026-06-06T05:31:47.032793+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:31:41.751677+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://alexispurslane.github.io/rsync-analysis/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2026-06-06T05:31:39.937636+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:31:36.725604+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpm05bzko2",
                    "https://alexispurslane.github.io/rsync-analysis/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2026-06-06T05:31:25.027504+00:00",
                "index_texts": [
                    "Data Analysis \u00b7 June 2026\n        \n        A simple distributional analysis of every rsync release with bug data. Nothing\n            complicated, answers only one question: are the Claude-assisted releases unusually buggy?\n        \n            Repository: RsyncProject/rsync\n            Method: severity-weighted bugs per 10 commits, exact permutation test\n        \n    \n\n    \n\n    \n        0 \u00b7 Disclaimer: How AI Assistance Was Used\n\n        In order to avoid accuastions of this \"just being Claude defending Claude,\" \"AI slop,\" \"probably all\n            hallucinations,\" etc., I've decided it's probably worth explaining a few key points about how this report\n            was\n            created:\n        \n            All metrics, methodology, and data sources were exclusively chosen by me, in consultation with my wife,\n                who has a Master's Degree in Statistics from Penn State University.\n\n            The methodology is directly based on my wife's input: she was the one that pointed out that trying to\n                just compare bugs per ten lines of code before and after would likely be too effected by noise because\n                of the low number of post-Claude samples, and that, for similar reasons, trying to build some kind of\n                linear regression model to ascertain the relative effects of different variables would probably also not\n                work. She specifically told me that looking at where the post-Claude releases fall into the historical\n                distribution, and how likely from the historical distribution we would be to get releases as \"bad\" or\n                worse than the post-Claude releases, was probably the best that could be done.\n\n            I spent several days on this, two before even creating the GitHub repo and had at least one major total\n                rewrite of the report to use a better methodology (given the feedback from my wife mentioned above).\n                This was a lot of manual, cognitive effort on my end.\n\n            The scripts used to fetch the data, collate it into a DuckDB database file, construct the views on that\n                DB, and then do the statistical analysis on that data, were indeed written by GLM 5.1, as was the\n                HTML and much of the original prose for the final report webpage you're looking at right now.\n\n            Crucially, however, all numbers, statistics, cards, and graphs in this report are automatically\n                    templated in directly by the Python script that ran the statistical analysis, thus avoiding any\n                possibility of hallucinations or inconsistencies in the numbers.\n\n            After posting this on Hacker News and\n                recieving almost no substantive input, discussion, or response on the actual content of the\n                article, I decided to rewrite all of the prose in my own voice. If anyone complains about my verbosity\n                or sentence structure \u2014 as they usually do, which is the reason I originally let the AI write the prose,\n                among other reasons obsoleted by templating \u2014 they can go fuck themselves.\n\n            If you want to replicate the data and results here, and inspect exactly how they were calculated, you\n                can find the repository here. I have\n                purposefully made it so that the pipeline can be run end to end completely from scratch, so you can see\n                the entire pipeline end-to end, with no mysterious DB blobs forcing you to trust that I didn't doctor\n                or screw up the data. If you want to be mad about the numbers, look there first.\n        \n\n\n        \n\n    \n\n    \n\n        1 \u00b7 Background: The rsync Outrage\n\n        In late May 2026, rsync blew up. First, an evidence-free Mastodon post\n                was made pointing to a spurious correlation between a regression that particular user experienced\n            upon upgrading to a release, and that release having Claude commits in it. It was viewed an unknown number\n            of times, but even likes and boosts passed the thousands mark handily, and it gained significant traction \u2014\n            as all spurious\n            anti-AI hate does \u2014, seeing 58 replies from 32 unique users. Someone rages about \"cognitive surrender\" with\n            no evidence; another suggests adding rsync to the famous open-slopware blacklist. From\n            there, it spread to Hacker News, with 81\n            comments,\n            full of mixed dread, anger, and crowing about how this finally proves once and for all no one can use LLMs\n            safely. Among all that was one particular\n                comment which spurred\n            further the view that the regressions and bugs were caused by Claude.\n\n        This On May 30, 2026, this burgeoning outrage emergently coalesced into a single focal point: a GitHub issue\n            titled \"Please Do\n                Not Vibe Fuck Up This Software\", opened against the rsync repository. It attached a screenshot of\n            the Mastodon post criticizing the project's use of Claude. That's it. No bug report, no technical content,\n            no\n            attempt to actually ascertain if the concern was real or justified; just 350+ comments\n            ranging from thoughtful concern to outright\n            harassment (most of the most egregious, unreasonable, and outright violent comments have since been deleted;\n            few thought to preserve them).\n\n        \n            \n            The GitHub issue that started it all. The original post was a screenshot of a Mastodon critique,\n                no bug report, no technical content. It has since accumulated 329 comments.\n            \n        \n\n        \n            \n            The thread quickly escalated, from \"the software is free, if you don't like it then fork it or\n                fuck off\" to: \"just because you are giving free soup to the homeless, does not mean you can piss in\n                    it\".\n        \n\n        The thread did not stop at words. It eventually escalated to, at one point, visual depictions of\n            fantasies of violence, when one user posted a now deleted comment including My Little Pony drawings of themselves\n            strangling the\n            \"project janitor that pushed vibecoded commits\":\n\n        \n            \n            A user posting drawings depicting violence against the rsync maintainer, one of several threats\n                that escalated the issue from heated debate to harassment.\n        \n\n        Completing the internet outrage cycle, this issue in turn spread to Hacker News, generating hundreds more comments.\n            Some attempted to point at the number of\n            regressions after the introduction of Claude \u2014\n            \"The Linux Mint Timeshift tool has an issue open documenting a number of regressions that are currently\n                open on the rsync issues page, that were only introduced post-vibecoding\" \u2014 as evidence that it was\n            worse. Others pointed out that those regressions\n            were not caused by Claude, and in response, the goalposts were moved again. Over and over, the core theme\n            was one\n            central claim, repeated everywhere: Claude-assisted development introduced bugs\n                into a previously stable tool. AI is cognitive surrender, is cocaine, is loss of craft, and the\n            users are right to be angry as a result:\n        \n\n        \n            People are very justifiably angry that a very stable, well trusted tool, has started to\n                immediately go downhill\u2026 all because the main dev is vibecoding that software.\n            \u2014 fao_ on Hacker News\n        \n\n        However, this isn't doesn't have to be a question solved only on the basis of \u2014 ironically \u2014 vibes. This is\n            something that could be, at least to a degree, empirically tested. Some even pointed that out:\n\n        On Lobste.rs, in response to the Medium\n                essay Tridge himself posted in response, finally some users like boramalper begin to\n            actually ask for evidence one way or another:\n\n        \n            It'd be interesting if someone actually did a timechart of regressions after each release (if at all\n                possible) to see if the number actually went up recently or not.\n            \u2014 boramalper on Lobsters\n        \n\n        User bitshift replied: \"I would also love to see such a chart. It wouldn't be completely\n                informative\u2026 But at least it would be something objective we could measure.\"\n\n        This analysis is that chart. Or, well, as best as it can be made, given the limitations of\n            the data (see the previous section).\n\n        \n\n    \n\n    \n\n    \n\n        2 \u00b7 Executive Summary\n\n        \n            36 releases with bug data, spanning v2.4.6 to\n                v3.4.3\n            2 releases have Claude commits: v3.4.2 (9 Claude, 0.00 sev/10c) and v3.4.3 (28 Claude, 3.29 sev/10c)\n            The Claude releases bracket the IQR in opposite directions: v3.4.2 is below the IQR, v3.4.3 is above it. Neither is an outlier.\n            Exact permutation test p-value = 46%: pick any\n                2 releases at random, you'd score as bad or worse 46% of the time.\n                This is the strongest available test and it finds nothing.\n            Fisher's exact test p-value = 74%: Claude releases are\n                no more likely to fall above the historical median than any other releases\n                (odds ratio 1.06).\n            The historical mean is 1.8\u00d7 the Claude mean (2.95 vs\n                1.65 sev/10c)\n            v3.4.1 (59 bugs / 9 commits, no Claude) is an outlier\n                but belongs in the baseline \u2014 it is a release, and the distribution already captures it\n        \n\n    \n\n    \n\n    \n\n    \n\n        3 \u00b7 The Metric\n\n        The analysis uses a single metric: severity-weighted bugs per 10 commits (sev/10c).\n            Each bug is normalized to a 0\u20131 severity score (its LLM-assigned severity divided by 100), and those\n            scores are summed per release instead of simply counting bugs. The raw bug count is also shown in the\n            table for reference, but sev/10c drives all statistical tests.\n\n        \n            sev/10c = (\u03a3 severity/100 \u00f7 total_commits) \u00d7 10\n        \n\n        How commits are assigned to releases\n\n        Every commit on the default branch was ordered by committer date to produce a sequential timeline. Each git\n            tag points to a specific commit in this timeline. A release's range is all commits between the previous tag\n            and its own tag. Pre-release tags (\"pre\", \"rc\") are skipped as boundaries and absorbed into their final\n            release. Every commit belongs to exactly one release.\n\n        How bugs are found and assigned to releases\n\n        Bug reports come from three sources:\n\n        \n            GitHub issues in the rsync repository (collated via the GitHub REST API),\n            the rsync Bugzilla instance (collected via the API),\n            and the rsync mailing list.\n        \n        \n\n        GitHub issues and mailing-list bugs are attributed to the most recent\n            release that shipped before the bug was reported. For Bugzilla, each entry has a \"Version\" field that\n            explicitly states which release the bug was reported against, and bugs\n            are attributed to that release.\n\n        Severity scoring\n\n        To control for bug severity \u2014 so that, as someone on HN said, a typo in a button and a CVE aren't rated\n            equally \u2014 every bug report was scored for severity on a\n            0\u2013100 scale. The scorer is Qwen\u00a03 35B,\n            a small open-weight language model, prompted as a senior reliability engineer assessing real-world impact.\n            Each bug report was given to the model as its title and body text (truncated to 3,000 characters),\n            along with the following rubric:\n\n        \nScoreCategoryDescription\n90\u2013100Data loss / corruptionSilent data corruption or data loss. The user's files or backups are wrong and they may not notice until it's too late. Security vulnerabilities that allow remote code execution or unauthorized access.70\u201389Crash / hang / broken backupsrsync crashes, hangs, or fails in a way that breaks automated backups or cron jobs. Data is not corrupted but backups are missed. High CPU or memory usage that makes rsync unusable in production. Build or compilation failures \u2014 if rsync cannot be built from source, users cannot install it at all. This is a blocking problem, not a minor inconvenience. Score these at least 70. Security vulnerabilities that expose sensitive data.50\u201369Feature regressionFeature regressed \u2014 something that used to work no longer works, but there's a workaround. Performance regressions large enough to disrupt production workflows. Incorrect output that is visible (errors, wrong filenames) but doesn't corrupt data.30\u201349Minor regressionMinor feature regression with easy workaround. Error messages are confusing but the operation still succeeds. Intermittent test failures. Portability issues on uncommon platforms.10\u201329Cosmetic / low impactCosmetic issues, documentation errors, minor UX annoyances. Test-only issues that don't affect users.0Feature requestIf the issue is asking for a new feature, a change in default behavior, or a packaging suggestion \u2014 no matter how reasonable \u2014 it is NOT a bug. Score it 0.0\u20139Not a real bugSpam, off-topic, duplicate. Issues that are clearly not about rsync or are empty/meaningless.\n\n\n        All three bug sources \u2014 GitHub issues, Bugzilla, and the rsync mailing list \u2014 were scored. Bugzilla and\n            mailing list reports had only a title (no body), so the model scored those from the title alone. The model\n            was instructed to fall back on the title and lean toward the middle of the range (40\u201360) when the body\n            didn't provide enough information. The model was also told to output only a severity integer via\n            structured output\n            (JSON schema), so there were no free-text responses to parse. Scoring was done at temperature\u00a00\n            for determinism \u2014 the same input always produces the same score.\n        \n\n        Issues scored severity\u00a00 \u2014 feature requests, spam, off-topic rants about AI, empty submissions \u2014 are\n            excluded from bug counts by default. This matters because some releases attracted a lot of noise on\n            GitHub. v3.4.2 had four issues filed; the model scored all four at severity\u00a00 (a feature-request\n            option, a missing tarball question, and two more feature requests).\n        \n\n        Example scores from the database, one per tier:\n\n        \nScoreReleaseTitle\n95v3.4.1Destination --chmod and --fake-super: chmod applied to fake-super, backup permissions lost75v3.4.3error in rsync protocol data stream (code 12) at token.c(490) [sender=3.4.3]55v3.4.1--dry-run doesn't work with --mkpath when copying files35v3.4.1manpages not installed in out-of-tree builds15v3.4.1Minor inconsistencies in options in manpage0v3.4.3PAM Support - Open for Discussion\n\n\n        Why the release is the unit of analysis\n\n        Why group commits by release, bugs by release, and then ascertain the correlation \u2014 or lack\n            thereof \u2014 between Claude commits and bugs through the intermediary of releases? This is for two\n            reasons.\n\n        First, because the claim that the critics are making is also, itself, made in terms of releases: that\n            having any Claude commits in a release makes the whole release more buggy as a whole in a noticeable way,\n            not just that Claude-authored commits may introduce more bugs; the latter is a different metric, because\n            later Claude- or human-authored commits could correct for those bugs within the same release, and\n            nobody would then notice as part of the release, and overall it wouldn't matter to users; additionally,\n            it's simply important, as stated elsewhere, to meet the claim of the critics where it's at. If this forces\n            them to make their claims more nuanced \u2014 or otherwise move the goalposts \u2014 then\n            mission accomplished.\n        \n\n        Second, it's a problem of attribution: the vast, vast majority of bugs do not state exactly which commit\n            caused them, because doing so would require extensive research and analysis that is often not worth it in\n            favor of simply fixing-forward, and even if that analysis was done \u2014 via something like\n            git bisect \u2014 it wouldn't necessarily result in anything useful, or anything at all.\n            Many bugs can result from a combination of multiple commits, often separated significantly over time,\n            where it's unclear whether one commit or the other really introduced the bug. Or, one commit can reveal\n            several latent bugs introduced by other commits at once, and so on.\n        \n\n        Why bugs and commits?\n\n        The critics' claim is simplistic, absolute, and universalistic: the rate of bugs in the Claude-exposed\n            releases went up. Therefore, the simplest honest response is to analyize precisely what is being claimed:\n            bugs, commits, releases, and Claude-exposed commits. If the Claude releases sit in the middle of the\n            historical distribution, the burden shifts to the critics to explain why this particular middle is somehow\n            worse than all the other middles that came before it. Even by weighting by severity, I feel that I am giving extensive generosity to the anti-AI point in all this, but enough of the more intelligent critics brought it up that I found it worth it.\n\n        Even if that results in is shifting the conversation\n            toward a more nuanced discussion of the quality and type and user impact of the bugs in\n            the releases, it will already have been a major win for the pro-AI crowd, and a shifting of the goalposts\n            for the anti-AI crowd, and then we can do further analysis based on that. And the ball's in the anti-AI\n            court for that game.\n\n        What this approach does not do\n\n        I'm aware that this metric does not control for commit complexity or security intensity. It is a blunt\n            instrument. But the critics' accusation is also\n            blunt: \"Claude is making things worse.\" A blunt instrument is what is required in response. Blood begets\n            blood.\n\n    \n\n    \n\n    \n\n    \n\n        4 \u00b7 Results\n\n        Claude Releases\n\n        Before we jump into deeper analysis, let's just look at the two Claude releases themselves, to get a sense\n            for them:\n\n        \n            v3.4.20.00 sev/10c0 bugs \u00b7 50 commits \u00b7 9 Claude0th percentile (rank 0 of 35)v3.4.33.29 sev/10c17 bugs \u00b7 34 commits \u00b7 28 Claude77th percentile (rank 27 of 35)\n        \n\n        If that doesn't look like a red flag to you, you'd be right.\n\n        Exact Permutation Test\n\n        So the question is: are the Claude releases unusually buggy, or could you easily pull a group just as bad\n            out of the historical distribution by dumb luck? The way you answer that question statistically is an\n            exact permutation test, which\n            just enumerates all pairs of two releases and asks: what fraction have a\n            mean bug rate as bad or worse than the one we actually observed? That fraction is the p-value of the\n            hypothesis under test.\n        \n\n        \n            46%\n            exact permutation test p-value (one-sided, H\u2081: Claude mean > historical)\n            \n            \n                272 of 595 possible groups of 2 historical releases have mean\n                sev/10c \u2265 1.65. Nearly half. The Claude releases sit right in the middle of the\n                permutation distribution \u2014 there is nothing extreme about them.\n                \n                \n                    Test statistic: mean sev/10c per group \u00b7 Claude group mean: 1.65 \u00b7 Historical mean:\n                    2.95\n                \n        \n\n        What this p-value tells us is that the hypothesis that Claude makes releases worse has, at least so far,\n            about as much predictive power as a coin flip: if you closed your eyes and picked 2\n            releases at random, you'd do as bad or\n            worse\n            nearly half the time. There's nothing unusual about the Claude group.\n\n        Fisher's Exact Test\n\n        The permutation test asks: how likely is it that a random group of releases scores as badly as the\n            Claude group? But there's another way to pose the question:\n            are Claude releases more likely than non-Claude releases to fall above the historical\n            median? That's a textbook 2\u00d72 contingency table, and the standard test for it is\n            Fisher's exact test.\n        \n\n        \n            \n                \n                    \n                    \u2264 median\n                    > median\n                \n            \n            \n                \n                    Non-Claude\n                    18\n                    17\n                \n                \n                    Claude\n                    1\n                    1\n                \n            \n        \n\n        \n            74%\n            one-sided p-value (H\u2081: Claude more likely above median)\n            \n                Fisher's exact test asks: if we split all releases at the historical median\n                (0.74 sev/10c), are these Claude releases significantly buggy than previous releases (more likely to land above the median)? With a p-value of 74%, the answer is a decisive no.\n                The odds ratio is 1.06 \u2014 essentially 1:1. Claude releases are\n                no more likely to be above the median than any other releases.\n                \n                \n                    Odds ratio: 1.06 \u00b7 Median: 0.74 sev/10c\n                \n        \n\n        To emphasize, this does not mean that all Claude releases in the future will not be more buggy. We don't have nearly enough data to build a model and extrapolate out like that, and that's not what a Fisher's exact test is for. The point that's being made here is that these specific releases are not at all notable; if no one had known they were AI, no one would have cared or noticed anything out of the ordinary, and there is no evidence with which to conclude that Claude made anything worse yet, unlike the objective, absolutist, universal claims made by critics.\n\n        The Distribution\n\n        In case you're not convinced, here's a visual aid, showing where these releases fall in the distribution of\n            all prior releases:\n\n        \n        0.010.1110100\n        \n         Historical\n             Claude\n            \n                \n                Middle 50% (IQR)\n            \n            \n                \n                Outside IQR\n            \n        \n        \n        \n            How to read this graph: Each dot is a release. The shaded green band is the\n            interquartile range \u2014\n            the middle 50% of historical releases, from 0.29 to 2.59 sev/10c.\n            The darker regions on either side are the lower and upper quarters.\n        \n\n        \n            This is another way of saying the same thing the previous two tests said, but more\n            intuitively: the Claude releases (green dots) bracket the IQR in opposite directions. v3.4.2, with zero real\n            bugs, sits just below the IQR; v3.4.3 sits just above it. They bracket the middle of the distribution\n            in opposite directions. Neither is a negative outlier, and since they're on either side of the IQR, there's\n            no evidence Claude usage bends releases in either direction.\n        \n\n        Commit Rate\n\n        One possible objection I've seen is that while perhaps the defect rate of Claude-authored commits is not\n            worse than human-authored ones, Claude sped up developmend so much \u2014 due to \"vibe coding\" perhaps \u2014 that the\n            total number of bugs in each release got too high for comfort anyway, in which case it doesn't matter that\n            the defect rate per commit isn't so bad, because that's not what downstream users experience. We can check\n            this, though:\n\n        \n            p=88%\n            exact permutation test: do Claude releases have more commits?\n            \n                Claude releases averaged 42 commits; non-Claude releases averaged\n                185. If you pick any 2 releases at random, you'd see as many\n                or more commits 88% of the time.\n            \n        \n\n        So it seems that Claude releases did not have meaningfully\n            more commits. If anything, they had a lot fewer, but of course, the data is underpowered to say that\n            for sure; just like the rest of this article, I'm only trying to show a startling lack of evidence\n            for the supposed massive harms of using Claude, not trying to prove there must have been none, or that it\n            was even \"beneficial\" in terms of focused releases.\n\n        But commit count is a blunt measure. A commit can be a one-line typo fix or a rewrite of the entire\n            protocol handler. What about lines of code changed? GitHub's compare API gives us additions and deletions\n            between each pair of consecutive tags:\n\n        \n            p=5%\n            exact permutation test: do Claude releases have more lines changed?\n            \n                Claude releases averaged 3756 lines changed; non-Claude releases averaged\n                696. If you pick any 2 releases at random, you'd see as many\n                or more lines changed only 5% of the time. Claude releases were larger\n                by this measure.\n            \n        \n\n        Lines changed went up by a lot. But did absolute bug numbers?\n\n        \n            p=77%\n            exact permutation test: do Claude releases have more severity-weighted bugs?\n            \n                Claude releases averaged 5.6 severity-weighted bugs; non-Claude releases averaged\n                14.9. If you pick any 2 releases at random, you'd see as many\n                or more severity-weighted bugs 77% of the time. More lines changed, but not more bugs.\n            \n        \n\n        So: the Claude releases changed way more lines of code than historical ones, but didn't have more bugs.\n            More code, same bugs. That's not what you'd expect if Claude were making things worse.\n\n        Regime Check\n\n        The obvious counterargument is that maybe earlier rsync releases were less in maintenence-mode, and so\n            had more bugs, but recent rsync releases have been more stable, so comparing the two Claude-exposed\n            releases to the full historical distribution is masking the fact that they're actually outliers for their\n            regime. Luckily, there's a way to test this statistically.\n\n        Yes, the historical mean (2.95 sev/10c) was driven by a bimodal distribution: v2.x releases\n            average 1.11 sev/10c; v3.x releases average 4.23. But even within the v3.x regime,\n            the Claude releases sit in the middle of the pack or better:\n\n        \n            \n  \n  \n  middle 50%\n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  \n  v3.4.2\n  \n  v3.4.3inside middle 50% \u2713\n        \n        0.010.1110100\n        \n         v3.x historical\n             Claude\n            \n                \n                Middle 50% (IQR)\n            \n        \n\n        \n\n        So the regime-shift argument doesn't just fail \u2014 it fails backwards. The v3.x era has a\n            much higher mean sev/10c than v2.x. If you restrict the comparison to v3.x only, Claude releases\n            don't stand out at all \u2014 and one of them is better than most. The only way to make Claude look\n            like an outlier is to compare it against a quieter era and then blame the shift on Claude, when\n            the data says the shift predates Claude entirely.\n\n        We can further test whether there are meaningfully different regimes in the version history, and thus\n            whether using the full historical data is valid, by doing a\n            runs test.\n            If such regimes existed, the runs test would detect non-random clustering:\n        \n\n        \n            p=0.060\n            Wald\u2013Wolfowitz runs test on 35 non-Claude releases\n            \n                13 runs observed (expected 18.5 under randomness, z=-1.88).\n                At p=0.060 this is marginal \u2014 not strong enough to reject randomness at the conventional\n                0.05 level, but close enough that it warrants the v3.x-only comparison above, which\n                already accounts for the v2.x/v3.x difference. Using the full historical baseline\n                is a conservative choice: it includes a noisier era, which makes it harder, not easier,\n                to show that Claude releases are normal.\n                \n                \n                    Observed runs: 13 \u00b7 Expected: 18.5 \u00b7 z=-1.88\n                \n        \n\n        The Pre-Claude Outlier\n\n        Here's my favorite part, though. Digging into the data, one of the first things that jumped out at me with\n            blinding clarity was that the worst release, by far, in rsync history was entirely prior to the\n                introduction of Claude:\n\n        \n            39.39\n            bugs per 10 commits \u2014 v3.4.1, no Claude\n            \n                The highest bug rate in the entire dataset. 59 bugs in 9 commits, a hotfix\n                release\n                the day after v3.4.0. It exceeds every other release by an order of magnitude.\n            \n        \n\n        And yet nobody noticed. There was no AI to blame so there was no GitHub\n            issue\n            with\n            300 comments, no death threats, no threats to fork or move to openrsync. A maintainer\n            shipped a broken release and fixed it, just like normal. The only thing that made v3.4.3\n            special was the availability of an enemy everyone had already decided to hate.\n\n        All Releases (chronological)\n\n        \n            \n                \n                    \n                        Release\n                        Bugs\n                        Sev\n                        Commits\n                        Claude\n                        Bugs/10c\n                        Sev/10c \u2195\n                        Percentile\n                    \n                \n                \n                    v2.4.621.21301.540.9254th percentile\n          v2.5.041.37300.550.1811th percentile\n          v2.5.141.86900.580.2620th percentile\n          v2.5.262.911700.510.2514th percentile\n          v2.5.453.52102.381.6469th percentile\n          v2.5.52211.68802.501.3263rd percentile\n          v2.5.6146.623900.590.2823rd percentile\n          v2.6.084.726700.300.189th percentile\n          v2.6.153.344400.110.080th percentile\n          v2.6.22916.817017.069.8894th percentile\n          v2.6.34922.738101.290.5934th percentile\n          v2.6.4229.476000.290.126th percentile\n          v2.6.5167.114601.100.4931st percentile\n          v2.6.7155.864900.230.093rd percentile\n          v2.6.8125.47201.670.7449th percentile\n          v2.6.95318.826102.030.7243rd percentile\n          v3.0.06427.990900.700.3126th percentile\n          v3.0.164.010200.590.4029th percentile\n          v3.0.2104.19011.114.5680th percentile\n          v3.0.32210.85504.001.9671st percentile\n          v3.1.017052.457102.980.9251st percentile\n          v3.1.16832.166010.304.8683rd percentile\n          v3.1.25518.45709.653.2274th percentile\n          v3.1.38530.961013.935.0786th percentile\n          v3.2.0227.830400.720.2517th percentile\n          v3.2.174.56301.110.7246th percentile\n          v3.2.2138.85802.241.5166th percentile\n          v3.2.39555.315706.053.5277th percentile\n          v3.2.42014.321300.940.6740th percentile\n          v3.2.595.55301.701.0457th percentile\n          v3.2.663.22802.141.1360th percentile\n          v3.2.78852.460014.678.7391st percentile\n          v3.3.04225.338011.056.6689th percentile\n          v3.4.064.06001.000.6637th percentile\n          v3.4.15935.59065.5639.3997th percentile\n          v3.4.200.05090.000.000th percentile\n          v3.4.31711.234285.003.2977th percentile\n          \n                \n            \n        \n\n    \n\n    \n\n    \n\n    \n\n        5 \u00b7 What the Data Is Consistent And Inconsistent With\n\n        \n\n            \n                \u2713\n                \n                    \"The Claude releases are statistically indistinguishable from historical\n                        releases\"\n                    One Claude release sits just below the IQR (v3.4.2, with zero real\n                        bugs), the other just above it.\n                        They bracket the middle of the distribution in opposite directions \u2014 neither is an outlier.\n                        The exact permutation test yields a p-value of 46% \u2014 pick any\n                        2 releases at random and you'd do as bad or worse nearly half the time.\n                        There is no signal of abnormality.\n                \n            \n\n            \n                \u2713\n                \n                    \"The outrage selected on a single tail event and narrativized it\"\n                    A Mastodon user noticed a regression in v3.4.3, saw Claude\n                        commits, and concluded causation. But v3.4.3 at 3.29 sev/10c is at\n                        the 77th percentile \u2014 elevated but not extreme. 8\n                        historical releases scored higher. The correlation is noise.\n                \n            \n            \n            \n                \u2717\n                \n                    \"Claude clearly made things worse\" &emdash; the main claim\n                    The Claude releases bracket the IQR \u2014  one below, one above. Neither is\n                        an outlier.\n                        There is no distributional evidence of harm. The claim rests entirely on a post-hoc correlation\n                        observed by a social media user.\n                \n            \n\n            \n                \u2717\n                \n                    \"Claude commits (in general) do not and will not make things worse\"\n                    This is a common misrepresentation of my claims here. I am not trying to extrapolate out into the future, and say something like \"in general, Claude won't make things worse\" or \"Claude will never make things worse.\" Instead, the point is this: there is no evidence of it having done so, and the two Claude releases we have currently are thoroughly unremarkable, so the outrage is totally unjustified.\n                \n            \n\n            \n                \u2717\n                \n                    \"The regressions speak for themselves\"\n                    v3.4.1 \u2014  a pre-Claude release \u2014  has the highest bug rate in the\n                        dataset (39.39 sev/10c). Nobody noticed, because there was no AI to be angry at.\n                        The regressions only \"speak\" when you ignore the historical distribution.\n                \n            \n\n            \n                \u2717\n                \n                    \"Just wait, more bugs will surface\"\n                    v3.4.3 has been out long enough that its rate\n                        (3.29) is already comparable to historical releases. The \"wait and see\"\n                        argument is an appeal to an unknowable future that shifts the burden of proof away from the\n                        critics. If more bugs surface, they will enter the distribution like every other release. There\n                        is no reason to expect a regime break.\n                \n            \n\n        \n\n    \n\n    \n\n    \n            Discussion, and Tridge's Response\n            So, why do people feel like they've been betrayed, and feel so sure that things have \"clearly\" gotten worse &emdash; that Claude \"broke rsync\" &emdash; when there is no evidence for this, and no data except two thoroughly unremarkable releases?\n            A lot of it is just sheer, blind outrage at the use of\n                LLMs. However, there are some confounders that might have caused people to feel that way:\n\n            On the HN thread, user zos_kia pointed at the confound directly:\n\n            \n                From a cursory look, it looks like a security fix in response to a CVE surfaced a coding error which\n                    has\n                    been present in the code since 2007. This is so banal that it's actually hilarious to see people lose their shit over it.\n                \u2014 zos_kia on Hacker News\n            \n\n            On Lobsters, user jbert spelled\n                    out the causal chain:\n\n            \n                The trigger for the increased volume of changes (and hence increased number of regressions) was the\n                    influx of (mostly) LLM-enabled security issues. i.e. the causal chain was: LLMs \u2192 more known security issues \u2192 more changes needed than usual \u2192 more\n                        regressions than usual.\n                \u2014 jbert on Lobsters\n            \n\n            Essentially, this isn't a \"Claude\" problem, it's a \"more security work\" problem, something that\n                Tridge himself confirmed in\n                his response, describing how a flood of AI-generated CVE reports forced rapid,\n                extensive\n                changes to rsync's attack surface.\n            \n\n            But, as with all things AI, it doesn't matter. In the end, the outrage isn't about whether rsync is worse\n                or better now, it's about people not liking AI, and arguing from a priori definitions, not\n                empirical results, to the desired conclusion: that AI is bad:\n\n            \n                Like I said, the author \"tried to balance security against feature regression.\" I don't dispute that\n                    he tried. I merely dispute that the chatbots are good at writing code; in fact, they are bad at\n                    writing code. If the author had approached these security bugs by hand with a mental model (a\n                        Naur theory!) which preserves their desired features and functionality then they would have\n                        caused fewer regressions...\n                \u2014 Corbin on\n                        Lobste.rs\n            \n\n            In response to this sweeping, absolute, causal claim made with no evidence \u2014 and in fact, counter to the\n                evidence \u2014 based on an old philosophical claim about the epistemology of programming, it is perhaps best\n                to leave the victim of this outrage himself with the final word:\n\n            \n                \u2026for the people saying things like \"I'm a PhD from xyz uni and I'm telling you LLMs are just\n                    stochastic tools that make everything up and the world will fall apart if you use them\", I'm here to\n                    tell you that you are out of date. The world of software engineering has changed dramatically in the\n                    last few months. The world of IT security and maintaining software in the face of the flood of\n                    reports has completely and utterly changed just in the last few weeks. Anything you learned about\n                    this stuff last year might as well be from another planet\u2026 Bottom line is I do know (well, roughly!)\n                    how LLMs work, but that doesn't make them not useful. It does mean you have to be cautious, but I am\n                    being cautious, or as cautious as I can be given my desire to be sailing and not dealing with a\n                    flood of gunk from so-called internet experts.\n                \u2014 Andrew\n                        Tridgell"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:31:19.765251+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://alexispurslane.github.io/rsync-analysis/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2026-06-06T05:31:19.086715+00:00",
                "index_texts": null,
                "output": "Did Claude Increase Bugs in rsync?",
                "pwd": "/data/archive/1780723852.876703",
                "schema": "ArchiveResult",
                "start_ts": "2026-06-06T05:31:19.059831+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://alexispurslane.github.io/rsync-analysis/']' timed out after 60 seconds",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "Did Claude Increase Bugs in rsync?",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1780723852.876703",
    "newest_archive_date": "2026-06-06T05:31:47.097428+00:00",
    "num_failures": 1,
    "num_outputs": 8,
    "oldest_archive_date": "2026-06-06T05:30:58.341796+00:00",
    "path": "/rsync-analysis/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KTDPK9KWE30B400E01N86XZR",
    "snapshot_id": "f5807c8e-4165-4e75-bbca-461d6a8377f8",
    "sources": [
        "/data/sources/1780723851-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1780723852.876703",
    "title": "Did Claude Increase Bugs in rsync?",
    "url": "https://alexispurslane.github.io/rsync-analysis/"
}