{
    "archive_path": "archive/1765073116.106155",
    "base_url": "bruceediger.com/posts/goofing-on-meta",
    "basename": "",
    "bookmarked_date": "2025-12-07 02:05",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/bruceediger.com/posts/goofing-on-meta",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=bruceediger.com",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "bruceediger.com",
    "downloaded_at": "2025-12-07T02:05:17.929731+00:00",
    "downloaded_datestr": "2025-12-07 02:05",
    "extension": "",
    "hash": "AK6DKM3Q0X39Q0HX5FWA",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://bruceediger.com/posts/goofing-on-meta/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-07T02:07:27.347934+00:00",
                "index_texts": null,
                "output": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://bruceediger.com/posts/goofing-on-meta/']' timed out after 60 seconds",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:06:27.249797+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://bruceediger.com/posts/goofing-on-meta/"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2025-12-07T02:05:54.886438+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:05:21.918010+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=bruceediger.com"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-07T02:05:21.379661+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:05:18.143498+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://bruceediger.com/posts/goofing-on-meta/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-07T02:05:21.737863+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:05:21.465102+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2025-12-07T02:06:10.218505+00:00",
                "index_texts": [
                    "Goofing on Meta's AI Crawler - Information Camouflage (https://fonts.gstatic.com) (//fonts.googleapis.com) (//fonts.gstatic.com) (https://fonts.googleapis.com/css?family=Open+Sans:400,400i,700) (/css/style.css) (/favicon.ico)  (/) (Information Camouflage)  Information Camouflage Building lifelong customer relationships    Menu  (/about/) About   (/advice/) Advice   (/blogroll/) Blogroll   (/contact/) Contact   (/cookie-policy/) Cookie Policy   (/more/) More   (/posts/) Posts       Goofing on Meta's AI Crawler   2025-11-12 (Last Modified: 2025-11-29)    Early in March 2025, I noticed that a web crawler with a user agent string of meta-externalagent/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/crawler)   was hitting my blog\u2019s machine at an unreasonable rate. 2025-11-29 : Hacker News and Slashdot (https://bruceediger.com/posts/tale-of-two-link-aggregators/) comparison for this post. I followed the URL and discovered this is what Meta uses to gather\npremium, human-generated content to train its LLMs.\nI found the rate of requests to be annoying. I already have a PHP program that creates the illusion of an (https://bruceediger.com/posts/anti-seo-infinite-website/) infinite website .\nI decided to answer any HTTP request that had \u201cmeta-externalagent\u201d\nin its user agent string with the contents of a bork.php generated file. I run the (https://httpd.apache.org/) Apache web server .\nTo feed bork.php generated content to Meta,\nI turned on (https://httpd.apache.org/docs/2.4/mod/mod_rewrite.html) mod_rewrite ,\nand put this in the relevant config file: RewriteEngine on\nRewriteCond %{HTTP_USER_AGENT} meta-externalagent\nRewriteRule  ^.*(\\?.*)*$ /bork.php [L]   That\u2019s as specific as I\u2019m willing to be,\ngiven how widely Apache configs vary. bork.php is a PHP program,\nand has a few (https://bruceediger.com/posts/anti-seo-infinite-website/#use-it) pre-reqs for running correctly. This worked brilliantly.\nMeta ramped up to requesting 270,000 URLs on May 30 and 31, 2025. (Meta&rsquo;s crawler requests per day)  Meta\u2019s crawler requests completely swamped any other traffic my blog/website got.\nI was essentially only feeding Meta\u2019s AI.\nAfter about 3 months, I got scared that Meta\u2019s insatiable\nconsumption of Super Great Pages about condiments,\nunderwear and circa 2010 C-List celebs would start costing me money.\nSo I switched to giving  \u201cmeta-externalagent\u201d a 404 status code.\nI decided to see how long it would take one of the highest valued\ncompanies in the world to decide to go away.\nThe answer is 5 months. Timeline Just to get this entirely out of the way: 2025-03-08 - approximate start date 2025-03-13 - well underway, first \u201ccombined\u201d format Apache log file I saved 2025-06-17 - started giving 404s to \u201cmeta-externalagent/1.1' 2025-10-23 - started giving 404s to \u201cfacebookexternalhit\u201d, \u201cmeta-externalagent\u201d,\n\u201cfacebookcatalog\u201d, \u201cmeta-externalads\u201d, \u201cmeta-externalfetcher\u201d, \u201cmeta-webindexer\u201d\ncrawlers, too, completely out of spite. 2025-11-10 - called an end to the experiment, starting writing this post.  Results From 2025-03-08 to 2025-06-17: 8898445 200 OK , issued from about 2025-03-08 until 2025-06-17 6225348 404 not found , issued from 2025-06-17 to 2025-11-10  Requested URLs My program bork.php generates HTML with links.\n25% of these links are in <img> tags.\nOne third of those have .png , .gif or .jpg suffixes each.\nAbout 20% of the <a> (anchor) tags are to external\nsites with randomly chosen names - they almost certainly do not exist in DNS.\nAbout 80% of <a> tags that bork.php generates\nwill not have a DNS name, so they\u2019re links back to my site. Of the links that are back to my site,\nrandomly-generated URLs have 11 suffixes (so called \u201cfile types\u201d).\nThose URLs are weighted towards .html suffixes.\nHere\u2019s what meta-externalagent crawlers asked for in\nrequests of my site: Suffix Expected % Requested %   cfm 7.1 12  gif 7.1 0  htm 7.1 10  html 28.5 40  jpg 7.1 0.4  jsp 7.1 12  mp3 7.1 10  png 7.1 0.4  shtml 7.1 13  tar.gz 7.1 0  torrent 7.1 none    I\u2019m not sure I\u2019m calculating the expected percentages correctly.\nThere are more ways for it to chose to make a .gif URL\nthan a .cfm URL, for example, but the \u201cExpected %\u201d column is roughly correct.\nThe actual proportions are far more interesting than\nany actual-to-expected correlation would be. The crawler asked for exactly 0 (zero) .torrent URLs,\nbut did ask for .mp3 URLs.\nIt heavily favors (87% of requests) URLs that indicate they contain text, .html , .htm and the like.\nIt heavily disfavors URLs ending in suffixes indicating an image. Since Meta is crawling the web to train its Large Language Model,\nit\u2019s not too surprising the crawler favors retrieving text. If I was paranoid, I\u2019d say that asking for .mp3 URLs way out of proportion\nfrom other non-text URLs indicates that Meta is looking for copyright violations.\nExcept why not retrieve indications of Torrenting in that case?\nIn any case, bork.php doesn\u2019t return appropriate content for non-text, non-image URLs,\njust a few randomly-chosen bytes, so I personally have nothing to fear. Note that the HTML and image file content is randomly generated.\nI believe I have no copyright on it at all.\nWhat Meta does with randomly generated content is on them, but my conscience is clear. Requesting IP Addresses Meta made requests from both IPv4 (230 addresses) and IPv6 (580 different) addresses.\nThe IPv6 addresses were all in the 2a03:2880::/29 block.\nThe IPv4 addresses were in 173.252.64.0/18, 57.141.0.0/24,\n66.220.144.0/20, 69.171.224.0/19, 69.63.176.0/20.\nMeta IPv6 addresses made 15115906 requests,\nwhile IPv4 addresses made 7887 requests.\nMeta distinctly prefers IPv6 addresses. The whois data on the addresses showed a variety\nof Facebook-related corporate entities as \u201cowners\u201d of the address ranges.\nMeta plays the \u201cwe\u2019re an Irish corporation\u201d game like all the other\ntech giants. Disturbing gap The 2025-08-19 to 2025-08-23 interval of very low Meta HTTP requests\nis not Meta\u2019s crawler stopping.\nI think something strange was going on at the data center hosting my VPS.\nApache on that VPS got very few requests from anywhere during those 5 days.\nThe systemd command journalctl doesn\u2019t show anything suspicious.\nThe VPS doesn\u2019t show a reboot during that period. I have set up (https://oss.oetiker.ch/smokeping/stats.en.html) Smokeping to track connectivity to my VPS.\nIt shows nothing amiss for the period. (Smokeping latency to my VPS)  Edit 2025-11-15 I nosed around in Smokeping a little more.\nThere was a problem 2025-08-21T23:30 -0600.\nLooks like for the 15 minutes between 23:30 and 24:45 -0600,\nSmokeping could read my server\u2019s PPP peer (CenturyLink Fiber),\nbut not my VPS. (zoom in on smokeping latency to my VPS)  Above, VPS Smokeping latency. (zoom in on smokeping latency to my PPP peer)  Above, PPP peer Smokeping latency. That\u2019s the only problem period I could find in Smokeping\nlatency charts for my VPS for the period 2025-08-19 to 2025-08-23 Confessions This is of course, completely unscientific.\nI forgot to write down the start date.\nI arbitrarily quit giving out fabricated, randomly-generated HTML\nin favor of 404s.\nMy infinite website program doesn\u2019t give out URLs that\nI could use for tracking,\nif and when some crawler asks for a URL\nmy Apache server it gave out previously.\nI didn\u2019t construct my \u201cinfinite website\u201d program so that I could\ndetermine what percentage of what kind of URLs it gave out. Exhortation This effort does show what a simple Linux guy with a $6 a month VPS can do.\nIf a lot of people ran scraper traps or junkyards,\nFacebook would have to behave properly,\nor behave a good deal more lawlessly.\nIf the latter,\nthe fig leaf of being a law-abiding citizen of The Internet would be removed. Meta is a terrible company.\nThey aren\u2019t being at all mannerly scraping everything.\nAt the very least,\nthe effects of copyright law on their use of human-written material is arguable.\nI feel that we should all give fake content to Meta\u2019s AI scraper,\nor (https://www.web.sp.am/) something similar .\nI believe that every time someone implements a scraper junkyard,\nit should be individual, highly customized, and idiosyncratic,\nin order to give the people at Meta, Google, OpenAI and others problems.\nI quit goofing on Meta because\nI was worried about costs of ridiculously high traffic to my $6-a-month VPS.\nI should probably have written my infinite website program\nwith some kind of rate limiting,\na fixed number of requests per day perhaps, and then give out 503s the rest of the day. bork.php already waits a randomly-chosen delay\nwith a mean of about 14 seconds\non each request.    (/tags/internet/) internet  (/tags/cybercrime/) cybercrime       (Bruce Ediger avatar)  About Bruce Ediger  Comitted to sharing healthy lifestyle ideas with the world   (/posts/pacman-fixup/) \u00ab\u2009Previous Pacman Fix up   (/posts/starship-troopers-6/) Next\u2009\u00bb Two podcasts reviewing Starship Troopers      \u00a9 2022-2025 Bruce Ediger. The Product of Inspiration, Pure Reason, And Intellect       "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:06:10.119937+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://bruceediger.com/posts/goofing-on-meta/"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2025-12-07T02:06:27.112873+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:06:11.456706+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://bruceediger.com/posts/goofing-on-meta/"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2025-12-07T02:06:10.035701+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:05:58.901000+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmpf3blynko",
                    "https://bruceediger.com/posts/goofing-on-meta/"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2025-12-07T02:05:58.372410+00:00",
                "index_texts": [
                    "Early in March 2025, I noticed that a web crawler with a user agent string of\nmeta-externalagent/1.1 (+https://developers.facebook.com/docs/sharing/webmasters/crawler)\nwas hitting my blog\u2019s machine at an unreasonable rate.\n2025-11-29: Hacker News and Slashdot comparison for this post.\nI followed the URL and discovered this is what Meta uses to gather\npremium, human-generated content to train its LLMs.\nI found the rate of requests to be annoying.\nI already have a PHP program that creates the illusion of an\ninfinite website.\nI decided to answer any HTTP request that had \u201cmeta-externalagent\u201d\nin its user agent string with the contents of a bork.php\ngenerated file.\nI run the Apache web server.\nTo feed bork.php generated content to Meta,\nI turned on mod_rewrite,\nand put this in the relevant config file:\nRewriteEngine on\nRewriteCond %{HTTP_USER_AGENT} meta-externalagent\nRewriteRule  ^.*(\\?.*)*$ /bork.php [L]\nThat\u2019s as specific as I\u2019m willing to be,\ngiven how widely Apache configs vary.\nbork.php is a PHP program,\nand has a few pre-reqs\nfor running correctly.\nThis worked brilliantly.\nMeta ramped up to requesting 270,000 URLs on May 30 and 31, 2025.\n\nMeta\u2019s crawler requests completely swamped any other traffic my blog/website got.\nI was essentially only feeding Meta\u2019s AI.\nAfter about 3 months, I got scared that Meta\u2019s insatiable\nconsumption of Super Great Pages about condiments,\nunderwear and circa 2010 C-List celebs would start costing me money.\nSo I switched to giving  \u201cmeta-externalagent\u201d a 404 status code.\nI decided to see how long it would take one of the highest valued\ncompanies in the world to decide to go away.\nThe answer is 5 months.\nTimeline\nJust to get this entirely out of the way:\n\n2025-03-08 - approximate start date\n2025-03-13 - well underway, first \u201ccombined\u201d format Apache log file I saved\n2025-06-17 - started giving 404s to \u201cmeta-externalagent/1.1'\n2025-10-23 - started giving 404s to \u201cfacebookexternalhit\u201d, \u201cmeta-externalagent\u201d,\n\u201cfacebookcatalog\u201d, \u201cmeta-externalads\u201d, \u201cmeta-externalfetcher\u201d, \u201cmeta-webindexer\u201d\ncrawlers, too, completely out of spite.\n2025-11-10 - called an end to the experiment, starting writing this post.\n\nResults\nFrom 2025-03-08 to 2025-06-17:\n\n8898445 200 OK, issued from about 2025-03-08 until 2025-06-17\n6225348 404 not found, issued from 2025-06-17 to 2025-11-10\n\nRequested URLs\nMy program bork.php generates HTML with links.\n25% of these links are in <img> tags.\nOne third of those have .png, .gif or .jpg suffixes each.\nAbout 20% of the <a> (anchor) tags are to external\nsites with randomly chosen names - they almost certainly do not exist in DNS.\nAbout 80% of <a> tags that bork.php generates\nwill not have a DNS name, so they\u2019re links back to my site.\nOf the links that are back to my site,\nrandomly-generated URLs have 11 suffixes (so called \u201cfile types\u201d).\nThose URLs are weighted towards .html suffixes.\nHere\u2019s what meta-externalagent crawlers asked for in\nrequests of my site:\n\n\n\nSuffix\nExpected %\nRequested %\n\n\n\n\ncfm\n7.1\n12\n\n\ngif\n7.1\n0\n\n\nhtm\n7.1\n10\n\n\nhtml\n28.5\n40\n\n\njpg\n7.1\n0.4\n\n\njsp\n7.1\n12\n\n\nmp3\n7.1\n10\n\n\npng\n7.1\n0.4\n\n\nshtml\n7.1\n13\n\n\ntar.gz\n7.1\n0\n\n\ntorrent\n7.1\nnone\n\n\n\nI\u2019m not sure I\u2019m calculating the expected percentages correctly.\nThere are more ways for it to chose to make a .gif URL\nthan a .cfm URL, for example, but the \u201cExpected %\u201d column is roughly correct.\nThe actual proportions are far more interesting than\nany actual-to-expected correlation would be.\nThe crawler asked for exactly 0 (zero) .torrent URLs,\nbut did ask for .mp3 URLs.\nIt heavily favors (87% of requests) URLs that indicate they contain text,\n.html, .htm and the like.\nIt heavily disfavors URLs ending in suffixes indicating an image.\nSince Meta is crawling the web to train its Large Language Model,\nit\u2019s not too surprising the crawler favors retrieving text.\nIf I was paranoid, I\u2019d say that asking for .mp3 URLs way out of proportion\nfrom other non-text URLs indicates that Meta is looking for copyright violations.\nExcept why not retrieve indications of Torrenting in that case?\nIn any case, bork.php doesn\u2019t return appropriate content for non-text, non-image URLs,\njust a few randomly-chosen bytes, so I personally have nothing to fear.\nNote that the HTML and image file content is randomly generated.\nI believe I have no copyright on it at all.\nWhat Meta does with randomly generated content is on them, but my conscience is clear.\nRequesting IP Addresses\nMeta made requests from both IPv4 (230 addresses) and IPv6 (580 different) addresses.\nThe IPv6 addresses were all in the 2a03:2880::/29 block.\nThe IPv4 addresses were in 173.252.64.0/18, 57.141.0.0/24,\n66.220.144.0/20, 69.171.224.0/19, 69.63.176.0/20.\nMeta IPv6 addresses made 15115906 requests,\nwhile IPv4 addresses made 7887 requests.\nMeta distinctly prefers IPv6 addresses.\nThe whois data on the addresses showed a variety\nof Facebook-related corporate entities as \u201cowners\u201d of the address ranges.\nMeta plays the \u201cwe\u2019re an Irish corporation\u201d game like all the other\ntech giants.\nDisturbing gap\nThe 2025-08-19 to 2025-08-23 interval of very low Meta HTTP requests\nis not Meta\u2019s crawler stopping.\nI think something strange was going on at the data center hosting my VPS.\nApache on that VPS got very few requests from anywhere during those 5 days.\nThe systemd command journalctl doesn\u2019t show anything suspicious.\nThe VPS doesn\u2019t show a reboot during that period.\nI have set up Smokeping\nto track connectivity to my VPS.\nIt shows nothing amiss for the period.\n\nEdit 2025-11-15 I nosed around in Smokeping a little more.\nThere was a problem 2025-08-21T23:30 -0600.\nLooks like for the 15 minutes between 23:30 and 24:45 -0600,\nSmokeping could read my server\u2019s PPP peer (CenturyLink Fiber),\nbut not my VPS.\n\nAbove, VPS Smokeping latency.\n\nAbove, PPP peer Smokeping latency.\nThat\u2019s the only problem period I could find in Smokeping\nlatency charts for my VPS for the period 2025-08-19 to 2025-08-23\nConfessions\nThis is of course, completely unscientific.\nI forgot to write down the start date.\nI arbitrarily quit giving out fabricated, randomly-generated HTML\nin favor of 404s.\nMy infinite website program doesn\u2019t give out URLs that\nI could use for tracking,\nif and when some crawler asks for a URL\nmy Apache server it gave out previously.\nI didn\u2019t construct my \u201cinfinite website\u201d program so that I could\ndetermine what percentage of what kind of URLs it gave out.\nExhortation\nThis effort does show what a simple Linux guy with a $6 a month VPS can do.\nIf a lot of people ran scraper traps or junkyards,\nFacebook would have to behave properly,\nor behave a good deal more lawlessly.\nIf the latter,\nthe fig leaf of being a law-abiding citizen of The Internet would be removed.\nMeta is a terrible company.\nThey aren\u2019t being at all mannerly scraping everything.\nAt the very least,\nthe effects of copyright law on their use of human-written material is arguable.\nI feel that we should all give fake content to Meta\u2019s AI scraper,\nor something similar.\nI believe that every time someone implements a scraper junkyard,\nit should be individual, highly customized, and idiosyncratic,\nin order to give the people at Meta, Google, OpenAI and others problems.\nI quit goofing on Meta because\nI was worried about costs of ridiculously high traffic to my $6-a-month VPS.\nI should probably have written my infinite website program\nwith some kind of rate limiting,\na fixed number of requests per day perhaps, and then give out 503s the rest of the day.\nbork.php already waits a randomly-chosen delay\nwith a mean of about 14 seconds\non each request."
                ],
                "output": "readability/",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:05:55.182804+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://bruceediger.com/posts/goofing-on-meta/"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-12-07T02:05:54.995547+00:00",
                "index_texts": null,
                "output": "Goofing on Meta's AI Crawler - Information Camouflage",
                "pwd": "/data/archive/1765073116.106155",
                "schema": "ArchiveResult",
                "start_ts": "2025-12-07T02:05:54.952868+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "TimeoutExpired: Command '['/usr/bin/curl', '--silent', '--location', '--compressed', '--proxy', 'socks5://tor-socks-proxy:9150', '--head', '--max-time', '60', '--user-agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)', 'https://web.archive.org/save/https://bruceediger.com/posts/goofing-on-meta/']' timed out after 60 seconds",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "Goofing on Meta's AI Crawler - Information Camouflage",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1765073116.106155",
    "newest_archive_date": "2025-12-07T02:06:27.249797+00:00",
    "num_failures": 1,
    "num_outputs": 8,
    "oldest_archive_date": "2025-12-07T02:05:18.143498+00:00",
    "path": "/posts/goofing-on-meta/",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01KBV8WPZ4F8A0ACBB01DNWXK6",
    "snapshot_id": "fc31a1a4-94e5-48dc-ad7c-b9c49b5e7666",
    "sources": [
        "/data/sources/1765073115-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1765073116.106155",
    "title": "Goofing on Meta's AI Crawler - Information Camouflage",
    "url": "https://bruceediger.com/posts/goofing-on-meta/"
}