{
    "archive_path": "archive/1762679350.080316",
    "base_url": "oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view",
    "basename": "view",
    "bookmarked_date": "2025-11-09 09:09",
    "canonical": {
        "archive_org_path": "https://web.archive.org/web/oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view",
        "dom_path": "output.html",
        "favicon_path": "favicon.ico",
        "git_path": "git/",
        "google_favicon_path": "https://www.google.com/s2/favicons?domain=oneuptime.com",
        "headers_path": "headers.json",
        "htmltotext_path": "htmltotext.txt",
        "index_path": "index.html",
        "media_path": "media/",
        "mercury_path": "mercury/content.html",
        "pdf_path": "output.pdf",
        "readability_path": "readability/content.html",
        "screenshot_path": "screenshot.png",
        "singlefile_path": "singlefile.html",
        "warc_path": "warc/",
        "wget_path": null
    },
    "domain": "oneuptime.com",
    "downloaded_at": "2025-11-09T09:09:13.168200+00:00",
    "downloaded_datestr": "2025-11-09 09:09",
    "extension": "",
    "hash": "WDZTK0MSREBZ04N5W9YP",
    "history": {
        "archive_org": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://web.archive.org/save/https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-11-09T09:09:56.271117+00:00",
                "index_texts": null,
                "output": "ArchiveError: Failed to find \"content-location\" URL header in Archive.org response.",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:48.668042+00:00",
                "status": "failed"
            }
        ],
        "dom": [
            {
                "cmd": [
                    "/usr/bin/chromium-browser",
                    "--proxy-server=socks5://tor-socks-proxy:9150",
                    "--disable-features=DarkMode",
                    "--run-all-compositor-stages-before-draw",
                    "--hide-scrollbars",
                    "--autoplay-policy=no-user-gesture-required",
                    "--no-first-run",
                    "--use-fake-ui-for-media-stream",
                    "--use-fake-device-for-media-stream",
                    "--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'",
                    "--headless=new",
                    "--no-sandbox",
                    "--no-zygote",
                    "--disable-dev-shm-usage",
                    "--disable-software-rasterizer",
                    "--disable-sync",
                    "--window-size=1440,2000",
                    "--user-agent=Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "--user-data-dir=/data/personas/Default/chrome_profile",
                    "--profile-directory=Default",
                    "--dump-dom",
                    "https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view"
                ],
                "cmd_version": "131.0.6778",
                "end_ts": "2025-11-09T09:09:35.751117+00:00",
                "index_texts": null,
                "output": "output.html",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:16.892520+00:00",
                "status": "succeeded"
            }
        ],
        "favicon": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--output",
                    "favicon.ico",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://www.google.com/s2/favicons?domain=oneuptime.com"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-11-09T09:09:16.523245+00:00",
                "index_texts": null,
                "output": "favicon.ico",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:13.200985+00:00",
                "status": "succeeded"
            }
        ],
        "git": [],
        "headers": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--head",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-11-09T09:09:16.783769+00:00",
                "index_texts": null,
                "output": "headers.json",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:16.546750+00:00",
                "status": "succeeded"
            }
        ],
        "htmltotext": [
            {
                "cmd": [
                    "(internal) archivebox.extractors.htmltotext",
                    "./{singlefile,dom}.html"
                ],
                "cmd_version": "0.8.5rc51",
                "end_ts": "2025-11-09T09:09:43.681486+00:00",
                "index_texts": [
                    "AWS to Bare Metal Two Years Later: Answering Your Toughest Questions About Leaving AWS  (https://fonts.googleapis.com) (https://fonts.gstatic.com) (https://avatars.githubusercontent.com) (//avatars.githubusercontent.com) (https://fonts.googleapis.com/css2?family=Inter:wght@100;200;300;400;500;600;700;800;900&display=swap) (/img/favicons/favicon.ico) (/img/favicons/apple-touch-icon.png) (/img/favicons/favicon-32x32.png) (/img/favicons/favicon-16x16.png) (/img/favicons/safari-pinned-tab.svg) (/img/ou-wb.svg) (/img/ou-wb.svg) (/img/hou-wb.svg) (/manifest.json) (https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view) (https://cdnjs.cloudflare.com/ajax/libs/highlight.js/11.9.0/styles/a11y-dark.min.css)  (/) OneUptime ()   Open menu     Products    (/product/status-page)   Status Pages Status Page helps you be more transparent with your customers\n                      and showcase reliability.   (/product/monitoring)   Monitoring Analyse uptime and performance of all of your resources.   (/product/incident-management)   Incident Management Protect revenue and improve customer experiences by resolving\n                      incidents faster.   (/product/on-call)   On-Call and Alerts Alert right people at the right time. Create on-call schedules\n                      and more.   (/product/logs-management)   Logs Management Fastest log software on the planet. Ingest logs from any\n                      source and search in seconds.   (/product/apm)   APM Monitor performance of any app, any service, any stack.   (/product/workflows)   Workflows Integrate with 5000+ different services and products without\n                      writing any code.    (/enterprise/demo)   Request Demo   (mailto:sales@oneuptime.com)   Contact Sales       (/pricing) Pricing (/enterprise/overview) Enterprise (/enterprise/demo) Request Demo More    (/docs)   Docs Learn more about OneUptime by reading our docs.   (/reference)   API Reference Connect OneUptime with the rest of your software stack.   (/blog)   Learning Resources and Blog Learn about observability and keep yourself updated.   (/support)   Help & Support Get all of your questions answered by contacting support.   (https://github.com/oneuptime/oneuptime)   GitHub Check the code out, create new feature requests all on GitHub.    (https://shop.oneuptime.com)   Merch Store Buy our merch and support open source development.   (/about)   About Us Learn more about why we are building OneUptime.   (/legal)   Legal Center See our terms, privacy, GDPR, SOC documents.        (/accounts) Sign\n          in (/accounts/register) Sign up    (OneUptime)  Close menu      (/product/status-page)   Status Page  (/product/monitoring)   Monitoring  (/product/incident-management)   Incident Management  (/product/on-call)   On-Call and Alerts  (/product/logs-management)   Logs Management  (/product/apm)   APM  (/product/workflows)   Workflows     (/pricing) Pricing (/enterprise/overview) Enterprise (/enterprise/demo) Request Demo (/support) Support  (/accounts/register) Sign\n            up Existing customer? (/accounts) Sign in       AWS to Bare Metal Two Years Later: Answering Your Toughest Questions About Leaving AWS  Two years after our AWS-to-bare-metal migration, we revisit the numbers, share what changed, and address the biggest questions from Hacker News and Reddit.  (https://github.com/devneelpatel) (Neel Patel) @devneelpatel  \u2022 Oct 29, 2025 \u2022   Reading time 9 min read  (/blog/tag/aws)  AWS   (/blog/tag/cloud)  Cloud   (/blog/tag/infrastructure)  Infrastructure   (/blog/tag/cost-optimization)  Cost Optimization       When we published (https://oneuptime.com/blog/post/2023-10-30-moving-from-aws-to-bare-metal/view) How moving from AWS to Bare-Metal saved us $230,000 /yr. in 2023, the story travelled far beyond our usual readership. The discussion threads on (https://news.ycombinator.com/item?id=38294569) Hacker News and (https://www.reddit.com/r/sysadmin/comments/17y6zbi/moving_from_aws_to_baremetal_saved_us_230000_yr/) Reddit were packed with sharp questions: did we skip Reserved Instances, how do we fail over a single rack, what about the people cost, and when is cloud still the better answer? This follow-up is our long-form reply. Over the last twenty-four months we: Ran the MicroK8s + Ceph stack in production for 730+ days with 99.993% measured availability. Added a second rack in Frankfurt, joined to our primary Paris cage over redundant DWDM, to kill the \u201csingle rack\u201d concern. Cut average customer-facing latency by 19% thanks to local NVMe and eliminating noisy neighbours. Reinvested the savings into buying bare metal AI servers to expand LLM-based alert / incident summarisation and auto code fixes based on log / traces and metrics in OneUptime.  Below we tackle the recurring themes from the community feedback, complete with the numbers we use internally. $230,000 / yr savings? That is just an engineers salary. In the US, it is. In the rest of the world. That's 2-5x engineers salary. We used to save $230,000 / yr but now the savings have exponentially grown. We now save over $1.2M / yr and we expect this to grow, as we grow as a business. \u201cWhy not just buy Savings Plans or Reserved Instances?\u201d We tried. Long answer: the maths still favoured bare metal once we priced everything in. We see a savings of over 76% if you compare our bare metal setup to AWS.  A few clarifications: Savings Plans do not reduce S3, egress, or Direct Connect. 37% off instances still leaves you paying list price for bandwidth, which was 22% of our AWS bill. EKS had an extra $1,260/month control-plane fee plus $600/month for NAT gateways. Those costs disappear once you run Kubernetes yourself. Our workload is 24/7 steady. We were already at >90% reservation coverage; there was no idle burst capacity to \u201cright size\u201d away. If we had the kind of bursty compute profile many commenters referenced, the choice would be different.  \u201cHow much did migration and ongoing ops really cost?\u201d We spent a week of engineers time (and that is the worst case estimate) on the initial migration, spread across SRE, platform, and database owners. Most of that time was work we needed anyway- formalising infrastructure-as-code, smoke testing charts, tightening backup policies. The incremental work that existed purely because of bare metal was roughly one week. Ongoing run-cost looks like this: Hands-on keyboard: ~24 engineer-hours/quarter across the entire platform team, including routine patching and firmware updates. That is comparable to the AWS time we used to burn on cost optimisation, IAM policy churn, and chasing deprecations and updating our VM's on AWS.  Remote hands: 2 interventions in 24 months (mainly disks). Mean response time: 27 minutes. We do not staff an on-site team. We rely on co-location provider to physically manage our rack. This means no traditional hardware admins.  Automation: We're now moving to Talos. We PXE boot with Tinkerbell, image with Talos, manage configs through Flux and Terraform, and run conformance suites before each Kubernetes upgrade. All of those tools also hardened our AWS estate, so they were not net-new effort.  The opportunity cost question from is fair. We track it the same way we track feature velocity: did the infra team ship less? The answer was \u201cno\u201d- our release cadence increased because we reclaimed few hours/month we used to spend in AWS \u201ccost council\u201d meetings. \u201cIsn\u2019t a single rack a single point of failure?\u201d We have multiple racks across two different DC / providers. We: Leased a secondary quarter rack in Frankfurt with a different provider and power utility. Currently: Deployed a second MicroK8s control plane, mirrored Ceph pools with asynchronous replication. Future: We're moving to Talos. Nothing against Microk8s, but we like the Talos way of managing the k8s cluster. Added isolated out-of-band management paths (4G / satellite) so we can reach the gear even during metro fibre events.  The AWS failover cluster we mentioned in 2023 still exists. We rehearse a full cutover quarterly using the same Helm releases we ship to customers. DNS failover remains the slowest leg (resolver caches can ignore TTL), so we added Anycast ingress via BGP with our transit provider to cut traffic shifting to sub-minute. \u201cWhat about hardware lifecycle and surprise CapEx?\u201d We amortise servers over five years, but we sized them with 2 \u00d7 AMD EPYC 9654 CPUs, 1 TB RAM, and NVMe sleds. At our current growth rate the boxes will hit CPU saturation before we hit year five. When that happens, the plan is to cascade the older gear into our regional analytics cluster (we use Posthog + Metabase for this) and buy a new batch. Thanks to the savings delta, we can refresh 40% of the fleet every 24 months and still spend less annually than the optimised AWS bill above. We also buy extended warranties from the OEM (Supermicro) and keep three cold spares in the cage. The hardware lasts 7-8 years and not 5, but we wtill count it as 5 to be very conservative.  \u201cAre you reinventing managed services?\u201d Another strong Reddit critique: why rebuild services AWS already offers? Three reasons we are comfortable with the trade: Portability is part of our product promise. OneUptime customers self-host in their own environments. Running the same open stack we ship (Postgres, Redis, ClickHouse, etc.) keeps us honest. We eun on Kubernetes and self-hosted customers run on Kubernetes as well.  Tooling maturity. Two years ago we relied on Terraform + EKS + RDS. Today we run MicroK8s (Talos in the future), Argo Rollouts, OpenTelemetry Collector, and Ceph dashboards. None of that is bespoke. We do not maintain a fork of anything. Selective cloud use. We still pay AWS for Glacier backups, CloudFront for edge caching, and short-lived burst capacity for load tests. Cloud makes sense when elasticity matters; bare metal wins when baseload dominates.  Managed services are phenomenal when you are short on expertise or need features beyond commodity compute. If we were all-in on DynamoDB streams or Step Functions we would almost certainly still be on AWS. \u201cHow do bandwidth and DoS scenarios work now?\u201d We committed to 5 Gbps 95th percentile across two carriers.  The same traffic on AWS egress would be 8x expensive in eu-west-1. For DDoS protection we front our ingress with Cloudflare.  \u201cHas reliability suffered?\u201d Short answer: No. Infact it was better than AWS (compared to recent AWS downtimes) We have 730+ days with 99.993% measured availability and we also escaped AWS region wide downtime that happened a week ago.  \u201cHow do audits and compliance work off-cloud now?\u201d We stayed SOC 2 Type II and ISO 27001 certified through the transition. The biggest deltas auditors cared about: Physical controls: We provide badge logs from the colo, camera footage on request, and quarterly access reviews. The colo already meets Tier III redundancy, so their reports roll into ours. Change management: Terraform plans, and now Talos machine configs give us immutable evidence of change. Auditors liked that more than AWS Console screenshots. Business continuity: We prove failover by moving workload to other DC.  If you are in a regulated space (HIPAA for instance), expect the paperwork to grow a little. We worked it in by leaning on the colo providers\u2019 standard compliance packets- they slotted straight into our risk register. \u201cWhy not stay in the cloud but switch providers?\u201d We priced Hetzner, OVH, Leaseweb, Equinix Metal, and AWS Outposts. The short version: Hyperscaler alternatives were cheaper on compute but still expensive on egress once you hit petabytes/month. Outposts also carried minimum commits that exceeded our needs. European dedicated hosts (Hetzner, OVH) are fantastic for lab clusters. The challenge was multi-100 TB Ceph clusters with redundant uplinks and smart-hands SLAs. Once we priced that tier, the savings narrowed. Equinix Metal got the closest, but bare metal on-demand still carried a 25-30% premium over our CapEx plan. Their global footprint is tempting; we may still use them for short-lived expansion.  Owning the hardware also let us plan power density (we run 15 kW racks) and reuse components. For our steady-state footprint, colocation won by a long shot. \u201cWhat does day-to-day toil look like now?\u201d We put real numbers to it because Reddit kept us honest: Weekly: Kernel and firmware patches (Talos makes this a redeploy), Ceph health checks,  Total time averages 1 hour/week on average over months.  Monthly: Kubernetes control plane upgrades in canary fashion. About 2 engineer-hours. We expect this to reduce when Talos kicks in. Quarterly: Disaster recovery drills, capacity planning, and contract audits with carriers. Roughly 12 hours across three engineers.  Total toil is ~14 engineer-hours/month, including prep. The AWS era had us spending similar time but on different work: chasing cost anomalies, expanding Security Hub exceptions, and mapping breaking changes in managed services. The toil moved; it did not multiply. \u201cDo you still use the cloud for anything substantial?\u201d Absolutely. Cloud still solves problems we would rather not own: Glacier keeps long-term log archives at a price point local object storage cannot match. CloudFront handles 14 edge PoPs we do not want to build. We terminate TLS at the edge for marketing assets and docs. We will soon move this to Cloudflare as they are cheaper. We spin up short-lived AWS environments for load testing.  So yes, we left AWS for the base workload, but we still swipe the corporate card when elasticity or geography outweighs fixed-cost savings. When the cloud is still the right answer It depends on your workload . We still recommend staying put if: Your usage pattern is spiky or seasonal and you can auto-scale to near zero between peaks. You lean heavily on managed services (Aurora Serverless, Kinesis, Step Functions) where the operational load is the value prop. You do not have the appetite to build a platform team comfortable with Kubernetes, Ceph, observability, and incident response.  Cloud-first was the right call for our first five years. Bare metal became the right call once our compute footprint, data gravity, and independence requirements stabilised. What is next We are working on a detailed runbook + Terraform module to help teams do capex forecasting for colo moves. Expect that on the blog later this year. A deep dive on Talos is in the queue, as requested by multiple folks in the HN thread.  Questions we did not cover? Let us know in the discussion threads- we are happy to keep sharing the gritty details. Related Reading:  (https://oneuptime.com/blog/post/2023-10-30-moving-from-aws-to-bare-metal/view) How moving from AWS to Bare-Metal saved us $230,000 /yr.  (https://oneuptime.com/blog/post/2025-02-01-datadog-dollars-why-monitoring-is-breaking-the-bank/view) Datadog Dollars: Why Your Monitoring Bill Is Breaking the Bank  (https://oneuptime.com/blog/post/2024-08-14-why-build-open-source-datadog/view) Why build open-source DataDog?    Share this article (Share on X) (https://twitter.com/intent/tweet?text=%20AWS%20to%20Bare%20Metal%20Two%20Years%20Later%3A%20Answering%20Your%20Toughest%20Questions%20About%20Leaving%20AWS&url=https%3A%2F%2Foneuptime.com%2Fblog%2Fpost%2F2025-10-29-aws-to-bare-metal-two-years-later%2Fview)    (Share on LinkedIn) (https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Foneuptime.com%2Fblog%2Fpost%2F2025-10-29-aws-to-bare-metal-two-years-later%2Fview)    (Discuss on Hacker News) (https://news.ycombinator.com/submitlink?u=https%3A%2F%2Foneuptime.com%2Fblog%2Fpost%2F2025-10-29-aws-to-bare-metal-two-years-later%2Fview&t=%20AWS%20to%20Bare%20Metal%20Two%20Years%20Later%3A%20Answering%20Your%20Toughest%20Questions%20About%20Leaving%20AWS)      (Neel Patel) Neel Patel @devneelpatel \u2022 Oct 29, 2025 \u2022 9 min read  Building reliable software at OneUptime. Follow along for more on observability & reliability. (https://github.com/devneelpatel) View GitHub       Our Commitment to Open Source  Everything we do at OneUptime is 100% open-source. You can contribute by writing a post just like this.\n                        Please check contributing guidelines (https://github.com/oneuptime/blog) here.  If you wish to contribute to this post, you can make edits and improve it (https://github.com/oneuptime/blog/tree/master/posts/2025-10-29-aws-to-bare-metal-two-years-later) here .        On this page $230,000 / yr savings? That is just an engineers salary. \u201cWhy not just buy Savings Plans or Reserved Instances?\u201d \u201cHow much did migration and ongoing ops really cost?\u201d \u201cIsn\u2019t a single rack a single point of failure?\u201d \u201cWhat about hardware lifecycle and surprise CapEx?\u201d \u201cAre you reinventing managed services?\u201d \u201cHow do bandwidth and DoS scenarios work now?\u201d \u201cHas reliability suffered?\u201d \u201cHow do audits and compliance work off-cloud now?\u201d \u201cWhy not stay in the cloud but switch providers?\u201d \u201cWhat does day-to-day toil look like now?\u201d \u201cDo you still use the cloud for anything substantial?\u201d When the cloud is still the right answer What is next          OneUptime is open source observability platform. Monitor, Observe, Debug, Resolve. Everything you need to build reliable software in one open source platform. Get started today. (/accounts/register) Get started (/enterprise/demo) Request Demo \u2192       We use cookies to enhance your browsing experience and provide\n            personalized content. By clicking \"Accept,\" you consent to the use of cookies. Our product uses both first-party and third-party cookies for session storage and for various other purposes. Please note that disabling certain cookies may affect the functionality and performance of our product. For more information about how we handle your data and cookies, please read our Privacy Policy. By continuing to use our site without changing your cookie settings, you agree to our use of cookies as\n            described above. See our (/legal/terms) terms and our (/legal/privacy) privacy policy  Accept\n                all Reject all    Footer (OneUptime) OneUptime is the complete observability platform. OneUptime is\n            open-source and always will be. (https://x.com/oneuptimehq) X    (https://github.com/oneuptime) GitHub    (https://www.reddit.com/r/oneuptimehq/) Reddit    (https://www.youtube.com/@OneUptimeHQ) YouTube      Products (/product/status-page) Status\n                    Page  (/product/incident-management) Incident Management  (/product/monitoring) Monitoring  (/product/on-call) On-Call and\n                    Alerts  (/product/workflows) Workflows    Enterprise (/enterprise/overview) Overview  (/enterprise/demo) Request\n                    Demo     Compare (/compare/statuspage.io) StatusPage.io  (/compare/incident.io) Incident.io  (/compare/pagerduty) PagerDuty  (/compare/pingdom) Pingdom    More (/docs) Docs  (/reference) API Reference  (/support) Help and Support  (/legal) Legal Center  (https://github.com/oneuptime/oneuptime) GitHub  (/blog) Blog  (https://github.com/OneUptime/interview) Careers  (/oss-friends) Open Source Friends  (https://shop.oneuptime.com) Merch Store  (/about) About Us       \u00a9 HackerBay, Inc. All rights reserved.   (https://github.com/oneuptime/interview) We're hiring remotely, worldwide!    See open roles \u2192        ()    "
                ],
                "output": "htmltotext.txt",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:43.566889+00:00",
                "status": "succeeded"
            }
        ],
        "media": [
            {
                "cmd": [
                    "/usr/local/bin/yt-dlp",
                    "--restrict-filenames",
                    "--trim-filenames",
                    "128",
                    "--write-description",
                    "--write-info-json",
                    "--write-annotations",
                    "--write-thumbnail",
                    "--no-call-home",
                    "--write-sub",
                    "--write-auto-subs",
                    "--convert-subs=srt",
                    "--yes-playlist",
                    "--continue",
                    "--no-abort-on-error",
                    "--ignore-errors",
                    "--geo-bypass",
                    "--add-metadata",
                    "--format=(bv*+ba/b)[filesize<=750m][filesize_approx<=?750m]/(bv*+ba/b)",
                    "--skip-download",
                    "--cache-dir=/data/yt-dlp-cache/",
                    "--cookies=/data/yt-dlp-cache/cookies.txt",
                    "--proxy=socks5://tor-socks-proxy:9150",
                    "--no-playlist",
                    "https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view"
                ],
                "cmd_version": "2024.10.7",
                "end_ts": "2025-11-09T09:09:48.637181+00:00",
                "index_texts": [],
                "output": "media/",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:43.952567+00:00",
                "status": "succeeded"
            }
        ],
        "mercury": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/postlight-parser",
                    "https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view"
                ],
                "cmd_version": "2.2.3",
                "end_ts": "2025-11-09T09:09:43.511669+00:00",
                "index_texts": null,
                "output": "mercury/",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:39.571004+00:00",
                "status": "succeeded"
            }
        ],
        "pdf": [],
        "readability": [
            {
                "cmd": [
                    "/home/archivebox/.npm/bin/readability-extractor",
                    "/tmp/tmp5es5dt7x",
                    "https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view"
                ],
                "cmd_version": "0.0.11",
                "end_ts": "2025-11-09T09:09:39.298610+00:00",
                "index_texts": [
                    "When we published How moving from AWS to Bare-Metal saved us $230,000 /yr. in 2023, the story travelled far beyond our usual readership. The discussion threads on Hacker News and Reddit were packed with sharp questions: did we skip Reserved Instances, how do we fail over a single rack, what about the people cost, and when is cloud still the better answer? This follow-up is our long-form reply.Over the last twenty-four months we:Ran the MicroK8s + Ceph stack in production for 730+ days with 99.993% measured availability.Added a second rack in Frankfurt, joined to our primary Paris cage over redundant DWDM, to kill the \u201csingle rack\u201d concern.Cut average customer-facing latency by 19% thanks to local NVMe and eliminating noisy neighbours.Reinvested the savings into buying bare metal AI servers to expand LLM-based alert / incident summarisation and auto code fixes based on log / traces and metrics in OneUptime.Below we tackle the recurring themes from the community feedback, complete with the numbers we use internally.$230,000 / yr savings? That is just an engineers salary.In the US, it is. In the rest of the world. That's 2-5x engineers salary. We used to save $230,000 / yr but now the savings have exponentially grown. We now save over $1.2M / yr and we expect this to grow, as we grow as a business.\u201cWhy not just buy Savings Plans or Reserved Instances?\u201dWe tried. Long answer: the maths still favoured bare metal once we priced everything in. We see a savings of over 76% if you compare our bare metal setup to AWS. A few clarifications:Savings Plans do not reduce S3, egress, or Direct Connect. 37% off instances still leaves you paying list price for bandwidth, which was 22% of our AWS bill.EKS had an extra $1,260/month control-plane fee plus $600/month for NAT gateways. Those costs disappear once you run Kubernetes yourself.Our workload is 24/7 steady. We were already at >90% reservation coverage; there was no idle burst capacity to \u201cright size\u201d away. If we had the kind of bursty compute profile many commenters referenced, the choice would be different.\u201cHow much did migration and ongoing ops really cost?\u201dWe spent a week of engineers time (and that is the worst case estimate) on the initial migration, spread across SRE, platform, and database owners. Most of that time was work we needed anyway- formalising infrastructure-as-code, smoke testing charts, tightening backup policies. The incremental work that existed purely because of bare metal was roughly one week.Ongoing run-cost looks like this:Hands-on keyboard: ~24 engineer-hours/quarter across the entire platform team, including routine patching and firmware updates. That is comparable to the AWS time we used to burn on cost optimisation, IAM policy churn, and chasing deprecations and updating our VM's on AWS. Remote hands: 2 interventions in 24 months (mainly disks). Mean response time: 27 minutes. We do not staff an on-site team. We rely on co-location provider to physically manage our rack. This means no traditional hardware admins. Automation: We're now moving to Talos. We PXE boot with Tinkerbell, image with Talos, manage configs through Flux and Terraform, and run conformance suites before each Kubernetes upgrade. All of those tools also hardened our AWS estate, so they were not net-new effort.The opportunity cost question from is fair. We track it the same way we track feature velocity: did the infra team ship less? The answer was \u201cno\u201d- our release cadence increased because we reclaimed few hours/month we used to spend in AWS \u201ccost council\u201d meetings.\u201cIsn\u2019t a single rack a single point of failure?\u201dWe have multiple racks across two different DC / providers. We:Leased a secondary quarter rack in Frankfurt with a different provider and power utility.Currently: Deployed a second MicroK8s control plane, mirrored Ceph pools with asynchronous replication. Future: We're moving to Talos. Nothing against Microk8s, but we like the Talos way of managing the k8s cluster.Added isolated out-of-band management paths (4G / satellite) so we can reach the gear even during metro fibre events.The AWS failover cluster we mentioned in 2023 still exists. We rehearse a full cutover quarterly using the same Helm releases we ship to customers. DNS failover remains the slowest leg (resolver caches can ignore TTL), so we added Anycast ingress via BGP with our transit provider to cut traffic shifting to sub-minute.\u201cWhat about hardware lifecycle and surprise CapEx?\u201dWe amortise servers over five years, but we sized them with 2 \u00d7 AMD EPYC 9654 CPUs, 1 TB RAM, and NVMe sleds. At our current growth rate the boxes will hit CPU saturation before we hit year five. When that happens, the plan is to cascade the older gear into our regional analytics cluster (we use Posthog + Metabase for this) and buy a new batch. Thanks to the savings delta, we can refresh 40% of the fleet every 24 months and still spend less annually than the optimised AWS bill above.We also buy extended warranties from the OEM (Supermicro) and keep three cold spares in the cage. The hardware lasts 7-8 years and not 5, but we wtill count it as 5 to be very conservative. \u201cAre you reinventing managed services?\u201dAnother strong Reddit critique: why rebuild services AWS already offers? Three reasons we are comfortable with the trade:Portability is part of our product promise. OneUptime customers self-host in their own environments. Running the same open stack we ship (Postgres, Redis, ClickHouse, etc.) keeps us honest. We eun on Kubernetes and self-hosted customers run on Kubernetes as well. Tooling maturity. Two years ago we relied on Terraform + EKS + RDS. Today we run MicroK8s (Talos in the future), Argo Rollouts, OpenTelemetry Collector, and Ceph dashboards. None of that is bespoke. We do not maintain a fork of anything.Selective cloud use. We still pay AWS for Glacier backups, CloudFront for edge caching, and short-lived burst capacity for load tests. Cloud makes sense when elasticity matters; bare metal wins when baseload dominates.Managed services are phenomenal when you are short on expertise or need features beyond commodity compute. If we were all-in on DynamoDB streams or Step Functions we would almost certainly still be on AWS.\u201cHow do bandwidth and DoS scenarios work now?\u201dWe committed to 5 Gbps 95th percentile across two carriers.  The same traffic on AWS egress would be 8x expensive in eu-west-1. For DDoS protection we front our ingress with Cloudflare. \u201cHas reliability suffered?\u201dShort answer: No. Infact it was better than AWS (compared to recent AWS downtimes)We have 730+ days with 99.993% measured availability and we also escaped AWS region wide downtime that happened a week ago. \u201cHow do audits and compliance work off-cloud now?\u201dWe stayed SOC 2 Type II and ISO 27001 certified through the transition. The biggest deltas auditors cared about:Physical controls: We provide badge logs from the colo, camera footage on request, and quarterly access reviews. The colo already meets Tier III redundancy, so their reports roll into ours.Change management: Terraform plans, and now Talos machine configs give us immutable evidence of change. Auditors liked that more than AWS Console screenshots.Business continuity: We prove failover by moving workload to other DC.If you are in a regulated space (HIPAA for instance), expect the paperwork to grow a little. We worked it in by leaning on the colo providers\u2019 standard compliance packets- they slotted straight into our risk register.\u201cWhy not stay in the cloud but switch providers?\u201dWe priced Hetzner, OVH, Leaseweb, Equinix Metal, and AWS Outposts. The short version:Hyperscaler alternatives were cheaper on compute but still expensive on egress once you hit petabytes/month. Outposts also carried minimum commits that exceeded our needs.European dedicated hosts (Hetzner, OVH) are fantastic for lab clusters. The challenge was multi-100 TB Ceph clusters with redundant uplinks and smart-hands SLAs. Once we priced that tier, the savings narrowed.Equinix Metal got the closest, but bare metal on-demand still carried a 25-30% premium over our CapEx plan. Their global footprint is tempting; we may still use them for short-lived expansion.Owning the hardware also let us plan power density (we run 15 kW racks) and reuse components. For our steady-state footprint, colocation won by a long shot.\u201cWhat does day-to-day toil look like now?\u201dWe put real numbers to it because Reddit kept us honest:Weekly: Kernel and firmware patches (Talos makes this a redeploy), Ceph health checks,  Total time averages 1 hour/week on average over months. Monthly: Kubernetes control plane upgrades in canary fashion. About 2 engineer-hours. We expect this to reduce when Talos kicks in.Quarterly: Disaster recovery drills, capacity planning, and contract audits with carriers. Roughly 12 hours across three engineers.Total toil is ~14 engineer-hours/month, including prep. The AWS era had us spending similar time but on different work: chasing cost anomalies, expanding Security Hub exceptions, and mapping breaking changes in managed services. The toil moved; it did not multiply.\u201cDo you still use the cloud for anything substantial?\u201dAbsolutely. Cloud still solves problems we would rather not own:Glacier keeps long-term log archives at a price point local object storage cannot match.CloudFront handles 14 edge PoPs we do not want to build. We terminate TLS at the edge for marketing assets and docs. We will soon move this to Cloudflare as they are cheaper.We spin up short-lived AWS environments for load testing.So yes, we left AWS for the base workload, but we still swipe the corporate card when elasticity or geography outweighs fixed-cost savings.When the cloud is still the right answerIt depends on your workload. We still recommend staying put if:Your usage pattern is spiky or seasonal and you can auto-scale to near zero between peaks.You lean heavily on managed services (Aurora Serverless, Kinesis, Step Functions) where the operational load is the value prop.You do not have the appetite to build a platform team comfortable with Kubernetes, Ceph, observability, and incident response.Cloud-first was the right call for our first five years. Bare metal became the right call once our compute footprint, data gravity, and independence requirements stabilised.What is nextWe are working on a detailed runbook + Terraform module to help teams do capex forecasting for colo moves. Expect that on the blog later this year.A deep dive on Talos is in the queue, as requested by multiple folks in the HN thread.Questions we did not cover? Let us know in the discussion threads- we are happy to keep sharing the gritty details.Related Reading:"
                ],
                "output": "readability/",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:36.511727+00:00",
                "status": "succeeded"
            }
        ],
        "screenshot": [],
        "singlefile": [],
        "title": [
            {
                "cmd": [
                    "/usr/bin/curl",
                    "--silent",
                    "--location",
                    "--compressed",
                    "--proxy",
                    "socks5://tor-socks-proxy:9150",
                    "--max-time",
                    "60",
                    "--user-agent",
                    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36 ArchiveBox/{VERSION} (+https://github.com/ArchiveBox/ArchiveBox/)",
                    "https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view"
                ],
                "cmd_version": "8.10.1",
                "end_ts": "2025-11-09T09:09:36.426437+00:00",
                "index_texts": null,
                "output": "AWS to Bare Metal Two Years Later: Answering Your Toughest Questions About Leaving AWS",
                "pwd": "/data/archive/1762679350.080316",
                "schema": "ArchiveResult",
                "start_ts": "2025-11-09T09:09:36.309789+00:00",
                "status": "succeeded"
            }
        ],
        "wget": []
    },
    "icons": null,
    "is_archived": true,
    "is_static": false,
    "latest": {
        "archive_org": "ArchiveError: Failed to find \"content-location\" URL header in Archive.org response.",
        "dom": "output.html",
        "favicon": "favicon.ico",
        "git": null,
        "media": "media/",
        "pdf": null,
        "screenshot": null,
        "singlefile": null,
        "title": "AWS to Bare Metal Two Years Later: Answering Your Toughest Questions About Leaving AWS",
        "warc": null,
        "wget": null
    },
    "link_dir": "/data/archive/1762679350.080316",
    "newest_archive_date": "2025-11-09T09:09:48.668042+00:00",
    "num_failures": 1,
    "num_outputs": 8,
    "oldest_archive_date": "2025-11-09T09:09:13.200985+00:00",
    "path": "/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view",
    "schema": "Link",
    "scheme": "https",
    "snapshot_abid": "snp_01K9KY0RVYD89C21E0018Q8TJV",
    "snapshot_id": "20350ddf-bfe6-48e6-8ce9-eddf11746a5b",
    "sources": [
        "/data/sources/1762679349-import.txt"
    ],
    "tags": null,
    "tags_str": "",
    "timestamp": "1762679350.080316",
    "title": "AWS to Bare Metal Two Years Later: Answering Your Toughest Questions About Leaving AWS",
    "url": "https://oneuptime.com/blog/post/2025-10-29-aws-to-bare-metal-two-years-later/view"
}