534 lines
46 KiB
HTML
534 lines
46 KiB
HTML
<!DOCTYPE html>
|
||
<html lang="en">
|
||
<head>
|
||
|
||
<title>The Always On Architecture - Moving Beyond Legacy Disaster Recovery - High Scalability -</title>
|
||
<meta charset="utf-8">
|
||
<meta name="viewport" content="width=device-width, initial-scale=1.0">
|
||
|
||
<link rel="preload" as="style" href="https://highscalability.com/assets/built/screen.css?v=QpNBjlSbNzyRyl6s">
|
||
<link rel="preload" as="script" href="https://highscalability.com/assets/built/source.js?v=uMULaSMoSjM4gY5C">
|
||
|
||
<link rel="preload" as="font" type="font/woff2" href="https://highscalability.com/assets/fonts/inter-roman.woff2?v=OecsB5TBLy27FKD2" crossorigin="anonymous">
|
||
<style>
|
||
@font-face {
|
||
font-family: "Inter";
|
||
font-style: normal;
|
||
font-weight: 100 900;
|
||
font-display: optional;
|
||
src: url(https://highscalability.com/assets/fonts/inter-roman.woff2?v=OecsB5TBLy27FKD2) format("woff2");
|
||
unicode-range: U+0000-00FF, U+0131, U+0152-0153, U+02BB-02BC, U+02C6, U+02DA, U+02DC, U+0304, U+0308, U+0329, U+2000-206F, U+2074, U+20AC, U+2122, U+2191, U+2193, U+2212, U+2215, U+FEFF, U+FFFD;
|
||
}
|
||
</style>
|
||
|
||
<link rel="stylesheet" type="text/css" href="https://highscalability.com/assets/built/screen.css?v=QpNBjlSbNzyRyl6s">
|
||
|
||
<style>
|
||
:root {
|
||
--background-color: #ffffff
|
||
}
|
||
</style>
|
||
|
||
<script>
|
||
/* The script for calculating the color contrast has been taken from
|
||
https://gomakethings.com/dynamically-changing-the-text-color-based-on-background-color-contrast-with-vanilla-js/ */
|
||
var accentColor = getComputedStyle(document.documentElement).getPropertyValue('--background-color');
|
||
accentColor = accentColor.trim().slice(1);
|
||
|
||
if (accentColor.length === 3) {
|
||
accentColor = accentColor[0] + accentColor[0] + accentColor[1] + accentColor[1] + accentColor[2] + accentColor[2];
|
||
}
|
||
|
||
var r = parseInt(accentColor.substr(0, 2), 16);
|
||
var g = parseInt(accentColor.substr(2, 2), 16);
|
||
var b = parseInt(accentColor.substr(4, 2), 16);
|
||
var yiq = ((r * 299) + (g * 587) + (b * 114)) / 1000;
|
||
var textColor = (yiq >= 128) ? 'dark' : 'light';
|
||
|
||
document.documentElement.className = `has-${textColor}-text`;
|
||
</script>
|
||
|
||
<link rel="canonical" href="https://highscalability.com/the-always-on-architecture-moving-beyond-legacy-disaster-rec/">
|
||
<meta name="referrer" content="no-referrer-when-downgrade">
|
||
|
||
<meta property="og:site_name" content="High Scalability">
|
||
<meta property="og:type" content="article">
|
||
<meta property="og:title" content="The Always On Architecture - Moving Beyond Legacy Disaster Recovery - High Scalability -">
|
||
<meta property="og:description" content="Failover does not cut it anymore. You need an ALWAYS ON architecture with multiple data center...">
|
||
<meta property="og:url" content="https://highscalability.com/the-always-on-architecture-moving-beyond-legacy-disaster-rec/">
|
||
<meta property="og:image" content="https://c1.staticflickr.com/9/8509/28579160224_44325cecec_o.jpg">
|
||
<meta property="article:published_time" content="2016-08-23T22:42:06.000Z">
|
||
<meta property="article:modified_time" content="2016-08-23T22:42:06.000Z">
|
||
<meta property="article:tag" content="Strategy">
|
||
|
||
<meta property="article:publisher" content="https://www.facebook.com/ghost">
|
||
<meta name="twitter:card" content="summary_large_image">
|
||
<meta name="twitter:title" content="The Always On Architecture - Moving Beyond Legacy Disaster Recovery - High Scalability -">
|
||
<meta name="twitter:description" content="Failover does not cut it anymore. You need an ALWAYS ON architecture with multiple data centers.-- Martin Van Ryswyk, VP of Engineering at DataStax
|
||
|
||
Failover, switching to a redundant or standby system when a component fails, has a long and checkered history as a way of dealing with failure. The">
|
||
<meta name="twitter:url" content="https://highscalability.com/the-always-on-architecture-moving-beyond-legacy-disaster-rec/">
|
||
<meta name="twitter:image" content="https://static.ghost.org/v5.0.0/images/publication-cover.jpg">
|
||
<meta name="twitter:label1" content="Written by">
|
||
<meta name="twitter:data1" content="High Scalability">
|
||
<meta name="twitter:label2" content="Filed under">
|
||
<meta name="twitter:data2" content="Strategy">
|
||
<meta name="twitter:site" content="@ghost">
|
||
<meta name="twitter:creator" content="@highscal">
|
||
|
||
<script type="application/ld+json">
|
||
{
|
||
"@context": "https://schema.org",
|
||
"@type": "Article",
|
||
"publisher": {
|
||
"@type": "Organization",
|
||
"name": "High Scalability",
|
||
"url": "https://highscalability.com/",
|
||
"logo": {
|
||
"@type": "ImageObject",
|
||
"url": "https://highscalability.com/favicon.ico",
|
||
"width": 48,
|
||
"height": 48
|
||
}
|
||
},
|
||
"author": {
|
||
"@type": "Person",
|
||
"name": "High Scalability",
|
||
"image": {
|
||
"@type": "ImageObject",
|
||
"url": "https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/2024/03/hs.jpeg",
|
||
"width": 400,
|
||
"height": 400
|
||
},
|
||
"url": "https://highscalability.com/author/hs/",
|
||
"sameAs": [
|
||
"https://blog.bytebytego.com/",
|
||
"https://x.com/highscal"
|
||
]
|
||
},
|
||
"headline": "The Always On Architecture - Moving Beyond Legacy Disaster Recovery - High Scalability -",
|
||
"url": "https://highscalability.com/the-always-on-architecture-moving-beyond-legacy-disaster-rec/",
|
||
"datePublished": "2016-08-23T22:42:06.000Z",
|
||
"dateModified": "2016-08-23T22:42:06.000Z",
|
||
"keywords": "Strategy",
|
||
"description": "Failover does not cut it anymore. You need an ALWAYS ON architecture with multiple data centers.-- Martin Van Ryswyk, VP of Engineering at DataStax\n\nFailover, switching to a redundant or standby system when a component fails, has a long and checkered history as a way of dealing with failure. The reason is your failover mechanism becomes a single point of failure that often fails just when it's needed most. Having worked on a few telecom systems that used a failover strategy I know exactly how st",
|
||
"mainEntityOfPage": "https://highscalability.com/the-always-on-architecture-moving-beyond-legacy-disaster-rec/"
|
||
}
|
||
</script>
|
||
|
||
<meta name="generator" content="Ghost 6.64">
|
||
<link rel="alternate" type="application/rss+xml" title="High Scalability" href="https://highscalability.com/rss/">
|
||
<script defer src="https://cdn.jsdelivr.net/ghost/portal@~2.71/umd/portal.min.js" data-i18n="true" data-ghost="https://highscalability.com/" data-key="e7388272b3fcb33a0abfc2f95c" data-api="https://high-scalability.ghost.io/ghost/api/content/" data-locale="en" crossorigin="anonymous"></script><style id="gh-members-styles">.gh-post-upgrade-cta-content,
|
||
.gh-post-upgrade-cta {
|
||
display: flex;
|
||
flex-direction: column;
|
||
align-items: center;
|
||
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, Oxygen, Ubuntu, Cantarell, 'Open Sans', 'Helvetica Neue', sans-serif;
|
||
text-align: center;
|
||
width: 100%;
|
||
color: #ffffff;
|
||
font-size: 16px;
|
||
}
|
||
|
||
.gh-post-upgrade-cta-content {
|
||
border-radius: 8px;
|
||
padding: 40px 4vw;
|
||
}
|
||
|
||
.gh-post-upgrade-cta h2 {
|
||
color: #ffffff;
|
||
font-size: 28px;
|
||
letter-spacing: -0.2px;
|
||
margin: 0;
|
||
padding: 0;
|
||
}
|
||
|
||
.gh-post-upgrade-cta p {
|
||
margin: 20px 0 0;
|
||
padding: 0;
|
||
}
|
||
|
||
.gh-post-upgrade-cta small {
|
||
font-size: 16px;
|
||
letter-spacing: -0.2px;
|
||
}
|
||
|
||
.gh-post-upgrade-cta a {
|
||
color: #ffffff;
|
||
cursor: pointer;
|
||
font-weight: 500;
|
||
box-shadow: none;
|
||
text-decoration: underline;
|
||
}
|
||
|
||
.gh-post-upgrade-cta a:hover {
|
||
color: #ffffff;
|
||
opacity: 0.8;
|
||
box-shadow: none;
|
||
text-decoration: underline;
|
||
}
|
||
|
||
.gh-post-upgrade-cta a.gh-btn {
|
||
display: block;
|
||
background: #ffffff;
|
||
text-decoration: none;
|
||
margin: 28px 0 0;
|
||
padding: 8px 18px;
|
||
border-radius: 4px;
|
||
font-size: 16px;
|
||
font-weight: 600;
|
||
}
|
||
|
||
.gh-post-upgrade-cta a.gh-btn:hover {
|
||
opacity: 0.92;
|
||
}</style>
|
||
<script defer src="https://cdn.jsdelivr.net/ghost/sodo-search@~1.8/umd/sodo-search.min.js" data-key="e7388272b3fcb33a0abfc2f95c" data-styles="https://cdn.jsdelivr.net/ghost/sodo-search@~1.8/umd/main.css" data-sodo-search="https://high-scalability.ghost.io/" data-locale="en" crossorigin="anonymous"></script>
|
||
|
||
<link href="https://highscalability.com/webmentions/receive/" rel="webmention">
|
||
<script defer src="/public/cards.min.js?v=ShRHxgy4po8zN-Wf"></script>
|
||
<link rel="stylesheet" type="text/css" href="/public/cards.min.css?v=WwnU9jw5ancNC8Gc">
|
||
<script defer src="/public/comment-counts.min.js?v=oFYkaGLdiMqB8VN9" data-ghost-comments-counts-api="https://highscalability.com/members/api/comments/counts/"></script>
|
||
<script defer src="/public/member-attribution.min.js?v=AKG4hWena9j3yX3I"></script>
|
||
<script defer src="/public/ghost-stats.min.js?v=vFcCUf6ZQ0Hyhc8h" data-stringify-payload="false" data-datasource="analytics_events" data-storage="localStorage" data-host="https://highscalability.com/.ghost/analytics/api/v1/page_hit" tb_site_uuid="647204cd-7ad2-4539-b98a-3489074b932d" tb_post_uuid="75f5d63b-5702-4244-915f-e358f0fe262e" tb_post_type="post" tb_member_uuid="undefined" tb_member_status="undefined" tb_gift_link=""></script><style>:root {--ghost-accent-color: #35cea0;}</style>
|
||
<!-- Fathom - beautiful, simple website analytics -->
|
||
<script src="https://cdn.usefathom.com/script.js" data-site="XBRJSZNU" defer></script>
|
||
<!-- / Fathom -->
|
||
<style>
|
||
/* Hide feature image on single post page in Source theme */
|
||
.post-template .gh-article-image {
|
||
display: none;
|
||
}
|
||
.gh-footer-copyright { display: none; }
|
||
</style>
|
||
|
||
</head>
|
||
<body class="post-template tag-strategy tag-hash-sqs has-sans-title has-sans-body">
|
||
|
||
<div class="gh-viewport">
|
||
|
||
<header id="gh-navigation" class="gh-navigation is-middle-logo gh-outer">
|
||
<div class="gh-navigation-inner gh-inner">
|
||
|
||
<div class="gh-navigation-brand">
|
||
<a class="gh-navigation-logo is-title" href="https://highscalability.com">
|
||
High Scalability
|
||
</a>
|
||
<button class="gh-search gh-icon-button" aria-label="Search this site" data-ghost-search>
|
||
<svg xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 24 24" stroke="currentColor" stroke-width="2" width="20" height="20"><path stroke-linecap="round" stroke-linejoin="round" d="M21 21l-6-6m2-5a7 7 0 11-14 0 7 7 0 0114 0z"></path></svg></button> <button class="gh-burger gh-icon-button" aria-label="Menu">
|
||
<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" fill="currentColor" viewBox="0 0 256 256"><path d="M224,128a8,8,0,0,1-8,8H40a8,8,0,0,1,0-16H216A8,8,0,0,1,224,128ZM40,72H216a8,8,0,0,0,0-16H40a8,8,0,0,0,0,16ZM216,184H40a8,8,0,0,0,0,16H216a8,8,0,0,0,0-16Z"></path></svg> <svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" fill="currentColor" viewBox="0 0 256 256"><path d="M205.66,194.34a8,8,0,0,1-11.32,11.32L128,139.31,61.66,205.66a8,8,0,0,1-11.32-11.32L116.69,128,50.34,61.66A8,8,0,0,1,61.66,50.34L128,116.69l66.34-66.35a8,8,0,0,1,11.32,11.32L139.31,128Z"></path></svg> </button>
|
||
</div>
|
||
|
||
<nav class="gh-navigation-menu">
|
||
<ul class="nav">
|
||
<li class="nav-home"><a href="https://highscalability.com/">Home</a></li>
|
||
<li class="nav-system-design-interview-course"><a href="https://bit.ly/hishscalcourse">System Design Interview Course</a></li>
|
||
</ul>
|
||
|
||
</nav>
|
||
|
||
<div class="gh-navigation-actions">
|
||
<button class="gh-search gh-icon-button" aria-label="Search this site" data-ghost-search>
|
||
<svg xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 24 24" stroke="currentColor" stroke-width="2" width="20" height="20"><path stroke-linecap="round" stroke-linejoin="round" d="M21 21l-6-6m2-5a7 7 0 11-14 0 7 7 0 0114 0z"></path></svg></button> <div class="gh-navigation-members">
|
||
<a href="#/portal/signin" data-portal="signin">Sign in</a>
|
||
<a class="gh-button" href="#/portal/signup" data-portal="signup">Subscribe</a>
|
||
</div>
|
||
</div>
|
||
|
||
</div>
|
||
</header>
|
||
|
||
|
||
|
||
<main class="gh-main">
|
||
|
||
<article class="gh-article post tag-strategy tag-hash-sqs no-image">
|
||
|
||
<header class="gh-article-header gh-canvas">
|
||
|
||
<a class="gh-article-tag" href="https://highscalability.com/tag/strategy/">Strategy</a>
|
||
<h1 class="gh-article-title is-title">The Always On Architecture - Moving Beyond Legacy Disaster Recovery</h1>
|
||
|
||
<div class="gh-meta-share">
|
||
<div class="gh-article-meta">
|
||
<div class="gh-article-author-image instapaper_ignore">
|
||
<a href="/author/hs/">
|
||
<img class="author-profile-image" src="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w160/2024/03/hs.jpeg" alt="High Scalability">
|
||
</a>
|
||
</div>
|
||
<div class="gh-article-meta-wrapper">
|
||
<h4 class="gh-article-author-name"><a href="/author/hs/">High Scalability</a></h4>
|
||
<div class="gh-article-meta-content">
|
||
<time class="gh-article-meta-date" datetime="2016-08-23">23 Aug 2016</time>
|
||
<span class="gh-article-meta-length"><span class="bull">—</span> 8 min read</span>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<a href="#/share" class="gh-button gh-button-share">
|
||
Share
|
||
</a>
|
||
</div>
|
||
|
||
|
||
</header>
|
||
|
||
<section class="gh-content gh-canvas is-body">
|
||
<figure class="kg-card kg-image-card"><img src="https://c1.staticflickr.com/9/8509/28579160224_44325cecec_o.jpg" class="kg-image" alt loading="lazy"></figure><blockquote>Failover does not cut it anymore. You need an ALWAYS ON architecture with multiple data centers.-- <a href="https://twitter.com/dsmvr/status/763084322152681472?ref=highscalability.com">Martin Van Ryswyk</a>, VP of Engineering at DataStax</blockquote><p><a href="https://en.wikipedia.org/wiki/Failover?ref=highscalability.com">Failover</a>, switching to a redundant or standby system when a component fails, has a long and checkered history as a way of dealing with failure. The reason is your failover mechanism becomes a single point of failure that often fails just when it's needed most. Having worked on a few telecom systems that used a failover strategy I know exactly how stressful failover events can be and how stupid you feel when your failover fails. If you have a double or triple fault in your system failover is exactly the time when it will happen.</p><p>For a long time the only real trick we had for achieving fault tolerance was to have a hot, warm, or cold standby (disk, interface, card, server, router, generator, datacenter, etc.) and failover to it when there's a problem. This old style of <a href="http://www.cioupdate.com/trends/article.php/3589851/Four-Steps-to-a-Successful-DR-Strategy.htm?ref=highscalability.com">Disaster Recovery</a> planning is no longer adequate or necessary.</p><p>Now, thanks to cloud infrastructures, at least at a software system level, we have an alternative: <strong>an always on architecture</strong>. Google calls this a <a href="https://highscalability.com/googles-transition-from-single-datacenter-to-failover-to-a-n/">natively multihomed architecture</a>. You can distribute data across multiple datacenters in such away that all your datacenters are always active. Each datacenter can automatically scale capacity up and down depending on what happens to other datacenters. You know, the usual sort of cloud propaganda. Robin Schumacher makes a good case here: <a href="http://www.datastax.com/2016/08/dear-cxo-when-will-what-happened-to-delta-happen-to-you?ref=highscalability.com">Long live Dear CXO – When Will What Happened to Delta Happen to You?</a></p><h2 id="recent-problems-with-disaster-recovery">Recent Problems With Disaster !Recovery</h2><p>Southwest had a service disruption <a href="http://www.nbcnews.com/business/travel/outdated-technology-likely-culprit-southwest-airlines-outage-n443176?ref=highscalability.com">a year ago</a> and again was recently bit by a "<a href="http://www.dallasnews.com/business/airline-industry/20160730-southwest-ceo-router-failure-that-grounded-flights-equated-to-once-in-a-thousand-year-flood.ece?ref=highscalability.com">once in a thousand-year event</a>" that caused a service outage (doesn't it seem like once in 1000 year events happen a lot more lately?). The first incident was blamed on "legacy systems" that could no longer handle the increased load of a now much larger airline. The most recent problem was caused by a router partially failed so the failover mechanism didn't kick in. 2,300 flights were canceled over four days at cost of perhaps $10 million. When you do your system's engineering review do you consider partial failures? Probably not. Yet they happen and are notoriously hard to detect and deal with.</p><p>Sprint has also experienced bad <a href="https://www.consumeraffairs.com/news/sprint-joins-southwest-delta-in-bad-backup-derby-081716.html?ref=highscalability.com">backup problems</a>:</p><blockquote>Sprint said a fire in D.C. caused problems at Sprint's data center in Reston, Va. How a fire across the street from Sprint's switch in D.C. caused issues 20 miles away wasn't quite clear, but apparently, emergency Sprint generators in D.C. didn't kick in as they were supposed to and, as so often happens, one thing led to another.</blockquote><p>And unless you were on Mars, you will have heard Delta recently <a href="http://money.cnn.com/2016/08/08/news/companies/delta-system-outage-flights/index.html?iid=EL&ref=highscalability.com">experienced</a> their <a href="http://www.flyertalk.com/forum/27032000-post135.html?ref=highscalability.com">own failover problems</a>:</p><blockquote>According to the flight captain of JFK-SLC this morning, a routine scheduled switch to the backup generator this morning at 2:30am caused a fire that destroyed both the backup and the primary. Firefighters took a while to extinguish the fire. Power is now back up and 400 out of the 500 servers rebooted, still waiting for the last 100 to have the whole system fully functional</blockquote><p>Delta has come under a lot of criticism. Why was the backup generator so close to the primary that a fire could destroy both? Why is the entire worldwide system running in a single datacenter? Why don't they test more? Why don't they have full failover to a backup datacenter? Why are they more interested in cutting costs the building a reliable system? Why do they still use those old mainframes? Why does a company that earns $42 billion a year have such crappy systems? It's only 500 servers, why not convert it to a cluster already? Why does management only care about their bonuses and cutting IT costs? Isn't that what you get from years out of outsourcing IT?</p><p>There's a lot of venom, as you might expect. If you want a little background on Delta's systems then here's a good article: <a href="http://www.wsj.com/articles/SB10001424052702303480304579575891541812918?ref=highscalability.com">Delta Air Lines to Take Control of Its Data Systems</a>. It appears that as of 2014 Delta spun in all its passenger-service and flight operations systems. They had 180 proprietary Delta technology applications controlling their ticketing, website, flight check-in, crew scheduling and more. And they spent about $250 million a year on IT.</p><h2 id="does-the-whole-system-need-a-refactoring">Does the whole system need a refactoring?</h2><p>Interesting comment on the technology involved in these systems from <a href="http://www.cringely.com/2016/08/08/outsourced-probably-hurt-delta-airlines-power-went/?ref=highscalability.com">FormerRTCoverageAA</a>:</p><blockquote>My advice is for ALL the major airlines to each put in about 10 million dollars (20-30 airlines would put a fund together about 200-300 million) to modernize and work on the Interfaces between them, and the hotel and car rental systems, tours, and other functions that SABRE/Amadeus/Apollo/etc. interface to. This would fund a research consortium to look at the current technology, and DEFINE THE INTERFACES for the next generation system. Maybe HP and IBM and Microsoft and whoever else wants to play could put in some money too. The key for this consortium is to have the INTERFACES defined. Give the specifications to the vendors (HP, IBM, Microsoft, Google, Priceline, Hilton, Hertz, whoever) that want to build the next generation reservations system. Then let them have 1 year and all have to work to inter-operate on the specification (just like they do on the “old” specs today for things like teletype, and last seat availability).</blockquote><blockquote>This has worked well in the healthcare space in getting payers and providers to work together. Each potential vendor needs to plan to spend 10-50 million dollars on their proposed solution. Then, we have the inter-operability technology fair (I would make it 2 weeks to 1 month) and each vendor can pitch to each airline, car rental, hotel, tour company, Uber, etc. Let each vendor do what he wants as long as the requirements for the specifications are met. Let the best tech vendor win.</blockquote><blockquote>It’s far past time to update these systems. Otherwise, more heartache pain and probably government bailouts to come. Possibly even larger travel and freight interruptions. A longer term blow up could put an airline out of business. Remember Eastern? I do….</blockquote><p>This all sounds like a great idea, but what could Delta do with its own architecture?</p><h2 id="the-always-on-architecture">The Always on Architecture</h2><p>Earlier this year I wrote on article on a paper from Google: <a href="https://static.googleusercontent.com/media/research.google.com/en//pubs/archive/44686.pdf?ref=highscalability.com">High-Availability at Massive Scale: Building Google’s Data Infrastructure for Ads</a> that explains their history with Always On. It seems appropriate. Here's the article:</p><figure class="kg-card kg-image-card"><img src="https://c2.staticflickr.com/2/1510/25190447426_c478660902_o.png" class="kg-image" alt loading="lazy"></figure><p>The main idea of the paper is that the typical <a href="https://en.wikipedia.org/wiki/Failover?ref=highscalability.com">failover</a> architecture used when moving from a single datacenter to multiple datacenters doesn’t work well in practice. What does work, where work means using fewer resources while providing high availability and consistency, is a <strong>natively multihomed architecture</strong>:</p><blockquote>Our current approach is to build natively multihomed systems. Such systems <strong>run hot in multiple datacenters all the time, and adaptively move load between datacenters</strong>, with the ability to handle outages of any scale completely transparently. Additionally, planned datacenter outages and maintenance events are completely transparent, causing minimal disruption to the operational systems. In the past, such events required labor-intensive efforts to move operational systems from one datacenter to another</blockquote><p>The use of “multihoming” in this context may be confusing because <a href="https://en.wikipedia.org/wiki/Multihoming?ref=highscalability.com">multihoming</a> usually refers to a computer connected to more than one network. At Google scale perhaps it’s just as natural to talk about connecting to multiple datacenters.</p><p>Google has built several multi-homed systems to guarantee high availability (4 to 5 nines) and consistency in the presence of datacenter level outages: <a href="http://research.google.com/pubs/pub38125.html?ref=highscalability.com">F1 / Spanner: Relational Database</a>; <a href="http://research.google.com/pubs/pub41318.html?ref=highscalability.com">Photon: Joining Continuous Data Streams</a>; <a href="http://research.google.com/pubs/pub42851.html?ref=highscalability.com">Mesa: Data Warehousing</a>. The approach taken by each of these systems is discussed in the paper, as are the many challenges is building a multi-homed system: Synchronous Global State; What to Checkpoint; Repeatable Input; Exactly Once Output.</p><p>The huge constraint here is <strong>having availability and consistency</strong>. This highlights the refreshing and continued emphasis Google puts on making even these complex systems <a href="https://highscalability.com/google-spanners-most-surprising-revelation-nosql-is-out-and/">easy for programmers to use</a>:</p><blockquote>The simplicity of a multi-homed system is particularly valuable for users. Without multi-homing, failover, recovery, and dealing with inconsistency are all application problems. With multi-homing, these hard problems are solved by the infrastructure, so the application developer gets high availability and consistency for free and can focus instead on building their application.</blockquote><p>The biggest surprise in the paper was the idea that a <strong>multihomed system can actually take far fewer resources than a failover system</strong>:</p><blockquote>In a multi-homed system deployed in three datacenters with 20% total catchup capacity, the total resource footprint is 170% of steady state. This is dramatically less than the 300% required in the failover design above</blockquote><h2 id="what-s-wrong-with-failover">What’s Wrong With Failover?</h2><blockquote>Failover-based approaches, however, do not truly achieve high availability, and can have excessive cost due to the deployment of standby resources.</blockquote><blockquote>Our teams have had several bad experiences dealing with failover-based systems in the past. Since unplanned outages are rare, failover procedures were often added as an afterthought, not automated and not well tested. On multiple occasions, teams spent days recovering from an outage, bringing systems back online component by component, recovering state with ad hoc tools like custom MapReduces, and gradually tuning the system as it tried to catch up processing the backlog starting from the initial outage. These situations not only cause extended unavailability, but are also extremely stressful for the teams running complex mission-critical systems.</blockquote><h2 id="how-do-multihomed-systems-work">How do Multihomed Systems Work?</h2><blockquote>In contrast, multi-homed systems are designed to run in multiple datacenters as a core design property, so there is no on-the-side failover concept. A multi-homed system runs live in multiple datacenters all the time. Each datacenter processes work all the time, and work is dynamically shared between datacenters to balance load. When one datacenter is slow, some fraction of work automatically moves to faster datacenters. When a datacenter is completely unavailable, all its work is automatically distributed to other datacenters.</blockquote><blockquote>There is no failover process other than the continuous dynamic load balancing. Multi-homed systems coordinate work across datacenters using shared global state that must be updated synchronously. All critical system state is replicated so that any work can be restarted in an alternate datacenter at any point, while still guaranteeing exactly once semantics. Multi-homed systems are uniquely able to provide high availability and full consistency in the presence of datacenter level failures.</blockquote><blockquote>In any of our typical streaming system, the events being processed are based on user interactions, and logged by systems serving user traffic in many datacenters around the world. A log collection service gathers these logs globally and copies them to two or more specific logs datacenters. Each logs datacenter gets a complete copy of the logs, with the guarantee that all events copied to any one datacenter will (eventually) be copied to all logs datacenters. The stream processing systems run in one or more of the logs datacenters and processes all events. Output from the stream processing system is usually stored into some globally replicated system so that the output can be consumed reliably from anywhere.</blockquote><blockquote>In a multi-homed system, all datacenters are live and processing all the time. Deploying three datacenters is typical. In steady state, each of the three datacenters process 33% of the traffic. After a failure where one datacenter is lost, the two remaining datacenters each process 50% of the traffic.</blockquote><p>Obviously Delta and other companies with extensive legacy systems are in a difficult position for this kind of approach. But if you consider IT something other than a cost center, and you plan to stay around for the long haul, and whole nations rely on the quality of your infrastructure, it's probably something you should consider. We have the technology.</p><h2 id="related-articles">Related Articles</h2><ul><li><a href="https://news.ycombinator.com/item?id=12353230&ref=highscalability.com">On HackerNews</a></li><li><a href="http://spectrum.ieee.org/static/lessons-from-a-decade-of-it-failures?ref=highscalability.com">Lessons From a Decade of IT Failures</a></li><li><a href="https://github.com/danluu/post-mortems?ref=highscalability.com">A List of Post-mortems!</a></li><li><a href="http://news.delta.com/ceo-apologizes-customers-flight-schedule-recovery-continues?ref=highscalability.com">CEO apologizes to customers; flight schedule recovery continues</a></li><li><a href="http://www.transtats.bts.gov/ot_delay/OT_DelayCause1.asp?ref=highscalability.com">Airline On-Time Statistics and Delay Causes</a></li><li><a href="http://www.ntsb.gov/_layouts/ntsb.aviation/index.aspx?ref=highscalability.com">Aviation Accident Database & Synopses</a></li><li><a href="http://searchdatacenter.techtarget.com/news/450302485/Delta-outage-raises-backup-data-center-power-questions?ref=highscalability.com">Delta outage raises backup data center, power questions</a></li><li><a href="http://t5datacenters.com/data-center-design-what-we-can-learn-from-delta-airlines-data-center-outage/?ref=highscalability.com">Data Center Design | What We Can Learn From Delta Airlines’ Data Center Outage</a></li><li><a href="http://www.foxnews.com/story/2004/12/25/comair-cancels-all-1100-flights.html?ref=highscalability.com">Comair Cancels All 1,100 Flights</a></li><li><a href="https://highscalability.com/f1-and-spanner-holistically-compared/">F1 And Spanner Holistically Compared</a></li><li><a href="https://highscalability.com/google-spanners-most-surprising-revelation-nosql-is-out-and/">Google Spanner's Most Surprising Revelation: NoSQL Is Out And NewSQL Is In</a></li><li><a href="https://highscalability.com/spanner-its-about-programmers-building-apps-using-sql-semant/">Spanner - It's About Programmers Building Apps Using SQL Semantics At NoSQL Scale</a></li><li><a href="https://highscalability.com/the-three-ages-of-google-batch-warehouse-instant/">The Three Ages Of Google - Batch, Warehouse, Instant</a></li></ul>
|
||
</section>
|
||
|
||
</article>
|
||
|
||
<div class="gh-comments gh-canvas">
|
||
|
||
<script defer src="https://cdn.jsdelivr.net/ghost/comments-ui@~1.6/umd/comments-ui.min.js" data-locale="en" data-ghost-comments="https://highscalability.com/" data-api="https://high-scalability.ghost.io/ghost/api/content/" data-admin="https://high-scalability.ghost.io/ghost/" data-key="e7388272b3fcb33a0abfc2f95c" data-title="null" data-count="true" data-post-id="65bceaca3953980001659985" data-color-scheme="auto" data-avatar-saturation="60" data-accent-color="#35cea0" data-comments-enabled="all" data-publication="High Scalability" crossorigin="anonymous"></script>
|
||
|
||
</div>
|
||
|
||
</main>
|
||
|
||
|
||
<section class="gh-container is-grid gh-outer">
|
||
<div class="gh-container-inner gh-inner">
|
||
<h2 class="gh-container-title">Read more</h2>
|
||
<div class="gh-feed">
|
||
<article class="gh-card post">
|
||
<a class="gh-card-link" href="/untitled-2/">
|
||
<figure class="gh-card-image">
|
||
<img
|
||
srcset="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w160/format/webp/2024/05/pasted-image-0-2.png 160w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w320/format/webp/2024/05/pasted-image-0-2.png 320w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w600/format/webp/2024/05/pasted-image-0-2.png 600w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w960/format/webp/2024/05/pasted-image-0-2.png 960w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w1200/format/webp/2024/05/pasted-image-0-2.png 1200w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w2000/format/webp/2024/05/pasted-image-0-2.png 2000w"
|
||
sizes="320px"
|
||
src="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w600/2024/05/pasted-image-0-2.png"
|
||
alt="Kafka 101"
|
||
loading="lazy"
|
||
>
|
||
</figure>
|
||
<div class="gh-card-wrapper">
|
||
<h3 class="gh-card-title is-title">Kafka 101</h3>
|
||
<p class="gh-card-excerpt is-body">This is a guest article by Stanislav Kozlovski, an Apache Kafka Committer. If you would like to connect with Stanislav, you can do so on Twitter and LinkedIn.
|
||
|
||
Originally developed in LinkedIn during 2011, Apache Kafka is one of the most popular open-source Apache projects out there. So far</p>
|
||
<footer class="gh-card-meta">
|
||
<!--
|
||
-->
|
||
<span class="gh-card-author">By ByteByteGo</span>
|
||
<time class="gh-card-date" datetime="2024-05-09">09 May 2024</time>
|
||
<!--
|
||
--></footer>
|
||
</div>
|
||
</a>
|
||
</article>
|
||
<article class="gh-card post">
|
||
<a class="gh-card-link" href="/capturing-a-billion-emo-j-i-ons/">
|
||
<figure class="gh-card-image">
|
||
<img
|
||
srcset="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w160/format/webp/2024/03/1-rSRWALA4XzOdDcn-5vv7Zw.gif 160w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w320/format/webp/2024/03/1-rSRWALA4XzOdDcn-5vv7Zw.gif 320w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w600/format/webp/2024/03/1-rSRWALA4XzOdDcn-5vv7Zw.gif 600w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w960/format/webp/2024/03/1-rSRWALA4XzOdDcn-5vv7Zw.gif 960w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w1200/format/webp/2024/03/1-rSRWALA4XzOdDcn-5vv7Zw.gif 1200w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w2000/format/webp/2024/03/1-rSRWALA4XzOdDcn-5vv7Zw.gif 2000w"
|
||
sizes="320px"
|
||
src="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w600/2024/03/1-rSRWALA4XzOdDcn-5vv7Zw.gif"
|
||
alt="Capturing A Billion Emo(j)i-ons"
|
||
loading="lazy"
|
||
>
|
||
</figure>
|
||
<div class="gh-card-wrapper">
|
||
<h3 class="gh-card-title is-title">Capturing A Billion Emo(j)i-ons</h3>
|
||
<p class="gh-card-excerpt is-body">This blog post was written by Dedeepya Bonthu. This is a repost from her Medium article, approved by the author.
|
||
|
||
In stadiums, sports fans love to express themselves by cheering for their favorite teams, holding up placards and team logos. Emoji’s allow fans at home to rapidly express themselves,</p>
|
||
<footer class="gh-card-meta">
|
||
<!--
|
||
-->
|
||
<span class="gh-card-author">By ByteByteGo</span>
|
||
<time class="gh-card-date" datetime="2024-03-26">26 Mar 2024</time>
|
||
<!--
|
||
--></footer>
|
||
</div>
|
||
</a>
|
||
</article>
|
||
<article class="gh-card post">
|
||
<a class="gh-card-link" href="/brief-history-of-scaling-uber/">
|
||
<figure class="gh-card-image">
|
||
<img
|
||
srcset="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w160/format/webp/2026/03/1704993859593.png 160w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w320/format/webp/2026/03/1704993859593.png 320w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w600/format/webp/2026/03/1704993859593.png 600w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w960/format/webp/2026/03/1704993859593.png 960w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w1200/format/webp/2026/03/1704993859593.png 1200w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w2000/format/webp/2026/03/1704993859593.png 2000w"
|
||
sizes="320px"
|
||
src="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w600/2026/03/1704993859593.png"
|
||
alt="Brief History of Scaling Uber"
|
||
loading="lazy"
|
||
>
|
||
</figure>
|
||
<div class="gh-card-wrapper">
|
||
<h3 class="gh-card-title is-title">Brief History of Scaling Uber</h3>
|
||
<p class="gh-card-excerpt is-body">This blog post was written by Josh Clemm, Senior Director of Engineering at Uber Eats. This is a repost from his LinkedIn article, approved by the author.
|
||
|
||
On a cold evening in Paris in 2008, Travis Kalanick and Garrett Camp couldn't get a cab. That's when</p>
|
||
<footer class="gh-card-meta">
|
||
<!--
|
||
-->
|
||
<span class="gh-card-author">By ByteByteGo</span>
|
||
<time class="gh-card-date" datetime="2024-03-14">14 Mar 2024</time>
|
||
<!--
|
||
--></footer>
|
||
</div>
|
||
</a>
|
||
</article>
|
||
<article class="gh-card post">
|
||
<a class="gh-card-link" href="/behind-aws-s3s-massive-scale/">
|
||
<figure class="gh-card-image">
|
||
<img
|
||
srcset="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w160/format/webp/2024/03/7.png 160w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w320/format/webp/2024/03/7.png 320w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w600/format/webp/2024/03/7.png 600w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w960/format/webp/2024/03/7.png 960w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w1200/format/webp/2024/03/7.png 1200w,
|
||
https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w2000/format/webp/2024/03/7.png 2000w"
|
||
sizes="320px"
|
||
src="https://storage.ghost.io/c/64/72/647204cd-7ad2-4539-b98a-3489074b932d/content/images/size/w600/2024/03/7.png"
|
||
alt="Behind AWS S3’s Massive Scale"
|
||
loading="lazy"
|
||
>
|
||
</figure>
|
||
<div class="gh-card-wrapper">
|
||
<h3 class="gh-card-title is-title">Behind AWS S3’s Massive Scale</h3>
|
||
<p class="gh-card-excerpt is-body">This is a guest article by Stanislav Kozlovski, an Apache Kafka Committer. If you would like to connect with Stanislav, you can do so on Twitter and LinkedIn.
|
||
|
||
AWS S3 is a service every engineer is familiar with.
|
||
|
||
It’s the service that popularized the notion of cold-storage to</p>
|
||
<footer class="gh-card-meta">
|
||
<!--
|
||
-->
|
||
<span class="gh-card-author">By ByteByteGo</span>
|
||
<time class="gh-card-date" datetime="2024-03-06">06 Mar 2024</time>
|
||
<!--
|
||
--></footer>
|
||
</div>
|
||
</a>
|
||
</article>
|
||
</div>
|
||
</div>
|
||
</section>
|
||
|
||
|
||
<footer class="gh-footer gh-outer">
|
||
<div class="gh-footer-inner gh-inner">
|
||
|
||
<section class="gh-footer-signup">
|
||
<h2 class="gh-footer-signup-header is-title">
|
||
High Scalability
|
||
</h2>
|
||
<p class="gh-footer-signup-subhead is-body">
|
||
Building bigger, faster, more reliable websites.
|
||
</p>
|
||
<form class="gh-form" data-members-form>
|
||
<input class="gh-form-input" id="footer-email" name="email" type="email" placeholder="jamie@example.com" required data-members-email>
|
||
<button class="gh-button" type="submit" aria-label="Subscribe">
|
||
<span><span>Subscribe</span> <svg xmlns="http://www.w3.org/2000/svg" width="32" height="32" fill="currentColor" viewBox="0 0 256 256"><path d="M224.49,136.49l-72,72a12,12,0,0,1-17-17L187,140H40a12,12,0,0,1,0-24H187L135.51,64.48a12,12,0,0,1,17-17l72,72A12,12,0,0,1,224.49,136.49Z"></path></svg></span>
|
||
<svg xmlns="http://www.w3.org/2000/svg" height="24" width="24" viewBox="0 0 24 24">
|
||
<g stroke-linecap="round" stroke-width="2" fill="currentColor" stroke="none" stroke-linejoin="round" class="nc-icon-wrapper">
|
||
<g class="nc-loop-dots-4-24-icon-o">
|
||
<circle cx="4" cy="12" r="3"></circle>
|
||
<circle cx="12" cy="12" r="3"></circle>
|
||
<circle cx="20" cy="12" r="3"></circle>
|
||
</g>
|
||
<style data-cap="butt">
|
||
.nc-loop-dots-4-24-icon-o{--animation-duration:0.8s}
|
||
.nc-loop-dots-4-24-icon-o *{opacity:.4;transform:scale(.75);animation:nc-loop-dots-4-anim var(--animation-duration) infinite}
|
||
.nc-loop-dots-4-24-icon-o :nth-child(1){transform-origin:4px 12px;animation-delay:-.3s;animation-delay:calc(var(--animation-duration)/-2.666)}
|
||
.nc-loop-dots-4-24-icon-o :nth-child(2){transform-origin:12px 12px;animation-delay:-.15s;animation-delay:calc(var(--animation-duration)/-5.333)}
|
||
.nc-loop-dots-4-24-icon-o :nth-child(3){transform-origin:20px 12px}
|
||
@keyframes nc-loop-dots-4-anim{0%,100%{opacity:.4;transform:scale(.75)}50%{opacity:1;transform:scale(1)}}
|
||
</style>
|
||
</g>
|
||
</svg> <span>Email sent</span>
|
||
</button>
|
||
<p data-members-error></p>
|
||
</form>
|
||
</section>
|
||
|
||
<div class="gh-social-links">
|
||
<a href="https://x.com/ghost" target="_blank" rel="noopener" aria-label="X">
|
||
<svg viewBox="0 0 24 24" fill="currentColor"><g><path d="M18.244 2.25h3.308l-7.227 8.26 8.502 11.24H16.17l-5.214-6.817L4.99 21.75H1.68l7.73-8.835L1.254 2.25H8.08l4.713 6.231zm-1.161 17.52h1.833L7.084 4.126H5.117z"></path></g></svg> </a>
|
||
<a href="https://www.facebook.com/ghost" target="_blank" rel="noopener" aria-label="Facebook">
|
||
<svg class="icon" viewBox="0 0 24 24" xmlns="http://www.w3.org/2000/svg" fill="currentColor"><path d="M23.9981 11.9991C23.9981 5.37216 18.626 0 11.9991 0C5.37216 0 0 5.37216 0 11.9991C0 17.9882 4.38789 22.9522 10.1242 23.8524V15.4676H7.07758V11.9991H10.1242V9.35553C10.1242 6.34826 11.9156 4.68714 14.6564 4.68714C15.9692 4.68714 17.3424 4.92149 17.3424 4.92149V7.87439H15.8294C14.3388 7.87439 13.8739 8.79933 13.8739 9.74824V11.9991H17.2018L16.6698 15.4676H13.8739V23.8524C19.6103 22.9522 23.9981 17.9882 23.9981 11.9991Z"/></svg> </a>
|
||
</div>
|
||
|
||
<div class="gh-footer-bar">
|
||
<span class="gh-footer-logo is-title">
|
||
High Scalability
|
||
</span>
|
||
<nav class="gh-footer-menu">
|
||
<ul class="nav">
|
||
<li class="nav-sign-up"><a href="#/portal/">Sign up</a></li>
|
||
</ul>
|
||
|
||
</nav>
|
||
<div class="gh-footer-copyright">
|
||
Powered by <a href="https://ghost.org/" target="_blank" rel="noopener">Ghost</a>
|
||
</div>
|
||
</div>
|
||
|
||
</div>
|
||
</footer>
|
||
|
||
</div>
|
||
|
||
<div class="pswp" tabindex="-1" role="dialog" aria-hidden="true">
|
||
<div class="pswp__bg"></div>
|
||
|
||
<div class="pswp__scroll-wrap">
|
||
<div class="pswp__container">
|
||
<div class="pswp__item"></div>
|
||
<div class="pswp__item"></div>
|
||
<div class="pswp__item"></div>
|
||
</div>
|
||
|
||
<div class="pswp__ui pswp__ui--hidden">
|
||
<div class="pswp__top-bar">
|
||
<div class="pswp__counter"></div>
|
||
|
||
<button class="pswp__button pswp__button--close" title="Close (Esc)"></button>
|
||
<button class="pswp__button pswp__button--share" title="Share"></button>
|
||
<button class="pswp__button pswp__button--fs" title="Toggle fullscreen"></button>
|
||
<button class="pswp__button pswp__button--zoom" title="Zoom in/out"></button>
|
||
|
||
<div class="pswp__preloader">
|
||
<div class="pswp__preloader__icn">
|
||
<div class="pswp__preloader__cut">
|
||
<div class="pswp__preloader__donut"></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
|
||
<div class="pswp__share-modal pswp__share-modal--hidden pswp__single-tap">
|
||
<div class="pswp__share-tooltip"></div>
|
||
</div>
|
||
|
||
<button class="pswp__button pswp__button--arrow--left" title="Previous (arrow left)"></button>
|
||
<button class="pswp__button pswp__button--arrow--right" title="Next (arrow right)"></button>
|
||
|
||
<div class="pswp__caption">
|
||
<div class="pswp__caption__center"></div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<script src="https://highscalability.com/assets/built/source.js?v=uMULaSMoSjM4gY5C"></script>
|
||
|
||
|
||
|
||
</body>
|
||
</html>
|