SRE weekly 所有文章

This commit is contained in:
2026-09-12 17:23:01 +08:00
parent 409b40ddcb
commit af7633f9dc
8486 changed files with 4489990 additions and 7 deletions

File diff suppressed because one or more lines are too long

View File

@@ -0,0 +1,322 @@
<!DOCTYPE html>
<html lang="en-US">
<head>
<meta http-equiv="X-Clacks-Overhead" content="GNU Terry Pratchett" />
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<link rel="shortcut icon" href="https://robertovitillo.com/images/favicon.png" />
<title>The costs of microservices | Roberto&#39;s blog</title>
<meta name="title" content="The costs of microservices" />
<meta name="description" content="An application typically starts its life as a monolith. Take a modern backend of a single-page Javascript application, for example - it starts out as a single stateless web service that exposes a RESTful HTTP API and uses a relational database as a backing store. The service is composed of a number of components, or libraries, that implement different business capabilities:
As the number of feature teams contributing to the same codebase increases, its components become increasingly coupled over time." />
<meta name="keywords" content="distributed systems," />
<meta property="og:title" content="The costs of microservices" />
<meta property="og:description" content="An application typically starts its life as a monolith. Take a modern backend of a single-page Javascript application, for example - it starts out as a single stateless web service that exposes a RESTful HTTP API and uses a relational database as a backing store. The service is composed of a number of components, or libraries, that implement different business capabilities:
As the number of feature teams contributing to the same codebase increases, its components become increasingly coupled over time." />
<meta property="og:type" content="article" />
<meta property="og:url" content="https://robertovitillo.com/costs-of-microservices/" /><meta property="article:section" content="blog" />
<meta property="article:published_time" content="2020-11-22T00:00:00+00:00" />
<meta property="article:modified_time" content="2020-11-22T00:00:00+00:00" /><meta property="og:site_name" content="Roberto&#39;s blog" />
<meta name="twitter:card" content="summary"/><meta name="twitter:title" content="The costs of microservices"/>
<meta name="twitter:description" content="An application typically starts its life as a monolith. Take a modern backend of a single-page Javascript application, for example - it starts out as a single stateless web service that exposes a RESTful HTTP API and uses a relational database as a backing store. The service is composed of a number of components, or libraries, that implement different business capabilities:
As the number of feature teams contributing to the same codebase increases, its components become increasingly coupled over time."/>
<meta itemprop="name" content="The costs of microservices">
<meta itemprop="description" content="An application typically starts its life as a monolith. Take a modern backend of a single-page Javascript application, for example - it starts out as a single stateless web service that exposes a RESTful HTTP API and uses a relational database as a backing store. The service is composed of a number of components, or libraries, that implement different business capabilities:
As the number of feature teams contributing to the same codebase increases, its components become increasingly coupled over time."><meta itemprop="datePublished" content="2020-11-22T00:00:00+00:00" />
<meta itemprop="dateModified" content="2020-11-22T00:00:00+00:00" />
<meta itemprop="wordCount" content="1326">
<meta itemprop="keywords" content="distributed systems," />
<meta name="referrer" content="no-referrer-when-downgrade" />
<style>
body {
font-family: Verdana, sans-serif;
margin: auto;
padding: 20px;
max-width: 720px;
text-align: left;
background-color: #fff;
word-wrap: break-word;
overflow-wrap: break-word;
line-height: 1.5;
color: #444;
}
h1,
h2,
h3,
h4,
h5,
h6,
strong,
b {
color: #222;
}
a {
color: #3273dc;
}
.title {
text-decoration: none;
border: 0;
}
.title span {
font-weight: 400;
}
nav a {
margin-right: 10px;
}
textarea {
width: 100%;
font-size: 16px;
}
input {
font-size: 16px;
}
content {
line-height: 1.6;
}
table {
width: 100%;
}
img {
max-width: 100%;
}
code {
padding: 2px 5px;
background-color: #f2f2f2;
}
pre code {
color: #222;
display: block;
padding: 20px;
white-space: pre-wrap;
font-size: 14px;
overflow-x: auto;
}
div.highlight pre {
background-color: initial;
color: initial;
}
div.highlight code {
background-color: unset;
color: unset;
}
blockquote {
border-left: 1px solid #999;
color: #222;
padding-left: 20px;
font-style: italic;
}
footer {
padding: 25px;
text-align: center;
}
.helptext {
color: #777;
font-size: small;
}
.errorlist {
color: #eba613;
font-size: small;
}
ul.blog-posts {
list-style-type: none;
padding: unset;
}
ul.blog-posts li {
display: flex;
}
ul.blog-posts li span {
flex: 0 0 130px;
}
ul.blog-posts li a:visited {
color: #8b6fcb;
}
@media (prefers-color-scheme: dark) {
body {
background-color: #333;
color: #ddd;
}
h1,
h2,
h3,
h4,
h5,
h6,
strong,
b {
color: #eee;
}
a {
color: #8cc2dd;
}
code {
background-color: #777;
}
pre code {
color: #ddd;
}
blockquote {
color: #ccc;
}
textarea,
input {
background-color: #252525;
color: #ddd;
}
.helptext {
color: #aaa;
}
}
</style>
<script id="MathJax-script" async src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-chtml.js"></script>
<script>
MathJax = {
tex: {
displayMath: [['\\[', '\\]'], ['$$', '$$']],
inlineMath: [['\\(', '\\)'], ['$', '$']]
}
};
</script>
<script async src="https://www.googletagmanager.com/gtag/js?id=G-93GY9CXFV2"></script>
<script>
var doNotTrack = false;
if (!doNotTrack) {
window.dataLayer = window.dataLayer || [];
function gtag(){dataLayer.push(arguments);}
gtag('js', new Date());
gtag('config', 'G-93GY9CXFV2', { 'anonymize_ip': false });
}
</script>
</head>
<body>
<header><a href="/" class="title">
<h2>Roberto&#39;s blog</h2>
</a>
<nav><a href="/">Home</a>
<a href="/blog">Blog</a>
</nav>
</header>
<main>
<h1>The costs of microservices</h1>
<p>
<i>
<time datetime='2020-11-22' pubdate>
22 Nov, 2020
</time>
</i>
</p>
<content>
<p>An application typically starts its life as a monolith. Take a modern backend of a single-page Javascript application, for example - it starts out as a single stateless web service that exposes a RESTful HTTP API and uses a relational database as a backing store. The service is composed of a number of components, or libraries, that implement different business capabilities:</p>
<p><figure>
<img src="monolith.png" alt="Monolith">
<figcaption style="text-align: center;"></figcaption>
</figure></p>
<p>As the number of feature teams contributing to the same codebase increases, its components become increasingly coupled over time. This leads the teams to step on each other&rsquo;s toes more and more frequently, decreasing their productivity.</p>
<p>The codebase becomes complex enough that nobody fully understands every part of it, and implementing new features or fixing bugs becomes time-consuming. Even if the backend is componentized into different libraries owned by different teams, a change to a library requires the service to be redeployed. And if a change introduces a bug - like a memory leak - the entire service can potentially be affected by it. Additionally, rolling back a faulty build affects the velocity of all teams, not just the one that introduced the bug.</p>
<p>One way to mitigate the growing pains of a <em>monolithic</em> backend is to split it into a set of independently deployable services that communicate via APIs. The APIs decouple the services from each other by creating boundaries that are hard to violate, unlike the ones between components running in the same process:</p>
<p><figure>
<img src="api_gw.png" alt="Microservices">
<figcaption style="text-align: center;"></figcaption>
</figure></p>
<p>This architectural style is also referred to as the microservice architecture. The term <em>micro</em> can be misleading, though - there doesn&rsquo;t have to be anything micro about the services. In fact, I would argue that if a service doesn&rsquo;t do much, it just creates more operational toll than benefits. A more appropriate name for this architecture is <a href="https://en.wikipedia.org/wiki/Service-oriented_architecture">service-oriented architecture</a>, but unfortunately, that name comes with some old baggage as well. Perhaps in 10 years, we will call the same concept with yet another name, but for now we will have to stick to microservices.</p>
<p>Breaking down the backend by business capabilities into a set of services with well-defined boundaries allows each service to be developed and operated by a single small team. The reduced team size increases the application&rsquo;s development speed for a variety of reasons:</p>
<ul>
<li>Smaller teams are more effective as the communication overhead grows <a href="https://en.wikipedia.org/wiki/The_Mythical_Man-Month">quadratically</a> with the team&rsquo;s size.</li>
<li>As each team dictates its own release schedule and has complete control over its codebase, less cross-team communication is required, and therefore decisions can be taken in less time.</li>
<li>The codebase of a service is smaller and easier to digest by its developers, reducing the time it takes to ramp up new hires. Also, a smaller codebase doesn&rsquo;t slow down IDEs, which makes the developers more productive.</li>
<li>The boundaries between services are much stronger than the boundaries between components in the same process. Because of that, when a developer needs to change a part of the backend, they only need to understand a small part of the whole.</li>
<li>Each service can be scaled independently and adopt a different technology stack based on its own needs. The consumers of the APIs don&rsquo;t care how the functionality is implemented after all. This makes it easy to experiment and evaluate new technologies without affecting other parts of the system.</li>
<li>Each microservice can have its own independent data model and data store(s) that best fit its use-cases, allowing developers to change its schema without affecting other services.</li>
</ul>
<h2 id="costs">Costs</h2>
<p>The microservices architecture adds more moving parts to the overall system, and this doesn&rsquo;t come for free. The cost of fully embracing microservices is only worth paying if it can be amortized across dozens of development teams.</p>
<p><strong>Development Experience</strong></p>
<p>Nothing forbids the use of different languages, libraries, and datastores for each microservice - but doing so transforms the application into an unmaintainable mess. For example, it makes it more challenging for a developer to move from one team to another if the software stack is completely different. And think of the sheer number of libraries - one for each language adopted - that need to be supported to provide common functionality that all services need, like logging.</p>
<p>It&rsquo;s only reasonable then that a certain degree of standardization is needed. One way to do that - while still allowing some degree of freedom - is to loosely encourage specific technologies by providing a great development experience for the teams that stick with the recommended portfolio of languages and technologies.</p>
<p><strong>Resource Provisioning</strong></p>
<p>To support a large number of independent services, it should be simple to spin up new servers, data stores, and other commodity resources - you don&rsquo;t want every team to come up with their own way of doing it. And once these resources have been provisioned, they have to be configured. To be able to pull this off, you will need a fair amount of automation.</p>
<p><strong>Communication</strong></p>
<p>Remote calls are expensive and introduce <a href="/default-timeouts/">new and fun ways</a> your systems can crumble. You will need defense mechanisms to protect against failures, like timeouts, retries and circuit breakers. You will also have to leverage asynchrony and batching to mitigate the performance hit of communicating across the network. All of which increases the system&rsquo;s complexity. A lot of what I describe in my <a href="https://understandingdistributed.systems/">book about distributed systems</a> is about dealing with this complexity.</p>
<p>That being said, even a monolith doesn&rsquo;t live in isolation as it&rsquo;s being accessed by remote clients, and it&rsquo;s likely to use third-party APIs as well. So eventually, these issues need to be solved there as well, albeit on a smaller scale.</p>
<p><strong>Continuous Integration, Delivery, and Deployment</strong></p>
<p>Continuous integration ensures that code changes are merged into the main branch after an automated build and test processes have run. Once a code change has been merged, it should be automatically published and deployed to a production-like environment, where a battery of integration and end-to-end tests run to ensure that the microservice doesn&rsquo;t break any service that depends on it.</p>
<p>While testing individual microservices is not more challenging than testing a monolith, testing the integration of all the microservices is an order of magnitude harder. Very subtle and unexpected behavior can emerge when individual services interact with each other.</p>
<p><strong>Operations</strong></p>
<p>Unlike with a monolith, it&rsquo;s much more expensive to staff each team responsible for a service with its own operations team. As a result, the team that develops a service is typically also on-call for it. This creates friction between development work and operational toll as the team needs to decide what to prioritize during each sprint.</p>
<p>Debugging systems failures becomes more challenging as well - you can&rsquo;t just load the whole application on your local machine and step through it with a debugger. The system has more ways to fail, as there are more moving parts. This is why good logging and monitoring becomes crucial at all levels.</p>
<p><strong>Eventual Consistency</strong></p>
<p>A side effect of splitting an application into separate services is that the data model no longer resides in a single data store. Atomically updating records stored in different data stores, and guaranteeing strong consistency, is slow, expensive, and hard to get right. Hence, this type of architecture usually requires embracing eventual consistency.</p>
<h2 id="practical-considerations">Practical Considerations</h2>
<p>Splitting an application into services adds a lot of complexity to the overall system. Because of that, it&rsquo;s generally best to start with a monolith and split it up only when there is a good reason to do so.</p>
<p>Getting the boundaries right between the services is challenging - it&rsquo;s much easier to move them around within a monolith until you find a sweet spot. Once the monolith is well matured and growing pains start to rise, then you can start to peel off one microservice at a time from it.</p>
<p>You should only start with a microservice first approach if you already have experience with it, and you either have built out a platform for it or have accounted for the time it will take you to build one.</p>
</content>
<p>
<a href="https://robertovitillo.com/blog/distributed-systems/">#Distributed Systems</a>
</p>
</main>
<footer>
<script async data-uid="9d8b0b1bf3" src="https://vitillo.ck.page/9d8b0b1bf3/index.js"></script>
</footer>
</body>
</html>

View File

@@ -0,0 +1,322 @@
<!DOCTYPE html>
<html lang="en-US">
<head>
<meta http-equiv="X-Clacks-Overhead" content="GNU Terry Pratchett" />
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<link rel="shortcut icon" href="https://robertovitillo.com/images/favicon.png" />
<title>How distributed systems fail | Roberto&#39;s blog</title>
<meta name="title" content="How distributed systems fail" />
<meta name="description" content="At scale, any failure that can happen will eventually happen. Hardware failures, software crashes, memory leaks - you name it. The more components you have, the more failures you will experience.
This nasty behavior is caused by cruel math - given an operation that has a certain probability of failing, as the total number of operations performed increases, so does the total number of failures. In other words, as you scale out your application to handle more load, the more failures it will experience." />
<meta name="keywords" content="distributed systems," />
<meta property="og:title" content="How distributed systems fail" />
<meta property="og:description" content="At scale, any failure that can happen will eventually happen. Hardware failures, software crashes, memory leaks - you name it. The more components you have, the more failures you will experience.
This nasty behavior is caused by cruel math - given an operation that has a certain probability of failing, as the total number of operations performed increases, so does the total number of failures. In other words, as you scale out your application to handle more load, the more failures it will experience." />
<meta property="og:type" content="article" />
<meta property="og:url" content="https://robertovitillo.com/how-distributed-systems-fail/" /><meta property="article:section" content="blog" />
<meta property="article:published_time" content="2020-12-05T00:00:00+00:00" />
<meta property="article:modified_time" content="2020-12-05T00:00:00+00:00" /><meta property="og:site_name" content="Roberto&#39;s blog" />
<meta name="twitter:card" content="summary"/><meta name="twitter:title" content="How distributed systems fail"/>
<meta name="twitter:description" content="At scale, any failure that can happen will eventually happen. Hardware failures, software crashes, memory leaks - you name it. The more components you have, the more failures you will experience.
This nasty behavior is caused by cruel math - given an operation that has a certain probability of failing, as the total number of operations performed increases, so does the total number of failures. In other words, as you scale out your application to handle more load, the more failures it will experience."/>
<meta itemprop="name" content="How distributed systems fail">
<meta itemprop="description" content="At scale, any failure that can happen will eventually happen. Hardware failures, software crashes, memory leaks - you name it. The more components you have, the more failures you will experience.
This nasty behavior is caused by cruel math - given an operation that has a certain probability of failing, as the total number of operations performed increases, so does the total number of failures. In other words, as you scale out your application to handle more load, the more failures it will experience."><meta itemprop="datePublished" content="2020-12-05T00:00:00+00:00" />
<meta itemprop="dateModified" content="2020-12-05T00:00:00+00:00" />
<meta itemprop="wordCount" content="1433">
<meta itemprop="keywords" content="distributed systems," />
<meta name="referrer" content="no-referrer-when-downgrade" />
<style>
body {
font-family: Verdana, sans-serif;
margin: auto;
padding: 20px;
max-width: 720px;
text-align: left;
background-color: #fff;
word-wrap: break-word;
overflow-wrap: break-word;
line-height: 1.5;
color: #444;
}
h1,
h2,
h3,
h4,
h5,
h6,
strong,
b {
color: #222;
}
a {
color: #3273dc;
}
.title {
text-decoration: none;
border: 0;
}
.title span {
font-weight: 400;
}
nav a {
margin-right: 10px;
}
textarea {
width: 100%;
font-size: 16px;
}
input {
font-size: 16px;
}
content {
line-height: 1.6;
}
table {
width: 100%;
}
img {
max-width: 100%;
}
code {
padding: 2px 5px;
background-color: #f2f2f2;
}
pre code {
color: #222;
display: block;
padding: 20px;
white-space: pre-wrap;
font-size: 14px;
overflow-x: auto;
}
div.highlight pre {
background-color: initial;
color: initial;
}
div.highlight code {
background-color: unset;
color: unset;
}
blockquote {
border-left: 1px solid #999;
color: #222;
padding-left: 20px;
font-style: italic;
}
footer {
padding: 25px;
text-align: center;
}
.helptext {
color: #777;
font-size: small;
}
.errorlist {
color: #eba613;
font-size: small;
}
ul.blog-posts {
list-style-type: none;
padding: unset;
}
ul.blog-posts li {
display: flex;
}
ul.blog-posts li span {
flex: 0 0 130px;
}
ul.blog-posts li a:visited {
color: #8b6fcb;
}
@media (prefers-color-scheme: dark) {
body {
background-color: #333;
color: #ddd;
}
h1,
h2,
h3,
h4,
h5,
h6,
strong,
b {
color: #eee;
}
a {
color: #8cc2dd;
}
code {
background-color: #777;
}
pre code {
color: #ddd;
}
blockquote {
color: #ccc;
}
textarea,
input {
background-color: #252525;
color: #ddd;
}
.helptext {
color: #aaa;
}
}
</style>
<script id="MathJax-script" async src="https://cdn.jsdelivr.net/npm/mathjax@3/es5/tex-chtml.js"></script>
<script>
MathJax = {
tex: {
displayMath: [['\\[', '\\]'], ['$$', '$$']],
inlineMath: [['\\(', '\\)'], ['$', '$']]
}
};
</script>
<script async src="https://www.googletagmanager.com/gtag/js?id=G-93GY9CXFV2"></script>
<script>
var doNotTrack = false;
if (!doNotTrack) {
window.dataLayer = window.dataLayer || [];
function gtag(){dataLayer.push(arguments);}
gtag('js', new Date());
gtag('config', 'G-93GY9CXFV2', { 'anonymize_ip': false });
}
</script>
</head>
<body>
<header><a href="/" class="title">
<h2>Roberto&#39;s blog</h2>
</a>
<nav><a href="/">Home</a>
<a href="/blog">Blog</a>
</nav>
</header>
<main>
<h1>How distributed systems fail</h1>
<p>
<i>
<time datetime='2020-12-05' pubdate>
05 Dec, 2020
</time>
</i>
</p>
<content>
<p>At scale, any failure that can happen will eventually happen. Hardware failures, software crashes, memory leaks - you name it. The more components you have, the more failures you will experience.</p>
<p>This nasty behavior is caused by <em>cruel math</em> - given an operation that has a certain probability of failing, as the total number of operations performed increases, so does the total number of failures. In other words, as you scale out your application to handle more load, the more failures it will experience.</p>
<p>To protect your application against failures, you first need to know what can go wrong. Assuming you are using a cloud provider and not maintaning your own datacenter, the most common failures you will encounter are caused by single points of failure, the network being unreliable, slow processes, and unexpected load.</p>
<h2 id="single-point-of-failure">Single Point of Failure</h2>
<p>A single point of failure is the most glaring cause of failure in a distributed system - it&rsquo;s that one component that when it fails brings down the entire system with it. In practice, distributed systems can have multiple single points of failure.</p>
<p>A service that to start up needs to read its configuration from a non-replicated database is an example of a single point of failure - if the database isn&rsquo;t reachable, the service won&rsquo;t be able to start.</p>
<p>A more subtle example is a service that exposes a HTTP API on top of TLS and uses a certificate that needs to be manually renewed. If the certificate isn&rsquo;t renewed by the time it expires, then most clients trying to connect to it wouldn&rsquo;t be able to open a connection with the service.</p>
<p>Single points of failure should be identified when the system is architected before they can cause any harm. The best way to detect them is to examine every component of the system and ask what would happen if that component were to fail. Some single points of failure can be architected away, e.g., by introducing redundancy, while others can&rsquo;t. In that case, the only option left is to minimize the blast radius.</p>
<h2 id="unreliable-network">Unreliable Network</h2>
<p>When a client make a remote network call, it sends a request to a server and expects to receive a response from it a while later. In the best case, the client receives a response shortly after sending the request. But what if the client waits and waits and still doesn&rsquo;t get a response?</p>
<p>In that case, the client doesn&rsquo;t know whether a response will eventually arrive or not. At that point it has only two options, it can either continue to wait, or fail the request with an exception or an error.</p>
<p>Slow network calls are the <a href="/default-timeouts/">silent killers</a> of distributed systems. Because the client doesn&rsquo;t know whether the response is on its way or not, it can spend a long time waiting before giving up, if it gives up at all. The wait can in turn cause degradations that are extremely hard to debug.</p>
<h2 id="slow-processes">Slow Processes</h2>
<p>From an observer&rsquo;s point of view, a very slow process is not very different from one that isn&rsquo;t running at all - neither can perform useful work. Resource leaks are one of the most common causes of slow processes.</p>
<p>Memory leaks are arguably the most well-known source of leaks. A memory leak manifests itself with a steady increase in memory consumption over time. Run-times with garbage collection don&rsquo;t help much either - if a reference to an object that isn&rsquo;t longer needed is kept somewhere, the object won&rsquo;t be deleted by the garbage collector.</p>
<p>A memory leak keeps consuming memory until there is no more of it, at which point the operating system starts swapping memory pages to the disk constantly, all the while the garbage collector kicks in more frequently trying its best to release any shred of memory. The constant paging and the garbage collector eating up CPU cycles make the process slower. Eventually, when there is no more physical memory, and there is no more space in the swap file, the process won&rsquo;t be able to allocate more memory, and most operations will fail.</p>
<p>Memory is just one of the many resources that can leak. For example, if you are using a thread pool, you can lose a thread when it blocks on a synchronous call that never returns. If a thread makes a synchronous, and blocking, HTTP call <a href="/default-timeouts/">without setting a timeout</a>, and the call never returns, the thread won&rsquo;t be returned to the pool. Since the pool has a fixed size and keeps losing threads, the pool will eventually run out of threads.</p>
<p>You might think that making <em>asynchronous</em> calls, rather than a synchronous ones, would mitigate the problem in the previous case. But, modern HTTP clients use socket pools to avoid recreating TCP connections and pay a <a href="/what-every-developer-should-know-about-tcp/">hefty performance fee</a>. If a request is made without a timeout, the connection is never returned to the pool. As the pool has a limited size, eventually there won&rsquo;t be any connections left to communicate with the host.</p>
<p>On top of all that, the code you write isn&rsquo;t the only one accessing memory, threads and sockets. The libraries your application depends on access the same resources, and they can do all kinds of shady things. Without digging into their implementation, assuming it&rsquo;s open in the first place, you can&rsquo;t be sure whether they can wreak havoc or not.</p>
<h2 id="unexpected-load">Unexpected Load</h2>
<p>Every system has a limit to how much load it can withstand without scaling. Depending on how the load increases, you are bound to hit that brick wall sooner or later. But one thing is an organic increase in load, which gives you the time to scale your service out accordingly, and another is a sudden and unexpected spike.</p>
<p>For example, consider the number of requests received by a service in a period of time. The rate and the type of incoming requests can change over time, and sometimes suddenly, for a variety of reasons:</p>
<ul>
<li>The requests might have a seasonality - depending on the hour of the day the service is going to get hit by users in different countries.</li>
<li>Some requests are much more expensive than others and abuse the system in ways you didn&rsquo;t really anticipate for, like scrapers slurping in data from your site at super human speed.</li>
<li>Some requests are malicious - think of DDoS attacks which try to saturate your service&rsquo;s bandwidth, denying access to the service to legitimate users.</li>
</ul>
<h2 id="cascading-failures">Cascading Failures</h2>
<p>You would think that if your system has hundreds of processes, it shouldn&rsquo;t make much of a difference if a small percentage are slow or unreachable. The thing about faults is that they tend to spread like cancer, propagating from one process to the other until the whole system crumbles to its knees. This effect is also referred to as a <em>cascading failure</em>, which occurs when a portion of an overall system fails, increasing the probability that other portions fail.</p>
<p>For example, suppose there are multiple clients querying two database replicas A and B, which are behind a load balancer. Each replica is handling about 50 transactions per second.</p>
<p><figure>
<img src="cascading_failure_1.png" alt="">
<figcaption style="text-align: center;"></figcaption>
</figure></p>
<p>Suddenly, replica B becomes unavailable because of a network fault. The load balancer detects that B is unavailable and removes it from its pool. Because of that, replica A has to pick up the slack for replica B, doubling the load it was previously under.</p>
<p><figure>
<img src="cascading_failure_2.png" alt="">
<figcaption style="text-align: center;"></figcaption>
</figure></p>
<p>As replica A starts to struggle to keep up with the incoming requests, the clients experience more failures and timeouts. In turn, they retry the same failing requests several times, adding insult to injury.</p>
<p>Eventually, replica A is under so much load that it can no longer serve requests promptly, and becomes for all intent and purposes unavailable, causing replica A to be removed from the load balancer&rsquo;s pool. In the meantime, replica B becomes available again and the load balancer puts it back in the pool, at which point it&rsquo;s flooded with requests that kill the replica instantaneously. This feedback loop of doom can repeat several time.</p>
<p>Cascading failures are very hard to get under control once they have started. The best way to mitigate one is to not have it in the first place by stopping the cracks in your services to propagate to others.</p>
<h2 id="defense-mechanisms">Defense Mechanisms</h2>
<p>There is a variety of best practices you can use to mitigate failures, like circuit breakers, load shedding, rate-limiting and bulkheads. I plan to blog about those in the future, but in the meantime Google is your friend. Also, I have an entire chapter dedicated to resiliency patterns in my <a href="https://understandingdistributed.systems/">book about distributed systems</a>.</p>
</content>
<p>
<a href="https://robertovitillo.com/blog/distributed-systems/">#Distributed Systems</a>
</p>
</main>
<footer>
<script async data-uid="9d8b0b1bf3" src="https://vitillo.ck.page/9d8b0b1bf3/index.js"></script>
</footer>
</body>
</html>

File diff suppressed because one or more lines are too long

View File

@@ -0,0 +1,93 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<link href="https://www.redditstatic.com/shreddit/assets/favicon/64x64.png" rel="icon shortcut" sizes="64x64" />
<title>Reddit</title>
<script nonce="cd4ea8c0-8fff-41a9-9ec7-56aea4eaddb3">
document.addEventListener("DOMContentLoaded",async function(){var e=document.forms[0],n=(e.onsubmit=function(t){return new URLSearchParams(document.location.search).forEach((e,n)=>t.target.appendChild(Object.assign(document.createElement("input"),{name:n,type:"hidden",value:e}))),!0},await(async e=>e+e)("8e58de0538fdb870"));e.elements.namedItem("solution").value=n,e.requestSubmit()},{once:!0});
</script>
<style>
main{align-items:center;display:flex;height:100vh;isolation:isolate;justify-content:center;position:relative;width:100vw}main:before{animation:scaleout 1.5s infinite ease-in-out;background-color:#d93900;border-radius:100%;content:'';height:8rem;opacity:.75;position:absolute;width:8rem}.logo{align-items:center;display:flex;fill:currentColor;font-size:4rem;justify-content:center;z-index:1}.logo svg{fill:currentColor;height:8rem;width:auto}@keyframes scaleout{0%{transform:scale(1)}100%{transform:scale(1.5);opacity:0}}.snoo-cls-1{fill:url(#snoo-radial-gragient) white}.snoo-cls-1,.snoo-cls-2,.snoo-cls-3,.snoo-cls-4,.snoo-cls-5,.snoo-cls-6,.snoo-cls-7,.snoo-cls-8,.snoo-cls-9,.snoo-cls-10,.snoo-cls-11{stroke-width:0}.snoo-cls-2{fill:url(#snoo-radial-gragient-2) white}.snoo-cls-3{fill:url(#snoo-radial-gragient-3) white}.snoo-cls-4{fill:url(#snoo-radial-gragient-4) #fc4301}.snoo-cls-5{fill:url(#snoo-radial-gragient-6) black}.snoo-cls-6{fill:url(#snoo-radial-gragient-8) black}.snoo-cls-7{fill:url(#snoo-radial-gragient-5) #fc4301}.snoo-cls-8{fill:url(#snoo-radial-gragient-7) white}.snoo-cls-9{fill:#842123}.snoo-cls-10{fill:#ff4500}.snoo-cls-11{fill:#ffc49c}
</style>
</head>
<body>
<main>
<div class='logo'>
<svg xmlns="http://www.w3.org/2000/svg"
viewBox="0 0 216 216"
xmlns:xlink="http://www.w3.org/1999/xlink"
xml:space="preserve">
<defs>
<linearGradient id="orangeredGradient" gradientTransform="rotate(90)">
<stop offset="0%" stop-color="#FE7B0E"></stop>
<stop offset="100%" stop-color="#EF0A22"></stop>
</linearGradient>
</defs>
<defs>
<radialGradient id="snoo-radial-gragient" cx="169.75" cy="92.19" fx="169.75" fy="92.19" r="50.98" gradientTransform="translate(0 11.64) scale(1 .87)" gradientUnits="userSpaceOnUse">
<stop offset="0" stop-color="#feffff"></stop>
<stop offset=".4" stop-color="#feffff"></stop>
<stop offset=".51" stop-color="#f9fcfc"></stop>
<stop offset=".62" stop-color="#edf3f5"></stop>
<stop offset=".7" stop-color="#dee9ec"></stop>
<stop offset=".72" stop-color="#d8e4e8"></stop>
<stop offset=".76" stop-color="#ccd8df"></stop>
<stop offset=".8" stop-color="#c8d5dd"></stop>
<stop offset=".83" stop-color="#ccd6de"></stop>
<stop offset=".85" stop-color="#d8dbe2"></stop>
<stop offset=".88" stop-color="#ede3e9"></stop>
<stop offset=".9" stop-color="#ffebef"></stop>
</radialGradient>
<radialGradient id="snoo-radial-gragient-2" cx="47.31" fx="47.31" r="50.98" xlink:href="#snoo-radial-gragient"></radialGradient>
<radialGradient id="snoo-radial-gragient-3" cx="109.61" cy="85.59" fx="109.61" fy="85.59" r="153.78" gradientTransform="translate(0 25.56) scale(1 .7)" xlink:href="#snoo-radial-gragient"></radialGradient>
<radialGradient id="snoo-radial-gragient-4" cx="-6.01" cy="64.68" fx="-6.01" fy="64.68" r="12.85" gradientTransform="translate(81.08 27.26) scale(1.07 1.55)" gradientUnits="userSpaceOnUse">
<stop offset="0" stop-color="#f60"></stop>
<stop offset=".5" stop-color="#ff4500"></stop>
<stop offset=".7" stop-color="#fc4301"></stop>
<stop offset=".82" stop-color="#f43f07"></stop>
<stop offset=".92" stop-color="#e53812"></stop>
<stop offset="1" stop-color="#d4301f"></stop>
</radialGradient>
<radialGradient id="snoo-radial-gragient-5" cx="-73.55" cy="64.68" fx="-73.55" fy="64.68" r="12.85" gradientTransform="translate(62.87 27.26) rotate(-180) scale(1.07 -1.55)" xlink:href="#snoo-radial-gragient-4"></radialGradient>
<radialGradient id="snoo-radial-gragient-6" cx="107.93" cy="166.96" fx="107.93" fy="166.96" r="45.3" gradientTransform="translate(0 57.4) scale(1 .66)" gradientUnits="userSpaceOnUse">
<stop offset="0" stop-color="#172e35"></stop>
<stop offset=".29" stop-color="#0e1c21"></stop>
<stop offset=".73" stop-color="#030708"></stop>
<stop offset="1" stop-color="#000"></stop>
</radialGradient>
<radialGradient id="snoo-radial-gragient-7" cx="147.88" cy="32.94" fx="147.88" fy="32.94" r="39.77" gradientTransform="translate(0 .54) scale(1 .98)" xlink:href="#snoo-radial-gragient"></radialGradient>
<radialGradient id="snoo-radial-gragient-8" cx="131.31" cy="73.08" fx="131.31" fy="73.08" r="32.6" gradientUnits="userSpaceOnUse">
<stop offset=".48" stop-color="#7a9299"></stop>
<stop offset=".67" stop-color="#172e35"></stop>
<stop offset=".75" stop-color="#000"></stop>
<stop offset=".82" stop-color="#172e35"></stop>
</radialGradient>
</defs>
<path class="snoo-cls-10" d="m108,0h0C48.35,0,0,48.35,0,108h0c0,29.82,12.09,56.82,31.63,76.37l-20.57,20.57c-4.08,4.08-1.19,11.06,4.58,11.06h92.36s0,0,0,0c59.65,0,108-48.35,108-108h0C216,48.35,167.65,0,108,0Z"></path>
<circle class="snoo-cls-1" cx="169.22" cy="106.98" r="25.22"></circle>
<circle class="snoo-cls-2" cx="46.78" cy="106.98" r="25.22"></circle>
<ellipse class="snoo-cls-3" cx="108.06" cy="128.64" rx="72" ry="54"></ellipse>
<path class="snoo-cls-4" d="m86.78,123.48c-.42,9.08-6.49,12.38-13.56,12.38s-12.46-4.93-12.04-14.01c.42-9.08,6.49-15.02,13.56-15.02s12.46,7.58,12.04,16.66Z"></path>
<path class="snoo-cls-7" d="m129.35,123.48c.42,9.08,6.49,12.38,13.56,12.38s12.46-4.93,12.04-14.01c-.42-9.08-6.49-15.02-13.56-15.02s-12.46,7.58-12.04,16.66Z"></path>
<ellipse class="snoo-cls-11" cx="79.63" cy="116.37" rx="2.8" ry="3.05"></ellipse>
<ellipse class="snoo-cls-11" cx="146.21" cy="116.37" rx="2.8" ry="3.05"></ellipse>
<path class="snoo-cls-5" d="m108.06,142.92c-8.76,0-17.16.43-24.92,1.22-1.33.13-2.17,1.51-1.65,2.74,4.35,10.39,14.61,17.69,26.57,17.69s22.23-7.3,26.57-17.69c.52-1.23-.33-2.61-1.65-2.74-7.77-.79-16.16-1.22-24.92-1.22Z"></path>
<circle class="snoo-cls-8" cx="147.49" cy="49.43" r="17.87"></circle>
<path class="snoo-cls-6" d="m107.8,76.92c-2.14,0-3.87-.89-3.87-2.27,0-16.01,13.03-29.04,29.04-29.04,2.14,0,3.87,1.73,3.87,3.87s-1.73,3.87-3.87,3.87c-11.74,0-21.29,9.55-21.29,21.29,0,1.38-1.73,2.27-3.87,2.27Z"></path>
<path class="snoo-cls-9" d="m62.82,122.65c.39-8.56,6.08-14.16,12.69-14.16,6.26,0,11.1,6.39,11.28,14.33.17-8.88-5.13-15.99-12.05-15.99s-13.14,6.05-13.56,15.2c-.42,9.15,4.97,13.83,12.04,13.83.17,0,.35,0,.52,0-6.44-.16-11.3-4.79-10.91-13.2Z"></path>
<path class="snoo-cls-9" d="m153.3,122.65c-.39-8.56-6.08-14.16-12.69-14.16-6.26,0-11.1,6.39-11.28,14.33-.17-8.88,5.13-15.99,12.05-15.99,7.07,0,13.14,6.05,13.56,15.2.42,9.15-4.97,13.83-12.04,13.83-.17,0-.35,0-.52,0,6.44-.16,11.3-4.79,10.91-13.2Z"></path>
</svg>
</div>
</main>
<form hidden method="GET" action="/r/sre/comments/177ob10/as_an_sre_how_often_are_you_directly_involved/">
<input type="hidden" name="solution" />
<input type="hidden" name="js_challenge" value="1"/>
<input type="hidden" name="jsc_token" value="7afd7253fec22262ff1c52b1703fe9ecb38e781e694617d7f0854dc4a82e11e9"/>
<input type="hidden" name="jsc_orig_r" value=""/>
</form>
</body>
</html>

View File

@@ -0,0 +1,12 @@
<!doctype html>
<meta charset="utf-8">
<title>Redirect</title>
<script>
const target = "https://blog.rust-lang.org/inside-rust/2023/07/21/crates-io-postmortem/";
const hash = window.location.hash || "";
window.location.replace(target + hash);
</script>
<noscript>
<meta http-equiv="refresh" content="0; url=https://blog.rust-lang.org/inside-rust/2023/07/21/crates-io-postmortem/">
</noscript>
<p><a href="https://blog.rust-lang.org/inside-rust/2023/07/21/crates-io-postmortem/">Click here</a> to be redirected.</p>

View File

@@ -0,0 +1,44 @@
[
{
"idx": 1,
"url": "https://thenewstack.io/translating-failures-into-service-level-objectives/",
"ok": true,
"error": null
},
{
"idx": 2,
"url": "https://robertovitillo.com/costs-of-microservices/",
"ok": true,
"error": null
},
{
"idx": 3,
"url": "https://robertovitillo.com/how-distributed-systems-fail/",
"ok": true,
"error": null
},
{
"idx": 4,
"url": "https://medium.com/@letathenasleep/alerting-the-dos-and-don-ts-for-effective-observability-139db9fb49d1",
"ok": false,
"error": "HTTP 403"
},
{
"idx": 5,
"url": "https://www.codereliant.io/retries-backoff-jitter/",
"ok": true,
"error": null
},
{
"idx": 6,
"url": "https://www.reddit.com/r/sre/comments/177ob10/as_an_sre_how_often_are_you_directly_involved/",
"ok": false,
"error": "trafilatura returned empty"
},
{
"idx": 7,
"url": "https://blog.rust-lang.org/inside-rust/2023/07/21/crates-io-postmortem.html",
"ok": true,
"error": null
}
]