658 lines
24 KiB
HTML
658 lines
24 KiB
HTML
<!DOCTYPE html>
|
||
<html class="no-js" lang="en">
|
||
<head>
|
||
<meta charset="utf-8">
|
||
<title>Adding latency: one step, two step, oops | Lawrence Jones</title>
|
||
<meta name="description"
|
||
content=" When it comes to complex systems, you can only go so far with synthetic experiments before you need to try something for real, and test in production. T...">
|
||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||
|
||
|
||
<!-- If this is an external_url, we want to redirect -->
|
||
|
||
|
||
<!-- Preload Google fonts, which is defined in sass -->
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://fonts.gstatic.com/s/sourcesanspro/v14/6xK1dSBYKcSV-LCoeQqfX1RYOo3qPZ7nsDc.ttf">
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://fonts.gstatic.com/s/sourcesanspro/v14/6xKwdSBYKcSV-LCoeQqfX1RYOo3qPZZclSds18E.ttf">
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://fonts.gstatic.com/s/sourcesanspro/v14/6xK3dSBYKcSV-LCoeQqfX1RYOo3qOK7g.ttf">
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://fonts.gstatic.com/s/sourcesanspro/v14/6xKydSBYKcSV-LCoeQqfX1RYOo3ig4vwlxdr.ttf">
|
||
|
||
|
||
<!--
|
||
Preload any font-awesome assets we might want to use
|
||
|
||
Identify these URLs by watching the network panel in Chrome when loading pages. Update
|
||
them whenever we change font-awesome version.
|
||
-->
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://use.fontawesome.com/releases/v5.8.2/webfonts/fa-brands-400.woff2">
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://use.fontawesome.com/releases/v5.8.2/webfonts/fa-solid-900.woff2">
|
||
|
||
|
||
<!-- CSS -->
|
||
<link rel="stylesheet" href="/assets/css/main.css">
|
||
|
||
<!-- Favicon -->
|
||
<link rel="shortcut icon" href="/assets/favicon.ico" type="image/x-icon">
|
||
|
||
<!-- RSS -->
|
||
<link rel="alternate" type="application/atom+xml" title="Lawrence Jones"
|
||
href="/feed.xml" />
|
||
|
||
<!--
|
||
Font Awesome
|
||
|
||
Configured to lazily load, so it doesn't block the page
|
||
-->
|
||
<link
|
||
rel="preload"
|
||
as="style"
|
||
onload="this.rel='stylesheet'"
|
||
href="https://use.fontawesome.com/releases/v5.8.2/css/all.css"
|
||
integrity="sha384-oS3vJWv+0UjzBfQzYUhtDYW+Pj2yciDJxpsK1OYPAYjqT085Qq/1cq5FLXAZQ7Ay"
|
||
crossorigin="anonymous">
|
||
|
||
<!-- KaTeX -->
|
||
|
||
|
||
<!-- Google Analytics, fast loading version -->
|
||
|
||
<script async src="https://www.googletagmanager.com/gtag/js?id=G-2FV47623W0"></script>
|
||
<script>
|
||
window.dataLayer = window.dataLayer || [];
|
||
function gtag(){dataLayer.push(arguments);}
|
||
gtag('js', new Date());
|
||
|
||
gtag('config', 'G-2FV47623W0');
|
||
</script>
|
||
|
||
|
||
|
||
<!-- Begin Jekyll SEO tag v2.8.0 -->
|
||
<title>Adding latency: one step, two step, oops</title>
|
||
<meta name="generator" content="Jekyll v4.2.2" />
|
||
<meta property="og:title" content="Adding latency: one step, two step, oops" />
|
||
<meta property="og:locale" content="en_US" />
|
||
<meta name="description" content="When it comes to complex systems, you can only go so far with synthetic experiments before you need to try something for real, and test in production. There's no substitute for it, and you're likely making the wrong decision if you avoid it. But I can say from experience it's not without risks, and this post shares an example of where we got as much as possible from that prework and testing, even if it was a bit of a bumpy ride." />
|
||
<meta property="og:description" content="When it comes to complex systems, you can only go so far with synthetic experiments before you need to try something for real, and test in production. There's no substitute for it, and you're likely making the wrong decision if you avoid it. But I can say from experience it's not without risks, and this post shares an example of where we got as much as possible from that prework and testing, even if it was a bit of a bumpy ride." />
|
||
<link rel="canonical" href="https://blog.lawrencejones.dev/latency/" />
|
||
<meta property="og:url" content="https://blog.lawrencejones.dev/latency/" />
|
||
<meta property="og:image" content="https://blog.lawrencejones.dev/assets/images/latency-social.png" />
|
||
<meta property="og:type" content="article" />
|
||
<meta property="article:published_time" content="2022-08-20T12:00:00+00:00" />
|
||
<meta name="twitter:card" content="summary_large_image" />
|
||
<meta property="twitter:image" content="https://blog.lawrencejones.dev/assets/images/latency-social.png" />
|
||
<meta property="twitter:title" content="Adding latency: one step, two step, oops" />
|
||
<meta name="twitter:site" content="@lawrjones" />
|
||
<script type="application/ld+json">
|
||
{"@context":"https://schema.org","@type":"BlogPosting","dateModified":"2022-08-20T12:00:00+00:00","datePublished":"2022-08-20T12:00:00+00:00","description":"When it comes to complex systems, you can only go so far with synthetic experiments before you need to try something for real, and test in production. There's no substitute for it, and you're likely making the wrong decision if you avoid it. But I can say from experience it's not without risks, and this post shares an example of where we got as much as possible from that prework and testing, even if it was a bit of a bumpy ride.","headline":"Adding latency: one step, two step, oops","image":"https://blog.lawrencejones.dev/assets/images/latency-social.png","mainEntityOfPage":{"@type":"WebPage","@id":"https://blog.lawrencejones.dev/latency/"},"url":"https://blog.lawrencejones.dev/latency/"}</script>
|
||
<!-- End Jekyll SEO tag -->
|
||
|
||
|
||
</head>
|
||
|
||
<body>
|
||
<header class="site-header">
|
||
<div class="header-content">
|
||
<div class="branding">
|
||
|
||
<a href="/">
|
||
<img class="avatar" src="https://secure.gravatar.com/avatar/a3d694b39e0e33fc479832b00dc128dc?s=105" alt="Gravatar picture of Lawrence">
|
||
</a>
|
||
|
||
<h1 class="site-title">
|
||
<a href="/">Lawrence Jones</a>
|
||
</h1>
|
||
</div>
|
||
<nav class="site-nav">
|
||
<ul>
|
||
|
||
|
||
|
||
|
||
|
||
<li>
|
||
<a class="page-link" href="/about/">
|
||
About
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<!-- Social icons from Font Awesome, if enabled -->
|
||
|
||
<li>
|
||
<a href="/feed.xml" title="Follow RSS feed">
|
||
<i class="fas fa-fw fa-rss"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
<li>
|
||
<a href="/cdn-cgi/l/email-protection#214c44614d405653444f42444b4e4f44520f454457" title="Email">
|
||
<i class="fas fa-fw fa-envelope"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li>
|
||
<a href="https://github.com/lawrencejones" title="Follow on GitHub" target="_blank" rel="noopener noreferrer">
|
||
<i class="fab fa-fw fa-github"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li>
|
||
<a href="https://twitter.com/lawrjones" title="Follow on Twitter" target="_blank" rel="noopener noreferrer">
|
||
<i style="color: rgba(29,161,242,1.00);" class="fab fa-fw fa-twitter"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<!-- Search bar -->
|
||
|
||
</ul>
|
||
</nav>
|
||
</div>
|
||
</header>
|
||
|
||
<div class="content">
|
||
<article>
|
||
<header style="background-image: url('/')">
|
||
<h1 class="title">Adding latency: one step, two step, oops</h1>
|
||
|
||
<p class="meta">
|
||
August 20, 2022
|
||
|
||
</p>
|
||
</header>
|
||
<section class="post-content">
|
||
<p>I’m a fan of testing in production, especially when it comes to complex systems
|
||
with a wide range of user behaviour. You can only go so far with synthetic
|
||
experiments before you need to try something out for real, but as with anything
|
||
in production, it’s not without risk.</p>
|
||
|
||
<p>Several years back I was at GoCardless, migrating our infrastructure from IBM
|
||
Softlayer to Google Cloud Platform. As a payments platform, GoCardless needed to
|
||
offer a reliable service: if we screwed up this migration, we’d drop requests
|
||
and prevent our merchant’s customers from completing checkout flows.</p>
|
||
|
||
<p>Keen to avoid this, we wanted to split the migration into milestones that could
|
||
answer specific concerns we had about the move, gradually increasing our
|
||
confidence until we felt comfortable moving even the most critical of our
|
||
systems.</p>
|
||
|
||
<p>Thankfully, GoCardless’ architecture was simple back then: we had a single Rails
|
||
app sat on top of a Postgres database, providing an HTTP API and running several
|
||
async workers.</p>
|
||
|
||
<p>The plan of attack would be:</p>
|
||
|
||
<ol>
|
||
<li>Move async workers into GCP, connecting to database in Softlayer</li>
|
||
<li>Move API into GCP, connecting to database in Softlayer</li>
|
||
<li>Move database into GCP</li>
|
||
</ol>
|
||
|
||
<p>Each one of these steps was risky, and could cause a customer facing problem.</p>
|
||
|
||
<p>As an example, we were moving the app into GKE, away from our home-grown
|
||
container management system in Softlayer: had we configured envvars correctly?
|
||
Would the networking perform the same? Might our resource limits be wrong?</p>
|
||
|
||
<p>That and several other binary it-works-or-it-doesn’t risks would be obvious from
|
||
the moment we try running work in GCP. Identifying these errors is why you move
|
||
the async workers first: they aren’t in a critical user path, will be retried,
|
||
and we could provide a fix before the next retry.</p>
|
||
|
||
<p>Those are the boring risks though, and this post is about a much nastier risk
|
||
than misconfiguration. The thing we were most concerned about, and what we
|
||
needed to really think about and test, was…</p>
|
||
|
||
<h2 id="performance">Performance.</h2>
|
||
|
||
<p>Notice that the plan explicitly mentions our async workers and API will be moved
|
||
to GCP, but continue to connect to the Postgres database inside of Softlayer.</p>
|
||
|
||
<p>That’s because adding the network hop via the IPSEC tunnel and over the public
|
||
internet is expensive, meaning all communication would suffer from a constant
|
||
latency penalty. It was going to change the cost of communicating with the
|
||
database from <strong>0.5ms in Softlayer to about 10ms when coming from GCP</strong>.</p>
|
||
|
||
<figure>
|
||
<img src="/assets/images/latency.png" alt="Diagram of Softlayer and GCP, with the two network hops compared and labelled with latency">
|
||
<figcaption style="margin-top: -20px; margin-bottom: 24px">
|
||
Softlayer and GCP, with the network hops labelled with latency.
|
||
</figcaption>
|
||
</figure>
|
||
|
||
<p>Rails apps are encouraged to assume latency to the database is essentially
|
||
‘free’, and ActiveRecord is an especially chatty ORM. Our app was no different,
|
||
and most codepaths would make huge numbers of queries to the database, where
|
||
those queries were about to become about 10x as expensive.</p>
|
||
|
||
<p>Going from 1ms to 10ms for every query was going to really suck, so our first
|
||
question was whether this was even possible: did the app even function when
|
||
queries took at least 10ms? What did that look like to a user of the app, was it
|
||
even usable?</p>
|
||
|
||
<h2 id="using-data">Using data</h2>
|
||
|
||
<p>When changing a fundamental operation cost like this, it’s best to establish the
|
||
lower bound of impact first, if only to rule out the plan entirely. If even the
|
||
best case scenario is intolerable, there’s no point discussing things further –
|
||
you’ll need another plan.</p>
|
||
|
||
<p>So before doing anything else, we wrote a <code class="language-plaintext highlighter-rouge">QueryMonitor</code> which added
|
||
instrumentation to the app, plugging into ActionSupport notifications about
|
||
ActiveRecord (the ORM) queries in order to measure the:</p>
|
||
|
||
<ul>
|
||
<li>Number of queries executed in a block</li>
|
||
<li>Time spent executing those queries in total</li>
|
||
</ul>
|
||
|
||
<p>We added it to the async worker code, to capture the database statistics from
|
||
running any async job and emit a <code class="language-plaintext highlighter-rouge">job.database_statistics</code> log when it
|
||
completes:</p>
|
||
|
||
<div class="language-ruby highlighter-rouge"><div class="highlight"><pre class="highlight"><code><span class="k">class</span> <span class="nc">Workers::BaseJob</span> <span class="o"><</span> <span class="no">Que</span><span class="o">::</span><span class="no">Job</span>
|
||
<span class="k">def</span> <span class="nf">run</span>
|
||
<span class="no">QueryMonitor</span><span class="p">.</span><span class="nf">trace</span> <span class="k">do</span>
|
||
<span class="k">super</span> <span class="c1"># run the job</span>
|
||
<span class="k">end</span>
|
||
<span class="k">ensure</span>
|
||
<span class="n">stats</span> <span class="o">=</span> <span class="no">QueryMonitor</span><span class="p">.</span><span class="nf">collect</span>
|
||
<span class="n">log</span><span class="p">(</span>
|
||
<span class="ss">event: </span><span class="s2">"job.database_statistics"</span><span class="p">,</span>
|
||
<span class="ss">job_name: </span><span class="n">job_name</span><span class="p">,</span>
|
||
<span class="ss">database_duration: </span><span class="n">stats</span><span class="p">.</span><span class="nf">duration</span><span class="p">,</span>
|
||
<span class="ss">database_query_count: </span><span class="n">stats</span><span class="p">.</span><span class="nf">query_count</span><span class="p">,</span>
|
||
<span class="p">)</span>
|
||
<span class="k">end</span>
|
||
<span class="k">end</span>
|
||
</code></pre></div></div>
|
||
|
||
<p>Then did somthing similar in our Rack middleware:</p>
|
||
|
||
<div class="language-ruby highlighter-rouge"><div class="highlight"><pre class="highlight"><code><span class="no">ActiveSupport</span><span class="o">::</span><span class="no">Notifications</span><span class="p">.</span>
|
||
<span class="nf">subscribe</span><span class="p">(</span><span class="s2">"start_handler.coach"</span><span class="p">)</span> <span class="k">do</span> <span class="o">|</span><span class="n">_</span><span class="p">,</span> <span class="n">event</span><span class="o">|</span>
|
||
<span class="no">QueryMonitor</span><span class="p">.</span><span class="nf">trace</span> <span class="c1"># start tracing</span>
|
||
<span class="k">end</span>
|
||
|
||
<span class="no">ActiveSupport</span><span class="o">::</span><span class="no">Notifications</span><span class="p">.</span>
|
||
<span class="nf">subscribe</span><span class="p">(</span><span class="s2">"request.coach"</span><span class="p">)</span> <span class="k">do</span> <span class="o">|</span><span class="nb">name</span><span class="p">,</span> <span class="n">event</span><span class="o">|</span>
|
||
<span class="n">stats</span> <span class="o">=</span> <span class="no">QueryMonitor</span><span class="p">.</span><span class="nf">collect</span> <span class="c1"># collect and clear trace</span>
|
||
<span class="n">log</span><span class="p">(</span>
|
||
<span class="ss">event: </span><span class="s2">"api_request.database_statistics"</span><span class="p">,</span>
|
||
<span class="ss">handler: </span><span class="n">event</span><span class="p">[</span><span class="ss">:handler</span><span class="p">],</span>
|
||
<span class="ss">database_duration: </span><span class="n">stats</span><span class="p">.</span><span class="nf">duration</span><span class="p">,</span>
|
||
<span class="ss">database_query_count: </span><span class="n">stats</span><span class="p">.</span><span class="nf">query_count</span><span class="p">,</span>
|
||
<span class="p">)</span>
|
||
<span class="k">end</span>
|
||
</code></pre></div></div>
|
||
|
||
<p>After running this for a week, I downloaded the data from our logging cluster
|
||
and built a spreadsheet for both workers and API requests that looked like this:</p>
|
||
|
||
<table>
|
||
<thead>
|
||
<tr>
|
||
<th>Workload</th>
|
||
<th>Query count</th>
|
||
<th>Duration</th>
|
||
<th>Duration @5ms</th>
|
||
<th>Duration @10ms</th>
|
||
</tr>
|
||
</thead>
|
||
<tbody>
|
||
<tr>
|
||
<td><code class="language-plaintext highlighter-rouge">API::Payments::Create</code></td>
|
||
<td>24</td>
|
||
<td>150ms</td>
|
||
<td>270ms</td>
|
||
<td>390ms</td>
|
||
</tr>
|
||
<tr>
|
||
<td><code class="language-plaintext highlighter-rouge">Workers::SubmitPayments</code></td>
|
||
<td>1,124,000</td>
|
||
<td>2,810s</td>
|
||
<td>8,430s</td>
|
||
<td>14,050s</td>
|
||
</tr>
|
||
</tbody>
|
||
</table>
|
||
|
||
<p>With a sprinkling of conditional formatting, the spreadsheets acted as a
|
||
heatmap, making it clear which endpoints or jobs were the worst offenders.
|
||
Printing them onto A3 poster paper, I visited each team and asked them to review
|
||
the list for workloads they were responsible for, asking them to think carefully
|
||
about how they’d adjust to each level of additional latency.</p>
|
||
|
||
<p>This exercise was really successful, with teams responding by:</p>
|
||
|
||
<ul>
|
||
<li>Reducing the number of queries their code made through use of joins and
|
||
preloads</li>
|
||
<li>Rearranging the scheduling of certain workers to ensure they finish in time</li>
|
||
<li>Splitting some jobs into parallel workers, and adding capacity to make up for
|
||
the slow down</li>
|
||
</ul>
|
||
|
||
<p>We rebuilt the spreadsheet after these changes and concluded that this could be
|
||
workable, if only for a short period of time. This was big! it meant the plan
|
||
was viable, but it didn’t answer all our concerns.</p>
|
||
|
||
<h2 id="not-that-easy">Not that easy</h2>
|
||
|
||
<p>Even with the data saying this should be fine, we’d made assumptions that were
|
||
unlikely to be valid, and meant our projections were a best case scenario only.
|
||
Probably the most critical and weakest assumption was that when something as
|
||
fundamental as the minimum cost for every query changes, the rest of the system
|
||
will continue otherwise unchanged: in other words, that the only thing the
|
||
network delay caused was individual queries becoming a bit slower.</p>
|
||
|
||
<p>That’s unlikely to be true, and one of the things we worried about was how
|
||
Postgres would respond in aggregate to longer individual queries. As an example,
|
||
we ran Postgres with pgBouncer as a connection pooler, allowing the Rails app to
|
||
share a small number of real Postgres connections with a much larger group of
|
||
clients, shared on a transaction basis.</p>
|
||
|
||
<p>This is best practice for Postgres, as key parts of the database need to iterate
|
||
through open connections to provide visibility guarantees: keeping the number of
|
||
connections low avoids making that critical path slow, improving database
|
||
health.</p>
|
||
|
||
<p>That’s great when your Rails app opens a transaction, issues a burst of quick
|
||
queries and immediately releases the connection back to the pool, but the
|
||
additional latency means we’re no longer doing that. That burst of queries is
|
||
now taking at least query count * 10ms, causing our average transaction duration
|
||
to increase, impacting the number of open connections in Postgres, impacting
|
||
core database performance, etc…</p>
|
||
|
||
<p>Accomodating this would require us to make many changes, some basic – like
|
||
measuring our new connection usage and increasing our connection limits to
|
||
permit them – and others more subtle, like tweaking the database for a higher
|
||
connection count (e.g. enabling huge-pages).</p>
|
||
|
||
<p>So while our data gave us confidence, there were still many questions we were
|
||
yet to answer, all of which could blow up the plan. And given the amount of
|
||
effort involved in performing this move, it would be good to be more confident
|
||
in the approach before committing to the move itself.</p>
|
||
|
||
<p>This was the point at which testing and hypothesising could take us no further,
|
||
and we needed to get our hands dirty to see how the systems actually behaved.</p>
|
||
|
||
<p>So how do we do that?</p>
|
||
|
||
<h2 id="time-to-experiment">Time to experiment</h2>
|
||
|
||
<p>As we anticipate certain issues will only appear at different levels of latency
|
||
increase (e.g. 3ms, 7ms, 9ms) then any test must be able to gradually increase
|
||
the latency. And because we know problems will occur, that latency should be
|
||
easily reversible, so we can revert to normal operations while finding a longer
|
||
term fix.</p>
|
||
|
||
<p>We decided to run an experiment over 1 week, which followed a process of:</p>
|
||
|
||
<ul>
|
||
<li>Check systems look healthy from:
|
||
<ul>
|
||
<li>Purpose built dashboards showing database healthy, API and worker capacity,
|
||
saturation, etc</li>
|
||
<li>Compare some hollistic statistics to ‘healthy’ parameters we’d defined
|
||
beforehand</li>
|
||
</ul>
|
||
</li>
|
||
<li>If healthy, increase latency by 1ms</li>
|
||
<li>Wait a day to gather data</li>
|
||
<li>Repeat until we reach 10ms</li>
|
||
</ul>
|
||
|
||
<p>Adding artificial network latency could be done via iptables - the kernel
|
||
network subsystem that controls all network packets - using a tool called
|
||
<a href="https://man7.org/linux/man-pages/man8/tc.8.html" target="_blank" rel="noopener noreferrer">tc</a> (traffic control) which we ran on the primary Postgres node, targeting
|
||
packets coming from the Rails app instances.</p>
|
||
|
||
<p>The tc command to add 3ms of latency would look like this:</p>
|
||
|
||
<div class="language-plaintext highlighter-rouge"><div class="highlight"><pre class="highlight"><code>tc qdisc add dev eth0 root netem delay 3ms
|
||
</code></pre></div></div>
|
||
|
||
<p>Just as expected, we hit problems at almost every additional 1ms of latency.</p>
|
||
|
||
<p>Sometimes it was simple, like workers overrunning, while others were more
|
||
complex. At one point we added more workers to adjust for the latency, but that
|
||
ate up our Postgres connections, so we increased the size of the pgBouncer pools
|
||
which caused another process to execute much quicker, causing other problems!</p>
|
||
|
||
<p>Especially for the more complex issues, we could never have predicted them in
|
||
advance: at least not in the detail we’d need to proactively fix them. It was
|
||
much safer to find them in a controlled environment where we could easily
|
||
rollback than it would have been during the real migration, with infrastructure
|
||
split over two providers.</p>
|
||
|
||
<p>There was a hitch, though. While we’d checked our internal processes were ok,
|
||
we happened to impact an API user who was submitting a big batch of payments
|
||
just before our payment deadline, and was doing so in sequence with no
|
||
parallelism.</p>
|
||
|
||
<p>While we offered no guarantee or advice on the performance of our API, in an
|
||
example of <a href="https://www.hyrumslaw.com/" target="_blank" rel="noopener noreferrer">Hyrum’s Law</a>, this customer had come to rely on a very
|
||
specific API performance in order to hit the payment deadline. As the API got
|
||
slower, they moved closer to the deadline, and we had to abort the experiment to
|
||
ensure they could make their end of month run.</p>
|
||
|
||
<p>This was difficult, and is the biggest downside of testing in production: shit
|
||
happens, and sometimes things go wrong.</p>
|
||
|
||
<p>My biggest learning was this didn’t invalidate the experiment, or meant we’d
|
||
made the wrong call. While painful at the time, this whole process had been
|
||
about reducing risk for the entire customer base, and the counterfactual where
|
||
we had a prolonged outage would have been much worse for all parties than a
|
||
single customer being impacted.</p>
|
||
|
||
<p>That said, I keep this experience in mind whenever doing similar work now. I’m
|
||
confident it was the right call, but it’s never comfortable when you’ve
|
||
negatively impacted a customer.</p>
|
||
|
||
<p>Burned fingers, and a hope to do better next time!</p>
|
||
|
||
|
||
<p>
|
||
<em>
|
||
|
||
If you liked this post and want to see more, follow me on <a target="_blank" href="https://www.linkedin.com/in/lawrence2jones/" rel="noopener noreferrer">LinkedIn</a>.
|
||
</em>
|
||
</p>
|
||
</section>
|
||
</article>
|
||
|
||
<!-- Disqus -->
|
||
|
||
|
||
<!-- Post navigation -->
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<div id="post-nav">
|
||
|
||
<div id="previous-post" class="post-nav-post">
|
||
<p>Previous post</p>
|
||
<a href="/growing-into-platform-engineering/">
|
||
Growing into Platform Engineering
|
||
</a>
|
||
</div>
|
||
|
||
|
||
<div id="next-post" class="post-nav-post">
|
||
<p>Next post</p>
|
||
<a href="/learn-at-scale-up/">
|
||
Want to found a start-up? Work at one first!
|
||
</a>
|
||
</div>
|
||
|
||
</div>
|
||
|
||
|
||
|
||
</div>
|
||
|
||
|
||
|
||
<footer class="site-footer">
|
||
<a href="https://www.linkedin.com/in/lawrence2jones/" target="_blank" rel="noopener noreferrer">
|
||
<i class="fab fa-linkedin" style="color: #0077b5;"></i> LinkedIn
|
||
</a>
|
||
</footer>
|
||
|
||
|
||
<script data-cfasync="false" src="/cdn-cgi/scripts/5c5dd728/cloudflare-static/email-decode.min.js"></script><script type="module" src="https://static.cloudflareinsights.com/beacon.min.js/v31edd6df95cf4e85bb4c19e7a9bdbcba1788362987495" integrity="sha512-iIg7k2xntmwu6/uSb5tpc/hySgZc4eoL31yB29W6tJFo2akwjPWcEqnCEdJvGexCL0KEQwVYv5BlowfhVz26hg==" data-cf-beacon='{"version":"2024.11.0","token":"292392cdf194479a91b07f11fd03de82","r":1,"spa":2}' crossorigin="anonymous"></script>
|
||
</body>
|
||
</html>
|