491 lines
17 KiB
HTML
491 lines
17 KiB
HTML
<!DOCTYPE html>
|
||
<html class="no-js" lang="en">
|
||
<head>
|
||
<meta charset="utf-8">
|
||
<title>When Game Days go wrong | Lawrence Jones</title>
|
||
<meta name="description"
|
||
content=" A story about how incident response training went wrong, with valuable lessons about pod priorities, isolation, and the importance of a healthy incident ...">
|
||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||
|
||
|
||
<!-- If this is an external_url, we want to redirect -->
|
||
|
||
|
||
<!-- Preload Google fonts, which is defined in sass -->
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://fonts.gstatic.com/s/sourcesanspro/v14/6xK1dSBYKcSV-LCoeQqfX1RYOo3qPZ7nsDc.ttf">
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://fonts.gstatic.com/s/sourcesanspro/v14/6xKwdSBYKcSV-LCoeQqfX1RYOo3qPZZclSds18E.ttf">
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://fonts.gstatic.com/s/sourcesanspro/v14/6xK3dSBYKcSV-LCoeQqfX1RYOo3qOK7g.ttf">
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://fonts.gstatic.com/s/sourcesanspro/v14/6xKydSBYKcSV-LCoeQqfX1RYOo3ig4vwlxdr.ttf">
|
||
|
||
|
||
<!--
|
||
Preload any font-awesome assets we might want to use
|
||
|
||
Identify these URLs by watching the network panel in Chrome when loading pages. Update
|
||
them whenever we change font-awesome version.
|
||
-->
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://use.fontawesome.com/releases/v5.8.2/webfonts/fa-brands-400.woff2">
|
||
|
||
<link rel="preload" as="font" crossorigin href="https://use.fontawesome.com/releases/v5.8.2/webfonts/fa-solid-900.woff2">
|
||
|
||
|
||
<!-- CSS -->
|
||
<link rel="stylesheet" href="/assets/css/main.css">
|
||
|
||
<!-- Favicon -->
|
||
<link rel="shortcut icon" href="/assets/favicon.ico" type="image/x-icon">
|
||
|
||
<!-- RSS -->
|
||
<link rel="alternate" type="application/atom+xml" title="Lawrence Jones"
|
||
href="/feed.xml" />
|
||
|
||
<!--
|
||
Font Awesome
|
||
|
||
Configured to lazily load, so it doesn't block the page
|
||
-->
|
||
<link
|
||
rel="preload"
|
||
as="style"
|
||
onload="this.rel='stylesheet'"
|
||
href="https://use.fontawesome.com/releases/v5.8.2/css/all.css"
|
||
integrity="sha384-oS3vJWv+0UjzBfQzYUhtDYW+Pj2yciDJxpsK1OYPAYjqT085Qq/1cq5FLXAZQ7Ay"
|
||
crossorigin="anonymous">
|
||
|
||
<!-- KaTeX -->
|
||
|
||
|
||
<!-- Google Analytics, fast loading version -->
|
||
|
||
<script async src="https://www.googletagmanager.com/gtag/js?id=G-2FV47623W0"></script>
|
||
<script>
|
||
window.dataLayer = window.dataLayer || [];
|
||
function gtag(){dataLayer.push(arguments);}
|
||
gtag('js', new Date());
|
||
|
||
gtag('config', 'G-2FV47623W0');
|
||
</script>
|
||
|
||
|
||
|
||
<!-- Begin Jekyll SEO tag v2.8.0 -->
|
||
<title>When Game Days go wrong</title>
|
||
<meta name="generator" content="Jekyll v4.2.2" />
|
||
<meta property="og:title" content="When Game Days go wrong" />
|
||
<meta property="og:locale" content="en_US" />
|
||
<meta name="description" content="A story about how incident response training went wrong, with valuable lessons about pod priorities, isolation, and the importance of a healthy incident response culture." />
|
||
<meta property="og:description" content="A story about how incident response training went wrong, with valuable lessons about pod priorities, isolation, and the importance of a healthy incident response culture." />
|
||
<link rel="canonical" href="https://blog.lawrencejones.dev/game-days-go-wrong/" />
|
||
<meta property="og:url" content="https://blog.lawrencejones.dev/game-days-go-wrong/" />
|
||
<meta property="og:type" content="article" />
|
||
<meta property="article:published_time" content="2024-12-08T12:00:00+00:00" />
|
||
<meta name="twitter:card" content="summary" />
|
||
<meta property="twitter:title" content="When Game Days go wrong" />
|
||
<meta name="twitter:site" content="@lawrjones" />
|
||
<script type="application/ld+json">
|
||
{"@context":"https://schema.org","@type":"BlogPosting","dateModified":"2024-12-08T12:00:00+00:00","datePublished":"2024-12-08T12:00:00+00:00","description":"A story about how incident response training went wrong, with valuable lessons about pod priorities, isolation, and the importance of a healthy incident response culture.","headline":"When Game Days go wrong","mainEntityOfPage":{"@type":"WebPage","@id":"https://blog.lawrencejones.dev/game-days-go-wrong/"},"url":"https://blog.lawrencejones.dev/game-days-go-wrong/"}</script>
|
||
<!-- End Jekyll SEO tag -->
|
||
|
||
|
||
</head>
|
||
|
||
<body>
|
||
<header class="site-header">
|
||
<div class="header-content">
|
||
<div class="branding">
|
||
|
||
<a href="/">
|
||
<img class="avatar" src="https://secure.gravatar.com/avatar/a3d694b39e0e33fc479832b00dc128dc?s=105" alt="Gravatar picture of Lawrence">
|
||
</a>
|
||
|
||
<h1 class="site-title">
|
||
<a href="/">Lawrence Jones</a>
|
||
</h1>
|
||
</div>
|
||
<nav class="site-nav">
|
||
<ul>
|
||
|
||
|
||
|
||
|
||
|
||
<li>
|
||
<a class="page-link" href="/about/">
|
||
About
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<!-- Social icons from Font Awesome, if enabled -->
|
||
|
||
<li>
|
||
<a href="/feed.xml" title="Follow RSS feed">
|
||
<i class="fas fa-fw fa-rss"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
<li>
|
||
<a href="/cdn-cgi/l/email-protection#cca1a98ca0adbbbea9a2afa9a6a3a2a9bfe2a8a9ba" title="Email">
|
||
<i class="fas fa-fw fa-envelope"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li>
|
||
<a href="https://github.com/lawrencejones" title="Follow on GitHub" target="_blank" rel="noopener noreferrer">
|
||
<i class="fab fa-fw fa-github"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<li>
|
||
<a href="https://twitter.com/lawrjones" title="Follow on Twitter" target="_blank" rel="noopener noreferrer">
|
||
<i style="color: rgba(29,161,242,1.00);" class="fab fa-fw fa-twitter"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<!-- Search bar -->
|
||
|
||
</ul>
|
||
</nav>
|
||
</div>
|
||
</header>
|
||
|
||
<div class="content">
|
||
<article>
|
||
<header style="background-image: url('/')">
|
||
<h1 class="title">When Game Days go wrong</h1>
|
||
|
||
<p class="meta">
|
||
December 8, 2024
|
||
|
||
</p>
|
||
</header>
|
||
<section class="post-content">
|
||
<p>It was the week before pandemic lockdowns began.</p>
|
||
|
||
<p>Like many companies, we were thinking about what a fully remote workforce might
|
||
mean for us. In an attempt to get ahead of things, we scheduled a <a href="https://incident.io/blog/game-day" target="_blank" rel="noopener noreferrer">‘Game
|
||
Day’</a> to test our work-from-home incident response capabilities. You
|
||
know, just in case we’d need to abandon the office (spoiler: we would).</p>
|
||
|
||
<p>The timing was perfect. We had several new team members and others who were
|
||
almost ready to join the on-call rotation but needed a confidence boost. Our
|
||
goal was simple: familiarise people with our systems and processes, deliberately
|
||
steering away from stress testing the infrastructure itself.</p>
|
||
|
||
<p>Focusing on how we’d respond remotely was the point of this, so we planned to
|
||
break things in obvious ways and watch how the incident response unfolded.</p>
|
||
|
||
<p>As the designated villain for the day, I had a straightforward plan:</p>
|
||
<ol>
|
||
<li>Detach a disk from our Postgres VM (running in a three-node cluster using
|
||
<a href="https://github.com/sorintlab/stolon" target="_blank" rel="noopener noreferrer">Stolon</a> and <a href="https://etcd.io/" target="_blank" rel="noopener noreferrer">Etcd</a>)</li>
|
||
<li>Fill up an <a href="https://www.elastic.co/elasticsearch" target="_blank" rel="noopener noreferrer">Elasticsearch</a> disk with junk (running in Kubernetes)</li>
|
||
<li>Create tons of consoles to exhaust cluster capacity</li>
|
||
</ol>
|
||
|
||
<p>Pretty standard stuff. In fact, it sounded like it might even be fun.</p>
|
||
|
||
<h2 id="all-according-to-plan">All according to plan</h2>
|
||
|
||
<p>For a while it was, with the Game Day starting smoothly.</p>
|
||
|
||
<p>When Postgres died and recovered, the team opened an incident and handled it
|
||
well. As they were wrapping up the Postgres recovery, I started filling up the
|
||
Elasticsearch disk, hoping to nudge the fairly large (~30TB) cluster into a
|
||
degraded state.</p>
|
||
|
||
<p>Alerts began firing, and the team efficiently split into two groups to handle
|
||
both issues. Perfect! This is exactly the type of response we were after,
|
||
proving we could handle ramping pressure even while remote.</p>
|
||
|
||
<p>It was time, then, to make things more interesting. So I started creating
|
||
a load of consoles in the Kubernetes cluster.</p>
|
||
|
||
<p>The console creation was meant to gradually fill the Kubernetes cluster with
|
||
batch jobs, exhausting our staging clusters capacity. This should have been safe
|
||
– we had <a href="https://kubernetes.io/docs/concepts/scheduling-eviction/pod-priority-preemption/" target="_blank" rel="noopener noreferrer">pod priorities</a> ensuring staging workloads were
|
||
assigned much lower priorities than production ones. No matter how many staging
|
||
consoles were created in the cluster there should be no impact on production
|
||
workloads, as Kubernetes will evict staging over production workloads first.</p>
|
||
|
||
<p>So when someone from support raised the alarm that the production app was down,
|
||
I was more than a little confused.</p>
|
||
|
||
<h2 id="going-sideways">Going sideways</h2>
|
||
|
||
<p>It was at this point that everything started going sideways.</p>
|
||
|
||
<p>Alerts were firing everywhere: Postgres alerts, HAProxy reporting no backends,
|
||
blackbox alerts confirming our API was unresponsive, and – most worryingly –
|
||
Etcd cluster out of quorum. That last one was particularly problematic because
|
||
Stolon relies on Etcd for leadership election. No Etcd quorum means no database
|
||
connections, which means no API requests, no batch jobs… no nothing.</p>
|
||
|
||
<p>The root cause? A perfect storm of “tomorrow’s problems” catching up with us.</p>
|
||
|
||
<p>At the time, we were running everything (production and staging) in a single
|
||
large Kubernetes cluster, separating environments and cluster infrastructure
|
||
using namespaces. We used pod priorities to ensure workloads would arrange
|
||
themselves correctly – cluster-level infrastructure above production, production
|
||
above staging, and so on.</p>
|
||
|
||
<p>But we’d made a critical mistake, one that would come back to haunt us in the
|
||
worst possible way. It all traced back to when we initially rolled out pod
|
||
priorities…</p>
|
||
|
||
<h2 id="the-mistake">The mistake</h2>
|
||
|
||
<p>When we initially rolled out pod priorities, we thought we’d be physically
|
||
separating the cluster within a month. To avoid doing substantial operational
|
||
work on something that would be rebuilt “soon,” we’d skipped applying pod
|
||
priorities to some sensitive deployments, including Etcd.</p>
|
||
|
||
<p>Here’s where things get interesting: in Kubernetes, pod priorities work through
|
||
a numeric class system. We had set this up pretty sensibly, with system
|
||
workloads at priority 300, production at 200, and staging at 100. The idea is
|
||
simple – when the cluster needs to make space, it evicts pods with lower
|
||
priorities first.</p>
|
||
|
||
<p>But there’s a subtle gotcha that bit us hard. Before you introduce any priority
|
||
classes to your cluster, every pod essentially has equal priority. They’re all
|
||
implicitly set to 0, but it doesn’t matter because they’re all the same. The
|
||
moment you add even a single priority class, though, that changes completely –
|
||
now any pod without an explicit priority gets assigned 0, making it lower
|
||
priority than everything else.</p>
|
||
|
||
<blockquote>
|
||
<p>There exists a ‘globalDefault’ property that can be set for a priority class
|
||
which means any pod without a class will get assigned that priority. We didn’t
|
||
have this set in our cluster, though I’d advise people set this if they’re
|
||
using pod priorities.</p>
|
||
</blockquote>
|
||
|
||
<p>By not setting a priority on our Etcd pods while having priorities everywhere
|
||
else, we’d accidentally marked our most critical infrastructure component as the
|
||
first thing to be evicted under pressure. When I came along and began filling
|
||
the cluster with staging consoles (a job I admit to doing with enthusiasm), we
|
||
started evicting Etcd pods. By the time alerts fired, we’d evicted enough pods
|
||
to lose quorum, having eaten well into our <a href="https://kubernetes.io/docs/concepts/workloads/pods/disruptions/" target="_blank" rel="noopener noreferrer">pod disruption
|
||
budgets</a>.</p>
|
||
|
||
<p>This prevented the Etcd pods from successfully rejoining the cluster when
|
||
restarted. We were, in technical terms, pretty screwed.</p>
|
||
|
||
<h2 id="righting-the-ship">Righting the ship</h2>
|
||
|
||
<p>The immediate focus was restoring database access. We deactivated the Stolon
|
||
Postgres cluster manager and booted Postgres manually, relying on muscle memory
|
||
for the right commands. We bypassed both PgBouncer and the Stolon fencing proxy,
|
||
pointing applications directly at Postgres. This brought things back up, albeit
|
||
in an unmanaged state.</p>
|
||
|
||
<p>After taking a moment to contemplate our life choices, we formulated a plan
|
||
using Etcd’s disaster recovery procedure. It took about six hours to bring
|
||
everything back and perform a Stolon failover. A good chunk of that time was
|
||
spent running <code class="language-plaintext highlighter-rouge">pg_basebackup</code> to restore another node – our database was about
|
||
6TB at this point.</p>
|
||
|
||
<p>It was a long day and not at all what we’d had planned. As drills go though,
|
||
this was probably the most intense way we could ever have tested our production
|
||
readiness in a remote context, even if it cost us more than we’d anticipated.</p>
|
||
|
||
<h2 id="infrastructure-debt-doesnt-age-well">Infrastructure debt doesn’t age well</h2>
|
||
|
||
<p>There’s a pattern in infrastructure work that’s worth recognising: we often
|
||
defer work because we think a better solution is coming “soon.” But “soon” has a
|
||
way of stretching into weeks or months, and the work we carefully documented and
|
||
planned to revisit gets buried under newer, more urgent tasks.</p>
|
||
|
||
<p>Over time, this creates hidden trapdoors in our infrastructure. When we leave
|
||
configuration half-applied or work partially complete, we’re not just creating
|
||
technical debt – we’re fragmenting our team’s mental model of the system. Every
|
||
“we’ll fix it later” adds another dimension of complexity, often temporal, that
|
||
someone needs to keep in mind. Eventually, someone (in this case, me) will
|
||
forget about that complexity and fall right through the trapdoor.</p>
|
||
|
||
<h2 id="the-real-test-isnt-the-outage">The real test isn’t the outage</h2>
|
||
|
||
<p>Of everything I took from this incident, though, it was how our organisation
|
||
responded that I found most interesting.</p>
|
||
|
||
<p>In preparing for an important change (remote work during COVID), while trying to
|
||
do the right thing for the company (training new incident responders), we
|
||
accidentally caused a significant outage. It would have been easy to say “well,
|
||
we won’t be doing Game Days again!” or worse, fire the person (me) who pressed
|
||
the buttons and wash your hands of their incomptence (ouch).</p>
|
||
|
||
<p>Instead, my boss, our CTO, caught up with me afterward to say “thank god we
|
||
found this while the whole team were on a call ready to respond!”. It’s true,
|
||
this would have been far worse had it happened out of hours, but you bet he’d
|
||
got a lot of heat for this outage. It would have been easy to push blame to me,
|
||
but he didn’t, and in that moment he created the essential conditions of a
|
||
‘blameless’ incident culture. One where an outage like this can balance its cost
|
||
with the lessons learned from it.</p>
|
||
|
||
<p>Sometimes the most valuable lessons come from the most uncomfortable situations.
|
||
The trick is making sure your team feels safe enough to learn from them.</p>
|
||
|
||
|
||
<p>
|
||
<em>
|
||
|
||
If you liked this post and want to see more, follow me on <a target="_blank" href="https://www.linkedin.com/in/lawrence2jones/" rel="noopener noreferrer">LinkedIn</a>.
|
||
</em>
|
||
</p>
|
||
</section>
|
||
</article>
|
||
|
||
<!-- Disqus -->
|
||
|
||
|
||
<!-- Post navigation -->
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<div id="post-nav">
|
||
|
||
<div id="previous-post" class="post-nav-post">
|
||
<p>Previous post</p>
|
||
<a href="/learn-one-thing/">
|
||
Learn one thing at a time
|
||
</a>
|
||
</div>
|
||
|
||
|
||
<div id="next-post" class="post-nav-post">
|
||
<p>Next post</p>
|
||
<a href="/2024/">
|
||
Looking back at 2024
|
||
</a>
|
||
</div>
|
||
|
||
</div>
|
||
|
||
|
||
|
||
</div>
|
||
|
||
|
||
|
||
<footer class="site-footer">
|
||
<a href="https://www.linkedin.com/in/lawrence2jones/" target="_blank" rel="noopener noreferrer">
|
||
<i class="fab fa-linkedin" style="color: #0077b5;"></i> LinkedIn
|
||
</a>
|
||
</footer>
|
||
|
||
|
||
<script data-cfasync="false" src="/cdn-cgi/scripts/5c5dd728/cloudflare-static/email-decode.min.js"></script><script type="module" src="https://static.cloudflareinsights.com/beacon.min.js/v31edd6df95cf4e85bb4c19e7a9bdbcba1788362987495" integrity="sha512-iIg7k2xntmwu6/uSb5tpc/hySgZc4eoL31yB29W6tJFo2akwjPWcEqnCEdJvGexCL0KEQwVYv5BlowfhVz26hg==" data-cf-beacon='{"version":"2024.11.0","token":"292392cdf194479a91b07f11fd03de82","r":1,"spa":2}' crossorigin="anonymous"></script>
|
||
</body>
|
||
</html>
|