Files
nexus/sreweekly/articles/468/04-scaling-prometheus-from-single-node-to-enterprise-grade-observability.html
2026-09-12 17:23:01 +08:00

852 lines
59 KiB
HTML
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
<!DOCTYPE html>
<html lang="en">
<head>
<title>Scaling Prometheus in 2026: Thanos vs Cortex vs Mimir</title>
<meta charset="utf-8" />
<meta http-equiv="X-UA-Compatible" content="IE=edge" />
<meta name="HandheldFriendly" content="True" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<link rel="preload" as="style" href="https://blog.oodle.ai/assets/built/screen.css?v=S0DSiw6jVHJaYsG0" />
<link rel="preload" as="script" href="https://blog.oodle.ai/assets/built/casper.js?v=YtPhrez3GuGLVDYs" />
<link rel="stylesheet" type="text/css" href="https://blog.oodle.ai/assets/built/screen.css?v=S0DSiw6jVHJaYsG0" />
<meta name="description" content="A single Prometheus server hits memory limits as series grow. How to scale it: sharding, federation, or moving to Thanos, Mimir, or VictoriaMetrics.">
<link rel="icon" href="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w256h256/2024/09/oodle_logo_1x.png" type="image/png">
<link rel="canonical" href="https://blog.oodle.ai/scaling-prometheus-from-single-node-to-enterprise-grade-observability/">
<meta name="referrer" content="no-referrer-when-downgrade">
<meta property="og:site_name" content="Oodle AI">
<meta property="og:type" content="article">
<meta property="og:title" content="Scaling Prometheus: How to Go From a Single Node to Enterprise-Grade Observability">
<meta property="og:description" content="Where a single Prometheus server breaks, and how Thanos, Cortex, Mimir, and VictoriaMetrics compare for scaling it. Updated for Prometheus 3.">
<meta property="og:url" content="https://blog.oodle.ai/scaling-prometheus-from-single-node-to-enterprise-grade-observability/">
<meta property="og:image" content="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/2026/07/oodle-scaling-prometheus-og-1.png">
<meta property="article:published_time" content="2025-03-10T04:12:24.000Z">
<meta property="article:modified_time" content="2026-07-31T13:18:21.000Z">
<meta property="article:tag" content="product">
<meta name="twitter:card" content="summary_large_image">
<meta name="twitter:title" content="Scaling Prometheus: How to Go From a Single Node to Enterprise-Grade Observability">
<meta name="twitter:description" content="Where a single Prometheus server breaks, and how Thanos, Cortex, Mimir, and VictoriaMetrics compare for scaling it. Updated for Prometheus 3.">
<meta name="twitter:url" content="https://blog.oodle.ai/scaling-prometheus-from-single-node-to-enterprise-grade-observability/">
<meta name="twitter:image" content="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/2026/07/oodle-scaling-prometheus-og.png">
<meta name="twitter:label1" content="Written by">
<meta name="twitter:data1" content="Gaurav Maheshwari">
<meta name="twitter:label2" content="Filed under">
<meta name="twitter:data2" content="product">
<meta name="twitter:site" content="@oodleai">
<meta name="twitter:creator" content="@gaurav25d">
<meta property="og:image:width" content="1200">
<meta property="og:image:height" content="630">
<script type="application/ld+json">
{
"@context": "https://schema.org",
"@type": "Article",
"publisher": {
"@type": "Organization",
"name": "Oodle AI",
"url": "https://blog.oodle.ai/",
"logo": {
"@type": "ImageObject",
"url": "https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/2025/04/logo_blue.png"
}
},
"author": {
"@type": "Person",
"name": "Gaurav Maheshwari",
"image": {
"@type": "ImageObject",
"url": "https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w1200/2024/11/_N2A0018-3.jpg",
"width": 1200,
"height": 1879
},
"url": "https://blog.oodle.ai/author/gaurav/",
"sameAs": [
"https://mgaurav.github.io/",
"https://x.com/gaurav25d"
]
},
"headline": "Scaling Prometheus in 2026: Thanos vs Cortex vs Mimir",
"url": "https://blog.oodle.ai/scaling-prometheus-from-single-node-to-enterprise-grade-observability/",
"datePublished": "2025-03-10T04:12:24.000Z",
"dateModified": "2026-07-31T13:18:21.000Z",
"image": {
"@type": "ImageObject",
"url": "https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w1200/2025/03/scalingprometheus.jpg",
"width": 1200,
"height": 481
},
"keywords": "product",
"description": "TL;DR\n\n\n * The problem: Prometheus is easy to run on one node, but its memory grows with every active time series, so a single server eventually hits a wall.\n * The scaling paths for running Prometheus at scale, in order of effort: reduce cardinality first, then split load with functional sharding, add federation for an aggregate view, and move to remote storage when you need durability and one query view.\n * The remote-storage verdict for 2026: Thanos to extend a working Prometheus fleet, Grafa",
"mainEntityOfPage": "https://blog.oodle.ai/scaling-prometheus-from-single-node-to-enterprise-grade-observability/"
}
</script>
<meta name="generator" content="Ghost 6.64">
<link rel="alternate" type="application/rss+xml" title="Oodle AI" href="https://blog.oodle.ai/rss/">
<script defer src="https://cdn.jsdelivr.net/ghost/portal@~2.71/umd/portal.min.js" data-i18n="true" data-ghost="https://blog.oodle.ai/" data-key="5d821cde2a2f12de6e45a55016" data-api="https://oodle-ai.ghost.io/ghost/api/content/" data-locale="en" crossorigin="anonymous"></script><style id="gh-members-styles">.gh-post-upgrade-cta-content,
.gh-post-upgrade-cta {
display: flex;
flex-direction: column;
align-items: center;
font-family: -apple-system, BlinkMacSystemFont, 'Segoe UI', Roboto, Oxygen, Ubuntu, Cantarell, 'Open Sans', 'Helvetica Neue', sans-serif;
text-align: center;
width: 100%;
color: #ffffff;
font-size: 16px;
}
.gh-post-upgrade-cta-content {
border-radius: 8px;
padding: 40px 4vw;
}
.gh-post-upgrade-cta h2 {
color: #ffffff;
font-size: 28px;
letter-spacing: -0.2px;
margin: 0;
padding: 0;
}
.gh-post-upgrade-cta p {
margin: 20px 0 0;
padding: 0;
}
.gh-post-upgrade-cta small {
font-size: 16px;
letter-spacing: -0.2px;
}
.gh-post-upgrade-cta a {
color: #ffffff;
cursor: pointer;
font-weight: 500;
box-shadow: none;
text-decoration: underline;
}
.gh-post-upgrade-cta a:hover {
color: #ffffff;
opacity: 0.8;
box-shadow: none;
text-decoration: underline;
}
.gh-post-upgrade-cta a.gh-btn {
display: block;
background: #ffffff;
text-decoration: none;
margin: 28px 0 0;
padding: 8px 18px;
border-radius: 4px;
font-size: 16px;
font-weight: 600;
}
.gh-post-upgrade-cta a.gh-btn:hover {
opacity: 0.92;
}</style>
<script defer src="https://cdn.jsdelivr.net/ghost/sodo-search@~1.8/umd/sodo-search.min.js" data-key="5d821cde2a2f12de6e45a55016" data-styles="https://cdn.jsdelivr.net/ghost/sodo-search@~1.8/umd/main.css" data-sodo-search="https://oodle-ai.ghost.io/" data-locale="en" crossorigin="anonymous"></script>
<link href="https://blog.oodle.ai/webmentions/receive/" rel="webmention">
<script defer src="/public/cards.min.js?v=ShRHxgy4po8zN-Wf"></script>
<link rel="stylesheet" type="text/css" href="/public/cards.min.css?v=WwnU9jw5ancNC8Gc">
<script defer src="/public/member-attribution.min.js?v=AKG4hWena9j3yX3I"></script>
<script defer src="/public/ghost-stats.min.js?v=vFcCUf6ZQ0Hyhc8h" data-stringify-payload="false" data-datasource="analytics_events" data-storage="localStorage" data-host="https://blog.oodle.ai/.ghost/analytics/api/v1/page_hit" tb_site_uuid="b2485d36-0dcd-4e46-b7bd-418ab6e38e18" tb_post_uuid="24b04aad-b255-41d2-8f36-1a84aebcb923" tb_post_type="post" tb_member_uuid="undefined" tb_member_status="undefined" tb_gift_link=""></script><style>:root {--ghost-accent-color: #0091cf;}</style>
<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/highlight.js/11.10.0/styles/atom-one-dark.min.css" integrity="sha512-Jk4AqjWsdSzSWCSuQTfYRIF84Rq/eV0G2+tu07byYwHcbTGfdmLrHjUSwvzp5HvbiqK4ibmNwdcG49Y5RGYPTg==" crossorigin="anonymous" referrerpolicy="no-referrer" />
<script src="https://cdnjs.cloudflare.com/ajax/libs/highlight.js/11.10.0/highlight.min.js" integrity="sha512-6yoqbrcLAHDWAdQmiRlHG4+m0g/CT/V9AGyxabG8j7Jk8j3r3K6due7oqpiRMZqcYe9WM2gPcaNNxnl2ux+3tA==" crossorigin="anonymous" referrerpolicy="no-referrer"></script>
<!-- and it's easy to individually load additional languages -->
<script src="https://cdn.jsdelivr.net/npm/prismjs/prism.min.js" defer></script>
<script src="https://cdn.jsdelivr.net/npm/prismjs/plugins/autoloader/prism-autoloader.min.js" defer></script>
<link rel="stylesheet" href="https://cdn.jsdelivr.net/npm/prismjs/themes/prism.min.css">
<script>
document.addEventListener('DOMContentLoaded', function () {
Prism.languages.d2 = {
comment: [
{ pattern: /"""[\s\S]*?"""/, greedy: true },
{ pattern: /#.*/ }
],
string: [
{ pattern: /"(?:\\.|[^"\\])*"/, greedy: true },
{ pattern: /'(?:\\.|[^'\\])*'/, greedy: true }
],
keyword:
/\b(?:style|shape|label|icon|link|tooltip|near|class|classes|vars|layers|steps|scenarios|direction|fill|stroke|opacity|font-size|font-color|shadow|multiple|animated|bold|italic|underline|border-radius|stroke-width|stroke-dash|double-border|3d|filled|width|height|top|left|grid-rows|grid-columns|grid-gap|vertical-gap|horizontal-gap|text-transform|fill-pattern|constraint|font)\b/,
boolean: /\b(?:true|false)\b/,
null: { pattern: /\bnull\b/, alias: 'keyword' },
arrow: {
pattern: /<?-+>?/,
alias: 'operator'
},
number: /\b\d+(?:\.\d+)?\b/,
punctuation: /[{}[\];:.]/
};
});
</script>
<!-- Google Analytics -->
<script async src="https://www.googletagmanager.com/gtag/js?id=G-M0CVSZP7TP"></script>
<script id="gtag-init">
window.dataLayer = window.dataLayer || [];
function gtag(){ window.dataLayer.push(arguments); };
gtag("js", new Date());
gtag("config", "G-M0CVSZP7TP");
</script>
<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/tocbot/4.32.2/tocbot.css" integrity="sha512-Di7Va5KC5NtXyMi+aEyVe2pUnniyhFoxfCrdCAOj8aSA42Te/bWKTz0iumaj5v1sN1nZdsaX8QTCn0k1nN4aLA==" crossorigin="anonymous" referrerpolicy="no-referrer" />
<style>
img{border-radius:10px;}.gh-content{position:relative}.gh-toc>.toc-list{position:relative}.toc-list{overflow:hidden;list-style:none}.gh-toc .is-active-link::before{background-color:var(--ghost-accent-color)}a.toc-link{display:inline-flex;font-weight:400;height:100%;line-height:1.2em;padding:6px 0;text-decoration:none;transition:.4s ease;font-size:100%!important}li.toc-list-item{color:#738a94!important}a.toc-link:hover{color:#15171a!important}.is-collapsible a.is-active-link,a.is-active-link{color:#15171a!important;font-weight:500}@media (max-width:1400px){.gh-toc{background:#fff;border-radius:1em;box-shadow:0 10px 50px rgba(25,37,52,.14),0 2px 5px rgba(25,37,52,.03);padding:30px;width:100%}}
.gh-sidebar .gh-toc .toc-list{padding-inline-start: 16px; margin-inline-start: 8px;}
@media (min-width:1300px){.gh-sidebar{position:absolute;top:0;bottom:0;margin-top:4vmin;grid-column:wide-start / main-start}.gh-toc{position:sticky;top:4vmin}li.toc-list-item{font-size:1.4rem}}
.blockquote {
position: relative;
align-self: center;
}
/* The Quote */
.blockquote .quote {
position: relative; /* for pseudos */
border-color: var(--ghost-accent-color);
color: var(--ghost-accent-color);
border-style: solid;
font-weight: normal;
font-size: 2rem;
font-weight: 400;
font-style: italic;
line-height: 1.6em;
margin: 0;
border-width: 2px;
border-radius:10px;
padding: 25px;
background: linear-gradient(180deg, rgba(186, 233, 255, 0.5) 0%, #fff 100%, #fff 100%);
}
/* Blockquote right double quotes */
.blockquote .quote:after {
content:"";
position: absolute;
border-color: var(--ghost-accent-color);
border-width: 2px;
border-style: solid;
border-radius: 0 100% 0 0;
width: 60px;
height: 60px;
bottom: -60px;
left: 50px;
border-bottom: none;
border-left: none;
z-index: 3;
}
.blockquote .quote:before {
content:"";
position: absolute;
width: 80px;
border: 6px solid #fff;
bottom: -3px;
left: 50px;
z-index: 2;
}
.blockquote .name {
margin-left: 130px;
margin-top: 10px;
font-weight: bold;
}
.blockquote .title {
margin-left: 130px;
}
</style>
<script type="application/ld+json">
{
"@context": "https://schema.org",
"@type": "FAQPage",
"mainEntity": [
{"@type": "Question", "name": "How do you scale Prometheus?",
"acceptedAnswer": {"@type": "Answer", "text": "In three stages: first reduce what you ingest (drop unused labels, use native histograms), then split load across servers with functional sharding and add federation for an aggregate global view, and finally move long-term storage to a horizontally scalable backend such as Thanos, Grafana Mimir, or VictoriaMetrics when you need durability and a single query view over raw data. That is the standard path to running Prometheus at scale."}},
{"@type": "Question", "name": "What is Prometheus sharding?",
"acceptedAnswer": {"@type": "Answer", "text": "Running multiple Prometheus servers that each scrape a subset of your targets, split by team, cluster, or service. Each server holds fewer time series, keeping memory manageable. The cost is that no single server can answer queries across all of your metrics."}},
{"@type": "Question", "name": "How many time series can a single Prometheus server handle?",
"acceptedAnswer": {"@type": "Answer", "text": "There is no hard limit; memory is the constraint. Well-provisioned single servers commonly run low single-digit millions of active series. Budget roughly a few kilobytes of RAM per active series, more with high churn, and watch query load, which competes for the same memory."}},
{"@type": "Question", "name": "What are the four types of metrics in Prometheus?",
"acceptedAnswer": {"@type": "Answer", "text": "Counter (a value that only goes up), gauge (a value that goes up and down), histogram (observations bucketed by size), and summary (like a histogram, with quantiles computed client-side). Prometheus 3 adds native histograms, a more efficient histogram representation."}},
{"@type": "Question", "name": "Is Cortex still maintained in 2026?",
"acceptedAnswer": {"@type": "Answer", "text": "Yes. Cortex remains a CNCF incubating project with regular releases (v1.21 shipped in 2026), and Amazon Managed Service for Prometheus is built on it. Its original maintainers moved to Grafana Mimir in 2022 and most new development happens there, which is why new self-hosted deployments usually pick Mimir."}},
{"@type": "Question", "name": "Does Prometheus scale horizontally?",
"acceptedAnswer": {"@type": "Answer", "text": "Not by default. Prometheus is a single-node system with no built-in clustering, so one server scales only vertically (more memory and CPU). Horizontal scale comes from functional sharding across servers, federation for an aggregate view, or remote-writing into a horizontally scalable backend such as Thanos, Mimir, or VictoriaMetrics."}},
{"@type": "Question", "name": "What are the limitations of Prometheus federation?",
"acceptedAnswer": {"@type": "Answer", "text": "Federation scrapes selected, usually pre-aggregated series from other Prometheus servers, so you still cannot query all raw data in one place. The federating server adds its own scrape delay, becomes another component to size and operate, and its load grows with every series it pulls. It also does nothing for long-term durability, which is why remote-storage backends exist."}},
{"@type": "Question", "name": "Which monitoring tools extend Prometheus for enterprises?",
"acceptedAnswer": {"@type": "Answer", "text": "The main open-source options are Thanos, Grafana Mimir, and VictoriaMetrics, which add durable storage, a global query view, and multi-tenancy on top of Prometheus. Managed options include Amazon Managed Service for Prometheus, Grafana Cloud, and fully Prometheus-compatible platforms such as Oodle."}}
]
}
</script>
<!-- Table fit: the theme forces nowrap + horizontal scroll on wide tables; let this post's
comparison table wrap and fit the content column instead -->
<style>
.gh-content table:not(.gist table){display:table;width:100%;white-space:normal}
.gh-content table:not(.gist table) td,.gh-content table:not(.gist table) th{white-space:normal;vertical-align:top}
.gh-content table:not(.gist table) code{white-space:normal;overflow-wrap:anywhere}
</style>
</head>
<body class="post-template tag-product is-head-left-logo has-sans-body">
<div class="viewport">
<header id="gh-head" class="gh-head outer is-header-hidden">
<div class="gh-head-inner inner">
<div class="gh-head-brand">
<a class="gh-head-logo" href="https://blog.oodle.ai">
<img src="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/2025/04/logo_blue.png" alt="Oodle AI">
</a>
<button class="gh-search gh-icon-btn" aria-label="Search this site" data-ghost-search><svg xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 24 24" stroke="currentColor" stroke-width="2" width="20" height="20"><path stroke-linecap="round" stroke-linejoin="round" d="M21 21l-6-6m2-5a7 7 0 11-14 0 7 7 0 0114 0z"></path></svg></button>
<button class="gh-burger" aria-label="Main Menu"></button>
</div>
<nav class="gh-head-menu">
<ul class="nav">
<li class="nav-home"><a href="https://blog.oodle.ai/">Home</a></li>
<li class="nav-discord"><a href="https://discord.gg/VdEuk9G3Gn">Discord</a></li>
<li class="nav-playground"><a href="https://play.oodle.ai/">Playground</a></li>
<li class="nav-sign-up"><a href="https://us1.oodle.ai/signup">Sign up</a></li>
<li class="nav-docs"><a href="https://docs.oodle.ai/">Docs</a></li>
<li class="nav-about"><a href="https://blog.oodle.ai/about/">About</a></li>
</ul>
</nav>
<div class="gh-head-actions">
<button class="gh-search gh-icon-btn" aria-label="Search this site" data-ghost-search><svg xmlns="http://www.w3.org/2000/svg" fill="none" viewBox="0 0 24 24" stroke="currentColor" stroke-width="2" width="20" height="20"><path stroke-linecap="round" stroke-linejoin="round" d="M21 21l-6-6m2-5a7 7 0 11-14 0 7 7 0 0114 0z"></path></svg></button>
<div class="gh-head-members">
<a class="gh-head-button" href="#/portal/signin" data-portal="signin">Sign in</a>
</div>
</div>
</div>
</header>
<div class="site-content">
<main id="site-main" class="site-main">
<article class="article post tag-product ">
<header class="article-header gh-canvas">
<div class="article-tag post-card-tags">
<span class="post-card-primary-tag">
<a href="/tag/product/">product</a>
</span>
</div>
<h1 class="article-title">Scaling Prometheus: How to Go From a Single Node to Enterprise-Grade Observability</h1>
<div class="article-byline">
<section class="article-byline-content">
<ul class="author-list instapaper_ignore">
<li class="author-list-item">
<a href="/author/gaurav/" class="author-avatar" aria-label="Read more of Gaurav Maheshwari">
<img class="author-profile-image" src="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w100/2024/11/_N2A0018-3.jpg" alt="Gaurav Maheshwari" loading="eager" />
</a>
</li>
</ul>
<div class="article-byline-meta">
<h4 class="author-name"><a href="/author/gaurav/">Gaurav Maheshwari</a></h4>
<div class="byline-meta-content">
<time class="byline-meta-date" datetime="2025-03-10">10 Mar 2025</time>
<span class="byline-reading-time"><span class="bull">&bull;</span> 11 min read</span>
</div>
</div>
</section>
<a href="#/share" class="gh-button gh-button-share">Share</a>
</div>
<figure class="article-image">
<img
class="gh-feature-image"
srcset="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w300/2025/03/scalingprometheus.jpg 300w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w600/2025/03/scalingprometheus.jpg 600w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w1000/2025/03/scalingprometheus.jpg 1000w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w2000/2025/03/scalingprometheus.jpg 2000w"
sizes="(min-width: 1400px) 1400px, 92vw"
src="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w2000/2025/03/scalingprometheus.jpg"
alt="Scaling Prometheus: How to Go From a Single Node to Enterprise-Grade Observability"
fetchpriority="high"
loading="eager"
/>
</figure>
</header>
<section class="gh-content gh-canvas">
<p><strong>TL;DR</strong></p>
<ul>
<li><strong>The problem:</strong> Prometheus is easy to run on one node, but its memory grows with every active time series, so a single server eventually hits a wall.</li>
<li><strong>The scaling paths for running Prometheus at scale, in order of effort:</strong> reduce cardinality first, then split load with <strong>functional sharding</strong>, add <strong>federation</strong> for an aggregate view, and move to <strong>remote storage</strong> when you need durability and one query view.</li>
<li><strong>The remote-storage verdict for 2026:</strong> Thanos to extend a working Prometheus fleet, Grafana Mimir for a new multi-team platform, VictoriaMetrics for the lightest operations. Cortex is maintained, but most new development happens in Mimir. Full comparison table below.</li>
<li><strong>New since this guide was written:</strong> Prometheus 3's native histograms cut series counts before you change any architecture.</li>
</ul>
<p>Scaling Prometheus is where most teams first meet the limits of a single monitoring server; this guide walks the full path from one node to enterprise-grade observability. Prometheus is an <a href="https://prometheus.io/?ref=blog.oodle.ai">open-source monitoring solution</a> that provides a streamlined way to store metrics data, query metrics using PromQL, and set up alerting. It has become the de-facto standard for monitoring infrastructure and applications.</p>
<ul>
<li><strong>Easy to setup</strong>: Prometheus is straightforward to deploy - simply run a single <code>prometheus</code> binary and you're ready to ingest and query metrics. It uses a single-node setup rather than a clustered architecture, making it simple to deploy but introducing scalability limitations that we'll discuss later.</li>
<li><strong>Service Discovery and Pull-based model</strong>: The Prometheus server pulls metrics from your services through service discovery. This simplifies deployment since services don't need to know about the central Prometheus setup. Each application service only needs to expose its metrics over an HTTP server in the <a href="https://github.com/prometheus/docs/blob/main/content/docs/instrumenting/exposition_formats.md?ref=blog.oodle.ai">Prometheus text exposition format</a>, and Prometheus will periodically pull these metrics based on the configured <code>scrape_interval</code>.</li>
<li><strong>Storage Engine:</strong> Prometheus stores time series data in memory and on local disk in an <a href="https://prometheus.io/docs/prometheus/latest/storage/?ref=blog.oodle.ai">efficient custom format</a>.</li>
<li><strong>Powerful Queries:</strong> Prometheus allows querying time series data via PromQL, a language that allows slicing and dicing of time series data over the labels and time ranges.</li>
<li><strong>Alerting:</strong> Prometheus supports alerting on the time series data with <a href="https://prometheus.io/docs/prometheus/latest/configuration/alerting_rules/?ref=blog.oodle.ai">alert rules</a> specified in PromQL, and notifications handled by <a href="https://prometheus.io/docs/alerting/latest/alertmanager/?ref=blog.oodle.ai">Alertmanager</a>.</li>
<li><strong>Third-party Integrations:</strong> Prometheus can easily collect data from third-party tools via <a href="https://prometheus.io/docs/instrumenting/exporters/?ref=blog.oodle.ai">exporters</a> and also serve as a data-source for dashboarding tools such as <a href="https://grafana.com/docs/grafana/latest/datasources/prometheus/?ref=blog.oodle.ai">Grafana</a>.</li>
</ul>
<p>Prometheus is a great starting point if you are just getting started with Observability. However, the reasons which make it easy to set up and operate at low scale are also the reasons why it can become challenging to operate Prometheus at scale.</p>
<p>To understand the limitations of Prometheus, we need to build a good mental model of how it stores data and what affects its scalability.</p>
<h2 id="time-series-and-high-cardinality-what-drives-memory-usage">Time Series and High Cardinality: What Drives Memory Usage</h2>
<p>Prometheus fundamentally stores all data as time series: streams of timestamped values belonging to the same metric and the same set of labeled dimensions.</p>
<p>For example, a time series with the metric name <code>api_http_requests_total</code> and the labels <code>method="POST"</code> and <code>handler="/messages"</code> could be written like this:</p>
<p><code>api_http_requests_total{method="POST", handler="/messages"}</code></p>
<p>For each time series, Prometheus stores a <code>(timestamp, value)</code> for the associated metric and set of labels at each <code>scrape_interval</code>. By default, Prometheus stores all time series in memory for two hours and flushes them to local on-disk storage at the end of the two-hour-interval.</p>
<p>Understanding Prometheus's data model helps us see that its memory usage directly correlates with the number of active time series it scrapes. This leads us to the <strong>cardinality explosion</strong> problem.</p>
<p>In the above example, if <code>method</code> can contain four possible values - <code>GET</code>, <code>PUT</code>, <code>POST</code>, and <code>DELETE</code>, and <code>handler</code> can contain ten possible values, the total cardinality of <code>api_http_requests_total</code> becomes <code>40 (4*10)</code>. If you then add a <code>status</code> label with two possible values (<code>success</code> or <code>failure</code>), the total cardinality increases to <code>80 (4*10*2)</code>.</p>
<p>High cardinality metrics can easily occur when you have dynamic labels or when the cross-product of multiple label cardinalities becomes large. Since the number of time series is determined by unique label combinations across all metrics, label cardinality directly impacts the Prometheus server's memory usage.</p>
<p>This highlights a key limitation of Prometheus: The memory required by a Prometheus server directly correlates with both the number of time series it scrapes and their associated scrape intervals. As the number of time series grows, vertical scaling of the Prometheus server becomes necessary.</p>
<p><strong>What’s new on native histograms:</strong> Prometheus 3 <a href="https://prometheus.io/docs/specs/native_histograms/?ref=blog.oodle.ai">native histograms</a> (stable since v3.8) store an entire histogram as a single series with exponential buckets, instead of one series per bucket label.</p>
<p>For histogram-heavy workloads this cuts active series counts, and therefore memory, substantially. Reducing cardinality this way is step zero of scaling Prometheus, before any architecture change.</p>
<h2 id="local-storage-and-data-retention-managing-data-on-disk">Local Storage and Data Retention: Managing Data on Disk</h2>
<p>As mentioned earlier, Prometheus stores data older than 2 hours (by default) to local on-disk storage. Prometheus will further run compaction on the data sitting on disk to merge them into larger blocks. The reliance on local on-disk storage impacts reliability and durability of Prometheus setup as noted in <a href="https://prometheus.io/docs/prometheus/latest/storage/?ref=blog.oodle.ai">Prometheus's Storage documentation</a>:</p>
<p>“Note that a limitation of local storage is that it is not clustered or replicated. Thus, it is not arbitrarily scalable or durable in the face of drive or node outages and should be managed like any other single node database.”</p>
<p>In addition to scalability limitation of on-disk storage, it also adds additional operational overhead:</p>
<ul>
<li>The local storage needs to be backed up with <a href="https://prometheus.io/docs/prometheus/latest/querying/api/?ref=blog.oodle.ai#snapshot">snapshots</a> to be able to recover from disk failures/operational errors.</li>
<li>It needs to be capacity planned periodically to account for increase in the number of time series/infrastructure growth.</li>
</ul>
<h2 id="query-performance-walking-the-performance-tightrope">Query Performance: Walking the Performance Tightrope</h2>
<p>PromQL allows slicing and dicing of data across various time ranges and labels. It allows complex queries which may require scanning through a lot of data. Given the single-process architecture of Prometheus, heavy queries may require excessive CPU or memory, causing OOM on the prometheus server or making it slow for other queries.</p>
<h2 id="beyond-single-node-strategies-for-scaling-prometheus">Beyond Single-Node: Strategies for Scaling Prometheus</h2>
<p>Having examined the key limitations of single-node Prometheus setups, let's explore various solutions for scaling Prometheus.</p>
<h2 id="divide-and-conquer-functional-sharding">Divide and Conquer: Functional Sharding</h2>
<p>To address Prometheus's limitations in handling large numbers of time series, one approach is to shard the time series data across multiple Prometheus servers. This can be achieved by dividing the data based on teams, clusters, service groups, or other logical boundaries. While this allows running multiple Prometheus servers with each handling its own subset of time series data, it introduces new challenges.</p>
<p>The main drawback is the loss of centralized query capability across all metrics. Users need to know which metrics are stored on which servers and query them accordingly. Additionally, this approach prevents joining data from different metrics at query time if they're stored on different servers.</p>
<h2 id="building-bridges-federation-and-global-views">Building Bridges: Federation and Global Views</h2>
<p>Prometheus's <a href="https://prometheus.io/docs/prometheus/latest/federation/?ref=blog.oodle.ai">Federation</a> feature enables one Prometheus server to scrape selected time series from another. This partially addresses the central visibility challenge by allowing a central Prometheus server to pull aggregated metrics from functionally sharded servers, providing an aggregate global view.</p>
<p>For example, you might set up multiple per-datacenter/region Prometheus servers that collect detailed data (instance-level drill-down), alongside global Prometheus servers that collect and store only aggregated data (job-level drill-down) from those local servers. This architecture provides both aggregate global views and detailed local views.</p>
<p>However, this approach still doesn't enable querying across all raw data.</p>
<h2 id="breaking-free-remote-storage-solutions">Breaking Free: Remote Storage Solutions</h2>
<p>Since neither Functional Sharding nor Federation fully solves central visibility or addresses data durability concerns, let's explore solutions that leverage remote storage.</p>
<h3 id="thanos-global-querying-with-object-storage">Thanos: Global Querying with Object Storage</h3>
<p>Thanos is an open-source project that extends Prometheus with long-term storage capabilities. Built on top of Prometheus, <a href="https://thanos.io/?ref=blog.oodle.ai">Thanos</a> enables object storage as a long-term storage solution for Prometheus data. It offers three key benefits:</p>
<ol>
<li>A <strong>global query view</strong> across all Prometheus servers</li>
<li><strong>Durable long-term storage</strong> through object storage</li>
<li>Deduplication of data from replicated Prometheus servers</li>
</ol>
<p>Thanos maintains the same data format on object storage as Prometheus uses for its on-disk storage. By enabling unlimited storage, Thanos allows for unlimited data retention.</p>
<p>While it solves the central visibility problem by allowing unified querying across multiple Prometheus servers and object storage, Thanos still depends on individual Prometheus servers for data collection and recent data serving. This means you'll still need to handle capacity planning and functional sharding of Prometheus servers, along with their associated operational overhead.</p>
<p>Thanos includes a query frontend that can cache results and split long time range queries into smaller ones. While this improves query performance against object storage, heavy queries (especially those spanning long time ranges) may still experience slower performance.</p>
<p><img src="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/2025/03/thanos.png" alt="Thanos architecture diagram: sidecars upload Prometheus blocks to object storage; a global query layer fans out across stores" loading="lazy"></p>
<p>Source: <a href="https://thanos.io/tip/thanos/quick-tutorial.md/?ref=blog.oodle.ai">https://thanos.io/tip/thanos/quick-tutorial.md/</a></p>
<h3 id="cortex-multi-tenant-scalability">Cortex: Multi-tenant Scalability</h3>
<p>Cortex provides a <a href="https://cortexmetrics.io/?ref=blog.oodle.ai">horizontally scalable, multi-tenant, long-term storage solution</a> for Prometheus. It accepts metrics data via the Prometheus remote-write protocol and stores them in object storage (similar to Thanos).</p>
<p>Unlike Thanos, <a href="https://cortexmetrics.io/?ref=blog.oodle.ai">Cortex</a> eliminates the need for Prometheus servers to serve recent data since all data is ingested directly into Cortex. It also adds multi-tenancy features and allows for configuring various quotas and limits per tenant. However, like Thanos, query performance for long-range queries may be slower due to the need to fetch data from object storage.</p>
<p><strong>Where Cortex stands in 2026:</strong> Cortex’s original maintainers <a href="https://thenewstack.io/the-great-grafana-mimir-and-cortex-split/?ref=blog.oodle.ai">moved to Grafana Mimir</a> in 2022, and most new development in this architecture now happens in Mimir. Cortex itself is still maintained: it remains a <a href="https://www.cncf.io/projects/cortex/?ref=blog.oodle.ai">CNCF incubating project</a> and shipped v1.21 in 2026.</p>
<p><strong><a href="https://aws.amazon.com/prometheus/faqs/?ref=blog.oodle.ai">Amazon Managed Service for Prometheus is powered by Cortex</a></strong>, per AWS’s own FAQ, so the architecture also runs at cloud-provider scale inside AWS. For a new self-hosted deployment the community’s default is Mimir, and <a href="https://grafana.com/docs/mimir/latest/set-up/migrate/migrate-from-cortex/?ref=blog.oodle.ai">Grafana publishes a migration guide</a> with a configuration converter for moving off Cortex.</p>
<p><img src="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/2025/03/cortex.png" alt="Cortex architecture diagram: metrics remote-written into a horizontally scalable multi-tenant backend over object storage" loading="lazy"></p>
<p>Source: <a href="https://cortexmetrics.io/docs/architecture/?ref=blog.oodle.ai">https://cortexmetrics.io/docs/architecture/</a></p>
<h3 id="grafana-mimir-enterprise-ready-cortex-evolution">Grafana Mimir: Enterprise-Ready Cortex Evolution</h3>
<p>Grafana Mimir has a similar architecture as that of Cortex and was started out as a fork of Cortex due to licensing issues. Grafana uses <a href="https://grafana.com/docs/mimir/latest/?ref=blog.oodle.ai">Mimir</a> architecture in their Grafana Cloud offering.</p>
<p>Since the fork, Mimir has become the actively developed branch of this architecture and the default choice for new centralized, multi-tenant metrics platforms. It keeps the Cortex model (horizontally scalable ingest over remote write, per-tenant limits, object storage for durability) and adds a split query engine that speeds up long-range queries.</p>
<p>Two caveats: it is a complex, many-component system to operate, and as a Grafana Labs project (AGPLv3), its roadmap follows Grafana Cloud's priorities.</p>
<h3 id="victoria-metrics-high-performance-at-scale">Victoria Metrics: High Performance at Scale</h3>
<p>Victoria Metrics offers a <a href="https://victoriametrics.com/?ref=blog.oodle.ai">high-performance, open-source time series database</a> and monitoring solution that can serve as Prometheus remote storage. It differs from Thanos and Cortex in two key ways:</p>
<ol>
<li>It uses disk storage rather than object storage</li>
<li>It employs its own optimized columnar file format instead of the Prometheus format</li>
</ol>
<p>Compared to Cortex, Victoria Metrics offers simpler setup and operation due to its streamlined architecture.</p>
<h2 id="making-the-right-choice-solution-comparison">Making the Right Choice: Solution Comparison</h2>
<p>We've explored how Prometheus excels as an easy-to-deploy metrics monitoring system at small scale, while examining its limitations at larger scales. We've also reviewed several open-source alternatives that address these limitations. Here's a summary of the solutions discussed:</p>
<table>
<thead>
<tr>
<th style="text-align:left">Dimension</th>
<th style="text-align:left">Prometheus<br>Federation</th>
<th style="text-align:left">Thanos</th>
<th style="text-align:left">Cortex</th>
<th style="text-align:left">Mimir</th>
<th style="text-align:left">Victoria<br>Metrics</th>
</tr>
</thead>
<tbody>
<tr>
<td style="text-align:left">Functional Sharding</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">Not Needed</td>
<td style="text-align:left">Not Needed</td>
<td style="text-align:left">Not Needed</td>
</tr>
<tr>
<td style="text-align:left">Global Query View</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
</tr>
<tr>
<td style="text-align:left">Data Durability</td>
<td style="text-align:left">🔴</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">🟠</td>
</tr>
<tr>
<td style="text-align:left">Query performance</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">✅</td>
</tr>
<tr>
<td style="text-align:left">Unlimited storage</td>
<td style="text-align:left">🔴</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">🔴</td>
</tr>
<tr>
<td style="text-align:left">Operational Overhead</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">🟠</td>
<td style="text-align:left">🟠</td>
</tr>
<tr>
<td style="text-align:left">High Cardinality</td>
<td style="text-align:left">🔴</td>
<td style="text-align:left">🔴</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
</tr>
<tr>
<td style="text-align:left">Actively developed (2026)</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">🟠 Maintained; new work in Mimir</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
</tr>
<tr>
<td style="text-align:left">Multi-tenancy / per-team limits</td>
<td style="text-align:left">🔴</td>
<td style="text-align:left">🟠 Basic</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">✅</td>
<td style="text-align:left">🟠 Cluster version</td>
</tr>
<tr>
<td style="text-align:left">Managed offering</td>
<td style="text-align:left">✅ AWS, Google Cloud, Azure</td>
<td style="text-align:left">✅ Exoscale (Managed Thanos)</td>
<td style="text-align:left">✅ AWS (Amazon Managed Prometheus)</td>
<td style="text-align:left">✅ Grafana Cloud</td>
<td style="text-align:left">✅ VictoriaMetrics Cloud</td>
</tr>
</tbody>
</table>
<p>Key: ✅ handled well · 🟠 partial or with caveats · 🔴 not addressed.</p>
<p><strong>On managed Prometheus:</strong> if you want Prometheus without operating it, <a href="https://www.oodle.ai/compare/amazon-managed-prometheus-alternative?ref=blog.oodle.ai">Amazon Managed Service for Prometheus</a>, <a href="https://docs.cloud.google.com/stackdriver/docs/managed-prometheus?ref=blog.oodle.ai">Google Cloud Managed Service for Prometheus</a>, <a href="https://learn.microsoft.com/en-us/azure/azure-monitor/metrics/prometheus-metrics-overview?ref=blog.oodle.ai">Azure Monitor managed service for Prometheus</a>, Grafana Cloud and <a href="https://www.oodle.ai/product/metrics?ref=blog.oodle.ai">Oodle</a> all offer it. None of them run stock Prometheus underneath: Amazon’s is powered by Cortex, Google’s is built on Monarch (the datastore behind Google’s own monitoring), Grafana Cloud runs Mimir, and Oodle runs its own object-storage engine. All speak PromQL and remote_write, so the choice is about operations, retention and cost rather than query language.</p>
<p><strong>Which should you choose?</strong></p>
<ul>
<li><strong>You have a working Prometheus fleet and mainly need retention and one query view:</strong> pick Thanos. It is the least disruptive option and extends what you run today.</li>
<li><strong>You are building a centralized, multi-team metrics platform from scratch:</strong> pick Mimir, or a managed equivalent. Not Cortex in 2026.</li>
<li><strong>You want the least operational burden and fast queries, with retention bounded by disk:</strong> pick VictoriaMetrics.</li>
<li><strong>You just hit your first memory wall:</strong> before adopting any of these, reduce cardinality by dropping unused labels and adopting native histograms. The cheapest scaling is fewer series.</li>
</ul>
<h2 id="what-prometheus-3-changes-for-scaling">What Prometheus 3 changes for scaling</h2>
<p>This guide was originally written for Prometheus 2.x. The 3.x line (v3.13 as of this update) changes the scaling picture in three ways:</p>
<ul>
<li><strong>Native histograms (stable in v3.8):</strong> <a href="https://prometheus.io/docs/specs/native_histograms/?ref=blog.oodle.ai">exponential-bucket histograms</a> stored as one series instead of a dozen bucket series. The single biggest lever against cardinality-driven memory growth.</li>
<li><strong>Native OTLP ingestion:</strong> Prometheus can <a href="https://prometheus.io/blog/2024/11/14/prometheus-3-0/?ref=blog.oodle.ai">receive OpenTelemetry metrics directly</a> at <code>/api/v1/otlp/v1/metrics</code> once the <code>--web.enable-otlp-receiver</code> flag is on, no collector required, which simplifies mixed OTel/Prometheus estates.</li>
<li><strong>Remote Write 2.0:</strong> <a href="https://prometheus.io/docs/specs/remote_write_spec_2_0/?ref=blog.oodle.ai">a leaner wire format</a> (string interning, native histogram support) that cuts bandwidth and CPU when shipping metrics to backends like Mimir or VictoriaMetrics.</li>
</ul>
<h2 id="frequently-asked-questions">Frequently asked questions</h2>
<h3 id="how-do-you-scale-prometheus">How do you scale Prometheus?</h3>
<p>In three stages: first reduce what you ingest (drop unused labels, use native histograms), then split load across servers with functional sharding and add federation for an aggregate global view, and finally move long-term storage to a horizontally scalable backend such as Thanos, Grafana Mimir, or VictoriaMetrics when you need durability and a single query view over raw data.</p>
<p>That is the standard path to running Prometheus at scale.</p>
<h3 id="what-is-prometheus-sharding">What is Prometheus sharding?</h3>
<p>Running multiple Prometheus servers that each scrape a subset of your targets, split by team, cluster, or service. Each server holds fewer time series, keeping memory manageable.</p>
<p>The cost is that no single server can answer queries across all of your metrics.</p>
<h3 id="how-many-time-series-can-a-single-prometheus-server-handle">How many time series can a single Prometheus server handle?</h3>
<p>There is no hard limit; memory is the constraint. Well-provisioned single servers commonly run low single-digit millions of active series.</p>
<p>Budget roughly a few kilobytes of RAM per active series, more with high churn, and watch query load, which competes for the same memory.</p>
<h3 id="what-are-the-four-types-of-metrics-in-prometheus">What are the four types of metrics in Prometheus?</h3>
<p>Counter (a value that only goes up, like requests served), gauge (a value that goes up and down, like memory in use), histogram (observations bucketed by size, like request latency), and summary (like a histogram, with quantiles computed client-side).</p>
<p>Prometheus 3 adds native histograms, a more efficient histogram representation.</p>
<h3 id="is-cortex-still-maintained-in-2026">Is Cortex still maintained in 2026?</h3>
<p>Yes. Cortex remains a CNCF incubating project with regular releases (v1.21 shipped in 2026), and Amazon Managed Service for Prometheus is built on it. Its original maintainers moved to Grafana Mimir in 2022 and most new development happens there, which is why new self-hosted deployments usually pick Mimir.</p>
<h3 id="does-prometheus-scale-horizontally">Does Prometheus scale horizontally?</h3>
<p>Not by default. Prometheus is a single-node system with no built-in clustering, so one server scales only vertically (more memory and CPU).</p>
<p>Horizontal scale comes from the patterns in this guide: functional sharding across servers, federation for an aggregate view, or remote-writing into a horizontally scalable backend such as Thanos, Mimir, or VictoriaMetrics.</p>
<h3 id="what-are-the-limitations-of-prometheus-federation">What are the limitations of Prometheus federation?</h3>
<p>Federation scrapes selected, usually pre-aggregated series from other Prometheus servers, so you still cannot query all raw data in one place.</p>
<p>The federating server adds its own scrape delay, becomes another component to size and operate, and its load grows with every series it pulls. It also does nothing for long-term durability, which is why remote-storage backends exist.</p>
<h3 id="which-monitoring-tools-extend-prometheus-for-enterprises">Which monitoring tools extend Prometheus for enterprises?</h3>
<p>The main open-source options are Thanos, Grafana Mimir, and VictoriaMetrics, which add durable storage, a global query view, and multi-tenancy on top of Prometheus.</p>
<p>Managed options include Amazon Managed Service for Prometheus, Grafana Cloud, and fully Prometheus-compatible platforms such as Oodle.</p>
<p><strong>Related reading on the Oodle blog</strong></p>
<ul>
<li>How these scaling trade-offs shaped Oodle's own architecture: <a href="https://blog.oodle.ai/building-a-high-performance-low-cost-metrics-observability-system/">Building a high-performance, low-cost metrics observability system</a></li>
<li>The query-performance side of the same problem: <a href="https://blog.oodle.ai/how-oodle-keeps-observability-fast-at-scale/">How Oodle keeps observability fast at scale</a></li>
<li>Finding the bottlenecks before you scale the monitoring around them: <a href="https://blog.oodle.ai/go-profiling-in-production/">Go profiling in production</a></li>
</ul>
<p>In the next part of this blog series, we will look at how Oodle solves the scalability issues of Prometheus in detail. Stay tuned!</p>
<p>In the meantime: every option above trades one kind of operational work for another. <a href="https://www.oodle.ai/product/metrics?ref=blog.oodle.ai">Oodle's serverless metrics engine</a> takes a different approach: an object-storage-native backend that is fully Prometheus-compatible (PromQL in, PromQL out), with no clusters to size and without the usual retention and cardinality trade-offs.</p>
</section>
</article>
</main>
<section class="footer-cta outer">
<div class="inner">
<h2 class="footer-cta-title">Sign up for more like this.</h2>
<a class="footer-cta-button" href="#/portal" data-portal>
<div class="footer-cta-input">Enter your email</div>
<span>Subscribe</span>
</a>
</div>
</section>
<aside class="read-more-wrap outer">
<div class="read-more inner">
<article class="post-card post">
<a class="post-card-image-link" href="/observability-for-ai-applications-using-oodle-and-openlit/">
<img class="post-card-image"
srcset="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w300/2026/08/Feature-Image-selection.png 300w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w600/2026/08/Feature-Image-selection.png 600w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w1000/2026/08/Feature-Image-selection.png 1000w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w2000/2026/08/Feature-Image-selection.png 2000w"
sizes="(max-width: 1000px) 400px, 800px"
src="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w600/2026/08/Feature-Image-selection.png"
alt="Observability for AI Applications Using Oodle and OpenLIT"
loading="lazy"
/>
</a>
<div class="post-card-content">
<a class="post-card-content-link" href="/observability-for-ai-applications-using-oodle-and-openlit/">
<header class="post-card-header">
<div class="post-card-tags">
</div>
<h2 class="post-card-title">
Observability for AI Applications Using Oodle and OpenLIT
</h2>
</header>
<div class="post-card-excerpt">Instrument LLM apps and agents with OpenLIT in one line and ship both metrics and traces to Oodle</div>
</a>
<footer class="post-card-meta">
<time class="post-card-meta-date" datetime="2026-08-21">21 Aug 2026</time>
<span class="post-card-meta-length">9 min read</span>
</footer>
</div>
</article>
<article class="post-card post">
<a class="post-card-image-link" href="/someones-hack-is-someone-elses-bug/">
<img class="post-card-image"
srcset="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w300/2026/08/bug-archaeologist.png 300w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w600/2026/08/bug-archaeologist.png 600w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w1000/2026/08/bug-archaeologist.png 1000w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w2000/2026/08/bug-archaeologist.png 2000w"
sizes="(max-width: 1000px) 400px, 800px"
src="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w600/2026/08/bug-archaeologist.png"
alt="Someone&#x27;s Hack is Someone Else&#x27;s Bug"
loading="lazy"
/>
</a>
<div class="post-card-content">
<a class="post-card-content-link" href="/someones-hack-is-someone-elses-bug/">
<header class="post-card-header">
<div class="post-card-tags">
</div>
<h2 class="post-card-title">
Someone&#x27;s Hack is Someone Else&#x27;s Bug
</h2>
</header>
<div class="post-card-excerpt">Story of how a 16 years old hack causes glitches even today</div>
</a>
<footer class="post-card-meta">
<time class="post-card-meta-date" datetime="2026-08-09">09 Aug 2026</time>
<span class="post-card-meta-length">3 min read</span>
</footer>
</div>
</article>
<article class="post-card post">
<a class="post-card-image-link" href="/ai-sensei-make-my-code-go-brrrrr/">
<img class="post-card-image"
srcset="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w300/2026/07/Gemini_Generated_Image_nkx6n5nkx6n5nkx6.png 300w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w600/2026/07/Gemini_Generated_Image_nkx6n5nkx6n5nkx6.png 600w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w1000/2026/07/Gemini_Generated_Image_nkx6n5nkx6n5nkx6.png 1000w,
https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w2000/2026/07/Gemini_Generated_Image_nkx6n5nkx6n5nkx6.png 2000w"
sizes="(max-width: 1000px) 400px, 800px"
src="https://storage.ghost.io/c/b2/48/b2485d36-0dcd-4e46-b7bd-418ab6e38e18/content/images/size/w600/2026/07/Gemini_Generated_Image_nkx6n5nkx6n5nkx6.png"
alt="AI Sensei, Make My Code Go Brrrrr!"
loading="lazy"
/>
</a>
<div class="post-card-content">
<a class="post-card-content-link" href="/ai-sensei-make-my-code-go-brrrrr/">
<header class="post-card-header">
<div class="post-card-tags">
</div>
<h2 class="post-card-title">
AI Sensei, Make My Code Go Brrrrr!
</h2>
</header>
<div class="post-card-excerpt">An engineer&#39;s account of how AI improved critical code, and made him more knowledgeable.</div>
</a>
<footer class="post-card-meta">
<time class="post-card-meta-date" datetime="2026-07-28">28 Jul 2026</time>
<span class="post-card-meta-length">5 min read</span>
</footer>
</div>
</article>
</div>
</aside>
</div>
<footer class="site-footer outer">
<div class="inner">
<section class="copyright"><a href="https://blog.oodle.ai">Oodle AI</a> &copy; 2026</section>
<div class="site-footer-center">
<div class="site-footer-social-links">
<a href="https://x.com/oodleai" target="_blank" rel="noopener" aria-label="X">
<svg class="icon" viewBox="0 0 24 24" fill="none" xmlns="http://www.w3.org/2000/svg">
<path d="M18.2439 2.25H21.5519L14.3249 10.51L22.8269 21.75H16.1699L10.9559 14.933L4.98991 21.75H1.67991L9.40991 12.915L1.25391 2.25H8.07991L12.7929 8.481L18.2439 2.25ZM17.0829 19.77H18.9159L7.08391 4.126H5.11691L17.0829 19.77Z" fill="currentColor"/>
</svg> </a>
</div>
<nav class="site-footer-nav">
<ul class="nav">
<li class="nav-sign-up"><a href="https://us1.oodle.ai/signup">Sign up</a></li>
</ul>
</nav>
</div>
<div class="gh-powered-by"><a href="https://ghost.org/" target="_blank" rel="noopener">Powered by Ghost</a></div>
</div>
</footer>
</div>
<div class="pswp" tabindex="-1" role="dialog" aria-hidden="true">
<div class="pswp__bg"></div>
<div class="pswp__scroll-wrap">
<div class="pswp__container">
<div class="pswp__item"></div>
<div class="pswp__item"></div>
<div class="pswp__item"></div>
</div>
<div class="pswp__ui pswp__ui--hidden">
<div class="pswp__top-bar">
<div class="pswp__counter"></div>
<button class="pswp__button pswp__button--close" title="Close (Esc)"></button>
<button class="pswp__button pswp__button--share" title="Share"></button>
<button class="pswp__button pswp__button--fs" title="Toggle fullscreen"></button>
<button class="pswp__button pswp__button--zoom" title="Zoom in/out"></button>
<div class="pswp__preloader">
<div class="pswp__preloader__icn">
<div class="pswp__preloader__cut">
<div class="pswp__preloader__donut"></div>
</div>
</div>
</div>
</div>
<div class="pswp__share-modal pswp__share-modal--hidden pswp__single-tap">
<div class="pswp__share-tooltip"></div>
</div>
<button class="pswp__button pswp__button--arrow--left" title="Previous (arrow left)"></button>
<button class="pswp__button pswp__button--arrow--right" title="Next (arrow right)"></button>
<div class="pswp__caption">
<div class="pswp__caption__center"></div>
</div>
</div>
</div>
</div>
<script src="https://blog.oodle.ai/assets/built/casper.js?v=YtPhrez3GuGLVDYs" defer></script>
<script src="https://cdnjs.cloudflare.com/ajax/libs/tocbot/4.32.2/tocbot.min.js" integrity="sha512-EovZhLMI2Z1FBS6Z37iDltDO4PkOZCmcamGhQHmcfporraL6LSfad582UieUBE3Boh165FVg3NKT1cz4g98V3A==" crossorigin="anonymous" referrerpolicy="no-referrer"></script>
<script>
tocbot.init({
tocSelector: '.gh-toc', // Selector for the ToC container
contentSelector: '.gh-content', // Selector for post content
headingSelector: 'h1, h2, h3, h4', // Select which headings to include
hasInnerContainers: true,
});
</script>
<script>document.addEventListener('DOMContentLoaded',function(){var contentElement=document.querySelector('.gh-content');if(contentElement){var asideElement=document.createElement('aside');asideElement.className='gh-sidebar';asideElement.innerHTML='<div class="gh-toc"></div>';var firstChildElement=contentElement.firstElementChild;if(firstChildElement){firstChildElement.insertAdjacentElement('beforebegin',asideElement)}else{contentElement.appendChild(asideElement)}tocbot.init({tocSelector:'.gh-toc',contentSelector:'.gh-content',headingSelector:'h1,h2,h3,h4',hasInnerContainers:true})}});</script>
</body>
</html>