1643 lines
461 KiB
HTML
1643 lines
461 KiB
HTML
<!DOCTYPE html><html lang="en-US" dir="ltr" data-scroll-behavior="smooth" class="inter_ed447d03-module__xJh6IW__variable inconsolata_5897f916-module__D6dAYa__variable basier_ff63df42-module__WL2v0q__variable font-inter scroll-smooth"><script>(function(){try{window.__cfbm=Object.freeze({"score":2,"verified":false,"category":"","bucket":"likely-automated"});window.sessionStorage.setItem('__cfbm', JSON.stringify(window.__cfbm));}catch(e){}})();</script>
|
||
<!-- Warm up third-party connections before the blocking consent defaults script. -->
|
||
<link rel="preconnect" href="https://cdn-prod.securiti.ai">
|
||
<link rel="dns-prefetch" href="//cdn-prod.securiti.ai">
|
||
<link rel="dns-prefetch" href="//discover.clickhouse.com">
|
||
|
||
<!-- Fetch banner CSS early, but do not apply it in <head>. -->
|
||
<link
|
||
rel="preload"
|
||
as="style"
|
||
href="https://discover.clickhouse.com/rs/238-FPC-317/images/securiti-cookie-banner-styles.css?version=0"
|
||
fetchpriority="low"
|
||
data-cookie-banner-css-preload
|
||
>
|
||
|
||
<!--
|
||
Must remain parser-blocking and before GTM.
|
||
This establishes Google Consent Mode defaults before GTM executes.
|
||
-->
|
||
<script
|
||
src="https://cdn-prod.securiti.ai/consent/cookie_banner/8555e54b-cd0b-45d7-9c1c-e9e088bf774a/e058d040-977c-4594-aa2c-84b844ce5cf0/google_consent_defaults.js"
|
||
data-cookie-banner-consent-defaults
|
||
></script>
|
||
|
||
<!--
|
||
The full Securiti banner UI can be deferred.
|
||
Consent defaults above are what GTM needs before it runs.
|
||
-->
|
||
<script
|
||
src="https://cdn-prod.securiti.ai/consent/cookie-consent-sdk-loader-strict-csp.js"
|
||
data-tenant-uuid="8555e54b-cd0b-45d7-9c1c-e9e088bf774a"
|
||
data-domain-uuid="e058d040-977c-4594-aa2c-84b844ce5cf0"
|
||
data-backend-url="https://app.securiti.ai"
|
||
data-skip-css="false"
|
||
data-strict-csp="true"
|
||
defer
|
||
data-cookie-banner-sdk
|
||
></script>
|
||
<head><meta charSet="utf-8"/><meta name="viewport" content="width=device-width, initial-scale=1"/><link rel="preload" href="/_next/static/immutable/media/83afe278b6a6bb3c.p.45535valc9rzk.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/basiersquare_bold_webfont-s.p.2u8jsokz8nhod.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/basiersquare_medium_webfont-s.p.26xz92gu9g5hg.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/basiersquare_semibold_webfont-s.p.1c0c__7s-vio-.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/c50f3c9c65fbdb75.p.18_orl2af6obj.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_breit_buch-s.p.444hxx5732eey.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_breit_dreiviertelfett-s.p.26mn1i0oa5he5.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_breit_extrafett-s.p.08ncyzqw0k4kl.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_breit_fett-s.p.0_t-j468o_cb9.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_buch-s.p.0ywwmkbterhyn.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_dreiviertelfett-s.p.03sa3pqhnqwzv.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_extrafett-s.p.32a0gi_a_59ac.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_fett-s.p.07es1xwfxq-cc.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_halbfett-s.p.106p6yn1xv5fc.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="preload" href="/_next/static/immutable/media/soehne_kraftig-s.p.2_vk7em30t5tz.woff2" as="font" crossorigin="" type="font/woff2"/><link rel="stylesheet" href="/_next/static/immutable/chunks/35oufqpfguwb5.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/immutable/chunks/0v_1jjtwkqx-_.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/immutable/chunks/023psz1z__79i.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/immutable/chunks/44fh1i3n6af1u.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/immutable/chunks/25ewdk32jdtfl.css" data-precedence="next"/><link rel="stylesheet" href="/_next/static/immutable/chunks/2osf4y-eei9lm.css" data-precedence="next"/><link rel="preload" as="script" fetchPriority="low" href="/_next/static/immutable/chunks/1f8tzf816i32j.js"/><script src="/_next/static/immutable/chunks/3q9kupw4qbhjb.js" async=""></script><script src="/_next/static/immutable/chunks/1krd7xlxk9v7b.js" async=""></script><script src="/_next/static/immutable/chunks/1eat_vqbhn7ur.js" async=""></script><script src="/_next/static/immutable/chunks/turbopack-3yzjdk7_0dx-4.js" async=""></script><script src="/_next/static/immutable/chunks/0esgme6slw2os.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/2pakapa8rw53j.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0f884239kz0zf.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0mugrkwyg2zkw.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/3_nn80b_suiek.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/420po0c6c-ox_.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/3biu5182d-lhi.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0xwnr3886mp47.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/1a9vq67yix_7x.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0tzu-v_27z1ur.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/2redqfc45k69c.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0bvw80ze-pqm7.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0o8_760sa9oqt.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/2uo20h-k3dnal.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/2qvo7870d8e3t.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/3cgheitv1dqkh.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/30m1qt8efl7us.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/3ero29qvrrxr-.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/200wl5-qkz6jp.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/44m5mm7tejp8y.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0k5y-yyer9qd2.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/11cralgf1s2pn.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0cohu1fgg5_3o.js" async="" crossorigin=""></script><script src="/_next/static/immutable/chunks/0a17ev_rxgc-h.js" async="" crossorigin=""></script><link rel="preload" href="https://clickhouse.com/gtmwksrxs8s/?id=GTM-WKSRXS8S" as="script"/><meta name="next-size-adjust" content=""/><link rel="ai-catalog" type="application/json" href="https://clickhouse.com/.well-known/ai-catalog.json"/><link rel="alternate" type="application/rss+xml" title="ClickHouse Blog" href="https://clickhouse.com/rss.xml"/><title>Can LLMs replace on call SREs today? | ClickHouse</title><meta name="description" content="We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found."/><meta name="application-name" content="ClickHouse"/><link rel="manifest" href="/manifest.webmanifest"/><meta name="referrer" content="origin-when-cross-origin"/><meta name="creator" content="ClickHouse"/><meta name="publisher" content="ClickHouse"/><meta name="last-modified" content="2026-03-03T12:38:31.061Z"/><link rel="canonical" href="https://clickhouse.com/blog/llm-observability-challenge"/><meta name="format-detection" content="telephone=no, date=no, address=no, email=no, url=no"/><meta property="og:title" content="Can LLMs replace on call SREs today? | ClickHouse"/><meta property="og:description" content="We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found."/><meta property="og:site_name" content="ClickHouse"/><meta property="og:locale" content="en_US"/><meta property="og:image" content="https://clickhouse.com/_next/image?url=%2Fuploads%2Fllm_observability_banner_22585788cf.png&w=1200&h=630&q=75"/><meta property="og:type" content="article"/><meta name="twitter:card" content="summary_large_image"/><meta name="twitter:site" content="@ClickHouseDB"/><meta name="twitter:creator" content="@ClickHouseDB"/><meta name="twitter:title" content="Can LLMs replace on call SREs today? | ClickHouse"/><meta name="twitter:description" content="We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found."/><meta name="twitter:image" content="https://clickhouse.com/_next/image?url=%2Fuploads%2Fllm_observability_banner_22585788cf.png&w=1200&h=630&q=75"/><meta name="twitter:image:alt" content="Can LLMs replace on call SREs today?"/><link rel="icon" href="/favicon.ico?favicon.1-9m4wq8d-w30.ico" sizes="48x48" type="image/x-icon"/><link rel="icon" href="/icon0.svg?icon0.14ll1zwu0qk95.svg" sizes="any" type="image/svg+xml"/><link rel="icon" href="/icon1.png?icon1.3b4swr1c2xvk2.png" sizes="96x96" type="image/png"/><link rel="apple-touch-icon" href="/apple-icon.png?apple-icon.18ftzencg_4sl.png" sizes="180x180" type="image/png"/><script src="/_next/static/immutable/chunks/0c0hxoamwjsbw.js" noModule=""></script></head><body class="font-inter antialiased"><div hidden=""><!--$--><!--/$--></div><!--$!--><template data-dgst="BAILOUT_TO_CLIENT_SIDE_RENDERING"></template><!--/$--><a href="#main" class="focus:ring-primary sr-only focus:not-sr-only focus:absolute focus:start-4 focus:top-4 focus:z-10000 focus:rounded focus:bg-neutral-900 focus:px-4 focus:py-2 focus:text-white focus:ring-2">Skip to content</a><div aria-hidden="true" class="fixed start-0 end-0 top-0 overflow-hidden bg-black pointer-events-none hidden" style="height:0;z-index:2147483646"><button type="button" title="Collapse the terminal (~)" class="absolute end-4 bottom-5 z-10 cursor-pointer rounded-md bg-neutral-800/70 p-1 text-neutral-400 transition-colors hover:bg-neutral-700 hover:text-white"><span class="sr-only">Collapse the terminal</span><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-up h-4 w-4" aria-hidden="true"><path d="m18 15-6-6-6 6"></path></svg></button><div class="to black from-neutral-725 absolute start-0 end-0 bottom-0 h-2 cursor-row-resize border-b border-neutral-700 bg-linear-to-t select-none"></div></div><header data-galaxy-component="header" class="sticky top-0 z-9999 w-full border-b border-white/5 backdrop-blur transition-colors bg-neutral-900/80"><div class="no-wrap relative container flex items-center py-4"><div class="relative "><a data-galaxy-event="logo" data-galaxy-click-event="topNav.logo.select" class="me-auto" aria-label="Go to homepage" href="/"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 135 40" width="135" height="40" fill="currentColor" role="img" aria-label="ClickHouse" class="text-white"><rect width="2.25" height="20.25" x="2.71" y="9.88" rx="0.24"></rect><rect width="2.25" height="20.25" x="7.21" y="9.88" rx="0.24"></rect><rect width="2.25" height="20.25" x="11.71" y="9.88" rx="0.24"></rect><rect width="2.25" height="20.25" x="16.21" y="9.88" rx="0.24"></rect><rect width="2.25" height="4.5" x="20.71" y="17.75" rx="0.24"></rect><path d="M40.03 15.14q-.95 0-1.7.34-.76.33-1.3.98-.52.64-.81 1.56-.27.91-.27 2.07a7 7 0 0 0 .45 2.63q.45 1.1 1.35 1.7t2.27.59q.83 0 1.58-.15.78-.15 1.57-.41v1.67q-.75.3-1.55.42-.8.15-1.84.14-1.96 0-3.27-.81a5 5 0 0 1-1.95-2.3q-.65-1.5-.65-3.5 0-1.46.4-2.66.42-1.23 1.19-2.1a5 5 0 0 1 1.9-1.36 7 7 0 0 1 2.65-.48 8.5 8.5 0 0 1 3.6.79l-.72 1.62q-.62-.3-1.37-.5a5 5 0 0 0-1.53-.24m7.6 11.36h-1.91V12.82h1.9zm4.9-9.7v9.7h-1.9v-9.7zm-.94-3.7q.44.01.76.26t.32.85q0 .57-.32.84a1.2 1.2 0 0 1-.76.25q-.45 0-.79-.25-.3-.27-.3-.84 0-.6.3-.85.33-.25.8-.25m7.84 13.58q-1.34 0-2.34-.52a3.6 3.6 0 0 1-1.56-1.62 6 6 0 0 1-.56-2.83q0-1.8.6-2.91.6-1.13 1.63-1.64a5 5 0 0 1 2.38-.54q.81 0 1.5.18.73.15 1.2.38l-.58 1.54q-.51-.2-1.08-.34-.56-.15-1.06-.14-.9 0-1.5.4-.57.37-.86 1.15-.27.76-.27 1.9 0 1.1.29 1.86.3.75.84 1.15.58.38 1.43.38a5 5 0 0 0 2.57-.65v1.66q-.53.3-1.13.45t-1.5.14m6.81-7.02q0 .38-.04.86-.01.5-.05.9h.05l.78-.97q.2-.26.4-.47l2.96-3.18h2.22l-3.9 4.16 4.15 5.54h-2.25l-3.2-4.34-1.12.94v3.4h-1.89V12.82h1.9zm18.4 6.84H82.7v-5.87h-6.13v5.87h-1.95V13.65h1.95v5.33h6.13v-5.33h1.95zm11.78-4.86q0 1.2-.32 2.14a5 5 0 0 1-.92 1.59q-.6.65-1.44.99a5.3 5.3 0 0 1-3.71 0 4.1 4.1 0 0 1-2.38-2.58 6 6 0 0 1-.34-2.16q0-1.6.54-2.72.56-1.1 1.59-1.69 1.04-.6 2.44-.6 1.34 0 2.34.6 1.03.58 1.6 1.7.6 1.11.6 2.73m-7.15 0q0 1.08.27 1.87.28.78.85 1.19t1.48.41q.9 0 1.47-.41.58-.42.85-1.19.27-.8.27-1.87 0-1.11-.29-1.87-.27-.76-.85-1.15a2.4 2.4 0 0 0-1.47-.42q-1.36 0-1.96.9-.62.9-.62 2.54m17.92-4.84v9.7h-1.53l-.27-1.28h-.09q-.3.5-.8.83-.47.33-1.05.47-.58.16-1.2.16-1.13 0-1.92-.36a2.6 2.6 0 0 1-1.18-1.15 4.5 4.5 0 0 1-.4-2.02V16.8h1.92v6.06q0 1.14.47 1.7.5.55 1.5.55t1.58-.4q.59-.39.81-1.14.25-.78.25-1.86V16.8zm9.43 6.96q0 .96-.47 1.6a3 3 0 0 1-1.35 1q-.88.32-2.12.32-1.03 0-1.77-.16-.71-.15-1.33-.43v-1.7q.65.31 1.5.56a6 6 0 0 0 1.65.23q1.08 0 1.55-.34.5-.34.49-.92 0-.31-.18-.57a2 2 0 0 0-.69-.54q-.48-.3-1.44-.65a15 15 0 0 1-1.56-.74 3 3 0 0 1-1-.88q-.34-.53-.34-1.33 0-1.26 1.01-1.93a5 5 0 0 1 2.7-.68 7 7 0 0 1 3.19.68l-.63 1.46a7 7 0 0 0-1.75-.56 4 4 0 0 0-.9-.09q-.86 0-1.31.27a.8.8 0 0 0-.45.76q0 .34.2.6.21.24.73.5.53.25 1.43.6t1.53.71q.64.36.97.88.34.53.34 1.33m6.1-7.14q1.27 0 2.19.54.92.52 1.4 1.51.5.99.5 2.34v1.04h-6.51q.03 1.5.77 2.29.75.8 2.11.8a8 8 0 0 0 1.66-.17q.74-.19 1.5-.52v1.58a7 7 0 0 1-3.23.65q-1.41 0-2.49-.56a4 4 0 0 1-1.69-1.65q-.6-1.12-.6-2.74 0-1.65.55-2.77a4.1 4.1 0 0 1 3.83-2.34m0 1.47q-1.04 0-1.66.67-.62.66-.72 1.89h4.57q0-.76-.24-1.33-.23-.59-.72-.9a2.2 2.2 0 0 0-1.24-.33"></path></svg></a><div class="absolute start-0 top-full z-50 w-56 origin-top-left pt-2 transition-all rtl:origin-top-right pointer-events-none scale-95 opacity-0"><div class="bg-neutral-750 rounded-lg border border-white/10 p-1.5 shadow-lg"><button type="button" class="cursor-pointer flex w-full rounded-lg px-3 py-2.5 text-start text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary">Copy logo as SVG</button><a href="/static/brand-assets/clickhouse-logo.zip" download="clickhouse-logo.zip" class="cursor-pointer flex w-full rounded-lg px-3 py-2.5 text-start text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary">Download full logo</a><a download="clickhouse-logomark.zip" class="cursor-pointer flex w-full rounded-lg px-3 py-2.5 text-start text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary" href="/static/brand-assets/clickhouse-logomark.zip">Download logomark</a></div></div></div><div class="ms-auto flex flex-row flex-nowrap items-center gap-4 min-[897px]:hidden"><button type="button" class="cursor-pointer undefined"><span class="sr-only">Open search</span><span class="hover:text-primary flex aspect-square w-10 items-center justify-center rounded-lg transition-colors hover:bg-white/5"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-search h-4 w-4" aria-hidden="true"><path d="m21 21-4.34-4.34"></path><circle cx="11" cy="11" r="8"></circle></svg></span></button><div class="relative "><button class="flex cursor-pointer items-center gap-2 "><span class="sr-only">Open region selector</span><svg xmlns="http://www.w3.org/2000/svg" width="15" height="15" fill="none" viewBox="0 0 15 15" class="shrink-0 grow-0"><path stroke="currentColor" stroke-linecap="round" stroke-linejoin="round" stroke-width="1.5" d="M13.75 7.5c0 3.45-2.8 6.25-6.25 6.25m6.25-6.25c0-3.45-2.8-6.25-6.25-6.25m6.25 6.25H1.25m6.25 6.25A6.25 6.25 0 0 1 1.25 7.5m6.25 6.25A9.56 9.56 0 0 0 10 7.5a9.56 9.56 0 0 0-2.5-6.25m0 12.5A9.56 9.56 0 0 1 5 7.5a9.56 9.56 0 0 1 2.5-6.25M1.25 7.5c0-3.45 2.8-6.25 6.25-6.25"></path></svg><svg xmlns="http://www.w3.org/2000/svg" width="6" height="10" fill="none" viewBox="0 0 6 10" class="flex w-2.5 shrink-0 grow-0 origin-center rotate-90 items-center justify-center transition-all opacity-50"><path stroke="currentColor" stroke-linecap="round" stroke-linejoin="round" stroke-width="1.5" d="m1 9 4-4-4-4"></path></svg></button><div class="absolute end-0 top-full origin-top-right pt-4 transition rtl:origin-top-left pointer-events-none scale-90 opacity-0"><ul class="bg-neutral-750 relative min-w-44 rounded-lg p-4 text-sm text-white shadow transition-all"><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/?lang=en">English</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/ja?lang=ja">Japanese</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/ko?lang=ko">Korean</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/zh?lang=zh">Chinese</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/fr?lang=fr">French</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/es?lang=es">Spanish</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/pt-BR?lang=pt-BR">Portuguese</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/ar?lang=ar">Arabic</a></li></ul></div></div><button class="bg-neutral-725 focus-visible:ring-primary inline-flex items-center justify-center rounded-md p-2 text-neutral-200 focus-visible:ring-2 focus-visible:outline-none" aria-expanded="false"><span class="sr-only">Open menu</span><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-menu size-4" aria-hidden="true"><path d="M4 5h16"></path><path d="M4 12h16"></path><path d="M4 19h16"></path></svg></button></div><div class="relative hidden flex-1 flex-row items-center transition-opacity min-[897px]:flex lg:ms-8"><nav data-galaxy-component="topNav" class="w-auto shrink-0"><div class="relative"><ul class="flex flex-row"><li><div class="group/nav-top-item relative"><button class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary px-4! group-hover/nav-top-item:bg-neutral-700/25 group-focus-within/nav-top-item:bg-neutral-700/25 group-hover/nav-top-item:text-primary group-focus-within/nav-top-item:text-primary"><span class="flex gap-x-1.5">Products<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-down size-4 self-center text-neutral-400 transition-colors group-focus-within/nav-top-item:text-current group-hover/nav-top-item:text-current" aria-hidden="true"><path d="m6 9 6 6 6-6"></path></svg></span></button><div class="pointer-events-none absolute -start-12 top-full w-max min-w-52 origin-top scale-90 pt-6 opacity-0 transition group-focus-within/nav-top-item:pointer-events-auto group-focus-within/nav-top-item:scale-100 group-focus-within/nav-top-item:opacity-100 group-hover/nav-top-item:pointer-events-auto group-hover/nav-top-item:scale-100 group-hover/nav-top-item:opacity-100"><div class="bg-neutral-750 rounded-lg"><ul class="grid auto-cols-auto grid-flow-col grid-rows-1"><li class="px-4 py-2 "><ul class="flex min-h-full flex-col"><li><div class="group/nav-sub-item"><div class="px-2 py-2 text-sm font-bold text-white">Products</div><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.cloudSelect" href="/cloud"><span class="flex items-center gap-x-4"><img alt="ClickHouse Cloud icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/clickhouse-cloud.2rsb11bgis-95.svg"/><span class="block leading-tight"><span class="block">ClickHouse Cloud</span><span class="text-xs text-nowrap text-slate-300 transition-colors">The best way to use ClickHouse.<br/>Available on AWS, GCP, and Azure.</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.byocSelect" href="/cloud/bring-your-own-cloud"><span class="flex items-center gap-x-4"><img alt="Bring Your Own Cloud icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/byoc.0h8f976g_o26-.svg"/><span class="block leading-tight"><span class="block">Bring Your Own Cloud</span><span class="text-xs text-nowrap text-slate-300 transition-colors">A fully managed ClickHouse service,<br/>deployed in your own AWS, GCP or Azure account.</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.postgresSelect" href="/cloud/postgres"><span class="flex items-center gap-x-4"><img alt="ClickHouse Managed Postgres icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/postgres.16wzg_24jsz5p.svg"/><span class="block leading-tight"><span class="block">ClickHouse Managed Postgres</span><span class="text-xs text-nowrap text-slate-300 transition-colors">Unified data stack for transactions<br/>and analytics.</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.managedClickstackSelect" href="/cloud/clickstack"><span class="flex items-center gap-x-4"><img alt="Managed ClickStack icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/managed-clickstack.1wt2__tf4rz6d.svg"/><span class="block leading-tight"><span class="block">Managed ClickStack</span><span class="text-xs text-nowrap text-slate-300 transition-colors">Managed observability with high-performance<br/>queries and long-term retention.</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a target="_blank" class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.langfuseCloudSelect" href="https://langfuse.com/?utm_source=clickhouse_topnav"><span class="flex items-center gap-x-4"><img alt="Langfuse Cloud icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/langfuse.09l1juwc1hfq6.svg"/><span class="block leading-tight"><span class="block">Langfuse Cloud<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-external-link ms-2 inline size-4 -translate-y-px align-middle" aria-hidden="true"><path d="M15 3h6v6"></path><path d="M10 14 21 3"></path><path d="M18 13v6a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2V8a2 2 0 0 1 2-2h6"></path></svg></span><span class="text-xs text-nowrap text-slate-300 transition-colors">LLM observability and evaluations<br/> for reliable AI applications and agents.</span></span></span></a></div></li></ul></li><li class="px-4 py-2 border-s border-neutral-700"><ul class="flex min-h-full flex-col"><li><div class="group/nav-sub-item"><div class="px-2 py-2 text-sm font-bold text-white">Open source</div><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.openSourceSelect" href="/clickhouse"><span class="flex items-center gap-x-4"><img alt="ClickHouse icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/clickhouse.0lnnr6dkfmisn.svg"/><span class="block leading-tight"><span class="block">ClickHouse</span><span class="text-xs text-nowrap text-slate-300 transition-colors">Fast open-source OLAP database for<br/>real-time analytics.</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.clickstackSelect" href="/clickstack"><span class="flex items-center gap-x-4"><img alt="ClickStack icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/clickstack.2evoafoj113lp.svg"/><span class="block leading-tight"><span class="block">ClickStack</span><span class="text-xs text-nowrap text-slate-300 transition-colors">Open-source observability stack for logs,<br/>metrics, traces, and session replays.</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.agenticDataStackSelect" href="/ai"><span class="flex items-center gap-x-4"><img alt="Agentic Data Stack icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/agentic-data-stack.295u5ndm0ogsy.svg"/><span class="block leading-tight"><span class="block">Agentic Data Stack</span><span class="text-xs text-nowrap text-slate-300 transition-colors">Build AI-powered applications<br/>with ClickHouse.</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.productMenu.chdbSelect" href="/chdb"><span class="flex items-center gap-x-4"><img alt="chDB icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/chdb.3adhkawsfckoc.svg"/><span class="block leading-tight"><span class="block">chDB</span><span class="text-xs text-nowrap text-slate-300 transition-colors">In-process SQL Engine powered by<br/>ClickHouse, with a Pandas-compatible API</span></span></span></a></div></li></ul></li></ul></div></div></div></li><li><div class="group/nav-top-item relative"><button class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary px-4! group-hover/nav-top-item:bg-neutral-700/25 group-focus-within/nav-top-item:bg-neutral-700/25 group-hover/nav-top-item:text-primary group-focus-within/nav-top-item:text-primary"><span class="flex gap-x-1.5">Solutions<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-down size-4 self-center text-neutral-400 transition-colors group-focus-within/nav-top-item:text-current group-hover/nav-top-item:text-current" aria-hidden="true"><path d="m6 9 6 6 6-6"></path></svg></span></button><div class="pointer-events-none absolute -start-12 top-full w-max min-w-52 origin-top scale-90 pt-6 opacity-0 transition group-focus-within/nav-top-item:pointer-events-auto group-focus-within/nav-top-item:scale-100 group-focus-within/nav-top-item:opacity-100 group-hover/nav-top-item:pointer-events-auto group-hover/nav-top-item:scale-100 group-hover/nav-top-item:opacity-100"><div class="bg-neutral-750 rounded-lg"><ul class="grid auto-cols-auto grid-flow-col grid-rows-1"><li class="px-4 py-2 "><ul class="flex min-h-full flex-col"><li><div class="group/nav-sub-item"><div class="px-2 py-2 text-sm font-bold text-white">Use cases</div><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.useCasesMenu.realTimeAnalyticsSelect" href="/use-cases/real-time-analytics"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Real-time analytics</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.useCasesMenu.observabilitySelect" href="/clickstack"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Observability</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.useCasesMenu.dataWarehousingSelect" href="/use-cases/data-warehousing"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Data warehousing</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.useCasesMenu.machineLearningSelect" href="/use-cases/machine-learning-and-data-science"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Machine learning and GenAI</span></span></span></a></div></li><li class="-mx-4 mt-auto border-t border-neutral-700 px-4 pt-2"><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.useCasesMenu.allUseCasesSelect" href="/use-cases"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">All use cases -></span></span></span></a></div></li></ul></li><li class="px-4 py-2 border-s border-neutral-700"><ul class="flex min-h-full flex-col"><li><div class="group/nav-sub-item"><div class="px-2 py-2 text-sm font-bold text-white">Industries</div><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.industriesMenu.financialServicesSelect" href="/industries/financial-services"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Financial services</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.industriesMenu.cybersecuritySelect" href="/industries/cybersecurity"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Cybersecurity</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.industriesMenu.gamingSelect" href="/industries/gaming"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Gaming and entertainment</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.industriesMenu.retailSelect" href="/industries/retail"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">E-commerce and retail</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.industriesMenu.automotiveSelect" href="/industries/automotive"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Automotive</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.industriesMenu.energySelect" href="/industries/energy"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Energy</span></span></span></a></div></li><li class="-mx-4 mt-auto border-t border-neutral-700 px-4 pt-2"><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.industriesMenu.allIndustriesSelect" href="/industries"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">All industries -></span></span></span></a></div></li></ul></li></ul></div></div></div></li><li><div class="group/nav-top-item relative"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary px-4! group-hover/nav-top-item:bg-neutral-700/25 group-focus-within/nav-top-item:bg-neutral-700/25 group-hover/nav-top-item:text-primary group-focus-within/nav-top-item:text-primary" data-galaxy-click-event="topNav.navItems.docsSelect" href="/docs"><span class="flex gap-x-1.5">Docs</span></a></div></li><li><div class="group/nav-top-item relative"><button class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary px-4! group-hover/nav-top-item:bg-neutral-700/25 group-focus-within/nav-top-item:bg-neutral-700/25 group-hover/nav-top-item:text-primary group-focus-within/nav-top-item:text-primary"><span class="flex gap-x-1.5">Resources<svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-down size-4 self-center text-neutral-400 transition-colors group-focus-within/nav-top-item:text-current group-hover/nav-top-item:text-current" aria-hidden="true"><path d="m6 9 6 6 6-6"></path></svg></span></button><div class="pointer-events-none absolute -start-12 top-full w-max min-w-52 origin-top scale-90 pt-6 opacity-0 transition group-focus-within/nav-top-item:pointer-events-auto group-focus-within/nav-top-item:scale-100 group-focus-within/nav-top-item:opacity-100 group-hover/nav-top-item:pointer-events-auto group-hover/nav-top-item:scale-100 group-hover/nav-top-item:opacity-100"><div class="bg-neutral-750 rounded-lg"><ul class="grid auto-cols-auto grid-flow-col grid-rows-1"><li class="px-4 py-2 "><ul class="flex min-h-full flex-col"><li><div class="group/nav-sub-item"><div class="px-2 py-2 text-sm font-bold text-white">Company resources</div><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.resourcesMenu.userStoriesSelect" href="/user-stories"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">User stories</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.resourcesMenu.blogSelect" href="/blog"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Blog</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.resourcesMenu.eventsSelect" href="/company/events"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Events</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.resourcesMenu.newsSelect" href="/company/news"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">News</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.resourcesMenu.learnAndCertificationSelect" href="/learn"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Learning and certification</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" href="/partners"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Partners</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.resourcesMenu.videosSelect" href="/videos"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Videos</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.resourcesMenu.demosSelect" href="/demos"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Demos</span></span></span></a></div></li></ul></li><li class="px-4 py-2 border-s border-neutral-700"><ul class="flex min-h-full flex-col"><li><div class="group/nav-sub-item"><div class="px-2 py-2 text-sm font-bold text-white">Comparisons</div><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.benchmarkHubSelect" href="/benchmarks"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Benchmark hub</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" href="/benchmarks/costbench"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">CostBench</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.bigQuerySelect" href="/comparison/bigquery"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">BigQuery</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.postgreSqlSelect" href="/comparison/postgresql"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">PostgreSQL</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.redshiftSelect" href="/comparison/redshift"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Redshift</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.snowflakeSelect" href="/comparison/snowflake"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Snowflake</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.elasticSelect" href="/comparison/elastic-for-observability"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Elastic Observability</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.splunkSelect" href="/comparison/splunk-for-observability"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Splunk</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.datadogSelect" href="/comparison/datadog-for-observability"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Datadog</span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.opensearchSelect" href="/comparison/opensearch-for-observability"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">OpenSearch<small class="-my-1 ms-2 inline-block rounded-sm bg-white/10 px-2 py-1 leading-none text-white">For observability</small></span></span></span></a></div></li><li><div class="group/nav-sub-item"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary w-full" data-galaxy-click-event="topNav.comparisonsMenu.databricksSelect" href="/clickhouse-for-databricks"><span class="flex items-center gap-x-4"><span class="block leading-tight"><span class="block">Databricks</span></span></span></a></div></li></ul></li></ul></div></div></div></li><li><div class="group/nav-top-item relative"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary px-4! group-hover/nav-top-item:bg-neutral-700/25 group-focus-within/nav-top-item:bg-neutral-700/25 group-hover/nav-top-item:text-primary group-focus-within/nav-top-item:text-primary" data-galaxy-click-event="topNav.navItems.pricingSelect" href="/pricing"><span class="flex gap-x-1.5">Pricing</span></a></div></li><li><div class="group/nav-top-item relative"><a class="inline-block rounded-lg text-neutral-100 text-start cursor-pointer px-2 py-2.5 text-sm font-medium transition-colors hover:bg-neutral-700/25 hover:text-primary px-4! group-hover/nav-top-item:bg-neutral-700/25 group-focus-within/nav-top-item:bg-neutral-700/25 group-hover/nav-top-item:text-primary group-focus-within/nav-top-item:text-primary" data-galaxy-click-event="topNav.navItems.contactUsSelect" href="/company/contact?loc=nav"><span class="flex gap-x-1.5">Contact us</span></a></div></li></ul></div></nav><div class="ms-auto flex flex-row flex-nowrap items-center gap-6"><div class="flex flex-row flex-nowrap items-center"><button type="button" class="cursor-pointer undefined"><span class="sr-only">Open search</span><span class="hover:text-primary flex aspect-square w-10 items-center justify-center rounded-lg transition-colors hover:bg-white/5"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-search h-4 w-4" aria-hidden="true"><path d="m21 21-4.34-4.34"></path><circle cx="11" cy="11" r="8"></circle></svg></span></button><button type="button" class="cursor-pointer " title="Toggle the web terminal (~)"><span class="sr-only">Toggle the web terminal</span><span class="hover:text-primary flex aspect-square w-10 items-center justify-center rounded-lg transition-colors hover:bg-white/5"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-square-terminal h-4 w-4" aria-hidden="true"><path d="m7 11 2-2-2-2"></path><path d="M11 13h4"></path><rect width="18" height="18" x="3" y="3" rx="2" ry="2"></rect></svg></span></button><div class="relative hidden p-3 md:block"><button class="flex cursor-pointer items-center gap-2 "><span class="sr-only">Open region selector</span><svg xmlns="http://www.w3.org/2000/svg" width="15" height="15" fill="none" viewBox="0 0 15 15" class="shrink-0 grow-0"><path stroke="currentColor" stroke-linecap="round" stroke-linejoin="round" stroke-width="1.5" d="M13.75 7.5c0 3.45-2.8 6.25-6.25 6.25m6.25-6.25c0-3.45-2.8-6.25-6.25-6.25m6.25 6.25H1.25m6.25 6.25A6.25 6.25 0 0 1 1.25 7.5m6.25 6.25A9.56 9.56 0 0 0 10 7.5a9.56 9.56 0 0 0-2.5-6.25m0 12.5A9.56 9.56 0 0 1 5 7.5a9.56 9.56 0 0 1 2.5-6.25M1.25 7.5c0-3.45 2.8-6.25 6.25-6.25"></path></svg><svg xmlns="http://www.w3.org/2000/svg" width="6" height="10" fill="none" viewBox="0 0 6 10" class="flex w-2.5 shrink-0 grow-0 origin-center rotate-90 items-center justify-center transition-all opacity-50"><path stroke="currentColor" stroke-linecap="round" stroke-linejoin="round" stroke-width="1.5" d="m1 9 4-4-4-4"></path></svg></button><div class="absolute end-0 top-full origin-top-right pt-4 transition rtl:origin-top-left pointer-events-none scale-90 opacity-0"><ul class="bg-neutral-750 relative min-w-44 rounded-lg p-4 text-sm text-white shadow transition-all"><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/?lang=en">English</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/ja?lang=ja">Japanese</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/ko?lang=ko">Korean</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/zh?lang=zh">Chinese</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/fr?lang=fr">French</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/es?lang=es">Spanish</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/pt-BR?lang=pt-BR">Portuguese</a></li><li><a class="hover:text-primary block w-full rounded-lg px-2 py-2.5 transition-colors hover:bg-neutral-700/25" href="/ar?lang=ar">Arabic</a></li></ul></div></div><a target="_blank" class="hover:text-primary hidden shrink-0 items-center gap-2 p-3 text-sm xl:inline-flex" data-galaxy-event="github" data-galaxy-click-event="topNav.navItems.githubSelect" href="https://github.com/ClickHouse/ClickHouse?utm_source=clickhouse&utm_medium=website&utm_campaign=website-nav"><img alt="GitHub Logo" loading="lazy" width="16" height="16" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/github.3ofxqpa2uzt_a.svg"/>49.8k</a></div><a class="hover:text-primary hidden shrink-0 items-center gap-2 text-sm lg:inline-flex" data-galaxy-event="signIn" data-galaxy-click-event="topNav.navItems.signInSelect" href="https://console.clickhouse.cloud/signIn">Sign in</a><a class="btn-primary" data-galaxy-event="signUp" data-galaxy-click-event="topNav.navItems.getStartedSelect" href="https://console.clickhouse.cloud/signUp?loc=nav-get-started"><span id="nav-bar-cta-button">Get Started</span></a></div></div></div></header><main id="main"><!--$!--><template data-dgst="BAILOUT_TO_CLIENT_SIDE_RENDERING"></template><!--/$--><script type="application/ld+json">[{"@context":"https://schema.org","@type":"BlogPosting","headline":"Can LLMs replace on call SREs today?","description":"We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found.","image":"/uploads/llm_observability_banner_22585788cf.png","publisher":{"@type":"Organization","name":"ClickHouse","url":"https://clickhouse.com/","logo":{"@type":"ImageObject","url":"https://clickhouse.com/_next/static/immutable/media/icon1.3b4swr1c2xvk2.png"}},"datePublished":"2025-08-13T16:18:03.962Z","dateModified":"2026-03-03T12:38:31.061Z","author":[{"@context":"https://schema.org","@type":"Person","name":"Lionel Palacin","url":"https://clickhouse.com/authors/lionel-palacin","@id":"https://clickhouse.com/authors/lionel-palacin#person","image":"/uploads/lio_headshot_singapore_7cc9852011.jpg","jobTitle":"Senior Product Marketing Engineer at ClickHouse"},{"@context":"https://schema.org","@type":"Person","name":"Al Brown","url":"https://clickhouse.com/authors/al-brown","@id":"https://clickhouse.com/authors/al-brown#person","image":"/uploads/al_brown_headshot_09ae0cbce6.jpg","jobTitle":"Product Marketing Engineer at ClickHouse"}]}]</script><div class="bg-neutral-725 w-full border-t border-b border-neutral-700 top-header sticky z-999"><div role="progressbar" aria-label="Reading progress" aria-valuemin="0" aria-valuemax="100" aria-valuenow="0" class="bg-primary h-0.5" style="width:0%"></div></div><div class="pointer-events-none fixed bottom-0 left-1/2 z-50 -translate-x-1/2 pb-4 transition duration-500 translate-y-full opacity-0"><span class="absolute -inset-x-20 -inset-y-10 rounded-full bg-black/40 blur-3xl"></span><button class="flex cursor-pointer gap-2 rounded-full border border-neutral-700/90 bg-neutral-950/90 px-3 py-1 text-sm text-neutral-200 shadow-lg backdrop-blur transition-colors hover:bg-neutral-900/90 pointer-events-none"><span class="-rotate-90">-></span>Scroll to top</button></div><div class="section-paddings-t relative border-b border-white/5 bg-neutral-900 pb-px"><div class="container flex flex-col items-start gap-8 lg:flex-row"><button type="button" class="group/backButton -mx-3 -my-1.5 me-8 inline-flex cursor-pointer items-center rounded px-3 py-1.5 text-base font-semibold whitespace-nowrap transition-colors hover:bg-white/5"><span class="me-2 w-4 transition-transform group-hover/backButton:-translate-x-1 rtl:group-hover/backButton:translate-x-1"><-</span>Back</button><div class="flex w-full flex-col gap-y-8 lg:grid lg:grid-cols-12 lg:gap-x-6"><div class="order-1 lg:order-0 lg:col-span-11 lg:mb-12 xl:col-span-9"><div class="flex flex-col gap-6 sm:-mt-0.5 sm:flex-row sm:items-center"><ul class="text-primary flex items-center gap-2 text-base font-semibold "><li><a class="hover:underline " href="/blog">Blog</a></li><li aria-hidden="true"><span>/</span></li><li><a class="hover:underline " href="/blog?category=engineering">Engineering</a></li></ul><div><div class="relative w-max text-sm"><span class="flex"><button type="button" disabled="" class="inline-flex cursor-pointer items-center gap-3 rounded-s border border-neutral-700 bg-neutral-900 px-3 py-1 text-start text-neutral-200 transition-colors hover:bg-white/10 disabled:pointer-events-none" title="Copy page as Markdown"><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-copy size-3 text-white" aria-hidden="true"><rect width="14" height="14" x="8" y="8" rx="2" ry="2"></rect><path d="M4 16c-1.1 0-2-.9-2-2V4c0-1.1.9-2 2-2h10c1.1 0 2 .9 2 2"></path></svg><span class="grid grid-cols-1 grid-rows-1"><span class="col-start-1 row-start-1 ">Copy page</span><span class="col-start-1 row-start-1 opacity-0">Copied!</span></span></button><button type="button" aria-haspopup="menu" aria-expanded="false" class="inline-flex cursor-pointer items-center rounded-e border border-s-0 border-neutral-700 bg-neutral-900 px-2 py-1 transition-colors hover:bg-white/10"><span class="sr-only">More actions</span><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-chevron-down size-4 transition-transform" aria-hidden="true"><path d="m6 9 6 6 6-6"></path></svg></button></span><ul role="menu" data-galaxy-component="aiActions" class="bg-neutral-750 absolute start-0 top-full z-10 w-max translate-y-2 space-y-1 rounded-lg border border-neutral-700 p-1 shadow-xl transition-opacity pointer-events-none opacity-0"><li role="none"><button type="button" role="menuitem" disabled="" class="flex w-full cursor-pointer items-center gap-3 rounded-lg px-3 py-1 transition-colors hover:bg-neutral-700/25 disabled:pointer-events-none disabled:opacity-50"><span><img alt="View as Markdown" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" class="aspect-square size-6 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/icon-markdown.2-p3ljls89_ww.svg"/></span><span class="flex flex-col text-start"><strong class="inline-flex items-center gap-2">View as Markdown<!-- --> </strong><small class="text-neutral-200">Open this page in Markdown</small></span></button></li><li role="none"><button type="button" role="menuitem" disabled="" class="flex w-full cursor-pointer items-center gap-3 rounded-lg px-3 py-1 transition-colors hover:bg-neutral-700/25 disabled:pointer-events-none disabled:opacity-50"><span><img alt="Open in ChatGPT" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" class="aspect-square size-6 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/icon-chatgpt.0aw932qxrira1.svg"/></span><span class="flex flex-col text-start"><strong class="inline-flex items-center gap-2">Open in ChatGPT<!-- --> <svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-external-link size-3 text-neutral-200" aria-hidden="true"><path d="M15 3h6v6"></path><path d="M10 14 21 3"></path><path d="M18 13v6a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2V8a2 2 0 0 1 2-2h6"></path></svg></strong><small class="text-neutral-200">Ask questions about this page</small></span></button></li><li role="none"><button type="button" role="menuitem" disabled="" class="flex w-full cursor-pointer items-center gap-3 rounded-lg px-3 py-1 transition-colors hover:bg-neutral-700/25 disabled:pointer-events-none disabled:opacity-50"><span><img alt="Open in Claude" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" class="aspect-square size-6 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/icon-claude.42qslf9ajhpwk.svg"/></span><span class="flex flex-col text-start"><strong class="inline-flex items-center gap-2">Open in Claude<!-- --> <svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-external-link size-3 text-neutral-200" aria-hidden="true"><path d="M15 3h6v6"></path><path d="M10 14 21 3"></path><path d="M18 13v6a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2V8a2 2 0 0 1 2-2h6"></path></svg></strong><small class="text-neutral-200">Ask questions about this page</small></span></button></li><li role="none"><button type="button" role="menuitem" disabled="" class="flex w-full cursor-pointer items-center gap-3 rounded-lg px-3 py-1 transition-colors hover:bg-neutral-700/25 disabled:pointer-events-none disabled:opacity-50"><span><img alt="Open in v0" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" class="aspect-square size-6 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/icon-v0.0mczhus58alob.svg"/></span><span class="flex flex-col text-start"><strong class="inline-flex items-center gap-2">Open in v0<!-- --> <svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-external-link size-3 text-neutral-200" aria-hidden="true"><path d="M15 3h6v6"></path><path d="M10 14 21 3"></path><path d="M18 13v6a2 2 0 0 1-2 2H5a2 2 0 0 1-2-2V8a2 2 0 0 1 2-2h6"></path></svg></strong><small class="text-neutral-200">Ask questions about this page</small></span></button></li></ul></div></div></div><h1 class="mt-6 mb-8 text-white">Can LLMs replace on call SREs today?</h1><div class="flex flex-row flex-wrap items-center gap-x-4 gap-y-2"><div class="shrink-0"><div class="flex -space-x-3"><span class="relative" style="z-index:2"><img alt="lio headshot singapore" loading="lazy" width="44" height="44" decoding="async" data-nimg="1" class="bg-neutral aspect-square h-auto! w-14! rounded-full border-4 border-neutral-900 object-cover" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Flio_headshot_singapore_7cc9852011.jpg&w=48&q=75 1x, /_next/image?url=%2Fuploads%2Flio_headshot_singapore_7cc9852011.jpg&w=96&q=75 2x" src="/_next/image?url=%2Fuploads%2Flio_headshot_singapore_7cc9852011.jpg&w=96&q=75"/></span><span class="relative" style="z-index:1"><img alt="Al Brown" loading="lazy" width="44" height="44" decoding="async" data-nimg="1" class="bg-neutral aspect-square h-auto! w-14! rounded-full border-4 border-neutral-900 object-cover" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fal_brown_headshot_09ae0cbce6.jpg&w=48&q=75 1x, /_next/image?url=%2Fuploads%2Fal_brown_headshot_09ae0cbce6.jpg&w=96&q=75 2x" src="/_next/image?url=%2Fuploads%2Fal_brown_headshot_09ae0cbce6.jpg&w=96&q=75"/></span></div></div><div><div class="text-base"><a class="hover:underline" data-state="closed" href="/authors/lionel-palacin">Lionel Palacin</a> and <a class="hover:underline" data-state="closed" href="/authors/al-brown">Al Brown</a></div><div class="text-sm text-neutral-300">Aug 14, 2025 · 21 minutes read</div></div></div></div><article class="order-3 lg:order-0 lg:col-span-11 xl:col-span-9"><div class="w-full space-y-6"><div class="rich-text rich-text-light "><style>
|
||
.llm-snippet {
|
||
background-color: #1e1e1e; /* dark gray background */
|
||
border: 1px solid #333; /* subtle border */
|
||
border-radius: 6px; /* rounded corners */
|
||
padding: 1rem 1.25rem; /* comfortable spacing */
|
||
font-size: 0.8rem;
|
||
line-height: 1.5;
|
||
color: #d4d4d4; /* light gray text */
|
||
overflow-x: auto; /* horizontal scroll if needed */
|
||
white-space: pre-wrap; /* preserve formatting but wrap text */
|
||
}
|
||
|
||
.rich_content details.llm p {
|
||
margin-bottom: 0.5rem;
|
||
margin-top: 0.5rem;
|
||
}
|
||
|
||
h4 {
|
||
font-size: 18px
|
||
}
|
||
|
||
h5 {
|
||
font-size: 16px
|
||
}
|
||
</style>
|
||
<p>There's a growing belief that AI-powered observability will soon reduce or even replace the role of Site Reliability Engineers (SREs). That's a bold claim and, at ClickHouse, we were curious to see how close we actually are.</p>
|
||
<p>For that we picked one particular task SRE does, among <a class="" href="https://en.wikipedia.org/wiki/Site_reliability_engineering">many other things</a>, which is the Root cause analysis, and ran an experiment to see how good they are at doing this task on their own.</p>
|
||
<p>Keep on reading to learn how we set up and ran the experiment, and more importantly what we learnt from it.</p>
|
||
<blockquote>
|
||
<p><strong>TL;DR</strong> Autonomous RCA is not there yet. The promise of using LLMs to find production issues faster and at lower cost fell short in our evaluation, and even GPT-5 did not outperform the others.</p>
|
||
<p>Use LLMs to assist investigations, summarize findings, draft updates, and suggest next steps while engineers stay in control with a fast, searchable observability stack.</p>
|
||
</blockquote>
|
||
<h2 id="the-experiment-llm-to-handle-rca-with-naive-prompt"><span class="group/mdHeader">The experiment: LLM to handle RCA with naive prompt<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h2>
|
||
<p>The experiment is straightforward. We gave a model access to observability data from a live application and, with a naive prompt, asked it to identify the root cause of a user-reported anomaly.</p>
|
||
<h3 id="the-contenders"><span class="group/mdHeader">The contenders<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>To run the experiment, first we needed contenders, so we picked five models. Some are well-known, others are less common but promising:</p>
|
||
<ul>
|
||
<li>
|
||
<p><strong>Claude Sonnet 4</strong>: Claude is known for its structured reasoning and detailed responses. It tends to be good at walking through steps logically, which could help when tracing complex issues across systems.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>OpenAI GPT-o3</strong>: A more advanced model from OpenAI, built for speed and multimodal input. It balances performance with fast response times.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>OpenAI GPT-4.1</strong>: Still strong when it comes to general reasoning and language understanding, but slower and not quite as sharp as GPT-o3. A good baseline for comparison.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Gemini 2.5 Pro</strong>: Google’s latest Pro model, integrated with their ecosystem. It does well with multi-step reasoning and has shown strong performance on code and troubleshooting tasks.</p>
|
||
</li>
|
||
</ul>
|
||
<p>We didn’t test every model out there, first because it’s an impossible task, but rather the ones we had easy access to and that seemed viable for the task. The idea is not to crown a winner, but to see how they each handle the reality of real-world incident data.</p>
|
||
<h3 id="the-anomalies"><span class="group/mdHeader">The anomalies<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Then we need observability data from a live application that contains an issue. We chose to run the <a class="" href="https://opentelemetry.io/docs/demo/">OpenTelemetry demo application</a> to generate four datasets containing unique anomalies.</p>
|
||
<p>As you can see in the architecture diagram below, the OpenTelemetry demo application is fairly complex, contains a lot of different services, a frontend users can interact with and a load generator. In brief, it's a good representation of a real-world application. Also the demo application comes out of the box with pre-canned anomalies that can be turned on using <a class="" href="https://opentelemetry.io/docs/demo/feature-flags/">feature flags</a>.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="blog-llm-diagram-otel.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fblog_llm_diagram_otel_757f32e326.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fblog_llm_diagram_otel_757f32e326.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fblog_llm_diagram_otel_757f32e326.png&w=2048&q=75"/></span></p>
|
||
<p>Using that application, we created three new datasets, each containing a distinct anomaly and data covering a 1 hour period. For the fourth test, we used our <a class="" href="https://play-clickstack.clickhouse.com/">existing ClickStack public demo</a> dataset which covers a 48 hour period.</p>
|
||
<p>The table below summarizes the four datasets we have and for each of them what feature flags were used.</p>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<div class="w-full overflow-x-auto"><table class="w-full min-w-fit border-collapse text-start text-sm "><thead><tr class="group/table-row "><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Name</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Database</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Duration</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Feature flag</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Description</th></tr></thead><tbody><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Anomaly 1</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">otel_anomaly_1</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1h</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">paymentFailure</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Generate an error when calling the charge method. This affects only users with loyalty status = gold.</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Anomaly 2</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">otel_anomaly_2</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1h</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">recommendationCacheFailure</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Create a memory leak due to an exponentially growing cache. 1.4x growth, 50% of requests trigger growth.</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Anomaly 3</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">otel_anomaly_3</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1h</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">productCatalogFailure</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Generate an error for GetProduct requests with product ID: OLJCESPC7Z</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Demo anomaly</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">otel_v2</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">48h</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">paymentCacheLeak</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Generate a memory leak in the payment cache service that slows down the application once the cache is full.</td></tr></tbody></table></div>
|
||
<p>To capture the datasets, we used the following methodology. </p>
|
||
<p>First, deploy the OpenTelemetry demo application and <a class="" href="https://clickhouse.com/docs/use-cases/observability/clickstack/ingesting-data/overview">instrument it</a> using ClickStack in Kubernetes. This <a class="" href="https://github.com/ClickHouse/opentelemetry-demo">repository</a> contains instructions on how to do so.</p>
|
||
<p>Once the application runs and the telemetry data starts to flow into ClickHouse, we increase the load to 1000 users. After the number of users reaches the desired target, we turn the feature flag on and let it run for about 40 minutes. Finally, we turn off the feature flag and let it run for another 10 minutes. The whole dataset captured now contains about 1 hour's worth of data.</p>
|
||
<div class="bg-neutral-725 mb-9 rounded-lg"><div class="bg-neutral-725 relative rounded-lg p-4 font-mono"><pre data-theme="dark" data-lang="sql" class="scrollbar-branded overflow-x-auto text-sm tabular-nums"><code class="hljs block w-fit min-w-full"><span class="line inline-block w-full pe-4" data-line="1"><span class="hljs-keyword">SELECT</span></span>
|
||
<span class="line inline-block w-full pe-4" data-line="2"> <span class="hljs-built_in">min</span>(TimestampTime),</span>
|
||
<span class="line inline-block w-full pe-4" data-line="3"> <span class="hljs-built_in">max</span>(TimestampTime)</span>
|
||
<span class="line inline-block w-full pe-4" data-line="4"><span class="hljs-keyword">FROM</span> otel_anomaly_3.otel_logs</span></code></pre><button type="button" class="absolute end-3 top-3 ms-auto hidden size-6 cursor-pointer rounded bg-neutral-900/5 text-neutral-400 backdrop-blur-2xl hover:bg-neutral-900/50 hover:text-neutral-500 sm:inline-block" data-galaxy-event="commandCopy" data-state="closed"><span class="sr-only">Copy command</span><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-copy mx-auto size-3" aria-hidden="true"><rect width="14" height="14" x="8" y="8" rx="2" ry="2"></rect><path d="M4 16c-1.1 0-2-.9-2-2V4c0-1.1.9-2 2-2h10c1.1 0 2 .9 2 2"></path></svg></button></div></div>
|
||
<!-- -->
|
||
<div class="bg-neutral-725 mb-9 rounded-lg"><div class="bg-neutral-725 relative rounded-lg p-4 font-mono"><pre data-theme="dark" data-lang="plaintext" class="scrollbar-branded overflow-x-auto text-sm tabular-nums"><code class="hljs block w-fit min-w-full"><span class="line inline-block w-full pe-4" data-line="1">┌──min(TimestampTime)─┬──max(TimestampTime)─┐</span>
|
||
<span class="line inline-block w-full pe-4" data-line="2">│ 2025-07-22 08:25:40 │ 2025-07-22 09:36:01 │ </span>
|
||
<span class="line inline-block w-full pe-4" data-line="3">└─────────────────────┴─────────────────────┘</span></code></pre><button type="button" class="absolute end-3 top-3 ms-auto hidden size-6 cursor-pointer rounded bg-neutral-900/5 text-neutral-400 backdrop-blur-2xl hover:bg-neutral-900/50 hover:text-neutral-500 sm:inline-block" data-galaxy-event="commandCopy" data-state="closed"><span class="sr-only">Copy command</span><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-copy mx-auto size-3" aria-hidden="true"><rect width="14" height="14" x="8" y="8" rx="2" ry="2"></rect><path d="M4 16c-1.1 0-2-.9-2-2V4c0-1.1.9-2 2-2h10c1.1 0 2 .9 2 2"></path></svg></button></div></div>
|
||
<p>Now we have our datasets captured, let’s investigate the issues using first a manual investigation then test the different models to see which one shines.</p>
|
||
<h2 id="how-we-ran-the-experiment"><span class="group/mdHeader">How we ran the experiment<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h2>
|
||
<h3 id="manual-investigation"><span class="group/mdHeader">Manual investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Before we ask the LLMs to find the problem, we need to know the actual root cause ourselves. That means doing a proper manual investigation for each anomaly, just like any SRE would. This gives us a baseline: if a model gets it right, we'll know. If it makes a mistake, or goes in the wrong direction, we'll be able to step in and guide it. </p>
|
||
<p>For this part, we use <a class="" href="https://clickhouse.com/use-cases/observability">ClickStack</a>, our observability stack built on top of ClickHouse. </p>
|
||
<p>We go through each issue manually, confirm what's wrong, and document the path we took. That becomes the reference point for comparing what the models do.</p>
|
||
<h3 id="ai-powered-investigation"><span class="group/mdHeader">AI-powered investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Then it's the models' turn.</p>
|
||
<p>We run each LLM through the same scenario using <a class="" href="https://www.librechat.ai/">LibreChat</a> connected to a <a class="" href="https://clickhouse.com/docs/use-cases/AI/MCP/librechat">ClickHouse MCP server</a>. This lets the models query real observability data and try to figure things out on their own.</p>
|
||
<p>We <a class="" href="https://clickhouse.com/blog/llm-observability-clickstack-mcp">instrument LibreChat and the ClickHouse MCP Server to track tokens usage</a>. Since the telemetry data is stored in ClickHouse, we can run the following query to obtain the number of tokens used throughout an investigation.</p>
|
||
<div class="bg-neutral-725 mb-9 rounded-lg"><div class="bg-neutral-725 relative rounded-lg p-4 font-mono"><pre data-theme="dark" data-lang="sql" class="scrollbar-branded overflow-x-auto text-sm tabular-nums"><code class="hljs block w-fit min-w-full"><span class="line inline-block w-full pe-4" data-line="1"><span class="hljs-keyword">SELECT</span></span>
|
||
<span class="line inline-block w-full pe-4" data-line="2"> LogAttributes[<span class="hljs-string">'conversationId'</span>] <span class="hljs-keyword">AS</span> conversationId,</span>
|
||
<span class="line inline-block w-full pe-4" data-line="3"> <span class="hljs-built_in">sum</span>(toUInt32OrZero(LogAttributes[<span class="hljs-string">'completionTokens'</span>])) <span class="hljs-keyword">AS</span> completionTokens,</span>
|
||
<span class="line inline-block w-full pe-4" data-line="4"> <span class="hljs-built_in">sum</span>(toUInt32OrZero(LogAttributes[<span class="hljs-string">'tokenCount'</span>])) <span class="hljs-keyword">AS</span> tokenCount,</span>
|
||
<span class="line inline-block w-full pe-4" data-line="5"> <span class="hljs-built_in">sum</span>(toUInt32OrZero(LogAttributes[<span class="hljs-string">'promptTokens'</span>])) <span class="hljs-keyword">AS</span> promptTokens,</span>
|
||
<span class="line inline-block w-full pe-4" data-line="6"> anyIf(LogAttributes[<span class="hljs-string">'text'</span>], (LogAttributes[<span class="hljs-string">'text'</span>]) <span class="hljs-operator">!=</span> <span class="hljs-string">''</span>) <span class="hljs-keyword">AS</span> prompt,</span>
|
||
<span class="line inline-block w-full pe-4" data-line="7"> <span class="hljs-built_in">min</span>(<span class="hljs-type">Timestamp</span>) <span class="hljs-keyword">AS</span> start_time,</span>
|
||
<span class="line inline-block w-full pe-4" data-line="8"> <span class="hljs-built_in">max</span>(<span class="hljs-type">Timestamp</span>) <span class="hljs-keyword">AS</span> end_time</span>
|
||
<span class="line inline-block w-full pe-4" data-line="9"><span class="hljs-keyword">FROM</span> otel_logs</span>
|
||
<span class="line inline-block w-full pe-4" data-line="10"><span class="hljs-keyword">WHERE</span> conversationId <span class="hljs-operator">=</span> </span>
|
||
<span class="line inline-block w-full pe-4" data-line="11"><span class="hljs-keyword">GROUP</span> <span class="hljs-keyword">BY</span> conversationId</span></code></pre><button type="button" class="absolute end-3 top-3 ms-auto hidden size-6 cursor-pointer rounded bg-neutral-900/5 text-neutral-400 backdrop-blur-2xl hover:bg-neutral-900/50 hover:text-neutral-500 sm:inline-block" data-galaxy-event="commandCopy" data-state="closed"><span class="sr-only">Copy command</span><svg xmlns="http://www.w3.org/2000/svg" width="24" height="24" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" class="lucide lucide-copy mx-auto size-3" aria-hidden="true"><rect width="14" height="14" x="8" y="8" rx="2" ry="2"></rect><path d="M4 16c-1.1 0-2-.9-2-2V4c0-1.1.9-2 2-2h10c1.1 0 2 .9 2 2"></path></svg></button></div></div>
|
||
<!-- -->
|
||
<p>We start each test with the same naive prompt:</p>
|
||
<p><em>"You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?"</em></p>
|
||
<p>If the model gets it right after the first prompt, great. If not, we ask follow up questions either based on the response it provided, or if it is completely off track, we provide additional context to help it get to a resolution.</p>
|
||
<p>Then for each anomaly, we report on:</p>
|
||
<ul>
|
||
<li>What the model finds</li>
|
||
<li>How accurate it is</li>
|
||
<li>How much guidance it requires</li>
|
||
<li>How many tokens it uses getting there</li>
|
||
<li>How long it takes to run the investigation</li>
|
||
</ul>
|
||
<p>This gives us a sense of how efficient and reliable each model is when dropped into a real-world SRE task.</p>
|
||
<h2 id="experiment-walkthrough"><span class="group/mdHeader">Experiment walkthrough<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h2>
|
||
<p>In this section, we go through each anomaly and first document the manual investigation, then for each model we start with our simple prompt and document the model's finding. If the model is not successful at diagnosing the issue right away, we provide additional prompts to guide it. </p>
|
||
<h3 id="anomaly-1-payment-service-failure"><span class="group/mdHeader">Anomaly 1: Payment service failure<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Users are reporting issues during the checkout process, after filling in all the information related to the order, clicking on place order leads to an error. </p>
|
||
<h4 id="manual-investigation"><span class="group/mdHeader">Manual investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
<p>This is a simple issue to diagnose manually in ClickStack. We start by looking at one of the client sessions containing an error.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-1-manual-1.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_1_manual_1_8d6ff13aa5.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_1_manual_1_8d6ff13aa5.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_1_manual_1_8d6ff13aa5.png&w=2048&q=75"/></span></p>
|
||
<p>From that specific session, we look at the trace view, where we can see an error message on the payment service. The message indicates that there is an invalid token for the loyalty level gold, which causes the payment request to fail.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-1-manual-2.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_1_manual_2_ffa9a04df4.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_1_manual_2_ffa9a04df4.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_1_manual_2_ffa9a04df4.png&w=2048&q=75"/></span></p>
|
||
<p>We navigate to the traces explorer view, and filter the dashboard for the payment service.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-1-manual-3.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_1_manual_3_fe0cbdef27.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_1_manual_3_fe0cbdef27.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_1_manual_3_fe0cbdef27.png&w=2048&q=75"/></span></p>
|
||
<p>This validates our finding, <strong>the payment service has issues when processing requests from clients with a Gold loyalty status.</strong></p>
|
||
<h4 id="llm-investigation"><span class="group/mdHeader">LLM investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
<h5 id="claude-sonnet-4-model"><span class="group/mdHeader">Claude Sonnet 4 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The model managed to find the RCA after the first prompt.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>Claude Sonnet 4 anomaly 1 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">OpenTelemetry Analysis Report
|
||
|
||
Issues Identified
|
||
|
||
1. Primary Issue: Payment Service Gold Tier Failure
|
||
|
||
- What: Starting at 11:52:35 on July 22, 2025, all payment transactions for users with "gold" loyalty level began failing with "Invalid token" errors
|
||
- Impact: 50% payment failure rate (13,505 failed transactions out of 27,000 total payment attempts)
|
||
- Error Message: "Payment request failed. Invalid token. app.loyalty.level=gold"
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="openai-o3-model"><span class="group/mdHeader">OpenAI o3 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The model managed to find the root cause and document it after the first prompt.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>OpenAI o3 anomaly 1 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Evidence
|
||
1. Traces
|
||
– Payment service: 13,512 error spans, all named
|
||
"Error: Payment request failed. Invalid token. app.loyalty.level=gold"
|
||
– Checkout service: 27,024 error spans (same TraceId chain) with gRPC method oteledemo.PaymentService/Charge returning INTERNAL.
|
||
– Frontend / frontend-proxy: ~29,000 error spans ("Failed to place order").
|
||
2. Logs
|
||
– 27,010 "failed to charge card: ... Invalid token ... loyalty.level=gold" log records.
|
||
– Small number (~6,000) of ECONNREFUSED errors, but they are an order of magnitude lower than the “Invalid token” failures.
|
||
3. Metrics
|
||
– No CPU / memory anomalies; latency spikes coincide with spikes in payment errors.
|
||
– Error rate is almost 100% for PlaceOrder when loyalty.level = gold.
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="openai-gpt4-1-model"><span class="group/mdHeader">OpenAI GPT4.1 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>Initially, the model didn’t query the data in ClickHouse via the MCP server. After a follow-up prompt, it queried ClickHouse and analyzed the data but couldn’t identify the root cause. It focused on the trace’s parent spans, which were impacted by the issue but not the actual source. The model only found the root cause after we gave it direct guidance.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>OpenAI GPT4.1 anomaly 1 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">To help you identify the issue, root cause, and suggest solutions, I’ll need to analyze relevant OpenTelemetry data from your demo application. This typically includes traces, metrics, and logs. Here’s how we’ll approach this:
|
||
1. Issue Identification: I’ll look for error traces, high latencies, dropped requests, or anomalies.
|
||
2. Root Cause Analysis: I’ll analyze the traces to see where bottlenecks, errors, or resource issues might exist (e.g., slow database calls, timeouts, dependency errors).
|
||
3. Potential Solutions: Based on findings, I’ll suggest actionable solutions.
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You have access to the data, can you start analyzing?
|
||
</div></pre>
|
||
<p>LLM response 2 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Observed Issue
|
||
- High average span durations in several services, especially load-generator, frontend-proxy, and frontend.
|
||
- No explicit errors in logs or trace statuses.
|
||
- This points toward a potential performance/bottleneck issue rather than a code crash or failure.
|
||
</div></pre>
|
||
<p>Prompt 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Users have been reported issues during the payment process
|
||
</div></pre>
|
||
<p>LLM response 3 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Impact
|
||
- Users, particularly with gold loyalty level, cannot complete payment, resulting in user-facing failures and high error rates in backend logs and traces.
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="gemini-2-5-pro"><span class="group/mdHeader">Gemini 2.5 Pro<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>To begin, Gemini made several errors trying to use the MCP tools to list databases. After the initial prompt, Gemini went in the wrong direction and focused on latency numbers of a specific API that was not causing a problem. After prompting it to look in the direction of the payment process, Gemini correctly identified the gold status issue.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>Gemini 2.5 Pro anomaly 1 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">### Issue
|
||
|
||
The application is experiencing high latency, with the `frontend-proxy` service being the most affected. The P95 latency for this service is over 1 second, which is significantly impacting the user experience.
|
||
|
||
### Root Cause
|
||
|
||
The root cause of the high latency is long-running calls from the `frontend-proxy` service to the `flagd.evaluation.v1.Service/EventStream` endpoint of the `flagd` service. The trace analysis shows that these calls can take over 3 minutes to complete. This is because the `EventStream` is a streaming endpoint designed to provide real-time updates of feature flags. The `frontend-proxy` is making a request to this endpoint and holding the connection open, which is causing the long-running spans and, consequently, the high latency.
|
||
|
||
The `canceled: "true"` attribute on the `router flagservice egress` span in the `frontend-proxy` service suggests that the proxy is timing out and canceling the request. However, the `flagd` service does not seem to be handling this cancellation correctly and continues to process the request, keeping the stream open for an extended period.
|
||
|
||
### Solution
|
||
|
||
To resolve this issue, the following actions are recommended:
|
||
|
||
* **Use a Short-Lived RPC for Feature Flag Evaluation**: Instead of using the `EventStream` endpoint, the `frontend-proxy` should use a more appropriate, short-lived RPC for fetching feature flags. This will ensure that the connections are not held open for extended periods, reducing the latency of the `frontend-proxy` service.
|
||
* **Investigate the `flagd` Service**: The `flagd` service should be investigated to understand why it's not handling client-side cancellations correctly. The developers should ensure that the service gracefully handles canceled requests and closes the streams promptly.
|
||
* **Review the `frontend-proxy` Configuration**: The `frontend-proxy`'s configuration should be reviewed to ensure that it's correctly handling streaming responses and that the timeouts are set to appropriate values.
|
||
|
||
By implementing these solutions, you should be able to significantly reduce the latency of the `frontend-proxy` service and improve the overall performance and reliability of your application.
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Users have reported issues during the payment process
|
||
</div></pre>
|
||
<p>LLM response 2 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Based on my investigation, I've identified the root cause of the payment issues and have recommendations for a solution.
|
||
|
||
### Summary of Findings
|
||
|
||
* **No Explicit Errors:** My initial search for traces and logs with an "ERROR" status for the `payment` service came up empty. This indicates that the application is not correctly flagging these issues as errors.
|
||
* **Anomalous Span Names:** An analysis of the service's trace data revealed a peculiar pattern: spans being created with names that are actually error messages. For example:
|
||
* `Error: Payment request failed. Invalid token. app.loyalty.level=gold`
|
||
* `Error: The credit card (ending 1791) expired on 1/2025.`
|
||
* **Error Context:** By examining these "error spans," I found that the "Invalid token" error consistently occurs for users with the `app.loyalty.level` attribute set to `gold`.
|
||
|
||
### Root Cause
|
||
|
||
The user-reported payment issues are caused by a bug in the `payment` service that specifically affects customers with a "gold" loyalty status. When these users attempt to make a payment, the service fails with an "Invalid token" error.
|
||
|
||
Furthermore, the application is handling these errors incorrectly from an observability standpoint. Instead of setting the status of the parent span (e.g., the `charge` or `grpc.oteldemo.PaymentService/Charge` span) to `ERROR`, the developers have written code that creates a new, zero-duration span with the error message as its name. This practice hides the errors from standard monitoring dashboards and makes it difficult to track the error rate of the service.
|
||
|
||
### Recommended Solutions
|
||
|
||
1. **Fix the Payment Logic:**
|
||
* **Immediate Priority:** The development team must investigate and fix the bug in the `payment` service that causes token validation to fail for "gold" loyalty members. This is the primary cause of the user-facing issue.
|
||
|
||
2. **Correct the Instrumentation:**
|
||
* **Best Practice:** The practice of creating new spans to log errors should be stopped. Instead, developers should adhere to OpenTelemetry best practices:
|
||
* When an error occurs, the `StatusCode` of the existing span in which the error happened should be set to `ERROR`.
|
||
* The error details should be recorded as an `Event` (also known as a `Log`) within that span, including attributes for the exception type, message, and stack trace. This ensures that errors are properly tracked, correlated with traces, and can be used to trigger alerts.
|
||
|
||
By addressing both the underlying bug in the payment logic and the incorrect instrumentation, you will resolve the user-reported issues and significantly improve the observability and reliability of your application.
|
||
</div></pre>
|
||
</details>
|
||
<h4 id="llm-investigation-summary"><span class="group/mdHeader">LLM investigation summary<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<div class="w-full overflow-x-auto"><table class="w-full min-w-fit border-collapse text-start text-sm "><thead><tr class="group/table-row "><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Model</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Resolution</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e "># of MCP calls</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Duration</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Input tokens</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Output tokens</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Cost</th></tr></thead><tbody><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Claude 4 sonnet</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">15</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">2 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1028123</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">4487</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$3.15</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">OpenAI o3</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">15</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">2 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">57397</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">2845</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.17</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">OpenAI GPT4.1</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with minor guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">14</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">3 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">43479</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">2224</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.10</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Gemini 2.5 Pro</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with minor guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">12</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">3 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">313892</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">7451</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.90</td></tr></tbody></table></div>
|
||
<h3 id="anomaly-2-recommendation-cache-leak"><span class="group/mdHeader">Anomaly 2: Recommendation cache leak<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>An issue on the recommendation cache service has been introduced which causes the service CPU usage to spike.</p>
|
||
<h4 id="manual-investigation"><span class="group/mdHeader">Manual investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
<p>Let’s start the investigation from the logs, we can see an increase in error messages.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-2-manual-1.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_1_1fb91f6e85.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_2_manual_1_1fb91f6e85.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_1_1fb91f6e85.png&w=2048&q=75"/></span></p>
|
||
<p>Scrolling through them, it’s not immediately obvious what the root cause is. We can see multiple error with a message about connection issue: <code class="inline rounded-sm border border-neutral-700 bg-neutral-800 px-1 py-0.5 ">⨯ Error: 14 UNAVAILABLE: No connection established. Last error: connect ECONNREFUSED 34.118.225.39:8080 (2025-07-22T11:05:44.834Z)</code></p>
|
||
<p>Let’s filter the trace explorer view to select only the traces in error and exclude the load-generator service as it is creating a lot of noise.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-2-manual-3.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_3_51ae663ca8.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_2_manual_3_51ae663ca8.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_3_51ae663ca8.png&w=2048&q=75"/></span></p>
|
||
<p>We see errors related to the recommendation service. Let’s filter only traces from the recommendation service.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-2-manual-4.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_4_cbd330d836.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_2_manual_4_cbd330d836.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_4_cbd330d836.png&w=2048&q=75"/></span></p>
|
||
<p>It seems there is a drop in traces during the experiment. Let’s validate that with the request throughput for the recommendation service using the Services view.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-2-manual-5.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_5_005ea71024.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_2_manual_5_005ea71024.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_5_005ea71024.png&w=2048&q=75"/></span></p>
|
||
<p>Throughput for the recommendation service dropped, while latency increased. This suggests the service is still responding but much more slowly, likely causing bottlenecks that lead to timeouts or connectivity errors.</p>
|
||
<p>Let’s return to the Traces and see if we can identify any common patterns among the traces using the Event deltas.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-2-manual-6.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_6_1141a55289.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_2_manual_6_1141a55289.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_6_1141a55289.png&w=2048&q=75"/></span></p>
|
||
<p>On the Event deltas screen, filtered by the get_product_list function, we can use the outliers attribute to compare fast and slow requests.</p>
|
||
<p>In this case, the outlier requests (slow requests) tend to return more than 10 products and have the recommendation cache enabled.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-2-manual-7.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_7_d4ba087db1.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_2_manual_7_d4ba087db1.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_2_manual_7_d4ba087db1.png&w=2048&q=75"/></span></p>
|
||
<p>In requests where the cache is enabled, we observed that when the number of products is high and if the cache misses, the request duration increases.</p>
|
||
<p><strong>The issue is related to the recommendation cache and the volume of selected products.</strong></p>
|
||
<h4 id="llm-investigation"><span class="group/mdHeader">LLM investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
<h5 id="claude-sonnet-4-model"><span class="group/mdHeader">Claude Sonnet 4 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The model was able to localize the root cause to the recommendation service but didn’t initially identify the cache issue.</p>
|
||
<p>Its progress was limited. It correctly observed that the recommendation service was failing due to resource exhaustion and noted that the service had restarted 12 times—something we hadn’t uncovered during our manual analysis.</p>
|
||
<p>It also identified that the restarts were caused by hitting the CPU limit, not a memory problem. While that’s a helpful detail, it still missed the actual cache-related issue.</p>
|
||
<blockquote>
|
||
<p>This reflects a common pattern: the model tends to lock onto a single line of reasoning and doesn’t explore other possibilities.</p>
|
||
</blockquote>
|
||
<p>When we followed up by asking why the pod was using 100% CPU, the model finally pinpointed a problem with how the recommendation service cache was behaving.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>Claude Sonnet 4 anomaly 2 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">The Recommendation Service experienced a complete outage, causing cascading failures throughout the application:
|
||
1. Primary Issue:
|
||
The recommendation service became completely unavailable between 10:13–11:05 UTC
|
||
• Normal span count dropped from 14,000 per 5-min window to 300–400
|
||
• Service was essentially down for nearly an hour
|
||
2. Connection Failures:
|
||
Frontend services couldn’t connect to recommendation service endpoints:
|
||
• IP addresses being refused: 34.118.225.39:8080, 34.118.236.87:8080, 34.118.226.49:8080
|
||
• 295,538 error logs from frontend service with "ECONNREFUSED" errors
|
||
• 51,190 failed gRPC calls to RecommendationService/ListRecommendations
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Why does the RecommendationService/ListRecommendations fail?
|
||
</div></pre>
|
||
<p>LLM response 2 (snippet 1/2):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root Cause Analysis: Why RecommendationService/ListRecommendations Fails
|
||
|
||
Based on my deep analysis of the observability data, here’s exactly why the RecommendationService/ListRecommendations is failing:
|
||
|
||
The Core Problem: Resource Exhaustion Leading to Cascading Failures
|
||
|
||
Primary Root Cause:
|
||
The recommendation service is experiencing resource starvation in its Kubernetes environment, leading to repeated container restarts.
|
||
</div></pre>
|
||
|
||
<p>LLM response 2 (snippet 2/2):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Restart Pattern:
|
||
• 12 service restarts in just 1 hour (10:13–11:07)
|
||
• Restart frequency: Every 3–7 minutes
|
||
• Restart timestamps show clear crash–restart cycles
|
||
</div></pre>
|
||
<p>Prompt 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Why do you think the recommendation service restarted?
|
||
</div></pre>
|
||
<p>LLM response 3 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Why the Recommendation Service Restarted – The Complete Analysis
|
||
|
||
Based on my deep dive into the observability data, I can now provide the exact reason for the recommendation service restarts:
|
||
⸻
|
||
|
||
Root Cause: CPU Throttling Leading to Kubernetes Health Check Failures
|
||
|
||
The Primary Trigger:
|
||
The recommendation service is being killed by Kubernetes due to failed health checks, not by memory issues (OOMKilled).
|
||
⸻
|
||
|
||
Evidence-Based Analysis:
|
||
|
||
Restart Pattern Analysis:
|
||
• 12 restarts in 54 minutes (10:13:40 – 11:07:23)
|
||
• Restart intervals: 54s, 96s, 77s, 179s, 248s, 436s, 411s, 488s, 414s, 400s, 420s
|
||
• Pattern: Initial rapid restarts (54–96s), then gradually increasing intervals as system stabilizes
|
||
</div></pre>
|
||
<p>Prompt 4:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Why is the pod using 100% CPUs?
|
||
</div></pre>
|
||
<p>LLM response 4 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root Cause Analysis:
|
||
|
||
The Problem: Algorithmic Complexity Explosion
|
||
What’s happening inside the recommendation service:
|
||
1. Cache Warming Issue:
|
||
At 10:12, something changes in the request patterns.
|
||
2. Cold Cache Performance:
|
||
When cache misses happen, the recommendation algorithm becomes computationally expensive.
|
||
3. Product List Processing:
|
||
The get_product_list function is doing heavy computation.
|
||
4. Cascade Effect:
|
||
As processing time increases, requests queue up, making everything slower.
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="openai-o3-model"><span class="group/mdHeader">OpenAI o3 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The model identified that the recommendation service was unavailable but didn’t explain why. When we asked for more details, it didn’t provide any useful insight. We then rephrased the question slightly. This time, the model discovered that the pod had crashed, though it couldn’t determine the exact cause. It suggested either a memory issue or a feature flag—both in the right direction.</p>
|
||
<p>We pressed further on the memory angle, which led the model to identify the cache-enabled flag as the likely root cause. However, when proposing a fix, the model generated a solution that wasn’t valid. It suggested code that doesn’t exist in the application, likely because it has no access to the source code.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>OpenAI o3 anomaly 2 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root cause (most likely)
|
||
|
||
The Recommendation Service is down or unreachable from the cluster:
|
||
• Pod(s) may be CrashLooping or scaled to 0
|
||
• Service/Endpoint list may be empty or pointing to the wrong port
|
||
• Container may start but not bind to 0.0.0.0:8080 (misconfiguration, wrong PORT env-var)
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Why does the RecommendationService/ListRecommendations fail?
|
||
</div></pre>
|
||
<p>LLM response 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">In short:
|
||
ListRecommendations fails whenever traffic is routed to an orphan / wrong IP (34.118.225.39).
|
||
Clean up that bad endpoint and the errors disappear.
|
||
</div></pre>
|
||
|
||
<p>Prompt 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Can you check before the recommendation service became unavailable what could have caused to crash?
|
||
</div></pre>
|
||
<p>LLM response 3 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">What probably caused the crash
|
||
|
||
– Memory leak or a code path that loads a huge product list into RAM
|
||
(note the 424 MiB / 332 MiB jumps right after fetching the product catalogue)
|
||
– A recent feature-flag or config change that increases the amount of data the service keeps in memory on start-up
|
||
</div></pre>
|
||
<p>Prompt 4:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">"note the 424 MiB/332 MiB jumps right after fetching the product catalogue" This is interesting. Can you check why fetching the product catalogue increased the memory a lot?
|
||
</div></pre>
|
||
<p>LLM response 4 (snippet 1/2):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">10:12 – flag flips → cache enabled
|
||
• ResolveBoolean/flagd traces appear for key "recommendation.cache_enabled"
|
||
• Subsequent get_product_list spans show cache_enabled = true
|
||
• In the very first minute after the flip, the value of app.products.count jumps from 10 to 20,226,406 (≈ 2 × 10⁷)
|
||
• container.memory.usage spikes to 830 MiB and the pod is OOM-killed
|
||
</div></pre>
|
||
<p>LLM response 4 (snippet 2/2):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">What the numbers tell us
|
||
|
||
The code that builds the cached product list is appending the entire list to itself on every cache miss / cold start instead of replacing it:
|
||
new_list = cache.get() # [] on first run
|
||
new_list += remote_product_list # +10
|
||
cache.set(new_list)
|
||
|
||
After a restart, the cache is empty again, so the first call fetches 10 products and stores them.
|
||
|
||
But a second call (still within the same process) does:
|
||
new_list = cache.get() # 10 items
|
||
new_list += remote_product_list # +10 ⇒ 20
|
||
cache.set(new_list)
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="openai-gpt4-1"><span class="group/mdHeader">OpenAI GPT4.1<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The model struggles to identify the root cause on its own. It doesn’t query data through the MCP tool unless explicitly instructed to do so. By default, it focuses only on logs and needs additional prompts to include traces and metrics in its analysis.</p>
|
||
<p>Once it reviews all the data, it flags the flagd service as a bottleneck affecting multiple components. While this is partially accurate, our manual investigation showed it’s not the main issue.</p>
|
||
<p>We directed the model to look into the Recommendation service, but it only mentioned the cache problem after we explicitly pointed it out.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>OpenAI GPT4.1 anomaly 2 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Absolutely, I can help you analyze OpenTelemetry data to diagnose issues within your demo application.
|
||
However, I’ll need more specific information or access to the relevant datasets (for example, traces, logs, or metrics from your observability backend).
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You have access to the OTel data in clickhouse, please query it.
|
||
</div></pre>
|
||
<p>LLM response 2 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">What this means
|
||
• The backend services (like API, web, or data processors) are not showing trace-level errors in the trace data.
|
||
• The only error detected is from the load-generator, which is likely the synthetic traffic generator for testing/demo purposes.
|
||
This could be expected and not user-impacting.
|
||
</div></pre>
|
||
|
||
<p>Prompt 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You also have access to traces and metrics data to further continue the investigation
|
||
</div></pre>
|
||
<p>LLM response 3 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Issue Identified
|
||
|
||
A major performance bottleneck is occurring in the /flagd.evaluation.v1.Service/EventStream span across multiple critical services (recommendation, fraud-detection, ad, flagd).
|
||
This is causing very high latencies, likely resulting in a degraded or unusable application experience for users.
|
||
</div></pre>
|
||
<p>Prompt 4:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Yes, can you focus on the recommendation service?
|
||
</div></pre>
|
||
<p>LLM response 4:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Conclusion & Root Cause:
|
||
• The recommendation service is blocked waiting for a response from flagd’s EventStream, hitting timeouts.
|
||
This means either the flagd service is slow, unavailable, or network connectivity to it is broken.
|
||
• Because flagd (feature flag evaluation) is used within critical request paths,
|
||
when it fails or hangs, the entire recommendation chain suffers, severely impacting user experience.
|
||
</div></pre>
|
||
<p>Prompt 5:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Top durations for /oteldemo.RecommendationService/ListRecommendations are also very high (over 21–22 seconds), suggesting calls to downstream dependencies (like flagd) are causing severe slowness in client-facing endpoints. Why do you think recommendation service (ListRecommendations) is calling flagd?
|
||
</div></pre>
|
||
<p>LLM response 5:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Trace Duration Patterns:
|
||
• The /flagd.evaluation.v1.Service/EventStream spans within the recommendation service show extremely high durations
|
||
(all stuck at 600+ seconds, matching the gRPC deadline).
|
||
• The /oteledemo.RecommendationService/ListRecommendations spans also have high durations (~21–22 seconds),
|
||
though not as long as the flagd EventStream spans, but still anomalous for a user-facing endpoint.
|
||
</div></pre>
|
||
<p>Prompt 6:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You have access to tracing data, you can check the direct trace linking.
|
||
</div></pre>
|
||
<p>LLM response 6:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet"> • Based on the evidence, I cannot confirm with this trace dataset that ListRecommendations is directly calling flagd via EventStream in a way that is instrumented by distributed traces.
|
||
• The slowness seen in both operations remains highly correlated.
|
||
Architectural practice suggests such a dependency may exist, but it is not proven in the current tracing data.
|
||
</div></pre>
|
||
<p>Prompt 7:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">if you can't find a direct link. Can you see another reason the recommendation service is running slow?
|
||
</div></pre>
|
||
<p>LLM response 7:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Key Observations:
|
||
• Durations are very high (around 19–22 seconds).
|
||
• All these slow requests have StatusCode “Unset” and rpc.grpc.status_code = "0"
|
||
(which means the gRPC call technically succeeded from a protocol point of view).
|
||
• The span attributes are consistent—each request recommended 5 products, the peer IP was the same (10.32.10.87), and the user agent is grpc-node-js/1.12.6.
|
||
• No explicit errors or useful StatusMessage are present.
|
||
</div></pre>
|
||
<p>Prompt 8:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">We recently upgraded the function get_product_list to optimize the cache logic. Can you check this didn't create a regression?
|
||
</div></pre>
|
||
<p>LLM response 8:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Is this a regression?
|
||
• A p95 response time of 200–250 ms might be acceptable for some backends, but if this represents a rise compared to previous levels, it could indicate a performance regression from the cache optimization.
|
||
• If this is much higher than before the change, it could mean the cache is not being hit effectively, or new cache logic is causing delays (e.g., added serialization, locking, or slower fallback path).
|
||
|
||
⸻
|
||
|
||
Signs of regression:
|
||
• There’s a jump from 204 ms to 251 ms during the 10:00–11:00 interval, possibly coinciding with the upgrade/deployment window.
|
||
</div></pre>
|
||
<p>Prompt 9:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Looking at the data it looks like the request latency for get_product_list went up though.
|
||
</div></pre>
|
||
<p>LLM response 9:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Summary:
|
||
|
||
There is strong evidence from tracing data that the cache logic upgrade in get_product_list introduced increased latency, likely hurting overall recommendation service responsiveness.
|
||
Rolling back or tuning the new cache logic should be prioritized.
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="gemini-2-5-pro"><span class="group/mdHeader">Gemini 2.5 Pro<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>Gemini initially explored the available database tables, acknowledging that traces, logs and metrics were available. It explained that it would begin with traces, but performed its entire investigation without ever considering logs or metrics. Using traces, it became locked on to the wrong problem.</p>
|
||
<p>After prompting it to also consider logs, Gemini then started to head in the right direction, suggesting that there was a timeout in the recommendation service, but did not discover why. After prompting it to investigate why the service timed out, it once again reverted back to only using traces, from which it decided that the issue was DNS. There are DNS errors present in the data set, but these are unrelated to the service in question. We then prompted it yet again to use the logs, but it got no further.</p>
|
||
<p>Next, we prompted it to consider the metrics table, which had some success; it discovered that the pods were reaching 100% CPU usage and suggested this was the cause - but did not dig any deeper. Prompting it to look into why it reached 100%, Gemini looped around and found nothing.</p>
|
||
<blockquote>
|
||
<p>However, it then began to formulate an imaginary cause (for which it had no evidence), and began trying to prove its case.</p>
|
||
</blockquote>
|
||
<p>Gemini was convinced that the cause must be a “poinsoned” product ID, and it began hunting for anything it could use to back this up. It was unsuccessful, but at least it admitted it.</p>
|
||
<p>We then asked it to take a look at resource usage around the time of the CPU spike, to see if there were other resources in contention. Somehow, this caused Gemini to discard the 100% CPU usage as a problem, and revert to telling us that DNS was, indeed, the problem (it isn’t).</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>Gemini 2.5 Pro anomaly 2 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">This is a very interesting finding. The top 10 longest-running spans are all related to the `flagd.evaluation.v1.Service/EventStream`. The durations are extremely high, in the order of 600 seconds (10 minutes), which is a strong indicator of a timeout or a serious performance bottleneck.
|
||
|
||
The services affected are `recommendation`, `fraud-detection`, and `ad`, which suggests a widespread problem with a shared dependency. In this case, the dependency is the `flagd` service, which appears to be a feature flagging service.
|
||
|
||
Based on this, I have identified the following:
|
||
|
||
**Issue:**
|
||
|
||
Users are experiencing significant performance issues with the application. These issues are not presenting as explicit errors but as very slow responses, leading to a poor user experience.
|
||
|
||
**Root Cause:**
|
||
|
||
The root cause of the performance degradation is the `/flagd.evaluation.v1.Service/EventStream` endpoint. This service is used by multiple other services (`recommendation`, `fraud-detection`, and `ad`) to receive feature flag updates. The extremely long duration of these calls indicates that the client services are waiting for a response from the `flagd` service for up to 10 minutes, which is causing them to become unresponsive.
|
||
|
||
**Potential Solutions:**
|
||
|
||
To resolve this issue, I recommend the following actions:
|
||
|
||
* **Investigate the `flagd` service:** The `flagd` service should be the primary focus of the investigation. Check its logs and metrics to understand why the `EventStream` endpoint is not responding in a timely manner.
|
||
* **Review network policies:** It is possible that a network policy is preventing the client services from connecting to the `flagd` service, leading to a timeout. Review the network policies to ensure that there are no rules blocking this traffic.
|
||
* **Implement connection timeouts:** The client services that connect to the `flagd` service should have a much shorter connection timeout. This will prevent them from being stuck for long periods and will allow them to fail fast and recover more gracefully. A timeout of a few seconds would be a good starting point.
|
||
* **Add readiness and liveness probes:** If the `flagd` service is running in a containerized environment (like Kubernetes), adding readiness and liveness probes will help automatically detect and recover from situations where the service is not responding.
|
||
|
||
By implementing these recommendations, you should be able to resolve the performance issues and improve the user experience of your application.
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">I dont think this is the problem. have you considered using the logs table as well?
|
||
</div></pre>
|
||
<p>LLM response 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Excellent! I've found the smoking gun. The logs from the `frontend-proxy` service show a clear pattern of `504 UT response_timeout` errors when calling the `frontend` service's `/api/recommendations` endpoint. This is a strong indicator that the `frontend` service is the bottleneck. The `response_timeout` message in the logs confirms that the `frontend-proxy` is waiting too long for a response from the `frontend` service.
|
||
|
||
Now that I've narrowed down the problem to the `frontend` service, I will investigate its traces to understand what's causing the delay. I'll focus on the `/api/recommendations` endpoint and look for the longest-running spans to pinpoint the exact operation that's causing the bottleneck.
|
||
|
||
I will now query the `otel_traces` table for the top 10 longest-running spans in the `frontend` service with the `SpanName` of `/api/recommendations`.
|
||
|
||
...
|
||
|
||
This is unexpected. I see that there are no traces for the `/api/recommendations` span in the `frontend` service, even though the logs from the `frontend-proxy` clearly show that this endpoint is being called.
|
||
|
||
This could mean that the `frontend` service is not correctly configured to generate traces for this endpoint. This is a common problem in distributed tracing setups, where some services might not be properly instrumented.
|
||
|
||
Even though I can't see the traces for the `/api/recommendations` endpoint, the logs from the `frontend-proxy` provide enough evidence to conclude that the `frontend` service is the root cause of the performance issues. The `504 UT response_timeout` errors are a clear indication that the `frontend` service is not responding in a timely manner.
|
||
|
||
Based on this, I can provide the following updated analysis:
|
||
|
||
**Issue:**
|
||
|
||
Users are experiencing significant performance issues with the application. These issues are not presenting as explicit errors but as very slow responses, leading to a poor user experience.
|
||
|
||
**Root Cause:**
|
||
|
||
The root cause of the performance degradation is the `frontend` service. The `frontend-proxy` is timing out when calling the `/api/recommendations` endpoint on the `frontend` service, as evidenced by the `504 UT response_timeout` errors in the `frontend-proxy`'s logs.
|
||
|
||
**Potential Solutions:**
|
||
|
||
To resolve this issue, I recommend the following actions:
|
||
|
||
* **Investigate the `frontend` service:** The `frontend` service should be the primary focus of the investigation. Check its logs and metrics to understand why the `/api/recommendations` endpoint is not responding in a timely manner.
|
||
* **Instrument the `frontend` service for tracing:** To get a better understanding of the performance of the `frontend` service, it's crucial to instrument it to generate traces. This will allow you to see the full lifecycle of a request and identify the exact operations that are causing the delay.
|
||
* **Review the code for the `/api/recommendations` endpoint:** The code for the `/api/recommendations` endpoint should be reviewed to identify any potential performance bottlenecks. This could include inefficient database queries, slow external service calls, or other performance anti-patterns.
|
||
* **Add readiness and liveness probes:** If the `frontend` service is running in a containerized environment (like Kubernetes), adding readiness and liveness probes will help automatically detect and recover from situations where the service is not responding.
|
||
|
||
By implementing these recommendations, you should be able to resolve the performance issues and improve the user experience of your application. The most critical step is to instrument the `frontend` service so you can get a better understanding of its performance and identify the root cause of the timeouts.
|
||
</div></pre>
|
||
|
||
<p>Prompt 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Why does the recommendation service timeout?
|
||
</div></pre>
|
||
<p>LLM response 3 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">This is the exact question we need to answer. The trace data has provided a crystal-clear picture of what's happening. Let's break it down.
|
||
|
||
**Analysis of the Trace (`425b78d730282b74af1a880fa7758a3b`)**
|
||
|
||
1. **The Client's Perspective (`frontend` service):**
|
||
* The `frontend` service makes a call to the `recommendation` service (`grpc.oteldemo.RecommendationService/ListRecommendations`).
|
||
* This call takes **307 milliseconds** from the `frontend` service's point of view.
|
||
|
||
2. **The Server's Perspective (`recommendation` service):**
|
||
* The `recommendation` service receives the request, processes it, and sends a response.
|
||
* The actual processing time inside the `recommendation` service (`/oteldemo.RecommendationService/ListRecommendations` span) is only **5 milliseconds**.
|
||
|
||
**The Root Cause: Massive Network Overhead**
|
||
|
||
The difference between the client's measured time (307 ms) and the server's processing time (5 ms) is **302 milliseconds**. This massive delay is happening *outside* the recommendation service's code.
|
||
|
||
I found the same pattern for all other downstream calls from the `frontend` service. For example, calls to the `product-catalog` service take about **218 ms** from the client's side but are processed in less than **1 ms** on the server side.
|
||
|
||
This means there is a consistent, large overhead of **~200-300 milliseconds** for every single outgoing gRPC call made by the `frontend` service.
|
||
|
||
**Conclusion**
|
||
|
||
The **recommendation service is not the problem**. It's extremely fast. The problem lies within the `frontend` service's environment, which is adding a significant delay to every network call it makes.
|
||
|
||
This is the root cause of the entire issue. This fixed delay makes the `frontend` service incredibly slow, which in turn causes the `frontend-proxy` to time out, just as we saw in the logs.
|
||
|
||
**Likely Causes for the Overhead and Next Steps**
|
||
|
||
This type of fixed overhead on outgoing network calls is often caused by misconfigurations in the environment, such as:
|
||
|
||
* **DNS Resolution Issues:** The `frontend` service's container might be trying to resolve service names (like `recommendation.default.svc.cluster.local`) using an external DNS server that fails or times out, before falling back to the correct internal cluster DNS.
|
||
* **Service Mesh Misconfiguration:** If you are using a service mesh like Istio or Linkerd, a misconfigured sidecar proxy could be adding this latency to every call.
|
||
* **Network Policies:** A network policy could be slowing down connection establishment.
|
||
|
||
**Recommendation:**
|
||
|
||
Your next step should be to **investigate the runtime environment of the `frontend` service pod**. Specifically, check its DNS configuration (`/etc/resolv.conf`) and any service mesh or network policy configurations that apply to it. Resolving this environmental issue should eliminate the network call overhead and fix the application's performance problems.
|
||
</div></pre>
|
||
<p>Prompt 4:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Ok but what is causing the timeout?
|
||
</div></pre>
|
||
<p>LLM response 4:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You are asking the central question. I will connect all the pieces of evidence to give you a definitive answer.
|
||
|
||
The timeout is caused by a **"death by a thousand cuts"** scenario happening inside the `frontend` service. The root cause is the **~250ms of network overhead** that is added to *every single* outbound call it makes.
|
||
|
||
Here is the step-by-step breakdown of how this leads to a 15-second timeout for a single `/api/recommendations` request:
|
||
|
||
1. **A Request Arrives:** A user requests recommendations, hitting the `frontend-proxy`, which forwards the request to the `frontend` service's `/api/recommendations` endpoint. The `frontend-proxy` starts a 15-second timer.
|
||
|
||
2. **The `frontend` Service's Job:** To build the response, the `frontend` service needs to gather data from other services. Looking at the trace data, we know for this single request it makes **5 downstream calls**:
|
||
* 1 call to the `recommendation` service.
|
||
* 4 calls to the `product-catalog` service.
|
||
|
||
3. **The Hidden "Network Tax":** As we discovered, every one of these 5 calls is burdened with a ~250 millisecond delay before it even leaves the `frontend` service's environment. The downstream services are incredibly fast, but this overhead is unavoidable on every call.
|
||
|
||
4. **Calculating the Cumulative Delay:** These small delays add up. The total *self-inflicted* delay caused by the network overhead for just this one request is:
|
||
> 5 calls × ~250 ms/call = **~1.25 seconds**
|
||
|
||
This means that even if the other services responded instantly, the `/api/recommendations` endpoint would still take over a second to respond.
|
||
|
||
5. **Amplification Under Load:** A 1.25-second response time is slow, but it's not a 15-second timeout. The timeout happens when the application is under load. The `load-generator` service is constantly sending requests. Because every request is slowed down by this 1.25-second (or more) overhead, the `frontend` service becomes a bottleneck. Requests start to pile up in a queue, waiting to be processed. This queuing and resource contention (CPU, connection pools) dramatically increases the total response time, pushing it far beyond the initial 1.25 seconds.
|
||
|
||
**Conclusion:**
|
||
|
||
The `frontend-proxy` times out because the `frontend` service, burdened by the cumulative effect of its network overhead, cannot process requests fast enough under normal load. The response time for some requests eventually exceeds the 15-second limit, triggering the `504 UT response_timeout` error you saw in the logs.
|
||
|
||
The problem isn't one slow downstream call, but rather the compounding effect of a small, consistent environmental problem on many fast downstream calls. To fix the timeout, you must fix the source of the network overhead within the `frontend` service's environment
|
||
</div></pre>
|
||
<p>Prompt 5:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">youve only used traces again, maybe you should look at the logs and there will be more info?
|
||
</div></pre>
|
||
<p>LLM response 5:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You are right to insist on a thorough investigation, and I appreciate you pushing me to use all the available data. My apologies for the repeated empty results. The lack of logs for that specific trace is, in itself, a critical discovery.
|
||
|
||
Let's synthesize everything we've found from both the traces and the logs to build a complete picture.
|
||
|
||
### The Story So Far:
|
||
|
||
1. **The Symptom (from Logs):** We started with your suggestion to check the logs. We found clear evidence in the `frontend-proxy` logs of `504 UT response_timeout` errors. This tells us the `frontend-proxy` is giving up on the `frontend` service after waiting for 15 seconds. This is the user-facing problem.
|
||
|
||
2. **The Bottleneck (from Traces):** The trace data allowed us to follow a request that was causing one of these timeouts. It showed that the `frontend` service was taking an extremely long time to respond.
|
||
|
||
3. **The "Why" (from Traces):** By breaking down the trace, we discovered something crucial: the `frontend` service makes many calls to other services (like `recommendation` and `product-catalog`). While these downstream services are very fast (responding in ~5ms), there is a consistent, large overhead of about **~250 milliseconds** added to *every single one* of these outgoing calls.
|
||
|
||
4. **The Missing Piece (from Logs... or lack thereof):** When we tried to find logs from the `frontend` service for that exact slow trace (`425b78d730282b74af1a880fa7758a3b`), we found nothing. This is a major clue. It strongly suggests the problem is happening at a level below the application code itself, likely in the infrastructure or environment where the `frontend` service is running.
|
||
|
||
### Final Conclusion: The Root Cause of the Timeout
|
||
|
||
The `frontend-proxy` times out because the **`frontend` service is suffering from a severe "death by a thousand cuts"**. Here's the sequence of events:
|
||
|
||
1. A request for recommendations arrives at the `frontend` service.
|
||
2. To fulfill this request, the `frontend` service must call multiple downstream services (we saw 5 such calls in our example trace).
|
||
3. Each of these 5 calls is saddled with a **~250ms "network tax"** before it even leaves the `frontend` service's environment. This tax is the real root cause.
|
||
4. This adds up to a minimum self-inflicted delay of **1.25 seconds** (5 calls x 250ms) for what should be a very fast operation.
|
||
5. Under normal load, multiple requests hit the `frontend` service at once. Because each one is already slow, they pile up, waiting for the CPU and network resources to become available. This queuing effect amplifies the initial 1.25-second delay, pushing the total response time over the `frontend-proxy`'s 15-second limit.
|
||
|
||
**Therefore, the timeout is not caused by a slow recommendation service, but by the compounding effect of a networking issue in the `frontend` service's environment.**
|
||
|
||
### Recommendations
|
||
|
||
The evidence strongly points to an environmental issue, not an application code issue.
|
||
|
||
1. **Primary Action: Investigate the `frontend` Service's Pod Environment.** The source of the ~250ms overhead must be found. The most common culprits for this kind of fixed network delay are:
|
||
* **DNS Resolution:** Check the `/etc/resolv.conf` file in the `frontend` service's container. It might be trying to contact an external DNS server before using the correct internal cluster DNS.
|
||
* **Service Mesh Misconfiguration:** If you're using a service mesh like Istio or Linkerd, inspect the configuration of the sidecar proxy attached to the `frontend` pod.
|
||
* **Network Policies:** Review any `NetworkPolicy` resources in your cluster that might be inspecting or slowing down traffic from the `frontend` pod.
|
||
|
||
2. **Secondary Action: Improve Observability.** This investigation was made much harder because the `frontend` service wasn't logging correctly. To prevent this in the future:
|
||
* **Fix the Logging:** Ensure the `frontend` service logs are being generated and, critically, that they are correlated with traces by including the `TraceId` in every log message. This will make future debugging much faster.
|
||
</div></pre>
|
||
<p>Prompt 6:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">what about looking at resource usage of the recommendation service?
|
||
</div></pre>
|
||
<p>LLM response 6:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You were absolutely right to insist on checking the resource usage. My apologies for focusing too narrowly on the network latency. The metrics you asked for have revealed the true root cause of the problem.
|
||
|
||
Here are the results for the resource utilization of the `recommendation` service:
|
||
|
||
* **Memory Usage (`k8s.container.memory_limit_utilization`):** The average memory usage is **~9.4%** of the limit, with a maximum peak at a healthy **~69.1%**. Memory is not the issue.
|
||
|
||
* **CPU Usage (`k8s.container.cpu_limit_utilization`):** The average CPU usage is **~29.8%**, but the maximum value is **~100.2%**.
|
||
|
||
### The Real Root Cause: CPU Throttling
|
||
|
||
The fact that the `MaxValue` for CPU utilization is at 100% is the smoking gun. This means the `recommendation` service's container is hitting its CPU limit. When this happens, Kubernetes **CPU throttles** the container, preventing it from using more CPU. This throttling directly translates into processing delays.
|
||
|
||
Here’s the corrected sequence of events:
|
||
|
||
1. A request arrives at the `recommendation` service.
|
||
2. Under load, the service experiences a spike in CPU usage, hitting 100% of its allocated limit.
|
||
3. Kubernetes steps in and throttles the container, effectively slowing down its ability to process the request. This causes the service's response time to jump from a few milliseconds to several hundred milliseconds.
|
||
4. This delay is passed back to the `frontend` service.
|
||
5. The `frontend` service makes multiple calls to other services (including the now-throttled `recommendation` service). The delays from these throttled calls accumulate.
|
||
6. This cumulative delay makes the `frontend` service slow to respond to the `frontend-proxy`.
|
||
7. Finally, the `frontend-proxy`, waiting on the slow `frontend` service, gives up after 15 seconds, resulting in the `504 UT response_timeout` errors.
|
||
|
||
My previous conclusion about a "network tax" was incorrect. The delay seen in the traces was not from the `frontend` service's network, but from waiting on the `recommendation` service, which was intermittently being slowed down by CPU throttling. Thank you for pushing me to investigate further.
|
||
|
||
### Recommendations
|
||
|
||
The solution is to alleviate the CPU pressure on the `recommendation` service.
|
||
|
||
1. **Immediate Fix: Increase the CPU Limit.** The most straightforward solution is to increase the CPU limit for the `recommendation` service deployment. This will give it more headroom to handle spikes in load without being throttled.
|
||
|
||
2. **Long-Term Fix: Investigate CPU Spikes.** While increasing the limit will fix the immediate problem, it's important to understand *why* the CPU is spiking. The developers should investigate the `recommendation` service's code to identify any inefficient operations or queries that could be optimized.
|
||
|
||
3. **Proactive Monitoring: Set Up Alerts.** To prevent this from happening in the future, you should configure alerts to fire whenever a service's CPU utilization approaches its limit. This will allow you to address resource constraints before they impact users.
|
||
</div></pre>
|
||
<p>Prompt 7:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">can you narrow down why the CPU reaches 100%?
|
||
</div></pre>
|
||
<p>LLM response 7:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Of course. Let's dig deeper into the "why." You are right to ask for the next level of detail. We know the `recommendation` service is hitting its CPU limit, but *why* is its CPU usage so high in the first place?
|
||
|
||
The traces we have examined show that the primary function of the `recommendation` service is to:
|
||
1. Call the `product-catalog` service to get a list of products.
|
||
2. Process this list to generate recommendations.
|
||
|
||
My hypothesis is that the CPU spike is directly related to the **size of the product list** it has to process for every single request. A large list could make the recommendation logic itself very CPU-intensive.
|
||
|
||
To test this, I need to find out how many products are in the catalog. The `product-catalog` service is the source of truth for this. I will inspect the traces originating from this service to find any logs or attributes that indicate the number of products it's handling.
|
||
|
||
I will now query the `otel_traces` table to look at the attributes of the `ListProducts` span within the `product-catalog` service. This might tell us the size of the payload being returned.
|
||
|
||
...
|
||
This is a fantastic find, but `10` seems far too low to cause a CPU spike. This suggests the sample of one trace might not be representative. The number of products is likely dynamic and can be much larger.
|
||
|
||
My next step is to find out the **maximum number of products** the `product-casting-service` has ever returned. This will tell us the peak load that the `recommendation` service is expected to handle.
|
||
|
||
I will now query the `otel_traces` table again for the `ListProducts` span, but this time I will look for the maximum value of the `app.products.count` attribute across all traces. This should give us a much better idea of the potential scale of the problem.
|
||
...
|
||
This is a dead end. The trace attributes consistently show a small number of products, which contradicts the CPU saturation we're seeing. It means my hypothesis that the *number* of products is the problem is likely wrong.
|
||
|
||
You've been right to push me to look at different data sources. I've been switching between traces, metrics, and logs, but I haven't used them together effectively enough. Let's try a new approach that directly links the high CPU usage to a specific action.
|
||
|
||
My new hypothesis is: **There is a "rogue" product or a specific type of product in the catalog that triggers a highly inefficient, CPU-intensive code path in the `recommendation` service.**
|
||
|
||
Here’s how we can prove this:
|
||
1. Find a time window when the `recommendation` service's CPU usage was at its peak (100%).
|
||
2. During that exact time window, find out which products were being requested by the `frontend` service.
|
||
|
||
This will directly correlate the CPU spike with the product being processed at that moment.
|
||
|
||
First, I need to find the exact timestamp when the CPU utilization hit 100%. I will query the metrics table for this.
|
||
...
|
||
This is another dead end. It seems the `productIds` are not being logged in the `url.query` attribute in the `frontend-proxy` logs.
|
||
|
||
My apologies. I have been jumping between different data sources without a clear, systematic approach. It's time to take a step back and think logically, using all the clues we've gathered so far.
|
||
|
||
Here's what we know for certain:
|
||
|
||
1. **The Symptom:** The application times out. The logs confirm this with `504` errors from the `frontend-proxy`.
|
||
2. **The Bottleneck:** The timeouts are caused by the `frontend` service responding too slowly.
|
||
3. **The Root Cause:** The `frontend` is slow because it's waiting on the `recommendation` service.
|
||
4. **The Core Problem:** The `recommendation` service is slow because its CPU is being throttled at 100%.
|
||
|
||
The question remains: **What in the `recommendation` service's code is so inefficient that it consumes 100% of the CPU?**
|
||
|
||
The traces show that the `recommendation` service's main job is to process a list of products. We have also seen that the *number* of products is small, so we can rule out "processing a large list" as the problem. This leaves us with one logical conclusion:
|
||
|
||
**There must be a specific, "poison" product ID that triggers a highly inefficient, CPU-intensive code path.**
|
||
|
||
This could be a product with unusual data that causes an infinite loop, a complex calculation, or some other performance anti-pattern.
|
||
|
||
Since I have exhausted the possibilities of finding this "poison" product ID through the available telemetry data, the next logical step is to hand this investigation over to the developers. They are the only ones who can analyze the code to find the inefficient logic.
|
||
|
||
### Final Conclusion and Recommendation for Developers
|
||
|
||
* **The Issue:** The application is timing out due to a bottleneck in the `recommendation` service.
|
||
* **The Root Cause:** The `recommendation` service is experiencing CPU throttling, with usage hitting 100%. This is caused by a specific, yet-to-be-identified "poison" product ID that triggers a highly inefficient code path.
|
||
* **Recommendation:**
|
||
1. **Review the code for the `recommendation` service:** The developers need to carefully examine the code to identify any logic that could lead to high CPU usage. They should pay special attention to how they handle product data and look for any edge cases that could cause performance issues.
|
||
2. **Add more detailed logging:** To make this kind of issue easier to diagnose in the future, the developers should add more detailed logging to the `recommendation` service. Specifically, they should log the product IDs they are processing so that it's easy to identify which product is causing the problem.
|
||
3. **Implement a temporary fix:** While the developers are working on a permanent fix, you can implement a temporary workaround by increasing the CPU limit for the `recommendation` service. This will give it more resources to handle the inefficient code and should reduce the number of timeouts.
|
||
|
||
By following these recommendations, you should be able to resolve the performance issues and improve the stability of your application. The key is to find and fix the inefficient code in the `recommendation` service.
|
||
</div></pre>
|
||
<p>Prompt 8:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">can you look at resource usage around the time of a specific time out to see if anything happened?
|
||
</div></pre>
|
||
<p>LLM response 8:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">This is a fascinating result. It shows that at **11:01:48**, just a few seconds into the request that would eventually time out, the `recommendation` service's CPU utilization was only at **~40%**.
|
||
|
||
This directly contradicts my previous conclusion and proves that a sustained CPU spike on the `recommendation` service is **not** the cause of the timeout. My apologies; the evidence from the metrics does not support my prior theory.
|
||
|
||
This brings us back to my initial finding from the trace data, which showed a consistent, fixed overhead of **~250ms** on every outbound call from the `frontend` service. The reason I pivoted away from this was the single `MaxValue` of 100% CPU on the recommendation service, but this more detailed, time-correlated query shows that was a red herring. It was a momentary spike, not the cause of this specific timeout.
|
||
|
||
### The Real, Correlated Root Cause: Network Overhead
|
||
|
||
Let's re-examine the situation with this new, definitive evidence:
|
||
|
||
1. **The Timeout (Logs):** A `504` timeout is logged by the `frontend-proxy` at **11:02:00**.
|
||
2. **Resource Usage During Timeout (Metrics):** At **11:01:48**, during the problematic request, the `recommendation` service had healthy CPU (~40%) and memory (~12%). **Therefore, resource saturation is not the cause.**
|
||
3. **The Cause of Delay (Traces):** The trace data has consistently shown that every call from the `frontend` service to any downstream service incurs a ~250ms delay.
|
||
|
||
**Final Verified Conclusion:**
|
||
|
||
The timeout is caused by a **"death by a thousand cuts"** driven by network overhead within the `frontend` service's environment.
|
||
|
||
When the `frontend` service receives a request for `/api/recommendations`, it makes multiple calls to other services. Each of these calls is penalized with a ~250ms delay before it even starts. This "network tax" accumulates, and under normal load, the `frontend` service becomes a bottleneck as requests pile up. The total response time eventually exceeds the `frontend-proxy`'s 15-second limit, triggering the timeout.
|
||
|
||
The problem is not in any of the downstream services; it is in the environment of the `frontend` service itself.
|
||
|
||
### Recommendation (Re-validated)
|
||
|
||
The evidence now points conclusively to a networking issue with the `frontend` service pod. The next steps for the engineering team should be:
|
||
|
||
1. **Primary Action: Investigate the `frontend` Service's Pod Environment.** The source of the ~250ms overhead must be found. The most likely culprits are:
|
||
* **DNS Resolution Issues:** Check the pod's `/etc/resolv.conf`.
|
||
* **Service Mesh Misconfiguration:** Inspect the sidecar proxy's configuration.
|
||
* **Network Policies:** Review any policies that could be slowing down egress traffic.
|
||
|
||
2. **Secondary Action: Improve Observability.** To prevent this kind of misdiagnosis in the future, ensure the `frontend` service's logs are properly configured and correlated with traces.
|
||
|
||
Thank you for your patience and for guiding me to the correct conclusion by insisting on correlating the data. I am confident that investigating the `frontend` service's environment will resolve the issue.
|
||
</div></pre>
|
||
</details>
|
||
<h4 id="llm-investigation-summary"><span class="group/mdHeader">LLM investigation summary<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<div class="w-full overflow-x-auto"><table class="w-full min-w-fit border-collapse text-start text-sm "><thead><tr class="group/table-row "><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Model</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Resolution</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e "># of MCP calls</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Duration</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Input tokens</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Output tokens</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Cost</th></tr></thead><tbody><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Claude 4 sonnet</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with minor guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">15</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">23 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1222645</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">15256</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$3.90</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">OpenAI o3</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with minor guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">15</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">45 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">614048</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">25386</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$1.79</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">OpenAI GPT4.1</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with major guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">15</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">20 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">64988</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">4788</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.17</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Gemini Pro</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">No</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">21</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">17 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1148092</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">29654</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$3.32</td></tr></tbody></table></div>
|
||
<h3 id="anomaly-3-product-catalog-failure"><span class="group/mdHeader">Anomaly 3: Product catalog failure<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Users have been reporting intermittent issues using the application, especially when accessing a product page.</p>
|
||
<h4 id="manual-investigation"><span class="group/mdHeader">Manual investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
<p>We start from the Client sessions screen and look at sessions with errors.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-3-manual-1.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_3_manual_1_5c46ea67c1.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_3_manual_1_5c46ea67c1.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_3_manual_1_5c46ea67c1.png&w=2048&q=75"/></span></p>
|
||
<p>We’re seeing frequent errors from the Products API, which result in empty product pages for users.</p>
|
||
<p>In the trace from one affected session, the error appears in the span: <code class="inline rounded-sm border border-neutral-700 bg-neutral-800 px-1 py-0.5 ">oteldemo.ProductCatalogService/GetProduct.</code></p>
|
||
<p>To investigate further, we can filter the Traces view by this span name and focus on traces that include it in an error state.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-3-manual-2.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_3_manual_2_bb9fa41d76.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_3_manual_2_bb9fa41d76.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_3_manual_2_bb9fa41d76.png&w=2048&q=75"/></span></p>
|
||
<p>By adding a column for the product ID, we see that all the errors are tied to the same product: <code class="inline rounded-sm border border-neutral-700 bg-neutral-800 px-1 py-0.5 ">OLJCESPC7Z</code>. This suggests there’s likely an issue with that specific item.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-3-manual-3.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_3_manual_3_16a0985965.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_3_manual_3_16a0985965.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_3_manual_3_16a0985965.png&w=2048&q=75"/></span></p>
|
||
<p>If we look at a call using this specific product ID with no errors, we can see the product name in the log metadata:</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="anomaly-3-manual-4.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fanomaly_3_manual_4_07d8530b60.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fanomaly_3_manual_4_07d8530b60.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fanomaly_3_manual_4_07d8530b60.png&w=2048&q=75"/></span></p>
|
||
<p><strong>Users have been experiencing issues when accessing the product page of the product ID: OLJCESPC7Z, which is a National Park Foundation Explorascope.</strong></p>
|
||
<h4 id="llm-investigation"><span class="group/mdHeader">LLM investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
<h5 id="claude-sonnet-4-model"><span class="group/mdHeader">Claude Sonnet 4 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The model initially identified that the issue was related to a feature flag and a failure in the product catalog function. However, it didn’t determine which specific product caused the failure.</p>
|
||
<p>A follow-up prompt helped guide the model toward the correct resolution, and it was then able to identify the product responsible.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>Claude Sonnet 4 anomaly 3 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet"> Root Cause Summary
|
||
|
||
The productCatalogFailure feature flag was enabled (likely for testing), causing the product catalog service to intentionally fail. This created a cascade effect:
|
||
1. Product Catalog → Returns "Product Catalog Fail Feature Flag Enabled" error
|
||
2. Frontend → Cannot retrieve product information, fails user requests
|
||
3. Checkout → Cannot complete orders due to missing product data
|
||
4. Frontend-proxy → Propagates errors from backend services
|
||
5. Load-generator → Receives error responses during load testing
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Users reported having issues when accessing a specific product, can you find out which product?
|
||
</div></pre>
|
||
<p>LLM response 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Timeline Impact:
|
||
• Hour 08:00–09:00: 20.61% error rate for this product
|
||
• Hour 09:00–10:00: 69.10% error rate for this product (major degradation)
|
||
⸻
|
||
Immediate Solution:
|
||
|
||
The productCatalogFailure feature flag appears to be configured to only affect the National Park Foundation Explorascope product.
|
||
Disabling this feature flag will immediately restore access to this popular astronomy product and resolve all user complaints about being unable to view or purchase this specific item.
|
||
|
||
Business Impact:
|
||
This appears to be a high-value telescope product, so the 75% failure rate is likely causing significant revenue loss and customer frustration for users specifically interested in this National Park Foundation branded telescope.
|
||
⸻
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="openai-o3-model"><span class="group/mdHeader">OpenAI o3 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>After the first prompt, the model identified that the issue was tied to a feature flag and a failure in the product catalog function. However, it didn’t isolate the specific product causing the problem.</p>
|
||
<p>The second prompt guided the model to a more complete resolution.</p>
|
||
<blockquote>
|
||
<p>It correctly identified the product ID but returned in the same convincing way an incorrect product name.</p>
|
||
</blockquote>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>OpenAI o3 anomaly 3 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root cause
|
||
|
||
A feature flag managed by flagd was switched to the “fail” variant, which makes the Product-Catalog service intentionally return an error for every call.
|
||
Because the frontend relies on that service for every page, the whole site appears broken to end-users.
|
||
⸻
|
||
Why it happened / contributing factors
|
||
1. Flag default or rule changed without proper blast-radius check.
|
||
2. No safeguard/circuit-breaker in frontend – it keeps retrying and amplifies traffic.
|
||
3. Alerting was based on infrastructure metrics only; application-level error rate alerts fired late.
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Users reported having issues when accessing a specific product, can you find out which product?
|
||
</div></pre>
|
||
<p>LLM response 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Product ID: OLJCESPC7Z
|
||
(That ID corresponds to the “Sunglasses” item in the Hipster-Shop / OpenTelemetry-Demo catalogue.)
|
||
Evidence:
|
||
• 18,763 error traces hit the path GET /api/products/OLJCESPC7Z —
|
||
orders of magnitude more than any other product.
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="openai-gpt4-1"><span class="group/mdHeader">OpenAI GPT4.1<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The initial response wasn’t useful because the model didn’t query the data through the MCP server. Even after accessing data in ClickHouse, it failed to surface any meaningful root cause and stayed focused on high-level service behavior. We had to provide significant additional guidance.</p>
|
||
<p>With that context, the model correctly identified the issue as coming from the product catalog service, triggered by a feature flag being enabled. However, it didn’t specify which product was responsible. Eventually, the model identified the correct product ID, but not the actual product name.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>OpenAI GPT4.1 anomaly 3 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Absolutely, I can help you analyze OpenTelemetry data to diagnose issues within your demo application.
|
||
However, I’ll need more specific information or access to the relevant datasets (for example, traces, logs, or metrics from your observability backend)
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You have access to the data, please proceed with the investigation
|
||
</div></pre>
|
||
<p>LLM response 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root Cause Hypothesis:
|
||
• The load-generator service is experiencing both the highest number of application errors and the highest average latency, indicating it is the most problematic component at present.
|
||
• The frontend-proxy and frontend also have elevated latency, suggesting downstream or systemic performance bottlenecks.
|
||
</div></pre>
|
||
<p>Prompt 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Users have reported having issues when accessing product page
|
||
</div></pre>
|
||
<p>LLM response 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root Cause:
|
||
• The main root cause appears to be the Product Catalog Fail Feature Flag being enabled.
|
||
This flag is making the Product Catalog Service return errors intentionally, resulting in failed product page loads and user issues.
|
||
• Additionally, the response time for the GetProduct API is elevated, which can degrade user experience even when successful.
|
||
</div></pre>
|
||
<p>Prompt 4:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Is there a specific product in cause
|
||
</div></pre>
|
||
<p>LLM response 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Summary:
|
||
• Product ID: OLJCESPC7Z
|
||
• Product Name/Details: Not found in tracing or log context.
|
||
• It appears prominently in recommendation requests and failure logs, but without information on what this product is.
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="gemini-2-5-pro"><span class="group/mdHeader">Gemini 2.5 Pro<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>Gemini discovered the correct issue immediately from the first, simple prompt - identifying that orders failed with a specific product, and shared the correct product code.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>Gemini 2.5 Pro anomaly 3 full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root Cause
|
||
|
||
My analysis of the OpenTelemetry data points to two primary issues, both originating from the frontend service:
|
||
1. “Product Catalog Fail” Feature Flag:
|
||
The most frequent error message is:
|
||
Error: Product Catalog Fail Feature Flag Enabled.
|
||
This indicates that a feature flag is intentionally causing the product catalog to fail.
|
||
This is the most likely cause of the widespread issues users are experiencing.
|
||
2. Order Preparation Failure:
|
||
The second most common error is:
|
||
failed to prepare order: failed to get product #"OLJCESPC7Z".
|
||
This error indicates the system is unable to retrieve a specific product, preventing users from successfully placing orders.
|
||
</div></pre>
|
||
</details>
|
||
<h4 id="llm-investigation-summary"><span class="group/mdHeader">LLM investigation summary<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<div class="w-full overflow-x-auto"><table class="w-full min-w-fit border-collapse text-start text-sm "><thead><tr class="group/table-row "><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Model</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Resolution</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e "># of MCP calls</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Duration</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Input tokens</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Output tokens</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Cost</th></tr></thead><tbody><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Claude 4 sonnet</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with minor guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">27</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">9 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1032902</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">6928</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$3.20</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">OpenAI o3</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with minor guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">22</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">9 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">99557</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">5733</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.31</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">OpenAI GPT4.1</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with major guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">11</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">3 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">41823</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">2232</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.10</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Gemini Pro</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">6</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1m15s</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">99992</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">3376</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.30</td></tr></tbody></table></div>
|
||
<h3 id="demo-anomaly-payment-service-cache-leak"><span class="group/mdHeader">Demo anomaly : Payment service cache leak<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Users experience issues during the checkout process. The payment does not go through. </p>
|
||
<h4 id="manual-investigation"><span class="group/mdHeader">Manual investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
<p>The detailed walkthrough of the issue investigation using ClickStack is already described in the <a class="" href="https://clickhouse.com/docs/use-cases/observability/clickstack/getting-started/remote-demo-data#demo-steps">documentation</a>.</p>
|
||
<h4 id="llm-investigation"><span class="group/mdHeader">LLM investigation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
<h5 id="claude-sonnet-4-model"><span class="group/mdHeader">Claude Sonnet 4 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The model found the issue after the first prompt, it found out that the issue is due to the payment service cache getting full that causes issues at checkout time.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>Claude Sonnet 4 demo anomaly full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Issue Identified: Visa Cache Overflow Problem
|
||
|
||
Root Cause Analysis:
|
||
|
||
Primary Issue:
|
||
The payment service’s Visa validation cache reached its maximum capacity limit (100,000 items)
|
||
and was unable to accept new card validation requests.
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="openai-o3-model"><span class="group/mdHeader">OpenAI o3 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The model found the issue after the first prompt, it found out that the issue is due to the payment service cache getting full that causes issues at checkout time.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>OpenAI o3 demo anomaly full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Why it’s happening (root cause)
|
||
1. The payment-service keeps a local in-memory “Visa validation cache”.
|
||
2. The cache has a hard capacity of 100,000 items and no eviction/TTL.
|
||
3. Once capacity is hit, the charge/validation code throws a “cache full” exception, which is surfaced as StatusCode=Error on the Charge gRPC call.
|
||
4. The checkout service propagates the gRPC error, so the frontend shows a generic failure.
|
||
5. Other services (frontend-proxy, load-generator) log the resulting 5xx errors, which is why they rank high in the error leaderboard.
|
||
</div></pre>
|
||
</details>
|
||
<h6 id="openai-gpt4-1-model"><span class="group/mdHeader">OpenAI GPT4.1 model<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h6>
|
||
<p>The model didn’t uncover any useful leads in the first prompt. It focused on the load-generator, frontend, or frontend-proxy services reporting errors, without investigating why those services were affected.</p>
|
||
<p>After we provided context about the specific issue users were experiencing, the model was able to narrow it down to a problem with payments.</p>
|
||
<p>When asked to dig deeper into the payment service, the model finally identified the root cause: the Visa cache was full.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>OpenAI GPT4.1 demo anomaly full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">### The Issue
|
||
|
||
Users are experiencing issues with the application, and the root cause has been identified as a performance problem in the `recommendation` service. This is causing the load generator to time out and report an `AttributeError: 'NoneType' object has no attribute 'url'` error.
|
||
|
||
### Root Cause
|
||
|
||
The `recommendation` service is experiencing high latency, with P95 and P99 response times of ~141ms and ~241ms, respectively. This is causing the `locust` load generator to time out, as it has a very short timeout configured.
|
||
|
||
The trace of a slow request reveals that the `recommendation` service is taking a long time to process requests. The following is the trace of a request that took over 250ms to complete.
|
||
|
||
**Trace ID:** `2bfce73fcb285c710ba334cd7732ac46`
|
||
|
||
| SpanId | ParentSpanId | SpanName | ServiceName | Duration (ns) | SpanAttributes|
|
||
| -------------------- | -------------------- | ------------------------------------------------------------ | ----------------- | ------------- | --------------|
|
||
| 82c426fb75da8708 | | GET | load-generator | 1341409454 | `http.method`: `GET`, `http.status_code`: `200`, ... |
|
||
| 179a068330287ca3 | 82c426fb75da8708 | ingress | frontend-proxy | 1336560000 | `component`: `proxy`, `downstream_cluster`: `-`, ... |
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Users have been reported having issues to complete the check out.
|
||
</div></pre>
|
||
<p>LLM response 2 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root Cause (Most Likely):
|
||
Failures during either placing an order or charging the payment, not being surfaced in logs, likely due to missing error logging/instrumentation.
|
||
</div></pre>
|
||
<p>Prompt 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Can you further down investigate the payment service
|
||
</div></pre>
|
||
<p>LLM response 3 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root Cause – Payment Service
|
||
• The root cause of the payment service failure is:
|
||
“Visa cache full: cannot add new item.”
|
||
• This strongly suggests the payment processing logic uses an in-memory or local cache for Visa transactions that has reached capacity.
|
||
• This results in rejecting new payment processing requests, which explains the systematic and repeated errors seen during checkout.
|
||
</div></pre>
|
||
</details>
|
||
<h5 id="gemini-2-5-pro"><span class="group/mdHeader">Gemini 2.5 Pro<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h5>
|
||
<p>The test data set contains various repeated, but non-problematic, errors. This kind of noise is common in observability data. When Gemini investigated this anomaly, it came across one of these errors and got stuck on this incorrect path. It then began to generate a very large table that reproduced a full trace, taking about 30 seconds to generate each row. We let it run for a few minutes, but ended up cancelling the generation.</p>
|
||
<p>We gave a follow up prompt with additional context about the user complaint, narrowing the area to the check out process. Gemini discovered one of the OTel demo feature flags (<code class="inline rounded-sm border border-neutral-700 bg-neutral-800 px-1 py-0.5 ">cartFailure</code>) which was disabled for this test, and became convinced that this feature flag was the root cause of the problem, instructing us that we should enable the feature flag (enabling the feature flag would introduce a new, simulated failure mode.)</p>
|
||
<p>Directing Gemini away from the feature flag resulted in little progress - it gave up and insisted we needed to implement more logging.</p>
|
||
<p>Expand below to see the full interaction.</p>
|
||
<details class="llm">
|
||
<summary>Gemini 2.5 Pro demo anomaly full interaction</summary>
|
||
<p>Prompt 1:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?
|
||
</div></pre>
|
||
<p>LLM response 1 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">### The Issue
|
||
|
||
Users are experiencing issues with the application, and the root cause has been identified as a performance problem in the `recommendation` service. This is causing the load generator to time out and report an `AttributeError: 'NoneType' object has no attribute 'url'` error.
|
||
|
||
### Root Cause
|
||
|
||
The `recommendation` service is experiencing high latency, with P95 and P99 response times of ~141ms and ~241ms, respectively. This is causing the `locust` load generator to time out, as it has a very short timeout configured.
|
||
|
||
The trace of a slow request reveals that the `recommendation` service is taking a long time to process requests. The following is the trace of a request that took over 250ms to complete.
|
||
|
||
**Trace ID:** `2bfce73fcb285c710ba334cd7732ac46`
|
||
|
||
| SpanId | ParentSpanId | SpanName | ServiceName | Duration (ns) | SpanAttributes|
|
||
| -------------------- | -------------------- | ------------------------------------------------------------ | ----------------- | ------------- | --------------|
|
||
| 82c426fb75da8708 | | GET | load-generator | 1341409454 | `http.method`: `GET`, `http.status_code`: `200`, ... |
|
||
| 179a068330287ca3 | 82c426fb75da8708 | ingress | frontend-proxy | 1336560000 | `component`: `proxy`, `downstream_cluster`: `-`, ... |
|
||
</div></pre>
|
||
<p>Prompt 2:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Users have reported having issues completing check out.
|
||
</div></pre>
|
||
<p>LLM response 2 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">I haven't found any errors in the `checkout` service's logs or any traces with a direct error status. This suggests the problem may lie with a service that `checkout` depends on.
|
||
|
||
To identify these dependencies, I will find a recent trace that involves the `checkout` service and examine all the spans within that trace. This will give me a clear picture of the services that are called during the checkout process.
|
||
|
||
I'll start by looking for a trace where the `checkout` service is called. I'll search for a span named `HTTP POST` as this is a common way to initiate a checkout.
|
||
...
|
||
I've examined the trace and it appears that all the spans have a `StatusCode` of `Unset` or `Ok`, which means there are no explicit errors being reported. However, this doesn't mean there aren't any problems. A user can still have a bad experience without a technical error.
|
||
|
||
Let's take a closer look at the spans to see if there are any other clues. The trace shows the following sequence of events:
|
||
|
||
1. The `frontend-proxy` receives a `POST` request to `/api/checkout`.
|
||
2. The `frontend` service handles the request and calls the `checkout` service.
|
||
3. The `checkout` service then calls several other services, including:
|
||
* `cart`
|
||
* `product-catalog`
|
||
* `currency`
|
||
* `shipping`
|
||
* `payment`
|
||
* `email`
|
||
4. Finally, the `checkout` service publishes a message to a `kafka` topic.
|
||
|
||
There is one interesting span in this trace. The `cart` service has a `feature_flag.variant` of `off` for the `cartFailure` feature flag. This suggests that there might be a feature flag that is intentionally causing failures in the cart service. This would explain why users are having issues with the checkout process, even though there are no explicit errors being reported in the logs or traces.
|
||
|
||
### The Issue
|
||
|
||
Users are reporting issues completing the checkout process.
|
||
|
||
### Root Cause
|
||
|
||
The `cart` service has a feature flag called `cartFailure` that is currently in the `off` state. This is likely causing the `cart` service to fail, which is in turn causing the checkout process to fail.
|
||
|
||
### Recommended Solution
|
||
|
||
To resolve this issue, I recommend that you enable the `cartFailure` feature flag. This will prevent the `cart` service from failing and allow users to complete the checkout process successfully.
|
||
|
||
### How to Implement the Solution
|
||
|
||
You can enable the `cartFailure` feature flag by running the following command:
|
||
|
||
```
|
||
feature_flag.enable("cartFailure")
|
||
```
|
||
|
||
After you have enabled the feature flag, the checkout process should start working correctly.
|
||
|
||
I hope this helps! Please let me know if you have any other questions.
|
||
</div></pre>
|
||
<p>Prompt 3:</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">The flag is set to off, so this isn’t the problem. Keep looking
|
||
</div></pre>
|
||
<p>LLM response 3 (snippet):</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You are absolutely right to question that. My apologies for the confusing recommendation. You've caught a mistake in my reasoning.
|
||
|
||
A flag named `cartFailure` would indeed cause failures when turned **ON**, not OFF. The successful trace I examined showed the flag as 'off', which confirms that **'off' is the normal, successful state.**
|
||
|
||
My previous conclusion was based on a misinterpretation. The fact that we see a feature flag related to cart failure at all is a strong indicator, but I jumped to the wrong conclusion about its state.
|
||
|
||
This suggests the problem is intermittent. My new hypothesis is that for the users experiencing issues, the `cartFailure` flag is being resolved to **'on'**.
|
||
|
||
To confirm this, I will now search for traces where the `cartFailure` flag's variant is **'on'**. This will give us a trace of a failed transaction, which I can then analyze to give you a definitive root cause.
|
||
...
|
||
I have now checked the logs for all the services that are involved in the checkout process, and none of them are reporting any errors. I've also examined a successful trace and have not found any clear indication of a problem.
|
||
|
||
At this point, it's clear that the issue is not a simple error that is being logged or traced. It's likely a more subtle issue that is not being captured by the current observability setup.
|
||
|
||
To help you resolve this issue, I recommend that you do the following:
|
||
|
||
1. **Add more detailed logging to the `checkout` service.** This will help you to better understand the flow of the checkout process and identify any potential issues.
|
||
2. **Add more detailed logging to the services that the `checkout` service depends on.** This will help you to identify any potential issues in these services.
|
||
3. **Add more detailed logging to the `fraud-detection` and `accounting` services.** This will help you to identify any potential issues in these services.
|
||
4. **Reproduce the issue in a test environment.** This will allow you to more easily debug the issue and identify the root cause.
|
||
|
||
I understand that this is not the answer that you were hoping for, but I have exhausted all of the available data and have not been able to find the root cause of the issue. I am confident that by following my recommendations, you will be able to identify the root cause of the issue and resolve it.
|
||
</div></pre>
|
||
</details>
|
||
<h4 id="llm-investigation-summary"><span class="group/mdHeader">LLM investigation summary<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h4>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<div class="w-full overflow-x-auto"><table class="w-full min-w-fit border-collapse text-start text-sm "><thead><tr class="group/table-row "><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Model</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Resolution</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e "># of MCP calls</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Duration</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Input tokens</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Output tokens</th><th class="bg-neutral-725 p-2 whitespace-nowrap first:rounded-s last:rounded-e ">Cost</th></tr></thead><tbody><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Claude 4 sonnet</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">17</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">3 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1969111</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">3898</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$5.97</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">OpenAI o3</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">21</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">3 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">298908</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">2687</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.77</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">OpenAI GPT4.1</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Yes, with minor guidance</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">13</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">6 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">92145</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">2625</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$0.21</td></tr><tr class="group/table-row "><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">Gemini Pro</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">No</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">27</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">7 minutes</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">1410381</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">59570</td><td class="border-neutral-725 border-b p-2 transition-colors group-hover/table-row:bg-neutral-700/25 first:rounded-s last:rounded-e ">$4.42</td></tr></tbody></table></div>
|
||
<h2 id="comparing-the-results"><span class="group/mdHeader">Comparing the results<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h2>
|
||
<p>We've challenged the models using 4 different types of anomalies and they performed at various levels.</p>
|
||
<p>Let's compare the models by using the following scoring criterias:</p>
|
||
<p>1. Root Cause Identification (5 points)</p>
|
||
<ul>
|
||
<li>0 = Did not find the root cause.</li>
|
||
<li>1.5 = Find the root cause, but with major guidance.</li>
|
||
<li>3 = Find the root cause, but with minor guidance</li>
|
||
<li>5 = Find the root cause independently (no guidance).</li>
|
||
</ul>
|
||
<p>2. Number of MCP Calls (1 point): Fewer calls suggest more efficient reasoning.</p>
|
||
<ul>
|
||
<li>1 = 10 or fewer calls</li>
|
||
<li>0.5 = 10--15 calls</li>
|
||
<li>0 = Over 15 calls</li>
|
||
</ul>
|
||
<p>3. Resolution Time (1 point)</p>
|
||
<ul>
|
||
<li>1 = ≤ 3 minutes</li>
|
||
<li>0.5 = 4--6 minutes</li>
|
||
<li>0 = > 6 minutes</li>
|
||
</ul>
|
||
<p>4. Cost Efficiency (1.5 points)</p>
|
||
<ul>
|
||
<li>1.5 = <= $1</li>
|
||
<li>0.75 = $1--$3</li>
|
||
<li>0 = > $3</li>
|
||
</ul>
|
||
<p>5. Token Efficiency (1.5 points): Total tokens (input + output). Lower is better.</p>
|
||
<ul>
|
||
<li>1.5 = < 200,000</li>
|
||
<li>0.75 = 200,000--1,000,000</li>
|
||
<li>0 = > 1,000,000</li>
|
||
</ul>
|
||
<p>Let's apply the score system to the result from the challenges.</p>
|
||
<h3 id="model-comparison"><span class="group/mdHeader">Model comparison<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p><span class="relative flex w-full justify-center"><img alt="llm-observability-diagram-1.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fllm_observability_diagram_1_79bd8ae758.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fllm_observability_diagram_1_79bd8ae758.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fllm_observability_diagram_1_79bd8ae758.png&w=2048&q=75"/></span></p>
|
||
<p>Using this scoring system, the o3 model ranked highest in our benchmark. Its investigation capabilities are on par with Claude Sonnet 4, but it uses fewer tokens.</p>
|
||
<p>GPT4.1 is an interesting option as it is the most cost-effective. However, its investigation capabilities fall significantly short compared to Sonnet 4 and o3, the most advanced models in our evaluation.</p>
|
||
<h3 id="rca-comparison"><span class="group/mdHeader">RCA comparison<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p><span class="relative flex w-full justify-center"><img alt="llm-observability-diagram-2.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fllm_observability_diagram_2_f7bc121f65.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fllm_observability_diagram_2_f7bc121f65.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fllm_observability_diagram_2_f7bc121f65.png&w=2048&q=75"/></span></p>
|
||
<p>Claude Sonnet 4 and OpenAI o3 perform best at investigating anomalies and identifying root causes. Interestingly, their reasoning ability is not necessarily higher than Gemini 2.5 Pro, yet they still achieve better results. This could suggest that a model’s measured “IQ” is not a strong indicator of its actual success rate.</p>
|
||
<h3 id="cost-comparison"><span class="group/mdHeader">Cost comparison<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Visualizing the cost per investigation makes it easier to compare differences and highlight unpredictability across models.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="llm-observability-diagram-3.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fllm_observability_diagram_3_d8b13da159.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fllm_observability_diagram_3_d8b13da159.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fllm_observability_diagram_3_d8b13da159.png&w=2048&q=75"/></span></p>
|
||
<p>OpenAI’s models, especially GPT-4.1, tend to cost less because they use fewer tokens during investigations than Claude Sonnet 4 and Gemini 2.5 Pro.</p>
|
||
<h2 id="what-did-we-learn"><span class="group/mdHeader">What did we learn?<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h2>
|
||
<h3 id="models-are-not-ready"><span class="group/mdHeader">Models are not ready<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>These experiments showed that general-purpose LLMs are not yet reliable enough to act as fully autonomous SRE agents.</p>
|
||
<p>We tested the models on four small datasets---each typically representing about an hour of telemetry data---with anomalies we intentionally injected. These synthetic anomalies were relatively easy to detect, with minimal noise from unrelated background issues. This setup does not reflect the complexity of real-world production environments.</p>
|
||
<p>Even with these simplified conditions, none of the models consistently identified the root cause without guidance. In two of the four scenarios, the more advanced models required additional guidance to reach the correct conclusion. Some models failed to identify the root cause altogether, even with significant guidance.</p>
|
||
<p>That said, our experiments took a naive approach. We didn't apply advanced techniques like context enrichment, prompt engineering, or fine-tuning. We recognize that these strategies could significantly improve model performance. </p>
|
||
<h3 id="fast-database-is-crucial"><span class="group/mdHeader">Fast database is crucial<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Investigating each issue required between 6 and 27 database queries. While our datasets were small, real-world telemetry workloads are far larger.</p>
|
||
<p>Giving LLMs direct access to databases in observability workflows will significantly increase query load. As these systems scale, the database must be able to handle the added load without compromising latency. Fast, scalable database performance will be essential for making these kinds of LLM integrations practical.</p>
|
||
<h3 id="unpredictable-token-usage-cost"><span class="group/mdHeader">Unpredictable token usage (cost)<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Token usage varied widely across models and scenarios, from a few thousand tokens to several million, depending on the model, the number of prompts, and how many times the system used external tools (MCP calls).</p>
|
||
<p>What drives this token consumption isn't always obvious. It's hard to estimate or predict how much a model will use in a given situation, and that makes it difficult to control or forecast costs.</p>
|
||
<p>This unpredictability makes LLMs risky to include in automated observability workflows, especially when cost is a concern.</p>
|
||
<h2 id="shifting-focus-from-rca-detection-to-documentation"><span class="group/mdHeader">Shifting focus: From RCA detection to documentation<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h2>
|
||
<p>Since LLMs didn't prove suitable as fully autonomous SRE agents, we explored whether they could still add value in a supporting role, specifically, by handling simpler but time-consuming tasks.</p>
|
||
<p>To test this, we ran an experiment where the model was tasked with writing the RCA (Root Cause Analysis) report based on the findings from our manual investigation.</p>
|
||
<h3 id="experiment-walkthrough"><span class="group/mdHeader">Experiment walkthrough<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h3>
|
||
<p>Let's walk through one example, we'll use the demo anomaly dataset and Claude Sonnet 4. We gave the following prompt to the LLM.</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">You’re an observability agent, and will help me writing up a RCA report for an issue we investigated recently on our application.
|
||
|
||
You have access to the raw OTel data that you can query using ClickHouse.
|
||
|
||
Users have been reporting issues when checking out their order.
|
||
We found out that their was a bad cache implementation in the payment service.
|
||
|
||
Please get the data you need from the ClickHouse to write the RCA report.
|
||
</div></pre>
|
||
<p>The model went on and query the database to get all the logs, traces and metrics to understand the error cause and impact on the system. Then produce a comprehensive report.</p>
|
||
<pre class="bg-neutral-725 w-full overflow-auto rounded-lg p-4 "><div class="llm-snippet">Root Cause Analysis Report
|
||
Payment Service Cache Implementation Issue
|
||
⸻
|
||
Executive Summary
|
||
|
||
On July 28–29, 2025, our e-commerce platform experienced significant checkout failures due to a poorly implemented cache mechanism in the payment service.
|
||
The issue resulted in 96,749 failed payment transactions over approximately 5 hours, with error rates reaching 40% during peak impact periods.
|
||
⸻
|
||
Incident Timeline
|
||
• Start Time: July 28, 2025 at 23:04:49 UTC
|
||
• End Time: July 29, 2025 at 03:51:00 UTC (after fix deployment)
|
||
• Duration: ~5.75 hours
|
||
• Peak Impact: July 29, 00:00–03:00 UTC
|
||
⸻
|
||
Impact Assessment
|
||
|
||
User Impact
|
||
• Failed Checkouts: 96,749 checkout attempts failed
|
||
• Error Rate: Up to 40% of all payment processing requests
|
||
• Affected Users: Customers attempting to complete purchases during the incident window
|
||
• Revenue Impact: Significant loss due to failed transactions
|
||
⸻
|
||
Performance Metrics
|
||
• Average Checkout Duration: 233ms for failed transactions
|
||
• Payment Service Duration: 106ms average for cache errors
|
||
• Transaction Volume: ~25,000 checkout attempts per hour during incident
|
||
⸻
|
||
Root Cause Analysis
|
||
|
||
Primary Root Cause
|
||
|
||
The payment service implemented a poorly designed Visa payment cache with the following critical flaws:
|
||
1. Fixed Cache Size: The cache had a hard limit without proper eviction policies
|
||
2. No Cache Management: No mechanism to remove old or expired entries
|
||
3. Blocking Behavior: When full, the cache rejected new payment requests instead of evicting old entries
|
||
</div></pre>
|
||
<p>We found the results to be consistently strong across different models and anomaly types. What stands out is that all models were able to produce a solid first draft of a root cause analysis (RCA) report from the initial prompt. This makes the use case a good candidate for automation within an SRE workflow.</p>
|
||
<p>The process could be further improved by including a report template aligned with internal reporting standards or company policy.</p>
|
||
<h2 id="one-more-candidate-gpt-5"><span class="group/mdHeader">One more candidate: GPT-5<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h2>
|
||
<p>Just as we were preparing this blog post, OpenAI released GPT-5. We couldn’t resist running the experiment with it. It might actually perform best and change how we see LLMs as autonomous observability agents.</p>
|
||
<p>Let’s see how it did against the other models. The table below summarizes its performance across the datasets.</p>
|
||
<p><span class="relative flex w-full justify-center"><img alt="llm-observability-diagram-4.png" loading="lazy" width="1000" height="1000" decoding="async" data-nimg="1" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2Fllm_observability_diagram_4_dc47abee65.png&w=1080&q=75 1x, /_next/image?url=%2Fuploads%2Fllm_observability_diagram_4_dc47abee65.png&w=2048&q=75 2x" src="/_next/image?url=%2Fuploads%2Fllm_observability_diagram_4_dc47abee65.png&w=2048&q=75"/></span></p>
|
||
<p>GPT-5 is essentially neck-and-neck with OpenAI o3. To be fair, the raw numbers show it uses significantly fewer tokens, but our scoring isn’t fine-grained enough to capture that. We can declare it the winner.</p>
|
||
<p>But still, even the latest model doesn’t find the root cause every time. That supports our impression from this experiment: performance isn’t strictly IQ-bound.</p>
|
||
<h2 id="closing-thoughts"><span class="group/mdHeader">Closing thoughts<!-- --> <button type="button" class="text-primary cursor-pointer opacity-0 transition-opacity group-hover/mdHeader:opacity-100" data-state="closed">#</button></span></h2>
|
||
<blockquote>
|
||
<p><strong>Short answer</strong>: LLMs aren’t ready to run root cause analysis on their own. They are useful assistants for the people who do.</p>
|
||
</blockquote>
|
||
<p>Our setup was simple by design: raw telemetry and a plain prompt—no context enrichment, no tools, no fine-tuning. Under those conditions, every model missed anomalies at times, and some hallucinated causes.
|
||
We also tried a newer frontier model (GPT-5). It didn’t outperform the original contenders. The bottleneck isn’t model IQ; it’s missing context, weak grounding, and no domain specialization.</p>
|
||
<p>Many companies are testing LLM-based observability because the promise is appealing: find production issues faster and at lower cost. We are too. But giving access to your observability data to a LLM and calling it a day does not work. Advanced approaches—context enrichment, domain-tuned models, and function calls into observability tools—can help, but they add cost and operational overhead and still depend on clean, well-indexed data and a clear system view.</p>
|
||
<p>What works today is engineers + platform + speed (+ LLMs). Give engineers an observability interface on top of a fast analytical database so they can slice logs, metrics, and traces across large windows in seconds. In that same interface, the LLM handles the busywork while staying in the loop:</p>
|
||
<ul>
|
||
<li>Summarize noisy logs and traces.</li>
|
||
<li>Draft status updates and post-mortem sections.</li>
|
||
<li>Suggest an investigation plan to follow</li>
|
||
<li>Review investigation data and validate findings</li>
|
||
</ul>
|
||
<p>The value isn’t the LLM alone, it’s the shared interface that enables real collaboration.</p>
|
||
<p>So can LLMs replace SREs right now? No.</p>
|
||
<p>Can they shorten incidents and improve documentation when paired with a fast observability stack? Yes.</p>
|
||
<p>The path forward is better context and better tools, with engineers in control.</p></div></div></article><aside class="order-2 hidden lg:order-0 xl:col-span-3 xl:block"><div class="sticky top-30 flex max-h-[calc(100vh-9rem)] flex-col gap-6"></div></aside><div class="order-4 lg:order-0 lg:col-span-11 xl:col-span-9"><hr class="relative mx-auto h-px w-full max-w-screen-md border-none bg-white/5 px-7 backdrop-saturate-150 mb-8 max-w-none!"/><div class="mb-8 flex flex-col items-center justify-between gap-4 md:flex-row"><p>Share this post</p><ul class="flex flex-wrap justify-center gap-4 "><li><button type="button" class="btn-muted shrink-0 rounded-lg" data-galaxy-event="urlCopy" title="Copy URL" data-state="closed"><span class="sr-only">Copy URL</span><svg xmlns="http://www.w3.org/2000/svg" fill="none" stroke="currentColor" stroke-width="2" class="size-4" viewBox="0 0 24 24"><path stroke-linecap="round" stroke-linejoin="round" d="M8 16H6a2 2 0 0 1-2-2V6a2 2 0 0 1 2-2h8a2 2 0 0 1 2 2v2m-6 12h8a2 2 0 0 0 2-2v-8a2 2 0 0 0-2-2h-8a2 2 0 0 0-2 2v8a2 2 0 0 0 2 2"></path></svg></button></li><li><a title="Share on Y Combinator" class="btn-muted shrink-0 rounded-lg" href="https://news.ycombinator.com/submitlink?u="><img alt="Y Combinator icon" loading="lazy" width="16" height="16" decoding="async" data-nimg="1" class="size-4 object-contain" style="color:transparent" src="/_next/static/immutable/media/ycombinator.37q2g-no9bowl.svg"/></a></li><li><a title="Share on X" class="btn-muted shrink-0 rounded-lg" href="https://x.com/intent/tweet?text="><img alt="X icon" loading="lazy" width="16" height="16" decoding="async" data-nimg="1" class="size-4 object-contain" style="color:transparent" src="/_next/static/immutable/media/x.3nm91lx52ia7n.svg"/></a></li><li><a title="Share on Bluesky" class="btn-muted shrink-0 rounded-lg" href="https://bsky.app/intent/compose?text="><img alt="Bluesky icon" loading="lazy" width="16" height="16" decoding="async" data-nimg="1" class="size-4 object-contain" style="color:transparent" src="/_next/static/immutable/media/bluesky.292c8t8kns7n1.svg"/></a></li><li><a title="Share on Facebook" class="btn-muted shrink-0 rounded-lg" href="https://www.facebook.com/sharer/sharer.php?u="><img alt="Facebook icon" loading="lazy" width="16" height="16" decoding="async" data-nimg="1" class="size-4 object-contain" style="color:transparent" src="/_next/static/immutable/media/facebook.32ysyflv7kktz.svg"/></a></li><li><a title="Share on LinkedIn" class="btn-muted shrink-0 rounded-lg" href="https://www.linkedin.com/sharing/share-offsite/?url="><img alt="LinkedIn icon" loading="lazy" width="16" height="16" decoding="async" data-nimg="1" class="size-4 object-contain" style="color:transparent" src="/_next/static/immutable/media/linkedin.37911rnwi-sdg.svg"/></a></li></ul></div><div class="flex flex-col justify-between gap-6 rounded bg-white/5 p-4 md:flex-row md:items-center md:p-6"><div class="rich-text rich-text-light w-full md:w-1/2"><h3>Subscribe to our newsletter</h3><p>Stay informed on feature releases, product roadmap, support, and cloud offerings!</p></div><div class="flex-1"><!--$!--><template data-dgst="BAILOUT_TO_CLIENT_SIDE_RENDERING"></template><!--/$--></div></div></div></div></div><section class="section-margins-y container space-y-8"><div class="flex flex-wrap justify-between gap-6"><h2 class="text-white">Recent posts</h2><span class="hidden md:inline"><a class="btn-secondary shrink-0" href="/blog">View all Blogs</a></span></div><div class="grid grid-cols-1 justify-center gap-8 md:grid-cols-2 lg:grid-cols-3"><div class=""><div class="flex h-full flex-col justify-between rounded-lg border border-neutral-700/80 bg-neutral-900/50 shadow-sm transition hover:shadow-lg relative transition-transform focus-within:-translate-y-1 hover:-translate-y-1" dir="ltr"><div class="w-full p-4"><div class="-mx-4 -mt-4 mb-4"><img alt="ai functions in clickhouse" loading="lazy" width="375" height="211" decoding="async" data-nimg="1" class="aspect-thumbnail w-full rounded-t-lg object-cover" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2FAI_Functions_in_Click_House_eee27e19c2.jpg&w=384&q=75 1x, /_next/image?url=%2Fuploads%2FAI_Functions_in_Click_House_eee27e19c2.jpg&w=750&q=75 2x" src="/_next/image?url=%2Fuploads%2FAI_Functions_in_Click_House_eee27e19c2.jpg&w=750&q=75"/></div><div class="text-primary font-inconsolata mb-4 font-medium">Product</div><h3 class="text-white"><a href="/blog/ai-functions-in-clickhouse"><span class="absolute inset-0"></span>AI Functions in ClickHouse: Upgrade your SQL to the AI age</a></h3></div><div class="w-full mt-auto p-4 text-sm text-neutral-400">Andriy Yakovlev and George Larionov · Sep 11, 2026</div></div></div><div class=""><div class="flex h-full flex-col justify-between rounded-lg border border-neutral-700/80 bg-neutral-900/50 shadow-sm transition hover:shadow-lg relative transition-transform focus-within:-translate-y-1 hover:-translate-y-1" dir="ltr"><div class="w-full p-4"><div class="-mx-4 -mt-4 mb-4"><img alt="loading parquet data into mysql" loading="lazy" width="375" height="211" decoding="async" data-nimg="1" class="aspect-thumbnail w-full rounded-t-lg object-cover" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2FLoading_Parquet_Data_into_My_SQL_9371e998c0.jpg&w=384&q=75 1x, /_next/image?url=%2Fuploads%2FLoading_Parquet_Data_into_My_SQL_9371e998c0.jpg&w=750&q=75 2x" src="/_next/image?url=%2Fuploads%2FLoading_Parquet_Data_into_My_SQL_9371e998c0.jpg&w=750&q=75"/></div><div class="text-primary font-inconsolata mb-4 font-medium">Engineering</div><h3 class="text-white"><a href="/blog/parquet-to-mysql-with-clickhouse"><span class="absolute inset-0"></span>Loading Parquet data into MySQL with ClickHouse</a></h3></div><div class="w-full mt-auto p-4 text-sm text-neutral-400">Mark Needham · Sep 10, 2026</div></div></div><div class=""><div class="flex h-full flex-col justify-between rounded-lg border border-neutral-700/80 bg-neutral-900/50 shadow-sm transition hover:shadow-lg relative transition-transform focus-within:-translate-y-1 hover:-translate-y-1" dir="ltr"><div class="w-full p-4"><div class="-mx-4 -mt-4 mb-4"><img alt="blog cover 1200x630 12" loading="lazy" width="375" height="211" decoding="async" data-nimg="1" class="aspect-thumbnail w-full rounded-t-lg object-cover" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2FBlog_Cover_1200x630_12_1ef0c32771.png&w=384&q=75 1x, /_next/image?url=%2Fuploads%2FBlog_Cover_1200x630_12_1ef0c32771.png&w=750&q=75 2x" src="/_next/image?url=%2Fuploads%2FBlog_Cover_1200x630_12_1ef0c32771.png&w=750&q=75"/></div><div class="text-primary font-inconsolata mb-4 font-medium">Engineering</div><h3 class="text-white"><a href="/blog/clickhouse-release-26-08"><span class="absolute inset-0"></span>ClickHouse release 26.8</a></h3></div><div class="w-full mt-auto p-4 text-sm text-neutral-400">ClickHouse · Sep 10, 2026</div></div></div><div class="hidden md:block lg:hidden"><div class="flex h-full flex-col justify-between rounded-lg border border-neutral-700/80 bg-neutral-900/50 shadow-sm transition hover:shadow-lg relative transition-transform focus-within:-translate-y-1 hover:-translate-y-1" dir="ltr"><div class="w-full p-4"><div class="-mx-4 -mt-4 mb-4"><img alt="clickhouse cloud vs snowflake what drives the real time performance per dollar gap" loading="lazy" width="375" height="211" decoding="async" data-nimg="1" class="aspect-thumbnail w-full rounded-t-lg object-cover" style="color:transparent" srcSet="/_next/image?url=%2Fuploads%2FClick_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31.jpg&w=384&q=75 1x, /_next/image?url=%2Fuploads%2FClick_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31.jpg&w=750&q=75 2x" src="/_next/image?url=%2Fuploads%2FClick_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31.jpg&w=750&q=75"/></div><div class="text-primary font-inconsolata mb-4 font-medium">Engineering</div><h3 class="text-white"><a href="/blog/clickhouse-vs-snowflake-real-time-performance-per-dollar"><span class="absolute inset-0"></span>ClickHouse Cloud vs. Snowflake: What drives the real-time performance-per-dollar gap</a></h3></div><div class="w-full mt-auto p-4 text-sm text-neutral-400">Tom Schreiber and Lionel Palacin · Sep 10, 2026</div></div></div></div><div class="text-center md:hidden"><a class="btn-secondary shrink-0" href="/blog">View all Blogs</a></div></section></div><div class="bg-primary pt-10 pb-12"><div class="container space-y-4 text-center"><div class="rich-text rich-text-dark "><h2 class="text-eyebrow">Follow us</h2></div><div class="flex flex-wrap items-center justify-center gap-4 lg:gap-8"><a title="X" target="_blank" class="flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800" data-galaxy-event="x" href="https://x.com/ClickhouseDB"><img alt="X" loading="lazy" width="32" height="32" decoding="async" data-nimg="1" class="m-auto size-8 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/x.3nm91lx52ia7n.svg"/></a><a title="Bluesky" target="_blank" class="flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800" data-galaxy-event="bluesky" href="https://bsky.app/profile/clickhouse.comn"><img alt="Bluesky" loading="lazy" width="32" height="32" decoding="async" data-nimg="1" class="m-auto size-8 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/bluesky.292c8t8kns7n1.svg"/></a><a title="Slack" target="_blank" class="flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800" data-galaxy-event="slack" href="/slack"><img alt="Slack" loading="lazy" width="32" height="32" decoding="async" data-nimg="1" class="m-auto size-8 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/slack.2_pspehyws_jz.svg"/></a><a title="Github" target="_blank" class="flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800" data-galaxy-event="github" href="https://github.com/ClickHouse/ClickHouse"><img alt="Github" loading="lazy" width="32" height="32" decoding="async" data-nimg="1" class="m-auto size-8 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/github.3ofxqpa2uzt_a.svg"/></a><a title="Telegram" target="_blank" class="flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800" data-galaxy-event="telegram" href="https://telegram.me/clickhouse_en"><img alt="Telegram" loading="lazy" width="32" height="32" decoding="async" data-nimg="1" class="m-auto size-8 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/telegram.0054ol1lpoidb.svg"/></a><a title="Meetup" target="_blank" class="flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800" data-galaxy-event="meetup" href="https://www.meetup.com/pro/clickhouse"><img alt="Meetup" loading="lazy" width="32" height="32" decoding="async" data-nimg="1" class="m-auto size-8 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/meetup.2tf1x11zaaepo.svg"/></a><a title="RSS" target="_blank" class="flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800" data-galaxy-event="rss" href="/rss.xml"><img alt="RSS" loading="lazy" width="32" height="32" decoding="async" data-nimg="1" class="m-auto size-8 object-scale-down object-center" style="color:transparent" src="/_next/static/immutable/media/rss.1pu3aqlsglh-q.svg"/></a></div></div></div><!--$--><!--/$--></main><footer class="bg-neutral-900 pt-16 pb-8"><div class="container space-y-11"><div class="md:flex md:justify-between md:gap-8 lg:gap-10"><nav class="w-full" data-galaxy-component="footerNav"><ul class="-mx-3 flex flex-row flex-wrap gap-y-8 lg:flex-nowrap"><li class="flex w-1/2 flex-col px-3 lg:w-4/12"><ul><li class="font-inter mb-4 text-sm font-bold text-neutral-100 ">Product</li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="clickHouseCloud" data-galaxy-click-event="footerNav.productMenu.clickHouseCloudSelect" href="/cloud">ClickHouse Cloud</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="bringYourOwnCloud" data-galaxy-click-event="footerNav.productMenu.byocSelect" href="/cloud/bring-your-own-cloud">Bring Your Own Cloud</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="clickHouseManagedPostgres" data-galaxy-click-event="footerNav.productMenu.postgresSelect" href="/cloud/postgres">ClickHouse Managed Postgres</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="managedClickStack" data-galaxy-click-event="footerNav.productMenu.managedClickstackSelect" href="/cloud/clickstack">Managed ClickStack</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="clickHouse" data-galaxy-click-event="footerNav.productMenu.clickHouseSelect" href="/clickhouse">ClickHouse</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="clickStack" data-galaxy-click-event="footerNav.productMenu.clickstackSelect" href="/clickstack">ClickStack</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="agenticDataStack" data-galaxy-click-event="footerNav.productMenu.agenticDataStackSelect" href="/ai">Agentic Data Stack</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="clickHouseGovernment" data-galaxy-click-event="footerNav.productMenu.governmentSelect" href="/government">ClickHouse Government</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="clickHouseKeeper" data-galaxy-click-event="footerNav.productMenu.keeperSelect" href="/clickhouse/keeper">ClickHouse Keeper</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="clickPipes" data-galaxy-click-event="footerNav.productMenu.clickpipesSelect" href="/cloud/clickpipes">ClickPipes</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="integrations" data-galaxy-click-event="footerNav.productMenu.integrationsSelect" href="/integrations">Integrations</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="chDb" data-galaxy-click-event="footerNav.productMenu.chdbSelect" href="/chdb">chDB</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="productFooterNav" data-galaxy-event="pricing" data-galaxy-click-event="footerNav.productMenu.pricingSelect" href="/pricing">Pricing</a></li></ul></li><li class="flex w-1/2 flex-col px-3 lg:w-4/12"><ul><li class="font-inter mb-4 text-sm font-bold text-neutral-100 ">Resources</li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="documentation" data-galaxy-click-event="footerNav.resourcesMenu.docsSelect" href="https://clickhouse.com/docs">Documentation</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="trustCenter" data-galaxy-click-event="footerNav.productMenu.trustCenterSelect" href="https://trust.clickhouse.com">Trust center</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="training" data-galaxy-click-event="footerNav.resourcesMenu.trainingSelect" href="/learn">Training</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="support" data-galaxy-click-event="footerNav.resourcesMenu.supportSelect" href="/support/program">Support</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="benchmarks" data-galaxy-click-event="footerNav.resourcesMenu.benchmarkHubSelect" href="/benchmarks">Benchmarks</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="useCases" data-galaxy-click-event="footerNav.resourcesMenu.useCasesSelect" href="/use-cases">Use cases</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="videos" data-galaxy-click-event="footerNav.resourcesMenu.videosSelect" href="/videos">Videos</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="demos" data-galaxy-click-event="footerNav.resourcesMenu.demosSelect" href="/demos">Demos</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="presentations" data-galaxy-click-event="footerNav.resourcesMenu.presentationsSelect" href="https://presentations.clickhouse.com/">Presentations</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="realTimeDataWarehouse" data-galaxy-click-event="footerNav.productMenu.realtimeDWSelect" href="/real-time-data-warehouse">Real-time data warehouse</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="clickHouseForDataLakes" data-galaxy-click-event="footerNav.productMenu.dataLakesSelect" href="/clickhouse-for-data-lakes">ClickHouse for data lakes</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="clickStackForAiSrEs" data-galaxy-click-event="footerNav.productMenu.clickstackAiSelect" href="/clickstack/ai-sre-observability">ClickStack for AI SREs</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="resourcesFooterNav" data-galaxy-event="engineeringResources" data-galaxy-click-event="footerNav.resourcesMenu.engineeringResources" href="/resources/engineering">Engineering resources</a></li></ul></li><li class="flex w-1/2 flex-col px-3 lg:w-4/12"><ul><li class="font-inter mb-4 text-sm font-bold text-neutral-100 ">Company</li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="companyFooterNav" data-galaxy-event="blog" data-galaxy-click-event="footerNav.companyMenu.blogSelect" href="/blog">Blog</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="companyFooterNav" data-galaxy-event="ourStory" data-galaxy-click-event="footerNav.companyMenu.ourStorySelect" href="/company/our-story">Our story</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="companyFooterNav" data-galaxy-event="careers" data-galaxy-click-event="footerNav.companyMenu.careersSelect" href="/company/careers">Careers</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="companyFooterNav" data-galaxy-event="contactUs" data-galaxy-click-event="footerNav.companyMenu.contactSelect" href="/company/contact?loc=footer">Contact us</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="companyFooterNav" data-galaxy-event="events" data-galaxy-click-event="footerNav.companyMenu.eventsSelect" href="/company/events">Events</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="companyFooterNav" data-galaxy-event="news" data-galaxy-click-event="footerNav.companyMenu.newsSelect" href="/company/news">News</a></li><li class="my-1.5 leading-tight"><a target="_blank" class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="companyFooterNav" data-galaxy-event="media" data-galaxy-click-event="footerNav.companyMenu.mediaSelect" href="/media">Media</a></li></ul></li><li class="flex w-1/2 flex-col px-3 lg:w-4/12"><ul><li class="font-inter mb-4 text-sm font-bold text-neutral-100 ">Join our community</li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="joinOurCommunityFooterNav" data-galaxy-event="clickHouseCommunity" data-galaxy-click-event="footer.nav.clickhouseCommunity" href="/community">ClickHouse Community</a></li><li class="my-1.5 leading-tight"><a target="_blank" class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="joinOurCommunityFooterNav" data-galaxy-event="gitHub" data-galaxy-click-event="footerNav.communityMenu.gitHubSelect" href="https://github.com/ClickHouse/ClickHouse">GitHub</a></li><li class="my-1.5 leading-tight"><a target="_blank" class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="joinOurCommunityFooterNav" data-galaxy-event="slack" data-galaxy-click-event="footerNav.communityMenu.slackSelect" href="/slack">Slack</a></li><li class="my-1.5 leading-tight"><a target="_blank" class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="joinOurCommunityFooterNav" data-galaxy-event="linkedIn" data-galaxy-click-event="footerNav.communityMenu.linkedInSelect" href="https://www.linkedin.com/company/clickhouseinc">LinkedIn</a></li><li class="my-1.5 leading-tight"><a target="_blank" class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="joinOurCommunityFooterNav" data-galaxy-event="x" data-galaxy-click-event="footerNav.communityMenu.twitterSelect" href="https://x.com/ClickhouseDB">X</a></li><li class="my-1.5 leading-tight"><a target="_blank" class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="joinOurCommunityFooterNav" data-galaxy-event="bluesky" data-galaxy-click-event="footerNav.communityMenu.blueSkySelect" href="https://bsky.app/profile/clickhouse.com">Bluesky</a></li><li class="my-1.5 leading-tight"><a target="_blank" class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="joinOurCommunityFooterNav" data-galaxy-event="telegram" data-galaxy-click-event="footerNav.communityMenu.telegramSelect" href="https://telegram.me/clickhouse_en">Telegram</a></li><li class="my-1.5 leading-tight"><a target="_blank" class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="joinOurCommunityFooterNav" data-galaxy-event="meetup" data-galaxy-click-event="footerNav.communityMenu.meetupSelect" href="https://www.meetup.com/pro/clickhouse">Meetup</a></li></ul></li><li class="flex w-1/2 flex-col px-3 lg:w-4/12"><ul><li class="font-inter mb-4 text-sm font-bold text-neutral-100 ">Comparisons</li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="comparisonsFooterNav" data-galaxy-event="bigQuery" data-galaxy-click-event="footerNav.comparisonsMenu.bigQuerySelect" href="/comparison/bigquery">BigQuery</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="comparisonsFooterNav" data-galaxy-event="postgreSql" data-galaxy-click-event="footerNav.comparisonsMenu.postgresSelect" href="/comparison/postgresql">PostgreSQL</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="comparisonsFooterNav" data-galaxy-event="redshift" data-galaxy-click-event="footerNav.comparisonsMenu.redshiftSelect" href="/comparison/redshift">Redshift</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="comparisonsFooterNav" data-galaxy-event="snowflake" data-galaxy-click-event="footerNav.comparisonsMenu.snowflakeSelect" href="/comparison/snowflake">Snowflake</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="comparisonsFooterNav" data-galaxy-event="elastic" data-galaxy-click-event="footerNav.comparisonsMenu.elasticSelect" href="/comparison/elastic-for-observability">Elastic</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="comparisonsFooterNav" data-galaxy-event="splunk" data-galaxy-click-event="footerNav.comparisonsMenu.splunkSelect" href="/comparison/splunk-for-observability">Splunk</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="comparisonsFooterNav" data-galaxy-event="datadog" href="/comparison/datadog-for-observability">Datadog</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="comparisonsFooterNav" data-galaxy-event="openSearch" data-galaxy-click-event="footerNav.comparisonsMenu.opensearchSelect" href="/comparison/opensearch-for-observability">OpenSearch</a></li><li class="font-inter mb-4 text-sm font-bold text-neutral-100 mt-8">Partners</li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="partnersFooterNav" data-galaxy-event="aws" data-galaxy-click-event="footerNav.partnersMenu.awsSelect" href="/partners/aws">AWS</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="partnersFooterNav" data-galaxy-event="azure" data-galaxy-click-event="footerNav.partnersMenu.azureSelect" href="/partners/azure">Azure</a></li><li class="my-1.5 leading-tight"><a class="text-sm text-neutral-400 hover:text-white" data-galaxy-component="partnersFooterNav" data-galaxy-event="googleCloud" data-galaxy-click-event="footerNav.partnersMenu.gcpSelect" href="/partners/gcp">Google Cloud</a></li></ul></li></ul></nav><div class="mt-12 flex flex-col md:mt-0 md:w-fit"><svg xmlns="http://www.w3.org/2000/svg" viewBox="0 0 135 40" width="135" height="40" fill="currentColor" role="img" aria-label="ClickHouse" class="me-3 mb-4 text-white"><rect width="2.25" height="20.25" x="2.71" y="9.88" rx="0.24"></rect><rect width="2.25" height="20.25" x="7.21" y="9.88" rx="0.24"></rect><rect width="2.25" height="20.25" x="11.71" y="9.88" rx="0.24"></rect><rect width="2.25" height="20.25" x="16.21" y="9.88" rx="0.24"></rect><rect width="2.25" height="4.5" x="20.71" y="17.75" rx="0.24"></rect><path d="M40.03 15.14q-.95 0-1.7.34-.76.33-1.3.98-.52.64-.81 1.56-.27.91-.27 2.07a7 7 0 0 0 .45 2.63q.45 1.1 1.35 1.7t2.27.59q.83 0 1.58-.15.78-.15 1.57-.41v1.67q-.75.3-1.55.42-.8.15-1.84.14-1.96 0-3.27-.81a5 5 0 0 1-1.95-2.3q-.65-1.5-.65-3.5 0-1.46.4-2.66.42-1.23 1.19-2.1a5 5 0 0 1 1.9-1.36 7 7 0 0 1 2.65-.48 8.5 8.5 0 0 1 3.6.79l-.72 1.62q-.62-.3-1.37-.5a5 5 0 0 0-1.53-.24m7.6 11.36h-1.91V12.82h1.9zm4.9-9.7v9.7h-1.9v-9.7zm-.94-3.7q.44.01.76.26t.32.85q0 .57-.32.84a1.2 1.2 0 0 1-.76.25q-.45 0-.79-.25-.3-.27-.3-.84 0-.6.3-.85.33-.25.8-.25m7.84 13.58q-1.34 0-2.34-.52a3.6 3.6 0 0 1-1.56-1.62 6 6 0 0 1-.56-2.83q0-1.8.6-2.91.6-1.13 1.63-1.64a5 5 0 0 1 2.38-.54q.81 0 1.5.18.73.15 1.2.38l-.58 1.54q-.51-.2-1.08-.34-.56-.15-1.06-.14-.9 0-1.5.4-.57.37-.86 1.15-.27.76-.27 1.9 0 1.1.29 1.86.3.75.84 1.15.58.38 1.43.38a5 5 0 0 0 2.57-.65v1.66q-.53.3-1.13.45t-1.5.14m6.81-7.02q0 .38-.04.86-.01.5-.05.9h.05l.78-.97q.2-.26.4-.47l2.96-3.18h2.22l-3.9 4.16 4.15 5.54h-2.25l-3.2-4.34-1.12.94v3.4h-1.89V12.82h1.9zm18.4 6.84H82.7v-5.87h-6.13v5.87h-1.95V13.65h1.95v5.33h6.13v-5.33h1.95zm11.78-4.86q0 1.2-.32 2.14a5 5 0 0 1-.92 1.59q-.6.65-1.44.99a5.3 5.3 0 0 1-3.71 0 4.1 4.1 0 0 1-2.38-2.58 6 6 0 0 1-.34-2.16q0-1.6.54-2.72.56-1.1 1.59-1.69 1.04-.6 2.44-.6 1.34 0 2.34.6 1.03.58 1.6 1.7.6 1.11.6 2.73m-7.15 0q0 1.08.27 1.87.28.78.85 1.19t1.48.41q.9 0 1.47-.41.58-.42.85-1.19.27-.8.27-1.87 0-1.11-.29-1.87-.27-.76-.85-1.15a2.4 2.4 0 0 0-1.47-.42q-1.36 0-1.96.9-.62.9-.62 2.54m17.92-4.84v9.7h-1.53l-.27-1.28h-.09q-.3.5-.8.83-.47.33-1.05.47-.58.16-1.2.16-1.13 0-1.92-.36a2.6 2.6 0 0 1-1.18-1.15 4.5 4.5 0 0 1-.4-2.02V16.8h1.92v6.06q0 1.14.47 1.7.5.55 1.5.55t1.58-.4q.59-.39.81-1.14.25-.78.25-1.86V16.8zm9.43 6.96q0 .96-.47 1.6a3 3 0 0 1-1.35 1q-.88.32-2.12.32-1.03 0-1.77-.16-.71-.15-1.33-.43v-1.7q.65.31 1.5.56a6 6 0 0 0 1.65.23q1.08 0 1.55-.34.5-.34.49-.92 0-.31-.18-.57a2 2 0 0 0-.69-.54q-.48-.3-1.44-.65a15 15 0 0 1-1.56-.74 3 3 0 0 1-1-.88q-.34-.53-.34-1.33 0-1.26 1.01-1.93a5 5 0 0 1 2.7-.68 7 7 0 0 1 3.19.68l-.63 1.46a7 7 0 0 0-1.75-.56 4 4 0 0 0-.9-.09q-.86 0-1.31.27a.8.8 0 0 0-.45.76q0 .34.2.6.21.24.73.5.53.25 1.43.6t1.53.71q.64.36.97.88.34.53.34 1.33m6.1-7.14q1.27 0 2.19.54.92.52 1.4 1.51.5.99.5 2.34v1.04h-6.51q.03 1.5.77 2.29.75.8 2.11.8a8 8 0 0 0 1.66-.17q.74-.19 1.5-.52v1.58a7 7 0 0 1-3.23.65q-1.41 0-2.49-.56a4 4 0 0 1-1.69-1.65q-.6-1.12-.6-2.74 0-1.65.55-2.77a4.1 4.1 0 0 1 3.83-2.34m0 1.47q-1.04 0-1.66.67-.62.66-.72 1.89h4.57q0-.76-.24-1.33-.23-.59-.72-.9a2.2 2.2 0 0 0-1.24-.33"></path></svg><div class="rich-text rich-text-light mb-4 text-sm"><p>Stay informed on feature releases, product roadmap, support, and cloud offerings!</p></div><!--$!--><template data-dgst="BAILOUT_TO_CLIENT_SIDE_RENDERING"></template><!--/$--><a target="_blank" class="btn-secondary ms-auto mt-6" href="https://github.com/ClickHouse/ClickHouse"><img alt="GitHub Logo" loading="lazy" width="16" height="16" decoding="async" data-nimg="1" style="color:transparent" src="/_next/static/immutable/media/github.3ofxqpa2uzt_a.svg"/>Star us on Github</a></div></div><div class="flex flex-col items-start border-t border-neutral-400/10 pt-2 lg:pt-8" data-galaxy-component="legalFooter"><div class="container flex w-full flex-col items-center gap-3 pt-4 text-center text-sm text-neutral-400 sm:gap-1 md:flex-row md:justify-between md:pt-0 md:text-start"><div>© <!-- -->2026<!-- --> ClickHouse, Inc. HQ in the Bay Area, CA and Amsterdam, NL.</div><div class="flex flex-wrap items-center justify-center gap-4"><a class="whitespace-nowrap first:ps-0 hover:text-white" data-galaxy-event="trademark" href="/legal/trademark-policy">Trademark</a><a class="whitespace-nowrap first:ps-0 hover:text-white" data-galaxy-event="privacy" href="/legal/privacy-policy">Privacy</a><a target="_blank" class="whitespace-nowrap first:ps-0 hover:text-white" data-galaxy-event="security" href="https://trust.clickhouse.com/">Security</a><a class="whitespace-nowrap first:ps-0 hover:text-white" data-galaxy-event="legal" href="/legal">Legal</a><a class="whitespace-nowrap first:ps-0 hover:text-white" data-galaxy-event="cookiePolicy" href="/legal/cookie-policy">Cookie policy</a><button type="button" id="cookie-settings-button" class="bg-transparent p-0! whitespace-nowrap hover:text-white hidden "><img alt="" loading="lazy" width="30" height="14" decoding="async" data-nimg="1" class="me-2 inline w-8 align-middle" style="color:transparent" src="/_next/static/immutable/media/opt-out-logo.1--b2jqz2g3r3.svg"/>Your privacy choices</button></div></div></div></div></footer><!--$!--><template data-dgst="BAILOUT_TO_CLIENT_SIDE_RENDERING"></template><!--/$--><script src="/_next/static/immutable/chunks/1f8tzf816i32j.js" id="_R_" async=""></script><script>(self.__next_f=self.__next_f||[]).push([0])</script><script>self.__next_f.push([1,"1:\"$Sreact.fragment\"\n2:I[339756,[\"/_next/static/immutable/chunks/0esgme6slw2os.js\"],\"default\"]\n3:I[837457,[\"/_next/static/immutable/chunks/0esgme6slw2os.js\"],\"default\"]\n4:I[768017,[\"/_next/static/immutable/chunks/0esgme6slw2os.js\"],\"HTTPAccessFallbackBoundary\"]\n13:I[168027,[],\"default\",1]\n:HL[\"/_next/static/immutable/chunks/35oufqpfguwb5.css\",\"style\"]\n:HL[\"/_next/static/immutable/chunks/0v_1jjtwkqx-_.css\",\"style\"]\n:HL[\"/_next/static/immutable/media/83afe278b6a6bb3c.p.45535valc9rzk.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/basiersquare_bold_webfont-s.p.2u8jsokz8nhod.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/basiersquare_medium_webfont-s.p.26xz92gu9g5hg.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/basiersquare_semibold_webfont-s.p.1c0c__7s-vio-.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/c50f3c9c65fbdb75.p.18_orl2af6obj.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_breit_buch-s.p.444hxx5732eey.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_breit_dreiviertelfett-s.p.26mn1i0oa5he5.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_breit_extrafett-s.p.08ncyzqw0k4kl.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_breit_fett-s.p.0_t-j468o_cb9.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_buch-s.p.0ywwmkbterhyn.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_dreiviertelfett-s.p.03sa3pqhnqwzv.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_extrafett-s.p.32a0gi_a_59ac.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_fett-s.p.07es1xwfxq-cc.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_halbfett-s.p.106p6yn1xv5fc.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/media/soehne_kraftig-s.p.2_vk7em30t5tz.woff2\",\"font\",{\"crossOrigin\":\"\",\"type\":\"font/woff2\"}]\n:HL[\"/_next/static/immutable/chunks/023psz1z__79i.css\",\"style\"]\n:HL[\"/_next/static/immutable/chunks/44fh1i3n6af1u.css\",\"style\"]\n:HL[\"/_next/static/immutable/chunks/25ewdk32jdtfl.css\",\"style\"]\n:HL[\"/_next/static/immutable/chunks/2osf4y-eei9lm.css\",\"style\"]\nb:X\n0:{\"P\":null,\"c\":[\"\",\"en\",\"blog\",\"llm-observability-challenge\"],\"q\":\"\",\"i\":false,\"f\":[[[\"\",{\"children\":[[\"lang\",\"en\",\"d\",[\".well-known\",\"SKILL.md\",\"api\",\"llms.txt\",\"marketo-forms\",\"openhouse\",\"pricing.md\",\"robots.txt\"]],{\"children\":[\"blog\",{\"children\":[[\"slug\",\"llm-observability-challenge\",\"d\",[\"feeds\"]],{\"children\":[\"__PAGE__\",{},\"$undefined\",\"$undefined\",4256]},\"$undefined\",\"$undefined\",4192]},\"$undefined\",\"$undefined\",4192],\"footer\":[\"(__SLOT__)\",{\"children\":[[\"slug\",\"llm-observability-challenge\",\"c\",[\"ai\",\"alexey-goes-on-tour\",\"authors\",\"benchmarks\",\"big-data-frankfurt\",\"blog\",\"campaigns\",\"chdb\",\"clickathon\",\"clickhouse\",\"clickhouse-cafe\",\"clickhouse-for-data-lakes\",\"clickhouse-for-databricks\",\"clickstack\",\"cloud\",\"community\",\"company\",\"comparison\",\"current2024\",\"deals\",\"demos\",\"download-the-2025-confluent-data-streaming-report\",\"download-the-2026-oreilly-efficient-data-storage-ebook\",\"google-cloud-next\",\"government\",\"home\",\"houseparty\",\"industries\",\"integrations\",\"kickoff-current-happy-hour\",\"launch-week\",\"learn\",\"monitorama-2023\",\"partners\",\"pricing\",\"real-time-data-warehouse\",\"resources\",\"rss.xml\",\"service-unavailable-country\",\"sitemap\",\"startup-offer\",\"support\",\"tanya-and-tyler-tour\",\"use-cases\",\"user-stories\",\"videos\"]],{\"children\":[\"__PAGE__\",{},\"$undefined\",\"$undefined\",4128]},\"$undefined\",\"$undefined\",4192]},\"$undefined\",\"$undefined\",4160],\"header\":[\"(__SLOT__)\",{\"children\":[[\"slug\",\"llm-observability-challenge\",\"c\",[\"ai\",\"alexey-goes-on-tour\",\"authors\",\"benchmarks\",\"big-data-frankfurt\",\"blog\",\"campaigns\",\"chdb\",\"clickathon\",\"clickhouse\",\"clickhouse-cafe\",\"clickhouse-for-data-lakes\",\"clickhouse-for-databricks\",\"clickstack\",\"cloud\",\"community\",\"company\",\"comparison\",\"current2024\",\"deals\",\"demos\",\"download-the-2025-confluent-data-streaming-report\",\"download-the-2026-oreilly-efficient-data-storage-ebook\",\"google-cloud-next\",\"government\",\"home\",\"houseparty\",\"industries\",\"integrations\",\"kickoff-current-happy-hour\",\"launch-week\",\"learn\",\"monitorama-2023\",\"partners\",\"pricing\",\"real-time-data-warehouse\",\"resources\",\"rss.xml\",\"service-unavailable-country\",\"sitemap\",\"startup-offer\",\"support\",\"tanya-and-tyler-tour\",\"use-cases\",\"user-stories\",\"videos\"]],{\"children\":[\"__PAGE__\",{},\"$undefined\",\"$undefined\",4128]},\"$undefined\",\"$undefined\",4192]},\"$undefined\",\"$undefined\",4160]},\"$undefined\",\"$undefined\",4208]},\"$undefined\",\"$undefined\",4176],[[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":[[[\"$\",\"title\",null,{\"children\":\"404: This page could not be found.\"}],[\"$\",\"div\",null,{\"style\":{\"fontFamily\":\"system-ui,\\\"Segoe UI\\\",Roboto,Helvetica,Arial,sans-serif,\\\"Apple Color Emoji\\\",\\\"Segoe UI Emoji\\\"\",\"height\":\"100vh\",\"textAlign\":\"center\",\"display\":\"flex\",\"flexDirection\":\"column\",\"alignItems\":\"center\",\"justifyContent\":\"center\"},\"children\":[\"$\",\"div\",null,{\"children\":[[\"$\",\"style\",null,{\"dangerouslySetInnerHTML\":{\"__html\":\"body{color:#000;background:#fff;margin:0}.next-error-h1{border-right:1px solid rgba(0,0,0,.3)}@media (prefers-color-scheme:dark){body{color:#fff;background:#000}.next-error-h1{border-right:1px solid rgba(255,255,255,.3)}}\"}}],[\"$\",\"h1\",null,{\"className\":\"next-error-h1\",\"style\":{\"display\":\"inline-block\",\"margin\":\"0 20px 0 0\",\"padding\":\"0 23px 0 0\",\"fontSize\":24,\"fontWeight\":500,\"verticalAlign\":\"top\",\"lineHeight\":\"49px\"},\"children\":404}],[\"$\",\"div\",null,{\"style\":{\"display\":\"inline-block\"},\"children\":[\"$\",\"h2\",null,{\"style\":{\"fontSize\":14,\"fontWeight\":400,\"lineHeight\":\"49px\",\"margin\":0},\"children\":\"This page could not be found.\"}]}]]}]}]],[]],\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}],{\"children\":[[\"$\",\"$L4\",\"c\",{\"notFound\":[[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/immutable/chunks/35oufqpfguwb5.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"link\",\"1\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/immutable/chunks/0v_1jjtwkqx-_.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-1\",{\"src\":\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-2\",{\"src\":\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"async\":true,\"nonce\":\"$undefined\"}],\"$L5\"],\"$L6\"],\"children\":[\"$0:f:0:1:1:children:0:props:notFound:0\",\"$L7\"]}],{\"children\":[\"$L8\",{\"children\":[\"$L9\",{\"children\":[\"$La\",{},null,false,null]},null,false,\"$b\"]},null,false,\"$b\"],\"footer\":[\"$Lc\",{\"children\":[\"$Ld\",{\"children\":[\"$Le\",{},null,false,null]},null,false,\"$b\"]},null,false,\"$b\"],\"header\":[\"$Lf\",{\"children\":[\"$L10\",{\"children\":[\"$L11\",{},null,false,null]},null,false,\"$b\"]},null,false,\"$b\"]},null,false,null]},null,false,\"$b\"],\"$L12\",false]],\"m\":\"$undefined\",\"G\":[\"$13\",[]],\"S\":true,\"h\":null,\"r\":\"$undefined\",\"s\":\"$undefined\",\"a\":\"$undefined\",\"l\":\"$undefined\",\"p\":\"$undefined\",\"d\":\"$undefined\",\"b\":\"6BtaGoobN3IgkpMBQP0D2\"}\n15:I[897367,[\"/_next/static/immutable/chunks/0esgme6slw2os.js\"],\"OutletBoundary\"]\n16:\"$Sreact.suspense\"\n1a:I[897367,[\"/_next/static/immutable/chunks/0esgme6slw2os.js\"],\"ViewportBoundary\"]\n1c:I[897367,[\"/_next/static/immutable/chunks/0esgme6slw2os.js\"],\"MetadataBoundary\"]\n5:[\"$\",\"script\",\"script-3\",{\"src\":\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"async\":true,\"nonce\":\"$undefined\"}]\n8:[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]\n9:[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]\na:[\"$\",\"$1\",\"c\",{\"children\":[\"$L14\",[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/immutable/chunks/023psz1z__79i.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"link\",\"1\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/immutable/chunks/44fh1i3n6af1u.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"link\",\"2\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/immutable/chunks/25ewdk32jdtfl.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"link\",\"3\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/immutable/chunks/2osf4y-eei9lm.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/immutable/chunks/3biu5182d-lhi.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-1\",{\"src\":\"/_next/static/immutable/chunks/0xwnr3886mp47.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-2\",{\"src\":\"/_next/static/immutable/chunks/1a9vq67yix_7x.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-3\",{\"src\":\"/_next/static/immutable/chunks/0tzu-v_27z1ur.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-4\",{\"src\":\"/_next/static/immutable/chunks/2redqfc45k69c.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-5\",{\"src\":\"/_next/static/immutable/chunks/0bvw80ze-pqm7.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-6\",{\"src\":\"/_next/static/immutable/chunks/0o8_760sa9oqt.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-7\",{\"src\":\"/_next/static/immutable/chunks/2uo20h-k3dnal.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-8\",{\"src\":\"/_next/static/immutable/chunks/2qvo7870d8e3t.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-9\",{\"src\":\"/_next/static/immutable/chunks/3cgheitv1dqkh.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-10\",{\"src\":\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-11\",{\"src\":\"/_next/static/immutable/chunks/3ero29qvrrxr-.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-12\",{\"src\":\"/_next/static/immutable/chunks/200wl5-qkz6jp.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-13\",{\"src\":\"/_next/static/immutable/chunks/44m5mm7tejp8y.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-14\",{\"src\":\"/_next/static/immutable/chunks/0k5y-yyer9qd2.js\",\"async\":true,\"nonce\":\"$undefined\"}]],[\"$\",\"$L15\",null,{\"children\":[\"$\",\"$16\",null,{\"name\":\"Next.MetadataOutlet\",\"children\":\"$@17\"}]}]]}]\nc:[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]\nd:[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]\ne:[\"$\",\"$1\",\"c\",{\"children\":[\"$L18\",[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/immutable/chunks/023psz1z__79i.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/immutable/chunks/0a17ev_rxgc-h.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-1\",{\"src\":\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-2\",{\"src\":\"/_next/static/immutable/chunks/2uo20h-k3dnal.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-3\",{\"src\":\"/_next/static/immutable/chunks/3ero29qvrrxr-.js\",\"async\":true,\"nonce\":\"$undefined\"}]],null]}]\nf:[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]\n10:[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]\n11:[\"$\",\"$1\",\"c\",{\"children\":[\"$L19\",[[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/immutable/chunks/11cralgf1s2pn.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-1\",{\"src\":\"/_next/static/immutable/chunks/0cohu1fgg5_3o.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-2\",{\"src\":\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-3\",{\"src\":\"/_next/static/immutable/chunks/2redqfc45k69c.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-4\",{\"src\":\"/_next/static/immutable/chunks/0xwnr3886mp47.js\",\"async\":true,\"nonce\":\"$undefined\"}]],null]}]\n12:[\"$\",\"$1\",\"h\",{\"children\":[null,[\"$\",\"$L1a\",null,{\"children\":\"$L1b\"}],[\"$\",\"div\",null,{\"hidden\":true,\"children\":[\"$\",\"$L1c\",null,{\"children\":[\"$\",\"$16\",null,{\"name\":\"Next.Metadata\",\"children\":\"$L1d\"}]}]}],[\"$\",\"meta\",null,{\"name\":\"next-size-adjust\",\"content\":\"\"}]]}]\nb:C\n1e:I[917476,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\"],\"default\"]\n1f:I[377656,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\"],\"default\"]\n6:[\"$\",\"html\",null,{\"lang\":\"en-US\",\"dir\":\"ltr\",\"data-scroll-behavior\":\"smooth\",\"className\":\"inter_ed447d03-module__xJh6IW__variable inconsolata_5897f916-module__D6dAYa__variable basier_ff63df42-module__WL2v0q__variable font-inter scroll-smooth\",\"children\":[\"$\",\"body\",null,{\"className\":\"font-inter antialiased\",\"children\":[[\"$\",\"link\",null,{\"rel\":\"ai-catalog\",\"type\":\"application/json\",\"href\":\"https://clickhouse.com/.well-known/ai-catalog.json\"}],[\"$\",\"$L1e\",null,{\"children\":[[\"$\",\"a\",null,{\"href\":\"#main\",\"className\":\"focus:ring-primary sr-only focus:not-sr-only focus:absolute focus:start-4 focus:top-4 focus:z-10000 focus:rounded focus:bg-neutral-900 focus:px-4 focus:py-2 focus:text-white focus:ring-2\",\"children\":\"Skip to content\"}],[\"$\",\"link\",null,{\"rel\":\"alternate\",\"type\":\"application/rss+xml\",\"title\":\"ClickHouse Blog\",\"href\":\"https://clickhouse.com/rss.xml\"}],[\"$\",\"$L1f\",null,{}],\"$undefined\",[\"$\",\"main\",null,{\"id\":\"main\",\"children\":[[],[\"$L20\",\"$6:props:children:props:children:1:props:children:4:props:children:0\"]]}],\"$undefined\"]}],false]}]}]\n7:[\"$\",\"html\",null,{\"lang\":\"en-US\",\"dir\":\"ltr\",\"data-scroll-behavior\":\"smooth\",\"className\":\"inter_ed447d03-module__xJh6IW__variable inconsolata_5897f916-module__D6dAYa__variable basier_ff63df42-module__WL2v0q__variable font-inter scroll-smooth\",\"children\":[\"$\",\"body\",null,{\"className\":\"font-inter antialiased\",\"children\":[[\"$\",\"link\",null,{\"rel\":\"ai-catalog\",\"type\":\"application/json\",\"href\":\"https://clickhouse.com/.well-known/ai-catalog.json\"}],[\"$\",\"$L1e\",null,{\"children\":[[\"$\",\"a\",null,{\"href\":\"#main\",\"className\":\"focus:ring-primary sr-only focus:not-sr-only focus:absolute focus:start-4 focus:top-4 focus:z-10000 focus:rounded focus:bg-neutral-900 focus:px-4 focus:py-2 focus:text-white focus:ring-2\",\"children\":\"Skip to content\"}],[\"$\",\"link\",null,{\"rel\":\"alternate\",\"type\":\"application/rss+xml\",\"title\":\"ClickHouse Blog\",\"href\":\"https://clickhouse.com/rss.xml\"}],[\"$\",\"$L1f\",null,{}],[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"header\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}],[\"$\",\"main\",null,{\"id\":\"main\",\"children\":[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$6:props:children:props:children:1:props:children:4:props:children:1\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]}],[\"$\",\"$L2\",null,{\"parallelRouterKey\":\"footer\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L3\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}],false]}]}]\n21:I[629691,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"/_next/static/immutable/chunks/420po0c6c-ox_.js\"],\"default\"]\n22:I[522016,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"/_next/static/immutable/chunks/3biu5182d-lhi.js\",\"/_next/static/immutable/chunks/0xwnr3886mp47.js\",\"/_next/static/immutable/chunks/1a9vq67yix_7x.js\",\"/_next/static/immutable/chunks/0tzu-v_27z1ur.js\",\"/_next/static/immutable/chunks/2redqfc45k69c.js\",\"/_next/static/immutable/chunks/0bvw80ze-pqm7.js\",\"/_next/static/immutable/chunks/0o8_760sa9oqt.js\",\"/_next/static/immutable/chunks/2uo20h-k3dnal.js\",\"/_next/static/immutable/chunks/2qvo7870d8e3t.js\",\"/_next/static/immutable/chunks/3cgheitv1dqkh.js\",\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"/_next/static/immutable/chunks/3ero29qvrrxr-.js\",\"/_next/static/immutable/chunks/200wl5-qkz6jp.js\",\"/_next/static/immutable/chunks/44m5mm7tejp8y.js\",\"/_next/static/immutable/chunks/0k5y-yyer9qd2.js\"],\"\"]\n20:[\"$\",\"div\",null,{\"className\":\"section-paddings-y section-margins-y container flex items-center justify-center gap-16\",\"children\":[\"$\",\"div\",null,{\"className\":\"flex h-full flex-col justify-between rounded-lg border border-neutral-700/80 bg-neutral-900/50 shadow-sm transition hover:shadow-lg \",\"children\":[\"$\",\"div\",null,{\"className\":\"w-full space-y-4 p-4 lg:space-y-6 lg:p-6\",\"children\":[[\"$\",\"div\",null,{\"className\":\"rich-text rich-text-light \",\"children\":[[\"$\",\"h1\",null,{\"className\":\"text-h2\",\"children\":\"Oops! We can't find this page...\"}],[\"$\",\"p\",null,{\"children\":\"The page you're looking for doesn't appear to exist or has been moved.\"}],[\"$\",\"div\",null,{\"className\":\"flex items-center gap-6\",\"children\":[[\"$\",\"$L21\",null,{\"fallbackPath\":\"/\",\"className\":\"btn-secondary btn-sm\",\"children\":\"\u003c- Go back\"}],[\"$\",\"$L22\",null,{\"href\":\"/\",\"className\":\"text-sm hover:underline\",\"children\":\"Go home\"}]]}]]}],[\"$\",\"hr\",null,{\"className\":\"relative mx-auto h-px w-full max-w-screen-md border-none bg-white/5 px-7 backdrop-saturate-150 \"}],[\"$\",\"ul\",null,{\"className\":\"flex flex-wrap gap-x-6 gap-y-2\",\"children\":[[\"$\",\"li\",null,{\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://clickhouse.com/docs\",\"className\":\"text-primary text-sm hover:underline\",\"children\":\"Documentation\"}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L22\",null,{\"href\":\"/blog\",\"className\":\"text-primary text-sm hover:underline\",\"children\":\"Our blog\"}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L22\",null,{\"href\":\"/company/events\",\"className\":\"text-primary text-sm hover:underline\",\"children\":\"Events\"}]}]]}]]}]}]}]\n2c:I[834816,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"/_next/static/immutable/chunks/11cralgf1s2pn.js\",\"/_next/static/immutable/chunks/0cohu1fgg5_3o.js\",\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"/_next/static/immutable/chunks/2redqfc45k69c.js\",\"/_next/static/immutable/chunks/0xwnr3886mp47.js\"],\"default\"]\n18:[\"$\",\"footer\",null,{\"className\":\"bg-neutral-900 pt-16 pb-8\",\"children\":[\"$\",\"div\",null,{\"className\":\"container space-y-11\",\"children\":[[\"$\",\"div\",null,{\"className\":\"md:flex md:justify-between md:gap-8 lg:gap-10\",\"children\":[[\"$\",\"nav\",null,{\"className\":\"w-full\",\"data-galaxy-component\":\"footerNav\",\"children\":[\"$\",\"ul\",null,{\"className\":\"-mx-3 flex flex-row flex-wrap gap-y-8 lg:flex-nowrap\",\"children\":[[\"$\",\"li\",\"0\",{\"className\":\"flex w-1/2 flex-col px-3 lg:w-4/12\",\"children\":[\"$\",\"ul\",null,{\"children\":[[\"$\",\"$1\",\"0\",{\"children\":[[\"$\",\"li\",null,{\"className\":\"font-inter mb-4 text-sm font-bold text-neutral-100 \",\"children\":\"Product\"}],[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/cloud\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"clickHouseCloud\",\"data-galaxy-click-event\":\"footerNav.productMenu.clickHouseCloudSelect\",\"children\":\"ClickHouse Cloud\"}]}]]}],[\"$\",\"$1\",\"1\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/cloud/bring-your-own-cloud\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"bringYourOwnCloud\",\"data-galaxy-click-event\":\"footerNav.productMenu.byocSelect\",\"children\":\"Bring Your Own Cloud\"}]}]]}],[\"$\",\"$1\",\"2\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/cloud/postgres\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"clickHouseManagedPostgres\",\"data-galaxy-click-event\":\"footerNav.productMenu.postgresSelect\",\"children\":\"ClickHouse Managed Postgres\"}]}]]}],[\"$\",\"$1\",\"3\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/cloud/clickstack\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"managedClickStack\",\"data-galaxy-click-event\":\"footerNav.productMenu.managedClickstackSelect\",\"children\":\"Managed ClickStack\"}]}]]}],[\"$\",\"$1\",\"4\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/clickhouse\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"clickHouse\",\"data-galaxy-click-event\":\"footerNav.productMenu.clickHouseSelect\",\"children\":\"ClickHouse\"}]}]]}],[\"$\",\"$1\",\"5\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/clickstack\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"clickStack\",\"data-galaxy-click-event\":\"footerNav.productMenu.clickstackSelect\",\"children\":\"ClickStack\"}]}]]}],[\"$\",\"$1\",\"6\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/ai\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"agenticDataStack\",\"data-galaxy-click-event\":\"footerNav.productMenu.agenticDataStackSelect\",\"children\":\"Agentic Data Stack\"}]}]]}],[\"$\",\"$1\",\"7\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/government\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"clickHouseGovernment\",\"data-galaxy-click-event\":\"footerNav.productMenu.governmentSelect\",\"children\":\"ClickHouse Government\"}]}]]}],[\"$\",\"$1\",\"8\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/clickhouse/keeper\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"clickHouseKeeper\",\"data-galaxy-click-event\":\"footerNav.productMenu.keeperSelect\",\"children\":\"ClickHouse Keeper\"}]}]]}],[\"$\",\"$1\",\"9\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/cloud/clickpipes\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"clickPipes\",\"data-galaxy-click-event\":\"footerNav.productMenu.clickpipesSelect\",\"children\":\"ClickPipes\"}]}]]}],\"$L23\",\"$L24\",\"$L25\"]}]}],\"$L26\",\"$L27\",\"$L28\",\"$L29\"]}]}],\"$L2a\"]}],\"$L2b\"]}]}]\n19:[\"$\",\"$L2c\",null,{\"logoLink\":\"/\",\"minimal\":\"$undefined\",\"signInLabel\":\"Sign in\",\"signUpLabel\":\"Get Started\",\"navItems\":[{\"label\":\"Products\",\"children\":[{\"heading\":\"Products\",\"label\":\"ClickHouse Cloud\",\"description\":\"The best way to use ClickHouse.\\nAvailable on AWS, GCP, and Azure.\",\"href\":\"/cloud\",\"icon\":{\"src\":\"/_next/static/immutable/media/clickhouse-cloud.2rsb11bgis-95.svg\",\"width\":24,\"height\":25,\"blurWidth\":0,\"blurHeight\":0}},{\"label\":\"Bring Your Own Cloud\",\"description\":\"A fully managed ClickHouse service,\\ndeployed in your own AWS, GCP or Azure account.\",\"href\":\"/cloud/bring-your-own-cloud\",\"icon\":{\"src\":\"/_next/static/immutable/media/byoc.0h8f976g_o26-.svg\",\"width\":24,\"height\":24,\"blurWidth\":0,\"blurHeight\":0}},{\"label\":\"ClickHouse Managed Postgres\",\"description\":\"Unified data stack for transactions\\nand analytics.\",\"href\":\"/cloud/postgres\",\"icon\":{\"src\":\"/_next/static/immutable/media/postgres.16wzg_24jsz5p.svg\",\"width\":24,\"height\":24,\"blurWidth\":0,\"blurHeight\":0}},{\"label\":\"Managed ClickStack\",\"description\":\"Managed observability with high-performance\\nqueries and long-term retention.\",\"href\":\"/cloud/clickstack\",\"icon\":{\"src\":\"/_next/static/immutable/media/managed-clickstack.1wt2__tf4rz6d.svg\",\"width\":45,\"height\":33,\"blurWidth\":0,\"blurHeight\":0}},{\"label\":\"Langfuse Cloud\",\"description\":\"LLM observability and evaluations\\n for reliable AI applications and agents.\",\"href\":\"https://langfuse.com/?utm_source=clickhouse_topnav\",\"target\":\"_blank\",\"icon\":{\"src\":\"/_next/static/immutable/media/langfuse.09l1juwc1hfq6.svg\",\"width\":403,\"height\":270,\"blurWidth\":0,\"blurHeight\":0}},{\"heading\":\"Open source\",\"label\":\"ClickHouse\",\"description\":\"Fast open-source OLAP database for\\nreal-time analytics.\",\"href\":\"/clickhouse\",\"icon\":{\"src\":\"/_next/static/immutable/media/clickhouse.0lnnr6dkfmisn.svg\",\"width\":24,\"height\":24,\"blurWidth\":0,\"blurHeight\":0},\"newColumn\":true},{\"label\":\"ClickStack\",\"description\":\"Open-source observability stack for logs,\\nmetrics, traces, and session replays.\",\"href\":\"/clickstack\",\"icon\":{\"src\":\"/_next/static/immutable/media/clickstack.2evoafoj113lp.svg\",\"width\":42,\"height\":42,\"blurWidth\":0,\"blurHeight\":0}},{\"label\":\"Agentic Data Stack\",\"description\":\"Build AI-powered applications\\nwith ClickHouse.\",\"href\":\"/ai\",\"icon\":{\"src\":\"/_next/static/immutable/media/agentic-data-stack.295u5ndm0ogsy.svg\",\"width\":20,\"height\":20,\"blurWidth\":0,\"blurHeight\":0}},{\"label\":\"chDB\",\"description\":\"In-process SQL Engine powered by\\nClickHouse, with a Pandas-compatible API\",\"href\":\"/chdb\",\"icon\":{\"src\":\"/_next/static/immutable/media/chdb.3adhkawsfckoc.svg\",\"width\":24,\"height\":24,\"blurWidth\":0,\"blurHeight\":0}}]},{\"label\":\"Solutions\",\"children\":[{\"heading\":\"Use cases\",\"label\":\"Real-time analytics\",\"href\":\"/use-cases/real-time-analytics\"},{\"label\":\"Observability\",\"href\":\"/clickstack\"},{\"label\":\"Data warehousing\",\"href\":\"/use-cases/data-warehousing\"},{\"label\":\"Machine learning and GenAI\",\"href\":\"/use-cases/machine-learning-and-data-science\"},{\"label\":\"All use cases -\u003e\",\"href\":\"/use-cases\",\"slot\":\"footer\"},{\"heading\":\"Industries\",\"label\":\"Financial services\",\"href\":\"/industries/financial-services\",\"newColumn\":true},{\"label\":\"Cybersecurity\",\"href\":\"/industries/cybersecurity\"},{\"label\":\"Gaming and entertainment\",\"href\":\"/industries/gaming\"},{\"label\":\"E-commerce and retail\",\"href\":\"/industries/retail\"},{\"label\":\"Automotive\",\"href\":\"/industries/automotive\"},{\"label\":\"Energy\",\"href\":\"/industries/energy\"},{\"label\":\"All industries -\u003e\",\"href\":\"/industries\",\"slot\":\"footer\"}]},{\"label\":\"Docs\",\"href\":\"/docs\"},{\"label\":\"Resources\",\"children\":[{\"heading\":\"Company resources\",\"label\":\"User stories\",\"href\":\"/user-stories\"},{\"label\":\"Blog\",\"href\":\"/blog\"},{\"label\":\"Events\",\"href\":\"/company/events\"},{\"label\":\"News\",\"href\":\"/company/news\"},{\"label\":\"Learning and certification\",\"href\":\"/learn\"},{\"label\":\"Partners\",\"href\":\"/partners\"},{\"label\":\"Videos\",\"href\":\"/videos\"},{\"label\":\"Demos\",\"href\":\"/demos\"},{\"newColumn\":true,\"heading\":\"Comparisons\",\"label\":\"Benchmark hub\",\"href\":\"/benchmarks\"},{\"label\":\"CostBench\",\"href\":\"/benchmarks/costbench\"},{\"label\":\"BigQuery\",\"href\":\"/comparison/bigquery\"},{\"label\":\"PostgreSQL\",\"href\":\"/comparison/postgresql\"},{\"label\":\"Redshift\",\"href\":\"/comparison/redshift\"},{\"label\":\"Snowflake\",\"href\":\"/comparison/snowflake\"},{\"label\":\"Elastic Observability\",\"href\":\"/comparison/elastic-for-observability\"},{\"label\":\"Splunk\",\"href\":\"/comparison/splunk-for-observability\"},{\"label\":\"Datadog\",\"href\":\"/comparison/datadog-for-observability\"},{\"label\":\"OpenSearch\",\"href\":\"/comparison/opensearch-for-observability\",\"badge\":\"For observability\"},{\"label\":\"Databricks\",\"href\":\"/clickhouse-for-databricks\"}]},{\"label\":\"Pricing\",\"href\":\"/pricing\"},{\"label\":\"Contact us\",\"href\":\"/company/contact?loc=nav\"}]}]\n30:I[841718,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"/_next/static/immutable/chunks/3biu5182d-lhi.js\",\"/_next/static/immutable/chunks/0xwnr3886mp47.js\",\"/_next/static/immutable/chunks/1a9vq67yix_7x.js\",\"/_next/static/immutable/chunks/0tzu-v_27z1ur.js\",\"/_next/static/immutable/chunks/2redqfc45k69c.js\",\"/_next/static/immutable/chunks/0bvw80ze-pqm7.js\",\"/_next/static/immutable/chunks/0o8_760sa9oqt.js\",\"/_next/static/immutable/chunks/2uo20h-k3dnal.js\",\"/_next/static/immutable/chunks/2qvo7870d8e3t.js\",\"/_next/static/immutable/chunks/3cgheitv1dqkh.js\",\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"/_next/static/immutable/chunks/3ero29qvrrxr-.js\",\"/_next/static/immutable/chunks/200wl5-qkz6jp.js\",\"/_next/static/immutable/chunks/44m5mm7tejp8y.js\",\"/_next/static/immutable/chunks/0k5y-yyer9qd2.js\"],\"default\"]\n31:I[605500,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"/_next/static/immutable/chunks/3biu5182d-lhi.js\",\"/_next/static/immutable/chunks/0xwnr3886mp47.js\",\"/_next/static/immutable/chunks/1a9vq67yix_7x.js\",\"/_next/static/immutable/chunks/0tzu-v_27z1ur.js\",\"/_next/static/immutable/chunks/2redqfc45k69c.js\",\"/_next/static/immutable/chunks/0bvw80ze-pqm7.js\",\"/_next/static/immutable/chunks/0o8_760sa9oqt.js\",\"/_next/static/immutable/chunks/2uo20h-k3dnal.js\",\"/_next/static/immutable/chunks/2qvo7870d8e3t.js\",\"/_next/static/immutable/chunks/3cgheitv1dqkh.js\",\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"/_next/static/immutable/chunks/3ero29qvrrxr-.js\",\"/_next/static/immutable/chunks/200wl5-qkz6jp.js\",\"/_next/static/immutable/chunks/44m5mm7tejp8y.js\",\"/_next/static/immutable/chunks/0k5y-yyer9qd2.js\"],\"Image\"]\n32:I[646080,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"/_next/static/immutable/chunks/0a17ev_rxgc-h.js\",\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"/_next/static/immutable/chunks/2uo20h-k3dnal.js\",\"/_next/static/immutable/chunks/3ero29qvrrxr-.js\"],\"default\"]\n23:[\"$\",\"$1\",\"10\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/integrations\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"integrations\",\"data-galaxy-click-event\":\"footerNav.productMenu.integrationsSelect\",\"children\":\"Integrations\"}]}]]}]\n24:[\"$\",\"$1\",\"11\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/chdb\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"chDb\",\"data-galaxy-click-event\":\"footerNav.productMenu.chdbSelect\",\"children\":\"chDB\"}]}]]}]\n25:[\"$\",\"$1\",\"12\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/pricing\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"productFooterNav\",\"data-galaxy-event\":\"pricing\",\"data-galaxy-click-event\":\"footerNav.productMenu.pricingSelect\",\"children\":\"Pricing\"}]}]]}]\n26:[\"$\",\"li\",\"1\",{\"className\":\"flex w-1/2 flex-col px-3 lg:w-4/12\",\"children\":[\"$\",\"ul\",null,{\"children\":[[\"$\",\"$1\",\"0\",{\"children\":[[\"$\",\"li\",null,{\"className\":\"font-inter mb-4 text-sm font-bold text-neutral-100 \",\"children\":\"Resources\"}],[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://clickhouse.com/docs\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"documentation\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.docsSelect\",\"children\":\"Documentation\"}]}]]}],[\"$\",\"$1\",\"1\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://trust.clickhouse.com\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"trustCenter\",\"data-galaxy-click-event\":\"footerNav.productMenu.trustCenterSelect\",\"children\":\"Trust center\"}]}]]}],[\"$\",\"$1\",\"2\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/learn\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"training\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.trainingSelect\",\"children\":\"Training\"}]}]]}],[\"$\",\"$1\",\"3\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/support/program\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"support\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.supportSelect\",\"children\":\"Support\"}]}]]}],[\"$\",\"$1\",\"4\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/benchmarks\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"benchmarks\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.benchmarkHubSelect\",\"children\":\"Benchmarks\"}]}]]}],[\"$\",\"$1\",\"5\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/use-cases\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"useCases\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.useCasesSelect\",\"children\":\"Use cases\"}]}]]}],[\"$\",\"$1\",\"6\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/videos\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"videos\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.videosSelect\",\"children\":\"Videos\"}]}]]}],[\"$\",\"$1\",\"7\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/demos\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"demos\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.demosSelect\",\"children\":\"Demos\"}]}]]}],[\"$\",\"$1\",\"8\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://presentations.clickhouse.com/\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"presentations\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.presentationsSelect\",\"children\":\"Presentations\"}]}]]}],[\"$\",\"$1\",\"9\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/real-time-data-warehouse\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"realTimeDataWarehouse\",\"data-galaxy-click-event\":\"footerNav.productMenu.realtimeDWSelect\",\"children\":\"Real-time data warehouse\"}]}]]}],[\"$\",\"$1\",\"10\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/clickhouse-for-data-lakes\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"clickHouseForDataLakes\",\"data-galaxy-click-event\":\"footerNav.productMenu.dataLakesSelect\",\"children\":\"ClickHouse for data lakes\"}]}]]}],\"$L2d\",\"$L2e\"]}]}]\n27:[\"$\",\"li\",\"2\",{\"className\":\"flex w-1/2 flex-col px-3 lg:w-4/12\",\"children\":[\"$\",\"ul\",null,{\"children\":[[\"$\",\"$1\",\"0\",{\"children\":[[\"$\",\"li\",null,{\"className\":\"font-inter mb-4 text-sm font-bold text-neutral-100 \",\"children\":\"Company\"}],[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/blog\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"companyFooterNav\",\"data-galaxy-event\":\"blog\",\"data-galaxy-click-event\":\"footerNav.companyMenu.blogSelect\",\"children\":\"Blog\"}]}]]}],[\"$\",\"$1\",\"1\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/company/our-story\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"companyFooterNav\",\"data-galaxy-event\":\"ourStory\",\"data-galaxy-click-event\":\"footerNav.companyMenu.ourStorySelect\",\"children\":\"Our story\"}]}]]}],[\"$\",\"$1\",\"2\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/company/careers\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"companyFooterNav\",\"data-galaxy-event\":\"careers\",\"data-galaxy-click-event\":\"footerNav.companyMenu.careersSelect\",\"children\":\"Careers\"}]}]]}],[\"$\",\"$1\",\"3\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/company/contact?loc=footer\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"companyFooterNav\",\"data-galaxy-event\":\"contactUs\",\"data-galaxy-click-event\":\"footerNav.companyMenu.contactSelect\",\"children\":\"Contact us\"}]}]]}],[\"$\",\"$1\",\"4\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/company/events\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"companyFooterNav\",\"data-galaxy-event\":\"events\",\"data-galaxy-click-event\":\"footerNav.companyMenu.eventsSelect\",\"children\":\"Events\"}]}]]}],[\"$\",\"$1\",\"5\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/company/news\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"companyFooterNav\",\"data-galaxy-event\":\"news\",\"data-galaxy-click-event\":\"footerNav.companyMenu.newsSelect\",\"children\":\"News\"}]}]]}],[\"$\",\"$1\",\"6\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/media\",\"target\":\"_blank\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"companyFooterNav\",\"data-galaxy-event\":\"media\",\"data-galaxy-click-event\":\"footerNav.companyMenu.mediaSelect\",\"children\":\"Media\"}]}]]}]]}]}]\n28:[\"$\",\"li\",\"3\",{\"className\":\"flex w-1/2 flex-col px-3 lg:w-4/12\",\"children\":[\"$\",\"ul\",null,{\"children\":[[\"$\",\"$1\",\"0\",{\"children\":[[\"$\",\"li\",null,{\"className\":\"font-inter mb-4 text-sm font-bold text-neutral-100 \",\"children\":\"Join our community\"}],[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/community\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"joinOurCommunityFooterNav\",\"data-galaxy-event\":\"clickHouseCommunity\",\"data-galaxy-click-event\":\"footer.nav.clickhouseCommunity\",\"children\":\"ClickHouse Community\"}]}]]}],[\"$\",\"$1\",\"1\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://github.com/ClickHouse/ClickHouse\",\"target\":\"_blank\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"joinOurCommunityFooterNav\",\"data-galaxy-event\":\"gitHub\",\"data-galaxy-click-event\":\"footerNav.communityMenu.gitHubSelect\",\"children\":\"GitHub\"}]}]]}],[\"$\",\"$1\",\"2\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/slack\",\"target\":\"_blank\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"joinOurCommunityFooterNav\",\"data-galaxy-event\":\"slack\",\"data-galaxy-click-event\":\"footerNav.communityMenu.slackSelect\",\"children\":\"Slack\"}]}]]}],[\"$\",\"$1\",\"3\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://www.linkedin.com/company/clickhouseinc\",\"target\":\"_blank\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"joinOurCommunityFooterNav\",\"data-galaxy-event\":\"linkedIn\",\"data-galaxy-click-event\":\"footerNav.communityMenu.linkedInSelect\",\"children\":\"LinkedIn\"}]}]]}],[\"$\",\"$1\",\"4\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://x.com/ClickhouseDB\",\"target\":\"_blank\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"joinOurCommunityFooterNav\",\"data-galaxy-event\":\"x\",\"data-galaxy-click-event\":\"footerNav.communityMenu.twitterSelect\",\"children\":\"X\"}]}]]}],[\"$\",\"$1\",\"5\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://bsky.app/profile/clickhouse.com\",\"target\":\"_blank\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"joinOurCommunityFooterNav\",\"data-galaxy-event\":\"bluesky\",\"data-galaxy-click-event\":\"footerNav.communityMenu.blueSkySelect\",\"children\":\"Bluesky\"}]}]]}],[\"$\",\"$1\",\"6\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://telegram.me/clickhouse_en\",\"target\":\"_blank\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"joinOurCommunityFooterNav\",\"data-galaxy-event\":\"telegram\",\"data-galaxy-click-event\":\"footerNav.communityMenu.telegramSelect\",\"children\":\"Telegram\"}]}]]}],[\"$\",\"$1\",\"7\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"https://www.meetup.com/pro/clickhouse\",\"target\":\"_blank\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"joinOurCommunityFooterNav\",\"data-galaxy-event\":\"meetup\",\"data-galaxy-click-event\":\"footerNav.communityMenu.meetupSelect\",\"children\":\"Meetup\"}]}]]}]]}]}]\n29:[\"$\",\"li\",\"4\",{\"className\":\"flex w-1/2 flex-col px-3 lg:w-4/12\",\"children\":[\"$\",\"ul\",null,{\"children\":[[\"$\",\"$1\",\"0\",{\"children\":[[\"$\",\"li\",null,{\"className\":\"font-inter mb-4 text-sm font-bold text-neutral-100 \",\"children\":\"Comparisons\"}],[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/comparison/bigquery\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"comparisonsFooterNav\",\"data-galaxy-event\":\"bigQuery\",\"data-galaxy-click-event\":\"footerNav.comparisonsMenu.bigQuerySelect\",\"children\":\"BigQuery\"}]}]]}],[\"$\",\"$1\",\"1\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/comparison/postgresql\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"comparisonsFooterNav\",\"data-galaxy-event\":\"postgreSql\",\"data-galaxy-click-event\":\"footerNav.comparisonsMenu.postgresSelect\",\"children\":\"PostgreSQL\"}]}]]}],[\"$\",\"$1\",\"2\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/comparison/redshift\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"comparisonsFooterNav\",\"data-galaxy-event\":\"redshift\",\"data-galaxy-click-event\":\"footerNav.comparisonsMenu.redshiftSelect\",\"children\":\"Redshift\"}]}]]}],[\"$\",\"$1\",\"3\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/comparison/snowflake\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"comparisonsFooterNav\",\"data-galaxy-event\":\"snowflake\",\"data-galaxy-click-event\":\"footerNav.comparisonsMenu.snowflakeSelect\",\"children\":\"Snowflake\"}]}]]}],[\"$\",\"$1\",\"4\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/comparison/elastic-for-observability\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"comparisonsFooterNav\",\"data-galaxy-event\":\"elastic\",\"data-galaxy-click-event\":\"footerNav.comparisonsMenu.elasticSelect\",\"children\":\"Elastic\"}]}]]}],[\"$\",\"$1\",\"5\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/comparison/splunk-for-observability\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"comparisonsFooterNav\",\"data-galaxy-event\":\"splunk\",\"data-galaxy-click-event\":\"footerNav.comparisonsMenu.splunkSelect\",\"children\":\"Splunk\"}]}]]}],[\"$\",\"$1\",\"6\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/comparison/datadog-for-observability\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"comparisonsFooterNav\",\"data-galaxy-event\":\"datadog\",\"data-galaxy-click-event\":\"$undefined\",\"children\":\"Datadog\"}]}]]}],[\"$\",\"$1\",\"7\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/comparison/opensearch-for-observability\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"comparisonsFooterNav\",\"data-galaxy-event\":\"openSearch\",\"data-galaxy-click-event\":\"footerNav.comparisonsMenu.opensearchSelect\",\"children\":\"OpenSearch\"}]}]]}],[\"$\",\"$1\",\"8\",{\"children\":[[\"$\",\"li\",null,{\"className\":\"font-inter mb-4 text-sm font-bold text-neutral-100 mt-8\",\"children\":\"Partners\"}],[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/partners/aws\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"partnersFooterNav\",\"data-galaxy-event\":\"aws\",\"data-galaxy-click-event\":\"footerNav.partnersMenu.awsSelect\",\"children\":\"AWS\"}]}]]}],[\"$\",\"$1\",\"9\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/partners/azure\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"partnersFooterNav\",\"data-galaxy-event\":\"azure\",\"data-galaxy-click-event\":\"footerNav.partnersMenu.azureSelect\",\"children\":\"Azure\"}]}]]}],[\"$\",\"$1\",\"10\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/partners/gcp\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"partnersFooterNav\",\"data-galaxy-event\":\"googleCloud\",\"data-galaxy-click-event\":\"footerNav.partnersMenu.gcpSelect\",\"children\":\"Google Cloud\"}]}]]}]]}]}]\n2f:T9ae,M40.03 15.14q-.95 0-1.7.34-.76.33-1.3.98-.52.64-.81 1.56-.27.91-.27 2.07a7 7 0 0 0 .45 2.63q.45 1.1 1.35 1.7t2.27.59q.83 0 1.58-.15.78-.15 1.57-.41v1.67q-.75.3-1.55.42-.8.15-1.84.14-1.96 0-3.27-.81a5 5 0 0 1-1.95-2.3q-.65-1.5-.65-3.5 0-1.46.4-2.66.42-1.23 1.19-2.1a5 5 0 0 1 1.9-1.36 7 7 0 0 1 2.65-.48 8.5 8.5 0 0 1 3.6.79l-.72 1.62q-.62-.3-1.37-.5a5 5 0 0 0-1.53-.24m7.6 11.36h-1.91V12.82h1.9zm4.9-9.7v9.7h-1.9v-9.7zm-.94-3.7q.44.01.76.26t.32.85q0 .57-.32.84a1.2 1.2 0 0 1-.76.25q-.45 0-.79-.25-.3-.27-.3-.84 0-.6.3-.85.33-.25.8-.25m7.84 13.58q-1.34 0-2.34-.52a3.6 3.6 0 0 1-1.56-1.62 6 6 0 0 1-.56-2.83q0-1.8.6-2.91.6-1.13 1.63-1.64a5 5 0 0 1 2.38-.54q.81 0 1.5.18.73.15 1.2.38l-.58 1.54q-.51-.2-1.08-.34-.56-.15-1.06-.14-.9 0-1.5.4-.57.37-.86 1.15-.27.76-.27 1.9 0 1.1.29 1.86.3.75.84 1.15.58.38 1.43.38a5 5 0 0 0 2.57-.65v1.66q-.53.3-1.13.45t-1.5.14m6.81-7.02q0 .38-.04.86-.01.5-.05.9h.05l.78-.97q.2-.26.4-.47l2.96-3.18h2.22l-3.9 4.16 4.15 5.54h-2.25l-3.2-4.34-1.12.94v3.4h-1.89V12.82h1.9zm18.4 6.84H82.7v-5.87h-6.13v5.87h-1.95V13.65h1.95v5.33h6.13v-5.33h1.95zm11.78-4.86q0 1.2-.32 2.14a5 5 0 0 1-.92 1.59q-.6.65-1.44.99a5.3 5.3 0 0 1-3.71 0 4.1 4.1 0 0 1-2.38-2.58 6 6 0 0 1-.34-2.16q0-1.6.54-2.72.56-1.1 1.59-1.69 1.04-.6 2.44-.6 1.34 0 2.34.6 1.03.58 1.6 1.7.6 1.11.6 2.73m-7.15 0q0 1.08.27 1.87.28.78.85 1.19t1.48.41q.9 0 1.47-.41.58-.42.85-1.19.27-.8.27-1.87 0-1.11-.29-1.87-.27-.76-.85-1.15a2.4 2.4 0 0 0-1.47-.42q-1.36 0-1.96.9-.62.9-.62 2.54m17.92-4.84v9.7h-1.53l-.27-1.28h-.09q-.3.5-.8.83-.47.33-1.05.47-.58.16-1.2.16-1.13 0-1.92-.36a2.6 2.6 0 0 1-1.18-1.15 4.5 4.5 0 0 1-.4-2.02V16.8h1.92v6.06q0 1.14.47 1.7.5.55 1.5.55t1.58-.4q.59-.39.81-1.14.25-.78.25-1.86V16.8zm9.43 6.96q0 .96-.47 1.6a3 3 0 0 1-1.35 1q-.88.32-2.12.32-1.03 0-1.77-.16-.71-.15-1.33-.43v-1.7q.65.31 1.5.56a6 6 0 0 0 1.65.23q1.08 0 1.55-.34.5-.34.49-.92 0-.31-.18-.57a2 2 0 0 0-.69-.54q-.48-.3-1.44-.65a15 15 0 0 1-1.56-.74 3 3 0 0 1-1-.88q-.34-.53-.34-1.33 0-1.26 1.01-1.93a5 5 0 0 1 2.7-.68 7 7 0 0 1 3.19.68l-.63 1.46a7 7 0 0 0-1.75-.56 4 4 0 0 0-.9-.09q-.86 0-1.31.27a.8.8 0 0 0-.45.76q0 .34.2.6.21.24.73.5.53.25 1.43.6t1.53.71q.64.36.97.88.34.53.34 1.33m6.1-7.14q1.27 0 2.19.54.92.52 1.4 1.51.5.99.5 2.34v1.04h-6.51q.03 1.5.77 2.29.75.8 2.11.8a8 8 0 0 0 1.66-.17q.74-.19 1.5-.52v1.58a7 7 0 0 1-3.23.65q-1.41 0-2.49-.56a4 4 0 0 1-1.69-1.65q-.6-1.12-.6-2.74 0-1.65.55-2.77a4.1 4.1 0 0 1 3.83-2.34m0 1.47q-1.04 0-1.66.67-.62.66-.72 1.89h4.57q0-.76-.24-1.33-.23-.59-.72-.9a2.2 2.2 0 0 0-1.24-.332a:[\"$\",\"div\",null,{\"className\":\"mt-12 flex flex-col md:mt-0 md:w-fit\",\"children\":[[\"$\",\"svg\",null,{\"xmlns\":\"http://www.w3.org/2000/svg\",\"viewBox\":\"0 0 135 40\",\"width\":135,\"height\":40,\"fill\":\"currentColor\",\"role\":\"img\",\"aria-label\":\"ClickHouse\",\"className\":\"me-3 mb-4 text-white\",\"children\":[[\"$\",\"rect\",null,{\"width\":2.25,\"height\":20.25,\"x\":2.71,\"y\":9.88,\"rx\":0.24}],[\"$\",\"rect\",null,{\"width\":2.25,\"height\":20.25,\"x\":7.21,\"y\":9.88,\"rx\":0.24}],[\"$\",\"rect\",null,{\"width\":2.25,\"height\":20.25,\"x\":11.71,\"y\":9.88,\"rx\":0.24}],[\"$\",\"rect\",null,{\"width\":2.25,\"height\":20.25,\"x\":16.21,\"y\":9.88,\"rx\":0.24}],[\"$\",\"rect\",null,{\"width\":2.25,\"height\":4.5,\"x\":20.71,\"y\":17.75,\"rx\":0.24}],[\"$\",\"path\",null,{\"d\":\"$2f\"}]]}],[\"$\",\"div\",null,{\"className\":\"rich-text rich-text-light mb-4 text-sm\",\"children\":[[\"$\",\"p\",\"p-0\",{\"children\":\"Stay informed on feature releases, product roadmap, support, and cloud offerings!\"}]]}],[\"$\",\"$L30\",null,{\"formId\":\"1122\",\"disclaimer\":false,\"success\":\"Thanks for registering to our newsletter!\",\"clearbitTracking\":true}],[\"$\",\"$L22\",null,{\"href\":\"https://github.com/ClickHouse/ClickHouse\",\"target\":\"_blank\",\"className\":\"btn-secondary ms-auto mt-6\",\"children\":[[\"$\",\"$L31\",null,{\"src\":{\"src\":\"/_next/static/immutable/media/github.3ofxqpa2uzt_a.svg\",\"width\":32,\"height\":32,\"blurWidth\":0,\"blurHeight\":0},\"width\":16,\"height\":16,\"alt\":\"GitHub Logo\"}],\"Star us on Github\"]}]]}]\n2b:[\"$\",\"div\",null,{\"className\":\"flex flex-col items-start border-t border-neutral-400/10 pt-2 lg:pt-8\",\"data-galaxy-component\":\"legalFooter\",\"children\":[\"$\",\"div\",null,{\"className\":\"container flex w-full flex-col items-center gap-3 pt-4 text-center text-sm text-neutral-400 sm:gap-1 md:flex-row md:justify-between md:pt-0 md:text-start\",\"children\":[[\"$\",\"div\",null,{\"children\":[\"© \",2026,\" ClickHouse, Inc. HQ in the Bay Area, CA and Amsterdam, NL.\"]}],[\"$\",\"div\",null,{\"className\":\"flex flex-wrap items-center justify-center gap-4\",\"children\":[[[\"$\",\"$L22\",\"0\",{\"href\":\"/legal/trademark-policy\",\"target\":\"$undefined\",\"className\":\"whitespace-nowrap first:ps-0 hover:text-white\",\"data-galaxy-event\":\"trademark\",\"children\":\"Trademark\"}],[\"$\",\"$L22\",\"1\",{\"href\":\"/legal/privacy-policy\",\"target\":\"$undefined\",\"className\":\"whitespace-nowrap first:ps-0 hover:text-white\",\"data-galaxy-event\":\"privacy\",\"children\":\"Privacy\"}],[\"$\",\"$L22\",\"2\",{\"href\":\"https://trust.clickhouse.com/\",\"target\":\"_blank\",\"className\":\"whitespace-nowrap first:ps-0 hover:text-white\",\"data-galaxy-event\":\"security\",\"children\":\"Security\"}],[\"$\",\"$L22\",\"3\",{\"href\":\"/legal\",\"target\":\"$undefined\",\"className\":\"whitespace-nowrap first:ps-0 hover:text-white\",\"data-galaxy-event\":\"legal\",\"children\":\"Legal\"}],[\"$\",\"$L22\",\"4\",{\"href\":\"/legal/cookie-policy\",\"target\":\"$undefined\",\"className\":\"whitespace-nowrap first:ps-0 hover:text-white\",\"data-galaxy-event\":\"cookiePolicy\",\"children\":\"Cookie policy\"}]],[\"$\",\"$L32\",null,{}]]}]]}]}]\n2d:[\"$\",\"$1\",\"11\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/clickstack/ai-sre-observability\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"clickStackForAiSrEs\",\"data-galaxy-click-event\":\"footerNav.productMenu.clickstackAiSelect\",\"children\":\"ClickStack for AI SREs\"}]}]]}]\n2e:[\"$\",\"$1\",\"12\",{\"children\":[\"$undefined\",[\"$\",\"li\",null,{\"className\":\"my-1.5 leading-tight\",\"children\":[\"$\",\"$L22\",null,{\"href\":\"/resources/engineering\",\"target\":\"$undefined\",\"className\":\"text-sm text-neutral-400 hover:text-white\",\"data-galaxy-component\":\"resourcesFooterNav\",\"data-galaxy-event\":\"engineeringResources\",\"data-galaxy-click-event\":\"footerNav.resourcesMenu.engineeringResources\",\"children\":\"Engineering resources\"}]}]]}]\n1b:[[\"$\",\"meta\",\"0\",{\"charSet\":\"utf-8\"}],[\"$\",\"meta\",\"1\",{\"name\":\"viewport\",\"content\":\"width=device-width, initial-scale=1\"}]]\n"])</script><script>self.__next_f.push([1,"33:I[27201,[\"/_next/static/immutable/chunks/0esgme6slw2os.js\"],\"IconMark\"]\n17:null\n1d:[[\"$\",\"title\",\"0\",{\"children\":\"Can LLMs replace on call SREs today? | ClickHouse\"}],[\"$\",\"meta\",\"1\",{\"name\":\"description\",\"content\":\"We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found.\"}],[\"$\",\"meta\",\"2\",{\"name\":\"application-name\",\"content\":\"ClickHouse\"}],[\"$\",\"link\",\"3\",{\"rel\":\"manifest\",\"href\":\"/manifest.webmanifest\",\"crossOrigin\":\"$undefined\"}],[\"$\",\"meta\",\"4\",{\"name\":\"referrer\",\"content\":\"origin-when-cross-origin\"}],[\"$\",\"meta\",\"5\",{\"name\":\"creator\",\"content\":\"ClickHouse\"}],[\"$\",\"meta\",\"6\",{\"name\":\"publisher\",\"content\":\"ClickHouse\"}],[\"$\",\"meta\",\"7\",{\"name\":\"last-modified\",\"content\":\"2026-03-03T12:38:31.061Z\"}],[\"$\",\"link\",\"8\",{\"rel\":\"canonical\",\"href\":\"https://clickhouse.com/blog/llm-observability-challenge\"}],[\"$\",\"meta\",\"9\",{\"name\":\"format-detection\",\"content\":\"telephone=no, date=no, address=no, email=no, url=no\"}],[\"$\",\"meta\",\"10\",{\"property\":\"og:title\",\"content\":\"Can LLMs replace on call SREs today? | ClickHouse\"}],[\"$\",\"meta\",\"11\",{\"property\":\"og:description\",\"content\":\"We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found.\"}],[\"$\",\"meta\",\"12\",{\"property\":\"og:site_name\",\"content\":\"ClickHouse\"}],[\"$\",\"meta\",\"13\",{\"property\":\"og:locale\",\"content\":\"en_US\"}],[\"$\",\"meta\",\"14\",{\"property\":\"og:image\",\"content\":\"https://clickhouse.com/_next/image?url=%2Fuploads%2Fllm_observability_banner_22585788cf.png\u0026w=1200\u0026h=630\u0026q=75\"}],[\"$\",\"meta\",\"15\",{\"property\":\"og:type\",\"content\":\"article\"}],[\"$\",\"meta\",\"16\",{\"name\":\"twitter:card\",\"content\":\"summary_large_image\"}],[\"$\",\"meta\",\"17\",{\"name\":\"twitter:site\",\"content\":\"@ClickHouseDB\"}],[\"$\",\"meta\",\"18\",{\"name\":\"twitter:creator\",\"content\":\"@ClickHouseDB\"}],[\"$\",\"meta\",\"19\",{\"name\":\"twitter:title\",\"content\":\"Can LLMs replace on call SREs today? | ClickHouse\"}],[\"$\",\"meta\",\"20\",{\"name\":\"twitter:description\",\"content\":\"We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found.\"}],[\"$\",\"meta\",\"21\",{\"name\":\"twitter:image\",\"content\":\"https://clickhouse.com/_next/image?url=%2Fuploads%2Fllm_observability_banner_22585788cf.png\u0026w=1200\u0026h=630\u0026q=75\"}],[\"$\",\"meta\",\"22\",{\"name\":\"twitter:image:alt\",\"content\":\"Can LLMs replace on call SREs today?\"}],[\"$\",\"link\",\"23\",{\"rel\":\"icon\",\"href\":\"/favicon.ico?favicon.1-9m4wq8d-w30.ico\",\"sizes\":\"48x48\",\"type\":\"image/x-icon\"}],[\"$\",\"link\",\"24\",{\"rel\":\"icon\",\"href\":\"/icon0.svg?icon0.14ll1zwu0qk95.svg\",\"sizes\":\"any\",\"type\":\"image/svg+xml\"}],[\"$\",\"link\",\"25\",{\"rel\":\"icon\",\"href\":\"/icon1.png?icon1.3b4swr1c2xvk2.png\",\"sizes\":\"96x96\",\"type\":\"image/png\"}],[\"$\",\"link\",\"26\",{\"rel\":\"apple-touch-icon\",\"href\":\"/apple-icon.png?apple-icon.18ftzencg_4sl.png\",\"sizes\":\"180x180\",\"type\":\"image/png\"}],[\"$\",\"$L33\",\"27\",{}]]\n"])</script><script>self.__next_f.push([1,"34:I[893372,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"/_next/static/immutable/chunks/3biu5182d-lhi.js\",\"/_next/static/immutable/chunks/0xwnr3886mp47.js\",\"/_next/static/immutable/chunks/1a9vq67yix_7x.js\",\"/_next/static/immutable/chunks/0tzu-v_27z1ur.js\",\"/_next/static/immutable/chunks/2redqfc45k69c.js\",\"/_next/static/immutable/chunks/0bvw80ze-pqm7.js\",\"/_next/static/immutable/chunks/0o8_760sa9oqt.js\",\"/_next/static/immutable/chunks/2uo20h-k3dnal.js\",\"/_next/static/immutable/chunks/2qvo7870d8e3t.js\",\"/_next/static/immutable/chunks/3cgheitv1dqkh.js\",\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"/_next/static/immutable/chunks/3ero29qvrrxr-.js\",\"/_next/static/immutable/chunks/200wl5-qkz6jp.js\",\"/_next/static/immutable/chunks/44m5mm7tejp8y.js\",\"/_next/static/immutable/chunks/0k5y-yyer9qd2.js\"],\"default\"]\n36:I[796949,[\"/_next/static/immutable/chunks/2pakapa8rw53j.js\",\"/_next/static/immutable/chunks/0f884239kz0zf.js\",\"/_next/static/immutable/chunks/0mugrkwyg2zkw.js\",\"/_next/static/immutable/chunks/3_nn80b_suiek.js\",\"/_next/static/immutable/chunks/3biu5182d-lhi.js\",\"/_next/static/immutable/chunks/0xwnr3886mp47.js\",\"/_next/static/immutable/chunks/1a9vq67yix_7x.js\",\"/_next/static/immutable/chunks/0tzu-v_27z1ur.js\",\"/_next/static/immutable/chunks/2redqfc45k69c.js\",\"/_next/static/immutable/chunks/0bvw80ze-pqm7.js\",\"/_next/static/immutable/chunks/0o8_760sa9oqt.js\",\"/_next/static/immutable/chunks/2uo20h-k3dnal.js\",\"/_next/static/immutable/chunks/2qvo7870d8e3t.js\",\"/_next/static/immutable/chunks/3cgheitv1dqkh.js\",\"/_next/static/immutable/chunks/30m1qt8efl7us.js\",\"/_next/static/immutable/chunks/3ero29qvrrxr-.js\",\"/_next/static/immutable/chunks/200wl5-qkz6jp.js\",\"/_next/static/immutable/chunks/44m5mm7tejp8y.js\",\"/_next/static/immutable/chunks/0k5y-yyer9qd2.js\"],\"default\"]\n35:T4b7,[{\"@context\":\"https://schema.org\",\"@type\":\"BlogPosting\",\"headline\":\"Can LLMs replace on call SREs today?\",\"description\":\"We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found.\",\"image\":\"/uploads/llm_observability_banner_22585788cf.png\",\"publisher\":{\"@type\":\"Organization\",\"name\":\"ClickHouse\",\"url\":\"https://clickhouse.com/\",\"logo\":{\"@type\":\"ImageObject\",\"url\":\"https://clickhouse.com/_next/static/immutable/media/icon1.3b4swr1c2xvk2.png\"}},\"datePublished\":\"2025-08-13T16:18:03.962Z\",\"dateModified\":\"2026-03-03T12:38:31.061Z\",\"author\":[{\"@context\":\"https://schema.org\",\"@type\":\"Person\",\"name\":\"Lionel Palacin\",\"url\":\"https://clickhouse.com/authors/lionel-palacin\",\"@id\":\"https://clickhouse.com/authors/lionel-palacin#person\",\"image\":\"/uploads/lio_headshot_singapore_7cc9852011.jpg\",\"jobTitle\":\"Senior Product Marketing Engineer at ClickHouse\"},{\"@context\":\"https://schema.org\",\"@type\":\"Person\",\"name\":\"Al Brown\",\"url\":\"https://clickhouse.com/authors/al-brown\",\"@id\":\"https://clickhouse.com/authors/al-brown#person\",\"image\":\"/uploads/al_brown_headshot_09ae0cbce6.jpg\",\"jobTitle\":\"Product Marketing Engineer at ClickHouse\"}]}]37:T1a2c0,\u003cstyle\u003e\n.llm-snippet {\n background-color: #1e1e1e; /* dark gray background */\n border: 1px solid #333; /* subtle border */\n border-radius: 6px; /* rounded corners */\n padding: 1rem 1.25rem; /* comfortable spacing */\n font-size: 0.8rem;\n line-height: 1.5;\n color: #d4d4d4; /* light gray text */\n overflow-x: auto; /* horizontal scroll if needed */\n white-space: pre-wrap; /* preserve formatting but wrap text */\n}\n\n.rich_content details.llm p {\n margin-bottom: 0.5rem;\n margin-top: 0.5rem;\n}\n\nh4 {\nfont-size: 18px\n}\n\nh5 {\nfont-size: 16px\n}\n\u003c/style\u003e\n\nThere's a growing belief that AI-powered observability will soon reduce or even replace the role of Site Reliability Engineers (SREs). That's a bold claim and, at ClickHouse, we were curious to see how close we actually are.\n\nFor that we picked one particular task SRE does, among [many other things](https://en.wikipedia.org/wiki/Site_reliability_engineering), which is the Root cause analysis, and ran an experiment to see how good they are at doing this task on their own.\n\nKeep on reading to learn how we set up and ran the experiment, and more importantly what we learnt from it.\n\n\u003e **TL;DR** Autonomous RCA is not there yet. The promise of using LLMs to find production issues faster and at lower cost fell short in our evaluation, and even GPT-5 did not outperform the others. \n\u003e \n\u003e Use LLMs to assist investigations, summarize findings, draft updates, and suggest next steps while engineers stay in control with a fast, searchable observability stack.\n\n## The experiment: LLM to handle RCA with naive prompt\n\nThe experiment is straightforward. We gave a model access to observability data from a live application and, with a naive prompt, asked it to identify the root cause of a user-reported anomaly.\n\n### The contenders\n\nTo run the experiment, first we needed contenders, so we picked five models. Some are well-known, others are less common but promising:\n\n- **Claude Sonnet 4**: Claude is known for its structured reasoning and detailed responses. It tends to be good at walking through steps logically, which could help when tracing complex issues across systems.\n\n- **OpenAI GPT-o3**: A more advanced model from OpenAI, built for speed and multimodal input. It balances performance with fast response times.\n\n- **OpenAI GPT-4.1**: Still strong when it comes to general reasoning and language understanding, but slower and not quite as sharp as GPT-o3. A good baseline for comparison.\n\n- **Gemini 2.5 Pro**: Google’s latest Pro model, integrated with their ecosystem. It does well with multi-step reasoning and has shown strong performance on code and troubleshooting tasks.\n\nWe didn’t test every model out there, first because it’s an impossible task, but rather the ones we had easy access to and that seemed viable for the task. The idea is not to crown a winner, but to see how they each handle the reality of real-world incident data.\n\n### The anomalies\n\nThen we need observability data from a live application that contains an issue. We chose to run the [OpenTelemetry demo application](https://opentelemetry.io/docs/demo/) to generate four datasets containing unique anomalies.\n\nAs you can see in the architecture diagram below, the OpenTelemetry demo application is fairly complex, contains a lot of different services, a frontend users can interact with and a load generator. In brief, it's a good representation of a real-world application. Also the demo application comes out of the box with pre-canned anomalies that can be turned on using [feature flags](https://opentelemetry.io/docs/demo/feature-flags/).\n\n\n\nUsing that application, we created three new datasets, each containing a distinct anomaly and data covering a 1 hour period. For the fourth test, we used our [existing ClickStack public demo](https://play-clickstack.clickhouse.com/) dataset which covers a 48 hour period.\n\nThe table below summarizes the four datasets we have and for each of them what feature flags were used.\n\n| Name | Database | Duration | Feature flag | Description |\n|---------------|-----------------|----------|---------------------------|-------------------|\n| Anomaly 1 | otel_anomaly_1 | 1h | paymentFailure | Generate an error when calling the charge method. This affects only users with loyalty status = gold. |\n| Anomaly 2 | otel_anomaly_2 | 1h | recommendationCacheFailure| Create a memory leak due to an exponentially growing cache. 1.4x growth, 50% of requests trigger growth. |\n| Anomaly 3 | otel_anomaly_3 | 1h | productCatalogFailure | Generate an error for GetProduct requests with product ID: OLJCESPC7Z |\n| Demo anomaly | otel_v2 | 48h | paymentCacheLeak | Generate a memory leak in the payment cache service that slows down the application once the cache is full. |\n\nTo capture the datasets, we used the following methodology. \n\nFirst, deploy the OpenTelemetry demo application and [instrument it](https://clickhouse.com/docs/use-cases/observability/clickstack/ingesting-data/overview) using ClickStack in Kubernetes. This [repository](https://github.com/ClickHouse/opentelemetry-demo) contains instructions on how to do so.\n\nOnce the application runs and the telemetry data starts to flow into ClickHouse, we increase the load to 1000 users. After the number of users reaches the desired target, we turn the feature flag on and let it run for about 40 minutes. Finally, we turn off the feature flag and let it run for another 10 minutes. The whole dataset captured now contains about 1 hour's worth of data.\n\n\u003cpre\u003e\n\u003ccode type='click-ui' language='sql' show_line_numbers='false'\u003e\nSELECT\n min(TimestampTime),\n max(TimestampTime)\nFROM otel_anomaly_3.otel_logs\n\u003c/code\u003e\n\u003c/pre\u003e\n\n\u003cpre\u003e\u003ccode type='click-ui' language='text' show_line_numbers='false'\u003e\n ┌──min(TimestampTime)─┬──max(TimestampTime)─┐\n│ 2025-07-22 08:25:40 │ 2025-07-22 09:36:01 │ \n└─────────────────────┴─────────────────────┘\n\u003c/code\u003e\u003c/pre\u003e\n\nNow we have our datasets captured, let’s investigate the issues using first a manual investigation then test the different models to see which one shines.\n\n## How we ran the experiment\n\n### Manual investigation\n\nBefore we ask the LLMs to find the problem, we need to know the actual root cause ourselves. That means doing a proper manual investigation for each anomaly, just like any SRE would. This gives us a baseline: if a model gets it right, we'll know. If it makes a mistake, or goes in the wrong direction, we'll be able to step in and guide it. \n\nFor this part, we use [ClickStack](https://clickhouse.com/use-cases/observability), our observability stack built on top of ClickHouse. \n\nWe go through each issue manually, confirm what's wrong, and document the path we took. That becomes the reference point for comparing what the models do.\n\n### AI-powered investigation\n\nThen it's the models' turn.\n\nWe run each LLM through the same scenario using [LibreChat](https://www.librechat.ai/) connected to a [ClickHouse MCP server](https://clickhouse.com/docs/use-cases/AI/MCP/librechat). This lets the models query real observability data and try to figure things out on their own.\n\nWe [instrument LibreChat and the ClickHouse MCP Server to track tokens usage](https://clickhouse.com/blog/llm-observability-clickstack-mcp). Since the telemetry data is stored in ClickHouse, we can run the following query to obtain the number of tokens used throughout an investigation.\n\n\u003cpre\u003e\n\u003ccode type='click-ui' language='sql' show_line_numbers='false'\u003e\nSELECT\n LogAttributes['conversationId'] AS conversationId,\n sum(toUInt32OrZero(LogAttributes['completionTokens'])) AS completionTokens,\n sum(toUInt32OrZero(LogAttributes['tokenCount'])) AS tokenCount,\n sum(toUInt32OrZero(LogAttributes['promptTokens'])) AS promptTokens,\n anyIf(LogAttributes['text'], (LogAttributes['text']) != '') AS prompt,\n min(Timestamp) AS start_time,\n max(Timestamp) AS end_time\nFROM otel_logs\nWHERE conversationId = \u003clibrechat_conversation_id\u003e\nGROUP BY conversationId\n\u003c/code\u003e\n\u003c/pre\u003e\nWe start each test with the same naive prompt:\n\n*\"You're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\"*\n\nIf the model gets it right after the first prompt, great. If not, we ask follow up questions either based on the response it provided, or if it is completely off track, we provide additional context to help it get to a resolution.\n\nThen for each anomaly, we report on:\n\n- What the model finds\n- How accurate it is\n- How much guidance it requires\n- How many tokens it uses getting there\n- How long it takes to run the investigation\n\nThis gives us a sense of how efficient and reliable each model is when dropped into a real-world SRE task.\n\nExperiment walkthrough\n----------------------\n\nIn this section, we go through each anomaly and first document the manual investigation, then for each model we start with our simple prompt and document the model's finding. If the model is not successful at diagnosing the issue right away, we provide additional prompts to guide it. \n\n### Anomaly 1: Payment service failure\n\nUsers are reporting issues during the checkout process, after filling in all the information related to the order, clicking on place order leads to an error. \n\n#### Manual investigation\n\nThis is a simple issue to diagnose manually in ClickStack. We start by looking at one of the client sessions containing an error.\n\n\n\nFrom that specific session, we look at the trace view, where we can see an error message on the payment service. The message indicates that there is an invalid token for the loyalty level gold, which causes the payment request to fail. \n\n\n\nWe navigate to the traces explorer view, and filter the dashboard for the payment service.\n\n\n\nThis validates our finding, **the payment service has issues when processing requests from clients with a Gold loyalty status.**\n\n#### LLM investigation\n\n##### Claude Sonnet 4 model\n\nThe model managed to find the RCA after the first prompt.\n\nExpand below to see the full interaction.\n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eClaude Sonnet 4 anomaly 1 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eOpenTelemetry Analysis Report\n\nIssues Identified\n\n1. Primary Issue: Payment Service Gold Tier Failure\n\n- What: Starting at 11:52:35 on July 22, 2025, all payment transactions for users with \"gold\" loyalty level began failing with \"Invalid token\" errors\n- Impact: 50% payment failure rate (13,505 failed transactions out of 27,000 total payment attempts)\n- Error Message: \"Payment request failed. Invalid token. app.loyalty.level=gold\"\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### OpenAI o3 model \n\nThe model managed to find the root cause and document it after the first prompt.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eOpenAI o3 anomaly 1 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eEvidence\n\t1. Traces\n– Payment service: 13,512 error spans, all named\n\"Error: Payment request failed. Invalid token. app.loyalty.level=gold\"\n– Checkout service: 27,024 error spans (same TraceId chain) with gRPC method oteledemo.PaymentService/Charge returning INTERNAL.\n– Frontend / frontend-proxy: ~29,000 error spans (\"Failed to place order\").\n\t2. Logs\n– 27,010 \"failed to charge card: ... Invalid token ... loyalty.level=gold\" log records.\n– Small number (~6,000) of ECONNREFUSED errors, but they are an order of magnitude lower than the “Invalid token” failures.\n\t3. Metrics\n– No CPU / memory anomalies; latency spikes coincide with spikes in payment errors.\n– Error rate is almost 100% for PlaceOrder when loyalty.level = gold.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### OpenAI GPT4.1 model \n\nInitially, the model didn’t query the data in ClickHouse via the MCP server. After a follow-up prompt, it queried ClickHouse and analyzed the data but couldn’t identify the root cause. It focused on the trace’s parent spans, which were impacted by the issue but not the actual source. The model only found the root cause after we gave it direct guidance.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eOpenAI GPT4.1 anomaly 1 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eTo help you identify the issue, root cause, and suggest solutions, I’ll need to analyze relevant OpenTelemetry data from your demo application. This typically includes traces, metrics, and logs. Here’s how we’ll approach this:\n\t1. Issue Identification: I’ll look for error traces, high latencies, dropped requests, or anomalies.\n\t2. Root Cause Analysis: I’ll analyze the traces to see where bottlenecks, errors, or resource issues might exist (e.g., slow database calls, timeouts, dependency errors).\n\t3. Potential Solutions: Based on findings, I’ll suggest actionable solutions.\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou have access to the data, can you start analyzing?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eObserved Issue\n- High average span durations in several services, especially load-generator, frontend-proxy, and frontend.\n- No explicit errors in logs or trace statuses.\n- This points toward a potential performance/bottleneck issue rather than a code crash or failure.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 3:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eUsers have been reported issues during the payment process\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3 (snippet):\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eImpact\n- Users, particularly with gold loyalty level, cannot complete payment, resulting in user-facing failures and high error rates in backend logs and traces.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### Gemini 2.5 Pro\n\nTo begin, Gemini made several errors trying to use the MCP tools to list databases. After the initial prompt, Gemini went in the wrong direction and focused on latency numbers of a specific API that was not causing a problem. After prompting it to look in the direction of the payment process, Gemini correctly identified the gold status issue.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eGemini 2.5 Pro anomaly 1 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003e### Issue\n\nThe application is experiencing high latency, with the `frontend-proxy` service being the most affected. The P95 latency for this service is over 1 second, which is significantly impacting the user experience.\n\n### Root Cause\n\nThe root cause of the high latency is long-running calls from the `frontend-proxy` service to the `flagd.evaluation.v1.Service/EventStream` endpoint of the `flagd` service. The trace analysis shows that these calls can take over 3 minutes to complete. This is because the `EventStream` is a streaming endpoint designed to provide real-time updates of feature flags. The `frontend-proxy` is making a request to this endpoint and holding the connection open, which is causing the long-running spans and, consequently, the high latency.\n\nThe `canceled: \"true\"` attribute on the `router flagservice egress` span in the `frontend-proxy` service suggests that the proxy is timing out and canceling the request. However, the `flagd` service does not seem to be handling this cancellation correctly and continues to process the request, keeping the stream open for an extended period.\n\n### Solution\n\nTo resolve this issue, the following actions are recommended:\n\n* **Use a Short-Lived RPC for Feature Flag Evaluation**: Instead of using the `EventStream` endpoint, the `frontend-proxy` should use a more appropriate, short-lived RPC for fetching feature flags. This will ensure that the connections are not held open for extended periods, reducing the latency of the `frontend-proxy` service.\n* **Investigate the `flagd` Service**: The `flagd` service should be investigated to understand why it's not handling client-side cancellations correctly. The developers should ensure that the service gracefully handles canceled requests and closes the streams promptly.\n* **Review the `frontend-proxy` Configuration**: The `frontend-proxy`'s configuration should be reviewed to ensure that it's correctly handling streaming responses and that the timeouts are set to appropriate values.\n\nBy implementing these solutions, you should be able to significantly reduce the latency of the `frontend-proxy` service and improve the overall performance and reliability of your application.\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eUsers have reported issues during the payment process\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eBased on my investigation, I've identified the root cause of the payment issues and have recommendations for a solution.\n\n### Summary of Findings\n\n* **No Explicit Errors:** My initial search for traces and logs with an \"ERROR\" status for the `payment` service came up empty. This indicates that the application is not correctly flagging these issues as errors.\n* **Anomalous Span Names:** An analysis of the service's trace data revealed a peculiar pattern: spans being created with names that are actually error messages. For example:\n * `Error: Payment request failed. Invalid token. app.loyalty.level=gold`\n * `Error: The credit card (ending 1791) expired on 1/2025.`\n* **Error Context:** By examining these \"error spans,\" I found that the \"Invalid token\" error consistently occurs for users with the `app.loyalty.level` attribute set to `gold`.\n\n### Root Cause\n\nThe user-reported payment issues are caused by a bug in the `payment` service that specifically affects customers with a \"gold\" loyalty status. When these users attempt to make a payment, the service fails with an \"Invalid token\" error.\n\nFurthermore, the application is handling these errors incorrectly from an observability standpoint. Instead of setting the status of the parent span (e.g., the `charge` or `grpc.oteldemo.PaymentService/Charge` span) to `ERROR`, the developers have written code that creates a new, zero-duration span with the error message as its name. This practice hides the errors from standard monitoring dashboards and makes it difficult to track the error rate of the service.\n\n### Recommended Solutions\n\n1. **Fix the Payment Logic:**\n * **Immediate Priority:** The development team must investigate and fix the bug in the `payment` service that causes token validation to fail for \"gold\" loyalty members. This is the primary cause of the user-facing issue.\n\n2. **Correct the Instrumentation:**\n * **Best Practice:** The practice of creating new spans to log errors should be stopped. Instead, developers should adhere to OpenTelemetry best practices:\n * When an error occurs, the `StatusCode` of the existing span in which the error happened should be set to `ERROR`.\n * The error details should be recorded as an `Event` (also known as a `Log`) within that span, including attributes for the exception type, message, and stack trace. This ensures that errors are properly tracked, correlated with traces, and can be used to trigger alerts.\n\nBy addressing both the underlying bug in the payment logic and the incorrect instrumentation, you will resolve the user-reported issues and significantly improve the observability and reliability of your application.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n#### LLM investigation summary \n\n| Model | Resolution | # of MCP calls | Duration | Input tokens | Output tokens | Cost |\n|-------|------------|----------------------|----------|--------------|---------------|------|\n| Claude 4 sonnet | Yes | 15 | 2 minutes | 1028123 | 4487 | $3.15 |\n| OpenAI o3 | Yes | 15 | 2 minutes | 57397 | 2845 | $0.17 |\n| OpenAI GPT4.1 | Yes, with minor guidance | 14 | 3 minutes | 43479 | 2224 | $0.10 |\n| Gemini 2.5 Pro | Yes, with minor guidance | 12 | 3 minutes | 313892 | 7451 | $0.90 |\n\n### Anomaly 2: Recommendation cache leak\n\nAn issue on the recommendation cache service has been introduced which causes the service CPU usage to spike. \n\n#### Manual investigation\n\nLet’s start the investigation from the logs, we can see an increase in error messages.\n\n\n\nScrolling through them, it’s not immediately obvious what the root cause is. We can see multiple error with a message about connection issue: `⨯ Error: 14 UNAVAILABLE: No connection established. Last error: connect ECONNREFUSED 34.118.225.39:8080 (2025-07-22T11:05:44.834Z)`\n\nLet’s filter the trace explorer view to select only the traces in error and exclude the load-generator service as it is creating a lot of noise. \n\n\n\nWe see errors related to the recommendation service. Let’s filter only traces from the recommendation service.\n\n\n\nIt seems there is a drop in traces during the experiment. Let’s validate that with the request throughput for the recommendation service using the Services view. \n\n\n\nThroughput for the recommendation service dropped, while latency increased. This suggests the service is still responding but much more slowly, likely causing bottlenecks that lead to timeouts or connectivity errors.\n\nLet’s return to the Traces and see if we can identify any common patterns among the traces using the Event deltas. \n\n\n\nOn the Event deltas screen, filtered by the get_product_list function, we can use the outliers attribute to compare fast and slow requests.\n\nIn this case, the outlier requests (slow requests) tend to return more than 10 products and have the recommendation cache enabled.\n\n\n\nIn requests where the cache is enabled, we observed that when the number of products is high and if the cache misses, the request duration increases.\n\n**The issue is related to the recommendation cache and the volume of selected products.**\n\n#### LLM investigation\n\n##### Claude Sonnet 4 model\n\nThe model was able to localize the root cause to the recommendation service but didn’t initially identify the cache issue.\n\nIts progress was limited. It correctly observed that the recommendation service was failing due to resource exhaustion and noted that the service had restarted 12 times—something we hadn’t uncovered during our manual analysis.\n\nIt also identified that the restarts were caused by hitting the CPU limit, not a memory problem. While that’s a helpful detail, it still missed the actual cache-related issue. \n\n\u003e This reflects a common pattern: the model tends to lock onto a single line of reasoning and doesn’t explore other possibilities.\n\nWhen we followed up by asking why the pod was using 100% CPU, the model finally pinpointed a problem with how the recommendation service cache was behaving.\n\nExpand below to see the full interaction.\n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eClaude Sonnet 4 anomaly 2 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eThe Recommendation Service experienced a complete outage, causing cascading failures throughout the application:\n\t1. Primary Issue:\nThe recommendation service became completely unavailable between 10:13–11:05 UTC\n\t• Normal span count dropped from 14,000 per 5-min window to 300–400\n\t• Service was essentially down for nearly an hour\n\t2. Connection Failures:\nFrontend services couldn’t connect to recommendation service endpoints:\n\t• IP addresses being refused: 34.118.225.39:8080, 34.118.236.87:8080, 34.118.226.49:8080\n\t• 295,538 error logs from frontend service with \"ECONNREFUSED\" errors\n\t• 51,190 failed gRPC calls to RecommendationService/ListRecommendations\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhy does the RecommendationService/ListRecommendations fail?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2 (snippet 1/2):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot Cause Analysis: Why RecommendationService/ListRecommendations Fails\n\nBased on my deep analysis of the observability data, here’s exactly why the RecommendationService/ListRecommendations is failing:\n\nThe Core Problem: Resource Exhaustion Leading to Cascading Failures\n\nPrimary Root Cause:\nThe recommendation service is experiencing resource starvation in its Kubernetes environment, leading to repeated container restarts.\n\u003c/div\u003e\u003c/pre\u003e\n\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 2 (snippet 2/2):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRestart Pattern:\n\t• 12 service restarts in just 1 hour (10:13–11:07)\n\t• Restart frequency: Every 3–7 minutes\n\t• Restart timestamps show clear crash–restart cycles\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 3:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhy do you think the recommendation service restarted?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhy the Recommendation Service Restarted – The Complete Analysis\n\nBased on my deep dive into the observability data, I can now provide the exact reason for the recommendation service restarts:\n⸻\n\nRoot Cause: CPU Throttling Leading to Kubernetes Health Check Failures\n\nThe Primary Trigger:\nThe recommendation service is being killed by Kubernetes due to failed health checks, not by memory issues (OOMKilled).\n⸻\n\nEvidence-Based Analysis:\n\nRestart Pattern Analysis:\n\t• 12 restarts in 54 minutes (10:13:40 – 11:07:23)\n\t• Restart intervals: 54s, 96s, 77s, 179s, 248s, 436s, 411s, 488s, 414s, 400s, 420s\n\t• Pattern: Initial rapid restarts (54–96s), then gradually increasing intervals as system stabilizes\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 4:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhy is the pod using 100% CPUs?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 4 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot Cause Analysis:\n\nThe Problem: Algorithmic Complexity Explosion\nWhat’s happening inside the recommendation service:\n\t1. Cache Warming Issue:\nAt 10:12, something changes in the request patterns.\n\t2. Cold Cache Performance:\nWhen cache misses happen, the recommendation algorithm becomes computationally expensive.\n\t3. Product List Processing:\nThe get_product_list function is doing heavy computation.\n\t4. Cascade Effect:\nAs processing time increases, requests queue up, making everything slower.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### OpenAI o3 model\n\nThe model identified that the recommendation service was unavailable but didn’t explain why. When we asked for more details, it didn’t provide any useful insight. We then rephrased the question slightly. This time, the model discovered that the pod had crashed, though it couldn’t determine the exact cause. It suggested either a memory issue or a feature flag—both in the right direction.\n\nWe pressed further on the memory angle, which led the model to identify the cache-enabled flag as the likely root cause. However, when proposing a fix, the model generated a solution that wasn’t valid. It suggested code that doesn’t exist in the application, likely because it has no access to the source code.\n\nExpand below to see the full interaction.\n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eOpenAI o3 anomaly 2 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot cause (most likely)\n\nThe Recommendation Service is down or unreachable from the cluster:\n\t• Pod(s) may be CrashLooping or scaled to 0\n\t• Service/Endpoint list may be empty or pointing to the wrong port\n\t• Container may start but not bind to 0.0.0.0:8080 (misconfiguration, wrong PORT env-var)\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhy does the RecommendationService/ListRecommendations fail?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eIn short:\nListRecommendations fails whenever traffic is routed to an orphan / wrong IP (34.118.225.39).\nClean up that bad endpoint and the errors disappear.\n\u003c/div\u003e\u003c/pre\u003e\n\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 3:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eCan you check before the recommendation service became unavailable what could have caused to crash?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhat probably caused the crash\n\n– Memory leak or a code path that loads a huge product list into RAM\n(note the 424 MiB / 332 MiB jumps right after fetching the product catalogue)\n– A recent feature-flag or config change that increases the amount of data the service keeps in memory on start-up\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 4:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003e\"note the 424 MiB/332 MiB jumps right after fetching the product catalogue\" This is interesting. Can you check why fetching the product catalogue increased the memory a lot?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 4 (snippet 1/2):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003e10:12 – flag flips → cache enabled\n\t• ResolveBoolean/flagd traces appear for key \"recommendation.cache_enabled\"\n\t• Subsequent get_product_list spans show cache_enabled = true\n\t• In the very first minute after the flip, the value of app.products.count jumps from 10 to 20,226,406 (≈ 2 × 10⁷)\n\t• container.memory.usage spikes to 830 MiB and the pod is OOM-killed\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 4 (snippet 2/2):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhat the numbers tell us\n\nThe code that builds the cached product list is appending the entire list to itself on every cache miss / cold start instead of replacing it:\nnew_list = cache.get() # [] on first run \nnew_list += remote_product_list # +10 \ncache.set(new_list)\n\nAfter a restart, the cache is empty again, so the first call fetches 10 products and stores them.\n\nBut a second call (still within the same process) does:\nnew_list = cache.get() # 10 items \nnew_list += remote_product_list # +10 ⇒ 20 \ncache.set(new_list)\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### OpenAI GPT4.1 \n\nThe model struggles to identify the root cause on its own. It doesn’t query data through the MCP tool unless explicitly instructed to do so. By default, it focuses only on logs and needs additional prompts to include traces and metrics in its analysis.\n\nOnce it reviews all the data, it flags the flagd service as a bottleneck affecting multiple components. While this is partially accurate, our manual investigation showed it’s not the main issue.\n\nWe directed the model to look into the Recommendation service, but it only mentioned the cache problem after we explicitly pointed it out.\n\nExpand below to see the full interaction.\n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eOpenAI GPT4.1 anomaly 2 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eAbsolutely, I can help you analyze OpenTelemetry data to diagnose issues within your demo application.\nHowever, I’ll need more specific information or access to the relevant datasets (for example, traces, logs, or metrics from your observability backend).\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou have access to the OTel data in clickhouse, please query it.\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhat this means\n\t• The backend services (like API, web, or data processors) are not showing trace-level errors in the trace data.\n\t• The only error detected is from the load-generator, which is likely the synthetic traffic generator for testing/demo purposes.\nThis could be expected and not user-impacting.\n\u003c/div\u003e\u003c/pre\u003e\n\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 3:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou also have access to traces and metrics data to further continue the investigation\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eIssue Identified\n\nA major performance bottleneck is occurring in the /flagd.evaluation.v1.Service/EventStream span across multiple critical services (recommendation, fraud-detection, ad, flagd).\nThis is causing very high latencies, likely resulting in a degraded or unusable application experience for users.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 4:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYes, can you focus on the recommendation service?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 4:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eConclusion \u0026 Root Cause:\n\t• The recommendation service is blocked waiting for a response from flagd’s EventStream, hitting timeouts.\nThis means either the flagd service is slow, unavailable, or network connectivity to it is broken.\n\t• Because flagd (feature flag evaluation) is used within critical request paths,\nwhen it fails or hangs, the entire recommendation chain suffers, severely impacting user experience.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 5:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eTop durations for /oteldemo.RecommendationService/ListRecommendations are also very high (over 21–22 seconds), suggesting calls to downstream dependencies (like flagd) are causing severe slowness in client-facing endpoints. Why do you think recommendation service (ListRecommendations) is calling flagd?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 5:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eTrace Duration Patterns:\n\t• The /flagd.evaluation.v1.Service/EventStream spans within the recommendation service show extremely high durations\n(all stuck at 600+ seconds, matching the gRPC deadline).\n\t• The /oteledemo.RecommendationService/ListRecommendations spans also have high durations (~21–22 seconds),\nthough not as long as the flagd EventStream spans, but still anomalous for a user-facing endpoint.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 6:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou have access to tracing data, you can check the direct trace linking.\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 6:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003e\t• Based on the evidence, I cannot confirm with this trace dataset that ListRecommendations is directly calling flagd via EventStream in a way that is instrumented by distributed traces.\n\t• The slowness seen in both operations remains highly correlated.\nArchitectural practice suggests such a dependency may exist, but it is not proven in the current tracing data.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 7:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eif you can't find a direct link. Can you see another reason the recommendation service is running slow?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 7:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eKey Observations:\n\t• Durations are very high (around 19–22 seconds).\n\t• All these slow requests have StatusCode “Unset” and rpc.grpc.status_code = \"0\"\n(which means the gRPC call technically succeeded from a protocol point of view).\n\t• The span attributes are consistent—each request recommended 5 products, the peer IP was the same (10.32.10.87), and the user agent is grpc-node-js/1.12.6.\n\t• No explicit errors or useful StatusMessage are present.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 8:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWe recently upgraded the function get_product_list to optimize the cache logic. Can you check this didn't create a regression?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 8:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eIs this a regression?\n\t• A p95 response time of 200–250 ms might be acceptable for some backends, but if this represents a rise compared to previous levels, it could indicate a performance regression from the cache optimization.\n\t• If this is much higher than before the change, it could mean the cache is not being hit effectively, or new cache logic is causing delays (e.g., added serialization, locking, or slower fallback path).\n\n⸻\n\nSigns of regression:\n\t• There’s a jump from 204 ms to 251 ms during the 10:00–11:00 interval, possibly coinciding with the upgrade/deployment window.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 9:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eLooking at the data it looks like the request latency for get_product_list went up though.\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 9:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eSummary:\n\nThere is strong evidence from tracing data that the cache logic upgrade in get_product_list introduced increased latency, likely hurting overall recommendation service responsiveness.\nRolling back or tuning the new cache logic should be prioritized.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### Gemini 2.5 Pro\n\nGemini initially explored the available database tables, acknowledging that traces, logs and metrics were available. It explained that it would begin with traces, but performed its entire investigation without ever considering logs or metrics. Using traces, it became locked on to the wrong problem.\n\nAfter prompting it to also consider logs, Gemini then started to head in the right direction, suggesting that there was a timeout in the recommendation service, but did not discover why. After prompting it to investigate why the service timed out, it once again reverted back to only using traces, from which it decided that the issue was DNS. There are DNS errors present in the data set, but these are unrelated to the service in question. We then prompted it yet again to use the logs, but it got no further.\n\nNext, we prompted it to consider the metrics table, which had some success; it discovered that the pods were reaching 100% CPU usage and suggested this was the cause - but did not dig any deeper. Prompting it to look into why it reached 100%, Gemini looped around and found nothing. \n\n\u003e However, it then began to formulate an imaginary cause (for which it had no evidence), and began trying to prove its case. \n \nGemini was convinced that the cause must be a “poinsoned” product ID, and it began hunting for anything it could use to back this up. It was unsuccessful, but at least it admitted it.\n\nWe then asked it to take a look at resource usage around the time of the CPU spike, to see if there were other resources in contention. Somehow, this caused Gemini to discard the 100% CPU usage as a problem, and revert to telling us that DNS was, indeed, the problem (it isn’t).\n\nExpand below to see the full interaction.\n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eGemini 2.5 Pro anomaly 2 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eThis is a very interesting finding. The top 10 longest-running spans are all related to the `flagd.evaluation.v1.Service/EventStream`. The durations are extremely high, in the order of 600 seconds (10 minutes), which is a strong indicator of a timeout or a serious performance bottleneck.\n\nThe services affected are `recommendation`, `fraud-detection`, and `ad`, which suggests a widespread problem with a shared dependency. In this case, the dependency is the `flagd` service, which appears to be a feature flagging service.\n\nBased on this, I have identified the following:\n\n**Issue:**\n\nUsers are experiencing significant performance issues with the application. These issues are not presenting as explicit errors but as very slow responses, leading to a poor user experience.\n\n**Root Cause:**\n\nThe root cause of the performance degradation is the `/flagd.evaluation.v1.Service/EventStream` endpoint. This service is used by multiple other services (`recommendation`, `fraud-detection`, and `ad`) to receive feature flag updates. The extremely long duration of these calls indicates that the client services are waiting for a response from the `flagd` service for up to 10 minutes, which is causing them to become unresponsive.\n\n**Potential Solutions:**\n\nTo resolve this issue, I recommend the following actions:\n\n* **Investigate the `flagd` service:** The `flagd` service should be the primary focus of the investigation. Check its logs and metrics to understand why the `EventStream` endpoint is not responding in a timely manner.\n* **Review network policies:** It is possible that a network policy is preventing the client services from connecting to the `flagd` service, leading to a timeout. Review the network policies to ensure that there are no rules blocking this traffic.\n* **Implement connection timeouts:** The client services that connect to the `flagd` service should have a much shorter connection timeout. This will prevent them from being stuck for long periods and will allow them to fail fast and recover more gracefully. A timeout of a few seconds would be a good starting point.\n* **Add readiness and liveness probes:** If the `flagd` service is running in a containerized environment (like Kubernetes), adding readiness and liveness probes will help automatically detect and recover from situations where the service is not responding.\n\nBy implementing these recommendations, you should be able to resolve the performance issues and improve the user experience of your application.\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eI dont think this is the problem. have you considered using the logs table as well?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eExcellent! I've found the smoking gun. The logs from the `frontend-proxy` service show a clear pattern of `504 UT response_timeout` errors when calling the `frontend` service's `/api/recommendations` endpoint. This is a strong indicator that the `frontend` service is the bottleneck. The `response_timeout` message in the logs confirms that the `frontend-proxy` is waiting too long for a response from the `frontend` service.\n\nNow that I've narrowed down the problem to the `frontend` service, I will investigate its traces to understand what's causing the delay. I'll focus on the `/api/recommendations` endpoint and look for the longest-running spans to pinpoint the exact operation that's causing the bottleneck.\n\nI will now query the `otel_traces` table for the top 10 longest-running spans in the `frontend` service with the `SpanName` of `/api/recommendations`.\n\n...\n\nThis is unexpected. I see that there are no traces for the `/api/recommendations` span in the `frontend` service, even though the logs from the `frontend-proxy` clearly show that this endpoint is being called.\n\nThis could mean that the `frontend` service is not correctly configured to generate traces for this endpoint. This is a common problem in distributed tracing setups, where some services might not be properly instrumented.\n\nEven though I can't see the traces for the `/api/recommendations` endpoint, the logs from the `frontend-proxy` provide enough evidence to conclude that the `frontend` service is the root cause of the performance issues. The `504 UT response_timeout` errors are a clear indication that the `frontend` service is not responding in a timely manner.\n\nBased on this, I can provide the following updated analysis:\n\n**Issue:**\n\nUsers are experiencing significant performance issues with the application. These issues are not presenting as explicit errors but as very slow responses, leading to a poor user experience.\n\n**Root Cause:**\n\nThe root cause of the performance degradation is the `frontend` service. The `frontend-proxy` is timing out when calling the `/api/recommendations` endpoint on the `frontend` service, as evidenced by the `504 UT response_timeout` errors in the `frontend-proxy`'s logs.\n\n**Potential Solutions:**\n\nTo resolve this issue, I recommend the following actions:\n\n* **Investigate the `frontend` service:** The `frontend` service should be the primary focus of the investigation. Check its logs and metrics to understand why the `/api/recommendations` endpoint is not responding in a timely manner.\n* **Instrument the `frontend` service for tracing:** To get a better understanding of the performance of the `frontend` service, it's crucial to instrument it to generate traces. This will allow you to see the full lifecycle of a request and identify the exact operations that are causing the delay.\n* **Review the code for the `/api/recommendations` endpoint:** The code for the `/api/recommendations` endpoint should be reviewed to identify any potential performance bottlenecks. This could include inefficient database queries, slow external service calls, or other performance anti-patterns.\n* **Add readiness and liveness probes:** If the `frontend` service is running in a containerized environment (like Kubernetes), adding readiness and liveness probes will help automatically detect and recover from situations where the service is not responding.\n\nBy implementing these recommendations, you should be able to resolve the performance issues and improve the user experience of your application. The most critical step is to instrument the `frontend` service so you can get a better understanding of its performance and identify the root cause of the timeouts.\n\u003c/div\u003e\u003c/pre\u003e\n\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 3:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhy does the recommendation service timeout?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eThis is the exact question we need to answer. The trace data has provided a crystal-clear picture of what's happening. Let's break it down.\n\n**Analysis of the Trace (`425b78d730282b74af1a880fa7758a3b`)**\n\n1. **The Client's Perspective (`frontend` service):**\n * The `frontend` service makes a call to the `recommendation` service (`grpc.oteldemo.RecommendationService/ListRecommendations`).\n * This call takes **307 milliseconds** from the `frontend` service's point of view.\n\n2. **The Server's Perspective (`recommendation` service):**\n * The `recommendation` service receives the request, processes it, and sends a response.\n * The actual processing time inside the `recommendation` service (`/oteldemo.RecommendationService/ListRecommendations` span) is only **5 milliseconds**.\n\n**The Root Cause: Massive Network Overhead**\n\nThe difference between the client's measured time (307 ms) and the server's processing time (5 ms) is **302 milliseconds**. This massive delay is happening *outside* the recommendation service's code.\n\nI found the same pattern for all other downstream calls from the `frontend` service. For example, calls to the `product-catalog` service take about **218 ms** from the client's side but are processed in less than **1 ms** on the server side.\n\nThis means there is a consistent, large overhead of **~200-300 milliseconds** for every single outgoing gRPC call made by the `frontend` service.\n\n**Conclusion**\n\nThe **recommendation service is not the problem**. It's extremely fast. The problem lies within the `frontend` service's environment, which is adding a significant delay to every network call it makes.\n\nThis is the root cause of the entire issue. This fixed delay makes the `frontend` service incredibly slow, which in turn causes the `frontend-proxy` to time out, just as we saw in the logs.\n\n**Likely Causes for the Overhead and Next Steps**\n\nThis type of fixed overhead on outgoing network calls is often caused by misconfigurations in the environment, such as:\n\n* **DNS Resolution Issues:** The `frontend` service's container might be trying to resolve service names (like `recommendation.default.svc.cluster.local`) using an external DNS server that fails or times out, before falling back to the correct internal cluster DNS.\n* **Service Mesh Misconfiguration:** If you are using a service mesh like Istio or Linkerd, a misconfigured sidecar proxy could be adding this latency to every call.\n* **Network Policies:** A network policy could be slowing down connection establishment.\n\n**Recommendation:**\n\nYour next step should be to **investigate the runtime environment of the `frontend` service pod**. Specifically, check its DNS configuration (`/etc/resolv.conf`) and any service mesh or network policy configurations that apply to it. Resolving this environmental issue should eliminate the network call overhead and fix the application's performance problems.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 4:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eOk but what is causing the timeout?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 4:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou are asking the central question. I will connect all the pieces of evidence to give you a definitive answer.\n\nThe timeout is caused by a **\"death by a thousand cuts\"** scenario happening inside the `frontend` service. The root cause is the **~250ms of network overhead** that is added to *every single* outbound call it makes.\n\nHere is the step-by-step breakdown of how this leads to a 15-second timeout for a single `/api/recommendations` request:\n\n1. **A Request Arrives:** A user requests recommendations, hitting the `frontend-proxy`, which forwards the request to the `frontend` service's `/api/recommendations` endpoint. The `frontend-proxy` starts a 15-second timer.\n\n2. **The `frontend` Service's Job:** To build the response, the `frontend` service needs to gather data from other services. Looking at the trace data, we know for this single request it makes **5 downstream calls**:\n * 1 call to the `recommendation` service.\n * 4 calls to the `product-catalog` service.\n\n3. **The Hidden \"Network Tax\":** As we discovered, every one of these 5 calls is burdened with a ~250 millisecond delay before it even leaves the `frontend` service's environment. The downstream services are incredibly fast, but this overhead is unavoidable on every call.\n\n4. **Calculating the Cumulative Delay:** These small delays add up. The total *self-inflicted* delay caused by the network overhead for just this one request is:\n \u003e 5 calls × ~250 ms/call = **~1.25 seconds**\n\n This means that even if the other services responded instantly, the `/api/recommendations` endpoint would still take over a second to respond.\n\n5. **Amplification Under Load:** A 1.25-second response time is slow, but it's not a 15-second timeout. The timeout happens when the application is under load. The `load-generator` service is constantly sending requests. Because every request is slowed down by this 1.25-second (or more) overhead, the `frontend` service becomes a bottleneck. Requests start to pile up in a queue, waiting to be processed. This queuing and resource contention (CPU, connection pools) dramatically increases the total response time, pushing it far beyond the initial 1.25 seconds.\n\n**Conclusion:**\n\nThe `frontend-proxy` times out because the `frontend` service, burdened by the cumulative effect of its network overhead, cannot process requests fast enough under normal load. The response time for some requests eventually exceeds the 15-second limit, triggering the `504 UT response_timeout` error you saw in the logs.\n\nThe problem isn't one slow downstream call, but rather the compounding effect of a small, consistent environmental problem on many fast downstream calls. To fix the timeout, you must fix the source of the network overhead within the `frontend` service's environment\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 5:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eyouve only used traces again, maybe you should look at the logs and there will be more info?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 5:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou are right to insist on a thorough investigation, and I appreciate you pushing me to use all the available data. My apologies for the repeated empty results. The lack of logs for that specific trace is, in itself, a critical discovery.\n\nLet's synthesize everything we've found from both the traces and the logs to build a complete picture.\n\n### The Story So Far:\n\n1. **The Symptom (from Logs):** We started with your suggestion to check the logs. We found clear evidence in the `frontend-proxy` logs of `504 UT response_timeout` errors. This tells us the `frontend-proxy` is giving up on the `frontend` service after waiting for 15 seconds. This is the user-facing problem.\n\n2. **The Bottleneck (from Traces):** The trace data allowed us to follow a request that was causing one of these timeouts. It showed that the `frontend` service was taking an extremely long time to respond.\n\n3. **The \"Why\" (from Traces):** By breaking down the trace, we discovered something crucial: the `frontend` service makes many calls to other services (like `recommendation` and `product-catalog`). While these downstream services are very fast (responding in ~5ms), there is a consistent, large overhead of about **~250 milliseconds** added to *every single one* of these outgoing calls.\n\n4. **The Missing Piece (from Logs... or lack thereof):** When we tried to find logs from the `frontend` service for that exact slow trace (`425b78d730282b74af1a880fa7758a3b`), we found nothing. This is a major clue. It strongly suggests the problem is happening at a level below the application code itself, likely in the infrastructure or environment where the `frontend` service is running.\n\n### Final Conclusion: The Root Cause of the Timeout\n\nThe `frontend-proxy` times out because the **`frontend` service is suffering from a severe \"death by a thousand cuts\"**. Here's the sequence of events:\n\n1. A request for recommendations arrives at the `frontend` service.\n2. To fulfill this request, the `frontend` service must call multiple downstream services (we saw 5 such calls in our example trace).\n3. Each of these 5 calls is saddled with a **~250ms \"network tax\"** before it even leaves the `frontend` service's environment. This tax is the real root cause.\n4. This adds up to a minimum self-inflicted delay of **1.25 seconds** (5 calls x 250ms) for what should be a very fast operation.\n5. Under normal load, multiple requests hit the `frontend` service at once. Because each one is already slow, they pile up, waiting for the CPU and network resources to become available. This queuing effect amplifies the initial 1.25-second delay, pushing the total response time over the `frontend-proxy`'s 15-second limit.\n\n**Therefore, the timeout is not caused by a slow recommendation service, but by the compounding effect of a networking issue in the `frontend` service's environment.**\n\n### Recommendations\n\nThe evidence strongly points to an environmental issue, not an application code issue.\n\n1. **Primary Action: Investigate the `frontend` Service's Pod Environment.** The source of the ~250ms overhead must be found. The most common culprits for this kind of fixed network delay are:\n * **DNS Resolution:** Check the `/etc/resolv.conf` file in the `frontend` service's container. It might be trying to contact an external DNS server before using the correct internal cluster DNS.\n * **Service Mesh Misconfiguration:** If you're using a service mesh like Istio or Linkerd, inspect the configuration of the sidecar proxy attached to the `frontend` pod.\n * **Network Policies:** Review any `NetworkPolicy` resources in your cluster that might be inspecting or slowing down traffic from the `frontend` pod.\n\n2. **Secondary Action: Improve Observability.** This investigation was made much harder because the `frontend` service wasn't logging correctly. To prevent this in the future:\n * **Fix the Logging:** Ensure the `frontend` service logs are being generated and, critically, that they are correlated with traces by including the `TraceId` in every log message. This will make future debugging much faster.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 6:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003ewhat about looking at resource usage of the recommendation service?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 6:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou were absolutely right to insist on checking the resource usage. My apologies for focusing too narrowly on the network latency. The metrics you asked for have revealed the true root cause of the problem.\n\nHere are the results for the resource utilization of the `recommendation` service:\n\n* **Memory Usage (`k8s.container.memory_limit_utilization`):** The average memory usage is **~9.4%** of the limit, with a maximum peak at a healthy **~69.1%**. Memory is not the issue.\n\n* **CPU Usage (`k8s.container.cpu_limit_utilization`):** The average CPU usage is **~29.8%**, but the maximum value is **~100.2%**.\n\n### The Real Root Cause: CPU Throttling\n\nThe fact that the `MaxValue` for CPU utilization is at 100% is the smoking gun. This means the `recommendation` service's container is hitting its CPU limit. When this happens, Kubernetes **CPU throttles** the container, preventing it from using more CPU. This throttling directly translates into processing delays.\n\nHere’s the corrected sequence of events:\n\n1. A request arrives at the `recommendation` service.\n2. Under load, the service experiences a spike in CPU usage, hitting 100% of its allocated limit.\n3. Kubernetes steps in and throttles the container, effectively slowing down its ability to process the request. This causes the service's response time to jump from a few milliseconds to several hundred milliseconds.\n4. This delay is passed back to the `frontend` service.\n5. The `frontend` service makes multiple calls to other services (including the now-throttled `recommendation` service). The delays from these throttled calls accumulate.\n6. This cumulative delay makes the `frontend` service slow to respond to the `frontend-proxy`.\n7. Finally, the `frontend-proxy`, waiting on the slow `frontend` service, gives up after 15 seconds, resulting in the `504 UT response_timeout` errors.\n\nMy previous conclusion about a \"network tax\" was incorrect. The delay seen in the traces was not from the `frontend` service's network, but from waiting on the `recommendation` service, which was intermittently being slowed down by CPU throttling. Thank you for pushing me to investigate further.\n\n### Recommendations\n\nThe solution is to alleviate the CPU pressure on the `recommendation` service.\n\n1. **Immediate Fix: Increase the CPU Limit.** The most straightforward solution is to increase the CPU limit for the `recommendation` service deployment. This will give it more headroom to handle spikes in load without being throttled.\n\n2. **Long-Term Fix: Investigate CPU Spikes.** While increasing the limit will fix the immediate problem, it's important to understand *why* the CPU is spiking. The developers should investigate the `recommendation` service's code to identify any inefficient operations or queries that could be optimized.\n\n3. **Proactive Monitoring: Set Up Alerts.** To prevent this from happening in the future, you should configure alerts to fire whenever a service's CPU utilization approaches its limit. This will allow you to address resource constraints before they impact users.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 7:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003ecan you narrow down why the CPU reaches 100%?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 7:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eOf course. Let's dig deeper into the \"why.\" You are right to ask for the next level of detail. We know the `recommendation` service is hitting its CPU limit, but *why* is its CPU usage so high in the first place?\n\nThe traces we have examined show that the primary function of the `recommendation` service is to:\n1. Call the `product-catalog` service to get a list of products.\n2. Process this list to generate recommendations.\n\nMy hypothesis is that the CPU spike is directly related to the **size of the product list** it has to process for every single request. A large list could make the recommendation logic itself very CPU-intensive.\n\nTo test this, I need to find out how many products are in the catalog. The `product-catalog` service is the source of truth for this. I will inspect the traces originating from this service to find any logs or attributes that indicate the number of products it's handling.\n\nI will now query the `otel_traces` table to look at the attributes of the `ListProducts` span within the `product-catalog` service. This might tell us the size of the payload being returned.\n\n...\nThis is a fantastic find, but `10` seems far too low to cause a CPU spike. This suggests the sample of one trace might not be representative. The number of products is likely dynamic and can be much larger.\n\nMy next step is to find out the **maximum number of products** the `product-casting-service` has ever returned. This will tell us the peak load that the `recommendation` service is expected to handle.\n\nI will now query the `otel_traces` table again for the `ListProducts` span, but this time I will look for the maximum value of the `app.products.count` attribute across all traces. This should give us a much better idea of the potential scale of the problem.\n...\nThis is a dead end. The trace attributes consistently show a small number of products, which contradicts the CPU saturation we're seeing. It means my hypothesis that the *number* of products is the problem is likely wrong.\n\nYou've been right to push me to look at different data sources. I've been switching between traces, metrics, and logs, but I haven't used them together effectively enough. Let's try a new approach that directly links the high CPU usage to a specific action.\n\nMy new hypothesis is: **There is a \"rogue\" product or a specific type of product in the catalog that triggers a highly inefficient, CPU-intensive code path in the `recommendation` service.**\n\nHere’s how we can prove this:\n1. Find a time window when the `recommendation` service's CPU usage was at its peak (100%).\n2. During that exact time window, find out which products were being requested by the `frontend` service.\n\nThis will directly correlate the CPU spike with the product being processed at that moment.\n\nFirst, I need to find the exact timestamp when the CPU utilization hit 100%. I will query the metrics table for this.\n...\nThis is another dead end. It seems the `productIds` are not being logged in the `url.query` attribute in the `frontend-proxy` logs.\n\nMy apologies. I have been jumping between different data sources without a clear, systematic approach. It's time to take a step back and think logically, using all the clues we've gathered so far.\n\nHere's what we know for certain:\n\n1. **The Symptom:** The application times out. The logs confirm this with `504` errors from the `frontend-proxy`.\n2. **The Bottleneck:** The timeouts are caused by the `frontend` service responding too slowly.\n3. **The Root Cause:** The `frontend` is slow because it's waiting on the `recommendation` service.\n4. **The Core Problem:** The `recommendation` service is slow because its CPU is being throttled at 100%.\n\nThe question remains: **What in the `recommendation` service's code is so inefficient that it consumes 100% of the CPU?**\n\nThe traces show that the `recommendation` service's main job is to process a list of products. We have also seen that the *number* of products is small, so we can rule out \"processing a large list\" as the problem. This leaves us with one logical conclusion:\n\n**There must be a specific, \"poison\" product ID that triggers a highly inefficient, CPU-intensive code path.**\n\nThis could be a product with unusual data that causes an infinite loop, a complex calculation, or some other performance anti-pattern.\n\nSince I have exhausted the possibilities of finding this \"poison\" product ID through the available telemetry data, the next logical step is to hand this investigation over to the developers. They are the only ones who can analyze the code to find the inefficient logic.\n\n### Final Conclusion and Recommendation for Developers\n\n* **The Issue:** The application is timing out due to a bottleneck in the `recommendation` service.\n* **The Root Cause:** The `recommendation` service is experiencing CPU throttling, with usage hitting 100%. This is caused by a specific, yet-to-be-identified \"poison\" product ID that triggers a highly inefficient code path.\n* **Recommendation:**\n 1. **Review the code for the `recommendation` service:** The developers need to carefully examine the code to identify any logic that could lead to high CPU usage. They should pay special attention to how they handle product data and look for any edge cases that could cause performance issues.\n 2. **Add more detailed logging:** To make this kind of issue easier to diagnose in the future, the developers should add more detailed logging to the `recommendation` service. Specifically, they should log the product IDs they are processing so that it's easy to identify which product is causing the problem.\n 3. **Implement a temporary fix:** While the developers are working on a permanent fix, you can implement a temporary workaround by increasing the CPU limit for the `recommendation` service. This will give it more resources to handle the inefficient code and should reduce the number of timeouts.\n\nBy following these recommendations, you should be able to resolve the performance issues and improve the stability of your application. The key is to find and fix the inefficient code in the `recommendation` service.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 8:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003ecan you look at resource usage around the time of a specific time out to see if anything happened?\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 8:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eThis is a fascinating result. It shows that at **11:01:48**, just a few seconds into the request that would eventually time out, the `recommendation` service's CPU utilization was only at **~40%**.\n\nThis directly contradicts my previous conclusion and proves that a sustained CPU spike on the `recommendation` service is **not** the cause of the timeout. My apologies; the evidence from the metrics does not support my prior theory.\n\nThis brings us back to my initial finding from the trace data, which showed a consistent, fixed overhead of **~250ms** on every outbound call from the `frontend` service. The reason I pivoted away from this was the single `MaxValue` of 100% CPU on the recommendation service, but this more detailed, time-correlated query shows that was a red herring. It was a momentary spike, not the cause of this specific timeout.\n\n### The Real, Correlated Root Cause: Network Overhead\n\nLet's re-examine the situation with this new, definitive evidence:\n\n1. **The Timeout (Logs):** A `504` timeout is logged by the `frontend-proxy` at **11:02:00**.\n2. **Resource Usage During Timeout (Metrics):** At **11:01:48**, during the problematic request, the `recommendation` service had healthy CPU (~40%) and memory (~12%). **Therefore, resource saturation is not the cause.**\n3. **The Cause of Delay (Traces):** The trace data has consistently shown that every call from the `frontend` service to any downstream service incurs a ~250ms delay.\n\n**Final Verified Conclusion:**\n\nThe timeout is caused by a **\"death by a thousand cuts\"** driven by network overhead within the `frontend` service's environment.\n\nWhen the `frontend` service receives a request for `/api/recommendations`, it makes multiple calls to other services. Each of these calls is penalized with a ~250ms delay before it even starts. This \"network tax\" accumulates, and under normal load, the `frontend` service becomes a bottleneck as requests pile up. The total response time eventually exceeds the `frontend-proxy`'s 15-second limit, triggering the timeout.\n\nThe problem is not in any of the downstream services; it is in the environment of the `frontend` service itself.\n\n### Recommendation (Re-validated)\n\nThe evidence now points conclusively to a networking issue with the `frontend` service pod. The next steps for the engineering team should be:\n\n1. **Primary Action: Investigate the `frontend` Service's Pod Environment.** The source of the ~250ms overhead must be found. The most likely culprits are:\n * **DNS Resolution Issues:** Check the pod's `/etc/resolv.conf`.\n * **Service Mesh Misconfiguration:** Inspect the sidecar proxy's configuration.\n * **Network Policies:** Review any policies that could be slowing down egress traffic.\n\n2. **Secondary Action: Improve Observability.** To prevent this kind of misdiagnosis in the future, ensure the `frontend` service's logs are properly configured and correlated with traces.\n\nThank you for your patience and for guiding me to the correct conclusion by insisting on correlating the data. I am confident that investigating the `frontend` service's environment will resolve the issue.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n#### LLM investigation summary \n\n| Model | Resolution | # of MCP calls | Duration | Input tokens | Output tokens | Cost |\n|-------|------------|----------------|----------|--------------|---------------|------|\n| Claude 4 sonnet | Yes, with minor guidance | 15 | 23 minutes | 1222645 | 15256 | $3.90 |\n| OpenAI o3 | Yes, with minor guidance | 15 | 45 minutes | 614048 | 25386 | $1.79 |\n| OpenAI GPT4.1 | Yes, with major guidance | 15 | 20 minutes | 64988 | 4788 | $0.17 |\n| Gemini Pro | No | 21 | 17 minutes | 1148092 | 29654 | $3.32 |\n\n### Anomaly 3: Product catalog failure\n\nUsers have been reporting intermittent issues using the application, especially when accessing a product page.\n\n#### Manual investigation\n\nWe start from the Client sessions screen and look at sessions with errors.\n\n\n\nWe’re seeing frequent errors from the Products API, which result in empty product pages for users.\n\nIn the trace from one affected session, the error appears in the span: `oteldemo.ProductCatalogService/GetProduct.`\n\nTo investigate further, we can filter the Traces view by this span name and focus on traces that include it in an error state.\n\n\n\nBy adding a column for the product ID, we see that all the errors are tied to the same product: `OLJCESPC7Z`. This suggests there’s likely an issue with that specific item.\n\n\n\nIf we look at a call using this specific product ID with no errors, we can see the product name in the log metadata: \n\n\n\n**Users have been experiencing issues when accessing the product page of the product ID: OLJCESPC7Z, which is a National Park Foundation Explorascope.**\n\n#### LLM investigation\n\n##### Claude Sonnet 4 model \n\nThe model initially identified that the issue was related to a feature flag and a failure in the product catalog function. However, it didn’t determine which specific product caused the failure.\n\nA follow-up prompt helped guide the model toward the correct resolution, and it was then able to identify the product responsible.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eClaude Sonnet 4 anomaly 3 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003e Root Cause Summary\n\nThe productCatalogFailure feature flag was enabled (likely for testing), causing the product catalog service to intentionally fail. This created a cascade effect:\n\t1. Product Catalog → Returns \"Product Catalog Fail Feature Flag Enabled\" error\n\t2. Frontend → Cannot retrieve product information, fails user requests\n\t3. Checkout → Cannot complete orders due to missing product data\n\t4. Frontend-proxy → Propagates errors from backend services\n\t5. Load-generator → Receives error responses during load testing\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eUsers reported having issues when accessing a specific product, can you find out which product?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eTimeline Impact:\n\t• Hour 08:00–09:00: 20.61% error rate for this product\n\t• Hour 09:00–10:00: 69.10% error rate for this product (major degradation)\n⸻\nImmediate Solution:\n\nThe productCatalogFailure feature flag appears to be configured to only affect the National Park Foundation Explorascope product.\nDisabling this feature flag will immediately restore access to this popular astronomy product and resolve all user complaints about being unable to view or purchase this specific item.\n\nBusiness Impact:\nThis appears to be a high-value telescope product, so the 75% failure rate is likely causing significant revenue loss and customer frustration for users specifically interested in this National Park Foundation branded telescope.\n⸻\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### OpenAI o3 model \n\nAfter the first prompt, the model identified that the issue was tied to a feature flag and a failure in the product catalog function. However, it didn’t isolate the specific product causing the problem.\n\nThe second prompt guided the model to a more complete resolution. \n\n\u003e It correctly identified the product ID but returned in the same convincing way an incorrect product name. \n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eOpenAI o3 anomaly 3 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot cause\n\nA feature flag managed by flagd was switched to the “fail” variant, which makes the Product-Catalog service intentionally return an error for every call.\nBecause the frontend relies on that service for every page, the whole site appears broken to end-users.\n⸻\nWhy it happened / contributing factors\n\t1. Flag default or rule changed without proper blast-radius check.\n\t2. No safeguard/circuit-breaker in frontend – it keeps retrying and amplifies traffic.\n\t3. Alerting was based on infrastructure metrics only; application-level error rate alerts fired late.\n\u003c/div\u003e\u003c/pre\u003e\n\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eUsers reported having issues when accessing a specific product, can you find out which product?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eProduct ID: OLJCESPC7Z\n(That ID corresponds to the “Sunglasses” item in the Hipster-Shop / OpenTelemetry-Demo catalogue.)\nEvidence:\n\t• 18,763 error traces hit the path GET /api/products/OLJCESPC7Z —\norders of magnitude more than any other product.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### OpenAI GPT4.1 \n\nThe initial response wasn’t useful because the model didn’t query the data through the MCP server. Even after accessing data in ClickHouse, it failed to surface any meaningful root cause and stayed focused on high-level service behavior. We had to provide significant additional guidance.\n\nWith that context, the model correctly identified the issue as coming from the product catalog service, triggered by a feature flag being enabled. However, it didn’t specify which product was responsible. Eventually, the model identified the correct product ID, but not the actual product name.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eOpenAI GPT4.1 anomaly 3 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eAbsolutely, I can help you analyze OpenTelemetry data to diagnose issues within your demo application.\nHowever, I’ll need more specific information or access to the relevant datasets (for example, traces, logs, or metrics from your observability backend)\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou have access to the data, please proceed with the investigation\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 2:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot Cause Hypothesis:\n\t• The load-generator service is experiencing both the highest number of application errors and the highest average latency, indicating it is the most problematic component at present.\n\t• The frontend-proxy and frontend also have elevated latency, suggesting downstream or systemic performance bottlenecks.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 3:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eUsers have reported having issues when accessing product page\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot Cause:\n\t• The main root cause appears to be the Product Catalog Fail Feature Flag being enabled.\nThis flag is making the Product Catalog Service return errors intentionally, resulting in failed product page loads and user issues.\n\t• Additionally, the response time for the GetProduct API is elevated, which can degrade user experience even when successful.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 4:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eIs there a specific product in cause\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3:\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eSummary:\n\t• Product ID: OLJCESPC7Z\n\t• Product Name/Details: Not found in tracing or log context.\n\t• It appears prominently in recommendation requests and failure logs, but without information on what this product is.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### Gemini 2.5 Pro\nGemini discovered the correct issue immediately from the first, simple prompt - identifying that orders failed with a specific product, and shared the correct product code.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eGemini 2.5 Pro anomaly 3 full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot Cause\n\nMy analysis of the OpenTelemetry data points to two primary issues, both originating from the frontend service:\n\t1. “Product Catalog Fail” Feature Flag:\nThe most frequent error message is:\nError: Product Catalog Fail Feature Flag Enabled.\nThis indicates that a feature flag is intentionally causing the product catalog to fail.\nThis is the most likely cause of the widespread issues users are experiencing.\n\t2. Order Preparation Failure:\nThe second most common error is:\nfailed to prepare order: failed to get product #\"OLJCESPC7Z\".\nThis error indicates the system is unable to retrieve a specific product, preventing users from successfully placing orders.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n#### LLM investigation summary \n\n| Model | Resolution | # of MCP calls | Duration | Input tokens | Output tokens | Cost |\n|-------|------------|----------------|----------|--------------|---------------|------|\n| Claude 4 sonnet | Yes, with minor guidance | 27 | 9 minutes | 1032902 | 6928 | $3.20 |\n| OpenAI o3 | Yes, with minor guidance | 22 | 9 minutes | 99557 | 5733 | $0.31 |\n| OpenAI GPT4.1 | Yes, with major guidance | 11 | 3 minutes | 41823 | 2232 | $0.10 |\n| Gemini Pro | Yes | 6 | 1m15s | 99992 | 3376 | $0.30 |\n\n### Demo anomaly : Payment service cache leak\n\nUsers experience issues during the checkout process. The payment does not go through. \n\n#### Manual investigation\n\nThe detailed walkthrough of the issue investigation using ClickStack is already described in the [documentation](https://clickhouse.com/docs/use-cases/observability/clickstack/getting-started/remote-demo-data#demo-steps).\n\n#### LLM investigation\n\n##### Claude Sonnet 4 model\n\nThe model found the issue after the first prompt, it found out that the issue is due to the payment service cache getting full that causes issues at checkout time.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eClaude Sonnet 4 demo anomaly full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eIssue Identified: Visa Cache Overflow Problem\n\nRoot Cause Analysis:\n\nPrimary Issue:\nThe payment service’s Visa validation cache reached its maximum capacity limit (100,000 items)\nand was unable to accept new card validation requests.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### OpenAI o3 model \n\nThe model found the issue after the first prompt, it found out that the issue is due to the payment service cache getting full that causes issues at checkout time.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eOpenAI o3 demo anomaly full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eWhy it’s happening (root cause)\n\t1. The payment-service keeps a local in-memory “Visa validation cache”.\n\t2. The cache has a hard capacity of 100,000 items and no eviction/TTL.\n\t3. Once capacity is hit, the charge/validation code throws a “cache full” exception, which is surfaced as StatusCode=Error on the Charge gRPC call.\n\t4. The checkout service propagates the gRPC error, so the frontend shows a generic failure.\n\t5. Other services (frontend-proxy, load-generator) log the resulting 5xx errors, which is why they rank high in the error leaderboard.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n###### OpenAI GPT4.1 model \n\nThe model didn’t uncover any useful leads in the first prompt. It focused on the load-generator, frontend, or frontend-proxy services reporting errors, without investigating why those services were affected.\n\nAfter we provided context about the specific issue users were experiencing, the model was able to narrow it down to a problem with payments.\n\nWhen asked to dig deeper into the payment service, the model finally identified the root cause: the Visa cache was full.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eOpenAI GPT4.1 demo anomaly full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003e### The Issue\n\nUsers are experiencing issues with the application, and the root cause has been identified as a performance problem in the `recommendation` service. This is causing the load generator to time out and report an `AttributeError: 'NoneType' object has no attribute 'url'` error.\n\n### Root Cause\n\nThe `recommendation` service is experiencing high latency, with P95 and P99 response times of ~141ms and ~241ms, respectively. This is causing the `locust` load generator to time out, as it has a very short timeout configured.\n\nThe trace of a slow request reveals that the `recommendation` service is taking a long time to process requests. The following is the trace of a request that took over 250ms to complete.\n\n**Trace ID:** `2bfce73fcb285c710ba334cd7732ac46`\n\n| SpanId | ParentSpanId | SpanName | ServiceName | Duration (ns) | SpanAttributes|\n| -------------------- | -------------------- | ------------------------------------------------------------ | ----------------- | ------------- | --------------|\n| 82c426fb75da8708 | | GET | load-generator | 1341409454 | `http.method`: `GET`, `http.status_code`: `200`, ... |\n| 179a068330287ca3 | 82c426fb75da8708 | ingress | frontend-proxy | 1336560000 | `component`: `proxy`, `downstream_cluster`: `-`, ... |\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 2:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eUsers have been reported having issues to complete the check out.\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 2 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot Cause (Most Likely):\nFailures during either placing an order or charging the payment, not being surfaced in logs, likely due to missing error logging/instrumentation.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 3:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eCan you further down investigate the payment service\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot Cause – Payment Service\n\t• The root cause of the payment service failure is:\n“Visa cache full: cannot add new item.”\n\t• This strongly suggests the payment processing logic uses an in-memory or local cache for Visa transactions that has reached capacity.\n\t• This results in rejecting new payment processing requests, which explains the systematic and repeated errors seen during checkout.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n##### Gemini 2.5 Pro\n\nThe test data set contains various repeated, but non-problematic, errors. This kind of noise is common in observability data. When Gemini investigated this anomaly, it came across one of these errors and got stuck on this incorrect path. It then began to generate a very large table that reproduced a full trace, taking about 30 seconds to generate each row. We let it run for a few minutes, but ended up cancelling the generation. \n\nWe gave a follow up prompt with additional context about the user complaint, narrowing the area to the check out process. Gemini discovered one of the OTel demo feature flags (`cartFailure`) which was disabled for this test, and became convinced that this feature flag was the root cause of the problem, instructing us that we should enable the feature flag (enabling the feature flag would introduce a new, simulated failure mode.)\n\nDirecting Gemini away from the feature flag resulted in little progress - it gave up and insisted we needed to implement more logging.\n\nExpand below to see the full interaction. \n\n\u003cdetails class=\"llm\"\u003e\n \u003csummary\u003eGemini 2.5 Pro demo anomaly full interaction\u003c/summary\u003e\n \nPrompt 1:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou're an Observability agent and have access to OpenTelemetry data from a demo application. Users have reported issues using the application, can you identify what is the issue, the root cause and suggest potential solutions?\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 1 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003e### The Issue\n\nUsers are experiencing issues with the application, and the root cause has been identified as a performance problem in the `recommendation` service. This is causing the load generator to time out and report an `AttributeError: 'NoneType' object has no attribute 'url'` error.\n\n### Root Cause\n\nThe `recommendation` service is experiencing high latency, with P95 and P99 response times of ~141ms and ~241ms, respectively. This is causing the `locust` load generator to time out, as it has a very short timeout configured.\n\nThe trace of a slow request reveals that the `recommendation` service is taking a long time to process requests. The following is the trace of a request that took over 250ms to complete.\n\n**Trace ID:** `2bfce73fcb285c710ba334cd7732ac46`\n\n| SpanId | ParentSpanId | SpanName | ServiceName | Duration (ns) | SpanAttributes|\n| -------------------- | -------------------- | ------------------------------------------------------------ | ----------------- | ------------- | --------------|\n| 82c426fb75da8708 | | GET | load-generator | 1341409454 | `http.method`: `GET`, `http.status_code`: `200`, ... |\n| 179a068330287ca3 | 82c426fb75da8708 | ingress | frontend-proxy | 1336560000 | `component`: `proxy`, `downstream_cluster`: `-`, ... |\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 2:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eUsers have reported having issues completing check out.\n\u003c/div\u003e\u003c/pre\u003e\n\nLLM response 2 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eI haven't found any errors in the `checkout` service's logs or any traces with a direct error status. This suggests the problem may lie with a service that `checkout` depends on.\n\nTo identify these dependencies, I will find a recent trace that involves the `checkout` service and examine all the spans within that trace. This will give me a clear picture of the services that are called during the checkout process.\n\nI'll start by looking for a trace where the `checkout` service is called. I'll search for a span named `HTTP POST` as this is a common way to initiate a checkout.\n...\nI've examined the trace and it appears that all the spans have a `StatusCode` of `Unset` or `Ok`, which means there are no explicit errors being reported. However, this doesn't mean there aren't any problems. A user can still have a bad experience without a technical error.\n\nLet's take a closer look at the spans to see if there are any other clues. The trace shows the following sequence of events:\n\n1. The `frontend-proxy` receives a `POST` request to `/api/checkout`.\n2. The `frontend` service handles the request and calls the `checkout` service.\n3. The `checkout` service then calls several other services, including:\n * `cart`\n * `product-catalog`\n * `currency`\n * `shipping`\n * `payment`\n * `email`\n4. Finally, the `checkout` service publishes a message to a `kafka` topic.\n\nThere is one interesting span in this trace. The `cart` service has a `feature_flag.variant` of `off` for the `cartFailure` feature flag. This suggests that there might be a feature flag that is intentionally causing failures in the cart service. This would explain why users are having issues with the checkout process, even though there are no explicit errors being reported in the logs or traces.\n\n### The Issue\n\nUsers are reporting issues completing the checkout process.\n\n### Root Cause\n\nThe `cart` service has a feature flag called `cartFailure` that is currently in the `off` state. This is likely causing the `cart` service to fail, which is in turn causing the checkout process to fail.\n\n### Recommended Solution\n\nTo resolve this issue, I recommend that you enable the `cartFailure` feature flag. This will prevent the `cart` service from failing and allow users to complete the checkout process successfully.\n\n### How to Implement the Solution\n\nYou can enable the `cartFailure` feature flag by running the following command:\n\n```\nfeature_flag.enable(\"cartFailure\")\n```\n\nAfter you have enabled the feature flag, the checkout process should start working correctly.\n\nI hope this helps! Please let me know if you have any other questions.\n\u003c/div\u003e\u003c/pre\u003e\n\nPrompt 3:\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eThe flag is set to off, so this isn’t the problem. Keep looking\n\u003c/div\u003e\u003c/pre\u003e\n\n\nLLM response 3 (snippet):\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou are absolutely right to question that. My apologies for the confusing recommendation. You've caught a mistake in my reasoning.\n\nA flag named `cartFailure` would indeed cause failures when turned **ON**, not OFF. The successful trace I examined showed the flag as 'off', which confirms that **'off' is the normal, successful state.**\n\nMy previous conclusion was based on a misinterpretation. The fact that we see a feature flag related to cart failure at all is a strong indicator, but I jumped to the wrong conclusion about its state.\n\nThis suggests the problem is intermittent. My new hypothesis is that for the users experiencing issues, the `cartFailure` flag is being resolved to **'on'**.\n\nTo confirm this, I will now search for traces where the `cartFailure` flag's variant is **'on'**. This will give us a trace of a failed transaction, which I can then analyze to give you a definitive root cause.\n...\nI have now checked the logs for all the services that are involved in the checkout process, and none of them are reporting any errors. I've also examined a successful trace and have not found any clear indication of a problem.\n\nAt this point, it's clear that the issue is not a simple error that is being logged or traced. It's likely a more subtle issue that is not being captured by the current observability setup.\n\nTo help you resolve this issue, I recommend that you do the following:\n\n1. **Add more detailed logging to the `checkout` service.** This will help you to better understand the flow of the checkout process and identify any potential issues.\n2. **Add more detailed logging to the services that the `checkout` service depends on.** This will help you to identify any potential issues in these services.\n3. **Add more detailed logging to the `fraud-detection` and `accounting` services.** This will help you to identify any potential issues in these services.\n4. **Reproduce the issue in a test environment.** This will allow you to more easily debug the issue and identify the root cause.\n\nI understand that this is not the answer that you were hoping for, but I have exhausted all of the available data and have not been able to find the root cause of the issue. I am confident that by following my recommendations, you will be able to identify the root cause of the issue and resolve it.\n\u003c/div\u003e\u003c/pre\u003e\n\u003c/details\u003e\n\n#### LLM investigation summary \n\n| Model | Resolution | # of MCP calls | Duration | Input tokens | Output tokens | Cost |\n|-------|------------|----------------|----------|--------------|---------------|------|\n| Claude 4 sonnet | Yes | 17 | 3 minutes | 1969111 | 3898 | $5.97 |\n| OpenAI o3 | Yes | 21 | 3 minutes | 298908 | 2687 | $0.77 |\n| OpenAI GPT4.1 | Yes, with minor guidance | 13 | 6 minutes | 92145 | 2625 | $0.21 |\n| Gemini Pro | No | 27 | 7 minutes | 1410381 | 59570 | $4.42 |\n\n## Comparing the results\n\nWe've challenged the models using 4 different types of anomalies and they performed at various levels.\n\nLet's compare the models by using the following scoring criterias:\n\n1\\. Root Cause Identification (5 points)\n\n- 0 = Did not find the root cause.\n- 1.5 = Find the root cause, but with major guidance.\n- 3 = Find the root cause, but with minor guidance\n- 5 = Find the root cause independently (no guidance).\n\n2\\. Number of MCP Calls (1 point): Fewer calls suggest more efficient reasoning.\n\n- 1 = 10 or fewer calls\n- 0.5 = 10--15 calls\n- 0 = Over 15 calls\n\n3\\. Resolution Time (1 point)\n\n- 1 = ≤ 3 minutes\n- 0.5 = 4--6 minutes\n- 0 = \u003e 6 minutes\n\n4\\. Cost Efficiency (1.5 points)\n\n- 1.5 = \u003c= $1\n- 0.75 = $1--$3\n- 0 = \u003e $3\n\n5\\. Token Efficiency (1.5 points): Total tokens (input + output). Lower is better.\n\n- 1.5 = \u003c 200,000\n- 0.75 = 200,000--1,000,000\n- 0 = \u003e 1,000,000\n\nLet's apply the score system to the result from the challenges.\n\n### Model comparison\n\n\n\nUsing this scoring system, the o3 model ranked highest in our benchmark. Its investigation capabilities are on par with Claude Sonnet 4, but it uses fewer tokens.\n\nGPT4.1 is an interesting option as it is the most cost-effective. However, its investigation capabilities fall significantly short compared to Sonnet 4 and o3, the most advanced models in our evaluation.\n\n### RCA comparison\n\n\n\nClaude Sonnet 4 and OpenAI o3 perform best at investigating anomalies and identifying root causes. Interestingly, their reasoning ability is not necessarily higher than Gemini 2.5 Pro, yet they still achieve better results. This could suggest that a model’s measured “IQ” is not a strong indicator of its actual success rate.\n\n### Cost comparison\n\nVisualizing the cost per investigation makes it easier to compare differences and highlight unpredictability across models.\n\n\n\nOpenAI’s models, especially GPT-4.1, tend to cost less because they use fewer tokens during investigations than Claude Sonnet 4 and Gemini 2.5 Pro.\n\n## What did we learn?\n\n### Models are not ready\n\nThese experiments showed that general-purpose LLMs are not yet reliable enough to act as fully autonomous SRE agents.\n\nWe tested the models on four small datasets---each typically representing about an hour of telemetry data---with anomalies we intentionally injected. These synthetic anomalies were relatively easy to detect, with minimal noise from unrelated background issues. This setup does not reflect the complexity of real-world production environments.\n\nEven with these simplified conditions, none of the models consistently identified the root cause without guidance. In two of the four scenarios, the more advanced models required additional guidance to reach the correct conclusion. Some models failed to identify the root cause altogether, even with significant guidance.\n\nThat said, our experiments took a naive approach. We didn't apply advanced techniques like context enrichment, prompt engineering, or fine-tuning. We recognize that these strategies could significantly improve model performance. \n\n### Fast database is crucial\n\nInvestigating each issue required between 6 and 27 database queries. While our datasets were small, real-world telemetry workloads are far larger.\n\nGiving LLMs direct access to databases in observability workflows will significantly increase query load. As these systems scale, the database must be able to handle the added load without compromising latency. Fast, scalable database performance will be essential for making these kinds of LLM integrations practical.\n\n### Unpredictable token usage (cost)\n\nToken usage varied widely across models and scenarios, from a few thousand tokens to several million, depending on the model, the number of prompts, and how many times the system used external tools (MCP calls).\n\nWhat drives this token consumption isn't always obvious. It's hard to estimate or predict how much a model will use in a given situation, and that makes it difficult to control or forecast costs.\n\nThis unpredictability makes LLMs risky to include in automated observability workflows, especially when cost is a concern.\n\n## Shifting focus: From RCA detection to documentation\n\nSince LLMs didn't prove suitable as fully autonomous SRE agents, we explored whether they could still add value in a supporting role, specifically, by handling simpler but time-consuming tasks.\n\nTo test this, we ran an experiment where the model was tasked with writing the RCA (Root Cause Analysis) report based on the findings from our manual investigation. \n\n### Experiment walkthrough\n\nLet's walk through one example, we'll use the demo anomaly dataset and Claude Sonnet 4. We gave the following prompt to the LLM.\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eYou’re an observability agent, and will help me writing up a RCA report for an issue we investigated recently on our application.\n\nYou have access to the raw OTel data that you can query using ClickHouse.\n\nUsers have been reporting issues when checking out their order. \nWe found out that their was a bad cache implementation in the payment service.\n\nPlease get the data you need from the ClickHouse to write the RCA report.\n\u003c/div\u003e\u003c/pre\u003e\n\nThe model went on and query the database to get all the logs, traces and metrics to understand the error cause and impact on the system. Then produce a comprehensive report.\n\n\u003cpre\u003e\u003cdiv class=\"llm-snippet\"\u003eRoot Cause Analysis Report\nPayment Service Cache Implementation Issue\n⸻\nExecutive Summary\n\nOn July 28–29, 2025, our e-commerce platform experienced significant checkout failures due to a poorly implemented cache mechanism in the payment service. \nThe issue resulted in 96,749 failed payment transactions over approximately 5 hours, with error rates reaching 40% during peak impact periods.\n⸻\nIncident Timeline\n\t• Start Time: July 28, 2025 at 23:04:49 UTC\n\t• End Time: July 29, 2025 at 03:51:00 UTC (after fix deployment)\n\t• Duration: ~5.75 hours\n\t• Peak Impact: July 29, 00:00–03:00 UTC\n⸻\nImpact Assessment\n\nUser Impact\n\t• Failed Checkouts: 96,749 checkout attempts failed\n\t• Error Rate: Up to 40% of all payment processing requests\n\t• Affected Users: Customers attempting to complete purchases during the incident window\n\t• Revenue Impact: Significant loss due to failed transactions\n⸻\nPerformance Metrics\n\t• Average Checkout Duration: 233ms for failed transactions\n\t• Payment Service Duration: 106ms average for cache errors\n\t• Transaction Volume: ~25,000 checkout attempts per hour during incident\n⸻\nRoot Cause Analysis\n\nPrimary Root Cause\n\nThe payment service implemented a poorly designed Visa payment cache with the following critical flaws:\n\t1. Fixed Cache Size: The cache had a hard limit without proper eviction policies\n\t2. No Cache Management: No mechanism to remove old or expired entries\n\t3. Blocking Behavior: When full, the cache rejected new payment requests instead of evicting old entries\n\u003c/div\u003e\u003c/pre\u003e\n\n\nWe found the results to be consistently strong across different models and anomaly types. What stands out is that all models were able to produce a solid first draft of a root cause analysis (RCA) report from the initial prompt. This makes the use case a good candidate for automation within an SRE workflow.\n\nThe process could be further improved by including a report template aligned with internal reporting standards or company policy. \n\n## One more candidate: GPT-5\n\nJust as we were preparing this blog post, OpenAI released GPT-5. We couldn’t resist running the experiment with it. It might actually perform best and change how we see LLMs as autonomous observability agents.\n\nLet’s see how it did against the other models. The table below summarizes its performance across the datasets.\n\n\n\nGPT-5 is essentially neck-and-neck with OpenAI o3. To be fair, the raw numbers show it uses significantly fewer tokens, but our scoring isn’t fine-grained enough to capture that. We can declare it the winner. \n\nBut still, even the latest model doesn’t find the root cause every time. That supports our impression from this experiment: performance isn’t strictly IQ-bound.\n\n## Closing thoughts\n\n\u003e **Short answer**: LLMs aren’t ready to run root cause analysis on their own. They are useful assistants for the people who do.\n\nOur setup was simple by design: raw telemetry and a plain prompt—no context enrichment, no tools, no fine-tuning. Under those conditions, every model missed anomalies at times, and some hallucinated causes.\nWe also tried a newer frontier model (GPT-5). It didn’t outperform the original contenders. The bottleneck isn’t model IQ; it’s missing context, weak grounding, and no domain specialization. \n\nMany companies are testing LLM-based observability because the promise is appealing: find production issues faster and at lower cost. We are too. But giving access to your observability data to a LLM and calling it a day does not work. Advanced approaches—context enrichment, domain-tuned models, and function calls into observability tools—can help, but they add cost and operational overhead and still depend on clean, well-indexed data and a clear system view.\n\nWhat works today is engineers + platform + speed (+ LLMs). Give engineers an observability interface on top of a fast analytical database so they can slice logs, metrics, and traces across large windows in seconds. In that same interface, the LLM handles the busywork while staying in the loop:\n\n- Summarize noisy logs and traces.\n- Draft status updates and post-mortem sections.\n- Suggest an investigation plan to follow\n- Review investigation data and validate findings\n\nThe value isn’t the LLM alone, it’s the shared interface that enables real collaboration.\n\nSo can LLMs replace SREs right now? No. \n\nCan they shorten incidents and improve documentation when paired with a fast observability stack? Yes. \n\nThe path forward is better context and better tools, with engineers in control.\n\n\n\n\n14:[[\"$\",\"$L34\",null,{\"name\":\"blogPage\"}],[\"$\",\"script\",null,{\"type\":\"application/ld+json\",\"dangerouslySetInnerHTML\":{\"__html\":\"$35\"}}],[\"$\",\"$L36\",null,{\"entry\":{\"id\":579,\"category\":\"Engineering\",\"title\":\"Can LLMs replace on call SREs today?\",\"shortDescription\":\"We often hear that LLMs will soon replace SREs. We wanted to test that claim, so we ran an experiment. Read the blog to see what we found.\",\"content\":\"$37\",\"createdAt\":\"2025-08-13T14:31:23.788Z\",\"updatedAt\":\"2026-03-03T12:38:31.061Z\",\"publishedAt\":\"2025-08-13T16:18:03.962Z\",\"slug\":\"llm-observability-challenge\",\"date\":\"2025-08-14\",\"keywords\":null,\"StagingOnly\":false,\"ShowCloudCTAHeader\":null,\"ShowCloudCTAFooter\":null,\"reading_time\":82,\"theme\":\"Feature Deep-dive\",\"use_case\":\"Logs, Metrics, \u0026 Traces\",\"canonical_url\":null,\"table_contents_headers\":\"h1, h2, h3\",\"ListOnBlogs\":true,\"enableSidebarGlobalCta\":false,\"reading_time_override\":21,\"language\":\"English\",\"documentId\":\"xw6qjaedyua0rsjci6uz03ur\",\"time\":null,\"thumbnailPng\":{\"id\":5237,\"name\":\"llm-observability-banner.png\",\"alternativeText\":null,\"caption\":null,\"width\":1200,\"height\":630,\"formats\":{\"large\":{\"ext\":\".png\",\"url\":\"/uploads/large_llm_observability_banner_22585788cf.png\",\"hash\":\"large_llm_observability_banner_22585788cf\",\"mime\":\"image/png\",\"name\":\"large_llm-observability-banner.png\",\"path\":null,\"size\":381.06,\"width\":1000,\"height\":525,\"sizeInBytes\":381059},\"small\":{\"ext\":\".png\",\"url\":\"/uploads/small_llm_observability_banner_22585788cf.png\",\"hash\":\"small_llm_observability_banner_22585788cf\",\"mime\":\"image/png\",\"name\":\"small_llm-observability-banner.png\",\"path\":null,\"size\":141.77,\"width\":500,\"height\":263,\"sizeInBytes\":141770},\"medium\":{\"ext\":\".png\",\"url\":\"/uploads/medium_llm_observability_banner_22585788cf.png\",\"hash\":\"medium_llm_observability_banner_22585788cf\",\"mime\":\"image/png\",\"name\":\"medium_llm-observability-banner.png\",\"path\":null,\"size\":253.51,\"width\":750,\"height\":394,\"sizeInBytes\":253505},\"thumbnail\":{\"ext\":\".png\",\"url\":\"/uploads/thumbnail_llm_observability_banner_22585788cf.png\",\"hash\":\"thumbnail_llm_observability_banner_22585788cf\",\"mime\":\"image/png\",\"name\":\"thumbnail_llm-observability-banner.png\",\"path\":null,\"size\":40.11,\"width\":245,\"height\":129,\"sizeInBytes\":40112}},\"hash\":\"llm_observability_banner_22585788cf\",\"ext\":\".png\",\"mime\":\"image/png\",\"size\":80.58,\"url\":\"/uploads/llm_observability_banner_22585788cf.png\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2025-08-14T09:37:47.487Z\",\"updatedAt\":\"2025-08-14T09:37:47.487Z\",\"documentId\":\"eeogmllxekh2r80er9rsmdti\",\"publishedAt\":\"2026-03-23T16:32:28.245Z\",\"focalPoint\":null},\"promotion\":null,\"sections\":[],\"author\":{\"id\":579,\"name\":\"Lionel Palacin and Al Brown\",\"profileLink\":null,\"avatarPng\":[{\"id\":3497,\"name\":\"lio-headshot-singapore.jpg\",\"alternativeText\":null,\"caption\":null,\"width\":2056,\"height\":1773,\"formats\":{\"large\":{\"ext\":\".jpg\",\"url\":\"/uploads/large_lio_headshot_singapore_7cc9852011.jpg\",\"hash\":\"large_lio_headshot_singapore_7cc9852011\",\"mime\":\"image/jpeg\",\"name\":\"large_lio-headshot-singapore.jpg\",\"path\":null,\"size\":77.21,\"width\":1000,\"height\":862,\"sizeInBytes\":77214},\"small\":{\"ext\":\".jpg\",\"url\":\"/uploads/small_lio_headshot_singapore_7cc9852011.jpg\",\"hash\":\"small_lio_headshot_singapore_7cc9852011\",\"mime\":\"image/jpeg\",\"name\":\"small_lio-headshot-singapore.jpg\",\"path\":null,\"size\":22.14,\"width\":500,\"height\":431,\"sizeInBytes\":22135},\"medium\":{\"ext\":\".jpg\",\"url\":\"/uploads/medium_lio_headshot_singapore_7cc9852011.jpg\",\"hash\":\"medium_lio_headshot_singapore_7cc9852011\",\"mime\":\"image/jpeg\",\"name\":\"medium_lio-headshot-singapore.jpg\",\"path\":null,\"size\":45.88,\"width\":750,\"height\":647,\"sizeInBytes\":45882},\"thumbnail\":{\"ext\":\".jpg\",\"url\":\"/uploads/thumbnail_lio_headshot_singapore_7cc9852011.jpg\",\"hash\":\"thumbnail_lio_headshot_singapore_7cc9852011\",\"mime\":\"image/jpeg\",\"name\":\"thumbnail_lio-headshot-singapore.jpg\",\"path\":null,\"size\":4.52,\"width\":181,\"height\":156,\"sizeInBytes\":4520}},\"hash\":\"lio_headshot_singapore_7cc9852011\",\"ext\":\".jpg\",\"mime\":\"image/jpeg\",\"size\":266.84,\"url\":\"/uploads/lio_headshot_singapore_7cc9852011.jpg\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2024-12-11T10:23:57.617Z\",\"updatedAt\":\"2025-11-27T13:40:04.710Z\",\"documentId\":\"h7ln6tnqdrqsatw7yrpwgbt8\",\"publishedAt\":\"2026-03-23T16:32:28.245Z\",\"focalPoint\":null},{\"id\":4957,\"name\":\"al-brown-headshot\",\"alternativeText\":\"Al Brown\",\"caption\":null,\"width\":512,\"height\":512,\"formats\":{\"small\":{\"ext\":\".jpg\",\"url\":\"/uploads/small_al_brown_headshot_09ae0cbce6.jpg\",\"hash\":\"small_al_brown_headshot_09ae0cbce6\",\"mime\":\"image/jpeg\",\"name\":\"small_al-brown-headshot\",\"path\":null,\"size\":34.73,\"width\":500,\"height\":500,\"sizeInBytes\":34725},\"thumbnail\":{\"ext\":\".jpg\",\"url\":\"/uploads/thumbnail_al_brown_headshot_09ae0cbce6.jpg\",\"hash\":\"thumbnail_al_brown_headshot_09ae0cbce6\",\"mime\":\"image/jpeg\",\"name\":\"thumbnail_al-brown-headshot\",\"path\":null,\"size\":4.81,\"width\":156,\"height\":156,\"sizeInBytes\":4811}},\"hash\":\"al_brown_headshot_09ae0cbce6\",\"ext\":\".jpg\",\"mime\":\"image/jpeg\",\"size\":26.51,\"url\":\"/uploads/al_brown_headshot_09ae0cbce6.jpg\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2025-07-16T12:04:06.145Z\",\"updatedAt\":\"2025-07-16T12:04:06.145Z\",\"documentId\":\"h9ydniag67r92mfaztfk41sl\",\"publishedAt\":\"2026-03-23T16:32:28.245Z\",\"focalPoint\":null}],\"profiles\":[{\"id\":46,\"name\":\"Lionel Palacin\",\"title\":\"Senior Product Marketing Engineer at ClickHouse\",\"description\":\"Lionel Palacin is a product marketing engineer at ClickHouse where he builds public demos to show users the product in action and write blog articles about it. Before that, he spent several years at Elastic working with customers and writing about the stack.\",\"linkedinUrl\":\"https://www.linkedin.com/in/lionelpalacin/\",\"twitterUrl\":null,\"githubUrl\":\"https://github.com/lio-p\",\"instagramUrl\":null,\"websiteUrl\":null,\"slug\":\"lionel-palacin\",\"createdAt\":\"2026-01-21T11:24:43.379Z\",\"updatedAt\":\"2026-02-24T15:46:42.867Z\",\"documentId\":\"ibmrx2xg3x25am08qs5rl1ap\",\"publishedAt\":\"2026-03-23T16:32:28.756Z\",\"avatar\":{\"id\":3497,\"name\":\"lio-headshot-singapore.jpg\",\"alternativeText\":null,\"caption\":null,\"width\":2056,\"height\":1773,\"formats\":{\"large\":{\"ext\":\".jpg\",\"url\":\"/uploads/large_lio_headshot_singapore_7cc9852011.jpg\",\"hash\":\"large_lio_headshot_singapore_7cc9852011\",\"mime\":\"image/jpeg\",\"name\":\"large_lio-headshot-singapore.jpg\",\"path\":null,\"size\":77.21,\"width\":1000,\"height\":862,\"sizeInBytes\":77214},\"small\":{\"ext\":\".jpg\",\"url\":\"/uploads/small_lio_headshot_singapore_7cc9852011.jpg\",\"hash\":\"small_lio_headshot_singapore_7cc9852011\",\"mime\":\"image/jpeg\",\"name\":\"small_lio-headshot-singapore.jpg\",\"path\":null,\"size\":22.14,\"width\":500,\"height\":431,\"sizeInBytes\":22135},\"medium\":{\"ext\":\".jpg\",\"url\":\"/uploads/medium_lio_headshot_singapore_7cc9852011.jpg\",\"hash\":\"medium_lio_headshot_singapore_7cc9852011\",\"mime\":\"image/jpeg\",\"name\":\"medium_lio-headshot-singapore.jpg\",\"path\":null,\"size\":45.88,\"width\":750,\"height\":647,\"sizeInBytes\":45882},\"thumbnail\":{\"ext\":\".jpg\",\"url\":\"/uploads/thumbnail_lio_headshot_singapore_7cc9852011.jpg\",\"hash\":\"thumbnail_lio_headshot_singapore_7cc9852011\",\"mime\":\"image/jpeg\",\"name\":\"thumbnail_lio-headshot-singapore.jpg\",\"path\":null,\"size\":4.52,\"width\":181,\"height\":156,\"sizeInBytes\":4520}},\"hash\":\"lio_headshot_singapore_7cc9852011\",\"ext\":\".jpg\",\"mime\":\"image/jpeg\",\"size\":266.84,\"url\":\"/uploads/lio_headshot_singapore_7cc9852011.jpg\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2024-12-11T10:23:57.617Z\",\"updatedAt\":\"2025-11-27T13:40:04.710Z\",\"documentId\":\"h7ln6tnqdrqsatw7yrpwgbt8\",\"publishedAt\":\"2026-03-23T16:32:28.245Z\",\"focalPoint\":null}},{\"id\":49,\"name\":\"Al Brown\",\"title\":\"Product Marketing Engineer at ClickHouse\",\"description\":\"I've spent the past 10 years working in the data analytics \u0026 engineering space, architecting nation-state cyber defence systems, consulting for global financial institutions, and leading developer relations \u0026 experience teams at early stage startups. \",\"linkedinUrl\":\"https://www.linkedin.com/in/alasdair-brown/\",\"twitterUrl\":null,\"githubUrl\":null,\"instagramUrl\":null,\"websiteUrl\":\"https://alasdairb.com/\",\"slug\":\"al-brown\",\"createdAt\":\"2026-01-21T11:24:43.420Z\",\"updatedAt\":\"2026-01-22T15:47:27.487Z\",\"documentId\":\"udfm6ane4lenjol9cd8eorjw\",\"publishedAt\":\"2026-03-23T16:32:28.756Z\",\"avatar\":{\"id\":4957,\"name\":\"al-brown-headshot\",\"alternativeText\":\"Al Brown\",\"caption\":null,\"width\":512,\"height\":512,\"formats\":{\"small\":{\"ext\":\".jpg\",\"url\":\"/uploads/small_al_brown_headshot_09ae0cbce6.jpg\",\"hash\":\"small_al_brown_headshot_09ae0cbce6\",\"mime\":\"image/jpeg\",\"name\":\"small_al-brown-headshot\",\"path\":null,\"size\":34.73,\"width\":500,\"height\":500,\"sizeInBytes\":34725},\"thumbnail\":{\"ext\":\".jpg\",\"url\":\"/uploads/thumbnail_al_brown_headshot_09ae0cbce6.jpg\",\"hash\":\"thumbnail_al_brown_headshot_09ae0cbce6\",\"mime\":\"image/jpeg\",\"name\":\"thumbnail_al-brown-headshot\",\"path\":null,\"size\":4.81,\"width\":156,\"height\":156,\"sizeInBytes\":4811}},\"hash\":\"al_brown_headshot_09ae0cbce6\",\"ext\":\".jpg\",\"mime\":\"image/jpeg\",\"size\":26.51,\"url\":\"/uploads/al_brown_headshot_09ae0cbce6.jpg\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2025-07-16T12:04:06.145Z\",\"updatedAt\":\"2025-07-16T12:04:06.145Z\",\"documentId\":\"h9ydniag67r92mfaztfk41sl\",\"publishedAt\":\"2026-03-23T16:32:28.245Z\",\"focalPoint\":null}}]}},\"related\":[{\"id\":4574,\"documentId\":\"g07wbk8cfvvk59z7maslo4ul\",\"category\":\"Product\",\"title\":\"AI Functions in ClickHouse: Upgrade your SQL to the AI age\",\"slug\":\"ai-functions-in-clickhouse\",\"date\":\"2026-09-11\",\"publishedAt\":\"2026-09-11T12:49:32.253Z\",\"reading_time\":11,\"reading_time_override\":6,\"thumbnailPng\":{\"id\":9525,\"name\":\"AI Functions in ClickHouse.jpg\",\"alternativeText\":null,\"caption\":null,\"width\":1200,\"height\":628,\"formats\":{\"large\":{\"ext\":\".jpg\",\"url\":\"/uploads/large_AI_Functions_in_Click_House_eee27e19c2.jpg\",\"hash\":\"large_AI_Functions_in_Click_House_eee27e19c2\",\"mime\":\"image/jpeg\",\"name\":\"large_AI Functions in ClickHouse.jpg\",\"path\":null,\"size\":57.39,\"width\":1000,\"height\":523,\"sizeInBytes\":57391},\"small\":{\"ext\":\".jpg\",\"url\":\"/uploads/small_AI_Functions_in_Click_House_eee27e19c2.jpg\",\"hash\":\"small_AI_Functions_in_Click_House_eee27e19c2\",\"mime\":\"image/jpeg\",\"name\":\"small_AI Functions in ClickHouse.jpg\",\"path\":null,\"size\":22.85,\"width\":500,\"height\":262,\"sizeInBytes\":22849},\"medium\":{\"ext\":\".jpg\",\"url\":\"/uploads/medium_AI_Functions_in_Click_House_eee27e19c2.jpg\",\"hash\":\"medium_AI_Functions_in_Click_House_eee27e19c2\",\"mime\":\"image/jpeg\",\"name\":\"medium_AI Functions in ClickHouse.jpg\",\"path\":null,\"size\":38.86,\"width\":750,\"height\":393,\"sizeInBytes\":38859},\"thumbnail\":{\"ext\":\".jpg\",\"url\":\"/uploads/thumbnail_AI_Functions_in_Click_House_eee27e19c2.jpg\",\"hash\":\"thumbnail_AI_Functions_in_Click_House_eee27e19c2\",\"mime\":\"image/jpeg\",\"name\":\"thumbnail_AI Functions in ClickHouse.jpg\",\"path\":null,\"size\":8.36,\"width\":245,\"height\":128,\"sizeInBytes\":8363}},\"hash\":\"AI_Functions_in_Click_House_eee27e19c2\",\"ext\":\".jpg\",\"mime\":\"image/jpeg\",\"size\":72.99,\"url\":\"/uploads/AI_Functions_in_Click_House_eee27e19c2.jpg\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2026-09-04T14:55:41.888Z\",\"updatedAt\":\"2026-09-04T14:56:58.353Z\",\"documentId\":\"i1q0xtr4xg137142er4iqm0p\",\"publishedAt\":\"2026-09-04T14:55:41.888Z\",\"focalPoint\":null},\"author\":{\"id\":5894,\"name\":null,\"profileLink\":null,\"profiles\":[{\"id\":101,\"name\":\"Andriy Yakovlev\",\"title\":\"Senior Product Manager\",\"description\":null,\"linkedinUrl\":null,\"twitterUrl\":null,\"githubUrl\":null,\"instagramUrl\":null,\"websiteUrl\":null,\"slug\":\"andriy-yakovlev\",\"createdAt\":\"2026-05-28T08:53:41.212Z\",\"updatedAt\":\"2026-05-28T08:53:41.212Z\",\"documentId\":\"xifghut0u5as9xcjmi8ptu7g\",\"publishedAt\":\"2026-05-28T08:53:41.206Z\"},{\"id\":129,\"name\":\"George Larionov\",\"title\":null,\"description\":null,\"linkedinUrl\":null,\"twitterUrl\":null,\"githubUrl\":null,\"instagramUrl\":null,\"websiteUrl\":null,\"slug\":\"george-larionov\",\"createdAt\":\"2026-09-04T13:51:51.359Z\",\"updatedAt\":\"2026-09-04T13:51:51.359Z\",\"documentId\":\"nkzdv8jndaq59uas90s7lvmh\",\"publishedAt\":\"2026-09-04T13:51:51.353Z\"}]}},{\"id\":4578,\"documentId\":\"ozf1p4vhgcqf8rdox9xrrjsf\",\"category\":\"Engineering\",\"title\":\"Loading Parquet data into MySQL with ClickHouse\",\"slug\":\"parquet-to-mysql-with-clickhouse\",\"date\":\"2026-09-10\",\"publishedAt\":\"2026-09-11T13:17:45.862Z\",\"reading_time\":7,\"reading_time_override\":null,\"thumbnailPng\":{\"id\":9647,\"name\":\"Loading Parquet Data into MySQL.jpg\",\"alternativeText\":null,\"caption\":null,\"width\":1200,\"height\":628,\"formats\":{\"large\":{\"ext\":\".jpg\",\"url\":\"/uploads/large_Loading_Parquet_Data_into_My_SQL_9371e998c0.jpg\",\"hash\":\"large_Loading_Parquet_Data_into_My_SQL_9371e998c0\",\"mime\":\"image/jpeg\",\"name\":\"large_Loading Parquet Data into MySQL.jpg\",\"path\":null,\"size\":53.7,\"width\":1000,\"height\":523,\"sizeInBytes\":53697},\"small\":{\"ext\":\".jpg\",\"url\":\"/uploads/small_Loading_Parquet_Data_into_My_SQL_9371e998c0.jpg\",\"hash\":\"small_Loading_Parquet_Data_into_My_SQL_9371e998c0\",\"mime\":\"image/jpeg\",\"name\":\"small_Loading Parquet Data into MySQL.jpg\",\"path\":null,\"size\":21.92,\"width\":500,\"height\":262,\"sizeInBytes\":21916},\"medium\":{\"ext\":\".jpg\",\"url\":\"/uploads/medium_Loading_Parquet_Data_into_My_SQL_9371e998c0.jpg\",\"hash\":\"medium_Loading_Parquet_Data_into_My_SQL_9371e998c0\",\"mime\":\"image/jpeg\",\"name\":\"medium_Loading Parquet Data into MySQL.jpg\",\"path\":null,\"size\":36.64,\"width\":750,\"height\":393,\"sizeInBytes\":36637},\"thumbnail\":{\"ext\":\".jpg\",\"url\":\"/uploads/thumbnail_Loading_Parquet_Data_into_My_SQL_9371e998c0.jpg\",\"hash\":\"thumbnail_Loading_Parquet_Data_into_My_SQL_9371e998c0\",\"mime\":\"image/jpeg\",\"name\":\"thumbnail_Loading Parquet Data into MySQL.jpg\",\"path\":null,\"size\":7.98,\"width\":245,\"height\":128,\"sizeInBytes\":7982}},\"hash\":\"Loading_Parquet_Data_into_My_SQL_9371e998c0\",\"ext\":\".jpg\",\"mime\":\"image/jpeg\",\"size\":66.29,\"url\":\"/uploads/Loading_Parquet_Data_into_My_SQL_9371e998c0.jpg\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2026-09-10T14:34:43.804Z\",\"updatedAt\":\"2026-09-10T14:34:43.804Z\",\"documentId\":\"d3dcfftmt2j9lsl3fce013jb\",\"publishedAt\":\"2026-09-10T14:34:43.805Z\",\"focalPoint\":null},\"author\":{\"id\":5898,\"name\":null,\"profileLink\":null,\"profiles\":[{\"id\":34,\"name\":\"Mark Needham\",\"title\":null,\"description\":null,\"linkedinUrl\":null,\"twitterUrl\":null,\"githubUrl\":null,\"instagramUrl\":null,\"websiteUrl\":null,\"slug\":\"mark-needham\",\"createdAt\":\"2026-01-21T11:24:43.061Z\",\"updatedAt\":\"2026-01-21T11:24:43.061Z\",\"documentId\":\"ng4i8fgtdz3xzfaqmkko9jv4\",\"publishedAt\":\"2026-03-23T16:32:28.756Z\"}]}},{\"id\":4577,\"documentId\":\"gjwbg7v5ae57yubnbqi3m38y\",\"category\":\"Engineering\",\"title\":\"ClickHouse release 26.8\",\"slug\":\"clickhouse-release-26-08\",\"date\":\"2026-09-10\",\"publishedAt\":\"2026-09-11T13:12:38.318Z\",\"reading_time\":31,\"reading_time_override\":null,\"thumbnailPng\":{\"id\":9593,\"name\":\"Blog Cover 1200x630 (12).png\",\"alternativeText\":null,\"caption\":null,\"width\":1200,\"height\":630,\"formats\":{\"large\":{\"ext\":\".png\",\"url\":\"/uploads/large_Blog_Cover_1200x630_12_1ef0c32771.png\",\"hash\":\"large_Blog_Cover_1200x630_12_1ef0c32771\",\"mime\":\"image/png\",\"name\":\"large_Blog Cover 1200x630 (12).png\",\"path\":null,\"size\":99,\"width\":1000,\"height\":525,\"sizeInBytes\":99002},\"small\":{\"ext\":\".png\",\"url\":\"/uploads/small_Blog_Cover_1200x630_12_1ef0c32771.png\",\"hash\":\"small_Blog_Cover_1200x630_12_1ef0c32771\",\"mime\":\"image/png\",\"name\":\"small_Blog Cover 1200x630 (12).png\",\"path\":null,\"size\":40.56,\"width\":500,\"height\":263,\"sizeInBytes\":40557},\"medium\":{\"ext\":\".png\",\"url\":\"/uploads/medium_Blog_Cover_1200x630_12_1ef0c32771.png\",\"hash\":\"medium_Blog_Cover_1200x630_12_1ef0c32771\",\"mime\":\"image/png\",\"name\":\"medium_Blog Cover 1200x630 (12).png\",\"path\":null,\"size\":64.58,\"width\":750,\"height\":394,\"sizeInBytes\":64577},\"thumbnail\":{\"ext\":\".png\",\"url\":\"/uploads/thumbnail_Blog_Cover_1200x630_12_1ef0c32771.png\",\"hash\":\"thumbnail_Blog_Cover_1200x630_12_1ef0c32771\",\"mime\":\"image/png\",\"name\":\"thumbnail_Blog Cover 1200x630 (12).png\",\"path\":null,\"size\":16.59,\"width\":245,\"height\":129,\"sizeInBytes\":16591}},\"hash\":\"Blog_Cover_1200x630_12_1ef0c32771\",\"ext\":\".png\",\"mime\":\"image/png\",\"size\":22.76,\"url\":\"/uploads/Blog_Cover_1200x630_12_1ef0c32771.png\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2026-09-09T10:41:08.920Z\",\"updatedAt\":\"2026-09-09T10:41:08.920Z\",\"documentId\":\"duv4biwluekqwrr456l257ae\",\"publishedAt\":\"2026-09-09T10:41:08.920Z\",\"focalPoint\":null},\"author\":{\"id\":5897,\"name\":null,\"profileLink\":null,\"profiles\":[{\"id\":63,\"name\":\"ClickHouse\",\"title\":null,\"description\":null,\"linkedinUrl\":null,\"twitterUrl\":null,\"githubUrl\":null,\"instagramUrl\":null,\"websiteUrl\":null,\"slug\":\"clickhouse\",\"createdAt\":\"2026-01-22T14:47:12.301Z\",\"updatedAt\":\"2026-01-22T14:47:12.301Z\",\"documentId\":\"uoa7v3flq60io41y1yycj9ak\",\"publishedAt\":\"2026-03-23T16:32:28.756Z\"}]}},{\"id\":4570,\"documentId\":\"ryszzwfihxip57vc0f85umf6\",\"category\":\"Engineering\",\"title\":\"ClickHouse Cloud vs. Snowflake: What drives the real-time performance-per-dollar gap\",\"slug\":\"clickhouse-vs-snowflake-real-time-performance-per-dollar\",\"date\":\"2026-09-10\",\"publishedAt\":\"2026-09-11T11:47:54.832Z\",\"reading_time\":42,\"reading_time_override\":15,\"thumbnailPng\":{\"id\":9543,\"name\":\"ClickHouse Cloud vs. Snowflake_ What drives the real-time performance-per-dollar gap.jpg\",\"alternativeText\":null,\"caption\":null,\"width\":1200,\"height\":630,\"formats\":{\"large\":{\"ext\":\".jpg\",\"url\":\"/uploads/large_Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31.jpg\",\"hash\":\"large_Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31\",\"mime\":\"image/jpeg\",\"name\":\"large_ClickHouse Cloud vs. Snowflake_ What drives the real-time performance-per-dollar gap.jpg\",\"path\":null,\"size\":66.69,\"width\":1000,\"height\":525,\"sizeInBytes\":66695},\"small\":{\"ext\":\".jpg\",\"url\":\"/uploads/small_Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31.jpg\",\"hash\":\"small_Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31\",\"mime\":\"image/jpeg\",\"name\":\"small_ClickHouse Cloud vs. Snowflake_ What drives the real-time performance-per-dollar gap.jpg\",\"path\":null,\"size\":26.86,\"width\":500,\"height\":263,\"sizeInBytes\":26855},\"medium\":{\"ext\":\".jpg\",\"url\":\"/uploads/medium_Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31.jpg\",\"hash\":\"medium_Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31\",\"mime\":\"image/jpeg\",\"name\":\"medium_ClickHouse Cloud vs. Snowflake_ What drives the real-time performance-per-dollar gap.jpg\",\"path\":null,\"size\":46.18,\"width\":750,\"height\":394,\"sizeInBytes\":46180},\"thumbnail\":{\"ext\":\".jpg\",\"url\":\"/uploads/thumbnail_Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31.jpg\",\"hash\":\"thumbnail_Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31\",\"mime\":\"image/jpeg\",\"name\":\"thumbnail_ClickHouse Cloud vs. Snowflake_ What drives the real-time performance-per-dollar gap.jpg\",\"path\":null,\"size\":9.14,\"width\":245,\"height\":129,\"sizeInBytes\":9141}},\"hash\":\"Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31\",\"ext\":\".jpg\",\"mime\":\"image/jpeg\",\"size\":82.96,\"url\":\"/uploads/Click_House_Cloud_vs_Snowflake_What_drives_the_real_time_performance_per_dollar_gap_da26aa6c31.jpg\",\"previewUrl\":null,\"provider\":\"local\",\"provider_metadata\":null,\"createdAt\":\"2026-09-07T13:38:27.075Z\",\"updatedAt\":\"2026-09-07T13:38:27.075Z\",\"documentId\":\"mc2xd0q92rgd3odnhezp9a6m\",\"publishedAt\":\"2026-09-07T13:38:27.075Z\",\"focalPoint\":null},\"author\":{\"id\":5890,\"name\":null,\"profileLink\":null,\"profiles\":[{\"id\":64,\"name\":\"Tom Schreiber\",\"title\":\"Principal Product Marketing Engineer at ClickHouse\",\"description\":\"Tom is a database architect and researcher with a background spanning Siemens, MongoDB, Elastic, and academic work published at VLDB. He is a co-author of the first ClickHouse research paper and focuses on translating deep technical foundations into benchmarks, training, architectures, and explainers that drive real-world adoption.\",\"linkedinUrl\":\"https://www.linkedin.com/in/schreibertom1\",\"twitterUrl\":\"https://x.com/SchreiberTom1\",\"githubUrl\":null,\"instagramUrl\":null,\"websiteUrl\":null,\"slug\":\"tom-schreiber\",\"createdAt\":\"2026-01-22T15:01:31.408Z\",\"updatedAt\":\"2026-01-23T14:15:53.146Z\",\"documentId\":\"prq3wxj1nsvdmb8wl6ial8h8\",\"publishedAt\":\"2026-03-23T16:32:28.756Z\"},{\"id\":46,\"name\":\"Lionel Palacin\",\"title\":\"Senior Product Marketing Engineer at ClickHouse\",\"description\":\"Lionel Palacin is a product marketing engineer at ClickHouse where he builds public demos to show users the product in action and write blog articles about it. Before that, he spent several years at Elastic working with customers and writing about the stack.\",\"linkedinUrl\":\"https://www.linkedin.com/in/lionelpalacin/\",\"twitterUrl\":null,\"githubUrl\":\"https://github.com/lio-p\",\"instagramUrl\":null,\"websiteUrl\":null,\"slug\":\"lionel-palacin\",\"createdAt\":\"2026-01-21T11:24:43.379Z\",\"updatedAt\":\"2026-02-24T15:46:42.867Z\",\"documentId\":\"ibmrx2xg3x25am08qs5rl1ap\",\"publishedAt\":\"2026-03-23T16:32:28.756Z\"}]}}],\"archiveUrl\":\"/blog\",\"dictionary\":{\"backLabel\":\"Back\",\"breadcrumbLabel\":\"Blog\",\"cloudCta\":{\"header\":\"[Get started](https://clickhouse.cloud/signUp?loc=blog-cta-header\u0026utm_source=clickhouse\u0026utm_medium=web\u0026utm_campaign=blog) with ClickHouse Cloud today and receive $300 in credits. To learn more about our volume-based discounts, [contact us](/company/contact?loc=blog-cta-header) or visit our [pricing page](/pricing?loc=blog-cta-header).\",\"footer\":\"[Get started](https://clickhouse.cloud/signUp?loc=blog-cta-footer\u0026utm_source=clickhouse\u0026utm_medium=web\u0026utm_campaign=blog) with ClickHouse Cloud today and receive $300 in credits. At the end of your 30-day trial, continue with a pay-as-you-go plan, or [contact us](/company/contact?loc=blog-cta-footer) to learn more about our volume-based discounts. Visit our [pricing page](/pricing?loc=blog-cta-header) for details.\"},\"sidebarCta\":{\"title\":\"Get started today\",\"description\":\"Interested in seeing how ClickHouse works on your data? Get started with ClickHouse Cloud in minutes and receive $300 in free credits.\",\"ctaLabel\":\"Sign up\"},\"shareLabel\":\"Share this post\",\"related\":{\"heading\":\"Recent posts\",\"viewAllLabel\":\"View all Blogs\"}},\"globalDict\":{\"blogCategories\":{\"Product\":\"Product\",\"Community\":\"Community\",\"Engineering\":\"Engineering\",\"User stories\":\"User stories\",\"Company and culture\":\"Company and culture\"},\"readingTime\":\"{minutes} minutes read\"},\"sectionsSlot\":\"$undefined\"}],\"$L38\"]\n"])</script><script>self.__next_f.push([1,"38:[\"$\",\"div\",null,{\"className\":\"bg-primary pt-10 pb-12\",\"data-galaxy-component\":\"$undefined\",\"children\":[\"$\",\"div\",null,{\"className\":\"container space-y-4 text-center\",\"children\":[[\"$\",\"div\",null,{\"className\":\"rich-text rich-text-dark \",\"children\":[\"$\",\"h2\",null,{\"className\":\"text-eyebrow\",\"children\":\"Follow us\"}]}],[\"$\",\"div\",null,{\"className\":\"flex flex-wrap items-center justify-center gap-4 lg:gap-8\",\"children\":[[\"$\",\"$L22\",null,{\"title\":\"X\",\"href\":\"https://x.com/ClickhouseDB\",\"target\":\"_blank\",\"className\":\"flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800\",\"data-galaxy-event\":\"x\",\"children\":[\"$\",\"$L31\",null,{\"src\":{\"src\":\"/_next/static/immutable/media/x.3nm91lx52ia7n.svg\",\"width\":36,\"height\":33,\"blurWidth\":0,\"blurHeight\":0},\"alt\":\"X\",\"width\":32,\"height\":32,\"className\":\"m-auto size-8 object-scale-down object-center\"}]}],[\"$\",\"$L22\",null,{\"title\":\"Bluesky\",\"href\":\"https://bsky.app/profile/clickhouse.comn\",\"target\":\"_blank\",\"className\":\"flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800\",\"data-galaxy-event\":\"bluesky\",\"children\":[\"$\",\"$L31\",null,{\"src\":{\"src\":\"/_next/static/immutable/media/bluesky.292c8t8kns7n1.svg\",\"width\":38,\"height\":32,\"blurWidth\":0,\"blurHeight\":0},\"alt\":\"Bluesky\",\"width\":32,\"height\":32,\"className\":\"m-auto size-8 object-scale-down object-center\"}]}],[\"$\",\"$L22\",null,{\"title\":\"Slack\",\"href\":\"/slack\",\"target\":\"_blank\",\"className\":\"flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800\",\"data-galaxy-event\":\"slack\",\"children\":[\"$\",\"$L31\",null,{\"src\":{\"src\":\"/_next/static/immutable/media/slack.2_pspehyws_jz.svg\",\"width\":33,\"height\":32,\"blurWidth\":0,\"blurHeight\":0},\"alt\":\"Slack\",\"width\":32,\"height\":32,\"className\":\"m-auto size-8 object-scale-down object-center\"}]}],[\"$\",\"$L22\",null,{\"title\":\"Github\",\"href\":\"https://github.com/ClickHouse/ClickHouse\",\"target\":\"_blank\",\"className\":\"flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800\",\"data-galaxy-event\":\"github\",\"children\":[\"$\",\"$L31\",null,{\"src\":\"$2a:props:children:3:props:children:0:props:src\",\"alt\":\"Github\",\"width\":32,\"height\":32,\"className\":\"m-auto size-8 object-scale-down object-center\"}]}],[\"$\",\"$L22\",null,{\"title\":\"Telegram\",\"href\":\"https://telegram.me/clickhouse_en\",\"target\":\"_blank\",\"className\":\"flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800\",\"data-galaxy-event\":\"telegram\",\"children\":[\"$\",\"$L31\",null,{\"src\":{\"src\":\"/_next/static/immutable/media/telegram.0054ol1lpoidb.svg\",\"width\":32,\"height\":33,\"blurWidth\":0,\"blurHeight\":0},\"alt\":\"Telegram\",\"width\":32,\"height\":32,\"className\":\"m-auto size-8 object-scale-down object-center\"}]}],[\"$\",\"$L22\",null,{\"title\":\"Meetup\",\"href\":\"https://www.meetup.com/pro/clickhouse\",\"target\":\"_blank\",\"className\":\"flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800\",\"data-galaxy-event\":\"meetup\",\"children\":[\"$\",\"$L31\",null,{\"src\":{\"src\":\"/_next/static/immutable/media/meetup.2tf1x11zaaepo.svg\",\"width\":33,\"height\":33,\"blurWidth\":0,\"blurHeight\":0},\"alt\":\"Meetup\",\"width\":32,\"height\":32,\"className\":\"m-auto size-8 object-scale-down object-center\"}]}],[\"$\",\"$L22\",null,{\"title\":\"RSS\",\"href\":\"/rss.xml\",\"target\":\"_blank\",\"className\":\"flex size-16 rounded border border-neutral-700/80 bg-neutral-900 transition-colors hover:bg-neutral-800\",\"data-galaxy-event\":\"rss\",\"children\":[\"$\",\"$L31\",null,{\"src\":{\"src\":\"/_next/static/immutable/media/rss.1pu3aqlsglh-q.svg\",\"width\":32,\"height\":32,\"blurWidth\":0,\"blurHeight\":0},\"alt\":\"RSS\",\"width\":32,\"height\":32,\"className\":\"m-auto size-8 object-scale-down object-center\"}]}]]}]]}]}]\n"])</script></body>
|
||
<!--
|
||
Apply the stylesheet late so it does not block initial render/LCP.
|
||
The earlier preload should make this reuse the already-fetched CSS.
|
||
-->
|
||
<link
|
||
rel="stylesheet"
|
||
href="https://discover.clickhouse.com/rs/238-FPC-317/images/securiti-cookie-banner-styles.css?version=0"
|
||
media="print"
|
||
onload="this.media='all'"
|
||
data-cookie-banner-css
|
||
>
|
||
<noscript>
|
||
<link
|
||
rel="stylesheet"
|
||
href="https://discover.clickhouse.com/rs/238-FPC-317/images/securiti-cookie-banner-styles.css?version=0"
|
||
data-cookie-banner-css-noscript
|
||
>
|
||
</noscript>
|
||
</html> |