3502 lines
147 KiB
HTML
3502 lines
147 KiB
HTML
<!DOCTYPE html>
|
||
<html>
|
||
<head>
|
||
<meta charset="utf-8">
|
||
|
||
|
||
<title>Docker lazy loading at Grab: Accelerating container startup times</title>
|
||
|
||
<meta name="description" content="Large container images were causing slow cold starts and poor auto-scaling for Grab's data platforms. This post explores how we implemented Docker image lazy...">
|
||
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
||
<meta name="theme-color" content="#00b14f" />
|
||
<meta http-equiv="X-UA-Compatible" content="IE=edge">
|
||
|
||
<link rel="canonical" href="https://engineering.grab.com/docker-lazy-loading">
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<script type="application/ld+json">
|
||
{
|
||
"@context": "https://schema.org",
|
||
"@type": "Organization",
|
||
"name": "Grab Tech",
|
||
"url": "https://engineering.grab.com",
|
||
"logo": {
|
||
"@type": "ImageObject",
|
||
"url": "https://engineering.grab.com/img/banner.png"
|
||
}
|
||
}
|
||
</script>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
<script type="application/ld+json">
|
||
{
|
||
"@context": "https://schema.org",
|
||
"@type": "BlogPosting",
|
||
"headline": "Docker lazy loading at Grab: Accelerating container startup times",
|
||
"description": "Large container images were causing slow cold starts and poor auto-scaling for Grab's data platforms. This post explores how we implemented Docker image lazy...",
|
||
"url": "https://engineering.grab.com/docker-lazy-loading",
|
||
"image": "https://engineering.grab.com/img/docker-lazy-loading/banner-1.png",
|
||
"datePublished": "2026-01-21T00:23:00+00:00",
|
||
"dateModified": "2026-01-21T00:23:00+00:00",
|
||
"author": {
|
||
"@type": "Person",
|
||
"name": "Huong Vuong"
|
||
},
|
||
"publisher": {
|
||
"@type": "Organization",
|
||
"name": "Grab Tech",
|
||
"logo": {
|
||
"@type": "ImageObject",
|
||
"url": "https://engineering.grab.com/img/banner.png"
|
||
}
|
||
},
|
||
"mainEntityOfPage": {
|
||
"@type": "WebPage",
|
||
"@id": "https://engineering.grab.com/docker-lazy-loading"
|
||
}
|
||
}
|
||
</script>
|
||
|
||
|
||
|
||
<script type="application/ld+json">
|
||
{
|
||
"@context": "https://schema.org",
|
||
"@type": "BreadcrumbList",
|
||
"itemListElement": [
|
||
{
|
||
"@type": "ListItem",
|
||
"position": 1,
|
||
"name": "Home",
|
||
"item": "https://engineering.grab.com"
|
||
},
|
||
{
|
||
"@type": "ListItem",
|
||
"position": 2,
|
||
"name": "Docker lazy loading at Grab: Accelerating container startup times",
|
||
"item": "https://engineering.grab.com/docker-lazy-loading"
|
||
}
|
||
]
|
||
}
|
||
</script>
|
||
|
||
|
||
|
||
<!-- Open Graph -->
|
||
<meta property="og:url" content="https://engineering.grab.com/docker-lazy-loading">
|
||
<meta property="og:title" content="Docker lazy loading at Grab: Accelerating container startup times">
|
||
<meta property="og:description" content="Large container images were causing slow cold starts and poor auto-scaling for Grab's data platforms. This post explores how we implemented Docker image lazy...">
|
||
<meta property="og:site_name" content="Grab Tech">
|
||
<meta property="og:type" content="article">
|
||
<meta property="og:image" content="https://engineering.grab.com/img/docker-lazy-loading/banner-1.png">
|
||
<meta property="og:image:width" content="1200">
|
||
<meta property="og:image:height" content="630">
|
||
<meta property="og:image:alt" content="Docker lazy loading at Grab: Accelerating container startup times">
|
||
|
||
<!-- Twitter -->
|
||
<meta name="twitter:card" content="summary_large_image">
|
||
<meta name="twitter:title" content="Docker lazy loading at Grab: Accelerating container startup times">
|
||
<meta name="twitter:description" content="Large container images were causing slow cold starts and poor auto-scaling for Grab's data platforms. This post explores how we implemented Docker image lazy...">
|
||
<meta name="twitter:image" content="https://engineering.grab.com/img/docker-lazy-loading/banner-1.png">
|
||
<meta name="twitter:image:alt" content="Docker lazy loading at Grab: Accelerating container startup times">
|
||
<meta name="twitter:site" content="@grabengineering">
|
||
|
||
<!-- Favicons -->
|
||
<link rel="icon" href="/favicon.ico">
|
||
<link rel="apple-touch-icon" href="/apple-touch-icon.png">
|
||
|
||
<!-- Preconnect for efficient external loading -->
|
||
<link rel="preconnect" href="https://fonts.googleapis.com" crossorigin>
|
||
<link rel="preconnect" href="https://fonts.gstatic.com" crossorigin>
|
||
<link rel="preconnect" href="https://maxcdn.bootstrapcdn.com" crossorigin>
|
||
<link rel="preconnect" href="https://code.jquery.com" crossorigin>
|
||
<link rel="dns-prefetch" href="//fonts.googleapis.com">
|
||
<link rel="dns-prefetch" href="//maxcdn.bootstrapcdn.com">
|
||
<link rel="dns-prefetch" href="//code.jquery.com">
|
||
|
||
<!-- CSS -->
|
||
<link href="https://fonts.googleapis.com/css?family=Droid+Serif:400,400i,700,700i&display=swap" rel="stylesheet">
|
||
<link rel="stylesheet" href="https://maxcdn.bootstrapcdn.com/bootstrap/3.3.1/css/bootstrap.min.css">
|
||
<link rel="stylesheet" href="https://maxcdn.bootstrapcdn.com/bootstrap/3.3.1/css/bootstrap-theme.min.css">
|
||
<script src="https://code.jquery.com/jquery-1.11.2.min.js" defer></script>
|
||
<script src="https://maxcdn.bootstrapcdn.com/bootstrap/3.3.1/js/bootstrap.min.js" defer></script>
|
||
<link href="https://maxcdn.bootstrapcdn.com/font-awesome/4.2.0/css/font-awesome.min.css" rel="stylesheet">
|
||
|
||
<link rel="stylesheet" href="/css/main.css">
|
||
|
||
<!-- RSS -->
|
||
<link rel="alternate" type="application/rss+xml" title="RSS for Official Grab Tech Blog" href="/feed.xml">
|
||
<!-- OneTrust Cookies Consent Notice (production only; skipped on jekyll serve) -->
|
||
|
||
<script type="text/javascript" src="https://cdn-apac.onetrust.com/consent/a3be3527-7455-48e0-ace6-557ddbd506d5/OtAutoBlock.js" ></script>
|
||
<script src="https://cdn-apac.onetrust.com/scripttemplates/otSDKStub.js" data-document-language="true" type="text/javascript" charset="UTF-8" data-domain-script="a3be3527-7455-48e0-ace6-557ddbd506d5" ></script>
|
||
<script type="text/javascript">
|
||
function OptanonWrapper() { }
|
||
</script>
|
||
|
||
</head>
|
||
|
||
<body>
|
||
<header class="site-header">
|
||
<div class="wrapper-navbar">
|
||
<div class="site-title-wrapper">
|
||
<div class="row site-title-wrapper-inner">
|
||
<div class="col-xs-1 visible-xs hamburger-nav" id="mobile-menu-btn">
|
||
<div class="menu-btn"></div>
|
||
</div>
|
||
<div class="col-sm-3 col-xs-3">
|
||
<div class="site-title-container">
|
||
<a class="site-title" href="/"></a>
|
||
<span class="site-subtitle"> Tech Blog</span>
|
||
</div>
|
||
</div>
|
||
<div class="col-sm-9 col-xs-8 text-right">
|
||
<ul class="nav-category hidden-xs">
|
||
|
||
|
||
<li>
|
||
<a href="/categories/engineering/">Engineering</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="/categories/data-science/">Data Science</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="/categories/design/">Design</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="/categories/product/">Product</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="/categories/security/">Security</a>
|
||
</li>
|
||
|
||
</ul>
|
||
<div class="site-search text-right">
|
||
<div class="blog-search" id="blog-search-container">
|
||
<button
|
||
type="button"
|
||
class="blog-search-trigger"
|
||
id="blog-search-trigger"
|
||
aria-haspopup="dialog"
|
||
aria-expanded="false"
|
||
aria-controls="blog-search-modal"
|
||
aria-label="Open search"
|
||
>
|
||
<svg class="blog-search-trigger-icon" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true">
|
||
<circle cx="11" cy="11" r="7"></circle>
|
||
<line x1="21" y1="21" x2="16.65" y2="16.65"></line>
|
||
</svg>
|
||
<kbd class="blog-search-kbd" data-blog-search-shortcut>Ctrl K</kbd>
|
||
</button>
|
||
</div>
|
||
|
||
</div>
|
||
</div>
|
||
</div>
|
||
<!-- Only visible on mobile view -->
|
||
<div class="mobile-menu-container">
|
||
<ul class="mobile-menu">
|
||
|
||
|
||
<li>
|
||
<a href="/categories/engineering/">Engineering</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="/categories/data-science/">Data Science</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="/categories/design/">Design</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="/categories/product/">Product</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="/categories/security/">Security</a>
|
||
</li>
|
||
|
||
</ul>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
</header>
|
||
<script src="/js/main.js" defer></script>
|
||
|
||
<div class="page-content">
|
||
|
||
<div class="wrapper post-page-wrapper">
|
||
<div class="post-with-sidebar">
|
||
<div class="post post-main">
|
||
<header class="post-header">
|
||
<img src="/img/docker-lazy-loading/banner-1.png" class="post-cover-photo" alt="Docker lazy loading at Grab: Accelerating container startup times cover photo" loading="eager" decoding="async" fetchpriority="high" width="1200" height="630">
|
||
|
||
<div class="post-tags">
|
||
|
||
|
||
<a href="/tags#database" class="label tags-label">Database</a>
|
||
|
||
</div>
|
||
|
||
|
||
|
||
<h1 class="post-title">Docker lazy loading at Grab: Accelerating container startup times</h1>
|
||
|
||
<div class="post-meta">
|
||
|
||
|
||
|
||
|
||
|
||
<div class="post-authors-row">
|
||
|
||
|
||
|
||
<div class="post-author-chip">
|
||
<button type="button"
|
||
class="post-author-chip-avatar-trigger"
|
||
data-toggle="tooltip"
|
||
data-placement="top"
|
||
data-trigger="hover focus"
|
||
aria-label="Huong Vuong"
|
||
title="Huong Vuong">
|
||
<img class="post-author-chip-avatar img-circle"
|
||
src="/img/authors/huong-vuong.png"
|
||
alt="Huong Vuong"
|
||
loading="lazy" decoding="async" width="40" height="40">
|
||
</button>
|
||
<a class="post-author-large" href="/authors#huong.vuong">Huong Vuong</a>
|
||
</div>
|
||
|
||
|
||
|
||
|
||
<div class="post-author-chip">
|
||
<button type="button"
|
||
class="post-author-chip-avatar-trigger"
|
||
data-toggle="tooltip"
|
||
data-placement="top"
|
||
data-trigger="hover focus"
|
||
aria-label="Joseph Sahayaraj"
|
||
title="Joseph Sahayaraj">
|
||
<img class="post-author-chip-avatar img-circle"
|
||
src="/img/authors/joseph-sahayaraj.JPG"
|
||
alt="Joseph Sahayaraj"
|
||
loading="lazy" decoding="async" width="40" height="40">
|
||
</button>
|
||
<a class="post-author-large" href="/authors#joseph.sahayaraj">Joseph Sahayaraj</a>
|
||
</div>
|
||
|
||
|
||
<span class="post-author-separator" aria-hidden="true">·</span>
|
||
<span class="post-date-large">21 Jan 2026 <span class="post-read-time-badge">12 min read</span></span>
|
||
</div>
|
||
|
||
</div>
|
||
|
||
|
||
<aside
|
||
class="article-listen"
|
||
id="article-listen"
|
||
aria-label="Listen to this article"
|
||
data-title="Docker lazy loading at Grab: Accelerating container startup times"
|
||
>
|
||
<div class="article-listen-inner">
|
||
<button
|
||
type="button"
|
||
class="article-listen-toggle"
|
||
id="article-listen-toggle"
|
||
aria-label="Play article audio"
|
||
aria-pressed="false"
|
||
>
|
||
<i class="fa fa-play article-listen-icon article-listen-icon--play" aria-hidden="true"></i>
|
||
<i class="fa fa-pause article-listen-icon article-listen-icon--pause" aria-hidden="true"></i>
|
||
</button>
|
||
<div class="article-listen-copy">
|
||
<p class="article-listen-label">🔊 Listen to Article</p>
|
||
<p class="article-listen-meta" id="article-listen-meta"></p>
|
||
</div>
|
||
<label class="article-listen-speed-label">
|
||
<span class="visually-hidden">Playback speed</span>
|
||
<select class="article-listen-speed" id="article-listen-speed" aria-label="Playback speed">
|
||
<option value="1">1×</option>
|
||
<option value="1.25">1.25×</option>
|
||
<option value="1.5">1.5×</option>
|
||
<option value="2">2×</option>
|
||
</select>
|
||
</label>
|
||
</div>
|
||
<p class="article-listen-unsupported" id="article-listen-unsupported" hidden>
|
||
Audio playback is not supported in this browser.
|
||
</p>
|
||
</aside>
|
||
<script src="/js/article-listen.js" defer></script>
|
||
|
||
|
||
|
||
</header>
|
||
<div class="wrapper-content">
|
||
<article class="post-content" id="post-content">
|
||
<h2 id="introduction">Introduction</h2>
|
||
|
||
<p>At Grab, we’ve been exploring ways to dramatically reduce container startup times for our data platforms. Large container images for services like Airflow and Spark Connect were taking minutes to download, causing slow cold starts and poor auto-scaling performance. This blog post shares our journey implementing Docker image lazy loading using eStargz and Seekable OCI (SOCI) technologies, the results we achieved, and the lessons learned along the way.</p>
|
||
|
||
<h2 id="results-the-numbers-speak-for-themselves">Results: The numbers speak for themselves</h2>
|
||
|
||
<h3 id="benchmark-results">Benchmark results</h3>
|
||
|
||
<p>Our initial testing on fresh nodes (nodes without cached images) showed dramatic improvements in image pull times as shown in <strong>Figure 1</strong>.</p>
|
||
|
||
<div class="post-image-section"><figure>
|
||
<img src="/img/docker-lazy-loading/figure-1.png" alt="" style="width:70%" /><figcaption align="middle">Figure 1. Table of results.</figcaption>
|
||
</figure>
|
||
</div>
|
||
|
||
<p>The key advantage of lazy loading is the reduction in image pull time, especially on “fresh” nodes that do not have the image cached. By analyzing detailed pod events, we can see the precise impact of using the stargz snapshotter.</p>
|
||
|
||
<p>During our SOCI benchmark testing, we observed an important distinction between SOCI and eStargz: <strong>SOCI maintains the same application startup time as standard images, while eStargz takes longer</strong>. For example, with Airflow, both overlayFS and SOCI achieved 5.0 seconds startup time, while eStargz took 25.0 seconds. This demonstrates that lazy loading doesn’t eliminate download time; it redistributes it. SOCI’s approach of maintaining separate indexes allows it to optimize the download-to-startup time trade-off more effectively, keeping application startup performance on par with standard images while still dramatically reducing image pull time.</p>
|
||
|
||
<h2 id="production-performance">Production performance</h2>
|
||
|
||
<p>The production deployment of SOCI lazy loading has delivered significant, measurable improvements across our data platforms. Both Airflow and Spark Connect now experience 30-40% faster startup times, directly improving our ability to handle traffic spikes and scale efficiently. These improvements translate to better auto-scaling responsiveness, reduced resource waste during initialization, and improved user experience for data processing workloads. The sustained performance gains observed over time demonstrate that lazy loading is a stable, production-ready optimization that delivers consistent value.</p>
|
||
|
||
<p><strong>Figure 2 and 3</strong> illustrates the P95 startup time improvements for both services:</p>
|
||
|
||
<div class="post-image-section"><figure>
|
||
<img src="/img/docker-lazy-loading/figure-2.png" alt="" style="width:70%" /><figcaption align="middle">Figure 2. Production results: Airflow P95 startup time. </figcaption>
|
||
</figure>
|
||
</div>
|
||
|
||
<div class="post-image-section"><figure>
|
||
<img src="/img/docker-lazy-loading/figure-3.png" alt="" style="width:70%" /><figcaption align="middle">Figure 3. Production results: Spark Connect P95 startup time.</figcaption>
|
||
</figure>
|
||
</div>
|
||
|
||
<p>It is important to note that P95 startup time includes both the image download/pull time and the application startup time itself. This metric captures the entire system performance for both cold and hot starts on fresh and hot nodes, showing the overall system improvement rather than just cold start performance.</p>
|
||
|
||
<p>During the production deployment and monitoring, we gained valuable insights on SOCI configuration tuning. Following AWS’s recommended configuration from their blog on <a href="https://aws.amazon.com/blogs/containers/introducing-seekable-oci-parallel-pull-mode-for-amazon-eks/">Introducing Seekable OCI: Parallel Pull Mode for Amazon EKS</a>, we optimized our SOCI snapshotter settings:</p>
|
||
|
||
<ul>
|
||
<li>
|
||
<p>Increased <em>max_concurrent_downloads_per_image</em> from 5 to 10.</p>
|
||
</li>
|
||
<li>
|
||
<p>Increased <em>max_concurrent_unpacks_per_image</em> from 3 to 10.</p>
|
||
</li>
|
||
<li>
|
||
<p>Increased <em>concurrent_download_chunk_size</em> from 8MB to 16MB (aligning with AWS’s recommendation for Elastic Container Registry (ECR)).</p>
|
||
</li>
|
||
</ul>
|
||
|
||
<p>This configuration tuning led to a significant performance improvement: <strong>image download time on a fresh node was reduced from 60 seconds to 24 seconds, representing a 60% improvement</strong>. The key lesson here is that default SOCI configurations may not be optimal for all environments, and tuning these parameters based on your infrastructure (especially when using ECR) can yield substantial gains.</p>
|
||
|
||
<h2 id="technical-background-how-docker-lazy-loading-works">Technical background: How Docker lazy loading works</h2>
|
||
|
||
<h3 id="container-root-filesystem-rootfs-and-file-organization">Container root filesystem (rootfs) and file organization</h3>
|
||
|
||
<p>A container’s root filesystem, or rootfs, is the directory structure that the container sees as its root <code class="language-plaintext highlighter-rouge">(/)</code>. It contains all the files and directories necessary for an application to run, including the application itself, its dependencies, system libraries, and configuration files. It’s an isolated filesystem, separate from the host machine’s filesystem.</p>
|
||
|
||
<p>The rootfs is built from a series of read-only layers that come from the container image. Each instruction in an image’s Dockerfile creates a new layer, representing a set of filesystem changes. When a container is launched, a new writable layer, often called the “container layer,” is added on top of the stack of read-only image layers. Any changes made to the running container, such as writing new files or modifying existing ones, are written to this writable layer. The underlying image layers remain untouched. This is known as a copy-on-write (CoW) mechanism.</p>
|
||
|
||
<p>In containerd, a snapshotter is a plugin responsible for managing container filesystems. Its primary job is to take the layers of an image and assemble them into a rootfs for a container. The default snapshotter in containerd is <strong>overlayFS</strong>, which uses the Linux kernel’s OverlayFS driver to efficiently stack layers. To assemble the rootfs, the overlayFS snapshotter creates a “merged” view of the read-only image layers:</p>
|
||
|
||
<div class="post-image-section"><figure>
|
||
<img src="/img/docker-lazy-loading/figure-4.png" alt="" style="width:70%" /><figcaption align="middle">Figure 4. How OverlayFS assembles the container filesystem.</figcaption>
|
||
</figure>
|
||
</div>
|
||
|
||
<ul>
|
||
<li>
|
||
<p><strong>lowerdir</strong>: The read-only image layers are used as the lowerdir in OverlayFS. These are the immutable layers from the container image.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>upperdir</strong>: A new, empty directory is created to be the upperdir. This is the writable layer for the container where any changes are stored.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>merged</strong>: The merged directory is the unified view of the lowerdir and upperdir. This is what is presented to the container as its rootfs.</p>
|
||
</li>
|
||
</ul>
|
||
|
||
<p>When a container reads a file, it’s read from the merged view. When a container writes a file, it’s written to the upperdir using a copy-on-write mechanism. This is an efficient way to manage container filesystems, as it avoids duplicating files and allows for fast container startup.</p>
|
||
|
||
<h3 id="the-problem-traditional-container-image-pull">The problem: Traditional container image pull</h3>
|
||
|
||
<p>To understand the benefits of lazy loading, we first need to understand the traditional container image pull process:</p>
|
||
|
||
<ol>
|
||
<li>
|
||
<p><strong>Download layers</strong>: The container runtime downloads all layer tarballs that make up the image.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Unpack layers</strong>: Each layer is unpacked and extracted onto the host’s disk.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Create snapshot</strong>: The snapshotter combines these layers into a single, unified filesystem, known as the container’s rootfs.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Start container</strong>: Only after all layers are downloaded and unpacked can the container start.</p>
|
||
</li>
|
||
</ol>
|
||
|
||
<p>This process is slow, especially for large images, as the entire image must be present on the host before the container can launch.</p>
|
||
|
||
<h3 id="the-solution-remote-snapshotter">The solution: Remote snapshotter</h3>
|
||
|
||
<p>To address the slow startup issue with large images, we use a <strong>remote snapshotter</strong> solution. A remote snapshotter is a special type of snapshotter that doesn’t require all image data to be locally present. Instead of downloading and unpacking all the layers, it creates a “snapshot” that points to the remote location of the data (like a container registry). The actual file content is then fetched on-demand when the container tries to read a file for the first time.</p>
|
||
|
||
<p>While a traditional snapshotter like overlayFS uses directories on the local disk as its lowerdir, a remote snapshotter creates a virtual lowerdir that is backed by the remote registry. This is typically done using FUSE (Filesystem in Userspace). The remote snapshotter creates a FUSE filesystem that presents the contents of the remote layer as if it were a local directory. This FUSE mount is then used as the lowerdir for the overlayFS driver. This allows the remote snapshotter to integrate with the existing overlayFS infrastructure while adding the capability of lazy-loading data from a remote source.</p>
|
||
|
||
<p>There are two main formats that enable remote snapshotters: <strong>eStargz</strong> and <strong>SOCI</strong>.</p>
|
||
|
||
<h3 id="estargz-format">eStargz format</h3>
|
||
|
||
<p>eStargz is a backward-compatible extension of the standard OCI <code class="language-plaintext highlighter-rouge">tar.gz</code> layer format. It has several key features that enable lazy loading:</p>
|
||
|
||
<ul>
|
||
<li>
|
||
<p><strong>Individually compressed files</strong>: Each file within the layer (and even chunks of large files) is compressed individually. This is the key that allows for random access to file contents.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>TOC (table of contents)</strong>: A JSON file named <code class="language-plaintext highlighter-rouge">stargz.index.json</code> is located at the end of the layer. This TOC contains metadata for every file, including its name, size, and, most importantly, its offset within the layer blob.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Footer</strong>: A small footer at the very end of the layer contains the offset of the TOC, allowing it to be easily located by reading only the last few bytes of the layer.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Chunking and verification</strong>: Large files can be broken down into smaller chunks, each with its own entry in the TOC. Each chunk also has a chunkDigest in its TOC entry, allowing for independent verification of each downloaded piece of data.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Prefetch landmark</strong>: A special file, <code class="language-plaintext highlighter-rouge">.prefetch.landmark</code>, can be placed in the layer to mark the end of “prioritized files”. This allows the snapshotter to intelligently prefetch the most important files for the container’s workload.</p>
|
||
</li>
|
||
</ul>
|
||
|
||
<p>The stargz snapshotter uses the eStargz format to enable lazy loading. Here’s how it works:</p>
|
||
|
||
<ol>
|
||
<li>
|
||
<p><strong>Mount request</strong>: When containerd calls the Mount function, it’s the main entry point for creating a new filesystem for a layer.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Resolve and read TOC</strong>: The snapshotter fetches the layer’s footer, then fetches the <code class="language-plaintext highlighter-rouge">stargz.index.json</code> TOC from the remote registry. This TOC contains all the file metadata needed to create a virtual filesystem.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Mount FUSE filesystem</strong>: With the TOC in memory, the snapshotter creates a virtual filesystem using FUSE. The container can now start, as it has a valid rootfs, even though most of the file content has not been downloaded.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>On-demand fetching</strong>: When the container performs a file operation like <code class="language-plaintext highlighter-rouge">read()</code>, the FUSE filesystem intercepts the call. The snapshotter checks a local disk cache for the requested bytes. If the data is not cached, it issues an HTTP Range request to the container registry to download only the required chunk of the layer.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Remote fetching and caching</strong>: The downloaded data is returned to the container and also written to the local cache for subsequent reads.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Prefetching for optimization</strong>: After the FUSE filesystem is mounted, a background goroutine begins downloading the prioritized files (up to the <em>.prefetch.landmark</em>) and can also be configured to download the entire rest of the layer in the background.</p>
|
||
</li>
|
||
</ol>
|
||
|
||
<p>For a deeper understanding of the eStargz format and stargz snapshotter, see the <a href="https://github.com/containerd/stargz-snapshotter/blob/main/docs/overview.md">stargz-snapshotter overview documentation</a>.</p>
|
||
|
||
<h3 id="soci-format">SOCI format</h3>
|
||
|
||
<p>SOCI is a technology open sourced by AWS that enables containers to launch faster by lazily loading the container image. SOCI works by creating an index (SOCI Index) of the files within an existing container image. SOCI borrows some of the design principles from stargz-snapshotter but takes a different approach:</p>
|
||
|
||
<ul>
|
||
<li>
|
||
<p><strong>Separate index</strong>: A SOCI index is generated separately from the container image and is stored in the registry as an OCI Artifact, linked back to the container image by OCI Reference Types.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>No image conversion</strong>: This means that the container images do not need to be converted, image digests do not change, and image signatures remain valid.</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>Native Bottlerocket support</strong>: SOCI is natively supported on Bottlerocket OS.</p>
|
||
</li>
|
||
</ul>
|
||
|
||
<p>For a deeper understanding of the SOCI format, see the <a href="https://github.com/awslabs/soci-snapshotter/blob/main/docs/index.md">soci-snapshotter documentation</a>.</p>
|
||
|
||
<h2 id="building-and-deploying-lazy-loaded-images">Building and deploying lazy-loaded images</h2>
|
||
|
||
<h3 id="setting-up-snapshotters-in-eks">Setting up snapshotters in EKS</h3>
|
||
|
||
<p>When using EKS with containerd as the container runtime, you can configure remote snapshotters to enable lazy loading. Here’s how to set them up:</p>
|
||
|
||
<p><strong>For stargz-snapshotter (eStargz)</strong>: You need to install the <code class="language-plaintext highlighter-rouge">containerd-stargz-grpc</code> service first, then register it as a proxy plugin in containerd’s configuration:</p>
|
||
|
||
<pre><code class="language-textproto"># /etc/containerd/config.toml
|
||
[proxy_plugins]
|
||
[proxy_plugins.stargz]
|
||
type = "snapshot"
|
||
address = "/run/containerd-stargz-grpc/containerd-stargz-grpc.sock"
|
||
</code></pre>
|
||
|
||
<p>For detailed installation instructions, see the <a href="https://github.com/containerd/stargz-snapshotter/blob/main/docs/INSTALL.md">stargz-snapshotter installation documentation</a>. The setup can be baked into an AMI for production use or tested via user data from node bootstrap scripts.</p>
|
||
|
||
<p><strong>For SOCI snapshotter (Bottlerocket)</strong>: On Bottlerocket nodes, enable the SOCI snapshotter via user data:</p>
|
||
|
||
<pre><code class="language-textproto"># Enable SOCI snapshotter
|
||
[settings.container-runtime]
|
||
snapshotter = "soci"
|
||
</code></pre>
|
||
|
||
<p>SOCI is natively supported on Bottlerocket, so no additional daemon installation is required.</p>
|
||
|
||
<h3 id="building-lazy-loaded-images">Building lazy-loaded images</h3>
|
||
|
||
<p>eStargz images can be built natively using Docker Buildx by setting the output compression to <code class="language-plaintext highlighter-rouge">estargz</code>:</p>
|
||
|
||
<div class="language-shell highlighter-rouge"><div class="highlight"><pre class="highlight"><code>docker buildx build
|
||
<span class="nt">--platform</span> linux/amd64
|
||
<span class="nt">--output</span> <span class="nb">type</span><span class="o">=</span>registry,oci-mediatypes<span class="o">=</span><span class="nb">true</span>,compression<span class="o">=</span>estargz,force-compression<span class="o">=</span><span class="nb">true</span>
|
||
<span class="nt">--tag</span> <span class="nv">$ECR_REGISTRY</span>/airflow:<span class="nv">$TAG</span>
|
||
<span class="nb">.</span>
|
||
</code></pre></div></div>
|
||
|
||
<p>SOCI doesn’t require rebuilding images; you only need to generate a SOCI index for existing images. Since Docker doesn’t natively support SOCI index generation yet, workaround solutions include using the <a href="https://awslabs.github.io/cfn-ecr-aws-soci-index-builder/#_overview">AWS SOCI Index Builder Using Lambda Functions</a> or integrating SOCI index generation into your CI/CD pipeline as described in this <a href="https://pabis.eu/blog/2025-06-17-Faster-ECS-Startup-SOCI-Index-GitLab-Pipeline.html">blog post</a>.</p>
|
||
|
||
<h2 id="key-takeaway-why-we-chose-soci">Key takeaway: Why we chose SOCI</h2>
|
||
|
||
<p>We started our exploration with eStargz but ultimately chose SOCI for production deployment. The key reason is scalability and alignment with our strategy to use Bottlerocket OS for enhancing Kubernetes pod startup and security. SOCI is natively supported by Bottlerocket, which means service teams don’t need to set up and maintain the more complicated stargz snapshotter across all EKS clusters. This makes the implementation easier to maintain and provides better support from AWS.</p>
|
||
|
||
<p>Additionally, we learned that lazy loading doesn’t eliminate the time required to download image data; it redistributes it from startup time to runtime. While this dramatically improves cold start performance, it’s important to monitor application performance closely and tune configuration parameters based on your workload and infrastructure. We achieved a 60% improvement by optimizing SOCI’s parallel pull mode settings, demonstrating the value of proper configuration tuning.</p>
|
||
|
||
<h2 id="conclusion">Conclusion</h2>
|
||
|
||
<p>Docker image lazy loading with SOCI offers a significant opportunity to improve the performance and efficiency of our services at Grab. Our testing and production deployments have shown:</p>
|
||
|
||
<ul>
|
||
<li>
|
||
<p>4x faster image pull times on fresh nodes.</p>
|
||
</li>
|
||
<li>
|
||
<p>29-34% improvement in P95 startup times for production workloads.</p>
|
||
</li>
|
||
<li>
|
||
<p>60% improvement in image download times with proper configuration tuning.</p>
|
||
</li>
|
||
</ul>
|
||
|
||
<p>The implementation path is clear, low-risk, and builds on proven components. This technology is production-ready, and we’re continuing to scale it across more services.</p>
|
||
|
||
<h3 id="references">References</h3>
|
||
|
||
<ul>
|
||
<li>
|
||
<p><strong>Databricks:</strong> <a href="https://www.databricks.com/blog/2021/09/08/booting-databricks-vms-7x-faster-for-serverless-compute.html">Booting Databricks VMs 7x Faster for Serverless Compute</a> - Industry case study showing how major tech companies achieve fast container startup at scale</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>BytePlus:</strong> <a href="https://docs.byteplus.com/en/docs/vke/Container-image-lazy-loading-solution">Container Image Lazy Loading Solution</a> - Enterprise implementation guide for lazy loading in production Kubernetes environments</p>
|
||
</li>
|
||
<li>
|
||
<p><strong>AWS:</strong> <a href="https://aws.amazon.com/blogs/containers/introducing-seekable-oci-parallel-pull-mode-for-amazon-eks/">Introducing Seekable OCI: Parallel Pull Mode for Amazon EKS</a> - AWS’s guide to SOCI configuration and optimization</p>
|
||
</li>
|
||
</ul>
|
||
|
||
<h2 id="join-us">Join us</h2>
|
||
|
||
<p>Grab is a leading superapp in Southeast Asia, operating across the deliveries, mobility and digital financial services sectors. Serving over 800 cities in eight Southeast Asian countries, Grab enables millions of people everyday to order food or groceries, send packages, hail a ride or taxi, pay for online purchases or access services such as lending and insurance, all through a single app. Grab was founded in 2012 with the mission to drive Southeast Asia forward by creating economic empowerment for everyone. Grab strives to serve a triple bottom line – we aim to simultaneously deliver financial performance for our shareholders and have a positive social impact, which includes economic empowerment for millions of people in the region, while mitigating our environmental footprint.</p>
|
||
|
||
<p>Powered by technology and driven by heart, our mission is to drive Southeast Asia forward by creating economic empowerment for everyone. If this mission speaks to you, <a href="https://grb.to/gebdockerlazyloading">join our team</a> today!</p>
|
||
|
||
</article>
|
||
<div class="sharing-links text-right">
|
||
Share on
|
||
<a href="https://twitter.com/intent/tweet?text=Docker lazy loading at Grab: Accelerating container startup times&url=https://engineering.grab.com/docker-lazy-loading&via=grabengineering&related=grabengineering" class="btn btn-sm btn-share btn-share-twitter" rel="nofollow" target="_new" title="Share on Twitter" onclick="onShareButtonClick(this); return false;"><i class="fa fa-lg fa-twitter"></i> Twitter</a>
|
||
<a href="https://facebook.com/sharer.php?u=https://engineering.grab.com/docker-lazy-loading" class="btn btn-sm btn-share btn-share-facebook" rel="nofollow" target="_new" title="Share on Facebook" onclick="onShareButtonClick(this); return false;"><i class="fa fa-lg fa-facebook"></i> Facebook</a>
|
||
<a href="https://www.linkedin.com/shareArticle?mini=true&url=https://engineering.grab.com/docker-lazy-loading&title=Docker lazy loading at Grab: Accelerating container startup times
|
||
&summary=Large container images were causing slow cold starts and poor auto-scaling for Grab's data platforms. This post explores how we implemented Docker image lazy loading with Seekable OCI (SOCI) technology, to achieve faster image pulls and startup times. The blog discusses how lazy loading works, the technology behind SOCI and eStargz, and finally how this configuration delivered a 60% improvement in download times.&source=Grab Tech" class="btn btn-sm btn-share btn-share-linkedin" rel="nofollow" target="_new" title="Share on LinkedIn" onclick="onShareButtonClick(this); return false;"><i class="fa fa-lg fa-linkedin"></i> LinkedIn</a>
|
||
</div>
|
||
<script>
|
||
function onShareButtonClick(button) {
|
||
var width = 600;
|
||
var height = 600;
|
||
var left = (window.screen.width / 2) - (width / 2);
|
||
var top = (window.screen.height / 2) - (height / 2);
|
||
window.open(button.href, '', 'menubar=no,toolbar=no,resizable=yes,scrollbars=yes,height=' + height + ',width=' + width + ',top=' + top + ',left=' + left);
|
||
return false;
|
||
}
|
||
</script>
|
||
|
||
<hr class="section-divider">
|
||
<section
|
||
class="related-jobs-section"
|
||
id="related-jobs"
|
||
data-tags='["Database"]'
|
||
data-limit="3"
|
||
aria-live="polite"
|
||
>
|
||
<h2 class="related-jobs-heading">Related jobs at Grab</h2>
|
||
<p class="related-jobs-status">Loading opportunities…</p>
|
||
<ul class="related-jobs-cards" hidden></ul>
|
||
<p class="related-jobs-fallback" hidden>
|
||
<a href="https://www.grab.careers/en/jobs/?orderby=0&pagesize=20&page=1&team=Engineering" target="_blank" rel="noopener noreferrer">Browse open roles at Grab Careers</a>
|
||
</p>
|
||
</section>
|
||
<script src="/js/related-jobs.js" defer></script>
|
||
|
||
</div>
|
||
</div>
|
||
|
||
<div class="post-sidebar">
|
||
|
||
|
||
|
||
|
||
<aside class="sidebar-panel recent-posts-panel">
|
||
<h2 class="sidebar-panel-heading">Recent articles</h2>
|
||
<ul class="recent-posts-list">
|
||
|
||
|
||
<li class="recent-posts-item">
|
||
<a href="/data-mesh-at-grab-part-three" class="recent-posts-link">
|
||
|
||
<img src="/img/datamesh-three/banner-img.png" class="recent-posts-thumb" alt="Data Mesh at Grab (Part III): Operationalizing data reliability with automated DPIs cover photo" loading="lazy" decoding="async" width="80" height="80">
|
||
|
||
<span class="recent-posts-text">
|
||
<span class="recent-posts-title">Data Mesh at Grab (Part III): Operationalizing data reliability with automated DPIs</span>
|
||
<time class="recent-posts-date" datetime="2026-08-28T00:00:00+00:00">28 Aug 2026</time>
|
||
</span>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
<li class="recent-posts-item">
|
||
<a href="/jarvis-pro-route-firsr-answer-later" class="recent-posts-link">
|
||
|
||
<img src="/img/jarvis-pro/banner-img.png" class="recent-posts-thumb" alt="Building Jarvis Pro: Route first, answer later cover photo" loading="lazy" decoding="async" width="80" height="80">
|
||
|
||
<span class="recent-posts-text">
|
||
<span class="recent-posts-title">Building Jarvis Pro: Route first, answer later</span>
|
||
<time class="recent-posts-date" datetime="2026-08-21T00:00:00+00:00">21 Aug 2026</time>
|
||
</span>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
<li class="recent-posts-item">
|
||
<a href="/grab-bench-evaluating-ai" class="recent-posts-link">
|
||
|
||
<img src="/img/grab-bench/banner-img.png" class="recent-posts-thumb" alt="Grab Bench: Evaluating AI on Grab-shaped production work cover photo" loading="lazy" decoding="async" width="80" height="80">
|
||
|
||
<span class="recent-posts-text">
|
||
<span class="recent-posts-title">Grab Bench: Evaluating AI on Grab-shaped production work</span>
|
||
<time class="recent-posts-date" datetime="2026-08-12T00:00:00+00:00">12 Aug 2026</time>
|
||
</span>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
<li class="recent-posts-item">
|
||
<a href="/how-ai-is-transforming-analytics" class="recent-posts-link">
|
||
|
||
<img src="/img/ai-improve-analytics/banner-image.png" class="recent-posts-thumb" alt="How AI is transforming analytics at Grab cover photo" loading="lazy" decoding="async" width="80" height="80">
|
||
|
||
<span class="recent-posts-text">
|
||
<span class="recent-posts-title">How AI is transforming analytics at Grab</span>
|
||
<time class="recent-posts-date" datetime="2026-08-01T00:23:00+00:00">1 Aug 2026</time>
|
||
</span>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
<li class="recent-posts-item">
|
||
<a href="/crowdsourced-taxonomy-verification" class="recent-posts-link">
|
||
|
||
<img src="/img/crowdsource-taxonomy/banner-img.png" class="recent-posts-thumb" alt="Crowdsourced taxonomy verification: A feedback-driven framework for refining knowledge graph relationships via online search interactions cover photo" loading="lazy" decoding="async" width="80" height="80">
|
||
|
||
<span class="recent-posts-text">
|
||
<span class="recent-posts-title">Crowdsourced taxonomy verification: A feedback-driven framework for refining knowledge graph relationships via online search interactions</span>
|
||
<time class="recent-posts-date" datetime="2026-07-30T00:00:00+00:00">30 Jul 2026</time>
|
||
</span>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
</ul>
|
||
<a class="sidebar-panel-cta" href="/">View all articles</a>
|
||
</aside>
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
|
||
</div>
|
||
</div>
|
||
</div>
|
||
|
||
</div>
|
||
<div class="progress-wrap">
|
||
<svg class="progress-circle svg-content" width="100%" height="100%" viewBox="-1 -1 102 102">
|
||
<path d="M50,1 a49,49 0 0,1 0,98 a49,49 0 0,1 0,-98" />
|
||
</svg>
|
||
<i class="fa fa-chevron-up btt-btn"></i>
|
||
</div>
|
||
<footer class="site-footer">
|
||
<div class="wrapper">
|
||
<div class="row">
|
||
<div class="col-sm-6 col-xs-12">
|
||
<h2 class="footer-heading">Grab Tech</h2>
|
||
<ul class="social-media-list">
|
||
|
||
<li>
|
||
<a href="https://github.com/grab" target="_blank" rel="nofollow noreferrer">
|
||
<i class="fa fa-github fa-lg"></i>
|
||
</a>
|
||
</li>
|
||
|
||
|
||
<li>
|
||
<a href="https://www.linkedin.com/company/grabapp" target="_blank" rel="nofollow noreferrer">
|
||
<i class="fa fa-linkedin fa-lg"></i>
|
||
</a>
|
||
</li>
|
||
|
||
<li>
|
||
<a href="https://engineering.grab.com/feed.xml" target="_blank">
|
||
<i class="fa fa-rss fa-lg"></i>
|
||
</a>
|
||
</li>
|
||
</ul>
|
||
|
||
<div>
|
||
<script src="//platform.linkedin.com/in.js" type="text/javascript"> lang: en_US</script>
|
||
<script type="IN/FollowCompany" data-id="5382086" data-counter="right"></script>
|
||
</div>
|
||
<br>
|
||
</div>
|
||
<div class="col-sm-6 col-xs-12 hiring-section">
|
||
<h2 class="footer-heading">Join Us</h2>
|
||
<p class="text">
|
||
Want to join us in our mission to revolutionize transportation?
|
||
</p>
|
||
<a class="btn" href="https://www.grab.careers/en/teams/engineering/" target="_blank" rel="noopener noreferrer">View open positions</a>
|
||
|
||
</div>
|
||
</div>
|
||
<script>
|
||
window.blogSearchIndex = [
|
||
|
||
{
|
||
"title": "Data Mesh at Grab (Part III): Operationalizing data reliability with automated DPIs",
|
||
"url": "/data-mesh-at-grab-part-three",
|
||
"tags": ["Data","Database","Engineering","Data Quality","Observability"],
|
||
"category": null,
|
||
"date": "28 Aug 2026",
|
||
"excerpt": "Make data reliability as dependable as running water. Learn how Grab operationalises this vision using automated Data Production Issues (DPIs) to convert dat..."
|
||
},
|
||
|
||
{
|
||
"title": "Building Jarvis Pro: Route first, answer later",
|
||
"url": "/jarvis-pro-route-firsr-answer-later",
|
||
"tags": ["Artificial Intelligence","Engineering","Product","Account Management"],
|
||
"category": null,
|
||
"date": "21 Aug 2026",
|
||
"excerpt": "Jarvis Pro routes account-manager questions before answering. This post explains why fluent AI replies are not enough, how routing, memory, and metric reconc..."
|
||
},
|
||
|
||
{
|
||
"title": "Grab Bench: Evaluating AI on Grab-shaped production work",
|
||
"url": "/grab-bench-evaluating-ai",
|
||
"tags": ["Artificial Intelligence","Engineering","Product","Machine Learning"],
|
||
"category": null,
|
||
"date": "12 Aug 2026",
|
||
"excerpt": "Broad AI benchmarks are good at telling us which models look strong on public leaderboards. They are less good at catching the quiet failures that break prod..."
|
||
},
|
||
|
||
{
|
||
"title": "How AI is transforming analytics at Grab",
|
||
"url": "/how-ai-is-transforming-analytics",
|
||
"tags": ["Engineering","Analytics","AI"],
|
||
"category": null,
|
||
"date": "1 Aug 2026",
|
||
"excerpt": "As AI agents take on more of the analytics loop, Grab's analysts are moving from producing artefacts to owning questions, judgement, and decisions. This post..."
|
||
},
|
||
|
||
{
|
||
"title": "Crowdsourced taxonomy verification: A feedback-driven framework for refining knowledge graph relationships via online search interactions",
|
||
"url": "/crowdsourced-taxonomy-verification",
|
||
"tags": ["Engineering","Data","Search","Machine Learning","LLM","Graphs"],
|
||
"category": null,
|
||
"date": "30 Jul 2026",
|
||
"excerpt": "Knowledge graphs power modern search and recommendation systems, but automated construction methods can introduce semantic inaccuracies. This framework verif..."
|
||
},
|
||
|
||
{
|
||
"title": "Agent platform (Part 1): How we help Grab build and run AI agents at scale",
|
||
"url": "/how-grab-builds-and-runs-ai-agents-at-scale",
|
||
"tags": ["Engineering","Generative AI","LLM","Experiment","Machine Learning"],
|
||
"category": null,
|
||
"date": "24 Jul 2026",
|
||
"excerpt": "Explains how Grab turned repeated operational pain points into reusable platform primitives that help teams build, test, and run agents faster and more safel..."
|
||
},
|
||
|
||
{
|
||
"title": "Scaling Grab's Data Lake: Our journey to Apache Iceberg adoption",
|
||
"url": "/our-journey-to-apache-iceberg-adoption",
|
||
"tags": ["Data","Database","Engineering","Apache Iceberg","Spark"],
|
||
"category": null,
|
||
"date": "10 Jul 2026",
|
||
"excerpt": "As Grab's Data Lake grew to petabytes, the limitations of Hive Parquet became clear, from catalog latency and small files to manual partition management. Thi..."
|
||
},
|
||
|
||
{
|
||
"title": "Migrating Counter Service storage: Design choices and learnings",
|
||
"url": "/counter-service-storage-migration",
|
||
"tags": ["Security","Artificial Intelligence","Kubernetes","DevSecOps","Platform","Engineering"],
|
||
"category": null,
|
||
"date": "3 Jul 2026",
|
||
"excerpt": "Counter Service powers real-time fraud detection at massive scale, handling billions of requests daily. Learn how we migrated its underlying storage with zer..."
|
||
},
|
||
|
||
{
|
||
"title": "Scaling out Distroless adoption With AI",
|
||
"url": "/scaling-out-distroless-adoption-with-ai",
|
||
"tags": ["Security","Containers","Artificial Intelligence","DevSecOps","Engineering"],
|
||
"category": null,
|
||
"date": "22 Jun 2026",
|
||
"excerpt": "Grab is moving hundreds of services to Distroless images to cut CVEs but only if workloads still run cleanly. Learn how medium tests became the migration gat..."
|
||
},
|
||
|
||
{
|
||
"title": "Palana (Part 2): Architecting isolation, identity, and auditability for AI agents",
|
||
"url": "/part-2-palana-architecture",
|
||
"tags": ["Security","Artificial Intelligence","Kubernetes","DevSecOps","Platform","Engineering"],
|
||
"category": null,
|
||
"date": "21 Jun 2026",
|
||
"excerpt": "How do you actually build a secure execution environment for artificial intelligence (AI) agents? In Part 2, we dive deep into Palana's architecture. Learn h..."
|
||
},
|
||
|
||
{
|
||
"title": "Palana (Part 1): Why Grab built a secure platform for autonomous AI Agents",
|
||
"url": "/palana-part-1-secure-platform-for-ai-agents",
|
||
"tags": ["Security","Artificial Intelligence","Kubernetes","DevSecOps","Platform","Engineering"],
|
||
"category": null,
|
||
"date": "19 Jun 2026",
|
||
"excerpt": "Artificial intelligence (AI) agents are evolving from chat interfaces into autonomous workloads, bringing new security risks. In Part 1, discover why Grab bu..."
|
||
},
|
||
|
||
{
|
||
"title": "From decentralized Docs-as-Code to a centralized repository: Evolving Grab's documentation strategy",
|
||
"url": "/evolving-documentation-strategy",
|
||
"tags": ["Blog","TechDocs","Engineering"],
|
||
"category": null,
|
||
"date": "29 May 2026",
|
||
"excerpt": "Building on Grab's Docs-as-Code approach, we reflect on our documentation journey, uncovering the benefits, challenges, and limitations along the way. Learn ..."
|
||
},
|
||
|
||
{
|
||
"title": "The Hugo evolution: Engineering Grab's unified, one-click data ingestion platform with Apache Flink",
|
||
"url": "/one-click-data-ingestion-platform-with-apache-flink",
|
||
"tags": ["Database","Hugo","FlinkSQL"],
|
||
"category": null,
|
||
"date": "22 May 2026",
|
||
"excerpt": "At Grab, we're transforming data ingestion and processing with Hugo, our self-service data platform. Now integrated with Apache Flink, Hugo empowers teams to..."
|
||
},
|
||
|
||
{
|
||
"title": "Scaling developer experience: How we improved Android Studio in a large monorepo",
|
||
"url": "/how-we-improved-android-studio-in-large-monorepo",
|
||
"tags": ["Engineering","Android","Performance","Monorepo"],
|
||
"category": null,
|
||
"date": "15 May 2026",
|
||
"excerpt": "Frustrated by 35-minute integrated development environment (IDE) syncs? In a large monorepo, slow builds were eroding productivity. Discover how we built a c..."
|
||
},
|
||
|
||
{
|
||
"title": "Enhancing Flink deployment with shadow testing",
|
||
"url": "/enchancing-flink-shadow-testing",
|
||
"tags": ["Database","Testing","FlinkSQL"],
|
||
"category": null,
|
||
"date": "7 May 2026",
|
||
"excerpt": "Discover how Grab's data streaming team has revolutionized Apache Flink deployments with Shadow Testing, ensuring seamless reliability for real-time applicat..."
|
||
},
|
||
|
||
{
|
||
"title": "Data Mesh at Grab (Part II): The foundational tools behind certification",
|
||
"url": "/data-mesh-part-2-the-foundational-tools-behind-certification",
|
||
"tags": ["Data","Database","Engineering"],
|
||
"category": null,
|
||
"date": "30 Apr 2026",
|
||
"excerpt": "How does Grab manage quality across hundreds of thousands of data assets? Discover the foundational tools powering our Signals Marketplace. We dive into Hubb..."
|
||
},
|
||
|
||
{
|
||
"title": "From firefighting to building: How AI agents restored our team’s core productivity",
|
||
"url": "/from-firefighting-to-building",
|
||
"tags": ["AI","Artificial Intelligence","Analytics","Database","Automation"],
|
||
"category": null,
|
||
"date": "19 Mar 2026",
|
||
"excerpt": "The Analytics Data Warehouse (ADW) team at Grab supports over 1,000 users. These users support an extensive repository of more than 15,000 tables. To allevia..."
|
||
},
|
||
|
||
{
|
||
"title": "Enabling R8 optimization at scale with AI-assisted debugging",
|
||
"url": "/r8-optimization-at-scale-with-ai-assisted-debugging",
|
||
"tags": ["AI","Artificial Intelligence"],
|
||
"category": null,
|
||
"date": "12 Mar 2026",
|
||
"excerpt": "How Grab enabled R8 optimization for its Android app at scale, over 9 million lines of code and more than engineers. Read how we achieved 25% ANR reduction, ..."
|
||
},
|
||
|
||
{
|
||
"title": "Reclaiming Terabytes: Optimizing Android image caching with TLRU",
|
||
"url": "/reclaiming-tetabytes-optimizing-android-image-caching-with-tlru",
|
||
"tags": ["App Disk","Disk Size","Optimization","Scalability"],
|
||
"category": null,
|
||
"date": "6 Mar 2026",
|
||
"excerpt": "In the quest to optimize app performance, managing the image cache was crucial. This blog takes us on a journey from a traditional Least Recently Used (LRU) ..."
|
||
},
|
||
|
||
{
|
||
"title": "Cursor at Grab: Adoption and impact",
|
||
"url": "/cursor-at-grab-adoption-and-impact",
|
||
"tags": ["AI","Artificial Intelligence"],
|
||
"category": null,
|
||
"date": "29 Jan 2026",
|
||
"excerpt": "A look inside how we scaled AI-assisted coding across Grab, moving Cursor from pilot to daily use to help us work faster and more reliably. Read what changed..."
|
||
},
|
||
|
||
{
|
||
"title": "Docker lazy loading at Grab: Accelerating container startup times",
|
||
"url": "/docker-lazy-loading",
|
||
"tags": ["Database"],
|
||
"category": null,
|
||
"date": "21 Jan 2026",
|
||
"excerpt": "Large container images were causing slow cold starts and poor auto-scaling for Grab's data platforms. This post explores how we implemented Docker image lazy..."
|
||
},
|
||
|
||
{
|
||
"title": "From deployment slop to production reality: How BriX bridges the gap with enterprise-grade AI infrastructure",
|
||
"url": "/brix",
|
||
"tags": ["AI","Artificial Intelligence","LLM","Deployment"],
|
||
"category": null,
|
||
"date": "16 Jan 2026",
|
||
"excerpt": "Built an AI tool that works locally but can't scale enterprise-wide? Learn how BriX tackles this deployment gap, turning prototypes into governed, production..."
|
||
},
|
||
|
||
{
|
||
"title": "Demystifying user journeys: Revolutionizing troubleshooting with auto tracking",
|
||
"url": "/auto-track-sdk",
|
||
"tags": ["Mobile","iOS","Android","Tracking"],
|
||
"category": null,
|
||
"date": "23 Dec 2025",
|
||
"excerpt": "In the dynamic realm of mobile development, understanding user journeys is key to effective troubleshooting. This blog delves into how Grab's innovative Auto..."
|
||
},
|
||
|
||
{
|
||
"title": "How Grab is accelerating growth with real-time personalization using Customer Data Platform scenarios",
|
||
"url": "/cdp-scenarios",
|
||
"tags": ["Database","FlinkSQL"],
|
||
"category": null,
|
||
"date": "18 Dec 2025",
|
||
"excerpt": "Grab’s Customer Data Platform (CDP) introduces Scenarios, enabling real-time personalization at scale. By leveraging event triggers, geo-fencing, historical ..."
|
||
},
|
||
|
||
{
|
||
"title": "A Decade of Defense: Celebrating Grab's 10th Year Bug Bounty Program",
|
||
"url": "/a-decade-of-defense",
|
||
"tags": ["Engineering","Performance"],
|
||
"category": null,
|
||
"date": "1 Dec 2025",
|
||
"excerpt": "Discover how Grab has championed cybersecurity for a decade with its Bug Bounty Program. This article delves into the milestones, insights, and the collabora..."
|
||
},
|
||
|
||
{
|
||
"title": "Real-time data quality monitoring: Kafka stream contracts with syntactic and semantic test",
|
||
"url": "/real-time-data-quality-monitoring",
|
||
"tags": ["Engineering","Kafka","Performance","Data Science","Data Processing","Real-Time Streaming"],
|
||
"category": null,
|
||
"date": "26 Nov 2025",
|
||
"excerpt": "Discover how Grab's Coban Platform revolutionizes real-time data quality monitoring for Kafka streams. Learn how syntactic and semantic tests empower stream ..."
|
||
},
|
||
|
||
{
|
||
"title": "SpellVault’s evolution: Beyond LLM apps, towards the agentic future",
|
||
"url": "/spellvault-evolution-beyond-llm",
|
||
"tags": ["Engineering","Performance"],
|
||
"category": null,
|
||
"date": "21 Nov 2025",
|
||
"excerpt": "Discover SpellVault’s evolution from its early RAG-based foundations and plugin ecosystem to its transformation into a tool-driven, agentic framework that em..."
|
||
},
|
||
|
||
{
|
||
"title": "Grab's Mac Cloud Exit supercharges macOS CI/CD",
|
||
"url": "/mac-cloud-exit",
|
||
"tags": ["Engineering","Performance"],
|
||
"category": null,
|
||
"date": "6 Nov 2025",
|
||
"excerpt": "Discover how our transition from cloud-based Mac hardware infrastructure to a colocation cluster within Southeast Asia has revolutionized our macOS CI/CD, en..."
|
||
},
|
||
|
||
{
|
||
"title": "How we built a custom vision LLM to improve document processing at Grab",
|
||
"url": "/custom-vision-llm-at-grab",
|
||
"tags": ["Engineering","Performance"],
|
||
"category": null,
|
||
"date": "4 Nov 2025",
|
||
"excerpt": "e-KYC faces challenges with unstandardized document formats and local SEA languages. Existing LLMs lack sufficient SEA language support. We trained a Vision ..."
|
||
},
|
||
|
||
{
|
||
"title": "Machine-learning predictive autoscaling for Flink",
|
||
"url": "/ml-predictive-autoscaling-for-flink",
|
||
"tags": ["Engineering","Performance","Data Science"],
|
||
"category": null,
|
||
"date": "30 Oct 2025",
|
||
"excerpt": "Explore how Grab uses machine learning to perform predictive scaling on our data processing workloads."
|
||
},
|
||
|
||
{
|
||
"title": "Modernising Grab’s model serving platform with NVIDIA Triton Inference Server",
|
||
"url": "/modernising-grab-model-serving-platform",
|
||
"tags": ["Engineering","Performance","Data Science"],
|
||
"category": null,
|
||
"date": "21 Oct 2025",
|
||
"excerpt": "Dive into Grab’s engineering journey to optimise a core ML model. Learn how we built the Triton Server Manager and used Triton Inference Server (TIS) to achi..."
|
||
},
|
||
|
||
{
|
||
"title": "Highly concurrent in-memory counter in GoLang",
|
||
"url": "/highly-concurrent-in-memory-counter-in-go-lang",
|
||
"tags": ["Database"],
|
||
"category": null,
|
||
"date": "6 Oct 2025",
|
||
"excerpt": "Dive into the chaos and triumph of real-time optimisation in the face of high database utilisation! This article recounts a developer's adrenaline-fueled jou..."
|
||
},
|
||
|
||
{
|
||
"title": "User foundation models for Grab",
|
||
"url": "/user-foundation-models-for-grab",
|
||
"tags": ["AI","Artificial Intelligence","Machine Learning","LLM"],
|
||
"category": null,
|
||
"date": "26 Sep 2025",
|
||
"excerpt": "Grab has developed a groundbreaking foundation model specifically designed to understand user behavior. Grab's custom solution addresses the unique challenge..."
|
||
},
|
||
|
||
{
|
||
"title": "Powering Partner Gateway metrics with Apache Pinot",
|
||
"url": "/pinot-partnergateway-tech-blog",
|
||
"tags": ["Database","Data","Apache"],
|
||
"category": null,
|
||
"date": "23 Sep 2025",
|
||
"excerpt": "Partner Gateway serves as Grab's secure interface for exposing APIs to third-party entities, facilitating seamless interactions between Grab's hosted service..."
|
||
},
|
||
|
||
{
|
||
"title": "Taming the monorepo beast: Our journey to a leaner, faster GitLab repo",
|
||
"url": "/taming-monorepo-beast",
|
||
"tags": ["Engineering","Monorepo","Go","Infrastructure","Performance"],
|
||
"category": null,
|
||
"date": "16 Sep 2025",
|
||
"excerpt": "At Grab, our decade-old Go monorepo had become a 214GB monster with 13 million commits, causing 4-minute replication delays and crippling developer productiv..."
|
||
},
|
||
|
||
{
|
||
"title": "Data mesh at Grab part I: Building trust through certification",
|
||
"url": "/signals-market-place",
|
||
"tags": ["Data","Database","Engineering"],
|
||
"category": null,
|
||
"date": "19 Aug 2025",
|
||
"excerpt": "Grab has embarked on a transformative journey to overhaul its enterprise data ecosystem, addressing challenges posed by the rapid growth of its business span..."
|
||
},
|
||
|
||
{
|
||
"title": "The evolution of Grab's machine learning feature store",
|
||
"url": "/evolution-of-grab-machine-learning-feature-store",
|
||
"tags": ["Database","AWS"],
|
||
"category": null,
|
||
"date": "24 Jul 2025",
|
||
"excerpt": "Learn how Grab is modernising its machine learning platform with a feature table-centric architecture powered by AWS Aurora for Postgres. This shift from a l..."
|
||
},
|
||
|
||
{
|
||
"title": "Grab's service mesh evolution: From Consul to Istio",
|
||
"url": "/service-mesh-evolution",
|
||
"tags": ["Microservice","Service Mesh","Kubernetes","AWS","GCP"],
|
||
"category": null,
|
||
"date": "16 Jul 2025",
|
||
"excerpt": "When you're running 1000+ microservices across Southeast Asia's most complex transport and delivery platform, 'good enough' stops being good enough. Discover..."
|
||
},
|
||
|
||
{
|
||
"title": "DispatchGym: Grab’s reinforcement learning research framework",
|
||
"url": "/techblog_-dispatchgym",
|
||
"tags": ["Dispatch","Python"],
|
||
"category": null,
|
||
"date": "7 Jul 2025",
|
||
"excerpt": "DispatchGym is a research framework that supports reinforcement learning (RL) studies for dispatch systems. A system that matches bookings with drivers. Desi..."
|
||
},
|
||
|
||
{
|
||
"title": "Counter Service: How we rewrote it in Rust",
|
||
"url": "/counter-service-how-we-rewrote-it-in-rust",
|
||
"tags": ["Database","Rust","Data"],
|
||
"category": null,
|
||
"date": "20 Jun 2025",
|
||
"excerpt": "The Integrity Data Platform team at Grab rewrote a QPS-heavy Golang microservice in Rust, achieving 70% infrastructure savings while maintaining similar perf..."
|
||
},
|
||
|
||
{
|
||
"title": "The complete stream processing journey on FlinkSQL",
|
||
"url": "/the-complete-stream-processing-journey-on-flinksql",
|
||
"tags": ["Database","FlinkSQL"],
|
||
"category": null,
|
||
"date": "12 Jun 2025",
|
||
"excerpt": "Introducing FlinkSQL interactive solution to enhance real-time stream processing exploration. The new system simplifies stream processing development, automa..."
|
||
},
|
||
|
||
{
|
||
"title": "Effortless enterprise authentication at Grab: Dex in action",
|
||
"url": "/dex-in-action",
|
||
"tags": ["Access Control","Engineering","Security"],
|
||
"category": null,
|
||
"date": "23 May 2025",
|
||
"excerpt": "This article outlines Grab's journey towards enabling a seamless single sign-on experience for its numerous internal applications. It addresses the challenge..."
|
||
},
|
||
|
||
{
|
||
"title": "From failure to success: The birth of GrabGPT, Grab’s internal ChatGPT",
|
||
"url": "/the-birth-of-grab-gpt",
|
||
"tags": ["Engineering","Optimization","AI","Artificial Intelligence"],
|
||
"category": null,
|
||
"date": "19 May 2025",
|
||
"excerpt": "When Grab's Machine Learning team sought to automate support queries, a failed chatbot experiment sparked an unexpected pivot: GrabGPT. Born from the need to..."
|
||
},
|
||
|
||
{
|
||
"title": "Streamlining RiskOps with the SOP agent framework",
|
||
"url": "/streamlining-riskops-with-sop",
|
||
"tags": ["Engineering","Generative AI","LLM","Experiment","Machine Learning"],
|
||
"category": null,
|
||
"date": "8 May 2025",
|
||
"excerpt": "Discover how the SOP-driven Large Language Model (LLM) agent framework is revolutionising Risk Operations (RiskOps) by automating Account Takeover (ATO) inve..."
|
||
},
|
||
|
||
{
|
||
"title": "Introducing the SOP-driven LLM agent frameworks",
|
||
"url": "/introducing-the-sop-drive-llm-agent-framework",
|
||
"tags": ["Engineering","Generative AI","LLM","Experiment","Machine Learning"],
|
||
"category": null,
|
||
"date": "25 Apr 2025",
|
||
"excerpt": "The SOP-driven Large Language Model (LLM) agent framework, revolutionises enterprise AI by integrating Standard Operating Procedures (SOPs) to ensure reliabl..."
|
||
},
|
||
|
||
{
|
||
"title": "Evaluating performance impact of removing Redis-cache from a Scylla-backed service",
|
||
"url": "/evaluate-performance-remove-redis-from-scylla-service",
|
||
"tags": ["Database","Engineering","Event Processing","Optimization","Redis"],
|
||
"category": null,
|
||
"date": "11 Apr 2025",
|
||
"excerpt": "At Grab, we recently reevaluated a setup that combined Scylla with an external Redis cache. We decided to remove Redis and adjusted our Scylla configurations..."
|
||
},
|
||
|
||
{
|
||
"title": "Facilitating Docs-as-Code implementation for users unfamiliar with Markdown",
|
||
"url": "/facilitating-docs-as-code-with-markdown",
|
||
"tags": ["Blog","TechDocs","Engineering"],
|
||
"category": null,
|
||
"date": "4 Apr 2025",
|
||
"excerpt": "In this article, we'll discuss how we've streamlined the Docs-as-Code process for technical contributors, specifically engineers, who are already familiar wi..."
|
||
},
|
||
|
||
{
|
||
"title": "Improving Hugo stability and addressing oncall challenges through automation",
|
||
"url": "/improving-hugo-stability",
|
||
"tags": ["Data Pipeline","Data Reliability","Data Observability","Platform","System Architecture"],
|
||
"category": null,
|
||
"date": "20 Mar 2025",
|
||
"excerpt": "Managing 4,000+ data pipelines demanded a smarter approach to stability. We built a comprehensive automation solution that enhances Hugo's monitoring capabil..."
|
||
},
|
||
|
||
{
|
||
"title": "Building a Spark observability product with StarRocks: Real-time and historical performance analysis",
|
||
"url": "/building-a-spark-observability",
|
||
"tags": ["Spark Observability","StarRocks","Data Engineering","Real-Time Analytics","System Architecture","Generative AI","LLM"],
|
||
"category": null,
|
||
"date": "6 Mar 2025",
|
||
"excerpt": "Discover how Grab revolutionised its Spark observability with StarRocks! We transformed our monitoring capabilities by moving from a fragmented system to a u..."
|
||
},
|
||
|
||
{
|
||
"title": "TechDocs at Grab: Cultivating a culture of quality documentation",
|
||
"url": "/techdocs-at-grab-cultivating-a-culture-of-quality-documentation",
|
||
"tags": ["Blog","TechDocs","Engineering"],
|
||
"category": null,
|
||
"date": "27 Feb 2025",
|
||
"excerpt": "Discover the steps taken in building a strong documentation culture that produces high-quality content, while making the tools easy to use for everyone invol..."
|
||
},
|
||
|
||
{
|
||
"title": "Grab AI Gateway: Connecting Grabbers to multiple GenAI providers",
|
||
"url": "/grab-ai-gateway",
|
||
"tags": ["Engineering","Data Science","Optimization","Generative AI","LLM","Machine Learning"],
|
||
"category": null,
|
||
"date": "19 Feb 2025",
|
||
"excerpt": "GenAI has become integral to innovation, powering the next generation of AI enabled applications. With easy integration with multiple AI providers, it brings..."
|
||
},
|
||
|
||
{
|
||
"title": "Embracing passwordless authentication with Grab’s Passkey",
|
||
"url": "/embracing-passwordless-authentication-with-passkey",
|
||
"tags": ["Engineering","Security"],
|
||
"category": null,
|
||
"date": "26 Dec 2024",
|
||
"excerpt": "Find out how Passkey makes logging in easier and safer by introducing passwordless authentication. Learn how this new feature works, the benefits it brings, ..."
|
||
},
|
||
|
||
{
|
||
"title": "Turbocharging GrabUnlimited with Temporal",
|
||
"url": "/turbocharging-grabunlimited-with-temporal",
|
||
"tags": ["Engineering","Optimization","Product","Database","Scalability"],
|
||
"category": null,
|
||
"date": "12 Dec 2024",
|
||
"excerpt": "Discover how Grab tackled the challenges of scaling its flagship membership program, GrabUnlimited. In this deep dive, we explore the migration from a legacy..."
|
||
},
|
||
|
||
{
|
||
"title": "How we seamlessly migrated high volume real-time streaming traffic from one service to another with zero data loss and duplication",
|
||
"url": "/seamless-migration",
|
||
"tags": ["Engineering","Optimization","Data Streaming","Real-Time Streaming","Service"],
|
||
"category": null,
|
||
"date": "5 Dec 2024",
|
||
"excerpt": "In the world of high-volume data processing, migrating services without disruption is a formidable challenge. At Grab, we recently undertook this task by spl..."
|
||
},
|
||
|
||
{
|
||
"title": "Supercharging LLM application development with LLM-Kit",
|
||
"url": "/supercharging-llm-application-development-with-llm-kit",
|
||
"tags": ["Engineering","Generative AI","LLM","Machine Learning"],
|
||
"category": null,
|
||
"date": "29 Nov 2024",
|
||
"excerpt": "Discover how Grab's LLM-Kit enhances AI app development by addressing scalability, security, and integration challenges. This article discusses the challenge..."
|
||
},
|
||
|
||
{
|
||
"title": "How we reduced initialisation time of Product Configuration Management SDK",
|
||
"url": "/how-we-reduced-grabx-sdk-initialisation-time",
|
||
"tags": ["Engineering","Optimization","Service"],
|
||
"category": null,
|
||
"date": "22 Nov 2024",
|
||
"excerpt": "Discover how we revolutionised our product configuration management SDK, reducing initialisation time by up to 90%. Learn about the challenges we faced with ..."
|
||
},
|
||
|
||
{
|
||
"title": "Metasense V2: Enhancing, improving and productionisation of LLM powered data governance",
|
||
"url": "/metasense-v2",
|
||
"tags": ["Engineering","Generative AI","LLM","Machine Learning"],
|
||
"category": null,
|
||
"date": "14 Nov 2024",
|
||
"excerpt": "In the initial article, we explored the integration of Large Language Models (LLM) to automate metadata generation, addressing challenges like limited custom..."
|
||
},
|
||
|
||
{
|
||
"title": "How we reduced peak memory and CPU usage of the product configuration management SDK",
|
||
"url": "/reduced-memory-cpu-usage-grabx-sdk",
|
||
"tags": ["Engineering","Optimization","Service"],
|
||
"category": null,
|
||
"date": "30 Oct 2024",
|
||
"excerpt": "Learn about GrabX, Grab’s central platform for product configuration management. This article discusses the steps taken to optimise the SDK, aiming to improv..."
|
||
},
|
||
|
||
{
|
||
"title": "LLM-assisted vector similarity search",
|
||
"url": "/llm-assisted-vector-similarity-search",
|
||
"tags": ["Engineering","Generative AI","LLM","Machine Learning","Experiment"],
|
||
"category": null,
|
||
"date": "23 Oct 2024",
|
||
"excerpt": "Vector similarity search has revolutionised data retrieval, particularly in the context of Retrieval-Augmented Generation in conjunction with advanced Large ..."
|
||
},
|
||
|
||
{
|
||
"title": "Leveraging RAG-powered LLMs for analytical tasks",
|
||
"url": "/transforming-the-analytics-landscape-with-RAG-powered-LLM",
|
||
"tags": ["Engineering","Generative AI","LLM","Experiment","Machine Learning"],
|
||
"category": null,
|
||
"date": "9 Oct 2024",
|
||
"excerpt": "The emergence of Retrieval-Augmented Generation (RAG) has significantly revolutionised Large Language Models (LLMs), propelling them to unprecedented heights..."
|
||
},
|
||
|
||
{
|
||
"title": "Evolution of Catwalk: Model serving platform at Grab",
|
||
"url": "/catwalk-evolution",
|
||
"tags": ["Machine Learning","Models","Data Science","TensorFlow","Kubernetes","Docker"],
|
||
"category": null,
|
||
"date": "1 Oct 2024",
|
||
"excerpt": "Read about the evolution of Catwalk, Grab's model serving platform, from its inception to its current state. Discover how it has evolved to meet the needs of..."
|
||
},
|
||
|
||
{
|
||
"title": "Enabling conversational data discovery with LLMs at Grab",
|
||
"url": "/hubble-data-discovery",
|
||
"tags": ["Data Discovery","AI","Artificial Intelligence","LLM","Documentation","Elasticsearch"],
|
||
"category": null,
|
||
"date": "26 Sep 2024",
|
||
"excerpt": "Discover how Grab is revolutionising data discovery with the power of AI and LLMs. Dive into our journey as we overcome challenges, build groundbreaking tool..."
|
||
},
|
||
|
||
{
|
||
"title": "Bringing Grab’s Live Activity to Android: Enhancing user experience through custom notifications",
|
||
"url": "/live-activity-2",
|
||
"tags": ["Engineering","Android","Exploration"],
|
||
"category": null,
|
||
"date": "23 Sep 2024",
|
||
"excerpt": "Unleashing Live Activity feature for iOS. Live Activity is a feature that enhances user experience by displaying a user interface (UI) outside of the app, de..."
|
||
},
|
||
|
||
{
|
||
"title": "Unveiling the process: The creation of our powerful campaign builder",
|
||
"url": "/the-creation-of-our-powerful-campaign-builder",
|
||
"tags": ["Engineering","Generative AI","LLM","Experiment","Machine Learning"],
|
||
"category": null,
|
||
"date": "10 Sep 2024",
|
||
"excerpt": "Dive into Trident, our real-time event-driven marketing tool at Grab. Explore the build of the core units powering our If This, Then That (IFTTT) logic. Lear..."
|
||
},
|
||
|
||
{
|
||
"title": "Chimera Sandbox: A scalable experimentation and development platform for Notebook services",
|
||
"url": "/chimera-sandbox",
|
||
"tags": ["Engineering","Generative AI","LLM","Experiment","Machine Learning"],
|
||
"category": null,
|
||
"date": "27 Aug 2024",
|
||
"excerpt": "Unleashing the potential of machine learning (ML) with Grab's Chimera Sandbox. This scalable platform facilitates rapid development and experimentation of ML..."
|
||
},
|
||
|
||
{
|
||
"title": "How we improved translation experience with cost efficiency",
|
||
"url": "/improved-translation-experience-with-cost-efficiency",
|
||
"tags": ["Chat","Chat Support","Engineering","GrabChat","Messaging","Translation","Generative AI","LLM"],
|
||
"category": null,
|
||
"date": "5 Aug 2024",
|
||
"excerpt": "Dive into our journey of improving in-app translation experience amidst a post-COVID tourism boom. Discover how we overcame language detection hurdles, craft..."
|
||
},
|
||
|
||
{
|
||
"title": "LLM-powered data classification for data entities at scale",
|
||
"url": "/llm-powered-data-classification",
|
||
"tags": ["Data","Machine Learning","Generative AI"],
|
||
"category": null,
|
||
"date": "15 Jul 2024",
|
||
"excerpt": "With the advent of the Large Language Model (LLM), new possibilities dawned for metadata generation and sensitive data identification at Grab. This prompted ..."
|
||
},
|
||
|
||
{
|
||
"title": "Profile-guided optimisation (PGO) on Grab services",
|
||
"url": "/profile-guided-optimisation",
|
||
"tags": ["Go","Optimization","Experiment","Performance"],
|
||
"category": null,
|
||
"date": "5 Jun 2024",
|
||
"excerpt": "Profile-guided optimisation (PGO) is a method that tracks CPU profile data and uses that data to optimise your application builds. The AI platform team enabl..."
|
||
},
|
||
|
||
{
|
||
"title": "How we evaluated the business impact of marketing campaigns",
|
||
"url": "/evaluate-business-impact-of-marketing-campaigns",
|
||
"tags": ["Marketing","Metrics","Optimization","Statistics","A/B Testing"],
|
||
"category": null,
|
||
"date": "23 May 2024",
|
||
"excerpt": "Discover how Grab assesses marketing effectiveness using advanced attribution models and strategic testing to improve campaign precision and impact."
|
||
},
|
||
|
||
{
|
||
"title": "No version left behind: Our epic journey of GitLab upgrades",
|
||
"url": "/no-version-left-behind-our-epic-journey-of-gitlab-upgrades",
|
||
"tags": ["Stability","Automation","Optimization"],
|
||
"category": null,
|
||
"date": "3 May 2024",
|
||
"excerpt": "Join us as we share our experience in developing and implementing a consistent upgrade routine. This process underscored the significance of adaptability, co..."
|
||
},
|
||
|
||
{
|
||
"title": "Ensuring data reliability and observability in risk systems",
|
||
"url": "/data-observability",
|
||
"tags": ["Data Science","Security","Risk","Data Observability","Data Reliability"],
|
||
"category": null,
|
||
"date": "23 Apr 2024",
|
||
"excerpt": "As the amount of data Grab handles grows, there is an increased need for quick detections for data anomalies (incompleteness or inaccuracy), while keeping it..."
|
||
},
|
||
|
||
{
|
||
"title": "Grab Experiment Decision Engine - a Unified Toolkit for Experimentation",
|
||
"url": "/grabx-decision-engine",
|
||
"tags": ["Data Science","Experiment","Statistics","Econometrics","Python Package"],
|
||
"category": null,
|
||
"date": "9 Apr 2024",
|
||
"excerpt": "Explore how the GrabX Decision Engine, an integral part of Grab's Experimentation platform, streamlines the testing of thousands of experimental variants wee..."
|
||
},
|
||
|
||
{
|
||
"title": "Iris - Turning observations into actionable insights for enhanced decision making",
|
||
"url": "/iris",
|
||
"tags": ["Data Insights","Metrics","Decision Making","Analytics"],
|
||
"category": null,
|
||
"date": "3 Apr 2024",
|
||
"excerpt": "With cross-platform monitoring, a common problem is the difficulty in getting comprehensive and in-depth views on metrics, making it tough to see the big pic..."
|
||
},
|
||
|
||
{
|
||
"title": "Android App Size at Scale with Project Bonsai",
|
||
"url": "/project-bonsai",
|
||
"tags": ["App Size","Optimization","Project Bonsai","App Download Size","App Disk Size","Scalability"],
|
||
"category": null,
|
||
"date": "1 Mar 2024",
|
||
"excerpt": "With the size of our app growing to include more features, Grab recognised it as a potential hurdle for new users with small storage capacities or restricted..."
|
||
},
|
||
|
||
{
|
||
"title": "Enabling near real-time data analytics on the data lake",
|
||
"url": "/enabling-near-realtime-data-analytics",
|
||
"tags": ["Data Analytics","Stream Processing","Kafka","Real-Time"],
|
||
"category": null,
|
||
"date": "23 Feb 2024",
|
||
"excerpt": "As the data lake landscape matures over the years, it presents opportunities to unlock more business value from the data. This correlates with the increased ..."
|
||
},
|
||
|
||
{
|
||
"title": "The journey of building a comprehensive attribution platform",
|
||
"url": "/attribution-platform",
|
||
"tags": ["Attribution Platform","User Journeys","Advertising"],
|
||
"category": null,
|
||
"date": "20 Feb 2024",
|
||
"excerpt": "The Grab superapp offers a comprehensive array of services from ride-hailing and food delivery to financial services. This creates multifaceted user journeys..."
|
||
},
|
||
|
||
{
|
||
"title": "Managing dynamic marketplace content at scale: Grab's approach to content moderation",
|
||
"url": "/dynamic-marketplace",
|
||
"tags": ["Dynamic Marketplace","Content Moderation","Scalability"],
|
||
"category": null,
|
||
"date": "1 Feb 2024",
|
||
"excerpt": "Understand how Grab employs a combination of automated and manual content moderation to manage its dynamic marketplace content efficiently, while also collab..."
|
||
},
|
||
|
||
{
|
||
"title": "Rethinking Stream Processing: Data Exploration",
|
||
"url": "/rethinking-streaming-processing-data-exploration",
|
||
"tags": ["Kafka","Kubernetes","Data Streaming","Deployment","Streaming Applications"],
|
||
"category": null,
|
||
"date": "31 Jan 2024",
|
||
"excerpt": "As Grab matures along the digitalisation journey, it is collecting and streaming event data generated from the end users of its superapp on a larger magnitud..."
|
||
},
|
||
|
||
{
|
||
"title": "Kafka on Kubernetes: Reloaded for fault tolerance",
|
||
"url": "/kafka-on-kubernetes",
|
||
"tags": ["Kafka","Kubernetes","AWS","Data Streaming"],
|
||
"category": null,
|
||
"date": "26 Dec 2023",
|
||
"excerpt": "Dive into this insightful post to explore how Coban, Grab's real-time data streaming platform, has drastically enhanced the fault tolerance on its Kafka on K..."
|
||
},
|
||
|
||
{
|
||
"title": "Championing CyberSecurity: Grab's bug bounty programme in 2023",
|
||
"url": "/cybersec-bug",
|
||
"tags": ["Security","Bug Bounty","HackerOne"],
|
||
"category": null,
|
||
"date": "19 Dec 2023",
|
||
"excerpt": "Since its launch in 2015, Grab’s Bug Bounty programme has made strides in giving back to the global security community and aiding research. Read this article..."
|
||
},
|
||
|
||
{
|
||
"title": "Sliding window rate limits in distributed systems",
|
||
"url": "/frequency-capping",
|
||
"tags": ["Data","Big Data","Rate-Limiting","Frequency Capping","Distributed Systems"],
|
||
"category": null,
|
||
"date": "14 Dec 2023",
|
||
"excerpt": "In the field of distributed systems, there are several common challenges, such as rate limiters and fast queries in big data. In this blog post, we delve int..."
|
||
},
|
||
|
||
{
|
||
"title": "An elegant platform",
|
||
"url": "/an-elegant-platform",
|
||
"tags": ["Data","Data Streaming","Real-Time Streaming","Platformization"],
|
||
"category": null,
|
||
"date": "30 Nov 2023",
|
||
"excerpt": "Supporting real-time data streaming enables our internal users to build intelligent applications and services, a crucial aspect of continuously out-serving o..."
|
||
},
|
||
|
||
{
|
||
"title": "Road localisation in GrabMaps",
|
||
"url": "/road-localisation-grabmaps",
|
||
"tags": ["Maps","Data","Big Data","Data Processing","Hyperlocalization","GrabMaps"],
|
||
"category": null,
|
||
"date": "17 Nov 2023",
|
||
"excerpt": "With GrabMaps powering the Grab superapp we have the opportunity to improve our services and enhance our map with hyperlocal data. No matter the use case, ro..."
|
||
},
|
||
|
||
{
|
||
"title": "Graph modelling guidelines",
|
||
"url": "/graph-modelling-guidelines",
|
||
"tags": ["Graph Technology","Graphs","Graph Networks","Security","Data"],
|
||
"category": null,
|
||
"date": "8 Nov 2023",
|
||
"excerpt": "Graphs are powerful data representations that detect relationships and data linkages between devices. This is very helpful in revealing fraudulent or malicio..."
|
||
},
|
||
|
||
{
|
||
"title": "Scaling marketing for merchants with targeted and intelligent promos",
|
||
"url": "/scaling-marketing-for-merchants",
|
||
"tags": ["Data","Advertising","Scalability","Data Science","Marketing"],
|
||
"category": null,
|
||
"date": "11 Oct 2023",
|
||
"excerpt": "Apart from ensuring advertisements reach the right audience, it is also important to make promos by merchants more targeted and intelligent to help scale the..."
|
||
},
|
||
|
||
{
|
||
"title": "Stepping up marketing for advertisers: Scalable lookalike audience",
|
||
"url": "/scalable-lookalike-audiences",
|
||
"tags": ["Data","Advertising","Scalability","Data Science","Marketing","Lookalike Audience"],
|
||
"category": null,
|
||
"date": "22 Sep 2023",
|
||
"excerpt": "A key challenge in advertising is reaching the right audience who are most likely to use your product. Read this article to find out how the Data Science tea..."
|
||
},
|
||
|
||
{
|
||
"title": "Building hyperlocal GrabMaps",
|
||
"url": "/building-hyperlocal-grabmaps",
|
||
"tags": ["Maps","Data","Big Data","Data Processing","Hyperlocalization","GrabMaps","Navigation"],
|
||
"category": null,
|
||
"date": "30 Aug 2023",
|
||
"excerpt": "Being hyperlocal is a key advantage for GrabMaps. In this article we will explain what being hyperlocal means and how it helps GrabMaps bring value to our dr..."
|
||
},
|
||
|
||
{
|
||
"title": "Streamlining Grab's Segmentation Platform with faster creation and lower latency",
|
||
"url": "/streamlining-grabs-segmentation-platform",
|
||
"tags": ["Backend","Performance"],
|
||
"category": null,
|
||
"date": "15 Aug 2023",
|
||
"excerpt": "Since 2019, Grab's Segmentation Platform has served as a comprehensive solution for user segmentation and audience creation across all business verticals. Th..."
|
||
},
|
||
|
||
{
|
||
"title": "Unsupervised graph anomaly detection - Catching new fraudulent behaviours",
|
||
"url": "/graph-anomaly-model",
|
||
"tags": ["Data Science","Graph Networks","Graphs","Graph Visualization","Security","Fraud Detection","Anomaly Detection","Machine Learning"],
|
||
"category": null,
|
||
"date": "2 Aug 2023",
|
||
"excerpt": "As fraudsters continue to evolve, it becomes more challenging to automatically detect new fraudulent behaviours. At Grab, we are committed to continuously im..."
|
||
},
|
||
|
||
{
|
||
"title": "Zero traffic cost for Kafka consumers",
|
||
"url": "/zero-traffic-cost",
|
||
"tags": ["Engineering","Kafka","Performance","Access Control"],
|
||
"category": null,
|
||
"date": "7 Jul 2023",
|
||
"excerpt": "Grab's data streaming infrastructure runs in the cloud across multiple Availability Zones for high availability and resilience, but this also incurs staggeri..."
|
||
},
|
||
|
||
{
|
||
"title": "Go module proxy at Grab",
|
||
"url": "/go-module-proxy",
|
||
"tags": ["Engineering"],
|
||
"category": null,
|
||
"date": "30 Jun 2023",
|
||
"excerpt": "While consolidating code into a single monorepo has its benefits, there are also several challenges that come with managing a large monorepo like slow perfor..."
|
||
},
|
||
|
||
{
|
||
"title": "PII masking for privacy-grade machine learning",
|
||
"url": "/pii-masking",
|
||
"tags": ["Engineering","Privacy","Data Masking","Machine Learning"],
|
||
"category": null,
|
||
"date": "1 Jun 2023",
|
||
"excerpt": "Data engineers at Grab work with large sets of data to build and train advanced machine learning models to continuously improve our user experience. However,..."
|
||
},
|
||
|
||
{
|
||
"title": "Performance bottlenecks of Go application on Kubernetes with non-integer (floating) CPU allocation",
|
||
"url": "/performance-bottlenecks-go-apps",
|
||
"tags": ["Engineering"],
|
||
"category": null,
|
||
"date": "23 May 2023",
|
||
"excerpt": "At Grab, we have been running our Go based stream processing framework (SPF) on Kubernetes for several years. But as the number of SPF pipelines increases, w..."
|
||
},
|
||
|
||
{
|
||
"title": "How we improved our iOS CI infrastructure with observability tools",
|
||
"url": "/iOS-CI-infrastructure-with-observability-tools",
|
||
"tags": ["iOS","Mobile","Engineering","UI Tests"],
|
||
"category": null,
|
||
"date": "18 May 2023",
|
||
"excerpt": "After upgrading to Xcode 13.1, we noticed a few issues such as instability of the CI tests and high CPU utilisation. Read to find out how the Test Automation..."
|
||
},
|
||
|
||
{
|
||
"title": "2.3x faster using the Go plugin to replace Lua virtual machine",
|
||
"url": "/faster-using-the-go-plugin-to-replace-Lua-VM",
|
||
"tags": ["Engineering","Virtual Machines","Faster","Go Plugin","Lua VM"],
|
||
"category": null,
|
||
"date": "15 May 2023",
|
||
"excerpt": "The Talaria open-source project has made significant improvements by replacing Lua VM with the Go plugin resulting in 2.3x faster performance and memory usag..."
|
||
},
|
||
|
||
{
|
||
"title": "Safer deployment of streaming applications",
|
||
"url": "/safer-flink-deployments",
|
||
"tags": ["Engineering","Deployment","Streaming Applications"],
|
||
"category": null,
|
||
"date": "2 May 2023",
|
||
"excerpt": "As Flink becomes more popular with real-time stream applications, we realise that Flink deployments are sometimes stressful and prone to errors. The Coban te..."
|
||
},
|
||
|
||
{
|
||
"title": "Message Center - Redesigning the messaging experience on the Grab superapp",
|
||
"url": "/message-center",
|
||
"tags": ["Engineering","GrabChat","Redesign","Messaging","Chat Support"],
|
||
"category": null,
|
||
"date": "17 Apr 2023",
|
||
"excerpt": "Grab’s messaging feature was designed for two-party communications, but as our superapp grew to include more features, we became more aware of the limitation..."
|
||
},
|
||
|
||
{
|
||
"title": "Evolution of quality at Grab",
|
||
"url": "/evolution-of-quality",
|
||
"tags": ["Engineering","Technology Stack","Exploration"],
|
||
"category": null,
|
||
"date": "31 Mar 2023",
|
||
"excerpt": "Testing is typically done after development is complete, which often results in bugs being discovered late in the process. Read to find out how Grab has impr..."
|
||
},
|
||
|
||
{
|
||
"title": "How OVO determined the right technology stack for their web-based projects",
|
||
"url": "/determining-tech-stack",
|
||
"tags": ["Engineering","Technology Stack","Exploration"],
|
||
"category": null,
|
||
"date": "21 Mar 2023",
|
||
"excerpt": "As companies grow in today's technology landscape, it often leads to a diverse set of technology stacks being used in different teams, which can lead to bigg..."
|
||
},
|
||
|
||
{
|
||
"title": "Migrating from Role to Attribute-based Access Control",
|
||
"url": "/migrating-to-abac",
|
||
"tags": ["Engineering","Access Control","Security"],
|
||
"category": null,
|
||
"date": "9 Mar 2023",
|
||
"excerpt": "To ensure our consumers continue to be well-protected, we need to ensure our data access measures are compliant with evolving security standards. With more s..."
|
||
},
|
||
|
||
{
|
||
"title": "Securing GitOps pipelines",
|
||
"url": "/securing-gitops-pipeline",
|
||
"tags": ["Engineering","Open Source","Pipelines","Continuous Delivery","Continuous Integration","Optimization"],
|
||
"category": null,
|
||
"date": "1 Mar 2023",
|
||
"excerpt": "This article illustrates how Grab’s real-time data platform team secured GitOps pipelines at scale with our in-house GitOps implementation."
|
||
},
|
||
|
||
{
|
||
"title": "New zoom freezing feature for Geohash plugin",
|
||
"url": "/geohash-plugin",
|
||
"tags": ["Engineering","Geohash","Maps","Open Source"],
|
||
"category": null,
|
||
"date": "21 Feb 2023",
|
||
"excerpt": "Built by Grab, the Geohash Java OpenStreetMap Editor (JOSM) plugin is widely used in map-making, but a common pain point is the inability to zoom in to a spe..."
|
||
},
|
||
|
||
{
|
||
"title": "Graph service platform",
|
||
"url": "/graph-service-platform",
|
||
"tags": ["Engineering","Graph Networks","Graphs","Graph Visualization","Security","Analytics","Fraud Detection"],
|
||
"category": null,
|
||
"date": "5 Jan 2023",
|
||
"excerpt": "Graphs are powerful data representations that detect relationships and data linkages between devices and help reveal fraudulent or malicious users. Learn how..."
|
||
},
|
||
|
||
{
|
||
"title": "Zero trust with Kafka",
|
||
"url": "/zero-trust-with-kafka",
|
||
"tags": ["Engineering","Kafka","Performance","Zero Trust","Access Control"],
|
||
"category": null,
|
||
"date": "7 Dec 2022",
|
||
"excerpt": "In addition to ensuring the high performance and availability of our services, security continues to be one of our highest priorities. Read this article to f..."
|
||
},
|
||
|
||
{
|
||
"title": "How KartaCam powers GrabMaps",
|
||
"url": "/kartacam-powers-grabmaps",
|
||
"tags": ["Engineering","GrabMaps","KartaCam","Maps","Edge AI"],
|
||
"category": null,
|
||
"date": "1 Dec 2022",
|
||
"excerpt": "The foundation for making maps lies in imagery and ensuring that it is fresh, high quality, and collected in an efficient yet low-cost manner. Read this to f..."
|
||
},
|
||
|
||
{
|
||
"title": "Graph for fraud detection",
|
||
"url": "/graph-for-fraud-detection",
|
||
"tags": ["Analytics","Data Science","Security","Graphs","Graph Visualization","Graph Networks","Fraud Detection"],
|
||
"category": null,
|
||
"date": "24 Nov 2022",
|
||
"excerpt": "Fraud detection has become increasingly important in a fast growing business as new fraud patterns arise when a business product is introduced. We need a sus..."
|
||
},
|
||
|
||
{
|
||
"title": "Query expansion based on user behaviour",
|
||
"url": "/query-expansion-based-on-user-behaviour",
|
||
"tags": ["Analytics","Data Science"],
|
||
"category": null,
|
||
"date": "16 Nov 2022",
|
||
"excerpt": "User behaviour data is a gold mine to gain insights about users and help us improve user experience. In this blog, we explore a query expansion framework bas..."
|
||
},
|
||
|
||
{
|
||
"title": "Using mobile sensor data to encourage safer driving",
|
||
"url": "/using-mobile-sensor-data-to-encourage-safer-driving",
|
||
"tags": ["Analytics","Driving Patterns","Data Science","GPS","Security"],
|
||
"category": null,
|
||
"date": "25 Oct 2022",
|
||
"excerpt": "Telematics is most commonly used to monitor vehicle movements and track driving safety, profiling, fleet optimisation and possible productivity improvements...."
|
||
},
|
||
|
||
{
|
||
"title": "Automatic rule backtesting with large quantities of data",
|
||
"url": "/automatic-rule-backtesting",
|
||
"tags": ["Testing","Automation","Backtesting","Data Science"],
|
||
"category": null,
|
||
"date": "8 Sep 2022",
|
||
"excerpt": "At Grab, real-time fraud detection is built on a rule engine. As data scientists and analysts, we need to analyse and simulate a rule on historical data to c..."
|
||
},
|
||
|
||
{
|
||
"title": "How we store and process millions of orders daily",
|
||
"url": "/how-we-store-millions-orders",
|
||
"tags": ["Database","Storage","Distributed Systems","Platform"],
|
||
"category": null,
|
||
"date": "15 Aug 2022",
|
||
"excerpt": "The Grab Order Platform is a distributed system that processes millions of GrabFood or GrabMart orders every day. Learn about how the Grab order platform sto..."
|
||
},
|
||
|
||
{
|
||
"title": "How we automated FAQ responses at Grab",
|
||
"url": "/automated-faq",
|
||
"tags": ["Automation","Knowledge Management","Productivity"],
|
||
"category": null,
|
||
"date": "13 Jul 2022",
|
||
"excerpt": "Most frequently asked questions (FAQ) are repetitive, which hinder on-call engineers' productivity. Read to find out how we automated FAQ responses at Grab, ..."
|
||
},
|
||
|
||
{
|
||
"title": "Graph Networks - 10X investigation with Graph Visualisations",
|
||
"url": "/graph-visualisation",
|
||
"tags": ["Security","Graphs Concepts","Graph Technology","Graph Visualization"],
|
||
"category": null,
|
||
"date": "30 Jun 2022",
|
||
"excerpt": "As fraud schemes get more complex, we need to stay one step ahead by improving fraud investigation methods. Read to find out more about graph visualisation, ..."
|
||
},
|
||
|
||
{
|
||
"title": "How facial recognition technology keeps you safe",
|
||
"url": "/facial-recognition",
|
||
"tags": ["Security","Facial Recognition"],
|
||
"category": null,
|
||
"date": "9 Jun 2022",
|
||
"excerpt": "Facial recognition technology has grown tremendously in recent years due to the rise of deep learning techniques and accelerated digital transformation. Read..."
|
||
},
|
||
|
||
{
|
||
"title": "Graph concepts and applications",
|
||
"url": "/graph-concepts",
|
||
"tags": ["Security","Graphs Concepts","Graph Technology"],
|
||
"category": null,
|
||
"date": "2 Jun 2022",
|
||
"excerpt": "Graph theory-based approaches show the concepts underlying the behaviour of massively complex systems and networks. Read to find out how graphs came about, w..."
|
||
},
|
||
|
||
{
|
||
"title": "Automated Experiment Analysis - Making experimental analysis scalable",
|
||
"url": "/automated-experiment-analysis",
|
||
"tags": ["Experiment","Experimental Analysis","Azure Databricks"],
|
||
"category": null,
|
||
"date": "30 May 2022",
|
||
"excerpt": "Analysts and data scientists invest lots of time into creating trustworthy experiments, which are key to making sound decisions. Read to find out how Automat..."
|
||
},
|
||
|
||
{
|
||
"title": "Search architecture revamp",
|
||
"url": "/search-architecture-revamp",
|
||
"tags": ["Architecture","Optimization","Search"],
|
||
"category": null,
|
||
"date": "17 May 2022",
|
||
"excerpt": "Grab’s search architecture was initially designed to only support exact text matching based on user queries. Find out what problems the Deliveries search tea..."
|
||
},
|
||
|
||
{
|
||
"title": "Embracing a Docs-as-Code approach",
|
||
"url": "/doc-as-code",
|
||
"tags": ["Docs-as-Code","Documentation","TechDocs","Engineering"],
|
||
"category": null,
|
||
"date": "4 May 2022",
|
||
"excerpt": "Read to find out how Grab is using the Docs-as-Code approach to improve technical documentation."
|
||
},
|
||
|
||
{
|
||
"title": "Graph Networks - Striking fraud syndicates in the dark",
|
||
"url": "/graph-networks",
|
||
"tags": ["Graph Networks","Graphs","Fraud Detection","Security"],
|
||
"category": null,
|
||
"date": "28 Apr 2022",
|
||
"excerpt": "As fraudulent entities evolve and get smarter, Grab needs to continuously enhance our defences to protect our consumers. Read to find out how Graph Networks ..."
|
||
},
|
||
|
||
{
|
||
"title": "How we reduced our CI YAML files from 1800 lines to 50 lines",
|
||
"url": "/how-we-reduced-our-ci-yaml",
|
||
"tags": ["CI","Machine Learning","Pipelines","Continuous Integration","Continuous Delivery","Optimization","Rust"],
|
||
"category": null,
|
||
"date": "19 Apr 2022",
|
||
"excerpt": "GitLab and its tooling are are an integral part of the machine learning platform team stack, for continuous delivery of machine learning. One of our core pro..."
|
||
},
|
||
|
||
{
|
||
"title": "How Kafka Connect helps move data seamlessly",
|
||
"url": "/kafka-connect",
|
||
"tags": ["Kafka","Data Processing","Real-Time"],
|
||
"category": null,
|
||
"date": "6 Apr 2022",
|
||
"excerpt": "Grab’s real-time data platform team (Coban) covers the importance of moving data in and out of Kafka easily and how Kafka Connect helps with that."
|
||
},
|
||
|
||
{
|
||
"title": "Supporting large campaigns at scale",
|
||
"url": "/supporting-large-campaigns-at-scale",
|
||
"tags": ["Kafka","Scheduling","Stream Processing","Batch Processing","Scheduled Job"],
|
||
"category": null,
|
||
"date": "1 Apr 2022",
|
||
"excerpt": "Running batch jobs targeting a large user base is a challenge. Find out how we designed our system to tackle the challenge at scale."
|
||
},
|
||
|
||
{
|
||
"title": "How telematics helps Grab to improve safety",
|
||
"url": "/telematics-at-grab",
|
||
"tags": ["Engineering","Data Science","Driving Patterns","Safety","Analytics"],
|
||
"category": null,
|
||
"date": "24 Mar 2022",
|
||
"excerpt": "Coupled with data science, telematics can help to detect traffic events such as harsh braking and unsafe lane changes so we can provide a safer experience fo..."
|
||
},
|
||
|
||
{
|
||
"title": "Real-time data ingestion in Grab",
|
||
"url": "/real-time-data-ingestion",
|
||
"tags": ["Engineering","Data Ingestion"],
|
||
"category": null,
|
||
"date": "14 Mar 2022",
|
||
"excerpt": "When it comes to data ingestion, there are several prevailing issues that come to mind: data inconsistency, integrity and maintenance. Find out how the Caspi..."
|
||
},
|
||
|
||
{
|
||
"title": "Abacus - Issuing points for multiple sources",
|
||
"url": "/abacus-issuing-points-for-multiple-sources",
|
||
"tags": ["Engineering","Event Processing","Optimization","Stream Processing"],
|
||
"category": null,
|
||
"date": "1 Mar 2022",
|
||
"excerpt": "Learn about the challenges of points rewarding and how GrabRewards Points are rewarded for different Grab offerings."
|
||
},
|
||
|
||
{
|
||
"title": "Exposing a Kafka Cluster via a VPC Endpoint Service",
|
||
"url": "/exposing-kafka-cluster",
|
||
"tags": ["Engineering","Cloud","Kafka"],
|
||
"category": null,
|
||
"date": "18 Feb 2022",
|
||
"excerpt": "Establishing communications between cloud resources that are hosted on different Virtual Private Clouds (VPC) can be complex and costly. Find out how the Cob..."
|
||
},
|
||
|
||
{
|
||
"title": "How Grab built a scalable, high-performance ad server",
|
||
"url": "/scalable-ads-server",
|
||
"tags": ["Engineering","Ads","Design"],
|
||
"category": null,
|
||
"date": "11 Feb 2022",
|
||
"excerpt": "Like many businesses, Grab leverages ads to create awareness and increase engagement with our consumers. Read to find out how the GrabAds team built an ad se..."
|
||
},
|
||
|
||
{
|
||
"title": "Biometric authentication - Why do we need it?",
|
||
"url": "/biometrics-authentication",
|
||
"tags": ["Engineering","Security"],
|
||
"category": null,
|
||
"date": "20 Jan 2022",
|
||
"excerpt": "As cyberattacks get more advanced, authentication methods like one-time passwords (OTPs) and personal identification numbers (PINs) are no longer enough to p..."
|
||
},
|
||
|
||
{
|
||
"title": "Using real-world patterns to improve matching in theory and practice",
|
||
"url": "/using-real-world-patterns-to-improve-matching",
|
||
"tags": ["Data Science","Research"],
|
||
"category": null,
|
||
"date": "22 Nov 2021",
|
||
"excerpt": "Find out how real-world patterns can be used to improve algorithm performance when performing bipartite matching for passengers and driver-partners."
|
||
},
|
||
|
||
{
|
||
"title": "Designing products and services based on Jobs to be Done",
|
||
"url": "/designing-products-and-services-based-on-jtbd",
|
||
"tags": ["Design","Product","Database","User Research"],
|
||
"category": null,
|
||
"date": "21 Oct 2021",
|
||
"excerpt": "In this post, we explain how the Jobs to be Done (JTBD) framework helps uncover the JTBD for consumers, as well as how we uncovered the core needs of GrabFoo..."
|
||
},
|
||
|
||
{
|
||
"title": "Search indexing optimisation",
|
||
"url": "/search-indexing-optimisation",
|
||
"tags": ["Engineering","Data","Database","Optimization"],
|
||
"category": null,
|
||
"date": "27 Sep 2021",
|
||
"excerpt": "Learn about the different optimisation techniques when building a search index."
|
||
},
|
||
|
||
{
|
||
"title": "Automating Multi-Armed Bandit testing during feature rollout",
|
||
"url": "/multi-armed-bandit-system-recommendation",
|
||
"tags": ["Engineering","Testing","Optimization"],
|
||
"category": null,
|
||
"date": "1 Sep 2021",
|
||
"excerpt": "Find out how you can run an automated test and simultaneously roll out a new feature."
|
||
},
|
||
|
||
{
|
||
"title": "How We Cut GrabFood.com’s Page JavaScript Asset Sizes by 3x",
|
||
"url": "/grabfood-bundle-size",
|
||
"tags": ["Product","Asset Size","Cloud","Optimization"],
|
||
"category": null,
|
||
"date": "29 Jul 2021",
|
||
"excerpt": "Find out how the GrabFood team cut their bundle size by 3 times with these 7 webpack bundle optimisation strategies."
|
||
},
|
||
|
||
{
|
||
"title": "Protecting Personal Data in Grab's Imagery",
|
||
"url": "/protecting-personal-data-in-grabs-imagery",
|
||
"tags": ["Engineering","Machine Learning","Data","Datasets","Data Science"],
|
||
"category": null,
|
||
"date": "26 Jul 2021",
|
||
"excerpt": "Learn how Grab improves privacy protection to cater to various geographical locations."
|
||
},
|
||
|
||
{
|
||
"title": "Processing ETL tasks with Ratchet",
|
||
"url": "/processing-etl-tasks-with-ratchet",
|
||
"tags": ["Pipelines","Data","ETL","Engineering"],
|
||
"category": null,
|
||
"date": "19 Jul 2021",
|
||
"excerpt": "Read about what Data and ETL pipelines are and how they are used for processing multiple tasks in the Lending Team at Grab."
|
||
},
|
||
|
||
{
|
||
"title": "App Modularisation at Scale",
|
||
"url": "/app-modularisation-at-scale",
|
||
"tags": ["App","Build Time","Engineering","Monorepo"],
|
||
"category": null,
|
||
"date": "13 Jul 2021",
|
||
"excerpt": "Read up to know how we improved our app’s build time performance and developer experience at Grab."
|
||
},
|
||
|
||
{
|
||
"title": "Reshaping Chat Support for Our Users",
|
||
"url": "/reshaping-chat-support",
|
||
"tags": ["Product","Design","Chat Support"],
|
||
"category": null,
|
||
"date": "7 Jul 2021",
|
||
"excerpt": "Learn how the Grab Support (GS) Tech Family reshaped chat support through experimentation and design updates."
|
||
},
|
||
|
||
{
|
||
"title": "Debugging High Latency Due to Context Leaks",
|
||
"url": "/debugging-high-latency-market-store",
|
||
"tags": ["Engineering","Latency","Debugging","Memory Leak"],
|
||
"category": null,
|
||
"date": "30 Jun 2021",
|
||
"excerpt": "Learn how the Marketplace Tech Family debugged and resolved Market-Store's high latency issues."
|
||
},
|
||
|
||
{
|
||
"title": "Building a Hyper Self-Service, Distributed Tracing and Feedback System for Rule & Machine Learning (ML) Predictions",
|
||
"url": "/building-hyper-self-service-distributed-tracing-feedback-system",
|
||
"tags": ["Engineering","Machine Learning","Statistics","Distributed Tracing"],
|
||
"category": null,
|
||
"date": "24 May 2021",
|
||
"excerpt": "Find out how the Trust, Identity, Safety, and Security (TISS) team improved machine learning predictions with Archivist, an in-house built solution."
|
||
},
|
||
|
||
{
|
||
"title": "Our Journey to Continuous Delivery at Grab (Part 2)",
|
||
"url": "/our-journey-to-continuous-delivery-at-grab-part2",
|
||
"tags": ["Deployment","CI","Continuous Integration","Continuous Deployment","Deployment Process","Continuous Delivery","Multi Cloud","Hermetic Deployments","Automation"],
|
||
"category": null,
|
||
"date": "10 May 2021",
|
||
"excerpt": "Read more about our long awaited piece on the automation work we have made through integration and hermeticity."
|
||
},
|
||
|
||
{
|
||
"title": "How We Improved Agent Chat Efficiency with Machine Learning",
|
||
"url": "/how-we-improved-agent-chat-efficiency-with-ml",
|
||
"tags": ["Engineering","Machine Learning","Consumer Support"],
|
||
"category": null,
|
||
"date": "19 Apr 2021",
|
||
"excerpt": "Read to find out how Customer Support Experience's Phoenix live chat team improved agent chat efficiency with machine learning."
|
||
},
|
||
|
||
{
|
||
"title": "How Grab Leveraged Performance Marketing Automation to Improve Conversion Rates by 30%",
|
||
"url": "/learn-how-grab-leveraged-performance-marketing-automation",
|
||
"tags": ["Automation","Engineering","Marketing"],
|
||
"category": null,
|
||
"date": "22 Mar 2021",
|
||
"excerpt": "Read to find out how Grab's Performance Marketing team leveraged on automation to improve conversion rates."
|
||
},
|
||
|
||
{
|
||
"title": "One Small Step Closer to Containerising Service Binaries",
|
||
"url": "/reducing-your-go-binary-size",
|
||
"tags": ["Backend","Engineering","Golang","Cloud-Native Transformations","Containerization","Kubernetes"],
|
||
"category": null,
|
||
"date": "23 Feb 2021",
|
||
"excerpt": "Learn how Grab is investigating and reducing service binary size for Golang projects."
|
||
},
|
||
|
||
{
|
||
"title": "Customer Support Workforce Routing",
|
||
"url": "/customer-support-workforce-routing",
|
||
"tags": ["Workforce Routing","Chat","Product","Routing","Queuing","Customer Support"],
|
||
"category": null,
|
||
"date": "5 Feb 2021",
|
||
"excerpt": "Read how we built our in-house workforce routing system at Grab."
|
||
},
|
||
|
||
{
|
||
"title": "Serving Driver-partners Data at Scale Using Mirror Cache",
|
||
"url": "/mirror-cache-blog",
|
||
"tags": ["Mirror Cache","Data at Scale"],
|
||
"category": null,
|
||
"date": "26 Jan 2021",
|
||
"excerpt": "Find out how a team at Grab used Mirror Cache, an in-memory local caching solution, to serve driver-partners data efficiently."
|
||
},
|
||
|
||
{
|
||
"title": "The GrabMart Journey",
|
||
"url": "/grabmart-product-team-experience",
|
||
"tags": ["GrabMart","Product"],
|
||
"category": null,
|
||
"date": "18 Jan 2021",
|
||
"excerpt": "Learn how Grab catered to user demand on GrabMart during COVID-19."
|
||
},
|
||
|
||
{
|
||
"title": "Trident - Real-time Event Processing at Scale",
|
||
"url": "/trident-real-time-event-processing-at-scale",
|
||
"tags": ["A/B Testing","Event Processing"],
|
||
"category": null,
|
||
"date": "13 Jan 2021",
|
||
"excerpt": "Find out where the messages and rewards come from, that arrive on your Grab app. Walk through scaling and processing optimisations that achieve tremendous th..."
|
||
},
|
||
|
||
{
|
||
"title": "Pharos - Searching Nearby Drivers on Road Network at Scale",
|
||
"url": "/pharos-searching-nearby-drivers-on-road-network-at-scale",
|
||
"tags": ["Real-Time K Nearest Neighbor Search","Spatial Data Store","Distributed Systems"],
|
||
"category": null,
|
||
"date": "22 Dec 2020",
|
||
"excerpt": "Learn how Grab stores driver locations and how these locations are used to find nearby drivers around you."
|
||
},
|
||
|
||
{
|
||
"title": "Reflecting on the Five Years of Bug Bounty at Grab",
|
||
"url": "/reflecting-on-the-five-years-of-bug-bounty-at-grab",
|
||
"tags": ["Security","HackerOne","Bug Bounty"],
|
||
"category": null,
|
||
"date": "16 Dec 2020",
|
||
"excerpt": "Read about how the product security team's bug bounty programme has helped keep Grab secure."
|
||
},
|
||
|
||
{
|
||
"title": "How Grab is Blazing Through the Superapp Bazel Migration",
|
||
"url": "/how-grab-is-blazing-through-the-super-app-bazel-migration",
|
||
"tags": ["Bazel","Android","iOS","Build Time","Xcode","Gradle"],
|
||
"category": null,
|
||
"date": "3 Dec 2020",
|
||
"excerpt": "Learn how we planned and started migrating our superapp to Bazel at Grab."
|
||
},
|
||
|
||
{
|
||
"title": "Democratising Fare Storage at Scale Using Event Sourcing",
|
||
"url": "/democratising-fare-storage-at-scale-using-event-sourcing",
|
||
"tags": ["Pricing","Event Sourcing","Fare Storage"],
|
||
"category": null,
|
||
"date": "23 Nov 2020",
|
||
"excerpt": "Read how we built Grab's single source of truth for fare storage and management. In this post, we explain how we used the Event Sourcing pattern to build our..."
|
||
},
|
||
|
||
{
|
||
"title": "Keeping 170 Libraries Up to Date on a Large Scale Android App",
|
||
"url": "/keeping-170-libraries-up-to-date-on-a-large-scale-android-app",
|
||
"tags": ["Mobile","Android","Engineering"],
|
||
"category": null,
|
||
"date": "30 Oct 2020",
|
||
"excerpt": "Learn how we maintain our libraries and prevent defect leaks in our Grab Passenger app."
|
||
},
|
||
|
||
{
|
||
"title": "Optimally Scaling Kafka Consumer Applications",
|
||
"url": "/optimally-scaling-kafka-consumer-applications",
|
||
"tags": ["Event Sourcing","Stream Processing","Kubernetes","Backend","Platform","Go"],
|
||
"category": null,
|
||
"date": "13 Oct 2020",
|
||
"excerpt": "Read this deep dive on our Kubernetes infrastructure setup for Grab's stream processing framework."
|
||
},
|
||
|
||
{
|
||
"title": "Our Journey to Continuous Delivery at Grab (Part 1)",
|
||
"url": "/our-journey-to-continuous-delivery-at-grab",
|
||
"tags": ["Deployment","CI","Continuous Integration","Continuous Deployment","Deployment Process","Cloud Agnostic","Spinnaker","Continuous Delivery","Multi Cloud"],
|
||
"category": null,
|
||
"date": "23 Sep 2020",
|
||
"excerpt": "Continuous Delivery is the principle of delivering software often, everyday. Read more to find out how we implemented continuous delivery at Grab."
|
||
},
|
||
|
||
{
|
||
"title": "Uncovering the Truth Behind Lua and Redis Data Consistency",
|
||
"url": "/uncovering-the-truth-behind-lua-and-redis-data-consistency",
|
||
"tags": ["Redis","Lua Scripts","High CPU Usage","Data Consistency"],
|
||
"category": null,
|
||
"date": "7 Sep 2020",
|
||
"excerpt": "Redis does not guarantee the consistency between master and its replica nodes when Lua scripts are used. Read more to find out why and how to guarantee data ..."
|
||
},
|
||
|
||
{
|
||
"title": "Securing and Managing Multi-cloud Presto Clusters with Grab’s DataGateway",
|
||
"url": "/data-gateway",
|
||
"tags": ["Engineering","Presto","Data","Data Pipeline","Access Control","Workload Distribution","Cluster"],
|
||
"category": null,
|
||
"date": "24 Aug 2020",
|
||
"excerpt": "This blog post discusses how Grab's DataGateway plays a key role in supporting hundreds of users in our entire Presto ecosystem - from managing user access, ..."
|
||
},
|
||
|
||
{
|
||
"title": "Go Modules- A Guide for monorepos (Part 2)",
|
||
"url": "/go-module-a-guide-for-monorepos-part-2",
|
||
"tags": ["Go","Monorepo","Vendoring","Vendors","Libraries"],
|
||
"category": null,
|
||
"date": "12 Aug 2020",
|
||
"excerpt": "This is the second post on the Go module series, which highlights Grab’s experience working with Go modules in a multi-module monorepo. Here, we discuss the ..."
|
||
},
|
||
|
||
{
|
||
"title": "The Journey of Deploying Apache Airflow at Grab",
|
||
"url": "/the-journey-of-deploying-apache-airflow-at-Grab",
|
||
"tags": ["Engineering","Data Pipeline","Scheduling","Airflow","Kubernetes","Platform"],
|
||
"category": null,
|
||
"date": "14 Jul 2020",
|
||
"excerpt": "This blog post shares how we designed and implemented an Apache Airflow-based scheduling and orchestration platform for teams across Grab."
|
||
},
|
||
|
||
{
|
||
"title": "How We Built Our In-house Chat Platform for the Web",
|
||
"url": "/how-we-built-our-in-house-chat-platform-for-the-web",
|
||
"tags": ["Chat","Web","Customer Support","Engineering"],
|
||
"category": null,
|
||
"date": "29 Jun 2020",
|
||
"excerpt": "This blog post shares our learnings from building our very own chat platform for the web."
|
||
},
|
||
|
||
{
|
||
"title": "Go Modules- A Guide for monorepos (Part 1)",
|
||
"url": "/go-module-a-guide-for-monorepos-part-1",
|
||
"tags": ["Go","Monorepo","Vendoring","Vendors","Libraries"],
|
||
"category": null,
|
||
"date": "29 May 2020",
|
||
"excerpt": "This post is the first in a series of blogs about Grab’s experience with Go modules in a multi-module monorepo. Here, we discuss the challenges we faced alon..."
|
||
},
|
||
|
||
{
|
||
"title": "Does Southeast Asia Run on Coffee?",
|
||
"url": "/does-southeast-asia-run-on-coffee",
|
||
"tags": ["Data","Data Analytics","Data Visualization"],
|
||
"category": null,
|
||
"date": "26 Mar 2020",
|
||
"excerpt": "This blog post shares insights on GrabFood data around how much our fellow Southeast Asians love coffee."
|
||
},
|
||
|
||
{
|
||
"title": "GrabChat Much? Talk Data to Me!",
|
||
"url": "/grabchat-much-talk-data-to-me",
|
||
"tags": ["Data","Data Analytics","Data Visualization"],
|
||
"category": null,
|
||
"date": "24 Mar 2020",
|
||
"excerpt": "This blog post uncovers some interesting insights from our GrabChat data in Singapore, Malaysia, and Indonesia."
|
||
},
|
||
|
||
{
|
||
"title": "7 Fun Facts about Grab’s Driver-Partners in Singapore",
|
||
"url": "/seven-facts-about-grab-driver-partners-in-sg",
|
||
"tags": ["Data","Data Analytics"],
|
||
"category": null,
|
||
"date": "20 Mar 2020",
|
||
"excerpt": "This blog post shares the most interesting data points from 2019 about our Singapore driver-partners."
|
||
},
|
||
|
||
{
|
||
"title": "Tackling UI Test Execution Time Imbalance for Xcode Parallel Testing",
|
||
"url": "/tackling-ui-test-execution-time-imbalance-for-xcode-parallel-testing",
|
||
"tags": ["Xcode","Testing","Mobile","Parallelism","UI Tests","CI","iOS"],
|
||
"category": null,
|
||
"date": "16 Mar 2020",
|
||
"excerpt": "This blog post introduces how we use Xcode parallel testing to balance test execution time and improve the parallelism of our systems. We also share how we o..."
|
||
},
|
||
|
||
{
|
||
"title": "Returning 575 Terabytes of Storage Space to Our Users",
|
||
"url": "/returning-storage-space-back-to-our-users",
|
||
"tags": ["Mobile","Android","Performance"],
|
||
"category": null,
|
||
"date": "25 Feb 2020",
|
||
"excerpt": "This blog explains how we measured and reduced our app's storage footprint on user devices."
|
||
},
|
||
|
||
{
|
||
"title": "Grab-Posisi - Southeast Asia’s First Comprehensive GPS Trajectory Dataset",
|
||
"url": "/grab-posisi",
|
||
"tags": ["GPS","Datasets","Maps"],
|
||
"category": null,
|
||
"date": "20 Feb 2020",
|
||
"excerpt": "This blog highlights Grab's latest GPS trajectory dataset - its content, format, applications, and how you can access the dataset for your research purpose."
|
||
},
|
||
|
||
{
|
||
"title": "How We Prevented App Performance Degradation from Sudden Ride Demand Spikes",
|
||
"url": "/preventing-app-performance-degradation-due-to-sudden-ride-demand-spikes",
|
||
"tags": ["Resiliency","Circuit Breakers"],
|
||
"category": null,
|
||
"date": "8 Jan 2020",
|
||
"excerpt": "This blog addresses how engineers overcame the challenges Grab faced during the initial days due to sudden spike in ride demand."
|
||
},
|
||
|
||
{
|
||
"title": "Plumbing At Scale",
|
||
"url": "/plumbing-at-scale",
|
||
"tags": ["Event Sourcing","Stream Processing","Kubernetes","Backend","Platform","Go"],
|
||
"category": null,
|
||
"date": "6 Jan 2020",
|
||
"excerpt": "This article details our journey building and deploying an event sourcing platform in Go, building a stream processing framework over it, and then scaling it..."
|
||
},
|
||
|
||
{
|
||
"title": "Journey to a Faster Everyday Superapp Where Every Millisecond Counts",
|
||
"url": "/journey-to-a-faster-everyday-super-app",
|
||
"tags": ["Superapp","Mobile","Performance"],
|
||
"category": null,
|
||
"date": "26 Dec 2019",
|
||
"excerpt": "This post narrates the journey of our performance improvement efforts on the Grab passenger app. It highlights how we were able to reduce the time spent star..."
|
||
},
|
||
|
||
{
|
||
"title": "Marionette - Enabling E2E User-scenario Simulation",
|
||
"url": "/marionette-enabling-e2e-user-scenario-simulation",
|
||
"tags": ["Backend","Testing","Microservice"],
|
||
"category": null,
|
||
"date": "23 Dec 2019",
|
||
"excerpt": "Do you know how we get early feedback on any breaking changes? Read through our blog to find out how Marionette, an in-house simulation platform, detects bre..."
|
||
},
|
||
|
||
{
|
||
"title": "How We Implemented Domain-Driven Development in Golang",
|
||
"url": "/domain-driven-development-in-golang",
|
||
"tags": ["Backend","Go"],
|
||
"category": null,
|
||
"date": "21 Nov 2019",
|
||
"excerpt": "Are you curious how we quickly enabled our partners to self-service using our platform? Have you wondered how some teams at Grab implemented domain-driven de..."
|
||
},
|
||
|
||
{
|
||
"title": "Driving Southeast Asia Forward Through People-Focused Design",
|
||
"url": "/driving-sea-forward-through-people-focused-design",
|
||
"tags": ["Design","User Research"],
|
||
"category": null,
|
||
"date": "5 Nov 2019",
|
||
"excerpt": "How do you design for a heavily diversified market like Southeast Asia? In this article, I’ll share key consumer insights that have guided my decisions and i..."
|
||
},
|
||
|
||
{
|
||
"title": "Griffin, an Anti-fraud Risk Rule Engine Making Billions of Predictions Daily",
|
||
"url": "/griffin",
|
||
"tags": ["Engineering","Anti-Fraud","Security","Fraud Detection","Data"],
|
||
"category": null,
|
||
"date": "28 Oct 2019",
|
||
"excerpt": "This blog highlights Grab’s high-performance risk rule engine that automates the creation of rules to detect fraudulent activities with minimal efforts by en..."
|
||
},
|
||
|
||
{
|
||
"title": "Using Grab’s Trust Counter Service to Detect Fraud Successfully",
|
||
"url": "/using-grabs-trust-counter-service-to-detect-fraud-successfully",
|
||
"tags": ["Engineering","Anti-Fraud","Security","Fraud Detection","Data"],
|
||
"category": null,
|
||
"date": "21 Oct 2019",
|
||
"excerpt": "This blog introduces Grab’s Trust Counter service for detecting fraud. It explains how the solution was designed so that different stakeholders like data ana..."
|
||
},
|
||
|
||
{
|
||
"title": "Being a Principal Engineer at Grab",
|
||
"url": "/about-being-a-principal-engineer-at-grab",
|
||
"tags": ["Career","Engineering","Microservice"],
|
||
"category": null,
|
||
"date": "25 Sep 2019",
|
||
"excerpt": "Curious about what a Principal Engineer role at Grab entails? Our Principal Engineers' responsibilities range from solving complex problems, taking care of t..."
|
||
},
|
||
|
||
{
|
||
"title": "Data First, SLA Always",
|
||
"url": "/data-first-sla-always",
|
||
"tags": ["Data Pipeline"],
|
||
"category": null,
|
||
"date": "1 Aug 2019",
|
||
"excerpt": "Introducing Trailblazer, the Data Engineering team’s solution to implementing change data capture of all upstream databases. In this article, we introduce th..."
|
||
},
|
||
|
||
{
|
||
"title": "Save Your Place with Grab!",
|
||
"url": "/save-your-place-with-grab",
|
||
"tags": ["Maps","Data"],
|
||
"category": null,
|
||
"date": "1 Aug 2019",
|
||
"excerpt": "Do you find it tedious to type and search for your destination or have a hard time remembering that address of the friend you are going to meet? Well...Grab ..."
|
||
},
|
||
|
||
{
|
||
"title": "No More Forgetting to Input ERP Charges - Hello Automated ERP!",
|
||
"url": "/automated-erp-charges",
|
||
"tags": ["Data","Maps","Tech"],
|
||
"category": null,
|
||
"date": "31 Jul 2019",
|
||
"excerpt": "Forgetting to input ERP charges was one of the biggest problems faced by our driver-partners. Read our blog to know how went about solving it."
|
||
},
|
||
|
||
{
|
||
"title": "How We Built a Logging Stack at Grab",
|
||
"url": "/how-built-logging-stack",
|
||
"tags": ["Logging"],
|
||
"category": null,
|
||
"date": "31 Jul 2019",
|
||
"excerpt": "This blog post explains what we did to solve our inhouse logging problem around the lack of visualizations and metrics for our service logs."
|
||
},
|
||
|
||
{
|
||
"title": "Making Grab’s Everyday App Super",
|
||
"url": "/grab-everyday-super-app",
|
||
"tags": ["Superapp","Feed","Recommendations","Data Science","Machine Learning"],
|
||
"category": null,
|
||
"date": "3 Jul 2019",
|
||
"excerpt": "To excel in a heavily diversified market like Southeast Asia, we leverage on the depth of our data to understand what sorts of information users want to see ..."
|
||
},
|
||
|
||
{
|
||
"title": "Catwalk: Serving Machine Learning Models at Scale",
|
||
"url": "/catwalk-serving-machine-learning-models-at-scale",
|
||
"tags": ["Machine Learning","Models","Data Science","TensorFlow"],
|
||
"category": null,
|
||
"date": "2 Jul 2019",
|
||
"excerpt": "This blog post explains why and how we came up with a machine learning model serving platform to accelerate the use of machine learning in Grab."
|
||
},
|
||
|
||
{
|
||
"title": "React Native in GrabPay",
|
||
"url": "/react-native-in-grabpay",
|
||
"tags": ["Grab","Mobile","GrabPay","React"],
|
||
"category": null,
|
||
"date": "30 May 2019",
|
||
"excerpt": "This blog post describes how we used React Native to optimize the Grab PAX app."
|
||
},
|
||
|
||
{
|
||
"title": "Connecting the Invisibles to Design Seamless Experiences",
|
||
"url": "/connecting-the-invisibles-to-design-seamless-experiences",
|
||
"tags": ["Design","Service Design"],
|
||
"category": null,
|
||
"date": "28 May 2019",
|
||
"excerpt": "Much of the work done by the service design team at Grab revolves around integrating people, processes, and systems to deliver seamless user experiences. In ..."
|
||
},
|
||
|
||
{
|
||
"title": "Tourists on GrabChat!",
|
||
"url": "/tourist-chat-data-story",
|
||
"tags": ["Data","Analytics","Data Analytics"],
|
||
"category": null,
|
||
"date": "22 May 2019",
|
||
"excerpt": "Just over two years ago we introduced GrabChat, Southeast Asia’s first of its kind in-app messaging platform. Since then we’ve added all sorts of useful feat..."
|
||
},
|
||
|
||
{
|
||
"title": "Bubble Tea Craze on GrabFood!",
|
||
"url": "/bubble-tea-craze-on-grabfood",
|
||
"tags": ["Data","Analytics","Data Analytics"],
|
||
"category": null,
|
||
"date": "9 May 2019",
|
||
"excerpt": "Bubble Tea’s popularity on GrabFood has captured our attention and we want to celebrate its fascinating growth with you! We have deep-dived into Grab’s Big D..."
|
||
},
|
||
|
||
{
|
||
"title": "Why You Should Organise an Immersion Trip for Your Next Project",
|
||
"url": "/why-you-should-organise-an-immersion-trip-for-your-next-project",
|
||
"tags": ["Hyperlocal","Immersion"],
|
||
"category": null,
|
||
"date": "7 May 2019",
|
||
"excerpt": "Sherizan Sheikh is a Design Lead at Grab Ventures, an incubation arm that looks at experiences beyond ride-hailing, for example, groceries, healthcare and au..."
|
||
},
|
||
|
||
{
|
||
"title": "Preventing Pipeline Calls from Crashing Redis Clusters",
|
||
"url": "/preventing-pipeline-calls-from-crashing-redis-clusters",
|
||
"tags": ["Grab","Backend","Redis","Redis Cluster","Go"],
|
||
"category": null,
|
||
"date": "5 May 2019",
|
||
"excerpt": "This blog post describes Grab’s post-mortem findings for the outage caused by the Redis Cluster failure."
|
||
},
|
||
|
||
{
|
||
"title": "Guiding You Door-to-Door via Our Superapp!",
|
||
"url": "/poi-entrances-venues-door-to-door",
|
||
"tags": ["Grab","Data","Tech","Maps","App"],
|
||
"category": null,
|
||
"date": "12 Apr 2019",
|
||
"excerpt": "Insights into how Grab is trying to solve pickup issues when you book from large venues such as airports or malls."
|
||
},
|
||
|
||
{
|
||
"title": "Loki, a Dynamic Mock Server for HTTP/TCP Testing",
|
||
"url": "/loki-dynamic-mock-server-http-tcp-testing",
|
||
"tags": ["Backend","Service","Mobile","Testing"],
|
||
"category": null,
|
||
"date": "10 Apr 2019",
|
||
"excerpt": "Read our blog to know how Loki, a dynamic mock server, makes local box testing of mobile apps easy, repeatable, and exhaustive. It supports both HTTP and TCP..."
|
||
},
|
||
|
||
{
|
||
"title": "How We Harnessed the Wisdom of Crowds to Improve Restaurant Location Accuracy",
|
||
"url": "/correcting-restaurant-locations-harnessing-wisdom-of-the-crowd",
|
||
"tags": ["Data Science"],
|
||
"category": null,
|
||
"date": "2 Apr 2019",
|
||
"excerpt": "We questioned some of the estimates that our algorithm for calculating restaurant wait times was making, and found that the \"errors\" were actually useful to ..."
|
||
},
|
||
|
||
{
|
||
"title": "Designing Resilient Systems Beyond Retries (Part 3): Architecture Patterns and Chaos Engineering",
|
||
"url": "/beyond-retries-part-3",
|
||
"tags": ["Resiliency","Microservice","Chaos Engineering"],
|
||
"category": null,
|
||
"date": "27 Mar 2019",
|
||
"excerpt": "This post is the third of a three-part series on going beyond retries and circuit breakers to improve system resiliency. This whole series covers techniques ..."
|
||
},
|
||
|
||
{
|
||
"title": "Designing Resilient Systems Beyond Retries (Part 2): Bulkheading, Load Balancing, and Fallbacks",
|
||
"url": "/beyond-retries-part-2",
|
||
"tags": ["Resiliency","Microservice","Bulkheading","Load Balancing","Fallbacks"],
|
||
"category": null,
|
||
"date": "25 Mar 2019",
|
||
"excerpt": "This post is the second of a three-part series on going beyond retries to improve system resiliency. We’ve previously discussed about rate-limiting as a stra..."
|
||
},
|
||
|
||
{
|
||
"title": "Designing Resilient Systems Beyond Retries (Part 1): Rate-Limiting",
|
||
"url": "/beyond-retries-part-1",
|
||
"tags": ["Resiliency","Microservice","Rate-Limiting"],
|
||
"category": null,
|
||
"date": "20 Mar 2019",
|
||
"excerpt": "This post is the first of a three-part series on going beyond retries to improve system resiliency. In this series, we will discuss other techniques and arch..."
|
||
},
|
||
|
||
{
|
||
"title": "Context Deadlines and How to Set Them",
|
||
"url": "/context-deadlines-and-how-to-set-them",
|
||
"tags": ["Resiliency","Microservice"],
|
||
"category": null,
|
||
"date": "11 Mar 2019",
|
||
"excerpt": "This blog post explains from the ground up a strategy for configuring timeouts and using context deadlines correctly, drawing from our experience developing ..."
|
||
},
|
||
|
||
{
|
||
"title": "Recipe for Building a Widget: How We Helped to “Peak-Shift” Demand by Helping Passengers Understand Travel Trends",
|
||
"url": "/peak-shift-demand-travel-trends",
|
||
"tags": ["Analytics","Data","Data Analytics"],
|
||
"category": null,
|
||
"date": "7 Mar 2019",
|
||
"excerpt": "We help to “peak-shift” demand by helping passengers understand travel trends with Grab’s data. Curious to know how we empower our passengers to make better ..."
|
||
},
|
||
|
||
{
|
||
"title": "Structured Logging: The Best Friend You’ll Want When Things Go Wrong",
|
||
"url": "/structured-logging",
|
||
"tags": ["Logging"],
|
||
"category": null,
|
||
"date": "5 Mar 2019",
|
||
"excerpt": "This blog post describes how we built a structured logging framework that integrates well with our existing Elastic stack-based logging backend, allowing us ..."
|
||
},
|
||
|
||
{
|
||
"title": "How We Simplified Our Data Ingestion & Transformation Process",
|
||
"url": "/data-ingestion-transformation-product-insights",
|
||
"tags": ["Big Data","Data Pipeline"],
|
||
"category": null,
|
||
"date": "3 Mar 2019",
|
||
"excerpt": "This blog post describes how Grab built a scalable data ingestion system and how we went from prototyping with Spark Streaming to running a production-grade ..."
|
||
},
|
||
|
||
{
|
||
"title": "Understanding Supply & Demand in Ride-hailing Through the Lens of Data",
|
||
"url": "/understanding-supply-demand-ride-hailing-data",
|
||
"tags": ["Analytics","Data","Data Analytics","Data Visualization","Data Storytelling"],
|
||
"category": null,
|
||
"date": "20 Feb 2019",
|
||
"excerpt": "Grab aims to ensure that our passengers can get a ride conveniently while providing our drivers better livelihood. To achieve this, balancing demand and supp..."
|
||
},
|
||
|
||
{
|
||
"title": "A Lean and Scalable Data Pipeline to Capture Large Scale Events and Support Experimentation Platform",
|
||
"url": "/experimentation-platform-data-pipeline",
|
||
"tags": ["Big Data","Data Pipeline","Experiment"],
|
||
"category": null,
|
||
"date": "16 Jan 2019",
|
||
"excerpt": "This blog post focuses on the lessons we learned while building our batch data pipeline."
|
||
},
|
||
|
||
{
|
||
"title": "Designing Resilient Systems: Circuit Breakers or Retries? (Part 2)",
|
||
"url": "/designing-resilient-systems-part-2",
|
||
"tags": ["Resiliency","Circuit Breakers"],
|
||
"category": null,
|
||
"date": "8 Jan 2019",
|
||
"excerpt": "Grab designs fault-tolerant systems that can withstand failures allowing us to continuously provide our consumers with the many services they expect from us."
|
||
},
|
||
|
||
{
|
||
"title": "Querying Big Data in Real-time with Presto & Grab's TalariaDB",
|
||
"url": "/big-data-real-time-presto-talariadb",
|
||
"tags": ["Big Data","Real-Time","Database","Presto","TalariaDB"],
|
||
"category": null,
|
||
"date": "2 Jan 2019",
|
||
"excerpt": "In this article, we focus on TalariaDB, a distributed, highly available, and low latency time-series database that stores real-time data. For example, logs, ..."
|
||
},
|
||
|
||
{
|
||
"title": "Designing Resilient Systems: Circuit Breakers or Retries? (Part 1)",
|
||
"url": "/designing-resilient-systems-part-1",
|
||
"tags": ["Resiliency","Circuit Breakers"],
|
||
"category": null,
|
||
"date": "21 Dec 2018",
|
||
"excerpt": "Grab designs fault-tolerant systems that can withstand failures allowing us to continuously provide our consumers with the many services they expect from us."
|
||
},
|
||
|
||
{
|
||
"title": "Orchestrating Chaos Using Grab's Experimentation Platform",
|
||
"url": "/chaos-engineering",
|
||
"tags": ["Chaos Engineering","Resiliency","Microservice"],
|
||
"category": null,
|
||
"date": "23 Nov 2018",
|
||
"excerpt": "At Grab, we practice chaos engineering by intentionally introducing failures in a service or component in the overall business flow. But the failed’ service ..."
|
||
},
|
||
|
||
{
|
||
"title": "Reliable and Scalable Feature Toggles and A/B Testing SDK at Grab",
|
||
"url": "/feature-toggles-ab-testing",
|
||
"tags": ["Experiment","Backend","Frontend","Feature Toggle","A/B Testing"],
|
||
"category": null,
|
||
"date": "2 Nov 2018",
|
||
"excerpt": "Grab’s feature toggle SDK provides a dynamic feature toggle capability to our engineering, data, product, and even business teams. Feature toggles also let t..."
|
||
},
|
||
|
||
{
|
||
"title": "Mockers - Overcoming Testing Challenges at Grab",
|
||
"url": "/mockers",
|
||
"tags": ["Backend","Service","Testing"],
|
||
"category": null,
|
||
"date": "18 Sep 2018",
|
||
"excerpt": "Sustaining quality in fast paced development is a challenge. At Grab, we use Mockers - a tool to expand the scope of local box testing. It helps us overcome ..."
|
||
},
|
||
|
||
{
|
||
"title": "Journey of a Tourist via Grab",
|
||
"url": "/journey-tourist-grab",
|
||
"tags": ["Analytics","Data","Data Analytics","Tourism","Tourists"],
|
||
"category": null,
|
||
"date": "11 Sep 2018",
|
||
"excerpt": "Grab's services to tourists are an integral part of connecting tourists to various destinations and attractions. Do tourists travel on Grab to outlandishly f..."
|
||
},
|
||
|
||
{
|
||
"title": "How We Designed the Quotas Microservice to Prevent Resource Abuse",
|
||
"url": "/quotas-service",
|
||
"tags": ["Quota","Backend","Service"],
|
||
"category": null,
|
||
"date": "10 Aug 2018",
|
||
"excerpt": "Reliable, scalable, and high performing solutions for common system level issues are essential for microservice success, and there is a Grab-wide initiative ..."
|
||
},
|
||
|
||
{
|
||
"title": "Grab Senior Data Scientist Liuqin Yang Wins Beale-Orchard-Hays Prize",
|
||
"url": "/boh-prize",
|
||
"tags": ["Data Science","BOH"],
|
||
"category": null,
|
||
"date": "20 Jul 2018",
|
||
"excerpt": "Grab Senior Data Scientist Dr. Liuqin Yang wins the 2018 Beale-Orchard-Hays Prize, the highest honor in Computational Mathematical Optimization. He has been ..."
|
||
},
|
||
|
||
{
|
||
"title": "Building Grab’s Experimentation Platform",
|
||
"url": "/building-grab-s-experimentation-platform",
|
||
"tags": ["Experiment","Backend","Frontend"],
|
||
"category": null,
|
||
"date": "13 Jul 2018",
|
||
"excerpt": "At Grab, we continuously strive to improve the user experience of our app for both our passengers and driver-partners. To do that, we’re constantly experimen..."
|
||
},
|
||
|
||
{
|
||
"title": "Introducing Grab-Kit: Distributed Service Design at Grab",
|
||
"url": "/introducing-grab-kit",
|
||
"tags": ["Backend","Engineering","Golang"],
|
||
"category": null,
|
||
"date": "8 Jun 2018",
|
||
"excerpt": "As we evolved from a single monolithic application to a microservices-based architecture, we were faced with a new challenge. How do we support exponential g..."
|
||
},
|
||
|
||
{
|
||
"title": "How Grab Experimented with Chat to Drive Down Booking Cancellations",
|
||
"url": "/experiment-chat-booking-cancellations",
|
||
"tags": ["Chat","Booking","Experiment"],
|
||
"category": null,
|
||
"date": "1 Mar 2018",
|
||
"excerpt": "At Grab, we consistently strive to build a platform that delivers excellent user experience to both our passengers and driver-partners. A major degradation t..."
|
||
},
|
||
|
||
{
|
||
"title": "Deep Dive into Database Timeouts in Rails",
|
||
"url": "/deep-dive-into-database-timeouts-in-rails",
|
||
"tags": ["Backend","Database","Distributed Systems","Ruby","Ruby on Rails"],
|
||
"category": null,
|
||
"date": "29 Jan 2018",
|
||
"excerpt": "Disaster strikes when you do not configure timeout values properly. In this post, we dive into the details of how timeouts work with Ruby on Rails and Databa..."
|
||
},
|
||
|
||
{
|
||
"title": "Dealing with the Meltdown Patch at Grab",
|
||
"url": "/dealing-with-the-meltdown-patch-at-grab",
|
||
"tags": ["AWS","Meltdown"],
|
||
"category": null,
|
||
"date": "7 Jan 2018",
|
||
"excerpt": "The meltdown attack reported recently had far reaching implications in terms of security as well as performance. This post is a quick rundown of what perform..."
|
||
},
|
||
|
||
{
|
||
"title": "GrabShare at the Intelligent Transportation Engineering Conference",
|
||
"url": "/grabshare-at-the-intelligent-transportation-engineering-conference",
|
||
"tags": ["Data Science","GrabShare"],
|
||
"category": null,
|
||
"date": "13 Dec 2017",
|
||
"excerpt": "We're excited to share the publication of our paper GrabShare: The Construction of a Realtime Ridesharing Service, which was Grab's contribution to the Intel..."
|
||
},
|
||
|
||
{
|
||
"title": "Grabbing Growth: A Growth Hacking Story",
|
||
"url": "/grabbing-growth-a-growth-hacking-story",
|
||
"tags": ["Growth Hacking"],
|
||
"category": null,
|
||
"date": "8 Dec 2017",
|
||
"excerpt": "Disrupt or be disrupted - that was exactly the spirit in which the Growth Hacking team was created this year. This was a deliberate decision to nurture our s..."
|
||
},
|
||
|
||
{
|
||
"title": "The Data and Science Behind GrabShare Part I: Verifying Potential and Developing the Algorithm",
|
||
"url": "/the-data-and-science-behind-grabshare-part-i",
|
||
"tags": ["Data Science","GrabShare"],
|
||
"category": null,
|
||
"date": "20 Oct 2017",
|
||
"excerpt": "Launching GrabShare was no easy feat. After reviewing the academic literature, we decided to take a different approach and build a new matching algorithm fro..."
|
||
},
|
||
|
||
{
|
||
"title": "The Art of Hiring Good Engineers",
|
||
"url": "/the-art-of-hiring-good-engineers",
|
||
"tags": ["Hiring"],
|
||
"category": null,
|
||
"date": "4 Oct 2017",
|
||
"excerpt": "Hiring the first five good engineers in your team requires a different approach to hiring the first twenty good engineers. The approach to designing this pro..."
|
||
},
|
||
|
||
{
|
||
"title": "Migrating Existing Datastores",
|
||
"url": "/migrating-existing-datastores",
|
||
"tags": ["Backend","Redis"],
|
||
"category": null,
|
||
"date": "8 Aug 2017",
|
||
"excerpt": "At Grab we take pride in creating solutions that impact millions of people in Southeast Asia and as they say, with great power comes great responsibility. As..."
|
||
},
|
||
|
||
{
|
||
"title": "So You Need to Hire Good Engineers",
|
||
"url": "/so-you-need-to-hire-good-engineers",
|
||
"tags": ["Hiring"],
|
||
"category": null,
|
||
"date": "24 Jul 2017",
|
||
"excerpt": "If you are in a fast growing tech startup, you're probably actively interviewing and hiring engineers to scale teams. My question to you is, what hiring stra..."
|
||
},
|
||
|
||
{
|
||
"title": "Come and #hackallthethings at Grab",
|
||
"url": "/come-and-hackallthethings-at-grab",
|
||
"tags": ["Security"],
|
||
"category": null,
|
||
"date": "11 Jul 2017",
|
||
"excerpt": "For the longest time, security has been at the center of our priorities. There’s nothing more self-evident about the trust our millions of driving partners a..."
|
||
},
|
||
|
||
{
|
||
"title": "How We Scaled Our Cache and Got a Good Night's Sleep",
|
||
"url": "/how-we-scaled-our-cache-and-got-a-good-nights-sleep",
|
||
"tags": ["Backend","Redis"],
|
||
"category": null,
|
||
"date": "19 Jun 2017",
|
||
"excerpt": "Caching is arguably the most important and widely used technique in computer industry, from CPU to Facebook live videos, cache is everywhere."
|
||
},
|
||
|
||
{
|
||
"title": "Grab's Front End Study Guide",
|
||
"url": "/grabs-front-end-study-guide",
|
||
"tags": ["Frontend","JavaScript","Web"],
|
||
"category": null,
|
||
"date": "3 Jun 2017",
|
||
"excerpt": "Grab is Southeast Asia (SEA)’s leading transportation platform and our mission is to drive SEA forward, leveraging on the latest technology and the talented ..."
|
||
},
|
||
|
||
{
|
||
"title": "DNS Resolution in Go and Cgo",
|
||
"url": "/dns-resolution-in-go-and-cgo",
|
||
"tags": ["Golang","Networking"],
|
||
"category": null,
|
||
"date": "24 May 2017",
|
||
"excerpt": "This article is part two of a two-part series. In this article, we will talk about RFC 6724 (3484), how DNS resolution works in Go and Cgo, and finally expla..."
|
||
},
|
||
|
||
{
|
||
"title": "Driving Southeast Asia Forward with AWS",
|
||
"url": "/driving-southeast-asia-forward-with-aws",
|
||
"tags": ["AWS"],
|
||
"category": null,
|
||
"date": "21 May 2017",
|
||
"excerpt": "My name is Arul Kumaravel, VP of Engineering at Grab. Grab's mission is to drive Southeast Asia (SEA) forwards. Today I would like to share with you how AWS ..."
|
||
},
|
||
|
||
{
|
||
"title": "How to Go from a Quick Idea to an Essential Feature in Four Steps",
|
||
"url": "/how-to-go-from-a-quick-idea-to-an-essential-feature-in-four-steps",
|
||
"tags": ["Data Science","Product Management"],
|
||
"category": ["Data Science","Product"],
|
||
"date": "16 May 2017",
|
||
"excerpt": "How do you work within a startup team and build a quick idea into a key feature for an app that impacts millions of people? It's one of those things that is ..."
|
||
},
|
||
|
||
{
|
||
"title": "Troubleshooting Unusual AWS ELB 5XX Error",
|
||
"url": "/troubleshooting-unusual-aws-elb-5xx-error",
|
||
"tags": ["AWS","Networking"],
|
||
"category": null,
|
||
"date": "10 May 2017",
|
||
"excerpt": "This article is part one of a two-part series. In this article we explain the ELB 5XX errors which we experience without an apparent reason. We walk you thro..."
|
||
},
|
||
|
||
{
|
||
"title": "Scaling Like a Boss with Presto",
|
||
"url": "/scaling-like-a-boss-with-presto",
|
||
"tags": ["Analytics","AWS","Data","Storage"],
|
||
"category": null,
|
||
"date": "1 May 2017",
|
||
"excerpt": "A year ago, the data volumes at Grab were much lower than the volume we currently use for data-driven analytics. We had a simple and robust infrastructure in..."
|
||
},
|
||
|
||
{
|
||
"title": "Deep Dive into iOS Automation at Grab - Continuous Delivery",
|
||
"url": "/deep-dive-into-ios-automation-at-grab-continuous-delivery",
|
||
"tags": ["Continuous Delivery","iOS","Mobile","Swift"],
|
||
"category": null,
|
||
"date": "23 Apr 2017",
|
||
"excerpt": "This is the second part of our series \"Deep Dive into iOS Automation at Grab\", where we will cover how we manage continuous delivery. As a common solution to..."
|
||
},
|
||
|
||
{
|
||
"title": "Deep Dive into iOS Automation at Grab - Integration Testing",
|
||
"url": "/deep-dive-into-ios-automation-at-grab-integration-testing",
|
||
"tags": ["Continuous Integration","iOS","Mobile","Testing"],
|
||
"category": null,
|
||
"date": "18 Apr 2017",
|
||
"excerpt": "This is the first part of our series \"Deep Dive Into iOS Automation At Grab\", where we will cover testing automation in the iOS team. Over the past two years..."
|
||
},
|
||
|
||
{
|
||
"title": "A Key Expired in Redis, You Won't Believe What Happened Next",
|
||
"url": "/a-key-expired-in-redis-you-wont-believe-what-happened-next",
|
||
"tags": ["Backend","Redis"],
|
||
"category": null,
|
||
"date": "27 Mar 2017",
|
||
"excerpt": "One of Grab's more popular caching solutions is Redis (often in the flavour of the misleadingly named ElastiCache), and for most cases, it works. Except for ..."
|
||
},
|
||
|
||
{
|
||
"title": "How Grab Hires Engineers in Singapore",
|
||
"url": "/how-grab-hires-engineers-in-singapore",
|
||
"tags": ["Hiring"],
|
||
"category": null,
|
||
"date": "16 Feb 2017",
|
||
"excerpt": "Working at Grab will be the “most challenging yet rewarding opportunity” any employee will ever encounter."
|
||
},
|
||
|
||
{
|
||
"title": "Battling with Tech Giants for the World's Best Talent",
|
||
"url": "/battling-with-tech-giants-for-the-worlds-best-talent",
|
||
"tags": ["Hiring"],
|
||
"category": null,
|
||
"date": "18 Jan 2017",
|
||
"excerpt": "Grab steadily attracts a diverse set of engineers from around the world in its three R&D centres: Singapore, Seattle, and Beijing. Right now, half of Gra..."
|
||
},
|
||
|
||
{
|
||
"title": "This Rocket Ain't Stopping - Achieving Zero Downtime for Rails to Golang API Migration",
|
||
"url": "/zero-downtime-migration",
|
||
"tags": ["AWS","Golang","Ruby"],
|
||
"category": null,
|
||
"date": "18 Oct 2016",
|
||
"excerpt": "Grab has been transitioning from a Rails + NodeJS stack to a full Golang Service Oriented Architecture. To contribute to a single common code base, we wanted..."
|
||
},
|
||
|
||
{
|
||
"title": "Grab Vietnam Careers Week",
|
||
"url": "/grab-vietnam-careers-week",
|
||
"tags": ["Hiring"],
|
||
"category": null,
|
||
"date": "14 Oct 2016",
|
||
"excerpt": "Grab is organising our first ever Grab Vietnam Careers Week in Ho Chi Minh City, Vietnam, from 22 to 26 October 2016. We are eager to have more engineers joi..."
|
||
},
|
||
|
||
{
|
||
"title": "GrabPay Wins Best Fraud Prevention Innovation at the Florin Awards",
|
||
"url": "/grabpay-wins-best-fraud-prevention-innovation-at-the-florin-awards",
|
||
"tags": ["User Trust"],
|
||
"category": null,
|
||
"date": "12 Oct 2016",
|
||
"excerpt": "I am honoured to receive the Best Fraud Prevention Innovation (Community Votes) Award at the 2016 Florin Awards on behalf of Grab. For those of you who voted..."
|
||
},
|
||
|
||
{
|
||
"title": "Round-robin in Distributed Systems",
|
||
"url": "/round-robin-in-distributed-systems",
|
||
"tags": ["Backend","Data","Distributed Systems","ELB","Golang"],
|
||
"category": null,
|
||
"date": "27 Sep 2016",
|
||
"excerpt": "While working on Grab's Common Data Service (CDS), there was the need to implement client side load balancing between CDS clients and servers. However, I kep..."
|
||
},
|
||
|
||
{
|
||
"title": "Why Test the Design with Only 5 Users",
|
||
"url": "/why-test-the-design-with-only-5-users",
|
||
"tags": ["User Research","UX"],
|
||
"category": null,
|
||
"date": "26 Aug 2016",
|
||
"excerpt": "The reasoning behind small sample sizes in qualitative usability research."
|
||
},
|
||
|
||
{
|
||
"title": "Programmers Beware - UX is Not Just for Designers",
|
||
"url": "/programmers-beware-ux-is-not-just-for-designers",
|
||
"tags": ["API","UX"],
|
||
"category": null,
|
||
"date": "5 Jul 2016",
|
||
"excerpt": "Perhaps one of the biggest missed opportunities in Tech in recent history is UX.Somehow, UX became the domain of Product Designers and User Interface Designe..."
|
||
},
|
||
|
||
{
|
||
"title": "Grab You Some Post-Mortem Reports",
|
||
"url": "/grab-you-some-post-mortem-reports",
|
||
"tags": ["Post Mortem"],
|
||
"category": null,
|
||
"date": "4 Feb 2016",
|
||
"excerpt": "Grab adopts a Service-Oriented Architecture (SOA) to rapidly develop and deploy new feature services. One of the drawbacks of such a design is that team memb..."
|
||
},
|
||
|
||
{
|
||
"title": "The Curious Case of the Phantom Instance",
|
||
"url": "/curious-case-of-the-phantom-instance",
|
||
"tags": ["AWS"],
|
||
"category": null,
|
||
"date": "28 Dec 2015",
|
||
"excerpt": "Here at the Grab Engineering team, we have built our entire backend stack on top of Amazon Web Services (AWS). Over time, it was inevitable that some habits ..."
|
||
}
|
||
|
||
];
|
||
</script>
|
||
|
||
|
||
<div class="blog-search-modal" id="blog-search-modal" role="dialog" aria-modal="true" aria-label="Search articles" hidden>
|
||
<div class="blog-search-modal-backdrop" data-blog-search-close></div>
|
||
<div class="blog-search-dialog" id="blog-search-dialog" role="document">
|
||
<form action="/search.html" role="search" class="blog-search-form" id="blog-search-form">
|
||
<svg class="blog-search-input-icon" viewBox="0 0 24 24" fill="none" stroke="currentColor" stroke-width="2" stroke-linecap="round" stroke-linejoin="round" aria-hidden="true">
|
||
<circle cx="11" cy="11" r="7"></circle>
|
||
<line x1="21" y1="21" x2="16.65" y2="16.65"></line>
|
||
</svg>
|
||
<input
|
||
type="search"
|
||
name="q"
|
||
id="blog-search-input"
|
||
class="blog-search-input"
|
||
placeholder="Search articles..."
|
||
autocomplete="off"
|
||
aria-autocomplete="list"
|
||
aria-controls="blog-search-results"
|
||
aria-expanded="false"
|
||
aria-label="Search articles"
|
||
>
|
||
<kbd class="blog-search-kbd">Esc</kbd>
|
||
</form>
|
||
<div id="blog-search-results" class="blog-search-results" role="listbox" aria-label="Search results"></div>
|
||
<div class="blog-search-footer">
|
||
<span class="blog-search-footer-hint">Enter opens top result · Esc closes</span>
|
||
<a href="/search.html" class="blog-search-full-link" id="blog-search-full-link" hidden>Search all articles</a>
|
||
</div>
|
||
</div>
|
||
</div>
|
||
|
||
<script src="/js/blog-search.js" defer></script>
|
||
|
||
<!-- Google Tag Manager -->
|
||
<script>
|
||
(function (w, d, s, l, i) {
|
||
w[l] = w[l] || [];
|
||
w[l].push({
|
||
'gtm.start': new Date().getTime(),
|
||
event: 'gtm.js'
|
||
});
|
||
var f = d.getElementsByTagName(s)[0],
|
||
j = d.createElement(s),
|
||
dl = l != 'dataLayer' ? '&l=' + l : '';
|
||
j.async = true;
|
||
j.src =
|
||
'https://www.googletagmanager.com/gtm.js?id=' + i + dl;
|
||
f.parentNode.insertBefore(j, f);
|
||
})(window, document, 'script', 'dataLayer', 'GTM-T3CT72T');
|
||
</script>
|
||
<!-- End Google Tag Manager -->
|
||
|
||
<!-- Old script
|
||
<script>
|
||
(function(i,s,o,g,r,a,m){i['GoogleAnalyticsObject']=r;i[r]=i[r]||function(){
|
||
(i[r].q=i[r].q||[]).push(arguments)},i[r].l=1*new Date();a=s.createElement(o),
|
||
m=s.getElementsByTagName(o)[0];a.async=1;a.src=g;m.parentNode.insertBefore(a,m)
|
||
})(window,document,'script','https://www.google-analytics.com/analytics.js','ga');
|
||
ga('create', 'GTM-T3CT72T', 'auto');
|
||
ga('send', 'pageview');
|
||
</script> -->
|
||
<!-- End of olf script -->
|
||
|
||
|
||
</body>
|
||
</html>
|