146 lines
145 KiB
HTML
146 lines
145 KiB
HTML
<!DOCTYPE html><html lang="en"><head><meta charSet="utf-8"/><meta name="viewport" content="width=device-width, initial-scale=1, maximum-scale=1, user-scalable=yes"/><link rel="preload" as="image" href="/images/logo.png"/><link rel="preload" as="image" href="/images/linked-in.svg"/><link rel="preload" as="image" href="/images/x.svg"/><link rel="preload" as="image" href="/images/yt.svg"/><link rel="preload" as="image" href="/images/insta.svg"/><link rel="preload" as="image" href="/images/f.svg"/><link rel="preload" as="image" href="/images/feed.svg"/><link rel="preload" as="image" href="https://images.ctfassets.net/p762jor363g1/attachment_065167b86d3cf061473c42f392138893/f28a6eb4a75e4f6778409d68d096e714/attachment_065167b86d3cf061473c42f392138893.png"/><link rel="preload" as="image" href="/images/x-blue.svg"/><link rel="preload" as="image" href="/images/f-blue.svg"/><link rel="stylesheet" href="/_next/static/chunks/0sxc27dsa.k6-.css" data-precedence="next"/><link rel="preload" as="script" fetchPriority="low" href="/_next/static/chunks/0zaia6i55l01..js"/><script src="/_next/static/chunks/0pqt~8bl3ukh4.js" async=""></script><script src="/_next/static/chunks/0m_p1bxtorv5i.js" async=""></script><script src="/_next/static/chunks/02g3221oh~3le.js" async=""></script><script src="/_next/static/chunks/turbopack-0~dymj8._h2gc.js" async=""></script><script src="/_next/static/chunks/0a_f9~n73ra-c.js" async=""></script><script src="/_next/static/chunks/0d3shmwh5_nmn.js" async=""></script><script src="/_next/static/chunks/08c8uq~xsq31o.js" async=""></script><script src="/_next/static/chunks/098votwdj80t9.js" async=""></script><script src="/_next/static/chunks/0wni0lk_-haja.js" async=""></script><link rel="preload" as="image" href="/images/linked-in-blue.svg"/><link rel="preload" as="image" href="/images/copy.svg"/><link rel="preload" as="image" href="https://images.ctfassets.net/p762jor363g1/7MYEdTbczd11VcDP5GMLrp/755c9f5d9d2bc9362c1b3c5baf6f6df8/image3.png"/><link rel="preload" as="image" href="https://images.ctfassets.net/p762jor363g1/2Bz9zncYu2jdd4ZeCYIa0w/dc9a472c7cb372e9be01de06984c0009/Option_1__2_.png"/><link rel="preload" as="image" href="https://images.ctfassets.net/p762jor363g1/69pB35ICFY1sL97klnboyf/aed20373e05048d11cdd12de739b4524/image1.png"/><link rel="preload" as="image" href="https://images.ctfassets.net/p762jor363g1/7l2eUSuEW1R60JzrWTVCTa/c29b1e8011108d7c79a3b4d4d6f6237f/IMG_5977.png"/><title>Spotify’s Event Delivery – The Road to the Cloud (Part II) | Spotify Engineering</title><meta name="description" content="Whenever a user performs an action in the Spotify client—such as listening to a song or searching for an artist—a small piece of information, an event, is sent to our servers. Event delivery, the process of making sure that all events gets transported safely from clients all over the world to our central processing system, is an interesting problem. In this series of blog posts, we are going to look at some of the work we have done in this area. More specifically, we are going to look at the architecture of our new event delivery system, and tell you why we choose to base our new system on Google Cloud managed services."/><meta name="application-name" content="Spotify Engineering"/><meta name="author" content="Igor Maravić"/><meta name="robots" content="index, follow, max-image-preview:large, max-snippet:-1, max-video-preview:-1"/><meta name="twitter:data1" content="Igor Maravić"/><meta name="twitter:data2" content="13 minutes"/><meta name="twitter:label1" content="Written by"/><meta name="twitter:label2" content="Est. reading time"/><meta property="og:title" content="Spotify’s Event Delivery – The Road to the Cloud (Part II) | Spotify Engineering"/><meta property="og:description" content="Whenever a user performs an action in the Spotify client—such as listening to a song or searching for an artist—a small piece of information, an event, is sent to our servers. Event delivery, the process of making sure that all events gets transported safely from clients all over the world to our central processing system, is an interesting problem. In this series of blog posts, we are going to look at some of the work we have done in this area. More specifically, we are going to look at the architecture of our new event delivery system, and tell you why we choose to base our new system on Google Cloud managed services."/><meta property="og:site_name" content="Spotify Engineering"/><meta property="og:image" content="https://images.ctfassets.net/p762jor363g1/attachment_065167b86d3cf061473c42f392138893/f28a6eb4a75e4f6778409d68d096e714/attachment_065167b86d3cf061473c42f392138893.png"/><meta property="og:image:type" content="image/jpeg"/><meta property="og:image:width" content="1200"/><meta property="og:image:height" content="630"/><meta property="og:image:alt" content="featured image"/><meta property="og:type" content="article"/><meta name="twitter:card" content="summary_large_image"/><meta name="twitter:creator" content="Igor Maravić"/><meta name="twitter:title" content="Spotify’s Event Delivery – The Road to the Cloud (Part II) | Spotify Engineering"/><meta name="twitter:description" content="Spotify’s Event Delivery – The Road to the Cloud (Part II) | Spotify Engineering"/><meta name="twitter:image" content="https://images.ctfassets.net/p762jor363g1/attachment_065167b86d3cf061473c42f392138893/f28a6eb4a75e4f6778409d68d096e714/attachment_065167b86d3cf061473c42f392138893.png"/><meta name="twitter:image:alt" content="featured image"/><meta name="twitter:image:type" content="image/jpeg"/><meta name="twitter:image:width" content="1200"/><meta name="twitter:image:height" content="630"/><link rel="icon" href="/favicon.ico?favicon.0cd2g1ew_zd4v.ico" sizes="250x250" type="image/x-icon"/><script src="/_next/static/chunks/03~yq9q893hmn.js" noModule=""></script></head><body><div hidden=""><!--$--><!--/$--></div><div class="page-wrapper show-notification"><div class="content-wrapper "><!--$?--><template id="B:0"></template><!--/$--><div class="single-post"><div class="container"><h1 class="single-post__title">Spotify’s Event Delivery – The Road to the Cloud (Part II)</h1></div><div class="container"><div class="single-post__content"><div class="single-post__featured-image"><img alt="Feature Image" width="740" height="404" decoding="async" data-nimg="1" style="color:transparent" src="https://images.ctfassets.net/p762jor363g1/attachment_065167b86d3cf061473c42f392138893/f28a6eb4a75e4f6778409d68d096e714/attachment_065167b86d3cf061473c42f392138893.png"/></div><div class="html-content"><p ><i>Whenever a user performs an action in the Spotify client—such as listening to a song or searching for an artist—a small piece of information, an event, is sent to our servers. Event delivery, the process of making sure that all events gets transported safely from clients all over the world to our central processing system, is an interesting problem. In this series of blog posts, we are going to look at some of the work we have done in this area. More specifically, we are going to look at the architecture of our new event delivery system, and tell you why we choose to base our new system on Google Cloud managed services.</i></p><p ><i>In the</i><a href="https://labs.spotify.com/2016/02/25/spotifys-event-delivery-the-road-to-the-cloud-part-i/" target="_blank" rel="noopener noreferrer"><i>first post</i></a><i>in this series, we talked about how our old event system worked and some of the lessons we learned from operating it. In this second post, we’ll cover the design of our new event delivery system, and why we choose</i><a href="https://cloud.google.com/pubsub/overview" target="_blank" rel="noopener noreferrer"><i>Cloud Pub/Sub</i></a><i>as the transport mechanism for all events.</i></p><h2>Designing Spotify’s New Event Delivery System</h2><p >Our experience in operating and maintaining our old event delivery system provided plenty of input into the design of the new and improved one. The current design was built on top of an even older system operating on hourly rotated log files. This design choice creates complexity such as the propagation and confirmation of end-of-file markers on each event producing machine. Moreover, the current implementation has some failure modes it can not recover from automatically. A piece of software that requires manual intervention for many failure modes running on each machine that produces logs incurs significant operational cost. In the new system we wanted a simpler design on the log producing machines, handing over events to a smaller set of machines close on the network for further processing.</p><p >The missing piece here is an event delivery system or queue that implements reliable transport of events and the persistence of undelivered messages in the queue. With such a system in place, we should be able to have the producer hand off events close to the producer at a very high rate; receive an acknowledgement back with low latency; and have the rest of the system be responsible for the complexity of making sure that submitted events gets passed to HDFS.</p><p >Another change we made was to have each event type have its own channel, or <i>topic,</i> and to convert to a more highly structured format early on in the process. Pushing more of the work onto the producer side means that less time needs to be spent converting the format in the Extract, Transform, Load (ETL) job later in the the process. Having separate topics per event is a key requirement for building efficient real time use cases.</p><p >Since event delivery is something that simply needs to work, we designed the new system in such a way that it could run in parallel with the current system. The interface, both at the producer and the consumer end, matched the current system and we can verify both performance and correctness of the new system rigorously before making the switch.</p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/new-system-design.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_4a878586c2177f60a5ed834a8cc09ace/08338d8cc7b144bfaec449788e834d6f/attachment_4a878586c2177f60a5ed834a8cc09ace.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_4a878586c2177f60a5ed834a8cc09ace"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p ><i>Figure 1. High Level System Design of New Event Delivery System</i></p><p >The four main components of the new system are the File Tailer, the Event Delivery Service, the Reliable Persistent Queue, and the ETL job.</p><p >In this design, the Tailer has a much narrower set of responsibilities than the Producer in our old system. It tails log files looking for new events, and forwards them to the Event Delivery Service. As soon as it gets a confirmation that the event has been received it’s responsibility ends. No more complexity handling end-of-file markers or making sure that data has reached it’s final destination in HDFS.</p><p >The Event Delivery Service accepts events from the Tailer, transform them to their final structured format and forwards them to the Queue. It is built as a RESTful microservice using the <a href="http://spotify.github.io/apollo/" target="_blank" rel="noopener noreferrer">Apollo</a> framework and deployed using the <a href="https://github.com/spotify/helios" target="_blank" rel="noopener noreferrer">Helios</a> orchestration platform, a common design pattern at Spotify. It enables clients to be decoupled from the specifics of a single persistence technology as well as enabling any underlying technology to be switched without service disruption.</p><p >The Queue is the core of our system and, as such, is important for it to scale with growing data volumes. To cope with Hadoop downtime, it needs to reliably store messages for a number of days.</p><p >The ETL job should reliably de-duplicate and export events from the Queue to hourly buckets in HDFS. Before it exposes a bucket to the downstream consumers, it needs to detect with high level of confidence that all data for the bucket has been consumed.</p><p >In Figure 1, you can see a box that says “Service Using API directly”. We have felt for some time that syslog was a less-than-awesome API for event producers. When the new system is in production and the old system has been fully retired, it makes sense to move away from syslog and start providing libraries that services can use to communicate directly with the Event Delivery Service.</p><h2>Choosing a Reliable Persistent Queue</h2><h2>Kafka 0.8</h2><p >Building a Reliable Persistent Queue system that reliably handles Spotify event volumes is a daunting task. Our intention was to leverage existing tools to do the heavy lifting. Since event delivery is the foundation of our data infrastructure, we wanted to play it safe. Our first choice was Kafka 0.8.</p><p >There are many reports that Kafka 0.8 is successfully used by companies of significant size around the world and Kafka 0.8 is a big improvement over the version in use in the current system. In particular, its improved Kafka brokers provide reliable persistent storage. The <a href="https://cwiki.apache.org/confluence/pages/viewpage.action?pageId=27846330" target="_blank" rel="noopener noreferrer">Mirror Maker</a> project introduced mirroring between data centers, and the <a href="http://docs.confluent.io/1.0/camus/docs/intro.html" target="_blank" rel="noopener noreferrer">Camus</a> project can be used for exporting Avro structured events to hourly buckets.</p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/gabo-kafka-system-design.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_7630604752550316b7850f9185b81327/827376fb232fb9157d5da2da9519adf9/attachment_7630604752550316b7850f9185b81327.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_7630604752550316b7850f9185b81327"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p ><i>Figure 2. Event delivery system design in which we use Kafka 0.8 as reliable persistent queue</i></p><p >To prove that event delivery can work as expected on Kafka 0.8, we deployed the test system shown in Figure 2. Embedding a simple Kafka producer in the Event Delivery Service also proved to be easy. To ensure the system worked correctly end-to-end—from Event Delivery Service to HDFS—we embedded various integration tests in our continuous integration and delivery process.</p><p >Sadly, as soon as this system started handling production traffic, it started to fall apart. The only component that proved to be stable was Camus (but since we didn’t push much load through the system, we still don’t know how Camus would perform under stress).</p><p >Mirror Maker gave us the most headaches. We assumed it would reliably mirror data between data centers, but this simply wasn’t the case. It only <a href="https://cwiki.apache.org/confluence/display/KAFKA/KIP-3+-+Mirror+Maker+Enhancement" target="_blank" rel="noopener noreferrer">mirrored data on a best effort basis</a>. If there were issues with the destination cluster, the Mirror Makers would just drop data while reporting to source cluster that data had been successfully mirrored. (Note that this behaviour should be fixed in Kafka 0.9.)</p><p >Mirror Makers occasionally got confused about who was the leader for consumption. The leader would sometimes forget that it was a leader, while the other Mirror Makers from the cluster would happily still try to follow it. When this happened, mirroring between data centers would stop.</p><p >The Kafka Producer also had serious issues with stability. If one or more brokers from a cluster was removed, or even just restarted, it was quite likely that the producer would enter a state from which it couldn’t recover by itself. While it was in such a state it wouldn’t produce any events. The only solution was to restart the whole service.</p><p >Even without solving these issues, we saw that a lot of work would be needed to make the system production-ready. We would need to define deployment strategies for Kafka Brokers and Mirror Makers, do capacity modelling and planning for all system components, and expose performance metrics to <a href="https://spotify.github.io/heroic/#!/index" target="_blank" rel="noopener noreferrer">Spotify’s monitoring system</a>.</p><p >We found ourself at a crossroads. Should we make a significant investment and try to get Kafka work for us? Or should we try something else?</p><h2>Cloud Pub/Sub</h2><p >While we were struggling with Kafka, various other Spotify teams were beginning to experiment with Google Cloud products. A particularly interesting products that was being assessed was <a href="https://cloud.google.com/pubsub/overview" target="_blank" rel="noopener noreferrer">Cloud Pub/Sub</a>. It seemed as if Cloud Pub/Sub might satisfy our basic need for a reliable, persistent queue: it can retain undelivered data for <a href="https://cloud.google.com/pubsub/quotas" target="_blank" rel="noopener noreferrer">7 days</a>, provide reliability through application-level acknowledgements, and has “at-least-once” delivery semantics.</p><p >As well as satisfying our basic needs, Cloud Pub/Sub came with extra goodies:</p><ul><li><p ><b>Global availability—as a global service, Pub/Sub is available in all </b><a href="https://cloud.google.com/compute/docs/zones#available" target="_blank" rel="noopener noreferrer"><b>Google Cloud Zones</b></a><b>; transferring data between our data center wouldn’t be through our normal internet provider but would use underlying Google network.</b></p></li><li><p ><b>A simple REST API</b>—if we didn’t like client library Google provided, we can easily write our own.</p></li><li><p ><b>Operational responsibility was handled by someone else</b>—there was no need to create a capacity model or deployment strategy, or to set up monitoring and alerting.</p></li></ul><p >It all sounded great on paper… but was it too good to be true? The solutions we’d built on Apache Kafka, while not perfect, has served us well. We had lots of experience of the different failure modes, access to the hardware and source code, and could—theoretically—find the root cause of any problem. Moving to a managed service would mean we’d have to trust operations of another organisation. And Cloud Pub/Sub was being advertised as beta software; we were unaware of any organisation other than Google who were using it at our scale.</p><p >With this in mind, we decided that we needed a comprehensive test plan to make absolutely sure that, if we were to go with Cloud Pub/Sub, it would meet <i>all</i>of our requirements.</p><h3>The Producer load test</h3><p >The first item on our agenda was testing Cloud Pub/Sub to see if it could handle the anticipated load. Currently our production load peaks at around 700K events per second. To account for the future growth and possible disaster recovery scenarios, we settled on a test load of 2M events per second. To make it extra hard for Pub/Sub, we wanted to publish this amount of traffic from a single data center, so that all the requests were hitting the Pub/Sub machines in the same zone. We made the assumption that Google plans zones as independent failure domains and that each zone can handle equal amounts of traffic. In theory, if we’re able to push 2M messages to a single zone, we should be able to push <i>number_of_zones</i>* 2M messages across all zones. Our hope was that the system would be able to handle this traffic on both the producing and consuming side for a long time without the service degrading.</p><p >Early on, we hit a stumbling block: the <a href="https://developers.google.com/api-client-library/java/apis/pubsub/v1" target="_blank" rel="noopener noreferrer">Cloud Pub/Sub Java client</a> simply didn’t perform well enough. The client, like many other Google Cloud API clients, is auto-generated from API specifications. That’s good if you want clients that support a wide variety of languages, but not so good if you want a high performance client for a single language.</p><p >Thankfully Pub/Sub has a REST API, so it was easy to write our own <a href="https://github.com/spotify/async-google-pubsub-client" target="_blank" rel="noopener noreferrer">library</a>. We designed the new client with its performance foremost in our minds. To enable better use of resources, we used asynchronous Java. We also added queuing and batching in the client. (This wasn’t first time that we needed to roll our sleeves up and reimplement a Google Cloud API client: in another <a href="https://github.com/spotify/async-datastore-client" target="_blank" rel="noopener noreferrer">project</a> we implemented a high performance client for the Datastore API.)</p><p >With our new client in place, we were ready to start pushing some serious load to Pub/Sub. We used a simple load generator to send mock traffic through the Event Service to Pub/Sub. The generated traffic was routed through two Pub/Sub topics with a ratio of 7:3. To push 2M messages per second, we ran the Event Service on 29 machines.</p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-2xx.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_4bebfa1fba3349ca8523e15fe0a2a4cd/5543229ae361cb0a8fc5e80c673a7fc5/attachment_4bebfa1fba3349ca8523e15fe0a2a4cd.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_4bebfa1fba3349ca8523e15fe0a2a4cd"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p ><i>Figure 3. Number of successful requests per second to Pub/Sub from all data centers</i></p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-5xx.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_6baaf6ce972cf5c497ef1b3c924699be/c21e0e21da1a2c0fe6a12b58c8524cdb/attachment_6baaf6ce972cf5c497ef1b3c924699be.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_6baaf6ce972cf5c497ef1b3c924699be"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p ><i>Figure 4. Number of failed requests per second to Pub/Sub from all data centers</i></p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/net_trafiic.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_37b2dba282bd7fb0a2da406c66453bd6/f352dc0a31b81881ae4bd7b318c0f952/attachment_37b2dba282bd7fb0a2da406c66453bd6.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_37b2dba282bd7fb0a2da406c66453bd6"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p ><i>Figure 5. Incoming and outgoing network traffic for Event Service machines in bps</i></p><p >Pub/Sub passed the test with flying colours. We published 2M messages without any service degradation and received almost no server errors from the Pub/Sub backend. Enabling batching and compression on the Event Service machines resulted in ~1Gbps of network traffic towards Pub/Sub.</p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_9efc288a516f91395d3dd62fb53b526d/5eb9ecea44b4a5d1805dfcbd1f41dc6a/attachment_9efc288a516f91395d3dd62fb53b526d.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_9efc288a516f91395d3dd62fb53b526d"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p ><i>Figure 6. Google Cloud Monitoring graph for total published messages to Pub/Sub</i></p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published-per-topic.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_7fb2c75f899c8879887e4691fe5de179/0a923e9f7542f09b60d3f97e81a3acfe/attachment_7fb2c75f899c8879887e4691fe5de179.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_7fb2c75f899c8879887e4691fe5de179"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p >Figure 7. Google Cloud Monitoring graph for per topic published messages to Pub/Sub</p><p >A useful side-effect of our test was that we could compare our internal metrics with the metrics exposed by Google. As it can been seen by looking Figure 3 and Figure 6, the graphs match perfectly.</p><h3>The Consumer stability test</h3><p >Our second major test focused on consumption. Over a period of 5 days, we measured the end-to-end latency of the system under heavy load. For the duration of the test we published, on average, around 800K messages per second. To mimic real world load variations, the publishing rate varied according to the time of day. To verify that we could use multiple topics concurrently, all data was published to two topics with ratio 7:3.</p><p >A slightly surprising behaviour of Cloud Pub/Sub is that subscriptions need to be created before messages can be persisted: until the subscription exists, no data is retained. Every subscription stores data independently and there is no limit to how many consumers a subscription can have. Consumers are coordinated on the server side, and the server is responsible for fairly allocating the messages to all the consumers that request data. This is a very different to Kafka: in Kafka, data is retained per created topic and the number of Kafka consumers per topic is limited by the number of topic partitions.</p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/long-term-load-test.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7/743803c106c9521d7fb3e0d53dc292b7/attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p ><i>Figure 8. The consumption test dashboard</i></p><p >In our test, we created a subscription, then one hour later we started to consume data. We consumed data in batches of 1000 messages. Since we didn’t try to push consumption as high as we could have, we only consumed events at a slightly higher rate than we were sending at peak. It took about 8 hours to catch up. Once we caught up, consumers kept consuming at a rate that matched the publishing rate.</p><p >The median end-to-end latency we measured during the test period—including backlog recovery—was around 20 seconds. We did not observe any lost messages whatsoever during the test period.</p><h2>Decision</h2><p >Based on these tests, we felt confident that Cloud Pub/Sub was the right choice for us. Latency was low and consistent, and the only capacity limitations we encountered was the one explicitly set by the available quota. In short, choosing Cloud Pub/Sub rather than Kafka 0.8 for our new event delivery platform was an obvious choice.</p>
|
||
<figure class="figure-image">
|
||
|
||
<a href="https://storage.googleapis.com/production-eng/1/2016/03/gabo-system-design-2x.png" target="_blank" rel="noopener noreferrer">
|
||
|
||
<img
|
||
src="//images.ctfassets.net/p762jor363g1/attachment_b78acd55ce53457d80eaf6cb49c7350e/b109c47ca3ed9476a365baef5383796c/attachment_b78acd55ce53457d80eaf6cb49c7350e.png"
|
||
style="width: 100%; max-width: 100%; height: auto; "
|
||
class="blog-image"
|
||
alt="attachment_b78acd55ce53457d80eaf6cb49c7350e"
|
||
/>
|
||
|
||
</a>
|
||
|
||
|
||
</figure>
|
||
<p ><i>Figure 9. Event delivery system design in which we use Cloud Pub/Sub as reliable persistent queue</i></p><h2>Next step</h2><p >After events are safely persisted in Pub/Sub it’s time to export them to HDFS. To fully leverage the Google Cloud offering we decided to give a chance to Dataflow.</p><p >In the last blog post from this series we’re going to go through our plan on leveraging Dataflow for this job. Stay tuned.</p></div><div class="social"><p>SHARE THIS ARTICLE</p><ul class="social__links"><li><a href="https://twitter.com/intent/tweet?url=https%3A%2F%2Fengineering.atspotify.com%2F2016%2F3%2Fspotifys-event-delivery-the-road-to-the-cloud-part-ii&text=Spotify%E2%80%99s%20Event%20Delivery%20%E2%80%93%20The%20Road%20to%20the%20Cloud%20(Part%20II)" target="_blank"><img alt="x" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/x-blue.svg"/></a></li><li><a href="https://www.facebook.com/sharer/sharer.php?u=https%3A%2F%2Fengineering.atspotify.com%2F2016%2F3%2Fspotifys-event-delivery-the-road-to-the-cloud-part-ii" target="_blank" rel="noopener noreferrer"><img alt="fb" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/f-blue.svg"/></a></li><li><a href="https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fengineering.atspotify.com%2F2016%2F3%2Fspotifys-event-delivery-the-road-to-the-cloud-part-ii" target="_blank" rel="noopener noreferrer"><img alt="linkedin" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/linked-in-blue.svg"/></a></li><li><a><img alt="copy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/copy.svg"/></a></li></ul></div></div><aside class="sidebar"><div class="date">Mar 3, 2016</div><div class="published">Published by Igor Maravić</div><div class="tag-wrapper"><a class="tag fill" href="/category/data">Data</a><a class="tag fill" href="/category/data-science">Data Science</a><a class="tag fill" href="/category/infrastructure">Infrastructure</a></div><div class="social"><p>SHARE THIS ARTICLE</p><ul class="social__links"><li><a href="https://twitter.com/intent/tweet?url=https%3A%2F%2Fengineering.atspotify.com%2F2016%2F3%2Fspotifys-event-delivery-the-road-to-the-cloud-part-ii&text=Spotify%E2%80%99s%20Event%20Delivery%20%E2%80%93%20The%20Road%20to%20the%20Cloud%20(Part%20II)" target="_blank"><img alt="x" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/x-blue.svg"/></a></li><li><a href="https://www.facebook.com/sharer/sharer.php?u=https%3A%2F%2Fengineering.atspotify.com%2F2016%2F3%2Fspotifys-event-delivery-the-road-to-the-cloud-part-ii" target="_blank" rel="noopener noreferrer"><img alt="fb" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/f-blue.svg"/></a></li><li><a href="https://www.linkedin.com/sharing/share-offsite/?url=https%3A%2F%2Fengineering.atspotify.com%2F2016%2F3%2Fspotifys-event-delivery-the-road-to-the-cloud-part-ii" target="_blank" rel="noopener noreferrer"><img alt="linkedin" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/linked-in-blue.svg"/></a></li><li><a><img alt="copy" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/copy.svg"/></a></li></ul></div></aside></div></div><div class="related-posts"><div class="container"><h3>Related articles</h3></div><div class="container"><div class="cards"><div class="post-card related"><div class="post-card__image"><a href="/2026/9/why-spotify-is-not-using-bayesian-a-b-testing"><img alt="sticky" width="400" height="197" decoding="async" data-nimg="1" style="color:transparent" src="https://images.ctfassets.net/p762jor363g1/7MYEdTbczd11VcDP5GMLrp/755c9f5d9d2bc9362c1b3c5baf6f6df8/image3.png"/></a></div><div class="post-card__content"><div class="top-content"><div class="post-card__date">Sep 8, 2026</div><a href="/2026/9/why-spotify-is-not-using-bayesian-a-b-testing"><h4 class="post-card__title">Why Spotify Is Not Using Bayesian A/B Testing</h4></a><div class="post-card__description">Clearing the confusion about what Bayesian A/B testing is.</div><div class="post-card__published"></div></div><div class="post-card__tags"></div></div></div><div class="post-card related"><div class="post-card__image"><a href="/2026/8/when-can-llms-replace-humans-in-a-b-tests"><img alt="sticky" width="400" height="197" decoding="async" data-nimg="1" style="color:transparent" src="https://images.ctfassets.net/p762jor363g1/2Bz9zncYu2jdd4ZeCYIa0w/dc9a472c7cb372e9be01de06984c0009/Option_1__2_.png"/></a></div><div class="post-card__content"><div class="top-content"><div class="post-card__date">Aug 13, 2026</div><a href="/2026/8/when-can-llms-replace-humans-in-a-b-tests"><h4 class="post-card__title">When Can LLMs Replace Humans in A/B Tests?</h4></a><div class="post-card__description">TL;DR: LLM predictions can stand in for human outcomes in A/B tests, but only by assumption, not by design....</div><div class="post-card__published"></div></div><div class="post-card__tags"></div></div></div><div class="post-card related"><div class="post-card__image"><a href="/2026/7/indexing-the-data-lake-for-online-point-queries"><img alt="sticky" width="400" height="197" decoding="async" data-nimg="1" style="color:transparent" src="https://images.ctfassets.net/p762jor363g1/69pB35ICFY1sL97klnboyf/aed20373e05048d11cdd12de739b4524/image1.png"/></a></div><div class="post-card__content"><div class="top-content"><div class="post-card__date">Jul 27, 2026</div><a href="/2026/7/indexing-the-data-lake-for-online-point-queries"><h4 class="post-card__title">Indexing the Data Lake for Online Point Queries</h4></a><div class="post-card__description">Companies like Spotify need vast quantities of data accessible at low latency for online services and,...</div><div class="post-card__published"></div></div><div class="post-card__tags"></div></div></div><div class="post-card related"><div class="post-card__image"><a href="/2026/6/encoding-your-domain-expert-the-context-layer-behind-spotifys-data-assistant"><img alt="sticky" width="400" height="197" decoding="async" data-nimg="1" style="color:transparent" src="https://images.ctfassets.net/p762jor363g1/7l2eUSuEW1R60JzrWTVCTa/c29b1e8011108d7c79a3b4d4d6f6237f/IMG_5977.png"/></a></div><div class="post-card__content"><div class="top-content"><div class="post-card__date">Jun 10, 2026</div><a href="/2026/6/encoding-your-domain-expert-the-context-layer-behind-spotifys-data-assistant"><h4 class="post-card__title">Encoding Your Domain Expert: The Context Layer Behind Spotify's Data Assistant</h4></a><div class="post-card__description">At Spotify, data problems used to follow a specific pattern. You'd look for the relevant dashboard, there...</div><div class="post-card__published"></div></div><div class="post-card__tags"></div></div></div></div></div></div><!--$--><!--/$--></div><footer class="footer"><div class="container"><section class="subscription"><div class="subscription__left"><h3>Want to stay in the loop?</h3><p>Subscribe to our newsletter to never miss an update!</p><small>By clicking sign up you’ll receive occasional emails from Spotify. You always have the choice to adjust your interest settings or unsubscribe.</small></div><div class="subscription__right"><div id="mc_embed_signup"><form id="mc-embedded-subscribe-form" name="mc-embedded-subscribe-form" class="validate" action="https://spotify.us20.list-manage.com/subscribe/post?u=99aa88d3fba458716d0cd1299&id=28cc680980" method="post" target="_blank"><div id="mc_embed_signup_scroll"><div class="mc-field-group"><input type="email" class="required email" id="mce-EMAIL" placeholder="Enter your best email" name="EMAIL"/><input type="submit" id="mc-embedded-subscribe" class="button button--submit" name="subscribe" value="Subscribe"/></div><div id="mce-responses" class="clear"><div class="response" id="mce-error-response" style="display:none"></div><div class="response" id="mce-success-response" style="display:none"></div></div><div style="left:-5000px;position:absolute" aria-hidden="true"><input type="text" name="b_99aa88d3fba458716d0cd1299_28cc680980"/></div></div></form></div></div></section><section class="links"><div class="links__left"><div class="logo"><a href="/"><img alt="feed" loading="lazy" width="167" height="50" decoding="async" data-nimg="1" style="color:transparent" src="/images/footer-logo.svg"/></a></div><div class="nav-wrapper"><nav class="nav"><ul><li><a target="_blank" rel="noopener noreferrer" href="https://spotify.com/">Spotify.com</a></li><li><a target="_blank" rel="noopener noreferrer" href="https://www.lifeatspotify.com/">Spotify Jobs</a></li><li><a target="_blank" rel="noopener noreferrer" href="https://newsroom.spotify.com/">Newsroom</a></li></ul><ul><li><a target="_blank" rel="noopener noreferrer" href="https://research.atspotify.com/">Spotify R&D Research</a></li><li><a target="_blank" rel="noopener noreferrer" href="https://backstage.spotify.com/">Spotify for Backstage</a></li></ul></nav></div></div><div class="links__right"><ul class="social"><li><a target="_blank" rel="noopener noreferrer" href="https://www.linkedin.com/showcase/spotify-r&d"><img alt="Linked In" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/linked-in.svg"/></a></li><li><a target="_blank" rel="noopener noreferrer" href="https://twitter.com/spotifyeng"><img alt="X" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/x.svg"/></a></li><li><a target="_blank" rel="noopener noreferrer" href="https://www.youtube.com/@SpotifyRnD"><img alt="Youtube" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/yt.svg"/></a></li><li><a target="_blank" rel="noopener noreferrer" href="http://www.instagram.com/lifeatspotify"><img alt="Instagram" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/insta.svg"/></a></li><li><a target="_blank" rel="noopener noreferrer" href="https://www.facebook.com/spotify"><img alt="facebook" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/f.svg"/></a></li><li><a target="_blank" rel="noopener noreferrer" href="/feed"><img alt="feed" width="24" height="24" decoding="async" data-nimg="1" style="color:transparent" src="/images/feed.svg"/></a></li></ul></div></section><section class="copy"><nav class="copy__nav"><ul><li><a target="_blank" rel="noopener noreferrer" href="https://www.spotify.com/us/legal/end-user-agreement/">Legal</a></li><li><a target="_blank" rel="noopener noreferrer" href="https://www.spotify.com/us/legal/privacy-policy/">Privacy</a></li><li><a target="_blank" rel="noopener noreferrer" href="https://www.spotify.com/legal/cookies-policy/">Cookies</a></li><li><a target="_blank" rel="noopener noreferrer" href="https://www.spotify.com/us/legal/privacy-policy/">About Ads</a></li></ul></nav><span>© <!-- -->2026<!-- --> Spotify AB</span></section></div></footer></div><script>requestAnimationFrame(function(){$RT=performance.now()});</script><script src="/_next/static/chunks/0zaia6i55l01..js" id="_R_" async=""></script><div hidden id="S:0"><div class="notification"><div class="content"></div></div><header class="header "><section class="container"><div class="container-left"><a href="/"><img alt="Engineering logo" width="90" height="36" decoding="async" data-nimg="1" style="color:transparent" src="/images/logo.png"/></a><nav class="nav "><ul><li class=""><a href="/">Blog</a></li><li class=""><a href="/podcasts">Podcasts</a></li><li class=""><a href="/opensource">Open Source</a></li><li class=""><a href="/jobs">Jobs</a></li><li class=""><a href="/about">About</a></li></ul></nav></div><div class="container-right"><div class="search "><div class="search__input"><input type="text" id="search-input" placeholder="Search" value=""/></div><div class="search__icon"><img alt="Search" id="search-icon" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" class="mag" style="color:transparent" src="/images/icon-search.svg"/><img alt="Search" loading="lazy" width="24" height="24" decoding="async" data-nimg="1" class="close" style="color:transparent" src="/images/icon-close.svg"/></div><div class="menu-button "><span></span><span></span><span></span></div></div></div></section></header></div><script>$RB=[];$RV=function(a){$RT=performance.now();for(var b=0;b<a.length;b+=2){var c=a[b],e=a[b+1];null!==e.parentNode&&e.parentNode.removeChild(e);var f=c.parentNode;if(f){var g=c.previousSibling,h=0;do{if(c&&8===c.nodeType){var d=c.data;if("/$"===d||"/&"===d)if(0===h)break;else h--;else"$"!==d&&"$?"!==d&&"$~"!==d&&"$!"!==d&&"&"!==d||h++}d=c.nextSibling;f.removeChild(c);c=d}while(c);for(;e.firstChild;)f.insertBefore(e.firstChild,c);g.data="$";g._reactRetry&&requestAnimationFrame(g._reactRetry)}}a.length=0};
|
||
$RC=function(a,b){if(b=document.getElementById(b))(a=document.getElementById(a))?(a.previousSibling.data="$~",$RB.push(a,b),2===$RB.length&&("number"!==typeof $RT?requestAnimationFrame($RV.bind(null,$RB)):(a=performance.now(),setTimeout($RV.bind(null,$RB),2300>a&&2E3<a?2300-a:$RT+300-a)))):b.parentNode.removeChild(b)};$RC("B:0","S:0")</script><script>(self.__next_f=self.__next_f||[]).push([0])</script><script>self.__next_f.push([1,"1:\"$Sreact.fragment\"\n3:I[39756,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"default\"]\n4:I[37457,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"default\"]\n6:I[97367,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"OutletBoundary\"]\n7:\"$Sreact.suspense\"\na:I[97367,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"ViewportBoundary\"]\nc:I[97367,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"MetadataBoundary\"]\ne:I[68027,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"default\",1]\n:HL[\"/_next/static/chunks/0sxc27dsa.k6-.css\",\"style\"]\n"])</script><script>self.__next_f.push([1,"0:{\"P\":null,\"c\":[\"\",\"2016\",\"3\",\"spotifys-event-delivery-the-road-to-the-cloud-part-ii\"],\"q\":\"\",\"i\":false,\"f\":[[[\"\",{\"children\":[[\"year\",\"2016\",\"d\",null],{\"children\":[[\"month\",\"3\",\"d\",null],{\"children\":[[\"slug\",\"spotifys-event-delivery-the-road-to-the-cloud-part-ii\",\"d\",null],{\"children\":[\"__PAGE__\",{}]}]}]}]},\"$undefined\",\"$undefined\",16],[[\"$\",\"$1\",\"c\",{\"children\":[[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/chunks/0sxc27dsa.k6-.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-1\",{\"src\":\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"async\":true,\"nonce\":\"$undefined\"}],[\"$\",\"script\",\"script-2\",{\"src\":\"/_next/static/chunks/08c8uq~xsq31o.js\",\"async\":true,\"nonce\":\"$undefined\"}]],\"$L2\"]}],{\"children\":[[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L3\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L4\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}],{\"children\":[[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L3\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L4\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}],{\"children\":[[\"$\",\"$1\",\"c\",{\"children\":[null,[\"$\",\"$L3\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$undefined\",\"errorStyles\":\"$undefined\",\"errorScripts\":\"$undefined\",\"template\":[\"$\",\"$L4\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":\"$undefined\",\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}],{\"children\":[[\"$\",\"$1\",\"c\",{\"children\":[\"$L5\",[[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/chunks/0wni0lk_-haja.js\",\"async\":true,\"nonce\":\"$undefined\"}]],[\"$\",\"$L6\",null,{\"children\":[\"$\",\"$7\",null,{\"name\":\"Next.MetadataOutlet\",\"children\":\"$@8\"}]}]]}],{},null,false,null]},null,false,\"$@9\"]},null,false,\"$@9\"]},null,false,\"$@9\"]},null,false,null],[\"$\",\"$1\",\"h\",{\"children\":[null,[\"$\",\"$La\",null,{\"children\":\"$Lb\"}],[\"$\",\"div\",null,{\"hidden\":true,\"children\":[\"$\",\"$Lc\",null,{\"children\":[\"$\",\"$7\",null,{\"name\":\"Next.Metadata\",\"children\":\"$Ld\"}]}]}],null]}],false]],\"m\":\"$undefined\",\"G\":[\"$e\",[[\"$\",\"link\",\"0\",{\"rel\":\"stylesheet\",\"href\":\"/_next/static/chunks/0sxc27dsa.k6-.css\",\"precedence\":\"next\",\"crossOrigin\":\"$undefined\",\"nonce\":\"$undefined\"}]]],\"S\":false,\"h\":null,\"s\":\"$undefined\",\"l\":\"$undefined\",\"p\":\"$undefined\",\"d\":\"$undefined\",\"b\":\"z1ONPGug6r-HUxPkOhj0S\"}\n"])</script><script>self.__next_f.push([1,"f:[]\n9:\"$Wf\"\nb:[[\"$\",\"meta\",\"0\",{\"charSet\":\"utf-8\"}],[\"$\",\"meta\",\"1\",{\"name\":\"viewport\",\"content\":\"width=device-width, initial-scale=1, maximum-scale=1, user-scalable=yes\"}]]\n"])</script><script>self.__next_f.push([1,"10:I[79520,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"\"]\n11:I[1013,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"HeaderContextProvider\"]\n12:I[32666,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"PostProvider\"]\n13:I[28044,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"UiTextProvider\"]\n14:I[70292,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"Initializer\"]\n15:I[97341,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"default\"]\n16:I[36676,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"default\"]\n17:I[58298,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\",\"/_next/static/chunks/098votwdj80t9.js\"],\"default\"]\n18:I[22016,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\",\"/_next/static/chunks/0wni0lk_-haja.js\"],\"\"]\n"])</script><script>self.__next_f.push([1,"2:[\"$\",\"html\",null,{\"lang\":\"en\",\"children\":[[\"$\",\"head\",null,{\"children\":[\"$\",\"$L10\",null,{\"id\":\"gtm\",\"dangerouslySetInnerHTML\":{\"__html\":\"(function(w, d, s, l, i) {\\n w[l] = w[l] || [];\\n w[l].push({\\n 'gtm.start': new Date().getTime(),\\n event: 'gtm.js'\\n });\\n var f = d.getElementsByTagName(s)[0],\\n j = d.createElement(s),\\n dl = l != 'dataLayer' ? '\u0026l=' + l : '';\\n j.async = true;\\n j.src =\\n 'https://www.googletagmanager.com/gtm.js?id=' + i + dl;\\n f.parentNode.insertBefore(j, f);\\n })(window, document, 'script', 'dataLayer', 'GTM-W5NPJP7');\"}}]}],[\"$\",\"body\",null,{\"children\":[\"$\",\"div\",null,{\"className\":\"page-wrapper show-notification\",\"children\":[[\"$\",\"$L11\",null,{\"children\":[\"$\",\"$L12\",null,{\"children\":[\"$\",\"$L13\",null,{\"defaultUiText\":{\"homepageBannerText\":{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Read about the magic behind the music \u0026 more. Welcome to our official technology blog.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"}],\"nodeType\":\"document\"},\"podcastTagline\":{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Two original shows from Spotify R\u0026D.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Listen and fill your brain with great tech stories.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"}],\"nodeType\":\"document\"},\"opensourceTagline\":{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"There’s never been a better time to take part in open source at Spotify. We don’t just want to contribute back to the community — we want to lead the way with projects we truly believe in. Jump into one of the projects below and join us.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"}],\"nodeType\":\"document\"},\"jobsTagline\":{\"nodeType\":\"document\",\"data\":{},\"content\":[{\"nodeType\":\"paragraph\",\"data\":{},\"content\":[{\"nodeType\":\"text\",\"value\":\"Complex problems, exceptional colleagues, and a direct impact on the joy of \",\"marks\":[],\"data\":{}},{\"nodeType\":\"text\",\"value\":\"713\",\"marks\":[],\"data\":{}},{\"nodeType\":\"text\",\"value\":\" million people. Our beliefs drive our success.\",\"marks\":[],\"data\":{}}]}]},\"aboutTagline\":{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Spotify Engineering is a diverse community of engineers, developers, researchers and more. Together, we build the infrastructure, features, and experiences you love – and help to shape the future of audio.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"}],\"nodeType\":\"document\"}},\"children\":[\"$\",\"$L14\",null,{\"children\":[[\"$\",\"$7\",null,{\"children\":[[\"$\",\"$L15\",null,{}],[\"$\",\"$L16\",null,{}]]}],[\"$\",\"$L3\",null,{\"parallelRouterKey\":\"children\",\"error\":\"$17\",\"errorStyles\":[],\"errorScripts\":[[\"$\",\"script\",\"script-0\",{\"src\":\"/_next/static/chunks/098votwdj80t9.js\",\"async\":true}]],\"template\":[\"$\",\"$L4\",null,{}],\"templateStyles\":\"$undefined\",\"templateScripts\":\"$undefined\",\"notFound\":[[\"$\",\"div\",null,{\"className\":\"not-found\",\"children\":[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[[\"$\",\"h1\",null,{\"children\":\"404\"}],[\"$\",\"div\",null,{\"className\":\"content\",\"children\":[[\"$\",\"h3\",null,{\"children\":\"OOOOPS... Something went wrong\"}],[\"$\",\"p\",null,{\"children\":\"We couldn't find the page you're looking for.\"}]]}],[\"$\",\"$L18\",null,{\"href\":\"/\",\"className\":\"button\",\"children\":\"Go back home\"}]]}]}],[]],\"forbidden\":\"$undefined\",\"unauthorized\":\"$undefined\"}]]}]}]}]}],[\"$\",\"footer\",null,{\"className\":\"footer\",\"children\":[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[[\"$\",\"section\",null,{\"className\":\"subscription\",\"children\":[[\"$\",\"div\",null,{\"className\":\"subscription__left\",\"children\":[[\"$\",\"h3\",null,{\"children\":\"Want to stay in the loop?\"}],[\"$\",\"p\",null,{\"children\":\"Subscribe to our newsletter to never miss an update!\"}],[\"$\",\"small\",null,{\"children\":\"By clicking sign up you’ll receive occasional emails from Spotify. You always have the choice to adjust your interest settings or unsubscribe.\"}]]}],[\"$\",\"div\",null,{\"className\":\"subscription__right\",\"children\":[\"$\",\"div\",null,{\"id\":\"mc_embed_signup\",\"children\":[\"$\",\"form\",null,{\"action\":\"https://spotify.us20.list-manage.com/subscribe/post?u=99aa88d3fba458716d0cd1299\u0026id=28cc680980\",\"method\":\"post\",\"id\":\"mc-embedded-subscribe-form\",\"name\":\"mc-embedded-subscribe-form\",\"className\":\"validate\",\"target\":\"_blank\",\"children\":\"$L19\"}]}]}]]}],\"$L1a\",\"$L1b\"]}]}]]}]}]]}]\n"])</script><script>self.__next_f.push([1,"1c:I[5500,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\",\"/_next/static/chunks/0wni0lk_-haja.js\"],\"Image\"]\n19:[\"$\",\"div\",null,{\"id\":\"mc_embed_signup_scroll\",\"children\":[[\"$\",\"div\",null,{\"className\":\"mc-field-group\",\"children\":[[\"$\",\"input\",null,{\"type\":\"email\",\"name\":\"EMAIL\",\"className\":\"required email\",\"id\":\"mce-EMAIL\",\"placeholder\":\"Enter your best email\"}],[\"$\",\"input\",null,{\"type\":\"submit\",\"name\":\"subscribe\",\"value\":\"Subscribe\",\"id\":\"mc-embedded-subscribe\",\"className\":\"button button--submit\"}]]}],[\"$\",\"div\",null,{\"id\":\"mce-responses\",\"className\":\"clear\",\"children\":[[\"$\",\"div\",null,{\"className\":\"response\",\"id\":\"mce-error-response\",\"style\":{\"display\":\"none\"}}],[\"$\",\"div\",null,{\"className\":\"response\",\"id\":\"mce-success-response\",\"style\":{\"display\":\"none\"}}]]}],[\"$\",\"div\",null,{\"style\":{\"left\":\"-5000px\",\"position\":\"absolute\"},\"aria-hidden\":\"true\",\"children\":[\"$\",\"input\",null,{\"type\":\"text\",\"name\":\"b_99aa88d3fba458716d0cd1299_28cc680980\"}]}]]}]\n"])</script><script>self.__next_f.push([1,"1a:[\"$\",\"section\",null,{\"className\":\"links\",\"children\":[[\"$\",\"div\",null,{\"className\":\"links__left\",\"children\":[[\"$\",\"div\",null,{\"className\":\"logo\",\"children\":[\"$\",\"$L18\",null,{\"href\":\"/\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"/images/footer-logo.svg\",\"alt\":\"feed\",\"width\":167,\"height\":50}]}]}],[\"$\",\"div\",null,{\"className\":\"nav-wrapper\",\"children\":[\"$\",\"nav\",null,{\"className\":\"nav\",\"children\":[[\"$\",\"ul\",null,{\"children\":[[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://spotify.com/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"Spotify.com\"}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://www.lifeatspotify.com/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"Spotify Jobs\"}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://newsroom.spotify.com/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"Newsroom\"}]}]]}],[\"$\",\"ul\",null,{\"children\":[[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://research.atspotify.com/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"Spotify R\u0026D Research\"}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://backstage.spotify.com/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"Spotify for Backstage\"}]}]]}]]}]}]]}],[\"$\",\"div\",null,{\"className\":\"links__right\",\"children\":[\"$\",\"ul\",null,{\"className\":\"social\",\"children\":[[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://www.linkedin.com/showcase/spotify-r\u0026d\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"/images/linked-in.svg\",\"alt\":\"Linked In\",\"width\":24,\"height\":24,\"priority\":true}]}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://twitter.com/spotifyeng\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"/images/x.svg\",\"alt\":\"X\",\"width\":24,\"height\":24,\"priority\":true}]}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://www.youtube.com/@SpotifyRnD\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"/images/yt.svg\",\"alt\":\"Youtube\",\"width\":24,\"height\":24,\"priority\":true}]}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"http://www.instagram.com/lifeatspotify\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"/images/insta.svg\",\"alt\":\"Instagram\",\"width\":24,\"height\":24,\"priority\":true}]}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://www.facebook.com/spotify\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"/images/f.svg\",\"alt\":\"facebook\",\"width\":24,\"height\":24,\"priority\":true}]}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"/feed\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"/images/feed.svg\",\"alt\":\"feed\",\"width\":24,\"height\":24,\"priority\":true}]}]}]]}]}]]}]\n"])</script><script>self.__next_f.push([1,"1b:[\"$\",\"section\",null,{\"className\":\"copy\",\"children\":[[\"$\",\"nav\",null,{\"className\":\"copy__nav\",\"children\":[\"$\",\"ul\",null,{\"children\":[[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://www.spotify.com/us/legal/end-user-agreement/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"Legal\"}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://www.spotify.com/us/legal/privacy-policy/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"Privacy\"}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://www.spotify.com/legal/cookies-policy/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"Cookies\"}]}],[\"$\",\"li\",null,{\"children\":[\"$\",\"$L18\",null,{\"href\":\"https://www.spotify.com/us/legal/privacy-policy/\",\"target\":\"_blank\",\"rel\":\"noopener noreferrer\",\"children\":\"About Ads\"}]}]]}]}],[\"$\",\"span\",null,{\"children\":[\"© \",2026,\" Spotify AB\"]}]]}]\n"])</script><script>self.__next_f.push([1,"8:null\n"])</script><script>self.__next_f.push([1,"d:[[\"$\",\"title\",\"0\",{\"children\":\"Spotify’s Event Delivery – The Road to the Cloud (Part II) | Spotify Engineering\"}],[\"$\",\"meta\",\"1\",{\"name\":\"description\",\"content\":\"Whenever a user performs an action in the Spotify client—such as listening to a song or searching for an artist—a small piece of information, an event, is sent to our servers. Event delivery, the process of making sure that all events gets transported safely from clients all over the world to our central processing system, is an interesting problem. In this series of blog posts, we are going to look at some of the work we have done in this area. More specifically, we are going to look at the architecture of our new event delivery system, and tell you why we choose to base our new system on Google Cloud managed services.\"}],[\"$\",\"meta\",\"2\",{\"name\":\"application-name\",\"content\":\"Spotify Engineering\"}],[\"$\",\"meta\",\"3\",{\"name\":\"author\",\"content\":\"Igor Maravić\"}],[\"$\",\"meta\",\"4\",{\"name\":\"robots\",\"content\":\"index, follow, max-image-preview:large, max-snippet:-1, max-video-preview:-1\"}],[\"$\",\"meta\",\"5\",{\"name\":\"twitter:data1\",\"content\":\"Igor Maravić\"}],[\"$\",\"meta\",\"6\",{\"name\":\"twitter:data2\",\"content\":\"13 minutes\"}],[\"$\",\"meta\",\"7\",{\"name\":\"twitter:label1\",\"content\":\"Written by\"}],[\"$\",\"meta\",\"8\",{\"name\":\"twitter:label2\",\"content\":\"Est. reading time\"}],[\"$\",\"meta\",\"9\",{\"property\":\"og:title\",\"content\":\"Spotify’s Event Delivery – The Road to the Cloud (Part II) | Spotify Engineering\"}],[\"$\",\"meta\",\"10\",{\"property\":\"og:description\",\"content\":\"Whenever a user performs an action in the Spotify client—such as listening to a song or searching for an artist—a small piece of information, an event, is sent to our servers. Event delivery, the process of making sure that all events gets transported safely from clients all over the world to our central processing system, is an interesting problem. In this series of blog posts, we are going to look at some of the work we have done in this area. More specifically, we are going to look at the architecture of our new event delivery system, and tell you why we choose to base our new system on Google Cloud managed services.\"}],[\"$\",\"meta\",\"11\",{\"property\":\"og:site_name\",\"content\":\"Spotify Engineering\"}],[\"$\",\"meta\",\"12\",{\"property\":\"og:image\",\"content\":\"https://images.ctfassets.net/p762jor363g1/attachment_065167b86d3cf061473c42f392138893/f28a6eb4a75e4f6778409d68d096e714/attachment_065167b86d3cf061473c42f392138893.png\"}],[\"$\",\"meta\",\"13\",{\"property\":\"og:image:type\",\"content\":\"image/jpeg\"}],[\"$\",\"meta\",\"14\",{\"property\":\"og:image:width\",\"content\":\"1200\"}],[\"$\",\"meta\",\"15\",{\"property\":\"og:image:height\",\"content\":\"630\"}],[\"$\",\"meta\",\"16\",{\"property\":\"og:image:alt\",\"content\":\"featured image\"}],[\"$\",\"meta\",\"17\",{\"property\":\"og:type\",\"content\":\"article\"}],[\"$\",\"meta\",\"18\",{\"name\":\"twitter:card\",\"content\":\"summary_large_image\"}],[\"$\",\"meta\",\"19\",{\"name\":\"twitter:creator\",\"content\":\"Igor Maravić\"}],[\"$\",\"meta\",\"20\",{\"name\":\"twitter:title\",\"content\":\"Spotify’s Event Delivery – The Road to the Cloud (Part II) | Spotify Engineering\"}],[\"$\",\"meta\",\"21\",{\"name\":\"twitter:description\",\"content\":\"Spotify’s Event Delivery – The Road to the Cloud (Part II) | Spotify Engineering\"}],[\"$\",\"meta\",\"22\",{\"name\":\"twitter:image\",\"content\":\"https://images.ctfassets.net/p762jor363g1/attachment_065167b86d3cf061473c42f392138893/f28a6eb4a75e4f6778409d68d096e714/attachment_065167b86d3cf061473c42f392138893.png\"}],[\"$\",\"meta\",\"23\",{\"name\":\"twitter:image:alt\",\"content\":\"featured image\"}],[\"$\",\"meta\",\"24\",{\"name\":\"twitter:image:type\",\"content\":\"image/jpeg\"}],[\"$\",\"meta\",\"25\",{\"name\":\"twitter:image:width\",\"content\":\"1200\"}],[\"$\",\"meta\",\"26\",{\"name\":\"twitter:image:height\",\"content\":\"630\"}],[\"$\",\"link\",\"27\",{\"rel\":\"icon\",\"href\":\"/favicon.ico?favicon.0cd2g1ew_zd4v.ico\",\"sizes\":\"250x250\",\"type\":\"image/x-icon\"}],\"$L1d\"]\n"])</script><script>self.__next_f.push([1,"1e:I[27201,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\"],\"IconMark\"]\n1d:[\"$\",\"$L1e\",\"28\",{}]\n"])</script><script>self.__next_f.push([1,"1f:T59ec,"])</script><script>self.__next_f.push([1,"\u003cp \u003e\u003ci\u003eWhenever a user performs an action in the Spotify client—such as listening to a song or searching for an artist—a small piece of information, an event, is sent to our servers. Event delivery, the process of making sure that all events gets transported safely from clients all over the world to our central processing system, is an interesting problem. In this series of blog posts, we are going to look at some of the work we have done in this area. More specifically, we are going to look at the architecture of our new event delivery system, and tell you why we choose to base our new system on Google Cloud managed services.\u003c/i\u003e\u003c/p\u003e\u003cp \u003e\u003ci\u003eIn the\u003c/i\u003e\u003ca href=\"https://labs.spotify.com/2016/02/25/spotifys-event-delivery-the-road-to-the-cloud-part-i/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\u003ci\u003efirst post\u003c/i\u003e\u003c/a\u003e\u003ci\u003ein this series, we talked about how our old event system worked and some of the lessons we learned from operating it. In this second post, we’ll cover the design of our new event delivery system, and why we choose\u003c/i\u003e\u003ca href=\"https://cloud.google.com/pubsub/overview\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\u003ci\u003eCloud Pub/Sub\u003c/i\u003e\u003c/a\u003e\u003ci\u003eas the transport mechanism for all events.\u003c/i\u003e\u003c/p\u003e\u003ch2\u003eDesigning Spotify’s New Event Delivery System\u003c/h2\u003e\u003cp \u003eOur experience in operating and maintaining our old event delivery system provided plenty of input into the design of the new and improved one. The current design was built on top of an even older system operating on hourly rotated log files. This design choice creates complexity such as the propagation and confirmation of end-of-file markers on each event producing machine. Moreover, the current implementation has some failure modes it can not recover from automatically. A piece of software that requires manual intervention for many failure modes running on each machine that produces logs incurs significant operational cost. In the new system we wanted a simpler design on the log producing machines, handing over events to a smaller set of machines close on the network for further processing.\u003c/p\u003e\u003cp \u003eThe missing piece here is an event delivery system or queue that implements reliable transport of events and the persistence of undelivered messages in the queue. With such a system in place, we should be able to have the producer hand off events close to the producer at a very high rate; receive an acknowledgement back with low latency; and have the rest of the system be responsible for the complexity of making sure that submitted events gets passed to HDFS.\u003c/p\u003e\u003cp \u003eAnother change we made was to have each event type have its own channel, or \u003ci\u003etopic,\u003c/i\u003e and to convert to a more highly structured format early on in the process. Pushing more of the work onto the producer side means that less time needs to be spent converting the format in the Extract, Transform, Load (ETL) job later in the the process. Having separate topics per event is a key requirement for building efficient real time use cases.\u003c/p\u003e\u003cp \u003eSince event delivery is something that simply needs to work, we designed the new system in such a way that it could run in parallel with the current system. The interface, both at the producer and the consumer end, matched the current system and we can verify both performance and correctness of the new system rigorously before making the switch.\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/new-system-design.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_4a878586c2177f60a5ed834a8cc09ace/08338d8cc7b144bfaec449788e834d6f/attachment_4a878586c2177f60a5ed834a8cc09ace.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_4a878586c2177f60a5ed834a8cc09ace\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003e\u003ci\u003eFigure 1. High Level System Design of New Event Delivery System\u003c/i\u003e\u003c/p\u003e\u003cp \u003eThe four main components of the new system are the File Tailer, the Event Delivery Service, the Reliable Persistent Queue, and the ETL job.\u003c/p\u003e\u003cp \u003eIn this design, the Tailer has a much narrower set of responsibilities than the Producer in our old system. It tails log files looking for new events, and forwards them to the Event Delivery Service. As soon as it gets a confirmation that the event has been received it’s responsibility ends. No more complexity handling end-of-file markers or making sure that data has reached it’s final destination in HDFS.\u003c/p\u003e\u003cp \u003eThe Event Delivery Service accepts events from the Tailer, transform them to their final structured format and forwards them to the Queue. It is built as a RESTful microservice using the \u003ca href=\"http://spotify.github.io/apollo/\" target=\"_blank\" rel=\"noopener noreferrer\"\u003eApollo\u003c/a\u003e framework and deployed using the \u003ca href=\"https://github.com/spotify/helios\" target=\"_blank\" rel=\"noopener noreferrer\"\u003eHelios\u003c/a\u003e orchestration platform, a common design pattern at Spotify. It enables clients to be decoupled from the specifics of a single persistence technology as well as enabling any underlying technology to be switched without service disruption.\u003c/p\u003e\u003cp \u003eThe Queue is the core of our system and, as such, is important for it to scale with growing data volumes. To cope with Hadoop downtime, it needs to reliably store messages for a number of days.\u003c/p\u003e\u003cp \u003eThe ETL job should reliably de-duplicate and export events from the Queue to hourly buckets in HDFS. Before it exposes a bucket to the downstream consumers, it needs to detect with high level of confidence that all data for the bucket has been consumed.\u003c/p\u003e\u003cp \u003eIn Figure 1, you can see a box that says “Service Using API directly”. We have felt for some time that syslog was a less-than-awesome API for event producers. When the new system is in production and the old system has been fully retired, it makes sense to move away from syslog and start providing libraries that services can use to communicate directly with the Event Delivery Service.\u003c/p\u003e\u003ch2\u003eChoosing a Reliable Persistent Queue\u003c/h2\u003e\u003ch2\u003eKafka 0.8\u003c/h2\u003e\u003cp \u003eBuilding a Reliable Persistent Queue system that reliably handles Spotify event volumes is a daunting task. Our intention was to leverage existing tools to do the heavy lifting. Since event delivery is the foundation of our data infrastructure, we wanted to play it safe. Our first choice was Kafka 0.8.\u003c/p\u003e\u003cp \u003eThere are many reports that Kafka 0.8 is successfully used by companies of significant size around the world and Kafka 0.8 is a big improvement over the version in use in the current system. In particular, its improved Kafka brokers provide reliable persistent storage. The \u003ca href=\"https://cwiki.apache.org/confluence/pages/viewpage.action?pageId=27846330\" target=\"_blank\" rel=\"noopener noreferrer\"\u003eMirror Maker\u003c/a\u003e project introduced mirroring between data centers, and the \u003ca href=\"http://docs.confluent.io/1.0/camus/docs/intro.html\" target=\"_blank\" rel=\"noopener noreferrer\"\u003eCamus\u003c/a\u003e project can be used for exporting Avro structured events to hourly buckets.\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-kafka-system-design.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_7630604752550316b7850f9185b81327/827376fb232fb9157d5da2da9519adf9/attachment_7630604752550316b7850f9185b81327.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_7630604752550316b7850f9185b81327\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003e\u003ci\u003eFigure 2. Event delivery system design in which we use Kafka 0.8 as reliable persistent queue\u003c/i\u003e\u003c/p\u003e\u003cp \u003eTo prove that event delivery can work as expected on Kafka 0.8, we deployed the test system shown in Figure 2. Embedding a simple Kafka producer in the Event Delivery Service also proved to be easy. To ensure the system worked correctly end-to-end—from Event Delivery Service to HDFS—we embedded various integration tests in our continuous integration and delivery process.\u003c/p\u003e\u003cp \u003eSadly, as soon as this system started handling production traffic, it started to fall apart. The only component that proved to be stable was Camus (but since we didn’t push much load through the system, we still don’t know how Camus would perform under stress).\u003c/p\u003e\u003cp \u003eMirror Maker gave us the most headaches. We assumed it would reliably mirror data between data centers, but this simply wasn’t the case. It only \u003ca href=\"https://cwiki.apache.org/confluence/display/KAFKA/KIP-3+-+Mirror+Maker+Enhancement\" target=\"_blank\" rel=\"noopener noreferrer\"\u003emirrored data on a best effort basis\u003c/a\u003e. If there were issues with the destination cluster, the Mirror Makers would just drop data while reporting to source cluster that data had been successfully mirrored. (Note that this behaviour should be fixed in Kafka 0.9.)\u003c/p\u003e\u003cp \u003eMirror Makers occasionally got confused about who was the leader for consumption. The leader would sometimes forget that it was a leader, while the other Mirror Makers from the cluster would happily still try to follow it. When this happened, mirroring between data centers would stop.\u003c/p\u003e\u003cp \u003eThe Kafka Producer also had serious issues with stability. If one or more brokers from a cluster was removed, or even just restarted, it was quite likely that the producer would enter a state from which it couldn’t recover by itself. While it was in such a state it wouldn’t produce any events. The only solution was to restart the whole service.\u003c/p\u003e\u003cp \u003eEven without solving these issues, we saw that a lot of work would be needed to make the system production-ready. We would need to define deployment strategies for Kafka Brokers and Mirror Makers, do capacity modelling and planning for all system components, and expose performance metrics to \u003ca href=\"https://spotify.github.io/heroic/#!/index\" target=\"_blank\" rel=\"noopener noreferrer\"\u003eSpotify’s monitoring system\u003c/a\u003e.\u003c/p\u003e\u003cp \u003eWe found ourself at a crossroads. Should we make a significant investment and try to get Kafka work for us? Or should we try something else?\u003c/p\u003e\u003ch2\u003eCloud Pub/Sub\u003c/h2\u003e\u003cp \u003eWhile we were struggling with Kafka, various other Spotify teams were beginning to experiment with Google Cloud products. A particularly interesting products that was being assessed was \u003ca href=\"https://cloud.google.com/pubsub/overview\" target=\"_blank\" rel=\"noopener noreferrer\"\u003eCloud Pub/Sub\u003c/a\u003e. It seemed as if Cloud Pub/Sub might satisfy our basic need for a reliable, persistent queue: it can retain undelivered data for \u003ca href=\"https://cloud.google.com/pubsub/quotas\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e7 days\u003c/a\u003e, provide reliability through application-level acknowledgements, and has “at-least-once” delivery semantics.\u003c/p\u003e\u003cp \u003eAs well as satisfying our basic needs, Cloud Pub/Sub came with extra goodies:\u003c/p\u003e\u003cul\u003e\u003cli\u003e\u003cp \u003e\u003cb\u003eGlobal availability—as a global service, Pub/Sub is available in all \u003c/b\u003e\u003ca href=\"https://cloud.google.com/compute/docs/zones#available\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\u003cb\u003eGoogle Cloud Zones\u003c/b\u003e\u003c/a\u003e\u003cb\u003e; transferring data between our data center wouldn’t be through our normal internet provider but would use underlying Google network.\u003c/b\u003e\u003c/p\u003e\u003c/li\u003e\u003cli\u003e\u003cp \u003e\u003cb\u003eA simple REST API\u003c/b\u003e—if we didn’t like client library Google provided, we can easily write our own.\u003c/p\u003e\u003c/li\u003e\u003cli\u003e\u003cp \u003e\u003cb\u003eOperational responsibility was handled by someone else\u003c/b\u003e—there was no need to create a capacity model or deployment strategy, or to set up monitoring and alerting.\u003c/p\u003e\u003c/li\u003e\u003c/ul\u003e\u003cp \u003eIt all sounded great on paper… but was it too good to be true? The solutions we’d built on Apache Kafka, while not perfect, has served us well. We had lots of experience of the different failure modes, access to the hardware and source code, and could—theoretically—find the root cause of any problem. Moving to a managed service would mean we’d have to trust operations of another organisation. And Cloud Pub/Sub was being advertised as beta software; we were unaware of any organisation other than Google who were using it at our scale.\u003c/p\u003e\u003cp \u003eWith this in mind, we decided that we needed a comprehensive test plan to make absolutely sure that, if we were to go with Cloud Pub/Sub, it would meet \u003ci\u003eall\u003c/i\u003eof our requirements.\u003c/p\u003e\u003ch3\u003eThe Producer load test\u003c/h3\u003e\u003cp \u003eThe first item on our agenda was testing Cloud Pub/Sub to see if it could handle the anticipated load. Currently our production load peaks at around 700K events per second. To account for the future growth and possible disaster recovery scenarios, we settled on a test load of 2M events per second. To make it extra hard for Pub/Sub, we wanted to publish this amount of traffic from a single data center, so that all the requests were hitting the Pub/Sub machines in the same zone. We made the assumption that Google plans zones as independent failure domains and that each zone can handle equal amounts of traffic. In theory, if we’re able to push 2M messages to a single zone, we should be able to push \u003ci\u003enumber_of_zones\u003c/i\u003e* 2M messages across all zones. Our hope was that the system would be able to handle this traffic on both the producing and consuming side for a long time without the service degrading.\u003c/p\u003e\u003cp \u003eEarly on, we hit a stumbling block: the \u003ca href=\"https://developers.google.com/api-client-library/java/apis/pubsub/v1\" target=\"_blank\" rel=\"noopener noreferrer\"\u003eCloud Pub/Sub Java client\u003c/a\u003e simply didn’t perform well enough. The client, like many other Google Cloud API clients, is auto-generated from API specifications. That’s good if you want clients that support a wide variety of languages, but not so good if you want a high performance client for a single language.\u003c/p\u003e\u003cp \u003eThankfully Pub/Sub has a REST API, so it was easy to write our own \u003ca href=\"https://github.com/spotify/async-google-pubsub-client\" target=\"_blank\" rel=\"noopener noreferrer\"\u003elibrary\u003c/a\u003e. We designed the new client with its performance foremost in our minds. To enable better use of resources, we used asynchronous Java. We also added queuing and batching in the client. (This wasn’t first time that we needed to roll our sleeves up and reimplement a Google Cloud API client: in another \u003ca href=\"https://github.com/spotify/async-datastore-client\" target=\"_blank\" rel=\"noopener noreferrer\"\u003eproject\u003c/a\u003e we implemented a high performance client for the Datastore API.)\u003c/p\u003e\u003cp \u003eWith our new client in place, we were ready to start pushing some serious load to Pub/Sub. We used a simple load generator to send mock traffic through the Event Service to Pub/Sub. The generated traffic was routed through two Pub/Sub topics with a ratio of 7:3. To push 2M messages per second, we ran the Event Service on 29 machines.\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-2xx.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_4bebfa1fba3349ca8523e15fe0a2a4cd/5543229ae361cb0a8fc5e80c673a7fc5/attachment_4bebfa1fba3349ca8523e15fe0a2a4cd.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_4bebfa1fba3349ca8523e15fe0a2a4cd\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003e\u003ci\u003eFigure 3. Number of successful requests per second to Pub/Sub from all data centers\u003c/i\u003e\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-5xx.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_6baaf6ce972cf5c497ef1b3c924699be/c21e0e21da1a2c0fe6a12b58c8524cdb/attachment_6baaf6ce972cf5c497ef1b3c924699be.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_6baaf6ce972cf5c497ef1b3c924699be\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003e\u003ci\u003eFigure 4. Number of failed requests per second to Pub/Sub from all data centers\u003c/i\u003e\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/net_trafiic.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_37b2dba282bd7fb0a2da406c66453bd6/f352dc0a31b81881ae4bd7b318c0f952/attachment_37b2dba282bd7fb0a2da406c66453bd6.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_37b2dba282bd7fb0a2da406c66453bd6\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003e\u003ci\u003eFigure 5. Incoming and outgoing network traffic for Event Service machines in bps\u003c/i\u003e\u003c/p\u003e\u003cp \u003ePub/Sub passed the test with flying colours. We published 2M messages without any service degradation and received almost no server errors from the Pub/Sub backend. Enabling batching and compression on the Event Service machines resulted in ~1Gbps of network traffic towards Pub/Sub.\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_9efc288a516f91395d3dd62fb53b526d/5eb9ecea44b4a5d1805dfcbd1f41dc6a/attachment_9efc288a516f91395d3dd62fb53b526d.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_9efc288a516f91395d3dd62fb53b526d\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003e\u003ci\u003eFigure 6. Google Cloud Monitoring graph for total published messages to Pub/Sub\u003c/i\u003e\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published-per-topic.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_7fb2c75f899c8879887e4691fe5de179/0a923e9f7542f09b60d3f97e81a3acfe/attachment_7fb2c75f899c8879887e4691fe5de179.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_7fb2c75f899c8879887e4691fe5de179\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003eFigure 7. Google Cloud Monitoring graph for per topic published messages to Pub/Sub\u003c/p\u003e\u003cp \u003eA useful side-effect of our test was that we could compare our internal metrics with the metrics exposed by Google. As it can been seen by looking Figure 3 and Figure 6, the graphs match perfectly.\u003c/p\u003e\u003ch3\u003eThe Consumer stability test\u003c/h3\u003e\u003cp \u003eOur second major test focused on consumption. Over a period of 5 days, we measured the end-to-end latency of the system under heavy load. For the duration of the test we published, on average, around 800K messages per second. To mimic real world load variations, the publishing rate varied according to the time of day. To verify that we could use multiple topics concurrently, all data was published to two topics with ratio 7:3.\u003c/p\u003e\u003cp \u003eA slightly surprising behaviour of Cloud Pub/Sub is that subscriptions need to be created before messages can be persisted: until the subscription exists, no data is retained. Every subscription stores data independently and there is no limit to how many consumers a subscription can have. Consumers are coordinated on the server side, and the server is responsible for fairly allocating the messages to all the consumers that request data. This is a very different to Kafka: in Kafka, data is retained per created topic and the number of Kafka consumers per topic is limited by the number of topic partitions.\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/long-term-load-test.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7/743803c106c9521d7fb3e0d53dc292b7/attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003e\u003ci\u003eFigure 8. The consumption test dashboard\u003c/i\u003e\u003c/p\u003e\u003cp \u003eIn our test, we created a subscription, then one hour later we started to consume data. We consumed data in batches of 1000 messages. Since we didn’t try to push consumption as high as we could have, we only consumed events at a slightly higher rate than we were sending at peak. It took about 8 hours to catch up. Once we caught up, consumers kept consuming at a rate that matched the publishing rate.\u003c/p\u003e\u003cp \u003eThe median end-to-end latency we measured during the test period—including backlog recovery—was around 20 seconds. We did not observe any lost messages whatsoever during the test period.\u003c/p\u003e\u003ch2\u003eDecision\u003c/h2\u003e\u003cp \u003eBased on these tests, we felt confident that Cloud Pub/Sub was the right choice for us. Latency was low and consistent, and the only capacity limitations we encountered was the one explicitly set by the available quota. In short, choosing Cloud Pub/Sub rather than Kafka 0.8 for our new event delivery platform was an obvious choice.\u003c/p\u003e\n \u003cfigure class=\"figure-image\"\u003e\n \n \u003ca href=\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-system-design-2x.png\" target=\"_blank\" rel=\"noopener noreferrer\"\u003e\n \n \u003cimg \n src=\"//images.ctfassets.net/p762jor363g1/attachment_b78acd55ce53457d80eaf6cb49c7350e/b109c47ca3ed9476a365baef5383796c/attachment_b78acd55ce53457d80eaf6cb49c7350e.png\" \n style=\"width: 100%; max-width: 100%; height: auto; \"\n class=\"blog-image\"\n alt=\"attachment_b78acd55ce53457d80eaf6cb49c7350e\"\n /\u003e\n \n \u003c/a\u003e\n \n \n \u003c/figure\u003e\n \u003cp \u003e\u003ci\u003eFigure 9. Event delivery system design in which we use Cloud Pub/Sub as reliable persistent queue\u003c/i\u003e\u003c/p\u003e\u003ch2\u003eNext step\u003c/h2\u003e\u003cp \u003eAfter events are safely persisted in Pub/Sub it’s time to export them to HDFS. To fully leverage the Google Cloud offering we decided to give a chance to Dataflow.\u003c/p\u003e\u003cp \u003eIn the last blog post from this series we’re going to go through our plan on leveraging Dataflow for this job. Stay tuned.\u003c/p\u003e"])</script><script>self.__next_f.push([1,"5:[[\"$\",\"div\",null,{\"className\":\"single-post\",\"children\":[[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[\"$\",\"h1\",null,{\"className\":\"single-post__title\",\"children\":\"Spotify’s Event Delivery – The Road to the Cloud (Part II)\"}]}],[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[[\"$\",\"div\",null,{\"className\":\"single-post__content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"single-post__featured-image\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"https://images.ctfassets.net/p762jor363g1/attachment_065167b86d3cf061473c42f392138893/f28a6eb4a75e4f6778409d68d096e714/attachment_065167b86d3cf061473c42f392138893.png\",\"alt\":\"Feature Image\",\"width\":740,\"height\":404,\"priority\":true}]}],[\"$\",\"div\",null,{\"className\":\"html-content\",\"dangerouslySetInnerHTML\":{\"__html\":\"$1f\"}}],null,\"$L20\"]}],\"$L21\"]}]]}],\"$L22\"]\n"])</script><script>self.__next_f.push([1,"23:I[58931,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\",\"/_next/static/chunks/0wni0lk_-haja.js\"],\"SocialShare\"]\n24:I[40367,[\"/_next/static/chunks/0a_f9~n73ra-c.js\",\"/_next/static/chunks/0d3shmwh5_nmn.js\",\"/_next/static/chunks/08c8uq~xsq31o.js\",\"/_next/static/chunks/0wni0lk_-haja.js\"],\"default\"]\n20:[\"$\",\"$L23\",null,{\"heading\":\"SHARE THIS ARTICLE\",\"title\":\"Spotify’s Event Delivery – The Road to the Cloud (Part II)\"}]\n"])</script><script>self.__next_f.push([1,"21:[\"$\",\"$L24\",null,{\"blogPost\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"blogpost_228800ecf71a7be6c3d43c7bb016e65a\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T22:26:20.011Z\",\"updatedAt\":\"2025-04-25T10:44:55.580Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":8,\"revision\":4,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"blogPost\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"Spotify’s Event Delivery – The Road to the Cloud (Part II)\",\"slug\":\"spotifys-event-delivery-the-road-to-the-cloud-part-ii\",\"content\":{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Whenever a user performs an action in the Spotify client—such as listening to a song or searching for an artist—a small piece of information, an event, is sent to our servers. Event delivery, the process of making sure that all events gets transported safely from clients all over the world to our central processing system, is an interesting problem. In this series of blog posts, we are going to look at some of the work we have done in this area. More specifically, we are going to look at the architecture of our new event delivery system, and tell you why we choose to base our new system on Google Cloud managed services.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"In the\",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://labs.spotify.com/2016/02/25/spotifys-event-delivery-the-road-to-the-cloud-part-i/\"},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"first post\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"in this series, we talked about how our old event system worked and some of the lessons we learned from operating it. In this second post, we’ll cover the design of our new event delivery system, and why we choose\",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://cloud.google.com/pubsub/overview\"},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Cloud Pub/Sub\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"as the transport mechanism for all events.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Designing Spotify’s New Event Delivery System\",\"nodeType\":\"text\"}],\"nodeType\":\"heading-2\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Our experience in operating and maintaining our old event delivery system provided plenty of input into the design of the new and improved one. The current design was built on top of an even older system operating on hourly rotated log files. This design choice creates complexity such as the propagation and confirmation of end-of-file markers on each event producing machine. Moreover, the current implementation has some failure modes it can not recover from automatically. A piece of software that requires manual intervention for many failure modes running on each machine that produces logs incurs significant operational cost. In the new system we wanted a simpler design on the log producing machines, handing over events to a smaller set of machines close on the network for further processing.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The missing piece here is an event delivery system or queue that implements reliable transport of events and the persistence of undelivered messages in the queue. With such a system in place, we should be able to have the producer hand off events close to the producer at a very high rate; receive an acknowledgement back with low latency; and have the rest of the system be responsible for the complexity of making sure that submitted events gets passed to HDFS.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Another change we made was to have each event type have its own channel, or \",\"nodeType\":\"text\"},{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"topic,\",\"nodeType\":\"text\"},{\"data\":{},\"marks\":[],\"value\":\" and to convert to a more highly structured format early on in the process. Pushing more of the work onto the producer side means that less time needs to be spent converting the format in the Extract, Transform, Load (ETL) job later in the the process. Having separate topics per event is a key requirement for building efficient real time use cases.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Since event delivery is something that simply needs to work, we designed the new system in such a way that it could run in parallel with the current system. The interface, both at the producer and the consumer end, matched the current system and we can verify both performance and correctness of the new system rigorously before making the switch.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_457f5a9451f7aefc8d3f6ea325f6727f\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:29:14.594Z\",\"updatedAt\":\"2025-03-13T22:10:49.943Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_4a878586c2177f60a5ed834a8cc09ace\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_4a878586c2177f60a5ed834a8cc09ace\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:44.641Z\",\"updatedAt\":\"2025-03-13T13:52:44.641Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_4a878586c2177f60a5ed834a8cc09ace\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_4a878586c2177f60a5ed834a8cc09ace/08338d8cc7b144bfaec449788e834d6f/attachment_4a878586c2177f60a5ed834a8cc09ace.png\",\"details\":{\"size\":45932,\"image\":{\"width\":941,\"height\":509}},\"fileName\":\"attachment_4a878586c2177f60a5ed834a8cc09ace.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"New System Design\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/new-system-design.png\",\"raw_html\":\"\u003cimg alt=\\\"New System Design\\\" class=\\\"wp-image-1439\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/new-system-design.png?w=730\u0026amp;h=395\\\"/\u003e\",\"html_attributes\":{\"alt\":\"New System Design\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/new-system-design.png?w=730\u0026h=395\",\"class\":[\"wp-image-1439\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Figure 1. High Level System Design of New Event Delivery System\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The four main components of the new system are the File Tailer, the Event Delivery Service, the Reliable Persistent Queue, and the ETL job.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"In this design, the Tailer has a much narrower set of responsibilities than the Producer in our old system. It tails log files looking for new events, and forwards them to the Event Delivery Service. As soon as it gets a confirmation that the event has been received it’s responsibility ends. No more complexity handling end-of-file markers or making sure that data has reached it’s final destination in HDFS.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The Event Delivery Service accepts events from the Tailer, transform them to their final structured format and forwards them to the Queue. It is built as a RESTful microservice using the \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"http://spotify.github.io/apollo/\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Apollo\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\" framework and deployed using the \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://github.com/spotify/helios\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Helios\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\" orchestration platform, a common design pattern at Spotify. It enables clients to be decoupled from the specifics of a single persistence technology as well as enabling any underlying technology to be switched without service disruption.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The Queue is the core of our system and, as such, is important for it to scale with growing data volumes. To cope with Hadoop downtime, it needs to reliably store messages for a number of days.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The ETL job should reliably de-duplicate and export events from the Queue to hourly buckets in HDFS. Before it exposes a bucket to the downstream consumers, it needs to detect with high level of confidence that all data for the bucket has been consumed.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"In Figure 1, you can see a box that says “Service Using API directly”. We have felt for some time that syslog was a less-than-awesome API for event producers. When the new system is in production and the old system has been fully retired, it makes sense to move away from syslog and start providing libraries that services can use to communicate directly with the Event Delivery Service.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Choosing a Reliable Persistent Queue\",\"nodeType\":\"text\"}],\"nodeType\":\"heading-2\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Kafka 0.8\",\"nodeType\":\"text\"}],\"nodeType\":\"heading-2\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Building a Reliable Persistent Queue system that reliably handles Spotify event volumes is a daunting task. Our intention was to leverage existing tools to do the heavy lifting. Since event delivery is the foundation of our data infrastructure, we wanted to play it safe. Our first choice was Kafka 0.8.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"There are many reports that Kafka 0.8 is successfully used by companies of significant size around the world and Kafka 0.8 is a big improvement over the version in use in the current system. In particular, its improved Kafka brokers provide reliable persistent storage. The \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://cwiki.apache.org/confluence/pages/viewpage.action?pageId=27846330\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Mirror Maker\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\" project introduced mirroring between data centers, and the \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"http://docs.confluent.io/1.0/camus/docs/intro.html\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Camus\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\" project can be used for exporting Avro structured events to hourly buckets.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_5ff031d7d855fc3035a1d2c8bc05134b\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:29:23.731Z\",\"updatedAt\":\"2025-03-13T22:10:52.389Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_7630604752550316b7850f9185b81327\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_7630604752550316b7850f9185b81327\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:45.470Z\",\"updatedAt\":\"2025-03-13T13:52:45.470Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_7630604752550316b7850f9185b81327\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_7630604752550316b7850f9185b81327/827376fb232fb9157d5da2da9519adf9/attachment_7630604752550316b7850f9185b81327.png\",\"details\":{\"size\":42531,\"image\":{\"width\":943,\"height\":454}},\"fileName\":\"attachment_7630604752550316b7850f9185b81327.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"Gabo Kafka System Design\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-kafka-system-design.png\",\"raw_html\":\"\u003cimg alt=\\\"Gabo Kafka System Design\\\" class=\\\"wp-image-1438\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-kafka-system-design.png?w=730\u0026amp;h=351\\\"/\u003e\",\"html_attributes\":{\"alt\":\"Gabo Kafka System Design\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-kafka-system-design.png?w=730\u0026h=351\",\"class\":[\"wp-image-1438\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Figure 2. Event delivery system design in which we use Kafka 0.8 as reliable persistent queue\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"To prove that event delivery can work as expected on Kafka 0.8, we deployed the test system shown in Figure 2. Embedding a simple Kafka producer in the Event Delivery Service also proved to be easy. To ensure the system worked correctly end-to-end—from Event Delivery Service to HDFS—we embedded various integration tests in our continuous integration and delivery process.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Sadly, as soon as this system started handling production traffic, it started to fall apart. The only component that proved to be stable was Camus (but since we didn’t push much load through the system, we still don’t know how Camus would perform under stress).\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Mirror Maker gave us the most headaches. We assumed it would reliably mirror data between data centers, but this simply wasn’t the case. It only \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://cwiki.apache.org/confluence/display/KAFKA/KIP-3+-+Mirror+Maker+Enhancement\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"mirrored data on a best effort basis\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\". If there were issues with the destination cluster, the Mirror Makers would just drop data while reporting to source cluster that data had been successfully mirrored. (Note that this behaviour should be fixed in Kafka 0.9.)\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Mirror Makers occasionally got confused about who was the leader for consumption. The leader would sometimes forget that it was a leader, while the other Mirror Makers from the cluster would happily still try to follow it. When this happened, mirroring between data centers would stop.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The Kafka Producer also had serious issues with stability. If one or more brokers from a cluster was removed, or even just restarted, it was quite likely that the producer would enter a state from which it couldn’t recover by itself. While it was in such a state it wouldn’t produce any events. The only solution was to restart the whole service.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Even without solving these issues, we saw that a lot of work would be needed to make the system production-ready. We would need to define deployment strategies for Kafka Brokers and Mirror Makers, do capacity modelling and planning for all system components, and expose performance metrics to \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://spotify.github.io/heroic/#!/index\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Spotify’s monitoring system\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\".\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"We found ourself at a crossroads. Should we make a significant investment and try to get Kafka work for us? Or should we try something else?\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Cloud Pub/Sub\",\"nodeType\":\"text\"}],\"nodeType\":\"heading-2\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"While we were struggling with Kafka, various other Spotify teams were beginning to experiment with Google Cloud products. A particularly interesting products that was being assessed was \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://cloud.google.com/pubsub/overview\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Cloud Pub/Sub\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\". It seemed as if Cloud Pub/Sub might satisfy our basic need for a reliable, persistent queue: it can retain undelivered data for \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://cloud.google.com/pubsub/quotas\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"7 days\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\", provide reliability through application-level acknowledgements, and has “at-least-once” delivery semantics.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"As well as satisfying our basic needs, Cloud Pub/Sub came with extra goodies:\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"bold\"}],\"value\":\"Global availability—as a global service, Pub/Sub is available in all \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://cloud.google.com/compute/docs/zones#available\"},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"bold\"}],\"value\":\"Google Cloud Zones\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[{\"type\":\"bold\"}],\"value\":\"; transferring data between our data center wouldn’t be through our normal internet provider but would use underlying Google network.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"}],\"nodeType\":\"list-item\"},{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"bold\"}],\"value\":\"A simple REST API\",\"nodeType\":\"text\"},{\"data\":{},\"marks\":[],\"value\":\"—if we didn’t like client library Google provided, we can easily write our own.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"}],\"nodeType\":\"list-item\"},{\"data\":{},\"content\":[{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"bold\"}],\"value\":\"Operational responsibility was handled by someone else\",\"nodeType\":\"text\"},{\"data\":{},\"marks\":[],\"value\":\"—there was no need to create a capacity model or deployment strategy, or to set up monitoring and alerting.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"}],\"nodeType\":\"list-item\"}],\"nodeType\":\"unordered-list\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"It all sounded great on paper… but was it too good to be true? The solutions we’d built on Apache Kafka, while not perfect, has served us well. We had lots of experience of the different failure modes, access to the hardware and source code, and could—theoretically—find the root cause of any problem. Moving to a managed service would mean we’d have to trust operations of another organisation. And Cloud Pub/Sub was being advertised as beta software; we were unaware of any organisation other than Google who were using it at our scale.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"With this in mind, we decided that we needed a comprehensive test plan to make absolutely sure that, if we were to go with Cloud Pub/Sub, it would meet \",\"nodeType\":\"text\"},{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"all\",\"nodeType\":\"text\"},{\"data\":{},\"marks\":[],\"value\":\"of our requirements.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The Producer load test\",\"nodeType\":\"text\"}],\"nodeType\":\"heading-3\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The first item on our agenda was testing Cloud Pub/Sub to see if it could handle the anticipated load. Currently our production load peaks at around 700K events per second. To account for the future growth and possible disaster recovery scenarios, we settled on a test load of 2M events per second. To make it extra hard for Pub/Sub, we wanted to publish this amount of traffic from a single data center, so that all the requests were hitting the Pub/Sub machines in the same zone. We made the assumption that Google plans zones as independent failure domains and that each zone can handle equal amounts of traffic. In theory, if we’re able to push 2M messages to a single zone, we should be able to push \",\"nodeType\":\"text\"},{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"number_of_zones\",\"nodeType\":\"text\"},{\"data\":{},\"marks\":[],\"value\":\"* 2M messages across all zones. Our hope was that the system would be able to handle this traffic on both the producing and consuming side for a long time without the service degrading.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Early on, we hit a stumbling block: the \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://developers.google.com/api-client-library/java/apis/pubsub/v1\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Cloud Pub/Sub Java client\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\" simply didn’t perform well enough. The client, like many other Google Cloud API clients, is auto-generated from API specifications. That’s good if you want clients that support a wide variety of languages, but not so good if you want a high performance client for a single language.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Thankfully Pub/Sub has a REST API, so it was easy to write our own \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://github.com/spotify/async-google-pubsub-client\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"library\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\". We designed the new client with its performance foremost in our minds. To enable better use of resources, we used asynchronous Java. We also added queuing and batching in the client. (This wasn’t first time that we needed to roll our sleeves up and reimplement a Google Cloud API client: in another \",\"nodeType\":\"text\"},{\"data\":{\"uri\":\"https://github.com/spotify/async-datastore-client\"},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"project\",\"nodeType\":\"text\"}],\"nodeType\":\"hyperlink\"},{\"data\":{},\"marks\":[],\"value\":\" we implemented a high performance client for the Datastore API.)\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"With our new client in place, we were ready to start pushing some serious load to Pub/Sub. We used a simple load generator to send mock traffic through the Event Service to Pub/Sub. The generated traffic was routed through two Pub/Sub topics with a ratio of 7:3. To push 2M messages per second, we ran the Event Service on 29 machines.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_43ef39ab6420c3e4fc0c006cc171ec2e\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:29:31.542Z\",\"updatedAt\":\"2025-03-13T22:10:54.944Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_4bebfa1fba3349ca8523e15fe0a2a4cd\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_4bebfa1fba3349ca8523e15fe0a2a4cd\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:46.298Z\",\"updatedAt\":\"2025-03-13T13:52:46.298Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_4bebfa1fba3349ca8523e15fe0a2a4cd\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_4bebfa1fba3349ca8523e15fe0a2a4cd/5543229ae361cb0a8fc5e80c673a7fc5/attachment_4bebfa1fba3349ca8523e15fe0a2a4cd.png\",\"details\":{\"size\":26034,\"image\":{\"width\":736,\"height\":241}},\"fileName\":\"attachment_4bebfa1fba3349ca8523e15fe0a2a4cd.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"gabo-pref-2xx\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-2xx.png\",\"raw_html\":\"\u003cimg alt=\\\"gabo-pref-2xx\\\" class=\\\"wp-image-1437\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-2xx.png?w=730\u0026amp;h=239\\\"/\u003e\",\"html_attributes\":{\"alt\":\"gabo-pref-2xx\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-2xx.png?w=730\u0026h=239\",\"class\":[\"wp-image-1437\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Figure 3. Number of successful requests per second to Pub/Sub from all data centers\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_d1779652e9d9721e5af7f3265b155df3\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:29:38.045Z\",\"updatedAt\":\"2025-03-13T22:10:57.578Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_6baaf6ce972cf5c497ef1b3c924699be\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_6baaf6ce972cf5c497ef1b3c924699be\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:47.113Z\",\"updatedAt\":\"2025-03-13T13:52:47.113Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_6baaf6ce972cf5c497ef1b3c924699be\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_6baaf6ce972cf5c497ef1b3c924699be/c21e0e21da1a2c0fe6a12b58c8524cdb/attachment_6baaf6ce972cf5c497ef1b3c924699be.png\",\"details\":{\"size\":18322,\"image\":{\"width\":732,\"height\":239}},\"fileName\":\"attachment_6baaf6ce972cf5c497ef1b3c924699be.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"gabo-pref-5xx\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-5xx.png\",\"raw_html\":\"\u003cimg alt=\\\"gabo-pref-5xx\\\" class=\\\"wp-image-1436\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-5xx.png?w=730\u0026amp;h=238\\\"/\u003e\",\"html_attributes\":{\"alt\":\"gabo-pref-5xx\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-pref-5xx.png?w=730\u0026h=238\",\"class\":[\"wp-image-1436\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Figure 4. Number of failed requests per second to Pub/Sub from all data centers\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_cd15d735a29f653ebddb503eabc9998c\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:29:42.243Z\",\"updatedAt\":\"2025-03-13T22:11:00.273Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_37b2dba282bd7fb0a2da406c66453bd6\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_37b2dba282bd7fb0a2da406c66453bd6\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:47.986Z\",\"updatedAt\":\"2025-03-13T13:52:47.986Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_37b2dba282bd7fb0a2da406c66453bd6\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_37b2dba282bd7fb0a2da406c66453bd6/f352dc0a31b81881ae4bd7b318c0f952/attachment_37b2dba282bd7fb0a2da406c66453bd6.png\",\"details\":{\"size\":62382,\"image\":{\"width\":1600,\"height\":532}},\"fileName\":\"attachment_37b2dba282bd7fb0a2da406c66453bd6.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"net_trafiic\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/net_trafiic.png\",\"raw_html\":\"\u003cimg alt=\\\"net_trafiic\\\" class=\\\"wp-image-1435\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/net_trafiic.png?w=730\u0026amp;h=243\\\"/\u003e\",\"html_attributes\":{\"alt\":\"net_trafiic\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/net_trafiic.png?w=730\u0026h=243\",\"class\":[\"wp-image-1435\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Figure 5. Incoming and outgoing network traffic for Event Service machines in bps\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Pub/Sub passed the test with flying colours. We published 2M messages without any service degradation and received almost no server errors from the Pub/Sub backend. Enabling batching and compression on the Event Service machines resulted in ~1Gbps of network traffic towards Pub/Sub.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_7acaf6e7419694e8a07c39289e7e17a4\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:30:08.549Z\",\"updatedAt\":\"2025-03-13T22:11:02.793Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_9efc288a516f91395d3dd62fb53b526d\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_9efc288a516f91395d3dd62fb53b526d\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:48.822Z\",\"updatedAt\":\"2025-03-13T13:52:48.822Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_9efc288a516f91395d3dd62fb53b526d\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_9efc288a516f91395d3dd62fb53b526d/5eb9ecea44b4a5d1805dfcbd1f41dc6a/attachment_9efc288a516f91395d3dd62fb53b526d.png\",\"details\":{\"size\":12629,\"image\":{\"width\":789,\"height\":218}},\"fileName\":\"attachment_9efc288a516f91395d3dd62fb53b526d.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"pubsub-published\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published.png\",\"raw_html\":\"\u003cimg alt=\\\"pubsub-published\\\" class=\\\"wp-image-1434\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published.png?w=730\u0026amp;h=202\\\"/\u003e\",\"html_attributes\":{\"alt\":\"pubsub-published\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published.png?w=730\u0026h=202\",\"class\":[\"wp-image-1434\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Figure 6. Google Cloud Monitoring graph for total published messages to Pub/Sub\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_d892e2594ec42a495604b956708d2b20\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:30:20.326Z\",\"updatedAt\":\"2025-03-13T22:11:05.299Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_7fb2c75f899c8879887e4691fe5de179\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_7fb2c75f899c8879887e4691fe5de179\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:49.784Z\",\"updatedAt\":\"2025-03-13T13:52:49.784Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_7fb2c75f899c8879887e4691fe5de179\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_7fb2c75f899c8879887e4691fe5de179/0a923e9f7542f09b60d3f97e81a3acfe/attachment_7fb2c75f899c8879887e4691fe5de179.png\",\"details\":{\"size\":13216,\"image\":{\"width\":786,\"height\":217}},\"fileName\":\"attachment_7fb2c75f899c8879887e4691fe5de179.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"pubsub-published-per-topic\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published-per-topic.png\",\"raw_html\":\"\u003cimg alt=\\\"pubsub-published-per-topic\\\" class=\\\"wp-image-1433\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published-per-topic.png?w=730\u0026amp;h=202\\\"/\u003e\",\"html_attributes\":{\"alt\":\"pubsub-published-per-topic\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/pubsub-published-per-topic.png?w=730\u0026h=202\",\"class\":[\"wp-image-1433\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Figure 7. Google Cloud Monitoring graph for per topic published messages to Pub/Sub\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"A useful side-effect of our test was that we could compare our internal metrics with the metrics exposed by Google. As it can been seen by looking Figure 3 and Figure 6, the graphs match perfectly.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The Consumer stability test\",\"nodeType\":\"text\"}],\"nodeType\":\"heading-3\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Our second major test focused on consumption. Over a period of 5 days, we measured the end-to-end latency of the system under heavy load. For the duration of the test we published, on average, around 800K messages per second. To mimic real world load variations, the publishing rate varied according to the time of day. To verify that we could use multiple topics concurrently, all data was published to two topics with ratio 7:3.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"A slightly surprising behaviour of Cloud Pub/Sub is that subscriptions need to be created before messages can be persisted: until the subscription exists, no data is retained. Every subscription stores data independently and there is no limit to how many consumers a subscription can have. Consumers are coordinated on the server side, and the server is responsible for fairly allocating the messages to all the consumers that request data. This is a very different to Kafka: in Kafka, data is retained per created topic and the number of Kafka consumers per topic is limited by the number of topic partitions.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_391052e9b876268f11a497cb5c3eee5c\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:30:26.178Z\",\"updatedAt\":\"2025-03-13T22:11:07.864Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:50.598Z\",\"updatedAt\":\"2025-03-13T13:52:50.598Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7/743803c106c9521d7fb3e0d53dc292b7/attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7.png\",\"details\":{\"size\":130320,\"image\":{\"width\":1600,\"height\":864}},\"fileName\":\"attachment_c8b66f47e8ad08cafe4c1634a8f8f3b7.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"long-term-load-test\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/long-term-load-test.png\",\"raw_html\":\"\u003cimg alt=\\\"long-term-load-test\\\" class=\\\"wp-image-1432\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/long-term-load-test.png?w=730\u0026amp;h=394\\\"/\u003e\",\"html_attributes\":{\"alt\":\"long-term-load-test\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/long-term-load-test.png?w=730\u0026h=394\",\"class\":[\"wp-image-1432\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Figure 8. The consumption test dashboard\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"In our test, we created a subscription, then one hour later we started to consume data. We consumed data in batches of 1000 messages. Since we didn’t try to push consumption as high as we could have, we only consumed events at a slightly higher rate than we were sending at peak. It took about 8 hours to catch up. Once we caught up, consumers kept consuming at a rate that matched the publishing rate.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"The median end-to-end latency we measured during the test period—including backlog recovery—was around 20 seconds. We did not observe any lost messages whatsoever during the test period.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Decision\",\"nodeType\":\"text\"}],\"nodeType\":\"heading-2\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Based on these tests, we felt confident that Cloud Pub/Sub was the right choice for us. Latency was low and consistent, and the only capacity limitations we encountered was the one explicitly set by the available quota. In short, choosing Cloud Pub/Sub rather than Kafka 0.8 for our new event delivery platform was an obvious choice.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{\"target\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"assetwrapper_b7191e84a8ecb65cb0b34f7bbc50ec0e\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T14:28:22.060Z\",\"updatedAt\":\"2025-03-13T22:10:31.231Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"assetWrapper\"}},\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_b78acd55ce53457d80eaf6cb49c7350e\",\"asset\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_b78acd55ce53457d80eaf6cb49c7350e\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:52:38.749Z\",\"updatedAt\":\"2025-03-13T13:52:38.749Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"attachment_b78acd55ce53457d80eaf6cb49c7350e\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_b78acd55ce53457d80eaf6cb49c7350e/b109c47ca3ed9476a365baef5383796c/attachment_b78acd55ce53457d80eaf6cb49c7350e.png\",\"details\":{\"size\":56490,\"image\":{\"width\":1473,\"height\":825}},\"fileName\":\"attachment_b78acd55ce53457d80eaf6cb49c7350e.png\",\"contentType\":\"image/png\"}}},\"caption\":null,\"alt\":\"Gabo System Design 2x\",\"width\":\"None\",\"height\":\"None\",\"type\":\"image\",\"alignment\":\"left\",\"hyperlink\":\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-system-design-2x.png\",\"raw_html\":\"\u003cimg alt=\\\"Gabo System Design 2x\\\" class=\\\"wp-image-1440\\\" decoding=\\\"async\\\" src=\\\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-system-design-2x.png?w=730\u0026amp;h=409\\\"/\u003e\",\"html_attributes\":{\"alt\":\"Gabo System Design 2x\",\"src\":\"https://storage.googleapis.com/production-eng/1/2016/03/gabo-system-design-2x.png?w=730\u0026h=409\",\"class\":[\"wp-image-1440\"],\"decoding\":\"async\"}}}},\"content\":[],\"nodeType\":\"embedded-entry-block\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[{\"type\":\"italic\"}],\"value\":\"Figure 9. Event delivery system design in which we use Cloud Pub/Sub as reliable persistent queue\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"Next step\",\"nodeType\":\"text\"}],\"nodeType\":\"heading-2\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"After events are safely persisted in Pub/Sub it’s time to export them to HDFS. To fully leverage the Google Cloud offering we decided to give a chance to Dataflow.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"},{\"data\":{},\"content\":[{\"data\":{},\"marks\":[],\"value\":\"In the last blog post from this series we’re going to go through our plan on leveraging Dataflow for this job. Stay tuned.\",\"nodeType\":\"text\"}],\"nodeType\":\"paragraph\"}],\"nodeType\":\"document\"},\"categories\":[{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"blogcategory_data\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T13:22:21.529Z\",\"updatedAt\":\"2025-05-26T09:52:46.244Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"blogCategory\"}},\"locale\":\"en-US\"},\"fields\":{\"name\":\"Data\",\"slug\":\"data\",\"has_blogs\":true}},{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"blogcategory_data-science\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T13:22:24.186Z\",\"updatedAt\":\"2025-05-26T09:52:51.710Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"blogCategory\"}},\"locale\":\"en-US\"},\"fields\":{\"name\":\"Data Science\",\"slug\":\"data-science\",\"has_blogs\":true}},{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"blogcategory_infrastructure\",\"type\":\"Entry\",\"createdAt\":\"2025-03-13T13:22:31.760Z\",\"updatedAt\":\"2025-05-26T09:53:14.772Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":4,\"revision\":2,\"contentType\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"ContentType\",\"id\":\"blogCategory\"}},\"locale\":\"en-US\"},\"fields\":{\"name\":\"Infrastructure\",\"slug\":\"infrastructure\",\"has_blogs\":true}}],\"featured_media\":{\"metadata\":{\"tags\":[],\"concepts\":[]},\"sys\":{\"space\":{\"sys\":{\"type\":\"Link\",\"linkType\":\"Space\",\"id\":\"p762jor363g1\"}},\"id\":\"attachment_065167b86d3cf061473c42f392138893\",\"type\":\"Asset\",\"createdAt\":\"2025-03-13T13:09:01.256Z\",\"updatedAt\":\"2025-03-13T13:09:01.256Z\",\"environment\":{\"sys\":{\"id\":\"spotify-engineering\",\"type\":\"Link\",\"linkType\":\"Environment\"}},\"publishedVersion\":3,\"revision\":1,\"locale\":\"en-US\"},\"fields\":{\"title\":\"Gabo System\",\"file\":{\"url\":\"//images.ctfassets.net/p762jor363g1/attachment_065167b86d3cf061473c42f392138893/f28a6eb4a75e4f6778409d68d096e714/attachment_065167b86d3cf061473c42f392138893.png\",\"details\":{\"size\":56490,\"image\":{\"width\":1473,\"height\":825}},\"fileName\":\"attachment_065167b86d3cf061473c42f392138893.png\",\"contentType\":\"image/png\"}}},\"is_sticky\":false,\"is_old\":true,\"created_at\":\"2016-03-03T12:58:23+00:00\",\"updated_at\":\"2020-07-06T17:02:51+00:00\",\"author\":\"Igor Maravić\",\"user_info\":{\"data\":{},\"content\":[],\"nodeType\":\"document\"}}}}]\n"])</script><script>self.__next_f.push([1,"22:[\"$\",\"div\",null,{\"className\":\"related-posts\",\"children\":[[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[\"$\",\"h3\",null,{\"children\":\"Related articles\"}]}],[\"$\",\"div\",null,{\"className\":\"container\",\"children\":[\"$\",\"div\",null,{\"className\":\"cards\",\"children\":[[\"$\",\"div\",\"4ode0KMR0ZNP2v8r180XjX\",{\"className\":\"post-card related\",\"children\":[[\"$\",\"div\",null,{\"className\":\"post-card__image\",\"children\":[\"$\",\"$L18\",null,{\"href\":\"/2026/9/why-spotify-is-not-using-bayesian-a-b-testing\",\"onClick\":\"$undefined\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"https://images.ctfassets.net/p762jor363g1/7MYEdTbczd11VcDP5GMLrp/755c9f5d9d2bc9362c1b3c5baf6f6df8/image3.png\",\"alt\":\"sticky\",\"width\":400,\"height\":197,\"quality\":100,\"priority\":true}]}]}],[\"$\",\"div\",null,{\"className\":\"post-card__content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"top-content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"post-card__date\",\"children\":\"Sep 8, 2026\"}],[\"$\",\"$L18\",null,{\"href\":\"/2026/9/why-spotify-is-not-using-bayesian-a-b-testing\",\"onClick\":\"$undefined\",\"children\":[\"$\",\"h4\",null,{\"className\":\"post-card__title\",\"children\":\"Why Spotify Is Not Using Bayesian A/B Testing\"}]}],[\"$\",\"div\",null,{\"className\":\"post-card__description\",\"children\":\"Clearing the confusion about what Bayesian A/B testing is.\"}],[\"$\",\"div\",null,{\"className\":\"post-card__published\",\"children\":\"$undefined\"}]]}],[\"$\",\"div\",null,{\"className\":\"post-card__tags\",\"children\":\"$undefined\"}]]}]]}],[\"$\",\"div\",\"ln7E2AsSaOEViz9VzHqlL\",{\"className\":\"post-card related\",\"children\":[[\"$\",\"div\",null,{\"className\":\"post-card__image\",\"children\":[\"$\",\"$L18\",null,{\"href\":\"/2026/8/when-can-llms-replace-humans-in-a-b-tests\",\"onClick\":\"$undefined\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"https://images.ctfassets.net/p762jor363g1/2Bz9zncYu2jdd4ZeCYIa0w/dc9a472c7cb372e9be01de06984c0009/Option_1__2_.png\",\"alt\":\"sticky\",\"width\":400,\"height\":197,\"quality\":100,\"priority\":true}]}]}],[\"$\",\"div\",null,{\"className\":\"post-card__content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"top-content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"post-card__date\",\"children\":\"Aug 13, 2026\"}],[\"$\",\"$L18\",null,{\"href\":\"/2026/8/when-can-llms-replace-humans-in-a-b-tests\",\"onClick\":\"$undefined\",\"children\":[\"$\",\"h4\",null,{\"className\":\"post-card__title\",\"children\":\"When Can LLMs Replace Humans in A/B Tests?\"}]}],[\"$\",\"div\",null,{\"className\":\"post-card__description\",\"children\":\"TL;DR: LLM predictions can stand in for human outcomes in A/B tests, but only by assumption, not by design....\"}],[\"$\",\"div\",null,{\"className\":\"post-card__published\",\"children\":\"$undefined\"}]]}],[\"$\",\"div\",null,{\"className\":\"post-card__tags\",\"children\":\"$undefined\"}]]}]]}],[\"$\",\"div\",\"2BHiaulFaaun4MVkEBakNn\",{\"className\":\"post-card related\",\"children\":[[\"$\",\"div\",null,{\"className\":\"post-card__image\",\"children\":[\"$\",\"$L18\",null,{\"href\":\"/2026/7/indexing-the-data-lake-for-online-point-queries\",\"onClick\":\"$undefined\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"https://images.ctfassets.net/p762jor363g1/69pB35ICFY1sL97klnboyf/aed20373e05048d11cdd12de739b4524/image1.png\",\"alt\":\"sticky\",\"width\":400,\"height\":197,\"quality\":100,\"priority\":true}]}]}],[\"$\",\"div\",null,{\"className\":\"post-card__content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"top-content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"post-card__date\",\"children\":\"Jul 27, 2026\"}],[\"$\",\"$L18\",null,{\"href\":\"/2026/7/indexing-the-data-lake-for-online-point-queries\",\"onClick\":\"$undefined\",\"children\":[\"$\",\"h4\",null,{\"className\":\"post-card__title\",\"children\":\"Indexing the Data Lake for Online Point Queries\"}]}],[\"$\",\"div\",null,{\"className\":\"post-card__description\",\"children\":\"Companies like Spotify need vast quantities of data accessible at low latency for online services and,...\"}],[\"$\",\"div\",null,{\"className\":\"post-card__published\",\"children\":\"$undefined\"}]]}],[\"$\",\"div\",null,{\"className\":\"post-card__tags\",\"children\":\"$undefined\"}]]}]]}],[\"$\",\"div\",\"6OwIaRIOSFOCZSHMjmLIxv\",{\"className\":\"post-card related\",\"children\":[[\"$\",\"div\",null,{\"className\":\"post-card__image\",\"children\":[\"$\",\"$L18\",null,{\"href\":\"/2026/6/encoding-your-domain-expert-the-context-layer-behind-spotifys-data-assistant\",\"onClick\":\"$undefined\",\"children\":[\"$\",\"$L1c\",null,{\"src\":\"https://images.ctfassets.net/p762jor363g1/7l2eUSuEW1R60JzrWTVCTa/c29b1e8011108d7c79a3b4d4d6f6237f/IMG_5977.png\",\"alt\":\"sticky\",\"width\":400,\"height\":197,\"quality\":100,\"priority\":true}]}]}],[\"$\",\"div\",null,{\"className\":\"post-card__content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"top-content\",\"children\":[[\"$\",\"div\",null,{\"className\":\"post-card__date\",\"children\":\"Jun 10, 2026\"}],[\"$\",\"$L18\",null,{\"href\":\"/2026/6/encoding-your-domain-expert-the-context-layer-behind-spotifys-data-assistant\",\"onClick\":\"$undefined\",\"children\":\"$L25\"}],\"$L26\",\"$L27\"]}],\"$L28\"]}]]}]]}]}]]}]\n"])</script><script>self.__next_f.push([1,"25:[\"$\",\"h4\",null,{\"className\":\"post-card__title\",\"children\":\"Encoding Your Domain Expert: The Context Layer Behind Spotify's Data Assistant\"}]\n26:[\"$\",\"div\",null,{\"className\":\"post-card__description\",\"children\":\"At Spotify, data problems used to follow a specific pattern. You'd look for the relevant dashboard, there...\"}]\n27:[\"$\",\"div\",null,{\"className\":\"post-card__published\",\"children\":\"$undefined\"}]\n28:[\"$\",\"div\",null,{\"className\":\"post-card__tags\",\"children\":\"$undefined\"}]\n"])</script></body></html> |