Files
nexus/sreweekly/articles/83/02-the-mystery-of-the-hanging-s3-downloads.html
2026-09-12 17:23:01 +08:00

635 lines
39 KiB
HTML

<html prefix='og: http://ogp.me/ns#'><head><title>The mystery of the hanging S3 downloads</title><meta name='viewport' content='width=device-width' /><style>
body {
font-family: helvetica, arial, sans-serif;
font-size: 110%;
color: black;
background: white;
margin: 0;
}
:link {
color: #009;
}
:visited {
color: #501;
}
a:active {
color: #900;
}
div.body {
width: 1040px;
margin-left: auto;
margin-right: auto;
}
div.navi {
width: 20%;
vertical-align: top;
padding: 0 2ex 0 2ex;
display: inline-block;
}
div.subbody {
width: 75%;
display: inline-block;
}
div.navi-block {
}
div.content {
vertical-align: top;
max-width: 90%;
margin-left: auto;
margin-right: auto;
line-height: 1.4em;
display: inline-block;
}
pre {
background-color: #eeeeee;
border: solid 1px #d0d0d0;
padding: 0.5em;
margin-left: 0.25em;
overflow: auto;
}
div.fp-post {
margin-bottom: 2em;
}
div.post-body {
padding-bottom: 1em;
}
div.post-head h2 {
font-size: 110%;
padding: 2px;
background-color: #ffffff;
margin-top: 0px;
border-bottom: solid;
border-width: 3px;
border-color: #004080;
}
div.post-head h2 a {
text-decoration: none;
color: #000;
}
h3.comment-head {
font-size: 110%;
padding: 2px;
background-color: #ffffff;
margin-top: 1em;
border-bottom: solid;
border-width: 3px;
border-color: #004080;
}
div.post-header {
color: #888;
}
div.post-comment-link {
font-weight: bold;
}
div.post-content {
margin-top: 1em;
}
div.navi-block {
margin-bottom: 1em;
}
div.navi-head {
background-color: #ffffff;
font-size: 110%;
font-weight: bold;
padding: 2px;
border-bottom: solid;
border-width: 3px;
border-color: #000000;
}
div.navi-body {
padding: 0.25em;
background-color: #ffffff;
}
div.top {
vertical-align: top;
width: 100%;
margin: 0;
margin-bottom: 2em;
padding: 1ex 0 1ex 0;
background-color: #246;
}
div.top div.top-body {
margin-left: auto;
margin-right: auto;
padding: 1ex;
width: 1040px;
}
div.top a {
color: #ddd;
}
span.site-name a {
font-weight: bold;
font-size: 125%;
}
div.site-navi {
padding-top: 3ex;
}
div.site-navi a {
display: inline-block;
}
div.site-navi span.divider {
padding: 0 1ex 0 1ex;
color: #66f;
}
div.comment-nil {
margin-right: 1em;
margin-left: 0.5em;
background-color: #f4f8f4;
padding: 3px;
}
div.comment-t {
margin-right: 1em;
margin-left: 0.5em;
padding: 3px;
}
div.comment-header {
margin-left: 0.25em;
margin-top: 1em;
font-weight: bold;
}
div.post-list-element {
margin-bottom: 0.5em;
}
div.post-body h2 {
font-size: 110%;
padding: 2px;
margin-top: 0px;
margin-left: 1em;
margin-right: 1em;
border-bottom: solid;
border-width: 3px;
border-color: #0060b0;
}
div.post-body h3 {
font-size: 100%;
padding: 2px;
margin-top: 0px;
margin-left: 2em;
margin-right: 2em;
border-bottom: solid;
border-width: 3px;
border-color: #4080c0;
}
div.list-item-title {
padding-bottom: 1ex;
font-size: 120%;
}
div.list-item-date {
color: #888;
display: inline-block;
vertical-align: top;
}
div.list-item-comments {
color: #888;
display: block;
padding-top: 1ex;
}
div.list-item-description {
display: inline-block;
width: 60%;
padding-left: 1ex;
vertical-align: top;
}
div.post-list-element-date {
color: #888;
}
div.post-list-element-comments {
}
blockquote {
border-left: 5px solid;
border-color: #888;
padding-left: 1ex;
margin-left: 2ex;
}
p.category-list span.category {
padding-right: 2ex;
}
sup a {
text-decoration: none;
}
div.footnotes {
font-size: 90%;
}
img.slide {
border-style: solid;
border-width: 1px;
padding: 3px;
margin-top: 2em;
margin-bottom: 2em;
}
promo {
display: block;
border: solid 3px #88a;
padding: 1ex;
background-color: #eef;
margin: 2em 3em 2em 3em;
}
@media only screen and (max-width: 1080px) {
div.navi {
display: none;
}
html {
overflow-x: hidden;
}
body {
overflow-x: hidden;
}
div.content {
width: 100%;
margin-left: 0px;
margin-right: 0px;
line-height: 1.4em;
display: block;
}
div.top div.top-body {
width: 98%;
}
div.body {
width: 98%;
max-width: 40em;
margin-left: 1ex;
margin-right: 1ex;
}
div.body img {
overflow: auto;
max-width: 100%;
}
div.subbody {
width: 100%;
}
}
</style>
<meta name='twitter:creator' content='@juhosnellman' /><meta property='og:title' content='The mystery of the hanging S3 downloads' /><meta property='og:url' content='https://www.snellman.net/blog/archive/2017-07-20-s3-mystery/' /><meta name='twitter:card' content='summary' /><meta property='og:description' content='Large downloads work from all other servers, but fail from S3. Must be a problem on their end, right?' /><meta description='Large downloads work from all other servers, but fail from S3. Must be a problem on their end, right?' /><link rel='icon' href='data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAACAAAAAgBAMAAACBVGfHAAAABGdBTUEAALGPC/xhBQAAACBjSFJNAAB6JgAAgIQAAPoAAACA6AAAdTAAAOpgAAA6mAAAF3CculE8AAAAJFBMVEUAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAD//wD///+IRkgVAAAACXRSTlMAMgoBBQYoBAKKFyCWAAAAAWJLR0QLH9fEwAAAAAd0SU1FB+ECDRUpM3gddlwAAACISURBVCjPdZFBEsMgCEWZSduMuxyhd+3GpdvPDfCUySgSpPHv/hu+IBARJahIxVVVFiANX2UBLDEyDmBO1N6YnUfxAGiPWELArY8BcO/DVlBn0AuujIExmr4h/AzE5rDEA0gOyP3bBSj3xhBX6EHLIJ5hAknBSPyf8qCgb/BbLHi9P3v+5dzdCTEd3f24qYSIAAAAJXRFWHRkYXRlOmNyZWF0ZQAyMDE3LTAyLTEzVDIyOjM4OjU3KzAxOjAw+4YYUwAAACV0RVh0ZGF0ZTptb2RpZnkAMjAxNy0wMi0xM1QyMjozODo1NyswMTowMIrboO8AAAAASUVORK5CYII=' /></head><body><link rel='alternate' type='application/rss+xml' title='RSS' href='https://www.snellman.net/blog/rss-index.xml' /><div><div class='top'><div class='top-body'><span class='site-name'><a href='https://www.snellman.net/blog/'>Juho Snellman's Weblog</a></span><div class='site-navi'><a href='https://www.snellman.net/blog/'>Home</a><span class='divider'> &ndash; </span><a href='/blog/archive/about/'>About</a><span class='divider'> &ndash; </span><a href='/blog/list/index/'>Archives</a><span class='divider'> &ndash; </span><a href='/blog/subscribe/'>Subscribe</a><span class='divider'> &ndash; </span><a href='/lastweek/'>Links</a></div></div></div><div class='body'><div class='subbody'><div class='content' valign='top'><div class='post-block'><div class='post-head'><h2><a href='https://www.snellman.net/blog/archive/2017-07-20-s3-mystery/'>The mystery of the hanging S3 downloads</a></h2></div><div class='post-body'><div class='post-header'>Posted on 2017-07-20 in <a href='/blog/list/networking/'>Networking</a></div><div class='post-content'>
<p>
A coworker was experiencing a strange problem with their Internet
connection at home. Large downloads from most sites worked
fine. The exception was that downloads from a Amazon S3 would get
up to a good speed (500Mbps), stall completely for a few seconds,
restart for a while, stall again, and eventually hang
completely. The problem seemed to be specific to S3,
downloads from generic AWS VMs were ok.
</p>
<p>
What could be going on? It shouldn't be a problem with
the ISP, or anything south of that: after all, connections to other
sites were working. It should not be a problem between the ISP
and Amazon, or there would have been problems with AWS too.
But it also seems very unlikely that S3 would
have a trivially reproducible problem causing large downloads to hang.
It's not like this is some minor use case of the service.
</p>
<p>
If it had been a problem with e.g. viewing Netflix, one might
suspect some kind of targeted traffic shaping. But an ISP
throttling or forcibly closing connections to S3 but not to AWS in
general? That's just silly talk.
</p>
<p>
The normal troubleshooting tips like reducing the MTU didn't help
either. This sounded like a fascinating networking whodunit,
so I couldn't resist butting in after hearing about it through
the grapevine.
</p>
<read-more></read-more>
<h3>The packet captures</h3>
<p>
The first step of debugging pretty much any networking problem is getting
a packet capture from as many points in the network as possible. In this
case we only had one capture point: the client machine. The problem
could not be reproduced on anything but S3, and obviously taking a capture
from S3 was not an option. Nor did we have access to any devices elsewhere
on the traffic path. <a id='fnref0'>[<a href='#fn0'>0</a>]
</p>
<p>
A superficial check of the ACK stream showed the following pattern.
The traffic would be humming along nicely, from the sequence numbers
we can see that about 57MB have already been downloaded in the first
2.5 seconds.
</p>
<pre>00:00:02.543596 client &gt; server: Flags [.], ack <b>57657817</b>
00:00:02.543623 client &gt; server: Flags [.], ack 57661318
00:00:02.543682 client &gt; server: Flags [.], ack 57667046
</pre>
<p>Then, a single packet loss occurs. We can tell from the SACK block that 1432 bytes of payload are missing. That's almost certainly a single packet.</p>
<pre><b>00:00:02.543734</b> client &gt; server: Flags [.], ack <b>57667046</b>,
options [sack 1 {<b>57668478</b>:57669910}]
</pre>
<p>After the single packet loss, more data continues to be delivered
with no problems. In the next 100ms a further 6MB gets delivered. But the
missing data never arrives.</p>
<pre>...
00:00:02.648316 client &gt; server: Flags [.], ack 57667046,
options [sack 1 {57668478:63829515}]
<b>00:00:02.648371</b> client &gt; server: Flags [.], ack 57667046,
options [sack 1 {57668478:<b>63830947</b>}]
</pre>
<p>In fact, no further ACKs are sent for 4 seconds. And even then it's not done
by one 1432 byte packet like we expected, but by two 512 byte packets and one
408 byte one. There's also a RTT-sized delay between the first and second
packets.
</p>
<pre>00:00:<b>06.751691</b> client &gt; server: Flags [.], ack <b>57667558</b>,
options [sack 1 {57668478:63830947}]
00:00:<b>06.792592</b> client &gt; server: Flags [.], ack <b>57668070</b>,
options [sack 1 {57668478:63830947}]
00:00:06.796277 client &gt; server: Flags [.], ack <b>63830947</b>
</pre>
<p>
After that, the connection continues merrily along, but the exact same thing
happens 3 seconds later.
</p>
<p>What can we tell from this? Clearly the actual server would be
retransmitting the lost packet much more quickly than with a 4 second
delay. It also would not be re-packetizing the 1432 byte packet into three
pieces. Instead what must be happening is that each retransmitted copy
is getting lost. After a few seconds RFC 4821-style path MTU probing kicks in,
and a smaller packet gets retransmitted. For some reason this retransmission
makes it through; this makes the sender believe that the path MTU has been
reduced, and it starts sending smaller packets.</p>
<p>Again this suggests there's something dodgy going on with MTUs, but as
mentioned in the beginning, reducing the MTU did not help.</p>
<p>But it also suggests a mechanism for why the connection eventually hangs
completely, rather than alternating between stalling and recovering.
There's a limit to how far
the MSS can be reduced. If nothing else, the segments will need to
have at least one byte of payload. In practice most operating systems have
a much higher limit on the MSS (something in the 80-160 byte range is
typical). If even packets of the minimum size aren't making it through,
the server can't react by sending smaller packets.</p>
<p>With the information from the ACK stream exhausted, it's time
to look at the packets in both directions. And what do you know?
We actually see the earlier retransmissions at the client, with
beautiful exponential backoff.
The packets were not lost in the network, but were silently rejected by
the client for some reason.</p>
<pre>00:00:02.685557 server &gt; client: Flags [.], seq 57667046:<b>57668478</b>, ack 4257, length 1432
00:00:02.960249 server &gt; client: Flags [.], seq 57667046:57668478, ack 4257, length 1432
00:00:03.500500 server &gt; client: Flags [.], seq 57667046:57668478, ack 4257, length 1432
00:00:04.580168 server &gt; client: Flags [.], seq 57667046:57668478, ack 4257, length 1432
00:00:06.751657 server &gt; client: Flags [.], seq 57667046:<b>57667558</b>, ack 4257, length 512
00:00:06.751691 client &gt; server: Flags [.], ack 57667558, win 65528,
options [sack 1 {57668478:63830947}]
00:00:06.792565 server &gt; client: Flags [.], seq <b>57667558:57668070</b>, ack 4257, length 512
00:00:06.792567 server &gt; client: Flags [.], seq <b>57668070:57668478</b>, ack 4257, length 408
00:00:06.792592 client &gt; server: Flags [.], ack 57668070,
options [sack 1 {57668478:63830947}]
</pre>
<p>There are really just two reasons this would happen. The IP or
TCP checksum could be wrong. But how could it be wrong for the
same packet six times in a row? That's crazy talk, the expected packet
corruption rate is more like one in a million. Alternatively
the packet is too large. But damn it, we know that's not the
problem, no matter how well this case is matching the common pattern.
Let's just have a look at the checksums, to rule it out...</p>
<pre>server &gt; client: Flags [.], cksum <b>0x0000</b> (incorrect -&gt; <b>0xd7a7</b>), seq 57667046:57668478, ack 4257, length 1432
server &gt; client: Flags [.], cksum 0x0000 (incorrect -&gt; 0xd7a7), seq 57667046:57668478, ack 4257, length 1432
server &gt; client: Flags [.], cksum 0x0000 (incorrect -&gt; 0xd7a7), seq 57667046:57668478, ack 4257, length 1432
...
</pre>
<p>Oh... Every single copy of that packet had a checksum of 0 instead of the
expected checksum of 0xd7a7. (Checksums of 0 are often not real errors,
but just artifacts of checksum offload. The packets being captured by software
before the checksum is computed by hardware.
That's not the case here; these are packets we're receiving rather than
transmitting.). And it gets crazier, when we look at the next
instance of the problem a few seconds later.</p>
<pre>server &gt; client: Flags [.], cksum 0x0000 (incorrect -&gt; 0xd7a7), seq 70927740:70928764, ack 4709, length 1024
server &gt; client: Flags [.], cksum 0x0000 (incorrect -&gt; 0xd7a7), seq 70927740:70928764, ack 4709, length 1024
server &gt; client: Flags [.], cksum 0x0000 (incorrect -&gt; 0xd7a7), seq 70927740:70928764, ack 4709, length 1024
...
</pre>
<p>It's the exact same problem, all the way down to the problem
appearing specifically with a TCP checksum of 0xd7a7. Further
analysis of the captures verified that this was a systematic
problem and not a coincidence. <b>Packets with an expected checksum of
0xd7a7 would always have the checksum replaced with
0. Packets with any other expected checksum would work just fine.</b>
<a id='fnref1'>[<a href='#fn1'>1</a>].</p>
<p>This explains why the path MTU probing temporarily fixes the problem:
the repacketized segments have different checksums, and make it through
unharmed.</p>
<h3>TCP Timestamps</h3>
<p>So, a problem internal to S3 is causing this very specific kind
of packet corruption then?</p>
<p>Not so fast! It turns out that most TCP implementations would
work around this kind of corruption by accident. The reason for
that is TCP Timestamps. And while you don't need to actually know
much about TCP Timestamps to understand this story, I have been
looking for an excuse to rant about them.
</p>
<p>With TCP Timestamps, every TCP packet will contain a TCP option
with two extra values. One of them is the sender's latest
timestamp. The other is an echo of the latest timestamp the sender
received from the other party. For example here the client is
sending the timestamp 805, and the server is echoing it back:</p>
<pre>client &gt; server: Flags [.], ack 89,
options [TS val <b>805</b> ecr 10087]
server &gt; client: Flags [P.], seq 89:450, ack 569,
options [TS val 10112 ecr <b>805</b>]
</pre>
<p>
TCP Timestamps were added to TCP very early on, for two
reasons, neither of which was very compelling in retrospect.</p>
<p>Reason number one was PAWS, Protection Against Wrapped-Around
Sequence-Numbers. The idea was that very fast connections might
require huge TCP window sizes, and minor packet reordering/duplication
might cause an old packet to be interpreted as a new packet, due to the
32 bit sequence number having wrapped around. I don't think that
world ever really arrived, and PAWS is irrelevant to practically
all TCP use cases.</p>
<p>The other original reason for timestamps was to enable TCP
senders to measure RTTs in the presence of packet loss. But this
can also be done with TCP Selective ACKs, a feature that's much
more useful in general (and thus was widely deployed a lot sooner,
despite being standardized later).
</p>
<p>In exchange for these dubious benefits, every TCP packet (both
data segments and pure control packets) is bloated by 12 bytes.
This is in contrast to something like selective ACKs, where most
packets don't grow in size. You only pay for selective ACKs when
packets are lost or reordered. I <a href='https://www.snellman.net/blog/archive/2016-12-01-quic-tou/'>think that the debuggability
of network protocols is important</a>, but with TCP you get basically
everything you need from other sources. TCP timestamps have a high
fixed cost, but give very little additional power.
</p>
<p>If TCP Timestamps suck so much, why does everyone use them
them? I don't know for sure anyone else's reasons. I ended up
implementing them purely due to an interoperability issue with the
FreeBSD TCP stack. Basically FreeBSD uses a small static receive
window for connections without TCP timestamps, while with TCP
timestamps on it'd scale the receive window up as necessary.
With connections with even a bit of latency, you needed
TCP timestamps to avoid the receive window becoming a bottleneck.
(This was <a href='https://svnweb.freebsd.org/base?view=revision&revision=316676'>fixed in FreeBSD a few months ago</a>. Yay!).</p>
<p>Now, performance of FreeBSD clients isn't a big deal for me as long as
the connections work. But you know who else uses a FreeBSD-derived
TCP stack? Apple. And when it comes to mobile networks, performance
of iOS devices is about as important as it gets. Anyone who cares about
large transfers to iOS or OS X clients must use TCP Timestamps,
no matter how distasteful they find the feature.</p>
<p><i>"But Juho, what does any of this have to do with S3?"</i>, you ask.
Well, S3 is one of those rare services that disable
timestamps. And that actually makes for a big difference
in this case. With timestamps, each retransmitted copy of a packet would use a
different timestamp value <a id='fn2'>[<a href='#fnref2'>2</a>].
And when any part of the TCP header changes, odds are that the
checksum changes as well. Even if some packets are lost due to the
having the magic checksum, at least the retransmissions will
make it through promptly.
</p>
<p>To check this theory, I asked for a test with TCP timestamps
disabled on the client. And immediately large downloads from
anywhere - even the ISP's own speedtest server - started hanging.
Success!</p>
<h3>Conclusion</h3>
<p>With this information I suggested my coworker call his ISP, and
report the problem.
He was smarter than that, and ran one more test: switching the
cable modem from router mode to bridging mode. Bam, the problem
was gone. In retrospect this makes sense: in router mode the cable
modem needs to update the checksums for each packet that pass
through the device. In bridging mode there's no NAT, so no
checksum update is needed.
</p>
<p>And that's how a dodgy cable modem caused downloads to fail with
one service, but one service only. I've seen many kinds of packet
corruption before, but never anything that was so absurdly specific.
</p>
<h3>Footnotes</h3>
<div class=footnotes>
<p>
<a id='fn0'>[<a href='#fnref0'>0</a>] There are techniques
around for routing the traffic such that we would have had a
measurement point. One would have been using something like a VPN or
a Socks proxy. But that's such a fundamental change to the traffic
pattern that it doesn't make for a very interesting test. Odds are
that the problem would just go away when you do that. The other
option would be to use a fully transparent generic TCP proxy on some
server with a public IP, have the client connect to the TCP
proxy and the proxy connect to the actual server. But setting that
up is tedious; certainly not worth doing as a first step.
</p>
<p>It's also pretty common to only have one trace point to start
with. For analysis I'd do for actual work purposes, we pretty
often have just a trace from somewhere in the middle of the
path, but nothing from the client or the server. Getting traces
from multiple points is so much trouble that we usually need
to roughly pinpoint the problem first with single-point packet
capture, and only then ask for more trace points.</p>
<p>
<a id='fn1'>[<a href='#fnref1'>1</a>] As far as I can tell 0xd7a7 has no interesting special
properties. The bytes are not printable ASCII characters. 0xd7a7
isn't a value with any special significance in another TCP header
field either. There are ways to screw up TCP checksum computations, but
I think they're mostly to do with the way 0x0 and 0xffff are both
zero values in a one's complement system.
<p>
<a id='fn2'>[<a href='#fnref2'>2</a>] Assuming sensible timestamp resolution. Not the rather unpractical 500ms tick that e.g. OpenBSD uses.
</div>
</div></div></div><p>If you liked this and want to be notified of new posts, <a href='https://twitter.com/intent/follow?screen_name=juhosnellman'>follow me on Twitter</a></p><promo>Like word games or puzzles? Try out my new word puzzle <a href='https://huewords.snellman.net'>Huewords</a></promo><div class='archive-prev-next'> Next &raquo; <a href='https://www.snellman.net/blog/archive/2017-08-19-slow-ps4-downloads/'>Why PS4 downloads are so slow</a></div><div class='archive-prev-next'> Previous &laquo; <a href='https://www.snellman.net/blog/archive/2017-07-18-wantarray/'>I don't want no 'wantarray'</a></div><a name='comments'></a><h3 class='comment-head'>Comments</h3><div class='post-area'><div class='comment-T'><div class='comment-header'>By Mark Entingh on 2017-07-20 </div><div class='comment-body'><p>amazing insight, thank you</p></div></div><div class='comment-NIL'><div class='comment-header'>By Matt Cross on 2017-07-20 </div><div class='comment-body'><p>I once worked on a CMTS (Cable Modem Termination System - the box at the cable company that talks to cable modems) on the fast-path forwarding software. We had a very similar bug that I tracked down to exactly this, and it did turn out to be a bug in our IP checksum calculation. The details are fuzzy but it had to do with handling carry bits and ones-complement arithmetic. Every router has to do this as it decrements TTL values. There was code in there that was optimized to not fully recalculate the checksum but instead "un-checksum" the old bytes from the old checksum and recalculate the checksum with the replacement bytes.</p><p> I suspect this is exactly what's going on here, just at the TCP header level as the router changes TCP port numbers.</p></div></div><div class='comment-T'><div class='comment-header'>By Will Mooney on 2017-07-20 </div><div class='comment-body'><p>Just curious as to what kind of client you were using on the client side. </p></div></div><div class='comment-NIL'><div class='comment-header'>By Vincent Bernat on 2017-07-20 </div><div class='comment-body'><p>Timestamps also enable to reuse timewait connections on Linux (through tcp_tw_reuse).</p></div></div><div class='comment-T'><div class='comment-header'>By Tim Schaller on 2017-07-20 </div><div class='comment-body'><p>Just curious, what kind of came modem was it?</p><p> Thanks.</p></div></div><div class='comment-NIL'><div class='comment-header'>By A Zepeda on 2017-07-20 </div><div class='comment-body'><p>Loved this article, good stuff</p></div></div><div class='comment-T'><div class='comment-header'>By Zan Lynx on 2017-07-20 </div><div class='comment-body'><p>Some TCP congestion algorithms rely on timestamps for measuring latency. The newest and best of these is BBR.</p></div></div><div class='comment-NIL'><div class='comment-header'>By Juho on 2017-07-20 </div><div class='comment-body'><p>Matt: Interesting! Yeah, getting the one's complement incremental checksum update computation is what I was referring to in footnote 1 (see RFC 1624 for the details of that). But it's still odd that it'd happen with this specific magic number. If it had been 0xffff replaced by 0x0000, it'd make more sense.</p><p> Will: I believe it was Chrome on Windows. But the client wouldn't matter for this, as long as it had TCP timestamps on by default. Any OS would have rejected the packets.</p><p> Tim: I don't know the exact make.</p><p> Zan: Measuring latency just isn't a good reason for timestamps. RTTs can be measured from the ACKs/SACKs with no need for TCP timestamps.</p><p> The only exception is LEDBAT, which tries to measure directional latency rather than RTTs. But it's making an assumption about just how timestamps work that's not actually in the spec. So I believe the only production use of LEDBAT doesn't actually use directional latency, but also uses RTTs. </p></div></div><div class='comment-T'><div class='comment-header'>By Dj on 2017-07-21 </div><div class='comment-body'><p>What modem model # are you testing with? Wouldn't happen to be a PUMA 6 based modem would it? You do know there is a big problem with PUMA 6 based modems right? <a href="http://www.dslreports.com/forum/r31122204-SB6190-Puma6-TCP-UDP-Network-Latency-Issue-Discussion">http://www.dslreports.com/forum/r31122204-SB6190-Puma6-TCP-UDP-Network-Latency-Issue-Discussion</a></p><p> <a href="http://www.dslreports.com/forum/r31079834-ALL-SB6190-is-a-terrible-modem-Intel-Puma-6-MaxLinear-mistake">http://www.dslreports.com/forum/r31079834-ALL-SB6190-is-a-terrible-modem-Intel-Puma-6-MaxLinear-mistake</a></p></div></div><div class='comment-NIL'><div class='comment-header'>By KRT on 2017-07-23 </div><div class='comment-body'><p>I've encountered a similar occurrence with packet handling in some professional kit before. Anything UDP that passed through without a checksum would be translated into a packet with an incorrect checksum being inserted. The vendor didn't believe me until they sent their own techs onsite and saw it with their own packet captures. Firmware was quickly issued that addressed the problem.</p></div></div><div class='comment-T'><div class='comment-header'>By osbjmg on 2017-07-25 </div><div class='comment-body'><p>Interesting story.</p><p> I understand that with these S3 downloads, you would commonly see this 0xd7a7 checksums, but it is present in all S3 downloads?</p><p> Do you know more about that datagram, and why it would be so reliable? </p></div></div><div class='comment-NIL'><div class='comment-header'>By osbjmg on 2017-07-26 </div><div class='comment-body'><p>Interesting story.</p><p> I understand that with these S3 downloads, you would commonly see this 0xd7a7 checksums, but it is present in all S3 downloads?</p><p> Do you know more about that datagram, and why it would be so reliable? </p></div></div><div class='comment-T'><div class='comment-header'>By Juho on 2017-07-26 </div><div class='comment-body'><p>osbjmg,</p><p> The checksums are effectively random, so 0xd7a7 will be present equally often in transmissions from any server. You'll get one on average every 100MB; so statistically one would be present in any sufficiently large download. It's just that on S3 the checksums would be "sticky" in retransmissions, unlike on other services. </p></div></div><div class='comment-NIL'><div class='comment-header'>By Juri Rischel Jensen on 2017-07-27 </div><div class='comment-body'><p>I'm facing a (looks like similar) problem right now, where I in the receiving end see the following:</p><p> REDACTED.22 &gt; REDACTED.52469: Flags [.], cksum 0x7ab7 (incorrect -&gt; 0x2a12), ack 656401448, win 1022, options [nop,nop,TS val 328489647 ecr 210124470], length 0</p><p> and every line reports the same checksum: 0x7ab7</p><p> The result is stalling connections (I do zfs send/recieve over SSH).</p><p> Anyone has any clue what's going on here...?</p></div></div><div class='comment-T'><div class='comment-header'>By Juri Rischel Jensen on 2017-07-27 </div><div class='comment-body'><p>Maybe I should add that I have a PFSense box in front of the sending part...</p><p> </p></div></div><div class='comment-NIL'><div class='comment-header'>By Juho on 2017-07-27 </div><div class='comment-body'><p>Juri,</p><p> Do you have the ability to take a capture on the sender or the PFSense box? These kinds of things are always a lot easier when you can see both sides of a connection. </p><p> And just to be clear (since sender/receiver could be either way), is the packet you pasted from the REDACTED.52469 host? If it was taken from REDACTED.22, that looks like just capture artifact from hardware TCP checksum offload, not the actual problem.</p><p> (This sounds like an interesting one, so if you get traces from both ends, feel free to send me an email. Contact information available at the front page of this site). </p></div></div><div class='comment-T'><div class='comment-header'>By Juri Rischel Jensen on 2017-07-28 </div><div class='comment-body'><p>Hi Juho</p><p> Thank you for your answer. After posting I did a capture on the sending end. There I got the same error (after a short while), but the checksum is different:</p><p> cksum 0x6b87 (incorrect -&gt; 0xb48b)</p><p> Every time the error occurs, the checksum is reported as 0x6b87 on the sending side, and 0x7ab7 on the receiving side.</p><p> I haven't done a capture on the pfsense box, as I haven't got access to it. But I've requested a dump.</p><p> And the packet I've pasted was captured on the receiver (REDACTED.22).</p><p> Lastly, the problem just appeared about 2 weeks ago. I've had these send/receives running for several years on the same equipment, with hundreds of TB's going through without problems.</p><p> I'll maybe take you up on your offer to take a look on the dumps. I'll try to do captures on all three points and send them to you.</p><p> Thank you.</p></div></div><div class='comment-NIL'><div class='comment-header'>By Fede on 2017-08-03 </div><div class='comment-body'><p>This is just the weirdest story ever. </p></div></div><div class='comment-T'><div class='comment-header'>By Diego on 2018-02-02 </div><div class='comment-body'><p>I'm experiencing the exact same behaviour from my Cloudways hosting. I guess they won't try to switch a cable...</p></div></div><div class='comment-NIL'><div class='comment-header'>By Nick on 2018-06-07 </div><div class='comment-body'><p>Great write up!</p><p> The CM is Hitron CODA 5482 .</p><p> Now a year later and many firmware updates since, the bug is still here.</p><p> I can't thank Juho enough for his assistance in finding the needle in the hay(tcp) stack!</p></div></div><div class='comment-T'><div class='comment-header'>By Juho on 2018-06-10 </div><div class='comment-body'><p>Thanks for the update Nick!</p><p> </p></div></div><div class='comment-NIL'><div class='comment-header'>By Mike on 2018-08-27 </div><div class='comment-body'><p>I had exactly the same problem with a Billion 7800N router and had been tearing my hair out for months over it. I replaced the router and it's fine now. I don't know how I would have diagnosed it without this page. Thanks.</p></div></div><div class='comment-T'><div class='comment-header'>By Terence on 2018-10-29 </div><div class='comment-body'><p>Thank you!!!!! This really helped me figure out a problem I would have never figured out!</p></div></div><div class='comment-NIL'><div class='comment-header'>By Mekkaz on 2019-01-18 </div><div class='comment-body'><p>Goddamn. Bro you are smart f**k!</p><p> I dont even work Networking. In fact, Networking is my weakest area despite. The most I do with packet level stuff like this is Wireshark a few things. I just landed on this page for searching if I can packet mangle and looking up [F,A]...that was 3-4 articles ago. This is a good damn website. Reading thru this joint. Im learning more and more stuff that I dont need to actually know but that is presented interestingly.</p><p> Anyway. keep up the good work. This is getting added into my favs like that Malware Analysis website. </p></div></div><div class='comment-T'><div class='comment-header'>By Mofoman on 2019-06-06 </div><div class='comment-body'><p>I ended up on this when investigating problem stalled S3 downloads on my iPad application. The files (300 - 500 MB in size) are downloaded for a while, but then they just stall. No error HTTP response, even one with an error is received from S3. </p><p> So if I understood correctly, this is not an issue that can be addressed from the application layer (meaning in my code), right? So somehow I would have to enable TCP timestamps on OS level?</p><p> Or is there anything I can do to this?</p></div></div><div class='comment-NIL'><div class='comment-header'>By Juho on 2019-06-12 </div><div class='comment-body'><p>Mofoman,</p><p> I'm afraid that enabling TCP timestamps from the client isn't possible if the server doesn't support it.</p><p> The obvious application-level fix would be to do the download in chunks. So download the file e.g. a 10MB range at a time. If a chunk appears to stall, open a new connection to S3 and restart that chunk but keep all the previous progress.</p><p> </p></div></div><div class='comment-T'><div class='comment-header'>By Anonymous on 2026-04-19 </div><div class='comment-body'><p></p></div></div></div><div id='show-post-new'><a onclick='document.getElementById("post-new").style.display="block"; document.getElementById("show-post-new").style.display="none"; ' style='text-decoration: underline'>Post comment</a></div><div class='post-new-area' id='post-new' style='display: none'><form action='/blog/post-comment/2017-07-20-s3-mystery/' method='POST'><table><tr><td valign='top'>Name</td><td><input name='AUTHOR' width='100%' /></td></tr><tr><td valign='top'>Message</td><td><textarea name='BODY' rows='10' cols='40'></textarea></td></tr><tr><td /><td><p>As an antispam measure, you need to write a super-secret password below. Today's password is "xyzzy" (without the quotes).</p></td></tr><tr><td valign='top'>Password</td><td><input name='PASSWORD' width='100%' /></td></tr><tr><td /><td><input type='submit' value='Send' /></td></tr></table></form><p><b>Formatting guidelines for comments:</b> No tags, two newlines can be used for separating paragraphs,
http://... is automatically turned into a link.</p></div></div></div><div class='navi right'><div class='navi-block'><div class='navi-head'>Links</div><div class='navi-body'><div class='post-list-element'><div class='post-list-element-title'><a href='/lastweek/page/36'>2020-07-13</a></div></div><div class='post-list-element'><div class='post-list-element-title'><a href='/lastweek/page/35'>2020-06-22</a></div></div><div class='post-list-element'><div class='post-list-element-title'><a href='/lastweek/page/34'>2020-06-15</a></div></div></div></div><div class='navi-block'><div class='navi-head'>Recent posts</div><div class='navi-body'><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2025-06-02-llms-are-cheap/'>LLMs are cheap</a></div><div class='post-list-element-date'>2025-06-02</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2023-07-25-web-integrity-api-vs-private-access-tokens/'>Web Environment Integrity vs. Private Access Tokens - They're the same thing!</a></div><div class='post-list-element-date'>2023-07-25</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2021-07-21-monorepo-atomic/'>A monorepo misconception - atomic cross-project commits</a></div><div class='post-list-element-date'>2021-07-21</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2019-05-14-procedural-puzzle-generator/'>Writing a procedural puzzle generator</a></div><div class='post-list-element-date'>2019-05-14</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2018-07-23-optimizing-breadth-first-search/'>Optimizing a breadth-first search</a></div><div class='post-list-element-date'>2018-07-23</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2017-09-04-lisp-numbers/'>Numbers and tagged pointers in early Lisp implementations</a></div><div class='post-list-element-date'>2017-09-04</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2017-08-19-slow-ps4-downloads/'>Why PS4 downloads are so slow</a></div><div class='post-list-element-date'>2017-08-19</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2017-07-20-s3-mystery/'>The mystery of the hanging S3 downloads</a></div><div class='post-list-element-date'>2017-07-20</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2017-07-18-wantarray/'>I don't want no 'wantarray'</a></div><div class='post-list-element-date'>2017-07-18</div></div><div class='post-list-element'><div class='post-list-element-title'><a href='https://www.snellman.net/blog/archive/2017-04-17-xxx-fixme/'>The origins of XXX as FIXME</a></div><div class='post-list-element-date'>2017-04-17</div></div></div></div><div class='navi-block'><div class='navi-head'>Recent comments</div><div class='navi-body'><div class='post-list-element'><div><a href='/blog/archive/2017-08-19-slow-ps4-downloads/#comments'>Why PS4 downloads are so slow</a></div><div class='post-list-element-date'>2026-08-26</div></div><div class='post-list-element'><div><a href='/blog/archive/2017-07-20-s3-mystery/#comments'>The mystery of the hanging S3 downloads</a></div><div class='post-list-element-date'>2026-04-19</div></div><div class='post-list-element'><div><a href='/blog/archive/2014-11-27-history-of-online-terra-mystica/#comments'>A brief history of Online Terra Mystica</a></div><div class='post-list-element-date'>2026-04-04</div></div></div></div></div><div></div></div></div></body></html>