diff --git a/target/linux/generic/backport-4.14/605-0001-tcp-avoid-min-RTT-overestimation-from-delayed-ACK.patch b/target/linux/generic/backport-4.14/605-0001-tcp-avoid-min-RTT-overestimation-from-delayed-ACK.patch new file mode 100644 index 0000000000..151ec1e66b --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0001-tcp-avoid-min-RTT-overestimation-from-delayed-ACK.patch @@ -0,0 +1,80 @@ +From 2a1bc5c57c50a519b02a0a4a827d943db3c0902f Mon Sep 17 00:00:00 2001 +From: Yuchung Cheng +Date: Wed, 17 Jan 2018 12:11:00 -0800 +Subject: [PATCH] tcp: avoid min-RTT overestimation from delayed ACKs + +This patch avoids having TCP sender or congestion control +overestimate the min RTT by orders of magnitude. This happens when +all the samples in the windowed filter are one-packet transfer +like small request and health-check like chit-chat, which is farily +common for applications using persistent connections. This patch +tries to conservatively labels and skip RTT samples obtained from +this type of workload. + +Signed-off-by: Yuchung Cheng +Signed-off-by: Soheil Hassas Yeganeh +Acked-by: Neal Cardwell +Acked-by: Eric Dumazet +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + net/ipv4/tcp_input.c | 23 +++++++++++++++++++++-- + 1 file changed, 21 insertions(+), 2 deletions(-) + +--- a/net/ipv4/tcp_input.c ++++ b/net/ipv4/tcp_input.c +@@ -112,6 +112,7 @@ int sysctl_tcp_invalid_ratelimit __read_ + #define FLAG_SACK_RENEGING 0x2000 /* snd_una advanced to a sacked seq */ + #define FLAG_UPDATE_TS_RECENT 0x4000 /* tcp_replace_ts_recent() */ + #define FLAG_NO_CHALLENGE_ACK 0x8000 /* do not call tcp_send_challenge_ack() */ ++#define FLAG_ACK_MAYBE_DELAYED 0x10000 /* Likely a delayed ACK */ + + #define FLAG_ACKED (FLAG_DATA_ACKED|FLAG_SYN_ACKED) + #define FLAG_NOT_DUP (FLAG_DATA|FLAG_WIN_UPDATE|FLAG_ACKED) +@@ -2938,11 +2939,18 @@ static void tcp_fastretrans_alert(struct + *rexmit = REXMIT_LOST; + } + +-static void tcp_update_rtt_min(struct sock *sk, u32 rtt_us) ++static void tcp_update_rtt_min(struct sock *sk, u32 rtt_us, const int flag) + { + struct tcp_sock *tp = tcp_sk(sk); + u32 wlen = sysctl_tcp_min_rtt_wlen * HZ; + ++ if ((flag & FLAG_ACK_MAYBE_DELAYED) && rtt_us > tcp_min_rtt(tp)) { ++ /* If the remote keeps returning delayed ACKs, eventually ++ * the min filter would pick it up and overestimate the ++ * prop. delay when it expires. Skip suspected delayed ACKs. ++ */ ++ return; ++ } + minmax_running_min(&tp->rtt_min, wlen, tcp_jiffies32, + rtt_us ? : jiffies_to_usecs(1)); + } +@@ -2982,7 +2990,7 @@ static bool tcp_ack_update_rtt(struct so + * always taken together with ACK, SACK, or TS-opts. Any negative + * values will be skipped with the seq_rtt_us < 0 check above. + */ +- tcp_update_rtt_min(sk, ca_rtt_us); ++ tcp_update_rtt_min(sk, ca_rtt_us, flag); + tcp_rtt_estimator(sk, seq_rtt_us); + tcp_set_rto(sk); + +@@ -3203,6 +3211,17 @@ static int tcp_clean_rtx_queue(struct so + if (likely(first_ackt) && !(flag & FLAG_RETRANS_DATA_ACKED)) { + seq_rtt_us = tcp_stamp_us_delta(tp->tcp_mstamp, first_ackt); + ca_rtt_us = tcp_stamp_us_delta(tp->tcp_mstamp, last_ackt); ++ ++ if (pkts_acked == 1 && last_in_flight < tp->mss_cache && ++ last_in_flight && !prior_sacked && fully_acked && ++ sack->rate->prior_delivered + 1 == tp->delivered && ++ !(flag & (FLAG_CA_ALERT | FLAG_SYN_ACKED))) { ++ /* Conservatively mark a delayed ACK. It's typically ++ * from a lone runt packet over the round trip to ++ * a receiver w/o out-of-order or CE events. ++ */ ++ flag |= FLAG_ACK_MAYBE_DELAYED; ++ } + } + if (sack->first_sackt) { + sack_rtt_us = tcp_stamp_us_delta(tp->tcp_mstamp, sack->first_sackt); diff --git a/target/linux/generic/backport-4.14/605-0002-tcp-avoid-min-R-bloat-by-skipping-R-from-delayed-K-in.patch b/target/linux/generic/backport-4.14/605-0002-tcp-avoid-min-R-bloat-by-skipping-R-from-delayed-K-in.patch new file mode 100644 index 0000000000..7c1cc326e9 --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0002-tcp-avoid-min-R-bloat-by-skipping-R-from-delayed-K-in.patch @@ -0,0 +1,60 @@ +From 6d6008138dcd01ff5489beb90b689da29d92d5b9 Mon Sep 17 00:00:00 2001 +From: Yuchung Cheng +Date: Wed, 17 Jan 2018 12:11:01 -0800 +Subject: [PATCH] tcp: avoid min RTT bloat by skipping RTT from delayed-ACK in + BBR + +A persistent connection may send tiny amount of data (e.g. health-check) +for a long period of time. BBR's windowed min RTT filter may only see +RTT samples from delayed ACKs causing BBR to grossly over-estimate +the path delay depending how much the ACK was delayed at the receiver. + +This patch skips RTT samples that are likely coming from delayed ACKs. Note +that it is possible the sender never obtains a valid measure to set the +min RTT. In this case BBR will continue to set cwnd to initial window +which seems fine because the connection is thin stream. + +Signed-off-by: Yuchung Cheng +Acked-by: Neal Cardwell +Acked-by: Soheil Hassas Yeganeh +Acked-by: Priyaranjan Jha +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + include/net/tcp.h | 1 + + net/ipv4/tcp_bbr.c | 3 ++- + net/ipv4/tcp_input.c | 1 + + 3 files changed, 4 insertions(+), 1 deletion(-) + +--- a/include/net/tcp.h ++++ b/include/net/tcp.h +@@ -998,6 +998,7 @@ struct rate_sample { + u32 prior_in_flight; /* in flight before this ACK */ + bool is_app_limited; /* is sample from packet with bubble in pipe? */ + bool is_retrans; /* is sample from retransmission? */ ++ bool is_ack_delayed; /* is this (likely) a delayed ACK? */ + }; + + struct tcp_congestion_ops { +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -769,7 +769,8 @@ static void bbr_update_min_rtt(struct so + filter_expired = after(tcp_jiffies32, + bbr->min_rtt_stamp + bbr_min_rtt_win_sec * HZ); + if (rs->rtt_us >= 0 && +- (rs->rtt_us < bbr->min_rtt_us || filter_expired)) { ++ (rs->rtt_us < bbr->min_rtt_us || ++ (filter_expired && !rs->is_ack_delayed))) { + bbr->min_rtt_us = rs->rtt_us; + bbr->min_rtt_stamp = tcp_jiffies32; + } +--- a/net/ipv4/tcp_input.c ++++ b/net/ipv4/tcp_input.c +@@ -3723,6 +3723,7 @@ static int tcp_ack(struct sock *sk, cons + + delivered = tp->delivered - delivered; /* freshly ACKed or SACKed */ + lost = tp->lost - lost; /* freshly marked lost */ ++ rs.is_ack_delayed = !!(flag & FLAG_ACK_MAYBE_DELAYED); + tcp_rate_gen(sk, delivered, lost, is_sack_reneg, sack_state.rate); + tcp_cong_control(sk, ack, delivered, flag, sack_state.rate); + tcp_xmit_recovery(sk, rexmit); diff --git a/target/linux/generic/backport-4.14/605-0003-tcp_bbr-better-deal-with-suboptimal-GSO-II.patch b/target/linux/generic/backport-4.14/605-0003-tcp_bbr-better-deal-with-suboptimal-GSO-II.patch new file mode 100644 index 0000000000..437bd91907 --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0003-tcp_bbr-better-deal-with-suboptimal-GSO-II.patch @@ -0,0 +1,138 @@ +From 1fbad3e6fe0e01f071941a5689fb5f72428ca31a Mon Sep 17 00:00:00 2001 +From: Eric Dumazet +Date: Wed, 28 Feb 2018 14:40:46 -0800 +Subject: [PATCH] tcp_bbr: better deal with suboptimal GSO (II) + +This is second part of dealing with suboptimal device gso parameters. +In first patch (350c9f484bde "tcp_bbr: better deal with suboptimal GSO") +we dealt with devices having low gso_max_segs + +Some devices lower gso_max_size from 64KB to 16 KB (r8152 is an example) + +In order to probe an optimal cwnd, we want BBR being not sensitive +to whatever GSO constraint a device can have. + +This patch removes tso_segs_goal() CC callback in favor of +min_tso_segs() for CC wanting to override sysctl_tcp_min_tso_segs + +Next patch will remove bbr->tso_segs_goal since it does not have +to be persistent. + +Signed-off-by: Eric Dumazet +Acked-by: Neal Cardwell +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + include/net/tcp.h | 6 ++---- + net/ipv4/tcp_bbr.c | 23 +++++++++++++---------- + net/ipv4/tcp_output.c | 15 ++++++++------- + 3 files changed, 23 insertions(+), 21 deletions(-) + +--- a/include/net/tcp.h ++++ b/include/net/tcp.h +@@ -551,8 +551,6 @@ __u32 cookie_v6_init_sequence(const stru + #endif + /* tcp_output.c */ + +-u32 tcp_tso_autosize(const struct sock *sk, unsigned int mss_now, +- int min_tso_segs); + void __tcp_push_pending_frames(struct sock *sk, unsigned int cur_mss, + int nonagle); + int __tcp_retransmit_skb(struct sock *sk, struct sk_buff *skb, int segs); +@@ -1025,8 +1023,8 @@ struct tcp_congestion_ops { + u32 (*undo_cwnd)(struct sock *sk); + /* hook for packet ack accounting (optional) */ + void (*pkts_acked)(struct sock *sk, const struct ack_sample *sample); +- /* suggest number of segments for each skb to transmit (optional) */ +- u32 (*tso_segs_goal)(struct sock *sk); ++ /* override sysctl_tcp_min_tso_segs */ ++ u32 (*min_tso_segs)(struct sock *sk); + /* returns the multiplier used in tcp_sndbuf_expand (optional) */ + u32 (*sndbuf_expand)(struct sock *sk); + /* call when packets are delivered to update cwnd and pacing rate, +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -261,23 +261,26 @@ static void bbr_set_pacing_rate(struct s + sk->sk_pacing_rate = rate; + } + +-/* Return count of segments we want in the skbs we send, or 0 for default. */ +-static u32 bbr_tso_segs_goal(struct sock *sk) ++/* override sysctl_tcp_min_tso_segs */ ++static u32 bbr_min_tso_segs(struct sock *sk) + { +- struct bbr *bbr = inet_csk_ca(sk); +- +- return bbr->tso_segs_goal; ++ return sk->sk_pacing_rate < (bbr_min_tso_rate >> 3) ? 1 : 2; + } + + static void bbr_set_tso_segs_goal(struct sock *sk) + { + struct tcp_sock *tp = tcp_sk(sk); + struct bbr *bbr = inet_csk_ca(sk); +- u32 min_segs; ++ u32 segs, bytes; ++ ++ /* Sort of tcp_tso_autosize() but ignoring ++ * driver provided sk_gso_max_size. ++ */ ++ bytes = min_t(u32, sk->sk_pacing_rate >> sk->sk_pacing_shift, ++ GSO_MAX_SIZE - 1 - MAX_TCP_HEADER); ++ segs = max_t(u32, bytes / tp->mss_cache, bbr_min_tso_segs(sk)); + +- min_segs = sk->sk_pacing_rate < (bbr_min_tso_rate >> 3) ? 1 : 2; +- bbr->tso_segs_goal = min(tcp_tso_autosize(sk, tp->mss_cache, min_segs), +- 0x7FU); ++ bbr->tso_segs_goal = min(segs, 0x7FU); + } + + /* Save "last known good" cwnd so we can restore it after losses or PROBE_RTT */ +@@ -941,7 +944,7 @@ static struct tcp_congestion_ops tcp_bbr + .undo_cwnd = bbr_undo_cwnd, + .cwnd_event = bbr_cwnd_event, + .ssthresh = bbr_ssthresh, +- .tso_segs_goal = bbr_tso_segs_goal, ++ .min_tso_segs = bbr_min_tso_segs, + .get_info = bbr_get_info, + .set_state = bbr_set_state, + }; +--- a/net/ipv4/tcp_output.c ++++ b/net/ipv4/tcp_output.c +@@ -1695,8 +1695,8 @@ static bool tcp_nagle_check(bool partial + /* Return how many segs we'd like on a TSO packet, + * to send one TSO packet per ms + */ +-u32 tcp_tso_autosize(const struct sock *sk, unsigned int mss_now, +- int min_tso_segs) ++static u32 tcp_tso_autosize(const struct sock *sk, unsigned int mss_now, ++ int min_tso_segs) + { + u32 bytes, segs; + +@@ -1712,7 +1712,6 @@ u32 tcp_tso_autosize(const struct sock * + + return segs; + } +-EXPORT_SYMBOL(tcp_tso_autosize); + + /* Return the number of segments we want in the skb we are transmitting. + * See if congestion control module wants to decide; otherwise, autosize. +@@ -1720,11 +1719,13 @@ EXPORT_SYMBOL(tcp_tso_autosize); + static u32 tcp_tso_segs(struct sock *sk, unsigned int mss_now) + { + const struct tcp_congestion_ops *ca_ops = inet_csk(sk)->icsk_ca_ops; +- u32 tso_segs = ca_ops->tso_segs_goal ? ca_ops->tso_segs_goal(sk) : 0; ++ u32 min_tso, tso_segs; + +- if (!tso_segs) +- tso_segs = tcp_tso_autosize(sk, mss_now, +- sysctl_tcp_min_tso_segs); ++ min_tso = ca_ops->min_tso_segs ? ++ ca_ops->min_tso_segs(sk) : ++ sysctl_tcp_min_tso_segs; ++ ++ tso_segs = tcp_tso_autosize(sk, mss_now, min_tso); + return min_t(u32, tso_segs, sk->sk_gso_max_segs); + } + diff --git a/target/linux/generic/backport-4.14/605-0004-tcp_bbr-remove-bbr-tso_segs_goal.patch b/target/linux/generic/backport-4.14/605-0004-tcp_bbr-remove-bbr-tso_segs_goal.patch new file mode 100644 index 0000000000..72f086dcc7 --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0004-tcp_bbr-remove-bbr-tso_segs_goal.patch @@ -0,0 +1,76 @@ +From 5325679a178db78f83fb069aa7bc760867e2eba6 Mon Sep 17 00:00:00 2001 +From: Eric Dumazet +Date: Wed, 28 Feb 2018 14:40:47 -0800 +Subject: [PATCH] tcp_bbr: remove bbr->tso_segs_goal + +Its value is computed then immediately used, +there is no need to store it. + +Signed-off-by: Eric Dumazet +Acked-by: Neal Cardwell +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + net/ipv4/tcp_bbr.c | 12 ++++-------- + 1 file changed, 4 insertions(+), 8 deletions(-) + +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -97,10 +97,9 @@ struct bbr { + packet_conservation:1, /* use packet conservation? */ + restore_cwnd:1, /* decided to revert cwnd to old value */ + round_start:1, /* start of packet-timed tx->ack round? */ +- tso_segs_goal:7, /* segments we want in each skb we send */ + idle_restart:1, /* restarting after idle? */ + probe_rtt_round_done:1, /* a BBR_PROBE_RTT round at 4 pkts? */ +- unused:5, ++ unused:12, + lt_is_sampling:1, /* taking long-term ("LT") samples now? */ + lt_rtt_cnt:7, /* round trips in long-term interval */ + lt_use_bw:1; /* use lt_bw as our bw estimate? */ +@@ -267,10 +266,9 @@ static u32 bbr_min_tso_segs(struct sock + return sk->sk_pacing_rate < (bbr_min_tso_rate >> 3) ? 1 : 2; + } + +-static void bbr_set_tso_segs_goal(struct sock *sk) ++static u32 bbr_tso_segs_goal(struct sock *sk) + { + struct tcp_sock *tp = tcp_sk(sk); +- struct bbr *bbr = inet_csk_ca(sk); + u32 segs, bytes; + + /* Sort of tcp_tso_autosize() but ignoring +@@ -280,7 +278,7 @@ static void bbr_set_tso_segs_goal(struct + GSO_MAX_SIZE - 1 - MAX_TCP_HEADER); + segs = max_t(u32, bytes / tp->mss_cache, bbr_min_tso_segs(sk)); + +- bbr->tso_segs_goal = min(segs, 0x7FU); ++ return min(segs, 0x7FU); + } + + /* Save "last known good" cwnd so we can restore it after losses or PROBE_RTT */ +@@ -351,7 +349,7 @@ static u32 bbr_target_cwnd(struct sock * + cwnd = (((w * gain) >> BBR_SCALE) + BW_UNIT - 1) / BW_UNIT; + + /* Allow enough full-sized skbs in flight to utilize end systems. */ +- cwnd += 3 * bbr->tso_segs_goal; ++ cwnd += 3 * bbr_tso_segs_goal(sk); + + /* Reduce delayed ACKs by rounding up cwnd to the next even number. */ + cwnd = (cwnd + 1) & ~1U; +@@ -832,7 +830,6 @@ static void bbr_main(struct sock *sk, co + + bw = bbr_bw(sk); + bbr_set_pacing_rate(sk, bw, bbr->pacing_gain); +- bbr_set_tso_segs_goal(sk); + bbr_set_cwnd(sk, rs, rs->acked_sacked, bw, bbr->cwnd_gain); + } + +@@ -842,7 +839,6 @@ static void bbr_init(struct sock *sk) + struct bbr *bbr = inet_csk_ca(sk); + + bbr->prior_cwnd = 0; +- bbr->tso_segs_goal = 0; /* default segs per skb until first ACK */ + bbr->rtt_cnt = 0; + bbr->next_rtt_delivered = 0; + bbr->prev_ca_state = TCP_CA_Open; diff --git a/target/linux/generic/backport-4.14/605-0005-net-tcp_bbr-set-tp-snd_ssthresh-to-BDP-upon-STARTUP-exit.patch b/target/linux/generic/backport-4.14/605-0005-net-tcp_bbr-set-tp-snd_ssthresh-to-BDP-upon-STARTUP-exit.patch new file mode 100644 index 0000000000..16a98e0e96 --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0005-net-tcp_bbr-set-tp-snd_ssthresh-to-BDP-upon-STARTUP-exit.patch @@ -0,0 +1,49 @@ +From 1f755aeceaf743d15530e31854395a185c33901e Mon Sep 17 00:00:00 2001 +From: Yousuk Seung +Date: Fri, 16 Mar 2018 10:51:49 -0700 +Subject: [PATCH] net-tcp_bbr: set tp->snd_ssthresh to BDP upon STARTUP exit + +Set tp->snd_ssthresh to BDP upon STARTUP exit. This allows us +to check if a BBR flow exited STARTUP and the BDP at the +time of STARTUP exit with SCM_TIMESTAMPING_OPT_STATS. Since BBR does not +use snd_ssthresh this fix has no impact on BBR's behavior. + +Signed-off-by: Yousuk Seung +Signed-off-by: Neal Cardwell +Signed-off-by: Priyaranjan Jha +Signed-off-by: Soheil Hassas Yeganeh +Signed-off-by: Yuchung Cheng +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + net/ipv4/tcp_bbr.c | 5 ++++- + 1 file changed, 4 insertions(+), 1 deletion(-) + +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -734,6 +734,8 @@ static void bbr_check_drain(struct sock + bbr->mode = BBR_DRAIN; /* drain queue we created */ + bbr->pacing_gain = bbr_drain_gain; /* pace slow to drain */ + bbr->cwnd_gain = bbr_high_gain; /* maintain cwnd */ ++ tcp_sk(sk)->snd_ssthresh = ++ bbr_target_cwnd(sk, bbr_max_bw(sk), BBR_UNIT); + } /* fall through to check if in-flight is already small: */ + if (bbr->mode == BBR_DRAIN && + tcp_packets_in_flight(tcp_sk(sk)) <= +@@ -839,6 +841,7 @@ static void bbr_init(struct sock *sk) + struct bbr *bbr = inet_csk_ca(sk); + + bbr->prior_cwnd = 0; ++ tp->snd_ssthresh = TCP_INFINITE_SSTHRESH; + bbr->rtt_cnt = 0; + bbr->next_rtt_delivered = 0; + bbr->prev_ca_state = TCP_CA_Open; +@@ -891,7 +894,7 @@ static u32 bbr_undo_cwnd(struct sock *sk + static u32 bbr_ssthresh(struct sock *sk) + { + bbr_save_cwnd(sk); +- return TCP_INFINITE_SSTHRESH; /* BBR does not use ssthresh */ ++ return tcp_sk(sk)->snd_ssthresh; + } + + static size_t bbr_get_info(struct sock *sk, u32 ext, int *attr, diff --git a/target/linux/generic/backport-4.14/605-0006-tcp_bbr-fix-bbr-pacing-rate-for-internal-pacing.patch b/target/linux/generic/backport-4.14/605-0006-tcp_bbr-fix-bbr-pacing-rate-for-internal-pacing.patch new file mode 100644 index 0000000000..50fb35efe1 --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0006-tcp_bbr-fix-bbr-pacing-rate-for-internal-pacing.patch @@ -0,0 +1,88 @@ +From 877916ac4797bda7b8b0674d6804bf1200268a2d Mon Sep 17 00:00:00 2001 +From: Eric Dumazet +Date: Wed, 20 Jun 2018 16:07:35 -0400 +Subject: [PATCH] tcp_bbr: fix bbr pacing rate for internal pacing + +This commit makes BBR use only the MSS (without any headers) to +calculate pacing rates when internal TCP-layer pacing is used. + +This is necessary to achieve the correct pacing behavior in this case, +since tcp_internal_pacing() uses only the payload length to calculate +pacing delays. + +Signed-off-by: Kevin Yang +Signed-off-by: Eric Dumazet +Reviewed-by: Neal Cardwell +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + include/net/tcp.h | 11 +++++++++++ + net/ipv4/tcp_bbr.c | 6 +++++- + net/ipv4/tcp_output.c | 14 -------------- + 3 files changed, 16 insertions(+), 15 deletions(-) + +--- a/include/net/tcp.h ++++ b/include/net/tcp.h +@@ -1233,6 +1233,17 @@ static inline bool tcp_is_cwnd_limited(c + return tp->is_cwnd_limited; + } + ++/* BBR congestion control needs pacing. ++ * Same remark for SO_MAX_PACING_RATE. ++ * sch_fq packet scheduler is efficiently handling pacing, ++ * but is not always installed/used. ++ * Return true if TCP stack should pace packets itself. ++ */ ++static inline bool tcp_needs_internal_pacing(const struct sock *sk) ++{ ++ return smp_load_acquire(&sk->sk_pacing_status) == SK_PACING_NEEDED; ++} ++ + /* Something is really bad, we could not queue an additional packet, + * because qdisc is full or receiver sent a 0 window. + * We do not want to add fuel to the fire, or abort too early, +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -205,7 +205,11 @@ static u32 bbr_bw(const struct sock *sk) + */ + static u64 bbr_rate_bytes_per_sec(struct sock *sk, u64 rate, int gain) + { +- rate *= tcp_mss_to_mtu(sk, tcp_sk(sk)->mss_cache); ++ unsigned int mss = tcp_sk(sk)->mss_cache; ++ ++ if (!tcp_needs_internal_pacing(sk)) ++ mss = tcp_mss_to_mtu(sk, mss); ++ rate *= mss; + rate *= gain; + rate >>= BBR_SCALE; + rate *= USEC_PER_SEC; +--- a/net/ipv4/tcp_output.c ++++ b/net/ipv4/tcp_output.c +@@ -949,17 +949,6 @@ enum hrtimer_restart tcp_pace_kick(struc + return HRTIMER_NORESTART; + } + +-/* BBR congestion control needs pacing. +- * Same remark for SO_MAX_PACING_RATE. +- * sch_fq packet scheduler is efficiently handling pacing, +- * but is not always installed/used. +- * Return true if TCP stack should pace packets itself. +- */ +-static bool tcp_needs_internal_pacing(const struct sock *sk) +-{ +- return smp_load_acquire(&sk->sk_pacing_status) == SK_PACING_NEEDED; +-} +- + static void tcp_internal_pacing(struct sock *sk, const struct sk_buff *skb) + { + u64 len_ns; +@@ -971,9 +960,6 @@ static void tcp_internal_pacing(struct s + if (!rate || rate == ~0U) + return; + +- /* Should account for header sizes as sch_fq does, +- * but lets make things simple. +- */ + len_ns = (u64)skb->len * NSEC_PER_SEC; + do_div(len_ns, rate); + hrtimer_start(&tcp_sk(sk)->pacing_timer, diff --git a/target/linux/generic/backport-4.14/605-0007-tcp_bbr-add-bbr_check_probe_rtt_done-helper.patch b/target/linux/generic/backport-4.14/605-0007-tcp_bbr-add-bbr_check_probe_rtt_done-helper.patch new file mode 100644 index 0000000000..4f91ddf854 --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0007-tcp_bbr-add-bbr_check_probe_rtt_done-helper.patch @@ -0,0 +1,98 @@ +From 092d9804a6479f34102f1017f14232a7731ef345 Mon Sep 17 00:00:00 2001 +From: Kevin Yang +Date: Wed, 22 Aug 2018 17:43:14 -0400 +Subject: [PATCH] tcp_bbr: add bbr_check_probe_rtt_done() helper + +This patch add a helper function bbr_check_probe_rtt_done() to + 1. check the condition to see if bbr should exit probe_rtt mode; + 2. process the logic of exiting probe_rtt mode. + +Fixes: 0f8782ea1497 ("tcp_bbr: add BBR congestion control") +Signed-off-by: Kevin Yang +Signed-off-by: Neal Cardwell +Signed-off-by: Yuchung Cheng +Reviewed-by: Soheil Hassas Yeganeh +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + net/ipv4/tcp_bbr.c | 34 ++++++++++++++++++---------------- + 1 file changed, 18 insertions(+), 16 deletions(-) + +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -95,11 +95,10 @@ struct bbr { + u32 mode:3, /* current bbr_mode in state machine */ + prev_ca_state:3, /* CA state on previous ACK */ + packet_conservation:1, /* use packet conservation? */ +- restore_cwnd:1, /* decided to revert cwnd to old value */ + round_start:1, /* start of packet-timed tx->ack round? */ + idle_restart:1, /* restarting after idle? */ + probe_rtt_round_done:1, /* a BBR_PROBE_RTT round at 4 pkts? */ +- unused:12, ++ unused:13, + lt_is_sampling:1, /* taking long-term ("LT") samples now? */ + lt_rtt_cnt:7, /* round trips in long-term interval */ + lt_use_bw:1; /* use lt_bw as our bw estimate? */ +@@ -396,17 +395,11 @@ static bool bbr_set_cwnd_to_recover_or_r + cwnd = tcp_packets_in_flight(tp) + acked; + } else if (prev_state >= TCP_CA_Recovery && state < TCP_CA_Recovery) { + /* Exiting loss recovery; restore cwnd saved before recovery. */ +- bbr->restore_cwnd = 1; ++ cwnd = max(cwnd, bbr->prior_cwnd); + bbr->packet_conservation = 0; + } + bbr->prev_ca_state = state; + +- if (bbr->restore_cwnd) { +- /* Restore cwnd after exiting loss recovery or PROBE_RTT. */ +- cwnd = max(cwnd, bbr->prior_cwnd); +- bbr->restore_cwnd = 0; +- } +- + if (bbr->packet_conservation) { + *new_cwnd = max(cwnd, tcp_packets_in_flight(tp) + acked); + return true; /* yes, using packet conservation */ +@@ -747,6 +740,20 @@ static void bbr_check_drain(struct sock + bbr_reset_probe_bw_mode(sk); /* we estimate queue is drained */ + } + ++static void bbr_check_probe_rtt_done(struct sock *sk) ++{ ++ struct tcp_sock *tp = tcp_sk(sk); ++ struct bbr *bbr = inet_csk_ca(sk); ++ ++ if (!(bbr->probe_rtt_done_stamp && ++ after(tcp_jiffies32, bbr->probe_rtt_done_stamp))) ++ return; ++ ++ bbr->min_rtt_stamp = tcp_jiffies32; /* wait a while until PROBE_RTT */ ++ tp->snd_cwnd = max(tp->snd_cwnd, bbr->prior_cwnd); ++ bbr_reset_mode(sk); ++} ++ + /* The goal of PROBE_RTT mode is to have BBR flows cooperatively and + * periodically drain the bottleneck queue, to converge to measure the true + * min_rtt (unloaded propagation delay). This allows the flows to keep queues +@@ -805,12 +812,8 @@ static void bbr_update_min_rtt(struct so + } else if (bbr->probe_rtt_done_stamp) { + if (bbr->round_start) + bbr->probe_rtt_round_done = 1; +- if (bbr->probe_rtt_round_done && +- after(tcp_jiffies32, bbr->probe_rtt_done_stamp)) { +- bbr->min_rtt_stamp = tcp_jiffies32; +- bbr->restore_cwnd = 1; /* snap to prior_cwnd */ +- bbr_reset_mode(sk); +- } ++ if (bbr->probe_rtt_round_done) ++ bbr_check_probe_rtt_done(sk); + } + } + /* Restart after idle ends only once we process a new S/ACK for data */ +@@ -861,7 +864,6 @@ static void bbr_init(struct sock *sk) + bbr->has_seen_rtt = 0; + bbr_init_pacing_rate_from_rtt(sk); + +- bbr->restore_cwnd = 0; + bbr->round_start = 0; + bbr->idle_restart = 0; + bbr->full_bw_reached = 0; diff --git a/target/linux/generic/backport-4.14/605-0008-tcp_bbr-in-restart-from-idle,-see-if-we-should-exit.patch b/target/linux/generic/backport-4.14/605-0008-tcp_bbr-in-restart-from-idle,-see-if-we-should-exit.patch new file mode 100644 index 0000000000..d1ee051d28 --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0008-tcp_bbr-in-restart-from-idle,-see-if-we-should-exit.patch @@ -0,0 +1,43 @@ +From 2c42bf86a0b5fd29a117b13d8f89f77347c53d51 Mon Sep 17 00:00:00 2001 +From: Kevin Yang +Date: Wed, 22 Aug 2018 17:43:15 -0400 +Subject: [PATCH] tcp_bbr: in restart from idle, see if we should exit + PROBE_RTT + +This patch fix the case where BBR does not exit PROBE_RTT mode when +it restarts from idle. When BBR restarts from idle and if BBR is in +PROBE_RTT mode, BBR should check if it's time to exit PROBE_RTT. If +yes, then BBR should exit PROBE_RTT mode and restore the cwnd to its +full value. + +Fixes: 0f8782ea1497 ("tcp_bbr: add BBR congestion control") +Signed-off-by: Kevin Yang +Signed-off-by: Neal Cardwell +Reviewed-by: Yuchung Cheng +Reviewed-by: Soheil Hassas Yeganeh +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + net/ipv4/tcp_bbr.c | 4 ++++ + 1 file changed, 4 insertions(+) + +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -174,6 +174,8 @@ static const u32 bbr_lt_bw_diff = 4000 / + /* If we estimate we're policed, use lt_bw for this many round trips: */ + static const u32 bbr_lt_bw_max_rtts = 48; + ++static void bbr_check_probe_rtt_done(struct sock *sk); ++ + /* Do we estimate that STARTUP filled the pipe? */ + static bool bbr_full_bw_reached(const struct sock *sk) + { +@@ -308,6 +310,8 @@ static void bbr_cwnd_event(struct sock * + */ + if (bbr->mode == BBR_PROBE_BW) + bbr_set_pacing_rate(sk, bbr_bw(sk), BBR_UNIT); ++ else if (bbr->mode == BBR_PROBE_RTT) ++ bbr_check_probe_rtt_done(sk); + } + } + diff --git a/target/linux/generic/backport-4.14/605-0009-tcp_bbr-apply-PROBE_RTT-cwnd-cap-even-if-acked==0.patch b/target/linux/generic/backport-4.14/605-0009-tcp_bbr-apply-PROBE_RTT-cwnd-cap-even-if-acked==0.patch new file mode 100644 index 0000000000..7a6f6f072d --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0009-tcp_bbr-apply-PROBE_RTT-cwnd-cap-even-if-acked==0.patch @@ -0,0 +1,41 @@ +From 41c14f81721f510b94b75276bcb47b2c3de18cfc Mon Sep 17 00:00:00 2001 +From: Kevin Yang +Date: Wed, 22 Aug 2018 17:43:16 -0400 +Subject: [PATCH] tcp_bbr: apply PROBE_RTT cwnd cap even if acked==0 + +This commit fixes a corner case where TCP BBR would enter PROBE_RTT +mode but not reduce its cwnd. If a TCP receiver ACKed less than one +full segment, the number of delivered/acked packets was 0, so that +bbr_set_cwnd() would short-circuit and exit early, without cutting +cwnd to the value we want for PROBE_RTT. + +The fix is to instead make sure that even when 0 full packets are +ACKed, we do apply all the appropriate caps, including the cap that +applies in PROBE_RTT mode. + +Fixes: 0f8782ea1497 ("tcp_bbr: add BBR congestion control") +Signed-off-by: Kevin Yang +Signed-off-by: Neal Cardwell +Reviewed-by: Yuchung Cheng +Reviewed-by: Soheil Hassas Yeganeh +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + net/ipv4/tcp_bbr.c | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) + +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -420,10 +420,10 @@ static void bbr_set_cwnd(struct sock *sk + { + struct tcp_sock *tp = tcp_sk(sk); + struct bbr *bbr = inet_csk_ca(sk); +- u32 cwnd = 0, target_cwnd = 0; ++ u32 cwnd = tp->snd_cwnd, target_cwnd = 0; + + if (!acked) +- return; ++ goto done; /* no packet fully ACKed; just apply caps */ + + if (bbr_set_cwnd_to_recover_or_restore(sk, rs, acked, &cwnd)) + goto done; diff --git a/target/linux/generic/backport-4.14/605-0010-tcp_bbr-centralize-code-to-set-gains.patch b/target/linux/generic/backport-4.14/605-0010-tcp_bbr-centralize-code-to-set-gains.patch new file mode 100644 index 0000000000..2b91a84c84 --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0010-tcp_bbr-centralize-code-to-set-gains.patch @@ -0,0 +1,112 @@ +From b4d63acbad2e0f829f720b93211321ef381fb2a2 Mon Sep 17 00:00:00 2001 +From: Neal Cardwell +Date: Tue, 16 Oct 2018 20:16:45 -0400 +Subject: [PATCH] tcp_bbr: centralize code to set gains + +Centralize the code that sets gains used for computing cwnd and pacing +rate. This simplifies the code and makes it easier to change the state +machine or (in the future) dynamically change the gain values and +ensure that the correct gain values are always used. + +Signed-off-by: Neal Cardwell +Signed-off-by: Yuchung Cheng +Signed-off-by: Soheil Hassas Yeganeh +Signed-off-by: Priyaranjan Jha +Signed-off-by: Eric Dumazet +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + net/ipv4/tcp_bbr.c | 40 ++++++++++++++++++++++++++++++---------- + 1 file changed, 30 insertions(+), 10 deletions(-) + +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -487,8 +487,6 @@ static void bbr_advance_cycle_phase(stru + + bbr->cycle_idx = (bbr->cycle_idx + 1) & (CYCLE_LEN - 1); + bbr->cycle_mstamp = tp->delivered_mstamp; +- bbr->pacing_gain = bbr->lt_use_bw ? BBR_UNIT : +- bbr_pacing_gain[bbr->cycle_idx]; + } + + /* Gain cycling: cycle pacing gain to converge to fair share of available bw. */ +@@ -506,8 +504,6 @@ static void bbr_reset_startup_mode(struc + struct bbr *bbr = inet_csk_ca(sk); + + bbr->mode = BBR_STARTUP; +- bbr->pacing_gain = bbr_high_gain; +- bbr->cwnd_gain = bbr_high_gain; + } + + static void bbr_reset_probe_bw_mode(struct sock *sk) +@@ -515,8 +511,6 @@ static void bbr_reset_probe_bw_mode(stru + struct bbr *bbr = inet_csk_ca(sk); + + bbr->mode = BBR_PROBE_BW; +- bbr->pacing_gain = BBR_UNIT; +- bbr->cwnd_gain = bbr_cwnd_gain; + bbr->cycle_idx = CYCLE_LEN - 1 - prandom_u32_max(bbr_cycle_rand); + bbr_advance_cycle_phase(sk); /* flip to next phase of gain cycle */ + } +@@ -733,8 +727,6 @@ static void bbr_check_drain(struct sock + + if (bbr->mode == BBR_STARTUP && bbr_full_bw_reached(sk)) { + bbr->mode = BBR_DRAIN; /* drain queue we created */ +- bbr->pacing_gain = bbr_drain_gain; /* pace slow to drain */ +- bbr->cwnd_gain = bbr_high_gain; /* maintain cwnd */ + tcp_sk(sk)->snd_ssthresh = + bbr_target_cwnd(sk, bbr_max_bw(sk), BBR_UNIT); + } /* fall through to check if in-flight is already small: */ +@@ -796,8 +788,6 @@ static void bbr_update_min_rtt(struct so + if (bbr_probe_rtt_mode_ms > 0 && filter_expired && + !bbr->idle_restart && bbr->mode != BBR_PROBE_RTT) { + bbr->mode = BBR_PROBE_RTT; /* dip, drain queue */ +- bbr->pacing_gain = BBR_UNIT; +- bbr->cwnd_gain = BBR_UNIT; + bbr_save_cwnd(sk); /* note cwnd so we can restore it */ + bbr->probe_rtt_done_stamp = 0; + } +@@ -825,6 +815,35 @@ static void bbr_update_min_rtt(struct so + bbr->idle_restart = 0; + } + ++static void bbr_update_gains(struct sock *sk) ++{ ++ struct bbr *bbr = inet_csk_ca(sk); ++ ++ switch (bbr->mode) { ++ case BBR_STARTUP: ++ bbr->pacing_gain = bbr_high_gain; ++ bbr->cwnd_gain = bbr_high_gain; ++ break; ++ case BBR_DRAIN: ++ bbr->pacing_gain = bbr_drain_gain; /* slow, to drain */ ++ bbr->cwnd_gain = bbr_high_gain; /* keep cwnd */ ++ break; ++ case BBR_PROBE_BW: ++ bbr->pacing_gain = (bbr->lt_use_bw ? ++ BBR_UNIT : ++ bbr_pacing_gain[bbr->cycle_idx]); ++ bbr->cwnd_gain = bbr_cwnd_gain; ++ break; ++ case BBR_PROBE_RTT: ++ bbr->pacing_gain = BBR_UNIT; ++ bbr->cwnd_gain = BBR_UNIT; ++ break; ++ default: ++ WARN_ONCE(1, "BBR bad mode: %u\n", bbr->mode); ++ break; ++ } ++} ++ + static void bbr_update_model(struct sock *sk, const struct rate_sample *rs) + { + bbr_update_bw(sk, rs); +@@ -832,6 +851,7 @@ static void bbr_update_model(struct sock + bbr_check_full_bw_reached(sk, rs); + bbr_check_drain(sk, rs); + bbr_update_min_rtt(sk, rs); ++ bbr_update_gains(sk); + } + + static void bbr_main(struct sock *sk, const struct rate_sample *rs) diff --git a/target/linux/generic/backport-4.14/605-0011-tcp_bbr-refactor-bbr_target_cwnd-for-general-inflight-pro.patch b/target/linux/generic/backport-4.14/605-0011-tcp_bbr-refactor-bbr_target_cwnd-for-general-inflight-pro.patch new file mode 100644 index 0000000000..5de70fc79e --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0011-tcp_bbr-refactor-bbr_target_cwnd-for-general-inflight-pro.patch @@ -0,0 +1,144 @@ +From cc9f592b4e2645119629622d6cb90d46a95cec74 Mon Sep 17 00:00:00 2001 +From: Priyaranjan Jha +Date: Wed, 23 Jan 2019 12:04:53 -0800 +Subject: [PATCH] tcp_bbr: refactor bbr_target_cwnd() for general inflight + provisioning + +Because bbr_target_cwnd() is really a general-purpose BBR helper for +computing some volume of inflight data as a function of the estimated +BDP, refactor it into following helper functions: +- bbr_bdp() +- bbr_quantization_budget() +- bbr_inflight() + +Signed-off-by: Priyaranjan Jha +Signed-off-by: Neal Cardwell +Signed-off-by: Yuchung Cheng +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + net/ipv4/tcp_bbr.c | 60 ++++++++++++++++++++++++++++++---------------- + 1 file changed, 39 insertions(+), 21 deletions(-) + +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -315,30 +315,19 @@ static void bbr_cwnd_event(struct sock * + } + } + +-/* Find target cwnd. Right-size the cwnd based on min RTT and the +- * estimated bottleneck bandwidth: ++/* Calculate bdp based on min RTT and the estimated bottleneck bandwidth: + * +- * cwnd = bw * min_rtt * gain = BDP * gain ++ * bdp = bw * min_rtt * gain + * + * The key factor, gain, controls the amount of queue. While a small gain + * builds a smaller queue, it becomes more vulnerable to noise in RTT + * measurements (e.g., delayed ACKs or other ACK compression effects). This + * noise may cause BBR to under-estimate the rate. +- * +- * To achieve full performance in high-speed paths, we budget enough cwnd to +- * fit full-sized skbs in-flight on both end hosts to fully utilize the path: +- * - one skb in sending host Qdisc, +- * - one skb in sending host TSO/GSO engine +- * - one skb being received by receiver host LRO/GRO/delayed-ACK engine +- * Don't worry, at low rates (bbr_min_tso_rate) this won't bloat cwnd because +- * in such cases tso_segs_goal is 1. The minimum cwnd is 4 packets, +- * which allows 2 outstanding 2-packet sequences, to try to keep pipe +- * full even with ACK-every-other-packet delayed ACKs. + */ +-static u32 bbr_target_cwnd(struct sock *sk, u32 bw, int gain) ++static u32 bbr_bdp(struct sock *sk, u32 bw, int gain) + { + struct bbr *bbr = inet_csk_ca(sk); +- u32 cwnd; ++ u32 bdp; + u64 w; + + /* If we've never had a valid RTT sample, cap cwnd at the initial +@@ -353,7 +342,24 @@ static u32 bbr_target_cwnd(struct sock * + w = (u64)bw * bbr->min_rtt_us; + + /* Apply a gain to the given value, then remove the BW_SCALE shift. */ +- cwnd = (((w * gain) >> BBR_SCALE) + BW_UNIT - 1) / BW_UNIT; ++ bdp = (((w * gain) >> BBR_SCALE) + BW_UNIT - 1) / BW_UNIT; ++ ++ return bdp; ++} ++ ++/* To achieve full performance in high-speed paths, we budget enough cwnd to ++ * fit full-sized skbs in-flight on both end hosts to fully utilize the path: ++ * - one skb in sending host Qdisc, ++ * - one skb in sending host TSO/GSO engine ++ * - one skb being received by receiver host LRO/GRO/delayed-ACK engine ++ * Don't worry, at low rates (bbr_min_tso_rate) this won't bloat cwnd because ++ * in such cases tso_segs_goal is 1. The minimum cwnd is 4 packets, ++ * which allows 2 outstanding 2-packet sequences, to try to keep pipe ++ * full even with ACK-every-other-packet delayed ACKs. ++ */ ++static u32 bbr_quantization_budget(struct sock *sk, u32 cwnd, int gain) ++{ ++ struct bbr *bbr = inet_csk_ca(sk); + + /* Allow enough full-sized skbs in flight to utilize end systems. */ + cwnd += 3 * bbr_tso_segs_goal(sk); +@@ -368,6 +374,17 @@ static u32 bbr_target_cwnd(struct sock * + return cwnd; + } + ++/* Find inflight based on min RTT and the estimated bottleneck bandwidth. */ ++static u32 bbr_inflight(struct sock *sk, u32 bw, int gain) ++{ ++ u32 inflight; ++ ++ inflight = bbr_bdp(sk, bw, gain); ++ inflight = bbr_quantization_budget(sk, inflight, gain); ++ ++ return inflight; ++} ++ + /* An optimization in BBR to reduce losses: On the first round of recovery, we + * follow the packet conservation principle: send P packets per P packets acked. + * After that, we slow-start and send at most 2*P packets per P packets acked. +@@ -429,7 +446,8 @@ static void bbr_set_cwnd(struct sock *sk + goto done; + + /* If we're below target cwnd, slow start cwnd toward target cwnd. */ +- target_cwnd = bbr_target_cwnd(sk, bw, gain); ++ target_cwnd = bbr_bdp(sk, bw, gain); ++ target_cwnd = bbr_quantization_budget(sk, target_cwnd, gain); + if (bbr_full_bw_reached(sk)) /* only cut cwnd if we filled the pipe */ + cwnd = min(cwnd + acked, target_cwnd); + else if (cwnd < target_cwnd || tp->delivered < TCP_INIT_CWND) +@@ -470,14 +488,14 @@ static bool bbr_is_next_cycle_phase(stru + if (bbr->pacing_gain > BBR_UNIT) + return is_full_length && + (rs->losses || /* perhaps pacing_gain*BDP won't fit */ +- inflight >= bbr_target_cwnd(sk, bw, bbr->pacing_gain)); ++ inflight >= bbr_inflight(sk, bw, bbr->pacing_gain)); + + /* A pacing_gain < 1.0 tries to drain extra queue we added if bw + * probing didn't find more bw. If inflight falls to match BDP then we + * estimate queue is drained; persisting would underutilize the pipe. + */ + return is_full_length || +- inflight <= bbr_target_cwnd(sk, bw, BBR_UNIT); ++ inflight <= bbr_inflight(sk, bw, BBR_UNIT); + } + + static void bbr_advance_cycle_phase(struct sock *sk) +@@ -728,11 +746,11 @@ static void bbr_check_drain(struct sock + if (bbr->mode == BBR_STARTUP && bbr_full_bw_reached(sk)) { + bbr->mode = BBR_DRAIN; /* drain queue we created */ + tcp_sk(sk)->snd_ssthresh = +- bbr_target_cwnd(sk, bbr_max_bw(sk), BBR_UNIT); ++ bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT); + } /* fall through to check if in-flight is already small: */ + if (bbr->mode == BBR_DRAIN && + tcp_packets_in_flight(tcp_sk(sk)) <= +- bbr_target_cwnd(sk, bbr_max_bw(sk), BBR_UNIT)) ++ bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT)) + bbr_reset_probe_bw_mode(sk); /* we estimate queue is drained */ + } + diff --git a/target/linux/generic/backport-4.14/605-0012-tcp_bbr-adapt-cwnd-based-on-ack-aggregation-estimation.patch b/target/linux/generic/backport-4.14/605-0012-tcp_bbr-adapt-cwnd-based-on-ack-aggregation-estimation.patch new file mode 100644 index 0000000000..0579a7352f --- /dev/null +++ b/target/linux/generic/backport-4.14/605-0012-tcp_bbr-adapt-cwnd-based-on-ack-aggregation-estimation.patch @@ -0,0 +1,262 @@ +From 4d43b58a4f987282c62e43fee39f37dedaa08d71 Mon Sep 17 00:00:00 2001 +From: Priyaranjan Jha +Date: Wed, 23 Jan 2019 12:04:54 -0800 +Subject: [PATCH] tcp_bbr: adapt cwnd based on ack aggregation estimation + +Aggregation effects are extremely common with wifi, cellular, and cable +modem link technologies, ACK decimation in middleboxes, and LRO and GRO +in receiving hosts. The aggregation can happen in either direction, +data or ACKs, but in either case the aggregation effect is visible +to the sender in the ACK stream. + +Previously BBR's sending was often limited by cwnd under severe ACK +aggregation/decimation because BBR sized the cwnd at 2*BDP. If packets +were acked in bursts after long delays (e.g. one ACK acking 5*BDP after +5*RTT), BBR's sending was halted after sending 2*BDP over 2*RTT, leaving +the bottleneck idle for potentially long periods. Note that loss-based +congestion control does not have this issue because when facing +aggregation it continues increasing cwnd after bursts of ACKs, growing +cwnd until the buffer is full. + +To achieve good throughput in the presence of aggregation effects, this +algorithm allows the BBR sender to put extra data in flight to keep the +bottleneck utilized during silences in the ACK stream that it has evidence +to suggest were caused by aggregation. + +A summary of the algorithm: when a burst of packets are acked by a +stretched ACK or a burst of ACKs or both, BBR first estimates the expected +amount of data that should have been acked, based on its estimated +bandwidth. Then the surplus ("extra_acked") is recorded in a windowed-max +filter to estimate the recent level of observed ACK aggregation. Then cwnd +is increased by the ACK aggregation estimate. The larger cwnd avoids BBR +being cwnd-limited in the face of ACK silences that recent history suggests +were caused by aggregation. As a sanity check, the ACK aggregation degree +is upper-bounded by the cwnd (at the time of measurement) and a global max +of BW * 100ms. The algorithm is further described by the following +presentation: +https://datatracker.ietf.org/meeting/101/materials/slides-101-iccrg-an-update-on-bbr-work-at-google-00 + +In our internal testing, we observed a significant increase in BBR +throughput (measured using netperf), in a basic wifi setup. +- Host1 (sender on ethernet) -> AP -> Host2 (receiver on wifi) +- 2.4 GHz -> BBR before: ~73 Mbps; BBR after: ~102 Mbps; CUBIC: ~100 Mbps +- 5.0 GHz -> BBR before: ~362 Mbps; BBR after: ~593 Mbps; CUBIC: ~601 Mbps + +Also, this code is running globally on YouTube TCP connections and produced +significant bandwidth increases for YouTube traffic. + +This is based on Ian Swett's max_ack_height_ algorithm from the +QUIC BBR implementation. + +Signed-off-by: Priyaranjan Jha +Signed-off-by: Neal Cardwell +Signed-off-by: Yuchung Cheng +Signed-off-by: David S. Miller +Signed-off-by: UtsavisGreat +--- + include/net/inet_connection_sock.h | 4 +- + net/ipv4/tcp_bbr.c | 122 ++++++++++++++++++++++++++++- + 2 files changed, 123 insertions(+), 3 deletions(-) + +--- a/include/net/inet_connection_sock.h ++++ b/include/net/inet_connection_sock.h +@@ -136,8 +136,8 @@ struct inet_connection_sock { + } icsk_mtup; + u32 icsk_user_timeout; + +- u64 icsk_ca_priv[88 / sizeof(u64)]; +-#define ICSK_CA_PRIV_SIZE (11 * sizeof(u64)) ++ u64 icsk_ca_priv[104 / sizeof(u64)]; ++#define ICSK_CA_PRIV_SIZE (13 * sizeof(u64)) + }; + + #define ICSK_TIME_RETRANS 1 /* Retransmit timer */ +--- a/net/ipv4/tcp_bbr.c ++++ b/net/ipv4/tcp_bbr.c +@@ -115,6 +115,14 @@ struct bbr { + unused_b:5; + u32 prior_cwnd; /* prior cwnd upon entering loss recovery */ + u32 full_bw; /* recent bw, to estimate if pipe is full */ ++ ++ /* For tracking ACK aggregation: */ ++ u64 ack_epoch_mstamp; /* start of ACK sampling epoch */ ++ u16 extra_acked[2]; /* max excess data ACKed in epoch */ ++ u32 ack_epoch_acked:20, /* packets (S)ACKed in sampling epoch */ ++ extra_acked_win_rtts:5, /* age of extra_acked, in round trips */ ++ extra_acked_win_idx:1, /* current index in extra_acked array */ ++ unused_c:6; + }; + + #define CYCLE_LEN 8 /* number of phases in a pacing gain cycle */ +@@ -174,6 +182,15 @@ static const u32 bbr_lt_bw_diff = 4000 / + /* If we estimate we're policed, use lt_bw for this many round trips: */ + static const u32 bbr_lt_bw_max_rtts = 48; + ++/* Gain factor for adding extra_acked to target cwnd: */ ++static const int bbr_extra_acked_gain = BBR_UNIT; ++/* Window length of extra_acked window. */ ++static const u32 bbr_extra_acked_win_rtts = 5; ++/* Max allowed val for ack_epoch_acked, after which sampling epoch is reset */ ++static const u32 bbr_ack_epoch_acked_reset_thresh = 1U << 20; ++/* Time period for clamping cwnd increment due to ack aggregation */ ++static const u32 bbr_extra_acked_max_us = 100 * 1000; ++ + static void bbr_check_probe_rtt_done(struct sock *sk); + + /* Do we estimate that STARTUP filled the pipe? */ +@@ -200,6 +217,16 @@ static u32 bbr_bw(const struct sock *sk) + return bbr->lt_use_bw ? bbr->lt_bw : bbr_max_bw(sk); + } + ++/* Return maximum extra acked in past k-2k round trips, ++ * where k = bbr_extra_acked_win_rtts. ++ */ ++static u16 bbr_extra_acked(const struct sock *sk) ++{ ++ struct bbr *bbr = inet_csk_ca(sk); ++ ++ return max(bbr->extra_acked[0], bbr->extra_acked[1]); ++} ++ + /* Return rate in bytes per second, optionally with a gain. + * The order here is chosen carefully to avoid overflow of u64. This should + * work for input rates of up to 2.9Tbit/sec and gain of 2.89x. +@@ -305,6 +332,8 @@ static void bbr_cwnd_event(struct sock * + + if (event == CA_EVENT_TX_START && tp->app_limited) { + bbr->idle_restart = 1; ++ bbr->ack_epoch_mstamp = tp->tcp_mstamp; ++ bbr->ack_epoch_acked = 0; + /* Avoid pointless buffer overflows: pace at est. bw if we don't + * need more speed (we're restarting from idle and app-limited). + */ +@@ -385,6 +414,22 @@ static u32 bbr_inflight(struct sock *sk, + return inflight; + } + ++/* Find the cwnd increment based on estimate of ack aggregation */ ++static u32 bbr_ack_aggregation_cwnd(struct sock *sk) ++{ ++ u32 max_aggr_cwnd, aggr_cwnd = 0; ++ ++ if (bbr_extra_acked_gain && bbr_full_bw_reached(sk)) { ++ max_aggr_cwnd = ((u64)bbr_bw(sk) * bbr_extra_acked_max_us) ++ / BW_UNIT; ++ aggr_cwnd = (bbr_extra_acked_gain * bbr_extra_acked(sk)) ++ >> BBR_SCALE; ++ aggr_cwnd = min(aggr_cwnd, max_aggr_cwnd); ++ } ++ ++ return aggr_cwnd; ++} ++ + /* An optimization in BBR to reduce losses: On the first round of recovery, we + * follow the packet conservation principle: send P packets per P packets acked. + * After that, we slow-start and send at most 2*P packets per P packets acked. +@@ -445,9 +490,15 @@ static void bbr_set_cwnd(struct sock *sk + if (bbr_set_cwnd_to_recover_or_restore(sk, rs, acked, &cwnd)) + goto done; + +- /* If we're below target cwnd, slow start cwnd toward target cwnd. */ + target_cwnd = bbr_bdp(sk, bw, gain); ++ ++ /* Increment the cwnd to account for excess ACKed data that seems ++ * due to aggregation (of data and/or ACKs) visible in the ACK stream. ++ */ ++ target_cwnd += bbr_ack_aggregation_cwnd(sk); + target_cwnd = bbr_quantization_budget(sk, target_cwnd, gain); ++ ++ /* If we're below target cwnd, slow start cwnd toward target cwnd. */ + if (bbr_full_bw_reached(sk)) /* only cut cwnd if we filled the pipe */ + cwnd = min(cwnd + acked, target_cwnd); + else if (cwnd < target_cwnd || tp->delivered < TCP_INIT_CWND) +@@ -711,6 +762,67 @@ static void bbr_update_bw(struct sock *s + } + } + ++/* Estimates the windowed max degree of ack aggregation. ++ * This is used to provision extra in-flight data to keep sending during ++ * inter-ACK silences. ++ * ++ * Degree of ack aggregation is estimated as extra data acked beyond expected. ++ * ++ * max_extra_acked = "maximum recent excess data ACKed beyond max_bw * interval" ++ * cwnd += max_extra_acked ++ * ++ * Max extra_acked is clamped by cwnd and bw * bbr_extra_acked_max_us (100 ms). ++ * Max filter is an approximate sliding window of 5-10 (packet timed) round ++ * trips. ++ */ ++static void bbr_update_ack_aggregation(struct sock *sk, ++ const struct rate_sample *rs) ++{ ++ u32 epoch_us, expected_acked, extra_acked; ++ struct bbr *bbr = inet_csk_ca(sk); ++ struct tcp_sock *tp = tcp_sk(sk); ++ ++ if (!bbr_extra_acked_gain || rs->acked_sacked <= 0 || ++ rs->delivered < 0 || rs->interval_us <= 0) ++ return; ++ ++ if (bbr->round_start) { ++ bbr->extra_acked_win_rtts = min(0x1F, ++ bbr->extra_acked_win_rtts + 1); ++ if (bbr->extra_acked_win_rtts >= bbr_extra_acked_win_rtts) { ++ bbr->extra_acked_win_rtts = 0; ++ bbr->extra_acked_win_idx = bbr->extra_acked_win_idx ? ++ 0 : 1; ++ bbr->extra_acked[bbr->extra_acked_win_idx] = 0; ++ } ++ } ++ ++ /* Compute how many packets we expected to be delivered over epoch. */ ++ epoch_us = tcp_stamp_us_delta(tp->delivered_mstamp, ++ bbr->ack_epoch_mstamp); ++ expected_acked = ((u64)bbr_bw(sk) * epoch_us) / BW_UNIT; ++ ++ /* Reset the aggregation epoch if ACK rate is below expected rate or ++ * significantly large no. of ack received since epoch (potentially ++ * quite old epoch). ++ */ ++ if (bbr->ack_epoch_acked <= expected_acked || ++ (bbr->ack_epoch_acked + rs->acked_sacked >= ++ bbr_ack_epoch_acked_reset_thresh)) { ++ bbr->ack_epoch_acked = 0; ++ bbr->ack_epoch_mstamp = tp->delivered_mstamp; ++ expected_acked = 0; ++ } ++ ++ /* Compute excess data delivered, beyond what was expected. */ ++ bbr->ack_epoch_acked = min_t(u32, 0xFFFFF, ++ bbr->ack_epoch_acked + rs->acked_sacked); ++ extra_acked = bbr->ack_epoch_acked - expected_acked; ++ extra_acked = min(extra_acked, tp->snd_cwnd); ++ if (extra_acked > bbr->extra_acked[bbr->extra_acked_win_idx]) ++ bbr->extra_acked[bbr->extra_acked_win_idx] = extra_acked; ++} ++ + /* Estimate when the pipe is full, using the change in delivery rate: BBR + * estimates that STARTUP filled the pipe if the estimated bw hasn't changed by + * at least bbr_full_bw_thresh (25%) after bbr_full_bw_cnt (3) non-app-limited +@@ -865,6 +977,7 @@ static void bbr_update_gains(struct sock + static void bbr_update_model(struct sock *sk, const struct rate_sample *rs) + { + bbr_update_bw(sk, rs); ++ bbr_update_ack_aggregation(sk, rs); + bbr_update_cycle_phase(sk, rs); + bbr_check_full_bw_reached(sk, rs); + bbr_check_drain(sk, rs); +@@ -916,6 +1029,13 @@ static void bbr_init(struct sock *sk) + bbr_reset_lt_bw_sampling(sk); + bbr_reset_startup_mode(sk); + ++ bbr->ack_epoch_mstamp = tp->tcp_mstamp; ++ bbr->ack_epoch_acked = 0; ++ bbr->extra_acked_win_rtts = 0; ++ bbr->extra_acked_win_idx = 0; ++ bbr->extra_acked[0] = 0; ++ bbr->extra_acked[1] = 0; ++ + cmpxchg(&sk->sk_pacing_status, SK_PACING_NONE, SK_PACING_NEEDED); + } + diff --git a/target/linux/generic/pending-4.14/607-tcp_bbr-adapt-cwnd-based-on-ack-aggregation-estimation.patch b/target/linux/generic/pending-4.14/607-tcp_bbr-adapt-cwnd-based-on-ack-aggregation-estimation.patch deleted file mode 100644 index 2aaafa19d4..0000000000 --- a/target/linux/generic/pending-4.14/607-tcp_bbr-adapt-cwnd-based-on-ack-aggregation-estimation.patch +++ /dev/null @@ -1,611 +0,0 @@ -From 232aa8ec3ed979d4716891540c03a806ecab0c37 Mon Sep 17 00:00:00 2001 -From: Priyaranjan Jha -Date: Wed, 23 Jan 2019 12:04:53 -0800 -Subject: tcp_bbr: refactor bbr_target_cwnd() for general inflight provisioning - -Because bbr_target_cwnd() is really a general-purpose BBR helper for -computing some volume of inflight data as a function of the estimated -BDP, refactor it into following helper functions: -- bbr_bdp() -- bbr_quantization_budget() -- bbr_inflight() - -Signed-off-by: Priyaranjan Jha -Signed-off-by: Neal Cardwell -Signed-off-by: Yuchung Cheng -Signed-off-by: David S. Miller ---- ---- a/include/net/inet_connection_sock.h -+++ b/include/net/inet_connection_sock.h -@@ -136,8 +136,8 @@ struct inet_connection_sock { - } icsk_mtup; - u32 icsk_user_timeout; - -- u64 icsk_ca_priv[88 / sizeof(u64)]; --#define ICSK_CA_PRIV_SIZE (11 * sizeof(u64)) -+ u64 icsk_ca_priv[104 / sizeof(u64)]; -+#define ICSK_CA_PRIV_SIZE (13 * sizeof(u64)) - }; - - #define ICSK_TIME_RETRANS 1 /* Retransmit timer */ ---- a/net/ipv4/tcp_bbr.c -+++ b/net/ipv4/tcp_bbr.c -@@ -95,12 +95,10 @@ struct bbr { - u32 mode:3, /* current bbr_mode in state machine */ - prev_ca_state:3, /* CA state on previous ACK */ - packet_conservation:1, /* use packet conservation? */ -- restore_cwnd:1, /* decided to revert cwnd to old value */ - round_start:1, /* start of packet-timed tx->ack round? */ -- tso_segs_goal:7, /* segments we want in each skb we send */ - idle_restart:1, /* restarting after idle? */ - probe_rtt_round_done:1, /* a BBR_PROBE_RTT round at 4 pkts? */ -- unused:5, -+ unused:13, - lt_is_sampling:1, /* taking long-term ("LT") samples now? */ - lt_rtt_cnt:7, /* round trips in long-term interval */ - lt_use_bw:1; /* use lt_bw as our bw estimate? */ -@@ -117,6 +115,14 @@ struct bbr { - unused_b:5; - u32 prior_cwnd; /* prior cwnd upon entering loss recovery */ - u32 full_bw; /* recent bw, to estimate if pipe is full */ -+ -+ /* For tracking ACK aggregation: */ -+ u64 ack_epoch_mstamp; /* start of ACK sampling epoch */ -+ u16 extra_acked[2]; /* max excess data ACKed in epoch */ -+ u32 ack_epoch_acked:20, /* packets (S)ACKed in sampling epoch */ -+ extra_acked_win_rtts:5, /* age of extra_acked, in round trips */ -+ extra_acked_win_idx:1, /* current index in extra_acked array */ -+ unused_c:6; - }; - - #define CYCLE_LEN 8 /* number of phases in a pacing gain cycle */ -@@ -176,6 +182,17 @@ static const u32 bbr_lt_bw_diff = 4000 / - /* If we estimate we're policed, use lt_bw for this many round trips: */ - static const u32 bbr_lt_bw_max_rtts = 48; - -+/* Gain factor for adding extra_acked to target cwnd: */ -+static const int bbr_extra_acked_gain = BBR_UNIT; -+/* Window length of extra_acked window. */ -+static const u32 bbr_extra_acked_win_rtts = 5; -+/* Max allowed val for ack_epoch_acked, after which sampling epoch is reset */ -+static const u32 bbr_ack_epoch_acked_reset_thresh = 1U << 20; -+/* Time period for clamping cwnd increment due to ack aggregation */ -+static const u32 bbr_extra_acked_max_us = 100 * 1000; -+ -+static void bbr_check_probe_rtt_done(struct sock *sk); -+ - /* Do we estimate that STARTUP filled the pipe? */ - static bool bbr_full_bw_reached(const struct sock *sk) - { -@@ -200,13 +217,31 @@ static u32 bbr_bw(const struct sock *sk) - return bbr->lt_use_bw ? bbr->lt_bw : bbr_max_bw(sk); - } - -+/* Return maximum extra acked in past k-2k round trips, -+ * where k = bbr_extra_acked_win_rtts. -+ */ -+static u16 bbr_extra_acked(const struct sock *sk) -+{ -+ struct bbr *bbr = inet_csk_ca(sk); -+ -+ return max(bbr->extra_acked[0], bbr->extra_acked[1]); -+} -+ - /* Return rate in bytes per second, optionally with a gain. - * The order here is chosen carefully to avoid overflow of u64. This should - * work for input rates of up to 2.9Tbit/sec and gain of 2.89x. - */ -+static bool tcp_needs_internal_pacing(const struct sock *sk) -+{ -+ return smp_load_acquire(&sk->sk_pacing_status) == SK_PACING_NEEDED; -+} - static u64 bbr_rate_bytes_per_sec(struct sock *sk, u64 rate, int gain) - { -- rate *= tcp_mss_to_mtu(sk, tcp_sk(sk)->mss_cache); -+ unsigned int mss = tcp_sk(sk)->mss_cache; -+ -+ if (!tcp_needs_internal_pacing(sk)) -+ mss = tcp_mss_to_mtu(sk, mss); -+ rate *= mss; - rate *= gain; - rate >>= BBR_SCALE; - rate *= USEC_PER_SEC; -@@ -261,23 +296,25 @@ static void bbr_set_pacing_rate(struct s - sk->sk_pacing_rate = rate; - } - --/* Return count of segments we want in the skbs we send, or 0 for default. */ --static u32 bbr_tso_segs_goal(struct sock *sk) -+/* override sysctl_tcp_min_tso_segs */ -+static u32 bbr_min_tso_segs(struct sock *sk) - { -- struct bbr *bbr = inet_csk_ca(sk); -- -- return bbr->tso_segs_goal; -+ return sk->sk_pacing_rate < (bbr_min_tso_rate >> 3) ? 1 : 2; - } - --static void bbr_set_tso_segs_goal(struct sock *sk) -+static u32 bbr_tso_segs_goal(struct sock *sk) - { - struct tcp_sock *tp = tcp_sk(sk); -- struct bbr *bbr = inet_csk_ca(sk); -- u32 min_segs; -+ u32 segs, bytes; -+ -+ /* Sort of tcp_tso_autosize() but ignoring -+ * driver provided sk_gso_max_size. -+ */ -+ bytes = min_t(u32, sk->sk_pacing_rate >> sk->sk_pacing_shift, -+ GSO_MAX_SIZE - 1 - MAX_TCP_HEADER); -+ segs = max_t(u32, bytes / tp->mss_cache, bbr_min_tso_segs(sk)); - -- min_segs = sk->sk_pacing_rate < (bbr_min_tso_rate >> 3) ? 1 : 2; -- bbr->tso_segs_goal = min(tcp_tso_autosize(sk, tp->mss_cache, min_segs), -- 0x7FU); -+ return min(segs, 0x7FU); - } - - /* Save "last known good" cwnd so we can restore it after losses or PROBE_RTT */ -@@ -299,38 +336,31 @@ static void bbr_cwnd_event(struct sock * - - if (event == CA_EVENT_TX_START && tp->app_limited) { - bbr->idle_restart = 1; -+ bbr->ack_epoch_mstamp = tp->tcp_mstamp; -+ bbr->ack_epoch_acked = 0; - /* Avoid pointless buffer overflows: pace at est. bw if we don't - * need more speed (we're restarting from idle and app-limited). - */ - if (bbr->mode == BBR_PROBE_BW) - bbr_set_pacing_rate(sk, bbr_bw(sk), BBR_UNIT); -+ else if (bbr->mode == BBR_PROBE_RTT) -+ bbr_check_probe_rtt_done(sk); - } - } - --/* Find target cwnd. Right-size the cwnd based on min RTT and the -- * estimated bottleneck bandwidth: -+/* Calculate bdp based on min RTT and the estimated bottleneck bandwidth: - * -- * cwnd = bw * min_rtt * gain = BDP * gain -+ * bdp = bw * min_rtt * gain - * - * The key factor, gain, controls the amount of queue. While a small gain - * builds a smaller queue, it becomes more vulnerable to noise in RTT - * measurements (e.g., delayed ACKs or other ACK compression effects). This - * noise may cause BBR to under-estimate the rate. -- * -- * To achieve full performance in high-speed paths, we budget enough cwnd to -- * fit full-sized skbs in-flight on both end hosts to fully utilize the path: -- * - one skb in sending host Qdisc, -- * - one skb in sending host TSO/GSO engine -- * - one skb being received by receiver host LRO/GRO/delayed-ACK engine -- * Don't worry, at low rates (bbr_min_tso_rate) this won't bloat cwnd because -- * in such cases tso_segs_goal is 1. The minimum cwnd is 4 packets, -- * which allows 2 outstanding 2-packet sequences, to try to keep pipe -- * full even with ACK-every-other-packet delayed ACKs. - */ --static u32 bbr_target_cwnd(struct sock *sk, u32 bw, int gain) -+static u32 bbr_bdp(struct sock *sk, u32 bw, int gain) - { - struct bbr *bbr = inet_csk_ca(sk); -- u32 cwnd; -+ u32 bdp; - u64 w; - - /* If we've never had a valid RTT sample, cap cwnd at the initial -@@ -345,21 +375,65 @@ static u32 bbr_target_cwnd(struct sock * - w = (u64)bw * bbr->min_rtt_us; - - /* Apply a gain to the given value, then remove the BW_SCALE shift. */ -- cwnd = (((w * gain) >> BBR_SCALE) + BW_UNIT - 1) / BW_UNIT; -+ bdp = (((w * gain) >> BBR_SCALE) + BW_UNIT - 1) / BW_UNIT; -+ -+ return bdp; -+} -+ -+/* To achieve full performance in high-speed paths, we budget enough cwnd to -+ * fit full-sized skbs in-flight on both end hosts to fully utilize the path: -+ * - one skb in sending host Qdisc, -+ * - one skb in sending host TSO/GSO engine -+ * - one skb being received by receiver host LRO/GRO/delayed-ACK engine -+ * Don't worry, at low rates (bbr_min_tso_rate) this won't bloat cwnd because -+ * in such cases tso_segs_goal is 1. The minimum cwnd is 4 packets, -+ * which allows 2 outstanding 2-packet sequences, to try to keep pipe -+ * full even with ACK-every-other-packet delayed ACKs. -+ */ -+static u32 bbr_quantization_budget(struct sock *sk, u32 cwnd) -+{ -+ struct bbr *bbr = inet_csk_ca(sk); - - /* Allow enough full-sized skbs in flight to utilize end systems. */ -- cwnd += 3 * bbr->tso_segs_goal; -+ cwnd += 3 * bbr_tso_segs_goal(sk); - - /* Reduce delayed ACKs by rounding up cwnd to the next even number. */ - cwnd = (cwnd + 1) & ~1U; - - /* Ensure gain cycling gets inflight above BDP even for small BDPs. */ -- if (bbr->mode == BBR_PROBE_BW && gain > BBR_UNIT) -+ if (bbr->mode == BBR_PROBE_BW && bbr->cycle_idx == 0) - cwnd += 2; - - return cwnd; - } - -+/* Find inflight based on min RTT and the estimated bottleneck bandwidth. */ -+static u32 bbr_inflight(struct sock *sk, u32 bw, int gain) -+{ -+ u32 inflight; -+ -+ inflight = bbr_bdp(sk, bw, gain); -+ inflight = bbr_quantization_budget(sk, inflight); -+ -+ return inflight; -+} -+ -+/* Find the cwnd increment based on estimate of ack aggregation */ -+static u32 bbr_ack_aggregation_cwnd(struct sock *sk) -+{ -+ u32 max_aggr_cwnd, aggr_cwnd = 0; -+ -+ if (bbr_extra_acked_gain && bbr_full_bw_reached(sk)) { -+ max_aggr_cwnd = ((u64)bbr_bw(sk) * bbr_extra_acked_max_us) -+ / BW_UNIT; -+ aggr_cwnd = (bbr_extra_acked_gain * bbr_extra_acked(sk)) -+ >> BBR_SCALE; -+ aggr_cwnd = min(aggr_cwnd, max_aggr_cwnd); -+ } -+ -+ return aggr_cwnd; -+} -+ - /* An optimization in BBR to reduce losses: On the first round of recovery, we - * follow the packet conservation principle: send P packets per P packets acked. - * After that, we slow-start and send at most 2*P packets per P packets acked. -@@ -391,17 +465,11 @@ static bool bbr_set_cwnd_to_recover_or_r - cwnd = tcp_packets_in_flight(tp) + acked; - } else if (prev_state >= TCP_CA_Recovery && state < TCP_CA_Recovery) { - /* Exiting loss recovery; restore cwnd saved before recovery. */ -- bbr->restore_cwnd = 1; -+ cwnd = max(cwnd, bbr->prior_cwnd); - bbr->packet_conservation = 0; - } - bbr->prev_ca_state = state; - -- if (bbr->restore_cwnd) { -- /* Restore cwnd after exiting loss recovery or PROBE_RTT. */ -- cwnd = max(cwnd, bbr->prior_cwnd); -- bbr->restore_cwnd = 0; -- } -- - if (bbr->packet_conservation) { - *new_cwnd = max(cwnd, tcp_packets_in_flight(tp) + acked); - return true; /* yes, using packet conservation */ -@@ -418,16 +486,23 @@ static void bbr_set_cwnd(struct sock *sk - { - struct tcp_sock *tp = tcp_sk(sk); - struct bbr *bbr = inet_csk_ca(sk); -- u32 cwnd = 0, target_cwnd = 0; -+ u32 cwnd = tp->snd_cwnd, target_cwnd = 0; - - if (!acked) -- return; -+ goto done; /* no packet fully ACKed; just apply caps */ - - if (bbr_set_cwnd_to_recover_or_restore(sk, rs, acked, &cwnd)) - goto done; - -+ target_cwnd = bbr_bdp(sk, bw, gain); -+ -+ /* Increment the cwnd to account for excess ACKed data that seems -+ * due to aggregation (of data and/or ACKs) visible in the ACK stream. -+ */ -+ target_cwnd += bbr_ack_aggregation_cwnd(sk); -+ target_cwnd = bbr_quantization_budget(sk, target_cwnd); -+ - /* If we're below target cwnd, slow start cwnd toward target cwnd. */ -- target_cwnd = bbr_target_cwnd(sk, bw, gain); - if (bbr_full_bw_reached(sk)) /* only cut cwnd if we filled the pipe */ - cwnd = min(cwnd + acked, target_cwnd); - else if (cwnd < target_cwnd || tp->delivered < TCP_INIT_CWND) -@@ -468,14 +543,14 @@ static bool bbr_is_next_cycle_phase(stru - if (bbr->pacing_gain > BBR_UNIT) - return is_full_length && - (rs->losses || /* perhaps pacing_gain*BDP won't fit */ -- inflight >= bbr_target_cwnd(sk, bw, bbr->pacing_gain)); -+ inflight >= bbr_inflight(sk, bw, bbr->pacing_gain)); - - /* A pacing_gain < 1.0 tries to drain extra queue we added if bw - * probing didn't find more bw. If inflight falls to match BDP then we - * estimate queue is drained; persisting would underutilize the pipe. - */ - return is_full_length || -- inflight <= bbr_target_cwnd(sk, bw, BBR_UNIT); -+ inflight <= bbr_inflight(sk, bw, BBR_UNIT); - } - - static void bbr_advance_cycle_phase(struct sock *sk) -@@ -485,8 +560,6 @@ static void bbr_advance_cycle_phase(stru - - bbr->cycle_idx = (bbr->cycle_idx + 1) & (CYCLE_LEN - 1); - bbr->cycle_mstamp = tp->delivered_mstamp; -- bbr->pacing_gain = bbr->lt_use_bw ? BBR_UNIT : -- bbr_pacing_gain[bbr->cycle_idx]; - } - - /* Gain cycling: cycle pacing gain to converge to fair share of available bw. */ -@@ -504,8 +577,6 @@ static void bbr_reset_startup_mode(struc - struct bbr *bbr = inet_csk_ca(sk); - - bbr->mode = BBR_STARTUP; -- bbr->pacing_gain = bbr_high_gain; -- bbr->cwnd_gain = bbr_high_gain; - } - - static void bbr_reset_probe_bw_mode(struct sock *sk) -@@ -513,8 +584,6 @@ static void bbr_reset_probe_bw_mode(stru - struct bbr *bbr = inet_csk_ca(sk); - - bbr->mode = BBR_PROBE_BW; -- bbr->pacing_gain = BBR_UNIT; -- bbr->cwnd_gain = bbr_cwnd_gain; - bbr->cycle_idx = CYCLE_LEN - 1 - prandom_u32_max(bbr_cycle_rand); - bbr_advance_cycle_phase(sk); /* flip to next phase of gain cycle */ - } -@@ -697,6 +766,67 @@ static void bbr_update_bw(struct sock *s - } - } - -+/* Estimates the windowed max degree of ack aggregation. -+ * This is used to provision extra in-flight data to keep sending during -+ * inter-ACK silences. -+ * -+ * Degree of ack aggregation is estimated as extra data acked beyond expected. -+ * -+ * max_extra_acked = "maximum recent excess data ACKed beyond max_bw * interval" -+ * cwnd += max_extra_acked -+ * -+ * Max extra_acked is clamped by cwnd and bw * bbr_extra_acked_max_us (100 ms). -+ * Max filter is an approximate sliding window of 5-10 (packet timed) round -+ * trips. -+ */ -+static void bbr_update_ack_aggregation(struct sock *sk, -+ const struct rate_sample *rs) -+{ -+ u32 epoch_us, expected_acked, extra_acked; -+ struct bbr *bbr = inet_csk_ca(sk); -+ struct tcp_sock *tp = tcp_sk(sk); -+ -+ if (!bbr_extra_acked_gain || rs->acked_sacked <= 0 || -+ rs->delivered < 0 || rs->interval_us <= 0) -+ return; -+ -+ if (bbr->round_start) { -+ bbr->extra_acked_win_rtts = min(0x1F, -+ bbr->extra_acked_win_rtts + 1); -+ if (bbr->extra_acked_win_rtts >= bbr_extra_acked_win_rtts) { -+ bbr->extra_acked_win_rtts = 0; -+ bbr->extra_acked_win_idx = bbr->extra_acked_win_idx ? -+ 0 : 1; -+ bbr->extra_acked[bbr->extra_acked_win_idx] = 0; -+ } -+ } -+ -+ /* Compute how many packets we expected to be delivered over epoch. */ -+ epoch_us = tcp_stamp_us_delta(tp->delivered_mstamp, -+ bbr->ack_epoch_mstamp); -+ expected_acked = ((u64)bbr_bw(sk) * epoch_us) / BW_UNIT; -+ -+ /* Reset the aggregation epoch if ACK rate is below expected rate or -+ * significantly large no. of ack received since epoch (potentially -+ * quite old epoch). -+ */ -+ if (bbr->ack_epoch_acked <= expected_acked || -+ (bbr->ack_epoch_acked + rs->acked_sacked >= -+ bbr_ack_epoch_acked_reset_thresh)) { -+ bbr->ack_epoch_acked = 0; -+ bbr->ack_epoch_mstamp = tp->delivered_mstamp; -+ expected_acked = 0; -+ } -+ -+ /* Compute excess data delivered, beyond what was expected. */ -+ bbr->ack_epoch_acked = min_t(u32, 0xFFFFF, -+ bbr->ack_epoch_acked + rs->acked_sacked); -+ extra_acked = bbr->ack_epoch_acked - expected_acked; -+ extra_acked = min(extra_acked, tp->snd_cwnd); -+ if (extra_acked > bbr->extra_acked[bbr->extra_acked_win_idx]) -+ bbr->extra_acked[bbr->extra_acked_win_idx] = extra_acked; -+} -+ - /* Estimate when the pipe is full, using the change in delivery rate: BBR - * estimates that STARTUP filled the pipe if the estimated bw hasn't changed by - * at least bbr_full_bw_thresh (25%) after bbr_full_bw_cnt (3) non-app-limited -@@ -731,15 +861,29 @@ static void bbr_check_drain(struct sock - - if (bbr->mode == BBR_STARTUP && bbr_full_bw_reached(sk)) { - bbr->mode = BBR_DRAIN; /* drain queue we created */ -- bbr->pacing_gain = bbr_drain_gain; /* pace slow to drain */ -- bbr->cwnd_gain = bbr_high_gain; /* maintain cwnd */ -+ tcp_sk(sk)->snd_ssthresh = -+ bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT); - } /* fall through to check if in-flight is already small: */ - if (bbr->mode == BBR_DRAIN && - tcp_packets_in_flight(tcp_sk(sk)) <= -- bbr_target_cwnd(sk, bbr_max_bw(sk), BBR_UNIT)) -+ bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT)) - bbr_reset_probe_bw_mode(sk); /* we estimate queue is drained */ - } - -+static void bbr_check_probe_rtt_done(struct sock *sk) -+{ -+ struct tcp_sock *tp = tcp_sk(sk); -+ struct bbr *bbr = inet_csk_ca(sk); -+ -+ if (!(bbr->probe_rtt_done_stamp && -+ after(tcp_jiffies32, bbr->probe_rtt_done_stamp))) -+ return; -+ -+ bbr->min_rtt_stamp = tcp_jiffies32; /* wait a while until PROBE_RTT */ -+ tp->snd_cwnd = max(tp->snd_cwnd, bbr->prior_cwnd); -+ bbr_reset_mode(sk); -+} -+ - /* The goal of PROBE_RTT mode is to have BBR flows cooperatively and - * periodically drain the bottleneck queue, to converge to measure the true - * min_rtt (unloaded propagation delay). This allows the flows to keep queues -@@ -769,7 +913,8 @@ static void bbr_update_min_rtt(struct so - filter_expired = after(tcp_jiffies32, - bbr->min_rtt_stamp + bbr_min_rtt_win_sec * HZ); - if (rs->rtt_us >= 0 && -- (rs->rtt_us < bbr->min_rtt_us || filter_expired)) { -+ (rs->rtt_us < bbr->min_rtt_us || -+ (filter_expired && !rs->is_ack_delayed))) { - bbr->min_rtt_us = rs->rtt_us; - bbr->min_rtt_stamp = tcp_jiffies32; - } -@@ -777,8 +922,6 @@ static void bbr_update_min_rtt(struct so - if (bbr_probe_rtt_mode_ms > 0 && filter_expired && - !bbr->idle_restart && bbr->mode != BBR_PROBE_RTT) { - bbr->mode = BBR_PROBE_RTT; /* dip, drain queue */ -- bbr->pacing_gain = BBR_UNIT; -- bbr->cwnd_gain = BBR_UNIT; - bbr_save_cwnd(sk); /* note cwnd so we can restore it */ - bbr->probe_rtt_done_stamp = 0; - } -@@ -797,12 +940,8 @@ static void bbr_update_min_rtt(struct so - } else if (bbr->probe_rtt_done_stamp) { - if (bbr->round_start) - bbr->probe_rtt_round_done = 1; -- if (bbr->probe_rtt_round_done && -- after(tcp_jiffies32, bbr->probe_rtt_done_stamp)) { -- bbr->min_rtt_stamp = tcp_jiffies32; -- bbr->restore_cwnd = 1; /* snap to prior_cwnd */ -- bbr_reset_mode(sk); -- } -+ if (bbr->probe_rtt_round_done) -+ bbr_check_probe_rtt_done(sk); - } - } - /* Restart after idle ends only once we process a new S/ACK for data */ -@@ -810,13 +949,44 @@ static void bbr_update_min_rtt(struct so - bbr->idle_restart = 0; - } - -+static void bbr_update_gains(struct sock *sk) -+{ -+ struct bbr *bbr = inet_csk_ca(sk); -+ -+ switch (bbr->mode) { -+ case BBR_STARTUP: -+ bbr->pacing_gain = bbr_high_gain; -+ bbr->cwnd_gain = bbr_high_gain; -+ break; -+ case BBR_DRAIN: -+ bbr->pacing_gain = bbr_drain_gain; /* slow, to drain */ -+ bbr->cwnd_gain = bbr_high_gain; /* keep cwnd */ -+ break; -+ case BBR_PROBE_BW: -+ bbr->pacing_gain = (bbr->lt_use_bw ? -+ BBR_UNIT : -+ bbr_pacing_gain[bbr->cycle_idx]); -+ bbr->cwnd_gain = bbr_cwnd_gain; -+ break; -+ case BBR_PROBE_RTT: -+ bbr->pacing_gain = BBR_UNIT; -+ bbr->cwnd_gain = BBR_UNIT; -+ break; -+ default: -+ WARN_ONCE(1, "BBR bad mode: %u\n", bbr->mode); -+ break; -+ } -+} -+ - static void bbr_update_model(struct sock *sk, const struct rate_sample *rs) - { - bbr_update_bw(sk, rs); -+ bbr_update_ack_aggregation(sk, rs); - bbr_update_cycle_phase(sk, rs); - bbr_check_full_bw_reached(sk, rs); - bbr_check_drain(sk, rs); - bbr_update_min_rtt(sk, rs); -+ bbr_update_gains(sk); - } - - static void bbr_main(struct sock *sk, const struct rate_sample *rs) -@@ -828,7 +998,6 @@ static void bbr_main(struct sock *sk, co - - bw = bbr_bw(sk); - bbr_set_pacing_rate(sk, bw, bbr->pacing_gain); -- bbr_set_tso_segs_goal(sk); - bbr_set_cwnd(sk, rs, rs->acked_sacked, bw, bbr->cwnd_gain); - } - -@@ -838,7 +1007,7 @@ static void bbr_init(struct sock *sk) - struct bbr *bbr = inet_csk_ca(sk); - - bbr->prior_cwnd = 0; -- bbr->tso_segs_goal = 0; /* default segs per skb until first ACK */ -+ tp->snd_ssthresh = TCP_INFINITE_SSTHRESH; - bbr->rtt_cnt = 0; - bbr->next_rtt_delivered = 0; - bbr->prev_ca_state = TCP_CA_Open; -@@ -854,7 +1023,6 @@ static void bbr_init(struct sock *sk) - bbr->has_seen_rtt = 0; - bbr_init_pacing_rate_from_rtt(sk); - -- bbr->restore_cwnd = 0; - bbr->round_start = 0; - bbr->idle_restart = 0; - bbr->full_bw_reached = 0; -@@ -865,6 +1033,13 @@ static void bbr_init(struct sock *sk) - bbr_reset_lt_bw_sampling(sk); - bbr_reset_startup_mode(sk); - -+ bbr->ack_epoch_mstamp = tp->tcp_mstamp; -+ bbr->ack_epoch_acked = 0; -+ bbr->extra_acked_win_rtts = 0; -+ bbr->extra_acked_win_idx = 0; -+ bbr->extra_acked[0] = 0; -+ bbr->extra_acked[1] = 0; -+ - cmpxchg(&sk->sk_pacing_status, SK_PACING_NONE, SK_PACING_NEEDED); - } - -@@ -891,7 +1066,7 @@ static u32 bbr_undo_cwnd(struct sock *sk - static u32 bbr_ssthresh(struct sock *sk) - { - bbr_save_cwnd(sk); -- return TCP_INFINITE_SSTHRESH; /* BBR does not use ssthresh */ -+ return tcp_sk(sk)->snd_ssthresh; - } - - static size_t bbr_get_info(struct sock *sk, u32 ext, int *attr, -@@ -940,7 +1115,7 @@ static struct tcp_congestion_ops tcp_bbr - .undo_cwnd = bbr_undo_cwnd, - .cwnd_event = bbr_cwnd_event, - .ssthresh = bbr_ssthresh, -- .tso_segs_goal = bbr_tso_segs_goal, -+ .min_tso_segs = bbr_min_tso_segs, - .get_info = bbr_get_info, - .set_state = bbr_set_state, - }; ---- a/include/net/tcp.h -+++ b/include/net/tcp.h -@@ -998,6 +998,7 @@ struct rate_sample { - u32 prior_in_flight; /* in flight before this ACK */ - bool is_app_limited; /* is sample from packet with bubble in pipe? */ - bool is_retrans; /* is sample from retransmission? */ -+ bool is_ack_delayed; /* is this (likely) a delayed ACK? */ - }; - - struct tcp_congestion_ops { -@@ -1024,6 +1025,8 @@ struct tcp_congestion_ops { - u32 (*undo_cwnd)(struct sock *sk); - /* hook for packet ack accounting (optional) */ - void (*pkts_acked)(struct sock *sk, const struct ack_sample *sample); -+ /* override sysctl_tcp_min_tso_segs */ -+ u32 (*min_tso_segs)(struct sock *sk); - /* suggest number of segments for each skb to transmit (optional) */ - u32 (*tso_segs_goal)(struct sock *sk); - /* returns the multiplier used in tcp_sndbuf_expand (optional) */