0001-net-tcp-backport-BBRv3-to-android16-6.12.patch 108 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877187818791880188118821883188418851886188718881889189018911892189318941895189618971898189919001901190219031904190519061907190819091910191119121913191419151916191719181919192019211922192319241925192619271928192919301931193219331934193519361937193819391940194119421943194419451946194719481949195019511952195319541955195619571958195919601961196219631964196519661967196819691970197119721973197419751976197719781979198019811982198319841985198619871988198919901991199219931994199519961997199819992000200120022003200420052006200720082009201020112012201320142015201620172018201920202021202220232024202520262027202820292030203120322033203420352036203720382039204020412042204320442045204620472048204920502051205220532054205520562057205820592060206120622063206420652066206720682069207020712072207320742075207620772078207920802081208220832084208520862087208820892090209120922093209420952096209720982099210021012102210321042105210621072108210921102111211221132114211521162117211821192120212121222123212421252126212721282129213021312132213321342135213621372138213921402141214221432144214521462147214821492150215121522153215421552156215721582159216021612162216321642165216621672168216921702171217221732174217521762177217821792180218121822183218421852186218721882189219021912192219321942195219621972198219922002201220222032204220522062207220822092210221122122213221422152216221722182219222022212222222322242225222622272228222922302231223222332234223522362237223822392240224122422243224422452246224722482249225022512252225322542255225622572258225922602261226222632264226522662267226822692270227122722273227422752276227722782279228022812282228322842285228622872288228922902291229222932294229522962297229822992300230123022303230423052306230723082309231023112312231323142315231623172318231923202321232223232324232523262327232823292330233123322333233423352336233723382339234023412342234323442345234623472348234923502351235223532354235523562357235823592360236123622363236423652366236723682369237023712372237323742375237623772378237923802381238223832384238523862387238823892390239123922393239423952396239723982399240024012402240324042405240624072408240924102411241224132414241524162417241824192420242124222423242424252426242724282429243024312432243324342435243624372438243924402441244224432444244524462447244824492450245124522453245424552456245724582459246024612462246324642465246624672468246924702471247224732474247524762477247824792480248124822483248424852486248724882489249024912492249324942495249624972498249925002501250225032504250525062507250825092510251125122513251425152516251725182519252025212522252325242525252625272528252925302531253225332534253525362537253825392540254125422543254425452546254725482549255025512552255325542555255625572558255925602561256225632564256525662567256825692570257125722573257425752576257725782579258025812582258325842585258625872588258925902591259225932594259525962597259825992600260126022603260426052606260726082609261026112612261326142615261626172618261926202621262226232624262526262627262826292630263126322633263426352636263726382639264026412642264326442645264626472648264926502651265226532654265526562657265826592660266126622663266426652666266726682669267026712672267326742675267626772678267926802681268226832684268526862687268826892690269126922693269426952696269726982699270027012702270327042705270627072708270927102711271227132714271527162717271827192720272127222723272427252726272727282729273027312732273327342735273627372738273927402741274227432744274527462747274827492750275127522753275427552756275727582759276027612762276327642765276627672768276927702771277227732774277527762777277827792780278127822783278427852786278727882789279027912792279327942795279627972798279928002801280228032804280528062807280828092810281128122813281428152816281728182819282028212822282328242825282628272828282928302831283228332834283528362837283828392840284128422843284428452846284728482849285028512852285328542855285628572858285928602861286228632864286528662867286828692870287128722873287428752876287728782879288028812882288328842885288628872888288928902891289228932894289528962897289828992900290129022903290429052906290729082909291029112912291329142915291629172918291929202921292229232924292529262927292829292930293129322933293429352936293729382939294029412942294329442945294629472948294929502951295229532954295529562957295829592960296129622963296429652966296729682969297029712972297329742975297629772978297929802981298229832984298529862987298829892990299129922993299429952996299729982999300030013002300330043005300630073008300930103011301230133014301530163017301830193020302130223023302430253026302730283029303030313032303330343035303630373038303930403041304230433044304530463047304830493050305130523053305430553056305730583059306030613062306330643065306630673068
  1. diff --git a/include/linux/tcp.h b/include/linux/tcp.h
  2. index 3bbf355f9..aaa9f7605 100644
  3. --- a/include/linux/tcp.h
  4. +++ b/include/linux/tcp.h
  5. @@ -370,7 +370,13 @@ struct tcp_sock {
  6. u8 compressed_ack;
  7. u8 dup_ack_counter:2,
  8. tlp_retrans:1, /* TLP is a retransmission */
  9. +#ifndef __GENKSYMS__
  10. + fast_ack_mode:2, /* which fast ack mode ? */
  11. + tlp_orig_data_app_limited:1, /* app-limited before TLP rtx? */
  12. + unused:2;
  13. +#else
  14. unused:5;
  15. +#endif
  16. u8 thin_lto : 1,/* Use linear timeouts for thin streams */
  17. fastopen_connect:1, /* FASTOPEN_CONNECT sockopt */
  18. fastopen_no_cookie:1, /* Allow send/recv SYN+data without a cookie */
  19. diff --git a/include/net/tcp.h b/include/net/tcp.h
  20. index 3255a199e..347d1ac2c 100644
  21. --- a/include/net/tcp.h
  22. +++ b/include/net/tcp.h
  23. @@ -376,6 +376,8 @@ static inline void tcp_dec_quickack_mode(struct sock *sk)
  24. #define TCP_ECN_QUEUE_CWR 2
  25. #define TCP_ECN_DEMAND_CWR 4
  26. #define TCP_ECN_SEEN 8
  27. +#define TCP_ECN_LOW 16
  28. +#define TCP_ECN_ECT_PERMANENT 3
  29. enum tcp_tw_status {
  30. TCP_TW_SUCCESS = 0,
  31. @@ -796,6 +798,15 @@ static inline void tcp_fast_path_check(struct sock *sk)
  32. u32 tcp_delack_max(const struct sock *sk);
  33. +static inline void tcp_set_ecn_low_from_dst(struct sock *sk,
  34. + const struct dst_entry *dst)
  35. +{
  36. + struct tcp_sock *tp = tcp_sk(sk);
  37. +
  38. + if (dst_feature(dst, RTAX_FEATURE_ECN_LOW))
  39. + tp->ecn_flags |= TCP_ECN_LOW;
  40. +}
  41. +
  42. /* Compute the actual rto_min value */
  43. static inline u32 tcp_rto_min(const struct sock *sk)
  44. {
  45. @@ -901,6 +912,11 @@ static inline u32 tcp_stamp_us_delta(u64 t1, u64 t0)
  46. return max_t(s64, t1 - t0, 0);
  47. }
  48. +static inline u32 tcp_stamp32_us_delta(u32 t1, u32 t0)
  49. +{
  50. + return max_t(s32, t1 - t0, 0);
  51. +}
  52. +
  53. /* provide the departure time in us unit */
  54. static inline u64 tcp_skb_timestamp_us(const struct sk_buff *skb)
  55. {
  56. @@ -989,10 +1005,33 @@ struct tcp_skb_cb {
  57. unused:11;
  58. /* pkts S/ACKed so far upon tx of skb, incl retrans: */
  59. __u32 delivered;
  60. +#ifdef __GENKSYMS__
  61. /* start of send pipeline phase */
  62. u64 first_tx_mstamp;
  63. /* when we reached the "delivered" count */
  64. u64 delivered_mstamp;
  65. +#else
  66. + union {
  67. + u64 __kabi_placeholder_f_tx;
  68. + struct {
  69. + /* start of send pipeline phase */
  70. + u32 first_tx_mstamp;
  71. +#define TCPCB_IN_FLIGHT_BITS 20
  72. +#define TCPCB_IN_FLIGHT_MAX ((1U << TCPCB_IN_FLIGHT_BITS) - 1)
  73. + u32 in_flight:20, /* packets in flight at transmit */
  74. + unused2:12;
  75. + };
  76. + };
  77. + union {
  78. + u64 __kabi_placeholder_del_mst;
  79. + struct {
  80. + /* when we reached the "delivered" count */
  81. + u32 delivered_mstamp;
  82. + /* packets lost so far upon tx of skb */
  83. + u32 lost;
  84. + };
  85. + };
  86. +#endif
  87. } tx; /* only used for outgoing skbs */
  88. union {
  89. struct inet_skb_parm h4;
  90. @@ -1105,6 +1144,9 @@ enum tcp_ca_event {
  91. CA_EVENT_LOSS, /* loss timeout */
  92. CA_EVENT_ECN_NO_CE, /* ECT set, but not CE marked */
  93. CA_EVENT_ECN_IS_CE, /* received CE marked IP packet */
  94. +#ifndef __GENKSYMS__
  95. + CA_EVENT_TLP_RECOVERY, /* a lost segment was repaired by TLP probe */
  96. +#endif
  97. };
  98. /* Information about inbound ACK, passed to cong_ops->in_ack_event() */
  99. @@ -1127,7 +1169,11 @@ enum tcp_ca_ack_event_flags {
  100. #define TCP_CONG_NON_RESTRICTED 0x1
  101. /* Requires ECN/ECT set on all packets */
  102. #define TCP_CONG_NEEDS_ECN 0x2
  103. -#define TCP_CONG_MASK (TCP_CONG_NON_RESTRICTED | TCP_CONG_NEEDS_ECN)
  104. +/* Wants notification of CE events (CA_EVENT_ECN_IS_CE, CA_EVENT_ECN_NO_CE). */
  105. +#define TCP_CONG_WANTS_CE_EVENTS 0x4
  106. +#define TCP_CONG_MASK (TCP_CONG_NON_RESTRICTED | \
  107. + TCP_CONG_NEEDS_ECN | \
  108. + TCP_CONG_WANTS_CE_EVENTS)
  109. union tcp_cc_info;
  110. @@ -1162,6 +1208,13 @@ struct rate_sample {
  111. bool is_app_limited; /* is sample from packet with bubble in pipe? */
  112. bool is_retrans; /* is sample from retransmission? */
  113. bool is_ack_delayed; /* is this (likely) a delayed ACK? */
  114. +#ifndef __GENKSYMS__
  115. + bool is_acking_tlp_retrans_seq; /* ACKed a TLP retransmit sequence? */
  116. + bool is_ece; /* did this ACK have ECN marked? */
  117. + u32 tx_in_flight; /* packets in flight at starting timestamp */
  118. + s32 lost; /* number of packets lost over interval */
  119. + u32 prior_lost; /* tp->lost at "prior_mstamp" */
  120. +#endif
  121. };
  122. struct tcp_congestion_ops {
  123. @@ -1252,6 +1305,14 @@ static inline char *tcp_ca_get_name_by_key(u32 key, char *buffer)
  124. }
  125. #endif
  126. +static inline bool tcp_ca_wants_ce_events(const struct sock *sk)
  127. +{
  128. + const struct inet_connection_sock *icsk = inet_csk(sk);
  129. +
  130. + return icsk->icsk_ca_ops->flags & (TCP_CONG_NEEDS_ECN |
  131. + TCP_CONG_WANTS_CE_EVENTS);
  132. +}
  133. +
  134. static inline bool tcp_ca_needs_ecn(const struct sock *sk)
  135. {
  136. const struct inet_connection_sock *icsk = inet_csk(sk);
  137. @@ -1283,6 +1344,21 @@ static inline bool tcp_skb_sent_after(u64 t1, u64 t2, u32 seq1, u32 seq2)
  138. return t1 > t2 || (t1 == t2 && after(seq1, seq2));
  139. }
  140. +/* If a retransmit failed due to local qdisc congestion or other local issues,
  141. + * then we may have called tcp_set_skb_tso_segs() to increase the number of
  142. + * segments in the skb without increasing the tx.in_flight. In all other cases,
  143. + * the tx.in_flight should be at least as big as the pcount of the sk_buff. We
  144. + * do not have the state to know whether a retransmit failed due to local qdisc
  145. + * congestion or other local issues, so to avoid spurious warnings we consider
  146. + * that any skb marked lost may have suffered that fate.
  147. + */
  148. +static inline bool tcp_skb_tx_in_flight_is_suspicious(u32 skb_pcount,
  149. + u32 skb_sacked_flags,
  150. + u32 tx_in_flight)
  151. +{
  152. + return (skb_pcount > tx_in_flight) && !(skb_sacked_flags & TCPCB_LOST);
  153. +}
  154. +
  155. /* These functions determine how the current flow behaves in respect of SACK
  156. * handling. SACK is negotiated with the peer, and therefore it can vary
  157. * between different flows.
  158. @@ -2447,6 +2523,83 @@ void tcp_plb_update_state(const struct sock *sk, struct tcp_plb_state *plb,
  159. void tcp_plb_check_rehash(struct sock *sk, struct tcp_plb_state *plb);
  160. void tcp_plb_update_state_upon_rto(struct sock *sk, struct tcp_plb_state *plb);
  161. +/* BBR3 congestion control block */
  162. +struct bbr3 {
  163. + u32 min_rtt_us; /* min RTT in min_rtt_win_sec window */
  164. + u32 min_rtt_stamp; /* timestamp of min_rtt_us */
  165. + u32 probe_rtt_done_stamp; /* end time for BBR_PROBE_RTT mode */
  166. + u32 probe_rtt_min_us; /* min RTT in probe_rtt_win_ms win */
  167. + u32 probe_rtt_min_stamp; /* timestamp of probe_rtt_min_us*/
  168. + u32 next_rtt_delivered; /* scb->tx.delivered at end of round */
  169. + u64 cycle_mstamp; /* time of this cycle phase start */
  170. + u32 mode:2, /* current bbr_mode in state machine */
  171. + prev_ca_state:3, /* CA state on previous ACK */
  172. + round_start:1, /* start of packet-timed tx->ack round? */
  173. + ce_state:1, /* If most recent data has CE bit set */
  174. + bw_probe_up_rounds:5, /* cwnd-limited rounds in PROBE_UP */
  175. + try_fast_path:1, /* can we take fast path? */
  176. + idle_restart:1, /* restarting after idle? */
  177. + probe_rtt_round_done:1, /* a BBR_PROBE_RTT round at 4 pkts? */
  178. + init_cwnd:7, /* initial cwnd */
  179. + unused_1:10;
  180. + u32 pacing_gain:10, /* current gain for setting pacing rate */
  181. + cwnd_gain:10, /* current gain for setting cwnd */
  182. + full_bw_reached:1, /* reached full bw in Startup? */
  183. + full_bw_cnt:2, /* number of rounds without large bw gains */
  184. + cycle_idx:2, /* current index in pacing_gain cycle array */
  185. + has_seen_rtt:1, /* have we seen an RTT sample yet? */
  186. + unused_2:6;
  187. + u32 prior_cwnd; /* prior cwnd upon entering loss recovery */
  188. + u32 full_bw; /* recent bw, to estimate if pipe is full */
  189. +
  190. + /* For tracking ACK aggregation: */
  191. + u64 ack_epoch_mstamp; /* start of ACK sampling epoch */
  192. + u16 extra_acked[2]; /* max excess data ACKed in epoch */
  193. + u32 ack_epoch_acked:20, /* packets (S)ACKed in sampling epoch */
  194. + extra_acked_win_rtts:5, /* age of extra_acked, in round trips */
  195. + extra_acked_win_idx:1, /* current index in extra_acked array */
  196. + /* BBR v3 state: */
  197. + full_bw_now:1, /* recently reached full bw plateau? */
  198. + startup_ecn_rounds:2, /* consecutive hi ECN STARTUP rounds */
  199. + loss_in_cycle:1, /* packet loss in this cycle? */
  200. + ecn_in_cycle:1, /* ECN in this cycle? */
  201. + unused_3:1;
  202. + u32 loss_round_delivered; /* scb->tx.delivered ending loss round */
  203. + u32 undo_bw_lo; /* bw_lo before latest losses */
  204. + u32 undo_inflight_lo; /* inflight_lo before latest losses */
  205. + u32 undo_inflight_hi; /* inflight_hi before latest losses */
  206. + u32 bw_latest; /* max delivered bw in last round trip */
  207. + u32 bw_lo; /* lower bound on sending bandwidth */
  208. + u32 bw_hi[2]; /* max recent measured bw sample */
  209. + u32 inflight_latest; /* max delivered data in last round trip */
  210. + u32 inflight_lo; /* lower bound of inflight data range */
  211. + u32 inflight_hi; /* upper bound of inflight data range */
  212. + u32 bw_probe_up_cnt; /* packets delivered per inflight_hi incr */
  213. + u32 bw_probe_up_acks; /* packets (S)ACKed since inflight_hi incr */
  214. + u32 probe_wait_us; /* PROBE_DOWN until next clock-driven probe */
  215. + u32 prior_rcv_nxt; /* tp->rcv_nxt when CE state last changed */
  216. + u32 ecn_eligible:1, /* sender can use ECN (RTT, handshake)? */
  217. + ecn_alpha:9, /* EWMA delivered_ce/delivered; 0..256 */
  218. + bw_probe_samples:1, /* rate samples reflect bw probing? */
  219. + prev_probe_too_high:1, /* did last PROBE_UP go too high? */
  220. + stopped_risky_probe:1, /* last PROBE_UP stopped due to risk? */
  221. + rounds_since_probe:8, /* packet-timed rounds since probed bw */
  222. + loss_round_start:1, /* loss_round_delivered round trip? */
  223. + loss_in_round:1, /* loss marked in this round trip? */
  224. + ecn_in_round:1, /* ECN marked in this round trip? */
  225. + ack_phase:3, /* bbr_ack_phase: meaning of ACKs */
  226. + loss_events_in_round:4,/* losses in STARTUP round */
  227. + initialized:1; /* has bbr_init() been called? */
  228. + u32 alpha_last_delivered; /* tp->delivered at alpha update */
  229. + u32 alpha_last_delivered_ce; /* tp->delivered_ce at alpha update */
  230. +
  231. + u8 unused_4; /* to preserve alignment */
  232. + struct tcp_plb_state plb;
  233. +
  234. + /* react to a specific lost skb (optional) */
  235. + void (*skb_marked_lost)(struct sock *sk, const struct sk_buff *skb);
  236. +};
  237. +
  238. /* At how many usecs into the future should the RTO fire? */
  239. static inline s64 tcp_rto_delta_us(const struct sock *sk)
  240. {
  241. diff --git a/include/uapi/linux/inet_diag.h b/include/uapi/linux/inet_diag.h
  242. index 86bb2e8b1..dcb697f46 100644
  243. --- a/include/uapi/linux/inet_diag.h
  244. +++ b/include/uapi/linux/inet_diag.h
  245. @@ -229,6 +229,31 @@ struct tcp_bbr_info {
  246. __u32 bbr_min_rtt; /* min-filtered RTT in uSec */
  247. __u32 bbr_pacing_gain; /* pacing gain shifted left 8 bits */
  248. __u32 bbr_cwnd_gain; /* cwnd gain shifted left 8 bits */
  249. +#ifndef __GENKSYMS__
  250. + __u32 bbr_bw_hi_lsb; /* lower 32 bits of bw_hi */
  251. + __u32 bbr_bw_hi_msb; /* upper 32 bits of bw_hi */
  252. + __u32 bbr_bw_lo_lsb; /* lower 32 bits of bw_lo */
  253. + __u32 bbr_bw_lo_msb; /* upper 32 bits of bw_lo */
  254. + __u8 bbr_mode; /* current bbr_mode in state machine */
  255. + __u8 bbr_phase; /* current state machine phase */
  256. + __u8 unused1; /* alignment padding; not used yet */
  257. + __u8 bbr_version; /* BBR algorithm version */
  258. + __u32 bbr_inflight_lo; /* lower short-term data volume bound */
  259. + __u32 bbr_inflight_hi; /* higher long-term data volume bound */
  260. + __u32 bbr_extra_acked; /* max excess packets ACKed in epoch */
  261. +#endif
  262. +};
  263. +
  264. +/* TCP BBR congestion control bbr_phase as reported in netlink/ss stats. */
  265. +enum tcp_bbr_phase {
  266. + BBR_PHASE_INVALID = 0,
  267. + BBR_PHASE_STARTUP = 1,
  268. + BBR_PHASE_DRAIN = 2,
  269. + BBR_PHASE_PROBE_RTT = 3,
  270. + BBR_PHASE_PROBE_BW_UP = 4,
  271. + BBR_PHASE_PROBE_BW_DOWN = 5,
  272. + BBR_PHASE_PROBE_BW_CRUISE = 6,
  273. + BBR_PHASE_PROBE_BW_REFILL = 7,
  274. };
  275. union tcp_cc_info {
  276. diff --git a/include/uapi/linux/rtnetlink.h b/include/uapi/linux/rtnetlink.h
  277. index db7254d52..38de18d92 100644
  278. --- a/include/uapi/linux/rtnetlink.h
  279. +++ b/include/uapi/linux/rtnetlink.h
  280. @@ -507,12 +507,14 @@ enum {
  281. #define RTAX_FEATURE_TIMESTAMP (1 << 2) /* unused */
  282. #define RTAX_FEATURE_ALLFRAG (1 << 3) /* unused */
  283. #define RTAX_FEATURE_TCP_USEC_TS (1 << 4)
  284. +#define RTAX_FEATURE_ECN_LOW (1 << 5)
  285. #define RTAX_FEATURE_MASK (RTAX_FEATURE_ECN | \
  286. RTAX_FEATURE_SACK | \
  287. RTAX_FEATURE_TIMESTAMP | \
  288. RTAX_FEATURE_ALLFRAG | \
  289. - RTAX_FEATURE_TCP_USEC_TS)
  290. + RTAX_FEATURE_TCP_USEC_TS | \
  291. + RTAX_FEATURE_ECN_LOW)
  292. struct rta_session {
  293. __u8 proto;
  294. diff --git a/net/ipv4/Kconfig b/net/ipv4/Kconfig
  295. index 4cfd87b28..3999393b1 100644
  296. --- a/net/ipv4/Kconfig
  297. +++ b/net/ipv4/Kconfig
  298. @@ -679,6 +679,24 @@ config TCP_CONG_BBR
  299. AQM schemes that do not provide a delay signal. It requires the fq
  300. ("Fair Queue") pacing packet scheduler.
  301. +config TCP_CONG_BBR3
  302. + tristate "BBRv3 TCP"
  303. + default n
  304. + help
  305. +
  306. + BBRv3 (Bottleneck Bandwidth and RTT version 3) TCP congestion control is a
  307. + model-based congestion control algorithm that aims to maximize
  308. + network utilization, keep queues and retransmit rates low, and to be
  309. + able to coexist with Reno/CUBIC in common scenarios. It builds an
  310. + explicit model of the network path. It tolerates a targeted degree
  311. + of random packet loss and delay. It can operate over LAN, WAN,
  312. + cellular, wifi, or cable modem links, and can use shallow-threshold
  313. + ECN signals. It can coexist to some degree with flows that use
  314. + loss-based congestion control, and can operate with shallow buffers,
  315. + deep buffers, bufferbloat, policers, or AQM schemes that do not
  316. + provide a delay signal. It requires pacing, using either TCP internal
  317. + pacing or the fq ("Fair Queue") pacing packet scheduler.
  318. +
  319. choice
  320. prompt "Default TCP congestion control"
  321. default DEFAULT_CUBIC
  322. @@ -716,6 +734,9 @@ choice
  323. config DEFAULT_BBR
  324. bool "BBR" if TCP_CONG_BBR=y
  325. + config DEFAULT_BBR3
  326. + bool "BBR3" if TCP_CONG_BBR3=y
  327. +
  328. config DEFAULT_RENO
  329. bool "Reno"
  330. endchoice
  331. @@ -740,6 +761,7 @@ config DEFAULT_TCP_CONG
  332. default "dctcp" if DEFAULT_DCTCP
  333. default "cdg" if DEFAULT_CDG
  334. default "bbr" if DEFAULT_BBR
  335. + default "bbr3" if DEFAULT_BBR3
  336. default "cubic"
  337. config TCP_SIGPOOL
  338. diff --git a/net/ipv4/Makefile b/net/ipv4/Makefile
  339. index ec36d2ec0..45324da63 100644
  340. --- a/net/ipv4/Makefile
  341. +++ b/net/ipv4/Makefile
  342. @@ -45,6 +45,7 @@ obj-$(CONFIG_INET_TCP_DIAG) += tcp_diag.o
  343. obj-$(CONFIG_INET_UDP_DIAG) += udp_diag.o
  344. obj-$(CONFIG_INET_RAW_DIAG) += raw_diag.o
  345. obj-$(CONFIG_TCP_CONG_BBR) += tcp_bbr.o
  346. +obj-$(CONFIG_TCP_CONG_BBR3) += tcp_bbr3.o
  347. obj-$(CONFIG_TCP_CONG_BIC) += tcp_bic.o
  348. obj-$(CONFIG_TCP_CONG_CDG) += tcp_cdg.o
  349. obj-$(CONFIG_TCP_CONG_CUBIC) += tcp_cubic.o
  350. diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c
  351. index 13c7b6b38..e2dd56670 100644
  352. --- a/net/ipv4/tcp.c
  353. +++ b/net/ipv4/tcp.c
  354. @@ -3415,6 +3415,7 @@ int tcp_disconnect(struct sock *sk, int flags)
  355. tp->rx_opt.dsack = 0;
  356. tp->rx_opt.num_sacks = 0;
  357. tp->rcv_ooopack = 0;
  358. + tp->fast_ack_mode = 0;
  359. /* Clean up fastopen related fields */
  360. diff --git a/net/ipv4/tcp_bbr3.c b/net/ipv4/tcp_bbr3.c
  361. new file mode 100644
  362. index 000000000..9d8885865
  363. --- /dev/null
  364. +++ b/net/ipv4/tcp_bbr3.c
  365. @@ -0,0 +1,2359 @@
  366. +/* BBR (Bottleneck Bandwidth and RTT) congestion control
  367. + *
  368. + * BBR is a model-based congestion control algorithm that aims for low queues,
  369. + * low loss, and (bounded) Reno/CUBIC coexistence. To maintain a model of the
  370. + * network path, it uses measurements of bandwidth and RTT, as well as (if they
  371. + * occur) packet loss and/or shallow-threshold ECN signals. Note that although
  372. + * it can use ECN or loss signals explicitly, it does not require either; it
  373. + * can bound its in-flight data based on its estimate of the BDP.
  374. + *
  375. + * The model has both higher and lower bounds for the operating range:
  376. + * lo: bw_lo, inflight_lo: conservative short-term lower bound
  377. + * hi: bw_hi, inflight_hi: robust long-term upper bound
  378. + * The bandwidth-probing time scale is (a) extended dynamically based on
  379. + * estimated BDP to improve coexistence with Reno/CUBIC; (b) bounded by
  380. + * an interactive wall-clock time-scale to be more scalable and responsive
  381. + * than Reno and CUBIC.
  382. + *
  383. + * Here is a state transition diagram for BBR:
  384. + *
  385. + * |
  386. + * V
  387. + * +---> STARTUP ----+
  388. + * | | |
  389. + * | V |
  390. + * | DRAIN ----+
  391. + * | | |
  392. + * | V |
  393. + * +---> PROBE_BW ----+
  394. + * | ^ | |
  395. + * | | | |
  396. + * | +----+ |
  397. + * | |
  398. + * +---- PROBE_RTT <--+
  399. + *
  400. + * A BBR flow starts in STARTUP, and ramps up its sending rate quickly.
  401. + * When it estimates the pipe is full, it enters DRAIN to drain the queue.
  402. + * In steady state a BBR flow only uses PROBE_BW and PROBE_RTT.
  403. + * A long-lived BBR flow spends the vast majority of its time remaining
  404. + * (repeatedly) in PROBE_BW, fully probing and utilizing the pipe's bandwidth
  405. + * in a fair manner, with a small, bounded queue. *If* a flow has been
  406. + * continuously sending for the entire min_rtt window, and hasn't seen an RTT
  407. + * sample that matches or decreases its min_rtt estimate for 10 seconds, then
  408. + * it briefly enters PROBE_RTT to cut inflight to a minimum value to re-probe
  409. + * the path's two-way propagation delay (min_rtt). When exiting PROBE_RTT, if
  410. + * we estimated that we reached the full bw of the pipe then we enter PROBE_BW;
  411. + * otherwise we enter STARTUP to try to fill the pipe.
  412. + *
  413. + * BBR is described in detail in:
  414. + * "BBR: Congestion-Based Congestion Control",
  415. + * Neal Cardwell, Yuchung Cheng, C. Stephen Gunn, Soheil Hassas Yeganeh,
  416. + * Van Jacobson. ACM Queue, Vol. 14 No. 5, September-October 2016.
  417. + *
  418. + * There is a public e-mail list for discussing BBR development and testing:
  419. + * https://groups.google.com/forum/#!forum/bbr-dev
  420. + *
  421. + * NOTE: BBR might be used with the fq qdisc ("man tc-fq") with pacing enabled,
  422. + * otherwise TCP stack falls back to an internal pacing using one high
  423. + * resolution timer per TCP socket and may use more resources.
  424. + */
  425. +
  426. +#include <linux/module.h>
  427. +#include <net/tcp.h>
  428. +#include <linux/inet_diag.h>
  429. +#include <linux/inet.h>
  430. +#include <linux/random.h>
  431. +#include <linux/win_minmax.h>
  432. +
  433. +#include <trace/events/tcp.h>
  434. +#include "tcp_dctcp.h"
  435. +
  436. +#define BBR_VERSION 3
  437. +
  438. +#define bbr_param(sk,name) (bbr_ ## name)
  439. +
  440. +/* Scale factor for rate in pkt/uSec unit to avoid truncation in bandwidth
  441. + * estimation. The rate unit ~= (1500 bytes / 1 usec / 2^24) ~= 715 bps.
  442. + * This handles bandwidths from 0.06pps (715bps) to 256Mpps (3Tbps) in a u32.
  443. + * Since the minimum window is >=4 packets, the lower bound isn't
  444. + * an issue. The upper bound isn't an issue with existing technologies.
  445. + */
  446. +#define BW_SCALE 24
  447. +#define BW_UNIT (1 << BW_SCALE)
  448. +
  449. +#define BBR_SCALE 8 /* scaling factor for fractions in BBR (e.g. gains) */
  450. +#define BBR_UNIT (1 << BBR_SCALE)
  451. +
  452. +/* BBR has the following modes for deciding how fast to send: */
  453. +enum bbr_mode {
  454. + BBR_STARTUP, /* ramp up sending rate rapidly to fill pipe */
  455. + BBR_DRAIN, /* drain any queue created during startup */
  456. + BBR_PROBE_BW, /* discover, share bw: pace around estimated bw */
  457. + BBR_PROBE_RTT, /* cut inflight to min to probe min_rtt */
  458. +};
  459. +
  460. +/* How does the incoming ACK stream relate to our bandwidth probing? */
  461. +enum bbr_ack_phase {
  462. + BBR_ACKS_INIT, /* not probing; not getting probe feedback */
  463. + BBR_ACKS_REFILLING, /* sending at est. bw to fill pipe */
  464. + BBR_ACKS_PROBE_STARTING, /* inflight rising to probe bw */
  465. + BBR_ACKS_PROBE_FEEDBACK, /* getting feedback from bw probing */
  466. + BBR_ACKS_PROBE_STOPPING, /* stopped probing; still getting feedback */
  467. +};
  468. +
  469. +struct bbr_context {
  470. + u32 sample_bw;
  471. +};
  472. +
  473. +/* Window length of min_rtt filter (in sec): */
  474. +static const u32 bbr_min_rtt_win_sec = 10;
  475. +/* Minimum time (in ms) spent at bbr_cwnd_min_target in BBR_PROBE_RTT mode: */
  476. +static const u32 bbr_probe_rtt_mode_ms = 200;
  477. +/* Window length of probe_rtt_min_us filter (in ms), and consequently the
  478. + * typical interval between PROBE_RTT mode entries. The default is 5000ms.
  479. + * Note that bbr_probe_rtt_win_ms must be <= bbr_min_rtt_win_sec * MSEC_PER_SEC
  480. + */
  481. +static const u32 bbr_probe_rtt_win_ms = 5000;
  482. +/* Proportion of cwnd to estimated BDP in PROBE_RTT, in units of BBR_UNIT: */
  483. +static const u32 bbr_probe_rtt_cwnd_gain = BBR_UNIT * 1 / 2;
  484. +
  485. +/* Use min_rtt to help adapt TSO burst size, with smaller min_rtt resulting
  486. + * in bigger TSO bursts. We cut the RTT-based allowance in half
  487. + * for every 2^9 usec (aka 512 us) of RTT, so that the RTT-based allowance
  488. + * is below 1500 bytes after 6 * ~500 usec = 3ms.
  489. + */
  490. +static const u32 bbr_tso_rtt_shift = 9;
  491. +
  492. +/* Pace at ~1% below estimated bw, on average, to reduce queue at bottleneck.
  493. + * In order to help drive the network toward lower queues and low latency while
  494. + * maintaining high utilization, the average pacing rate aims to be slightly
  495. + * lower than the estimated bandwidth. This is an important aspect of the
  496. + * design.
  497. + */
  498. +static const int bbr_pacing_margin_percent = 1;
  499. +
  500. +/* We use a startup_pacing_gain of 4*ln(2) because it's the smallest value
  501. + * that will allow a smoothly increasing pacing rate that will double each RTT
  502. + * and send the same number of packets per RTT that an un-paced, slow-starting
  503. + * Reno or CUBIC flow would:
  504. + */
  505. +static const int bbr_startup_pacing_gain = BBR_UNIT * 277 / 100 + 1;
  506. +/* The gain for deriving startup cwnd: */
  507. +static const int bbr_startup_cwnd_gain = BBR_UNIT * 2;
  508. +/* The pacing gain in BBR_DRAIN is calculated to typically drain
  509. + * the queue created in BBR_STARTUP in a single round:
  510. + */
  511. +static const int bbr_drain_gain = BBR_UNIT * 1000 / 2885;
  512. +/* The gain for deriving steady-state cwnd tolerates delayed/stretched ACKs: */
  513. +static const int bbr_cwnd_gain = BBR_UNIT * 2;
  514. +/* The pacing_gain values for the PROBE_BW gain cycle, to discover/share bw: */
  515. +static const int bbr_pacing_gain[] = {
  516. + BBR_UNIT * 5 / 4, /* UP: probe for more available bw */
  517. + BBR_UNIT * 91 / 100, /* DOWN: drain queue and/or yield bw */
  518. + BBR_UNIT, /* CRUISE: try to use pipe w/ some headroom */
  519. + BBR_UNIT, /* REFILL: refill pipe to estimated 100% */
  520. +};
  521. +enum bbr_pacing_gain_phase {
  522. + BBR_BW_PROBE_UP = 0, /* push up inflight to probe for bw/vol */
  523. + BBR_BW_PROBE_DOWN = 1, /* drain excess inflight from the queue */
  524. + BBR_BW_PROBE_CRUISE = 2, /* use pipe, w/ headroom in queue/pipe */
  525. + BBR_BW_PROBE_REFILL = 3, /* refill the pipe again to 100% */
  526. +};
  527. +
  528. +/* Try to keep at least this many packets in flight, if things go smoothly. For
  529. + * smooth functioning, a sliding window protocol ACKing every other packet
  530. + * needs at least 4 packets in flight:
  531. + */
  532. +static const u32 bbr_cwnd_min_target = 4;
  533. +
  534. +/* To estimate if BBR_STARTUP or BBR_BW_PROBE_UP has filled pipe... */
  535. +/* If bw has increased significantly (1.25x), there may be more bw available: */
  536. +static const u32 bbr_full_bw_thresh = BBR_UNIT * 5 / 4;
  537. +/* But after 3 rounds w/o significant bw growth, estimate pipe is full: */
  538. +static const u32 bbr_full_bw_cnt = 3;
  539. +
  540. +/* Gain factor for adding extra_acked to target cwnd: */
  541. +static const int bbr_extra_acked_gain = BBR_UNIT;
  542. +/* Window length of extra_acked window. */
  543. +static const u32 bbr_extra_acked_win_rtts = 5;
  544. +/* Max allowed val for ack_epoch_acked, after which sampling epoch is reset */
  545. +static const u32 bbr_ack_epoch_acked_reset_thresh = 1U << 20;
  546. +/* Time period for clamping cwnd increment due to ack aggregation */
  547. +static const u32 bbr_extra_acked_max_us = 100 * 1000;
  548. +
  549. +/* Flags to control BBR ECN-related behavior... */
  550. +
  551. +/* Ensure ACKs only ACK packets with consistent ECN CE status? */
  552. +static const bool bbr_precise_ece_ack = true;
  553. +
  554. +/* Max RTT (in usec) at which to use sender-side ECN logic.
  555. + * Disabled when 0 (ECN allowed at any RTT).
  556. + */
  557. +static const u32 bbr_ecn_max_rtt_us = 5000;
  558. +
  559. +/* On losses, scale down inflight and pacing rate by beta scaled by BBR_SCALE.
  560. + * No loss response when 0.
  561. + */
  562. +static const u32 bbr_beta = BBR_UNIT * 30 / 100;
  563. +
  564. +/* Gain factor for ECN mark ratio samples, scaled by BBR_SCALE (1/16 = 6.25%) */
  565. +static const u32 bbr_ecn_alpha_gain = BBR_UNIT * 1 / 16;
  566. +
  567. +/* The initial value for ecn_alpha; 1.0 allows a flow to respond quickly
  568. + * to congestion if the bottleneck is congested when the flow starts up.
  569. + */
  570. +static const u32 bbr_ecn_alpha_init = BBR_UNIT;
  571. +
  572. +/* On ECN, cut inflight_lo to (1 - ecn_factor * ecn_alpha) scaled by BBR_SCALE.
  573. + * No ECN based bounding when 0.
  574. + */
  575. +static const u32 bbr_ecn_factor = BBR_UNIT * 1 / 3; /* 1/3 = 33% */
  576. +
  577. +/* Estimate bw probing has gone too far if CE ratio exceeds this threshold.
  578. + * Scaled by BBR_SCALE. Disabled when 0.
  579. + */
  580. +static const u32 bbr_ecn_thresh = BBR_UNIT * 1 / 2; /* 1/2 = 50% */
  581. +
  582. +/* If non-zero, if in a cycle with no losses but some ECN marks, after ECN
  583. + * clears then make the first round's increment to inflight_hi the following
  584. + * fraction of inflight_hi.
  585. + */
  586. +static const u32 bbr_ecn_reprobe_gain = BBR_UNIT * 1 / 2;
  587. +
  588. +/* Estimate bw probing has gone too far if loss rate exceeds this level. */
  589. +static const u32 bbr_loss_thresh = BBR_UNIT * 2 / 100; /* 2% loss */
  590. +
  591. +/* Slow down for a packet loss recovered by TLP? */
  592. +static const bool bbr_loss_probe_recovery = true;
  593. +
  594. +/* Exit STARTUP if number of loss marking events in a Recovery round is >= N,
  595. + * and loss rate is higher than bbr_loss_thresh.
  596. + * Disabled if 0.
  597. + */
  598. +static const u32 bbr_full_loss_cnt = 6;
  599. +
  600. +/* Exit STARTUP if number of round trips with ECN mark rate above ecn_thresh
  601. + * meets this count.
  602. + */
  603. +static const u32 bbr_full_ecn_cnt = 2;
  604. +
  605. +/* Fraction of unutilized headroom to try to leave in path upon high loss. */
  606. +static const u32 bbr_inflight_headroom = BBR_UNIT * 15 / 100;
  607. +
  608. +/* How much do we increase cwnd_gain when probing for bandwidth in
  609. + * BBR_BW_PROBE_UP? This specifies the increment in units of
  610. + * BBR_UNIT/4. The default is 1, meaning 0.25.
  611. + * The min value is 0 (meaning 0.0); max is 3 (meaning 0.75).
  612. + */
  613. +static const u32 bbr_bw_probe_cwnd_gain = 1;
  614. +
  615. +/* Max number of packet-timed rounds to wait before probing for bandwidth. If
  616. + * we want to tolerate 1% random loss per round, and not have this cut our
  617. + * inflight too much, we must probe for bw periodically on roughly this scale.
  618. + * If low, limits Reno/CUBIC coexistence; if high, limits loss tolerance.
  619. + * We aim to be fair with Reno/CUBIC up to a BDP of at least:
  620. + * BDP = 25Mbps * .030sec /(1514bytes) = 61.9 packets
  621. + */
  622. +static const u32 bbr_bw_probe_max_rounds = 63;
  623. +
  624. +/* Max amount of randomness to inject in round counting for Reno-coexistence.
  625. + */
  626. +static const u32 bbr_bw_probe_rand_rounds = 2;
  627. +
  628. +/* Use BBR-native probe time scale starting at this many usec.
  629. + * We aim to be fair with Reno/CUBIC up to an inter-loss time epoch of at least:
  630. + * BDP*RTT = 25Mbps * .030sec /(1514bytes) * 0.030sec = 1.9 secs
  631. + */
  632. +static const u32 bbr_bw_probe_base_us = 2 * USEC_PER_SEC; /* 2 secs */
  633. +
  634. +/* Use BBR-native probes spread over this many usec: */
  635. +static const u32 bbr_bw_probe_rand_us = 1 * USEC_PER_SEC; /* 1 secs */
  636. +
  637. +/* Use fast path if app-limited, no loss/ECN, and target cwnd was reached? */
  638. +static const bool bbr_fast_path = true;
  639. +
  640. +/* Use fast ack mode? */
  641. +static const bool bbr_fast_ack_mode = true;
  642. +
  643. +static u32 bbr_max_bw(const struct sock *sk);
  644. +static u32 bbr_bw(const struct sock *sk);
  645. +static void bbr_exit_probe_rtt(struct sock *sk);
  646. +static void bbr_reset_congestion_signals(struct sock *sk);
  647. +static void bbr_run_loss_probe_recovery(struct sock *sk);
  648. +
  649. +static void bbr_check_probe_rtt_done(struct sock *sk);
  650. +
  651. +static void bbr3_skb_marked_lost(struct sock *sk, const struct sk_buff *skb);
  652. +
  653. +static inline struct bbr3 *bbr3_get_priv(const struct sock *sk)
  654. +{
  655. + /* Read the address value out of slot 0 of the 104-byte array space */
  656. + return *(struct bbr3 **)inet_csk_ca(sk);
  657. +}
  658. +
  659. +/* This connection can use ECN if both endpoints have signaled ECN support in
  660. + * the handshake and the per-route settings indicated this is a
  661. + * shallow-threshold ECN environment, meaning both:
  662. + * (a) ECN CE marks indicate low-latency/shallow-threshold congestion, and
  663. + * (b) TCP endpoints provide precise ACKs that only ACK data segments
  664. + * with consistent ECN CE status
  665. + */
  666. +static bool bbr_can_use_ecn(const struct sock *sk)
  667. +{
  668. + return (tcp_sk(sk)->ecn_flags & TCP_ECN_OK) &&
  669. + (tcp_sk(sk)->ecn_flags & TCP_ECN_LOW);
  670. +}
  671. +
  672. +/* Do we estimate that STARTUP filled the pipe? */
  673. +static bool bbr_full_bw_reached(const struct sock *sk)
  674. +{
  675. + const struct bbr3 *bbr = bbr3_get_priv(sk);
  676. +
  677. + return bbr->full_bw_reached;
  678. +}
  679. +
  680. +/* Return the windowed max recent bandwidth sample, in pkts/uS << BW_SCALE. */
  681. +static u32 bbr_max_bw(const struct sock *sk)
  682. +{
  683. + const struct bbr3 *bbr = bbr3_get_priv(sk);
  684. +
  685. + return max(bbr->bw_hi[0], bbr->bw_hi[1]);
  686. +}
  687. +
  688. +/* Return the estimated bandwidth of the path, in pkts/uS << BW_SCALE. */
  689. +static u32 bbr_bw(const struct sock *sk)
  690. +{
  691. + const struct bbr3 *bbr = bbr3_get_priv(sk);
  692. +
  693. + return min(bbr_max_bw(sk), bbr->bw_lo);
  694. +}
  695. +
  696. +/* Return maximum extra acked in past k-2k round trips,
  697. + * where k = bbr_extra_acked_win_rtts.
  698. + */
  699. +static u16 bbr_extra_acked(const struct sock *sk)
  700. +{
  701. + struct bbr3 *bbr = bbr3_get_priv(sk);
  702. +
  703. + return max(bbr->extra_acked[0], bbr->extra_acked[1]);
  704. +}
  705. +
  706. +/* Return rate in bytes per second, optionally with a gain.
  707. + * The order here is chosen carefully to avoid overflow of u64. This should
  708. + * work for input rates of up to 2.9Tbit/sec and gain of 2.89x.
  709. + */
  710. +static u64 bbr_rate_bytes_per_sec(struct sock *sk, u64 rate, int gain,
  711. + int margin)
  712. +{
  713. + unsigned int mss = tcp_sk(sk)->mss_cache;
  714. +
  715. + rate *= mss;
  716. + rate *= gain;
  717. + rate >>= BBR_SCALE;
  718. + rate *= USEC_PER_SEC / 100 * (100 - margin);
  719. + rate >>= BW_SCALE;
  720. + rate = max(rate, 1ULL);
  721. + return rate;
  722. +}
  723. +
  724. +static u64 bbr_bw_bytes_per_sec(struct sock *sk, u64 rate)
  725. +{
  726. + return bbr_rate_bytes_per_sec(sk, rate, BBR_UNIT, 0);
  727. +}
  728. +
  729. +/* Convert a BBR bw and gain factor to a pacing rate in bytes per second. */
  730. +static unsigned long bbr_bw_to_pacing_rate(struct sock *sk, u32 bw, int gain)
  731. +{
  732. + u64 rate = bw;
  733. +
  734. + rate = bbr_rate_bytes_per_sec(sk, rate, gain,
  735. + bbr_pacing_margin_percent);
  736. + rate = min_t(u64, rate, READ_ONCE(sk->sk_max_pacing_rate));
  737. + return rate;
  738. +}
  739. +
  740. +/* Initialize pacing rate to: startup_pacing_gain * init_cwnd / RTT. */
  741. +static void bbr_init_pacing_rate_from_rtt(struct sock *sk)
  742. +{
  743. + struct tcp_sock *tp = tcp_sk(sk);
  744. + struct bbr3 *bbr = bbr3_get_priv(sk);
  745. + u64 bw;
  746. + u32 rtt_us;
  747. +
  748. + if (tp->srtt_us) { /* any RTT sample yet? */
  749. + rtt_us = max(tp->srtt_us >> 3, 1U);
  750. + bbr->has_seen_rtt = 1;
  751. + } else { /* no RTT sample yet */
  752. + rtt_us = USEC_PER_MSEC; /* use nominal default RTT */
  753. + }
  754. + bw = (u64)tcp_snd_cwnd(tp) * BW_UNIT;
  755. + do_div(bw, rtt_us);
  756. + WRITE_ONCE(sk->sk_pacing_rate,
  757. + bbr_bw_to_pacing_rate(sk, bw,
  758. + bbr_param(sk, startup_pacing_gain)));
  759. +}
  760. +
  761. +/* Pace using current bw estimate and a gain factor. */
  762. +static void bbr_set_pacing_rate(struct sock *sk, u32 bw, int gain)
  763. +{
  764. + struct tcp_sock *tp = tcp_sk(sk);
  765. + struct bbr3 *bbr = bbr3_get_priv(sk);
  766. + unsigned long rate = bbr_bw_to_pacing_rate(sk, bw, gain);
  767. +
  768. + if (unlikely(!bbr->has_seen_rtt && tp->srtt_us))
  769. + bbr_init_pacing_rate_from_rtt(sk);
  770. + if (bbr_full_bw_reached(sk) || rate > READ_ONCE(sk->sk_pacing_rate))
  771. + WRITE_ONCE(sk->sk_pacing_rate, rate);
  772. +}
  773. +
  774. +/* Return the number of segments BBR would like in a TSO/GSO skb, given a
  775. + * particular max gso size as a constraint. TODO: make this simpler and more
  776. + * consistent by switching bbr to just call tcp_tso_autosize().
  777. + */
  778. +static u32 bbr_tso_segs_generic(struct sock *sk, unsigned int mss_now,
  779. + u32 gso_max_size)
  780. +{
  781. + struct bbr3 *bbr = bbr3_get_priv(sk);
  782. + u32 segs, r;
  783. + u64 bytes;
  784. +
  785. + if(!bbr || !bbr->initialized)
  786. + return 2;
  787. +
  788. + /* Budget a TSO/GSO burst size allowance based on bw (pacing_rate). */
  789. + bytes = READ_ONCE(sk->sk_pacing_rate) >> READ_ONCE(sk->sk_pacing_shift);
  790. +
  791. + /* Budget a TSO/GSO burst size allowance based on min_rtt. For every
  792. + * K = 2^tso_rtt_shift microseconds of min_rtt, halve the burst.
  793. + * The min_rtt-based burst allowance is: 64 KBytes / 2^(min_rtt/K)
  794. + */
  795. + if (bbr_param(sk, tso_rtt_shift)) {
  796. + r = bbr->min_rtt_us >> bbr_param(sk, tso_rtt_shift);
  797. + if (r < BITS_PER_TYPE(u32)) /* prevent undefined behavior */
  798. + bytes += GSO_LEGACY_MAX_SIZE >> r;
  799. + }
  800. +
  801. + bytes = min_t(u32, bytes, gso_max_size - 1 - MAX_TCP_HEADER);
  802. + segs = max_t(u32, bytes / mss_now,
  803. + sock_net(sk)->ipv4.sysctl_tcp_min_tso_segs);
  804. + return segs;
  805. +}
  806. +
  807. +/* Custom tcp_tso_autosize() for BBR, used at transmit time to cap skb size. */
  808. +static u32 bbr3_tso_segs(struct sock *sk)
  809. +{
  810. + unsigned int mss_now = tcp_current_mss(sk);
  811. + return bbr_tso_segs_generic(sk, mss_now, sk->sk_gso_max_size);
  812. +}
  813. +
  814. +/* Like bbr2_tso_segs(), using mss_cache, ignoring driver's sk_gso_max_size. */
  815. +static u32 bbr_tso_segs_goal(struct sock *sk)
  816. +{
  817. + struct tcp_sock *tp = tcp_sk(sk);
  818. +
  819. + return bbr_tso_segs_generic(sk, tp->mss_cache, GSO_LEGACY_MAX_SIZE);
  820. +}
  821. +
  822. +/* Save "last known good" cwnd so we can restore it after losses or PROBE_RTT */
  823. +static void bbr_save_cwnd(struct sock *sk)
  824. +{
  825. + struct tcp_sock *tp = tcp_sk(sk);
  826. + struct bbr3 *bbr = bbr3_get_priv(sk);
  827. +
  828. + if (bbr->prev_ca_state < TCP_CA_Recovery && bbr->mode != BBR_PROBE_RTT)
  829. + bbr->prior_cwnd = tcp_snd_cwnd(tp); /* this cwnd is good enough */
  830. + else /* loss recovery or BBR_PROBE_RTT have temporarily cut cwnd */
  831. + bbr->prior_cwnd = max(bbr->prior_cwnd, tcp_snd_cwnd(tp));
  832. +}
  833. +
  834. +static void bbr3_cwnd_event(struct sock *sk, enum tcp_ca_event event)
  835. +{
  836. + struct tcp_sock *tp = tcp_sk(sk);
  837. + struct bbr3 *bbr = bbr3_get_priv(sk);
  838. +
  839. + if (event == CA_EVENT_TX_START) {
  840. + if (!tp->app_limited)
  841. + return;
  842. + bbr->idle_restart = 1;
  843. + bbr->ack_epoch_mstamp = tp->tcp_mstamp;
  844. + bbr->ack_epoch_acked = 0;
  845. + /* Avoid pointless buffer overflows: pace at est. bw if we don't
  846. + * need more speed (we're restarting from idle and app-limited).
  847. + */
  848. + if (bbr->mode == BBR_PROBE_BW)
  849. + bbr_set_pacing_rate(sk, bbr_bw(sk), BBR_UNIT);
  850. + else if (bbr->mode == BBR_PROBE_RTT)
  851. + bbr_check_probe_rtt_done(sk);
  852. + } else if ((event == CA_EVENT_ECN_IS_CE ||
  853. + event == CA_EVENT_ECN_NO_CE) &&
  854. + bbr_can_use_ecn(sk) &&
  855. + bbr_param(sk, precise_ece_ack)) {
  856. + u32 state = bbr->ce_state;
  857. + dctcp_ece_ack_update(sk, event, &bbr->prior_rcv_nxt, &state);
  858. + bbr->ce_state = state;
  859. + } else if (event == CA_EVENT_TLP_RECOVERY &&
  860. + bbr_param(sk, loss_probe_recovery)) {
  861. + bbr_run_loss_probe_recovery(sk);
  862. + }
  863. +}
  864. +
  865. +/* Calculate bdp based on min RTT and the estimated bottleneck bandwidth:
  866. + *
  867. + * bdp = ceil(bw * min_rtt * gain)
  868. + *
  869. + * The key factor, gain, controls the amount of queue. While a small gain
  870. + * builds a smaller queue, it becomes more vulnerable to noise in RTT
  871. + * measurements (e.g., delayed ACKs or other ACK compression effects). This
  872. + * noise may cause BBR to under-estimate the rate.
  873. + */
  874. +static u32 bbr_bdp(struct sock *sk, u32 bw, int gain)
  875. +{
  876. + struct bbr3 *bbr = bbr3_get_priv(sk);
  877. + u32 bdp;
  878. + u64 w;
  879. +
  880. + /* If we've never had a valid RTT sample, cap cwnd at the initial
  881. + * default. This should only happen when the connection is not using TCP
  882. + * timestamps and has retransmitted all of the SYN/SYNACK/data packets
  883. + * ACKed so far. In this case, an RTO can cut cwnd to 1, in which
  884. + * case we need to slow-start up toward something safe: initial cwnd.
  885. + */
  886. + if (unlikely(bbr->min_rtt_us == ~0U)) /* no valid RTT samples yet? */
  887. + return bbr->init_cwnd; /* be safe: cap at initial cwnd */
  888. +
  889. + w = (u64)bw * bbr->min_rtt_us;
  890. +
  891. + /* Apply a gain to the given value, remove the BW_SCALE shift, and
  892. + * round the value up to avoid a negative feedback loop.
  893. + */
  894. + bdp = (((w * gain) >> BBR_SCALE) + BW_UNIT - 1) / BW_UNIT;
  895. +
  896. + return bdp;
  897. +}
  898. +
  899. +/* To achieve full performance in high-speed paths, we budget enough cwnd to
  900. + * fit full-sized skbs in-flight on both end hosts to fully utilize the path:
  901. + * - one skb in sending host Qdisc,
  902. + * - one skb in sending host TSO/GSO engine
  903. + * - one skb being received by receiver host LRO/GRO/delayed-ACK engine
  904. + * Don't worry, at low rates this won't bloat cwnd because
  905. + * in such cases tso_segs_goal is small. The minimum cwnd is 4 packets,
  906. + * which allows 2 outstanding 2-packet sequences, to try to keep pipe
  907. + * full even with ACK-every-other-packet delayed ACKs.
  908. + */
  909. +static u32 bbr_quantization_budget(struct sock *sk, u32 cwnd)
  910. +{
  911. + struct bbr3 *bbr = bbr3_get_priv(sk);
  912. + u32 tso_segs_goal;
  913. +
  914. + tso_segs_goal = 3 * bbr_tso_segs_goal(sk);
  915. +
  916. + /* Allow enough full-sized skbs in flight to utilize end systems. */
  917. + cwnd = max_t(u32, cwnd, tso_segs_goal);
  918. + cwnd = max_t(u32, cwnd, bbr_param(sk, cwnd_min_target));
  919. + /* Ensure gain cycling gets inflight above BDP even for small BDPs. */
  920. + if (bbr->mode == BBR_PROBE_BW && bbr->cycle_idx == BBR_BW_PROBE_UP)
  921. + cwnd += 2;
  922. +
  923. + return cwnd;
  924. +}
  925. +
  926. +/* Find inflight based on min RTT and the estimated bottleneck bandwidth. */
  927. +static u32 bbr_inflight(struct sock *sk, u32 bw, int gain)
  928. +{
  929. + u32 inflight;
  930. +
  931. + inflight = bbr_bdp(sk, bw, gain);
  932. + inflight = bbr_quantization_budget(sk, inflight);
  933. +
  934. + return inflight;
  935. +}
  936. +
  937. +/* With pacing at lower layers, there's often less data "in the network" than
  938. + * "in flight". With TSQ and departure time pacing at lower layers (e.g. fq),
  939. + * we often have several skbs queued in the pacing layer with a pre-scheduled
  940. + * earliest departure time (EDT). BBR adapts its pacing rate based on the
  941. + * inflight level that it estimates has already been "baked in" by previous
  942. + * departure time decisions. We calculate a rough estimate of the number of our
  943. + * packets that might be in the network at the earliest departure time for the
  944. + * next skb scheduled:
  945. + * in_network_at_edt = inflight_at_edt - (EDT - now) * bw
  946. + * If we're increasing inflight, then we want to know if the transmit of the
  947. + * EDT skb will push inflight above the target, so inflight_at_edt includes
  948. + * bbr_tso_segs_goal() from the skb departing at EDT. If decreasing inflight,
  949. + * then estimate if inflight will sink too low just before the EDT transmit.
  950. + */
  951. +static u32 bbr_packets_in_net_at_edt(struct sock *sk, u32 inflight_now)
  952. +{
  953. + struct tcp_sock *tp = tcp_sk(sk);
  954. + struct bbr3 *bbr = bbr3_get_priv(sk);
  955. + u64 now_ns, edt_ns, interval_us;
  956. + u32 interval_delivered, inflight_at_edt;
  957. +
  958. + now_ns = tp->tcp_clock_cache;
  959. + edt_ns = max(tp->tcp_wstamp_ns, now_ns);
  960. + interval_us = div_u64(edt_ns - now_ns, NSEC_PER_USEC);
  961. + interval_delivered = (u64)bbr_bw(sk) * interval_us >> BW_SCALE;
  962. + inflight_at_edt = inflight_now;
  963. + if (bbr->pacing_gain > BBR_UNIT) /* increasing inflight */
  964. + inflight_at_edt += bbr_tso_segs_goal(sk); /* include EDT skb */
  965. + if (interval_delivered >= inflight_at_edt)
  966. + return 0;
  967. + return inflight_at_edt - interval_delivered;
  968. +}
  969. +
  970. +/* Find the cwnd increment based on estimate of ack aggregation */
  971. +static u32 bbr_ack_aggregation_cwnd(struct sock *sk)
  972. +{
  973. + u32 max_aggr_cwnd, aggr_cwnd = 0;
  974. +
  975. + if (bbr_param(sk, extra_acked_gain)) {
  976. + max_aggr_cwnd = ((u64)bbr_bw(sk) * bbr_extra_acked_max_us)
  977. + / BW_UNIT;
  978. + aggr_cwnd = (bbr_param(sk, extra_acked_gain) * bbr_extra_acked(sk))
  979. + >> BBR_SCALE;
  980. + aggr_cwnd = min(aggr_cwnd, max_aggr_cwnd);
  981. + }
  982. +
  983. + return aggr_cwnd;
  984. +}
  985. +
  986. +/* Returns the cwnd for PROBE_RTT mode. */
  987. +static u32 bbr_probe_rtt_cwnd(struct sock *sk)
  988. +{
  989. + return max_t(u32, bbr_param(sk, cwnd_min_target),
  990. + bbr_bdp(sk, bbr_bw(sk), bbr_param(sk, probe_rtt_cwnd_gain)));
  991. +}
  992. +
  993. +/* Slow-start up toward target cwnd (if bw estimate is growing, or packet loss
  994. + * has drawn us down below target), or snap down to target if we're above it.
  995. + */
  996. +static void bbr_set_cwnd(struct sock *sk, const struct rate_sample *rs,
  997. + u32 acked, u32 bw, int gain, u32 cwnd,
  998. + struct bbr_context *ctx)
  999. +{
  1000. + struct tcp_sock *tp = tcp_sk(sk);
  1001. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1002. + u32 target_cwnd = 0;
  1003. +
  1004. + if (!acked)
  1005. + goto done; /* no packet fully ACKed; just apply caps */
  1006. +
  1007. + target_cwnd = bbr_bdp(sk, bw, gain);
  1008. +
  1009. + /* Increment the cwnd to account for excess ACKed data that seems
  1010. + * due to aggregation (of data and/or ACKs) visible in the ACK stream.
  1011. + */
  1012. + target_cwnd += bbr_ack_aggregation_cwnd(sk);
  1013. + target_cwnd = bbr_quantization_budget(sk, target_cwnd);
  1014. +
  1015. + /* Update cwnd and enable fast path if cwnd reaches target_cwnd. */
  1016. + bbr->try_fast_path = 0;
  1017. + if (bbr_full_bw_reached(sk)) { /* only cut cwnd if we filled the pipe */
  1018. + cwnd += acked;
  1019. + if (cwnd >= target_cwnd) {
  1020. + cwnd = target_cwnd;
  1021. + bbr->try_fast_path = 1;
  1022. + }
  1023. + } else if (cwnd < target_cwnd || cwnd < 2 * bbr->init_cwnd) {
  1024. + cwnd += acked;
  1025. + } else {
  1026. + bbr->try_fast_path = 1;
  1027. + }
  1028. +
  1029. + cwnd = max_t(u32, cwnd, bbr_param(sk, cwnd_min_target));
  1030. +done:
  1031. + tcp_snd_cwnd_set(tp, min(cwnd, tp->snd_cwnd_clamp)); /* global cap */
  1032. + if (bbr->mode == BBR_PROBE_RTT) /* drain queue, refresh min_rtt */
  1033. + tcp_snd_cwnd_set(tp, min_t(u32, tcp_snd_cwnd(tp),
  1034. + bbr_probe_rtt_cwnd(sk)));
  1035. +}
  1036. +
  1037. +static void bbr_reset_startup_mode(struct sock *sk)
  1038. +{
  1039. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1040. +
  1041. + bbr->mode = BBR_STARTUP;
  1042. +}
  1043. +
  1044. +/* See if we have reached next round trip. Upon start of the new round,
  1045. + * returns packets delivered since previous round start plus this ACK.
  1046. + */
  1047. +static u32 bbr_update_round_start(struct sock *sk,
  1048. + const struct rate_sample *rs, struct bbr_context *ctx)
  1049. +{
  1050. + struct tcp_sock *tp = tcp_sk(sk);
  1051. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1052. + u32 round_delivered = 0;
  1053. +
  1054. + bbr->round_start = 0;
  1055. +
  1056. + /* See if we've reached the next RTT */
  1057. + if (rs->interval_us > 0 &&
  1058. + !before(rs->prior_delivered, bbr->next_rtt_delivered)) {
  1059. + round_delivered = tp->delivered - bbr->next_rtt_delivered;
  1060. + bbr->next_rtt_delivered = tp->delivered;
  1061. + bbr->round_start = 1;
  1062. + }
  1063. + return round_delivered;
  1064. +}
  1065. +
  1066. +/* Calculate the bandwidth based on how fast packets are delivered */
  1067. +static void bbr_calculate_bw_sample(struct sock *sk,
  1068. + const struct rate_sample *rs, struct bbr_context *ctx)
  1069. +{
  1070. + u64 bw = 0;
  1071. +
  1072. + /* Divide delivered by the interval to find a (lower bound) bottleneck
  1073. + * bandwidth sample. Delivered is in packets and interval_us in uS and
  1074. + * ratio will be <<1 for most connections. So delivered is first scaled.
  1075. + * Round up to allow growth at low rates, even with integer division.
  1076. + */
  1077. + if (rs->interval_us > 0) {
  1078. + if (WARN_ONCE(rs->delivered < 0,
  1079. + "negative delivered: %d interval_us: %ld\n",
  1080. + rs->delivered, rs->interval_us))
  1081. + return;
  1082. +
  1083. + bw = DIV_ROUND_UP_ULL((u64)rs->delivered * BW_UNIT, rs->interval_us);
  1084. + }
  1085. +
  1086. + ctx->sample_bw = bw;
  1087. +}
  1088. +
  1089. +/* Estimates the windowed max degree of ack aggregation.
  1090. + * This is used to provision extra in-flight data to keep sending during
  1091. + * inter-ACK silences.
  1092. + *
  1093. + * Degree of ack aggregation is estimated as extra data acked beyond expected.
  1094. + *
  1095. + * max_extra_acked = "maximum recent excess data ACKed beyond max_bw * interval"
  1096. + * cwnd += max_extra_acked
  1097. + *
  1098. + * Max extra_acked is clamped by cwnd and bw * bbr_extra_acked_max_us (100 ms).
  1099. + * Max filter is an approximate sliding window of 5-10 (packet timed) round
  1100. + * trips for non-startup phase, and 1-2 round trips for startup.
  1101. + */
  1102. +static void bbr_update_ack_aggregation(struct sock *sk,
  1103. + const struct rate_sample *rs)
  1104. +{
  1105. + u32 epoch_us, expected_acked, extra_acked;
  1106. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1107. + struct tcp_sock *tp = tcp_sk(sk);
  1108. + u32 extra_acked_win_rtts_thresh = bbr_param(sk, extra_acked_win_rtts);
  1109. +
  1110. + if (!bbr_param(sk, extra_acked_gain) || rs->acked_sacked <= 0 ||
  1111. + rs->delivered < 0 || rs->interval_us <= 0)
  1112. + return;
  1113. +
  1114. + if (bbr->round_start) {
  1115. + bbr->extra_acked_win_rtts = min(0x1F,
  1116. + bbr->extra_acked_win_rtts + 1);
  1117. + if (!bbr_full_bw_reached(sk))
  1118. + extra_acked_win_rtts_thresh = 1;
  1119. + if (bbr->extra_acked_win_rtts >=
  1120. + extra_acked_win_rtts_thresh) {
  1121. + bbr->extra_acked_win_rtts = 0;
  1122. + bbr->extra_acked_win_idx = bbr->extra_acked_win_idx ?
  1123. + 0 : 1;
  1124. + bbr->extra_acked[bbr->extra_acked_win_idx] = 0;
  1125. + }
  1126. + }
  1127. +
  1128. + /* Compute how many packets we expected to be delivered over epoch. */
  1129. + epoch_us = tcp_stamp_us_delta(tp->delivered_mstamp,
  1130. + bbr->ack_epoch_mstamp);
  1131. + expected_acked = ((u64)bbr_bw(sk) * epoch_us) / BW_UNIT;
  1132. +
  1133. + /* Reset the aggregation epoch if ACK rate is below expected rate or
  1134. + * significantly large no. of ack received since epoch (potentially
  1135. + * quite old epoch).
  1136. + */
  1137. + if (bbr->ack_epoch_acked <= expected_acked ||
  1138. + (bbr->ack_epoch_acked + rs->acked_sacked >=
  1139. + bbr_ack_epoch_acked_reset_thresh)) {
  1140. + bbr->ack_epoch_acked = 0;
  1141. + bbr->ack_epoch_mstamp = tp->delivered_mstamp;
  1142. + expected_acked = 0;
  1143. + }
  1144. +
  1145. + /* Compute excess data delivered, beyond what was expected. */
  1146. + bbr->ack_epoch_acked = min_t(u32, 0xFFFFF,
  1147. + bbr->ack_epoch_acked + rs->acked_sacked);
  1148. + extra_acked = bbr->ack_epoch_acked - expected_acked;
  1149. + extra_acked = min(extra_acked, tcp_snd_cwnd(tp));
  1150. + if (extra_acked > bbr->extra_acked[bbr->extra_acked_win_idx])
  1151. + bbr->extra_acked[bbr->extra_acked_win_idx] = extra_acked;
  1152. +}
  1153. +
  1154. +static void bbr_check_probe_rtt_done(struct sock *sk)
  1155. +{
  1156. + struct tcp_sock *tp = tcp_sk(sk);
  1157. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1158. +
  1159. + if (!(bbr->probe_rtt_done_stamp &&
  1160. + after(tcp_jiffies32, bbr->probe_rtt_done_stamp)))
  1161. + return;
  1162. +
  1163. + bbr->probe_rtt_min_stamp = tcp_jiffies32; /* schedule next PROBE_RTT */
  1164. + tcp_snd_cwnd_set(tp, max(tcp_snd_cwnd(tp), bbr->prior_cwnd));
  1165. + bbr_exit_probe_rtt(sk);
  1166. +}
  1167. +
  1168. +/* The goal of PROBE_RTT mode is to have BBR flows cooperatively and
  1169. + * periodically drain the bottleneck queue, to converge to measure the true
  1170. + * min_rtt (unloaded propagation delay). This allows the flows to keep queues
  1171. + * small (reducing queuing delay and packet loss) and achieve fairness among
  1172. + * BBR flows.
  1173. + *
  1174. + * The min_rtt filter window is 10 seconds. When the min_rtt estimate expires,
  1175. + * we enter PROBE_RTT mode and cap the cwnd at bbr_cwnd_min_target=4 packets.
  1176. + * After at least bbr_probe_rtt_mode_ms=200ms and at least one packet-timed
  1177. + * round trip elapsed with that flight size <= 4, we leave PROBE_RTT mode and
  1178. + * re-enter the previous mode. BBR uses 200ms to approximately bound the
  1179. + * performance penalty of PROBE_RTT's cwnd capping to roughly 2% (200ms/10s).
  1180. + *
  1181. + * Note that flows need only pay 2% if they are busy sending over the last 10
  1182. + * seconds. Interactive applications (e.g., Web, RPCs, video chunks) often have
  1183. + * natural silences or low-rate periods within 10 seconds where the rate is low
  1184. + * enough for long enough to drain its queue in the bottleneck. We pick up
  1185. + * these min RTT measurements opportunistically with our min_rtt filter. :-)
  1186. + */
  1187. +static void bbr_update_min_rtt(struct sock *sk, const struct rate_sample *rs)
  1188. +{
  1189. + struct tcp_sock *tp = tcp_sk(sk);
  1190. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1191. + bool probe_rtt_expired, min_rtt_expired;
  1192. + u32 expire;
  1193. +
  1194. + /* Track min RTT in probe_rtt_win_ms to time next PROBE_RTT state. */
  1195. + expire = bbr->probe_rtt_min_stamp +
  1196. + msecs_to_jiffies(bbr_param(sk, probe_rtt_win_ms));
  1197. + probe_rtt_expired = after(tcp_jiffies32, expire);
  1198. + if (rs->rtt_us >= 0 &&
  1199. + (rs->rtt_us < bbr->probe_rtt_min_us ||
  1200. + (probe_rtt_expired && !rs->is_ack_delayed))) {
  1201. + bbr->probe_rtt_min_us = rs->rtt_us;
  1202. + bbr->probe_rtt_min_stamp = tcp_jiffies32;
  1203. + }
  1204. + /* Track min RTT seen in the min_rtt_win_sec filter window: */
  1205. + expire = bbr->min_rtt_stamp + bbr_param(sk, min_rtt_win_sec) * HZ;
  1206. + min_rtt_expired = after(tcp_jiffies32, expire);
  1207. + if (bbr->probe_rtt_min_us <= bbr->min_rtt_us ||
  1208. + min_rtt_expired) {
  1209. + bbr->min_rtt_us = bbr->probe_rtt_min_us;
  1210. + bbr->min_rtt_stamp = bbr->probe_rtt_min_stamp;
  1211. + }
  1212. +
  1213. + if (bbr_param(sk, probe_rtt_mode_ms) > 0 && probe_rtt_expired &&
  1214. + !bbr->idle_restart && bbr->mode != BBR_PROBE_RTT) {
  1215. + bbr->mode = BBR_PROBE_RTT; /* dip, drain queue */
  1216. + bbr_save_cwnd(sk); /* note cwnd so we can restore it */
  1217. + bbr->probe_rtt_done_stamp = 0;
  1218. + bbr->ack_phase = BBR_ACKS_PROBE_STOPPING;
  1219. + bbr->next_rtt_delivered = tp->delivered;
  1220. + }
  1221. +
  1222. + if (bbr->mode == BBR_PROBE_RTT) {
  1223. + /* Ignore low rate samples during this mode. */
  1224. + tp->app_limited =
  1225. + (tp->delivered + tcp_packets_in_flight(tp)) ? : 1;
  1226. + /* Maintain min packets in flight for max(200 ms, 1 round). */
  1227. + if (!bbr->probe_rtt_done_stamp &&
  1228. + tcp_packets_in_flight(tp) <= bbr_probe_rtt_cwnd(sk)) {
  1229. + bbr->probe_rtt_done_stamp = tcp_jiffies32 +
  1230. + msecs_to_jiffies(bbr_param(sk, probe_rtt_mode_ms));
  1231. + bbr->probe_rtt_round_done = 0;
  1232. + bbr->next_rtt_delivered = tp->delivered;
  1233. + } else if (bbr->probe_rtt_done_stamp) {
  1234. + if (bbr->round_start)
  1235. + bbr->probe_rtt_round_done = 1;
  1236. + if (bbr->probe_rtt_round_done)
  1237. + bbr_check_probe_rtt_done(sk);
  1238. + }
  1239. + }
  1240. + /* Restart after idle ends only once we process a new S/ACK for data */
  1241. + if (rs->delivered > 0)
  1242. + bbr->idle_restart = 0;
  1243. +}
  1244. +
  1245. +static void bbr_update_gains(struct sock *sk)
  1246. +{
  1247. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1248. +
  1249. + switch (bbr->mode) {
  1250. + case BBR_STARTUP:
  1251. + bbr->pacing_gain = bbr_param(sk, startup_pacing_gain);
  1252. + bbr->cwnd_gain = bbr_param(sk, startup_cwnd_gain);
  1253. + break;
  1254. + case BBR_DRAIN:
  1255. + bbr->pacing_gain = bbr_param(sk, drain_gain); /* slow, to drain */
  1256. + bbr->cwnd_gain = bbr_param(sk, startup_cwnd_gain); /* keep cwnd */
  1257. + break;
  1258. + case BBR_PROBE_BW:
  1259. + bbr->pacing_gain = bbr_pacing_gain[bbr->cycle_idx];
  1260. + bbr->cwnd_gain = bbr_param(sk, cwnd_gain);
  1261. + if (bbr_param(sk, bw_probe_cwnd_gain) &&
  1262. + bbr->cycle_idx == BBR_BW_PROBE_UP)
  1263. + bbr->cwnd_gain +=
  1264. + BBR_UNIT * bbr_param(sk, bw_probe_cwnd_gain) / 4;
  1265. + break;
  1266. + case BBR_PROBE_RTT:
  1267. + bbr->pacing_gain = BBR_UNIT;
  1268. + bbr->cwnd_gain = BBR_UNIT;
  1269. + break;
  1270. + default:
  1271. + WARN_ONCE(1, "BBR bad mode: %u\n", bbr->mode);
  1272. + break;
  1273. + }
  1274. +}
  1275. +
  1276. +static u32 bbr3_sndbuf_expand(struct sock *sk)
  1277. +{
  1278. + /* Provision 3 * cwnd since BBR may slow-start even during recovery. */
  1279. + return 3;
  1280. +}
  1281. +
  1282. +/* Incorporate a new bw sample into the current window of our max filter. */
  1283. +static void bbr_take_max_bw_sample(struct sock *sk, u32 bw)
  1284. +{
  1285. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1286. +
  1287. + bbr->bw_hi[1] = max(bw, bbr->bw_hi[1]);
  1288. +}
  1289. +
  1290. +/* Keep max of last 1-2 cycles. Each PROBE_BW cycle, flip filter window. */
  1291. +static void bbr_advance_max_bw_filter(struct sock *sk)
  1292. +{
  1293. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1294. +
  1295. + if (!bbr->bw_hi[1])
  1296. + return; /* no samples in this window; remember old window */
  1297. + bbr->bw_hi[0] = bbr->bw_hi[1];
  1298. + bbr->bw_hi[1] = 0;
  1299. +}
  1300. +
  1301. +/* Reset the estimator for reaching full bandwidth based on bw plateau. */
  1302. +static void bbr_reset_full_bw(struct sock *sk)
  1303. +{
  1304. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1305. +
  1306. + bbr->full_bw = 0;
  1307. + bbr->full_bw_cnt = 0;
  1308. + bbr->full_bw_now = 0;
  1309. +}
  1310. +
  1311. +/* How much do we want in flight? Our BDP, unless congestion cut cwnd. */
  1312. +static u32 bbr_target_inflight(struct sock *sk)
  1313. +{
  1314. + u32 bdp = bbr_inflight(sk, bbr_bw(sk), BBR_UNIT);
  1315. +
  1316. + return min(bdp, tcp_sk(sk)->snd_cwnd);
  1317. +}
  1318. +
  1319. +static bool bbr_is_probing_bandwidth(struct sock *sk)
  1320. +{
  1321. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1322. +
  1323. + return (bbr->mode == BBR_STARTUP) ||
  1324. + (bbr->mode == BBR_PROBE_BW &&
  1325. + (bbr->cycle_idx == BBR_BW_PROBE_REFILL ||
  1326. + bbr->cycle_idx == BBR_BW_PROBE_UP));
  1327. +}
  1328. +
  1329. +/* Has the given amount of time elapsed since we marked the phase start? */
  1330. +static bool bbr_has_elapsed_in_phase(const struct sock *sk, u32 interval_us)
  1331. +{
  1332. + const struct tcp_sock *tp = tcp_sk(sk);
  1333. + const struct bbr3 *bbr = bbr3_get_priv(sk);
  1334. +
  1335. + return tcp_stamp_us_delta(tp->tcp_mstamp,
  1336. + bbr->cycle_mstamp + interval_us) > 0;
  1337. +}
  1338. +
  1339. +static void bbr_handle_queue_too_high_in_startup(struct sock *sk)
  1340. +{
  1341. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1342. + u32 bdp; /* estimated BDP in packets, with quantization budget */
  1343. +
  1344. + bbr->full_bw_reached = 1;
  1345. +
  1346. + bdp = bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT);
  1347. + bbr->inflight_hi = max(bdp, bbr->inflight_latest);
  1348. +}
  1349. +
  1350. +/* Exit STARTUP upon N consecutive rounds with ECN mark rate > ecn_thresh. */
  1351. +static void bbr_check_ecn_too_high_in_startup(struct sock *sk, u32 ce_ratio)
  1352. +{
  1353. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1354. +
  1355. + if (bbr_full_bw_reached(sk) || !bbr->ecn_eligible ||
  1356. + !bbr_param(sk, full_ecn_cnt) || !bbr_param(sk, ecn_thresh))
  1357. + return;
  1358. +
  1359. + if (ce_ratio >= bbr_param(sk, ecn_thresh))
  1360. + bbr->startup_ecn_rounds++;
  1361. + else
  1362. + bbr->startup_ecn_rounds = 0;
  1363. +
  1364. + if (bbr->startup_ecn_rounds >= bbr_param(sk, full_ecn_cnt)) {
  1365. + bbr_handle_queue_too_high_in_startup(sk);
  1366. + return;
  1367. + }
  1368. +}
  1369. +
  1370. +/* Updates ecn_alpha and returns ce_ratio. -1 if not available. */
  1371. +static int bbr_update_ecn_alpha(struct sock *sk)
  1372. +{
  1373. + struct tcp_sock *tp = tcp_sk(sk);
  1374. + struct net *net = sock_net(sk);
  1375. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1376. + s32 delivered, delivered_ce;
  1377. + u64 alpha, ce_ratio;
  1378. + u32 gain;
  1379. + bool want_ecn_alpha;
  1380. +
  1381. + /* See if we should use ECN sender logic for this connection. */
  1382. + if (!bbr->ecn_eligible && bbr_can_use_ecn(sk) &&
  1383. + !!bbr_param(sk, ecn_factor) &&
  1384. + (bbr->min_rtt_us <= bbr_ecn_max_rtt_us ||
  1385. + !bbr_ecn_max_rtt_us))
  1386. + bbr->ecn_eligible = 1;
  1387. +
  1388. + /* Skip updating alpha only if not ECN-eligible and PLB is disabled. */
  1389. + want_ecn_alpha = (bbr->ecn_eligible ||
  1390. + (bbr_can_use_ecn(sk) &&
  1391. + READ_ONCE(net->ipv4.sysctl_tcp_plb_enabled)));
  1392. + if (!want_ecn_alpha)
  1393. + return -1;
  1394. +
  1395. + delivered = tp->delivered - bbr->alpha_last_delivered;
  1396. + delivered_ce = tp->delivered_ce - bbr->alpha_last_delivered_ce;
  1397. +
  1398. + if (delivered == 0 || /* avoid divide by zero */
  1399. + WARN_ON_ONCE(delivered < 0 || delivered_ce < 0)) /* backwards? */
  1400. + return -1;
  1401. +
  1402. + BUILD_BUG_ON(BBR_SCALE != TCP_PLB_SCALE);
  1403. + ce_ratio = (u64)delivered_ce << BBR_SCALE;
  1404. + do_div(ce_ratio, delivered);
  1405. +
  1406. + gain = bbr_param(sk, ecn_alpha_gain);
  1407. + alpha = ((BBR_UNIT - gain) * bbr->ecn_alpha) >> BBR_SCALE;
  1408. + alpha += (gain * ce_ratio) >> BBR_SCALE;
  1409. + bbr->ecn_alpha = min_t(u32, alpha, BBR_UNIT);
  1410. +
  1411. + bbr->alpha_last_delivered = tp->delivered;
  1412. + bbr->alpha_last_delivered_ce = tp->delivered_ce;
  1413. +
  1414. + bbr_check_ecn_too_high_in_startup(sk, ce_ratio);
  1415. + return (int)ce_ratio;
  1416. +}
  1417. +
  1418. +/* Protective Load Balancing (PLB). PLB rehashes outgoing data (to a new IPv6
  1419. + * flow label) if it encounters sustained congestion in the form of ECN marks.
  1420. + */
  1421. +static void bbr_plb(struct sock *sk, const struct rate_sample *rs, int ce_ratio)
  1422. +{
  1423. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1424. +
  1425. + if (bbr->round_start && ce_ratio >= 0)
  1426. + tcp_plb_update_state(sk, &bbr->plb, ce_ratio);
  1427. +
  1428. + tcp_plb_check_rehash(sk, &bbr->plb);
  1429. +}
  1430. +
  1431. +/* Each round trip of BBR_BW_PROBE_UP, double volume of probing data. */
  1432. +static void bbr_raise_inflight_hi_slope(struct sock *sk)
  1433. +{
  1434. + struct tcp_sock *tp = tcp_sk(sk);
  1435. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1436. + u32 growth_this_round, cnt;
  1437. +
  1438. + /* Calculate "slope": packets S/Acked per inflight_hi increment. */
  1439. + growth_this_round = 1 << bbr->bw_probe_up_rounds;
  1440. + bbr->bw_probe_up_rounds = min(bbr->bw_probe_up_rounds + 1, 30);
  1441. + cnt = tcp_snd_cwnd(tp) / growth_this_round;
  1442. + cnt = max(cnt, 1U);
  1443. + bbr->bw_probe_up_cnt = cnt;
  1444. +}
  1445. +
  1446. +/* In BBR_BW_PROBE_UP, not seeing high loss/ECN/queue, so raise inflight_hi. */
  1447. +static void bbr_probe_inflight_hi_upward(struct sock *sk,
  1448. + const struct rate_sample *rs)
  1449. +{
  1450. + struct tcp_sock *tp = tcp_sk(sk);
  1451. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1452. + u32 delta;
  1453. +
  1454. + if (!tp->is_cwnd_limited || tcp_snd_cwnd(tp) < bbr->inflight_hi)
  1455. + return; /* not fully using inflight_hi, so don't grow it */
  1456. +
  1457. + /* For each bw_probe_up_cnt packets ACKed, increase inflight_hi by 1. */
  1458. + bbr->bw_probe_up_acks += rs->acked_sacked;
  1459. + if (bbr->bw_probe_up_acks >= bbr->bw_probe_up_cnt) {
  1460. + delta = bbr->bw_probe_up_acks / bbr->bw_probe_up_cnt;
  1461. + bbr->bw_probe_up_acks -= delta * bbr->bw_probe_up_cnt;
  1462. + bbr->inflight_hi += delta;
  1463. + bbr->try_fast_path = 0; /* Need to update cwnd */
  1464. + }
  1465. +
  1466. + if (bbr->round_start)
  1467. + bbr_raise_inflight_hi_slope(sk);
  1468. +}
  1469. +
  1470. +/* Does loss/ECN rate for this sample say inflight is "too high"?
  1471. + * This is used by both the bbr_check_loss_too_high_in_startup() function,
  1472. + * and in PROBE_UP.
  1473. + */
  1474. +static bool bbr_is_inflight_too_high(const struct sock *sk,
  1475. + const struct rate_sample *rs)
  1476. +{
  1477. + const struct bbr3 *bbr = bbr3_get_priv(sk);
  1478. + u32 loss_thresh, ecn_thresh;
  1479. +
  1480. + if (rs->lost > 0 && rs->tx_in_flight) {
  1481. + loss_thresh = (u64)rs->tx_in_flight * bbr_param(sk, loss_thresh) >>
  1482. + BBR_SCALE;
  1483. + if (rs->lost > loss_thresh) {
  1484. + return true;
  1485. + }
  1486. + }
  1487. +
  1488. + if (rs->delivered_ce > 0 && rs->delivered > 0 &&
  1489. + bbr->ecn_eligible && !!bbr_param(sk, ecn_thresh)) {
  1490. + ecn_thresh = (u64)rs->delivered * bbr_param(sk, ecn_thresh) >>
  1491. + BBR_SCALE;
  1492. + if (rs->delivered_ce > ecn_thresh) {
  1493. + return true;
  1494. + }
  1495. + }
  1496. +
  1497. + return false;
  1498. +}
  1499. +
  1500. +/* Calculate the tx_in_flight level that corresponded to excessive loss.
  1501. + * We find "lost_prefix" segs of the skb where loss rate went too high,
  1502. + * by solving for "lost_prefix" in the following equation:
  1503. + * lost / inflight >= loss_thresh
  1504. + * (lost_prev + lost_prefix) / (inflight_prev + lost_prefix) >= loss_thresh
  1505. + * Then we take that equation, convert it to fixed point, and
  1506. + * round up to the nearest packet.
  1507. + */
  1508. +static u32 bbr_inflight_hi_from_lost_skb(const struct sock *sk,
  1509. + const struct rate_sample *rs,
  1510. + const struct sk_buff *skb)
  1511. +{
  1512. + const struct tcp_sock *tp = tcp_sk(sk);
  1513. + u32 loss_thresh = bbr_param(sk, loss_thresh);
  1514. + u32 pcount, divisor, inflight_hi;
  1515. + s32 inflight_prev, lost_prev;
  1516. + u64 loss_budget, lost_prefix;
  1517. +
  1518. + pcount = tcp_skb_pcount(skb);
  1519. +
  1520. + /* How much data was in flight before this skb? */
  1521. + inflight_prev = rs->tx_in_flight - pcount;
  1522. + if (inflight_prev < 0) {
  1523. + WARN_ONCE(tcp_skb_tx_in_flight_is_suspicious(
  1524. + pcount,
  1525. + TCP_SKB_CB(skb)->sacked,
  1526. + rs->tx_in_flight),
  1527. + "tx_in_flight: %u pcount: %u reneg: %u",
  1528. + rs->tx_in_flight, pcount, tcp_sk(sk)->is_sack_reneg);
  1529. + return ~0U;
  1530. + }
  1531. +
  1532. + /* How much inflight data was marked lost before this skb? */
  1533. + lost_prev = rs->lost - pcount;
  1534. + if (WARN_ONCE(lost_prev < 0,
  1535. + "cwnd: %u ca: %d out: %u lost: %u pif: %u "
  1536. + "tx_in_flight: %u tx.lost: %u tp->lost: %u rs->lost: %d "
  1537. + "lost_prev: %d pcount: %d seq: %u end_seq: %u reneg: %u",
  1538. + tcp_snd_cwnd(tp), inet_csk(sk)->icsk_ca_state,
  1539. + tp->packets_out, tp->lost_out, tcp_packets_in_flight(tp),
  1540. + rs->tx_in_flight, TCP_SKB_CB(skb)->tx.lost, tp->lost,
  1541. + rs->lost, lost_prev, pcount,
  1542. + TCP_SKB_CB(skb)->seq, TCP_SKB_CB(skb)->end_seq,
  1543. + tp->is_sack_reneg))
  1544. + return ~0U;
  1545. +
  1546. + /* At what prefix of this lost skb did losss rate exceed loss_thresh? */
  1547. + loss_budget = (u64)inflight_prev * loss_thresh + BBR_UNIT - 1;
  1548. + loss_budget >>= BBR_SCALE;
  1549. + if (lost_prev >= loss_budget) {
  1550. + lost_prefix = 0; /* previous losses crossed loss_thresh */
  1551. + } else {
  1552. + lost_prefix = loss_budget - lost_prev;
  1553. + lost_prefix <<= BBR_SCALE;
  1554. + divisor = BBR_UNIT - loss_thresh;
  1555. + if (WARN_ON_ONCE(!divisor)) /* loss_thresh is 8 bits */
  1556. + return ~0U;
  1557. + do_div(lost_prefix, divisor);
  1558. + }
  1559. +
  1560. + inflight_hi = inflight_prev + lost_prefix;
  1561. + return inflight_hi;
  1562. +}
  1563. +
  1564. +/* If loss/ECN rates during probing indicated we may have overfilled a
  1565. + * buffer, return an operating point that tries to leave unutilized headroom in
  1566. + * the path for other flows, for fairness convergence and lower RTTs and loss.
  1567. + */
  1568. +static u32 bbr_inflight_with_headroom(const struct sock *sk)
  1569. +{
  1570. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1571. + u32 headroom, headroom_fraction;
  1572. +
  1573. + if (bbr->inflight_hi == ~0U)
  1574. + return ~0U;
  1575. +
  1576. + headroom_fraction = bbr_param(sk, inflight_headroom);
  1577. + headroom = ((u64)bbr->inflight_hi * headroom_fraction) >> BBR_SCALE;
  1578. + headroom = max(headroom, 1U);
  1579. + return max_t(s32, bbr->inflight_hi - headroom,
  1580. + bbr_param(sk, cwnd_min_target));
  1581. +}
  1582. +
  1583. +/* Bound cwnd to a sensible level, based on our current probing state
  1584. + * machine phase and model of a good inflight level (inflight_lo, inflight_hi).
  1585. + */
  1586. +static void bbr_bound_cwnd_for_inflight_model(struct sock *sk)
  1587. +{
  1588. + struct tcp_sock *tp = tcp_sk(sk);
  1589. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1590. + u32 cap;
  1591. +
  1592. + /* tcp_rcv_synsent_state_process() currently calls tcp_ack()
  1593. + * and thus cong_control() without first initializing us(!).
  1594. + */
  1595. + if (!bbr->initialized)
  1596. + return;
  1597. +
  1598. + cap = ~0U;
  1599. + if (bbr->mode == BBR_PROBE_BW &&
  1600. + bbr->cycle_idx != BBR_BW_PROBE_CRUISE) {
  1601. + /* Probe to see if more packets fit in the path. */
  1602. + cap = bbr->inflight_hi;
  1603. + } else {
  1604. + if (bbr->mode == BBR_PROBE_RTT ||
  1605. + (bbr->mode == BBR_PROBE_BW &&
  1606. + bbr->cycle_idx == BBR_BW_PROBE_CRUISE))
  1607. + cap = bbr_inflight_with_headroom(sk);
  1608. + }
  1609. + /* Adapt to any loss/ECN since our last bw probe. */
  1610. + cap = min(cap, bbr->inflight_lo);
  1611. +
  1612. + cap = max_t(u32, cap, bbr_param(sk, cwnd_min_target));
  1613. + tcp_snd_cwnd_set(tp, min(cap, tcp_snd_cwnd(tp)));
  1614. +}
  1615. +
  1616. +/* How should we multiplicatively cut bw or inflight limits based on ECN? */
  1617. +static u32 bbr_ecn_cut(struct sock *sk)
  1618. +{
  1619. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1620. +
  1621. + return BBR_UNIT -
  1622. + ((bbr->ecn_alpha * bbr_param(sk, ecn_factor)) >> BBR_SCALE);
  1623. +}
  1624. +
  1625. +/* Init lower bounds if have not inited yet. */
  1626. +static void bbr_init_lower_bounds(struct sock *sk, bool init_bw)
  1627. +{
  1628. + struct tcp_sock *tp = tcp_sk(sk);
  1629. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1630. +
  1631. + if (init_bw && bbr->bw_lo == ~0U)
  1632. + bbr->bw_lo = bbr_max_bw(sk);
  1633. + if (bbr->inflight_lo == ~0U)
  1634. + bbr->inflight_lo = tcp_snd_cwnd(tp);
  1635. +}
  1636. +
  1637. +/* Reduce bw and inflight to (1 - beta). */
  1638. +static void bbr_loss_lower_bounds(struct sock *sk, u32 *bw, u32 *inflight)
  1639. +{
  1640. + struct bbr3* bbr = bbr3_get_priv(sk);
  1641. + u32 loss_cut = BBR_UNIT - bbr_param(sk, beta);
  1642. +
  1643. + *bw = max_t(u32, bbr->bw_latest,
  1644. + (u64)bbr->bw_lo * loss_cut >> BBR_SCALE);
  1645. + *inflight = max_t(u32, bbr->inflight_latest,
  1646. + (u64)bbr->inflight_lo * loss_cut >> BBR_SCALE);
  1647. +}
  1648. +
  1649. +/* Reduce inflight to (1 - alpha*ecn_factor). */
  1650. +static void bbr_ecn_lower_bounds(struct sock *sk, u32 *inflight)
  1651. +{
  1652. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1653. + u32 ecn_cut = bbr_ecn_cut(sk);
  1654. +
  1655. + *inflight = (u64)bbr->inflight_lo * ecn_cut >> BBR_SCALE;
  1656. +}
  1657. +
  1658. +/* Estimate a short-term lower bound on the capacity available now, based
  1659. + * on measurements of the current delivery process and recent history. When we
  1660. + * are seeing loss/ECN at times when we are not probing bw, then conservatively
  1661. + * move toward flow balance by multiplicatively cutting our short-term
  1662. + * estimated safe rate and volume of data (bw_lo and inflight_lo). We use a
  1663. + * multiplicative decrease in order to converge to a lower capacity in time
  1664. + * logarithmic in the magnitude of the decrease.
  1665. + *
  1666. + * However, we do not cut our short-term estimates lower than the current rate
  1667. + * and volume of delivered data from this round trip, since from the current
  1668. + * delivery process we can estimate the measured capacity available now.
  1669. + *
  1670. + * Anything faster than that approach would knowingly risk high loss, which can
  1671. + * cause low bw for Reno/CUBIC and high loss recovery latency for
  1672. + * request/response flows using any congestion control.
  1673. + */
  1674. +static void bbr_adapt_lower_bounds(struct sock *sk,
  1675. + const struct rate_sample *rs)
  1676. +{
  1677. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1678. + u32 ecn_inflight_lo = ~0U;
  1679. +
  1680. + /* We only use lower-bound estimates when not probing bw.
  1681. + * When probing we need to push inflight higher to probe bw.
  1682. + */
  1683. + if (bbr_is_probing_bandwidth(sk))
  1684. + return;
  1685. +
  1686. + /* ECN response. */
  1687. + if (bbr->ecn_in_round && !!bbr_param(sk, ecn_factor)) {
  1688. + bbr_init_lower_bounds(sk, false);
  1689. + bbr_ecn_lower_bounds(sk, &ecn_inflight_lo);
  1690. + }
  1691. +
  1692. + /* Loss response. */
  1693. + if (bbr->loss_in_round) {
  1694. + bbr_init_lower_bounds(sk, true);
  1695. + bbr_loss_lower_bounds(sk, &bbr->bw_lo, &bbr->inflight_lo);
  1696. + }
  1697. +
  1698. + /* Adjust to the lower of the levels implied by loss/ECN. */
  1699. + bbr->inflight_lo = min(bbr->inflight_lo, ecn_inflight_lo);
  1700. + bbr->bw_lo = max(1U, bbr->bw_lo);
  1701. +}
  1702. +
  1703. +/* Reset any short-term lower-bound adaptation to congestion, so that we can
  1704. + * push our inflight up.
  1705. + */
  1706. +static void bbr_reset_lower_bounds(struct sock *sk)
  1707. +{
  1708. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1709. +
  1710. + bbr->bw_lo = ~0U;
  1711. + bbr->inflight_lo = ~0U;
  1712. +}
  1713. +
  1714. +/* After bw probing (STARTUP/PROBE_UP), reset signals before entering a state
  1715. + * machine phase where we adapt our lower bound based on congestion signals.
  1716. + */
  1717. +static void bbr_reset_congestion_signals(struct sock *sk)
  1718. +{
  1719. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1720. +
  1721. + bbr->loss_in_round = 0;
  1722. + bbr->ecn_in_round = 0;
  1723. + bbr->loss_in_cycle = 0;
  1724. + bbr->ecn_in_cycle = 0;
  1725. + bbr->bw_latest = 0;
  1726. + bbr->inflight_latest = 0;
  1727. +}
  1728. +
  1729. +static void bbr_exit_loss_recovery(struct sock *sk)
  1730. +{
  1731. + struct tcp_sock *tp = tcp_sk(sk);
  1732. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1733. +
  1734. + tcp_snd_cwnd_set(tp, max(tcp_snd_cwnd(tp), bbr->prior_cwnd));
  1735. + bbr->try_fast_path = 0; /* bound cwnd using latest model */
  1736. +}
  1737. +
  1738. +/* Update rate and volume of delivered data from latest round trip. */
  1739. +static void bbr_update_latest_delivery_signals(
  1740. + struct sock *sk, const struct rate_sample *rs, struct bbr_context *ctx)
  1741. +{
  1742. + struct tcp_sock *tp = tcp_sk(sk);
  1743. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1744. +
  1745. + bbr->loss_round_start = 0;
  1746. + if (rs->interval_us <= 0 || !rs->acked_sacked)
  1747. + return; /* Not a valid observation */
  1748. +
  1749. + bbr->bw_latest = max_t(u32, bbr->bw_latest, ctx->sample_bw);
  1750. + bbr->inflight_latest = max_t(u32, bbr->inflight_latest, rs->delivered);
  1751. +
  1752. + if (!before(rs->prior_delivered, bbr->loss_round_delivered)) {
  1753. + bbr->loss_round_delivered = tp->delivered;
  1754. + bbr->loss_round_start = 1; /* mark start of new round trip */
  1755. + }
  1756. +}
  1757. +
  1758. +/* Once per round, reset filter for latest rate and volume of delivered data. */
  1759. +static void bbr_advance_latest_delivery_signals(
  1760. + struct sock *sk, const struct rate_sample *rs, struct bbr_context *ctx)
  1761. +{
  1762. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1763. +
  1764. + /* If ACK matches a TLP retransmit, persist the filter. If we detect
  1765. + * that a TLP retransmit plugged a tail loss, we'll want to remember
  1766. + * how much data the path delivered before the tail loss.
  1767. + */
  1768. + if (bbr->loss_round_start && !rs->is_acking_tlp_retrans_seq) {
  1769. + bbr->bw_latest = ctx->sample_bw;
  1770. + bbr->inflight_latest = rs->delivered;
  1771. + }
  1772. +}
  1773. +
  1774. +/* Update (most of) our congestion signals: track the recent rate and volume of
  1775. + * delivered data, presence of loss, and EWMA degree of ECN marking.
  1776. + */
  1777. +static void bbr_update_congestion_signals(
  1778. + struct sock *sk, const struct rate_sample *rs, struct bbr_context *ctx)
  1779. +{
  1780. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1781. + u64 bw;
  1782. +
  1783. + if (rs->interval_us <= 0 || !rs->acked_sacked)
  1784. + return; /* Not a valid observation */
  1785. + bw = ctx->sample_bw;
  1786. +
  1787. + if (!rs->is_app_limited || bw >= bbr_max_bw(sk))
  1788. + bbr_take_max_bw_sample(sk, bw);
  1789. +
  1790. + bbr->loss_in_round |= (rs->losses > 0);
  1791. +
  1792. + if (!bbr->loss_round_start)
  1793. + return; /* skip the per-round-trip updates */
  1794. + /* Now do per-round-trip updates. */
  1795. + bbr_adapt_lower_bounds(sk, rs);
  1796. +
  1797. + bbr->loss_in_round = 0;
  1798. + bbr->ecn_in_round = 0;
  1799. +}
  1800. +
  1801. +/* Bandwidth probing can cause loss. To help coexistence with loss-based
  1802. + * congestion control we spread out our probing in a Reno-conscious way. Due to
  1803. + * the shape of the Reno sawtooth, the time required between loss epochs for an
  1804. + * idealized Reno flow is a number of round trips that is the BDP of that
  1805. + * flow. We count packet-timed round trips directly, since measured RTT can
  1806. + * vary widely, and Reno is driven by packet-timed round trips.
  1807. + */
  1808. +static bool bbr_is_reno_coexistence_probe_time(struct sock *sk)
  1809. +{
  1810. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1811. + u32 rounds;
  1812. +
  1813. + /* Random loss can shave some small percentage off of our inflight
  1814. + * in each round. To survive this, flows need robust periodic probes.
  1815. + */
  1816. + rounds = min_t(u32, bbr_param(sk, bw_probe_max_rounds), bbr_target_inflight(sk));
  1817. + return bbr->rounds_since_probe >= rounds;
  1818. +}
  1819. +
  1820. +/* How long do we want to wait before probing for bandwidth (and risking
  1821. + * loss)? We randomize the wait, for better mixing and fairness convergence.
  1822. + *
  1823. + * We bound the Reno-coexistence inter-bw-probe time to be 62-63 round trips.
  1824. + * This is calculated to allow fairness with a 25Mbps, 30ms Reno flow,
  1825. + * (eg 4K video to a broadband user):
  1826. + * BDP = 25Mbps * .030sec /(1514bytes) = 61.9 packets
  1827. + *
  1828. + * We bound the BBR-native inter-bw-probe wall clock time to be:
  1829. + * (a) higher than 2 sec: to try to avoid causing loss for a long enough time
  1830. + * to allow Reno at 30ms to get 4K video bw, the inter-bw-probe time must
  1831. + * be at least: 25Mbps * .030sec / (1514bytes) * 0.030sec = 1.9secs
  1832. + * (b) lower than 3 sec: to ensure flows can start probing in a reasonable
  1833. + * amount of time to discover unutilized bw on human-scale interactive
  1834. + * time-scales (e.g. perhaps traffic from a web page download that we
  1835. + * were competing with is now complete).
  1836. + */
  1837. +static void bbr_pick_probe_wait(struct sock *sk)
  1838. +{
  1839. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1840. +
  1841. + /* Decide the random round-trip bound for wait until probe: */
  1842. + bbr->rounds_since_probe =
  1843. + get_random_u32_below(bbr_param(sk, bw_probe_rand_rounds));
  1844. + /* Decide the random wall clock bound for wait until probe: */
  1845. + bbr->probe_wait_us = bbr_param(sk, bw_probe_base_us) +
  1846. + get_random_u32_below(bbr_param(sk, bw_probe_rand_us));
  1847. +}
  1848. +
  1849. +static void bbr_set_cycle_idx(struct sock *sk, int cycle_idx)
  1850. +{
  1851. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1852. +
  1853. + bbr->cycle_idx = cycle_idx;
  1854. + /* New phase, so need to update cwnd and pacing rate. */
  1855. + bbr->try_fast_path = 0;
  1856. +}
  1857. +
  1858. +/* Send at estimated bw to fill the pipe, but not queue. We need this phase
  1859. + * before PROBE_UP, because as soon as we send faster than the available bw
  1860. + * we will start building a queue, and if the buffer is shallow we can cause
  1861. + * loss. If we do not fill the pipe before we cause this loss, our bw_hi and
  1862. + * inflight_hi estimates will underestimate.
  1863. + */
  1864. +static void bbr_start_bw_probe_refill(struct sock *sk, u32 bw_probe_up_rounds)
  1865. +{
  1866. + struct tcp_sock *tp = tcp_sk(sk);
  1867. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1868. +
  1869. + bbr_reset_lower_bounds(sk);
  1870. + bbr->bw_probe_up_rounds = bw_probe_up_rounds;
  1871. + bbr->bw_probe_up_acks = 0;
  1872. + bbr->stopped_risky_probe = 0;
  1873. + bbr->ack_phase = BBR_ACKS_REFILLING;
  1874. + bbr->next_rtt_delivered = tp->delivered;
  1875. + bbr_set_cycle_idx(sk, BBR_BW_PROBE_REFILL);
  1876. +}
  1877. +
  1878. +/* Now probe max deliverable data rate and volume. */
  1879. +static void bbr_start_bw_probe_up(struct sock *sk, struct bbr_context *ctx)
  1880. +{
  1881. + struct tcp_sock *tp = tcp_sk(sk);
  1882. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1883. +
  1884. + bbr->ack_phase = BBR_ACKS_PROBE_STARTING;
  1885. + bbr->next_rtt_delivered = tp->delivered;
  1886. + bbr->cycle_mstamp = tp->tcp_mstamp;
  1887. + bbr_reset_full_bw(sk);
  1888. + bbr->full_bw = ctx->sample_bw;
  1889. + bbr_set_cycle_idx(sk, BBR_BW_PROBE_UP);
  1890. + bbr_raise_inflight_hi_slope(sk);
  1891. +}
  1892. +
  1893. +/* Start a new PROBE_BW probing cycle of some wall clock length. Pick a wall
  1894. + * clock time at which to probe beyond an inflight that we think to be
  1895. + * safe. This will knowingly risk packet loss, so we want to do this rarely, to
  1896. + * keep packet loss rates low. Also start a round-trip counter, to probe faster
  1897. + * if we estimate a Reno flow at our BDP would probe faster.
  1898. + */
  1899. +static void bbr_start_bw_probe_down(struct sock *sk)
  1900. +{
  1901. + struct tcp_sock *tp = tcp_sk(sk);
  1902. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1903. +
  1904. + bbr_reset_congestion_signals(sk);
  1905. + bbr->bw_probe_up_cnt = ~0U; /* not growing inflight_hi any more */
  1906. + bbr_pick_probe_wait(sk);
  1907. + bbr->cycle_mstamp = tp->tcp_mstamp; /* start wall clock */
  1908. + bbr->ack_phase = BBR_ACKS_PROBE_STOPPING;
  1909. + bbr->next_rtt_delivered = tp->delivered;
  1910. + bbr_set_cycle_idx(sk, BBR_BW_PROBE_DOWN);
  1911. +}
  1912. +
  1913. +/* Cruise: maintain what we estimate to be a neutral, conservative
  1914. + * operating point, without attempting to probe up for bandwidth or down for
  1915. + * RTT, and only reducing inflight in response to loss/ECN signals.
  1916. + */
  1917. +static void bbr_start_bw_probe_cruise(struct sock *sk)
  1918. +{
  1919. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1920. +
  1921. + if (bbr->inflight_lo != ~0U)
  1922. + bbr->inflight_lo = min(bbr->inflight_lo, bbr->inflight_hi);
  1923. +
  1924. + bbr_set_cycle_idx(sk, BBR_BW_PROBE_CRUISE);
  1925. +}
  1926. +
  1927. +/* Loss and/or ECN rate is too high while probing.
  1928. + * Adapt (once per bw probe) by cutting inflight_hi and then restarting cycle.
  1929. + */
  1930. +static void bbr_handle_inflight_too_high(struct sock *sk,
  1931. + const struct rate_sample *rs)
  1932. +{
  1933. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1934. + const u32 beta = bbr_param(sk, beta);
  1935. +
  1936. + bbr->prev_probe_too_high = 1;
  1937. + bbr->bw_probe_samples = 0; /* only react once per probe */
  1938. + /* If we are app-limited then we are not robustly
  1939. + * probing the max volume of inflight data we think
  1940. + * might be safe (analogous to how app-limited bw
  1941. + * samples are not known to be robustly probing bw).
  1942. + */
  1943. + if (!rs->is_app_limited) {
  1944. + bbr->inflight_hi = max_t(u32, rs->tx_in_flight,
  1945. + (u64)bbr_target_inflight(sk) *
  1946. + (BBR_UNIT - beta) >> BBR_SCALE);
  1947. + }
  1948. + if (bbr->mode == BBR_PROBE_BW && bbr->cycle_idx == BBR_BW_PROBE_UP)
  1949. + bbr_start_bw_probe_down(sk);
  1950. +}
  1951. +
  1952. +/* If we're seeing bw and loss samples reflecting our bw probing, adapt
  1953. + * using the signals we see. If loss or ECN mark rate gets too high, then adapt
  1954. + * inflight_hi downward. If we're able to push inflight higher without such
  1955. + * signals, push higher: adapt inflight_hi upward.
  1956. + */
  1957. +static bool bbr_adapt_upper_bounds(struct sock *sk,
  1958. + const struct rate_sample *rs,
  1959. + struct bbr_context *ctx)
  1960. +{
  1961. + struct bbr3 *bbr = bbr3_get_priv(sk);
  1962. +
  1963. + /* Track when we'll see bw/loss samples resulting from our bw probes. */
  1964. + if (bbr->ack_phase == BBR_ACKS_PROBE_STARTING && bbr->round_start)
  1965. + bbr->ack_phase = BBR_ACKS_PROBE_FEEDBACK;
  1966. + if (bbr->ack_phase == BBR_ACKS_PROBE_STOPPING && bbr->round_start) {
  1967. + /* End of samples from bw probing phase. */
  1968. + bbr->bw_probe_samples = 0;
  1969. + bbr->ack_phase = BBR_ACKS_INIT;
  1970. + /* At this point in the cycle, our current bw sample is also
  1971. + * our best recent chance at finding the highest available bw
  1972. + * for this flow. So now is the best time to forget the bw
  1973. + * samples from the previous cycle, by advancing the window.
  1974. + */
  1975. + if (bbr->mode == BBR_PROBE_BW && !rs->is_app_limited)
  1976. + bbr_advance_max_bw_filter(sk);
  1977. + /* If we had an inflight_hi, then probed and pushed inflight all
  1978. + * the way up to hit that inflight_hi without seeing any
  1979. + * high loss/ECN in all the resulting ACKs from that probing,
  1980. + * then probe up again, this time letting inflight persist at
  1981. + * inflight_hi for a round trip, then accelerating beyond.
  1982. + */
  1983. + if (bbr->mode == BBR_PROBE_BW &&
  1984. + bbr->stopped_risky_probe && !bbr->prev_probe_too_high) {
  1985. + bbr_start_bw_probe_refill(sk, 0);
  1986. + return true; /* yes, decided state transition */
  1987. + }
  1988. + }
  1989. + if (bbr_is_inflight_too_high(sk, rs)) {
  1990. + if (bbr->bw_probe_samples) /* sample is from bw probing? */
  1991. + bbr_handle_inflight_too_high(sk, rs);
  1992. + } else {
  1993. + /* Loss/ECN rate is declared safe. Adjust upper bound upward. */
  1994. +
  1995. + if (bbr->inflight_hi == ~0U)
  1996. + return false; /* no excess queue signals yet */
  1997. +
  1998. + /* To be resilient to random loss, we must raise bw/inflight_hi
  1999. + * if we observe in any phase that a higher level is safe.
  2000. + */
  2001. + if (rs->tx_in_flight > bbr->inflight_hi) {
  2002. + bbr->inflight_hi = rs->tx_in_flight;
  2003. + }
  2004. +
  2005. + if (bbr->mode == BBR_PROBE_BW &&
  2006. + bbr->cycle_idx == BBR_BW_PROBE_UP)
  2007. + bbr_probe_inflight_hi_upward(sk, rs);
  2008. + }
  2009. +
  2010. + return false;
  2011. +}
  2012. +
  2013. +/* Check if it's time to probe for bandwidth now, and if so, kick it off. */
  2014. +static bool bbr_check_time_to_probe_bw(struct sock *sk,
  2015. + const struct rate_sample *rs)
  2016. +{
  2017. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2018. + u32 n;
  2019. +
  2020. + /* If we seem to be at an operating point where we are not seeing loss
  2021. + * but we are seeing ECN marks, then when the ECN marks cease we reprobe
  2022. + * quickly (in case cross-traffic has ceased and freed up bw).
  2023. + */
  2024. + if (bbr_param(sk, ecn_reprobe_gain) && bbr->ecn_eligible &&
  2025. + bbr->ecn_in_cycle && !bbr->loss_in_cycle &&
  2026. + inet_csk(sk)->icsk_ca_state == TCP_CA_Open) {
  2027. + /* Calculate n so that when bbr_raise_inflight_hi_slope()
  2028. + * computes growth_this_round as 2^n it will be roughly the
  2029. + * desired volume of data (inflight_hi*ecn_reprobe_gain).
  2030. + */
  2031. + n = ilog2((((u64)bbr->inflight_hi *
  2032. + bbr_param(sk, ecn_reprobe_gain)) >> BBR_SCALE));
  2033. + bbr_start_bw_probe_refill(sk, n);
  2034. + return true;
  2035. + }
  2036. +
  2037. + if (bbr_has_elapsed_in_phase(sk, bbr->probe_wait_us) ||
  2038. + bbr_is_reno_coexistence_probe_time(sk)) {
  2039. + bbr_start_bw_probe_refill(sk, 0);
  2040. + return true;
  2041. + }
  2042. + return false;
  2043. +}
  2044. +
  2045. +/* Is it time to transition from PROBE_DOWN to PROBE_CRUISE? */
  2046. +static bool bbr_check_time_to_cruise(struct sock *sk, u32 inflight, u32 bw)
  2047. +{
  2048. + /* Always need to pull inflight down to leave headroom in queue. */
  2049. + if (inflight > bbr_inflight_with_headroom(sk))
  2050. + return false;
  2051. +
  2052. + return inflight <= bbr_inflight(sk, bw, BBR_UNIT);
  2053. +}
  2054. +
  2055. +/* PROBE_BW state machine: cruise, refill, probe for bw, or drain? */
  2056. +static void bbr_update_cycle_phase(struct sock *sk,
  2057. + const struct rate_sample *rs,
  2058. + struct bbr_context *ctx)
  2059. +{
  2060. + struct tcp_sock *tp = tcp_sk(sk);
  2061. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2062. + bool is_bw_probe_done = false;
  2063. + u32 inflight, bw;
  2064. +
  2065. + if (!bbr_full_bw_reached(sk))
  2066. + return;
  2067. +
  2068. + /* In DRAIN, PROBE_BW, or PROBE_RTT, adjust upper bounds. */
  2069. + if (bbr_adapt_upper_bounds(sk, rs, ctx))
  2070. + return; /* already decided state transition */
  2071. +
  2072. + if (bbr->mode != BBR_PROBE_BW)
  2073. + return;
  2074. +
  2075. + inflight = bbr_packets_in_net_at_edt(sk, rs->prior_in_flight);
  2076. + bw = bbr_max_bw(sk);
  2077. +
  2078. + switch (bbr->cycle_idx) {
  2079. + /* First we spend most of our time cruising with a pacing_gain of 1.0,
  2080. + * which paces at the estimated bw, to try to fully use the pipe
  2081. + * without building queue. If we encounter loss/ECN marks, we adapt
  2082. + * by slowing down.
  2083. + */
  2084. + case BBR_BW_PROBE_CRUISE:
  2085. + if (bbr_check_time_to_probe_bw(sk, rs))
  2086. + return; /* already decided state transition */
  2087. + break;
  2088. +
  2089. + /* After cruising, when it's time to probe, we first "refill": we send
  2090. + * at the estimated bw to fill the pipe, before probing higher and
  2091. + * knowingly risking overflowing the bottleneck buffer (causing loss).
  2092. + */
  2093. + case BBR_BW_PROBE_REFILL:
  2094. + if (bbr->round_start) {
  2095. + /* After one full round trip of sending in REFILL, we
  2096. + * start to see bw samples reflecting our REFILL, which
  2097. + * may be putting too much data in flight.
  2098. + */
  2099. + bbr->bw_probe_samples = 1;
  2100. + bbr_start_bw_probe_up(sk, ctx);
  2101. + }
  2102. + break;
  2103. +
  2104. + /* After we refill the pipe, we probe by using a pacing_gain > 1.0, to
  2105. + * probe for bw. If we have not seen loss/ECN, we try to raise inflight
  2106. + * to at least pacing_gain*BDP; note that this may take more than
  2107. + * min_rtt if min_rtt is small (e.g. on a LAN).
  2108. + *
  2109. + * We terminate PROBE_UP bandwidth probing upon any of the following:
  2110. + *
  2111. + * (1) We've pushed inflight up to hit the inflight_hi target set in the
  2112. + * most recent previous bw probe phase. Thus we want to start
  2113. + * draining the queue immediately because it's very likely the most
  2114. + * recently sent packets will fill the queue and cause drops.
  2115. + * (2) If inflight_hi has not limited bandwidth growth recently, and
  2116. + * yet delivered bandwidth has not increased much recently
  2117. + * (bbr->full_bw_now).
  2118. + * (3) Loss filter says loss rate is "too high".
  2119. + * (4) ECN filter says ECN mark rate is "too high".
  2120. + *
  2121. + * (1) (2) checked here, (3) (4) checked in bbr_is_inflight_too_high()
  2122. + */
  2123. + case BBR_BW_PROBE_UP:
  2124. + if (bbr->prev_probe_too_high &&
  2125. + inflight >= bbr->inflight_hi) {
  2126. + bbr->stopped_risky_probe = 1;
  2127. + is_bw_probe_done = true;
  2128. + } else {
  2129. + if (tp->is_cwnd_limited &&
  2130. + tcp_snd_cwnd(tp) >= bbr->inflight_hi) {
  2131. + /* inflight_hi is limiting bw growth */
  2132. + bbr_reset_full_bw(sk);
  2133. + bbr->full_bw = ctx->sample_bw;
  2134. + } else if (bbr->full_bw_now) {
  2135. + /* Plateau in estimated bw. Pipe looks full. */
  2136. + is_bw_probe_done = true;
  2137. + }
  2138. + }
  2139. + if (is_bw_probe_done) {
  2140. + bbr->prev_probe_too_high = 0; /* no loss/ECN (yet) */
  2141. + bbr_start_bw_probe_down(sk); /* restart w/ down */
  2142. + }
  2143. + break;
  2144. +
  2145. + /* After probing in PROBE_UP, we have usually accumulated some data in
  2146. + * the bottleneck buffer (if bw probing didn't find more bw). We next
  2147. + * enter PROBE_DOWN to try to drain any excess data from the queue. To
  2148. + * do this, we use a pacing_gain < 1.0. We hold this pacing gain until
  2149. + * our inflight is less then that target cruising point, which is the
  2150. + * minimum of (a) the amount needed to leave headroom, and (b) the
  2151. + * estimated BDP. Once inflight falls to match the target, we estimate
  2152. + * the queue is drained; persisting would underutilize the pipe.
  2153. + */
  2154. + case BBR_BW_PROBE_DOWN:
  2155. + if (bbr_check_time_to_probe_bw(sk, rs))
  2156. + return; /* already decided state transition */
  2157. + if (bbr_check_time_to_cruise(sk, inflight, bw))
  2158. + bbr_start_bw_probe_cruise(sk);
  2159. + break;
  2160. +
  2161. + default:
  2162. + WARN_ONCE(1, "BBR invalid cycle index %u\n", bbr->cycle_idx);
  2163. + }
  2164. +}
  2165. +
  2166. +/* Exiting PROBE_RTT, so return to bandwidth probing in STARTUP or PROBE_BW. */
  2167. +static void bbr_exit_probe_rtt(struct sock *sk)
  2168. +{
  2169. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2170. +
  2171. + bbr_reset_lower_bounds(sk);
  2172. + if (bbr_full_bw_reached(sk)) {
  2173. + bbr->mode = BBR_PROBE_BW;
  2174. + /* Raising inflight after PROBE_RTT may cause loss, so reset
  2175. + * the PROBE_BW clock and schedule the next bandwidth probe for
  2176. + * a friendly and randomized future point in time.
  2177. + */
  2178. + bbr_start_bw_probe_down(sk);
  2179. + /* Since we are exiting PROBE_RTT, we know inflight is
  2180. + * below our estimated BDP, so it is reasonable to cruise.
  2181. + */
  2182. + bbr_start_bw_probe_cruise(sk);
  2183. + } else {
  2184. + bbr->mode = BBR_STARTUP;
  2185. + }
  2186. +}
  2187. +
  2188. +/* Exit STARTUP based on loss rate > 1% and loss gaps in round >= N. Wait until
  2189. + * the end of the round in recovery to get a good estimate of how many packets
  2190. + * have been lost, and how many we need to drain with a low pacing rate.
  2191. + */
  2192. +static void bbr_check_loss_too_high_in_startup(struct sock *sk,
  2193. + const struct rate_sample *rs)
  2194. +{
  2195. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2196. +
  2197. + if (bbr_full_bw_reached(sk))
  2198. + return;
  2199. +
  2200. + /* For STARTUP exit, check the loss rate at the end of each round trip
  2201. + * of Recovery episodes in STARTUP. We check the loss rate at the end
  2202. + * of the round trip to filter out noisy/low loss and have a better
  2203. + * sense of inflight (extent of loss), so we can drain more accurately.
  2204. + */
  2205. + if (rs->losses && bbr->loss_events_in_round < 0xf)
  2206. + bbr->loss_events_in_round++; /* update saturating counter */
  2207. + if (bbr_param(sk, full_loss_cnt) && bbr->loss_round_start &&
  2208. + inet_csk(sk)->icsk_ca_state == TCP_CA_Recovery &&
  2209. + bbr->loss_events_in_round >= bbr_param(sk, full_loss_cnt) &&
  2210. + bbr_is_inflight_too_high(sk, rs)) {
  2211. + bbr_handle_queue_too_high_in_startup(sk);
  2212. + return;
  2213. + }
  2214. + if (bbr->loss_round_start)
  2215. + bbr->loss_events_in_round = 0;
  2216. +}
  2217. +
  2218. +/* Estimate when the pipe is full, using the change in delivery rate: BBR
  2219. + * estimates bw probing filled the pipe if the estimated bw hasn't changed by
  2220. + * at least bbr_full_bw_thresh (25%) after bbr_full_bw_cnt (3) non-app-limited
  2221. + * rounds. Why 3 rounds: 1: rwin autotuning grows the rwin, 2: we fill the
  2222. + * higher rwin, 3: we get higher delivery rate samples. Or transient
  2223. + * cross-traffic or radio noise can go away. CUBIC Hystart shares a similar
  2224. + * design goal, but uses delay and inter-ACK spacing instead of bandwidth.
  2225. + */
  2226. +static void bbr_check_full_bw_reached(struct sock *sk,
  2227. + const struct rate_sample *rs,
  2228. + struct bbr_context *ctx)
  2229. +{
  2230. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2231. + u32 bw_thresh, full_cnt, thresh;
  2232. +
  2233. + if (bbr->full_bw_now || rs->is_app_limited)
  2234. + return;
  2235. +
  2236. + thresh = bbr_param(sk, full_bw_thresh);
  2237. + full_cnt = bbr_param(sk, full_bw_cnt);
  2238. + bw_thresh = (u64)bbr->full_bw * thresh >> BBR_SCALE;
  2239. + if (ctx->sample_bw >= bw_thresh) {
  2240. + bbr_reset_full_bw(sk);
  2241. + bbr->full_bw = ctx->sample_bw;
  2242. + return;
  2243. + }
  2244. + if (!bbr->round_start)
  2245. + return;
  2246. + ++bbr->full_bw_cnt;
  2247. + bbr->full_bw_now = bbr->full_bw_cnt >= full_cnt;
  2248. + bbr->full_bw_reached |= bbr->full_bw_now;
  2249. +}
  2250. +
  2251. +/* If pipe is probably full, drain the queue and then enter steady-state. */
  2252. +static void bbr_check_drain(struct sock *sk, const struct rate_sample *rs,
  2253. + struct bbr_context *ctx)
  2254. +{
  2255. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2256. +
  2257. + if (bbr->mode == BBR_STARTUP && bbr_full_bw_reached(sk)) {
  2258. + bbr->mode = BBR_DRAIN; /* drain queue we created */
  2259. + /* Set ssthresh to export purely for monitoring, to signal
  2260. + * completion of initial STARTUP by setting to a non-
  2261. + * TCP_INFINITE_SSTHRESH value (ssthresh is not used by BBR).
  2262. + */
  2263. + tcp_sk(sk)->snd_ssthresh =
  2264. + bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT);
  2265. + bbr_reset_congestion_signals(sk);
  2266. + } /* fall through to check if in-flight is already small: */
  2267. + if (bbr->mode == BBR_DRAIN &&
  2268. + bbr_packets_in_net_at_edt(sk, tcp_packets_in_flight(tcp_sk(sk))) <=
  2269. + bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT)) {
  2270. + bbr->mode = BBR_PROBE_BW;
  2271. + bbr_start_bw_probe_down(sk);
  2272. + }
  2273. +}
  2274. +
  2275. +static void bbr_update_model(struct sock *sk, const struct rate_sample *rs,
  2276. + struct bbr_context *ctx)
  2277. +{
  2278. + bbr_update_congestion_signals(sk, rs, ctx);
  2279. + bbr_update_ack_aggregation(sk, rs);
  2280. + bbr_check_loss_too_high_in_startup(sk, rs);
  2281. + bbr_check_full_bw_reached(sk, rs, ctx);
  2282. + bbr_check_drain(sk, rs, ctx);
  2283. + bbr_update_cycle_phase(sk, rs, ctx);
  2284. + bbr_update_min_rtt(sk, rs);
  2285. +}
  2286. +
  2287. +/* Fast path for app-limited case.
  2288. + *
  2289. + * On each ack, we execute bbr state machine, which primarily consists of:
  2290. + * 1) update model based on new rate sample, and
  2291. + * 2) update control based on updated model or state change.
  2292. + *
  2293. + * There are certain workload/scenarios, e.g. app-limited case, where
  2294. + * either we can skip updating model or we can skip update of both model
  2295. + * as well as control. This provides signifcant softirq cpu savings for
  2296. + * processing incoming acks.
  2297. + *
  2298. + * In case of app-limited, if there is no congestion (loss/ecn) and
  2299. + * if observed bw sample is less than current estimated bw, then we can
  2300. + * skip some of the computation in bbr state processing:
  2301. + *
  2302. + * - if there is no rtt/mode/phase change: In this case, since all the
  2303. + * parameters of the network model are constant, we can skip model
  2304. + * as well control update.
  2305. + *
  2306. + * - else we can skip rest of the model update. But we still need to
  2307. + * update the control to account for the new rtt/mode/phase.
  2308. + *
  2309. + * Returns whether we can take fast path or not.
  2310. + */
  2311. +static bool bbr_run_fast_path(struct sock *sk, bool *update_model,
  2312. + const struct rate_sample *rs, struct bbr_context *ctx)
  2313. +{
  2314. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2315. + u32 prev_min_rtt_us, prev_mode;
  2316. +
  2317. + if (bbr_param(sk, fast_path) && bbr->try_fast_path &&
  2318. + rs->is_app_limited && ctx->sample_bw < bbr_max_bw(sk) &&
  2319. + !bbr->loss_in_round && !bbr->ecn_in_round ) {
  2320. + prev_mode = bbr->mode;
  2321. + prev_min_rtt_us = bbr->min_rtt_us;
  2322. + bbr_check_drain(sk, rs, ctx);
  2323. + bbr_update_cycle_phase(sk, rs, ctx);
  2324. + bbr_update_min_rtt(sk, rs);
  2325. +
  2326. + if (bbr->mode == prev_mode &&
  2327. + bbr->min_rtt_us == prev_min_rtt_us &&
  2328. + bbr->try_fast_path) {
  2329. + return true;
  2330. + }
  2331. +
  2332. + /* Skip model update, but control still needs to be updated */
  2333. + *update_model = false;
  2334. + }
  2335. + return false;
  2336. +}
  2337. +
  2338. +static void bbr3_main(struct sock *sk, u32 ack, int flag, const struct rate_sample *rs)
  2339. +{
  2340. + struct tcp_sock *tp = tcp_sk(sk);
  2341. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2342. + struct bbr_context ctx = { 0 };
  2343. + bool update_model = true;
  2344. + u32 bw, round_delivered;
  2345. + int ce_ratio = -1;
  2346. +
  2347. + if(!bbr || !bbr->initialized)
  2348. + return;
  2349. +
  2350. + round_delivered = bbr_update_round_start(sk, rs, &ctx);
  2351. + if (bbr->round_start) {
  2352. + bbr->rounds_since_probe =
  2353. + min_t(s32, bbr->rounds_since_probe + 1, 0xFF);
  2354. + ce_ratio = bbr_update_ecn_alpha(sk);
  2355. + }
  2356. + bbr_plb(sk, rs, ce_ratio);
  2357. +
  2358. + bbr->ecn_in_round |= (bbr->ecn_eligible && rs->is_ece);
  2359. + bbr_calculate_bw_sample(sk, rs, &ctx);
  2360. + bbr_update_latest_delivery_signals(sk, rs, &ctx);
  2361. +
  2362. + if (bbr_run_fast_path(sk, &update_model, rs, &ctx))
  2363. + goto out;
  2364. +
  2365. + if (update_model)
  2366. + bbr_update_model(sk, rs, &ctx);
  2367. +
  2368. + bbr_update_gains(sk);
  2369. + bw = bbr_bw(sk);
  2370. + bbr_set_pacing_rate(sk, bw, bbr->pacing_gain);
  2371. + bbr_set_cwnd(sk, rs, rs->acked_sacked, bw, bbr->cwnd_gain,
  2372. + tcp_snd_cwnd(tp), &ctx);
  2373. + bbr_bound_cwnd_for_inflight_model(sk);
  2374. +
  2375. +out:
  2376. + bbr_advance_latest_delivery_signals(sk, rs, &ctx);
  2377. + bbr->prev_ca_state = inet_csk(sk)->icsk_ca_state;
  2378. + bbr->loss_in_cycle |= rs->lost > 0;
  2379. + bbr->ecn_in_cycle |= rs->delivered_ce > 0;
  2380. +}
  2381. +
  2382. +static void bbr3_init(struct sock *sk)
  2383. +{
  2384. + struct tcp_sock *tp = tcp_sk(sk);
  2385. + struct bbr3 **bbr3_ptr = (struct bbr3 **)inet_csk_ca(sk);
  2386. + struct bbr3 *bbr;
  2387. +
  2388. + bbr = kmalloc(sizeof(struct bbr3), GFP_ATOMIC);
  2389. + if (unlikely(!bbr))
  2390. + return;
  2391. +
  2392. + memset(bbr, 0, sizeof(struct bbr3));
  2393. +
  2394. + *bbr3_ptr = bbr;
  2395. +
  2396. + bbr->skb_marked_lost = bbr3_skb_marked_lost;
  2397. +
  2398. + bbr->initialized = 1;
  2399. +
  2400. + bbr->init_cwnd = min(0x7FU, tcp_snd_cwnd(tp));
  2401. + bbr->prior_cwnd = tp->prior_cwnd;
  2402. + tp->snd_ssthresh = TCP_INFINITE_SSTHRESH;
  2403. + bbr->next_rtt_delivered = tp->delivered;
  2404. + bbr->prev_ca_state = TCP_CA_Open;
  2405. +
  2406. + bbr->probe_rtt_done_stamp = 0;
  2407. + bbr->probe_rtt_round_done = 0;
  2408. + bbr->probe_rtt_min_us = tcp_min_rtt(tp);
  2409. + bbr->probe_rtt_min_stamp = tcp_jiffies32;
  2410. + bbr->min_rtt_us = tcp_min_rtt(tp);
  2411. + bbr->min_rtt_stamp = tcp_jiffies32;
  2412. +
  2413. + bbr->has_seen_rtt = 0;
  2414. + bbr_init_pacing_rate_from_rtt(sk);
  2415. +
  2416. + bbr->round_start = 0;
  2417. + bbr->idle_restart = 0;
  2418. + bbr->full_bw_reached = 0;
  2419. + bbr->full_bw = 0;
  2420. + bbr->full_bw_cnt = 0;
  2421. + bbr->cycle_mstamp = 0;
  2422. + bbr->cycle_idx = 0;
  2423. +
  2424. + bbr_reset_startup_mode(sk);
  2425. +
  2426. + bbr->ack_epoch_mstamp = tp->tcp_mstamp;
  2427. + bbr->ack_epoch_acked = 0;
  2428. + bbr->extra_acked_win_rtts = 0;
  2429. + bbr->extra_acked_win_idx = 0;
  2430. + bbr->extra_acked[0] = 0;
  2431. + bbr->extra_acked[1] = 0;
  2432. +
  2433. + bbr->ce_state = 0;
  2434. + bbr->prior_rcv_nxt = tp->rcv_nxt;
  2435. + bbr->try_fast_path = 0;
  2436. +
  2437. + cmpxchg(&sk->sk_pacing_status, SK_PACING_NONE, SK_PACING_NEEDED);
  2438. +
  2439. + /* Start sampling ECN mark rate after first full flight is ACKed: */
  2440. + bbr->loss_round_delivered = tp->delivered + 1;
  2441. + bbr->loss_round_start = 0;
  2442. + bbr->undo_bw_lo = 0;
  2443. + bbr->undo_inflight_lo = 0;
  2444. + bbr->undo_inflight_hi = 0;
  2445. + bbr->loss_events_in_round = 0;
  2446. + bbr->startup_ecn_rounds = 0;
  2447. + bbr_reset_congestion_signals(sk);
  2448. + bbr->bw_lo = ~0U;
  2449. + bbr->bw_hi[0] = 0;
  2450. + bbr->bw_hi[1] = 0;
  2451. + bbr->inflight_lo = ~0U;
  2452. + bbr->inflight_hi = ~0U;
  2453. + bbr_reset_full_bw(sk);
  2454. + bbr->bw_probe_up_cnt = ~0U;
  2455. + bbr->bw_probe_up_acks = 0;
  2456. + bbr->bw_probe_up_rounds = 0;
  2457. + bbr->probe_wait_us = 0;
  2458. + bbr->stopped_risky_probe = 0;
  2459. + bbr->ack_phase = BBR_ACKS_INIT;
  2460. + bbr->rounds_since_probe = 0;
  2461. + bbr->bw_probe_samples = 0;
  2462. + bbr->prev_probe_too_high = 0;
  2463. + bbr->ecn_eligible = 0;
  2464. + bbr->ecn_alpha = bbr_param(sk, ecn_alpha_init);
  2465. + bbr->alpha_last_delivered = 0;
  2466. + bbr->alpha_last_delivered_ce = 0;
  2467. + bbr->plb.pause_until = 0;
  2468. +
  2469. + tp->fast_ack_mode = bbr_fast_ack_mode ? 1 : 0;
  2470. +
  2471. + if (bbr_can_use_ecn(sk))
  2472. + tp->ecn_flags |= TCP_ECN_ECT_PERMANENT;
  2473. +}
  2474. +
  2475. +__used noinline static void bbr3_release(struct sock *sk)
  2476. +{
  2477. + struct bbr3 **bbr3_ptr = (struct bbr3 **)inet_csk_ca(sk);
  2478. +
  2479. + if (*bbr3_ptr) {
  2480. + kfree(*bbr3_ptr);
  2481. + *bbr3_ptr = NULL;
  2482. + }
  2483. +}
  2484. +
  2485. +/* BBR marks the current round trip as a loss round. */
  2486. +static void bbr_note_loss(struct sock *sk)
  2487. +{
  2488. + struct tcp_sock *tp = tcp_sk(sk);
  2489. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2490. +
  2491. + /* Capture "current" data over the full round trip of loss, to
  2492. + * have a better chance of observing the full capacity of the path.
  2493. + */
  2494. + if (!bbr->loss_in_round) /* first loss in this round trip? */
  2495. + bbr->loss_round_delivered = tp->delivered; /* set round trip */
  2496. + bbr->loss_in_round = 1;
  2497. + bbr->loss_in_cycle = 1;
  2498. +}
  2499. +
  2500. +/* Core TCP stack informs us that the given skb was just marked lost. */
  2501. +static void bbr3_skb_marked_lost(struct sock *sk,
  2502. + const struct sk_buff *skb)
  2503. +{
  2504. + struct tcp_sock *tp = tcp_sk(sk);
  2505. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2506. + struct tcp_skb_cb *scb = TCP_SKB_CB(skb);
  2507. + struct rate_sample rs = {};
  2508. +
  2509. + bbr_note_loss(sk);
  2510. +
  2511. + if (!bbr->bw_probe_samples)
  2512. + return; /* not an skb sent while probing for bandwidth */
  2513. + if (unlikely(!scb->tx.delivered_mstamp))
  2514. + return; /* skb was SACKed, reneged, marked lost; ignore it */
  2515. + /* We are probing for bandwidth. Construct a rate sample that
  2516. + * estimates what happened in the flight leading up to this lost skb,
  2517. + * then see if the loss rate went too high, and if so at which packet.
  2518. + */
  2519. + rs.tx_in_flight = scb->tx.in_flight;
  2520. + rs.lost = tp->lost - scb->tx.lost;
  2521. + rs.is_app_limited = scb->tx.is_app_limited;
  2522. + if (bbr_is_inflight_too_high(sk, &rs)) {
  2523. + rs.tx_in_flight = bbr_inflight_hi_from_lost_skb(sk, &rs, skb);
  2524. + bbr_handle_inflight_too_high(sk, &rs);
  2525. + }
  2526. +}
  2527. +
  2528. +static void bbr_run_loss_probe_recovery(struct sock *sk)
  2529. +{
  2530. + struct tcp_sock *tp = tcp_sk(sk);
  2531. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2532. + struct rate_sample rs = {0};
  2533. +
  2534. + bbr_note_loss(sk);
  2535. +
  2536. + if (!bbr->bw_probe_samples)
  2537. + return; /* not sent while probing for bandwidth */
  2538. + /* We are probing for bandwidth. Construct a rate sample that
  2539. + * estimates what happened in the flight leading up to this
  2540. + * loss, then see if the loss rate went too high.
  2541. + */
  2542. + rs.lost = 1; /* TLP probe repaired loss of a single segment */
  2543. + rs.tx_in_flight = bbr->inflight_latest + rs.lost;
  2544. + rs.is_app_limited = tp->tlp_orig_data_app_limited;
  2545. + if (bbr_is_inflight_too_high(sk, &rs))
  2546. + bbr_handle_inflight_too_high(sk, &rs);
  2547. +}
  2548. +
  2549. +/* Revert short-term model if current loss recovery event was spurious. */
  2550. +static u32 bbr3_undo_cwnd(struct sock *sk)
  2551. +{
  2552. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2553. +
  2554. + if(!bbr || !bbr->initialized)
  2555. + return tcp_sk(sk)->snd_cwnd;
  2556. +
  2557. + bbr_reset_full_bw(sk); /* spurious slow-down; reset full bw detector */
  2558. + bbr->loss_in_round = 0;
  2559. +
  2560. + /* Revert to cwnd and other state saved before loss episode. */
  2561. + bbr->bw_lo = max(bbr->bw_lo, bbr->undo_bw_lo);
  2562. + bbr->inflight_lo = max(bbr->inflight_lo, bbr->undo_inflight_lo);
  2563. + bbr->inflight_hi = max(bbr->inflight_hi, bbr->undo_inflight_hi);
  2564. + bbr->try_fast_path = 0; /* take slow path to set proper cwnd, pacing */
  2565. + return bbr->prior_cwnd;
  2566. +}
  2567. +
  2568. +/* Entering loss recovery, so save state for when we undo recovery. */
  2569. +static u32 bbr3_ssthresh(struct sock *sk)
  2570. +{
  2571. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2572. +
  2573. + if(!bbr || !bbr->initialized)
  2574. + return tcp_sk(sk)->snd_ssthresh;
  2575. +
  2576. + bbr_save_cwnd(sk);
  2577. + /* For undo, save state that adapts based on loss signal. */
  2578. + bbr->undo_bw_lo = bbr->bw_lo;
  2579. + bbr->undo_inflight_lo = bbr->inflight_lo;
  2580. + bbr->undo_inflight_hi = bbr->inflight_hi;
  2581. + return tcp_sk(sk)->snd_ssthresh;
  2582. +}
  2583. +
  2584. +static enum tcp_bbr_phase bbr_get_phase(struct bbr3 *bbr)
  2585. +{
  2586. + switch (bbr->mode) {
  2587. + case BBR_STARTUP:
  2588. + return BBR_PHASE_STARTUP;
  2589. + case BBR_DRAIN:
  2590. + return BBR_PHASE_DRAIN;
  2591. + case BBR_PROBE_BW:
  2592. + break;
  2593. + case BBR_PROBE_RTT:
  2594. + return BBR_PHASE_PROBE_RTT;
  2595. + default:
  2596. + return BBR_PHASE_INVALID;
  2597. + }
  2598. + switch (bbr->cycle_idx) {
  2599. + case BBR_BW_PROBE_UP:
  2600. + return BBR_PHASE_PROBE_BW_UP;
  2601. + case BBR_BW_PROBE_DOWN:
  2602. + return BBR_PHASE_PROBE_BW_DOWN;
  2603. + case BBR_BW_PROBE_CRUISE:
  2604. + return BBR_PHASE_PROBE_BW_CRUISE;
  2605. + case BBR_BW_PROBE_REFILL:
  2606. + return BBR_PHASE_PROBE_BW_REFILL;
  2607. + default:
  2608. + return BBR_PHASE_INVALID;
  2609. + }
  2610. +}
  2611. +
  2612. +static size_t bbr3_get_info(struct sock *sk, u32 ext, int *attr,
  2613. + union tcp_cc_info *info)
  2614. +{
  2615. + if (ext & (1 << (INET_DIAG_BBRINFO - 1)) ||
  2616. + ext & (1 << (INET_DIAG_VEGASINFO - 1))) {
  2617. + struct tcp_bbr_info *bbr_info;
  2618. + u64 bw, bw_hi, bw_lo;
  2619. +
  2620. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2621. + if(!bbr || !bbr->initialized)
  2622. + return 0;
  2623. +
  2624. + bw = bbr_bw_bytes_per_sec(sk, bbr_bw(sk));
  2625. + bw_hi = bbr_bw_bytes_per_sec(sk, bbr_max_bw(sk));
  2626. + bw_lo = bbr->bw_lo == ~0U ?
  2627. + ~0ULL : bbr_bw_bytes_per_sec(sk, bbr->bw_lo);
  2628. + bbr_info = &info->bbr;
  2629. +
  2630. + memset(bbr_info, 0, sizeof(*bbr_info));
  2631. + bbr_info->bbr_bw_lo = (u32)bw;
  2632. + bbr_info->bbr_bw_hi = (u32)(bw >> 32);
  2633. + bbr_info->bbr_min_rtt = bbr->min_rtt_us;
  2634. + bbr_info->bbr_pacing_gain = bbr->pacing_gain;
  2635. + bbr_info->bbr_cwnd_gain = bbr->cwnd_gain;
  2636. + bbr_info->bbr_bw_hi_lsb = (u32)bw_hi;
  2637. + bbr_info->bbr_bw_hi_msb = (u32)(bw_hi >> 32);
  2638. + bbr_info->bbr_bw_lo_lsb = (u32)bw_lo;
  2639. + bbr_info->bbr_bw_lo_msb = (u32)(bw_lo >> 32);
  2640. + bbr_info->bbr_mode = bbr->mode;
  2641. + bbr_info->bbr_phase = (__u8)bbr_get_phase(bbr);
  2642. + bbr_info->bbr_version = (__u8)BBR_VERSION;
  2643. + bbr_info->bbr_inflight_lo = bbr->inflight_lo;
  2644. + bbr_info->bbr_inflight_hi = bbr->inflight_hi;
  2645. + bbr_info->bbr_extra_acked = bbr_extra_acked(sk);
  2646. + *attr = INET_DIAG_BBRINFO;
  2647. + return sizeof(*bbr_info);
  2648. + }
  2649. + return 0;
  2650. +}
  2651. +
  2652. +static void bbr3_set_state(struct sock *sk, u8 new_state)
  2653. +{
  2654. + struct tcp_sock *tp = tcp_sk(sk);
  2655. + struct bbr3 *bbr = bbr3_get_priv(sk);
  2656. +
  2657. + if(!bbr || !bbr->initialized)
  2658. + return;
  2659. +
  2660. + if (new_state == TCP_CA_Loss) {
  2661. +
  2662. + bbr->prev_ca_state = TCP_CA_Loss;
  2663. + tcp_plb_update_state_upon_rto(sk, &bbr->plb);
  2664. + /* The tcp_write_timeout() call to sk_rethink_txhash() likely
  2665. + * repathed this flow, so re-learn the min network RTT on the
  2666. + * new path:
  2667. + */
  2668. + bbr_reset_full_bw(sk);
  2669. + if (!bbr_is_probing_bandwidth(sk) && bbr->inflight_lo == ~0U) {
  2670. + /* bbr_adapt_lower_bounds() needs cwnd before
  2671. + * we suffered an RTO, to update inflight_lo:
  2672. + */
  2673. + bbr->inflight_lo =
  2674. + max(tcp_snd_cwnd(tp), bbr->prior_cwnd);
  2675. + }
  2676. + } else if (bbr->prev_ca_state == TCP_CA_Loss &&
  2677. + new_state != TCP_CA_Loss) {
  2678. + bbr_exit_loss_recovery(sk);
  2679. + }
  2680. +}
  2681. +
  2682. +
  2683. +static struct tcp_congestion_ops tcp_bbr_cong_ops __read_mostly = {
  2684. + .flags = TCP_CONG_NON_RESTRICTED | TCP_CONG_WANTS_CE_EVENTS,
  2685. + .name = "bbr3",
  2686. + .owner = THIS_MODULE,
  2687. + .init = bbr3_init,
  2688. + .release = bbr3_release,
  2689. + .cong_control = bbr3_main,
  2690. + .sndbuf_expand = bbr3_sndbuf_expand,
  2691. + .undo_cwnd = bbr3_undo_cwnd,
  2692. + .cwnd_event = bbr3_cwnd_event,
  2693. + .ssthresh = bbr3_ssthresh,
  2694. + .min_tso_segs = bbr3_tso_segs,
  2695. + .get_info = bbr3_get_info,
  2696. + .set_state = bbr3_set_state,
  2697. +};
  2698. +
  2699. +static int __init bbr3_register(void)
  2700. +{
  2701. + return tcp_register_congestion_control(&tcp_bbr_cong_ops);
  2702. +}
  2703. +
  2704. +static void __exit bbr3_unregister(void)
  2705. +{
  2706. + tcp_unregister_congestion_control(&tcp_bbr_cong_ops);
  2707. +}
  2708. +
  2709. +module_init(bbr3_register);
  2710. +module_exit(bbr3_unregister);
  2711. +
  2712. +MODULE_AUTHOR("Van Jacobson <vanj@google.com>");
  2713. +MODULE_AUTHOR("Neal Cardwell <ncardwell@google.com>");
  2714. +MODULE_AUTHOR("Yuchung Cheng <ycheng@google.com>");
  2715. +MODULE_AUTHOR("Soheil Hassas Yeganeh <soheil@google.com>");
  2716. +MODULE_AUTHOR("Priyaranjan Jha <priyarjha@google.com>");
  2717. +MODULE_AUTHOR("Yousuk Seung <ysseung@google.com>");
  2718. +MODULE_AUTHOR("Kevin Yang <yyd@google.com>");
  2719. +MODULE_AUTHOR("Arjun Roy <arjunroy@google.com>");
  2720. +MODULE_AUTHOR("David Morley <morleyd@google.com>");
  2721. +
  2722. +MODULE_LICENSE("Dual BSD/GPL");
  2723. +MODULE_DESCRIPTION("TCP BBR (Bottleneck Bandwidth and RTT)");
  2724. +MODULE_VERSION(__stringify(BBR_VERSION));
  2725. diff --git a/net/ipv4/tcp_cong.c b/net/ipv4/tcp_cong.c
  2726. index 0306d257f..28f581c0d 100644
  2727. --- a/net/ipv4/tcp_cong.c
  2728. +++ b/net/ipv4/tcp_cong.c
  2729. @@ -237,6 +237,7 @@ void tcp_init_congestion_control(struct sock *sk)
  2730. struct inet_connection_sock *icsk = inet_csk(sk);
  2731. tcp_sk(sk)->prior_ssthresh = 0;
  2732. + tcp_sk(sk)->fast_ack_mode = 0;
  2733. if (icsk->icsk_ca_ops->init)
  2734. icsk->icsk_ca_ops->init(sk);
  2735. if (tcp_ca_needs_ecn(sk))
  2736. diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
  2737. index 5158676a7..b7a4c9a30 100644
  2738. --- a/net/ipv4/tcp_input.c
  2739. +++ b/net/ipv4/tcp_input.c
  2740. @@ -376,7 +376,7 @@ static void __tcp_ecn_check_ce(struct sock *sk, const struct sk_buff *skb)
  2741. tcp_enter_quickack_mode(sk, 2);
  2742. break;
  2743. case INET_ECN_CE:
  2744. - if (tcp_ca_needs_ecn(sk))
  2745. + if (tcp_ca_wants_ce_events(sk))
  2746. tcp_ca_event(sk, CA_EVENT_ECN_IS_CE);
  2747. if (!(tp->ecn_flags & TCP_ECN_DEMAND_CWR)) {
  2748. @@ -387,7 +387,7 @@ static void __tcp_ecn_check_ce(struct sock *sk, const struct sk_buff *skb)
  2749. tp->ecn_flags |= TCP_ECN_SEEN;
  2750. break;
  2751. default:
  2752. - if (tcp_ca_needs_ecn(sk))
  2753. + if (tcp_ca_wants_ce_events(sk))
  2754. tcp_ca_event(sk, CA_EVENT_ECN_NO_CE);
  2755. tp->ecn_flags |= TCP_ECN_SEEN;
  2756. break;
  2757. @@ -1140,7 +1140,18 @@ static void tcp_verify_retransmit_hint(struct tcp_sock *tp, struct sk_buff *skb)
  2758. */
  2759. static void tcp_notify_skb_loss_event(struct tcp_sock *tp, const struct sk_buff *skb)
  2760. {
  2761. + struct bbr3 *bbr3 = NULL;
  2762. + struct sock *sk = (struct sock *)tp;
  2763. + const struct tcp_congestion_ops *ca_ops = inet_csk(sk)->icsk_ca_ops;
  2764. +
  2765. tp->lost += tcp_skb_pcount(skb);
  2766. +
  2767. + if(ca_ops && strncmp(ca_ops->name, "bbr3", 4) == 0)
  2768. + {
  2769. + bbr3 = *(struct bbr3 **)inet_csk_ca(sk);
  2770. + if (bbr3 && bbr3->skb_marked_lost)
  2771. + bbr3->skb_marked_lost(sk, skb);
  2772. + }
  2773. }
  2774. void tcp_mark_skb_lost(struct sock *sk, struct sk_buff *skb)
  2775. @@ -1512,6 +1523,17 @@ static bool tcp_shifted_skb(struct sock *sk, struct sk_buff *prev,
  2776. WARN_ON_ONCE(tcp_skb_pcount(skb) < pcount);
  2777. tcp_skb_pcount_add(skb, -pcount);
  2778. + /* Adjust tx.in_flight as pcount is shifted from skb to prev. */
  2779. + if (WARN_ONCE(TCP_SKB_CB(skb)->tx.in_flight < pcount,
  2780. + "prev in_flight: %u skb in_flight: %u pcount: %u",
  2781. + TCP_SKB_CB(prev)->tx.in_flight,
  2782. + TCP_SKB_CB(skb)->tx.in_flight,
  2783. + pcount))
  2784. + TCP_SKB_CB(skb)->tx.in_flight = 0;
  2785. + else
  2786. + TCP_SKB_CB(skb)->tx.in_flight -= pcount;
  2787. + TCP_SKB_CB(prev)->tx.in_flight += pcount;
  2788. +
  2789. /* When we're adding to gso_segs == 1, gso_size will be zero,
  2790. * in theory this shouldn't be necessary but as long as DSACK
  2791. * code can come after this skb later on it's better to keep
  2792. @@ -3853,7 +3875,8 @@ static void tcp_replace_ts_recent(struct tcp_sock *tp, u32 seq)
  2793. /* This routine deals with acks during a TLP episode and ends an episode by
  2794. * resetting tlp_high_seq. Ref: TLP algorithm in draft-ietf-tcpm-rack
  2795. */
  2796. -static void tcp_process_tlp_ack(struct sock *sk, u32 ack, int flag)
  2797. +static void tcp_process_tlp_ack(struct sock *sk, u32 ack, int flag,
  2798. + struct rate_sample *rs)
  2799. {
  2800. struct tcp_sock *tp = tcp_sk(sk);
  2801. @@ -3870,6 +3893,7 @@ static void tcp_process_tlp_ack(struct sock *sk, u32 ack, int flag)
  2802. /* ACK advances: there was a loss, so reduce cwnd. Reset
  2803. * tlp_high_seq in tcp_init_cwnd_reduction()
  2804. */
  2805. + tcp_ca_event(sk, CA_EVENT_TLP_RECOVERY);
  2806. tcp_init_cwnd_reduction(sk);
  2807. tcp_set_ca_state(sk, TCP_CA_CWR);
  2808. tcp_end_cwnd_reduction(sk);
  2809. @@ -3880,6 +3904,11 @@ static void tcp_process_tlp_ack(struct sock *sk, u32 ack, int flag)
  2810. FLAG_NOT_DUP | FLAG_DATA_SACKED))) {
  2811. /* Pure dupack: original and TLP probe arrived; no loss */
  2812. tp->tlp_high_seq = 0;
  2813. + } else {
  2814. + /* This ACK matches a TLP retransmit. We cannot yet tell if
  2815. + * this ACK is for the original or the TLP retransmit.
  2816. + */
  2817. + rs->is_acking_tlp_retrans_seq = 1;
  2818. }
  2819. }
  2820. @@ -3999,6 +4028,7 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
  2821. prior_fack = tcp_is_sack(tp) ? tcp_highest_sack_seq(tp) : tp->snd_una;
  2822. rs.prior_in_flight = tcp_packets_in_flight(tp);
  2823. + tcp_rate_check_app_limited(sk);
  2824. /* ts_recent update must be made after we are sure that the packet
  2825. * is in window.
  2826. @@ -4064,7 +4094,7 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
  2827. tcp_in_ack_event(sk, flag);
  2828. if (tp->tlp_high_seq)
  2829. - tcp_process_tlp_ack(sk, ack, flag);
  2830. + tcp_process_tlp_ack(sk, ack, flag, &rs);
  2831. if (tcp_ack_is_dubious(sk, flag)) {
  2832. if (!(flag & (FLAG_SND_UNA_ADVANCED |
  2833. @@ -4088,6 +4118,7 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
  2834. delivered = tcp_newly_delivered(sk, delivered, flag);
  2835. lost = tp->lost - lost; /* freshly marked lost */
  2836. rs.is_ack_delayed = !!(flag & FLAG_ACK_MAYBE_DELAYED);
  2837. + rs.is_ece = !!(flag & FLAG_ECE);
  2838. tcp_rate_gen(sk, delivered, lost, is_sack_reneg, sack_state.rate);
  2839. tcp_cong_control(sk, ack, delivered, flag, sack_state.rate);
  2840. tcp_xmit_recovery(sk, rexmit);
  2841. @@ -4108,7 +4139,7 @@ static int tcp_ack(struct sock *sk, const struct sk_buff *skb, int flag)
  2842. tcp_ack_probe(sk);
  2843. if (tp->tlp_high_seq)
  2844. - tcp_process_tlp_ack(sk, ack, flag);
  2845. + tcp_process_tlp_ack(sk, ack, flag, &rs);
  2846. return 1;
  2847. old_ack:
  2848. @@ -5750,7 +5781,7 @@ static void tcp_new_space(struct sock *sk)
  2849. sk);
  2850. }
  2851. -/* Caller made space either from:
  2852. +/* Caller made space either from:->
  2853. * 1) Freeing skbs in rtx queues (after tp->snd_una has advanced)
  2854. * 2) Sent skbs from output queue (and thus advancing tp->snd_nxt)
  2855. *
  2856. @@ -5788,13 +5819,14 @@ static void __tcp_ack_snd_check(struct sock *sk, int ofo_possible)
  2857. /* More than one full frame received... */
  2858. if (((tp->rcv_nxt - tp->rcv_wup) > inet_csk(sk)->icsk_ack.rcv_mss &&
  2859. + (tp->fast_ack_mode == 1 ||
  2860. /* ... and right edge of window advances far enough.
  2861. * (tcp_recvmsg() will send ACK otherwise).
  2862. * If application uses SO_RCVLOWAT, we want send ack now if
  2863. * we have not received enough bytes to satisfy the condition.
  2864. */
  2865. (tp->rcv_nxt - tp->copied_seq < sk->sk_rcvlowat ||
  2866. - __tcp_select_window(sk) >= tp->rcv_wnd)) ||
  2867. + __tcp_select_window(sk) >= tp->rcv_wnd))) ||
  2868. /* We ACK each frame or... */
  2869. tcp_in_quickack_mode(sk) ||
  2870. /* Protocol state mandates a one-time immediate ACK */
  2871. diff --git a/net/ipv4/tcp_minisocks.c b/net/ipv4/tcp_minisocks.c
  2872. index adddfb7d9..806a53064 100644
  2873. --- a/net/ipv4/tcp_minisocks.c
  2874. +++ b/net/ipv4/tcp_minisocks.c
  2875. @@ -462,6 +462,8 @@ void tcp_ca_openreq_child(struct sock *sk, const struct dst_entry *dst)
  2876. u32 ca_key = dst_metric(dst, RTAX_CC_ALGO);
  2877. bool ca_got_dst = false;
  2878. + tcp_set_ecn_low_from_dst(sk, dst);
  2879. +
  2880. if (ca_key != TCP_CA_UNSPEC) {
  2881. const struct tcp_congestion_ops *ca;
  2882. diff --git a/net/ipv4/tcp_output.c b/net/ipv4/tcp_output.c
  2883. index d75c1cdfa..2c7c91a28 100644
  2884. --- a/net/ipv4/tcp_output.c
  2885. +++ b/net/ipv4/tcp_output.c
  2886. @@ -346,9 +346,9 @@ static void tcp_ecn_send_syn(struct sock *sk, struct sk_buff *skb)
  2887. bool use_ecn = READ_ONCE(sock_net(sk)->ipv4.sysctl_tcp_ecn) == 1 ||
  2888. tcp_ca_needs_ecn(sk) || bpf_needs_ecn;
  2889. + const struct dst_entry *dst = __sk_dst_get(sk);
  2890. +
  2891. if (!use_ecn) {
  2892. - const struct dst_entry *dst = __sk_dst_get(sk);
  2893. -
  2894. if (dst && dst_feature(dst, RTAX_FEATURE_ECN))
  2895. use_ecn = true;
  2896. }
  2897. @@ -360,6 +360,9 @@ static void tcp_ecn_send_syn(struct sock *sk, struct sk_buff *skb)
  2898. tp->ecn_flags = TCP_ECN_OK;
  2899. if (tcp_ca_needs_ecn(sk) || bpf_needs_ecn)
  2900. INET_ECN_xmit(sk);
  2901. +
  2902. + if (dst)
  2903. + tcp_set_ecn_low_from_dst(sk, dst);
  2904. }
  2905. }
  2906. @@ -397,7 +400,8 @@ static void tcp_ecn_send(struct sock *sk, struct sk_buff *skb,
  2907. th->cwr = 1;
  2908. skb_shinfo(skb)->gso_type |= SKB_GSO_TCP_ECN;
  2909. }
  2910. - } else if (!tcp_ca_needs_ecn(sk)) {
  2911. + } else if (!(tp->ecn_flags & TCP_ECN_ECT_PERMANENT) &&
  2912. + !tcp_ca_needs_ecn(sk)) {
  2913. /* ACK or retransmitted segment: clear ECT|CE */
  2914. INET_ECN_dontxmit(sk);
  2915. }
  2916. @@ -1612,7 +1616,7 @@ int tcp_fragment(struct sock *sk, enum tcp_queue tcp_queue,
  2917. {
  2918. struct tcp_sock *tp = tcp_sk(sk);
  2919. struct sk_buff *buff;
  2920. - int old_factor;
  2921. + int old_factor, inflight_prev;
  2922. long limit;
  2923. int nlen;
  2924. u8 flags;
  2925. @@ -1687,6 +1691,30 @@ int tcp_fragment(struct sock *sk, enum tcp_queue tcp_queue,
  2926. if (diff)
  2927. tcp_adjust_pcount(sk, skb, diff);
  2928. +
  2929. + inflight_prev = TCP_SKB_CB(skb)->tx.in_flight - old_factor;
  2930. + if (inflight_prev < 0) {
  2931. + WARN_ONCE(tcp_skb_tx_in_flight_is_suspicious(
  2932. + old_factor,
  2933. + TCP_SKB_CB(skb)->sacked,
  2934. + TCP_SKB_CB(skb)->tx.in_flight),
  2935. + "inconsistent: tx.in_flight: %u "
  2936. + "old_factor: %d mss: %u sacked: %u "
  2937. + "1st pcount: %d 2nd pcount: %d "
  2938. + "1st len: %u 2nd len: %u ",
  2939. + TCP_SKB_CB(skb)->tx.in_flight, old_factor,
  2940. + mss_now, TCP_SKB_CB(skb)->sacked,
  2941. + tcp_skb_pcount(skb), tcp_skb_pcount(buff),
  2942. + skb->len, buff->len);
  2943. + inflight_prev = 0;
  2944. + }
  2945. + /* Set 1st tx.in_flight as if 1st were sent by itself: */
  2946. + TCP_SKB_CB(skb)->tx.in_flight = inflight_prev +
  2947. + tcp_skb_pcount(skb);
  2948. + /* Set 2nd tx.in_flight with new 1st and 2nd pcounts: */
  2949. + TCP_SKB_CB(buff)->tx.in_flight = inflight_prev +
  2950. + tcp_skb_pcount(skb) +
  2951. + tcp_skb_pcount(buff);
  2952. }
  2953. /* Link BUFF into the send queue. */
  2954. @@ -3002,6 +3030,7 @@ void tcp_send_loss_probe(struct sock *sk)
  2955. if (WARN_ON(!skb || !tcp_skb_pcount(skb)))
  2956. goto rearm_timer;
  2957. + tp->tlp_orig_data_app_limited = TCP_SKB_CB(skb)->tx.is_app_limited;
  2958. if (__tcp_retransmit_skb(sk, skb, 1))
  2959. goto rearm_timer;
  2960. diff --git a/net/ipv4/tcp_rate.c b/net/ipv4/tcp_rate.c
  2961. index a8f6d9d06..3ad92f7b3 100644
  2962. --- a/net/ipv4/tcp_rate.c
  2963. +++ b/net/ipv4/tcp_rate.c
  2964. @@ -34,6 +34,24 @@
  2965. * ready to send in the write queue.
  2966. */
  2967. +void tcp_set_tx_in_flight(struct sock *sk, struct sk_buff *skb)
  2968. +{
  2969. + struct tcp_sock *tp = tcp_sk(sk);
  2970. + u32 in_flight;
  2971. +
  2972. + /* Check, sanitize, and record packets in flight after skb was sent. */
  2973. + in_flight = tcp_packets_in_flight(tp) + tcp_skb_pcount(skb);
  2974. + if (WARN_ONCE(in_flight > TCPCB_IN_FLIGHT_MAX,
  2975. + "insane in_flight %u cc %s mss %u "
  2976. + "cwnd %u pif %u %u %u %u\n",
  2977. + in_flight, inet_csk(sk)->icsk_ca_ops->name,
  2978. + tp->mss_cache, tp->snd_cwnd,
  2979. + tp->packets_out, tp->retrans_out,
  2980. + tp->sacked_out, tp->lost_out))
  2981. + in_flight = TCPCB_IN_FLIGHT_MAX;
  2982. + TCP_SKB_CB(skb)->tx.in_flight = in_flight;
  2983. +}
  2984. +
  2985. /* Snapshot the current delivery information in the skb, to generate
  2986. * a rate sample later when the skb is (s)acked in tcp_rate_skb_delivered().
  2987. */
  2988. @@ -66,7 +84,9 @@ void tcp_rate_skb_sent(struct sock *sk, struct sk_buff *skb)
  2989. TCP_SKB_CB(skb)->tx.delivered_mstamp = tp->delivered_mstamp;
  2990. TCP_SKB_CB(skb)->tx.delivered = tp->delivered;
  2991. TCP_SKB_CB(skb)->tx.delivered_ce = tp->delivered_ce;
  2992. + TCP_SKB_CB(skb)->tx.lost = tp->lost;
  2993. TCP_SKB_CB(skb)->tx.is_app_limited = tp->app_limited ? 1 : 0;
  2994. + tcp_set_tx_in_flight(sk, skb);
  2995. }
  2996. /* When an skb is sacked or acked, we fill in the rate sample with the (prior)
  2997. @@ -91,17 +111,19 @@ void tcp_rate_skb_delivered(struct sock *sk, struct sk_buff *skb,
  2998. if (!rs->prior_delivered ||
  2999. tcp_skb_sent_after(tx_tstamp, tp->first_tx_mstamp,
  3000. scb->end_seq, rs->last_end_seq)) {
  3001. + rs->prior_lost = scb->tx.lost;
  3002. rs->prior_delivered_ce = scb->tx.delivered_ce;
  3003. rs->prior_delivered = scb->tx.delivered;
  3004. rs->prior_mstamp = scb->tx.delivered_mstamp;
  3005. rs->is_app_limited = scb->tx.is_app_limited;
  3006. rs->is_retrans = scb->sacked & TCPCB_RETRANS;
  3007. + rs->tx_in_flight = scb->tx.in_flight;
  3008. rs->last_end_seq = scb->end_seq;
  3009. /* Record send time of most recently ACKed packet: */
  3010. tp->first_tx_mstamp = tx_tstamp;
  3011. /* Find the duration of the "send phase" of this window: */
  3012. - rs->interval_us = tcp_stamp_us_delta(tp->first_tx_mstamp,
  3013. + rs->interval_us = tcp_stamp32_us_delta(tp->first_tx_mstamp,
  3014. scb->tx.first_tx_mstamp);
  3015. }
  3016. @@ -144,6 +166,7 @@ void tcp_rate_gen(struct sock *sk, u32 delivered, u32 lost,
  3017. return;
  3018. }
  3019. rs->delivered = tp->delivered - rs->prior_delivered;
  3020. + rs->lost = tp->lost - rs->prior_lost;
  3021. rs->delivered_ce = tp->delivered_ce - rs->prior_delivered_ce;
  3022. /* delivered_ce occupies less than 32 bits in the skb control block */