/* Scale factor for rate in pkt/uSec unit to avoid truncation in bandwidth *estimation.Therateunit~=(1500bytes/1usec/2^24)~=715bps. *Thishandlesbandwidthsfrom0.06pps(715bps)to256Mpps(3Tbps)inau32. *Sincetheminimumwindowis>=4packets,thelowerboundisn't *anissue.Theupperboundisn'tanissuewithexistingtechnologies.
*/ #define BW_SCALE 24 #define BW_UNIT (1 << BW_SCALE)
#define BBR_SCALE 8/* scaling factor for fractions in BBR (e.g. gains) */ #define BBR_UNIT (1 << BBR_SCALE)
/* BBR has the following modes for deciding how fast to send: */ enum bbr_mode {
BBR_STARTUP, /* ramp up sending rate rapidly to fill pipe */
BBR_DRAIN, /* drain any queue created during startup */
BBR_PROBE_BW, /* discover, share bw: pace around estimated bw */
BBR_PROBE_RTT, /* cut inflight to min to probe min_rtt */
};
/* BBR congestion control block */ struct bbr {
u32 min_rtt_us; /* min RTT in min_rtt_win_sec window */
u32 min_rtt_stamp; /* timestamp of min_rtt_us */
u32 probe_rtt_done_stamp; /* end time for BBR_PROBE_RTT mode */ struct minmax bw; /* Max recent delivery rate in pkts/uS << 24 */
u32 rtt_cnt; /* count of packet-timed rounds elapsed */
u32 next_rtt_delivered; /* scb->tx.delivered at end of round */
u64 cycle_mstamp; /* time of this cycle phase start */
u32 mode:3, /* current bbr_mode in state machine */
prev_ca_state:3, /* CA state on previous ACK */
packet_conservation:1, /* use packet conservation? */
round_start:1, /* start of packet-timed tx->ack round? */
idle_restart:1, /* restarting after idle? */
probe_rtt_round_done:1, /* a BBR_PROBE_RTT round at 4 pkts? */
unused:13,
lt_is_sampling:1, /* taking long-term ("LT") samples now? */
lt_rtt_cnt:7, /* round trips in long-term interval */
lt_use_bw:1; /* use lt_bw as our bw estimate? */
u32 lt_bw; /* LT est delivery rate in pkts/uS << 24 */
u32 lt_last_delivered; /* LT intvl start: tp->delivered */
u32 lt_last_stamp; /* LT intvl start: tp->delivered_mstamp */
u32 lt_last_lost; /* LT intvl start: tp->lost */
u32 pacing_gain:10, /* current gain for setting pacing rate */
cwnd_gain:10, /* current gain for setting cwnd */
full_bw_reached:1, /* reached full bw in Startup? */
full_bw_cnt:2, /* number of rounds without large bw gains */
cycle_idx:3, /* current index in pacing_gain cycle array */
has_seen_rtt:1, /* have we seen an RTT sample yet? */
unused_b:5;
u32 prior_cwnd; /* prior cwnd upon entering loss recovery */
u32 full_bw; /* recent bw, to estimate if pipe is full */
/* For tracking ACK aggregation: */
u64 ack_epoch_mstamp; /* start of ACK sampling epoch */
u16 extra_acked[2]; /* max excess data ACKed in epoch */
u32 ack_epoch_acked:20, /* packets (S)ACKed in sampling epoch */
extra_acked_win_rtts:5, /* age of extra_acked, in round trips */
extra_acked_win_idx:1, /* current index in extra_acked array */
unused_c:6;
};
#define CYCLE_LEN 8/* number of phases in a pacing gain cycle */
/* Window length of bw filter (in rounds): */ staticconstint bbr_bw_rtts = CYCLE_LEN + 2; /* Window length of min_rtt filter (in sec): */ staticconst u32 bbr_min_rtt_win_sec = 10; /* Minimum time (in ms) spent at bbr_cwnd_min_target in BBR_PROBE_RTT mode: */ staticconst u32 bbr_probe_rtt_mode_ms = 200; /* Skip TSO below the following bandwidth (bits/sec): */ staticconstint bbr_min_tso_rate = 1200000;
/* Pace at ~1% below estimated bw, on average, to reduce queue at bottleneck. *Inordertohelpdrivethenetworktowardlowerqueuesandlowlatencywhile *maintaininghighutilization,theaveragepacingrateaimstobeslightly *lowerthantheestimatedbandwidth.Thisisanimportantaspectofthe *design.
*/ staticconstint bbr_pacing_margin_percent = 1;
/* We use a high_gain value of 2/ln(2) because it's the smallest pacing gain *thatwillallowasmoothlyincreasingpacingratethatwilldoubleeachRTT *andsendthesamenumberofpacketsperRTTthatanun-paced,slow-starting *RenoorCUBICflowwould:
*/ staticconstint bbr_high_gain = BBR_UNIT * 2885 / 1000 + 1; /* The pacing gain of 1/high_gain in BBR_DRAIN is calculated to typically drain *thequeuecreatedinBBR_STARTUPinasingleround:
*/ staticconstint bbr_drain_gain = BBR_UNIT * 1000 / 2885; /* The gain for deriving steady-state cwnd tolerates delayed/stretched ACKs: */ staticconstint bbr_cwnd_gain = BBR_UNIT * 2; /* The pacing_gain values for the PROBE_BW gain cycle, to discover/share bw: */ staticconstint bbr_pacing_gain[] = {
BBR_UNIT * 5 / 4, /* probe for more available bw */
BBR_UNIT * 3 / 4, /* drain queue and/or yield bw to other flows */
BBR_UNIT, BBR_UNIT, BBR_UNIT, /* cruise at 1.0*bw to utilize pipe, */
BBR_UNIT, BBR_UNIT, BBR_UNIT /* without creating excess queue... */
}; /* Randomize the starting gain cycling phase over N phases: */ staticconst u32 bbr_cycle_rand = 7;
/* Try to keep at least this many packets in flight, if things go smoothly. For *smoothfunctioning,aslidingwindowprotocolACKingeveryotherpacket *needsatleast4packetsinflight:
*/ staticconst u32 bbr_cwnd_min_target = 4;
/* To estimate if BBR_STARTUP mode (i.e. high_gain) has filled pipe... */ /* If bw has increased significantly (1.25x), there may be more bw available: */ staticconst u32 bbr_full_bw_thresh = BBR_UNIT * 5 / 4; /* But after 3 rounds w/o significant bw growth, estimate pipe is full: */ staticconst u32 bbr_full_bw_cnt = 3;
/* "long-term" ("LT") bandwidth estimator parameters... */ /* The minimum number of rounds in an LT bw sampling interval: */ staticconst u32 bbr_lt_intvl_min_rtts = 4; /* If lost/delivered ratio > 20%, interval is "lossy" and we may be policed: */ staticconst u32 bbr_lt_loss_thresh = 50; /* If 2 intervals have a bw ratio <= 1/8, their bw is "consistent": */ staticconst u32 bbr_lt_bw_ratio = BBR_UNIT / 8; /* If 2 intervals have a bw diff <= 4 Kbit/sec their bw is "consistent": */ staticconst u32 bbr_lt_bw_diff = 4000 / 8; /* If we estimate we're policed, use lt_bw for this many round trips: */ staticconst u32 bbr_lt_bw_max_rtts = 48;
/* Gain factor for adding extra_acked to target cwnd: */ staticconstint bbr_extra_acked_gain = BBR_UNIT; /* Window length of extra_acked window. */ staticconst u32 bbr_extra_acked_win_rtts = 5; /* Max allowed val for ack_epoch_acked, after which sampling epoch is reset */ staticconst u32 bbr_ack_epoch_acked_reset_thresh = 1U << 20; /* Time period for clamping cwnd increment due to ack aggregation */ staticconst u32 bbr_extra_acked_max_us = 100 * 1000;
/* Convert a BBR bw and gain factor to a pacing rate in bytes per second. */ staticunsignedlong bbr_bw_to_pacing_rate(struct sock *sk, u32 bw, int gain)
{
u64 rate = bw;
/* Save "last known good" cwnd so we can restore it after losses or PROBE_RTT */ staticvoid bbr_save_cwnd(struct sock *sk)
{ struct tcp_sock *tp = tcp_sk(sk); struct bbr *bbr = inet_csk_ca(sk);
if (bbr->prev_ca_state < TCP_CA_Recovery && bbr->mode != BBR_PROBE_RTT)
bbr->prior_cwnd = tcp_snd_cwnd(tp); /* this cwnd is good enough */ else/* loss recovery or BBR_PROBE_RTT have temporarily cut cwnd */
bbr->prior_cwnd = max(bbr->prior_cwnd, tcp_snd_cwnd(tp));
}
if (event == CA_EVENT_TX_START && tp->app_limited) {
bbr->idle_restart = 1;
bbr->ack_epoch_mstamp = tp->tcp_mstamp;
bbr->ack_epoch_acked = 0; /* Avoid pointless buffer overflows: pace at est. bw if we don't *needmorespeed(we'rerestartingfromidleandapp-limited).
*/ if (bbr->mode == BBR_PROBE_BW)
bbr_set_pacing_rate(sk, bbr_bw(sk), BBR_UNIT); elseif (bbr->mode == BBR_PROBE_RTT)
bbr_check_probe_rtt_done(sk);
}
}
/* Calculate bdp based on min RTT and the estimated bottleneck bandwidth: * *bdp=ceil(bw*min_rtt*gain) * *Thekeyfactor,gain,controlstheamountofqueue.Whileasmallgain *buildsasmallerqueue,itbecomesmorevulnerabletonoiseinRTT *measurements(e.g.,delayedACKsorotherACKcompressioneffects).This *noisemaycauseBBRtounder-estimatetherate.
*/ static u32 bbr_bdp(struct sock *sk, u32 bw, int gain)
{ struct bbr *bbr = inet_csk_ca(sk);
u32 bdp;
u64 w;
/* If we've never had a valid RTT sample, cap cwnd at the initial *default.ThisshouldonlyhappenwhentheconnectionisnotusingTCP *timestampsandhasretransmittedalloftheSYN/SYNACK/datapackets *ACKedsofar.Inthiscase,anRTOcancutcwndto1,inwhich *caseweneedtoslow-startuptowardsomethingsafe:TCP_INIT_CWND.
*/ if (unlikely(bbr->min_rtt_us == ~0U)) /* no valid RTT samples yet? */ return TCP_INIT_CWND; /* be safe: cap at default initial cwnd*/
w = (u64)bw * bbr->min_rtt_us;
/* Apply a gain to the given value, remove the BW_SCALE shift, and *roundthevalueuptoavoidanegativefeedbackloop.
*/
bdp = (((w * gain) >> BBR_SCALE) + BW_UNIT - 1) / BW_UNIT;
return bdp;
}
/* To achieve full performance in high-speed paths, we budget enough cwnd to *fitfull-sizedskbsin-flightonbothendhoststofullyutilizethepath: *-oneskbinsendinghostQdisc, *-oneskbinsendinghostTSO/GSOengine *-oneskbbeingreceivedbyreceiverhostLRO/GRO/delayed-ACKengine *Don'tworry,atlowrates(bbr_min_tso_rate)thiswon'tbloatcwndbecause *insuchcasestso_segs_goalis1.Theminimumcwndis4packets, *whichallows2outstanding2-packetsequences,totrytokeeppipe *fullevenwithACK-every-other-packetdelayedACKs.
*/ static u32 bbr_quantization_budget(struct sock *sk, u32 cwnd)
{ struct bbr *bbr = inet_csk_ca(sk);
/* Allow enough full-sized skbs in flight to utilize end systems. */
cwnd += 3 * bbr_tso_segs_goal(sk);
/* Reduce delayed ACKs by rounding up cwnd to the next even number. */
cwnd = (cwnd + 1) & ~1U;
/* Ensure gain cycling gets inflight above BDP even for small BDPs. */ if (bbr->mode == BBR_PROBE_BW && bbr->cycle_idx == 0)
cwnd += 2;
return cwnd;
}
/* Find inflight based on min RTT and the estimated bottleneck bandwidth. */ static u32 bbr_inflight(struct sock *sk, u32 bw, int gain)
{
u32 inflight;
/* An optimization in BBR to reduce losses: On the first round of recovery, we *followthepacketconservationprinciple:sendPpacketsperPpacketsacked. *Afterthat,weslow-startandsendatmost2*PpacketsperPpacketsacked. *Afterrecoveryfinishes,oruponundo,werestorethecwndwehadwhen *recoverystarted(cappedbythetargetcwndbasedonestimatedBDP). * *TODO(ycheng/ncardwell):implementarate-basedapproach.
*/ staticbool bbr_set_cwnd_to_recover_or_restore( struct sock *sk, conststruct rate_sample *rs, u32 acked, u32 *new_cwnd)
{ struct tcp_sock *tp = tcp_sk(sk); struct bbr *bbr = inet_csk_ca(sk);
u8 prev_state = bbr->prev_ca_state, state = inet_csk(sk)->icsk_ca_state;
u32 cwnd = tcp_snd_cwnd(tp);
/* An ACK for P pkts should release at most 2*P packets. We do this *intwosteps.First,herewedeductthenumberoflostpackets. *Then,inbbr_set_cwnd()weslowstartuptowardthetargetcwnd.
*/ if (rs->losses > 0)
cwnd = max_t(s32, cwnd - rs->losses, 1);
if (state == TCP_CA_Recovery && prev_state != TCP_CA_Recovery) { /* Starting 1st round of Recovery, so do packet conservation. */
bbr->packet_conservation = 1;
bbr->next_rtt_delivered = tp->delivered; /* start round now */ /* Cut unused cwnd from app behavior, TSQ, or TSO deferral: */
cwnd = tcp_packets_in_flight(tp) + acked;
} elseif (prev_state >= TCP_CA_Recovery && state < TCP_CA_Recovery) { /* Exiting loss recovery; restore cwnd saved before recovery. */
cwnd = max(cwnd, bbr->prior_cwnd);
bbr->packet_conservation = 0;
}
bbr->prev_ca_state = state;
/* Slow-start up toward target cwnd (if bw estimate is growing, or packet loss *hasdrawnusdownbelowtarget),orsnapdowntotargetifwe'reaboveit.
*/ staticvoid bbr_set_cwnd(struct sock *sk, conststruct rate_sample *rs,
u32 acked, u32 bw, int gain)
{ struct tcp_sock *tp = tcp_sk(sk); struct bbr *bbr = inet_csk_ca(sk);
u32 cwnd = tcp_snd_cwnd(tp), target_cwnd = 0;
if (!acked) goto done; /* no packet fully ACKed; just apply caps */
if (bbr_set_cwnd_to_recover_or_restore(sk, rs, acked, &cwnd)) goto done;
target_cwnd = bbr_bdp(sk, bw, gain);
/* Increment the cwnd to account for excess ACKed data that seems *duetoaggregation(ofdataand/orACKs)visibleintheACKstream.
*/
target_cwnd += bbr_ack_aggregation_cwnd(sk);
target_cwnd = bbr_quantization_budget(sk, target_cwnd);
/* If we're below target cwnd, slow start cwnd toward target cwnd. */ if (bbr_full_bw_reached(sk)) /* only cut cwnd if we filled the pipe */
cwnd = min(cwnd + acked, target_cwnd); elseif (cwnd < target_cwnd || tp->delivered < TCP_INIT_CWND)
cwnd = cwnd + acked;
cwnd = max(cwnd, bbr_cwnd_min_target);
done:
tcp_snd_cwnd_set(tp, min(cwnd, tp->snd_cwnd_clamp)); /* apply global cap */ if (bbr->mode == BBR_PROBE_RTT) /* drain queue, refresh min_rtt */
tcp_snd_cwnd_set(tp, min(tcp_snd_cwnd(tp), bbr_cwnd_min_target));
}
/* End cycle phase if it's time and/or we hit the phase's in-flight target. */ staticbool bbr_is_next_cycle_phase(struct sock *sk, conststruct rate_sample *rs)
{ struct tcp_sock *tp = tcp_sk(sk); struct bbr *bbr = inet_csk_ca(sk); bool is_full_length =
tcp_stamp_us_delta(tp->delivered_mstamp, bbr->cycle_mstamp) >
bbr->min_rtt_us;
u32 inflight, bw;
/* The pacing_gain of 1.0 paces at the estimated bw to try to fully *usethepipewithoutincreasingthequeue.
*/ if (bbr->pacing_gain == BBR_UNIT) return is_full_length; /* just use wall clock time */
/* A pacing_gain > 1.0 probes for bw by trying to raise inflight to at *leastpacing_gain*BDP;thismaytakemorethanmin_rttifmin_rttis *small(e.g.onaLAN).Wedonotpersistifpacketsarelost,since *apathwithsmallbuffersmaynotholdthatmuch.
*/ if (bbr->pacing_gain > BBR_UNIT) return is_full_length &&
(rs->losses || /* perhaps pacing_gain*BDP won't fit */
inflight >= bbr_inflight(sk, bw, bbr->pacing_gain));
/* A pacing_gain < 1.0 tries to drain extra queue we added if bw *probingdidn'tfindmorebw.IfinflightfallstomatchBDPthenwe *estimatequeueisdrained;persistingwouldunderutilizethepipe.
*/ return is_full_length ||
inflight <= bbr_inflight(sk, bw, BBR_UNIT);
}
/* Gain cycling: cycle pacing gain to converge to fair share of available bw. */ staticvoid bbr_update_cycle_phase(struct sock *sk, conststruct rate_sample *rs)
{ struct bbr *bbr = inet_csk_ca(sk);
if (bbr->mode == BBR_PROBE_BW && bbr_is_next_cycle_phase(sk, rs))
bbr_advance_cycle_phase(sk);
}
if (bbr->lt_use_bw) { /* already using long-term rate, lt_bw? */ if (bbr->mode == BBR_PROBE_BW && bbr->round_start &&
++bbr->lt_rtt_cnt >= bbr_lt_bw_max_rtts) {
bbr_reset_lt_bw_sampling(sk); /* stop using lt_bw */
bbr_reset_probe_bw_mode(sk); /* restart gain cycling */
} return;
}
/* Wait for the first loss before sampling, to let the policer exhaust *itstokensandestimatethesteady-staterateallowedbythepolicer. *Startingsamplesearlierincludesburststhatover-estimatethebw.
*/ if (!bbr->lt_is_sampling) { if (!rs->losses) return;
bbr_reset_lt_bw_sampling_interval(sk);
bbr->lt_is_sampling = true;
}
/* To avoid underestimates, reset sampling if we run out of data. */ if (rs->is_app_limited) {
bbr_reset_lt_bw_sampling(sk); return;
}
if (bbr->round_start)
bbr->lt_rtt_cnt++; /* count round trips in this interval */ if (bbr->lt_rtt_cnt < bbr_lt_intvl_min_rtts) return; /* sampling interval needs to be longer */ if (bbr->lt_rtt_cnt > 4 * bbr_lt_intvl_min_rtts) {
bbr_reset_lt_bw_sampling(sk); /* interval is too long */ return;
}
/* End sampling interval when a packet is lost, so we estimate the *policertokenswereexhausted.Stoppingthesamplingbeforethe *tokensareexhaustedunder-estimatesthepolicedrate.
*/ if (!rs->losses) return;
/* Calculate packets lost and delivered in sampling interval. */
lost = tp->lost - bbr->lt_last_lost;
delivered = tp->delivered - bbr->lt_last_delivered; /* Is loss rate (lost/delivered) >= lt_loss_thresh? If not, wait. */ if (!delivered || (lost << BBR_SCALE) < bbr_lt_loss_thresh * delivered) return;
/* Find average delivery rate in this sampling interval. */
t = div_u64(tp->delivered_mstamp, USEC_PER_MSEC) - bbr->lt_last_stamp; if ((s32)t < 1) return; /* interval is less than one ms, so wait */ /* Check if can multiply without overflow */ if (t >= ~0U / USEC_PER_MSEC) {
bbr_reset_lt_bw_sampling(sk); /* interval too long; reset */ return;
}
t *= USEC_PER_MSEC;
bw = (u64)delivered * BW_UNIT;
do_div(bw, t);
bbr_lt_bw_interval_done(sk, bw);
}
/* Estimate the bandwidth based on how fast packets are delivered */ staticvoid bbr_update_bw(struct sock *sk, conststruct rate_sample *rs)
{ struct tcp_sock *tp = tcp_sk(sk); struct bbr *bbr = inet_csk_ca(sk);
u64 bw;
bbr->round_start = 0; if (rs->delivered < 0 || rs->interval_us <= 0) return; /* Not a valid observation */
/* See if we've reached the next RTT */ if (!before(rs->prior_delivered, bbr->next_rtt_delivered)) {
bbr->next_rtt_delivered = tp->delivered;
bbr->rtt_cnt++;
bbr->round_start = 1;
bbr->packet_conservation = 0;
}
bbr_lt_bw_sampling(sk, rs);
/* Divide delivered by the interval to find a (lower bound) bottleneck *bandwidthsample.Deliveredisinpacketsandinterval_usinuSand *ratiowillbe<<1formostconnections.Sodeliveredisfirstscaled.
*/
bw = div64_long((u64)rs->delivered * BW_UNIT, rs->interval_us);
/* If this sample is application-limited, it is likely to have a very *lowdeliveredcountthatrepresentsapplicationbehaviorratherthan *theavailablenetworkrate.Suchasamplecoulddragdownestimated *bw,causingneedlessslow-down.Thus,tocontinuetosendatthe *lastmeasurednetworkrate,wefilteroutapp-limitedsamplesunless *theydescribethepathbwatleastaswellasourbwmodel. * *Sothegoalduringapp-limitedphaseistoproceedwiththebest *networkratenomatterhowlong.Weautomaticallyleavethis *phasewhenappwritesfasterthanthenetworkcandeliver:)
*/ if (!rs->is_app_limited || bw >= bbr_max_bw(sk)) { /* Incorporate new sample into our max bw filter. */
minmax_running_max(&bbr->bw, bbr_bw_rtts, bbr->rtt_cnt, bw);
}
}
/* Compute how many packets we expected to be delivered over epoch. */
epoch_us = tcp_stamp_us_delta(tp->delivered_mstamp,
bbr->ack_epoch_mstamp);
expected_acked = ((u64)bbr_bw(sk) * epoch_us) / BW_UNIT;
/* Reset the aggregation epoch if ACK rate is below expected rate or *significantlylargeno.ofackreceivedsinceepoch(potentially *quiteoldepoch).
*/ if (bbr->ack_epoch_acked <= expected_acked ||
(bbr->ack_epoch_acked + rs->acked_sacked >=
bbr_ack_epoch_acked_reset_thresh)) {
bbr->ack_epoch_acked = 0;
bbr->ack_epoch_mstamp = tp->delivered_mstamp;
expected_acked = 0;
}
/* Compute excess data delivered, beyond what was expected. */
bbr->ack_epoch_acked = min_t(u32, 0xFFFFF,
bbr->ack_epoch_acked + rs->acked_sacked);
extra_acked = bbr->ack_epoch_acked - expected_acked;
extra_acked = min(extra_acked, tcp_snd_cwnd(tp)); if (extra_acked > bbr->extra_acked[bbr->extra_acked_win_idx])
bbr->extra_acked[bbr->extra_acked_win_idx] = extra_acked;
}
/* Estimate when the pipe is full, using the change in delivery rate: BBR *estimatesthatSTARTUPfilledthepipeiftheestimatedbwhasn'tchangedby *atleastbbr_full_bw_thresh(25%)afterbbr_full_bw_cnt(3)non-app-limited *rounds.Why3rounds:1:rwinautotuninggrowstherwin,2:wefillthe *higherrwin,3:wegethigherdeliveryratesamples.Ortransient *cross-trafficorradionoisecangoaway.CUBICHystartsharesasimilar *designgoal,butusesdelayandinter-ACKspacinginsteadofbandwidth.
*/ staticvoid bbr_check_full_bw_reached(struct sock *sk, conststruct rate_sample *rs)
{ struct bbr *bbr = inet_csk_ca(sk);
u32 bw_thresh;
if (bbr_full_bw_reached(sk) || !bbr->round_start || rs->is_app_limited) return;
/* If pipe is probably full, drain the queue and then enter steady-state. */ staticvoid bbr_check_drain(struct sock *sk, conststruct rate_sample *rs)
{ struct bbr *bbr = inet_csk_ca(sk);
if (bbr->mode == BBR_STARTUP && bbr_full_bw_reached(sk)) {
bbr->mode = BBR_DRAIN; /* drain queue we created */
tcp_sk(sk)->snd_ssthresh =
bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT);
} /* fall through to check if in-flight is already small: */ if (bbr->mode == BBR_DRAIN &&
bbr_packets_in_net_at_edt(sk, tcp_packets_in_flight(tcp_sk(sk))) <=
bbr_inflight(sk, bbr_max_bw(sk), BBR_UNIT))
bbr_reset_probe_bw_mode(sk); /* we estimate queue is drained */
}
if (!(bbr->probe_rtt_done_stamp &&
after(tcp_jiffies32, bbr->probe_rtt_done_stamp))) return;
bbr->min_rtt_stamp = tcp_jiffies32; /* wait a while until PROBE_RTT */
tcp_snd_cwnd_set(tp, max(tcp_snd_cwnd(tp), bbr->prior_cwnd));
bbr_reset_mode(sk);
}
/* The goal of PROBE_RTT mode is to have BBR flows cooperatively and *periodicallydrainthebottleneckqueue,toconvergetomeasurethetrue *min_rtt(unloadedpropagationdelay).Thisallowstheflowstokeepqueues *small(reducingqueuingdelayandpacketloss)andachievefairnessamong *BBRflows. * *Themin_rttfilterwindowis10seconds.Whenthemin_rttestimateexpires, *weenterPROBE_RTTmodeandcapthecwndatbbr_cwnd_min_target=4packets. *Afteratleastbbr_probe_rtt_mode_ms=200msandatleastonepacket-timed *roundtripelapsedwiththatflightsize<=4,weleavePROBE_RTTmodeand *re-enterthepreviousmode.BBRuses200mstoapproximatelyboundthe *performancepenaltyofPROBE_RTT'scwndcappingtoroughly2%(200ms/10s). * *Notethatflowsneedonlypay2%iftheyarebusysendingoverthelast10 *seconds.Interactiveapplications(e.g.,Web,RPCs,videochunks)oftenhave *naturalsilencesorlow-rateperiodswithin10secondswheretherateislow *enoughforlongenoughtodrainitsqueueinthebottleneck.Wepickup *theseminRTTmeasurementsopportunisticallywithourmin_rttfilter.:-)
*/ staticvoid bbr_update_min_rtt(struct sock *sk, conststruct rate_sample *rs)
{ struct tcp_sock *tp = tcp_sk(sk); struct bbr *bbr = inet_csk_ca(sk); bool filter_expired;
/* Track min RTT seen in the min_rtt_win_sec filter window: */
filter_expired = after(tcp_jiffies32,
bbr->min_rtt_stamp + bbr_min_rtt_win_sec * HZ); if (rs->rtt_us >= 0 &&
(rs->rtt_us < bbr->min_rtt_us ||
(filter_expired && !rs->is_ack_delayed))) {
bbr->min_rtt_us = rs->rtt_us;
bbr->min_rtt_stamp = tcp_jiffies32;
}
if (bbr_probe_rtt_mode_ms > 0 && filter_expired &&
!bbr->idle_restart && bbr->mode != BBR_PROBE_RTT) {
bbr->mode = BBR_PROBE_RTT; /* dip, drain queue */
bbr_save_cwnd(sk); /* note cwnd so we can restore it */
bbr->probe_rtt_done_stamp = 0;
}
if (bbr->mode == BBR_PROBE_RTT) { /* Ignore low rate samples during this mode. */
tp->app_limited =
(tp->delivered + tcp_packets_in_flight(tp)) ? : 1; /* Maintain min packets in flight for max(200 ms, 1 round). */ if (!bbr->probe_rtt_done_stamp &&
tcp_packets_in_flight(tp) <= bbr_cwnd_min_target) {
bbr->probe_rtt_done_stamp = tcp_jiffies32 +
msecs_to_jiffies(bbr_probe_rtt_mode_ms);
bbr->probe_rtt_round_done = 0;
bbr->next_rtt_delivered = tp->delivered;
} elseif (bbr->probe_rtt_done_stamp) { if (bbr->round_start)
bbr->probe_rtt_round_done = 1; if (bbr->probe_rtt_round_done)
bbr_check_probe_rtt_done(sk);
}
} /* Restart after idle ends only once we process a new S/ACK for data */ if (rs->delivered > 0)
bbr->idle_restart = 0;
}
__bpf_kfunc static u32 bbr_sndbuf_expand(struct sock *sk)
{ /* Provision 3 * cwnd since BBR may slow-start even during recovery. */ return3;
}
/* In theory BBR does not need to undo the cwnd since it does not *alwaysreducecwndonlosses(seebbr_main()).Keepitfornow.
*/
__bpf_kfunc static u32 bbr_undo_cwnd(struct sock *sk)
{ struct bbr *bbr = inet_csk_ca(sk);
/* Entering loss recovery, so save cwnd for when we exit or undo recovery. */
__bpf_kfunc static u32 bbr_ssthresh(struct sock *sk)
{
bbr_save_cwnd(sk); return tcp_sk(sk)->snd_ssthresh;
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.