/* Returns end sequence number of the receiver's advertised window */ static u64 mptcp_wnd_end(conststruct mptcp_sock *msk)
{ return READ_ONCE(msk->wnd_end);
}
/* This is the first subflow, always with id 0 */
WRITE_ONCE(subflow->local_id, 0);
mptcp_sock_graft(msk->first, sk->sk_socket);
iput(SOCK_INODE(ssock));
return0;
}
/* If the MPC handshake is not started, returns the first subflow, *eventuallyallocatingit.
*/ struct sock *__mptcp_nmpc_sk(struct mptcp_sock *msk)
{ struct sock *sk = (struct sock *)msk; int ret;
if (!((1 << sk->sk_state) & (TCPF_CLOSE | TCPF_LISTEN))) return ERR_PTR(-EINVAL);
if (!msk->first) {
ret = __mptcp_socket_create(msk); if (ret) return ERR_PTR(ret);
}
pr_debug("colesced seq %llx into %llx new len %d new end seq %llx\n",
MPTCP_SKB_CB(from)->map_seq, MPTCP_SKB_CB(to)->map_seq,
to->len, MPTCP_SKB_CB(from)->end_seq);
MPTCP_SKB_CB(to)->end_seq = MPTCP_SKB_CB(from)->end_seq;
/* note the fwd memory can reach a negative value after accounting *forthedelta,butthelaterskbfreewillrestoreanon *negativeone
*/
atomic_add(delta, &sk->sk_rmem_alloc);
sk_mem_charge(sk, delta);
kfree_skb_partial(from, fragstolen);
pr_debug("msk=%p seq=%llx limit=%llx empty=%d\n", msk, seq, max_seq,
RB_EMPTY_ROOT(&msk->out_of_order_queue)); if (after64(end_seq, max_seq)) { /* out of window */
mptcp_drop(sk, skb);
pr_debug("oow by %lld, rcv_wnd_sent %llu\n",
(unsignedlonglong)end_seq - (unsignedlong)max_seq,
(unsignedlonglong)atomic64_read(&msk->rcv_wnd_sent));
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_NODSSWINDOW); return;
}
p = &msk->out_of_order_queue.rb_node;
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_OFOQUEUE); if (RB_EMPTY_ROOT(&msk->out_of_order_queue)) {
rb_link_node(&skb->rbnode, NULL, p);
rb_insert_color(&skb->rbnode, &msk->out_of_order_queue);
msk->ooo_last_skb = skb; goto end;
}
/* with 2 subflows, adding at end of ooo queue is quite likely *Useofooo_last_skbavoidstheO(Log(N))rbtreelookup.
*/ if (mptcp_ooo_try_coalesce(msk, msk->ooo_last_skb, skb)) {
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_OFOMERGE);
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_OFOQUEUETAIL); return;
}
/* Can avoid an rbtree lookup if we are adding skb after ooo_last_skb */ if (!before64(seq, MPTCP_SKB_CB(msk->ooo_last_skb)->end_seq)) {
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_OFOQUEUETAIL);
parent = &msk->ooo_last_skb->rbnode;
p = &parent->rb_right; goto insert;
}
/* Find place to insert this segment. Handle overlaps on the way. */
parent = NULL; while (*p) {
parent = *p;
skb1 = rb_to_skb(parent); if (before64(seq, MPTCP_SKB_CB(skb1)->map_seq)) {
p = &parent->rb_left; continue;
} if (before64(seq, MPTCP_SKB_CB(skb1)->end_seq)) { if (!after64(end_seq, MPTCP_SKB_CB(skb1)->end_seq)) { /* All the bits are present. Drop. */
mptcp_drop(sk, skb);
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_DUPDATA); return;
} if (after64(seq, MPTCP_SKB_CB(skb1)->map_seq)) { /* partial overlap: *|skb| *|skb1| *continuetraversing
*/
} else { /* skb's seq == skb1's seq and skb covers skb1. *Replaceskb1withskb.
*/
rb_replace_node(&skb1->rbnode, &skb->rbnode,
&msk->out_of_order_queue);
mptcp_drop(sk, skb1);
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_DUPDATA); goto merge_right;
}
} elseif (mptcp_ooo_try_coalesce(msk, skb1, skb)) {
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_OFOMERGE); return;
}
p = &parent->rb_right;
}
merge_right: /* Remove other segments covered by skb. */ while ((skb1 = skb_rb_next(skb)) != NULL) { if (before64(end_seq, MPTCP_SKB_CB(skb1)->end_seq)) break;
rb_erase(&skb1->rbnode, &msk->out_of_order_queue);
mptcp_drop(sk, skb1);
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_DUPDATA);
} /* If there is no skb after us, we are the last_skb ! */ if (!skb1)
msk->ooo_last_skb = skb;
/* old data, keep it simple and drop the whole pkt, sender *willretransmitasneeded,ifneeded.
*/
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_DUPDATA);
drop:
mptcp_drop(sk, skb); returnfalse;
}
/* Need to ack a DATA_FIN received from a peer while this side *oftheconnectionisinESTABLISHED,FIN_WAIT1,orFIN_WAIT2. *msk->rcv_data_finwassetwhenparsingtheincomingoptions *atthesubflowlevelandthemsklockwasnotheld,sothis *isthefirstopportunitytoactontheDATA_FINandchange *themskstate. * *Ifwearecaughtuptothesequencenumberoftheincoming *DATA_FIN,sendtheDATA_ACKnowanddostatetransition.If *notcaughtup,donothingandlettherecvcodesendDATA_ACK *whencatchingup.
*/
if (mptcp_pending_data_fin(sk, &rcv_data_fin_seq)) {
WRITE_ONCE(msk->ack_seq, msk->ack_seq + 1);
WRITE_ONCE(msk->rcv_data_fin, 0);
WRITE_ONCE(sk->sk_shutdown, sk->sk_shutdown | RCV_SHUTDOWN);
smp_mb__before_atomic(); /* SHUTDOWN must be visible first */
switch (sk->sk_state) { case TCP_ESTABLISHED:
mptcp_set_state(sk, TCP_CLOSE_WAIT); break; case TCP_FIN_WAIT1:
mptcp_set_state(sk, TCP_CLOSING); break; case TCP_FIN_WAIT2:
mptcp_shutdown_subflows(msk);
mptcp_set_state(sk, TCP_CLOSE); break; default: /* Other states not expected */
WARN_ON_ONCE(1); break;
}
ret = true; if (!__mptcp_check_fallback(msk))
mptcp_send_ack(msk);
mptcp_close_wake_up(sk);
} return ret;
}
/* try to move as much data as available */
map_remaining = subflow->map_data_len -
mptcp_subflow_get_map_offset(subflow);
skb = skb_peek(&ssk->sk_receive_queue); if (unlikely(!skb)) break;
if (__mptcp_check_fallback(msk)) { /* Under fallback skbs have no MPTCP extension and TCP could *collapsethembetweenthedummymapcreationandthe *currentdequeue.Besuretoadjustthemapsize.
*/
map_remaining = skb->len;
subflow->map_data_len = skb->len;
}
offset = seq - TCP_SKB_CB(skb)->seq;
fin = TCP_SKB_CB(skb)->tcp_flags & TCPHDR_FIN; if (fin)
seq++;
if (offset < skb->len) {
size_t len = skb->len - offset;
staticbool __mptcp_subflow_error_report(struct sock *sk, struct sock *ssk)
{ int err = sock_error(ssk); int ssk_state;
if (!err) returnfalse;
/* only propagate errors on fallen-back sockets or *onMPCconnect
*/ if (sk->sk_state != TCP_SYN_SENT && !__mptcp_check_fallback(mptcp_sk(sk))) returnfalse;
/* We need to propagate only transition to CLOSE state. *Orphanedsocketwillseesuchstatechangevia *subflow_sched_work_if_closed()andthatpathwillproperly *destroythemskasneeded.
*/
ssk_state = inet_sk_state_load(ssk); if (ssk_state == TCP_CLOSE && !sock_flag(sk, SOCK_DEAD))
mptcp_set_state(sk, ssk_state);
WRITE_ONCE(sk->sk_err, -err);
/* This barrier is coupled with smp_rmb() in mptcp_poll() */
smp_wmb();
sk_error_report(sk); returntrue;
}
mptcp_for_each_subflow(msk, subflow) if (__mptcp_subflow_error_report(sk, mptcp_subflow_tcp_sock(subflow))) break;
}
/* In most cases we will be able to lock the mptcp socket. If its already *owned,weneedtodefertotheworkqueuetoavoidABBAdeadlock.
*/ staticbool move_skbs_to_msk(struct mptcp_sock *msk, struct sock *ssk)
{ struct sock *sk = (struct sock *)msk; bool moved;
moved = __mptcp_move_skbs_from_subflow(msk, ssk);
__mptcp_ofo_queue(msk); if (unlikely(ssk->sk_err)) { if (!sock_owned_by_user(sk))
__mptcp_error_report(sk); else
__set_bit(MPTCP_ERROR_REPORT, &msk->cb_flags);
}
/* If the moves have caught up with the DATA_FIN sequence number *it'stimetoacktheDATA_FINandchangesocketstate,but *thisisnotagoodplacetochangestate.Lettheworkqueue *doit.
*/ if (mptcp_pending_data_fin(sk, NULL))
mptcp_schedule_work(sk); return moved;
}
/* The peer can send data while we are shutting down this *subflowatmskdestructiontime,butwemustavoidenqueuing *moredatatothemskreceivequeue
*/ if (unlikely(subflow->disposable)) return;
mptcp_data_lock(sk); if (!sock_owned_by_user(sk))
__mptcp_data_ready(sk, ssk); else
__set_bit(MPTCP_DEQUEUE, &mptcp_sk(sk)->cb_flags);
mptcp_data_unlock(sk);
}
spin_lock_bh(&msk->fallback_lock); if (!msk->allow_subflows) {
spin_unlock_bh(&msk->fallback_lock); returnfalse;
}
mptcp_subflow_joined(msk, ssk);
spin_unlock_bh(&msk->fallback_lock);
/* attach to msk socket only after we are sure we will deal with it *atclosetime
*/ if (sk->sk_socket && !ssk->sk_socket)
mptcp_sock_graft(ssk, sk->sk_socket);
/* prevent rescheduling on close */ if (unlikely(inet_sk_state_load(sk) == TCP_CLOSE)) return;
tout = mptcp_sk(sk)->timer_ival;
sk_reset_timer(sk, &icsk->icsk_retransmit_timer, jiffies + tout);
}
bool mptcp_schedule_work(struct sock *sk)
{ if (inet_sk_state_load(sk) != TCP_CLOSE &&
schedule_work(&mptcp_sk(sk)->work)) { /* each subflow already holds a reference to the sk, and the *workqueueisinvokedbyasubflow,soskcan'tgoawayhere.
*/
sock_hold(sk); returntrue;
} returnfalse;
}
/* can collapse only if MPTCP level sequence is in order and this *mappinghasnotbeenxmittedyet
*/ return mpext && mpext->data_seq + mpext->data_len == write_seq &&
!mpext->frozen;
}
/* we can append data to the given data frag if: *-thereisspaceavailableinthebackingpage_frag *-thedatafragtailmatchesthecurrentpage_fragfreeoffset *-thedatafragendsequencenumbermatchesthecurrentwriteseq
*/ staticbool mptcp_frag_can_collapse_to(conststruct mptcp_sock *msk, conststruct page_frag *pfrag, conststruct mptcp_data_frag *df)
{ return df && pfrag->page == df->page &&
pfrag->size - pfrag->offset > 0 &&
pfrag->offset == (df->offset + df->data_len) &&
df->data_seq + df->data_len == msk->write_seq;
}
/* called under both the msk socket lock and the data lock */ staticvoid __mptcp_clean_una(struct sock *sk)
{ struct mptcp_sock *msk = mptcp_sk(sk); struct mptcp_data_frag *dtmp, *dfrag;
u64 snd_una;
if (first)
tcp_enter_memory_pressure(ssk);
sk_stream_moderate_sndbuf(ssk);
first = false;
}
__mptcp_sync_sndbuf(sk);
}
/* ensure we get enough memory for the frag hdr, beyond some minimal amount of *data
*/ staticbool mptcp_page_frag_refill(struct sock *sk, struct page_frag *pfrag)
{ if (likely(skb_page_frag_refill(32U + sizeof(struct mptcp_data_frag),
pfrag, sk->sk_allocation))) returntrue;
mptcp_enter_memory_pressure(sk); returnfalse;
}
staticstruct mptcp_data_frag *
mptcp_carve_data_frag(conststruct mptcp_sock *msk, struct page_frag *pfrag, int orig_offset)
{ int offset = ALIGN(orig_offset, sizeof(long)); struct mptcp_data_frag *dfrag;
i = skb_shinfo(skb)->nr_frags;
reuse_skb = false;
mpext = mptcp_get_ext(skb);
}
/* Zero window and all data acked? Probe. */
copy = mptcp_check_allowed_size(msk, ssk, data_seq, copy); if (copy == 0) {
u64 snd_una = READ_ONCE(msk->snd_una);
/* No need for zero probe if there are any data pending *eitheratthemskorssklevel;skbisthecurrentwrite *queuetailandcanbeemptyatthispoint.
*/ if (snd_una != msk->snd_nxt || skb->len ||
skb != tcp_send_head(ssk)) {
tcp_remove_empty_skb(ssk); return0;
}
trace_mptcp_subflow_get_send(subflow);
ssk = mptcp_subflow_tcp_sock(subflow); if (!mptcp_subflow_active(subflow)) continue;
tout = max(tout, mptcp_timeout_from_subflow(subflow));
nr_active += !backup;
pace = subflow->avg_pacing_rate; if (unlikely(!pace)) { /* init pacing rate from socket */
subflow->avg_pacing_rate = READ_ONCE(ssk->sk_pacing_rate);
pace = subflow->avg_pacing_rate; if (!pace) continue;
}
/* pick the best backup if no other subflow is active */ if (!nr_active)
send_info[SSK_MODE_ACTIVE].ssk = send_info[SSK_MODE_BACKUP].ssk;
/* According to the blest algorithm, to avoid HoL blocking for the *fasterflow,weneedto: *-estimatethefasterflowlingertime *-usetheabovetoestimatetheamountofbytetransferred *bythefasterflow *-checkthattheamountofqueueddataisgreaterthantheabove, *otherwisedonotusethepicked,slower,subflow *Weselectthesubflowwiththeshorterestimatedtimetoflush *thequeuedmem,whichbasicallyensuretheabove.Wejustneed *tocheckthatsubflowhasanonemptycwin.
*/
ssk = send_info[SSK_MODE_ACTIVE].ssk; if (!ssk || !sk_stream_memory_free(ssk)) return NULL;
while ((dfrag = mptcp_send_head(sk))) {
info->sent = dfrag->already_sent;
info->limit = dfrag->data_len;
len = dfrag->data_len - dfrag->already_sent; while (len > 0) { int ret = 0;
ret = mptcp_sendmsg_frag(sk, ssk, dfrag, info); if (ret <= 0) {
err = copied ? : ret; goto out;
}
while (mptcp_send_head(sk) && (push_count > 0)) { struct mptcp_subflow_context *subflow; int ret = 0;
if (mptcp_sched_get_send(msk)) break;
push_count = 0;
mptcp_for_each_subflow(msk, subflow) { if (READ_ONCE(subflow->scheduled)) {
mptcp_subflow_set_scheduled(subflow, false);
prev_ssk = ssk;
ssk = mptcp_subflow_tcp_sock(subflow); if (ssk != prev_ssk) { /* First check. If the ssk has changed since *thelastround,releaseprev_ssk
*/ if (prev_ssk)
mptcp_push_release(prev_ssk, &info);
/* Need to lock the new subflow only if different *fromthepreviousone,otherwisewearestill *heldingtherelevantlock
*/
lock_sock(ssk);
}
/* at this point we held the socket lock for the last subflow we used */ if (ssk)
mptcp_push_release(ssk, &info);
/* ensure the rtx timer is running */ if (!mptcp_rtx_timer_pending(sk))
mptcp_reset_rtx_timer(sk); if (do_check_data_fin)
mptcp_check_send_data_fin(sk);
}
info.flags = 0; while (mptcp_send_head(sk) && keep_pushing) { struct mptcp_subflow_context *subflow = mptcp_subflow_ctx(ssk); int ret = 0;
/* check for a different subflow usage only after *spoolingthefirstchunkofdata
*/ if (first) {
mptcp_subflow_set_scheduled(subflow, false);
ret = __subflow_push_pending(sk, ssk, &info);
first = false; if (ret <= 0) break;
copied += ret; continue;
}
if (mptcp_sched_get_send(msk)) goto out;
if (READ_ONCE(subflow->scheduled)) {
mptcp_subflow_set_scheduled(subflow, false);
ret = __subflow_push_pending(sk, ssk, &info); if (ret <= 0)
keep_pushing = false;
copied += ret;
}
out: /* __mptcp_alloc_tx_skb could have released some wmem and we are *notgoingtoflushitviarelease_sock()
*/ if (copied) {
tcp_push(ssk, 0, info.mss_now, tcp_sk(ssk)->nonagle,
info.size_goal); if (!mptcp_rtx_timer_pending(sk))
mptcp_reset_rtx_timer(sk);
/* on flags based fastopen the mptcp is supposed to create the *firstsubflowrightnow.Otherwiseweareinthedefer_connect *path,andthefirstsubflowmustbealreadypresent. *Sincethedefer_connectflagisclearedafterthefirstsuccsful *fastopenattempt,noneedtocheckforadditionalsubflowstatus.
*/ if (msg->msg_flags & MSG_FASTOPEN) {
ssk = __mptcp_nmpc_sk(msk); if (IS_ERR(ssk)) return PTR_ERR(ssk);
} if (!msk->first) return -EINVAL;
/* do the blocking bits of inet_stream_connect outside the ssk socket lock */ if (ret == -EINPROGRESS && !(msg->msg_flags & MSG_DONTWAIT)) {
ret = __inet_stream_connect(sk->sk_socket, msg->msg_name,
msg->msg_namelen, msg->msg_flags, 1);
/* Keep the same behaviour of plain TCP: zero the copied bytes in *caseofanyerror,excepttimeoutorsignal
*/ if (ret && ret != -EINPROGRESS && ret != -ERESTARTSYS && ret != -EINTR)
*copied_syn = 0;
} elseif (ret && ret != -EINPROGRESS) { /* The disconnect() op called by tcp_sendmsg_fastopen()/ *__inet_stream_connect()canfail,duetolookingcheck, *seemptcp_disconnect(). *Attemptitagainoutsidetheproblematicscope.
*/ if (!mptcp_disconnect(sk, 0)) {
sk->sk_disconnects++;
sk->sk_socket->state = SS_UNCONNECTED;
}
}
inet_clear_bit(DEFER_CONNECT, sk);
if ((1 << sk->sk_state) & ~(TCPF_ESTABLISHED | TCPF_CLOSE_WAIT)) {
ret = sk_stream_wait_connect(sk, &timeo); if (ret) goto do_error;
}
ret = -EPIPE; if (unlikely(sk->sk_err || (sk->sk_shutdown & SEND_SHUTDOWN))) goto do_error;
pfrag = sk_page_frag(sk);
while (msg_data_left(msg)) { int total_ts, frag_truesize = 0; struct mptcp_data_frag *dfrag; bool dfrag_collapsed;
size_t psize, offset;
u32 copy_limit;
/* ensure fitting the notsent_lowat() constraint */
copy_limit = mptcp_send_limit(sk); if (!copy_limit) goto wait_for_memory;
/* reuse tail pfrag, if possible, or carve a new one from the *pageallocator
*/
dfrag = mptcp_pending_tail(sk);
dfrag_collapsed = mptcp_frag_can_collapse_to(msk, pfrag, dfrag); if (!dfrag_collapsed) { if (!mptcp_page_frag_refill(sk, pfrag)) goto wait_for_memory;
/* Make subflows follow along. If we do not do this, we *getdropsatsubflowlevelifskbscan'tbemovedto *themptcprxqueuefastenough(announcedrcv_wincan *exceedssk->sk_rcvbuf).
*/
mptcp_for_each_subflow(msk, subflow) { struct sock *ssk; bool slow;
/* verify we can move any data from the subflow, eventually updating */ if (!(sk->sk_userlocks & SOCK_RCVBUF_LOCK))
mptcp_for_each_subflow(msk, subflow)
__mptcp_rcvbuf_update(sk, subflow->tcp_sock);
if (unlikely(msk->recvmsg_inq))
cmsg_flags = MPTCP_CMSG_INQ;
while (copied < len) { int err, bytes_read;
bytes_read = __mptcp_recvmsg_mskq(sk, msg, len - copied, flags,
copied, &tss, &cmsg_flags); if (unlikely(bytes_read < 0)) { if (!copied)
copied = bytes_read; goto out_err;
}
copied += bytes_read;
if (skb_queue_empty(&sk->sk_receive_queue) && __mptcp_move_skbs(sk)) continue;
/* only the MPTCP socket status is relevant here. The exit *conditionsmirrorcloselytcp_recvmsg()
*/ if (copied >= target) break;
if (copied) { if (sk->sk_err ||
sk->sk_state == TCP_CLOSE ||
(sk->sk_shutdown & RCV_SHUTDOWN) ||
!timeo ||
signal_pending(current)) break;
} else { if (sk->sk_err) {
copied = sock_error(sk); break;
}
if (sk->sk_shutdown & RCV_SHUTDOWN) { /* race breaker: the shutdown could be after the *previousreceivequeuecheck
*/ if (__mptcp_move_skbs(sk)) continue; break;
}
if (sk->sk_state == TCP_CLOSE) {
copied = -ENOTCONN; break;
}
if (!timeo) {
copied = -EAGAIN; break;
}
if (signal_pending(current)) {
copied = sock_intr_errno(timeo); break;
}
}
bh_lock_sock(sk); if (!sock_owned_by_user(sk)) { /* we need a process context to retransmit */ if (!test_and_set_bit(MPTCP_WORK_RTX, &msk->flags))
mptcp_schedule_work(sk);
} else { /* delegate our work to tcp_release_cb() */
__set_bit(MPTCP_RETRANSMIT, &msk->cb_flags);
}
bh_unlock_sock(sk);
sock_put(sk);
}
/* still data outstanding at TCP level? skip this */ if (!tcp_rtx_and_write_queues_empty(ssk)) {
mptcp_pm_subflow_chk_stale(msk, ssk);
min_stale_count = min_t(int, min_stale_count, subflow->stale_count); continue;
}
if (subflow->backup || subflow->request_bkup) { if (!backup)
backup = ssk; continue;
}
if (!pick)
pick = ssk;
}
if (pick) return pick;
/* use backup only if there are no progresses anywhere */ return min_stale_count > 1 ? backup : NULL;
}
/* the closing socket has some data untransmitted and/or unacked: *somedatainthemptcprtxqueuehasnotreallyxmittedyet. *keepitsimpleandre-injectthewholemptcplevelrtxqueue
*/
mptcp_data_lock(sk);
__mptcp_clean_una_wakeup(sk);
rtx_head = mptcp_rtx_head(sk); if (!rtx_head) {
mptcp_data_unlock(sk); returnfalse;
}
/* be sure to clear the "sent status" on all re-injected fragments */
list_for_each_entry(cur, &msk->rtx_queue, list) { if (!cur->already_sent) break;
cur->already_sent = 0;
}
/* be sure to send a reset only if the caller asked for it, also *cleancompletelythesubflowstatuswhenthesubflowreaches *TCP_CLOSEstate
*/ staticvoid __mptcp_subflow_disconnect(struct sock *ssk, struct mptcp_subflow_context *subflow, unsignedint flags)
{ if (((1 << ssk->sk_state) & (TCPF_CLOSE | TCPF_LISTEN)) ||
(flags & MPTCP_CF_FASTCLOSE)) { /* The MPTCP code never wait on the subflow sockets, TCP-level *disconnectshouldneverfail
*/
WARN_ON_ONCE(tcp_disconnect(ssk, 0));
mptcp_subflow_ctx_reset(subflow);
} else {
tcp_shutdown(ssk, SEND_SHUTDOWN);
}
}
/* subflow sockets can be either outgoing (connect) or incoming *(accept). * *Outgoingsubflowsusein-kernelsockets. *Incomingsubflowsdonothavetheirown'structsocket'allocated, *soweneedtousetcp_close()afterdetachingthemfromthemptcp *parentsocket.
*/ staticvoid __mptcp_close_ssk(struct sock *sk, struct sock *ssk, struct mptcp_subflow_context *subflow, unsignedint flags)
{ struct mptcp_sock *msk = mptcp_sk(sk); bool dispose_it, need_push = false;
/* If the first subflow moved to a close state before accept, e.g. due *toanincomingresetorlistenershutdown,thesubflowsocketis *alreadydeletedbyinet_child_forget()andthemptcpsocketcan't *survivetoo.
*/ if (msk->in_accept_queue && msk->first == ssk &&
(sock_flag(sk, SOCK_DEAD) || sock_flag(ssk, SOCK_DEAD))) { /* ensure later check in mptcp_worker() will dispose the msk */
sock_set_flag(sk, SOCK_DEAD);
mptcp_set_close_tout(sk, tcp_jiffies32 - (mptcp_close_timeout(sk) + 1));
lock_sock_nested(ssk, SINGLE_DEPTH_NESTING);
mptcp_subflow_drop_ctx(ssk); goto out_release;
}
dispose_it = msk->free_first || ssk != msk->first; if (dispose_it)
list_del(&subflow->node);
lock_sock_nested(ssk, SINGLE_DEPTH_NESTING);
if ((flags & MPTCP_CF_FASTCLOSE) && !__mptcp_check_fallback(msk)) { /* be sure to force the tcp_close path *togeneratetheegressreset
*/
ssk->sk_lingertime = 0;
sock_set_flag(ssk, SOCK_LINGER);
subflow->send_fastclose = 1;
}
/* if ssk hit tcp_done(), tcp_cleanup_ulp() cleared the related ops *thesskhasbeenalreadydestroyed,wejustneedtoreleasethe *referenceownedbymsk;
*/ if (!inet_csk(ssk)->icsk_ulp_ops) {
WARN_ON_ONCE(!sock_flag(ssk, SOCK_DEAD));
kfree_rcu(subflow, rcu);
} else { /* otherwise tcp will dispose of the ssk and subflow ctx */
__tcp_close(ssk, 0);
/* close acquired an extra ref */
__sock_put(ssk);
}
void mptcp_close_ssk(struct sock *sk, struct sock *ssk, struct mptcp_subflow_context *subflow)
{ /* The first subflow can already be closed and still in the list */ if (subflow->close_event_done) return;
subflow->close_event_done = true;
if (sk->sk_state == TCP_ESTABLISHED)
mptcp_event(MPTCP_EVENT_SUB_CLOSED, mptcp_sk(sk), ssk, GFP_KERNEL);
/* subflow aborted before reaching the fully_established status *attemptthecreationofthenextsubflow
*/
mptcp_pm_subflow_check_next(mptcp_sk(sk), subflow);
/* Mirror the tcp_reset() error propagation */ switch (sk->sk_state) { case TCP_SYN_SENT:
WRITE_ONCE(sk->sk_err, ECONNREFUSED); break; case TCP_CLOSE_WAIT:
WRITE_ONCE(sk->sk_err, EPIPE); break; case TCP_CLOSE: return; default:
WRITE_ONCE(sk->sk_err, ECONNRESET);
}
mptcp_set_state(sk, TCP_CLOSE);
WRITE_ONCE(sk->sk_shutdown, SHUTDOWN_MASK);
smp_mb__before_atomic(); /* SHUTDOWN must be visible first */
set_bit(MPTCP_WORK_CLOSE_SUBFLOW, &msk->flags);
/* the calling mptcp_worker will properly destroy the socket */ if (sock_flag(sk, SOCK_DEAD)) return;
mptcp_for_each_subflow(msk, subflow) { if (READ_ONCE(subflow->scheduled)) {
u16 copied = 0;
mptcp_subflow_set_scheduled(subflow, false);
ssk = mptcp_subflow_tcp_sock(subflow);
lock_sock(ssk);
/* limit retransmission to the bytes already sent on some subflows */
info.sent = 0;
info.limit = READ_ONCE(msk->csum_enabled) ? dfrag->data_len :
dfrag->already_sent;
/* the close timeout takes precedence on the fail one, and here at least one of *themisactive
*/
timeout = inet_csk(sk)->icsk_mtup.probe_timestamp ? close_timeout : fail_tout;
switch (ssk->sk_state) { case TCP_LISTEN: if (!(how & RCV_SHUTDOWN)) break;
fallthrough; case TCP_SYN_SENT:
WARN_ON_ONCE(tcp_disconnect(ssk, O_NONBLOCK)); break; default: if (__mptcp_check_fallback(mptcp_sk(sk))) {
pr_debug("Fallback\n");
ssk->sk_shutdown |= how;
tcp_shutdown(ssk, how);
/* simulate the data_fin ack reception to let the state *machinemoveforward
*/
WRITE_ONCE(mptcp_sk(sk)->snd_una, mptcp_sk(sk)->snd_nxt);
mptcp_schedule_work(sk);
} else {
pr_debug("Sending DATA_FIN on subflow %p\n", ssk);
tcp_send_ack(ssk); if (!mptcp_rtx_timer_pending(sk))
mptcp_reset_rtx_timer(sk);
} break;
}
release_sock(ssk);
}
void mptcp_set_state(struct sock *sk, int state)
{ int oldstate = sk->sk_state;
switch (state) { case TCP_ESTABLISHED: if (oldstate != TCP_ESTABLISHED)
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_CURRESTAB); break; case TCP_CLOSE_WAIT: /* Unlike TCP, MPTCP sk would not have the TCP_SYN_RECV state: *MPTCP"accepted"socketswillbecreatedlateron.Sono *transitionfromTCP_SYN_RECVtoTCP_CLOSE_WAIT.
*/ break; default: if (oldstate == TCP_ESTABLISHED || oldstate == TCP_CLOSE_WAIT)
MPTCP_DEC_STATS(sock_net(sk), MPTCP_MIB_CURRESTAB);
}
/* we still need to enqueue subflows or not really shutting down, *skipthis
*/ if (!msk->snd_data_fin_enable || msk->snd_nxt + 1 != msk->write_seq ||
mptcp_send_head(sk)) return;
if (mptcp_data_avail(msk) || timeout < 0) { /* If the msk has read data, or the caller explicitly ask it, *dotheMPTCPequivalentofTCPreset,akaMPTCPfastclose
*/
mptcp_do_fastclose(sk);
timeout = 0;
} elseif (mptcp_close_state(sk)) {
__mptcp_wr_shutdown(sk);
}
sk_stream_wait_close(sk, timeout);
cleanup: /* orphan all the subflows */
mptcp_for_each_subflow(msk, subflow) { struct sock *ssk = mptcp_subflow_tcp_sock(subflow); bool slow = lock_sock_fast_nested(ssk);
subflows_alive += ssk->sk_state != TCP_CLOSE;
/* since the close timeout takes precedence on the fail one, *cancelthelatter
*/ if (ssk == msk->first)
subflow->fail_tout = 0;
/* detach from the parent socket, but allow data_ready to *pushincomingdataintothemptcpstack,toproperlyackit
*/
ssk->sk_socket = NULL;
ssk->sk_wq = NULL;
unlock_sock_fast(ssk, slow);
}
sock_orphan(sk);
/* all the subflows are closed, only timeout can change the msk *state,let'snotkeepresourcesbusyfornoreasons
*/ if (subflows_alive == 0)
mptcp_set_state(sk, TCP_CLOSE);
/* We are on the fastopen error path. We can't call straight into the *subflowscleanupcodeduetolocknesting(wearealreadyunder *msk->firstsocketlock).
*/ if (msk->fastopening) return -EBUSY;
/* msk->subflow is still intact, the following will not free the first *subflow
*/
mptcp_destroy_common(msk, MPTCP_CF_FASTCLOSE);
/* The first subflow is already in TCP_CLOSE status, the following *can'toverlapwithafallbackanymore
*/
spin_lock_bh(&msk->fallback_lock);
msk->allow_subflows = true;
msk->allow_infinite_fallback = true;
WRITE_ONCE(msk->flags, 0);
spin_unlock_bh(&msk->fallback_lock);
/* this can't race with mptcp_close(), as the msk is *notyetexpostedtouser-space
*/
mptcp_set_state(nsk, TCP_ESTABLISHED);
/* The msk maintain a ref to each subflow in the connections list */
WRITE_ONCE(msk->first, ssk);
subflow = mptcp_subflow_ctx(ssk);
list_add(&subflow->node, &msk->conn_list);
sock_hold(ssk);
/* new mpc subflow takes ownership of the newly *createdmptcpsocket
*/
mptcp_token_accept(subflow_req, msk);
/* set msk addresses early to ensure mptcp_pm_get_local_id() *usesthecorrectdata
*/
mptcp_copy_inaddrs(nsk, ssk);
__mptcp_propagate_sndbuf(nsk, ssk);
mptcp_rcv_space_init(msk, ssk);
if (mp_opt->suboptions & OPTION_MPTCP_MPC_ACK)
__mptcp_subflow_fully_established(msk, subflow, mp_opt);
bh_unlock_sock(nsk);
/* note: the newly allocated socket refcount is 2 now */ return nsk;
}
/* join list will be eventually flushed (with rst) at sock lock release time */
mptcp_for_each_subflow_safe(msk, subflow, tmp)
__mptcp_close_ssk(sk, mptcp_subflow_tcp_sock(subflow), subflow, flags);
if (__test_and_clear_bit(MPTCP_CLEAN_UNA, &msk->cb_flags))
__mptcp_clean_una_wakeup(sk); if (unlikely(msk->cb_flags)) { /* be sure to sync the msk state before taking actions *dependingonsk_state(MPTCP_ERROR_REPORT) *Onskreleaseavoidactionsdependingonthefirstsubflow
*/ if (__test_and_clear_bit(MPTCP_SYNC_STATE, &msk->cb_flags) && msk->first)
__mptcp_sync_state(sk, msk->pending_state); if (__test_and_clear_bit(MPTCP_ERROR_REPORT, &msk->cb_flags))
__mptcp_error_report(sk); if (__test_and_clear_bit(MPTCP_SYNC_SNDBUF, &msk->cb_flags))
__mptcp_sync_sndbuf(sk);
}
}
/* MP_JOIN client subflow must wait for 4th ack before sending any data: *TCPcan'tscheduledelacktimerbeforethesubflowisfullyestablished. *MPTCPusesthedelacktimertodo3rdackretransmissions
*/ staticvoid schedule_3rdack_retransmission(struct sock *ssk)
{ struct inet_connection_sock *icsk = inet_csk(ssk); struct tcp_sock *tp = tcp_sk(ssk); unsignedlong timeout;
if (READ_ONCE(mptcp_subflow_ctx(ssk)->fully_established)) return;
/* reschedule with a timeout above RTT, as we must look only for drop */ if (tp->srtt_us)
timeout = usecs_to_jiffies(tp->srtt_us >> (3 - 1)); else
timeout = TCP_TIMEOUT_INIT;
timeout += jiffies;
/* the first subflow is disconnected after close - see *__mptcp_close_ssk().tcp_disconnect()movesthewrite_seq *soignorethatstatus,too.
*/ if (!((1 << msk->first->sk_state) &
(TCPF_SYN_SENT | TCPF_SYN_RECV | TCPF_CLOSE)))
delta += READ_ONCE(tp->write_seq) - tp->snd_una;
} if (delta > INT_MAX)
delta = INT_MAX;
return (int)delta;
}
staticint mptcp_ioctl(struct sock *sk, int cmd, int *karg)
{ struct mptcp_sock *msk = mptcp_sk(sk); bool slow;
switch (cmd) { case SIOCINQ: if (sk->sk_state == TCP_LISTEN) return -EINVAL;
ssk = __mptcp_nmpc_sk(msk); if (IS_ERR(ssk)) return PTR_ERR(ssk);
mptcp_set_state(sk, TCP_SYN_SENT);
subflow = mptcp_subflow_ctx(ssk); #ifdef CONFIG_TCP_MD5SIG /* no MPTCP if MD5SIG is enabled on this socket or we may run out of *TCPoptionspace.
*/ if (rcu_access_pointer(tcp_sk(ssk)->md5sig_info))
mptcp_early_fallback(msk, subflow, MPTCP_MIB_MD5SIGFALLBACK); #endif if (subflow->request_mptcp) { if (mptcp_active_should_disable(sk))
mptcp_early_fallback(msk, subflow,
MPTCP_MIB_MPCAPABLEACTIVEDISABLED); elseif (mptcp_token_new_connect(ssk) < 0)
mptcp_early_fallback(msk, subflow,
MPTCP_MIB_TOKENFALLBACKINIT);
}
WRITE_ONCE(msk->write_seq, subflow->idsn);
WRITE_ONCE(msk->snd_nxt, subflow->idsn);
WRITE_ONCE(msk->snd_una, subflow->idsn); if (likely(!__mptcp_check_fallback(msk)))
MPTCP_INC_STATS(sock_net(sk), MPTCP_MIB_MPCAPABLEACTIVE);
/* if reaching here via the fastopen/sendmsg path, the caller already *acquiredthesubflowsocketlock,too.
*/ if (!msk->fastopening)
lock_sock(ssk);
/* the following mirrors closely a very small chunk of code from *__inet_stream_connect()
*/ if (ssk->sk_state != TCP_CLOSE) goto out;
if (BPF_CGROUP_PRE_CONNECT_ENABLED(ssk)) {
err = ssk->sk_prot->pre_connect(ssk, uaddr, addr_len); if (err) goto out;
}
/* on successful connect, the msk state will be moved to established by *subflow_finish_connect()
*/ if (unlikely(err)) { /* avoid leaving a dangling token in an unconnected socket */
mptcp_token_destroy(msk);
mptcp_set_state(sk, TCP_CLOSE); return err;
}
/* Buggy applications can call accept on socket states other then LISTEN *butnoneedtoallocatethefirstsubflowjusttoerrorout.
*/
ssk = READ_ONCE(msk->first); if (!ssk) return -EINVAL;
/* is_mptcp should be false if subflow->conn is missing, see *subflow_syn_recv_sock()
*/ if (WARN_ON_ONCE(!new_mptcp_sock)) {
tcp_sk(newsk)->is_mptcp = 0; goto tcpfallback;
}
/* set ssk->sk_socket of accept()ed flows to mptcp socket. *ThisisneededsoNOSPACEflagcanbesetfromtcpstack.
*/
mptcp_for_each_subflow(msk, subflow) { struct sock *ssk = mptcp_subflow_tcp_sock(subflow);
if (!ssk->sk_socket)
mptcp_sock_graft(ssk, newsock);
}
/* Do late cleanup for the first subflow as necessary. Also *dealwithbadpeersnotdoingacompleteshutdown.
*/ if (unlikely(inet_sk_state_load(msk->first) == TCP_CLOSE)) {
__mptcp_close_ssk(newsk, msk->first,
mptcp_subflow_ctx(msk->first), 0); if (unlikely(list_is_singular(&msk->conn_list)))
mptcp_set_state(newsk, TCP_CLOSE);
}
} else {
tcpfallback:
newsk->sk_kern_sock = arg->kern;
lock_sock(newsk);
__inet_accept(sock, newsock, newsk); /* we are being invoked after accepting a non-mp-capable *flow:skisatcp_sk,notanmptcpone. * *Handthesocketovertotcpsoallfurthersocketops *bypassmptcp.
*/
WRITE_ONCE(newsock->sk->sk_socket->ops,
mptcp_fallback_tcp_ops(newsock->sk));
}
release_sock(newsk);
/* always provide a 0 'work_done' argument, so that napi_complete_done *willnottryaccessingtheNULLnapi->devptr
*/
napi_complete_done(napi, 0); return work_done;
}
void __init mptcp_proto_init(void)
{ struct mptcp_delegated_action *delegated; int cpu;
mptcp_prot.h.hashinfo = tcp_prot.h.hashinfo;
if (percpu_counter_init(&mptcp_sockets_allocated, 0, GFP_KERNEL))
panic("Failed to allocate MPTCP pcpu counter\n");
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.