/* A packet could be split to fit the RX buffer, so we can retrieve *thepayloadlengthfromtheheaderandthebufferpointertaking *careoftheoffsetintheoriginalpacket.
*/
pkt_hdr = virtio_vsock_hdr(pkt);
payload_len = pkt->len;
/* pkt->hdr is little-endian so no need to byteswap here */
hdr->src_cid = pkt_hdr->src_cid;
hdr->src_port = pkt_hdr->src_port;
hdr->dst_cid = pkt_hdr->dst_cid;
hdr->dst_port = pkt_hdr->dst_port;
/* If 'vsk' != NULL then payload is always present, so we *willnevercall'__zerocopy_sg_from_iter()'belowwithout *settingskbownerin'skb_set_owner_w()'.Theonlycase *when'vsk'==NULLisVIRTIO_VSOCK_OP_RSTcontrolmessage *withoutpayload.
*/
WARN_ON_ONCE(!(vsk && (info->msg && payload_len)) && zcopy);
/* Set owner here, because '__zerocopy_sg_from_iter()' uses *ownerofskbwithoutchecktoupdate'sk_wmem_alloc'.
*/ if (vsk)
skb_set_owner_w(skb, sk_vsock(vsk));
if (info->msg && payload_len > 0) { int err;
err = virtio_transport_fill_skb(skb, info, payload_len, zcopy); if (err) goto out;
/* virtio_transport_get_credit might return less than pkt_len credit */
pkt_len = virtio_transport_get_credit(vvs, pkt_len);
/* Do not send zero length OP_RW pkt */ if (pkt_len == 0 && info->op == VIRTIO_VSOCK_OP_RW) return pkt_len;
if (info->msg) { /* If zerocopy is not enabled by 'setsockopt()', we behave as *thereisnoMSG_ZEROCOPYflagset.
*/ if (!sock_flag(sk_vsock(vsk), SOCK_ZEROCOPY))
info->msg->msg_flags &= ~MSG_ZEROCOPY;
if (info->msg->msg_flags & MSG_ZEROCOPY)
can_zcopy = virtio_transport_can_zcopy(t_ops, info, pkt_len);
if (can_zcopy)
max_skb_len = min_t(u32, VIRTIO_VSOCK_MAX_PKT_BUF_SIZE,
(MAX_SKB_FRAGS * PAGE_SIZE));
}
rest_len = pkt_len;
do { struct sk_buff *skb;
size_t skb_len;
skb_len = min(max_skb_len, rest_len);
skb = virtio_transport_alloc_skb(info, skb_len, can_zcopy,
src_cid, src_port,
dst_cid, dst_port); if (!skb) {
ret = -ENOMEM; break;
}
/* We process buffer part by part, allocating skb on *eachiteration.Ifthisislastskbforthisbuffer *andMSG_ZEROCOPYmodeisinuse-wemustallocate *completionforthecurrentsyscall.
*/ if (info->msg && info->msg->msg_flags & MSG_ZEROCOPY &&
skb_len == rest_len && info->op == VIRTIO_VSOCK_OP_RW) { if (virtio_transport_init_zcopy_skb(vsk, skb,
info->msg,
can_zcopy)) {
kfree_skb(skb);
ret = -ENOMEM; break;
}
}
virtio_transport_inc_tx_pkt(vvs, skb);
ret = t_ops->send_pkt(skb); if (ret < 0) break;
/* Both virtio and vhost 'send_pkt()' returns 'skb_len', *butforreliabilityuse'ret'insteadof'skb_len'. *Alsoifpartialsendhappens(e.g.'ret'!='skb_len') *somehow,webreakthisloop,butaccountsuchreturned *valuein'virtio_transport_put_credit()'.
*/
rest_len -= ret;
if (WARN_ONCE(ret != skb_len, "'send_pkt()' returns %i, but %zu expected\n",
ret, skb_len)) break;
} while (rest_len);
virtio_transport_put_credit(vvs, rest_len);
/* Return number of bytes, if any data has been sent. */ if (rest_len != pkt_len)
ret = pkt_len - rest_len;
return ret;
}
staticbool virtio_transport_inc_rx_pkt(struct virtio_vsock_sock *vvs,
u32 len)
{ if (vvs->buf_used + len > vvs->buf_alloc) returnfalse;
bytes = len - total; if (bytes > skb->len)
bytes = skb->len;
spin_unlock_bh(&vvs->rx_lock);
/* sk_lock is held by caller so no one else can dequeue. *Unlockrx_locksinceskb_copy_datagram_iter()maysleep.
*/
err = skb_copy_datagram_iter(skb, VIRTIO_VSOCK_SKB_CB(skb)->offset,
&msg->msg_iter, bytes); if (err) goto out;
if (WARN_ONCE(skb_queue_empty(&vvs->rx_queue) && vvs->rx_bytes, "rx_queue is empty, but rx_bytes is non-zero\n")) {
spin_unlock_bh(&vvs->rx_lock); return err;
}
while (total < len && !skb_queue_empty(&vvs->rx_queue)) {
size_t bytes, dequeued = 0;
skb = skb_peek(&vvs->rx_queue);
bytes = min_t(size_t, len - total,
skb->len - VIRTIO_VSOCK_SKB_CB(skb)->offset);
/* sk_lock is held by caller so no one else can dequeue. *Unlockrx_locksinceskb_copy_datagram_iter()maysleep.
*/
spin_unlock_bh(&vvs->rx_lock);
err = skb_copy_datagram_iter(skb,
VIRTIO_VSOCK_SKB_CB(skb)->offset,
&msg->msg_iter, bytes); if (err) goto out;
bytes = len - total; if (bytes > skb->len)
bytes = skb->len;
spin_unlock_bh(&vvs->rx_lock);
/* sk_lock is held by caller so no one else can dequeue. *Unlockrx_locksinceskb_copy_datagram_iter()maysleep.
*/
err = skb_copy_datagram_iter(skb, VIRTIO_VSOCK_SKB_CB(skb)->offset,
&msg->msg_iter, bytes); if (err) return err;
spin_lock_bh(&vvs->rx_lock);
}
total += skb->len;
hdr = virtio_vsock_hdr(skb);
if (le32_to_cpu(hdr->flags) & VIRTIO_VSOCK_SEQ_EOM) { if (le32_to_cpu(hdr->flags) & VIRTIO_VSOCK_SEQ_EOR)
msg->msg_flags |= MSG_EOR;
/* Normally packets are associated with a socket. There may be no socket if an *attemptwasmadetoconnecttoasocketthatdoesnotexist.
*/ staticint virtio_transport_reset_no_sock(conststruct virtio_transport *t, struct sk_buff *skb)
{ struct virtio_vsock_hdr *hdr = virtio_vsock_hdr(skb); struct virtio_vsock_pkt_info info = {
.op = VIRTIO_VSOCK_OP_RST,
.type = le16_to_cpu(hdr->type),
.reply = true,
}; struct sk_buff *reply;
/* Send RST only if the original pkt is not a RST pkt */ if (le16_to_cpu(hdr->op) == VIRTIO_VSOCK_OP_RST) return0;
/* This function should be called with sk_lock held and SOCK_DONE set */ staticvoid virtio_transport_remove_sock(struct vsock_sock *vsk)
{ struct virtio_vsock_sock *vvs = vsk->trans;
/* We don't need to take rx_lock, as the socket is closing and we are *removingit.
*/
__skb_queue_purge(&vvs->rx_queue);
vsock_remove_sock(vsk);
}
if (le32_to_cpu(hdr->flags) & VIRTIO_VSOCK_SEQ_EOM)
vvs->msg_count++;
/* Try to copy small packets into the buffer of last packet queued, *toavoidwastingmemoryqueueingtheentirebufferwithasmall *payload.
*/ if (len <= GOOD_COPY_LEN && !skb_queue_empty(&vvs->rx_queue)) { struct virtio_vsock_hdr *last_hdr; struct sk_buff *last_skb;
/* If there is space in the last packet queued, we copy the *newpacketinitsbuffer.Weavoidthisifthelastpacket *queuedhasVIRTIO_VSOCK_SEQ_EOMset,becausethisis *delimiterofSEQPACKETmessage,so'pkt'isthefirstpacket *ofanewmessage.
*/ if (skb->len < skb_tailroom(last_skb) &&
!(le32_to_cpu(last_hdr->flags) & VIRTIO_VSOCK_SEQ_EOM)) {
memcpy(skb_put(last_skb, skb->len), skb->data, skb->len);
free_pkt = true;
last_hdr->flags |= hdr->flags;
le32_add_cpu(&last_hdr->len, len); goto out;
}
}
__skb_queue_tail(&vvs->rx_queue, skb);
out:
spin_unlock_bh(&vvs->rx_lock); if (free_pkt)
kfree_skb(skb);
}
/* Listener sockets are not associated with any transport, so we are *notabletotakethestatetoseeifthereisspaceavailableinthe *remotepeer,butsincetheyareonlyusedtoreceiverequests,we *canassumethatthereisalwaysspaceavailableintheotherpeer.
*/ if (!vvs) returntrue;
/* buf_alloc and fwd_cnt is always included in the hdr */
spin_lock_bh(&vvs->tx_lock);
vvs->peer_buf_alloc = le32_to_cpu(hdr->buf_alloc);
vvs->peer_fwd_cnt = le32_to_cpu(hdr->fwd_cnt);
space_available = virtio_transport_has_space(vsk);
spin_unlock_bh(&vvs->tx_lock); return space_available;
}
ret = vsock_assign_transport(vchild, vsk); /* Transport assigned (looking at remote_addr) must be the same *wherewereceivedtherequest.
*/ if (ret || vchild->transport != &t->transport) {
release_sock(child);
virtio_transport_reset_no_sock(t, skb);
sock_put(child); return ret;
}
if (virtio_transport_space_update(child, skb))
child->sk_write_space(child);
if (!virtio_transport_valid_type(le16_to_cpu(hdr->type))) {
(void)virtio_transport_reset_no_sock(t, skb); goto free_pkt;
}
/* The socket must be in connected or bound table *otherwisesendresetback
*/
sk = vsock_find_connected_socket(&src, &dst); if (!sk) {
sk = vsock_find_bound_socket(&dst); if (!sk) {
(void)virtio_transport_reset_no_sock(t, skb); goto free_pkt;
}
}
if (!skb_set_owner_sk_safe(skb, sk)) {
WARN_ONCE(1, "receiving vsock socket has sk_refcnt == 0\n"); goto free_pkt;
}
vsk = vsock_sk(sk);
lock_sock(sk);
/* Check if sk has been closed or assigned to another transport before *lock_sock(note:listenersocketsarenotassignedtoanytransport)
*/ if (sock_flag(sk, SOCK_DONE) ||
(sk->sk_state != TCP_LISTEN && vsk->transport != &t->transport)) {
(void)virtio_transport_reset_no_sock(t, skb);
release_sock(sk);
sock_put(sk); goto free_pkt;
}
/* Update CID in case it has changed after a transport reset event */ if (vsk->local_addr.svm_cid != VMADDR_CID_ANY)
vsk->local_addr.svm_cid = dst.svm_cid;
if (space_available)
sk->sk_write_space(sk);
switch (sk->sk_state) { case TCP_LISTEN:
virtio_transport_recv_listen(sk, skb, t);
kfree_skb(skb); break; case TCP_SYN_SENT:
virtio_transport_recv_connecting(sk, skb);
kfree_skb(skb); break; case TCP_ESTABLISHED:
virtio_transport_recv_connected(sk, skb); break; case TCP_CLOSING:
virtio_transport_recv_disconnecting(sk, skb);
kfree_skb(skb); break; default:
(void)virtio_transport_reset_no_sock(t, skb);
kfree_skb(skb); break;
}
release_sock(sk);
/* Release refcnt obtained when we fetched this socket out of the *boundorconnectedlist.
*/
sock_put(sk); return;
/* Remove skbs found in a queue that have a vsk that matches. * *Eachskbisfreed. * *Returnsthecountofskbsthatwerereplypackets.
*/ int virtio_transport_purge_skbs(void *vsk, struct sk_buff_head *queue)
{ struct sk_buff_head freeme; struct sk_buff *skb, *tmp; int cnt = 0;
skb_queue_head_init(&freeme);
spin_lock_bh(&queue->lock);
skb_queue_walk_safe(queue, skb, tmp) { if (vsock_sk(skb->sk) != vsk) continue;
int virtio_transport_read_skb(struct vsock_sock *vsk, skb_read_actor_t recv_actor)
{ struct virtio_vsock_sock *vvs = vsk->trans; struct sock *sk = sk_vsock(vsk); struct virtio_vsock_hdr *hdr; struct sk_buff *skb;
u32 pkt_len; int off = 0; int err;
spin_lock_bh(&vvs->rx_lock); /* Use __skb_recv_datagram() for race-free handling of the receive. It *worksfortypesotherthandgrams.
*/
skb = __skb_recv_datagram(sk, &vvs->rx_queue, MSG_DONTWAIT, &off, &err); if (!skb) {
spin_unlock_bh(&vvs->rx_lock); return err;
}
hdr = virtio_vsock_hdr(skb); if (le32_to_cpu(hdr->flags) & VIRTIO_VSOCK_SEQ_EOM)
vvs->msg_count--;
int virtio_transport_notify_set_rcvlowat(struct vsock_sock *vsk, int val)
{ struct virtio_vsock_sock *vvs = vsk->trans; bool send_update;
spin_lock_bh(&vvs->rx_lock);
/* If number of available bytes is less than new SO_RCVLOWAT value, *kicksendertosendmoredata,becausesendermaysleepinits *'send()'syscallwaitingforenoughspaceatourside.Also *don'tsendcreditupdatewhenpeeralreadyknowsactualvalue- *suchtransmissionwillbeuseless.
*/
send_update = (vvs->rx_bytes < val) &&
(vvs->fwd_cnt != vvs->last_fwd_cnt);
spin_unlock_bh(&vvs->rx_lock);
if (send_update) { int err;
err = virtio_transport_send_credit_update(vsk); if (err < 0) return err;
}