From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-05 09:12:47
The first patch made some adjustments to xsk.
The second patch itself can be used as an independent patch to solve the problem
that XDP may fail to load when the number of queues is insufficient.
The third to last patch implements support for xsk in virtio-net.
A practical problem with virtio is that tx interrupts are not very reliable.
There will always be some missing or delayed tx interrupts. So I specially added
a point timer to solve this problem. Of course, considering performance issues,
The timer only triggers when the ring of the network card is full.
Regarding the issue of virtio-net supporting xsk's zero copy rx, I am also
developing it, but I found that the modification may be relatively large, so I
consider this patch set to be separated from the code related to xsk zero copy
rx.
Xuan Zhuo (5):
xsk: support get page for drv
virtio-net: support XDP_TX when not more queues
virtio-net, xsk: distinguish XDP_TX and XSK XMIT ctx
xsk, virtio-net: prepare for support xsk
virtio-net, xsk: virtio-net support xsk zero copy tx
drivers/net/virtio_net.c | 643 +++++++++++++++++++++++++++++++++++++++-----
include/linux/netdevice.h | 1 +
include/net/xdp_sock_drv.h | 10 +
include/net/xsk_buff_pool.h | 1 +
net/xdp/xsk_buff_pool.c | 10 +-
5 files changed, 597 insertions(+), 68 deletions(-)
--
1.8.3.1
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-05 09:12:47
The number of queues implemented by many virtio backends is limited,
especially some machines have a large number of CPUs. In this case, it
is often impossible to allocate a separate queue for XDP_TX.
This patch allows XDP_TX to run by lock when not enough queue.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 42 ++++++++++++++++++++++++++++++++----------
1 file changed, 32 insertions(+), 10 deletions(-)
@@ -194,6 +194,7 @@ struct virtnet_info {/* # of XDP queue pairs currently used by the driver */u16xdp_queue_pairs;+boolxdp_enable;/* I like... big packets and I cannot lie! */boolbig_packets;
@@ -481,14 +482,34 @@ static int __virtnet_xdp_xmit_one(struct virtnet_info *vi,return0;}-staticstructsend_queue*virtnet_xdp_sq(structvirtnet_info*vi)+staticstructsend_queue*virtnet_get_xdp_sq(structvirtnet_info*vi){unsignedintqp;+structnetdev_queue*txq;++if(vi->curr_queue_pairs>nr_cpu_ids){+qp=vi->curr_queue_pairs-vi->xdp_queue_pairs+smp_processor_id();+}else{+qp=smp_processor_id()%vi->curr_queue_pairs;+txq=netdev_get_tx_queue(vi->dev,qp);+__netif_tx_lock(txq,raw_smp_processor_id());+}-qp=vi->curr_queue_pairs-vi->xdp_queue_pairs+smp_processor_id();return&vi->sq[qp];}+staticvoidvirtnet_put_xdp_sq(structvirtnet_info*vi)+{+unsignedintqp;+structnetdev_queue*txq;++if(vi->curr_queue_pairs<=nr_cpu_ids){+qp=smp_processor_id()%vi->curr_queue_pairs;+txq=netdev_get_tx_queue(vi->dev,qp);+__netif_tx_unlock(txq);+}+}+staticintvirtnet_xdp_xmit(structnet_device*dev,intn,structxdp_frame**frames,u32flags){
@@ -512,7 +533,7 @@ static int virtnet_xdp_xmit(struct net_device *dev,if(!xdp_prog)return-ENXIO;-sq=virtnet_xdp_sq(vi);+sq=virtnet_get_xdp_sq(vi);if(unlikely(flags&~XDP_XMIT_FLAGS_MASK)){ret=-EINVAL;
@@ -560,12 +581,13 @@ static int virtnet_xdp_xmit(struct net_device *dev,sq->stats.kicks+=kicks;u64_stats_update_end(&sq->stats.syncp);+virtnet_put_xdp_sq(vi);returnret;}staticunsignedintvirtnet_get_headroom(structvirtnet_info*vi){-returnvi->xdp_queue_pairs?VIRTIO_XDP_HEADROOM:0;+returnvi->xdp_enable?VIRTIO_XDP_HEADROOM:0;}/* We copy the packet for XDP in the following cases:
@@ -1457,12 +1479,13 @@ static int virtnet_poll(struct napi_struct *napi, int budget)xdp_do_flush();if(xdp_xmit&VIRTIO_XDP_TX){-sq=virtnet_xdp_sq(vi);+sq=virtnet_get_xdp_sq(vi);if(virtqueue_kick_prepare(sq->vq)&&virtqueue_notify(sq->vq)){u64_stats_update_begin(&sq->stats.syncp);sq->stats.kicks++;u64_stats_update_end(&sq->stats.syncp);}+virtnet_put_xdp_sq(vi);}returnreceived;
@@ -2415,10 +2438,7 @@ static int virtnet_xdp_set(struct net_device *dev, struct bpf_prog *prog,/* XDP requires extra queues for XDP_TX */if(curr_qp+xdp_qp>vi->max_queue_pairs){-NL_SET_ERR_MSG_MOD(extack,"Too few free TX rings available");-netdev_warn(dev,"request %i queues but max is %i\n",-curr_qp+xdp_qp,vi->max_queue_pairs);-return-ENOMEM;+xdp_qp=0;}old_prog=rtnl_dereference(vi->rq[0].xdp_prog);
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-05 09:12:47
Split function free_old_xmit_skbs, add sub-function __free_old_xmit_ptr,
which is convenient to call with other statistical information, and
supports the parameter 'xsk_wakeup' required for processing xsk.
Use netif stop check as a function virtnet_sq_stop_check, which will be
used when adding xsk support.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 95 ++++++++++++++++++++++++++----------------------
1 file changed, 52 insertions(+), 43 deletions(-)
@@ -376,6 +381,37 @@ static void skb_xmit_done(struct virtqueue *vq)netif_wake_subqueue(vi->dev,vq2txq(vq));}+staticvoidvirtnet_sq_stop_check(structsend_queue*sq,boolin_napi)+{+structvirtnet_info*vi=sq->vq->vdev->priv;+structnet_device*dev=vi->dev;+intqnum=sq-vi->sq;++/* If running out of space, stop queue to avoid getting packets that we+*arethenunabletotransmit.+*Analternativewouldbetoforcequeuinglayertorequeuetheskbby+*returningNETDEV_TX_BUSY.However,NETDEV_TX_BUSYshouldnotbe+*returnedinanormalpathofoperation:itmeansthatdriverisnot+*maintainingtheTXqueuestop/startstateproperly,andcauses+*thestacktodoanon-trivialamountofuselesswork.+*Sincemostpacketsonlytake1or2ringslots,stoppingthequeue+*earlymeans16slotsaretypicallywasted.+*/++if(sq->vq->num_free<2+MAX_SKB_FRAGS){+netif_stop_subqueue(dev,qnum);+if(!sq->napi.weight&&+unlikely(!virtqueue_enable_cb_delayed(sq->vq))){+/* More just got used, free them then recheck. */+free_old_xmit_skbs(sq,in_napi);+if(sq->vq->num_free>=2+MAX_SKB_FRAGS){+netif_start_subqueue(dev,qnum);+virtqueue_disable_cb(sq->vq);+}+}+}+}+#define MRG_CTX_HEADER_SHIFT 22staticvoid*mergeable_len_to_ctx(unsignedinttruesize,unsignedintheadroom)
@@ -543,13 +579,11 @@ static int virtnet_xdp_xmit(struct net_device *dev,structreceive_queue*rq=vi->rq;structbpf_prog*xdp_prog;structsend_queue*sq;-unsignedintlen;intpackets=0;intbytes=0;intdrops=0;intkicks=0;intret,err;-void*ptr;inti;/* Only allow ndo_xdp_xmit if XDP is loaded on dev, as this
@@ -567,24 +601,7 @@ static int virtnet_xdp_xmit(struct net_device *dev,gotoout;}-/* Free up any pending old buffers before queueing new ones. */-while((ptr=virtqueue_get_buf(sq->vq,&len))!=NULL){-if(likely(is_xdp_frame(ptr))){-structvirtnet_xdp_type*xtype;-structxdp_frame*frame;--xtype=ptr_to_xtype(ptr);-frame=xtype_got_ptr(xtype);-bytes+=frame->len;-xdp_return_frame(frame);-}else{-structsk_buff*skb=ptr;--bytes+=skb->len;-napi_consume_skb(skb,false);-}-packets++;-}+__free_old_xmit_ptr(sq,false,true,&packets,&bytes);for(i=0;i<n;i++){structxdp_frame*xdpf=frames[i];
@@ -1422,7 +1439,9 @@ static int virtnet_receive(struct receive_queue *rq, int budget,returnstats.packets;}-staticvoidfree_old_xmit_skbs(structsend_queue*sq,boolin_napi)+staticvoid__free_old_xmit_ptr(structsend_queue*sq,boolin_napi,+boolxsk_wakeup,+unsignedint*_packets,unsignedint*_bytes){unsignedintpackets=0;unsignedintbytes=0;
@@ -1456,6 +1475,17 @@ static void free_old_xmit_skbs(struct send_queue *sq, bool in_napi)packets++;}+*_packets=packets;+*_bytes=bytes;+}++staticvoidfree_old_xmit_skbs(structsend_queue*sq,boolin_napi)+{+unsignedintpackets=0;+unsignedintbytes=0;++__free_old_xmit_ptr(sq,in_napi,true,&packets,&bytes);+/* Avoid overhead when no packets have been processed*happenswhencalledspeculativelyfromstart_xmit.*/
@@ -1672,28 +1702,7 @@ static netdev_tx_t start_xmit(struct sk_buff *skb, struct net_device *dev)nf_reset_ct(skb);}-/* If running out of space, stop queue to avoid getting packets that we-*arethenunabletotransmit.-*Analternativewouldbetoforcequeuinglayertorequeuetheskbby-*returningNETDEV_TX_BUSY.However,NETDEV_TX_BUSYshouldnotbe-*returnedinanormalpathofoperation:itmeansthatdriverisnot-*maintainingtheTXqueuestop/startstateproperly,andcauses-*thestacktodoanon-trivialamountofuselesswork.-*Sincemostpacketsonlytake1or2ringslots,stoppingthequeue-*earlymeans16slotsaretypicallywasted.-*/-if(sq->vq->num_free<2+MAX_SKB_FRAGS){-netif_stop_subqueue(dev,qnum);-if(!use_napi&&-unlikely(!virtqueue_enable_cb_delayed(sq->vq))){-/* More just got used, free them then recheck. */-free_old_xmit_skbs(sq,false);-if(sq->vq->num_free>=2+MAX_SKB_FRAGS){-netif_start_subqueue(dev,qnum);-virtqueue_disable_cb(sq->vq);-}-}-}+virtnet_sq_stop_check(sq,false);if(kick||netif_xmit_stopped(txq)){if(virtqueue_kick_prepare(sq->vq)&&virtqueue_notify(sq->vq)){
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-05 09:12:47
If support xsk, a new ptr will be recovered during the
process of freeing the old ptr. In order to distinguish between ctx sent
by XDP_TX and ctx sent by xsk, a struct is added here to distinguish
between these two situations. virtnet_xdp_type.type It is used to
distinguish different ctx, and virtnet_xdp_type.offset is used to record
the offset between "true ctx" and virtnet_xdp_type.
The newly added virtnet_xsk_hdr will be used for xsk.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 77 ++++++++++++++++++++++++++++++++++++++----------
1 file changed, 62 insertions(+), 15 deletions(-)
@@ -252,14 +268,19 @@ static bool is_xdp_frame(void *ptr)return(unsignedlong)ptr&VIRTIO_XDP_FLAG;}-staticvoid*xdp_to_ptr(structxdp_frame*ptr)+staticvoid*xdp_to_ptr(structvirtnet_xdp_type*ptr){return(void*)((unsignedlong)ptr|VIRTIO_XDP_FLAG);}-staticstructxdp_frame*ptr_to_xdp(void*ptr)+staticstructvirtnet_xdp_type*ptr_to_xtype(void*ptr){-return(structxdp_frame*)((unsignedlong)ptr&~VIRTIO_XDP_FLAG);+return(structvirtnet_xdp_type*)((unsignedlong)ptr&~VIRTIO_XDP_FLAG);+}++staticvoid*xtype_got_ptr(structvirtnet_xdp_type*xdptype)+{+return(char*)xdptype+xdptype->offset;}/* Converting between virtqueue no. and kernel tx/rx queue no.
@@ -460,11 +481,16 @@ static int __virtnet_xdp_xmit_one(struct virtnet_info *vi,structxdp_frame*xdpf){structvirtio_net_hdr_mrg_rxbuf*hdr;+structvirtnet_xdp_type*xdptype;interr;-if(unlikely(xdpf->headroom<vi->hdr_len))+if(unlikely(xdpf->headroom<vi->hdr_len+sizeof(*xdptype)))return-EOVERFLOW;+xdptype=(structvirtnet_xdp_type*)(xdpf+1);+xdptype->offset=(char*)xdpf-(char*)xdptype;+xdptype->type=XDP_TYPE_TX;+/* Make room for virtqueue hdr (also change xdpf->headroom?) */xdpf->data-=vi->hdr_len;/* Zero header and leave csum up to XDP layers */
@@ -544,8 +570,11 @@ static int virtnet_xdp_xmit(struct net_device *dev,/* Free up any pending old buffers before queueing new ones. */while((ptr=virtqueue_get_buf(sq->vq,&len))!=NULL){if(likely(is_xdp_frame(ptr))){-structxdp_frame*frame=ptr_to_xdp(ptr);+structvirtnet_xdp_type*xtype;+structxdp_frame*frame;+xtype=ptr_to_xtype(ptr);+frame=xtype_got_ptr(xtype);bytes+=frame->len;xdp_return_frame(frame);}else{
@@ -1395,24 +1424,34 @@ static int virtnet_receive(struct receive_queue *rq, int budget,staticvoidfree_old_xmit_skbs(structsend_queue*sq,boolin_napi){-unsignedintlen;unsignedintpackets=0;unsignedintbytes=0;-void*ptr;+unsignedintlen;+structvirtnet_xdp_type*xtype;+structxdp_frame*frame;+structvirtnet_xsk_hdr*xskhdr;+structsk_buff*skb;+void*ptr;while((ptr=virtqueue_get_buf(sq->vq,&len))!=NULL){if(likely(!is_xdp_frame(ptr))){-structsk_buff*skb=ptr;+skb=ptr;pr_debug("Sent skb %p\n",skb);bytes+=skb->len;napi_consume_skb(skb,in_napi);}else{-structxdp_frame*frame=ptr_to_xdp(ptr);+xtype=ptr_to_xtype(ptr);-bytes+=frame->len;-xdp_return_frame(frame);+if(xtype->type==XDP_TYPE_XSK){+xskhdr=(structvirtnet_xsk_hdr*)xtype;+bytes+=xskhdr->len;+}else{+frame=xtype_got_ptr(xtype);+xdp_return_frame(frame);+bytes+=frame->len;+}}packets++;}
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-05 09:12:48
Virtio net support xdp socket.
We should open the module param "napi_tx" for using this feature.
In fact, various virtio implementations have some problems:
1. The tx interrupt may be lost
2. The tx interrupt may have a relatively large delay
This brings us to several questions:
1. Wakeup wakes up a tx interrupt or directly starts a napi on the
current cpu, which will cause a delay in sending packets.
2. When the tx ring is full, the tx interrupt may be lost or delayed,
resulting in untimely recovery.
I choose to send part of the data directly during wakeup. If the sending
has not been completed, I will start a napi to complete the subsequent
sending work.
Since the possible delay or loss of tx interrupt occurs when the tx ring
is full, I added a timer to solve this problem.
The performance of udp sending based on virtio net + xsk is 6 times that
of ordinary kernel udp send.
* xsk_check_timeout: when the dev full or all xsk.hdr used, start timer
to check the xsk.hdr is avail. the unit is us.
* xsk_num_max: the xsk.hdr max num
* xsk_num_percent: the max hdr num be the percent of the virtio ring
size. The real xsk hdr num will the min of xsk_num_max and the percent
of the num of virtio ring
* xsk_budget: the budget for xsk run
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 437 ++++++++++++++++++++++++++++++++++++++++++++++-
1 file changed, 434 insertions(+), 3 deletions(-)
@@ -1595,6 +1676,8 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget)structvirtnet_info*vi=sq->vq->vdev->priv;unsignedintindex=vq2txq(sq->vq);structnetdev_queue*txq;+structxsk_buff_pool*pool;+intwork=0;if(unlikely(is_xdp_raw_buffer_queue(vi,index))){/* We don't need to enable cb for XDP */
@@ -1604,15 +1687,26 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget)txq=netdev_get_tx_queue(vi->dev,index);__netif_tx_lock(txq,raw_smp_processor_id());-free_old_xmit_skbs(sq,true);++rcu_read_lock();+pool=rcu_dereference(sq->xsk.pool);+if(pool){+work=virtnet_xsk_run(sq,pool,budget);+rcu_read_unlock();+}else{+rcu_read_unlock();+free_old_xmit_skbs(sq,true);+}+__netif_tx_unlock(txq);-virtqueue_napi_complete(napi,sq->vq,0);+if(work<budget)+virtqueue_napi_complete(napi,sq->vq,0);if(sq->vq->num_free>=2+MAX_SKB_FRAGS)netif_tx_wake_queue(txq);-return0;+returnwork;}staticintxmit_skb(structsend_queue*sq,structsk_buff*skb)
@@ -2560,16 +2654,346 @@ static int virtnet_xdp_set(struct net_device *dev, struct bpf_prog *prog,returnerr;}+staticenumhrtimer_restartvirtnet_xsk_timeout(structhrtimer*timer)+{+structsend_queue*sq;++sq=container_of(timer,structsend_queue,xsk.timer);++clear_bit(VIRTNET_STATE_XSK_TIMER,&sq->xsk.state);++virtqueue_napi_schedule(&sq->napi,sq->vq);++returnHRTIMER_NORESTART;+}++staticintvirtnet_xsk_pool_enable(structnet_device*dev,+structxsk_buff_pool*pool,+u16qid)+{+structvirtnet_info*vi=netdev_priv(dev);+structsend_queue*sq=&vi->sq[qid];+structvirtnet_xsk_hdr*hdr;+intn,ret=0;++if(qid>=dev->real_num_rx_queues||qid>=dev->real_num_tx_queues)+return-EINVAL;++if(qid>=vi->curr_queue_pairs)+return-EINVAL;++rcu_read_lock();++ret=-EBUSY;+if(rcu_dereference(sq->xsk.pool))+gotoend;++/* check last xsk wait for hdr been free */+if(rcu_dereference(sq->xsk.hdr))+gotoend;++n=virtqueue_get_vring_size(sq->vq);+n=min(xsk_num_max,n*(xsk_num_percent%100)/100);++ret=-ENOMEM;+hdr=kcalloc(n,sizeof(structvirtnet_xsk_hdr),GFP_ATOMIC);+if(!hdr)+gotoend;++memset(&sq->xsk,0,sizeof(sq->xsk));++sq->xsk.hdr_pro=n;+sq->xsk.hdr_n=n;++hrtimer_init(&sq->xsk.timer,CLOCK_MONOTONIC,HRTIMER_MODE_REL_PINNED);+sq->xsk.timer.function=virtnet_xsk_timeout;++rcu_assign_pointer(sq->xsk.pool,pool);+rcu_assign_pointer(sq->xsk.hdr,hdr);++ret=0;+end:+rcu_read_unlock();++returnret;+}++staticintvirtnet_xsk_pool_disable(structnet_device*dev,u16qid)+{+structvirtnet_info*vi=netdev_priv(dev);+structsend_queue*sq=&vi->sq[qid];++if(qid>=dev->real_num_rx_queues||qid>=dev->real_num_tx_queues)+return-EINVAL;++if(qid>=vi->curr_queue_pairs)+return-EINVAL;++rcu_assign_pointer(sq->xsk.pool,NULL);++hrtimer_cancel(&sq->xsk.timer);++synchronize_rcu();/* Sync with the XSK wakeup and with NAPI. */++if(sq->xsk.hdr_pro-sq->xsk.hdr_con==sq->xsk.hdr_n){+kfree(sq->xsk.hdr);+rcu_assign_pointer(sq->xsk.hdr,NULL);+synchronize_rcu();+}++return0;+}+staticintvirtnet_xdp(structnet_device*dev,structnetdev_bpf*xdp){switch(xdp->command){caseXDP_SETUP_PROG:returnvirtnet_xdp_set(dev,xdp->prog,xdp->extack);+caseXDP_SETUP_XSK_POOL:+xdp->xsk.need_dma=false;+if(xdp->xsk.pool)+returnvirtnet_xsk_pool_enable(dev,xdp->xsk.pool,+xdp->xsk.queue_id);+else+returnvirtnet_xsk_pool_disable(dev,xdp->xsk.queue_id);default:return-EINVAL;}}+staticintvirtnet_xsk_xmit(structsend_queue*sq,structxsk_buff_pool*pool,+structxdp_desc*desc)+{+structvirtnet_info*vi=sq->vq->vdev->priv;+void*data,*ptr;+structpage*page;+structvirtnet_xsk_hdr*xskhdr;+u32idx,offset,n,i,copy,copied;+u64addr;+interr,m;++addr=desc->addr;++data=xsk_buff_raw_get_data(pool,addr);+offset=offset_in_page(data);++/* one for hdr, one for the first page */+n=2;+m=desc->len-(PAGE_SIZE-offset);+if(m>0){+n+=m>>PAGE_SHIFT;+if(m&PAGE_MASK)+++n;++n=min_t(u32,n,ARRAY_SIZE(sq->sg));+}++idx=sq->xsk.hdr_con%sq->xsk.hdr_n;+xskhdr=&sq->xsk.hdr[idx];++/* xskhdr->hdr has been memset to zero, so not need to clear again */++sg_init_table(sq->sg,n);+sg_set_buf(sq->sg,&xskhdr->hdr,vi->hdr_len);++copied=0;+for(i=1;i<n;++i){+copy=min_t(int,desc->len-copied,PAGE_SIZE-offset);++page=xsk_buff_raw_get_page(pool,addr+copied);++sg_set_page(sq->sg+i,page,copy,offset);+copied+=copy;+if(offset)+offset=0;+}++xskhdr->len=desc->len;+ptr=xdp_to_ptr(&xskhdr->type);++err=virtqueue_add_outbuf(sq->vq,sq->sg,n,ptr,GFP_ATOMIC);+if(unlikely(err))+sq->xsk.last_desc=*desc;+else+sq->xsk.hdr_con++;++returnerr;+}++staticboolvirtnet_xsk_dev_is_full(structsend_queue*sq)+{+if(sq->vq->num_free<2+MAX_SKB_FRAGS)+returntrue;++if(sq->xsk.hdr_con==sq->xsk.hdr_pro)+returntrue;++returnfalse;+}++staticintvirtnet_xsk_xmit_zc(structsend_queue*sq,+structxsk_buff_pool*pool,unsignedintbudget)+{+structxdp_descdesc;+interr,packet=0;+intret=-EAGAIN;++if(sq->xsk.last_desc.addr){+err=virtnet_xsk_xmit(sq,pool,&sq->xsk.last_desc);+if(unlikely(err))+return-EBUSY;++++packet;+sq->xsk.last_desc.addr=0;+}++while(budget-->0){+if(virtnet_xsk_dev_is_full(sq)){+ret=-EBUSY;+break;+}++if(!xsk_tx_peek_desc(pool,&desc)){+/* done */+ret=0;+break;+}++err=virtnet_xsk_xmit(sq,pool,&desc);+if(unlikely(err)){+ret=-EBUSY;+break;+}++++packet;+}++if(packet){+xsk_tx_release(pool);++if(virtqueue_kick_prepare(sq->vq)&&virtqueue_notify(sq->vq)){+u64_stats_update_begin(&sq->stats.syncp);+sq->stats.kicks++;+u64_stats_update_end(&sq->stats.syncp);+}+}++returnret;+}++staticintvirtnet_xsk_run(structsend_queue*sq,+structxsk_buff_pool*pool,intbudget)+{+interr,ret=0;+unsignedint_packets=0;+unsignedint_bytes=0;++sq->xsk.wait_slot=false;++if(test_and_clear_bit(VIRTNET_STATE_XSK_TIMER,&sq->xsk.state))+hrtimer_try_to_cancel(&sq->xsk.timer);++__free_old_xmit_ptr(sq,true,false,&_packets,&_bytes);++err=virtnet_xsk_xmit_zc(sq,pool,xsk_budget);+if(!err){+structxdp_descdesc;++clear_bit(VIRTNET_STATE_XSK_WAKEUP,&sq->xsk.state);+xsk_set_tx_need_wakeup(pool);++/* Race breaker. If new is coming after last xmit+*butbeforeflagchange+*/++if(!xsk_tx_peek_desc(pool,&desc))+gotoend;++set_bit(VIRTNET_STATE_XSK_WAKEUP,&sq->xsk.state);+xsk_clear_tx_need_wakeup(pool);++sq->xsk.last_desc=desc;+ret=budget;+gotoend;+}++xsk_clear_tx_need_wakeup(pool);++if(err==-EAGAIN){+ret=budget;+gotoend;+}++/* -EBUSY: wait tx ring avali.+*bytxinterruptorrxinterruptorstart_xmitortimer+*/++__free_old_xmit_ptr(sq,true,false,&_packets,&_bytes);++if(!virtnet_xsk_dev_is_full(sq)){+ret=budget;+gotoend;+}++sq->xsk.wait_slot=true;++if(xsk_check_timeout){+hrtimer_start(&sq->xsk.timer,+ns_to_ktime(xsk_check_timeout*1000),+HRTIMER_MODE_REL_PINNED);++set_bit(VIRTNET_STATE_XSK_TIMER,&sq->xsk.state);+}++virtnet_sq_stop_check(sq,true);++end:+returnret;+}++staticintvirtnet_xsk_wakeup(structnet_device*dev,u32qid,u32flag)+{+structvirtnet_info*vi=netdev_priv(dev);+structsend_queue*sq;+structxsk_buff_pool*pool;+structnetdev_queue*txq;+intwork=0;++if(!netif_running(dev))+return-ENETDOWN;++if(qid>=vi->curr_queue_pairs)+return-EINVAL;++sq=&vi->sq[qid];++rcu_read_lock();++pool=rcu_dereference(sq->xsk.pool);+if(!pool)+gotoend;++if(test_and_set_bit(VIRTNET_STATE_XSK_WAKEUP,&sq->xsk.state))+gotoend;++txq=netdev_get_tx_queue(dev,qid);++local_bh_disable();+__netif_tx_lock(txq,raw_smp_processor_id());++work=virtnet_xsk_run(sq,pool,xsk_budget);++__netif_tx_unlock(txq);+local_bh_enable();++if(work==xsk_budget)+virtqueue_napi_schedule(&sq->napi,sq->vq);++end:+rcu_read_unlock();+return0;+}+staticintvirtnet_get_phys_port_name(structnet_device*dev,char*buf,size_tlen){
@@ -2624,6 +3048,7 @@ static int virtnet_set_features(struct net_device *dev,.ndo_vlan_rx_kill_vid=virtnet_vlan_rx_kill_vid,.ndo_bpf=virtnet_xdp,.ndo_xdp_xmit=virtnet_xdp_xmit,+.ndo_xsk_wakeup=virtnet_xsk_wakeup,.ndo_features_check=passthru_features_check,.ndo_get_phys_port_name=virtnet_get_phys_port_name,.ndo_set_features=virtnet_set_features,
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-05 09:13:31
For some drivers, such as virtio-net, we do not configure dma when
binding xsk. We will get the page when sending.
This patch participates in a field need_dma during the setup pool. If
the device does not use dma, this value should be set to false.
And a function xsk_buff_raw_get_page is added to get the page based on
addr in drv.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
include/linux/netdevice.h | 1 +
include/net/xdp_sock_drv.h | 10 ++++++++++
include/net/xsk_buff_pool.h | 1 +
net/xdp/xsk_buff_pool.c | 10 +++++++++-
4 files changed, 21 insertions(+), 1 deletion(-)
@@ -167,12 +167,13 @@ static int __xp_assign_dev(struct xsk_buff_pool *pool,bpf.command=XDP_SETUP_XSK_POOL;bpf.xsk.pool=pool;bpf.xsk.queue_id=queue_id;+bpf.xsk.need_dma=true;err=netdev->netdev_ops->ndo_bpf(netdev,&bpf);if(err)gotoerr_unreg_pool;-if(!pool->dma_pages){+if(bpf.xsk.need_dma&&!pool->dma_pages){WARN(1,"Driver did not DMA map zero-copy buffers");err=-EINVAL;gotoerr_unreg_xsk;
From: Jason Wang <hidden> Date: 2021-01-05 09:34:07
On 2021/1/5 下午5:11, Xuan Zhuo wrote:
The first patch made some adjustments to xsk.
Thanks a lot for the work. It's rather interesting.
The second patch itself can be used as an independent patch to solve the problem
that XDP may fail to load when the number of queues is insufficient.
It would be better to send this as a separated patch. Several people
asked for this before.
The third to last patch implements support for xsk in virtio-net.
A practical problem with virtio is that tx interrupts are not very reliable.
There will always be some missing or delayed tx interrupts. So I specially added
a point timer to solve this problem. Of course, considering performance issues,
The timer only triggers when the ring of the network card is full.
This is sub-optimal. We need figure out the root cause. We don't meet
such issue before.
Several questions:
- is tx interrupt enabled?
- can you still see the issue if you disable event index?
- what's backend did you use? qemu or vhost(user)?
Regarding the issue of virtio-net supporting xsk's zero copy rx, I am also
developing it, but I found that the modification may be relatively large, so I
consider this patch set to be separated from the code related to xsk zero copy
rx.
That's fine, but a question here.
How is the multieuque being handled here. I'm asking since there's no
programmable filters/directors support in virtio spec now.
Thanks
Xuan Zhuo (5):
xsk: support get page for drv
virtio-net: support XDP_TX when not more queues
virtio-net, xsk: distinguish XDP_TX and XSK XMIT ctx
xsk, virtio-net: prepare for support xsk
virtio-net, xsk: virtio-net support xsk zero copy tx
drivers/net/virtio_net.c | 643 +++++++++++++++++++++++++++++++++++++++-----
include/linux/netdevice.h | 1 +
include/net/xdp_sock_drv.h | 10 +
include/net/xsk_buff_pool.h | 1 +
net/xdp/xsk_buff_pool.c | 10 +-
5 files changed, 597 insertions(+), 68 deletions(-)
--
1.8.3.1
From: "Michael S. Tsirkin" <mst@redhat.com> Date: 2021-01-05 12:27:19
On Tue, Jan 05, 2021 at 05:11:38PM +0800, Xuan Zhuo wrote:
The first patch made some adjustments to xsk.
The second patch itself can be used as an independent patch to solve the problem
that XDP may fail to load when the number of queues is insufficient.
The third to last patch implements support for xsk in virtio-net.
A practical problem with virtio is that tx interrupts are not very reliable.
There will always be some missing or delayed tx interrupts.
Would appreciate a bit more data on this one. Is this a host bug? Device bug?
Can we limit the work around somehow?
So I specially added
a point timer to solve this problem. Of course, considering performance issues,
The timer only triggers when the ring of the network card is full.
Regarding the issue of virtio-net supporting xsk's zero copy rx, I am also
developing it, but I found that the modification may be relatively large, so I
consider this patch set to be separated from the code related to xsk zero copy
rx.
Xuan Zhuo (5):
xsk: support get page for drv
virtio-net: support XDP_TX when not more queues
virtio-net, xsk: distinguish XDP_TX and XSK XMIT ctx
xsk, virtio-net: prepare for support xsk
virtio-net, xsk: virtio-net support xsk zero copy tx
drivers/net/virtio_net.c | 643 +++++++++++++++++++++++++++++++++++++++-----
include/linux/netdevice.h | 1 +
include/net/xdp_sock_drv.h | 10 +
include/net/xsk_buff_pool.h | 1 +
net/xdp/xsk_buff_pool.c | 10 +-
5 files changed, 597 insertions(+), 68 deletions(-)
--
1.8.3.1
From: "Michael S. Tsirkin" <mst@redhat.com> Date: 2021-01-05 13:23:45
On Tue, Jan 05, 2021 at 05:11:43PM +0800, Xuan Zhuo wrote:
Virtio net support xdp socket.
We should open the module param "napi_tx" for using this feature.
what does this imply exactly?
In fact, various virtio implementations have some problems:
1. The tx interrupt may be lost
2. The tx interrupt may have a relatively large delay
This brings us to several questions:
1. Wakeup wakes up a tx interrupt or directly starts a napi on the
current cpu, which will cause a delay in sending packets.
2. When the tx ring is full, the tx interrupt may be lost or delayed,
resulting in untimely recovery.
I choose to send part of the data directly during wakeup. If the sending
has not been completed, I will start a napi to complete the subsequent
sending work.
Since the possible delay or loss of tx interrupt occurs when the tx ring
is full, I added a timer to solve this problem.
A lost interrupt sounds like a bug somewhere.
Why isn't this device already broken then, even without zero copy?
Won't a full ring stall forever? And won't a significantly delayed
tx interrupt block userspace?
How about putting work arounds were in a separate patch for now,
so it's easier to figure out whether anything in the patch itself
is causing issues?
quoted hunk
The performance of udp sending based on virtio net + xsk is 6 times that
of ordinary kernel udp send.
* xsk_check_timeout: when the dev full or all xsk.hdr used, start timer
to check the xsk.hdr is avail. the unit is us.
* xsk_num_max: the xsk.hdr max num
* xsk_num_percent: the max hdr num be the percent of the virtio ring
size. The real xsk hdr num will the min of xsk_num_max and the percent
of the num of virtio ring
* xsk_budget: the budget for xsk run
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 437 ++++++++++++++++++++++++++++++++++++++++++++++-
1 file changed, 434 insertions(+), 3 deletions(-)
I mean, we call virtqueue_enable_cb_delayed on each start_xmit.
The point is explicitly to reduce the # of tx interrupts,
is this the issue?
+ *
+ * timer for:
+ * 1. recycle the desc.(no check for performance, see below)
+ * 2. check the nic ring is avali. when nic ring is full
+ *
+ * Here, the regular check is performed for dev full. The
+ * application layer must ensure that the number of cq is
+ * sufficient, otherwise there may be insufficient cq in use.
Can't really parse this. cq as in control vq?
quoted hunk
+ *
+ */
+ struct hrtimer timer;
+ } xsk;
};
/* Internal representation of a receive virtqueue */
@@ -267,6 +307,8 @@ static void __free_old_xmit_ptr(struct send_queue *sq, bool in_napi, bool xsk_wakeup, unsigned int *_packets, unsigned int *_bytes); static void free_old_xmit_skbs(struct send_queue *sq, bool in_napi);+static int virtnet_xsk_run(struct send_queue *sq,+ struct xsk_buff_pool *pool, int budget); static bool is_xdp_frame(void *ptr) {
@@ -1595,6 +1676,8 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget) struct virtnet_info *vi = sq->vq->vdev->priv; unsigned int index = vq2txq(sq->vq); struct netdev_queue *txq;+ struct xsk_buff_pool *pool;+ int work = 0; if (unlikely(is_xdp_raw_buffer_queue(vi, index))) { /* We don't need to enable cb for XDP */
@@ -1604,15 +1687,26 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget) txq = netdev_get_tx_queue(vi->dev, index); __netif_tx_lock(txq, raw_smp_processor_id());- free_old_xmit_skbs(sq, true);++ rcu_read_lock();+ pool = rcu_dereference(sq->xsk.pool);+ if (pool) {+ work = virtnet_xsk_run(sq, pool, budget);+ rcu_read_unlock();+ } else {+ rcu_read_unlock();+ free_old_xmit_skbs(sq, true);+ }+ __netif_tx_unlock(txq);- virtqueue_napi_complete(napi, sq->vq, 0);+ if (work < budget)+ virtqueue_napi_complete(napi, sq->vq, 0); if (sq->vq->num_free >= 2 + MAX_SKB_FRAGS) netif_tx_wake_queue(txq);- return 0;+ return work; } static int xmit_skb(struct send_queue *sq, struct sk_buff *skb)
@@ -2560,16 +2654,346 @@ static int virtnet_xdp_set(struct net_device *dev, struct bpf_prog *prog, return err; }+static enum hrtimer_restart virtnet_xsk_timeout(struct hrtimer *timer)+{+ struct send_queue *sq;++ sq = container_of(timer, struct send_queue, xsk.timer);++ clear_bit(VIRTNET_STATE_XSK_TIMER, &sq->xsk.state);++ virtqueue_napi_schedule(&sq->napi, sq->vq);++ return HRTIMER_NORESTART;+}++static int virtnet_xsk_pool_enable(struct net_device *dev,+ struct xsk_buff_pool *pool,+ u16 qid)+{+ struct virtnet_info *vi = netdev_priv(dev);+ struct send_queue *sq = &vi->sq[qid];+ struct virtnet_xsk_hdr *hdr;+ int n, ret = 0;++ if (qid >= dev->real_num_rx_queues || qid >= dev->real_num_tx_queues)+ return -EINVAL;++ if (qid >= vi->curr_queue_pairs)+ return -EINVAL;++ rcu_read_lock();++ ret = -EBUSY;+ if (rcu_dereference(sq->xsk.pool))+ goto end;++ /* check last xsk wait for hdr been free */+ if (rcu_dereference(sq->xsk.hdr))+ goto end;++ n = virtqueue_get_vring_size(sq->vq);+ n = min(xsk_num_max, n * (xsk_num_percent % 100) / 100);++ ret = -ENOMEM;+ hdr = kcalloc(n, sizeof(struct virtnet_xsk_hdr), GFP_ATOMIC);+ if (!hdr)+ goto end;++ memset(&sq->xsk, 0, sizeof(sq->xsk));++ sq->xsk.hdr_pro = n;+ sq->xsk.hdr_n = n;++ hrtimer_init(&sq->xsk.timer, CLOCK_MONOTONIC, HRTIMER_MODE_REL_PINNED);+ sq->xsk.timer.function = virtnet_xsk_timeout;++ rcu_assign_pointer(sq->xsk.pool, pool);+ rcu_assign_pointer(sq->xsk.hdr, hdr);++ ret = 0;+end:+ rcu_read_unlock();++ return ret;+}++static int virtnet_xsk_pool_disable(struct net_device *dev, u16 qid)+{+ struct virtnet_info *vi = netdev_priv(dev);+ struct send_queue *sq = &vi->sq[qid];++ if (qid >= dev->real_num_rx_queues || qid >= dev->real_num_tx_queues)+ return -EINVAL;++ if (qid >= vi->curr_queue_pairs)+ return -EINVAL;++ rcu_assign_pointer(sq->xsk.pool, NULL);++ hrtimer_cancel(&sq->xsk.timer);++ synchronize_rcu(); /* Sync with the XSK wakeup and with NAPI. */++ if (sq->xsk.hdr_pro - sq->xsk.hdr_con == sq->xsk.hdr_n) {+ kfree(sq->xsk.hdr);+ rcu_assign_pointer(sq->xsk.hdr, NULL);+ synchronize_rcu();+ }++ return 0;+}+ static int virtnet_xdp(struct net_device *dev, struct netdev_bpf *xdp) { switch (xdp->command) { case XDP_SETUP_PROG: return virtnet_xdp_set(dev, xdp->prog, xdp->extack);+ case XDP_SETUP_XSK_POOL:+ xdp->xsk.need_dma = false;+ if (xdp->xsk.pool)+ return virtnet_xsk_pool_enable(dev, xdp->xsk.pool,+ xdp->xsk.queue_id);+ else+ return virtnet_xsk_pool_disable(dev, xdp->xsk.queue_id); default: return -EINVAL; } }+static int virtnet_xsk_xmit(struct send_queue *sq, struct xsk_buff_pool *pool,+ struct xdp_desc *desc)+{+ struct virtnet_info *vi = sq->vq->vdev->priv;+ void *data, *ptr;+ struct page *page;+ struct virtnet_xsk_hdr *xskhdr;+ u32 idx, offset, n, i, copy, copied;+ u64 addr;+ int err, m;++ addr = desc->addr;++ data = xsk_buff_raw_get_data(pool, addr);+ offset = offset_in_page(data);++ /* one for hdr, one for the first page */+ n = 2;+ m = desc->len - (PAGE_SIZE - offset);+ if (m > 0) {+ n += m >> PAGE_SHIFT;+ if (m & PAGE_MASK)+ ++n;++ n = min_t(u32, n, ARRAY_SIZE(sq->sg));+ }++ idx = sq->xsk.hdr_con % sq->xsk.hdr_n;+ xskhdr = &sq->xsk.hdr[idx];++ /* xskhdr->hdr has been memset to zero, so not need to clear again */++ sg_init_table(sq->sg, n);+ sg_set_buf(sq->sg, &xskhdr->hdr, vi->hdr_len);++ copied = 0;+ for (i = 1; i < n; ++i) {+ copy = min_t(int, desc->len - copied, PAGE_SIZE - offset);++ page = xsk_buff_raw_get_page(pool, addr + copied);++ sg_set_page(sq->sg + i, page, copy, offset);+ copied += copy;+ if (offset)+ offset = 0;+ }++ xskhdr->len = desc->len;+ ptr = xdp_to_ptr(&xskhdr->type);++ err = virtqueue_add_outbuf(sq->vq, sq->sg, n, ptr, GFP_ATOMIC);+ if (unlikely(err))+ sq->xsk.last_desc = *desc;+ else+ sq->xsk.hdr_con++;++ return err;+}++static bool virtnet_xsk_dev_is_full(struct send_queue *sq)+{+ if (sq->vq->num_free < 2 + MAX_SKB_FRAGS)+ return true;++ if (sq->xsk.hdr_con == sq->xsk.hdr_pro)+ return true;++ return false;+}++static int virtnet_xsk_xmit_zc(struct send_queue *sq,+ struct xsk_buff_pool *pool, unsigned int budget)+{+ struct xdp_desc desc;+ int err, packet = 0;+ int ret = -EAGAIN;++ if (sq->xsk.last_desc.addr) {+ err = virtnet_xsk_xmit(sq, pool, &sq->xsk.last_desc);+ if (unlikely(err))+ return -EBUSY;++ ++packet;+ sq->xsk.last_desc.addr = 0;+ }++ while (budget-- > 0) {+ if (virtnet_xsk_dev_is_full(sq)) {+ ret = -EBUSY;+ break;+ }++ if (!xsk_tx_peek_desc(pool, &desc)) {+ /* done */+ ret = 0;+ break;+ }++ err = virtnet_xsk_xmit(sq, pool, &desc);+ if (unlikely(err)) {+ ret = -EBUSY;+ break;+ }++ ++packet;+ }++ if (packet) {+ xsk_tx_release(pool);++ if (virtqueue_kick_prepare(sq->vq) && virtqueue_notify(sq->vq)) {+ u64_stats_update_begin(&sq->stats.syncp);+ sq->stats.kicks++;+ u64_stats_update_end(&sq->stats.syncp);+ }+ }++ return ret;+}++static int virtnet_xsk_run(struct send_queue *sq,+ struct xsk_buff_pool *pool, int budget)+{+ int err, ret = 0;+ unsigned int _packets = 0;+ unsigned int _bytes = 0;++ sq->xsk.wait_slot = false;++ if (test_and_clear_bit(VIRTNET_STATE_XSK_TIMER, &sq->xsk.state))+ hrtimer_try_to_cancel(&sq->xsk.timer);++ __free_old_xmit_ptr(sq, true, false, &_packets, &_bytes);++ err = virtnet_xsk_xmit_zc(sq, pool, xsk_budget);+ if (!err) {+ struct xdp_desc desc;++ clear_bit(VIRTNET_STATE_XSK_WAKEUP, &sq->xsk.state);+ xsk_set_tx_need_wakeup(pool);++ /* Race breaker. If new is coming after last xmit+ * but before flag change+ */
A bit more text explaining the rules for the two bits please.
+
+ if (!xsk_tx_peek_desc(pool, &desc))
+ goto end;
+
+ set_bit(VIRTNET_STATE_XSK_WAKEUP, &sq->xsk.state);
+ xsk_clear_tx_need_wakeup(pool);
+
+ sq->xsk.last_desc = desc;
+ ret = budget;
+ goto end;
+ }
+
+ xsk_clear_tx_need_wakeup(pool);
+
+ if (err == -EAGAIN) {
+ ret = budget;
+ goto end;
+ }
+
+ /* -EBUSY: wait tx ring avali.
+ * by tx interrupt or rx interrupt or start_xmit or timer
@@ -2740,6 +3166,11 @@ static void free_unused_bufs(struct virtnet_info *vi) xdp_return_frame(xtype_got_ptr(xtype)); } }++ n = sq->xsk.hdr_con + sq->xsk.hdr_n;+ n -= sq->xsk.hdr_pro;+ if (n)+ virt_xsk_complete(sq, n, false); } for (i = 0; i < vi->max_queue_pairs; i++) {
From: Dan Carpenter <hidden> Date: 2021-01-05 13:37:03
Hi Xuan,
url: https://github.com/0day-ci/linux/commits/Xuan-Zhuo/virtio-net-support-xdp-socket-zero-copy-xmit/20210105-171505
base: https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs.git master
config: i386-randconfig-m021-20210105 (attached as .config)
compiler: gcc-9 (Debian 9.3.0-15) 9.3.0
If you fix the issue, kindly add following tag as appropriate
Reported-by: kernel test robot <redacted>
Reported-by: Dan Carpenter <redacted>
New smatch warnings:
drivers/net/virtio_net.c:2669 virtnet_xsk_timeout() warn: test_bit() takes a bit number
drivers/net/virtio_net.c:2899 virtnet_xsk_run() warn: test_bit() takes a bit number
drivers/net/virtio_net.c:2982 virtnet_xsk_wakeup() warn: test_bit() takes a bit number
Old smatch warnings:
drivers/net/virtio_net.c:2908 virtnet_xsk_run() warn: test_bit() takes a bit number
drivers/net/virtio_net.c:2918 virtnet_xsk_run() warn: test_bit() takes a bit number
drivers/net/virtio_net.c:2951 virtnet_xsk_run() warn: test_bit() takes a bit number
drivers/net/virtio_net.c:3073 virtnet_config_changed_work() error: uninitialized symbol 'v'.
vim +2669 drivers/net/virtio_net.c
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2663 static enum hrtimer_restart virtnet_xsk_timeout(struct hrtimer *timer)
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2664 {
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2665 struct send_queue *sq;
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2666
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2667 sq = container_of(timer, struct send_queue, xsk.timer);
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2668
265d3cdead3bd6 Xuan Zhuo 2021-01-05 @2669 clear_bit(VIRTNET_STATE_XSK_TIMER, &sq->xsk.state);
This is a double shift bug like BIT(BIT(foo)).
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2670
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2671 virtqueue_napi_schedule(&sq->napi, sq->vq);
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2672
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2673 return HRTIMER_NORESTART;
265d3cdead3bd6 Xuan Zhuo 2021-01-05 2674 }
---
0-DAY CI Kernel Test Service, Intel Corporation
https://lists.01.org/hyperkitty/list/kbuild-all@lists.01.org
From: kernel test robot <hidden> Date: 2021-01-05 13:37:13
Hi Xuan,
Thank you for the patch! Perhaps something to improve:
[auto build test WARNING on ipvs/master]
[also build test WARNING on linus/master v5.11-rc2 next-20210104]
[cannot apply to sparc-next/master]
[If your patch is applied to the wrong git tree, kindly drop us a note.
And when submitting patch, we suggest to use '--base' as documented in
https://git-scm.com/docs/git-format-patch]
url: https://github.com/0day-ci/linux/commits/Xuan-Zhuo/virtio-net-support-xdp-socket-zero-copy-xmit/20210105-171505
base: https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs.git master
config: x86_64-randconfig-s032-20210105 (attached as .config)
compiler: gcc-9 (Debian 9.3.0-15) 9.3.0
reproduce:
# apt-get install sparse
# sparse version: v0.6.3-208-g46a52ca4-dirty
# https://github.com/0day-ci/linux/commit/265d3cdead3bd609bec53e66dfb7c5bf7339f73f
git remote add linux-review https://github.com/0day-ci/linux
git fetch --no-tags linux-review Xuan-Zhuo/virtio-net-support-xdp-socket-zero-copy-xmit/20210105-171505
git checkout 265d3cdead3bd609bec53e66dfb7c5bf7339f73f
# save the attached .config to linux build tree
make W=1 C=1 CF='-fdiagnostic-prefix -D__CHECK_ENDIAN__' ARCH=x86_64
If you fix the issue, kindly add following tag as appropriate
Reported-by: kernel test robot <redacted>
"sparse warnings: (new ones prefixed by >>)"
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-16 03:00:29
XDP socket is an excellent by pass kernel network transmission framework. The
zero copy feature of xsk (XDP socket) needs to be supported by the driver. The
performance of zero copy is very good. mlx5 and intel ixgbe already support this
feature, This patch set allows virtio-net to support xsk's zerocopy xmit
feature.
And xsk's zerocopy rx has made major changes to virtio-net, and I hope to submit
it after this patch set are received.
Compared with other drivers, virtio-net does not directly obtain the dma
address, so I first obtain the xsk page, and then pass the page to virtio.
When recycling the sent packets, we have to distinguish between skb and xdp.
Now we have to distinguish between skb, xdp, xsk. So the second patch solves
this problem first.
The last four patches are used to support xsk zerocopy in virtio-net:
1. support xsk enable/disable
2. realize the function of xsk packet sending
3. implement xsk wakeup callback
4. set xsk completed when packet sent done
---------------- Performance Testing ------------
The udp package tool implemented by the interface of xsk vs sockperf(kernel udp)
for performance testing:
xsk zero copy in virtio-net:
CPU PPS MSGSIZE
28.7% 3833857 64
38.5% 3689491 512
38.9% 2787096 1456
xsk without zero copy in virtio-net:
CPU PPS MSGSIZE
100% 1916747 64
100% 1775988 512
100% 1440054 1456
sockperf:
CPU PPS MSGSIZE
100% 713274 64
100% 701024 512
100% 695832 1456
Xuan Zhuo (7):
xsk: support get page for drv
virtio-net, xsk: distinguish XDP_TX and XSK XMIT ctx
xsk, virtio-net: prepare for support xsk zerocopy xmit
virtio-net, xsk: support xsk enable/disable
virtio-net, xsk: realize the function of xsk packet sending
virtio-net, xsk: implement xsk wakeup callback
virtio-net, xsk: set xsk completed when packet sent done
drivers/net/virtio_net.c | 559 +++++++++++++++++++++++++++++++++++++++-----
include/linux/netdevice.h | 1 +
include/net/xdp_sock_drv.h | 10 +
include/net/xsk_buff_pool.h | 1 +
net/xdp/xsk_buff_pool.c | 10 +-
5 files changed, 523 insertions(+), 58 deletions(-)
--
1.8.3.1
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-16 03:00:31
For some drivers, such as virtio-net, we do not configure dma when
binding xsk. We will get the page when sending.
This patch participates in a field need_dma during the setup pool. If
the device does not use dma, this value should be set to false.
And a function xsk_buff_raw_get_page is added to get the page based on
addr in drv.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
include/linux/netdevice.h | 1 +
include/net/xdp_sock_drv.h | 10 ++++++++++
include/net/xsk_buff_pool.h | 1 +
net/xdp/xsk_buff_pool.c | 10 +++++++++-
4 files changed, 21 insertions(+), 1 deletion(-)
@@ -166,12 +166,13 @@ static int __xp_assign_dev(struct xsk_buff_pool *pool,bpf.command=XDP_SETUP_XSK_POOL;bpf.xsk.pool=pool;bpf.xsk.queue_id=queue_id;+bpf.xsk.need_dma=true;err=netdev->netdev_ops->ndo_bpf(netdev,&bpf);if(err)gotoerr_unreg_pool;-if(!pool->dma_pages){+if(bpf.xsk.need_dma&&!pool->dma_pages){WARN(1,"Driver did not DMA map zero-copy buffers");err=-EINVAL;gotoerr_unreg_xsk;
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-16 03:00:32
Split function free_old_xmit_skbs, add sub-function __free_old_xmit_ptr,
which is convenient to call with other statistical information, and
supports the parameter 'xsk_wakeup' required for processing xsk.
Use netif stop check as a function virtnet_sq_stop_check, which will be
used when adding xsk support.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 95 ++++++++++++++++++++++++++----------------------
1 file changed, 52 insertions(+), 43 deletions(-)
@@ -375,6 +380,37 @@ static void skb_xmit_done(struct virtqueue *vq)netif_wake_subqueue(vi->dev,vq2txq(vq));}+staticvoidvirtnet_sq_stop_check(structsend_queue*sq,boolin_napi)+{+structvirtnet_info*vi=sq->vq->vdev->priv;+structnet_device*dev=vi->dev;+intqnum=sq-vi->sq;++/* If running out of space, stop queue to avoid getting packets that we+*arethenunabletotransmit.+*Analternativewouldbetoforcequeuinglayertorequeuetheskbby+*returningNETDEV_TX_BUSY.However,NETDEV_TX_BUSYshouldnotbe+*returnedinanormalpathofoperation:itmeansthatdriverisnot+*maintainingtheTXqueuestop/startstateproperly,andcauses+*thestacktodoanon-trivialamountofuselesswork.+*Sincemostpacketsonlytake1or2ringslots,stoppingthequeue+*earlymeans16slotsaretypicallywasted.+*/++if(sq->vq->num_free<2+MAX_SKB_FRAGS){+netif_stop_subqueue(dev,qnum);+if(!sq->napi.weight&&+unlikely(!virtqueue_enable_cb_delayed(sq->vq))){+/* More just got used, free them then recheck. */+free_old_xmit_skbs(sq,in_napi);+if(sq->vq->num_free>=2+MAX_SKB_FRAGS){+netif_start_subqueue(dev,qnum);+virtqueue_disable_cb(sq->vq);+}+}+}+}+#define MRG_CTX_HEADER_SHIFT 22staticvoid*mergeable_len_to_ctx(unsignedinttruesize,unsignedintheadroom)
@@ -522,13 +558,11 @@ static int virtnet_xdp_xmit(struct net_device *dev,structreceive_queue*rq=vi->rq;structbpf_prog*xdp_prog;structsend_queue*sq;-unsignedintlen;intpackets=0;intbytes=0;intdrops=0;intkicks=0;intret,err;-void*ptr;inti;/* Only allow ndo_xdp_xmit if XDP is loaded on dev, as this
@@ -546,24 +580,7 @@ static int virtnet_xdp_xmit(struct net_device *dev,gotoout;}-/* Free up any pending old buffers before queueing new ones. */-while((ptr=virtqueue_get_buf(sq->vq,&len))!=NULL){-if(likely(is_xdp_frame(ptr))){-structvirtnet_xdp_type*xtype;-structxdp_frame*frame;--xtype=ptr_to_xtype(ptr);-frame=xtype_get_ptr(xtype);-bytes+=frame->len;-xdp_return_frame(frame);-}else{-structsk_buff*skb=ptr;--bytes+=skb->len;-napi_consume_skb(skb,false);-}-packets++;-}+__free_old_xmit_ptr(sq,false,true,&packets,&bytes);for(i=0;i<n;i++){structxdp_frame*xdpf=frames[i];
@@ -1400,7 +1417,9 @@ static int virtnet_receive(struct receive_queue *rq, int budget,returnstats.packets;}-staticvoidfree_old_xmit_skbs(structsend_queue*sq,boolin_napi)+staticvoid__free_old_xmit_ptr(structsend_queue*sq,boolin_napi,+boolxsk_wakeup,+unsignedint*_packets,unsignedint*_bytes){unsignedintpackets=0;unsignedintbytes=0;
@@ -1434,6 +1453,17 @@ static void free_old_xmit_skbs(struct send_queue *sq, bool in_napi)packets++;}+*_packets=packets;+*_bytes=bytes;+}++staticvoidfree_old_xmit_skbs(structsend_queue*sq,boolin_napi)+{+unsignedintpackets=0;+unsignedintbytes=0;++__free_old_xmit_ptr(sq,in_napi,true,&packets,&bytes);+/* Avoid overhead when no packets have been processed*happenswhencalledspeculativelyfromstart_xmit.*/
@@ -1649,28 +1679,7 @@ static netdev_tx_t start_xmit(struct sk_buff *skb, struct net_device *dev)nf_reset_ct(skb);}-/* If running out of space, stop queue to avoid getting packets that we-*arethenunabletotransmit.-*Analternativewouldbetoforcequeuinglayertorequeuetheskbby-*returningNETDEV_TX_BUSY.However,NETDEV_TX_BUSYshouldnotbe-*returnedinanormalpathofoperation:itmeansthatdriverisnot-*maintainingtheTXqueuestop/startstateproperly,andcauses-*thestacktodoanon-trivialamountofuselesswork.-*Sincemostpacketsonlytake1or2ringslots,stoppingthequeue-*earlymeans16slotsaretypicallywasted.-*/-if(sq->vq->num_free<2+MAX_SKB_FRAGS){-netif_stop_subqueue(dev,qnum);-if(!use_napi&&-unlikely(!virtqueue_enable_cb_delayed(sq->vq))){-/* More just got used, free them then recheck. */-free_old_xmit_skbs(sq,false);-if(sq->vq->num_free>=2+MAX_SKB_FRAGS){-netif_start_subqueue(dev,qnum);-virtqueue_disable_cb(sq->vq);-}-}-}+virtnet_sq_stop_check(sq,false);if(kick||netif_xmit_stopped(txq)){if(virtqueue_kick_prepare(sq->vq)&&virtqueue_notify(sq->vq)){
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-16 03:00:32
Since I did not find an interface to directly notify virtio to generate
a tx interrupt, I sent some data to trigger a new tx interrupt.
Another advantage of this is that the transmission delay will be
relatively small, and there is no need to wait for the tx interrupt to
start softirq.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 51 ++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 51 insertions(+)
@@ -2841,6 +2841,56 @@ static int virtnet_xsk_run(struct send_queue *sq,returnret;}+staticintvirtnet_xsk_wakeup(structnet_device*dev,u32qid,u32flag)+{+structvirtnet_info*vi=netdev_priv(dev);+structsend_queue*sq;+structxsk_buff_pool*pool;+structnetdev_queue*txq;++if(!netif_running(dev))+return-ENETDOWN;++if(qid>=vi->curr_queue_pairs)+return-EINVAL;++sq=&vi->sq[qid];++rcu_read_lock();++pool=rcu_dereference(sq->xsk.pool);+if(!pool)+gotoend;++if(test_and_set_bit(VIRTNET_STATE_XSK_WAKEUP,&sq->xsk.state))+gotoend;++txq=netdev_get_tx_queue(dev,qid);++local_bh_disable();+__netif_tx_lock(txq,raw_smp_processor_id());++/* Send part of the package directly to reduce the delay in sending the+*package,andthiscanactivelytriggerthetxinterrupts.+*+*Ifthepackageisnotprocessed,thencontinueprocessinginthe+*subsequenttxinterrupt(virtnet_poll_tx).+*+*Ifnopacketissentout,theringofthedeviceisfull.Inthis+*case,wewillstillgetatxinterruptresponse.Thenwewilldeal+*withthesubsequentpacketsendingwork.+*/++virtnet_xsk_run(sq,pool,xsk_budget);++__netif_tx_unlock(txq);+local_bh_enable();++end:+rcu_read_unlock();+return0;+}+staticintvirtnet_get_phys_port_name(structnet_device*dev,char*buf,size_tlen){
@@ -2895,6 +2945,7 @@ static int virtnet_set_features(struct net_device *dev,.ndo_vlan_rx_kill_vid=virtnet_vlan_rx_kill_vid,.ndo_bpf=virtnet_xdp,.ndo_xdp_xmit=virtnet_xdp_xmit,+.ndo_xsk_wakeup=virtnet_xsk_wakeup,.ndo_features_check=passthru_features_check,.ndo_get_phys_port_name=virtnet_get_phys_port_name,.ndo_set_features=virtnet_set_features,
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-16 03:00:32
virtnet_xsk_run will be called in the tx interrupt handling function
virtnet_poll_tx.
The sending process gets desc from the xsk tx queue, and assembles it to
send the data.
Compared with other drivers, a special place is that the page of the
data in xsk is used here instead of the dma address. Because the virtio
interface does not use the dma address.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 200 ++++++++++++++++++++++++++++++++++++++++++++++-
1 file changed, 197 insertions(+), 3 deletions(-)
@@ -1590,6 +1597,8 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget)structvirtnet_info*vi=sq->vq->vdev->priv;unsignedintindex=vq2txq(sq->vq);structnetdev_queue*txq;+structxsk_buff_pool*pool;+intwork=0;if(unlikely(is_xdp_raw_buffer_queue(vi,index))){/* We don't need to enable cb for XDP */
@@ -1599,15 +1608,26 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget)txq=netdev_get_tx_queue(vi->dev,index);__netif_tx_lock(txq,raw_smp_processor_id());-free_old_xmit_skbs(sq,true);++rcu_read_lock();+pool=rcu_dereference(sq->xsk.pool);+if(pool){+work=virtnet_xsk_run(sq,pool,budget);+rcu_read_unlock();+}else{+rcu_read_unlock();+free_old_xmit_skbs(sq,true);+}+__netif_tx_unlock(txq);-virtqueue_napi_complete(napi,sq->vq,0);+if(work<budget)+virtqueue_napi_complete(napi,sq->vq,0);if(sq->vq->num_free>=2+MAX_SKB_FRAGS)netif_tx_wake_queue(txq);-return0;+returnwork;}staticintxmit_skb(structsend_queue*sq,structsk_buff*skb)
@@ -2647,6 +2667,180 @@ static int virtnet_xdp(struct net_device *dev, struct netdev_bpf *xdp)}}+staticintvirtnet_xsk_xmit(structsend_queue*sq,structxsk_buff_pool*pool,+structxdp_desc*desc)+{+structvirtnet_info*vi=sq->vq->vdev->priv;+void*data,*ptr;+structpage*page;+structvirtnet_xsk_hdr*xskhdr;+u32idx,offset,n,i,copy,copied;+u64addr;+interr,m;++addr=desc->addr;++data=xsk_buff_raw_get_data(pool,addr);+offset=offset_in_page(data);++/* one for hdr, one for the first page */+n=2;+m=desc->len-(PAGE_SIZE-offset);+if(m>0){+n+=m>>PAGE_SHIFT;+if(m&PAGE_MASK)+++n;++n=min_t(u32,n,ARRAY_SIZE(sq->sg));+}++idx=sq->xsk.hdr_con%sq->xsk.hdr_n;+xskhdr=&sq->xsk.hdr[idx];++/* xskhdr->hdr has been memset to zero, so not need to clear again */++sg_init_table(sq->sg,n);+sg_set_buf(sq->sg,&xskhdr->hdr,vi->hdr_len);++copied=0;+for(i=1;i<n;++i){+copy=min_t(int,desc->len-copied,PAGE_SIZE-offset);++page=xsk_buff_raw_get_page(pool,addr+copied);++sg_set_page(sq->sg+i,page,copy,offset);+copied+=copy;+if(offset)+offset=0;+}++xskhdr->len=desc->len;+ptr=xdp_to_ptr(&xskhdr->type);++err=virtqueue_add_outbuf(sq->vq,sq->sg,n,ptr,GFP_ATOMIC);+if(unlikely(err))+sq->xsk.last_desc=*desc;+else+sq->xsk.hdr_con++;++returnerr;+}++staticboolvirtnet_xsk_dev_is_full(structsend_queue*sq)+{+if(sq->vq->num_free<2+MAX_SKB_FRAGS)+returntrue;++if(sq->xsk.hdr_con==sq->xsk.hdr_pro)+returntrue;++returnfalse;+}++staticintvirtnet_xsk_xmit_zc(structsend_queue*sq,+structxsk_buff_pool*pool,unsignedintbudget)+{+structxdp_descdesc;+interr,packet=0;+intret=-EAGAIN;++if(sq->xsk.last_desc.addr){+err=virtnet_xsk_xmit(sq,pool,&sq->xsk.last_desc);+if(unlikely(err))+return-EBUSY;++++packet;+sq->xsk.last_desc.addr=0;+}++while(budget-->0){+if(virtnet_xsk_dev_is_full(sq)){+ret=-EBUSY;+break;+}++if(!xsk_tx_peek_desc(pool,&desc)){+/* done */+ret=0;+break;+}++err=virtnet_xsk_xmit(sq,pool,&desc);+if(unlikely(err)){+ret=-EBUSY;+break;+}++++packet;+}++if(packet){+xsk_tx_release(pool);++if(virtqueue_kick_prepare(sq->vq)&&virtqueue_notify(sq->vq)){+u64_stats_update_begin(&sq->stats.syncp);+sq->stats.kicks++;+u64_stats_update_end(&sq->stats.syncp);+}+}++returnret;+}++staticintvirtnet_xsk_run(structsend_queue*sq,+structxsk_buff_pool*pool,intbudget)+{+interr,ret=0;+unsignedint_packets=0;+unsignedint_bytes=0;++sq->xsk.wait_slot=false;++__free_old_xmit_ptr(sq,true,false,&_packets,&_bytes);++err=virtnet_xsk_xmit_zc(sq,pool,xsk_budget);+if(!err){+structxdp_descdesc;++clear_bit(VIRTNET_STATE_XSK_WAKEUP,&sq->xsk.state);+xsk_set_tx_need_wakeup(pool);++/* Race breaker. If new is coming after last xmit+*butbeforeflagchange+*/++if(!xsk_tx_peek_desc(pool,&desc))+gotoend;++set_bit(VIRTNET_STATE_XSK_WAKEUP,&sq->xsk.state);+xsk_clear_tx_need_wakeup(pool);++sq->xsk.last_desc=desc;+ret=budget;+gotoend;+}++xsk_clear_tx_need_wakeup(pool);++if(err==-EAGAIN){+ret=budget;+gotoend;+}++__free_old_xmit_ptr(sq,true,false,&_packets,&_bytes);++if(!virtnet_xsk_dev_is_full(sq)){+ret=budget;+gotoend;+}++sq->xsk.wait_slot=true;++virtnet_sq_stop_check(sq,true);+end:+returnret;+}+staticintvirtnet_get_phys_port_name(structnet_device*dev,char*buf,size_tlen){
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-16 03:01:01
If support xsk, a new ptr will be recovered during the
process of freeing the old ptr. In order to distinguish between ctx sent
by XDP_TX and ctx sent by xsk, a struct is added here to distinguish
between these two situations. virtnet_xdp_type.type It is used to
distinguish different ctx, and virtnet_xdp_type.offset is used to record
the offset between "true ctx" and virtnet_xdp_type.
The newly added virtnet_xsk_hdr will be used for xsk.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 75 ++++++++++++++++++++++++++++++++++++++----------
1 file changed, 60 insertions(+), 15 deletions(-)
@@ -251,14 +267,19 @@ static bool is_xdp_frame(void *ptr)return(unsignedlong)ptr&VIRTIO_XDP_FLAG;}-staticvoid*xdp_to_ptr(structxdp_frame*ptr)+staticvoid*xdp_to_ptr(structvirtnet_xdp_type*ptr){return(void*)((unsignedlong)ptr|VIRTIO_XDP_FLAG);}-staticstructxdp_frame*ptr_to_xdp(void*ptr)+staticstructvirtnet_xdp_type*ptr_to_xtype(void*ptr)+{+return(structvirtnet_xdp_type*)((unsignedlong)ptr&~VIRTIO_XDP_FLAG);+}++staticvoid*xtype_get_ptr(structvirtnet_xdp_type*xdptype){-return(structxdp_frame*)((unsignedlong)ptr&~VIRTIO_XDP_FLAG);+return(char*)xdptype+xdptype->offset;}/* Converting between virtqueue no. and kernel tx/rx queue no.
@@ -459,11 +480,16 @@ static int __virtnet_xdp_xmit_one(struct virtnet_info *vi,structxdp_frame*xdpf){structvirtio_net_hdr_mrg_rxbuf*hdr;+structvirtnet_xdp_type*xdptype;interr;-if(unlikely(xdpf->headroom<vi->hdr_len))+if(unlikely(xdpf->headroom<vi->hdr_len+sizeof(*xdptype)))return-EOVERFLOW;+xdptype=(structvirtnet_xdp_type*)(xdpf+1);+xdptype->offset=(char*)xdpf-(char*)xdptype;+xdptype->type=XDP_TYPE_TX;+/* Make room for virtqueue hdr (also change xdpf->headroom?) */xdpf->data-=vi->hdr_len;/* Zero header and leave csum up to XDP layers */
@@ -523,8 +549,11 @@ static int virtnet_xdp_xmit(struct net_device *dev,/* Free up any pending old buffers before queueing new ones. */while((ptr=virtqueue_get_buf(sq->vq,&len))!=NULL){if(likely(is_xdp_frame(ptr))){-structxdp_frame*frame=ptr_to_xdp(ptr);+structvirtnet_xdp_type*xtype;+structxdp_frame*frame;+xtype=ptr_to_xtype(ptr);+frame=xtype_get_ptr(xtype);bytes+=frame->len;xdp_return_frame(frame);}else{
@@ -1373,24 +1402,34 @@ static int virtnet_receive(struct receive_queue *rq, int budget,staticvoidfree_old_xmit_skbs(structsend_queue*sq,boolin_napi){-unsignedintlen;unsignedintpackets=0;unsignedintbytes=0;-void*ptr;+unsignedintlen;+structvirtnet_xdp_type*xtype;+structxdp_frame*frame;+structvirtnet_xsk_hdr*xskhdr;+structsk_buff*skb;+void*ptr;while((ptr=virtqueue_get_buf(sq->vq,&len))!=NULL){if(likely(!is_xdp_frame(ptr))){-structsk_buff*skb=ptr;+skb=ptr;pr_debug("Sent skb %p\n",skb);bytes+=skb->len;napi_consume_skb(skb,in_napi);}else{-structxdp_frame*frame=ptr_to_xdp(ptr);+xtype=ptr_to_xtype(ptr);-bytes+=frame->len;-xdp_return_frame(frame);+if(xtype->type==XDP_TYPE_XSK){+xskhdr=(structvirtnet_xsk_hdr*)xtype;+bytes+=xskhdr->len;+}else{+frame=xtype_get_ptr(xtype);+xdp_return_frame(frame);+bytes+=frame->len;+}}packets++;}
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-16 03:01:22
When enable, a certain number of struct virtnet_xsk_hdr is allocated to
save the information of each packet and virtio hdr.This number is the
limit of the received module parameters.
When struct virtnet_xsk_hdr is used up, or the sq->vq->num_free of
virtio-net is too small, it will be considered that the device is busy.
* xsk_num_max: the xsk.hdr max num
* xsk_num_percent: the max hdr num be the percent of the virtio ring
size. The real xsk hdr num will the min of xsk_num_max and the percent
of the num of virtio ring
* xsk_budget: the budget for xsk run
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 97 ++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 97 insertions(+)
@@ -149,6 +158,15 @@ struct send_queue {structvirtnet_sq_statsstats;structnapi_structnapi;++struct{+structxsk_buff_pool__rcu*pool;+structvirtnet_xsk_hdr__rcu*hdr;++u64hdr_con;+u64hdr_pro;+u64hdr_n;+}xsk;};/* Internal representation of a receive virtqueue */
@@ -2540,11 +2558,90 @@ static int virtnet_xdp_set(struct net_device *dev, struct bpf_prog *prog,returnerr;}+staticintvirtnet_xsk_pool_enable(structnet_device*dev,+structxsk_buff_pool*pool,+u16qid)+{+structvirtnet_info*vi=netdev_priv(dev);+structsend_queue*sq=&vi->sq[qid];+structvirtnet_xsk_hdr*hdr;+intn,ret=0;++if(qid>=dev->real_num_rx_queues||qid>=dev->real_num_tx_queues)+return-EINVAL;++if(qid>=vi->curr_queue_pairs)+return-EINVAL;++rcu_read_lock();++ret=-EBUSY;+if(rcu_dereference(sq->xsk.pool))+gotoend;++/* check last xsk wait for hdr been free */+if(rcu_dereference(sq->xsk.hdr))+gotoend;++n=virtqueue_get_vring_size(sq->vq);+n=min(xsk_num_max,n*(xsk_num_percent%100)/100);++ret=-ENOMEM;+hdr=kcalloc(n,sizeof(structvirtnet_xsk_hdr),GFP_ATOMIC);+if(!hdr)+gotoend;++memset(&sq->xsk,0,sizeof(sq->xsk));++sq->xsk.hdr_pro=n;+sq->xsk.hdr_n=n;++rcu_assign_pointer(sq->xsk.pool,pool);+rcu_assign_pointer(sq->xsk.hdr,hdr);++ret=0;+end:+rcu_read_unlock();++returnret;+}++staticintvirtnet_xsk_pool_disable(structnet_device*dev,u16qid)+{+structvirtnet_info*vi=netdev_priv(dev);+structsend_queue*sq=&vi->sq[qid];+structvirtnet_xsk_hdr*hdr=NULL;++if(qid>=dev->real_num_rx_queues||qid>=dev->real_num_tx_queues)+return-EINVAL;++if(qid>=vi->curr_queue_pairs)+return-EINVAL;++rcu_assign_pointer(sq->xsk.pool,NULL);++if(sq->xsk.hdr_pro-sq->xsk.hdr_con==sq->xsk.hdr_n)+hdr=rcu_replace_pointer(sq->xsk.hdr,hdr,true);++synchronize_rcu();/* Sync with the XSK wakeup and with NAPI. */++kfree(hdr);++return0;+}+staticintvirtnet_xdp(structnet_device*dev,structnetdev_bpf*xdp){switch(xdp->command){caseXDP_SETUP_PROG:returnvirtnet_xdp_set(dev,xdp->prog,xdp->extack);+caseXDP_SETUP_XSK_POOL:+xdp->xsk.need_dma=false;+if(xdp->xsk.pool)+returnvirtnet_xsk_pool_enable(dev,xdp->xsk.pool,+xdp->xsk.queue_id);+else+returnvirtnet_xsk_pool_disable(dev,xdp->xsk.queue_id);default:return-EINVAL;}
From: Xuan Zhuo <xuanzhuo@linux.alibaba.com> Date: 2021-01-16 03:01:22
When recycling packets that have been sent, call xsk_tx_completed to
inform xsk which packets have been sent.
If necessary, start napi to process the packets in the xsk queue.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 49 ++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 49 insertions(+)
From: Jakub Kicinski <kuba@kernel.org> Date: 2021-01-16 04:48:27
On Sat, 16 Jan 2021 10:59:26 +0800 Xuan Zhuo wrote:
+ idx = sq->xsk.hdr_con % sq->xsk.hdr_n;
The arguments here are 64 bit, this code will not build on 32 bit
machines:
ERROR: modpost: "__umoddi3" [drivers/net/virtio_net.ko] undefined!
There's also a sparse warning in this patch:
drivers/net/virtio_net.c:2704:16: warning: incorrect type in assignment (different address spaces)
drivers/net/virtio_net.c:2704:16: expected struct virtnet_xsk_hdr *xskhdr
drivers/net/virtio_net.c:2704:16: got struct virtnet_xsk_hdr [noderef] __rcu *
From: Jason Wang <hidden> Date: 2021-01-18 06:30:57
On 2021/1/16 上午10:59, Xuan Zhuo wrote:
XDP socket is an excellent by pass kernel network transmission framework. The
zero copy feature of xsk (XDP socket) needs to be supported by the driver. The
performance of zero copy is very good. mlx5 and intel ixgbe already support this
feature, This patch set allows virtio-net to support xsk's zerocopy xmit
feature.
And xsk's zerocopy rx has made major changes to virtio-net, and I hope to submit
it after this patch set are received.
Compared with other drivers, virtio-net does not directly obtain the dma
address, so I first obtain the xsk page, and then pass the page to virtio.
When recycling the sent packets, we have to distinguish between skb and xdp.
Now we have to distinguish between skb, xdp, xsk. So the second patch solves
this problem first.
The last four patches are used to support xsk zerocopy in virtio-net:
1. support xsk enable/disable
2. realize the function of xsk packet sending
3. implement xsk wakeup callback
4. set xsk completed when packet sent done
---------------- Performance Testing ------------
The udp package tool implemented by the interface of xsk vs sockperf(kernel udp)
for performance testing:
xsk zero copy in virtio-net:
CPU PPS MSGSIZE
28.7% 3833857 64
38.5% 3689491 512
38.9% 2787096 1456
Some questions on the results:
1) What's the setup on the vhost?
2) What's the setup of the mitigation in both host and guest?
3) Any analyze on the possible bottleneck via perf or other tools?
Thanks
xsk without zero copy in virtio-net:
CPU PPS MSGSIZE
100% 1916747 64
100% 1775988 512
100% 1440054 1456
sockperf:
CPU PPS MSGSIZE
100% 713274 64
100% 701024 512
100% 695832 1456
Xuan Zhuo (7):
xsk: support get page for drv
virtio-net, xsk: distinguish XDP_TX and XSK XMIT ctx
xsk, virtio-net: prepare for support xsk zerocopy xmit
virtio-net, xsk: support xsk enable/disable
virtio-net, xsk: realize the function of xsk packet sending
virtio-net, xsk: implement xsk wakeup callback
virtio-net, xsk: set xsk completed when packet sent done
drivers/net/virtio_net.c | 559 +++++++++++++++++++++++++++++++++++++++-----
include/linux/netdevice.h | 1 +
include/net/xdp_sock_drv.h | 10 +
include/net/xsk_buff_pool.h | 1 +
net/xdp/xsk_buff_pool.c | 10 +-
5 files changed, 523 insertions(+), 58 deletions(-)
--
1.8.3.1
From: Jason Wang <hidden> Date: 2021-01-18 06:47:28
On 2021/1/16 上午10:59, Xuan Zhuo wrote:
If support xsk, a new ptr will be recovered during the
process of freeing the old ptr. In order to distinguish between ctx sent
by XDP_TX and ctx sent by xsk, a struct is added here to distinguish
between these two situations. virtnet_xdp_type.type It is used to
distinguish different ctx, and virtnet_xdp_type.offset is used to record
the offset between "true ctx" and virtnet_xdp_type.
The newly added virtnet_xsk_hdr will be used for xsk.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
Any reason that you can't simply encode the type in the pointer itself
as we used to do?
#define VIRTIO_XSK_FLAG BIT(1)
?
@@ -251,14 +267,19 @@ static bool is_xdp_frame(void *ptr)return(unsignedlong)ptr&VIRTIO_XDP_FLAG;}-staticvoid*xdp_to_ptr(structxdp_frame*ptr)+staticvoid*xdp_to_ptr(structvirtnet_xdp_type*ptr){return(void*)((unsignedlong)ptr|VIRTIO_XDP_FLAG);}-staticstructxdp_frame*ptr_to_xdp(void*ptr)+staticstructvirtnet_xdp_type*ptr_to_xtype(void*ptr)+{+return(structvirtnet_xdp_type*)((unsignedlong)ptr&~VIRTIO_XDP_FLAG);+}++staticvoid*xtype_get_ptr(structvirtnet_xdp_type*xdptype){-return(structxdp_frame*)((unsignedlong)ptr&~VIRTIO_XDP_FLAG);+return(char*)xdptype+xdptype->offset;}/* Converting between virtqueue no. and kernel tx/rx queue no.
@@ -459,11 +480,16 @@ static int __virtnet_xdp_xmit_one(struct virtnet_info *vi,structxdp_frame*xdpf){structvirtio_net_hdr_mrg_rxbuf*hdr;+structvirtnet_xdp_type*xdptype;interr;-if(unlikely(xdpf->headroom<vi->hdr_len))+if(unlikely(xdpf->headroom<vi->hdr_len+sizeof(*xdptype)))return-EOVERFLOW;+xdptype=(structvirtnet_xdp_type*)(xdpf+1);+xdptype->offset=(char*)xdpf-(char*)xdptype;+xdptype->type=XDP_TYPE_TX;+/* Make room for virtqueue hdr (also change xdpf->headroom?) */xdpf->data-=vi->hdr_len;/* Zero header and leave csum up to XDP layers */
@@ -523,8 +549,11 @@ static int virtnet_xdp_xmit(struct net_device *dev,/* Free up any pending old buffers before queueing new ones. */while((ptr=virtqueue_get_buf(sq->vq,&len))!=NULL){if(likely(is_xdp_frame(ptr))){-structxdp_frame*frame=ptr_to_xdp(ptr);+structvirtnet_xdp_type*xtype;+structxdp_frame*frame;+xtype=ptr_to_xtype(ptr);+frame=xtype_get_ptr(xtype);bytes+=frame->len;xdp_return_frame(frame);}else{
@@ -1373,24 +1402,34 @@ static int virtnet_receive(struct receive_queue *rq, int budget,staticvoidfree_old_xmit_skbs(structsend_queue*sq,boolin_napi){-unsignedintlen;unsignedintpackets=0;unsignedintbytes=0;-void*ptr;+unsignedintlen;+structvirtnet_xdp_type*xtype;+structxdp_frame*frame;+structvirtnet_xsk_hdr*xskhdr;+structsk_buff*skb;+void*ptr;while((ptr=virtqueue_get_buf(sq->vq,&len))!=NULL){if(likely(!is_xdp_frame(ptr))){-structsk_buff*skb=ptr;+skb=ptr;pr_debug("Sent skb %p\n",skb);bytes+=skb->len;napi_consume_skb(skb,in_napi);}else{-structxdp_frame*frame=ptr_to_xdp(ptr);+xtype=ptr_to_xtype(ptr);-bytes+=frame->len;-xdp_return_frame(frame);+if(xtype->type==XDP_TYPE_XSK){+xskhdr=(structvirtnet_xsk_hdr*)xtype;+bytes+=xskhdr->len;+}else{+frame=xtype_get_ptr(xtype);+xdp_return_frame(frame);+bytes+=frame->len;+}}packets++;}
From: "Michael S. Tsirkin" <mst@redhat.com> Date: 2021-01-18 12:29:15
On Mon, Jan 18, 2021 at 05:10:24PM +0800, Jason Wang wrote:
On 2021/1/16 上午10:59, Xuan Zhuo wrote:
quoted
virtnet_xsk_run will be called in the tx interrupt handling function
virtnet_poll_tx.
The sending process gets desc from the xsk tx queue, and assembles it to
send the data.
Compared with other drivers, a special place is that the page of the
data in xsk is used here instead of the dma address. Because the virtio
interface does not use the dma address.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 200 ++++++++++++++++++++++++++++++++++++++++++++++-
1 file changed, 197 insertions(+), 3 deletions(-)
Please add documentation about the new fields/defines, how are they
accessed, what locking/ordering is in place.
quoted
@@ -284,6 +289,8 @@ static void __free_old_xmit_ptr(struct send_queue *sq, bool in_napi, bool xsk_wakeup, unsigned int *_packets, unsigned int *_bytes); static void free_old_xmit_skbs(struct send_queue *sq, bool in_napi);+static int virtnet_xsk_run(struct send_queue *sq,+ struct xsk_buff_pool *pool, int budget); static bool is_xdp_frame(void *ptr) {
@@ -1590,6 +1597,8 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget) struct virtnet_info *vi = sq->vq->vdev->priv; unsigned int index = vq2txq(sq->vq); struct netdev_queue *txq;+ struct xsk_buff_pool *pool;+ int work = 0; if (unlikely(is_xdp_raw_buffer_queue(vi, index))) { /* We don't need to enable cb for XDP */
@@ -1599,15 +1608,26 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget) txq = netdev_get_tx_queue(vi->dev, index); __netif_tx_lock(txq, raw_smp_processor_id());- free_old_xmit_skbs(sq, true);++ rcu_read_lock();+ pool = rcu_dereference(sq->xsk.pool);+ if (pool) {+ work = virtnet_xsk_run(sq, pool, budget);+ rcu_read_unlock();+ } else {+ rcu_read_unlock();+ free_old_xmit_skbs(sq, true);+ }+ __netif_tx_unlock(txq);- virtqueue_napi_complete(napi, sq->vq, 0);+ if (work < budget)+ virtqueue_napi_complete(napi, sq->vq, 0); if (sq->vq->num_free >= 2 + MAX_SKB_FRAGS) netif_tx_wake_queue(txq);- return 0;+ return work; } static int xmit_skb(struct send_queue *sq, struct sk_buff *skb)
@@ -2647,6 +2667,180 @@ static int virtnet_xdp(struct net_device *dev, struct netdev_bpf *xdp) } }+static int virtnet_xsk_xmit(struct send_queue *sq, struct xsk_buff_pool *pool,+ struct xdp_desc *desc)+{+ struct virtnet_info *vi = sq->vq->vdev->priv;+ void *data, *ptr;+ struct page *page;+ struct virtnet_xsk_hdr *xskhdr;+ u32 idx, offset, n, i, copy, copied;+ u64 addr;+ int err, m;++ addr = desc->addr;++ data = xsk_buff_raw_get_data(pool, addr);+ offset = offset_in_page(data);++ /* one for hdr, one for the first page */+ n = 2;+ m = desc->len - (PAGE_SIZE - offset);+ if (m > 0) {+ n += m >> PAGE_SHIFT;+ if (m & PAGE_MASK)+ ++n;++ n = min_t(u32, n, ARRAY_SIZE(sq->sg));+ }++ idx = sq->xsk.hdr_con % sq->xsk.hdr_n;
I don't understand the reason of the hdr array. It looks to me all of them
are zero and read only from device.
Any reason for not reusing a single hdr for all xdp descriptors? Or maybe
it's time to introduce VIRTIO_NET_F_NO_HDR.
I'm not sure it's worth it, since
- xdp can be enabled/disabled dynamically
- there's intent to add offload support to xdp
quoted
+ xskhdr = &sq->xsk.hdr[idx];
+
+ /* xskhdr->hdr has been memset to zero, so not need to clear again */
+
+ sg_init_table(sq->sg, n);
+ sg_set_buf(sq->sg, &xskhdr->hdr, vi->hdr_len);
+
+ copied = 0;
+ for (i = 1; i < n; ++i) {
+ copy = min_t(int, desc->len - copied, PAGE_SIZE - offset);
+
+ page = xsk_buff_raw_get_page(pool, addr + copied);
+
+ sg_set_page(sq->sg + i, page, copy, offset);
+ copied += copy;
+ if (offset)
+ offset = 0;
+ }
It looks to me we need to terminate the sg:
**
* virtqueue_add_outbuf - expose output buffers to other end
* @vq: the struct virtqueue we're talking about.
* @sg: scatterlist (must be well-formed and terminated!)
From: Jason Wang <hidden> Date: 2021-01-18 20:31:43
On 2021/1/16 上午10:59, Xuan Zhuo wrote:
quoted hunk
virtnet_xsk_run will be called in the tx interrupt handling function
virtnet_poll_tx.
The sending process gets desc from the xsk tx queue, and assembles it to
send the data.
Compared with other drivers, a special place is that the page of the
data in xsk is used here instead of the dma address. Because the virtio
interface does not use the dma address.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 200 ++++++++++++++++++++++++++++++++++++++++++++++-
1 file changed, 197 insertions(+), 3 deletions(-)
@@ -1590,6 +1597,8 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget)structvirtnet_info*vi=sq->vq->vdev->priv;unsignedintindex=vq2txq(sq->vq);structnetdev_queue*txq;+structxsk_buff_pool*pool;+intwork=0;if(unlikely(is_xdp_raw_buffer_queue(vi,index))){/* We don't need to enable cb for XDP */
@@ -1599,15 +1608,26 @@ static int virtnet_poll_tx(struct napi_struct *napi, int budget)txq=netdev_get_tx_queue(vi->dev,index);__netif_tx_lock(txq,raw_smp_processor_id());-free_old_xmit_skbs(sq,true);++rcu_read_lock();+pool=rcu_dereference(sq->xsk.pool);+if(pool){+work=virtnet_xsk_run(sq,pool,budget);+rcu_read_unlock();+}else{+rcu_read_unlock();+free_old_xmit_skbs(sq,true);+}+__netif_tx_unlock(txq);-virtqueue_napi_complete(napi,sq->vq,0);+if(work<budget)+virtqueue_napi_complete(napi,sq->vq,0);if(sq->vq->num_free>=2+MAX_SKB_FRAGS)netif_tx_wake_queue(txq);-return0;+returnwork;}staticintxmit_skb(structsend_queue*sq,structsk_buff*skb)
@@ -2647,6 +2667,180 @@ static int virtnet_xdp(struct net_device *dev, struct netdev_bpf *xdp)}}+staticintvirtnet_xsk_xmit(structsend_queue*sq,structxsk_buff_pool*pool,+structxdp_desc*desc)+{+structvirtnet_info*vi=sq->vq->vdev->priv;+void*data,*ptr;+structpage*page;+structvirtnet_xsk_hdr*xskhdr;+u32idx,offset,n,i,copy,copied;+u64addr;+interr,m;++addr=desc->addr;++data=xsk_buff_raw_get_data(pool,addr);+offset=offset_in_page(data);++/* one for hdr, one for the first page */+n=2;+m=desc->len-(PAGE_SIZE-offset);+if(m>0){+n+=m>>PAGE_SHIFT;+if(m&PAGE_MASK)+++n;++n=min_t(u32,n,ARRAY_SIZE(sq->sg));+}++idx=sq->xsk.hdr_con%sq->xsk.hdr_n;
I don't understand the reason of the hdr array. It looks to me all of
them are zero and read only from device.
Any reason for not reusing a single hdr for all xdp descriptors? Or
maybe it's time to introduce VIRTIO_NET_F_NO_HDR.
+ xskhdr = &sq->xsk.hdr[idx];
+
+ /* xskhdr->hdr has been memset to zero, so not need to clear again */
+
+ sg_init_table(sq->sg, n);
+ sg_set_buf(sq->sg, &xskhdr->hdr, vi->hdr_len);
+
+ copied = 0;
+ for (i = 1; i < n; ++i) {
+ copy = min_t(int, desc->len - copied, PAGE_SIZE - offset);
+
+ page = xsk_buff_raw_get_page(pool, addr + copied);
+
+ sg_set_page(sq->sg + i, page, copy, offset);
+ copied += copy;
+ if (offset)
+ offset = 0;
+ }
It looks to me we need to terminate the sg:
**
* virtqueue_add_outbuf - expose output buffers to other end
* @vq: the struct virtqueue we're talking about.
* @sg: scatterlist (must be well-formed and terminated!)
From: Jason Wang <hidden> Date: 2021-01-19 05:02:04
On 2021/1/16 上午10:59, Xuan Zhuo wrote:
quoted hunk
Since I did not find an interface to directly notify virtio to generate
a tx interrupt, I sent some data to trigger a new tx interrupt.
Another advantage of this is that the transmission delay will be
relatively small, and there is no need to wait for the tx interrupt to
start softirq.
Signed-off-by: Xuan Zhuo <xuanzhuo@linux.alibaba.com>
---
drivers/net/virtio_net.c | 51 ++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 51 insertions(+)
@@ -2841,6 +2841,56 @@ static int virtnet_xsk_run(struct send_queue *sq,returnret;}+staticintvirtnet_xsk_wakeup(structnet_device*dev,u32qid,u32flag)+{+structvirtnet_info*vi=netdev_priv(dev);+structsend_queue*sq;+structxsk_buff_pool*pool;+structnetdev_queue*txq;++if(!netif_running(dev))+return-ENETDOWN;++if(qid>=vi->curr_queue_pairs)+return-EINVAL;++sq=&vi->sq[qid];++rcu_read_lock();++pool=rcu_dereference(sq->xsk.pool);+if(!pool)+gotoend;++if(test_and_set_bit(VIRTNET_STATE_XSK_WAKEUP,&sq->xsk.state))+gotoend;++txq=netdev_get_tx_queue(dev,qid);++local_bh_disable();+__netif_tx_lock(txq,raw_smp_processor_id());
You can use __netif_tx_lock_bh().
Thanks
quoted hunk
+
+ /* Send part of the package directly to reduce the delay in sending the
+ * package, and this can actively trigger the tx interrupts.
+ *
+ * If the package is not processed, then continue processing in the
+ * subsequent tx interrupt(virtnet_poll_tx).
+ *
+ * If no packet is sent out, the ring of the device is full. In this
+ * case, we will still get a tx interrupt response. Then we will deal
+ * with the subsequent packet sending work.
+ */
+
+ virtnet_xsk_run(sq, pool, xsk_budget);
+
+ __netif_tx_unlock(txq);
+ local_bh_enable();
+
+end:
+ rcu_read_unlock();
+ return 0;
+}
+
static int virtnet_get_phys_port_name(struct net_device *dev, char *buf,
size_t len)
{