[RFC net-next 11/15] xdp: Copy provider buffers on pass and redirect
flat view
From: Björn Töpel <bjorn@kernel.org>
Date: 2026-10-02 19:02:00
Also in:
bpf, io-uring, linux-doc, lkml
Subsystem:
bpf [general] (safe dynamic programs and tools), bpf [networking] (tcx & tc bpf, sock_addr), networking [general], the rest, xdp (express data path) · Maintainers:
Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko, Eduard Zingerman, Kumar Kartikeya Dwivedi, "David S. Miller", Eric Dumazet, Jakub Kicinski, Paolo Abeni, Linus Torvalds, Jesper Dangaard Brouer, John Fastabend
A provider buffer must go back to its page pool from the NAPI poll that allocated it. Userspace gets UMEM ownership back through the FILL ring, without a reference count. An skb or xdp_frame can outlive the poll, so it cannot hold a provider buffer. Copy the packet, fragments included, into kernel pages on XDP_PASS and on redirects to targets other than XSKMAP. On redirect, split the copy from the release: return the provider buffer only after the target accepts the copy. The classic zero-copy redirect path is unchanged. The copy and the BPF fragment helpers get a fragment's address from the new xdp_frag_address(). It works for pages and for provider net_iovs; skb_frag_address() works only for pages. The next patch handles delivery to an XSKMAP socket. Signed-off-by: Björn Töpel <bjorn@kernel.org> --- include/net/xdp.h | 8 +++- net/core/dev.h | 5 +++ net/core/filter.c | 28 +++++++++++--- net/core/xdp.c | 98 ++++++++++++++++++++++++++++++++++++++++++++--- 4 files changed, 126 insertions(+), 13 deletions(-)
diff --git a/include/net/xdp.h b/include/net/xdp.h
index 80931490ca3a..4d7040debcca 100644
--- a/include/net/xdp.h
+++ b/include/net/xdp.h@@ -420,11 +420,17 @@ xdp_update_skb_frags_info(struct sk_buff *skb, u8 nr_frags, skb->unreadable |= !!(xdp_flags & XDP_FLAGS_FRAGS_UNREADABLE); } +/* Page and readable provider fragments alike resolve through their netmem. */ +static inline void *xdp_frag_address(const skb_frag_t *frag) +{ + return netmem_address(skb_frag_netmem(frag)) + skb_frag_off(frag); +} + /* Avoids inlining WARN macro in fast-path */ void xdp_warn(const char *msg, const char *func, const int line); #define XDP_WARN(msg) xdp_warn(msg, __func__, __LINE__) -struct sk_buff *xdp_build_skb_from_buff(const struct xdp_buff *xdp); +struct sk_buff *xdp_build_skb_from_buff(struct xdp_buff *xdp); struct sk_buff *xdp_build_skb_from_zc(struct xdp_buff *xdp); struct xdp_frame *xdp_convert_zc_to_xdp_frame(struct xdp_buff *xdp); struct sk_buff *__xdp_build_skb_from_frame(struct xdp_frame *xdpf,
diff --git a/net/core/dev.h b/net/core/dev.h
index 4793a994f7f1..6598a4c5cfec 100644
--- a/net/core/dev.h
+++ b/net/core/dev.h@@ -13,6 +13,11 @@ struct netlink_ext_ack; struct netdev_queue_config; struct cpumask; struct pp_memory_provider_params; +struct xdp_buff; +struct xdp_frame; + +struct xdp_frame *xdp_copy_zc_to_xdp_frame(struct xdp_buff *xdp); +void xdp_release_zc_buff(struct xdp_buff *xdp); /* Random bits of netdevice that don't need to be exposed */ #define FLOW_LIMIT_HISTORY (1 << 7) /* must be ^2 and !overflow buckets */
diff --git a/net/core/filter.c b/net/core/filter.c
index 70dc621672f2..23b3923c0869 100644
--- a/net/core/filter.c
+++ b/net/core/filter.c@@ -4221,7 +4221,7 @@ void bpf_xdp_copy_buf(struct xdp_buff *xdp, unsigned long off, break; ptr_off += ptr_len; - ptr_buf = skb_frag_address(next_frag); + ptr_buf = xdp_frag_address(next_frag); ptr_len = skb_frag_size(next_frag); next_frag++; }
@@ -4249,7 +4249,7 @@ void *bpf_xdp_pointer(struct xdp_buff *xdp, u32 offset, u32 len) u32 frag_size = skb_frag_size(&sinfo->frags[i]); if (offset < frag_size) { - addr = skb_frag_address(&sinfo->frags[i]); + addr = xdp_frag_address(&sinfo->frags[i]); size = frag_size; break; }
@@ -4339,7 +4339,7 @@ static int bpf_xdp_frags_increase_tail(struct xdp_buff *xdp, int offset) if (unlikely(offset > tailroom)) return -EINVAL; - memset(skb_frag_address(frag) + skb_frag_size(frag), 0, offset); + memset(xdp_frag_address(frag) + skb_frag_size(frag), 0, offset); skb_frag_size_add(frag, offset); sinfo->xdp_frags_size += offset; if (rxq->mem.type == MEM_TYPE_XSK_BUFF_POOL)
@@ -4684,12 +4684,28 @@ int xdp_do_redirect(struct net_device *dev, struct xdp_buff *xdp, { struct bpf_redirect_info *ri = bpf_net_ctx_get_ri(); enum bpf_map_type map_type = ri->map_type; + struct xdp_frame *xdpf; + int err; if (map_type == BPF_MAP_TYPE_XSKMAP) return __xdp_do_redirect_xsk(ri, dev, xdp, xdp_prog); - return __xdp_do_redirect_frame(ri, dev, xdp_convert_buff_to_frame(xdp), - xdp_prog); + if (!xdp_buff_has_netmem(xdp)) + return __xdp_do_redirect_frame(ri, dev, + xdp_convert_buff_to_frame(xdp), + xdp_prog); + + /* Return provider buffers only after the target accepts the copy. */ + xdpf = xdp_copy_zc_to_xdp_frame(xdp); + err = __xdp_do_redirect_frame(ri, dev, xdpf, xdp_prog); + if (err) { + if (xdpf) + xdp_return_frame_rx_napi(xdpf); + return err; + } + + xdp_release_zc_buff(xdp); + return 0; } EXPORT_SYMBOL_GPL(xdp_do_redirect);
@@ -12677,7 +12693,7 @@ __bpf_kfunc int bpf_xdp_pull_data(struct xdp_md *x, u32 len) skb_frag_t *frag = &sinfo->frags[i]; u32 shrink = min_t(u32, delta, skb_frag_size(frag)); - memcpy(xdp->data_end, skb_frag_address(frag), shrink); + memcpy(xdp->data_end, xdp_frag_address(frag), shrink); xdp->data_end += shrink; sinfo->xdp_frags_size -= shrink;
diff --git a/net/core/xdp.c b/net/core/xdp.c
index 386240bd24c9..b7f16f44dac2 100644
--- a/net/core/xdp.c
+++ b/net/core/xdp.c@@ -23,6 +23,8 @@ #include <trace/events/xdp.h> #include <net/xdp_sock_drv.h> +#include "dev.h" + #define REG_STATE_NEW 0x0 #define REG_STATE_REGISTERED 0x1 #define REG_STATE_UNREGISTERED 0x2
@@ -559,7 +561,8 @@ void xdp_return_buff(struct xdp_buff *xdp) xdp->rxq->mem.type, true, xdp); out: - __xdp_return(virt_to_netmem(xdp->data), xdp->rxq->mem.type, true, xdp); + __xdp_return(xdp_buff_get_netmem(xdp), xdp->rxq->mem.type, true, + xdp); } EXPORT_SYMBOL_GPL(xdp_return_buff);
@@ -573,7 +576,56 @@ void xdp_attachment_setup(struct xdp_attachment_info *info, } EXPORT_SYMBOL_GPL(xdp_attachment_setup); -struct xdp_frame *xdp_convert_zc_to_xdp_frame(struct xdp_buff *xdp) +static bool xdp_copy_frags_to_frame(struct xdp_frame *xdpf, + const struct xdp_buff *xdp) +{ + const struct skb_shared_info *xinfo; + struct skb_shared_info *sinfo; + u32 i, nr_frags; + + xinfo = xdp_get_shared_info_from_buff(xdp); + nr_frags = xinfo->nr_frags; + + sinfo = xdp_get_shared_info_from_frame(xdpf); + memset(sinfo, 0, sizeof(*sinfo)); + + for (i = 0; i < nr_frags; i++) { + const skb_frag_t *frag = &xinfo->frags[i]; + u32 len = skb_frag_size(frag); + struct page *page; + + page = dev_alloc_page(); + if (!page) + goto err; + + memcpy(page_address(page), xdp_frag_address(frag), len); + __skb_fill_page_desc_noacc(sinfo, i, page, 0, len); + if (page_is_pfmemalloc(page)) + xdpf->flags |= XDP_FLAGS_FRAGS_PF_MEMALLOC; + } + + sinfo->nr_frags = nr_frags; + sinfo->xdp_frags_size = xinfo->xdp_frags_size; + sinfo->xdp_frags_truesize = nr_frags * PAGE_SIZE; + return true; + +err: + while (i--) + put_page(skb_frag_page(&sinfo->frags[i])); + + return false; +} + +/** + * xdp_copy_zc_to_xdp_frame - copy a zero-copy buff into a page-backed frame + * @xdp: zero-copy &xdp_buff to copy + * + * The source keeps its buffers, so a caller that cannot hand the frame on + * must free the frame and may still return the source to its owner. + * + * Return: new frame on success, %NULL on failure. + */ +struct xdp_frame *xdp_copy_zc_to_xdp_frame(struct xdp_buff *xdp) { unsigned int metasize, totsize; void *addr, *data_to_copy;
@@ -607,7 +659,34 @@ struct xdp_frame *xdp_convert_zc_to_xdp_frame(struct xdp_buff *xdp) xdpf->frame_sz = PAGE_SIZE; xdpf->mem_type = MEM_TYPE_PAGE_ORDER0; - xsk_buff_free(xdp); + /* Classic zero-copy frames keep only the linear part. */ + if (xdp_buff_has_netmem(xdp) && xdp_buff_has_frags(xdp)) { + xdpf->flags = XDP_FLAGS_HAS_FRAGS; + if (!xdp_copy_frags_to_frame(xdpf, xdp)) { + put_page(page); + return NULL; + } + } + + return xdpf; +} + +void xdp_release_zc_buff(struct xdp_buff *xdp) +{ + if (xdp->rxq->mem.type == MEM_TYPE_XSK_BUFF_POOL) + xsk_buff_free(xdp); + else + xdp_return_buff(xdp); +} + +struct xdp_frame *xdp_convert_zc_to_xdp_frame(struct xdp_buff *xdp) +{ + struct xdp_frame *xdpf; + + xdpf = xdp_copy_zc_to_xdp_frame(xdp); + if (xdpf) + xdp_release_zc_buff(xdp); + return xdpf; } EXPORT_SYMBOL_GPL(xdp_convert_zc_to_xdp_frame);
@@ -627,10 +706,11 @@ EXPORT_SYMBOL_GPL(xdp_warn); * &xdp_buff: allocate an skb head from the NAPI percpu cache, initialize * skb data pointers and offsets, set the recycle bit if the buff is * PP-backed, Rx queue index, protocol and update frags info. + * Provider-backed netmem is copied into kernel memory and released. * * Return: new &sk_buff on success, %NULL on error. */ -struct sk_buff *xdp_build_skb_from_buff(const struct xdp_buff *xdp) +struct sk_buff *xdp_build_skb_from_buff(struct xdp_buff *xdp) { const struct xdp_rxq_info *rxq = xdp->rxq; const struct skb_shared_info *sinfo;
@@ -638,6 +718,9 @@ struct sk_buff *xdp_build_skb_from_buff(const struct xdp_buff *xdp) u32 nr_frags = 0; int metalen; + if (xdp_buff_has_netmem(xdp)) + return xdp_build_skb_from_zc(xdp); + if (unlikely(xdp_buff_has_frags(xdp))) { sinfo = xdp_get_shared_info_from_buff(xdp); nr_frags = sinfo->nr_frags;
@@ -708,7 +791,7 @@ static noinline bool xdp_copy_frags_from_zc(struct sk_buff *skb, return false; } - memcpy(page_address(page) + offset, skb_frag_address(frag), + memcpy(page_address(page) + offset, xdp_frag_address(frag), len); __skb_fill_page_desc_noacc(sinfo, i, page, offset, len);
@@ -787,7 +870,10 @@ struct sk_buff *xdp_build_skb_from_zc(struct xdp_buff *xdp) goto out; } - xsk_buff_free(xdp); + if (xdp->rxq->mem.type == MEM_TYPE_XSK_BUFF_POOL) + xsk_buff_free(xdp); + else + xdp_return_buff(xdp); skb->protocol = eth_type_trans(skb, rxq->dev);
--
2.55.0