Thread (5 messages) 5 messages, 2 authors, 11h ago

[RFC bpf-next v2 2/3] bpf: Add PPPoE decap support to bpf_skb_adjust_room

From: ThisSeanZhang <hidden>
Date: 2026-10-03 19:34:04
Also in: bpf
Subsystem: bpf [general] (safe dynamic programs and tools), bpf [networking] (tcx & tc bpf, sock_addr), networking [general], the rest · Maintainers: Alexei Starovoitov, Daniel Borkmann, Andrii Nakryiko, Eduard Zingerman, Kumar Kartikeya Dwivedi, "David S. Miller", Eric Dumazet, Jakub Kicinski, Paolo Abeni, Linus Torvalds

Add a new BPF_F_ADJ_ROOM_DECAP_PPPOE flag to bpf_skb_adjust_room() so
that a TC BPF program can decapsulate a PPPoE session packet:

    bpf_skb_adjust_room(skb, -PPPOE_SES_HLEN, BPF_ADJ_ROOM_MAC,
                        BPF_F_ADJ_ROOM_DECAP_PPPOE);

The flag removes the six-byte PPPoE session header and the following
two-byte PPP protocol field between the MAC and network headers. It
restores consistent skb metadata for later consumers: the PPP protocol
field selects ETH_P_IP or ETH_P_IPV6, and any other PPP protocol is
rejected.

Before this patch, bpf_skb_adjust_room() rejected PPPoE packets because
it only accepted ETH_P_IP or ETH_P_IPV6 in skb->protocol. A TC program
therefore could not remove the PPPoE header and restore the inner
protocol metadata with the existing helper.

This is the counterpart to BPF_F_ADJ_ROOM_ENCAP_PPPOE introduced in
the previous patch and restores plain IPv4 or IPv6 skb metadata after
PPPoE decapsulation.

The flag also resets mac_len after removing the header. On flows where
a packet encapsulated on the same host re-enters TC ingress without
passing through the receive path, mac_len may still include the
removed PPPoE header. The reset mirrors what pppoe_gso_segment() does
for its segments.

Signed-off-by: ThisSeanZhang <redacted>
---
 include/uapi/linux/bpf.h       | 10 +++++
 net/core/filter.c              | 69 +++++++++++++++++++++++++++++++---
 tools/include/uapi/linux/bpf.h | 10 +++++
 3 files changed, 83 insertions(+), 6 deletions(-)
diff --git a/include/uapi/linux/bpf.h b/include/uapi/linux/bpf.h
index 30481d040..52a509060 100644
--- a/include/uapi/linux/bpf.h
+++ b/include/uapi/linux/bpf.h
@@ -3071,6 +3071,15 @@ union bpf_attr {
  *		  decapsulating a tunnel with an outer IPv6 header (IPv6-in-IPv6
  *		  or IPv4-in-IPv6).
  *
+ *		* **BPF_F_ADJ_ROOM_DECAP_PPPOE**:
+ *		  Decapsulate a PPPoE session header. Must be used with
+ *		  **BPF_ADJ_ROOM_MAC** mode and a negative *len_diff* equal to
+ *		  the size of the PPPoE session header plus the PPP protocol
+ *		  field (8 bytes in total). The PPP protocol field determines
+ *		  the new *skb->protocol* (**ETH_P_IP** or **ETH_P_IPV6**);
+ *		  a payload with any other PPP protocol is rejected. The Ethernet
+ *		  type in the MAC header is left for the BPF program to restore.
+ *
  *		When using the decapsulation flags above, the skb->encapsulation
  *		flag is automatically cleared if all tunnel-specific GSO flags
  *		(SKB_GSO_UDP_TUNNEL, SKB_GSO_UDP_TUNNEL_CSUM, SKB_GSO_GRE,
@@ -6335,6 +6344,7 @@ enum bpf_adj_room_flags {
 	BPF_F_ADJ_ROOM_DECAP_IPXIP4	= (1ULL << 11),
 	BPF_F_ADJ_ROOM_DECAP_IPXIP6	= (1ULL << 12),
 	BPF_F_ADJ_ROOM_ENCAP_PPPOE	= (1ULL << 13),
+	BPF_F_ADJ_ROOM_DECAP_PPPOE	= (1ULL << 14),
 };
 
 enum {
diff --git a/net/core/filter.c b/net/core/filter.c
index 5b204e316..46986a84e 100644
--- a/net/core/filter.c
+++ b/net/core/filter.c
@@ -48,6 +48,7 @@
 #include <linux/seccomp.h>
 #include <linux/if_vlan.h>
 #include <linux/if_pppox.h>
+#include <linux/ppp_defs.h>
 #include <linux/bpf.h>
 #include <linux/btf.h>
 #include <net/sch_generic.h>
@@ -3590,7 +3591,8 @@ static u32 bpf_skb_net_base_len(const struct sk_buff *skb)
 
 #define BPF_F_ADJ_ROOM_DECAP_MASK	(BPF_F_ADJ_ROOM_DECAP_L3_MASK | \
 					 BPF_F_ADJ_ROOM_DECAP_L4_MASK | \
-					 BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK)
+					 BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK | \
+					 BPF_F_ADJ_ROOM_DECAP_PPPOE)
 
 #define BPF_F_ADJ_ROOM_MASK		(BPF_F_ADJ_ROOM_FIXED_GSO | \
 					 BPF_F_ADJ_ROOM_ENCAP_MASK | \
@@ -3724,6 +3726,8 @@ static int bpf_skb_net_shrink(struct sk_buff *skb, u32 off, u32 len_diff,
 			      u64 flags)
 {
 	bool decap = flags & BPF_F_ADJ_ROOM_DECAP_L3_MASK;
+	__be16 inner_proto = 0;
+	u32 inner_len = 0;
 	int ret;
 
 	if (unlikely(flags & ~(BPF_F_ADJ_ROOM_DECAP_MASK |
@@ -3742,6 +3746,33 @@ static int bpf_skb_net_shrink(struct sk_buff *skb, u32 off, u32 len_diff,
 	if (unlikely(ret < 0))
 		return ret;
 
+	if (flags & BPF_F_ADJ_ROOM_DECAP_PPPOE) {
+		u16 ppp_proto;
+
+		if (unlikely(!pskb_may_pull(skb, off + PPPOE_SES_HLEN)))
+			return -ENOMEM;
+
+		/* PPP protocol field follows the PPPoE session header. */
+		ppp_proto = get_unaligned_be16(skb->data + off +
+					       sizeof(struct pppoe_hdr));
+		switch (ppp_proto) {
+		case PPP_IP:
+			inner_proto = htons(ETH_P_IP);
+			inner_len = sizeof(struct iphdr);
+			break;
+		case PPP_IPV6:
+			inner_proto = htons(ETH_P_IPV6);
+			inner_len = sizeof(struct ipv6hdr);
+			break;
+		default:
+			return -ENOTSUPP;
+		}
+
+		/* A full inner L3 header must remain after decapsulation. */
+		if (skb->len - off - PPPOE_SES_HLEN < inner_len)
+			return -EINVAL;
+	}
+
 	ret = bpf_skb_net_hdr_pop(skb, off, len_diff);
 	if (unlikely(ret < 0))
 		return ret;
@@ -3757,6 +3788,13 @@ static int bpf_skb_net_shrink(struct sk_buff *skb, u32 off, u32 len_diff,
 			skb_dst_drop(skb);
 	}
 
+	if (flags & BPF_F_ADJ_ROOM_DECAP_PPPOE) {
+		skb->protocol = inner_proto;
+		skb_reset_mac_len(skb);
+		if (skb_valid_dst(skb))
+			skb_dst_drop(skb);
+	}
+
 	if (skb_is_gso(skb)) {
 		struct skb_shared_info *shinfo = skb_shinfo(skb);
 
@@ -3869,9 +3907,13 @@ BPF_CALL_4(bpf_skb_adjust_room, struct sk_buff *, skb, s32, len_diff,
 		return -EINVAL;
 	if (unlikely(len_diff_abs > 0xfffU))
 		return -EFAULT;
-	if (unlikely(proto != htons(ETH_P_IP) &&
-		     proto != htons(ETH_P_IPV6)))
+	if (unlikely(flags & BPF_F_ADJ_ROOM_DECAP_PPPOE)) {
+		if (proto != htons(ETH_P_PPP_SES))
+			return -ENOTSUPP;
+	} else if (unlikely(proto != htons(ETH_P_IP) &&
+			    proto != htons(ETH_P_IPV6))) {
 		return -ENOTSUPP;
+	}
 
 	off = skb_mac_header_len(skb);
 	switch (mode) {
@@ -3885,9 +3927,7 @@ BPF_CALL_4(bpf_skb_adjust_room, struct sk_buff *, skb, s32, len_diff,
 	}
 
 	if (flags & BPF_F_ADJ_ROOM_ENCAP_PPPOE) {
-		/* The PPPoE session header has a fixed size and is
-		 * inserted directly after the MAC header.
-		 */
+		/* Fixed-size header, inserted after the MAC header. */
 		if (shrink || mode != BPF_ADJ_ROOM_MAC ||
 		    len_diff != PPPOE_SES_HLEN ||
 		    flags & ((BPF_F_ADJ_ROOM_ENCAP_MASK |
@@ -3920,6 +3960,23 @@ BPF_CALL_4(bpf_skb_adjust_room, struct sk_buff *, skb, s32, len_diff,
 		    (flags & BPF_F_ADJ_ROOM_DECAP_IPXIP_MASK))
 			return -EINVAL;
 
+		/* PPPoE decapsulation is mutually exclusive with the
+		 * other decapsulation types.
+		 */
+		if ((flags & BPF_F_ADJ_ROOM_DECAP_PPPOE) &&
+		    (flags & (BPF_F_ADJ_ROOM_DECAP_MASK &
+			      ~BPF_F_ADJ_ROOM_DECAP_PPPOE)))
+			return -EINVAL;
+
+		if (flags & BPF_F_ADJ_ROOM_DECAP_PPPOE) {
+			/* Fixed-size header; require a full inner L3 header. */
+			if (mode != BPF_ADJ_ROOM_MAC ||
+			    len_diff_abs != PPPOE_SES_HLEN)
+				return -EINVAL;
+
+			len_min = sizeof(struct iphdr);
+		}
+
 		if (flags & BPF_F_ADJ_ROOM_DECAP_L4_MASK)
 			len_decap_min += bpf_skb_net_base_len(skb);
 
diff --git a/tools/include/uapi/linux/bpf.h b/tools/include/uapi/linux/bpf.h
index 30481d040..52a509060 100644
--- a/tools/include/uapi/linux/bpf.h
+++ b/tools/include/uapi/linux/bpf.h
@@ -3071,6 +3071,15 @@ union bpf_attr {
  *		  decapsulating a tunnel with an outer IPv6 header (IPv6-in-IPv6
  *		  or IPv4-in-IPv6).
  *
+ *		* **BPF_F_ADJ_ROOM_DECAP_PPPOE**:
+ *		  Decapsulate a PPPoE session header. Must be used with
+ *		  **BPF_ADJ_ROOM_MAC** mode and a negative *len_diff* equal to
+ *		  the size of the PPPoE session header plus the PPP protocol
+ *		  field (8 bytes in total). The PPP protocol field determines
+ *		  the new *skb->protocol* (**ETH_P_IP** or **ETH_P_IPV6**);
+ *		  a payload with any other PPP protocol is rejected. The Ethernet
+ *		  type in the MAC header is left for the BPF program to restore.
+ *
  *		When using the decapsulation flags above, the skb->encapsulation
  *		flag is automatically cleared if all tunnel-specific GSO flags
  *		(SKB_GSO_UDP_TUNNEL, SKB_GSO_UDP_TUNNEL_CSUM, SKB_GSO_GRE,
@@ -6335,6 +6344,7 @@ enum bpf_adj_room_flags {
 	BPF_F_ADJ_ROOM_DECAP_IPXIP4	= (1ULL << 11),
 	BPF_F_ADJ_ROOM_DECAP_IPXIP6	= (1ULL << 12),
 	BPF_F_ADJ_ROOM_ENCAP_PPPOE	= (1ULL << 13),
+	BPF_F_ADJ_ROOM_DECAP_PPPOE	= (1ULL << 14),
 };
 
 enum {
-- 
2.47.3
Keyboard shortcuts
hback out one level
jnext message in thread
kprevious message in thread
ldrill in
Escclose help / fold thread tree
?toggle this help