From 04d2feaed8d0103c498727191ba04001d5100e67 Mon Sep 17 00:00:00 2001 From: Julian Anastasov Date: Fri, 31 Jul 2026 22:27:41 +0800 Subject: ipvs: add totalconns for dest Replace the inactconns dest counter with totalconns, now inactconns can be obtained from totalconns - activeconns. This reduces the atomic inc/dec ops for TCP/SCTP from 6 to 4 if the connection is established and then closed. Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Cc: stable@vger.kernel.org Signed-off-by: Julian Anastasov Signed-off-by: Yizhou Zhao Signed-off-by: Pablo Neira Ayuso --- include/net/ip_vs.h | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) (limited to 'include/net') diff --git a/include/net/ip_vs.h b/include/net/ip_vs.h index d2813eb795be..11f430646db8 100644 --- a/include/net/ip_vs.h +++ b/include/net/ip_vs.h @@ -987,7 +987,7 @@ struct ip_vs_dest { /* connection counters and thresholds */ atomic_t activeconns; /* active connections */ - atomic_t inactconns; /* inactive connections */ + atomic_t totalconns; /* total connections */ atomic_t persistconns; /* persistent connections */ __u32 u_threshold; /* upper threshold */ __u32 l_threshold; /* lower threshold */ @@ -2220,14 +2220,21 @@ void ip_vs_unregister_hooks(struct netns_ipvs *ipvs, unsigned int af); static inline int ip_vs_dest_conn_overhead(struct ip_vs_dest *dest) { - /* We think the overhead of processing active connections is 256 + /* We think the overhead of processing active connections is 257 * times higher than that of inactive connections in average. (This - * 256 times might not be accurate, we will change it later) We + * 257 times might not be accurate, we will change it later) We * use the following formula to estimate the overhead now: - * dest->activeconns*256 + dest->inactconns + * dest->activeconns*256 + dest->totalconns */ return (atomic_read(&dest->activeconns) << 8) + - atomic_read(&dest->inactconns); + atomic_read(&dest->totalconns); +} + +static inline int +ip_vs_dest_inactconns(const struct ip_vs_dest *dest) +{ + return max(atomic_read(&dest->totalconns) - + atomic_read(&dest->activeconns), 0); } #ifdef CONFIG_IP_VS_PROTO_TCP -- cgit From 8f843441c4e7eae8ea83491e8c203c2b192edcf5 Mon Sep 17 00:00:00 2001 From: Julian Anastasov Date: Fri, 31 Jul 2026 22:27:42 +0800 Subject: ipvs: properly update the overload flag on dest edit The upper/lower connection thresholds for dest can be changed, so use ip_vs_dest_update_overload() to properly update the dest overload flag. The thresholds were not limited, fit them in the 0 .. INT_MAX range as already done in ipvsadm. As the thresholds are also read when connections are created and expired, use WRITE_ONCE/READ_ONCE to access them. As the lower threshold is optional, use (u - (u >> 2)) to calculate the 75% default value based on the upper threshold by preserving the integer rounding, as suggested by Yizhou Zhao. Trigger flag update when totalconns reaches one of the thresholds and use dst_lock to serialize the updating. Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Cc: stable@vger.kernel.org Signed-off-by: Julian Anastasov Signed-off-by: Yizhou Zhao Signed-off-by: Pablo Neira Ayuso --- include/net/ip_vs.h | 3 ++ net/netfilter/ipvs/ip_vs_conn.c | 27 ++++++----------- net/netfilter/ipvs/ip_vs_ctl.c | 67 ++++++++++++++++++++++++++++++++++++----- 3 files changed, 72 insertions(+), 25 deletions(-) (limited to 'include/net') diff --git a/include/net/ip_vs.h b/include/net/ip_vs.h index 11f430646db8..e99382930617 100644 --- a/include/net/ip_vs.h +++ b/include/net/ip_vs.h @@ -991,6 +991,7 @@ struct ip_vs_dest { atomic_t persistconns; /* persistent connections */ __u32 u_threshold; /* upper threshold */ __u32 l_threshold; /* lower threshold */ + __u32 l_threshold_val;/* used lower threshold */ /* for destination cache */ spinlock_t dst_lock; /* lock of dst_cache */ @@ -1907,6 +1908,8 @@ static inline void ip_vs_dest_put_and_free(struct ip_vs_dest *dest) kfree(dest); } +void ip_vs_dest_update_overload(struct ip_vs_dest *dest, int mode); + /* IPVS sync daemon data and function prototypes * (from ip_vs_sync.c) */ diff --git a/net/netfilter/ipvs/ip_vs_conn.c b/net/netfilter/ipvs/ip_vs_conn.c index 4d0a6f718ced..abf52a226fee 100644 --- a/net/netfilter/ipvs/ip_vs_conn.c +++ b/net/netfilter/ipvs/ip_vs_conn.c @@ -1141,22 +1141,22 @@ ip_vs_bind_dest(struct ip_vs_conn *cp, struct ip_vs_dest *dest) /* Update the connection counters */ if (!(flags & IP_VS_CONN_F_TEMPLATE)) { + int tc; + /* It is a normal connection, so modify the counters * according to the flags, later the protocol can * update them on state change */ if (!(flags & IP_VS_CONN_F_INACTIVE)) atomic_inc(&dest->activeconns); - atomic_inc(&dest->totalconns); + tc = atomic_inc_return(&dest->totalconns); + if (tc == READ_ONCE(dest->u_threshold)) + ip_vs_dest_update_overload(dest, 1); } else { /* It is a persistent connection/template, so increase the persistent connection counter */ atomic_inc(&dest->persistconns); } - - if (dest->u_threshold != 0 && - atomic_read(&dest->totalconns) >= dest->u_threshold) - dest->flags |= IP_VS_DEST_F_OVERLOAD; } @@ -1237,27 +1237,20 @@ static inline void ip_vs_unbind_dest(struct ip_vs_conn *cp) /* Update the connection counters */ if (!(cp->flags & IP_VS_CONN_F_TEMPLATE)) { + int tc; + /* It is a normal connection, so decrease the counters */ if (!(cp->flags & IP_VS_CONN_F_INACTIVE)) atomic_dec(&dest->activeconns); - atomic_dec(&dest->totalconns); + tc = atomic_fetch_dec(&dest->totalconns); + if (tc == READ_ONCE(dest->l_threshold_val)) + ip_vs_dest_update_overload(dest, -1); } else { /* It is a persistent connection/template, so decrease the persistent connection counter */ atomic_dec(&dest->persistconns); } - if (dest->l_threshold != 0) { - if (atomic_read(&dest->totalconns) < dest->l_threshold) - dest->flags &= ~IP_VS_DEST_F_OVERLOAD; - } else if (dest->u_threshold != 0) { - if (atomic_read(&dest->totalconns) * 4 < dest->u_threshold * 3) - dest->flags &= ~IP_VS_DEST_F_OVERLOAD; - } else { - if (dest->flags & IP_VS_DEST_F_OVERLOAD) - dest->flags &= ~IP_VS_DEST_F_OVERLOAD; - } - ip_vs_dest_put(dest); } diff --git a/net/netfilter/ipvs/ip_vs_ctl.c b/net/netfilter/ipvs/ip_vs_ctl.c index 45f534427d23..974773642af8 100644 --- a/net/netfilter/ipvs/ip_vs_ctl.c +++ b/net/netfilter/ipvs/ip_vs_ctl.c @@ -1304,6 +1304,40 @@ void ip_vs_stats_free(struct ip_vs_stats *stats) } } +/* Update overload flag based on number of dest conns and lower/upper + * connection thresholds: + * - conns reach u_threshold and exceed it: set the flag + * - conns go below l_threshold (or 75% of u_threshold): clear the flag + */ +static void __ip_vs_dest_update_overload(struct ip_vs_dest *dest, int mode) +{ + int conns; + u32 l, u; + + lockdep_assert_held(&dest->dst_lock); + u = READ_ONCE(dest->u_threshold); + if (!u) + goto unset; + l = READ_ONCE(dest->l_threshold_val); + conns = atomic_read(&dest->totalconns); + if (conns >= (mode > 0 ? l : u)) { + dest->flags |= IP_VS_DEST_F_OVERLOAD; + return; + } + if (conns >= (mode < 0 ? u : l)) + return; + +unset: + dest->flags &= ~IP_VS_DEST_F_OVERLOAD; +} + +void ip_vs_dest_update_overload(struct ip_vs_dest *dest, int mode) +{ + spin_lock_bh(&dest->dst_lock); + __ip_vs_dest_update_overload(dest, mode); + spin_unlock_bh(&dest->dst_lock); +} + /* * Update a destination in the given service */ @@ -1370,10 +1404,19 @@ __ip_vs_update_dest(struct ip_vs_service *svc, struct ip_vs_dest *dest, /* set the dest status flags */ dest->flags |= IP_VS_DEST_F_AVAILABLE; - if (udest->u_threshold == 0 || udest->u_threshold > dest->u_threshold) - dest->flags &= ~IP_VS_DEST_F_OVERLOAD; - dest->u_threshold = udest->u_threshold; - dest->l_threshold = udest->l_threshold; + if (READ_ONCE(dest->u_threshold) != udest->u_threshold || + READ_ONCE(dest->l_threshold) != udest->l_threshold) { + spin_lock_bh(&dest->dst_lock); + WRITE_ONCE(dest->u_threshold, udest->u_threshold); + WRITE_ONCE(dest->l_threshold, udest->l_threshold); + /* Low threshold defaults to 75% of upper threshold */ + WRITE_ONCE(dest->l_threshold_val, + udest->l_threshold ? : + (udest->u_threshold - + (udest->u_threshold >> 2))); + __ip_vs_dest_update_overload(dest, 0); + spin_unlock_bh(&dest->dst_lock); + } dest->af = udest->af; @@ -1486,6 +1529,9 @@ ip_vs_add_dest(struct ip_vs_service *svc, struct ip_vs_dest_user_kern *udest) return -ERANGE; } + if (udest->u_threshold > INT_MAX) + return -EINVAL; + if (udest->tun_type == IP_VS_CONN_F_TUNNEL_TYPE_GUE) { if (udest->tun_port == 0) { pr_err("%s(): tunnel port is zero\n", __func__); @@ -1559,6 +1605,9 @@ ip_vs_edit_dest(struct ip_vs_service *svc, struct ip_vs_dest_user_kern *udest) return -ERANGE; } + if (udest->u_threshold > INT_MAX) + return -EINVAL; + if (udest->tun_type == IP_VS_CONN_F_TUNNEL_TYPE_GUE) { if (udest->tun_port == 0) { pr_err("%s(): tunnel port is zero\n", __func__); @@ -3667,8 +3716,8 @@ __ip_vs_get_dest_entries(struct netns_ipvs *ipvs, const struct ip_vs_get_dests * entry.port = dest->port; entry.conn_flags = atomic_read(&dest->conn_flags); entry.weight = atomic_read(&dest->weight); - entry.u_threshold = dest->u_threshold; - entry.l_threshold = dest->l_threshold; + entry.u_threshold = READ_ONCE(dest->u_threshold); + entry.l_threshold = READ_ONCE(dest->l_threshold); entry.activeconns = atomic_read(&dest->activeconns); entry.inactconns = ip_vs_dest_inactconns(dest); entry.persistconns = atomic_read(&dest->persistconns); @@ -4277,8 +4326,10 @@ static int ip_vs_genl_fill_dest(struct sk_buff *skb, struct ip_vs_dest *dest) dest->tun_port) || nla_put_u16(skb, IPVS_DEST_ATTR_TUN_FLAGS, dest->tun_flags) || - nla_put_u32(skb, IPVS_DEST_ATTR_U_THRESH, dest->u_threshold) || - nla_put_u32(skb, IPVS_DEST_ATTR_L_THRESH, dest->l_threshold) || + nla_put_u32(skb, IPVS_DEST_ATTR_U_THRESH, + READ_ONCE(dest->u_threshold)) || + nla_put_u32(skb, IPVS_DEST_ATTR_L_THRESH, + READ_ONCE(dest->l_threshold)) || nla_put_u32(skb, IPVS_DEST_ATTR_ACTIVE_CONNS, atomic_read(&dest->activeconns)) || nla_put_u32(skb, IPVS_DEST_ATTR_INACT_CONNS, -- cgit From cdcc4e46180df8161f4d2f3c6fd6beaf6990133d Mon Sep 17 00:00:00 2001 From: Yizhou Zhao Date: Fri, 31 Jul 2026 22:27:43 +0800 Subject: ipvs: separate destination availability state IPVS configuration paths update destination availability while connection accounting updates destination overload state. The two independent states share dest->flags, so their read-modify-write updates can race and lose one another. Keep OVERLOAD in flags, where the preceding patch serializes its updates with dst_lock, and move AVAILABLE to cflags. This keeps configuration- controlled availability out of the scheduler hot cacheline until a scheduler needs to check it. It also prevents availability updates from clobbering overload state. The destination status bits are not exposed through the IPVS sockopt or netlink interfaces, so keep their definitions in the internal IPVS header. Readers can still observe stale destination state; this does not provide a cross-field snapshot. Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Cc: stable@vger.kernel.org Reported-by: Yizhou Zhao Reported-by: Yuxiang Yang Reported-by: Ao Wang Reported-by: Xuewei Feng Reported-by: Qi Li Reported-by: Ke Xu Link: https://lore.kernel.org/all/8913381c-1e02-35c7-0ec4-61de5a12fd35@ssi.bg/ Assisted-by: Claude-Code:GLM-5.2 Suggested-by: Julian Anastasov Signed-off-by: Yizhou Zhao Acked-by: Julian Anastasov Signed-off-by: Pablo Neira Ayuso --- include/net/ip_vs.h | 7 +++++++ include/uapi/linux/ip_vs.h | 6 ------ net/netfilter/ipvs/ip_vs_conn.c | 4 ++-- net/netfilter/ipvs/ip_vs_core.c | 6 +++--- net/netfilter/ipvs/ip_vs_ctl.c | 4 ++-- net/netfilter/ipvs/ip_vs_dh.c | 4 ++-- net/netfilter/ipvs/ip_vs_lblc.c | 2 +- net/netfilter/ipvs/ip_vs_lblcr.c | 8 ++++---- net/netfilter/ipvs/ip_vs_xmit.c | 4 ++-- 9 files changed, 23 insertions(+), 22 deletions(-) (limited to 'include/net') diff --git a/include/net/ip_vs.h b/include/net/ip_vs.h index e99382930617..fc2ef5ef31a6 100644 --- a/include/net/ip_vs.h +++ b/include/net/ip_vs.h @@ -36,6 +36,12 @@ #define IP_VS_HDR_INVERSE 1 #define IP_VS_HDR_ICMP 2 +/* Destination Server Flags */ +#define IP_VS_DEST_F_OVERLOAD 0x0002 /* server is overloaded */ + +/* Destination Server Config Flags */ +#define IP_VS_DEST_CF_AVAILABLE 0x0001 /* server is available */ + /* conn_tab limits (as per Kconfig) */ #define IP_VS_CONN_TAB_MIN_BITS 8 #if BITS_PER_LONG > 32 @@ -976,6 +982,7 @@ struct ip_vs_dest { volatile unsigned int flags; /* dest status flags */ atomic_t conn_flags; /* flags to copy to conn */ atomic_t weight; /* server weight */ + unsigned long cflags; /* config flags */ atomic_t last_weight; /* server latest weight */ __u16 tun_type; /* tunnel type */ __be16 tun_port; /* tunnel port */ diff --git a/include/uapi/linux/ip_vs.h b/include/uapi/linux/ip_vs.h index 1ed234e7f251..2c37c6ac7525 100644 --- a/include/uapi/linux/ip_vs.h +++ b/include/uapi/linux/ip_vs.h @@ -28,12 +28,6 @@ #define IP_VS_SVC_F_SCHED_SH_FALLBACK IP_VS_SVC_F_SCHED1 /* SH fallback */ #define IP_VS_SVC_F_SCHED_SH_PORT IP_VS_SVC_F_SCHED2 /* SH use port */ -/* - * Destination Server Flags - */ -#define IP_VS_DEST_F_AVAILABLE 0x0001 /* server is available */ -#define IP_VS_DEST_F_OVERLOAD 0x0002 /* server is overloaded */ - /* * IPVS sync daemon states */ diff --git a/net/netfilter/ipvs/ip_vs_conn.c b/net/netfilter/ipvs/ip_vs_conn.c index abf52a226fee..6fa3e1dc534c 100644 --- a/net/netfilter/ipvs/ip_vs_conn.c +++ b/net/netfilter/ipvs/ip_vs_conn.c @@ -1279,7 +1279,7 @@ int ip_vs_check_template(struct ip_vs_conn *ct, struct ip_vs_dest *cdest) * Checking the dest server status. */ if ((dest == NULL) || - !(dest->flags & IP_VS_DEST_F_AVAILABLE) || + !(dest->cflags & IP_VS_DEST_CF_AVAILABLE) || expire_quiescent_template(ipvs, dest) || (cdest && (dest != cdest))) { IP_VS_DBG_BUF(9, "check_template: dest not available for " @@ -2020,7 +2020,7 @@ repeat: cp = ip_vs_hn0_to_conn(hn); resched_score++; dest = cp->dest; - if (!dest || (dest->flags & IP_VS_DEST_F_AVAILABLE)) + if (!dest || (dest->cflags & IP_VS_DEST_CF_AVAILABLE)) continue; if (atomic_read(&cp->n_control)) diff --git a/net/netfilter/ipvs/ip_vs_core.c b/net/netfilter/ipvs/ip_vs_core.c index 0bdaeb4ed61e..95af77b68851 100644 --- a/net/netfilter/ipvs/ip_vs_core.c +++ b/net/netfilter/ipvs/ip_vs_core.c @@ -302,7 +302,7 @@ ip_vs_in_stats(struct ip_vs_conn *cp, struct sk_buff *skb) struct ip_vs_dest *dest = cp->dest; struct netns_ipvs *ipvs = cp->ipvs; - if (dest && (dest->flags & IP_VS_DEST_F_AVAILABLE)) { + if (dest && (dest->cflags & IP_VS_DEST_CF_AVAILABLE)) { struct ip_vs_cpu_stats *s; struct ip_vs_service *svc; @@ -338,7 +338,7 @@ ip_vs_out_stats(struct ip_vs_conn *cp, struct sk_buff *skb) struct ip_vs_dest *dest = cp->dest; struct netns_ipvs *ipvs = cp->ipvs; - if (dest && (dest->flags & IP_VS_DEST_F_AVAILABLE)) { + if (dest && (dest->cflags & IP_VS_DEST_CF_AVAILABLE)) { struct ip_vs_cpu_stats *s; struct ip_vs_service *svc; @@ -2210,7 +2210,7 @@ ip_vs_in_hook(void *priv, struct sk_buff *skb, const struct nf_hook_state *state } /* Check the server status */ - if (cp && cp->dest && !(cp->dest->flags & IP_VS_DEST_F_AVAILABLE)) { + if (cp && cp->dest && !(cp->dest->cflags & IP_VS_DEST_CF_AVAILABLE)) { /* the destination server is not available */ if (sysctl_expire_nodest_conn(ipvs)) { bool old_ct = ip_vs_conn_uses_old_conntrack(cp, skb); diff --git a/net/netfilter/ipvs/ip_vs_ctl.c b/net/netfilter/ipvs/ip_vs_ctl.c index 974773642af8..8f9a8e491ad6 100644 --- a/net/netfilter/ipvs/ip_vs_ctl.c +++ b/net/netfilter/ipvs/ip_vs_ctl.c @@ -1402,7 +1402,7 @@ __ip_vs_update_dest(struct ip_vs_service *svc, struct ip_vs_dest *dest, } /* set the dest status flags */ - dest->flags |= IP_VS_DEST_F_AVAILABLE; + dest->cflags |= IP_VS_DEST_CF_AVAILABLE; if (READ_ONCE(dest->u_threshold) != udest->u_threshold || READ_ONCE(dest->l_threshold) != udest->l_threshold) { @@ -1662,7 +1662,7 @@ static void __ip_vs_unlink_dest(struct ip_vs_service *svc, struct ip_vs_dest *dest, int svcupd) { - dest->flags &= ~IP_VS_DEST_F_AVAILABLE; + dest->cflags &= ~IP_VS_DEST_CF_AVAILABLE; spin_lock_bh(&dest->dst_lock); __ip_vs_dst_cache_reset(dest); diff --git a/net/netfilter/ipvs/ip_vs_dh.c b/net/netfilter/ipvs/ip_vs_dh.c index e1f62f6b25e2..43abed7a26a6 100644 --- a/net/netfilter/ipvs/ip_vs_dh.c +++ b/net/netfilter/ipvs/ip_vs_dh.c @@ -219,8 +219,8 @@ ip_vs_dh_schedule(struct ip_vs_service *svc, const struct sk_buff *skb, s = (struct ip_vs_dh_state *) svc->sched_data; dest = ip_vs_dh_get(svc->af, s, &iph->daddr); - if (!dest - || !(dest->flags & IP_VS_DEST_F_AVAILABLE) + if (!dest || + !(dest->cflags & IP_VS_DEST_CF_AVAILABLE) || atomic_read(&dest->weight) <= 0 || is_overloaded(dest)) { ip_vs_scheduler_err(svc, "no destination available"); diff --git a/net/netfilter/ipvs/ip_vs_lblc.c b/net/netfilter/ipvs/ip_vs_lblc.c index 15ccb2b2fa1f..693bcc82ccb7 100644 --- a/net/netfilter/ipvs/ip_vs_lblc.c +++ b/net/netfilter/ipvs/ip_vs_lblc.c @@ -502,7 +502,7 @@ ip_vs_lblc_schedule(struct ip_vs_service *svc, const struct sk_buff *skb, */ dest = en->dest; - if ((dest->flags & IP_VS_DEST_F_AVAILABLE) && + if ((dest->cflags & IP_VS_DEST_CF_AVAILABLE) && atomic_read(&dest->weight) > 0 && !is_overloaded(dest, svc)) goto out; } diff --git a/net/netfilter/ipvs/ip_vs_lblcr.c b/net/netfilter/ipvs/ip_vs_lblcr.c index c90ea897c3f7..f53f05ceea36 100644 --- a/net/netfilter/ipvs/ip_vs_lblcr.c +++ b/net/netfilter/ipvs/ip_vs_lblcr.c @@ -169,8 +169,8 @@ static inline struct ip_vs_dest *ip_vs_dest_set_min(struct ip_vs_dest_set *set) if (least->flags & IP_VS_DEST_F_OVERLOAD) continue; - if ((atomic_read(&least->weight) > 0) - && (least->flags & IP_VS_DEST_F_AVAILABLE)) { + if ((atomic_read(&least->weight) > 0) && + (least->cflags & IP_VS_DEST_CF_AVAILABLE)) { loh = ip_vs_dest_conn_overhead(least); goto nextstage; } @@ -186,8 +186,8 @@ static inline struct ip_vs_dest *ip_vs_dest_set_min(struct ip_vs_dest_set *set) doh = ip_vs_dest_conn_overhead(dest); if (((__s64)loh * atomic_read(&dest->weight) > - (__s64)doh * atomic_read(&least->weight)) - && (dest->flags & IP_VS_DEST_F_AVAILABLE)) { + (__s64)doh * atomic_read(&least->weight)) && + (dest->cflags & IP_VS_DEST_CF_AVAILABLE)) { least = dest; loh = doh; } diff --git a/net/netfilter/ipvs/ip_vs_xmit.c b/net/netfilter/ipvs/ip_vs_xmit.c index c4508f3f43dd..fc7403186394 100644 --- a/net/netfilter/ipvs/ip_vs_xmit.c +++ b/net/netfilter/ipvs/ip_vs_xmit.c @@ -351,7 +351,7 @@ __ip_vs_get_out_rt(struct netns_ipvs *ipvs, int skb_af, struct sk_buff *skb, * stored in dest_trash. */ if (!rt_dev_is_down(dst_dev_rcu(&rt->dst)) && - dest->flags & IP_VS_DEST_F_AVAILABLE) + dest->cflags & IP_VS_DEST_CF_AVAILABLE) __ip_vs_dst_set(dest, dest_dst, &rt->dst, 0); else noref = 0; @@ -530,7 +530,7 @@ __ip_vs_get_out_rt_v6(struct netns_ipvs *ipvs, int skb_af, struct sk_buff *skb, * stored in dest_trash. */ if (!rt_dev_is_down(dst_dev_rcu(&rt->dst)) && - dest->flags & IP_VS_DEST_F_AVAILABLE) + dest->cflags & IP_VS_DEST_CF_AVAILABLE) __ip_vs_dst_set(dest, dest_dst, &rt->dst, cookie); else noref = 0; -- cgit From d93660df4dd1d116f608ada4a29a80a5d6f0a6ed Mon Sep 17 00:00:00 2001 From: Julian Anastasov Date: Thu, 6 Aug 2026 13:52:11 +0300 Subject: ipvs: revalidate ihl to prevent out-of-bounds access While the outer IP header is already pulled into the skb head, we must be careful and revalidate the embedded headers after reading them from the skb frags to prevent out-of-bounds access. One such place reported by Sashiko is ip_vs_nat_icmp() where local process can change the ihl field and after skb_ensure_writable() we can see larger value which is a problem for the ip_send_check(cih) calls. Add check to drop the packet if the ihl field is changed. Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Link: https://sashiko.dev/#/patchset/20260730183506.87473-1-ja%40ssi.bg Signed-off-by: Julian Anastasov Signed-off-by: Pablo Neira Ayuso --- include/net/ip_vs.h | 2 +- net/netfilter/ipvs/ip_vs_core.c | 11 +++++++++-- net/netfilter/ipvs/ip_vs_xmit.c | 3 ++- 3 files changed, 12 insertions(+), 4 deletions(-) (limited to 'include/net') diff --git a/include/net/ip_vs.h b/include/net/ip_vs.h index fc2ef5ef31a6..be3a6617adf4 100644 --- a/include/net/ip_vs.h +++ b/include/net/ip_vs.h @@ -2068,7 +2068,7 @@ static inline bool ip_vs_conn_use_hash2(struct ip_vs_conn *cp) !(cp->flags & IP_VS_CONN_F_TEMPLATE); } -void ip_vs_nat_icmp(struct sk_buff *skb, struct ip_vs_protocol *pp, +bool ip_vs_nat_icmp(struct sk_buff *skb, struct ip_vs_protocol *pp, struct ip_vs_conn *cp, int dir, unsigned int toff, bool has_ports, struct ip_vs_iphdr *ciph); diff --git a/net/netfilter/ipvs/ip_vs_core.c b/net/netfilter/ipvs/ip_vs_core.c index a46e7acdd8e1..eb806813292a 100644 --- a/net/netfilter/ipvs/ip_vs_core.c +++ b/net/netfilter/ipvs/ip_vs_core.c @@ -923,7 +923,7 @@ static int ip_vs_route_me_harder(struct netns_ipvs *ipvs, int af, * Packet has been made sufficiently writable in caller * - inout: 1=in->out, 0=out->in */ -void ip_vs_nat_icmp(struct sk_buff *skb, struct ip_vs_protocol *pp, +bool ip_vs_nat_icmp(struct sk_buff *skb, struct ip_vs_protocol *pp, struct ip_vs_conn *cp, int inout, unsigned int toff, bool has_ports, struct ip_vs_iphdr *ciph) { @@ -931,6 +931,11 @@ void ip_vs_nat_icmp(struct sk_buff *skb, struct ip_vs_protocol *pp, struct icmphdr *icmph = (struct icmphdr *)(skb->data + toff); struct iphdr *cih = (struct iphdr *)(icmph + 1); + /* Before now we may used ihl from skb frag, revalidate it after + * copying it into skb head to prevent out-of-bounds access + */ + if (cih->ihl * 4 != ciph->len - ciph->off) + return false; if (inout) { iph->saddr = cp->vaddr.ip; ip_send_check(iph); @@ -964,6 +969,7 @@ void ip_vs_nat_icmp(struct sk_buff *skb, struct ip_vs_protocol *pp, else IP_VS_DBG_PKT(11, AF_INET, pp, skb, ciph->off, "Forwarding altered incoming ICMP"); + return true; } #ifdef CONFIG_IP_VS_IPV6 @@ -1055,7 +1061,8 @@ static int handle_response_icmp(int af, struct sk_buff *skb, ip_vs_nat_icmp_v6(skb, pp, cp, 1, toff, has_ports, ciph); else #endif - ip_vs_nat_icmp(skb, pp, cp, 1, toff, has_ports, ciph); + if (!ip_vs_nat_icmp(skb, pp, cp, 1, toff, has_ports, ciph)) + goto out; if (ip_vs_route_me_harder(cp->ipvs, af, skb, hooknum)) goto out; diff --git a/net/netfilter/ipvs/ip_vs_xmit.c b/net/netfilter/ipvs/ip_vs_xmit.c index fc7403186394..04450a48f01a 100644 --- a/net/netfilter/ipvs/ip_vs_xmit.c +++ b/net/netfilter/ipvs/ip_vs_xmit.c @@ -1580,7 +1580,8 @@ ip_vs_icmp_xmit(struct sk_buff *skb, struct ip_vs_conn *cp, if (skb_cow(skb, rt->dst.dev->hard_header_len)) goto tx_error; - ip_vs_nat_icmp(skb, pp, cp, 0, toff, has_ports, ciph); + if (!ip_vs_nat_icmp(skb, pp, cp, 0, toff, has_ports, ciph)) + goto tx_error; /* Another hack: avoid icmp_send in ip_fragment */ skb->ignore_df = 1; -- cgit