From ef6cb145e216b0378686bce356f3317284d54231 Mon Sep 17 00:00:00 2001 From: Joel Granados Date: Mon, 10 Aug 2026 15:01:02 +0200 Subject: [PATCH 1/3] net: enforce net sysctl registration Replace the warning and file permission change with an error when an "unsafe" net sysctl registration is detected. One of the barriers preventing the const qualification of the ctl_tables in the net directory is the permission (->mode) change in ensure_safe_net_sysctl. This prep commit removes that barrier and ensures that the received ctl_table pointer to the net ctl_table register function is const. Signed-off-by: Joel Granados Link: https://patch.msgid.link/20260810-jag-net_const_qualify-v4-1-77e888237c69@kernel.org Reviewed-by: Simon Horman Signed-off-by: Paolo Abeni --- include/net/net_namespace.h | 5 +++-- net/sysctl_net.c | 23 ++++++++++++----------- 2 files changed, 15 insertions(+), 13 deletions(-) diff --git a/include/net/net_namespace.h b/include/net/net_namespace.h index 501af1999fe8..e5ee673b9fcf 100644 --- a/include/net/net_namespace.h +++ b/include/net/net_namespace.h @@ -525,12 +525,13 @@ struct ctl_table; #ifdef CONFIG_SYSCTL int net_sysctl_init(void); struct ctl_table_header *register_net_sysctl_sz(struct net *net, const char *path, - struct ctl_table *table, size_t table_size); + const struct ctl_table *table, + size_t table_size); void unregister_net_sysctl_table(struct ctl_table_header *header); #else static inline int net_sysctl_init(void) { return 0; } static inline struct ctl_table_header *register_net_sysctl_sz(struct net *net, - const char *path, struct ctl_table *table, size_t table_size) + const char *path, const struct ctl_table *table, size_t table_size) { return NULL; } diff --git a/net/sysctl_net.c b/net/sysctl_net.c index 19e8048241ba..e190a639eef2 100644 --- a/net/sysctl_net.c +++ b/net/sysctl_net.c @@ -114,16 +114,17 @@ __init int net_sysctl_init(void) goto out; } -/* Verify that sysctls for non-init netns are safe by either: +/* Return error when sysctls for non-init netns are unsafe by verifying: * 1) being read-only, or * 2) having a data pointer which points outside of the global kernel/module * data segment, and rather into the heap where a per-net object was * allocated. */ -static void ensure_safe_net_sysctl(struct net *net, const char *path, - struct ctl_table *table, size_t table_size) +static int ensure_safe_net_sysctl(struct net *net, const char *path, + const struct ctl_table *table, + size_t table_size) { - struct ctl_table *ent; + const struct ctl_table *ent; pr_debug("Registering net sysctl (net %p): %s\n", net, path); ent = table; @@ -149,24 +150,24 @@ static void ensure_safe_net_sysctl(struct net *net, const char *path, else continue; - /* If it is writable and points to kernel/module global - * data, then it's probably a netns leak. - */ + /* Warn on netns leak. */ WARN(1, "sysctl %s/%s: data points to %s global data: %ps\n", path, ent->procname, where, ent->data); - /* Make it "safe" by dropping writable perms */ - ent->mode &= ~0222; + return -EACCES; } + + return 0; } struct ctl_table_header *register_net_sysctl_sz(struct net *net, const char *path, - struct ctl_table *table, + const struct ctl_table *table, size_t table_size) { if (!net_eq(net, &init_net)) - ensure_safe_net_sysctl(net, path, table, table_size); + if (ensure_safe_net_sysctl(net, path, table, table_size)) + return NULL; return __register_sysctl_table(&net->sysctls, path, table, table_size); } From 09190c59cd101e0bf87a1c5a32ebae25c91e6d81 Mon Sep 17 00:00:00 2001 From: Joel Granados Date: Mon, 10 Aug 2026 15:01:03 +0200 Subject: [PATCH 2/3] net: Const qualify ctl_tables that kmemdup unconditionally Const qualify clt_table arrays in the net directory that always pass a memory duplicate to sysctl register. The template would then be in .rodata and the kmemdup'ed array would be outside. Signed-off-by: Joel Granados Link: https://patch.msgid.link/20260810-jag-net_const_qualify-v4-2-77e888237c69@kernel.org Reviewed-by: Simon Horman Signed-off-by: Paolo Abeni --- net/ipv4/devinet.c | 2 +- net/ipv6/icmp.c | 2 +- net/ipv6/route.c | 2 +- net/ipv6/sysctl_net_ipv6.c | 2 +- net/netfilter/nf_conntrack_standalone.c | 2 +- net/sctp/sysctl.c | 2 +- net/xfrm/xfrm_sysctl.c | 2 +- 7 files changed, 7 insertions(+), 7 deletions(-) diff --git a/net/ipv4/devinet.c b/net/ipv4/devinet.c index 47ded0f607d4..a90be57c63be 100644 --- a/net/ipv4/devinet.c +++ b/net/ipv4/devinet.c @@ -2796,7 +2796,7 @@ static void devinet_sysctl_unregister(struct in_device *idev) neigh_sysctl_unregister(idev->arp_parms); } -static struct ctl_table ctl_forward_entry[] = { +static const struct ctl_table ctl_forward_entry[] = { { .procname = "ip_forward", .data = &ipv4_devconf.data[ diff --git a/net/ipv6/icmp.c b/net/ipv6/icmp.c index efb23807a026..a95b0351824f 100644 --- a/net/ipv6/icmp.c +++ b/net/ipv6/icmp.c @@ -1374,7 +1374,7 @@ EXPORT_SYMBOL(icmpv6_err_convert); static u32 icmpv6_errors_extension_mask_all = GENMASK_U8(ICMP_ERR_EXT_COUNT - 1, 0); -static struct ctl_table ipv6_icmp_table_template[] = { +static const struct ctl_table ipv6_icmp_table_template[] = { { .procname = "ratelimit", .data = &init_net.ipv6.sysctl.icmpv6_time, diff --git a/net/ipv6/route.c b/net/ipv6/route.c index ae2f93a19dd9..16dfac54a259 100644 --- a/net/ipv6/route.c +++ b/net/ipv6/route.c @@ -6590,7 +6590,7 @@ static int ipv6_sysctl_rtcache_flush(const struct ctl_table *ctl, int write, return 0; } -static struct ctl_table ipv6_route_table_template[] = { +static const struct ctl_table ipv6_route_table_template[] = { { .procname = "max_size", .data = &init_net.ipv6.sysctl.ip6_rt_max_size, diff --git a/net/ipv6/sysctl_net_ipv6.c b/net/ipv6/sysctl_net_ipv6.c index d2cd33e2698d..1a0a36dcdabc 100644 --- a/net/ipv6/sysctl_net_ipv6.c +++ b/net/ipv6/sysctl_net_ipv6.c @@ -61,7 +61,7 @@ proc_rt6_multipath_hash_fields(const struct ctl_table *table, int write, void *b return ret; } -static struct ctl_table ipv6_table_template[] = { +static const struct ctl_table ipv6_table_template[] = { { .procname = "bindv6only", .data = &init_net.ipv6.sysctl.bindv6only, diff --git a/net/netfilter/nf_conntrack_standalone.c b/net/netfilter/nf_conntrack_standalone.c index be2953c7d702..f4f2d82192d5 100644 --- a/net/netfilter/nf_conntrack_standalone.c +++ b/net/netfilter/nf_conntrack_standalone.c @@ -639,7 +639,7 @@ enum nf_ct_sysctl_index { NF_SYSCTL_CT_LAST_SYSCTL, }; -static struct ctl_table nf_ct_sysctl_table[] = { +static const struct ctl_table nf_ct_sysctl_table[] = { [NF_SYSCTL_CT_MAX] = { .procname = "nf_conntrack_max", .data = &nf_conntrack_max, diff --git a/net/sctp/sysctl.c b/net/sctp/sysctl.c index fca840484ebf..2b94c211427d 100644 --- a/net/sctp/sysctl.c +++ b/net/sctp/sysctl.c @@ -92,7 +92,7 @@ static struct ctl_table sctp_table[] = { #define SCTP_PF_RETRANS_IDX 2 #define SCTP_PS_RETRANS_IDX 3 -static struct ctl_table sctp_net_table[] = { +static const struct ctl_table sctp_net_table[] = { [SCTP_RTO_MIN_IDX] = { .procname = "rto_min", .data = &init_net.sctp.rto_min, diff --git a/net/xfrm/xfrm_sysctl.c b/net/xfrm/xfrm_sysctl.c index ca003e8a0376..357152a50faf 100644 --- a/net/xfrm/xfrm_sysctl.c +++ b/net/xfrm/xfrm_sysctl.c @@ -13,7 +13,7 @@ static void __net_init __xfrm_sysctl_init(struct net *net) } #ifdef CONFIG_SYSCTL -static struct ctl_table xfrm_table[] = { +static const struct ctl_table xfrm_table[] = { { .procname = "xfrm_aevent_etime", .maxlen = sizeof(u32), From 0abc76bc20826e2582c4589e43b7fb8f3612911c Mon Sep 17 00:00:00 2001 From: Joel Granados Date: Mon, 10 Aug 2026 15:01:04 +0200 Subject: [PATCH 3/3] net: Const qualify network templated ctl_tables Arrays Add duplication helpers in the cases where the ctl_table array elements are modified after duplication. Helpers return a ctl_table as const pointer allowing the const qualification of the static global ctl_table array. Signed-off-by: Joel Granados Link: https://patch.msgid.link/20260810-jag-net_const_qualify-v4-3-77e888237c69@kernel.org Reviewed-by: Simon Horman Signed-off-by: Paolo Abeni --- net/core/sysctl_net_core.c | 38 ++++++++++++++-------- net/ipv4/sysctl_net_ipv4.c | 54 ++++++++++++++++++------------- net/ipv4/xfrm4_policy.c | 22 ++++++++++--- net/ipv6/xfrm6_policy.c | 22 ++++++++++--- net/netfilter/nf_hooks_lwtunnel.c | 4 +-- net/smc/smc_sysctl.c | 26 +++++++++++---- net/unix/sysctl_net_unix.c | 21 +++++++++--- net/vmw_vsock/af_vsock.c | 25 ++++++++++---- 8 files changed, 146 insertions(+), 66 deletions(-) diff --git a/net/core/sysctl_net_core.c b/net/core/sysctl_net_core.c index b508618bfc12..eb35da3556f4 100644 --- a/net/core/sysctl_net_core.c +++ b/net/core/sysctl_net_core.c @@ -678,7 +678,7 @@ static struct ctl_table net_core_table[] = { }, }; -static struct ctl_table netns_core_table[] = { +static const struct ctl_table netns_core_table[] = { #if IS_ENABLED(CONFIG_RPS) { .procname = "rps_default_mask", @@ -787,26 +787,38 @@ static int __init fb_tunnels_only_for_init_net_sysctl_setup(char *str) } __setup("fb_tunnels=", fb_tunnels_only_for_init_net_sysctl_setup); -static __net_init int sysctl_core_net_init(struct net *net) +static const struct ctl_table *netns_core_table_dup(struct net *net) { size_t table_size = ARRAY_SIZE(netns_core_table); struct ctl_table *tbl; + int i; + + tbl = kmemdup(netns_core_table, sizeof(netns_core_table), GFP_KERNEL); + if (!tbl) + return NULL; + + for (i = 0; i < table_size; ++i) { + if (tbl[i].data == &sysctl_wmem_max) + break; + + tbl[i].data += (char *)net - (char *)&init_net; + } + for (; i < table_size; ++i) + tbl[i].mode &= ~0222; + + return tbl; +} + +static __net_init int sysctl_core_net_init(struct net *net) +{ + size_t table_size = ARRAY_SIZE(netns_core_table); + const struct ctl_table *tbl; tbl = netns_core_table; if (!net_eq(net, &init_net)) { - int i; - tbl = kmemdup(tbl, sizeof(netns_core_table), GFP_KERNEL); + tbl = netns_core_table_dup(net); if (tbl == NULL) goto err_dup; - - for (i = 0; i < table_size; ++i) { - if (tbl[i].data == &sysctl_wmem_max) - break; - - tbl[i].data += (char *)net - (char *)&init_net; - } - for (; i < table_size; ++i) - tbl[i].mode &= ~0222; } net->core.sysctl_hdr = register_net_sysctl_sz(net, "net/core", tbl, table_size); diff --git a/net/ipv4/sysctl_net_ipv4.c b/net/ipv4/sysctl_net_ipv4.c index ca1180dba1de..2f0363bca2a8 100644 --- a/net/ipv4/sysctl_net_ipv4.c +++ b/net/ipv4/sysctl_net_ipv4.c @@ -624,7 +624,7 @@ static struct ctl_table ipv4_table[] = { }, }; -static struct ctl_table ipv4_net_table[] = { +static const struct ctl_table ipv4_net_table[] = { { .procname = "tcp_max_tw_buckets", .data = &init_net.ipv4.tcp_death_row.sysctl_max_tw_buckets, @@ -1654,35 +1654,45 @@ static struct ctl_table ipv4_net_table[] = { }, }; -static __net_init int ipv4_sysctl_init_net(struct net *net) +static const struct ctl_table *ipv4_net_table_dup(struct net *net) { size_t table_size = ARRAY_SIZE(ipv4_net_table); struct ctl_table *table; + int i; + + table = kmemdup(ipv4_net_table, sizeof(ipv4_net_table), GFP_KERNEL); + if (!table) + return NULL; + + for (i = 0; i < table_size; i++) { + if (table[i].data) { + /* Update the variables to point into + * the current struct net + */ + table[i].data += (void *)net - (void *)&init_net; + } else { + /* Entries without data pointer are global; + * Make them read-only in non-init_net ns + */ + table[i].mode &= ~0222; + } + if (table[i].extra2 >= (void *)&init_net.ipv4 && + table[i].extra2 < (void *)(&init_net.ipv4 + 1)) + table[i].extra2 += (void *)net - (void *)&init_net; + } + return table; +} + +static __net_init int ipv4_sysctl_init_net(struct net *net) +{ + size_t table_size = ARRAY_SIZE(ipv4_net_table); + const struct ctl_table *table; table = ipv4_net_table; if (!net_eq(net, &init_net)) { - int i; - - table = kmemdup(table, sizeof(ipv4_net_table), GFP_KERNEL); + table = ipv4_net_table_dup(net); if (!table) goto err_alloc; - - for (i = 0; i < table_size; i++) { - if (table[i].data) { - /* Update the variables to point into - * the current struct net - */ - table[i].data += (void *)net - (void *)&init_net; - } else { - /* Entries without data pointer are global; - * Make them read-only in non-init_net ns - */ - table[i].mode &= ~0222; - } - if (table[i].extra2 >= (void *)&init_net.ipv4 && - table[i].extra2 < (void *)(&init_net.ipv4 + 1)) - table[i].extra2 += (void *)net - (void *)&init_net; - } } net->ipv4.ipv4_hdr = register_net_sysctl_sz(net, "net/ipv4", table, diff --git a/net/ipv4/xfrm4_policy.c b/net/ipv4/xfrm4_policy.c index 58faf1ddd2b1..ab7a01029d49 100644 --- a/net/ipv4/xfrm4_policy.c +++ b/net/ipv4/xfrm4_policy.c @@ -141,7 +141,7 @@ static const struct xfrm_policy_afinfo xfrm4_policy_afinfo = { }; #ifdef CONFIG_SYSCTL -static struct ctl_table xfrm4_policy_table[] = { +static const struct ctl_table xfrm4_policy_table[] = { { .procname = "xfrm4_gc_thresh", .data = &init_net.xfrm.xfrm4_dst_ops.gc_thresh, @@ -151,18 +151,30 @@ static struct ctl_table xfrm4_policy_table[] = { }, }; -static __net_init int xfrm4_net_sysctl_init(struct net *net) +static const struct ctl_table *xfrm4_policy_table_dup(struct net *net) { struct ctl_table *table; + + table = kmemdup(xfrm4_policy_table, sizeof(xfrm4_policy_table), + GFP_KERNEL); + if (!table) + return NULL; + + table[0].data = &net->xfrm.xfrm4_dst_ops.gc_thresh; + + return table; +} + +static __net_init int xfrm4_net_sysctl_init(struct net *net) +{ + const struct ctl_table *table; struct ctl_table_header *hdr; table = xfrm4_policy_table; if (!net_eq(net, &init_net)) { - table = kmemdup(table, sizeof(xfrm4_policy_table), GFP_KERNEL); + table = xfrm4_policy_table_dup(net); if (!table) goto err_alloc; - - table[0].data = &net->xfrm.xfrm4_dst_ops.gc_thresh; } hdr = register_net_sysctl_sz(net, "net/ipv4", table, diff --git a/net/ipv6/xfrm6_policy.c b/net/ipv6/xfrm6_policy.c index 3b749475f6ed..5ec063cb4aa4 100644 --- a/net/ipv6/xfrm6_policy.c +++ b/net/ipv6/xfrm6_policy.c @@ -187,7 +187,7 @@ static void xfrm6_policy_fini(void) } #ifdef CONFIG_SYSCTL -static struct ctl_table xfrm6_policy_table[] = { +static const struct ctl_table xfrm6_policy_table[] = { { .procname = "xfrm6_gc_thresh", .data = &init_net.xfrm.xfrm6_dst_ops.gc_thresh, @@ -197,18 +197,30 @@ static struct ctl_table xfrm6_policy_table[] = { }, }; -static int __net_init xfrm6_net_sysctl_init(struct net *net) +static const struct ctl_table *xfrm6_policy_table_dup(struct net *net) { struct ctl_table *table; + + table = kmemdup(xfrm6_policy_table, sizeof(xfrm6_policy_table), + GFP_KERNEL); + if (!table) + return NULL; + + table[0].data = &net->xfrm.xfrm6_dst_ops.gc_thresh; + + return table; +} + +static int __net_init xfrm6_net_sysctl_init(struct net *net) +{ + const struct ctl_table *table; struct ctl_table_header *hdr; table = xfrm6_policy_table; if (!net_eq(net, &init_net)) { - table = kmemdup(table, sizeof(xfrm6_policy_table), GFP_KERNEL); + table = xfrm6_policy_table_dup(net); if (!table) goto err_alloc; - - table[0].data = &net->xfrm.xfrm6_dst_ops.gc_thresh; } hdr = register_net_sysctl_sz(net, "net/ipv6", table, diff --git a/net/netfilter/nf_hooks_lwtunnel.c b/net/netfilter/nf_hooks_lwtunnel.c index 2d890dd04ff8..4e1eef1ba0f1 100644 --- a/net/netfilter/nf_hooks_lwtunnel.c +++ b/net/netfilter/nf_hooks_lwtunnel.c @@ -54,7 +54,7 @@ int nf_hooks_lwtunnel_sysctl_handler(const struct ctl_table *table, int write, } EXPORT_SYMBOL_GPL(nf_hooks_lwtunnel_sysctl_handler); -static struct ctl_table nf_lwtunnel_sysctl_table[] = { +static const struct ctl_table nf_lwtunnel_sysctl_table[] = { { .procname = "nf_hooks_lwtunnel", .data = NULL, @@ -66,8 +66,8 @@ static struct ctl_table nf_lwtunnel_sysctl_table[] = { static int __net_init nf_lwtunnel_net_init(struct net *net) { + const struct ctl_table *table; struct ctl_table_header *hdr; - struct ctl_table *table; table = nf_lwtunnel_sysctl_table; if (!net_eq(net, &init_net)) { diff --git a/net/smc/smc_sysctl.c b/net/smc/smc_sysctl.c index b1efed546243..09dad48337f6 100644 --- a/net/smc/smc_sysctl.c +++ b/net/smc/smc_sysctl.c @@ -97,7 +97,7 @@ static int proc_smc_hs_ctrl(const struct ctl_table *ctl, int write, } #endif /* CONFIG_SMC_HS_CTRL_BPF */ -static struct ctl_table smc_table[] = { +static const struct ctl_table smc_table[] = { { .procname = "autocorking_size", .data = &init_net.smc.sysctl_autocorking_size, @@ -195,14 +195,29 @@ static struct ctl_table smc_table[] = { #endif /* CONFIG_SMC_HS_CTRL_BPF */ }; -int __net_init smc_sysctl_net_init(struct net *net) +static const struct ctl_table *smc_table_dup(struct net *net) { size_t table_size = ARRAY_SIZE(smc_table); struct ctl_table *table; + int i; + + table = kmemdup(smc_table, sizeof(smc_table), GFP_KERNEL); + if (!table) + return NULL; + + for (i = 0; i < table_size; i++) + table[i].data += (void *)net - (void *)&init_net; + + return table; +} + +int __net_init smc_sysctl_net_init(struct net *net) +{ + size_t table_size = ARRAY_SIZE(smc_table); + const struct ctl_table *table; table = smc_table; if (!net_eq(net, &init_net)) { - int i; #if IS_ENABLED(CONFIG_SMC_HS_CTRL_BPF) struct smc_hs_ctrl *ctrl; @@ -214,12 +229,9 @@ int __net_init smc_sysctl_net_init(struct net *net) rcu_read_unlock(); #endif /* CONFIG_SMC_HS_CTRL_BPF */ - table = kmemdup(table, sizeof(smc_table), GFP_KERNEL); + table = smc_table_dup(net); if (!table) goto err_alloc; - - for (i = 0; i < table_size; i++) - table[i].data += (void *)net - (void *)&init_net; } net->smc.smc_hdr = register_net_sysctl_sz(net, "net/smc", table, diff --git a/net/unix/sysctl_net_unix.c b/net/unix/sysctl_net_unix.c index e02ed6e3955c..47660d5726bb 100644 --- a/net/unix/sysctl_net_unix.c +++ b/net/unix/sysctl_net_unix.c @@ -13,7 +13,7 @@ #include "af_unix.h" -static struct ctl_table unix_table[] = { +static const struct ctl_table unix_table[] = { { .procname = "max_dgram_qlen", .data = &init_net.unx.sysctl_max_dgram_qlen, @@ -23,18 +23,29 @@ static struct ctl_table unix_table[] = { }, }; -int __net_init unix_sysctl_register(struct net *net) +static const struct ctl_table *unix_table_dup(struct net *net) { struct ctl_table *table; + table = kmemdup(unix_table, sizeof(unix_table), GFP_KERNEL); + if (!table) + return NULL; + + table[0].data = &net->unx.sysctl_max_dgram_qlen; + + return table; +} + +int __net_init unix_sysctl_register(struct net *net) +{ + const struct ctl_table *table; + if (net_eq(net, &init_net)) { table = unix_table; } else { - table = kmemdup(unix_table, sizeof(unix_table), GFP_KERNEL); + table = unix_table_dup(net); if (!table) goto err_alloc; - - table[0].data = &net->unx.sysctl_max_dgram_qlen; } net->unx.ctl = register_net_sysctl_sz(net, "net/unix", table, diff --git a/net/vmw_vsock/af_vsock.c b/net/vmw_vsock/af_vsock.c index 622dbd046799..caebef73ea58 100644 --- a/net/vmw_vsock/af_vsock.c +++ b/net/vmw_vsock/af_vsock.c @@ -2899,7 +2899,7 @@ static int vsock_net_child_mode_string(const struct ctl_table *table, int write, return 0; } -static struct ctl_table vsock_table[] = { +static const struct ctl_table vsock_table[] = { { .procname = "ns_mode", .data = &init_net.vsock.mode, @@ -2925,20 +2925,31 @@ static struct ctl_table vsock_table[] = { }, }; -static int __net_init vsock_sysctl_register(struct net *net) +static const struct ctl_table *vsock_table_dup(struct net *net) { struct ctl_table *table; + table = kmemdup(vsock_table, sizeof(vsock_table), GFP_KERNEL); + if (!table) + return NULL; + + table[0].data = &net->vsock.mode; + table[1].data = &net->vsock.child_ns_mode; + table[2].data = &net->vsock.g2h_fallback; + + return table; +} + +static int __net_init vsock_sysctl_register(struct net *net) +{ + const struct ctl_table *table; + if (net_eq(net, &init_net)) { table = vsock_table; } else { - table = kmemdup(vsock_table, sizeof(vsock_table), GFP_KERNEL); + table = vsock_table_dup(net); if (!table) goto err_alloc; - - table[0].data = &net->vsock.mode; - table[1].data = &net->vsock.child_ns_mode; - table[2].data = &net->vsock.g2h_fallback; } net->vsock.sysctl_hdr = register_net_sysctl_sz(net, "net/vsock", table,