Merge branch 'net-sysctl-const-qualify-sysctl-ctl_table-arrays'

Joel Granados says:

====================
net: sysctl: Const Qualify sysctl ctl_table arrays

What?
=====
We do two things:
1. Reject netns-unsafe: Replace warning and file permission change with
   an error (reject registration) when an "unsafe" net sysctl
   registration is detected.
2. Const qualify: Const qualify network templated ctl_table arrays and
   unconditional kmemdup'ed ctl_table arrays.

Why?
====
The main motivation for this is to continue with the const qualification
of the ctl_table arrays [1]. The permission change inside
ensure_safe_net_sysctl disallows cons qualifiaction as it basically
modifies the entries before running the sysctl registration.

      ent->mode &= ~0222;

On reject netns-unsafe?
=======================
* I believe that there is currently now way that the permission change
  gets executed [2]
* I found one case where the warning message was posted to lore
  (vsock_sysctl_register) [3], but it made its to mainline as part of
  the second case in [2].
* We should error anyway because writing to the global sysctl value
  through a child netns is indicative of a bug [4].

On Const qualification?
=======================
We can separate the places where network registers sysctl tables into
three groups:
1. Static global: The unchanged global static arrays are passed along to
   sysctl register.
2. Always kmemdup: The global static arrays are always kmemdup'ed before
   passing them along to sysctl register.
3. Dynamic global: The global static array is changed in place before
   passing it along to sysctl register.

This series handles case 1 and 2. It leaves 3 for a later point as
const qualifying those global ctl_tables is more involved.

I would be very thankful if you point me to anything that I have missed
in my analysis that shows that this cannot/shouldn't be done.

[1]
  https://git.kernel.org/pub/scm/linux/kernel/git/sysctl/sysctl.git/commit/?h=constfy-sysctl-6.14-rc1&id=1751f872cc97f992ed5c4c72c55588db1f0021e1

[2]
  I have identified 4 contexts relevant to the ensure_safe_net_sysctl call
  inside the network sysctl registration.

  1. When the (struct net) == &init_net (like in iw_cm_init): In this case
     ensure_safe_net_sysctl is not executed and permission modification
     never happens.

  2. When the ctl_table data (->data) gets "manually" assigned to
     something other init_net (like in vsock_sysctl_register): In this
     case ensure_safe_net_sysctl *is* executed but the data that is passed
     is neither a module address (!is_module_address) nor a kernel core
     address (!is_kernel_core_data); so the permission modification never
     happens.

  3. When the permissions are explicitly changed on a kmemdup'ed ctl_table
     array (like in sysctl_core_net_init): in this case
     ensure_safe_net_sysctl *is* executed but the permission modification
     never happens as the mode is not writable.

  4. When ctl have custom proc_handlers (like in nf_lwtunnel_net_init): In
     this case ->data is NULL so it is not a module address
     (!is_module_address) nor a kernel core address
     (!is_kernel_core_data), so permission modification never happens.

  It seems like there is no way of executing the permission change in
  ensure_safe_net_sysctl. Please correct me if this is inaccurate and help
  me find the case that I missed.

[3]
  https://lore.kernel.org/all/20260302194926.90378-1-graf@amazon.com/

[4]
  The ensure_safe_net_sysctl function was introduced in Commit:
  31c4d2f160 ("net: Ensure net namespace
  isolation of sysctls") which states that it is trying to prevent a
  leak (indicative of a bug).

[5]
  https://patchwork.kernel.org/project/netdevbpf/patch/20260713-jag-net_const_qualify-v3-1-7289fe9eaea6@kernel.org/
====================

Link: https://patch.msgid.link/20260810-jag-net_const_qualify-v4-0-77e888237c69@kernel.org
Signed-off-by: Paolo Abeni <pabeni@redhat.com>
This commit is contained in:
Paolo Abeni
2026-08-13 13:12:24 +02:00
17 changed files with 168 additions and 86 deletions

View File

@@ -525,12 +525,13 @@ struct ctl_table;
#ifdef CONFIG_SYSCTL
int net_sysctl_init(void);
struct ctl_table_header *register_net_sysctl_sz(struct net *net, const char *path,
struct ctl_table *table, size_t table_size);
const struct ctl_table *table,
size_t table_size);
void unregister_net_sysctl_table(struct ctl_table_header *header);
#else
static inline int net_sysctl_init(void) { return 0; }
static inline struct ctl_table_header *register_net_sysctl_sz(struct net *net,
const char *path, struct ctl_table *table, size_t table_size)
const char *path, const struct ctl_table *table, size_t table_size)
{
return NULL;
}

View File

@@ -678,7 +678,7 @@ static struct ctl_table net_core_table[] = {
},
};
static struct ctl_table netns_core_table[] = {
static const struct ctl_table netns_core_table[] = {
#if IS_ENABLED(CONFIG_RPS)
{
.procname = "rps_default_mask",
@@ -787,26 +787,38 @@ static int __init fb_tunnels_only_for_init_net_sysctl_setup(char *str)
}
__setup("fb_tunnels=", fb_tunnels_only_for_init_net_sysctl_setup);
static __net_init int sysctl_core_net_init(struct net *net)
static const struct ctl_table *netns_core_table_dup(struct net *net)
{
size_t table_size = ARRAY_SIZE(netns_core_table);
struct ctl_table *tbl;
int i;
tbl = kmemdup(netns_core_table, sizeof(netns_core_table), GFP_KERNEL);
if (!tbl)
return NULL;
for (i = 0; i < table_size; ++i) {
if (tbl[i].data == &sysctl_wmem_max)
break;
tbl[i].data += (char *)net - (char *)&init_net;
}
for (; i < table_size; ++i)
tbl[i].mode &= ~0222;
return tbl;
}
static __net_init int sysctl_core_net_init(struct net *net)
{
size_t table_size = ARRAY_SIZE(netns_core_table);
const struct ctl_table *tbl;
tbl = netns_core_table;
if (!net_eq(net, &init_net)) {
int i;
tbl = kmemdup(tbl, sizeof(netns_core_table), GFP_KERNEL);
tbl = netns_core_table_dup(net);
if (tbl == NULL)
goto err_dup;
for (i = 0; i < table_size; ++i) {
if (tbl[i].data == &sysctl_wmem_max)
break;
tbl[i].data += (char *)net - (char *)&init_net;
}
for (; i < table_size; ++i)
tbl[i].mode &= ~0222;
}
net->core.sysctl_hdr = register_net_sysctl_sz(net, "net/core", tbl, table_size);

View File

@@ -2796,7 +2796,7 @@ static void devinet_sysctl_unregister(struct in_device *idev)
neigh_sysctl_unregister(idev->arp_parms);
}
static struct ctl_table ctl_forward_entry[] = {
static const struct ctl_table ctl_forward_entry[] = {
{
.procname = "ip_forward",
.data = &ipv4_devconf.data[

View File

@@ -624,7 +624,7 @@ static struct ctl_table ipv4_table[] = {
},
};
static struct ctl_table ipv4_net_table[] = {
static const struct ctl_table ipv4_net_table[] = {
{
.procname = "tcp_max_tw_buckets",
.data = &init_net.ipv4.tcp_death_row.sysctl_max_tw_buckets,
@@ -1654,35 +1654,45 @@ static struct ctl_table ipv4_net_table[] = {
},
};
static __net_init int ipv4_sysctl_init_net(struct net *net)
static const struct ctl_table *ipv4_net_table_dup(struct net *net)
{
size_t table_size = ARRAY_SIZE(ipv4_net_table);
struct ctl_table *table;
int i;
table = kmemdup(ipv4_net_table, sizeof(ipv4_net_table), GFP_KERNEL);
if (!table)
return NULL;
for (i = 0; i < table_size; i++) {
if (table[i].data) {
/* Update the variables to point into
* the current struct net
*/
table[i].data += (void *)net - (void *)&init_net;
} else {
/* Entries without data pointer are global;
* Make them read-only in non-init_net ns
*/
table[i].mode &= ~0222;
}
if (table[i].extra2 >= (void *)&init_net.ipv4 &&
table[i].extra2 < (void *)(&init_net.ipv4 + 1))
table[i].extra2 += (void *)net - (void *)&init_net;
}
return table;
}
static __net_init int ipv4_sysctl_init_net(struct net *net)
{
size_t table_size = ARRAY_SIZE(ipv4_net_table);
const struct ctl_table *table;
table = ipv4_net_table;
if (!net_eq(net, &init_net)) {
int i;
table = kmemdup(table, sizeof(ipv4_net_table), GFP_KERNEL);
table = ipv4_net_table_dup(net);
if (!table)
goto err_alloc;
for (i = 0; i < table_size; i++) {
if (table[i].data) {
/* Update the variables to point into
* the current struct net
*/
table[i].data += (void *)net - (void *)&init_net;
} else {
/* Entries without data pointer are global;
* Make them read-only in non-init_net ns
*/
table[i].mode &= ~0222;
}
if (table[i].extra2 >= (void *)&init_net.ipv4 &&
table[i].extra2 < (void *)(&init_net.ipv4 + 1))
table[i].extra2 += (void *)net - (void *)&init_net;
}
}
net->ipv4.ipv4_hdr = register_net_sysctl_sz(net, "net/ipv4", table,

View File

@@ -141,7 +141,7 @@ static const struct xfrm_policy_afinfo xfrm4_policy_afinfo = {
};
#ifdef CONFIG_SYSCTL
static struct ctl_table xfrm4_policy_table[] = {
static const struct ctl_table xfrm4_policy_table[] = {
{
.procname = "xfrm4_gc_thresh",
.data = &init_net.xfrm.xfrm4_dst_ops.gc_thresh,
@@ -151,18 +151,30 @@ static struct ctl_table xfrm4_policy_table[] = {
},
};
static __net_init int xfrm4_net_sysctl_init(struct net *net)
static const struct ctl_table *xfrm4_policy_table_dup(struct net *net)
{
struct ctl_table *table;
table = kmemdup(xfrm4_policy_table, sizeof(xfrm4_policy_table),
GFP_KERNEL);
if (!table)
return NULL;
table[0].data = &net->xfrm.xfrm4_dst_ops.gc_thresh;
return table;
}
static __net_init int xfrm4_net_sysctl_init(struct net *net)
{
const struct ctl_table *table;
struct ctl_table_header *hdr;
table = xfrm4_policy_table;
if (!net_eq(net, &init_net)) {
table = kmemdup(table, sizeof(xfrm4_policy_table), GFP_KERNEL);
table = xfrm4_policy_table_dup(net);
if (!table)
goto err_alloc;
table[0].data = &net->xfrm.xfrm4_dst_ops.gc_thresh;
}
hdr = register_net_sysctl_sz(net, "net/ipv4", table,

View File

@@ -1374,7 +1374,7 @@ EXPORT_SYMBOL(icmpv6_err_convert);
static u32 icmpv6_errors_extension_mask_all =
GENMASK_U8(ICMP_ERR_EXT_COUNT - 1, 0);
static struct ctl_table ipv6_icmp_table_template[] = {
static const struct ctl_table ipv6_icmp_table_template[] = {
{
.procname = "ratelimit",
.data = &init_net.ipv6.sysctl.icmpv6_time,

View File

@@ -6590,7 +6590,7 @@ static int ipv6_sysctl_rtcache_flush(const struct ctl_table *ctl, int write,
return 0;
}
static struct ctl_table ipv6_route_table_template[] = {
static const struct ctl_table ipv6_route_table_template[] = {
{
.procname = "max_size",
.data = &init_net.ipv6.sysctl.ip6_rt_max_size,

View File

@@ -61,7 +61,7 @@ proc_rt6_multipath_hash_fields(const struct ctl_table *table, int write, void *b
return ret;
}
static struct ctl_table ipv6_table_template[] = {
static const struct ctl_table ipv6_table_template[] = {
{
.procname = "bindv6only",
.data = &init_net.ipv6.sysctl.bindv6only,

View File

@@ -187,7 +187,7 @@ static void xfrm6_policy_fini(void)
}
#ifdef CONFIG_SYSCTL
static struct ctl_table xfrm6_policy_table[] = {
static const struct ctl_table xfrm6_policy_table[] = {
{
.procname = "xfrm6_gc_thresh",
.data = &init_net.xfrm.xfrm6_dst_ops.gc_thresh,
@@ -197,18 +197,30 @@ static struct ctl_table xfrm6_policy_table[] = {
},
};
static int __net_init xfrm6_net_sysctl_init(struct net *net)
static const struct ctl_table *xfrm6_policy_table_dup(struct net *net)
{
struct ctl_table *table;
table = kmemdup(xfrm6_policy_table, sizeof(xfrm6_policy_table),
GFP_KERNEL);
if (!table)
return NULL;
table[0].data = &net->xfrm.xfrm6_dst_ops.gc_thresh;
return table;
}
static int __net_init xfrm6_net_sysctl_init(struct net *net)
{
const struct ctl_table *table;
struct ctl_table_header *hdr;
table = xfrm6_policy_table;
if (!net_eq(net, &init_net)) {
table = kmemdup(table, sizeof(xfrm6_policy_table), GFP_KERNEL);
table = xfrm6_policy_table_dup(net);
if (!table)
goto err_alloc;
table[0].data = &net->xfrm.xfrm6_dst_ops.gc_thresh;
}
hdr = register_net_sysctl_sz(net, "net/ipv6", table,

View File

@@ -639,7 +639,7 @@ enum nf_ct_sysctl_index {
NF_SYSCTL_CT_LAST_SYSCTL,
};
static struct ctl_table nf_ct_sysctl_table[] = {
static const struct ctl_table nf_ct_sysctl_table[] = {
[NF_SYSCTL_CT_MAX] = {
.procname = "nf_conntrack_max",
.data = &nf_conntrack_max,

View File

@@ -54,7 +54,7 @@ int nf_hooks_lwtunnel_sysctl_handler(const struct ctl_table *table, int write,
}
EXPORT_SYMBOL_GPL(nf_hooks_lwtunnel_sysctl_handler);
static struct ctl_table nf_lwtunnel_sysctl_table[] = {
static const struct ctl_table nf_lwtunnel_sysctl_table[] = {
{
.procname = "nf_hooks_lwtunnel",
.data = NULL,
@@ -66,8 +66,8 @@ static struct ctl_table nf_lwtunnel_sysctl_table[] = {
static int __net_init nf_lwtunnel_net_init(struct net *net)
{
const struct ctl_table *table;
struct ctl_table_header *hdr;
struct ctl_table *table;
table = nf_lwtunnel_sysctl_table;
if (!net_eq(net, &init_net)) {

View File

@@ -92,7 +92,7 @@ static struct ctl_table sctp_table[] = {
#define SCTP_PF_RETRANS_IDX 2
#define SCTP_PS_RETRANS_IDX 3
static struct ctl_table sctp_net_table[] = {
static const struct ctl_table sctp_net_table[] = {
[SCTP_RTO_MIN_IDX] = {
.procname = "rto_min",
.data = &init_net.sctp.rto_min,

View File

@@ -97,7 +97,7 @@ static int proc_smc_hs_ctrl(const struct ctl_table *ctl, int write,
}
#endif /* CONFIG_SMC_HS_CTRL_BPF */
static struct ctl_table smc_table[] = {
static const struct ctl_table smc_table[] = {
{
.procname = "autocorking_size",
.data = &init_net.smc.sysctl_autocorking_size,
@@ -195,14 +195,29 @@ static struct ctl_table smc_table[] = {
#endif /* CONFIG_SMC_HS_CTRL_BPF */
};
int __net_init smc_sysctl_net_init(struct net *net)
static const struct ctl_table *smc_table_dup(struct net *net)
{
size_t table_size = ARRAY_SIZE(smc_table);
struct ctl_table *table;
int i;
table = kmemdup(smc_table, sizeof(smc_table), GFP_KERNEL);
if (!table)
return NULL;
for (i = 0; i < table_size; i++)
table[i].data += (void *)net - (void *)&init_net;
return table;
}
int __net_init smc_sysctl_net_init(struct net *net)
{
size_t table_size = ARRAY_SIZE(smc_table);
const struct ctl_table *table;
table = smc_table;
if (!net_eq(net, &init_net)) {
int i;
#if IS_ENABLED(CONFIG_SMC_HS_CTRL_BPF)
struct smc_hs_ctrl *ctrl;
@@ -214,12 +229,9 @@ int __net_init smc_sysctl_net_init(struct net *net)
rcu_read_unlock();
#endif /* CONFIG_SMC_HS_CTRL_BPF */
table = kmemdup(table, sizeof(smc_table), GFP_KERNEL);
table = smc_table_dup(net);
if (!table)
goto err_alloc;
for (i = 0; i < table_size; i++)
table[i].data += (void *)net - (void *)&init_net;
}
net->smc.smc_hdr = register_net_sysctl_sz(net, "net/smc", table,

View File

@@ -114,16 +114,17 @@ __init int net_sysctl_init(void)
goto out;
}
/* Verify that sysctls for non-init netns are safe by either:
/* Return error when sysctls for non-init netns are unsafe by verifying:
* 1) being read-only, or
* 2) having a data pointer which points outside of the global kernel/module
* data segment, and rather into the heap where a per-net object was
* allocated.
*/
static void ensure_safe_net_sysctl(struct net *net, const char *path,
struct ctl_table *table, size_t table_size)
static int ensure_safe_net_sysctl(struct net *net, const char *path,
const struct ctl_table *table,
size_t table_size)
{
struct ctl_table *ent;
const struct ctl_table *ent;
pr_debug("Registering net sysctl (net %p): %s\n", net, path);
ent = table;
@@ -149,24 +150,24 @@ static void ensure_safe_net_sysctl(struct net *net, const char *path,
else
continue;
/* If it is writable and points to kernel/module global
* data, then it's probably a netns leak.
*/
/* Warn on netns leak. */
WARN(1, "sysctl %s/%s: data points to %s global data: %ps\n",
path, ent->procname, where, ent->data);
/* Make it "safe" by dropping writable perms */
ent->mode &= ~0222;
return -EACCES;
}
return 0;
}
struct ctl_table_header *register_net_sysctl_sz(struct net *net,
const char *path,
struct ctl_table *table,
const struct ctl_table *table,
size_t table_size)
{
if (!net_eq(net, &init_net))
ensure_safe_net_sysctl(net, path, table, table_size);
if (ensure_safe_net_sysctl(net, path, table, table_size))
return NULL;
return __register_sysctl_table(&net->sysctls, path, table, table_size);
}

View File

@@ -13,7 +13,7 @@
#include "af_unix.h"
static struct ctl_table unix_table[] = {
static const struct ctl_table unix_table[] = {
{
.procname = "max_dgram_qlen",
.data = &init_net.unx.sysctl_max_dgram_qlen,
@@ -23,18 +23,29 @@ static struct ctl_table unix_table[] = {
},
};
int __net_init unix_sysctl_register(struct net *net)
static const struct ctl_table *unix_table_dup(struct net *net)
{
struct ctl_table *table;
table = kmemdup(unix_table, sizeof(unix_table), GFP_KERNEL);
if (!table)
return NULL;
table[0].data = &net->unx.sysctl_max_dgram_qlen;
return table;
}
int __net_init unix_sysctl_register(struct net *net)
{
const struct ctl_table *table;
if (net_eq(net, &init_net)) {
table = unix_table;
} else {
table = kmemdup(unix_table, sizeof(unix_table), GFP_KERNEL);
table = unix_table_dup(net);
if (!table)
goto err_alloc;
table[0].data = &net->unx.sysctl_max_dgram_qlen;
}
net->unx.ctl = register_net_sysctl_sz(net, "net/unix", table,

View File

@@ -2899,7 +2899,7 @@ static int vsock_net_child_mode_string(const struct ctl_table *table, int write,
return 0;
}
static struct ctl_table vsock_table[] = {
static const struct ctl_table vsock_table[] = {
{
.procname = "ns_mode",
.data = &init_net.vsock.mode,
@@ -2925,20 +2925,31 @@ static struct ctl_table vsock_table[] = {
},
};
static int __net_init vsock_sysctl_register(struct net *net)
static const struct ctl_table *vsock_table_dup(struct net *net)
{
struct ctl_table *table;
table = kmemdup(vsock_table, sizeof(vsock_table), GFP_KERNEL);
if (!table)
return NULL;
table[0].data = &net->vsock.mode;
table[1].data = &net->vsock.child_ns_mode;
table[2].data = &net->vsock.g2h_fallback;
return table;
}
static int __net_init vsock_sysctl_register(struct net *net)
{
const struct ctl_table *table;
if (net_eq(net, &init_net)) {
table = vsock_table;
} else {
table = kmemdup(vsock_table, sizeof(vsock_table), GFP_KERNEL);
table = vsock_table_dup(net);
if (!table)
goto err_alloc;
table[0].data = &net->vsock.mode;
table[1].data = &net->vsock.child_ns_mode;
table[2].data = &net->vsock.g2h_fallback;
}
net->vsock.sysctl_hdr = register_net_sysctl_sz(net, "net/vsock", table,

View File

@@ -13,7 +13,7 @@ static void __net_init __xfrm_sysctl_init(struct net *net)
}
#ifdef CONFIG_SYSCTL
static struct ctl_table xfrm_table[] = {
static const struct ctl_table xfrm_table[] = {
{
.procname = "xfrm_aevent_etime",
.maxlen = sizeof(u32),