diff --git a/include/net/net_namespace.h b/include/net/net_namespace.h index 501af1999fe8..e5ee673b9fcf 100644 --- a/include/net/net_namespace.h +++ b/include/net/net_namespace.h @@ -525,12 +525,13 @@ struct ctl_table; #ifdef CONFIG_SYSCTL int net_sysctl_init(void); struct ctl_table_header *register_net_sysctl_sz(struct net *net, const char *path, - struct ctl_table *table, size_t table_size); + const struct ctl_table *table, + size_t table_size); void unregister_net_sysctl_table(struct ctl_table_header *header); #else static inline int net_sysctl_init(void) { return 0; } static inline struct ctl_table_header *register_net_sysctl_sz(struct net *net, - const char *path, struct ctl_table *table, size_t table_size) + const char *path, const struct ctl_table *table, size_t table_size) { return NULL; } diff --git a/net/core/sysctl_net_core.c b/net/core/sysctl_net_core.c index b508618bfc12..eb35da3556f4 100644 --- a/net/core/sysctl_net_core.c +++ b/net/core/sysctl_net_core.c @@ -678,7 +678,7 @@ static struct ctl_table net_core_table[] = { }, }; -static struct ctl_table netns_core_table[] = { +static const struct ctl_table netns_core_table[] = { #if IS_ENABLED(CONFIG_RPS) { .procname = "rps_default_mask", @@ -787,26 +787,38 @@ static int __init fb_tunnels_only_for_init_net_sysctl_setup(char *str) } __setup("fb_tunnels=", fb_tunnels_only_for_init_net_sysctl_setup); -static __net_init int sysctl_core_net_init(struct net *net) +static const struct ctl_table *netns_core_table_dup(struct net *net) { size_t table_size = ARRAY_SIZE(netns_core_table); struct ctl_table *tbl; + int i; + + tbl = kmemdup(netns_core_table, sizeof(netns_core_table), GFP_KERNEL); + if (!tbl) + return NULL; + + for (i = 0; i < table_size; ++i) { + if (tbl[i].data == &sysctl_wmem_max) + break; + + tbl[i].data += (char *)net - (char *)&init_net; + } + for (; i < table_size; ++i) + tbl[i].mode &= ~0222; + + return tbl; +} + +static __net_init int sysctl_core_net_init(struct net *net) +{ + size_t table_size = ARRAY_SIZE(netns_core_table); + const struct ctl_table *tbl; tbl = netns_core_table; if (!net_eq(net, &init_net)) { - int i; - tbl = kmemdup(tbl, sizeof(netns_core_table), GFP_KERNEL); + tbl = netns_core_table_dup(net); if (tbl == NULL) goto err_dup; - - for (i = 0; i < table_size; ++i) { - if (tbl[i].data == &sysctl_wmem_max) - break; - - tbl[i].data += (char *)net - (char *)&init_net; - } - for (; i < table_size; ++i) - tbl[i].mode &= ~0222; } net->core.sysctl_hdr = register_net_sysctl_sz(net, "net/core", tbl, table_size); diff --git a/net/ipv4/devinet.c b/net/ipv4/devinet.c index 47ded0f607d4..a90be57c63be 100644 --- a/net/ipv4/devinet.c +++ b/net/ipv4/devinet.c @@ -2796,7 +2796,7 @@ static void devinet_sysctl_unregister(struct in_device *idev) neigh_sysctl_unregister(idev->arp_parms); } -static struct ctl_table ctl_forward_entry[] = { +static const struct ctl_table ctl_forward_entry[] = { { .procname = "ip_forward", .data = &ipv4_devconf.data[ diff --git a/net/ipv4/sysctl_net_ipv4.c b/net/ipv4/sysctl_net_ipv4.c index ca1180dba1de..2f0363bca2a8 100644 --- a/net/ipv4/sysctl_net_ipv4.c +++ b/net/ipv4/sysctl_net_ipv4.c @@ -624,7 +624,7 @@ static struct ctl_table ipv4_table[] = { }, }; -static struct ctl_table ipv4_net_table[] = { +static const struct ctl_table ipv4_net_table[] = { { .procname = "tcp_max_tw_buckets", .data = &init_net.ipv4.tcp_death_row.sysctl_max_tw_buckets, @@ -1654,35 +1654,45 @@ static struct ctl_table ipv4_net_table[] = { }, }; -static __net_init int ipv4_sysctl_init_net(struct net *net) +static const struct ctl_table *ipv4_net_table_dup(struct net *net) { size_t table_size = ARRAY_SIZE(ipv4_net_table); struct ctl_table *table; + int i; + + table = kmemdup(ipv4_net_table, sizeof(ipv4_net_table), GFP_KERNEL); + if (!table) + return NULL; + + for (i = 0; i < table_size; i++) { + if (table[i].data) { + /* Update the variables to point into + * the current struct net + */ + table[i].data += (void *)net - (void *)&init_net; + } else { + /* Entries without data pointer are global; + * Make them read-only in non-init_net ns + */ + table[i].mode &= ~0222; + } + if (table[i].extra2 >= (void *)&init_net.ipv4 && + table[i].extra2 < (void *)(&init_net.ipv4 + 1)) + table[i].extra2 += (void *)net - (void *)&init_net; + } + return table; +} + +static __net_init int ipv4_sysctl_init_net(struct net *net) +{ + size_t table_size = ARRAY_SIZE(ipv4_net_table); + const struct ctl_table *table; table = ipv4_net_table; if (!net_eq(net, &init_net)) { - int i; - - table = kmemdup(table, sizeof(ipv4_net_table), GFP_KERNEL); + table = ipv4_net_table_dup(net); if (!table) goto err_alloc; - - for (i = 0; i < table_size; i++) { - if (table[i].data) { - /* Update the variables to point into - * the current struct net - */ - table[i].data += (void *)net - (void *)&init_net; - } else { - /* Entries without data pointer are global; - * Make them read-only in non-init_net ns - */ - table[i].mode &= ~0222; - } - if (table[i].extra2 >= (void *)&init_net.ipv4 && - table[i].extra2 < (void *)(&init_net.ipv4 + 1)) - table[i].extra2 += (void *)net - (void *)&init_net; - } } net->ipv4.ipv4_hdr = register_net_sysctl_sz(net, "net/ipv4", table, diff --git a/net/ipv4/xfrm4_policy.c b/net/ipv4/xfrm4_policy.c index 58faf1ddd2b1..ab7a01029d49 100644 --- a/net/ipv4/xfrm4_policy.c +++ b/net/ipv4/xfrm4_policy.c @@ -141,7 +141,7 @@ static const struct xfrm_policy_afinfo xfrm4_policy_afinfo = { }; #ifdef CONFIG_SYSCTL -static struct ctl_table xfrm4_policy_table[] = { +static const struct ctl_table xfrm4_policy_table[] = { { .procname = "xfrm4_gc_thresh", .data = &init_net.xfrm.xfrm4_dst_ops.gc_thresh, @@ -151,18 +151,30 @@ static struct ctl_table xfrm4_policy_table[] = { }, }; -static __net_init int xfrm4_net_sysctl_init(struct net *net) +static const struct ctl_table *xfrm4_policy_table_dup(struct net *net) { struct ctl_table *table; + + table = kmemdup(xfrm4_policy_table, sizeof(xfrm4_policy_table), + GFP_KERNEL); + if (!table) + return NULL; + + table[0].data = &net->xfrm.xfrm4_dst_ops.gc_thresh; + + return table; +} + +static __net_init int xfrm4_net_sysctl_init(struct net *net) +{ + const struct ctl_table *table; struct ctl_table_header *hdr; table = xfrm4_policy_table; if (!net_eq(net, &init_net)) { - table = kmemdup(table, sizeof(xfrm4_policy_table), GFP_KERNEL); + table = xfrm4_policy_table_dup(net); if (!table) goto err_alloc; - - table[0].data = &net->xfrm.xfrm4_dst_ops.gc_thresh; } hdr = register_net_sysctl_sz(net, "net/ipv4", table, diff --git a/net/ipv6/icmp.c b/net/ipv6/icmp.c index efb23807a026..a95b0351824f 100644 --- a/net/ipv6/icmp.c +++ b/net/ipv6/icmp.c @@ -1374,7 +1374,7 @@ EXPORT_SYMBOL(icmpv6_err_convert); static u32 icmpv6_errors_extension_mask_all = GENMASK_U8(ICMP_ERR_EXT_COUNT - 1, 0); -static struct ctl_table ipv6_icmp_table_template[] = { +static const struct ctl_table ipv6_icmp_table_template[] = { { .procname = "ratelimit", .data = &init_net.ipv6.sysctl.icmpv6_time, diff --git a/net/ipv6/route.c b/net/ipv6/route.c index ae2f93a19dd9..16dfac54a259 100644 --- a/net/ipv6/route.c +++ b/net/ipv6/route.c @@ -6590,7 +6590,7 @@ static int ipv6_sysctl_rtcache_flush(const struct ctl_table *ctl, int write, return 0; } -static struct ctl_table ipv6_route_table_template[] = { +static const struct ctl_table ipv6_route_table_template[] = { { .procname = "max_size", .data = &init_net.ipv6.sysctl.ip6_rt_max_size, diff --git a/net/ipv6/sysctl_net_ipv6.c b/net/ipv6/sysctl_net_ipv6.c index d2cd33e2698d..1a0a36dcdabc 100644 --- a/net/ipv6/sysctl_net_ipv6.c +++ b/net/ipv6/sysctl_net_ipv6.c @@ -61,7 +61,7 @@ proc_rt6_multipath_hash_fields(const struct ctl_table *table, int write, void *b return ret; } -static struct ctl_table ipv6_table_template[] = { +static const struct ctl_table ipv6_table_template[] = { { .procname = "bindv6only", .data = &init_net.ipv6.sysctl.bindv6only, diff --git a/net/ipv6/xfrm6_policy.c b/net/ipv6/xfrm6_policy.c index 3b749475f6ed..5ec063cb4aa4 100644 --- a/net/ipv6/xfrm6_policy.c +++ b/net/ipv6/xfrm6_policy.c @@ -187,7 +187,7 @@ static void xfrm6_policy_fini(void) } #ifdef CONFIG_SYSCTL -static struct ctl_table xfrm6_policy_table[] = { +static const struct ctl_table xfrm6_policy_table[] = { { .procname = "xfrm6_gc_thresh", .data = &init_net.xfrm.xfrm6_dst_ops.gc_thresh, @@ -197,18 +197,30 @@ static struct ctl_table xfrm6_policy_table[] = { }, }; -static int __net_init xfrm6_net_sysctl_init(struct net *net) +static const struct ctl_table *xfrm6_policy_table_dup(struct net *net) { struct ctl_table *table; + + table = kmemdup(xfrm6_policy_table, sizeof(xfrm6_policy_table), + GFP_KERNEL); + if (!table) + return NULL; + + table[0].data = &net->xfrm.xfrm6_dst_ops.gc_thresh; + + return table; +} + +static int __net_init xfrm6_net_sysctl_init(struct net *net) +{ + const struct ctl_table *table; struct ctl_table_header *hdr; table = xfrm6_policy_table; if (!net_eq(net, &init_net)) { - table = kmemdup(table, sizeof(xfrm6_policy_table), GFP_KERNEL); + table = xfrm6_policy_table_dup(net); if (!table) goto err_alloc; - - table[0].data = &net->xfrm.xfrm6_dst_ops.gc_thresh; } hdr = register_net_sysctl_sz(net, "net/ipv6", table, diff --git a/net/netfilter/nf_conntrack_standalone.c b/net/netfilter/nf_conntrack_standalone.c index be2953c7d702..f4f2d82192d5 100644 --- a/net/netfilter/nf_conntrack_standalone.c +++ b/net/netfilter/nf_conntrack_standalone.c @@ -639,7 +639,7 @@ enum nf_ct_sysctl_index { NF_SYSCTL_CT_LAST_SYSCTL, }; -static struct ctl_table nf_ct_sysctl_table[] = { +static const struct ctl_table nf_ct_sysctl_table[] = { [NF_SYSCTL_CT_MAX] = { .procname = "nf_conntrack_max", .data = &nf_conntrack_max, diff --git a/net/netfilter/nf_hooks_lwtunnel.c b/net/netfilter/nf_hooks_lwtunnel.c index 2d890dd04ff8..4e1eef1ba0f1 100644 --- a/net/netfilter/nf_hooks_lwtunnel.c +++ b/net/netfilter/nf_hooks_lwtunnel.c @@ -54,7 +54,7 @@ int nf_hooks_lwtunnel_sysctl_handler(const struct ctl_table *table, int write, } EXPORT_SYMBOL_GPL(nf_hooks_lwtunnel_sysctl_handler); -static struct ctl_table nf_lwtunnel_sysctl_table[] = { +static const struct ctl_table nf_lwtunnel_sysctl_table[] = { { .procname = "nf_hooks_lwtunnel", .data = NULL, @@ -66,8 +66,8 @@ static struct ctl_table nf_lwtunnel_sysctl_table[] = { static int __net_init nf_lwtunnel_net_init(struct net *net) { + const struct ctl_table *table; struct ctl_table_header *hdr; - struct ctl_table *table; table = nf_lwtunnel_sysctl_table; if (!net_eq(net, &init_net)) { diff --git a/net/sctp/sysctl.c b/net/sctp/sysctl.c index fca840484ebf..2b94c211427d 100644 --- a/net/sctp/sysctl.c +++ b/net/sctp/sysctl.c @@ -92,7 +92,7 @@ static struct ctl_table sctp_table[] = { #define SCTP_PF_RETRANS_IDX 2 #define SCTP_PS_RETRANS_IDX 3 -static struct ctl_table sctp_net_table[] = { +static const struct ctl_table sctp_net_table[] = { [SCTP_RTO_MIN_IDX] = { .procname = "rto_min", .data = &init_net.sctp.rto_min, diff --git a/net/smc/smc_sysctl.c b/net/smc/smc_sysctl.c index b1efed546243..09dad48337f6 100644 --- a/net/smc/smc_sysctl.c +++ b/net/smc/smc_sysctl.c @@ -97,7 +97,7 @@ static int proc_smc_hs_ctrl(const struct ctl_table *ctl, int write, } #endif /* CONFIG_SMC_HS_CTRL_BPF */ -static struct ctl_table smc_table[] = { +static const struct ctl_table smc_table[] = { { .procname = "autocorking_size", .data = &init_net.smc.sysctl_autocorking_size, @@ -195,14 +195,29 @@ static struct ctl_table smc_table[] = { #endif /* CONFIG_SMC_HS_CTRL_BPF */ }; -int __net_init smc_sysctl_net_init(struct net *net) +static const struct ctl_table *smc_table_dup(struct net *net) { size_t table_size = ARRAY_SIZE(smc_table); struct ctl_table *table; + int i; + + table = kmemdup(smc_table, sizeof(smc_table), GFP_KERNEL); + if (!table) + return NULL; + + for (i = 0; i < table_size; i++) + table[i].data += (void *)net - (void *)&init_net; + + return table; +} + +int __net_init smc_sysctl_net_init(struct net *net) +{ + size_t table_size = ARRAY_SIZE(smc_table); + const struct ctl_table *table; table = smc_table; if (!net_eq(net, &init_net)) { - int i; #if IS_ENABLED(CONFIG_SMC_HS_CTRL_BPF) struct smc_hs_ctrl *ctrl; @@ -214,12 +229,9 @@ int __net_init smc_sysctl_net_init(struct net *net) rcu_read_unlock(); #endif /* CONFIG_SMC_HS_CTRL_BPF */ - table = kmemdup(table, sizeof(smc_table), GFP_KERNEL); + table = smc_table_dup(net); if (!table) goto err_alloc; - - for (i = 0; i < table_size; i++) - table[i].data += (void *)net - (void *)&init_net; } net->smc.smc_hdr = register_net_sysctl_sz(net, "net/smc", table, diff --git a/net/sysctl_net.c b/net/sysctl_net.c index 19e8048241ba..e190a639eef2 100644 --- a/net/sysctl_net.c +++ b/net/sysctl_net.c @@ -114,16 +114,17 @@ __init int net_sysctl_init(void) goto out; } -/* Verify that sysctls for non-init netns are safe by either: +/* Return error when sysctls for non-init netns are unsafe by verifying: * 1) being read-only, or * 2) having a data pointer which points outside of the global kernel/module * data segment, and rather into the heap where a per-net object was * allocated. */ -static void ensure_safe_net_sysctl(struct net *net, const char *path, - struct ctl_table *table, size_t table_size) +static int ensure_safe_net_sysctl(struct net *net, const char *path, + const struct ctl_table *table, + size_t table_size) { - struct ctl_table *ent; + const struct ctl_table *ent; pr_debug("Registering net sysctl (net %p): %s\n", net, path); ent = table; @@ -149,24 +150,24 @@ static void ensure_safe_net_sysctl(struct net *net, const char *path, else continue; - /* If it is writable and points to kernel/module global - * data, then it's probably a netns leak. - */ + /* Warn on netns leak. */ WARN(1, "sysctl %s/%s: data points to %s global data: %ps\n", path, ent->procname, where, ent->data); - /* Make it "safe" by dropping writable perms */ - ent->mode &= ~0222; + return -EACCES; } + + return 0; } struct ctl_table_header *register_net_sysctl_sz(struct net *net, const char *path, - struct ctl_table *table, + const struct ctl_table *table, size_t table_size) { if (!net_eq(net, &init_net)) - ensure_safe_net_sysctl(net, path, table, table_size); + if (ensure_safe_net_sysctl(net, path, table, table_size)) + return NULL; return __register_sysctl_table(&net->sysctls, path, table, table_size); } diff --git a/net/unix/sysctl_net_unix.c b/net/unix/sysctl_net_unix.c index e02ed6e3955c..47660d5726bb 100644 --- a/net/unix/sysctl_net_unix.c +++ b/net/unix/sysctl_net_unix.c @@ -13,7 +13,7 @@ #include "af_unix.h" -static struct ctl_table unix_table[] = { +static const struct ctl_table unix_table[] = { { .procname = "max_dgram_qlen", .data = &init_net.unx.sysctl_max_dgram_qlen, @@ -23,18 +23,29 @@ static struct ctl_table unix_table[] = { }, }; -int __net_init unix_sysctl_register(struct net *net) +static const struct ctl_table *unix_table_dup(struct net *net) { struct ctl_table *table; + table = kmemdup(unix_table, sizeof(unix_table), GFP_KERNEL); + if (!table) + return NULL; + + table[0].data = &net->unx.sysctl_max_dgram_qlen; + + return table; +} + +int __net_init unix_sysctl_register(struct net *net) +{ + const struct ctl_table *table; + if (net_eq(net, &init_net)) { table = unix_table; } else { - table = kmemdup(unix_table, sizeof(unix_table), GFP_KERNEL); + table = unix_table_dup(net); if (!table) goto err_alloc; - - table[0].data = &net->unx.sysctl_max_dgram_qlen; } net->unx.ctl = register_net_sysctl_sz(net, "net/unix", table, diff --git a/net/vmw_vsock/af_vsock.c b/net/vmw_vsock/af_vsock.c index 622dbd046799..caebef73ea58 100644 --- a/net/vmw_vsock/af_vsock.c +++ b/net/vmw_vsock/af_vsock.c @@ -2899,7 +2899,7 @@ static int vsock_net_child_mode_string(const struct ctl_table *table, int write, return 0; } -static struct ctl_table vsock_table[] = { +static const struct ctl_table vsock_table[] = { { .procname = "ns_mode", .data = &init_net.vsock.mode, @@ -2925,20 +2925,31 @@ static struct ctl_table vsock_table[] = { }, }; -static int __net_init vsock_sysctl_register(struct net *net) +static const struct ctl_table *vsock_table_dup(struct net *net) { struct ctl_table *table; + table = kmemdup(vsock_table, sizeof(vsock_table), GFP_KERNEL); + if (!table) + return NULL; + + table[0].data = &net->vsock.mode; + table[1].data = &net->vsock.child_ns_mode; + table[2].data = &net->vsock.g2h_fallback; + + return table; +} + +static int __net_init vsock_sysctl_register(struct net *net) +{ + const struct ctl_table *table; + if (net_eq(net, &init_net)) { table = vsock_table; } else { - table = kmemdup(vsock_table, sizeof(vsock_table), GFP_KERNEL); + table = vsock_table_dup(net); if (!table) goto err_alloc; - - table[0].data = &net->vsock.mode; - table[1].data = &net->vsock.child_ns_mode; - table[2].data = &net->vsock.g2h_fallback; } net->vsock.sysctl_hdr = register_net_sysctl_sz(net, "net/vsock", table, diff --git a/net/xfrm/xfrm_sysctl.c b/net/xfrm/xfrm_sysctl.c index ca003e8a0376..357152a50faf 100644 --- a/net/xfrm/xfrm_sysctl.c +++ b/net/xfrm/xfrm_sysctl.c @@ -13,7 +13,7 @@ static void __net_init __xfrm_sysctl_init(struct net *net) } #ifdef CONFIG_SYSCTL -static struct ctl_table xfrm_table[] = { +static const struct ctl_table xfrm_table[] = { { .procname = "xfrm_aevent_etime", .maxlen = sizeof(u32),