mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git
synced 2026-08-28 16:54:57 -04:00
Move the hash table to the super block to remove excessive overhead in case
of small number of xattrs per inode.
Add linked list to the inode, used for listxattr and eviction. Listxattr
uses rcu protection to iterate the list of xattrs.
Before being made per-sb, lazy allocation was protected by inode lock. Now
inode lock no longer provides sufficient exclusion, so use cmpxchg() to
ensure atomicity.
Though I haven't found a description of this pattern, after some research
it seems that cmpxchg_release() and READ_ONCE() should provide the
necessary memory barriers.
Use simple_xattr_free_rcu() in simple_xattrs_free(). This is needed because
the hash table is now shared between inodes and lookup on a different inode
might be running the compare function on the just freed element within the
RCU grace period.
Following stats are based on slabinfo diff, after creating 100k empty
files, then adding a "user.test=foo" xattr to each:
v7.0 (no rhashtable):
File creation: 993.40 bytes/file
Xattr addition: 79.99 bytes/file
v7.1-rc2 (per-inode rhashtable):
File creation: 939.73 bytes/file
Xattr addition: 1296.08 bytes/file
v7.1-rc2 + this patch (per-sb rhashtable)
File creation: 946.84 bytes/file
Xattr addition: 111.86 bytes/file
The overhead of a single xattr is reduced to nearly v7.0 levels. The per
xattr overhead is slightly larger due to the addition of three pointers to
struct simple_xattr.
Fixes: b32c4a2136 ("xattr: add rhashtable-based simple_xattr infrastructure")
Signed-off-by: Miklos Szeredi <mszeredi@redhat.com>
Link: https://patch.msgid.link/20260605135322.2632068-5-mszeredi@redhat.com
Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
237 lines
6.4 KiB
C
237 lines
6.4 KiB
C
/* SPDX-License-Identifier: GPL-2.0-only */
|
|
/*
|
|
* fs/kernfs/kernfs-internal.h - kernfs internal header file
|
|
*
|
|
* Copyright (c) 2001-3 Patrick Mochel
|
|
* Copyright (c) 2007 SUSE Linux Products GmbH
|
|
* Copyright (c) 2007, 2013 Tejun Heo <teheo@suse.de>
|
|
*/
|
|
|
|
#ifndef __KERNFS_INTERNAL_H
|
|
#define __KERNFS_INTERNAL_H
|
|
|
|
#include <linux/lockdep.h>
|
|
#include <linux/fs.h>
|
|
#include <linux/mutex.h>
|
|
#include <linux/rwsem.h>
|
|
#include <linux/xattr.h>
|
|
|
|
#include <linux/kernfs.h>
|
|
#include <linux/fs_context.h>
|
|
|
|
struct kernfs_iattrs {
|
|
kuid_t ia_uid;
|
|
kgid_t ia_gid;
|
|
struct timespec64 ia_atime;
|
|
struct timespec64 ia_mtime;
|
|
struct timespec64 ia_ctime;
|
|
|
|
struct list_head xattrs;
|
|
struct simple_xattr_limits xattr_limits;
|
|
};
|
|
|
|
struct kernfs_root {
|
|
/* published fields */
|
|
struct kernfs_node *kn;
|
|
unsigned int flags; /* KERNFS_ROOT_* flags */
|
|
|
|
/* private fields, do not use outside kernfs proper */
|
|
struct idr ino_idr;
|
|
spinlock_t kernfs_idr_lock; /* root->ino_idr */
|
|
u32 last_id_lowbits;
|
|
u32 id_highbits;
|
|
struct kernfs_syscall_ops *syscall_ops;
|
|
|
|
/* list of kernfs_super_info of this root, protected by kernfs_rwsem */
|
|
struct list_head supers;
|
|
|
|
wait_queue_head_t deactivate_waitq;
|
|
struct rw_semaphore kernfs_rwsem;
|
|
struct rw_semaphore kernfs_iattr_rwsem;
|
|
struct rw_semaphore kernfs_supers_rwsem;
|
|
|
|
/* kn->parent and kn->name */
|
|
rwlock_t kernfs_rename_lock;
|
|
|
|
struct rcu_head rcu;
|
|
|
|
struct simple_xattr_cache xa_cache;
|
|
};
|
|
|
|
/* +1 to avoid triggering overflow warning when negating it */
|
|
#define KN_DEACTIVATED_BIAS (INT_MIN + 1)
|
|
|
|
/* KERNFS_TYPE_MASK and types are defined in include/linux/kernfs.h */
|
|
|
|
/**
|
|
* kernfs_root - find out the kernfs_root a kernfs_node belongs to
|
|
* @kn: kernfs_node of interest
|
|
*
|
|
* Return: the kernfs_root @kn belongs to.
|
|
*/
|
|
static inline struct kernfs_root *kernfs_root(const struct kernfs_node *kn)
|
|
{
|
|
const struct kernfs_node *knp;
|
|
/* if parent exists, it's always a dir; otherwise, @sd is a dir */
|
|
guard(rcu)();
|
|
knp = rcu_dereference(kn->__parent);
|
|
if (knp)
|
|
kn = knp;
|
|
return kn->dir.root;
|
|
}
|
|
|
|
/*
|
|
* mount.c
|
|
*/
|
|
struct kernfs_super_info {
|
|
struct super_block *sb;
|
|
|
|
/*
|
|
* The root associated with this super_block. Each super_block is
|
|
* identified by the root and ns it's associated with.
|
|
*/
|
|
struct kernfs_root *root;
|
|
|
|
/*
|
|
* Each sb is associated with one namespace tag, currently the
|
|
* network namespace of the task which mounted this kernfs
|
|
* instance. If multiple tags become necessary, make the following
|
|
* an array and compare kernfs_node tag against every entry.
|
|
*/
|
|
const struct ns_common *ns;
|
|
|
|
/* anchored at kernfs_root->supers, protected by kernfs_rwsem */
|
|
struct list_head node;
|
|
};
|
|
#define kernfs_info(SB) ((struct kernfs_super_info *)(SB->s_fs_info))
|
|
|
|
static inline bool kernfs_root_is_locked(const struct kernfs_node *kn)
|
|
{
|
|
return lockdep_is_held(&kernfs_root(kn)->kernfs_rwsem);
|
|
}
|
|
|
|
static inline bool kernfs_rename_is_locked(const struct kernfs_node *kn)
|
|
{
|
|
return lockdep_is_held(&kernfs_root(kn)->kernfs_rename_lock);
|
|
}
|
|
|
|
static inline const char *kernfs_rcu_name(const struct kernfs_node *kn)
|
|
{
|
|
return rcu_dereference_check(kn->name, kernfs_root_is_locked(kn));
|
|
}
|
|
|
|
static inline struct kernfs_node *kernfs_parent(const struct kernfs_node *kn)
|
|
{
|
|
/*
|
|
* The kernfs_node::__parent remains valid within a RCU section. The kn
|
|
* can be reparented (and renamed) which changes the entry. This can be
|
|
* avoided by locking kernfs_root::kernfs_rwsem or
|
|
* kernfs_root::kernfs_rename_lock.
|
|
* Both locks can be used to obtain a reference on __parent. Once the
|
|
* reference count reaches 0 then the node is about to be freed
|
|
* and can not be renamed (or become a different parent) anymore.
|
|
*/
|
|
return rcu_dereference_check(kn->__parent,
|
|
kernfs_root_is_locked(kn) ||
|
|
kernfs_rename_is_locked(kn) ||
|
|
!atomic_read(&kn->count));
|
|
}
|
|
|
|
static inline struct kernfs_node *kernfs_dentry_node(struct dentry *dentry)
|
|
{
|
|
if (d_really_is_negative(dentry))
|
|
return NULL;
|
|
return d_inode(dentry)->i_private;
|
|
}
|
|
|
|
static inline void kernfs_set_rev(struct kernfs_node *parent,
|
|
struct dentry *dentry)
|
|
{
|
|
dentry->d_time = parent->dir.rev;
|
|
}
|
|
|
|
static inline void kernfs_inc_rev(struct kernfs_node *parent)
|
|
{
|
|
parent->dir.rev++;
|
|
}
|
|
|
|
static inline bool kernfs_dir_changed(struct kernfs_node *parent,
|
|
struct dentry *dentry)
|
|
{
|
|
if (parent->dir.rev != dentry->d_time)
|
|
return true;
|
|
return false;
|
|
}
|
|
|
|
extern const struct super_operations kernfs_sops;
|
|
extern struct kmem_cache *kernfs_node_cache, *kernfs_iattrs_cache;
|
|
|
|
/*
|
|
* inode.c
|
|
*/
|
|
extern const struct xattr_handler * const kernfs_xattr_handlers[];
|
|
void kernfs_evict_inode(struct inode *inode);
|
|
int kernfs_iop_permission(struct mnt_idmap *idmap,
|
|
struct inode *inode, int mask);
|
|
int kernfs_iop_setattr(struct mnt_idmap *idmap, struct dentry *dentry,
|
|
struct iattr *iattr);
|
|
int kernfs_iop_getattr(struct mnt_idmap *idmap,
|
|
const struct path *path, struct kstat *stat,
|
|
u32 request_mask, unsigned int query_flags);
|
|
ssize_t kernfs_iop_listxattr(struct dentry *dentry, char *buf, size_t size);
|
|
int __kernfs_setattr(struct kernfs_node *kn, const struct iattr *iattr);
|
|
|
|
/*
|
|
* dir.c
|
|
*/
|
|
extern const struct dentry_operations kernfs_dops;
|
|
extern const struct file_operations kernfs_dir_fops;
|
|
extern const struct inode_operations kernfs_dir_iops;
|
|
|
|
struct kernfs_node *kernfs_get_active(struct kernfs_node *kn);
|
|
void kernfs_put_active(struct kernfs_node *kn);
|
|
int kernfs_add_one(struct kernfs_node *kn);
|
|
struct kernfs_node *kernfs_new_node(struct kernfs_node *parent,
|
|
const char *name, umode_t mode,
|
|
kuid_t uid, kgid_t gid,
|
|
unsigned flags);
|
|
|
|
/*
|
|
* file.c
|
|
*/
|
|
extern const struct file_operations kernfs_file_fops;
|
|
|
|
bool kernfs_should_drain_open_files(struct kernfs_node *kn);
|
|
void kernfs_drain_open_files(struct kernfs_node *kn);
|
|
|
|
/*
|
|
* symlink.c
|
|
*/
|
|
extern const struct inode_operations kernfs_symlink_iops;
|
|
|
|
/*
|
|
* kernfs locks
|
|
*/
|
|
extern struct kernfs_global_locks *kernfs_locks;
|
|
|
|
/* Hashed mutex helpers - protect per-node data structures */
|
|
static inline struct mutex *kernfs_node_lock_ptr(struct kernfs_node *kn)
|
|
{
|
|
int idx = hash_ptr(kn, NR_KERNFS_LOCK_BITS);
|
|
|
|
return &kernfs_locks->node_mutex[idx];
|
|
}
|
|
|
|
static inline struct mutex *kernfs_node_lock(struct kernfs_node *kn)
|
|
{
|
|
struct mutex *lock = kernfs_node_lock_ptr(kn);
|
|
|
|
mutex_lock(lock);
|
|
return lock;
|
|
}
|
|
|
|
DEFINE_CLASS(kernfs_node_lock, struct mutex *,
|
|
mutex_unlock(_T), kernfs_node_lock(kn), struct kernfs_node *kn)
|
|
|
|
#endif /* __KERNFS_INTERNAL_H */
|