mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git
synced 2026-07-23 03:57:32 -04:00
Pull misc vfs updates from Christian Brauner:
"Features:
- Reduce pipe->mutex contention by pre-allocating pages outside the
lock in anon_pipe_write().
anon_pipe_write() called alloc_page() once per page while holding
pipe->mutex. The allocation can sleep doing direct reclaim and runs
memcg charging, which extends the critical section and stalls any
concurrent reader on the same mutex. Now up to 8 pages are
pre-allocated before the mutex is taken, leftovers are recycled
into the per-pipe tmp_page[] cache before unlock, and any remainder
is released after unlock, keeping the allocator out of the critical
section on both sides. On a writers x readers sweep with 64KB
writes against a 1 MB pipe throughput improves 6-28% and average
write latency drops 5-22%; under memory pressure - when the cost of
holding the mutex across reclaim is highest - throughput improves
21-48% and latency drops 17-33%. The microbenchmark is added to
selftests.
- uaccess/sockptr: fix the ignored_trailing logic in
copy_struct_to_user() to behave as documented and the usize check
in copy_struct_from_sockptr() for user pointers, and add
copy_struct_{from,to}_bounce_buffer() and copy_struct_to_sockptr()
helpers for upcoming users (IPPROTO_SMBDIRECT, IPPROTO_QUIC).
- bpf: add a sleepable bpf_real_inode() kfunc that resolves the real
inode backing a dentry via d_real_inode(). On overlayfs the inode
attached to the dentry doesn't carry the underlying device
information; this is used by the filesystem restriction BPF program
that was merged into systemd.
- docs: add guidelines for submitting new filesystems, motivated by
the maintenance burden abandoned and untestable filesystems impose
on VFS developers, blocking infrastructure work like folio
conversions and iomap migration.
Fixes:
- libfs: set SB_I_NOEXEC and SB_I_NODEV by default in init_pseudo()
and drop the now-redundant assignments in callers. This began as a
one-line dma-buf fix for a path_noexec() warning; a pseudo
filesystem has no reason not to set SB_I_NOEXEC. All init_pseudo()
callers were audited: the only visible effect is on dma-buf where
SB_I_NOEXEC silences the warning.
- Handle set_blocksize() failures in legacy filesystems (bfs, hpfs,
qnx4, jfs, befs, affs, isofs, minix, ntfs3, omfs). Mounting a
device with a sector size > PAGE_SIZE crashed roughly half of them;
the rest had the same missing error handling pattern. Plus a
follow-up releasing the superblock buffer_head when setting the
minix v3 block size fails.
- mount: honour SB_NOUSER in the new mount API.
- fs/fcntl: fix a SOFTIRQ-unsafe lock order in fasync signaling by
switching the process-group paths of send_sigio() and send_sigurg()
from read_lock(&tasklist_lock) to RCU, matching the single-PID
path.
- vfs: add an FS_USERNS_DELEGATABLE flag and set it for NFS, fixing
delegated NFS mounts (fsopen() in a container with the mount
performed by a privileged daemon) that broke when non-init
s_user_ns was tied to FS_USERNS_MOUNT.
- selftests/namespaces: fix a hang in nsid_test where an unreaped
grandchild kept the TAP pipe write-end open, a waitpid(-1) race in
listns_efault_test, and a false FAIL on kernels without listns()
where the tests should SKIP.
- filelock: fix the break_lease() stub signature for
CONFIG_FILE_LOCKING=n.
- init/initramfs_test: wait for the async initramfs unpacking before
running; the test and do_populate_rootfs() share the parser state.
- fs/coredump: reduce redundant log noise in
validate_coredump_safety().
- iomap: pass the correct length to fserror_report_io() in
__iomap_write_begin().
- backing-file: fix the backing_file_open() kerneldoc.
Cleanups:
- initramfs: refactor the cpio hex header parsing to use hex2bin()
instead of the hand-rolled simple_strntoul() which is reverted, and
extend the initramfs KUnit tests to cover header fields with 0x
prefixes.
- Replace __get_free_pages() and friends with kmalloc()/kzalloc()
across quota, proc, ocfs2/dlm, nilfs2, nfs, nfsd, libfs, jfs, jbd2,
isofs, fuse, select, namespace, configfs, binfmt_misc, bfs, and the
do_mounts init code - part of the larger work of replacing page
allocator calls with kmalloc().
- Use clear_and_wake_up_bit() in unlock_buffer() and
journal_end_buffer_io_sync() instead of open-coding the sequence.
- Drop unused VFS exports: unexport drop_super_exclusive(), remove
start_removing_user_path_at(), and fold __start_removing_path()
into start_removing_path().
- fs/read_write: narrow the __kernel_write() export with
EXPORT_SYMBOL_FOR_MODULES().
- vfs: uapi: retire octal and hex constants in favor of (1 << n) for
the O_ flags. Finding a free bit for a new flag across the
architectures was needlessly hard with the mixed bases.
- dcache: add extra sanity checks of dead dentries in dentry_free()
via a new DENTRY_WARN_ONCE() that also prints d_flags.
- iov_iter: use kmemdup_array() in dup_iter() to harden the
allocation against multiplication overflow.
- fs/pipe: write to ->poll_usage only once.
- vfs: remove an always-taken if-branch in find_next_fd().
- dcache: use kmalloc_flex() for struct external_name in __d_alloc().
- namei: use QSTR() instead of QSTR_INIT() in path_pts().
- sync_file_range: delete dead S_ISLNK code.
- Comment fixes: retire a stale comment in fget_task_next() and fix
assorted spelling mistakes"
* tag 'vfs-7.2-rc1.misc' of git://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs: (73 commits)
backing-file: fix backing_file_open() kerneldoc parameter
iomap: pass the correct len to fserror_report_io in __iomap_write_begin
vfs: add FS_USERNS_DELEGATABLE flag and set it for NFS
filelock: fix break_lease() stub signature for CONFIG_FILE_LOCKING=n
vfs: uapi: retire octal and hex numbers in favor of (1 << n) for O_ flags
bpf: add bpf_real_inode() kfunc
fs/read_write: Do not export __kernel_write() to the entire world
libfs: drop redundant SB_I_NOEXEC/SB_I_NODEV in init_pseudo() callers
libfs: set SB_I_NOEXEC and SB_I_NODEV by default in init_pseudo()
mount: honour SB_NOUSER in the new mount API
fs/fcntl: fix SOFTIRQ-unsafe lock order in fasync signaling
selftests/pipe: add pipe_bench microbenchmark
fs/pipe: pre-allocate pages outside pipe->mutex in anon_pipe_write
fs: retire stale comment in fget_task_next()
fs: fix spelling mistakes in comment
bfs: replace get_zeroed_page() with kzalloc()
binfmt_misc: replace __get_free_page() with kmalloc()
configfs: replace __get_free_pages() with kzalloc()
fs/namespace: use __getname() to allocate mntpath buffer
fs/select: replace __get_free_page() with kmalloc()
...
304 lines
7.1 KiB
C
304 lines
7.1 KiB
C
// SPDX-License-Identifier: GPL-2.0
|
|
/*
|
|
* linux/fs/isofs/dir.c
|
|
*
|
|
* (C) 1992, 1993, 1994 Eric Youngdale Modified for ISO 9660 filesystem.
|
|
*
|
|
* (C) 1991 Linus Torvalds - minix filesystem
|
|
*
|
|
* Steve Beynon : Missing last directory entries fixed
|
|
* (stephen@askone.demon.co.uk) : 21st June 1996
|
|
*
|
|
* isofs directory handling functions
|
|
*/
|
|
#include <linux/gfp.h>
|
|
#include <linux/filelock.h>
|
|
#include <linux/slab.h>
|
|
#include "isofs.h"
|
|
#include <linux/fileattr.h>
|
|
|
|
int isofs_name_translate(struct iso_directory_record *de, char *new, struct inode *inode)
|
|
{
|
|
char * old = de->name;
|
|
int len = de->name_len[0];
|
|
int i;
|
|
|
|
for (i = 0; i < len; i++) {
|
|
unsigned char c = old[i];
|
|
if (!c)
|
|
break;
|
|
|
|
if (c >= 'A' && c <= 'Z')
|
|
c |= 0x20; /* lower case */
|
|
|
|
/* Drop trailing '.;1' (ISO 9660:1988 7.5.1 requires period) */
|
|
if (c == '.' && i == len - 3 && old[i + 1] == ';' && old[i + 2] == '1')
|
|
break;
|
|
|
|
/* Drop trailing ';1' */
|
|
if (c == ';' && i == len - 2 && old[i + 1] == '1')
|
|
break;
|
|
|
|
/* Convert remaining ';' to '.' */
|
|
/* Also '/' to '.' (broken Acorn-generated ISO9660 images) */
|
|
if (c == ';' || c == '/')
|
|
c = '.';
|
|
|
|
new[i] = c;
|
|
}
|
|
return i;
|
|
}
|
|
|
|
/* Acorn extensions written by Matthew Wilcox <willy@infradead.org> 1998 */
|
|
int get_acorn_filename(struct iso_directory_record *de,
|
|
char *retname, struct inode *inode)
|
|
{
|
|
int std;
|
|
unsigned char *chr;
|
|
int retnamlen = isofs_name_translate(de, retname, inode);
|
|
|
|
if (retnamlen == 0)
|
|
return 0;
|
|
std = sizeof(struct iso_directory_record) + de->name_len[0];
|
|
if (std & 1)
|
|
std++;
|
|
if (de->length[0] - std != 32)
|
|
return retnamlen;
|
|
chr = ((unsigned char *) de) + std;
|
|
if (strncmp(chr, "ARCHIMEDES", 10))
|
|
return retnamlen;
|
|
if ((*retname == '_') && ((chr[19] & 1) == 1))
|
|
*retname = '!';
|
|
if (((de->flags[0] & 2) == 0) && (chr[13] == 0xff)
|
|
&& ((chr[12] & 0xf0) == 0xf0)) {
|
|
retname[retnamlen] = ',';
|
|
sprintf(retname+retnamlen+1, "%3.3x",
|
|
((chr[12] & 0xf) << 8) | chr[11]);
|
|
retnamlen += 4;
|
|
}
|
|
return retnamlen;
|
|
}
|
|
|
|
/*
|
|
* This should _really_ be cleaned up some day..
|
|
*/
|
|
static int do_isofs_readdir(struct inode *inode, struct file *file,
|
|
struct dir_context *ctx,
|
|
char *tmpname, struct iso_directory_record *tmpde)
|
|
{
|
|
unsigned long bufsize = ISOFS_BUFFER_SIZE(inode);
|
|
unsigned char bufbits = ISOFS_BUFFER_BITS(inode);
|
|
unsigned long block, offset, block_saved, offset_saved;
|
|
unsigned long inode_number = 0; /* Quiet GCC */
|
|
struct buffer_head *bh = NULL;
|
|
int len;
|
|
int map;
|
|
int first_de = 1;
|
|
char *p = NULL; /* Quiet GCC */
|
|
struct iso_directory_record *de;
|
|
struct isofs_sb_info *sbi = ISOFS_SB(inode->i_sb);
|
|
|
|
offset = ctx->pos & (bufsize - 1);
|
|
block = ctx->pos >> bufbits;
|
|
|
|
while (ctx->pos < inode->i_size) {
|
|
int de_len;
|
|
|
|
if (!bh) {
|
|
bh = isofs_bread(inode, block);
|
|
if (!bh)
|
|
return 0;
|
|
}
|
|
|
|
de = (struct iso_directory_record *) (bh->b_data + offset);
|
|
|
|
de_len = *(unsigned char *)de;
|
|
|
|
/*
|
|
* If the length byte is zero, we should move on to the next
|
|
* CDROM sector. If we are at the end of the directory, we
|
|
* kick out of the while loop.
|
|
*/
|
|
|
|
if (de_len == 0) {
|
|
brelse(bh);
|
|
bh = NULL;
|
|
ctx->pos = (ctx->pos + ISOFS_BLOCK_SIZE) & ~(ISOFS_BLOCK_SIZE - 1);
|
|
block = ctx->pos >> bufbits;
|
|
offset = 0;
|
|
continue;
|
|
}
|
|
|
|
block_saved = block;
|
|
offset_saved = offset;
|
|
offset += de_len;
|
|
|
|
/* Make sure we have a full directory entry */
|
|
if (offset >= bufsize) {
|
|
int slop = bufsize - offset + de_len;
|
|
memcpy(tmpde, de, slop);
|
|
offset &= bufsize - 1;
|
|
block++;
|
|
brelse(bh);
|
|
bh = NULL;
|
|
if (offset) {
|
|
bh = isofs_bread(inode, block);
|
|
if (!bh)
|
|
return 0;
|
|
memcpy((void *) tmpde + slop, bh->b_data, offset);
|
|
}
|
|
de = tmpde;
|
|
}
|
|
/* Basic sanity check, whether name doesn't exceed dir entry */
|
|
if (de_len < sizeof(struct iso_directory_record) ||
|
|
de_len < de->name_len[0] +
|
|
sizeof(struct iso_directory_record)) {
|
|
printk(KERN_NOTICE "iso9660: Corrupted directory entry"
|
|
" in block %lu of inode %llu\n", block,
|
|
inode->i_ino);
|
|
brelse(bh);
|
|
return -EIO;
|
|
}
|
|
|
|
if (first_de) {
|
|
isofs_normalize_block_and_offset(de,
|
|
&block_saved,
|
|
&offset_saved);
|
|
inode_number = isofs_get_ino(block_saved,
|
|
offset_saved, bufbits);
|
|
}
|
|
|
|
if (de->flags[-sbi->s_high_sierra] & 0x80) {
|
|
first_de = 0;
|
|
ctx->pos += de_len;
|
|
continue;
|
|
}
|
|
first_de = 1;
|
|
|
|
/* Handle the case of the '.' directory */
|
|
if (de->name_len[0] == 1 && de->name[0] == 0) {
|
|
if (!dir_emit_dot(file, ctx))
|
|
break;
|
|
ctx->pos += de_len;
|
|
continue;
|
|
}
|
|
|
|
len = 0;
|
|
|
|
/* Handle the case of the '..' directory */
|
|
if (de->name_len[0] == 1 && de->name[0] == 1) {
|
|
if (!dir_emit_dotdot(file, ctx))
|
|
break;
|
|
ctx->pos += de_len;
|
|
continue;
|
|
}
|
|
|
|
/* Handle everything else. Do name translation if there
|
|
is no Rock Ridge NM field. */
|
|
|
|
/*
|
|
* Do not report hidden files if so instructed, or associated
|
|
* files unless instructed to do so
|
|
*/
|
|
if ((sbi->s_hide && (de->flags[-sbi->s_high_sierra] & 1)) ||
|
|
(!sbi->s_showassoc &&
|
|
(de->flags[-sbi->s_high_sierra] & 4))) {
|
|
ctx->pos += de_len;
|
|
continue;
|
|
}
|
|
|
|
map = 1;
|
|
if (sbi->s_rock) {
|
|
len = get_rock_ridge_filename(de, tmpname, inode);
|
|
if (len != 0) { /* may be -1 */
|
|
p = tmpname;
|
|
map = 0;
|
|
}
|
|
}
|
|
if (map) {
|
|
#ifdef CONFIG_JOLIET
|
|
if (sbi->s_joliet_level) {
|
|
len = get_joliet_filename(de, tmpname, inode);
|
|
p = tmpname;
|
|
} else
|
|
#endif
|
|
if (sbi->s_mapping == 'a') {
|
|
len = get_acorn_filename(de, tmpname, inode);
|
|
p = tmpname;
|
|
} else
|
|
if (sbi->s_mapping == 'n') {
|
|
len = isofs_name_translate(de, tmpname, inode);
|
|
p = tmpname;
|
|
} else {
|
|
p = de->name;
|
|
len = de->name_len[0];
|
|
}
|
|
}
|
|
if (len > 0) {
|
|
if (!dir_emit(ctx, p, len, inode_number, DT_UNKNOWN))
|
|
break;
|
|
}
|
|
ctx->pos += de_len;
|
|
}
|
|
if (bh)
|
|
brelse(bh);
|
|
return 0;
|
|
}
|
|
|
|
/*
|
|
* Handle allocation of temporary space for name translation and
|
|
* handling split directory entries.. The real work is done by
|
|
* "do_isofs_readdir()".
|
|
*/
|
|
static int isofs_readdir(struct file *file, struct dir_context *ctx)
|
|
{
|
|
int result;
|
|
char *tmpname;
|
|
struct iso_directory_record *tmpde;
|
|
struct inode *inode = file_inode(file);
|
|
|
|
tmpname = kmalloc(PAGE_SIZE, GFP_KERNEL);
|
|
if (tmpname == NULL)
|
|
return -ENOMEM;
|
|
|
|
tmpde = (struct iso_directory_record *) (tmpname+1024);
|
|
|
|
result = do_isofs_readdir(inode, file, ctx, tmpname, tmpde);
|
|
|
|
kfree(tmpname);
|
|
return result;
|
|
}
|
|
|
|
int isofs_fileattr_get(struct dentry *dentry, struct file_kattr *fa)
|
|
{
|
|
struct isofs_sb_info *sbi = ISOFS_SB(dentry->d_sb);
|
|
|
|
if (sbi->s_check == 'r') {
|
|
fa->fsx_xflags |= FS_XFLAG_CASEFOLD;
|
|
fa->flags |= FS_CASEFOLD_FL;
|
|
}
|
|
if (!sbi->s_joliet_level && !sbi->s_rock &&
|
|
(sbi->s_mapping == 'n' || sbi->s_mapping == 'a'))
|
|
fa->fsx_xflags |= FS_XFLAG_CASENONPRESERVING;
|
|
return 0;
|
|
}
|
|
|
|
const struct file_operations isofs_dir_operations =
|
|
{
|
|
.llseek = generic_file_llseek,
|
|
.read = generic_read_dir,
|
|
.iterate_shared = isofs_readdir,
|
|
.setlease = generic_setlease,
|
|
};
|
|
|
|
/*
|
|
* directories can handle most operations...
|
|
*/
|
|
const struct inode_operations isofs_dir_inode_operations =
|
|
{
|
|
.lookup = isofs_lookup,
|
|
.fileattr_get = isofs_fileattr_get,
|
|
};
|
|
|
|
|