Merge patch series "fs: support freeze/thaw/mark_dead/sync with shared devices"

Christian Brauner <brauner@kernel.org> says:

fs: support freeze/thaw/mark_dead/sync with shared devices

This is a generalization of the device number to superblock so it works
for actual block device and anonymous (or even mtd) devices.

fs_holder_ops recovers the affected superblock from bdev->bd_holder. That
forces the holder of a block device to be exactly one superblock and makes
it impossible for several superblocks to share a single device.

erofs does exactly that. It can mount read-only "blob" devices that are
shared between many superblocks: a metadata-only erofs that indexes a set
of per-layer blobs (one filesystem instead of one per OCI layer), or an
incremental image whose base device is shared by several updates. Because
the block layer only tracks a single holder, a freeze, thaw, removal or
sync on such a device is never propagated to all the superblocks using it,
and the current infrastructure has no way to find them.

This series replaces the bd_holder-based lookup with a global, dev_t-keyed
table mapping each block device to the superblock(s) using it. The holder
argument becomes purely the block layer's exclusivity token -- a superblock,
or the file_system_type for a device shared within one filesystem type --
and the fs_holder_ops callbacks look the device up in the table and act on
every superblock registered for it: 1:1 for most filesystems, 1:many for
erofs.

Filesystems claim and release their devices through new
fs_bdev_file_open_by_{dev,path}() and fs_bdev_file_release() helpers; the
per-fs patches convert xfs, btrfs, ext4, f2fs and erofs over to them and
fix cramfs and romfs, which released the registered main device with a
raw bdev_fput().

Since every superblock is registered under its s_dev the table also
replaces the last s_dev-keyed walk of the super_blocks list:
user_get_super() resolves device numbers through it, so ustat() and
quotactl() now work on any device a filesystem claims and no longer
take sb_lock.

The longer-term motivation is to let userspace decide which devices may be
onlined from one central place, without having to teach every filesystem
about it individually.

* patches from https://patch.msgid.link/20260616-work-super-bdev_holder_global-v2-0-7df6b864028e@kernel.org:
  selftests/filesystems: add ustat() coverage
  fs: look up the superblock via the device table in user_get_super()
  super: make fs_holder_ops private
  f2fs: open via dedicated fs bdev helpers
  erofs: open via dedicated fs bdev helpers
  fs: tolerate per-superblock freeze errors on shared devices
  fs: look up superblocks via the device table in fs_holder_ops
  ext4: open via dedicated fs bdev helpers
  btrfs: open via dedicated fs bdev helpers
  xfs: port to fs_bdev_file_open_by_path()
  fs: add dedicated block device open helpers for filesystems
  fs: maintain a global device-to-superblock table
  ocfs2: don't reset s_dev on dismount
  ext4: use anonymous devices for KUnit test superblocks
  fs, block: move blk_mode_t and fop_flags_t into <linux/types.h>
  super: take lock after last reference count
  super: convert s_count to refcount_t s_passive

Link: https://patch.msgid.link/20260616-work-super-bdev_holder_global-v2-0-7df6b864028e@kernel.org
Signed-off-by: Christian Brauner (Amutable) <brauner@kernel.org>
This commit is contained in:
Christian Brauner 2026-06-25 14:15:59 +02:00
commit 041a0885f7
No known key found for this signature in database
GPG Key ID: 91C61BC06578DCA2
22 changed files with 644 additions and 261 deletions

View File

@ -480,7 +480,12 @@ btrfs_get_bdev_and_sb(const char *device_path, blk_mode_t flags, void *holder,
struct block_device *bdev;
int ret;
*bdev_file = bdev_file_open_by_path(device_path, flags, holder, &fs_holder_ops);
if (holder)
*bdev_file = fs_bdev_file_open_by_path(device_path, flags,
holder, holder);
else
*bdev_file = bdev_file_open_by_path(device_path, flags, NULL,
NULL);
if (IS_ERR(*bdev_file)) {
ret = PTR_ERR(*bdev_file);
@ -495,7 +500,7 @@ btrfs_get_bdev_and_sb(const char *device_path, blk_mode_t flags, void *holder,
if (holder) {
ret = set_blocksize(*bdev_file, BTRFS_BDEV_BLOCKSIZE);
if (ret) {
bdev_fput(*bdev_file);
fs_bdev_file_release(*bdev_file, holder);
goto error;
}
}
@ -503,7 +508,10 @@ btrfs_get_bdev_and_sb(const char *device_path, blk_mode_t flags, void *holder,
*disk_super = btrfs_read_disk_super(bdev, 0, false);
if (IS_ERR(*disk_super)) {
ret = PTR_ERR(*disk_super);
bdev_fput(*bdev_file);
if (holder)
fs_bdev_file_release(*bdev_file, holder);
else
bdev_fput(*bdev_file);
goto error;
}
@ -727,7 +735,7 @@ static int btrfs_open_one_device(struct btrfs_fs_devices *fs_devices,
error_free_page:
btrfs_release_disk_super(disk_super);
bdev_fput(bdev_file);
fs_bdev_file_release(bdev_file, holder);
return -EINVAL;
}
@ -1087,7 +1095,7 @@ static void __btrfs_free_extra_devids(struct btrfs_fs_devices *fs_devices,
continue;
if (device->bdev_file) {
bdev_fput(device->bdev_file);
fs_bdev_file_release(device->bdev_file, device->bdev_file->private_data);
device->bdev = NULL;
device->bdev_file = NULL;
fs_devices->open_devices--;
@ -1127,8 +1135,10 @@ void btrfs_free_extra_devids(struct btrfs_fs_devices *fs_devices)
/* Release a device that was made unfreezable for a membership change. */
void btrfs_release_device_allow_freeze(struct file *bdev_file)
{
/* Yield before allow (strand-safe); file still open for the allow (UAF-safe). */
bdev_yield_claim(bdev_file);
struct super_block *sb = bdev_file->private_data;
/* Unregister before re-allowing (strand-safe); file still open (UAF-safe). */
fs_bdev_unregister(bdev_file, sb);
bdev_allow_freeze(file_bdev(bdev_file));
bdev_fput(bdev_file);
}
@ -1147,7 +1157,8 @@ static void btrfs_close_bdev(struct btrfs_device *device, bool allow_freeze)
if (allow_freeze)
btrfs_release_device_allow_freeze(device->bdev_file);
else
bdev_fput(device->bdev_file);
fs_bdev_file_release(device->bdev_file,
device->bdev_file->private_data);
}
static void btrfs_close_one_device(struct btrfs_device *device)
@ -2894,8 +2905,8 @@ struct file *btrfs_open_device_deny_freeze(const char *path,
return ERR_PTR(ret);
}
bdev_file = bdev_file_open_by_dev(file_bdev(probe_file)->bd_dev,
BLK_OPEN_WRITE, sb, &fs_holder_ops);
bdev_file = fs_bdev_file_open_by_dev(file_bdev(probe_file)->bd_dev,
BLK_OPEN_WRITE, sb, sb);
if (IS_ERR(bdev_file))
bdev_allow_freeze(file_bdev(probe_file));
bdev_fput(probe_file);

View File

@ -504,7 +504,7 @@ static void cramfs_kill_sb(struct super_block *sb)
sb->s_mtd = NULL;
} else if (IS_ENABLED(CONFIG_CRAMFS_BLOCKDEV) && sb->s_bdev) {
sync_blockdev(sb->s_bdev);
bdev_fput(sb->s_bdev_file);
fs_bdev_file_release(sb->s_bdev_file, sb);
}
kfree(sbi);
}

View File

@ -147,8 +147,8 @@ static int erofs_init_device(struct erofs_buf *buf, struct super_block *sb,
if (!sbi->devs->flatdev) {
file = erofs_is_fileio_mode(sbi) ?
filp_open(dif->path, O_RDONLY | O_LARGEFILE, 0) :
bdev_file_open_by_path(dif->path,
BLK_OPEN_READ, sb->s_type, NULL);
fs_bdev_file_open_by_path(dif->path,
BLK_OPEN_READ, sb->s_type, sb);
if (IS_ERR(file)) {
if (file == ERR_PTR(-ENOTBLK))
return -EINVAL;
@ -799,28 +799,34 @@ static int erofs_fc_reconfigure(struct fs_context *fc)
static int erofs_release_device_info(int id, void *ptr, void *data)
{
struct super_block *sb = data;
struct erofs_device_info *dif = ptr;
fs_put_dax(dif->dax_dev, NULL);
if (dif->file)
fput(dif->file);
if (dif->file) {
if (S_ISBLK(file_inode(dif->file)->i_mode))
fs_bdev_file_release(dif->file, sb);
else
fput(dif->file);
}
kfree(dif->path);
kfree(dif);
return 0;
}
static void erofs_free_dev_context(struct erofs_dev_context *devs)
static void erofs_free_dev_context(struct erofs_dev_context *devs,
struct super_block *sb)
{
if (!devs)
return;
idr_for_each(&devs->tree, &erofs_release_device_info, NULL);
idr_for_each(&devs->tree, &erofs_release_device_info, sb);
idr_destroy(&devs->tree);
kfree(devs);
}
static void erofs_sb_free(struct erofs_sb_info *sbi)
static void erofs_sb_free(struct erofs_sb_info *sbi, struct super_block *sb)
{
erofs_free_dev_context(sbi->devs);
erofs_free_dev_context(sbi->devs, sb);
kfree_sensitive(sbi->domain_id);
if (sbi->dif0.file)
fput(sbi->dif0.file);
@ -832,8 +838,13 @@ static void erofs_fc_free(struct fs_context *fc)
{
struct erofs_sb_info *sbi = fc->s_fs_info;
if (sbi) /* free here if an error occurs before transferring to sb */
erofs_sb_free(sbi);
/*
* Freed here only if an error occurs before the sb is set up; at that
* point no block-backed device has been claimed (that happens in
* fill_super), so the NULL sb never reaches fs_bdev_file_release().
*/
if (sbi)
erofs_sb_free(sbi, NULL);
}
static const struct fs_context_operations erofs_context_ops = {
@ -887,7 +898,7 @@ static void erofs_kill_sb(struct super_block *sb)
kill_block_super(sb);
erofs_drop_internal_inodes(sbi);
fs_put_dax(sbi->dif0.dax_dev, NULL);
erofs_sb_free(sbi);
erofs_sb_free(sbi, sb);
sb->s_fs_info = NULL;
}
@ -899,7 +910,7 @@ static void erofs_put_super(struct super_block *sb)
erofs_shrinker_unregister(sb);
erofs_xattr_prefixes_cleanup(sb);
erofs_drop_internal_inodes(sbi);
erofs_free_dev_context(sbi->devs);
erofs_free_dev_context(sbi->devs, sb);
sbi->devs = NULL;
}

View File

@ -126,11 +126,6 @@ struct kunit_ext_test_param {
struct kunit_ext_data_state exp_data_state[3];
};
static void ext_kill_sb(struct super_block *sb)
{
generic_shutdown_super(sb);
}
static int ext_init_fs_context(struct fs_context *fc)
{
return 0;
@ -138,13 +133,13 @@ static int ext_init_fs_context(struct fs_context *fc)
static int ext_set(struct super_block *sb, struct fs_context *fc)
{
return 0;
return set_anon_super_fc(sb, fc);
}
static struct file_system_type ext_fs_type = {
.name = "extents test",
.init_fs_context = ext_init_fs_context,
.kill_sb = ext_kill_sb,
.kill_sb = kill_anon_super,
};
static void extents_kunit_exit(struct kunit *test)

View File

@ -59,11 +59,6 @@ static const struct super_operations mbt_sops = {
.free_inode = mbt_free_inode,
};
static void mbt_kill_sb(struct super_block *sb)
{
generic_shutdown_super(sb);
}
static int mbt_init_fs_context(struct fs_context *fc)
{
return 0;
@ -72,7 +67,7 @@ static int mbt_init_fs_context(struct fs_context *fc)
static struct file_system_type mbt_fs_type = {
.name = "mballoc test",
.init_fs_context = mbt_init_fs_context,
.kill_sb = mbt_kill_sb,
.kill_sb = kill_anon_super,
};
static int mbt_mb_init(struct super_block *sb)
@ -136,7 +131,7 @@ static void mbt_mb_release(struct super_block *sb)
static int mbt_set(struct super_block *sb, struct fs_context *fc)
{
return 0;
return set_anon_super_fc(sb, fc);
}
static struct super_block *mbt_ext4_alloc_super_block(void)

View File

@ -5797,7 +5797,7 @@ failed_mount8: __maybe_unused
brelse(sbi->s_sbh);
if (sbi->s_journal_bdev_file) {
invalidate_bdev(file_bdev(sbi->s_journal_bdev_file));
bdev_fput(sbi->s_journal_bdev_file);
fs_bdev_file_release(sbi->s_journal_bdev_file, sb);
}
out_fail:
invalidate_bdev(sb->s_bdev);
@ -5981,9 +5981,9 @@ static struct file *ext4_get_journal_blkdev(struct super_block *sb,
struct ext4_super_block *es;
int errno;
bdev_file = bdev_file_open_by_dev(j_dev,
bdev_file = fs_bdev_file_open_by_dev(j_dev,
BLK_OPEN_READ | BLK_OPEN_WRITE | BLK_OPEN_RESTRICT_WRITES,
sb, &fs_holder_ops);
sb, sb);
if (IS_ERR(bdev_file)) {
ext4_msg(sb, KERN_ERR,
"failed to open journal device unknown-block(%u,%u) %pe",
@ -6043,7 +6043,7 @@ static struct file *ext4_get_journal_blkdev(struct super_block *sb,
out_bh:
brelse(bh);
out_bdev:
bdev_fput(bdev_file);
fs_bdev_file_release(bdev_file, sb);
return ERR_PTR(errno);
}
@ -6082,7 +6082,7 @@ static journal_t *ext4_open_dev_journal(struct super_block *sb,
out_journal:
ext4_journal_destroy(EXT4_SB(sb), journal);
out_bdev:
bdev_fput(bdev_file);
fs_bdev_file_release(bdev_file, sb);
return ERR_PTR(errno);
}
@ -7499,7 +7499,7 @@ static void ext4_kill_sb(struct super_block *sb)
kill_block_super(sb);
if (bdev_file)
bdev_fput(bdev_file);
fs_bdev_file_release(bdev_file, sb);
}
static struct file_system_type ext4_fs_type = {

View File

@ -1971,7 +1971,7 @@ static void destroy_device_list(struct f2fs_sb_info *sbi)
for (i = 0; i < sbi->s_ndevs; i++) {
if (i > 0)
bdev_fput(FDEV(i).bdev_file);
fs_bdev_file_release(FDEV(i).bdev_file, sbi->sb);
#ifdef CONFIG_BLK_DEV_ZONED
kvfree(FDEV(i).blkz_seq);
#endif
@ -4898,8 +4898,8 @@ static int f2fs_scan_devices(struct f2fs_sb_info *sbi)
FDEV(i).end_blk = FDEV(i).start_blk +
SEGS_TO_BLKS(sbi,
FDEV(i).total_segments) - 1;
FDEV(i).bdev_file = bdev_file_open_by_path(
FDEV(i).path, mode, sbi->sb, NULL);
FDEV(i).bdev_file = fs_bdev_file_open_by_path(
FDEV(i).path, mode, sbi->sb, sbi->sb);
}
}
if (IS_ERR(FDEV(i).bdev_file))

View File

@ -137,6 +137,7 @@ extern int reconfigure_super(struct fs_context *);
extern bool super_trylock_shared(struct super_block *sb);
struct super_block *user_get_super(dev_t, bool excl);
void put_super(struct super_block *sb);
void __init super_dev_init(void);
extern bool mount_capable(struct fs_context *);
/*

View File

@ -6262,6 +6262,8 @@ void __init mnt_init(void)
if (!mount_hashtable || !mountpoint_hashtable)
panic("Failed to allocate mount hash table\n");
super_dev_init();
kernfs_init();
err = sysfs_init();

View File

@ -1882,7 +1882,6 @@ static void ocfs2_dismount_volume(struct super_block *sb, int mnt_err)
ocfs2_delete_osb(osb);
kfree(osb);
sb->s_dev = 0;
sb->s_fs_info = NULL;
}

View File

@ -587,7 +587,7 @@ static void romfs_kill_sb(struct super_block *sb)
#ifdef CONFIG_ROMFS_ON_BLOCK
if (sb->s_bdev) {
sync_blockdev(sb->s_bdev);
bdev_fput(sb->s_bdev_file);
fs_bdev_file_release(sb->s_bdev_file, sb);
}
#endif
}

View File

@ -24,6 +24,7 @@
#include <linux/export.h>
#include <linux/slab.h>
#include <linux/blkdev.h>
#include <linux/rhashtable.h>
#include <linux/mount.h>
#include <linux/security.h>
#include <linux/writeback.h> /* for the emergency remount stuff */
@ -102,7 +103,7 @@ static bool super_flags(const struct super_block *sb, unsigned int flags)
* creation will succeed and SB_BORN is set by vfs_get_tree() or we're
* woken and we'll see SB_DYING.
*
* The caller must have acquired a temporary reference on @sb->s_count.
* The caller must have acquired a temporary reference on @sb->s_passive.
*
* Return: The function returns true if SB_BORN was set and with
* s_umount held. The function returns false if SB_DYING was
@ -272,6 +273,8 @@ static unsigned long super_cache_count(struct shrinker *shrink,
return total_objects;
}
static struct super_dev *super_dev_alloc(dev_t dev, struct super_block *sb);
static void destroy_super_work(struct work_struct *work)
{
struct super_block *s = container_of(work, struct super_block,
@ -279,6 +282,8 @@ static void destroy_super_work(struct work_struct *work)
fsnotify_sb_free(s);
security_sb_free(s);
put_user_ns(s->s_user_ns);
/* Only an unregistered entry is still owned by the superblock. */
kfree(s->s_super_dev);
kfree(s->s_subtype);
for (int i = 0; i < SB_FREEZE_LEVELS; i++)
percpu_free_rwsem(&s->s_writers.rw_sem[i]);
@ -367,7 +372,7 @@ static struct super_block *alloc_super(struct file_system_type *type, int flags,
spin_lock_init(&s->s_inode_wblist_lock);
fserror_mount(s);
s->s_count = 1;
refcount_set(&s->s_passive, 1);
atomic_set(&s->s_active, 1);
mutex_init(&s->s_vfs_rename_mutex);
lockdep_set_class(&s->s_vfs_rename_mutex, &type->s_vfs_rename_key);
@ -392,6 +397,10 @@ static struct super_block *alloc_super(struct file_system_type *type, int flags,
goto fail;
if (list_lru_init_memcg(&s->s_inode_lru, s->s_shrink))
goto fail;
s->s_super_dev = super_dev_alloc(0, s);
if (!s->s_super_dev)
goto fail;
s->s_min_writeback_pages = MIN_WRITEBACK_PAGES;
return s;
@ -403,12 +412,17 @@ static struct super_block *alloc_super(struct file_system_type *type, int flags,
/* Superblock refcounting */
/*
* Drop a superblock's refcount. The caller must hold sb_lock.
* Drop a superblock's passive reference. Must be called WITHOUT sb_lock held;
* put_super() acquires sb_lock itself when the final reference is dropped.
*/
static void __put_super(struct super_block *s)
void put_super(struct super_block *s)
{
if (!--s->s_count) {
if (refcount_dec_and_test(&s->s_passive)) {
spin_lock(&sb_lock);
list_del_init(&s->s_list);
spin_unlock(&sb_lock);
WARN_ON(s->s_dentry_lru.node);
WARN_ON(s->s_inode_lru.node);
WARN_ON(s->s_mounts);
@ -416,18 +430,109 @@ static void __put_super(struct super_block *s)
}
}
/**
* put_super - drop a temporary reference to superblock
* @sb: superblock in question
*
* Drops a temporary reference, frees superblock if there's no
* references left.
*/
void put_super(struct super_block *sb)
struct super_dev {
dev_t sd_dev;
struct super_block *sd_sb;
refcount_t sd_ref;
struct rhlist_head sd_node;
struct rcu_head sd_rcu;
};
static struct rhltable super_dev_table;
static const struct rhashtable_params super_dev_params = {
.key_len = sizeof(dev_t),
.key_offset = offsetof(struct super_dev, sd_dev),
.head_offset = offsetof(struct super_dev, sd_node),
};
static struct super_dev *super_dev_alloc(dev_t dev, struct super_block *sb)
{
spin_lock(&sb_lock);
__put_super(sb);
spin_unlock(&sb_lock);
struct super_dev *fsd;
fsd = kzalloc_obj(*fsd);
if (!fsd)
return NULL;
fsd->sd_dev = dev;
fsd->sd_sb = sb;
refcount_set(&fsd->sd_ref, 1);
return fsd;
}
static void super_dev_put(struct super_dev *fsd)
{
/* Unlink only once unpinned, so a cursor never resumes from a removed node. */
if (fsd && refcount_dec_and_test(&fsd->sd_ref)) {
rhltable_remove(&super_dev_table, &fsd->sd_node, super_dev_params);
put_super(fsd->sd_sb);
kfree_rcu(fsd, sd_rcu);
}
}
void __init super_dev_init(void)
{
if (rhltable_init(&super_dev_table, &super_dev_params))
panic("VFS: Cannot initialise super_dev_table\n");
}
static int super_dev_insert(struct super_dev *fsd)
{
int err;
err = rhltable_insert(&super_dev_table, &fsd->sd_node, super_dev_params);
if (!err)
refcount_inc(&fsd->sd_sb->s_passive);
return err;
}
/* Register @sb under @sb->s_dev as the final fallible act of a set callback. */
static int super_dev_register(struct super_block *sb)
{
struct super_dev *fsd = sb->s_super_dev;
int err;
lockdep_assert_held(&sb_lock);
VFS_WARN_ON_ONCE(!sb->s_dev);
VFS_WARN_ON_ONCE(!fsd || fsd->sd_dev);
fsd->sd_dev = sb->s_dev;
err = super_dev_insert(fsd);
if (err)
fsd->sd_dev = 0;
return err;
}
static struct super_dev *super_dev_get(struct rhlist_head *pos)
{
struct super_dev *sb_dev;
for (; pos; pos = rcu_dereference_all(pos->next)) {
sb_dev = container_of(pos, struct super_dev, sd_node);
if (refcount_inc_not_zero(&sb_dev->sd_ref))
return sb_dev;
}
return NULL;
}
static struct super_dev *super_dev_first(dev_t dev)
{
struct super_dev *sb_dev;
rcu_read_lock();
sb_dev = super_dev_get(rhltable_lookup(&super_dev_table, &dev, super_dev_params));
rcu_read_unlock();
return sb_dev;
}
static struct super_dev *super_dev_next(struct super_dev *prev)
{
struct super_dev *sb_dev;
rcu_read_lock();
sb_dev = super_dev_get(rcu_dereference_all(prev->sd_node.next));
rcu_read_unlock();
super_dev_put(prev);
return sb_dev;
}
static void kill_super_notify(struct super_block *sb)
@ -449,6 +554,12 @@ static void kill_super_notify(struct super_block *sb)
hlist_del_init(&sb->s_instances);
spin_unlock(&sb_lock);
/* Drop sget_fc()'s claim; a never-registered entry stays with the sb. */
if (sb->s_super_dev->sd_dev) {
super_dev_put(sb->s_super_dev);
sb->s_super_dev = NULL;
}
/*
* Let concurrent mounts know that this thing is really dead.
* We don't need @sb->s_umount here as every concurrent caller
@ -478,11 +589,7 @@ void deactivate_locked_super(struct super_block *s)
kill_super_notify(s);
/*
* Since list_lru_destroy() may sleep, we cannot call it from
* put_super(), where we hold the sb_lock. Therefore we destroy
* the lru lists right now.
*/
/* list_lru_destroy() may sleep; put_super() callers may not. */
list_lru_destroy(&s->s_dentry_lru);
list_lru_destroy(&s->s_inode_lru);
@ -529,7 +636,7 @@ static bool grab_super(struct super_block *sb)
{
bool locked;
sb->s_count++;
refcount_inc(&sb->s_passive);
spin_unlock(&sb_lock);
locked = super_lock_excl(sb);
if (locked) {
@ -556,7 +663,7 @@ static bool grab_super(struct super_block *sb)
* lock held in read mode in case of success. On successful return,
* the caller must drop the s_umount lock when done.
*
* Note that unlike get_super() et.al. this one does *not* bump ->s_count.
* Note that unlike get_super() et.al. this one does *not* bump ->s_passive.
* The reason why it's safe is that we are OK with doing trylock instead
* of down_read(). There's a couple of places that are OK with that, but
* it's very much not a general-purpose interface.
@ -763,6 +870,7 @@ struct super_block *sget_fc(struct fs_context *fc,
}
if (!s) {
spin_unlock(&sb_lock);
s = alloc_super(fc->fs_type, fc->sb_flags, user_ns);
if (!s)
return ERR_PTR(-ENOMEM);
@ -772,11 +880,13 @@ struct super_block *sget_fc(struct fs_context *fc,
s->s_fs_info = fc->s_fs_info;
err = set(s, fc);
if (err) {
VFS_WARN_ON_ONCE(s->s_super_dev->sd_dev);
s->s_fs_info = NULL;
spin_unlock(&sb_lock);
destroy_unused_super(s);
return ERR_PTR(err);
}
VFS_WARN_ON_ONCE(!s->s_super_dev->sd_dev);
fc->s_fs_info = NULL;
s->s_type = fc->fs_type;
s->s_iflags |= fc->s_iflags;
@ -851,14 +961,17 @@ static void __iterate_supers(void (*f)(struct super_block *, void *), void *arg,
struct super_block *sb, *p = NULL;
bool excl = flags & SUPER_ITER_EXCL;
guard(spinlock)(&sb_lock);
spin_lock(&sb_lock);
for (sb = first_super(flags);
!list_entry_is_head(sb, &super_blocks, s_list);
sb = next_super(sb, flags)) {
if (super_flags(sb, SB_DYING))
continue;
sb->s_count++;
if (!refcount_inc_not_zero(&sb->s_passive))
continue;
spin_unlock(&sb_lock);
if (flags & SUPER_ITER_UNLOCKED) {
@ -868,13 +981,14 @@ static void __iterate_supers(void (*f)(struct super_block *, void *), void *arg,
super_unlock(sb, excl);
}
spin_lock(&sb_lock);
if (p)
__put_super(p);
put_super(p);
p = sb;
spin_lock(&sb_lock);
}
spin_unlock(&sb_lock);
if (p)
__put_super(p);
put_super(p);
}
void iterate_supers(void (*f)(struct super_block *, void *), void *arg)
@ -903,7 +1017,9 @@ void iterate_supers_type(struct file_system_type *type,
if (super_flags(sb, SB_DYING))
continue;
sb->s_count++;
if (!refcount_inc_not_zero(&sb->s_passive))
continue;
spin_unlock(&sb_lock);
locked = super_lock_shared(sb);
@ -912,41 +1028,33 @@ void iterate_supers_type(struct file_system_type *type,
super_unlock_shared(sb);
}
spin_lock(&sb_lock);
if (p)
__put_super(p);
put_super(p);
p = sb;
spin_lock(&sb_lock);
}
if (p)
__put_super(p);
spin_unlock(&sb_lock);
if (p)
put_super(p);
}
EXPORT_SYMBOL(iterate_supers_type);
struct super_block *user_get_super(dev_t dev, bool excl)
{
struct super_block *sb;
struct super_dev *sb_dev;
spin_lock(&sb_lock);
list_for_each_entry(sb, &super_blocks, s_list) {
bool locked;
for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
struct super_block *sb = sb_dev->sd_sb;
if (sb->s_dev != dev)
if (!super_lock(sb, excl))
continue;
sb->s_count++;
spin_unlock(&sb_lock);
locked = super_lock(sb, excl);
if (locked)
return sb;
spin_lock(&sb_lock);
__put_super(sb);
break;
/* The pinned entry holds a passive reference, take our own. */
refcount_inc(&sb->s_passive);
super_dev_put(sb_dev);
return sb;
}
spin_unlock(&sb_lock);
return NULL;
}
@ -1222,7 +1330,16 @@ EXPORT_SYMBOL(free_anon_bdev);
int set_anon_super(struct super_block *s, void *data)
{
return get_anon_bdev(&s->s_dev);
int error;
error = get_anon_bdev(&s->s_dev);
if (error)
return error;
error = super_dev_register(s);
if (error)
free_anon_bdev(s->s_dev);
return error;
}
EXPORT_SYMBOL(set_anon_super);
@ -1308,7 +1425,7 @@ EXPORT_SYMBOL(get_tree_keyed);
static int set_bdev_super(struct super_block *s, void *data)
{
s->s_dev = *(dev_t *)data;
return 0;
return super_dev_register(s);
}
static int super_s_dev_set(struct super_block *s, struct fs_context *fc)
@ -1350,197 +1467,313 @@ struct super_block *sget_dev(struct fs_context *fc, dev_t dev)
EXPORT_SYMBOL(sget_dev);
#ifdef CONFIG_BLOCK
/*
* Lock the superblock that is holder of the bdev. Returns the superblock
* pointer if we successfully locked the superblock and it is alive. Otherwise
* we return NULL and just unlock bdev->bd_holder_lock.
*
* The function must be called with bdev->bd_holder_lock and releases it.
*/
static struct super_block *bdev_super_lock(struct block_device *bdev, bool excl)
__releases(&bdev->bd_holder_lock)
static int fs_super_freeze(struct super_block *sb)
{
struct super_block *sb = bdev->bd_holder;
bool locked;
if (sb->s_op->freeze_super)
return sb->s_op->freeze_super(sb,
FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
return freeze_super(sb, FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
}
lockdep_assert_held(&bdev->bd_holder_lock);
lockdep_assert_not_held(&sb->s_umount);
lockdep_assert_not_held(&bdev->bd_disk->open_mutex);
/* Make sure sb doesn't go away from under us */
spin_lock(&sb_lock);
sb->s_count++;
spin_unlock(&sb_lock);
mutex_unlock(&bdev->bd_holder_lock);
locked = super_lock(sb, excl);
/*
* If the superblock wasn't already SB_DYING then we hold
* s_umount and can safely drop our temporary reference.
*/
put_super(sb);
if (!locked)
return NULL;
if (!sb->s_root || !(sb->s_flags & SB_ACTIVE)) {
super_unlock(sb, excl);
return NULL;
}
return sb;
static int fs_super_thaw(struct super_block *sb)
{
if (sb->s_op->thaw_super)
return sb->s_op->thaw_super(sb,
FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
return thaw_super(sb, FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
}
static void fs_bdev_mark_dead(struct block_device *bdev, bool surprise)
{
struct super_block *sb;
struct super_dev *sb_dev;
dev_t dev = bdev->bd_dev;
sb = bdev_super_lock(bdev, false);
if (!sb)
return;
mutex_unlock(&bdev->bd_holder_lock);
if (sb->s_op->remove_bdev) {
int ret;
for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
struct super_block *sb = sb_dev->sd_sb;
ret = sb->s_op->remove_bdev(sb, bdev);
if (!ret) {
super_unlock_shared(sb);
return;
if (!super_lock_shared(sb))
continue;
if (sb->s_root && (sb->s_flags & SB_ACTIVE)) {
if (!sb->s_op->remove_bdev ||
sb->s_op->remove_bdev(sb, bdev)) {
if (!surprise)
sync_filesystem(sb);
shrink_dcache_sb(sb);
evict_inodes(sb);
if (sb->s_op->shutdown)
sb->s_op->shutdown(sb);
}
}
/* Fallback to shutdown. */
super_unlock_shared(sb);
}
if (!surprise)
sync_filesystem(sb);
shrink_dcache_sb(sb);
evict_inodes(sb);
if (sb->s_op->shutdown)
sb->s_op->shutdown(sb);
super_unlock_shared(sb);
}
static void fs_bdev_sync(struct block_device *bdev)
{
struct super_block *sb;
struct super_dev *sb_dev;
dev_t dev = bdev->bd_dev;
sb = bdev_super_lock(bdev, false);
if (!sb)
return;
mutex_unlock(&bdev->bd_holder_lock);
sync_filesystem(sb);
super_unlock_shared(sb);
}
for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
struct super_block *sb = sb_dev->sd_sb;
static struct super_block *get_bdev_super(struct block_device *bdev)
{
bool active = false;
struct super_block *sb;
sb = bdev_super_lock(bdev, true);
if (sb) {
active = atomic_inc_not_zero(&sb->s_active);
super_unlock_excl(sb);
if (!super_lock_shared(sb))
continue;
if (sb->s_root && (sb->s_flags & SB_ACTIVE))
sync_filesystem(sb);
super_unlock_shared(sb);
}
if (!active)
return NULL;
return sb;
}
/**
* fs_bdev_freeze - freeze owning filesystem of block device
* fs_bdev_freeze - freeze every superblock using a block device
* @bdev: block device
*
* Freeze the filesystem that owns this block device if it is still
* active.
* Freeze each live superblock using @bdev. A superblock owning several block
* devices is frozen once per device and stays frozen until all are thawed; the
* block layer nests these freezes so the count stays balanced.
*
* A filesystem that owns multiple block devices may be frozen from each
* block device and won't be unfrozen until all block devices are
* unfrozen. Each block device can only freeze the filesystem once as we
* nest freezes for block devices in the block layer.
*
* Return: If the freeze was successful zero is returned. If the freeze
* failed a negative error code is returned.
* Return: 0, or the error from the one superblock on a single-fs device. When
* several superblocks share @bdev a per-superblock failure is swallowed
* (see below), but a sync_blockdev() failure is always reported.
*/
static int fs_bdev_freeze(struct block_device *bdev)
{
struct super_block *sb;
int error = 0;
dev_t dev = bdev->bd_dev;
struct super_dev *sb_dev;
unsigned int count = 0;
int error = 0, err;
lockdep_assert_held(&bdev->bd_fsfreeze_mutex);
sb = get_bdev_super(bdev);
if (!sb)
return -EINVAL;
mutex_unlock(&bdev->bd_holder_lock);
if (sb->s_op->freeze_super)
error = sb->s_op->freeze_super(sb,
FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
else
error = freeze_super(sb,
FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
if (!get_active_super(sb_dev->sd_sb))
continue;
err = fs_super_freeze(sb_dev->sd_sb);
if (err && !error)
error = err;
deactivate_super(sb_dev->sd_sb);
count++;
}
/*
* When several superblocks share the device, keep it frozen even if some
* of them failed to freeze and swallow the error: rolling the rest back
* via thaw_super() can fail too, so neither is a clear win. A single
* filesystem (count == 1) still reports its error.
*/
if (error && count > 1)
error = 0;
if (!error)
error = sync_blockdev(bdev);
deactivate_super(sb);
return error;
}
/**
* fs_bdev_thaw - thaw owning filesystem of block device
* fs_bdev_thaw - thaw every superblock using a block device
* @bdev: block device
*
* Thaw the filesystem that owns this block device.
* The counterpart to fs_bdev_freeze(): thaw each live superblock using @bdev.
* A zero return does not imply a superblock is fully unfrozen; it may have been
* frozen more than once (by the kernel or via another device).
*
* A filesystem that owns multiple block devices may be frozen from each
* block device and won't be unfrozen until all block devices are
* unfrozen. Each block device can only freeze the filesystem once as we
* nest freezes for block devices in the block layer.
*
* Return: If the thaw was successful zero is returned. If the thaw
* failed a negative error code is returned. If this function
* returns zero it doesn't mean that the filesystem is unfrozen
* as it may have been frozen multiple times (kernel may hold a
* freeze or might be frozen from other block devices).
* Return: 0, or the first error on a single-fs device; a shared device swallows
* per-superblock errors, as fs_bdev_freeze() does.
*/
static int fs_bdev_thaw(struct block_device *bdev)
{
struct super_block *sb;
int error;
dev_t dev = bdev->bd_dev;
struct super_dev *sb_dev;
unsigned int count = 0;
int error = 0, err;
lockdep_assert_held(&bdev->bd_fsfreeze_mutex);
/*
* The block device may have been frozen before it was claimed by a
* filesystem. Concurrently another process might try to mount that
* frozen block device and has temporarily claimed the block device for
* that purpose causing a concurrent fs_bdev_thaw() to end up here. The
* mounter is already about to abort mounting because they still saw an
* elevanted bdev->bd_fsfreeze_count so get_bdev_super() will return
* NULL in that case.
*/
sb = get_bdev_super(bdev);
if (!sb)
return -EINVAL;
mutex_unlock(&bdev->bd_holder_lock);
if (sb->s_op->thaw_super)
error = sb->s_op->thaw_super(sb,
FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
else
error = thaw_super(sb,
FREEZE_MAY_NEST | FREEZE_HOLDER_USERSPACE, NULL);
deactivate_super(sb);
for (sb_dev = super_dev_first(dev); sb_dev; sb_dev = super_dev_next(sb_dev)) {
if (!get_active_super(sb_dev->sd_sb))
continue;
err = fs_super_thaw(sb_dev->sd_sb);
if (err && !error)
error = err;
deactivate_super(sb_dev->sd_sb);
count++;
}
/* Shared device: swallow per-superblock errors, like fs_bdev_freeze(). */
if (error && count > 1)
error = 0;
return error;
}
const struct blk_holder_ops fs_holder_ops = {
static const struct blk_holder_ops fs_holder_ops = {
.mark_dead = fs_bdev_mark_dead,
.sync = fs_bdev_sync,
.freeze = fs_bdev_freeze,
.thaw = fs_bdev_thaw,
};
EXPORT_SYMBOL_GPL(fs_holder_ops);
static struct super_dev *super_dev_lookup(dev_t dev, struct super_block *sb)
{
struct super_dev *it;
struct rhlist_head *list, *pos;
RCU_LOCKDEP_WARN(!rcu_read_lock_held(), "suspicious super_dev_lookup() usage");
VFS_WARN_ON_ONCE(!dev);
VFS_WARN_ON_ONCE(!sb);
list = rhltable_lookup(&super_dev_table, &dev, super_dev_params);
rhl_for_each_entry_rcu(it, pos, list, sd_node) {
if (it->sd_sb == sb)
return it;
}
return NULL;
}
static int fs_bdev_register(struct file *bdev_file, struct super_block *sb)
{
struct super_dev *sb_dev __free(kfree) = NULL;
dev_t dev = file_bdev(bdev_file)->bd_dev;
int err;
scoped_guard(rcu) {
sb_dev = super_dev_lookup(dev, sb);
if (sb_dev && refcount_inc_not_zero(&sb_dev->sd_ref)) {
retain_and_null_ptr(sb_dev);
return 0;
}
}
sb_dev = super_dev_alloc(dev, sb);
if (!sb_dev)
return -ENOMEM;
err = super_dev_insert(sb_dev);
if (err)
return err;
/* Publish the entry before reading the count; pairs with bdev_freeze(). */
smp_mb();
if (atomic_read(&file_bdev(bdev_file)->bd_fsfreeze_count) > 0) {
err = -EBUSY;
super_dev_put(sb_dev);
}
retain_and_null_ptr(sb_dev);
return err;
}
/**
* fs_bdev_file_open_by_dev - claim a block device on behalf of a superblock
* @dev: block device number
* @mode: open mode
* @holder: block-layer exclusivity token (a superblock, or the file_system_type
* when the device may be shared by several superblocks of that type)
* @sb: superblock to drive fs_holder_ops events for
*
* Open @dev with &fs_holder_ops and register that @sb uses it, so device
* removal/sync/freeze/thaw are propagated to @sb (and any other superblock
* sharing @dev). Must be paired with fs_bdev_file_release().
*
* Return: an opened block-device file or an ERR_PTR().
*/
struct file *fs_bdev_file_open_by_dev(dev_t dev, blk_mode_t mode, void *holder,
struct super_block *sb)
{
struct file *bdev_file;
int err;
bdev_file = bdev_file_open_by_dev(dev, mode, holder, &fs_holder_ops);
if (IS_ERR(bdev_file))
return bdev_file;
err = fs_bdev_register(bdev_file, sb);
if (err) {
bdev_fput(bdev_file);
return ERR_PTR(err);
}
return bdev_file;
}
EXPORT_SYMBOL_GPL(fs_bdev_file_open_by_dev);
/**
* fs_bdev_file_open_by_path - claim a block device on behalf of a superblock
* @path: path to the block device
* @mode: open mode
* @holder: block-layer exclusivity token (a superblock, or the file_system_type
* when the device may be shared by several superblocks of that type)
* @sb: superblock to drive fs_holder_ops events for
*
* Open the block device at @path with &fs_holder_ops and register that @sb
* uses it, so device removal/sync/freeze/thaw are propagated to @sb (and any
* other superblock sharing the device). Must be paired with
* fs_bdev_file_release().
*
* Return: an opened block-device file or an ERR_PTR().
*/
struct file *fs_bdev_file_open_by_path(const char *path, blk_mode_t mode,
void *holder, struct super_block *sb)
{
struct file *bdev_file;
int err;
bdev_file = bdev_file_open_by_path(path, mode, holder, &fs_holder_ops);
if (IS_ERR(bdev_file))
return bdev_file;
err = fs_bdev_register(bdev_file, sb);
if (err) {
bdev_fput(bdev_file);
return ERR_PTR(err);
}
return bdev_file;
}
EXPORT_SYMBOL_GPL(fs_bdev_file_open_by_path);
/**
* fs_bdev_unregister - drop a superblock's claim on a block device
* @bdev_file: file returned by fs_bdev_file_open_by_{dev,path}()
* @sb: superblock the device was claimed for
*
* The inverse of fs_bdev_register(): drop one claim on the {dev, @sb} entry
* (the last claim unregisters it; a pinning cursor defers the actual unlink)
* without closing the device. A caller that must act on the still-open device
* between unregistering and closing - e.g. re-allow freezing one denied for a
* membership change - pairs this with bdev_fput(). fs_bdev_file_release() is
* the common unregister-and-close.
*/
void fs_bdev_unregister(struct file *bdev_file, struct super_block *sb)
{
dev_t dev = file_bdev(bdev_file)->bd_dev;
struct super_dev *sb_dev;
rcu_read_lock();
sb_dev = super_dev_lookup(dev, sb);
rcu_read_unlock();
super_dev_put(sb_dev);
}
EXPORT_SYMBOL_GPL(fs_bdev_unregister);
/**
* fs_bdev_file_release - release a block device claimed for a superblock
* @bdev_file: file returned by fs_bdev_file_open_by_{dev,path}()
* @sb: superblock the device was claimed for
*
* Unregister the {dev, @sb} entry, then close the block device.
*/
void fs_bdev_file_release(struct file *bdev_file, struct super_block *sb)
{
fs_bdev_unregister(bdev_file, sb);
bdev_fput(bdev_file);
}
EXPORT_SYMBOL_GPL(fs_bdev_file_release);
int setup_bdev_super(struct super_block *sb, int sb_flags,
struct fs_context *fc)
@ -1549,7 +1782,7 @@ int setup_bdev_super(struct super_block *sb, int sb_flags,
struct file *bdev_file;
struct block_device *bdev;
bdev_file = bdev_file_open_by_dev(sb->s_dev, mode, sb, &fs_holder_ops);
bdev_file = fs_bdev_file_open_by_dev(sb->s_dev, mode, sb, sb);
if (IS_ERR(bdev_file)) {
if (fc)
errorf(fc, "%s: Can't open blockdev", fc->source);
@ -1563,20 +1796,19 @@ int setup_bdev_super(struct super_block *sb, int sb_flags,
* writable from userspace even for a read-only block device.
*/
if ((mode & BLK_OPEN_WRITE) && bdev_read_only(bdev)) {
bdev_fput(bdev_file);
fs_bdev_file_release(bdev_file, sb);
return -EACCES;
}
/*
* It is enough to check bdev was not frozen before we set
* s_bdev as freezing will wait until SB_BORN is set.
*/
/* The sget_fc() entry is already published; pairs with bdev_freeze(). */
smp_mb();
if (atomic_read(&bdev->bd_fsfreeze_count) > 0) {
if (fc)
warnf(fc, "%pg: Can't mount, blockdev is frozen", bdev);
bdev_fput(bdev_file);
fs_bdev_file_release(bdev_file, sb);
return -EBUSY;
}
spin_lock(&sb_lock);
sb->s_bdev_file = bdev_file;
sb->s_bdev = bdev;
@ -1665,7 +1897,7 @@ void kill_block_super(struct super_block *sb)
generic_shutdown_super(sb);
if (bdev) {
sync_blockdev(bdev);
bdev_fput(sb->s_bdev_file);
fs_bdev_file_release(sb->s_bdev_file, sb);
}
}

View File

@ -1615,7 +1615,7 @@ xfs_free_buftarg(
fs_put_dax(btp->bt_daxdev, btp->bt_mount);
/* the main block device is closed by kill_block_super */
if (btp->bt_bdev != btp->bt_mount->m_super->s_bdev)
bdev_fput(btp->bt_file);
fs_bdev_file_release(btp->bt_file, btp->bt_mount->m_super);
kfree(btp);
}

View File

@ -400,8 +400,8 @@ xfs_blkdev_get(
blk_mode_t mode;
mode = sb_open_mode(mp->m_super->s_flags);
*bdev_filep = bdev_file_open_by_path(name, mode,
mp->m_super, &fs_holder_ops);
*bdev_filep = fs_bdev_file_open_by_path(name, mode,
mp->m_super, mp->m_super);
if (IS_ERR(*bdev_filep)) {
error = PTR_ERR(*bdev_filep);
*bdev_filep = NULL;
@ -526,7 +526,7 @@ xfs_open_devices(
mp->m_logdev_targp = mp->m_ddev_targp;
/* Handle won't be used, drop it */
if (logdev_file)
bdev_fput(logdev_file);
fs_bdev_file_release(logdev_file, mp->m_super);
}
return 0;
@ -538,10 +538,10 @@ xfs_open_devices(
xfs_free_buftarg(mp->m_ddev_targp);
out_close_rtdev:
if (rtdev_file)
bdev_fput(rtdev_file);
fs_bdev_file_release(rtdev_file, mp->m_super);
out_close_logdev:
if (logdev_file)
bdev_fput(logdev_file);
fs_bdev_file_release(logdev_file, mp->m_super);
return error;
}

View File

@ -126,8 +126,6 @@ struct blk_integrity {
unsigned char pi_tuple_size;
};
typedef unsigned int __bitwise blk_mode_t;
/* open for reading */
#define BLK_OPEN_READ ((__force blk_mode_t)(1 << 0))
/* open for writing */
@ -1770,13 +1768,6 @@ struct blk_holder_ops {
__releases(&bdev->bd_holder_lock);
};
/*
* For filesystems using @fs_holder_ops, the @holder argument passed to
* helpers used to open and claim block devices via
* bd_prepare_to_claim() must point to a superblock.
*/
extern const struct blk_holder_ops fs_holder_ops;
/*
* Return the correct open flags for blkdev_get_by_* for super block flags
* as stored in sb->s_flags.

View File

@ -1916,8 +1916,6 @@ struct dir_context {
struct io_uring_cmd;
struct offset_ctx;
typedef unsigned int __bitwise fop_flags_t;
struct file_operations {
struct module *owner;
fop_flags_t fop_flags;

View File

@ -237,4 +237,12 @@ int thaw_super(struct super_block *super, enum freeze_holder who,
int sb_init_dio_done_wq(struct super_block *sb);
struct file;
struct file *fs_bdev_file_open_by_dev(dev_t dev, blk_mode_t mode, void *holder,
struct super_block *sb);
struct file *fs_bdev_file_open_by_path(const char *path, blk_mode_t mode,
void *holder, struct super_block *sb);
void fs_bdev_unregister(struct file *bdev_file, struct super_block *sb);
void fs_bdev_file_release(struct file *bdev_file, struct super_block *sb);
#endif /* _LINUX_FS_SUPER_H */

View File

@ -30,6 +30,7 @@ struct mount;
struct mtd_info;
struct quotactl_ops;
struct shrinker;
struct super_dev;
struct unicode_map;
struct user_namespace;
struct workqueue_struct;
@ -132,6 +133,7 @@ struct super_operations {
struct super_block {
struct list_head s_list; /* Keep this first */
dev_t s_dev; /* search index; _not_ kdev_t */
struct super_dev *s_super_dev; /* sget_fc()'s device table claim */
unsigned char s_blocksize_bits;
unsigned long s_blocksize;
loff_t s_maxbytes; /* Max file size */
@ -145,7 +147,7 @@ struct super_block {
unsigned long s_magic;
struct dentry *s_root;
struct rw_semaphore s_umount;
int s_count;
refcount_t s_passive;
atomic_t s_active;
#ifdef CONFIG_SECURITY
void *s_security;

View File

@ -163,6 +163,8 @@ typedef u32 dma_addr_t;
typedef unsigned int __bitwise gfp_t;
typedef unsigned int __bitwise slab_flags_t;
typedef unsigned int __bitwise fmode_t;
typedef unsigned int __bitwise blk_mode_t;
typedef unsigned int __bitwise fop_flags_t;
#ifdef CONFIG_PHYS_ADDR_T_64BIT
typedef u64 phys_addr_t;

View File

@ -5,3 +5,4 @@ fclog
file_stressor
anon_inode_test
kernfs_test
ustat_test

View File

@ -1,7 +1,7 @@
# SPDX-License-Identifier: GPL-2.0
CFLAGS += $(KHDR_INCLUDES)
TEST_GEN_PROGS := devpts_pts file_stressor anon_inode_test kernfs_test fclog
TEST_GEN_PROGS := devpts_pts file_stressor anon_inode_test kernfs_test fclog ustat_test
TEST_GEN_PROGS_EXTENDED := dnotify_test
include ../lib.mk

View File

@ -0,0 +1,135 @@
// SPDX-License-Identifier: GPL-2.0
/*
* Test ustat(2): looking up superblocks by device number.
*
* ustat() resolves a device number to a mounted superblock via
* user_get_super(). Check that the device number of a mounted tmpfs (an
* anonymous device) resolves, that it stops resolving once the filesystem
* is unmounted and that bogus device numbers report EINVAL.
*/
#define _GNU_SOURCE
#include <errno.h>
#include <fcntl.h>
#include <sched.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/mount.h>
#include <sys/stat.h>
#include <sys/syscall.h>
#include <unistd.h>
#include "../kselftest_harness.h"
/* struct ustat is not exported through UAPI, mirror include/linux/types.h. */
struct ustat_buf {
int f_tfree;
unsigned long f_tinode;
char f_fname[6];
char f_fpack[6];
/* slack in case an architecture lays the struct out differently */
char pad[64];
};
#ifdef __NR_ustat
/*
* The kernel decodes @dev with new_decode_dev(), which matches the low 32
* bits of the st_dev encoding stat(2) returns for any major below 4096.
*/
static int sys_ustat(unsigned int dev, struct ustat_buf *buf)
{
return syscall(__NR_ustat, dev, buf);
}
static int write_string(const char *path, const char *string)
{
ssize_t len = strlen(string);
int fd;
fd = open(path, O_WRONLY);
if (fd < 0)
return -1;
if (write(fd, string, len) != len) {
close(fd);
return -1;
}
return close(fd);
}
/* Enter namespaces in which mounting a tmpfs instance is allowed. */
static int setup_namespaces(void)
{
uid_t uid = getuid();
gid_t gid = getgid();
char map[64];
if (unshare(CLONE_NEWNS | (uid ? CLONE_NEWUSER : 0)))
return -1;
if (uid) {
if (write_string("/proc/self/setgroups", "deny"))
return -1;
snprintf(map, sizeof(map), "0 %d 1", uid);
if (write_string("/proc/self/uid_map", map))
return -1;
snprintf(map, sizeof(map), "0 %d 1", gid);
if (write_string("/proc/self/gid_map", map))
return -1;
}
return mount(NULL, "/", NULL, MS_REC | MS_PRIVATE, NULL);
}
TEST(resolves_mounted_superblock)
{
char dir[] = "/tmp/ustat_test.XXXXXX";
struct ustat_buf ub;
struct stat st;
ASSERT_NE(NULL, mkdtemp(dir));
if (setup_namespaces()) {
rmdir(dir);
SKIP(return, "cannot set up namespaces: %s", strerror(errno));
}
ASSERT_EQ(0, mount("ustat_test", dir, "tmpfs", 0, NULL));
ASSERT_EQ(0, stat(dir, &st));
memset(&ub, 0xff, sizeof(ub));
ASSERT_EQ(0, sys_ustat(st.st_dev, &ub))
TH_LOG("ustat(%u): %s", (unsigned int)st.st_dev,
strerror(errno));
ASSERT_EQ(0, umount(dir));
/* The unmount removed the superblock, the device is gone. */
ASSERT_EQ(-1, sys_ustat(st.st_dev, &ub));
ASSERT_EQ(EINVAL, errno);
rmdir(dir);
}
TEST(bogus_device_numbers)
{
struct ustat_buf ub;
ASSERT_EQ(-1, sys_ustat(0, &ub));
ASSERT_EQ(EINVAL, errno);
/* major 4095, minor 1048575: nothing plausible lives there */
ASSERT_EQ(-1, sys_ustat((0xfffu << 8) | 0xffu | (0xfff00u << 12), &ub));
ASSERT_EQ(EINVAL, errno);
}
#else /* !__NR_ustat */
TEST(unsupported)
{
SKIP(return, "ustat(2) is not available on this architecture");
}
#endif /* __NR_ustat */
TEST_HARNESS_MAIN