mirror of
https://github.com/torvalds/linux.git
synced 2026-07-28 01:55:51 +02:00
Merge patch series "proc: subset=pid: Relax check of mount visibility"
Alexey Gladkov <legion@kernel.org> says: When mounting procfs with the subset=pids option, all static files become unavailable and only the dynamic part with information about pids is accessible. In this case, there is no point in imposing additional restrictions on the visibility of the entire filesystem for the mounter. Everything that can be hidden in procfs is already inaccessible. Currently, these restrictions prevent procfs from being mounted inside rootless containers, as almost all container implementations override part of procfs to hide certain directories. Relaxing these restrictions will allow pidfs to be used in nested containerization. * patches from https://patch.msgid.link/cover.1777278334.git.legion@kernel.org: docs: proc: add documentation about mount restrictions proc: handle subset=pid separately in userns visibility checks proc: prevent reconfiguring subset=pid proc: subset=pid: Show /proc/self/net only for CAP_NET_ADMIN sysfs: remove trivial sysfs_get_tree() wrapper fs: move SB_I_USERNS_VISIBLE to FS_USERNS_MOUNT_RESTRICTED namespace: record fully visible mounts in list Link: https://patch.msgid.link/cover.1777278334.git.legion@kernel.org Signed-off-by: Christian Brauner <brauner@kernel.org>
This commit is contained in:
commit
a76640171b
|
|
@ -52,6 +52,7 @@ fixes/update part 1.1 Stefani Seibold <stefani@seibold.net> June 9 2009
|
|||
|
||||
4 Configuring procfs
|
||||
4.1 Mount options
|
||||
4.2 Mount restrictions
|
||||
|
||||
5 Filesystem behavior
|
||||
|
||||
|
|
@ -2425,7 +2426,9 @@ prohibited by hidepid=. If you use some daemon like identd which needs to learn
|
|||
information about processes information, just add identd to this group.
|
||||
|
||||
subset=pid hides all top level files and directories in the procfs that
|
||||
are not related to tasks.
|
||||
are not related to tasks. This option cannot be changed on an existing
|
||||
procfs instance because overmounts that existed before the change could
|
||||
otherwise remain reachable after the top level procfs entries are hidden.
|
||||
|
||||
pidns= specifies a pid namespace (either as a string path to something like
|
||||
`/proc/$pid/ns/pid`, or a file descriptor when using `FSCONFIG_SET_FD`) that
|
||||
|
|
@ -2434,6 +2437,20 @@ will use the calling process's active pid namespace. Note that the pid
|
|||
namespace of an existing procfs instance cannot be modified (attempting to do
|
||||
so will give an `-EBUSY` error).
|
||||
|
||||
4.2 Mount restrictions
|
||||
--------------------------
|
||||
|
||||
If user namespaces are in use, the kernel additionally checks the instances of
|
||||
procfs available to the mounter and will not allow procfs to be mounted if:
|
||||
|
||||
1. This mount is not fully visible unless the new procfs is going to be
|
||||
mounted with subset=pid option.
|
||||
|
||||
a. Its root directory is not the root directory of the filesystem.
|
||||
b. If any file or non-empty procfs directory is hidden by another mount.
|
||||
|
||||
2. A new mount overrides the readonly option or any option from atime family.
|
||||
|
||||
Chapter 5: Filesystem behavior
|
||||
==============================
|
||||
|
||||
|
|
|
|||
|
|
@ -25,6 +25,7 @@ struct mnt_namespace {
|
|||
__u32 n_fsnotify_mask;
|
||||
struct fsnotify_mark_connector __rcu *n_fsnotify_marks;
|
||||
#endif
|
||||
struct hlist_head mnt_visible_mounts; /* SB_I_USERNS_VISIBLE mounts */
|
||||
unsigned int nr_mounts; /* # of mounts in the namespace */
|
||||
unsigned int pending_mounts;
|
||||
refcount_t passive; /* number references not pinning @mounts */
|
||||
|
|
@ -90,6 +91,7 @@ struct mount {
|
|||
int mnt_expiry_mark; /* true if marked for expiry */
|
||||
struct hlist_head mnt_pins;
|
||||
struct hlist_head mnt_stuck_children;
|
||||
struct hlist_node mnt_ns_visible; /* link in ns->mnt_visible_mounts */
|
||||
struct mount *overmount; /* mounted on ->mnt_root */
|
||||
} __randomize_layout;
|
||||
|
||||
|
|
@ -207,6 +209,8 @@ static inline void move_from_ns(struct mount *mnt)
|
|||
ns->mnt_first_node = rb_next(&mnt->mnt_node);
|
||||
rb_erase(&mnt->mnt_node, &ns->mounts);
|
||||
RB_CLEAR_NODE(&mnt->mnt_node);
|
||||
if (!hlist_unhashed(&mnt->mnt_ns_visible))
|
||||
hlist_del_init(&mnt->mnt_ns_visible);
|
||||
}
|
||||
|
||||
bool has_locked_children(struct mount *mnt, struct dentry *dentry);
|
||||
|
|
|
|||
|
|
@ -321,6 +321,7 @@ static struct mount *alloc_vfsmnt(const char *name)
|
|||
INIT_HLIST_NODE(&mnt->mnt_slave);
|
||||
INIT_HLIST_NODE(&mnt->mnt_mp_list);
|
||||
INIT_HLIST_HEAD(&mnt->mnt_stuck_children);
|
||||
INIT_HLIST_NODE(&mnt->mnt_ns_visible);
|
||||
RB_CLEAR_NODE(&mnt->mnt_node);
|
||||
mnt->mnt.mnt_idmap = &nop_mnt_idmap;
|
||||
}
|
||||
|
|
@ -1098,6 +1099,10 @@ static void mnt_add_to_ns(struct mnt_namespace *ns, struct mount *mnt)
|
|||
rb_link_node(&mnt->mnt_node, parent, link);
|
||||
rb_insert_color(&mnt->mnt_node, &ns->mounts);
|
||||
|
||||
if ((mnt->mnt.mnt_sb->s_type->fs_flags & FS_USERNS_MOUNT_RESTRICTED) &&
|
||||
mnt->mnt.mnt_root == mnt->mnt.mnt_sb->s_root)
|
||||
hlist_add_head(&mnt->mnt_ns_visible, &ns->mnt_visible_mounts);
|
||||
|
||||
mnt_notify_add(mnt);
|
||||
}
|
||||
|
||||
|
|
@ -6340,20 +6345,26 @@ static bool mnt_already_visible(struct mnt_namespace *ns,
|
|||
int *new_mnt_flags)
|
||||
{
|
||||
int new_flags = *new_mnt_flags;
|
||||
struct mount *mnt, *n;
|
||||
struct mount *mnt;
|
||||
|
||||
/* Don't acquire namespace semaphore without a good reason. */
|
||||
if (hlist_empty(&ns->mnt_visible_mounts))
|
||||
return false;
|
||||
|
||||
guard(namespace_shared)();
|
||||
rbtree_postorder_for_each_entry_safe(mnt, n, &ns->mounts, mnt_node) {
|
||||
hlist_for_each_entry(mnt, &ns->mnt_visible_mounts, mnt_ns_visible) {
|
||||
const struct super_block *sb_visible = mnt->mnt.mnt_sb;
|
||||
struct mount *child;
|
||||
int mnt_flags;
|
||||
|
||||
if (mnt->mnt.mnt_sb->s_type != sb->s_type)
|
||||
if (sb_visible->s_type != sb->s_type)
|
||||
continue;
|
||||
|
||||
/* This mount is not fully visible if it's root directory
|
||||
* is not the root directory of the filesystem.
|
||||
/*
|
||||
* Restricted variants are not compatible with anything, even
|
||||
* other restricted variants.
|
||||
*/
|
||||
if (mnt->mnt.mnt_root != mnt->mnt.mnt_sb->s_root)
|
||||
if (sb_visible->s_iflags & SB_I_RESTRICTED_VARIANT)
|
||||
continue;
|
||||
|
||||
/* A local view of the mount flags */
|
||||
|
|
@ -6405,16 +6416,23 @@ static bool mount_too_revealing(const struct super_block *sb, int *new_mnt_flags
|
|||
return false;
|
||||
|
||||
/* Can this filesystem be too revealing? */
|
||||
s_iflags = sb->s_iflags;
|
||||
if (!(s_iflags & SB_I_USERNS_VISIBLE))
|
||||
if (!(sb->s_type->fs_flags & FS_USERNS_MOUNT_RESTRICTED))
|
||||
return false;
|
||||
|
||||
s_iflags = sb->s_iflags;
|
||||
if ((s_iflags & required_iflags) != required_iflags) {
|
||||
WARN_ONCE(1, "Expected s_iflags to contain 0x%lx\n",
|
||||
required_iflags);
|
||||
return true;
|
||||
}
|
||||
|
||||
/*
|
||||
* Restricted variants don't need an already visible mount because they
|
||||
* don't expose the full filesystem view.
|
||||
*/
|
||||
if (s_iflags & SB_I_RESTRICTED_VARIANT)
|
||||
return false;
|
||||
|
||||
return !mnt_already_visible(ns, sb, new_mnt_flags);
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@
|
|||
#include <linux/uidgid.h>
|
||||
#include <net/net_namespace.h>
|
||||
#include <linux/seq_file.h>
|
||||
#include <linux/security.h>
|
||||
|
||||
#include "internal.h"
|
||||
|
||||
|
|
@ -270,6 +271,7 @@ static struct net *get_proc_task_net(struct inode *dir)
|
|||
struct task_struct *task;
|
||||
struct nsproxy *ns;
|
||||
struct net *net = NULL;
|
||||
struct proc_fs_info *fs_info = proc_sb_info(dir->i_sb);
|
||||
|
||||
rcu_read_lock();
|
||||
task = pid_task(proc_pid(dir), PIDTYPE_PID);
|
||||
|
|
@ -282,6 +284,12 @@ static struct net *get_proc_task_net(struct inode *dir)
|
|||
}
|
||||
rcu_read_unlock();
|
||||
|
||||
if (net && (fs_info->pidonly == PROC_PIDONLY_ON) &&
|
||||
security_capable(fs_info->mounter_cred, net->user_ns, CAP_NET_ADMIN, CAP_OPT_NONE) < 0) {
|
||||
put_net(net);
|
||||
net = NULL;
|
||||
}
|
||||
|
||||
return net;
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -223,12 +223,17 @@ static int proc_parse_param(struct fs_context *fc, struct fs_parameter *param)
|
|||
return 0;
|
||||
}
|
||||
|
||||
static void proc_apply_options(struct proc_fs_info *fs_info,
|
||||
static int proc_apply_options(struct proc_fs_info *fs_info,
|
||||
struct fs_context *fc,
|
||||
struct user_namespace *user_ns)
|
||||
{
|
||||
struct proc_fs_context *ctx = fc->fs_private;
|
||||
|
||||
if ((ctx->mask & (1 << Opt_subset)) &&
|
||||
fc->purpose == FS_CONTEXT_FOR_RECONFIGURE &&
|
||||
ctx->pidonly != fs_info->pidonly)
|
||||
return invalf(fc, "proc: subset=pid cannot be changed\n");
|
||||
|
||||
if (ctx->mask & (1 << Opt_gid))
|
||||
fs_info->pid_gid = make_kgid(user_ns, ctx->gid);
|
||||
if (ctx->mask & (1 << Opt_hidepid))
|
||||
|
|
@ -240,6 +245,7 @@ static void proc_apply_options(struct proc_fs_info *fs_info,
|
|||
put_pid_ns(fs_info->pid_ns);
|
||||
fs_info->pid_ns = get_pid_ns(ctx->pid_ns);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int proc_fill_super(struct super_block *s, struct fs_context *fc)
|
||||
|
|
@ -254,10 +260,13 @@ static int proc_fill_super(struct super_block *s, struct fs_context *fc)
|
|||
return -ENOMEM;
|
||||
|
||||
fs_info->pid_ns = get_pid_ns(ctx->pid_ns);
|
||||
proc_apply_options(fs_info, fc, current_user_ns());
|
||||
fs_info->mounter_cred = get_cred(fc->cred);
|
||||
ret = proc_apply_options(fs_info, fc, current_user_ns());
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
/* User space would break if executables or devices appear on proc */
|
||||
s->s_iflags |= SB_I_USERNS_VISIBLE | SB_I_NOEXEC | SB_I_NODEV;
|
||||
s->s_iflags |= SB_I_NOEXEC | SB_I_NODEV;
|
||||
s->s_flags |= SB_NODIRATIME | SB_NOSUID | SB_NOEXEC;
|
||||
s->s_blocksize = 1024;
|
||||
s->s_blocksize_bits = 10;
|
||||
|
|
@ -266,6 +275,9 @@ static int proc_fill_super(struct super_block *s, struct fs_context *fc)
|
|||
s->s_time_gran = 1;
|
||||
s->s_fs_info = fs_info;
|
||||
|
||||
if (fs_info->pidonly == PROC_PIDONLY_ON)
|
||||
s->s_iflags |= SB_I_RESTRICTED_VARIANT;
|
||||
|
||||
/*
|
||||
* procfs isn't actually a stacking filesystem; however, there is
|
||||
* too much magic going on inside it to permit stacking things on
|
||||
|
|
@ -303,8 +315,7 @@ static int proc_reconfigure(struct fs_context *fc)
|
|||
|
||||
sync_filesystem(sb);
|
||||
|
||||
proc_apply_options(fs_info, fc, current_user_ns());
|
||||
return 0;
|
||||
return proc_apply_options(fs_info, fc, current_user_ns());
|
||||
}
|
||||
|
||||
static int proc_get_tree(struct fs_context *fc)
|
||||
|
|
@ -350,6 +361,7 @@ static void proc_kill_sb(struct super_block *sb)
|
|||
kill_anon_super(sb);
|
||||
if (fs_info) {
|
||||
put_pid_ns(fs_info->pid_ns);
|
||||
put_cred(fs_info->mounter_cred);
|
||||
kfree_rcu(fs_info, rcu);
|
||||
}
|
||||
}
|
||||
|
|
@ -359,7 +371,7 @@ static struct file_system_type proc_fs_type = {
|
|||
.init_fs_context = proc_init_fs_context,
|
||||
.parameters = proc_fs_parameters,
|
||||
.kill_sb = proc_kill_sb,
|
||||
.fs_flags = FS_USERNS_MOUNT | FS_DISALLOW_NOTIFY_PERM,
|
||||
.fs_flags = FS_USERNS_MOUNT | FS_USERNS_MOUNT_RESTRICTED | FS_DISALLOW_NOTIFY_PERM,
|
||||
};
|
||||
|
||||
void __init proc_root_init(void)
|
||||
|
|
|
|||
|
|
@ -23,20 +23,6 @@
|
|||
static struct kernfs_root *sysfs_root;
|
||||
struct kernfs_node *sysfs_root_kn;
|
||||
|
||||
static int sysfs_get_tree(struct fs_context *fc)
|
||||
{
|
||||
struct kernfs_fs_context *kfc = fc->fs_private;
|
||||
int ret;
|
||||
|
||||
ret = kernfs_get_tree(fc);
|
||||
if (ret)
|
||||
return ret;
|
||||
|
||||
if (kfc->new_sb_created)
|
||||
fc->root->d_sb->s_iflags |= SB_I_USERNS_VISIBLE;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void sysfs_fs_context_free(struct fs_context *fc)
|
||||
{
|
||||
struct kernfs_fs_context *kfc = fc->fs_private;
|
||||
|
|
@ -49,7 +35,7 @@ static void sysfs_fs_context_free(struct fs_context *fc)
|
|||
|
||||
static const struct fs_context_operations sysfs_fs_context_ops = {
|
||||
.free = sysfs_fs_context_free,
|
||||
.get_tree = sysfs_get_tree,
|
||||
.get_tree = kernfs_get_tree,
|
||||
};
|
||||
|
||||
static int sysfs_init_fs_context(struct fs_context *fc)
|
||||
|
|
@ -93,7 +79,7 @@ static struct file_system_type sysfs_fs_type = {
|
|||
.name = "sysfs",
|
||||
.init_fs_context = sysfs_init_fs_context,
|
||||
.kill_sb = sysfs_kill_sb,
|
||||
.fs_flags = FS_USERNS_MOUNT,
|
||||
.fs_flags = FS_USERNS_MOUNT | FS_USERNS_MOUNT_RESTRICTED,
|
||||
};
|
||||
|
||||
int __init sysfs_init(void)
|
||||
|
|
|
|||
|
|
@ -2281,6 +2281,7 @@ struct file_system_type {
|
|||
#define FS_MGTIME 64 /* FS uses multigrain timestamps */
|
||||
#define FS_LBS 128 /* FS supports LBS */
|
||||
#define FS_POWER_FREEZE 256 /* Always freeze on suspend/hibernate */
|
||||
#define FS_USERNS_MOUNT_RESTRICTED 512 /* Restrict mount in userns if not already visible */
|
||||
#define FS_RENAME_DOES_D_MOVE 32768 /* FS will handle d_move() during rename() internally. */
|
||||
int (*init_fs_context)(struct fs_context *);
|
||||
const struct fs_parameter_spec *parameters;
|
||||
|
|
|
|||
|
|
@ -326,7 +326,7 @@ struct super_block {
|
|||
#define SB_I_STABLE_WRITES 0x00000008 /* don't modify blks until WB is done */
|
||||
|
||||
/* sb->s_iflags to limit user namespace mounts */
|
||||
#define SB_I_USERNS_VISIBLE 0x00000010 /* fstype already mounted */
|
||||
#define SB_I_RESTRICTED_VARIANT 0x00000010
|
||||
#define SB_I_IMA_UNVERIFIABLE_SIGNATURE 0x00000020
|
||||
#define SB_I_UNTRUSTED_MOUNTER 0x00000040
|
||||
#define SB_I_EVM_HMAC_UNSUPPORTED 0x00000080
|
||||
|
|
|
|||
|
|
@ -67,6 +67,7 @@ enum proc_pidonly {
|
|||
struct proc_fs_info {
|
||||
struct pid_namespace *pid_ns;
|
||||
kgid_t pid_gid;
|
||||
const struct cred *mounter_cred;
|
||||
enum proc_hidepid hide_pid;
|
||||
enum proc_pidonly pidonly;
|
||||
struct rcu_head rcu;
|
||||
|
|
|
|||
|
|
@ -249,7 +249,7 @@ static int acct_on(const char __user *name)
|
|||
return -EINVAL;
|
||||
|
||||
/* Exclude procfs and sysfs. */
|
||||
if (file_inode(file)->i_sb->s_iflags & SB_I_USERNS_VISIBLE)
|
||||
if (file_inode(file)->i_sb->s_type->fs_flags & FS_USERNS_MOUNT_RESTRICTED)
|
||||
return -EINVAL;
|
||||
|
||||
if (!(file->f_mode & FMODE_CAN_WRITE))
|
||||
|
|
|
|||
Loading…
Reference in New Issue
Block a user