vfs-7.3-rc3.fixes

Please consider pulling these changes from the signed vfs-7.3-rc3.fixes tag.
 
 Thanks!
 Christian
 -----BEGIN PGP SIGNATURE-----
 
 iHUEABYKAB0WIQRAhzRXHqcMeLMyaSiRxhvAZXjcogUCaqFWOQAKCRCRxhvAZXjc
 otLlAP9X02ybdUt9NndBK8LjslDWwB9hOXzPgYsOKYODEqODjQD/aLpbXVEsA1yy
 SLdSDtbtpf+01z4KHorvAakBzk/jrw4=
 =OzBX
 -----END PGP SIGNATURE-----

Merge tag 'vfs-7.3-rc3.fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs

Pull vfs fixes from Christian Brauner:

 - netfs:

     - Fix an uninitialized return value in netfs_unbuffered_write()
       when preparing the first subrequest fails

     - For partial unbuffered/DIO writes return the amount transferred
       rather than an error

     - Update i_size with the amount actually written when a partial
       transfer ends in an error

     - Fix a subrequest reference leak when the io_iter ends up empty

     - Handle netfs_alloc_subrequest() failure during unbuffered writes

     - Load all readahead folios into the rolling buffer upfront and
       drop the readahead references once the first subrequest is
       dispatched

     - Mark folios for copy-to-cache while issuing subrequests

     - Fix read progress reporting

 - afs:

     - Add the missing kunmap in the error path of afs_dir_search_bucket()

     - Fix a double kunmap in afs_edit_dir_remove()

     - Don't free an existing server's endpoint state when cleaning up a
       candidate server in afs_lookup_server()

     - Unbind peers removed from a server's address list

 - ufs:

     - Load the cylinder group metadata before creating the root dentry

     - Validate the cylinder group index and rotor positions before
       caching them

     - Treat an unreadable directory block as not empty

 - exec:

     - Close the close-on-exec files before taking exec_update_lock

       Closing a file can block on the filesystem, so a hung filesystem
       blocked everything that takes exec_update_lock and a FUSE server
       inspecting the calling process could deadlock

     - Drop the bprm loader before closing bprm->file in free_bprm()

 - exit: Hold a reference to thread_pid across proc_flush_pid()

 - reboot: Fix a use-after-free on cad_pid

 - nsfs: Keep the namespace tree fields out of the rcu_head used by
   kfree_rcu()

 - nstree: Check listing permission before taking a namespace
   reference in listns()

 - super: Return 0 when a nested thaw drops its hold while other
   freezers remain

 - ext4: Don't set I_METADATA_WRITEBACK during fastcommit replay

 - adfs: Free s_fs_info in ->kill_sb()

 - autofs: Free the inode info allocated in autofs_fill_super() when
   the root inode allocation fails

 - ovl: Return EINVAL instead of EIO on a user namespace mismatch now
   that it's a plain refusal and not an internal error

 - cachefiles: Don't cast the variable-length coherency data to a
   __be64 in the coherency tracepoint

* tag 'vfs-7.3-rc3.fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs: (28 commits)
  nstree: check listing permission before taking a namespace reference
  exec: do_close_on_exec() before taking exec_update_lock
  exit: hold a reference to thread_pid across proc_flush_pid
  fs: autofs: fix memory leak in autofs_fill_super()
  exec: Drop bprm loader before closing bprm->file
  afs: Clear stale peer app data after address list changes
  afs: Fix incorrect free in candidate cleanup in afs_lookup_server()
  afs: Fix double-unmap of directory block
  afs: Fix missing kunmap in afs_dir_search_bucket()
  ovl: return EINVAL instead of EIO in case of mismatched user_ns
  reboot: fix cad_pid use-after-free race
  cachefiles: Fix potential UAF/KASAN warning
  netfs: Fix read progress reporting
  netfs: Mark folios with COPY_TO_CACHE whilst issuing subreqs
  netfs: Fix readahead synchronisation issues by loading all folios upfront
  netfs: break unbuffered write when netfs_alloc_subrequest() fails
  netfs: Fix subreq ref leak
  netfs: Fix i_size update for partial transfer
  netfs: Fix error vs transferred passed to ->ki_complete()
  netfs: Fix unbuffered/DIO write partial transfer error return
  ...
This commit is contained in:
Linus Torvalds 2026-09-09 09:38:03 -07:00
commit 5e1287972b
39 changed files with 532 additions and 235 deletions

View File

@ -92,10 +92,7 @@ static int adfs_checkdiscrecord(struct adfs_discrecord *dr)
static void adfs_put_super(struct super_block *sb)
{
struct adfs_sb_info *asb = ADFS_SB(sb);
adfs_free_map(sb);
kfree_rcu(asb, rcu);
}
static int adfs_show_options(struct seq_file *seq, struct dentry *root)
@ -365,7 +362,7 @@ static int adfs_fill_super(struct super_block *sb, struct fs_context *fc)
ret = -EINVAL;
}
if (ret)
goto error;
return ret;
/* set up enough so that we can read an inode */
sb->s_op = &adfs_sops;
@ -406,15 +403,9 @@ static int adfs_fill_super(struct super_block *sb, struct fs_context *fc)
if (!sb->s_root) {
adfs_free_map(sb);
adfs_error(sb, "get root inode failed\n");
ret = -EIO;
goto error;
return -EIO;
}
return 0;
error:
sb->s_fs_info = NULL;
kfree(asb);
return ret;
}
static int adfs_get_tree(struct fs_context *fc)
@ -465,10 +456,19 @@ static int adfs_init_fs_context(struct fs_context *fc)
return 0;
}
static void adfs_kill_sb(struct super_block *sb)
{
struct adfs_sb_info *asb = ADFS_SB(sb);
kill_block_super(sb);
kfree_rcu(asb, rcu);
}
static struct file_system_type adfs_fs_type = {
.owner = THIS_MODULE,
.name = "adfs",
.kill_sb = kill_block_super,
.kill_sb = adfs_kill_sb,
.fs_flags = FS_REQUIRES_DEV,
.init_fs_context = adfs_init_fs_context,
.parameters = adfs_param_spec,

View File

@ -394,8 +394,11 @@ void afs_set_peer_appdata(struct afs_server *server,
struct rxrpc_peer *pn = new_alist->addrs[n].peer;
struct rxrpc_peer *po = old_alist->addrs[o].peer;
if (pn == po)
if (pn == po) {
n++;
o++;
continue;
}
if (pn < po) {
rxrpc_kernel_set_peer_data(pn, data);
n++;

View File

@ -442,7 +442,7 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
/* Check and clear the entry. */
de = &block->dirents[slot];
if (de->u.valid != 1)
goto error_unmap;
goto error;
trace_afs_edit_dir(vnode, why, afs_edit_dir_delete, b, slot,
ntohl(de->u.vnode), ntohl(de->u.unique),
@ -458,7 +458,6 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
/* Clear the constituent entries. */
next = de->u.hash_next;
memset(de, 0, sizeof(*de) * iter.nr_slots);
kunmap_local(block);
/* Adjust the hash chain: if iter->prev_entry is 0, the hashtable head
* index is previous; otherwise it's slot number of the previous entry.
@ -485,7 +484,6 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
pde = &pblock->dirents[ps];
prev_next = pde->u.hash_next;
if (prev_next != htons(entry)) {
kunmap_local(pblock);
pr_warn("%llx:%llx:%x: not prev in chain b=%x p=%x,%x e=%x %*s",
vnode->fid.vid, vnode->fid.vnode, vnode->fid.unique,
iter.bucket, iter.prev_entry, prev_next, entry,
@ -493,7 +491,6 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
goto error;
}
pde->u.hash_next = next;
kunmap_local(pblock);
}
netfs_single_mark_inode_dirty(&vnode->netfs.inode);
@ -503,18 +500,16 @@ void afs_edit_dir_remove(struct afs_vnode *vnode,
_debug("Remove %s from %u[%u]", name->name, b, slot);
out_unmap:
afs_dir_end_iter(&iter);
kunmap_local(meta);
_leave("");
return;
already_invalidated:
kunmap_local(block);
trace_afs_edit_dir(vnode, why, afs_edit_dir_delete_inval,
0, 0, 0, 0, name->name);
goto out_unmap;
error_unmap:
kunmap_local(block);
error:
trace_afs_edit_dir(vnode, why, afs_edit_dir_delete_error,
0, 0, 0, 0, name->name);

View File

@ -75,10 +75,7 @@ union afs_xdr_dir_block *afs_dir_find_block(struct afs_dir_iter *iter, size_t bl
_enter("%zx,%d", block, slot);
if (iter->block) {
kunmap_local(iter->block);
iter->block = NULL;
}
afs_dir_end_iter(iter);
if (dvnode->directory_size < blend)
goto fail;
@ -173,12 +170,8 @@ int afs_dir_search_bucket(struct afs_dir_iter *iter, const struct qstr *name,
ret = -ENOENT;
found:
if (iter->block) {
kunmap_local(iter->block);
iter->block = NULL;
}
bad:
afs_dir_end_iter(iter);
if (ret == -ESTALE)
afs_invalidate_dir(iter->dvnode, afs_dir_invalid_iter_stale);
_leave(" = %d", ret);

View File

@ -258,6 +258,7 @@ int afs_fs_probe_fileserver(struct afs_net *net, struct afs_server *server,
lockdep_is_held(&server->fs_lock));
if (old) {
estate->responsive_set = old->responsive_set;
old_alist = old->addresses;
if (!new_alist)
new_alist = old->addresses;
}

View File

@ -1133,6 +1133,14 @@ int afs_dir_search_bucket(struct afs_dir_iter *iter, const struct qstr *name,
int afs_dir_search(struct afs_vnode *dvnode, const struct qstr *name,
struct afs_fid *_fid, afs_dataversion_t *_dir_version);
static inline void afs_dir_end_iter(struct afs_dir_iter *iter)
{
if (iter->block) {
kunmap_local(iter->block);
iter->block = NULL;
}
}
/*
* dir_silly.c
*/

View File

@ -242,7 +242,6 @@ struct afs_server *afs_lookup_server(struct afs_cell *cell, struct key *key,
out:
afs_put_addrlist(alist, afs_alist_trace_put_server_create);
if (candidate) {
kfree(rcu_access_pointer(server->endpoint_state));
kfree(candidate);
afs_dec_servers_outstanding(cell->net);
}

View File

@ -323,8 +323,10 @@ static int autofs_fill_super(struct super_block *s, struct fs_context *fc)
return -ENOMEM;
root_inode = autofs_get_inode(s, S_IFDIR | 0755);
if (!root_inode)
if (!root_inode) {
autofs_free_ino(ino);
return -ENOMEM;
}
root_inode->i_uid = ctx->uid;
root_inode->i_gid = ctx->gid;

View File

@ -13,6 +13,7 @@
#include <linux/quotaops.h>
#include <linux/xattr.h>
#include <linux/slab.h>
#include <linux/unaligned.h>
#include "internal.h"
#define CACHEFILES_COOKIE_TYPE_DATA 1
@ -50,7 +51,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object)
_enter("%x,#%d", object->debug_id, len);
buf = kmalloc(sizeof(struct cachefiles_xattr) + len, GFP_KERNEL);
buf = kmalloc(sizeof(struct cachefiles_xattr) + max(len, sizeof(__be64)), GFP_KERNEL);
if (!buf)
return -ENOMEM;
@ -60,6 +61,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object)
buf->content = object->content_info;
if (test_bit(FSCACHE_COOKIE_LOCAL_WRITE, &object->cookie->flags))
buf->content = CACHEFILES_CONTENT_DIRTY;
put_unaligned_be64(0, (__be64 *)buf->data);
if (len > 0)
memcpy(buf->data, fscache_get_aux(object->cookie), len);
@ -77,8 +79,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object)
trace_cachefiles_vfs_error(object, file_inode(file), ret,
cachefiles_trace_setxattr_error);
trace_cachefiles_coherency(object, file_inode(file)->i_ino,
be64_to_cpup((__be64 *)buf->data),
buf->content,
buf->data, buf->content,
cachefiles_coherency_set_fail);
if (ret != -ENOMEM)
cachefiles_io_error_obj(
@ -86,8 +87,7 @@ int cachefiles_set_object_xattr(struct cachefiles_object *object)
"Failed to set xattr with error %d", ret);
} else {
trace_cachefiles_coherency(object, file_inode(file)->i_ino,
be64_to_cpup((__be64 *)buf->data),
buf->content,
buf->data, buf->content,
cachefiles_coherency_set_ok);
}
@ -110,9 +110,10 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file
int ret = -ESTALE;
tlen = sizeof(struct cachefiles_xattr) + len;
buf = kmalloc(tlen, GFP_KERNEL);
buf = kmalloc(sizeof(struct cachefiles_xattr) + max(len, sizeof(__be64)), GFP_KERNEL);
if (!buf)
return -ENOMEM;
put_unaligned_be64(0, (__be64 *)buf->data);
xlen = cachefiles_inject_read_error();
if (xlen == 0)
@ -148,8 +149,7 @@ int cachefiles_check_auxdata(struct cachefiles_object *object, struct file *file
out:
trace_cachefiles_coherency(object, file_inode(file)->i_ino,
be64_to_cpup((__be64 *)buf->data),
buf->content, why);
buf->data, buf->content, why);
kfree(buf);
return ret;
}

View File

@ -1164,6 +1164,20 @@ int begin_new_exec(struct linux_binprm * bprm)
if (retval)
goto out;
/*
* We have to apply CLOEXEC before we change whether the process is
* dumpable (in setup_new_exec) to avoid a race with a process in userspace
* trying to access the should-be-closed file descriptors of a process
* undergoing exec(2).
*
* This can block on filesystem ->flush() handlers, including waiting
* for FUSE daemons, so do it before exec_mmap takes the
* exec_update_lock.
* This must happen after the point of no return, and after unsharing
* the FD table.
*/
do_close_on_exec(me->files);
/*
* Must be called _before_ exec_mmap() as bprm->mm is
* not visible until then. Doing it here also ensures
@ -1214,14 +1228,6 @@ int begin_new_exec(struct linux_binprm * bprm)
clear_syscall_work_syscall_user_dispatch(me);
/*
* We have to apply CLOEXEC before we change whether the process is
* dumpable (in setup_new_exec) to avoid a race with a process in userspace
* trying to access the should-be-closed file descriptors of a process
* undergoing exec(2).
*/
do_close_on_exec(me->files);
if (bprm->secureexec) {
/* Make sure parent cannot signal privileged process. */
me->pdeath_signal = 0;
@ -1472,9 +1478,9 @@ static void free_bprm(struct linux_binprm *bprm)
/* exec swapped the mm but failed before setup_new_exec() freed it */
if (bprm->old_mm)
exec_mm_put_old(bprm->old_mm);
do_close_execat(bprm->file);
/* An unconsumed PT_INTERP substitute from a binfmt_misc loader entry. */
bprm_drop_loader(bprm);
do_close_execat(bprm->file);
do_close_execat(bprm->executable);
/* If a binfmt changed the interp, free it. */
if (bprm->interp != bprm->filename)

View File

@ -6456,9 +6456,10 @@ int ext4_chunk_trans_blocks(struct inode *inode, int nrblocks)
int ext4_mark_iloc_dirty(handle_t *handle,
struct inode *inode, struct ext4_iloc *iloc)
{
struct super_block *sb = inode->i_sb;
int err = 0;
err = ext4_emergency_state(inode->i_sb);
err = ext4_emergency_state(sb);
if (unlikely(err)) {
put_bh(iloc->bh);
return err;
@ -6473,9 +6474,13 @@ int ext4_mark_iloc_dirty(handle_t *handle,
put_bh(iloc->bh);
/*
* Mark that there's metadata writeout pending for the inode so that it
* gets properly flushed on fsync(2) and similar.
* gets properly flushed on fsync(2) and similar. We don't bother for
* fastcommit replay as that flushes the whole bdev afterwards anyway.
* It is faster this way and we avoid entering fs writeback paths which
* aren't fully initialized yet.
*/
if (!EXT4_SB(inode->i_sb)->s_journal) {
if (!ext4_handle_valid(handle) &&
!(EXT4_SB(sb)->s_mount_state & EXT4_FC_REPLAY)) {
/*
* Inode didn't need to go through dirtying, make sure it is
* attached to wb so that writeback can handle it.

View File

@ -54,6 +54,42 @@ static void netfs_rreq_expand(struct netfs_io_request *rreq,
}
}
/*
* Drop the folio refs acquired from the readahead API.
*/
static void netfs_bulk_drop_ra_refs(struct netfs_io_request *rreq)
{
struct folio_batch fbatch;
struct folio *folio;
pgoff_t nr_pages = DIV_ROUND_UP(rreq->len, PAGE_SIZE);
pgoff_t first = rreq->start / PAGE_SIZE;
XA_STATE(xas, &rreq->mapping->i_pages, first);
folio_batch_init(&fbatch);
rcu_read_lock();
xas_for_each(&xas, folio, first + nr_pages - 1) {
if (xas_retry(&xas, folio))
continue;
if (!folio_batch_add(&fbatch, folio))
folio_batch_release(&fbatch);
}
rcu_read_unlock();
folio_batch_release(&fbatch);
trace_netfs_rreq(rreq, netfs_rreq_trace_ra_put_ref);
clear_bit_unlock(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags);
wake_up(&rreq->waitq);
}
static void netfs_maybe_bulk_drop_ra_refs(struct netfs_io_request *rreq)
{
if (test_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags))
netfs_bulk_drop_ra_refs(rreq);
}
/*
* Begin an operation, and fetch the stored zero point value from the cookie if
* available.
@ -74,12 +110,8 @@ static int netfs_begin_cache_read(struct netfs_io_request *rreq, struct netfs_in
*
* Returns the limited size if successful and -ENOMEM if insufficient memory
* available.
*
* [!] NOTE: This must be run in the same thread as ->issue_read() was called
* in as we access the readahead_control struct.
*/
static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq,
struct readahead_control *ractl)
static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq)
{
struct netfs_io_request *rreq = subreq->rreq;
size_t rsize = subreq->len;
@ -87,30 +119,6 @@ static ssize_t netfs_prepare_read_iterator(struct netfs_io_subrequest *subreq,
if (subreq->source == NETFS_DOWNLOAD_FROM_SERVER)
rsize = umin(rsize, rreq->io_streams[0].sreq_max_len);
if (ractl) {
/* If we don't have sufficient folios in the rolling buffer,
* extract a folioq's worth from the readahead region at a time
* into the buffer. Note that this acquires a ref on each page
* that we will need to release later - but we don't want to do
* that until after we've started the I/O.
*/
struct folio_batch put_batch;
folio_batch_init(&put_batch);
while (rreq->submitted < subreq->start + rsize) {
ssize_t added;
added = rolling_buffer_load_from_ra(&rreq->buffer, ractl,
&put_batch);
if (added < 0) {
folio_batch_release(&put_batch);
return added;
}
rreq->submitted += added;
}
folio_batch_release(&put_batch);
}
subreq->len = rsize;
if (unlikely(rreq->io_streams[0].sreq_max_segs)) {
size_t limit = netfs_limit_iter(&rreq->buffer.iter, 0, rsize,
@ -203,17 +211,68 @@ static void netfs_issue_read(struct netfs_io_request *rreq,
}
}
/*
* Mark folios that we want to copy to the cache. For filesystems that use
* netfslib fully, we set folio->private to NETFS_FOLIO_COPY_TO_CACHE;
* otherwise we set the deprecated PG_private_2.
*/
static void netfs_mark_copy_to_cache(struct netfs_io_request *rreq,
struct folio_queue **fq,
unsigned int *offset,
int *slot,
size_t len,
bool copy)
{
while (len > 0) {
struct folio *folio;
size_t fsize, overlap;
if (!*fq)
break;
if (*slot >= folioq_count(*fq)) {
*fq = (*fq)->next;
*slot = 0;
*offset = 0;
continue;
}
/* Determine how much the subreq overlaps the folio, if at all. */
fsize = folioq_folio_size(*fq, *slot);
overlap = min(len, fsize - *offset);
if (overlap > 0 && copy) {
folio = folioq_folio(*fq, *slot);
if (unlikely(test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags))) {
if (!folio_test_private_2(folio))
folio_start_private_2(folio);
} else {
if (!folio_get_private(folio))
folio_attach_private(folio, NETFS_FOLIO_COPY_TO_CACHE);
}
trace_netfs_folio(folio, netfs_folio_trace_mark_copy);
}
len -= overlap;
*offset += overlap;
if (*offset >= fsize) {
*slot += 1;
*offset = 0;
}
}
}
/*
* Perform a read to the pagecache from a series of sources of different types,
* slicing up the region to be read according to available cache blocks and
* network rsize.
*/
static void netfs_read_to_pagecache(struct netfs_io_request *rreq,
struct readahead_control *ractl)
static void netfs_read_to_pagecache(struct netfs_io_request *rreq)
{
struct folio_queue *fq = rreq->buffer.tail;
unsigned long long start = rreq->start;
unsigned int offset = 0;
ssize_t size = rreq->len;
int ret = 0;
int ret = 0, slot = 0;
do {
struct netfs_io_subrequest *subreq;
@ -288,7 +347,7 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq,
break;
issue:
slice = netfs_prepare_read_iterator(subreq, ractl);
slice = netfs_prepare_read_iterator(subreq);
if (slice < 0) {
ret = slice;
netfs_cancel_read(subreq, ret);
@ -301,7 +360,15 @@ static void netfs_read_to_pagecache(struct netfs_io_request *rreq,
set_bit(NETFS_RREQ_ALL_QUEUED, &rreq->flags);
}
if (fq) {
/* See if the cache indicated this should be cached. */
bool copy = test_bit(NETFS_SREQ_COPY_TO_CACHE, &subreq->flags);
netfs_mark_copy_to_cache(rreq, &fq, &slot, &offset, slice, copy);
}
netfs_issue_read(rreq, subreq);
netfs_maybe_bulk_drop_ra_refs(rreq);
if (test_bit(NETFS_RREQ_PAUSE, &rreq->flags))
netfs_wait_for_paused_read(rreq);
@ -339,7 +406,8 @@ void netfs_readahead(struct readahead_control *ractl)
{
struct netfs_io_request *rreq;
struct netfs_inode *ictx = netfs_inode(ractl->mapping->host);
unsigned long long start = readahead_pos(ractl);
ssize_t added;
uoff_t start = readahead_pos(ractl);
size_t size = readahead_length(ractl);
int ret;
@ -360,11 +428,24 @@ void netfs_readahead(struct readahead_control *ractl)
netfs_rreq_expand(rreq, ractl);
rreq->submitted = rreq->start;
if (rolling_buffer_init(&rreq->buffer, rreq->debug_id, ITER_DEST, rreq->gfp) < 0)
/* Load the folios to be read into a bvecq chain. Note that this
* acquires a ref on each folio that we will need to release later -
* but we don't want to do that until after we've started the I/O.
*/
added = rolling_buffer_bulk_load_from_ra(&rreq->buffer, ractl,
rreq->debug_id, rreq->gfp);
if (added < 0) {
ret = added;
goto cleanup_free;
netfs_read_to_pagecache(rreq, ractl);
}
__set_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags);
rreq->submitted = rreq->start + added;
rreq->cleaned_to = rreq->start;
netfs_read_set_unlock_at(rreq);
netfs_read_to_pagecache(rreq);
netfs_maybe_bulk_drop_ra_refs(rreq);
return netfs_put_request(rreq, netfs_rreq_trace_put_return);
cleanup_free:
@ -387,6 +468,7 @@ static int netfs_create_singular_buffer(struct netfs_io_request *rreq, struct fo
if (added < 0)
return added;
rreq->submitted = rreq->start + added;
rreq->progress_at = added;
return 0;
}
@ -457,7 +539,7 @@ static int netfs_read_gaps(struct file *file, struct folio *folio)
iov_iter_bvec(&rreq->buffer.iter, ITER_DEST, bvec, i, rreq->len);
rreq->submitted = rreq->start + flen;
netfs_read_to_pagecache(rreq, NULL);
netfs_read_to_pagecache(rreq);
ret = netfs_wait_for_read(rreq);
if (ret >= 0) {
@ -532,7 +614,7 @@ int netfs_read_folio(struct file *file, struct folio *folio)
if (ret < 0)
goto discard;
netfs_read_to_pagecache(rreq, NULL);
netfs_read_to_pagecache(rreq);
ret = netfs_wait_for_read(rreq);
netfs_put_request(rreq, netfs_rreq_trace_put_return);
return ret < 0 ? ret : 0;
@ -689,7 +771,7 @@ int netfs_write_begin(struct netfs_inode *ctx,
if (ret < 0)
goto error_put;
netfs_read_to_pagecache(rreq, NULL);
netfs_read_to_pagecache(rreq);
ret = netfs_wait_for_read(rreq);
netfs_put_request(rreq, netfs_rreq_trace_put_return);
if (ret < 0)
@ -754,7 +836,7 @@ int netfs_prefetch_for_write(struct file *file, struct folio *folio,
if (ret < 0)
goto error_put;
netfs_read_to_pagecache(rreq, NULL);
netfs_read_to_pagecache(rreq);
ret = netfs_wait_for_read(rreq);
netfs_put_request(rreq, netfs_rreq_trace_put_return);
return ret < 0 ? ret : 0;

View File

@ -21,7 +21,7 @@ static void netfs_unbuffered_write_done(struct netfs_io_request *wreq)
/* Okay, declare that all I/O is complete. */
trace_netfs_rreq(wreq, netfs_rreq_trace_write_done);
if (!wreq->error)
if (wreq->transferred)
netfs_update_i_size(ictx, &ictx->inode, wreq->start, wreq->transferred);
if (wreq->origin == NETFS_DIO_WRITE &&
@ -51,7 +51,7 @@ static void netfs_unbuffered_write_done(struct netfs_io_request *wreq)
wreq->iocb->ki_pos += written;
if (wreq->iocb->ki_complete) {
trace_netfs_rreq(wreq, netfs_rreq_trace_ki_complete);
wreq->iocb->ki_complete(wreq->iocb, wreq->error ?: written);
wreq->iocb->ki_complete(wreq->iocb, written ?: wreq->error);
}
wreq->iocb = VFS_PTR_POISON;
}
@ -95,7 +95,7 @@ static int netfs_unbuffered_write(struct netfs_io_request *wreq)
{
struct netfs_io_subrequest *subreq = NULL;
struct netfs_io_stream *stream = &wreq->io_streams[0];
int ret;
int ret = 0;
_enter("%llx", wreq->len);
@ -110,6 +110,11 @@ static int netfs_unbuffered_write(struct netfs_io_request *wreq)
if (!subreq) {
netfs_prepare_write(wreq, stream, wreq->start + wreq->transferred);
subreq = stream->construct;
if (!subreq) {
wreq->error = -ENOMEM;
ret = -ENOMEM;
break;
}
stream->construct = NULL;
}
@ -121,8 +126,14 @@ static int netfs_unbuffered_write(struct netfs_io_request *wreq)
}
iov_iter_truncate(&subreq->io_iter, wreq->len - wreq->transferred);
if (!iov_iter_count(&subreq->io_iter))
if (!iov_iter_count(&subreq->io_iter)) {
pr_warn("netfs: Unexpected zero-length iterator R=%08x\n",
wreq->debug_id);
__set_bit(NETFS_SREQ_FAILED, &subreq->flags);
netfs_write_subrequest_terminated(subreq, -EIO);
wreq->error = -EIO;
break;
}
subreq->len = netfs_limit_iter(&subreq->io_iter, 0,
stream->sreq_max_len,
@ -139,13 +150,11 @@ static int netfs_unbuffered_write(struct netfs_io_request *wreq)
if (test_bit(NETFS_SREQ_NEED_RETRY, &subreq->flags)) {
retry = true;
} else if (test_bit(NETFS_SREQ_FAILED, &subreq->flags)) {
ret = subreq->error;
wreq->error = ret;
wreq->error = subreq->error;
netfs_see_subrequest(subreq, netfs_sreq_trace_see_failed);
subreq = NULL;
break;
}
ret = 0;
if (!retry) {
netfs_unbuffered_write_collect(wreq, stream, subreq);
@ -288,11 +297,11 @@ ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *
ret = -EIOCBQUEUED;
} else {
ret = netfs_unbuffered_write(wreq);
if (ret < 0) {
_debug("begin = %zd", ret);
} else {
if (wreq->transferred) {
iocb->ki_pos += wreq->transferred;
ret = wreq->transferred ?: wreq->error;
ret = wreq->transferred;
} else if (wreq->error) {
ret = wreq->error;
}
netfs_put_request(wreq, netfs_rreq_trace_put_complete);

View File

@ -79,6 +79,7 @@ ssize_t netfs_wait_for_read(struct netfs_io_request *rreq);
ssize_t netfs_wait_for_write(struct netfs_io_request *rreq);
void netfs_wait_for_paused_read(struct netfs_io_request *rreq);
void netfs_wait_for_paused_write(struct netfs_io_request *rreq);
void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq);
/*
* objects.c
@ -109,6 +110,8 @@ static inline void netfs_see_subrequest(struct netfs_io_subrequest *subreq,
/*
* read_collect.c
*/
void netfs_cancel_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio);
void netfs_read_set_unlock_at(struct netfs_io_request *rreq);
bool netfs_read_collection(struct netfs_io_request *rreq);
void netfs_read_collection_worker(struct work_struct *work);
void netfs_cancel_read(struct netfs_io_subrequest *subreq, int error);

View File

@ -563,3 +563,22 @@ void netfs_wait_for_paused_write(struct netfs_io_request *rreq)
{
return netfs_wait_for_pause(rreq, netfs_write_collection);
}
/*
* Wait for the readahead-acquired refs to be put.
*/
void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq)
{
DEFINE_WAIT(myself);
for (;;) {
trace_netfs_rreq(rreq, netfs_rreq_trace_wait_put_ra_refs);
prepare_to_wait(&rreq->waitq, &myself, TASK_UNINTERRUPTIBLE);
if (!test_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags))
break;
schedule();
}
trace_netfs_rreq(rreq, netfs_rreq_trace_waited_put_ra_refs);
finish_wait(&rreq->waitq, &myself);
}

View File

@ -41,24 +41,32 @@ struct netfs_io_request *netfs_alloc_request(struct address_space *mapping,
memset(rreq, 0, kmem_cache_size(cache));
INIT_WORK(&rreq->cleanup_work, netfs_free_request);
rreq->gfp = gfp;
rreq->start = start;
rreq->len = len;
rreq->origin = origin;
rreq->netfs_ops = ctx->ops;
rreq->mapping = mapping;
rreq->inode = inode;
rreq->i_size = i_size_read(inode);
rreq->debug_id = atomic_inc_return(&debug_ids);
rreq->wsize = INT_MAX;
rreq->gfp = gfp;
rreq->start = start;
rreq->collected_to = start;
rreq->cleaned_to = start;
rreq->len = len;
rreq->progress_at = 0;
rreq->origin = origin;
rreq->netfs_ops = ctx->ops;
rreq->mapping = mapping;
rreq->inode = inode;
rreq->i_size = i_size_read(inode);
rreq->debug_id = atomic_inc_return(&debug_ids);
rreq->wsize = INT_MAX;
rreq->io_streams[0].sreq_max_len = ULONG_MAX;
rreq->io_streams[0].sreq_max_segs = 0;
spin_lock_init(&rreq->lock);
INIT_LIST_HEAD(&rreq->io_streams[0].subrequests);
INIT_LIST_HEAD(&rreq->io_streams[1].subrequests);
init_waitqueue_head(&rreq->waitq);
refcount_set(&rreq->ref, 2);
for (int s = 0; s < NR_IO_STREAMS; s++) {
struct netfs_io_stream *stream = &rreq->io_streams[s];
INIT_LIST_HEAD(&stream->subrequests);
stream->collected_to = rreq->start;
}
if (origin == NETFS_READAHEAD ||
origin == NETFS_READPAGE ||
origin == NETFS_READ_GAPS ||

View File

@ -19,7 +19,6 @@
#define MADE_PROGRESS 0x04 /* Made progress cleaning up a stream or the folio set */
#define BUFFERED 0x08 /* The pagecache needs cleaning up */
#define NEED_RETRY 0x10 /* A front op requests retrying */
#define COPY_TO_CACHE 0x40 /* Need to copy subrequest to cache */
#define ABANDON_SREQ 0x80 /* Need to abandon untransferred part of subrequest */
/*
@ -34,6 +33,30 @@ static void netfs_clear_unread(struct netfs_io_subrequest *subreq)
__set_bit(NETFS_SREQ_HIT_EOF, &subreq->flags);
}
/*
* Cancel the copy-to-cache mark on a folio.
*/
void netfs_cancel_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio)
{
if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) {
if (folio_get_private(folio) == NETFS_FOLIO_COPY_TO_CACHE) {
folio_detach_private(folio);
trace_netfs_folio(folio, netfs_folio_trace_cancel_copy);
} else if (netfs_folio_group(folio) == NETFS_FOLIO_COPY_TO_CACHE) {
struct netfs_folio *finfo = netfs_folio_info(folio);
finfo->netfs_group = NULL;
trace_netfs_folio(folio, netfs_folio_trace_cancel_copy);
}
} else {
// TODO: Use of PG_private_2 is deprecated.
if (folio_test_private_2(folio)) {
folio_end_private_2(folio);
trace_netfs_folio(folio, netfs_folio_trace_cancel_copy);
}
}
}
/*
* Flush, mark and unlock a folio that's now completely read. If we want to
* cache the folio, we set the group to NETFS_FOLIO_COPY_TO_CACHE, mark it
@ -48,37 +71,37 @@ static void netfs_unlock_read_folio(struct netfs_io_request *rreq,
if (unlikely(folio_pos(folio) < rreq->abandon_to)) {
trace_netfs_folio(folio, netfs_folio_trace_abandon);
netfs_cancel_copy_to_cache(rreq, folio);
goto just_unlock;
}
flush_dcache_folio(folio);
folio_mark_uptodate(folio);
if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) {
finfo = netfs_folio_info(folio);
if (finfo) {
trace_netfs_folio(folio, netfs_folio_trace_filled_gaps);
if (finfo->netfs_group)
folio_change_private(folio, finfo->netfs_group);
else
folio_detach_private(folio);
kfree(finfo);
}
if (unlikely(test_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags)))
netfs_cancel_copy_to_cache(rreq, folio);
if (test_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags)) {
if (!WARN_ON_ONCE(folio_get_private(folio) != NULL)) {
trace_netfs_folio(folio, netfs_folio_trace_copy_to_cache);
folio_attach_private(folio, NETFS_FOLIO_COPY_TO_CACHE);
folio_mark_dirty(folio);
}
if (!test_bit(NETFS_RREQ_USE_PGPRIV2, &rreq->flags)) {
if (netfs_folio_group(folio) == NETFS_FOLIO_COPY_TO_CACHE) {
trace_netfs_folio(folio, netfs_folio_trace_sched_copy);
folio_mark_dirty(folio);
} else {
finfo = netfs_folio_info(folio);
if (finfo) {
trace_netfs_folio(folio, netfs_folio_trace_filled_gaps);
if (finfo->netfs_group)
folio_change_private(folio, finfo->netfs_group);
else
folio_detach_private(folio);
kfree(finfo);
}
trace_netfs_folio(folio, netfs_folio_trace_read_done);
}
folioq_clear(folioq, slot);
} else {
// TODO: Use of PG_private_2 is deprecated.
if (test_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags))
if (folio_test_private_2(folio))
netfs_pgpriv2_copy_to_cache(rreq, folio);
}
@ -94,6 +117,35 @@ static void netfs_unlock_read_folio(struct netfs_io_request *rreq,
folioq_clear(folioq, slot);
}
/*
* Determine how much to gather before unlocking more folios.
*/
void netfs_read_set_unlock_at(struct netfs_io_request *rreq)
{
struct folio_queue *folioq = rreq->buffer.tail;
unsigned int slot = rreq->buffer.first_tail_slot;
size_t cleaned_to = rreq->cleaned_to - rreq->start;
size_t progress_at = cleaned_to;
size_t minimum = 256 * 1024;
while (progress_at < rreq->len) {
if (slot >= folioq_count(folioq)) {
folioq = folioq->next;
if (!folioq)
break;
slot = 0;
}
progress_at += folioq_folio_size(folioq, slot);
if (progress_at - cleaned_to >= minimum)
break;
slot++;
}
WRITE_ONCE(rreq->progress_at, progress_at);
trace_netfs_read_progress_at(rreq);
}
/*
* Unlock any folios we've finished with.
*/
@ -112,30 +164,31 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq,
if (slot >= folioq_nr_slots(folioq)) {
folioq = rolling_buffer_delete_spent(&rreq->buffer);
if (!folioq) {
rreq->front_folio_order = 0;
WRITE_ONCE(rreq->progress_at, rreq->len);
return;
}
slot = 0;
}
/* We have to wait for readahead refs to have been released before we
* can unlock any folios as the ref-dropper walks i_pages and the only
* thing preventing these folios from being removed is the folio lock.
*/
if (test_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags))
netfs_wait_for_put_ra_refs(rreq);
for (;;) {
struct folio *folio;
unsigned long long fpos, fend;
unsigned int order;
size_t fsize;
if (*notes & COPY_TO_CACHE)
set_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags);
folio = folioq_folio(folioq, slot);
if (WARN_ONCE(!folio_test_locked(folio),
"R=%08x: folio %lx is not locked\n",
rreq->debug_id, folio->index))
trace_netfs_folio(folio, netfs_folio_trace_not_locked);
order = folioq_folio_order(folioq, slot);
rreq->front_folio_order = order;
fsize = PAGE_SIZE << order;
fsize = folioq_folio_size(folioq, slot);
fpos = folio_pos(folio);
fend = fpos + fsize;
@ -149,8 +202,6 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq,
WRITE_ONCE(rreq->cleaned_to, fpos + fsize);
*notes |= MADE_PROGRESS;
clear_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags);
/* Clean up the head folioq. If we clear an entire folioq, then
* we can get rid of it provided it's not also the tail folioq
* being filled by the issuer.
@ -172,6 +223,8 @@ static void netfs_read_unlock_folios(struct netfs_io_request *rreq,
rreq->buffer.tail = folioq;
done:
rreq->buffer.first_tail_slot = slot;
netfs_read_set_unlock_at(rreq);
}
/*
@ -232,7 +285,7 @@ static void netfs_collect_read_results(struct netfs_io_request *rreq)
* subreqs.
*/
if (notes & BUFFERED) {
size_t fsize = PAGE_SIZE << rreq->front_folio_order;
uoff_t unlock_at = rreq->start + rreq->progress_at;
/* Clear the tail of a short read. */
if (!(notes & HIT_PENDING) &&
@ -248,16 +301,13 @@ static void netfs_collect_read_results(struct netfs_io_request *rreq)
stream->collected_to = front->start + transferred;
rreq->collected_to = stream->collected_to;
if (test_bit(NETFS_SREQ_COPY_TO_CACHE, &front->flags))
notes |= COPY_TO_CACHE;
if (test_bit(NETFS_SREQ_FAILED, &front->flags)) {
rreq->abandon_to = front->start + front->len;
front->transferred = front->len;
transferred = front->len;
trace_netfs_rreq(rreq, netfs_rreq_trace_set_abandon);
}
if (front->start + transferred >= rreq->cleaned_to + fsize ||
if (front->start + transferred >= unlock_at ||
test_bit(NETFS_SREQ_HIT_EOF, &front->flags))
netfs_read_unlock_folios(rreq, &notes);
} else {
@ -477,20 +527,22 @@ void netfs_read_collection_worker(struct work_struct *work)
void netfs_read_subreq_progress(struct netfs_io_subrequest *subreq)
{
struct netfs_io_request *rreq = subreq->rreq;
struct netfs_io_stream *stream = &rreq->io_streams[0];
size_t fsize = PAGE_SIZE << rreq->front_folio_order;
trace_netfs_sreq(subreq, netfs_sreq_trace_progress);
struct netfs_io_stream *stream = &rreq->io_streams[subreq->stream_nr];
size_t progress_at = READ_ONCE(rreq->progress_at);
uoff_t update_at = rreq->start + progress_at;
uoff_t transferred_to = subreq->start + subreq->transferred;
/* If we are at the head of the queue, wake up the collector,
* getting a ref to it if we were the ones to do so.
*/
if (subreq->start + subreq->transferred > rreq->cleaned_to + fsize &&
if (progress_at < rreq->len &&
transferred_to >= update_at &&
(rreq->origin == NETFS_READAHEAD ||
rreq->origin == NETFS_READPAGE ||
rreq->origin == NETFS_READ_FOR_WRITE) &&
list_is_first(&subreq->rreq_link, &stream->subrequests)
) {
trace_netfs_sreq(subreq, netfs_sreq_trace_progress);
__set_bit(NETFS_SREQ_MADE_PROGRESS, &subreq->flags);
netfs_wake_collector(rreq);
}

View File

@ -54,8 +54,8 @@ static void netfs_pgpriv2_copy_folio(struct netfs_io_request *creq, struct folio
/* Attach the folio to the rolling buffer. */
if (rolling_buffer_append(&creq->buffer, folio, 0, creq->gfp) < 0) {
set_bit(NETFS_RREQ_CANCEL_CACHING, &creq->flags);
folio_end_private_2(folio);
clear_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &creq->flags);
return;
}
@ -122,13 +122,14 @@ static struct netfs_io_request *netfs_pgpriv2_begin_copy_to_cache(
netfs_put_failed_request(creq);
cancel:
rreq->copy_to_cache = ERR_PTR(-ENOBUFS);
clear_bit(NETFS_RREQ_FOLIO_COPY_TO_CACHE, &rreq->flags);
set_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags);
return ERR_PTR(-ENOBUFS);
}
/*
* [DEPRECATED] Mark page as requiring copy-to-cache using PG_private_2 and add
* it to the copy write request.
* it to the copy write request. PG_private_2 should already be set on the
* folio.
*/
void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *folio)
{
@ -136,11 +137,13 @@ void netfs_pgpriv2_copy_to_cache(struct netfs_io_request *rreq, struct folio *fo
if (!creq)
creq = netfs_pgpriv2_begin_copy_to_cache(rreq, folio);
if (IS_ERR(creq))
if (IS_ERR(creq)) {
set_bit(NETFS_RREQ_CANCEL_CACHING, &rreq->flags);
netfs_cancel_copy_to_cache(rreq, folio);
return;
}
trace_netfs_folio(folio, netfs_folio_trace_copy_to_cache);
folio_start_private_2(folio);
trace_netfs_folio(folio, netfs_folio_trace_pgpriv2_copy);
netfs_pgpriv2_copy_folio(creq, folio);
}

View File

@ -292,11 +292,22 @@ void netfs_unlock_abandoned_read_pages(struct netfs_io_request *rreq)
{
struct folio_queue *p;
/* We have to wait for readahead refs to have been released before we
* can unlock any folios as the ref-dropper walks i_pages and the only
* thing preventing these folios from being removed is the folio lock.
*/
if (test_bit(NETFS_RREQ_NEED_PUT_RA_REFS, &rreq->flags))
netfs_wait_for_put_ra_refs(rreq);
for (p = rreq->buffer.tail; p; p = p->next) {
for (int slot = 0; slot < folioq_count(p); slot++) {
struct folio *folio = folioq_folio(p, slot);
if (folio && !folioq_is_marked2(p, slot)) {
if (!folio)
continue;
netfs_cancel_copy_to_cache(rreq, folio);
if (!folioq_is_marked2(p, slot)) {
if (folio == rreq->no_unlock_folio &&
test_bit(NETFS_RREQ_NO_UNLOCK_FOLIO,
&rreq->flags)) {

View File

@ -170,6 +170,8 @@ ssize_t netfs_read_single(struct inode *inode, struct file *file, struct iov_ite
if (IS_ERR(rreq))
return PTR_ERR(rreq);
rreq->progress_at = rreq->len;
ret = netfs_single_begin_cache_read(rreq, ictx);
if (ret == -ENOMEM || ret == -EINTR || ret == -ERESTARTSYS)
goto cleanup_free;

View File

@ -115,42 +115,65 @@ int rolling_buffer_make_space(struct rolling_buffer *roll, gfp_t gfp)
}
/*
* Decant the list of folios to read into a rolling buffer.
* Decant the entire list of folios to read into a rolling buffer.
*/
ssize_t rolling_buffer_load_from_ra(struct rolling_buffer *roll,
struct readahead_control *ractl,
struct folio_batch *put_batch)
ssize_t rolling_buffer_bulk_load_from_ra(struct rolling_buffer *roll,
struct readahead_control *ractl,
unsigned int rreq_id, gfp_t gfp)
{
struct folio_queue *fq;
struct page **vec;
int nr, ix, to;
ssize_t size = 0;
ssize_t loaded = 0;
if (rolling_buffer_make_space(roll, GFP_KERNEL) < 0)
return -ENOMEM;
while (ractl->_nr_pages - ractl->_batch_count > 0) {
unsigned int nr;
fq = roll->head;
vec = (struct page **)fq->vec.folios;
nr = __readahead_batch(ractl, vec + folio_batch_count(&fq->vec),
folio_batch_space(&fq->vec));
ix = fq->vec.nr;
to = ix + nr;
fq->vec.nr = to;
for (; ix < to; ix++) {
struct folio *folio = folioq_folio(fq, ix);
unsigned int order = folio_order(folio);
/* Allocate a folioq to put some folios into and attach it to
* the rolling buffer.
*/
fq = netfs_folioq_alloc(rreq_id, gfp,
netfs_trace_folioq_make_space);
if (!fq)
goto nomem_unlock;
fq->prev = roll->head;
if (!roll->tail)
roll->tail = fq;
else
roll->head->next = fq;
roll->head = fq;
fq->orders[ix] = order;
size += PAGE_SIZE << order;
trace_netfs_folio(folio, netfs_folio_trace_read);
if (!folio_batch_add(put_batch, folio))
folio_batch_release(put_batch);
/* Get a batch of folios and note their orders. */
nr = __readahead_batch(ractl, (struct page **)fq->vec.folios,
folioq_nr_slots(fq));
if (WARN_ON_ONCE(!nr))
break;
fq->vec.nr = nr;
for (int slot = 0; slot < nr; slot++) {
struct folio *folio = folioq_folio(fq, slot);
unsigned int order;
order = folio_order(folio);
fq->orders[slot] = order;
loaded += PAGE_SIZE << order;
trace_netfs_folio(folio, netfs_folio_trace_read);
}
}
WRITE_ONCE(roll->iter.count, roll->iter.count + size);
/* Store the counter after setting the slot. */
smp_store_release(&roll->next_head_slot, to);
return size;
WRITE_ONCE(roll->iter.count, loaded);
iov_iter_folio_queue(&roll->iter, ITER_DEST, roll->tail, 0, 0, loaded);
return loaded;
nomem_unlock:
for (fq = roll->tail; fq; fq = fq->next) {
for (int slot = 0; slot < folioq_count(fq); slot++) {
folio_unlock(fq->vec.folios[slot]);
folioq_mark(fq, slot);
}
}
rolling_buffer_clear(roll);
roll->head = NULL;
roll->tail = NULL;
return -ENOMEM;
}
/*

View File

@ -170,6 +170,8 @@ void netfs_prepare_write(struct netfs_io_request *wreq,
rolling_buffer_make_space(&wreq->buffer, wreq->gfp);
subreq = netfs_alloc_subrequest(wreq);
if (!subreq)
return;
subreq->source = stream->source;
subreq->start = start;
subreq->stream_nr = stream->stream_nr;

View File

@ -1543,7 +1543,7 @@ int ovl_fill_super(struct super_block *sb, struct fs_context *fc)
struct ovl_fs *ofs = sb->s_fs_info;
int err;
err = -EIO;
err = -EINVAL;
/* The fscontext fd may have been passed to another user namespace. */
if (fc->user_ns != current_user_ns())
goto out_err;

View File

@ -2369,11 +2369,14 @@ static int thaw_super_locked(struct super_block *sb, enum freeze_holder who,
goto out_unlock;
/*
* All freezers share a single active reference.
* So just unlock in case there are any left.
* All freezers share a single active reference. If other freezers
* remain, drop our hold and report success; the superblock stays
* frozen until the last holder thaws it.
*/
if (freeze_dec(sb, who))
if (freeze_dec(sb, who)) {
error = 0;
goto out_unlock;
}
if (sb_rdonly(sb)) {
sb->s_writers.frozen = SB_UNFROZEN;

View File

@ -68,6 +68,16 @@ static bool ufs_read_cylinder(struct super_block *sb,
ucpi->c_clustersumoff = fs32_to_cpu(sb, ucg->cg_u.cg_44.cg_clustersumoff);
ucpi->c_clusteroff = fs32_to_cpu(sb, ucg->cg_u.cg_44.cg_clusteroff);
ucpi->c_nclusterblks = fs32_to_cpu(sb, ucg->cg_u.cg_44.cg_nclusterblks);
/* these on-disk values become array and bitmap indices */
if (ucpi->c_cgx != cgno ||
ucpi->c_rotor >= uspi->s_fpg ||
ucpi->c_frotor >= uspi->s_fpg ||
ucpi->c_irotor >= uspi->s_ipg) {
ufs_error(sb, __func__,
"inconsistent metadata in cylinder group %u\n", cgno);
goto failed;
}
UFSD("EXIT\n");
return true;

View File

@ -590,7 +590,7 @@ int ufs_empty_dir(struct inode * inode)
kaddr = ufs_get_folio(inode, i, &folio);
if (IS_ERR(kaddr))
continue;
return 0;
de = (struct ufs_dir_entry *)kaddr;
kaddr += ufs_last_byte(inode, i) - UFS_DIR_REC_LEN(1);

View File

@ -1199,6 +1199,15 @@ static int ufs_fill_super(struct super_block *sb, struct fs_context *fc)
sb->s_maxbytes = ufs_max_bytes(sb);
sb->s_max_links = UFS_LINK_MAX;
ufs_setup_cstotal(sb);
/*
* Read cylinder group structures
*/
if (!sb_rdonly(sb))
if (!ufs_read_cylinder_structures(sb))
goto failed;
/* create the root dentry last, once UFS_SB(sb) is fully set up */
inode = ufs_iget(sb, UFS_ROOTINO);
if (IS_ERR(inode)) {
ret = PTR_ERR(inode);
@ -1210,14 +1219,6 @@ static int ufs_fill_super(struct super_block *sb, struct fs_context *fc)
goto failed;
}
ufs_setup_cstotal(sb);
/*
* Read cylinder group structures
*/
if (!sb_rdonly(sb))
if (!ufs_read_cylinder_structures(sb))
goto failed;
UFSD("EXIT\n");
return 0;

View File

@ -246,6 +246,7 @@ struct netfs_io_request {
unsigned long long submitted; /* Amount submitted for I/O so far */
unsigned long long len; /* Length of the request */
size_t transferred; /* Amount to be indicated as transferred */
size_t progress_at; /* Report read progress when hit this much read */
long error; /* 0 or error that occurred */
unsigned long long i_size; /* Size of the file */
unsigned long long start; /* Start position */
@ -262,7 +263,6 @@ struct netfs_io_request {
atomic_t subreq_counter; /* Next subreq->debug_index */
unsigned int nr_group_rel; /* Number of refs to release on ->group */
spinlock_t lock; /* Lock for queuing subreqs */
unsigned char front_folio_order; /* Order (size) of front folio */
enum netfs_io_origin origin; /* Origin of the request */
bool direct_bv_unpin; /* T if direct_bv[] must be unpinned */
refcount_t ref;
@ -275,9 +275,10 @@ struct netfs_io_request {
#define NETFS_RREQ_SHORT_TRANSFER 5 /* Set if we have a short transfer */
#define NETFS_RREQ_OFFLOAD_COLLECTION 8 /* Offload collection to workqueue */
#define NETFS_RREQ_NO_UNLOCK_FOLIO 9 /* Don't unlock no_unlock_folio on completion */
#define NETFS_RREQ_FOLIO_COPY_TO_CACHE 10 /* Copy current folio to cache from read */
#define NETFS_RREQ_CANCEL_CACHING 10 /* Set to cancel caching */
#define NETFS_RREQ_UPLOAD_TO_SERVER 11 /* Need to write to the server */
#define NETFS_RREQ_USE_IO_ITER 12 /* Use ->io_iter rather than ->i_pages */
#define NETFS_RREQ_NEED_PUT_RA_REFS 17 /* Need to put the folio refs RA gave us */
#define NETFS_RREQ_USE_PGPRIV2 31 /* [DEPRECATED] Use PG_private_2 to mark
* write to cache on read */
const struct netfs_request_ops *netfs_ops;

View File

@ -116,10 +116,8 @@ struct ns_common {
struct dentry *stashed;
const struct proc_ns_operations *ops;
unsigned int inum;
union {
struct ns_tree;
struct rcu_head ns_rcu;
};
struct ns_tree;
struct rcu_head ns_rcu;
};
#define to_ns_common(__ns) \

View File

@ -45,9 +45,9 @@ struct rolling_buffer_snapshot {
int rolling_buffer_init(struct rolling_buffer *roll, unsigned int rreq_id,
unsigned int direction, gfp_t gfp);
int rolling_buffer_make_space(struct rolling_buffer *roll, gfp_t gfp);
ssize_t rolling_buffer_load_from_ra(struct rolling_buffer *roll,
struct readahead_control *ractl,
struct folio_batch *put_batch);
ssize_t rolling_buffer_bulk_load_from_ra(struct rolling_buffer *roll,
struct readahead_control *ractl,
unsigned int rreq_id, gfp_t gfp);
ssize_t rolling_buffer_append(struct rolling_buffer *roll, struct folio *folio,
unsigned int flags, gfp_t gfp);
struct folio_queue *rolling_buffer_delete_spent(struct rolling_buffer *roll);

View File

@ -1787,7 +1787,7 @@ static inline bool is_lazy_mmu_mode_active(void)
}
#endif
extern struct pid *cad_pid;
extern struct pid __rcu *cad_pid;
/*
* Per process flags

View File

@ -562,10 +562,7 @@ static inline sigset_t *sigmask_to_save(void)
return res;
}
static inline int kill_cad_pid(int sig, int priv)
{
return kill_pid(cad_pid, sig, priv);
}
int kill_cad_pid(int sig, int priv);
/* These can be the second arg to send_sig_info/send_group_sig_info. */
#define SEND_SIG_NOINFO ((struct kernel_siginfo *) 0)

View File

@ -372,7 +372,7 @@ TRACE_EVENT(cachefiles_rename,
TRACE_EVENT(cachefiles_coherency,
TP_PROTO(struct cachefiles_object *obj,
ino_t ino,
u64 disk_aux,
const void *disk_aux,
enum cachefiles_content content,
enum cachefiles_coherency_trace why),
@ -389,12 +389,27 @@ TRACE_EVENT(cachefiles_coherency,
),
TP_fast_assign(
union {
__be16 s[4];
__be64 ll;
} x;
__entry->obj = obj->debug_id;
__entry->why = why;
__entry->content = content;
__entry->ino = ino;
__entry->aux = be64_to_cpup((__be64 *)obj->cookie->inline_aux);
__entry->disk_aux = disk_aux;
/* cachefiles_xattr::data is 2-byte aligned but not 8-byte aligned. */
if (disk_aux) {
x.s[0] = ((__be16 *)disk_aux)[0];
x.s[1] = ((__be16 *)disk_aux)[1];
x.s[2] = ((__be16 *)disk_aux)[2];
x.s[3] = ((__be16 *)disk_aux)[3];
__entry->disk_aux = be64_to_cpu(x.ll);
} else {
__entry->disk_aux = 0;
}
),
TP_printk("o=%08x %s B=%llx c=%u aux=%llx dsk=%llx",

View File

@ -59,6 +59,7 @@
EM(netfs_rreq_trace_free, "FREE ") \
EM(netfs_rreq_trace_intr, "INTR ") \
EM(netfs_rreq_trace_ki_complete, "KI-CMPL") \
EM(netfs_rreq_trace_ra_put_ref, "RA-PUT ") \
EM(netfs_rreq_trace_recollect, "RECLLCT") \
EM(netfs_rreq_trace_redirty, "REDIRTY") \
EM(netfs_rreq_trace_resubmit, "RESUBMT") \
@ -70,9 +71,11 @@
EM(netfs_rreq_trace_unpause, "UNPAUSE") \
EM(netfs_rreq_trace_wait_ip, "WAIT-IP") \
EM(netfs_rreq_trace_wait_pause, "--PAUSED--") \
EM(netfs_rreq_trace_wait_put_ra_refs, "WAIT-P-RA") \
EM(netfs_rreq_trace_wait_quiesce, "WAIT-QUIESCE") \
EM(netfs_rreq_trace_waited_ip, "DONE-IP") \
EM(netfs_rreq_trace_waited_pause, "--UNPAUSED--") \
EM(netfs_rreq_trace_waited_put_ra_refs, "DONE-P-RA") \
EM(netfs_rreq_trace_waited_quiesce, "DONE-QUIESCE") \
EM(netfs_rreq_trace_wake_ip, "WAKE-IP") \
EM(netfs_rreq_trace_wake_queue, "WAKE-Q ") \
@ -195,7 +198,6 @@
EM(netfs_folio_trace_clear_cc, "clear-cc") \
EM(netfs_folio_trace_clear_g, "clear-g") \
EM(netfs_folio_trace_clear_s, "clear-s") \
EM(netfs_folio_trace_copy_to_cache, "mark-copy") \
EM(netfs_folio_trace_end_copy, "end-copy") \
EM(netfs_folio_trace_filled_gaps, "filled-gaps") \
EM(netfs_folio_trace_invalidate_all, "inval-all") \
@ -206,16 +208,19 @@
EM(netfs_folio_trace_kill_cc, "kill-cc") \
EM(netfs_folio_trace_kill_g, "kill-g") \
EM(netfs_folio_trace_kill_s, "kill-s") \
EM(netfs_folio_trace_mark_copy, "mark-copy") \
EM(netfs_folio_trace_mkwrite, "mkwrite") \
EM(netfs_folio_trace_mkwrite_plus, "mkwrite+") \
EM(netfs_folio_trace_not_under_wback, "!wback") \
EM(netfs_folio_trace_not_locked, "!locked") \
EM(netfs_folio_trace_not_under_wback, "!wback") \
EM(netfs_folio_trace_pgpriv2_copy, "pgpriv2-copy") \
EM(netfs_folio_trace_put, "put") \
EM(netfs_folio_trace_read, "read") \
EM(netfs_folio_trace_read_done, "read-done") \
EM(netfs_folio_trace_read_gaps, "read-gaps") \
EM(netfs_folio_trace_read_unlock, "read-unlock") \
EM(netfs_folio_trace_redirtied, "redirtied") \
EM(netfs_folio_trace_sched_copy, "sched-copy") \
EM(netfs_folio_trace_store, "store") \
EM(netfs_folio_trace_store_copy, "store-copy") \
EM(netfs_folio_trace_store_plus, "store+") \
@ -786,6 +791,27 @@ TRACE_EVENT(netfs_folioq,
__print_symbolic(__entry->trace, netfs_folioq_traces))
);
TRACE_EVENT(netfs_read_progress_at,
TP_PROTO(const struct netfs_io_request *rreq),
TP_ARGS(rreq),
TP_STRUCT__entry(
__field(unsigned int, rreq)
__field(size_t, progress_at)
__field(size_t, cleaned_to)
),
TP_fast_assign(
__entry->rreq = rreq->debug_id;
__entry->cleaned_to = rreq->cleaned_to - rreq->start;
__entry->progress_at = rreq->progress_at;
),
TP_printk("R=%08x cln=%zx prg=%zx",
__entry->rreq, __entry->cleaned_to, __entry->progress_at)
);
#undef EM
#undef E_
#endif /* _TRACE_NETFS_H */

View File

@ -1648,7 +1648,7 @@ static noinline void __init kernel_init_freeable(void)
*/
set_mems_allowed(node_states[N_MEMORY]);
cad_pid = get_pid(task_pid(current));
rcu_assign_pointer(cad_pid, get_pid(task_pid(current)));
smp_prepare_cpus(setup_max_cpus);

View File

@ -261,8 +261,11 @@ void release_task(struct task_struct *p)
pidfs_exit(p);
cgroup_task_release(p);
/* Retrieve @thread_pid before __unhash_process() may set it to NULL. */
thread_pid = task_pid(p);
/*
* Pin @thread_pid before __unhash_process() clears it. The last
* PIDTYPE detach can otherwise free it before proc_flush_pid().
*/
thread_pid = get_pid(task_pid(p));
write_lock_irq(&tasklist_lock);
ptrace_release_task(p);
@ -291,8 +294,8 @@ void release_task(struct task_struct *p)
}
write_unlock_irq(&tasklist_lock);
/* @thread_pid can't go away until free_pids() below */
proc_flush_pid(thread_pid);
put_pid(thread_pid);
exit_cred_namespaces(p);
add_device_randomness(&p->se.sum_exec_runtime,
sizeof(p->se.sum_exec_runtime));

View File

@ -533,19 +533,13 @@ DEFINE_FREE(ns_put, struct ns_common *, if (!IS_ERR_OR_NULL(_T)) ns_put(_T))
static inline struct ns_common *__must_check legitimize_ns(const struct klistns *kls,
struct ns_common *candidate)
{
struct ns_common *ns __free(ns_put) = NULL;
if (!ns_requested(kls, candidate))
return NULL;
ns = ns_get_unless_inactive(candidate);
if (!ns)
if (!may_list_ns(kls, candidate))
return NULL;
if (!may_list_ns(kls, ns))
return NULL;
return no_free_ptr(ns);
return ns_get_unless_inactive(candidate);
}
static ssize_t do_listns_userns(struct klistns *kls)

View File

@ -13,7 +13,9 @@
#include <linux/kexec.h>
#include <linux/kmod.h>
#include <linux/kmsg_dump.h>
#include <linux/rcupdate.h>
#include <linux/reboot.h>
#include <linux/sched/signal.h>
#include <linux/suspend.h>
#include <linux/syscalls.h>
#include <linux/syscore_ops.h>
@ -24,8 +26,7 @@
*/
static int C_A_D = 1;
struct pid *cad_pid;
EXPORT_SYMBOL(cad_pid);
struct pid __rcu *cad_pid;
#if defined(CONFIG_ARM)
#define DEFAULT_REBOOT_MODE = REBOOT_HARD
@ -1371,10 +1372,14 @@ static int proc_do_cad_pid(const struct ctl_table *table, int write, void *buffe
{
struct ctl_table tmp_table = *table;
struct pid *new_pid;
struct pid *old_pid;
pid_t tmp_pid;
int r;
tmp_pid = pid_vnr(cad_pid);
rcu_read_lock();
tmp_pid = pid_vnr(rcu_dereference(cad_pid));
rcu_read_unlock();
tmp_table.data = &tmp_pid;
r = proc_dointvec(&tmp_table, write, buffer, lenp, ppos);
@ -1385,7 +1390,13 @@ static int proc_do_cad_pid(const struct ctl_table *table, int write, void *buffe
if (!new_pid)
return -ESRCH;
put_pid(xchg(&cad_pid, new_pid));
old_pid = unrcu_pointer(xchg(&cad_pid, RCU_INITIALIZER(new_pid)));
/*
* Wait for cad_pid readers before put_pid(). We cannot use
* call_rcu() here because free_pid() already owns pid->rcu.
*/
synchronize_rcu();
put_pid(old_pid);
return 0;
}

View File

@ -1892,6 +1892,18 @@ int kill_pid(struct pid *pid, int sig, int priv)
}
EXPORT_SYMBOL(kill_pid);
int kill_cad_pid(int sig, int priv)
{
int ret;
rcu_read_lock();
ret = kill_pid(rcu_dereference(cad_pid), sig, priv);
rcu_read_unlock();
return ret;
}
EXPORT_SYMBOL(kill_cad_pid);
#ifdef CONFIG_POSIX_TIMERS
/*
* These functions handle POSIX timer signals. POSIX timers use