mirror of
https://github.com/torvalds/linux.git
synced 2026-09-24 06:24:02 +02:00
ext4: use fsdata to track inline data write state and fix race
Instead of checking the live inode state (ext4_has_inline_data(inode)
and ext4_test_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA)) in the
write_end handlers, use the fsdata parameter of the address space
operations to explicitly pass down the state in which write_begin
prepared the write.
A concurrent thread (such as ext4_page_mkwrite()) can convert the
inline data to an extent between write_begin and write_end. If this
happens, the write_end handlers would previously miss the inline
write_end path and fall through to extent-based write_end logic.
However, since block buffers were never allocated in write_begin,
this resulted in NULL pointer dereferences or data loss because
folio_buffers(folio) was NULL.
Define EXT4_WRITE_DATA_INLINE (4) as a bit flag (Bit 2), treating
fsdata as bitwise flags rather than mutually exclusive enums to keep
states of the write path independent. Communicate this state via
fsdata:
1) ext4_write_begin() and ext4_da_write_begin() set the
EXT4_WRITE_DATA_INLINE bit in *fsdata via bitwise OR when an inline
write is successfully prepared.
2) On entry, ext4_write_begin() clears the EXT4_WRITE_DATA_INLINE bit
to safely handle VFS retries (where generic_perform_write() bypasses
the fsdata initialization on its retry jump).
3) The write_end handlers perform a bitwise AND to check if the
EXT4_WRITE_DATA_INLINE bit is set and invoke the inline write_end
helper accordingly.
Furthermore, during a buffered write, ext4_write_inline_data_end()
acquires the xattr lock after preparing the write. If a concurrent
page fault (ext4_page_mkwrite()) converts the inline data to an extent
after the write_end handlers check the state but before
ext4_write_inline_data_end() acquires the xattr write lock, the
subsequent check will trigger a kernel panic via
BUG_ON(!ext4_has_inline_data(inode)).
To keep git history working and bisectability clean, replace the
BUG_ON check in ext4_write_inline_data_end() with a graceful error-
handling retry path in this same commit. If the inline data is cleared
after locking the xattr, we safely release all resources (releasing
iloc.bh, unlocking/putting the folio, stopping the active journal
transaction handle) and return 0 (VFS retry) to let the generic write
path retry the operation safely.
Reported-by: syzbot+0c89d865531d053abb2d@syzkaller.appspotmail.com
Closes: https://syzkaller.appspot.com/bug?extid=0c89d865531d053abb2d
Fixes: 3fdcfb668f ("ext4: add journalled write support for inline data")
Suggested-by: Jan Kara <jack@suse.cz>
Signed-off-by: Aditya Prakash Srivastava <aditya.ansh182@gmail.com>
Reviewed-by: Jan Kara <jack@suse.cz>
Link: https://patch.msgid.link/20260703045414.1768-1-aditya.ansh182@gmail.com
Signed-off-by: Theodore Ts'o <tytso@mit.edu>
This commit is contained in:
parent
802b8aa8ff
commit
7edbb323ba
|
|
@ -3138,6 +3138,7 @@ int do_journal_get_write_access(handle_t *handle, struct inode *inode,
|
|||
void ext4_set_inode_mapping_order(struct inode *inode);
|
||||
#define FALL_BACK_TO_NONDELALLOC 1
|
||||
#define CONVERT_INLINE_DATA 2
|
||||
#define EXT4_WRITE_DATA_INLINE 4
|
||||
|
||||
typedef enum {
|
||||
EXT4_IGET_NORMAL = 0,
|
||||
|
|
|
|||
|
|
@ -812,7 +812,19 @@ int ext4_write_inline_data_end(struct inode *inode, loff_t pos, unsigned len,
|
|||
goto out;
|
||||
}
|
||||
ext4_write_lock_xattr(inode, &no_expand);
|
||||
BUG_ON(!ext4_has_inline_data(inode));
|
||||
/*
|
||||
* We could have raced with ext4_page_mkwrite() converting
|
||||
* the inode and clearing the inline data flag, so we just
|
||||
* release resources and retry the whole write.
|
||||
*/
|
||||
if (unlikely(!ext4_has_inline_data(inode))) {
|
||||
ext4_write_unlock_xattr(inode, &no_expand);
|
||||
brelse(iloc.bh);
|
||||
folio_unlock(folio);
|
||||
folio_put(folio);
|
||||
ext4_journal_stop(handle);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
* ei->i_inline_off may have changed since
|
||||
|
|
|
|||
|
|
@ -1304,6 +1304,8 @@ static int ext4_write_begin(const struct kiocb *iocb,
|
|||
if (unlikely(ret))
|
||||
return ret;
|
||||
|
||||
*fsdata = (void *)((unsigned long)*fsdata & ~EXT4_WRITE_DATA_INLINE);
|
||||
|
||||
trace_ext4_write_begin(inode, pos, len);
|
||||
/*
|
||||
* Reserve one block more for addition to orphan list in case
|
||||
|
|
@ -1318,8 +1320,10 @@ static int ext4_write_begin(const struct kiocb *iocb,
|
|||
foliop);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
if (ret == 1)
|
||||
if (ret == 1) {
|
||||
*fsdata = (void *)((unsigned long)*fsdata | EXT4_WRITE_DATA_INLINE);
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
|
|
@ -1452,8 +1456,7 @@ static int ext4_write_end(const struct kiocb *iocb,
|
|||
|
||||
trace_ext4_write_end(inode, pos, len, copied);
|
||||
|
||||
if (ext4_has_inline_data(inode) &&
|
||||
ext4_test_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA))
|
||||
if ((unsigned long)fsdata & EXT4_WRITE_DATA_INLINE)
|
||||
return ext4_write_inline_data_end(inode, pos, len, copied,
|
||||
folio);
|
||||
|
||||
|
|
@ -1562,8 +1565,7 @@ static int ext4_journalled_write_end(const struct kiocb *iocb,
|
|||
|
||||
BUG_ON(!ext4_handle_valid(handle));
|
||||
|
||||
if (ext4_has_inline_data(inode) &&
|
||||
ext4_test_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA))
|
||||
if ((unsigned long)fsdata & EXT4_WRITE_DATA_INLINE)
|
||||
return ext4_write_inline_data_end(inode, pos, len, copied,
|
||||
folio);
|
||||
|
||||
|
|
@ -3175,8 +3177,10 @@ static int ext4_da_write_begin(const struct kiocb *iocb,
|
|||
foliop, fsdata, true);
|
||||
if (ret < 0)
|
||||
return ret;
|
||||
if (ret == 1)
|
||||
if (ret == 1) {
|
||||
*fsdata = (void *)((unsigned long)*fsdata | EXT4_WRITE_DATA_INLINE);
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
retry:
|
||||
|
|
@ -3305,17 +3309,15 @@ static int ext4_da_write_end(const struct kiocb *iocb,
|
|||
struct folio *folio, void *fsdata)
|
||||
{
|
||||
struct inode *inode = mapping->host;
|
||||
int write_mode = (int)(unsigned long)fsdata;
|
||||
unsigned long write_mode = (unsigned long)fsdata;
|
||||
|
||||
if (write_mode == FALL_BACK_TO_NONDELALLOC)
|
||||
if (write_mode & FALL_BACK_TO_NONDELALLOC)
|
||||
return ext4_write_end(iocb, mapping, pos,
|
||||
len, copied, folio, fsdata);
|
||||
|
||||
trace_ext4_da_write_end(inode, pos, len, copied);
|
||||
|
||||
if (write_mode != CONVERT_INLINE_DATA &&
|
||||
ext4_test_inode_state(inode, EXT4_STATE_MAY_INLINE_DATA) &&
|
||||
ext4_has_inline_data(inode))
|
||||
if (write_mode & EXT4_WRITE_DATA_INLINE)
|
||||
return ext4_write_inline_data_end(inode, pos, len, copied,
|
||||
folio);
|
||||
|
||||
|
|
|
|||
Loading…
Reference in New Issue
Block a user