mirror of
https://github.com/torvalds/linux.git
synced 2026-10-09 20:09:02 +02:00
entry error handling, allocation cleanup, FITRIM bounds, and locking:
- Fix valid_size extension over shared writable mappings by lazily
zeroing page-cache gaps and ensuring racing stores cannot be
overwritten or extend beyond valid_size.
- Clean up partially written directory entries on add-entry failure and
safely free new directory clusters.
- Free a newly allocated directory cluster when zeroing it fails.
- Write the destination entry before removing the source entry during
rename, preserving the source if the destination write fails.
- Keep FITRIM within the requested range, even when the free-space search
wraps around the allocation bitmap.
- Fix truncated FS_IOC_GETFSLABEL results by passing the destination
buffer size in bytes to the UTF-16-to-NLS conversion.
- Fix overflow in the cluster-to-dentry calculation, which could cause
readdir to stop early on large volumes.
- Replace the per-inode truncate_lock with inode_lock, using shared
locking in bmap to serialize against concurrent truncation.
- Update the exfat mailing list address in MAINTAINERS.
-----BEGIN PGP SIGNATURE-----
iQJKBAABCgA0FiEE6NzKS6Uv/XAAGHgyZwv7A1FEIQgFAmqJqrAWHGxpbmtpbmpl
b25Aa2VybmVsLm9yZwAKCRBnC/sDUUQhCLIXEACnqHgNmrvIM6QtesU6iSVd6eGT
+7q+6ahkVv6dFks9KMc2hbzIs9VWmDpD3omU2OuqbJYiL3vuhv4+CwOnie6/5umc
3JCAG5jVOq5ee+7Tz6xZoxz2JxoxXm6LLmHXU0WALN4k8IsE7veGXAGkMgfnAQ44
lFDTByhvyf7vmXGHsz2FewOjdT6pdl4SUVAZQkvcZB9ujCqCWzx53s1+XZ2RCsT3
trepvTXQkxk2UyAKGs30JUiHDYwv01sYJCV84YKFBzTBCQ9AIIxKLCWlhYfERmLi
iL2nTBoNQPdDKiQK/R+axjkUABIQD3L4EhhqdLqcDBOFGsRlxUfd8pjyp+O5/yRw
Ev3ydGBDOULyU6Fmtx7mpzWm3EHklssMoGGuI9b81WcQc0PLp613egglgr1gKNZs
PaB74ea3ehSfHUIn2hF6Uli+OfZhXuHiOE/YKp6QTuer3rwUl0nJgIPDpkfWDuBH
hBGnTyiafMS0TVnd2yS0z8R/JipVpUniAPKLL5kRO3uIWcUwrRGjU8vfjsY5BSvU
W8BrYBu69zRQ603m/ARJ2gFU3jPmD7a/Su/m2GKaPjB/90RDbjRjcJ+r78pDfa1b
FgmL0cgyZg6uYFB0K/PCbSDsXA+x7xpjmOwZGgZNzrsFDQ+C9p5fizuOVG6N9MOg
zxOmKrpSSLYx0ahHgQ==
=kunx
-----END PGP SIGNATURE-----
Merge tag 'exfat-for-7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat
Pull exfat updates from Namjae Jeon:
"A set of exfat fixes covering shared writable mappings, entry error
handling, allocation cleanup, FITRIM bounds, and locking:
- Fix valid_size extension over shared writable mappings by lazily
zeroing page-cache gaps and ensuring racing stores cannot be
overwritten or extend beyond valid_size
- Clean up partially written directory entries on add-entry failure
and safely free new directory clusters
- Free a newly allocated directory cluster when zeroing it fails
- Write the destination entry before removing the source entry during
rename, preserving the source if the destination write fails
- Keep FITRIM within the requested range, even when the free-space
search wraps around the allocation bitmap
- Fix truncated FS_IOC_GETFSLABEL results by passing the destination
buffer size in bytes to the UTF-16-to-NLS conversion
- Fix overflow in the cluster-to-dentry calculation, which could
cause readdir to stop early on large volumes
- Replace the per-inode truncate_lock with inode_lock, using shared
locking in bmap to serialize against concurrent truncation
- Update the exfat mailing list address in MAINTAINERS"
* tag 'exfat-for-7.3-rc1' of git://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat:
exfat: replace truncate_lock with inode_lock
exfat: keep FITRIM within the requested range
exfat: fix truncated volume labels returned by FS_IOC_GETFSLABEL
exfat: fix overflow in cluster-to-dentry conversion
exfat: clean up new entry on add entry failure
exfat: free new directory cluster if zeroing fails
exfat: write moved entry before removing source
MAINTAINERS: update mailing list address for exfat
exfat: fix valid_size extension over a shared writable mapping
280 lines
7.9 KiB
C
280 lines
7.9 KiB
C
// SPDX-License-Identifier: GPL-2.0-or-later
|
|
/*
|
|
* iomap callack functions
|
|
*
|
|
* Copyright (C) 2026 Namjae Jeon <linkinjeon@kernel.org>
|
|
*/
|
|
|
|
#include <linux/iomap.h>
|
|
#include <linux/pagemap.h>
|
|
|
|
#include "exfat_raw.h"
|
|
#include "exfat_fs.h"
|
|
#include "iomap.h"
|
|
|
|
/*
|
|
* exfat_file_write_dio_end_io - Direct I/O write completion handler
|
|
*
|
|
* Updates i_size if the write extended the file. Called from the dio layer
|
|
* after I/O completion.
|
|
*/
|
|
static int exfat_file_write_dio_end_io(struct kiocb *iocb, ssize_t size,
|
|
int error, unsigned int flags)
|
|
{
|
|
struct inode *inode = file_inode(iocb->ki_filp);
|
|
|
|
if (error)
|
|
return error;
|
|
|
|
if (size && i_size_read(inode) < iocb->ki_pos + size) {
|
|
i_size_write(inode, iocb->ki_pos + size);
|
|
mark_inode_dirty(inode);
|
|
}
|
|
|
|
return 0;
|
|
}
|
|
|
|
const struct iomap_dio_ops exfat_write_dio_ops = {
|
|
.end_io = exfat_file_write_dio_end_io,
|
|
};
|
|
|
|
static int __exfat_iomap_begin(struct inode *inode, loff_t offset, loff_t length,
|
|
unsigned int flags, struct iomap *iomap, bool may_alloc)
|
|
{
|
|
struct super_block *sb = inode->i_sb;
|
|
struct exfat_sb_info *sbi = EXFAT_SB(sb);
|
|
struct exfat_inode_info *ei = EXFAT_I(inode);
|
|
unsigned int cluster, num_clusters;
|
|
loff_t cluster_offset, cluster_length;
|
|
int err;
|
|
bool balloc = false;
|
|
|
|
if (!may_alloc) {
|
|
/* Completely beyond EOF. Treat as hole */
|
|
if (i_size_read(inode) <= offset) {
|
|
iomap->type = IOMAP_HOLE;
|
|
iomap->addr = IOMAP_NULL_ADDR;
|
|
iomap->offset = offset;
|
|
iomap->length = length;
|
|
return 0;
|
|
}
|
|
|
|
/* Clamp length if the requested range goes beyond i_size */
|
|
if (offset + length > i_size_read(inode))
|
|
length = round_up(i_size_read(inode),
|
|
i_blocksize(inode)) - offset;
|
|
}
|
|
|
|
num_clusters = exfat_bytes_to_cluster_round_up(sbi,
|
|
offset + length) - exfat_bytes_to_cluster(sbi, offset);
|
|
|
|
mutex_lock(&sbi->s_lock);
|
|
iomap->bdev = inode->i_sb->s_bdev;
|
|
iomap->offset = offset;
|
|
|
|
err = exfat_map_cluster(inode, exfat_bytes_to_cluster(sbi, offset),
|
|
&cluster, &num_clusters, may_alloc, &balloc);
|
|
if (err)
|
|
goto out;
|
|
|
|
cluster_offset = exfat_cluster_offset(sbi, offset);
|
|
cluster_length = exfat_cluster_to_bytes(sbi, num_clusters);
|
|
|
|
iomap->length = min_t(loff_t, length, cluster_length - cluster_offset);
|
|
iomap->addr = exfat_cluster_to_phys_bytes(sbi, cluster) + cluster_offset;
|
|
iomap->type = IOMAP_MAPPED;
|
|
iomap->flags = IOMAP_F_MERGED;
|
|
|
|
if (may_alloc || flags & IOMAP_ZERO) {
|
|
if (balloc)
|
|
iomap->flags |= IOMAP_F_NEW;
|
|
else if (iomap->offset + iomap->length >= ei->valid_size) {
|
|
/*
|
|
* This is a write that starts at or extends beyond
|
|
* the current valid_size. The region between the old
|
|
* valid_size and the end of this write needs to be
|
|
* zeroed in the page cache to prevent stale data
|
|
* exposure (see IOMAP_F_ZERO_TAIL handling in
|
|
* __iomap_write_begin()).
|
|
*/
|
|
iomap->flags |= IOMAP_F_ZERO_TAIL;
|
|
}
|
|
} else {
|
|
/*
|
|
* valid_size is tracked in byte granularity and
|
|
* marks the exact boundary between valid data and
|
|
* holes (or unwritten space).
|
|
*
|
|
* When IOMAP_REPORT is set (used by lseek(SEEK_HOLE)
|
|
* and SEEK_DATA), we return IOMAP_HOLE. This allows
|
|
* iomap_seek_hole_iter() to directly return the
|
|
* precise byte position.
|
|
*
|
|
* For normal I/O paths (without IOMAP_REPORT) we
|
|
* return IOMAP_UNWRITTEN so the write path can
|
|
* distinguish it from a real hole.
|
|
*/
|
|
if (offset >= ei->valid_size) {
|
|
iomap->type = flags & IOMAP_REPORT ?
|
|
IOMAP_HOLE : IOMAP_UNWRITTEN;
|
|
} else if (offset + iomap->length > ei->valid_size) {
|
|
if (flags & IOMAP_REPORT) {
|
|
/*
|
|
* For SEEK_HOLE/SEEK_DATA, clip the length
|
|
* to the exact byte boundary (valid_size).
|
|
* This ensures the caller gets the precise
|
|
* hole position in byte units.
|
|
*/
|
|
iomap->length = ei->valid_size - iomap->offset;
|
|
} else
|
|
iomap->length = round_up(ei->valid_size,
|
|
i_blocksize(inode)) -
|
|
iomap->offset;
|
|
}
|
|
}
|
|
|
|
iomap->flags |= IOMAP_F_MERGED;
|
|
out:
|
|
mutex_unlock(&sbi->s_lock);
|
|
return err;
|
|
}
|
|
|
|
static int exfat_iomap_begin(struct inode *inode, loff_t offset, loff_t length,
|
|
unsigned int flags, struct iomap *iomap, struct iomap *srcmap)
|
|
{
|
|
return __exfat_iomap_begin(inode, offset, length, flags, iomap, false);
|
|
}
|
|
|
|
static int exfat_write_iomap_begin(struct inode *inode, loff_t offset, loff_t length,
|
|
unsigned int flags, struct iomap *iomap, struct iomap *srcmap)
|
|
{
|
|
return __exfat_iomap_begin(inode, offset, length, flags, iomap, true);
|
|
}
|
|
|
|
static DEFINE_IOMAP_ITER_NEXT(exfat_iomap_next, exfat_iomap_begin);
|
|
|
|
const struct iomap_ops exfat_iomap_ops = {
|
|
.iomap_next = exfat_iomap_next,
|
|
};
|
|
|
|
/*
|
|
* exfat_write_iomap_end - Update the state after write
|
|
*
|
|
* Extends ->valid_size to cover the newly written range.
|
|
* Marks the inode dirty if metadata was changed.
|
|
*/
|
|
static int exfat_write_iomap_end(struct inode *inode, loff_t pos, loff_t length,
|
|
ssize_t written, unsigned int flags, struct iomap *iomap)
|
|
{
|
|
struct exfat_inode_info *ei = EXFAT_I(inode);
|
|
bool dirtied = false;
|
|
loff_t end;
|
|
|
|
if (!written)
|
|
return 0;
|
|
|
|
end = pos + written;
|
|
|
|
if (ei->valid_size < end) {
|
|
ei->valid_size = end;
|
|
dirtied = true;
|
|
}
|
|
|
|
/*
|
|
* IOMAP_F_ZERO_TAIL zeroes the remainder of the last block. Track that
|
|
* block as zeroed so later valid_size extensions do not zero it again.
|
|
*/
|
|
if (iomap->flags & IOMAP_F_ZERO_TAIL)
|
|
end = round_up(end, i_blocksize(inode));
|
|
if (ei->zeroed_size < end)
|
|
ei->zeroed_size = end;
|
|
|
|
if (dirtied || iomap->flags & IOMAP_F_SIZE_CHANGED)
|
|
mark_inode_dirty(inode);
|
|
|
|
return written;
|
|
}
|
|
|
|
static DEFINE_IOMAP_ITER_NEXT_END(exfat_write_iomap_next,
|
|
exfat_write_iomap_begin, exfat_write_iomap_end);
|
|
|
|
const struct iomap_ops exfat_write_iomap_ops = {
|
|
.iomap_next = exfat_write_iomap_next,
|
|
};
|
|
|
|
/*
|
|
* exfat_writeback_range - Map folio during writeback
|
|
*
|
|
* Called for each folio during writeback. If the folio falls outside the
|
|
* current iomap, remaps by calling read_iomap_begin.
|
|
*/
|
|
static ssize_t exfat_writeback_range(struct iomap_writepage_ctx *wpc,
|
|
struct folio *folio, u64 offset, unsigned int len, u64 end_pos)
|
|
{
|
|
if (offset < wpc->iomap.offset ||
|
|
offset >= wpc->iomap.offset + wpc->iomap.length) {
|
|
int error;
|
|
|
|
error = __exfat_iomap_begin(wpc->inode, offset, len,
|
|
0, &wpc->iomap, false);
|
|
if (error)
|
|
return error;
|
|
}
|
|
|
|
return iomap_add_to_ioend(wpc, folio, offset, end_pos, len);
|
|
}
|
|
|
|
const struct iomap_writeback_ops exfat_writeback_ops = {
|
|
.writeback_range = exfat_writeback_range,
|
|
.writeback_submit = iomap_ioend_writeback_submit,
|
|
};
|
|
|
|
/**
|
|
* exfat_iomap_read_end_io - iomap read bio completion handler for exFAT
|
|
* @bio: bio that has completed reading
|
|
*
|
|
* exfat_iomap_begin() rounds up MAPPED extents to the block boundary of
|
|
* valid_size. This ensures that any subsequent blocks are treated as
|
|
* IOMAP_UNWRITTEN, but it also causes the "straddle block" containing
|
|
* valid_size to be read from disk. The disk data beyond valid_size in
|
|
* this block is stale and must be zeroed to prevent data leakage.
|
|
*/
|
|
static void exfat_iomap_read_end_io(struct bio *bio)
|
|
{
|
|
int error = blk_status_to_errno(bio->bi_status);
|
|
struct folio_iter iter;
|
|
|
|
bio_for_each_folio_all(iter, bio) {
|
|
struct folio *folio = iter.folio;
|
|
struct exfat_inode_info *ei = EXFAT_I(folio->mapping->host);
|
|
s64 valid_size;
|
|
loff_t pos = folio_pos(folio);
|
|
|
|
valid_size = ei->valid_size;
|
|
if (pos + iter.offset < valid_size &&
|
|
pos + iter.offset + iter.length > valid_size)
|
|
folio_zero_segment(folio, offset_in_folio(folio, valid_size),
|
|
iter.offset + iter.length);
|
|
|
|
iomap_finish_folio_read(folio, iter.offset, iter.length, error);
|
|
}
|
|
bio_put(bio);
|
|
}
|
|
|
|
static void exfat_iomap_bio_submit_read(const struct iomap_iter *iter,
|
|
struct iomap_read_folio_ctx *ctx)
|
|
{
|
|
iomap_bio_submit_read_endio(iter, ctx, exfat_iomap_read_end_io);
|
|
}
|
|
|
|
const struct iomap_read_ops exfat_iomap_bio_read_ops = {
|
|
.read_folio_range = iomap_bio_read_folio_range,
|
|
.submit_read = exfat_iomap_bio_submit_read,
|
|
};
|
|
|
|
int exfat_iomap_swap_activate(struct swap_info_struct *sis,
|
|
struct file *file, sector_t *span)
|
|
{
|
|
return iomap_swapfile_activate(sis, file, span, &exfat_iomap_ops);
|
|
}
|