[PATCH] ext4: do not advance i_size on a buffered write that copied nothing

From: Hengyu Liang

Date: Fri Oct 09 2026 - 00:48:41 EST


Commit 665575cff098 ("filemap: move prefaulting out of hot write path")
made generic_perform_write() call ->write_begin() and ->write_end()
before it finds out that the source buffer cannot be read. In that
case ->write_end() is called with copied == 0.

However, ext4_write_end(), ext4_journalled_write_end() and
ext4_da_do_write_end() set i_size to pos + copied without checking
copied. As of now, a buffered write() past EOF that fails with EFAULT
grows the file to the write offset, and the new size can reach the
disk.

The issue can be reproduced on an ext4 mount with:

char *p = mmap(NULL, 4096, PROT_NONE,
MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
int fd = open("/mnt/ext4/file", O_CREAT | O_TRUNC | O_RDWR, 0600);
pwrite(fd, p, 4096, 1024 * 1024); /* fails with EFAULT */
fstat(fd, &st);

Before that commit, the result is st_size = 0.
After it, the result is st_size = 1048576.

This patch will leave i_size alone when nothing was copied, as
ext4_write_inline_data_end() already does. The delalloc path does not
trim after a short write, so in this case it also has to drop the
folio and the delayed blocks that ->write_begin() set up past i_size.
Otherwise that folio stays dirty past EOF and writeback never cleans
it.

Fixes: 665575cff098 ("filemap: move prefaulting out of hot write path")
Cc: stable@xxxxxxxxxxxxxxx
Signed-off-by: Hengyu Liang <hengyul@xxxxxxxxxx>
---
fs/ext4/inode.c | 42 +++++++++++++++++++++++++++++++++++-------
1 file changed, 35 insertions(+), 7 deletions(-)

diff --git a/fs/ext4/inode.c b/fs/ext4/inode.c
index 26f0f9714f03..36e75c89b8c8 100644
--- a/fs/ext4/inode.c
+++ b/fs/ext4/inode.c
@@ -1452,15 +1452,18 @@ static int ext4_write_end(const struct kiocb *iocb,
* it's important to update i_size while still holding folio lock:
* page writeout could otherwise come in and zero beyond i_size.
*
+ * If nothing was copied (e.g. the source buffer was unreadable), do
+ * not move i_size up to pos.
+ *
* If FS_IOC_ENABLE_VERITY is running on this inode, then Merkle tree
* blocks are being written past EOF, so skip the i_size update.
*/
- if (!verity)
+ if (copied && !verity)
i_size_changed = ext4_update_inode_size(inode, pos + copied);
folio_unlock(folio);
folio_put(folio);

- if (old_size < pos && !verity)
+ if (copied && old_size < pos && !verity)
pagecache_isize_extended(inode, old_size, pos);

/*
@@ -1571,13 +1574,13 @@ static int ext4_journalled_write_end(const struct kiocb *iocb,
if (!partial)
folio_mark_uptodate(folio);
}
- if (!verity)
+ if (copied && !verity)
size_changed = ext4_update_inode_size(inode, pos + copied);
EXT4_I(inode)->i_datasync_tid = handle->h_transaction->t_tid;
folio_unlock(folio);
folio_put(folio);

- if (old_size < pos && !verity)
+ if (copied && old_size < pos && !verity)
pagecache_isize_extended(inode, old_size, pos);

if (size_changed) {
@@ -3210,6 +3213,24 @@ static int ext4_da_should_update_i_disksize(struct folio *folio,
return 1;
}

+/*
+ * Delalloc counterpart of ext4_truncate_failed_write(): ->write_begin() only
+ * reserved delayed blocks, so drop the page cache and the delayed extents
+ * beyond i_size and leave the blocks that are allocated there alone.
+ */
+static void ext4_da_truncate_failed_write(struct inode *inode)
+{
+ struct address_space *mapping = inode->i_mapping;
+ ext4_lblk_t lblk = EXT4_B_TO_LBLK(inode, inode->i_size);
+
+ filemap_invalidate_lock(mapping);
+ truncate_inode_pages(mapping, inode->i_size);
+ down_write(&EXT4_I(inode)->i_data_sem);
+ ext4_es_remove_extent(inode, lblk, EXT_MAX_BLOCKS - lblk);
+ up_write(&EXT4_I(inode)->i_data_sem);
+ filemap_invalidate_unlock(mapping);
+}
+
static int ext4_da_do_write_end(struct address_space *mapping,
loff_t pos, unsigned len, unsigned copied,
struct folio *folio)
@@ -3247,12 +3268,12 @@ static int ext4_da_do_write_end(struct address_space *mapping,
* checked, we need to update i_disksize here as certain
* ext4_writepages() paths not allocating blocks and update i_disksize.
*/
- if (new_i_size > inode->i_size) {
+ if (copied && new_i_size > inode->i_size) {
unsigned long end;

i_size_write(inode, new_i_size);
end = offset_in_folio(folio, new_i_size - 1);
- if (copied && ext4_da_should_update_i_disksize(folio, end)) {
+ if (ext4_da_should_update_i_disksize(folio, end)) {
ext4_update_i_disksize(inode, new_i_size);
disksize_changed = true;
}
@@ -3261,9 +3282,16 @@ static int ext4_da_do_write_end(struct address_space *mapping,
folio_unlock(folio);
folio_put(folio);

- if (pos > old_size)
+ if (copied && pos > old_size)
pagecache_isize_extended(inode, old_size, pos);

+ /*
+ * A write past i_size that copied nothing leaves i_size alone, so
+ * trim what ->write_begin() set up beyond it.
+ */
+ if (!copied && pos > old_size)
+ ext4_da_truncate_failed_write(inode);
+
if (!disksize_changed)
return copied;

--
2.53.0