[PATCH -next v5 23/32] ext4: submit and wait for pending disksize-grow I/O on writeback

Zhang Yi <[email protected]>
Newsgroups org.kernel.vger.linux-fsdevel,org.kernel.vger.linux-ext4,org.kernel.vger.linux-kernel
Message-ID <[email protected]>
From: Zhang Yi <[email protected]>

When the current writeback pass begins beyond the disksize-grow-pending
zeroed EOF block, the ioend worker would otherwise have to wait for the
pending EOF block to complete before it can advance i_disksize.
Otherwise the old EOF block could be exposed as stale data once
i_disksize advances past it.

Therefore, introduce the ioend mechanism for the pending range, tag
ioends that cover the pending zeroed EOF block which straddles
i_disksize with EXT4_IOMAP_IOEND_DISKSIZE_GROW_IO in
ext4_iomap_writeback_submit(), and clear the bit and wake up all waiters
in ext4_iomap_end_bio() when such an ioend completes.

In order to avoid the ioend that passes the pending range waiting for a
long time, proactively submit the pending range first in
ext4_iomap_writepages() so it completes in parallel with the rest of the
writeback.

Signed-off-by: Zhang Yi <[email protected]>
---
 fs/ext4/ext4.h    |  6 ++++++
 fs/ext4/inode.c   | 49 ++++++++++++++++++++++++++++++++++++++++++++++-
 fs/ext4/page-io.c | 40 ++++++++++++++++++++++++++++++++++++++
 3 files changed, 94 insertions(+), 1 deletion(-)

diff --git a/fs/ext4/ext4.h b/fs/ext4/ext4.h
index 1c3d736fb700..089dbd39c5c2 100644
--- a/fs/ext4/ext4.h
+++ b/fs/ext4/ext4.h
@@ -3986,6 +3986,12 @@ extern int ext4_move_extents(struct file *o_filp, struct file *d_filp,
 			     __u64 len, __u64 *moved_len);
 
 /* page-io.c */
+/*
+ * The I/O range covers the zeroed EOF block that straddles i_disksize
+ * and will advance it upon completion.
+ */
+#define EXT4_IOMAP_IOEND_DISKSIZE_GROW_IO	1UL
+
 extern int __init ext4_init_pageio(void);
 extern void ext4_exit_pageio(void);
 extern ext4_io_end_t *ext4_init_io_end(struct inode *inode, gfp_t flags);
diff --git a/fs/ext4/inode.c b/fs/ext4/inode.c
index e4a4396eaf87..a0707310b464 100644
--- a/fs/ext4/inode.c
+++ b/fs/ext4/inode.c
@@ -4366,7 +4366,10 @@ static int ext4_iomap_writeback_submit(struct iomap_writepage_ctx *wpc,
 				       int error)
 {
 	struct iomap_ioend *ioend = wpc->wb_ctx;
-	struct ext4_inode_info *ei = EXT4_I(ioend->io_inode);
+	struct inode *inode = ioend->io_inode;
+	struct ext4_inode_info *ei = EXT4_I(inode);
+	unsigned int blocksize = i_blocksize(inode);
+	loff_t pstart, plen;
 
 	/*
 	 * After I/O completion, a worker needs to be scheduled when:
@@ -4379,6 +4382,21 @@ static int ext4_iomap_writeback_submit(struct iomap_writepage_ctx *wpc,
 	    test_opt(ioend->io_inode->i_sb, DATA_ERR_ABORT))
 		ioend->io_bio.bi_end_io = ext4_iomap_end_bio;
 
+	/*
+	 * Mark the I/O as DISKSIZE_GROW_IO by setting io_private to
+	 * EXT4_IOMAP_IOEND_DISKSIZE_GROW_IO if it covers the pending range.
+	 * Such I/O will allow or trigger i_disksize advancement in the
+	 * ioend worker.
+	 */
+	plen = ext4_iomap_get_disksize_pending_range(inode, &pstart);
+	if (plen &&
+	    round_down(ioend->io_offset, blocksize) <= pstart &&
+	    round_up(ioend->io_offset + ioend->io_size, blocksize) >=
+			pstart + plen) {
+		ioend->io_bio.bi_end_io = ext4_iomap_end_bio;
+		ioend->io_private = (void *)EXT4_IOMAP_IOEND_DISKSIZE_GROW_IO;
+	}
+
 	/*
 	 * ext4_iomap_end_bio() always defers endio processing, disable
 	 * generic BIO in task to avoid double deferral since we will use
@@ -4398,6 +4416,29 @@ static const struct iomap_writeback_ops ext4_writeback_ops = {
 	.writeback_submit = ext4_iomap_writeback_submit,
 };
 
+/*
+ * If the current writeback range begins after the pending zeroed EOF
+ * block range which straddles i_disksize, issue a separate writeback to
+ * flush it first, so as to avoid prolonged waiting.
+ */
+static void ext4_iomap_wb_submit_zeroed_eof(struct inode *inode,
+					    struct writeback_control *wbc)
+{
+	struct address_space *mapping = inode->i_mapping;
+	loff_t pstart, plen, range_start;
+
+	if (wbc->range_cyclic)
+		range_start = (loff_t)mapping->writeback_index << PAGE_SHIFT;
+	else
+		range_start = wbc->range_start;
+
+	plen = ext4_iomap_get_disksize_pending_range(inode, &pstart);
+	if (!plen || range_start < pstart + plen)
+		return;
+
+	filemap_fdatawrite_range(mapping, pstart, pstart + plen - 1);
+}
+
 static int ext4_iomap_writepages(struct address_space *mapping,
 				 struct writeback_control *wbc)
 {
@@ -4415,6 +4456,12 @@ static int ext4_iomap_writepages(struct address_space *mapping,
 	if (unlikely(ret))
 		return ret;
 
+	/*
+	 * Submit the pending zeroed EOF block range if the entire
+	 * writeback range lies beyond it.
+	 */
+	ext4_iomap_wb_submit_zeroed_eof(inode, wbc);
+
 	alloc_ctx = ext4_writepages_down_read(sb);
 	trace_ext4_writepages(inode, wbc);
 	ret = iomap_writepages(&wpc);
diff --git a/fs/ext4/page-io.c b/fs/ext4/page-io.c
index 2888e0057561..955ff88045db 100644
--- a/fs/ext4/page-io.c
+++ b/fs/ext4/page-io.c
@@ -549,6 +549,34 @@ void ext4_bio_write_folio(struct ext4_io_submit *io, struct folio *folio,
 	} while ((bh = bh->b_this_page) != head);
 }
 
+/*
+ * If the current writeback range starts beyond the zeroed EOF pending
+ * range that straddles i_disksize, wait for the zeroed data from
+ * ext4_block_zero_eof() to be written out first. Otherwise, extending
+ * i_disksize may expose stale data in the old EOF block.
+ */
+static void ext4_iomap_wb_disksize_pending_wait(struct inode *inode,
+						loff_t pos, size_t size)
+{
+	loff_t disksize = READ_ONCE(EXT4_I(inode)->i_disksize);
+	loff_t pstart, plen;
+
+	/*
+	 * Overwrite I/Os and I/Os covering the EOF block do not need to
+	 * wait: the former do not advance i_disksize past the pending
+	 * boundary, and the latter are the pending I/O itself (cleared in
+	 * the bio completion path).
+	 */
+	if (pos < round_up(disksize, i_blocksize(inode)))
+		return;
+
+	plen = ext4_iomap_get_disksize_pending_range(inode, &pstart);
+	if (!plen || pos < pstart + plen)
+		return;
+
+	ext4_iomap_wait_disksize_pending(inode);
+}
+
 static int ext4_iomap_wb_update_disksize(handle_t *handle, struct inode *inode,
 					 loff_t end)
 {
@@ -594,6 +622,9 @@ static void ext4_iomap_finish_ioend(struct iomap_ioend *ioend)
 	    end <= READ_ONCE(EXT4_I(inode)->i_disksize))
 		goto out;
 
+	/* Wait for disksize-pending zeroed data to be written out. */
+	ext4_iomap_wb_disksize_pending_wait(inode, pos, size);
+
 	/*
 	 * We may need to convert one extent, update the i_disksize and
 	 * dirty the inode.
@@ -660,8 +691,17 @@ void ext4_iomap_end_bio(struct bio *bio)
 {
 	struct iomap_ioend *ioend = iomap_ioend_from_bio(bio);
 	struct ext4_inode_info *ei = EXT4_I(ioend->io_inode);
+	unsigned long io_mode = (unsigned long)ioend->io_private;
 	unsigned long flags;
 
+	/*
+	 * This is a disksize-pending I/O: clear the disksize-pending
+	 * state set in ext4_block_zero_eof() and wake up all waiters
+	 * that will update the inode i_disksize.
+	 */
+	if (io_mode == EXT4_IOMAP_IOEND_DISKSIZE_GROW_IO)
+		ext4_iomap_clear_disksize_pending(ioend->io_inode);
+
 	spin_lock_irqsave(&ei->i_completed_io_lock, flags);
 	if (list_empty(&ei->i_rsv_conversion_list))
 		queue_work(EXT4_SB(ioend->io_inode->i_sb)->rsv_conversion_wq,
-- 
2.52.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.