summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--Documentation/filesystems/netfs_library.rst26
-rw-r--r--fs/netfs/buffered_write.c9
-rw-r--r--fs/netfs/direct_write.c9
-rw-r--r--fs/netfs/internal.h2
-rw-r--r--fs/netfs/misc.c99
-rw-r--r--include/linux/netfs.h2
6 files changed, 147 insertions, 0 deletions
diff --git a/Documentation/filesystems/netfs_library.rst b/Documentation/filesystems/netfs_library.rst
index ddd799df6ce3..a9de281db8ee 100644
--- a/Documentation/filesystems/netfs_library.rst
+++ b/Documentation/filesystems/netfs_library.rst
@@ -451,6 +451,32 @@ one.
The inode should be marked ``NETFS_ICTX_SINGLE_NO_UPLOAD`` if this API is to be
used. The writeback function requires the buffer to be of ITER_FOLIOQ type.
+Clearing Stale Post-EOF Pagecache
+---------------------------------
+
+When a file is extended, data left in the pagecache past the old EOF by a write
+through an mmap must not be exposed as file content. Netfslib clears this on
+its own write paths, and exports a helper so a filesystem can do the same from
+a resize path (truncate, setattr, fallocate and the like) that holds the
+inode's ``i_rwsem`` exclusively across the whole resize::
+
+ void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from,
+ uoff_t to);
+
+This zeroes any such data within the ``[from, to)`` hole to be made, where
+@from is the old EOF and @to is the new one, and the caller must have already
+updated ``i_size`` to @to before calling it. Only the folio straddling @from
+can hold data written past the EOF through an mmap, as pages wholly beyond the
+EOF can't be faulted in, so the zeroing is limited to that folio. The folio is
+zeroed rather than dropped so that a concurrent extending write can't lose
+data.
+
+This overlaps with ``pagecache_isize_extended()`` but can't reuse it: that
+helper is keyed on a sub-page block size and is a no-op when the block size is
+``>= PAGE_SIZE`` (as on network filesystems), and it doesn't wait for
+writeback. As both address the same problem, a change to one should probably
+be reflected in the other to keep them in sync.
+
High-Level VM API
==================
diff --git a/fs/netfs/buffered_write.c b/fs/netfs/buffered_write.c
index 2cdb68e6b16f..ecf119b4f916 100644
--- a/fs/netfs/buffered_write.c
+++ b/fs/netfs/buffered_write.c
@@ -469,6 +469,7 @@ ssize_t netfs_buffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *fr
struct netfs_group *netfs_group)
{
struct file *file = iocb->ki_filp;
+ struct inode *inode = file_inode(file);
ssize_t ret;
trace_netfs_write_iter(iocb, from);
@@ -481,6 +482,14 @@ ssize_t netfs_buffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *fr
if (ret)
return ret;
+ if (iocb->ki_pos > i_size_read(inode)) {
+ ret = netfs_clear_stale_pre_isize(inode, i_size_read(inode),
+ iocb->ki_pos,
+ iocb->ki_flags & IOCB_NOWAIT);
+ if (ret)
+ return ret;
+ }
+
return netfs_perform_write(iocb, from, netfs_group);
}
EXPORT_SYMBOL(netfs_buffered_write_iter_locked);
diff --git a/fs/netfs/direct_write.c b/fs/netfs/direct_write.c
index 2361277416c7..8be74706d85e 100644
--- a/fs/netfs/direct_write.c
+++ b/fs/netfs/direct_write.c
@@ -359,6 +359,15 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from)
ret = file_update_time(file);
if (ret < 0)
goto out;
+
+ if (iocb->ki_pos > i_size_read(inode)) {
+ ret = netfs_clear_stale_pre_isize(inode, i_size_read(inode),
+ iocb->ki_pos,
+ iocb->ki_flags & IOCB_NOWAIT);
+ if (ret < 0)
+ goto out;
+ }
+
if (iocb->ki_flags & IOCB_NOWAIT) {
/* We could block if there are any pages in the range. */
ret = -EAGAIN;
diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h
index c79c8e69d60c..786af76da98c 100644
--- a/fs/netfs/internal.h
+++ b/fs/netfs/internal.h
@@ -80,6 +80,8 @@ ssize_t netfs_wait_for_write(struct netfs_io_request *rreq);
void netfs_wait_for_paused_read(struct netfs_io_request *rreq);
void netfs_wait_for_paused_write(struct netfs_io_request *rreq);
void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq);
+int netfs_clear_stale_pre_isize(struct inode *inode, uoff_t from,
+ uoff_t to, bool nowait);
/*
* objects.c
diff --git a/fs/netfs/misc.c b/fs/netfs/misc.c
index f5c1c463f4ff..523057390d6f 100644
--- a/fs/netfs/misc.c
+++ b/fs/netfs/misc.c
@@ -6,6 +6,7 @@
*/
#include <linux/swap.h>
+#include <linux/rmap.h>
#include "internal.h"
/**
@@ -582,3 +583,101 @@ void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq)
trace_netfs_rreq(rreq, netfs_rreq_trace_waited_put_ra_refs);
finish_wait(&rreq->waitq, &myself);
}
+
+/**
+ * netfs_clear_stale_isize - Clear stale pagecache in a to-be-created hole
+ * @inode: The inode to act upon.
+ * @from: The base of the hole to be made.
+ * @to: The top of the hole to be made.
+ * @nowait: True to return -EAGAIN rather than block.
+ * @exclusive: True if the caller holds i_rwsem exclusively for the resize.
+ *
+ * Zero any data left in the pagecache within the [@from, @to) hole by a
+ * write through an mmap so that it isn't exposed as file content once the
+ * file is extended. Only the uptodate folio straddling @from can hold such
+ * data as pages wholly beyond the EOF can't be faulted in, so the zeroing
+ * is limited to that folio. The folio is zeroed rather than dropped so
+ * that a concurrent extending write can't lose data.
+ *
+ * If @exclusive is false, @from is re-read from i_size and used to clamp
+ * the zeroed range, for callers that may race with another writer also
+ * extending the file (eg. multiple buffered writes extending the same file
+ * under a shared i_rwsem). If @exclusive is true, for callers that hold
+ * i_rwsem exclusively across the whole resize and have already updated
+ * i_size to @to, staleness is decided from the folio's dirty state instead:
+ * since no genuine concurrent buffered writer can be racing, a lockless
+ * stat() adopting a server-confirmed size mid-resize has no data behind it
+ * and never dirties the folio, so it can't fool this check into skipping
+ * the zeroing the way it could fool the @exclusive false clamp.
+ *
+ * pagecache_isize_extended() can't be reused here: it is keyed on a
+ * sub-page block size (a no-op when the block size is >= PAGE_SIZE, as on
+ * network filesystems), runs after i_size is updated, can't honour
+ * @nowait, and doesn't wait for writeback. Keep the two in sync if either
+ * is changed.
+ *
+ * Return: 0 on success, or -EAGAIN if @nowait is set and the folio is
+ * mapped or under writeback and so can't be cleaned without blocking.
+ */
+static int netfs_clear_stale_isize(struct inode *inode, uoff_t from,
+ uoff_t to, bool nowait, bool exclusive)
+{
+ struct address_space *mapping = inode->i_mapping;
+ fgf_t fgp = FGP_LOCK;
+ struct folio *folio;
+ int ret;
+
+ if (from >= to)
+ return 0;
+
+ if (nowait)
+ fgp |= FGP_NOWAIT;
+
+ folio = __filemap_get_folio(mapping, from >> PAGE_SHIFT, fgp, 0);
+ if (IS_ERR(folio))
+ return PTR_ERR(folio) == -EAGAIN ? -EAGAIN : 0;
+
+ ret = 0;
+ if (nowait && (folio_mapped(folio) || folio_test_writeback(folio))) {
+ ret = -EAGAIN;
+ goto out;
+ }
+
+ folio_wait_writeback(folio);
+
+ if (folio_mkclean(folio))
+ folio_mark_dirty(folio);
+
+ if (folio_test_uptodate(folio) &&
+ (!exclusive || folio_test_dirty(folio))) {
+ uoff_t fpos = folio_pos(folio);
+
+ if (!exclusive)
+ from = umax(from, i_size_read(inode));
+ if (from < to && from < fpos + folio_size(folio)) {
+ size_t end = umin(to - fpos, folio_size(folio));
+ size_t offset = from - fpos;
+
+ folio_zero_segment(folio, offset, end);
+ }
+ }
+out:
+ folio_unlock(folio);
+ folio_put(folio);
+ return ret;
+}
+
+/* Clear stale pagecache before an extending buffered/DIO write. */
+int netfs_clear_stale_pre_isize(struct inode *inode, uoff_t from,
+ uoff_t to, bool nowait)
+{
+ return netfs_clear_stale_isize(inode, from, to, nowait, false);
+}
+
+/* Clear stale pagecache when extending a file under an exclusive resize. */
+void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from,
+ uoff_t to)
+{
+ netfs_clear_stale_isize(inode, from, to, false, true);
+}
+EXPORT_SYMBOL(netfs_clear_stale_post_isize);
diff --git a/include/linux/netfs.h b/include/linux/netfs.h
index b4dd32863dd4..fe6275e55fc2 100644
--- a/include/linux/netfs.h
+++ b/include/linux/netfs.h
@@ -397,6 +397,8 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from);
ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *iter,
struct netfs_group *netfs_group);
ssize_t netfs_file_write_iter(struct kiocb *iocb, struct iov_iter *from);
+void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from,
+ uoff_t to);
/* Single, monolithic object read/write API. */
void netfs_single_mark_inode_dirty(struct inode *inode);