diff options
| -rw-r--r-- | Documentation/filesystems/netfs_library.rst | 26 | ||||
| -rw-r--r-- | fs/netfs/buffered_write.c | 9 | ||||
| -rw-r--r-- | fs/netfs/direct_write.c | 9 | ||||
| -rw-r--r-- | fs/netfs/internal.h | 2 | ||||
| -rw-r--r-- | fs/netfs/misc.c | 99 | ||||
| -rw-r--r-- | include/linux/netfs.h | 2 |
6 files changed, 147 insertions, 0 deletions
diff --git a/Documentation/filesystems/netfs_library.rst b/Documentation/filesystems/netfs_library.rst index ddd799df6ce3..a9de281db8ee 100644 --- a/Documentation/filesystems/netfs_library.rst +++ b/Documentation/filesystems/netfs_library.rst @@ -451,6 +451,32 @@ one. The inode should be marked ``NETFS_ICTX_SINGLE_NO_UPLOAD`` if this API is to be used. The writeback function requires the buffer to be of ITER_FOLIOQ type. +Clearing Stale Post-EOF Pagecache +--------------------------------- + +When a file is extended, data left in the pagecache past the old EOF by a write +through an mmap must not be exposed as file content. Netfslib clears this on +its own write paths, and exports a helper so a filesystem can do the same from +a resize path (truncate, setattr, fallocate and the like) that holds the +inode's ``i_rwsem`` exclusively across the whole resize:: + + void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from, + uoff_t to); + +This zeroes any such data within the ``[from, to)`` hole to be made, where +@from is the old EOF and @to is the new one, and the caller must have already +updated ``i_size`` to @to before calling it. Only the folio straddling @from +can hold data written past the EOF through an mmap, as pages wholly beyond the +EOF can't be faulted in, so the zeroing is limited to that folio. The folio is +zeroed rather than dropped so that a concurrent extending write can't lose +data. + +This overlaps with ``pagecache_isize_extended()`` but can't reuse it: that +helper is keyed on a sub-page block size and is a no-op when the block size is +``>= PAGE_SIZE`` (as on network filesystems), and it doesn't wait for +writeback. As both address the same problem, a change to one should probably +be reflected in the other to keep them in sync. + High-Level VM API ================== diff --git a/fs/netfs/buffered_write.c b/fs/netfs/buffered_write.c index 2cdb68e6b16f..ecf119b4f916 100644 --- a/fs/netfs/buffered_write.c +++ b/fs/netfs/buffered_write.c @@ -469,6 +469,7 @@ ssize_t netfs_buffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *fr struct netfs_group *netfs_group) { struct file *file = iocb->ki_filp; + struct inode *inode = file_inode(file); ssize_t ret; trace_netfs_write_iter(iocb, from); @@ -481,6 +482,14 @@ ssize_t netfs_buffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *fr if (ret) return ret; + if (iocb->ki_pos > i_size_read(inode)) { + ret = netfs_clear_stale_pre_isize(inode, i_size_read(inode), + iocb->ki_pos, + iocb->ki_flags & IOCB_NOWAIT); + if (ret) + return ret; + } + return netfs_perform_write(iocb, from, netfs_group); } EXPORT_SYMBOL(netfs_buffered_write_iter_locked); diff --git a/fs/netfs/direct_write.c b/fs/netfs/direct_write.c index 2361277416c7..8be74706d85e 100644 --- a/fs/netfs/direct_write.c +++ b/fs/netfs/direct_write.c @@ -359,6 +359,15 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from) ret = file_update_time(file); if (ret < 0) goto out; + + if (iocb->ki_pos > i_size_read(inode)) { + ret = netfs_clear_stale_pre_isize(inode, i_size_read(inode), + iocb->ki_pos, + iocb->ki_flags & IOCB_NOWAIT); + if (ret < 0) + goto out; + } + if (iocb->ki_flags & IOCB_NOWAIT) { /* We could block if there are any pages in the range. */ ret = -EAGAIN; diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h index c79c8e69d60c..786af76da98c 100644 --- a/fs/netfs/internal.h +++ b/fs/netfs/internal.h @@ -80,6 +80,8 @@ ssize_t netfs_wait_for_write(struct netfs_io_request *rreq); void netfs_wait_for_paused_read(struct netfs_io_request *rreq); void netfs_wait_for_paused_write(struct netfs_io_request *rreq); void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq); +int netfs_clear_stale_pre_isize(struct inode *inode, uoff_t from, + uoff_t to, bool nowait); /* * objects.c diff --git a/fs/netfs/misc.c b/fs/netfs/misc.c index f5c1c463f4ff..523057390d6f 100644 --- a/fs/netfs/misc.c +++ b/fs/netfs/misc.c @@ -6,6 +6,7 @@ */ #include <linux/swap.h> +#include <linux/rmap.h> #include "internal.h" /** @@ -582,3 +583,101 @@ void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq) trace_netfs_rreq(rreq, netfs_rreq_trace_waited_put_ra_refs); finish_wait(&rreq->waitq, &myself); } + +/** + * netfs_clear_stale_isize - Clear stale pagecache in a to-be-created hole + * @inode: The inode to act upon. + * @from: The base of the hole to be made. + * @to: The top of the hole to be made. + * @nowait: True to return -EAGAIN rather than block. + * @exclusive: True if the caller holds i_rwsem exclusively for the resize. + * + * Zero any data left in the pagecache within the [@from, @to) hole by a + * write through an mmap so that it isn't exposed as file content once the + * file is extended. Only the uptodate folio straddling @from can hold such + * data as pages wholly beyond the EOF can't be faulted in, so the zeroing + * is limited to that folio. The folio is zeroed rather than dropped so + * that a concurrent extending write can't lose data. + * + * If @exclusive is false, @from is re-read from i_size and used to clamp + * the zeroed range, for callers that may race with another writer also + * extending the file (eg. multiple buffered writes extending the same file + * under a shared i_rwsem). If @exclusive is true, for callers that hold + * i_rwsem exclusively across the whole resize and have already updated + * i_size to @to, staleness is decided from the folio's dirty state instead: + * since no genuine concurrent buffered writer can be racing, a lockless + * stat() adopting a server-confirmed size mid-resize has no data behind it + * and never dirties the folio, so it can't fool this check into skipping + * the zeroing the way it could fool the @exclusive false clamp. + * + * pagecache_isize_extended() can't be reused here: it is keyed on a + * sub-page block size (a no-op when the block size is >= PAGE_SIZE, as on + * network filesystems), runs after i_size is updated, can't honour + * @nowait, and doesn't wait for writeback. Keep the two in sync if either + * is changed. + * + * Return: 0 on success, or -EAGAIN if @nowait is set and the folio is + * mapped or under writeback and so can't be cleaned without blocking. + */ +static int netfs_clear_stale_isize(struct inode *inode, uoff_t from, + uoff_t to, bool nowait, bool exclusive) +{ + struct address_space *mapping = inode->i_mapping; + fgf_t fgp = FGP_LOCK; + struct folio *folio; + int ret; + + if (from >= to) + return 0; + + if (nowait) + fgp |= FGP_NOWAIT; + + folio = __filemap_get_folio(mapping, from >> PAGE_SHIFT, fgp, 0); + if (IS_ERR(folio)) + return PTR_ERR(folio) == -EAGAIN ? -EAGAIN : 0; + + ret = 0; + if (nowait && (folio_mapped(folio) || folio_test_writeback(folio))) { + ret = -EAGAIN; + goto out; + } + + folio_wait_writeback(folio); + + if (folio_mkclean(folio)) + folio_mark_dirty(folio); + + if (folio_test_uptodate(folio) && + (!exclusive || folio_test_dirty(folio))) { + uoff_t fpos = folio_pos(folio); + + if (!exclusive) + from = umax(from, i_size_read(inode)); + if (from < to && from < fpos + folio_size(folio)) { + size_t end = umin(to - fpos, folio_size(folio)); + size_t offset = from - fpos; + + folio_zero_segment(folio, offset, end); + } + } +out: + folio_unlock(folio); + folio_put(folio); + return ret; +} + +/* Clear stale pagecache before an extending buffered/DIO write. */ +int netfs_clear_stale_pre_isize(struct inode *inode, uoff_t from, + uoff_t to, bool nowait) +{ + return netfs_clear_stale_isize(inode, from, to, nowait, false); +} + +/* Clear stale pagecache when extending a file under an exclusive resize. */ +void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from, + uoff_t to) +{ + netfs_clear_stale_isize(inode, from, to, false, true); +} +EXPORT_SYMBOL(netfs_clear_stale_post_isize); diff --git a/include/linux/netfs.h b/include/linux/netfs.h index b4dd32863dd4..fe6275e55fc2 100644 --- a/include/linux/netfs.h +++ b/include/linux/netfs.h @@ -397,6 +397,8 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from); ssize_t netfs_unbuffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *iter, struct netfs_group *netfs_group); ssize_t netfs_file_write_iter(struct kiocb *iocb, struct iov_iter *from); +void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from, + uoff_t to); /* Single, monolithic object read/write API. */ void netfs_single_mark_inode_dirty(struct inode *inode); |
