summaryrefslogtreecommitdiffstats
path: root/fs
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 14:04:11 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 14:04:11 -0700
commitff47652a4b66c067c765a7ad464d930b5a9367cc (patch)
treeec2b19252312016e6f521c287f69ced542eb0c85 /fs
parent8f150ccedfbd610aa25509ff42365d70fb20478f (diff)
parent19465a9aeb664f1d710d0d22c07c31bfb7c85446 (diff)
downloadlinux-stable-ff47652a4b66c067c765a7ad464d930b5a9367cc.tar.gz
linux-stable-ff47652a4b66c067c765a7ad464d930b5a9367cc.zip
Merge tag 'cifs-fixes-7.3-rc6' of https://git.manguebit.org/linux
Pull smb client fixes from Paulo Alcantara: "Fix a series of data corruption and I/O error bugs found by running generic/363 (fsx) in a loop against Windows Server 2022 and Samba. - Stop data dirtied past EOF through an mmap from reappearing as file content once the file is extended by a write, truncate, zero range, copy range or clone range - Flush dirty data and drain in-flight I/O before operations that assume the pagecache and the server agree on the file: querying allocated ranges, the O_TRUNC open, interior zero range, and server-side copy/clone - Stop a genuine size-extending zero range or preallocate from being refused with -EOPNOTSUPP when the inode is not read caching, by querying the server's authoritative EOF instead of trusting a stale cached i_size - Zero the untransferred tail of a short read, both in the netfs read-gaps path (where stale folio content could otherwise be written back to the server) and in the DIO/unbuffered read collector, and tell a real EOF apart from a stale cached remote_i_size after a lease downgrade - Require stable pages on signed connections so a buffered write can't modify a folio whose signature has already been computed and is in flight, which the server rejected with STATUS_ACCESS_DENIED and the client surfaced as -EIO - Split several cifsFileInfo flags out of a shared bitfield byte so concurrent updates taken under different locks no longer clobber each other through a byte-level RMW" * tag 'cifs-fixes-7.3-rc6' of https://git.manguebit.org/linux: smb: client: split cifsFileInfo bitfields to avoid shared-byte RMW races smb: client: require stable pages for signed connections smb: client: distinguish real EOF from a stale remote_i_size on read netfs: zero the tail of a short DIO/unbuffered read smb: client: only require read lease for size-extending preallocate netfs: zero gaps in read-gaps folio to avoid writing back stale data smb: client: only require read lease for size-extending zero range smb: client: drain and invalidate before server-side copy/clone smb: client: flush dirty data before zeroing a range smb: client: drain outstanding I/O before truncating on O_TRUNC open smb: client: flush and commit data before querying allocated ranges smb: client: discard post-EOF pagecache when extending a file via clone range smb: client: discard post-EOF pagecache when extending a file via copy range smb: client: discard post-EOF pagecache when extending a file via zero range smb: client: clear post-EOF pagecache when extending a file via truncate netfs: clear post-EOF pagecache when extending a file via write
Diffstat (limited to 'fs')
-rw-r--r--fs/netfs/buffered_read.c7
-rw-r--r--fs/netfs/buffered_write.c9
-rw-r--r--fs/netfs/direct_write.c9
-rw-r--r--fs/netfs/internal.h2
-rw-r--r--fs/netfs/misc.c99
-rw-r--r--fs/netfs/read_collect.c24
-rw-r--r--fs/smb/client/cifsfs.c41
-rw-r--r--fs/smb/client/cifsfs.h5
-rw-r--r--fs/smb/client/cifsglob.h10
-rw-r--r--fs/smb/client/file.c5
-rw-r--r--fs/smb/client/inode.c24
-rw-r--r--fs/smb/client/smb2ops.c130
-rw-r--r--fs/smb/client/smb2pdu.c10
13 files changed, 319 insertions, 56 deletions
diff --git a/fs/netfs/buffered_read.c b/fs/netfs/buffered_read.c
index 105194de6..e6506942a 100644
--- a/fs/netfs/buffered_read.c
+++ b/fs/netfs/buffered_read.c
@@ -541,6 +541,13 @@ static int netfs_read_gaps(struct file *file, struct folio *folio)
ret = netfs_wait_for_read(rreq);
if (ret >= 0) {
+ if (ret < flen) {
+ struct iov_iter iter;
+
+ iov_iter_bvec(&iter, ITER_DEST, bvec, i, flen);
+ iov_iter_advance(&iter, ret);
+ iov_iter_zero(flen - ret, &iter);
+ }
if (group)
folio_change_private(folio, group);
else
diff --git a/fs/netfs/buffered_write.c b/fs/netfs/buffered_write.c
index 2cdb68e6b..ecf119b4f 100644
--- a/fs/netfs/buffered_write.c
+++ b/fs/netfs/buffered_write.c
@@ -469,6 +469,7 @@ ssize_t netfs_buffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *fr
struct netfs_group *netfs_group)
{
struct file *file = iocb->ki_filp;
+ struct inode *inode = file_inode(file);
ssize_t ret;
trace_netfs_write_iter(iocb, from);
@@ -481,6 +482,14 @@ ssize_t netfs_buffered_write_iter_locked(struct kiocb *iocb, struct iov_iter *fr
if (ret)
return ret;
+ if (iocb->ki_pos > i_size_read(inode)) {
+ ret = netfs_clear_stale_pre_isize(inode, i_size_read(inode),
+ iocb->ki_pos,
+ iocb->ki_flags & IOCB_NOWAIT);
+ if (ret)
+ return ret;
+ }
+
return netfs_perform_write(iocb, from, netfs_group);
}
EXPORT_SYMBOL(netfs_buffered_write_iter_locked);
diff --git a/fs/netfs/direct_write.c b/fs/netfs/direct_write.c
index 236127741..8be74706d 100644
--- a/fs/netfs/direct_write.c
+++ b/fs/netfs/direct_write.c
@@ -359,6 +359,15 @@ ssize_t netfs_unbuffered_write_iter(struct kiocb *iocb, struct iov_iter *from)
ret = file_update_time(file);
if (ret < 0)
goto out;
+
+ if (iocb->ki_pos > i_size_read(inode)) {
+ ret = netfs_clear_stale_pre_isize(inode, i_size_read(inode),
+ iocb->ki_pos,
+ iocb->ki_flags & IOCB_NOWAIT);
+ if (ret < 0)
+ goto out;
+ }
+
if (iocb->ki_flags & IOCB_NOWAIT) {
/* We could block if there are any pages in the range. */
ret = -EAGAIN;
diff --git a/fs/netfs/internal.h b/fs/netfs/internal.h
index c79c8e69d..786af76da 100644
--- a/fs/netfs/internal.h
+++ b/fs/netfs/internal.h
@@ -80,6 +80,8 @@ ssize_t netfs_wait_for_write(struct netfs_io_request *rreq);
void netfs_wait_for_paused_read(struct netfs_io_request *rreq);
void netfs_wait_for_paused_write(struct netfs_io_request *rreq);
void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq);
+int netfs_clear_stale_pre_isize(struct inode *inode, uoff_t from,
+ uoff_t to, bool nowait);
/*
* objects.c
diff --git a/fs/netfs/misc.c b/fs/netfs/misc.c
index f5c1c463f..523057390 100644
--- a/fs/netfs/misc.c
+++ b/fs/netfs/misc.c
@@ -6,6 +6,7 @@
*/
#include <linux/swap.h>
+#include <linux/rmap.h>
#include "internal.h"
/**
@@ -582,3 +583,101 @@ void netfs_wait_for_put_ra_refs(struct netfs_io_request *rreq)
trace_netfs_rreq(rreq, netfs_rreq_trace_waited_put_ra_refs);
finish_wait(&rreq->waitq, &myself);
}
+
+/**
+ * netfs_clear_stale_isize - Clear stale pagecache in a to-be-created hole
+ * @inode: The inode to act upon.
+ * @from: The base of the hole to be made.
+ * @to: The top of the hole to be made.
+ * @nowait: True to return -EAGAIN rather than block.
+ * @exclusive: True if the caller holds i_rwsem exclusively for the resize.
+ *
+ * Zero any data left in the pagecache within the [@from, @to) hole by a
+ * write through an mmap so that it isn't exposed as file content once the
+ * file is extended. Only the uptodate folio straddling @from can hold such
+ * data as pages wholly beyond the EOF can't be faulted in, so the zeroing
+ * is limited to that folio. The folio is zeroed rather than dropped so
+ * that a concurrent extending write can't lose data.
+ *
+ * If @exclusive is false, @from is re-read from i_size and used to clamp
+ * the zeroed range, for callers that may race with another writer also
+ * extending the file (eg. multiple buffered writes extending the same file
+ * under a shared i_rwsem). If @exclusive is true, for callers that hold
+ * i_rwsem exclusively across the whole resize and have already updated
+ * i_size to @to, staleness is decided from the folio's dirty state instead:
+ * since no genuine concurrent buffered writer can be racing, a lockless
+ * stat() adopting a server-confirmed size mid-resize has no data behind it
+ * and never dirties the folio, so it can't fool this check into skipping
+ * the zeroing the way it could fool the @exclusive false clamp.
+ *
+ * pagecache_isize_extended() can't be reused here: it is keyed on a
+ * sub-page block size (a no-op when the block size is >= PAGE_SIZE, as on
+ * network filesystems), runs after i_size is updated, can't honour
+ * @nowait, and doesn't wait for writeback. Keep the two in sync if either
+ * is changed.
+ *
+ * Return: 0 on success, or -EAGAIN if @nowait is set and the folio is
+ * mapped or under writeback and so can't be cleaned without blocking.
+ */
+static int netfs_clear_stale_isize(struct inode *inode, uoff_t from,
+ uoff_t to, bool nowait, bool exclusive)
+{
+ struct address_space *mapping = inode->i_mapping;
+ fgf_t fgp = FGP_LOCK;
+ struct folio *folio;
+ int ret;
+
+ if (from >= to)
+ return 0;
+
+ if (nowait)
+ fgp |= FGP_NOWAIT;
+
+ folio = __filemap_get_folio(mapping, from >> PAGE_SHIFT, fgp, 0);
+ if (IS_ERR(folio))
+ return PTR_ERR(folio) == -EAGAIN ? -EAGAIN : 0;
+
+ ret = 0;
+ if (nowait && (folio_mapped(folio) || folio_test_writeback(folio))) {
+ ret = -EAGAIN;
+ goto out;
+ }
+
+ folio_wait_writeback(folio);
+
+ if (folio_mkclean(folio))
+ folio_mark_dirty(folio);
+
+ if (folio_test_uptodate(folio) &&
+ (!exclusive || folio_test_dirty(folio))) {
+ uoff_t fpos = folio_pos(folio);
+
+ if (!exclusive)
+ from = umax(from, i_size_read(inode));
+ if (from < to && from < fpos + folio_size(folio)) {
+ size_t end = umin(to - fpos, folio_size(folio));
+ size_t offset = from - fpos;
+
+ folio_zero_segment(folio, offset, end);
+ }
+ }
+out:
+ folio_unlock(folio);
+ folio_put(folio);
+ return ret;
+}
+
+/* Clear stale pagecache before an extending buffered/DIO write. */
+int netfs_clear_stale_pre_isize(struct inode *inode, uoff_t from,
+ uoff_t to, bool nowait)
+{
+ return netfs_clear_stale_isize(inode, from, to, nowait, false);
+}
+
+/* Clear stale pagecache when extending a file under an exclusive resize. */
+void netfs_clear_stale_post_isize(struct inode *inode, uoff_t from,
+ uoff_t to)
+{
+ netfs_clear_stale_isize(inode, from, to, false, true);
+}
+EXPORT_SYMBOL(netfs_clear_stale_post_isize);
diff --git a/fs/netfs/read_collect.c b/fs/netfs/read_collect.c
index a94197ef0..c5bfaf2c6 100644
--- a/fs/netfs/read_collect.c
+++ b/fs/netfs/read_collect.c
@@ -33,6 +33,22 @@ static void netfs_clear_unread(struct netfs_io_subrequest *subreq)
__set_bit(NETFS_SREQ_HIT_EOF, &subreq->flags);
}
+static void netfs_clear_unread_dio(struct netfs_io_subrequest *subreq)
+{
+ uoff_t pos = subreq->start + subreq->transferred;
+ struct netfs_io_request *rreq = subreq->rreq;
+ size_t fill;
+
+ if (pos >= rreq->i_size)
+ return;
+
+ fill = min_t(uoff_t, rreq->i_size - pos,
+ subreq->len - subreq->transferred);
+
+ netfs_reset_iter(subreq);
+ subreq->transferred += iov_iter_zero(fill, &subreq->io_iter);
+}
+
/*
* Cancel the copy-to-cache mark on a folio.
*/
@@ -311,6 +327,14 @@ reassess:
test_bit(NETFS_SREQ_HIT_EOF, &front->flags))
netfs_read_unlock_folios(rreq, &notes);
} else {
+ if (!(notes & HIT_PENDING) &&
+ front->error == 0 &&
+ transferred < front->len &&
+ test_bit(NETFS_SREQ_CLEAR_TAIL, &front->flags)) {
+ netfs_clear_unread_dio(front);
+ transferred = front->transferred;
+ trace_netfs_sreq(front, netfs_sreq_trace_clear);
+ }
stream->collected_to = front->start + transferred;
rreq->collected_to = stream->collected_to;
}
diff --git a/fs/smb/client/cifsfs.c b/fs/smb/client/cifsfs.c
index 7ecd70efd..98b610b2a 100644
--- a/fs/smb/client/cifsfs.c
+++ b/fs/smb/client/cifsfs.c
@@ -1412,6 +1412,7 @@ static loff_t cifs_remap_file_range(struct file *src_file, loff_t off,
* server could even support copy of range where source = target
*/
lock_two_nondirectories(target_inode, src_inode);
+ filemap_invalidate_lock(target_inode->i_mapping);
if (len == 0) {
loff_t src_size = i_size_read(src_inode);
@@ -1467,9 +1468,15 @@ static loff_t cifs_remap_file_range(struct file *src_file, loff_t off,
i_size = target_inode->i_size;
spin_unlock(&target_inode->i_lock);
- /* Discard all the folios that overlap the destination region. */
+ /*
+ * Discard all the folios that overlap the destination region. Start at
+ * the old EOF when extending so the folio straddling it, which may hold
+ * data written past EOF through an mmap, is dropped too.
+ */
cifs_dbg(FYI, "about to discard pages %llx-%llx\n", fstart, fend);
- truncate_inode_pages_range(&target_inode->i_data, fstart, fend);
+ truncate_inode_pages_range(&target_inode->i_data,
+ min(fstart, i_size), fend);
+ netfs_wait_for_outstanding_io(target_inode);
fscache_invalidate(cifs_inode_cookie(target_inode), NULL, i_size, 0);
@@ -1508,6 +1515,7 @@ static loff_t cifs_remap_file_range(struct file *src_file, loff_t off,
if (rc)
CIFS_I(target_inode)->time = 0;
unlock:
+ filemap_invalidate_unlock(target_inode->i_mapping);
/* although unlocking in the reverse order from locking is not
strictly necessary here it is a little cleaner to be consistent */
unlock_two_nondirectories(src_inode, target_inode);
@@ -1531,6 +1539,9 @@ ssize_t cifs_file_copychunk_range(unsigned int xid,
struct cifs_tcon *target_tcon;
ssize_t rc;
+ if (len == 0)
+ return 0;
+
cifs_dbg(FYI, "copychunk range\n");
if (!src_file->private_data || !dst_file->private_data) {
@@ -1560,6 +1571,7 @@ ssize_t cifs_file_copychunk_range(unsigned int xid,
* server could even support copy of range where source = target
*/
lock_two_nondirectories(target_inode, src_inode);
+ filemap_invalidate_lock(target_inode->i_mapping);
cifs_dbg(FYI, "about to flush pages\n");
@@ -1581,10 +1593,28 @@ ssize_t cifs_file_copychunk_range(unsigned int xid,
/* Flush and invalidate all the folios in the destination region. If
* the copy was successful, then some of the flush is extra overhead,
* but we need to allow for the copy failing in some way (eg. ENOSPC).
+ *
+ * Start at the old EOF when extending so the folio straddling it, which
+ * may hold data written past EOF through an mmap, is dropped too.
*/
- rc = filemap_invalidate_inode(target_inode, true, destoff, destoff + len - 1);
- if (rc)
- goto unlock;
+ if (target_inode->i_mapping->nrpages) {
+ loff_t fstart = min(destoff, i_size_read(target_inode));
+ loff_t fend = destoff + len - 1;
+
+ unmap_mapping_pages(target_inode->i_mapping,
+ fstart >> PAGE_SHIFT,
+ (fend >> PAGE_SHIFT) -
+ (fstart >> PAGE_SHIFT) + 1,
+ false);
+ rc = filemap_write_and_wait_range(target_inode->i_mapping,
+ fstart, fend);
+ if (rc)
+ goto unlock;
+ invalidate_inode_pages2_range(target_inode->i_mapping,
+ fstart >> PAGE_SHIFT,
+ fend >> PAGE_SHIFT);
+ }
+ netfs_wait_for_outstanding_io(target_inode);
fscache_invalidate(cifs_inode_cookie(target_inode), NULL,
i_size_read(target_inode), 0);
@@ -1616,6 +1646,7 @@ ssize_t cifs_file_copychunk_range(unsigned int xid,
CIFS_I(target_inode)->time = 0;
unlock:
+ filemap_invalidate_unlock(target_inode->i_mapping);
/* although unlocking in the reverse order from locking is not
* strictly necessary here it is a little cleaner to be consistent
*/
diff --git a/fs/smb/client/cifsfs.h b/fs/smb/client/cifsfs.h
index 0c85daa83..e3d820c97 100644
--- a/fs/smb/client/cifsfs.h
+++ b/fs/smb/client/cifsfs.h
@@ -146,8 +146,9 @@ ssize_t cifs_file_copychunk_range(unsigned int xid, struct file *src_file,
unsigned int flags);
long cifs_ioctl(struct file *filep, unsigned int command, unsigned long arg);
-void cifs_setsize(struct inode *inode, loff_t offset);
-void cifs_resize_file_locked(struct inode *inode, loff_t offset);
+void cifs_setsize(struct inode *inode, loff_t old_size, loff_t offset);
+void cifs_resize_file_locked(struct inode *inode, loff_t old_size,
+ loff_t offset);
struct fs_context;
struct smb3_fs_context;
diff --git a/fs/smb/client/cifsglob.h b/fs/smb/client/cifsglob.h
index 79e4e84f8..5c5b76a9e 100644
--- a/fs/smb/client/cifsglob.h
+++ b/fs/smb/client/cifsglob.h
@@ -1450,11 +1450,11 @@ struct cifsFileInfo {
struct dentry *dentry;
struct tcon_link *tlink;
unsigned int f_flags;
- bool invalidHandle:1; /* file closed via session abend */
- bool swapfile:1;
- bool oplock_break_cancelled:1;
- bool status_file_deleted:1; /* file has been deleted */
- bool offload:1; /* offload final part of _put to a wq */
+ bool invalidHandle; /* file closed via session abend */
+ bool swapfile;
+ bool oplock_break_cancelled;
+ bool status_file_deleted; /* file has been deleted */
+ bool offload; /* offload final part of _put to a wq */
__u16 oplock_epoch; /* epoch from the lease break */
__u32 oplock_level; /* oplock/lease level from the lease break */
int count;
diff --git a/fs/smb/client/file.c b/fs/smb/client/file.c
index 0d428517f..11355d6bf 100644
--- a/fs/smb/client/file.c
+++ b/fs/smb/client/file.c
@@ -1012,11 +1012,14 @@ static int cifs_do_truncate(const unsigned int xid, struct dentry *dentry)
}
mapping_set_error(inode->i_mapping, rc);
+ netfs_wait_for_outstanding_io(inode);
+
cfile = find_writable_file(cinode, FIND_FSUID_ONLY);
rc = cifs_file_flush(xid, inode, cfile);
if (!rc) {
if (cfile) {
struct netfs_inode *ictx = netfs_inode(inode);
+ loff_t old_size = i_size_read(inode);
tcon = tlink_tcon(cfile->tlink);
server = tcon->ses->server;
@@ -1025,7 +1028,7 @@ static int cifs_do_truncate(const unsigned int xid, struct dentry *dentry)
cfile, 0, false);
if (!rc) {
netfs_resize_file(&cinode->netfs, 0, true);
- cifs_setsize(inode, 0);
+ cifs_setsize(inode, old_size, 0);
cifs_invalidate_cache(inode, 0);
}
netfs_wb_end(ictx);
diff --git a/fs/smb/client/inode.c b/fs/smb/client/inode.c
index 1fe0ef0a9..f29555f3c 100644
--- a/fs/smb/client/inode.c
+++ b/fs/smb/client/inode.c
@@ -88,6 +88,8 @@ static void cifs_set_ops(struct inode *inode)
else
inode->i_data.a_ops = &cifs_addr_ops;
mapping_set_large_folios(inode->i_mapping);
+ if (tcon->ses->server->sign)
+ mapping_set_stable_writes(inode->i_mapping);
break;
case S_IFDIR:
if (IS_AUTOMOUNT(inode)) {
@@ -3053,15 +3055,12 @@ int cifs_fiemap(struct inode *inode, struct fiemap_extent_info *fei, u64 start,
return -EOPNOTSUPP;
}
-void cifs_setsize(struct inode *inode, loff_t offset)
+void cifs_setsize(struct inode *inode, loff_t old_size, loff_t offset)
{
- loff_t old_size;
u64 blocks = CIFS_INO_BLOCKS(offset);
spin_lock(&inode->i_lock);
- old_size = i_size_read(inode);
i_size_write(inode, offset);
-
/*
* Extending EOF does not allocate the intervening range. Only clamp
* i_blocks on shrink; allocation growth comes from writes or from the
@@ -3071,20 +3070,28 @@ void cifs_setsize(struct inode *inode, loff_t offset)
inode->i_blocks = blocks;
spin_unlock(&inode->i_lock);
inode_set_mtime_to_ts(inode, inode_set_ctime_current(inode));
+
+ /*
+ * Zero the tail of the folio straddling the old EOF so data dirtied
+ * past EOF through an mmap isn't exposed. truncate_pagecache() then
+ * drops any pagecache beyond the new EOF, as in truncate_setsize().
+ */
if (offset > old_size)
- pagecache_isize_extended(inode, old_size, offset);
+ netfs_clear_stale_post_isize(inode, old_size, offset);
+
truncate_pagecache(inode, offset);
netfs_wait_for_outstanding_io(inode);
}
-void cifs_resize_file_locked(struct inode *inode, loff_t offset)
+void cifs_resize_file_locked(struct inode *inode, loff_t old_size,
+ loff_t offset)
{
struct fscache_cookie *cookie = cifs_inode_cookie(inode);
lockdep_assert_held_write(&inode->i_rwsem);
netfs_resize_file(netfs_inode(inode), offset, true);
- cifs_setsize(inode, offset);
+ cifs_setsize(inode, old_size, offset);
if (!cookie)
return;
@@ -3101,6 +3108,7 @@ int cifs_file_set_size(const unsigned int xid, struct dentry *dentry,
struct inode *inode = d_inode(dentry);
struct cifs_sb_info *cifs_sb = CIFS_SB(inode->i_sb);
struct cifsInodeInfo *cifsInode = CIFS_I(inode);
+ loff_t old_size = i_size_read(inode);
struct tcon_link *tlink = NULL;
struct cifs_tcon *tcon = NULL;
struct TCP_Server_Info *server;
@@ -3159,7 +3167,7 @@ int cifs_file_set_size(const unsigned int xid, struct dentry *dentry,
set_size_out:
if (rc == 0)
- cifs_resize_file_locked(inode, size);
+ cifs_resize_file_locked(inode, old_size, size);
return rc;
}
diff --git a/fs/smb/client/smb2ops.c b/fs/smb/client/smb2ops.c
index aa142420d..583ab4c4d 100644
--- a/fs/smb/client/smb2ops.c
+++ b/fs/smb/client/smb2ops.c
@@ -2287,6 +2287,7 @@ smb2_duplicate_extents(const unsigned int xid,
struct duplicate_extents_to_file dup_ext_buf;
struct timespec64 ts;
struct cifs_tcon *tcon = tlink_tcon(trgtfile->tlink);
+ loff_t old_size;
u64 asize;
/* server fileays advertise duplicate extent support with this flag */
@@ -2305,11 +2306,12 @@ smb2_duplicate_extents(const unsigned int xid,
trgtfile->fid.volatile_fid, tcon->tid,
tcon->ses->Suid, src_off, dest_off, len);
inode = d_inode(trgtfile->dentry);
- if (i_size_read(inode) < dest_off + len) {
+ old_size = i_size_read(inode);
+ if (old_size < dest_off + len) {
rc = smb2_set_file_size(xid, tcon, trgtfile, dest_off + len, false);
if (rc)
goto duplicate_extents_out;
- cifs_resize_file_locked(inode, dest_off + len);
+ cifs_resize_file_locked(inode, old_size, dest_off + len);
}
rc = SMB2_ioctl(xid, tcon, trgtfile->fid.persistent_fid,
trgtfile->fid.volatile_fid,
@@ -3520,6 +3522,21 @@ static long smb3_zero_data(struct file *file, struct cifs_tcon *tcon,
0, NULL, NULL);
}
+static long query_server_eof(const unsigned int xid,
+ struct cifs_tcon *tcon,
+ struct cifsFileInfo *cfile,
+ unsigned long long *eof)
+{
+ struct smb2_file_all_info file_inf = {};
+ long rc;
+
+ rc = SMB2_query_info(xid, tcon, cfile->fid.persistent_fid,
+ cfile->fid.volatile_fid, &file_inf);
+ if (!rc)
+ *eof = le64_to_cpu(file_inf.EndOfFile);
+ return rc;
+}
+
static long smb3_zero_range(struct file *file, struct cifs_tcon *tcon,
unsigned long long offset, unsigned long long len,
bool keep_size)
@@ -3547,25 +3564,32 @@ static long smb3_zero_range(struct file *file, struct cifs_tcon *tcon,
filemap_invalidate_lock(inode->i_mapping);
netfs_read_sizes(inode, &i_size, &remote_i_size, &zero_point);
- if (offset + len >= remote_i_size && offset < i_size) {
- unsigned long long top = umin(offset + len, i_size);
- rc = filemap_write_and_wait_range(inode->i_mapping, offset, top - 1);
- if (rc < 0)
- goto zero_range_exit;
- }
+ rc = filemap_write_and_wait_range(inode->i_mapping, offset,
+ offset + len - 1);
+ if (rc < 0)
+ goto zero_range_exit;
/*
* We zero the range through ioctl, so we need remove the page caches
* first, otherwise the data may be inconsistent with the server.
+ *
+ * Start at the old EOF when extending so the folio straddling it, which
+ * may hold data written past EOF through an mmap, is dropped too.
*/
- truncate_pagecache_range(inode, offset, offset + len - 1);
+ truncate_pagecache_range(inode, min(offset, i_size), offset + len - 1);
netfs_wait_for_outstanding_io(inode);
- /* if file not oplocked can't be sure whether asking to extend size */
- rc = -EOPNOTSUPP;
- if (keep_size == false && !CIFS_CACHE_READ(cifsi))
- goto zero_range_exit;
+ if (!keep_size && !CIFS_CACHE_READ(cifsi)) {
+ rc = query_server_eof(xid, tcon, cfile, &remote_i_size);
+ if (rc)
+ goto zero_range_exit;
+ i_size = max(i_size, remote_i_size);
+ if (i_size < new_size) {
+ rc = -EOPNOTSUPP;
+ goto zero_range_exit;
+ }
+ }
fscache_invalidate(cifs_inode_cookie(inode), NULL,
i_size_read(inode), 0);
@@ -3577,7 +3601,7 @@ static long smb3_zero_range(struct file *file, struct cifs_tcon *tcon,
/*
* do we also need to change the size of the file?
*/
- if (keep_size == false && (unsigned long long)i_size_read(inode) < new_size) {
+ if (!keep_size && umax(i_size, i_size_read(inode)) < new_size) {
rc = SMB2_set_eof(xid, tcon, cfile->fid.persistent_fid,
cfile->fid.volatile_fid, cfile->pid, new_size);
if (rc >= 0) {
@@ -3723,7 +3747,8 @@ static int smb3_simple_fallocate_write_range(unsigned int xid,
static int smb3_simple_fallocate_range(unsigned int xid,
struct cifs_tcon *tcon,
struct cifsFileInfo *cfile,
- loff_t off, loff_t len)
+ loff_t off, loff_t len,
+ loff_t old_eof)
{
struct file_allocated_range_buffer in_data, *out_data = NULL, *tmp_data;
struct inode *inode = d_inode(cfile->dentry);
@@ -3739,12 +3764,30 @@ static int smb3_simple_fallocate_range(unsigned int xid,
goto out;
}
- if (off >= i_size_read(inode)) {
+ if (off >= old_eof) {
rc = smb3_simple_fallocate_write_range(xid, tcon, cfile,
off, len, buf);
goto out;
}
+ filemap_invalidate_lock(inode->i_mapping);
+
+ /*
+ * Flush and commit the data to the server, otherwise
+ * FSCTL_QUERY_ALLOCATED_RANGES might report recently written data as
+ * unallocated holes on Windows Servers, and the loop below would
+ * then zero-fill them and corrupt the file.
+ */
+ rc = filemap_write_and_wait_range(inode->i_mapping, off,
+ off + len - 1);
+ if (rc)
+ goto out_unlock;
+ netfs_wait_for_outstanding_io(inode);
+ rc = SMB2_flush(xid, tcon, cfile->fid.persistent_fid,
+ cfile->fid.volatile_fid);
+ if (rc)
+ goto out_unlock;
+
in_data.file_offset = cpu_to_le64(off);
in_data.length = cpu_to_le64(len);
rc = SMB2_ioctl(xid, tcon, cfile->fid.persistent_fid,
@@ -3754,7 +3797,7 @@ static int smb3_simple_fallocate_range(unsigned int xid,
1024 * sizeof(struct file_allocated_range_buffer),
(char **)&out_data, &out_data_len);
if (rc)
- goto out;
+ goto out_unlock;
tmp_data = out_data;
while (len) {
@@ -3764,12 +3807,12 @@ static int smb3_simple_fallocate_range(unsigned int xid,
if (out_data_len == 0) {
rc = smb3_simple_fallocate_write_range(xid, tcon,
cfile, off, len, buf);
- goto out;
+ goto out_unlock;
}
if (out_data_len < sizeof(struct file_allocated_range_buffer)) {
rc = -EINVAL;
- goto out;
+ goto out_unlock;
}
range_start = le64_to_cpu(tmp_data->file_offset);
@@ -3777,7 +3820,7 @@ static int smb3_simple_fallocate_range(unsigned int xid,
if (check_add_overflow(range_start, range_len, &range_end) ||
range_end > S64_MAX) {
rc = -EINVAL;
- goto out;
+ goto out_unlock;
}
if (off < range_start) {
@@ -3792,11 +3835,11 @@ static int smb3_simple_fallocate_range(unsigned int xid,
rc = smb3_simple_fallocate_write_range(xid, tcon,
cfile, off, l, buf);
if (rc)
- goto out;
+ goto out_unlock;
off = off + l;
len = len - l;
if (len == 0)
- goto out;
+ goto out_unlock;
}
/*
* We are at a section of allocated data, just skip forward
@@ -3815,6 +3858,8 @@ static int smb3_simple_fallocate_range(unsigned int xid,
out_data_len -= sizeof(struct file_allocated_range_buffer);
}
+ out_unlock:
+ filemap_invalidate_unlock(inode->i_mapping);
out:
kfree(out_data);
kvfree(buf);
@@ -3830,7 +3875,7 @@ static long smb3_simple_falloc(struct file *file, struct cifs_tcon *tcon,
struct cifsFileInfo *cfile = file->private_data;
long rc = -EOPNOTSUPP;
unsigned int xid;
- loff_t old_eof, new_eof;
+ loff_t old_eof, new_eof, local_eof;
struct smb2_file_all_info file_inf;
u64 asize;
int qrc;
@@ -3839,18 +3884,37 @@ static long smb3_simple_falloc(struct file *file, struct cifs_tcon *tcon,
inode = d_inode(cfile->dentry);
cifsi = CIFS_I(inode);
- old_eof = i_size_read(inode);
+ old_eof = local_eof = i_size_read(inode);
trace_smb3_falloc_enter(xid, cfile->fid.persistent_fid, tcon->tid,
tcon->ses->Suid, off, len);
- /* if file not oplocked can't be sure whether asking to extend size */
- if (!CIFS_CACHE_READ(cifsi))
- if (!keep_size) {
+
+ if (!keep_size && !CIFS_CACHE_READ(cifsi)) {
+ unsigned long long server_eof;
+
+ rc = filemap_write_and_wait(inode->i_mapping);
+ if (rc) {
trace_smb3_falloc_err(xid, cfile->fid.persistent_fid,
tcon->tid, tcon->ses->Suid, off, len, rc);
free_xid(xid);
return rc;
}
+ netfs_wait_for_outstanding_io(inode);
+
+ rc = query_server_eof(xid, tcon, cfile, &server_eof);
+ if (rc) {
+ trace_smb3_falloc_err(xid, cfile->fid.persistent_fid,
+ tcon->tid, tcon->ses->Suid, off, len, rc);
+ free_xid(xid);
+ return rc;
+ }
+ /*
+ * Only use the larger EOF to decide whether we're extending.
+ * The pagecache zeroing below must still key off the local
+ * i_size, so keep local_eof for that.
+ */
+ old_eof = max_t(loff_t, old_eof, server_eof);
+ }
/*
* Extending the file
@@ -3874,7 +3938,7 @@ static long smb3_simple_falloc(struct file *file, struct cifs_tcon *tcon,
}
rc = smb3_simple_fallocate_range(xid, tcon, cfile,
- off, len);
+ off, len, old_eof);
if (rc) {
spin_lock(&inode->i_lock);
cifsi->time = 0;
@@ -3883,7 +3947,7 @@ static long smb3_simple_falloc(struct file *file, struct cifs_tcon *tcon,
}
new_eof = off + len;
- cifs_resize_file_locked(inode, new_eof);
+ cifs_resize_file_locked(inode, local_eof, new_eof);
qrc = SMB2_query_info(xid, tcon,
cfile->fid.persistent_fid,
@@ -3931,7 +3995,7 @@ static long smb3_simple_falloc(struct file *file, struct cifs_tcon *tcon,
if (rc)
goto out;
- cifs_resize_file_locked(inode, new_eof);
+ cifs_resize_file_locked(inode, local_eof, new_eof);
qrc = SMB2_query_info(xid, tcon,
cfile->fid.persistent_fid,
@@ -3976,7 +4040,7 @@ static long smb3_simple_falloc(struct file *file, struct cifs_tcon *tcon,
}
}
- if ((keep_size == true) || (i_size_read(inode) >= off + len)) {
+ if (keep_size || old_eof >= off + len) {
/*
* At this point, we are trying to fallocate an internal
* regions of a sparse file. Since smb2 does not have a
@@ -3993,7 +4057,7 @@ static long smb3_simple_falloc(struct file *file, struct cifs_tcon *tcon,
*/
if (len <= 1024 * 1024) {
rc = smb3_simple_fallocate_range(xid, tcon, cfile,
- off, len);
+ off, len, old_eof);
goto out;
}
@@ -4005,7 +4069,7 @@ static long smb3_simple_falloc(struct file *file, struct cifs_tcon *tcon,
* ie potentially making a few extra pages at the beginning
* or end of the file non-sparse via set_sparse is harmless.
*/
- if ((off > 8192) || (off + len + 8192 < i_size_read(inode))) {
+ if (off > 8192 || off + len + 8192 < old_eof) {
rc = -EOPNOTSUPP;
goto out;
}
diff --git a/fs/smb/client/smb2pdu.c b/fs/smb/client/smb2pdu.c
index 3d7ead36d..cfe3c4b4a 100644
--- a/fs/smb/client/smb2pdu.c
+++ b/fs/smb/client/smb2pdu.c
@@ -4796,14 +4796,20 @@ do_retry:
rdata->got_bytes);
if (rdata->result == -ENODATA) {
- __set_bit(NETFS_SREQ_HIT_EOF, &rdata->subreq.flags);
rdata->result = 0;
+ if (rdata->subreq.start + rdata->subreq.transferred >= i_size_read(inode))
+ __set_bit(NETFS_SREQ_HIT_EOF, &rdata->subreq.flags);
+ else
+ __set_bit(NETFS_SREQ_CLEAR_TAIL, &rdata->subreq.flags);
} else {
size_t trans = rdata->subreq.transferred + rdata->got_bytes;
if (trans < rdata->subreq.len &&
rdata->subreq.start + trans >= netfs_read_remote_i_size(inode)) {
- __set_bit(NETFS_SREQ_HIT_EOF, &rdata->subreq.flags);
rdata->result = 0;
+ if (rdata->subreq.start + trans >= i_size_read(inode))
+ __set_bit(NETFS_SREQ_HIT_EOF, &rdata->subreq.flags);
+ else
+ __set_bit(NETFS_SREQ_CLEAR_TAIL, &rdata->subreq.flags);
}
if (rdata->got_bytes)
__set_bit(NETFS_SREQ_MADE_PROGRESS, &rdata->subreq.flags);