]> git.hungrycats.org Git - linux/commitdiff
ceph: revalidate ki_pos for O_APPEND writes after cap acquisition
authorXiubo Li <xiubo.li@clyso.com>
Tue, 21 Jul 2026 05:06:53 +0000 (13:06 +0800)
committerGreg Kroah-Hartman <gregkh@linuxfoundation.org>
Mon, 14 Sep 2026 11:36:13 +0000 (13:36 +0200)
[ Upstream commit d2a8d446a09c74c8ddfe108b50dd791c889983fc ]

For O_APPEND writes, ki_pos is set to the current EOF via
generic_write_checks() after fetching i_size from the MDS.  However,
ceph_get_caps() may need to wait for Fwx exclusive caps if the write
extends the file (endoff > i_max_size).  While waiting for Fwx, the
previous Fwx holder (another client) may have already extended the
file.  When the MDS grants us Fwx, the cap grant message updates the
local i_size, but ki_pos remains at the old EOF, causing the append
write to land at a stale offset and overwrite data from the other
client.

Fix by re-reading i_size_read(inode) after ceph_get_caps() returns.
At this point we hold Fwx exclusive caps, no other client can modify
the file, and i_size reflects the true EOF from the MDS cap grant.
No extra MDS round-trip is needed.  Only adjust ki_pos when the EOF
has actually changed.

After adjusting ki_pos forward, the write range [pos, pos+count) may
now exceed the i_max_size that was validated by ceph_get_caps() for
the old range.  Re-check against i_max_size and truncate the write
if necessary to stay within the MDS-granted limit.

Link: https://tracker.ceph.com/issues/7333
Fixes: 8e4473bb50a1 ("ceph: do not execute direct write in parallel if O_APPEND is specified")
Signed-off-by: Xiubo Li <xiubo.li@clyso.com>
Reviewed-by: Viacheslav Dubeyko <slava@dubeyko.com>
Signed-off-by: Ilya Dryomov <idryomov@gmail.com>
Signed-off-by: Sasha Levin <sashal@kernel.org>
fs/ceph/file.c

index f640aff749edcec38ae424f07b63a5d7c8d8cc1f..fa582c8537c5fe0ffdc15dbba32c746b19228b8c 100644 (file)
@@ -2412,6 +2412,54 @@ retry_snap:
        if (err < 0)
                goto out;
 
+       /*
+        * For O_APPEND writes we may have waited for Fwx exclusive caps
+        * while the previous Fwx holder (another client) extended the
+        * file.  i_size has been updated via the cap grant message from
+        * the MDS, but ki_pos is still the old EOF.  Re-read i_size here
+        * (no extra MDS round-trip needed) and adjust ki_pos to the true
+        * EOF.  Since we hold Fwx, no other client can change the file.
+        */
+       if (iocb->ki_flags & IOCB_APPEND) {
+               loff_t cur_eof = i_size_read(inode);
+
+               if (cur_eof != pos) {
+                       doutc(cl,
+                             "%p %llx.%llx O_APPEND: pos adjusted %lld -> %lld\n",
+                             inode, ceph_vinop(inode), pos, cur_eof);
+                       iocb->ki_pos = cur_eof;
+                       pos = cur_eof;
+                       if (pos >= limit) {
+                               err = -EFBIG;
+                               goto out_caps;
+                       }
+                       iov_iter_truncate(from, limit - pos);
+                       count = iov_iter_count(from);
+
+                       /*
+                        * ceph_get_caps() validated the old endoff
+                        * against i_max_size; adjusting ki_pos forward
+                        * may have shifted the write range beyond the
+                        * granted max_size.  Re-check and truncate if
+                        * necessary.
+                        */
+                       spin_lock(&ci->i_ceph_lock);
+                       if (pos + count > (loff_t)ci->i_max_size) {
+                               loff_t max_size = ci->i_max_size;
+
+                               spin_unlock(&ci->i_ceph_lock);
+                               if (pos >= max_size) {
+                                       err = -EFBIG;
+                                       goto out_caps;
+                               }
+                               iov_iter_truncate(from, max_size - pos);
+                               count = iov_iter_count(from);
+                       } else {
+                               spin_unlock(&ci->i_ceph_lock);
+                       }
+               }
+       }
+
        err = file_update_time(file);
        if (err)
                goto out_caps;