summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorXiubo Li <xiubo.li@clyso.com>2026-07-21 13:06:53 +0800
committerIlya Dryomov <idryomov@gmail.com>2026-08-26 19:57:27 +0200
commitd2a8d446a09c74c8ddfe108b50dd791c889983fc (patch)
treef4ef7d96246a8ac7cd455f3962fc506dd36b315c
parenta354d7eaa1a57f1532c8072a424cc2d339a73cc0 (diff)
ceph: revalidate ki_pos for O_APPEND writes after cap acquisition
For O_APPEND writes, ki_pos is set to the current EOF via generic_write_checks() after fetching i_size from the MDS. However, ceph_get_caps() may need to wait for Fwx exclusive caps if the write extends the file (endoff > i_max_size). While waiting for Fwx, the previous Fwx holder (another client) may have already extended the file. When the MDS grants us Fwx, the cap grant message updates the local i_size, but ki_pos remains at the old EOF, causing the append write to land at a stale offset and overwrite data from the other client. Fix by re-reading i_size_read(inode) after ceph_get_caps() returns. At this point we hold Fwx exclusive caps, no other client can modify the file, and i_size reflects the true EOF from the MDS cap grant. No extra MDS round-trip is needed. Only adjust ki_pos when the EOF has actually changed. After adjusting ki_pos forward, the write range [pos, pos+count) may now exceed the i_max_size that was validated by ceph_get_caps() for the old range. Re-check against i_max_size and truncate the write if necessary to stay within the MDS-granted limit. Link: https://tracker.ceph.com/issues/7333 Fixes: 8e4473bb50a1 ("ceph: do not execute direct write in parallel if O_APPEND is specified") Signed-off-by: Xiubo Li <xiubo.li@clyso.com> Reviewed-by: Viacheslav Dubeyko <slava@dubeyko.com> Signed-off-by: Ilya Dryomov <idryomov@gmail.com>
-rw-r--r--fs/ceph/file.c48
1 files changed, 48 insertions, 0 deletions
diff --git a/fs/ceph/file.c b/fs/ceph/file.c
index a4a2a4b6a027..a0b9c2b5a583 100644
--- a/fs/ceph/file.c
+++ b/fs/ceph/file.c
@@ -2477,6 +2477,54 @@ retry_snap:
if (err < 0)
goto out;
+ /*
+ * For O_APPEND writes we may have waited for Fwx exclusive caps
+ * while the previous Fwx holder (another client) extended the
+ * file. i_size has been updated via the cap grant message from
+ * the MDS, but ki_pos is still the old EOF. Re-read i_size here
+ * (no extra MDS round-trip needed) and adjust ki_pos to the true
+ * EOF. Since we hold Fwx, no other client can change the file.
+ */
+ if (iocb->ki_flags & IOCB_APPEND) {
+ loff_t cur_eof = i_size_read(inode);
+
+ if (cur_eof != pos) {
+ doutc(cl,
+ "%p %llx.%llx O_APPEND: pos adjusted %lld -> %lld\n",
+ inode, ceph_vinop(inode), pos, cur_eof);
+ iocb->ki_pos = cur_eof;
+ pos = cur_eof;
+ if (pos >= limit) {
+ err = -EFBIG;
+ goto out_caps;
+ }
+ iov_iter_truncate(from, limit - pos);
+ count = iov_iter_count(from);
+
+ /*
+ * ceph_get_caps() validated the old endoff
+ * against i_max_size; adjusting ki_pos forward
+ * may have shifted the write range beyond the
+ * granted max_size. Re-check and truncate if
+ * necessary.
+ */
+ spin_lock(&ci->i_ceph_lock);
+ if (pos + count > (loff_t)ci->i_max_size) {
+ loff_t max_size = ci->i_max_size;
+
+ spin_unlock(&ci->i_ceph_lock);
+ if (pos >= max_size) {
+ err = -EFBIG;
+ goto out_caps;
+ }
+ iov_iter_truncate(from, max_size - pos);
+ count = iov_iter_count(from);
+ } else {
+ spin_unlock(&ci->i_ceph_lock);
+ }
+ }
+ }
+
err = file_update_time(file);
if (err)
goto out_caps;