From: "Darrick J. Wong" <darrick.wong@oracle.com> To: david@fromorbit.com, darrick.wong@oracle.com Cc: sandeen@redhat.com, linux-nfs@vger.kernel.org, linux-cifs@vger.kernel.org, Amir Goldstein <amir73il@gmail.com>, linux-unionfs@vger.kernel.org, linux-xfs@vger.kernel.org, linux-mm@kvack.org, linux-btrfs@vger.kernel.org, linux-fsdevel@vger.kernel.org, ocfs2-devel@oss.oracle.com Subject: [PATCH 19/25] vfs: implement opportunistic short dedupe Date: Wed, 10 Oct 2018 21:14:45 -0700 [thread overview] Message-ID: <153923128529.5546.6430455638279784448.stgit@magnolia> (raw) In-Reply-To: <153923113649.5546.9840926895953408273.stgit@magnolia> From: Darrick J. Wong <darrick.wong@oracle.com> For a given dedupe request, the bytes_deduped field in the control structure tells userspace if we managed to deduplicate some, but not all of, the requested regions starting from the file offsets supplied. However, due to sloppy coding, the current dedupe code returns FILE_DEDUPE_RANGE_DIFFERS if any part of the range is different. Fix this so that we can actually support partial request completion. Signed-off-by: Darrick J. Wong <darrick.wong@oracle.com> Reviewed-by: Amir Goldstein <amir73il@gmail.com> --- fs/read_write.c | 48 ++++++++++++++++++++++++++++++++++++++---------- include/linux/fs.h | 7 +++++-- 2 files changed, 43 insertions(+), 12 deletions(-) diff --git a/fs/read_write.c b/fs/read_write.c index c88a443d9eb2..de055cb9c5ae 100644 --- a/fs/read_write.c +++ b/fs/read_write.c @@ -1737,13 +1737,26 @@ static struct page *vfs_dedupe_get_page(struct inode *inode, loff_t offset) return page; } +static unsigned int vfs_dedupe_memcmp(const char *s1, const char *s2, + unsigned int len) +{ + const char *orig_s1; + + for (orig_s1 = s1; len > 0; s1++, s2++, len--) + if (*s1 != *s2) + break; + + return s1 - orig_s1; +} + /* * Compare extents of two files to see if they are the same. * Caller must have locked both inodes to prevent write races. */ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, struct inode *dest, loff_t destoff, - loff_t len, bool *is_same) + loff_t *req_len, + unsigned int remap_flags) { loff_t src_poff; loff_t dest_poff; @@ -1751,8 +1764,11 @@ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, void *dest_addr; struct page *src_page; struct page *dest_page; - loff_t cmp_len; + loff_t len = *req_len; + loff_t same_len = 0; bool same; + unsigned int cmp_len; + unsigned int cmp_same; int error; error = -EINVAL; @@ -1762,7 +1778,7 @@ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, dest_poff = destoff & (PAGE_SIZE - 1); cmp_len = min(PAGE_SIZE - src_poff, PAGE_SIZE - dest_poff); - cmp_len = min(cmp_len, len); + cmp_len = min_t(loff_t, cmp_len, len); if (cmp_len <= 0) goto out_error; @@ -1784,7 +1800,10 @@ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, flush_dcache_page(src_page); flush_dcache_page(dest_page); - if (memcmp(src_addr + src_poff, dest_addr + dest_poff, cmp_len)) + cmp_same = vfs_dedupe_memcmp(src_addr + src_poff, + dest_addr + dest_poff, cmp_len); + same_len += cmp_same; + if (cmp_same != cmp_len) same = false; kunmap_atomic(dest_addr); @@ -1802,7 +1821,17 @@ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, len -= cmp_len; } - *is_same = same; + /* + * If less than the whole range matched, we have to back down to the + * nearest block boundary. + */ + if (*req_len != same_len) { + if (!(remap_flags & RFR_SHORT_DEDUPE)) + return -EBADE; + + *req_len = ALIGN_DOWN(same_len, dest->i_sb->s_blocksize); + } + return 0; out_error: @@ -1881,13 +1910,11 @@ int generic_remap_file_range_prep(struct file *file_in, loff_t pos_in, * Check that the extents are the same. */ if (is_dedupe) { - bool is_same = false; - ret = vfs_dedupe_file_range_compare(inode_in, pos_in, - inode_out, pos_out, *len, &is_same); + inode_out, pos_out, len, remap_flags); if (ret) return ret; - if (!is_same) + if (*len == 0) return -EBADE; } @@ -2013,7 +2040,8 @@ loff_t vfs_dedupe_file_range_one(struct file *src_file, loff_t src_pos, { loff_t ret; - WARN_ON_ONCE(remap_flags & ~(RFR_SAME_DATA)); + WARN_ON_ONCE(remap_flags & ~(RFR_SAME_DATA | RFR_CAN_SHORTEN | + RFR_SHORT_DEDUPE)); ret = mnt_want_write_file(dst_file); if (ret) diff --git a/include/linux/fs.h b/include/linux/fs.h index f0603ed007e9..18b6db85ab64 100644 --- a/include/linux/fs.h +++ b/include/linux/fs.h @@ -1727,16 +1727,19 @@ struct block_device_operations; * RFR_SAME_DATA: only remap if contents identical (i.e. deduplicate) * RFR_TO_SRC_EOF: remap to the end of the source file * RFR_CAN_SHORTEN: caller can handle a shortened request + * RFR_SHORT_DEDUPE: deduplicate from byte 0 until the file data don't match */ #define RFR_SAME_DATA (1 << 0) #define RFR_TO_SRC_EOF (1 << 1) #define RFR_CAN_SHORTEN (1 << 2) +#define RFR_SHORT_DEDUPE (1 << 3) #define RFR_VALID_FLAGS (RFR_SAME_DATA | RFR_TO_SRC_EOF | \ - RFR_CAN_SHORTEN) + RFR_CAN_SHORTEN | RFR_SHORT_DEDUPE) /* Implemented by the VFS, so these are advisory. */ -#define RFR_VFS_FLAGS (RFR_TO_SRC_EOF | RFR_CAN_SHORTEN) +#define RFR_VFS_FLAGS (RFR_TO_SRC_EOF | RFR_CAN_SHORTEN | \ + RFR_SHORT_DEDUPE) /* * Filesystem remapping implementations should call this helper on their
WARNING: multiple messages have this Message-ID (diff)
From: Darrick J. Wong <darrick.wong@oracle.com> To: david@fromorbit.com, darrick.wong@oracle.com Cc: sandeen@redhat.com, linux-nfs@vger.kernel.org, linux-cifs@vger.kernel.org, Amir Goldstein <amir73il@gmail.com>, linux-unionfs@vger.kernel.org, linux-xfs@vger.kernel.org, linux-mm@kvack.org, linux-btrfs@vger.kernel.org, linux-fsdevel@vger.kernel.org, ocfs2-devel@oss.oracle.com Subject: [Ocfs2-devel] [PATCH 19/25] vfs: implement opportunistic short dedupe Date: Wed, 10 Oct 2018 21:14:45 -0700 [thread overview] Message-ID: <153923128529.5546.6430455638279784448.stgit@magnolia> (raw) In-Reply-To: <153923113649.5546.9840926895953408273.stgit@magnolia> From: Darrick J. Wong <darrick.wong@oracle.com> For a given dedupe request, the bytes_deduped field in the control structure tells userspace if we managed to deduplicate some, but not all of, the requested regions starting from the file offsets supplied. However, due to sloppy coding, the current dedupe code returns FILE_DEDUPE_RANGE_DIFFERS if any part of the range is different. Fix this so that we can actually support partial request completion. Signed-off-by: Darrick J. Wong <darrick.wong@oracle.com> Reviewed-by: Amir Goldstein <amir73il@gmail.com> --- fs/read_write.c | 48 ++++++++++++++++++++++++++++++++++++++---------- include/linux/fs.h | 7 +++++-- 2 files changed, 43 insertions(+), 12 deletions(-) diff --git a/fs/read_write.c b/fs/read_write.c index c88a443d9eb2..de055cb9c5ae 100644 --- a/fs/read_write.c +++ b/fs/read_write.c @@ -1737,13 +1737,26 @@ static struct page *vfs_dedupe_get_page(struct inode *inode, loff_t offset) return page; } +static unsigned int vfs_dedupe_memcmp(const char *s1, const char *s2, + unsigned int len) +{ + const char *orig_s1; + + for (orig_s1 = s1; len > 0; s1++, s2++, len--) + if (*s1 != *s2) + break; + + return s1 - orig_s1; +} + /* * Compare extents of two files to see if they are the same. * Caller must have locked both inodes to prevent write races. */ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, struct inode *dest, loff_t destoff, - loff_t len, bool *is_same) + loff_t *req_len, + unsigned int remap_flags) { loff_t src_poff; loff_t dest_poff; @@ -1751,8 +1764,11 @@ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, void *dest_addr; struct page *src_page; struct page *dest_page; - loff_t cmp_len; + loff_t len = *req_len; + loff_t same_len = 0; bool same; + unsigned int cmp_len; + unsigned int cmp_same; int error; error = -EINVAL; @@ -1762,7 +1778,7 @@ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, dest_poff = destoff & (PAGE_SIZE - 1); cmp_len = min(PAGE_SIZE - src_poff, PAGE_SIZE - dest_poff); - cmp_len = min(cmp_len, len); + cmp_len = min_t(loff_t, cmp_len, len); if (cmp_len <= 0) goto out_error; @@ -1784,7 +1800,10 @@ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, flush_dcache_page(src_page); flush_dcache_page(dest_page); - if (memcmp(src_addr + src_poff, dest_addr + dest_poff, cmp_len)) + cmp_same = vfs_dedupe_memcmp(src_addr + src_poff, + dest_addr + dest_poff, cmp_len); + same_len += cmp_same; + if (cmp_same != cmp_len) same = false; kunmap_atomic(dest_addr); @@ -1802,7 +1821,17 @@ static int vfs_dedupe_file_range_compare(struct inode *src, loff_t srcoff, len -= cmp_len; } - *is_same = same; + /* + * If less than the whole range matched, we have to back down to the + * nearest block boundary. + */ + if (*req_len != same_len) { + if (!(remap_flags & RFR_SHORT_DEDUPE)) + return -EBADE; + + *req_len = ALIGN_DOWN(same_len, dest->i_sb->s_blocksize); + } + return 0; out_error: @@ -1881,13 +1910,11 @@ int generic_remap_file_range_prep(struct file *file_in, loff_t pos_in, * Check that the extents are the same. */ if (is_dedupe) { - bool is_same = false; - ret = vfs_dedupe_file_range_compare(inode_in, pos_in, - inode_out, pos_out, *len, &is_same); + inode_out, pos_out, len, remap_flags); if (ret) return ret; - if (!is_same) + if (*len == 0) return -EBADE; } @@ -2013,7 +2040,8 @@ loff_t vfs_dedupe_file_range_one(struct file *src_file, loff_t src_pos, { loff_t ret; - WARN_ON_ONCE(remap_flags & ~(RFR_SAME_DATA)); + WARN_ON_ONCE(remap_flags & ~(RFR_SAME_DATA | RFR_CAN_SHORTEN | + RFR_SHORT_DEDUPE)); ret = mnt_want_write_file(dst_file); if (ret) diff --git a/include/linux/fs.h b/include/linux/fs.h index f0603ed007e9..18b6db85ab64 100644 --- a/include/linux/fs.h +++ b/include/linux/fs.h @@ -1727,16 +1727,19 @@ struct block_device_operations; * RFR_SAME_DATA: only remap if contents identical (i.e. deduplicate) * RFR_TO_SRC_EOF: remap to the end of the source file * RFR_CAN_SHORTEN: caller can handle a shortened request + * RFR_SHORT_DEDUPE: deduplicate from byte 0 until the file data don't match */ #define RFR_SAME_DATA (1 << 0) #define RFR_TO_SRC_EOF (1 << 1) #define RFR_CAN_SHORTEN (1 << 2) +#define RFR_SHORT_DEDUPE (1 << 3) #define RFR_VALID_FLAGS (RFR_SAME_DATA | RFR_TO_SRC_EOF | \ - RFR_CAN_SHORTEN) + RFR_CAN_SHORTEN | RFR_SHORT_DEDUPE) /* Implemented by the VFS, so these are advisory. */ -#define RFR_VFS_FLAGS (RFR_TO_SRC_EOF | RFR_CAN_SHORTEN) +#define RFR_VFS_FLAGS (RFR_TO_SRC_EOF | RFR_CAN_SHORTEN | \ + RFR_SHORT_DEDUPE) /* * Filesystem remapping implementations should call this helper on their
next prev parent reply other threads:[~2018-10-11 4:14 UTC|newest] Thread overview: 96+ messages / expand[flat|nested] mbox.gz Atom feed top 2018-10-11 4:12 [PATCH v3 00/25] fs: fixes for serious clone/dedupe problems Darrick J. Wong 2018-10-11 4:12 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:12 ` [PATCH 01/25] xfs: add a per-xfs trace_printk macro Darrick J. Wong 2018-10-11 4:12 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 13:39 ` Christoph Hellwig 2018-10-11 13:39 ` [Ocfs2-devel] " Christoph Hellwig 2018-10-11 23:34 ` Darrick J. Wong 2018-10-11 23:34 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:12 ` [PATCH 02/25] vfs: vfs_clone_file_prep_inodes should return EINVAL for a clone from beyond EOF Darrick J. Wong 2018-10-11 4:12 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 13:40 ` Christoph Hellwig 2018-10-11 13:40 ` [Ocfs2-devel] " Christoph Hellwig 2018-10-11 4:12 ` [PATCH 03/25] vfs: check file ranges before cloning files Darrick J. Wong 2018-10-11 4:12 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 13:42 ` Christoph Hellwig 2018-10-11 13:42 ` [Ocfs2-devel] " Christoph Hellwig 2018-10-11 14:13 ` Amir Goldstein 2018-10-11 4:12 ` [PATCH 04/25] vfs: strengthen checking of file range inputs to generic_remap_checks Darrick J. Wong 2018-10-11 4:12 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 13:43 ` Christoph Hellwig 2018-10-11 13:43 ` [Ocfs2-devel] " Christoph Hellwig 2018-10-11 4:12 ` [PATCH 05/25] vfs: avoid problematic remapping requests into partial EOF block Darrick J. Wong 2018-10-11 4:12 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-12 0:16 ` Dave Chinner 2018-10-12 0:16 ` [Ocfs2-devel] " Dave Chinner 2018-10-12 16:07 ` Darrick J. Wong 2018-10-12 16:07 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-12 20:22 ` Filipe Manana 2018-10-12 20:22 ` Filipe Manana 2018-10-15 0:31 ` Dave Chinner 2018-10-15 0:31 ` [Ocfs2-devel] " Dave Chinner 2018-11-02 12:04 ` Filipe Manana 2018-11-02 12:04 ` Filipe Manana 2018-11-02 17:42 ` Darrick J. Wong 2018-11-02 17:42 ` Darrick J. Wong 2018-11-02 17:42 ` [Ocfs2-devel] " Darrick J. Wong 2018-11-02 18:18 ` Filipe Manana 2018-11-02 19:05 ` Filipe Manana 2018-10-11 4:13 ` [PATCH 06/25] vfs: skip zero-length dedupe requests Darrick J. Wong 2018-10-11 4:13 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:13 ` [PATCH 07/25] vfs: combine the clone and dedupe into a single remap_file_range Darrick J. Wong 2018-10-11 4:13 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:13 ` [PATCH 08/25] vfs: rename vfs_clone_file_prep to be more descriptive Darrick J. Wong 2018-10-11 4:13 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:13 ` [PATCH 09/25] vfs: rename clone_verify_area to remap_verify_area Darrick J. Wong 2018-10-11 4:13 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:13 ` [PATCH 10/25] vfs: create generic_remap_file_range_touch to update inode metadata Darrick J. Wong 2018-10-11 4:13 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:13 ` [PATCH 11/25] vfs: pass remap flags to generic_remap_file_range_prep Darrick J. Wong 2018-10-11 4:13 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:13 ` [PATCH 12/25] vfs: pass remap flags to generic_remap_checks Darrick J. Wong 2018-10-11 4:13 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:13 ` [PATCH 13/25] vfs: make remap_file_range functions take and return bytes completed Darrick J. Wong 2018-10-11 4:13 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:14 ` [PATCH 14/25] vfs: plumb RFR_* remap flags through the vfs clone functions Darrick J. Wong 2018-10-11 4:14 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:14 ` [PATCH 15/25] vfs: plumb RFR_* remap flags through the vfs dedupe functions Darrick J. Wong 2018-10-11 4:14 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:14 ` [PATCH 16/25] vfs: make remapping to source file eof more explicit Darrick J. Wong 2018-10-11 4:14 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:14 ` [PATCH 17/25] vfs: enable remap callers that can handle short operations Darrick J. Wong 2018-10-11 4:14 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 5:15 ` Amir Goldstein 2018-10-11 16:04 ` Darrick J. Wong 2018-10-11 16:04 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 16:05 ` [PATCH v2 " Darrick J. Wong 2018-10-11 16:05 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:14 ` [PATCH 18/25] vfs: hide file range comparison function Darrick J. Wong 2018-10-11 4:14 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:14 ` Darrick J. Wong [this message] 2018-10-11 4:14 ` [Ocfs2-devel] [PATCH 19/25] vfs: implement opportunistic short dedupe Darrick J. Wong 2018-10-11 4:14 ` [PATCH 20/25] ocfs2: truncate page cache for clone destination file before remapping Darrick J. Wong 2018-10-11 4:14 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:14 ` [PATCH 21/25] ocfs2: fix pagecache truncation prior to reflink Darrick J. Wong 2018-10-11 4:14 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:15 ` [PATCH 22/25] ocfs2: support partial clone range and dedupe range Darrick J. Wong 2018-10-11 4:15 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:15 ` [PATCH 23/25] xfs: fix pagecache truncation prior to reflink Darrick J. Wong 2018-10-11 4:15 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-12 1:15 ` Dave Chinner 2018-10-12 1:15 ` [Ocfs2-devel] " Dave Chinner 2018-10-11 4:15 ` [PATCH 24/25] xfs: support returning partial reflink results Darrick J. Wong 2018-10-11 4:15 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-12 1:22 ` Dave Chinner 2018-10-12 1:22 ` [Ocfs2-devel] " Dave Chinner 2018-10-12 16:06 ` Darrick J. Wong 2018-10-12 16:06 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-11 4:15 ` [PATCH 25/25] xfs: remove redundant remap partial EOF block checks Darrick J. Wong 2018-10-11 4:15 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-12 1:22 ` Dave Chinner 2018-10-12 1:22 ` [Ocfs2-devel] " Dave Chinner 2018-10-11 8:33 ` [PATCH v3 00/25] fs: fixes for serious clone/dedupe problems Amir Goldstein 2018-10-11 15:55 ` Darrick J. Wong 2018-10-11 15:55 ` [Ocfs2-devel] " Darrick J. Wong 2018-10-13 0:05 [PATCH v4 " Darrick J. Wong 2018-10-13 0:07 ` [PATCH 19/25] vfs: implement opportunistic short dedupe Darrick J. Wong 2018-10-14 17:26 ` Christoph Hellwig
Reply instructions: You may reply publicly to this message via plain-text email using any one of the following methods: * Save the following mbox file, import it into your mail client, and reply-to-all from there: mbox Avoid top-posting and favor interleaved quoting: https://en.wikipedia.org/wiki/Posting_style#Interleaved_style * Reply using the --to, --cc, and --in-reply-to switches of git-send-email(1): git send-email \ --in-reply-to=153923128529.5546.6430455638279784448.stgit@magnolia \ --to=darrick.wong@oracle.com \ --cc=amir73il@gmail.com \ --cc=david@fromorbit.com \ --cc=linux-btrfs@vger.kernel.org \ --cc=linux-cifs@vger.kernel.org \ --cc=linux-fsdevel@vger.kernel.org \ --cc=linux-mm@kvack.org \ --cc=linux-nfs@vger.kernel.org \ --cc=linux-unionfs@vger.kernel.org \ --cc=linux-xfs@vger.kernel.org \ --cc=ocfs2-devel@oss.oracle.com \ --cc=sandeen@redhat.com \ /path/to/YOUR_REPLY https://kernel.org/pub/software/scm/git/docs/git-send-email.html * If your mail client supports setting the In-Reply-To header via mailto: links, try the mailto: linkBe sure your reply has a Subject: header at the top and a blank line before the message body.
This is an external index of several public inboxes, see mirroring instructions on how to clone and mirror all data and code used by this external index.