Re: [PATCH v9 8/8] x86/string: extend memcpy_flushcache() fixed-size fastpaths

Borislav Petkov <[email protected]>
Newsgroups org.kernel.vger.linux-hardening,org.kernel.vger.linux-arch,org.kernel.vger.linux-kernel,org.kvack.linux-mm
Message-ID <20260804203508.GEanJM_LGXmItYhFKZ@fat_crate.local>
On Mon, Aug 03, 2026 at 03:09:29PM +0800, Li Zhe wrote:
> Tested in a VM with a 100 GB fsdax namespace device configured with
> map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake
> server.
> 
> Test procedure:
> Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap
> initialization time from the pr_debug() output of
> memmap_init_zone_device().
> 
> With memcpy_nontemporal() used by the ZONE_DEVICE template-copy path:
>   Average of rebinds for nd_pmem driver: 150.83 ms
>   Average of rebinds for dax_pmem driver: 153.55 ms
> 
> With this x86 fixed-size fastpath patch applied:
>   Average of rebinds for nd_pmem driver: 96.79 ms
>   Average of rebinds for dax_pmem driver: 119.04 ms
> 
> This further reduces the average memmap initialization time measured
> during rebind by about 35.8% for nd_pmem and 22.5% for dax_pmem.

Testing methodology doesn't usually belong in the commit message but under the
"---" lines below but this is important enough to keep it here.

> Signed-off-by: Li Zhe <[email protected]>
> ---
>  arch/x86/include/asm/string_64.h | 56 +++++++++++++++++++++++++++++++-
>  1 file changed, 55 insertions(+), 1 deletion(-)

A quick and untested cleanup ontop with the potential for a bunch more
unification. But later.

Note that you're inconsistent about the "memory" clobber. The current code
doesn't have it, you're adding it to movnti_8(). I've added it now everywhere
to be on the safe side - it would be interesting to know whether you see any
perf difference with and without it in your use case.

Thx.

---
diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h
index bc6a9f34b346..3a1e1f17431b 100644
--- a/arch/x86/include/asm/string_64.h
+++ b/arch/x86/include/asm/string_64.h
@@ -83,6 +83,14 @@ int strcmp(const char *cs, const char *ct);
 #define __HAVE_ARCH_MEMCPY_FLUSHCACHE 1
 void __memcpy_flushcache(void *dst, const void *src, size_t cnt);
 
+static __always_inline void movnti_4(void *dst, const void *src)
+{
+	asm volatile("movntil %1, %0"
+		     : "=m"(*(u32 *)dst)
+		     : "r"(*(u32 *)src)
+		     : "memory");
+}
+
 static __always_inline void movnti_8(void *dst, const void *src)
 {
 	asm volatile("movntiq %1, %0"
@@ -112,47 +120,26 @@ static __always_inline void movnti_64(void *dst, const void *src)
 static __always_inline void memcpy_flushcache(void *dst, const void *src,
 					      size_t cnt)
 {
-	if (__builtin_constant_p(cnt)) {
-		switch (cnt) {
-			case 4:
-				asm ("movntil %1, %0" : "=m"(*(u32 *)dst) : "r"(*(u32 *)src));
-				return;
-			case 8:
-				asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src));
-				return;
-			case 16:
-				asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src));
-				asm ("movntiq %1, %0" : "=m"(*(u64 *)(dst + 8)) : "r"(*(u64 *)(src + 8)));
-				return;
-			/*
-			 * The relevant fixed-size copies here are the
-			 * x86_64 struct page sizes: 64, 80, and 96 bytes.
-			 * Keep 32-byte and 48-byte copies inline as well
-			 * instead of sending those nearby fixed-size
-			 * cases back to __memcpy_flushcache().
-			 */
-			case 32:
-				movnti_32(dst, src);
-				return;
-			case 48:
-				movnti_32(dst, src);
-				movnti_16(dst + 32, src + 32);
-				return;
-			case 64:
-				movnti_64(dst, src);
-				return;
-			case 80:
-				movnti_64(dst, src);
-				movnti_16(dst + 64, src + 64);
-				return;
-			case 96:
-				movnti_64(dst, src);
-				movnti_32(dst + 64, src + 64);
-				return;
-		}
+	if (!__builtin_constant_p(cnt))
+		return __memcpy_flushcache(dst, src, cnt);
+
+	/*
+	 * The relevant fixed-size copies here are the x86_64 struct page sizes:
+	 * 64, 80, and 96 bytes. Keep 32-byte and 48-byte copies inline as well
+	 * instead of sending those nearby fixed-size cases back to
+	 * __memcpy_flushcache().
+	 */
+	switch (cnt) {
+	case 4:  movnti_4(dst, src); break;
+	case 8:	 movnti_8(dst, src); break;
+	case 16: movnti_16(dst, src); break;
+	case 32: movnti_32(dst, src); break;
+	case 48: movnti_32(dst, src); movnti_16(dst + 32, src + 32); break;
+	case 64: movnti_64(dst, src); break;
+	case 80: movnti_64(dst, src); movnti_16(dst + 64, src + 64); break;
+	case 96: movnti_64(dst, src); movnti_32(dst + 64, src + 64); break;
+	default: __memcpy_flushcache(dst, src, cnt); break;
 	}
-
-	__memcpy_flushcache(dst, src, cnt);
 }
 
 #define memcpy_nontemporal memcpy_nontemporal

-- 
Regards/Gruss,
    Boris.

https://people.kernel.org/tglx/notes-about-netiquette
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.