Re: [PATCH v9 8/8] x86/string: extend memcpy_flushcache() fixed-size fastpaths
Borislav Petkov <[email protected]> Tue, 4 Aug 2026 13:35:08 -0700
| Newsgroups | org.kernel.vger.linux-arch,org.kernel.vger.linux-hardening,org.kernel.vger.linux-kernel,org.kvack.linux-mm |
|---|---|
| Message-ID | <20260804203508.GEanJM_LGXmItYhFKZ@fat_crate.local> |
On Mon, Aug 03, 2026 at 03:09:29PM +0800, Li Zhe wrote: > Tested in a VM with a 100 GB fsdax namespace device configured with > map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake > server. > > Test procedure: > Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap > initialization time from the pr_debug() output of > memmap_init_zone_device(). > > With memcpy_nontemporal() used by the ZONE_DEVICE template-copy path: > Average of rebinds for nd_pmem driver: 150.83 ms > Average of rebinds for dax_pmem driver: 153.55 ms > > With this x86 fixed-size fastpath patch applied: > Average of rebinds for nd_pmem driver: 96.79 ms > Average of rebinds for dax_pmem driver: 119.04 ms > > This further reduces the average memmap initialization time measured > during rebind by about 35.8% for nd_pmem and 22.5% for dax_pmem. Testing methodology doesn't usually belong in the commit message but under the "---" lines below but this is important enough to keep it here. > Signed-off-by: Li Zhe <[email protected]> > --- > arch/x86/include/asm/string_64.h | 56 +++++++++++++++++++++++++++++++- > 1 file changed, 55 insertions(+), 1 deletion(-) A quick and untested cleanup ontop with the potential for a bunch more unification. But later. Note that you're inconsistent about the "memory" clobber. The current code doesn't have it, you're adding it to movnti_8(). I've added it now everywhere to be on the safe side - it would be interesting to know whether you see any perf difference with and without it in your use case. Thx. --- diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h index bc6a9f34b346..3a1e1f17431b 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -83,6 +83,14 @@ int strcmp(const char *cs, const char *ct); #define __HAVE_ARCH_MEMCPY_FLUSHCACHE 1 void __memcpy_flushcache(void *dst, const void *src, size_t cnt); +static __always_inline void movnti_4(void *dst, const void *src) +{ + asm volatile("movntil %1, %0" + : "=m"(*(u32 *)dst) + : "r"(*(u32 *)src) + : "memory"); +} + static __always_inline void movnti_8(void *dst, const void *src) { asm volatile("movntiq %1, %0" @@ -112,47 +120,26 @@ static __always_inline void movnti_64(void *dst, const void *src) static __always_inline void memcpy_flushcache(void *dst, const void *src, size_t cnt) { - if (__builtin_constant_p(cnt)) { - switch (cnt) { - case 4: - asm ("movntil %1, %0" : "=m"(*(u32 *)dst) : "r"(*(u32 *)src)); - return; - case 8: - asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src)); - return; - case 16: - asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src)); - asm ("movntiq %1, %0" : "=m"(*(u64 *)(dst + 8)) : "r"(*(u64 *)(src + 8))); - return; - /* - * The relevant fixed-size copies here are the - * x86_64 struct page sizes: 64, 80, and 96 bytes. - * Keep 32-byte and 48-byte copies inline as well - * instead of sending those nearby fixed-size - * cases back to __memcpy_flushcache(). - */ - case 32: - movnti_32(dst, src); - return; - case 48: - movnti_32(dst, src); - movnti_16(dst + 32, src + 32); - return; - case 64: - movnti_64(dst, src); - return; - case 80: - movnti_64(dst, src); - movnti_16(dst + 64, src + 64); - return; - case 96: - movnti_64(dst, src); - movnti_32(dst + 64, src + 64); - return; - } + if (!__builtin_constant_p(cnt)) + return __memcpy_flushcache(dst, src, cnt); + + /* + * The relevant fixed-size copies here are the x86_64 struct page sizes: + * 64, 80, and 96 bytes. Keep 32-byte and 48-byte copies inline as well + * instead of sending those nearby fixed-size cases back to + * __memcpy_flushcache(). + */ + switch (cnt) { + case 4: movnti_4(dst, src); break; + case 8: movnti_8(dst, src); break; + case 16: movnti_16(dst, src); break; + case 32: movnti_32(dst, src); break; + case 48: movnti_32(dst, src); movnti_16(dst + 32, src + 32); break; + case 64: movnti_64(dst, src); break; + case 80: movnti_64(dst, src); movnti_16(dst + 64, src + 64); break; + case 96: movnti_64(dst, src); movnti_32(dst + 64, src + 64); break; + default: __memcpy_flushcache(dst, src, cnt); break; } - - __memcpy_flushcache(dst, src, cnt); } #define memcpy_nontemporal memcpy_nontemporal -- Regards/Gruss, Boris. https://people.kernel.org/tglx/notes-about-netiquette