diff options
author | Andrew Senkevich <andrew.senkevich@intel.com> | 2016-01-16 00:49:45 +0300 |
---|---|---|
committer | Andrew Senkevich <andrew.senkevich@intel.com> | 2016-01-16 00:49:45 +0300 |
commit | 72276d6e8843db6df5971b06787f0a5e39bda138 (patch) | |
tree | ad7ed01db58285d38559773305d5d8b16eca39d3 /sysdeps/x86_64/multiarch/memcpy.S | |
parent | b02840bacdefde318d2ad2f920e50785b9b25d69 (diff) | |
download | glibc-72276d6e8843db6df5971b06787f0a5e39bda138.tar.gz glibc-72276d6e8843db6df5971b06787f0a5e39bda138.tar.xz glibc-72276d6e8843db6df5971b06787f0a5e39bda138.zip |
Added memcpy/memmove family optimized with AVX512 for KNL hardware.
Added AVX512 implementations of memcpy, mempcpy, memmove, memcpy_chk, mempcpy_chk, memmove_chk. It shows average improvement more than 30% over AVX versions on KNL hardware (performance results in the thread <https://sourceware.org/ml/libc-alpha/2016-01/msg00258.html>). * sysdeps/x86_64/multiarch/Makefile (sysdep_routines): Added new files. * sysdeps/x86_64/multiarch/ifunc-impl-list.c: Added new tests. * sysdeps/x86_64/multiarch/memcpy-avx512-no-vzeroupper.S: New file. * sysdeps/x86_64/multiarch/mempcpy-avx512-no-vzeroupper.S: Likewise. * sysdeps/x86_64/multiarch/memmove-avx512-no-vzeroupper.S: Likewise. * sysdeps/x86_64/multiarch/memcpy.S: Added new IFUNC branch. * sysdeps/x86_64/multiarch/memcpy_chk.S: Likewise. * sysdeps/x86_64/multiarch/memmove.c: Likewise. * sysdeps/x86_64/multiarch/memmove_chk.c: Likewise. * sysdeps/x86_64/multiarch/mempcpy.S: Likewise. * sysdeps/x86_64/multiarch/mempcpy_chk.S: Likewise.
Diffstat (limited to 'sysdeps/x86_64/multiarch/memcpy.S')
-rw-r--r-- | sysdeps/x86_64/multiarch/memcpy.S | 22 |
1 files changed, 15 insertions, 7 deletions
diff --git a/sysdeps/x86_64/multiarch/memcpy.S b/sysdeps/x86_64/multiarch/memcpy.S index 27fca2957e..64a1bcd137 100644 --- a/sysdeps/x86_64/multiarch/memcpy.S +++ b/sysdeps/x86_64/multiarch/memcpy.S @@ -30,19 +30,27 @@ ENTRY(__new_memcpy) .type __new_memcpy, @gnu_indirect_function LOAD_RTLD_GLOBAL_RO_RDX - leaq __memcpy_avx_unaligned(%rip), %rax +#ifdef HAVE_AVX512_ASM_SUPPORT + HAS_ARCH_FEATURE (AVX512F_Usable) + jz 1f + HAS_ARCH_FEATURE (Prefer_No_VZEROUPPER) + jz 1f + leaq __memcpy_avx512_no_vzeroupper(%rip), %rax + ret +#endif +1: leaq __memcpy_avx_unaligned(%rip), %rax HAS_ARCH_FEATURE (AVX_Fast_Unaligned_Load) - jz 1f + jz 2f ret -1: leaq __memcpy_sse2(%rip), %rax +2: leaq __memcpy_sse2(%rip), %rax HAS_ARCH_FEATURE (Slow_BSF) - jnz 2f + jnz 3f leaq __memcpy_sse2_unaligned(%rip), %rax ret -2: HAS_CPU_FEATURE (SSSE3) - jz 3f +3: HAS_CPU_FEATURE (SSSE3) + jz 4f leaq __memcpy_ssse3(%rip), %rax -3: ret +4: ret END(__new_memcpy) # undef ENTRY |