ef9c4cb6c7
The difference between memset and wmemset is byte vs int. Add stubs to SSE2/AVX2/AVX512 memset for wmemset with updated constant and size: SSE2 wmemset: shl $0x2,%rdx movd %esi,%xmm0 mov %rdi,%rax pshufd $0x0,%xmm0,%xmm0 jmp entry_from_wmemset SSE2 memset: movd %esi,%xmm0 mov %rdi,%rax punpcklbw %xmm0,%xmm0 punpcklwd %xmm0,%xmm0 pshufd $0x0,%xmm0,%xmm0 entry_from_wmemset: Since the ERMS versions of wmemset requires "rep stosl" instead of "rep stosb", only the vector store stubs of SSE2/AVX2/AVX512 wmemset are added. The SSE2 wmemset is about 3X faster and the AVX2 wmemset is about 6X faster on Haswell. * include/wchar.h (__wmemset_chk): New. * sysdeps/x86_64/memset.S (VDUP_TO_VEC0_AND_SET_RETURN): Renamed to MEMSET_VDUP_TO_VEC0_AND_SET_RETURN. (WMEMSET_VDUP_TO_VEC0_AND_SET_RETURN): New. (WMEMSET_CHK_SYMBOL): Likewise. (WMEMSET_SYMBOL): Likewise. (__wmemset): Add hidden definition. (wmemset): Add weak hidden definition. * sysdeps/x86_64/multiarch/Makefile (sysdep_routines): Add wmemset_chk-nonshared. * sysdeps/x86_64/multiarch/ifunc-impl-list.c (__libc_ifunc_impl_list): Add __wmemset_sse2_unaligned, __wmemset_avx2_unaligned, __wmemset_avx512_unaligned, __wmemset_chk_sse2_unaligned, __wmemset_chk_avx2_unaligned and __wmemset_chk_avx512_unaligned. * sysdeps/x86_64/multiarch/memset-avx2-unaligned-erms.S (VDUP_TO_VEC0_AND_SET_RETURN): Renamed to ... (MEMSET_VDUP_TO_VEC0_AND_SET_RETURN): This. (WMEMSET_VDUP_TO_VEC0_AND_SET_RETURN): New. (WMEMSET_SYMBOL): Likewise. * sysdeps/x86_64/multiarch/memset-avx512-unaligned-erms.S (VDUP_TO_VEC0_AND_SET_RETURN): Renamed to ... (MEMSET_VDUP_TO_VEC0_AND_SET_RETURN): This. (WMEMSET_VDUP_TO_VEC0_AND_SET_RETURN): New. (WMEMSET_SYMBOL): Likewise. * sysdeps/x86_64/multiarch/memset-vec-unaligned-erms.S: Updated. (WMEMSET_CHK_SYMBOL): New. (WMEMSET_CHK_SYMBOL (__wmemset_chk, unaligned)): Likewise. (WMEMSET_SYMBOL (__wmemset, unaligned)): Likewise. * sysdeps/x86_64/multiarch/memset.S (WMEMSET_SYMBOL): New. (libc_hidden_builtin_def): Also define __GI_wmemset and __GI___wmemset. (weak_alias): New. * sysdeps/x86_64/multiarch/wmemset.c: New file. * sysdeps/x86_64/multiarch/wmemset.h: Likewise. * sysdeps/x86_64/multiarch/wmemset_chk-nonshared.S: Likewise. * sysdeps/x86_64/multiarch/wmemset_chk.c: Likewise. * sysdeps/x86_64/wmemset.c: Likewise. * sysdeps/x86_64/wmemset_chk.c: Likewise.
68 lines
2.0 KiB
ArmAsm
68 lines
2.0 KiB
ArmAsm
/* memset/bzero -- set memory area to CH/0
|
|
Optimized version for x86-64.
|
|
Copyright (C) 2002-2017 Free Software Foundation, Inc.
|
|
This file is part of the GNU C Library.
|
|
|
|
The GNU C Library is free software; you can redistribute it and/or
|
|
modify it under the terms of the GNU Lesser General Public
|
|
License as published by the Free Software Foundation; either
|
|
version 2.1 of the License, or (at your option) any later version.
|
|
|
|
The GNU C Library is distributed in the hope that it will be useful,
|
|
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
|
|
Lesser General Public License for more details.
|
|
|
|
You should have received a copy of the GNU Lesser General Public
|
|
License along with the GNU C Library; if not, see
|
|
<http://www.gnu.org/licenses/>. */
|
|
|
|
#include <sysdep.h>
|
|
|
|
#define VEC_SIZE 16
|
|
#define VEC(i) xmm##i
|
|
/* Don't use movups and movaps since it will get larger nop paddings for
|
|
alignment. */
|
|
#define VMOVU movdqu
|
|
#define VMOVA movdqa
|
|
|
|
#define MEMSET_VDUP_TO_VEC0_AND_SET_RETURN(d, r) \
|
|
movd d, %xmm0; \
|
|
movq r, %rax; \
|
|
punpcklbw %xmm0, %xmm0; \
|
|
punpcklwd %xmm0, %xmm0; \
|
|
pshufd $0, %xmm0, %xmm0
|
|
|
|
#define WMEMSET_VDUP_TO_VEC0_AND_SET_RETURN(d, r) \
|
|
movd d, %xmm0; \
|
|
movq r, %rax; \
|
|
pshufd $0, %xmm0, %xmm0
|
|
|
|
#define SECTION(p) p
|
|
|
|
#ifndef MEMSET_SYMBOL
|
|
# define MEMSET_CHK_SYMBOL(p,s) p
|
|
# define MEMSET_SYMBOL(p,s) memset
|
|
#endif
|
|
|
|
#ifndef WMEMSET_SYMBOL
|
|
# define WMEMSET_CHK_SYMBOL(p,s) p
|
|
# define WMEMSET_SYMBOL(p,s) __wmemset
|
|
#endif
|
|
|
|
#include "multiarch/memset-vec-unaligned-erms.S"
|
|
|
|
libc_hidden_builtin_def (memset)
|
|
|
|
#if IS_IN (libc)
|
|
libc_hidden_def (__wmemset)
|
|
weak_alias (__wmemset, wmemset)
|
|
libc_hidden_weak (wmemset)
|
|
#endif
|
|
|
|
#if defined SHARED && IS_IN (libc) && !defined USE_MULTIARCH
|
|
strong_alias (__memset_chk, __memset_zero_constant_len_parameter)
|
|
.section .gnu.warning.__memset_zero_constant_len_parameter
|
|
.string "memset used with constant zero length parameter; this could be due to transposed parameters"
|
|
#endif
|