diff --git a/glibc-RHEL-175520-1.patch b/glibc-RHEL-175520-1.patch new file mode 100644 index 0000000..dd544ac --- /dev/null +++ b/glibc-RHEL-175520-1.patch @@ -0,0 +1,153 @@ +commit 43388b3ac09c757af1d94c9071e52b79b4bd8bb4 +Author: Noah Goldstein +Date: Mon Jan 10 15:35:36 2022 -0600 + + string/test-str*cmp: remove stupid_[strcmp, strncmp, wcscmp, wcsncmp]. + + These implementations just add to test duration. Since we have + simple_* implementations we already have a safe reference + implementation. + + Signed-off-by: Noah Goldstein + +diff --git a/string/test-strcmp.c b/string/test-strcmp.c +index a0255b9625fbcedd..a6b5344f5710effc 100644 +--- a/string/test-strcmp.c ++++ b/string/test-strcmp.c +@@ -36,7 +36,6 @@ + # define STRLEN wcslen + # define MEMCPY wmemcpy + # define SIMPLE_STRCMP simple_wcscmp +-# define STUPID_STRCMP stupid_wcscmp + # define CHAR wchar_t + # define UCHAR wchar_t + # define CHARBYTES 4 +@@ -66,25 +65,6 @@ simple_wcscmp (const wchar_t *s1, const wchar_t *s2) + return c1 < c2 ? -1 : 1; + } + +-int +-stupid_wcscmp (const wchar_t *s1, const wchar_t *s2) +-{ +- size_t ns1 = wcslen (s1) + 1; +- size_t ns2 = wcslen (s2) + 1; +- size_t n = ns1 < ns2 ? ns1 : ns2; +- int ret = 0; +- +- wchar_t c1, c2; +- +- while (n--) { +- c1 = *s1++; +- c2 = *s2++; +- if ((ret = c1 < c2 ? -1 : c1 == c2 ? 0 : 1) != 0) +- break; +- } +- return ret; +-} +- + #else + # include + +@@ -94,7 +74,6 @@ stupid_wcscmp (const wchar_t *s1, const wchar_t *s2) + # define STRLEN strlen + # define MEMCPY memcpy + # define SIMPLE_STRCMP simple_strcmp +-# define STUPID_STRCMP stupid_strcmp + # define CHAR char + # define UCHAR unsigned char + # define CHARBYTES 1 +@@ -115,24 +94,10 @@ simple_strcmp (const char *s1, const char *s2) + return ret; + } + +-int +-stupid_strcmp (const char *s1, const char *s2) +-{ +- size_t ns1 = strlen (s1) + 1; +- size_t ns2 = strlen (s2) + 1; +- size_t n = ns1 < ns2 ? ns1 : ns2; +- int ret = 0; +- +- while (n--) +- if ((ret = *(unsigned char *) s1++ - *(unsigned char *) s2++) != 0) +- break; +- return ret; +-} + #endif + + typedef int (*proto_t) (const CHAR *, const CHAR *); + +-IMPL (STUPID_STRCMP, 1) + IMPL (SIMPLE_STRCMP, 1) + IMPL (STRCMP, 1) + +diff --git a/string/test-strncmp.c b/string/test-strncmp.c +index 56e23670ae7f90e4..a761136e94d65efc 100644 +--- a/string/test-strncmp.c ++++ b/string/test-strncmp.c +@@ -34,7 +34,6 @@ + # define STRDUP wcsdup + # define MEMCPY wmemcpy + # define SIMPLE_STRNCMP simple_wcsncmp +-# define STUPID_STRNCMP stupid_wcsncmp + # define CHAR wchar_t + # define UCHAR wchar_t + # define CHARBYTES 4 +@@ -58,25 +57,6 @@ simple_wcsncmp (const CHAR *s1, const CHAR *s2, size_t n) + return 0; + } + +-int +-stupid_wcsncmp (const CHAR *s1, const CHAR *s2, size_t n) +-{ +- wchar_t c1, c2; +- size_t ns1 = wcsnlen (s1, n) + 1, ns2 = wcsnlen (s2, n) + 1; +- +- n = ns1 < n ? ns1 : n; +- n = ns2 < n ? ns2 : n; +- +- while (n--) +- { +- c1 = *s1++; +- c2 = *s2++; +- if (c1 != c2) +- return c1 > c2 ? 1 : -1; +- } +- return 0; +-} +- + #else + # define L(str) str + # define STRNCMP strncmp +@@ -84,7 +64,6 @@ stupid_wcsncmp (const CHAR *s1, const CHAR *s2, size_t n) + # define STRDUP strdup + # define MEMCPY memcpy + # define SIMPLE_STRNCMP simple_strncmp +-# define STUPID_STRNCMP stupid_strncmp + # define CHAR char + # define UCHAR unsigned char + # define CHARBYTES 1 +@@ -102,23 +81,10 @@ simple_strncmp (const char *s1, const char *s2, size_t n) + return ret; + } + +-int +-stupid_strncmp (const char *s1, const char *s2, size_t n) +-{ +- size_t ns1 = strnlen (s1, n) + 1, ns2 = strnlen (s2, n) + 1; +- int ret = 0; +- +- n = ns1 < n ? ns1 : n; +- n = ns2 < n ? ns2 : n; +- while (n-- && (ret = *(unsigned char *) s1++ - * (unsigned char *) s2++) == 0); +- return ret; +-} +- + #endif + + typedef int (*proto_t) (const CHAR *, const CHAR *, size_t); + +-IMPL (STUPID_STRNCMP, 0) + IMPL (SIMPLE_STRNCMP, 0) + IMPL (STRNCMP, 1) + diff --git a/glibc-RHEL-175520-10.patch b/glibc-RHEL-175520-10.patch new file mode 100644 index 0000000..a7e990b --- /dev/null +++ b/glibc-RHEL-175520-10.patch @@ -0,0 +1,977 @@ +commit a3c50bf46a1ca6d9d2b7d879176d345abf95a9de +Author: Noah Goldstein +Date: Thu Sep 21 09:38:37 2023 -0500 + + x86: Prepare `strrchr-evex` and `strrchr-evex512` for AVX10 + + This commit refactors `strrchr-evex` and `strrchr-evex512` to use a + common implementation: `strrchr-evex-base.S`. + + The motivation is `strrchr-evex` needed to be refactored to not use + 64-bit masked registers in preperation for AVX10. + + Once vec-width masked register combining was removed, the EVEX and + EVEX512 implementations can easily be implemented in the same file + without any major overhead. + + The net result is performance improvements (measured on TGL) for both + `strrchr-evex` and `strrchr-evex512`. Although, note there are some + regressions in the test suite and it may be many of the cases that + make the total-geomean of improvement/regression across bench-strrchr + are cold. The point of the performance measurement is to show there + are no major regressions, but the primary motivation is preperation + for AVX10. + + Benchmarks where taken on TGL: + https://www.intel.com/content/www/us/en/products/sku/213799/intel-core-i711850h-processor-24m-cache-up-to-4-80-ghz/specifications.html + + EVEX geometric_mean(N=5) of all benchmarks New / Original : 0.74 + EVEX512 geometric_mean(N=5) of all benchmarks New / Original: 0.87 + + Full check passes on x86. + +Conflicts: + sysdeps/x86_64/multiarch/strrchr-evex-base.S + (fixup context, missing spell fixes for removed comments) + sysdeps/x86_64/multiarch/strrchr-evex.S + (adapted downstream) + +diff --git a/sysdeps/x86_64/multiarch/strrchr-evex-base.S b/sysdeps/x86_64/multiarch/strrchr-evex-base.S +index 81cab3e0178c8b1e..8cb38be69fef5a45 100644 +--- a/sysdeps/x86_64/multiarch/strrchr-evex-base.S ++++ b/sysdeps/x86_64/multiarch/strrchr-evex-base.S +@@ -1,4 +1,4 @@ +-/* Placeholder function, not used by any processor at the moment. ++/* Implementation for strrchr using evex256 and evex512. + Copyright (C) 2022 Free Software Foundation, Inc. + This file is part of the GNU C Library. + +@@ -16,8 +16,6 @@ + License along with the GNU C Library; if not, see + . */ + +-/* UNUSED. Exists purely as reference implementation. */ +- + #include + + #if ISA_SHOULD_BUILD (4) +@@ -25,240 +23,351 @@ + # include + + # ifdef USE_AS_WCSRCHR ++# if VEC_SIZE == 64 ++# define RCX_M cx ++# define KORTEST_M kortestw ++# else ++# define RCX_M cl ++# define KORTEST_M kortestb ++# endif ++ ++# define SHIFT_REG VRCX + # define CHAR_SIZE 4 +-# define VPBROADCAST vpbroadcastd +-# define VPCMPEQ vpcmpeqd +-# define VPMINU vpminud ++# define VPCMP vpcmpd ++# define VPMIN vpminud ++# define VPCOMPRESS vpcompressd + # define VPTESTN vptestnmd ++# define VPTEST vptestmd ++# define VPBROADCAST vpbroadcastd ++# define VPCMPEQ vpcmpeqd ++ + # else ++# define SHIFT_REG VRDI + # define CHAR_SIZE 1 +-# define VPBROADCAST vpbroadcastb +-# define VPCMPEQ vpcmpeqb +-# define VPMINU vpminub ++# define VPCMP vpcmpb ++# define VPMIN vpminub ++# define VPCOMPRESS vpcompressb + # define VPTESTN vptestnmb ++# define VPTEST vptestmb ++# define VPBROADCAST vpbroadcastb ++# define VPCMPEQ vpcmpeqb ++ ++# define RCX_M VRCX ++# define KORTEST_M KORTEST + # endif + +-# define PAGE_SIZE 4096 ++# define VMATCH VMM(0) + # define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) ++# define PAGE_SIZE 4096 + + .section SECTION(.text), "ax", @progbits +-/* Aligning entry point to 64 byte, provides better performance for +- one vector length string. */ +-ENTRY_P2ALIGN (STRRCHR, 6) +- +- /* Broadcast CHAR to VMM(0). */ +- VPBROADCAST %esi, %VMM(0) ++ /* Aligning entry point to 64 byte, provides better performance for ++ one vector length string. */ ++ENTRY_P2ALIGN(STRRCHR, 6) + movl %edi, %eax +- sall $20, %eax +- cmpl $((PAGE_SIZE - VEC_SIZE) << 20), %eax +- ja L(page_cross) ++ /* Broadcast CHAR to VMATCH. */ ++ VPBROADCAST %esi, %VMATCH + +-L(page_cross_continue): +- /* Compare [w]char for null, mask bit will be set for match. */ +- VMOVU (%rdi), %VMM(1) ++ andl $(PAGE_SIZE - 1), %eax ++ cmpl $(PAGE_SIZE - VEC_SIZE), %eax ++ jg L(cross_page_boundary) + +- VPTESTN %VMM(1), %VMM(1), %k1 +- KMOV %k1, %VRCX +- test %VRCX, %VRCX +- jz L(align_more) +- +- VPCMPEQ %VMM(1), %VMM(0), %k0 +- KMOV %k0, %VRAX +- BLSMSK %VRCX, %VRCX +- and %VRCX, %VRAX +- jz L(ret) +- +- BSR %VRAX, %VRAX ++ VMOVU (%rdi), %VMM(1) ++ /* k0 has a 1 for each zero CHAR in YMM1. */ ++ VPTESTN %VMM(1), %VMM(1), %k0 ++ KMOV %k0, %VGPR(rsi) ++ test %VGPR(rsi), %VGPR(rsi) ++ jz L(aligned_more) ++ /* fallthrough: zero CHAR in first VEC. */ ++L(page_cross_return): ++ /* K1 has a 1 for each search CHAR match in VEC(1). */ ++ VPCMPEQ %VMATCH, %VMM(1), %k1 ++ KMOV %k1, %VGPR(rax) ++ /* Build mask up until first zero CHAR (used to mask of ++ potential search CHAR matches past the end of the string). */ ++ blsmsk %VGPR(rsi), %VGPR(rsi) ++ /* Use `and` here to remove any out of bounds matches so we can ++ do a reverse scan on `rax` to find the last match. */ ++ and %VGPR(rsi), %VGPR(rax) ++ jz L(ret0) ++ /* Get last match. */ ++ bsr %VGPR(rax), %VGPR(rax) + # ifdef USE_AS_WCSRCHR + leaq (%rdi, %rax, CHAR_SIZE), %rax + # else +- add %rdi, %rax ++ addq %rdi, %rax + # endif +-L(ret): ++L(ret0): + ret + +-L(vector_x2_end): +- VPCMPEQ %VMM(2), %VMM(0), %k2 +- KMOV %k2, %VRAX +- BLSMSK %VRCX, %VRCX +- and %VRCX, %VRAX +- jz L(vector_x1_ret) +- +- BSR %VRAX, %VRAX +- leaq (VEC_SIZE)(%rdi, %rax, CHAR_SIZE), %rax +- ret +- +- /* Check the first vector at very last to look for match. */ +-L(vector_x1_ret): +- VPCMPEQ %VMM(1), %VMM(0), %k2 +- KMOV %k2, %VRAX +- test %VRAX, %VRAX +- jz L(ret) +- +- BSR %VRAX, %VRAX ++ /* Returns for first vec x1/x2/x3 have hard coded backward ++ search path for earlier matches. */ ++ .p2align 4,, 6 ++L(first_vec_x1): ++ VPCMPEQ %VMATCH, %VMM(2), %k1 ++ KMOV %k1, %VGPR(rax) ++ blsmsk %VGPR(rcx), %VGPR(rcx) ++ /* eax non-zero if search CHAR in range. */ ++ and %VGPR(rcx), %VGPR(rax) ++ jnz L(first_vec_x1_return) ++ ++ /* fallthrough: no match in YMM2 then need to check for earlier ++ matches (in YMM1). */ ++ .p2align 4,, 4 ++L(first_vec_x0_test): ++ VPCMPEQ %VMATCH, %VMM(1), %k1 ++ KMOV %k1, %VGPR(rax) ++ test %VGPR(rax), %VGPR(rax) ++ jz L(ret1) ++ bsr %VGPR(rax), %VGPR(rax) + # ifdef USE_AS_WCSRCHR + leaq (%rsi, %rax, CHAR_SIZE), %rax + # else +- add %rsi, %rax ++ addq %rsi, %rax + # endif ++L(ret1): + ret + +-L(align_more): +- /* Zero r8 to store match result. */ +- xorl %r8d, %r8d +- /* Save pointer of first vector, in case if no match found. */ ++ .p2align 4,, 10 ++L(first_vec_x3): ++ VPCMPEQ %VMATCH, %VMM(4), %k1 ++ KMOV %k1, %VGPR(rax) ++ blsmsk %VGPR(rcx), %VGPR(rcx) ++ /* If no search CHAR match in range check YMM1/YMM2/YMM3. */ ++ and %VGPR(rcx), %VGPR(rax) ++ jz L(first_vec_x1_or_x2) ++ bsr %VGPR(rax), %VGPR(rax) ++ leaq (VEC_SIZE * 3)(%rdi, %rax, CHAR_SIZE), %rax ++ ret ++ .p2align 4,, 4 ++ ++L(first_vec_x2): ++ VPCMPEQ %VMATCH, %VMM(3), %k1 ++ KMOV %k1, %VGPR(rax) ++ blsmsk %VGPR(rcx), %VGPR(rcx) ++ /* Check YMM3 for last match first. If no match try YMM2/YMM1. */ ++ and %VGPR(rcx), %VGPR(rax) ++ jz L(first_vec_x0_x1_test) ++ bsr %VGPR(rax), %VGPR(rax) ++ leaq (VEC_SIZE * 2)(%r8, %rax, CHAR_SIZE), %rax ++ ret ++ ++ .p2align 4,, 6 ++L(first_vec_x0_x1_test): ++ VPCMPEQ %VMATCH, %VMM(2), %k1 ++ KMOV %k1, %VGPR(rax) ++ /* Check YMM2 for last match first. If no match try YMM1. */ ++ test %VGPR(rax), %VGPR(rax) ++ jz L(first_vec_x0_test) ++ .p2align 4,, 4 ++L(first_vec_x1_return): ++ bsr %VGPR(rax), %VGPR(rax) ++ leaq (VEC_SIZE)(%r8, %rax, CHAR_SIZE), %rax ++ ret ++ ++ .p2align 4,, 12 ++L(aligned_more): ++L(page_cross_continue): ++ /* Need to keep original pointer incase VEC(1) has last match. */ + movq %rdi, %rsi +- /* Align pointer to vector size. */ + andq $-VEC_SIZE, %rdi +- /* Loop unroll for 2 vector loop. */ +- VMOVA (VEC_SIZE)(%rdi), %VMM(2) ++ ++ VMOVU VEC_SIZE(%rdi), %VMM(2) + VPTESTN %VMM(2), %VMM(2), %k0 + KMOV %k0, %VRCX ++ movq %rdi, %r8 + test %VRCX, %VRCX +- jnz L(vector_x2_end) ++ jnz L(first_vec_x1) ++ ++ VMOVU (VEC_SIZE * 2)(%rdi), %VMM(3) ++ VPTESTN %VMM(3), %VMM(3), %k0 ++ KMOV %k0, %VRCX ++ ++ test %VRCX, %VRCX ++ jnz L(first_vec_x2) ++ ++ VMOVU (VEC_SIZE * 3)(%rdi), %VMM(4) ++ VPTESTN %VMM(4), %VMM(4), %k0 ++ KMOV %k0, %VRCX ++ ++ /* Intentionally use 64-bit here. EVEX256 version needs 1-byte ++ padding for efficient nop before loop alignment. */ ++ test %rcx, %rcx ++ jnz L(first_vec_x3) + +- /* Save pointer of second vector, in case if no match +- found. */ +- movq %rdi, %r9 +- /* Align address to VEC_SIZE * 2 for loop. */ + andq $-(VEC_SIZE * 2), %rdi ++ .p2align 4 ++L(first_aligned_loop): ++ /* Preserve VEC(1), VEC(2), VEC(3), and VEC(4) until we can ++ gurantee they don't store a match. */ ++ VMOVA (VEC_SIZE * 4)(%rdi), %VMM(5) ++ VMOVA (VEC_SIZE * 5)(%rdi), %VMM(6) + +- .p2align 4,,11 +-L(loop): +- /* 2 vector loop, as it provide better performance as compared +- to 4 vector loop. */ +- VMOVA (VEC_SIZE * 2)(%rdi), %VMM(3) +- VMOVA (VEC_SIZE * 3)(%rdi), %VMM(4) +- VPCMPEQ %VMM(3), %VMM(0), %k1 +- VPCMPEQ %VMM(4), %VMM(0), %k2 +- VPMINU %VMM(3), %VMM(4), %VMM(5) +- VPTESTN %VMM(5), %VMM(5), %k0 +- KOR %k1, %k2, %k3 +- subq $-(VEC_SIZE * 2), %rdi +- /* If k0 and k3 zero, match and end of string not found. */ +- KORTEST %k0, %k3 +- jz L(loop) +- +- /* If k0 is non zero, end of string found. */ +- KORTEST %k0, %k0 +- jnz L(endloop) +- +- lea VEC_SIZE(%rdi), %r8 +- /* A match found, it need to be stored in r8 before loop +- continue. */ +- /* Check second vector first. */ +- KMOV %k2, %VRDX +- test %VRDX, %VRDX +- jnz L(loop_vec_x2_match) ++ VPCMP $4, %VMM(5), %VMATCH, %k2 ++ VPCMP $4, %VMM(6), %VMATCH, %k3{%k2} + ++ VPMIN %VMM(5), %VMM(6), %VMM(7) ++ ++ VPTEST %VMM(7), %VMM(7), %k1{%k3} ++ subq $(VEC_SIZE * -2), %rdi ++ KORTEST_M %k1, %k1 ++ jc L(first_aligned_loop) ++ ++ VPTESTN %VMM(7), %VMM(7), %k1 + KMOV %k1, %VRDX +- /* Match is in first vector, rdi offset need to be substracted +- by VEC_SIZE. */ +- sub $VEC_SIZE, %r8 +- +- /* If second vector doesn't have match, first vector must +- have match. */ +-L(loop_vec_x2_match): +- BSR %VRDX, %VRDX +-# ifdef USE_AS_WCSRCHR +- sal $2, %rdx +-# endif +- add %rdx, %r8 +- jmp L(loop) ++ test %VRDX, %VRDX ++ jz L(second_aligned_loop_prep) + +-L(endloop): +- /* Check if string end in first loop vector. */ +- VPTESTN %VMM(3), %VMM(3), %k0 +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(loop_vector_x1_end) ++ KORTEST_M %k3, %k3 ++ jnc L(return_first_aligned_loop) + +- /* Check if it has match in first loop vector. */ +- KMOV %k1, %VRAX ++ .p2align 4,, 6 ++L(first_vec_x1_or_x2_or_x3): ++ VPCMPEQ %VMM(4), %VMATCH, %k4 ++ KMOV %k4, %VRAX + test %VRAX, %VRAX +- jz L(loop_vector_x2_end) +- +- BSR %VRAX, %VRAX +- leaq (%rdi, %rax, CHAR_SIZE), %r8 ++ jz L(first_vec_x1_or_x2) ++ bsr %VRAX, %VRAX ++ leaq (VEC_SIZE * 3)(%r8, %rax, CHAR_SIZE), %rax ++ ret + +- /* String must end in second loop vector. */ +-L(loop_vector_x2_end): +- VPTESTN %VMM(4), %VMM(4), %k0 ++ .p2align 4,, 8 ++L(return_first_aligned_loop): ++ VPTESTN %VMM(5), %VMM(5), %k0 + KMOV %k0, %VRCX ++ blsmsk %VRCX, %VRCX ++ jnc L(return_first_new_match_first) ++ blsmsk %VRDX, %VRDX ++ VPCMPEQ %VMM(6), %VMATCH, %k0 ++ KMOV %k0, %VRAX ++ addq $VEC_SIZE, %rdi ++ and %VRDX, %VRAX ++ jnz L(return_first_new_match_ret) ++ subq $VEC_SIZE, %rdi ++L(return_first_new_match_first): + KMOV %k2, %VRAX +- BLSMSK %VRCX, %VRCX +- /* Check if it has match in second loop vector. */ ++# ifdef USE_AS_WCSRCHR ++ xorl $((1 << CHAR_PER_VEC)- 1), %VRAX + and %VRCX, %VRAX +- jz L(check_last_match) ++# else ++ andn %VRCX, %VRAX, %VRAX ++# endif ++ jz L(first_vec_x1_or_x2_or_x3) ++L(return_first_new_match_ret): ++ bsr %VRAX, %VRAX ++ leaq (VEC_SIZE * 2)(%rdi, %rax, CHAR_SIZE), %rax ++ ret + +- BSR %VRAX, %VRAX +- leaq (VEC_SIZE)(%rdi, %rax, CHAR_SIZE), %rax ++ .p2align 4,, 10 ++L(first_vec_x1_or_x2): ++ VPCMPEQ %VMM(3), %VMATCH, %k3 ++ KMOV %k3, %VRAX ++ test %VRAX, %VRAX ++ jz L(first_vec_x0_x1_test) ++ bsr %VRAX, %VRAX ++ leaq (VEC_SIZE * 2)(%r8, %rax, CHAR_SIZE), %rax + ret + +- /* String end in first loop vector. */ +-L(loop_vector_x1_end): +- KMOV %k1, %VRAX +- BLSMSK %VRCX, %VRCX +- /* Check if it has match in second loop vector. */ +- and %VRCX, %VRAX +- jz L(check_last_match) ++ .p2align 4 ++ /* We can throw away the work done for the first 4x checks here ++ as we have a later match. This is the 'fast' path persay. */ ++L(second_aligned_loop_prep): ++L(second_aligned_loop_set_furthest_match): ++ movq %rdi, %rsi ++ VMOVA %VMM(5), %VMM(7) ++ VMOVA %VMM(6), %VMM(8) ++ .p2align 4 ++L(second_aligned_loop): ++ VMOVU (VEC_SIZE * 4)(%rdi), %VMM(5) ++ VMOVU (VEC_SIZE * 5)(%rdi), %VMM(6) ++ VPCMP $4, %VMM(5), %VMATCH, %k2 ++ VPCMP $4, %VMM(6), %VMATCH, %k3{%k2} ++ ++ VPMIN %VMM(5), %VMM(6), %VMM(4) ++ ++ VPTEST %VMM(4), %VMM(4), %k1{%k3} ++ subq $(VEC_SIZE * -2), %rdi ++ KMOV %k1, %VRCX ++ inc %RCX_M ++ jz L(second_aligned_loop) ++ VPTESTN %VMM(4), %VMM(4), %k1 ++ KMOV %k1, %VRDX ++ test %VRDX, %VRDX ++ jz L(second_aligned_loop_set_furthest_match) + +- BSR %VRAX, %VRAX +- leaq (%rdi, %rax, CHAR_SIZE), %rax +- ret ++ KORTEST_M %k3, %k3 ++ jnc L(return_new_match) ++ /* branch here because there is a significant advantage interms ++ of output dependency chance in using edx. */ + +- /* No match in first and second loop vector. */ +-L(check_last_match): +- /* Check if any match recorded in r8. */ +- test %r8, %r8 +- jz L(vector_x2_ret) +- movq %r8, %rax ++L(return_old_match): ++ VPCMPEQ %VMM(8), %VMATCH, %k0 ++ KMOV %k0, %VRCX ++ bsr %VRCX, %VRCX ++ jnz L(return_old_match_ret) ++ ++ VPCMPEQ %VMM(7), %VMATCH, %k0 ++ KMOV %k0, %VRCX ++ bsr %VRCX, %VRCX ++ subq $VEC_SIZE, %rsi ++L(return_old_match_ret): ++ leaq (VEC_SIZE * 3)(%rsi, %rcx, CHAR_SIZE), %rax + ret + +- /* No match recorded in r8. Check the second saved vector +- in begining. */ +-L(vector_x2_ret): +- VPCMPEQ %VMM(2), %VMM(0), %k2 ++L(return_new_match): ++ VPTESTN %VMM(5), %VMM(5), %k0 ++ KMOV %k0, %VRCX ++ blsmsk %VRCX, %VRCX ++ jnc L(return_new_match_first) ++ dec %VRDX ++ VPCMPEQ %VMM(6), %VMATCH, %k0 ++ KMOV %k0, %VRAX ++ addq $VEC_SIZE, %rdi ++ and %VRDX, %VRAX ++ jnz L(return_new_match_ret) ++ subq $VEC_SIZE, %rdi ++L(return_new_match_first): + KMOV %k2, %VRAX +- test %VRAX, %VRAX +- jz L(vector_x1_ret) +- +- /* Match found in the second saved vector. */ +- BSR %VRAX, %VRAX +- leaq (VEC_SIZE)(%r9, %rax, CHAR_SIZE), %rax ++# ifdef USE_AS_WCSRCHR ++ xorl $((1 << CHAR_PER_VEC)- 1), %VRAX ++ and %VRCX, %VRAX ++# else ++ andn %VRCX, %VRAX, %VRAX ++# endif ++ jz L(return_old_match) ++L(return_new_match_ret): ++ bsr %VRAX, %VRAX ++ leaq (VEC_SIZE * 2)(%rdi, %rax, CHAR_SIZE), %rax + ret + +-L(page_cross): +- mov %rdi, %rax +- movl %edi, %ecx ++ .p2align 4,, 4 ++L(cross_page_boundary): ++ xorq %rdi, %rax ++ mov $-1, %VRDX ++ VMOVU (PAGE_SIZE - VEC_SIZE)(%rax), %VMM(6) ++ VPTESTN %VMM(6), %VMM(6), %k0 ++ KMOV %k0, %VRSI + + # ifdef USE_AS_WCSRCHR +- /* Calculate number of compare result bits to be skipped for +- wide string alignment adjustment. */ +- andl $(VEC_SIZE - 1), %ecx +- sarl $2, %ecx ++ movl %edi, %ecx ++ and $(VEC_SIZE - 1), %ecx ++ shrl $2, %ecx + # endif +- /* ecx contains number of w[char] to be skipped as a result +- of address alignment. */ +- andq $-VEC_SIZE, %rax +- VMOVA (%rax), %VMM(1) +- VPTESTN %VMM(1), %VMM(1), %k1 +- KMOV %k1, %VRAX +- SHR %cl, %VRAX +- jz L(page_cross_continue) +- VPCMPEQ %VMM(1), %VMM(0), %k0 +- KMOV %k0, %VRDX +- SHR %cl, %VRDX +- BLSMSK %VRAX, %VRAX +- and %VRDX, %VRAX +- jz L(ret) +- BSR %VRAX, %VRAX ++ shlx %SHIFT_REG, %VRDX, %VRDX ++ + # ifdef USE_AS_WCSRCHR +- leaq (%rdi, %rax, CHAR_SIZE), %rax ++ kmovw %edx, %k1 + # else +- add %rdi, %rax ++ KMOV %VRDX, %k1 + # endif + +- ret +-END (STRRCHR) ++ VPCOMPRESS %VMM(6), %VMM(1){%k1}{z} ++ /* We could technically just jmp back after the vpcompress but ++ it doesn't save any 16-byte blocks. */ ++ shrx %SHIFT_REG, %VRSI, %VRSI ++ test %VRSI, %VRSI ++ jnz L(page_cross_return) ++ jmp L(page_cross_continue) ++ /* 1-byte from cache line. */ ++END(STRRCHR) + #endif +diff --git a/sysdeps/x86_64/multiarch/strrchr-evex.S b/sysdeps/x86_64/multiarch/strrchr-evex.S +index f5b6d755ceb85ae2..3bf6a5101422e4d1 100644 +--- a/sysdeps/x86_64/multiarch/strrchr-evex.S ++++ b/sysdeps/x86_64/multiarch/strrchr-evex.S +@@ -1,374 +1,8 @@ +-/* strrchr/wcsrchr optimized with 256-bit EVEX instructions. +- Copyright (C) 2021 Free Software Foundation, Inc. +- This file is part of the GNU C Library. +- +- The GNU C Library is free software; you can redistribute it and/or +- modify it under the terms of the GNU Lesser General Public +- License as published by the Free Software Foundation; either +- version 2.1 of the License, or (at your option) any later version. +- +- The GNU C Library is distributed in the hope that it will be useful, +- but WITHOUT ANY WARRANTY; without even the implied warranty of +- MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +- Lesser General Public License for more details. +- +- You should have received a copy of the GNU Lesser General Public +- License along with the GNU C Library; if not, see +- . */ +- +-#if IS_IN (libc) +- +-# include +- + # ifndef STRRCHR + # define STRRCHR __strrchr_evex + # endif + +-# define VMOVU vmovdqu64 +-# define VMOVA vmovdqa64 +- +-# ifdef USE_AS_WCSRCHR +-# define SHIFT_REG esi +- +-# define kunpck kunpckbw +-# define kmov_2x kmovd +-# define maskz_2x ecx +-# define maskm_2x eax +-# define CHAR_SIZE 4 +-# define VPMIN vpminud +-# define VPTESTN vptestnmd +-# define VPBROADCAST vpbroadcastd +-# define VPCMP vpcmpd +-# else +-# define SHIFT_REG edi +- +-# define kunpck kunpckdq +-# define kmov_2x kmovq +-# define maskz_2x rcx +-# define maskm_2x rax +- +-# define CHAR_SIZE 1 +-# define VPMIN vpminub +-# define VPTESTN vptestnmb +-# define VPBROADCAST vpbroadcastb +-# define VPCMP vpcmpb +-# endif +- +-# define XMMZERO xmm16 +-# define YMMZERO ymm16 +-# define YMMMATCH ymm17 +-# define YMMSAVE ymm18 +- +-# define YMM1 ymm19 +-# define YMM2 ymm20 +-# define YMM3 ymm21 +-# define YMM4 ymm22 +-# define YMM5 ymm23 +-# define YMM6 ymm24 +-# define YMM7 ymm25 +-# define YMM8 ymm26 +- +- +-# define VEC_SIZE 32 +-# define PAGE_SIZE 4096 +- .section .text.evex, "ax", @progbits +-ENTRY(STRRCHR) +- movl %edi, %eax +- /* Broadcast CHAR to YMMMATCH. */ +- VPBROADCAST %esi, %YMMMATCH +- +- andl $(PAGE_SIZE - 1), %eax +- cmpl $(PAGE_SIZE - VEC_SIZE), %eax +- jg L(cross_page_boundary) +- +-L(page_cross_continue): +- VMOVU (%rdi), %YMM1 +- /* k0 has a 1 for each zero CHAR in YMM1. */ +- VPTESTN %YMM1, %YMM1, %k0 +- kmovd %k0, %ecx +- testl %ecx, %ecx +- jz L(aligned_more) +- /* fallthrough: zero CHAR in first VEC. */ +- +- /* K1 has a 1 for each search CHAR match in YMM1. */ +- VPCMP $0, %YMMMATCH, %YMM1, %k1 +- kmovd %k1, %eax +- /* Build mask up until first zero CHAR (used to mask of +- potential search CHAR matches past the end of the string). +- */ +- blsmskl %ecx, %ecx +- andl %ecx, %eax +- jz L(ret0) +- /* Get last match (the `andl` removed any out of bounds +- matches). */ +- bsrl %eax, %eax +-# ifdef USE_AS_WCSRCHR +- leaq (%rdi, %rax, CHAR_SIZE), %rax +-# else +- addq %rdi, %rax +-# endif +-L(ret0): +- ret +- +- /* Returns for first vec x1/x2/x3 have hard coded backward +- search path for earlier matches. */ +- .p2align 4,, 6 +-L(first_vec_x1): +- VPCMP $0, %YMMMATCH, %YMM2, %k1 +- kmovd %k1, %eax +- blsmskl %ecx, %ecx +- /* eax non-zero if search CHAR in range. */ +- andl %ecx, %eax +- jnz L(first_vec_x1_return) +- +- /* fallthrough: no match in YMM2 then need to check for earlier +- matches (in YMM1). */ +- .p2align 4,, 4 +-L(first_vec_x0_test): +- VPCMP $0, %YMMMATCH, %YMM1, %k1 +- kmovd %k1, %eax +- testl %eax, %eax +- jz L(ret1) +- bsrl %eax, %eax +-# ifdef USE_AS_WCSRCHR +- leaq (%rsi, %rax, CHAR_SIZE), %rax +-# else +- addq %rsi, %rax +-# endif +-L(ret1): +- ret +- +- .p2align 4,, 10 +-L(first_vec_x1_or_x2): +- VPCMP $0, %YMM3, %YMMMATCH, %k3 +- VPCMP $0, %YMM2, %YMMMATCH, %k2 +- /* K2 and K3 have 1 for any search CHAR match. Test if any +- matches between either of them. Otherwise check YMM1. */ +- kortestd %k2, %k3 +- jz L(first_vec_x0_test) +- +- /* Guranteed that YMM2 and YMM3 are within range so merge the +- two bitmasks then get last result. */ +- kunpck %k2, %k3, %k3 +- kmovq %k3, %rax +- bsrq %rax, %rax +- leaq (VEC_SIZE)(%r8, %rax, CHAR_SIZE), %rax +- ret +- +- .p2align 4,, 6 +-L(first_vec_x3): +- VPCMP $0, %YMMMATCH, %YMM4, %k1 +- kmovd %k1, %eax +- blsmskl %ecx, %ecx +- /* If no search CHAR match in range check YMM1/YMM2/YMM3. */ +- andl %ecx, %eax +- jz L(first_vec_x1_or_x2) +- bsrl %eax, %eax +- leaq (VEC_SIZE * 3)(%rdi, %rax, CHAR_SIZE), %rax +- ret +- +- .p2align 4,, 6 +-L(first_vec_x0_x1_test): +- VPCMP $0, %YMMMATCH, %YMM2, %k1 +- kmovd %k1, %eax +- /* Check YMM2 for last match first. If no match try YMM1. */ +- testl %eax, %eax +- jz L(first_vec_x0_test) +- .p2align 4,, 4 +-L(first_vec_x1_return): +- bsrl %eax, %eax +- leaq (VEC_SIZE)(%rdi, %rax, CHAR_SIZE), %rax +- ret +- +- .p2align 4,, 10 +-L(first_vec_x2): +- VPCMP $0, %YMMMATCH, %YMM3, %k1 +- kmovd %k1, %eax +- blsmskl %ecx, %ecx +- /* Check YMM3 for last match first. If no match try YMM2/YMM1. +- */ +- andl %ecx, %eax +- jz L(first_vec_x0_x1_test) +- bsrl %eax, %eax +- leaq (VEC_SIZE * 2)(%rdi, %rax, CHAR_SIZE), %rax +- ret +- +- +- .p2align 4 +-L(aligned_more): +- /* Need to keep original pointer incase YMM1 has last match. */ +- movq %rdi, %rsi +- andq $-VEC_SIZE, %rdi +- VMOVU VEC_SIZE(%rdi), %YMM2 +- VPTESTN %YMM2, %YMM2, %k0 +- kmovd %k0, %ecx +- testl %ecx, %ecx +- jnz L(first_vec_x1) +- +- VMOVU (VEC_SIZE * 2)(%rdi), %YMM3 +- VPTESTN %YMM3, %YMM3, %k0 +- kmovd %k0, %ecx +- testl %ecx, %ecx +- jnz L(first_vec_x2) +- +- VMOVU (VEC_SIZE * 3)(%rdi), %YMM4 +- VPTESTN %YMM4, %YMM4, %k0 +- kmovd %k0, %ecx +- movq %rdi, %r8 +- testl %ecx, %ecx +- jnz L(first_vec_x3) +- +- andq $-(VEC_SIZE * 2), %rdi +- .p2align 4 +-L(first_aligned_loop): +- /* Preserve YMM1, YMM2, YMM3, and YMM4 until we can gurantee +- they don't store a match. */ +- VMOVA (VEC_SIZE * 4)(%rdi), %YMM5 +- VMOVA (VEC_SIZE * 5)(%rdi), %YMM6 +- +- VPCMP $0, %YMM5, %YMMMATCH, %k2 +- vpxord %YMM6, %YMMMATCH, %YMM7 +- +- VPMIN %YMM5, %YMM6, %YMM8 +- VPMIN %YMM8, %YMM7, %YMM7 +- +- VPTESTN %YMM7, %YMM7, %k1 +- subq $(VEC_SIZE * -2), %rdi +- kortestd %k1, %k2 +- jz L(first_aligned_loop) +- +- VPCMP $0, %YMM6, %YMMMATCH, %k3 +- VPTESTN %YMM8, %YMM8, %k1 +- ktestd %k1, %k1 +- jz L(second_aligned_loop_prep) +- +- kortestd %k2, %k3 +- jnz L(return_first_aligned_loop) +- +- .p2align 4,, 6 +-L(first_vec_x1_or_x2_or_x3): +- VPCMP $0, %YMM4, %YMMMATCH, %k4 +- kmovd %k4, %eax +- testl %eax, %eax +- jz L(first_vec_x1_or_x2) +- bsrl %eax, %eax +- leaq (VEC_SIZE * 3)(%r8, %rax, CHAR_SIZE), %rax +- ret +- +- .p2align 4,, 8 +-L(return_first_aligned_loop): +- VPTESTN %YMM5, %YMM5, %k0 +- kunpck %k0, %k1, %k0 +- kmov_2x %k0, %maskz_2x +- +- blsmsk %maskz_2x, %maskz_2x +- kunpck %k2, %k3, %k3 +- kmov_2x %k3, %maskm_2x +- and %maskz_2x, %maskm_2x +- jz L(first_vec_x1_or_x2_or_x3) +- +- bsr %maskm_2x, %maskm_2x +- leaq (VEC_SIZE * 2)(%rdi, %rax, CHAR_SIZE), %rax +- ret +- +- .p2align 4 +- /* We can throw away the work done for the first 4x checks here +- as we have a later match. This is the 'fast' path persay. +- */ +-L(second_aligned_loop_prep): +-L(second_aligned_loop_set_furthest_match): +- movq %rdi, %rsi +- kunpck %k2, %k3, %k4 +- +- .p2align 4 +-L(second_aligned_loop): +- VMOVU (VEC_SIZE * 4)(%rdi), %YMM1 +- VMOVU (VEC_SIZE * 5)(%rdi), %YMM2 +- +- VPCMP $0, %YMM1, %YMMMATCH, %k2 +- vpxord %YMM2, %YMMMATCH, %YMM3 +- +- VPMIN %YMM1, %YMM2, %YMM4 +- VPMIN %YMM3, %YMM4, %YMM3 +- +- VPTESTN %YMM3, %YMM3, %k1 +- subq $(VEC_SIZE * -2), %rdi +- kortestd %k1, %k2 +- jz L(second_aligned_loop) +- +- VPCMP $0, %YMM2, %YMMMATCH, %k3 +- VPTESTN %YMM4, %YMM4, %k1 +- ktestd %k1, %k1 +- jz L(second_aligned_loop_set_furthest_match) +- +- kortestd %k2, %k3 +- /* branch here because there is a significant advantage interms +- of output dependency chance in using edx. */ +- jnz L(return_new_match) +-L(return_old_match): +- kmovq %k4, %rax +- bsrq %rax, %rax +- leaq (VEC_SIZE * 2)(%rsi, %rax, CHAR_SIZE), %rax +- ret +- +-L(return_new_match): +- VPTESTN %YMM1, %YMM1, %k0 +- kunpck %k0, %k1, %k0 +- kmov_2x %k0, %maskz_2x +- +- blsmsk %maskz_2x, %maskz_2x +- kunpck %k2, %k3, %k3 +- kmov_2x %k3, %maskm_2x +- and %maskz_2x, %maskm_2x +- jz L(return_old_match) +- +- bsr %maskm_2x, %maskm_2x +- leaq (VEC_SIZE * 2)(%rdi, %rax, CHAR_SIZE), %rax +- ret +- +-L(cross_page_boundary): +- /* eax contains all the page offset bits of src (rdi). `xor rdi, +- rax` sets pointer will all page offset bits cleared so +- offset of (PAGE_SIZE - VEC_SIZE) will get last aligned VEC +- before page cross (guranteed to be safe to read). Doing this +- as opposed to `movq %rdi, %rax; andq $-VEC_SIZE, %rax` saves +- a bit of code size. */ +- xorq %rdi, %rax +- VMOVU (PAGE_SIZE - VEC_SIZE)(%rax), %YMM1 +- VPTESTN %YMM1, %YMM1, %k0 +- kmovd %k0, %ecx +- +- /* Shift out zero CHAR matches that are before the begining of +- src (rdi). */ +-# ifdef USE_AS_WCSRCHR +- movl %edi, %esi +- andl $(VEC_SIZE - 1), %esi +- shrl $2, %esi +-# endif +- shrxl %SHIFT_REG, %ecx, %ecx +- +- testl %ecx, %ecx +- jz L(page_cross_continue) +- +- /* Found zero CHAR so need to test for search CHAR. */ +- VPCMP $0, %YMMMATCH, %YMM1, %k1 +- kmovd %k1, %eax +- /* Shift out search CHAR matches that are before the begining of +- src (rdi). */ +- shrxl %SHIFT_REG, %eax, %eax +- +- /* Check if any search CHAR match in range. */ +- blsmskl %ecx, %ecx +- andl %ecx, %eax +- jz L(ret3) +- bsrl %eax, %eax +-# ifdef USE_AS_WCSRCHR +- leaq (%rdi, %rax, CHAR_SIZE), %rax +-# else +- addq %rdi, %rax +-# endif +-L(ret3): +- ret ++#include "x86-evex256-vecs.h" ++#include "reg-macros.h" + +-END(STRRCHR) +-#endif ++#include "strrchr-evex-base.S" +diff --git a/sysdeps/x86_64/multiarch/wcsrchr-evex.S b/sysdeps/x86_64/multiarch/wcsrchr-evex.S +index c64602f7dc154ad6..7bd265a24977f339 100644 +--- a/sysdeps/x86_64/multiarch/wcsrchr-evex.S ++++ b/sysdeps/x86_64/multiarch/wcsrchr-evex.S +@@ -1,3 +1,4 @@ + #define STRRCHR __wcsrchr_evex + #define USE_AS_WCSRCHR 1 ++#define USE_WIDE_CHAR 1 + #include "strrchr-evex.S" diff --git a/glibc-RHEL-175520-11.patch b/glibc-RHEL-175520-11.patch new file mode 100644 index 0000000..56ab246 --- /dev/null +++ b/glibc-RHEL-175520-11.patch @@ -0,0 +1,157 @@ +commit b7f8b6b64b135e6dba5083d688675874d6809c91 +Author: Noah Goldstein +Date: Wed Nov 1 16:42:07 2023 -0500 + + x86: Fix unchecked AVX512-VBMI2 usage in strrchr-evex-base.S + + strrchr-evex-base used `vpcompress{b|d}` in the page cross logic but + was missing the CPU_FEATURE checks for VBMI2 in the + ifunc/ifunc-impl-list. + + The fix is either to add those checks or change the logic to not use + `vpcompress{b|d}`. Choosing the latter here so that the strrchr-evex + implementation is usable on SKX. + + New implementation is a bit slower, but this is in a cold path so its + probably okay. + +diff --git a/sysdeps/x86_64/multiarch/strrchr-evex-base.S b/sysdeps/x86_64/multiarch/strrchr-evex-base.S +index 8cb38be69fef5a45..2c1db616294bdc2c 100644 +--- a/sysdeps/x86_64/multiarch/strrchr-evex-base.S ++++ b/sysdeps/x86_64/multiarch/strrchr-evex-base.S +@@ -35,18 +35,20 @@ + # define CHAR_SIZE 4 + # define VPCMP vpcmpd + # define VPMIN vpminud +-# define VPCOMPRESS vpcompressd + # define VPTESTN vptestnmd + # define VPTEST vptestmd + # define VPBROADCAST vpbroadcastd + # define VPCMPEQ vpcmpeqd + + # else +-# define SHIFT_REG VRDI ++# if VEC_SIZE == 64 ++# define SHIFT_REG VRCX ++# else ++# define SHIFT_REG VRDI ++# endif + # define CHAR_SIZE 1 + # define VPCMP vpcmpb + # define VPMIN vpminub +-# define VPCOMPRESS vpcompressb + # define VPTESTN vptestnmb + # define VPTEST vptestmb + # define VPBROADCAST vpbroadcastb +@@ -56,6 +58,12 @@ + # define KORTEST_M KORTEST + # endif + ++# if VEC_SIZE == 32 || (defined USE_AS_WCSRCHR) ++# define SHIFT_R(cnt, val) shrx cnt, val, val ++# else ++# define SHIFT_R(cnt, val) shr %cl, val ++# endif ++ + # define VMATCH VMM(0) + # define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) + # define PAGE_SIZE 4096 +@@ -71,7 +79,7 @@ ENTRY_P2ALIGN(STRRCHR, 6) + andl $(PAGE_SIZE - 1), %eax + cmpl $(PAGE_SIZE - VEC_SIZE), %eax + jg L(cross_page_boundary) +- ++L(page_cross_continue): + VMOVU (%rdi), %VMM(1) + /* k0 has a 1 for each zero CHAR in YMM1. */ + VPTESTN %VMM(1), %VMM(1), %k0 +@@ -79,7 +87,7 @@ ENTRY_P2ALIGN(STRRCHR, 6) + test %VGPR(rsi), %VGPR(rsi) + jz L(aligned_more) + /* fallthrough: zero CHAR in first VEC. */ +-L(page_cross_return): ++ + /* K1 has a 1 for each search CHAR match in VEC(1). */ + VPCMPEQ %VMATCH, %VMM(1), %k1 + KMOV %k1, %VGPR(rax) +@@ -167,7 +175,6 @@ L(first_vec_x1_return): + + .p2align 4,, 12 + L(aligned_more): +-L(page_cross_continue): + /* Need to keep original pointer incase VEC(1) has last match. */ + movq %rdi, %rsi + andq $-VEC_SIZE, %rdi +@@ -340,34 +347,54 @@ L(return_new_match_ret): + leaq (VEC_SIZE * 2)(%rdi, %rax, CHAR_SIZE), %rax + ret + +- .p2align 4,, 4 + L(cross_page_boundary): ++ /* eax contains all the page offset bits of src (rdi). `xor rdi, ++ rax` sets pointer will all page offset bits cleared so ++ offset of (PAGE_SIZE - VEC_SIZE) will get last aligned VEC ++ before page cross (guaranteed to be safe to read). Doing this ++ as opposed to `movq %rdi, %rax; andq $-VEC_SIZE, %rax` saves ++ a bit of code size. */ + xorq %rdi, %rax +- mov $-1, %VRDX +- VMOVU (PAGE_SIZE - VEC_SIZE)(%rax), %VMM(6) +- VPTESTN %VMM(6), %VMM(6), %k0 ++ VMOVU (PAGE_SIZE - VEC_SIZE)(%rax), %VMM(1) ++ VPTESTN %VMM(1), %VMM(1), %k0 + KMOV %k0, %VRSI + +-# ifdef USE_AS_WCSRCHR ++ /* Shift out zero CHAR matches that are before the beginning of ++ src (rdi). */ ++# if VEC_SIZE == 64 || (defined USE_AS_WCSRCHR) + movl %edi, %ecx +- and $(VEC_SIZE - 1), %ecx ++# endif ++# ifdef USE_AS_WCSRCHR ++ andl $(VEC_SIZE - 1), %ecx + shrl $2, %ecx + # endif +- shlx %SHIFT_REG, %VRDX, %VRDX ++ SHIFT_R (%SHIFT_REG, %VRSI) ++# if VEC_SIZE == 32 || (defined USE_AS_WCSRCHR) ++ /* For strrchr-evex512 we use SHIFT_R as shr which will set zero ++ flag. */ ++ test %VRSI, %VRSI ++# endif ++ jz L(page_cross_continue) + ++ /* Found zero CHAR so need to test for search CHAR. */ ++ VPCMPEQ %VMATCH, %VMM(1), %k1 ++ KMOV %k1, %VRAX ++ /* Shift out search CHAR matches that are before the beginning of ++ src (rdi). */ ++ SHIFT_R (%SHIFT_REG, %VRAX) ++ /* Check if any search CHAR match in range. */ ++ blsmsk %VRSI, %VRSI ++ and %VRSI, %VRAX ++ jz L(ret2) ++ bsr %VRAX, %VRAX + # ifdef USE_AS_WCSRCHR +- kmovw %edx, %k1 ++ leaq (%rdi, %rax, CHAR_SIZE), %rax + # else +- KMOV %VRDX, %k1 ++ addq %rdi, %rax + # endif +- +- VPCOMPRESS %VMM(6), %VMM(1){%k1}{z} +- /* We could technically just jmp back after the vpcompress but +- it doesn't save any 16-byte blocks. */ +- shrx %SHIFT_REG, %VRSI, %VRSI +- test %VRSI, %VRSI +- jnz L(page_cross_return) +- jmp L(page_cross_continue) +- /* 1-byte from cache line. */ ++L(ret2): ++ ret ++ /* 3 bytes from cache-line for evex. */ ++ /* 0 bytes from cache-line for evex512. */ + END(STRRCHR) + #endif diff --git a/glibc-RHEL-175520-12.patch b/glibc-RHEL-175520-12.patch new file mode 100644 index 0000000..980e401 --- /dev/null +++ b/glibc-RHEL-175520-12.patch @@ -0,0 +1,1276 @@ +commit e957308723ac2e55dad360d602298632980bbd38 +Author: Matthew Sterrett +Date: Fri Dec 15 12:04:05 2023 -0800 + + x86: Unifies 'strlen-evex' and 'strlen-evex512' implementations. + + This commit uses a common implementation 'strlen-evex-base.S' for both + 'strlen-evex' and 'strlen-evex512' + + The motivation is to reduce the number of implementations to maintain. + This incidentally gives a small performance improvement. + + All tests pass on x86. + + Benchmarks were taken on SKX. + https://www.intel.com/content/www/us/en/products/sku/123613/intel-core-i97900x-xseries-processor-13-75m-cache-up-to-4-30-ghz/specifications.html + + Geometric mean for strlen-evex512 over all benchmarks (N=10) was (new/old) 0.939 + Geometric mean for wcslen-evex512 over all benchmarks (N=10) was (new/old) 0.965 + + Code Size Changes: + strlen-evex512.S : +24 bytes + wcslen-evex512.S : +54 bytes + + Reviewed-by: Noah Goldstein + +Conflicts: + sysdeps/x86_64/multiarch/strlen-evex-base.S + (adapt to missing ceabdcd130ca7043b0fcf2676183d79431d10493) + sysdeps/x86_64/multiarch/strlen-evex.S + (adapt downstream) + +diff --git a/sysdeps/x86_64/multiarch/strlen-evex-base.S b/sysdeps/x86_64/multiarch/strlen-evex-base.S +index 742ac9c9253ecb60..c3fda529d5687f13 100644 +--- a/sysdeps/x86_64/multiarch/strlen-evex-base.S ++++ b/sysdeps/x86_64/multiarch/strlen-evex-base.S +@@ -1,5 +1,5 @@ +-/* Placeholder function, not used by any processor at the moment. +- Copyright (C) 2022 Free Software Foundation, Inc. ++/* strlen/wcslen optimized with 256/512-bit EVEX instructions. ++ Copyright (C) 2021-2023 Free Software Foundation, Inc. + This file is part of the GNU C Library. + + The GNU C Library is free software; you can redistribute it and/or +@@ -22,272 +22,211 @@ + + # ifdef USE_AS_WCSLEN + # define VPCMPEQ vpcmpeqd ++# define VPCMPNEQ vpcmpneqd + # define VPTESTN vptestnmd ++# define VPTEST vptestmd + # define VPMINU vpminud + # define CHAR_SIZE 4 ++# define CHAR_SIZE_SHIFT_REG(reg) sar $2, %reg + # else + # define VPCMPEQ vpcmpeqb ++# define VPCMPNEQ vpcmpneqb + # define VPTESTN vptestnmb ++# define VPTEST vptestmb + # define VPMINU vpminub + # define CHAR_SIZE 1 ++# define CHAR_SIZE_SHIFT_REG(reg) ++ ++# define REG_WIDTH VEC_SIZE + # endif + +-# define PAGE_SIZE 4096 + # define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) + +- .section SECTION(.text),"ax",@progbits +-/* Aligning entry point to 64 byte, provides better performance for +- one vector length string. */ +-ENTRY_P2ALIGN (STRLEN, 6) +-# ifdef USE_AS_STRNLEN +- /* Check zero length. */ +- test %RSI_LP, %RSI_LP +- jz L(ret_max) +-# ifdef __ILP32__ +- /* Clear the upper 32 bits. */ +- movl %esi, %esi +-# endif ++# include "reg-macros.h" ++ ++# if CHAR_PER_VEC == 64 ++ ++# define TAIL_RETURN_LBL first_vec_x2 ++# define TAIL_RETURN_OFFSET (CHAR_PER_VEC * 2) ++ ++# define FALLTHROUGH_RETURN_LBL first_vec_x3 ++# define FALLTHROUGH_RETURN_OFFSET (CHAR_PER_VEC * 3) ++ ++# else ++ ++# define TAIL_RETURN_LBL first_vec_x3 ++# define TAIL_RETURN_OFFSET (CHAR_PER_VEC * 3) ++ ++# define FALLTHROUGH_RETURN_LBL first_vec_x2 ++# define FALLTHROUGH_RETURN_OFFSET (CHAR_PER_VEC * 2) + # endif + ++# define XZERO VMM_128(0) ++# define VZERO VMM(0) ++# define PAGE_SIZE 4096 ++ ++ .section SECTION(.text), "ax", @progbits ++ENTRY_P2ALIGN(STRLEN, 6) + movl %edi, %eax +- vpxorq %VMM_128(0), %VMM_128(0), %VMM_128(0) +- sall $20, %eax +- cmpl $((PAGE_SIZE - VEC_SIZE) << 20), %eax +- ja L(page_cross) +- +- /* Compare [w]char for null, mask bit will be set for match. */ +- VPCMPEQ (%rdi), %VMM(0), %k0 +-# ifdef USE_AS_STRNLEN +- KMOV %k0, %VRCX +- /* Store max length in rax. */ +- mov %rsi, %rax +- /* If rcx is 0, rax will have max length. We can not use VRCX +- and VRAX here for evex256 because, upper 32 bits may be +- undefined for ecx and eax. */ +- bsfq %rcx, %rax +- cmp $CHAR_PER_VEC, %rax +- ja L(align_more) +- cmpq %rax, %rsi +- cmovb %esi, %eax +-# else ++ vpxorq %XZERO, %XZERO, %XZERO ++ andl $(PAGE_SIZE - 1), %eax ++ cmpl $(PAGE_SIZE - VEC_SIZE), %eax ++ ja L(cross_page_boundary) ++ ++ /* Check the first VEC_SIZE bytes. Each bit in K0 represents a ++ null byte. */ ++ VPCMPEQ (%rdi), %VZERO, %k0 + KMOV %k0, %VRAX + test %VRAX, %VRAX +- jz L(align_more) ++ jz L(aligned_more) + bsf %VRAX, %VRAX +-# endif + ret + +- /* At this point vector max length reached. */ +-# ifdef USE_AS_STRNLEN +- .p2align 4,,3 +-L(ret_max): +- movq %rsi, %rax ++ .p2align 4,, 8 ++L(first_vec_x4): ++ bsf %VRAX, %VRAX ++ subl %ecx, %edi ++ CHAR_SIZE_SHIFT_REG (edi) ++ leal (CHAR_PER_VEC * 4)(%rdi, %rax), %eax + ret +-# endif + +-L(align_more): +- mov %rdi, %rax +- /* Align rax to VEC_SIZE. */ +- andq $-VEC_SIZE, %rax +-# ifdef USE_AS_STRNLEN +- movq %rdi, %rdx +- subq %rax, %rdx +-# ifdef USE_AS_WCSLEN +- shr $2, %VRDX +-# endif +- /* At this point rdx contains [w]chars already compared. */ +- leaq -CHAR_PER_VEC(%rsi, %rdx), %rdx +- /* At this point rdx contains number of w[char] needs to go. +- Now onwards rdx will keep decrementing with each compare. */ +-# endif +- +- /* Loop unroll 4 times for 4 vector loop. */ +- VPCMPEQ VEC_SIZE(%rax), %VMM(0), %k0 +- subq $-VEC_SIZE, %rax +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x1) + +-# ifdef USE_AS_STRNLEN +- subq $CHAR_PER_VEC, %rdx +- jbe L(ret_max) +-# endif + +- VPCMPEQ VEC_SIZE(%rax), %VMM(0), %k0 +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x2) ++ /* Aligned more for strnlen compares remaining length vs 2 * ++ CHAR_PER_VEC, 4 * CHAR_PER_VEC, and 8 * CHAR_PER_VEC before ++ going to the loop. */ ++ .p2align 4,, 10 ++L(aligned_more): ++ movq %rdi, %rcx ++ andq $(VEC_SIZE * -1), %rdi ++L(cross_page_continue): ++ /* Remaining length >= 2 * CHAR_PER_VEC so do VEC0/VEC1 without ++ rechecking bounds. */ ++ VPCMPEQ (VEC_SIZE * 1)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(first_vec_x1) + +-# ifdef USE_AS_STRNLEN +- subq $CHAR_PER_VEC, %rdx +- jbe L(ret_max) +-# endif ++ VPCMPEQ (VEC_SIZE * 2)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(first_vec_x2) + +- VPCMPEQ (VEC_SIZE * 2)(%rax), %VMM(0), %k0 +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x3) ++ VPCMPEQ (VEC_SIZE * 3)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(first_vec_x3) + +-# ifdef USE_AS_STRNLEN +- subq $CHAR_PER_VEC, %rdx +- jbe L(ret_max) +-# endif ++ VPCMPEQ (VEC_SIZE * 4)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(first_vec_x4) + +- VPCMPEQ (VEC_SIZE * 3)(%rax), %VMM(0), %k0 +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x4) ++ subq $(VEC_SIZE * -1), %rdi + +-# ifdef USE_AS_STRNLEN +- subq $CHAR_PER_VEC, %rdx +- jbe L(ret_max) +- /* Save pointer before 4 x VEC_SIZE alignment. */ +- movq %rax, %rcx ++# if CHAR_PER_VEC == 64 ++ /* No partial register stalls on processors that we use evex512 ++ on and this saves code size. */ ++ xorb %dil, %dil ++# else ++ andq $-(VEC_SIZE * 4), %rdi + # endif + +- /* Align address to VEC_SIZE * 4 for loop. */ +- andq $-(VEC_SIZE * 4), %rax +- +-# ifdef USE_AS_STRNLEN +- subq %rax, %rcx +-# ifdef USE_AS_WCSLEN +- shr $2, %VRCX +-# endif +- /* rcx contains number of [w]char will be recompared due to +- alignment fixes. rdx must be incremented by rcx to offset +- alignment adjustment. */ +- addq %rcx, %rdx +- /* Need jump as we don't want to add/subtract rdx for first +- iteration of 4 x VEC_SIZE aligned loop. */ +-# endif + +- .p2align 4,,11 +-L(loop): +- /* VPMINU and VPCMP combination provide better performance as +- compared to alternative combinations. */ +- VMOVA (VEC_SIZE * 4)(%rax), %VMM(1) +- VPMINU (VEC_SIZE * 5)(%rax), %VMM(1), %VMM(2) +- VMOVA (VEC_SIZE * 6)(%rax), %VMM(3) +- VPMINU (VEC_SIZE * 7)(%rax), %VMM(3), %VMM(4) + ++ /* Compare 4 * VEC at a time forward. */ ++ .p2align 4 ++L(loop_4x_vec): ++ VMOVA (VEC_SIZE * 4)(%rdi), %VMM(1) ++ VPMINU (VEC_SIZE * 5)(%rdi), %VMM(1), %VMM(2) ++ VMOVA (VEC_SIZE * 6)(%rdi), %VMM(3) ++ VPMINU (VEC_SIZE * 7)(%rdi), %VMM(3), %VMM(4) + VPTESTN %VMM(2), %VMM(2), %k0 +- VPTESTN %VMM(4), %VMM(4), %k1 ++ VPTESTN %VMM(4), %VMM(4), %k2 + +- subq $-(VEC_SIZE * 4), %rax +- KORTEST %k0, %k1 ++ subq $-(VEC_SIZE * 4), %rdi ++ KORTEST %k0, %k2 ++ jz L(loop_4x_vec) + +-# ifndef USE_AS_STRNLEN +- jz L(loop) ++ VPTESTN %VMM(1), %VMM(1), %k1 ++ KMOV %k1, %VRAX ++ test %VRAX, %VRAX ++ jnz L(first_vec_x0) ++ ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(first_vec_x1) ++ ++ VPTESTN %VMM(3), %VMM(3), %k0 ++ ++# if CHAR_PER_VEC == 64 ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(first_vec_x2) ++ KMOV %k2, %VRAX + # else +- jnz L(loopend) +- subq $(CHAR_PER_VEC * 4), %rdx +- ja L(loop) +- mov %rsi, %rax ++ /* We can only combine last 2x VEC masks if CHAR_PER_VEC <= 32. */ ++ kmovd %k2, %edx ++ kmovd %k0, %eax ++ salq $CHAR_PER_VEC, %rdx ++ orq %rdx, %rax ++# endif ++ ++ /* first_vec_x3 for strlen-ZMM and first_vec_x2 for strlen-YMM. */ ++ .p2align 4,, 2 ++L(FALLTHROUGH_RETURN_LBL): ++ bsfq %rax, %rax ++ subq %rcx, %rdi ++ CHAR_SIZE_SHIFT_REG (rdi) ++ leaq (FALLTHROUGH_RETURN_OFFSET)(%rdi, %rax), %rax + ret +-# endif + +-L(loopend): +- +- VPTESTN %VMM(1), %VMM(1), %k2 +- KMOV %k2, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x1) +- +- KMOV %k0, %VRCX +- /* At this point, if k0 is non zero, null char must be in the +- second vector. */ +- test %VRCX, %VRCX +- jnz L(ret_vec_x2) +- +- VPTESTN %VMM(3), %VMM(3), %k3 +- KMOV %k3, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x3) +- /* At this point null [w]char must be in the fourth vector so no +- need to check. */ +- KMOV %k1, %VRCX +- +- /* Fourth, third, second vector terminating are pretty much +- same, implemented this way to avoid branching and reuse code +- from pre loop exit condition. */ +-L(ret_vec_x4): +- bsf %VRCX, %VRCX +- subq %rdi, %rax +-# ifdef USE_AS_WCSLEN +- subq $-(VEC_SIZE * 3), %rax +- shrq $2, %rax +- addq %rcx, %rax +-# else +- leaq (VEC_SIZE * 3)(%rcx, %rax), %rax +-# endif +-# ifdef USE_AS_STRNLEN +- cmpq %rsi, %rax +- cmovnb %rsi, %rax +-# endif ++ .p2align 4,, 8 ++L(first_vec_x0): ++ bsf %VRAX, %VRAX ++ sub %rcx, %rdi ++ CHAR_SIZE_SHIFT_REG (rdi) ++ addq %rdi, %rax + ret + +-L(ret_vec_x3): +- bsf %VRCX, %VRCX +- subq %rdi, %rax +-# ifdef USE_AS_WCSLEN +- subq $-(VEC_SIZE * 2), %rax +- shrq $2, %rax +- addq %rcx, %rax +-# else +- leaq (VEC_SIZE * 2)(%rcx, %rax), %rax +-# endif +-# ifdef USE_AS_STRNLEN +- cmpq %rsi, %rax +- cmovnb %rsi, %rax +-# endif ++ .p2align 4,, 10 ++L(first_vec_x1): ++ bsf %VRAX, %VRAX ++ sub %rcx, %rdi ++ CHAR_SIZE_SHIFT_REG (rdi) ++ leaq (CHAR_PER_VEC)(%rdi, %rax), %rax + ret + +-L(ret_vec_x2): +- subq $-VEC_SIZE, %rax +-L(ret_vec_x1): +- bsf %VRCX, %VRCX +- subq %rdi, %rax +-# ifdef USE_AS_WCSLEN +- shrq $2, %rax +-# endif +- addq %rcx, %rax +-# ifdef USE_AS_STRNLEN +- cmpq %rsi, %rax +- cmovnb %rsi, %rax +-# endif ++ .p2align 4,, 10 ++ /* first_vec_x2 for strlen-ZMM and first_vec_x3 for strlen-YMM. */ ++L(TAIL_RETURN_LBL): ++ bsf %VRAX, %VRAX ++ sub %VRCX, %VRDI ++ CHAR_SIZE_SHIFT_REG (VRDI) ++ lea (TAIL_RETURN_OFFSET)(%rdi, %rax), %VRAX + ret + +-L(page_cross): +- mov %rdi, %rax +- movl %edi, %ecx +- andl $(VEC_SIZE - 1), %ecx ++ .p2align 4,, 8 ++L(cross_page_boundary): ++ movq %rdi, %rcx ++ /* Align data to VEC_SIZE. */ ++ andq $-VEC_SIZE, %rdi ++ ++ VPCMPEQ (%rdi), %VZERO, %k0 ++ ++ KMOV %k0, %VRAX + # ifdef USE_AS_WCSLEN +- sarl $2, %ecx +-# endif +- /* ecx contains number of w[char] to be skipped as a result +- of address alignment. */ +- andq $-VEC_SIZE, %rax +- VPCMPEQ (%rax), %VMM(0), %k0 +- KMOV %k0, %VRDX +- /* Ignore number of character for alignment adjustment. */ +- shr %cl, %VRDX +-# ifdef USE_AS_STRNLEN +- jnz L(page_cross_end) +- movl $CHAR_PER_VEC, %eax +- sub %ecx, %eax +- cmp %rax, %rsi +- ja L(align_more) ++ movl %ecx, %edx ++ shrl $2, %edx ++ andl $(CHAR_PER_VEC - 1), %edx ++ shrx %edx, %eax, %eax ++ testl %eax, %eax + # else +- jz L(align_more) +-# endif +- +-L(page_cross_end): +- bsf %VRDX, %VRAX +-# ifdef USE_AS_STRNLEN +- cmpq %rsi, %rax +- cmovnb %esi, %eax ++ shr %cl, %VRAX + # endif ++ jz L(cross_page_continue) ++ bsf %VRAX, %VRAX + ret + +-END (STRLEN) ++END(STRLEN) + #endif +diff --git a/sysdeps/x86_64/multiarch/strlen-evex.S b/sysdeps/x86_64/multiarch/strlen-evex.S +index 4bf6874b82335519..93ad15e3561937e4 100644 +--- a/sysdeps/x86_64/multiarch/strlen-evex.S ++++ b/sysdeps/x86_64/multiarch/strlen-evex.S +@@ -1,489 +1,7 @@ +-/* strlen/strnlen/wcslen/wcsnlen optimized with 256-bit EVEX instructions. +- Copyright (C) 2021 Free Software Foundation, Inc. +- This file is part of the GNU C Library. +- +- The GNU C Library is free software; you can redistribute it and/or +- modify it under the terms of the GNU Lesser General Public +- License as published by the Free Software Foundation; either +- version 2.1 of the License, or (at your option) any later version. +- +- The GNU C Library is distributed in the hope that it will be useful, +- but WITHOUT ANY WARRANTY; without even the implied warranty of +- MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +- Lesser General Public License for more details. +- +- You should have received a copy of the GNU Lesser General Public +- License along with the GNU C Library; if not, see +- . */ +- +-#if IS_IN (libc) +- +-# include +- +-# ifndef STRLEN +-# define STRLEN __strlen_evex +-# endif +- +-# define VMOVA vmovdqa64 +- +-# ifdef USE_AS_WCSLEN +-# define VPCMP vpcmpd +-# define VPMINU vpminud +-# define SHIFT_REG ecx +-# define CHAR_SIZE 4 +-# else +-# define VPCMP vpcmpb +-# define VPMINU vpminub +-# define SHIFT_REG edx +-# define CHAR_SIZE 1 +-# endif +- +-# define XMMZERO xmm16 +-# define YMMZERO ymm16 +-# define YMM1 ymm17 +-# define YMM2 ymm18 +-# define YMM3 ymm19 +-# define YMM4 ymm20 +-# define YMM5 ymm21 +-# define YMM6 ymm22 +- +-# define VEC_SIZE 32 +-# define PAGE_SIZE 4096 +-# define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) +- +- .section .text.evex,"ax",@progbits +-ENTRY (STRLEN) +-# ifdef USE_AS_STRNLEN +- /* Check zero length. */ +- test %RSI_LP, %RSI_LP +- jz L(zero) +-# ifdef __ILP32__ +- /* Clear the upper 32 bits. */ +- movl %esi, %esi +-# endif +- mov %RSI_LP, %R8_LP +-# endif +- movl %edi, %eax +- vpxorq %XMMZERO, %XMMZERO, %XMMZERO +- /* Clear high bits from edi. Only keeping bits relevant to page +- cross check. */ +- andl $(PAGE_SIZE - 1), %eax +- /* Check if we may cross page boundary with one vector load. */ +- cmpl $(PAGE_SIZE - VEC_SIZE), %eax +- ja L(cross_page_boundary) +- +- /* Check the first VEC_SIZE bytes. Each bit in K0 represents a +- null byte. */ +- VPCMP $0, (%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +-# ifdef USE_AS_STRNLEN +- /* If length < CHAR_PER_VEC handle special. */ +- cmpq $CHAR_PER_VEC, %rsi +- jbe L(first_vec_x0) +-# endif +- testl %eax, %eax +- jz L(aligned_more) +- tzcntl %eax, %eax +- ret +-# ifdef USE_AS_STRNLEN +-L(zero): +- xorl %eax, %eax +- ret +- +- .p2align 4 +-L(first_vec_x0): +- /* Set bit for max len so that tzcnt will return min of max len +- and position of first match. */ +- btsq %rsi, %rax +- tzcntl %eax, %eax +- ret +-# endif +- +- .p2align 4 +-L(first_vec_x1): +- tzcntl %eax, %eax +- /* Safe to use 32 bit instructions as these are only called for +- size = [1, 159]. */ +-# ifdef USE_AS_STRNLEN +- /* Use ecx which was computed earlier to compute correct value. +- */ +- leal -(CHAR_PER_VEC * 4 + 1)(%rcx, %rax), %eax +-# else +- subl %edx, %edi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarl $2, %edi +-# endif +- leal CHAR_PER_VEC(%rdi, %rax), %eax +-# endif +- ret +- +- .p2align 4 +-L(first_vec_x2): +- tzcntl %eax, %eax +- /* Safe to use 32 bit instructions as these are only called for +- size = [1, 159]. */ +-# ifdef USE_AS_STRNLEN +- /* Use ecx which was computed earlier to compute correct value. +- */ +- leal -(CHAR_PER_VEC * 3 + 1)(%rcx, %rax), %eax +-# else +- subl %edx, %edi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarl $2, %edi +-# endif +- leal (CHAR_PER_VEC * 2)(%rdi, %rax), %eax +-# endif +- ret +- +- .p2align 4 +-L(first_vec_x3): +- tzcntl %eax, %eax +- /* Safe to use 32 bit instructions as these are only called for +- size = [1, 159]. */ +-# ifdef USE_AS_STRNLEN +- /* Use ecx which was computed earlier to compute correct value. +- */ +- leal -(CHAR_PER_VEC * 2 + 1)(%rcx, %rax), %eax +-# else +- subl %edx, %edi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarl $2, %edi +-# endif +- leal (CHAR_PER_VEC * 3)(%rdi, %rax), %eax +-# endif +- ret +- +- .p2align 4 +-L(first_vec_x4): +- tzcntl %eax, %eax +- /* Safe to use 32 bit instructions as these are only called for +- size = [1, 159]. */ +-# ifdef USE_AS_STRNLEN +- /* Use ecx which was computed earlier to compute correct value. +- */ +- leal -(CHAR_PER_VEC + 1)(%rcx, %rax), %eax +-# else +- subl %edx, %edi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarl $2, %edi +-# endif +- leal (CHAR_PER_VEC * 4)(%rdi, %rax), %eax +-# endif +- ret +- +- .p2align 5 +-L(aligned_more): +- movq %rdi, %rdx +- /* Align data to VEC_SIZE. */ +- andq $-(VEC_SIZE), %rdi +-L(cross_page_continue): +- /* Check the first 4 * VEC_SIZE. Only one VEC_SIZE at a time +- since data is only aligned to VEC_SIZE. */ +-# ifdef USE_AS_STRNLEN +- /* + CHAR_SIZE because it simplies the logic in +- last_4x_vec_or_less. */ +- leaq (VEC_SIZE * 5 + CHAR_SIZE)(%rdi), %rcx +- subq %rdx, %rcx +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarl $2, %ecx +-# endif +-# endif +- /* Load first VEC regardless. */ +- VPCMP $0, VEC_SIZE(%rdi), %YMMZERO, %k0 +-# ifdef USE_AS_STRNLEN +- /* Adjust length. If near end handle specially. */ +- subq %rcx, %rsi +- jb L(last_4x_vec_or_less) +-# endif +- kmovd %k0, %eax +- testl %eax, %eax +- jnz L(first_vec_x1) +- +- VPCMP $0, (VEC_SIZE * 2)(%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +- test %eax, %eax +- jnz L(first_vec_x2) +- +- VPCMP $0, (VEC_SIZE * 3)(%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +- testl %eax, %eax +- jnz L(first_vec_x3) +- +- VPCMP $0, (VEC_SIZE * 4)(%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +- testl %eax, %eax +- jnz L(first_vec_x4) +- +- addq $VEC_SIZE, %rdi +-# ifdef USE_AS_STRNLEN +- /* Check if at last VEC_SIZE * 4 length. */ +- cmpq $(CHAR_PER_VEC * 4 - 1), %rsi +- jbe L(last_4x_vec_or_less_load) +- movl %edi, %ecx +- andl $(VEC_SIZE * 4 - 1), %ecx +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarl $2, %ecx +-# endif +- /* Readjust length. */ +- addq %rcx, %rsi +-# endif +- /* Align data to VEC_SIZE * 4. */ +- andq $-(VEC_SIZE * 4), %rdi +- +- /* Compare 4 * VEC at a time forward. */ +- .p2align 4 +-L(loop_4x_vec): +- /* Load first VEC regardless. */ +- VMOVA (VEC_SIZE * 4)(%rdi), %YMM1 +-# ifdef USE_AS_STRNLEN +- /* Break if at end of length. */ +- subq $(CHAR_PER_VEC * 4), %rsi +- jb L(last_4x_vec_or_less_cmpeq) +-# endif +- /* Save some code size by microfusing VPMINU with the load. Since +- the matches in ymm2/ymm4 can only be returned if there where no +- matches in ymm1/ymm3 respectively there is no issue with overlap. +- */ +- VPMINU (VEC_SIZE * 5)(%rdi), %YMM1, %YMM2 +- VMOVA (VEC_SIZE * 6)(%rdi), %YMM3 +- VPMINU (VEC_SIZE * 7)(%rdi), %YMM3, %YMM4 +- +- VPCMP $0, %YMM2, %YMMZERO, %k0 +- VPCMP $0, %YMM4, %YMMZERO, %k1 +- subq $-(VEC_SIZE * 4), %rdi +- kortestd %k0, %k1 +- jz L(loop_4x_vec) +- +- /* Check if end was in first half. */ +- kmovd %k0, %eax +- subq %rdx, %rdi +-# ifdef USE_AS_WCSLEN +- shrq $2, %rdi +-# endif +- testl %eax, %eax +- jz L(second_vec_return) +- +- VPCMP $0, %YMM1, %YMMZERO, %k2 +- kmovd %k2, %edx +- /* Combine VEC1 matches (edx) with VEC2 matches (eax). */ +-# ifdef USE_AS_WCSLEN +- sall $CHAR_PER_VEC, %eax +- orl %edx, %eax +- tzcntl %eax, %eax +-# else +- salq $CHAR_PER_VEC, %rax +- orq %rdx, %rax +- tzcntq %rax, %rax +-# endif +- addq %rdi, %rax +- ret +- +- +-# ifdef USE_AS_STRNLEN +- +-L(last_4x_vec_or_less_load): +- /* Depending on entry adjust rdi / prepare first VEC in YMM1. */ +- VMOVA (VEC_SIZE * 4)(%rdi), %YMM1 +-L(last_4x_vec_or_less_cmpeq): +- VPCMP $0, %YMM1, %YMMZERO, %k0 +- addq $(VEC_SIZE * 3), %rdi +-L(last_4x_vec_or_less): +- kmovd %k0, %eax +- /* If remaining length > VEC_SIZE * 2. This works if esi is off by +- VEC_SIZE * 4. */ +- testl $(CHAR_PER_VEC * 2), %esi +- jnz L(last_4x_vec) +- +- /* length may have been negative or positive by an offset of +- CHAR_PER_VEC * 4 depending on where this was called from. This +- fixes that. */ +- andl $(CHAR_PER_VEC * 4 - 1), %esi +- testl %eax, %eax +- jnz L(last_vec_x1_check) +- +- /* Check the end of data. */ +- subl $CHAR_PER_VEC, %esi +- jb L(max) +- +- VPCMP $0, (VEC_SIZE * 2)(%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +- tzcntl %eax, %eax +- /* Check the end of data. */ +- cmpl %eax, %esi +- jb L(max) +- +- subq %rdx, %rdi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarq $2, %rdi +-# endif +- leaq (CHAR_PER_VEC * 2)(%rdi, %rax), %rax +- ret +-L(max): +- movq %r8, %rax +- ret +-# endif +- +- /* Placed here in strnlen so that the jcc L(last_4x_vec_or_less) +- in the 4x VEC loop can use 2 byte encoding. */ +- .p2align 4 +-L(second_vec_return): +- VPCMP $0, %YMM3, %YMMZERO, %k0 +- /* Combine YMM3 matches (k0) with YMM4 matches (k1). */ +-# ifdef USE_AS_WCSLEN +- kunpckbw %k0, %k1, %k0 +- kmovd %k0, %eax +- tzcntl %eax, %eax +-# else +- kunpckdq %k0, %k1, %k0 +- kmovq %k0, %rax +- tzcntq %rax, %rax +-# endif +- leaq (CHAR_PER_VEC * 2)(%rdi, %rax), %rax +- ret +- +- +-# ifdef USE_AS_STRNLEN +-L(last_vec_x1_check): +- tzcntl %eax, %eax +- /* Check the end of data. */ +- cmpl %eax, %esi +- jb L(max) +- subq %rdx, %rdi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarq $2, %rdi +-# endif +- leaq (CHAR_PER_VEC)(%rdi, %rax), %rax +- ret +- +- .p2align 4 +-L(last_4x_vec): +- /* Test first 2x VEC normally. */ +- testl %eax, %eax +- jnz L(last_vec_x1) +- +- VPCMP $0, (VEC_SIZE * 2)(%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +- testl %eax, %eax +- jnz L(last_vec_x2) +- +- /* Normalize length. */ +- andl $(CHAR_PER_VEC * 4 - 1), %esi +- VPCMP $0, (VEC_SIZE * 3)(%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +- testl %eax, %eax +- jnz L(last_vec_x3) +- +- /* Check the end of data. */ +- subl $(CHAR_PER_VEC * 3), %esi +- jb L(max) +- +- VPCMP $0, (VEC_SIZE * 4)(%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +- tzcntl %eax, %eax +- /* Check the end of data. */ +- cmpl %eax, %esi +- jb L(max_end) +- +- subq %rdx, %rdi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarq $2, %rdi +-# endif +- leaq (CHAR_PER_VEC * 4)(%rdi, %rax), %rax +- ret +- +- .p2align 4 +-L(last_vec_x1): +- tzcntl %eax, %eax +- subq %rdx, %rdi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarq $2, %rdi +-# endif +- leaq (CHAR_PER_VEC)(%rdi, %rax), %rax +- ret +- +- .p2align 4 +-L(last_vec_x2): +- tzcntl %eax, %eax +- subq %rdx, %rdi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarq $2, %rdi +-# endif +- leaq (CHAR_PER_VEC * 2)(%rdi, %rax), %rax +- ret +- +- .p2align 4 +-L(last_vec_x3): +- tzcntl %eax, %eax +- subl $(CHAR_PER_VEC * 2), %esi +- /* Check the end of data. */ +- cmpl %eax, %esi +- jb L(max_end) +- subq %rdx, %rdi +-# ifdef USE_AS_WCSLEN +- /* NB: Divide bytes by 4 to get the wchar_t count. */ +- sarq $2, %rdi +-# endif +- leaq (CHAR_PER_VEC * 3)(%rdi, %rax), %rax +- ret +-L(max_end): +- movq %r8, %rax +- ret +-# endif +- +- /* Cold case for crossing page with first load. */ +- .p2align 4 +-L(cross_page_boundary): +- movq %rdi, %rdx +- /* Align data to VEC_SIZE. */ +- andq $-VEC_SIZE, %rdi +- VPCMP $0, (%rdi), %YMMZERO, %k0 +- kmovd %k0, %eax +- /* Remove the leading bytes. */ +-# ifdef USE_AS_WCSLEN +- /* NB: Divide shift count by 4 since each bit in K0 represent 4 +- bytes. */ +- movl %edx, %ecx +- shrl $2, %ecx +- andl $(CHAR_PER_VEC - 1), %ecx +-# endif +- /* SHIFT_REG is ecx for USE_AS_WCSLEN and edx otherwise. */ +- sarxl %SHIFT_REG, %eax, %eax +- testl %eax, %eax +-# ifndef USE_AS_STRNLEN +- jz L(cross_page_continue) +- tzcntl %eax, %eax +- ret +-# else +- jnz L(cross_page_less_vec) +-# ifndef USE_AS_WCSLEN +- movl %edx, %ecx +- andl $(CHAR_PER_VEC - 1), %ecx +-# endif +- movl $CHAR_PER_VEC, %eax +- subl %ecx, %eax +- /* Check the end of data. */ +- cmpq %rax, %rsi +- ja L(cross_page_continue) +- movl %esi, %eax +- ret +-L(cross_page_less_vec): +- tzcntl %eax, %eax +- /* Select min of length and position of first null. */ +- cmpq %rax, %rsi +- cmovb %esi, %eax +- ret +-# endif +- +-END (STRLEN) ++#ifndef STRLEN ++# define STRLEN __strlen_evex + #endif ++ ++#include "x86-evex256-vecs.h" ++#include "reg-macros.h" ++#include "strlen-evex-base.S" +diff --git a/sysdeps/x86_64/multiarch/strnlen-evex512.S b/sysdeps/x86_64/multiarch/strnlen-evex512.S +index 0b7f220214a7c33c..ebf22c259f9b6362 100644 +--- a/sysdeps/x86_64/multiarch/strnlen-evex512.S ++++ b/sysdeps/x86_64/multiarch/strnlen-evex512.S +@@ -1,4 +1,264 @@ +-#define STRLEN __strnlen_evex512 +-#define USE_AS_STRNLEN 1 ++/* Placeholder function, not used by any processor at the moment. ++ Copyright (C) 2022-2023 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. + +-#include "strlen-evex512.S" ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef STRNLEN ++#define STRNLEN __strnlen_evex512 ++#endif ++ ++#include "x86-evex512-vecs.h" ++#include "reg-macros.h" ++ ++#include ++ ++#if ISA_SHOULD_BUILD (4) ++ ++# include ++ ++# ifdef USE_AS_WCSLEN ++# define VPCMPEQ vpcmpeqd ++# define VPTESTN vptestnmd ++# define VPMINU vpminud ++# define CHAR_SIZE 4 ++# else ++# define VPCMPEQ vpcmpeqb ++# define VPTESTN vptestnmb ++# define VPMINU vpminub ++# define CHAR_SIZE 1 ++# endif ++ ++# define PAGE_SIZE 4096 ++# define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) ++ ++ .section SECTION(.text),"ax",@progbits ++/* Aligning entry point to 64 byte, provides better performance for ++ one vector length string. */ ++ENTRY_P2ALIGN (STRNLEN, 6) ++ /* Check zero length. */ ++ test %RSI_LP, %RSI_LP ++ jz L(ret_max) ++# ifdef __ILP32__ ++ /* Clear the upper 32 bits. */ ++ movl %esi, %esi ++# endif ++ ++ movl %edi, %eax ++ vpxorq %VMM_128(0), %VMM_128(0), %VMM_128(0) ++ sall $20, %eax ++ cmpl $((PAGE_SIZE - VEC_SIZE) << 20), %eax ++ ja L(page_cross) ++ ++ /* Compare [w]char for null, mask bit will be set for match. */ ++ VPCMPEQ (%rdi), %VMM(0), %k0 ++ KMOV %k0, %VRCX ++ /* Store max length in rax. */ ++ mov %rsi, %rax ++ /* If rcx is 0, rax will have max length. We can not use VRCX ++ and VRAX here for evex256 because, upper 32 bits may be ++ undefined for ecx and eax. */ ++ bsfq %rcx, %rax ++ cmp $CHAR_PER_VEC, %rax ++ ja L(align_more) ++ cmpq %rax, %rsi ++ cmovb %esi, %eax ++ ret ++ ++ /* At this point vector max length reached. */ ++ .p2align 4,,3 ++L(ret_max): ++ movq %rsi, %rax ++ ret ++ ++L(align_more): ++ mov %rdi, %rax ++ /* Align rax to VEC_SIZE. */ ++ andq $-VEC_SIZE, %rax ++ movq %rdi, %rdx ++ subq %rax, %rdx ++# ifdef USE_AS_WCSLEN ++ shr $2, %VRDX ++# endif ++ /* At this point rdx contains [w]chars already compared. */ ++ leaq -CHAR_PER_VEC(%rsi, %rdx), %rdx ++ /* At this point rdx contains number of w[char] needs to go. ++ Now onwards rdx will keep decrementing with each compare. */ ++ ++ /* Loop unroll 4 times for 4 vector loop. */ ++ VPCMPEQ VEC_SIZE(%rax), %VMM(0), %k0 ++ subq $-VEC_SIZE, %rax ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX ++ jnz L(ret_vec_x1) ++ ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(ret_max) ++ ++ VPCMPEQ VEC_SIZE(%rax), %VMM(0), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX ++ jnz L(ret_vec_x2) ++ ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(ret_max) ++ ++ VPCMPEQ (VEC_SIZE * 2)(%rax), %VMM(0), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX ++ jnz L(ret_vec_x3) ++ ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(ret_max) ++ ++ VPCMPEQ (VEC_SIZE * 3)(%rax), %VMM(0), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX ++ jnz L(ret_vec_x4) ++ ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(ret_max) ++ /* Save pointer before 4 x VEC_SIZE alignment. */ ++ movq %rax, %rcx ++ ++ /* Align address to VEC_SIZE * 4 for loop. */ ++ andq $-(VEC_SIZE * 4), %rax ++ ++ subq %rax, %rcx ++# ifdef USE_AS_WCSLEN ++ shr $2, %VRCX ++# endif ++ /* rcx contains number of [w]char will be recompared due to ++ alignment fixes. rdx must be incremented by rcx to offset ++ alignment adjustment. */ ++ addq %rcx, %rdx ++ /* Need jump as we don't want to add/subtract rdx for first ++ iteration of 4 x VEC_SIZE aligned loop. */ ++ ++ .p2align 4,,11 ++L(loop): ++ /* VPMINU and VPCMP combination provide better performance as ++ compared to alternative combinations. */ ++ VMOVA (VEC_SIZE * 4)(%rax), %VMM(1) ++ VPMINU (VEC_SIZE * 5)(%rax), %VMM(1), %VMM(2) ++ VMOVA (VEC_SIZE * 6)(%rax), %VMM(3) ++ VPMINU (VEC_SIZE * 7)(%rax), %VMM(3), %VMM(4) ++ ++ VPTESTN %VMM(2), %VMM(2), %k0 ++ VPTESTN %VMM(4), %VMM(4), %k1 ++ ++ subq $-(VEC_SIZE * 4), %rax ++ KORTEST %k0, %k1 ++ ++ jnz L(loopend) ++ subq $(CHAR_PER_VEC * 4), %rdx ++ ja L(loop) ++ mov %rsi, %rax ++ ret ++ ++L(loopend): ++ ++ VPTESTN %VMM(1), %VMM(1), %k2 ++ KMOV %k2, %VRCX ++ test %VRCX, %VRCX ++ jnz L(ret_vec_x1) ++ ++ KMOV %k0, %VRCX ++ /* At this point, if k0 is non zero, null char must be in the ++ second vector. */ ++ test %VRCX, %VRCX ++ jnz L(ret_vec_x2) ++ ++ VPTESTN %VMM(3), %VMM(3), %k3 ++ KMOV %k3, %VRCX ++ test %VRCX, %VRCX ++ jnz L(ret_vec_x3) ++ /* At this point null [w]char must be in the fourth vector so no ++ need to check. */ ++ KMOV %k1, %VRCX ++ ++ /* Fourth, third, second vector terminating are pretty much ++ same, implemented this way to avoid branching and reuse code ++ from pre loop exit condition. */ ++L(ret_vec_x4): ++ bsf %VRCX, %VRCX ++ subq %rdi, %rax ++# ifdef USE_AS_WCSLEN ++ subq $-(VEC_SIZE * 3), %rax ++ shrq $2, %rax ++ addq %rcx, %rax ++# else ++ leaq (VEC_SIZE * 3)(%rcx, %rax), %rax ++# endif ++ ++ cmpq %rsi, %rax ++ cmovnb %rsi, %rax ++ ret ++ ++L(ret_vec_x3): ++ bsf %VRCX, %VRCX ++ subq %rdi, %rax ++# ifdef USE_AS_WCSLEN ++ subq $-(VEC_SIZE * 2), %rax ++ shrq $2, %rax ++ addq %rcx, %rax ++# else ++ leaq (VEC_SIZE * 2)(%rcx, %rax), %rax ++# endif ++ cmpq %rsi, %rax ++ cmovnb %rsi, %rax ++ ret ++ ++L(ret_vec_x2): ++ subq $-VEC_SIZE, %rax ++L(ret_vec_x1): ++ bsf %VRCX, %VRCX ++ subq %rdi, %rax ++# ifdef USE_AS_WCSLEN ++ shrq $2, %rax ++# endif ++ addq %rcx, %rax ++ cmpq %rsi, %rax ++ cmovnb %rsi, %rax ++ ret ++ ++L(page_cross): ++ mov %rdi, %rax ++ movl %edi, %ecx ++ andl $(VEC_SIZE - 1), %ecx ++# ifdef USE_AS_WCSLEN ++ sarl $2, %ecx ++# endif ++ /* ecx contains number of w[char] to be skipped as a result ++ of address alignment. */ ++ andq $-VEC_SIZE, %rax ++ VPCMPEQ (%rax), %VMM(0), %k0 ++ KMOV %k0, %VRDX ++ /* Ignore number of character for alignment adjustment. */ ++ shr %cl, %VRDX ++ jnz L(page_cross_end) ++ movl $CHAR_PER_VEC, %eax ++ sub %ecx, %eax ++ cmp %rax, %rsi ++ ja L(align_more) ++ ++L(page_cross_end): ++ bsf %VRDX, %VRAX ++ cmpq %rsi, %rax ++ cmovnb %esi, %eax ++ ret ++ ++END (STRNLEN) ++#endif +diff --git a/sysdeps/x86_64/multiarch/wcslen-evex512.S b/sysdeps/x86_64/multiarch/wcslen-evex512.S +index f59c372b78b4fb8c..aff288a66b2831fb 100644 +--- a/sysdeps/x86_64/multiarch/wcslen-evex512.S ++++ b/sysdeps/x86_64/multiarch/wcslen-evex512.S +@@ -1,4 +1,8 @@ +-#define STRLEN __wcslen_evex512 ++#ifndef WCSLEN ++# define WCSLEN __wcslen_evex512 ++#endif ++ ++#define STRLEN WCSLEN + #define USE_AS_WCSLEN 1 + + #include "strlen-evex512.S" +diff --git a/sysdeps/x86_64/multiarch/wcsnlen-evex512.S b/sysdeps/x86_64/multiarch/wcsnlen-evex512.S +index 73dcf2f210a85aac..1c37d74fc90ff60c 100644 +--- a/sysdeps/x86_64/multiarch/wcsnlen-evex512.S ++++ b/sysdeps/x86_64/multiarch/wcsnlen-evex512.S +@@ -1,5 +1,8 @@ +-#define STRLEN __wcsnlen_evex512 ++#ifndef WCSNLEN ++# define WCSNLEN __wcsnlen_evex512 ++#endif ++ ++#define STRNLEN WCSNLEN + #define USE_AS_WCSLEN 1 +-#define USE_AS_STRNLEN 1 + +-#include "strlen-evex512.S" ++#include "strnlen-evex512.S" diff --git a/glibc-RHEL-175520-13.patch b/glibc-RHEL-175520-13.patch new file mode 100644 index 0000000..76a367e --- /dev/null +++ b/glibc-RHEL-175520-13.patch @@ -0,0 +1,154 @@ +commit 7da08862471dfec6fdae731c2a5f351ad485c71f +Author: Noah Goldstein +Date: Tue Aug 13 23:29:14 2024 +0800 + + x86: Fix bug in strchrnul-evex512 [BZ #32078] + + Issue was we were expecting not matches with CHAR before the start of + the string in the page cross case. + + The check code in the page cross case: + ``` + and $0xffffffffffffffc0,%rax + vmovdqa64 (%rax),%zmm17 + vpcmpneqb %zmm17,%zmm16,%k1 + vptestmb %zmm17,%zmm17,%k0{%k1} + kmovq %k0,%rax + inc %rax + shr %cl,%rax + je L(continue) + ``` + + expects that all characters that neither match null nor CHAR will be + 1s in `rax` prior to the `inc`. Then the `inc` will overflow all of + the 1s where no relevant match was found. + + This is incorrect in the page-cross case, as the + `vmovdqa64 (%rax),%zmm17` loads from before the start of the input + string. + + If there are matches with CHAR before the start of the string, `rax` + won't properly overflow. + + The fix is quite simple. Just replace: + + ``` + inc %rax + shr %cl,%rax + ``` + With: + ``` + sar %cl,%rax + inc %rax + ``` + + The arithmetic shift will clear any matches prior to the start of the + string while maintaining the signbit so the 1s can properly overflow + to zero in the case of no matches. + Reviewed-by: H.J. Lu + +diff --git a/string/test-strchr.c b/string/test-strchr.c +index f24ade10485853c3..e5e55cdccd156914 100644 +--- a/string/test-strchr.c ++++ b/string/test-strchr.c +@@ -227,6 +227,69 @@ check1 (void) + check_result (impl, s, c, exp_result); + } + ++static void ++check2 (void) ++{ ++ CHAR *s = (CHAR *) (buf1 + getpagesize () - 4 * sizeof (CHAR)); ++ CHAR *s_begin = (CHAR *) (buf1 + getpagesize () - 64); ++#ifndef USE_FOR_STRCHRNUL ++ CHAR *exp_result = NULL; ++#else ++ CHAR *exp_result = s + 1; ++#endif ++ CHAR val = 0x12; ++ for (; s_begin != s; ++s_begin) ++ *s_begin = val; ++ ++ s[0] = val + 1; ++ s[1] = 0; ++ s[2] = val + 1; ++ s[3] = val + 1; ++ ++ { ++ FOR_EACH_IMPL (impl, 0) ++ check_result (impl, s, val, exp_result); ++ } ++ s[3] = val; ++ { ++ FOR_EACH_IMPL (impl, 0) ++ check_result (impl, s, val, exp_result); ++ } ++ exp_result = s; ++ s[0] = val; ++ { ++ FOR_EACH_IMPL (impl, 0) ++ check_result (impl, s, val, exp_result); ++ } ++ ++ s[3] = val + 1; ++ { ++ FOR_EACH_IMPL (impl, 0) ++ check_result (impl, s, val, exp_result); ++ } ++ ++ s[0] = val + 1; ++ s[1] = val + 1; ++ s[2] = val + 1; ++ s[3] = val + 1; ++ s[4] = val; ++ exp_result = s + 4; ++ { ++ FOR_EACH_IMPL (impl, 0) ++ check_result (impl, s, val, exp_result); ++ } ++ s[4] = 0; ++#ifndef USE_FOR_STRCHRNUL ++ exp_result = NULL; ++#else ++ exp_result = s + 4; ++#endif ++ { ++ FOR_EACH_IMPL (impl, 0) ++ check_result (impl, s, val, exp_result); ++ } ++} ++ + int + test_main (void) + { +@@ -235,7 +298,7 @@ test_main (void) + test_init (); + + check1 (); +- ++ check2 (); + printf ("%20s", ""); + FOR_EACH_IMPL (impl, 0) + printf ("\t%s", impl->name); +diff --git a/sysdeps/x86_64/multiarch/strchr-evex-base.S b/sysdeps/x86_64/multiarch/strchr-evex-base.S +index 75fee8c82ade14f1..f50923ec839e0599 100644 +--- a/sysdeps/x86_64/multiarch/strchr-evex-base.S ++++ b/sysdeps/x86_64/multiarch/strchr-evex-base.S +@@ -124,13 +124,13 @@ L(page_cross): + VPCMPNE %VMM(1), %VMM(0), %k1 + VPTEST %VMM(1), %VMM(1), %k0{%k1} + KMOV %k0, %VRAX +-# ifdef USE_AS_WCSCHR ++ sar %cl, %VRAX ++#ifdef USE_AS_WCSCHR + sub $VEC_MATCH_MASK, %VRAX +-# else ++#else + inc %VRAX +-# endif ++#endif + /* Ignore number of character for alignment adjustment. */ +- shr %cl, %VRAX + jz L(align_more) + + bsf %VRAX, %VRAX diff --git a/glibc-RHEL-175520-14.patch b/glibc-RHEL-175520-14.patch new file mode 100644 index 0000000..02a5475 --- /dev/null +++ b/glibc-RHEL-175520-14.patch @@ -0,0 +1,783 @@ +commit 294a8927694ed866ffc40833f1b6d96cd649df0a +Author: Matthew Sterrett +Date: Fri Aug 9 15:05:09 2024 -0700 + + x86: Unifies 'strnlen-evex' and 'strnlen-evex512' implementations. + + This commit uses a common implementation 'strnlen-evex-base.S' for both + 'strnlen-evex' and 'strnlen-evex512' + + This patch serves both to reduce the number of implementations, and it also does some small optimizations that benefit strnlen-evex and strnlen-evex512. + + All tests pass on x86. + + Benchmarks were taken on SKX. + https://www.intel.com/content/www/us/en/products/sku/123613/intel-core-i97900x-xseries-processor-13-75m-cache-up-to-4-30-ghz/specifications.html + + Geometric mean for strnlen-evex over all benchmarks (N=10) was (new/old) 0.881 + Geometric mean for strnlen-evex512 over all benchmarks (N=10) was (new/old) 0.953 + + Code Size Changes: + strnlen-evex : +31 bytes + strnlen-evex512 : +156 bytes + Reviewed-by: Noah Goldstein + +Conflicts: + sysdeps/x86_64/multiarch/strnlen-evex.S + (adapt to 2.34 branch) + sysdeps/x86_64/multiarch/strnlen-evex512.S + (fixup context) + +diff --git a/sysdeps/x86_64/multiarch/strnlen-evex-base.S b/sysdeps/x86_64/multiarch/strnlen-evex-base.S +new file mode 100644 +index 0000000000000000..1c2cfdfe067140f1 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/strnlen-evex-base.S +@@ -0,0 +1,462 @@ ++/* strnlen/wcsnlen optimized with 256/512-bit EVEX instructions. ++ Copyright (C) 2022-2024 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++ ++#include ++ ++#if ISA_SHOULD_BUILD (4) ++ ++# include ++ ++#ifdef USE_AS_WCSLEN ++# define VPCMPEQ vpcmpeqd ++# define VPTESTN vptestnmd ++# define VPMINU vpminud ++# define CHAR_SIZE 4 ++#else ++# define VPCMPEQ vpcmpeqb ++# define VPTESTN vptestnmb ++# define VPMINU vpminub ++# define CHAR_SIZE 1 ++#endif ++ ++#define XZERO VMM_128(0) ++#define VZERO VMM(0) ++#define PAGE_SIZE 4096 ++#define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) ++ ++#if CHAR_PER_VEC == 32 ++# define SUB_SHORT(imm, reg) subb $(imm), %VGPR_SZ(reg, 8) ++#else ++# define SUB_SHORT(imm, reg) subl $(imm), %VGPR_SZ(reg, 32) ++#endif ++ ++#ifdef USE_AS_WCSLEN ++/* For wide-character, we care more about limitting code size ++ than optimally aligning targets, so just cap nop padding ++ reasonably low. */ ++# define P2ALIGN(...) .p2align 4,, 6 ++# define P2ALIGN_CLAMPED(...) P2ALIGN(__VA_ARGS__) ++#else ++# define P2ALIGN(x) .p2align x ++# define P2ALIGN_CLAMPED(x, y) .p2align x,, y ++#endif ++ ++ .section SECTION(.text), "ax", @progbits ++ /* Aligning entry point to 64 byte, provides better performance for ++ one vector length string. */ ++ENTRY_P2ALIGN(STRNLEN, 6) ++ /* rdi is pointer to array, rsi is the upper limit. */ ++ ++ /* Check zero length. */ ++ test %RSI_LP, %RSI_LP ++ jz L(zero) ++ ++#ifdef __ILP32__ ++ /* Clear the upper 32 bits. */ ++ movl %esi, %esi ++#endif ++ ++ vpxorq %XZERO, %XZERO, %XZERO ++ ++ /* Check that we won't cross a page boundary with our first load. */ ++ movl %edi, %eax ++ shll $20, %eax ++ cmpl $((PAGE_SIZE - VEC_SIZE) << 20), %eax ++ ja L(crosses_page_boundary) ++ ++ /* Check the first VEC_SIZE bytes. Each bit in K0 represents a ++ null byte. */ ++ VPCMPEQ (%rdi), %VZERO, %k0 ++ KMOV %k0, %VRCX ++ ++ /* If src (rcx) is zero, bsf does not change the result. NB: ++ Must use 64-bit bsf here so that upper bits of len are not ++ cleared. */ ++ movq %rsi, %rax ++ bsfq %rcx, %rax ++ ++ /* If rax > CHAR_PER_VEC then rcx must have been zero (no null ++ CHAR) and rsi must be > CHAR_PER_VEC. */ ++ cmpq $CHAR_PER_VEC, %rax ++ ja L(more_1x_vec) ++ ++ /* Check if first match in bounds. */ ++ cmpq %rax, %rsi ++ cmovb %esi, %eax ++ ret ++ ++#if VEC_SIZE == 32 ++ P2ALIGN_CLAMPED(4, 2) ++L(zero): ++L(max_0): ++ movl %esi, %eax ++ ret ++#endif ++ ++ P2ALIGN_CLAMPED(4, 10) ++L(more_1x_vec): ++L(cross_page_continue): ++ /* After this calculation, rax stores the number of elements ++ left to be processed The complexity comes from the fact some ++ elements get read twice due to alignment and we need to be ++ sure we don't count them twice (else, it would just be rsi - ++ CHAR_PER_VEC). */ ++ ++#ifdef USE_AS_WCSLEN ++ /* Need to compute directly for wcslen as CHAR_SIZE * rsi can ++ overflow. */ ++ movq %rdi, %rax ++ andq $(VEC_SIZE * -1), %rdi ++ subq %rdi, %rax ++ sarq $2, %rax ++ leaq -(CHAR_PER_VEC * 1)(%rax, %rsi), %rax ++#else ++ /* Calculate ptr + N - VEC_SIZE, then mask off the low bits, ++ then subtract ptr to get the new aligned limit value. */ ++ leaq (VEC_SIZE * -1)(%rsi, %rdi), %rax ++ andq $(VEC_SIZE * -1), %rdi ++ subq %rdi, %rax ++#endif ++ ++ VPCMPEQ VEC_SIZE(%rdi), %VZERO, %k0 ++ ++ /* Checking here is faster for 256-bit but not 512-bit */ ++#if VEC_SIZE == 0 ++ KMOV %k0, %VRDX ++ test %VRDX, %VRDX ++ jnz L(last_vec_check) ++#endif ++ ++ cmpq $(CHAR_PER_VEC * 2), %rax ++ ja L(more_2x_vec) ++ ++L(last_2x_vec_or_less): ++ ++ /* Checking here is faster for 512-bit but not 256-bit */ ++#if VEC_SIZE != 0 ++ KMOV %k0, %VRDX ++ test %VRDX, %VRDX ++ jnz L(last_vec_check) ++#endif ++ ++ /* Check for the end of data. */ ++ SUB_SHORT (CHAR_PER_VEC, rax) ++ jbe L(max_0) ++ ++ /* Check the final remaining vector. */ ++ VPCMPEQ (VEC_SIZE * 2)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRDX ++ test %VRDX, %VRDX ++#if VEC_SIZE == 32 ++ jz L(max_0) ++#else ++ jnz L(last_vec_check) ++ P2ALIGN_CLAMPED(4, 2) ++L(zero): ++L(max_0): ++ movl %esi, %eax ++ ret ++ ++#endif ++ P2ALIGN_CLAMPED(4, 4) ++L(last_vec_check): ++ bsf %VRDX, %VRDX ++ sub %eax, %edx ++ lea (%rsi, %rdx), %eax ++ cmovae %esi, %eax ++ ret ++ ++ ++#if VEC_SIZE == 32 ++ P2ALIGN_CLAMPED(4, 8) ++#endif ++L(last_4x_vec_or_less): ++ addl $(CHAR_PER_VEC * -4), %eax ++ VPCMPEQ (VEC_SIZE * 5)(%rdi), %VZERO, %k0 ++ ++#if VEC_SIZE == 64 ++ KMOV %k0, %VRDX ++ test %VRDX, %VRDX ++ jnz L(last_vec_check) ++#endif ++ ++ subq $(VEC_SIZE * -4), %rdi ++ cmpl $(CHAR_PER_VEC * 2), %eax ++ jbe L(last_2x_vec_or_less) ++ ++ P2ALIGN_CLAMPED(4, 6) ++L(more_2x_vec): ++ /* Remaining length >= 2 * CHAR_PER_VEC so do VEC0/VEC1 without ++ rechecking bounds. */ ++ ++ /* Already checked in 256-bit case */ ++#if VEC_SIZE != 0 ++ KMOV %k0, %VRDX ++ ++ test %VRDX, %VRDX ++ jnz L(first_vec_x1) ++#endif ++ ++ VPCMPEQ (VEC_SIZE * 2)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRDX ++ ++ test %VRDX, %VRDX ++ jnz L(first_vec_x2) ++ ++ cmpq $(CHAR_PER_VEC * 4), %rax ++ ja L(more_4x_vec) ++ ++ ++ VPCMPEQ (VEC_SIZE * 3)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRDX ++ addl $(CHAR_PER_VEC * -2), %eax ++ test %VRDX, %VRDX ++ jnz L(last_vec_check) ++ ++ subb $(CHAR_PER_VEC), %al ++ jbe L(max_1) ++ ++ VPCMPEQ (VEC_SIZE * 4)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRDX ++ ++ test %VRDX, %VRDX ++ jnz L(last_vec_check) ++L(max_1): ++ movl %esi, %eax ++ ret ++ ++ ++ P2ALIGN_CLAMPED(4, 14) ++L(first_vec_x2): ++#if VEC_SIZE == 64 ++ /* If VEC_SIZE == 64 we can fit logic for full return label in ++ spare bytes before next cache line. */ ++ bsf %VRDX, %VRDX ++ sub %eax, %esi ++ leal (CHAR_PER_VEC * 1)(%rsi, %rdx), %eax ++ ret ++ P2ALIGN_CLAMPED(4, 6) ++#else ++ addl $CHAR_PER_VEC, %esi ++#endif ++L(first_vec_x1): ++ bsf %VRDX, %VRDX ++ sub %eax, %esi ++ leal (CHAR_PER_VEC * 0)(%rsi, %rdx), %eax ++ ret ++ ++#if VEC_SIZE == 64 ++ P2ALIGN_CLAMPED(4, 6) ++L(first_vec_x4): ++# if VEC_SIZE == 64 ++ /* If VEC_SIZE == 64 we can fit logic for full return label in ++ spare bytes before next cache line. */ ++ bsf %VRDX, %VRDX ++ sub %eax, %esi ++ leal (CHAR_PER_VEC * 3)(%rsi, %rdx), %eax ++ ret ++ P2ALIGN_CLAMPED(4, 6) ++# else ++ addl $CHAR_PER_VEC, %esi ++# endif ++L(first_vec_x3): ++ bsf %VRDX, %VRDX ++ sub %eax, %esi ++ leal (CHAR_PER_VEC * 2)(%rsi, %rdx), %eax ++ ret ++#endif ++ ++ P2ALIGN_CLAMPED(6, 20) ++L(more_4x_vec): ++ VPCMPEQ (VEC_SIZE * 3)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRDX ++ test %VRDX, %VRDX ++ jnz L(first_vec_x3) ++ ++ VPCMPEQ (VEC_SIZE * 4)(%rdi), %VZERO, %k0 ++ KMOV %k0, %VRDX ++ test %VRDX, %VRDX ++ jnz L(first_vec_x4) ++ ++ /* Check if at last VEC_SIZE * 4 length before aligning for the ++ loop. */ ++ cmpq $(CHAR_PER_VEC * 8), %rax ++ jbe L(last_4x_vec_or_less) ++ ++ ++ /* Compute number of words checked after aligning. */ ++#ifdef USE_AS_WCSLEN ++ /* Need to compute directly for wcslen as CHAR_SIZE * rsi can ++ overflow. */ ++ leaq (VEC_SIZE * -3)(%rdi), %rdx ++#else ++ leaq (VEC_SIZE * -3)(%rdi, %rax), %rax ++#endif ++ ++ subq $(VEC_SIZE * -1), %rdi ++ ++ /* Align data to VEC_SIZE * 4. */ ++#if VEC_SIZE == 64 ++ /* Saves code size. No evex512 processor has partial register ++ stalls. If that change this can be replaced with `andq ++ $-(VEC_SIZE * 4), %rdi`. */ ++ xorb %dil, %dil ++#else ++ andq $-(VEC_SIZE * 4), %rdi ++#endif ++ ++#ifdef USE_AS_WCSLEN ++ subq %rdi, %rdx ++ sarq $2, %rdx ++ addq %rdx, %rax ++#else ++ subq %rdi, %rax ++#endif ++ ++ // mov %rdi, %rdx ++ ++ P2ALIGN(6) ++L(loop): ++ /* VPMINU and VPCMP combination provide better performance as ++ compared to alternative combinations. */ ++ VMOVA (VEC_SIZE * 4)(%rdi), %VMM(1) ++ VPMINU (VEC_SIZE * 5)(%rdi), %VMM(1), %VMM(2) ++ VMOVA (VEC_SIZE * 6)(%rdi), %VMM(3) ++ VPMINU (VEC_SIZE * 7)(%rdi), %VMM(3), %VMM(4) ++ ++ VPTESTN %VMM(2), %VMM(2), %k0 ++ VPTESTN %VMM(4), %VMM(4), %k1 ++ ++ subq $-(VEC_SIZE * 4), %rdi ++ KORTEST %k0, %k1 ++ ++ jnz L(loopend) ++ subq $(CHAR_PER_VEC * 4), %rax ++ ja L(loop) ++ mov %rsi, %rax ++ ret ++ ++ ++#if VEC_SIZE == 32 ++ P2ALIGN_CLAMPED(4, 6) ++L(first_vec_x4): ++# if VEC_SIZE == 64 ++ /* If VEC_SIZE == 64 we can fit logic for full return label in ++ spare bytes before next cache line. */ ++ bsf %VRDX, %VRDX ++ sub %eax, %esi ++ leal (CHAR_PER_VEC * 3)(%rsi, %rdx), %eax ++ ret ++ P2ALIGN_CLAMPED(4, 6) ++# else ++ addl $CHAR_PER_VEC, %esi ++# endif ++L(first_vec_x3): ++ bsf %VRDX, %VRDX ++ sub %eax, %esi ++ leal (CHAR_PER_VEC * 2)(%rsi, %rdx), %eax ++ ret ++#endif ++ ++ ++ P2ALIGN_CLAMPED(4, 11) ++L(loopend): ++ /* We found a null terminator in one of the 4 vectors. */ ++ ++ /* Check the first vector. */ ++ movq %rax, %r8 ++ VPTESTN %VMM(1), %VMM(1), %k2 ++ KMOV %k2, %VRCX ++ bsf %rcx, %r8 ++ ++ cmpq $(CHAR_PER_VEC), %r8 ++ jbe L(end_vec) ++ ++ /* Check the second vector. */ ++ subq $(CHAR_PER_VEC), %rax ++ movq %rax, %r8 ++ KMOV %k0, %VRCX ++ bsf %rcx, %r8 ++ ++ cmpq $(CHAR_PER_VEC), %r8 ++ jbe L(end_vec) ++ ++ /* Check the third vector. */ ++ subq $(CHAR_PER_VEC), %rax ++ movq %rax, %r8 ++ VPTESTN %VMM(3), %VMM(3), %k2 ++ KMOV %k2, %VRCX ++ bsf %rcx, %r8 ++ ++ cmpq $(CHAR_PER_VEC), %r8 ++ jbe L(end_vec) ++ ++ /* It is in the fourth vector. */ ++ subq $(CHAR_PER_VEC), %rax ++ movq %rax, %r8 ++ KMOV %k1, %VRCX ++ bsf %rcx, %r8 ++ ++ P2ALIGN_CLAMPED(4, 3) ++L(end_vec): ++ /* Get the number that has been processed. */ ++ movq %rsi, %rcx ++ subq %rax, %rcx ++ ++ /* Add that to the offset we found the null terminator at. */ ++ leaq (%r8, %rcx), %rax ++ ++ /* Take the min of that and the limit. */ ++ cmpq %rsi, %rax ++ cmovnb %rsi, %rax ++ ret ++ ++ P2ALIGN_CLAMPED(4, 11) ++L(crosses_page_boundary): ++ /* Align data backwards to VEC_SIZE. */ ++ shrl $20, %eax ++ movq %rdi, %rcx ++ andq $-VEC_SIZE, %rcx ++ VPCMPEQ (%rcx), %VZERO, %k0 ++ ++ KMOV %k0, %VRCX ++#ifdef USE_AS_WCSLEN ++ shrl $2, %eax ++ andl $(CHAR_PER_VEC - 1), %eax ++#endif ++ /* By this point rax contains number of bytes we need to skip. */ ++ shrx %VRAX, %VRCX, %VRCX ++ ++ /* Calculates CHAR_PER_VEC - eax and stores in eax. */ ++ negl %eax ++ andl $(CHAR_PER_VEC - 1), %eax ++ ++ movq %rsi, %rdx ++ bsf %VRCX, %VRDX ++ cmpq %rax, %rdx ++ ja L(cross_page_continue) ++ ++ /* The vector had a null terminator or we are at the limit. */ ++ movl %edx, %eax ++ cmpq %rdx, %rsi ++ cmovb %esi, %eax ++ ret ++ ++END(STRNLEN) ++#endif +diff --git a/sysdeps/x86_64/multiarch/strnlen-evex.S b/sysdeps/x86_64/multiarch/strnlen-evex.S +index 722022f303cc0ab5..c41288906cdd2bc4 100644 +--- a/sysdeps/x86_64/multiarch/strnlen-evex.S ++++ b/sysdeps/x86_64/multiarch/strnlen-evex.S +@@ -1,4 +1,7 @@ +-#define STRLEN __strnlen_evex +-#define USE_AS_STRNLEN 1 ++#ifndef STRNLEN ++#define STRNLEN __strnlen_evex ++#endif + +-#include "strlen-evex.S" ++#include "x86-evex256-vecs.h" ++#include "reg-macros.h" ++#include "strnlen-evex-base.S" +diff --git a/sysdeps/x86_64/multiarch/strnlen-evex512.S b/sysdeps/x86_64/multiarch/strnlen-evex512.S +index ebf22c259f9b6362..07f0bb375cb671d7 100644 +--- a/sysdeps/x86_64/multiarch/strnlen-evex512.S ++++ b/sysdeps/x86_64/multiarch/strnlen-evex512.S +@@ -1,264 +1,7 @@ +-/* Placeholder function, not used by any processor at the moment. +- Copyright (C) 2022-2023 Free Software Foundation, Inc. +- This file is part of the GNU C Library. +- +- The GNU C Library is free software; you can redistribute it and/or +- modify it under the terms of the GNU Lesser General Public +- License as published by the Free Software Foundation; either +- version 2.1 of the License, or (at your option) any later version. +- +- The GNU C Library is distributed in the hope that it will be useful, +- but WITHOUT ANY WARRANTY; without even the implied warranty of +- MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU +- Lesser General Public License for more details. +- +- You should have received a copy of the GNU Lesser General Public +- License along with the GNU C Library; if not, see +- . */ +- + #ifndef STRNLEN + #define STRNLEN __strnlen_evex512 + #endif + + #include "x86-evex512-vecs.h" + #include "reg-macros.h" +- +-#include +- +-#if ISA_SHOULD_BUILD (4) +- +-# include +- +-# ifdef USE_AS_WCSLEN +-# define VPCMPEQ vpcmpeqd +-# define VPTESTN vptestnmd +-# define VPMINU vpminud +-# define CHAR_SIZE 4 +-# else +-# define VPCMPEQ vpcmpeqb +-# define VPTESTN vptestnmb +-# define VPMINU vpminub +-# define CHAR_SIZE 1 +-# endif +- +-# define PAGE_SIZE 4096 +-# define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) +- +- .section SECTION(.text),"ax",@progbits +-/* Aligning entry point to 64 byte, provides better performance for +- one vector length string. */ +-ENTRY_P2ALIGN (STRNLEN, 6) +- /* Check zero length. */ +- test %RSI_LP, %RSI_LP +- jz L(ret_max) +-# ifdef __ILP32__ +- /* Clear the upper 32 bits. */ +- movl %esi, %esi +-# endif +- +- movl %edi, %eax +- vpxorq %VMM_128(0), %VMM_128(0), %VMM_128(0) +- sall $20, %eax +- cmpl $((PAGE_SIZE - VEC_SIZE) << 20), %eax +- ja L(page_cross) +- +- /* Compare [w]char for null, mask bit will be set for match. */ +- VPCMPEQ (%rdi), %VMM(0), %k0 +- KMOV %k0, %VRCX +- /* Store max length in rax. */ +- mov %rsi, %rax +- /* If rcx is 0, rax will have max length. We can not use VRCX +- and VRAX here for evex256 because, upper 32 bits may be +- undefined for ecx and eax. */ +- bsfq %rcx, %rax +- cmp $CHAR_PER_VEC, %rax +- ja L(align_more) +- cmpq %rax, %rsi +- cmovb %esi, %eax +- ret +- +- /* At this point vector max length reached. */ +- .p2align 4,,3 +-L(ret_max): +- movq %rsi, %rax +- ret +- +-L(align_more): +- mov %rdi, %rax +- /* Align rax to VEC_SIZE. */ +- andq $-VEC_SIZE, %rax +- movq %rdi, %rdx +- subq %rax, %rdx +-# ifdef USE_AS_WCSLEN +- shr $2, %VRDX +-# endif +- /* At this point rdx contains [w]chars already compared. */ +- leaq -CHAR_PER_VEC(%rsi, %rdx), %rdx +- /* At this point rdx contains number of w[char] needs to go. +- Now onwards rdx will keep decrementing with each compare. */ +- +- /* Loop unroll 4 times for 4 vector loop. */ +- VPCMPEQ VEC_SIZE(%rax), %VMM(0), %k0 +- subq $-VEC_SIZE, %rax +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x1) +- +- subq $CHAR_PER_VEC, %rdx +- jbe L(ret_max) +- +- VPCMPEQ VEC_SIZE(%rax), %VMM(0), %k0 +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x2) +- +- subq $CHAR_PER_VEC, %rdx +- jbe L(ret_max) +- +- VPCMPEQ (VEC_SIZE * 2)(%rax), %VMM(0), %k0 +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x3) +- +- subq $CHAR_PER_VEC, %rdx +- jbe L(ret_max) +- +- VPCMPEQ (VEC_SIZE * 3)(%rax), %VMM(0), %k0 +- KMOV %k0, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x4) +- +- subq $CHAR_PER_VEC, %rdx +- jbe L(ret_max) +- /* Save pointer before 4 x VEC_SIZE alignment. */ +- movq %rax, %rcx +- +- /* Align address to VEC_SIZE * 4 for loop. */ +- andq $-(VEC_SIZE * 4), %rax +- +- subq %rax, %rcx +-# ifdef USE_AS_WCSLEN +- shr $2, %VRCX +-# endif +- /* rcx contains number of [w]char will be recompared due to +- alignment fixes. rdx must be incremented by rcx to offset +- alignment adjustment. */ +- addq %rcx, %rdx +- /* Need jump as we don't want to add/subtract rdx for first +- iteration of 4 x VEC_SIZE aligned loop. */ +- +- .p2align 4,,11 +-L(loop): +- /* VPMINU and VPCMP combination provide better performance as +- compared to alternative combinations. */ +- VMOVA (VEC_SIZE * 4)(%rax), %VMM(1) +- VPMINU (VEC_SIZE * 5)(%rax), %VMM(1), %VMM(2) +- VMOVA (VEC_SIZE * 6)(%rax), %VMM(3) +- VPMINU (VEC_SIZE * 7)(%rax), %VMM(3), %VMM(4) +- +- VPTESTN %VMM(2), %VMM(2), %k0 +- VPTESTN %VMM(4), %VMM(4), %k1 +- +- subq $-(VEC_SIZE * 4), %rax +- KORTEST %k0, %k1 +- +- jnz L(loopend) +- subq $(CHAR_PER_VEC * 4), %rdx +- ja L(loop) +- mov %rsi, %rax +- ret +- +-L(loopend): +- +- VPTESTN %VMM(1), %VMM(1), %k2 +- KMOV %k2, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x1) +- +- KMOV %k0, %VRCX +- /* At this point, if k0 is non zero, null char must be in the +- second vector. */ +- test %VRCX, %VRCX +- jnz L(ret_vec_x2) +- +- VPTESTN %VMM(3), %VMM(3), %k3 +- KMOV %k3, %VRCX +- test %VRCX, %VRCX +- jnz L(ret_vec_x3) +- /* At this point null [w]char must be in the fourth vector so no +- need to check. */ +- KMOV %k1, %VRCX +- +- /* Fourth, third, second vector terminating are pretty much +- same, implemented this way to avoid branching and reuse code +- from pre loop exit condition. */ +-L(ret_vec_x4): +- bsf %VRCX, %VRCX +- subq %rdi, %rax +-# ifdef USE_AS_WCSLEN +- subq $-(VEC_SIZE * 3), %rax +- shrq $2, %rax +- addq %rcx, %rax +-# else +- leaq (VEC_SIZE * 3)(%rcx, %rax), %rax +-# endif +- +- cmpq %rsi, %rax +- cmovnb %rsi, %rax +- ret +- +-L(ret_vec_x3): +- bsf %VRCX, %VRCX +- subq %rdi, %rax +-# ifdef USE_AS_WCSLEN +- subq $-(VEC_SIZE * 2), %rax +- shrq $2, %rax +- addq %rcx, %rax +-# else +- leaq (VEC_SIZE * 2)(%rcx, %rax), %rax +-# endif +- cmpq %rsi, %rax +- cmovnb %rsi, %rax +- ret +- +-L(ret_vec_x2): +- subq $-VEC_SIZE, %rax +-L(ret_vec_x1): +- bsf %VRCX, %VRCX +- subq %rdi, %rax +-# ifdef USE_AS_WCSLEN +- shrq $2, %rax +-# endif +- addq %rcx, %rax +- cmpq %rsi, %rax +- cmovnb %rsi, %rax +- ret +- +-L(page_cross): +- mov %rdi, %rax +- movl %edi, %ecx +- andl $(VEC_SIZE - 1), %ecx +-# ifdef USE_AS_WCSLEN +- sarl $2, %ecx +-# endif +- /* ecx contains number of w[char] to be skipped as a result +- of address alignment. */ +- andq $-VEC_SIZE, %rax +- VPCMPEQ (%rax), %VMM(0), %k0 +- KMOV %k0, %VRDX +- /* Ignore number of character for alignment adjustment. */ +- shr %cl, %VRDX +- jnz L(page_cross_end) +- movl $CHAR_PER_VEC, %eax +- sub %ecx, %eax +- cmp %rax, %rsi +- ja L(align_more) +- +-L(page_cross_end): +- bsf %VRDX, %VRAX +- cmpq %rsi, %rax +- cmovnb %esi, %eax +- ret +- +-END (STRNLEN) +-#endif ++#include "strnlen-evex-base.S" diff --git a/glibc-RHEL-175520-15.patch b/glibc-RHEL-175520-15.patch new file mode 100644 index 0000000..ae0580e --- /dev/null +++ b/glibc-RHEL-175520-15.patch @@ -0,0 +1,59 @@ +Partial backport of b79f8ff26aa6. +Upstream commit was skipped during backporting because most of the files +impacted were re-written later. This missing backport left +wcsnlen-evex.S pointing to strlen-evex.S. After strlen-evex.S dropped +bounds checking (Patch 12) and strnlen-evex-base.S was introduced (Patch +14), wcsnlen-evex.S must be updated to point to the new strnlen +implementation. + +commit b79f8ff26aa6151d2d2167afcddcd1ec46cfbc81 +Author: Noah Goldstein +Date: Tue Oct 18 17:44:05 2022 -0700 + + x86: Optimize strnlen-evex.S and implement with VMM headers + + Optimizations are: + 1. Use the fact that bsf(0) leaves the destination unchanged to save a + branch in short string case. + 2. Restructure code so that small strings are given the hot path. + - This is a net-zero on the benchmark suite but in general makes + sense as smaller sizes are far more common. + 3. Use more code-size efficient instructions. + - tzcnt ... -> bsf ... + - vpcmpb $0 ... -> vpcmpeq ... + 4. Align labels less aggressively, especially if it doesn't save fetch + blocks / causes the basic-block to span extra cache-lines. + + The optimizations (especially for point 2) make the strnlen and + strlen code essentially incompatible so split strnlen-evex + to a new file. + + Code Size Changes: + strlen-evex.S : -23 bytes + strnlen-evex.S : -167 bytes + + Net perf changes: + + Reported as geometric mean of all improvements / regressions from N=10 + runs of the benchtests. Value as New Time / Old Time so < 1.0 is + improvement and 1.0 is regression. + + strlen-evex.S : 0.992 (No real change) + strnlen-evex.S : 0.947 + + Full results attached in email. + + Full check passes on x86-64. + +diff --git a/sysdeps/x86_64/multiarch/wcsnlen-evex.S b/sysdeps/x86_64/multiarch/wcsnlen-evex.S +index 24773bb4e2327ca4..920d860047e5a96c 100644 +--- a/sysdeps/x86_64/multiarch/wcsnlen-evex.S ++++ b/sysdeps/x86_64/multiarch/wcsnlen-evex.S +@@ -1,5 +1,4 @@ +-#define STRLEN __wcsnlen_evex ++#define STRNLEN __wcsnlen_evex + #define USE_AS_WCSLEN 1 +-#define USE_AS_STRNLEN 1 + +-#include "strlen-evex.S" ++#include "strnlen-evex.S" diff --git a/glibc-RHEL-175520-16.patch b/glibc-RHEL-175520-16.patch new file mode 100644 index 0000000..b62d93d --- /dev/null +++ b/glibc-RHEL-175520-16.patch @@ -0,0 +1,257 @@ +commit cd5fda114ece002945ace3d54a8f80a4f67d1fbb +Author: Sajan Karumanchi +Date: Thu Mar 26 09:21:30 2026 +0000 + + x86_64: Prefer EVEX512 code-path on AMD Zen5 CPUs + + Introduced a synthetic architecture preference flag (Prefer_EVEX512) + and enabled it for AMD Zen5 (CPUID Family 0x1A) when AVX-512 is supported. + + This flag modifies IFUNC dispatch to prefer 512-bit EVEX variants over + 256-bit EVEX variants for string and memory functions on Zen5 processors, + leveraging their native 512-bit execution units for improved throughput. + When Prefer_EVEX512 is set, the dispatcher selects evex512 implementations; + otherwise, it falls back to evex (256-bit) variants. + + The implementation updates the IFUNC selection logic in ifunc-avx2.h and + ifunc-evex.h to check for the Prefer_EVEX512 flag before dispatching to + EVEX512 implementations. This change affects six string/memory functions: + + - strchr + - strlen + - strnlen + - strrchr + - strchrnul + - memchr + + Benchmarks conducted on AMD Zen5 hardware demonstrate significant + performance improvements across all affected functions: + + Function Baseline Patched Avg Avg Avg Max + Variant Variant Baseline Patched Change Improve + (ns) (ns) % % + ------------+----------+----------+-----------+----------+--------+-------- + STRCHR evex evex512 16.408 12.293 25.08% 37.69% + STRLEN evex evex512 16.862 11.436 32.18% 56.74% + STRNLEN evex evex512 18.493 11.762 36.40% 64.40% + STRRCHR evex evex512 15.154 10.874 28.24% 44.38% + STRCHRNUL evex evex512 16.464 12.605 23.44% 45.56% + MEMCHR evex evex512 9.984 8.268 17.19% 39.99% + + Additionally, a tunable option (glibc.cpu.x86_cpu_features.preferred) + is provided to allow runtime control of the Prefer_EVEX512 flag for testing + and compatibility. + + Reviewed-by: Ganesh Gopalasubramanian + Reviewed-by: H.J. Lu + +Conflicts: + sysdeps/x86/cpu-features.c + (adapt to missing b93dddfaf440aa12f45d7c356f6ffe9f27d35577) + sysdeps/x86/cpu-tunables.c + (fixup: extra "disable" parameter downstream) + sysdeps/x86/include/cpu-features-preferred_feature_index_1.def + (fixup context: missing 5bcf6265f215326d14dfacdce8532792c2c7f8f8) + sysdeps/x86_64/multiarch/ifunc-avx2.h + (fixup context) + sysdeps/x86_64/multiarch/ifunc-evex.h + (fixup context) + sysdeps/x86_64/multiarch/strchr.c + (fixup context) + +diff --git a/sysdeps/x86/cpu-features.c b/sysdeps/x86/cpu-features.c +index 846c17366a38abc2..fd0c8be971b3119c 100644 +--- a/sysdeps/x86/cpu-features.c ++++ b/sysdeps/x86/cpu-features.c +@@ -855,6 +855,12 @@ https://www.intel.com/content/www/us/en/support/articles/000059422/processors.ht + + ecx = cpu_features->features[CPUID_INDEX_1].cpuid.ecx; + ++ /* Prefer EVEX512 string/memory variants on AMD Zen5 (Family 0x1A) ++ when AVX-512 is usable. */ ++ if (family == 0x1A && CPU_FEATURE_USABLE_P (cpu_features, AVX512F)) ++ cpu_features->preferred[index_arch_Prefer_EVEX512] ++ |= bit_arch_Prefer_EVEX512; ++ + if (CPU_FEATURE_USABLE_P (cpu_features, AVX)) + { + /* Since the FMA4 bit is in CPUID_INDEX_80000001 and +diff --git a/sysdeps/x86/cpu-tunables.c b/sysdeps/x86/cpu-tunables.c +index 957db3ad229ba39f..1d1da66ee4e9f639 100644 +--- a/sysdeps/x86/cpu-tunables.c ++++ b/sysdeps/x86/cpu-tunables.c +@@ -220,6 +220,12 @@ TUNABLE_CALLBACK (set_hwcaps) (tunable_val_t *valp) + disable, 11); + } + break; ++ case 14: ++ { ++ CHECK_GLIBC_IFUNC_PREFERRED_NEED_BOTH ++ (n, cpu_features, Prefer_EVEX512, AVX512F, disable, 14); ++ } ++ break; + case 15: + { + CHECK_GLIBC_IFUNC_PREFERRED_BOTH (n, cpu_features, +diff --git a/sysdeps/x86/include/cpu-features-preferred_feature_index_1.def b/sysdeps/x86/include/cpu-features-preferred_feature_index_1.def +index 1530d594b3a0c88e..21955eabc7a4fe91 100644 +--- a/sysdeps/x86/include/cpu-features-preferred_feature_index_1.def ++++ b/sysdeps/x86/include/cpu-features-preferred_feature_index_1.def +@@ -33,3 +33,4 @@ BIT (Prefer_No_AVX512) + BIT (MathVec_Prefer_No_AVX512) + BIT (Prefer_FSRM) + BIT (Avoid_Short_Distance_REP_MOVSB) ++BIT (Prefer_EVEX512) +diff --git a/sysdeps/x86_64/multiarch/ifunc-avx2.h b/sysdeps/x86_64/multiarch/ifunc-avx2.h +index 877f007dd6e38fe8..dad8377750c39ec3 100644 +--- a/sysdeps/x86_64/multiarch/ifunc-avx2.h ++++ b/sysdeps/x86_64/multiarch/ifunc-avx2.h +@@ -1,4 +1,4 @@ +-/* Common definition for ifunc selections optimized with SSE2 and AVX2. ++/* Common definition for ifunc selections optimized with SSE2, AVX2 and EVEX512. + All versions must be listed in ifunc-impl-list.c. + Copyright (C) 2017-2021 Free Software Foundation, Inc. + This file is part of the GNU C Library. +@@ -23,6 +23,9 @@ extern __typeof (REDIRECT_NAME) OPTIMIZE (sse2) attribute_hidden; + extern __typeof (REDIRECT_NAME) OPTIMIZE (avx2) attribute_hidden; + extern __typeof (REDIRECT_NAME) OPTIMIZE (avx2_rtm) attribute_hidden; + extern __typeof (REDIRECT_NAME) OPTIMIZE (evex) attribute_hidden; ++#ifdef USE_EVEX512 ++extern __typeof (REDIRECT_NAME) OPTIMIZE (evex512) attribute_hidden; ++#endif + + static inline void * + IFUNC_SELECTOR (void) +@@ -37,7 +40,13 @@ IFUNC_SELECTOR (void) + { + if (CPU_FEATURE_USABLE_P (cpu_features, AVX512VL) + && CPU_FEATURE_USABLE_P (cpu_features, AVX512BW)) +- return OPTIMIZE (evex); ++ { ++#ifdef USE_EVEX512 ++ if (CPU_FEATURES_ARCH_P (cpu_features, Prefer_EVEX512)) ++ return OPTIMIZE (evex512); ++#endif ++ return OPTIMIZE (evex); ++ } + + if (CPU_FEATURE_USABLE_P (cpu_features, RTM)) + return OPTIMIZE (avx2_rtm); +diff --git a/sysdeps/x86_64/multiarch/ifunc-evex.h b/sysdeps/x86_64/multiarch/ifunc-evex.h +index fc391edb8abc7d7f..440c1cc2e2cc0e73 100644 +--- a/sysdeps/x86_64/multiarch/ifunc-evex.h ++++ b/sysdeps/x86_64/multiarch/ifunc-evex.h +@@ -1,4 +1,4 @@ +-/* Common definition for ifunc selection optimized with EVEX. ++/* Common definition for ifunc selection optimized with EVEX and EVEX512. + All versions must be listed in ifunc-impl-list.c. + Copyright (C) 2017-2021 Free Software Foundation, Inc. + This file is part of the GNU C Library. +@@ -25,6 +25,9 @@ extern __typeof (REDIRECT_NAME) OPTIMIZE (avx2_rtm) attribute_hidden; + extern __typeof (REDIRECT_NAME) OPTIMIZE (evex) attribute_hidden; + extern __typeof (REDIRECT_NAME) OPTIMIZE (evex_rtm) attribute_hidden; + ++#ifdef USE_EVEX512 ++extern __typeof (REDIRECT_NAME) OPTIMIZE (evex512) attribute_hidden; ++#endif + + static inline void * + IFUNC_SELECTOR (void) +@@ -38,6 +41,11 @@ IFUNC_SELECTOR (void) + if (CPU_FEATURE_USABLE_P (cpu_features, AVX512VL) + && CPU_FEATURE_USABLE_P (cpu_features, AVX512BW)) + { ++#ifdef USE_EVEX512 ++ if (CPU_FEATURES_ARCH_P (cpu_features, Prefer_EVEX512)) ++ return OPTIMIZE (evex512); ++#endif ++ + if (CPU_FEATURE_USABLE_P (cpu_features, RTM)) + return OPTIMIZE (evex_rtm); + +diff --git a/sysdeps/x86_64/multiarch/memchr.c b/sysdeps/x86_64/multiarch/memchr.c +index 5a4131cb8fd6d860..ae1c2b121d62b9a2 100644 +--- a/sysdeps/x86_64/multiarch/memchr.c ++++ b/sysdeps/x86_64/multiarch/memchr.c +@@ -24,6 +24,7 @@ + # undef memchr + + # define SYMBOL_NAME memchr ++# define USE_EVEX512 1 + # include "ifunc-evex.h" + + libc_ifunc_redirected (__redirect_memchr, memchr, IFUNC_SELECTOR ()); +diff --git a/sysdeps/x86_64/multiarch/strchr.c b/sysdeps/x86_64/multiarch/strchr.c +index 691770f335b70b5e..de7a338d53047d78 100644 +--- a/sysdeps/x86_64/multiarch/strchr.c ++++ b/sysdeps/x86_64/multiarch/strchr.c +@@ -31,6 +31,7 @@ extern __typeof (REDIRECT_NAME) OPTIMIZE (sse2_no_bsf) attribute_hidden; + extern __typeof (REDIRECT_NAME) OPTIMIZE (avx2) attribute_hidden; + extern __typeof (REDIRECT_NAME) OPTIMIZE (avx2_rtm) attribute_hidden; + extern __typeof (REDIRECT_NAME) OPTIMIZE (evex) attribute_hidden; ++extern __typeof (REDIRECT_NAME) OPTIMIZE (evex512) attribute_hidden; + + static inline void * + IFUNC_SELECTOR (void) +@@ -43,7 +44,12 @@ IFUNC_SELECTOR (void) + { + if (CPU_FEATURE_USABLE_P (cpu_features, AVX512VL) + && CPU_FEATURE_USABLE_P (cpu_features, AVX512BW)) +- return OPTIMIZE (evex); ++ { ++ if (CPU_FEATURES_ARCH_P (cpu_features, Prefer_EVEX512)) ++ return OPTIMIZE (evex512); ++ ++ return OPTIMIZE (evex); ++ } + + if (CPU_FEATURE_USABLE_P (cpu_features, RTM)) + return OPTIMIZE (avx2_rtm); +diff --git a/sysdeps/x86_64/multiarch/strchrnul.c b/sysdeps/x86_64/multiarch/strchrnul.c +index 7631927dd44d8bf1..eadb2d41ed86636a 100644 +--- a/sysdeps/x86_64/multiarch/strchrnul.c ++++ b/sysdeps/x86_64/multiarch/strchrnul.c +@@ -26,6 +26,7 @@ + # undef strchrnul + + # define SYMBOL_NAME strchrnul ++# define USE_EVEX512 1 + # include "ifunc-avx2.h" + + libc_ifunc_redirected (__redirect_strchrnul, __strchrnul, +diff --git a/sysdeps/x86_64/multiarch/strlen.c b/sysdeps/x86_64/multiarch/strlen.c +index f438b00375c8e791..13a6e63a898784e9 100644 +--- a/sysdeps/x86_64/multiarch/strlen.c ++++ b/sysdeps/x86_64/multiarch/strlen.c +@@ -24,6 +24,7 @@ + # undef strlen + + # define SYMBOL_NAME strlen ++# define USE_EVEX512 1 + # include "ifunc-avx2.h" + + libc_ifunc_redirected (__redirect_strlen, strlen, IFUNC_SELECTOR ()); +diff --git a/sysdeps/x86_64/multiarch/strnlen.c b/sysdeps/x86_64/multiarch/strnlen.c +index 413c2f38ab01d205..02d97b437918f4ed 100644 +--- a/sysdeps/x86_64/multiarch/strnlen.c ++++ b/sysdeps/x86_64/multiarch/strnlen.c +@@ -26,6 +26,7 @@ + # undef strnlen + + # define SYMBOL_NAME strnlen ++# define USE_EVEX512 1 + # include "ifunc-avx2.h" + + libc_ifunc_redirected (__redirect_strnlen, __strnlen, IFUNC_SELECTOR ()); +diff --git a/sysdeps/x86_64/multiarch/strrchr.c b/sysdeps/x86_64/multiarch/strrchr.c +index d46b23543f08333f..2b5cc738d2d2a74d 100644 +--- a/sysdeps/x86_64/multiarch/strrchr.c ++++ b/sysdeps/x86_64/multiarch/strrchr.c +@@ -23,6 +23,7 @@ + # undef strrchr + + # define SYMBOL_NAME strrchr ++# define USE_EVEX512 1 + # include "ifunc-avx2.h" + + libc_ifunc_redirected (__redirect_strrchr, strrchr, IFUNC_SELECTOR ()); diff --git a/glibc-RHEL-175520-2.patch b/glibc-RHEL-175520-2.patch new file mode 100644 index 0000000..65197a2 --- /dev/null +++ b/glibc-RHEL-175520-2.patch @@ -0,0 +1,78 @@ +commit 0281c7a7ec8f3f46d8e6f5f3d7fca548946dbfce +Author: Noah Goldstein +Date: Sun Feb 13 19:40:37 2022 -0600 + + String: Strength memset tests in test-memset.c + + The prior sentinel logic was broken and was checking the SIMPLE_MEMSET + as opposed to the tested implementation. As well `s` (the test buffer) + was not reset between implementation tests so it was possible for a + buggy implementation to be hidden by a previously executed correct + one. + Reviewed-by: Siddhesh Poyarekar + +diff --git a/string/test-memset.c b/string/test-memset.c +index 82bfcd6ad4a7a0fb..fe89bb57c1c85892 100644 +--- a/string/test-memset.c ++++ b/string/test-memset.c +@@ -107,26 +107,28 @@ SIMPLE_MEMSET (CHAR *s, int c, size_t n) + } + + static void +-do_one_test (impl_t *impl, CHAR *s, int c __attribute ((unused)), size_t n) ++do_one_test (impl_t *impl, CHAR *s, int c __attribute ((unused)), size_t n, int space_below, int space_above) + { +- CHAR buf[n + 2]; +- CHAR *tstbuf = buf + 1; +- CHAR sentinel = c - 1; +- buf[0] = sentinel; +- buf[n + 1] = sentinel; ++ CHAR buf[n]; ++ CHAR sentinel = ~c; ++ if (space_below) ++ s[-1] = sentinel; ++ if (space_above) ++ s[n] = sentinel; ++ SIMPLE_MEMSET(s, ~c, n); + #ifdef TEST_BZERO +- simple_bzero (tstbuf, n); ++ simple_bzero (buf, n); + CALL (impl, s, n); +- if (memcmp (s, tstbuf, n) != 0 +- || buf[0] != sentinel +- || buf[n + 1] != sentinel) ++ if (memcmp (s, buf, n) != 0 ++ || (space_below && s[-1] != sentinel) ++ || (space_above && s[n] != sentinel)) + #else + CHAR *res = CALL (impl, s, c, n); + if (res != s +- || SIMPLE_MEMSET (tstbuf, c, n) != tstbuf +- || MEMCMP (s, tstbuf, n) != 0 +- || buf[0] != sentinel +- || buf[n + 1] != sentinel) ++ || SIMPLE_MEMSET (buf, c, n) != buf ++ || MEMCMP (s, buf, n) != 0 ++ || (space_below && s[-1] != sentinel) ++ || (space_above && s[n] != sentinel)) + #endif /* !TEST_BZERO */ + { + error (0, 0, "Wrong result in function %s", impl->name); +@@ -138,12 +140,16 @@ do_one_test (impl_t *impl, CHAR *s, int c __attribute ((unused)), size_t n) + static void + do_test (size_t align, int c, size_t len) + { ++ int space_below, space_above; + align &= 4095; + if ((align + len) * sizeof (CHAR) > page_size) + return; + ++ space_below = !!align; ++ space_above = !((align + len + 1) * sizeof (CHAR) > page_size); ++ + FOR_EACH_IMPL (impl, 0) +- do_one_test (impl, (CHAR *) (buf1) + align, c, len); ++ do_one_test (impl, (CHAR *) (buf1) + align, c, len, space_below, space_above); + } + + #ifndef TEST_BZERO diff --git a/glibc-RHEL-175520-3.patch b/glibc-RHEL-175520-3.patch new file mode 100644 index 0000000..8385164 --- /dev/null +++ b/glibc-RHEL-175520-3.patch @@ -0,0 +1,830 @@ +commit 67e3b0c63c35769c1ba28fa2a32446332bb4fcef +Author: Siddhesh Poyarekar +Date: Wed Apr 6 20:53:24 2022 +0530 + + tests/string: Drop simple/stupid/builtin tests + + In most cases the simple/stupid/builtin functions were in there to + benchmark optimized implementations against. Only in some cases the + functions are used to check expected results. + + Remove these tests from IMPL() and only keep them in wherever they're + used for a specific purpose, e.g. to generate expected results. + + This improves timing of `make subdirs=string` by over a minute and a + half (over 15%) on a Whiskey Lake laptop. + + Signed-off-by: Siddhesh Poyarekar + Reviewed-by: Noah Goldstein + +Conflicts: + string/test-memcpy-support.h + (modified string/test-memcpy.c instead) + string/test-memset.c + (fixup context) + +diff --git a/string/test-memccpy.c b/string/test-memccpy.c +index e0b1d5ae5283ccb1..5f7eeda5a224d7f5 100644 +--- a/string/test-memccpy.c ++++ b/string/test-memccpy.c +@@ -21,13 +21,9 @@ + #define TEST_NAME "memccpy" + #include "test-string.h" + +-void *simple_memccpy (void *, const void *, int, size_t); +-void *stupid_memccpy (void *, const void *, int, size_t); +- +-IMPL (stupid_memccpy, 0) +-IMPL (simple_memccpy, 0) + IMPL (memccpy, 1) + ++/* Naive implementation to verify results. */ + void * + simple_memccpy (void *dst, const void *src, int c, size_t n) + { +@@ -41,18 +37,6 @@ simple_memccpy (void *dst, const void *src, int c, size_t n) + return NULL; + } + +-void * +-stupid_memccpy (void *dst, const void *src, int c, size_t n) +-{ +- void *p = memchr (src, c, n); +- +- if (p != NULL) +- return mempcpy (dst, src, p - src + 1); +- +- memcpy (dst, src, n); +- return NULL; +-} +- + typedef void *(*proto_t) (void *, const void *, int c, size_t); + + static void +diff --git a/string/test-memchr.c b/string/test-memchr.c +index ce964284aa16508f..6e5f7795a09febcb 100644 +--- a/string/test-memchr.c ++++ b/string/test-memchr.c +@@ -45,11 +45,10 @@ + #endif /* WIDE */ + + typedef CHAR *(*proto_t) (const CHAR *, int, size_t); +-CHAR *SIMPLE_MEMCHR (const CHAR *, int, size_t); + +-IMPL (SIMPLE_MEMCHR, 0) + IMPL (MEMCHR, 1) + ++/* Naive implementation to verify results. */ + CHAR * + SIMPLE_MEMCHR (const CHAR *s, int c, size_t n) + { +diff --git a/string/test-memcmp.c b/string/test-memcmp.c +index fbda26a41e9731a5..480b3a0adcf3a644 100644 +--- a/string/test-memcmp.c ++++ b/string/test-memcmp.c +@@ -73,7 +73,6 @@ simple_memcmp (const char *s1, const char *s2, size_t n) + + typedef int (*proto_t) (const CHAR *, const CHAR *, size_t); + +-IMPL (SIMPLE_MEMCMP, 0) + IMPL (MEMCMP, 1) + + static int +diff --git a/string/test-memcpy.c b/string/test-memcpy.c +index c9dfc88fedddc775..63c930627821ed93 100644 +--- a/string/test-memcpy.c ++++ b/string/test-memcpy.c +@@ -24,13 +24,9 @@ + # define TEST_NAME "memcpy" + # include "test-string.h" + +-char *simple_memcpy (char *, const char *, size_t); +-char *builtin_memcpy (char *, const char *, size_t); +- +-IMPL (simple_memcpy, 0) +-IMPL (builtin_memcpy, 0) + IMPL (memcpy, 1) + ++/* Naive implementation to verify results. */ + char * + simple_memcpy (char *dst, const char *src, size_t n) + { +@@ -40,11 +36,6 @@ simple_memcpy (char *dst, const char *src, size_t n) + return ret; + } + +-char * +-builtin_memcpy (char *dst, const char *src, size_t n) +-{ +- return __builtin_memcpy (dst, src, n); +-} + #endif + + typedef char *(*proto_t) (char *, const char *, size_t); +diff --git a/string/test-memmem.c b/string/test-memmem.c +index 624d3106bfebf628..260116fcd80072cc 100644 +--- a/string/test-memmem.c ++++ b/string/test-memmem.c +@@ -24,11 +24,10 @@ + #include "test-string.h" + + typedef char *(*proto_t) (const void *, size_t, const void *, size_t); +-void *simple_memmem (const void *, size_t, const void *, size_t); + +-IMPL (simple_memmem, 0) + IMPL (memmem, 1) + ++/* Naive implementation to verify results. */ + void * + simple_memmem (const void *haystack, size_t haystack_len, const void *needle, + size_t needle_len) +diff --git a/string/test-memmove.c b/string/test-memmove.c +index b271248b1d903634..c4f9f6d11330016d 100644 +--- a/string/test-memmove.c ++++ b/string/test-memmove.c +@@ -30,23 +30,23 @@ char *simple_memmove (char *, const char *, size_t); + + #ifdef TEST_BCOPY + typedef void (*proto_t) (const char *, char *, size_t); +-void simple_bcopy (const char *, char *, size_t); + +-IMPL (simple_bcopy, 0) + IMPL (bcopy, 1) + ++/* Naive implementation to verify results. */ + void + simple_bcopy (const char *src, char *dst, size_t n) + { + simple_memmove (dst, src, n); + } ++ + #else + typedef char *(*proto_t) (char *, const char *, size_t); + +-IMPL (simple_memmove, 0) + IMPL (memmove, 1) + #endif + ++/* Naive implementation to verify results. */ + char * + inhibit_loop_to_libcall + simple_memmove (char *dst, const char *src, size_t n) +diff --git a/string/test-mempcpy.c b/string/test-mempcpy.c +index f80b014d41921462..1554af9b1587d5ca 100644 +--- a/string/test-mempcpy.c ++++ b/string/test-mempcpy.c +@@ -23,11 +23,9 @@ + #define TEST_NAME "mempcpy" + #include "test-string.h" + +-char *simple_mempcpy (char *, const char *, size_t); +- +-IMPL (simple_mempcpy, 0) + IMPL (mempcpy, 1) + ++/* Naive implementation to verify results. */ + char * + simple_mempcpy (char *dst, const char *src, size_t n) + { +diff --git a/string/test-memrchr.c b/string/test-memrchr.c +index 1a3b9ce69d2729c2..5166f22a37893c98 100644 +--- a/string/test-memrchr.c ++++ b/string/test-memrchr.c +@@ -22,11 +22,10 @@ + #include "test-string.h" + + typedef char *(*proto_t) (const char *, int, size_t); +-char *simple_memrchr (const char *, int, size_t); + +-IMPL (simple_memrchr, 0) + IMPL (memrchr, 1) + ++/* Naive implementation to verify results. */ + char * + simple_memrchr (const char *s, int c, size_t n) + { +diff --git a/string/test-memset.c b/string/test-memset.c +index fe89bb57c1c85892..bca571860ff82a19 100644 +--- a/string/test-memset.c ++++ b/string/test-memset.c +@@ -51,51 +51,19 @@ + # define BIG_CHAR WCHAR_MAX + #endif /* WIDE */ + +-CHAR *SIMPLE_MEMSET (CHAR *, int, size_t); +- + #ifdef TEST_BZERO + typedef void (*proto_t) (char *, size_t); +-void simple_bzero (char *, size_t); +-void builtin_bzero (char *, size_t); +- +-IMPL (simple_bzero, 0) +-IMPL (builtin_bzero, 0) +-#ifdef TEST_EXPLICIT_BZERO ++# ifdef TEST_EXPLICIT_BZERO + IMPL (explicit_bzero, 1) +-#else ++# else + IMPL (bzero, 1) +-#endif +- +-void +-simple_bzero (char *s, size_t n) +-{ +- SIMPLE_MEMSET (s, 0, n); +-} +- +-void +-builtin_bzero (char *s, size_t n) +-{ +- __builtin_bzero (s, n); +-} ++# endif + #else + typedef CHAR *(*proto_t) (CHAR *, int, size_t); +- +-IMPL (SIMPLE_MEMSET, 0) +-# ifndef WIDE +-char *builtin_memset (char *, int, size_t); +-IMPL (builtin_memset, 0) +-# endif /* !WIDE */ + IMPL (MEMSET, 1) +- +-# ifndef WIDE +-char * +-builtin_memset (char *s, int c, size_t n) +-{ +- return __builtin_memset (s, c, n); +-} +-# endif /* !WIDE */ + #endif /* !TEST_BZERO */ + ++/* Naive implementation to verify results. */ + CHAR * + inhibit_loop_to_libcall + SIMPLE_MEMSET (CHAR *s, int c, size_t n) +@@ -117,7 +85,7 @@ do_one_test (impl_t *impl, CHAR *s, int c __attribute ((unused)), size_t n, int + s[n] = sentinel; + SIMPLE_MEMSET(s, ~c, n); + #ifdef TEST_BZERO +- simple_bzero (buf, n); ++ SIMPLE_MEMSET (buf, 0, n); + CALL (impl, s, n); + if (memcmp (s, buf, n) != 0 + || (space_below && s[-1] != sentinel) +diff --git a/string/test-strcasecmp.c b/string/test-strcasecmp.c +index b380150cb74b5a8e..f4e002510cb6797b 100644 +--- a/string/test-strcasecmp.c ++++ b/string/test-strcasecmp.c +@@ -24,14 +24,11 @@ + #include "test-string.h" + + typedef int (*proto_t) (const char *, const char *); +-static int simple_strcasecmp (const char *, const char *); +-static int stupid_strcasecmp (const char *, const char *); + +-IMPL (stupid_strcasecmp, 0) +-IMPL (simple_strcasecmp, 0) + IMPL (strcasecmp, 1) + +-static int ++/* Naive implementation to verify results. */ ++int + simple_strcasecmp (const char *s1, const char *s2) + { + int ret; +@@ -43,24 +40,6 @@ simple_strcasecmp (const char *s1, const char *s2) + return ret; + } + +-static int +-stupid_strcasecmp (const char *s1, const char *s2) +-{ +- size_t ns1 = strlen (s1) + 1, ns2 = strlen (s2) + 1; +- size_t n = ns1 < ns2 ? ns1 : ns2; +- int ret = 0; +- +- while (n--) +- { +- if ((ret = ((unsigned char) tolower (*s1) +- - (unsigned char) tolower (*s2))) != 0) +- break; +- ++s1; +- ++s2; +- } +- return ret; +-} +- + static void + do_one_test (impl_t *impl, const char *s1, const char *s2, int exp_result) + { +diff --git a/string/test-strcasestr.c b/string/test-strcasestr.c +index 045156e6b8b87f51..0cdc5dcd97974694 100644 +--- a/string/test-strcasestr.c ++++ b/string/test-strcasestr.c +@@ -22,15 +22,15 @@ + #include "test-string.h" + + +-#define STRCASESTR simple_strcasestr ++#define STRCASESTR c_strcasestr + #define NO_ALIAS + #define __strncasecmp strncasecmp + #define __strnlen strnlen + #include "strcasestr.c" + +- ++/* Naive implementation to verify results. */ + static char * +-stupid_strcasestr (const char *s1, const char *s2) ++simple_strcasestr (const char *s1, const char *s2) + { + ssize_t s1len = strlen (s1); + ssize_t s2len = strlen (s2); +@@ -54,8 +54,7 @@ stupid_strcasestr (const char *s1, const char *s2) + + typedef char *(*proto_t) (const char *, const char *); + +-IMPL (stupid_strcasestr, 0) +-IMPL (simple_strcasestr, 0) ++IMPL (c_strcasestr, 0) + IMPL (strcasestr, 1) + + +@@ -130,7 +129,7 @@ check1 (void) + const char s2[] = "OK"; + char *exp_result; + +- exp_result = stupid_strcasestr (s1, s2); ++ exp_result = simple_strcasestr (s1, s2); + FOR_EACH_IMPL (impl, 0) + check_result (impl, s1, s2, exp_result); + } +diff --git a/string/test-strcat.c b/string/test-strcat.c +index dea1558b0c64c311..97cc1a2136800ee9 100644 +--- a/string/test-strcat.c ++++ b/string/test-strcat.c +@@ -55,11 +55,10 @@ + #endif /* WIDE */ + + typedef CHAR *(*proto_t) (CHAR *, const CHAR *); +-CHAR *SIMPLE_STRCAT (CHAR *, const CHAR *); + +-IMPL (SIMPLE_STRCAT, 0) + IMPL (STRCAT, 1) + ++/* Naive implementation to verify results. */ + CHAR * + SIMPLE_STRCAT (CHAR *dst, const CHAR *src) + { +diff --git a/string/test-strchr.c b/string/test-strchr.c +index 6c8ca54a7df8fa2b..f24ade10485853c3 100644 +--- a/string/test-strchr.c ++++ b/string/test-strchr.c +@@ -37,7 +37,6 @@ + #ifndef WIDE + # ifdef USE_FOR_STRCHRNUL + # define STRCHR strchrnul +-# define stupid_STRCHR stupid_STRCHRNUL + # define simple_STRCHR simple_STRCHRNUL + # else + # define STRCHR strchr +@@ -53,7 +52,6 @@ + # include + # ifdef USE_FOR_STRCHRNUL + # define STRCHR wcschrnul +-# define stupid_STRCHR stupid_WCSCHRNUL + # define simple_STRCHR simple_WCSCHRNUL + # else + # define STRCHR wcschr +@@ -76,17 +74,9 @@ + + typedef CHAR *(*proto_t) (const CHAR *, int); + ++/* Naive implementation to verify results. */ + CHAR * + simple_STRCHR (const CHAR *s, int c) +-{ +- for (; *s != (CHAR) c; ++s) +- if (*s == '\0') +- return NULLRET ((CHAR *) s); +- return (CHAR *) s; +-} +- +-CHAR * +-stupid_STRCHR (const CHAR *s, int c) + { + size_t n = STRLEN (s) + 1; + +@@ -96,8 +86,6 @@ stupid_STRCHR (const CHAR *s, int c) + return NULLRET ((CHAR *) s - 1); + } + +-IMPL (stupid_STRCHR, 0) +-IMPL (simple_STRCHR, 0) + IMPL (STRCHR, 1) + + static int +@@ -233,7 +221,7 @@ check1 (void) + { + CHAR s[] __attribute__((aligned(16))) = L ("\xff"); + CHAR c = L ('\xfe'); +- CHAR *exp_result = stupid_STRCHR (s, c); ++ CHAR *exp_result = simple_STRCHR (s, c); + + FOR_EACH_IMPL (impl, 0) + check_result (impl, s, c, exp_result); +diff --git a/string/test-strcmp.c b/string/test-strcmp.c +index a6b5344f5710effc..7eeaf472de5ce178 100644 +--- a/string/test-strcmp.c ++++ b/string/test-strcmp.c +@@ -98,7 +98,6 @@ simple_strcmp (const char *s1, const char *s2) + + typedef int (*proto_t) (const CHAR *, const CHAR *); + +-IMPL (SIMPLE_STRCMP, 1) + IMPL (STRCMP, 1) + + static int +diff --git a/string/test-strcpy.c b/string/test-strcpy.c +index 4c0a04f85d70104f..271b9251154fb0f6 100644 +--- a/string/test-strcpy.c ++++ b/string/test-strcpy.c +@@ -56,11 +56,9 @@ + # define STRCPY wcscpy + # endif + +-CHAR *SIMPLE_STRCPY (CHAR *, const CHAR *); +- +-IMPL (SIMPLE_STRCPY, 0) + IMPL (STRCPY, 1) + ++/* Naive implementation to verify results. */ + CHAR * + SIMPLE_STRCPY (CHAR *dst, const CHAR *src) + { +diff --git a/string/test-strlen.c b/string/test-strlen.c +index c9a7afb339a28114..06919fa9ce289b39 100644 +--- a/string/test-strlen.c ++++ b/string/test-strlen.c +@@ -39,6 +39,7 @@ + + typedef size_t (*proto_t) (const CHAR *); + ++/* Naive implementation to verify results. */ + size_t + simple_STRLEN (const CHAR *s) + { +@@ -57,7 +58,6 @@ builtin_strlen (const CHAR *p) + IMPL (builtin_strlen, 0) + #endif + +-IMPL (simple_STRLEN, 0) + IMPL (STRLEN, 1) + + +diff --git a/string/test-strncasecmp.c b/string/test-strncasecmp.c +index ace94e83eca9bd72..6662dab672378ae1 100644 +--- a/string/test-strncasecmp.c ++++ b/string/test-strncasecmp.c +@@ -25,12 +25,10 @@ + + typedef int (*proto_t) (const char *, const char *, size_t); + static int simple_strncasecmp (const char *, const char *, size_t); +-static int stupid_strncasecmp (const char *, const char *, size_t); + +-IMPL (stupid_strncasecmp, 0) +-IMPL (simple_strncasecmp, 0) + IMPL (strncasecmp, 1) + ++/* Naive implementation to verify results. */ + static int + simple_strncasecmp (const char *s1, const char *s2, size_t n) + { +@@ -50,27 +48,6 @@ simple_strncasecmp (const char *s1, const char *s2, size_t n) + return ret; + } + +-static int +-stupid_strncasecmp (const char *s1, const char *s2, size_t max) +-{ +- size_t ns1 = strlen (s1) + 1; +- size_t ns2 = strlen (s2) + 1; +- size_t n = ns1 < ns2 ? ns1 : ns2; +- if (n > max) +- n = max; +- int ret = 0; +- +- while (n--) +- { +- if ((ret = ((unsigned char) tolower (*s1) +- - (unsigned char) tolower (*s2))) != 0) +- break; +- ++s1; +- ++s2; +- } +- return ret; +-} +- + static int + check_result (impl_t *impl, const char *s1, const char *s2, size_t n, + int exp_result) +diff --git a/string/test-strncat.c b/string/test-strncat.c +index 37ea26ea057938a4..982c1dff0421ce45 100644 +--- a/string/test-strncat.c ++++ b/string/test-strncat.c +@@ -29,7 +29,6 @@ + # define CHAR char + # define UCHAR unsigned char + # define SIMPLE_STRNCAT simple_strncat +-# define STUPID_STRNCAT stupid_strncat + # define STRLEN strlen + # define MEMSET memset + # define MEMCPY memcpy +@@ -42,7 +41,6 @@ + # define CHAR wchar_t + # define UCHAR wchar_t + # define SIMPLE_STRNCAT simple_wcsncat +-# define STUPID_STRNCAT stupid_wcsncat + # define STRLEN wcslen + # define MEMSET wmemset + # define MEMCPY wmemcpy +@@ -52,14 +50,12 @@ + #endif /* WIDE */ + + typedef CHAR *(*proto_t) (CHAR *, const CHAR *, size_t); +-CHAR *STUPID_STRNCAT (CHAR *, const CHAR *, size_t); +-CHAR *SIMPLE_STRNCAT (CHAR *, const CHAR *, size_t); + +-IMPL (STUPID_STRNCAT, 0) + IMPL (STRNCAT, 2) + ++/* Naive implementation to verify results. */ + CHAR * +-STUPID_STRNCAT (CHAR *dst, const CHAR *src, size_t n) ++SIMPLE_STRNCAT (CHAR *dst, const CHAR *src, size_t n) + { + CHAR *ret = dst; + while (*dst++ != '\0'); +diff --git a/string/test-strncmp.c b/string/test-strncmp.c +index a761136e94d65efc..5e481d260b52ff9d 100644 +--- a/string/test-strncmp.c ++++ b/string/test-strncmp.c +@@ -85,7 +85,6 @@ simple_strncmp (const char *s1, const char *s2, size_t n) + + typedef int (*proto_t) (const CHAR *, const CHAR *, size_t); + +-IMPL (SIMPLE_STRNCMP, 0) + IMPL (STRNCMP, 1) + + +diff --git a/string/test-strncpy.c b/string/test-strncpy.c +index 84a3b83bbc185802..7ee1b53ef5a219fb 100644 +--- a/string/test-strncpy.c ++++ b/string/test-strncpy.c +@@ -48,21 +48,16 @@ + # include "test-string.h" + # ifndef WIDE + # define SIMPLE_STRNCPY simple_strncpy +-# define STUPID_STRNCPY stupid_strncpy + # define STRNCPY strncpy + # else + # define SIMPLE_STRNCPY simple_wcsncpy +-# define STUPID_STRNCPY stupid_wcsncpy + # define STRNCPY wcsncpy + # endif /* WIDE */ + +-CHAR *SIMPLE_STRNCPY (CHAR *, const CHAR *, size_t); +-CHAR *STUPID_STRNCPY (CHAR *, const CHAR *, size_t); + +-IMPL (STUPID_STRNCPY, 0) +-IMPL (SIMPLE_STRNCPY, 0) + IMPL (STRNCPY, 1) + ++/* Naive implementation to verify results. */ + CHAR * + SIMPLE_STRNCPY (CHAR *dst, const CHAR *src, size_t n) + { +@@ -77,18 +72,6 @@ SIMPLE_STRNCPY (CHAR *dst, const CHAR *src, size_t n) + return ret; + } + +-CHAR * +-STUPID_STRNCPY (CHAR *dst, const CHAR *src, size_t n) +-{ +- size_t nc = STRNLEN (src, n); +- size_t i; +- +- for (i = 0; i < nc; ++i) +- dst[i] = src[i]; +- for (; i < n; ++i) +- dst[i] = '\0'; +- return dst; +-} + #endif /* !STRNCPY_RESULT */ + + typedef CHAR *(*proto_t) (CHAR *, const CHAR *, size_t); +diff --git a/string/test-strnlen.c b/string/test-strnlen.c +index eac84cd17526d5d9..18d1c93f3e0cd315 100644 +--- a/string/test-strnlen.c ++++ b/string/test-strnlen.c +@@ -43,11 +43,10 @@ + #endif /* !WIDE */ + + typedef size_t (*proto_t) (const CHAR *, size_t); +-size_t SIMPLE_STRNLEN (const CHAR *, size_t); + +-IMPL (SIMPLE_STRNLEN, 0) + IMPL (STRNLEN, 1) + ++/* Naive implementation to verify results. */ + size_t + SIMPLE_STRNLEN (const CHAR *s, size_t maxlen) + { +diff --git a/string/test-strpbrk.c b/string/test-strpbrk.c +index 3f81c2edc35af161..89586a9ab685eec6 100644 +--- a/string/test-strpbrk.c ++++ b/string/test-strpbrk.c +@@ -48,22 +48,17 @@ + # ifndef WIDE + # define STRPBRK strpbrk + # define SIMPLE_STRPBRK simple_strpbrk +-# define STUPID_STRPBRK stupid_strpbrk + # else + # include + # define STRPBRK wcspbrk + # define SIMPLE_STRPBRK simple_wcspbrk +-# define STUPID_STRPBRK stupid_wcspbrk + # endif /* WIDE */ + + typedef CHAR *(*proto_t) (const CHAR *, const CHAR *); +-CHAR *SIMPLE_STRPBRK (const CHAR *, const CHAR *); +-CHAR *STUPID_STRPBRK (const CHAR *, const CHAR *); + +-IMPL (STUPID_STRPBRK, 0) +-IMPL (SIMPLE_STRPBRK, 0) + IMPL (STRPBRK, 1) + ++/* Naive implementation to verify results. */ + CHAR * + SIMPLE_STRPBRK (const CHAR *s, const CHAR *rej) + { +@@ -73,22 +68,10 @@ SIMPLE_STRPBRK (const CHAR *s, const CHAR *rej) + while ((c = *s++) != '\0') + for (r = rej; *r != '\0'; ++r) + if (*r == c) +- return (CHAR *) s - 1; ++ return (CHAR *) s - 1; + return NULL; + } + +-CHAR * +-STUPID_STRPBRK (const CHAR *s, const CHAR *rej) +-{ +- size_t ns = STRLEN (s), nrej = STRLEN (rej); +- size_t i, j; +- +- for (i = 0; i < ns; ++i) +- for (j = 0; j < nrej; ++j) +- if (s[i] == rej[j]) +- return (CHAR *) s + i; +- return NULL; +-} + #endif /* !STRPBRK_RESULT */ + + static void +diff --git a/string/test-strrchr.c b/string/test-strrchr.c +index 9f85dc681ef075b8..d6018f2324eb8ad0 100644 +--- a/string/test-strrchr.c ++++ b/string/test-strrchr.c +@@ -45,11 +45,10 @@ + #endif + + typedef CHAR *(*proto_t) (const CHAR *, int); +-CHAR *SIMPLE_STRRCHR (const CHAR *, int); + +-IMPL (SIMPLE_STRRCHR, 0) + IMPL (STRRCHR, 1) + ++/* Naive implementation to verify results. */ + CHAR * + SIMPLE_STRRCHR (const CHAR *s, int c) + { +diff --git a/string/test-strspn.c b/string/test-strspn.c +index ccbed1ca2f6cf8f0..24d79770cde1f28d 100644 +--- a/string/test-strspn.c ++++ b/string/test-strspn.c +@@ -30,7 +30,6 @@ + # define CHAR char + # define UCHAR unsigned char + # define SIMPLE_STRSPN simple_strspn +-# define STUPID_STRSPN stupid_strspn + # define STRLEN strlen + # define STRCHR strchr + # define BIG_CHAR CHAR_MAX +@@ -41,7 +40,6 @@ + # define CHAR wchar_t + # define UCHAR wchar_t + # define SIMPLE_STRSPN simple_wcsspn +-# define STUPID_STRSPN stupid_wcsspn + # define STRLEN wcslen + # define STRCHR wcschr + # define BIG_CHAR WCHAR_MAX +@@ -49,13 +47,10 @@ + #endif /* WIDE */ + + typedef size_t (*proto_t) (const CHAR *, const CHAR *); +-size_t SIMPLE_STRSPN (const CHAR *, const CHAR *); +-size_t STUPID_STRSPN (const CHAR *, const CHAR *); + +-IMPL (STUPID_STRSPN, 0) +-IMPL (SIMPLE_STRSPN, 0) + IMPL (STRSPN, 1) + ++/* Naive implementation to verify results. */ + size_t + SIMPLE_STRSPN (const CHAR *s, const CHAR *acc) + { +@@ -73,23 +68,6 @@ SIMPLE_STRSPN (const CHAR *s, const CHAR *acc) + return s - str - 1; + } + +-size_t +-STUPID_STRSPN (const CHAR *s, const CHAR *acc) +-{ +- size_t ns = STRLEN (s), nacc = STRLEN (acc); +- size_t i, j; +- +- for (i = 0; i < ns; ++i) +- { +- for (j = 0; j < nacc; ++j) +- if (s[i] == acc[j]) +- break; +- if (j == nacc) +- return i; +- } +- return i; +-} +- + static void + do_one_test (impl_t *impl, const CHAR *s, const CHAR *acc, size_t exp_res) + { +diff --git a/string/test-strstr.c b/string/test-strstr.c +index 2c40d929f27843fa..958c4842af92c1f1 100644 +--- a/string/test-strstr.c ++++ b/string/test-strstr.c +@@ -22,14 +22,14 @@ + #include "test-string.h" + + +-#define STRSTR simple_strstr ++#define STRSTR c_strstr + #define libc_hidden_builtin_def(arg) /* nothing */ + #define __strnlen strnlen + #include "strstr.c" + +- ++/* Naive implementation to verify results. */ + static char * +-stupid_strstr (const char *s1, const char *s2) ++simple_strstr (const char *s1, const char *s2) + { + ssize_t s1len = strlen (s1); + ssize_t s2len = strlen (s2); +@@ -53,8 +53,7 @@ stupid_strstr (const char *s1, const char *s2) + + typedef char *(*proto_t) (const char *, const char *); + +-IMPL (stupid_strstr, 0) +-IMPL (simple_strstr, 0) ++IMPL (c_strstr, 0) + IMPL (strstr, 1) + + +@@ -131,7 +130,7 @@ check1 (void) + const char s2[] = "_EF_BF_BD_EF_BF_BD_EF_BF_BD_EF_BF_BD_EF_BF_BD"; + char *exp_result; + +- exp_result = stupid_strstr (s1, s2); ++ exp_result = simple_strstr (s1, s2); + FOR_EACH_IMPL (impl, 0) + check_result (impl, s1, s2, exp_result); + } +@@ -164,7 +163,7 @@ check2 (void) + char *s2_page_cross = (void *) buf2 + page_size_real - 8; + strcpy (s2_page_cross, s2_stack); + +- exp_result = stupid_strstr (s1_stack, s2_stack); ++ exp_result = simple_strstr (s1_stack, s2_stack); + FOR_EACH_IMPL (impl, 0) + { + check_result (impl, s1_stack, s2_stack, exp_result); +@@ -202,7 +201,7 @@ pr23637 (void) + /* Ensure we don't match at the first 'x'. */ + h[0] = 'x'; + +- char *exp_result = stupid_strstr (h, n); ++ char *exp_result = simple_strstr (h, n); + FOR_EACH_IMPL (impl, 0) + check_result (impl, h, n, exp_result); + } diff --git a/glibc-RHEL-175520-4.patch b/glibc-RHEL-175520-4.patch new file mode 100644 index 0000000..e9cd9a1 --- /dev/null +++ b/glibc-RHEL-175520-4.patch @@ -0,0 +1,716 @@ +commit 52ab7604db35e0421bc3d2468a3af52b2c513a7b +Author: Noah Goldstein +Date: Fri Oct 14 22:00:25 2022 -0500 + + x86: Update VEC macros to complete API for evex/evex512 impls + + 1) Copy so that backport will be easier. + 2) Make section only define if there is not a previous definition + 3) Add `VEC_lo` definition for proper reg-width but in the + ymm/zmm0-15 range. + 4) Add macros for accessing GPRs based on VEC_SIZE + This is to make it easier to do think like: + ``` + vpcmpb %VEC(0), %VEC(1), %k0 + kmov{d|q} %k0, %{eax|rax} + test %{eax|rax} + ``` + It adds macro s.t any GPR can get the proper width with: + `V{upcase_GPR_name}` + + and any mask insn can get the proper width with: + `{upcase_mask_insn_without_postfix}` + + This commit does not change libc.so + + Tested build on x86-64 + +diff --git a/sysdeps/x86_64/multiarch/reg-macros.h b/sysdeps/x86_64/multiarch/reg-macros.h +new file mode 100644 +index 0000000000000000..c8ea330256d00e5c +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/reg-macros.h +@@ -0,0 +1,168 @@ ++/* This file was generated by: gen-reg-macros.py. ++ ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef _REG_MACROS_H ++#define _REG_MACROS_H 1 ++ ++#define rax_8 al ++#define rax_16 ax ++#define rax_32 eax ++#define rax_64 rax ++#define rbx_8 bl ++#define rbx_16 bx ++#define rbx_32 ebx ++#define rbx_64 rbx ++#define rcx_8 cl ++#define rcx_16 cx ++#define rcx_32 ecx ++#define rcx_64 rcx ++#define rdx_8 dl ++#define rdx_16 dx ++#define rdx_32 edx ++#define rdx_64 rdx ++#define rbp_8 bpl ++#define rbp_16 bp ++#define rbp_32 ebp ++#define rbp_64 rbp ++#define rsp_8 spl ++#define rsp_16 sp ++#define rsp_32 esp ++#define rsp_64 rsp ++#define rsi_8 sil ++#define rsi_16 si ++#define rsi_32 esi ++#define rsi_64 rsi ++#define rdi_8 dil ++#define rdi_16 di ++#define rdi_32 edi ++#define rdi_64 rdi ++#define r8_8 r8b ++#define r8_16 r8w ++#define r8_32 r8d ++#define r8_64 r8 ++#define r9_8 r9b ++#define r9_16 r9w ++#define r9_32 r9d ++#define r9_64 r9 ++#define r10_8 r10b ++#define r10_16 r10w ++#define r10_32 r10d ++#define r10_64 r10 ++#define r11_8 r11b ++#define r11_16 r11w ++#define r11_32 r11d ++#define r11_64 r11 ++#define r12_8 r12b ++#define r12_16 r12w ++#define r12_32 r12d ++#define r12_64 r12 ++#define r13_8 r13b ++#define r13_16 r13w ++#define r13_32 r13d ++#define r13_64 r13 ++#define r14_8 r14b ++#define r14_16 r14w ++#define r14_32 r14d ++#define r14_64 r14 ++#define r15_8 r15b ++#define r15_16 r15w ++#define r15_32 r15d ++#define r15_64 r15 ++ ++#define kmov_8 kmovb ++#define kmov_16 kmovw ++#define kmov_32 kmovd ++#define kmov_64 kmovq ++#define kortest_8 kortestb ++#define kortest_16 kortestw ++#define kortest_32 kortestd ++#define kortest_64 kortestq ++#define kor_8 korb ++#define kor_16 korw ++#define kor_32 kord ++#define kor_64 korq ++#define ktest_8 ktestb ++#define ktest_16 ktestw ++#define ktest_32 ktestd ++#define ktest_64 ktestq ++#define kand_8 kandb ++#define kand_16 kandw ++#define kand_32 kandd ++#define kand_64 kandq ++#define kxor_8 kxorb ++#define kxor_16 kxorw ++#define kxor_32 kxord ++#define kxor_64 kxorq ++#define knot_8 knotb ++#define knot_16 knotw ++#define knot_32 knotd ++#define knot_64 knotq ++#define kxnor_8 kxnorb ++#define kxnor_16 kxnorw ++#define kxnor_32 kxnord ++#define kxnor_64 kxnorq ++#define kunpack_8 kunpackbw ++#define kunpack_16 kunpackwd ++#define kunpack_32 kunpackdq ++ ++/* Common API for accessing proper width GPR is V{upcase_GPR_name}. */ ++#define VRAX VGPR(rax) ++#define VRBX VGPR(rbx) ++#define VRCX VGPR(rcx) ++#define VRDX VGPR(rdx) ++#define VRBP VGPR(rbp) ++#define VRSP VGPR(rsp) ++#define VRSI VGPR(rsi) ++#define VRDI VGPR(rdi) ++#define VR8 VGPR(r8) ++#define VR9 VGPR(r9) ++#define VR10 VGPR(r10) ++#define VR11 VGPR(r11) ++#define VR12 VGPR(r12) ++#define VR13 VGPR(r13) ++#define VR14 VGPR(r14) ++#define VR15 VGPR(r15) ++ ++/* Common API for accessing proper width mask insn is {upcase_mask_insn}. */ ++#define KMOV VKINSN(kmov) ++#define KORTEST VKINSN(kortest) ++#define KOR VKINSN(kor) ++#define KTEST VKINSN(ktest) ++#define KAND VKINSN(kand) ++#define KXOR VKINSN(kxor) ++#define KNOT VKINSN(knot) ++#define KXNOR VKINSN(kxnor) ++#define KUNPACK VKINSN(kunpack) ++ ++#ifdef USE_WIDE_CHAR ++# define REG_WIDTH 32 ++#else ++# define REG_WIDTH VEC_SIZE ++#endif ++ ++#define VPASTER(x, y) x##_##y ++#define VEVALUATOR(x, y) VPASTER(x, y) ++ ++#define VGPR_SZ(reg_name, reg_size) VEVALUATOR(reg_name, reg_size) ++#define VKINSN_SZ(insn, reg_size) VEVALUATOR(insn, reg_size) ++ ++#define VGPR(reg_name) VGPR_SZ(reg_name, REG_WIDTH) ++#define VKINSN(mask_insn) VKINSN_SZ(mask_insn, REG_WIDTH) ++ ++#endif +diff --git a/sysdeps/x86_64/multiarch/scripts/gen-reg-macros.py b/sysdeps/x86_64/multiarch/scripts/gen-reg-macros.py +new file mode 100644 +index 0000000000000000..9fb6903212492034 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/scripts/gen-reg-macros.py +@@ -0,0 +1,133 @@ ++#!/usr/bin/python3 ++# Copyright (C) 2022 Free Software Foundation, Inc. ++# This file is part of the GNU C Library. ++# ++# The GNU C Library is free software; you can redistribute it and/or ++# modify it under the terms of the GNU Lesser General Public ++# License as published by the Free Software Foundation; either ++# version 2.1 of the License, or (at your option) any later version. ++# ++# The GNU C Library is distributed in the hope that it will be useful, ++# but WITHOUT ANY WARRANTY; without even the implied warranty of ++# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++# Lesser General Public License for more details. ++# ++# You should have received a copy of the GNU Lesser General Public ++# License along with the GNU C Library; if not, see ++# . ++"""Generate macros for getting GPR name of a certain size ++ ++Inputs: None ++Output: Prints header fill to stdout ++ ++API: ++ V{upcase_GPR_name} ++ - Get register name REG_WIDTH component of `upcase_GPR_name` ++ {upcase_mask_insn_without_postfix} ++ - Get proper REG_WIDTH mask insn for `upcase_mask_insn_without_postfix` ++ VGPR(reg_name) ++ - Get register name REG_WIDTH component of `reg_name` ++ VKINSN(mask_insn) ++ - Get proper REG_WIDTH mask insn for `mask_insn` ++ VGPR_SZ(reg_name, reg_size) ++ - Get register name `reg_size` component of `reg_name` ++ VKINSN_SZ(mask_insn, insn_size) ++ - Get proper `insn_size` mask insn for `mask_insn` ++""" ++ ++import sys ++import os ++from datetime import datetime ++ ++registers = [["rax", "eax", "ax", "al"], ["rbx", "ebx", "bx", "bl"], ++ ["rcx", "ecx", "cx", "cl"], ["rdx", "edx", "dx", "dl"], ++ ["rbp", "ebp", "bp", "bpl"], ["rsp", "esp", "sp", "spl"], ++ ["rsi", "esi", "si", "sil"], ["rdi", "edi", "di", "dil"], ++ ["r8", "r8d", "r8w", "r8b"], ["r9", "r9d", "r9w", "r9b"], ++ ["r10", "r10d", "r10w", "r10b"], ["r11", "r11d", "r11w", "r11b"], ++ ["r12", "r12d", "r12w", "r12b"], ["r13", "r13d", "r13w", "r13b"], ++ ["r14", "r14d", "r14w", "r14b"], ["r15", "r15d", "r15w", "r15b"]] ++ ++mask_insns = [ ++ "kmov", ++ "kortest", ++ "kor", ++ "ktest", ++ "kand", ++ "kxor", ++ "knot", ++ "kxnor", ++] ++mask_insns_ext = ["b", "w", "d", "q"] ++ ++cr = """ ++ Copyright (C) {} Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++""" ++ ++print("/* This file was generated by: {}.".format(os.path.basename( ++ sys.argv[0]))) ++print(cr.format(datetime.today().year)) ++ ++print("#ifndef _REG_MACROS_H") ++print("#define _REG_MACROS_H\t1") ++print("") ++for reg in registers: ++ for i in range(0, 4): ++ print("#define {}_{}\t{}".format(reg[0], 8 << i, reg[3 - i])) ++ ++print("") ++for mask_insn in mask_insns: ++ for i in range(0, 4): ++ print("#define {}_{}\t{}{}".format(mask_insn, 8 << i, mask_insn, ++ mask_insns_ext[i])) ++for i in range(0, 3): ++ print("#define kunpack_{}\tkunpack{}{}".format(8 << i, mask_insns_ext[i], ++ mask_insns_ext[i + 1])) ++mask_insns.append("kunpack") ++ ++print("") ++print( ++ "/* Common API for accessing proper width GPR is V{upcase_GPR_name}. */") ++for reg in registers: ++ print("#define V{}\tVGPR({})".format(reg[0].upper(), reg[0])) ++ ++print("") ++ ++print( ++ "/* Common API for accessing proper width mask insn is {upcase_mask_insn}. */" ++) ++for mask_insn in mask_insns: ++ print("#define {} \tVKINSN({})".format(mask_insn.upper(), mask_insn)) ++print("") ++ ++print("#ifdef USE_WIDE_CHAR") ++print("# define REG_WIDTH 32") ++print("#else") ++print("# define REG_WIDTH VEC_SIZE") ++print("#endif") ++print("") ++print("#define VPASTER(x, y)\tx##_##y") ++print("#define VEVALUATOR(x, y)\tVPASTER(x, y)") ++print("") ++print("#define VGPR_SZ(reg_name, reg_size)\tVEVALUATOR(reg_name, reg_size)") ++print("#define VKINSN_SZ(insn, reg_size)\tVEVALUATOR(insn, reg_size)") ++print("") ++print("#define VGPR(reg_name)\tVGPR_SZ(reg_name, REG_WIDTH)") ++print("#define VKINSN(mask_insn)\tVKINSN_SZ(mask_insn, REG_WIDTH)") ++ ++print("\n#endif") +diff --git a/sysdeps/x86_64/multiarch/x86-avx-rtm-vecs.h b/sysdeps/x86_64/multiarch/x86-avx-rtm-vecs.h +new file mode 100644 +index 0000000000000000..0b326c8a703eabb3 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/x86-avx-rtm-vecs.h +@@ -0,0 +1,35 @@ ++/* Common config for AVX-RTM VECs ++ All versions must be listed in ifunc-impl-list.c. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef _X86_AVX_RTM_VECS_H ++#define _X86_AVX_RTM_VECS_H 1 ++ ++#define COND_VZEROUPPER COND_VZEROUPPER_XTEST ++#define ZERO_UPPER_VEC_REGISTERS_RETURN \ ++ ZERO_UPPER_VEC_REGISTERS_RETURN_XTEST ++ ++#define VZEROUPPER_RETURN jmp L(return_vzeroupper) ++ ++#define USE_WITH_RTM 1 ++#include "x86-avx-vecs.h" ++ ++#undef SECTION ++#define SECTION(p) p##.avx.rtm ++ ++#endif +diff --git a/sysdeps/x86_64/multiarch/x86-avx-vecs.h b/sysdeps/x86_64/multiarch/x86-avx-vecs.h +new file mode 100644 +index 0000000000000000..dca1089060eeba93 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/x86-avx-vecs.h +@@ -0,0 +1,47 @@ ++/* Common config for AVX VECs ++ All versions must be listed in ifunc-impl-list.c. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef _X86_AVX_VECS_H ++#define _X86_AVX_VECS_H 1 ++ ++#ifdef VEC_SIZE ++# error "Multiple VEC configs included!" ++#endif ++ ++#define VEC_SIZE 32 ++#include "x86-vec-macros.h" ++ ++#define USE_WITH_AVX 1 ++#define SECTION(p) p##.avx ++ ++/* 4-byte mov instructions with AVX2. */ ++#define MOV_SIZE 4 ++/* 1 (ret) + 3 (vzeroupper). */ ++#define RET_SIZE 4 ++#define VZEROUPPER vzeroupper ++ ++#define VMOVU vmovdqu ++#define VMOVA vmovdqa ++#define VMOVNT vmovntdq ++ ++/* Often need to access xmm portion. */ ++#define VMM_128 VMM_any_xmm ++#define VMM VMM_any_ymm ++ ++#endif +diff --git a/sysdeps/x86_64/multiarch/x86-evex-vecs-common.h b/sysdeps/x86_64/multiarch/x86-evex-vecs-common.h +new file mode 100644 +index 0000000000000000..f331e9d8ecfc371e +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/x86-evex-vecs-common.h +@@ -0,0 +1,39 @@ ++/* Common config for EVEX256 and EVEX512 VECs ++ All versions must be listed in ifunc-impl-list.c. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef _X86_EVEX_VECS_COMMON_H ++#define _X86_EVEX_VECS_COMMON_H 1 ++ ++#include "x86-vec-macros.h" ++ ++/* 6-byte mov instructions with EVEX. */ ++#define MOV_SIZE 6 ++/* No vzeroupper needed. */ ++#define RET_SIZE 1 ++#define VZEROUPPER ++ ++#define VMOVU vmovdqu64 ++#define VMOVA vmovdqa64 ++#define VMOVNT vmovntdq ++ ++#define VMM_128 VMM_hi_xmm ++#define VMM_256 VMM_hi_ymm ++#define VMM_512 VMM_hi_zmm ++ ++#endif +diff --git a/sysdeps/x86_64/multiarch/x86-evex256-vecs.h b/sysdeps/x86_64/multiarch/x86-evex256-vecs.h +new file mode 100644 +index 0000000000000000..8337b955045deb64 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/x86-evex256-vecs.h +@@ -0,0 +1,38 @@ ++/* Common config for EVEX256 VECs ++ All versions must be listed in ifunc-impl-list.c. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef _EVEX256_VECS_H ++#define _EVEX256_VECS_H 1 ++ ++#ifdef VEC_SIZE ++# error "Multiple VEC configs included!" ++#endif ++ ++#define VEC_SIZE 32 ++#include "x86-evex-vecs-common.h" ++ ++#define USE_WITH_EVEX256 1 ++ ++#ifndef SECTION ++# define SECTION(p) p##.evex ++#endif ++ ++#define VMM VMM_256 ++#define VMM_lo VMM_any_ymm ++#endif +diff --git a/sysdeps/x86_64/multiarch/x86-evex512-vecs.h b/sysdeps/x86_64/multiarch/x86-evex512-vecs.h +new file mode 100644 +index 0000000000000000..7dc5c23ad04e6128 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/x86-evex512-vecs.h +@@ -0,0 +1,38 @@ ++/* Common config for EVEX512 VECs ++ All versions must be listed in ifunc-impl-list.c. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef _EVEX512_VECS_H ++#define _EVEX512_VECS_H 1 ++ ++#ifdef VEC_SIZE ++# error "Multiple VEC configs included!" ++#endif ++ ++#define VEC_SIZE 64 ++#include "x86-evex-vecs-common.h" ++ ++#define USE_WITH_EVEX512 1 ++ ++#ifndef SECTION ++# define SECTION(p) p##.evex512 ++#endif ++ ++#define VMM VMM_512 ++#define VMM_lo VMM_any_zmm ++#endif +diff --git a/sysdeps/x86_64/multiarch/x86-sse2-vecs.h b/sysdeps/x86_64/multiarch/x86-sse2-vecs.h +new file mode 100644 +index 0000000000000000..b8bbd5dc29cf2975 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/x86-sse2-vecs.h +@@ -0,0 +1,47 @@ ++/* Common config for SSE2 VECs ++ All versions must be listed in ifunc-impl-list.c. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef _X86_SSE2_VECS_H ++#define _X86_SSE2_VECS_H 1 ++ ++#ifdef VEC_SIZE ++# error "Multiple VEC configs included!" ++#endif ++ ++#define VEC_SIZE 16 ++#include "x86-vec-macros.h" ++ ++#define USE_WITH_SSE2 1 ++#define SECTION(p) p ++ ++/* 3-byte mov instructions with SSE2. */ ++#define MOV_SIZE 3 ++/* No vzeroupper needed. */ ++#define RET_SIZE 1 ++#define VZEROUPPER ++ ++#define VMOVU movups ++#define VMOVA movaps ++#define VMOVNT movntdq ++ ++#define VMM_128 VMM_any_xmm ++#define VMM VMM_any_xmm ++ ++ ++#endif +diff --git a/sysdeps/x86_64/multiarch/x86-vec-macros.h b/sysdeps/x86_64/multiarch/x86-vec-macros.h +new file mode 100644 +index 0000000000000000..7d6bb31d55f1a0d1 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/x86-vec-macros.h +@@ -0,0 +1,90 @@ ++/* Macro helpers for VEC_{type}({vec_num}) ++ All versions must be listed in ifunc-impl-list.c. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++#ifndef _X86_VEC_MACROS_H ++#define _X86_VEC_MACROS_H 1 ++ ++#ifndef VEC_SIZE ++# error "Never include this file directly. Always include a vector config." ++#endif ++ ++/* Defines so we can use SSE2 / AVX2 / EVEX / EVEX512 encoding with same ++ VMM(N) values. */ ++#define VMM_hi_xmm0 xmm16 ++#define VMM_hi_xmm1 xmm17 ++#define VMM_hi_xmm2 xmm18 ++#define VMM_hi_xmm3 xmm19 ++#define VMM_hi_xmm4 xmm20 ++#define VMM_hi_xmm5 xmm21 ++#define VMM_hi_xmm6 xmm22 ++#define VMM_hi_xmm7 xmm23 ++#define VMM_hi_xmm8 xmm24 ++#define VMM_hi_xmm9 xmm25 ++#define VMM_hi_xmm10 xmm26 ++#define VMM_hi_xmm11 xmm27 ++#define VMM_hi_xmm12 xmm28 ++#define VMM_hi_xmm13 xmm29 ++#define VMM_hi_xmm14 xmm30 ++#define VMM_hi_xmm15 xmm31 ++ ++#define VMM_hi_ymm0 ymm16 ++#define VMM_hi_ymm1 ymm17 ++#define VMM_hi_ymm2 ymm18 ++#define VMM_hi_ymm3 ymm19 ++#define VMM_hi_ymm4 ymm20 ++#define VMM_hi_ymm5 ymm21 ++#define VMM_hi_ymm6 ymm22 ++#define VMM_hi_ymm7 ymm23 ++#define VMM_hi_ymm8 ymm24 ++#define VMM_hi_ymm9 ymm25 ++#define VMM_hi_ymm10 ymm26 ++#define VMM_hi_ymm11 ymm27 ++#define VMM_hi_ymm12 ymm28 ++#define VMM_hi_ymm13 ymm29 ++#define VMM_hi_ymm14 ymm30 ++#define VMM_hi_ymm15 ymm31 ++ ++#define VMM_hi_zmm0 zmm16 ++#define VMM_hi_zmm1 zmm17 ++#define VMM_hi_zmm2 zmm18 ++#define VMM_hi_zmm3 zmm19 ++#define VMM_hi_zmm4 zmm20 ++#define VMM_hi_zmm5 zmm21 ++#define VMM_hi_zmm6 zmm22 ++#define VMM_hi_zmm7 zmm23 ++#define VMM_hi_zmm8 zmm24 ++#define VMM_hi_zmm9 zmm25 ++#define VMM_hi_zmm10 zmm26 ++#define VMM_hi_zmm11 zmm27 ++#define VMM_hi_zmm12 zmm28 ++#define VMM_hi_zmm13 zmm29 ++#define VMM_hi_zmm14 zmm30 ++#define VMM_hi_zmm15 zmm31 ++ ++#define PRIMITIVE_VMM(vec, num) vec##num ++ ++#define VMM_any_xmm(i) PRIMITIVE_VMM(xmm, i) ++#define VMM_any_ymm(i) PRIMITIVE_VMM(ymm, i) ++#define VMM_any_zmm(i) PRIMITIVE_VMM(zmm, i) ++ ++#define VMM_hi_xmm(i) PRIMITIVE_VMM(VMM_hi_xmm, i) ++#define VMM_hi_ymm(i) PRIMITIVE_VMM(VMM_hi_ymm, i) ++#define VMM_hi_zmm(i) PRIMITIVE_VMM(VMM_hi_zmm, i) ++ ++#endif diff --git a/glibc-RHEL-175520-5.patch b/glibc-RHEL-175520-5.patch new file mode 100644 index 0000000..8ce00aa --- /dev/null +++ b/glibc-RHEL-175520-5.patch @@ -0,0 +1,264 @@ +commit be066536bd313f1eec6e36fb92a96d39bf76f483 +Author: Noah Goldstein +Date: Fri Oct 14 22:00:30 2022 -0500 + + x86: Update strlen-evex-base to use new reg/vec macros. + + To avoid duplicate the VMM / GPR / mask insn macros in all incoming + evex512 files use the macros defined in 'reg-macros.h' and + '{vec}-macros.h' + + This commit does not change libc.so + + Tested build on x86-64 + +diff --git a/sysdeps/x86_64/multiarch/strlen-evex-base.S b/sysdeps/x86_64/multiarch/strlen-evex-base.S +index 278c899691d89ba7..176babee1e0a9e89 100644 +--- a/sysdeps/x86_64/multiarch/strlen-evex-base.S ++++ b/sysdeps/x86_64/multiarch/strlen-evex-base.S +@@ -32,42 +32,10 @@ + # define CHAR_SIZE 1 + # endif + +-# define XMM0 xmm16 + # define PAGE_SIZE 4096 + # define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) + +-# if VEC_SIZE == 64 +-# define KMOV kmovq +-# define KORTEST kortestq +-# define RAX rax +-# define RCX rcx +-# define RDX rdx +-# define SHR shrq +-# define TEXTSUFFIX evex512 +-# define VMM0 zmm16 +-# define VMM1 zmm17 +-# define VMM2 zmm18 +-# define VMM3 zmm19 +-# define VMM4 zmm20 +-# define VMOVA vmovdqa64 +-# elif VEC_SIZE == 32 +-/* Currently Unused. */ +-# define KMOV kmovd +-# define KORTEST kortestd +-# define RAX eax +-# define RCX ecx +-# define RDX edx +-# define SHR shrl +-# define TEXTSUFFIX evex256 +-# define VMM0 ymm16 +-# define VMM1 ymm17 +-# define VMM2 ymm18 +-# define VMM3 ymm19 +-# define VMM4 ymm20 +-# define VMOVA vmovdqa32 +-# endif +- +- .section .text.TEXTSUFFIX, "ax", @progbits ++ .section SECTION(.text),"ax",@progbits + /* Aligning entry point to 64 byte, provides better performance for + one vector length string. */ + ENTRY_P2ALIGN (STRLEN, 6) +@@ -82,18 +50,18 @@ ENTRY_P2ALIGN (STRLEN, 6) + # endif + + movl %edi, %eax +- vpxorq %XMM0, %XMM0, %XMM0 ++ vpxorq %VMM_128(0), %VMM_128(0), %VMM_128(0) + andl $(PAGE_SIZE - 1), %eax + cmpl $(PAGE_SIZE - VEC_SIZE), %eax + ja L(page_cross) + + /* Compare [w]char for null, mask bit will be set for match. */ +- VPCMP $0, (%rdi), %VMM0, %k0 +- KMOV %k0, %RAX +- test %RAX, %RAX ++ VPCMP $0, (%rdi), %VMM(0), %k0 ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX + jz L(align_more) + +- bsf %RAX, %RAX ++ bsf %VRAX, %VRAX + # ifdef USE_AS_STRNLEN + cmpq %rsi, %rax + cmovnb %rsi, %rax +@@ -116,7 +84,7 @@ L(align_more): + movq %rax, %rdx + subq %rdi, %rdx + # ifdef USE_AS_WCSLEN +- SHR $2, %RDX ++ shr $2, %VRDX + # endif + /* At this point rdx contains [w]chars already compared. */ + subq %rsi, %rdx +@@ -127,9 +95,9 @@ L(align_more): + # endif + + /* Loop unroll 4 times for 4 vector loop. */ +- VPCMP $0, (%rax), %VMM0, %k0 +- KMOV %k0, %RCX +- test %RCX, %RCX ++ VPCMP $0, (%rax), %VMM(0), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX + jnz L(ret_vec_x1) + + # ifdef USE_AS_STRNLEN +@@ -137,9 +105,9 @@ L(align_more): + jbe L(ret_max) + # endif + +- VPCMP $0, VEC_SIZE(%rax), %VMM0, %k0 +- KMOV %k0, %RCX +- test %RCX, %RCX ++ VPCMP $0, VEC_SIZE(%rax), %VMM(0), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX + jnz L(ret_vec_x2) + + # ifdef USE_AS_STRNLEN +@@ -147,9 +115,9 @@ L(align_more): + jbe L(ret_max) + # endif + +- VPCMP $0, (VEC_SIZE * 2)(%rax), %VMM0, %k0 +- KMOV %k0, %RCX +- test %RCX, %RCX ++ VPCMP $0, (VEC_SIZE * 2)(%rax), %VMM(0), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX + jnz L(ret_vec_x3) + + # ifdef USE_AS_STRNLEN +@@ -157,9 +125,9 @@ L(align_more): + jbe L(ret_max) + # endif + +- VPCMP $0, (VEC_SIZE * 3)(%rax), %VMM0, %k0 +- KMOV %k0, %RCX +- test %RCX, %RCX ++ VPCMP $0, (VEC_SIZE * 3)(%rax), %VMM(0), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX + jnz L(ret_vec_x4) + + # ifdef USE_AS_STRNLEN +@@ -175,7 +143,7 @@ L(align_more): + # ifdef USE_AS_STRNLEN + subq %rax, %rcx + # ifdef USE_AS_WCSLEN +- SHR $2, %RCX ++ shr $2, %VRCX + # endif + /* rcx contains number of [w]char will be recompared due to + alignment fixes. rdx must be incremented by rcx to offset +@@ -195,42 +163,42 @@ L(loop_entry): + # endif + /* VPMINU and VPCMP combination provide better performance as + compared to alternative combinations. */ +- VMOVA (VEC_SIZE * 4)(%rax), %VMM1 +- VPMINU (VEC_SIZE * 5)(%rax), %VMM1, %VMM2 +- VMOVA (VEC_SIZE * 6)(%rax), %VMM3 +- VPMINU (VEC_SIZE * 7)(%rax), %VMM3, %VMM4 ++ VMOVA (VEC_SIZE * 4)(%rax), %VMM(1) ++ VPMINU (VEC_SIZE * 5)(%rax), %VMM(1), %VMM(2) ++ VMOVA (VEC_SIZE * 6)(%rax), %VMM(3) ++ VPMINU (VEC_SIZE * 7)(%rax), %VMM(3), %VMM(4) + +- VPTESTN %VMM2, %VMM2, %k0 +- VPTESTN %VMM4, %VMM4, %k1 ++ VPTESTN %VMM(2), %VMM(2), %k0 ++ VPTESTN %VMM(4), %VMM(4), %k1 + + subq $-(VEC_SIZE * 4), %rax + KORTEST %k0, %k1 + jz L(loop) + +- VPTESTN %VMM1, %VMM1, %k2 +- KMOV %k2, %RCX +- test %RCX, %RCX ++ VPTESTN %VMM(1), %VMM(1), %k2 ++ KMOV %k2, %VRCX ++ test %VRCX, %VRCX + jnz L(ret_vec_x1) + +- KMOV %k0, %RCX ++ KMOV %k0, %VRCX + /* At this point, if k0 is non zero, null char must be in the + second vector. */ +- test %RCX, %RCX ++ test %VRCX, %VRCX + jnz L(ret_vec_x2) + +- VPTESTN %VMM3, %VMM3, %k3 +- KMOV %k3, %RCX +- test %RCX, %RCX ++ VPTESTN %VMM(3), %VMM(3), %k3 ++ KMOV %k3, %VRCX ++ test %VRCX, %VRCX + jnz L(ret_vec_x3) + /* At this point null [w]char must be in the fourth vector so no + need to check. */ +- KMOV %k1, %RCX ++ KMOV %k1, %VRCX + + /* Fourth, third, second vector terminating are pretty much + same, implemented this way to avoid branching and reuse code + from pre loop exit condition. */ + L(ret_vec_x4): +- bsf %RCX, %RCX ++ bsf %VRCX, %VRCX + subq %rdi, %rax + # ifdef USE_AS_WCSLEN + subq $-(VEC_SIZE * 3), %rax +@@ -246,7 +214,7 @@ L(ret_vec_x4): + ret + + L(ret_vec_x3): +- bsf %RCX, %RCX ++ bsf %VRCX, %VRCX + subq %rdi, %rax + # ifdef USE_AS_WCSLEN + subq $-(VEC_SIZE * 2), %rax +@@ -264,7 +232,7 @@ L(ret_vec_x3): + L(ret_vec_x2): + subq $-VEC_SIZE, %rax + L(ret_vec_x1): +- bsf %RCX, %RCX ++ bsf %VRCX, %VRCX + subq %rdi, %rax + # ifdef USE_AS_WCSLEN + shrq $2, %rax +@@ -285,13 +253,13 @@ L(page_cross): + /* ecx contains number of w[char] to be skipped as a result + of address alignment. */ + xorq %rdi, %rax +- VPCMP $0, (PAGE_SIZE - VEC_SIZE)(%rax), %VMM0, %k0 +- KMOV %k0, %RAX ++ VPCMP $0, (PAGE_SIZE - VEC_SIZE)(%rax), %VMM(0), %k0 ++ KMOV %k0, %VRAX + /* Ignore number of character for alignment adjustment. */ +- SHR %cl, %RAX ++ shr %cl, %VRAX + jz L(align_more) + +- bsf %RAX, %RAX ++ bsf %VRAX, %VRAX + # ifdef USE_AS_STRNLEN + cmpq %rsi, %rax + cmovnb %rsi, %rax +diff --git a/sysdeps/x86_64/multiarch/strlen-evex512.S b/sysdeps/x86_64/multiarch/strlen-evex512.S +index 116f8981c8954e2e..10c3415c8a1e5380 100644 +--- a/sysdeps/x86_64/multiarch/strlen-evex512.S ++++ b/sysdeps/x86_64/multiarch/strlen-evex512.S +@@ -2,6 +2,6 @@ + # define STRLEN __strlen_evex512 + #endif + +-#define VEC_SIZE 64 +- ++#include "x86-evex512-vecs.h" ++#include "reg-macros.h" + #include "strlen-evex-base.S" diff --git a/glibc-RHEL-175520-6.patch b/glibc-RHEL-175520-6.patch new file mode 100644 index 0000000..0e53c73 --- /dev/null +++ b/glibc-RHEL-175520-6.patch @@ -0,0 +1,453 @@ +commit 451c6e58540e8571e31581c04c4829e5d2cfe8ac +Author: Sunil K Pandey +Date: Thu Aug 18 06:48:07 2022 -0700 + + x86_64: Implement evex512 version of memchr, rawmemchr and wmemchr + + This patch implements following evex512 version of string functions. + evex512 version takes up to 30% less cycle as compared to evex, + depending on length and alignment. + + - memchr function using 512 bit vectors. + - rawmemchr function using 512 bit vectors. + - wmemchr function using 512 bit vectors. + + Code size data: + + memchr-evex.o 762 byte + memchr-evex512.o 576 byte (-24%) + + rawmemchr-evex.o 461 byte + rawmemchr-evex512.o 412 byte (-11%) + + wmemchr-evex.o 794 byte + wmemchr-evex512.o 552 byte (-30%) + + Placeholder function, not used by any processor at the moment. + + Reviewed-by: Noah Goldstein + +Conflicts: + sysdeps/x86_64/multiarch/ifunc-impl-list.c + (Adapting to missing commit 703f4341083afa7d71987aa96a35eab81309e634) + +diff --git a/sysdeps/x86_64/multiarch/Makefile b/sysdeps/x86_64/multiarch/Makefile +index b9ea5b60c2be1b0a..67d4b3df693b7110 100644 +--- a/sysdeps/x86_64/multiarch/Makefile ++++ b/sysdeps/x86_64/multiarch/Makefile +@@ -4,6 +4,7 @@ sysdep_routines += \ + memchr-avx2 \ + memchr-avx2-rtm \ + memchr-evex \ ++ memchr-evex512 \ + memchr-evex-rtm \ + memchr-sse2 \ + memcmp-avx2-movbe \ +@@ -36,6 +37,7 @@ sysdep_routines += \ + rawmemchr-avx2 \ + rawmemchr-avx2-rtm \ + rawmemchr-evex \ ++ rawmemchr-evex512 \ + rawmemchr-evex-rtm \ + rawmemchr-sse2 \ + stpcpy-avx2 \ +@@ -174,6 +176,7 @@ sysdep_routines += \ + wmemchr-avx2 \ + wmemchr-avx2-rtm \ + wmemchr-evex \ ++ wmemchr-evex512 \ + wmemchr-evex-rtm \ + wmemchr-sse2 \ + wmemcmp-avx2-movbe \ +diff --git a/sysdeps/x86_64/multiarch/ifunc-impl-list.c b/sysdeps/x86_64/multiarch/ifunc-impl-list.c +index 84f9e73e2b7df816..6037de5c422a6f7a 100644 +--- a/sysdeps/x86_64/multiarch/ifunc-impl-list.c ++++ b/sysdeps/x86_64/multiarch/ifunc-impl-list.c +@@ -54,6 +54,11 @@ __libc_ifunc_impl_list (const char *name, struct libc_ifunc_impl *array, + && CPU_FEATURE_USABLE (AVX512BW) + && CPU_FEATURE_USABLE (BMI2)), + __memchr_evex) ++ IFUNC_IMPL_ADD (array, i, memchr, ++ (CPU_FEATURE_USABLE (AVX512VL) ++ && CPU_FEATURE_USABLE (AVX512BW) ++ && CPU_FEATURE_USABLE (BMI2)), ++ __memchr_evex512) + IFUNC_IMPL_ADD (array, i, memchr, + (CPU_FEATURE_USABLE (AVX512VL) + && CPU_FEATURE_USABLE (AVX512BW) +@@ -304,6 +309,11 @@ __libc_ifunc_impl_list (const char *name, struct libc_ifunc_impl *array, + && CPU_FEATURE_USABLE (AVX512BW) + && CPU_FEATURE_USABLE (BMI2)), + __rawmemchr_evex) ++ IFUNC_IMPL_ADD (array, i, rawmemchr, ++ (CPU_FEATURE_USABLE (AVX512VL) ++ && CPU_FEATURE_USABLE (AVX512BW) ++ && CPU_FEATURE_USABLE (BMI2)), ++ __rawmemchr_evex512) + IFUNC_IMPL_ADD (array, i, rawmemchr, + (CPU_FEATURE_USABLE (AVX512VL) + && CPU_FEATURE_USABLE (AVX512BW) +@@ -821,6 +831,11 @@ __libc_ifunc_impl_list (const char *name, struct libc_ifunc_impl *array, + && CPU_FEATURE_USABLE (AVX512BW) + && CPU_FEATURE_USABLE (BMI2)), + __wmemchr_evex) ++ IFUNC_IMPL_ADD (array, i, wmemchr, ++ (CPU_FEATURE_USABLE (AVX512VL) ++ && CPU_FEATURE_USABLE (AVX512BW) ++ && CPU_FEATURE_USABLE (BMI2)), ++ __wmemchr_evex512) + IFUNC_IMPL_ADD (array, i, wmemchr, + (CPU_FEATURE_USABLE (AVX512VL) + && CPU_FEATURE_USABLE (AVX512BW) +diff --git a/sysdeps/x86_64/multiarch/memchr-evex-base.S b/sysdeps/x86_64/multiarch/memchr-evex-base.S +new file mode 100644 +index 0000000000000000..6ebc9a66e812a644 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/memchr-evex-base.S +@@ -0,0 +1,304 @@ ++/* Placeholder function, not used by any processor at the moment. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++/* UNUSED. Exists purely as reference implementation. */ ++ ++#include ++ ++#if ISA_SHOULD_BUILD (4) ++ ++# include ++ ++# ifdef USE_AS_WMEMCHR ++# define CHAR_SIZE 4 ++# define VPBROADCAST vpbroadcastd ++# define VPCMPEQ vpcmpeqd ++# define VPCMPNE vpcmpneqd ++# define VPMINU vpminud ++# define VPTESTNM vptestnmd ++# else ++# define CHAR_SIZE 1 ++# define VPBROADCAST vpbroadcastb ++# define VPCMPEQ vpcmpeqb ++# define VPCMPNE vpcmpneqb ++# define VPMINU vpminub ++# define VPTESTNM vptestnmb ++# endif ++ ++# define PAGE_SIZE 4096 ++# define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) ++ ++ .section SECTION(.text), "ax", @progbits ++/* Aligning entry point to 64 byte, provides better performance for ++ one vector length string. */ ++ENTRY_P2ALIGN (MEMCHR, 6) ++# ifndef USE_AS_RAWMEMCHR ++ /* Check for zero length. */ ++ test %RDX_LP, %RDX_LP ++ jz L(zero) ++ ++# ifdef __ILP32__ ++ /* Clear the upper 32 bits. */ ++ movl %edx, %edx ++# endif ++# endif ++ ++ /* Broadcast CHAR to VMM(1). */ ++ VPBROADCAST %esi, %VMM(1) ++ movl %edi, %eax ++ andl $(PAGE_SIZE - 1), %eax ++ cmpl $(PAGE_SIZE - VEC_SIZE), %eax ++ ja L(page_cross) ++ ++ /* Compare [w]char for null, mask bit will be set for match. */ ++ VPCMPEQ (%rdi), %VMM(1), %k0 ++ ++ KMOV %k0, %VRCX ++# ifndef USE_AS_RAWMEMCHR ++ mov %rdx, %rsi ++ /* Need to use bsfq here as upper 32 bit of rsi may zero out ++ for 'bsf %ecx, %esi', if %ecx is 0. */ ++ bsfq %rcx, %rsi ++ cmp $CHAR_PER_VEC, %rsi ++ ja L(align_more) ++# ifdef USE_AS_WMEMCHR ++ leaq (%rdi, %rsi, CHAR_SIZE), %rdi ++# else ++ addq %rsi, %rdi ++# endif ++ xor %eax, %eax ++ cmp %rsi, %rdx ++ cmova %rdi, %rax ++# else ++ bsf %VRCX, %VRAX ++ jz L(align_more) ++ add %rdi, %rax ++# endif ++ ret ++ ++ .p2align 5,,5 ++L(page_cross): ++ movl %eax, %ecx ++ andl $(VEC_SIZE - 1), %ecx ++# ifdef USE_AS_WMEMCHR ++ shrl $2, %ecx ++# endif ++ xorq %rdi, %rax ++ VPCMPEQ (PAGE_SIZE - VEC_SIZE)(%rax), %VMM(1), %k0 ++ KMOV %k0, %VRSI ++ shr %cl, %VRSI ++# ifndef USE_AS_RAWMEMCHR ++ jnz L(page_cross_end) ++ movl $CHAR_PER_VEC, %eax ++ sub %ecx, %eax ++ cmp %rax, %rdx ++ ja L(align_more) ++# else ++ jz L(align_more) ++# endif ++ ++L(page_cross_end): ++# ifndef USE_AS_RAWMEMCHR ++ bsf %VRSI, %VRCX ++ jz L(zero) ++ leaq (%rdi, %rcx, CHAR_SIZE), %rdi ++ xor %eax, %eax ++ cmp %rcx, %rdx ++ cmova %rdi, %rax ++# else ++ bsf %VRSI, %VRAX ++ add %rdi, %rax ++# endif ++ ret ++ ++# ifndef USE_AS_RAWMEMCHR ++L(zero): ++ xorl %eax, %eax ++ ret ++# endif ++ ++L(ret_vec_x2): ++ subq $-VEC_SIZE, %rdi ++L(ret_vec_x1): ++ bsf %VRAX, %VRAX ++# ifndef USE_AS_RAWMEMCHR ++ cmp %rax, %rdx ++ jbe L(zero) ++# endif ++# ifdef USE_AS_WMEMCHR ++ leaq (%rdi, %rax, CHAR_SIZE), %rax ++# else ++ add %rdi, %rax ++# endif ++ ret ++ ++ .p2align 5,,5 ++L(align_more): ++# ifndef USE_AS_RAWMEMCHR ++ mov %rdi, %rax ++# endif ++ subq $-VEC_SIZE, %rdi ++ /* Align rdi to VEC_SIZE. */ ++ andq $-VEC_SIZE, %rdi ++ ++# ifndef USE_AS_RAWMEMCHR ++ subq %rdi, %rax ++# ifdef USE_AS_WMEMCHR ++ sar $2, %rax ++# endif ++ addq %rax, %rdx ++# endif ++ ++ /* Loop unroll 4 times for 4 vector loop. */ ++ VPCMPEQ (%rdi), %VMM(1), %k0 ++ ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(ret_vec_x1) ++ ++# ifndef USE_AS_RAWMEMCHR ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(zero) ++# endif ++ ++ VPCMPEQ VEC_SIZE(%rdi), %VMM(1), %k0 ++ ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(ret_vec_x2) ++ ++# ifndef USE_AS_RAWMEMCHR ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(zero) ++# endif ++ ++ VPCMPEQ (VEC_SIZE * 2)(%rdi), %VMM(1), %k0 ++ ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(ret_vec_x3) ++ ++# ifndef USE_AS_RAWMEMCHR ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(zero) ++# endif ++ ++ VPCMPEQ (VEC_SIZE * 3)(%rdi), %VMM(1), %k0 ++ ++ KMOV %k0, %VRAX ++ test %VRAX, %VRAX ++ jnz L(ret_vec_x4) ++ ++# ifndef USE_AS_RAWMEMCHR ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(zero) ++ /* Save pointer to find alignment adjustment. */ ++ movq %rdi, %rax ++# endif ++ /* Align address to VEC_SIZE * 4 for loop. */ ++ andq $-(VEC_SIZE * 4), %rdi ++ ++ /* Add alignment difference to rdx. */ ++# ifndef USE_AS_RAWMEMCHR ++ subq %rdi, %rax ++# ifdef USE_AS_WMEMCHR ++ shr $2, %VRAX ++# endif ++ addq %rax, %rdx ++# endif ++ ++ /* 4 vector loop. */ ++ .p2align 5,,11 ++L(loop): ++ ++ VPCMPNE (VEC_SIZE * 4)(%rdi), %VMM(1), %k1 ++ vpxorq (VEC_SIZE * 5)(%rdi), %VMM(1), %VMM(2) ++ vpxorq (VEC_SIZE * 6)(%rdi), %VMM(1), %VMM(3) ++ VPCMPEQ (VEC_SIZE * 7)(%rdi), %VMM(1), %k3 ++ VPMINU %VMM(2), %VMM(3), %VMM(3){%k1}{z} ++ VPTESTNM %VMM(3), %VMM(3), %k2 ++ ++ subq $-(VEC_SIZE * 4), %rdi ++ KORTEST %k2, %k3 ++# ifdef USE_AS_RAWMEMCHR ++ jz L(loop) ++# else ++ jnz L(loopend) ++ subq $(CHAR_PER_VEC * 4), %rdx ++ ja L(loop) ++L(zero_2): ++ xor %eax, %eax ++ ret ++# endif ++ ++L(loopend): ++ VPCMPEQ (%rdi), %VMM(1), %k1 ++ KMOV %k1, %VRAX ++ test %VRAX, %VRAX ++ jnz L(ret_vec_x1) ++ ++# ifndef USE_AS_RAWMEMCHR ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(zero_2) ++# endif ++ ++ VPCMPEQ VEC_SIZE(%rdi), %VMM(1), %k1 ++ KMOV %k1, %VRAX ++ test %VRAX, %VRAX ++ jnz L(ret_vec_x2) ++ ++# ifndef USE_AS_RAWMEMCHR ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(zero_2) ++# endif ++ ++ VPCMPEQ (VEC_SIZE * 2)(%rdi), %VMM(1), %k1 ++ KMOV %k1, %VRAX ++ test %VRAX, %VRAX ++ jnz L(ret_vec_x3) ++ ++# ifndef USE_AS_RAWMEMCHR ++ subq $CHAR_PER_VEC, %rdx ++ jbe L(zero_2) ++# endif ++ ++ /* At this point null [w]char must be in the fourth vector so no ++ need to check. */ ++ KMOV %k3, %VRAX ++ ++L(ret_vec_x4): ++ bsf %VRAX, %VRAX ++# ifndef USE_AS_RAWMEMCHR ++ cmp %rax, %rdx ++ jbe L(zero) ++# endif ++ leaq (VEC_SIZE * 3)(%rdi, %rax, CHAR_SIZE), %rax ++ ret ++ ++ .p2align 5,,5 ++L(ret_vec_x3): ++ bsf %VRAX, %VRAX ++# ifndef USE_AS_RAWMEMCHR ++ cmp %rax, %rdx ++ jbe L(zero) ++# endif ++ leaq (VEC_SIZE * 2)(%rdi, %rax, CHAR_SIZE), %rax ++ ret ++ ++END (MEMCHR) ++#endif +diff --git a/sysdeps/x86_64/multiarch/memchr-evex512.S b/sysdeps/x86_64/multiarch/memchr-evex512.S +new file mode 100644 +index 0000000000000000..002f8c84893181e0 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/memchr-evex512.S +@@ -0,0 +1,8 @@ ++# ifndef MEMCHR ++# define MEMCHR __memchr_evex512 ++# endif ++ ++#include "x86-evex512-vecs.h" ++#include "reg-macros.h" ++ ++#include "memchr-evex-base.S" +diff --git a/sysdeps/x86_64/multiarch/rawmemchr-evex512.S b/sysdeps/x86_64/multiarch/rawmemchr-evex512.S +new file mode 100644 +index 0000000000000000..302d3cb0554e736e +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/rawmemchr-evex512.S +@@ -0,0 +1,7 @@ ++#ifndef RAWMEMCHR ++# define RAWMEMCHR __rawmemchr_evex512 ++#endif ++#define USE_AS_RAWMEMCHR 1 ++#define MEMCHR RAWMEMCHR ++ ++#include "memchr-evex512.S" +diff --git a/sysdeps/x86_64/multiarch/wmemchr-evex512.S b/sysdeps/x86_64/multiarch/wmemchr-evex512.S +new file mode 100644 +index 0000000000000000..78ec4ee5ad7fb7b5 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/wmemchr-evex512.S +@@ -0,0 +1,9 @@ ++#ifndef WMEMCHR ++# define WMEMCHR __wmemchr_evex512 ++#endif ++ ++#define MEMCHR WMEMCHR ++#define USE_AS_WMEMCHR 1 ++ ++#define USE_WIDE_CHAR 1 ++#include "memchr-evex512.S" diff --git a/glibc-RHEL-175520-7.patch b/glibc-RHEL-175520-7.patch new file mode 100644 index 0000000..0ed7b2e --- /dev/null +++ b/glibc-RHEL-175520-7.patch @@ -0,0 +1,427 @@ +commit 59e501f204fa196d6571b523459ba528bbef7783 +Author: Sunil K Pandey +Date: Tue Jul 26 13:54:56 2022 -0700 + + x86_64: Implement evex512 version of strchrnul, strchr and wcschr + + This patch implements following evex512 version of string functions. + evex512 version takes up to 30% less cycle as compared to evex, + depending on length and alignment. + + - strchrnul function using 512 bit vectors. + - strchr function using 512 bit vectors. + - wcschr function using 512 bit vectors. + + Code size data: + + strchrnul-evex.o 599 byte + strchrnul-evex512.o 569 byte (-5%) + + strchr-evex.o 639 byte + strchr-evex512.o 595 byte (-7%) + + wcschr-evex.o 644 byte + wcschr-evex512.o 607 byte (-6%) + + Placeholder function, not used by any processor at the moment. + + Reviewed-by: Noah Goldstein + +Conflicts: + sysdeps/x86_64/multiarch/ifunc-impl-list.c + (fixup macro) + +diff --git a/sysdeps/x86_64/multiarch/Makefile b/sysdeps/x86_64/multiarch/Makefile +index 67d4b3df693b7110..1c8bc2af0ef9c367 100644 +--- a/sysdeps/x86_64/multiarch/Makefile ++++ b/sysdeps/x86_64/multiarch/Makefile +@@ -67,11 +67,13 @@ sysdep_routines += \ + strchr-avx2 \ + strchr-avx2-rtm \ + strchr-evex \ ++ strchr-evex512 \ + strchr-sse2 \ + strchr-sse2-no-bsf \ + strchrnul-avx2 \ + strchrnul-avx2-rtm \ + strchrnul-evex \ ++ strchrnul-evex512 \ + strchrnul-sse2 \ + strcmp-avx2 \ + strcmp-avx2-rtm \ +@@ -146,6 +148,7 @@ sysdep_routines += \ + wcschr-avx2 \ + wcschr-avx2-rtm \ + wcschr-evex \ ++ wcschr-evex512 \ + wcschr-sse2 \ + wcscmp-avx2 \ + wcscmp-avx2-rtm \ +diff --git a/sysdeps/x86_64/multiarch/ifunc-impl-list.c b/sysdeps/x86_64/multiarch/ifunc-impl-list.c +index 6037de5c422a6f7a..cf482eff56fb663c 100644 +--- a/sysdeps/x86_64/multiarch/ifunc-impl-list.c ++++ b/sysdeps/x86_64/multiarch/ifunc-impl-list.c +@@ -484,6 +484,10 @@ __libc_ifunc_impl_list (const char *name, struct libc_ifunc_impl *array, + && CPU_FEATURE_USABLE (AVX512BW) + && CPU_FEATURE_USABLE (BMI2)), + __strchr_evex) ++ IFUNC_IMPL_ADD (array, i, strchr, ++ (CPU_FEATURE_USABLE (AVX512VL) ++ && CPU_FEATURE_USABLE (AVX512BW)), ++ __strchr_evex512) + IFUNC_IMPL_ADD (array, i, strchr, 1, __strchr_sse2_no_bsf) + IFUNC_IMPL_ADD (array, i, strchr, 1, __strchr_sse2)) + +@@ -503,6 +507,10 @@ __libc_ifunc_impl_list (const char *name, struct libc_ifunc_impl *array, + && CPU_FEATURE_USABLE (AVX512BW) + && CPU_FEATURE_USABLE (BMI2)), + __strchrnul_evex) ++ IFUNC_IMPL_ADD (array, i, strchrnul, ++ (CPU_FEATURE_USABLE (AVX512VL) ++ && CPU_FEATURE_USABLE (AVX512BW)), ++ __strchrnul_evex512) + IFUNC_IMPL_ADD (array, i, strchrnul, 1, __strchrnul_sse2)) + + /* Support sysdeps/x86_64/multiarch/strrchr.c. */ +@@ -698,6 +706,10 @@ __libc_ifunc_impl_list (const char *name, struct libc_ifunc_impl *array, + && CPU_FEATURE_USABLE (AVX512BW) + && CPU_FEATURE_USABLE (BMI2)), + __wcschr_evex) ++ IFUNC_IMPL_ADD (array, i, wcschr, ++ (CPU_FEATURE_USABLE (AVX512VL) ++ && CPU_FEATURE_USABLE (AVX512BW)), ++ __wcschr_evex512) + IFUNC_IMPL_ADD (array, i, wcschr, 1, __wcschr_sse2)) + + /* Support sysdeps/x86_64/multiarch/wcsrchr.c. */ +diff --git a/sysdeps/x86_64/multiarch/strchr-evex-base.S b/sysdeps/x86_64/multiarch/strchr-evex-base.S +new file mode 100644 +index 0000000000000000..75fee8c82ade14f1 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/strchr-evex-base.S +@@ -0,0 +1,282 @@ ++/* Placeholder function, not used by any processor at the moment. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++/* UNUSED. Exists purely as reference implementation. */ ++ ++#include ++ ++#if ISA_SHOULD_BUILD (4) ++ ++# include ++ ++# ifdef USE_AS_WCSCHR ++# define CHAR_REG esi ++# define CHAR_SIZE 4 ++# define VPBROADCAST vpbroadcastd ++# define VPCMP vpcmpd ++# define VPCMPNE vpcmpneqd ++# define VPMINU vpminud ++# define VPTEST vptestmd ++# define VPTESTN vptestnmd ++# else ++# define CHAR_REG sil ++# define CHAR_SIZE 1 ++# define VPBROADCAST vpbroadcastb ++# define VPCMP vpcmpb ++# define VPCMPNE vpcmpneqb ++# define VPMINU vpminub ++# define VPTEST vptestmb ++# define VPTESTN vptestnmb ++# endif ++ ++# define PAGE_SIZE 4096 ++# define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) ++# define VEC_MATCH_MASK ((1 << CHAR_PER_VEC) - 1) ++ ++ .section SECTION(.text), "ax", @progbits ++/* Aligning entry point to 64 byte, provides better performance for ++ one vector length string. */ ++ENTRY_P2ALIGN (STRCHR, 6) ++ ++ /* Broadcast CHAR to VMM(0). */ ++ VPBROADCAST %esi, %VMM(0) ++ movl %edi, %eax ++ sall $20,%eax ++ cmpl $((PAGE_SIZE - VEC_SIZE) << 20), %eax ++ ja L(page_cross) ++ ++ VMOVU (%rdi), %VMM(1) ++ VPCMPNE %VMM(1), %VMM(0), %k1 ++ VPTEST %VMM(1), %VMM(1), %k0{%k1} ++ KMOV %k0, %VRAX ++ /* Compare [w]char for null, mask bit will be set for match. */ ++ ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRAX ++# else ++ inc %VRAX ++# endif ++ jz L(align_more) ++ ++ bsf %VRAX, %VRAX ++ ++# ifdef USE_AS_WCSCHR ++ leaq (%rdi, %rax, CHAR_SIZE), %rax ++# else ++ add %rdi, %rax ++# endif ++# ifndef USE_AS_STRCHRNUL ++ cmp (%rax), %CHAR_REG ++ jne L(zero) ++ ret ++L(zero): ++ xorl %eax, %eax ++# endif ++ ret ++ ++L(ret_vec_x3): ++ subq $-VEC_SIZE, %rdi ++L(ret_vec_x2): ++ subq $-VEC_SIZE, %rdi ++L(ret_vec_x1): ++ bsf %VRAX, %VRAX ++# ifdef USE_AS_WCSCHR ++ leaq (%rdi, %rax, CHAR_SIZE), %rax ++# else ++ add %rdi, %rax ++# endif ++ ++# ifndef USE_AS_STRCHRNUL ++ cmp (%rax), %CHAR_REG ++ jne L(zero) ++# endif ++ ret ++ ++L(page_cross): ++ mov %rdi, %rax ++ movl %edi, %ecx ++# ifdef USE_AS_WCSCHR ++ /* Calculate number of compare result bits to be skipped for ++ wide string alignment adjustment. */ ++ andl $(VEC_SIZE - 1), %ecx ++ sarl $2, %ecx ++# endif ++ /* ecx contains number of w[char] to be skipped as a result ++ of address alignment. */ ++ andq $-VEC_SIZE, %rax ++ ++ VMOVA (%rax), %VMM(1) ++ VPCMPNE %VMM(1), %VMM(0), %k1 ++ VPTEST %VMM(1), %VMM(1), %k0{%k1} ++ KMOV %k0, %VRAX ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRAX ++# else ++ inc %VRAX ++# endif ++ /* Ignore number of character for alignment adjustment. */ ++ shr %cl, %VRAX ++ jz L(align_more) ++ ++ bsf %VRAX, %VRAX ++# ifdef USE_AS_WCSCHR ++ leaq (%rdi, %rax, CHAR_SIZE), %rax ++# else ++ addq %rdi, %rax ++# endif ++ ++# ifndef USE_AS_STRCHRNUL ++ cmp (%rax), %CHAR_REG ++ jne L(zero) ++# endif ++ ret ++ ++L(align_more): ++ /* Align rax to VEC_SIZE. */ ++ andq $-VEC_SIZE, %rdi ++ ++ /* Loop unroll 4 times for 4 vector loop. */ ++ VMOVA VEC_SIZE(%rdi), %VMM(1) ++ VPCMPNE %VMM(1), %VMM(0), %k1 ++ VPTEST %VMM(1), %VMM(1), %k0{%k1} ++ ++ /* Increment rdi by vector size for further comparison and ++ return. */ ++ subq $-VEC_SIZE, %rdi ++ KMOV %k0, %VRAX ++ ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRAX ++# else ++ inc %VRAX ++# endif ++ jnz L(ret_vec_x1) ++ ++ VMOVA VEC_SIZE(%rdi), %VMM(1) ++ VPCMPNE %VMM(1), %VMM(0), %k1 ++ VPTEST %VMM(1), %VMM(1), %k0{%k1} ++ KMOV %k0, %VRAX ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRAX ++# else ++ inc %VRAX ++# endif ++ jnz L(ret_vec_x2) ++ ++ VMOVA (VEC_SIZE * 2)(%rdi), %VMM(1) ++ VPCMPNE %VMM(1), %VMM(0), %k1 ++ VPTEST %VMM(1), %VMM(1), %k0{%k1} ++ KMOV %k0, %VRAX ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRAX ++# else ++ inc %VRAX ++# endif ++ jnz L(ret_vec_x3) ++ ++ VMOVA (VEC_SIZE * 3)(%rdi), %VMM(1) ++ VPCMPNE %VMM(1), %VMM(0), %k1 ++ VPTEST %VMM(1), %VMM(1), %k0{%k1} ++ KMOV %k0, %VRDX ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRDX ++# else ++ inc %VRDX ++# endif ++ jnz L(ret_vec_x4) ++ ++ ++ /* Align address to VEC_SIZE * 4 for loop. */ ++ andq $-(VEC_SIZE * 4), %rdi ++L(loop): ++ /* VPMINU and VPCMP combination provide better performance as ++ compared to alternative combinations. */ ++ VMOVA (VEC_SIZE * 4)(%rdi), %VMM(1) ++ VMOVA (VEC_SIZE * 5)(%rdi), %VMM(2) ++ VMOVA (VEC_SIZE * 6)(%rdi), %VMM(3) ++ VMOVA (VEC_SIZE * 7)(%rdi), %VMM(4) ++ ++ VPCMPNE %VMM(1), %VMM(0), %k1 ++ VPCMPNE %VMM(2), %VMM(0), %k2 ++ ++ VPMINU %VMM(2), %VMM(1), %VMM(2) ++ ++ VPCMPNE %VMM(3), %VMM(0), %k3{%k1} ++ VPCMPNE %VMM(4), %VMM(0), %k4{%k2} ++ ++ VPMINU %VMM(4), %VMM(3), %VMM(4) ++ VPMINU %VMM(2), %VMM(4), %VMM(4){%k3}{z} ++ ++ VPTEST %VMM(4), %VMM(4), %k5{%k4} ++ ++ KMOV %k5, %VRDX ++ subq $-(VEC_SIZE * 4), %rdi ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRDX ++# else ++ inc %VRDX ++# endif ++ jz L(loop) ++ ++ VPTEST %VMM(1), %VMM(1), %k0{%k1} ++ KMOV %k0, %VRAX ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRAX ++# else ++ inc %VRAX ++# endif ++ jnz L(ret_vec_x1) ++ ++ VPTEST %VMM(2), %VMM(2), %k0{%k2} ++ KMOV %k0, %VRAX ++ /* At this point, if k1 is non zero, null char must be in the ++ second vector. */ ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRAX ++# else ++ inc %VRAX ++# endif ++ jnz L(ret_vec_x2) ++ ++ VPTEST %VMM(3), %VMM(3), %k0{%k3} ++ KMOV %k0, %VRAX ++# ifdef USE_AS_WCSCHR ++ sub $VEC_MATCH_MASK, %VRAX ++# else ++ inc %VRAX ++# endif ++ jnz L(ret_vec_x3) ++ /* At this point null [w]char must be in the fourth vector so no ++ need to check. */ ++ ++L(ret_vec_x4): ++ bsf %VRDX, %VRDX ++ leaq (VEC_SIZE * 3)(%rdi, %rdx, CHAR_SIZE), %rax ++# ifndef USE_AS_STRCHRNUL ++ cmp (%rax), %CHAR_REG ++ jne L(zero_2) ++# endif ++ ret ++ ++# ifndef USE_AS_STRCHRNUL ++L(zero_2): ++ xor %eax, %eax ++ ret ++# endif ++END (STRCHR) ++#endif +diff --git a/sysdeps/x86_64/multiarch/strchr-evex512.S b/sysdeps/x86_64/multiarch/strchr-evex512.S +new file mode 100644 +index 0000000000000000..a4ac0229523d1b01 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/strchr-evex512.S +@@ -0,0 +1,8 @@ ++# ifndef STRCHR ++# define STRCHR __strchr_evex512 ++# endif ++ ++#include "x86-evex512-vecs.h" ++#include "reg-macros.h" ++ ++#include "strchr-evex-base.S" +diff --git a/sysdeps/x86_64/multiarch/strchrnul-evex512.S b/sysdeps/x86_64/multiarch/strchrnul-evex512.S +new file mode 100644 +index 0000000000000000..1be0b12f385da936 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/strchrnul-evex512.S +@@ -0,0 +1,8 @@ ++#ifndef STRCHRNUL ++# define STRCHRNUL __strchrnul_evex512 ++#endif ++ ++#define STRCHR STRCHRNUL ++#define USE_AS_STRCHRNUL 1 ++ ++#include "strchr-evex512.S" +diff --git a/sysdeps/x86_64/multiarch/wcschr-evex512.S b/sysdeps/x86_64/multiarch/wcschr-evex512.S +new file mode 100644 +index 0000000000000000..3fe4e77a706b2bce +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/wcschr-evex512.S +@@ -0,0 +1,9 @@ ++#ifndef WCSCHR ++# define WCSCHR __wcschr_evex512 ++#endif ++ ++#define STRCHR WCSCHR ++#define USE_AS_WCSCHR 1 ++ ++#define USE_WIDE_CHAR 1 ++#include "strchr-evex512.S" diff --git a/glibc-RHEL-175520-8.patch b/glibc-RHEL-175520-8.patch new file mode 100644 index 0000000..7f39782 --- /dev/null +++ b/glibc-RHEL-175520-8.patch @@ -0,0 +1,210 @@ +commit e96971482de05eff92c1408b694c320cedd2d167 +Author: Sunil K Pandey +Date: Mon Oct 3 12:00:53 2022 -0700 + + x86-64: Improve evex512 version of strlen functions + + This patch improves following functionality + - Replace VPCMP with VPCMPEQ. + - Replace page cross check logic with sall. + - Remove extra lea from align_more. + - Remove uncondition loop jump. + - Use bsf to check max length in first vector. + + Reviewed-by: Noah Goldstein + +diff --git a/sysdeps/x86_64/multiarch/strlen-evex-base.S b/sysdeps/x86_64/multiarch/strlen-evex-base.S +index 176babee1e0a9e89..742ac9c9253ecb60 100644 +--- a/sysdeps/x86_64/multiarch/strlen-evex-base.S ++++ b/sysdeps/x86_64/multiarch/strlen-evex-base.S +@@ -21,12 +21,12 @@ + # include + + # ifdef USE_AS_WCSLEN +-# define VPCMP vpcmpd ++# define VPCMPEQ vpcmpeqd + # define VPTESTN vptestnmd + # define VPMINU vpminud + # define CHAR_SIZE 4 + # else +-# define VPCMP vpcmpb ++# define VPCMPEQ vpcmpeqb + # define VPTESTN vptestnmb + # define VPMINU vpminub + # define CHAR_SIZE 1 +@@ -51,20 +51,29 @@ ENTRY_P2ALIGN (STRLEN, 6) + + movl %edi, %eax + vpxorq %VMM_128(0), %VMM_128(0), %VMM_128(0) +- andl $(PAGE_SIZE - 1), %eax +- cmpl $(PAGE_SIZE - VEC_SIZE), %eax ++ sall $20, %eax ++ cmpl $((PAGE_SIZE - VEC_SIZE) << 20), %eax + ja L(page_cross) + + /* Compare [w]char for null, mask bit will be set for match. */ +- VPCMP $0, (%rdi), %VMM(0), %k0 ++ VPCMPEQ (%rdi), %VMM(0), %k0 ++# ifdef USE_AS_STRNLEN ++ KMOV %k0, %VRCX ++ /* Store max length in rax. */ ++ mov %rsi, %rax ++ /* If rcx is 0, rax will have max length. We can not use VRCX ++ and VRAX here for evex256 because, upper 32 bits may be ++ undefined for ecx and eax. */ ++ bsfq %rcx, %rax ++ cmp $CHAR_PER_VEC, %rax ++ ja L(align_more) ++ cmpq %rax, %rsi ++ cmovb %esi, %eax ++# else + KMOV %k0, %VRAX + test %VRAX, %VRAX + jz L(align_more) +- + bsf %VRAX, %VRAX +-# ifdef USE_AS_STRNLEN +- cmpq %rsi, %rax +- cmovnb %rsi, %rax + # endif + ret + +@@ -77,25 +86,24 @@ L(ret_max): + # endif + + L(align_more): +- leaq VEC_SIZE(%rdi), %rax ++ mov %rdi, %rax + /* Align rax to VEC_SIZE. */ + andq $-VEC_SIZE, %rax + # ifdef USE_AS_STRNLEN +- movq %rax, %rdx +- subq %rdi, %rdx ++ movq %rdi, %rdx ++ subq %rax, %rdx + # ifdef USE_AS_WCSLEN + shr $2, %VRDX + # endif + /* At this point rdx contains [w]chars already compared. */ +- subq %rsi, %rdx +- jae L(ret_max) +- negq %rdx ++ leaq -CHAR_PER_VEC(%rsi, %rdx), %rdx + /* At this point rdx contains number of w[char] needs to go. + Now onwards rdx will keep decrementing with each compare. */ + # endif + + /* Loop unroll 4 times for 4 vector loop. */ +- VPCMP $0, (%rax), %VMM(0), %k0 ++ VPCMPEQ VEC_SIZE(%rax), %VMM(0), %k0 ++ subq $-VEC_SIZE, %rax + KMOV %k0, %VRCX + test %VRCX, %VRCX + jnz L(ret_vec_x1) +@@ -105,7 +113,7 @@ L(align_more): + jbe L(ret_max) + # endif + +- VPCMP $0, VEC_SIZE(%rax), %VMM(0), %k0 ++ VPCMPEQ VEC_SIZE(%rax), %VMM(0), %k0 + KMOV %k0, %VRCX + test %VRCX, %VRCX + jnz L(ret_vec_x2) +@@ -115,7 +123,7 @@ L(align_more): + jbe L(ret_max) + # endif + +- VPCMP $0, (VEC_SIZE * 2)(%rax), %VMM(0), %k0 ++ VPCMPEQ (VEC_SIZE * 2)(%rax), %VMM(0), %k0 + KMOV %k0, %VRCX + test %VRCX, %VRCX + jnz L(ret_vec_x3) +@@ -125,7 +133,7 @@ L(align_more): + jbe L(ret_max) + # endif + +- VPCMP $0, (VEC_SIZE * 3)(%rax), %VMM(0), %k0 ++ VPCMPEQ (VEC_SIZE * 3)(%rax), %VMM(0), %k0 + KMOV %k0, %VRCX + test %VRCX, %VRCX + jnz L(ret_vec_x4) +@@ -151,16 +159,10 @@ L(align_more): + addq %rcx, %rdx + /* Need jump as we don't want to add/subtract rdx for first + iteration of 4 x VEC_SIZE aligned loop. */ +- jmp L(loop_entry) + # endif + + .p2align 4,,11 + L(loop): +-# ifdef USE_AS_STRNLEN +- subq $(CHAR_PER_VEC * 4), %rdx +- jbe L(ret_max) +-L(loop_entry): +-# endif + /* VPMINU and VPCMP combination provide better performance as + compared to alternative combinations. */ + VMOVA (VEC_SIZE * 4)(%rax), %VMM(1) +@@ -173,7 +175,18 @@ L(loop_entry): + + subq $-(VEC_SIZE * 4), %rax + KORTEST %k0, %k1 +- jz L(loop) ++ ++# ifndef USE_AS_STRNLEN ++ jz L(loop) ++# else ++ jnz L(loopend) ++ subq $(CHAR_PER_VEC * 4), %rdx ++ ja L(loop) ++ mov %rsi, %rax ++ ret ++# endif ++ ++L(loopend): + + VPTESTN %VMM(1), %VMM(1), %k2 + KMOV %k2, %VRCX +@@ -245,24 +258,34 @@ L(ret_vec_x1): + ret + + L(page_cross): +- movl %eax, %ecx +-# ifdef USE_AS_WCSLEN ++ mov %rdi, %rax ++ movl %edi, %ecx + andl $(VEC_SIZE - 1), %ecx ++# ifdef USE_AS_WCSLEN + sarl $2, %ecx + # endif + /* ecx contains number of w[char] to be skipped as a result + of address alignment. */ +- xorq %rdi, %rax +- VPCMP $0, (PAGE_SIZE - VEC_SIZE)(%rax), %VMM(0), %k0 +- KMOV %k0, %VRAX ++ andq $-VEC_SIZE, %rax ++ VPCMPEQ (%rax), %VMM(0), %k0 ++ KMOV %k0, %VRDX + /* Ignore number of character for alignment adjustment. */ +- shr %cl, %VRAX ++ shr %cl, %VRDX ++# ifdef USE_AS_STRNLEN ++ jnz L(page_cross_end) ++ movl $CHAR_PER_VEC, %eax ++ sub %ecx, %eax ++ cmp %rax, %rsi ++ ja L(align_more) ++# else + jz L(align_more) ++# endif + +- bsf %VRAX, %VRAX ++L(page_cross_end): ++ bsf %VRDX, %VRAX + # ifdef USE_AS_STRNLEN + cmpq %rsi, %rax +- cmovnb %rsi, %rax ++ cmovnb %esi, %eax + # endif + ret + diff --git a/glibc-RHEL-175520-9.patch b/glibc-RHEL-175520-9.patch new file mode 100644 index 0000000..5abe02f --- /dev/null +++ b/glibc-RHEL-175520-9.patch @@ -0,0 +1,388 @@ +commit faaf733f49211439475e50f06716b303ee2644bf +Author: Sunil K Pandey +Date: Tue Aug 9 07:57:29 2022 -0700 + + x86_64: Implement evex512 version of strrchr and wcsrchr + + Changes from v1: + Use vec api for register. + Replace VPCMP with VPCMPEQ + Restructure and remove 1 unconditional jump. + Change page cross logic to use sall. + + This patch implements following evex512 version of string functions. + evex512 version takes up to 30% less cycle as compared to evex, + depending on length and alignment. + + - strrchr function using 512 bit vectors. + - wcsrchr function using 512 bit vectors. + + Code size data: + + strrchr-evex.o 879 byte + strrchr-evex512.o 601 byte (-32%) + + wcsrchr-evex.o 882 byte + wcsrchr-evex512.o 572 byte (-35%) + + Placeholder function, not used by any processor at the moment. + + Reviewed-by: Noah Goldstein + +Conflicts: + sysdeps/x86_64/multiarch/Makefile + (fixup context) + sysdeps/x86_64/multiarch/ifunc-impl-list.c + (fixup macro name) + +diff --git a/sysdeps/x86_64/multiarch/Makefile b/sysdeps/x86_64/multiarch/Makefile +index 1c8bc2af0ef9c367..80009869b2a76c19 100644 +--- a/sysdeps/x86_64/multiarch/Makefile ++++ b/sysdeps/x86_64/multiarch/Makefile +@@ -129,6 +129,7 @@ sysdep_routines += \ + strrchr-avx2 \ + strrchr-avx2-rtm \ + strrchr-evex \ ++ strrchr-evex512 \ + strrchr-sse2 \ + strspn-c \ + strspn-sse2 \ +@@ -175,6 +176,7 @@ sysdep_routines += \ + wcsrchr-avx2 \ + wcsrchr-avx2-rtm \ + wcsrchr-evex \ ++ wcsrchr-evex512 \ + wcsrchr-sse2 \ + wmemchr-avx2 \ + wmemchr-avx2-rtm \ +diff --git a/sysdeps/x86_64/multiarch/ifunc-impl-list.c b/sysdeps/x86_64/multiarch/ifunc-impl-list.c +index cf482eff56fb663c..00bb9a56c039fcae 100644 +--- a/sysdeps/x86_64/multiarch/ifunc-impl-list.c ++++ b/sysdeps/x86_64/multiarch/ifunc-impl-list.c +@@ -532,6 +532,11 @@ __libc_ifunc_impl_list (const char *name, struct libc_ifunc_impl *array, + && CPU_FEATURE_USABLE (BMI1) + && CPU_FEATURE_USABLE (BMI2)), + __strrchr_evex) ++ IFUNC_IMPL_ADD (array, i, strrchr, ++ (CPU_FEATURE_USABLE (AVX512VL) ++ && CPU_FEATURE_USABLE (AVX512BW) ++ && CPU_FEATURE_USABLE (BMI2)), ++ __strrchr_evex512) + IFUNC_IMPL_ADD (array, i, strrchr, 1, __strrchr_sse2)) + + /* Support sysdeps/x86_64/multiarch/strcmp.c. */ +@@ -731,6 +736,11 @@ __libc_ifunc_impl_list (const char *name, struct libc_ifunc_impl *array, + && CPU_FEATURE_USABLE (BMI1) + && CPU_FEATURE_USABLE (BMI2)), + __wcsrchr_evex) ++ IFUNC_IMPL_ADD (array, i, wcsrchr, ++ (CPU_FEATURE_USABLE (AVX512VL) ++ && CPU_FEATURE_USABLE (AVX512BW) ++ && CPU_FEATURE_USABLE (BMI2)), ++ __wcsrchr_evex512) + IFUNC_IMPL_ADD (array, i, wcsrchr, 1, __wcsrchr_sse2)) + + /* Support sysdeps/x86_64/multiarch/wcscmp.c. */ +diff --git a/sysdeps/x86_64/multiarch/strrchr-evex-base.S b/sysdeps/x86_64/multiarch/strrchr-evex-base.S +new file mode 100644 +index 0000000000000000..81cab3e0178c8b1e +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/strrchr-evex-base.S +@@ -0,0 +1,264 @@ ++/* Placeholder function, not used by any processor at the moment. ++ Copyright (C) 2022 Free Software Foundation, Inc. ++ This file is part of the GNU C Library. ++ ++ The GNU C Library is free software; you can redistribute it and/or ++ modify it under the terms of the GNU Lesser General Public ++ License as published by the Free Software Foundation; either ++ version 2.1 of the License, or (at your option) any later version. ++ ++ The GNU C Library is distributed in the hope that it will be useful, ++ but WITHOUT ANY WARRANTY; without even the implied warranty of ++ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU ++ Lesser General Public License for more details. ++ ++ You should have received a copy of the GNU Lesser General Public ++ License along with the GNU C Library; if not, see ++ . */ ++ ++/* UNUSED. Exists purely as reference implementation. */ ++ ++#include ++ ++#if ISA_SHOULD_BUILD (4) ++ ++# include ++ ++# ifdef USE_AS_WCSRCHR ++# define CHAR_SIZE 4 ++# define VPBROADCAST vpbroadcastd ++# define VPCMPEQ vpcmpeqd ++# define VPMINU vpminud ++# define VPTESTN vptestnmd ++# else ++# define CHAR_SIZE 1 ++# define VPBROADCAST vpbroadcastb ++# define VPCMPEQ vpcmpeqb ++# define VPMINU vpminub ++# define VPTESTN vptestnmb ++# endif ++ ++# define PAGE_SIZE 4096 ++# define CHAR_PER_VEC (VEC_SIZE / CHAR_SIZE) ++ ++ .section SECTION(.text), "ax", @progbits ++/* Aligning entry point to 64 byte, provides better performance for ++ one vector length string. */ ++ENTRY_P2ALIGN (STRRCHR, 6) ++ ++ /* Broadcast CHAR to VMM(0). */ ++ VPBROADCAST %esi, %VMM(0) ++ movl %edi, %eax ++ sall $20, %eax ++ cmpl $((PAGE_SIZE - VEC_SIZE) << 20), %eax ++ ja L(page_cross) ++ ++L(page_cross_continue): ++ /* Compare [w]char for null, mask bit will be set for match. */ ++ VMOVU (%rdi), %VMM(1) ++ ++ VPTESTN %VMM(1), %VMM(1), %k1 ++ KMOV %k1, %VRCX ++ test %VRCX, %VRCX ++ jz L(align_more) ++ ++ VPCMPEQ %VMM(1), %VMM(0), %k0 ++ KMOV %k0, %VRAX ++ BLSMSK %VRCX, %VRCX ++ and %VRCX, %VRAX ++ jz L(ret) ++ ++ BSR %VRAX, %VRAX ++# ifdef USE_AS_WCSRCHR ++ leaq (%rdi, %rax, CHAR_SIZE), %rax ++# else ++ add %rdi, %rax ++# endif ++L(ret): ++ ret ++ ++L(vector_x2_end): ++ VPCMPEQ %VMM(2), %VMM(0), %k2 ++ KMOV %k2, %VRAX ++ BLSMSK %VRCX, %VRCX ++ and %VRCX, %VRAX ++ jz L(vector_x1_ret) ++ ++ BSR %VRAX, %VRAX ++ leaq (VEC_SIZE)(%rdi, %rax, CHAR_SIZE), %rax ++ ret ++ ++ /* Check the first vector at very last to look for match. */ ++L(vector_x1_ret): ++ VPCMPEQ %VMM(1), %VMM(0), %k2 ++ KMOV %k2, %VRAX ++ test %VRAX, %VRAX ++ jz L(ret) ++ ++ BSR %VRAX, %VRAX ++# ifdef USE_AS_WCSRCHR ++ leaq (%rsi, %rax, CHAR_SIZE), %rax ++# else ++ add %rsi, %rax ++# endif ++ ret ++ ++L(align_more): ++ /* Zero r8 to store match result. */ ++ xorl %r8d, %r8d ++ /* Save pointer of first vector, in case if no match found. */ ++ movq %rdi, %rsi ++ /* Align pointer to vector size. */ ++ andq $-VEC_SIZE, %rdi ++ /* Loop unroll for 2 vector loop. */ ++ VMOVA (VEC_SIZE)(%rdi), %VMM(2) ++ VPTESTN %VMM(2), %VMM(2), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX ++ jnz L(vector_x2_end) ++ ++ /* Save pointer of second vector, in case if no match ++ found. */ ++ movq %rdi, %r9 ++ /* Align address to VEC_SIZE * 2 for loop. */ ++ andq $-(VEC_SIZE * 2), %rdi ++ ++ .p2align 4,,11 ++L(loop): ++ /* 2 vector loop, as it provide better performance as compared ++ to 4 vector loop. */ ++ VMOVA (VEC_SIZE * 2)(%rdi), %VMM(3) ++ VMOVA (VEC_SIZE * 3)(%rdi), %VMM(4) ++ VPCMPEQ %VMM(3), %VMM(0), %k1 ++ VPCMPEQ %VMM(4), %VMM(0), %k2 ++ VPMINU %VMM(3), %VMM(4), %VMM(5) ++ VPTESTN %VMM(5), %VMM(5), %k0 ++ KOR %k1, %k2, %k3 ++ subq $-(VEC_SIZE * 2), %rdi ++ /* If k0 and k3 zero, match and end of string not found. */ ++ KORTEST %k0, %k3 ++ jz L(loop) ++ ++ /* If k0 is non zero, end of string found. */ ++ KORTEST %k0, %k0 ++ jnz L(endloop) ++ ++ lea VEC_SIZE(%rdi), %r8 ++ /* A match found, it need to be stored in r8 before loop ++ continue. */ ++ /* Check second vector first. */ ++ KMOV %k2, %VRDX ++ test %VRDX, %VRDX ++ jnz L(loop_vec_x2_match) ++ ++ KMOV %k1, %VRDX ++ /* Match is in first vector, rdi offset need to be substracted ++ by VEC_SIZE. */ ++ sub $VEC_SIZE, %r8 ++ ++ /* If second vector doesn't have match, first vector must ++ have match. */ ++L(loop_vec_x2_match): ++ BSR %VRDX, %VRDX ++# ifdef USE_AS_WCSRCHR ++ sal $2, %rdx ++# endif ++ add %rdx, %r8 ++ jmp L(loop) ++ ++L(endloop): ++ /* Check if string end in first loop vector. */ ++ VPTESTN %VMM(3), %VMM(3), %k0 ++ KMOV %k0, %VRCX ++ test %VRCX, %VRCX ++ jnz L(loop_vector_x1_end) ++ ++ /* Check if it has match in first loop vector. */ ++ KMOV %k1, %VRAX ++ test %VRAX, %VRAX ++ jz L(loop_vector_x2_end) ++ ++ BSR %VRAX, %VRAX ++ leaq (%rdi, %rax, CHAR_SIZE), %r8 ++ ++ /* String must end in second loop vector. */ ++L(loop_vector_x2_end): ++ VPTESTN %VMM(4), %VMM(4), %k0 ++ KMOV %k0, %VRCX ++ KMOV %k2, %VRAX ++ BLSMSK %VRCX, %VRCX ++ /* Check if it has match in second loop vector. */ ++ and %VRCX, %VRAX ++ jz L(check_last_match) ++ ++ BSR %VRAX, %VRAX ++ leaq (VEC_SIZE)(%rdi, %rax, CHAR_SIZE), %rax ++ ret ++ ++ /* String end in first loop vector. */ ++L(loop_vector_x1_end): ++ KMOV %k1, %VRAX ++ BLSMSK %VRCX, %VRCX ++ /* Check if it has match in second loop vector. */ ++ and %VRCX, %VRAX ++ jz L(check_last_match) ++ ++ BSR %VRAX, %VRAX ++ leaq (%rdi, %rax, CHAR_SIZE), %rax ++ ret ++ ++ /* No match in first and second loop vector. */ ++L(check_last_match): ++ /* Check if any match recorded in r8. */ ++ test %r8, %r8 ++ jz L(vector_x2_ret) ++ movq %r8, %rax ++ ret ++ ++ /* No match recorded in r8. Check the second saved vector ++ in begining. */ ++L(vector_x2_ret): ++ VPCMPEQ %VMM(2), %VMM(0), %k2 ++ KMOV %k2, %VRAX ++ test %VRAX, %VRAX ++ jz L(vector_x1_ret) ++ ++ /* Match found in the second saved vector. */ ++ BSR %VRAX, %VRAX ++ leaq (VEC_SIZE)(%r9, %rax, CHAR_SIZE), %rax ++ ret ++ ++L(page_cross): ++ mov %rdi, %rax ++ movl %edi, %ecx ++ ++# ifdef USE_AS_WCSRCHR ++ /* Calculate number of compare result bits to be skipped for ++ wide string alignment adjustment. */ ++ andl $(VEC_SIZE - 1), %ecx ++ sarl $2, %ecx ++# endif ++ /* ecx contains number of w[char] to be skipped as a result ++ of address alignment. */ ++ andq $-VEC_SIZE, %rax ++ VMOVA (%rax), %VMM(1) ++ VPTESTN %VMM(1), %VMM(1), %k1 ++ KMOV %k1, %VRAX ++ SHR %cl, %VRAX ++ jz L(page_cross_continue) ++ VPCMPEQ %VMM(1), %VMM(0), %k0 ++ KMOV %k0, %VRDX ++ SHR %cl, %VRDX ++ BLSMSK %VRAX, %VRAX ++ and %VRDX, %VRAX ++ jz L(ret) ++ BSR %VRAX, %VRAX ++# ifdef USE_AS_WCSRCHR ++ leaq (%rdi, %rax, CHAR_SIZE), %rax ++# else ++ add %rdi, %rax ++# endif ++ ++ ret ++END (STRRCHR) ++#endif +diff --git a/sysdeps/x86_64/multiarch/strrchr-evex512.S b/sysdeps/x86_64/multiarch/strrchr-evex512.S +new file mode 100644 +index 0000000000000000..7d81e663759a8b28 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/strrchr-evex512.S +@@ -0,0 +1,8 @@ ++# ifndef STRRCHR ++# define STRRCHR __strrchr_evex512 ++# endif ++ ++#include "x86-evex512-vecs.h" ++#include "reg-macros.h" ++ ++#include "strrchr-evex-base.S" +diff --git a/sysdeps/x86_64/multiarch/wcsrchr-evex512.S b/sysdeps/x86_64/multiarch/wcsrchr-evex512.S +new file mode 100644 +index 0000000000000000..f241f63716401a97 +--- /dev/null ++++ b/sysdeps/x86_64/multiarch/wcsrchr-evex512.S +@@ -0,0 +1,13 @@ ++#ifndef WCSRCHR ++# define WCSRCHR __wcsrchr_evex512 ++#endif ++ ++#define STRRCHR WCSRCHR ++#define USE_AS_WCSRCHR 1 ++ ++#define USE_WIDE_CHAR 1 ++ ++#include "x86-evex512-vecs.h" ++#include "reg-macros.h" ++ ++#include "strrchr-evex512.S"