diff --git a/llvm/test/CodeGen/X86/vector-reduce-and-bool.ll b/llvm/test/CodeGen/X86/vector-reduce-and-bool.ll index 068148d69498d..7db9ad872326d 100644 --- a/llvm/test/CodeGen/X86/vector-reduce-and-bool.ll +++ b/llvm/test/CodeGen/X86/vector-reduce-and-bool.ll @@ -111,30 +111,18 @@ define i1 @trunc_v4i32_v4i1(<4 x i32>) { ret i1 %b } -define i1 @trunc_v8i16_v8i1(<8 x i8>) { -; SSE2-LABEL: trunc_v8i16_v8i1: -; SSE2: # %bb.0: -; SSE2-NEXT: punpcklbw {{.*#+}} xmm0 = xmm0[0,0,1,1,2,2,3,3,4,4,5,5,6,6,7,7] -; SSE2-NEXT: psllw $15, %xmm0 -; SSE2-NEXT: packsswb %xmm0, %xmm0 -; SSE2-NEXT: pmovmskb %xmm0, %eax -; SSE2-NEXT: cmpb $-1, %al -; SSE2-NEXT: sete %al -; SSE2-NEXT: retq -; -; SSE41-LABEL: trunc_v8i16_v8i1: -; SSE41: # %bb.0: -; SSE41-NEXT: pmovzxbw {{.*#+}} xmm0 = xmm0[0],zero,xmm0[1],zero,xmm0[2],zero,xmm0[3],zero,xmm0[4],zero,xmm0[5],zero,xmm0[6],zero,xmm0[7],zero -; SSE41-NEXT: psllw $15, %xmm0 -; SSE41-NEXT: packsswb %xmm0, %xmm0 -; SSE41-NEXT: pmovmskb %xmm0, %eax -; SSE41-NEXT: cmpb $-1, %al -; SSE41-NEXT: sete %al -; SSE41-NEXT: retq +define i1 @trunc_v8i16_v8i1(<8 x i16>) { +; SSE-LABEL: trunc_v8i16_v8i1: +; SSE: # %bb.0: +; SSE-NEXT: psllw $15, %xmm0 +; SSE-NEXT: packsswb %xmm0, %xmm0 +; SSE-NEXT: pmovmskb %xmm0, %eax +; SSE-NEXT: cmpb $-1, %al +; SSE-NEXT: sete %al +; SSE-NEXT: retq ; ; AVX-LABEL: trunc_v8i16_v8i1: ; AVX: # %bb.0: -; AVX-NEXT: vpmovzxbw {{.*#+}} xmm0 = xmm0[0],zero,xmm0[1],zero,xmm0[2],zero,xmm0[3],zero,xmm0[4],zero,xmm0[5],zero,xmm0[6],zero,xmm0[7],zero ; AVX-NEXT: vpsllw $15, %xmm0, %xmm0 ; AVX-NEXT: vpacksswb %xmm0, %xmm0, %xmm0 ; AVX-NEXT: vpmovmskb %xmm0, %eax @@ -144,9 +132,9 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512F-LABEL: trunc_v8i16_v8i1: ; AVX512F: # %bb.0: -; AVX512F-NEXT: vpmovsxbd %xmm0, %zmm0 -; AVX512F-NEXT: vpslld $31, %zmm0, %zmm0 -; AVX512F-NEXT: vptestmd %zmm0, %zmm0, %k0 +; AVX512F-NEXT: vpmovsxwq %xmm0, %zmm0 +; AVX512F-NEXT: vpsllq $63, %zmm0, %zmm0 +; AVX512F-NEXT: vptestmq %zmm0, %zmm0, %k0 ; AVX512F-NEXT: kmovw %k0, %eax ; AVX512F-NEXT: cmpb $-1, %al ; AVX512F-NEXT: sete %al @@ -155,8 +143,8 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512BW-LABEL: trunc_v8i16_v8i1: ; AVX512BW: # %bb.0: -; AVX512BW-NEXT: vpsllw $7, %xmm0, %xmm0 -; AVX512BW-NEXT: vpmovb2m %zmm0, %k0 +; AVX512BW-NEXT: vpsllw $15, %xmm0, %xmm0 +; AVX512BW-NEXT: vpmovw2m %zmm0, %k0 ; AVX512BW-NEXT: kmovd %k0, %eax ; AVX512BW-NEXT: cmpb $-1, %al ; AVX512BW-NEXT: sete %al @@ -165,13 +153,13 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512VL-LABEL: trunc_v8i16_v8i1: ; AVX512VL: # %bb.0: -; AVX512VL-NEXT: vpsllw $7, %xmm0, %xmm0 -; AVX512VL-NEXT: vpmovb2m %xmm0, %k0 +; AVX512VL-NEXT: vpsllw $15, %xmm0, %xmm0 +; AVX512VL-NEXT: vpmovw2m %xmm0, %k0 ; AVX512VL-NEXT: kmovd %k0, %eax ; AVX512VL-NEXT: cmpb $-1, %al ; AVX512VL-NEXT: sete %al ; AVX512VL-NEXT: retq - %a = trunc <8 x i8> %0 to <8 x i1> + %a = trunc <8 x i16> %0 to <8 x i1> %b = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> %a) ret i1 %b } @@ -949,11 +937,12 @@ define i1 @icmp_v4i32_v4i1(<4 x i32>) { ret i1 %b } -define i1 @icmp_v8i16_v8i1(<8 x i8>) { +define i1 @icmp_v8i16_v8i1(<8 x i16>) { ; SSE-LABEL: icmp_v8i16_v8i1: ; SSE: # %bb.0: ; SSE-NEXT: pxor %xmm1, %xmm1 -; SSE-NEXT: pcmpeqb %xmm0, %xmm1 +; SSE-NEXT: pcmpeqw %xmm0, %xmm1 +; SSE-NEXT: packsswb %xmm1, %xmm1 ; SSE-NEXT: pmovmskb %xmm1, %eax ; SSE-NEXT: cmpb $-1, %al ; SSE-NEXT: sete %al @@ -962,7 +951,8 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; AVX-LABEL: icmp_v8i16_v8i1: ; AVX: # %bb.0: ; AVX-NEXT: vpxor %xmm1, %xmm1, %xmm1 -; AVX-NEXT: vpcmpeqb %xmm1, %xmm0, %xmm0 +; AVX-NEXT: vpcmpeqw %xmm1, %xmm0, %xmm0 +; AVX-NEXT: vpacksswb %xmm0, %xmm0, %xmm0 ; AVX-NEXT: vpmovmskb %xmm0, %eax ; AVX-NEXT: cmpb $-1, %al ; AVX-NEXT: sete %al @@ -971,9 +961,9 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; AVX512F-LABEL: icmp_v8i16_v8i1: ; AVX512F: # %bb.0: ; AVX512F-NEXT: vpxor %xmm1, %xmm1, %xmm1 -; AVX512F-NEXT: vpcmpeqb %xmm1, %xmm0, %xmm0 -; AVX512F-NEXT: vpmovsxbd %xmm0, %zmm0 -; AVX512F-NEXT: vptestmd %zmm0, %zmm0, %k0 +; AVX512F-NEXT: vpcmpeqw %xmm1, %xmm0, %xmm0 +; AVX512F-NEXT: vpmovsxwq %xmm0, %zmm0 +; AVX512F-NEXT: vptestmq %zmm0, %zmm0, %k0 ; AVX512F-NEXT: kmovw %k0, %eax ; AVX512F-NEXT: cmpb $-1, %al ; AVX512F-NEXT: sete %al @@ -983,7 +973,7 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; AVX512BW-LABEL: icmp_v8i16_v8i1: ; AVX512BW: # %bb.0: ; AVX512BW-NEXT: # kill: def $xmm0 killed $xmm0 def $zmm0 -; AVX512BW-NEXT: vptestnmb %zmm0, %zmm0, %k0 +; AVX512BW-NEXT: vptestnmw %zmm0, %zmm0, %k0 ; AVX512BW-NEXT: kmovd %k0, %eax ; AVX512BW-NEXT: cmpb $-1, %al ; AVX512BW-NEXT: sete %al @@ -992,12 +982,12 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; ; AVX512VL-LABEL: icmp_v8i16_v8i1: ; AVX512VL: # %bb.0: -; AVX512VL-NEXT: vptestnmb %xmm0, %xmm0, %k0 +; AVX512VL-NEXT: vptestnmw %xmm0, %xmm0, %k0 ; AVX512VL-NEXT: kmovd %k0, %eax ; AVX512VL-NEXT: cmpb $-1, %al ; AVX512VL-NEXT: sete %al ; AVX512VL-NEXT: retq - %a = icmp eq <8 x i8> %0, zeroinitializer + %a = icmp eq <8 x i16> %0, zeroinitializer %b = call i1 @llvm.vector.reduce.and.v8i1(<8 x i1> %a) ret i1 %b } diff --git a/llvm/test/CodeGen/X86/vector-reduce-or-bool.ll b/llvm/test/CodeGen/X86/vector-reduce-or-bool.ll index 6269cd4df1119..b06822b93fd95 100644 --- a/llvm/test/CodeGen/X86/vector-reduce-or-bool.ll +++ b/llvm/test/CodeGen/X86/vector-reduce-or-bool.ll @@ -111,28 +111,17 @@ define i1 @trunc_v4i32_v4i1(<4 x i32>) { ret i1 %b } -define i1 @trunc_v8i16_v8i1(<8 x i8>) { -; SSE2-LABEL: trunc_v8i16_v8i1: -; SSE2: # %bb.0: -; SSE2-NEXT: punpcklbw {{.*#+}} xmm0 = xmm0[0,0,1,1,2,2,3,3,4,4,5,5,6,6,7,7] -; SSE2-NEXT: psllw $15, %xmm0 -; SSE2-NEXT: pmovmskb %xmm0, %eax -; SSE2-NEXT: testl $43690, %eax # imm = 0xAAAA -; SSE2-NEXT: setne %al -; SSE2-NEXT: retq -; -; SSE41-LABEL: trunc_v8i16_v8i1: -; SSE41: # %bb.0: -; SSE41-NEXT: pmovzxbw {{.*#+}} xmm0 = xmm0[0],zero,xmm0[1],zero,xmm0[2],zero,xmm0[3],zero,xmm0[4],zero,xmm0[5],zero,xmm0[6],zero,xmm0[7],zero -; SSE41-NEXT: psllw $15, %xmm0 -; SSE41-NEXT: pmovmskb %xmm0, %eax -; SSE41-NEXT: testl $43690, %eax # imm = 0xAAAA -; SSE41-NEXT: setne %al -; SSE41-NEXT: retq +define i1 @trunc_v8i16_v8i1(<8 x i16>) { +; SSE-LABEL: trunc_v8i16_v8i1: +; SSE: # %bb.0: +; SSE-NEXT: psllw $15, %xmm0 +; SSE-NEXT: pmovmskb %xmm0, %eax +; SSE-NEXT: testl $43690, %eax # imm = 0xAAAA +; SSE-NEXT: setne %al +; SSE-NEXT: retq ; ; AVX-LABEL: trunc_v8i16_v8i1: ; AVX: # %bb.0: -; AVX-NEXT: vpmovzxbw {{.*#+}} xmm0 = xmm0[0],zero,xmm0[1],zero,xmm0[2],zero,xmm0[3],zero,xmm0[4],zero,xmm0[5],zero,xmm0[6],zero,xmm0[7],zero ; AVX-NEXT: vpsllw $15, %xmm0, %xmm0 ; AVX-NEXT: vpmovmskb %xmm0, %eax ; AVX-NEXT: testl $43690, %eax # imm = 0xAAAA @@ -141,9 +130,9 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512F-LABEL: trunc_v8i16_v8i1: ; AVX512F: # %bb.0: -; AVX512F-NEXT: vpmovsxbd %xmm0, %zmm0 -; AVX512F-NEXT: vpslld $31, %zmm0, %zmm0 -; AVX512F-NEXT: vptestmd %zmm0, %zmm0, %k0 +; AVX512F-NEXT: vpmovsxwq %xmm0, %zmm0 +; AVX512F-NEXT: vpsllq $63, %zmm0, %zmm0 +; AVX512F-NEXT: vptestmq %zmm0, %zmm0, %k0 ; AVX512F-NEXT: kmovw %k0, %eax ; AVX512F-NEXT: testb %al, %al ; AVX512F-NEXT: setne %al @@ -152,8 +141,8 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512BW-LABEL: trunc_v8i16_v8i1: ; AVX512BW: # %bb.0: -; AVX512BW-NEXT: vpsllw $7, %xmm0, %xmm0 -; AVX512BW-NEXT: vpmovb2m %zmm0, %k0 +; AVX512BW-NEXT: vpsllw $15, %xmm0, %xmm0 +; AVX512BW-NEXT: vpmovw2m %zmm0, %k0 ; AVX512BW-NEXT: kmovd %k0, %eax ; AVX512BW-NEXT: testb %al, %al ; AVX512BW-NEXT: setne %al @@ -162,13 +151,13 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512VL-LABEL: trunc_v8i16_v8i1: ; AVX512VL: # %bb.0: -; AVX512VL-NEXT: vpsllw $7, %xmm0, %xmm0 -; AVX512VL-NEXT: vpmovb2m %xmm0, %k0 +; AVX512VL-NEXT: vpsllw $15, %xmm0, %xmm0 +; AVX512VL-NEXT: vpmovw2m %xmm0, %k0 ; AVX512VL-NEXT: kmovd %k0, %eax ; AVX512VL-NEXT: testb %al, %al ; AVX512VL-NEXT: setne %al ; AVX512VL-NEXT: retq - %a = trunc <8 x i8> %0 to <8 x i1> + %a = trunc <8 x i16> %0 to <8 x i1> %b = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> %a) ret i1 %b } @@ -947,31 +936,31 @@ define i1 @icmp_v4i32_v4i1(<4 x i32>) { ret i1 %b } -define i1 @icmp_v8i16_v8i1(<8 x i8>) { +define i1 @icmp_v8i16_v8i1(<8 x i16>) { ; SSE-LABEL: icmp_v8i16_v8i1: ; SSE: # %bb.0: ; SSE-NEXT: pxor %xmm1, %xmm1 -; SSE-NEXT: pcmpeqb %xmm0, %xmm1 +; SSE-NEXT: pcmpeqw %xmm0, %xmm1 ; SSE-NEXT: pmovmskb %xmm1, %eax -; SSE-NEXT: testb %al, %al +; SSE-NEXT: testl %eax, %eax ; SSE-NEXT: setne %al ; SSE-NEXT: retq ; ; AVX-LABEL: icmp_v8i16_v8i1: ; AVX: # %bb.0: ; AVX-NEXT: vpxor %xmm1, %xmm1, %xmm1 -; AVX-NEXT: vpcmpeqb %xmm1, %xmm0, %xmm0 +; AVX-NEXT: vpcmpeqw %xmm1, %xmm0, %xmm0 ; AVX-NEXT: vpmovmskb %xmm0, %eax -; AVX-NEXT: testb %al, %al +; AVX-NEXT: testl %eax, %eax ; AVX-NEXT: setne %al ; AVX-NEXT: retq ; ; AVX512F-LABEL: icmp_v8i16_v8i1: ; AVX512F: # %bb.0: ; AVX512F-NEXT: vpxor %xmm1, %xmm1, %xmm1 -; AVX512F-NEXT: vpcmpeqb %xmm1, %xmm0, %xmm0 -; AVX512F-NEXT: vpmovsxbd %xmm0, %zmm0 -; AVX512F-NEXT: vptestmd %zmm0, %zmm0, %k0 +; AVX512F-NEXT: vpcmpeqw %xmm1, %xmm0, %xmm0 +; AVX512F-NEXT: vpmovsxwq %xmm0, %zmm0 +; AVX512F-NEXT: vptestmq %zmm0, %zmm0, %k0 ; AVX512F-NEXT: kmovw %k0, %eax ; AVX512F-NEXT: testb %al, %al ; AVX512F-NEXT: setne %al @@ -981,7 +970,7 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; AVX512BW-LABEL: icmp_v8i16_v8i1: ; AVX512BW: # %bb.0: ; AVX512BW-NEXT: # kill: def $xmm0 killed $xmm0 def $zmm0 -; AVX512BW-NEXT: vptestnmb %zmm0, %zmm0, %k0 +; AVX512BW-NEXT: vptestnmw %zmm0, %zmm0, %k0 ; AVX512BW-NEXT: kmovd %k0, %eax ; AVX512BW-NEXT: testb %al, %al ; AVX512BW-NEXT: setne %al @@ -990,12 +979,12 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; ; AVX512VL-LABEL: icmp_v8i16_v8i1: ; AVX512VL: # %bb.0: -; AVX512VL-NEXT: vptestnmb %xmm0, %xmm0, %k0 +; AVX512VL-NEXT: vptestnmw %xmm0, %xmm0, %k0 ; AVX512VL-NEXT: kmovd %k0, %eax ; AVX512VL-NEXT: testb %al, %al ; AVX512VL-NEXT: setne %al ; AVX512VL-NEXT: retq - %a = icmp eq <8 x i8> %0, zeroinitializer + %a = icmp eq <8 x i16> %0, zeroinitializer %b = call i1 @llvm.vector.reduce.or.v8i1(<8 x i1> %a) ret i1 %b } diff --git a/llvm/test/CodeGen/X86/vector-reduce-xor-bool.ll b/llvm/test/CodeGen/X86/vector-reduce-xor-bool.ll index 7ef49de025531..0d0accd3f1f18 100644 --- a/llvm/test/CodeGen/X86/vector-reduce-xor-bool.ll +++ b/llvm/test/CodeGen/X86/vector-reduce-xor-bool.ll @@ -111,30 +111,18 @@ define i1 @trunc_v4i32_v4i1(<4 x i32>) { ret i1 %b } -define i1 @trunc_v8i16_v8i1(<8 x i8>) { -; SSE2-LABEL: trunc_v8i16_v8i1: -; SSE2: # %bb.0: -; SSE2-NEXT: punpcklbw {{.*#+}} xmm0 = xmm0[0,0,1,1,2,2,3,3,4,4,5,5,6,6,7,7] -; SSE2-NEXT: psllw $15, %xmm0 -; SSE2-NEXT: packsswb %xmm0, %xmm0 -; SSE2-NEXT: pmovmskb %xmm0, %eax -; SSE2-NEXT: testb %al, %al -; SSE2-NEXT: setnp %al -; SSE2-NEXT: retq -; -; SSE41-LABEL: trunc_v8i16_v8i1: -; SSE41: # %bb.0: -; SSE41-NEXT: pmovzxbw {{.*#+}} xmm0 = xmm0[0],zero,xmm0[1],zero,xmm0[2],zero,xmm0[3],zero,xmm0[4],zero,xmm0[5],zero,xmm0[6],zero,xmm0[7],zero -; SSE41-NEXT: psllw $15, %xmm0 -; SSE41-NEXT: packsswb %xmm0, %xmm0 -; SSE41-NEXT: pmovmskb %xmm0, %eax -; SSE41-NEXT: testb %al, %al -; SSE41-NEXT: setnp %al -; SSE41-NEXT: retq +define i1 @trunc_v8i16_v8i1(<8 x i16>) { +; SSE-LABEL: trunc_v8i16_v8i1: +; SSE: # %bb.0: +; SSE-NEXT: psllw $15, %xmm0 +; SSE-NEXT: packsswb %xmm0, %xmm0 +; SSE-NEXT: pmovmskb %xmm0, %eax +; SSE-NEXT: testb %al, %al +; SSE-NEXT: setnp %al +; SSE-NEXT: retq ; ; AVX-LABEL: trunc_v8i16_v8i1: ; AVX: # %bb.0: -; AVX-NEXT: vpmovzxbw {{.*#+}} xmm0 = xmm0[0],zero,xmm0[1],zero,xmm0[2],zero,xmm0[3],zero,xmm0[4],zero,xmm0[5],zero,xmm0[6],zero,xmm0[7],zero ; AVX-NEXT: vpsllw $15, %xmm0, %xmm0 ; AVX-NEXT: vpacksswb %xmm0, %xmm0, %xmm0 ; AVX-NEXT: vpmovmskb %xmm0, %eax @@ -144,9 +132,9 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512F-LABEL: trunc_v8i16_v8i1: ; AVX512F: # %bb.0: -; AVX512F-NEXT: vpmovsxbd %xmm0, %zmm0 -; AVX512F-NEXT: vpslld $31, %zmm0, %zmm0 -; AVX512F-NEXT: vptestmd %zmm0, %zmm0, %k0 +; AVX512F-NEXT: vpmovsxwq %xmm0, %zmm0 +; AVX512F-NEXT: vpsllq $63, %zmm0, %zmm0 +; AVX512F-NEXT: vptestmq %zmm0, %zmm0, %k0 ; AVX512F-NEXT: kmovw %k0, %eax ; AVX512F-NEXT: testb %al, %al ; AVX512F-NEXT: setnp %al @@ -155,8 +143,8 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512BW-LABEL: trunc_v8i16_v8i1: ; AVX512BW: # %bb.0: -; AVX512BW-NEXT: vpsllw $7, %xmm0, %xmm0 -; AVX512BW-NEXT: vpmovb2m %zmm0, %k0 +; AVX512BW-NEXT: vpsllw $15, %xmm0, %xmm0 +; AVX512BW-NEXT: vpmovw2m %zmm0, %k0 ; AVX512BW-NEXT: kmovd %k0, %eax ; AVX512BW-NEXT: testb %al, %al ; AVX512BW-NEXT: setnp %al @@ -165,13 +153,13 @@ define i1 @trunc_v8i16_v8i1(<8 x i8>) { ; ; AVX512VL-LABEL: trunc_v8i16_v8i1: ; AVX512VL: # %bb.0: -; AVX512VL-NEXT: vpsllw $7, %xmm0, %xmm0 -; AVX512VL-NEXT: vpmovb2m %xmm0, %k0 +; AVX512VL-NEXT: vpsllw $15, %xmm0, %xmm0 +; AVX512VL-NEXT: vpmovw2m %xmm0, %k0 ; AVX512VL-NEXT: kmovd %k0, %eax ; AVX512VL-NEXT: testb %al, %al ; AVX512VL-NEXT: setnp %al ; AVX512VL-NEXT: retq - %a = trunc <8 x i8> %0 to <8 x i1> + %a = trunc <8 x i16> %0 to <8 x i1> %b = call i1 @llvm.vector.reduce.xor.v8i1(<8 x i1> %a) ret i1 %b } @@ -1027,11 +1015,12 @@ define i1 @icmp_v4i32_v4i1(<4 x i32>) { ret i1 %b } -define i1 @icmp_v8i16_v8i1(<8 x i8>) { +define i1 @icmp_v8i16_v8i1(<8 x i16>) { ; SSE-LABEL: icmp_v8i16_v8i1: ; SSE: # %bb.0: ; SSE-NEXT: pxor %xmm1, %xmm1 -; SSE-NEXT: pcmpeqb %xmm0, %xmm1 +; SSE-NEXT: pcmpeqw %xmm0, %xmm1 +; SSE-NEXT: packsswb %xmm1, %xmm1 ; SSE-NEXT: pmovmskb %xmm1, %eax ; SSE-NEXT: testb %al, %al ; SSE-NEXT: setnp %al @@ -1040,7 +1029,8 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; AVX-LABEL: icmp_v8i16_v8i1: ; AVX: # %bb.0: ; AVX-NEXT: vpxor %xmm1, %xmm1, %xmm1 -; AVX-NEXT: vpcmpeqb %xmm1, %xmm0, %xmm0 +; AVX-NEXT: vpcmpeqw %xmm1, %xmm0, %xmm0 +; AVX-NEXT: vpacksswb %xmm0, %xmm0, %xmm0 ; AVX-NEXT: vpmovmskb %xmm0, %eax ; AVX-NEXT: testb %al, %al ; AVX-NEXT: setnp %al @@ -1049,9 +1039,9 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; AVX512F-LABEL: icmp_v8i16_v8i1: ; AVX512F: # %bb.0: ; AVX512F-NEXT: vpxor %xmm1, %xmm1, %xmm1 -; AVX512F-NEXT: vpcmpeqb %xmm1, %xmm0, %xmm0 -; AVX512F-NEXT: vpmovsxbd %xmm0, %zmm0 -; AVX512F-NEXT: vptestmd %zmm0, %zmm0, %k0 +; AVX512F-NEXT: vpcmpeqw %xmm1, %xmm0, %xmm0 +; AVX512F-NEXT: vpmovsxwq %xmm0, %zmm0 +; AVX512F-NEXT: vptestmq %zmm0, %zmm0, %k0 ; AVX512F-NEXT: kmovw %k0, %eax ; AVX512F-NEXT: testb %al, %al ; AVX512F-NEXT: setnp %al @@ -1061,7 +1051,7 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; AVX512BW-LABEL: icmp_v8i16_v8i1: ; AVX512BW: # %bb.0: ; AVX512BW-NEXT: # kill: def $xmm0 killed $xmm0 def $zmm0 -; AVX512BW-NEXT: vptestnmb %zmm0, %zmm0, %k0 +; AVX512BW-NEXT: vptestnmw %zmm0, %zmm0, %k0 ; AVX512BW-NEXT: kmovd %k0, %eax ; AVX512BW-NEXT: testb %al, %al ; AVX512BW-NEXT: setnp %al @@ -1070,12 +1060,12 @@ define i1 @icmp_v8i16_v8i1(<8 x i8>) { ; ; AVX512VL-LABEL: icmp_v8i16_v8i1: ; AVX512VL: # %bb.0: -; AVX512VL-NEXT: vptestnmb %xmm0, %xmm0, %k0 +; AVX512VL-NEXT: vptestnmw %xmm0, %xmm0, %k0 ; AVX512VL-NEXT: kmovd %k0, %eax ; AVX512VL-NEXT: testb %al, %al ; AVX512VL-NEXT: setnp %al ; AVX512VL-NEXT: retq - %a = icmp eq <8 x i8> %0, zeroinitializer + %a = icmp eq <8 x i16> %0, zeroinitializer %b = call i1 @llvm.vector.reduce.xor.v8i1(<8 x i1> %a) ret i1 %b }