xref: /llvm-project/llvm/test/CodeGen/X86/vector-idiv-udiv-512.ll (revision 61d5addd942a5ef8128e48d3617419e6320d8280)
1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py
2; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx512f | FileCheck %s --check-prefix=AVX --check-prefix=AVX512F
3; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx512bw | FileCheck %s --check-prefix=AVX --check-prefix=AVX512BW
4
5;
6; udiv by 7
7;
8
9define <8 x i64> @test_div7_8i64(<8 x i64> %a) nounwind {
10; AVX-LABEL: test_div7_8i64:
11; AVX:       # %bb.0:
12; AVX-NEXT:    vextracti32x4 $3, %zmm0, %xmm1
13; AVX-NEXT:    vpextrq $1, %xmm1, %rcx
14; AVX-NEXT:    movabsq $2635249153387078803, %rsi # imm = 0x2492492492492493
15; AVX-NEXT:    movq %rcx, %rax
16; AVX-NEXT:    mulq %rsi
17; AVX-NEXT:    subq %rdx, %rcx
18; AVX-NEXT:    shrq %rcx
19; AVX-NEXT:    addq %rdx, %rcx
20; AVX-NEXT:    vmovq %rcx, %xmm2
21; AVX-NEXT:    vmovq %xmm1, %rcx
22; AVX-NEXT:    movq %rcx, %rax
23; AVX-NEXT:    mulq %rsi
24; AVX-NEXT:    subq %rdx, %rcx
25; AVX-NEXT:    shrq %rcx
26; AVX-NEXT:    addq %rdx, %rcx
27; AVX-NEXT:    vmovq %rcx, %xmm1
28; AVX-NEXT:    vpunpcklqdq {{.*#+}} xmm1 = xmm1[0],xmm2[0]
29; AVX-NEXT:    vextracti32x4 $2, %zmm0, %xmm2
30; AVX-NEXT:    vpextrq $1, %xmm2, %rcx
31; AVX-NEXT:    movq %rcx, %rax
32; AVX-NEXT:    mulq %rsi
33; AVX-NEXT:    subq %rdx, %rcx
34; AVX-NEXT:    shrq %rcx
35; AVX-NEXT:    addq %rdx, %rcx
36; AVX-NEXT:    vmovq %rcx, %xmm3
37; AVX-NEXT:    vmovq %xmm2, %rcx
38; AVX-NEXT:    movq %rcx, %rax
39; AVX-NEXT:    mulq %rsi
40; AVX-NEXT:    subq %rdx, %rcx
41; AVX-NEXT:    shrq %rcx
42; AVX-NEXT:    addq %rdx, %rcx
43; AVX-NEXT:    vmovq %rcx, %xmm2
44; AVX-NEXT:    vpunpcklqdq {{.*#+}} xmm2 = xmm2[0],xmm3[0]
45; AVX-NEXT:    vinserti128 $1, %xmm1, %ymm2, %ymm1
46; AVX-NEXT:    vextracti128 $1, %ymm0, %xmm2
47; AVX-NEXT:    vpextrq $1, %xmm2, %rcx
48; AVX-NEXT:    movq %rcx, %rax
49; AVX-NEXT:    mulq %rsi
50; AVX-NEXT:    subq %rdx, %rcx
51; AVX-NEXT:    shrq %rcx
52; AVX-NEXT:    addq %rdx, %rcx
53; AVX-NEXT:    vmovq %rcx, %xmm3
54; AVX-NEXT:    vmovq %xmm2, %rcx
55; AVX-NEXT:    movq %rcx, %rax
56; AVX-NEXT:    mulq %rsi
57; AVX-NEXT:    subq %rdx, %rcx
58; AVX-NEXT:    shrq %rcx
59; AVX-NEXT:    addq %rdx, %rcx
60; AVX-NEXT:    vmovq %rcx, %xmm2
61; AVX-NEXT:    vpunpcklqdq {{.*#+}} xmm2 = xmm2[0],xmm3[0]
62; AVX-NEXT:    vpextrq $1, %xmm0, %rcx
63; AVX-NEXT:    movq %rcx, %rax
64; AVX-NEXT:    mulq %rsi
65; AVX-NEXT:    subq %rdx, %rcx
66; AVX-NEXT:    shrq %rcx
67; AVX-NEXT:    addq %rdx, %rcx
68; AVX-NEXT:    vmovq %rcx, %xmm3
69; AVX-NEXT:    vmovq %xmm0, %rcx
70; AVX-NEXT:    movq %rcx, %rax
71; AVX-NEXT:    mulq %rsi
72; AVX-NEXT:    subq %rdx, %rcx
73; AVX-NEXT:    shrq %rcx
74; AVX-NEXT:    addq %rdx, %rcx
75; AVX-NEXT:    vmovq %rcx, %xmm0
76; AVX-NEXT:    vpunpcklqdq {{.*#+}} xmm0 = xmm0[0],xmm3[0]
77; AVX-NEXT:    vinserti128 $1, %xmm2, %ymm0, %ymm0
78; AVX-NEXT:    vinserti64x4 $1, %ymm1, %zmm0, %zmm0
79; AVX-NEXT:    vpsrlq $2, %zmm0, %zmm0
80; AVX-NEXT:    retq
81  %res = udiv <8 x i64> %a, <i64 7, i64 7, i64 7, i64 7, i64 7, i64 7, i64 7, i64 7>
82  ret <8 x i64> %res
83}
84
85define <16 x i32> @test_div7_16i32(<16 x i32> %a) nounwind {
86; AVX-LABEL: test_div7_16i32:
87; AVX:       # %bb.0:
88; AVX-NEXT:    vpbroadcastd {{.*#+}} zmm1 = [613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757]
89; AVX-NEXT:    vpmuludq %zmm1, %zmm0, %zmm2
90; AVX-NEXT:    vpshufd {{.*#+}} zmm3 = zmm0[1,1,3,3,5,5,7,7,9,9,11,11,13,13,15,15]
91; AVX-NEXT:    vpmuludq %zmm1, %zmm3, %zmm1
92; AVX-NEXT:    vpmovsxbd {{.*#+}} zmm3 = [1,17,3,19,5,21,7,23,9,25,11,27,13,29,15,31]
93; AVX-NEXT:    vpermi2d %zmm1, %zmm2, %zmm3
94; AVX-NEXT:    vpsubd %zmm3, %zmm0, %zmm0
95; AVX-NEXT:    vpsrld $1, %zmm0, %zmm0
96; AVX-NEXT:    vpaddd %zmm3, %zmm0, %zmm0
97; AVX-NEXT:    vpsrld $2, %zmm0, %zmm0
98; AVX-NEXT:    retq
99  %res = udiv <16 x i32> %a, <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7>
100  ret <16 x i32> %res
101}
102
103define <32 x i16> @test_div7_32i16(<32 x i16> %a) nounwind {
104; AVX512F-LABEL: test_div7_32i16:
105; AVX512F:       # %bb.0:
106; AVX512F-NEXT:    vpbroadcastw {{.*#+}} ymm1 = [9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363]
107; AVX512F-NEXT:    vpmulhuw %ymm1, %ymm0, %ymm2
108; AVX512F-NEXT:    vpsubw %ymm2, %ymm0, %ymm3
109; AVX512F-NEXT:    vpsrlw $1, %ymm3, %ymm3
110; AVX512F-NEXT:    vpaddw %ymm2, %ymm3, %ymm2
111; AVX512F-NEXT:    vpsrlw $2, %ymm2, %ymm2
112; AVX512F-NEXT:    vextracti64x4 $1, %zmm0, %ymm0
113; AVX512F-NEXT:    vpmulhuw %ymm1, %ymm0, %ymm1
114; AVX512F-NEXT:    vpsubw %ymm1, %ymm0, %ymm0
115; AVX512F-NEXT:    vpsrlw $1, %ymm0, %ymm0
116; AVX512F-NEXT:    vpaddw %ymm1, %ymm0, %ymm0
117; AVX512F-NEXT:    vpsrlw $2, %ymm0, %ymm0
118; AVX512F-NEXT:    vinserti64x4 $1, %ymm0, %zmm2, %zmm0
119; AVX512F-NEXT:    retq
120;
121; AVX512BW-LABEL: test_div7_32i16:
122; AVX512BW:       # %bb.0:
123; AVX512BW-NEXT:    vpmulhuw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm0, %zmm1 # [9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363]
124; AVX512BW-NEXT:    vpsubw %zmm1, %zmm0, %zmm0
125; AVX512BW-NEXT:    vpsrlw $1, %zmm0, %zmm0
126; AVX512BW-NEXT:    vpaddw %zmm1, %zmm0, %zmm0
127; AVX512BW-NEXT:    vpsrlw $2, %zmm0, %zmm0
128; AVX512BW-NEXT:    retq
129  %res = udiv <32 x i16> %a, <i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7>
130  ret <32 x i16> %res
131}
132
133define <64 x i8> @test_div7_64i8(<64 x i8> %a) nounwind {
134; AVX512F-LABEL: test_div7_64i8:
135; AVX512F:       # %bb.0:
136; AVX512F-NEXT:    vpxor %xmm1, %xmm1, %xmm1
137; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm2 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31]
138; AVX512F-NEXT:    vpbroadcastw {{.*#+}} ymm3 = [37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37]
139; AVX512F-NEXT:    vpmullw %ymm3, %ymm2, %ymm2
140; AVX512F-NEXT:    vpsrlw $8, %ymm2, %ymm2
141; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm4 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23]
142; AVX512F-NEXT:    vpmullw %ymm3, %ymm4, %ymm4
143; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
144; AVX512F-NEXT:    vpackuswb %ymm2, %ymm4, %ymm2
145; AVX512F-NEXT:    vpsubb %ymm2, %ymm0, %ymm4
146; AVX512F-NEXT:    vpsrlw $1, %ymm4, %ymm4
147; AVX512F-NEXT:    vpbroadcastb {{.*#+}} ymm5 = [127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127]
148; AVX512F-NEXT:    vpand %ymm5, %ymm4, %ymm4
149; AVX512F-NEXT:    vpaddb %ymm2, %ymm4, %ymm2
150; AVX512F-NEXT:    vpsrlw $2, %ymm2, %ymm2
151; AVX512F-NEXT:    vextracti64x4 $1, %zmm0, %ymm0
152; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm4 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31]
153; AVX512F-NEXT:    vpmullw %ymm3, %ymm4, %ymm4
154; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
155; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm1 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23]
156; AVX512F-NEXT:    vpmullw %ymm3, %ymm1, %ymm1
157; AVX512F-NEXT:    vpsrlw $8, %ymm1, %ymm1
158; AVX512F-NEXT:    vpackuswb %ymm4, %ymm1, %ymm1
159; AVX512F-NEXT:    vpsubb %ymm1, %ymm0, %ymm0
160; AVX512F-NEXT:    vpsrlw $1, %ymm0, %ymm0
161; AVX512F-NEXT:    vpand %ymm5, %ymm0, %ymm0
162; AVX512F-NEXT:    vpaddb %ymm1, %ymm0, %ymm0
163; AVX512F-NEXT:    vpsrlw $2, %ymm0, %ymm0
164; AVX512F-NEXT:    vinserti64x4 $1, %ymm0, %zmm2, %zmm0
165; AVX512F-NEXT:    vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm0, %zmm0
166; AVX512F-NEXT:    retq
167;
168; AVX512BW-LABEL: test_div7_64i8:
169; AVX512BW:       # %bb.0:
170; AVX512BW-NEXT:    vpxor %xmm1, %xmm1, %xmm1
171; AVX512BW-NEXT:    vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63]
172; AVX512BW-NEXT:    vpbroadcastw {{.*#+}} zmm3 = [37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37]
173; AVX512BW-NEXT:    vpmullw %zmm3, %zmm2, %zmm2
174; AVX512BW-NEXT:    vpsrlw $8, %zmm2, %zmm2
175; AVX512BW-NEXT:    vpunpcklbw {{.*#+}} zmm1 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55]
176; AVX512BW-NEXT:    vpmullw %zmm3, %zmm1, %zmm1
177; AVX512BW-NEXT:    vpsrlw $8, %zmm1, %zmm1
178; AVX512BW-NEXT:    vpackuswb %zmm2, %zmm1, %zmm1
179; AVX512BW-NEXT:    vpsubb %zmm1, %zmm0, %zmm0
180; AVX512BW-NEXT:    vpsrlw $1, %zmm0, %zmm0
181; AVX512BW-NEXT:    vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm0, %zmm0
182; AVX512BW-NEXT:    vpaddb %zmm1, %zmm0, %zmm0
183; AVX512BW-NEXT:    vpsrlw $2, %zmm0, %zmm0
184; AVX512BW-NEXT:    vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm0, %zmm0
185; AVX512BW-NEXT:    retq
186  %res = udiv <64 x i8> %a, <i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7>
187  ret <64 x i8> %res
188}
189
190;
191; udiv by non-splat constant
192;
193
194define <64 x i8> @test_divconstant_64i8(<64 x i8> %a) nounwind {
195; AVX512F-LABEL: test_divconstant_64i8:
196; AVX512F:       # %bb.0:
197; AVX512F-NEXT:    vextracti64x4 $1, %zmm0, %ymm2
198; AVX512F-NEXT:    vpxor %xmm1, %xmm1, %xmm1
199; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm3 = ymm2[8],ymm1[8],ymm2[9],ymm1[9],ymm2[10],ymm1[10],ymm2[11],ymm1[11],ymm2[12],ymm1[12],ymm2[13],ymm1[13],ymm2[14],ymm1[14],ymm2[15],ymm1[15],ymm2[24],ymm1[24],ymm2[25],ymm1[25],ymm2[26],ymm1[26],ymm2[27],ymm1[27],ymm2[28],ymm1[28],ymm2[29],ymm1[29],ymm2[30],ymm1[30],ymm2[31],ymm1[31]
200; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [256,256,64,256,256,256,256,256,128,256,256,256,256,256,256,256]
201; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
202; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [137,27,37,19,79,41,171,101,147,79,171,117,205,57,32,37]
203; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
204; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm4 = ymm2[0],ymm1[0],ymm2[1],ymm1[1],ymm2[2],ymm1[2],ymm2[3],ymm1[3],ymm2[4],ymm1[4],ymm2[5],ymm1[5],ymm2[6],ymm1[6],ymm2[7],ymm1[7],ymm2[16],ymm1[16],ymm2[17],ymm1[17],ymm2[18],ymm1[18],ymm2[19],ymm1[19],ymm2[20],ymm1[20],ymm2[21],ymm1[21],ymm2[22],ymm1[22],ymm2[23],ymm1[23]
205; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [256,256,256,256,256,256,256,256,128,256,256,256,256,256,256,256]
206; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
207; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [27,111,57,235,241,249,8,9,187,135,205,27,57,241,16,137]
208; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
209; AVX512F-NEXT:    vpackuswb %ymm3, %ymm4, %ymm3
210; AVX512F-NEXT:    vpsubb %ymm3, %ymm2, %ymm2
211; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm4 = ymm2[8],ymm1[8],ymm2[9],ymm1[9],ymm2[10],ymm1[10],ymm2[11],ymm1[11],ymm2[12],ymm1[12],ymm2[13],ymm1[13],ymm2[14],ymm1[14],ymm2[15],ymm1[15],ymm2[24],ymm1[24],ymm2[25],ymm1[25],ymm2[26],ymm1[26],ymm2[27],ymm1[27],ymm2[28],ymm1[28],ymm2[29],ymm1[29],ymm2[30],ymm1[30],ymm2[31],ymm1[31]
212; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [0,128,0,0,0,0,0,128,0,0,0,128,0,0,0,128]
213; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
214; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm2 = ymm2[0],ymm1[0],ymm2[1],ymm1[1],ymm2[2],ymm1[2],ymm2[3],ymm1[3],ymm2[4],ymm1[4],ymm2[5],ymm1[5],ymm2[6],ymm1[6],ymm2[7],ymm1[7],ymm2[16],ymm1[16],ymm2[17],ymm1[17],ymm2[18],ymm1[18],ymm2[19],ymm1[19],ymm2[20],ymm1[20],ymm2[21],ymm1[21],ymm2[22],ymm1[22],ymm2[23],ymm1[23]
215; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm2, %ymm2 # [0,0,0,0,0,0,0,128,0,128,0,0,0,0,0,0]
216; AVX512F-NEXT:    vpsrlw $8, %ymm2, %ymm2
217; AVX512F-NEXT:    vpackuswb %ymm4, %ymm2, %ymm2
218; AVX512F-NEXT:    vpaddb %ymm3, %ymm2, %ymm2
219; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm3 = ymm2[8],ymm1[8],ymm2[9],ymm1[9],ymm2[10],ymm1[10],ymm2[11],ymm1[11],ymm2[12],ymm1[12],ymm2[13],ymm1[13],ymm2[14],ymm1[14],ymm2[15],ymm1[15],ymm2[24],ymm1[24],ymm2[25],ymm1[25],ymm2[26],ymm1[26],ymm2[27],ymm1[27],ymm2[28],ymm1[28],ymm2[29],ymm1[29],ymm2[30],ymm1[30],ymm2[31],ymm1[31]
220; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [16,16,256,128,32,64,16,16,64,64,32,32,32,128,256,64]
221; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
222; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm2 = ymm2[0],ymm1[0],ymm2[1],ymm1[1],ymm2[2],ymm1[2],ymm2[3],ymm1[3],ymm2[4],ymm1[4],ymm2[5],ymm1[5],ymm2[6],ymm1[6],ymm2[7],ymm1[7],ymm2[16],ymm1[16],ymm2[17],ymm1[17],ymm2[18],ymm1[18],ymm2[19],ymm1[19],ymm2[20],ymm1[20],ymm2[21],ymm1[21],ymm2[22],ymm1[22],ymm2[23],ymm1[23]
223; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm2, %ymm2 # [64,16,32,8,8,8,256,16,32,16,16,128,64,16,256,32]
224; AVX512F-NEXT:    vpsrlw $8, %ymm2, %ymm2
225; AVX512F-NEXT:    vpackuswb %ymm3, %ymm2, %ymm2
226; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm3 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31]
227; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [256,256,256,256,256,256,256,128,256,256,256,256,256,256,256,256]
228; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
229; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [137,16,241,57,27,205,135,187,9,8,249,241,235,57,111,27]
230; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
231; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm4 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23]
232; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [256,256,256,256,256,256,256,128,256,256,256,256,256,64,256,256]
233; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
234; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [37,32,57,205,117,171,79,147,101,171,41,79,19,37,27,137]
235; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
236; AVX512F-NEXT:    vpackuswb %ymm3, %ymm4, %ymm3
237; AVX512F-NEXT:    vpsubb %ymm3, %ymm0, %ymm0
238; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm4 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31]
239; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [0,0,0,0,0,0,128,0,128,0,0,0,0,0,0,0]
240; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
241; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm0 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23]
242; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm0, %ymm0 # [128,0,0,0,128,0,0,0,128,0,0,0,0,0,128,0]
243; AVX512F-NEXT:    vpsrlw $8, %ymm0, %ymm0
244; AVX512F-NEXT:    vpackuswb %ymm4, %ymm0, %ymm0
245; AVX512F-NEXT:    vpaddb %ymm3, %ymm0, %ymm0
246; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm3 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31]
247; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [32,256,16,64,128,16,16,32,16,256,8,8,8,32,16,64]
248; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
249; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm0 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23]
250; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm0, %ymm0 # [64,256,128,32,32,32,64,64,16,16,64,32,128,256,16,16]
251; AVX512F-NEXT:    vpsrlw $8, %ymm0, %ymm0
252; AVX512F-NEXT:    vpackuswb %ymm3, %ymm0, %ymm0
253; AVX512F-NEXT:    vinserti64x4 $1, %ymm2, %zmm0, %zmm0
254; AVX512F-NEXT:    retq
255;
256; AVX512BW-LABEL: test_divconstant_64i8:
257; AVX512BW:       # %bb.0:
258; AVX512BW-NEXT:    vpxor %xmm1, %xmm1, %xmm1
259; AVX512BW-NEXT:    vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63]
260; AVX512BW-NEXT:    vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2
261; AVX512BW-NEXT:    vpsrlw $8, %zmm2, %zmm2
262; AVX512BW-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2 # [137,16,241,57,27,205,135,187,9,8,249,241,235,57,111,27,137,27,37,19,79,41,171,101,147,79,171,117,205,57,32,37]
263; AVX512BW-NEXT:    vpsrlw $8, %zmm2, %zmm2
264; AVX512BW-NEXT:    vpunpcklbw {{.*#+}} zmm3 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55]
265; AVX512BW-NEXT:    vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3
266; AVX512BW-NEXT:    vpsrlw $8, %zmm3, %zmm3
267; AVX512BW-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 # [37,32,57,205,117,171,79,147,101,171,41,79,19,37,27,137,27,111,57,235,241,249,8,9,187,135,205,27,57,241,16,137]
268; AVX512BW-NEXT:    vpsrlw $8, %zmm3, %zmm3
269; AVX512BW-NEXT:    vpackuswb %zmm2, %zmm3, %zmm2
270; AVX512BW-NEXT:    vpsubb %zmm2, %zmm0, %zmm0
271; AVX512BW-NEXT:    vpunpckhbw {{.*#+}} zmm3 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63]
272; AVX512BW-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 # [0,0,0,0,0,0,128,0,128,0,0,0,0,0,0,0,0,128,0,0,0,0,0,128,0,0,0,128,0,0,0,128]
273; AVX512BW-NEXT:    vpsrlw $8, %zmm3, %zmm3
274; AVX512BW-NEXT:    vpunpcklbw {{.*#+}} zmm0 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55]
275; AVX512BW-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm0, %zmm0 # [128,0,0,0,128,0,0,0,128,0,0,0,0,0,128,0,0,0,0,0,0,0,0,128,0,128,0,0,0,0,0,0]
276; AVX512BW-NEXT:    vpsrlw $8, %zmm0, %zmm0
277; AVX512BW-NEXT:    vpackuswb %zmm3, %zmm0, %zmm0
278; AVX512BW-NEXT:    vpaddb %zmm2, %zmm0, %zmm0
279; AVX512BW-NEXT:    vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63]
280; AVX512BW-NEXT:    vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2
281; AVX512BW-NEXT:    vpsrlw $8, %zmm2, %zmm2
282; AVX512BW-NEXT:    vpunpcklbw {{.*#+}} zmm0 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55]
283; AVX512BW-NEXT:    vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm0, %zmm0
284; AVX512BW-NEXT:    vpsrlw $8, %zmm0, %zmm0
285; AVX512BW-NEXT:    vpackuswb %zmm2, %zmm0, %zmm0
286; AVX512BW-NEXT:    retq
287  %res = udiv <64 x i8> %a, <i8 7, i8 8, i8 9, i8 10, i8 11, i8 12, i8 13, i8 14, i8 15, i8 16, i8 17, i8 18, i8 19, i8 20, i8 21, i8 22, i8 23, i8 24, i8 25, i8 26, i8 27, i8 28, i8 29, i8 30, i8 31, i8 32, i8 33, i8 34, i8 35, i8 36, i8 37, i8 38, i8 38, i8 37, i8 36, i8 35, i8 34, i8 33, i8 32, i8 31, i8 30, i8 29, i8 28, i8 27, i8 26, i8 25, i8 24, i8 23, i8 22, i8 21, i8 20, i8 19, i8 18, i8 17, i8 16, i8 15, i8 14, i8 13, i8 12, i8 11, i8 10, i8 9, i8 8, i8 7>
288  ret <64 x i8> %res
289}
290
291;
292; urem by 7
293;
294
295define <8 x i64> @test_rem7_8i64(<8 x i64> %a) nounwind {
296; AVX-LABEL: test_rem7_8i64:
297; AVX:       # %bb.0:
298; AVX-NEXT:    vextracti32x4 $3, %zmm0, %xmm1
299; AVX-NEXT:    vpextrq $1, %xmm1, %rcx
300; AVX-NEXT:    movabsq $2635249153387078803, %rsi # imm = 0x2492492492492493
301; AVX-NEXT:    movq %rcx, %rax
302; AVX-NEXT:    mulq %rsi
303; AVX-NEXT:    movq %rcx, %rax
304; AVX-NEXT:    subq %rdx, %rax
305; AVX-NEXT:    shrq %rax
306; AVX-NEXT:    addq %rdx, %rax
307; AVX-NEXT:    shrq $2, %rax
308; AVX-NEXT:    leaq (,%rax,8), %rdx
309; AVX-NEXT:    subq %rdx, %rax
310; AVX-NEXT:    addq %rcx, %rax
311; AVX-NEXT:    vmovq %rax, %xmm2
312; AVX-NEXT:    vmovq %xmm1, %rcx
313; AVX-NEXT:    movq %rcx, %rax
314; AVX-NEXT:    mulq %rsi
315; AVX-NEXT:    movq %rcx, %rax
316; AVX-NEXT:    subq %rdx, %rax
317; AVX-NEXT:    shrq %rax
318; AVX-NEXT:    addq %rdx, %rax
319; AVX-NEXT:    shrq $2, %rax
320; AVX-NEXT:    leaq (,%rax,8), %rdx
321; AVX-NEXT:    subq %rdx, %rax
322; AVX-NEXT:    addq %rcx, %rax
323; AVX-NEXT:    vmovq %rax, %xmm1
324; AVX-NEXT:    vpunpcklqdq {{.*#+}} xmm1 = xmm1[0],xmm2[0]
325; AVX-NEXT:    vextracti32x4 $2, %zmm0, %xmm2
326; AVX-NEXT:    vpextrq $1, %xmm2, %rcx
327; AVX-NEXT:    movq %rcx, %rax
328; AVX-NEXT:    mulq %rsi
329; AVX-NEXT:    movq %rcx, %rax
330; AVX-NEXT:    subq %rdx, %rax
331; AVX-NEXT:    shrq %rax
332; AVX-NEXT:    addq %rdx, %rax
333; AVX-NEXT:    shrq $2, %rax
334; AVX-NEXT:    leaq (,%rax,8), %rdx
335; AVX-NEXT:    subq %rdx, %rax
336; AVX-NEXT:    addq %rcx, %rax
337; AVX-NEXT:    vmovq %rax, %xmm3
338; AVX-NEXT:    vmovq %xmm2, %rcx
339; AVX-NEXT:    movq %rcx, %rax
340; AVX-NEXT:    mulq %rsi
341; AVX-NEXT:    movq %rcx, %rax
342; AVX-NEXT:    subq %rdx, %rax
343; AVX-NEXT:    shrq %rax
344; AVX-NEXT:    addq %rdx, %rax
345; AVX-NEXT:    shrq $2, %rax
346; AVX-NEXT:    leaq (,%rax,8), %rdx
347; AVX-NEXT:    subq %rdx, %rax
348; AVX-NEXT:    addq %rcx, %rax
349; AVX-NEXT:    vmovq %rax, %xmm2
350; AVX-NEXT:    vpunpcklqdq {{.*#+}} xmm2 = xmm2[0],xmm3[0]
351; AVX-NEXT:    vinserti128 $1, %xmm1, %ymm2, %ymm1
352; AVX-NEXT:    vextracti128 $1, %ymm0, %xmm2
353; AVX-NEXT:    vpextrq $1, %xmm2, %rcx
354; AVX-NEXT:    movq %rcx, %rax
355; AVX-NEXT:    mulq %rsi
356; AVX-NEXT:    movq %rcx, %rax
357; AVX-NEXT:    subq %rdx, %rax
358; AVX-NEXT:    shrq %rax
359; AVX-NEXT:    addq %rdx, %rax
360; AVX-NEXT:    shrq $2, %rax
361; AVX-NEXT:    leaq (,%rax,8), %rdx
362; AVX-NEXT:    subq %rdx, %rax
363; AVX-NEXT:    addq %rcx, %rax
364; AVX-NEXT:    vmovq %rax, %xmm3
365; AVX-NEXT:    vmovq %xmm2, %rcx
366; AVX-NEXT:    movq %rcx, %rax
367; AVX-NEXT:    mulq %rsi
368; AVX-NEXT:    movq %rcx, %rax
369; AVX-NEXT:    subq %rdx, %rax
370; AVX-NEXT:    shrq %rax
371; AVX-NEXT:    addq %rdx, %rax
372; AVX-NEXT:    shrq $2, %rax
373; AVX-NEXT:    leaq (,%rax,8), %rdx
374; AVX-NEXT:    subq %rdx, %rax
375; AVX-NEXT:    addq %rcx, %rax
376; AVX-NEXT:    vmovq %rax, %xmm2
377; AVX-NEXT:    vpunpcklqdq {{.*#+}} xmm2 = xmm2[0],xmm3[0]
378; AVX-NEXT:    vpextrq $1, %xmm0, %rcx
379; AVX-NEXT:    movq %rcx, %rax
380; AVX-NEXT:    mulq %rsi
381; AVX-NEXT:    movq %rcx, %rax
382; AVX-NEXT:    subq %rdx, %rax
383; AVX-NEXT:    shrq %rax
384; AVX-NEXT:    addq %rdx, %rax
385; AVX-NEXT:    shrq $2, %rax
386; AVX-NEXT:    leaq (,%rax,8), %rdx
387; AVX-NEXT:    subq %rdx, %rax
388; AVX-NEXT:    addq %rcx, %rax
389; AVX-NEXT:    vmovq %rax, %xmm3
390; AVX-NEXT:    vmovq %xmm0, %rcx
391; AVX-NEXT:    movq %rcx, %rax
392; AVX-NEXT:    mulq %rsi
393; AVX-NEXT:    movq %rcx, %rax
394; AVX-NEXT:    subq %rdx, %rax
395; AVX-NEXT:    shrq %rax
396; AVX-NEXT:    addq %rdx, %rax
397; AVX-NEXT:    shrq $2, %rax
398; AVX-NEXT:    leaq (,%rax,8), %rdx
399; AVX-NEXT:    subq %rdx, %rax
400; AVX-NEXT:    addq %rcx, %rax
401; AVX-NEXT:    vmovq %rax, %xmm0
402; AVX-NEXT:    vpunpcklqdq {{.*#+}} xmm0 = xmm0[0],xmm3[0]
403; AVX-NEXT:    vinserti128 $1, %xmm2, %ymm0, %ymm0
404; AVX-NEXT:    vinserti64x4 $1, %ymm1, %zmm0, %zmm0
405; AVX-NEXT:    retq
406  %res = urem <8 x i64> %a, <i64 7, i64 7, i64 7, i64 7, i64 7, i64 7, i64 7, i64 7>
407  ret <8 x i64> %res
408}
409
410define <16 x i32> @test_rem7_16i32(<16 x i32> %a) nounwind {
411; AVX-LABEL: test_rem7_16i32:
412; AVX:       # %bb.0:
413; AVX-NEXT:    vpbroadcastd {{.*#+}} zmm1 = [613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757]
414; AVX-NEXT:    vpmuludq %zmm1, %zmm0, %zmm2
415; AVX-NEXT:    vpshufd {{.*#+}} zmm3 = zmm0[1,1,3,3,5,5,7,7,9,9,11,11,13,13,15,15]
416; AVX-NEXT:    vpmuludq %zmm1, %zmm3, %zmm1
417; AVX-NEXT:    vpmovsxbd {{.*#+}} zmm3 = [1,17,3,19,5,21,7,23,9,25,11,27,13,29,15,31]
418; AVX-NEXT:    vpermi2d %zmm1, %zmm2, %zmm3
419; AVX-NEXT:    vpsubd %zmm3, %zmm0, %zmm1
420; AVX-NEXT:    vpsrld $1, %zmm1, %zmm1
421; AVX-NEXT:    vpaddd %zmm3, %zmm1, %zmm1
422; AVX-NEXT:    vpsrld $2, %zmm1, %zmm1
423; AVX-NEXT:    vpslld $3, %zmm1, %zmm2
424; AVX-NEXT:    vpsubd %zmm2, %zmm1, %zmm1
425; AVX-NEXT:    vpaddd %zmm1, %zmm0, %zmm0
426; AVX-NEXT:    retq
427  %res = urem <16 x i32> %a, <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7>
428  ret <16 x i32> %res
429}
430
431define <32 x i16> @test_rem7_32i16(<32 x i16> %a) nounwind {
432; AVX512F-LABEL: test_rem7_32i16:
433; AVX512F:       # %bb.0:
434; AVX512F-NEXT:    vextracti64x4 $1, %zmm0, %ymm1
435; AVX512F-NEXT:    vpbroadcastw {{.*#+}} ymm2 = [9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363]
436; AVX512F-NEXT:    vpmulhuw %ymm2, %ymm1, %ymm3
437; AVX512F-NEXT:    vpsubw %ymm3, %ymm1, %ymm4
438; AVX512F-NEXT:    vpsrlw $1, %ymm4, %ymm4
439; AVX512F-NEXT:    vpaddw %ymm3, %ymm4, %ymm3
440; AVX512F-NEXT:    vpsrlw $2, %ymm3, %ymm3
441; AVX512F-NEXT:    vpsllw $3, %ymm3, %ymm4
442; AVX512F-NEXT:    vpsubw %ymm4, %ymm3, %ymm3
443; AVX512F-NEXT:    vpaddw %ymm3, %ymm1, %ymm1
444; AVX512F-NEXT:    vpmulhuw %ymm2, %ymm0, %ymm2
445; AVX512F-NEXT:    vpsubw %ymm2, %ymm0, %ymm3
446; AVX512F-NEXT:    vpsrlw $1, %ymm3, %ymm3
447; AVX512F-NEXT:    vpaddw %ymm2, %ymm3, %ymm2
448; AVX512F-NEXT:    vpsrlw $2, %ymm2, %ymm2
449; AVX512F-NEXT:    vpsllw $3, %ymm2, %ymm3
450; AVX512F-NEXT:    vpsubw %ymm3, %ymm2, %ymm2
451; AVX512F-NEXT:    vpaddw %ymm2, %ymm0, %ymm0
452; AVX512F-NEXT:    vinserti64x4 $1, %ymm1, %zmm0, %zmm0
453; AVX512F-NEXT:    retq
454;
455; AVX512BW-LABEL: test_rem7_32i16:
456; AVX512BW:       # %bb.0:
457; AVX512BW-NEXT:    vpmulhuw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm0, %zmm1 # [9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363]
458; AVX512BW-NEXT:    vpsubw %zmm1, %zmm0, %zmm2
459; AVX512BW-NEXT:    vpsrlw $1, %zmm2, %zmm2
460; AVX512BW-NEXT:    vpaddw %zmm1, %zmm2, %zmm1
461; AVX512BW-NEXT:    vpsrlw $2, %zmm1, %zmm1
462; AVX512BW-NEXT:    vpsllw $3, %zmm1, %zmm2
463; AVX512BW-NEXT:    vpsubw %zmm2, %zmm1, %zmm1
464; AVX512BW-NEXT:    vpaddw %zmm1, %zmm0, %zmm0
465; AVX512BW-NEXT:    retq
466  %res = urem <32 x i16> %a, <i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7>
467  ret <32 x i16> %res
468}
469
470define <64 x i8> @test_rem7_64i8(<64 x i8> %a) nounwind {
471; AVX512F-LABEL: test_rem7_64i8:
472; AVX512F:       # %bb.0:
473; AVX512F-NEXT:    vextracti64x4 $1, %zmm0, %ymm1
474; AVX512F-NEXT:    vpxor %xmm2, %xmm2, %xmm2
475; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm3 = ymm1[8],ymm2[8],ymm1[9],ymm2[9],ymm1[10],ymm2[10],ymm1[11],ymm2[11],ymm1[12],ymm2[12],ymm1[13],ymm2[13],ymm1[14],ymm2[14],ymm1[15],ymm2[15],ymm1[24],ymm2[24],ymm1[25],ymm2[25],ymm1[26],ymm2[26],ymm1[27],ymm2[27],ymm1[28],ymm2[28],ymm1[29],ymm2[29],ymm1[30],ymm2[30],ymm1[31],ymm2[31]
476; AVX512F-NEXT:    vpbroadcastw {{.*#+}} ymm4 = [37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37]
477; AVX512F-NEXT:    vpmullw %ymm4, %ymm3, %ymm3
478; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
479; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm5 = ymm1[0],ymm2[0],ymm1[1],ymm2[1],ymm1[2],ymm2[2],ymm1[3],ymm2[3],ymm1[4],ymm2[4],ymm1[5],ymm2[5],ymm1[6],ymm2[6],ymm1[7],ymm2[7],ymm1[16],ymm2[16],ymm1[17],ymm2[17],ymm1[18],ymm2[18],ymm1[19],ymm2[19],ymm1[20],ymm2[20],ymm1[21],ymm2[21],ymm1[22],ymm2[22],ymm1[23],ymm2[23]
480; AVX512F-NEXT:    vpmullw %ymm4, %ymm5, %ymm5
481; AVX512F-NEXT:    vpsrlw $8, %ymm5, %ymm5
482; AVX512F-NEXT:    vpackuswb %ymm3, %ymm5, %ymm3
483; AVX512F-NEXT:    vpsubb %ymm3, %ymm1, %ymm5
484; AVX512F-NEXT:    vpsrlw $1, %ymm5, %ymm5
485; AVX512F-NEXT:    vpbroadcastb {{.*#+}} ymm6 = [127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127]
486; AVX512F-NEXT:    vpand %ymm6, %ymm5, %ymm5
487; AVX512F-NEXT:    vpaddb %ymm3, %ymm5, %ymm3
488; AVX512F-NEXT:    vpsllw $1, %ymm3, %ymm5
489; AVX512F-NEXT:    vpbroadcastb {{.*#+}} ymm7 = [248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248]
490; AVX512F-NEXT:    vpand %ymm7, %ymm5, %ymm5
491; AVX512F-NEXT:    vpsrlw $2, %ymm3, %ymm3
492; AVX512F-NEXT:    vpbroadcastb {{.*#+}} ymm8 = [63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63]
493; AVX512F-NEXT:    vpand %ymm3, %ymm8, %ymm3
494; AVX512F-NEXT:    vpsubb %ymm5, %ymm3, %ymm3
495; AVX512F-NEXT:    vpaddb %ymm3, %ymm1, %ymm1
496; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm3 = ymm0[8],ymm2[8],ymm0[9],ymm2[9],ymm0[10],ymm2[10],ymm0[11],ymm2[11],ymm0[12],ymm2[12],ymm0[13],ymm2[13],ymm0[14],ymm2[14],ymm0[15],ymm2[15],ymm0[24],ymm2[24],ymm0[25],ymm2[25],ymm0[26],ymm2[26],ymm0[27],ymm2[27],ymm0[28],ymm2[28],ymm0[29],ymm2[29],ymm0[30],ymm2[30],ymm0[31],ymm2[31]
497; AVX512F-NEXT:    vpmullw %ymm4, %ymm3, %ymm3
498; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
499; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm2 = ymm0[0],ymm2[0],ymm0[1],ymm2[1],ymm0[2],ymm2[2],ymm0[3],ymm2[3],ymm0[4],ymm2[4],ymm0[5],ymm2[5],ymm0[6],ymm2[6],ymm0[7],ymm2[7],ymm0[16],ymm2[16],ymm0[17],ymm2[17],ymm0[18],ymm2[18],ymm0[19],ymm2[19],ymm0[20],ymm2[20],ymm0[21],ymm2[21],ymm0[22],ymm2[22],ymm0[23],ymm2[23]
500; AVX512F-NEXT:    vpmullw %ymm4, %ymm2, %ymm2
501; AVX512F-NEXT:    vpsrlw $8, %ymm2, %ymm2
502; AVX512F-NEXT:    vpackuswb %ymm3, %ymm2, %ymm2
503; AVX512F-NEXT:    vpsubb %ymm2, %ymm0, %ymm3
504; AVX512F-NEXT:    vpsrlw $1, %ymm3, %ymm3
505; AVX512F-NEXT:    vpand %ymm6, %ymm3, %ymm3
506; AVX512F-NEXT:    vpaddb %ymm2, %ymm3, %ymm2
507; AVX512F-NEXT:    vpsllw $1, %ymm2, %ymm3
508; AVX512F-NEXT:    vpand %ymm7, %ymm3, %ymm3
509; AVX512F-NEXT:    vpsrlw $2, %ymm2, %ymm2
510; AVX512F-NEXT:    vpand %ymm2, %ymm8, %ymm2
511; AVX512F-NEXT:    vpsubb %ymm3, %ymm2, %ymm2
512; AVX512F-NEXT:    vpaddb %ymm2, %ymm0, %ymm0
513; AVX512F-NEXT:    vinserti64x4 $1, %ymm1, %zmm0, %zmm0
514; AVX512F-NEXT:    retq
515;
516; AVX512BW-LABEL: test_rem7_64i8:
517; AVX512BW:       # %bb.0:
518; AVX512BW-NEXT:    vpxor %xmm1, %xmm1, %xmm1
519; AVX512BW-NEXT:    vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63]
520; AVX512BW-NEXT:    vpbroadcastw {{.*#+}} zmm3 = [37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37]
521; AVX512BW-NEXT:    vpmullw %zmm3, %zmm2, %zmm2
522; AVX512BW-NEXT:    vpsrlw $8, %zmm2, %zmm2
523; AVX512BW-NEXT:    vpunpcklbw {{.*#+}} zmm1 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55]
524; AVX512BW-NEXT:    vpmullw %zmm3, %zmm1, %zmm1
525; AVX512BW-NEXT:    vpsrlw $8, %zmm1, %zmm1
526; AVX512BW-NEXT:    vpackuswb %zmm2, %zmm1, %zmm1
527; AVX512BW-NEXT:    vpsubb %zmm1, %zmm0, %zmm2
528; AVX512BW-NEXT:    vpsrlw $1, %zmm2, %zmm2
529; AVX512BW-NEXT:    vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm2, %zmm2
530; AVX512BW-NEXT:    vpaddb %zmm1, %zmm2, %zmm1
531; AVX512BW-NEXT:    vpsllw $1, %zmm1, %zmm2
532; AVX512BW-NEXT:    vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm2, %zmm2
533; AVX512BW-NEXT:    vpsrlw $2, %zmm1, %zmm1
534; AVX512BW-NEXT:    vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm1, %zmm1
535; AVX512BW-NEXT:    vpsubb %zmm2, %zmm1, %zmm1
536; AVX512BW-NEXT:    vpaddb %zmm1, %zmm0, %zmm0
537; AVX512BW-NEXT:    retq
538  %res = urem <64 x i8> %a, <i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7>
539  ret <64 x i8> %res
540}
541
542;
543; urem by non-splat constant
544;
545
546define <64 x i8> @test_remconstant_64i8(<64 x i8> %a) nounwind {
547; AVX512F-LABEL: test_remconstant_64i8:
548; AVX512F:       # %bb.0:
549; AVX512F-NEXT:    vextracti64x4 $1, %zmm0, %ymm2
550; AVX512F-NEXT:    vpxor %xmm1, %xmm1, %xmm1
551; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm3 = ymm2[8],ymm1[8],ymm2[9],ymm1[9],ymm2[10],ymm1[10],ymm2[11],ymm1[11],ymm2[12],ymm1[12],ymm2[13],ymm1[13],ymm2[14],ymm1[14],ymm2[15],ymm1[15],ymm2[24],ymm1[24],ymm2[25],ymm1[25],ymm2[26],ymm1[26],ymm2[27],ymm1[27],ymm2[28],ymm1[28],ymm2[29],ymm1[29],ymm2[30],ymm1[30],ymm2[31],ymm1[31]
552; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [256,256,64,256,256,256,256,256,128,256,256,256,256,256,256,256]
553; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
554; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [137,27,37,19,79,41,171,101,147,79,171,117,205,57,32,37]
555; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
556; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm4 = ymm2[0],ymm1[0],ymm2[1],ymm1[1],ymm2[2],ymm1[2],ymm2[3],ymm1[3],ymm2[4],ymm1[4],ymm2[5],ymm1[5],ymm2[6],ymm1[6],ymm2[7],ymm1[7],ymm2[16],ymm1[16],ymm2[17],ymm1[17],ymm2[18],ymm1[18],ymm2[19],ymm1[19],ymm2[20],ymm1[20],ymm2[21],ymm1[21],ymm2[22],ymm1[22],ymm2[23],ymm1[23]
557; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [256,256,256,256,256,256,256,256,128,256,256,256,256,256,256,256]
558; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
559; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [27,111,57,235,241,249,8,9,187,135,205,27,57,241,16,137]
560; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
561; AVX512F-NEXT:    vpackuswb %ymm3, %ymm4, %ymm3
562; AVX512F-NEXT:    vpsubb %ymm3, %ymm2, %ymm4
563; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm5 = ymm4[8],ymm1[8],ymm4[9],ymm1[9],ymm4[10],ymm1[10],ymm4[11],ymm1[11],ymm4[12],ymm1[12],ymm4[13],ymm1[13],ymm4[14],ymm1[14],ymm4[15],ymm1[15],ymm4[24],ymm1[24],ymm4[25],ymm1[25],ymm4[26],ymm1[26],ymm4[27],ymm1[27],ymm4[28],ymm1[28],ymm4[29],ymm1[29],ymm4[30],ymm1[30],ymm4[31],ymm1[31]
564; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [0,128,0,0,0,0,0,128,0,0,0,128,0,0,0,128]
565; AVX512F-NEXT:    vpsrlw $8, %ymm5, %ymm5
566; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm4 = ymm4[0],ymm1[0],ymm4[1],ymm1[1],ymm4[2],ymm1[2],ymm4[3],ymm1[3],ymm4[4],ymm1[4],ymm4[5],ymm1[5],ymm4[6],ymm1[6],ymm4[7],ymm1[7],ymm4[16],ymm1[16],ymm4[17],ymm1[17],ymm4[18],ymm1[18],ymm4[19],ymm1[19],ymm4[20],ymm1[20],ymm4[21],ymm1[21],ymm4[22],ymm1[22],ymm4[23],ymm1[23]
567; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [0,0,0,0,0,0,0,128,0,128,0,0,0,0,0,0]
568; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
569; AVX512F-NEXT:    vpackuswb %ymm5, %ymm4, %ymm4
570; AVX512F-NEXT:    vpaddb %ymm3, %ymm4, %ymm3
571; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm4 = ymm3[8],ymm1[8],ymm3[9],ymm1[9],ymm3[10],ymm1[10],ymm3[11],ymm1[11],ymm3[12],ymm1[12],ymm3[13],ymm1[13],ymm3[14],ymm1[14],ymm3[15],ymm1[15],ymm3[24],ymm1[24],ymm3[25],ymm1[25],ymm3[26],ymm1[26],ymm3[27],ymm1[27],ymm3[28],ymm1[28],ymm3[29],ymm1[29],ymm3[30],ymm1[30],ymm3[31],ymm1[31]
572; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [16,16,256,128,32,64,16,16,64,64,32,32,32,128,256,64]
573; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
574; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm3 = ymm3[0],ymm1[0],ymm3[1],ymm1[1],ymm3[2],ymm1[2],ymm3[3],ymm1[3],ymm3[4],ymm1[4],ymm3[5],ymm1[5],ymm3[6],ymm1[6],ymm3[7],ymm1[7],ymm3[16],ymm1[16],ymm3[17],ymm1[17],ymm3[18],ymm1[18],ymm3[19],ymm1[19],ymm3[20],ymm1[20],ymm3[21],ymm1[21],ymm3[22],ymm1[22],ymm3[23],ymm1[23]
575; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [64,16,32,8,8,8,256,16,32,16,16,128,64,16,256,32]
576; AVX512F-NEXT:    vpsrlw $8, %ymm3, %ymm3
577; AVX512F-NEXT:    vpackuswb %ymm4, %ymm3, %ymm4
578; AVX512F-NEXT:    vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm5 # [38,0,36,0,34,0,32,0,30,0,28,0,26,0,24,0,22,0,20,0,18,0,16,0,14,0,12,0,10,0,8,0]
579; AVX512F-NEXT:    vpbroadcastw {{.*#+}} ymm3 = [255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255]
580; AVX512F-NEXT:    vpand %ymm3, %ymm5, %ymm5
581; AVX512F-NEXT:    vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [0,37,0,35,0,33,0,31,0,29,0,27,0,25,0,23,0,21,0,19,0,17,0,15,0,13,0,11,0,9,0,7]
582; AVX512F-NEXT:    vpsllw $8, %ymm4, %ymm4
583; AVX512F-NEXT:    vpor %ymm4, %ymm5, %ymm4
584; AVX512F-NEXT:    vpsubb %ymm4, %ymm2, %ymm2
585; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm4 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31]
586; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [256,256,256,256,256,256,256,128,256,256,256,256,256,256,256,256]
587; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
588; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [137,16,241,57,27,205,135,187,9,8,249,241,235,57,111,27]
589; AVX512F-NEXT:    vpsrlw $8, %ymm4, %ymm4
590; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm5 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23]
591; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [256,256,256,256,256,256,256,128,256,256,256,256,256,64,256,256]
592; AVX512F-NEXT:    vpsrlw $8, %ymm5, %ymm5
593; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [37,32,57,205,117,171,79,147,101,171,41,79,19,37,27,137]
594; AVX512F-NEXT:    vpsrlw $8, %ymm5, %ymm5
595; AVX512F-NEXT:    vpackuswb %ymm4, %ymm5, %ymm4
596; AVX512F-NEXT:    vpsubb %ymm4, %ymm0, %ymm5
597; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm6 = ymm5[8],ymm1[8],ymm5[9],ymm1[9],ymm5[10],ymm1[10],ymm5[11],ymm1[11],ymm5[12],ymm1[12],ymm5[13],ymm1[13],ymm5[14],ymm1[14],ymm5[15],ymm1[15],ymm5[24],ymm1[24],ymm5[25],ymm1[25],ymm5[26],ymm1[26],ymm5[27],ymm1[27],ymm5[28],ymm1[28],ymm5[29],ymm1[29],ymm5[30],ymm1[30],ymm5[31],ymm1[31]
598; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm6, %ymm6 # [0,0,0,0,0,0,128,0,128,0,0,0,0,0,0,0]
599; AVX512F-NEXT:    vpsrlw $8, %ymm6, %ymm6
600; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm5 = ymm5[0],ymm1[0],ymm5[1],ymm1[1],ymm5[2],ymm1[2],ymm5[3],ymm1[3],ymm5[4],ymm1[4],ymm5[5],ymm1[5],ymm5[6],ymm1[6],ymm5[7],ymm1[7],ymm5[16],ymm1[16],ymm5[17],ymm1[17],ymm5[18],ymm1[18],ymm5[19],ymm1[19],ymm5[20],ymm1[20],ymm5[21],ymm1[21],ymm5[22],ymm1[22],ymm5[23],ymm1[23]
601; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [128,0,0,0,128,0,0,0,128,0,0,0,0,0,128,0]
602; AVX512F-NEXT:    vpsrlw $8, %ymm5, %ymm5
603; AVX512F-NEXT:    vpackuswb %ymm6, %ymm5, %ymm5
604; AVX512F-NEXT:    vpaddb %ymm4, %ymm5, %ymm4
605; AVX512F-NEXT:    vpunpckhbw {{.*#+}} ymm5 = ymm4[8],ymm1[8],ymm4[9],ymm1[9],ymm4[10],ymm1[10],ymm4[11],ymm1[11],ymm4[12],ymm1[12],ymm4[13],ymm1[13],ymm4[14],ymm1[14],ymm4[15],ymm1[15],ymm4[24],ymm1[24],ymm4[25],ymm1[25],ymm4[26],ymm1[26],ymm4[27],ymm1[27],ymm4[28],ymm1[28],ymm4[29],ymm1[29],ymm4[30],ymm1[30],ymm4[31],ymm1[31]
606; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [32,256,16,64,128,16,16,32,16,256,8,8,8,32,16,64]
607; AVX512F-NEXT:    vpsrlw $8, %ymm5, %ymm5
608; AVX512F-NEXT:    vpunpcklbw {{.*#+}} ymm1 = ymm4[0],ymm1[0],ymm4[1],ymm1[1],ymm4[2],ymm1[2],ymm4[3],ymm1[3],ymm4[4],ymm1[4],ymm4[5],ymm1[5],ymm4[6],ymm1[6],ymm4[7],ymm1[7],ymm4[16],ymm1[16],ymm4[17],ymm1[17],ymm4[18],ymm1[18],ymm4[19],ymm1[19],ymm4[20],ymm1[20],ymm4[21],ymm1[21],ymm4[22],ymm1[22],ymm4[23],ymm1[23]
609; AVX512F-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm1, %ymm1 # [64,256,128,32,32,32,64,64,16,16,64,32,128,256,16,16]
610; AVX512F-NEXT:    vpsrlw $8, %ymm1, %ymm1
611; AVX512F-NEXT:    vpackuswb %ymm5, %ymm1, %ymm1
612; AVX512F-NEXT:    vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm1, %ymm4 # [7,0,9,0,11,0,13,0,15,0,17,0,19,0,21,0,23,0,25,0,27,0,29,0,31,0,33,0,35,0,37,0]
613; AVX512F-NEXT:    vpand %ymm3, %ymm4, %ymm3
614; AVX512F-NEXT:    vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm1, %ymm1 # [0,8,0,10,0,12,0,14,0,16,0,18,0,20,0,22,0,24,0,26,0,28,0,30,0,32,0,34,0,36,0,38]
615; AVX512F-NEXT:    vpsllw $8, %ymm1, %ymm1
616; AVX512F-NEXT:    vpor %ymm1, %ymm3, %ymm1
617; AVX512F-NEXT:    vpsubb %ymm1, %ymm0, %ymm0
618; AVX512F-NEXT:    vinserti64x4 $1, %ymm2, %zmm0, %zmm0
619; AVX512F-NEXT:    retq
620;
621; AVX512BW-LABEL: test_remconstant_64i8:
622; AVX512BW:       # %bb.0:
623; AVX512BW-NEXT:    vpxor %xmm1, %xmm1, %xmm1
624; AVX512BW-NEXT:    vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63]
625; AVX512BW-NEXT:    vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2
626; AVX512BW-NEXT:    vpsrlw $8, %zmm2, %zmm2
627; AVX512BW-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2 # [137,16,241,57,27,205,135,187,9,8,249,241,235,57,111,27,137,27,37,19,79,41,171,101,147,79,171,117,205,57,32,37]
628; AVX512BW-NEXT:    vpsrlw $8, %zmm2, %zmm2
629; AVX512BW-NEXT:    vpunpcklbw {{.*#+}} zmm3 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55]
630; AVX512BW-NEXT:    vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3
631; AVX512BW-NEXT:    vpsrlw $8, %zmm3, %zmm3
632; AVX512BW-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 # [37,32,57,205,117,171,79,147,101,171,41,79,19,37,27,137,27,111,57,235,241,249,8,9,187,135,205,27,57,241,16,137]
633; AVX512BW-NEXT:    vpsrlw $8, %zmm3, %zmm3
634; AVX512BW-NEXT:    vpackuswb %zmm2, %zmm3, %zmm2
635; AVX512BW-NEXT:    vpsubb %zmm2, %zmm0, %zmm3
636; AVX512BW-NEXT:    vpunpckhbw {{.*#+}} zmm4 = zmm3[8],zmm1[8],zmm3[9],zmm1[9],zmm3[10],zmm1[10],zmm3[11],zmm1[11],zmm3[12],zmm1[12],zmm3[13],zmm1[13],zmm3[14],zmm1[14],zmm3[15],zmm1[15],zmm3[24],zmm1[24],zmm3[25],zmm1[25],zmm3[26],zmm1[26],zmm3[27],zmm1[27],zmm3[28],zmm1[28],zmm3[29],zmm1[29],zmm3[30],zmm1[30],zmm3[31],zmm1[31],zmm3[40],zmm1[40],zmm3[41],zmm1[41],zmm3[42],zmm1[42],zmm3[43],zmm1[43],zmm3[44],zmm1[44],zmm3[45],zmm1[45],zmm3[46],zmm1[46],zmm3[47],zmm1[47],zmm3[56],zmm1[56],zmm3[57],zmm1[57],zmm3[58],zmm1[58],zmm3[59],zmm1[59],zmm3[60],zmm1[60],zmm3[61],zmm1[61],zmm3[62],zmm1[62],zmm3[63],zmm1[63]
637; AVX512BW-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm4, %zmm4 # [0,0,0,0,0,0,128,0,128,0,0,0,0,0,0,0,0,128,0,0,0,0,0,128,0,0,0,128,0,0,0,128]
638; AVX512BW-NEXT:    vpsrlw $8, %zmm4, %zmm4
639; AVX512BW-NEXT:    vpunpcklbw {{.*#+}} zmm3 = zmm3[0],zmm1[0],zmm3[1],zmm1[1],zmm3[2],zmm1[2],zmm3[3],zmm1[3],zmm3[4],zmm1[4],zmm3[5],zmm1[5],zmm3[6],zmm1[6],zmm3[7],zmm1[7],zmm3[16],zmm1[16],zmm3[17],zmm1[17],zmm3[18],zmm1[18],zmm3[19],zmm1[19],zmm3[20],zmm1[20],zmm3[21],zmm1[21],zmm3[22],zmm1[22],zmm3[23],zmm1[23],zmm3[32],zmm1[32],zmm3[33],zmm1[33],zmm3[34],zmm1[34],zmm3[35],zmm1[35],zmm3[36],zmm1[36],zmm3[37],zmm1[37],zmm3[38],zmm1[38],zmm3[39],zmm1[39],zmm3[48],zmm1[48],zmm3[49],zmm1[49],zmm3[50],zmm1[50],zmm3[51],zmm1[51],zmm3[52],zmm1[52],zmm3[53],zmm1[53],zmm3[54],zmm1[54],zmm3[55],zmm1[55]
640; AVX512BW-NEXT:    vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 # [128,0,0,0,128,0,0,0,128,0,0,0,0,0,128,0,0,0,0,0,0,0,0,128,0,128,0,0,0,0,0,0]
641; AVX512BW-NEXT:    vpsrlw $8, %zmm3, %zmm3
642; AVX512BW-NEXT:    vpackuswb %zmm4, %zmm3, %zmm3
643; AVX512BW-NEXT:    vpaddb %zmm2, %zmm3, %zmm2
644; AVX512BW-NEXT:    vpunpckhbw {{.*#+}} zmm3 = zmm2[8],zmm1[8],zmm2[9],zmm1[9],zmm2[10],zmm1[10],zmm2[11],zmm1[11],zmm2[12],zmm1[12],zmm2[13],zmm1[13],zmm2[14],zmm1[14],zmm2[15],zmm1[15],zmm2[24],zmm1[24],zmm2[25],zmm1[25],zmm2[26],zmm1[26],zmm2[27],zmm1[27],zmm2[28],zmm1[28],zmm2[29],zmm1[29],zmm2[30],zmm1[30],zmm2[31],zmm1[31],zmm2[40],zmm1[40],zmm2[41],zmm1[41],zmm2[42],zmm1[42],zmm2[43],zmm1[43],zmm2[44],zmm1[44],zmm2[45],zmm1[45],zmm2[46],zmm1[46],zmm2[47],zmm1[47],zmm2[56],zmm1[56],zmm2[57],zmm1[57],zmm2[58],zmm1[58],zmm2[59],zmm1[59],zmm2[60],zmm1[60],zmm2[61],zmm1[61],zmm2[62],zmm1[62],zmm2[63],zmm1[63]
645; AVX512BW-NEXT:    vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3
646; AVX512BW-NEXT:    vpsrlw $8, %zmm3, %zmm3
647; AVX512BW-NEXT:    vpunpcklbw {{.*#+}} zmm1 = zmm2[0],zmm1[0],zmm2[1],zmm1[1],zmm2[2],zmm1[2],zmm2[3],zmm1[3],zmm2[4],zmm1[4],zmm2[5],zmm1[5],zmm2[6],zmm1[6],zmm2[7],zmm1[7],zmm2[16],zmm1[16],zmm2[17],zmm1[17],zmm2[18],zmm1[18],zmm2[19],zmm1[19],zmm2[20],zmm1[20],zmm2[21],zmm1[21],zmm2[22],zmm1[22],zmm2[23],zmm1[23],zmm2[32],zmm1[32],zmm2[33],zmm1[33],zmm2[34],zmm1[34],zmm2[35],zmm1[35],zmm2[36],zmm1[36],zmm2[37],zmm1[37],zmm2[38],zmm1[38],zmm2[39],zmm1[39],zmm2[48],zmm1[48],zmm2[49],zmm1[49],zmm2[50],zmm1[50],zmm2[51],zmm1[51],zmm2[52],zmm1[52],zmm2[53],zmm1[53],zmm2[54],zmm1[54],zmm2[55],zmm1[55]
648; AVX512BW-NEXT:    vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm1
649; AVX512BW-NEXT:    vpsrlw $8, %zmm1, %zmm1
650; AVX512BW-NEXT:    vpackuswb %zmm3, %zmm1, %zmm1
651; AVX512BW-NEXT:    vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm2 # [7,0,9,0,11,0,13,0,15,0,17,0,19,0,21,0,23,0,25,0,27,0,29,0,31,0,33,0,35,0,37,0,38,0,36,0,34,0,32,0,30,0,28,0,26,0,24,0,22,0,20,0,18,0,16,0,14,0,12,0,10,0,8,0]
652; AVX512BW-NEXT:    vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm1 # [0,8,0,10,0,12,0,14,0,16,0,18,0,20,0,22,0,24,0,26,0,28,0,30,0,32,0,34,0,36,0,38,0,37,0,35,0,33,0,31,0,29,0,27,0,25,0,23,0,21,0,19,0,17,0,15,0,13,0,11,0,9,0,7]
653; AVX512BW-NEXT:    vpsllw $8, %zmm1, %zmm1
654; AVX512BW-NEXT:    vpternlogd {{.*#+}} zmm1 = zmm1 | (zmm2 & mem)
655; AVX512BW-NEXT:    vpsubb %zmm1, %zmm0, %zmm0
656; AVX512BW-NEXT:    retq
657  %res = urem <64 x i8> %a, <i8 7, i8 8, i8 9, i8 10, i8 11, i8 12, i8 13, i8 14, i8 15, i8 16, i8 17, i8 18, i8 19, i8 20, i8 21, i8 22, i8 23, i8 24, i8 25, i8 26, i8 27, i8 28, i8 29, i8 30, i8 31, i8 32, i8 33, i8 34, i8 35, i8 36, i8 37, i8 38, i8 38, i8 37, i8 36, i8 35, i8 34, i8 33, i8 32, i8 31, i8 30, i8 29, i8 28, i8 27, i8 26, i8 25, i8 24, i8 23, i8 22, i8 21, i8 20, i8 19, i8 18, i8 17, i8 16, i8 15, i8 14, i8 13, i8 12, i8 11, i8 10, i8 9, i8 8, i8 7>
658  ret <64 x i8> %res
659}
660