1; NOTE: Assertions have been autogenerated by utils/update_llc_test_checks.py 2; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx512f | FileCheck %s --check-prefix=AVX --check-prefix=AVX512F 3; RUN: llc < %s -mtriple=x86_64-unknown-unknown -mattr=+avx512bw | FileCheck %s --check-prefix=AVX --check-prefix=AVX512BW 4 5; 6; udiv by 7 7; 8 9define <8 x i64> @test_div7_8i64(<8 x i64> %a) nounwind { 10; AVX-LABEL: test_div7_8i64: 11; AVX: # %bb.0: 12; AVX-NEXT: vextracti32x4 $3, %zmm0, %xmm1 13; AVX-NEXT: vpextrq $1, %xmm1, %rcx 14; AVX-NEXT: movabsq $2635249153387078803, %rsi # imm = 0x2492492492492493 15; AVX-NEXT: movq %rcx, %rax 16; AVX-NEXT: mulq %rsi 17; AVX-NEXT: subq %rdx, %rcx 18; AVX-NEXT: shrq %rcx 19; AVX-NEXT: addq %rdx, %rcx 20; AVX-NEXT: vmovq %rcx, %xmm2 21; AVX-NEXT: vmovq %xmm1, %rcx 22; AVX-NEXT: movq %rcx, %rax 23; AVX-NEXT: mulq %rsi 24; AVX-NEXT: subq %rdx, %rcx 25; AVX-NEXT: shrq %rcx 26; AVX-NEXT: addq %rdx, %rcx 27; AVX-NEXT: vmovq %rcx, %xmm1 28; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm1 = xmm1[0],xmm2[0] 29; AVX-NEXT: vextracti32x4 $2, %zmm0, %xmm2 30; AVX-NEXT: vpextrq $1, %xmm2, %rcx 31; AVX-NEXT: movq %rcx, %rax 32; AVX-NEXT: mulq %rsi 33; AVX-NEXT: subq %rdx, %rcx 34; AVX-NEXT: shrq %rcx 35; AVX-NEXT: addq %rdx, %rcx 36; AVX-NEXT: vmovq %rcx, %xmm3 37; AVX-NEXT: vmovq %xmm2, %rcx 38; AVX-NEXT: movq %rcx, %rax 39; AVX-NEXT: mulq %rsi 40; AVX-NEXT: subq %rdx, %rcx 41; AVX-NEXT: shrq %rcx 42; AVX-NEXT: addq %rdx, %rcx 43; AVX-NEXT: vmovq %rcx, %xmm2 44; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm2 = xmm2[0],xmm3[0] 45; AVX-NEXT: vinserti128 $1, %xmm1, %ymm2, %ymm1 46; AVX-NEXT: vextracti128 $1, %ymm0, %xmm2 47; AVX-NEXT: vpextrq $1, %xmm2, %rcx 48; AVX-NEXT: movq %rcx, %rax 49; AVX-NEXT: mulq %rsi 50; AVX-NEXT: subq %rdx, %rcx 51; AVX-NEXT: shrq %rcx 52; AVX-NEXT: addq %rdx, %rcx 53; AVX-NEXT: vmovq %rcx, %xmm3 54; AVX-NEXT: vmovq %xmm2, %rcx 55; AVX-NEXT: movq %rcx, %rax 56; AVX-NEXT: mulq %rsi 57; AVX-NEXT: subq %rdx, %rcx 58; AVX-NEXT: shrq %rcx 59; AVX-NEXT: addq %rdx, %rcx 60; AVX-NEXT: vmovq %rcx, %xmm2 61; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm2 = xmm2[0],xmm3[0] 62; AVX-NEXT: vpextrq $1, %xmm0, %rcx 63; AVX-NEXT: movq %rcx, %rax 64; AVX-NEXT: mulq %rsi 65; AVX-NEXT: subq %rdx, %rcx 66; AVX-NEXT: shrq %rcx 67; AVX-NEXT: addq %rdx, %rcx 68; AVX-NEXT: vmovq %rcx, %xmm3 69; AVX-NEXT: vmovq %xmm0, %rcx 70; AVX-NEXT: movq %rcx, %rax 71; AVX-NEXT: mulq %rsi 72; AVX-NEXT: subq %rdx, %rcx 73; AVX-NEXT: shrq %rcx 74; AVX-NEXT: addq %rdx, %rcx 75; AVX-NEXT: vmovq %rcx, %xmm0 76; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm0 = xmm0[0],xmm3[0] 77; AVX-NEXT: vinserti128 $1, %xmm2, %ymm0, %ymm0 78; AVX-NEXT: vinserti64x4 $1, %ymm1, %zmm0, %zmm0 79; AVX-NEXT: vpsrlq $2, %zmm0, %zmm0 80; AVX-NEXT: retq 81 %res = udiv <8 x i64> %a, <i64 7, i64 7, i64 7, i64 7, i64 7, i64 7, i64 7, i64 7> 82 ret <8 x i64> %res 83} 84 85define <16 x i32> @test_div7_16i32(<16 x i32> %a) nounwind { 86; AVX-LABEL: test_div7_16i32: 87; AVX: # %bb.0: 88; AVX-NEXT: vpbroadcastd {{.*#+}} zmm1 = [613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757] 89; AVX-NEXT: vpmuludq %zmm1, %zmm0, %zmm2 90; AVX-NEXT: vpshufd {{.*#+}} zmm3 = zmm0[1,1,3,3,5,5,7,7,9,9,11,11,13,13,15,15] 91; AVX-NEXT: vpmuludq %zmm1, %zmm3, %zmm1 92; AVX-NEXT: vpmovsxbd {{.*#+}} zmm3 = [1,17,3,19,5,21,7,23,9,25,11,27,13,29,15,31] 93; AVX-NEXT: vpermi2d %zmm1, %zmm2, %zmm3 94; AVX-NEXT: vpsubd %zmm3, %zmm0, %zmm0 95; AVX-NEXT: vpsrld $1, %zmm0, %zmm0 96; AVX-NEXT: vpaddd %zmm3, %zmm0, %zmm0 97; AVX-NEXT: vpsrld $2, %zmm0, %zmm0 98; AVX-NEXT: retq 99 %res = udiv <16 x i32> %a, <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7> 100 ret <16 x i32> %res 101} 102 103define <32 x i16> @test_div7_32i16(<32 x i16> %a) nounwind { 104; AVX512F-LABEL: test_div7_32i16: 105; AVX512F: # %bb.0: 106; AVX512F-NEXT: vpbroadcastw {{.*#+}} ymm1 = [9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363] 107; AVX512F-NEXT: vpmulhuw %ymm1, %ymm0, %ymm2 108; AVX512F-NEXT: vpsubw %ymm2, %ymm0, %ymm3 109; AVX512F-NEXT: vpsrlw $1, %ymm3, %ymm3 110; AVX512F-NEXT: vpaddw %ymm2, %ymm3, %ymm2 111; AVX512F-NEXT: vpsrlw $2, %ymm2, %ymm2 112; AVX512F-NEXT: vextracti64x4 $1, %zmm0, %ymm0 113; AVX512F-NEXT: vpmulhuw %ymm1, %ymm0, %ymm1 114; AVX512F-NEXT: vpsubw %ymm1, %ymm0, %ymm0 115; AVX512F-NEXT: vpsrlw $1, %ymm0, %ymm0 116; AVX512F-NEXT: vpaddw %ymm1, %ymm0, %ymm0 117; AVX512F-NEXT: vpsrlw $2, %ymm0, %ymm0 118; AVX512F-NEXT: vinserti64x4 $1, %ymm0, %zmm2, %zmm0 119; AVX512F-NEXT: retq 120; 121; AVX512BW-LABEL: test_div7_32i16: 122; AVX512BW: # %bb.0: 123; AVX512BW-NEXT: vpmulhuw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm0, %zmm1 # [9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363] 124; AVX512BW-NEXT: vpsubw %zmm1, %zmm0, %zmm0 125; AVX512BW-NEXT: vpsrlw $1, %zmm0, %zmm0 126; AVX512BW-NEXT: vpaddw %zmm1, %zmm0, %zmm0 127; AVX512BW-NEXT: vpsrlw $2, %zmm0, %zmm0 128; AVX512BW-NEXT: retq 129 %res = udiv <32 x i16> %a, <i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7> 130 ret <32 x i16> %res 131} 132 133define <64 x i8> @test_div7_64i8(<64 x i8> %a) nounwind { 134; AVX512F-LABEL: test_div7_64i8: 135; AVX512F: # %bb.0: 136; AVX512F-NEXT: vpxor %xmm1, %xmm1, %xmm1 137; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm2 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31] 138; AVX512F-NEXT: vpbroadcastw {{.*#+}} ymm3 = [37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37] 139; AVX512F-NEXT: vpmullw %ymm3, %ymm2, %ymm2 140; AVX512F-NEXT: vpsrlw $8, %ymm2, %ymm2 141; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm4 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23] 142; AVX512F-NEXT: vpmullw %ymm3, %ymm4, %ymm4 143; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 144; AVX512F-NEXT: vpackuswb %ymm2, %ymm4, %ymm2 145; AVX512F-NEXT: vpsubb %ymm2, %ymm0, %ymm4 146; AVX512F-NEXT: vpsrlw $1, %ymm4, %ymm4 147; AVX512F-NEXT: vpbroadcastb {{.*#+}} ymm5 = [127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127] 148; AVX512F-NEXT: vpand %ymm5, %ymm4, %ymm4 149; AVX512F-NEXT: vpaddb %ymm2, %ymm4, %ymm2 150; AVX512F-NEXT: vpsrlw $2, %ymm2, %ymm2 151; AVX512F-NEXT: vextracti64x4 $1, %zmm0, %ymm0 152; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm4 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31] 153; AVX512F-NEXT: vpmullw %ymm3, %ymm4, %ymm4 154; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 155; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm1 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23] 156; AVX512F-NEXT: vpmullw %ymm3, %ymm1, %ymm1 157; AVX512F-NEXT: vpsrlw $8, %ymm1, %ymm1 158; AVX512F-NEXT: vpackuswb %ymm4, %ymm1, %ymm1 159; AVX512F-NEXT: vpsubb %ymm1, %ymm0, %ymm0 160; AVX512F-NEXT: vpsrlw $1, %ymm0, %ymm0 161; AVX512F-NEXT: vpand %ymm5, %ymm0, %ymm0 162; AVX512F-NEXT: vpaddb %ymm1, %ymm0, %ymm0 163; AVX512F-NEXT: vpsrlw $2, %ymm0, %ymm0 164; AVX512F-NEXT: vinserti64x4 $1, %ymm0, %zmm2, %zmm0 165; AVX512F-NEXT: vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm0, %zmm0 166; AVX512F-NEXT: retq 167; 168; AVX512BW-LABEL: test_div7_64i8: 169; AVX512BW: # %bb.0: 170; AVX512BW-NEXT: vpxor %xmm1, %xmm1, %xmm1 171; AVX512BW-NEXT: vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63] 172; AVX512BW-NEXT: vpbroadcastw {{.*#+}} zmm3 = [37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37] 173; AVX512BW-NEXT: vpmullw %zmm3, %zmm2, %zmm2 174; AVX512BW-NEXT: vpsrlw $8, %zmm2, %zmm2 175; AVX512BW-NEXT: vpunpcklbw {{.*#+}} zmm1 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55] 176; AVX512BW-NEXT: vpmullw %zmm3, %zmm1, %zmm1 177; AVX512BW-NEXT: vpsrlw $8, %zmm1, %zmm1 178; AVX512BW-NEXT: vpackuswb %zmm2, %zmm1, %zmm1 179; AVX512BW-NEXT: vpsubb %zmm1, %zmm0, %zmm0 180; AVX512BW-NEXT: vpsrlw $1, %zmm0, %zmm0 181; AVX512BW-NEXT: vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm0, %zmm0 182; AVX512BW-NEXT: vpaddb %zmm1, %zmm0, %zmm0 183; AVX512BW-NEXT: vpsrlw $2, %zmm0, %zmm0 184; AVX512BW-NEXT: vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm0, %zmm0 185; AVX512BW-NEXT: retq 186 %res = udiv <64 x i8> %a, <i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7> 187 ret <64 x i8> %res 188} 189 190; 191; udiv by non-splat constant 192; 193 194define <64 x i8> @test_divconstant_64i8(<64 x i8> %a) nounwind { 195; AVX512F-LABEL: test_divconstant_64i8: 196; AVX512F: # %bb.0: 197; AVX512F-NEXT: vextracti64x4 $1, %zmm0, %ymm2 198; AVX512F-NEXT: vpxor %xmm1, %xmm1, %xmm1 199; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm3 = ymm2[8],ymm1[8],ymm2[9],ymm1[9],ymm2[10],ymm1[10],ymm2[11],ymm1[11],ymm2[12],ymm1[12],ymm2[13],ymm1[13],ymm2[14],ymm1[14],ymm2[15],ymm1[15],ymm2[24],ymm1[24],ymm2[25],ymm1[25],ymm2[26],ymm1[26],ymm2[27],ymm1[27],ymm2[28],ymm1[28],ymm2[29],ymm1[29],ymm2[30],ymm1[30],ymm2[31],ymm1[31] 200; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [256,256,64,256,256,256,256,256,128,256,256,256,256,256,256,256] 201; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 202; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [137,27,37,19,79,41,171,101,147,79,171,117,205,57,32,37] 203; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 204; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm4 = ymm2[0],ymm1[0],ymm2[1],ymm1[1],ymm2[2],ymm1[2],ymm2[3],ymm1[3],ymm2[4],ymm1[4],ymm2[5],ymm1[5],ymm2[6],ymm1[6],ymm2[7],ymm1[7],ymm2[16],ymm1[16],ymm2[17],ymm1[17],ymm2[18],ymm1[18],ymm2[19],ymm1[19],ymm2[20],ymm1[20],ymm2[21],ymm1[21],ymm2[22],ymm1[22],ymm2[23],ymm1[23] 205; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [256,256,256,256,256,256,256,256,128,256,256,256,256,256,256,256] 206; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 207; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [27,111,57,235,241,249,8,9,187,135,205,27,57,241,16,137] 208; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 209; AVX512F-NEXT: vpackuswb %ymm3, %ymm4, %ymm3 210; AVX512F-NEXT: vpsubb %ymm3, %ymm2, %ymm2 211; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm4 = ymm2[8],ymm1[8],ymm2[9],ymm1[9],ymm2[10],ymm1[10],ymm2[11],ymm1[11],ymm2[12],ymm1[12],ymm2[13],ymm1[13],ymm2[14],ymm1[14],ymm2[15],ymm1[15],ymm2[24],ymm1[24],ymm2[25],ymm1[25],ymm2[26],ymm1[26],ymm2[27],ymm1[27],ymm2[28],ymm1[28],ymm2[29],ymm1[29],ymm2[30],ymm1[30],ymm2[31],ymm1[31] 212; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [0,128,0,0,0,0,0,128,0,0,0,128,0,0,0,128] 213; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 214; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm2 = ymm2[0],ymm1[0],ymm2[1],ymm1[1],ymm2[2],ymm1[2],ymm2[3],ymm1[3],ymm2[4],ymm1[4],ymm2[5],ymm1[5],ymm2[6],ymm1[6],ymm2[7],ymm1[7],ymm2[16],ymm1[16],ymm2[17],ymm1[17],ymm2[18],ymm1[18],ymm2[19],ymm1[19],ymm2[20],ymm1[20],ymm2[21],ymm1[21],ymm2[22],ymm1[22],ymm2[23],ymm1[23] 215; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm2, %ymm2 # [0,0,0,0,0,0,0,128,0,128,0,0,0,0,0,0] 216; AVX512F-NEXT: vpsrlw $8, %ymm2, %ymm2 217; AVX512F-NEXT: vpackuswb %ymm4, %ymm2, %ymm2 218; AVX512F-NEXT: vpaddb %ymm3, %ymm2, %ymm2 219; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm3 = ymm2[8],ymm1[8],ymm2[9],ymm1[9],ymm2[10],ymm1[10],ymm2[11],ymm1[11],ymm2[12],ymm1[12],ymm2[13],ymm1[13],ymm2[14],ymm1[14],ymm2[15],ymm1[15],ymm2[24],ymm1[24],ymm2[25],ymm1[25],ymm2[26],ymm1[26],ymm2[27],ymm1[27],ymm2[28],ymm1[28],ymm2[29],ymm1[29],ymm2[30],ymm1[30],ymm2[31],ymm1[31] 220; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [16,16,256,128,32,64,16,16,64,64,32,32,32,128,256,64] 221; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 222; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm2 = ymm2[0],ymm1[0],ymm2[1],ymm1[1],ymm2[2],ymm1[2],ymm2[3],ymm1[3],ymm2[4],ymm1[4],ymm2[5],ymm1[5],ymm2[6],ymm1[6],ymm2[7],ymm1[7],ymm2[16],ymm1[16],ymm2[17],ymm1[17],ymm2[18],ymm1[18],ymm2[19],ymm1[19],ymm2[20],ymm1[20],ymm2[21],ymm1[21],ymm2[22],ymm1[22],ymm2[23],ymm1[23] 223; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm2, %ymm2 # [64,16,32,8,8,8,256,16,32,16,16,128,64,16,256,32] 224; AVX512F-NEXT: vpsrlw $8, %ymm2, %ymm2 225; AVX512F-NEXT: vpackuswb %ymm3, %ymm2, %ymm2 226; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm3 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31] 227; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [256,256,256,256,256,256,256,128,256,256,256,256,256,256,256,256] 228; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 229; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [137,16,241,57,27,205,135,187,9,8,249,241,235,57,111,27] 230; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 231; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm4 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23] 232; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [256,256,256,256,256,256,256,128,256,256,256,256,256,64,256,256] 233; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 234; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [37,32,57,205,117,171,79,147,101,171,41,79,19,37,27,137] 235; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 236; AVX512F-NEXT: vpackuswb %ymm3, %ymm4, %ymm3 237; AVX512F-NEXT: vpsubb %ymm3, %ymm0, %ymm0 238; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm4 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31] 239; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [0,0,0,0,0,0,128,0,128,0,0,0,0,0,0,0] 240; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 241; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm0 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23] 242; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm0, %ymm0 # [128,0,0,0,128,0,0,0,128,0,0,0,0,0,128,0] 243; AVX512F-NEXT: vpsrlw $8, %ymm0, %ymm0 244; AVX512F-NEXT: vpackuswb %ymm4, %ymm0, %ymm0 245; AVX512F-NEXT: vpaddb %ymm3, %ymm0, %ymm0 246; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm3 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31] 247; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [32,256,16,64,128,16,16,32,16,256,8,8,8,32,16,64] 248; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 249; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm0 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23] 250; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm0, %ymm0 # [64,256,128,32,32,32,64,64,16,16,64,32,128,256,16,16] 251; AVX512F-NEXT: vpsrlw $8, %ymm0, %ymm0 252; AVX512F-NEXT: vpackuswb %ymm3, %ymm0, %ymm0 253; AVX512F-NEXT: vinserti64x4 $1, %ymm2, %zmm0, %zmm0 254; AVX512F-NEXT: retq 255; 256; AVX512BW-LABEL: test_divconstant_64i8: 257; AVX512BW: # %bb.0: 258; AVX512BW-NEXT: vpxor %xmm1, %xmm1, %xmm1 259; AVX512BW-NEXT: vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63] 260; AVX512BW-NEXT: vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2 261; AVX512BW-NEXT: vpsrlw $8, %zmm2, %zmm2 262; AVX512BW-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2 # [137,16,241,57,27,205,135,187,9,8,249,241,235,57,111,27,137,27,37,19,79,41,171,101,147,79,171,117,205,57,32,37] 263; AVX512BW-NEXT: vpsrlw $8, %zmm2, %zmm2 264; AVX512BW-NEXT: vpunpcklbw {{.*#+}} zmm3 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55] 265; AVX512BW-NEXT: vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 266; AVX512BW-NEXT: vpsrlw $8, %zmm3, %zmm3 267; AVX512BW-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 # [37,32,57,205,117,171,79,147,101,171,41,79,19,37,27,137,27,111,57,235,241,249,8,9,187,135,205,27,57,241,16,137] 268; AVX512BW-NEXT: vpsrlw $8, %zmm3, %zmm3 269; AVX512BW-NEXT: vpackuswb %zmm2, %zmm3, %zmm2 270; AVX512BW-NEXT: vpsubb %zmm2, %zmm0, %zmm0 271; AVX512BW-NEXT: vpunpckhbw {{.*#+}} zmm3 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63] 272; AVX512BW-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 # [0,0,0,0,0,0,128,0,128,0,0,0,0,0,0,0,0,128,0,0,0,0,0,128,0,0,0,128,0,0,0,128] 273; AVX512BW-NEXT: vpsrlw $8, %zmm3, %zmm3 274; AVX512BW-NEXT: vpunpcklbw {{.*#+}} zmm0 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55] 275; AVX512BW-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm0, %zmm0 # [128,0,0,0,128,0,0,0,128,0,0,0,0,0,128,0,0,0,0,0,0,0,0,128,0,128,0,0,0,0,0,0] 276; AVX512BW-NEXT: vpsrlw $8, %zmm0, %zmm0 277; AVX512BW-NEXT: vpackuswb %zmm3, %zmm0, %zmm0 278; AVX512BW-NEXT: vpaddb %zmm2, %zmm0, %zmm0 279; AVX512BW-NEXT: vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63] 280; AVX512BW-NEXT: vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2 281; AVX512BW-NEXT: vpsrlw $8, %zmm2, %zmm2 282; AVX512BW-NEXT: vpunpcklbw {{.*#+}} zmm0 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55] 283; AVX512BW-NEXT: vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm0, %zmm0 284; AVX512BW-NEXT: vpsrlw $8, %zmm0, %zmm0 285; AVX512BW-NEXT: vpackuswb %zmm2, %zmm0, %zmm0 286; AVX512BW-NEXT: retq 287 %res = udiv <64 x i8> %a, <i8 7, i8 8, i8 9, i8 10, i8 11, i8 12, i8 13, i8 14, i8 15, i8 16, i8 17, i8 18, i8 19, i8 20, i8 21, i8 22, i8 23, i8 24, i8 25, i8 26, i8 27, i8 28, i8 29, i8 30, i8 31, i8 32, i8 33, i8 34, i8 35, i8 36, i8 37, i8 38, i8 38, i8 37, i8 36, i8 35, i8 34, i8 33, i8 32, i8 31, i8 30, i8 29, i8 28, i8 27, i8 26, i8 25, i8 24, i8 23, i8 22, i8 21, i8 20, i8 19, i8 18, i8 17, i8 16, i8 15, i8 14, i8 13, i8 12, i8 11, i8 10, i8 9, i8 8, i8 7> 288 ret <64 x i8> %res 289} 290 291; 292; urem by 7 293; 294 295define <8 x i64> @test_rem7_8i64(<8 x i64> %a) nounwind { 296; AVX-LABEL: test_rem7_8i64: 297; AVX: # %bb.0: 298; AVX-NEXT: vextracti32x4 $3, %zmm0, %xmm1 299; AVX-NEXT: vpextrq $1, %xmm1, %rcx 300; AVX-NEXT: movabsq $2635249153387078803, %rsi # imm = 0x2492492492492493 301; AVX-NEXT: movq %rcx, %rax 302; AVX-NEXT: mulq %rsi 303; AVX-NEXT: movq %rcx, %rax 304; AVX-NEXT: subq %rdx, %rax 305; AVX-NEXT: shrq %rax 306; AVX-NEXT: addq %rdx, %rax 307; AVX-NEXT: shrq $2, %rax 308; AVX-NEXT: leaq (,%rax,8), %rdx 309; AVX-NEXT: subq %rdx, %rax 310; AVX-NEXT: addq %rcx, %rax 311; AVX-NEXT: vmovq %rax, %xmm2 312; AVX-NEXT: vmovq %xmm1, %rcx 313; AVX-NEXT: movq %rcx, %rax 314; AVX-NEXT: mulq %rsi 315; AVX-NEXT: movq %rcx, %rax 316; AVX-NEXT: subq %rdx, %rax 317; AVX-NEXT: shrq %rax 318; AVX-NEXT: addq %rdx, %rax 319; AVX-NEXT: shrq $2, %rax 320; AVX-NEXT: leaq (,%rax,8), %rdx 321; AVX-NEXT: subq %rdx, %rax 322; AVX-NEXT: addq %rcx, %rax 323; AVX-NEXT: vmovq %rax, %xmm1 324; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm1 = xmm1[0],xmm2[0] 325; AVX-NEXT: vextracti32x4 $2, %zmm0, %xmm2 326; AVX-NEXT: vpextrq $1, %xmm2, %rcx 327; AVX-NEXT: movq %rcx, %rax 328; AVX-NEXT: mulq %rsi 329; AVX-NEXT: movq %rcx, %rax 330; AVX-NEXT: subq %rdx, %rax 331; AVX-NEXT: shrq %rax 332; AVX-NEXT: addq %rdx, %rax 333; AVX-NEXT: shrq $2, %rax 334; AVX-NEXT: leaq (,%rax,8), %rdx 335; AVX-NEXT: subq %rdx, %rax 336; AVX-NEXT: addq %rcx, %rax 337; AVX-NEXT: vmovq %rax, %xmm3 338; AVX-NEXT: vmovq %xmm2, %rcx 339; AVX-NEXT: movq %rcx, %rax 340; AVX-NEXT: mulq %rsi 341; AVX-NEXT: movq %rcx, %rax 342; AVX-NEXT: subq %rdx, %rax 343; AVX-NEXT: shrq %rax 344; AVX-NEXT: addq %rdx, %rax 345; AVX-NEXT: shrq $2, %rax 346; AVX-NEXT: leaq (,%rax,8), %rdx 347; AVX-NEXT: subq %rdx, %rax 348; AVX-NEXT: addq %rcx, %rax 349; AVX-NEXT: vmovq %rax, %xmm2 350; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm2 = xmm2[0],xmm3[0] 351; AVX-NEXT: vinserti128 $1, %xmm1, %ymm2, %ymm1 352; AVX-NEXT: vextracti128 $1, %ymm0, %xmm2 353; AVX-NEXT: vpextrq $1, %xmm2, %rcx 354; AVX-NEXT: movq %rcx, %rax 355; AVX-NEXT: mulq %rsi 356; AVX-NEXT: movq %rcx, %rax 357; AVX-NEXT: subq %rdx, %rax 358; AVX-NEXT: shrq %rax 359; AVX-NEXT: addq %rdx, %rax 360; AVX-NEXT: shrq $2, %rax 361; AVX-NEXT: leaq (,%rax,8), %rdx 362; AVX-NEXT: subq %rdx, %rax 363; AVX-NEXT: addq %rcx, %rax 364; AVX-NEXT: vmovq %rax, %xmm3 365; AVX-NEXT: vmovq %xmm2, %rcx 366; AVX-NEXT: movq %rcx, %rax 367; AVX-NEXT: mulq %rsi 368; AVX-NEXT: movq %rcx, %rax 369; AVX-NEXT: subq %rdx, %rax 370; AVX-NEXT: shrq %rax 371; AVX-NEXT: addq %rdx, %rax 372; AVX-NEXT: shrq $2, %rax 373; AVX-NEXT: leaq (,%rax,8), %rdx 374; AVX-NEXT: subq %rdx, %rax 375; AVX-NEXT: addq %rcx, %rax 376; AVX-NEXT: vmovq %rax, %xmm2 377; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm2 = xmm2[0],xmm3[0] 378; AVX-NEXT: vpextrq $1, %xmm0, %rcx 379; AVX-NEXT: movq %rcx, %rax 380; AVX-NEXT: mulq %rsi 381; AVX-NEXT: movq %rcx, %rax 382; AVX-NEXT: subq %rdx, %rax 383; AVX-NEXT: shrq %rax 384; AVX-NEXT: addq %rdx, %rax 385; AVX-NEXT: shrq $2, %rax 386; AVX-NEXT: leaq (,%rax,8), %rdx 387; AVX-NEXT: subq %rdx, %rax 388; AVX-NEXT: addq %rcx, %rax 389; AVX-NEXT: vmovq %rax, %xmm3 390; AVX-NEXT: vmovq %xmm0, %rcx 391; AVX-NEXT: movq %rcx, %rax 392; AVX-NEXT: mulq %rsi 393; AVX-NEXT: movq %rcx, %rax 394; AVX-NEXT: subq %rdx, %rax 395; AVX-NEXT: shrq %rax 396; AVX-NEXT: addq %rdx, %rax 397; AVX-NEXT: shrq $2, %rax 398; AVX-NEXT: leaq (,%rax,8), %rdx 399; AVX-NEXT: subq %rdx, %rax 400; AVX-NEXT: addq %rcx, %rax 401; AVX-NEXT: vmovq %rax, %xmm0 402; AVX-NEXT: vpunpcklqdq {{.*#+}} xmm0 = xmm0[0],xmm3[0] 403; AVX-NEXT: vinserti128 $1, %xmm2, %ymm0, %ymm0 404; AVX-NEXT: vinserti64x4 $1, %ymm1, %zmm0, %zmm0 405; AVX-NEXT: retq 406 %res = urem <8 x i64> %a, <i64 7, i64 7, i64 7, i64 7, i64 7, i64 7, i64 7, i64 7> 407 ret <8 x i64> %res 408} 409 410define <16 x i32> @test_rem7_16i32(<16 x i32> %a) nounwind { 411; AVX-LABEL: test_rem7_16i32: 412; AVX: # %bb.0: 413; AVX-NEXT: vpbroadcastd {{.*#+}} zmm1 = [613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757,613566757] 414; AVX-NEXT: vpmuludq %zmm1, %zmm0, %zmm2 415; AVX-NEXT: vpshufd {{.*#+}} zmm3 = zmm0[1,1,3,3,5,5,7,7,9,9,11,11,13,13,15,15] 416; AVX-NEXT: vpmuludq %zmm1, %zmm3, %zmm1 417; AVX-NEXT: vpmovsxbd {{.*#+}} zmm3 = [1,17,3,19,5,21,7,23,9,25,11,27,13,29,15,31] 418; AVX-NEXT: vpermi2d %zmm1, %zmm2, %zmm3 419; AVX-NEXT: vpsubd %zmm3, %zmm0, %zmm1 420; AVX-NEXT: vpsrld $1, %zmm1, %zmm1 421; AVX-NEXT: vpaddd %zmm3, %zmm1, %zmm1 422; AVX-NEXT: vpsrld $2, %zmm1, %zmm1 423; AVX-NEXT: vpslld $3, %zmm1, %zmm2 424; AVX-NEXT: vpsubd %zmm2, %zmm1, %zmm1 425; AVX-NEXT: vpaddd %zmm1, %zmm0, %zmm0 426; AVX-NEXT: retq 427 %res = urem <16 x i32> %a, <i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7, i32 7> 428 ret <16 x i32> %res 429} 430 431define <32 x i16> @test_rem7_32i16(<32 x i16> %a) nounwind { 432; AVX512F-LABEL: test_rem7_32i16: 433; AVX512F: # %bb.0: 434; AVX512F-NEXT: vextracti64x4 $1, %zmm0, %ymm1 435; AVX512F-NEXT: vpbroadcastw {{.*#+}} ymm2 = [9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363] 436; AVX512F-NEXT: vpmulhuw %ymm2, %ymm1, %ymm3 437; AVX512F-NEXT: vpsubw %ymm3, %ymm1, %ymm4 438; AVX512F-NEXT: vpsrlw $1, %ymm4, %ymm4 439; AVX512F-NEXT: vpaddw %ymm3, %ymm4, %ymm3 440; AVX512F-NEXT: vpsrlw $2, %ymm3, %ymm3 441; AVX512F-NEXT: vpsllw $3, %ymm3, %ymm4 442; AVX512F-NEXT: vpsubw %ymm4, %ymm3, %ymm3 443; AVX512F-NEXT: vpaddw %ymm3, %ymm1, %ymm1 444; AVX512F-NEXT: vpmulhuw %ymm2, %ymm0, %ymm2 445; AVX512F-NEXT: vpsubw %ymm2, %ymm0, %ymm3 446; AVX512F-NEXT: vpsrlw $1, %ymm3, %ymm3 447; AVX512F-NEXT: vpaddw %ymm2, %ymm3, %ymm2 448; AVX512F-NEXT: vpsrlw $2, %ymm2, %ymm2 449; AVX512F-NEXT: vpsllw $3, %ymm2, %ymm3 450; AVX512F-NEXT: vpsubw %ymm3, %ymm2, %ymm2 451; AVX512F-NEXT: vpaddw %ymm2, %ymm0, %ymm0 452; AVX512F-NEXT: vinserti64x4 $1, %ymm1, %zmm0, %zmm0 453; AVX512F-NEXT: retq 454; 455; AVX512BW-LABEL: test_rem7_32i16: 456; AVX512BW: # %bb.0: 457; AVX512BW-NEXT: vpmulhuw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm0, %zmm1 # [9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363,9363] 458; AVX512BW-NEXT: vpsubw %zmm1, %zmm0, %zmm2 459; AVX512BW-NEXT: vpsrlw $1, %zmm2, %zmm2 460; AVX512BW-NEXT: vpaddw %zmm1, %zmm2, %zmm1 461; AVX512BW-NEXT: vpsrlw $2, %zmm1, %zmm1 462; AVX512BW-NEXT: vpsllw $3, %zmm1, %zmm2 463; AVX512BW-NEXT: vpsubw %zmm2, %zmm1, %zmm1 464; AVX512BW-NEXT: vpaddw %zmm1, %zmm0, %zmm0 465; AVX512BW-NEXT: retq 466 %res = urem <32 x i16> %a, <i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7, i16 7> 467 ret <32 x i16> %res 468} 469 470define <64 x i8> @test_rem7_64i8(<64 x i8> %a) nounwind { 471; AVX512F-LABEL: test_rem7_64i8: 472; AVX512F: # %bb.0: 473; AVX512F-NEXT: vextracti64x4 $1, %zmm0, %ymm1 474; AVX512F-NEXT: vpxor %xmm2, %xmm2, %xmm2 475; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm3 = ymm1[8],ymm2[8],ymm1[9],ymm2[9],ymm1[10],ymm2[10],ymm1[11],ymm2[11],ymm1[12],ymm2[12],ymm1[13],ymm2[13],ymm1[14],ymm2[14],ymm1[15],ymm2[15],ymm1[24],ymm2[24],ymm1[25],ymm2[25],ymm1[26],ymm2[26],ymm1[27],ymm2[27],ymm1[28],ymm2[28],ymm1[29],ymm2[29],ymm1[30],ymm2[30],ymm1[31],ymm2[31] 476; AVX512F-NEXT: vpbroadcastw {{.*#+}} ymm4 = [37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37] 477; AVX512F-NEXT: vpmullw %ymm4, %ymm3, %ymm3 478; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 479; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm5 = ymm1[0],ymm2[0],ymm1[1],ymm2[1],ymm1[2],ymm2[2],ymm1[3],ymm2[3],ymm1[4],ymm2[4],ymm1[5],ymm2[5],ymm1[6],ymm2[6],ymm1[7],ymm2[7],ymm1[16],ymm2[16],ymm1[17],ymm2[17],ymm1[18],ymm2[18],ymm1[19],ymm2[19],ymm1[20],ymm2[20],ymm1[21],ymm2[21],ymm1[22],ymm2[22],ymm1[23],ymm2[23] 480; AVX512F-NEXT: vpmullw %ymm4, %ymm5, %ymm5 481; AVX512F-NEXT: vpsrlw $8, %ymm5, %ymm5 482; AVX512F-NEXT: vpackuswb %ymm3, %ymm5, %ymm3 483; AVX512F-NEXT: vpsubb %ymm3, %ymm1, %ymm5 484; AVX512F-NEXT: vpsrlw $1, %ymm5, %ymm5 485; AVX512F-NEXT: vpbroadcastb {{.*#+}} ymm6 = [127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127,127] 486; AVX512F-NEXT: vpand %ymm6, %ymm5, %ymm5 487; AVX512F-NEXT: vpaddb %ymm3, %ymm5, %ymm3 488; AVX512F-NEXT: vpsllw $1, %ymm3, %ymm5 489; AVX512F-NEXT: vpbroadcastb {{.*#+}} ymm7 = [248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248,248] 490; AVX512F-NEXT: vpand %ymm7, %ymm5, %ymm5 491; AVX512F-NEXT: vpsrlw $2, %ymm3, %ymm3 492; AVX512F-NEXT: vpbroadcastb {{.*#+}} ymm8 = [63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63,63] 493; AVX512F-NEXT: vpand %ymm3, %ymm8, %ymm3 494; AVX512F-NEXT: vpsubb %ymm5, %ymm3, %ymm3 495; AVX512F-NEXT: vpaddb %ymm3, %ymm1, %ymm1 496; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm3 = ymm0[8],ymm2[8],ymm0[9],ymm2[9],ymm0[10],ymm2[10],ymm0[11],ymm2[11],ymm0[12],ymm2[12],ymm0[13],ymm2[13],ymm0[14],ymm2[14],ymm0[15],ymm2[15],ymm0[24],ymm2[24],ymm0[25],ymm2[25],ymm0[26],ymm2[26],ymm0[27],ymm2[27],ymm0[28],ymm2[28],ymm0[29],ymm2[29],ymm0[30],ymm2[30],ymm0[31],ymm2[31] 497; AVX512F-NEXT: vpmullw %ymm4, %ymm3, %ymm3 498; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 499; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm2 = ymm0[0],ymm2[0],ymm0[1],ymm2[1],ymm0[2],ymm2[2],ymm0[3],ymm2[3],ymm0[4],ymm2[4],ymm0[5],ymm2[5],ymm0[6],ymm2[6],ymm0[7],ymm2[7],ymm0[16],ymm2[16],ymm0[17],ymm2[17],ymm0[18],ymm2[18],ymm0[19],ymm2[19],ymm0[20],ymm2[20],ymm0[21],ymm2[21],ymm0[22],ymm2[22],ymm0[23],ymm2[23] 500; AVX512F-NEXT: vpmullw %ymm4, %ymm2, %ymm2 501; AVX512F-NEXT: vpsrlw $8, %ymm2, %ymm2 502; AVX512F-NEXT: vpackuswb %ymm3, %ymm2, %ymm2 503; AVX512F-NEXT: vpsubb %ymm2, %ymm0, %ymm3 504; AVX512F-NEXT: vpsrlw $1, %ymm3, %ymm3 505; AVX512F-NEXT: vpand %ymm6, %ymm3, %ymm3 506; AVX512F-NEXT: vpaddb %ymm2, %ymm3, %ymm2 507; AVX512F-NEXT: vpsllw $1, %ymm2, %ymm3 508; AVX512F-NEXT: vpand %ymm7, %ymm3, %ymm3 509; AVX512F-NEXT: vpsrlw $2, %ymm2, %ymm2 510; AVX512F-NEXT: vpand %ymm2, %ymm8, %ymm2 511; AVX512F-NEXT: vpsubb %ymm3, %ymm2, %ymm2 512; AVX512F-NEXT: vpaddb %ymm2, %ymm0, %ymm0 513; AVX512F-NEXT: vinserti64x4 $1, %ymm1, %zmm0, %zmm0 514; AVX512F-NEXT: retq 515; 516; AVX512BW-LABEL: test_rem7_64i8: 517; AVX512BW: # %bb.0: 518; AVX512BW-NEXT: vpxor %xmm1, %xmm1, %xmm1 519; AVX512BW-NEXT: vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63] 520; AVX512BW-NEXT: vpbroadcastw {{.*#+}} zmm3 = [37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37,37] 521; AVX512BW-NEXT: vpmullw %zmm3, %zmm2, %zmm2 522; AVX512BW-NEXT: vpsrlw $8, %zmm2, %zmm2 523; AVX512BW-NEXT: vpunpcklbw {{.*#+}} zmm1 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55] 524; AVX512BW-NEXT: vpmullw %zmm3, %zmm1, %zmm1 525; AVX512BW-NEXT: vpsrlw $8, %zmm1, %zmm1 526; AVX512BW-NEXT: vpackuswb %zmm2, %zmm1, %zmm1 527; AVX512BW-NEXT: vpsubb %zmm1, %zmm0, %zmm2 528; AVX512BW-NEXT: vpsrlw $1, %zmm2, %zmm2 529; AVX512BW-NEXT: vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm2, %zmm2 530; AVX512BW-NEXT: vpaddb %zmm1, %zmm2, %zmm1 531; AVX512BW-NEXT: vpsllw $1, %zmm1, %zmm2 532; AVX512BW-NEXT: vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm2, %zmm2 533; AVX512BW-NEXT: vpsrlw $2, %zmm1, %zmm1 534; AVX512BW-NEXT: vpandd {{\.?LCPI[0-9]+_[0-9]+}}(%rip){1to16}, %zmm1, %zmm1 535; AVX512BW-NEXT: vpsubb %zmm2, %zmm1, %zmm1 536; AVX512BW-NEXT: vpaddb %zmm1, %zmm0, %zmm0 537; AVX512BW-NEXT: retq 538 %res = urem <64 x i8> %a, <i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7, i8 7,i8 7, i8 7, i8 7, i8 7> 539 ret <64 x i8> %res 540} 541 542; 543; urem by non-splat constant 544; 545 546define <64 x i8> @test_remconstant_64i8(<64 x i8> %a) nounwind { 547; AVX512F-LABEL: test_remconstant_64i8: 548; AVX512F: # %bb.0: 549; AVX512F-NEXT: vextracti64x4 $1, %zmm0, %ymm2 550; AVX512F-NEXT: vpxor %xmm1, %xmm1, %xmm1 551; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm3 = ymm2[8],ymm1[8],ymm2[9],ymm1[9],ymm2[10],ymm1[10],ymm2[11],ymm1[11],ymm2[12],ymm1[12],ymm2[13],ymm1[13],ymm2[14],ymm1[14],ymm2[15],ymm1[15],ymm2[24],ymm1[24],ymm2[25],ymm1[25],ymm2[26],ymm1[26],ymm2[27],ymm1[27],ymm2[28],ymm1[28],ymm2[29],ymm1[29],ymm2[30],ymm1[30],ymm2[31],ymm1[31] 552; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [256,256,64,256,256,256,256,256,128,256,256,256,256,256,256,256] 553; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 554; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [137,27,37,19,79,41,171,101,147,79,171,117,205,57,32,37] 555; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 556; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm4 = ymm2[0],ymm1[0],ymm2[1],ymm1[1],ymm2[2],ymm1[2],ymm2[3],ymm1[3],ymm2[4],ymm1[4],ymm2[5],ymm1[5],ymm2[6],ymm1[6],ymm2[7],ymm1[7],ymm2[16],ymm1[16],ymm2[17],ymm1[17],ymm2[18],ymm1[18],ymm2[19],ymm1[19],ymm2[20],ymm1[20],ymm2[21],ymm1[21],ymm2[22],ymm1[22],ymm2[23],ymm1[23] 557; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [256,256,256,256,256,256,256,256,128,256,256,256,256,256,256,256] 558; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 559; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [27,111,57,235,241,249,8,9,187,135,205,27,57,241,16,137] 560; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 561; AVX512F-NEXT: vpackuswb %ymm3, %ymm4, %ymm3 562; AVX512F-NEXT: vpsubb %ymm3, %ymm2, %ymm4 563; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm5 = ymm4[8],ymm1[8],ymm4[9],ymm1[9],ymm4[10],ymm1[10],ymm4[11],ymm1[11],ymm4[12],ymm1[12],ymm4[13],ymm1[13],ymm4[14],ymm1[14],ymm4[15],ymm1[15],ymm4[24],ymm1[24],ymm4[25],ymm1[25],ymm4[26],ymm1[26],ymm4[27],ymm1[27],ymm4[28],ymm1[28],ymm4[29],ymm1[29],ymm4[30],ymm1[30],ymm4[31],ymm1[31] 564; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [0,128,0,0,0,0,0,128,0,0,0,128,0,0,0,128] 565; AVX512F-NEXT: vpsrlw $8, %ymm5, %ymm5 566; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm4 = ymm4[0],ymm1[0],ymm4[1],ymm1[1],ymm4[2],ymm1[2],ymm4[3],ymm1[3],ymm4[4],ymm1[4],ymm4[5],ymm1[5],ymm4[6],ymm1[6],ymm4[7],ymm1[7],ymm4[16],ymm1[16],ymm4[17],ymm1[17],ymm4[18],ymm1[18],ymm4[19],ymm1[19],ymm4[20],ymm1[20],ymm4[21],ymm1[21],ymm4[22],ymm1[22],ymm4[23],ymm1[23] 567; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [0,0,0,0,0,0,0,128,0,128,0,0,0,0,0,0] 568; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 569; AVX512F-NEXT: vpackuswb %ymm5, %ymm4, %ymm4 570; AVX512F-NEXT: vpaddb %ymm3, %ymm4, %ymm3 571; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm4 = ymm3[8],ymm1[8],ymm3[9],ymm1[9],ymm3[10],ymm1[10],ymm3[11],ymm1[11],ymm3[12],ymm1[12],ymm3[13],ymm1[13],ymm3[14],ymm1[14],ymm3[15],ymm1[15],ymm3[24],ymm1[24],ymm3[25],ymm1[25],ymm3[26],ymm1[26],ymm3[27],ymm1[27],ymm3[28],ymm1[28],ymm3[29],ymm1[29],ymm3[30],ymm1[30],ymm3[31],ymm1[31] 572; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [16,16,256,128,32,64,16,16,64,64,32,32,32,128,256,64] 573; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 574; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm3 = ymm3[0],ymm1[0],ymm3[1],ymm1[1],ymm3[2],ymm1[2],ymm3[3],ymm1[3],ymm3[4],ymm1[4],ymm3[5],ymm1[5],ymm3[6],ymm1[6],ymm3[7],ymm1[7],ymm3[16],ymm1[16],ymm3[17],ymm1[17],ymm3[18],ymm1[18],ymm3[19],ymm1[19],ymm3[20],ymm1[20],ymm3[21],ymm1[21],ymm3[22],ymm1[22],ymm3[23],ymm1[23] 575; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm3, %ymm3 # [64,16,32,8,8,8,256,16,32,16,16,128,64,16,256,32] 576; AVX512F-NEXT: vpsrlw $8, %ymm3, %ymm3 577; AVX512F-NEXT: vpackuswb %ymm4, %ymm3, %ymm4 578; AVX512F-NEXT: vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm5 # [38,0,36,0,34,0,32,0,30,0,28,0,26,0,24,0,22,0,20,0,18,0,16,0,14,0,12,0,10,0,8,0] 579; AVX512F-NEXT: vpbroadcastw {{.*#+}} ymm3 = [255,255,255,255,255,255,255,255,255,255,255,255,255,255,255,255] 580; AVX512F-NEXT: vpand %ymm3, %ymm5, %ymm5 581; AVX512F-NEXT: vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [0,37,0,35,0,33,0,31,0,29,0,27,0,25,0,23,0,21,0,19,0,17,0,15,0,13,0,11,0,9,0,7] 582; AVX512F-NEXT: vpsllw $8, %ymm4, %ymm4 583; AVX512F-NEXT: vpor %ymm4, %ymm5, %ymm4 584; AVX512F-NEXT: vpsubb %ymm4, %ymm2, %ymm2 585; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm4 = ymm0[8],ymm1[8],ymm0[9],ymm1[9],ymm0[10],ymm1[10],ymm0[11],ymm1[11],ymm0[12],ymm1[12],ymm0[13],ymm1[13],ymm0[14],ymm1[14],ymm0[15],ymm1[15],ymm0[24],ymm1[24],ymm0[25],ymm1[25],ymm0[26],ymm1[26],ymm0[27],ymm1[27],ymm0[28],ymm1[28],ymm0[29],ymm1[29],ymm0[30],ymm1[30],ymm0[31],ymm1[31] 586; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [256,256,256,256,256,256,256,128,256,256,256,256,256,256,256,256] 587; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 588; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm4, %ymm4 # [137,16,241,57,27,205,135,187,9,8,249,241,235,57,111,27] 589; AVX512F-NEXT: vpsrlw $8, %ymm4, %ymm4 590; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm5 = ymm0[0],ymm1[0],ymm0[1],ymm1[1],ymm0[2],ymm1[2],ymm0[3],ymm1[3],ymm0[4],ymm1[4],ymm0[5],ymm1[5],ymm0[6],ymm1[6],ymm0[7],ymm1[7],ymm0[16],ymm1[16],ymm0[17],ymm1[17],ymm0[18],ymm1[18],ymm0[19],ymm1[19],ymm0[20],ymm1[20],ymm0[21],ymm1[21],ymm0[22],ymm1[22],ymm0[23],ymm1[23] 591; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [256,256,256,256,256,256,256,128,256,256,256,256,256,64,256,256] 592; AVX512F-NEXT: vpsrlw $8, %ymm5, %ymm5 593; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [37,32,57,205,117,171,79,147,101,171,41,79,19,37,27,137] 594; AVX512F-NEXT: vpsrlw $8, %ymm5, %ymm5 595; AVX512F-NEXT: vpackuswb %ymm4, %ymm5, %ymm4 596; AVX512F-NEXT: vpsubb %ymm4, %ymm0, %ymm5 597; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm6 = ymm5[8],ymm1[8],ymm5[9],ymm1[9],ymm5[10],ymm1[10],ymm5[11],ymm1[11],ymm5[12],ymm1[12],ymm5[13],ymm1[13],ymm5[14],ymm1[14],ymm5[15],ymm1[15],ymm5[24],ymm1[24],ymm5[25],ymm1[25],ymm5[26],ymm1[26],ymm5[27],ymm1[27],ymm5[28],ymm1[28],ymm5[29],ymm1[29],ymm5[30],ymm1[30],ymm5[31],ymm1[31] 598; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm6, %ymm6 # [0,0,0,0,0,0,128,0,128,0,0,0,0,0,0,0] 599; AVX512F-NEXT: vpsrlw $8, %ymm6, %ymm6 600; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm5 = ymm5[0],ymm1[0],ymm5[1],ymm1[1],ymm5[2],ymm1[2],ymm5[3],ymm1[3],ymm5[4],ymm1[4],ymm5[5],ymm1[5],ymm5[6],ymm1[6],ymm5[7],ymm1[7],ymm5[16],ymm1[16],ymm5[17],ymm1[17],ymm5[18],ymm1[18],ymm5[19],ymm1[19],ymm5[20],ymm1[20],ymm5[21],ymm1[21],ymm5[22],ymm1[22],ymm5[23],ymm1[23] 601; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [128,0,0,0,128,0,0,0,128,0,0,0,0,0,128,0] 602; AVX512F-NEXT: vpsrlw $8, %ymm5, %ymm5 603; AVX512F-NEXT: vpackuswb %ymm6, %ymm5, %ymm5 604; AVX512F-NEXT: vpaddb %ymm4, %ymm5, %ymm4 605; AVX512F-NEXT: vpunpckhbw {{.*#+}} ymm5 = ymm4[8],ymm1[8],ymm4[9],ymm1[9],ymm4[10],ymm1[10],ymm4[11],ymm1[11],ymm4[12],ymm1[12],ymm4[13],ymm1[13],ymm4[14],ymm1[14],ymm4[15],ymm1[15],ymm4[24],ymm1[24],ymm4[25],ymm1[25],ymm4[26],ymm1[26],ymm4[27],ymm1[27],ymm4[28],ymm1[28],ymm4[29],ymm1[29],ymm4[30],ymm1[30],ymm4[31],ymm1[31] 606; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm5, %ymm5 # [32,256,16,64,128,16,16,32,16,256,8,8,8,32,16,64] 607; AVX512F-NEXT: vpsrlw $8, %ymm5, %ymm5 608; AVX512F-NEXT: vpunpcklbw {{.*#+}} ymm1 = ymm4[0],ymm1[0],ymm4[1],ymm1[1],ymm4[2],ymm1[2],ymm4[3],ymm1[3],ymm4[4],ymm1[4],ymm4[5],ymm1[5],ymm4[6],ymm1[6],ymm4[7],ymm1[7],ymm4[16],ymm1[16],ymm4[17],ymm1[17],ymm4[18],ymm1[18],ymm4[19],ymm1[19],ymm4[20],ymm1[20],ymm4[21],ymm1[21],ymm4[22],ymm1[22],ymm4[23],ymm1[23] 609; AVX512F-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm1, %ymm1 # [64,256,128,32,32,32,64,64,16,16,64,32,128,256,16,16] 610; AVX512F-NEXT: vpsrlw $8, %ymm1, %ymm1 611; AVX512F-NEXT: vpackuswb %ymm5, %ymm1, %ymm1 612; AVX512F-NEXT: vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm1, %ymm4 # [7,0,9,0,11,0,13,0,15,0,17,0,19,0,21,0,23,0,25,0,27,0,29,0,31,0,33,0,35,0,37,0] 613; AVX512F-NEXT: vpand %ymm3, %ymm4, %ymm3 614; AVX512F-NEXT: vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %ymm1, %ymm1 # [0,8,0,10,0,12,0,14,0,16,0,18,0,20,0,22,0,24,0,26,0,28,0,30,0,32,0,34,0,36,0,38] 615; AVX512F-NEXT: vpsllw $8, %ymm1, %ymm1 616; AVX512F-NEXT: vpor %ymm1, %ymm3, %ymm1 617; AVX512F-NEXT: vpsubb %ymm1, %ymm0, %ymm0 618; AVX512F-NEXT: vinserti64x4 $1, %ymm2, %zmm0, %zmm0 619; AVX512F-NEXT: retq 620; 621; AVX512BW-LABEL: test_remconstant_64i8: 622; AVX512BW: # %bb.0: 623; AVX512BW-NEXT: vpxor %xmm1, %xmm1, %xmm1 624; AVX512BW-NEXT: vpunpckhbw {{.*#+}} zmm2 = zmm0[8],zmm1[8],zmm0[9],zmm1[9],zmm0[10],zmm1[10],zmm0[11],zmm1[11],zmm0[12],zmm1[12],zmm0[13],zmm1[13],zmm0[14],zmm1[14],zmm0[15],zmm1[15],zmm0[24],zmm1[24],zmm0[25],zmm1[25],zmm0[26],zmm1[26],zmm0[27],zmm1[27],zmm0[28],zmm1[28],zmm0[29],zmm1[29],zmm0[30],zmm1[30],zmm0[31],zmm1[31],zmm0[40],zmm1[40],zmm0[41],zmm1[41],zmm0[42],zmm1[42],zmm0[43],zmm1[43],zmm0[44],zmm1[44],zmm0[45],zmm1[45],zmm0[46],zmm1[46],zmm0[47],zmm1[47],zmm0[56],zmm1[56],zmm0[57],zmm1[57],zmm0[58],zmm1[58],zmm0[59],zmm1[59],zmm0[60],zmm1[60],zmm0[61],zmm1[61],zmm0[62],zmm1[62],zmm0[63],zmm1[63] 625; AVX512BW-NEXT: vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2 626; AVX512BW-NEXT: vpsrlw $8, %zmm2, %zmm2 627; AVX512BW-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm2, %zmm2 # [137,16,241,57,27,205,135,187,9,8,249,241,235,57,111,27,137,27,37,19,79,41,171,101,147,79,171,117,205,57,32,37] 628; AVX512BW-NEXT: vpsrlw $8, %zmm2, %zmm2 629; AVX512BW-NEXT: vpunpcklbw {{.*#+}} zmm3 = zmm0[0],zmm1[0],zmm0[1],zmm1[1],zmm0[2],zmm1[2],zmm0[3],zmm1[3],zmm0[4],zmm1[4],zmm0[5],zmm1[5],zmm0[6],zmm1[6],zmm0[7],zmm1[7],zmm0[16],zmm1[16],zmm0[17],zmm1[17],zmm0[18],zmm1[18],zmm0[19],zmm1[19],zmm0[20],zmm1[20],zmm0[21],zmm1[21],zmm0[22],zmm1[22],zmm0[23],zmm1[23],zmm0[32],zmm1[32],zmm0[33],zmm1[33],zmm0[34],zmm1[34],zmm0[35],zmm1[35],zmm0[36],zmm1[36],zmm0[37],zmm1[37],zmm0[38],zmm1[38],zmm0[39],zmm1[39],zmm0[48],zmm1[48],zmm0[49],zmm1[49],zmm0[50],zmm1[50],zmm0[51],zmm1[51],zmm0[52],zmm1[52],zmm0[53],zmm1[53],zmm0[54],zmm1[54],zmm0[55],zmm1[55] 630; AVX512BW-NEXT: vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 631; AVX512BW-NEXT: vpsrlw $8, %zmm3, %zmm3 632; AVX512BW-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 # [37,32,57,205,117,171,79,147,101,171,41,79,19,37,27,137,27,111,57,235,241,249,8,9,187,135,205,27,57,241,16,137] 633; AVX512BW-NEXT: vpsrlw $8, %zmm3, %zmm3 634; AVX512BW-NEXT: vpackuswb %zmm2, %zmm3, %zmm2 635; AVX512BW-NEXT: vpsubb %zmm2, %zmm0, %zmm3 636; AVX512BW-NEXT: vpunpckhbw {{.*#+}} zmm4 = zmm3[8],zmm1[8],zmm3[9],zmm1[9],zmm3[10],zmm1[10],zmm3[11],zmm1[11],zmm3[12],zmm1[12],zmm3[13],zmm1[13],zmm3[14],zmm1[14],zmm3[15],zmm1[15],zmm3[24],zmm1[24],zmm3[25],zmm1[25],zmm3[26],zmm1[26],zmm3[27],zmm1[27],zmm3[28],zmm1[28],zmm3[29],zmm1[29],zmm3[30],zmm1[30],zmm3[31],zmm1[31],zmm3[40],zmm1[40],zmm3[41],zmm1[41],zmm3[42],zmm1[42],zmm3[43],zmm1[43],zmm3[44],zmm1[44],zmm3[45],zmm1[45],zmm3[46],zmm1[46],zmm3[47],zmm1[47],zmm3[56],zmm1[56],zmm3[57],zmm1[57],zmm3[58],zmm1[58],zmm3[59],zmm1[59],zmm3[60],zmm1[60],zmm3[61],zmm1[61],zmm3[62],zmm1[62],zmm3[63],zmm1[63] 637; AVX512BW-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm4, %zmm4 # [0,0,0,0,0,0,128,0,128,0,0,0,0,0,0,0,0,128,0,0,0,0,0,128,0,0,0,128,0,0,0,128] 638; AVX512BW-NEXT: vpsrlw $8, %zmm4, %zmm4 639; AVX512BW-NEXT: vpunpcklbw {{.*#+}} zmm3 = zmm3[0],zmm1[0],zmm3[1],zmm1[1],zmm3[2],zmm1[2],zmm3[3],zmm1[3],zmm3[4],zmm1[4],zmm3[5],zmm1[5],zmm3[6],zmm1[6],zmm3[7],zmm1[7],zmm3[16],zmm1[16],zmm3[17],zmm1[17],zmm3[18],zmm1[18],zmm3[19],zmm1[19],zmm3[20],zmm1[20],zmm3[21],zmm1[21],zmm3[22],zmm1[22],zmm3[23],zmm1[23],zmm3[32],zmm1[32],zmm3[33],zmm1[33],zmm3[34],zmm1[34],zmm3[35],zmm1[35],zmm3[36],zmm1[36],zmm3[37],zmm1[37],zmm3[38],zmm1[38],zmm3[39],zmm1[39],zmm3[48],zmm1[48],zmm3[49],zmm1[49],zmm3[50],zmm1[50],zmm3[51],zmm1[51],zmm3[52],zmm1[52],zmm3[53],zmm1[53],zmm3[54],zmm1[54],zmm3[55],zmm1[55] 640; AVX512BW-NEXT: vpmullw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 # [128,0,0,0,128,0,0,0,128,0,0,0,0,0,128,0,0,0,0,0,0,0,0,128,0,128,0,0,0,0,0,0] 641; AVX512BW-NEXT: vpsrlw $8, %zmm3, %zmm3 642; AVX512BW-NEXT: vpackuswb %zmm4, %zmm3, %zmm3 643; AVX512BW-NEXT: vpaddb %zmm2, %zmm3, %zmm2 644; AVX512BW-NEXT: vpunpckhbw {{.*#+}} zmm3 = zmm2[8],zmm1[8],zmm2[9],zmm1[9],zmm2[10],zmm1[10],zmm2[11],zmm1[11],zmm2[12],zmm1[12],zmm2[13],zmm1[13],zmm2[14],zmm1[14],zmm2[15],zmm1[15],zmm2[24],zmm1[24],zmm2[25],zmm1[25],zmm2[26],zmm1[26],zmm2[27],zmm1[27],zmm2[28],zmm1[28],zmm2[29],zmm1[29],zmm2[30],zmm1[30],zmm2[31],zmm1[31],zmm2[40],zmm1[40],zmm2[41],zmm1[41],zmm2[42],zmm1[42],zmm2[43],zmm1[43],zmm2[44],zmm1[44],zmm2[45],zmm1[45],zmm2[46],zmm1[46],zmm2[47],zmm1[47],zmm2[56],zmm1[56],zmm2[57],zmm1[57],zmm2[58],zmm1[58],zmm2[59],zmm1[59],zmm2[60],zmm1[60],zmm2[61],zmm1[61],zmm2[62],zmm1[62],zmm2[63],zmm1[63] 645; AVX512BW-NEXT: vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm3, %zmm3 646; AVX512BW-NEXT: vpsrlw $8, %zmm3, %zmm3 647; AVX512BW-NEXT: vpunpcklbw {{.*#+}} zmm1 = zmm2[0],zmm1[0],zmm2[1],zmm1[1],zmm2[2],zmm1[2],zmm2[3],zmm1[3],zmm2[4],zmm1[4],zmm2[5],zmm1[5],zmm2[6],zmm1[6],zmm2[7],zmm1[7],zmm2[16],zmm1[16],zmm2[17],zmm1[17],zmm2[18],zmm1[18],zmm2[19],zmm1[19],zmm2[20],zmm1[20],zmm2[21],zmm1[21],zmm2[22],zmm1[22],zmm2[23],zmm1[23],zmm2[32],zmm1[32],zmm2[33],zmm1[33],zmm2[34],zmm1[34],zmm2[35],zmm1[35],zmm2[36],zmm1[36],zmm2[37],zmm1[37],zmm2[38],zmm1[38],zmm2[39],zmm1[39],zmm2[48],zmm1[48],zmm2[49],zmm1[49],zmm2[50],zmm1[50],zmm2[51],zmm1[51],zmm2[52],zmm1[52],zmm2[53],zmm1[53],zmm2[54],zmm1[54],zmm2[55],zmm1[55] 648; AVX512BW-NEXT: vpsllvw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm1 649; AVX512BW-NEXT: vpsrlw $8, %zmm1, %zmm1 650; AVX512BW-NEXT: vpackuswb %zmm3, %zmm1, %zmm1 651; AVX512BW-NEXT: vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm2 # [7,0,9,0,11,0,13,0,15,0,17,0,19,0,21,0,23,0,25,0,27,0,29,0,31,0,33,0,35,0,37,0,38,0,36,0,34,0,32,0,30,0,28,0,26,0,24,0,22,0,20,0,18,0,16,0,14,0,12,0,10,0,8,0] 652; AVX512BW-NEXT: vpmaddubsw {{\.?LCPI[0-9]+_[0-9]+}}(%rip), %zmm1, %zmm1 # [0,8,0,10,0,12,0,14,0,16,0,18,0,20,0,22,0,24,0,26,0,28,0,30,0,32,0,34,0,36,0,38,0,37,0,35,0,33,0,31,0,29,0,27,0,25,0,23,0,21,0,19,0,17,0,15,0,13,0,11,0,9,0,7] 653; AVX512BW-NEXT: vpsllw $8, %zmm1, %zmm1 654; AVX512BW-NEXT: vpternlogd {{.*#+}} zmm1 = zmm1 | (zmm2 & mem) 655; AVX512BW-NEXT: vpsubb %zmm1, %zmm0, %zmm0 656; AVX512BW-NEXT: retq 657 %res = urem <64 x i8> %a, <i8 7, i8 8, i8 9, i8 10, i8 11, i8 12, i8 13, i8 14, i8 15, i8 16, i8 17, i8 18, i8 19, i8 20, i8 21, i8 22, i8 23, i8 24, i8 25, i8 26, i8 27, i8 28, i8 29, i8 30, i8 31, i8 32, i8 33, i8 34, i8 35, i8 36, i8 37, i8 38, i8 38, i8 37, i8 36, i8 35, i8 34, i8 33, i8 32, i8 31, i8 30, i8 29, i8 28, i8 27, i8 26, i8 25, i8 24, i8 23, i8 22, i8 21, i8 20, i8 19, i8 18, i8 17, i8 16, i8 15, i8 14, i8 13, i8 12, i8 11, i8 10, i8 9, i8 8, i8 7> 658 ret <64 x i8> %res 659} 660