Target/AArch64/AArch64TargetTransformInfo.cpp

09467b48Spatrick//===-- AArch64TargetTransformInfo.cpp - AArch64 specific TTI -------------===//
09467b48Spatrick//
09467b48Spatrick// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
09467b48Spatrick// See https://llvm.org/LICENSE.txt for license information.
09467b48Spatrick// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
09467b48Spatrick//
09467b48Spatrick//===----------------------------------------------------------------------===//
09467b48Spatrick
09467b48Spatrick#include "AArch64TargetTransformInfo.h"
73471bf0Spatrick#include "AArch64ExpandImm.h"
*d415bd75Srobert#include "AArch64PerfectShuffle.h"
09467b48Spatrick#include "MCTargetDesc/AArch64AddressingModes.h"
*d415bd75Srobert#include "llvm/Analysis/IVDescriptors.h"
09467b48Spatrick#include "llvm/Analysis/LoopInfo.h"
09467b48Spatrick#include "llvm/Analysis/TargetTransformInfo.h"
09467b48Spatrick#include "llvm/CodeGen/BasicTTIImpl.h"
09467b48Spatrick#include "llvm/CodeGen/CostTable.h"
09467b48Spatrick#include "llvm/CodeGen/TargetLowering.h"
09467b48Spatrick#include "llvm/IR/IntrinsicInst.h"
*d415bd75Srobert#include "llvm/IR/Intrinsics.h"
09467b48Spatrick#include "llvm/IR/IntrinsicsAArch64.h"
73471bf0Spatrick#include "llvm/IR/PatternMatch.h"
09467b48Spatrick#include "llvm/Support/Debug.h"
73471bf0Spatrick#include "llvm/Transforms/InstCombine/InstCombiner.h"
*d415bd75Srobert#include "llvm/Transforms/Vectorize/LoopVectorizationLegality.h"
09467b48Spatrick#include <algorithm>
*d415bd75Srobert#include <optional>
09467b48Spatrickusing namespace llvm;
73471bf0Spatrickusing namespace llvm::PatternMatch;
09467b48Spatrick
09467b48Spatrick#define DEBUG_TYPE "aarch64tti"
09467b48Spatrick
09467b48Spatrickstatic cl::opt<bool> EnableFalkorHWPFUnrollFix("enable-falkor-hwpf-unroll-fix",
09467b48Spatrick                                               cl::init(true), cl::Hidden);
09467b48Spatrick
*d415bd75Srobertstatic cl::opt<unsigned> SVEGatherOverhead("sve-gather-overhead", cl::init(10),
*d415bd75Srobert                                           cl::Hidden);
*d415bd75Srobert
*d415bd75Srobertstatic cl::opt<unsigned> SVEScatterOverhead("sve-scatter-overhead",
*d415bd75Srobert                                            cl::init(10), cl::Hidden);
*d415bd75Srobert
*d415bd75Srobertnamespace {
*d415bd75Srobertclass TailFoldingKind {
*d415bd75Srobertprivate:
*d415bd75Srobert  uint8_t Bits = 0; // Currently defaults to disabled.
*d415bd75Srobert
*d415bd75Srobertpublic:
*d415bd75Srobert  enum TailFoldingOpts {
*d415bd75Srobert    TFDisabled = 0x0,
*d415bd75Srobert    TFReductions = 0x01,
*d415bd75Srobert    TFRecurrences = 0x02,
*d415bd75Srobert    TFSimple = 0x80,
*d415bd75Srobert    TFAll = TFReductions | TFRecurrences | TFSimple
*d415bd75Srobert  };
*d415bd75Srobert
*d415bd75Srobert  void operator=(const std::string &Val) {
*d415bd75Srobert    if (Val.empty())
*d415bd75Srobert      return;
*d415bd75Srobert    SmallVector<StringRef, 6> TailFoldTypes;
*d415bd75Srobert    StringRef(Val).split(TailFoldTypes, '+', -1, false);
*d415bd75Srobert    for (auto TailFoldType : TailFoldTypes) {
*d415bd75Srobert      if (TailFoldType == "disabled")
*d415bd75Srobert        Bits = 0;
*d415bd75Srobert      else if (TailFoldType == "all")
*d415bd75Srobert        Bits = TFAll;
*d415bd75Srobert      else if (TailFoldType == "default")
*d415bd75Srobert        Bits = 0; // Currently defaults to never tail-folding.
*d415bd75Srobert      else if (TailFoldType == "simple")
*d415bd75Srobert        add(TFSimple);
*d415bd75Srobert      else if (TailFoldType == "reductions")
*d415bd75Srobert        add(TFReductions);
*d415bd75Srobert      else if (TailFoldType == "recurrences")
*d415bd75Srobert        add(TFRecurrences);
*d415bd75Srobert      else if (TailFoldType == "noreductions")
*d415bd75Srobert        remove(TFReductions);
*d415bd75Srobert      else if (TailFoldType == "norecurrences")
*d415bd75Srobert        remove(TFRecurrences);
*d415bd75Srobert      else {
*d415bd75Srobert        errs()
*d415bd75Srobert            << "invalid argument " << TailFoldType.str()
*d415bd75Srobert            << " to -sve-tail-folding=; each element must be one of: disabled, "
*d415bd75Srobert               "all, default, simple, reductions, noreductions, recurrences, "
*d415bd75Srobert               "norecurrences\n";
*d415bd75Srobert      }
*d415bd75Srobert    }
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  operator uint8_t() const { return Bits; }
*d415bd75Srobert
*d415bd75Srobert  void add(uint8_t Flag) { Bits |= Flag; }
*d415bd75Srobert  void remove(uint8_t Flag) { Bits &= ~Flag; }
*d415bd75Srobert};
*d415bd75Srobert} // namespace
*d415bd75Srobert
*d415bd75SrobertTailFoldingKind TailFoldingKindLoc;
*d415bd75Srobert
*d415bd75Srobertcl::opt<TailFoldingKind, true, cl::parser<std::string>> SVETailFolding(
*d415bd75Srobert    "sve-tail-folding",
*d415bd75Srobert    cl::desc(
*d415bd75Srobert        "Control the use of vectorisation using tail-folding for SVE:"
*d415bd75Srobert        "\ndisabled    No loop types will vectorize using tail-folding"
*d415bd75Srobert        "\ndefault     Uses the default tail-folding settings for the target "
*d415bd75Srobert        "CPU"
*d415bd75Srobert        "\nall         All legal loop types will vectorize using tail-folding"
*d415bd75Srobert        "\nsimple      Use tail-folding for simple loops (not reductions or "
*d415bd75Srobert        "recurrences)"
*d415bd75Srobert        "\nreductions  Use tail-folding for loops containing reductions"
*d415bd75Srobert        "\nrecurrences Use tail-folding for loops containing fixed order "
*d415bd75Srobert        "recurrences"),
*d415bd75Srobert    cl::location(TailFoldingKindLoc));
*d415bd75Srobert
*d415bd75Srobert// Experimental option that will only be fully functional when the
*d415bd75Srobert// code-generator is changed to use SVE instead of NEON for all fixed-width
*d415bd75Srobert// operations.
*d415bd75Srobertstatic cl::opt<bool> EnableFixedwidthAutovecInStreamingMode(
*d415bd75Srobert    "enable-fixedwidth-autovec-in-streaming-mode", cl::init(false), cl::Hidden);
*d415bd75Srobert
*d415bd75Srobert// Experimental option that will only be fully functional when the cost-model
*d415bd75Srobert// and code-generator have been changed to avoid using scalable vector
*d415bd75Srobert// instructions that are not legal in streaming SVE mode.
*d415bd75Srobertstatic cl::opt<bool> EnableScalableAutovecInStreamingMode(
*d415bd75Srobert    "enable-scalable-autovec-in-streaming-mode", cl::init(false), cl::Hidden);
*d415bd75Srobert
09467b48Spatrickbool AArch64TTIImpl::areInlineCompatible(const Function *Caller,
09467b48Spatrick                                         const Function *Callee) const {
*d415bd75Srobert  SMEAttrs CallerAttrs(*Caller);
*d415bd75Srobert  SMEAttrs CalleeAttrs(*Callee);
*d415bd75Srobert  if (CallerAttrs.requiresSMChange(CalleeAttrs,
*d415bd75Srobert                                   /*BodyOverridesInterface=*/true) ||
*d415bd75Srobert      CallerAttrs.requiresLazySave(CalleeAttrs) ||
*d415bd75Srobert      CalleeAttrs.hasNewZAInterface())
*d415bd75Srobert    return false;
*d415bd75Srobert
09467b48Spatrick  const TargetMachine &TM = getTLI()->getTargetMachine();
09467b48Spatrick
09467b48Spatrick  const FeatureBitset &CallerBits =
09467b48Spatrick      TM.getSubtargetImpl(*Caller)->getFeatureBits();
09467b48Spatrick  const FeatureBitset &CalleeBits =
09467b48Spatrick      TM.getSubtargetImpl(*Callee)->getFeatureBits();
09467b48Spatrick
09467b48Spatrick  // Inline a callee if its target-features are a subset of the callers
09467b48Spatrick  // target-features.
09467b48Spatrick  return (CallerBits & CalleeBits) == CalleeBits;
09467b48Spatrick}
09467b48Spatrick
*d415bd75Srobertbool AArch64TTIImpl::shouldMaximizeVectorBandwidth(
*d415bd75Srobert    TargetTransformInfo::RegisterKind K) const {
*d415bd75Srobert  assert(K != TargetTransformInfo::RGK_Scalar);
*d415bd75Srobert  return K == TargetTransformInfo::RGK_FixedWidthVector;
*d415bd75Srobert}
*d415bd75Srobert
09467b48Spatrick/// Calculate the cost of materializing a 64-bit value. This helper
09467b48Spatrick/// method might only calculate a fraction of a larger immediate. Therefore it
09467b48Spatrick/// is valid to return a cost of ZERO.
73471bf0SpatrickInstructionCost AArch64TTIImpl::getIntImmCost(int64_t Val) {
09467b48Spatrick  // Check if the immediate can be encoded within an instruction.
09467b48Spatrick  if (Val == 0 || AArch64_AM::isLogicalImmediate(Val, 64))
09467b48Spatrick    return 0;
09467b48Spatrick
09467b48Spatrick  if (Val < 0)
09467b48Spatrick    Val = ~Val;
09467b48Spatrick
09467b48Spatrick  // Calculate how many moves we will need to materialize this constant.
09467b48Spatrick  SmallVector<AArch64_IMM::ImmInsnModel, 4> Insn;
09467b48Spatrick  AArch64_IMM::expandMOVImm(Val, 64, Insn);
09467b48Spatrick  return Insn.size();
09467b48Spatrick}
09467b48Spatrick
09467b48Spatrick/// Calculate the cost of materializing the given constant.
73471bf0SpatrickInstructionCost AArch64TTIImpl::getIntImmCost(const APInt &Imm, Type *Ty,
097a140dSpatrick                                              TTI::TargetCostKind CostKind) {
09467b48Spatrick  assert(Ty->isIntegerTy());
09467b48Spatrick
09467b48Spatrick  unsigned BitSize = Ty->getPrimitiveSizeInBits();
09467b48Spatrick  if (BitSize == 0)
09467b48Spatrick    return ~0U;
09467b48Spatrick
09467b48Spatrick  // Sign-extend all constants to a multiple of 64-bit.
09467b48Spatrick  APInt ImmVal = Imm;
09467b48Spatrick  if (BitSize & 0x3f)
09467b48Spatrick    ImmVal = Imm.sext((BitSize + 63) & ~0x3fU);
09467b48Spatrick
09467b48Spatrick  // Split the constant into 64-bit chunks and calculate the cost for each
09467b48Spatrick  // chunk.
73471bf0Spatrick  InstructionCost Cost = 0;
09467b48Spatrick  for (unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
09467b48Spatrick    APInt Tmp = ImmVal.ashr(ShiftVal).sextOrTrunc(64);
09467b48Spatrick    int64_t Val = Tmp.getSExtValue();
09467b48Spatrick    Cost += getIntImmCost(Val);
09467b48Spatrick  }
09467b48Spatrick  // We need at least one instruction to materialze the constant.
73471bf0Spatrick  return std::max<InstructionCost>(1, Cost);
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getIntImmCostInst(unsigned Opcode, unsigned Idx,
097a140dSpatrick                                                  const APInt &Imm, Type *Ty,
73471bf0Spatrick                                                  TTI::TargetCostKind CostKind,
73471bf0Spatrick                                                  Instruction *Inst) {
09467b48Spatrick  assert(Ty->isIntegerTy());
09467b48Spatrick
09467b48Spatrick  unsigned BitSize = Ty->getPrimitiveSizeInBits();
09467b48Spatrick  // There is no cost model for constants with a bit size of 0. Return TCC_Free
09467b48Spatrick  // here, so that constant hoisting will ignore this constant.
09467b48Spatrick  if (BitSize == 0)
09467b48Spatrick    return TTI::TCC_Free;
09467b48Spatrick
09467b48Spatrick  unsigned ImmIdx = ~0U;
09467b48Spatrick  switch (Opcode) {
09467b48Spatrick  default:
09467b48Spatrick    return TTI::TCC_Free;
09467b48Spatrick  case Instruction::GetElementPtr:
09467b48Spatrick    // Always hoist the base address of a GetElementPtr.
09467b48Spatrick    if (Idx == 0)
09467b48Spatrick      return 2 * TTI::TCC_Basic;
09467b48Spatrick    return TTI::TCC_Free;
09467b48Spatrick  case Instruction::Store:
09467b48Spatrick    ImmIdx = 0;
09467b48Spatrick    break;
09467b48Spatrick  case Instruction::Add:
09467b48Spatrick  case Instruction::Sub:
09467b48Spatrick  case Instruction::Mul:
09467b48Spatrick  case Instruction::UDiv:
09467b48Spatrick  case Instruction::SDiv:
09467b48Spatrick  case Instruction::URem:
09467b48Spatrick  case Instruction::SRem:
09467b48Spatrick  case Instruction::And:
09467b48Spatrick  case Instruction::Or:
09467b48Spatrick  case Instruction::Xor:
09467b48Spatrick  case Instruction::ICmp:
09467b48Spatrick    ImmIdx = 1;
09467b48Spatrick    break;
09467b48Spatrick  // Always return TCC_Free for the shift value of a shift instruction.
09467b48Spatrick  case Instruction::Shl:
09467b48Spatrick  case Instruction::LShr:
09467b48Spatrick  case Instruction::AShr:
09467b48Spatrick    if (Idx == 1)
09467b48Spatrick      return TTI::TCC_Free;
09467b48Spatrick    break;
09467b48Spatrick  case Instruction::Trunc:
09467b48Spatrick  case Instruction::ZExt:
09467b48Spatrick  case Instruction::SExt:
09467b48Spatrick  case Instruction::IntToPtr:
09467b48Spatrick  case Instruction::PtrToInt:
09467b48Spatrick  case Instruction::BitCast:
09467b48Spatrick  case Instruction::PHI:
09467b48Spatrick  case Instruction::Call:
09467b48Spatrick  case Instruction::Select:
09467b48Spatrick  case Instruction::Ret:
09467b48Spatrick  case Instruction::Load:
09467b48Spatrick    break;
09467b48Spatrick  }
09467b48Spatrick
09467b48Spatrick  if (Idx == ImmIdx) {
09467b48Spatrick    int NumConstants = (BitSize + 63) / 64;
73471bf0Spatrick    InstructionCost Cost = AArch64TTIImpl::getIntImmCost(Imm, Ty, CostKind);
09467b48Spatrick    return (Cost <= NumConstants * TTI::TCC_Basic)
09467b48Spatrick               ? static_cast<int>(TTI::TCC_Free)
09467b48Spatrick               : Cost;
09467b48Spatrick  }
097a140dSpatrick  return AArch64TTIImpl::getIntImmCost(Imm, Ty, CostKind);
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost
73471bf0SpatrickAArch64TTIImpl::getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx,
097a140dSpatrick                                    const APInt &Imm, Type *Ty,
097a140dSpatrick                                    TTI::TargetCostKind CostKind) {
09467b48Spatrick  assert(Ty->isIntegerTy());
09467b48Spatrick
09467b48Spatrick  unsigned BitSize = Ty->getPrimitiveSizeInBits();
09467b48Spatrick  // There is no cost model for constants with a bit size of 0. Return TCC_Free
09467b48Spatrick  // here, so that constant hoisting will ignore this constant.
09467b48Spatrick  if (BitSize == 0)
09467b48Spatrick    return TTI::TCC_Free;
09467b48Spatrick
09467b48Spatrick  // Most (all?) AArch64 intrinsics do not support folding immediates into the
09467b48Spatrick  // selected instruction, so we compute the materialization cost for the
09467b48Spatrick  // immediate directly.
09467b48Spatrick  if (IID >= Intrinsic::aarch64_addg && IID <= Intrinsic::aarch64_udiv)
097a140dSpatrick    return AArch64TTIImpl::getIntImmCost(Imm, Ty, CostKind);
09467b48Spatrick
09467b48Spatrick  switch (IID) {
09467b48Spatrick  default:
09467b48Spatrick    return TTI::TCC_Free;
09467b48Spatrick  case Intrinsic::sadd_with_overflow:
09467b48Spatrick  case Intrinsic::uadd_with_overflow:
09467b48Spatrick  case Intrinsic::ssub_with_overflow:
09467b48Spatrick  case Intrinsic::usub_with_overflow:
09467b48Spatrick  case Intrinsic::smul_with_overflow:
09467b48Spatrick  case Intrinsic::umul_with_overflow:
09467b48Spatrick    if (Idx == 1) {
09467b48Spatrick      int NumConstants = (BitSize + 63) / 64;
73471bf0Spatrick      InstructionCost Cost = AArch64TTIImpl::getIntImmCost(Imm, Ty, CostKind);
09467b48Spatrick      return (Cost <= NumConstants * TTI::TCC_Basic)
09467b48Spatrick                 ? static_cast<int>(TTI::TCC_Free)
09467b48Spatrick                 : Cost;
09467b48Spatrick    }
09467b48Spatrick    break;
09467b48Spatrick  case Intrinsic::experimental_stackmap:
09467b48Spatrick    if ((Idx < 2) || (Imm.getBitWidth() <= 64 && isInt<64>(Imm.getSExtValue())))
09467b48Spatrick      return TTI::TCC_Free;
09467b48Spatrick    break;
09467b48Spatrick  case Intrinsic::experimental_patchpoint_void:
09467b48Spatrick  case Intrinsic::experimental_patchpoint_i64:
09467b48Spatrick    if ((Idx < 4) || (Imm.getBitWidth() <= 64 && isInt<64>(Imm.getSExtValue())))
09467b48Spatrick      return TTI::TCC_Free;
09467b48Spatrick    break;
73471bf0Spatrick  case Intrinsic::experimental_gc_statepoint:
73471bf0Spatrick    if ((Idx < 5) || (Imm.getBitWidth() <= 64 && isInt<64>(Imm.getSExtValue())))
73471bf0Spatrick      return TTI::TCC_Free;
73471bf0Spatrick    break;
09467b48Spatrick  }
097a140dSpatrick  return AArch64TTIImpl::getIntImmCost(Imm, Ty, CostKind);
09467b48Spatrick}
09467b48Spatrick
09467b48SpatrickTargetTransformInfo::PopcntSupportKind
09467b48SpatrickAArch64TTIImpl::getPopcntSupport(unsigned TyWidth) {
09467b48Spatrick  assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2");
09467b48Spatrick  if (TyWidth == 32 || TyWidth == 64)
09467b48Spatrick    return TTI::PSK_FastHardware;
09467b48Spatrick  // TODO: AArch64TargetLowering::LowerCTPOP() supports 128bit popcount.
09467b48Spatrick  return TTI::PSK_Software;
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost
73471bf0SpatrickAArch64TTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA,
73471bf0Spatrick                                      TTI::TargetCostKind CostKind) {
73471bf0Spatrick  auto *RetTy = ICA.getReturnType();
73471bf0Spatrick  switch (ICA.getID()) {
73471bf0Spatrick  case Intrinsic::umin:
*d415bd75Srobert  case Intrinsic::umax:
73471bf0Spatrick  case Intrinsic::smin:
73471bf0Spatrick  case Intrinsic::smax: {
73471bf0Spatrick    static const auto ValidMinMaxTys = {MVT::v8i8,  MVT::v16i8, MVT::v4i16,
73471bf0Spatrick                                        MVT::v8i16, MVT::v2i32, MVT::v4i32};
*d415bd75Srobert    auto LT = getTypeLegalizationCost(RetTy);
*d415bd75Srobert    // v2i64 types get converted to cmp+bif hence the cost of 2
*d415bd75Srobert    if (LT.second == MVT::v2i64)
*d415bd75Srobert      return LT.first * 2;
73471bf0Spatrick    if (any_of(ValidMinMaxTys, [&LT](MVT M) { return M == LT.second; }))
73471bf0Spatrick      return LT.first;
73471bf0Spatrick    break;
73471bf0Spatrick  }
73471bf0Spatrick  case Intrinsic::sadd_sat:
73471bf0Spatrick  case Intrinsic::ssub_sat:
73471bf0Spatrick  case Intrinsic::uadd_sat:
73471bf0Spatrick  case Intrinsic::usub_sat: {
73471bf0Spatrick    static const auto ValidSatTys = {MVT::v8i8,  MVT::v16i8, MVT::v4i16,
73471bf0Spatrick                                     MVT::v8i16, MVT::v2i32, MVT::v4i32,
73471bf0Spatrick                                     MVT::v2i64};
*d415bd75Srobert    auto LT = getTypeLegalizationCost(RetTy);
73471bf0Spatrick    // This is a base cost of 1 for the vadd, plus 3 extract shifts if we
73471bf0Spatrick    // need to extend the type, as it uses shr(qadd(shl, shl)).
73471bf0Spatrick    unsigned Instrs =
73471bf0Spatrick        LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits() ? 1 : 4;
73471bf0Spatrick    if (any_of(ValidSatTys, [&LT](MVT M) { return M == LT.second; }))
73471bf0Spatrick      return LT.first * Instrs;
73471bf0Spatrick    break;
73471bf0Spatrick  }
73471bf0Spatrick  case Intrinsic::abs: {
73471bf0Spatrick    static const auto ValidAbsTys = {MVT::v8i8,  MVT::v16i8, MVT::v4i16,
73471bf0Spatrick                                     MVT::v8i16, MVT::v2i32, MVT::v4i32,
73471bf0Spatrick                                     MVT::v2i64};
*d415bd75Srobert    auto LT = getTypeLegalizationCost(RetTy);
73471bf0Spatrick    if (any_of(ValidAbsTys, [&LT](MVT M) { return M == LT.second; }))
73471bf0Spatrick      return LT.first;
73471bf0Spatrick    break;
73471bf0Spatrick  }
73471bf0Spatrick  case Intrinsic::experimental_stepvector: {
73471bf0Spatrick    InstructionCost Cost = 1; // Cost of the `index' instruction
*d415bd75Srobert    auto LT = getTypeLegalizationCost(RetTy);
73471bf0Spatrick    // Legalisation of illegal vectors involves an `index' instruction plus
73471bf0Spatrick    // (LT.first - 1) vector adds.
73471bf0Spatrick    if (LT.first > 1) {
73471bf0Spatrick      Type *LegalVTy = EVT(LT.second).getTypeForEVT(RetTy->getContext());
73471bf0Spatrick      InstructionCost AddCost =
73471bf0Spatrick          getArithmeticInstrCost(Instruction::Add, LegalVTy, CostKind);
73471bf0Spatrick      Cost += AddCost * (LT.first - 1);
73471bf0Spatrick    }
73471bf0Spatrick    return Cost;
73471bf0Spatrick  }
73471bf0Spatrick  case Intrinsic::bitreverse: {
73471bf0Spatrick    static const CostTblEntry BitreverseTbl[] = {
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::i32, 1},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::i64, 1},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::v8i8, 1},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::v16i8, 1},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::v4i16, 2},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::v8i16, 2},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::v2i32, 2},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::v4i32, 2},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::v1i64, 2},
73471bf0Spatrick        {Intrinsic::bitreverse, MVT::v2i64, 2},
73471bf0Spatrick    };
*d415bd75Srobert    const auto LegalisationCost = getTypeLegalizationCost(RetTy);
73471bf0Spatrick    const auto *Entry =
73471bf0Spatrick        CostTableLookup(BitreverseTbl, ICA.getID(), LegalisationCost.second);
*d415bd75Srobert    if (Entry) {
*d415bd75Srobert      // Cost Model is using the legal type(i32) that i8 and i16 will be
*d415bd75Srobert      // converted to +1 so that we match the actual lowering cost
73471bf0Spatrick      if (TLI->getValueType(DL, RetTy, true) == MVT::i8 ||
73471bf0Spatrick          TLI->getValueType(DL, RetTy, true) == MVT::i16)
73471bf0Spatrick        return LegalisationCost.first * Entry->Cost + 1;
*d415bd75Srobert
73471bf0Spatrick      return LegalisationCost.first * Entry->Cost;
*d415bd75Srobert    }
73471bf0Spatrick    break;
73471bf0Spatrick  }
73471bf0Spatrick  case Intrinsic::ctpop: {
*d415bd75Srobert    if (!ST->hasNEON()) {
*d415bd75Srobert      // 32-bit or 64-bit ctpop without NEON is 12 instructions.
*d415bd75Srobert      return getTypeLegalizationCost(RetTy).first * 12;
*d415bd75Srobert    }
73471bf0Spatrick    static const CostTblEntry CtpopCostTbl[] = {
73471bf0Spatrick        {ISD::CTPOP, MVT::v2i64, 4},
73471bf0Spatrick        {ISD::CTPOP, MVT::v4i32, 3},
73471bf0Spatrick        {ISD::CTPOP, MVT::v8i16, 2},
73471bf0Spatrick        {ISD::CTPOP, MVT::v16i8, 1},
73471bf0Spatrick        {ISD::CTPOP, MVT::i64,   4},
73471bf0Spatrick        {ISD::CTPOP, MVT::v2i32, 3},
73471bf0Spatrick        {ISD::CTPOP, MVT::v4i16, 2},
73471bf0Spatrick        {ISD::CTPOP, MVT::v8i8,  1},
73471bf0Spatrick        {ISD::CTPOP, MVT::i32,   5},
73471bf0Spatrick    };
*d415bd75Srobert    auto LT = getTypeLegalizationCost(RetTy);
73471bf0Spatrick    MVT MTy = LT.second;
73471bf0Spatrick    if (const auto *Entry = CostTableLookup(CtpopCostTbl, ISD::CTPOP, MTy)) {
73471bf0Spatrick      // Extra cost of +1 when illegal vector types are legalized by promoting
73471bf0Spatrick      // the integer type.
73471bf0Spatrick      int ExtraCost = MTy.isVector() && MTy.getScalarSizeInBits() !=
73471bf0Spatrick                                            RetTy->getScalarSizeInBits()
73471bf0Spatrick                          ? 1
73471bf0Spatrick                          : 0;
73471bf0Spatrick      return LT.first * Entry->Cost + ExtraCost;
73471bf0Spatrick    }
73471bf0Spatrick    break;
73471bf0Spatrick  }
*d415bd75Srobert  case Intrinsic::sadd_with_overflow:
*d415bd75Srobert  case Intrinsic::uadd_with_overflow:
*d415bd75Srobert  case Intrinsic::ssub_with_overflow:
*d415bd75Srobert  case Intrinsic::usub_with_overflow:
*d415bd75Srobert  case Intrinsic::smul_with_overflow:
*d415bd75Srobert  case Intrinsic::umul_with_overflow: {
*d415bd75Srobert    static const CostTblEntry WithOverflowCostTbl[] = {
*d415bd75Srobert        {Intrinsic::sadd_with_overflow, MVT::i8, 3},
*d415bd75Srobert        {Intrinsic::uadd_with_overflow, MVT::i8, 3},
*d415bd75Srobert        {Intrinsic::sadd_with_overflow, MVT::i16, 3},
*d415bd75Srobert        {Intrinsic::uadd_with_overflow, MVT::i16, 3},
*d415bd75Srobert        {Intrinsic::sadd_with_overflow, MVT::i32, 1},
*d415bd75Srobert        {Intrinsic::uadd_with_overflow, MVT::i32, 1},
*d415bd75Srobert        {Intrinsic::sadd_with_overflow, MVT::i64, 1},
*d415bd75Srobert        {Intrinsic::uadd_with_overflow, MVT::i64, 1},
*d415bd75Srobert        {Intrinsic::ssub_with_overflow, MVT::i8, 3},
*d415bd75Srobert        {Intrinsic::usub_with_overflow, MVT::i8, 3},
*d415bd75Srobert        {Intrinsic::ssub_with_overflow, MVT::i16, 3},
*d415bd75Srobert        {Intrinsic::usub_with_overflow, MVT::i16, 3},
*d415bd75Srobert        {Intrinsic::ssub_with_overflow, MVT::i32, 1},
*d415bd75Srobert        {Intrinsic::usub_with_overflow, MVT::i32, 1},
*d415bd75Srobert        {Intrinsic::ssub_with_overflow, MVT::i64, 1},
*d415bd75Srobert        {Intrinsic::usub_with_overflow, MVT::i64, 1},
*d415bd75Srobert        {Intrinsic::smul_with_overflow, MVT::i8, 5},
*d415bd75Srobert        {Intrinsic::umul_with_overflow, MVT::i8, 4},
*d415bd75Srobert        {Intrinsic::smul_with_overflow, MVT::i16, 5},
*d415bd75Srobert        {Intrinsic::umul_with_overflow, MVT::i16, 4},
*d415bd75Srobert        {Intrinsic::smul_with_overflow, MVT::i32, 2}, // eg umull;tst
*d415bd75Srobert        {Intrinsic::umul_with_overflow, MVT::i32, 2}, // eg umull;cmp sxtw
*d415bd75Srobert        {Intrinsic::smul_with_overflow, MVT::i64, 3}, // eg mul;smulh;cmp
*d415bd75Srobert        {Intrinsic::umul_with_overflow, MVT::i64, 3}, // eg mul;umulh;cmp asr
*d415bd75Srobert    };
*d415bd75Srobert    EVT MTy = TLI->getValueType(DL, RetTy->getContainedType(0), true);
*d415bd75Srobert    if (MTy.isSimple())
*d415bd75Srobert      if (const auto *Entry = CostTableLookup(WithOverflowCostTbl, ICA.getID(),
*d415bd75Srobert                                              MTy.getSimpleVT()))
*d415bd75Srobert        return Entry->Cost;
*d415bd75Srobert    break;
*d415bd75Srobert  }
*d415bd75Srobert  case Intrinsic::fptosi_sat:
*d415bd75Srobert  case Intrinsic::fptoui_sat: {
*d415bd75Srobert    if (ICA.getArgTypes().empty())
*d415bd75Srobert      break;
*d415bd75Srobert    bool IsSigned = ICA.getID() == Intrinsic::fptosi_sat;
*d415bd75Srobert    auto LT = getTypeLegalizationCost(ICA.getArgTypes()[0]);
*d415bd75Srobert    EVT MTy = TLI->getValueType(DL, RetTy);
*d415bd75Srobert    // Check for the legal types, which are where the size of the input and the
*d415bd75Srobert    // output are the same, or we are using cvt f64->i32 or f32->i64.
*d415bd75Srobert    if ((LT.second == MVT::f32 || LT.second == MVT::f64 ||
*d415bd75Srobert         LT.second == MVT::v2f32 || LT.second == MVT::v4f32 ||
*d415bd75Srobert         LT.second == MVT::v2f64) &&
*d415bd75Srobert        (LT.second.getScalarSizeInBits() == MTy.getScalarSizeInBits() ||
*d415bd75Srobert         (LT.second == MVT::f64 && MTy == MVT::i32) ||
*d415bd75Srobert         (LT.second == MVT::f32 && MTy == MVT::i64)))
*d415bd75Srobert      return LT.first;
*d415bd75Srobert    // Similarly for fp16 sizes
*d415bd75Srobert    if (ST->hasFullFP16() &&
*d415bd75Srobert        ((LT.second == MVT::f16 && MTy == MVT::i32) ||
*d415bd75Srobert         ((LT.second == MVT::v4f16 || LT.second == MVT::v8f16) &&
*d415bd75Srobert          (LT.second.getScalarSizeInBits() == MTy.getScalarSizeInBits()))))
*d415bd75Srobert      return LT.first;
*d415bd75Srobert
*d415bd75Srobert    // Otherwise we use a legal convert followed by a min+max
*d415bd75Srobert    if ((LT.second.getScalarType() == MVT::f32 ||
*d415bd75Srobert         LT.second.getScalarType() == MVT::f64 ||
*d415bd75Srobert         (ST->hasFullFP16() && LT.second.getScalarType() == MVT::f16)) &&
*d415bd75Srobert        LT.second.getScalarSizeInBits() >= MTy.getScalarSizeInBits()) {
*d415bd75Srobert      Type *LegalTy =
*d415bd75Srobert          Type::getIntNTy(RetTy->getContext(), LT.second.getScalarSizeInBits());
*d415bd75Srobert      if (LT.second.isVector())
*d415bd75Srobert        LegalTy = VectorType::get(LegalTy, LT.second.getVectorElementCount());
*d415bd75Srobert      InstructionCost Cost = 1;
*d415bd75Srobert      IntrinsicCostAttributes Attrs1(IsSigned ? Intrinsic::smin : Intrinsic::umin,
*d415bd75Srobert                                    LegalTy, {LegalTy, LegalTy});
*d415bd75Srobert      Cost += getIntrinsicInstrCost(Attrs1, CostKind);
*d415bd75Srobert      IntrinsicCostAttributes Attrs2(IsSigned ? Intrinsic::smax : Intrinsic::umax,
*d415bd75Srobert                                    LegalTy, {LegalTy, LegalTy});
*d415bd75Srobert      Cost += getIntrinsicInstrCost(Attrs2, CostKind);
*d415bd75Srobert      return LT.first * Cost;
*d415bd75Srobert    }
*d415bd75Srobert    break;
*d415bd75Srobert  }
73471bf0Spatrick  default:
73471bf0Spatrick    break;
73471bf0Spatrick  }
73471bf0Spatrick  return BaseT::getIntrinsicInstrCost(ICA, CostKind);
73471bf0Spatrick}
73471bf0Spatrick
73471bf0Spatrick/// The function will remove redundant reinterprets casting in the presence
73471bf0Spatrick/// of the control flow
*d415bd75Srobertstatic std::optional<Instruction *> processPhiNode(InstCombiner &IC,
73471bf0Spatrick                                                   IntrinsicInst &II) {
73471bf0Spatrick  SmallVector<Instruction *, 32> Worklist;
73471bf0Spatrick  auto RequiredType = II.getType();
73471bf0Spatrick
73471bf0Spatrick  auto *PN = dyn_cast<PHINode>(II.getArgOperand(0));
73471bf0Spatrick  assert(PN && "Expected Phi Node!");
73471bf0Spatrick
73471bf0Spatrick  // Don't create a new Phi unless we can remove the old one.
73471bf0Spatrick  if (!PN->hasOneUse())
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  for (Value *IncValPhi : PN->incoming_values()) {
73471bf0Spatrick    auto *Reinterpret = dyn_cast<IntrinsicInst>(IncValPhi);
73471bf0Spatrick    if (!Reinterpret ||
73471bf0Spatrick        Reinterpret->getIntrinsicID() !=
73471bf0Spatrick            Intrinsic::aarch64_sve_convert_to_svbool ||
73471bf0Spatrick        RequiredType != Reinterpret->getArgOperand(0)->getType())
*d415bd75Srobert      return std::nullopt;
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  // Create the new Phi
73471bf0Spatrick  LLVMContext &Ctx = PN->getContext();
73471bf0Spatrick  IRBuilder<> Builder(Ctx);
73471bf0Spatrick  Builder.SetInsertPoint(PN);
73471bf0Spatrick  PHINode *NPN = Builder.CreatePHI(RequiredType, PN->getNumIncomingValues());
73471bf0Spatrick  Worklist.push_back(PN);
73471bf0Spatrick
73471bf0Spatrick  for (unsigned I = 0; I < PN->getNumIncomingValues(); I++) {
73471bf0Spatrick    auto *Reinterpret = cast<Instruction>(PN->getIncomingValue(I));
73471bf0Spatrick    NPN->addIncoming(Reinterpret->getOperand(0), PN->getIncomingBlock(I));
73471bf0Spatrick    Worklist.push_back(Reinterpret);
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  // Cleanup Phi Node and reinterprets
73471bf0Spatrick  return IC.replaceInstUsesWith(II, NPN);
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobert// (from_svbool (binop (to_svbool pred) (svbool_t _) (svbool_t _))))
*d415bd75Srobert// => (binop (pred) (from_svbool _) (from_svbool _))
*d415bd75Srobert//
*d415bd75Srobert// The above transformation eliminates a `to_svbool` in the predicate
*d415bd75Srobert// operand of bitwise operation `binop` by narrowing the vector width of
*d415bd75Srobert// the operation. For example, it would convert a `<vscale x 16 x i1>
*d415bd75Srobert// and` into a `<vscale x 4 x i1> and`. This is profitable because
*d415bd75Srobert// to_svbool must zero the new lanes during widening, whereas
*d415bd75Srobert// from_svbool is free.
*d415bd75Srobertstatic std::optional<Instruction *>
*d415bd75SroberttryCombineFromSVBoolBinOp(InstCombiner &IC, IntrinsicInst &II) {
*d415bd75Srobert  auto BinOp = dyn_cast<IntrinsicInst>(II.getOperand(0));
*d415bd75Srobert  if (!BinOp)
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  auto IntrinsicID = BinOp->getIntrinsicID();
*d415bd75Srobert  switch (IntrinsicID) {
*d415bd75Srobert  case Intrinsic::aarch64_sve_and_z:
*d415bd75Srobert  case Intrinsic::aarch64_sve_bic_z:
*d415bd75Srobert  case Intrinsic::aarch64_sve_eor_z:
*d415bd75Srobert  case Intrinsic::aarch64_sve_nand_z:
*d415bd75Srobert  case Intrinsic::aarch64_sve_nor_z:
*d415bd75Srobert  case Intrinsic::aarch64_sve_orn_z:
*d415bd75Srobert  case Intrinsic::aarch64_sve_orr_z:
*d415bd75Srobert    break;
*d415bd75Srobert  default:
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  auto BinOpPred = BinOp->getOperand(0);
*d415bd75Srobert  auto BinOpOp1 = BinOp->getOperand(1);
*d415bd75Srobert  auto BinOpOp2 = BinOp->getOperand(2);
*d415bd75Srobert
*d415bd75Srobert  auto PredIntr = dyn_cast<IntrinsicInst>(BinOpPred);
*d415bd75Srobert  if (!PredIntr ||
*d415bd75Srobert      PredIntr->getIntrinsicID() != Intrinsic::aarch64_sve_convert_to_svbool)
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  auto PredOp = PredIntr->getOperand(0);
*d415bd75Srobert  auto PredOpTy = cast<VectorType>(PredOp->getType());
*d415bd75Srobert  if (PredOpTy != II.getType())
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert
*d415bd75Srobert  SmallVector<Value *> NarrowedBinOpArgs = {PredOp};
*d415bd75Srobert  auto NarrowBinOpOp1 = Builder.CreateIntrinsic(
*d415bd75Srobert      Intrinsic::aarch64_sve_convert_from_svbool, {PredOpTy}, {BinOpOp1});
*d415bd75Srobert  NarrowedBinOpArgs.push_back(NarrowBinOpOp1);
*d415bd75Srobert  if (BinOpOp1 == BinOpOp2)
*d415bd75Srobert    NarrowedBinOpArgs.push_back(NarrowBinOpOp1);
*d415bd75Srobert  else
*d415bd75Srobert    NarrowedBinOpArgs.push_back(Builder.CreateIntrinsic(
*d415bd75Srobert        Intrinsic::aarch64_sve_convert_from_svbool, {PredOpTy}, {BinOpOp2}));
*d415bd75Srobert
*d415bd75Srobert  auto NarrowedBinOp =
*d415bd75Srobert      Builder.CreateIntrinsic(IntrinsicID, {PredOpTy}, NarrowedBinOpArgs);
*d415bd75Srobert  return IC.replaceInstUsesWith(II, NarrowedBinOp);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *>
*d415bd75SrobertinstCombineConvertFromSVBool(InstCombiner &IC, IntrinsicInst &II) {
73471bf0Spatrick  // If the reinterpret instruction operand is a PHI Node
73471bf0Spatrick  if (isa<PHINode>(II.getArgOperand(0)))
73471bf0Spatrick    return processPhiNode(IC, II);
73471bf0Spatrick
*d415bd75Srobert  if (auto BinOpCombine = tryCombineFromSVBoolBinOp(IC, II))
*d415bd75Srobert    return BinOpCombine;
*d415bd75Srobert
73471bf0Spatrick  SmallVector<Instruction *, 32> CandidatesForRemoval;
73471bf0Spatrick  Value *Cursor = II.getOperand(0), *EarliestReplacement = nullptr;
73471bf0Spatrick
73471bf0Spatrick  const auto *IVTy = cast<VectorType>(II.getType());
73471bf0Spatrick
73471bf0Spatrick  // Walk the chain of conversions.
73471bf0Spatrick  while (Cursor) {
73471bf0Spatrick    // If the type of the cursor has fewer lanes than the final result, zeroing
73471bf0Spatrick    // must take place, which breaks the equivalence chain.
73471bf0Spatrick    const auto *CursorVTy = cast<VectorType>(Cursor->getType());
73471bf0Spatrick    if (CursorVTy->getElementCount().getKnownMinValue() <
73471bf0Spatrick        IVTy->getElementCount().getKnownMinValue())
73471bf0Spatrick      break;
73471bf0Spatrick
73471bf0Spatrick    // If the cursor has the same type as I, it is a viable replacement.
73471bf0Spatrick    if (Cursor->getType() == IVTy)
73471bf0Spatrick      EarliestReplacement = Cursor;
73471bf0Spatrick
73471bf0Spatrick    auto *IntrinsicCursor = dyn_cast<IntrinsicInst>(Cursor);
73471bf0Spatrick
73471bf0Spatrick    // If this is not an SVE conversion intrinsic, this is the end of the chain.
73471bf0Spatrick    if (!IntrinsicCursor || !(IntrinsicCursor->getIntrinsicID() ==
73471bf0Spatrick                                  Intrinsic::aarch64_sve_convert_to_svbool ||
73471bf0Spatrick                              IntrinsicCursor->getIntrinsicID() ==
73471bf0Spatrick                                  Intrinsic::aarch64_sve_convert_from_svbool))
73471bf0Spatrick      break;
73471bf0Spatrick
73471bf0Spatrick    CandidatesForRemoval.insert(CandidatesForRemoval.begin(), IntrinsicCursor);
73471bf0Spatrick    Cursor = IntrinsicCursor->getOperand(0);
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  // If no viable replacement in the conversion chain was found, there is
73471bf0Spatrick  // nothing to do.
73471bf0Spatrick  if (!EarliestReplacement)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  return IC.replaceInstUsesWith(II, EarliestReplacement);
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVESel(InstCombiner &IC,
*d415bd75Srobert                                                      IntrinsicInst &II) {
*d415bd75Srobert  IRBuilder<> Builder(&II);
*d415bd75Srobert  auto Select = Builder.CreateSelect(II.getOperand(0), II.getOperand(1),
*d415bd75Srobert                                     II.getOperand(2));
*d415bd75Srobert  return IC.replaceInstUsesWith(II, Select);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEDup(InstCombiner &IC,
73471bf0Spatrick                                                      IntrinsicInst &II) {
73471bf0Spatrick  IntrinsicInst *Pg = dyn_cast<IntrinsicInst>(II.getArgOperand(1));
73471bf0Spatrick  if (!Pg)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  if (Pg->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  const auto PTruePattern =
73471bf0Spatrick      cast<ConstantInt>(Pg->getOperand(0))->getZExtValue();
73471bf0Spatrick  if (PTruePattern != AArch64SVEPredPattern::vl1)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  // The intrinsic is inserting into lane zero so use an insert instead.
73471bf0Spatrick  auto *IdxTy = Type::getInt64Ty(II.getContext());
73471bf0Spatrick  auto *Insert = InsertElementInst::Create(
73471bf0Spatrick      II.getArgOperand(0), II.getArgOperand(2), ConstantInt::get(IdxTy, 0));
73471bf0Spatrick  Insert->insertBefore(&II);
73471bf0Spatrick  Insert->takeName(&II);
73471bf0Spatrick
73471bf0Spatrick  return IC.replaceInstUsesWith(II, Insert);
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEDupX(InstCombiner &IC,
*d415bd75Srobert                                                       IntrinsicInst &II) {
*d415bd75Srobert  // Replace DupX with a regular IR splat.
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert  auto *RetTy = cast<ScalableVectorType>(II.getType());
*d415bd75Srobert  Value *Splat =
*d415bd75Srobert      Builder.CreateVectorSplat(RetTy->getElementCount(), II.getArgOperand(0));
*d415bd75Srobert  Splat->takeName(&II);
*d415bd75Srobert  return IC.replaceInstUsesWith(II, Splat);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVECmpNE(InstCombiner &IC,
73471bf0Spatrick                                                        IntrinsicInst &II) {
73471bf0Spatrick  LLVMContext &Ctx = II.getContext();
73471bf0Spatrick  IRBuilder<> Builder(Ctx);
73471bf0Spatrick  Builder.SetInsertPoint(&II);
73471bf0Spatrick
73471bf0Spatrick  // Check that the predicate is all active
73471bf0Spatrick  auto *Pg = dyn_cast<IntrinsicInst>(II.getArgOperand(0));
73471bf0Spatrick  if (!Pg || Pg->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  const auto PTruePattern =
73471bf0Spatrick      cast<ConstantInt>(Pg->getOperand(0))->getZExtValue();
73471bf0Spatrick  if (PTruePattern != AArch64SVEPredPattern::all)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  // Check that we have a compare of zero..
*d415bd75Srobert  auto *SplatValue =
*d415bd75Srobert      dyn_cast_or_null<ConstantInt>(getSplatValue(II.getArgOperand(2)));
*d415bd75Srobert  if (!SplatValue || !SplatValue->isZero())
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  // ..against a dupq
73471bf0Spatrick  auto *DupQLane = dyn_cast<IntrinsicInst>(II.getArgOperand(1));
73471bf0Spatrick  if (!DupQLane ||
73471bf0Spatrick      DupQLane->getIntrinsicID() != Intrinsic::aarch64_sve_dupq_lane)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  // Where the dupq is a lane 0 replicate of a vector insert
73471bf0Spatrick  if (!cast<ConstantInt>(DupQLane->getArgOperand(1))->isZero())
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  auto *VecIns = dyn_cast<IntrinsicInst>(DupQLane->getArgOperand(0));
*d415bd75Srobert  if (!VecIns || VecIns->getIntrinsicID() != Intrinsic::vector_insert)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  // Where the vector insert is a fixed constant vector insert into undef at
73471bf0Spatrick  // index zero
73471bf0Spatrick  if (!isa<UndefValue>(VecIns->getArgOperand(0)))
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  if (!cast<ConstantInt>(VecIns->getArgOperand(2))->isZero())
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  auto *ConstVec = dyn_cast<Constant>(VecIns->getArgOperand(1));
73471bf0Spatrick  if (!ConstVec)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  auto *VecTy = dyn_cast<FixedVectorType>(ConstVec->getType());
73471bf0Spatrick  auto *OutTy = dyn_cast<ScalableVectorType>(II.getType());
73471bf0Spatrick  if (!VecTy || !OutTy || VecTy->getNumElements() != OutTy->getMinNumElements())
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  unsigned NumElts = VecTy->getNumElements();
73471bf0Spatrick  unsigned PredicateBits = 0;
73471bf0Spatrick
73471bf0Spatrick  // Expand intrinsic operands to a 16-bit byte level predicate
73471bf0Spatrick  for (unsigned I = 0; I < NumElts; ++I) {
73471bf0Spatrick    auto *Arg = dyn_cast<ConstantInt>(ConstVec->getAggregateElement(I));
73471bf0Spatrick    if (!Arg)
*d415bd75Srobert      return std::nullopt;
73471bf0Spatrick    if (!Arg->isZero())
73471bf0Spatrick      PredicateBits |= 1 << (I * (16 / NumElts));
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  // If all bits are zero bail early with an empty predicate
73471bf0Spatrick  if (PredicateBits == 0) {
73471bf0Spatrick    auto *PFalse = Constant::getNullValue(II.getType());
73471bf0Spatrick    PFalse->takeName(&II);
73471bf0Spatrick    return IC.replaceInstUsesWith(II, PFalse);
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  // Calculate largest predicate type used (where byte predicate is largest)
73471bf0Spatrick  unsigned Mask = 8;
73471bf0Spatrick  for (unsigned I = 0; I < 16; ++I)
73471bf0Spatrick    if ((PredicateBits & (1 << I)) != 0)
73471bf0Spatrick      Mask |= (I % 8);
73471bf0Spatrick
73471bf0Spatrick  unsigned PredSize = Mask & -Mask;
73471bf0Spatrick  auto *PredType = ScalableVectorType::get(
73471bf0Spatrick      Type::getInt1Ty(Ctx), AArch64::SVEBitsPerBlock / (PredSize * 8));
73471bf0Spatrick
73471bf0Spatrick  // Ensure all relevant bits are set
73471bf0Spatrick  for (unsigned I = 0; I < 16; I += PredSize)
73471bf0Spatrick    if ((PredicateBits & (1 << I)) == 0)
*d415bd75Srobert      return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  auto *PTruePat =
73471bf0Spatrick      ConstantInt::get(Type::getInt32Ty(Ctx), AArch64SVEPredPattern::all);
73471bf0Spatrick  auto *PTrue = Builder.CreateIntrinsic(Intrinsic::aarch64_sve_ptrue,
73471bf0Spatrick                                        {PredType}, {PTruePat});
73471bf0Spatrick  auto *ConvertToSVBool = Builder.CreateIntrinsic(
73471bf0Spatrick      Intrinsic::aarch64_sve_convert_to_svbool, {PredType}, {PTrue});
73471bf0Spatrick  auto *ConvertFromSVBool =
73471bf0Spatrick      Builder.CreateIntrinsic(Intrinsic::aarch64_sve_convert_from_svbool,
73471bf0Spatrick                              {II.getType()}, {ConvertToSVBool});
73471bf0Spatrick
73471bf0Spatrick  ConvertFromSVBool->takeName(&II);
73471bf0Spatrick  return IC.replaceInstUsesWith(II, ConvertFromSVBool);
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVELast(InstCombiner &IC,
73471bf0Spatrick                                                       IntrinsicInst &II) {
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
73471bf0Spatrick  Value *Pg = II.getArgOperand(0);
73471bf0Spatrick  Value *Vec = II.getArgOperand(1);
*d415bd75Srobert  auto IntrinsicID = II.getIntrinsicID();
*d415bd75Srobert  bool IsAfter = IntrinsicID == Intrinsic::aarch64_sve_lasta;
73471bf0Spatrick
73471bf0Spatrick  // lastX(splat(X)) --> X
73471bf0Spatrick  if (auto *SplatVal = getSplatValue(Vec))
73471bf0Spatrick    return IC.replaceInstUsesWith(II, SplatVal);
73471bf0Spatrick
*d415bd75Srobert  // If x and/or y is a splat value then:
*d415bd75Srobert  // lastX (binop (x, y)) --> binop(lastX(x), lastX(y))
*d415bd75Srobert  Value *LHS, *RHS;
*d415bd75Srobert  if (match(Vec, m_OneUse(m_BinOp(m_Value(LHS), m_Value(RHS))))) {
*d415bd75Srobert    if (isSplatValue(LHS) || isSplatValue(RHS)) {
*d415bd75Srobert      auto *OldBinOp = cast<BinaryOperator>(Vec);
*d415bd75Srobert      auto OpC = OldBinOp->getOpcode();
*d415bd75Srobert      auto *NewLHS =
*d415bd75Srobert          Builder.CreateIntrinsic(IntrinsicID, {Vec->getType()}, {Pg, LHS});
*d415bd75Srobert      auto *NewRHS =
*d415bd75Srobert          Builder.CreateIntrinsic(IntrinsicID, {Vec->getType()}, {Pg, RHS});
*d415bd75Srobert      auto *NewBinOp = BinaryOperator::CreateWithCopiedFlags(
*d415bd75Srobert          OpC, NewLHS, NewRHS, OldBinOp, OldBinOp->getName(), &II);
*d415bd75Srobert      return IC.replaceInstUsesWith(II, NewBinOp);
*d415bd75Srobert    }
*d415bd75Srobert  }
*d415bd75Srobert
73471bf0Spatrick  auto *C = dyn_cast<Constant>(Pg);
73471bf0Spatrick  if (IsAfter && C && C->isNullValue()) {
73471bf0Spatrick    // The intrinsic is extracting lane 0 so use an extract instead.
73471bf0Spatrick    auto *IdxTy = Type::getInt64Ty(II.getContext());
73471bf0Spatrick    auto *Extract = ExtractElementInst::Create(Vec, ConstantInt::get(IdxTy, 0));
73471bf0Spatrick    Extract->insertBefore(&II);
73471bf0Spatrick    Extract->takeName(&II);
73471bf0Spatrick    return IC.replaceInstUsesWith(II, Extract);
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  auto *IntrPG = dyn_cast<IntrinsicInst>(Pg);
73471bf0Spatrick  if (!IntrPG)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  if (IntrPG->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  const auto PTruePattern =
73471bf0Spatrick      cast<ConstantInt>(IntrPG->getOperand(0))->getZExtValue();
73471bf0Spatrick
73471bf0Spatrick  // Can the intrinsic's predicate be converted to a known constant index?
*d415bd75Srobert  unsigned MinNumElts = getNumElementsFromSVEPredPattern(PTruePattern);
*d415bd75Srobert  if (!MinNumElts)
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
*d415bd75Srobert  unsigned Idx = MinNumElts - 1;
73471bf0Spatrick  // Increment the index if extracting the element after the last active
73471bf0Spatrick  // predicate element.
73471bf0Spatrick  if (IsAfter)
73471bf0Spatrick    ++Idx;
73471bf0Spatrick
73471bf0Spatrick  // Ignore extracts whose index is larger than the known minimum vector
73471bf0Spatrick  // length. NOTE: This is an artificial constraint where we prefer to
73471bf0Spatrick  // maintain what the user asked for until an alternative is proven faster.
73471bf0Spatrick  auto *PgVTy = cast<ScalableVectorType>(Pg->getType());
73471bf0Spatrick  if (Idx >= PgVTy->getMinNumElements())
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  // The intrinsic is extracting a fixed lane so use an extract instead.
73471bf0Spatrick  auto *IdxTy = Type::getInt64Ty(II.getContext());
73471bf0Spatrick  auto *Extract = ExtractElementInst::Create(Vec, ConstantInt::get(IdxTy, Idx));
73471bf0Spatrick  Extract->insertBefore(&II);
73471bf0Spatrick  Extract->takeName(&II);
73471bf0Spatrick  return IC.replaceInstUsesWith(II, Extract);
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVECondLast(InstCombiner &IC,
*d415bd75Srobert                                                           IntrinsicInst &II) {
*d415bd75Srobert  // The SIMD&FP variant of CLAST[AB] is significantly faster than the scalar
*d415bd75Srobert  // integer variant across a variety of micro-architectures. Replace scalar
*d415bd75Srobert  // integer CLAST[AB] intrinsic with optimal SIMD&FP variant. A simple
*d415bd75Srobert  // bitcast-to-fp + clast[ab] + bitcast-to-int will cost a cycle or two more
*d415bd75Srobert  // depending on the micro-architecture, but has been observed as generally
*d415bd75Srobert  // being faster, particularly when the CLAST[AB] op is a loop-carried
*d415bd75Srobert  // dependency.
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert  Value *Pg = II.getArgOperand(0);
*d415bd75Srobert  Value *Fallback = II.getArgOperand(1);
*d415bd75Srobert  Value *Vec = II.getArgOperand(2);
*d415bd75Srobert  Type *Ty = II.getType();
*d415bd75Srobert
*d415bd75Srobert  if (!Ty->isIntegerTy())
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  Type *FPTy;
*d415bd75Srobert  switch (cast<IntegerType>(Ty)->getBitWidth()) {
*d415bd75Srobert  default:
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert  case 16:
*d415bd75Srobert    FPTy = Builder.getHalfTy();
*d415bd75Srobert    break;
*d415bd75Srobert  case 32:
*d415bd75Srobert    FPTy = Builder.getFloatTy();
*d415bd75Srobert    break;
*d415bd75Srobert  case 64:
*d415bd75Srobert    FPTy = Builder.getDoubleTy();
*d415bd75Srobert    break;
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  Value *FPFallBack = Builder.CreateBitCast(Fallback, FPTy);
*d415bd75Srobert  auto *FPVTy = VectorType::get(
*d415bd75Srobert      FPTy, cast<VectorType>(Vec->getType())->getElementCount());
*d415bd75Srobert  Value *FPVec = Builder.CreateBitCast(Vec, FPVTy);
*d415bd75Srobert  auto *FPII = Builder.CreateIntrinsic(II.getIntrinsicID(), {FPVec->getType()},
*d415bd75Srobert                                       {Pg, FPFallBack, FPVec});
*d415bd75Srobert  Value *FPIItoInt = Builder.CreateBitCast(FPII, II.getType());
*d415bd75Srobert  return IC.replaceInstUsesWith(II, FPIItoInt);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineRDFFR(InstCombiner &IC,
73471bf0Spatrick                                                     IntrinsicInst &II) {
73471bf0Spatrick  LLVMContext &Ctx = II.getContext();
73471bf0Spatrick  IRBuilder<> Builder(Ctx);
73471bf0Spatrick  Builder.SetInsertPoint(&II);
73471bf0Spatrick  // Replace rdffr with predicated rdffr.z intrinsic, so that optimizePTestInstr
73471bf0Spatrick  // can work with RDFFR_PP for ptest elimination.
73471bf0Spatrick  auto *AllPat =
73471bf0Spatrick      ConstantInt::get(Type::getInt32Ty(Ctx), AArch64SVEPredPattern::all);
73471bf0Spatrick  auto *PTrue = Builder.CreateIntrinsic(Intrinsic::aarch64_sve_ptrue,
73471bf0Spatrick                                        {II.getType()}, {AllPat});
73471bf0Spatrick  auto *RDFFR =
73471bf0Spatrick      Builder.CreateIntrinsic(Intrinsic::aarch64_sve_rdffr_z, {}, {PTrue});
73471bf0Spatrick  RDFFR->takeName(&II);
73471bf0Spatrick  return IC.replaceInstUsesWith(II, RDFFR);
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobertstatic std::optional<Instruction *>
73471bf0SpatrickinstCombineSVECntElts(InstCombiner &IC, IntrinsicInst &II, unsigned NumElts) {
73471bf0Spatrick  const auto Pattern = cast<ConstantInt>(II.getArgOperand(0))->getZExtValue();
73471bf0Spatrick
73471bf0Spatrick  if (Pattern == AArch64SVEPredPattern::all) {
73471bf0Spatrick    LLVMContext &Ctx = II.getContext();
73471bf0Spatrick    IRBuilder<> Builder(Ctx);
73471bf0Spatrick    Builder.SetInsertPoint(&II);
73471bf0Spatrick
73471bf0Spatrick    Constant *StepVal = ConstantInt::get(II.getType(), NumElts);
73471bf0Spatrick    auto *VScale = Builder.CreateVScale(StepVal);
73471bf0Spatrick    VScale->takeName(&II);
73471bf0Spatrick    return IC.replaceInstUsesWith(II, VScale);
73471bf0Spatrick  }
73471bf0Spatrick
*d415bd75Srobert  unsigned MinNumElts = getNumElementsFromSVEPredPattern(Pattern);
73471bf0Spatrick
*d415bd75Srobert  return MinNumElts && NumElts >= MinNumElts
*d415bd75Srobert             ? std::optional<Instruction *>(IC.replaceInstUsesWith(
73471bf0Spatrick                   II, ConstantInt::get(II.getType(), MinNumElts)))
*d415bd75Srobert             : std::nullopt;
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEPTest(InstCombiner &IC,
73471bf0Spatrick                                                        IntrinsicInst &II) {
*d415bd75Srobert  Value *PgVal = II.getArgOperand(0);
*d415bd75Srobert  Value *OpVal = II.getArgOperand(1);
73471bf0Spatrick
73471bf0Spatrick  IRBuilder<> Builder(II.getContext());
73471bf0Spatrick  Builder.SetInsertPoint(&II);
73471bf0Spatrick
*d415bd75Srobert  // PTEST_<FIRST|LAST>(X, X) is equivalent to PTEST_ANY(X, X).
*d415bd75Srobert  // Later optimizations prefer this form.
*d415bd75Srobert  if (PgVal == OpVal &&
*d415bd75Srobert      (II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_first ||
*d415bd75Srobert       II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_last)) {
*d415bd75Srobert    Value *Ops[] = {PgVal, OpVal};
*d415bd75Srobert    Type *Tys[] = {PgVal->getType()};
*d415bd75Srobert
*d415bd75Srobert    auto *PTest =
*d415bd75Srobert        Builder.CreateIntrinsic(Intrinsic::aarch64_sve_ptest_any, Tys, Ops);
*d415bd75Srobert    PTest->takeName(&II);
*d415bd75Srobert
*d415bd75Srobert    return IC.replaceInstUsesWith(II, PTest);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  IntrinsicInst *Pg = dyn_cast<IntrinsicInst>(PgVal);
*d415bd75Srobert  IntrinsicInst *Op = dyn_cast<IntrinsicInst>(OpVal);
*d415bd75Srobert
*d415bd75Srobert  if (!Pg || !Op)
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  Intrinsic::ID OpIID = Op->getIntrinsicID();
*d415bd75Srobert
*d415bd75Srobert  if (Pg->getIntrinsicID() == Intrinsic::aarch64_sve_convert_to_svbool &&
*d415bd75Srobert      OpIID == Intrinsic::aarch64_sve_convert_to_svbool &&
*d415bd75Srobert      Pg->getArgOperand(0)->getType() == Op->getArgOperand(0)->getType()) {
*d415bd75Srobert    Value *Ops[] = {Pg->getArgOperand(0), Op->getArgOperand(0)};
*d415bd75Srobert    Type *Tys[] = {Pg->getArgOperand(0)->getType()};
73471bf0Spatrick
73471bf0Spatrick    auto *PTest = Builder.CreateIntrinsic(II.getIntrinsicID(), Tys, Ops);
73471bf0Spatrick
73471bf0Spatrick    PTest->takeName(&II);
73471bf0Spatrick    return IC.replaceInstUsesWith(II, PTest);
73471bf0Spatrick  }
73471bf0Spatrick
*d415bd75Srobert  // Transform PTEST_ANY(X=OP(PG,...), X) -> PTEST_ANY(PG, X)).
*d415bd75Srobert  // Later optimizations may rewrite sequence to use the flag-setting variant
*d415bd75Srobert  // of instruction X to remove PTEST.
*d415bd75Srobert  if ((Pg == Op) && (II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_any) &&
*d415bd75Srobert      ((OpIID == Intrinsic::aarch64_sve_brka_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_brkb_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_brkpa_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_brkpb_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_rdffr_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_and_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_bic_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_eor_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_nand_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_nor_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_orn_z) ||
*d415bd75Srobert       (OpIID == Intrinsic::aarch64_sve_orr_z))) {
*d415bd75Srobert    Value *Ops[] = {Pg->getArgOperand(0), Pg};
*d415bd75Srobert    Type *Tys[] = {Pg->getType()};
*d415bd75Srobert
*d415bd75Srobert    auto *PTest = Builder.CreateIntrinsic(II.getIntrinsicID(), Tys, Ops);
*d415bd75Srobert    PTest->takeName(&II);
*d415bd75Srobert
*d415bd75Srobert    return IC.replaceInstUsesWith(II, PTest);
73471bf0Spatrick  }
73471bf0Spatrick
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Sroberttemplate <Intrinsic::ID MulOpc, typename Intrinsic::ID FuseOpc>
*d415bd75Srobertstatic std::optional<Instruction *>
*d415bd75SrobertinstCombineSVEVectorFuseMulAddSub(InstCombiner &IC, IntrinsicInst &II,
*d415bd75Srobert                                  bool MergeIntoAddendOp) {
*d415bd75Srobert  Value *P = II.getOperand(0);
*d415bd75Srobert  Value *MulOp0, *MulOp1, *AddendOp, *Mul;
*d415bd75Srobert  if (MergeIntoAddendOp) {
*d415bd75Srobert    AddendOp = II.getOperand(1);
*d415bd75Srobert    Mul = II.getOperand(2);
*d415bd75Srobert  } else {
*d415bd75Srobert    AddendOp = II.getOperand(2);
*d415bd75Srobert    Mul = II.getOperand(1);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  if (!match(Mul, m_Intrinsic<MulOpc>(m_Specific(P), m_Value(MulOp0),
*d415bd75Srobert                                      m_Value(MulOp1))))
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  if (!Mul->hasOneUse())
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  Instruction *FMFSource = nullptr;
*d415bd75Srobert  if (II.getType()->isFPOrFPVectorTy()) {
*d415bd75Srobert    llvm::FastMathFlags FAddFlags = II.getFastMathFlags();
*d415bd75Srobert    // Stop the combine when the flags on the inputs differ in case dropping
*d415bd75Srobert    // flags would lead to us missing out on more beneficial optimizations.
*d415bd75Srobert    if (FAddFlags != cast<CallInst>(Mul)->getFastMathFlags())
*d415bd75Srobert      return std::nullopt;
*d415bd75Srobert    if (!FAddFlags.allowContract())
*d415bd75Srobert      return std::nullopt;
*d415bd75Srobert    FMFSource = &II;
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert
*d415bd75Srobert  CallInst *Res;
*d415bd75Srobert  if (MergeIntoAddendOp)
*d415bd75Srobert    Res = Builder.CreateIntrinsic(FuseOpc, {II.getType()},
*d415bd75Srobert                                  {P, AddendOp, MulOp0, MulOp1}, FMFSource);
*d415bd75Srobert  else
*d415bd75Srobert    Res = Builder.CreateIntrinsic(FuseOpc, {II.getType()},
*d415bd75Srobert                                  {P, MulOp0, MulOp1, AddendOp}, FMFSource);
*d415bd75Srobert
*d415bd75Srobert  return IC.replaceInstUsesWith(II, Res);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic bool isAllActivePredicate(Value *Pred) {
*d415bd75Srobert  // Look through convert.from.svbool(convert.to.svbool(...) chain.
*d415bd75Srobert  Value *UncastedPred;
*d415bd75Srobert  if (match(Pred, m_Intrinsic<Intrinsic::aarch64_sve_convert_from_svbool>(
*d415bd75Srobert                      m_Intrinsic<Intrinsic::aarch64_sve_convert_to_svbool>(
*d415bd75Srobert                          m_Value(UncastedPred)))))
*d415bd75Srobert    // If the predicate has the same or less lanes than the uncasted
*d415bd75Srobert    // predicate then we know the casting has no effect.
*d415bd75Srobert    if (cast<ScalableVectorType>(Pred->getType())->getMinNumElements() <=
*d415bd75Srobert        cast<ScalableVectorType>(UncastedPred->getType())->getMinNumElements())
*d415bd75Srobert      Pred = UncastedPred;
*d415bd75Srobert
*d415bd75Srobert  return match(Pred, m_Intrinsic<Intrinsic::aarch64_sve_ptrue>(
*d415bd75Srobert                         m_ConstantInt<AArch64SVEPredPattern::all>()));
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *>
*d415bd75SrobertinstCombineSVELD1(InstCombiner &IC, IntrinsicInst &II, const DataLayout &DL) {
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert
*d415bd75Srobert  Value *Pred = II.getOperand(0);
*d415bd75Srobert  Value *PtrOp = II.getOperand(1);
*d415bd75Srobert  Type *VecTy = II.getType();
*d415bd75Srobert  Value *VecPtr = Builder.CreateBitCast(PtrOp, VecTy->getPointerTo());
*d415bd75Srobert
*d415bd75Srobert  if (isAllActivePredicate(Pred)) {
*d415bd75Srobert    LoadInst *Load = Builder.CreateLoad(VecTy, VecPtr);
*d415bd75Srobert    Load->copyMetadata(II);
*d415bd75Srobert    return IC.replaceInstUsesWith(II, Load);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  CallInst *MaskedLoad =
*d415bd75Srobert      Builder.CreateMaskedLoad(VecTy, VecPtr, PtrOp->getPointerAlignment(DL),
*d415bd75Srobert                               Pred, ConstantAggregateZero::get(VecTy));
*d415bd75Srobert  MaskedLoad->copyMetadata(II);
*d415bd75Srobert  return IC.replaceInstUsesWith(II, MaskedLoad);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *>
*d415bd75SrobertinstCombineSVEST1(InstCombiner &IC, IntrinsicInst &II, const DataLayout &DL) {
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert
*d415bd75Srobert  Value *VecOp = II.getOperand(0);
*d415bd75Srobert  Value *Pred = II.getOperand(1);
*d415bd75Srobert  Value *PtrOp = II.getOperand(2);
*d415bd75Srobert  Value *VecPtr =
*d415bd75Srobert      Builder.CreateBitCast(PtrOp, VecOp->getType()->getPointerTo());
*d415bd75Srobert
*d415bd75Srobert  if (isAllActivePredicate(Pred)) {
*d415bd75Srobert    StoreInst *Store = Builder.CreateStore(VecOp, VecPtr);
*d415bd75Srobert    Store->copyMetadata(II);
*d415bd75Srobert    return IC.eraseInstFromFunction(II);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  CallInst *MaskedStore = Builder.CreateMaskedStore(
*d415bd75Srobert      VecOp, VecPtr, PtrOp->getPointerAlignment(DL), Pred);
*d415bd75Srobert  MaskedStore->copyMetadata(II);
*d415bd75Srobert  return IC.eraseInstFromFunction(II);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic Instruction::BinaryOps intrinsicIDToBinOpCode(unsigned Intrinsic) {
*d415bd75Srobert  switch (Intrinsic) {
*d415bd75Srobert  case Intrinsic::aarch64_sve_fmul:
*d415bd75Srobert    return Instruction::BinaryOps::FMul;
*d415bd75Srobert  case Intrinsic::aarch64_sve_fadd:
*d415bd75Srobert    return Instruction::BinaryOps::FAdd;
*d415bd75Srobert  case Intrinsic::aarch64_sve_fsub:
*d415bd75Srobert    return Instruction::BinaryOps::FSub;
*d415bd75Srobert  default:
*d415bd75Srobert    return Instruction::BinaryOpsEnd;
*d415bd75Srobert  }
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *>
*d415bd75SrobertinstCombineSVEVectorBinOp(InstCombiner &IC, IntrinsicInst &II) {
*d415bd75Srobert  auto *OpPredicate = II.getOperand(0);
*d415bd75Srobert  auto BinOpCode = intrinsicIDToBinOpCode(II.getIntrinsicID());
*d415bd75Srobert  if (BinOpCode == Instruction::BinaryOpsEnd ||
*d415bd75Srobert      !match(OpPredicate, m_Intrinsic<Intrinsic::aarch64_sve_ptrue>(
*d415bd75Srobert                              m_ConstantInt<AArch64SVEPredPattern::all>())))
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert  Builder.setFastMathFlags(II.getFastMathFlags());
*d415bd75Srobert  auto BinOp =
*d415bd75Srobert      Builder.CreateBinOp(BinOpCode, II.getOperand(1), II.getOperand(2));
*d415bd75Srobert  return IC.replaceInstUsesWith(II, BinOp);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEVectorAdd(InstCombiner &IC,
*d415bd75Srobert                                                            IntrinsicInst &II) {
*d415bd75Srobert  if (auto FMLA =
*d415bd75Srobert          instCombineSVEVectorFuseMulAddSub<Intrinsic::aarch64_sve_fmul,
*d415bd75Srobert                                            Intrinsic::aarch64_sve_fmla>(IC, II,
*d415bd75Srobert                                                                         true))
*d415bd75Srobert    return FMLA;
*d415bd75Srobert  if (auto MLA = instCombineSVEVectorFuseMulAddSub<Intrinsic::aarch64_sve_mul,
*d415bd75Srobert                                                   Intrinsic::aarch64_sve_mla>(
*d415bd75Srobert          IC, II, true))
*d415bd75Srobert    return MLA;
*d415bd75Srobert  if (auto FMAD =
*d415bd75Srobert          instCombineSVEVectorFuseMulAddSub<Intrinsic::aarch64_sve_fmul,
*d415bd75Srobert                                            Intrinsic::aarch64_sve_fmad>(IC, II,
*d415bd75Srobert                                                                         false))
*d415bd75Srobert    return FMAD;
*d415bd75Srobert  if (auto MAD = instCombineSVEVectorFuseMulAddSub<Intrinsic::aarch64_sve_mul,
*d415bd75Srobert                                                   Intrinsic::aarch64_sve_mad>(
*d415bd75Srobert          IC, II, false))
*d415bd75Srobert    return MAD;
*d415bd75Srobert  return instCombineSVEVectorBinOp(IC, II);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEVectorSub(InstCombiner &IC,
*d415bd75Srobert                                                            IntrinsicInst &II) {
*d415bd75Srobert  if (auto FMLS =
*d415bd75Srobert          instCombineSVEVectorFuseMulAddSub<Intrinsic::aarch64_sve_fmul,
*d415bd75Srobert                                            Intrinsic::aarch64_sve_fmls>(IC, II,
*d415bd75Srobert                                                                         true))
*d415bd75Srobert    return FMLS;
*d415bd75Srobert  if (auto MLS = instCombineSVEVectorFuseMulAddSub<Intrinsic::aarch64_sve_mul,
*d415bd75Srobert                                                   Intrinsic::aarch64_sve_mls>(
*d415bd75Srobert          IC, II, true))
*d415bd75Srobert    return MLS;
*d415bd75Srobert  if (auto FMSB =
*d415bd75Srobert          instCombineSVEVectorFuseMulAddSub<Intrinsic::aarch64_sve_fmul,
*d415bd75Srobert                                            Intrinsic::aarch64_sve_fnmsb>(
*d415bd75Srobert              IC, II, false))
*d415bd75Srobert    return FMSB;
*d415bd75Srobert  return instCombineSVEVectorBinOp(IC, II);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEVectorMul(InstCombiner &IC,
73471bf0Spatrick                                                            IntrinsicInst &II) {
73471bf0Spatrick  auto *OpPredicate = II.getOperand(0);
73471bf0Spatrick  auto *OpMultiplicand = II.getOperand(1);
73471bf0Spatrick  auto *OpMultiplier = II.getOperand(2);
73471bf0Spatrick
73471bf0Spatrick  IRBuilder<> Builder(II.getContext());
73471bf0Spatrick  Builder.SetInsertPoint(&II);
73471bf0Spatrick
*d415bd75Srobert  // Return true if a given instruction is a unit splat value, false otherwise.
*d415bd75Srobert  auto IsUnitSplat = [](auto *I) {
*d415bd75Srobert    auto *SplatValue = getSplatValue(I);
*d415bd75Srobert    if (!SplatValue)
73471bf0Spatrick      return false;
73471bf0Spatrick    return match(SplatValue, m_FPOne()) || match(SplatValue, m_One());
73471bf0Spatrick  };
73471bf0Spatrick
73471bf0Spatrick  // Return true if a given instruction is an aarch64_sve_dup intrinsic call
73471bf0Spatrick  // with a unit splat value, false otherwise.
73471bf0Spatrick  auto IsUnitDup = [](auto *I) {
73471bf0Spatrick    auto *IntrI = dyn_cast<IntrinsicInst>(I);
73471bf0Spatrick    if (!IntrI || IntrI->getIntrinsicID() != Intrinsic::aarch64_sve_dup)
73471bf0Spatrick      return false;
73471bf0Spatrick
73471bf0Spatrick    auto *SplatValue = IntrI->getOperand(2);
73471bf0Spatrick    return match(SplatValue, m_FPOne()) || match(SplatValue, m_One());
73471bf0Spatrick  };
73471bf0Spatrick
*d415bd75Srobert  if (IsUnitSplat(OpMultiplier)) {
*d415bd75Srobert    // [f]mul pg %n, (dupx 1) => %n
73471bf0Spatrick    OpMultiplicand->takeName(&II);
73471bf0Spatrick    return IC.replaceInstUsesWith(II, OpMultiplicand);
73471bf0Spatrick  } else if (IsUnitDup(OpMultiplier)) {
*d415bd75Srobert    // [f]mul pg %n, (dup pg 1) => %n
73471bf0Spatrick    auto *DupInst = cast<IntrinsicInst>(OpMultiplier);
73471bf0Spatrick    auto *DupPg = DupInst->getOperand(1);
73471bf0Spatrick    // TODO: this is naive. The optimization is still valid if DupPg
73471bf0Spatrick    // 'encompasses' OpPredicate, not only if they're the same predicate.
73471bf0Spatrick    if (OpPredicate == DupPg) {
73471bf0Spatrick      OpMultiplicand->takeName(&II);
73471bf0Spatrick      return IC.replaceInstUsesWith(II, OpMultiplicand);
73471bf0Spatrick    }
73471bf0Spatrick  }
73471bf0Spatrick
*d415bd75Srobert  return instCombineSVEVectorBinOp(IC, II);
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEUnpack(InstCombiner &IC,
*d415bd75Srobert                                                         IntrinsicInst &II) {
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert  Value *UnpackArg = II.getArgOperand(0);
*d415bd75Srobert  auto *RetTy = cast<ScalableVectorType>(II.getType());
*d415bd75Srobert  bool IsSigned = II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpkhi ||
*d415bd75Srobert                  II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpklo;
*d415bd75Srobert
*d415bd75Srobert  // Hi = uunpkhi(splat(X)) --> Hi = splat(extend(X))
*d415bd75Srobert  // Lo = uunpklo(splat(X)) --> Lo = splat(extend(X))
*d415bd75Srobert  if (auto *ScalarArg = getSplatValue(UnpackArg)) {
*d415bd75Srobert    ScalarArg =
*d415bd75Srobert        Builder.CreateIntCast(ScalarArg, RetTy->getScalarType(), IsSigned);
*d415bd75Srobert    Value *NewVal =
*d415bd75Srobert        Builder.CreateVectorSplat(RetTy->getElementCount(), ScalarArg);
*d415bd75Srobert    NewVal->takeName(&II);
*d415bd75Srobert    return IC.replaceInstUsesWith(II, NewVal);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVETBL(InstCombiner &IC,
73471bf0Spatrick                                                      IntrinsicInst &II) {
73471bf0Spatrick  auto *OpVal = II.getOperand(0);
73471bf0Spatrick  auto *OpIndices = II.getOperand(1);
73471bf0Spatrick  VectorType *VTy = cast<VectorType>(II.getType());
73471bf0Spatrick
*d415bd75Srobert  // Check whether OpIndices is a constant splat value < minimal element count
*d415bd75Srobert  // of result.
*d415bd75Srobert  auto *SplatValue = dyn_cast_or_null<ConstantInt>(getSplatValue(OpIndices));
73471bf0Spatrick  if (!SplatValue ||
73471bf0Spatrick      SplatValue->getValue().uge(VTy->getElementCount().getKnownMinValue()))
*d415bd75Srobert    return std::nullopt;
73471bf0Spatrick
73471bf0Spatrick  // Convert sve_tbl(OpVal sve_dup_x(SplatValue)) to
73471bf0Spatrick  // splat_vector(extractelement(OpVal, SplatValue)) for further optimization.
73471bf0Spatrick  IRBuilder<> Builder(II.getContext());
73471bf0Spatrick  Builder.SetInsertPoint(&II);
73471bf0Spatrick  auto *Extract = Builder.CreateExtractElement(OpVal, SplatValue);
73471bf0Spatrick  auto *VectorSplat =
73471bf0Spatrick      Builder.CreateVectorSplat(VTy->getElementCount(), Extract);
73471bf0Spatrick
73471bf0Spatrick  VectorSplat->takeName(&II);
73471bf0Spatrick  return IC.replaceInstUsesWith(II, VectorSplat);
73471bf0Spatrick}
73471bf0Spatrick
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEZip(InstCombiner &IC,
*d415bd75Srobert                                                      IntrinsicInst &II) {
*d415bd75Srobert  // zip1(uzp1(A, B), uzp2(A, B)) --> A
*d415bd75Srobert  // zip2(uzp1(A, B), uzp2(A, B)) --> B
*d415bd75Srobert  Value *A, *B;
*d415bd75Srobert  if (match(II.getArgOperand(0),
*d415bd75Srobert            m_Intrinsic<Intrinsic::aarch64_sve_uzp1>(m_Value(A), m_Value(B))) &&
*d415bd75Srobert      match(II.getArgOperand(1), m_Intrinsic<Intrinsic::aarch64_sve_uzp2>(
*d415bd75Srobert                                     m_Specific(A), m_Specific(B))))
*d415bd75Srobert    return IC.replaceInstUsesWith(
*d415bd75Srobert        II, (II.getIntrinsicID() == Intrinsic::aarch64_sve_zip1 ? A : B));
*d415bd75Srobert
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *>
*d415bd75SrobertinstCombineLD1GatherIndex(InstCombiner &IC, IntrinsicInst &II) {
*d415bd75Srobert  Value *Mask = II.getOperand(0);
*d415bd75Srobert  Value *BasePtr = II.getOperand(1);
*d415bd75Srobert  Value *Index = II.getOperand(2);
*d415bd75Srobert  Type *Ty = II.getType();
*d415bd75Srobert  Value *PassThru = ConstantAggregateZero::get(Ty);
*d415bd75Srobert
*d415bd75Srobert  // Contiguous gather => masked load.
*d415bd75Srobert  // (sve.ld1.gather.index Mask BasePtr (sve.index IndexBase 1))
*d415bd75Srobert  // => (masked.load (gep BasePtr IndexBase) Align Mask zeroinitializer)
*d415bd75Srobert  Value *IndexBase;
*d415bd75Srobert  if (match(Index, m_Intrinsic<Intrinsic::aarch64_sve_index>(
*d415bd75Srobert                       m_Value(IndexBase), m_SpecificInt(1)))) {
*d415bd75Srobert    IRBuilder<> Builder(II.getContext());
*d415bd75Srobert    Builder.SetInsertPoint(&II);
*d415bd75Srobert
*d415bd75Srobert    Align Alignment =
*d415bd75Srobert        BasePtr->getPointerAlignment(II.getModule()->getDataLayout());
*d415bd75Srobert
*d415bd75Srobert    Type *VecPtrTy = PointerType::getUnqual(Ty);
*d415bd75Srobert    Value *Ptr = Builder.CreateGEP(cast<VectorType>(Ty)->getElementType(),
*d415bd75Srobert                                   BasePtr, IndexBase);
*d415bd75Srobert    Ptr = Builder.CreateBitCast(Ptr, VecPtrTy);
*d415bd75Srobert    CallInst *MaskedLoad =
*d415bd75Srobert        Builder.CreateMaskedLoad(Ty, Ptr, Alignment, Mask, PassThru);
*d415bd75Srobert    MaskedLoad->takeName(&II);
*d415bd75Srobert    return IC.replaceInstUsesWith(II, MaskedLoad);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *>
*d415bd75SrobertinstCombineST1ScatterIndex(InstCombiner &IC, IntrinsicInst &II) {
*d415bd75Srobert  Value *Val = II.getOperand(0);
*d415bd75Srobert  Value *Mask = II.getOperand(1);
*d415bd75Srobert  Value *BasePtr = II.getOperand(2);
*d415bd75Srobert  Value *Index = II.getOperand(3);
*d415bd75Srobert  Type *Ty = Val->getType();
*d415bd75Srobert
*d415bd75Srobert  // Contiguous scatter => masked store.
*d415bd75Srobert  // (sve.st1.scatter.index Value Mask BasePtr (sve.index IndexBase 1))
*d415bd75Srobert  // => (masked.store Value (gep BasePtr IndexBase) Align Mask)
*d415bd75Srobert  Value *IndexBase;
*d415bd75Srobert  if (match(Index, m_Intrinsic<Intrinsic::aarch64_sve_index>(
*d415bd75Srobert                       m_Value(IndexBase), m_SpecificInt(1)))) {
*d415bd75Srobert    IRBuilder<> Builder(II.getContext());
*d415bd75Srobert    Builder.SetInsertPoint(&II);
*d415bd75Srobert
*d415bd75Srobert    Align Alignment =
*d415bd75Srobert        BasePtr->getPointerAlignment(II.getModule()->getDataLayout());
*d415bd75Srobert
*d415bd75Srobert    Value *Ptr = Builder.CreateGEP(cast<VectorType>(Ty)->getElementType(),
*d415bd75Srobert                                   BasePtr, IndexBase);
*d415bd75Srobert    Type *VecPtrTy = PointerType::getUnqual(Ty);
*d415bd75Srobert    Ptr = Builder.CreateBitCast(Ptr, VecPtrTy);
*d415bd75Srobert
*d415bd75Srobert    (void)Builder.CreateMaskedStore(Val, Ptr, Alignment, Mask);
*d415bd75Srobert
*d415bd75Srobert    return IC.eraseInstFromFunction(II);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVESDIV(InstCombiner &IC,
*d415bd75Srobert                                                       IntrinsicInst &II) {
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert  Type *Int32Ty = Builder.getInt32Ty();
*d415bd75Srobert  Value *Pred = II.getOperand(0);
*d415bd75Srobert  Value *Vec = II.getOperand(1);
*d415bd75Srobert  Value *DivVec = II.getOperand(2);
*d415bd75Srobert
*d415bd75Srobert  Value *SplatValue = getSplatValue(DivVec);
*d415bd75Srobert  ConstantInt *SplatConstantInt = dyn_cast_or_null<ConstantInt>(SplatValue);
*d415bd75Srobert  if (!SplatConstantInt)
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert  APInt Divisor = SplatConstantInt->getValue();
*d415bd75Srobert
*d415bd75Srobert  if (Divisor.isPowerOf2()) {
*d415bd75Srobert    Constant *DivisorLog2 = ConstantInt::get(Int32Ty, Divisor.logBase2());
*d415bd75Srobert    auto ASRD = Builder.CreateIntrinsic(
*d415bd75Srobert        Intrinsic::aarch64_sve_asrd, {II.getType()}, {Pred, Vec, DivisorLog2});
*d415bd75Srobert    return IC.replaceInstUsesWith(II, ASRD);
*d415bd75Srobert  }
*d415bd75Srobert  if (Divisor.isNegatedPowerOf2()) {
*d415bd75Srobert    Divisor.negate();
*d415bd75Srobert    Constant *DivisorLog2 = ConstantInt::get(Int32Ty, Divisor.logBase2());
*d415bd75Srobert    auto ASRD = Builder.CreateIntrinsic(
*d415bd75Srobert        Intrinsic::aarch64_sve_asrd, {II.getType()}, {Pred, Vec, DivisorLog2});
*d415bd75Srobert    auto NEG = Builder.CreateIntrinsic(Intrinsic::aarch64_sve_neg,
*d415bd75Srobert                                       {ASRD->getType()}, {ASRD, Pred, ASRD});
*d415bd75Srobert    return IC.replaceInstUsesWith(II, NEG);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertbool SimplifyValuePattern(SmallVector<Value *> &Vec, bool AllowPoison) {
*d415bd75Srobert  size_t VecSize = Vec.size();
*d415bd75Srobert  if (VecSize == 1)
*d415bd75Srobert    return true;
*d415bd75Srobert  if (!isPowerOf2_64(VecSize))
*d415bd75Srobert    return false;
*d415bd75Srobert  size_t HalfVecSize = VecSize / 2;
*d415bd75Srobert
*d415bd75Srobert  for (auto LHS = Vec.begin(), RHS = Vec.begin() + HalfVecSize;
*d415bd75Srobert       RHS != Vec.end(); LHS++, RHS++) {
*d415bd75Srobert    if (*LHS != nullptr && *RHS != nullptr) {
*d415bd75Srobert      if (*LHS == *RHS)
*d415bd75Srobert        continue;
*d415bd75Srobert      else
*d415bd75Srobert        return false;
*d415bd75Srobert    }
*d415bd75Srobert    if (!AllowPoison)
*d415bd75Srobert      return false;
*d415bd75Srobert    if (*LHS == nullptr && *RHS != nullptr)
*d415bd75Srobert      *LHS = *RHS;
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  Vec.resize(HalfVecSize);
*d415bd75Srobert  SimplifyValuePattern(Vec, AllowPoison);
*d415bd75Srobert  return true;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobert// Try to simplify dupqlane patterns like dupqlane(f32 A, f32 B, f32 A, f32 B)
*d415bd75Srobert// to dupqlane(f64(C)) where C is A concatenated with B
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVEDupqLane(InstCombiner &IC,
*d415bd75Srobert                                                           IntrinsicInst &II) {
*d415bd75Srobert  Value *CurrentInsertElt = nullptr, *Default = nullptr;
*d415bd75Srobert  if (!match(II.getOperand(0),
*d415bd75Srobert             m_Intrinsic<Intrinsic::vector_insert>(
*d415bd75Srobert                 m_Value(Default), m_Value(CurrentInsertElt), m_Value())) ||
*d415bd75Srobert      !isa<FixedVectorType>(CurrentInsertElt->getType()))
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert  auto IIScalableTy = cast<ScalableVectorType>(II.getType());
*d415bd75Srobert
*d415bd75Srobert  // Insert the scalars into a container ordered by InsertElement index
*d415bd75Srobert  SmallVector<Value *> Elts(IIScalableTy->getMinNumElements(), nullptr);
*d415bd75Srobert  while (auto InsertElt = dyn_cast<InsertElementInst>(CurrentInsertElt)) {
*d415bd75Srobert    auto Idx = cast<ConstantInt>(InsertElt->getOperand(2));
*d415bd75Srobert    Elts[Idx->getValue().getZExtValue()] = InsertElt->getOperand(1);
*d415bd75Srobert    CurrentInsertElt = InsertElt->getOperand(0);
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  bool AllowPoison =
*d415bd75Srobert      isa<PoisonValue>(CurrentInsertElt) && isa<PoisonValue>(Default);
*d415bd75Srobert  if (!SimplifyValuePattern(Elts, AllowPoison))
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  // Rebuild the simplified chain of InsertElements. e.g. (a, b, a, b) as (a, b)
*d415bd75Srobert  IRBuilder<> Builder(II.getContext());
*d415bd75Srobert  Builder.SetInsertPoint(&II);
*d415bd75Srobert  Value *InsertEltChain = PoisonValue::get(CurrentInsertElt->getType());
*d415bd75Srobert  for (size_t I = 0; I < Elts.size(); I++) {
*d415bd75Srobert    if (Elts[I] == nullptr)
*d415bd75Srobert      continue;
*d415bd75Srobert    InsertEltChain = Builder.CreateInsertElement(InsertEltChain, Elts[I],
*d415bd75Srobert                                                 Builder.getInt64(I));
*d415bd75Srobert  }
*d415bd75Srobert  if (InsertEltChain == nullptr)
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  // Splat the simplified sequence, e.g. (f16 a, f16 b, f16 c, f16 d) as one i64
*d415bd75Srobert  // value or (f16 a, f16 b) as one i32 value. This requires an InsertSubvector
*d415bd75Srobert  // be bitcast to a type wide enough to fit the sequence, be splatted, and then
*d415bd75Srobert  // be narrowed back to the original type.
*d415bd75Srobert  unsigned PatternWidth = IIScalableTy->getScalarSizeInBits() * Elts.size();
*d415bd75Srobert  unsigned PatternElementCount = IIScalableTy->getScalarSizeInBits() *
*d415bd75Srobert                                 IIScalableTy->getMinNumElements() /
*d415bd75Srobert                                 PatternWidth;
*d415bd75Srobert
*d415bd75Srobert  IntegerType *WideTy = Builder.getIntNTy(PatternWidth);
*d415bd75Srobert  auto *WideScalableTy = ScalableVectorType::get(WideTy, PatternElementCount);
*d415bd75Srobert  auto *WideShuffleMaskTy =
*d415bd75Srobert      ScalableVectorType::get(Builder.getInt32Ty(), PatternElementCount);
*d415bd75Srobert
*d415bd75Srobert  auto ZeroIdx = ConstantInt::get(Builder.getInt64Ty(), APInt(64, 0));
*d415bd75Srobert  auto InsertSubvector = Builder.CreateInsertVector(
*d415bd75Srobert      II.getType(), PoisonValue::get(II.getType()), InsertEltChain, ZeroIdx);
*d415bd75Srobert  auto WideBitcast =
*d415bd75Srobert      Builder.CreateBitOrPointerCast(InsertSubvector, WideScalableTy);
*d415bd75Srobert  auto WideShuffleMask = ConstantAggregateZero::get(WideShuffleMaskTy);
*d415bd75Srobert  auto WideShuffle = Builder.CreateShuffleVector(
*d415bd75Srobert      WideBitcast, PoisonValue::get(WideScalableTy), WideShuffleMask);
*d415bd75Srobert  auto NarrowBitcast =
*d415bd75Srobert      Builder.CreateBitOrPointerCast(WideShuffle, II.getType());
*d415bd75Srobert
*d415bd75Srobert  return IC.replaceInstUsesWith(II, NarrowBitcast);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineMaxMinNM(InstCombiner &IC,
*d415bd75Srobert                                                        IntrinsicInst &II) {
*d415bd75Srobert  Value *A = II.getArgOperand(0);
*d415bd75Srobert  Value *B = II.getArgOperand(1);
*d415bd75Srobert  if (A == B)
*d415bd75Srobert    return IC.replaceInstUsesWith(II, A);
*d415bd75Srobert
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic std::optional<Instruction *> instCombineSVESrshl(InstCombiner &IC,
*d415bd75Srobert                                                        IntrinsicInst &II) {
*d415bd75Srobert  IRBuilder<> Builder(&II);
*d415bd75Srobert  Value *Pred = II.getOperand(0);
*d415bd75Srobert  Value *Vec = II.getOperand(1);
*d415bd75Srobert  Value *Shift = II.getOperand(2);
*d415bd75Srobert
*d415bd75Srobert  // Convert SRSHL into the simpler LSL intrinsic when fed by an ABS intrinsic.
*d415bd75Srobert  Value *AbsPred, *MergedValue;
*d415bd75Srobert  if (!match(Vec, m_Intrinsic<Intrinsic::aarch64_sve_sqabs>(
*d415bd75Srobert                      m_Value(MergedValue), m_Value(AbsPred), m_Value())) &&
*d415bd75Srobert      !match(Vec, m_Intrinsic<Intrinsic::aarch64_sve_abs>(
*d415bd75Srobert                      m_Value(MergedValue), m_Value(AbsPred), m_Value())))
*d415bd75Srobert
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  // Transform is valid if any of the following are true:
*d415bd75Srobert  // * The ABS merge value is an undef or non-negative
*d415bd75Srobert  // * The ABS predicate is all active
*d415bd75Srobert  // * The ABS predicate and the SRSHL predicates are the same
*d415bd75Srobert  if (!isa<UndefValue>(MergedValue) && !match(MergedValue, m_NonNegative()) &&
*d415bd75Srobert      AbsPred != Pred && !isAllActivePredicate(AbsPred))
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  // Only valid when the shift amount is non-negative, otherwise the rounding
*d415bd75Srobert  // behaviour of SRSHL cannot be ignored.
*d415bd75Srobert  if (!match(Shift, m_NonNegative()))
*d415bd75Srobert    return std::nullopt;
*d415bd75Srobert
*d415bd75Srobert  auto LSL = Builder.CreateIntrinsic(Intrinsic::aarch64_sve_lsl, {II.getType()},
*d415bd75Srobert                                     {Pred, Vec, Shift});
*d415bd75Srobert
*d415bd75Srobert  return IC.replaceInstUsesWith(II, LSL);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstd::optional<Instruction *>
73471bf0SpatrickAArch64TTIImpl::instCombineIntrinsic(InstCombiner &IC,
73471bf0Spatrick                                     IntrinsicInst &II) const {
73471bf0Spatrick  Intrinsic::ID IID = II.getIntrinsicID();
73471bf0Spatrick  switch (IID) {
73471bf0Spatrick  default:
73471bf0Spatrick    break;
*d415bd75Srobert  case Intrinsic::aarch64_neon_fmaxnm:
*d415bd75Srobert  case Intrinsic::aarch64_neon_fminnm:
*d415bd75Srobert    return instCombineMaxMinNM(IC, II);
73471bf0Spatrick  case Intrinsic::aarch64_sve_convert_from_svbool:
73471bf0Spatrick    return instCombineConvertFromSVBool(IC, II);
73471bf0Spatrick  case Intrinsic::aarch64_sve_dup:
73471bf0Spatrick    return instCombineSVEDup(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_dup_x:
*d415bd75Srobert    return instCombineSVEDupX(IC, II);
73471bf0Spatrick  case Intrinsic::aarch64_sve_cmpne:
73471bf0Spatrick  case Intrinsic::aarch64_sve_cmpne_wide:
73471bf0Spatrick    return instCombineSVECmpNE(IC, II);
73471bf0Spatrick  case Intrinsic::aarch64_sve_rdffr:
73471bf0Spatrick    return instCombineRDFFR(IC, II);
73471bf0Spatrick  case Intrinsic::aarch64_sve_lasta:
73471bf0Spatrick  case Intrinsic::aarch64_sve_lastb:
73471bf0Spatrick    return instCombineSVELast(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_clasta_n:
*d415bd75Srobert  case Intrinsic::aarch64_sve_clastb_n:
*d415bd75Srobert    return instCombineSVECondLast(IC, II);
73471bf0Spatrick  case Intrinsic::aarch64_sve_cntd:
73471bf0Spatrick    return instCombineSVECntElts(IC, II, 2);
73471bf0Spatrick  case Intrinsic::aarch64_sve_cntw:
73471bf0Spatrick    return instCombineSVECntElts(IC, II, 4);
73471bf0Spatrick  case Intrinsic::aarch64_sve_cnth:
73471bf0Spatrick    return instCombineSVECntElts(IC, II, 8);
73471bf0Spatrick  case Intrinsic::aarch64_sve_cntb:
73471bf0Spatrick    return instCombineSVECntElts(IC, II, 16);
73471bf0Spatrick  case Intrinsic::aarch64_sve_ptest_any:
73471bf0Spatrick  case Intrinsic::aarch64_sve_ptest_first:
73471bf0Spatrick  case Intrinsic::aarch64_sve_ptest_last:
73471bf0Spatrick    return instCombineSVEPTest(IC, II);
73471bf0Spatrick  case Intrinsic::aarch64_sve_mul:
73471bf0Spatrick  case Intrinsic::aarch64_sve_fmul:
73471bf0Spatrick    return instCombineSVEVectorMul(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_fadd:
*d415bd75Srobert  case Intrinsic::aarch64_sve_add:
*d415bd75Srobert    return instCombineSVEVectorAdd(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_fsub:
*d415bd75Srobert  case Intrinsic::aarch64_sve_sub:
*d415bd75Srobert    return instCombineSVEVectorSub(IC, II);
73471bf0Spatrick  case Intrinsic::aarch64_sve_tbl:
73471bf0Spatrick    return instCombineSVETBL(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_uunpkhi:
*d415bd75Srobert  case Intrinsic::aarch64_sve_uunpklo:
*d415bd75Srobert  case Intrinsic::aarch64_sve_sunpkhi:
*d415bd75Srobert  case Intrinsic::aarch64_sve_sunpklo:
*d415bd75Srobert    return instCombineSVEUnpack(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_zip1:
*d415bd75Srobert  case Intrinsic::aarch64_sve_zip2:
*d415bd75Srobert    return instCombineSVEZip(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_ld1_gather_index:
*d415bd75Srobert    return instCombineLD1GatherIndex(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_st1_scatter_index:
*d415bd75Srobert    return instCombineST1ScatterIndex(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_ld1:
*d415bd75Srobert    return instCombineSVELD1(IC, II, DL);
*d415bd75Srobert  case Intrinsic::aarch64_sve_st1:
*d415bd75Srobert    return instCombineSVEST1(IC, II, DL);
*d415bd75Srobert  case Intrinsic::aarch64_sve_sdiv:
*d415bd75Srobert    return instCombineSVESDIV(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_sel:
*d415bd75Srobert    return instCombineSVESel(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_srshl:
*d415bd75Srobert    return instCombineSVESrshl(IC, II);
*d415bd75Srobert  case Intrinsic::aarch64_sve_dupq_lane:
*d415bd75Srobert    return instCombineSVEDupqLane(IC, II);
73471bf0Spatrick  }
73471bf0Spatrick
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstd::optional<Value *> AArch64TTIImpl::simplifyDemandedVectorEltsIntrinsic(
*d415bd75Srobert    InstCombiner &IC, IntrinsicInst &II, APInt OrigDemandedElts,
*d415bd75Srobert    APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3,
*d415bd75Srobert    std::function<void(Instruction *, unsigned, APInt, APInt &)>
*d415bd75Srobert        SimplifyAndSetOp) const {
*d415bd75Srobert  switch (II.getIntrinsicID()) {
*d415bd75Srobert  default:
*d415bd75Srobert    break;
*d415bd75Srobert  case Intrinsic::aarch64_neon_fcvtxn:
*d415bd75Srobert  case Intrinsic::aarch64_neon_rshrn:
*d415bd75Srobert  case Intrinsic::aarch64_neon_sqrshrn:
*d415bd75Srobert  case Intrinsic::aarch64_neon_sqrshrun:
*d415bd75Srobert  case Intrinsic::aarch64_neon_sqshrn:
*d415bd75Srobert  case Intrinsic::aarch64_neon_sqshrun:
*d415bd75Srobert  case Intrinsic::aarch64_neon_sqxtn:
*d415bd75Srobert  case Intrinsic::aarch64_neon_sqxtun:
*d415bd75Srobert  case Intrinsic::aarch64_neon_uqrshrn:
*d415bd75Srobert  case Intrinsic::aarch64_neon_uqshrn:
*d415bd75Srobert  case Intrinsic::aarch64_neon_uqxtn:
*d415bd75Srobert    SimplifyAndSetOp(&II, 0, OrigDemandedElts, UndefElts);
*d415bd75Srobert    break;
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  return std::nullopt;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75SrobertTypeSize
*d415bd75SrobertAArch64TTIImpl::getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const {
*d415bd75Srobert  switch (K) {
*d415bd75Srobert  case TargetTransformInfo::RGK_Scalar:
*d415bd75Srobert    return TypeSize::getFixed(64);
*d415bd75Srobert  case TargetTransformInfo::RGK_FixedWidthVector:
*d415bd75Srobert    if (!ST->isStreamingSVEModeDisabled() &&
*d415bd75Srobert        !EnableFixedwidthAutovecInStreamingMode)
*d415bd75Srobert      return TypeSize::getFixed(0);
*d415bd75Srobert
*d415bd75Srobert    if (ST->hasSVE())
*d415bd75Srobert      return TypeSize::getFixed(
*d415bd75Srobert          std::max(ST->getMinSVEVectorSizeInBits(), 128u));
*d415bd75Srobert
*d415bd75Srobert    return TypeSize::getFixed(ST->hasNEON() ? 128 : 0);
*d415bd75Srobert  case TargetTransformInfo::RGK_ScalableVector:
*d415bd75Srobert    if (!ST->isStreamingSVEModeDisabled() && !EnableScalableAutovecInStreamingMode)
*d415bd75Srobert      return TypeSize::getScalable(0);
*d415bd75Srobert
*d415bd75Srobert    return TypeSize::getScalable(ST->hasSVE() ? 128 : 0);
*d415bd75Srobert  }
*d415bd75Srobert  llvm_unreachable("Unsupported register kind");
73471bf0Spatrick}
73471bf0Spatrick
09467b48Spatrickbool AArch64TTIImpl::isWideningInstruction(Type *DstTy, unsigned Opcode,
09467b48Spatrick                                           ArrayRef<const Value *> Args) {
09467b48Spatrick
09467b48Spatrick  // A helper that returns a vector type from the given type. The number of
*d415bd75Srobert  // elements in type Ty determines the vector width.
09467b48Spatrick  auto toVectorTy = [&](Type *ArgTy) {
73471bf0Spatrick    return VectorType::get(ArgTy->getScalarType(),
73471bf0Spatrick                           cast<VectorType>(DstTy)->getElementCount());
09467b48Spatrick  };
09467b48Spatrick
09467b48Spatrick  // Exit early if DstTy is not a vector type whose elements are at least
*d415bd75Srobert  // 16-bits wide. SVE doesn't generally have the same set of instructions to
*d415bd75Srobert  // perform an extend with the add/sub/mul. There are SMULLB style
*d415bd75Srobert  // instructions, but they operate on top/bottom, requiring some sort of lane
*d415bd75Srobert  // interleaving to be used with zext/sext.
*d415bd75Srobert  if (!useNeonVector(DstTy) || DstTy->getScalarSizeInBits() < 16)
09467b48Spatrick    return false;
09467b48Spatrick
09467b48Spatrick  // Determine if the operation has a widening variant. We consider both the
09467b48Spatrick  // "long" (e.g., usubl) and "wide" (e.g., usubw) versions of the
09467b48Spatrick  // instructions.
09467b48Spatrick  //
*d415bd75Srobert  // TODO: Add additional widening operations (e.g., shl, etc.) once we
09467b48Spatrick  //       verify that their extending operands are eliminated during code
09467b48Spatrick  //       generation.
09467b48Spatrick  switch (Opcode) {
09467b48Spatrick  case Instruction::Add: // UADDL(2), SADDL(2), UADDW(2), SADDW(2).
09467b48Spatrick  case Instruction::Sub: // USUBL(2), SSUBL(2), USUBW(2), SSUBW(2).
*d415bd75Srobert  case Instruction::Mul: // SMULL(2), UMULL(2)
09467b48Spatrick    break;
09467b48Spatrick  default:
09467b48Spatrick    return false;
09467b48Spatrick  }
09467b48Spatrick
09467b48Spatrick  // To be a widening instruction (either the "wide" or "long" versions), the
*d415bd75Srobert  // second operand must be a sign- or zero extend.
09467b48Spatrick  if (Args.size() != 2 ||
*d415bd75Srobert      (!isa<SExtInst>(Args[1]) && !isa<ZExtInst>(Args[1])))
09467b48Spatrick    return false;
09467b48Spatrick  auto *Extend = cast<CastInst>(Args[1]);
*d415bd75Srobert  auto *Arg0 = dyn_cast<CastInst>(Args[0]);
*d415bd75Srobert
*d415bd75Srobert  // A mul only has a mull version (not like addw). Both operands need to be
*d415bd75Srobert  // extending and the same type.
*d415bd75Srobert  if (Opcode == Instruction::Mul &&
*d415bd75Srobert      (!Arg0 || Arg0->getOpcode() != Extend->getOpcode() ||
*d415bd75Srobert       Arg0->getOperand(0)->getType() != Extend->getOperand(0)->getType()))
*d415bd75Srobert    return false;
09467b48Spatrick
09467b48Spatrick  // Legalize the destination type and ensure it can be used in a widening
09467b48Spatrick  // operation.
*d415bd75Srobert  auto DstTyL = getTypeLegalizationCost(DstTy);
09467b48Spatrick  unsigned DstElTySize = DstTyL.second.getScalarSizeInBits();
09467b48Spatrick  if (!DstTyL.second.isVector() || DstElTySize != DstTy->getScalarSizeInBits())
09467b48Spatrick    return false;
09467b48Spatrick
09467b48Spatrick  // Legalize the source type and ensure it can be used in a widening
09467b48Spatrick  // operation.
097a140dSpatrick  auto *SrcTy = toVectorTy(Extend->getSrcTy());
*d415bd75Srobert  auto SrcTyL = getTypeLegalizationCost(SrcTy);
09467b48Spatrick  unsigned SrcElTySize = SrcTyL.second.getScalarSizeInBits();
09467b48Spatrick  if (!SrcTyL.second.isVector() || SrcElTySize != SrcTy->getScalarSizeInBits())
09467b48Spatrick    return false;
09467b48Spatrick
09467b48Spatrick  // Get the total number of vector elements in the legalized types.
73471bf0Spatrick  InstructionCost NumDstEls =
73471bf0Spatrick      DstTyL.first * DstTyL.second.getVectorMinNumElements();
73471bf0Spatrick  InstructionCost NumSrcEls =
73471bf0Spatrick      SrcTyL.first * SrcTyL.second.getVectorMinNumElements();
09467b48Spatrick
09467b48Spatrick  // Return true if the legalized types have the same number of vector elements
09467b48Spatrick  // and the destination element type size is twice that of the source type.
09467b48Spatrick  return NumDstEls == NumSrcEls && 2 * SrcElTySize == DstElTySize;
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getCastInstrCost(unsigned Opcode, Type *Dst,
73471bf0Spatrick                                                 Type *Src,
73471bf0Spatrick                                                 TTI::CastContextHint CCH,
097a140dSpatrick                                                 TTI::TargetCostKind CostKind,
09467b48Spatrick                                                 const Instruction *I) {
09467b48Spatrick  int ISD = TLI->InstructionOpcodeToISD(Opcode);
09467b48Spatrick  assert(ISD && "Invalid opcode");
09467b48Spatrick
09467b48Spatrick  // If the cast is observable, and it is used by a widening instruction (e.g.,
09467b48Spatrick  // uaddl, saddw, etc.), it may be free.
*d415bd75Srobert  if (I && I->hasOneUser()) {
09467b48Spatrick    auto *SingleUser = cast<Instruction>(*I->user_begin());
09467b48Spatrick    SmallVector<const Value *, 4> Operands(SingleUser->operand_values());
09467b48Spatrick    if (isWideningInstruction(Dst, SingleUser->getOpcode(), Operands)) {
09467b48Spatrick      // If the cast is the second operand, it is free. We will generate either
09467b48Spatrick      // a "wide" or "long" version of the widening instruction.
09467b48Spatrick      if (I == SingleUser->getOperand(1))
09467b48Spatrick        return 0;
09467b48Spatrick      // If the cast is not the second operand, it will be free if it looks the
09467b48Spatrick      // same as the second operand. In this case, we will generate a "long"
09467b48Spatrick      // version of the widening instruction.
09467b48Spatrick      if (auto *Cast = dyn_cast<CastInst>(SingleUser->getOperand(1)))
09467b48Spatrick        if (I->getOpcode() == unsigned(Cast->getOpcode()) &&
09467b48Spatrick            cast<CastInst>(I)->getSrcTy() == Cast->getSrcTy())
09467b48Spatrick          return 0;
09467b48Spatrick    }
09467b48Spatrick  }
09467b48Spatrick
097a140dSpatrick  // TODO: Allow non-throughput costs that aren't binary.
73471bf0Spatrick  auto AdjustCost = [&CostKind](InstructionCost Cost) -> InstructionCost {
097a140dSpatrick    if (CostKind != TTI::TCK_RecipThroughput)
097a140dSpatrick      return Cost == 0 ? 0 : 1;
097a140dSpatrick    return Cost;
097a140dSpatrick  };
097a140dSpatrick
09467b48Spatrick  EVT SrcTy = TLI->getValueType(DL, Src);
09467b48Spatrick  EVT DstTy = TLI->getValueType(DL, Dst);
09467b48Spatrick
09467b48Spatrick  if (!SrcTy.isSimple() || !DstTy.isSimple())
73471bf0Spatrick    return AdjustCost(
73471bf0Spatrick        BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I));
09467b48Spatrick
09467b48Spatrick  static const TypeConversionCostTblEntry
09467b48Spatrick  ConversionTbl[] = {
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v2i8,   MVT::v2i64,  1},  // xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v2i16,  MVT::v2i64,  1},  // xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v2i32,  MVT::v2i64,  1},  // xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v4i8,   MVT::v4i32,  1},  // xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v4i8,   MVT::v4i64,  3},  // 2 xtn + 1 uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v4i16,  MVT::v4i32,  1},  // xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v4i16,  MVT::v4i64,  2},  // 1 uzp1 + 1 xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v4i32,  MVT::v4i64,  1},  // 1 uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v8i8,   MVT::v8i16,  1},  // 1 xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v8i8,   MVT::v8i32,  2},  // 1 uzp1 + 1 xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v8i8,   MVT::v8i64,  4},  // 3 x uzp1 + xtn
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v8i16,  MVT::v8i32,  1},  // 1 uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v8i16,  MVT::v8i64,  3},  // 3 x uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v8i32,  MVT::v8i64,  2},  // 2 x uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v16i8,  MVT::v16i16, 1},  // uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v16i8,  MVT::v16i32, 3},  // (2 + 1) x uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v16i8,  MVT::v16i64, 7},  // (4 + 2 + 1) x uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v16i16, MVT::v16i32, 2},  // 2 x uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v16i16, MVT::v16i64, 6},  // (4 + 2) x uzp1
*d415bd75Srobert    { ISD::TRUNCATE, MVT::v16i32, MVT::v16i64, 4},  // 4 x uzp1
09467b48Spatrick
73471bf0Spatrick    // Truncations on nxvmiN
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv2i1, MVT::nxv2i16, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv2i1, MVT::nxv2i32, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv2i1, MVT::nxv2i64, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv4i1, MVT::nxv4i16, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv4i1, MVT::nxv4i32, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv4i1, MVT::nxv4i64, 2 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv8i1, MVT::nxv8i16, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv8i1, MVT::nxv8i32, 3 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv8i1, MVT::nxv8i64, 5 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv16i1, MVT::nxv16i8, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv2i16, MVT::nxv2i32, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv2i32, MVT::nxv2i64, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv4i16, MVT::nxv4i32, 1 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv4i32, MVT::nxv4i64, 2 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv8i16, MVT::nxv8i32, 3 },
73471bf0Spatrick    { ISD::TRUNCATE, MVT::nxv8i32, MVT::nxv8i64, 6 },
73471bf0Spatrick
09467b48Spatrick    // The number of shll instructions for the extension.
09467b48Spatrick    { ISD::SIGN_EXTEND, MVT::v4i64,  MVT::v4i16, 3 },
09467b48Spatrick    { ISD::ZERO_EXTEND, MVT::v4i64,  MVT::v4i16, 3 },
09467b48Spatrick    { ISD::SIGN_EXTEND, MVT::v4i64,  MVT::v4i32, 2 },
09467b48Spatrick    { ISD::ZERO_EXTEND, MVT::v4i64,  MVT::v4i32, 2 },
09467b48Spatrick    { ISD::SIGN_EXTEND, MVT::v8i32,  MVT::v8i8,  3 },
09467b48Spatrick    { ISD::ZERO_EXTEND, MVT::v8i32,  MVT::v8i8,  3 },
09467b48Spatrick    { ISD::SIGN_EXTEND, MVT::v8i32,  MVT::v8i16, 2 },
09467b48Spatrick    { ISD::ZERO_EXTEND, MVT::v8i32,  MVT::v8i16, 2 },
09467b48Spatrick    { ISD::SIGN_EXTEND, MVT::v8i64,  MVT::v8i8,  7 },
09467b48Spatrick    { ISD::ZERO_EXTEND, MVT::v8i64,  MVT::v8i8,  7 },
09467b48Spatrick    { ISD::SIGN_EXTEND, MVT::v8i64,  MVT::v8i16, 6 },
09467b48Spatrick    { ISD::ZERO_EXTEND, MVT::v8i64,  MVT::v8i16, 6 },
09467b48Spatrick    { ISD::SIGN_EXTEND, MVT::v16i16, MVT::v16i8, 2 },
09467b48Spatrick    { ISD::ZERO_EXTEND, MVT::v16i16, MVT::v16i8, 2 },
09467b48Spatrick    { ISD::SIGN_EXTEND, MVT::v16i32, MVT::v16i8, 6 },
09467b48Spatrick    { ISD::ZERO_EXTEND, MVT::v16i32, MVT::v16i8, 6 },
09467b48Spatrick
09467b48Spatrick    // LowerVectorINT_TO_FP:
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v2f32, MVT::v2i32, 1 },
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v4f32, MVT::v4i32, 1 },
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v2f64, MVT::v2i64, 1 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i32, 1 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i32, 1 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i64, 1 },
09467b48Spatrick
09467b48Spatrick    // Complex: to v2f32
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v2f32, MVT::v2i8,  3 },
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v2f32, MVT::v2i16, 3 },
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v2f32, MVT::v2i64, 2 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i8,  3 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i16, 3 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v2f32, MVT::v2i64, 2 },
09467b48Spatrick
09467b48Spatrick    // Complex: to v4f32
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v4f32, MVT::v4i8,  4 },
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v4f32, MVT::v4i16, 2 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i8,  3 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v4f32, MVT::v4i16, 2 },
09467b48Spatrick
09467b48Spatrick    // Complex: to v8f32
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i8,  10 },
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v8f32, MVT::v8i16, 4 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i8,  10 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v8f32, MVT::v8i16, 4 },
09467b48Spatrick
09467b48Spatrick    // Complex: to v16f32
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v16f32, MVT::v16i8, 21 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v16f32, MVT::v16i8, 21 },
09467b48Spatrick
09467b48Spatrick    // Complex: to v2f64
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v2f64, MVT::v2i8,  4 },
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v2f64, MVT::v2i16, 4 },
09467b48Spatrick    { ISD::SINT_TO_FP, MVT::v2f64, MVT::v2i32, 2 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i8,  4 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i16, 4 },
09467b48Spatrick    { ISD::UINT_TO_FP, MVT::v2f64, MVT::v2i32, 2 },
09467b48Spatrick
*d415bd75Srobert    // Complex: to v4f64
*d415bd75Srobert    { ISD::SINT_TO_FP, MVT::v4f64, MVT::v4i32,  4 },
*d415bd75Srobert    { ISD::UINT_TO_FP, MVT::v4f64, MVT::v4i32,  4 },
09467b48Spatrick
09467b48Spatrick    // LowerVectorFP_TO_INT
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v2i32, MVT::v2f32, 1 },
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v4i32, MVT::v4f32, 1 },
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v2i64, MVT::v2f64, 1 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v2i32, MVT::v2f32, 1 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f32, 1 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v2i64, MVT::v2f64, 1 },
09467b48Spatrick
09467b48Spatrick    // Complex, from v2f32: legal type is v2i32 (no cost) or v2i64 (1 ext).
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v2i64, MVT::v2f32, 2 },
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v2i16, MVT::v2f32, 1 },
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v2i8,  MVT::v2f32, 1 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v2i64, MVT::v2f32, 2 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v2i16, MVT::v2f32, 1 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v2i8,  MVT::v2f32, 1 },
09467b48Spatrick
09467b48Spatrick    // Complex, from v4f32: legal type is v4i16, 1 narrowing => ~2
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v4i16, MVT::v4f32, 2 },
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v4i8,  MVT::v4f32, 2 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v4i16, MVT::v4f32, 2 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v4i8,  MVT::v4f32, 2 },
09467b48Spatrick
73471bf0Spatrick    // Complex, from nxv2f32.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i64, MVT::nxv2f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i32, MVT::nxv2f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i16, MVT::nxv2f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i8,  MVT::nxv2f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i64, MVT::nxv2f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i32, MVT::nxv2f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i16, MVT::nxv2f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i8,  MVT::nxv2f32, 1 },
73471bf0Spatrick
09467b48Spatrick    // Complex, from v2f64: legal type is v2i32, 1 narrowing => ~2.
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v2i32, MVT::v2f64, 2 },
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v2i16, MVT::v2f64, 2 },
09467b48Spatrick    { ISD::FP_TO_SINT, MVT::v2i8,  MVT::v2f64, 2 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v2i32, MVT::v2f64, 2 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v2i16, MVT::v2f64, 2 },
09467b48Spatrick    { ISD::FP_TO_UINT, MVT::v2i8,  MVT::v2f64, 2 },
73471bf0Spatrick
73471bf0Spatrick    // Complex, from nxv2f64.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i64, MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i32, MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i16, MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i8,  MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i64, MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i32, MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i16, MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i8,  MVT::nxv2f64, 1 },
73471bf0Spatrick
73471bf0Spatrick    // Complex, from nxv4f32.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i64, MVT::nxv4f32, 4 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i32, MVT::nxv4f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i16, MVT::nxv4f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i8,  MVT::nxv4f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i64, MVT::nxv4f32, 4 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i32, MVT::nxv4f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i16, MVT::nxv4f32, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i8,  MVT::nxv4f32, 1 },
73471bf0Spatrick
73471bf0Spatrick    // Complex, from nxv8f64. Illegal -> illegal conversions not required.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv8i16, MVT::nxv8f64, 7 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv8i8,  MVT::nxv8f64, 7 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv8i16, MVT::nxv8f64, 7 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv8i8,  MVT::nxv8f64, 7 },
73471bf0Spatrick
73471bf0Spatrick    // Complex, from nxv4f64. Illegal -> illegal conversions not required.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i32, MVT::nxv4f64, 3 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i16, MVT::nxv4f64, 3 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i8,  MVT::nxv4f64, 3 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i32, MVT::nxv4f64, 3 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i16, MVT::nxv4f64, 3 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i8,  MVT::nxv4f64, 3 },
73471bf0Spatrick
73471bf0Spatrick    // Complex, from nxv8f32. Illegal -> illegal conversions not required.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv8i16, MVT::nxv8f32, 3 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv8i8,  MVT::nxv8f32, 3 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv8i16, MVT::nxv8f32, 3 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv8i8,  MVT::nxv8f32, 3 },
73471bf0Spatrick
73471bf0Spatrick    // Complex, from nxv8f16.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv8i64, MVT::nxv8f16, 10 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv8i32, MVT::nxv8f16, 4 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv8i16, MVT::nxv8f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv8i8,  MVT::nxv8f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv8i64, MVT::nxv8f16, 10 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv8i32, MVT::nxv8f16, 4 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv8i16, MVT::nxv8f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv8i8,  MVT::nxv8f16, 1 },
73471bf0Spatrick
73471bf0Spatrick    // Complex, from nxv4f16.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i64, MVT::nxv4f16, 4 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i32, MVT::nxv4f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i16, MVT::nxv4f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv4i8,  MVT::nxv4f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i64, MVT::nxv4f16, 4 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i32, MVT::nxv4f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i16, MVT::nxv4f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv4i8,  MVT::nxv4f16, 1 },
73471bf0Spatrick
73471bf0Spatrick    // Complex, from nxv2f16.
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i64, MVT::nxv2f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i32, MVT::nxv2f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i16, MVT::nxv2f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_SINT, MVT::nxv2i8,  MVT::nxv2f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i64, MVT::nxv2f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i32, MVT::nxv2f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i16, MVT::nxv2f16, 1 },
73471bf0Spatrick    { ISD::FP_TO_UINT, MVT::nxv2i8,  MVT::nxv2f16, 1 },
73471bf0Spatrick
73471bf0Spatrick    // Truncate from nxvmf32 to nxvmf16.
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv2f16, MVT::nxv2f32, 1 },
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv4f16, MVT::nxv4f32, 1 },
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv8f16, MVT::nxv8f32, 3 },
73471bf0Spatrick
73471bf0Spatrick    // Truncate from nxvmf64 to nxvmf16.
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv2f16, MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv4f16, MVT::nxv4f64, 3 },
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv8f16, MVT::nxv8f64, 7 },
73471bf0Spatrick
73471bf0Spatrick    // Truncate from nxvmf64 to nxvmf32.
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv2f32, MVT::nxv2f64, 1 },
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv4f32, MVT::nxv4f64, 3 },
73471bf0Spatrick    { ISD::FP_ROUND, MVT::nxv8f32, MVT::nxv8f64, 6 },
73471bf0Spatrick
73471bf0Spatrick    // Extend from nxvmf16 to nxvmf32.
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv2f32, MVT::nxv2f16, 1},
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv4f32, MVT::nxv4f16, 1},
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv8f32, MVT::nxv8f16, 2},
73471bf0Spatrick
73471bf0Spatrick    // Extend from nxvmf16 to nxvmf64.
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv2f64, MVT::nxv2f16, 1},
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv4f64, MVT::nxv4f16, 2},
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv8f64, MVT::nxv8f16, 4},
73471bf0Spatrick
73471bf0Spatrick    // Extend from nxvmf32 to nxvmf64.
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv2f64, MVT::nxv2f32, 1},
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv4f64, MVT::nxv4f32, 2},
73471bf0Spatrick    { ISD::FP_EXTEND, MVT::nxv8f64, MVT::nxv8f32, 6},
73471bf0Spatrick
*d415bd75Srobert    // Bitcasts from float to integer
*d415bd75Srobert    { ISD::BITCAST, MVT::nxv2f16, MVT::nxv2i16, 0 },
*d415bd75Srobert    { ISD::BITCAST, MVT::nxv4f16, MVT::nxv4i16, 0 },
*d415bd75Srobert    { ISD::BITCAST, MVT::nxv2f32, MVT::nxv2i32, 0 },
*d415bd75Srobert
*d415bd75Srobert    // Bitcasts from integer to float
*d415bd75Srobert    { ISD::BITCAST, MVT::nxv2i16, MVT::nxv2f16, 0 },
*d415bd75Srobert    { ISD::BITCAST, MVT::nxv4i16, MVT::nxv4f16, 0 },
*d415bd75Srobert    { ISD::BITCAST, MVT::nxv2i32, MVT::nxv2f32, 0 },
09467b48Spatrick  };
09467b48Spatrick
09467b48Spatrick  if (const auto *Entry = ConvertCostTableLookup(ConversionTbl, ISD,
09467b48Spatrick                                                 DstTy.getSimpleVT(),
09467b48Spatrick                                                 SrcTy.getSimpleVT()))
097a140dSpatrick    return AdjustCost(Entry->Cost);
09467b48Spatrick
*d415bd75Srobert  static const TypeConversionCostTblEntry FP16Tbl[] = {
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v4i8, MVT::v4f16, 1}, // fcvtzs
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v4i8, MVT::v4f16, 1},
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v4i16, MVT::v4f16, 1}, // fcvtzs
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v4i16, MVT::v4f16, 1},
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v4i32, MVT::v4f16, 2}, // fcvtl+fcvtzs
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v4i32, MVT::v4f16, 2},
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v8i8, MVT::v8f16, 2}, // fcvtzs+xtn
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v8i8, MVT::v8f16, 2},
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v8i16, MVT::v8f16, 1}, // fcvtzs
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v8i16, MVT::v8f16, 1},
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v8i32, MVT::v8f16, 4}, // 2*fcvtl+2*fcvtzs
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v8i32, MVT::v8f16, 4},
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v16i8, MVT::v16f16, 3}, // 2*fcvtzs+xtn
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v16i8, MVT::v16f16, 3},
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v16i16, MVT::v16f16, 2}, // 2*fcvtzs
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v16i16, MVT::v16f16, 2},
*d415bd75Srobert      {ISD::FP_TO_SINT, MVT::v16i32, MVT::v16f16, 8}, // 4*fcvtl+4*fcvtzs
*d415bd75Srobert      {ISD::FP_TO_UINT, MVT::v16i32, MVT::v16f16, 8},
*d415bd75Srobert      {ISD::UINT_TO_FP, MVT::v8f16, MVT::v8i8, 2},   // ushll + ucvtf
*d415bd75Srobert      {ISD::SINT_TO_FP, MVT::v8f16, MVT::v8i8, 2},   // sshll + scvtf
*d415bd75Srobert      {ISD::UINT_TO_FP, MVT::v16f16, MVT::v16i8, 4}, // 2 * ushl(2) + 2 * ucvtf
*d415bd75Srobert      {ISD::SINT_TO_FP, MVT::v16f16, MVT::v16i8, 4}, // 2 * sshl(2) + 2 * scvtf
*d415bd75Srobert  };
*d415bd75Srobert
*d415bd75Srobert  if (ST->hasFullFP16())
*d415bd75Srobert    if (const auto *Entry = ConvertCostTableLookup(
*d415bd75Srobert            FP16Tbl, ISD, DstTy.getSimpleVT(), SrcTy.getSimpleVT()))
*d415bd75Srobert      return AdjustCost(Entry->Cost);
*d415bd75Srobert
73471bf0Spatrick  return AdjustCost(
73471bf0Spatrick      BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I));
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getExtractWithExtendCost(unsigned Opcode,
73471bf0Spatrick                                                         Type *Dst,
09467b48Spatrick                                                         VectorType *VecTy,
09467b48Spatrick                                                         unsigned Index) {
09467b48Spatrick
09467b48Spatrick  // Make sure we were given a valid extend opcode.
09467b48Spatrick  assert((Opcode == Instruction::SExt || Opcode == Instruction::ZExt) &&
09467b48Spatrick         "Invalid opcode");
09467b48Spatrick
09467b48Spatrick  // We are extending an element we extract from a vector, so the source type
09467b48Spatrick  // of the extend is the element type of the vector.
09467b48Spatrick  auto *Src = VecTy->getElementType();
09467b48Spatrick
09467b48Spatrick  // Sign- and zero-extends are for integer types only.
09467b48Spatrick  assert(isa<IntegerType>(Dst) && isa<IntegerType>(Src) && "Invalid type");
09467b48Spatrick
09467b48Spatrick  // Get the cost for the extract. We compute the cost (if any) for the extend
09467b48Spatrick  // below.
*d415bd75Srobert  TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
*d415bd75Srobert  InstructionCost Cost = getVectorInstrCost(Instruction::ExtractElement, VecTy,
*d415bd75Srobert                                            CostKind, Index, nullptr, nullptr);
09467b48Spatrick
09467b48Spatrick  // Legalize the types.
*d415bd75Srobert  auto VecLT = getTypeLegalizationCost(VecTy);
09467b48Spatrick  auto DstVT = TLI->getValueType(DL, Dst);
09467b48Spatrick  auto SrcVT = TLI->getValueType(DL, Src);
09467b48Spatrick
09467b48Spatrick  // If the resulting type is still a vector and the destination type is legal,
09467b48Spatrick  // we may get the extension for free. If not, get the default cost for the
09467b48Spatrick  // extend.
09467b48Spatrick  if (!VecLT.second.isVector() || !TLI->isTypeLegal(DstVT))
73471bf0Spatrick    return Cost + getCastInstrCost(Opcode, Dst, Src, TTI::CastContextHint::None,
73471bf0Spatrick                                   CostKind);
09467b48Spatrick
09467b48Spatrick  // The destination type should be larger than the element type. If not, get
09467b48Spatrick  // the default cost for the extend.
73471bf0Spatrick  if (DstVT.getFixedSizeInBits() < SrcVT.getFixedSizeInBits())
73471bf0Spatrick    return Cost + getCastInstrCost(Opcode, Dst, Src, TTI::CastContextHint::None,
73471bf0Spatrick                                   CostKind);
09467b48Spatrick
09467b48Spatrick  switch (Opcode) {
09467b48Spatrick  default:
09467b48Spatrick    llvm_unreachable("Opcode should be either SExt or ZExt");
09467b48Spatrick
09467b48Spatrick  // For sign-extends, we only need a smov, which performs the extension
09467b48Spatrick  // automatically.
09467b48Spatrick  case Instruction::SExt:
09467b48Spatrick    return Cost;
09467b48Spatrick
09467b48Spatrick  // For zero-extends, the extend is performed automatically by a umov unless
09467b48Spatrick  // the destination type is i64 and the element type is i8 or i16.
09467b48Spatrick  case Instruction::ZExt:
09467b48Spatrick    if (DstVT.getSizeInBits() != 64u || SrcVT.getSizeInBits() == 32u)
09467b48Spatrick      return Cost;
09467b48Spatrick  }
09467b48Spatrick
09467b48Spatrick  // If we are unable to perform the extend for free, get the default cost.
73471bf0Spatrick  return Cost + getCastInstrCost(Opcode, Dst, Src, TTI::CastContextHint::None,
73471bf0Spatrick                                 CostKind);
097a140dSpatrick}
097a140dSpatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getCFInstrCost(unsigned Opcode,
73471bf0Spatrick                                               TTI::TargetCostKind CostKind,
73471bf0Spatrick                                               const Instruction *I) {
097a140dSpatrick  if (CostKind != TTI::TCK_RecipThroughput)
097a140dSpatrick    return Opcode == Instruction::PHI ? 0 : 1;
097a140dSpatrick  assert(CostKind == TTI::TCK_RecipThroughput && "unexpected CostKind");
097a140dSpatrick  // Branches are assumed to be predicted.
097a140dSpatrick  return 0;
09467b48Spatrick}
09467b48Spatrick
*d415bd75SrobertInstructionCost AArch64TTIImpl::getVectorInstrCostHelper(Type *Val,
*d415bd75Srobert                                                         unsigned Index,
*d415bd75Srobert                                                         bool HasRealUse) {
09467b48Spatrick  assert(Val->isVectorTy() && "This must be a vector type");
09467b48Spatrick
09467b48Spatrick  if (Index != -1U) {
09467b48Spatrick    // Legalize the type.
*d415bd75Srobert    std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Val);
09467b48Spatrick
09467b48Spatrick    // This type is legalized to a scalar type.
09467b48Spatrick    if (!LT.second.isVector())
09467b48Spatrick      return 0;
09467b48Spatrick
*d415bd75Srobert    // The type may be split. For fixed-width vectors we can normalize the
*d415bd75Srobert    // index to the new type.
*d415bd75Srobert    if (LT.second.isFixedLengthVector()) {
09467b48Spatrick      unsigned Width = LT.second.getVectorNumElements();
09467b48Spatrick      Index = Index % Width;
*d415bd75Srobert    }
09467b48Spatrick
09467b48Spatrick    // The element at index zero is already inside the vector.
*d415bd75Srobert    // - For a physical (HasRealUse==true) insert-element or extract-element
*d415bd75Srobert    // instruction that extracts integers, an explicit FPR -> GPR move is
*d415bd75Srobert    // needed. So it has non-zero cost.
*d415bd75Srobert    // - For the rest of cases (virtual instruction or element type is float),
*d415bd75Srobert    // consider the instruction free.
*d415bd75Srobert    //
*d415bd75Srobert    // FIXME:
*d415bd75Srobert    // If the extract-element and insert-element instructions could be
*d415bd75Srobert    // simplified away (e.g., could be combined into users by looking at use-def
*d415bd75Srobert    // context), they have no cost. This is not done in the first place for
*d415bd75Srobert    // compile-time considerations.
*d415bd75Srobert    if (Index == 0 && (!HasRealUse || !Val->getScalarType()->isIntegerTy()))
09467b48Spatrick      return 0;
09467b48Spatrick  }
09467b48Spatrick
09467b48Spatrick  // All other insert/extracts cost this much.
09467b48Spatrick  return ST->getVectorInsertExtractBaseCost();
09467b48Spatrick}
09467b48Spatrick
*d415bd75SrobertInstructionCost AArch64TTIImpl::getVectorInstrCost(unsigned Opcode, Type *Val,
*d415bd75Srobert                                                   TTI::TargetCostKind CostKind,
*d415bd75Srobert                                                   unsigned Index, Value *Op0,
*d415bd75Srobert                                                   Value *Op1) {
*d415bd75Srobert  return getVectorInstrCostHelper(Val, Index, false /* HasRealUse */);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75SrobertInstructionCost AArch64TTIImpl::getVectorInstrCost(const Instruction &I,
*d415bd75Srobert                                                   Type *Val,
*d415bd75Srobert                                                   TTI::TargetCostKind CostKind,
*d415bd75Srobert                                                   unsigned Index) {
*d415bd75Srobert  return getVectorInstrCostHelper(Val, Index, true /* HasRealUse */);
*d415bd75Srobert}
*d415bd75Srobert
73471bf0SpatrickInstructionCost AArch64TTIImpl::getArithmeticInstrCost(
097a140dSpatrick    unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
*d415bd75Srobert    TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info,
*d415bd75Srobert    ArrayRef<const Value *> Args,
09467b48Spatrick    const Instruction *CxtI) {
*d415bd75Srobert
097a140dSpatrick  // TODO: Handle more cost kinds.
097a140dSpatrick  if (CostKind != TTI::TCK_RecipThroughput)
*d415bd75Srobert    return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info,
*d415bd75Srobert                                         Op2Info, Args, CxtI);
097a140dSpatrick
09467b48Spatrick  // Legalize the type.
*d415bd75Srobert  std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
09467b48Spatrick  int ISD = TLI->InstructionOpcodeToISD(Opcode);
09467b48Spatrick
09467b48Spatrick  switch (ISD) {
09467b48Spatrick  default:
*d415bd75Srobert    return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info,
*d415bd75Srobert                                         Op2Info);
09467b48Spatrick  case ISD::SDIV:
*d415bd75Srobert    if (Op2Info.isConstant() && Op2Info.isUniform() && Op2Info.isPowerOf2()) {
09467b48Spatrick      // On AArch64, scalar signed division by constants power-of-two are
09467b48Spatrick      // normally expanded to the sequence ADD + CMP + SELECT + SRA.
09467b48Spatrick      // The OperandValue properties many not be same as that of previous
09467b48Spatrick      // operation; conservatively assume OP_None.
*d415bd75Srobert      InstructionCost Cost = getArithmeticInstrCost(
*d415bd75Srobert          Instruction::Add, Ty, CostKind,
*d415bd75Srobert          Op1Info.getNoProps(), Op2Info.getNoProps());
097a140dSpatrick      Cost += getArithmeticInstrCost(Instruction::Sub, Ty, CostKind,
*d415bd75Srobert                                     Op1Info.getNoProps(), Op2Info.getNoProps());
*d415bd75Srobert      Cost += getArithmeticInstrCost(
*d415bd75Srobert          Instruction::Select, Ty, CostKind,
*d415bd75Srobert          Op1Info.getNoProps(), Op2Info.getNoProps());
097a140dSpatrick      Cost += getArithmeticInstrCost(Instruction::AShr, Ty, CostKind,
*d415bd75Srobert                                     Op1Info.getNoProps(), Op2Info.getNoProps());
09467b48Spatrick      return Cost;
09467b48Spatrick    }
*d415bd75Srobert    [[fallthrough]];
*d415bd75Srobert  case ISD::UDIV: {
*d415bd75Srobert    if (Op2Info.isConstant() && Op2Info.isUniform()) {
09467b48Spatrick      auto VT = TLI->getValueType(DL, Ty);
09467b48Spatrick      if (TLI->isOperationLegalOrCustom(ISD::MULHU, VT)) {
09467b48Spatrick        // Vector signed division by constant are expanded to the
09467b48Spatrick        // sequence MULHS + ADD/SUB + SRA + SRL + ADD, and unsigned division
09467b48Spatrick        // to MULHS + SUB + SRL + ADD + SRL.
73471bf0Spatrick        InstructionCost MulCost = getArithmeticInstrCost(
*d415bd75Srobert            Instruction::Mul, Ty, CostKind, Op1Info.getNoProps(), Op2Info.getNoProps());
73471bf0Spatrick        InstructionCost AddCost = getArithmeticInstrCost(
*d415bd75Srobert            Instruction::Add, Ty, CostKind, Op1Info.getNoProps(), Op2Info.getNoProps());
73471bf0Spatrick        InstructionCost ShrCost = getArithmeticInstrCost(
*d415bd75Srobert            Instruction::AShr, Ty, CostKind, Op1Info.getNoProps(), Op2Info.getNoProps());
09467b48Spatrick        return MulCost * 2 + AddCost * 2 + ShrCost * 2 + 1;
09467b48Spatrick      }
09467b48Spatrick    }
09467b48Spatrick
*d415bd75Srobert    InstructionCost Cost = BaseT::getArithmeticInstrCost(
*d415bd75Srobert        Opcode, Ty, CostKind, Op1Info, Op2Info);
09467b48Spatrick    if (Ty->isVectorTy()) {
*d415bd75Srobert      if (TLI->isOperationLegalOrCustom(ISD, LT.second) && ST->hasSVE()) {
*d415bd75Srobert        // SDIV/UDIV operations are lowered using SVE, then we can have less
*d415bd75Srobert        // costs.
*d415bd75Srobert        if (isa<FixedVectorType>(Ty) && cast<FixedVectorType>(Ty)
*d415bd75Srobert                                                ->getPrimitiveSizeInBits()
*d415bd75Srobert                                                .getFixedValue() < 128) {
*d415bd75Srobert          EVT VT = TLI->getValueType(DL, Ty);
*d415bd75Srobert          static const CostTblEntry DivTbl[]{
*d415bd75Srobert              {ISD::SDIV, MVT::v2i8, 5},  {ISD::SDIV, MVT::v4i8, 8},
*d415bd75Srobert              {ISD::SDIV, MVT::v8i8, 8},  {ISD::SDIV, MVT::v2i16, 5},
*d415bd75Srobert              {ISD::SDIV, MVT::v4i16, 5}, {ISD::SDIV, MVT::v2i32, 1},
*d415bd75Srobert              {ISD::UDIV, MVT::v2i8, 5},  {ISD::UDIV, MVT::v4i8, 8},
*d415bd75Srobert              {ISD::UDIV, MVT::v8i8, 8},  {ISD::UDIV, MVT::v2i16, 5},
*d415bd75Srobert              {ISD::UDIV, MVT::v4i16, 5}, {ISD::UDIV, MVT::v2i32, 1}};
*d415bd75Srobert
*d415bd75Srobert          const auto *Entry = CostTableLookup(DivTbl, ISD, VT.getSimpleVT());
*d415bd75Srobert          if (nullptr != Entry)
*d415bd75Srobert            return Entry->Cost;
*d415bd75Srobert        }
*d415bd75Srobert        // For 8/16-bit elements, the cost is higher because the type
*d415bd75Srobert        // requires promotion and possibly splitting:
*d415bd75Srobert        if (LT.second.getScalarType() == MVT::i8)
*d415bd75Srobert          Cost *= 8;
*d415bd75Srobert        else if (LT.second.getScalarType() == MVT::i16)
*d415bd75Srobert          Cost *= 4;
*d415bd75Srobert        return Cost;
*d415bd75Srobert      } else {
*d415bd75Srobert        // If one of the operands is a uniform constant then the cost for each
*d415bd75Srobert        // element is Cost for insertion, extraction and division.
*d415bd75Srobert        // Insertion cost = 2, Extraction Cost = 2, Division = cost for the
*d415bd75Srobert        // operation with scalar type
*d415bd75Srobert        if ((Op1Info.isConstant() && Op1Info.isUniform()) ||
*d415bd75Srobert            (Op2Info.isConstant() && Op2Info.isUniform())) {
*d415bd75Srobert          if (auto *VTy = dyn_cast<FixedVectorType>(Ty)) {
*d415bd75Srobert            InstructionCost DivCost = BaseT::getArithmeticInstrCost(
*d415bd75Srobert                Opcode, Ty->getScalarType(), CostKind, Op1Info, Op2Info);
*d415bd75Srobert            return (4 + DivCost) * VTy->getNumElements();
*d415bd75Srobert          }
*d415bd75Srobert        }
*d415bd75Srobert        // On AArch64, without SVE, vector divisions are expanded
*d415bd75Srobert        // into scalar divisions of each pair of elements.
*d415bd75Srobert        Cost += getArithmeticInstrCost(Instruction::ExtractElement, Ty,
*d415bd75Srobert                                       CostKind, Op1Info, Op2Info);
097a140dSpatrick        Cost += getArithmeticInstrCost(Instruction::InsertElement, Ty, CostKind,
*d415bd75Srobert                                       Op1Info, Op2Info);
*d415bd75Srobert      }
*d415bd75Srobert
09467b48Spatrick      // TODO: if one of the arguments is scalar, then it's not necessary to
09467b48Spatrick      // double the cost of handling the vector elements.
09467b48Spatrick      Cost += Cost;
09467b48Spatrick    }
09467b48Spatrick    return Cost;
*d415bd75Srobert  }
09467b48Spatrick  case ISD::MUL:
*d415bd75Srobert    // When SVE is available, then we can lower the v2i64 operation using
*d415bd75Srobert    // the SVE mul instruction, which has a lower cost.
*d415bd75Srobert    if (LT.second == MVT::v2i64 && ST->hasSVE())
*d415bd75Srobert      return LT.first;
*d415bd75Srobert
*d415bd75Srobert    // When SVE is not available, there is no MUL.2d instruction,
*d415bd75Srobert    // which means mul <2 x i64> is expensive as elements are extracted
*d415bd75Srobert    // from the vectors and the muls scalarized.
*d415bd75Srobert    // As getScalarizationOverhead is a bit too pessimistic, we
*d415bd75Srobert    // estimate the cost for a i64 vector directly here, which is:
*d415bd75Srobert    // - four 2-cost i64 extracts,
*d415bd75Srobert    // - two 2-cost i64 inserts, and
*d415bd75Srobert    // - two 1-cost muls.
*d415bd75Srobert    // So, for a v2i64 with LT.First = 1 the cost is 14, and for a v4i64 with
*d415bd75Srobert    // LT.first = 2 the cost is 28. If both operands are extensions it will not
*d415bd75Srobert    // need to scalarize so the cost can be cheaper (smull or umull).
*d415bd75Srobert    // so the cost can be cheaper (smull or umull).
*d415bd75Srobert    if (LT.second != MVT::v2i64 || isWideningInstruction(Ty, Opcode, Args))
*d415bd75Srobert      return LT.first;
*d415bd75Srobert    return LT.first * 14;
73471bf0Spatrick  case ISD::ADD:
09467b48Spatrick  case ISD::XOR:
09467b48Spatrick  case ISD::OR:
09467b48Spatrick  case ISD::AND:
*d415bd75Srobert  case ISD::SRL:
*d415bd75Srobert  case ISD::SRA:
*d415bd75Srobert  case ISD::SHL:
09467b48Spatrick    // These nodes are marked as 'custom' for combining purposes only.
09467b48Spatrick    // We know that they are legal. See LowerAdd in ISelLowering.
*d415bd75Srobert    return LT.first;
097a140dSpatrick
097a140dSpatrick  case ISD::FADD:
*d415bd75Srobert  case ISD::FSUB:
*d415bd75Srobert  case ISD::FMUL:
*d415bd75Srobert  case ISD::FDIV:
*d415bd75Srobert  case ISD::FNEG:
097a140dSpatrick    // These nodes are marked as 'custom' just to lower them to SVE.
097a140dSpatrick    // We know said lowering will incur no additional cost.
*d415bd75Srobert    if (!Ty->getScalarType()->isFP128Ty())
*d415bd75Srobert      return 2 * LT.first;
097a140dSpatrick
*d415bd75Srobert    return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info,
*d415bd75Srobert                                         Op2Info);
09467b48Spatrick  }
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getAddressComputationCost(Type *Ty,
73471bf0Spatrick                                                          ScalarEvolution *SE,
09467b48Spatrick                                                          const SCEV *Ptr) {
09467b48Spatrick  // Address computations in vectorized code with non-consecutive addresses will
09467b48Spatrick  // likely result in more instructions compared to scalar code where the
09467b48Spatrick  // computation can more often be merged into the index mode. The resulting
09467b48Spatrick  // extra micro-ops can significantly decrease throughput.
09467b48Spatrick  unsigned NumVectorInstToHideOverhead = 10;
09467b48Spatrick  int MaxMergeDistance = 64;
09467b48Spatrick
09467b48Spatrick  if (Ty->isVectorTy() && SE &&
09467b48Spatrick      !BaseT::isConstantStridedAccessLessThan(SE, Ptr, MaxMergeDistance + 1))
09467b48Spatrick    return NumVectorInstToHideOverhead;
09467b48Spatrick
09467b48Spatrick  // In many cases the address computation is not merged into the instruction
09467b48Spatrick  // addressing mode.
09467b48Spatrick  return 1;
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getCmpSelInstrCost(unsigned Opcode, Type *ValTy,
097a140dSpatrick                                                   Type *CondTy,
73471bf0Spatrick                                                   CmpInst::Predicate VecPred,
097a140dSpatrick                                                   TTI::TargetCostKind CostKind,
097a140dSpatrick                                                   const Instruction *I) {
097a140dSpatrick  // TODO: Handle other cost kinds.
097a140dSpatrick  if (CostKind != TTI::TCK_RecipThroughput)
73471bf0Spatrick    return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
73471bf0Spatrick                                     I);
09467b48Spatrick
09467b48Spatrick  int ISD = TLI->InstructionOpcodeToISD(Opcode);
09467b48Spatrick  // We don't lower some vector selects well that are wider than the register
09467b48Spatrick  // width.
73471bf0Spatrick  if (isa<FixedVectorType>(ValTy) && ISD == ISD::SELECT) {
09467b48Spatrick    // We would need this many instructions to hide the scalarization happening.
09467b48Spatrick    const int AmortizationCost = 20;
73471bf0Spatrick
73471bf0Spatrick    // If VecPred is not set, check if we can get a predicate from the context
73471bf0Spatrick    // instruction, if its type matches the requested ValTy.
73471bf0Spatrick    if (VecPred == CmpInst::BAD_ICMP_PREDICATE && I && I->getType() == ValTy) {
73471bf0Spatrick      CmpInst::Predicate CurrentPred;
73471bf0Spatrick      if (match(I, m_Select(m_Cmp(CurrentPred, m_Value(), m_Value()), m_Value(),
73471bf0Spatrick                            m_Value())))
73471bf0Spatrick        VecPred = CurrentPred;
73471bf0Spatrick    }
*d415bd75Srobert    // Check if we have a compare/select chain that can be lowered using
*d415bd75Srobert    // a (F)CMxx & BFI pair.
*d415bd75Srobert    if (CmpInst::isIntPredicate(VecPred) || VecPred == CmpInst::FCMP_OLE ||
*d415bd75Srobert        VecPred == CmpInst::FCMP_OLT || VecPred == CmpInst::FCMP_OGT ||
*d415bd75Srobert        VecPred == CmpInst::FCMP_OGE || VecPred == CmpInst::FCMP_OEQ ||
*d415bd75Srobert        VecPred == CmpInst::FCMP_UNE) {
*d415bd75Srobert      static const auto ValidMinMaxTys = {
*d415bd75Srobert          MVT::v8i8,  MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v2i32,
*d415bd75Srobert          MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32, MVT::v2f64};
*d415bd75Srobert      static const auto ValidFP16MinMaxTys = {MVT::v4f16, MVT::v8f16};
*d415bd75Srobert
*d415bd75Srobert      auto LT = getTypeLegalizationCost(ValTy);
*d415bd75Srobert      if (any_of(ValidMinMaxTys, [&LT](MVT M) { return M == LT.second; }) ||
*d415bd75Srobert          (ST->hasFullFP16() &&
*d415bd75Srobert           any_of(ValidFP16MinMaxTys, [&LT](MVT M) { return M == LT.second; })))
73471bf0Spatrick        return LT.first;
73471bf0Spatrick    }
73471bf0Spatrick
09467b48Spatrick    static const TypeConversionCostTblEntry
09467b48Spatrick    VectorSelectTbl[] = {
09467b48Spatrick      { ISD::SELECT, MVT::v16i1, MVT::v16i16, 16 },
09467b48Spatrick      { ISD::SELECT, MVT::v8i1, MVT::v8i32, 8 },
09467b48Spatrick      { ISD::SELECT, MVT::v16i1, MVT::v16i32, 16 },
09467b48Spatrick      { ISD::SELECT, MVT::v4i1, MVT::v4i64, 4 * AmortizationCost },
09467b48Spatrick      { ISD::SELECT, MVT::v8i1, MVT::v8i64, 8 * AmortizationCost },
09467b48Spatrick      { ISD::SELECT, MVT::v16i1, MVT::v16i64, 16 * AmortizationCost }
09467b48Spatrick    };
09467b48Spatrick
09467b48Spatrick    EVT SelCondTy = TLI->getValueType(DL, CondTy);
09467b48Spatrick    EVT SelValTy = TLI->getValueType(DL, ValTy);
09467b48Spatrick    if (SelCondTy.isSimple() && SelValTy.isSimple()) {
09467b48Spatrick      if (const auto *Entry = ConvertCostTableLookup(VectorSelectTbl, ISD,
09467b48Spatrick                                                     SelCondTy.getSimpleVT(),
09467b48Spatrick                                                     SelValTy.getSimpleVT()))
09467b48Spatrick        return Entry->Cost;
09467b48Spatrick    }
09467b48Spatrick  }
73471bf0Spatrick  // The base case handles scalable vectors fine for now, since it treats the
73471bf0Spatrick  // cost as 1 * legalization cost.
73471bf0Spatrick  return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind, I);
09467b48Spatrick}
09467b48Spatrick
09467b48SpatrickAArch64TTIImpl::TTI::MemCmpExpansionOptions
09467b48SpatrickAArch64TTIImpl::enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
09467b48Spatrick  TTI::MemCmpExpansionOptions Options;
097a140dSpatrick  if (ST->requiresStrictAlign()) {
097a140dSpatrick    // TODO: Add cost modeling for strict align. Misaligned loads expand to
097a140dSpatrick    // a bunch of instructions when strict align is enabled.
097a140dSpatrick    return Options;
097a140dSpatrick  }
097a140dSpatrick  Options.AllowOverlappingLoads = true;
09467b48Spatrick  Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
09467b48Spatrick  Options.NumLoadsPerBlock = Options.MaxNumLoads;
09467b48Spatrick  // TODO: Though vector loads usually perform well on AArch64, in some targets
09467b48Spatrick  // they may wake up the FP unit, which raises the power consumption.  Perhaps
09467b48Spatrick  // they could be used with no holds barred (-O3).
09467b48Spatrick  Options.LoadSizes = {8, 4, 2, 1};
09467b48Spatrick  return Options;
09467b48Spatrick}
09467b48Spatrick
*d415bd75Srobertbool AArch64TTIImpl::prefersVectorizedAddressing() const {
*d415bd75Srobert  return ST->hasSVE();
*d415bd75Srobert}
*d415bd75Srobert
73471bf0SpatrickInstructionCost
73471bf0SpatrickAArch64TTIImpl::getMaskedMemoryOpCost(unsigned Opcode, Type *Src,
73471bf0Spatrick                                      Align Alignment, unsigned AddressSpace,
73471bf0Spatrick                                      TTI::TargetCostKind CostKind) {
*d415bd75Srobert  if (useNeonVector(Src))
73471bf0Spatrick    return BaseT::getMaskedMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
73471bf0Spatrick                                        CostKind);
*d415bd75Srobert  auto LT = getTypeLegalizationCost(Src);
73471bf0Spatrick  if (!LT.first.isValid())
73471bf0Spatrick    return InstructionCost::getInvalid();
73471bf0Spatrick
73471bf0Spatrick  // The code-generator is currently not able to handle scalable vectors
73471bf0Spatrick  // of <vscale x 1 x eltty> yet, so return an invalid cost to avoid selecting
73471bf0Spatrick  // it. This change will be removed when code-generation for these types is
73471bf0Spatrick  // sufficiently reliable.
73471bf0Spatrick  if (cast<VectorType>(Src)->getElementCount() == ElementCount::getScalable(1))
73471bf0Spatrick    return InstructionCost::getInvalid();
73471bf0Spatrick
*d415bd75Srobert  return LT.first;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertstatic unsigned getSVEGatherScatterOverhead(unsigned Opcode) {
*d415bd75Srobert  return Opcode == Instruction::Load ? SVEGatherOverhead : SVEScatterOverhead;
73471bf0Spatrick}
73471bf0Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getGatherScatterOpCost(
73471bf0Spatrick    unsigned Opcode, Type *DataTy, const Value *Ptr, bool VariableMask,
73471bf0Spatrick    Align Alignment, TTI::TargetCostKind CostKind, const Instruction *I) {
*d415bd75Srobert  if (useNeonVector(DataTy))
73471bf0Spatrick    return BaseT::getGatherScatterOpCost(Opcode, DataTy, Ptr, VariableMask,
73471bf0Spatrick                                         Alignment, CostKind, I);
73471bf0Spatrick  auto *VT = cast<VectorType>(DataTy);
*d415bd75Srobert  auto LT = getTypeLegalizationCost(DataTy);
73471bf0Spatrick  if (!LT.first.isValid())
73471bf0Spatrick    return InstructionCost::getInvalid();
73471bf0Spatrick
73471bf0Spatrick  // The code-generator is currently not able to handle scalable vectors
73471bf0Spatrick  // of <vscale x 1 x eltty> yet, so return an invalid cost to avoid selecting
73471bf0Spatrick  // it. This change will be removed when code-generation for these types is
73471bf0Spatrick  // sufficiently reliable.
73471bf0Spatrick  if (cast<VectorType>(DataTy)->getElementCount() ==
73471bf0Spatrick      ElementCount::getScalable(1))
73471bf0Spatrick    return InstructionCost::getInvalid();
73471bf0Spatrick
73471bf0Spatrick  ElementCount LegalVF = LT.second.getVectorElementCount();
73471bf0Spatrick  InstructionCost MemOpCost =
*d415bd75Srobert      getMemoryOpCost(Opcode, VT->getElementType(), Alignment, 0, CostKind,
*d415bd75Srobert                      {TTI::OK_AnyValue, TTI::OP_None}, I);
*d415bd75Srobert  // Add on an overhead cost for using gathers/scatters.
*d415bd75Srobert  // TODO: At the moment this is applied unilaterally for all CPUs, but at some
*d415bd75Srobert  // point we may want a per-CPU overhead.
*d415bd75Srobert  MemOpCost *= getSVEGatherScatterOverhead(Opcode);
73471bf0Spatrick  return LT.first * MemOpCost * getMaxNumElements(LegalVF);
73471bf0Spatrick}
73471bf0Spatrick
73471bf0Spatrickbool AArch64TTIImpl::useNeonVector(const Type *Ty) const {
73471bf0Spatrick  return isa<FixedVectorType>(Ty) && !ST->useSVEForFixedLengthVectors();
73471bf0Spatrick}
73471bf0Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getMemoryOpCost(unsigned Opcode, Type *Ty,
73471bf0Spatrick                                                MaybeAlign Alignment,
73471bf0Spatrick                                                unsigned AddressSpace,
097a140dSpatrick                                                TTI::TargetCostKind CostKind,
*d415bd75Srobert                                                TTI::OperandValueInfo OpInfo,
09467b48Spatrick                                                const Instruction *I) {
73471bf0Spatrick  EVT VT = TLI->getValueType(DL, Ty, true);
097a140dSpatrick  // Type legalization can't handle structs
73471bf0Spatrick  if (VT == MVT::Other)
097a140dSpatrick    return BaseT::getMemoryOpCost(Opcode, Ty, Alignment, AddressSpace,
097a140dSpatrick                                  CostKind);
097a140dSpatrick
*d415bd75Srobert  auto LT = getTypeLegalizationCost(Ty);
73471bf0Spatrick  if (!LT.first.isValid())
73471bf0Spatrick    return InstructionCost::getInvalid();
73471bf0Spatrick
73471bf0Spatrick  // The code-generator is currently not able to handle scalable vectors
73471bf0Spatrick  // of <vscale x 1 x eltty> yet, so return an invalid cost to avoid selecting
73471bf0Spatrick  // it. This change will be removed when code-generation for these types is
73471bf0Spatrick  // sufficiently reliable.
73471bf0Spatrick  if (auto *VTy = dyn_cast<ScalableVectorType>(Ty))
73471bf0Spatrick    if (VTy->getElementCount() == ElementCount::getScalable(1))
73471bf0Spatrick      return InstructionCost::getInvalid();
73471bf0Spatrick
73471bf0Spatrick  // TODO: consider latency as well for TCK_SizeAndLatency.
73471bf0Spatrick  if (CostKind == TTI::TCK_CodeSize || CostKind == TTI::TCK_SizeAndLatency)
73471bf0Spatrick    return LT.first;
73471bf0Spatrick
73471bf0Spatrick  if (CostKind != TTI::TCK_RecipThroughput)
73471bf0Spatrick    return 1;
09467b48Spatrick
09467b48Spatrick  if (ST->isMisaligned128StoreSlow() && Opcode == Instruction::Store &&
09467b48Spatrick      LT.second.is128BitVector() && (!Alignment || *Alignment < Align(16))) {
09467b48Spatrick    // Unaligned stores are extremely inefficient. We don't split all
09467b48Spatrick    // unaligned 128-bit stores because the negative impact that has shown in
09467b48Spatrick    // practice on inlined block copy code.
09467b48Spatrick    // We make such stores expensive so that we will only vectorize if there
09467b48Spatrick    // are 6 other instructions getting vectorized.
09467b48Spatrick    const int AmortizationCost = 6;
09467b48Spatrick
09467b48Spatrick    return LT.first * 2 * AmortizationCost;
09467b48Spatrick  }
09467b48Spatrick
*d415bd75Srobert  // Opaque ptr or ptr vector types are i64s and can be lowered to STP/LDPs.
*d415bd75Srobert  if (Ty->isPtrOrPtrVectorTy())
*d415bd75Srobert    return LT.first;
*d415bd75Srobert
73471bf0Spatrick  // Check truncating stores and extending loads.
73471bf0Spatrick  if (useNeonVector(Ty) &&
73471bf0Spatrick      Ty->getScalarSizeInBits() != LT.second.getScalarSizeInBits()) {
73471bf0Spatrick    // v4i8 types are lowered to scalar a load/store and sshll/xtn.
73471bf0Spatrick    if (VT == MVT::v4i8)
73471bf0Spatrick      return 2;
73471bf0Spatrick    // Otherwise we need to scalarize.
73471bf0Spatrick    return cast<FixedVectorType>(Ty)->getNumElements() * 2;
09467b48Spatrick  }
09467b48Spatrick
09467b48Spatrick  return LT.first;
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getInterleavedMemoryOpCost(
097a140dSpatrick    unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
097a140dSpatrick    Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
097a140dSpatrick    bool UseMaskForCond, bool UseMaskForGaps) {
09467b48Spatrick  assert(Factor >= 2 && "Invalid interleave factor");
097a140dSpatrick  auto *VecVTy = cast<FixedVectorType>(VecTy);
09467b48Spatrick
09467b48Spatrick  if (!UseMaskForCond && !UseMaskForGaps &&
09467b48Spatrick      Factor <= TLI->getMaxSupportedInterleaveFactor()) {
097a140dSpatrick    unsigned NumElts = VecVTy->getNumElements();
097a140dSpatrick    auto *SubVecTy =
097a140dSpatrick        FixedVectorType::get(VecTy->getScalarType(), NumElts / Factor);
09467b48Spatrick
09467b48Spatrick    // ldN/stN only support legal vector types of size 64 or 128 in bits.
09467b48Spatrick    // Accesses having vector types that are a multiple of 128 bits can be
09467b48Spatrick    // matched to more than one ldN/stN instruction.
*d415bd75Srobert    bool UseScalable;
09467b48Spatrick    if (NumElts % Factor == 0 &&
*d415bd75Srobert        TLI->isLegalInterleavedAccessType(SubVecTy, DL, UseScalable))
*d415bd75Srobert      return Factor * TLI->getNumInterleavedAccesses(SubVecTy, DL, UseScalable);
09467b48Spatrick  }
09467b48Spatrick
09467b48Spatrick  return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
097a140dSpatrick                                           Alignment, AddressSpace, CostKind,
09467b48Spatrick                                           UseMaskForCond, UseMaskForGaps);
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost
73471bf0SpatrickAArch64TTIImpl::getCostOfKeepingLiveOverCall(ArrayRef<Type *> Tys) {
73471bf0Spatrick  InstructionCost Cost = 0;
097a140dSpatrick  TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
09467b48Spatrick  for (auto *I : Tys) {
09467b48Spatrick    if (!I->isVectorTy())
09467b48Spatrick      continue;
097a140dSpatrick    if (I->getScalarSizeInBits() * cast<FixedVectorType>(I)->getNumElements() ==
097a140dSpatrick        128)
097a140dSpatrick      Cost += getMemoryOpCost(Instruction::Store, I, Align(128), 0, CostKind) +
097a140dSpatrick              getMemoryOpCost(Instruction::Load, I, Align(128), 0, CostKind);
09467b48Spatrick  }
09467b48Spatrick  return Cost;
09467b48Spatrick}
09467b48Spatrick
09467b48Spatrickunsigned AArch64TTIImpl::getMaxInterleaveFactor(unsigned VF) {
09467b48Spatrick  return ST->getMaxInterleaveFactor();
09467b48Spatrick}
09467b48Spatrick
09467b48Spatrick// For Falkor, we want to avoid having too many strided loads in a loop since
09467b48Spatrick// that can exhaust the HW prefetcher resources.  We adjust the unroller
09467b48Spatrick// MaxCount preference below to attempt to ensure unrolling doesn't create too
09467b48Spatrick// many strided loads.
09467b48Spatrickstatic void
09467b48SpatrickgetFalkorUnrollingPreferences(Loop *L, ScalarEvolution &SE,
09467b48Spatrick                              TargetTransformInfo::UnrollingPreferences &UP) {
09467b48Spatrick  enum { MaxStridedLoads = 7 };
09467b48Spatrick  auto countStridedLoads = [](Loop *L, ScalarEvolution &SE) {
09467b48Spatrick    int StridedLoads = 0;
09467b48Spatrick    // FIXME? We could make this more precise by looking at the CFG and
09467b48Spatrick    // e.g. not counting loads in each side of an if-then-else diamond.
09467b48Spatrick    for (const auto BB : L->blocks()) {
09467b48Spatrick      for (auto &I : *BB) {
09467b48Spatrick        LoadInst *LMemI = dyn_cast<LoadInst>(&I);
09467b48Spatrick        if (!LMemI)
09467b48Spatrick          continue;
09467b48Spatrick
09467b48Spatrick        Value *PtrValue = LMemI->getPointerOperand();
09467b48Spatrick        if (L->isLoopInvariant(PtrValue))
09467b48Spatrick          continue;
09467b48Spatrick
09467b48Spatrick        const SCEV *LSCEV = SE.getSCEV(PtrValue);
09467b48Spatrick        const SCEVAddRecExpr *LSCEVAddRec = dyn_cast<SCEVAddRecExpr>(LSCEV);
09467b48Spatrick        if (!LSCEVAddRec || !LSCEVAddRec->isAffine())
09467b48Spatrick          continue;
09467b48Spatrick
09467b48Spatrick        // FIXME? We could take pairing of unrolled load copies into account
09467b48Spatrick        // by looking at the AddRec, but we would probably have to limit this
09467b48Spatrick        // to loops with no stores or other memory optimization barriers.
09467b48Spatrick        ++StridedLoads;
09467b48Spatrick        // We've seen enough strided loads that seeing more won't make a
09467b48Spatrick        // difference.
09467b48Spatrick        if (StridedLoads > MaxStridedLoads / 2)
09467b48Spatrick          return StridedLoads;
09467b48Spatrick      }
09467b48Spatrick    }
09467b48Spatrick    return StridedLoads;
09467b48Spatrick  };
09467b48Spatrick
09467b48Spatrick  int StridedLoads = countStridedLoads(L, SE);
09467b48Spatrick  LLVM_DEBUG(dbgs() << "falkor-hwpf: detected " << StridedLoads
09467b48Spatrick                    << " strided loads\n");
09467b48Spatrick  // Pick the largest power of 2 unroll count that won't result in too many
09467b48Spatrick  // strided loads.
09467b48Spatrick  if (StridedLoads) {
09467b48Spatrick    UP.MaxCount = 1 << Log2_32(MaxStridedLoads / StridedLoads);
09467b48Spatrick    LLVM_DEBUG(dbgs() << "falkor-hwpf: setting unroll MaxCount to "
09467b48Spatrick                      << UP.MaxCount << '\n');
09467b48Spatrick  }
09467b48Spatrick}
09467b48Spatrick
09467b48Spatrickvoid AArch64TTIImpl::getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
*d415bd75Srobert                                             TTI::UnrollingPreferences &UP,
*d415bd75Srobert                                             OptimizationRemarkEmitter *ORE) {
09467b48Spatrick  // Enable partial unrolling and runtime unrolling.
*d415bd75Srobert  BaseT::getUnrollingPreferences(L, SE, UP, ORE);
*d415bd75Srobert
*d415bd75Srobert  UP.UpperBound = true;
09467b48Spatrick
09467b48Spatrick  // For inner loop, it is more likely to be a hot one, and the runtime check
09467b48Spatrick  // can be promoted out from LICM pass, so the overhead is less, let's try
09467b48Spatrick  // a larger threshold to unroll more loops.
09467b48Spatrick  if (L->getLoopDepth() > 1)
09467b48Spatrick    UP.PartialThreshold *= 2;
09467b48Spatrick
09467b48Spatrick  // Disable partial & runtime unrolling on -Os.
09467b48Spatrick  UP.PartialOptSizeThreshold = 0;
09467b48Spatrick
09467b48Spatrick  if (ST->getProcFamily() == AArch64Subtarget::Falkor &&
09467b48Spatrick      EnableFalkorHWPFUnrollFix)
09467b48Spatrick    getFalkorUnrollingPreferences(L, SE, UP);
73471bf0Spatrick
73471bf0Spatrick  // Scan the loop: don't unroll loops with calls as this could prevent
73471bf0Spatrick  // inlining. Don't unroll vector loops either, as they don't benefit much from
73471bf0Spatrick  // unrolling.
73471bf0Spatrick  for (auto *BB : L->getBlocks()) {
73471bf0Spatrick    for (auto &I : *BB) {
73471bf0Spatrick      // Don't unroll vectorised loop.
73471bf0Spatrick      if (I.getType()->isVectorTy())
73471bf0Spatrick        return;
73471bf0Spatrick
73471bf0Spatrick      if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
73471bf0Spatrick        if (const Function *F = cast<CallBase>(I).getCalledFunction()) {
73471bf0Spatrick          if (!isLoweredToCall(F))
73471bf0Spatrick            continue;
73471bf0Spatrick        }
73471bf0Spatrick        return;
73471bf0Spatrick      }
73471bf0Spatrick    }
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  // Enable runtime unrolling for in-order models
73471bf0Spatrick  // If mcpu is omitted, getProcFamily() returns AArch64Subtarget::Others, so by
73471bf0Spatrick  // checking for that case, we can ensure that the default behaviour is
73471bf0Spatrick  // unchanged
73471bf0Spatrick  if (ST->getProcFamily() != AArch64Subtarget::Others &&
73471bf0Spatrick      !ST->getSchedModel().isOutOfOrder()) {
73471bf0Spatrick    UP.Runtime = true;
73471bf0Spatrick    UP.Partial = true;
73471bf0Spatrick    UP.UnrollRemainder = true;
73471bf0Spatrick    UP.DefaultUnrollRuntimeCount = 4;
73471bf0Spatrick
73471bf0Spatrick    UP.UnrollAndJam = true;
73471bf0Spatrick    UP.UnrollAndJamInnerLoopThreshold = 60;
73471bf0Spatrick  }
09467b48Spatrick}
09467b48Spatrick
097a140dSpatrickvoid AArch64TTIImpl::getPeelingPreferences(Loop *L, ScalarEvolution &SE,
097a140dSpatrick                                           TTI::PeelingPreferences &PP) {
097a140dSpatrick  BaseT::getPeelingPreferences(L, SE, PP);
097a140dSpatrick}
097a140dSpatrick
09467b48SpatrickValue *AArch64TTIImpl::getOrCreateResultFromMemIntrinsic(IntrinsicInst *Inst,
09467b48Spatrick                                                         Type *ExpectedType) {
09467b48Spatrick  switch (Inst->getIntrinsicID()) {
09467b48Spatrick  default:
09467b48Spatrick    return nullptr;
09467b48Spatrick  case Intrinsic::aarch64_neon_st2:
09467b48Spatrick  case Intrinsic::aarch64_neon_st3:
09467b48Spatrick  case Intrinsic::aarch64_neon_st4: {
09467b48Spatrick    // Create a struct type
09467b48Spatrick    StructType *ST = dyn_cast<StructType>(ExpectedType);
09467b48Spatrick    if (!ST)
09467b48Spatrick      return nullptr;
*d415bd75Srobert    unsigned NumElts = Inst->arg_size() - 1;
09467b48Spatrick    if (ST->getNumElements() != NumElts)
09467b48Spatrick      return nullptr;
09467b48Spatrick    for (unsigned i = 0, e = NumElts; i != e; ++i) {
09467b48Spatrick      if (Inst->getArgOperand(i)->getType() != ST->getElementType(i))
09467b48Spatrick        return nullptr;
09467b48Spatrick    }
*d415bd75Srobert    Value *Res = PoisonValue::get(ExpectedType);
09467b48Spatrick    IRBuilder<> Builder(Inst);
09467b48Spatrick    for (unsigned i = 0, e = NumElts; i != e; ++i) {
09467b48Spatrick      Value *L = Inst->getArgOperand(i);
09467b48Spatrick      Res = Builder.CreateInsertValue(Res, L, i);
09467b48Spatrick    }
09467b48Spatrick    return Res;
09467b48Spatrick  }
09467b48Spatrick  case Intrinsic::aarch64_neon_ld2:
09467b48Spatrick  case Intrinsic::aarch64_neon_ld3:
09467b48Spatrick  case Intrinsic::aarch64_neon_ld4:
09467b48Spatrick    if (Inst->getType() == ExpectedType)
09467b48Spatrick      return Inst;
09467b48Spatrick    return nullptr;
09467b48Spatrick  }
09467b48Spatrick}
09467b48Spatrick
09467b48Spatrickbool AArch64TTIImpl::getTgtMemIntrinsic(IntrinsicInst *Inst,
09467b48Spatrick                                        MemIntrinsicInfo &Info) {
09467b48Spatrick  switch (Inst->getIntrinsicID()) {
09467b48Spatrick  default:
09467b48Spatrick    break;
09467b48Spatrick  case Intrinsic::aarch64_neon_ld2:
09467b48Spatrick  case Intrinsic::aarch64_neon_ld3:
09467b48Spatrick  case Intrinsic::aarch64_neon_ld4:
09467b48Spatrick    Info.ReadMem = true;
09467b48Spatrick    Info.WriteMem = false;
09467b48Spatrick    Info.PtrVal = Inst->getArgOperand(0);
09467b48Spatrick    break;
09467b48Spatrick  case Intrinsic::aarch64_neon_st2:
09467b48Spatrick  case Intrinsic::aarch64_neon_st3:
09467b48Spatrick  case Intrinsic::aarch64_neon_st4:
09467b48Spatrick    Info.ReadMem = false;
09467b48Spatrick    Info.WriteMem = true;
*d415bd75Srobert    Info.PtrVal = Inst->getArgOperand(Inst->arg_size() - 1);
09467b48Spatrick    break;
09467b48Spatrick  }
09467b48Spatrick
09467b48Spatrick  switch (Inst->getIntrinsicID()) {
09467b48Spatrick  default:
09467b48Spatrick    return false;
09467b48Spatrick  case Intrinsic::aarch64_neon_ld2:
09467b48Spatrick  case Intrinsic::aarch64_neon_st2:
09467b48Spatrick    Info.MatchingId = VECTOR_LDST_TWO_ELEMENTS;
09467b48Spatrick    break;
09467b48Spatrick  case Intrinsic::aarch64_neon_ld3:
09467b48Spatrick  case Intrinsic::aarch64_neon_st3:
09467b48Spatrick    Info.MatchingId = VECTOR_LDST_THREE_ELEMENTS;
09467b48Spatrick    break;
09467b48Spatrick  case Intrinsic::aarch64_neon_ld4:
09467b48Spatrick  case Intrinsic::aarch64_neon_st4:
09467b48Spatrick    Info.MatchingId = VECTOR_LDST_FOUR_ELEMENTS;
09467b48Spatrick    break;
09467b48Spatrick  }
09467b48Spatrick  return true;
09467b48Spatrick}
09467b48Spatrick
09467b48Spatrick/// See if \p I should be considered for address type promotion. We check if \p
09467b48Spatrick/// I is a sext with right type and used in memory accesses. If it used in a
09467b48Spatrick/// "complex" getelementptr, we allow it to be promoted without finding other
09467b48Spatrick/// sext instructions that sign extended the same initial value. A getelementptr
09467b48Spatrick/// is considered as "complex" if it has more than 2 operands.
09467b48Spatrickbool AArch64TTIImpl::shouldConsiderAddressTypePromotion(
09467b48Spatrick    const Instruction &I, bool &AllowPromotionWithoutCommonHeader) {
09467b48Spatrick  bool Considerable = false;
09467b48Spatrick  AllowPromotionWithoutCommonHeader = false;
09467b48Spatrick  if (!isa<SExtInst>(&I))
09467b48Spatrick    return false;
09467b48Spatrick  Type *ConsideredSExtType =
09467b48Spatrick      Type::getInt64Ty(I.getParent()->getParent()->getContext());
09467b48Spatrick  if (I.getType() != ConsideredSExtType)
09467b48Spatrick    return false;
09467b48Spatrick  // See if the sext is the one with the right type and used in at least one
09467b48Spatrick  // GetElementPtrInst.
09467b48Spatrick  for (const User *U : I.users()) {
09467b48Spatrick    if (const GetElementPtrInst *GEPInst = dyn_cast<GetElementPtrInst>(U)) {
09467b48Spatrick      Considerable = true;
09467b48Spatrick      // A getelementptr is considered as "complex" if it has more than 2
09467b48Spatrick      // operands. We will promote a SExt used in such complex GEP as we
09467b48Spatrick      // expect some computation to be merged if they are done on 64 bits.
09467b48Spatrick      if (GEPInst->getNumOperands() > 2) {
09467b48Spatrick        AllowPromotionWithoutCommonHeader = true;
09467b48Spatrick        break;
09467b48Spatrick      }
09467b48Spatrick    }
09467b48Spatrick  }
09467b48Spatrick  return Considerable;
09467b48Spatrick}
09467b48Spatrick
73471bf0Spatrickbool AArch64TTIImpl::isLegalToVectorizeReduction(
73471bf0Spatrick    const RecurrenceDescriptor &RdxDesc, ElementCount VF) const {
73471bf0Spatrick  if (!VF.isScalable())
73471bf0Spatrick    return true;
73471bf0Spatrick
73471bf0Spatrick  Type *Ty = RdxDesc.getRecurrenceType();
73471bf0Spatrick  if (Ty->isBFloatTy() || !isElementTypeLegalForScalableVector(Ty))
09467b48Spatrick    return false;
73471bf0Spatrick
73471bf0Spatrick  switch (RdxDesc.getRecurrenceKind()) {
73471bf0Spatrick  case RecurKind::Add:
73471bf0Spatrick  case RecurKind::FAdd:
73471bf0Spatrick  case RecurKind::And:
73471bf0Spatrick  case RecurKind::Or:
73471bf0Spatrick  case RecurKind::Xor:
73471bf0Spatrick  case RecurKind::SMin:
73471bf0Spatrick  case RecurKind::SMax:
73471bf0Spatrick  case RecurKind::UMin:
73471bf0Spatrick  case RecurKind::UMax:
73471bf0Spatrick  case RecurKind::FMin:
73471bf0Spatrick  case RecurKind::FMax:
*d415bd75Srobert  case RecurKind::SelectICmp:
*d415bd75Srobert  case RecurKind::SelectFCmp:
*d415bd75Srobert  case RecurKind::FMulAdd:
73471bf0Spatrick    return true;
09467b48Spatrick  default:
09467b48Spatrick    return false;
09467b48Spatrick  }
73471bf0Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost
73471bf0SpatrickAArch64TTIImpl::getMinMaxReductionCost(VectorType *Ty, VectorType *CondTy,
73471bf0Spatrick                                       bool IsUnsigned,
097a140dSpatrick                                       TTI::TargetCostKind CostKind) {
*d415bd75Srobert  std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
09467b48Spatrick
*d415bd75Srobert  if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
*d415bd75Srobert    return BaseT::getMinMaxReductionCost(Ty, CondTy, IsUnsigned, CostKind);
*d415bd75Srobert
*d415bd75Srobert  assert((isa<ScalableVectorType>(Ty) == isa<ScalableVectorType>(CondTy)) &&
*d415bd75Srobert         "Both vector needs to be equally scalable");
*d415bd75Srobert
73471bf0Spatrick  InstructionCost LegalizationCost = 0;
73471bf0Spatrick  if (LT.first > 1) {
73471bf0Spatrick    Type *LegalVTy = EVT(LT.second).getTypeForEVT(Ty->getContext());
*d415bd75Srobert    unsigned MinMaxOpcode =
*d415bd75Srobert        Ty->isFPOrFPVectorTy()
*d415bd75Srobert            ? Intrinsic::maxnum
*d415bd75Srobert            : (IsUnsigned ? Intrinsic::umin : Intrinsic::smin);
*d415bd75Srobert    IntrinsicCostAttributes Attrs(MinMaxOpcode, LegalVTy, {LegalVTy, LegalVTy});
*d415bd75Srobert    LegalizationCost = getIntrinsicInstrCost(Attrs, CostKind) * (LT.first - 1);
73471bf0Spatrick  }
09467b48Spatrick
73471bf0Spatrick  return LegalizationCost + /*Cost of horizontal reduction*/ 2;
73471bf0Spatrick}
73471bf0Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getArithmeticReductionCostSVE(
73471bf0Spatrick    unsigned Opcode, VectorType *ValTy, TTI::TargetCostKind CostKind) {
*d415bd75Srobert  std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
73471bf0Spatrick  InstructionCost LegalizationCost = 0;
73471bf0Spatrick  if (LT.first > 1) {
73471bf0Spatrick    Type *LegalVTy = EVT(LT.second).getTypeForEVT(ValTy->getContext());
73471bf0Spatrick    LegalizationCost = getArithmeticInstrCost(Opcode, LegalVTy, CostKind);
73471bf0Spatrick    LegalizationCost *= LT.first - 1;
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  int ISD = TLI->InstructionOpcodeToISD(Opcode);
73471bf0Spatrick  assert(ISD && "Invalid opcode");
73471bf0Spatrick  // Add the final reduction cost for the legal horizontal reduction
73471bf0Spatrick  switch (ISD) {
73471bf0Spatrick  case ISD::ADD:
73471bf0Spatrick  case ISD::AND:
73471bf0Spatrick  case ISD::OR:
73471bf0Spatrick  case ISD::XOR:
73471bf0Spatrick  case ISD::FADD:
73471bf0Spatrick    return LegalizationCost + 2;
73471bf0Spatrick  default:
73471bf0Spatrick    return InstructionCost::getInvalid();
73471bf0Spatrick  }
73471bf0Spatrick}
73471bf0Spatrick
73471bf0SpatrickInstructionCost
73471bf0SpatrickAArch64TTIImpl::getArithmeticReductionCost(unsigned Opcode, VectorType *ValTy,
*d415bd75Srobert                                           std::optional<FastMathFlags> FMF,
73471bf0Spatrick                                           TTI::TargetCostKind CostKind) {
73471bf0Spatrick  if (TTI::requiresOrderedReduction(FMF)) {
*d415bd75Srobert    if (auto *FixedVTy = dyn_cast<FixedVectorType>(ValTy)) {
*d415bd75Srobert      InstructionCost BaseCost =
*d415bd75Srobert          BaseT::getArithmeticReductionCost(Opcode, ValTy, FMF, CostKind);
*d415bd75Srobert      // Add on extra cost to reflect the extra overhead on some CPUs. We still
*d415bd75Srobert      // end up vectorizing for more computationally intensive loops.
*d415bd75Srobert      return BaseCost + FixedVTy->getNumElements();
*d415bd75Srobert    }
73471bf0Spatrick
73471bf0Spatrick    if (Opcode != Instruction::FAdd)
73471bf0Spatrick      return InstructionCost::getInvalid();
73471bf0Spatrick
73471bf0Spatrick    auto *VTy = cast<ScalableVectorType>(ValTy);
73471bf0Spatrick    InstructionCost Cost =
73471bf0Spatrick        getArithmeticInstrCost(Opcode, VTy->getScalarType(), CostKind);
73471bf0Spatrick    Cost *= getMaxNumElements(VTy->getElementCount());
73471bf0Spatrick    return Cost;
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  if (isa<ScalableVectorType>(ValTy))
73471bf0Spatrick    return getArithmeticReductionCostSVE(Opcode, ValTy, CostKind);
73471bf0Spatrick
*d415bd75Srobert  std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
09467b48Spatrick  MVT MTy = LT.second;
09467b48Spatrick  int ISD = TLI->InstructionOpcodeToISD(Opcode);
09467b48Spatrick  assert(ISD && "Invalid opcode");
09467b48Spatrick
09467b48Spatrick  // Horizontal adds can use the 'addv' instruction. We model the cost of these
73471bf0Spatrick  // instructions as twice a normal vector add, plus 1 for each legalization
73471bf0Spatrick  // step (LT.first). This is the only arithmetic vector reduction operation for
73471bf0Spatrick  // which we have an instruction.
73471bf0Spatrick  // OR, XOR and AND costs should match the codegen from:
73471bf0Spatrick  // OR: llvm/test/CodeGen/AArch64/reduce-or.ll
73471bf0Spatrick  // XOR: llvm/test/CodeGen/AArch64/reduce-xor.ll
73471bf0Spatrick  // AND: llvm/test/CodeGen/AArch64/reduce-and.ll
09467b48Spatrick  static const CostTblEntry CostTblNoPairwise[]{
73471bf0Spatrick      {ISD::ADD, MVT::v8i8,   2},
73471bf0Spatrick      {ISD::ADD, MVT::v16i8,  2},
73471bf0Spatrick      {ISD::ADD, MVT::v4i16,  2},
73471bf0Spatrick      {ISD::ADD, MVT::v8i16,  2},
73471bf0Spatrick      {ISD::ADD, MVT::v4i32,  2},
*d415bd75Srobert      {ISD::ADD, MVT::v2i64,  2},
73471bf0Spatrick      {ISD::OR,  MVT::v8i8,  15},
73471bf0Spatrick      {ISD::OR,  MVT::v16i8, 17},
73471bf0Spatrick      {ISD::OR,  MVT::v4i16,  7},
73471bf0Spatrick      {ISD::OR,  MVT::v8i16,  9},
73471bf0Spatrick      {ISD::OR,  MVT::v2i32,  3},
73471bf0Spatrick      {ISD::OR,  MVT::v4i32,  5},
73471bf0Spatrick      {ISD::OR,  MVT::v2i64,  3},
73471bf0Spatrick      {ISD::XOR, MVT::v8i8,  15},
73471bf0Spatrick      {ISD::XOR, MVT::v16i8, 17},
73471bf0Spatrick      {ISD::XOR, MVT::v4i16,  7},
73471bf0Spatrick      {ISD::XOR, MVT::v8i16,  9},
73471bf0Spatrick      {ISD::XOR, MVT::v2i32,  3},
73471bf0Spatrick      {ISD::XOR, MVT::v4i32,  5},
73471bf0Spatrick      {ISD::XOR, MVT::v2i64,  3},
73471bf0Spatrick      {ISD::AND, MVT::v8i8,  15},
73471bf0Spatrick      {ISD::AND, MVT::v16i8, 17},
73471bf0Spatrick      {ISD::AND, MVT::v4i16,  7},
73471bf0Spatrick      {ISD::AND, MVT::v8i16,  9},
73471bf0Spatrick      {ISD::AND, MVT::v2i32,  3},
73471bf0Spatrick      {ISD::AND, MVT::v4i32,  5},
73471bf0Spatrick      {ISD::AND, MVT::v2i64,  3},
09467b48Spatrick  };
73471bf0Spatrick  switch (ISD) {
73471bf0Spatrick  default:
73471bf0Spatrick    break;
73471bf0Spatrick  case ISD::ADD:
09467b48Spatrick    if (const auto *Entry = CostTableLookup(CostTblNoPairwise, ISD, MTy))
73471bf0Spatrick      return (LT.first - 1) + Entry->Cost;
73471bf0Spatrick    break;
73471bf0Spatrick  case ISD::XOR:
73471bf0Spatrick  case ISD::AND:
73471bf0Spatrick  case ISD::OR:
73471bf0Spatrick    const auto *Entry = CostTableLookup(CostTblNoPairwise, ISD, MTy);
73471bf0Spatrick    if (!Entry)
73471bf0Spatrick      break;
73471bf0Spatrick    auto *ValVTy = cast<FixedVectorType>(ValTy);
73471bf0Spatrick    if (!ValVTy->getElementType()->isIntegerTy(1) &&
73471bf0Spatrick        MTy.getVectorNumElements() <= ValVTy->getNumElements() &&
73471bf0Spatrick        isPowerOf2_32(ValVTy->getNumElements())) {
73471bf0Spatrick      InstructionCost ExtraCost = 0;
73471bf0Spatrick      if (LT.first != 1) {
73471bf0Spatrick        // Type needs to be split, so there is an extra cost of LT.first - 1
73471bf0Spatrick        // arithmetic ops.
73471bf0Spatrick        auto *Ty = FixedVectorType::get(ValTy->getElementType(),
73471bf0Spatrick                                        MTy.getVectorNumElements());
73471bf0Spatrick        ExtraCost = getArithmeticInstrCost(Opcode, Ty, CostKind);
73471bf0Spatrick        ExtraCost *= LT.first - 1;
73471bf0Spatrick      }
73471bf0Spatrick      return Entry->Cost + ExtraCost;
73471bf0Spatrick    }
73471bf0Spatrick    break;
73471bf0Spatrick  }
73471bf0Spatrick  return BaseT::getArithmeticReductionCost(Opcode, ValTy, FMF, CostKind);
09467b48Spatrick}
09467b48Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getSpliceCost(VectorType *Tp, int Index) {
73471bf0Spatrick  static const CostTblEntry ShuffleTbl[] = {
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv16i8,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv8i16,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv4i32,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv2i64,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv2f16,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv4f16,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv8f16,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv2bf16, 1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv4bf16, 1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv8bf16, 1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv2f32,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv4f32,  1 },
73471bf0Spatrick      { TTI::SK_Splice, MVT::nxv2f64,  1 },
73471bf0Spatrick  };
73471bf0Spatrick
*d415bd75Srobert  // The code-generator is currently not able to handle scalable vectors
*d415bd75Srobert  // of <vscale x 1 x eltty> yet, so return an invalid cost to avoid selecting
*d415bd75Srobert  // it. This change will be removed when code-generation for these types is
*d415bd75Srobert  // sufficiently reliable.
*d415bd75Srobert  if (Tp->getElementCount() == ElementCount::getScalable(1))
*d415bd75Srobert    return InstructionCost::getInvalid();
*d415bd75Srobert
*d415bd75Srobert  std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
73471bf0Spatrick  Type *LegalVTy = EVT(LT.second).getTypeForEVT(Tp->getContext());
73471bf0Spatrick  TTI::TargetCostKind CostKind = TTI::TCK_RecipThroughput;
73471bf0Spatrick  EVT PromotedVT = LT.second.getScalarType() == MVT::i1
73471bf0Spatrick                       ? TLI->getPromotedVTForPredicate(EVT(LT.second))
73471bf0Spatrick                       : LT.second;
73471bf0Spatrick  Type *PromotedVTy = EVT(PromotedVT).getTypeForEVT(Tp->getContext());
73471bf0Spatrick  InstructionCost LegalizationCost = 0;
73471bf0Spatrick  if (Index < 0) {
73471bf0Spatrick    LegalizationCost =
73471bf0Spatrick        getCmpSelInstrCost(Instruction::ICmp, PromotedVTy, PromotedVTy,
73471bf0Spatrick                           CmpInst::BAD_ICMP_PREDICATE, CostKind) +
73471bf0Spatrick        getCmpSelInstrCost(Instruction::Select, PromotedVTy, LegalVTy,
73471bf0Spatrick                           CmpInst::BAD_ICMP_PREDICATE, CostKind);
73471bf0Spatrick  }
73471bf0Spatrick
73471bf0Spatrick  // Predicated splice are promoted when lowering. See AArch64ISelLowering.cpp
73471bf0Spatrick  // Cost performed on a promoted type.
73471bf0Spatrick  if (LT.second.getScalarType() == MVT::i1) {
73471bf0Spatrick    LegalizationCost +=
73471bf0Spatrick        getCastInstrCost(Instruction::ZExt, PromotedVTy, LegalVTy,
73471bf0Spatrick                         TTI::CastContextHint::None, CostKind) +
73471bf0Spatrick        getCastInstrCost(Instruction::Trunc, LegalVTy, PromotedVTy,
73471bf0Spatrick                         TTI::CastContextHint::None, CostKind);
73471bf0Spatrick  }
73471bf0Spatrick  const auto *Entry =
73471bf0Spatrick      CostTableLookup(ShuffleTbl, TTI::SK_Splice, PromotedVT.getSimpleVT());
73471bf0Spatrick  assert(Entry && "Illegal Type for Splice");
73471bf0Spatrick  LegalizationCost += Entry->Cost;
73471bf0Spatrick  return LegalizationCost * LT.first;
73471bf0Spatrick}
73471bf0Spatrick
73471bf0SpatrickInstructionCost AArch64TTIImpl::getShuffleCost(TTI::ShuffleKind Kind,
73471bf0Spatrick                                               VectorType *Tp,
*d415bd75Srobert                                               ArrayRef<int> Mask,
*d415bd75Srobert                                               TTI::TargetCostKind CostKind,
*d415bd75Srobert                                               int Index, VectorType *SubTp,
*d415bd75Srobert                                               ArrayRef<const Value *> Args) {
*d415bd75Srobert  std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
*d415bd75Srobert  // If we have a Mask, and the LT is being legalized somehow, split the Mask
*d415bd75Srobert  // into smaller vectors and sum the cost of each shuffle.
*d415bd75Srobert  if (!Mask.empty() && isa<FixedVectorType>(Tp) && LT.second.isVector() &&
*d415bd75Srobert      Tp->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
*d415bd75Srobert      cast<FixedVectorType>(Tp)->getNumElements() >
*d415bd75Srobert          LT.second.getVectorNumElements() &&
*d415bd75Srobert      !Index && !SubTp) {
*d415bd75Srobert    unsigned TpNumElts = cast<FixedVectorType>(Tp)->getNumElements();
*d415bd75Srobert    assert(Mask.size() == TpNumElts && "Expected Mask and Tp size to match!");
*d415bd75Srobert    unsigned LTNumElts = LT.second.getVectorNumElements();
*d415bd75Srobert    unsigned NumVecs = (TpNumElts + LTNumElts - 1) / LTNumElts;
*d415bd75Srobert    VectorType *NTp =
*d415bd75Srobert        VectorType::get(Tp->getScalarType(), LT.second.getVectorElementCount());
*d415bd75Srobert    InstructionCost Cost;
*d415bd75Srobert    for (unsigned N = 0; N < NumVecs; N++) {
*d415bd75Srobert      SmallVector<int> NMask;
*d415bd75Srobert      // Split the existing mask into chunks of size LTNumElts. Track the source
*d415bd75Srobert      // sub-vectors to ensure the result has at most 2 inputs.
*d415bd75Srobert      unsigned Source1, Source2;
*d415bd75Srobert      unsigned NumSources = 0;
*d415bd75Srobert      for (unsigned E = 0; E < LTNumElts; E++) {
*d415bd75Srobert        int MaskElt = (N * LTNumElts + E < TpNumElts) ? Mask[N * LTNumElts + E]
*d415bd75Srobert                                                      : UndefMaskElem;
*d415bd75Srobert        if (MaskElt < 0) {
*d415bd75Srobert          NMask.push_back(UndefMaskElem);
*d415bd75Srobert          continue;
*d415bd75Srobert        }
*d415bd75Srobert
*d415bd75Srobert        // Calculate which source from the input this comes from and whether it
*d415bd75Srobert        // is new to us.
*d415bd75Srobert        unsigned Source = MaskElt / LTNumElts;
*d415bd75Srobert        if (NumSources == 0) {
*d415bd75Srobert          Source1 = Source;
*d415bd75Srobert          NumSources = 1;
*d415bd75Srobert        } else if (NumSources == 1 && Source != Source1) {
*d415bd75Srobert          Source2 = Source;
*d415bd75Srobert          NumSources = 2;
*d415bd75Srobert        } else if (NumSources >= 2 && Source != Source1 && Source != Source2) {
*d415bd75Srobert          NumSources++;
*d415bd75Srobert        }
*d415bd75Srobert
*d415bd75Srobert        // Add to the new mask. For the NumSources>2 case these are not correct,
*d415bd75Srobert        // but are only used for the modular lane number.
*d415bd75Srobert        if (Source == Source1)
*d415bd75Srobert          NMask.push_back(MaskElt % LTNumElts);
*d415bd75Srobert        else if (Source == Source2)
*d415bd75Srobert          NMask.push_back(MaskElt % LTNumElts + LTNumElts);
*d415bd75Srobert        else
*d415bd75Srobert          NMask.push_back(MaskElt % LTNumElts);
*d415bd75Srobert      }
*d415bd75Srobert      // If the sub-mask has at most 2 input sub-vectors then re-cost it using
*d415bd75Srobert      // getShuffleCost. If not then cost it using the worst case.
*d415bd75Srobert      if (NumSources <= 2)
*d415bd75Srobert        Cost += getShuffleCost(NumSources <= 1 ? TTI::SK_PermuteSingleSrc
*d415bd75Srobert                                               : TTI::SK_PermuteTwoSrc,
*d415bd75Srobert                               NTp, NMask, CostKind, 0, nullptr, Args);
*d415bd75Srobert      else if (any_of(enumerate(NMask), [&](const auto &ME) {
*d415bd75Srobert                 return ME.value() % LTNumElts == ME.index();
*d415bd75Srobert               }))
*d415bd75Srobert        Cost += LTNumElts - 1;
*d415bd75Srobert      else
*d415bd75Srobert        Cost += LTNumElts;
*d415bd75Srobert    }
*d415bd75Srobert    return Cost;
*d415bd75Srobert  }
*d415bd75Srobert
73471bf0Spatrick  Kind = improveShuffleKindFromMask(Kind, Mask);
*d415bd75Srobert
*d415bd75Srobert  // Check for broadcast loads.
*d415bd75Srobert  if (Kind == TTI::SK_Broadcast) {
*d415bd75Srobert    bool IsLoad = !Args.empty() && isa<LoadInst>(Args[0]);
*d415bd75Srobert    if (IsLoad && LT.second.isVector() &&
*d415bd75Srobert        isLegalBroadcastLoad(Tp->getElementType(),
*d415bd75Srobert                             LT.second.getVectorElementCount()))
*d415bd75Srobert      return 0; // broadcast is handled by ld1r
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  // If we have 4 elements for the shuffle and a Mask, get the cost straight
*d415bd75Srobert  // from the perfect shuffle tables.
*d415bd75Srobert  if (Mask.size() == 4 && Tp->getElementCount() == ElementCount::getFixed(4) &&
*d415bd75Srobert      (Tp->getScalarSizeInBits() == 16 || Tp->getScalarSizeInBits() == 32) &&
*d415bd75Srobert      all_of(Mask, [](int E) { return E < 8; }))
*d415bd75Srobert    return getPerfectShuffleCost(Mask);
*d415bd75Srobert
09467b48Spatrick  if (Kind == TTI::SK_Broadcast || Kind == TTI::SK_Transpose ||
73471bf0Spatrick      Kind == TTI::SK_Select || Kind == TTI::SK_PermuteSingleSrc ||
*d415bd75Srobert      Kind == TTI::SK_Reverse || Kind == TTI::SK_Splice) {
09467b48Spatrick    static const CostTblEntry ShuffleTbl[] = {
09467b48Spatrick        // Broadcast shuffle kinds can be performed with 'dup'.
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v8i8, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v16i8, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v4i16, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v8i16, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v2i32, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v4i32, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v2i64, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v2f32, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v4f32, 1},
09467b48Spatrick        {TTI::SK_Broadcast, MVT::v2f64, 1},
09467b48Spatrick        // Transpose shuffle kinds can be performed with 'trn1/trn2' and
09467b48Spatrick        // 'zip1/zip2' instructions.
09467b48Spatrick        {TTI::SK_Transpose, MVT::v8i8, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v16i8, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v4i16, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v8i16, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v2i32, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v4i32, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v2i64, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v2f32, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v4f32, 1},
09467b48Spatrick        {TTI::SK_Transpose, MVT::v2f64, 1},
09467b48Spatrick        // Select shuffle kinds.
09467b48Spatrick        // TODO: handle vXi8/vXi16.
09467b48Spatrick        {TTI::SK_Select, MVT::v2i32, 1}, // mov.
09467b48Spatrick        {TTI::SK_Select, MVT::v4i32, 2}, // rev+trn (or similar).
09467b48Spatrick        {TTI::SK_Select, MVT::v2i64, 1}, // mov.
09467b48Spatrick        {TTI::SK_Select, MVT::v2f32, 1}, // mov.
09467b48Spatrick        {TTI::SK_Select, MVT::v4f32, 2}, // rev+trn (or similar).
09467b48Spatrick        {TTI::SK_Select, MVT::v2f64, 1}, // mov.
09467b48Spatrick        // PermuteSingleSrc shuffle kinds.
09467b48Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v2i32, 1}, // mov.
09467b48Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v4i32, 3}, // perfectshuffle worst case.
09467b48Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v2i64, 1}, // mov.
09467b48Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v2f32, 1}, // mov.
09467b48Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v4f32, 3}, // perfectshuffle worst case.
09467b48Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v2f64, 1}, // mov.
73471bf0Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v4i16, 3}, // perfectshuffle worst case.
73471bf0Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v4f16, 3}, // perfectshuffle worst case.
*d415bd75Srobert        {TTI::SK_PermuteSingleSrc, MVT::v4bf16, 3}, // same
73471bf0Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v8i16, 8},  // constpool + load + tbl
73471bf0Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v8f16, 8},  // constpool + load + tbl
73471bf0Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v8bf16, 8}, // constpool + load + tbl
73471bf0Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v8i8, 8},   // constpool + load + tbl
73471bf0Spatrick        {TTI::SK_PermuteSingleSrc, MVT::v16i8, 8},  // constpool + load + tbl
73471bf0Spatrick        // Reverse can be lowered with `rev`.
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v2i32, 1}, // REV64
73471bf0Spatrick        {TTI::SK_Reverse, MVT::v4i32, 2}, // REV64; EXT
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v2i64, 1}, // EXT
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v2f32, 1}, // REV64
73471bf0Spatrick        {TTI::SK_Reverse, MVT::v4f32, 2}, // REV64; EXT
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v2f64, 1}, // EXT
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v8f16, 2}, // REV64; EXT
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v8i16, 2}, // REV64; EXT
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v16i8, 2}, // REV64; EXT
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v4f16, 1}, // REV64
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v4i16, 1}, // REV64
*d415bd75Srobert        {TTI::SK_Reverse, MVT::v8i8, 1},  // REV64
*d415bd75Srobert        // Splice can all be lowered as `ext`.
*d415bd75Srobert        {TTI::SK_Splice, MVT::v2i32, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v4i32, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v2i64, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v2f32, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v4f32, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v2f64, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v8f16, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v8bf16, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v8i16, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v16i8, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v4bf16, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v4f16, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v4i16, 1},
*d415bd75Srobert        {TTI::SK_Splice, MVT::v8i8, 1},
73471bf0Spatrick        // Broadcast shuffle kinds for scalable vectors
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv16i8, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv8i16, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv4i32, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv2i64, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv2f16, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv4f16, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv8f16, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv2bf16, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv4bf16, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv8bf16, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv2f32, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv4f32, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv2f64, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv16i1, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv8i1, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv4i1, 1},
73471bf0Spatrick        {TTI::SK_Broadcast, MVT::nxv2i1, 1},
73471bf0Spatrick        // Handle the cases for vector.reverse with scalable vectors
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv16i8, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv8i16, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv4i32, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv2i64, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv2f16, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv4f16, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv8f16, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv2bf16, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv4bf16, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv8bf16, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv2f32, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv4f32, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv2f64, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv16i1, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv8i1, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv4i1, 1},
73471bf0Spatrick        {TTI::SK_Reverse, MVT::nxv2i1, 1},
09467b48Spatrick    };
09467b48Spatrick    if (const auto *Entry = CostTableLookup(ShuffleTbl, Kind, LT.second))
09467b48Spatrick      return LT.first * Entry->Cost;
09467b48Spatrick  }
*d415bd75Srobert
73471bf0Spatrick  if (Kind == TTI::SK_Splice && isa<ScalableVectorType>(Tp))
73471bf0Spatrick    return getSpliceCost(Tp, Index);
*d415bd75Srobert
*d415bd75Srobert  // Inserting a subvector can often be done with either a D, S or H register
*d415bd75Srobert  // move, so long as the inserted vector is "aligned".
*d415bd75Srobert  if (Kind == TTI::SK_InsertSubvector && LT.second.isFixedLengthVector() &&
*d415bd75Srobert      LT.second.getSizeInBits() <= 128 && SubTp) {
*d415bd75Srobert    std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(SubTp);
*d415bd75Srobert    if (SubLT.second.isVector()) {
*d415bd75Srobert      int NumElts = LT.second.getVectorNumElements();
*d415bd75Srobert      int NumSubElts = SubLT.second.getVectorNumElements();
*d415bd75Srobert      if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
*d415bd75Srobert        return SubLT.first;
*d415bd75Srobert    }
*d415bd75Srobert  }
*d415bd75Srobert
*d415bd75Srobert  return BaseT::getShuffleCost(Kind, Tp, Mask, CostKind, Index, SubTp);
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75Srobertbool AArch64TTIImpl::preferPredicateOverEpilogue(
*d415bd75Srobert    Loop *L, LoopInfo *LI, ScalarEvolution &SE, AssumptionCache &AC,
*d415bd75Srobert    TargetLibraryInfo *TLI, DominatorTree *DT, LoopVectorizationLegality *LVL,
*d415bd75Srobert    InterleavedAccessInfo *IAI) {
*d415bd75Srobert  if (!ST->hasSVE() || TailFoldingKindLoc == TailFoldingKind::TFDisabled)
*d415bd75Srobert    return false;
*d415bd75Srobert
*d415bd75Srobert  // We don't currently support vectorisation with interleaving for SVE - with
*d415bd75Srobert  // such loops we're better off not using tail-folding. This gives us a chance
*d415bd75Srobert  // to fall back on fixed-width vectorisation using NEON's ld2/st2/etc.
*d415bd75Srobert  if (IAI->hasGroups())
*d415bd75Srobert    return false;
*d415bd75Srobert
*d415bd75Srobert  TailFoldingKind Required; // Defaults to 0.
*d415bd75Srobert  if (LVL->getReductionVars().size())
*d415bd75Srobert    Required.add(TailFoldingKind::TFReductions);
*d415bd75Srobert  if (LVL->getFixedOrderRecurrences().size())
*d415bd75Srobert    Required.add(TailFoldingKind::TFRecurrences);
*d415bd75Srobert  if (!Required)
*d415bd75Srobert    Required.add(TailFoldingKind::TFSimple);
*d415bd75Srobert
*d415bd75Srobert  return (TailFoldingKindLoc & Required) == Required;
*d415bd75Srobert}
*d415bd75Srobert
*d415bd75SrobertInstructionCost
*d415bd75SrobertAArch64TTIImpl::getScalingFactorCost(Type *Ty, GlobalValue *BaseGV,
*d415bd75Srobert                                     int64_t BaseOffset, bool HasBaseReg,
*d415bd75Srobert                                     int64_t Scale, unsigned AddrSpace) const {
*d415bd75Srobert  // Scaling factors are not free at all.
*d415bd75Srobert  // Operands                     | Rt Latency
*d415bd75Srobert  // -------------------------------------------
*d415bd75Srobert  // Rt, [Xn, Xm]                 | 4
*d415bd75Srobert  // -------------------------------------------
*d415bd75Srobert  // Rt, [Xn, Xm, lsl #imm]       | Rn: 4 Rm: 5
*d415bd75Srobert  // Rt, [Xn, Wm, <extend> #imm]  |
*d415bd75Srobert  TargetLoweringBase::AddrMode AM;
*d415bd75Srobert  AM.BaseGV = BaseGV;
*d415bd75Srobert  AM.BaseOffs = BaseOffset;
*d415bd75Srobert  AM.HasBaseReg = HasBaseReg;
*d415bd75Srobert  AM.Scale = Scale;
*d415bd75Srobert  if (getTLI()->isLegalAddressingMode(DL, AM, Ty, AddrSpace))
*d415bd75Srobert    // Scale represents reg2 * scale, thus account for 1 if
*d415bd75Srobert    // it is not equal to 0 or 1.
*d415bd75Srobert    return AM.Scale != 0 && AM.Scale != 1;
*d415bd75Srobert  return -1;
09467b48Spatrick}