LLVM8Doxygen/ARMTargetTransformInfo_8cpp_source.html

 //===- ARMTargetTransformInfo.cpp - ARM specific TTI ----------------------===//
 //
 //                     The LLVM Compiler Infrastructure
 //
 // This file is distributed under the University of Illinois Open Source
 // License. See LICENSE.TXT for details.
 //
 //===----------------------------------------------------------------------===//

 #include "ARMTargetTransformInfo.h"
 #include "ARMSubtarget.h"
 #include "MCTargetDesc/ARMAddressingModes.h"
 #include "llvm/ADT/APInt.h"
 #include "llvm/ADT/SmallVector.h"
 #include "llvm/Analysis/LoopInfo.h"
 #include "llvm/CodeGen/CostTable.h"
 #include "llvm/CodeGen/ISDOpcodes.h"
 #include "llvm/CodeGen/ValueTypes.h"
 #include "llvm/IR/BasicBlock.h"
 #include "llvm/IR/CallSite.h"
 #include "llvm/IR/DataLayout.h"
 #include "llvm/IR/DerivedTypes.h"
 #include "llvm/IR/Instruction.h"
 #include "llvm/IR/Instructions.h"
 #include "llvm/IR/Type.h"
 #include "llvm/MC/SubtargetFeature.h"
 #include "llvm/Support/Casting.h"
 #include "llvm/Support/MachineValueType.h"
 #include "llvm/Target/TargetMachine.h"
 #include <algorithm>
 #include <cassert>
 #include <cstdint>
 #include <utility>

 using namespace llvm;

 #define DEBUG_TYPE "armtti"

 bool ARMTTIImpl::areInlineCompatible(const Function *Caller,
                                      const Function *Callee) const {
   const TargetMachine &TM = getTLI()->getTargetMachine();
   const FeatureBitset &CallerBits =
       TM.getSubtargetImpl(*Caller)->getFeatureBits();
   const FeatureBitset &CalleeBits =
       TM.getSubtargetImpl(*Callee)->getFeatureBits();

   // To inline a callee, all features not in the whitelist must match exactly.
   bool MatchExact = (CallerBits & ~InlineFeatureWhitelist) ==
                     (CalleeBits & ~InlineFeatureWhitelist);
   // For features in the whitelist, the callee's features must be a subset of
   // the callers'.
   bool MatchSubset = ((CallerBits & CalleeBits) & InlineFeatureWhitelist) ==
                      (CalleeBits & InlineFeatureWhitelist);
   return MatchExact && MatchSubset;
 }

 int ARMTTIImpl::getIntImmCost(const APInt &Imm, Type *Ty) {
   assert(Ty->isIntegerTy());

  unsigned Bits = Ty->getPrimitiveSizeInBits();
  if (Bits == 0 || Imm.getActiveBits() >= 64)
    return 4;

   int64_t SImmVal = Imm.getSExtValue();
   uint64_t ZImmVal = Imm.getZExtValue();
   if (!ST->isThumb()) {
     if ((SImmVal >= 0 && SImmVal < 65536) ||
         (ARM_AM::getSOImmVal(ZImmVal) != -1) ||
         (ARM_AM::getSOImmVal(~ZImmVal) != -1))
       return 1;
     return ST->hasV6T2Ops() ? 2 : 3;
   }
   if (ST->isThumb2()) {
     if ((SImmVal >= 0 && SImmVal < 65536) ||
         (ARM_AM::getT2SOImmVal(ZImmVal) != -1) ||
         (ARM_AM::getT2SOImmVal(~ZImmVal) != -1))
       return 1;
     return ST->hasV6T2Ops() ? 2 : 3;
   }
   // Thumb1, any i8 imm cost 1.
   if (Bits == 8 || (SImmVal >= 0 && SImmVal < 256))
     return 1;
   if ((~SImmVal < 256) || ARM_AM::isThumbImmShiftedVal(ZImmVal))
     return 2;
   // Load from constantpool.
   return 3;
 }

 // Constants smaller than 256 fit in the immediate field of
 // Thumb1 instructions so we return a zero cost and 1 otherwise.
 int ARMTTIImpl::getIntImmCodeSizeCost(unsigned Opcode, unsigned Idx,
                                       const APInt &Imm, Type *Ty) {
   if (Imm.isNonNegative() && Imm.getLimitedValue() < 256)
     return 0;

   return 1;
 }

 int ARMTTIImpl::getIntImmCost(unsigned Opcode, unsigned Idx, const APInt &Imm,
                               Type *Ty) {
   // Division by a constant can be turned into multiplication, but only if we
   // know it's constant. So it's not so much that the immediate is cheap (it's
   // not), but that the alternative is worse.
   // FIXME: this is probably unneeded with GlobalISel.
   if ((Opcode == Instruction::SDiv || Opcode == Instruction::UDiv ||
        Opcode == Instruction::SRem || Opcode == Instruction::URem) &&
       Idx == 1)
     return 0;

   if (Opcode == Instruction::And)
       // Conversion to BIC is free, and means we can use ~Imm instead.
       return std::min(getIntImmCost(Imm, Ty), getIntImmCost(~Imm, Ty));

   if (Opcode == Instruction::Add)
     // Conversion to SUB is free, and means we can use -Imm instead.
     return std::min(getIntImmCost(Imm, Ty), getIntImmCost(-Imm, Ty));

   if (Opcode == Instruction::ICmp && Imm.isNegative() &&
       Ty->getIntegerBitWidth() == 32) {
     int64_t NegImm = -Imm.getSExtValue();
     if (ST->isThumb2() && NegImm < 1<<12)
       // icmp X, #-C -> cmn X, #C
       return 0;
     if (ST->isThumb() && NegImm < 1<<8)
       // icmp X, #-C -> adds X, #C
       return 0;
   }

   // xor a, -1 can always be folded to MVN
   if (Opcode == Instruction::Xor && Imm.isAllOnesValue())
     return 0;

   return getIntImmCost(Imm, Ty);
 }

 int ARMTTIImpl::getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src,
                                  const Instruction *I) {
   int ISD = TLI->InstructionOpcodeToISD(Opcode);
   assert(ISD && "Invalid opcode");

   // Single to/from double precision conversions.
   static const CostTblEntry NEONFltDblTbl[] = {
     // Vector fptrunc/fpext conversions.
     { ISD::FP_ROUND,   MVT::v2f64, 2 },
     { ISD::FP_EXTEND,  MVT::v2f32, 2 },
     { ISD::FP_EXTEND,  MVT::v4f32, 4 }
   };

   if (Src->isVectorTy() && ST->hasNEON() && (ISD == ISD::FP_ROUND ||
                                           ISD == ISD::FP_EXTEND)) {
     std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, Src);
     if (const auto *Entry = CostTableLookup(NEONFltDblTbl, ISD, LT.second))
       return LT.first * Entry->Cost;
   }

   EVT SrcTy = TLI->getValueType(DL, Src);
   EVT DstTy = TLI->getValueType(DL, Dst);

   if (!SrcTy.isSimple() || !DstTy.isSimple())
     return BaseT::getCastInstrCost(Opcode, Dst, Src);

   // Some arithmetic, load and store operations have specific instructions
   // to cast up/down their types automatically at no extra cost.
   // TODO: Get these tables to know at least what the related operations are.
   static const TypeConversionCostTblEntry NEONVectorConversionTbl[] = {
     { ISD::SIGN_EXTEND, MVT::v4i32, MVT::v4i16, 0 },
     { ISD::ZERO_EXTEND, MVT::v4i32, MVT::v4i16, 0 },
     { ISD::SIGN_EXTEND, MVT::v2i64, MVT::v2i32, 1 },
     { ISD::ZERO_EXTEND, MVT::v2i64, MVT::v2i32, 1 },
     { ISD::TRUNCATE,    MVT::v4i32, MVT::v4i64, 0 },
     { ISD::TRUNCATE,    MVT::v4i16, MVT::v4i32, 1 },

     // The number of vmovl instructions for the extension.
     { ISD::SIGN_EXTEND, MVT::v4i64, MVT::v4i16, 3 },
     { ISD::ZERO_EXTEND, MVT::v4i64, MVT::v4i16, 3 },
     { ISD::SIGN_EXTEND, MVT::v8i32, MVT::v8i8, 3 },
     { ISD::ZERO_EXTEND, MVT::v8i32, MVT::v8i8, 3 },
     { ISD::SIGN_EXTEND, MVT::v8i64, MVT::v8i8, 7 },
     { ISD::ZERO_EXTEND, MVT::v8i64, MVT::v8i8, 7 },
     { ISD::SIGN_EXTEND, MVT::v8i64, MVT::v8i16, 6 },
     { ISD::ZERO_EXTEND, MVT::v8i64, MVT::v8i16, 6 },
     { ISD::SIGN_EXTEND, MVT::v16i32, MVT::v16i8, 6 },
     { ISD::ZERO_EXTEND, MVT::v16i32, MVT::v16i8, 6 },

     // Operations that we legalize using splitting.
     { ISD::TRUNCATE,    MVT::v16i8, MVT::v16i32, 6 },
     { ISD::TRUNCATE,    MVT::v8i8, MVT::v8i32, 3 },

     // Vector float <-> i32 conversions.
     { ISD::SINT_TO_FP,  MVT::v4f32, MVT::v4i32, 1 },
     { ISD::UINT_TO_FP,  MVT::v4f32, MVT::v4i32, 1 },

     { ISD::SINT_TO_FP,  MVT::v2f32, MVT::v2i8, 3 },
     { ISD::UINT_TO_FP,  MVT::v2f32, MVT::v2i8, 3 },
     { ISD::SINT_TO_FP,  MVT::v2f32, MVT::v2i16, 2 },
     { ISD::UINT_TO_FP,  MVT::v2f32, MVT::v2i16, 2 },
     { ISD::SINT_TO_FP,  MVT::v2f32, MVT::v2i32, 1 },
     { ISD::UINT_TO_FP,  MVT::v2f32, MVT::v2i32, 1 },
     { ISD::SINT_TO_FP,  MVT::v4f32, MVT::v4i1, 3 },
     { ISD::UINT_TO_FP,  MVT::v4f32, MVT::v4i1, 3 },
     { ISD::SINT_TO_FP,  MVT::v4f32, MVT::v4i8, 3 },
     { ISD::UINT_TO_FP,  MVT::v4f32, MVT::v4i8, 3 },
     { ISD::SINT_TO_FP,  MVT::v4f32, MVT::v4i16, 2 },
     { ISD::UINT_TO_FP,  MVT::v4f32, MVT::v4i16, 2 },
     { ISD::SINT_TO_FP,  MVT::v8f32, MVT::v8i16, 4 },
     { ISD::UINT_TO_FP,  MVT::v8f32, MVT::v8i16, 4 },
     { ISD::SINT_TO_FP,  MVT::v8f32, MVT::v8i32, 2 },
     { ISD::UINT_TO_FP,  MVT::v8f32, MVT::v8i32, 2 },
     { ISD::SINT_TO_FP,  MVT::v16f32, MVT::v16i16, 8 },
     { ISD::UINT_TO_FP,  MVT::v16f32, MVT::v16i16, 8 },
     { ISD::SINT_TO_FP,  MVT::v16f32, MVT::v16i32, 4 },
     { ISD::UINT_TO_FP,  MVT::v16f32, MVT::v16i32, 4 },

     { ISD::FP_TO_SINT,  MVT::v4i32, MVT::v4f32, 1 },
     { ISD::FP_TO_UINT,  MVT::v4i32, MVT::v4f32, 1 },
     { ISD::FP_TO_SINT,  MVT::v4i8, MVT::v4f32, 3 },
     { ISD::FP_TO_UINT,  MVT::v4i8, MVT::v4f32, 3 },
     { ISD::FP_TO_SINT,  MVT::v4i16, MVT::v4f32, 2 },
     { ISD::FP_TO_UINT,  MVT::v4i16, MVT::v4f32, 2 },

     // Vector double <-> i32 conversions.
     { ISD::SINT_TO_FP,  MVT::v2f64, MVT::v2i32, 2 },
     { ISD::UINT_TO_FP,  MVT::v2f64, MVT::v2i32, 2 },

     { ISD::SINT_TO_FP,  MVT::v2f64, MVT::v2i8, 4 },
     { ISD::UINT_TO_FP,  MVT::v2f64, MVT::v2i8, 4 },
     { ISD::SINT_TO_FP,  MVT::v2f64, MVT::v2i16, 3 },
     { ISD::UINT_TO_FP,  MVT::v2f64, MVT::v2i16, 3 },
     { ISD::SINT_TO_FP,  MVT::v2f64, MVT::v2i32, 2 },
     { ISD::UINT_TO_FP,  MVT::v2f64, MVT::v2i32, 2 },

     { ISD::FP_TO_SINT,  MVT::v2i32, MVT::v2f64, 2 },
     { ISD::FP_TO_UINT,  MVT::v2i32, MVT::v2f64, 2 },
     { ISD::FP_TO_SINT,  MVT::v8i16, MVT::v8f32, 4 },
     { ISD::FP_TO_UINT,  MVT::v8i16, MVT::v8f32, 4 },
     { ISD::FP_TO_SINT,  MVT::v16i16, MVT::v16f32, 8 },
     { ISD::FP_TO_UINT,  MVT::v16i16, MVT::v16f32, 8 }
   };

   if (SrcTy.isVector() && ST->hasNEON()) {
     if (const auto *Entry = ConvertCostTableLookup(NEONVectorConversionTbl, ISD,
                                                    DstTy.getSimpleVT(),
                                                    SrcTy.getSimpleVT()))
       return Entry->Cost;
   }

   // Scalar float to integer conversions.
   static const TypeConversionCostTblEntry NEONFloatConversionTbl[] = {
     { ISD::FP_TO_SINT,  MVT::i1, MVT::f32, 2 },
     { ISD::FP_TO_UINT,  MVT::i1, MVT::f32, 2 },
     { ISD::FP_TO_SINT,  MVT::i1, MVT::f64, 2 },
     { ISD::FP_TO_UINT,  MVT::i1, MVT::f64, 2 },
     { ISD::FP_TO_SINT,  MVT::i8, MVT::f32, 2 },
     { ISD::FP_TO_UINT,  MVT::i8, MVT::f32, 2 },
     { ISD::FP_TO_SINT,  MVT::i8, MVT::f64, 2 },
     { ISD::FP_TO_UINT,  MVT::i8, MVT::f64, 2 },
     { ISD::FP_TO_SINT,  MVT::i16, MVT::f32, 2 },
     { ISD::FP_TO_UINT,  MVT::i16, MVT::f32, 2 },
     { ISD::FP_TO_SINT,  MVT::i16, MVT::f64, 2 },
     { ISD::FP_TO_UINT,  MVT::i16, MVT::f64, 2 },
     { ISD::FP_TO_SINT,  MVT::i32, MVT::f32, 2 },
     { ISD::FP_TO_UINT,  MVT::i32, MVT::f32, 2 },
     { ISD::FP_TO_SINT,  MVT::i32, MVT::f64, 2 },
     { ISD::FP_TO_UINT,  MVT::i32, MVT::f64, 2 },
     { ISD::FP_TO_SINT,  MVT::i64, MVT::f32, 10 },
     { ISD::FP_TO_UINT,  MVT::i64, MVT::f32, 10 },
     { ISD::FP_TO_SINT,  MVT::i64, MVT::f64, 10 },
     { ISD::FP_TO_UINT,  MVT::i64, MVT::f64, 10 }
   };
   if (SrcTy.isFloatingPoint() && ST->hasNEON()) {
     if (const auto *Entry = ConvertCostTableLookup(NEONFloatConversionTbl, ISD,
                                                    DstTy.getSimpleVT(),
                                                    SrcTy.getSimpleVT()))
       return Entry->Cost;
   }

   // Scalar integer to float conversions.
   static const TypeConversionCostTblEntry NEONIntegerConversionTbl[] = {
     { ISD::SINT_TO_FP,  MVT::f32, MVT::i1, 2 },
     { ISD::UINT_TO_FP,  MVT::f32, MVT::i1, 2 },
     { ISD::SINT_TO_FP,  MVT::f64, MVT::i1, 2 },
     { ISD::UINT_TO_FP,  MVT::f64, MVT::i1, 2 },
     { ISD::SINT_TO_FP,  MVT::f32, MVT::i8, 2 },
     { ISD::UINT_TO_FP,  MVT::f32, MVT::i8, 2 },
     { ISD::SINT_TO_FP,  MVT::f64, MVT::i8, 2 },
     { ISD::UINT_TO_FP,  MVT::f64, MVT::i8, 2 },
     { ISD::SINT_TO_FP,  MVT::f32, MVT::i16, 2 },
     { ISD::UINT_TO_FP,  MVT::f32, MVT::i16, 2 },
     { ISD::SINT_TO_FP,  MVT::f64, MVT::i16, 2 },
     { ISD::UINT_TO_FP,  MVT::f64, MVT::i16, 2 },
     { ISD::SINT_TO_FP,  MVT::f32, MVT::i32, 2 },
     { ISD::UINT_TO_FP,  MVT::f32, MVT::i32, 2 },
     { ISD::SINT_TO_FP,  MVT::f64, MVT::i32, 2 },
     { ISD::UINT_TO_FP,  MVT::f64, MVT::i32, 2 },
     { ISD::SINT_TO_FP,  MVT::f32, MVT::i64, 10 },
     { ISD::UINT_TO_FP,  MVT::f32, MVT::i64, 10 },
     { ISD::SINT_TO_FP,  MVT::f64, MVT::i64, 10 },
     { ISD::UINT_TO_FP,  MVT::f64, MVT::i64, 10 }
   };

   if (SrcTy.isInteger() && ST->hasNEON()) {
     if (const auto *Entry = ConvertCostTableLookup(NEONIntegerConversionTbl,
                                                    ISD, DstTy.getSimpleVT(),
                                                    SrcTy.getSimpleVT()))
       return Entry->Cost;
   }

   // Scalar integer conversion costs.
   static const TypeConversionCostTblEntry ARMIntegerConversionTbl[] = {
     // i16 -> i64 requires two dependent operations.
     { ISD::SIGN_EXTEND, MVT::i64, MVT::i16, 2 },

     // Truncates on i64 are assumed to be free.
     { ISD::TRUNCATE,    MVT::i32, MVT::i64, 0 },
     { ISD::TRUNCATE,    MVT::i16, MVT::i64, 0 },
     { ISD::TRUNCATE,    MVT::i8,  MVT::i64, 0 },
     { ISD::TRUNCATE,    MVT::i1,  MVT::i64, 0 }
   };

   if (SrcTy.isInteger()) {
     if (const auto *Entry = ConvertCostTableLookup(ARMIntegerConversionTbl, ISD,
                                                    DstTy.getSimpleVT(),
                                                    SrcTy.getSimpleVT()))
       return Entry->Cost;
   }

   return BaseT::getCastInstrCost(Opcode, Dst, Src);
 }

 int ARMTTIImpl::getVectorInstrCost(unsigned Opcode, Type *ValTy,
                                    unsigned Index) {
   // Penalize inserting into an D-subregister. We end up with a three times
   // lower estimated throughput on swift.
   if (ST->hasSlowLoadDSubregister() && Opcode == Instruction::InsertElement &&
       ValTy->isVectorTy() && ValTy->getScalarSizeInBits() <= 32)
     return 3;

   if ((Opcode == Instruction::InsertElement ||
        Opcode == Instruction::ExtractElement)) {
     // Cross-class copies are expensive on many microarchitectures,
     // so assume they are expensive by default.
     if (ValTy->getVectorElementType()->isIntegerTy())
       return 3;

     // Even if it's not a cross class copy, this likely leads to mixing
     // of NEON and VFP code and should be therefore penalized.
     if (ValTy->isVectorTy() &&
         ValTy->getScalarSizeInBits() <= 32)
       return std::max(BaseT::getVectorInstrCost(Opcode, ValTy, Index), 2U);
   }

   return BaseT::getVectorInstrCost(Opcode, ValTy, Index);
 }

 int ARMTTIImpl::getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy,
                                    const Instruction *I) {
   int ISD = TLI->InstructionOpcodeToISD(Opcode);
   // On NEON a vector select gets lowered to vbsl.
   if (ST->hasNEON() && ValTy->isVectorTy() && ISD == ISD::SELECT) {
     // Lowering of some vector selects is currently far from perfect.
     static const TypeConversionCostTblEntry NEONVectorSelectTbl[] = {
       { ISD::SELECT, MVT::v4i1, MVT::v4i64, 4*4 + 1*2 + 1 },
       { ISD::SELECT, MVT::v8i1, MVT::v8i64, 50 },
       { ISD::SELECT, MVT::v16i1, MVT::v16i64, 100 }
     };

     EVT SelCondTy = TLI->getValueType(DL, CondTy);
     EVT SelValTy = TLI->getValueType(DL, ValTy);
     if (SelCondTy.isSimple() && SelValTy.isSimple()) {
       if (const auto *Entry = ConvertCostTableLookup(NEONVectorSelectTbl, ISD,
                                                      SelCondTy.getSimpleVT(),
                                                      SelValTy.getSimpleVT()))
         return Entry->Cost;
     }

     std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, ValTy);
     return LT.first;
   }

   return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, I);
 }

 int ARMTTIImpl::getAddressComputationCost(Type *Ty, ScalarEvolution *SE,
                                           const SCEV *Ptr) {
   // Address computations in vectorized code with non-consecutive addresses will
   // likely result in more instructions compared to scalar code where the
   // computation can more often be merged into the index mode. The resulting
   // extra micro-ops can significantly decrease throughput.
   unsigned NumVectorInstToHideOverhead = 10;
   int MaxMergeDistance = 64;

   if (Ty->isVectorTy() && SE &&
       !BaseT::isConstantStridedAccessLessThan(SE, Ptr, MaxMergeDistance + 1))
     return NumVectorInstToHideOverhead;

   // In many cases the address computation is not merged into the instruction
   // addressing mode.
   return 1;
 }

 int ARMTTIImpl::getShuffleCost(TTI::ShuffleKind Kind, Type *Tp, int Index,
                                Type *SubTp) {
   if (Kind == TTI::SK_Broadcast) {
     static const CostTblEntry NEONDupTbl[] = {
         // VDUP handles these cases.
         {ISD::VECTOR_SHUFFLE, MVT::v2i32, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2f32, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2i64, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2f64, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v4i16, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v8i8,  1},

         {ISD::VECTOR_SHUFFLE, MVT::v4i32, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v4f32, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v8i16, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v16i8, 1}};

     std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, Tp);

     if (const auto *Entry = CostTableLookup(NEONDupTbl, ISD::VECTOR_SHUFFLE,
                                             LT.second))
       return LT.first * Entry->Cost;

     return BaseT::getShuffleCost(Kind, Tp, Index, SubTp);
   }
   if (Kind == TTI::SK_Reverse) {
     static const CostTblEntry NEONShuffleTbl[] = {
         // Reverse shuffle cost one instruction if we are shuffling within a
         // double word (vrev) or two if we shuffle a quad word (vrev, vext).
         {ISD::VECTOR_SHUFFLE, MVT::v2i32, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2f32, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2i64, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2f64, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v4i16, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v8i8,  1},

         {ISD::VECTOR_SHUFFLE, MVT::v4i32, 2},
         {ISD::VECTOR_SHUFFLE, MVT::v4f32, 2},
         {ISD::VECTOR_SHUFFLE, MVT::v8i16, 2},
         {ISD::VECTOR_SHUFFLE, MVT::v16i8, 2}};

     std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, Tp);

     if (const auto *Entry = CostTableLookup(NEONShuffleTbl, ISD::VECTOR_SHUFFLE,
                                             LT.second))
       return LT.first * Entry->Cost;

     return BaseT::getShuffleCost(Kind, Tp, Index, SubTp);
   }
   if (Kind == TTI::SK_Select) {
     static const CostTblEntry NEONSelShuffleTbl[] = {
         // Select shuffle cost table for ARM. Cost is the number of instructions
         // required to create the shuffled vector.

         {ISD::VECTOR_SHUFFLE, MVT::v2f32, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2i64, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2f64, 1},
         {ISD::VECTOR_SHUFFLE, MVT::v2i32, 1},

         {ISD::VECTOR_SHUFFLE, MVT::v4i32, 2},
         {ISD::VECTOR_SHUFFLE, MVT::v4f32, 2},
         {ISD::VECTOR_SHUFFLE, MVT::v4i16, 2},

         {ISD::VECTOR_SHUFFLE, MVT::v8i16, 16},

         {ISD::VECTOR_SHUFFLE, MVT::v16i8, 32}};

     std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, Tp);
     if (const auto *Entry = CostTableLookup(NEONSelShuffleTbl,
                                             ISD::VECTOR_SHUFFLE, LT.second))
       return LT.first * Entry->Cost;
     return BaseT::getShuffleCost(Kind, Tp, Index, SubTp);
   }
   return BaseT::getShuffleCost(Kind, Tp, Index, SubTp);
 }

 int ARMTTIImpl::getArithmeticInstrCost(
     unsigned Opcode, Type *Ty, TTI::OperandValueKind Op1Info,
     TTI::OperandValueKind Op2Info, TTI::OperandValueProperties Opd1PropInfo,
     TTI::OperandValueProperties Opd2PropInfo,
     ArrayRef<const Value *> Args) {
   int ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
   std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, Ty);

   const unsigned FunctionCallDivCost = 20;
   const unsigned ReciprocalDivCost = 10;
   static const CostTblEntry CostTbl[] = {
     // Division.
     // These costs are somewhat random. Choose a cost of 20 to indicate that
     // vectorizing devision (added function call) is going to be very expensive.
     // Double registers types.
     { ISD::SDIV, MVT::v1i64, 1 * FunctionCallDivCost},
     { ISD::UDIV, MVT::v1i64, 1 * FunctionCallDivCost},
     { ISD::SREM, MVT::v1i64, 1 * FunctionCallDivCost},
     { ISD::UREM, MVT::v1i64, 1 * FunctionCallDivCost},
     { ISD::SDIV, MVT::v2i32, 2 * FunctionCallDivCost},
     { ISD::UDIV, MVT::v2i32, 2 * FunctionCallDivCost},
     { ISD::SREM, MVT::v2i32, 2 * FunctionCallDivCost},
     { ISD::UREM, MVT::v2i32, 2 * FunctionCallDivCost},
     { ISD::SDIV, MVT::v4i16,     ReciprocalDivCost},
     { ISD::UDIV, MVT::v4i16,     ReciprocalDivCost},
     { ISD::SREM, MVT::v4i16, 4 * FunctionCallDivCost},
     { ISD::UREM, MVT::v4i16, 4 * FunctionCallDivCost},
     { ISD::SDIV, MVT::v8i8,      ReciprocalDivCost},
     { ISD::UDIV, MVT::v8i8,      ReciprocalDivCost},
     { ISD::SREM, MVT::v8i8,  8 * FunctionCallDivCost},
     { ISD::UREM, MVT::v8i8,  8 * FunctionCallDivCost},
     // Quad register types.
     { ISD::SDIV, MVT::v2i64, 2 * FunctionCallDivCost},
     { ISD::UDIV, MVT::v2i64, 2 * FunctionCallDivCost},
     { ISD::SREM, MVT::v2i64, 2 * FunctionCallDivCost},
     { ISD::UREM, MVT::v2i64, 2 * FunctionCallDivCost},
     { ISD::SDIV, MVT::v4i32, 4 * FunctionCallDivCost},
     { ISD::UDIV, MVT::v4i32, 4 * FunctionCallDivCost},
     { ISD::SREM, MVT::v4i32, 4 * FunctionCallDivCost},
     { ISD::UREM, MVT::v4i32, 4 * FunctionCallDivCost},
     { ISD::SDIV, MVT::v8i16, 8 * FunctionCallDivCost},
     { ISD::UDIV, MVT::v8i16, 8 * FunctionCallDivCost},
     { ISD::SREM, MVT::v8i16, 8 * FunctionCallDivCost},
     { ISD::UREM, MVT::v8i16, 8 * FunctionCallDivCost},
     { ISD::SDIV, MVT::v16i8, 16 * FunctionCallDivCost},
     { ISD::UDIV, MVT::v16i8, 16 * FunctionCallDivCost},
     { ISD::SREM, MVT::v16i8, 16 * FunctionCallDivCost},
     { ISD::UREM, MVT::v16i8, 16 * FunctionCallDivCost},
     // Multiplication.
   };

   if (ST->hasNEON())
     if (const auto *Entry = CostTableLookup(CostTbl, ISDOpcode, LT.second))
       return LT.first * Entry->Cost;

   int Cost = BaseT::getArithmeticInstrCost(Opcode, Ty, Op1Info, Op2Info,
                                            Opd1PropInfo, Opd2PropInfo);

   // This is somewhat of a hack. The problem that we are facing is that SROA
   // creates a sequence of shift, and, or instructions to construct values.
   // These sequences are recognized by the ISel and have zero-cost. Not so for
   // the vectorized code. Because we have support for v2i64 but not i64 those
   // sequences look particularly beneficial to vectorize.
   // To work around this we increase the cost of v2i64 operations to make them
   // seem less beneficial.
   if (LT.second == MVT::v2i64 &&
       Op2Info == TargetTransformInfo::OK_UniformConstantValue)
     Cost += 4;

   return Cost;
 }

 int ARMTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src, unsigned Alignment,
                                 unsigned AddressSpace, const Instruction *I) {
   std::pair<int, MVT> LT = TLI->getTypeLegalizationCost(DL, Src);

   if (Src->isVectorTy() && Alignment != 16 &&
       Src->getVectorElementType()->isDoubleTy()) {
     // Unaligned loads/stores are extremely inefficient.
     // We need 4 uops for vst.1/vld.1 vs 1uop for vldr/vstr.
     return LT.first * 4;
   }
   return LT.first;
 }

 int ARMTTIImpl::getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy,
                                            unsigned Factor,
                                            ArrayRef<unsigned> Indices,
                                            unsigned Alignment,
                                            unsigned AddressSpace,
                                            bool UseMaskForCond,
                                            bool UseMaskForGaps) {
   assert(Factor >= 2 && "Invalid interleave factor");
   assert(isa<VectorType>(VecTy) && "Expect a vector type");

   // vldN/vstN doesn't support vector types of i64/f64 element.
   bool EltIs64Bits = DL.getTypeSizeInBits(VecTy->getScalarType()) == 64;

   if (Factor <= TLI->getMaxSupportedInterleaveFactor() && !EltIs64Bits &&
       !UseMaskForCond && !UseMaskForGaps) {
     unsigned NumElts = VecTy->getVectorNumElements();
     auto *SubVecTy = VectorType::get(VecTy->getScalarType(), NumElts / Factor);

     // vldN/vstN only support legal vector types of size 64 or 128 in bits.
     // Accesses having vector types that are a multiple of 128 bits can be
     // matched to more than one vldN/vstN instruction.
     if (NumElts % Factor == 0 &&
         TLI->isLegalInterleavedAccessType(SubVecTy, DL))
       return Factor * TLI->getNumInterleavedAccesses(SubVecTy, DL);
   }

   return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
                                            Alignment, AddressSpace,
                                            UseMaskForCond, UseMaskForGaps);
 }

 void ARMTTIImpl::getUnrollingPreferences(Loop *L, ScalarEvolution &SE,
                                          TTI::UnrollingPreferences &UP) {
   // Only currently enable these preferences for M-Class cores.
   if (!ST->isMClass())
     return BasicTTIImplBase::getUnrollingPreferences(L, SE, UP);

   // Disable loop unrolling for Oz and Os.
   UP.OptSizeThreshold = 0;
   UP.PartialOptSizeThreshold = 0;
   if (L->getHeader()->getParent()->optForSize())
     return;

   // Only enable on Thumb-2 targets.
   if (!ST->isThumb2())
     return;

   SmallVector<BasicBlock*, 4> ExitingBlocks;
   L->getExitingBlocks(ExitingBlocks);
   LLVM_DEBUG(dbgs() << "Loop has:\n"
                     << "Blocks: " << L->getNumBlocks() << "\n"
                     << "Exit blocks: " << ExitingBlocks.size() << "\n");

   // Only allow another exit other than the latch. This acts as an early exit
   // as it mirrors the profitability calculation of the runtime unroller.
   if (ExitingBlocks.size() > 2)
     return;

   // Limit the CFG of the loop body for targets with a branch predictor.
   // Allowing 4 blocks permits if-then-else diamonds in the body.
   if (ST->hasBranchPredictor() && L->getNumBlocks() > 4)
     return;

   // Scan the loop: don't unroll loops with calls as this could prevent
   // inlining.
   unsigned Cost = 0;
   for (auto *BB : L->getBlocks()) {
     for (auto &I : *BB) {
       if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
         ImmutableCallSite CS(&I);
         if (const Function *F = CS.getCalledFunction()) {
           if (!isLoweredToCall(F))
             continue;
         }
         return;
       }
       SmallVector<const Value*, 4> Operands(I.value_op_begin(),
                                             I.value_op_end());
       Cost += getUserCost(&I, Operands);
     }
   }

   LLVM_DEBUG(dbgs() << "Cost of loop: " << Cost << "\n");

   UP.Partial = true;
   UP.Runtime = true;
   UP.UnrollRemainder = true;
   UP.DefaultUnrollRuntimeCount = 4;
   UP.UnrollAndJam = true;
   UP.UnrollAndJamInnerLoopThreshold = 60;

   // Force unrolling small loops can be very useful because of the branch
   // taken cost of the backedge.
   if (Cost < 12)
     UP.Force = true;
 }
llvm::Type::getVectorElementType
Type * getVectorElementType() const
Definition: Type.h:371

llvm::ISD::FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding &#39;Y&#39; from a larger floating point type down to the precision of the ...
Definition: ISDOpcodes.h:538

MachineValueType.h

Instruction.h

llvm::TargetTransformInfo::UnrollingPreferences::Partial
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
Definition: TargetTransformInfo.h:406

llvm::ARMTTIImpl::getCmpSelInstrCost
int getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, const Instruction *I=nullptr)
Definition: ARMTargetTransformInfo.cpp:355

llvm::BasicTTIImplBase< ARMTTIImpl >::getArithmeticInstrCost
unsigned getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::OperandValueKind Opd1Info=TTI::OK_AnyValue, TTI::OperandValueKind Opd2Info=TTI::OK_AnyValue, TTI::OperandValueProperties Opd1PropInfo=TTI::OP_None, TTI::OperandValueProperties Opd2PropInfo=TTI::OP_None, ArrayRef< const Value * > Args=ArrayRef< const Value * >())
Definition: BasicTTIImpl.h:568

llvm::MVT::v4f32
Definition: MachineValueType.h:154

llvm::TargetTransformInfoImplBase::isConstantStridedAccessLessThan
bool isConstantStridedAccessLessThan(ScalarEvolution *SE, const SCEV *Ptr, int64_t MergeDistance)
Definition: TargetTransformInfoImpl.h:656

llvm::APInt::getZExtValue
uint64_t getZExtValue() const
Get zero extended value.
Definition: APInt.h:1563

llvm::max
GCNRegPressure max(const GCNRegPressure &P1, const GCNRegPressure &P2)
Definition: GCNRegPressure.h:89

Instructions.h

llvm::ARMSubtarget::isThumb
bool isThumb() const
Definition: ARMSubtarget.h:712

llvm
This class represents lattice values for constants.
Definition: AllocatorList.h:24

CostTable.h
Cost tables and simple lookup functions.

llvm::ISD::VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition: ISDOpcodes.h:367

DerivedTypes.h

Type.h

llvm::MVT::v4i64
Definition: MachineValueType.h:100

llvm::ARMTargetLowering::getNumInterleavedAccesses
unsigned getNumInterleavedAccesses(VectorType *VecTy, const DataLayout &DL) const
Returns the number of interleaved accesses that will be generated when lowering accesses of the given...
Definition: ARMISelLowering.cpp:14724

llvm::ARMTTIImpl::getCastInstrCost
int getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, const Instruction *I=nullptr)
Definition: ARMTargetTransformInfo.cpp:136

llvm::ARMSubtarget::hasBranchPredictor
bool hasBranchPredictor() const
Definition: ARMSubtarget.h:626

DataLayout.h

llvm::ARMTTIImpl::getArithmeticInstrCost
int getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::OperandValueKind Op1Info=TTI::OK_AnyValue, TTI::OperandValueKind Op2Info=TTI::OK_AnyValue, TTI::OperandValueProperties Opd1PropInfo=TTI::OP_None, TTI::OperandValueProperties Opd2PropInfo=TTI::OP_None, ArrayRef< const Value *> Args=ArrayRef< const Value *>())
Definition: ARMTargetTransformInfo.cpp:477

llvm::ScalarEvolution
The main scalar evolution driver.
Definition: ScalarEvolution.h:454

llvm::TargetTransformInfo::UnrollingPreferences::PartialOptSizeThreshold
unsigned PartialOptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size, like OptSizeThreshold, but used for partial/runtime unrolling (set to UINT_MAX to disable).
Definition: TargetTransformInfo.h:377

llvm::MVT::v8i16
Definition: MachineValueType.h:84

llvm::EVT::getSimpleVT
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition: ValueTypes.h:253

TargetMachine.h

llvm::ISD::UINT_TO_FP
Definition: ISDOpcodes.h:479

BasicBlock.h

ISDOpcodes.h

llvm::TargetTransformInfo::UnrollingPreferences::Force
bool Force
Apply loop unroll on any kind of loop (mainly to loops that fail runtime unrolling).
Definition: TargetTransformInfo.h:418

llvm::MVT::v16i16
Definition: MachineValueType.h:85

llvm::EVT::isInteger
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition: ValueTypes.h:141

F
F(f)

llvm::TypeConversionCostTblEntry
Type Conversion Cost Table.
Definition: CostTable.h:45

llvm::Function
Definition: Function.h:60

llvm::MVT::v16i1
Definition: MachineValueType.h:64

llvm::Type::isVectorTy
bool isVectorTy() const
True if this is an instance of VectorType.
Definition: Type.h:230

llvm::TargetTransformInfo::UnrollingPreferences::UnrollAndJam
bool UnrollAndJam
Allow unroll and jam. Used to enable unroll and jam for the target.
Definition: TargetTransformInfo.h:426

llvm::ARMTTIImpl::getInterleavedMemoryOpCost
int getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, unsigned Alignment, unsigned AddressSpace, bool UseMaskForCond=false, bool UseMaskForGaps=false)
Definition: ARMTargetTransformInfo.cpp:562

llvm::ISD::FP_TO_UINT
Definition: ISDOpcodes.h:525

llvm::CostTblEntry
Cost Table Entry.
Definition: CostTable.h:25

CallSite.h

llvm::EVT::isFloatingPoint
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition: ValueTypes.h:136

llvm::MVT::v1i64
Definition: MachineValueType.h:98

llvm::MVT::v8f32
Definition: MachineValueType.h:155

llvm::ARMTargetLowering::isLegalInterleavedAccessType
bool isLegalInterleavedAccessType(VectorType *VecTy, const DataLayout &DL) const
Returns true if VecTy is a legal interleaved access type.
Definition: ARMISelLowering.cpp:14729

llvm::MVT::v4i1
Definition: MachineValueType.h:62

llvm::MCSubtargetInfo::getFeatureBits
const FeatureBitset & getFeatureBits() const
Definition: MCSubtargetInfo.h:71

llvm::APInt::isNonNegative
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition: APInt.h:369

LoopInfo.h

llvm::ARMTTIImpl::getMemoryOpCost
int getMemoryOpCost(unsigned Opcode, Type *Src, unsigned Alignment, unsigned AddressSpace, const Instruction *I=nullptr)
Definition: ARMTargetTransformInfo.cpp:549

llvm::Type::isIntegerTy
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition: Type.h:197

llvm::ConvertCostTableLookup
const TypeConversionCostTblEntry * ConvertCostTableLookup(ArrayRef< TypeConversionCostTblEntry > Tbl, int ISD, MVT Dst, MVT Src)
Find in type conversion cost table, TypeTy must be comparable to CompareTy by ==. ...
Definition: CostTable.h:55

llvm::Instruction
Definition: Instruction.h:44

APInt.h
This file implements a class to represent arbitrary precision integral constant values and operations...

llvm::LoopBase::getHeader
BlockT * getHeader() const
Definition: LoopInfo.h:100

llvm::APInt::getActiveBits
unsigned getActiveBits() const
Compute the number of active bits in the value.
Definition: APInt.h:1533

llvm::BasicTTIImplBase< ARMTTIImpl >::getCmpSelInstrCost
unsigned getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, const Instruction *I)
Definition: BasicTTIImpl.h:772

ARMTargetTransformInfo.h
This file a TargetTransformInfo::Concept conforming object specific to the ARM target machine...

llvm::ARMTTIImpl::getShuffleCost
int getShuffleCost(TTI::ShuffleKind Kind, Type *Tp, int Index, Type *SubTp)
Definition: ARMTargetTransformInfo.cpp:401

llvm::APInt::getSExtValue
int64_t getSExtValue() const
Get sign extended value.
Definition: APInt.h:1575

llvm::BasicTTIImplBase< ARMTTIImpl >::getCastInstrCost
unsigned getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, const Instruction *I=nullptr)
Definition: BasicTTIImpl.h:634

llvm::ISD::SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition: ISDOpcodes.h:478

llvm::TargetTransformInfoImplBase::isLoweredToCall
bool isLoweredToCall(const Function *F)
Definition: TargetTransformInfoImpl.h:195

llvm::ARMSubtarget::hasV6T2Ops
bool hasV6T2Ops() const
Definition: ARMSubtarget.h:539

llvm::MVT::v2i32
Definition: MachineValueType.h:91

llvm::ArrayRef
ArrayRef - Represent a constant reference to an array (0 or more elements consecutively in memory)...
Definition: APInt.h:33

llvm::TargetTransformInfo::SK_Select
Selects elements from the corresponding lane of either source operand.
Definition: TargetTransformInfo.h:660

llvm::ISD::UDIV
Definition: ISDOpcodes.h:201

llvm::TargetTransformInfoImplBase::DL
const DataLayout & DL
Definition: TargetTransformInfoImpl.h:36

llvm::MVT::v2f64
Definition: MachineValueType.h:158

llvm::SystemZISD::TM
Definition: SystemZISelLowering.h:68

llvm::TargetTransformInfo::SK_Reverse
Reverse the order of the vector.
Definition: TargetTransformInfo.h:659

llvm::dwarf::Index
Index
Definition: Dwarf.h:337

llvm::ISD::FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition: ISDOpcodes.h:524

llvm::Type::getScalarType
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return &#39;this&#39;.
Definition: Type.h:304

llvm::APInt::isNegative
bool isNegative() const
Determine sign of this APInt.
Definition: APInt.h:364

llvm::BasicTTIImplBase::getUnrollingPreferences
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP)
Definition: BasicTTIImpl.h:424

llvm::APInt::isAllOnesValue
bool isAllOnesValue() const
Determine if all bits are set.
Definition: APInt.h:396

llvm::FeatureBitset
Container class for subtarget features.
Definition: SubtargetFeature.h:37

llvm::Type
The instances of the Type class are immutable: once they are created, they are never changed...
Definition: Type.h:46

llvm::BasicTTIImplBase< ARMTTIImpl >::getInterleavedMemoryOpCost
unsigned getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, unsigned Alignment, unsigned AddressSpace, bool UseMaskForCond=false, bool UseMaskForGaps=false)
Definition: BasicTTIImpl.h:850

llvm::ARMSubtarget::isMClass
bool isMClass() const
Definition: ARMSubtarget.h:716

llvm::ARM_AM::isThumbImmShiftedVal
bool isThumbImmShiftedVal(unsigned V)
isThumbImmShiftedVal - Return true if the specified value can be obtained by left shifting a 8-bit im...
Definition: ARMAddressingModes.h:220

llvm::TargetTransformInfo::UnrollingPreferences::UnrollAndJamInnerLoopThreshold
unsigned UnrollAndJamInnerLoopThreshold
Threshold for unroll and jam, for inner loop size.
Definition: TargetTransformInfo.h:431

llvm::TargetTransformInfoImplCRTPBase< ARMTTIImpl >::getUserCost
unsigned getUserCost(const User *U, ArrayRef< const Value * > Operands)
Definition: TargetTransformInfoImpl.h:790

llvm::MVT::v2i8
Definition: MachineValueType.h:72

llvm::MVT::v2f32
Definition: MachineValueType.h:153

llvm::Function::optForSize
bool optForSize() const
Optimize this function for size (-Os) or minimum size (-Oz).
Definition: Function.h:598

llvm::ARMSubtarget::hasSlowLoadDSubregister
bool hasSlowLoadDSubregister() const
Definition: ARMSubtarget.h:613

llvm::ARM_AM::getT2SOImmVal
int getT2SOImmVal(unsigned Arg)
getT2SOImmVal - Given a 32-bit immediate, if it is something that can fit into a Thumb-2 shifter_oper...
Definition: ARMAddressingModes.h:305

llvm::MVT::v2i64
Definition: MachineValueType.h:99

llvm::AArch64CC::LT
Definition: AArch64BaseInfo.h:205

ValueTypes.h

llvm::ISD::FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition: ISDOpcodes.h:556

llvm::LoopBase::getExitingBlocks
void getExitingBlocks(SmallVectorImpl< BlockT *> &ExitingBlocks) const
Return all blocks inside the loop that have successors outside of the loop.
Definition: LoopInfoImpl.h:35

llvm::tgtok::Bits
Definition: TGLexer.h:49

llvm::EVT
Extended Value Type.
Definition: ValueTypes.h:34

llvm::SmallVectorBase::size
size_t size() const
Definition: SmallVector.h:53

llvm::TargetLoweringBase::getTargetMachine
const TargetMachine & getTargetMachine() const
Definition: TargetLowering.h:231

llvm::TargetLoweringBase::getValueType
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
Definition: TargetLowering.h:1140

llvm::MVT::f64
Definition: MachineValueType.h:52

llvm::ISD::UREM
Definition: ISDOpcodes.h:201

llvm::TargetTransformInfo::OperandValueProperties
OperandValueProperties
Additional properties of an operand&#39;s values.
Definition: TargetTransformInfo.h:681

llvm::BasicTTIImplBase< ARMTTIImpl >::getShuffleCost
unsigned getShuffleCost(TTI::ShuffleKind Kind, Type *Tp, int Index, Type *SubTp)
Definition: BasicTTIImpl.h:615

llvm::MVT::v8i32
Definition: MachineValueType.h:93

ARMSubtarget.h

llvm::ARM_AM::getSOImmVal
int getSOImmVal(unsigned Arg)
getSOImmVal - Given a 32-bit immediate, if it is something that can fit into an shifter_operand immed...
Definition: ARMAddressingModes.h:162

llvm::Type::getScalarSizeInBits
unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type...
Definition: Type.cpp:130

llvm::SmallVector
This is a &#39;vector&#39; (really, a variable-sized array), optimized for the case when the array is small...
Definition: SmallVector.h:847

llvm::MVT::v16i32
Definition: MachineValueType.h:94

llvm::MVT::f32
Definition: MachineValueType.h:51

llvm::AddressSpace
AddressSpace
Definition: NVPTXBaseInfo.h:22

llvm::ARMTTIImpl::getAddressComputationCost
int getAddressComputationCost(Type *Val, ScalarEvolution *SE, const SCEV *Ptr)
Definition: ARMTargetTransformInfo.cpp:383

llvm::ARMSubtarget::hasNEON
bool hasNEON() const
Definition: ARMSubtarget.h:571

llvm::TargetTransformInfo::UnrollingPreferences::DefaultUnrollRuntimeCount
unsigned DefaultUnrollRuntimeCount
Default unroll count for loops with run-time trip count.
Definition: TargetTransformInfo.h:389

llvm::MVT::i8
Definition: MachineValueType.h:41

llvm::TargetTransformInfo::UnrollingPreferences::Runtime
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
Definition: TargetTransformInfo.h:410

llvm::TargetMachine::getSubtargetImpl
virtual const TargetSubtargetInfo * getSubtargetImpl(const Function &) const
Virtual method implemented by subclasses that returns a reference to that target&#39;s TargetSubtargetInf...
Definition: TargetMachine.h:111

llvm::dbgs
raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition: Debug.cpp:133

llvm::Type::getVectorNumElements
unsigned getVectorNumElements() const
Definition: DerivedTypes.h:462

llvm::APInt
Class for arbitrary precision integers.
Definition: APInt.h:70

llvm::ISD::SREM
Definition: ISDOpcodes.h:201

llvm::ISD::SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition: ISDOpcodes.h:420

llvm::TargetTransformInfo::UnrollingPreferences::UnrollRemainder
bool UnrollRemainder
Allow unrolling of all the iterations of the runtime loop remainder.
Definition: TargetTransformInfo.h:424

llvm::ISD::ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition: ISDOpcodes.h:468

llvm::MVT::v4i32
Definition: MachineValueType.h:92

llvm::TargetLoweringBase::InstructionOpcodeToISD
int InstructionOpcodeToISD(unsigned Opcode) const
Get the ISD node that corresponds to the Instruction class opcode.
Definition: TargetLoweringBase.cpp:1438

llvm::ISD::SDIV
Definition: ISDOpcodes.h:201

llvm::MVT::i16
Definition: MachineValueType.h:42

llvm::DataLayout::getTypeSizeInBits
uint64_t getTypeSizeInBits(Type *Ty) const
Size examples:
Definition: DataLayout.h:568

llvm::CostTableLookup
const CostTblEntry * CostTableLookup(ArrayRef< CostTblEntry > Tbl, int ISD, MVT Ty)
Find in cost table, TypeTy must be comparable to CompareTy by ==.
Definition: CostTable.h:32

llvm::ARMSubtarget::isThumb2
bool isThumb2() const
Definition: ARMSubtarget.h:714

llvm::LoopBase::getNumBlocks
unsigned getNumBlocks() const
Get the number of blocks in this loop in constant time.
Definition: LoopInfo.h:163

llvm::ARMTTIImpl::areInlineCompatible
bool areInlineCompatible(const Function *Caller, const Function *Callee) const
Definition: ARMTargetTransformInfo.cpp:39

llvm::MVT::v8i64
Definition: MachineValueType.h:101

llvm::EVT::isVector
bool isVector() const
Return true if this is a vector value type.
Definition: ValueTypes.h:151

llvm::MVT::v16f32
Definition: MachineValueType.h:156

llvm::SCEV
This class represents an analyzed expression in the program.
Definition: ScalarEvolution.h:77

llvm::Type::getIntegerBitWidth
unsigned getIntegerBitWidth() const
Definition: DerivedTypes.h:97

llvm::Loop
Represents a single loop in the control flow graph.
Definition: LoopInfo.h:465

llvm::APInt::getLimitedValue
uint64_t getLimitedValue(uint64_t Limit=UINT64_MAX) const
If this value is smaller than the specified limit, return it, otherwise return the limit value...
Definition: APInt.h:482

llvm::LoopBase::getBlocks
ArrayRef< BlockT * > getBlocks() const
Get a list of the basic blocks which make up this loop.
Definition: LoopInfo.h:149

llvm::TargetTransformInfo::UnrollingPreferences
Parameters that control the generic loop unrolling transformation.
Definition: TargetTransformInfo.h:348

llvm::TargetTransformInfo::UnrollingPreferences::OptSizeThreshold
unsigned OptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size (set to UINT_MAX to disable)...
Definition: TargetTransformInfo.h:370

llvm::ImmutableCallSite
Establish a view to a call site for examination.
Definition: CallSite.h:711

llvm::BasicBlock::getParent
const Function * getParent() const
Return the enclosing method, or null if none.
Definition: BasicBlock.h:107

llvm::ARMTTIImpl::getIntImmCost
int getIntImmCost(const APInt &Imm, Type *Ty)
Definition: ARMTargetTransformInfo.cpp:57

I
#define I(x, y, z)
Definition: MD5.cpp:58

llvm::BasicTTIImplBase< ARMTTIImpl >::getVectorInstrCost
unsigned getVectorInstrCost(unsigned Opcode, Type *Val, unsigned Index)
Definition: BasicTTIImpl.h:812

llvm::MVT::v4i16
Definition: MachineValueType.h:83

llvm::TargetTransformInfo::OK_UniformConstantValue
Definition: TargetTransformInfo.h:676

llvm::MVT::v4i8
Definition: MachineValueType.h:73

llvm::ARMTTIImpl::getUnrollingPreferences
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP)
Definition: ARMTargetTransformInfo.cpp:593

llvm::MVT::i32
Definition: MachineValueType.h:43

llvm::MVT::i64
Definition: MachineValueType.h:44

llvm::CallSiteBase::getCalledFunction
FunTy * getCalledFunction() const
Return the function being called if this is a direct call, otherwise return null (if it&#39;s an indirect...
Definition: CallSite.h:107

Kind
const unsigned Kind
Definition: ARMAsmParser.cpp:10586

llvm::MCID::Add
Definition: MCInstrDesc.h:153

assert
assert(ImpDefSCC.getReg()==AMDGPU::SCC &&ImpDefSCC.isDef())

llvm::MVT::i1
Definition: MachineValueType.h:40

llvm::Type::getPrimitiveSizeInBits
unsigned getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition: Type.cpp:115

llvm::MVT::v16i64
Definition: MachineValueType.h:102

llvm::MVT::v16i8
Definition: MachineValueType.h:75

llvm::VectorType::get
static VectorType * get(Type *ElementType, unsigned NumElements)
This static method is the primary way to construct an VectorType.
Definition: Type.cpp:606

SmallVector.h

llvm::ARMTTIImpl::getIntImmCodeSizeCost
int getIntImmCodeSizeCost(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty)
Definition: ARMTargetTransformInfo.cpp:91

llvm::TargetTransformInfo::SK_Broadcast
Broadcast element 0 to all other elements.
Definition: TargetTransformInfo.h:658

llvm::MVT::v2i16
Definition: MachineValueType.h:82

llvm::ARMTTIImpl::getVectorInstrCost
int getVectorInstrCost(unsigned Opcode, Type *Val, unsigned Index)
Definition: ARMTargetTransformInfo.cpp:330

Casting.h

llvm::TargetMachine
Primary interface to the complete machine description for the target machine.
Definition: TargetMachine.h:59

llvm::TargetTransformInfo::OperandValueKind
OperandValueKind
Additional information about an operand&#39;s possible values.
Definition: TargetTransformInfo.h:673

llvm::MVT::v8i8
Definition: MachineValueType.h:74

llvm::ISD::SIGN_EXTEND
Conversion operators.
Definition: ISDOpcodes.h:465

llvm::ISD::TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition: ISDOpcodes.h:474

llvm::EVT::isSimple
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition: ValueTypes.h:126

SubtargetFeature.h

LLVM_DEBUG
#define LLVM_DEBUG(X)
Definition: Debug.h:123

llvm::Type::isDoubleTy
bool isDoubleTy() const
Return true if this is &#39;double&#39;, a 64-bit IEEE fp type.
Definition: Type.h:150

ARMAddressingModes.h

llvm::AMDGPU::HSAMD::Kernel::Key::Args
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
Definition: AMDGPUMetadata.h:374

llvm::TargetLoweringBase::getTypeLegalizationCost
std::pair< int, MVT > getTypeLegalizationCost(const DataLayout &DL, Type *Ty) const
Estimate the cost of type-legalization and the legalized type.
Definition: TargetLoweringBase.cpp:1516

llvm::MVT::v8i1
Definition: MachineValueType.h:63

llvm::TargetTransformInfo::ShuffleKind
ShuffleKind
The various kinds of shuffle patterns for vector queries.
Definition: TargetTransformInfo.h:657