diff options
Diffstat (limited to 'llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp')
| -rw-r--r-- | llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp | 164 |
1 files changed, 98 insertions, 66 deletions
diff --git a/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp b/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp index 2b9483293941..f32c9bd2bdea 100644 --- a/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp +++ b/llvm/lib/Target/SystemZ/SystemZTargetTransformInfo.cpp @@ -18,6 +18,7 @@ #include "llvm/CodeGen/BasicTTIImpl.h" #include "llvm/CodeGen/TargetLowering.h" #include "llvm/IR/DerivedTypes.h" +#include "llvm/IR/InstIterator.h" #include "llvm/IR/IntrinsicInst.h" #include "llvm/IR/Intrinsics.h" #include "llvm/Support/Debug.h" @@ -80,7 +81,6 @@ unsigned SystemZTTIImpl::adjustInliningThreshold(const CallBase *CB) const { const Function *Callee = CB->getCalledFunction(); if (!Callee) return 0; - const Module *M = Caller->getParent(); // Increase the threshold if an incoming argument is used only as a memcpy // source. @@ -92,25 +92,37 @@ unsigned SystemZTTIImpl::adjustInliningThreshold(const CallBase *CB) const { } } - // Give bonus for globals used much in both caller and callee. - std::set<const GlobalVariable *> CalleeGlobals; - std::set<const GlobalVariable *> CallerGlobals; - for (const GlobalVariable &Global : M->globals()) - for (const User *U : Global.users()) - if (const Instruction *User = dyn_cast<Instruction>(U)) { - if (User->getParent()->getParent() == Callee) - CalleeGlobals.insert(&Global); - if (User->getParent()->getParent() == Caller) - CallerGlobals.insert(&Global); + // Give bonus for globals used much in both caller and a relatively small + // callee. + unsigned InstrCount = 0; + SmallDenseMap<const Value *, unsigned> Ptr2NumUses; + for (auto &I : instructions(Callee)) { + if (++InstrCount == 200) { + Ptr2NumUses.clear(); + break; + } + if (const auto *SI = dyn_cast<StoreInst>(&I)) { + if (!SI->isVolatile()) + if (auto *GV = dyn_cast<GlobalVariable>(SI->getPointerOperand())) + Ptr2NumUses[GV]++; + } else if (const auto *LI = dyn_cast<LoadInst>(&I)) { + if (!LI->isVolatile()) + if (auto *GV = dyn_cast<GlobalVariable>(LI->getPointerOperand())) + Ptr2NumUses[GV]++; + } else if (const auto *GEP = dyn_cast<GetElementPtrInst>(&I)) { + if (auto *GV = dyn_cast<GlobalVariable>(GEP->getPointerOperand())) { + unsigned NumStores = 0, NumLoads = 0; + countNumMemAccesses(GEP, NumStores, NumLoads, Callee); + Ptr2NumUses[GV] += NumLoads + NumStores; } - for (auto *GV : CalleeGlobals) - if (CallerGlobals.count(GV)) { - unsigned CalleeStores = 0, CalleeLoads = 0; + } + } + + for (auto [Ptr, NumCalleeUses] : Ptr2NumUses) + if (NumCalleeUses > 10) { unsigned CallerStores = 0, CallerLoads = 0; - countNumMemAccesses(GV, CalleeStores, CalleeLoads, Callee); - countNumMemAccesses(GV, CallerStores, CallerLoads, Caller); - if ((CalleeStores + CalleeLoads) > 10 && - (CallerStores + CallerLoads) > 10) { + countNumMemAccesses(Ptr, CallerStores, CallerLoads, Caller); + if (CallerStores + CallerLoads > 10) { Bonus = 1000; break; } @@ -136,8 +148,9 @@ unsigned SystemZTTIImpl::adjustInliningThreshold(const CallBase *CB) const { return Bonus; } -InstructionCost SystemZTTIImpl::getIntImmCost(const APInt &Imm, Type *Ty, - TTI::TargetCostKind CostKind) { +InstructionCost +SystemZTTIImpl::getIntImmCost(const APInt &Imm, Type *Ty, + TTI::TargetCostKind CostKind) const { assert(Ty->isIntegerTy()); unsigned BitSize = Ty->getPrimitiveSizeInBits(); @@ -173,7 +186,7 @@ InstructionCost SystemZTTIImpl::getIntImmCost(const APInt &Imm, Type *Ty, InstructionCost SystemZTTIImpl::getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, - Instruction *Inst) { + Instruction *Inst) const { assert(Ty->isIntegerTy()); unsigned BitSize = Ty->getPrimitiveSizeInBits(); @@ -293,7 +306,7 @@ InstructionCost SystemZTTIImpl::getIntImmCostInst(unsigned Opcode, unsigned Idx, InstructionCost SystemZTTIImpl::getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, - TTI::TargetCostKind CostKind) { + TTI::TargetCostKind CostKind) const { assert(Ty->isIntegerTy()); unsigned BitSize = Ty->getPrimitiveSizeInBits(); @@ -342,16 +355,16 @@ SystemZTTIImpl::getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, } TargetTransformInfo::PopcntSupportKind -SystemZTTIImpl::getPopcntSupport(unsigned TyWidth) { +SystemZTTIImpl::getPopcntSupport(unsigned TyWidth) const { assert(isPowerOf2_32(TyWidth) && "Type width must be power of 2"); if (ST->hasPopulationCount() && TyWidth <= 64) return TTI::PSK_FastHardware; return TTI::PSK_Software; } -void SystemZTTIImpl::getUnrollingPreferences(Loop *L, ScalarEvolution &SE, - TTI::UnrollingPreferences &UP, - OptimizationRemarkEmitter *ORE) { +void SystemZTTIImpl::getUnrollingPreferences( + Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, + OptimizationRemarkEmitter *ORE) const { // Find out if L contains a call, what the machine instruction count // estimate is, and how many stores there are. bool HasCall = false; @@ -371,15 +384,15 @@ void SystemZTTIImpl::getUnrollingPreferences(Loop *L, ScalarEvolution &SE, } if (isa<StoreInst>(&I)) { Type *MemAccessTy = I.getOperand(0)->getType(); - NumStores += getMemoryOpCost(Instruction::Store, MemAccessTy, - std::nullopt, 0, TTI::TCK_RecipThroughput); + NumStores += getMemoryOpCost(Instruction::Store, MemAccessTy, Align(), + 0, TTI::TCK_RecipThroughput); } } // The z13 processor will run out of store tags if too many stores // are fed into it too quickly. Therefore make sure there are not // too many stores in the resulting unrolled loop. - unsigned const NumStoresVal = *NumStores.getValue(); + unsigned const NumStoresVal = NumStores.getValue(); unsigned const Max = (NumStoresVal ? (12 / NumStoresVal) : UINT_MAX); if (HasCall) { @@ -406,12 +419,13 @@ void SystemZTTIImpl::getUnrollingPreferences(Loop *L, ScalarEvolution &SE, } void SystemZTTIImpl::getPeelingPreferences(Loop *L, ScalarEvolution &SE, - TTI::PeelingPreferences &PP) { + TTI::PeelingPreferences &PP) const { BaseT::getPeelingPreferences(L, SE, PP); } -bool SystemZTTIImpl::isLSRCostLess(const TargetTransformInfo::LSRCost &C1, - const TargetTransformInfo::LSRCost &C2) { +bool SystemZTTIImpl::isLSRCostLess( + const TargetTransformInfo::LSRCost &C1, + const TargetTransformInfo::LSRCost &C2) const { // SystemZ specific: check instruction count (first), and don't care about // ImmCost, since offsets are checked explicitly. return std::tie(C1.Insns, C1.NumRegs, C1.AddRecCost, @@ -422,6 +436,20 @@ bool SystemZTTIImpl::isLSRCostLess(const TargetTransformInfo::LSRCost &C1, C2.ScaleCost, C2.SetupCost); } +bool SystemZTTIImpl::areInlineCompatible(const Function *Caller, + const Function *Callee) const { + const TargetMachine &TM = getTLI()->getTargetMachine(); + + const FeatureBitset &CallerBits = + TM.getSubtargetImpl(*Caller)->getFeatureBits(); + const FeatureBitset &CalleeBits = + TM.getSubtargetImpl(*Callee)->getFeatureBits(); + + // Support only equal feature bitsets. Restriction should be relaxed in the + // future to allow inlining when callee's bits are subset of the caller's. + return CallerBits == CalleeBits; +} + unsigned SystemZTTIImpl::getNumberOfRegisters(unsigned ClassID) const { bool Vector = (ClassID == 1); if (!Vector) @@ -464,12 +492,12 @@ unsigned SystemZTTIImpl::getMinPrefetchStride(unsigned NumMemAccesses, return ST->hasMiscellaneousExtensions3() ? 8192 : 2048; } -bool SystemZTTIImpl::hasDivRemOp(Type *DataType, bool IsSigned) { +bool SystemZTTIImpl::hasDivRemOp(Type *DataType, bool IsSigned) const { EVT VT = TLI->getValueType(DL, DataType); return (VT.isScalarInteger() && TLI->isTypeLegal(VT)); } -static bool isFreeEltLoad(Value *Op) { +static bool isFreeEltLoad(const Value *Op) { if (isa<LoadInst>(Op) && Op->hasOneUse()) { const Instruction *UserI = cast<Instruction>(*Op->user_begin()); return !isa<StoreInst>(UserI); // Prefer MVC @@ -479,7 +507,8 @@ static bool isFreeEltLoad(Value *Op) { InstructionCost SystemZTTIImpl::getScalarizationOverhead( VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, - TTI::TargetCostKind CostKind, ArrayRef<Value *> VL) { + TTI::TargetCostKind CostKind, bool ForPoisonSrc, + ArrayRef<Value *> VL) const { unsigned NumElts = cast<FixedVectorType>(Ty)->getNumElements(); InstructionCost Cost = 0; @@ -501,7 +530,7 @@ InstructionCost SystemZTTIImpl::getScalarizationOverhead( } Cost += BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract, - CostKind, VL); + CostKind, ForPoisonSrc, VL); return Cost; } @@ -527,8 +556,7 @@ static unsigned getNumVectorRegs(Type *Ty) { InstructionCost SystemZTTIImpl::getArithmeticInstrCost( unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, - ArrayRef<const Value *> Args, - const Instruction *CxtI) { + ArrayRef<const Value *> Args, const Instruction *CxtI) const { // TODO: Handle more cost kinds. if (CostKind != TTI::TCK_RecipThroughput) @@ -710,20 +738,22 @@ InstructionCost SystemZTTIImpl::getArithmeticInstrCost( Args, CxtI); } -InstructionCost SystemZTTIImpl::getShuffleCost( - TTI::ShuffleKind Kind, VectorType *Tp, ArrayRef<int> Mask, - TTI::TargetCostKind CostKind, int Index, VectorType *SubTp, - ArrayRef<const Value *> Args, const Instruction *CxtI) { - Kind = improveShuffleKindFromMask(Kind, Mask, Tp, Index, SubTp); +InstructionCost +SystemZTTIImpl::getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, + VectorType *SrcTy, ArrayRef<int> Mask, + TTI::TargetCostKind CostKind, int Index, + VectorType *SubTp, ArrayRef<const Value *> Args, + const Instruction *CxtI) const { + Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp); if (ST->hasVector()) { - unsigned NumVectors = getNumVectorRegs(Tp); + unsigned NumVectors = getNumVectorRegs(SrcTy); // TODO: Since fp32 is expanded, the shuffle cost should always be 0. // FP128 values are always in scalar registers, so there is no work // involved with a shuffle, except for broadcast. In that case register // moves are done with a single instruction per element. - if (Tp->getScalarType()->isFP128Ty()) + if (SrcTy->getScalarType()->isFP128Ty()) return (Kind == TargetTransformInfo::SK_Broadcast ? NumVectors - 1 : 0); switch (Kind) { @@ -747,7 +777,8 @@ InstructionCost SystemZTTIImpl::getShuffleCost( } } - return BaseT::getShuffleCost(Kind, Tp, Mask, CostKind, Index, SubTp); + return BaseT::getShuffleCost(Kind, DstTy, SrcTy, Mask, CostKind, Index, + SubTp); } // Return the log2 difference of the element sizes of the two vector types. @@ -762,8 +793,7 @@ static unsigned getElSizeLog2Diff(Type *Ty0, Type *Ty1) { } // Return the number of instructions needed to truncate SrcTy to DstTy. -unsigned SystemZTTIImpl:: -getVectorTruncCost(Type *SrcTy, Type *DstTy) { +unsigned SystemZTTIImpl::getVectorTruncCost(Type *SrcTy, Type *DstTy) const { assert (SrcTy->isVectorTy() && DstTy->isVectorTy()); assert(SrcTy->getPrimitiveSizeInBits().getFixedValue() > DstTy->getPrimitiveSizeInBits().getFixedValue() && @@ -804,8 +834,8 @@ getVectorTruncCost(Type *SrcTy, Type *DstTy) { // Return the cost of converting a vector bitmask produced by a compare // (SrcTy), to the type of the select or extend instruction (DstTy). -unsigned SystemZTTIImpl:: -getVectorBitmaskConversionCost(Type *SrcTy, Type *DstTy) { +unsigned SystemZTTIImpl::getVectorBitmaskConversionCost(Type *SrcTy, + Type *DstTy) const { assert (SrcTy->isVectorTy() && DstTy->isVectorTy() && "Should only be called with vector types."); @@ -855,9 +885,9 @@ static Type *getCmpOpsType(const Instruction *I, unsigned VF = 1) { // Get the cost of converting a boolean vector to a vector with same width // and element size as Dst, plus the cost of zero extending if needed. -unsigned SystemZTTIImpl:: -getBoolVecToIntConversionCost(unsigned Opcode, Type *Dst, - const Instruction *I) { +unsigned +SystemZTTIImpl::getBoolVecToIntConversionCost(unsigned Opcode, Type *Dst, + const Instruction *I) const { auto *DstVTy = cast<FixedVectorType>(Dst); unsigned VF = DstVTy->getNumElements(); unsigned Cost = 0; @@ -876,7 +906,7 @@ InstructionCost SystemZTTIImpl::getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, - const Instruction *I) { + const Instruction *I) const { // FIXME: Can the logic below also be used for these cost kinds? if (CostKind == TTI::TCK_CodeSize || CostKind == TTI::TCK_SizeAndLatency) { auto BaseCost = BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I); @@ -887,7 +917,8 @@ InstructionCost SystemZTTIImpl::getCastInstrCost(unsigned Opcode, Type *Dst, unsigned SrcScalarBits = Src->getScalarSizeInBits(); if (!Src->isVectorTy()) { - assert (!Dst->isVectorTy()); + if (Dst->isVectorTy()) + return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I); if (Opcode == Instruction::SIToFP || Opcode == Instruction::UIToFP) { if (Src->isIntegerTy(128)) @@ -1072,7 +1103,7 @@ static unsigned getOperandsExtensionCost(const Instruction *I) { InstructionCost SystemZTTIImpl::getCmpSelInstrCost( unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, - TTI::OperandValueInfo Op2Info, const Instruction *I) { + TTI::OperandValueInfo Op2Info, const Instruction *I) const { if (CostKind != TTI::TCK_RecipThroughput) return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind, Op1Info, Op2Info); @@ -1166,8 +1197,9 @@ InstructionCost SystemZTTIImpl::getCmpSelInstrCost( InstructionCost SystemZTTIImpl::getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, - unsigned Index, Value *Op0, - Value *Op1) { + unsigned Index, + const Value *Op0, + const Value *Op1) const { if (Opcode == Instruction::InsertElement) { // Vector Element Load. if (Op1 != nullptr && isFreeEltLoad(Op1)) @@ -1194,8 +1226,8 @@ InstructionCost SystemZTTIImpl::getVectorInstrCost(unsigned Opcode, Type *Val, } // Check if a load may be folded as a memory operand in its user. -bool SystemZTTIImpl:: -isFoldableLoad(const LoadInst *Ld, const Instruction *&FoldedValue) { +bool SystemZTTIImpl::isFoldableLoad(const LoadInst *Ld, + const Instruction *&FoldedValue) const { if (!Ld->hasOneUse()) return false; FoldedValue = Ld; @@ -1283,11 +1315,11 @@ static bool isBswapIntrinsicCall(const Value *V) { } InstructionCost SystemZTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src, - MaybeAlign Alignment, + Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo, - const Instruction *I) { + const Instruction *I) const { assert(!Src->isVoidTy() && "Invalid type"); // TODO: Handle other cost kinds. @@ -1362,7 +1394,7 @@ InstructionCost SystemZTTIImpl::getMemoryOpCost(unsigned Opcode, Type *Src, InstructionCost SystemZTTIImpl::getInterleavedMemoryOpCost( unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, - bool UseMaskForCond, bool UseMaskForGaps) { + bool UseMaskForCond, bool UseMaskForGaps) const { if (UseMaskForCond || UseMaskForGaps) return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices, Alignment, AddressSpace, CostKind, @@ -1442,7 +1474,7 @@ inline bool customCostReductions(unsigned Opcode) { InstructionCost SystemZTTIImpl::getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional<FastMathFlags> FMF, - TTI::TargetCostKind CostKind) { + TTI::TargetCostKind CostKind) const { unsigned ScalarBits = Ty->getScalarSizeInBits(); // The following is only for subtargets with vector math, non-ordered // reductions, and reasonable scalar sizes for int and fp add/mul. @@ -1469,7 +1501,7 @@ SystemZTTIImpl::getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, InstructionCost SystemZTTIImpl::getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, - TTI::TargetCostKind CostKind) { + TTI::TargetCostKind CostKind) const { // Return custom costs only on subtargets with vector enhancements. if (ST->hasVectorEnhancements1()) { unsigned NumVectors = getNumVectorRegs(Ty); @@ -1498,7 +1530,7 @@ getVectorIntrinsicInstrCost(Intrinsic::ID ID, Type *RetTy, InstructionCost SystemZTTIImpl::getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, - TTI::TargetCostKind CostKind) { + TTI::TargetCostKind CostKind) const { InstructionCost Cost = getVectorIntrinsicInstrCost( ICA.getID(), ICA.getReturnType(), ICA.getArgTypes()); if (Cost != -1) |
