19#include "llvm/IR/IntrinsicsRISCV.h"
28#define DEBUG_TYPE "riscvtti"
31 "riscv-v-register-bit-width-lmul",
33 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
34 "by autovectorized code. Fractional LMULs are not supported."),
40 "Overrides result used for getMaximumVF query which is used "
41 "exclusively by SLP vectorizer."),
46 cl::desc(
"Set the lower bound of a trip count to decide on "
47 "vectorization while tail-folding."),
59 size_t NumInstr = OpCodes.size();
64 return LMULCost * NumInstr;
66 for (
auto Op : OpCodes) {
68 case RISCV::VRGATHER_VI:
71 case RISCV::VRGATHER_VV:
74 case RISCV::VSLIDEUP_VI:
75 case RISCV::VSLIDEDOWN_VI:
78 case RISCV::VSLIDEUP_VX:
79 case RISCV::VSLIDEDOWN_VX:
82 case RISCV::VREDMAX_VS:
83 case RISCV::VREDMIN_VS:
84 case RISCV::VREDMAXU_VS:
85 case RISCV::VREDMINU_VS:
86 case RISCV::VREDSUM_VS:
87 case RISCV::VREDAND_VS:
88 case RISCV::VREDOR_VS:
89 case RISCV::VREDXOR_VS:
90 case RISCV::VFREDMAX_VS:
91 case RISCV::VFREDMIN_VS:
92 case RISCV::VFREDUSUM_VS: {
99 case RISCV::VFREDOSUM_VS: {
107 case RISCV::VFMV_F_S:
112 case RISCV::VFMV_S_F:
114 case RISCV::VMXOR_MM:
115 case RISCV::VMAND_MM:
116 case RISCV::VMANDN_MM:
117 case RISCV::VMNAND_MM:
119 case RISCV::VFIRST_M:
138 assert(Ty->isIntegerTy() &&
139 "getIntImmCost can only estimate cost of materialising integers");
162 if (!BO || !BO->hasOneUse())
165 if (BO->getOpcode() != Instruction::Shl)
176 if (ShAmt == Trailing)
193 if (!Cmp || !Cmp->isEquality())
209 if ((CmpC & Mask) != CmpC)
216 return NewCmpC >= -2048 && NewCmpC <= 2048;
223 assert(Ty->isIntegerTy() &&
224 "getIntImmCost can only estimate cost of materialising integers");
232 bool Takes12BitImm =
false;
233 unsigned ImmArgIdx = ~0U;
236 case Instruction::GetElementPtr:
241 case Instruction::Store: {
246 if (Idx == 1 || !Inst)
251 if (!getTLI()->allowsMemoryAccessForAlignment(
252 Ty->getContext(),
DL, getTLI()->getValueType(
DL, Ty),
259 case Instruction::Load:
262 case Instruction::And:
264 if (
Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
267 if (
Imm == UINT64_C(0xffffffff) && (!ST->is64Bit() || ST->hasStdExtZba()))
270 if (ST->hasStdExtZbs() && (~
Imm).isPowerOf2())
272 if (Inst && Idx == 1 &&
Imm.getBitWidth() <= ST->getXLen() &&
275 if (Inst && Idx == 1 &&
Imm.getBitWidth() == 64 &&
278 Takes12BitImm =
true;
280 case Instruction::Add:
281 Takes12BitImm =
true;
283 case Instruction::Or:
284 case Instruction::Xor:
286 if (ST->hasStdExtZbs() &&
Imm.isPowerOf2())
288 Takes12BitImm =
true;
290 case Instruction::Mul:
292 if (
Imm.isPowerOf2() ||
Imm.isNegatedPowerOf2())
295 if ((
Imm + 1).isPowerOf2() || (
Imm - 1).isPowerOf2())
298 Takes12BitImm =
true;
300 case Instruction::Sub:
301 case Instruction::Shl:
302 case Instruction::LShr:
303 case Instruction::AShr:
304 Takes12BitImm =
true;
315 if (
Imm.getSignificantBits() <= 64 &&
338 return ST->hasVInstructions();
348 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
352 if (Opcode == Instruction::FAdd)
361 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
362 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
363 InputTypeA != InputTypeB || !InputTypeA->
isIntegerTy(8) ||
379 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second,
CostKind);
388 std::pair<InstructionCost, MVT> AccLT =
396 bool WidenFirst =
false;
397 if (VF.
isScalable() && AccLT.second.isScalableVector()) {
398 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
411 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
414 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second,
CostKind);
418 std::pair<InstructionCost, MVT> RedLT =
420 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
422 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
426 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
438 switch (
II->getIntrinsicID()) {
442 case Intrinsic::vector_reduce_mul:
443 case Intrinsic::vector_reduce_fmul:
449 if (ST->hasVInstructions())
450 if (
unsigned MinVLen = ST->getRealMinVLen();
465 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
468 (ST->hasVInstructions() &&
491 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
497RISCVTTIImpl::getConstantPoolLoadCost(
Type *Ty,
502 return getStaticDataAddrGenerationCost(
CostKind) +
508 unsigned Size = Mask.size();
511 for (
unsigned I = 0;
I !=
Size; ++
I) {
512 if (
static_cast<unsigned>(Mask[
I]) ==
I)
518 for (
unsigned J =
I + 1; J !=
Size; ++J)
520 if (
static_cast<unsigned>(Mask[J]) != J %
I)
548 "Expected fixed vector type and non-empty mask");
551 unsigned NumOfDests =
divideCeil(Mask.size(), LegalNumElts);
555 if (NumOfDests <= 1 ||
557 Tp->getElementType()->getPrimitiveSizeInBits() ||
558 LegalNumElts >= Tp->getElementCount().getFixedValue())
561 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
564 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
568 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
569 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
570 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
572 assert(NormalizedVF >= Mask.size() &&
573 "Normalized mask expected to be not shorter than original mask.");
578 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
579 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
582 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
585 Cost +=
TTI.getShuffleCost(
588 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
590 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
591 Cost +=
TTI.getShuffleCost(
594 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
617 if (!VLen || Mask.empty())
621 LegalVT =
TTI.getTypeLegalizationCost(
627 if (NumOfDests <= 1 ||
629 Tp->getElementType()->getPrimitiveSizeInBits() ||
633 unsigned VecTySize =
TTI.getDataLayout().getTypeStoreSize(Tp);
636 unsigned NumOfSrcs =
divideCeil(VecTySize, LegalVTSize);
642 unsigned NormalizedVF =
647 assert(NormalizedVF >= Mask.size() &&
648 "Normalized mask expected to be not shorter than original mask.");
654 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
655 [&](
ArrayRef<int> RegMask,
unsigned SrcReg,
unsigned DestReg) {
658 if (!ReusedSingleSrcShuffles.
insert(std::make_pair(RegMask, SrcReg))
663 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
665 [&](
ArrayRef<int> RegMask,
unsigned Idx1,
unsigned Idx2,
bool NewReg) {
667 SingleOpTy,
CostKind, RegMask, 0,
nullptr);
674 if ((NumOfDestRegs > 2 && NumShuffles <=
static_cast<int>(NumOfDestRegs)) ||
675 (NumOfDestRegs <= 2 && NumShuffles < 4))
690 if (!
LT.second.isFixedLengthVector())
698 auto GetSlideOpcode = [&](
int SlideAmt) {
700 bool IsVI =
isUInt<5>(std::abs(SlideAmt));
702 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
703 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
706 std::array<std::pair<int, int>, 2> SrcInfo;
710 if (SrcInfo[1].second == 0)
713 if (ST->hasStdExtZvzip() &&
LT.second.getScalarSizeInBits() != 1) {
715 if (
isPairEven(SrcInfo, Mask, Factor) && Factor == 1)
716 return getRISCVInstructionCost(RISCV::VPAIRE_VV,
LT.second,
CostKind);
717 if (
isPairOdd(SrcInfo, Mask, Factor) && Factor == 1)
718 return getRISCVInstructionCost(RISCV::VPAIRO_VV,
LT.second,
CostKind);
722 if (SrcInfo[0].second != 0) {
723 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
724 FirstSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
727 if (SrcInfo[1].first == -1)
728 return FirstSlideCost;
731 if (SrcInfo[1].second != 0) {
732 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
733 SecondSlideCost = getRISCVInstructionCost(Opcode,
LT.second,
CostKind);
736 getRISCVInstructionCost(RISCV::VMERGE_VVM,
LT.second,
CostKind);
743 return FirstSlideCost + SecondSlideCost + MaskCost;
746std::optional<MVT> RISCVTTIImpl::getZvzipVZIPCostVT(
MVT InterleavedVT)
const {
756 LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
758 return InterleavedVT;
761std::optional<MVT> RISCVTTIImpl::getZvzipVUNZIPCostVT(
MVT InterleavedVT)
const {
769 return InterleavedVT;
779 "Expected the Mask to match the return size if given");
781 "Expected the same scalar types");
784 if (VIC == TTI::VectorInstrContext::SplatOpFolded &&
800 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
802 *
this, LT.second, ST->getRealVLen(),
804 if (VRegSplittingCost.
isValid())
805 return VRegSplittingCost;
810 if (Mask.size() >= 2) {
811 MVT EltTp = LT.second.getVectorElementType();
822 return 2 * LT.first * TLI->getLMULCost(LT.second);
824 if (Mask[0] == 0 || Mask[0] == 1) {
828 if (
equal(DeinterleaveMask, Mask))
829 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
834 if (LT.second.getScalarSizeInBits() != 1 &&
837 unsigned NumSlides =
Log2_32(Mask.size() / SubVectorSize);
839 for (
unsigned I = 0;
I != NumSlides; ++
I) {
840 unsigned InsertIndex = SubVectorSize * (1 <<
I);
845 std::pair<InstructionCost, MVT> DestLT =
850 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
864 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
865 LT.second.getVectorNumElements() <= 256)) {
870 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second,
CostKind);
884 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
885 LT.second.getVectorNumElements() <= 256)) {
886 auto &
C = SrcTy->getContext();
887 auto EC = SrcTy->getElementCount();
892 return 2 * IndexCost +
893 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
912 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
940 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
941 if (std::optional<unsigned> VLen = ST->getRealVLen();
942 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
943 SubLT.second.getSizeInBits() <= *VLen)
951 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second,
CostKind);
958 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second,
CostKind);
970 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
977 if (IsLoad && LT.second.isVector() &&
979 LT.second.getVectorElementCount()))
983 Instruction::InsertElement);
984 if (LT.second.getScalarSizeInBits() == 1) {
992 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
1005 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
1006 RISCV::VMV_X_S, RISCV::VMV_V_X,
1015 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second,
CostKind);
1021 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second,
CostKind);
1027 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
1028 if (Index >= 0 && Index < 32)
1029 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
1030 else if (Index < 0 && Index > -32)
1031 Opcodes[1] = RISCV::VSLIDEUP_VI;
1032 return LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1036 if (!LT.second.isVector())
1042 if (SrcTy->getElementType()->isIntegerTy(1)) {
1054 MVT ContainerVT = LT.second;
1055 if (LT.second.isFixedLengthVector())
1056 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1058 if (ContainerVT.
bitsLE(M1VT)) {
1068 if (LT.second.isFixedLengthVector())
1070 LenCost =
isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1071 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1072 if (LT.second.isFixedLengthVector() &&
1073 isInt<5>(LT.second.getVectorNumElements() - 1))
1074 Opcodes[1] = RISCV::VRSUB_VI;
1076 getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1077 return LT.first * (LenCost + GatherCost);
1084 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1086 getRISCVInstructionCost(M1Opcodes, M1VT,
CostKind) + 3;
1090 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT,
CostKind) * Ratio;
1092 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second,
CostKind);
1093 return FixedCost + LT.first * (GatherCost + SlideCost);
1127 Ty, DemandedElts, Insert, Extract,
CostKind);
1129 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1130 if (Ty->getScalarSizeInBits() == 1) {
1140 assert(LT.second.isFixedLengthVector());
1141 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1145 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second,
CostKind);
1158 switch (MICA.
getID()) {
1159 case Intrinsic::vp_load_ff: {
1160 EVT DataTypeVT = TLI->getValueType(
DL, DataTy);
1161 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1168 case Intrinsic::experimental_vp_strided_load:
1169 case Intrinsic::experimental_vp_strided_store:
1171 case Intrinsic::masked_compressstore:
1172 case Intrinsic::masked_expandload:
1174 case Intrinsic::vp_scatter:
1175 case Intrinsic::vp_gather:
1176 case Intrinsic::masked_scatter:
1177 case Intrinsic::masked_gather:
1179 case Intrinsic::vp_load:
1180 case Intrinsic::vp_store:
1181 case Intrinsic::masked_load:
1182 case Intrinsic::masked_store:
1191 unsigned Opcode = MICA.
getID() == Intrinsic::masked_load ? Instruction::Load
1192 : Instruction::Store;
1203 if (MICA.
getID() == Intrinsic::vp_load ||
1204 MICA.
getID() == Intrinsic::vp_store) {
1216 bool UseMaskForCond,
bool UseMaskForGaps)
const {
1222 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1226 if (LT.second.isVector()) {
1232 VTy->getElementCount().divideCoefficientBy(Factor));
1233 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1234 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1239 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1240 unsigned VecSizeInBits =
1241 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1242 unsigned VLENForTuning =
1244 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1246 MVT SubVecVT = getTLI()->getValueType(
DL, SubVecTy).getSimpleVT();
1247 Cost += Factor * TLI->getLMULCost(SubVecVT);
1253 unsigned NumLoads = getEstimatedVLFor(VTy);
1269 if (UseMaskForGaps) {
1272 "Indices should not contain duplicate elements");
1273 unsigned NumOfFields = Indices.
size();
1274 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.
back() + 1);
1275 if (IsTailGapOnly &&
1276 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1278 if (LT.second.isVector() &&
1279 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1281 FVTy->getElementType(),
1282 FVTy->getElementCount().divideCoefficientBy(Factor));
1283 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1286 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1295 unsigned VF = FVTy->getNumElements() / Factor;
1302 if (Opcode == Instruction::Load) {
1304 for (
unsigned Index : Indices) {
1308 Mask.resize(VF * Factor, -1);
1312 Cost += ShuffleCost;
1330 UseMaskForCond, UseMaskForGaps);
1332 assert(Opcode == Instruction::Store &&
"Opcode must be a store");
1339 return MemCost + ShuffleCost;
1346 bool IsLoad = MICA.
getID() == Intrinsic::masked_gather ||
1347 MICA.
getID() == Intrinsic::vp_gather;
1348 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1356 if ((Opcode == Instruction::Load &&
1358 (Opcode == Instruction::Store &&
1364 if (MICA.
getID() == Intrinsic::vp_gather ||
1365 MICA.
getID() == Intrinsic::vp_scatter) {
1368 if (DataLT.first > 1)
1370 if (PtrLT.first > 1)
1378 unsigned NumLoads = getEstimatedVLFor(&VTy);
1385 unsigned Opcode = MICA.
getID() == Intrinsic::masked_expandload
1387 : Instruction::Store;
1391 bool IsLegal = (Opcode == Instruction::Store &&
1393 (Opcode == Instruction::Load &&
1417 if (Opcode == Instruction::Store)
1418 Opcodes.
append({RISCV::VCOMPRESS_VM});
1420 Opcodes.
append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1422 LT.first * getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
1447 unsigned NumLoads = getEstimatedVLFor(&VTy);
1450 uint64_t CacheLineBytes = ST->getCacheLineSize();
1451 if (!CacheLineBytes)
1452 CacheLineBytes = 64;
1455 int64_t Stride = StrideCI->getSExtValue();
1457 if (Stride != std::numeric_limits<int64_t>::min() && Stride != 0) {
1458 uint64_t AbsStride = (uint64_t)std::abs(Stride);
1459 if (AbsStride < CacheLineBytes) {
1460 uint64_t MaxCombines = ST->getMaxVectorCoalesceElts();
1461 if ((CacheLineBytes / AbsStride) >= MaxCombines)
1462 NumLoads =
divideCeil(NumLoads, MaxCombines);
1466 NumLoads =
divideCeil((NumLoads * AbsStride), CacheLineBytes);
1480 for (
auto *Ty : Tys) {
1481 if (!Ty->isVectorTy())
1495 {Intrinsic::floor, MVT::f32, 9},
1496 {Intrinsic::floor, MVT::f64, 9},
1497 {Intrinsic::ceil, MVT::f32, 9},
1498 {Intrinsic::ceil, MVT::f64, 9},
1499 {Intrinsic::trunc, MVT::f32, 7},
1500 {Intrinsic::trunc, MVT::f64, 7},
1501 {Intrinsic::round, MVT::f32, 9},
1502 {Intrinsic::round, MVT::f64, 9},
1503 {Intrinsic::roundeven, MVT::f32, 9},
1504 {Intrinsic::roundeven, MVT::f64, 9},
1505 {Intrinsic::rint, MVT::f32, 7},
1506 {Intrinsic::rint, MVT::f64, 7},
1507 {Intrinsic::nearbyint, MVT::f32, 9},
1508 {Intrinsic::nearbyint, MVT::f64, 9},
1509 {Intrinsic::bswap, MVT::i16, 3},
1510 {Intrinsic::bswap, MVT::i32, 12},
1511 {Intrinsic::bswap, MVT::i64, 31},
1512 {Intrinsic::bitreverse, MVT::i8, 17},
1513 {Intrinsic::bitreverse, MVT::i16, 24},
1514 {Intrinsic::bitreverse, MVT::i32, 33},
1515 {Intrinsic::bitreverse, MVT::i64, 52},
1516 {Intrinsic::ctpop, MVT::i8, 12},
1517 {Intrinsic::ctpop, MVT::i16, 19},
1518 {Intrinsic::ctpop, MVT::i32, 20},
1519 {Intrinsic::ctpop, MVT::i64, 21},
1520 {Intrinsic::ctlz, MVT::i8, 19},
1521 {Intrinsic::ctlz, MVT::i16, 28},
1522 {Intrinsic::ctlz, MVT::i32, 31},
1523 {Intrinsic::ctlz, MVT::i64, 35},
1524 {Intrinsic::cttz, MVT::i8, 16},
1525 {Intrinsic::cttz, MVT::i16, 23},
1526 {Intrinsic::cttz, MVT::i32, 24},
1527 {Intrinsic::cttz, MVT::i64, 25},
1534 switch (ICA.
getID()) {
1535 case Intrinsic::lrint:
1536 case Intrinsic::llrint:
1537 case Intrinsic::lround:
1538 case Intrinsic::llround: {
1542 if (ST->hasVInstructions() && LT.second.isVector()) {
1544 unsigned SrcEltSz =
DL.getTypeSizeInBits(SrcTy->getScalarType());
1545 unsigned DstEltSz =
DL.getTypeSizeInBits(RetTy->getScalarType());
1546 if (LT.second.getVectorElementType() == MVT::bf16) {
1547 if (!ST->hasVInstructionsBF16Minimal())
1550 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1552 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1553 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1554 !ST->hasVInstructionsF16()) {
1555 if (!ST->hasVInstructionsF16Minimal())
1558 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1560 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1562 }
else if (SrcEltSz > DstEltSz) {
1563 Ops = {RISCV::VFNCVT_X_F_W};
1564 }
else if (SrcEltSz < DstEltSz) {
1565 Ops = {RISCV::VFWCVT_X_F_V};
1567 Ops = {RISCV::VFCVT_X_F_V};
1572 if (SrcEltSz > DstEltSz)
1573 return SrcLT.first *
1574 getRISCVInstructionCost(
Ops, SrcLT.second,
CostKind);
1575 return LT.first * getRISCVInstructionCost(
Ops, LT.second,
CostKind);
1579 case Intrinsic::ceil:
1580 case Intrinsic::floor:
1581 case Intrinsic::trunc:
1582 case Intrinsic::rint:
1583 case Intrinsic::round:
1584 case Intrinsic::roundeven: {
1587 if (!LT.second.isVector() && TLI->isOperationCustom(
ISD::FCEIL, LT.second))
1588 return LT.first * 8;
1591 case Intrinsic::umin:
1592 case Intrinsic::umax:
1593 case Intrinsic::smin:
1594 case Intrinsic::smax: {
1596 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1599 if (ST->hasVInstructions() && LT.second.isVector()) {
1601 switch (ICA.
getID()) {
1602 case Intrinsic::umin:
1603 Op = RISCV::VMINU_VV;
1605 case Intrinsic::umax:
1606 Op = RISCV::VMAXU_VV;
1608 case Intrinsic::smin:
1609 Op = RISCV::VMIN_VV;
1611 case Intrinsic::smax:
1612 Op = RISCV::VMAX_VV;
1615 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1619 case Intrinsic::sadd_sat:
1620 case Intrinsic::ssub_sat:
1621 case Intrinsic::uadd_sat:
1622 case Intrinsic::usub_sat: {
1624 if (ST->hasVInstructions() && LT.second.isVector()) {
1626 switch (ICA.
getID()) {
1627 case Intrinsic::sadd_sat:
1628 Op = RISCV::VSADD_VV;
1630 case Intrinsic::ssub_sat:
1631 Op = RISCV::VSSUB_VV;
1633 case Intrinsic::uadd_sat:
1634 Op = RISCV::VSADDU_VV;
1636 case Intrinsic::usub_sat:
1637 Op = RISCV::VSSUBU_VV;
1640 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1644 case Intrinsic::fma:
1645 case Intrinsic::fmuladd: {
1648 if (ST->hasVInstructions() && LT.second.isVector())
1650 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second,
CostKind);
1653 case Intrinsic::fabs: {
1655 if (ST->hasVInstructions() && LT.second.isVector()) {
1661 if (LT.second.getVectorElementType() == MVT::bf16 ||
1662 (LT.second.getVectorElementType() == MVT::f16 &&
1663 !ST->hasVInstructionsF16()))
1664 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1669 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second,
CostKind);
1673 case Intrinsic::sqrt: {
1675 if (ST->hasVInstructions() && LT.second.isVector()) {
1678 MVT ConvType = LT.second;
1679 MVT FsqrtType = LT.second;
1682 if (LT.second.getVectorElementType() == MVT::bf16) {
1683 if (LT.second == MVT::nxv32bf16) {
1684 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1685 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1686 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1687 ConvType = MVT::nxv16f16;
1688 FsqrtType = MVT::nxv16f32;
1690 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1691 FsqrtOp = {RISCV::VFSQRT_V};
1692 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1694 }
else if (LT.second.getVectorElementType() == MVT::f16 &&
1695 !ST->hasVInstructionsF16()) {
1696 if (LT.second == MVT::nxv32f16) {
1697 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1698 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1699 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1700 ConvType = MVT::nxv16f16;
1701 FsqrtType = MVT::nxv16f32;
1703 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1704 FsqrtOp = {RISCV::VFSQRT_V};
1705 FsqrtType = TLI->getTypeToPromoteTo(
ISD::FSQRT, FsqrtType);
1708 FsqrtOp = {RISCV::VFSQRT_V};
1711 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType,
CostKind) +
1712 getRISCVInstructionCost(ConvOp, ConvType,
CostKind));
1716 case Intrinsic::cttz:
1717 case Intrinsic::ctlz:
1718 case Intrinsic::ctpop: {
1720 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1722 switch (ICA.
getID()) {
1723 case Intrinsic::cttz:
1726 case Intrinsic::ctlz:
1729 case Intrinsic::ctpop:
1730 Op = RISCV::VCPOP_V;
1733 return LT.first * getRISCVInstructionCost(
Op, LT.second,
CostKind);
1737 case Intrinsic::abs: {
1739 if (ST->hasVInstructions() && LT.second.isVector()) {
1741 if (ST->hasStdExtZvabd())
1743 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second,
CostKind);
1748 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1753 case Intrinsic::fshl:
1754 case Intrinsic::fshr: {
1761 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1763 (RetTy->getIntegerBitWidth() == 32 ||
1764 RetTy->getIntegerBitWidth() == 64) &&
1765 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1770 case Intrinsic::clmul: {
1772 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbkc()) {
1775 if (!ST->is64Bit() || LT.second != MVT::i64)
1781 return LT.first * getRISCVInstructionCost(
1782 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1787 case Intrinsic::masked_udiv:
1790 case Intrinsic::masked_sdiv:
1793 case Intrinsic::masked_urem:
1796 case Intrinsic::masked_srem:
1799 case Intrinsic::get_active_lane_mask: {
1800 if (ST->hasVInstructions()) {
1809 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1815 case Intrinsic::stepvector: {
1819 if (ST->hasVInstructions())
1820 return getRISCVInstructionCost(RISCV::VID_V, LT.second,
CostKind) +
1822 getRISCVInstructionCost(RISCV::VADD_VX, LT.second,
CostKind);
1823 return 1 + (LT.first - 1);
1825 case Intrinsic::vector_splice_left:
1826 case Intrinsic::vector_splice_right: {
1831 if (ST->hasVInstructions() && LT.second.isVector()) {
1833 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1838 case Intrinsic::experimental_cttz_elts: {
1839 if (!ST->hasVInstructions())
1844 if (!LT.second.isVector())
1848 if (LT.second.getVectorElementType() != MVT::i1)
1849 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second,
CostKind);
1851 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind);
1863 return LT.first *
Cost;
1865 case Intrinsic::experimental_vp_splice: {
1873 case Intrinsic::vp_merge: {
1881 case Intrinsic::fptoui_sat:
1882 case Intrinsic::fptosi_sat: {
1884 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
1889 if (!SrcTy->isVectorTy())
1892 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1909 case Intrinsic::experimental_vector_extract_last_active: {
1931 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1932 TLI->getVectorIdxTy(
getDataLayout()), MaskTy->getElementCount(),
1933 true, &VScaleRange);
1934 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1942 if (StepLT.first > 1)
1946 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1948 Cost += MaskLT.first *
1949 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second,
CostKind);
1951 Cost += StepLT.first *
1952 getRISCVInstructionCost(Opcodes, StepLT.second,
CostKind);
1956 Cost += ValLT.first *
1957 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1961 case Intrinsic::vector_interleave2:
1962 case Intrinsic::vector_deinterleave2: {
1963 if (!ST->hasStdExtZvzip())
1966 bool IsInterleave = ICA.
getID() == Intrinsic::vector_interleave2;
1967 Type *InterleavedTy = IsInterleave ? RetTy : ICA.
getArgTypes().front();
1971 [](
const Value *Arg) { return isa<UndefValue>(Arg); }))
1978 unsigned HalfVF = HalfFVT->getNumElements();
1983 for (
unsigned Start = 0; Start != 2; ++Start)
1990 if (!LT.second.isScalableVector())
1993 if (std::optional<MVT> CostVT = getZvzipVZIPCostVT(LT.second))
1995 getRISCVInstructionCost(RISCV::VZIP_VV, *CostVT,
CostKind);
1996 }
else if (std::optional<MVT> CostVT = getZvzipVUNZIPCostVT(LT.second)) {
1998 getRISCVInstructionCost({RISCV::VUNZIPE_V, RISCV::VUNZIPO_V},
2005 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
2007 LT.second.isVector()) {
2008 MVT EltTy = LT.second.getVectorElementType();
2010 ICA.
getID(), EltTy))
2011 return LT.first * Entry->Cost;
2024 if (ST->hasVInstructions() && PtrTy->
isVectorTy())
2042 if (ST->hasStdExtP() &&
2050 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
2051 Dst->getScalarSizeInBits() > ST->getELen())
2054 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2069 if (Src->getScalarSizeInBits() == 1) {
2074 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second,
CostKind) +
2075 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
2081 if (Dst->getScalarSizeInBits() == 1) {
2087 return SrcLT.first *
2088 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
2100 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
2101 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
2103 SrcLT.second.getSizeInBits()) ||
2105 DstLT.second.getSizeInBits()) ||
2106 SrcLT.first > 1 || DstLT.first > 1)
2110 assert((SrcLT.first == 1) && (DstLT.first == 1) &&
"Illegal type");
2112 int PowDiff = (int)
Log2_32(DstLT.second.getScalarSizeInBits()) -
2113 (int)
Log2_32(SrcLT.second.getScalarSizeInBits());
2117 if ((PowDiff < 1) || (PowDiff > 3))
2119 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
2120 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
2123 return getRISCVInstructionCost(
Op, DstLT.second,
CostKind);
2129 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
2130 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
2134 : RISCV::VFNCVT_F_F_W;
2136 for (; SrcEltSize != DstEltSize;) {
2140 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2142 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2150 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2152 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2154 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2155 unsigned SrcEltSize = Src->getScalarSizeInBits();
2156 unsigned DstEltSize = Dst->getScalarSizeInBits();
2158 if ((SrcEltSize == 16) &&
2159 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2165 std::pair<InstructionCost, MVT> VecF32LT =
2168 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2173 if (DstEltSize == SrcEltSize)
2174 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2175 else if (DstEltSize > SrcEltSize)
2176 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2181 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2182 Cost += getRISCVInstructionCost(FNCVT, VecVT,
CostKind);
2183 if ((SrcEltSize / 2) > DstEltSize) {
2194 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2195 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2196 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2197 unsigned SrcEltSize = Src->getScalarSizeInBits();
2198 unsigned DstEltSize = Dst->getScalarSizeInBits();
2201 if ((DstEltSize == 16) &&
2202 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2208 std::pair<InstructionCost, MVT> VecF32LT =
2211 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2216 if (DstEltSize == SrcEltSize)
2217 Cost += getRISCVInstructionCost(FCVT, DstLT.second,
CostKind);
2218 else if (DstEltSize > SrcEltSize) {
2219 if ((DstEltSize / 2) > SrcEltSize) {
2223 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2226 Cost += getRISCVInstructionCost(FWCVT, DstLT.second,
CostKind);
2228 Cost += getRISCVInstructionCost(FNCVT, DstLT.second,
CostKind);
2235unsigned RISCVTTIImpl::getEstimatedVLFor(
VectorType *Ty)
const {
2237 const unsigned EltSize =
DL.getTypeSizeInBits(Ty->getElementType());
2238 const unsigned MinSize =
DL.getTypeSizeInBits(Ty).getKnownMinValue();
2253 if (Ty->getScalarSizeInBits() > ST->getELen())
2257 if (Ty->getElementType()->isIntegerTy(1)) {
2261 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2267 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2271 case Intrinsic::maximum:
2273 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2275 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2290 case Intrinsic::minimum:
2292 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2294 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2300 const unsigned EltTyBits =
DL.getTypeSizeInBits(DstTy);
2309 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2318 case Intrinsic::smax:
2319 SplitOp = RISCV::VMAX_VV;
2320 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2322 case Intrinsic::smin:
2323 SplitOp = RISCV::VMIN_VV;
2324 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2326 case Intrinsic::umax:
2327 SplitOp = RISCV::VMAXU_VV;
2328 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2330 case Intrinsic::umin:
2331 SplitOp = RISCV::VMINU_VV;
2332 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2334 case Intrinsic::maxnum:
2335 SplitOp = RISCV::VFMAX_VV;
2336 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2338 case Intrinsic::minnum:
2339 SplitOp = RISCV::VFMIN_VV;
2340 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2345 (LT.first > 1) ? (LT.first - 1) *
2346 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2348 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2353 std::optional<FastMathFlags> FMF,
2359 if (Ty->getScalarSizeInBits() > ST->getELen())
2362 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2370 Type *ElementTy = Ty->getElementType();
2375 if (LT.second == MVT::v1i1)
2376 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second,
CostKind) +
2394 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2395 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second,
CostKind) +
2396 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second,
CostKind) +
2397 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2406 return (LT.first - 1) *
2407 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind) +
2408 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) + 1;
2416 return (LT.first - 1) *
2417 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second,
CostKind) +
2418 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind) +
2431 SplitOp = RISCV::VADD_VV;
2432 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2435 SplitOp = RISCV::VOR_VV;
2436 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2439 SplitOp = RISCV::VXOR_VV;
2440 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2443 SplitOp = RISCV::VAND_VV;
2444 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2448 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2449 LT.second.getScalarType() == MVT::bf16)
2453 for (
unsigned i = 0; i < LT.first.getValue(); i++)
2456 return getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2458 SplitOp = RISCV::VFADD_VV;
2459 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2464 (LT.first > 1) ? (LT.first - 1) *
2465 getRISCVInstructionCost(SplitOp, LT.second,
CostKind)
2467 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second,
CostKind);
2471 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *ValTy,
2482 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2488 if (IsUnsigned && Opcode == Instruction::Add &&
2489 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2493 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second,
CostKind);
2500 return (LT.first - 1) +
2507 assert(OpInfo.isConstant() &&
"non constant operand?");
2514 if (OpInfo.isUniform())
2520 return getConstantPoolLoadCost(Ty,
CostKind);
2529 EVT VT = TLI->getValueType(
DL, Src,
true);
2531 if (VT == MVT::Other ||
2537 if (Opcode == Instruction::Store && OpInfo.isConstant())
2552 if (Src->
isVectorTy() && LT.second.isVector() &&
2554 LT.second.getSizeInBits()))
2564 if (ST->hasVInstructions() && LT.second.isVector() &&
2566 BaseCost *= TLI->getLMULCost(LT.second);
2567 return Cost + BaseCost;
2576 Op1Info, Op2Info,
I);
2580 Op1Info, Op2Info,
I);
2585 Op1Info, Op2Info,
I);
2587 auto GetConstantMatCost =
2589 if (OpInfo.isUniform())
2594 return getConstantPoolLoadCost(ValTy,
CostKind);
2599 ConstantMatCost += GetConstantMatCost(Op1Info);
2601 ConstantMatCost += GetConstantMatCost(Op2Info);
2604 if (Opcode == Instruction::Select && LT.second.isVector()) {
2605 if (CondTy->isVectorTy()) {
2610 return ConstantMatCost +
2612 getRISCVInstructionCost(
2613 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2617 return ConstantMatCost +
2618 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2628 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2629 return ConstantMatCost +
2631 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2633 LT.first * getRISCVInstructionCost(
2634 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2641 return ConstantMatCost +
2642 LT.first * getRISCVInstructionCost(
2643 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2647 if ((Opcode == Instruction::ICmp) && ValTy->
isVectorTy() &&
2651 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2656 if ((Opcode == Instruction::FCmp) && ValTy->
isVectorTy() &&
2661 return ConstantMatCost +
2662 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second,
CostKind);
2672 Op1Info, Op2Info,
I);
2681 return ConstantMatCost +
2682 LT.first * getRISCVInstructionCost(
2683 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2690 return ConstantMatCost +
2692 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2701 return ConstantMatCost +
2703 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second,
CostKind);
2716 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2717 U->getType()->isIntegerTy() &&
2718 !isa<ConstantData>(U->getOperand(1)) &&
2719 !isa<ConstantData>(U->getOperand(2));
2727 Op1Info, Op2Info,
I);
2734 return Opcode == Instruction::PHI ? 0 : 1;
2751 if (Opcode != Instruction::ExtractElement &&
2752 Opcode != Instruction::InsertElement)
2758 if (Opcode == Instruction::InsertElement &&
2759 VIC == TTI::VectorInstrContext::SplatOpFolded &&
2760 ST->sinkSplatOperands() && Index == 0)
2767 if (!LT.second.isVector()) {
2777 auto NumElems = FixedVecTy->getNumElements();
2783 return Opcode == Instruction::ExtractElement
2784 ? StoreCost * NumElems + LoadCost
2785 : (StoreCost + LoadCost) * NumElems + StoreCost;
2789 if (LT.second.isScalableVector() && !LT.first.isValid())
2797 if (Opcode == Instruction::ExtractElement) {
2803 return ExtendCost + ExtractCost;
2813 return ExtendCost + InsertCost + TruncCost;
2820 if (LT.second.isFloatingPoint())
2821 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2825 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2827 getRISCVInstructionCost(MoveOpc, LT.second,
CostKind);
2829 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2834 if (LT.second.isFixedLengthVector()) {
2835 unsigned Width = LT.second.getVectorNumElements();
2836 Index = Index % Width;
2841 if (
auto VLEN = ST->getRealVLen()) {
2842 unsigned EltSize = LT.second.getScalarSizeInBits();
2843 unsigned M1Max = *VLEN / EltSize;
2844 Index = Index % M1Max;
2850 else if (Opcode == Instruction::InsertElement)
2858 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2859 LT.second.isScalableVector()))) {
2861 Align VecAlign =
DL.getPrefTypeAlign(Val);
2862 Align SclAlign =
DL.getPrefTypeAlign(ScalarType);
2867 if (Opcode == Instruction::ExtractElement)
2903 Opcode == Instruction::InsertElement
2904 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2905 RISCV::VSLIDE1DOWN_VX,
2906 RISCV::VSLIDEUP_VX},
2908 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2909 RISCV::VSRL_VX, RISCV::VMV_X_S},
2912 return BaseCost + SlideCost;
2918 unsigned Index)
const {
2927 assert(Index < EC.getKnownMinValue() &&
"Unexpected reverse index");
2929 EC.getKnownMinValue() - 1 - Index,
nullptr,
2938std::optional<InstructionCost>
2944 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2946 if (Opcode == Instruction::UDiv)
2953 return std::nullopt;
2975 if (std::optional<InstructionCost> CombinedCost =
2977 Op2Info, Args, CtxI))
2978 return *CombinedCost;
2982 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2985 if (!LT.second.isVector()) {
2995 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
2996 if (
const auto *Entry =
CostTableLookup(DivTbl, ISDOpcode, LT.second))
2997 return Entry->Cost * LT.first;
3006 if ((LT.second.getVectorElementType() == MVT::f16 ||
3007 LT.second.getVectorElementType() == MVT::bf16) &&
3008 TLI->getOperationAction(ISDOpcode, LT.second) ==
3010 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
3014 CastCost += LT.first * Args.size() *
3022 LT.second = PromotedVT;
3025 auto getConstantMatCost =
3035 return getConstantPoolLoadCost(Ty,
CostKind);
3041 ConstantMatCost += getConstantMatCost(0, Op1Info);
3043 ConstantMatCost += getConstantMatCost(1, Op2Info);
3046 switch (ISDOpcode) {
3049 Op = RISCV::VADD_VV;
3054 Op = RISCV::VSLL_VV;
3059 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
3064 Op = RISCV::VMUL_VV;
3068 Op = RISCV::VDIV_VV;
3072 Op = RISCV::VREM_VV;
3076 Op = RISCV::VFADD_VV;
3079 Op = RISCV::VFMUL_VV;
3082 Op = RISCV::VFDIV_VV;
3085 Op = RISCV::VFSGNJN_VV;
3090 return CastCost + ConstantMatCost +
3099 if (Ty->isFPOrFPVectorTy())
3101 return CastCost + ConstantMatCost + LT.first *
InstrCost;
3124 if (Info.isSameBase() && V !=
Base) {
3125 if (
GEP->hasAllConstantIndices())
3131 unsigned Stride =
DL.getTypeStoreSize(AccessTy);
3132 if (Info.isUnitStride() &&
3138 GEP->getType()->getPointerAddressSpace()))
3141 {TTI::OK_AnyValue, TTI::OP_None},
3142 {TTI::OK_AnyValue, TTI::OP_None}, {});
3159 if (ST->enableDefaultUnroll())
3169 if (L->getHeader()->getParent()->hasOptSize())
3173 L->getExitingBlocks(ExitingBlocks);
3175 <<
"Blocks: " << L->getNumBlocks() <<
"\n"
3176 <<
"Exit blocks: " << ExitingBlocks.
size() <<
"\n");
3180 if (ExitingBlocks.
size() > 2)
3185 if (L->getNumBlocks() > 4)
3193 for (
auto *BB : L->getBlocks()) {
3194 for (
auto &
I : *BB) {
3198 if (IsVectorized && (
I.getType()->isVectorTy() ||
3200 return V->getType()->isVectorTy();
3239 bool HasMask =
false;
3242 bool IsWrite) -> int64_t {
3243 if (
auto *TarExtTy =
3245 return TarExtTy->getIntParameter(0);
3251 case Intrinsic::riscv_vle_mask:
3252 case Intrinsic::riscv_vse_mask:
3253 case Intrinsic::riscv_vlseg2_mask:
3254 case Intrinsic::riscv_vlseg3_mask:
3255 case Intrinsic::riscv_vlseg4_mask:
3256 case Intrinsic::riscv_vlseg5_mask:
3257 case Intrinsic::riscv_vlseg6_mask:
3258 case Intrinsic::riscv_vlseg7_mask:
3259 case Intrinsic::riscv_vlseg8_mask:
3260 case Intrinsic::riscv_vsseg2_mask:
3261 case Intrinsic::riscv_vsseg3_mask:
3262 case Intrinsic::riscv_vsseg4_mask:
3263 case Intrinsic::riscv_vsseg5_mask:
3264 case Intrinsic::riscv_vsseg6_mask:
3265 case Intrinsic::riscv_vsseg7_mask:
3266 case Intrinsic::riscv_vsseg8_mask:
3269 case Intrinsic::riscv_vle:
3270 case Intrinsic::riscv_vse:
3271 case Intrinsic::riscv_vlseg2:
3272 case Intrinsic::riscv_vlseg3:
3273 case Intrinsic::riscv_vlseg4:
3274 case Intrinsic::riscv_vlseg5:
3275 case Intrinsic::riscv_vlseg6:
3276 case Intrinsic::riscv_vlseg7:
3277 case Intrinsic::riscv_vlseg8:
3278 case Intrinsic::riscv_vsseg2:
3279 case Intrinsic::riscv_vsseg3:
3280 case Intrinsic::riscv_vsseg4:
3281 case Intrinsic::riscv_vsseg5:
3282 case Intrinsic::riscv_vsseg6:
3283 case Intrinsic::riscv_vsseg7:
3284 case Intrinsic::riscv_vsseg8: {
3301 Ty = TarExtTy->getTypeParameter(0U);
3306 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3307 unsigned VLIndex = RVVIInfo->VLOperand;
3308 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3316 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3319 unsigned ElemSize = Ty->getScalarSizeInBits();
3323 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3324 Alignment, Mask, EVL);
3327 case Intrinsic::riscv_vlse_mask:
3328 case Intrinsic::riscv_vsse_mask:
3329 case Intrinsic::riscv_vlsseg2_mask:
3330 case Intrinsic::riscv_vlsseg3_mask:
3331 case Intrinsic::riscv_vlsseg4_mask:
3332 case Intrinsic::riscv_vlsseg5_mask:
3333 case Intrinsic::riscv_vlsseg6_mask:
3334 case Intrinsic::riscv_vlsseg7_mask:
3335 case Intrinsic::riscv_vlsseg8_mask:
3336 case Intrinsic::riscv_vssseg2_mask:
3337 case Intrinsic::riscv_vssseg3_mask:
3338 case Intrinsic::riscv_vssseg4_mask:
3339 case Intrinsic::riscv_vssseg5_mask:
3340 case Intrinsic::riscv_vssseg6_mask:
3341 case Intrinsic::riscv_vssseg7_mask:
3342 case Intrinsic::riscv_vssseg8_mask:
3345 case Intrinsic::riscv_vlse:
3346 case Intrinsic::riscv_vsse:
3347 case Intrinsic::riscv_vlsseg2:
3348 case Intrinsic::riscv_vlsseg3:
3349 case Intrinsic::riscv_vlsseg4:
3350 case Intrinsic::riscv_vlsseg5:
3351 case Intrinsic::riscv_vlsseg6:
3352 case Intrinsic::riscv_vlsseg7:
3353 case Intrinsic::riscv_vlsseg8:
3354 case Intrinsic::riscv_vssseg2:
3355 case Intrinsic::riscv_vssseg3:
3356 case Intrinsic::riscv_vssseg4:
3357 case Intrinsic::riscv_vssseg5:
3358 case Intrinsic::riscv_vssseg6:
3359 case Intrinsic::riscv_vssseg7:
3360 case Intrinsic::riscv_vssseg8: {
3377 Ty = TarExtTy->getTypeParameter(0U);
3382 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3383 unsigned VLIndex = RVVIInfo->VLOperand;
3384 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3393 unsigned PointerAlign = Alignment.valueOrOne().value();
3396 Alignment =
Align(1);
3403 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3406 unsigned ElemSize = Ty->getScalarSizeInBits();
3410 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3411 Alignment, Mask, EVL, Stride);
3414 case Intrinsic::riscv_vloxei_mask:
3415 case Intrinsic::riscv_vluxei_mask:
3416 case Intrinsic::riscv_vsoxei_mask:
3417 case Intrinsic::riscv_vsuxei_mask:
3418 case Intrinsic::riscv_vloxseg2_mask:
3419 case Intrinsic::riscv_vloxseg3_mask:
3420 case Intrinsic::riscv_vloxseg4_mask:
3421 case Intrinsic::riscv_vloxseg5_mask:
3422 case Intrinsic::riscv_vloxseg6_mask:
3423 case Intrinsic::riscv_vloxseg7_mask:
3424 case Intrinsic::riscv_vloxseg8_mask:
3425 case Intrinsic::riscv_vluxseg2_mask:
3426 case Intrinsic::riscv_vluxseg3_mask:
3427 case Intrinsic::riscv_vluxseg4_mask:
3428 case Intrinsic::riscv_vluxseg5_mask:
3429 case Intrinsic::riscv_vluxseg6_mask:
3430 case Intrinsic::riscv_vluxseg7_mask:
3431 case Intrinsic::riscv_vluxseg8_mask:
3432 case Intrinsic::riscv_vsoxseg2_mask:
3433 case Intrinsic::riscv_vsoxseg3_mask:
3434 case Intrinsic::riscv_vsoxseg4_mask:
3435 case Intrinsic::riscv_vsoxseg5_mask:
3436 case Intrinsic::riscv_vsoxseg6_mask:
3437 case Intrinsic::riscv_vsoxseg7_mask:
3438 case Intrinsic::riscv_vsoxseg8_mask:
3439 case Intrinsic::riscv_vsuxseg2_mask:
3440 case Intrinsic::riscv_vsuxseg3_mask:
3441 case Intrinsic::riscv_vsuxseg4_mask:
3442 case Intrinsic::riscv_vsuxseg5_mask:
3443 case Intrinsic::riscv_vsuxseg6_mask:
3444 case Intrinsic::riscv_vsuxseg7_mask:
3445 case Intrinsic::riscv_vsuxseg8_mask:
3448 case Intrinsic::riscv_vloxei:
3449 case Intrinsic::riscv_vluxei:
3450 case Intrinsic::riscv_vsoxei:
3451 case Intrinsic::riscv_vsuxei:
3452 case Intrinsic::riscv_vloxseg2:
3453 case Intrinsic::riscv_vloxseg3:
3454 case Intrinsic::riscv_vloxseg4:
3455 case Intrinsic::riscv_vloxseg5:
3456 case Intrinsic::riscv_vloxseg6:
3457 case Intrinsic::riscv_vloxseg7:
3458 case Intrinsic::riscv_vloxseg8:
3459 case Intrinsic::riscv_vluxseg2:
3460 case Intrinsic::riscv_vluxseg3:
3461 case Intrinsic::riscv_vluxseg4:
3462 case Intrinsic::riscv_vluxseg5:
3463 case Intrinsic::riscv_vluxseg6:
3464 case Intrinsic::riscv_vluxseg7:
3465 case Intrinsic::riscv_vluxseg8:
3466 case Intrinsic::riscv_vsoxseg2:
3467 case Intrinsic::riscv_vsoxseg3:
3468 case Intrinsic::riscv_vsoxseg4:
3469 case Intrinsic::riscv_vsoxseg5:
3470 case Intrinsic::riscv_vsoxseg6:
3471 case Intrinsic::riscv_vsoxseg7:
3472 case Intrinsic::riscv_vsoxseg8:
3473 case Intrinsic::riscv_vsuxseg2:
3474 case Intrinsic::riscv_vsuxseg3:
3475 case Intrinsic::riscv_vsuxseg4:
3476 case Intrinsic::riscv_vsuxseg5:
3477 case Intrinsic::riscv_vsuxseg6:
3478 case Intrinsic::riscv_vsuxseg7:
3479 case Intrinsic::riscv_vsuxseg8: {
3496 Ty = TarExtTy->getTypeParameter(0U);
3501 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3502 unsigned VLIndex = RVVIInfo->VLOperand;
3503 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3516 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3519 unsigned ElemSize = Ty->getScalarSizeInBits();
3524 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3525 Align(1), Mask, EVL,
3534 if (Ty->isVectorTy()) {
3537 if ((EltTy->
isHalfTy() && !ST->hasVInstructionsF16()) ||
3543 if (
Size.isScalable() && ST->hasVInstructions())
3546 if (ST->useRVVForFixedLengthVectors())
3566 return std::max<unsigned>(1U, RegWidth.
getFixedValue() / ElemWidth);
3574 return ST->enableUnalignedVectorMem();
3580 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3602 Align Alignment)
const {
3612 if (VTy->getElementType()->isIntegerTy(8)) {
3613 uint64_t MaxEltCount = VTy->getElementCount().getKnownMinValue();
3614 if (VTy->isScalableTy())
3617 if (MaxEltCount > 256)
3626 Align Alignment)
const {
3633 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3636 return TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, ElementTy));
3645 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
3646 bool Considerable =
false;
3647 AllowPromotionWithoutCommonHeader =
false;
3650 Type *ConsideredSExtType =
3652 if (
I.getType() != ConsideredSExtType)
3656 for (
const User *U :
I.users()) {
3658 Considerable =
true;
3662 if (GEPInst->getNumOperands() > 2) {
3663 AllowPromotionWithoutCommonHeader =
true;
3668 return Considerable;
3673 case Instruction::Add:
3674 case Instruction::Sub:
3675 case Instruction::Mul:
3676 case Instruction::And:
3677 case Instruction::Or:
3678 case Instruction::Xor:
3679 case Instruction::FAdd:
3680 case Instruction::FSub:
3681 case Instruction::FMul:
3682 case Instruction::FDiv:
3683 case Instruction::ICmp:
3684 case Instruction::FCmp:
3686 case Instruction::Shl:
3687 case Instruction::LShr:
3688 case Instruction::AShr:
3689 case Instruction::UDiv:
3690 case Instruction::SDiv:
3691 case Instruction::URem:
3692 case Instruction::SRem:
3693 case Instruction::Select:
3694 return Operand == 1;
3701 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3711 switch (
II->getIntrinsicID()) {
3712 case Intrinsic::fma:
3713 case Intrinsic::fmuladd:
3714 return Operand == 0 || Operand == 1;
3715 case Intrinsic::vp_udiv:
3716 case Intrinsic::vp_sdiv:
3717 case Intrinsic::vp_urem:
3718 case Intrinsic::vp_srem:
3719 case Intrinsic::ssub_sat:
3720 case Intrinsic::usub_sat:
3721 return Operand == 1;
3723 case Intrinsic::smin:
3724 case Intrinsic::umin:
3725 case Intrinsic::smax:
3726 case Intrinsic::umax:
3727 case Intrinsic::sadd_sat:
3728 case Intrinsic::uadd_sat:
3729 return Operand == 0 || Operand == 1;
3738 GatherUseOps)
const {
3739 if (Scalars.
empty() || !ST->hasVInstructions() || !ST->sinkSplatOperands() ||
3744 if (SplatIt == Scalars.
end() || (*SplatIt)->getType()->isIntegerTy(1) ||
3750 if (!GatherUseOps(UserOps) || UserOps.
empty())
3769 if (
I->isBitwiseLogicOp()) {
3770 if (!
I->getType()->isVectorTy()) {
3771 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3772 for (
auto &
Op :
I->operands()) {
3780 }
else if (
I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3781 for (
auto &
Op :
I->operands()) {
3793 Ops.push_back(&Not);
3794 Ops.push_back(&InsertElt);
3802 if (!
I->getType()->isVectorTy() || !ST->hasVInstructions())
3810 if (!ST->sinkSplatOperands())
3813 for (
auto OpIdx :
enumerate(
I->operands())) {
3833 for (
Use &U :
Op->uses()) {
3840 Use *InsertEltUse = &
Op->getOperandUse(0);
3843 Ops.push_back(&InsertElt->getOperandUse(1));
3844 Ops.push_back(InsertEltUse);
3845 Ops.push_back(&OpIdx.value());
3854 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3857 Options.AllowOverlappingLoads =
true;
3858 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3860 if (ST->is64Bit()) {
3861 Options.LoadSizes = {8, 4, 2, 1};
3862 Options.AllowedTailExpansions = {3, 5, 6};
3864 Options.LoadSizes = {4, 2, 1};
3865 Options.AllowedTailExpansions = {3};
3868 if (IsZeroCmp && ST->hasVInstructions()) {
3869 unsigned VLenB = ST->getRealMinVLen() / 8;
3872 unsigned MinSize = ST->getXLen() / 8 + 1;
3873 unsigned MaxSize = VLenB * 8;
3887 if (
I->getOpcode() == Instruction::Or &&
3891 if (
I->getOpcode() == Instruction::Add ||
3892 I->getOpcode() == Instruction::Sub)
3910std::optional<Instruction *>
3916 if (
is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3917 II.getIntrinsicID())) {
3920 if (!ST->hasVInstructions())
3923 bool HasAVL =
II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3924 unsigned Offset = HasAVL ? 1 : 0;
3925 unsigned BitWidth =
II.getType()->getIntegerBitWidth();
3950 Value *AVL =
II.getArgOperand(0);
3979 II.getRange().value_or(ConstantRange::getFull(
BitWidth));
3981 if (NewRange != OldRange) {
3982 II.addRangeRetAttr(NewRange);
3992 if (
II.user_empty())
3997 const APInt *Scalar;
4002 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
4006 unsigned TargetEltBW =
DL.getTypeSizeInBits(TargetVecTy->getElementType());
4007 unsigned SourceEltBW =
DL.getTypeSizeInBits(SourceVecTy->getElementType());
4008 if (TargetEltBW % SourceEltBW)
4010 unsigned TargetScale = TargetEltBW / SourceEltBW;
4011 if (VL % TargetScale || TargetScale == 1)
4013 Type *VLTy =
II.getOperand(2)->getType();
4014 ElementCount SourceEC = SourceVecTy->getElementCount();
4015 unsigned NewEltBW = SourceEltBW * TargetScale;
4017 !
DL.fitsInLegalInteger(NewEltBW))
4020 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(
DL, NewEltTy)))
4024 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
4025 "Lossless bitcast between types expected");
4031 RetTy, Intrinsic::riscv_vmv_v_x,
4032 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
4033 ConstantInt::get(VLTy, VL / TargetScale)}),
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
uint64_t IntrinsicInst * II
This file describes how to lower LLVM code to machine code.
Class for arbitrary precision integers.
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
@ ICMP_SLT
signed less than
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ FCMP_ULT
1 1 0 0 True if unordered or less than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Convenience struct for specifying and reasoning about fast-math flags.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
static FixedVectorType * getHalfElementsVectorType(FixedVectorType *VTy)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
user_iterator user_begin()
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
Intrinsic::ID getID() const
bool isTypeBasedOnly() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Represents a single loop in the control flow graph.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector value type where the runtime length is machine dependent.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
ElementCount getVectorElementCount() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getHalfNumVectorElementsVT() const
Return a VT for a vector type with the same element type but half the number of elements.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
Align getAlignment() const
unsigned getAddressSpace() const
Type * getDataType() const
bool getVariableMask() const
const Value * getStrideVal() const
Intrinsic::ID getID() const
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
TargetTransformInfo::VectorInstrContext getBuildVectorContextHint(ArrayRef< int > Mask, ArrayRef< Value * > Scalars, function_ref< bool(SmallVectorImpl< TargetTransformInfo::BuildVectorUseOp > &)> GatherUseOps) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CtxI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
The main scalar evolution driver.
static LLVM_ABI bool isZeroEltSplatMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses all elements with the same value as the first element of exa...
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
bool isVoidTy() const
Return true if this is 'void'.
A Use represents the edge between a Value definition and its users.
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
constexpr ScalarTy getFixedValue() const
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isKnownEven() const
A return value of true indicates we know at compile time that the number of elements (vscale * Min) i...
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
An efficient, type-erasing, non-owning reference to a callable.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
@ None
The instruction is not folded.
@ BinaryOp
One of the operands is a binary op.
@ SplatOpFolded
All of the value's users support splatting the value.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
bool isPairEven(const std::array< std::pair< int, int >, 2 > &SrcInfo, ArrayRef< int > Mask, unsigned &Factor)
Given a shuffle which can be represented as a pair of two slides, see if it is a pair-even idiom.
bool isPairOdd(const std::array< std::pair< int, int >, 2 > &SrcInfo, ArrayRef< int > Mask, unsigned &Factor)
Given a shuffle which can be represented as a pair of two slides, see if it is a pair-odd idiom.
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
auto find_if_not(R &&Range, UnaryPredicate P)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
This struct is a compact representation of a valid (non-zero power of two) alignment.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const