24#include "llvm/IR/IntrinsicsAArch64.h"
36#define DEBUG_TYPE "aarch64tti"
42 "sve-prefer-fixed-over-scalable-if-equal",
cl::Hidden);
60 "Penalty of calling a function that requires a change to PSTATE.SM"));
64 cl::desc(
"Penalty of inlining a call that requires a change to PSTATE.SM"));
75 cl::desc(
"The cost of a histcnt instruction"));
79 cl::desc(
"The number of instructions to search for a redundant dmb"));
83 cl::desc(
"Threshold for forced unrolling of small loops in AArch64"));
86class TailFoldingOption {
101 bool NeedsDefault =
true;
105 void setNeedsDefault(
bool V) { NeedsDefault =
V; }
120 assert((InitialBits == TailFoldingOpts::Disabled || !NeedsDefault) &&
121 "Initial bits should only include one of "
122 "(disabled|all|simple|default)");
123 Bits = NeedsDefault ? DefaultBits : InitialBits;
125 Bits &= ~DisableBits;
131 errs() <<
"invalid argument '" << Opt
132 <<
"' to -sve-tail-folding=; the option should be of the form\n"
133 " (disabled|all|default|simple)[+(reductions|recurrences"
134 "|reverse|noreductions|norecurrences|noreverse)]\n";
140 void operator=(
const std::string &Val) {
149 setNeedsDefault(
false);
152 StringRef(Val).split(TailFoldTypes,
'+', -1,
false);
154 unsigned StartIdx = 1;
155 if (TailFoldTypes[0] ==
"disabled")
156 setInitialBits(TailFoldingOpts::Disabled);
157 else if (TailFoldTypes[0] ==
"all")
158 setInitialBits(TailFoldingOpts::All);
159 else if (TailFoldTypes[0] ==
"default")
160 setNeedsDefault(
true);
161 else if (TailFoldTypes[0] ==
"simple")
162 setInitialBits(TailFoldingOpts::Simple);
165 setInitialBits(TailFoldingOpts::Disabled);
168 for (
unsigned I = StartIdx;
I < TailFoldTypes.
size();
I++) {
169 if (TailFoldTypes[
I] ==
"reductions")
170 setEnableBit(TailFoldingOpts::Reductions);
171 else if (TailFoldTypes[
I] ==
"recurrences")
172 setEnableBit(TailFoldingOpts::Recurrences);
173 else if (TailFoldTypes[
I] ==
"reverse")
174 setEnableBit(TailFoldingOpts::Reverse);
175 else if (TailFoldTypes[
I] ==
"noreductions")
176 setDisableBit(TailFoldingOpts::Reductions);
177 else if (TailFoldTypes[
I] ==
"norecurrences")
178 setDisableBit(TailFoldingOpts::Recurrences);
179 else if (TailFoldTypes[
I] ==
"noreverse")
180 setDisableBit(TailFoldingOpts::Reverse);
197 "Control the use of vectorisation using tail-folding for SVE where the"
198 " option is specified in the form (Initial)[+(Flag1|Flag2|...)]:"
199 "\ndisabled (Initial) No loop types will vectorize using "
201 "\ndefault (Initial) Uses the default tail-folding settings for "
203 "\nall (Initial) All legal loop types will vectorize using "
205 "\nsimple (Initial) Use tail-folding for simple loops (not "
206 "reductions or recurrences)"
207 "\nreductions Use tail-folding for loops containing reductions"
208 "\nnoreductions Inverse of above"
209 "\nrecurrences Use tail-folding for loops containing fixed order "
211 "\nnorecurrences Inverse of above"
212 "\nreverse Use tail-folding for loops requiring reversed "
214 "\nnoreverse Inverse of above"),
259 TTI->isMultiversionedFunction(
F) ?
"fmv-features" :
"target-features";
260 StringRef FeatureStr =
F.getFnAttribute(AttributeStr).getValueAsString();
261 FeatureStr.
split(Features,
",");
277 return F.hasFnAttribute(
"fmv-features");
287 if (
CallAttrs.caller().hasNonStreamingInterfaceAndBody() &&
288 CallAttrs.callee().hasStreamingInterfaceOrBody())
293 if (
CallAttrs.callee().hasStreamingBody()) {
303 CallAttrs.requiresPreservingAllZAState()) {
326 auto FVTy = dyn_cast<FixedVectorType>(Ty);
328 FVTy->getScalarSizeInBits() * FVTy->getNumElements() > 128;
337 unsigned DefaultCallPenalty)
const {
362 if (
F ==
Call.getCaller())
368 return DefaultCallPenalty;
379 ST->isSVEorStreamingSVEAvailable() &&
380 !ST->disableMaximizeScalableBandwidth();
404 assert(Ty->isIntegerTy());
406 unsigned BitSize = Ty->getPrimitiveSizeInBits();
413 ImmVal =
Imm.sext((BitSize + 63) & ~0x3fU);
418 for (
unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
424 return std::max<InstructionCost>(1,
Cost);
431 assert(Ty->isIntegerTy());
433 unsigned BitSize = Ty->getPrimitiveSizeInBits();
439 unsigned ImmIdx = ~0U;
443 case Instruction::GetElementPtr:
448 case Instruction::Store:
451 case Instruction::Add:
452 case Instruction::Sub:
453 case Instruction::Mul:
454 case Instruction::UDiv:
455 case Instruction::SDiv:
456 case Instruction::URem:
457 case Instruction::SRem:
458 case Instruction::And:
459 case Instruction::Or:
460 case Instruction::Xor:
461 case Instruction::ICmp:
465 case Instruction::Shl:
466 case Instruction::LShr:
467 case Instruction::AShr:
471 case Instruction::Trunc:
472 case Instruction::ZExt:
473 case Instruction::SExt:
474 case Instruction::IntToPtr:
475 case Instruction::PtrToInt:
476 case Instruction::BitCast:
477 case Instruction::PHI:
478 case Instruction::Call:
479 case Instruction::Select:
480 case Instruction::Ret:
481 case Instruction::Load:
486 int NumConstants = (BitSize + 63) / 64;
499 assert(Ty->isIntegerTy());
501 unsigned BitSize = Ty->getPrimitiveSizeInBits();
510 if (IID >= Intrinsic::aarch64_addg && IID <= Intrinsic::aarch64_udiv)
516 case Intrinsic::sadd_with_overflow:
517 case Intrinsic::uadd_with_overflow:
518 case Intrinsic::ssub_with_overflow:
519 case Intrinsic::usub_with_overflow:
520 case Intrinsic::smul_with_overflow:
521 case Intrinsic::umul_with_overflow:
523 int NumConstants = (BitSize + 63) / 64;
530 case Intrinsic::experimental_stackmap:
531 if ((Idx < 2) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
534 case Intrinsic::experimental_patchpoint_void:
535 case Intrinsic::experimental_patchpoint:
536 if ((Idx < 4) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
539 case Intrinsic::experimental_gc_statepoint:
540 if ((Idx < 5) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
550 if (TyWidth == 32 || TyWidth == 64)
559 return ST->getMispredictionPenalty();
580 unsigned TotalHistCnts = 1;
590 unsigned EC = VTy->getElementCount().getKnownMinValue();
595 unsigned LegalEltSize = EltSize <= 32 ? 32 : 64;
597 if (EC == 2 || (LegalEltSize == 32 && EC == 4))
601 TotalHistCnts = EC / NaturalVectorWidth;
621 switch (ICA.
getID()) {
622 case Intrinsic::experimental_vector_histogram_add: {
629 case Intrinsic::clmul: {
634 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
638 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8) {
643 -1,
nullptr,
nullptr) *
646 -1,
nullptr,
nullptr);
650 if (LT.second.SimpleTy == MVT::nxv2i64)
651 if (ST->hasSVEAES() && (ST->isSVEAvailable() || ST->hasSSVE_AES()))
654 if (ST->hasSVE2() || ST->hasSME()) {
655 switch (LT.second.SimpleTy) {
670 if (LT.second.SimpleTy == MVT::nxv2i64)
674 switch (LT.second.SimpleTy) {
684 -1,
nullptr,
nullptr) *
687 -1,
nullptr,
nullptr));
696 return LT.first * 11;
698 return LT.first * 14;
705 case Intrinsic::umin:
706 case Intrinsic::umax:
707 case Intrinsic::smin:
708 case Intrinsic::smax: {
709 static const auto ValidMinMaxTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
710 MVT::v8i16, MVT::v2i32, MVT::v4i32,
711 MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32,
718 ICA.
getID() == Intrinsic::smin || ICA.
getID() == Intrinsic::smax;
719 EVT VT = TLI->getValueType(
DL, RetTy,
true);
720 if (VT == MVT::v2i8 || VT == MVT::v2i16 || VT == MVT::v4i8)
721 return LT.first * (IsSigned ? 5 : 3);
723 if (LT.second == MVT::v2i64)
729 case Intrinsic::scmp:
730 case Intrinsic::ucmp: {
732 {Intrinsic::scmp, MVT::i32, 3},
733 {Intrinsic::scmp, MVT::i64, 3},
734 {Intrinsic::scmp, MVT::v8i8, 3},
735 {Intrinsic::scmp, MVT::v16i8, 3},
736 {Intrinsic::scmp, MVT::v4i16, 3},
737 {Intrinsic::scmp, MVT::v8i16, 3},
738 {Intrinsic::scmp, MVT::v2i32, 3},
739 {Intrinsic::scmp, MVT::v4i32, 3},
740 {Intrinsic::scmp, MVT::v1i64, 3},
741 {Intrinsic::scmp, MVT::v2i64, 3},
747 return Entry->Cost * LT.first;
750 case Intrinsic::sadd_sat:
751 case Intrinsic::ssub_sat:
752 case Intrinsic::uadd_sat:
753 case Intrinsic::usub_sat: {
754 static const auto ValidSatTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
755 MVT::v8i16, MVT::v2i32, MVT::v4i32,
761 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits() ? 1 : 4;
763 return LT.first * Instrs;
768 if (ST->isSVEAvailable() && VectorSize >= 128 &&
isPowerOf2_64(VectorSize))
769 return LT.first * Instrs;
773 case Intrinsic::abs: {
774 static const auto ValidAbsTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
775 MVT::v8i16, MVT::v2i32, MVT::v4i32,
776 MVT::v2i64, MVT::nxv16i8, MVT::nxv8i16,
777 MVT::nxv4i32, MVT::nxv2i64};
783 case Intrinsic::bswap: {
784 static const auto ValidAbsTys = {MVT::v4i16, MVT::v8i16, MVT::v2i32,
785 MVT::v4i32, MVT::v2i64};
788 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits())
793 case Intrinsic::fmuladd: {
798 (EltTy->
isHalfTy() && ST->hasFullFP16()))
802 case Intrinsic::stepvector: {
811 Cost += AddCost * (LT.first - 1);
815 case Intrinsic::vector_extract:
816 case Intrinsic::vector_insert: {
829 bool IsExtract = ICA.
getID() == Intrinsic::vector_extract;
830 EVT SubVecVT = IsExtract ? getTLI()->getValueType(
DL, RetTy)
838 getTLI()->getTypeConversion(
C, SubVecVT);
840 getTLI()->getTypeConversion(
C, VecVT);
848 case Intrinsic::bitreverse: {
850 {Intrinsic::bitreverse, MVT::i32, 1},
851 {Intrinsic::bitreverse, MVT::i64, 1},
852 {Intrinsic::bitreverse, MVT::v8i8, 1},
853 {Intrinsic::bitreverse, MVT::v16i8, 1},
854 {Intrinsic::bitreverse, MVT::v4i16, 2},
855 {Intrinsic::bitreverse, MVT::v8i16, 2},
856 {Intrinsic::bitreverse, MVT::v2i32, 2},
857 {Intrinsic::bitreverse, MVT::v4i32, 2},
858 {Intrinsic::bitreverse, MVT::v1i64, 2},
859 {Intrinsic::bitreverse, MVT::v2i64, 2},
867 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8 ||
868 TLI->getValueType(
DL, RetTy,
true) == MVT::i16)
869 return LegalisationCost.first * Entry->Cost + 1;
871 return LegalisationCost.first * Entry->Cost;
875 case Intrinsic::ctpop: {
879 if (ST->hasCSSC() && !RetTy->isVectorTy()) {
882 return LT.first + ExtraCost;
884 if (!ST->hasNEON()) {
914 RetTy->getScalarSizeInBits()
917 return LT.first * Entry->Cost + ExtraCost;
921 case Intrinsic::sadd_with_overflow:
922 case Intrinsic::uadd_with_overflow:
923 case Intrinsic::ssub_with_overflow:
924 case Intrinsic::usub_with_overflow:
925 case Intrinsic::smul_with_overflow:
926 case Intrinsic::umul_with_overflow: {
928 {Intrinsic::sadd_with_overflow, MVT::i8, 3},
929 {Intrinsic::uadd_with_overflow, MVT::i8, 3},
930 {Intrinsic::sadd_with_overflow, MVT::i16, 3},
931 {Intrinsic::uadd_with_overflow, MVT::i16, 3},
932 {Intrinsic::sadd_with_overflow, MVT::i32, 1},
933 {Intrinsic::uadd_with_overflow, MVT::i32, 1},
934 {Intrinsic::sadd_with_overflow, MVT::i64, 1},
935 {Intrinsic::uadd_with_overflow, MVT::i64, 1},
936 {Intrinsic::ssub_with_overflow, MVT::i8, 3},
937 {Intrinsic::usub_with_overflow, MVT::i8, 3},
938 {Intrinsic::ssub_with_overflow, MVT::i16, 3},
939 {Intrinsic::usub_with_overflow, MVT::i16, 3},
940 {Intrinsic::ssub_with_overflow, MVT::i32, 1},
941 {Intrinsic::usub_with_overflow, MVT::i32, 1},
942 {Intrinsic::ssub_with_overflow, MVT::i64, 1},
943 {Intrinsic::usub_with_overflow, MVT::i64, 1},
944 {Intrinsic::smul_with_overflow, MVT::i8, 5},
945 {Intrinsic::umul_with_overflow, MVT::i8, 4},
946 {Intrinsic::smul_with_overflow, MVT::i16, 5},
947 {Intrinsic::umul_with_overflow, MVT::i16, 4},
948 {Intrinsic::smul_with_overflow, MVT::i32, 2},
949 {Intrinsic::umul_with_overflow, MVT::i32, 2},
950 {Intrinsic::smul_with_overflow, MVT::i64, 3},
951 {Intrinsic::umul_with_overflow, MVT::i64, 3},
953 EVT MTy = TLI->getValueType(
DL, RetTy->getContainedType(0),
true);
960 case Intrinsic::fptosi_sat:
961 case Intrinsic::fptoui_sat: {
964 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
966 EVT MTy = TLI->getValueType(
DL, RetTy);
969 if ((LT.second == MVT::f32 || LT.second == MVT::f64 ||
970 LT.second == MVT::v2f32 || LT.second == MVT::v4f32 ||
971 LT.second == MVT::v2f64)) {
973 (LT.second == MVT::f64 && MTy == MVT::i32) ||
974 (LT.second == MVT::f32 && MTy == MVT::i64)))
983 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
990 if ((LT.second == MVT::f16 && MTy == MVT::i32) ||
991 (LT.second == MVT::f16 && MTy == MVT::i64) ||
992 ((LT.second == MVT::v4f16 || LT.second == MVT::v8f16) &&
1006 if ((LT.second.getScalarType() == MVT::f32 ||
1007 LT.second.getScalarType() == MVT::f64 ||
1008 LT.second.getScalarType() == MVT::f16) &&
1011 Type::getIntNTy(RetTy->getContext(), LT.second.getScalarSizeInBits());
1012 if (LT.second.isVector())
1013 LegalTy =
VectorType::get(LegalTy, LT.second.getVectorElementCount());
1017 LegalTy, {LegalTy, LegalTy});
1021 LegalTy, {LegalTy, LegalTy});
1023 return LT.first *
Cost +
1024 ((LT.second.getScalarType() != MVT::f16 || ST->hasFullFP16()) ? 0
1030 RetTy = RetTy->getScalarType();
1031 if (LT.second.isVector()) {
1049 return LT.first *
Cost;
1051 case Intrinsic::fshl:
1052 case Intrinsic::fshr: {
1061 if (RetTy->isIntegerTy() && ICA.
getArgs()[0] == ICA.
getArgs()[1] &&
1062 (RetTy->getPrimitiveSizeInBits() == 32 ||
1063 RetTy->getPrimitiveSizeInBits() == 64)) {
1076 {Intrinsic::fshl, MVT::v4i32, 2},
1077 {Intrinsic::fshl, MVT::v2i64, 2}, {Intrinsic::fshl, MVT::v16i8, 2},
1078 {Intrinsic::fshl, MVT::v8i16, 2}, {Intrinsic::fshl, MVT::v2i32, 2},
1079 {Intrinsic::fshl, MVT::v8i8, 2}, {Intrinsic::fshl, MVT::v4i16, 2}};
1085 return LegalisationCost.first * Entry->Cost;
1089 if (!RetTy->isIntegerTy())
1094 bool HigherCost = (RetTy->getScalarSizeInBits() != 32 &&
1095 RetTy->getScalarSizeInBits() < 64) ||
1096 (RetTy->getScalarSizeInBits() % 64 != 0);
1097 unsigned ExtraCost = HigherCost ? 1 : 0;
1098 if (RetTy->getScalarSizeInBits() == 32 ||
1099 RetTy->getScalarSizeInBits() == 64)
1102 else if (HigherCost)
1106 return TyL.first + ExtraCost;
1108 case Intrinsic::get_active_lane_mask: {
1110 EVT RetVT = getTLI()->getValueType(
DL, RetTy);
1112 if (getTLI()->shouldExpandGetActiveLaneMask(RetVT, OpVT))
1115 if (RetTy->isScalableTy()) {
1116 if (TLI->getTypeAction(RetTy->getContext(), RetVT) !=
1126 if (ST->hasSVE2p1() || ST->hasSME2()) {
1138 Type *CondTy =
OpTy->getWithNewBitWidth(1);
1141 return Cost + (SplitCost * (
Cost - 1));
1156 case Intrinsic::experimental_vector_match: {
1157 if (!ST->hasSVE2() || !ST->isSVEAvailable())
1163 unsigned SearchSize = NeedleTy->getNumElements();
1164 if (SearchSize <= 2)
1169 {MVT::nxv8i16, MVT::nxv16i8, MVT::v8i16, MVT::v16i8, MVT::v8i8},
1173 unsigned ElementSizeInBits = SearchVT.getScalarSizeInBits();
1179 unsigned MatchesRequiredForNeedle =
1191 return Cost * LegalParts * MatchesRequiredForNeedle;
1193 case Intrinsic::cttz: {
1195 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
1196 return LT.first * 2;
1197 if (LT.second == MVT::v4i16 || LT.second == MVT::v8i16 ||
1198 LT.second == MVT::v2i32 || LT.second == MVT::v4i32)
1199 return LT.first * 3;
1202 case Intrinsic::experimental_cttz_elts: {
1212 case Intrinsic::loop_dependence_raw_mask:
1213 case Intrinsic::loop_dependence_war_mask: {
1215 if (ST->hasSVE2() || ST->hasSME()) {
1216 EVT VecVT = getTLI()->getValueType(
DL, RetTy);
1217 unsigned EltSizeInBytes =
1227 case Intrinsic::experimental_vector_extract_last_active:
1228 if (ST->isSVEorStreamingSVEAvailable()) {
1234 case Intrinsic::pow: {
1237 EVT VT = getTLI()->getValueType(
DL, RetTy);
1238 RTLIB::Libcall LC = RTLIB::getPOW(VT);
1239 bool HasLibcall = getTLI()->getLibcallImpl(LC) != RTLIB::Unsupported;
1254 bool Is025 = ExpF->getValueAPF().isExactlyValue(0.25);
1255 bool Is075 = ExpF->getValueAPF().isExactlyValue(0.75);
1265 return (Sqrt * 2) +
FMul;
1276 case Intrinsic::sqrt:
1277 case Intrinsic::fabs:
1278 case Intrinsic::ceil:
1279 case Intrinsic::floor:
1280 case Intrinsic::nearbyint:
1281 case Intrinsic::round:
1282 case Intrinsic::rint:
1283 case Intrinsic::roundeven:
1284 case Intrinsic::trunc:
1285 case Intrinsic::minnum:
1286 case Intrinsic::maxnum:
1287 case Intrinsic::minimum:
1288 case Intrinsic::maximum: {
1306 auto RequiredType =
II.getType();
1309 assert(PN &&
"Expected Phi Node!");
1312 if (!PN->hasOneUse())
1313 return std::nullopt;
1315 for (
Value *IncValPhi : PN->incoming_values()) {
1318 Reinterpret->getIntrinsicID() !=
1319 Intrinsic::aarch64_sve_convert_to_svbool ||
1320 RequiredType != Reinterpret->getArgOperand(0)->getType())
1321 return std::nullopt;
1329 for (
unsigned I = 0;
I < PN->getNumIncomingValues();
I++) {
1331 NPN->
addIncoming(Reinterpret->getOperand(0), PN->getIncomingBlock(
I));
1404 return GoverningPredicateIdx != std::numeric_limits<unsigned>::max();
1409 return GoverningPredicateIdx;
1414 GoverningPredicateIdx = Index;
1436 return UndefIntrinsic;
1441 UndefIntrinsic = IID;
1468 return CmpPredicate;
1473 CmpPredicate = Pred;
1489 return ResultLanes == InactiveLanesTakenFromOperand;
1494 return OperandIdxForInactiveLanes;
1498 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1499 ResultLanes = InactiveLanesTakenFromOperand;
1500 OperandIdxForInactiveLanes = Index;
1505 return ResultLanes == InactiveLanesAreNotDefined;
1509 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1510 ResultLanes = InactiveLanesAreNotDefined;
1515 return ResultLanes == InactiveLanesAreUnused;
1519 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1520 ResultLanes = InactiveLanesAreUnused;
1530 ResultIsZeroInitialized =
true;
1541 return OperandIdxWithNoActiveLanes != std::numeric_limits<unsigned>::max();
1546 return OperandIdxWithNoActiveLanes;
1551 OperandIdxWithNoActiveLanes = Index;
1556 unsigned GoverningPredicateIdx = std::numeric_limits<unsigned>::max();
1559 unsigned IROpcode = 0;
1562 enum PredicationStyle {
1564 InactiveLanesTakenFromOperand,
1565 InactiveLanesAreNotDefined,
1566 InactiveLanesAreUnused
1569 bool ResultIsZeroInitialized =
false;
1570 unsigned OperandIdxForInactiveLanes = std::numeric_limits<unsigned>::max();
1571 unsigned OperandIdxWithNoActiveLanes = std::numeric_limits<unsigned>::max();
1579 return !isa<ScalableVectorType>(V->getType());
1587 case Intrinsic::aarch64_sve_fcvt_bf16f32_v2:
1588 case Intrinsic::aarch64_sve_fcvt_f16f32:
1589 case Intrinsic::aarch64_sve_fcvt_f16f64:
1590 case Intrinsic::aarch64_sve_fcvt_f32f16:
1591 case Intrinsic::aarch64_sve_fcvt_f32f64:
1592 case Intrinsic::aarch64_sve_fcvt_f64f16:
1593 case Intrinsic::aarch64_sve_fcvt_f64f32:
1594 case Intrinsic::aarch64_sve_fcvtlt_f32f16:
1595 case Intrinsic::aarch64_sve_fcvtlt_f64f32:
1596 case Intrinsic::aarch64_sve_fcvtx_f32f64:
1597 case Intrinsic::aarch64_sve_fcvtzs:
1598 case Intrinsic::aarch64_sve_fcvtzs_i32f16:
1599 case Intrinsic::aarch64_sve_fcvtzs_i32f64:
1600 case Intrinsic::aarch64_sve_fcvtzs_i64f16:
1601 case Intrinsic::aarch64_sve_fcvtzs_i64f32:
1602 case Intrinsic::aarch64_sve_fcvtzu:
1603 case Intrinsic::aarch64_sve_fcvtzu_i32f16:
1604 case Intrinsic::aarch64_sve_fcvtzu_i32f64:
1605 case Intrinsic::aarch64_sve_fcvtzu_i64f16:
1606 case Intrinsic::aarch64_sve_fcvtzu_i64f32:
1607 case Intrinsic::aarch64_sve_revb:
1608 case Intrinsic::aarch64_sve_revh:
1609 case Intrinsic::aarch64_sve_revw:
1610 case Intrinsic::aarch64_sve_revd:
1611 case Intrinsic::aarch64_sve_scvtf:
1612 case Intrinsic::aarch64_sve_scvtf_f16i32:
1613 case Intrinsic::aarch64_sve_scvtf_f16i64:
1614 case Intrinsic::aarch64_sve_scvtf_f32i64:
1615 case Intrinsic::aarch64_sve_scvtf_f64i32:
1616 case Intrinsic::aarch64_sve_ucvtf:
1617 case Intrinsic::aarch64_sve_ucvtf_f16i32:
1618 case Intrinsic::aarch64_sve_ucvtf_f16i64:
1619 case Intrinsic::aarch64_sve_ucvtf_f32i64:
1620 case Intrinsic::aarch64_sve_ucvtf_f64i32:
1623 case Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2:
1624 case Intrinsic::aarch64_sve_fcvtnt_f16f32:
1625 case Intrinsic::aarch64_sve_fcvtnt_f32f64:
1626 case Intrinsic::aarch64_sve_fcvtxnt_f32f64:
1629 case Intrinsic::aarch64_sve_fabd:
1631 case Intrinsic::aarch64_sve_fadd:
1634 case Intrinsic::aarch64_sve_fdiv:
1637 case Intrinsic::aarch64_sve_fmax:
1639 case Intrinsic::aarch64_sve_fmaxnm:
1641 case Intrinsic::aarch64_sve_fmin:
1643 case Intrinsic::aarch64_sve_fminnm:
1645 case Intrinsic::aarch64_sve_fmla:
1647 case Intrinsic::aarch64_sve_fmls:
1649 case Intrinsic::aarch64_sve_fmul:
1652 case Intrinsic::aarch64_sve_fmulx:
1654 case Intrinsic::aarch64_sve_fnmla:
1656 case Intrinsic::aarch64_sve_fnmls:
1658 case Intrinsic::aarch64_sve_fsub:
1661 case Intrinsic::aarch64_sve_add:
1664 case Intrinsic::aarch64_sve_mla:
1666 case Intrinsic::aarch64_sve_mls:
1668 case Intrinsic::aarch64_sve_mul:
1671 case Intrinsic::aarch64_sve_sabd:
1673 case Intrinsic::aarch64_sve_sdiv:
1676 case Intrinsic::aarch64_sve_smax:
1678 case Intrinsic::aarch64_sve_smin:
1680 case Intrinsic::aarch64_sve_smulh:
1682 case Intrinsic::aarch64_sve_sub:
1685 case Intrinsic::aarch64_sve_uabd:
1687 case Intrinsic::aarch64_sve_udiv:
1690 case Intrinsic::aarch64_sve_umax:
1692 case Intrinsic::aarch64_sve_umin:
1694 case Intrinsic::aarch64_sve_umulh:
1696 case Intrinsic::aarch64_sve_asr:
1699 case Intrinsic::aarch64_sve_lsl:
1702 case Intrinsic::aarch64_sve_lsr:
1705 case Intrinsic::aarch64_sve_and:
1708 case Intrinsic::aarch64_sve_bic:
1710 case Intrinsic::aarch64_sve_eor:
1713 case Intrinsic::aarch64_sve_orr:
1716 case Intrinsic::aarch64_sve_shsub:
1718 case Intrinsic::aarch64_sve_shsubr:
1720 case Intrinsic::aarch64_sve_sqrshl:
1722 case Intrinsic::aarch64_sve_sqshl:
1724 case Intrinsic::aarch64_sve_sqsub:
1726 case Intrinsic::aarch64_sve_srshl:
1728 case Intrinsic::aarch64_sve_uhsub:
1730 case Intrinsic::aarch64_sve_uhsubr:
1732 case Intrinsic::aarch64_sve_uqrshl:
1734 case Intrinsic::aarch64_sve_uqshl:
1736 case Intrinsic::aarch64_sve_uqsub:
1738 case Intrinsic::aarch64_sve_urshl:
1741 case Intrinsic::aarch64_sve_add_u:
1744 case Intrinsic::aarch64_sve_and_u:
1747 case Intrinsic::aarch64_sve_asr_u:
1750 case Intrinsic::aarch64_sve_eor_u:
1753 case Intrinsic::aarch64_sve_fadd_u:
1756 case Intrinsic::aarch64_sve_fdiv_u:
1759 case Intrinsic::aarch64_sve_fmul_u:
1762 case Intrinsic::aarch64_sve_fsub_u:
1765 case Intrinsic::aarch64_sve_lsl_u:
1768 case Intrinsic::aarch64_sve_lsr_u:
1771 case Intrinsic::aarch64_sve_mul_u:
1774 case Intrinsic::aarch64_sve_orr_u:
1777 case Intrinsic::aarch64_sve_sdiv_u:
1780 case Intrinsic::aarch64_sve_sub_u:
1783 case Intrinsic::aarch64_sve_udiv_u:
1787 case Intrinsic::aarch64_sve_addqv:
1788 case Intrinsic::aarch64_sve_bic_z:
1789 case Intrinsic::aarch64_sve_brka_z:
1790 case Intrinsic::aarch64_sve_brkb_z:
1791 case Intrinsic::aarch64_sve_brkn_z:
1792 case Intrinsic::aarch64_sve_brkpa_z:
1793 case Intrinsic::aarch64_sve_brkpb_z:
1794 case Intrinsic::aarch64_sve_cntp:
1795 case Intrinsic::aarch64_sve_compact:
1796 case Intrinsic::aarch64_sve_eorv:
1797 case Intrinsic::aarch64_sve_eorqv:
1798 case Intrinsic::aarch64_sve_nand_z:
1799 case Intrinsic::aarch64_sve_nor_z:
1800 case Intrinsic::aarch64_sve_orn_z:
1801 case Intrinsic::aarch64_sve_orv:
1802 case Intrinsic::aarch64_sve_orqv:
1803 case Intrinsic::aarch64_sve_pnext:
1804 case Intrinsic::aarch64_sve_rdffr_z:
1805 case Intrinsic::aarch64_sve_saddv:
1806 case Intrinsic::aarch64_sve_uaddv:
1807 case Intrinsic::aarch64_sve_umaxv:
1808 case Intrinsic::aarch64_sve_umaxqv:
1809 case Intrinsic::aarch64_sve_facge:
1810 case Intrinsic::aarch64_sve_facgt:
1811 case Intrinsic::aarch64_sve_ld1:
1812 case Intrinsic::aarch64_sve_ld1_gather:
1813 case Intrinsic::aarch64_sve_ld1_gather_index:
1814 case Intrinsic::aarch64_sve_ld1_gather_scalar_offset:
1815 case Intrinsic::aarch64_sve_ld1_gather_sxtw:
1816 case Intrinsic::aarch64_sve_ld1_gather_sxtw_index:
1817 case Intrinsic::aarch64_sve_ld1_gather_uxtw:
1818 case Intrinsic::aarch64_sve_ld1_gather_uxtw_index:
1819 case Intrinsic::aarch64_sve_ld1q_gather_index:
1820 case Intrinsic::aarch64_sve_ld1q_gather_scalar_offset:
1821 case Intrinsic::aarch64_sve_ld1q_gather_vector_offset:
1822 case Intrinsic::aarch64_sve_ld1ro:
1823 case Intrinsic::aarch64_sve_ld1rq:
1824 case Intrinsic::aarch64_sve_ld1udq:
1825 case Intrinsic::aarch64_sve_ld1uwq:
1826 case Intrinsic::aarch64_sve_ld2_sret:
1827 case Intrinsic::aarch64_sve_ld2q_sret:
1828 case Intrinsic::aarch64_sve_ld3_sret:
1829 case Intrinsic::aarch64_sve_ld3q_sret:
1830 case Intrinsic::aarch64_sve_ld4_sret:
1831 case Intrinsic::aarch64_sve_ld4q_sret:
1832 case Intrinsic::aarch64_sve_ldff1:
1833 case Intrinsic::aarch64_sve_ldff1_gather:
1834 case Intrinsic::aarch64_sve_ldff1_gather_index:
1835 case Intrinsic::aarch64_sve_ldff1_gather_scalar_offset:
1836 case Intrinsic::aarch64_sve_ldff1_gather_sxtw:
1837 case Intrinsic::aarch64_sve_ldff1_gather_sxtw_index:
1838 case Intrinsic::aarch64_sve_ldff1_gather_uxtw:
1839 case Intrinsic::aarch64_sve_ldff1_gather_uxtw_index:
1840 case Intrinsic::aarch64_sve_ldnf1:
1841 case Intrinsic::aarch64_sve_ldnt1:
1842 case Intrinsic::aarch64_sve_ldnt1_gather:
1843 case Intrinsic::aarch64_sve_ldnt1_gather_index:
1844 case Intrinsic::aarch64_sve_ldnt1_gather_scalar_offset:
1845 case Intrinsic::aarch64_sve_ldnt1_gather_uxtw:
1848 case Intrinsic::aarch64_sve_and_z:
1851 case Intrinsic::aarch64_sve_orr_z:
1854 case Intrinsic::aarch64_sve_eor_z:
1858 case Intrinsic::aarch64_sve_cmpeq:
1859 case Intrinsic::aarch64_sve_cmpeq_wide:
1862 case Intrinsic::aarch64_sve_cmpge:
1863 case Intrinsic::aarch64_sve_cmpge_wide:
1866 case Intrinsic::aarch64_sve_cmpgt:
1867 case Intrinsic::aarch64_sve_cmpgt_wide:
1870 case Intrinsic::aarch64_sve_cmphi:
1871 case Intrinsic::aarch64_sve_cmphi_wide:
1874 case Intrinsic::aarch64_sve_cmphs:
1875 case Intrinsic::aarch64_sve_cmphs_wide:
1878 case Intrinsic::aarch64_sve_cmple_wide:
1881 case Intrinsic::aarch64_sve_cmplo_wide:
1884 case Intrinsic::aarch64_sve_cmpls_wide:
1887 case Intrinsic::aarch64_sve_cmplt_wide:
1890 case Intrinsic::aarch64_sve_cmpne:
1891 case Intrinsic::aarch64_sve_cmpne_wide:
1894 case Intrinsic::aarch64_sve_fcmpeq:
1897 case Intrinsic::aarch64_sve_fcmpge:
1900 case Intrinsic::aarch64_sve_fcmpgt:
1903 case Intrinsic::aarch64_sve_fcmpne:
1906 case Intrinsic::aarch64_sve_fcmpuo:
1910 case Intrinsic::aarch64_sve_prf:
1911 case Intrinsic::aarch64_sve_prfb_gather_index:
1912 case Intrinsic::aarch64_sve_prfb_gather_scalar_offset:
1913 case Intrinsic::aarch64_sve_prfb_gather_sxtw_index:
1914 case Intrinsic::aarch64_sve_prfb_gather_uxtw_index:
1915 case Intrinsic::aarch64_sve_prfd_gather_index:
1916 case Intrinsic::aarch64_sve_prfd_gather_scalar_offset:
1917 case Intrinsic::aarch64_sve_prfd_gather_sxtw_index:
1918 case Intrinsic::aarch64_sve_prfd_gather_uxtw_index:
1919 case Intrinsic::aarch64_sve_prfh_gather_index:
1920 case Intrinsic::aarch64_sve_prfh_gather_scalar_offset:
1921 case Intrinsic::aarch64_sve_prfh_gather_sxtw_index:
1922 case Intrinsic::aarch64_sve_prfh_gather_uxtw_index:
1923 case Intrinsic::aarch64_sve_prfw_gather_index:
1924 case Intrinsic::aarch64_sve_prfw_gather_scalar_offset:
1925 case Intrinsic::aarch64_sve_prfw_gather_sxtw_index:
1926 case Intrinsic::aarch64_sve_prfw_gather_uxtw_index:
1929 case Intrinsic::aarch64_sve_st1_scatter:
1930 case Intrinsic::aarch64_sve_st1_scatter_scalar_offset:
1931 case Intrinsic::aarch64_sve_st1_scatter_sxtw:
1932 case Intrinsic::aarch64_sve_st1_scatter_sxtw_index:
1933 case Intrinsic::aarch64_sve_st1_scatter_uxtw:
1934 case Intrinsic::aarch64_sve_st1_scatter_uxtw_index:
1935 case Intrinsic::aarch64_sve_st1dq:
1936 case Intrinsic::aarch64_sve_st1q_scatter_index:
1937 case Intrinsic::aarch64_sve_st1q_scatter_scalar_offset:
1938 case Intrinsic::aarch64_sve_st1q_scatter_vector_offset:
1939 case Intrinsic::aarch64_sve_st1wq:
1940 case Intrinsic::aarch64_sve_stnt1:
1941 case Intrinsic::aarch64_sve_stnt1_scatter:
1942 case Intrinsic::aarch64_sve_stnt1_scatter_index:
1943 case Intrinsic::aarch64_sve_stnt1_scatter_scalar_offset:
1944 case Intrinsic::aarch64_sve_stnt1_scatter_uxtw:
1946 case Intrinsic::aarch64_sve_st2:
1947 case Intrinsic::aarch64_sve_st2q:
1949 case Intrinsic::aarch64_sve_st3:
1950 case Intrinsic::aarch64_sve_st3q:
1952 case Intrinsic::aarch64_sve_st4:
1953 case Intrinsic::aarch64_sve_st4q:
1961 Value *UncastedPred;
1967 Pred = UncastedPred;
1973 if (OrigPredTy->getMinNumElements() <=
1975 ->getMinNumElements())
1976 Pred = UncastedPred;
1980 return C &&
C->isAllOnesValue();
1987 if (Dup && Dup->getIntrinsicID() == Intrinsic::aarch64_sve_dup &&
1988 Dup->getOperand(1) == Pg &&
isa<Constant>(Dup->getOperand(2)))
1996static std::optional<Instruction *>
2003 Value *Op1 =
II.getOperand(1);
2004 Value *Op2 =
II.getOperand(2);
2029 Value *NarrowOp1, *NarrowOp2;
2042 else if (SimpleNarrow == NarrowOp1)
2044 else if (SimpleNarrow == NarrowOp2)
2048 Intrinsic::aarch64_sve_convert_to_svbool, {SimpleNarrow->
getType()},
2058 return std::nullopt;
2069 if (SimpleII == Inactive)
2077static std::optional<Instruction *>
2081 assert((
Opc == Instruction::ICmp ||
Opc == Instruction::FCmp) &&
2082 "Expected a compare operation!");
2089 Opc == Instruction::ICmp &&
LHS->getType() !=
RHS->getType();
2090 assert((IsWideICmp ||
LHS->getType() ==
RHS->getType()) &&
2091 "Unexpected wide compare!");
2107 const APInt *LHSVal, *RHSVal;
2109 return std::nullopt;
2132 return std::nullopt;
2146static std::optional<Instruction *>
2150 return std::nullopt;
2179 II.setCalledFunction(NewDecl);
2185 return std::nullopt;
2196 if (
Opc == Instruction::FCmp ||
Opc == Instruction::ICmp)
2199 return std::nullopt;
2211static std::optional<Instruction *>
2213 auto m_ConvertToSVBool = [](
auto P) {
2217 Intrinsic::aarch64_sve_convert_from_svbool;
2240 return std::nullopt;
2244 case Intrinsic::aarch64_sve_and_z:
2245 case Intrinsic::aarch64_sve_bic_z:
2246 case Intrinsic::aarch64_sve_eor_z:
2247 case Intrinsic::aarch64_sve_nand_z:
2248 case Intrinsic::aarch64_sve_nor_z:
2249 case Intrinsic::aarch64_sve_orn_z:
2250 case Intrinsic::aarch64_sve_orr_z:
2253 return std::nullopt;
2256 Value *BinOpPred = BinOp->getOperand(0);
2257 Value *BinOpOp1 = BinOp->getOperand(1);
2258 Value *BinOpOp2 = BinOp->getOperand(2);
2260 Value *NarrowBinOpPred;
2262 return std::nullopt;
2264 Value *NarrowBinOpOp1 =
2266 Value *NarrowBinOpOp2 = NarrowBinOpOp1;
2267 if (BinOpOp1 != BinOpOp2)
2271 BinOpIID, Ty, {NarrowBinOpPred, NarrowBinOpOp1, NarrowBinOpOp2});
2275static std::optional<Instruction *>
2282 return BinOpCombine;
2287 return std::nullopt;
2290 Value *Cursor =
II.getOperand(0), *EarliestReplacement =
nullptr;
2299 if (CursorVTy->getElementCount().getKnownMinValue() <
2300 IVTy->getElementCount().getKnownMinValue())
2304 if (Cursor->getType() == IVTy)
2305 EarliestReplacement = Cursor;
2310 if (!IntrinsicCursor || !(IntrinsicCursor->getIntrinsicID() ==
2311 Intrinsic::aarch64_sve_convert_to_svbool ||
2312 IntrinsicCursor->getIntrinsicID() ==
2313 Intrinsic::aarch64_sve_convert_from_svbool))
2316 CandidatesForRemoval.
insert(CandidatesForRemoval.
begin(), IntrinsicCursor);
2317 Cursor = IntrinsicCursor->getOperand(0);
2322 if (!EarliestReplacement)
2323 return std::nullopt;
2331 auto *OpPredicate =
II.getOperand(0);
2348 II.getArgOperand(2));
2354 return std::nullopt;
2358 II.getArgOperand(0),
II.getArgOperand(2),
uint64_t(0));
2367 II.getArgOperand(0));
2376 if (!
II.hasOneUse())
2377 return std::nullopt;
2380 return std::nullopt;
2383 switch (
II.getIntrinsicID()) {
2384 case Intrinsic::aarch64_sve_cmpne:
2385 IID = Intrinsic::aarch64_sve_cmpeq;
2387 case Intrinsic::aarch64_sve_cmpne_wide:
2388 IID = Intrinsic::aarch64_sve_cmpeq_wide;
2390 case Intrinsic::aarch64_sve_cmpeq:
2391 IID = Intrinsic::aarch64_sve_cmpne;
2393 case Intrinsic::aarch64_sve_cmpeq_wide:
2394 IID = Intrinsic::aarch64_sve_cmpne_wide;
2397 return std::nullopt;
2402 IID,
II.getOperand(1)->getType(),
2403 {II.getOperand(0), II.getOperand(1), II.getOperand(2)});
2415 return std::nullopt;
2417 for (
auto *U :
II.users()) {
2420 Type *Ty =
II.getOperand(1)->getType();
2425 Intrinsic::aarch64_sve_umin, Ty,
2426 {
II.getOperand(0),
II.getOperand(1), ConstantInt::get(Ty, 1)});
2432 return std::nullopt;
2446 return std::nullopt;
2451 if (!SplatValue || !SplatValue->isZero())
2452 return std::nullopt;
2457 DupQLane->getIntrinsicID() != Intrinsic::aarch64_sve_dupq_lane)
2458 return std::nullopt;
2462 if (!DupQLaneIdx || !DupQLaneIdx->isZero())
2463 return std::nullopt;
2466 if (!VecIns || VecIns->getIntrinsicID() != Intrinsic::vector_insert)
2467 return std::nullopt;
2472 return std::nullopt;
2475 return std::nullopt;
2479 return std::nullopt;
2483 if (!VecTy || !OutTy || VecTy->getNumElements() != OutTy->getMinNumElements())
2484 return std::nullopt;
2486 unsigned NumElts = VecTy->getNumElements();
2487 unsigned PredicateBits = 0;
2490 for (
unsigned I = 0;
I < NumElts; ++
I) {
2493 return std::nullopt;
2495 PredicateBits |= 1 << (
I * (16 / NumElts));
2499 if (PredicateBits == 0) {
2501 PFalse->takeName(&
II);
2507 for (
unsigned I = 0;
I < 16; ++
I)
2508 if ((PredicateBits & (1 <<
I)) != 0)
2511 unsigned PredSize = Mask & -Mask;
2516 for (
unsigned I = 0;
I < 16;
I += PredSize)
2517 if ((PredicateBits & (1 <<
I)) == 0)
2518 return std::nullopt;
2520 auto *ConvertToSVBool =
2523 auto *ConvertFromSVBool =
2525 II.getType(), ConvertToSVBool);
2533 Value *Pg =
II.getArgOperand(0);
2534 Value *Vec =
II.getArgOperand(1);
2535 auto IntrinsicID =
II.getIntrinsicID();
2536 bool IsAfter = IntrinsicID == Intrinsic::aarch64_sve_lasta;
2548 auto OpC = OldBinOp->getOpcode();
2554 OpC, NewLHS, NewRHS, OldBinOp, OldBinOp->getName(),
II.getIterator());
2560 if (IsAfter &&
C &&
C->isNullValue()) {
2564 Extract->insertBefore(
II.getIterator());
2565 Extract->takeName(&
II);
2571 return std::nullopt;
2573 if (IntrPG->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
2574 return std::nullopt;
2576 const auto PTruePattern =
2582 return std::nullopt;
2584 unsigned Idx = MinNumElts - 1;
2594 if (Idx >= PgVTy->getMinNumElements())
2595 return std::nullopt;
2600 Extract->insertBefore(
II.getIterator());
2601 Extract->takeName(&
II);
2614 Value *Pg =
II.getArgOperand(0);
2616 Value *Vec =
II.getArgOperand(2);
2619 if (!Ty->isIntegerTy())
2620 return std::nullopt;
2625 return std::nullopt;
2642 II.getIntrinsicID(), {FPVec->getType()}, {Pg, FPFallBack, FPVec});
2657static std::optional<Instruction *>
2661 if (
Pattern == AArch64SVEPredPattern::all) {
2670 return MinNumElts && NumElts >= MinNumElts
2672 II, ConstantInt::get(
II.getType(), MinNumElts)))
2676static std::optional<Instruction *>
2679 if (!ST->isStreaming())
2680 return std::nullopt;
2692 Value *PgVal =
II.getArgOperand(0);
2693 Value *OpVal =
II.getArgOperand(1);
2697 if (PgVal == OpVal &&
2698 (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_first ||
2699 II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_last)) {
2714 return std::nullopt;
2718 if (Pg->
getIntrinsicID() == Intrinsic::aarch64_sve_convert_to_svbool &&
2719 OpIID == Intrinsic::aarch64_sve_convert_to_svbool &&
2733 if ((Pg ==
Op) && (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_any) &&
2734 ((OpIID == Intrinsic::aarch64_sve_brka_z) ||
2735 (OpIID == Intrinsic::aarch64_sve_brkb_z) ||
2736 (OpIID == Intrinsic::aarch64_sve_brkpa_z) ||
2737 (OpIID == Intrinsic::aarch64_sve_brkpb_z) ||
2738 (OpIID == Intrinsic::aarch64_sve_rdffr_z) ||
2739 (OpIID == Intrinsic::aarch64_sve_and_z) ||
2740 (OpIID == Intrinsic::aarch64_sve_bic_z) ||
2741 (OpIID == Intrinsic::aarch64_sve_eor_z) ||
2742 (OpIID == Intrinsic::aarch64_sve_nand_z) ||
2743 (OpIID == Intrinsic::aarch64_sve_nor_z) ||
2744 (OpIID == Intrinsic::aarch64_sve_orn_z) ||
2745 (OpIID == Intrinsic::aarch64_sve_orr_z))) {
2755 return std::nullopt;
2758template <Intrinsic::ID MulOpc, Intrinsic::ID FuseOpc>
2759static std::optional<Instruction *>
2761 bool MergeIntoAddendOp) {
2763 Value *MulOp0, *MulOp1, *AddendOp, *
Mul;
2764 if (MergeIntoAddendOp) {
2765 AddendOp =
II.getOperand(1);
2766 Mul =
II.getOperand(2);
2768 AddendOp =
II.getOperand(2);
2769 Mul =
II.getOperand(1);
2774 return std::nullopt;
2776 if (!
Mul->hasOneUse())
2777 return std::nullopt;
2780 if (
II.getType()->isFPOrFPVectorTy()) {
2785 return std::nullopt;
2787 return std::nullopt;
2792 if (MergeIntoAddendOp)
2802static std::optional<Instruction *>
2804 Value *Pred =
II.getOperand(0);
2805 Value *PtrOp =
II.getOperand(1);
2806 Type *VecTy =
II.getType();
2821static std::optional<Instruction *>
2823 Value *VecOp =
II.getOperand(0);
2824 Value *Pred =
II.getOperand(1);
2825 Value *PtrOp =
II.getOperand(2);
2841 case Intrinsic::aarch64_sve_fmul_u:
2842 return Instruction::BinaryOps::FMul;
2843 case Intrinsic::aarch64_sve_fadd_u:
2844 return Instruction::BinaryOps::FAdd;
2845 case Intrinsic::aarch64_sve_fsub_u:
2846 return Instruction::BinaryOps::FSub;
2848 return Instruction::BinaryOpsEnd;
2852static std::optional<Instruction *>
2855 if (
II.isStrictFP())
2856 return std::nullopt;
2858 auto *OpPredicate =
II.getOperand(0);
2860 if (BinOpCode == Instruction::BinaryOpsEnd ||
2862 return std::nullopt;
2864 BinOpCode,
II.getOperand(1),
II.getOperand(2),
II.getFastMathFlags());
2868static std::optional<Instruction *>
2870 assert(
II.getIntrinsicID() == Intrinsic::aarch64_sve_mla_u &&
2871 "Expected MLA_U intrinsic");
2872 Value *Acc =
II.getArgOperand(1);
2873 Value *MulOp0 =
II.getArgOperand(2);
2874 Value *MulOp1 =
II.getArgOperand(3);
2889 II.setArgOperand(2, MulOp1);
2890 II.setArgOperand(3, MulOp0);
2894 return std::nullopt;
2897static std::optional<Instruction *>
2899 assert((
II.getIntrinsicID() == Intrinsic::aarch64_sve_sadalp ||
2900 II.getIntrinsicID() == Intrinsic::aarch64_sve_uadalp) &&
2901 "Expected SADALP or UADALP intrinsic");
2907 return std::nullopt;
2911 return std::nullopt;
2915 II.getIntrinsicID(), {II.getType()},
2916 {II.getArgOperand(0), Acc, II.getArgOperand(2)});
2926 Intrinsic::aarch64_sve_mla>(
2930 Intrinsic::aarch64_sve_mad>(
2933 return std::nullopt;
2936static std::optional<Instruction *>
2940 Intrinsic::aarch64_sve_fmla>(IC,
II,
2945 Intrinsic::aarch64_sve_fmad>(IC,
II,
2950 Intrinsic::aarch64_sve_fmla>(IC,
II,
2953 return std::nullopt;
2956static std::optional<Instruction *>
2960 Intrinsic::aarch64_sve_fmla>(IC,
II,
2965 Intrinsic::aarch64_sve_fmad>(IC,
II,
2970 Intrinsic::aarch64_sve_fmla_u>(
2976static std::optional<Instruction *>
2980 Intrinsic::aarch64_sve_fmls>(IC,
II,
2985 Intrinsic::aarch64_sve_fnmsb>(
2990 Intrinsic::aarch64_sve_fmls>(IC,
II,
2993 return std::nullopt;
2996static std::optional<Instruction *>
3000 Intrinsic::aarch64_sve_fmls>(IC,
II,
3005 Intrinsic::aarch64_sve_fnmsb>(
3010 Intrinsic::aarch64_sve_fmls_u>(
3019 Intrinsic::aarch64_sve_mls>(
3022 return std::nullopt;
3027 Value *UnpackArg =
II.getArgOperand(0);
3029 bool IsSigned =
II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpkhi ||
3030 II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpklo;
3043 return std::nullopt;
3047 auto *OpVal =
II.getOperand(0);
3048 auto *OpIndices =
II.getOperand(1);
3055 SplatValue->getValue().uge(VTy->getElementCount().getKnownMinValue()))
3056 return std::nullopt;
3071 Type *RetTy =
II.getType();
3072 constexpr Intrinsic::ID FromSVB = Intrinsic::aarch64_sve_convert_from_svbool;
3073 constexpr Intrinsic::ID ToSVB = Intrinsic::aarch64_sve_convert_to_svbool;
3077 if ((
match(
II.getArgOperand(0),
3084 if (TyA ==
B->getType() &&
3089 TyA->getMinNumElements());
3095 return std::nullopt;
3103 if (
match(
II.getArgOperand(0),
3108 II, (
II.getIntrinsicID() == Intrinsic::aarch64_sve_zip1 ?
A :
B));
3110 return std::nullopt;
3113static std::optional<Instruction *>
3115 Value *Mask =
II.getOperand(0);
3116 Value *BasePtr =
II.getOperand(1);
3117 Value *Index =
II.getOperand(2);
3128 BasePtr->getPointerAlignment(
II.getDataLayout());
3131 BasePtr, IndexBase);
3138 return std::nullopt;
3141static std::optional<Instruction *>
3143 Value *Val =
II.getOperand(0);
3144 Value *Mask =
II.getOperand(1);
3145 Value *BasePtr =
II.getOperand(2);
3146 Value *Index =
II.getOperand(3);
3156 BasePtr->getPointerAlignment(
II.getDataLayout());
3159 BasePtr, IndexBase);
3165 return std::nullopt;
3171 Value *Pred =
II.getOperand(0);
3172 Value *Vec =
II.getOperand(1);
3173 Value *DivVec =
II.getOperand(2);
3177 if (!SplatConstantInt)
3178 return std::nullopt;
3182 if (DivisorValue == -1)
3183 return std::nullopt;
3184 if (DivisorValue == 1)
3190 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3197 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3199 Intrinsic::aarch64_sve_neg, {ASRD->getType()}, {ASRD, Pred, ASRD});
3203 return std::nullopt;
3207 size_t VecSize = Vec.
size();
3212 size_t HalfVecSize = VecSize / 2;
3216 if (*
LHS !=
nullptr && *
RHS !=
nullptr) {
3224 if (*
LHS ==
nullptr && *
RHS !=
nullptr)
3242 return std::nullopt;
3249 Elts[Idx->getValue().getZExtValue()] = InsertElt->getOperand(1);
3250 CurrentInsertElt = InsertElt->getOperand(0);
3256 return std::nullopt;
3260 for (
size_t I = 0;
I < Elts.
size();
I++) {
3261 if (Elts[
I] ==
nullptr)
3266 if (InsertEltChain ==
nullptr)
3267 return std::nullopt;
3273 unsigned PatternWidth = IIScalableTy->getScalarSizeInBits() * Elts.
size();
3274 unsigned PatternElementCount = IIScalableTy->getScalarSizeInBits() *
3275 IIScalableTy->getMinNumElements() /
3280 auto *WideShuffleMaskTy =
3291 auto NarrowBitcast =
3304 return std::nullopt;
3309 Value *Pred =
II.getOperand(0);
3310 Value *Vec =
II.getOperand(1);
3311 Value *Shift =
II.getOperand(2);
3314 Value *AbsPred, *MergedValue;
3320 return std::nullopt;
3328 return std::nullopt;
3333 return std::nullopt;
3336 {
II.getType()}, {Pred, Vec, Shift});
3343 Value *Vec =
II.getOperand(0);
3348 return std::nullopt;
3354 auto *NI =
II.getNextNode();
3357 return !
I->mayReadOrWriteMemory() && !
I->mayHaveSideEffects();
3359 while (LookaheadThreshold-- && CanSkipOver(NI)) {
3360 auto *NIBB = NI->getParent();
3361 NI = NI->getNextNode();
3363 if (
auto *SuccBB = NIBB->getUniqueSuccessor())
3364 NI = &*SuccBB->getFirstNonPHIOrDbgOrLifetime();
3370 if (NextII &&
II.isIdenticalTo(NextII))
3373 return std::nullopt;
3381 {II.getType(), II.getOperand(0)->getType()},
3382 {II.getOperand(0), II.getOperand(1)}));
3389 if (PredPattern == AArch64SVEPredPattern::all ||
3390 PredPattern == AArch64SVEPredPattern::pow2)
3392 return std::nullopt;
3398 Value *Passthru =
II.getOperand(0);
3406 auto *Mask = ConstantInt::get(Ty, MaskValue);
3412 return std::nullopt;
3415static std::optional<Instruction *>
3422 return std::nullopt;
3428 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3438 UMinID,
II.getType(), {Pg, NewUMin, ConstantInt::get(II.getType(), 1)});
3448 return std::nullopt;
3454 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3462 return std::nullopt;
3465 II.getType(), {Pg, A, B});
3467 UMinID,
II.getType(), {Pg, NewOrr, ConstantInt::get(II.getType(), 1)});
3476 constexpr Intrinsic::ID CmphsID = Intrinsic::aarch64_sve_cmphs;
3481 Value *
A, *PgLHS, *PgRHS;
3487 !
LHS->hasOneUser() || !
RHS->hasOneUser())
3488 return std::nullopt;
3491 if (ConstB > ConstA)
3497 if (PgLHS != PgRHS || (Pg !=
LHS && Pg !=
RHS && Pg != PgLHS))
3498 return std::nullopt;
3500 Type *VecTy =
A->getType();
3504 Constant *Limit = ConstantInt::get(VecTy, ConstA - ConstB);
3511std::optional<Instruction *>
3522 case Intrinsic::aarch64_dmb:
3524 case Intrinsic::aarch64_neon_fmaxnm:
3525 case Intrinsic::aarch64_neon_fminnm:
3527 case Intrinsic::aarch64_sve_convert_from_svbool:
3529 case Intrinsic::aarch64_sve_dup:
3531 case Intrinsic::aarch64_sve_dup_x:
3533 case Intrinsic::aarch64_sve_cmpeq:
3534 case Intrinsic::aarch64_sve_cmpeq_wide:
3536 case Intrinsic::aarch64_sve_cmpne:
3537 case Intrinsic::aarch64_sve_cmpne_wide:
3539 case Intrinsic::aarch64_sve_rdffr:
3541 case Intrinsic::aarch64_sve_lasta:
3542 case Intrinsic::aarch64_sve_lastb:
3544 case Intrinsic::aarch64_sve_clasta_n:
3545 case Intrinsic::aarch64_sve_clastb_n:
3547 case Intrinsic::aarch64_sve_cntd:
3549 case Intrinsic::aarch64_sve_cntw:
3551 case Intrinsic::aarch64_sve_cnth:
3553 case Intrinsic::aarch64_sve_cntb:
3555 case Intrinsic::aarch64_sme_cntsd:
3557 case Intrinsic::aarch64_sve_ptest_any:
3558 case Intrinsic::aarch64_sve_ptest_first:
3559 case Intrinsic::aarch64_sve_ptest_last:
3561 case Intrinsic::aarch64_sve_fadd:
3563 case Intrinsic::aarch64_sve_fadd_u:
3565 case Intrinsic::aarch64_sve_fmul_u:
3567 case Intrinsic::aarch64_sve_fsub:
3569 case Intrinsic::aarch64_sve_fsub_u:
3571 case Intrinsic::aarch64_sve_add:
3573 case Intrinsic::aarch64_sve_add_u:
3575 Intrinsic::aarch64_sve_mla_u>(
3577 case Intrinsic::aarch64_sve_mla_u:
3579 case Intrinsic::aarch64_sve_sadalp:
3580 case Intrinsic::aarch64_sve_uadalp:
3582 case Intrinsic::aarch64_sve_sub:
3584 case Intrinsic::aarch64_sve_sub_u:
3586 Intrinsic::aarch64_sve_mls_u>(
3588 case Intrinsic::aarch64_sve_tbl:
3590 case Intrinsic::aarch64_sve_uunpkhi:
3591 case Intrinsic::aarch64_sve_uunpklo:
3592 case Intrinsic::aarch64_sve_sunpkhi:
3593 case Intrinsic::aarch64_sve_sunpklo:
3595 case Intrinsic::aarch64_sve_uzp1:
3597 case Intrinsic::aarch64_sve_zip1:
3598 case Intrinsic::aarch64_sve_zip2:
3600 case Intrinsic::aarch64_sve_ld1_gather_index:
3602 case Intrinsic::aarch64_sve_st1_scatter_index:
3604 case Intrinsic::aarch64_sve_ld1:
3606 case Intrinsic::aarch64_sve_st1:
3608 case Intrinsic::aarch64_sve_sdiv:
3610 case Intrinsic::aarch64_sve_sel:
3612 case Intrinsic::aarch64_sve_srshl:
3614 case Intrinsic::aarch64_sve_dupq_lane:
3616 case Intrinsic::aarch64_sve_insr:
3618 case Intrinsic::aarch64_sve_whilelo:
3620 case Intrinsic::aarch64_sve_ptrue:
3622 case Intrinsic::aarch64_sve_uxtb:
3624 case Intrinsic::aarch64_sve_uxth:
3626 case Intrinsic::aarch64_sve_uxtw:
3628 case Intrinsic::aarch64_sme_in_streaming_mode:
3630 case Intrinsic::aarch64_sve_umin_u:
3632 case Intrinsic::aarch64_sve_orr_u:
3634 case Intrinsic::aarch64_sve_and_z:
3638 return std::nullopt;
3645 SimplifyAndSetOp)
const {
3646 switch (
II.getIntrinsicID()) {
3649 case Intrinsic::aarch64_neon_fcvtxn:
3650 case Intrinsic::aarch64_neon_rshrn:
3651 case Intrinsic::aarch64_neon_sqrshrn:
3652 case Intrinsic::aarch64_neon_sqrshrun:
3653 case Intrinsic::aarch64_neon_sqshrn:
3654 case Intrinsic::aarch64_neon_sqshrun:
3655 case Intrinsic::aarch64_neon_sqxtn:
3656 case Intrinsic::aarch64_neon_sqxtun:
3657 case Intrinsic::aarch64_neon_uqrshrn:
3658 case Intrinsic::aarch64_neon_uqshrn:
3659 case Intrinsic::aarch64_neon_uqxtn:
3660 SimplifyAndSetOp(&
II, 0, OrigDemandedElts, UndefElts);
3664 return std::nullopt;
3668 return ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3678 if (ST->useSVEForFixedLengthVectors() &&
3681 std::max(ST->getMinSVEVectorSizeInBits(), 128u));
3682 else if (ST->isNeonAvailable())
3687 if (ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3696bool AArch64TTIImpl::isSingleExtWideningInstruction(
3698 Type *SrcOverrideTy)
const {
3713 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3716 Type *SrcTy = SrcOverrideTy;
3718 case Instruction::Add:
3719 case Instruction::Sub: {
3728 if (Opcode == Instruction::Sub)
3752 assert(SrcTy &&
"Expected some SrcTy");
3754 unsigned SrcElTySize = SrcTyL.second.getScalarSizeInBits();
3760 DstTyL.first * DstTyL.second.getVectorMinNumElements();
3762 SrcTyL.first * SrcTyL.second.getVectorMinNumElements();
3766 return NumDstEls == NumSrcEls && 2 * SrcElTySize == DstEltSize;
3769Type *AArch64TTIImpl::isBinExtWideningInstruction(
unsigned Opcode,
Type *DstTy,
3771 Type *SrcOverrideTy)
const {
3772 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3773 Opcode != Instruction::Mul)
3783 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3786 auto getScalarSizeWithOverride = [&](
const Value *
V) {
3792 ->getScalarSizeInBits();
3795 unsigned MaxEltSize = 0;
3798 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3799 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3800 MaxEltSize = std::max(EltSize0, EltSize1);
3803 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3804 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3807 if (EltSize0 >= DstEltSize / 2 || EltSize1 >= DstEltSize / 2)
3809 MaxEltSize = DstEltSize / 2;
3810 }
else if (Opcode == Instruction::Mul &&
3818 Known.Zero.countLeadingOnes() >
3823 getScalarSizeWithOverride(
isa<ZExtInst>(Args[0]) ? Args[0] : Args[1]);
3827 if (MaxEltSize * 2 > DstEltSize)
3845 if (!Src->isVectorTy() || !TLI->isTypeLegal(TLI->getValueType(
DL, Src)) ||
3846 (Src->isScalableTy() && !ST->hasSVE2()))
3856 if (AddUser && AddUser->getOpcode() == Instruction::Add)
3860 if (!Shr || Shr->getOpcode() != Instruction::LShr)
3864 if (!Trunc || Trunc->getOpcode() != Instruction::Trunc ||
3865 Src->getScalarSizeInBits() !=
3889 int ISD = TLI->InstructionOpcodeToISD(Opcode);
3893 if (
I &&
I->hasOneUser()) {
3896 if (
Type *ExtTy = isBinExtWideningInstruction(
3897 SingleUser->getOpcode(), Dst,
Operands,
3898 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3911 if (isSingleExtWideningInstruction(
3912 SingleUser->getOpcode(), Dst,
Operands,
3913 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3917 if (SingleUser->getOpcode() == Instruction::Add) {
3918 if (
I == SingleUser->getOperand(1) ||
3920 cast<CastInst>(SingleUser->getOperand(1))->getOpcode() == Opcode))
3935 EVT SrcTy = TLI->getValueType(
DL, Src);
3936 EVT DstTy = TLI->getValueType(
DL, Dst);
3944 Instruction::ExtractElement, Src,
CostKind, -1,
nullptr,
nullptr);
3946 Opcode, Dst->getScalarType(), Src->getScalarType(), CCH,
CostKind);
3950 if (!SrcTy.isSimple() || !DstTy.
isSimple())
3955 if (!ST->hasSVE2() && !ST->isStreamingSVEAvailable() &&
3984 EVT WiderTy = SrcTy.
bitsGT(DstTy) ? SrcTy : DstTy;
3987 ST->useSVEForFixedLengthVectors(WiderTy)) {
3988 std::pair<InstructionCost, MVT> LT =
3990 unsigned NumElements =
4006 const unsigned int SVE_EXT_COST = 1;
4007 const unsigned int SVE_FCVT_COST = 1;
4008 const unsigned int SVE_UNPACK_ONCE = 4;
4009 const unsigned int SVE_UNPACK_TWICE = 16;
4138 SVE_EXT_COST + SVE_FCVT_COST},
4143 SVE_EXT_COST + SVE_FCVT_COST},
4150 SVE_EXT_COST + SVE_FCVT_COST},
4154 SVE_EXT_COST + SVE_FCVT_COST},
4160 SVE_EXT_COST + SVE_FCVT_COST},
4163 SVE_EXT_COST + SVE_FCVT_COST},
4168 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4170 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4180 SVE_EXT_COST + SVE_FCVT_COST},
4185 SVE_EXT_COST + SVE_FCVT_COST},
4198 SVE_EXT_COST + SVE_FCVT_COST},
4202 SVE_EXT_COST + SVE_FCVT_COST},
4214 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4216 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4218 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4220 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4224 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4226 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4242 SVE_EXT_COST + SVE_FCVT_COST},
4247 SVE_EXT_COST + SVE_FCVT_COST},
4258 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4260 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4262 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4264 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4266 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4268 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4272 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4274 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4276 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4278 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4503 if (ST->hasFullFP16())
4515 Src->getScalarType(), CCH,
CostKind) +
4523 ST->isSVEorStreamingSVEAvailable() &&
4524 TLI->getTypeAction(Src->getContext(), SrcTy) ==
4526 TLI->getTypeAction(Dst->getContext(), DstTy) ==
4535 Opcode, LegalTy, Src, CCH,
CostKind,
I);
4538 return Part1 + Part2;
4545 ST->isSVEorStreamingSVEAvailable() && TLI->isTypeLegal(DstTy))
4557 assert((Opcode == Instruction::SExt || Opcode == Instruction::ZExt) &&
4570 CostKind, Index,
nullptr,
nullptr);
4574 auto DstVT = TLI->getValueType(
DL, Dst);
4575 auto SrcVT = TLI->getValueType(
DL, Src);
4580 if (!VecLT.second.isVector() || !TLI->isTypeLegal(DstVT))
4586 if (DstVT.getFixedSizeInBits() < SrcVT.getFixedSizeInBits())
4596 case Instruction::SExt:
4601 case Instruction::ZExt:
4602 if (DstVT.getSizeInBits() != 64u || SrcVT.getSizeInBits() == 32u)
4615 return Opcode == Instruction::PHI ? 0 : 1;
4624 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4626 assert(Ty->isVectorTy() &&
"This must be a vector type");
4633 if (!LT.second.isVector())
4638 if (LT.second.isFixedLengthVector()) {
4639 unsigned Width = LT.second.getVectorNumElements();
4640 Index = Index % Width;
4647 if (Index == 0 && !Ty->getScalarType()->isIntegerTy())
4654 if (VIC == TTI::VectorInstrContext::Load) {
4655 if (ST->hasFastLD1Single())
4667 : ST->getVectorInsertExtractBaseCost() + 1;
4691 auto ExtractCanFuseWithFmul = [&]() {
4698 auto IsAllowedScalarTy = [&](
const Type *
T) {
4699 return T->isFloatTy() ||
T->isDoubleTy() ||
4700 (
T->isHalfTy() && ST->hasFullFP16());
4704 auto IsUserFMulScalarTy = [](
const Value *EEUser) {
4707 return BO && BO->getOpcode() == BinaryOperator::FMul &&
4708 !BO->getType()->isVectorTy();
4713 auto IsExtractLaneEquivalentToZero = [&](
unsigned Idx,
unsigned EltSz) {
4717 return Idx == 0 || (RegWidth != 0 && (Idx * EltSz) % RegWidth == 0);
4726 DenseMap<User *, unsigned> UserToExtractIdx;
4727 for (
auto *U :
Scalar->users()) {
4728 if (!IsUserFMulScalarTy(U))
4732 UserToExtractIdx[
U];
4734 if (UserToExtractIdx.
empty())
4736 for (
auto &[S, U, L] : ScalarUserAndIdx) {
4737 for (
auto *U : S->users()) {
4738 if (UserToExtractIdx.
contains(U)) {
4740 auto *Op0 =
FMul->getOperand(0);
4741 auto *Op1 =
FMul->getOperand(1);
4742 if ((Op0 == S && Op1 == S) || Op0 != S || Op1 != S) {
4743 UserToExtractIdx[
U] =
L;
4749 for (
auto &[U, L] : UserToExtractIdx) {
4761 return !EE->users().empty() &&
all_of(EE->users(), [&](
const User *U) {
4762 if (!IsUserFMulScalarTy(U))
4767 const auto *BO = cast<BinaryOperator>(U);
4768 const auto *OtherEE = dyn_cast<ExtractElementInst>(
4769 BO->getOperand(0) == EE ? BO->getOperand(1) : BO->getOperand(0));
4771 const auto *IdxOp = dyn_cast<ConstantInt>(OtherEE->getIndexOperand());
4774 return IsExtractLaneEquivalentToZero(
4775 cast<ConstantInt>(OtherEE->getIndexOperand())
4778 OtherEE->getType()->getScalarSizeInBits());
4786 if (Opcode == Instruction::ExtractElement && (
I || Scalar) &&
4787 ExtractCanFuseWithFmul())
4792 :
ST->getVectorInsertExtractBaseCost();
4801 if (Opcode == Instruction::InsertElement && Index == 0 && Op0 &&
4804 return getVectorInstrCostHelper(Opcode, Ty,
CostKind, Index,
nullptr,
nullptr,
4810 Value *Scalar,
ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4812 return getVectorInstrCostHelper(Opcode, Ty,
CostKind, Index,
nullptr, Scalar,
4813 ScalarUserAndIdx, VIC);
4820 return getVectorInstrCostHelper(
I.getOpcode(), Ty,
CostKind, Index, &
I,
4827 unsigned Index)
const {
4838 : ST->getVectorInsertExtractBaseCost() + 1;
4847 if (Ty->getElementType()->isFloatingPointTy())
4850 unsigned VecInstCost =
4852 return DemandedElts.
popcount() * (Insert + Extract) * VecInstCost;
4859 if (!Ty->getScalarType()->isHalfTy() && !Ty->getScalarType()->isBFloatTy())
4860 return std::nullopt;
4861 if (Ty->getScalarType()->isHalfTy() && ST->hasFullFP16())
4862 return std::nullopt;
4864 if (CanUseSVE && ST->hasSVEB16B16() && ST->isNonStreamingSVEorSME2Available())
4865 return std::nullopt;
4872 Cost += InstCost(PromotedTy);
4894 int ISD = TLI->InstructionOpcodeToISD(Opcode);
4901 Op2Info, Args, CxtI);
4908 Ty,
CostKind, Op1Info, Op2Info,
true,
4911 [&](
Type *PromotedTy) {
4915 return *PromotedCost;
4918 if (Ty->getScalarType()->isFP128Ty())
4926 if (
Type *ExtTy = isBinExtWideningInstruction(Opcode, Ty, Args)) {
4946 ST->hasLimited64bitVectorMulBandwidth())
4949 if (Ty->getScalarSizeInBits() > 64) {
4954 return CostPerLane * CostPerLane * NumLanes * Mul64CostFactor;
4957 if (LT.second == MVT::v2i64) {
4961 return LT.first * Mul64CostFactor;
4982 if (LT.second == MVT::nxv2i64)
4983 return LT.first * Mul64CostFactor;
5042 auto VT = TLI->getValueType(
DL, Ty);
5043 if (VT.isScalarInteger() && VT.getSizeInBits() <= 64) {
5047 : (3 * AsrCost + AddCost);
5049 return MulCost + AsrCost + 2 * AddCost;
5051 }
else if (VT.isVector()) {
5061 if (Ty->isScalableTy() && ST->hasSVE())
5062 Cost += 2 * AsrCost;
5067 ? (LT.second.getScalarType() == MVT::i64 ? 1 : 2) * AsrCost
5071 }
else if (LT.second == MVT::v2i64) {
5072 return VT.getVectorNumElements() *
5079 if (Ty->isScalableTy() && ST->hasSVE())
5080 return MulCost + 2 * AddCost + 2 * AsrCost;
5081 return 2 * MulCost + AddCost + AsrCost + UsraCost;
5086 LT.second.isFixedLengthVector()) {
5096 return ExtractCost + InsertCost +
5104 auto VT = TLI->getValueType(
DL, Ty);
5120 bool HasMULH = VT == MVT::i64 || LT.second == MVT::nxv2i64 ||
5121 LT.second == MVT::nxv4i32 || LT.second == MVT::nxv8i16 ||
5122 LT.second == MVT::nxv16i8;
5123 bool Is128bit = LT.second.is128BitVector();
5135 (HasMULH ? 0 : ShrCost) +
5136 AddCost * 2 + ShrCost;
5137 return DivCost + (
ISD ==
ISD::UREM ? MulCost + AddCost : 0);
5144 if (!VT.isVector() && VT.getSizeInBits() > 64)
5148 Opcode, Ty,
CostKind, Op1Info, Op2Info);
5150 if (TLI->isOperationLegalOrCustom(
ISD, LT.second) && ST->hasSVE()) {
5154 Ty->getPrimitiveSizeInBits().getFixedValue() < 128) {
5164 if (
nullptr != Entry)
5172 FVTy && LT.second.isFixedLengthVector()) {
5173 unsigned NumElts = FVTy->getNumElements();
5174 unsigned RegElts = LT.second.getVectorNumElements();
5176 Cost = (NumElts / RegElts +
popcount(NumElts % RegElts)) * 2;
5180 if (LT.second.getScalarType() == MVT::i8)
5182 else if (LT.second.getScalarType() == MVT::i16)
5194 Opcode, Ty->getScalarType(),
CostKind, Op1Info, Op2Info);
5195 return (4 + DivCost) * VTy->getNumElements();
5201 -1,
nullptr,
nullptr);
5228 LT.second.isFixedLengthVector())
5229 return 2 * LT.first + 1;
5238 if ((Ty->isFloatTy() || Ty->isDoubleTy() ||
5239 (Ty->isHalfTy() && ST->hasFullFP16())) &&
5248 if (!Ty->getScalarType()->isFP128Ty())
5255 if (!Ty->getScalarType()->isFP128Ty())
5256 return 2 * LT.first;
5263 if (!Ty->isVectorTy())
5279 int MaxMergeDistance = 64;
5283 return NumVectorInstToHideOverhead;
5293 unsigned Opcode1,
unsigned Opcode2)
const {
5296 if (!
Sched.hasInstrSchedModel())
5300 Sched.getSchedClassDesc(
TII->get(Opcode1).getSchedClass());
5302 Sched.getSchedClassDesc(
TII->get(Opcode2).getSchedClass());
5308 "Cannot handle variant scheduling classes without an MI");
5324 const int AmortizationCost = 20;
5332 VecPred = CurrentPred;
5340 static const auto ValidMinMaxTys = {
5341 MVT::v8i8, MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v2i32,
5342 MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32, MVT::v2f64};
5343 static const auto ValidFP16MinMaxTys = {MVT::v4f16, MVT::v8f16};
5347 (ST->hasFullFP16() &&
5353 {Instruction::Select, MVT::v2i1, MVT::v2f32, 2},
5354 {Instruction::Select, MVT::v2i1, MVT::v2f64, 2},
5355 {Instruction::Select, MVT::v4i1, MVT::v4f32, 2},
5356 {Instruction::Select, MVT::v4i1, MVT::v4f16, 2},
5357 {Instruction::Select, MVT::v8i1, MVT::v8f16, 2},
5358 {Instruction::Select, MVT::v16i1, MVT::v16i16, 16},
5359 {Instruction::Select, MVT::v8i1, MVT::v8i32, 8},
5360 {Instruction::Select, MVT::v16i1, MVT::v16i32, 16},
5361 {Instruction::Select, MVT::v4i1, MVT::v4i64, 4 * AmortizationCost},
5362 {Instruction::Select, MVT::v8i1, MVT::v8i64, 8 * AmortizationCost},
5363 {Instruction::Select, MVT::v16i1, MVT::v16i64, 16 * AmortizationCost}};
5365 EVT SelCondTy = TLI->getValueType(
DL, CondTy);
5366 EVT SelValTy = TLI->getValueType(
DL, ValTy);
5375 if (Opcode == Instruction::FCmp) {
5377 ValTy,
CostKind, Op1Info, Op2Info,
false,
5379 false, [&](
Type *PromotedTy) {
5391 return *PromotedCost;
5395 if (LT.second.getScalarType() != MVT::f64 &&
5396 LT.second.getScalarType() != MVT::f32 &&
5397 LT.second.getScalarType() != MVT::f16)
5402 unsigned Factor = 1;
5403 if (!CondTy->isVectorTy() &&
5417 AArch64::FCMEQv4f32))
5429 TLI->isTypeLegal(TLI->getValueType(
DL, ValTy)) &&
5448 Op1Info, Op2Info,
I);
5454 if (ST->requiresStrictAlign()) {
5459 Options.AllowOverlappingLoads =
true;
5460 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
5465 Options.LoadSizes = {8, 4, 2, 1};
5466 Options.AllowedTailExpansions = {3, 5, 6};
5471 return ST->hasSVE();
5477 switch (MICA.
getID()) {
5478 case Intrinsic::masked_scatter:
5479 case Intrinsic::masked_gather:
5481 case Intrinsic::masked_load:
5482 case Intrinsic::masked_store:
5483 case Intrinsic::masked_expandload:
5484 case Intrinsic::masked_compressstore:
5498 if (!LT.first.isValid())
5503 if (VT->getElementType()->isIntegerTy(1))
5514 if (MICA.
getID() == Intrinsic::masked_expandload) {
5522 if (MICA.
getID() == Intrinsic::masked_compressstore) {
5543 if (LT.first > 1 && LT.second.getScalarSizeInBits() > 8)
5544 return MemOpCost * 2;
5553 assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
5554 "Should be called on only load or stores.");
5556 case Instruction::Load:
5559 return ST->getGatherOverhead();
5561 case Instruction::Store:
5564 return ST->getScatterOverhead();
5575 unsigned Opcode = (MICA.
getID() == Intrinsic::masked_gather ||
5576 MICA.
getID() == Intrinsic::vp_gather)
5578 : Instruction::Store;
5588 if (!LT.first.isValid())
5592 if (!LT.second.isVector() ||
5594 VT->getElementType()->isIntegerTy(1))
5604 ElementCount LegalVF = LT.second.getVectorElementCount();
5607 {TTI::OK_AnyValue, TTI::OP_None},
I);
5623 EVT VT = TLI->getValueType(
DL, Ty,
true);
5625 if (VT == MVT::Other)
5630 if (!LT.first.isValid())
5640 (VTy->getElementType()->isIntegerTy(1) &&
5641 !VTy->getElementCount().isKnownMultipleOf(
5651 if (Opcode == Instruction::Store)
5655 if (ST->getFixedLoadLatency())
5656 return (LT.first - 1) + ST->getFixedLoadLatency();
5665 if (LT.second.isScalableVector() ||
5666 ST->useSVEForFixedLengthVectors(LT.second)) {
5667 Inst = AArch64::LDR_ZXI;
5668 }
else if (LT.second.isVector() || LT.second.isFloatingPoint()) {
5669 switch (LT.second.getSizeInBits()) {
5671 Inst = AArch64::LDRBui;
5674 Inst = AArch64::LDRHui;
5677 Inst = AArch64::LDRSui;
5680 Inst = AArch64::LDRDui;
5683 Inst = AArch64::LDRQui;
5689 switch (LT.second.getSizeInBits()) {
5691 Inst = AArch64::LDRBBui;
5694 Inst = AArch64::LDRHHui;
5697 Inst = AArch64::LDRWui;
5700 Inst = AArch64::LDRXui;
5708 unsigned SchedClass =
TII->get(Inst).getSchedClass();
5712 float NumLoads = (LT.first - 1).
getValue();
5713 return NumLoads *
Sched.getReciprocalThroughput(*ST, *SCD) +
5714 Sched.computeInstrLatency(*ST, *SCD);
5717 if (ST->isMisaligned128StoreSlow() && Opcode == Instruction::Store &&
5718 LT.second.is128BitVector() && Alignment <
Align(16)) {
5724 const int AmortizationCost = 6;
5726 return LT.first * 2 * AmortizationCost;
5730 if (Ty->isPtrOrPtrVectorTy())
5735 if (Ty->getScalarSizeInBits() != LT.second.getScalarSizeInBits()) {
5737 if (VT == MVT::v4i8)
5744 if (!
isPowerOf2_32(EltSize) || EltSize < 8 || EltSize > 64 ||
5745 Alignment !=
Align(1))
5757 if (Remainder != 0) {
5760 while (!TypeWorklist.
empty()) {
5770 TypeWorklist.
push_back({CurrNumElements - PrevPow2,
Offset + PrevPow2});
5782 bool UseMaskForCond,
bool UseMaskForGaps)
const {
5783 assert(Factor >= 2 &&
"Invalid interleave factor");
5792 if (Factor > TLI->getMaxSupportedInterleaveFactor())
5796 DL.getTypeSizeInBits(VecTy).getKnownMinValue() != (3 * 128))
5802 if (!VecTy->
isScalableTy() && (UseMaskForCond || UseMaskForGaps))
5805 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
5808 EC.divideCoefficientBy(Factor));
5814 if (EC.isKnownMultipleOf(Factor) &&
5815 TLI->isLegalInterleavedAccessType(SubVecTy,
DL, UseScalable))
5816 return Factor * TLI->getNumInterleavedAccesses(SubVecTy,
DL, UseScalable);
5821 if (VecTy->
isScalableTy() && EC.isKnownMultipleOf(Factor)) {
5827 if (UseMaskForCond) {
5828 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
5829 : Intrinsic::masked_store;
5849 if (Opcode == Instruction::Store && Factor == 4 &&
5850 SubVecCost.second.getScalarSizeInBits() ==
5851 (4 * ResultCost.second.getScalarSizeInBits()))
5852 LegalizationCost *= 4;
5854 return MemCost + (Factor * LegalizationCost) + (Factor *
Log2_64(Factor));
5860 UseMaskForCond, UseMaskForGaps);
5867 for (
auto *
I : Tys) {
5868 if (!
I->isVectorTy())
5879 Align Alignment)
const {
5886 return (ST->isSVEAvailable() && ST->hasSVE2p2()) ||
5887 (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
5892 bool HasUnorderedReductions)
const {
5895 return ST->getMaxInterleaveFactor();
5905 enum { MaxStridedLoads = 7 };
5907 int StridedLoads = 0;
5910 for (
const auto BB : L->blocks()) {
5911 for (
auto &
I : *BB) {
5917 if (L->isLoopInvariant(PtrValue))
5922 if (!LSCEVAddRec || !LSCEVAddRec->
isAffine())
5931 if (StridedLoads > MaxStridedLoads / 2)
5932 return StridedLoads;
5935 return StridedLoads;
5938 int StridedLoads = countStridedLoads(L, SE);
5940 <<
" strided loads\n");
5956 unsigned *FinalSize) {
5960 for (
auto *BB : L->getBlocks()) {
5961 for (
auto &
I : *BB) {
5967 if (!Cost.isValid())
5971 if (LoopCost > Budget)
5993 if (MaxTC > 0 && MaxTC <= 32)
6004 if (Blocks.
size() != 2)
6026 if (!L->isInnermost() || L->getNumBlocks() > 8)
6030 if (!L->getExitBlock())
6036 bool HasParellelizableReductions =
6037 L->getNumBlocks() == 1 &&
6038 any_of(L->getHeader()->phis(),
6040 return canParallelizeReductionWhenUnrolling(Phi, L, &SE);
6043 if (HasParellelizableReductions &&
6065 if (HasParellelizableReductions) {
6076 if (Header == Latch) {
6079 unsigned Width = 10;
6085 unsigned MaxInstsPerLine = 16;
6087 unsigned BestUC = 1;
6088 unsigned SizeWithBestUC = BestUC *
Size;
6090 unsigned SizeWithUC = UC *
Size;
6091 if (SizeWithUC > 48)
6093 if ((SizeWithUC % MaxInstsPerLine) == 0 ||
6094 (SizeWithBestUC % MaxInstsPerLine) < (SizeWithUC % MaxInstsPerLine)) {
6096 SizeWithBestUC = BestUC *
Size;
6106 for (
auto *BB : L->blocks()) {
6107 for (
auto &
I : *BB) {
6117 for (
auto *U :
I.users())
6119 LoadedValuesPlus.
insert(U);
6126 return LoadedValuesPlus.
contains(
SI->getOperand(0));
6152 auto *I = dyn_cast<Instruction>(V);
6153 return I && DependsOnLoopLoad(I, Depth + 1);
6160 DependsOnLoopLoad(
I, 0)) {
6192 if (L->getLoopDepth() > 1)
6203 for (
auto *BB : L->getBlocks()) {
6204 for (
auto &
I : *BB) {
6208 if (IsVectorized &&
I.getType()->isVectorTy())
6225 if (ST->isAppleMLike())
6227 else if (ST->getProcFamily() == AArch64Subtarget::Falkor &&
6249 !ST->getSchedModel().isOutOfOrder()) {
6272 bool CanCreate)
const {
6276 case Intrinsic::aarch64_neon_st1x2:
6277 case Intrinsic::aarch64_neon_st1x3:
6278 case Intrinsic::aarch64_neon_st1x4:
6279 case Intrinsic::aarch64_neon_st2:
6280 case Intrinsic::aarch64_neon_st3:
6281 case Intrinsic::aarch64_neon_st4: {
6284 if (!CanCreate || !ST)
6286 unsigned NumElts = Inst->
arg_size() - 1;
6287 if (ST->getNumElements() != NumElts)
6289 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6295 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6297 Res = Builder.CreateInsertValue(Res, L, i);
6301 case Intrinsic::aarch64_neon_ld1x2:
6302 case Intrinsic::aarch64_neon_ld1x3:
6303 case Intrinsic::aarch64_neon_ld1x4:
6304 case Intrinsic::aarch64_neon_ld2:
6305 case Intrinsic::aarch64_neon_ld3:
6306 case Intrinsic::aarch64_neon_ld4:
6307 if (Inst->
getType() == ExpectedType)
6318 case Intrinsic::aarch64_neon_ld1x2:
6319 case Intrinsic::aarch64_neon_ld1x3:
6320 case Intrinsic::aarch64_neon_ld1x4:
6321 case Intrinsic::aarch64_neon_ld2:
6322 case Intrinsic::aarch64_neon_ld3:
6323 case Intrinsic::aarch64_neon_ld4:
6324 Info.ReadMem =
true;
6325 Info.WriteMem =
false;
6328 case Intrinsic::aarch64_neon_st1x2:
6329 case Intrinsic::aarch64_neon_st1x3:
6330 case Intrinsic::aarch64_neon_st1x4:
6331 case Intrinsic::aarch64_neon_st2:
6332 case Intrinsic::aarch64_neon_st3:
6333 case Intrinsic::aarch64_neon_st4:
6334 Info.ReadMem =
false;
6335 Info.WriteMem =
true;
6344 case Intrinsic::aarch64_neon_ld1x2:
6345 case Intrinsic::aarch64_neon_st1x2:
6346 Info.MatchingId = Intrinsic::aarch64_neon_ld1x2;
6348 case Intrinsic::aarch64_neon_ld1x3:
6349 case Intrinsic::aarch64_neon_st1x3:
6350 Info.MatchingId = Intrinsic::aarch64_neon_ld1x3;
6352 case Intrinsic::aarch64_neon_ld1x4:
6353 case Intrinsic::aarch64_neon_st1x4:
6354 Info.MatchingId = Intrinsic::aarch64_neon_ld1x4;
6356 case Intrinsic::aarch64_neon_ld2:
6357 case Intrinsic::aarch64_neon_st2:
6358 Info.MatchingId = Intrinsic::aarch64_neon_ld2;
6360 case Intrinsic::aarch64_neon_ld3:
6361 case Intrinsic::aarch64_neon_st3:
6362 Info.MatchingId = Intrinsic::aarch64_neon_ld3;
6364 case Intrinsic::aarch64_neon_ld4:
6365 case Intrinsic::aarch64_neon_st4:
6366 Info.MatchingId = Intrinsic::aarch64_neon_ld4;
6378 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
6379 bool Considerable =
false;
6380 AllowPromotionWithoutCommonHeader =
false;
6383 Type *ConsideredSExtType =
6385 if (
I.getType() != ConsideredSExtType)
6389 for (
const User *U :
I.users()) {
6391 Considerable =
true;
6395 if (GEPInst->getNumOperands() > 2) {
6396 AllowPromotionWithoutCommonHeader =
true;
6401 return Considerable;
6452 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
6462 return LegalizationCost + 2;
6472 LegalizationCost *= LT.first - 1;
6475 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6484 return LegalizationCost + 2;
6492 std::optional<FastMathFlags> FMF,
6508 return BaseCost + FixedVTy->getNumElements();
6522 MVT MTy = LT.second;
6527 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6575 MTy.
isVector() && (EltTy->isFloatTy() || EltTy->isDoubleTy() ||
6576 (EltTy->isHalfTy() && ST->hasFullFP16()))) {
6588 return (LT.first - 1) +
Log2_32(NElts);
6593 return (LT.first - 1) + Entry->Cost;
6605 if (LT.first != 1) {
6611 ExtraCost *= LT.first - 1;
6614 auto Cost = ValVTy->getElementType()->isIntegerTy(1) ? 2 : Entry->Cost;
6615 return Cost + ExtraCost;
6623 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *VecTy,
6625 EVT VecVT = TLI->getValueType(
DL, VecTy);
6626 EVT ResVT = TLI->getValueType(
DL, ResTy);
6636 if (((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6638 ((LT.second == MVT::v4i16 || LT.second == MVT::v8i16) &&
6640 ((LT.second == MVT::v2i32 || LT.second == MVT::v4i32) &&
6642 return (LT.first - 1) * 2 + 2;
6653 EVT VecVT = TLI->getValueType(
DL, VecTy);
6654 EVT ResVT = TLI->getValueType(
DL, ResTy);
6657 RedOpcode == Instruction::Add) {
6663 if ((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6665 return LT.first + 2;
6700 EVT PromotedVT = LT.second.getScalarType() == MVT::i1
6701 ? TLI->getPromotedVTForPredicate(
EVT(LT.second))
6715 if (LT.second.getScalarType() == MVT::i1) {
6724 assert(Entry &&
"Illegal Type for Splice");
6725 LegalizationCost += Entry->Cost;
6726 return LegalizationCost * LT.first;
6730 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
6739 if ((Opcode != Instruction::Add && Opcode != Instruction::Sub &&
6740 Opcode != Instruction::FAdd && Opcode != Instruction::FSub))
6746 assert(FMF &&
"Missing FastMathFlags for floating-point partial reduction");
6747 if (!FMF->allowReassoc() || !FMF->allowContract())
6751 "FastMathFlags only apply to floating-point partial reductions");
6755 (!BinOp || (OpBExtend !=
TTI::PR_None && InputTypeB)) &&
6756 "Unexpected values for OpBExtend or InputTypeB");
6760 if (BinOp && ((*BinOp != Instruction::Mul && *BinOp != Instruction::FMul) ||
6761 InputTypeA != InputTypeB))
6772 assert(!OpBExtend &&
"Extended second operand without extended first.");
6773 assert(InputTypeA == AccumType &&
"Type mismatch with no extensions.");
6779 bool IsUSDot = OpBExtend !=
TTI::PR_None && OpAExtend != OpBExtend;
6782 if (IsUSDot && !ST->hasMatMulInt8() && !ST->hasDotProd())
6795 auto TC = TLI->getTypeConversion(AccumVectorType->
getContext(),
6804 if (TLI->getTypeAction(AccumVectorType->
getContext(), TC.second) !=
6810 std::pair<InstructionCost, MVT> AccumLT =
6812 std::pair<InstructionCost, MVT> InputLT =
6816 auto IsSupported = [&](
bool SVEPred,
bool NEONPred) ->
bool {
6817 return (ST->isSVEorStreamingSVEAvailable() && SVEPred) ||
6818 (AccumLT.second.isFixedLengthVector() &&
6819 AccumLT.second.getSizeInBits() <= 128 && ST->isNeonAvailable() &&
6823 bool IsSub = Opcode == Instruction::Sub || Opcode == Instruction::FSub;
6831 if (AccumLT.second.getScalarType() == MVT::i32 &&
6832 InputLT.second.getScalarType() == MVT::i8) {
6834 if (!IsUSDot && IsSupported(
true, ST->hasDotProd()))
6835 return Cost + INegCost;
6837 if (IsUSDot && IsSupported(ST->hasMatMulInt8(), ST->hasMatMulInt8()))
6838 return Cost + INegCost;
6843 if (IsUSDot && IsSupported(
false, ST->hasDotProd()))
6844 return Cost * 3 + INegCost;
6847 if (ST->isSVEorStreamingSVEAvailable() && !IsUSDot) {
6849 if (AccumLT.second.getScalarType() == MVT::i64 &&
6850 InputLT.second.getScalarType() == MVT::i16)
6851 return Cost + INegCost;
6854 if (AccumLT.second.getScalarType() == MVT::i32 &&
6855 InputLT.second.getScalarType() == MVT::i16 &&
6856 (ST->hasSVE2p1() || ST->hasSME2()) && !IsSub)
6859 if (AccumLT.second.getScalarType() == MVT::i64 &&
6860 InputLT.second.getScalarType() == MVT::i8)
6866 return Cost + INegCost;
6869 if (AccumLT.second.getScalarType() == MVT::i16 &&
6870 InputLT.second.getScalarType() == MVT::i8 &&
6871 (ST->hasSVE2p3() || ST->hasSME2p3()) && !IsSub)
6877 if (Opcode == Instruction::FAdd && !IsSub &&
6878 IsSupported(ST->hasSME2() || ST->hasSVE2p1(), ST->hasF16F32DOT()) &&
6879 AccumLT.second.getScalarType() == MVT::f32 &&
6880 InputLT.second.getScalarType() == MVT::f16)
6884 if (Ratio == 2 && !IsUSDot) {
6885 MVT InVT = InputLT.second.getScalarType();
6889 if (IsSupported(ST->hasSVE2() || ST->hasSME(),
true) &&
6891 return (BinOp || IsSub) ?
Cost * 2 :
Cost;
6894 if (IsSupported(ST->hasSVE2(), ST->hasFP16FML()) && InVT == MVT::f16)
6898 if (IsSupported(ST->hasSVE2p1() || ST->hasSME2(),
false) &&
6899 InVT == MVT::bf16 && IsSub)
6909 if (IsSupported(ST->hasBF16(), ST->hasBF16()) && InVT == MVT::bf16)
6910 return Cost * 2 + FNegCost;
6914 AccumType, VF, OpAExtend, OpBExtend,
6926 "Expected the Mask to match the return size if given");
6928 "Expected the same scalar types");
6934 LT.second.getScalarSizeInBits() * Mask.size() > 128 &&
6935 SrcTy->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
6936 Mask.size() > LT.second.getVectorNumElements() && !Index && !SubTp) {
6944 return std::max<InstructionCost>(1, LT.first / 4);
6952 Mask, 4, SrcTy->getElementCount().getKnownMinValue() * 2) ||
6954 Mask, 3, SrcTy->getElementCount().getKnownMinValue() * 2)))
6957 unsigned TpNumElts = Mask.size();
6958 unsigned LTNumElts = LT.second.getVectorNumElements();
6959 unsigned NumVecs = (TpNumElts + LTNumElts - 1) / LTNumElts;
6961 LT.second.getVectorElementCount());
6963 std::map<std::tuple<unsigned, unsigned, SmallVector<int>>,
InstructionCost>
6965 for (
unsigned N = 0;
N < NumVecs;
N++) {
6969 unsigned Source1 = -1U, Source2 = -1U;
6970 unsigned NumSources = 0;
6971 for (
unsigned E = 0; E < LTNumElts; E++) {
6972 int MaskElt = (
N * LTNumElts + E < TpNumElts) ? Mask[
N * LTNumElts + E]
6981 unsigned Source = MaskElt / LTNumElts;
6982 if (NumSources == 0) {
6985 }
else if (NumSources == 1 && Source != Source1) {
6988 }
else if (NumSources >= 2 && Source != Source1 && Source != Source2) {
6994 if (Source == Source1)
6996 else if (Source == Source2)
6997 NMask.
push_back(MaskElt % LTNumElts + LTNumElts);
7006 PreviousCosts.insert({std::make_tuple(Source1, Source2, NMask), 0});
7017 NTp, NTp,
CostKind, NMask, 0,
nullptr, Args,
7020 Result.first->second = NCost;
7034 if (IsExtractSubvector && LT.second.isFixedLengthVector()) {
7035 if (LT.second.getFixedSizeInBits() >= 128 &&
7037 LT.second.getVectorNumElements() / 2) {
7040 if (Index == (
int)LT.second.getVectorNumElements() / 2)
7054 if (!Mask.empty() && LT.second.isFixedLengthVector() &&
7057 return M.value() < 0 || M.value() == (int)M.index();
7063 !Mask.empty() && SrcTy->getPrimitiveSizeInBits().isNonZero() &&
7064 SrcTy->getPrimitiveSizeInBits().isKnownMultipleOf(
7073 if ((ST->hasSVE2p1() || ST->hasSME2p1()) &&
7074 ST->isSVEorStreamingSVEAvailable() &&
7079 if (ST->isSVEorStreamingSVEAvailable() &&
7093 if (IsLoad && LT.second.isVector() &&
7095 LT.second.getVectorElementCount()))
7101 if (Mask.size() == 4 &&
7103 (SrcTy->getScalarSizeInBits() == 16 ||
7104 SrcTy->getScalarSizeInBits() == 32) &&
7105 all_of(Mask, [](
int E) {
return E < 8; }))
7111 if (LT.second.isFixedLengthVector() &&
7112 LT.second.getVectorNumElements() == Mask.size() &&
7118 (
isZIPMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7119 isTRNMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7120 isUZPMask(Mask, LT.second.getVectorNumElements(), Unused) ||
7121 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7122 LT.second.getVectorNumElements(), 16) ||
7123 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7124 LT.second.getVectorNumElements(), 32) ||
7125 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7126 LT.second.getVectorNumElements(), 64) ||
7129 [&Mask](
int M) {
return M < 0 || M == Mask[0]; })))
7258 return LT.first * Entry->Cost;
7267 LT.second.getSizeInBits() <= 128 && SubTp) {
7269 if (SubLT.second.isVector()) {
7270 int NumElts = LT.second.getVectorNumElements();
7271 int NumSubElts = SubLT.second.getVectorNumElements();
7272 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
7278 if (IsExtractSubvector)
7299 if (
getPtrStride(*PSE, AccessTy, Ptr, TheLoop, DT, Strides,
7312 return ST->useFixedOverScalableIfEqualCost();
7316 return ST->getEpilogueVectorizationMinVF();
7351 unsigned NumInsns = 0;
7353 NumInsns += BB->size();
7363 int64_t Scale,
unsigned AddrSpace)
const {
7391 if (
I->getOpcode() == Instruction::Or &&
7395 if (
I->getOpcode() == Instruction::Add ||
7396 I->getOpcode() == Instruction::Sub)
7421 return all_equal(Shuf->getShuffleMask());
7428 bool AllowSplat =
false) {
7433 auto areTypesHalfed = [](
Value *FullV,
Value *HalfV) {
7434 auto *FullTy = FullV->
getType();
7435 auto *HalfTy = HalfV->getType();
7437 2 * HalfTy->getPrimitiveSizeInBits().getFixedValue();
7440 auto extractHalf = [](
Value *FullV,
Value *HalfV) {
7443 return FullVT->getNumElements() == 2 * HalfVT->getNumElements();
7447 Value *S1Op1 =
nullptr, *S2Op1 =
nullptr;
7461 if ((S1Op1 && (!areTypesHalfed(S1Op1, Op1) || !extractHalf(S1Op1, Op1))) ||
7462 (S2Op1 && (!areTypesHalfed(S2Op1, Op2) || !extractHalf(S2Op1, Op2))))
7476 if ((M1Start != 0 && M1Start != (NumElements / 2)) ||
7477 (M2Start != 0 && M2Start != (NumElements / 2)))
7479 if (S1Op1 && S2Op1 && M1Start != M2Start)
7489 return Ext->getType()->getScalarSizeInBits() ==
7490 2 * Ext->getOperand(0)->getType()->getScalarSizeInBits();
7504 Value *VectorOperand =
nullptr;
7521 if (!
GEP ||
GEP->getNumOperands() != 2)
7525 Value *Offsets =
GEP->getOperand(1);
7528 if (
Base->getType()->isVectorTy() || !Offsets->getType()->isVectorTy())
7534 if (OffsetsInst->getType()->getScalarSizeInBits() > 32 &&
7535 OffsetsInst->getOperand(0)->getType()->getScalarSizeInBits() <= 32)
7536 Ops.push_back(&
GEP->getOperandUse(1));
7572 switch (
II->getIntrinsicID()) {
7573 case Intrinsic::aarch64_neon_smull:
7574 case Intrinsic::aarch64_neon_umull:
7577 Ops.push_back(&
II->getOperandUse(0));
7578 Ops.push_back(&
II->getOperandUse(1));
7583 case Intrinsic::fma:
7584 case Intrinsic::fmuladd:
7591 Ops.push_back(&
II->getOperandUse(0));
7593 Ops.push_back(&
II->getOperandUse(1));
7596 case Intrinsic::aarch64_neon_sqdmull:
7597 case Intrinsic::aarch64_neon_sqdmulh:
7598 case Intrinsic::aarch64_neon_sqrdmulh:
7601 Ops.push_back(&
II->getOperandUse(0));
7603 Ops.push_back(&
II->getOperandUse(1));
7604 return !
Ops.empty();
7605 case Intrinsic::aarch64_neon_fmlal:
7606 case Intrinsic::aarch64_neon_fmlal2:
7607 case Intrinsic::aarch64_neon_fmlsl:
7608 case Intrinsic::aarch64_neon_fmlsl2:
7611 Ops.push_back(&
II->getOperandUse(1));
7613 Ops.push_back(&
II->getOperandUse(2));
7614 return !
Ops.empty();
7615 case Intrinsic::aarch64_sve_ptest_first:
7616 case Intrinsic::aarch64_sve_ptest_last:
7618 if (IIOp->getIntrinsicID() == Intrinsic::aarch64_sve_ptrue)
7619 Ops.push_back(&
II->getOperandUse(0));
7620 return !
Ops.empty();
7621 case Intrinsic::aarch64_sme_write_horiz:
7622 case Intrinsic::aarch64_sme_write_vert:
7623 case Intrinsic::aarch64_sme_writeq_horiz:
7624 case Intrinsic::aarch64_sme_writeq_vert: {
7626 if (!Idx || Idx->getOpcode() != Instruction::Add)
7628 Ops.push_back(&
II->getOperandUse(1));
7631 case Intrinsic::aarch64_sme_read_horiz:
7632 case Intrinsic::aarch64_sme_read_vert:
7633 case Intrinsic::aarch64_sme_readq_horiz:
7634 case Intrinsic::aarch64_sme_readq_vert:
7635 case Intrinsic::aarch64_sme_ld1b_vert:
7636 case Intrinsic::aarch64_sme_ld1h_vert:
7637 case Intrinsic::aarch64_sme_ld1w_vert:
7638 case Intrinsic::aarch64_sme_ld1d_vert:
7639 case Intrinsic::aarch64_sme_ld1q_vert:
7640 case Intrinsic::aarch64_sme_st1b_vert:
7641 case Intrinsic::aarch64_sme_st1h_vert:
7642 case Intrinsic::aarch64_sme_st1w_vert:
7643 case Intrinsic::aarch64_sme_st1d_vert:
7644 case Intrinsic::aarch64_sme_st1q_vert:
7645 case Intrinsic::aarch64_sme_ld1b_horiz:
7646 case Intrinsic::aarch64_sme_ld1h_horiz:
7647 case Intrinsic::aarch64_sme_ld1w_horiz:
7648 case Intrinsic::aarch64_sme_ld1d_horiz:
7649 case Intrinsic::aarch64_sme_ld1q_horiz:
7650 case Intrinsic::aarch64_sme_st1b_horiz:
7651 case Intrinsic::aarch64_sme_st1h_horiz:
7652 case Intrinsic::aarch64_sme_st1w_horiz:
7653 case Intrinsic::aarch64_sme_st1d_horiz:
7654 case Intrinsic::aarch64_sme_st1q_horiz: {
7656 if (!Idx || Idx->getOpcode() != Instruction::Add)
7658 Ops.push_back(&
II->getOperandUse(3));
7661 case Intrinsic::aarch64_neon_pmull:
7664 Ops.push_back(&
II->getOperandUse(0));
7665 Ops.push_back(&
II->getOperandUse(1));
7667 case Intrinsic::aarch64_neon_pmull64:
7669 II->getArgOperand(1)))
7671 Ops.push_back(&
II->getArgOperandUse(0));
7672 Ops.push_back(&
II->getArgOperandUse(1));
7674 case Intrinsic::masked_gather:
7677 Ops.push_back(&
II->getArgOperandUse(0));
7679 case Intrinsic::masked_scatter:
7682 Ops.push_back(&
II->getArgOperandUse(1));
7689 auto ShouldSinkCondition = [](
Value *
Cond,
7694 if (
II->getIntrinsicID() != Intrinsic::vector_reduce_or ||
7698 Ops.push_back(&
II->getOperandUse(0));
7702 switch (
I->getOpcode()) {
7703 case Instruction::GetElementPtr:
7704 case Instruction::Add:
7705 case Instruction::Sub:
7707 for (
unsigned Op = 0;
Op <
I->getNumOperands(); ++
Op) {
7709 Ops.push_back(&
I->getOperandUse(
Op));
7714 case Instruction::Select: {
7715 if (!ShouldSinkCondition(
I->getOperand(0),
Ops))
7718 Ops.push_back(&
I->getOperandUse(0));
7721 case Instruction::UncondBr:
7723 case Instruction::CondBr: {
7727 Ops.push_back(&
I->getOperandUse(0));
7730 case Instruction::FMul:
7735 Ops.push_back(&
I->getOperandUse(0));
7737 Ops.push_back(&
I->getOperandUse(1));
7747 case Instruction::Xor:
7750 if (
I->getType()->isVectorTy() && ST->isNeonAvailable()) {
7752 ST->isSVEorStreamingSVEAvailable() && (ST->hasSVE2() || ST->hasSME());
7757 case Instruction::And:
7758 case Instruction::Or:
7761 if (
I->getOpcode() == Instruction::Or &&
7766 if (!(
I->getType()->isVectorTy() && ST->hasNEON()) &&
7769 for (
auto &
Op :
I->operands()) {
7781 Ops.push_back(&Not);
7782 Ops.push_back(&InsertElt);
7792 if (!
I->getType()->isVectorTy())
7793 return !
Ops.empty();
7795 switch (
I->getOpcode()) {
7796 case Instruction::Sub:
7797 case Instruction::Add: {
7806 Ops.push_back(&Ext1->getOperandUse(0));
7807 Ops.push_back(&Ext2->getOperandUse(0));
7810 Ops.push_back(&
I->getOperandUse(0));
7811 Ops.push_back(&
I->getOperandUse(1));
7815 case Instruction::Or: {
7818 if (ST->hasNEON()) {
7832 if (
I->getParent() != MainAnd->
getParent() ||
7837 if (
I->getParent() != IA->getParent() ||
7838 I->getParent() != IB->getParent())
7843 Ops.push_back(&
I->getOperandUse(0));
7844 Ops.push_back(&
I->getOperandUse(1));
7853 case Instruction::Mul: {
7854 auto ShouldSinkSplatForIndexedVariant = [](
Value *V) {
7857 if (Ty->isScalableTy())
7861 return Ty->getScalarSizeInBits() == 16 || Ty->getScalarSizeInBits() == 32;
7864 int NumZExts = 0, NumSExts = 0;
7865 for (
auto &
Op :
I->operands()) {
7872 auto *ExtOp = Ext->getOperand(0);
7873 if (
isSplatShuffle(ExtOp) && ShouldSinkSplatForIndexedVariant(ExtOp))
7874 Ops.push_back(&Ext->getOperandUse(0));
7882 if (Ext->getOperand(0)->getType()->getScalarSizeInBits() * 2 <
7883 I->getType()->getScalarSizeInBits())
7920 if (!ElementConstant || !ElementConstant->
isZero())
7923 unsigned Opcode = OperandInstr->
getOpcode();
7924 if (Opcode == Instruction::SExt)
7926 else if (Opcode == Instruction::ZExt)
7931 unsigned Bitwidth =
I->getType()->getScalarSizeInBits();
7941 Ops.push_back(&Insert->getOperandUse(1));
7947 if (!
Ops.empty() && (NumSExts == 2 || NumZExts == 2))
7951 if (!ShouldSinkSplatForIndexedVariant(
I))
7956 Ops.push_back(&
I->getOperandUse(0));
7958 Ops.push_back(&
I->getOperandUse(1));
7960 return !
Ops.empty();
7962 case Instruction::FMul: {
7964 if (
I->getType()->isScalableTy())
7965 return !
Ops.empty();
7969 return !
Ops.empty();
7973 Ops.push_back(&
I->getOperandUse(0));
7975 Ops.push_back(&
I->getOperandUse(1));
7976 return !
Ops.empty();
static bool isAllActivePredicate(const SelectionDAG &DAG, SDValue N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static Error reportError(StringRef Message)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
This file defines the DenseMap class.
static Value * getCondition(Instruction *I)
const HexagonInstrInfo * TII
This file provides the interface for the instcombine pass implementation.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file defines the LoopVectorizationLegality class.
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > & Cond
static uint64_t getBits(uint64_t Val, int Start, int End)
static unsigned getFastMathFlags(const MachineInstr &I, const SPIRVSubtarget &ST)
static SymbolRef::Type getType(const Symbol *Sym)
This file describes how to lower LLVM code to machine code.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
This file implements the C++20 <bit> header.
unsigned getVectorInsertExtractBaseCost() const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
bool isExtPartOfAvgExpr(const Instruction *ExtUser, Type *Dst, Type *Src) const
InstructionCost getIntImmCost(int64_t Val) const
Calculate the cost of materializing a 64-bit value.
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index) const override
std::optional< InstructionCost > getFP16BF16PromoteCost(Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, bool IncludeTrunc, bool CanUseSVE, std::function< InstructionCost(Type *)> InstCost) const
FP16 and BF16 operations are lowered to fptrunc(op(fpext, fpext) if the architecture features are not...
bool prefersVectorizedAddressing() const override
bool preferFixedOverScalableIfEqualCost() const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind=TTI::TCK_RecipThroughput) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
bool isElementTypeLegalForScalableVector(Type *Ty) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
APInt getPriorityMask(const Function &F) const override
bool shouldMaximizeVectorBandwidth(TargetTransformInfo::RegisterKind K) const override
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
bool useNeonVector(const Type *Ty) const
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override
TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const override
unsigned getInlineCallPenalty(const Function *F, const CallBase &Call, unsigned DefaultCallPenalty) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
unsigned getMaxNumElements(ElementCount VF) const
Try to return an estimate cost factor that can be used as a multiplier when scalarizing an operation ...
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
bool isMultiversionedFunction(const Function &F) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool isLegalToVectorizeReduction(const RecurrenceDescriptor &RdxDesc, ElementCount VF) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedGatherScatter(Type *DataType) const
InstructionCost getBranchMispredictPenalty() const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
APInt getFeatureMask(const Function &F) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool areTypesABICompatible(const Function *Caller, const Function *Callee, ArrayRef< Type * > Types) const override
bool enableScalableVectorization() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Value * getOrCreateResultFromMemIntrinsic(IntrinsicInst *Inst, Type *ExpectedType, bool CanCreate=true) const override
bool hasKnownLowerThroughputFromSchedulingModel(unsigned Opcode1, unsigned Opcode2) const
Check whether Opcode1 has less throughput according to the scheduling model than Opcode2.
unsigned getEpilogueVectorizationMinVF() const override
InstructionCost getSpliceCost(VectorType *Tp, int Index, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCostSVE(unsigned Opcode, VectorType *ValTy, TTI::TargetCostKind CostKind) const
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
Return the cost of the scaling factor used in the addressing mode represented by AM for this target,...
bool isLegalMaskedCompressStore(Type *DataType, Align Alignment) const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
Class for arbitrary precision integers.
bool isNegatedPowerOf2() const
Check if this APInt's negated value is a power of two greater than zero.
uint64_t getZExtValue() const
Get zero extended value.
unsigned popcount() const
Count the number of bits set.
void negate()
Negate this APInt in place.
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
unsigned logBase2() const
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
int64_t getSExtValue() const
Get sign extended value.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
LLVM Basic Block Representation.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isTypeLegal(Type *Ty) const override
static BinaryOperator * CreateWithCopiedFlags(BinaryOps Opc, Value *V1, Value *V2, Value *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ ICMP_SGE
signed greater or equal
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
const APInt & getValue() const
Return the constant as an APInt value reference.
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
This is an important base class in LLVM.
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeSizeInBits(Type *Ty) const
Size examples:
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
static constexpr ElementCount getFixed(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
static bool isCommutative(Predicate Pred)
This provides a helper for copying FMF from an instruction or setting specified flags.
Convenience struct for specifying and reasoning about fast-math flags.
bool noSignedZeros() const
bool allowContract() const
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
static bool isCommutative(Predicate P)
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
IntegerType * getIntNTy(unsigned N)
Fetch the type representing an N-bit integer.
Type * getDoubleTy()
Fetch the type representing a 64-bit floating point value.
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
LLVM_ABI CallInst * CreateMaskedLoad(Type *Ty, Value *Ptr, Align Alignment, Value *Mask, Value *PassThru=nullptr, const Twine &Name="")
Create a call to Masked Load intrinsic.
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Type * getHalfTy()
Fetch the type representing a 16-bit floating point value.
Value * CreateGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Value * CreateLogicalAnd(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateBitOrPointerCast(Value *V, Type *DestTy, const Twine &Name="")
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
Value * CreateBinOpFMF(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, FMFSource FMFSource, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
LLVM_ABI CallInst * CreateMaskedStore(Value *Val, Value *Ptr, Align Alignment, Value *Mask)
Create a call to Masked Store intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Type * getFloatTy()
Fetch the type representing a 32-bit floating point value.
Value * CreateIntCast(Value *V, Type *DestTy, bool isSigned, const Twine &Name="")
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateInsertVector(Type *DstType, Value *SrcVec, Value *SubVec, Value *Idx, const Twine &Name="")
Create a call to the vector.insert intrinsic.
LLVM_ABI Value * CreateElementCount(Type *Ty, ElementCount EC)
Create an expression which evaluates to the number of elements in EC at runtime.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
This instruction inserts a single (scalar) element into a VectorType value.
The core instruction combiner logic.
virtual Instruction * eraseInstFromFunction(Instruction &I)=0
Combiner aware instruction erasure.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
user_iterator user_begin()
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
bool hasGroups() const
Returns true if we have any interleave groups.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
const IntrinsicInst * getInst() const
Intrinsic::ID getID() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
Value * getPointerOperand()
iterator_range< block_iterator > blocks() const
RecurrenceSet & getFixedOrderRecurrences()
Return the fixed-order recurrences found in the loop.
DominatorTree * getDominatorTree() const
PredicatedScalarEvolution * getPredicatedScalarEvolution() const
const ReductionList & getReductionVars() const
Returns the reduction variables found in the loop.
Represents a single loop in the control flow graph.
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
bool isFixedLengthVector() const
MVT getVectorElementType() const
Information for memory intrinsic cost model.
Align getAlignment() const
Type * getDataType() const
Intrinsic::ID getID() const
const Instruction * getInst() const
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
Type * getRecurrenceType() const
Returns the type of the recurrence.
RecurKind getRecurrenceKind() const
This node represents a polynomial recurrence on the trip count of the specified loop.
bool isAffine() const
Return true if this represents an expression A + B*x where A and B are loop invariant values.
This class represents an analyzed expression in the program.
SMEAttrs is a utility class to parse the SME ACLE attributes on functions.
bool hasStreamingCompatibleInterface() const
bool hasStreamingInterfaceOrBody() const
bool isSMEABIRoutine() const
SMECallAttrs is a utility class to hold the SMEAttrs for a callsite.
bool requiresSMChange() const
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
static ScalableVectorType * getDoubleElementsVectorType(ScalableVectorType *VTy)
The main scalar evolution driver.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
LLVM_ABI unsigned getSmallConstantTripMultiple(const Loop *L, const SCEV *ExitCount)
Returns the largest constant divisor of the trip count as a normal unsigned value,...
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI unsigned getSmallConstantMaxTripCount(const Loop *L, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Returns the upper bound of the loop trip count as a normal unsigned value.
LLVM_ABI bool isBackedgeTakenCountMaxOrZero(const Loop *L)
Return true if the backedge taken count is either the value returned by getConstantMaxBackedgeTakenCo...
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
const SCEV * getSymbolicMaxBackedgeTakenCount(const Loop *L)
When successful, this returns a SCEV that is greater than or equal to (i.e.
This instruction constructs a fixed permutation of two input vectors.
static LLVM_ABI bool isDeInterleaveMaskOfFactor(ArrayRef< int > Mask, unsigned Factor, unsigned &Index)
Check if the mask is a DE-interleave mask of the given factor Factor like: <Index,...
static LLVM_ABI bool isExtractSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is an extract subvector mask.
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator insert(iterator I, T &&Elt)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
static StackOffset getScalable(int64_t Scalable)
static StackOffset getFixed(int64_t Fixed)
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Class to represent struct types.
TargetInstrInfo - Interface to description of machine instruction set.
std::pair< LegalizeTypeAction, EVT > LegalizeKind
LegalizeKind holds the legalization kind that needs to happen to EVT in order to type-legalize it.
const RTLIB::RuntimeLibcallsInfo & getRuntimeLibcallsInfo() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
bool isPointerTy() const
True if this is an instance of PointerType.
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static VectorType * getInteger(VectorType *VTy)
This static method gets a VectorType with the same number of elements as the input type,...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
LLVM_ABI APInt getCpuSupportsMask(ArrayRef< StringRef > Features)
static constexpr unsigned SVEBitsPerBlock
LLVM_ABI APInt getFMVPriority(ArrayRef< StringRef > Features)
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ CTTZ_ELTS
Returns the number of number of trailing (least significant) zero elements in a vector.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
CheckType m_SpecificType(LLT Ty)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
cst_pred_ty< is_nonnegative > m_NonNegative()
Match an integer or vector of non-negative values.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_VScale()
Matches a call to llvm.vscale().
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
AnyBinaryOp_match< LHS, RHS, true > m_c_BinOp(const LHS &L, const RHS &R)
Matches a BinaryOperator with LHS and RHS in either order.
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
brc_match< Cond_t, match_bind< BasicBlock >, match_bind< BasicBlock > > m_Br(const Cond_t &C, BasicBlock *&T, BasicBlock *&F)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
initializer< Ty > init(const Ty &Val)
LocationClass< Ty > location(Ty &L)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
std::optional< unsigned > isDUPQMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPQMask - matches a splat of equivalent lanes within segments of a given number of elements.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
bool isZIPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for zip1 or zip2 masks of the form: <0, 8, 1, 9, 2, 10, 3, 11> (WhichResultOut = 0,...
TailFoldingOpts
An enum to describe what types of loops we should attempt to tail-fold: Disabled: None Reductions: Lo...
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
bool isDUPFirstSegmentMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPFirstSegmentMask - matches a splat of the first 128b segment.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::optional< const MDOperand * > findStringMetadataForLoop(const Loop *TheLoop, StringRef Name)
Find string metadata for loop.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
LLVM_ABI std::optional< int64_t > getPtrStride(PredicatedScalarEvolution &PSE, Type *AccessTy, Value *Ptr, const Loop *Lp, const DominatorTree &DT, const SymbolicStrideMap &StridesMap=SymbolicStrideMap(), bool ShouldCheckWrap=true, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
If the pointer has a constant stride return it in units of the access type size.
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI bool MaskedValueIsZero(const Value *V, const APInt &Mask, const SimplifyQuery &SQ, unsigned Depth=0)
Return true if 'V & Mask' is known to be zero.
unsigned M1(unsigned Val)
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
unsigned getPerfectShuffleCost(llvm::ArrayRef< int > M)
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
bool isUZPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut)
Return true for uzp1 or uzp2 masks of the form: <0, 2, 4, 6, 8, 10, 12, 14> or <1,...
bool isREVMask(ArrayRef< int > M, unsigned EltSize, unsigned NumElts, unsigned BlockSize)
isREVMask - Check if a vector shuffle corresponds to a REV instruction with the specified blocksize.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ Or
Bitwise or logical OR of integers.
@ FSub
Subtraction of floats.
@ FAddChainWithSubs
A chain of fadds and fsubs.
@ AnyOf
AnyOf reduction with select(cmp(),x,y) where one of (x,y) is loop invariant, and both x and y are int...
@ Xor
Bitwise or logical XOR of integers.
@ FindLast
FindLast reduction with select(cmp(),x,y) where x and y.
@ FMax
FP max implemented in terms of select(cmp()).
@ FMulAdd
Sum of float products with llvm.fmuladd(a * b + sum).
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ FMin
FP min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
DWARFExpression::Operation Op
TypeConversionCostTblEntryT< uint16_t > TypeConversionCostTblEntry
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
unsigned getNumElementsFromSVEPredPattern(unsigned Pattern)
Return the number of active elements for VL1 to VL256 predicate pattern, zero for all other patterns.
auto predecessors(const MachineBasicBlock *BB)
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
LLVM_ABI Value * simplifyCmpInst(CmpPredicate Predicate, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a CmpInst, fold the result or return null.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
const TypeConversionCostTblEntryT< CostType > * ConvertCostTableLookup(ArrayRef< TypeConversionCostTblEntryT< CostType > > Tbl, int ISD, MVT Dst, MVT Src)
Find in type conversion cost table.
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
bool isTRNMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for trn1 or trn2 masks of the form: <0, 8, 2, 10, 4, 12, 6, 14> (WhichResultOut = 0,...
unsigned getMatchingIROpode() const
bool inactiveLanesAreUnused() const
bool inactiveLanesAreNotDefined() const
bool hasMatchingUndefIntrinsic() const
static SVEIntrinsicInfo defaultMergingUnaryNarrowingTopOp()
static SVEIntrinsicInfo defaultZeroingOp()
bool hasGoverningPredicate() const
SVEIntrinsicInfo & setOperandIdxInactiveLanesTakenFrom(unsigned Index)
static SVEIntrinsicInfo defaultMergingOp(Intrinsic::ID IID=Intrinsic::not_intrinsic)
SVEIntrinsicInfo & setOperandIdxWithNoActiveLanes(unsigned Index)
unsigned getOperandIdxWithNoActiveLanes() const
CmpInst::Predicate getCmpPredicate() const
SVEIntrinsicInfo & setInactiveLanesAreUnused()
SVEIntrinsicInfo & setInactiveLanesAreNotDefined()
SVEIntrinsicInfo & setGoverningPredicateOperandIdx(unsigned Index)
bool inactiveLanesTakenFromOperand() const
static SVEIntrinsicInfo defaultUndefOp()
bool hasOperandWithNoActiveLanes() const
Intrinsic::ID getMatchingUndefIntrinsic() const
SVEIntrinsicInfo & setResultIsZeroInitialized()
bool hasCmpPredicate() const
static SVEIntrinsicInfo defaultMergingUnaryOp()
SVEIntrinsicInfo & setMatchingUndefIntrinsic(Intrinsic::ID IID)
unsigned getGoverningPredicateOperandIdx() const
bool hasMatchingIROpode() const
SVEIntrinsicInfo & setCmpPredicate(CmpInst::Predicate Pred)
bool resultIsZeroInitialized() const
SVEIntrinsicInfo & setMatchingIROpcode(unsigned Opcode)
unsigned getOperandIdxInactiveLanesTakenFrom() const
static SVEIntrinsicInfo defaultVoidOp(unsigned GPIndex)
This struct is a compact representation of a valid (non-zero power of two) alignment.
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
bool isFixedLengthVector() const
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Machine model for scheduling, bundling, and heuristics.
static LLVM_ABI double getReciprocalThroughput(const MCSubtargetInfo &STI, const MCSchedClassDesc &SCDesc)
Information about a load/store intrinsic defined by the target.
InterleavedAccessInfo * IAI
LoopVectorizationLegality * LVL
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...