24#include "llvm/IR/IntrinsicsAArch64.h"
36#define DEBUG_TYPE "aarch64tti"
42 "sve-prefer-fixed-over-scalable-if-equal",
cl::Hidden);
60 "Penalty of calling a function that requires a change to PSTATE.SM"));
64 cl::desc(
"Penalty of inlining a call that requires a change to PSTATE.SM"));
75 cl::desc(
"The cost of a histcnt instruction"));
79 cl::desc(
"The number of instructions to search for a redundant dmb"));
83 cl::desc(
"Threshold for forced unrolling of small loops in AArch64"));
86class TailFoldingOption {
101 bool NeedsDefault =
true;
105 void setNeedsDefault(
bool V) { NeedsDefault =
V; }
120 assert((InitialBits == TailFoldingOpts::Disabled || !NeedsDefault) &&
121 "Initial bits should only include one of "
122 "(disabled|all|simple|default)");
123 Bits = NeedsDefault ? DefaultBits : InitialBits;
125 Bits &= ~DisableBits;
131 errs() <<
"invalid argument '" << Opt
132 <<
"' to -sve-tail-folding=; the option should be of the form\n"
133 " (disabled|all|default|simple)[+(reductions|recurrences"
134 "|reverse|noreductions|norecurrences|noreverse)]\n";
140 void operator=(
const std::string &Val) {
149 setNeedsDefault(
false);
152 StringRef(Val).split(TailFoldTypes,
'+', -1,
false);
154 unsigned StartIdx = 1;
155 if (TailFoldTypes[0] ==
"disabled")
156 setInitialBits(TailFoldingOpts::Disabled);
157 else if (TailFoldTypes[0] ==
"all")
158 setInitialBits(TailFoldingOpts::All);
159 else if (TailFoldTypes[0] ==
"default")
160 setNeedsDefault(
true);
161 else if (TailFoldTypes[0] ==
"simple")
162 setInitialBits(TailFoldingOpts::Simple);
165 setInitialBits(TailFoldingOpts::Disabled);
168 for (
unsigned I = StartIdx;
I < TailFoldTypes.
size();
I++) {
169 if (TailFoldTypes[
I] ==
"reductions")
170 setEnableBit(TailFoldingOpts::Reductions);
171 else if (TailFoldTypes[
I] ==
"recurrences")
172 setEnableBit(TailFoldingOpts::Recurrences);
173 else if (TailFoldTypes[
I] ==
"reverse")
174 setEnableBit(TailFoldingOpts::Reverse);
175 else if (TailFoldTypes[
I] ==
"noreductions")
176 setDisableBit(TailFoldingOpts::Reductions);
177 else if (TailFoldTypes[
I] ==
"norecurrences")
178 setDisableBit(TailFoldingOpts::Recurrences);
179 else if (TailFoldTypes[
I] ==
"noreverse")
180 setDisableBit(TailFoldingOpts::Reverse);
187 return getBits(DefaultBits) == TailFoldingOpts::Disabled;
201 "Control the use of vectorisation using tail-folding for SVE where the"
202 " option is specified in the form (Initial)[+(Flag1|Flag2|...)]:"
203 "\ndisabled (Initial) No loop types will vectorize using "
205 "\ndefault (Initial) Uses the default tail-folding settings for "
207 "\nall (Initial) All legal loop types will vectorize using "
209 "\nsimple (Initial) Use tail-folding for simple loops (not "
210 "reductions or recurrences)"
211 "\nreductions Use tail-folding for loops containing reductions"
212 "\nnoreductions Inverse of above"
213 "\nrecurrences Use tail-folding for loops containing fixed order "
215 "\nnorecurrences Inverse of above"
216 "\nreverse Use tail-folding for loops requiring reversed "
218 "\nnoreverse Inverse of above"),
263 TTI->isMultiversionedFunction(
F) ?
"fmv-features" :
"target-features";
264 StringRef FeatureStr =
F.getFnAttribute(AttributeStr).getValueAsString();
265 FeatureStr.
split(Features,
",");
281 return F.hasFnAttribute(
"fmv-features");
291 if (
CallAttrs.caller().hasNonStreamingInterfaceAndBody() &&
292 CallAttrs.callee().hasStreamingInterfaceOrBody())
297 if (
CallAttrs.callee().hasStreamingBody()) {
307 CallAttrs.requiresPreservingAllZAState()) {
330 auto FVTy = dyn_cast<FixedVectorType>(Ty);
332 FVTy->getScalarSizeInBits() * FVTy->getNumElements() > 128;
341 unsigned DefaultCallPenalty)
const {
366 if (
F ==
Call.getCaller())
372 return DefaultCallPenalty;
383 ST->isSVEorStreamingSVEAvailable() &&
384 !ST->disableMaximizeScalableBandwidth();
408 assert(Ty->isIntegerTy());
410 unsigned BitSize = Ty->getPrimitiveSizeInBits();
417 ImmVal =
Imm.sext((BitSize + 63) & ~0x3fU);
422 for (
unsigned ShiftVal = 0; ShiftVal < BitSize; ShiftVal += 64) {
428 return std::max<InstructionCost>(1,
Cost);
435 assert(Ty->isIntegerTy());
437 unsigned BitSize = Ty->getPrimitiveSizeInBits();
443 unsigned ImmIdx = ~0U;
447 case Instruction::GetElementPtr:
452 case Instruction::Store:
455 case Instruction::Add:
456 case Instruction::Sub:
457 case Instruction::Mul:
458 case Instruction::UDiv:
459 case Instruction::SDiv:
460 case Instruction::URem:
461 case Instruction::SRem:
462 case Instruction::And:
463 case Instruction::Or:
464 case Instruction::Xor:
465 case Instruction::ICmp:
469 case Instruction::Shl:
470 case Instruction::LShr:
471 case Instruction::AShr:
475 case Instruction::Trunc:
476 case Instruction::ZExt:
477 case Instruction::SExt:
478 case Instruction::IntToPtr:
479 case Instruction::PtrToInt:
480 case Instruction::BitCast:
481 case Instruction::PHI:
482 case Instruction::Call:
483 case Instruction::Select:
484 case Instruction::Ret:
485 case Instruction::Load:
490 int NumConstants = (BitSize + 63) / 64;
503 assert(Ty->isIntegerTy());
505 unsigned BitSize = Ty->getPrimitiveSizeInBits();
514 if (IID >= Intrinsic::aarch64_addg && IID <= Intrinsic::aarch64_udiv)
520 case Intrinsic::sadd_with_overflow:
521 case Intrinsic::uadd_with_overflow:
522 case Intrinsic::ssub_with_overflow:
523 case Intrinsic::usub_with_overflow:
524 case Intrinsic::smul_with_overflow:
525 case Intrinsic::umul_with_overflow:
527 int NumConstants = (BitSize + 63) / 64;
534 case Intrinsic::experimental_stackmap:
535 if ((Idx < 2) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
538 case Intrinsic::experimental_patchpoint_void:
539 case Intrinsic::experimental_patchpoint:
540 if ((Idx < 4) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
543 case Intrinsic::experimental_gc_statepoint:
544 if ((Idx < 5) || (
Imm.getBitWidth() <= 64 &&
isInt<64>(
Imm.getSExtValue())))
554 if (TyWidth == 32 || TyWidth == 64)
563 return ST->getMispredictionPenalty();
584 unsigned TotalHistCnts = 1;
594 unsigned EC = VTy->getElementCount().getKnownMinValue();
599 unsigned LegalEltSize = EltSize <= 32 ? 32 : 64;
601 if (EC == 2 || (LegalEltSize == 32 && EC == 4))
605 TotalHistCnts = EC / NaturalVectorWidth;
624 !
is_contained({Intrinsic::masked_load, Intrinsic::masked_store},
628 switch (ICA.
getID()) {
629 case Intrinsic::experimental_vector_histogram_add: {
636 case Intrinsic::clmul: {
641 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
645 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8) {
650 -1,
nullptr,
nullptr) *
653 -1,
nullptr,
nullptr);
657 if (LT.second.SimpleTy == MVT::nxv2i64)
658 if (ST->hasSVEAES() && (ST->isSVEAvailable() || ST->hasSSVE_AES()))
661 if (ST->hasSVE2() || ST->hasSME()) {
662 switch (LT.second.SimpleTy) {
677 if (LT.second.SimpleTy == MVT::nxv2i64)
681 switch (LT.second.SimpleTy) {
691 -1,
nullptr,
nullptr) *
694 -1,
nullptr,
nullptr));
703 return LT.first * 11;
705 return LT.first * 14;
712 case Intrinsic::umin:
713 case Intrinsic::umax:
714 case Intrinsic::smin:
715 case Intrinsic::smax: {
716 static const auto ValidMinMaxTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
717 MVT::v8i16, MVT::v2i32, MVT::v4i32,
718 MVT::nxv16i8, MVT::nxv8i16, MVT::nxv4i32,
725 ICA.
getID() == Intrinsic::smin || ICA.
getID() == Intrinsic::smax;
726 EVT VT = TLI->getValueType(
DL, RetTy,
true);
727 if (VT == MVT::v2i8 || VT == MVT::v2i16 || VT == MVT::v4i8)
728 return LT.first * (IsSigned ? 5 : 3);
730 if (LT.second == MVT::v2i64)
736 case Intrinsic::scmp:
737 case Intrinsic::ucmp: {
739 {Intrinsic::scmp, MVT::i32, 3},
740 {Intrinsic::scmp, MVT::i64, 3},
741 {Intrinsic::scmp, MVT::v8i8, 3},
742 {Intrinsic::scmp, MVT::v16i8, 3},
743 {Intrinsic::scmp, MVT::v4i16, 3},
744 {Intrinsic::scmp, MVT::v8i16, 3},
745 {Intrinsic::scmp, MVT::v2i32, 3},
746 {Intrinsic::scmp, MVT::v4i32, 3},
747 {Intrinsic::scmp, MVT::v1i64, 3},
748 {Intrinsic::scmp, MVT::v2i64, 3},
754 return Entry->Cost * LT.first;
757 case Intrinsic::sadd_sat:
758 case Intrinsic::ssub_sat:
759 case Intrinsic::uadd_sat:
760 case Intrinsic::usub_sat: {
761 static const auto ValidSatTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
762 MVT::v8i16, MVT::v2i32, MVT::v4i32,
768 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits() ? 1 : 4;
770 return LT.first * Instrs;
775 if (ST->isSVEAvailable() && VectorSize >= 128 &&
isPowerOf2_64(VectorSize))
776 return LT.first * Instrs;
780 case Intrinsic::abs: {
781 static const auto ValidAbsTys = {MVT::v8i8, MVT::v16i8, MVT::v4i16,
782 MVT::v8i16, MVT::v2i32, MVT::v4i32,
783 MVT::v2i64, MVT::nxv16i8, MVT::nxv8i16,
784 MVT::nxv4i32, MVT::nxv2i64};
790 case Intrinsic::bswap: {
791 static const auto ValidAbsTys = {MVT::v4i16, MVT::v8i16, MVT::v2i32,
792 MVT::v4i32, MVT::v2i64};
795 LT.second.getScalarSizeInBits() == RetTy->getScalarSizeInBits())
800 case Intrinsic::fmuladd: {
805 (EltTy->
isHalfTy() && ST->hasFullFP16()))
809 case Intrinsic::stepvector: {
818 Cost += AddCost * (LT.first - 1);
822 case Intrinsic::vector_extract:
823 case Intrinsic::vector_insert: {
836 bool IsExtract = ICA.
getID() == Intrinsic::vector_extract;
837 EVT SubVecVT = IsExtract ? getTLI()->getValueType(
DL, RetTy)
845 getTLI()->getTypeConversion(
C, SubVecVT);
847 getTLI()->getTypeConversion(
C, VecVT);
855 case Intrinsic::bitreverse: {
857 {Intrinsic::bitreverse, MVT::i32, 1},
858 {Intrinsic::bitreverse, MVT::i64, 1},
859 {Intrinsic::bitreverse, MVT::v8i8, 1},
860 {Intrinsic::bitreverse, MVT::v16i8, 1},
861 {Intrinsic::bitreverse, MVT::v4i16, 2},
862 {Intrinsic::bitreverse, MVT::v8i16, 2},
863 {Intrinsic::bitreverse, MVT::v2i32, 2},
864 {Intrinsic::bitreverse, MVT::v4i32, 2},
865 {Intrinsic::bitreverse, MVT::v1i64, 2},
866 {Intrinsic::bitreverse, MVT::v2i64, 2},
874 if (TLI->getValueType(
DL, RetTy,
true) == MVT::i8 ||
875 TLI->getValueType(
DL, RetTy,
true) == MVT::i16)
876 return LegalisationCost.first * Entry->Cost + 1;
878 return LegalisationCost.first * Entry->Cost;
882 case Intrinsic::ctpop: {
886 if (ST->hasCSSC() && !RetTy->isVectorTy()) {
889 return LT.first + ExtraCost;
891 if (!ST->hasNEON()) {
921 RetTy->getScalarSizeInBits()
924 return LT.first * Entry->Cost + ExtraCost;
928 case Intrinsic::sadd_with_overflow:
929 case Intrinsic::uadd_with_overflow:
930 case Intrinsic::ssub_with_overflow:
931 case Intrinsic::usub_with_overflow:
932 case Intrinsic::smul_with_overflow:
933 case Intrinsic::umul_with_overflow: {
935 {Intrinsic::sadd_with_overflow, MVT::i8, 3},
936 {Intrinsic::uadd_with_overflow, MVT::i8, 3},
937 {Intrinsic::sadd_with_overflow, MVT::i16, 3},
938 {Intrinsic::uadd_with_overflow, MVT::i16, 3},
939 {Intrinsic::sadd_with_overflow, MVT::i32, 1},
940 {Intrinsic::uadd_with_overflow, MVT::i32, 1},
941 {Intrinsic::sadd_with_overflow, MVT::i64, 1},
942 {Intrinsic::uadd_with_overflow, MVT::i64, 1},
943 {Intrinsic::ssub_with_overflow, MVT::i8, 3},
944 {Intrinsic::usub_with_overflow, MVT::i8, 3},
945 {Intrinsic::ssub_with_overflow, MVT::i16, 3},
946 {Intrinsic::usub_with_overflow, MVT::i16, 3},
947 {Intrinsic::ssub_with_overflow, MVT::i32, 1},
948 {Intrinsic::usub_with_overflow, MVT::i32, 1},
949 {Intrinsic::ssub_with_overflow, MVT::i64, 1},
950 {Intrinsic::usub_with_overflow, MVT::i64, 1},
951 {Intrinsic::smul_with_overflow, MVT::i8, 5},
952 {Intrinsic::umul_with_overflow, MVT::i8, 4},
953 {Intrinsic::smul_with_overflow, MVT::i16, 5},
954 {Intrinsic::umul_with_overflow, MVT::i16, 4},
955 {Intrinsic::smul_with_overflow, MVT::i32, 2},
956 {Intrinsic::umul_with_overflow, MVT::i32, 2},
957 {Intrinsic::smul_with_overflow, MVT::i64, 3},
958 {Intrinsic::umul_with_overflow, MVT::i64, 3},
960 EVT MTy = TLI->getValueType(
DL, RetTy->getContainedType(0),
true);
967 case Intrinsic::fptosi_sat:
968 case Intrinsic::fptoui_sat: {
971 bool IsSigned = ICA.
getID() == Intrinsic::fptosi_sat;
973 EVT MTy = TLI->getValueType(
DL, RetTy);
976 if ((LT.second == MVT::f32 || LT.second == MVT::f64 ||
977 LT.second == MVT::v2f32 || LT.second == MVT::v4f32 ||
978 LT.second == MVT::v2f64)) {
980 (LT.second == MVT::f64 && MTy == MVT::i32) ||
981 (LT.second == MVT::f32 && MTy == MVT::i64)))
990 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
997 if ((LT.second == MVT::f16 && MTy == MVT::i32) ||
998 (LT.second == MVT::f16 && MTy == MVT::i64) ||
999 ((LT.second == MVT::v4f16 || LT.second == MVT::v8f16) &&
1013 if ((LT.second.getScalarType() == MVT::f32 ||
1014 LT.second.getScalarType() == MVT::f64 ||
1015 LT.second.getScalarType() == MVT::f16) &&
1018 Type::getIntNTy(RetTy->getContext(), LT.second.getScalarSizeInBits());
1019 if (LT.second.isVector())
1020 LegalTy =
VectorType::get(LegalTy, LT.second.getVectorElementCount());
1024 LegalTy, {LegalTy, LegalTy});
1028 LegalTy, {LegalTy, LegalTy});
1030 return LT.first *
Cost +
1031 ((LT.second.getScalarType() != MVT::f16 || ST->hasFullFP16()) ? 0
1037 RetTy = RetTy->getScalarType();
1038 if (LT.second.isVector()) {
1056 return LT.first *
Cost;
1058 case Intrinsic::fshl:
1059 case Intrinsic::fshr: {
1068 if (RetTy->isIntegerTy() && ICA.
getArgs()[0] == ICA.
getArgs()[1] &&
1069 (RetTy->getPrimitiveSizeInBits() == 32 ||
1070 RetTy->getPrimitiveSizeInBits() == 64)) {
1083 {Intrinsic::fshl, MVT::v4i32, 2},
1084 {Intrinsic::fshl, MVT::v2i64, 2}, {Intrinsic::fshl, MVT::v16i8, 2},
1085 {Intrinsic::fshl, MVT::v8i16, 2}, {Intrinsic::fshl, MVT::v2i32, 2},
1086 {Intrinsic::fshl, MVT::v8i8, 2}, {Intrinsic::fshl, MVT::v4i16, 2}};
1092 return LegalisationCost.first * Entry->Cost;
1096 if (!RetTy->isIntegerTy())
1101 bool HigherCost = (RetTy->getScalarSizeInBits() != 32 &&
1102 RetTy->getScalarSizeInBits() < 64) ||
1103 (RetTy->getScalarSizeInBits() % 64 != 0);
1104 unsigned ExtraCost = HigherCost ? 1 : 0;
1105 if (RetTy->getScalarSizeInBits() == 32 ||
1106 RetTy->getScalarSizeInBits() == 64)
1109 else if (HigherCost)
1113 return TyL.first + ExtraCost;
1115 case Intrinsic::get_active_lane_mask: {
1117 EVT RetVT = getTLI()->getValueType(
DL, RetTy);
1119 if (getTLI()->shouldExpandGetActiveLaneMask(RetVT, OpVT))
1122 if (RetTy->isScalableTy()) {
1123 if (TLI->getTypeAction(RetTy->getContext(), RetVT) !=
1133 if (ST->hasSVE2p1() || ST->hasSME2()) {
1145 Type *CondTy =
OpTy->getWithNewBitWidth(1);
1148 return Cost + (SplitCost * (
Cost - 1));
1163 case Intrinsic::experimental_vector_match: {
1164 if (!ST->hasSVE2() || !ST->isSVEAvailable())
1170 unsigned SearchSize = NeedleTy->getNumElements();
1171 if (SearchSize <= 2)
1176 {MVT::nxv8i16, MVT::nxv16i8, MVT::v8i16, MVT::v16i8, MVT::v8i8},
1180 unsigned ElementSizeInBits = SearchVT.getScalarSizeInBits();
1186 unsigned MatchesRequiredForNeedle =
1198 return Cost * LegalParts * MatchesRequiredForNeedle;
1200 case Intrinsic::cttz: {
1202 if (LT.second == MVT::v8i8 || LT.second == MVT::v16i8)
1203 return LT.first * 2;
1204 if (LT.second == MVT::v4i16 || LT.second == MVT::v8i16 ||
1205 LT.second == MVT::v2i32 || LT.second == MVT::v4i32)
1206 return LT.first * 3;
1209 case Intrinsic::experimental_cttz_elts: {
1219 case Intrinsic::loop_dependence_raw_mask:
1220 case Intrinsic::loop_dependence_war_mask: {
1222 if (ST->hasSVE2() || ST->hasSME()) {
1223 EVT VecVT = getTLI()->getValueType(
DL, RetTy);
1224 unsigned EltSizeInBytes =
1234 case Intrinsic::experimental_vector_extract_last_active:
1235 if (ST->isSVEorStreamingSVEAvailable()) {
1241 case Intrinsic::pow: {
1244 EVT VT = getTLI()->getValueType(
DL, RetTy);
1245 RTLIB::Libcall LC = RTLIB::getPOW(VT);
1246 bool HasLibcall = getTLI()->getLibcallImpl(LC) != RTLIB::Unsupported;
1261 bool Is025 = ExpF->getValueAPF().isExactlyValue(0.25);
1262 bool Is075 = ExpF->getValueAPF().isExactlyValue(0.75);
1272 return (Sqrt * 2) +
FMul;
1283 case Intrinsic::sqrt:
1284 case Intrinsic::fabs:
1285 case Intrinsic::ceil:
1286 case Intrinsic::floor:
1287 case Intrinsic::nearbyint:
1288 case Intrinsic::round:
1289 case Intrinsic::rint:
1290 case Intrinsic::roundeven:
1291 case Intrinsic::trunc:
1292 case Intrinsic::minnum:
1293 case Intrinsic::maxnum:
1294 case Intrinsic::minimum:
1295 case Intrinsic::maximum: {
1313 auto RequiredType =
II.getType();
1316 assert(PN &&
"Expected Phi Node!");
1319 if (!PN->hasOneUse())
1320 return std::nullopt;
1322 for (
Value *IncValPhi : PN->incoming_values()) {
1325 Reinterpret->getIntrinsicID() !=
1326 Intrinsic::aarch64_sve_convert_to_svbool ||
1327 RequiredType != Reinterpret->getArgOperand(0)->getType())
1328 return std::nullopt;
1336 for (
unsigned I = 0;
I < PN->getNumIncomingValues();
I++) {
1338 NPN->
addIncoming(Reinterpret->getOperand(0), PN->getIncomingBlock(
I));
1411 return GoverningPredicateIdx != std::numeric_limits<unsigned>::max();
1416 return GoverningPredicateIdx;
1421 GoverningPredicateIdx = Index;
1443 return UndefIntrinsic;
1448 UndefIntrinsic = IID;
1475 return CmpPredicate;
1480 CmpPredicate = Pred;
1496 return ResultLanes == InactiveLanesTakenFromOperand;
1501 return OperandIdxForInactiveLanes;
1505 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1506 ResultLanes = InactiveLanesTakenFromOperand;
1507 OperandIdxForInactiveLanes = Index;
1512 return ResultLanes == InactiveLanesAreNotDefined;
1516 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1517 ResultLanes = InactiveLanesAreNotDefined;
1522 return ResultLanes == InactiveLanesAreUnused;
1526 assert(ResultLanes == Uninitialized &&
"Cannot set property twice!");
1527 ResultLanes = InactiveLanesAreUnused;
1537 ResultIsZeroInitialized =
true;
1548 return OperandIdxWithNoActiveLanes != std::numeric_limits<unsigned>::max();
1553 return OperandIdxWithNoActiveLanes;
1558 OperandIdxWithNoActiveLanes = Index;
1563 unsigned GoverningPredicateIdx = std::numeric_limits<unsigned>::max();
1566 unsigned IROpcode = 0;
1569 enum PredicationStyle {
1571 InactiveLanesTakenFromOperand,
1572 InactiveLanesAreNotDefined,
1573 InactiveLanesAreUnused
1576 bool ResultIsZeroInitialized =
false;
1577 unsigned OperandIdxForInactiveLanes = std::numeric_limits<unsigned>::max();
1578 unsigned OperandIdxWithNoActiveLanes = std::numeric_limits<unsigned>::max();
1586 return !isa<ScalableVectorType>(V->getType());
1594 case Intrinsic::aarch64_sve_fcvt_bf16f32_v2:
1595 case Intrinsic::aarch64_sve_fcvt_f16f32:
1596 case Intrinsic::aarch64_sve_fcvt_f16f64:
1597 case Intrinsic::aarch64_sve_fcvt_f32f16:
1598 case Intrinsic::aarch64_sve_fcvt_f32f64:
1599 case Intrinsic::aarch64_sve_fcvt_f64f16:
1600 case Intrinsic::aarch64_sve_fcvt_f64f32:
1601 case Intrinsic::aarch64_sve_fcvtlt_f32f16:
1602 case Intrinsic::aarch64_sve_fcvtlt_f64f32:
1603 case Intrinsic::aarch64_sve_fcvtx_f32f64:
1604 case Intrinsic::aarch64_sve_fcvtzs:
1605 case Intrinsic::aarch64_sve_fcvtzs_i32f16:
1606 case Intrinsic::aarch64_sve_fcvtzs_i32f64:
1607 case Intrinsic::aarch64_sve_fcvtzs_i64f16:
1608 case Intrinsic::aarch64_sve_fcvtzs_i64f32:
1609 case Intrinsic::aarch64_sve_fcvtzu:
1610 case Intrinsic::aarch64_sve_fcvtzu_i32f16:
1611 case Intrinsic::aarch64_sve_fcvtzu_i32f64:
1612 case Intrinsic::aarch64_sve_fcvtzu_i64f16:
1613 case Intrinsic::aarch64_sve_fcvtzu_i64f32:
1614 case Intrinsic::aarch64_sve_revb:
1615 case Intrinsic::aarch64_sve_revh:
1616 case Intrinsic::aarch64_sve_revw:
1617 case Intrinsic::aarch64_sve_revd:
1618 case Intrinsic::aarch64_sve_scvtf:
1619 case Intrinsic::aarch64_sve_scvtf_f16i32:
1620 case Intrinsic::aarch64_sve_scvtf_f16i64:
1621 case Intrinsic::aarch64_sve_scvtf_f32i64:
1622 case Intrinsic::aarch64_sve_scvtf_f64i32:
1623 case Intrinsic::aarch64_sve_ucvtf:
1624 case Intrinsic::aarch64_sve_ucvtf_f16i32:
1625 case Intrinsic::aarch64_sve_ucvtf_f16i64:
1626 case Intrinsic::aarch64_sve_ucvtf_f32i64:
1627 case Intrinsic::aarch64_sve_ucvtf_f64i32:
1630 case Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2:
1631 case Intrinsic::aarch64_sve_fcvtnt_f16f32:
1632 case Intrinsic::aarch64_sve_fcvtnt_f32f64:
1633 case Intrinsic::aarch64_sve_fcvtxnt_f32f64:
1636 case Intrinsic::aarch64_sve_fabd:
1638 case Intrinsic::aarch64_sve_fadd:
1641 case Intrinsic::aarch64_sve_fdiv:
1644 case Intrinsic::aarch64_sve_fmax:
1646 case Intrinsic::aarch64_sve_fmaxnm:
1648 case Intrinsic::aarch64_sve_fmin:
1650 case Intrinsic::aarch64_sve_fminnm:
1652 case Intrinsic::aarch64_sve_fmla:
1654 case Intrinsic::aarch64_sve_fmls:
1656 case Intrinsic::aarch64_sve_fmul:
1659 case Intrinsic::aarch64_sve_fmulx:
1661 case Intrinsic::aarch64_sve_fnmla:
1663 case Intrinsic::aarch64_sve_fnmls:
1665 case Intrinsic::aarch64_sve_fsub:
1668 case Intrinsic::aarch64_sve_add:
1671 case Intrinsic::aarch64_sve_mla:
1673 case Intrinsic::aarch64_sve_mls:
1675 case Intrinsic::aarch64_sve_mul:
1678 case Intrinsic::aarch64_sve_sabd:
1680 case Intrinsic::aarch64_sve_sdiv:
1683 case Intrinsic::aarch64_sve_smax:
1685 case Intrinsic::aarch64_sve_smin:
1687 case Intrinsic::aarch64_sve_smulh:
1689 case Intrinsic::aarch64_sve_sub:
1692 case Intrinsic::aarch64_sve_uabd:
1694 case Intrinsic::aarch64_sve_udiv:
1697 case Intrinsic::aarch64_sve_umax:
1699 case Intrinsic::aarch64_sve_umin:
1701 case Intrinsic::aarch64_sve_umulh:
1703 case Intrinsic::aarch64_sve_asr:
1706 case Intrinsic::aarch64_sve_lsl:
1709 case Intrinsic::aarch64_sve_lsr:
1712 case Intrinsic::aarch64_sve_and:
1715 case Intrinsic::aarch64_sve_bic:
1717 case Intrinsic::aarch64_sve_eor:
1720 case Intrinsic::aarch64_sve_orr:
1723 case Intrinsic::aarch64_sve_shsub:
1725 case Intrinsic::aarch64_sve_shsubr:
1727 case Intrinsic::aarch64_sve_sqrshl:
1729 case Intrinsic::aarch64_sve_sqshl:
1731 case Intrinsic::aarch64_sve_sqsub:
1733 case Intrinsic::aarch64_sve_srshl:
1735 case Intrinsic::aarch64_sve_uhsub:
1737 case Intrinsic::aarch64_sve_uhsubr:
1739 case Intrinsic::aarch64_sve_uqrshl:
1741 case Intrinsic::aarch64_sve_uqshl:
1743 case Intrinsic::aarch64_sve_uqsub:
1745 case Intrinsic::aarch64_sve_urshl:
1748 case Intrinsic::aarch64_sve_add_u:
1751 case Intrinsic::aarch64_sve_and_u:
1754 case Intrinsic::aarch64_sve_asr_u:
1757 case Intrinsic::aarch64_sve_eor_u:
1760 case Intrinsic::aarch64_sve_fadd_u:
1763 case Intrinsic::aarch64_sve_fdiv_u:
1766 case Intrinsic::aarch64_sve_fmul_u:
1769 case Intrinsic::aarch64_sve_fsub_u:
1772 case Intrinsic::aarch64_sve_lsl_u:
1775 case Intrinsic::aarch64_sve_lsr_u:
1778 case Intrinsic::aarch64_sve_mul_u:
1781 case Intrinsic::aarch64_sve_orr_u:
1784 case Intrinsic::aarch64_sve_sdiv_u:
1787 case Intrinsic::aarch64_sve_sub_u:
1790 case Intrinsic::aarch64_sve_udiv_u:
1794 case Intrinsic::aarch64_sve_addqv:
1795 case Intrinsic::aarch64_sve_bic_z:
1796 case Intrinsic::aarch64_sve_brka_z:
1797 case Intrinsic::aarch64_sve_brkb_z:
1798 case Intrinsic::aarch64_sve_brkn_z:
1799 case Intrinsic::aarch64_sve_brkpa_z:
1800 case Intrinsic::aarch64_sve_brkpb_z:
1801 case Intrinsic::aarch64_sve_cntp:
1802 case Intrinsic::aarch64_sve_compact:
1803 case Intrinsic::aarch64_sve_eorv:
1804 case Intrinsic::aarch64_sve_eorqv:
1805 case Intrinsic::aarch64_sve_nand_z:
1806 case Intrinsic::aarch64_sve_nor_z:
1807 case Intrinsic::aarch64_sve_orn_z:
1808 case Intrinsic::aarch64_sve_orv:
1809 case Intrinsic::aarch64_sve_orqv:
1810 case Intrinsic::aarch64_sve_pnext:
1811 case Intrinsic::aarch64_sve_rdffr_z:
1812 case Intrinsic::aarch64_sve_saddv:
1813 case Intrinsic::aarch64_sve_uaddv:
1814 case Intrinsic::aarch64_sve_umaxv:
1815 case Intrinsic::aarch64_sve_umaxqv:
1816 case Intrinsic::aarch64_sve_facge:
1817 case Intrinsic::aarch64_sve_facgt:
1818 case Intrinsic::aarch64_sve_ld1:
1819 case Intrinsic::aarch64_sve_ld1_gather:
1820 case Intrinsic::aarch64_sve_ld1_gather_index:
1821 case Intrinsic::aarch64_sve_ld1_gather_scalar_offset:
1822 case Intrinsic::aarch64_sve_ld1_gather_sxtw:
1823 case Intrinsic::aarch64_sve_ld1_gather_sxtw_index:
1824 case Intrinsic::aarch64_sve_ld1_gather_uxtw:
1825 case Intrinsic::aarch64_sve_ld1_gather_uxtw_index:
1826 case Intrinsic::aarch64_sve_ld1q_gather_index:
1827 case Intrinsic::aarch64_sve_ld1q_gather_scalar_offset:
1828 case Intrinsic::aarch64_sve_ld1q_gather_vector_offset:
1829 case Intrinsic::aarch64_sve_ld1ro:
1830 case Intrinsic::aarch64_sve_ld1rq:
1831 case Intrinsic::aarch64_sve_ld1udq:
1832 case Intrinsic::aarch64_sve_ld1uwq:
1833 case Intrinsic::aarch64_sve_ld2_sret:
1834 case Intrinsic::aarch64_sve_ld2q_sret:
1835 case Intrinsic::aarch64_sve_ld3_sret:
1836 case Intrinsic::aarch64_sve_ld3q_sret:
1837 case Intrinsic::aarch64_sve_ld4_sret:
1838 case Intrinsic::aarch64_sve_ld4q_sret:
1839 case Intrinsic::aarch64_sve_ldff1:
1840 case Intrinsic::aarch64_sve_ldff1_gather:
1841 case Intrinsic::aarch64_sve_ldff1_gather_index:
1842 case Intrinsic::aarch64_sve_ldff1_gather_scalar_offset:
1843 case Intrinsic::aarch64_sve_ldff1_gather_sxtw:
1844 case Intrinsic::aarch64_sve_ldff1_gather_sxtw_index:
1845 case Intrinsic::aarch64_sve_ldff1_gather_uxtw:
1846 case Intrinsic::aarch64_sve_ldff1_gather_uxtw_index:
1847 case Intrinsic::aarch64_sve_ldnf1:
1848 case Intrinsic::aarch64_sve_ldnt1:
1849 case Intrinsic::aarch64_sve_ldnt1_gather:
1850 case Intrinsic::aarch64_sve_ldnt1_gather_index:
1851 case Intrinsic::aarch64_sve_ldnt1_gather_scalar_offset:
1852 case Intrinsic::aarch64_sve_ldnt1_gather_uxtw:
1855 case Intrinsic::aarch64_sve_and_z:
1858 case Intrinsic::aarch64_sve_orr_z:
1861 case Intrinsic::aarch64_sve_eor_z:
1865 case Intrinsic::aarch64_sve_cmpeq:
1866 case Intrinsic::aarch64_sve_cmpeq_wide:
1869 case Intrinsic::aarch64_sve_cmpge:
1870 case Intrinsic::aarch64_sve_cmpge_wide:
1873 case Intrinsic::aarch64_sve_cmpgt:
1874 case Intrinsic::aarch64_sve_cmpgt_wide:
1877 case Intrinsic::aarch64_sve_cmphi:
1878 case Intrinsic::aarch64_sve_cmphi_wide:
1881 case Intrinsic::aarch64_sve_cmphs:
1882 case Intrinsic::aarch64_sve_cmphs_wide:
1885 case Intrinsic::aarch64_sve_cmple_wide:
1888 case Intrinsic::aarch64_sve_cmplo_wide:
1891 case Intrinsic::aarch64_sve_cmpls_wide:
1894 case Intrinsic::aarch64_sve_cmplt_wide:
1897 case Intrinsic::aarch64_sve_cmpne:
1898 case Intrinsic::aarch64_sve_cmpne_wide:
1901 case Intrinsic::aarch64_sve_fcmpeq:
1904 case Intrinsic::aarch64_sve_fcmpge:
1907 case Intrinsic::aarch64_sve_fcmpgt:
1910 case Intrinsic::aarch64_sve_fcmpne:
1913 case Intrinsic::aarch64_sve_fcmpuo:
1917 case Intrinsic::aarch64_sve_prf:
1918 case Intrinsic::aarch64_sve_prfb_gather_index:
1919 case Intrinsic::aarch64_sve_prfb_gather_scalar_offset:
1920 case Intrinsic::aarch64_sve_prfb_gather_sxtw_index:
1921 case Intrinsic::aarch64_sve_prfb_gather_uxtw_index:
1922 case Intrinsic::aarch64_sve_prfd_gather_index:
1923 case Intrinsic::aarch64_sve_prfd_gather_scalar_offset:
1924 case Intrinsic::aarch64_sve_prfd_gather_sxtw_index:
1925 case Intrinsic::aarch64_sve_prfd_gather_uxtw_index:
1926 case Intrinsic::aarch64_sve_prfh_gather_index:
1927 case Intrinsic::aarch64_sve_prfh_gather_scalar_offset:
1928 case Intrinsic::aarch64_sve_prfh_gather_sxtw_index:
1929 case Intrinsic::aarch64_sve_prfh_gather_uxtw_index:
1930 case Intrinsic::aarch64_sve_prfw_gather_index:
1931 case Intrinsic::aarch64_sve_prfw_gather_scalar_offset:
1932 case Intrinsic::aarch64_sve_prfw_gather_sxtw_index:
1933 case Intrinsic::aarch64_sve_prfw_gather_uxtw_index:
1936 case Intrinsic::aarch64_sve_st1_scatter:
1937 case Intrinsic::aarch64_sve_st1_scatter_scalar_offset:
1938 case Intrinsic::aarch64_sve_st1_scatter_sxtw:
1939 case Intrinsic::aarch64_sve_st1_scatter_sxtw_index:
1940 case Intrinsic::aarch64_sve_st1_scatter_uxtw:
1941 case Intrinsic::aarch64_sve_st1_scatter_uxtw_index:
1942 case Intrinsic::aarch64_sve_st1dq:
1943 case Intrinsic::aarch64_sve_st1q_scatter_index:
1944 case Intrinsic::aarch64_sve_st1q_scatter_scalar_offset:
1945 case Intrinsic::aarch64_sve_st1q_scatter_vector_offset:
1946 case Intrinsic::aarch64_sve_st1wq:
1947 case Intrinsic::aarch64_sve_stnt1:
1948 case Intrinsic::aarch64_sve_stnt1_scatter:
1949 case Intrinsic::aarch64_sve_stnt1_scatter_index:
1950 case Intrinsic::aarch64_sve_stnt1_scatter_scalar_offset:
1951 case Intrinsic::aarch64_sve_stnt1_scatter_uxtw:
1953 case Intrinsic::aarch64_sve_st2:
1954 case Intrinsic::aarch64_sve_st2q:
1956 case Intrinsic::aarch64_sve_st3:
1957 case Intrinsic::aarch64_sve_st3q:
1959 case Intrinsic::aarch64_sve_st4:
1960 case Intrinsic::aarch64_sve_st4q:
1968 Value *UncastedPred;
1974 Pred = UncastedPred;
1980 if (OrigPredTy->getMinNumElements() <=
1982 ->getMinNumElements())
1983 Pred = UncastedPred;
1987 return C &&
C->isAllOnesValue();
1994 if (Dup && Dup->getIntrinsicID() == Intrinsic::aarch64_sve_dup &&
1995 Dup->getOperand(1) == Pg &&
isa<Constant>(Dup->getOperand(2)))
2003static std::optional<Instruction *>
2010 Value *Op1 =
II.getOperand(1);
2011 Value *Op2 =
II.getOperand(2);
2036 Value *NarrowOp1, *NarrowOp2;
2047 else if (SimpleNarrow == NarrowOp1)
2049 else if (SimpleNarrow == NarrowOp2)
2054 SimpleNarrow->
getType(), SimpleNarrow);
2063 return std::nullopt;
2074 if (SimpleII == Inactive)
2082static std::optional<Instruction *>
2086 assert((
Opc == Instruction::ICmp ||
Opc == Instruction::FCmp) &&
2087 "Expected a compare operation!");
2094 Opc == Instruction::ICmp &&
LHS->getType() !=
RHS->getType();
2095 assert((IsWideICmp ||
LHS->getType() ==
RHS->getType()) &&
2096 "Unexpected wide compare!");
2112 const APInt *LHSVal, *RHSVal;
2114 return std::nullopt;
2137 return std::nullopt;
2151static std::optional<Instruction *>
2155 return std::nullopt;
2184 II.setCalledFunction(NewDecl);
2190 return std::nullopt;
2201 if (
Opc == Instruction::FCmp ||
Opc == Instruction::ICmp)
2204 return std::nullopt;
2216static std::optional<Instruction *>
2218 auto m_ConvertToSVBool = [](
auto P) {
2222 Intrinsic::aarch64_sve_convert_from_svbool;
2245 return std::nullopt;
2249 case Intrinsic::aarch64_sve_and_z:
2250 case Intrinsic::aarch64_sve_bic_z:
2251 case Intrinsic::aarch64_sve_eor_z:
2252 case Intrinsic::aarch64_sve_nand_z:
2253 case Intrinsic::aarch64_sve_nor_z:
2254 case Intrinsic::aarch64_sve_orn_z:
2255 case Intrinsic::aarch64_sve_orr_z:
2258 return std::nullopt;
2261 Value *BinOpPred = BinOp->getOperand(0);
2262 Value *BinOpOp1 = BinOp->getOperand(1);
2263 Value *BinOpOp2 = BinOp->getOperand(2);
2265 Value *NarrowBinOpPred;
2267 return std::nullopt;
2269 Value *NarrowBinOpOp1 =
2271 Value *NarrowBinOpOp2 = NarrowBinOpOp1;
2272 if (BinOpOp1 != BinOpOp2)
2276 BinOpIID, Ty, {NarrowBinOpPred, NarrowBinOpOp1, NarrowBinOpOp2});
2280static std::optional<Instruction *>
2287 return BinOpCombine;
2292 return std::nullopt;
2295 Value *Cursor =
II.getOperand(0), *EarliestReplacement =
nullptr;
2304 if (CursorVTy->getElementCount().getKnownMinValue() <
2305 IVTy->getElementCount().getKnownMinValue())
2309 if (Cursor->getType() == IVTy)
2310 EarliestReplacement = Cursor;
2315 if (!IntrinsicCursor || !(IntrinsicCursor->getIntrinsicID() ==
2316 Intrinsic::aarch64_sve_convert_to_svbool ||
2317 IntrinsicCursor->getIntrinsicID() ==
2318 Intrinsic::aarch64_sve_convert_from_svbool))
2321 CandidatesForRemoval.
insert(CandidatesForRemoval.
begin(), IntrinsicCursor);
2322 Cursor = IntrinsicCursor->getOperand(0);
2327 if (!EarliestReplacement)
2328 return std::nullopt;
2336 auto *OpPredicate =
II.getOperand(0);
2353 II.getArgOperand(2));
2359 return std::nullopt;
2363 II.getArgOperand(0),
II.getArgOperand(2),
uint64_t(0));
2372 II.getArgOperand(0));
2381 if (!
II.hasOneUse())
2382 return std::nullopt;
2385 return std::nullopt;
2388 switch (
II.getIntrinsicID()) {
2389 case Intrinsic::aarch64_sve_cmpne:
2390 IID = Intrinsic::aarch64_sve_cmpeq;
2392 case Intrinsic::aarch64_sve_cmpne_wide:
2393 IID = Intrinsic::aarch64_sve_cmpeq_wide;
2395 case Intrinsic::aarch64_sve_cmpeq:
2396 IID = Intrinsic::aarch64_sve_cmpne;
2398 case Intrinsic::aarch64_sve_cmpeq_wide:
2399 IID = Intrinsic::aarch64_sve_cmpne_wide;
2402 return std::nullopt;
2407 IID,
II.getOperand(1)->getType(),
2408 {II.getOperand(0), II.getOperand(1), II.getOperand(2)});
2420 return std::nullopt;
2422 for (
auto *U :
II.users()) {
2425 Type *Ty =
II.getOperand(1)->getType();
2430 Intrinsic::aarch64_sve_umin, Ty,
2431 {
II.getOperand(0),
II.getOperand(1), ConstantInt::get(Ty, 1)});
2437 return std::nullopt;
2451 return std::nullopt;
2456 if (!SplatValue || !SplatValue->isZero())
2457 return std::nullopt;
2462 DupQLane->getIntrinsicID() != Intrinsic::aarch64_sve_dupq_lane)
2463 return std::nullopt;
2467 if (!DupQLaneIdx || !DupQLaneIdx->isZero())
2468 return std::nullopt;
2471 if (!VecIns || VecIns->getIntrinsicID() != Intrinsic::vector_insert)
2472 return std::nullopt;
2477 return std::nullopt;
2480 return std::nullopt;
2484 return std::nullopt;
2488 if (!VecTy || !OutTy || VecTy->getNumElements() != OutTy->getMinNumElements())
2489 return std::nullopt;
2491 unsigned NumElts = VecTy->getNumElements();
2492 unsigned PredicateBits = 0;
2495 for (
unsigned I = 0;
I < NumElts; ++
I) {
2498 return std::nullopt;
2500 PredicateBits |= 1 << (
I * (16 / NumElts));
2504 if (PredicateBits == 0) {
2506 PFalse->takeName(&
II);
2512 for (
unsigned I = 0;
I < 16; ++
I)
2513 if ((PredicateBits & (1 <<
I)) != 0)
2516 unsigned PredSize = Mask & -Mask;
2521 for (
unsigned I = 0;
I < 16;
I += PredSize)
2522 if ((PredicateBits & (1 <<
I)) == 0)
2523 return std::nullopt;
2525 auto *ConvertToSVBool =
2528 auto *ConvertFromSVBool =
2530 II.getType(), ConvertToSVBool);
2538 Value *Pg =
II.getArgOperand(0);
2539 Value *Vec =
II.getArgOperand(1);
2540 auto IntrinsicID =
II.getIntrinsicID();
2541 bool IsAfter = IntrinsicID == Intrinsic::aarch64_sve_lasta;
2553 auto OpC = OldBinOp->getOpcode();
2559 OpC, NewLHS, NewRHS, OldBinOp, OldBinOp->getName(),
II.getIterator());
2565 if (IsAfter &&
C &&
C->isNullValue()) {
2569 Extract->insertBefore(
II.getIterator());
2570 Extract->takeName(&
II);
2576 return std::nullopt;
2578 if (IntrPG->getIntrinsicID() != Intrinsic::aarch64_sve_ptrue)
2579 return std::nullopt;
2581 const auto PTruePattern =
2587 return std::nullopt;
2589 unsigned Idx = MinNumElts - 1;
2599 if (Idx >= PgVTy->getMinNumElements())
2600 return std::nullopt;
2605 Extract->insertBefore(
II.getIterator());
2606 Extract->takeName(&
II);
2619 Value *Pg =
II.getArgOperand(0);
2621 Value *Vec =
II.getArgOperand(2);
2624 if (!Ty->isIntegerTy())
2625 return std::nullopt;
2630 return std::nullopt;
2647 II.getIntrinsicID(), {FPVec->getType()}, {Pg, FPFallBack, FPVec});
2662static std::optional<Instruction *>
2666 if (
Pattern == AArch64SVEPredPattern::all) {
2675 return MinNumElts && NumElts >= MinNumElts
2677 II, ConstantInt::get(
II.getType(), MinNumElts)))
2681static std::optional<Instruction *>
2684 if (!ST->isStreaming())
2685 return std::nullopt;
2697 Value *PgVal =
II.getArgOperand(0);
2698 Value *OpVal =
II.getArgOperand(1);
2702 if (PgVal == OpVal &&
2703 (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_first ||
2704 II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_last)) {
2719 return std::nullopt;
2723 if (Pg->
getIntrinsicID() == Intrinsic::aarch64_sve_convert_to_svbool &&
2724 OpIID == Intrinsic::aarch64_sve_convert_to_svbool &&
2738 if ((Pg ==
Op) && (
II.getIntrinsicID() == Intrinsic::aarch64_sve_ptest_any) &&
2739 ((OpIID == Intrinsic::aarch64_sve_brka_z) ||
2740 (OpIID == Intrinsic::aarch64_sve_brkb_z) ||
2741 (OpIID == Intrinsic::aarch64_sve_brkpa_z) ||
2742 (OpIID == Intrinsic::aarch64_sve_brkpb_z) ||
2743 (OpIID == Intrinsic::aarch64_sve_rdffr_z) ||
2744 (OpIID == Intrinsic::aarch64_sve_and_z) ||
2745 (OpIID == Intrinsic::aarch64_sve_bic_z) ||
2746 (OpIID == Intrinsic::aarch64_sve_eor_z) ||
2747 (OpIID == Intrinsic::aarch64_sve_nand_z) ||
2748 (OpIID == Intrinsic::aarch64_sve_nor_z) ||
2749 (OpIID == Intrinsic::aarch64_sve_orn_z) ||
2750 (OpIID == Intrinsic::aarch64_sve_orr_z))) {
2760 return std::nullopt;
2763template <Intrinsic::ID MulOpc, Intrinsic::ID FuseOpc>
2764static std::optional<Instruction *>
2766 bool MergeIntoAddendOp) {
2768 Value *MulOp0, *MulOp1, *AddendOp, *
Mul;
2769 if (MergeIntoAddendOp) {
2770 AddendOp =
II.getOperand(1);
2771 Mul =
II.getOperand(2);
2773 AddendOp =
II.getOperand(2);
2774 Mul =
II.getOperand(1);
2779 return std::nullopt;
2781 if (!
Mul->hasOneUse())
2782 return std::nullopt;
2785 if (
II.getType()->isFPOrFPVectorTy()) {
2790 return std::nullopt;
2792 return std::nullopt;
2797 if (MergeIntoAddendOp)
2807static std::optional<Instruction *>
2809 Value *Pred =
II.getOperand(0);
2810 Value *PtrOp =
II.getOperand(1);
2811 Type *VecTy =
II.getType();
2826static std::optional<Instruction *>
2828 Value *VecOp =
II.getOperand(0);
2829 Value *Pred =
II.getOperand(1);
2830 Value *PtrOp =
II.getOperand(2);
2846 case Intrinsic::aarch64_sve_fmul_u:
2847 return Instruction::BinaryOps::FMul;
2848 case Intrinsic::aarch64_sve_fadd_u:
2849 return Instruction::BinaryOps::FAdd;
2850 case Intrinsic::aarch64_sve_fsub_u:
2851 return Instruction::BinaryOps::FSub;
2853 return Instruction::BinaryOpsEnd;
2857static std::optional<Instruction *>
2860 if (
II.isStrictFP())
2861 return std::nullopt;
2863 auto *OpPredicate =
II.getOperand(0);
2865 if (BinOpCode == Instruction::BinaryOpsEnd ||
2867 return std::nullopt;
2869 BinOpCode,
II.getOperand(1),
II.getOperand(2),
II.getFastMathFlags());
2873static std::optional<Instruction *>
2875 assert(
II.getIntrinsicID() == Intrinsic::aarch64_sve_mla_u &&
2876 "Expected MLA_U intrinsic");
2877 Value *Acc =
II.getArgOperand(1);
2878 Value *MulOp0 =
II.getArgOperand(2);
2879 Value *MulOp1 =
II.getArgOperand(3);
2894 II.setArgOperand(2, MulOp1);
2895 II.setArgOperand(3, MulOp0);
2899 return std::nullopt;
2902static std::optional<Instruction *>
2904 assert((
II.getIntrinsicID() == Intrinsic::aarch64_sve_sadalp ||
2905 II.getIntrinsicID() == Intrinsic::aarch64_sve_uadalp) &&
2906 "Expected SADALP or UADALP intrinsic");
2912 return std::nullopt;
2916 return std::nullopt;
2920 II.getIntrinsicID(), {II.getType()},
2921 {II.getArgOperand(0), Acc, II.getArgOperand(2)});
2931 Intrinsic::aarch64_sve_mla>(
2935 Intrinsic::aarch64_sve_mad>(
2938 return std::nullopt;
2941static std::optional<Instruction *>
2945 Intrinsic::aarch64_sve_fmla>(IC,
II,
2950 Intrinsic::aarch64_sve_fmad>(IC,
II,
2955 Intrinsic::aarch64_sve_fmla>(IC,
II,
2958 return std::nullopt;
2961static std::optional<Instruction *>
2965 Intrinsic::aarch64_sve_fmla>(IC,
II,
2970 Intrinsic::aarch64_sve_fmad>(IC,
II,
2975 Intrinsic::aarch64_sve_fmla_u>(
2981static std::optional<Instruction *>
2985 Intrinsic::aarch64_sve_fmls>(IC,
II,
2990 Intrinsic::aarch64_sve_fnmsb>(
2995 Intrinsic::aarch64_sve_fmls>(IC,
II,
2998 return std::nullopt;
3001static std::optional<Instruction *>
3005 Intrinsic::aarch64_sve_fmls>(IC,
II,
3010 Intrinsic::aarch64_sve_fnmsb>(
3015 Intrinsic::aarch64_sve_fmls_u>(
3024 Intrinsic::aarch64_sve_mls>(
3027 return std::nullopt;
3032 Value *UnpackArg =
II.getArgOperand(0);
3034 bool IsSigned =
II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpkhi ||
3035 II.getIntrinsicID() == Intrinsic::aarch64_sve_sunpklo;
3048 return std::nullopt;
3052 auto *OpVal =
II.getOperand(0);
3053 auto *OpIndices =
II.getOperand(1);
3060 SplatValue->getValue().uge(VTy->getElementCount().getKnownMinValue()))
3061 return std::nullopt;
3076 Type *RetTy =
II.getType();
3077 constexpr Intrinsic::ID FromSVB = Intrinsic::aarch64_sve_convert_from_svbool;
3078 constexpr Intrinsic::ID ToSVB = Intrinsic::aarch64_sve_convert_to_svbool;
3082 if ((
match(
II.getArgOperand(0),
3089 if (TyA ==
B->getType() &&
3094 TyA->getMinNumElements());
3100 return std::nullopt;
3108 if (
match(
II.getArgOperand(0),
3113 II, (
II.getIntrinsicID() == Intrinsic::aarch64_sve_zip1 ?
A :
B));
3115 return std::nullopt;
3118static std::optional<Instruction *>
3120 Value *Mask =
II.getOperand(0);
3121 Value *BasePtr =
II.getOperand(1);
3122 Value *Index =
II.getOperand(2);
3133 BasePtr->getPointerAlignment(
II.getDataLayout());
3136 BasePtr, IndexBase);
3143 return std::nullopt;
3146static std::optional<Instruction *>
3148 Value *Val =
II.getOperand(0);
3149 Value *Mask =
II.getOperand(1);
3150 Value *BasePtr =
II.getOperand(2);
3151 Value *Index =
II.getOperand(3);
3161 BasePtr->getPointerAlignment(
II.getDataLayout());
3164 BasePtr, IndexBase);
3170 return std::nullopt;
3176 Value *Pred =
II.getOperand(0);
3177 Value *Vec =
II.getOperand(1);
3178 Value *DivVec =
II.getOperand(2);
3182 if (!SplatConstantInt)
3183 return std::nullopt;
3187 if (DivisorValue == -1)
3188 return std::nullopt;
3189 if (DivisorValue == 1)
3195 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3202 Intrinsic::aarch64_sve_asrd, {
II.getType()}, {Pred, Vec, DivisorLog2});
3204 Intrinsic::aarch64_sve_neg, {ASRD->getType()}, {ASRD, Pred, ASRD});
3208 return std::nullopt;
3212 size_t VecSize = Vec.
size();
3217 size_t HalfVecSize = VecSize / 2;
3221 if (*
LHS !=
nullptr && *
RHS !=
nullptr) {
3229 if (*
LHS ==
nullptr && *
RHS !=
nullptr)
3247 return std::nullopt;
3254 Elts[Idx->getValue().getZExtValue()] = InsertElt->getOperand(1);
3255 CurrentInsertElt = InsertElt->getOperand(0);
3261 return std::nullopt;
3265 for (
size_t I = 0;
I < Elts.
size();
I++) {
3266 if (Elts[
I] ==
nullptr)
3271 if (InsertEltChain ==
nullptr)
3272 return std::nullopt;
3278 unsigned PatternWidth = IIScalableTy->getScalarSizeInBits() * Elts.
size();
3279 unsigned PatternElementCount = IIScalableTy->getScalarSizeInBits() *
3280 IIScalableTy->getMinNumElements() /
3285 auto *WideShuffleMaskTy =
3296 auto NarrowBitcast =
3309 return std::nullopt;
3314 Value *Pred =
II.getOperand(0);
3315 Value *Vec =
II.getOperand(1);
3316 Value *Shift =
II.getOperand(2);
3319 Value *AbsPred, *MergedValue;
3325 return std::nullopt;
3333 return std::nullopt;
3338 return std::nullopt;
3341 {
II.getType()}, {Pred, Vec, Shift});
3348 Value *Vec =
II.getOperand(0);
3353 return std::nullopt;
3359 auto *NI =
II.getNextNode();
3362 return !
I->mayReadOrWriteMemory() && !
I->mayHaveSideEffects();
3364 while (LookaheadThreshold-- && CanSkipOver(NI)) {
3365 auto *NIBB = NI->getParent();
3366 NI = NI->getNextNode();
3368 if (
auto *SuccBB = NIBB->getUniqueSuccessor())
3369 NI = &*SuccBB->getFirstNonPHIOrDbgOrLifetime();
3375 if (NextII &&
II.isIdenticalTo(NextII))
3378 return std::nullopt;
3386 {II.getType(), II.getOperand(0)->getType()},
3387 {II.getOperand(0), II.getOperand(1)}));
3394 if (PredPattern == AArch64SVEPredPattern::all ||
3395 PredPattern == AArch64SVEPredPattern::pow2)
3397 return std::nullopt;
3403 Value *Passthru =
II.getOperand(0);
3411 auto *Mask = ConstantInt::get(Ty, MaskValue);
3417 return std::nullopt;
3420static std::optional<Instruction *>
3427 return std::nullopt;
3433 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3443 UMinID,
II.getType(), {Pg, NewUMin, ConstantInt::get(II.getType(), 1)});
3453 return std::nullopt;
3459 constexpr Intrinsic::ID UMinID = Intrinsic::aarch64_sve_umin_u;
3467 return std::nullopt;
3470 II.getType(), {Pg, A, B});
3472 UMinID,
II.getType(), {Pg, NewOrr, ConstantInt::get(II.getType(), 1)});
3481 constexpr Intrinsic::ID CmphsID = Intrinsic::aarch64_sve_cmphs;
3486 Value *
A, *PgLHS, *PgRHS;
3492 !
LHS->hasOneUser() || !
RHS->hasOneUser())
3493 return std::nullopt;
3496 if (ConstB > ConstA)
3502 if (PgLHS != PgRHS || (Pg !=
LHS && Pg !=
RHS && Pg != PgLHS))
3503 return std::nullopt;
3505 Type *VecTy =
A->getType();
3509 Constant *Limit = ConstantInt::get(VecTy, ConstA - ConstB);
3516std::optional<Instruction *>
3527 case Intrinsic::aarch64_dmb:
3529 case Intrinsic::aarch64_neon_fmaxnm:
3530 case Intrinsic::aarch64_neon_fminnm:
3532 case Intrinsic::aarch64_sve_convert_from_svbool:
3534 case Intrinsic::aarch64_sve_dup:
3536 case Intrinsic::aarch64_sve_dup_x:
3538 case Intrinsic::aarch64_sve_cmpeq:
3539 case Intrinsic::aarch64_sve_cmpeq_wide:
3541 case Intrinsic::aarch64_sve_cmpne:
3542 case Intrinsic::aarch64_sve_cmpne_wide:
3544 case Intrinsic::aarch64_sve_rdffr:
3546 case Intrinsic::aarch64_sve_lasta:
3547 case Intrinsic::aarch64_sve_lastb:
3549 case Intrinsic::aarch64_sve_clasta_n:
3550 case Intrinsic::aarch64_sve_clastb_n:
3552 case Intrinsic::aarch64_sve_cntd:
3554 case Intrinsic::aarch64_sve_cntw:
3556 case Intrinsic::aarch64_sve_cnth:
3558 case Intrinsic::aarch64_sve_cntb:
3560 case Intrinsic::aarch64_sme_cntsd:
3562 case Intrinsic::aarch64_sve_ptest_any:
3563 case Intrinsic::aarch64_sve_ptest_first:
3564 case Intrinsic::aarch64_sve_ptest_last:
3566 case Intrinsic::aarch64_sve_fadd:
3568 case Intrinsic::aarch64_sve_fadd_u:
3570 case Intrinsic::aarch64_sve_fmul_u:
3572 case Intrinsic::aarch64_sve_fsub:
3574 case Intrinsic::aarch64_sve_fsub_u:
3576 case Intrinsic::aarch64_sve_add:
3578 case Intrinsic::aarch64_sve_add_u:
3580 Intrinsic::aarch64_sve_mla_u>(
3582 case Intrinsic::aarch64_sve_mla_u:
3584 case Intrinsic::aarch64_sve_sadalp:
3585 case Intrinsic::aarch64_sve_uadalp:
3587 case Intrinsic::aarch64_sve_sub:
3589 case Intrinsic::aarch64_sve_sub_u:
3591 Intrinsic::aarch64_sve_mls_u>(
3593 case Intrinsic::aarch64_sve_tbl:
3595 case Intrinsic::aarch64_sve_uunpkhi:
3596 case Intrinsic::aarch64_sve_uunpklo:
3597 case Intrinsic::aarch64_sve_sunpkhi:
3598 case Intrinsic::aarch64_sve_sunpklo:
3600 case Intrinsic::aarch64_sve_uzp1:
3602 case Intrinsic::aarch64_sve_zip1:
3603 case Intrinsic::aarch64_sve_zip2:
3605 case Intrinsic::aarch64_sve_ld1_gather_index:
3607 case Intrinsic::aarch64_sve_st1_scatter_index:
3609 case Intrinsic::aarch64_sve_ld1:
3611 case Intrinsic::aarch64_sve_st1:
3613 case Intrinsic::aarch64_sve_sdiv:
3615 case Intrinsic::aarch64_sve_sel:
3617 case Intrinsic::aarch64_sve_srshl:
3619 case Intrinsic::aarch64_sve_dupq_lane:
3621 case Intrinsic::aarch64_sve_insr:
3623 case Intrinsic::aarch64_sve_whilelo:
3625 case Intrinsic::aarch64_sve_ptrue:
3627 case Intrinsic::aarch64_sve_uxtb:
3629 case Intrinsic::aarch64_sve_uxth:
3631 case Intrinsic::aarch64_sve_uxtw:
3633 case Intrinsic::aarch64_sme_in_streaming_mode:
3635 case Intrinsic::aarch64_sve_umin_u:
3637 case Intrinsic::aarch64_sve_orr_u:
3639 case Intrinsic::aarch64_sve_and_z:
3643 return std::nullopt;
3650 SimplifyAndSetOp)
const {
3651 switch (
II.getIntrinsicID()) {
3654 case Intrinsic::aarch64_neon_fcvtxn:
3655 case Intrinsic::aarch64_neon_rshrn:
3656 case Intrinsic::aarch64_neon_sqrshrn:
3657 case Intrinsic::aarch64_neon_sqrshrun:
3658 case Intrinsic::aarch64_neon_sqshrn:
3659 case Intrinsic::aarch64_neon_sqshrun:
3660 case Intrinsic::aarch64_neon_sqxtn:
3661 case Intrinsic::aarch64_neon_sqxtun:
3662 case Intrinsic::aarch64_neon_uqrshrn:
3663 case Intrinsic::aarch64_neon_uqshrn:
3664 case Intrinsic::aarch64_neon_uqxtn:
3665 SimplifyAndSetOp(&
II, 0, OrigDemandedElts, UndefElts);
3669 return std::nullopt;
3673 return ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3683 if (ST->useSVEForFixedLengthVectors() &&
3686 std::max(ST->getMinSVEVectorSizeInBits(), 128u));
3687 else if (ST->isNeonAvailable())
3692 if (ST->isSVEAvailable() || (ST->isSVEorStreamingSVEAvailable() &&
3701bool AArch64TTIImpl::isSingleExtWideningInstruction(
3703 Type *SrcOverrideTy)
const {
3718 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3721 Type *SrcTy = SrcOverrideTy;
3723 case Instruction::Add:
3724 case Instruction::Sub: {
3733 if (Opcode == Instruction::Sub)
3757 assert(SrcTy &&
"Expected some SrcTy");
3759 unsigned SrcElTySize = SrcTyL.second.getScalarSizeInBits();
3765 DstTyL.first * DstTyL.second.getVectorMinNumElements();
3767 SrcTyL.first * SrcTyL.second.getVectorMinNumElements();
3771 return NumDstEls == NumSrcEls && 2 * SrcElTySize == DstEltSize;
3774Type *AArch64TTIImpl::isBinExtWideningInstruction(
unsigned Opcode,
Type *DstTy,
3776 Type *SrcOverrideTy)
const {
3777 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3778 Opcode != Instruction::Mul)
3788 (DstEltSize != 16 && DstEltSize != 32 && DstEltSize != 64))
3791 auto getScalarSizeWithOverride = [&](
const Value *
V) {
3797 ->getScalarSizeInBits();
3800 unsigned MaxEltSize = 0;
3803 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3804 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3805 MaxEltSize = std::max(EltSize0, EltSize1);
3808 unsigned EltSize0 = getScalarSizeWithOverride(Args[0]);
3809 unsigned EltSize1 = getScalarSizeWithOverride(Args[1]);
3812 if (EltSize0 >= DstEltSize / 2 || EltSize1 >= DstEltSize / 2)
3814 MaxEltSize = DstEltSize / 2;
3815 }
else if (Opcode == Instruction::Mul &&
3823 Known.Zero.countLeadingOnes() >
3828 getScalarSizeWithOverride(
isa<ZExtInst>(Args[0]) ? Args[0] : Args[1]);
3832 if (MaxEltSize * 2 > DstEltSize)
3850 if (!Src->isVectorTy() || !TLI->isTypeLegal(TLI->getValueType(
DL, Src)) ||
3851 (Src->isScalableTy() && !ST->hasSVE2()))
3861 if (AddUser && AddUser->getOpcode() == Instruction::Add)
3865 if (!Shr || Shr->getOpcode() != Instruction::LShr)
3869 if (!Trunc || Trunc->getOpcode() != Instruction::Trunc ||
3870 Src->getScalarSizeInBits() !=
3894 int ISD = TLI->InstructionOpcodeToISD(Opcode);
3898 if (
I && !
I->users().empty()) {
3902 auto GetUserAbsorbedCastCost =
3903 [&](
const Instruction *Usr) -> std::optional<InstructionCost> {
3906 if (
Type *ExtTy = isBinExtWideningInstruction(
3908 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3921 if (isSingleExtWideningInstruction(
3923 Src !=
I->getOperand(0)->getType() ? Src :
nullptr)) {
3927 if (Usr->getOpcode() == Instruction::Add) {
3928 if (
I == Usr->getOperand(1) ||
3944 return std::nullopt;
3948 bool AllUsersAbsorbCast =
true;
3949 for (
const User *U :
I->users()) {
3951 std::optional<InstructionCost> UserCost = GetUserAbsorbedCastCost(Usr);
3953 AllUsersAbsorbCast =
false;
3956 MaxAbsorbedCost = std::max(MaxAbsorbedCost, *UserCost);
3959 if (AllUsersAbsorbCast)
3960 return MaxAbsorbedCost;
3963 EVT SrcTy = TLI->getValueType(
DL, Src);
3964 EVT DstTy = TLI->getValueType(
DL, Dst);
3972 Instruction::ExtractElement, Src,
CostKind, -1,
nullptr,
nullptr);
3974 Opcode, Dst->getScalarType(), Src->getScalarType(), CCH,
CostKind);
3978 if (!SrcTy.isSimple() || !DstTy.
isSimple())
3983 if (!ST->hasSVE2() && !ST->isStreamingSVEAvailable() &&
4012 EVT WiderTy = SrcTy.
bitsGT(DstTy) ? SrcTy : DstTy;
4015 ST->useSVEForFixedLengthVectors(WiderTy)) {
4016 std::pair<InstructionCost, MVT> LT =
4018 unsigned NumElements =
4034 const unsigned int SVE_EXT_COST = 1;
4035 const unsigned int SVE_FCVT_COST = 1;
4036 const unsigned int SVE_UNPACK_ONCE = 4;
4037 const unsigned int SVE_UNPACK_TWICE = 16;
4166 SVE_EXT_COST + SVE_FCVT_COST},
4171 SVE_EXT_COST + SVE_FCVT_COST},
4178 SVE_EXT_COST + SVE_FCVT_COST},
4182 SVE_EXT_COST + SVE_FCVT_COST},
4188 SVE_EXT_COST + SVE_FCVT_COST},
4191 SVE_EXT_COST + SVE_FCVT_COST},
4196 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4198 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4208 SVE_EXT_COST + SVE_FCVT_COST},
4213 SVE_EXT_COST + SVE_FCVT_COST},
4226 SVE_EXT_COST + SVE_FCVT_COST},
4230 SVE_EXT_COST + SVE_FCVT_COST},
4242 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4244 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4246 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4248 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4252 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4254 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4270 SVE_EXT_COST + SVE_FCVT_COST},
4275 SVE_EXT_COST + SVE_FCVT_COST},
4286 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4288 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4290 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4292 SVE_EXT_COST + SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4294 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4296 SVE_UNPACK_ONCE + 2 * SVE_FCVT_COST},
4300 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4302 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4304 SVE_EXT_COST + SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4306 SVE_UNPACK_TWICE + 4 * SVE_FCVT_COST},
4531 if (ST->hasFullFP16())
4543 Src->getScalarType(), CCH,
CostKind) +
4551 ST->isSVEorStreamingSVEAvailable() &&
4552 TLI->getTypeAction(Src->getContext(), SrcTy) ==
4554 TLI->getTypeAction(Dst->getContext(), DstTy) ==
4563 Opcode, LegalTy, Src, CCH,
CostKind,
I);
4566 return Part1 + Part2;
4573 ST->isSVEorStreamingSVEAvailable() && TLI->isTypeLegal(DstTy))
4585 assert((Opcode == Instruction::SExt || Opcode == Instruction::ZExt) &&
4598 CostKind, Index,
nullptr,
nullptr);
4602 auto DstVT = TLI->getValueType(
DL, Dst);
4603 auto SrcVT = TLI->getValueType(
DL, Src);
4608 if (!VecLT.second.isVector() || !TLI->isTypeLegal(DstVT))
4614 if (DstVT.getFixedSizeInBits() < SrcVT.getFixedSizeInBits())
4624 case Instruction::SExt:
4629 case Instruction::ZExt:
4630 if (DstVT.getSizeInBits() != 64u || SrcVT.getSizeInBits() == 32u)
4643 return Opcode == Instruction::PHI ? 0 : 1;
4652 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4654 assert(Ty->isVectorTy() &&
"This must be a vector type");
4661 if (!LT.second.isVector())
4666 if (LT.second.isFixedLengthVector()) {
4667 unsigned Width = LT.second.getVectorNumElements();
4668 Index = Index % Width;
4675 if (Index == 0 && !Ty->getScalarType()->isIntegerTy())
4687 if (Index * Ty->getScalarSizeInBits() < 128)
4689 if (Index * Ty->getScalarSizeInBits() < 512 &&
4690 Opcode == Instruction::ExtractElement)
4692 return Ty->getScalarType()->isIntegerTy() ? Cost + 1 : Cost;
4693 if (Opcode == Instruction::ExtractElement)
4695 if (Opcode == Instruction::InsertElement)
4704 if (VIC == TTI::VectorInstrContext::Load) {
4705 if (ST->hasFastLD1Single())
4709 : ST->getVectorInsertExtractBaseCost() + 1;
4717 : ST->getVectorInsertExtractBaseCost() + 1;
4741 auto ExtractCanFuseWithFmul = [&]() {
4748 auto IsAllowedScalarTy = [&](
const Type *
T) {
4749 return T->isFloatTy() ||
T->isDoubleTy() ||
4750 (
T->isHalfTy() && ST->hasFullFP16());
4754 auto IsUserFMulScalarTy = [](
const Value *EEUser) {
4757 return BO && BO->getOpcode() == BinaryOperator::FMul &&
4758 !BO->getType()->isVectorTy();
4763 auto IsExtractLaneEquivalentToZero = [&](
unsigned Idx,
unsigned EltSz) {
4767 return Idx == 0 || (RegWidth != 0 && (Idx * EltSz) % RegWidth == 0);
4776 DenseMap<User *, unsigned> UserToExtractIdx;
4777 for (
auto *U :
Scalar->users()) {
4778 if (!IsUserFMulScalarTy(U))
4782 UserToExtractIdx[
U];
4784 if (UserToExtractIdx.
empty())
4786 for (
auto &[S, U, L] : ScalarUserAndIdx) {
4787 for (
auto *U : S->users()) {
4788 if (UserToExtractIdx.
contains(U)) {
4790 auto *Op0 =
FMul->getOperand(0);
4791 auto *Op1 =
FMul->getOperand(1);
4792 if ((Op0 == S && Op1 == S) || Op0 != S || Op1 != S) {
4793 UserToExtractIdx[
U] =
L;
4799 for (
auto &[U, L] : UserToExtractIdx) {
4811 return !EE->users().empty() &&
all_of(EE->users(), [&](
const User *U) {
4812 if (!IsUserFMulScalarTy(U))
4817 const auto *BO = cast<BinaryOperator>(U);
4818 const auto *OtherEE = dyn_cast<ExtractElementInst>(
4819 BO->getOperand(0) == EE ? BO->getOperand(1) : BO->getOperand(0));
4821 const auto *IdxOp = dyn_cast<ConstantInt>(OtherEE->getIndexOperand());
4824 return IsExtractLaneEquivalentToZero(
4825 cast<ConstantInt>(OtherEE->getIndexOperand())
4828 OtherEE->getType()->getScalarSizeInBits());
4836 if (Opcode == Instruction::ExtractElement && (
I || Scalar) &&
4837 ExtractCanFuseWithFmul())
4842 :
ST->getVectorInsertExtractBaseCost();
4851 if (Opcode == Instruction::InsertElement && Index == 0 && Op0 &&
4854 return getVectorInstrCostHelper(Opcode, Ty,
CostKind, Index,
nullptr,
nullptr,
4860 Value *Scalar,
ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
4862 return getVectorInstrCostHelper(Opcode, Ty,
CostKind, Index,
nullptr, Scalar,
4863 ScalarUserAndIdx, VIC);
4870 return getVectorInstrCostHelper(
I.getOpcode(), Ty,
CostKind, Index, &
I,
4877 unsigned Index)
const {
4888 : ST->getVectorInsertExtractBaseCost() + 1;
4897 if (Ty->getElementType()->isFloatingPointTy())
4900 unsigned VecInstCost =
4902 return DemandedElts.
popcount() * (Insert + Extract) * VecInstCost;
4909 if (!Ty->getScalarType()->isHalfTy() && !Ty->getScalarType()->isBFloatTy())
4910 return std::nullopt;
4911 if (Ty->getScalarType()->isHalfTy() && ST->hasFullFP16())
4912 return std::nullopt;
4914 if (CanUseSVE && ST->hasSVEB16B16() && ST->isNonStreamingSVEorSME2Available())
4915 return std::nullopt;
4922 Cost += InstCost(PromotedTy);
4937 int ISD = TLI->InstructionOpcodeToISD(Opcode);
4951 Op2Info, Args, CtxI);
4958 Ty,
CostKind, Op1Info, Op2Info,
true,
4961 [&](
Type *PromotedTy) {
4965 return *PromotedCost;
4968 if (Ty->getScalarType()->isFP128Ty())
4976 if (
Type *ExtTy = isBinExtWideningInstruction(Opcode, Ty, Args)) {
4996 ST->hasLimited64bitVectorMulBandwidth())
4999 if (Ty->getScalarSizeInBits() > 64) {
5004 return CostPerLane * CostPerLane * NumLanes * Mul64CostFactor;
5007 if (LT.second == MVT::v2i64) {
5011 return LT.first * Mul64CostFactor;
5032 if (LT.second == MVT::nxv2i64)
5033 return LT.first * Mul64CostFactor;
5092 auto VT = TLI->getValueType(
DL, Ty);
5093 if (VT.isScalarInteger() && VT.getSizeInBits() <= 64) {
5097 : (3 * AsrCost + AddCost);
5099 return MulCost + AsrCost + 2 * AddCost;
5101 }
else if (VT.isVector()) {
5111 if (Ty->isScalableTy() && ST->hasSVE())
5112 Cost += 2 * AsrCost;
5117 ? (LT.second.getScalarType() == MVT::i64 ? 1 : 2) * AsrCost
5121 }
else if (LT.second == MVT::v2i64) {
5122 return VT.getVectorNumElements() *
5129 if (Ty->isScalableTy() && ST->hasSVE())
5130 return MulCost + 2 * AddCost + 2 * AsrCost;
5131 return 2 * MulCost + AddCost + AsrCost + UsraCost;
5136 LT.second.isFixedLengthVector()) {
5146 return ExtractCost + InsertCost +
5154 auto VT = TLI->getValueType(
DL, Ty);
5170 bool HasMULH = VT == MVT::i64 || LT.second == MVT::nxv2i64 ||
5171 LT.second == MVT::nxv4i32 || LT.second == MVT::nxv8i16 ||
5172 LT.second == MVT::nxv16i8;
5173 bool Is128bit = LT.second.is128BitVector();
5185 (HasMULH ? 0 : ShrCost) +
5186 AddCost * 2 + ShrCost;
5187 return DivCost + (
ISD ==
ISD::UREM ? MulCost + AddCost : 0);
5194 if (!VT.isVector() && VT.getSizeInBits() > 64)
5198 Opcode, Ty,
CostKind, Op1Info, Op2Info);
5200 if (TLI->isOperationLegalOrCustom(
ISD, LT.second) && ST->hasSVE()) {
5204 Ty->getPrimitiveSizeInBits().getFixedValue() < 128) {
5214 if (
nullptr != Entry)
5222 FVTy && LT.second.isFixedLengthVector()) {
5223 unsigned NumElts = FVTy->getNumElements();
5224 unsigned RegElts = LT.second.getVectorNumElements();
5226 Cost = (NumElts / RegElts +
popcount(NumElts % RegElts)) * 2;
5230 if (LT.second.getScalarType() == MVT::i8)
5232 else if (LT.second.getScalarType() == MVT::i16)
5244 Opcode, Ty->getScalarType(),
CostKind, Op1Info, Op2Info);
5245 return (4 + DivCost) * VTy->getNumElements();
5251 -1,
nullptr,
nullptr);
5278 LT.second.isFixedLengthVector())
5279 return 2 * LT.first + 1;
5288 if ((Ty->isFloatTy() || Ty->isDoubleTy() ||
5289 (Ty->isHalfTy() && ST->hasFullFP16())) &&
5298 if (!Ty->getScalarType()->isFP128Ty())
5305 if (!Ty->getScalarType()->isFP128Ty())
5306 return 2 * LT.first;
5313 if (!Ty->isVectorTy())
5329 int MaxMergeDistance = 64;
5333 return NumVectorInstToHideOverhead;
5343 unsigned Opcode1,
unsigned Opcode2)
const {
5346 if (!
Sched.hasInstrSchedModel())
5350 Sched.getSchedClassDesc(
TII->get(Opcode1).getSchedClass());
5352 Sched.getSchedClassDesc(
TII->get(Opcode2).getSchedClass());
5358 "Cannot handle variant scheduling classes without an MI");
5374 const int AmortizationCost = 20;
5382 VecPred = CurrentPred;
5390 static const auto ValidMinMaxTys = {
5391 MVT::v8i8, MVT::v16i8, MVT::v4i16, MVT::v8i16, MVT::v2i32,
5392 MVT::v4i32, MVT::v2i64, MVT::v2f32, MVT::v4f32, MVT::v2f64};
5393 static const auto ValidFP16MinMaxTys = {MVT::v4f16, MVT::v8f16};
5397 (ST->hasFullFP16() &&
5403 {Instruction::Select, MVT::v2i1, MVT::v2f32, 2},
5404 {Instruction::Select, MVT::v2i1, MVT::v2f64, 2},
5405 {Instruction::Select, MVT::v4i1, MVT::v4f32, 2},
5406 {Instruction::Select, MVT::v4i1, MVT::v4f16, 2},
5407 {Instruction::Select, MVT::v8i1, MVT::v8f16, 2},
5408 {Instruction::Select, MVT::v16i1, MVT::v16i16, 16},
5409 {Instruction::Select, MVT::v8i1, MVT::v8i32, 8},
5410 {Instruction::Select, MVT::v16i1, MVT::v16i32, 16},
5411 {Instruction::Select, MVT::v4i1, MVT::v4i64, 4 * AmortizationCost},
5412 {Instruction::Select, MVT::v8i1, MVT::v8i64, 8 * AmortizationCost},
5413 {Instruction::Select, MVT::v16i1, MVT::v16i64, 16 * AmortizationCost}};
5415 EVT SelCondTy = TLI->getValueType(
DL, CondTy);
5416 EVT SelValTy = TLI->getValueType(
DL, ValTy);
5425 if (Opcode == Instruction::FCmp) {
5427 ValTy,
CostKind, Op1Info, Op2Info,
false,
5429 false, [&](
Type *PromotedTy) {
5441 return *PromotedCost;
5445 if (LT.second.getScalarType() != MVT::f64 &&
5446 LT.second.getScalarType() != MVT::f32 &&
5447 LT.second.getScalarType() != MVT::f16)
5452 unsigned Factor = 1;
5453 if (!CondTy->isVectorTy() &&
5467 AArch64::FCMEQv4f32))
5479 TLI->isTypeLegal(TLI->getValueType(
DL, ValTy)) &&
5498 Op1Info, Op2Info,
I);
5504 if (ST->requiresStrictAlign()) {
5509 Options.AllowOverlappingLoads =
true;
5510 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
5515 Options.LoadSizes = {8, 4, 2, 1};
5516 Options.AllowedTailExpansions = {3, 5, 6};
5521 return ST->hasSVE();
5527 switch (MICA.
getID()) {
5528 case Intrinsic::masked_scatter:
5529 case Intrinsic::masked_gather:
5531 case Intrinsic::masked_load:
5532 case Intrinsic::masked_store:
5533 case Intrinsic::masked_expandload:
5534 case Intrinsic::masked_compressstore:
5548 if (!LT.first.isValid())
5553 if (VT->getElementType()->isIntegerTy(1))
5559 return is_contained({Intrinsic::masked_load, Intrinsic::masked_store},
5565 if (MICA.
getID() == Intrinsic::masked_expandload) {
5578 if (MICA.
getID() == Intrinsic::masked_compressstore) {
5598 if (LT.first > 1 && LT.second.getScalarSizeInBits() > 8)
5599 return MemOpCost * 2;
5608 assert((Opcode == Instruction::Load || Opcode == Instruction::Store) &&
5609 "Should be called on only load or stores.");
5611 case Instruction::Load:
5614 return ST->getGatherOverhead();
5616 case Instruction::Store:
5619 return ST->getScatterOverhead();
5630 unsigned Opcode = (MICA.
getID() == Intrinsic::masked_gather ||
5631 MICA.
getID() == Intrinsic::vp_gather)
5633 : Instruction::Store;
5643 if (!LT.first.isValid())
5647 if (!LT.second.isVector() ||
5649 VT->getElementType()->isIntegerTy(1))
5659 ElementCount LegalVF = LT.second.getVectorElementCount();
5662 {TTI::OK_AnyValue, TTI::OP_None},
I);
5678 EVT VT = TLI->getValueType(
DL, Ty,
true);
5680 if (VT == MVT::Other)
5685 if (!LT.first.isValid())
5691 if (VTy->getElementType()->isIntegerTy(1) &&
5692 !VTy->getElementCount().isKnownMultipleOf(
5698 Intrinsic::ID IID = Opcode == Instruction::Load ? Intrinsic::masked_load
5699 : Intrinsic::masked_store;
5713 if (Opcode == Instruction::Store)
5717 if (ST->getFixedLoadLatency())
5718 return (LT.first - 1) + ST->getFixedLoadLatency();
5727 if (LT.second.isScalableVector() ||
5728 ST->useSVEForFixedLengthVectors(LT.second)) {
5729 Inst = AArch64::LDR_ZXI;
5730 }
else if (LT.second.isVector() || LT.second.isFloatingPoint()) {
5731 switch (LT.second.getSizeInBits()) {
5733 Inst = AArch64::LDRBui;
5736 Inst = AArch64::LDRHui;
5739 Inst = AArch64::LDRSui;
5742 Inst = AArch64::LDRDui;
5745 Inst = AArch64::LDRQui;
5751 switch (LT.second.getSizeInBits()) {
5753 Inst = AArch64::LDRBBui;
5756 Inst = AArch64::LDRHHui;
5759 Inst = AArch64::LDRWui;
5762 Inst = AArch64::LDRXui;
5770 unsigned SchedClass =
TII->get(Inst).getSchedClass();
5772 ?
Sched.getSchedClassDesc(SchedClass)
5778 return (LT.first - 1) + ST->getLoadLatency();
5781 float NumLoads = (LT.first - 1).
getValue();
5782 return NumLoads *
Sched.getReciprocalThroughput(*ST, *SCD) +
5783 Sched.computeInstrLatency(*ST, *SCD);
5786 if (ST->isMisaligned128StoreSlow() && Opcode == Instruction::Store &&
5787 LT.second.is128BitVector() && Alignment <
Align(16)) {
5793 const int AmortizationCost = 6;
5795 return LT.first * 2 * AmortizationCost;
5799 if (Ty->isPtrOrPtrVectorTy())
5804 if (Ty->getScalarSizeInBits() != LT.second.getScalarSizeInBits()) {
5806 if (VT == MVT::v4i8)
5813 if (!
isPowerOf2_32(EltSize) || EltSize < 8 || EltSize > 64 ||
5814 Alignment !=
Align(1))
5826 if (Remainder != 0) {
5829 while (!TypeWorklist.
empty()) {
5839 TypeWorklist.
push_back({CurrNumElements - PrevPow2,
Offset + PrevPow2});
5851 bool UseMaskForCond,
bool UseMaskForGaps)
const {
5852 assert(Factor >= 2 &&
"Invalid interleave factor");
5861 if (Factor > TLI->getMaxSupportedInterleaveFactor())
5865 DL.getTypeSizeInBits(VecTy).getKnownMinValue() != (3 * 128))
5870 unsigned MaxNativeInterleaveFactor = TLI->getMaxSupportedInterleaveFactor();
5875 (UseMaskForCond || UseMaskForGaps ||
5876 (Factor > MaxNativeInterleaveFactor &&
5877 TLI->useSVEForFixedLengthVectorVT(LT.second))))
5880 if (!UseMaskForGaps && Factor <= MaxNativeInterleaveFactor) {
5883 EC.divideCoefficientBy(Factor));
5889 if (EC.isKnownMultipleOf(Factor) &&
5890 TLI->isLegalInterleavedAccessType(SubVecTy,
DL, UseScalable))
5891 return Factor * TLI->getNumInterleavedAccesses(SubVecTy,
DL, UseScalable);
5896 if (VecTy->
isScalableTy() && EC.isKnownMultipleOf(Factor)) {
5902 if (UseMaskForCond) {
5903 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
5904 : Intrinsic::masked_store;
5924 if (Opcode == Instruction::Store && Factor == 4 &&
5925 SubVecCost.second.getScalarSizeInBits() ==
5926 (4 * ResultCost.second.getScalarSizeInBits()))
5927 LegalizationCost *= 4;
5929 return MemCost + (Factor * LegalizationCost) + (Factor *
Log2_64(Factor));
5935 UseMaskForCond, UseMaskForGaps);
5942 for (
auto *
I : Tys) {
5943 if (!
I->isVectorTy())
5954 Align Alignment)
const {
5961 return (ST->isSVEAvailable() && ST->hasSVE2p2()) ||
5962 (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2());
5974 Size.getFixedValue() <= 16;
5979 bool HasUnorderedReductions)
const {
5982 return ST->getMaxInterleaveFactor();
5992 enum { MaxStridedLoads = 7 };
5994 int StridedLoads = 0;
5997 for (
const auto BB : L->blocks()) {
5998 for (
auto &
I : *BB) {
6004 if (L->isLoopInvariant(PtrValue))
6009 if (!LSCEVAddRec || !LSCEVAddRec->
isAffine())
6018 if (StridedLoads > MaxStridedLoads / 2)
6019 return StridedLoads;
6022 return StridedLoads;
6025 int StridedLoads = countStridedLoads(L, SE);
6027 <<
" strided loads\n");
6043 unsigned *FinalSize) {
6047 for (
auto *BB : L->getBlocks()) {
6048 for (
auto &
I : *BB) {
6054 if (!Cost.isValid())
6058 if (LoopCost > Budget)
6080 if (MaxTC > 0 && MaxTC <= 32)
6091 if (Blocks.
size() != 2)
6113 if (!L->isInnermost() || L->getNumBlocks() > 8)
6117 if (!L->getExitBlock())
6123 bool HasParellelizableReductions =
6124 L->getNumBlocks() == 1 &&
6125 any_of(L->getHeader()->phis(),
6127 return canParallelizeReductionWhenUnrolling(Phi, L, &SE);
6130 if (HasParellelizableReductions &&
6152 if (HasParellelizableReductions) {
6163 if (Header == Latch) {
6166 unsigned Width = 10;
6172 unsigned MaxInstsPerLine = 16;
6174 unsigned BestUC = 1;
6175 unsigned SizeWithBestUC = BestUC *
Size;
6177 unsigned SizeWithUC = UC *
Size;
6178 if (SizeWithUC > 48)
6180 if ((SizeWithUC % MaxInstsPerLine) == 0 ||
6181 (SizeWithBestUC % MaxInstsPerLine) < (SizeWithUC % MaxInstsPerLine)) {
6183 SizeWithBestUC = BestUC *
Size;
6193 for (
auto *BB : L->blocks()) {
6194 for (
auto &
I : *BB) {
6204 for (
auto *U :
I.users())
6206 LoadedValuesPlus.
insert(U);
6213 return LoadedValuesPlus.
contains(
SI->getOperand(0));
6239 auto *I = dyn_cast<Instruction>(V);
6240 return I && DependsOnLoopLoad(I, Depth + 1);
6247 DependsOnLoopLoad(
I, 0)) {
6279 if (L->getLoopDepth() > 1)
6290 for (
auto *BB : L->getBlocks()) {
6291 for (
auto &
I : *BB) {
6295 if (IsVectorized &&
I.getType()->isVectorTy())
6312 if (ST->isAppleMLike())
6314 else if (ST->getProcFamily() == AArch64Subtarget::Falkor &&
6336 !ST->getSchedModel().isOutOfOrder()) {
6359 bool CanCreate)
const {
6363 case Intrinsic::aarch64_neon_st1x2:
6364 case Intrinsic::aarch64_neon_st1x3:
6365 case Intrinsic::aarch64_neon_st1x4:
6366 case Intrinsic::aarch64_neon_st2:
6367 case Intrinsic::aarch64_neon_st3:
6368 case Intrinsic::aarch64_neon_st4: {
6371 if (!CanCreate || !ST)
6373 unsigned NumElts = Inst->
arg_size() - 1;
6374 if (ST->getNumElements() != NumElts)
6376 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6382 for (
unsigned i = 0, e = NumElts; i != e; ++i) {
6384 Res = Builder.CreateInsertValue(Res, L, i);
6388 case Intrinsic::aarch64_neon_ld1x2:
6389 case Intrinsic::aarch64_neon_ld1x3:
6390 case Intrinsic::aarch64_neon_ld1x4:
6391 case Intrinsic::aarch64_neon_ld2:
6392 case Intrinsic::aarch64_neon_ld3:
6393 case Intrinsic::aarch64_neon_ld4:
6394 if (Inst->
getType() == ExpectedType)
6405 case Intrinsic::aarch64_neon_ld1x2:
6406 case Intrinsic::aarch64_neon_ld1x3:
6407 case Intrinsic::aarch64_neon_ld1x4:
6408 case Intrinsic::aarch64_neon_ld2:
6409 case Intrinsic::aarch64_neon_ld3:
6410 case Intrinsic::aarch64_neon_ld4:
6411 Info.ReadMem =
true;
6412 Info.WriteMem =
false;
6415 case Intrinsic::aarch64_neon_st1x2:
6416 case Intrinsic::aarch64_neon_st1x3:
6417 case Intrinsic::aarch64_neon_st1x4:
6418 case Intrinsic::aarch64_neon_st2:
6419 case Intrinsic::aarch64_neon_st3:
6420 case Intrinsic::aarch64_neon_st4:
6421 Info.ReadMem =
false;
6422 Info.WriteMem =
true;
6431 case Intrinsic::aarch64_neon_ld1x2:
6432 case Intrinsic::aarch64_neon_st1x2:
6433 Info.MatchingId = Intrinsic::aarch64_neon_ld1x2;
6435 case Intrinsic::aarch64_neon_ld1x3:
6436 case Intrinsic::aarch64_neon_st1x3:
6437 Info.MatchingId = Intrinsic::aarch64_neon_ld1x3;
6439 case Intrinsic::aarch64_neon_ld1x4:
6440 case Intrinsic::aarch64_neon_st1x4:
6441 Info.MatchingId = Intrinsic::aarch64_neon_ld1x4;
6443 case Intrinsic::aarch64_neon_ld2:
6444 case Intrinsic::aarch64_neon_st2:
6445 Info.MatchingId = Intrinsic::aarch64_neon_ld2;
6447 case Intrinsic::aarch64_neon_ld3:
6448 case Intrinsic::aarch64_neon_st3:
6449 Info.MatchingId = Intrinsic::aarch64_neon_ld3;
6451 case Intrinsic::aarch64_neon_ld4:
6452 case Intrinsic::aarch64_neon_st4:
6453 Info.MatchingId = Intrinsic::aarch64_neon_ld4;
6465 const Instruction &
I,
bool &AllowPromotionWithoutCommonHeader)
const {
6466 bool Considerable =
false;
6467 AllowPromotionWithoutCommonHeader =
false;
6470 Type *ConsideredSExtType =
6472 if (
I.getType() != ConsideredSExtType)
6476 for (
const User *U :
I.users()) {
6478 Considerable =
true;
6482 if (GEPInst->getNumOperands() > 2) {
6483 AllowPromotionWithoutCommonHeader =
true;
6488 return Considerable;
6539 if (LT.second.getScalarType() == MVT::f16 && !ST->hasFullFP16())
6549 return LegalizationCost + 2;
6559 LegalizationCost *= LT.first - 1;
6562 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6571 return LegalizationCost + 2;
6579 std::optional<FastMathFlags> FMF,
6595 return BaseCost + FixedVTy->getNumElements();
6609 MVT MTy = LT.second;
6614 int ISD = TLI->InstructionOpcodeToISD(Opcode);
6662 MTy.
isVector() && (EltTy->isFloatTy() || EltTy->isDoubleTy() ||
6663 (EltTy->isHalfTy() && ST->hasFullFP16()))) {
6675 return (LT.first - 1) +
Log2_32(NElts);
6680 return (LT.first - 1) + Entry->Cost;
6692 if (LT.first != 1) {
6698 ExtraCost *= LT.first - 1;
6701 auto Cost = ValVTy->getElementType()->isIntegerTy(1) ? 2 : Entry->Cost;
6702 return Cost + ExtraCost;
6710 unsigned Opcode,
bool IsUnsigned,
Type *ResTy,
VectorType *VecTy,
6712 EVT VecVT = TLI->getValueType(
DL, VecTy);
6713 EVT ResVT = TLI->getValueType(
DL, ResTy);
6723 if (((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6725 ((LT.second == MVT::v4i16 || LT.second == MVT::v8i16) &&
6727 ((LT.second == MVT::v2i32 || LT.second == MVT::v4i32) &&
6729 return (LT.first - 1) * 2 + 2;
6740 EVT VecVT = TLI->getValueType(
DL, VecTy);
6741 EVT ResVT = TLI->getValueType(
DL, ResTy);
6744 RedOpcode == Instruction::Add) {
6750 if ((LT.second == MVT::v8i8 || LT.second == MVT::v16i8) &&
6752 return LT.first + 2;
6787 EVT PromotedVT = LT.second.getScalarType() == MVT::i1
6788 ? TLI->getPromotedVTForPredicate(
EVT(LT.second))
6802 if (LT.second.getScalarType() == MVT::i1) {
6811 assert(Entry &&
"Illegal Type for Splice");
6812 LegalizationCost += Entry->Cost;
6813 return LegalizationCost * LT.first;
6817 unsigned Opcode,
Type *InputTypeA,
Type *InputTypeB,
Type *AccumType,
6826 if ((Opcode != Instruction::Add && Opcode != Instruction::Sub &&
6827 Opcode != Instruction::FAdd && Opcode != Instruction::FSub))
6833 assert(FMF &&
"Missing FastMathFlags for floating-point partial reduction");
6834 if (!FMF->allowReassoc() || !FMF->allowContract())
6838 "FastMathFlags only apply to floating-point partial reductions");
6842 (!BinOp || (OpBExtend !=
TTI::PR_None && InputTypeB)) &&
6843 "Unexpected values for OpBExtend or InputTypeB");
6847 if (BinOp && ((*BinOp != Instruction::Mul && *BinOp != Instruction::FMul) ||
6848 InputTypeA != InputTypeB))
6859 assert(!OpBExtend &&
"Extended second operand without extended first.");
6860 assert(InputTypeA == AccumType &&
"Type mismatch with no extensions.");
6866 bool IsUSDot = OpBExtend !=
TTI::PR_None && OpAExtend != OpBExtend;
6869 if (IsUSDot && !ST->hasMatMulInt8() && !ST->hasDotProd())
6882 auto TC = TLI->getTypeConversion(AccumVectorType->
getContext(),
6891 if (TLI->getTypeAction(AccumVectorType->
getContext(), TC.second) !=
6897 std::pair<InstructionCost, MVT> AccumLT =
6899 std::pair<InstructionCost, MVT> InputLT =
6903 auto IsSupported = [&](
bool SVEPred,
bool NEONPred) ->
bool {
6904 return (ST->isSVEorStreamingSVEAvailable() && SVEPred) ||
6905 (AccumLT.second.isFixedLengthVector() &&
6906 AccumLT.second.getSizeInBits() <= 128 && ST->isNeonAvailable() &&
6910 bool IsSub = Opcode == Instruction::Sub || Opcode == Instruction::FSub;
6918 if (AccumLT.second.getScalarType() == MVT::i32 &&
6919 InputLT.second.getScalarType() == MVT::i8) {
6921 if (!IsUSDot && IsSupported(
true, ST->hasDotProd()))
6922 return Cost + INegCost;
6924 if (IsUSDot && IsSupported(ST->hasMatMulInt8(), ST->hasMatMulInt8()))
6925 return Cost + INegCost;
6930 if (IsUSDot && IsSupported(
false, ST->hasDotProd()))
6931 return Cost * 3 + INegCost;
6934 if (ST->isSVEorStreamingSVEAvailable() && !IsUSDot) {
6936 if (AccumLT.second.getScalarType() == MVT::i64 &&
6937 InputLT.second.getScalarType() == MVT::i16)
6938 return Cost + INegCost;
6941 if (AccumLT.second.getScalarType() == MVT::i32 &&
6942 InputLT.second.getScalarType() == MVT::i16 &&
6943 (ST->hasSVE2p1() || ST->hasSME2()) && !IsSub)
6946 if (AccumLT.second.getScalarType() == MVT::i64 &&
6947 InputLT.second.getScalarType() == MVT::i8)
6953 return Cost + INegCost;
6956 if (AccumLT.second.getScalarType() == MVT::i16 &&
6957 InputLT.second.getScalarType() == MVT::i8 &&
6958 (ST->hasSVE2p3() || ST->hasSME2p3()) && !IsSub)
6964 if (Opcode == Instruction::FAdd && !IsSub &&
6965 IsSupported(ST->hasSME2() || ST->hasSVE2p1(), ST->hasF16F32DOT()) &&
6966 AccumLT.second.getScalarType() == MVT::f32 &&
6967 InputLT.second.getScalarType() == MVT::f16)
6971 if (Ratio == 2 && !IsUSDot) {
6972 MVT InVT = InputLT.second.getScalarType();
6976 if (IsSupported(ST->hasSVE2() || ST->hasSME(),
true) &&
6978 return (BinOp || IsSub) ?
Cost * 2 :
Cost;
6981 if (IsSupported(ST->hasSVE2(), ST->hasFP16FML()) && InVT == MVT::f16)
6985 if (IsSupported(ST->hasSVE2p1() || ST->hasSME2(),
false) &&
6986 InVT == MVT::bf16 && IsSub)
6996 if (IsSupported(ST->hasBF16(), ST->hasBF16()) && InVT == MVT::bf16)
6997 return Cost * 2 + FNegCost;
7001 AccumType, VF, OpAExtend, OpBExtend,
7012 "Expected the Mask to match the return size if given");
7014 "Expected the same scalar types");
7020 LT.second.getScalarSizeInBits() * Mask.size() > 128 &&
7021 SrcTy->getScalarSizeInBits() == LT.second.getScalarSizeInBits() &&
7022 Mask.size() > LT.second.getVectorNumElements() && !Index && !SubTp) {
7030 return std::max<InstructionCost>(1, LT.first / 4);
7038 Mask, 4, SrcTy->getElementCount().getKnownMinValue() * 2) ||
7040 Mask, 3, SrcTy->getElementCount().getKnownMinValue() * 2)))
7043 unsigned TpNumElts = Mask.size();
7044 unsigned LTNumElts = LT.second.getVectorNumElements();
7045 unsigned NumVecs = (TpNumElts + LTNumElts - 1) / LTNumElts;
7047 LT.second.getVectorElementCount());
7049 std::map<std::tuple<unsigned, unsigned, SmallVector<int>>,
InstructionCost>
7051 for (
unsigned N = 0;
N < NumVecs;
N++) {
7055 unsigned Source1 = -1U, Source2 = -1U;
7056 unsigned NumSources = 0;
7057 for (
unsigned E = 0; E < LTNumElts; E++) {
7058 int MaskElt = (
N * LTNumElts + E < TpNumElts) ? Mask[
N * LTNumElts + E]
7067 unsigned Source = MaskElt / LTNumElts;
7068 if (NumSources == 0) {
7071 }
else if (NumSources == 1 && Source != Source1) {
7074 }
else if (NumSources >= 2 && Source != Source1 && Source != Source2) {
7080 if (Source == Source1)
7082 else if (Source == Source2)
7083 NMask.
push_back(MaskElt % LTNumElts + LTNumElts);
7092 PreviousCosts.insert({std::make_tuple(Source1, Source2, NMask), 0});
7103 NTp, NTp,
CostKind, NMask, 0,
nullptr, Args,
7106 Result.first->second = NCost;
7120 if (IsExtractSubvector && LT.second.isFixedLengthVector()) {
7121 if (LT.second.getFixedSizeInBits() >= 128 &&
7123 LT.second.getVectorNumElements() / 2) {
7126 if (Index == (
int)LT.second.getVectorNumElements() / 2)
7140 if (!Mask.empty() && LT.second.isFixedLengthVector() &&
7143 return M.value() < 0 || M.value() == (int)M.index();
7149 !Mask.empty() && SrcTy->getPrimitiveSizeInBits().isNonZero() &&
7150 SrcTy->getPrimitiveSizeInBits().isKnownMultipleOf(
7159 if ((ST->hasSVE2p1() || ST->hasSME2p1()) &&
7160 ST->isSVEorStreamingSVEAvailable() &&
7165 if (ST->isSVEorStreamingSVEAvailable() &&
7179 if (IsLoad && LT.second.isVector() &&
7181 LT.second.getVectorElementCount()))
7187 if (Mask.size() == 4 &&
7189 (SrcTy->getScalarSizeInBits() == 16 ||
7190 SrcTy->getScalarSizeInBits() == 32) &&
7191 all_of(Mask, [](
int E) {
return E < 8; }))
7197 if (LT.second.isFixedLengthVector() &&
7198 LT.second.getVectorNumElements() == Mask.size() &&
7204 (
isZIPMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7205 isTRNMask(Mask, LT.second.getVectorNumElements(), Unused, Unused) ||
7206 isUZPMask(Mask, LT.second.getVectorNumElements(), Unused) ||
7207 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7208 LT.second.getVectorNumElements(), 16) ||
7209 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7210 LT.second.getVectorNumElements(), 32) ||
7211 isREVMask(Mask, LT.second.getScalarSizeInBits(),
7212 LT.second.getVectorNumElements(), 64) ||
7215 [&Mask](
int M) {
return M < 0 || M == Mask[0]; })))
7344 return LT.first * Entry->Cost;
7353 LT.second.getSizeInBits() <= 128 && SubTp) {
7355 if (SubLT.second.isVector()) {
7356 int NumElts = LT.second.getVectorNumElements();
7357 int NumSubElts = SubLT.second.getVectorNumElements();
7358 if ((Index % NumSubElts) == 0 && (NumElts % NumSubElts) == 0)
7364 if (IsExtractSubvector)
7385 if (
getPtrStride(*PSE, AccessTy, Ptr, TheLoop, DT, Strides,
7398 return ST->useFixedOverScalableIfEqualCost();
7402 return ST->getEpilogueVectorizationMinVF();
7437 unsigned NumInsns = 0;
7439 NumInsns += BB->size();
7449 int64_t Scale,
unsigned AddrSpace)
const {
7477 if (
I->getOpcode() == Instruction::Or &&
7481 if (
I->getOpcode() == Instruction::Add ||
7482 I->getOpcode() == Instruction::Sub)
7507 return all_equal(Shuf->getShuffleMask());
7514 bool AllowSplat =
false) {
7519 auto areTypesHalfed = [](
Value *FullV,
Value *HalfV) {
7520 auto *FullTy = FullV->
getType();
7521 auto *HalfTy = HalfV->getType();
7523 2 * HalfTy->getPrimitiveSizeInBits().getFixedValue();
7526 auto extractHalf = [](
Value *FullV,
Value *HalfV) {
7529 return FullVT->getNumElements() == 2 * HalfVT->getNumElements();
7533 Value *S1Op1 =
nullptr, *S2Op1 =
nullptr;
7547 if ((S1Op1 && (!areTypesHalfed(S1Op1, Op1) || !extractHalf(S1Op1, Op1))) ||
7548 (S2Op1 && (!areTypesHalfed(S2Op1, Op2) || !extractHalf(S2Op1, Op2))))
7562 if ((M1Start != 0 && M1Start != (NumElements / 2)) ||
7563 (M2Start != 0 && M2Start != (NumElements / 2)))
7565 if (S1Op1 && S2Op1 && M1Start != M2Start)
7575 return Ext->getType()->getScalarSizeInBits() ==
7576 2 * Ext->getOperand(0)->getType()->getScalarSizeInBits();
7590 Value *VectorOperand =
nullptr;
7607 if (!
GEP ||
GEP->getNumOperands() != 2)
7611 Value *Offsets =
GEP->getOperand(1);
7614 if (
Base->getType()->isVectorTy() || !Offsets->getType()->isVectorTy())
7620 if (OffsetsInst->getType()->getScalarSizeInBits() > 32 &&
7621 OffsetsInst->getOperand(0)->getType()->getScalarSizeInBits() <= 32)
7622 Ops.push_back(&
GEP->getOperandUse(1));
7658 switch (
II->getIntrinsicID()) {
7659 case Intrinsic::aarch64_neon_smull:
7660 case Intrinsic::aarch64_neon_umull:
7663 Ops.push_back(&
II->getOperandUse(0));
7664 Ops.push_back(&
II->getOperandUse(1));
7669 case Intrinsic::fma:
7670 case Intrinsic::fmuladd:
7677 Ops.push_back(&
II->getOperandUse(0));
7679 Ops.push_back(&
II->getOperandUse(1));
7682 case Intrinsic::aarch64_neon_sqdmull:
7683 case Intrinsic::aarch64_neon_sqdmulh:
7684 case Intrinsic::aarch64_neon_sqrdmulh:
7687 Ops.push_back(&
II->getOperandUse(0));
7689 Ops.push_back(&
II->getOperandUse(1));
7690 return !
Ops.empty();
7691 case Intrinsic::aarch64_neon_fmlal:
7692 case Intrinsic::aarch64_neon_fmlal2:
7693 case Intrinsic::aarch64_neon_fmlsl:
7694 case Intrinsic::aarch64_neon_fmlsl2:
7697 Ops.push_back(&
II->getOperandUse(1));
7699 Ops.push_back(&
II->getOperandUse(2));
7700 return !
Ops.empty();
7701 case Intrinsic::aarch64_sve_ptest_first:
7702 case Intrinsic::aarch64_sve_ptest_last:
7704 if (IIOp->getIntrinsicID() == Intrinsic::aarch64_sve_ptrue)
7705 Ops.push_back(&
II->getOperandUse(0));
7706 return !
Ops.empty();
7707 case Intrinsic::aarch64_sme_write_horiz:
7708 case Intrinsic::aarch64_sme_write_vert:
7709 case Intrinsic::aarch64_sme_writeq_horiz:
7710 case Intrinsic::aarch64_sme_writeq_vert: {
7712 if (!Idx || Idx->getOpcode() != Instruction::Add)
7714 Ops.push_back(&
II->getOperandUse(1));
7717 case Intrinsic::aarch64_sme_read_horiz:
7718 case Intrinsic::aarch64_sme_read_vert:
7719 case Intrinsic::aarch64_sme_readq_horiz:
7720 case Intrinsic::aarch64_sme_readq_vert:
7721 case Intrinsic::aarch64_sme_ld1b_vert:
7722 case Intrinsic::aarch64_sme_ld1h_vert:
7723 case Intrinsic::aarch64_sme_ld1w_vert:
7724 case Intrinsic::aarch64_sme_ld1d_vert:
7725 case Intrinsic::aarch64_sme_ld1q_vert:
7726 case Intrinsic::aarch64_sme_st1b_vert:
7727 case Intrinsic::aarch64_sme_st1h_vert:
7728 case Intrinsic::aarch64_sme_st1w_vert:
7729 case Intrinsic::aarch64_sme_st1d_vert:
7730 case Intrinsic::aarch64_sme_st1q_vert:
7731 case Intrinsic::aarch64_sme_ld1b_horiz:
7732 case Intrinsic::aarch64_sme_ld1h_horiz:
7733 case Intrinsic::aarch64_sme_ld1w_horiz:
7734 case Intrinsic::aarch64_sme_ld1d_horiz:
7735 case Intrinsic::aarch64_sme_ld1q_horiz:
7736 case Intrinsic::aarch64_sme_st1b_horiz:
7737 case Intrinsic::aarch64_sme_st1h_horiz:
7738 case Intrinsic::aarch64_sme_st1w_horiz:
7739 case Intrinsic::aarch64_sme_st1d_horiz:
7740 case Intrinsic::aarch64_sme_st1q_horiz: {
7742 if (!Idx || Idx->getOpcode() != Instruction::Add)
7744 Ops.push_back(&
II->getOperandUse(3));
7747 case Intrinsic::aarch64_neon_pmull:
7750 Ops.push_back(&
II->getOperandUse(0));
7751 Ops.push_back(&
II->getOperandUse(1));
7753 case Intrinsic::aarch64_neon_pmull64:
7755 II->getArgOperand(1)))
7757 Ops.push_back(&
II->getArgOperandUse(0));
7758 Ops.push_back(&
II->getArgOperandUse(1));
7760 case Intrinsic::masked_gather:
7763 Ops.push_back(&
II->getArgOperandUse(0));
7765 case Intrinsic::masked_scatter:
7768 Ops.push_back(&
II->getArgOperandUse(1));
7775 auto ShouldSinkCondition = [](
Value *
Cond,
7780 if (
II->getIntrinsicID() != Intrinsic::vector_reduce_or ||
7784 Ops.push_back(&
II->getOperandUse(0));
7788 switch (
I->getOpcode()) {
7789 case Instruction::GetElementPtr:
7790 case Instruction::Add:
7791 case Instruction::Sub:
7793 for (
unsigned Op = 0;
Op <
I->getNumOperands(); ++
Op) {
7795 Ops.push_back(&
I->getOperandUse(
Op));
7800 case Instruction::Select: {
7801 if (!ShouldSinkCondition(
I->getOperand(0),
Ops))
7804 Ops.push_back(&
I->getOperandUse(0));
7807 case Instruction::UncondBr:
7809 case Instruction::CondBr: {
7813 Ops.push_back(&
I->getOperandUse(0));
7816 case Instruction::FMul:
7821 Ops.push_back(&
I->getOperandUse(0));
7823 Ops.push_back(&
I->getOperandUse(1));
7833 case Instruction::Xor:
7836 if (
I->getType()->isVectorTy() && ST->isNeonAvailable()) {
7838 ST->isSVEorStreamingSVEAvailable() && (ST->hasSVE2() || ST->hasSME());
7843 case Instruction::And:
7844 case Instruction::Or:
7847 if (
I->getOpcode() == Instruction::Or &&
7852 if (!(
I->getType()->isVectorTy() && ST->hasNEON()) &&
7855 for (
auto &
Op :
I->operands()) {
7867 Ops.push_back(&Not);
7868 Ops.push_back(&InsertElt);
7878 if (!
I->getType()->isVectorTy())
7879 return !
Ops.empty();
7881 switch (
I->getOpcode()) {
7882 case Instruction::Sub:
7883 case Instruction::Add: {
7892 Ops.push_back(&Ext1->getOperandUse(0));
7893 Ops.push_back(&Ext2->getOperandUse(0));
7896 Ops.push_back(&
I->getOperandUse(0));
7897 Ops.push_back(&
I->getOperandUse(1));
7901 case Instruction::Or: {
7904 if (ST->hasNEON()) {
7918 if (
I->getParent() != MainAnd->
getParent() ||
7923 if (
I->getParent() != IA->getParent() ||
7924 I->getParent() != IB->getParent())
7929 Ops.push_back(&
I->getOperandUse(0));
7930 Ops.push_back(&
I->getOperandUse(1));
7939 case Instruction::Mul: {
7940 auto ShouldSinkSplatForIndexedVariant = [](
Value *V) {
7943 if (Ty->isScalableTy())
7947 return Ty->getScalarSizeInBits() == 16 || Ty->getScalarSizeInBits() == 32;
7950 int NumZExts = 0, NumSExts = 0;
7951 for (
auto &
Op :
I->operands()) {
7958 auto *ExtOp = Ext->getOperand(0);
7959 if (
isSplatShuffle(ExtOp) && ShouldSinkSplatForIndexedVariant(ExtOp))
7960 Ops.push_back(&Ext->getOperandUse(0));
7968 if (Ext->getOperand(0)->getType()->getScalarSizeInBits() * 2 <
7969 I->getType()->getScalarSizeInBits())
8006 if (!ElementConstant || !ElementConstant->
isZero())
8009 unsigned Opcode = OperandInstr->
getOpcode();
8010 if (Opcode == Instruction::SExt)
8012 else if (Opcode == Instruction::ZExt)
8017 unsigned Bitwidth =
I->getType()->getScalarSizeInBits();
8027 Ops.push_back(&Insert->getOperandUse(1));
8033 if (!
Ops.empty() && (NumSExts == 2 || NumZExts == 2))
8037 if (!ShouldSinkSplatForIndexedVariant(
I))
8042 Ops.push_back(&
I->getOperandUse(0));
8044 Ops.push_back(&
I->getOperandUse(1));
8046 return !
Ops.empty();
8048 case Instruction::FMul: {
8050 if (
I->getType()->isScalableTy())
8051 return !
Ops.empty();
8055 return !
Ops.empty();
8059 Ops.push_back(&
I->getOperandUse(0));
8061 Ops.push_back(&
I->getOperandUse(1));
8062 return !
Ops.empty();
8071 Align Alignment)
const {
8072 if (!(ST->isSVEAvailable() ||
8073 (ST->isSVEorStreamingSVEAvailable() && ST->hasSME2p2())))
8077 DataType->getPrimitiveSizeInBits().getFixedValue() < 128)
8084 if (!LT.first.isValid())
8089 switch (LT.second.SimpleTy) {
static bool isAllActivePredicate(const SelectionDAG &DAG, SDValue N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static Error reportError(StringRef Message)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
This file defines the DenseMap class.
static Value * getCondition(Instruction *I)
const HexagonInstrInfo * TII
This file provides the interface for the instcombine pass implementation.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file defines the LoopVectorizationLegality class.
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > & Cond
static uint64_t getBits(uint64_t Val, int Start, int End)
static unsigned getFastMathFlags(const MachineInstr &I, const SPIRVSubtarget &ST)
static SymbolRef::Type getType(const Symbol *Sym)
This file describes how to lower LLVM code to machine code.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
This file implements the C++20 <bit> header.
unsigned getVectorInsertExtractBaseCost() const
bool useSVEForFixedLengthVectors() const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
bool isExtPartOfAvgExpr(const Instruction *ExtUser, Type *Dst, Type *Src) const
InstructionCost getIntImmCost(int64_t Val) const
Calculate the cost of materializing a 64-bit value.
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index) const override
std::optional< InstructionCost > getFP16BF16PromoteCost(Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, bool IncludeTrunc, bool CanUseSVE, std::function< InstructionCost(Type *)> InstCost) const
FP16 and BF16 operations are lowered to fptrunc(op(fpext, fpext) if the architecture features are not...
bool prefersVectorizedAddressing() const override
bool preferFixedOverScalableIfEqualCost() const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind=TTI::TCK_RecipThroughput) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
bool isElementTypeLegalForScalableVector(Type *Ty) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
APInt getPriorityMask(const Function &F) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool shouldMaximizeVectorBandwidth(TargetTransformInfo::RegisterKind K) const override
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
bool useNeonVector(const Type *Ty) const
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
bool isLegalSpeculativeLoad(Type *DataType, unsigned AddressSpace) const override
bool isLegalMaskedExpandLoad(Type *DataTy, Align Alignment) const override
TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
bool isElementTypeLegalForCompressStore(Type *Ty) const
InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const override
unsigned getInlineCallPenalty(const Function *F, const CallBase &Call, unsigned DefaultCallPenalty) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
unsigned getMaxNumElements(ElementCount VF) const
Try to return an estimate cost factor that can be used as a multiplier when scalarizing an operation ...
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
bool isMultiversionedFunction(const Function &F) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool isLegalToVectorizeReduction(const RecurrenceDescriptor &RdxDesc, ElementCount VF) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
bool isLegalMaskedCompressStore(Type *DataType, Align Alignment) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
bool isLegalMaskedGatherScatter(Type *DataType) const
InstructionCost getBranchMispredictPenalty() const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
APInt getFeatureMask(const Function &F) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool areTypesABICompatible(const Function *Caller, const Function *Callee, ArrayRef< Type * > Types) const override
bool enableScalableVectorization() const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Value * getOrCreateResultFromMemIntrinsic(IntrinsicInst *Inst, Type *ExpectedType, bool CanCreate=true) const override
bool hasKnownLowerThroughputFromSchedulingModel(unsigned Opcode1, unsigned Opcode2) const
Check whether Opcode1 has less throughput according to the scheduling model than Opcode2.
unsigned getEpilogueVectorizationMinVF() const override
InstructionCost getSpliceCost(VectorType *Tp, int Index, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCostSVE(unsigned Opcode, VectorType *ValTy, TTI::TargetCostKind CostKind) const
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
Return the cost of the scaling factor used in the addressing mode represented by AM for this target,...
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
Class for arbitrary precision integers.
bool isNegatedPowerOf2() const
Check if this APInt's negated value is a power of two greater than zero.
uint64_t getZExtValue() const
Get zero extended value.
unsigned popcount() const
Count the number of bits set.
void negate()
Negate this APInt in place.
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
unsigned logBase2() const
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
int64_t getSExtValue() const
Get sign extended value.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
LLVM Basic Block Representation.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isTypeLegal(Type *Ty) const override
static BinaryOperator * CreateWithCopiedFlags(BinaryOps Opc, Value *V1, Value *V2, Value *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
@ ICMP_UGE
unsigned greater or equal
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
@ ICMP_ULT
unsigned less than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
@ ICMP_SGE
signed greater or equal
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
static bool isFPPredicate(Predicate P)
static bool isIntPredicate(Predicate P)
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
This is the shared class of boolean and integer constants.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
const APInt & getValue() const
Return the constant as an APInt value reference.
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
This is an important base class in LLVM.
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeSizeInBits(Type *Ty) const
Size examples:
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
static constexpr ElementCount getFixed(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
static bool isCommutative(Predicate Pred)
This provides a helper for copying FMF from an instruction or setting specified flags.
Convenience struct for specifying and reasoning about fast-math flags.
bool noSignedZeros() const
bool allowContract() const
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
static bool isCommutative(Predicate P)
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
IntegerType * getIntNTy(unsigned N)
Fetch the type representing an N-bit integer.
Type * getDoubleTy()
Fetch the type representing a 64-bit floating point value.
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
LLVM_ABI CallInst * CreateMaskedLoad(Type *Ty, Value *Ptr, Align Alignment, Value *Mask, Value *PassThru=nullptr, const Twine &Name="")
Create a call to Masked Load intrinsic.
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Type * getHalfTy()
Fetch the type representing a 16-bit floating point value.
Value * CreateGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Value * CreateLogicalAnd(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateBitOrPointerCast(Value *V, Type *DestTy, const Twine &Name="")
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
Value * CreateBinOpFMF(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, FMFSource FMFSource, const Twine &Name="", MDNode *FPMathTag=nullptr)
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
LLVM_ABI CallInst * CreateMaskedStore(Value *Val, Value *Ptr, Align Alignment, Value *Mask)
Create a call to Masked Store intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Type * getFloatTy()
Fetch the type representing a 32-bit floating point value.
Value * CreateIntCast(Value *V, Type *DestTy, bool isSigned, const Twine &Name="")
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateInsertVector(Type *DstType, Value *SrcVec, Value *SubVec, Value *Idx, const Twine &Name="")
Create a call to the vector.insert intrinsic.
LLVM_ABI Value * CreateElementCount(Type *Ty, ElementCount EC)
Create an expression which evaluates to the number of elements in EC at runtime.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
This instruction inserts a single (scalar) element into a VectorType value.
The core instruction combiner logic.
virtual Instruction * eraseInstFromFunction(Instruction &I)=0
Combiner aware instruction erasure.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
user_iterator user_begin()
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
bool hasGroups() const
Returns true if we have any interleave groups.
const SmallVectorImpl< Type * > & getArgTypes() const
Type * getReturnType() const
const SmallVectorImpl< const Value * > & getArgs() const
const IntrinsicInst * getInst() const
Intrinsic::ID getID() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
Value * getPointerOperand()
iterator_range< block_iterator > blocks() const
RecurrenceSet & getFixedOrderRecurrences()
Return the fixed-order recurrences found in the loop.
DominatorTree * getDominatorTree() const
PredicatedScalarEvolution * getPredicatedScalarEvolution() const
const ReductionList & getReductionVars() const
Returns the reduction variables found in the loop.
Represents a single loop in the control flow graph.
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
bool isFixedLengthVector() const
MVT getVectorElementType() const
Information for memory intrinsic cost model.
Align getAlignment() const
Type * getDataType() const
Intrinsic::ID getID() const
const Instruction * getInst() const
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
Type * getRecurrenceType() const
Returns the type of the recurrence.
RecurKind getRecurrenceKind() const
This node represents a polynomial recurrence on the trip count of the specified loop.
bool isAffine() const
Return true if this represents an expression A + B*x where A and B are loop invariant values.
This class represents an analyzed expression in the program.
SMEAttrs is a utility class to parse the SME ACLE attributes on functions.
bool hasStreamingCompatibleInterface() const
bool hasStreamingInterfaceOrBody() const
bool isSMEABIRoutine() const
SMECallAttrs is a utility class to hold the SMEAttrs for a callsite.
bool requiresSMChange() const
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
static ScalableVectorType * getDoubleElementsVectorType(ScalableVectorType *VTy)
The main scalar evolution driver.
LLVM_ABI const SCEV * getBackedgeTakenCount(const Loop *L, ExitCountKind Kind=Exact)
If the specified loop has a predictable backedge-taken count, return it, otherwise return a SCEVCould...
LLVM_ABI unsigned getSmallConstantTripMultiple(const Loop *L, const SCEV *ExitCount)
Returns the largest constant divisor of the trip count as a normal unsigned value,...
LLVM_ABI const SCEV * getSCEV(Value *V)
Return a SCEV expression for the full generality of the specified expression.
LLVM_ABI unsigned getSmallConstantMaxTripCount(const Loop *L, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Returns the upper bound of the loop trip count as a normal unsigned value.
LLVM_ABI bool isBackedgeTakenCountMaxOrZero(const Loop *L)
Return true if the backedge taken count is either the value returned by getConstantMaxBackedgeTakenCo...
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
const SCEV * getSymbolicMaxBackedgeTakenCount(const Loop *L)
When successful, this returns a SCEV that is greater than or equal to (i.e.
This instruction constructs a fixed permutation of two input vectors.
static LLVM_ABI bool isDeInterleaveMaskOfFactor(ArrayRef< int > Mask, unsigned Factor, unsigned &Index)
Check if the mask is a DE-interleave mask of the given factor Factor like: <Index,...
static LLVM_ABI bool isExtractSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is an extract subvector mask.
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator insert(iterator I, T &&Elt)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
static StackOffset getScalable(int64_t Scalable)
static StackOffset getFixed(int64_t Fixed)
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Class to represent struct types.
TargetInstrInfo - Interface to description of machine instruction set.
std::pair< LegalizeTypeAction, EVT > LegalizeKind
LegalizeKind holds the legalization kind that needs to happen to EVT in order to type-legalize it.
const RTLIB::RuntimeLibcallsInfo & getRuntimeLibcallsInfo() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
bool isVectorTy() const
True if this is an instance of VectorType.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static VectorType * getInteger(VectorType *VTy)
This static method gets a VectorType with the same number of elements as the input type,...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
LLVM_ABI APInt getCpuSupportsMask(ArrayRef< StringRef > Features)
static constexpr unsigned SVEBitsPerBlock
LLVM_ABI APInt getFMVPriority(ArrayRef< StringRef > Features)
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
@ ADD
Simple integer binary arithmetic operators.
@ CTTZ_ELTS
Returns the number of number of trailing (least significant) zero elements in a vector.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
@ FADD
Simple binary floating point operators.
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ SHL
Shift and rotation operations.
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
@ AND
Bitwise operators - logical and, logical or, logical xor.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
CheckType m_SpecificType(LLT Ty)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
cst_pred_ty< is_nonnegative > m_NonNegative()
Match an integer or vector of non-negative values.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_VScale()
Matches a call to llvm.vscale().
OneOps_match< OpTy, Instruction::Load > m_Load(const OpTy &Op)
Matches LoadInst.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
AnyBinaryOp_match< LHS, RHS, true > m_c_BinOp(const LHS &L, const RHS &R)
Matches a BinaryOperator with LHS and RHS in either order.
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
brc_match< Cond_t, match_bind< BasicBlock >, match_bind< BasicBlock > > m_Br(const Cond_t &C, BasicBlock *&T, BasicBlock *&F)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
initializer< Ty > init(const Ty &Val)
LocationClass< Ty > location(Ty &L)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
std::optional< unsigned > isDUPQMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPQMask - matches a splat of equivalent lanes within segments of a given number of elements.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
bool isZIPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for zip1 or zip2 masks of the form: <0, 8, 1, 9, 2, 10, 3, 11> (WhichResultOut = 0,...
TailFoldingOpts
An enum to describe what types of loops we should attempt to tail-fold: Disabled: None Reductions: Lo...
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
bool isDUPFirstSegmentMask(ArrayRef< int > Mask, unsigned Segments, unsigned SegmentSize)
isDUPFirstSegmentMask - matches a splat of the first 128b segment.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::optional< const MDOperand * > findStringMetadataForLoop(const Loop *TheLoop, StringRef Name)
Find string metadata for loop.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
LLVM_ABI std::optional< int64_t > getPtrStride(PredicatedScalarEvolution &PSE, Type *AccessTy, Value *Ptr, const Loop *Lp, const DominatorTree &DT, const SymbolicStrideMap &StridesMap=SymbolicStrideMap(), bool ShouldCheckWrap=true, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
If the pointer has a constant stride return it in units of the access type size.
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
LLVM_ABI bool MaskedValueIsZero(const Value *V, const APInt &Mask, const SimplifyQuery &SQ, unsigned Depth=0)
Return true if 'V & Mask' is known to be zero.
unsigned M1(unsigned Val)
auto dyn_cast_or_null(const Y &Val)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
unsigned getPerfectShuffleCost(llvm::ArrayRef< int > M)
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
bool isUZPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut)
Return true for uzp1 or uzp2 masks of the form: <0, 2, 4, 6, 8, 10, 12, 14> or <1,...
bool isREVMask(ArrayRef< int > M, unsigned EltSize, unsigned NumElts, unsigned BlockSize)
isREVMask - Check if a vector shuffle corresponds to a REV instruction with the specified blocksize.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ Or
Bitwise or logical OR of integers.
@ FSub
Subtraction of floats.
@ FAddChainWithSubs
A chain of fadds and fsubs.
@ AnyOf
AnyOf reduction with select(cmp(),x,y) where one of (x,y) is loop invariant, and both x and y are int...
@ Xor
Bitwise or logical XOR of integers.
@ FindLast
FindLast reduction with select(cmp(),x,y) where x and y.
@ FMax
FP max implemented in terms of select(cmp()).
@ FMulAdd
Sum of float products with llvm.fmuladd(a * b + sum).
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ FMin
FP min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
DWARFExpression::Operation Op
TypeConversionCostTblEntryT< uint16_t > TypeConversionCostTblEntry
CostTblEntryT< uint16_t > CostTblEntry
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
unsigned getNumElementsFromSVEPredPattern(unsigned Pattern)
Return the number of active elements for VL1 to VL256 predicate pattern, zero for all other patterns.
auto predecessors(const MachineBasicBlock *BB)
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
LLVM_ABI Value * simplifyCmpInst(CmpPredicate Predicate, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a CmpInst, fold the result or return null.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
const TypeConversionCostTblEntryT< CostType > * ConvertCostTableLookup(ArrayRef< TypeConversionCostTblEntryT< CostType > > Tbl, int ISD, MVT Dst, MVT Src)
Find in type conversion cost table.
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
bool isTRNMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for trn1 or trn2 masks of the form: <0, 8, 2, 10, 4, 12, 6, 14> (WhichResultOut = 0,...
unsigned getMatchingIROpode() const
bool inactiveLanesAreUnused() const
bool inactiveLanesAreNotDefined() const
bool hasMatchingUndefIntrinsic() const
static SVEIntrinsicInfo defaultMergingUnaryNarrowingTopOp()
static SVEIntrinsicInfo defaultZeroingOp()
bool hasGoverningPredicate() const
SVEIntrinsicInfo & setOperandIdxInactiveLanesTakenFrom(unsigned Index)
static SVEIntrinsicInfo defaultMergingOp(Intrinsic::ID IID=Intrinsic::not_intrinsic)
SVEIntrinsicInfo & setOperandIdxWithNoActiveLanes(unsigned Index)
unsigned getOperandIdxWithNoActiveLanes() const
CmpInst::Predicate getCmpPredicate() const
SVEIntrinsicInfo & setInactiveLanesAreUnused()
SVEIntrinsicInfo & setInactiveLanesAreNotDefined()
SVEIntrinsicInfo & setGoverningPredicateOperandIdx(unsigned Index)
bool inactiveLanesTakenFromOperand() const
static SVEIntrinsicInfo defaultUndefOp()
bool hasOperandWithNoActiveLanes() const
Intrinsic::ID getMatchingUndefIntrinsic() const
SVEIntrinsicInfo & setResultIsZeroInitialized()
bool hasCmpPredicate() const
static SVEIntrinsicInfo defaultMergingUnaryOp()
SVEIntrinsicInfo & setMatchingUndefIntrinsic(Intrinsic::ID IID)
unsigned getGoverningPredicateOperandIdx() const
bool hasMatchingIROpode() const
SVEIntrinsicInfo & setCmpPredicate(CmpInst::Predicate Pred)
bool resultIsZeroInitialized() const
SVEIntrinsicInfo & setMatchingIROpcode(unsigned Opcode)
unsigned getOperandIdxInactiveLanesTakenFrom() const
static SVEIntrinsicInfo defaultVoidOp(unsigned GPIndex)
This struct is a compact representation of a valid (non-zero power of two) alignment.
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
bool isFixedLengthVector() const
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Summarize the scheduling resources required for an instruction of a particular scheduling class.
Machine model for scheduling, bundling, and heuristics.
static LLVM_ABI double getReciprocalThroughput(const MCSubtargetInfo &STI, const MCSchedClassDesc &SCDesc)
Information about a load/store intrinsic defined by the target.
InterleavedAccessInfo * IAI
LoopVectorizationLegality * LVL
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...