LLVM 24.0.0git
RISCVTargetTransformInfo.cpp
Go to the documentation of this file.
1//===-- RISCVTargetTransformInfo.cpp - RISC-V specific TTI ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
11#include "llvm/ADT/STLExtras.h"
18#include "llvm/IR/IntrinsicsRISCV.h"
21#include <cmath>
22#include <optional>
23using namespace llvm;
24using namespace llvm::PatternMatch;
25
26#define DEBUG_TYPE "riscvtti"
27
29 "riscv-v-register-bit-width-lmul",
31 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
32 "by autovectorized code. Fractional LMULs are not supported."),
34
36 "riscv-v-slp-max-vf",
38 "Overrides result used for getMaximumVF query which is used "
39 "exclusively by SLP vectorizer."),
41
43 RVVMinTripCount("riscv-v-min-trip-count",
44 cl::desc("Set the lower bound of a trip count to decide on "
45 "vectorization while tail-folding."),
47
48static cl::opt<bool> EnableOrLikeSelectOpt("enable-riscv-or-like-select",
49 cl::init(true), cl::Hidden);
50
52RISCVTTIImpl::getRISCVInstructionCost(ArrayRef<unsigned> OpCodes, MVT VT,
54 // Check if the type is valid for all CostKind
55 if (!VT.isVector())
57 size_t NumInstr = OpCodes.size();
59 return NumInstr;
60 InstructionCost LMULCost = TLI->getLMULCost(VT);
62 return LMULCost * NumInstr;
63 InstructionCost Cost = 0;
64 for (auto Op : OpCodes) {
65 switch (Op) {
66 case RISCV::VRGATHER_VI:
67 Cost += TLI->getVRGatherVICost(VT);
68 break;
69 case RISCV::VRGATHER_VV:
70 Cost += TLI->getVRGatherVVCost(VT);
71 break;
72 case RISCV::VSLIDEUP_VI:
73 case RISCV::VSLIDEDOWN_VI:
74 Cost += TLI->getVSlideVICost(VT);
75 break;
76 case RISCV::VSLIDEUP_VX:
77 case RISCV::VSLIDEDOWN_VX:
78 Cost += TLI->getVSlideVXCost(VT);
79 break;
80 case RISCV::VREDMAX_VS:
81 case RISCV::VREDMIN_VS:
82 case RISCV::VREDMAXU_VS:
83 case RISCV::VREDMINU_VS:
84 case RISCV::VREDSUM_VS:
85 case RISCV::VREDAND_VS:
86 case RISCV::VREDOR_VS:
87 case RISCV::VREDXOR_VS:
88 case RISCV::VFREDMAX_VS:
89 case RISCV::VFREDMIN_VS:
90 case RISCV::VFREDUSUM_VS: {
91 unsigned VL = VT.getVectorMinNumElements();
92 if (!VT.isFixedLengthVector())
93 VL *= *getVScaleForTuning();
94 Cost += Log2_32_Ceil(VL);
95 break;
96 }
97 case RISCV::VFREDOSUM_VS: {
98 unsigned VL = VT.getVectorMinNumElements();
99 if (!VT.isFixedLengthVector())
100 VL *= *getVScaleForTuning();
101 Cost += VL;
102 break;
103 }
104 case RISCV::VMV_X_S:
105 case RISCV::VFMV_F_S:
106 // Domain crossings from vector -> scalar are usually more expensive.
107 Cost += 2;
108 break;
109 case RISCV::VMV_S_X:
110 case RISCV::VFMV_S_F:
111 case RISCV::VMOR_MM:
112 case RISCV::VMXOR_MM:
113 case RISCV::VMAND_MM:
114 case RISCV::VMANDN_MM:
115 case RISCV::VMNAND_MM:
116 case RISCV::VCPOP_M:
117 case RISCV::VFIRST_M:
118 Cost += 1;
119 break;
120 case RISCV::VDIV_VV:
121 case RISCV::VREM_VV:
122 Cost += LMULCost * TTI::TCC_Expensive;
123 break;
124 default:
125 Cost += LMULCost;
126 }
127 }
128 return Cost;
129}
130
132 const RISCVSubtarget *ST,
133 const APInt &Imm, Type *Ty,
135 bool FreeZeroes) {
136 assert(Ty->isIntegerTy() &&
137 "getIntImmCost can only estimate cost of materialising integers");
138
139 // We have a Zero register, so 0 is always free.
140 if (Imm == 0)
141 return TTI::TCC_Free;
142
143 // Otherwise, we check how many instructions it will take to materialise.
144 return RISCVMatInt::getIntMatCost(Imm, DL.getTypeSizeInBits(Ty), *ST,
145 /*CompressionCost=*/false, FreeZeroes);
146}
147
151 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind, false);
152}
153
154// Look for patterns of shift followed by AND that can be turned into a pair of
155// shifts. We won't need to materialize an immediate for the AND so these can
156// be considered free.
157static bool canUseShiftPair(Instruction *Inst, const APInt &Imm) {
158 uint64_t Mask = Imm.getZExtValue();
159 auto *BO = dyn_cast<BinaryOperator>(Inst->getOperand(0));
160 if (!BO || !BO->hasOneUse())
161 return false;
162
163 if (BO->getOpcode() != Instruction::Shl)
164 return false;
165
166 if (!isa<ConstantInt>(BO->getOperand(1)))
167 return false;
168
169 unsigned ShAmt = cast<ConstantInt>(BO->getOperand(1))->getZExtValue();
170 // (and (shl x, c2), c1) will be matched to (srli (slli x, c2+c3), c3) if c1
171 // is a mask shifted by c2 bits with c3 leading zeros.
172 if (isShiftedMask_64(Mask)) {
173 unsigned Trailing = llvm::countr_zero(Mask);
174 if (ShAmt == Trailing)
175 return true;
176 }
177
178 return false;
179}
180
181// If this is i64 AND is part of (X & -(1 << C1) & 0xffffffff) == C2 << C1),
182// DAGCombiner can convert this to (sraiw X, C1) == sext(C2) for RV64. On RV32,
183// the type will be split so only the lower 32 bits need to be compared using
184// (srai/srli X, C) == C2.
185static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm) {
186 if (!Inst->hasOneUse())
187 return false;
188
189 // Look for equality comparison.
190 auto *Cmp = dyn_cast<ICmpInst>(*Inst->user_begin());
191 if (!Cmp || !Cmp->isEquality())
192 return false;
193
194 // Right hand side of comparison should be a constant.
195 auto *C = dyn_cast<ConstantInt>(Cmp->getOperand(1));
196 if (!C)
197 return false;
198
199 uint64_t Mask = Imm.getZExtValue();
200
201 // Mask should be of the form -(1 << C) in the lower 32 bits.
202 if (!isUInt<32>(Mask) || !isPowerOf2_32(-uint32_t(Mask)))
203 return false;
204
205 // Comparison constant should be a subset of Mask.
206 uint64_t CmpC = C->getZExtValue();
207 if ((CmpC & Mask) != CmpC)
208 return false;
209
210 // We'll need to sign extend the comparison constant and shift it right. Make
211 // sure the new constant can use addi/xori+seqz/snez.
212 unsigned ShiftBits = llvm::countr_zero(Mask);
213 int64_t NewCmpC = SignExtend64<32>(CmpC) >> ShiftBits;
214 return NewCmpC >= -2048 && NewCmpC <= 2048;
215}
216
218 const APInt &Imm, Type *Ty,
220 Instruction *Inst) const {
221 assert(Ty->isIntegerTy() &&
222 "getIntImmCost can only estimate cost of materialising integers");
223
224 // We have a Zero register, so 0 is always free.
225 if (Imm == 0)
226 return TTI::TCC_Free;
227
228 // Some instructions in RISC-V can take a 12-bit immediate. Some of these are
229 // commutative, in others the immediate comes from a specific argument index.
230 bool Takes12BitImm = false;
231 unsigned ImmArgIdx = ~0U;
232
233 switch (Opcode) {
234 case Instruction::GetElementPtr:
235 // Never hoist any arguments to a GetElementPtr. CodeGenPrepare will
236 // split up large offsets in GEP into better parts than ConstantHoisting
237 // can.
238 return TTI::TCC_Free;
239 case Instruction::Store: {
240 // Use the materialization cost regardless of if it's the address or the
241 // value that is constant, except for if the store is misaligned and
242 // misaligned accesses are not legal (experience shows constant hoisting
243 // can sometimes be harmful in such cases).
244 if (Idx == 1 || !Inst)
245 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
246 /*FreeZeroes=*/true);
247
248 StoreInst *ST = cast<StoreInst>(Inst);
249 if (!getTLI()->allowsMemoryAccessForAlignment(
250 Ty->getContext(), DL, getTLI()->getValueType(DL, Ty),
251 ST->getPointerAddressSpace(), ST->getAlign()))
252 return TTI::TCC_Free;
253
254 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
255 /*FreeZeroes=*/true);
256 }
257 case Instruction::Load:
258 // If the address is a constant, use the materialization cost.
259 return getIntImmCost(Imm, Ty, CostKind);
260 case Instruction::And:
261 // zext.h
262 if (Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
263 return TTI::TCC_Free;
264 // zext.w
265 if (Imm == UINT64_C(0xffffffff) &&
266 ((ST->hasStdExtZba() && ST->isRV64()) || ST->isRV32()))
267 return TTI::TCC_Free;
268 // bclri
269 if (ST->hasStdExtZbs() && (~Imm).isPowerOf2())
270 return TTI::TCC_Free;
271 if (Inst && Idx == 1 && Imm.getBitWidth() <= ST->getXLen() &&
272 canUseShiftPair(Inst, Imm))
273 return TTI::TCC_Free;
274 if (Inst && Idx == 1 && Imm.getBitWidth() == 64 &&
275 canUseShiftCmp(Inst, Imm))
276 return TTI::TCC_Free;
277 Takes12BitImm = true;
278 break;
279 case Instruction::Add:
280 Takes12BitImm = true;
281 break;
282 case Instruction::Or:
283 case Instruction::Xor:
284 // bseti/binvi
285 if (ST->hasStdExtZbs() && Imm.isPowerOf2())
286 return TTI::TCC_Free;
287 Takes12BitImm = true;
288 break;
289 case Instruction::Mul:
290 // Power of 2 is a shift. Negated power of 2 is a shift and a negate.
291 if (Imm.isPowerOf2() || Imm.isNegatedPowerOf2())
292 return TTI::TCC_Free;
293 // One more or less than a power of 2 can use SLLI+ADD/SUB.
294 if ((Imm + 1).isPowerOf2() || (Imm - 1).isPowerOf2())
295 return TTI::TCC_Free;
296 // FIXME: There is no MULI instruction.
297 Takes12BitImm = true;
298 break;
299 case Instruction::Sub:
300 case Instruction::Shl:
301 case Instruction::LShr:
302 case Instruction::AShr:
303 Takes12BitImm = true;
304 ImmArgIdx = 1;
305 break;
306 default:
307 break;
308 }
309
310 if (Takes12BitImm) {
311 // Check immediate is the correct argument...
312 if (Instruction::isCommutative(Opcode) || Idx == ImmArgIdx) {
313 // ... and fits into the 12-bit immediate.
314 if (Imm.getSignificantBits() <= 64 &&
315 getTLI()->isLegalAddImmediate(Imm.getSExtValue())) {
316 return TTI::TCC_Free;
317 }
318 }
319
320 // Otherwise, use the full materialisation cost.
321 return getIntImmCost(Imm, Ty, CostKind);
322 }
323
324 // By default, prevent hoisting.
325 return TTI::TCC_Free;
326}
327
330 const APInt &Imm, Type *Ty,
332 // Prevent hoisting in unknown cases.
333 return TTI::TCC_Free;
334}
335
337 return ST->hasVInstructions();
338}
339
341RISCVTTIImpl::getPopcntSupport(unsigned TyWidth) const {
342 assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2");
343 return ST->hasCPOPLike() ? TTI::PSK_FastHardware : TTI::PSK_Software;
344}
345
347 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
349 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
350 TTI::TargetCostKind CostKind, std::optional<FastMathFlags> FMF) const {
351 if (Opcode == Instruction::FAdd)
353
354 // zve32x is broken for partial_reduce_umla, but let's make sure we
355 // don't generate them.
356 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
357 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
358 InputTypeA != InputTypeB || !InputTypeA->isIntegerTy(8) ||
359 !AccumType->isIntegerTy(32) || !VF.isKnownMultipleOf(4))
361
362 Type *Tp = VectorType::get(AccumType, VF.divideCoefficientBy(4));
363 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
364 // Note: Asuming all vdot4a* variants are equal cost
365 return LT.first *
366 getRISCVInstructionCost(RISCV::VDOT4A_VV, LT.second, CostKind);
367}
368
370 // Currently, the ExpandReductions pass can't expand scalable-vector
371 // reductions, but we still request expansion as RVV doesn't support certain
372 // reductions and the SelectionDAG can't legalize them either.
373 switch (II->getIntrinsicID()) {
374 default:
375 return false;
376 // These reductions have no equivalent in RVV
377 case Intrinsic::vector_reduce_mul:
378 case Intrinsic::vector_reduce_fmul:
379 return true;
380 }
381}
382
383std::optional<unsigned> RISCVTTIImpl::getMaxVScale() const {
384 if (ST->hasVInstructions())
385 return ST->getRealMaxVLen() / RISCV::RVVBitsPerBlock;
386 return BaseT::getMaxVScale();
387}
388
389std::optional<unsigned> RISCVTTIImpl::getVScaleForTuning() const {
390 if (ST->hasVInstructions())
391 if (unsigned MinVLen = ST->getRealMinVLen();
392 MinVLen >= RISCV::RVVBitsPerBlock)
393 return MinVLen / RISCV::RVVBitsPerBlock;
395}
396
399 unsigned LMUL =
400 llvm::bit_floor(std::clamp<unsigned>(RVVRegisterWidthLMUL, 1, 8));
401 switch (K) {
403 return TypeSize::getFixed(ST->getXLen());
405 return TypeSize::getFixed(
406 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
409 (ST->hasVInstructions() &&
410 ST->getRealMinVLen() >= RISCV::RVVBitsPerBlock)
412 : 0);
413 }
414
415 llvm_unreachable("Unsupported register kind");
416}
417
418InstructionCost RISCVTTIImpl::getStaticDataAddrGenerationCost(
419 const TTI::TargetCostKind CostKind) const {
420 switch (CostKind) {
423 // Always 2 instructions
424 return 2;
425 case TTI::TCK_Latency:
427 // Depending on the memory model the address generation will
428 // require AUIPC + ADDI (medany) or LUI + ADDI (medlow). Don't
429 // have a way of getting this information here, so conservatively
430 // require both.
431 // In practice, these are generally implemented together.
432 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
433 }
434 llvm_unreachable("Unsupported cost kind");
435}
436
438RISCVTTIImpl::getConstantPoolLoadCost(Type *Ty,
440 // Add a cost of address generation + the cost of the load. The address
441 // is expected to be a PC relative offset to a constant pool entry
442 // using auipc/addi.
443 return getStaticDataAddrGenerationCost(CostKind) +
444 getMemoryOpCost(Instruction::Load, Ty, DL.getABITypeAlign(Ty),
445 /*AddressSpace=*/0, CostKind);
446}
447
448static bool isRepeatedConcatMask(ArrayRef<int> Mask, int &SubVectorSize) {
449 unsigned Size = Mask.size();
450 if (!isPowerOf2_32(Size))
451 return false;
452 for (unsigned I = 0; I != Size; ++I) {
453 if (static_cast<unsigned>(Mask[I]) == I)
454 continue;
455 if (Mask[I] != 0)
456 return false;
457 if (Size % I != 0)
458 return false;
459 for (unsigned J = I + 1; J != Size; ++J)
460 // Check the pattern is repeated.
461 if (static_cast<unsigned>(Mask[J]) != J % I)
462 return false;
463 SubVectorSize = I;
464 return true;
465 }
466 // That means Mask is <0, 1, 2, 3>. This is not a concatenation.
467 return false;
468}
469
471 LLVMContext &C) {
472 assert((DataVT.getScalarSizeInBits() != 8 ||
473 DataVT.getVectorNumElements() <= 256) && "unhandled case in lowering");
474 MVT IndexVT = DataVT.changeTypeToInteger();
475 if (IndexVT.getScalarType().bitsGT(ST.getXLenVT()))
476 IndexVT = IndexVT.changeVectorElementType(MVT::i16);
477 return cast<VectorType>(EVT(IndexVT).getTypeForEVT(C));
478}
479
480/// Attempt to approximate the cost of a shuffle which will require splitting
481/// during legalization. Note that processShuffleMasks is not an exact proxy
482/// for the algorithm used in LegalizeVectorTypes, but hopefully it's a
483/// reasonably close upperbound.
485 MVT LegalVT, VectorType *Tp,
486 ArrayRef<int> Mask,
488 assert(LegalVT.isFixedLengthVector() && !Mask.empty() &&
489 "Expected fixed vector type and non-empty mask");
490 unsigned LegalNumElts = LegalVT.getVectorNumElements();
491 // Number of destination vectors after legalization:
492 unsigned NumOfDests = divideCeil(Mask.size(), LegalNumElts);
493 // We are going to permute multiple sources and the result will be in
494 // multiple destinations. Providing an accurate cost only for splits where
495 // the element type remains the same.
496 if (NumOfDests <= 1 ||
498 Tp->getElementType()->getPrimitiveSizeInBits() ||
499 LegalNumElts >= Tp->getElementCount().getFixedValue())
501
502 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
503 unsigned LegalVTSize = LegalVT.getStoreSize();
504 // Number of source vectors after legalization:
505 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
506
507 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(), LegalNumElts);
508
509 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
510 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
511 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
512 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
513 assert(NormalizedVF >= Mask.size() &&
514 "Normalized mask expected to be not shorter than original mask.");
515 copy(Mask, NormalizedMask.begin());
516 InstructionCost Cost = 0;
517 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
519 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
520 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
521 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
522 return;
523 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
524 .second)
525 return;
526 Cost += TTI.getShuffleCost(
528 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
529 SingleOpTy, RegMask, CostKind, 0, nullptr);
530 },
531 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
532 Cost += TTI.getShuffleCost(
534 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
535 SingleOpTy, RegMask, CostKind, 0, nullptr);
536 });
537 return Cost;
538}
539
540/// Try to perform better estimation of the permutation.
541/// 1. Split the source/destination vectors into real registers.
542/// 2. Do the mask analysis to identify which real registers are
543/// permuted. If more than 1 source registers are used for the
544/// destination register building, the cost for this destination register
545/// is (Number_of_source_register - 1) * Cost_PermuteTwoSrc. If only one
546/// source register is used, build mask and calculate the cost as a cost
547/// of PermuteSingleSrc.
548/// Also, for the single register permute we try to identify if the
549/// destination register is just a copy of the source register or the
550/// copy of the previous destination register (the cost is
551/// TTI::TCC_Basic). If the source register is just reused, the cost for
552/// this operation is 0.
553static InstructionCost
555 std::optional<unsigned> VLen, VectorType *Tp,
557 assert(LegalVT.isFixedLengthVector());
558 if (!VLen || Mask.empty())
560 MVT ElemVT = LegalVT.getVectorElementType();
561 unsigned ElemsPerVReg = *VLen / ElemVT.getFixedSizeInBits();
562 LegalVT = TTI.getTypeLegalizationCost(
563 FixedVectorType::get(Tp->getElementType(), ElemsPerVReg))
564 .second;
565 // Number of destination vectors after legalization:
566 InstructionCost NumOfDests =
567 divideCeil(Mask.size(), LegalVT.getVectorNumElements());
568 if (NumOfDests <= 1 ||
570 Tp->getElementType()->getPrimitiveSizeInBits() ||
571 LegalVT.getVectorNumElements() >= Tp->getElementCount().getFixedValue())
573
574 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
575 unsigned LegalVTSize = LegalVT.getStoreSize();
576 // Number of source vectors after legalization:
577 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
578
579 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(),
580 LegalVT.getVectorNumElements());
581
582 unsigned E = NumOfDests.getValue();
583 unsigned NormalizedVF =
584 LegalVT.getVectorNumElements() * std::max(NumOfSrcs, E);
585 unsigned NumOfSrcRegs = NormalizedVF / LegalVT.getVectorNumElements();
586 unsigned NumOfDestRegs = NormalizedVF / LegalVT.getVectorNumElements();
587 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
588 assert(NormalizedVF >= Mask.size() &&
589 "Normalized mask expected to be not shorter than original mask.");
590 copy(Mask, NormalizedMask.begin());
591 InstructionCost Cost = 0;
592 int NumShuffles = 0;
593 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
595 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
596 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
597 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
598 return;
599 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
600 .second)
601 return;
602 ++NumShuffles;
603 Cost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, SingleOpTy,
604 SingleOpTy, RegMask, CostKind, 0, nullptr);
605 },
606 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
607 Cost += TTI.getShuffleCost(TTI::SK_PermuteTwoSrc, SingleOpTy,
608 SingleOpTy, RegMask, CostKind, 0, nullptr);
609 NumShuffles += 2;
610 });
611 // Note: check that we do not emit too many shuffles here to prevent code
612 // size explosion.
613 // TODO: investigate, if it can be improved by extra analysis of the masks
614 // to check if the code is more profitable.
615 if ((NumOfDestRegs > 2 && NumShuffles <= static_cast<int>(NumOfDestRegs)) ||
616 (NumOfDestRegs <= 2 && NumShuffles < 4))
617 return Cost;
619}
620
621InstructionCost RISCVTTIImpl::getSlideCost(FixedVectorType *Tp,
622 ArrayRef<int> Mask,
624 // Avoid missing masks and length changing shuffles
625 if (Mask.size() <= 2 || Mask.size() != Tp->getNumElements())
627
628 int NumElts = Tp->getNumElements();
629 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
630 // Avoid scalarization cases
631 if (!LT.second.isFixedLengthVector())
633
634 // Requires moving elements between parts, which requires additional
635 // unmodeled instructions.
636 if (LT.first != 1)
638
639 auto GetSlideOpcode = [&](int SlideAmt) {
640 assert(SlideAmt != 0);
641 bool IsVI = isUInt<5>(std::abs(SlideAmt));
642 if (SlideAmt < 0)
643 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
644 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
645 };
646
647 std::array<std::pair<int, int>, 2> SrcInfo;
648 if (!isMaskedSlidePair(Mask, NumElts, SrcInfo))
650
651 if (SrcInfo[1].second == 0)
652 std::swap(SrcInfo[0], SrcInfo[1]);
653
654 InstructionCost FirstSlideCost = 0;
655 if (SrcInfo[0].second != 0) {
656 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
657 FirstSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
658 }
659
660 if (SrcInfo[1].first == -1)
661 return FirstSlideCost;
662
663 InstructionCost SecondSlideCost = 0;
664 if (SrcInfo[1].second != 0) {
665 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
666 SecondSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
667 } else {
668 SecondSlideCost =
669 getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second, CostKind);
670 }
671
672 auto EC = Tp->getElementCount();
673 VectorType *MaskTy =
675 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
676 return FirstSlideCost + SecondSlideCost + MaskCost;
677}
678
681 VectorType *SrcTy, ArrayRef<int> Mask,
682 TTI::TargetCostKind CostKind, int Index,
684 const Instruction *CxtI) const {
685 assert((Mask.empty() || DstTy->isScalableTy() ||
686 Mask.size() == DstTy->getElementCount().getKnownMinValue()) &&
687 "Expected the Mask to match the return size if given");
688 assert(SrcTy->getScalarType() == DstTy->getScalarType() &&
689 "Expected the same scalar types");
690
691 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
692
693 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
694 // For now, skip all fixed vector cost analysis when P extension is available
695 // to avoid crashes in getMinRVVVectorSizeInBits()
696 if (ST->hasStdExtP() && isa<FixedVectorType>(SrcTy))
697 return 1;
698
699 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
700
701 // First, handle cases where having a fixed length vector enables us to
702 // give a more accurate cost than falling back to generic scalable codegen.
703 // TODO: Each of these cases hints at a modeling gap around scalable vectors.
704 if (auto *FVTp = dyn_cast<FixedVectorType>(SrcTy);
705 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
707 *this, LT.second, ST->getRealVLen(),
708 Kind == TTI::SK_InsertSubvector ? DstTy : SrcTy, Mask, CostKind);
709 if (VRegSplittingCost.isValid())
710 return VRegSplittingCost;
711 switch (Kind) {
712 default:
713 break;
715 if (Mask.size() >= 2) {
716 MVT EltTp = LT.second.getVectorElementType();
717 // If the size of the element is < ELEN then shuffles of interleaves and
718 // deinterleaves of 2 vectors can be lowered into the following
719 // sequences
720 if (EltTp.getScalarSizeInBits() < ST->getELen()) {
721 // Example sequence:
722 // vsetivli zero, 4, e8, mf4, ta, ma (ignored)
723 // vwaddu.vv v10, v8, v9
724 // li a0, -1 (ignored)
725 // vwmaccu.vx v10, a0, v9
726 if (ShuffleVectorInst::isInterleaveMask(Mask, 2, Mask.size()))
727 return 2 * LT.first * TLI->getLMULCost(LT.second);
728
729 if (Mask[0] == 0 || Mask[0] == 1) {
730 auto DeinterleaveMask = createStrideMask(Mask[0], 2, Mask.size());
731 // Example sequence:
732 // vnsrl.wi v10, v8, 0
733 if (equal(DeinterleaveMask, Mask))
734 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
735 LT.second, CostKind);
736 }
737 }
738 int SubVectorSize;
739 if (LT.second.getScalarSizeInBits() != 1 &&
740 isRepeatedConcatMask(Mask, SubVectorSize)) {
742 unsigned NumSlides = Log2_32(Mask.size() / SubVectorSize);
743 // The cost of extraction from a subvector is 0 if the index is 0.
744 for (unsigned I = 0; I != NumSlides; ++I) {
745 unsigned InsertIndex = SubVectorSize * (1 << I);
746 FixedVectorType *SubTp =
747 FixedVectorType::get(SrcTy->getElementType(), InsertIndex);
748 FixedVectorType *DestTp =
750 std::pair<InstructionCost, MVT> DestLT =
752 // Add the cost of whole vector register move because the
753 // destination vector register group for vslideup cannot overlap the
754 // source.
755 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
756 Cost += getShuffleCost(TTI::SK_InsertSubvector, DestTp, DestTp, {},
757 CostKind, InsertIndex, SubTp);
758 }
759 return Cost;
760 }
761 }
762
763 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
764 SlideCost.isValid())
765 return SlideCost;
766
767 // vrgather + cost of generating the mask constant.
768 // We model this for an unknown mask with a single vrgather.
769 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
770 LT.second.getVectorNumElements() <= 256)) {
771 VectorType *IdxTy =
772 getVRGatherIndexType(LT.second, *ST, SrcTy->getContext());
773 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
774 return IndexCost +
775 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second, CostKind);
776 }
777 break;
778 }
781
782 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
783 SlideCost.isValid())
784 return SlideCost;
785
786 // 2 x (vrgather + cost of generating the mask constant) + cost of mask
787 // register for the second vrgather. We model this for an unknown
788 // (shuffle) mask.
789 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
790 LT.second.getVectorNumElements() <= 256)) {
791 auto &C = SrcTy->getContext();
792 auto EC = SrcTy->getElementCount();
793 VectorType *IdxTy = getVRGatherIndexType(LT.second, *ST, C);
795 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
796 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
797 return 2 * IndexCost +
798 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
799 LT.second, CostKind) +
800 MaskCost;
801 }
802 break;
803 }
804 }
805
806 auto shouldSplit = [](TTI::ShuffleKind Kind) {
807 switch (Kind) {
808 default:
809 return false;
813 return true;
814 }
815 };
816
817 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
818 shouldSplit(Kind)) {
819 InstructionCost SplitCost =
820 costShuffleViaSplitting(*this, LT.second, FVTp, Mask, CostKind);
821 if (SplitCost.isValid())
822 return SplitCost;
823 }
824 }
825
826 // Handle scalable vectors (and fixed vectors legalized to scalable vectors).
827 switch (Kind) {
828 default:
829 // Fallthrough to generic handling.
830 // TODO: Most of these cases will return getInvalid in generic code, and
831 // must be implemented here.
832 break;
834 // Extract at zero is always a subregister extract
835 if (Index == 0)
836 return TTI::TCC_Free;
837
838 // If we're extracting a subvector of at most m1 size at a sub-register
839 // boundary - which unfortunately we need exact vlen to identify - this is
840 // a subregister extract at worst and thus won't require a vslidedown.
841 // TODO: Extend for aligned m2, m4 subvector extracts
842 // TODO: Extend for misalgined (but contained) extracts
843 // TODO: Extend for scalable subvector types
844 if (std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(SubTp);
845 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
846 if (std::optional<unsigned> VLen = ST->getRealVLen();
847 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
848 SubLT.second.getSizeInBits() <= *VLen)
849 return TTI::TCC_Free;
850 }
851
852 // Example sequence:
853 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
854 // vslidedown.vi v8, v9, 2
855 return LT.first *
856 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second, CostKind);
858 // Example sequence:
859 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
860 // vslideup.vi v8, v9, 2
861 LT = getTypeLegalizationCost(DstTy);
862 return LT.first *
863 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second, CostKind);
864 case TTI::SK_Select: {
865 // Example sequence:
866 // li a0, 90
867 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
868 // vmv.s.x v0, a0
869 // vmerge.vvm v8, v9, v8, v0
870 // We use 2 for the cost of the mask materialization as this is the true
871 // cost for small masks and most shuffles are small. At worst, this cost
872 // should be a very small constant for the constant pool load. As such,
873 // we may bias towards large selects slightly more than truly warranted.
874 return LT.first *
875 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
876 LT.second, CostKind));
877 }
878 case TTI::SK_Broadcast: {
879 // Check for broadcast loads, which are synthesized by optimized zero-stride
880 // loads (this is checked in RISCVTTIImpl::isLegalBroadcastLoad).
881 bool IsLoad = !Args.empty() && isa<LoadInst>(Args[0]);
882 if (IsLoad && LT.second.isVector() &&
883 isLegalBroadcastLoad(SrcTy->getElementType(),
884 LT.second.getVectorElementCount()))
885 return 0;
886
887 bool HasScalar = (Args.size() > 0) && (Operator::getOpcode(Args[0]) ==
888 Instruction::InsertElement);
889 if (LT.second.getScalarSizeInBits() == 1) {
890 if (HasScalar) {
891 // Example sequence:
892 // andi a0, a0, 1
893 // vsetivli zero, 2, e8, mf8, ta, ma (ignored)
894 // vmv.v.x v8, a0
895 // vmsne.vi v0, v8, 0
896 return LT.first *
897 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
898 LT.second, CostKind));
899 }
900 // Example sequence:
901 // vsetivli zero, 2, e8, mf8, ta, mu (ignored)
902 // vmv.v.i v8, 0
903 // vmerge.vim v8, v8, 1, v0
904 // vmv.x.s a0, v8
905 // andi a0, a0, 1
906 // vmv.v.x v8, a0
907 // vmsne.vi v0, v8, 0
908
909 return LT.first *
910 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
911 RISCV::VMV_X_S, RISCV::VMV_V_X,
912 RISCV::VMSNE_VI},
913 LT.second, CostKind));
914 }
915
916 if (HasScalar) {
917 // Example sequence:
918 // vmv.v.x v8, a0
919 return LT.first *
920 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second, CostKind);
921 }
922
923 // Example sequence:
924 // vrgather.vi v9, v8, 0
925 return LT.first *
926 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second, CostKind);
927 }
928 case TTI::SK_Splice: {
929 // vslidedown+vslideup.
930 // TODO: Multiplying by LT.first implies this legalizes into multiple copies
931 // of similar code, but I think we expand through memory.
932 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
933 if (Index >= 0 && Index < 32)
934 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
935 else if (Index < 0 && Index > -32)
936 Opcodes[1] = RISCV::VSLIDEUP_VI;
937 return LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
938 }
939 case TTI::SK_Reverse: {
940
941 if (!LT.second.isVector())
943
944 // TODO: Cases to improve here:
945 // * Illegal vector types
946 // * i64 on RV32
947 if (SrcTy->getElementType()->isIntegerTy(1)) {
948 VectorType *WideTy =
949 VectorType::get(IntegerType::get(SrcTy->getContext(), 8),
950 cast<VectorType>(SrcTy)->getElementCount());
951 return getCastInstrCost(Instruction::ZExt, WideTy, SrcTy,
953 getShuffleCost(TTI::SK_Reverse, WideTy, WideTy, {}, CostKind, 0,
954 nullptr) +
955 getCastInstrCost(Instruction::Trunc, SrcTy, WideTy,
957 }
958
959 MVT ContainerVT = LT.second;
960 if (LT.second.isFixedLengthVector())
961 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
962 MVT M1VT = RISCVTargetLowering::getM1VT(ContainerVT);
963 if (ContainerVT.bitsLE(M1VT)) {
964 // Example sequence:
965 // csrr a0, vlenb
966 // srli a0, a0, 3
967 // addi a0, a0, -1
968 // vsetvli a1, zero, e8, mf8, ta, mu (ignored)
969 // vid.v v9
970 // vrsub.vx v10, v9, a0
971 // vrgather.vv v9, v8, v10
972 InstructionCost LenCost = 3;
973 if (LT.second.isFixedLengthVector())
974 // vrsub.vi has a 5 bit immediate field, otherwise an li suffices
975 LenCost = isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
976 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
977 if (LT.second.isFixedLengthVector() &&
978 isInt<5>(LT.second.getVectorNumElements() - 1))
979 Opcodes[1] = RISCV::VRSUB_VI;
980 InstructionCost GatherCost =
981 getRISCVInstructionCost(Opcodes, LT.second, CostKind);
982 return LT.first * (LenCost + GatherCost);
983 }
984
985 // At high LMUL, we split into a series of M1 reverses (see
986 // lowerVECTOR_REVERSE) and then do a single slide at the end to eliminate
987 // the resulting gap at the bottom (for fixed vectors only). The important
988 // bit is that the cost scales linearly, not quadratically with LMUL.
989 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
990 InstructionCost FixedCost =
991 getRISCVInstructionCost(M1Opcodes, M1VT, CostKind) + 3;
992 unsigned Ratio =
994 InstructionCost GatherCost =
995 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT, CostKind) * Ratio;
996 InstructionCost SlideCost = !LT.second.isFixedLengthVector() ? 0 :
997 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second, CostKind);
998 return FixedCost + LT.first * (GatherCost + SlideCost);
999 }
1000 }
1001 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, Mask, CostKind, Index,
1002 SubTp);
1003}
1004
1005static unsigned isM1OrSmaller(MVT VT) {
1007 return (LMUL == RISCVVType::VLMUL::LMUL_F8 ||
1011}
1012
1014 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
1015 TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef<Value *> VL,
1016 TTI::VectorInstrContext VIC) const {
1019
1020 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
1021 // For now, skip all fixed vector cost analysis when P extension is available
1022 // to avoid crashes in getMinRVVVectorSizeInBits()
1023 if (ST->hasStdExtP() && isa<FixedVectorType>(Ty)) {
1024 return 1; // Treat as single instruction cost for now
1025 }
1026
1027 // A build_vector (which is m1 sized or smaller) can be done in no
1028 // worse than one vslide1down.vx per element in the type. We could
1029 // in theory do an explode_vector in the inverse manner, but our
1030 // lowering today does not have a first class node for this pattern.
1032 Ty, DemandedElts, Insert, Extract, CostKind);
1033 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1034 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1035 if (Ty->getScalarSizeInBits() == 1) {
1036 auto *WideVecTy = cast<VectorType>(Ty->getWithNewBitWidth(8));
1037 // Note: Implicit scalar anyextend is assumed to be free since the i1
1038 // must be stored in a GPR.
1039 return getScalarizationOverhead(WideVecTy, DemandedElts, Insert, Extract,
1040 CostKind) +
1041 getCastInstrCost(Instruction::Trunc, Ty, WideVecTy,
1043 }
1044
1045 assert(LT.second.isFixedLengthVector());
1046 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1047 if (isM1OrSmaller(ContainerVT)) {
1048 InstructionCost BV =
1049 cast<FixedVectorType>(Ty)->getNumElements() *
1050 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second, CostKind);
1051 if (BV < Cost)
1052 Cost = BV;
1053 }
1054 }
1055 return Cost;
1056}
1057
1061 Type *DataTy = MICA.getDataType();
1062 Align Alignment = MICA.getAlignment();
1063 switch (MICA.getID()) {
1064 case Intrinsic::vp_load_ff: {
1065 EVT DataTypeVT = TLI->getValueType(DL, DataTy);
1066 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1068
1069 unsigned AS = MICA.getAddressSpace();
1070 return getMemoryOpCost(Instruction::Load, DataTy, Alignment, AS, CostKind,
1071 {TTI::OK_AnyValue, TTI::OP_None}, nullptr);
1072 }
1073 case Intrinsic::experimental_vp_strided_load:
1074 case Intrinsic::experimental_vp_strided_store:
1075 return getStridedMemoryOpCost(MICA, CostKind);
1076 case Intrinsic::masked_compressstore:
1077 case Intrinsic::masked_expandload:
1079 case Intrinsic::vp_scatter:
1080 case Intrinsic::vp_gather:
1081 case Intrinsic::masked_scatter:
1082 case Intrinsic::masked_gather:
1083 return getGatherScatterOpCost(MICA, CostKind);
1084 case Intrinsic::vp_load:
1085 case Intrinsic::vp_store:
1086 case Intrinsic::masked_load:
1087 case Intrinsic::masked_store:
1088 return getMaskedMemoryOpCost(MICA, CostKind);
1089 }
1091}
1092
1096 unsigned Opcode = MICA.getID() == Intrinsic::masked_load ? Instruction::Load
1097 : Instruction::Store;
1098 Type *Src = MICA.getDataType();
1099 Align Alignment = MICA.getAlignment();
1100 unsigned AddressSpace = MICA.getAddressSpace();
1101
1102 if (!isLegalMaskedLoadStore(Src, Alignment) ||
1105
1106 return getMemoryOpCost(Opcode, Src, Alignment, AddressSpace, CostKind);
1107}
1108
1110 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
1111 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
1112 bool UseMaskForCond, bool UseMaskForGaps) const {
1113
1114 // The interleaved memory access pass will lower (de)interleave ops combined
1115 // with an adjacent appropriate memory to vlseg/vsseg intrinsics. vlseg/vsseg
1116 // only support masking per-iteration (i.e. condition), not per-segment (i.e.
1117 // gap).
1118 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1119 auto *VTy = cast<VectorType>(VecTy);
1120 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
1121 // Need to make sure type has't been scalarized
1122 if (LT.second.isVector()) {
1124 return LT.first * TTI::TCC_Basic;
1125
1126 auto *SubVecTy =
1127 VectorType::get(VTy->getElementType(),
1128 VTy->getElementCount().divideCoefficientBy(Factor));
1129 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1130 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1131 AddressSpace, DL)) {
1132
1133 // Some processors optimize segment loads/stores as N * DLEN sized
1134 // load ops + Factor * LMUL shuffle ops.
1135 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1136 unsigned VecSizeInBits =
1137 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1138 unsigned VLENForTuning =
1140 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1141 InstructionCost Cost = divideCeil(VecSizeInBits, DLENForTuning);
1142 MVT SubVecVT = getTLI()->getValueType(DL, SubVecTy).getSimpleVT();
1143 Cost += Factor * TLI->getLMULCost(SubVecVT);
1144 return Cost;
1145 }
1146
1147 // Otherwise, the cost is proportional to the number of elements (VL *
1148 // Factor ops).
1149 unsigned NumLoads = getEstimatedVLFor(VTy);
1150 return NumLoads * TTI::TCC_Basic;
1151 }
1152 }
1153 }
1154
1155 // TODO: Return the cost of interleaved accesses for scalable vector when
1156 // unable to convert to segment accesses instructions.
1157 if (isa<ScalableVectorType>(VecTy))
1159
1160 auto *FVTy = cast<FixedVectorType>(VecTy);
1161 // When gaps are only at the tail, for interleaved load, we can emit a wide
1162 // masked load and shufflevectors. For interleaved store, we can emit
1163 // shufflevectors and a wide masked store. The interleaved memory access pass
1164 // will lower them into vlsseg/vssseg intrinsics.
1165 if (UseMaskForGaps) {
1166 assert(llvm::is_sorted(Indices) && "Indices must be sorted");
1167 assert(llvm::adjacent_find(Indices) == Indices.end() &&
1168 "Indices should not contain duplicate elements");
1169 unsigned NumOfFields = Indices.size();
1170 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.back() + 1);
1171 if (IsTailGapOnly &&
1172 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1173 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(FVTy);
1174 if (LT.second.isVector() &&
1175 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1176 auto *SubVecTy = VectorType::get(
1177 FVTy->getElementType(),
1178 FVTy->getElementCount().divideCoefficientBy(Factor));
1179 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1180 AddressSpace, DL)) {
1181 // The cost is proportional to the total number of element accesses.
1182 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1183 return NumAccesses * TTI::TCC_Basic;
1184 }
1185 }
1186 }
1187 }
1188
1189 InstructionCost MemCost =
1190 getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
1191 unsigned VF = FVTy->getNumElements() / Factor;
1192
1193 // An interleaved load will look like this for Factor=3:
1194 // %wide.vec = load <12 x i32>, ptr %3, align 4
1195 // %strided.vec = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1196 // %strided.vec1 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1197 // %strided.vec2 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1198 if (Opcode == Instruction::Load) {
1199 InstructionCost Cost = MemCost;
1200 for (unsigned Index : Indices) {
1201 FixedVectorType *VecTy =
1202 FixedVectorType::get(FVTy->getElementType(), VF * Factor);
1203 auto Mask = createStrideMask(Index, Factor, VF);
1204 Mask.resize(VF * Factor, -1);
1205 InstructionCost ShuffleCost =
1207 Mask, CostKind, 0, nullptr, {});
1208 Cost += ShuffleCost;
1209 }
1210 return Cost;
1211 }
1212
1213 // TODO: Model for NF > 2
1214 // We'll need to enhance getShuffleCost to model shuffles that are just
1215 // inserts and extracts into subvectors, since they won't have the full cost
1216 // of a vrgather.
1217 // An interleaved store for 3 vectors of 4 lanes will look like
1218 // %11 = shufflevector <4 x i32> %4, <4 x i32> %6, <8 x i32> <0...7>
1219 // %12 = shufflevector <4 x i32> %9, <4 x i32> poison, <8 x i32> <0...3>
1220 // %13 = shufflevector <8 x i32> %11, <8 x i32> %12, <12 x i32> <0...11>
1221 // %interleaved.vec = shufflevector %13, poison, <12 x i32> <interleave mask>
1222 // store <12 x i32> %interleaved.vec, ptr %10, align 4
1223 if (Factor != 2)
1224 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
1225 Alignment, AddressSpace, CostKind,
1226 UseMaskForCond, UseMaskForGaps);
1227
1228 assert(Opcode == Instruction::Store && "Opcode must be a store");
1229 // For an interleaving store of 2 vectors, we perform one large interleaving
1230 // shuffle that goes into the wide store
1231 auto Mask = createInterleaveMask(VF, Factor);
1232 InstructionCost ShuffleCost =
1234 CostKind, 0, nullptr, {});
1235 return MemCost + ShuffleCost;
1236}
1237
1241
1242 bool IsLoad = MICA.getID() == Intrinsic::masked_gather ||
1243 MICA.getID() == Intrinsic::vp_gather;
1244 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1245 Type *DataTy = MICA.getDataType();
1246 Align Alignment = MICA.getAlignment();
1249
1250 if ((Opcode == Instruction::Load &&
1251 !isLegalMaskedGather(DataTy, Align(Alignment))) ||
1252 (Opcode == Instruction::Store &&
1253 !isLegalMaskedScatter(DataTy, Align(Alignment))))
1255
1256 // Cost is proportional to the number of memory operations implied. For
1257 // scalable vectors, we use an estimate on that number since we don't
1258 // know exactly what VL will be.
1259 auto &VTy = *cast<VectorType>(DataTy);
1260 unsigned NumLoads = getEstimatedVLFor(&VTy);
1261 return NumLoads * TTI::TCC_Basic;
1262}
1263
1265 const MemIntrinsicCostAttributes &MICA,
1267 unsigned Opcode = MICA.getID() == Intrinsic::masked_expandload
1268 ? Instruction::Load
1269 : Instruction::Store;
1270 Type *DataTy = MICA.getDataType();
1271 bool VariableMask = MICA.getVariableMask();
1272 Align Alignment = MICA.getAlignment();
1273 bool IsLegal = (Opcode == Instruction::Store &&
1274 isLegalMaskedCompressStore(DataTy, Alignment)) ||
1275 (Opcode == Instruction::Load &&
1276 isLegalMaskedExpandLoad(DataTy, Alignment));
1277 if (!IsLegal || CostKind != TTI::TCK_RecipThroughput)
1279 // Example compressstore sequence:
1280 // vsetivli zero, 8, e32, m2, ta, ma (ignored)
1281 // vcompress.vm v10, v8, v0
1282 // vcpop.m a1, v0
1283 // vsetvli zero, a1, e32, m2, ta, ma
1284 // vse32.v v10, (a0)
1285 // Example expandload sequence:
1286 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
1287 // vcpop.m a1, v0
1288 // vsetvli zero, a1, e32, m2, ta, ma
1289 // vle32.v v10, (a0)
1290 // vsetivli zero, 8, e32, m2, ta, ma
1291 // viota.m v12, v0
1292 // vrgather.vv v8, v10, v12, v0.t
1293 auto MemOpCost =
1294 getMemoryOpCost(Opcode, DataTy, Alignment, /*AddressSpace*/ 0, CostKind);
1295 auto LT = getTypeLegalizationCost(DataTy);
1296 SmallVector<unsigned, 4> Opcodes{RISCV::VSETVLI};
1297 if (VariableMask)
1298 Opcodes.push_back(RISCV::VCPOP_M);
1299 if (Opcode == Instruction::Store)
1300 Opcodes.append({RISCV::VCOMPRESS_VM});
1301 else
1302 Opcodes.append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1303 return MemOpCost +
1304 LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1305}
1306
1310 Type *DataTy = MICA.getDataType();
1311 Align Alignment = MICA.getAlignment();
1312
1313 if (!isLegalStridedLoadStore(DataTy, Alignment))
1315
1317 return TTI::TCC_Basic;
1318
1319 // Cost is proportional to the number of memory operations implied. For
1320 // scalable vectors, we use an estimate on that number since we don't
1321 // know exactly what VL will be.
1322 auto &VTy = *cast<VectorType>(DataTy);
1323 unsigned NumLoads = getEstimatedVLFor(&VTy);
1324 return NumLoads * TTI::TCC_Basic;
1325}
1326
1329 // FIXME: This is a property of the default vector convention, not
1330 // all possible calling conventions. Fixing that will require
1331 // some TTI API and SLP rework.
1334 for (auto *Ty : Tys) {
1335 if (!Ty->isVectorTy())
1336 continue;
1337 Align A = DL.getPrefTypeAlign(Ty);
1338 Cost += getMemoryOpCost(Instruction::Store, Ty, A, 0, CostKind) +
1339 getMemoryOpCost(Instruction::Load, Ty, A, 0, CostKind);
1340 }
1341 return Cost;
1342}
1343
1344// Currently, these represent both throughput and codesize costs
1345// for the respective intrinsics. The costs in this table are simply
1346// instruction counts with the following adjustments made:
1347// * One vsetvli is considered free.
1349 {Intrinsic::floor, MVT::f32, 9},
1350 {Intrinsic::floor, MVT::f64, 9},
1351 {Intrinsic::ceil, MVT::f32, 9},
1352 {Intrinsic::ceil, MVT::f64, 9},
1353 {Intrinsic::trunc, MVT::f32, 7},
1354 {Intrinsic::trunc, MVT::f64, 7},
1355 {Intrinsic::round, MVT::f32, 9},
1356 {Intrinsic::round, MVT::f64, 9},
1357 {Intrinsic::roundeven, MVT::f32, 9},
1358 {Intrinsic::roundeven, MVT::f64, 9},
1359 {Intrinsic::rint, MVT::f32, 7},
1360 {Intrinsic::rint, MVT::f64, 7},
1361 {Intrinsic::nearbyint, MVT::f32, 9},
1362 {Intrinsic::nearbyint, MVT::f64, 9},
1363 {Intrinsic::bswap, MVT::i16, 3},
1364 {Intrinsic::bswap, MVT::i32, 12},
1365 {Intrinsic::bswap, MVT::i64, 31},
1366 {Intrinsic::bitreverse, MVT::i8, 17},
1367 {Intrinsic::bitreverse, MVT::i16, 24},
1368 {Intrinsic::bitreverse, MVT::i32, 33},
1369 {Intrinsic::bitreverse, MVT::i64, 52},
1370 {Intrinsic::ctpop, MVT::i8, 12},
1371 {Intrinsic::ctpop, MVT::i16, 19},
1372 {Intrinsic::ctpop, MVT::i32, 20},
1373 {Intrinsic::ctpop, MVT::i64, 21},
1374 {Intrinsic::ctlz, MVT::i8, 19},
1375 {Intrinsic::ctlz, MVT::i16, 28},
1376 {Intrinsic::ctlz, MVT::i32, 31},
1377 {Intrinsic::ctlz, MVT::i64, 35},
1378 {Intrinsic::cttz, MVT::i8, 16},
1379 {Intrinsic::cttz, MVT::i16, 23},
1380 {Intrinsic::cttz, MVT::i32, 24},
1381 {Intrinsic::cttz, MVT::i64, 25},
1382};
1383
1387 auto *RetTy = ICA.getReturnType();
1388 switch (ICA.getID()) {
1389 case Intrinsic::lrint:
1390 case Intrinsic::llrint:
1391 case Intrinsic::lround:
1392 case Intrinsic::llround: {
1393 auto LT = getTypeLegalizationCost(RetTy);
1394 Type *SrcTy = ICA.getArgTypes().front();
1395 auto SrcLT = getTypeLegalizationCost(SrcTy);
1396 if (ST->hasVInstructions() && LT.second.isVector()) {
1398 unsigned SrcEltSz = DL.getTypeSizeInBits(SrcTy->getScalarType());
1399 unsigned DstEltSz = DL.getTypeSizeInBits(RetTy->getScalarType());
1400 if (LT.second.getVectorElementType() == MVT::bf16) {
1401 if (!ST->hasVInstructionsBF16Minimal())
1403 if (DstEltSz == 32)
1404 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1405 else
1406 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1407 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1408 !ST->hasVInstructionsF16()) {
1409 if (!ST->hasVInstructionsF16Minimal())
1411 if (DstEltSz == 32)
1412 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1413 else
1414 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1415
1416 } else if (SrcEltSz > DstEltSz) {
1417 Ops = {RISCV::VFNCVT_X_F_W};
1418 } else if (SrcEltSz < DstEltSz) {
1419 Ops = {RISCV::VFWCVT_X_F_V};
1420 } else {
1421 Ops = {RISCV::VFCVT_X_F_V};
1422 }
1423
1424 // We need to use the source LMUL in the case of a narrowing op, and the
1425 // destination LMUL otherwise.
1426 if (SrcEltSz > DstEltSz)
1427 return SrcLT.first *
1428 getRISCVInstructionCost(Ops, SrcLT.second, CostKind);
1429 return LT.first * getRISCVInstructionCost(Ops, LT.second, CostKind);
1430 }
1431 break;
1432 }
1433 case Intrinsic::ceil:
1434 case Intrinsic::floor:
1435 case Intrinsic::trunc:
1436 case Intrinsic::rint:
1437 case Intrinsic::round:
1438 case Intrinsic::roundeven: {
1439 // These all use the same code.
1440 auto LT = getTypeLegalizationCost(RetTy);
1441 if (!LT.second.isVector() && TLI->isOperationCustom(ISD::FCEIL, LT.second))
1442 return LT.first * 8;
1443 break;
1444 }
1445 case Intrinsic::umin:
1446 case Intrinsic::umax:
1447 case Intrinsic::smin:
1448 case Intrinsic::smax: {
1449 auto LT = getTypeLegalizationCost(RetTy);
1450 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1451 return LT.first;
1452
1453 if (ST->hasVInstructions() && LT.second.isVector()) {
1454 unsigned Op;
1455 switch (ICA.getID()) {
1456 case Intrinsic::umin:
1457 Op = RISCV::VMINU_VV;
1458 break;
1459 case Intrinsic::umax:
1460 Op = RISCV::VMAXU_VV;
1461 break;
1462 case Intrinsic::smin:
1463 Op = RISCV::VMIN_VV;
1464 break;
1465 case Intrinsic::smax:
1466 Op = RISCV::VMAX_VV;
1467 break;
1468 }
1469 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1470 }
1471 break;
1472 }
1473 case Intrinsic::sadd_sat:
1474 case Intrinsic::ssub_sat:
1475 case Intrinsic::uadd_sat:
1476 case Intrinsic::usub_sat: {
1477 auto LT = getTypeLegalizationCost(RetTy);
1478 if (ST->hasVInstructions() && LT.second.isVector()) {
1479 unsigned Op;
1480 switch (ICA.getID()) {
1481 case Intrinsic::sadd_sat:
1482 Op = RISCV::VSADD_VV;
1483 break;
1484 case Intrinsic::ssub_sat:
1485 Op = RISCV::VSSUB_VV;
1486 break;
1487 case Intrinsic::uadd_sat:
1488 Op = RISCV::VSADDU_VV;
1489 break;
1490 case Intrinsic::usub_sat:
1491 Op = RISCV::VSSUBU_VV;
1492 break;
1493 }
1494 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1495 }
1496 break;
1497 }
1498 case Intrinsic::fma:
1499 case Intrinsic::fmuladd: {
1500 // TODO: handle promotion with f16/bf16 with zvfhmin/zvfbfmin
1501 auto LT = getTypeLegalizationCost(RetTy);
1502 if (ST->hasVInstructions() && LT.second.isVector())
1503 return LT.first *
1504 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second, CostKind);
1505 break;
1506 }
1507 case Intrinsic::fabs: {
1508 auto LT = getTypeLegalizationCost(RetTy);
1509 if (ST->hasVInstructions() && LT.second.isVector()) {
1510 // lui a0, 8
1511 // addi a0, a0, -1
1512 // vsetvli a1, zero, e16, m1, ta, ma
1513 // vand.vx v8, v8, a0
1514 // f16 with zvfhmin and bf16 with zvfhbmin
1515 if (LT.second.getVectorElementType() == MVT::bf16 ||
1516 (LT.second.getVectorElementType() == MVT::f16 &&
1517 !ST->hasVInstructionsF16()))
1518 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1519 CostKind) +
1520 2;
1521 else
1522 return LT.first *
1523 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second, CostKind);
1524 }
1525 break;
1526 }
1527 case Intrinsic::sqrt: {
1528 auto LT = getTypeLegalizationCost(RetTy);
1529 if (ST->hasVInstructions() && LT.second.isVector()) {
1532 MVT ConvType = LT.second;
1533 MVT FsqrtType = LT.second;
1534 // f16 with zvfhmin and bf16 with zvfbfmin and the type of nxv32[b]f16
1535 // will be spilt.
1536 if (LT.second.getVectorElementType() == MVT::bf16) {
1537 if (LT.second == MVT::nxv32bf16) {
1538 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1539 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1540 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1541 ConvType = MVT::nxv16f16;
1542 FsqrtType = MVT::nxv16f32;
1543 } else {
1544 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1545 FsqrtOp = {RISCV::VFSQRT_V};
1546 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1547 }
1548 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1549 !ST->hasVInstructionsF16()) {
1550 if (LT.second == MVT::nxv32f16) {
1551 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1552 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1553 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1554 ConvType = MVT::nxv16f16;
1555 FsqrtType = MVT::nxv16f32;
1556 } else {
1557 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1558 FsqrtOp = {RISCV::VFSQRT_V};
1559 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1560 }
1561 } else {
1562 FsqrtOp = {RISCV::VFSQRT_V};
1563 }
1564
1565 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType, CostKind) +
1566 getRISCVInstructionCost(ConvOp, ConvType, CostKind));
1567 }
1568 break;
1569 }
1570 case Intrinsic::cttz:
1571 case Intrinsic::ctlz:
1572 case Intrinsic::ctpop: {
1573 auto LT = getTypeLegalizationCost(RetTy);
1574 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1575 unsigned Op;
1576 switch (ICA.getID()) {
1577 case Intrinsic::cttz:
1578 Op = RISCV::VCTZ_V;
1579 break;
1580 case Intrinsic::ctlz:
1581 Op = RISCV::VCLZ_V;
1582 break;
1583 case Intrinsic::ctpop:
1584 Op = RISCV::VCPOP_V;
1585 break;
1586 }
1587 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1588 }
1589 break;
1590 }
1591 case Intrinsic::abs: {
1592 auto LT = getTypeLegalizationCost(RetTy);
1593 if (ST->hasVInstructions() && LT.second.isVector()) {
1594 // vabs.v v10, v8
1595 if (ST->hasStdExtZvabd())
1596 return LT.first *
1597 getRISCVInstructionCost({RISCV::VABS_V}, LT.second, CostKind);
1598
1599 // vrsub.vi v10, v8, 0
1600 // vmax.vv v8, v8, v10
1601 return LT.first *
1602 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1603 LT.second, CostKind);
1604 }
1605 break;
1606 }
1607 case Intrinsic::fshl:
1608 case Intrinsic::fshr: {
1609 if (ICA.getArgs().empty())
1610 break;
1611
1612 // Funnel-shifts are ROTL/ROTR when the first and second operand are equal.
1613 // When Zbb/Zbkb is enabled we can use a single ROL(W)/ROR(I)(W)
1614 // instruction.
1615 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1616 ICA.getArgs()[0] == ICA.getArgs()[1] &&
1617 (RetTy->getIntegerBitWidth() == 32 ||
1618 RetTy->getIntegerBitWidth() == 64) &&
1619 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1620 return 1;
1621 }
1622 break;
1623 }
1624 case Intrinsic::clmul: {
1625 auto LT = getTypeLegalizationCost(RetTy);
1626 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbc() &&
1627 !ST->hasStdExtZbkc()) {
1628 // TODO: Once custom lowering in this case for RV32 is added, this guard
1629 // should be removed and the cost model should be updated.
1630 if (!ST->is64Bit() || LT.second != MVT::i64)
1631 break;
1632 // vmv.s.x v8, a0
1633 // vclmul.vx v8, v8, a1
1634 // vmv.x.s a0, v8
1635 MVT VecVT = MVT::getScalableVectorVT(LT.second, 1);
1636 return LT.first * getRISCVInstructionCost(
1637 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1638 VecVT, CostKind);
1639 }
1640 break;
1641 }
1642 case Intrinsic::masked_udiv:
1643 return getArithmeticInstrCost(Instruction::UDiv, ICA.getReturnType(),
1644 CostKind);
1645 case Intrinsic::masked_sdiv:
1646 return getArithmeticInstrCost(Instruction::SDiv, ICA.getReturnType(),
1647 CostKind);
1648 case Intrinsic::masked_urem:
1649 return getArithmeticInstrCost(Instruction::URem, ICA.getReturnType(),
1650 CostKind);
1651 case Intrinsic::masked_srem:
1652 return getArithmeticInstrCost(Instruction::SRem, ICA.getReturnType(),
1653 CostKind);
1654 case Intrinsic::get_active_lane_mask: {
1655 if (ST->hasVInstructions()) {
1656 Type *ExpRetTy = VectorType::get(
1657 ICA.getArgTypes()[0], cast<VectorType>(RetTy)->getElementCount());
1658 auto LT = getTypeLegalizationCost(ExpRetTy);
1659
1660 // vid.v v8 // considered hoisted
1661 // vsaddu.vx v8, v8, a0
1662 // vmsltu.vx v0, v8, a1
1663 return LT.first *
1664 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1665 LT.second, CostKind);
1666 }
1667 break;
1668 }
1669 // TODO: add more intrinsic
1670 case Intrinsic::stepvector: {
1671 auto LT = getTypeLegalizationCost(RetTy);
1672 // Legalisation of illegal types involves an `index' instruction plus
1673 // (LT.first - 1) vector adds.
1674 if (ST->hasVInstructions())
1675 return getRISCVInstructionCost(RISCV::VID_V, LT.second, CostKind) +
1676 (LT.first - 1) *
1677 getRISCVInstructionCost(RISCV::VADD_VX, LT.second, CostKind);
1678 return 1 + (LT.first - 1);
1679 }
1680 case Intrinsic::vector_splice_left:
1681 case Intrinsic::vector_splice_right: {
1682 auto LT = getTypeLegalizationCost(RetTy);
1683 // Constant offsets fall through to getShuffleCost.
1684 if (!ICA.isTypeBasedOnly() && isa<ConstantInt>(ICA.getArgs()[2]))
1685 break;
1686 if (ST->hasVInstructions() && LT.second.isVector()) {
1687 return LT.first *
1688 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1689 LT.second, CostKind);
1690 }
1691 break;
1692 }
1693 case Intrinsic::experimental_cttz_elts: {
1694 if (!ST->hasVInstructions())
1695 break;
1697 Type *ArgTy = ICA.getArgTypes()[0];
1698 auto LT = getTypeLegalizationCost(ArgTy);
1699
1700 // If the element type is not i1, do a comparison with all-zeros.
1701 if (LT.second.getVectorElementType() != MVT::i1)
1702 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second, CostKind);
1703
1704 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind);
1705
1706 // If zero_is_poison is false, then we will generate additional
1707 // cmp + select instructions to convert -1 to EVL.
1708 Type *BoolTy = Type::getInt1Ty(RetTy->getContext());
1709 if (ICA.getArgs().size() > 1 &&
1710 cast<ConstantInt>(ICA.getArgs()[1])->isZero())
1711 Cost += getCmpSelInstrCost(Instruction::ICmp, BoolTy, RetTy,
1713 getCmpSelInstrCost(Instruction::Select, RetTy, BoolTy,
1715
1716 return LT.first * Cost;
1717 }
1718 case Intrinsic::experimental_vp_splice: {
1719 // To support type-based query from vectorizer, set the index to 0.
1720 // Note that index only change the cost from vslide.vx to vslide.vi and in
1721 // current implementations they have same costs.
1723 cast<VectorType>(ICA.getArgTypes()[0]), {}, CostKind,
1725 }
1726 case Intrinsic::vp_merge: {
1727 // If an operand is a binary op and the type is legal, RISCVVectorPeephole
1728 // will likely fold the resulting vmerge.vvm away.
1730 getTypeLegalizationCost(RetTy).first == 1)
1731 return TTI::TCC_Free;
1732 break;
1733 }
1734 case Intrinsic::fptoui_sat:
1735 case Intrinsic::fptosi_sat: {
1737 bool IsSigned = ICA.getID() == Intrinsic::fptosi_sat;
1738 Type *SrcTy = ICA.getArgTypes()[0];
1739
1740 auto SrcLT = getTypeLegalizationCost(SrcTy);
1741 auto DstLT = getTypeLegalizationCost(RetTy);
1742 if (!SrcTy->isVectorTy())
1743 break;
1744
1745 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1747
1748 Cost +=
1749 getCastInstrCost(IsSigned ? Instruction::FPToSI : Instruction::FPToUI,
1750 RetTy, SrcTy, TTI::CastContextHint::None, CostKind);
1751
1752 // Handle NaN.
1753 // vmfne v0, v8, v8 # If v8[i] is NaN set v0[i] to 1.
1754 // vmerge.vim v8, v8, 0, v0 # Convert NaN to 0.
1755 Type *CondTy = RetTy->getWithNewBitWidth(1);
1756 Cost += getCmpSelInstrCost(BinaryOperator::FCmp, SrcTy, CondTy,
1758 Cost += getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
1760 return Cost;
1761 }
1762 case Intrinsic::experimental_vector_extract_last_active: {
1763 auto *ValTy = cast<VectorType>(ICA.getArgTypes()[0]);
1764 auto *MaskTy = cast<VectorType>(ICA.getArgTypes()[1]);
1765
1766 auto ValLT = getTypeLegalizationCost(ValTy);
1767 auto MaskLT = getTypeLegalizationCost(MaskTy);
1768
1769 // TODO: Return cheaper cost when the entire lane is inactive.
1770 // The expected asm sequence is:
1771 // vcpop.m a0, v0
1772 // beqz a0, exit # Return passthru when the entire lane is inactive.
1773 // vid v10, v0.t
1774 // vredmaxu.vs v10, v10, v10
1775 // vmv.x.s a0, v10
1776 // zext.b a0, a0
1777 // vslidedown.vx v8, v8, a0
1778 // vmv.x.s a0, v8
1779 // exit:
1780 // ...
1781
1782 // Find a suitable type for a stepvector.
1783 ConstantRange VScaleRange(APInt(64, 1), APInt::getZero(64));
1784 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1785 TLI->getVectorIdxTy(getDataLayout()), MaskTy->getElementCount(),
1786 /*ZeroIsPoison=*/true, &VScaleRange);
1787 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1788 Type *StepTy = Type::getIntNTy(MaskTy->getContext(), EltWidth);
1789 auto *StepVecTy = VectorType::get(StepTy, ValTy->getElementCount());
1790 auto StepLT = getTypeLegalizationCost(StepVecTy);
1791
1792 // Currently expandVectorFindLastActive cannot handle step vector split.
1793 // So return invalid when the type needs split.
1794 // FIXME: Remove this if expandVectorFindLastActive supports split vector.
1795 if (StepLT.first > 1)
1797
1799 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1800
1801 Cost += MaskLT.first *
1802 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second, CostKind);
1803 Cost += getCFInstrCost(Instruction::CondBr, CostKind, nullptr);
1804 Cost += StepLT.first *
1805 getRISCVInstructionCost(Opcodes, StepLT.second, CostKind);
1806 Cost += getCastInstrCost(Instruction::ZExt,
1807 Type::getInt64Ty(ValTy->getContext()), StepTy,
1809 Cost += ValLT.first *
1810 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1811 ValLT.second, CostKind);
1812 return Cost;
1813 }
1814 }
1815
1816 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
1817 if (auto LT = getTypeLegalizationCost(RetTy);
1818 LT.second.isVector()) {
1819 MVT EltTy = LT.second.getVectorElementType();
1820 if (const auto *Entry = CostTableLookup(VectorIntrinsicCostTable,
1821 ICA.getID(), EltTy))
1822 return LT.first * Entry->Cost;
1823 }
1824 }
1825
1827}
1828
1831 const SCEV *Ptr,
1833 // Address computations for vector indexed load/store likely require an offset
1834 // and/or scaling.
1835 if (ST->hasVInstructions() && PtrTy->isVectorTy())
1836 return getArithmeticInstrCost(Instruction::Add, PtrTy, CostKind);
1837
1838 return BaseT::getAddressComputationCost(PtrTy, SE, Ptr, CostKind);
1839}
1840
1842 Type *Src,
1845 const Instruction *I) const {
1846 bool IsVectorType = isa<VectorType>(Dst) && isa<VectorType>(Src);
1847 if (!IsVectorType)
1848 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1849
1850 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
1851 // For now, skip all fixed vector cost analysis when P extension is available
1852 // to avoid crashes in getMinRVVVectorSizeInBits()
1853 if (ST->hasStdExtP() &&
1855 return 1; // Treat as single instruction cost for now
1856 }
1857
1858 // FIXME: Need to compute legalizing cost for illegal types. The current
1859 // code handles only legal types and those which can be trivially
1860 // promoted to legal.
1861 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
1862 Dst->getScalarSizeInBits() > ST->getELen())
1863 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1864
1865 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1866 assert(ISD && "Invalid opcode");
1867 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Src);
1868 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Dst);
1869
1870 // Handle i1 source and dest cases *before* calling logic in BasicTTI.
1871 // The shared implementation doesn't model vector widening during legalization
1872 // and instead assumes scalarization. In order to scalarize an <N x i1>
1873 // vector, we need to extend/trunc to/from i8. If we don't special case
1874 // this, we can get an infinite recursion cycle.
1875 switch (ISD) {
1876 default:
1877 break;
1878 case ISD::SIGN_EXTEND:
1879 case ISD::ZERO_EXTEND:
1880 if (Src->getScalarSizeInBits() == 1) {
1881 // We do not use vsext/vzext to extend from mask vector.
1882 // Instead we use the following instructions to extend from mask vector:
1883 // vmv.v.i v8, 0
1884 // vmerge.vim v8, v8, -1, v0 (repeated per split)
1885 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second, CostKind) +
1886 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
1887 DstLT.second, CostKind) +
1888 DstLT.first - 1;
1889 }
1890 break;
1891 case ISD::TRUNCATE:
1892 if (Dst->getScalarSizeInBits() == 1) {
1893 // We do not use several vncvt to truncate to mask vector. So we could
1894 // not use PowDiff to calculate it.
1895 // Instead we use the following instructions to truncate to mask vector:
1896 // vand.vi v8, v8, 1
1897 // vmsne.vi v0, v8, 0
1898 return SrcLT.first *
1899 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
1900 SrcLT.second, CostKind) +
1901 SrcLT.first - 1;
1902 }
1903 break;
1904 };
1905
1906 // Our actual lowering for the case where a wider legal type is available
1907 // uses promotion to the wider type. This is reflected in the result of
1908 // getTypeLegalizationCost, but BasicTTI assumes the widened cases are
1909 // scalarized if the legalized Src and Dst are not equal sized.
1910 const DataLayout &DL = this->getDataLayout();
1911 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
1912 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
1913 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Src),
1914 SrcLT.second.getSizeInBits()) ||
1915 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Dst),
1916 DstLT.second.getSizeInBits()) ||
1917 SrcLT.first > 1 || DstLT.first > 1)
1918 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1919
1920 // The split cost is handled by the base getCastInstrCost
1921 assert((SrcLT.first == 1) && (DstLT.first == 1) && "Illegal type");
1922
1923 int PowDiff = (int)Log2_32(DstLT.second.getScalarSizeInBits()) -
1924 (int)Log2_32(SrcLT.second.getScalarSizeInBits());
1925 switch (ISD) {
1926 case ISD::SIGN_EXTEND:
1927 case ISD::ZERO_EXTEND: {
1928 if ((PowDiff < 1) || (PowDiff > 3))
1929 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1930 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
1931 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
1932 unsigned Op =
1933 (ISD == ISD::SIGN_EXTEND) ? SExtOp[PowDiff - 1] : ZExtOp[PowDiff - 1];
1934 return getRISCVInstructionCost(Op, DstLT.second, CostKind);
1935 }
1936 case ISD::TRUNCATE:
1937 case ISD::FP_EXTEND:
1938 case ISD::FP_ROUND: {
1939 // Counts of narrow/widen instructions.
1940 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
1941 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
1942
1943 unsigned Op = (ISD == ISD::TRUNCATE) ? RISCV::VNSRL_WI
1944 : (ISD == ISD::FP_EXTEND) ? RISCV::VFWCVT_F_F_V
1945 : RISCV::VFNCVT_F_F_W;
1947 for (; SrcEltSize != DstEltSize;) {
1948 MVT ElementMVT = (ISD == ISD::TRUNCATE)
1949 ? MVT::getIntegerVT(DstEltSize)
1950 : MVT::getFloatingPointVT(DstEltSize);
1951 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
1952 DstEltSize =
1953 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
1954 Cost += getRISCVInstructionCost(Op, DstMVT, CostKind);
1955 }
1956 return Cost;
1957 }
1958 case ISD::FP_TO_SINT:
1959 case ISD::FP_TO_UINT: {
1960 unsigned IsSigned = ISD == ISD::FP_TO_SINT;
1961 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
1962 unsigned FWCVT =
1963 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
1964 unsigned FNCVT =
1965 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
1966 unsigned SrcEltSize = Src->getScalarSizeInBits();
1967 unsigned DstEltSize = Dst->getScalarSizeInBits();
1969 if ((SrcEltSize == 16) &&
1970 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
1971 // If the target only supports zvfhmin or it is fp16-to-i64 conversion
1972 // pre-widening to f32 and then convert f32 to integer
1973 VectorType *VecF32Ty =
1974 VectorType::get(Type::getFloatTy(Dst->getContext()),
1975 cast<VectorType>(Dst)->getElementCount());
1976 std::pair<InstructionCost, MVT> VecF32LT =
1977 getTypeLegalizationCost(VecF32Ty);
1978 Cost +=
1979 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
1980 VecF32LT.second, CostKind);
1981 Cost += getCastInstrCost(Opcode, Dst, VecF32Ty, CCH, CostKind, I);
1982 return Cost;
1983 }
1984 if (DstEltSize == SrcEltSize)
1985 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
1986 else if (DstEltSize > SrcEltSize)
1987 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
1988 else { // (SrcEltSize > DstEltSize)
1989 // First do a narrowing conversion to an integer half the size, then
1990 // truncate if needed.
1991 MVT ElementVT = MVT::getIntegerVT(SrcEltSize / 2);
1992 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
1993 Cost += getRISCVInstructionCost(FNCVT, VecVT, CostKind);
1994 if ((SrcEltSize / 2) > DstEltSize) {
1995 Type *VecTy = EVT(VecVT).getTypeForEVT(Dst->getContext());
1996 Cost +=
1997 getCastInstrCost(Instruction::Trunc, Dst, VecTy, CCH, CostKind, I);
1998 }
1999 }
2000 return Cost;
2001 }
2002 case ISD::SINT_TO_FP:
2003 case ISD::UINT_TO_FP: {
2004 unsigned IsSigned = ISD == ISD::SINT_TO_FP;
2005 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2006 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2007 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2008 unsigned SrcEltSize = Src->getScalarSizeInBits();
2009 unsigned DstEltSize = Dst->getScalarSizeInBits();
2010
2012 if ((DstEltSize == 16) &&
2013 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2014 // If the target only supports zvfhmin or it is i64-to-fp16 conversion
2015 // it is converted to f32 and then converted to f16
2016 VectorType *VecF32Ty =
2017 VectorType::get(Type::getFloatTy(Dst->getContext()),
2018 cast<VectorType>(Dst)->getElementCount());
2019 std::pair<InstructionCost, MVT> VecF32LT =
2020 getTypeLegalizationCost(VecF32Ty);
2021 Cost += getCastInstrCost(Opcode, VecF32Ty, Src, CCH, CostKind, I);
2022 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2023 DstLT.second, CostKind);
2024 return Cost;
2025 }
2026
2027 if (DstEltSize == SrcEltSize)
2028 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2029 else if (DstEltSize > SrcEltSize) {
2030 if ((DstEltSize / 2) > SrcEltSize) {
2031 VectorType *VecTy =
2032 VectorType::get(IntegerType::get(Dst->getContext(), DstEltSize / 2),
2033 cast<VectorType>(Dst)->getElementCount());
2034 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2035 Cost += getCastInstrCost(Op, VecTy, Src, CCH, CostKind, I);
2036 }
2037 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2038 } else
2039 Cost += getRISCVInstructionCost(FNCVT, DstLT.second, CostKind);
2040 return Cost;
2041 }
2042 }
2043 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2044}
2045
2046unsigned RISCVTTIImpl::getEstimatedVLFor(VectorType *Ty) const {
2047 if (isa<ScalableVectorType>(Ty)) {
2048 const unsigned EltSize = DL.getTypeSizeInBits(Ty->getElementType());
2049 const unsigned MinSize = DL.getTypeSizeInBits(Ty).getKnownMinValue();
2050 const unsigned VectorBits = *getVScaleForTuning() * RISCV::RVVBitsPerBlock;
2051 return RISCVTargetLowering::computeVLMAX(VectorBits, EltSize, MinSize);
2052 }
2053 return cast<FixedVectorType>(Ty)->getNumElements();
2054}
2055
2058 FastMathFlags FMF,
2060 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2061 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2062
2063 // Skip if scalar size of Ty is bigger than ELEN.
2064 if (Ty->getScalarSizeInBits() > ST->getELen())
2065 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2066
2067 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2068 if (Ty->getElementType()->isIntegerTy(1)) {
2069 // SelectionDAGBuilder does following transforms:
2070 // vector_reduce_{smin,umax}(<n x i1>) --> vector_reduce_or(<n x i1>)
2071 // vector_reduce_{smax,umin}(<n x i1>) --> vector_reduce_and(<n x i1>)
2072 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2073 return getArithmeticReductionCost(Instruction::Or, Ty, FMF, CostKind);
2074 else
2075 return getArithmeticReductionCost(Instruction::And, Ty, FMF, CostKind);
2076 }
2077
2078 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2080 InstructionCost ExtraCost = 0;
2081 switch (IID) {
2082 case Intrinsic::maximum:
2083 if (FMF.noNaNs()) {
2084 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2085 } else {
2086 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2087 RISCV::VFMV_F_S};
2088 // Cost of Canonical Nan + branch
2089 // lui a0, 523264
2090 // fmv.w.x fa0, a0
2091 Type *DstTy = Ty->getScalarType();
2092 const unsigned EltTyBits = DstTy->getScalarSizeInBits();
2093 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2094 ExtraCost = 1 +
2095 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2097 getCFInstrCost(Instruction::CondBr, CostKind);
2098 }
2099 break;
2100
2101 case Intrinsic::minimum:
2102 if (FMF.noNaNs()) {
2103 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2104 } else {
2105 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2106 RISCV::VFMV_F_S};
2107 // Cost of Canonical Nan + branch
2108 // lui a0, 523264
2109 // fmv.w.x fa0, a0
2110 Type *DstTy = Ty->getScalarType();
2111 const unsigned EltTyBits = DL.getTypeSizeInBits(DstTy);
2112 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2113 ExtraCost = 1 +
2114 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2116 getCFInstrCost(Instruction::CondBr, CostKind);
2117 }
2118 break;
2119 }
2120 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2121 }
2122
2123 // IR Reduction is composed by one rvv reduction instruction and vmv
2124 unsigned SplitOp;
2126 switch (IID) {
2127 default:
2128 llvm_unreachable("Unsupported intrinsic");
2129 case Intrinsic::smax:
2130 SplitOp = RISCV::VMAX_VV;
2131 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2132 break;
2133 case Intrinsic::smin:
2134 SplitOp = RISCV::VMIN_VV;
2135 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2136 break;
2137 case Intrinsic::umax:
2138 SplitOp = RISCV::VMAXU_VV;
2139 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2140 break;
2141 case Intrinsic::umin:
2142 SplitOp = RISCV::VMINU_VV;
2143 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2144 break;
2145 case Intrinsic::maxnum:
2146 SplitOp = RISCV::VFMAX_VV;
2147 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2148 break;
2149 case Intrinsic::minnum:
2150 SplitOp = RISCV::VFMIN_VV;
2151 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2152 break;
2153 }
2154 // Add a cost for data larger than LMUL8
2155 InstructionCost SplitCost =
2156 (LT.first > 1) ? (LT.first - 1) *
2157 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2158 : 0;
2159 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2160}
2161
2164 std::optional<FastMathFlags> FMF,
2166 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2167 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2168
2169 // Skip if scalar size of Ty is bigger than ELEN.
2170 if (Ty->getScalarSizeInBits() > ST->getELen())
2171 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2172
2173 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2174 assert(ISD && "Invalid opcode");
2175
2176 if (ISD != ISD::ADD && ISD != ISD::OR && ISD != ISD::XOR && ISD != ISD::AND &&
2177 ISD != ISD::FADD)
2178 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2179
2180 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2181 Type *ElementTy = Ty->getElementType();
2182 if (ElementTy->isIntegerTy(1)) {
2183 // Example sequences:
2184 // vfirst.m a0, v0
2185 // seqz a0, a0
2186 if (LT.second == MVT::v1i1)
2187 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind) +
2188 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2190
2191 if (ISD == ISD::AND) {
2192 // Example sequences:
2193 // vmand.mm v8, v9, v8 ; needed every time type is split
2194 // vmnot.m v8, v0 ; alias for vmnand
2195 // vcpop.m a0, v8
2196 // seqz a0, a0
2197
2198 // See the discussion: https://github.com/llvm/llvm-project/pull/119160
2199 // For LMUL <= 8, there is no splitting,
2200 // the sequences are vmnot, vcpop and seqz.
2201 // When LMUL > 8 and split = 1,
2202 // the sequences are vmnand, vcpop and seqz.
2203 // When LMUL > 8 and split > 1,
2204 // the sequences are (LT.first-2) * vmand, vmnand, vcpop and seqz.
2205 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2206 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second, CostKind) +
2207 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second, CostKind) +
2208 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2209 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2211 } else if (ISD == ISD::XOR || ISD == ISD::ADD) {
2212 // Example sequences:
2213 // vsetvli a0, zero, e8, mf8, ta, ma
2214 // vmxor.mm v8, v0, v8 ; needed every time type is split
2215 // vcpop.m a0, v8
2216 // andi a0, a0, 1
2217 return (LT.first - 1) *
2218 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind) +
2219 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) + 1;
2220 } else {
2221 assert(ISD == ISD::OR);
2222 // Example sequences:
2223 // vsetvli a0, zero, e8, mf8, ta, ma
2224 // vmor.mm v8, v9, v8 ; needed every time type is split
2225 // vcpop.m a0, v0
2226 // snez a0, a0
2227 return (LT.first - 1) *
2228 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second, CostKind) +
2229 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2230 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2232 }
2233 }
2234
2235 // IR Reduction of or/and is composed by one vmv and one rvv reduction
2236 // instruction, and others is composed by two vmv and one rvv reduction
2237 // instruction
2238 unsigned SplitOp;
2240 switch (ISD) {
2241 case ISD::ADD:
2242 SplitOp = RISCV::VADD_VV;
2243 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2244 break;
2245 case ISD::OR:
2246 SplitOp = RISCV::VOR_VV;
2247 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2248 break;
2249 case ISD::XOR:
2250 SplitOp = RISCV::VXOR_VV;
2251 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2252 break;
2253 case ISD::AND:
2254 SplitOp = RISCV::VAND_VV;
2255 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2256 break;
2257 case ISD::FADD:
2258 // We can't promote f16/bf16 fadd reductions.
2259 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2260 LT.second.getScalarType() == MVT::bf16)
2261 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2263 Opcodes.push_back(RISCV::VFMV_S_F);
2264 for (unsigned i = 0; i < LT.first.getValue(); i++)
2265 Opcodes.push_back(RISCV::VFREDOSUM_VS);
2266 Opcodes.push_back(RISCV::VFMV_F_S);
2267 return getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2268 }
2269 SplitOp = RISCV::VFADD_VV;
2270 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2271 break;
2272 }
2273 // Add a cost for data larger than LMUL8
2274 InstructionCost SplitCost =
2275 (LT.first > 1) ? (LT.first - 1) *
2276 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2277 : 0;
2278 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2279}
2280
2282 unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy,
2283 std::optional<FastMathFlags> FMF, TTI::TargetCostKind CostKind) const {
2284 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2285 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2286 FMF, CostKind);
2287
2288 // Skip if scalar size of ResTy is bigger than ELEN.
2289 if (ResTy->getScalarSizeInBits() > ST->getELen())
2290 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2291 FMF, CostKind);
2292
2293 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2294 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2295 FMF, CostKind);
2296
2297 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2298
2299 if (IsUnsigned && Opcode == Instruction::Add &&
2300 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2301 // Represent vector_reduce_add(ZExt(<n x i1>)) as
2302 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
2303 return LT.first *
2304 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind);
2305 }
2306
2307 if (ResTy->getScalarSizeInBits() != 2 * LT.second.getScalarSizeInBits())
2308 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2309 FMF, CostKind);
2310
2311 return (LT.first - 1) +
2312 getArithmeticReductionCost(Opcode, ValTy, FMF, CostKind);
2313}
2314
2318 assert(OpInfo.isConstant() && "non constant operand?");
2319 if (!isa<VectorType>(Ty))
2320 // FIXME: We need to account for immediate materialization here, but doing
2321 // a decent job requires more knowledge about the immediate than we
2322 // currently have here.
2323 return 0;
2324
2325 if (OpInfo.isUniform())
2326 // vmv.v.i, vmv.v.x, or vfmv.v.f
2327 // We ignore the cost of the scalar constant materialization to be consistent
2328 // with how we treat scalar constants themselves just above.
2329 return 1;
2330
2331 return getConstantPoolLoadCost(Ty, CostKind);
2332}
2333
2335 Align Alignment,
2336 unsigned AddressSpace,
2338 TTI::OperandValueInfo OpInfo,
2339 const Instruction *I) const {
2340 EVT VT = TLI->getValueType(DL, Src, true);
2341 // Type legalization can't handle structs, and load latency isn't handled here
2342 if (VT == MVT::Other ||
2343 (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency))
2344 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2345 CostKind, OpInfo, I);
2346
2348 if (Opcode == Instruction::Store && OpInfo.isConstant())
2349 Cost += getStoreImmCost(Src, OpInfo, CostKind);
2350
2351 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Src);
2352
2353 InstructionCost BaseCost = [&]() {
2354 InstructionCost Cost = LT.first;
2356 return Cost;
2357
2358 // Our actual lowering for the case where a wider legal type is available
2359 // uses the a VL predicated load on the wider type. This is reflected in
2360 // the result of getTypeLegalizationCost, but BasicTTI assumes the
2361 // widened cases are scalarized.
2362 const DataLayout &DL = this->getDataLayout();
2363 if (Src->isVectorTy() && LT.second.isVector() &&
2364 TypeSize::isKnownLT(DL.getTypeStoreSizeInBits(Src),
2365 LT.second.getSizeInBits()))
2366 return Cost;
2367
2368 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2369 CostKind, OpInfo, I);
2370 }();
2371
2372 // Assume memory ops cost scale with the number of vector registers
2373 // possible accessed by the instruction. Note that BasicTTI already
2374 // handles the LT.first term for us.
2375 if (ST->hasVInstructions() && LT.second.isVector() &&
2377 BaseCost *= TLI->getLMULCost(LT.second);
2378 return Cost + BaseCost;
2379}
2380
2382 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
2384 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
2386 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2387 Op1Info, Op2Info, I);
2388
2389 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2390 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2391 Op1Info, Op2Info, I);
2392
2393 // Skip if scalar size of ValTy is bigger than ELEN.
2394 if (ValTy->isVectorTy() && ValTy->getScalarSizeInBits() > ST->getELen())
2395 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2396 Op1Info, Op2Info, I);
2397
2398 auto GetConstantMatCost =
2399 [&](TTI::OperandValueInfo OpInfo) -> InstructionCost {
2400 if (OpInfo.isUniform())
2401 // We return 0 we currently ignore the cost of materializing scalar
2402 // constants in GPRs.
2403 return 0;
2404
2405 return getConstantPoolLoadCost(ValTy, CostKind);
2406 };
2407
2408 InstructionCost ConstantMatCost;
2409 if (Op1Info.isConstant())
2410 ConstantMatCost += GetConstantMatCost(Op1Info);
2411 if (Op2Info.isConstant())
2412 ConstantMatCost += GetConstantMatCost(Op2Info);
2413
2414 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2415 if (Opcode == Instruction::Select && LT.second.isVector()) {
2416 if (CondTy->isVectorTy()) {
2417 if (ValTy->getScalarSizeInBits() == 1) {
2418 // vmandn.mm v8, v8, v9
2419 // vmand.mm v9, v0, v9
2420 // vmor.mm v0, v9, v8
2421 return ConstantMatCost +
2422 LT.first *
2423 getRISCVInstructionCost(
2424 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2425 LT.second, CostKind);
2426 }
2427 // vselect and max/min are supported natively.
2428 return ConstantMatCost +
2429 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2430 CostKind);
2431 }
2432
2433 if (ValTy->getScalarSizeInBits() == 1) {
2434 // vmv.v.x v9, a0
2435 // vmsne.vi v9, v9, 0
2436 // vmandn.mm v8, v8, v9
2437 // vmand.mm v9, v0, v9
2438 // vmor.mm v0, v9, v8
2439 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2440 return ConstantMatCost +
2441 LT.first *
2442 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2443 InterimVT, CostKind) +
2444 LT.first * getRISCVInstructionCost(
2445 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2446 LT.second, CostKind);
2447 }
2448
2449 // vmv.v.x v10, a0
2450 // vmsne.vi v0, v10, 0
2451 // vmerge.vvm v8, v9, v8, v0
2452 return ConstantMatCost +
2453 LT.first * getRISCVInstructionCost(
2454 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2455 LT.second, CostKind);
2456 }
2457
2458 if ((Opcode == Instruction::ICmp) && ValTy->isVectorTy() &&
2459 CmpInst::isIntPredicate(VecPred)) {
2460 // Use VMSLT_VV to represent VMSEQ, VMSNE, VMSLTU, VMSLEU, VMSLT, VMSLE
2461 // provided they incur the same cost across all implementations
2462 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2463 LT.second,
2464 CostKind);
2465 }
2466
2467 if ((Opcode == Instruction::FCmp) && ValTy->isVectorTy() &&
2468 CmpInst::isFPPredicate(VecPred)) {
2469
2470 // Use VMXOR_MM and VMXNOR_MM to generate all true/false mask
2471 if ((VecPred == CmpInst::FCMP_FALSE) || (VecPred == CmpInst::FCMP_TRUE))
2472 return ConstantMatCost +
2473 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind);
2474
2475 // If we do not support the input floating point vector type, use the base
2476 // one which will calculate as:
2477 // ScalarizeCost + Num * Cost for fixed vector,
2478 // InvalidCost for scalable vector.
2479 if ((ValTy->getScalarSizeInBits() == 16 && !ST->hasVInstructionsF16()) ||
2480 (ValTy->getScalarSizeInBits() == 32 && !ST->hasVInstructionsF32()) ||
2481 (ValTy->getScalarSizeInBits() == 64 && !ST->hasVInstructionsF64()))
2482 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2483 Op1Info, Op2Info, I);
2484
2485 // Assuming vector fp compare and mask instructions are all the same cost
2486 // until a need arises to differentiate them.
2487 switch (VecPred) {
2488 case CmpInst::FCMP_ONE: // vmflt.vv + vmflt.vv + vmor.mm
2489 case CmpInst::FCMP_ORD: // vmfeq.vv + vmfeq.vv + vmand.mm
2490 case CmpInst::FCMP_UNO: // vmfne.vv + vmfne.vv + vmor.mm
2491 case CmpInst::FCMP_UEQ: // vmflt.vv + vmflt.vv + vmnor.mm
2492 return ConstantMatCost +
2493 LT.first * getRISCVInstructionCost(
2494 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2495 LT.second, CostKind);
2496
2497 case CmpInst::FCMP_UGT: // vmfle.vv + vmnot.m
2498 case CmpInst::FCMP_UGE: // vmflt.vv + vmnot.m
2499 case CmpInst::FCMP_ULT: // vmfle.vv + vmnot.m
2500 case CmpInst::FCMP_ULE: // vmflt.vv + vmnot.m
2501 return ConstantMatCost +
2502 LT.first *
2503 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2504 LT.second, CostKind);
2505
2506 case CmpInst::FCMP_OEQ: // vmfeq.vv
2507 case CmpInst::FCMP_OGT: // vmflt.vv
2508 case CmpInst::FCMP_OGE: // vmfle.vv
2509 case CmpInst::FCMP_OLT: // vmflt.vv
2510 case CmpInst::FCMP_OLE: // vmfle.vv
2511 case CmpInst::FCMP_UNE: // vmfne.vv
2512 return ConstantMatCost +
2513 LT.first *
2514 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second, CostKind);
2515 default:
2516 break;
2517 }
2518 }
2519
2520 // With ShortForwardBranchOpt or ConditionalMoveFusion, scalar icmp + select
2521 // instructions will lower to SELECT_CC and lower to PseudoCCMOVGPR which will
2522 // generate a conditional branch + mv. The cost of scalar (icmp + select) will
2523 // be (0 + select instr cost).
2524 if (ST->hasConditionalMoveFusion() && I && isa<ICmpInst>(I) &&
2525 ValTy->isIntegerTy() && !I->user_empty()) {
2526 if (all_of(I->users(), [&](const User *U) {
2527 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2528 U->getType()->isIntegerTy() &&
2529 !isa<ConstantData>(U->getOperand(1)) &&
2530 !isa<ConstantData>(U->getOperand(2));
2531 }))
2532 return 0;
2533 }
2534
2535 // TODO: Add cost for scalar type.
2536
2537 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2538 Op1Info, Op2Info, I);
2539}
2540
2543 const Instruction *I) const {
2545 return Opcode == Instruction::PHI ? 0 : 1;
2546 // Branches are assumed to be predicted.
2547 return 0;
2548}
2549
2551 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
2552 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
2553 assert(Val->isVectorTy() && "This must be a vector type");
2554
2555 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
2556 // For now, skip all fixed vector cost analysis when P extension is available
2557 // to avoid crashes in getMinRVVVectorSizeInBits()
2558 if (ST->hasStdExtP() && isa<FixedVectorType>(Val)) {
2559 return 1; // Treat as single instruction cost for now
2560 }
2561
2562 if (Opcode != Instruction::ExtractElement &&
2563 Opcode != Instruction::InsertElement)
2564 return BaseT::getVectorInstrCost(Opcode, Val, CostKind, Index, Op0, Op1,
2565 VIC);
2566
2567 // Legalize the type.
2568 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Val);
2569
2570 // This type is legalized to a scalar type.
2571 if (!LT.second.isVector()) {
2572 auto *FixedVecTy = cast<FixedVectorType>(Val);
2573 // If Index is a known constant, cost is zero.
2574 if (Index != -1U)
2575 return 0;
2576 // Extract/InsertElement with non-constant index is very costly when
2577 // scalarized; estimate cost of loads/stores sequence via the stack:
2578 // ExtractElement cost: store vector to stack, load scalar;
2579 // InsertElement cost: store vector to stack, store scalar, load vector.
2580 Type *ElemTy = FixedVecTy->getElementType();
2581 auto NumElems = FixedVecTy->getNumElements();
2582 auto Align = DL.getPrefTypeAlign(ElemTy);
2583 InstructionCost LoadCost =
2584 getMemoryOpCost(Instruction::Load, ElemTy, Align, 0, CostKind);
2585 InstructionCost StoreCost =
2586 getMemoryOpCost(Instruction::Store, ElemTy, Align, 0, CostKind);
2587 return Opcode == Instruction::ExtractElement
2588 ? StoreCost * NumElems + LoadCost
2589 : (StoreCost + LoadCost) * NumElems + StoreCost;
2590 }
2591
2592 // For unsupported scalable vector.
2593 if (LT.second.isScalableVector() && !LT.first.isValid())
2594 return LT.first;
2595
2596 // Mask vector extract/insert is expanded via e8.
2597 if (Val->getScalarSizeInBits() == 1) {
2598 VectorType *WideTy =
2600 cast<VectorType>(Val)->getElementCount());
2601 if (Opcode == Instruction::ExtractElement) {
2602 InstructionCost ExtendCost
2603 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2605 InstructionCost ExtractCost
2606 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2607 return ExtendCost + ExtractCost;
2608 }
2609 InstructionCost ExtendCost
2610 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2612 InstructionCost InsertCost
2613 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2614 InstructionCost TruncCost
2615 = getCastInstrCost(Instruction::Trunc, Val, WideTy,
2617 return ExtendCost + InsertCost + TruncCost;
2618 }
2619
2620
2621 // In RVV, we could use vslidedown + vmv.x.s to extract element from vector
2622 // and vslideup + vmv.s.x to insert element to vector.
2623 unsigned MoveOpc;
2624 if (LT.second.isFloatingPoint())
2625 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2626 : RISCV::VFMV_F_S;
2627 else
2628 MoveOpc =
2629 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2630 InstructionCost BaseCost =
2631 getRISCVInstructionCost(MoveOpc, LT.second, CostKind);
2632 // When insertelement we should add the index with 1 as the input of vslideup.
2633 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2634
2635 if (Index != -1U) {
2636 // The type may be split. For fixed-width vectors we can normalize the
2637 // index to the new type.
2638 if (LT.second.isFixedLengthVector()) {
2639 unsigned Width = LT.second.getVectorNumElements();
2640 Index = Index % Width;
2641 }
2642
2643 // If exact VLEN is known, we will insert/extract into the appropriate
2644 // subvector with no additional subvector insert/extract cost.
2645 if (auto VLEN = ST->getRealVLen()) {
2646 unsigned EltSize = LT.second.getScalarSizeInBits();
2647 unsigned M1Max = *VLEN / EltSize;
2648 Index = Index % M1Max;
2649 }
2650
2651 if (Index == 0)
2652 // We can extract/insert the first element without vslidedown/vslideup.
2653 SlideCost = 0;
2654 else if (Opcode == Instruction::InsertElement)
2655 SlideCost = 1; // With a constant index, we do not need to use addi.
2656 }
2657
2658 // When the vector needs to split into multiple register groups and the index
2659 // exceeds single vector register group, we need to insert/extract the element
2660 // via stack.
2661 if (LT.first > 1 &&
2662 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2663 LT.second.isScalableVector()))) {
2664 Type *ScalarType = Val->getScalarType();
2665 Align VecAlign = DL.getPrefTypeAlign(Val);
2666 Align SclAlign = DL.getPrefTypeAlign(ScalarType);
2667 // Extra addi for unknown index.
2668 InstructionCost IdxCost = Index == -1U ? 1 : 0;
2669
2670 // Store all split vectors into stack and load the target element.
2671 if (Opcode == Instruction::ExtractElement)
2672 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2673 getMemoryOpCost(Instruction::Load, ScalarType, SclAlign, 0,
2674 CostKind) +
2675 IdxCost;
2676
2677 // Store all split vectors into stack and store the target element and load
2678 // vectors back.
2679 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2680 getMemoryOpCost(Instruction::Load, Val, VecAlign, 0, CostKind) +
2681 getMemoryOpCost(Instruction::Store, ScalarType, SclAlign, 0,
2682 CostKind) +
2683 IdxCost;
2684 }
2685
2686 // Extract i64 in the target that has XLEN=32 need more instruction.
2687 if (Val->getScalarType()->isIntegerTy() &&
2688 ST->getXLen() < Val->getScalarSizeInBits()) {
2689 // For extractelement, we need the following instructions:
2690 // vsetivli zero, 1, e64, m1, ta, mu (not count)
2691 // vslidedown.vx v8, v8, a0
2692 // vmv.x.s a0, v8
2693 // li a1, 32
2694 // vsrl.vx v8, v8, a1
2695 // vmv.x.s a1, v8
2696
2697 // For insertelement, we need the following instructions:
2698 // vsetivli zero, 2, e32, m4, ta, ma (don't count)
2699 // vslide1down.vx v12, v8, a0
2700 // vslide1down.vx v12, v12, a1
2701 // addi a0, a2, 1
2702 // vsetvli zero, a0, e64, m4, tu, ma (don't count)
2703 // vslideup.vx v8, v12, a2
2704
2705 // TODO: should we count these special vsetvlis?
2706 BaseCost =
2707 Opcode == Instruction::InsertElement
2708 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2709 RISCV::VSLIDE1DOWN_VX,
2710 RISCV::VSLIDEUP_VX},
2711 LT.second, CostKind)
2712 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2713 RISCV::VSRL_VX, RISCV::VMV_X_S},
2714 LT.second, CostKind);
2715 }
2716 return BaseCost + SlideCost;
2717}
2718
2722 unsigned Index) const {
2723 if (isa<FixedVectorType>(Val))
2725 Index);
2726
2727 // TODO: This code replicates what LoopVectorize.cpp used to do when asking
2728 // for the cost of extracting the last lane of a scalable vector. It probably
2729 // needs a more accurate cost.
2730 ElementCount EC = cast<VectorType>(Val)->getElementCount();
2731 assert(Index < EC.getKnownMinValue() && "Unexpected reverse index");
2732 return getVectorInstrCost(Opcode, Val, CostKind,
2733 EC.getKnownMinValue() - 1 - Index, nullptr,
2734 nullptr);
2735}
2736
2737/// Check to see if this instruction is expected to be combined to a simpler
2738/// operation during/before lowering. If so return the cost of the combined
2739/// operation rather than provided one. For instance, `udiv i16 %X, 2` is likely
2740/// to be combined to `lshr i16 %X, 1`, so return the cost of a `lshr` rather
2741/// than the cost of a `udiv`
2742std::optional<InstructionCost>
2744 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2746 ArrayRef<const Value *> Args, const Instruction *CxtI) const {
2747 // Vector unsigned division/remainder will be simplified to shifts/masks.
2748 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2749 Opd2Info.isConstant() && Opd2Info.isPowerOf2()) {
2750 if (Opcode == Instruction::UDiv)
2751 return getArithmeticInstrCost(Instruction::LShr, Ty, CostKind, Opd1Info,
2752 Opd2Info.getNoProps());
2753 // UREM
2754 return getArithmeticInstrCost(Instruction::And, Ty, CostKind, Opd1Info,
2755 Opd2Info.getNoProps());
2756 }
2757 return std::nullopt;
2758}
2759
2761 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2763 ArrayRef<const Value *> Args, const Instruction *CxtI) const {
2764
2765 // TODO: Handle more cost kinds.
2767 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2768 Args, CxtI);
2769
2770 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2771 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2772 Args, CxtI);
2773
2774 // Skip if scalar size of Ty is bigger than ELEN.
2775 if (isa<VectorType>(Ty) && Ty->getScalarSizeInBits() > ST->getELen())
2776 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2777 Args, CxtI);
2778
2779 if (std::optional<InstructionCost> CombinedCost =
2781 Op2Info, Args, CxtI))
2782 return *CombinedCost;
2783
2784 // Legalize the type.
2785 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2786 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2787
2788 // TODO: Handle scalar type.
2789 if (!LT.second.isVector()) {
2790 static const CostTblEntry DivTbl[]{
2791 {ISD::UDIV, MVT::i32, TTI::TCC_Expensive},
2792 {ISD::UDIV, MVT::i64, TTI::TCC_Expensive},
2793 {ISD::SDIV, MVT::i32, TTI::TCC_Expensive},
2794 {ISD::SDIV, MVT::i64, TTI::TCC_Expensive},
2795 {ISD::UREM, MVT::i32, TTI::TCC_Expensive},
2796 {ISD::UREM, MVT::i64, TTI::TCC_Expensive},
2797 {ISD::SREM, MVT::i32, TTI::TCC_Expensive},
2798 {ISD::SREM, MVT::i64, TTI::TCC_Expensive}};
2799 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
2800 if (const auto *Entry = CostTableLookup(DivTbl, ISDOpcode, LT.second))
2801 return Entry->Cost * LT.first;
2802
2803 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2804 Args, CxtI);
2805 }
2806
2807 // f16 with zvfhmin and bf16 will be promoted to f32.
2808 // FIXME: nxv32[b]f16 will be custom lowered and split.
2809 InstructionCost CastCost = 0;
2810 if ((LT.second.getVectorElementType() == MVT::f16 ||
2811 LT.second.getVectorElementType() == MVT::bf16) &&
2812 TLI->getOperationAction(ISDOpcode, LT.second) ==
2814 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
2815 Type *PromotedTy = EVT(PromotedVT).getTypeForEVT(Ty->getContext());
2816 Type *LegalTy = EVT(LT.second).getTypeForEVT(Ty->getContext());
2817 // Add cost of extending arguments
2818 CastCost += LT.first * Args.size() *
2819 getCastInstrCost(Instruction::FPExt, PromotedTy, LegalTy,
2821 // Add cost of truncating result
2822 CastCost +=
2823 LT.first * getCastInstrCost(Instruction::FPTrunc, LegalTy, PromotedTy,
2825 // Compute cost of op in promoted type
2826 LT.second = PromotedVT;
2827 }
2828
2829 auto getConstantMatCost =
2830 [&](unsigned Operand, TTI::OperandValueInfo OpInfo) -> InstructionCost {
2831 if (OpInfo.isUniform() && canSplatOperand(Opcode, Operand))
2832 // Two sub-cases:
2833 // * Has a 5 bit immediate operand which can be splatted.
2834 // * Has a larger immediate which must be materialized in scalar register
2835 // We return 0 for both as we currently ignore the cost of materializing
2836 // scalar constants in GPRs.
2837 return 0;
2838
2839 return getConstantPoolLoadCost(Ty, CostKind);
2840 };
2841
2842 // Add the cost of materializing any constant vectors required.
2843 InstructionCost ConstantMatCost = 0;
2844 if (Op1Info.isConstant())
2845 ConstantMatCost += getConstantMatCost(0, Op1Info);
2846 if (Op2Info.isConstant())
2847 ConstantMatCost += getConstantMatCost(1, Op2Info);
2848
2849 unsigned Op;
2850 switch (ISDOpcode) {
2851 case ISD::ADD:
2852 case ISD::SUB:
2853 Op = RISCV::VADD_VV;
2854 break;
2855 case ISD::SHL:
2856 case ISD::SRL:
2857 case ISD::SRA:
2858 Op = RISCV::VSLL_VV;
2859 break;
2860 case ISD::AND:
2861 case ISD::OR:
2862 case ISD::XOR:
2863 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
2864 break;
2865 case ISD::MUL:
2866 case ISD::MULHS:
2867 case ISD::MULHU:
2868 Op = RISCV::VMUL_VV;
2869 break;
2870 case ISD::SDIV:
2871 case ISD::UDIV:
2872 Op = RISCV::VDIV_VV;
2873 break;
2874 case ISD::SREM:
2875 case ISD::UREM:
2876 Op = RISCV::VREM_VV;
2877 break;
2878 case ISD::FADD:
2879 case ISD::FSUB:
2880 Op = RISCV::VFADD_VV;
2881 break;
2882 case ISD::FMUL:
2883 Op = RISCV::VFMUL_VV;
2884 break;
2885 case ISD::FDIV:
2886 Op = RISCV::VFDIV_VV;
2887 break;
2888 case ISD::FNEG:
2889 Op = RISCV::VFSGNJN_VV;
2890 break;
2891 default:
2892 // Assuming all other instructions have the same cost until a need arises to
2893 // differentiate them.
2894 return CastCost + ConstantMatCost +
2895 BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2896 Args, CxtI);
2897 }
2898
2899 InstructionCost InstrCost = getRISCVInstructionCost(Op, LT.second, CostKind);
2900 // We use BasicTTIImpl to calculate scalar costs, which assumes floating point
2901 // ops are twice as expensive as integer ops. Do the same for vectors so
2902 // scalar floating point ops aren't cheaper than their vector equivalents.
2903 if (Ty->isFPOrFPVectorTy())
2904 InstrCost *= 2;
2905 return CastCost + ConstantMatCost + LT.first * InstrCost;
2906}
2907
2908// TODO: Deduplicate from TargetTransformInfoImplCRTPBase.
2910 ArrayRef<const Value *> Ptrs, const Value *Base,
2911 const TTI::PointersChainInfo &Info, Type *AccessTy,
2912 const TTI::TargetCostKind CostKind) const {
2914 // In the basic model we take into account GEP instructions only
2915 // (although here can come alloca instruction, a value, constants and/or
2916 // constant expressions, PHIs, bitcasts ... whatever allowed to be used as a
2917 // pointer). Typically, if Base is a not a GEP-instruction and all the
2918 // pointers are relative to the same base address, all the rest are
2919 // either GEP instructions, PHIs, bitcasts or constants. When we have same
2920 // base, we just calculate cost of each non-Base GEP as an ADD operation if
2921 // any their index is a non-const.
2922 // If no known dependencies between the pointers cost is calculated as a sum
2923 // of costs of GEP instructions.
2924 for (auto [I, V] : enumerate(Ptrs)) {
2925 const auto *GEP = dyn_cast<GetElementPtrInst>(V);
2926 if (!GEP)
2927 continue;
2928 if (Info.isSameBase() && V != Base) {
2929 if (GEP->hasAllConstantIndices())
2930 continue;
2931 // If the chain is unit-stride and BaseReg + stride*i is a legal
2932 // addressing mode, then presume the base GEP is sitting around in a
2933 // register somewhere and check if we can fold the offset relative to
2934 // it.
2935 unsigned Stride = DL.getTypeStoreSize(AccessTy);
2936 if (Info.isUnitStride() &&
2937 isLegalAddressingMode(AccessTy,
2938 /* BaseGV */ nullptr,
2939 /* BaseOffset */ Stride * I,
2940 /* HasBaseReg */ true,
2941 /* Scale */ 0,
2942 GEP->getType()->getPointerAddressSpace()))
2943 continue;
2944 Cost += getArithmeticInstrCost(Instruction::Add, GEP->getType(), CostKind,
2945 {TTI::OK_AnyValue, TTI::OP_None},
2946 {TTI::OK_AnyValue, TTI::OP_None}, {});
2947 } else {
2948 SmallVector<const Value *> Indices(GEP->indices());
2949 Cost += getGEPCost(GEP->getSourceElementType(), GEP->getPointerOperand(),
2950 Indices, AccessTy, CostKind);
2951 }
2952 }
2953 return Cost;
2954}
2955
2958 OptimizationRemarkEmitter *ORE) const {
2959 // TODO: More tuning on benchmarks and metrics with changes as needed
2960 // would apply to all settings below to enable performance.
2961
2962
2963 if (ST->enableDefaultUnroll())
2964 return BasicTTIImplBase::getUnrollingPreferences(L, SE, UP, ORE);
2965
2966 // Enable Upper bound unrolling universally, not dependent upon the conditions
2967 // below.
2968 UP.UpperBound = true;
2969
2970 // Disable loop unrolling for Oz and Os.
2971 UP.OptSizeThreshold = 0;
2973 if (L->getHeader()->getParent()->hasOptSize())
2974 return;
2975
2976 SmallVector<BasicBlock *, 4> ExitingBlocks;
2977 L->getExitingBlocks(ExitingBlocks);
2978 LLVM_DEBUG(dbgs() << "Loop has:\n"
2979 << "Blocks: " << L->getNumBlocks() << "\n"
2980 << "Exit blocks: " << ExitingBlocks.size() << "\n");
2981
2982 // Only allow another exit other than the latch. This acts as an early exit
2983 // as it mirrors the profitability calculation of the runtime unroller.
2984 if (ExitingBlocks.size() > 2)
2985 return;
2986
2987 // Limit the CFG of the loop body for targets with a branch predictor.
2988 // Allowing 4 blocks permits if-then-else diamonds in the body.
2989 if (L->getNumBlocks() > 4)
2990 return;
2991
2992 // Scan the loop: don't unroll loops with calls as this could prevent
2993 // inlining. Don't unroll auto-vectorized loops either, though do allow
2994 // unrolling of the scalar remainder.
2995 bool IsVectorized = getBooleanLoopAttribute(L, "llvm.loop.isvectorized");
2997 for (auto *BB : L->getBlocks()) {
2998 for (auto &I : *BB) {
2999 // Both auto-vectorized loops and the scalar remainder have the
3000 // isvectorized attribute, so differentiate between them by the presence
3001 // of vector instructions.
3002 if (IsVectorized && (I.getType()->isVectorTy() ||
3003 llvm::any_of(I.operand_values(), [](Value *V) {
3004 return V->getType()->isVectorTy();
3005 })))
3006 return;
3007
3008 if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
3009 if (const Function *F = cast<CallBase>(I).getCalledFunction()) {
3010 if (!isLoweredToCall(F))
3011 continue;
3012 }
3013 return;
3014 }
3015
3016 SmallVector<const Value *> Operands(I.operand_values());
3019 }
3020 }
3021
3022 LLVM_DEBUG(dbgs() << "Cost of loop: " << Cost << "\n");
3023
3024 UP.Partial = true;
3025 UP.Runtime = true;
3026 UP.UnrollRemainder = true;
3027 UP.UnrollAndJam = true;
3028
3029 // Force unrolling small loops can be very useful because of the branch
3030 // taken cost of the backedge.
3031 if (Cost < 12)
3032 UP.Force = true;
3033}
3034
3039
3041 MemIntrinsicInfo &Info) const {
3042 const DataLayout &DL = getDataLayout();
3043 Intrinsic::ID IID = Inst->getIntrinsicID();
3044 LLVMContext &C = Inst->getContext();
3045 bool HasMask = false;
3046
3047 auto getSegNum = [](const IntrinsicInst *II, unsigned PtrOperandNo,
3048 bool IsWrite) -> int64_t {
3049 if (auto *TarExtTy =
3050 dyn_cast<TargetExtType>(II->getArgOperand(0)->getType()))
3051 return TarExtTy->getIntParameter(0);
3052
3053 return 1;
3054 };
3055
3056 switch (IID) {
3057 case Intrinsic::riscv_vle_mask:
3058 case Intrinsic::riscv_vse_mask:
3059 case Intrinsic::riscv_vlseg2_mask:
3060 case Intrinsic::riscv_vlseg3_mask:
3061 case Intrinsic::riscv_vlseg4_mask:
3062 case Intrinsic::riscv_vlseg5_mask:
3063 case Intrinsic::riscv_vlseg6_mask:
3064 case Intrinsic::riscv_vlseg7_mask:
3065 case Intrinsic::riscv_vlseg8_mask:
3066 case Intrinsic::riscv_vsseg2_mask:
3067 case Intrinsic::riscv_vsseg3_mask:
3068 case Intrinsic::riscv_vsseg4_mask:
3069 case Intrinsic::riscv_vsseg5_mask:
3070 case Intrinsic::riscv_vsseg6_mask:
3071 case Intrinsic::riscv_vsseg7_mask:
3072 case Intrinsic::riscv_vsseg8_mask:
3073 HasMask = true;
3074 [[fallthrough]];
3075 case Intrinsic::riscv_vle:
3076 case Intrinsic::riscv_vse:
3077 case Intrinsic::riscv_vlseg2:
3078 case Intrinsic::riscv_vlseg3:
3079 case Intrinsic::riscv_vlseg4:
3080 case Intrinsic::riscv_vlseg5:
3081 case Intrinsic::riscv_vlseg6:
3082 case Intrinsic::riscv_vlseg7:
3083 case Intrinsic::riscv_vlseg8:
3084 case Intrinsic::riscv_vsseg2:
3085 case Intrinsic::riscv_vsseg3:
3086 case Intrinsic::riscv_vsseg4:
3087 case Intrinsic::riscv_vsseg5:
3088 case Intrinsic::riscv_vsseg6:
3089 case Intrinsic::riscv_vsseg7:
3090 case Intrinsic::riscv_vsseg8: {
3091 // Intrinsic interface:
3092 // riscv_vle(merge, ptr, vl)
3093 // riscv_vle_mask(merge, ptr, mask, vl, policy)
3094 // riscv_vse(val, ptr, vl)
3095 // riscv_vse_mask(val, ptr, mask, vl, policy)
3096 // riscv_vlseg#(merge, ptr, vl, sew)
3097 // riscv_vlseg#_mask(merge, ptr, mask, vl, policy, sew)
3098 // riscv_vsseg#(val, ptr, vl, sew)
3099 // riscv_vsseg#_mask(val, ptr, mask, vl, sew)
3100 bool IsWrite = Inst->getType()->isVoidTy();
3101 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3102 // The results of segment loads are TargetExtType.
3103 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3104 unsigned SEW =
3105 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3106 ->getZExtValue();
3107 Ty = TarExtTy->getTypeParameter(0U);
3109 IntegerType::get(C, SEW),
3110 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3111 }
3112 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3113 unsigned VLIndex = RVVIInfo->VLOperand;
3114 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3115 MaybeAlign Alignment =
3116 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3117 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3118 Value *Mask = ConstantInt::getTrue(MaskType);
3119 if (HasMask)
3120 Mask = Inst->getArgOperand(VLIndex - 1);
3121 Value *EVL = Inst->getArgOperand(VLIndex);
3122 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3123 // RVV uses contiguous elements as a segment.
3124 if (SegNum > 1) {
3125 unsigned ElemSize = Ty->getScalarSizeInBits();
3126 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3127 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3128 }
3129 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3130 Alignment, Mask, EVL);
3131 return true;
3132 }
3133 case Intrinsic::riscv_vlse_mask:
3134 case Intrinsic::riscv_vsse_mask:
3135 case Intrinsic::riscv_vlsseg2_mask:
3136 case Intrinsic::riscv_vlsseg3_mask:
3137 case Intrinsic::riscv_vlsseg4_mask:
3138 case Intrinsic::riscv_vlsseg5_mask:
3139 case Intrinsic::riscv_vlsseg6_mask:
3140 case Intrinsic::riscv_vlsseg7_mask:
3141 case Intrinsic::riscv_vlsseg8_mask:
3142 case Intrinsic::riscv_vssseg2_mask:
3143 case Intrinsic::riscv_vssseg3_mask:
3144 case Intrinsic::riscv_vssseg4_mask:
3145 case Intrinsic::riscv_vssseg5_mask:
3146 case Intrinsic::riscv_vssseg6_mask:
3147 case Intrinsic::riscv_vssseg7_mask:
3148 case Intrinsic::riscv_vssseg8_mask:
3149 HasMask = true;
3150 [[fallthrough]];
3151 case Intrinsic::riscv_vlse:
3152 case Intrinsic::riscv_vsse:
3153 case Intrinsic::riscv_vlsseg2:
3154 case Intrinsic::riscv_vlsseg3:
3155 case Intrinsic::riscv_vlsseg4:
3156 case Intrinsic::riscv_vlsseg5:
3157 case Intrinsic::riscv_vlsseg6:
3158 case Intrinsic::riscv_vlsseg7:
3159 case Intrinsic::riscv_vlsseg8:
3160 case Intrinsic::riscv_vssseg2:
3161 case Intrinsic::riscv_vssseg3:
3162 case Intrinsic::riscv_vssseg4:
3163 case Intrinsic::riscv_vssseg5:
3164 case Intrinsic::riscv_vssseg6:
3165 case Intrinsic::riscv_vssseg7:
3166 case Intrinsic::riscv_vssseg8: {
3167 // Intrinsic interface:
3168 // riscv_vlse(merge, ptr, stride, vl)
3169 // riscv_vlse_mask(merge, ptr, stride, mask, vl, policy)
3170 // riscv_vsse(val, ptr, stride, vl)
3171 // riscv_vsse_mask(val, ptr, stride, mask, vl, policy)
3172 // riscv_vlsseg#(merge, ptr, offset, vl, sew)
3173 // riscv_vlsseg#_mask(merge, ptr, offset, mask, vl, policy, sew)
3174 // riscv_vssseg#(val, ptr, offset, vl, sew)
3175 // riscv_vssseg#_mask(val, ptr, offset, mask, vl, sew)
3176 bool IsWrite = Inst->getType()->isVoidTy();
3177 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3178 // The results of segment loads are TargetExtType.
3179 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3180 unsigned SEW =
3181 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3182 ->getZExtValue();
3183 Ty = TarExtTy->getTypeParameter(0U);
3185 IntegerType::get(C, SEW),
3186 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3187 }
3188 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3189 unsigned VLIndex = RVVIInfo->VLOperand;
3190 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3191 MaybeAlign Alignment =
3192 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3193
3194 Value *Stride = Inst->getArgOperand(PtrOperandNo + 1);
3195 // Use the pointer alignment as the element alignment if the stride is a
3196 // multiple of the pointer alignment. Otherwise, the element alignment
3197 // should be the greatest common divisor of pointer alignment and stride.
3198 // For simplicity, just consider unalignment for elements.
3199 unsigned PointerAlign = Alignment.valueOrOne().value();
3200 if (!isa<ConstantInt>(Stride) ||
3201 cast<ConstantInt>(Stride)->getZExtValue() % PointerAlign != 0)
3202 Alignment = Align(1);
3203
3204 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3205 Value *Mask = ConstantInt::getTrue(MaskType);
3206 if (HasMask)
3207 Mask = Inst->getArgOperand(VLIndex - 1);
3208 Value *EVL = Inst->getArgOperand(VLIndex);
3209 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3210 // RVV uses contiguous elements as a segment.
3211 if (SegNum > 1) {
3212 unsigned ElemSize = Ty->getScalarSizeInBits();
3213 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3214 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3215 }
3216 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3217 Alignment, Mask, EVL, Stride);
3218 return true;
3219 }
3220 case Intrinsic::riscv_vloxei_mask:
3221 case Intrinsic::riscv_vluxei_mask:
3222 case Intrinsic::riscv_vsoxei_mask:
3223 case Intrinsic::riscv_vsuxei_mask:
3224 case Intrinsic::riscv_vloxseg2_mask:
3225 case Intrinsic::riscv_vloxseg3_mask:
3226 case Intrinsic::riscv_vloxseg4_mask:
3227 case Intrinsic::riscv_vloxseg5_mask:
3228 case Intrinsic::riscv_vloxseg6_mask:
3229 case Intrinsic::riscv_vloxseg7_mask:
3230 case Intrinsic::riscv_vloxseg8_mask:
3231 case Intrinsic::riscv_vluxseg2_mask:
3232 case Intrinsic::riscv_vluxseg3_mask:
3233 case Intrinsic::riscv_vluxseg4_mask:
3234 case Intrinsic::riscv_vluxseg5_mask:
3235 case Intrinsic::riscv_vluxseg6_mask:
3236 case Intrinsic::riscv_vluxseg7_mask:
3237 case Intrinsic::riscv_vluxseg8_mask:
3238 case Intrinsic::riscv_vsoxseg2_mask:
3239 case Intrinsic::riscv_vsoxseg3_mask:
3240 case Intrinsic::riscv_vsoxseg4_mask:
3241 case Intrinsic::riscv_vsoxseg5_mask:
3242 case Intrinsic::riscv_vsoxseg6_mask:
3243 case Intrinsic::riscv_vsoxseg7_mask:
3244 case Intrinsic::riscv_vsoxseg8_mask:
3245 case Intrinsic::riscv_vsuxseg2_mask:
3246 case Intrinsic::riscv_vsuxseg3_mask:
3247 case Intrinsic::riscv_vsuxseg4_mask:
3248 case Intrinsic::riscv_vsuxseg5_mask:
3249 case Intrinsic::riscv_vsuxseg6_mask:
3250 case Intrinsic::riscv_vsuxseg7_mask:
3251 case Intrinsic::riscv_vsuxseg8_mask:
3252 HasMask = true;
3253 [[fallthrough]];
3254 case Intrinsic::riscv_vloxei:
3255 case Intrinsic::riscv_vluxei:
3256 case Intrinsic::riscv_vsoxei:
3257 case Intrinsic::riscv_vsuxei:
3258 case Intrinsic::riscv_vloxseg2:
3259 case Intrinsic::riscv_vloxseg3:
3260 case Intrinsic::riscv_vloxseg4:
3261 case Intrinsic::riscv_vloxseg5:
3262 case Intrinsic::riscv_vloxseg6:
3263 case Intrinsic::riscv_vloxseg7:
3264 case Intrinsic::riscv_vloxseg8:
3265 case Intrinsic::riscv_vluxseg2:
3266 case Intrinsic::riscv_vluxseg3:
3267 case Intrinsic::riscv_vluxseg4:
3268 case Intrinsic::riscv_vluxseg5:
3269 case Intrinsic::riscv_vluxseg6:
3270 case Intrinsic::riscv_vluxseg7:
3271 case Intrinsic::riscv_vluxseg8:
3272 case Intrinsic::riscv_vsoxseg2:
3273 case Intrinsic::riscv_vsoxseg3:
3274 case Intrinsic::riscv_vsoxseg4:
3275 case Intrinsic::riscv_vsoxseg5:
3276 case Intrinsic::riscv_vsoxseg6:
3277 case Intrinsic::riscv_vsoxseg7:
3278 case Intrinsic::riscv_vsoxseg8:
3279 case Intrinsic::riscv_vsuxseg2:
3280 case Intrinsic::riscv_vsuxseg3:
3281 case Intrinsic::riscv_vsuxseg4:
3282 case Intrinsic::riscv_vsuxseg5:
3283 case Intrinsic::riscv_vsuxseg6:
3284 case Intrinsic::riscv_vsuxseg7:
3285 case Intrinsic::riscv_vsuxseg8: {
3286 // Intrinsic interface (only listed ordered version):
3287 // riscv_vloxei(merge, ptr, index, vl)
3288 // riscv_vloxei_mask(merge, ptr, index, mask, vl, policy)
3289 // riscv_vsoxei(val, ptr, index, vl)
3290 // riscv_vsoxei_mask(val, ptr, index, mask, vl, policy)
3291 // riscv_vloxseg#(merge, ptr, index, vl, sew)
3292 // riscv_vloxseg#_mask(merge, ptr, index, mask, vl, policy, sew)
3293 // riscv_vsoxseg#(val, ptr, index, vl, sew)
3294 // riscv_vsoxseg#_mask(val, ptr, index, mask, vl, sew)
3295 bool IsWrite = Inst->getType()->isVoidTy();
3296 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3297 // The results of segment loads are TargetExtType.
3298 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3299 unsigned SEW =
3300 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3301 ->getZExtValue();
3302 Ty = TarExtTy->getTypeParameter(0U);
3304 IntegerType::get(C, SEW),
3305 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3306 }
3307 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3308 unsigned VLIndex = RVVIInfo->VLOperand;
3309 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3310 Value *Mask;
3311 if (HasMask) {
3312 Mask = Inst->getArgOperand(VLIndex - 1);
3313 } else {
3314 // Mask cannot be nullptr here: vector GEP produces <vscale x N x ptr>,
3315 // and casting that to scalar i64 triggers a vector/scalar mismatch
3316 // assertion in CreatePointerCast. Use an all-true mask so ASan lowers it
3317 // via extractelement instead.
3318 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3319 Mask = ConstantInt::getTrue(MaskType);
3320 }
3321 Value *EVL = Inst->getArgOperand(VLIndex);
3322 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3323 // RVV uses contiguous elements as a segment.
3324 if (SegNum > 1) {
3325 unsigned ElemSize = Ty->getScalarSizeInBits();
3326 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3327 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3328 }
3329 Value *OffsetOp = Inst->getArgOperand(PtrOperandNo + 1);
3330 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3331 Align(1), Mask, EVL,
3332 /* Stride */ nullptr, OffsetOp);
3333 return true;
3334 }
3335 }
3336 return false;
3337}
3338
3340 if (Ty->isVectorTy()) {
3341 // f16 with only zvfhmin and bf16 will be promoted to f32
3342 Type *EltTy = cast<VectorType>(Ty)->getElementType();
3343 if ((EltTy->isHalfTy() && !ST->hasVInstructionsF16()) ||
3344 EltTy->isBFloatTy())
3345 Ty = VectorType::get(Type::getFloatTy(Ty->getContext()),
3346 cast<VectorType>(Ty));
3347
3348 TypeSize Size = DL.getTypeSizeInBits(Ty);
3349 if (Size.isScalable() && ST->hasVInstructions())
3350 return divideCeil(Size.getKnownMinValue(), RISCV::RVVBitsPerBlock);
3351
3352 if (ST->useRVVForFixedLengthVectors())
3353 return divideCeil(Size, ST->getRealMinVLen());
3354 }
3355
3356 return BaseT::getRegUsageForType(Ty);
3357}
3358
3359unsigned RISCVTTIImpl::getMaximumVF(unsigned ElemWidth, unsigned Opcode) const {
3360 if (SLPMaxVF.getNumOccurrences())
3361 return SLPMaxVF;
3362
3363 // Return how many elements can fit in getRegisterBitwidth. This is the
3364 // same routine as used in LoopVectorizer. We should probably be
3365 // accounting for whether we actually have instructions with the right
3366 // lane type, but we don't have enough information to do that without
3367 // some additional plumbing which hasn't been justified yet.
3368 TypeSize RegWidth =
3370 // If no vector registers, or absurd element widths, disable
3371 // vectorization by returning 1.
3372 return std::max<unsigned>(1U, RegWidth.getFixedValue() / ElemWidth);
3373}
3374
3378
3380 return ST->enableUnalignedVectorMem();
3381}
3382
3385 ScalarEvolution *SE) const {
3386 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3387 return TTI::AMK_PostIndexed;
3388
3390}
3391
3393 const TargetTransformInfo::LSRCost &C2) const {
3394 // RISC-V specific here are "instruction number 1st priority".
3395 // If we need to emit adds inside the loop to add up base registers, then
3396 // we need at least one extra temporary register.
3397 unsigned C1NumRegs = C1.NumRegs + (C1.NumBaseAdds != 0);
3398 unsigned C2NumRegs = C2.NumRegs + (C2.NumBaseAdds != 0);
3399 return std::tie(C1.Insns, C1NumRegs, C1.AddRecCost,
3400 C1.NumIVMuls, C1.NumBaseAdds,
3401 C1.ScaleCost, C1.ImmCost, C1.SetupCost) <
3402 std::tie(C2.Insns, C2NumRegs, C2.AddRecCost,
3403 C2.NumIVMuls, C2.NumBaseAdds,
3404 C2.ScaleCost, C2.ImmCost, C2.SetupCost);
3405}
3406
3408 Align Alignment) const {
3409 auto *VTy = dyn_cast<VectorType>(DataTy);
3410 if (!VTy || VTy->isScalableTy())
3411 return false;
3412
3413 if (!isLegalMaskedLoadStore(DataTy, Alignment))
3414 return false;
3415
3416 // FIXME: If it is an i8 vector and the element count exceeds 256, we should
3417 // scalarize these types with LMUL >= maximum fixed-length LMUL.
3418 if (VTy->getElementType()->isIntegerTy(8))
3419 if (VTy->getElementCount().getFixedValue() > 256)
3420 return VTy->getPrimitiveSizeInBits() / ST->getRealMinVLen() <
3421 ST->getMaxLMULForFixedLengthVectors();
3422 return true;
3423}
3424
3426 Align Alignment) const {
3427 auto *VTy = dyn_cast<VectorType>(DataTy);
3428 if (!VTy || VTy->isScalableTy())
3429 return false;
3430
3431 if (!isLegalMaskedLoadStore(DataTy, Alignment))
3432 return false;
3433 return true;
3434}
3435
3437 ElementCount NumElements) const {
3438 // Optimized zero-stride loads can be treated as broadcasts.
3439 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3440 return false;
3441
3442 return TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, ElementTy));
3443}
3444
3445/// See if \p I should be considered for address type promotion. We check if \p
3446/// I is a sext with right type and used in memory accesses. If it used in a
3447/// "complex" getelementptr, we allow it to be promoted without finding other
3448/// sext instructions that sign extended the same initial value. A getelementptr
3449/// is considered as "complex" if it has more than 2 operands.
3451 const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const {
3452 bool Considerable = false;
3453 AllowPromotionWithoutCommonHeader = false;
3454 if (!isa<SExtInst>(&I))
3455 return false;
3456 Type *ConsideredSExtType =
3457 Type::getInt64Ty(I.getParent()->getParent()->getContext());
3458 if (I.getType() != ConsideredSExtType)
3459 return false;
3460 // See if the sext is the one with the right type and used in at least one
3461 // GetElementPtrInst.
3462 for (const User *U : I.users()) {
3463 if (const GetElementPtrInst *GEPInst = dyn_cast<GetElementPtrInst>(U)) {
3464 Considerable = true;
3465 // A getelementptr is considered as "complex" if it has more than 2
3466 // operands. We will promote a SExt used in such complex GEP as we
3467 // expect some computation to be merged if they are done on 64 bits.
3468 if (GEPInst->getNumOperands() > 2) {
3469 AllowPromotionWithoutCommonHeader = true;
3470 break;
3471 }
3472 }
3473 }
3474 return Considerable;
3475}
3476
3477bool RISCVTTIImpl::canSplatOperand(unsigned Opcode, int Operand) const {
3478 switch (Opcode) {
3479 case Instruction::Add:
3480 case Instruction::Sub:
3481 case Instruction::Mul:
3482 case Instruction::And:
3483 case Instruction::Or:
3484 case Instruction::Xor:
3485 case Instruction::FAdd:
3486 case Instruction::FSub:
3487 case Instruction::FMul:
3488 case Instruction::FDiv:
3489 case Instruction::ICmp:
3490 case Instruction::FCmp:
3491 return true;
3492 case Instruction::Shl:
3493 case Instruction::LShr:
3494 case Instruction::AShr:
3495 case Instruction::UDiv:
3496 case Instruction::SDiv:
3497 case Instruction::URem:
3498 case Instruction::SRem:
3499 case Instruction::Select:
3500 return Operand == 1;
3501 default:
3502 return false;
3503 }
3504}
3505
3507 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3508 return false;
3509
3510 if (canSplatOperand(I->getOpcode(), Operand))
3511 return true;
3512
3513 auto *II = dyn_cast<IntrinsicInst>(I);
3514 if (!II)
3515 return false;
3516
3517 switch (II->getIntrinsicID()) {
3518 case Intrinsic::fma:
3519 case Intrinsic::fmuladd:
3520 return Operand == 0 || Operand == 1;
3521 case Intrinsic::vp_udiv:
3522 case Intrinsic::vp_sdiv:
3523 case Intrinsic::vp_urem:
3524 case Intrinsic::vp_srem:
3525 case Intrinsic::ssub_sat:
3526 case Intrinsic::usub_sat:
3527 return Operand == 1;
3528 // These intrinsics are commutative.
3529 case Intrinsic::smin:
3530 case Intrinsic::umin:
3531 case Intrinsic::smax:
3532 case Intrinsic::umax:
3533 case Intrinsic::sadd_sat:
3534 case Intrinsic::uadd_sat:
3535 return Operand == 0 || Operand == 1;
3536 default:
3537 return false;
3538 }
3539}
3540
3541/// Check if sinking \p I's operands to I's basic block is profitable, because
3542/// the operands can be folded into a target instruction, e.g.
3543/// splats of scalars can fold into vector instructions.
3546 using namespace llvm::PatternMatch;
3547
3548 if (I->isBitwiseLogicOp()) {
3549 if (!I->getType()->isVectorTy()) {
3550 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3551 for (auto &Op : I->operands()) {
3552 // (and/or/xor X, (not Y)) -> (andn/orn/xnor X, Y)
3553 if (match(Op.get(), m_Not(m_Value()))) {
3554 Ops.push_back(&Op);
3555 return true;
3556 }
3557 }
3558 }
3559 } else if (I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3560 for (auto &Op : I->operands()) {
3561 // (and X, (not Y)) -> (vandn.vv X, Y)
3562 if (match(Op.get(), m_Not(m_Value()))) {
3563 Ops.push_back(&Op);
3564 return true;
3565 }
3566 // (and X, (splat (not Y))) -> (vandn.vx X, Y)
3568 m_ZeroInt()),
3569 m_Value(), m_ZeroMask()))) {
3570 Use &InsertElt = cast<Instruction>(Op)->getOperandUse(0);
3571 Use &Not = cast<Instruction>(InsertElt)->getOperandUse(1);
3572 Ops.push_back(&Not);
3573 Ops.push_back(&InsertElt);
3574 Ops.push_back(&Op);
3575 return true;
3576 }
3577 }
3578 }
3579 }
3580
3581 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3582 return false;
3583
3584 // Don't sink splat operands if the target prefers it. Some targets requires
3585 // S2V transfer buffers and we can run out of them copying the same value
3586 // repeatedly.
3587 // FIXME: It could still be worth doing if it would improve vector register
3588 // pressure and prevent a vector spill.
3589 if (!ST->sinkSplatOperands())
3590 return false;
3591
3592 for (auto OpIdx : enumerate(I->operands())) {
3593 if (!canSplatOperand(I, OpIdx.index()))
3594 continue;
3595
3596 Instruction *Op = dyn_cast<Instruction>(OpIdx.value().get());
3597 // Make sure we are not already sinking this operand
3598 if (!Op || any_of(Ops, [&](Use *U) { return U->get() == Op; }))
3599 continue;
3600
3601 // We are looking for a splat that can be sunk.
3603 m_Value(), m_ZeroMask())))
3604 continue;
3605
3606 // Don't sink i1 splats.
3607 if (cast<VectorType>(Op->getType())->getElementType()->isIntegerTy(1))
3608 continue;
3609
3610 // All uses of the shuffle should be sunk to avoid duplicating it across gpr
3611 // and vector registers
3612 for (Use &U : Op->uses()) {
3613 Instruction *Insn = cast<Instruction>(U.getUser());
3614 if (!canSplatOperand(Insn, U.getOperandNo()))
3615 return false;
3616 }
3617
3618 // Sink any fpexts since they might be used in a widening fp pattern.
3619 Use *InsertEltUse = &Op->getOperandUse(0);
3620 auto *InsertElt = cast<InsertElementInst>(InsertEltUse);
3621 if (isa<FPExtInst>(InsertElt->getOperand(1)))
3622 Ops.push_back(&InsertElt->getOperandUse(1));
3623 Ops.push_back(InsertEltUse);
3624 Ops.push_back(&OpIdx.value());
3625 }
3626 return true;
3627}
3628
3630RISCVTTIImpl::enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
3632
3633 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3634 return Options;
3635
3636 Options.AllowOverlappingLoads = true;
3637 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3638 Options.NumLoadsPerBlock = Options.MaxNumLoads;
3639 if (ST->is64Bit()) {
3640 Options.LoadSizes = {8, 4, 2, 1};
3641 Options.AllowedTailExpansions = {3, 5, 6};
3642 } else {
3643 Options.LoadSizes = {4, 2, 1};
3644 Options.AllowedTailExpansions = {3};
3645 }
3646
3647 if (IsZeroCmp && ST->hasVInstructions()) {
3648 unsigned VLenB = ST->getRealMinVLen() / 8;
3649 // The minimum size should be `XLen / 8 + 1`, and the maxinum size should be
3650 // `VLenB * MaxLMUL` so that it fits in a single register group.
3651 unsigned MinSize = ST->getXLen() / 8 + 1;
3652 unsigned MaxSize = VLenB * ST->getMaxLMULForFixedLengthVectors();
3653 for (unsigned Size = MinSize; Size <= MaxSize; Size++)
3654 Options.LoadSizes.insert(Options.LoadSizes.begin(), Size);
3655 }
3656 return Options;
3657}
3658
3660 const Instruction *I) const {
3662 // For the binary operators (e.g. or) we need to be more careful than
3663 // selects, here we only transform them if they are already at a natural
3664 // break point in the code - the end of a block with an unconditional
3665 // terminator.
3666 if (I->getOpcode() == Instruction::Or &&
3667 isa<UncondBrInst>(I->getNextNode()))
3668 return true;
3669
3670 if (I->getOpcode() == Instruction::Add ||
3671 I->getOpcode() == Instruction::Sub)
3672 return true;
3673 }
3675}
3676
3678 const Function *Caller, const Attribute &Attr) const {
3679 // "interrupt" controls the prolog/epilog of interrupt handlers (and includes
3680 // restrictions on their signatures). We can outline from the bodies of these
3681 // handlers, but when we do we need to make sure we don't mark the outlined
3682 // function as an interrupt handler too.
3683 if (Attr.isStringAttribute() && Attr.getKindAsString() == "interrupt")
3684 return false;
3685
3687}
3688
3689std::optional<Instruction *>
3691 // If all operands of a vmv.v.x are constant, fold a bitcast(vmv.v.x) to scale
3692 // the vmv.v.x, enabling removal of the bitcast. The transform helps avoid
3693 // creating redundant masks.
3694 const DataLayout &DL = IC.getDataLayout();
3695 if (II.user_empty())
3696 return {};
3697 auto *TargetVecTy = dyn_cast<ScalableVectorType>(II.user_back()->getType());
3698 if (!TargetVecTy)
3699 return {};
3700 const APInt *Scalar;
3701 uint64_t VL;
3703 m_Poison(), m_APInt(Scalar), m_ConstantInt(VL))) ||
3704 !all_of(II.users(), [TargetVecTy](User *U) {
3705 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
3706 }))
3707 return {};
3708 auto *SourceVecTy = cast<ScalableVectorType>(II.getType());
3709 unsigned TargetEltBW = DL.getTypeSizeInBits(TargetVecTy->getElementType());
3710 unsigned SourceEltBW = DL.getTypeSizeInBits(SourceVecTy->getElementType());
3711 if (TargetEltBW % SourceEltBW)
3712 return {};
3713 unsigned TargetScale = TargetEltBW / SourceEltBW;
3714 if (VL % TargetScale || TargetScale == 1)
3715 return {};
3716 Type *VLTy = II.getOperand(2)->getType();
3717 ElementCount SourceEC = SourceVecTy->getElementCount();
3718 unsigned NewEltBW = SourceEltBW * TargetScale;
3719 if (!SourceEC.isKnownMultipleOf(TargetScale) ||
3720 !DL.fitsInLegalInteger(NewEltBW))
3721 return {};
3722 auto *NewEltTy = IntegerType::get(II.getContext(), NewEltBW);
3723 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, NewEltTy)))
3724 return {};
3725 ElementCount NewEC = SourceEC.divideCoefficientBy(TargetScale);
3726 Type *RetTy = VectorType::get(NewEltTy, NewEC);
3727 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
3728 "Lossless bitcast between types expected");
3729 APInt NewScalar = APInt::getSplat(NewEltBW, *Scalar);
3730 return IC.replaceInstUsesWith(
3731 II,
3734 RetTy, Intrinsic::riscv_vmv_v_x,
3735 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
3736 ConstantInt::get(VLTy, VL / TargetScale)}),
3737 SourceVecTy));
3738}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > EnableOrLikeSelectOpt("enable-aarch64-or-like-select", cl::init(true), cl::Hidden)
unsigned Imm
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
Hexagon Common GEP
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static LVOptions Options
Definition LVOptions.cpp:25
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
static InstructionCost costShuffleViaVRegSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, std::optional< unsigned > VLen, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Try to perform better estimation of the permutation.
static InstructionCost costShuffleViaSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Attempt to approximate the cost of a shuffle which will require splitting during legalization.
static bool isRepeatedConcatMask(ArrayRef< int > Mask, int &SubVectorSize)
static unsigned isM1OrSmaller(MVT VT)
static cl::opt< bool > EnableOrLikeSelectOpt("enable-riscv-or-like-select", cl::init(true), cl::Hidden)
static cl::opt< unsigned > SLPMaxVF("riscv-v-slp-max-vf", cl::desc("Overrides result used for getMaximumVF query which is used " "exclusively by SLP vectorizer."), cl::Hidden)
static cl::opt< unsigned > RVVRegisterWidthLMUL("riscv-v-register-bit-width-lmul", cl::desc("The LMUL to use for getRegisterBitWidth queries. Affects LMUL used " "by autovectorized code. Fractional LMULs are not supported."), cl::init(2), cl::Hidden)
static cl::opt< unsigned > RVVMinTripCount("riscv-v-min-trip-count", cl::desc("Set the lower bound of a trip count to decide on " "vectorization while tail-folding."), cl::init(5), cl::Hidden)
static InstructionCost getIntImmCostImpl(const DataLayout &DL, const RISCVSubtarget *ST, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, bool FreeZeroes)
static VectorType * getVRGatherIndexType(MVT DataVT, const RISCVSubtarget &ST, LLVMContext &C)
static const CostTblEntry VectorIntrinsicCostTable[]
static bool canUseShiftPair(Instruction *Inst, const APInt &Imm)
static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm)
This file defines a TargetTransformInfoImplBase conforming object specific to the RISC-V target machi...
SI Fold Operands
static Type * getValueType(Value *V, bool LookThroughCmp=false)
Returns the "element type" of the given value/instruction V.
This file contains some templates that are useful if you are working with the STL at all.
#define LLVM_DEBUG(...)
Definition Debug.h:119
This file describes how to lower LLVM code to machine code.
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:647
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:197
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
iterator end() const
Definition ArrayRef.h:130
size_t size() const
Get the array size.
Definition ArrayRef.h:141
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:105
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, Type *AccessType, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, ArrayRef< int > Mask, TTI::TargetCostKind CostKind, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
std::optional< unsigned > getMaxVScale() const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
Definition InstrTypes.h:757
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
Definition InstrTypes.h:742
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static bool isFPPredicate(Predicate P)
Definition InstrTypes.h:833
static bool isIntPredicate(Predicate P)
Definition InstrTypes.h:839
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool noNaNs() const
Definition FMF.h:65
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:867
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2243
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
const SmallVectorImpl< Type * > & getArgTypes() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Machine Value Type.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
The optimization diagnostic interface.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CxtI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
std::optional< unsigned > getMaxVScale() const override
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, ArrayRef< int > Mask, TTI::TargetCostKind CostKind, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
Definition Type.cpp:889
The main scalar evolution driver.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
Definition DenseSet.h:293
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
virtual const DataLayout & getDataLayout() const
virtual bool shouldTreatInstructionLikeSelect(const Instruction *I) const
virtual TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const
virtual bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const
virtual bool isLoweredToCall(const Function *F) const
InstructionCost getInstructionCost(const User *U, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind) const override
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
PopcntSupportKind
Flags indicating the kind of support for population count.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Expensive
The cost of a 'div' instruction on x86.
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
AddressingModeKind
Which addressing mode Loop Strength Reduction will try to generate.
@ AMK_PostIndexed
Prefer post-indexed addressing mode.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:343
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:346
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:310
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:288
LLVM_ABI bool isScalableTy(SmallPtrSetImpl< const Type * > &Visited) const
Return true if this is a type whose size is a known multiple of vscale.
Definition Type.cpp:61
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:306
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:313
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:286
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
user_iterator user_begin()
Definition Value.h:402
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:439
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Definition Value.cpp:1002
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
Definition TypeSize.h:180
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:230
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
Definition ISDOpcodes.h:24
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:890
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:417
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:706
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:771
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:860
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:988
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:936
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:741
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
Definition CostTable.h:36
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
InstructionCost Cost
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ BinaryOp
One of the operands is a binary op.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
Definition STLExtras.h:1818
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
Definition STLExtras.h:1970
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
TargetTransformInfo TTI
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
Definition STLExtras.h:1885
CostTblEntryT< uint16_t > CostTblEntry
Definition CostTable.h:31
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2146
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
Extended Value Type.
Definition ValueTypes.h:35
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Align valueOrOne() const
For convenience, returns a valid alignment or 1 if undefined.
Definition Alignment.h:130
Information about a load/store intrinsic defined by the target.
unsigned Insns
TODO: Some of these could be merged.
Returns options for expansion of memcmp. IsZeroCmp is.
Describe known properties for a set of pointers.
Parameters that control the generic loop unrolling transformation.
bool UpperBound
Allow using trip count upper bound to unroll loops.
bool Force
Apply loop unroll on any kind of loop (mainly to loops that fail runtime unrolling).
unsigned PartialOptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size, like OptSizeThreshold,...
bool UnrollAndJam
Allow unroll and jam. Used to enable unroll and jam for the target.
bool UnrollRemainder
Allow unrolling of all the iterations of the runtime loop remainder.
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
unsigned OptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size (set to UINT_MAX to disable).