LLVM 24.0.0git
RISCVTargetTransformInfo.cpp
Go to the documentation of this file.
1//===-- RISCVTargetTransformInfo.cpp - RISC-V specific TTI ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
11#include "RISCVVectorUtils.h"
12#include "llvm/ADT/STLExtras.h"
19#include "llvm/IR/IntrinsicsRISCV.h"
22#include <cmath>
23#include <limits>
24#include <optional>
25using namespace llvm;
26using namespace llvm::PatternMatch;
27
28#define DEBUG_TYPE "riscvtti"
29
31 "riscv-v-register-bit-width-lmul",
33 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
34 "by autovectorized code. Fractional LMULs are not supported."),
36
38 "riscv-v-slp-max-vf",
40 "Overrides result used for getMaximumVF query which is used "
41 "exclusively by SLP vectorizer."),
43
45 RVVMinTripCount("riscv-v-min-trip-count",
46 cl::desc("Set the lower bound of a trip count to decide on "
47 "vectorization while tail-folding."),
49
50static cl::opt<bool> EnableOrLikeSelectOpt("enable-riscv-or-like-select",
51 cl::init(true), cl::Hidden);
52
54RISCVTTIImpl::getRISCVInstructionCost(ArrayRef<unsigned> OpCodes, MVT VT,
56 // Check if the type is valid for all CostKind
57 if (!VT.isVector())
59 size_t NumInstr = OpCodes.size();
61 return NumInstr;
62 InstructionCost LMULCost = TLI->getLMULCost(VT);
64 return LMULCost * NumInstr;
65 InstructionCost Cost = 0;
66 for (auto Op : OpCodes) {
67 switch (Op) {
68 case RISCV::VRGATHER_VI:
69 Cost += TLI->getVRGatherVICost(VT);
70 break;
71 case RISCV::VRGATHER_VV:
72 Cost += TLI->getVRGatherVVCost(VT);
73 break;
74 case RISCV::VSLIDEUP_VI:
75 case RISCV::VSLIDEDOWN_VI:
76 Cost += TLI->getVSlideVICost(VT);
77 break;
78 case RISCV::VSLIDEUP_VX:
79 case RISCV::VSLIDEDOWN_VX:
80 Cost += TLI->getVSlideVXCost(VT);
81 break;
82 case RISCV::VREDMAX_VS:
83 case RISCV::VREDMIN_VS:
84 case RISCV::VREDMAXU_VS:
85 case RISCV::VREDMINU_VS:
86 case RISCV::VREDSUM_VS:
87 case RISCV::VREDAND_VS:
88 case RISCV::VREDOR_VS:
89 case RISCV::VREDXOR_VS:
90 case RISCV::VFREDMAX_VS:
91 case RISCV::VFREDMIN_VS:
92 case RISCV::VFREDUSUM_VS: {
93 unsigned VL = VT.getVectorMinNumElements();
94 if (!VT.isFixedLengthVector())
95 VL *= *getVScaleForTuning();
96 Cost += Log2_32_Ceil(VL);
97 break;
98 }
99 case RISCV::VFREDOSUM_VS: {
100 unsigned VL = VT.getVectorMinNumElements();
101 if (!VT.isFixedLengthVector())
102 VL *= *getVScaleForTuning();
103 Cost += VL;
104 break;
105 }
106 case RISCV::VMV_X_S:
107 case RISCV::VFMV_F_S:
108 // Domain crossings from vector -> scalar are usually more expensive.
109 Cost += 2;
110 break;
111 case RISCV::VMV_S_X:
112 case RISCV::VFMV_S_F:
113 case RISCV::VMOR_MM:
114 case RISCV::VMXOR_MM:
115 case RISCV::VMAND_MM:
116 case RISCV::VMANDN_MM:
117 case RISCV::VMNAND_MM:
118 case RISCV::VCPOP_M:
119 case RISCV::VFIRST_M:
120 Cost += 1;
121 break;
122 case RISCV::VDIV_VV:
123 case RISCV::VREM_VV:
124 Cost += LMULCost * TTI::TCC_Expensive;
125 break;
126 default:
127 Cost += LMULCost;
128 }
129 }
130 return Cost;
131}
132
134 const RISCVSubtarget *ST,
135 const APInt &Imm, Type *Ty,
137 bool FreeZeroes) {
138 assert(Ty->isIntegerTy() &&
139 "getIntImmCost can only estimate cost of materialising integers");
140
141 // We have a Zero register, so 0 is always free.
142 if (Imm == 0)
143 return TTI::TCC_Free;
144
145 // Otherwise, we check how many instructions it will take to materialise.
146 return RISCVMatInt::getIntMatCost(Imm, DL.getTypeSizeInBits(Ty), *ST,
147 /*CompressionCost=*/false, FreeZeroes);
148}
149
153 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind, false);
154}
155
156// Look for patterns of shift followed by AND that can be turned into a pair of
157// shifts. We won't need to materialize an immediate for the AND so these can
158// be considered free.
159static bool canUseShiftPair(Instruction *Inst, const APInt &Imm) {
160 uint64_t Mask = Imm.getZExtValue();
161 auto *BO = dyn_cast<BinaryOperator>(Inst->getOperand(0));
162 if (!BO || !BO->hasOneUse())
163 return false;
164
165 if (BO->getOpcode() != Instruction::Shl)
166 return false;
167
168 if (!isa<ConstantInt>(BO->getOperand(1)))
169 return false;
170
171 unsigned ShAmt = cast<ConstantInt>(BO->getOperand(1))->getZExtValue();
172 // (and (shl x, c2), c1) will be matched to (srli (slli x, c2+c3), c3) if c1
173 // is a mask shifted by c2 bits with c3 leading zeros.
174 if (isShiftedMask_64(Mask)) {
175 unsigned Trailing = llvm::countr_zero(Mask);
176 if (ShAmt == Trailing)
177 return true;
178 }
179
180 return false;
181}
182
183// If this is i64 AND is part of (X & -(1 << C1) & 0xffffffff) == C2 << C1),
184// DAGCombiner can convert this to (sraiw X, C1) == sext(C2) for RV64. On RV32,
185// the type will be split so only the lower 32 bits need to be compared using
186// (srai/srli X, C) == C2.
187static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm) {
188 if (!Inst->hasOneUse())
189 return false;
190
191 // Look for equality comparison.
192 auto *Cmp = dyn_cast<ICmpInst>(*Inst->user_begin());
193 if (!Cmp || !Cmp->isEquality())
194 return false;
195
196 // Right hand side of comparison should be a constant.
197 auto *C = dyn_cast<ConstantInt>(Cmp->getOperand(1));
198 if (!C)
199 return false;
200
201 uint64_t Mask = Imm.getZExtValue();
202
203 // Mask should be of the form -(1 << C) in the lower 32 bits.
204 if (!isUInt<32>(Mask) || !isPowerOf2_32(-uint32_t(Mask)))
205 return false;
206
207 // Comparison constant should be a subset of Mask.
208 uint64_t CmpC = C->getZExtValue();
209 if ((CmpC & Mask) != CmpC)
210 return false;
211
212 // We'll need to sign extend the comparison constant and shift it right. Make
213 // sure the new constant can use addi/xori+seqz/snez.
214 unsigned ShiftBits = llvm::countr_zero(Mask);
215 int64_t NewCmpC = SignExtend64<32>(CmpC) >> ShiftBits;
216 return NewCmpC >= -2048 && NewCmpC <= 2048;
217}
218
220 const APInt &Imm, Type *Ty,
222 Instruction *Inst) const {
223 assert(Ty->isIntegerTy() &&
224 "getIntImmCost can only estimate cost of materialising integers");
225
226 // We have a Zero register, so 0 is always free.
227 if (Imm == 0)
228 return TTI::TCC_Free;
229
230 // Some instructions in RISC-V can take a 12-bit immediate. Some of these are
231 // commutative, in others the immediate comes from a specific argument index.
232 bool Takes12BitImm = false;
233 unsigned ImmArgIdx = ~0U;
234
235 switch (Opcode) {
236 case Instruction::GetElementPtr:
237 // Never hoist any arguments to a GetElementPtr. CodeGenPrepare will
238 // split up large offsets in GEP into better parts than ConstantHoisting
239 // can.
240 return TTI::TCC_Free;
241 case Instruction::Store: {
242 // Use the materialization cost regardless of if it's the address or the
243 // value that is constant, except for if the store is misaligned and
244 // misaligned accesses are not legal (experience shows constant hoisting
245 // can sometimes be harmful in such cases).
246 if (Idx == 1 || !Inst)
247 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
248 /*FreeZeroes=*/true);
249
250 StoreInst *ST = cast<StoreInst>(Inst);
251 if (!getTLI()->allowsMemoryAccessForAlignment(
252 Ty->getContext(), DL, getTLI()->getValueType(DL, Ty),
253 ST->getPointerAddressSpace(), ST->getAlign()))
254 return TTI::TCC_Free;
255
256 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
257 /*FreeZeroes=*/true);
258 }
259 case Instruction::Load:
260 // If the address is a constant, use the materialization cost.
261 return getIntImmCost(Imm, Ty, CostKind);
262 case Instruction::And:
263 // zext.h
264 if (Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
265 return TTI::TCC_Free;
266 // zext.w
267 if (Imm == UINT64_C(0xffffffff) && (!ST->is64Bit() || ST->hasStdExtZba()))
268 return TTI::TCC_Free;
269 // bclri
270 if (ST->hasStdExtZbs() && (~Imm).isPowerOf2())
271 return TTI::TCC_Free;
272 if (Inst && Idx == 1 && Imm.getBitWidth() <= ST->getXLen() &&
273 canUseShiftPair(Inst, Imm))
274 return TTI::TCC_Free;
275 if (Inst && Idx == 1 && Imm.getBitWidth() == 64 &&
276 canUseShiftCmp(Inst, Imm))
277 return TTI::TCC_Free;
278 Takes12BitImm = true;
279 break;
280 case Instruction::Add:
281 Takes12BitImm = true;
282 break;
283 case Instruction::Or:
284 case Instruction::Xor:
285 // bseti/binvi
286 if (ST->hasStdExtZbs() && Imm.isPowerOf2())
287 return TTI::TCC_Free;
288 Takes12BitImm = true;
289 break;
290 case Instruction::Mul:
291 // Power of 2 is a shift. Negated power of 2 is a shift and a negate.
292 if (Imm.isPowerOf2() || Imm.isNegatedPowerOf2())
293 return TTI::TCC_Free;
294 // One more or less than a power of 2 can use SLLI+ADD/SUB.
295 if ((Imm + 1).isPowerOf2() || (Imm - 1).isPowerOf2())
296 return TTI::TCC_Free;
297 // FIXME: There is no MULI instruction.
298 Takes12BitImm = true;
299 break;
300 case Instruction::Sub:
301 case Instruction::Shl:
302 case Instruction::LShr:
303 case Instruction::AShr:
304 Takes12BitImm = true;
305 ImmArgIdx = 1;
306 break;
307 default:
308 break;
309 }
310
311 if (Takes12BitImm) {
312 // Check immediate is the correct argument...
313 if (Instruction::isCommutative(Opcode) || Idx == ImmArgIdx) {
314 // ... and fits into the 12-bit immediate.
315 if (Imm.getSignificantBits() <= 64 &&
316 getTLI()->isLegalAddImmediate(Imm.getSExtValue())) {
317 return TTI::TCC_Free;
318 }
319 }
320
321 // Otherwise, use the full materialisation cost.
322 return getIntImmCost(Imm, Ty, CostKind);
323 }
324
325 // By default, prevent hoisting.
326 return TTI::TCC_Free;
327}
328
331 const APInt &Imm, Type *Ty,
333 // Prevent hoisting in unknown cases.
334 return TTI::TCC_Free;
335}
336
338 return ST->hasVInstructions();
339}
340
342RISCVTTIImpl::getPopcntSupport(unsigned TyWidth) const {
343 assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2");
344 return ST->hasCPOPLike() ? TTI::PSK_FastHardware : TTI::PSK_Software;
345}
346
348 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
350 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
351 TTI::TargetCostKind CostKind, std::optional<FastMathFlags> FMF) const {
352 if (Opcode == Instruction::FAdd)
354
355 // zve32x is broken for partial_reduce_umla, but let's make sure we
356 // don't generate them.
357 // vdot4a* reduces four i8 products into an i32 result; an i64 accumulator is
358 // additionally supported by widening the i32 partial sums to i64 (see
359 // lowerPARTIAL_REDUCE_MLA). VF is the number of i8 input elements, so the
360 // reduction factor is AccumBits / 8 (4 for i32, 8 for i64).
361 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
362 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
363 InputTypeA != InputTypeB || !InputTypeA->isIntegerTy(8) ||
364 (!AccumType->isIntegerTy(32) && !AccumType->isIntegerTy(64)))
366
367 unsigned Ratio = AccumType->getScalarSizeInBits() / 8;
368 if (!VF.isKnownMultipleOf(Ratio))
370
371 // Cost of the vdot4a* itself, which operates on the i32 intermediate type
372 // holding VF/4 elements.
373 Type *DotTp = VectorType::get(Type::getInt32Ty(AccumType->getContext()),
374 VF.divideCoefficientBy(4));
375 std::pair<InstructionCost, MVT> DotLT = getTypeLegalizationCost(DotTp);
376 // Note: Asuming all vdot4a* variants are equal cost
378 DotLT.first *
379 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second, CostKind);
380
381 // Account for reducing the i32 partial sums down to the i64 accumulator's
382 // element count and accumulating into it (see lowerPARTIAL_REDUCE_MLA), which
383 // has two shapes depending on the accumulator's LMUL.
384 if (AccumType->isIntegerTy(64)) {
385 LLVMContext &Ctx = AccumType->getContext();
386 Type *I32Ty = Type::getInt32Ty(Ctx);
387 ElementCount AccVF = VF.divideCoefficientBy(Ratio);
388 std::pair<InstructionCost, MVT> AccLT =
389 getTypeLegalizationCost(VectorType::get(AccumType, AccVF));
390
391 // When the i32 subvectors of a single-vector scalable accumulator are a
392 // fractional LMUL, extracting the high subvector would need a vslidedown,
393 // so instead the i32 dot result is widened to i64 first (vsext.vf2 /
394 // vzext.vf2) and then reduced and accumulated with register-aligned i64
395 // vadd.vv.
396 bool WidenFirst = false;
397 if (VF.isScalable() && AccLT.second.isScalableVector()) {
398 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
399 WidenFirst =
401 .second;
402 }
403
404 if (WidenFirst) {
405 // The widened i64 dot result has VF/4 elements, i.e. twice the
406 // accumulator's element count, so the reduction plus the accumulate are
407 // two i64 vadd.vv.
408 std::pair<InstructionCost, MVT> WideLT = getTypeLegalizationCost(
409 VectorType::get(AccumType, VF.divideCoefficientBy(4)));
410 Cost +=
411 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
412 WideLT.second, CostKind) +
413 2 * AccLT.first *
414 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second, CostKind);
415 } else {
416 // Otherwise the scale-4 i32 sums are halved with a single i32 vadd.vv,
417 // then widened and added into the i64 result with a vwadd.wv.
418 std::pair<InstructionCost, MVT> RedLT =
420 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
421 RedLT.second, CostKind) +
422 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
423 AccLT.second, CostKind);
424 // Fixed-length vectors extract the high i32 subvector with a vslidedown.
425 if (VF.isFixed())
426 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
427 DotLT.second, CostKind);
428 }
429 }
430
431 return Cost;
432}
433
435 // Currently, the ExpandReductions pass can't expand scalable-vector
436 // reductions, but we still request expansion as RVV doesn't support certain
437 // reductions and the SelectionDAG can't legalize them either.
438 switch (II->getIntrinsicID()) {
439 default:
440 return false;
441 // These reductions have no equivalent in RVV
442 case Intrinsic::vector_reduce_mul:
443 case Intrinsic::vector_reduce_fmul:
444 return true;
445 }
446}
447
448std::optional<unsigned> RISCVTTIImpl::getVScaleForTuning() const {
449 if (ST->hasVInstructions())
450 if (unsigned MinVLen = ST->getRealMinVLen();
451 MinVLen >= RISCV::RVVBitsPerBlock)
452 return MinVLen / RISCV::RVVBitsPerBlock;
454}
455
458 unsigned LMUL =
459 llvm::bit_floor(std::clamp<unsigned>(RVVRegisterWidthLMUL, 1, 8));
460 switch (K) {
462 return TypeSize::getFixed(ST->getXLen());
464 return TypeSize::getFixed(
465 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
468 (ST->hasVInstructions() &&
469 ST->getRealMinVLen() >= RISCV::RVVBitsPerBlock)
471 : 0);
472 }
473
474 llvm_unreachable("Unsupported register kind");
475}
476
477InstructionCost RISCVTTIImpl::getStaticDataAddrGenerationCost(
478 const TTI::TargetCostKind CostKind) const {
479 switch (CostKind) {
482 // Always 2 instructions
483 return 2;
484 case TTI::TCK_Latency:
486 // Depending on the memory model the address generation will
487 // require AUIPC + ADDI (medany) or LUI + ADDI (medlow). Don't
488 // have a way of getting this information here, so conservatively
489 // require both.
490 // In practice, these are generally implemented together.
491 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
492 }
493 llvm_unreachable("Unsupported cost kind");
494}
495
497RISCVTTIImpl::getConstantPoolLoadCost(Type *Ty,
499 // Add a cost of address generation + the cost of the load. The address
500 // is expected to be a PC relative offset to a constant pool entry
501 // using auipc/addi.
502 return getStaticDataAddrGenerationCost(CostKind) +
503 getMemoryOpCost(Instruction::Load, Ty, DL.getABITypeAlign(Ty),
504 /*AddressSpace=*/0, CostKind);
505}
506
507static bool isRepeatedConcatMask(ArrayRef<int> Mask, int &SubVectorSize) {
508 unsigned Size = Mask.size();
509 if (!isPowerOf2_32(Size))
510 return false;
511 for (unsigned I = 0; I != Size; ++I) {
512 if (static_cast<unsigned>(Mask[I]) == I)
513 continue;
514 if (Mask[I] != 0)
515 return false;
516 if (Size % I != 0)
517 return false;
518 for (unsigned J = I + 1; J != Size; ++J)
519 // Check the pattern is repeated.
520 if (static_cast<unsigned>(Mask[J]) != J % I)
521 return false;
522 SubVectorSize = I;
523 return true;
524 }
525 // That means Mask is <0, 1, 2, 3>. This is not a concatenation.
526 return false;
527}
528
530 LLVMContext &C) {
531 assert((DataVT.getScalarSizeInBits() != 8 ||
532 DataVT.getVectorNumElements() <= 256) && "unhandled case in lowering");
533 MVT IndexVT = DataVT.changeTypeToInteger();
534 if (IndexVT.getScalarType().bitsGT(ST.getXLenVT()))
535 IndexVT = IndexVT.changeVectorElementType(MVT::i16);
536 return cast<VectorType>(EVT(IndexVT).getTypeForEVT(C));
537}
538
539/// Attempt to approximate the cost of a shuffle which will require splitting
540/// during legalization. Note that processShuffleMasks is not an exact proxy
541/// for the algorithm used in LegalizeVectorTypes, but hopefully it's a
542/// reasonably close upperbound.
544 MVT LegalVT, VectorType *Tp,
545 ArrayRef<int> Mask,
547 assert(LegalVT.isFixedLengthVector() && !Mask.empty() &&
548 "Expected fixed vector type and non-empty mask");
549 unsigned LegalNumElts = LegalVT.getVectorNumElements();
550 // Number of destination vectors after legalization:
551 unsigned NumOfDests = divideCeil(Mask.size(), LegalNumElts);
552 // We are going to permute multiple sources and the result will be in
553 // multiple destinations. Providing an accurate cost only for splits where
554 // the element type remains the same.
555 if (NumOfDests <= 1 ||
557 Tp->getElementType()->getPrimitiveSizeInBits() ||
558 LegalNumElts >= Tp->getElementCount().getFixedValue())
560
561 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
562 unsigned LegalVTSize = LegalVT.getStoreSize();
563 // Number of source vectors after legalization:
564 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
565
566 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(), LegalNumElts);
567
568 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
569 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
570 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
571 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
572 assert(NormalizedVF >= Mask.size() &&
573 "Normalized mask expected to be not shorter than original mask.");
574 copy(Mask, NormalizedMask.begin());
575 InstructionCost Cost = 0;
576 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
578 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
579 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
580 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
581 return;
582 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
583 .second)
584 return;
585 Cost += TTI.getShuffleCost(
587 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
588 SingleOpTy, CostKind, RegMask, 0, nullptr);
589 },
590 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
591 Cost += TTI.getShuffleCost(
593 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
594 SingleOpTy, CostKind, RegMask, 0, nullptr);
595 });
596 return Cost;
597}
598
599/// Try to perform better estimation of the permutation.
600/// 1. Split the source/destination vectors into real registers.
601/// 2. Do the mask analysis to identify which real registers are
602/// permuted. If more than 1 source registers are used for the
603/// destination register building, the cost for this destination register
604/// is (Number_of_source_register - 1) * Cost_PermuteTwoSrc. If only one
605/// source register is used, build mask and calculate the cost as a cost
606/// of PermuteSingleSrc.
607/// Also, for the single register permute we try to identify if the
608/// destination register is just a copy of the source register or the
609/// copy of the previous destination register (the cost is
610/// TTI::TCC_Basic). If the source register is just reused, the cost for
611/// this operation is 0.
612static InstructionCost
614 std::optional<unsigned> VLen, VectorType *Tp,
616 assert(LegalVT.isFixedLengthVector());
617 if (!VLen || Mask.empty())
619 MVT ElemVT = LegalVT.getVectorElementType();
620 unsigned ElemsPerVReg = *VLen / ElemVT.getFixedSizeInBits();
621 LegalVT = TTI.getTypeLegalizationCost(
622 FixedVectorType::get(Tp->getElementType(), ElemsPerVReg))
623 .second;
624 // Number of destination vectors after legalization:
625 InstructionCost NumOfDests =
626 divideCeil(Mask.size(), LegalVT.getVectorNumElements());
627 if (NumOfDests <= 1 ||
629 Tp->getElementType()->getPrimitiveSizeInBits() ||
630 LegalVT.getVectorNumElements() >= Tp->getElementCount().getFixedValue())
632
633 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
634 unsigned LegalVTSize = LegalVT.getStoreSize();
635 // Number of source vectors after legalization:
636 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
637
638 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(),
639 LegalVT.getVectorNumElements());
640
641 unsigned E = NumOfDests.getValue();
642 unsigned NormalizedVF =
643 LegalVT.getVectorNumElements() * std::max(NumOfSrcs, E);
644 unsigned NumOfSrcRegs = NormalizedVF / LegalVT.getVectorNumElements();
645 unsigned NumOfDestRegs = NormalizedVF / LegalVT.getVectorNumElements();
646 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
647 assert(NormalizedVF >= Mask.size() &&
648 "Normalized mask expected to be not shorter than original mask.");
649 copy(Mask, NormalizedMask.begin());
650 InstructionCost Cost = 0;
651 int NumShuffles = 0;
652 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
654 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
655 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
656 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
657 return;
658 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
659 .second)
660 return;
661 ++NumShuffles;
662 Cost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, SingleOpTy,
663 SingleOpTy, CostKind, RegMask, 0, nullptr);
664 },
665 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
666 Cost += TTI.getShuffleCost(TTI::SK_PermuteTwoSrc, SingleOpTy,
667 SingleOpTy, CostKind, RegMask, 0, nullptr);
668 NumShuffles += 2;
669 });
670 // Note: check that we do not emit too many shuffles here to prevent code
671 // size explosion.
672 // TODO: investigate, if it can be improved by extra analysis of the masks
673 // to check if the code is more profitable.
674 if ((NumOfDestRegs > 2 && NumShuffles <= static_cast<int>(NumOfDestRegs)) ||
675 (NumOfDestRegs <= 2 && NumShuffles < 4))
676 return Cost;
678}
679
680InstructionCost RISCVTTIImpl::getSlideCost(FixedVectorType *Tp,
681 ArrayRef<int> Mask,
683 // Avoid missing masks and length changing shuffles
684 if (Mask.size() <= 2 || Mask.size() != Tp->getNumElements())
686
687 int NumElts = Tp->getNumElements();
688 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
689 // Avoid scalarization cases
690 if (!LT.second.isFixedLengthVector())
692
693 // Requires moving elements between parts, which requires additional
694 // unmodeled instructions.
695 if (LT.first != 1)
697
698 auto GetSlideOpcode = [&](int SlideAmt) {
699 assert(SlideAmt != 0);
700 bool IsVI = isUInt<5>(std::abs(SlideAmt));
701 if (SlideAmt < 0)
702 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
703 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
704 };
705
706 std::array<std::pair<int, int>, 2> SrcInfo;
707 if (!isMaskedSlidePair(Mask, NumElts, SrcInfo))
709
710 if (SrcInfo[1].second == 0)
711 std::swap(SrcInfo[0], SrcInfo[1]);
712
713 if (ST->hasStdExtZvzip() && LT.second.getScalarSizeInBits() != 1) {
714 unsigned Factor;
715 if (isPairEven(SrcInfo, Mask, Factor) && Factor == 1)
716 return getRISCVInstructionCost(RISCV::VPAIRE_VV, LT.second, CostKind);
717 if (isPairOdd(SrcInfo, Mask, Factor) && Factor == 1)
718 return getRISCVInstructionCost(RISCV::VPAIRO_VV, LT.second, CostKind);
719 }
720
721 InstructionCost FirstSlideCost = 0;
722 if (SrcInfo[0].second != 0) {
723 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
724 FirstSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
725 }
726
727 if (SrcInfo[1].first == -1)
728 return FirstSlideCost;
729
730 InstructionCost SecondSlideCost = 0;
731 if (SrcInfo[1].second != 0) {
732 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
733 SecondSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
734 } else {
735 SecondSlideCost =
736 getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second, CostKind);
737 }
738
739 auto EC = Tp->getElementCount();
740 VectorType *MaskTy =
742 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
743 return FirstSlideCost + SecondSlideCost + MaskCost;
744}
745
746std::optional<MVT> RISCVTTIImpl::getZvzipVZIPCostVT(MVT InterleavedVT) const {
747 assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
748 if (!InterleavedVT.getVectorElementCount().isKnownEven())
749 return std::nullopt;
750
751 unsigned EltBits = InterleavedVT.getScalarSizeInBits();
752 unsigned MinSize = InterleavedVT.getSizeInBits().getKnownMinValue();
753 unsigned LMULOctuple = MinSize / (RISCV::RVVBitsPerBlock / 8);
754 // Perform the 2 * SEW <= LMUL * min(ELEN, VLEN) check.
755 if (EltBits * 16 >
756 LMULOctuple * std::min(ST->getELen(), ST->getRealMinVLen()))
757 return std::nullopt;
758 return InterleavedVT;
759}
760
761std::optional<MVT> RISCVTTIImpl::getZvzipVUNZIPCostVT(MVT InterleavedVT) const {
762 assert(InterleavedVT.isScalableVector() && "Expected a scalable vector type");
763 if (!InterleavedVT.getVectorElementCount().isKnownEven())
764 return std::nullopt;
765
766 MVT DeinterleavedVT = InterleavedVT.getHalfNumVectorElementsVT();
767 if (RISCVTargetLowering::getLMUL(DeinterleavedVT) == RISCVVType::LMUL_8)
768 return std::nullopt;
769 return InterleavedVT;
770}
771
773 TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy,
775 VectorType *SubTp, ArrayRef<const Value *> Args, const Instruction *CtxI,
776 TTI::VectorInstrContext VIC) const {
777 assert((Mask.empty() || DstTy->isScalableTy() ||
778 Mask.size() == DstTy->getElementCount().getKnownMinValue()) &&
779 "Expected the Mask to match the return size if given");
780 assert(SrcTy->getScalarType() == DstTy->getScalarType() &&
781 "Expected the same scalar types");
782
783 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
784 if (VIC == TTI::VectorInstrContext::SplatOpFolded &&
785 ST->sinkSplatOperands() && Kind == TTI::SK_Broadcast)
786 return TTI::TCC_Free;
787
788 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
789 // For now, skip all fixed vector cost analysis when P extension is available
790 // to avoid crashes in getMinRVVVectorSizeInBits()
791 if (ST->hasStdExtP() && isa<FixedVectorType>(SrcTy))
792 return 1;
793
794 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
795
796 // First, handle cases where having a fixed length vector enables us to
797 // give a more accurate cost than falling back to generic scalable codegen.
798 // TODO: Each of these cases hints at a modeling gap around scalable vectors.
799 if (auto *FVTp = dyn_cast<FixedVectorType>(SrcTy);
800 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
802 *this, LT.second, ST->getRealVLen(),
803 Kind == TTI::SK_InsertSubvector ? DstTy : SrcTy, Mask, CostKind);
804 if (VRegSplittingCost.isValid())
805 return VRegSplittingCost;
806 switch (Kind) {
807 default:
808 break;
810 if (Mask.size() >= 2) {
811 MVT EltTp = LT.second.getVectorElementType();
812 // If the size of the element is < ELEN then shuffles of interleaves and
813 // deinterleaves of 2 vectors can be lowered into the following
814 // sequences
815 if (EltTp.getScalarSizeInBits() < ST->getELen()) {
816 // Example sequence:
817 // vsetivli zero, 4, e8, mf4, ta, ma (ignored)
818 // vwaddu.vv v10, v8, v9
819 // li a0, -1 (ignored)
820 // vwmaccu.vx v10, a0, v9
821 if (ShuffleVectorInst::isInterleaveMask(Mask, 2, Mask.size()))
822 return 2 * LT.first * TLI->getLMULCost(LT.second);
823
824 if (Mask[0] == 0 || Mask[0] == 1) {
825 auto DeinterleaveMask = createStrideMask(Mask[0], 2, Mask.size());
826 // Example sequence:
827 // vnsrl.wi v10, v8, 0
828 if (equal(DeinterleaveMask, Mask))
829 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
830 LT.second, CostKind);
831 }
832 }
833 int SubVectorSize;
834 if (LT.second.getScalarSizeInBits() != 1 &&
835 isRepeatedConcatMask(Mask, SubVectorSize)) {
837 unsigned NumSlides = Log2_32(Mask.size() / SubVectorSize);
838 // The cost of extraction from a subvector is 0 if the index is 0.
839 for (unsigned I = 0; I != NumSlides; ++I) {
840 unsigned InsertIndex = SubVectorSize * (1 << I);
841 FixedVectorType *SubTp =
842 FixedVectorType::get(SrcTy->getElementType(), InsertIndex);
843 FixedVectorType *DestTp =
845 std::pair<InstructionCost, MVT> DestLT =
847 // Add the cost of whole vector register move because the
848 // destination vector register group for vslideup cannot overlap the
849 // source.
850 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
852 CostKind, {}, InsertIndex, SubTp);
853 }
854 return Cost;
855 }
856 }
857
858 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
859 SlideCost.isValid())
860 return SlideCost;
861
862 // vrgather + cost of generating the mask constant.
863 // We model this for an unknown mask with a single vrgather.
864 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
865 LT.second.getVectorNumElements() <= 256)) {
866 VectorType *IdxTy =
867 getVRGatherIndexType(LT.second, *ST, SrcTy->getContext());
868 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
869 return IndexCost +
870 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second, CostKind);
871 }
872 break;
873 }
876
877 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
878 SlideCost.isValid())
879 return SlideCost;
880
881 // 2 x (vrgather + cost of generating the mask constant) + cost of mask
882 // register for the second vrgather. We model this for an unknown
883 // (shuffle) mask.
884 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
885 LT.second.getVectorNumElements() <= 256)) {
886 auto &C = SrcTy->getContext();
887 auto EC = SrcTy->getElementCount();
888 VectorType *IdxTy = getVRGatherIndexType(LT.second, *ST, C);
890 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
891 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
892 return 2 * IndexCost +
893 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
894 LT.second, CostKind) +
895 MaskCost;
896 }
897 break;
898 }
899 }
900
901 auto shouldSplit = [](TTI::ShuffleKind Kind) {
902 switch (Kind) {
903 default:
904 return false;
908 return true;
909 }
910 };
911
912 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
913 shouldSplit(Kind)) {
914 InstructionCost SplitCost =
915 costShuffleViaSplitting(*this, LT.second, FVTp, Mask, CostKind);
916 if (SplitCost.isValid())
917 return SplitCost;
918 }
919 }
920
921 // Handle scalable vectors (and fixed vectors legalized to scalable vectors).
922 switch (Kind) {
923 default:
924 // Fallthrough to generic handling.
925 // TODO: Most of these cases will return getInvalid in generic code, and
926 // must be implemented here.
927 break;
929 // Extract at zero is always a subregister extract
930 if (Index == 0)
931 return TTI::TCC_Free;
932
933 // If we're extracting a subvector of at most m1 size at a sub-register
934 // boundary - which unfortunately we need exact vlen to identify - this is
935 // a subregister extract at worst and thus won't require a vslidedown.
936 // TODO: Extend for aligned m2, m4 subvector extracts
937 // TODO: Extend for misalgined (but contained) extracts
938 // TODO: Extend for scalable subvector types
939 if (std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(SubTp);
940 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
941 if (std::optional<unsigned> VLen = ST->getRealVLen();
942 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
943 SubLT.second.getSizeInBits() <= *VLen)
944 return TTI::TCC_Free;
945 }
946
947 // Example sequence:
948 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
949 // vslidedown.vi v8, v9, 2
950 return LT.first *
951 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second, CostKind);
953 // Example sequence:
954 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
955 // vslideup.vi v8, v9, 2
956 LT = getTypeLegalizationCost(DstTy);
957 return LT.first *
958 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second, CostKind);
959 case TTI::SK_Select: {
960 // Example sequence:
961 // li a0, 90
962 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
963 // vmv.s.x v0, a0
964 // vmerge.vvm v8, v9, v8, v0
965 // We use 2 for the cost of the mask materialization as this is the true
966 // cost for small masks and most shuffles are small. At worst, this cost
967 // should be a very small constant for the constant pool load. As such,
968 // we may bias towards large selects slightly more than truly warranted.
969 return LT.first *
970 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
971 LT.second, CostKind));
972 }
973 case TTI::SK_Broadcast: {
974 // Check for broadcast loads, which are synthesized by optimized zero-stride
975 // loads (this is checked in RISCVTTIImpl::isLegalBroadcastLoad).
976 bool IsLoad = !Args.empty() && isa<LoadInst>(Args[0]);
977 if (IsLoad && LT.second.isVector() &&
978 isLegalBroadcastLoad(SrcTy->getElementType(),
979 LT.second.getVectorElementCount()))
980 return 0;
981
982 bool HasScalar = (Args.size() > 0) && (Operator::getOpcode(Args[0]) ==
983 Instruction::InsertElement);
984 if (LT.second.getScalarSizeInBits() == 1) {
985 if (HasScalar) {
986 // Example sequence:
987 // andi a0, a0, 1
988 // vsetivli zero, 2, e8, mf8, ta, ma (ignored)
989 // vmv.v.x v8, a0
990 // vmsne.vi v0, v8, 0
991 return LT.first *
992 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
993 LT.second, CostKind));
994 }
995 // Example sequence:
996 // vsetivli zero, 2, e8, mf8, ta, mu (ignored)
997 // vmv.v.i v8, 0
998 // vmerge.vim v8, v8, 1, v0
999 // vmv.x.s a0, v8
1000 // andi a0, a0, 1
1001 // vmv.v.x v8, a0
1002 // vmsne.vi v0, v8, 0
1003
1004 return LT.first *
1005 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
1006 RISCV::VMV_X_S, RISCV::VMV_V_X,
1007 RISCV::VMSNE_VI},
1008 LT.second, CostKind));
1009 }
1010
1011 if (HasScalar) {
1012 // Example sequence:
1013 // vmv.v.x v8, a0
1014 return LT.first *
1015 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second, CostKind);
1016 }
1017
1018 // Example sequence:
1019 // vrgather.vi v9, v8, 0
1020 return LT.first *
1021 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second, CostKind);
1022 }
1023 case TTI::SK_Splice: {
1024 // vslidedown+vslideup.
1025 // TODO: Multiplying by LT.first implies this legalizes into multiple copies
1026 // of similar code, but I think we expand through memory.
1027 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
1028 if (Index >= 0 && Index < 32)
1029 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
1030 else if (Index < 0 && Index > -32)
1031 Opcodes[1] = RISCV::VSLIDEUP_VI;
1032 return LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1033 }
1034 case TTI::SK_Reverse: {
1035
1036 if (!LT.second.isVector())
1038
1039 // TODO: Cases to improve here:
1040 // * Illegal vector types
1041 // * i64 on RV32
1042 if (SrcTy->getElementType()->isIntegerTy(1)) {
1043 VectorType *WideTy =
1044 VectorType::get(IntegerType::get(SrcTy->getContext(), 8),
1045 cast<VectorType>(SrcTy)->getElementCount());
1046 return getCastInstrCost(Instruction::ZExt, WideTy, SrcTy,
1048 getShuffleCost(TTI::SK_Reverse, WideTy, WideTy, CostKind, {}, 0,
1049 nullptr) +
1050 getCastInstrCost(Instruction::Trunc, SrcTy, WideTy,
1052 }
1053
1054 MVT ContainerVT = LT.second;
1055 if (LT.second.isFixedLengthVector())
1056 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1057 MVT M1VT = RISCVTargetLowering::getM1VT(ContainerVT);
1058 if (ContainerVT.bitsLE(M1VT)) {
1059 // Example sequence:
1060 // csrr a0, vlenb
1061 // srli a0, a0, 3
1062 // addi a0, a0, -1
1063 // vsetvli a1, zero, e8, mf8, ta, mu (ignored)
1064 // vid.v v9
1065 // vrsub.vx v10, v9, a0
1066 // vrgather.vv v9, v8, v10
1067 InstructionCost LenCost = 3;
1068 if (LT.second.isFixedLengthVector())
1069 // vrsub.vi has a 5 bit immediate field, otherwise an li suffices
1070 LenCost = isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1071 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1072 if (LT.second.isFixedLengthVector() &&
1073 isInt<5>(LT.second.getVectorNumElements() - 1))
1074 Opcodes[1] = RISCV::VRSUB_VI;
1075 InstructionCost GatherCost =
1076 getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1077 return LT.first * (LenCost + GatherCost);
1078 }
1079
1080 // At high LMUL, we split into a series of M1 reverses (see
1081 // lowerVECTOR_REVERSE) and then do a single slide at the end to eliminate
1082 // the resulting gap at the bottom (for fixed vectors only). The important
1083 // bit is that the cost scales linearly, not quadratically with LMUL.
1084 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1085 InstructionCost FixedCost =
1086 getRISCVInstructionCost(M1Opcodes, M1VT, CostKind) + 3;
1087 unsigned Ratio =
1088 ContainerVT.getVectorMinNumElements() / M1VT.getVectorMinNumElements();
1089 InstructionCost GatherCost =
1090 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT, CostKind) * Ratio;
1091 InstructionCost SlideCost = !LT.second.isFixedLengthVector() ? 0 :
1092 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second, CostKind);
1093 return FixedCost + LT.first * (GatherCost + SlideCost);
1094 }
1095 }
1096 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
1097 SubTp);
1098}
1099
1100static unsigned isM1OrSmaller(MVT VT) {
1102 return (LMUL == RISCVVType::VLMUL::LMUL_F8 ||
1106}
1107
1109 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
1110 TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef<Value *> VL,
1111 TTI::VectorInstrContext VIC) const {
1114
1115 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
1116 // For now, skip all fixed vector cost analysis when P extension is available
1117 // to avoid crashes in getMinRVVVectorSizeInBits()
1118 if (ST->hasStdExtP() && isa<FixedVectorType>(Ty)) {
1119 return 1; // Treat as single instruction cost for now
1120 }
1121
1122 // A build_vector (which is m1 sized or smaller) can be done in no
1123 // worse than one vslide1down.vx per element in the type. We could
1124 // in theory do an explode_vector in the inverse manner, but our
1125 // lowering today does not have a first class node for this pattern.
1127 Ty, DemandedElts, Insert, Extract, CostKind);
1128 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1129 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1130 if (Ty->getScalarSizeInBits() == 1) {
1131 auto *WideVecTy = cast<VectorType>(Ty->getWithNewBitWidth(8));
1132 // Note: Implicit scalar anyextend is assumed to be free since the i1
1133 // must be stored in a GPR.
1134 return getScalarizationOverhead(WideVecTy, DemandedElts, Insert, Extract,
1135 CostKind) +
1136 getCastInstrCost(Instruction::Trunc, Ty, WideVecTy,
1138 }
1139
1140 assert(LT.second.isFixedLengthVector());
1141 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1142 if (isM1OrSmaller(ContainerVT)) {
1143 InstructionCost BV =
1144 cast<FixedVectorType>(Ty)->getNumElements() *
1145 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second, CostKind);
1146 if (BV < Cost)
1147 Cost = BV;
1148 }
1149 }
1150 return Cost;
1151}
1152
1156 Type *DataTy = MICA.getDataType();
1157 Align Alignment = MICA.getAlignment();
1158 switch (MICA.getID()) {
1159 case Intrinsic::vp_load_ff: {
1160 EVT DataTypeVT = TLI->getValueType(DL, DataTy);
1161 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1163
1164 unsigned AS = MICA.getAddressSpace();
1165 return getMemoryOpCost(Instruction::Load, DataTy, Alignment, AS, CostKind,
1166 {TTI::OK_AnyValue, TTI::OP_None}, nullptr);
1167 }
1168 case Intrinsic::experimental_vp_strided_load:
1169 case Intrinsic::experimental_vp_strided_store:
1170 return getStridedMemoryOpCost(MICA, CostKind);
1171 case Intrinsic::masked_compressstore:
1172 case Intrinsic::masked_expandload:
1174 case Intrinsic::vp_scatter:
1175 case Intrinsic::vp_gather:
1176 case Intrinsic::masked_scatter:
1177 case Intrinsic::masked_gather:
1178 return getGatherScatterOpCost(MICA, CostKind);
1179 case Intrinsic::vp_load:
1180 case Intrinsic::vp_store:
1181 case Intrinsic::masked_load:
1182 case Intrinsic::masked_store:
1183 return getMaskedMemoryOpCost(MICA, CostKind);
1184 }
1186}
1187
1191 unsigned Opcode = MICA.getID() == Intrinsic::masked_load ? Instruction::Load
1192 : Instruction::Store;
1193 Type *Src = MICA.getDataType();
1194 Align Alignment = MICA.getAlignment();
1195 unsigned AddressSpace = MICA.getAddressSpace();
1196
1197 if (!isLegalMaskedLoadStore(Src, Alignment) ||
1200
1201 // Splitting involves additional evl arithmetic and vl toggles.
1202 InstructionCost SplitCost = 0;
1203 if (MICA.getID() == Intrinsic::vp_load ||
1204 MICA.getID() == Intrinsic::vp_store) {
1205 auto LT = getTypeLegalizationCost(Src);
1206 if (LT.first > 1)
1207 SplitCost += LT.first * TTI::TCC_Expensive;
1208 }
1209
1210 return getMemoryOpCost(Opcode, Src, Alignment, AddressSpace, CostKind);
1211}
1212
1214 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
1215 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
1216 bool UseMaskForCond, bool UseMaskForGaps) const {
1217
1218 // The interleaved memory access pass will lower (de)interleave ops combined
1219 // with an adjacent appropriate memory to vlseg/vsseg intrinsics. vlseg/vsseg
1220 // only support masking per-iteration (i.e. condition), not per-segment (i.e.
1221 // gap).
1222 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1223 auto *VTy = cast<VectorType>(VecTy);
1224 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
1225 // Need to make sure type has't been scalarized
1226 if (LT.second.isVector()) {
1228 return LT.first * TTI::TCC_Basic;
1229
1230 auto *SubVecTy =
1231 VectorType::get(VTy->getElementType(),
1232 VTy->getElementCount().divideCoefficientBy(Factor));
1233 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1234 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1235 AddressSpace, DL)) {
1236
1237 // Some processors optimize segment loads/stores as N * DLEN sized
1238 // load ops + Factor * LMUL shuffle ops.
1239 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1240 unsigned VecSizeInBits =
1241 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1242 unsigned VLENForTuning =
1244 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1245 InstructionCost Cost = divideCeil(VecSizeInBits, DLENForTuning);
1246 MVT SubVecVT = getTLI()->getValueType(DL, SubVecTy).getSimpleVT();
1247 Cost += Factor * TLI->getLMULCost(SubVecVT);
1248 return Cost;
1249 }
1250
1251 // Otherwise, the cost is proportional to the number of elements (VL *
1252 // Factor ops).
1253 unsigned NumLoads = getEstimatedVLFor(VTy);
1254 return NumLoads * TTI::TCC_Basic;
1255 }
1256 }
1257 }
1258
1259 // TODO: Return the cost of interleaved accesses for scalable vector when
1260 // unable to convert to segment accesses instructions.
1261 if (isa<ScalableVectorType>(VecTy))
1263
1264 auto *FVTy = cast<FixedVectorType>(VecTy);
1265 // When gaps are only at the tail, for interleaved load, we can emit a wide
1266 // masked load and shufflevectors. For interleaved store, we can emit
1267 // shufflevectors and a wide masked store. The interleaved memory access pass
1268 // will lower them into vlsseg/vssseg intrinsics.
1269 if (UseMaskForGaps) {
1270 assert(llvm::is_sorted(Indices) && "Indices must be sorted");
1271 assert(llvm::adjacent_find(Indices) == Indices.end() &&
1272 "Indices should not contain duplicate elements");
1273 unsigned NumOfFields = Indices.size();
1274 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.back() + 1);
1275 if (IsTailGapOnly &&
1276 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1277 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(FVTy);
1278 if (LT.second.isVector() &&
1279 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1280 auto *SubVecTy = VectorType::get(
1281 FVTy->getElementType(),
1282 FVTy->getElementCount().divideCoefficientBy(Factor));
1283 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1284 AddressSpace, DL)) {
1285 // The cost is proportional to the total number of element accesses.
1286 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1287 return NumAccesses * TTI::TCC_Basic;
1288 }
1289 }
1290 }
1291 }
1292
1293 InstructionCost MemCost =
1294 getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
1295 unsigned VF = FVTy->getNumElements() / Factor;
1296
1297 // An interleaved load will look like this for Factor=3:
1298 // %wide.vec = load <12 x i32>, ptr %3, align 4
1299 // %strided.vec = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1300 // %strided.vec1 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1301 // %strided.vec2 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1302 if (Opcode == Instruction::Load) {
1303 InstructionCost Cost = MemCost;
1304 for (unsigned Index : Indices) {
1305 FixedVectorType *VecTy =
1306 FixedVectorType::get(FVTy->getElementType(), VF * Factor);
1307 auto Mask = createStrideMask(Index, Factor, VF);
1308 Mask.resize(VF * Factor, -1);
1309 InstructionCost ShuffleCost =
1311 CostKind, Mask, 0, nullptr, {});
1312 Cost += ShuffleCost;
1313 }
1314 return Cost;
1315 }
1316
1317 // TODO: Model for NF > 2
1318 // We'll need to enhance getShuffleCost to model shuffles that are just
1319 // inserts and extracts into subvectors, since they won't have the full cost
1320 // of a vrgather.
1321 // An interleaved store for 3 vectors of 4 lanes will look like
1322 // %11 = shufflevector <4 x i32> %4, <4 x i32> %6, <8 x i32> <0...7>
1323 // %12 = shufflevector <4 x i32> %9, <4 x i32> poison, <8 x i32> <0...3>
1324 // %13 = shufflevector <8 x i32> %11, <8 x i32> %12, <12 x i32> <0...11>
1325 // %interleaved.vec = shufflevector %13, poison, <12 x i32> <interleave mask>
1326 // store <12 x i32> %interleaved.vec, ptr %10, align 4
1327 if (Factor != 2)
1328 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
1329 Alignment, AddressSpace, CostKind,
1330 UseMaskForCond, UseMaskForGaps);
1331
1332 assert(Opcode == Instruction::Store && "Opcode must be a store");
1333 // For an interleaving store of 2 vectors, we perform one large interleaving
1334 // shuffle that goes into the wide store
1335 auto Mask = createInterleaveMask(VF, Factor);
1336 InstructionCost ShuffleCost =
1338 CostKind, Mask, 0, nullptr, {});
1339 return MemCost + ShuffleCost;
1340}
1341
1345
1346 bool IsLoad = MICA.getID() == Intrinsic::masked_gather ||
1347 MICA.getID() == Intrinsic::vp_gather;
1348 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1349 Type *DataTy = MICA.getDataType();
1350 Type *PtrTy = DataTy->getWithNewType(
1351 DL.getAddressType(DataTy->getContext(), MICA.getAddressSpace()));
1352 Align Alignment = MICA.getAlignment();
1355
1356 if ((Opcode == Instruction::Load &&
1357 !isLegalMaskedGather(DataTy, Align(Alignment))) ||
1358 (Opcode == Instruction::Store &&
1359 !isLegalMaskedScatter(DataTy, Align(Alignment))))
1361
1362 // Splitting vp intrinsics involves additional evl arithmetic and vl toggles.
1363 InstructionCost SplitCost = 0;
1364 if (MICA.getID() == Intrinsic::vp_gather ||
1365 MICA.getID() == Intrinsic::vp_scatter) {
1366 auto DataLT = getTypeLegalizationCost(DataTy);
1367 auto PtrLT = getTypeLegalizationCost(PtrTy);
1368 if (DataLT.first > 1)
1369 SplitCost += DataLT.first * TTI::TCC_Expensive;
1370 if (PtrLT.first > 1)
1371 SplitCost += PtrLT.first * TTI::TCC_Expensive;
1372 }
1373
1374 // Cost is proportional to the number of memory operations implied. For
1375 // scalable vectors, we use an estimate on that number since we don't
1376 // know exactly what VL will be.
1377 auto &VTy = *cast<VectorType>(DataTy);
1378 unsigned NumLoads = getEstimatedVLFor(&VTy);
1379 return SplitCost + NumLoads * TTI::TCC_Basic;
1380}
1381
1383 const MemIntrinsicCostAttributes &MICA,
1385 unsigned Opcode = MICA.getID() == Intrinsic::masked_expandload
1386 ? Instruction::Load
1387 : Instruction::Store;
1388 Type *DataTy = MICA.getDataType();
1389 bool VariableMask = MICA.getVariableMask();
1390 Align Alignment = MICA.getAlignment();
1391 bool IsLegal = (Opcode == Instruction::Store &&
1392 isLegalMaskedCompressStore(DataTy, Alignment)) ||
1393 (Opcode == Instruction::Load &&
1394 isLegalMaskedExpandLoad(DataTy, Alignment));
1395 if (!IsLegal || CostKind != TTI::TCK_RecipThroughput)
1397 // Example compressstore sequence:
1398 // vsetivli zero, 8, e32, m2, ta, ma (ignored)
1399 // vcompress.vm v10, v8, v0
1400 // vcpop.m a1, v0
1401 // vsetvli zero, a1, e32, m2, ta, ma
1402 // vse32.v v10, (a0)
1403 // Example expandload sequence:
1404 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
1405 // vcpop.m a1, v0
1406 // vsetvli zero, a1, e32, m2, ta, ma
1407 // vle32.v v10, (a0)
1408 // vsetivli zero, 8, e32, m2, ta, ma
1409 // viota.m v12, v0
1410 // vrgather.vv v8, v10, v12, v0.t
1411 auto MemOpCost =
1412 getMemoryOpCost(Opcode, DataTy, Alignment, /*AddressSpace*/ 0, CostKind);
1413 auto LT = getTypeLegalizationCost(DataTy);
1414 SmallVector<unsigned, 4> Opcodes{RISCV::VSETVLI};
1415 if (VariableMask)
1416 Opcodes.push_back(RISCV::VCPOP_M);
1417 if (Opcode == Instruction::Store)
1418 Opcodes.append({RISCV::VCOMPRESS_VM});
1419 else
1420 Opcodes.append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1421 return MemOpCost +
1422 LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1423}
1424
1428 Type *DataTy = MICA.getDataType();
1429 Align Alignment = MICA.getAlignment();
1430
1431 if (!isLegalStridedLoadStore(DataTy, Alignment))
1433
1435 return TTI::TCC_Basic;
1436
1437 // Splitting vp intrinsics involves additional evl arithmetic and vl toggles.
1438 InstructionCost SplitCost = 0;
1439 auto LT = getTypeLegalizationCost(DataTy);
1440 if (LT.first > 1)
1441 SplitCost += LT.first * TTI::TCC_Expensive;
1442
1443 // Cost is proportional to the number of memory operations implied. For
1444 // scalable vectors, we use an estimate on that number since we don't
1445 // know exactly what VL will be.
1446 auto &VTy = *cast<VectorType>(DataTy);
1447 unsigned NumLoads = getEstimatedVLFor(&VTy);
1448 // Performant implementations of the vector extension will coalesce
1449 // elements if they fall on the same cache line
1450 uint64_t CacheLineBytes = ST->getCacheLineSize();
1451 if (!CacheLineBytes) // If no value, use default value of 64
1452 CacheLineBytes = 64;
1453 if (const ConstantInt *StrideCI =
1455 int64_t Stride = StrideCI->getSExtValue();
1456 // Bail early to avoid UB with std:abs() call
1457 if (Stride != std::numeric_limits<int64_t>::min() && Stride != 0) {
1458 uint64_t AbsStride = (uint64_t)std::abs(Stride);
1459 if (AbsStride < CacheLineBytes) {
1460 uint64_t MaxCombines = ST->getMaxVectorCoalesceElts();
1461 if ((CacheLineBytes / AbsStride) >= MaxCombines)
1462 NumLoads = divideCeil(NumLoads, MaxCombines);
1463 else
1464 // If we were to calculate CacheLineBytes / AbsStride first, would
1465 // lose accuracy
1466 NumLoads = divideCeil((NumLoads * AbsStride), CacheLineBytes);
1467 }
1468 }
1469 }
1470 return SplitCost + NumLoads * TTI::TCC_Basic;
1471}
1472
1475 // FIXME: This is a property of the default vector convention, not
1476 // all possible calling conventions. Fixing that will require
1477 // some TTI API and SLP rework.
1480 for (auto *Ty : Tys) {
1481 if (!Ty->isVectorTy())
1482 continue;
1483 Align A = DL.getPrefTypeAlign(Ty);
1484 Cost += getMemoryOpCost(Instruction::Store, Ty, A, 0, CostKind) +
1485 getMemoryOpCost(Instruction::Load, Ty, A, 0, CostKind);
1486 }
1487 return Cost;
1488}
1489
1490// Currently, these represent both throughput and codesize costs
1491// for the respective intrinsics. The costs in this table are simply
1492// instruction counts with the following adjustments made:
1493// * One vsetvli is considered free.
1495 {Intrinsic::floor, MVT::f32, 9},
1496 {Intrinsic::floor, MVT::f64, 9},
1497 {Intrinsic::ceil, MVT::f32, 9},
1498 {Intrinsic::ceil, MVT::f64, 9},
1499 {Intrinsic::trunc, MVT::f32, 7},
1500 {Intrinsic::trunc, MVT::f64, 7},
1501 {Intrinsic::round, MVT::f32, 9},
1502 {Intrinsic::round, MVT::f64, 9},
1503 {Intrinsic::roundeven, MVT::f32, 9},
1504 {Intrinsic::roundeven, MVT::f64, 9},
1505 {Intrinsic::rint, MVT::f32, 7},
1506 {Intrinsic::rint, MVT::f64, 7},
1507 {Intrinsic::nearbyint, MVT::f32, 9},
1508 {Intrinsic::nearbyint, MVT::f64, 9},
1509 {Intrinsic::bswap, MVT::i16, 3},
1510 {Intrinsic::bswap, MVT::i32, 12},
1511 {Intrinsic::bswap, MVT::i64, 31},
1512 {Intrinsic::bitreverse, MVT::i8, 17},
1513 {Intrinsic::bitreverse, MVT::i16, 24},
1514 {Intrinsic::bitreverse, MVT::i32, 33},
1515 {Intrinsic::bitreverse, MVT::i64, 52},
1516 {Intrinsic::ctpop, MVT::i8, 12},
1517 {Intrinsic::ctpop, MVT::i16, 19},
1518 {Intrinsic::ctpop, MVT::i32, 20},
1519 {Intrinsic::ctpop, MVT::i64, 21},
1520 {Intrinsic::ctlz, MVT::i8, 19},
1521 {Intrinsic::ctlz, MVT::i16, 28},
1522 {Intrinsic::ctlz, MVT::i32, 31},
1523 {Intrinsic::ctlz, MVT::i64, 35},
1524 {Intrinsic::cttz, MVT::i8, 16},
1525 {Intrinsic::cttz, MVT::i16, 23},
1526 {Intrinsic::cttz, MVT::i32, 24},
1527 {Intrinsic::cttz, MVT::i64, 25},
1528};
1529
1533 auto *RetTy = ICA.getReturnType();
1534 switch (ICA.getID()) {
1535 case Intrinsic::lrint:
1536 case Intrinsic::llrint:
1537 case Intrinsic::lround:
1538 case Intrinsic::llround: {
1539 auto LT = getTypeLegalizationCost(RetTy);
1540 Type *SrcTy = ICA.getArgTypes().front();
1541 auto SrcLT = getTypeLegalizationCost(SrcTy);
1542 if (ST->hasVInstructions() && LT.second.isVector()) {
1544 unsigned SrcEltSz = DL.getTypeSizeInBits(SrcTy->getScalarType());
1545 unsigned DstEltSz = DL.getTypeSizeInBits(RetTy->getScalarType());
1546 if (LT.second.getVectorElementType() == MVT::bf16) {
1547 if (!ST->hasVInstructionsBF16Minimal())
1549 if (DstEltSz == 32)
1550 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1551 else
1552 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1553 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1554 !ST->hasVInstructionsF16()) {
1555 if (!ST->hasVInstructionsF16Minimal())
1557 if (DstEltSz == 32)
1558 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1559 else
1560 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1561
1562 } else if (SrcEltSz > DstEltSz) {
1563 Ops = {RISCV::VFNCVT_X_F_W};
1564 } else if (SrcEltSz < DstEltSz) {
1565 Ops = {RISCV::VFWCVT_X_F_V};
1566 } else {
1567 Ops = {RISCV::VFCVT_X_F_V};
1568 }
1569
1570 // We need to use the source LMUL in the case of a narrowing op, and the
1571 // destination LMUL otherwise.
1572 if (SrcEltSz > DstEltSz)
1573 return SrcLT.first *
1574 getRISCVInstructionCost(Ops, SrcLT.second, CostKind);
1575 return LT.first * getRISCVInstructionCost(Ops, LT.second, CostKind);
1576 }
1577 break;
1578 }
1579 case Intrinsic::ceil:
1580 case Intrinsic::floor:
1581 case Intrinsic::trunc:
1582 case Intrinsic::rint:
1583 case Intrinsic::round:
1584 case Intrinsic::roundeven: {
1585 // These all use the same code.
1586 auto LT = getTypeLegalizationCost(RetTy);
1587 if (!LT.second.isVector() && TLI->isOperationCustom(ISD::FCEIL, LT.second))
1588 return LT.first * 8;
1589 break;
1590 }
1591 case Intrinsic::umin:
1592 case Intrinsic::umax:
1593 case Intrinsic::smin:
1594 case Intrinsic::smax: {
1595 auto LT = getTypeLegalizationCost(RetTy);
1596 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1597 return LT.first;
1598
1599 if (ST->hasVInstructions() && LT.second.isVector()) {
1600 unsigned Op;
1601 switch (ICA.getID()) {
1602 case Intrinsic::umin:
1603 Op = RISCV::VMINU_VV;
1604 break;
1605 case Intrinsic::umax:
1606 Op = RISCV::VMAXU_VV;
1607 break;
1608 case Intrinsic::smin:
1609 Op = RISCV::VMIN_VV;
1610 break;
1611 case Intrinsic::smax:
1612 Op = RISCV::VMAX_VV;
1613 break;
1614 }
1615 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1616 }
1617 break;
1618 }
1619 case Intrinsic::sadd_sat:
1620 case Intrinsic::ssub_sat:
1621 case Intrinsic::uadd_sat:
1622 case Intrinsic::usub_sat: {
1623 auto LT = getTypeLegalizationCost(RetTy);
1624 if (ST->hasVInstructions() && LT.second.isVector()) {
1625 unsigned Op;
1626 switch (ICA.getID()) {
1627 case Intrinsic::sadd_sat:
1628 Op = RISCV::VSADD_VV;
1629 break;
1630 case Intrinsic::ssub_sat:
1631 Op = RISCV::VSSUB_VV;
1632 break;
1633 case Intrinsic::uadd_sat:
1634 Op = RISCV::VSADDU_VV;
1635 break;
1636 case Intrinsic::usub_sat:
1637 Op = RISCV::VSSUBU_VV;
1638 break;
1639 }
1640 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1641 }
1642 break;
1643 }
1644 case Intrinsic::fma:
1645 case Intrinsic::fmuladd: {
1646 // TODO: handle promotion with f16/bf16 with zvfhmin/zvfbfmin
1647 auto LT = getTypeLegalizationCost(RetTy);
1648 if (ST->hasVInstructions() && LT.second.isVector())
1649 return LT.first *
1650 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second, CostKind);
1651 break;
1652 }
1653 case Intrinsic::fabs: {
1654 auto LT = getTypeLegalizationCost(RetTy);
1655 if (ST->hasVInstructions() && LT.second.isVector()) {
1656 // lui a0, 8
1657 // addi a0, a0, -1
1658 // vsetvli a1, zero, e16, m1, ta, ma
1659 // vand.vx v8, v8, a0
1660 // f16 with zvfhmin and bf16 with zvfhbmin
1661 if (LT.second.getVectorElementType() == MVT::bf16 ||
1662 (LT.second.getVectorElementType() == MVT::f16 &&
1663 !ST->hasVInstructionsF16()))
1664 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1665 CostKind) +
1666 2;
1667 else
1668 return LT.first *
1669 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second, CostKind);
1670 }
1671 break;
1672 }
1673 case Intrinsic::sqrt: {
1674 auto LT = getTypeLegalizationCost(RetTy);
1675 if (ST->hasVInstructions() && LT.second.isVector()) {
1678 MVT ConvType = LT.second;
1679 MVT FsqrtType = LT.second;
1680 // f16 with zvfhmin and bf16 with zvfbfmin and the type of nxv32[b]f16
1681 // will be spilt.
1682 if (LT.second.getVectorElementType() == MVT::bf16) {
1683 if (LT.second == MVT::nxv32bf16) {
1684 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1685 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1686 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1687 ConvType = MVT::nxv16f16;
1688 FsqrtType = MVT::nxv16f32;
1689 } else {
1690 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1691 FsqrtOp = {RISCV::VFSQRT_V};
1692 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1693 }
1694 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1695 !ST->hasVInstructionsF16()) {
1696 if (LT.second == MVT::nxv32f16) {
1697 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1698 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1699 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1700 ConvType = MVT::nxv16f16;
1701 FsqrtType = MVT::nxv16f32;
1702 } else {
1703 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1704 FsqrtOp = {RISCV::VFSQRT_V};
1705 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1706 }
1707 } else {
1708 FsqrtOp = {RISCV::VFSQRT_V};
1709 }
1710
1711 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType, CostKind) +
1712 getRISCVInstructionCost(ConvOp, ConvType, CostKind));
1713 }
1714 break;
1715 }
1716 case Intrinsic::cttz:
1717 case Intrinsic::ctlz:
1718 case Intrinsic::ctpop: {
1719 auto LT = getTypeLegalizationCost(RetTy);
1720 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1721 unsigned Op;
1722 switch (ICA.getID()) {
1723 case Intrinsic::cttz:
1724 Op = RISCV::VCTZ_V;
1725 break;
1726 case Intrinsic::ctlz:
1727 Op = RISCV::VCLZ_V;
1728 break;
1729 case Intrinsic::ctpop:
1730 Op = RISCV::VCPOP_V;
1731 break;
1732 }
1733 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1734 }
1735 break;
1736 }
1737 case Intrinsic::abs: {
1738 auto LT = getTypeLegalizationCost(RetTy);
1739 if (ST->hasVInstructions() && LT.second.isVector()) {
1740 // vabs.v v10, v8 (alias for vabd.vx v10, v8, zero)
1741 if (ST->hasStdExtZvabd())
1742 return LT.first *
1743 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second, CostKind);
1744
1745 // vrsub.vi v10, v8, 0
1746 // vmax.vv v8, v8, v10
1747 return LT.first *
1748 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1749 LT.second, CostKind);
1750 }
1751 break;
1752 }
1753 case Intrinsic::fshl:
1754 case Intrinsic::fshr: {
1755 if (ICA.getArgs().empty())
1756 break;
1757
1758 // Funnel-shifts are ROTL/ROTR when the first and second operand are equal.
1759 // When Zbb/Zbkb is enabled we can use a single ROL(W)/ROR(I)(W)
1760 // instruction.
1761 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1762 ICA.getArgs()[0] == ICA.getArgs()[1] &&
1763 (RetTy->getIntegerBitWidth() == 32 ||
1764 RetTy->getIntegerBitWidth() == 64) &&
1765 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1766 return 1;
1767 }
1768 break;
1769 }
1770 case Intrinsic::clmul: {
1771 auto LT = getTypeLegalizationCost(RetTy);
1772 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbkc()) {
1773 // TODO: Once custom lowering in this case for RV32 is added, this guard
1774 // should be removed and the cost model should be updated.
1775 if (!ST->is64Bit() || LT.second != MVT::i64)
1776 break;
1777 // vmv.s.x v8, a0
1778 // vclmul.vx v8, v8, a1
1779 // vmv.x.s a0, v8
1780 MVT VecVT = MVT::getScalableVectorVT(LT.second, 1);
1781 return LT.first * getRISCVInstructionCost(
1782 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1783 VecVT, CostKind);
1784 }
1785 break;
1786 }
1787 case Intrinsic::masked_udiv:
1788 return getArithmeticInstrCost(Instruction::UDiv, ICA.getReturnType(),
1789 CostKind);
1790 case Intrinsic::masked_sdiv:
1791 return getArithmeticInstrCost(Instruction::SDiv, ICA.getReturnType(),
1792 CostKind);
1793 case Intrinsic::masked_urem:
1794 return getArithmeticInstrCost(Instruction::URem, ICA.getReturnType(),
1795 CostKind);
1796 case Intrinsic::masked_srem:
1797 return getArithmeticInstrCost(Instruction::SRem, ICA.getReturnType(),
1798 CostKind);
1799 case Intrinsic::get_active_lane_mask: {
1800 if (ST->hasVInstructions()) {
1801 Type *ExpRetTy = VectorType::get(
1802 ICA.getArgTypes()[0], cast<VectorType>(RetTy)->getElementCount());
1803 auto LT = getTypeLegalizationCost(ExpRetTy);
1804
1805 // vid.v v8 // considered hoisted
1806 // vsaddu.vx v8, v8, a0
1807 // vmsltu.vx v0, v8, a1
1808 return LT.first *
1809 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1810 LT.second, CostKind);
1811 }
1812 break;
1813 }
1814 // TODO: add more intrinsic
1815 case Intrinsic::stepvector: {
1816 auto LT = getTypeLegalizationCost(RetTy);
1817 // Legalisation of illegal types involves an `index' instruction plus
1818 // (LT.first - 1) vector adds.
1819 if (ST->hasVInstructions())
1820 return getRISCVInstructionCost(RISCV::VID_V, LT.second, CostKind) +
1821 (LT.first - 1) *
1822 getRISCVInstructionCost(RISCV::VADD_VX, LT.second, CostKind);
1823 return 1 + (LT.first - 1);
1824 }
1825 case Intrinsic::vector_splice_left:
1826 case Intrinsic::vector_splice_right: {
1827 auto LT = getTypeLegalizationCost(RetTy);
1828 // Constant offsets fall through to getShuffleCost.
1829 if (!ICA.isTypeBasedOnly() && isa<ConstantInt>(ICA.getArgs()[2]))
1830 break;
1831 if (ST->hasVInstructions() && LT.second.isVector()) {
1832 return LT.first *
1833 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1834 LT.second, CostKind);
1835 }
1836 break;
1837 }
1838 case Intrinsic::experimental_cttz_elts: {
1839 if (!ST->hasVInstructions())
1840 break;
1842 Type *ArgTy = ICA.getArgTypes()[0];
1843 auto LT = getTypeLegalizationCost(ArgTy);
1844 if (!LT.second.isVector())
1845 break;
1846
1847 // If the element type is not i1, do a comparison with all-zeros.
1848 if (LT.second.getVectorElementType() != MVT::i1)
1849 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second, CostKind);
1850
1851 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind);
1852
1853 // If zero_is_poison is false, then we will generate additional
1854 // cmp + select instructions to convert -1 to EVL.
1855 Type *BoolTy = Type::getInt1Ty(RetTy->getContext());
1856 if (ICA.getArgs().size() > 1 &&
1857 cast<ConstantInt>(ICA.getArgs()[1])->isZero())
1858 Cost += getCmpSelInstrCost(Instruction::ICmp, BoolTy, RetTy,
1860 getCmpSelInstrCost(Instruction::Select, RetTy, BoolTy,
1862
1863 return LT.first * Cost;
1864 }
1865 case Intrinsic::experimental_vp_splice: {
1866 // To support type-based query from vectorizer, set the index to 0.
1867 // Note that index only change the cost from vslide.vx to vslide.vi and in
1868 // current implementations they have same costs.
1870 cast<VectorType>(ICA.getArgTypes()[0]), CostKind, {},
1872 }
1873 case Intrinsic::vp_merge: {
1874 // If an operand is a binary op and the type is legal, RISCVVectorPeephole
1875 // will likely fold the resulting vmerge.vvm away.
1877 getTypeLegalizationCost(RetTy).first == 1)
1878 return TTI::TCC_Free;
1879 break;
1880 }
1881 case Intrinsic::fptoui_sat:
1882 case Intrinsic::fptosi_sat: {
1884 bool IsSigned = ICA.getID() == Intrinsic::fptosi_sat;
1885 Type *SrcTy = ICA.getArgTypes()[0];
1886
1887 auto SrcLT = getTypeLegalizationCost(SrcTy);
1888 auto DstLT = getTypeLegalizationCost(RetTy);
1889 if (!SrcTy->isVectorTy())
1890 break;
1891
1892 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1894
1895 Cost +=
1896 getCastInstrCost(IsSigned ? Instruction::FPToSI : Instruction::FPToUI,
1897 RetTy, SrcTy, TTI::CastContextHint::None, CostKind);
1898
1899 // Handle NaN.
1900 // vmfne v0, v8, v8 # If v8[i] is NaN set v0[i] to 1.
1901 // vmerge.vim v8, v8, 0, v0 # Convert NaN to 0.
1902 Type *CondTy = RetTy->getWithNewBitWidth(1);
1903 Cost += getCmpSelInstrCost(BinaryOperator::FCmp, SrcTy, CondTy,
1905 Cost += getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
1907 return Cost;
1908 }
1909 case Intrinsic::experimental_vector_extract_last_active: {
1910 auto *ValTy = cast<VectorType>(ICA.getArgTypes()[0]);
1911 auto *MaskTy = cast<VectorType>(ICA.getArgTypes()[1]);
1912
1913 auto ValLT = getTypeLegalizationCost(ValTy);
1914 auto MaskLT = getTypeLegalizationCost(MaskTy);
1915
1916 // TODO: Return cheaper cost when the entire lane is inactive.
1917 // The expected asm sequence is:
1918 // vcpop.m a0, v0
1919 // beqz a0, exit # Return passthru when the entire lane is inactive.
1920 // vid v10, v0.t
1921 // vredmaxu.vs v10, v10, v10
1922 // vmv.x.s a0, v10
1923 // zext.b a0, a0
1924 // vslidedown.vx v8, v8, a0
1925 // vmv.x.s a0, v8
1926 // exit:
1927 // ...
1928
1929 // Find a suitable type for a stepvector.
1930 ConstantRange VScaleRange(APInt(64, 1), APInt::getZero(64));
1931 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1932 TLI->getVectorIdxTy(getDataLayout()), MaskTy->getElementCount(),
1933 /*ZeroIsPoison=*/true, &VScaleRange);
1934 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1935 Type *StepTy = Type::getIntNTy(MaskTy->getContext(), EltWidth);
1936 auto *StepVecTy = VectorType::get(StepTy, ValTy->getElementCount());
1937 auto StepLT = getTypeLegalizationCost(StepVecTy);
1938
1939 // Currently expandVectorFindLastActive cannot handle step vector split.
1940 // So return invalid when the type needs split.
1941 // FIXME: Remove this if expandVectorFindLastActive supports split vector.
1942 if (StepLT.first > 1)
1944
1946 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1947
1948 Cost += MaskLT.first *
1949 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second, CostKind);
1950 Cost += getCFInstrCost(Instruction::CondBr, CostKind, nullptr);
1951 Cost += StepLT.first *
1952 getRISCVInstructionCost(Opcodes, StepLT.second, CostKind);
1953 Cost += getCastInstrCost(Instruction::ZExt,
1954 Type::getInt64Ty(ValTy->getContext()), StepTy,
1956 Cost += ValLT.first *
1957 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1958 ValLT.second, CostKind);
1959 return Cost;
1960 }
1961 case Intrinsic::vector_interleave2:
1962 case Intrinsic::vector_deinterleave2: {
1963 if (!ST->hasStdExtZvzip())
1964 break;
1965
1966 bool IsInterleave = ICA.getID() == Intrinsic::vector_interleave2;
1967 Type *InterleavedTy = IsInterleave ? RetTy : ICA.getArgTypes().front();
1968 // ISel does not select vzip.vv if either interleave2 input is undef.
1969 if (IsInterleave && !ICA.isTypeBasedOnly() &&
1970 any_of(ICA.getArgs(),
1971 [](const Value *Arg) { return isa<UndefValue>(Arg); }))
1972 break;
1973 if (InterleavedTy->getScalarSizeInBits() == 1)
1974 break;
1975
1976 if (auto *FVT = dyn_cast<FixedVectorType>(InterleavedTy)) {
1977 auto *HalfFVT = FixedVectorType::getHalfElementsVectorType(FVT);
1978 unsigned HalfVF = HalfFVT->getNumElements();
1979 if (IsInterleave)
1980 return getShuffleCost(TTI::SK_PermuteTwoSrc, FVT, HalfFVT, CostKind,
1981 createInterleaveMask(HalfVF, 2), 0, nullptr);
1983 for (unsigned Start = 0; Start != 2; ++Start)
1985 createStrideMask(Start, 2, HalfVF), 0, nullptr);
1986 return Cost;
1987 }
1988
1989 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(InterleavedTy);
1990 if (!LT.second.isScalableVector())
1991 break;
1992 if (IsInterleave) {
1993 if (std::optional<MVT> CostVT = getZvzipVZIPCostVT(LT.second))
1994 return LT.first *
1995 getRISCVInstructionCost(RISCV::VZIP_VV, *CostVT, CostKind);
1996 } else if (std::optional<MVT> CostVT = getZvzipVUNZIPCostVT(LT.second)) {
1997 return LT.first *
1998 getRISCVInstructionCost({RISCV::VUNZIPE_V, RISCV::VUNZIPO_V},
1999 *CostVT, CostKind);
2000 }
2001 break;
2002 }
2003 }
2004
2005 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
2006 if (auto LT = getTypeLegalizationCost(RetTy);
2007 LT.second.isVector()) {
2008 MVT EltTy = LT.second.getVectorElementType();
2009 if (const auto *Entry = CostTableLookup(VectorIntrinsicCostTable,
2010 ICA.getID(), EltTy))
2011 return LT.first * Entry->Cost;
2012 }
2013 }
2014
2016}
2017
2020 const SCEV *Ptr,
2022 // Address computations for vector indexed load/store likely require an offset
2023 // and/or scaling.
2024 if (ST->hasVInstructions() && PtrTy->isVectorTy())
2025 return getArithmeticInstrCost(Instruction::Add, PtrTy, CostKind);
2026
2027 return BaseT::getAddressComputationCost(PtrTy, SE, Ptr, CostKind);
2028}
2029
2031 Type *Src,
2034 const Instruction *I) const {
2035 bool IsVectorType = isa<VectorType>(Dst) && isa<VectorType>(Src);
2036 if (!IsVectorType)
2037 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2038
2039 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
2040 // For now, skip all fixed vector cost analysis when P extension is available
2041 // to avoid crashes in getMinRVVVectorSizeInBits()
2042 if (ST->hasStdExtP() &&
2044 return 1; // Treat as single instruction cost for now
2045 }
2046
2047 // FIXME: Need to compute legalizing cost for illegal types. The current
2048 // code handles only legal types and those which can be trivially
2049 // promoted to legal.
2050 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
2051 Dst->getScalarSizeInBits() > ST->getELen())
2052 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2053
2054 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2055 assert(ISD && "Invalid opcode");
2056 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Src);
2057 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Dst);
2058
2059 // Handle i1 source and dest cases *before* calling logic in BasicTTI.
2060 // The shared implementation doesn't model vector widening during legalization
2061 // and instead assumes scalarization. In order to scalarize an <N x i1>
2062 // vector, we need to extend/trunc to/from i8. If we don't special case
2063 // this, we can get an infinite recursion cycle.
2064 switch (ISD) {
2065 default:
2066 break;
2067 case ISD::SIGN_EXTEND:
2068 case ISD::ZERO_EXTEND:
2069 if (Src->getScalarSizeInBits() == 1) {
2070 // We do not use vsext/vzext to extend from mask vector.
2071 // Instead we use the following instructions to extend from mask vector:
2072 // vmv.v.i v8, 0
2073 // vmerge.vim v8, v8, -1, v0 (repeated per split)
2074 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second, CostKind) +
2075 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
2076 DstLT.second, CostKind) +
2077 DstLT.first - 1;
2078 }
2079 break;
2080 case ISD::TRUNCATE:
2081 if (Dst->getScalarSizeInBits() == 1) {
2082 // We do not use several vncvt to truncate to mask vector. So we could
2083 // not use PowDiff to calculate it.
2084 // Instead we use the following instructions to truncate to mask vector:
2085 // vand.vi v8, v8, 1
2086 // vmsne.vi v0, v8, 0
2087 return SrcLT.first *
2088 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
2089 SrcLT.second, CostKind) +
2090 SrcLT.first - 1;
2091 }
2092 break;
2093 };
2094
2095 // Our actual lowering for the case where a wider legal type is available
2096 // uses promotion to the wider type. This is reflected in the result of
2097 // getTypeLegalizationCost, but BasicTTI assumes the widened cases are
2098 // scalarized if the legalized Src and Dst are not equal sized.
2099 const DataLayout &DL = this->getDataLayout();
2100 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
2101 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
2102 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Src),
2103 SrcLT.second.getSizeInBits()) ||
2104 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Dst),
2105 DstLT.second.getSizeInBits()) ||
2106 SrcLT.first > 1 || DstLT.first > 1)
2107 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2108
2109 // The split cost is handled by the base getCastInstrCost
2110 assert((SrcLT.first == 1) && (DstLT.first == 1) && "Illegal type");
2111
2112 int PowDiff = (int)Log2_32(DstLT.second.getScalarSizeInBits()) -
2113 (int)Log2_32(SrcLT.second.getScalarSizeInBits());
2114 switch (ISD) {
2115 case ISD::SIGN_EXTEND:
2116 case ISD::ZERO_EXTEND: {
2117 if ((PowDiff < 1) || (PowDiff > 3))
2118 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2119 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
2120 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
2121 unsigned Op =
2122 (ISD == ISD::SIGN_EXTEND) ? SExtOp[PowDiff - 1] : ZExtOp[PowDiff - 1];
2123 return getRISCVInstructionCost(Op, DstLT.second, CostKind);
2124 }
2125 case ISD::TRUNCATE:
2126 case ISD::FP_EXTEND:
2127 case ISD::FP_ROUND: {
2128 // Counts of narrow/widen instructions.
2129 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
2130 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
2131
2132 unsigned Op = (ISD == ISD::TRUNCATE) ? RISCV::VNSRL_WI
2133 : (ISD == ISD::FP_EXTEND) ? RISCV::VFWCVT_F_F_V
2134 : RISCV::VFNCVT_F_F_W;
2136 for (; SrcEltSize != DstEltSize;) {
2137 MVT ElementMVT = (ISD == ISD::TRUNCATE)
2138 ? MVT::getIntegerVT(DstEltSize)
2139 : MVT::getFloatingPointVT(DstEltSize);
2140 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2141 DstEltSize =
2142 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2143 Cost += getRISCVInstructionCost(Op, DstMVT, CostKind);
2144 }
2145 return Cost;
2146 }
2147 case ISD::FP_TO_SINT:
2148 case ISD::FP_TO_UINT: {
2149 unsigned IsSigned = ISD == ISD::FP_TO_SINT;
2150 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2151 unsigned FWCVT =
2152 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2153 unsigned FNCVT =
2154 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2155 unsigned SrcEltSize = Src->getScalarSizeInBits();
2156 unsigned DstEltSize = Dst->getScalarSizeInBits();
2158 if ((SrcEltSize == 16) &&
2159 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2160 // If the target only supports zvfhmin or it is fp16-to-i64 conversion
2161 // pre-widening to f32 and then convert f32 to integer
2162 VectorType *VecF32Ty =
2163 VectorType::get(Type::getFloatTy(Dst->getContext()),
2164 cast<VectorType>(Dst)->getElementCount());
2165 std::pair<InstructionCost, MVT> VecF32LT =
2166 getTypeLegalizationCost(VecF32Ty);
2167 Cost +=
2168 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2169 VecF32LT.second, CostKind);
2170 Cost += getCastInstrCost(Opcode, Dst, VecF32Ty, CCH, CostKind, I);
2171 return Cost;
2172 }
2173 if (DstEltSize == SrcEltSize)
2174 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2175 else if (DstEltSize > SrcEltSize)
2176 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2177 else { // (SrcEltSize > DstEltSize)
2178 // First do a narrowing conversion to an integer half the size, then
2179 // truncate if needed.
2180 MVT ElementVT = MVT::getIntegerVT(SrcEltSize / 2);
2181 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2182 Cost += getRISCVInstructionCost(FNCVT, VecVT, CostKind);
2183 if ((SrcEltSize / 2) > DstEltSize) {
2184 Type *VecTy = EVT(VecVT).getTypeForEVT(Dst->getContext());
2185 Cost +=
2186 getCastInstrCost(Instruction::Trunc, Dst, VecTy, CCH, CostKind, I);
2187 }
2188 }
2189 return Cost;
2190 }
2191 case ISD::SINT_TO_FP:
2192 case ISD::UINT_TO_FP: {
2193 unsigned IsSigned = ISD == ISD::SINT_TO_FP;
2194 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2195 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2196 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2197 unsigned SrcEltSize = Src->getScalarSizeInBits();
2198 unsigned DstEltSize = Dst->getScalarSizeInBits();
2199
2201 if ((DstEltSize == 16) &&
2202 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2203 // If the target only supports zvfhmin or it is i64-to-fp16 conversion
2204 // it is converted to f32 and then converted to f16
2205 VectorType *VecF32Ty =
2206 VectorType::get(Type::getFloatTy(Dst->getContext()),
2207 cast<VectorType>(Dst)->getElementCount());
2208 std::pair<InstructionCost, MVT> VecF32LT =
2209 getTypeLegalizationCost(VecF32Ty);
2210 Cost += getCastInstrCost(Opcode, VecF32Ty, Src, CCH, CostKind, I);
2211 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2212 DstLT.second, CostKind);
2213 return Cost;
2214 }
2215
2216 if (DstEltSize == SrcEltSize)
2217 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2218 else if (DstEltSize > SrcEltSize) {
2219 if ((DstEltSize / 2) > SrcEltSize) {
2220 VectorType *VecTy =
2221 VectorType::get(IntegerType::get(Dst->getContext(), DstEltSize / 2),
2222 cast<VectorType>(Dst)->getElementCount());
2223 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2224 Cost += getCastInstrCost(Op, VecTy, Src, CCH, CostKind, I);
2225 }
2226 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2227 } else
2228 Cost += getRISCVInstructionCost(FNCVT, DstLT.second, CostKind);
2229 return Cost;
2230 }
2231 }
2232 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2233}
2234
2235unsigned RISCVTTIImpl::getEstimatedVLFor(VectorType *Ty) const {
2236 if (isa<ScalableVectorType>(Ty)) {
2237 const unsigned EltSize = DL.getTypeSizeInBits(Ty->getElementType());
2238 const unsigned MinSize = DL.getTypeSizeInBits(Ty).getKnownMinValue();
2239 const unsigned VectorBits = *getVScaleForTuning() * RISCV::RVVBitsPerBlock;
2240 return RISCVTargetLowering::computeVLMAX(VectorBits, EltSize, MinSize);
2241 }
2242 return cast<FixedVectorType>(Ty)->getNumElements();
2243}
2244
2247 FastMathFlags FMF,
2249 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2250 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2251
2252 // Skip if scalar size of Ty is bigger than ELEN.
2253 if (Ty->getScalarSizeInBits() > ST->getELen())
2254 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2255
2256 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2257 if (Ty->getElementType()->isIntegerTy(1)) {
2258 // SelectionDAGBuilder does following transforms:
2259 // vector_reduce_{smin,umax}(<n x i1>) --> vector_reduce_or(<n x i1>)
2260 // vector_reduce_{smax,umin}(<n x i1>) --> vector_reduce_and(<n x i1>)
2261 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2262 return getArithmeticReductionCost(Instruction::Or, Ty, FMF, CostKind);
2263 else
2264 return getArithmeticReductionCost(Instruction::And, Ty, FMF, CostKind);
2265 }
2266
2267 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2269 InstructionCost ExtraCost = 0;
2270 switch (IID) {
2271 case Intrinsic::maximum:
2272 if (FMF.noNaNs()) {
2273 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2274 } else {
2275 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2276 RISCV::VFMV_F_S};
2277 // Cost of Canonical Nan + branch
2278 // lui a0, 523264
2279 // fmv.w.x fa0, a0
2280 Type *DstTy = Ty->getScalarType();
2281 const unsigned EltTyBits = DstTy->getScalarSizeInBits();
2282 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2283 ExtraCost = 1 +
2284 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2286 getCFInstrCost(Instruction::CondBr, CostKind);
2287 }
2288 break;
2289
2290 case Intrinsic::minimum:
2291 if (FMF.noNaNs()) {
2292 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2293 } else {
2294 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2295 RISCV::VFMV_F_S};
2296 // Cost of Canonical Nan + branch
2297 // lui a0, 523264
2298 // fmv.w.x fa0, a0
2299 Type *DstTy = Ty->getScalarType();
2300 const unsigned EltTyBits = DL.getTypeSizeInBits(DstTy);
2301 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2302 ExtraCost = 1 +
2303 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2305 getCFInstrCost(Instruction::CondBr, CostKind);
2306 }
2307 break;
2308 }
2309 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2310 }
2311
2312 // IR Reduction is composed by one rvv reduction instruction and vmv
2313 unsigned SplitOp;
2315 switch (IID) {
2316 default:
2317 llvm_unreachable("Unsupported intrinsic");
2318 case Intrinsic::smax:
2319 SplitOp = RISCV::VMAX_VV;
2320 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2321 break;
2322 case Intrinsic::smin:
2323 SplitOp = RISCV::VMIN_VV;
2324 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2325 break;
2326 case Intrinsic::umax:
2327 SplitOp = RISCV::VMAXU_VV;
2328 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2329 break;
2330 case Intrinsic::umin:
2331 SplitOp = RISCV::VMINU_VV;
2332 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2333 break;
2334 case Intrinsic::maxnum:
2335 SplitOp = RISCV::VFMAX_VV;
2336 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2337 break;
2338 case Intrinsic::minnum:
2339 SplitOp = RISCV::VFMIN_VV;
2340 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2341 break;
2342 }
2343 // Add a cost for data larger than LMUL8
2344 InstructionCost SplitCost =
2345 (LT.first > 1) ? (LT.first - 1) *
2346 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2347 : 0;
2348 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2349}
2350
2353 std::optional<FastMathFlags> FMF,
2355 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2356 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2357
2358 // Skip if scalar size of Ty is bigger than ELEN.
2359 if (Ty->getScalarSizeInBits() > ST->getELen())
2360 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2361
2362 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2363 assert(ISD && "Invalid opcode");
2364
2365 if (ISD != ISD::ADD && ISD != ISD::OR && ISD != ISD::XOR && ISD != ISD::AND &&
2366 ISD != ISD::FADD)
2367 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2368
2369 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2370 Type *ElementTy = Ty->getElementType();
2371 if (ElementTy->isIntegerTy(1)) {
2372 // Example sequences:
2373 // vfirst.m a0, v0
2374 // seqz a0, a0
2375 if (LT.second == MVT::v1i1)
2376 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind) +
2377 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2379
2380 if (ISD == ISD::AND) {
2381 // Example sequences:
2382 // vmand.mm v8, v9, v8 ; needed every time type is split
2383 // vmnot.m v8, v0 ; alias for vmnand
2384 // vcpop.m a0, v8
2385 // seqz a0, a0
2386
2387 // See the discussion: https://github.com/llvm/llvm-project/pull/119160
2388 // For LMUL <= 8, there is no splitting,
2389 // the sequences are vmnot, vcpop and seqz.
2390 // When LMUL > 8 and split = 1,
2391 // the sequences are vmnand, vcpop and seqz.
2392 // When LMUL > 8 and split > 1,
2393 // the sequences are (LT.first-2) * vmand, vmnand, vcpop and seqz.
2394 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2395 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second, CostKind) +
2396 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second, CostKind) +
2397 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2398 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2400 } else if (ISD == ISD::XOR || ISD == ISD::ADD) {
2401 // Example sequences:
2402 // vsetvli a0, zero, e8, mf8, ta, ma
2403 // vmxor.mm v8, v0, v8 ; needed every time type is split
2404 // vcpop.m a0, v8
2405 // andi a0, a0, 1
2406 return (LT.first - 1) *
2407 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind) +
2408 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) + 1;
2409 } else {
2410 assert(ISD == ISD::OR);
2411 // Example sequences:
2412 // vsetvli a0, zero, e8, mf8, ta, ma
2413 // vmor.mm v8, v9, v8 ; needed every time type is split
2414 // vcpop.m a0, v0
2415 // snez a0, a0
2416 return (LT.first - 1) *
2417 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second, CostKind) +
2418 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2419 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2421 }
2422 }
2423
2424 // IR Reduction of or/and is composed by one vmv and one rvv reduction
2425 // instruction, and others is composed by two vmv and one rvv reduction
2426 // instruction
2427 unsigned SplitOp;
2429 switch (ISD) {
2430 case ISD::ADD:
2431 SplitOp = RISCV::VADD_VV;
2432 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2433 break;
2434 case ISD::OR:
2435 SplitOp = RISCV::VOR_VV;
2436 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2437 break;
2438 case ISD::XOR:
2439 SplitOp = RISCV::VXOR_VV;
2440 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2441 break;
2442 case ISD::AND:
2443 SplitOp = RISCV::VAND_VV;
2444 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2445 break;
2446 case ISD::FADD:
2447 // We can't promote f16/bf16 fadd reductions.
2448 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2449 LT.second.getScalarType() == MVT::bf16)
2450 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2452 Opcodes.push_back(RISCV::VFMV_S_F);
2453 for (unsigned i = 0; i < LT.first.getValue(); i++)
2454 Opcodes.push_back(RISCV::VFREDOSUM_VS);
2455 Opcodes.push_back(RISCV::VFMV_F_S);
2456 return getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2457 }
2458 SplitOp = RISCV::VFADD_VV;
2459 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2460 break;
2461 }
2462 // Add a cost for data larger than LMUL8
2463 InstructionCost SplitCost =
2464 (LT.first > 1) ? (LT.first - 1) *
2465 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2466 : 0;
2467 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2468}
2469
2471 unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy,
2472 std::optional<FastMathFlags> FMF, TTI::TargetCostKind CostKind) const {
2473 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2474 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2475 FMF, CostKind);
2476
2477 // Skip if scalar size of ResTy is bigger than ELEN.
2478 if (ResTy->getScalarSizeInBits() > ST->getELen())
2479 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2480 FMF, CostKind);
2481
2482 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2483 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2484 FMF, CostKind);
2485
2486 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2487
2488 if (IsUnsigned && Opcode == Instruction::Add &&
2489 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2490 // Represent vector_reduce_add(ZExt(<n x i1>)) as
2491 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
2492 return LT.first *
2493 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind);
2494 }
2495
2496 if (ResTy->getScalarSizeInBits() != 2 * LT.second.getScalarSizeInBits())
2497 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2498 FMF, CostKind);
2499
2500 return (LT.first - 1) +
2501 getArithmeticReductionCost(Opcode, ValTy, FMF, CostKind);
2502}
2503
2507 assert(OpInfo.isConstant() && "non constant operand?");
2508 if (!isa<VectorType>(Ty))
2509 // FIXME: We need to account for immediate materialization here, but doing
2510 // a decent job requires more knowledge about the immediate than we
2511 // currently have here.
2512 return 0;
2513
2514 if (OpInfo.isUniform())
2515 // vmv.v.i, vmv.v.x, or vfmv.v.f
2516 // We ignore the cost of the scalar constant materialization to be consistent
2517 // with how we treat scalar constants themselves just above.
2518 return 1;
2519
2520 return getConstantPoolLoadCost(Ty, CostKind);
2521}
2522
2524 Align Alignment,
2525 unsigned AddressSpace,
2527 TTI::OperandValueInfo OpInfo,
2528 const Instruction *I) const {
2529 EVT VT = TLI->getValueType(DL, Src, true);
2530 // Type legalization can't handle structs, and load latency isn't handled here
2531 if (VT == MVT::Other ||
2532 (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency))
2533 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2534 CostKind, OpInfo, I);
2535
2537 if (Opcode == Instruction::Store && OpInfo.isConstant())
2538 Cost += getStoreImmCost(Src, OpInfo, CostKind);
2539
2540 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Src);
2541
2542 InstructionCost BaseCost = [&]() {
2543 InstructionCost Cost = LT.first;
2545 return Cost;
2546
2547 // Our actual lowering for the case where a wider legal type is available
2548 // uses the a VL predicated load on the wider type. This is reflected in
2549 // the result of getTypeLegalizationCost, but BasicTTI assumes the
2550 // widened cases are scalarized.
2551 const DataLayout &DL = this->getDataLayout();
2552 if (Src->isVectorTy() && LT.second.isVector() &&
2553 TypeSize::isKnownLT(DL.getTypeStoreSizeInBits(Src),
2554 LT.second.getSizeInBits()))
2555 return Cost;
2556
2557 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2558 CostKind, OpInfo, I);
2559 }();
2560
2561 // Assume memory ops cost scale with the number of vector registers
2562 // possible accessed by the instruction. Note that BasicTTI already
2563 // handles the LT.first term for us.
2564 if (ST->hasVInstructions() && LT.second.isVector() &&
2566 BaseCost *= TLI->getLMULCost(LT.second);
2567 return Cost + BaseCost;
2568}
2569
2571 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
2573 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
2575 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2576 Op1Info, Op2Info, I);
2577
2578 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2579 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2580 Op1Info, Op2Info, I);
2581
2582 // Skip if scalar size of ValTy is bigger than ELEN.
2583 if (ValTy->isVectorTy() && ValTy->getScalarSizeInBits() > ST->getELen())
2584 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2585 Op1Info, Op2Info, I);
2586
2587 auto GetConstantMatCost =
2588 [&](TTI::OperandValueInfo OpInfo) -> InstructionCost {
2589 if (OpInfo.isUniform())
2590 // We return 0 we currently ignore the cost of materializing scalar
2591 // constants in GPRs.
2592 return 0;
2593
2594 return getConstantPoolLoadCost(ValTy, CostKind);
2595 };
2596
2597 InstructionCost ConstantMatCost;
2598 if (Op1Info.isConstant())
2599 ConstantMatCost += GetConstantMatCost(Op1Info);
2600 if (Op2Info.isConstant())
2601 ConstantMatCost += GetConstantMatCost(Op2Info);
2602
2603 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2604 if (Opcode == Instruction::Select && LT.second.isVector()) {
2605 if (CondTy->isVectorTy()) {
2606 if (ValTy->getScalarSizeInBits() == 1) {
2607 // vmandn.mm v8, v8, v9
2608 // vmand.mm v9, v0, v9
2609 // vmor.mm v0, v9, v8
2610 return ConstantMatCost +
2611 LT.first *
2612 getRISCVInstructionCost(
2613 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2614 LT.second, CostKind);
2615 }
2616 // vselect and max/min are supported natively.
2617 return ConstantMatCost +
2618 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2619 CostKind);
2620 }
2621
2622 if (ValTy->getScalarSizeInBits() == 1) {
2623 // vmv.v.x v9, a0
2624 // vmsne.vi v9, v9, 0
2625 // vmandn.mm v8, v8, v9
2626 // vmand.mm v9, v0, v9
2627 // vmor.mm v0, v9, v8
2628 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2629 return ConstantMatCost +
2630 LT.first *
2631 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2632 InterimVT, CostKind) +
2633 LT.first * getRISCVInstructionCost(
2634 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2635 LT.second, CostKind);
2636 }
2637
2638 // vmv.v.x v10, a0
2639 // vmsne.vi v0, v10, 0
2640 // vmerge.vvm v8, v9, v8, v0
2641 return ConstantMatCost +
2642 LT.first * getRISCVInstructionCost(
2643 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2644 LT.second, CostKind);
2645 }
2646
2647 if ((Opcode == Instruction::ICmp) && ValTy->isVectorTy() &&
2648 CmpInst::isIntPredicate(VecPred)) {
2649 // Use VMSLT_VV to represent VMSEQ, VMSNE, VMSLTU, VMSLEU, VMSLT, VMSLE
2650 // provided they incur the same cost across all implementations
2651 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2652 LT.second,
2653 CostKind);
2654 }
2655
2656 if ((Opcode == Instruction::FCmp) && ValTy->isVectorTy() &&
2657 CmpInst::isFPPredicate(VecPred)) {
2658
2659 // Use VMXOR_MM and VMXNOR_MM to generate all true/false mask
2660 if ((VecPred == CmpInst::FCMP_FALSE) || (VecPred == CmpInst::FCMP_TRUE))
2661 return ConstantMatCost +
2662 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind);
2663
2664 // If we do not support the input floating point vector type, use the base
2665 // one which will calculate as:
2666 // ScalarizeCost + Num * Cost for fixed vector,
2667 // InvalidCost for scalable vector.
2668 if ((ValTy->getScalarSizeInBits() == 16 && !ST->hasVInstructionsF16()) ||
2669 (ValTy->getScalarSizeInBits() == 32 && !ST->hasVInstructionsF32()) ||
2670 (ValTy->getScalarSizeInBits() == 64 && !ST->hasVInstructionsF64()))
2671 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2672 Op1Info, Op2Info, I);
2673
2674 // Assuming vector fp compare and mask instructions are all the same cost
2675 // until a need arises to differentiate them.
2676 switch (VecPred) {
2677 case CmpInst::FCMP_ONE: // vmflt.vv + vmflt.vv + vmor.mm
2678 case CmpInst::FCMP_ORD: // vmfeq.vv + vmfeq.vv + vmand.mm
2679 case CmpInst::FCMP_UNO: // vmfne.vv + vmfne.vv + vmor.mm
2680 case CmpInst::FCMP_UEQ: // vmflt.vv + vmflt.vv + vmnor.mm
2681 return ConstantMatCost +
2682 LT.first * getRISCVInstructionCost(
2683 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2684 LT.second, CostKind);
2685
2686 case CmpInst::FCMP_UGT: // vmfle.vv + vmnot.m
2687 case CmpInst::FCMP_UGE: // vmflt.vv + vmnot.m
2688 case CmpInst::FCMP_ULT: // vmfle.vv + vmnot.m
2689 case CmpInst::FCMP_ULE: // vmflt.vv + vmnot.m
2690 return ConstantMatCost +
2691 LT.first *
2692 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2693 LT.second, CostKind);
2694
2695 case CmpInst::FCMP_OEQ: // vmfeq.vv
2696 case CmpInst::FCMP_OGT: // vmflt.vv
2697 case CmpInst::FCMP_OGE: // vmfle.vv
2698 case CmpInst::FCMP_OLT: // vmflt.vv
2699 case CmpInst::FCMP_OLE: // vmfle.vv
2700 case CmpInst::FCMP_UNE: // vmfne.vv
2701 return ConstantMatCost +
2702 LT.first *
2703 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second, CostKind);
2704 default:
2705 break;
2706 }
2707 }
2708
2709 // With ShortForwardBranchOpt or ConditionalMoveFusion, scalar icmp + select
2710 // instructions will lower to SELECT_CC and lower to PseudoCCMOVGPR which will
2711 // generate a conditional branch + mv. The cost of scalar (icmp + select) will
2712 // be (0 + select instr cost).
2713 if (ST->hasConditionalMoveFusion() && I && isa<ICmpInst>(I) &&
2714 ValTy->isIntegerTy() && !I->user_empty()) {
2715 if (all_of(I->users(), [&](const User *U) {
2716 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2717 U->getType()->isIntegerTy() &&
2718 !isa<ConstantData>(U->getOperand(1)) &&
2719 !isa<ConstantData>(U->getOperand(2));
2720 }))
2721 return 0;
2722 }
2723
2724 // TODO: Add cost for scalar type.
2725
2726 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2727 Op1Info, Op2Info, I);
2728}
2729
2732 const Instruction *I) const {
2734 return Opcode == Instruction::PHI ? 0 : 1;
2735 // Branches are assumed to be predicted.
2736 return 0;
2737}
2738
2740 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
2741 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
2742 assert(Val->isVectorTy() && "This must be a vector type");
2743
2744 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
2745 // For now, skip all fixed vector cost analysis when P extension is available
2746 // to avoid crashes in getMinRVVVectorSizeInBits()
2747 if (ST->hasStdExtP() && isa<FixedVectorType>(Val)) {
2748 return 1; // Treat as single instruction cost for now
2749 }
2750
2751 if (Opcode != Instruction::ExtractElement &&
2752 Opcode != Instruction::InsertElement)
2753 return BaseT::getVectorInstrCost(Opcode, Val, CostKind, Index, Op0, Op1,
2754 VIC);
2755
2756 // Scalar splat operand can be folded for vector ops that support splatting
2757 // the scalar operand, so the explicit insertelement is free in this context.
2758 if (Opcode == Instruction::InsertElement &&
2759 VIC == TTI::VectorInstrContext::SplatOpFolded &&
2760 ST->sinkSplatOperands() && Index == 0)
2761 return TTI::TCC_Free;
2762
2763 // Legalize the type.
2764 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Val);
2765
2766 // This type is legalized to a scalar type.
2767 if (!LT.second.isVector()) {
2768 auto *FixedVecTy = cast<FixedVectorType>(Val);
2769 // If Index is a known constant, cost is zero.
2770 if (Index != -1U)
2771 return 0;
2772 // Extract/InsertElement with non-constant index is very costly when
2773 // scalarized; estimate cost of loads/stores sequence via the stack:
2774 // ExtractElement cost: store vector to stack, load scalar;
2775 // InsertElement cost: store vector to stack, store scalar, load vector.
2776 Type *ElemTy = FixedVecTy->getElementType();
2777 auto NumElems = FixedVecTy->getNumElements();
2778 auto Align = DL.getPrefTypeAlign(ElemTy);
2779 InstructionCost LoadCost =
2780 getMemoryOpCost(Instruction::Load, ElemTy, Align, 0, CostKind);
2781 InstructionCost StoreCost =
2782 getMemoryOpCost(Instruction::Store, ElemTy, Align, 0, CostKind);
2783 return Opcode == Instruction::ExtractElement
2784 ? StoreCost * NumElems + LoadCost
2785 : (StoreCost + LoadCost) * NumElems + StoreCost;
2786 }
2787
2788 // For unsupported scalable vector.
2789 if (LT.second.isScalableVector() && !LT.first.isValid())
2790 return LT.first;
2791
2792 // Mask vector extract/insert is expanded via e8.
2793 if (Val->getScalarSizeInBits() == 1) {
2794 VectorType *WideTy =
2796 cast<VectorType>(Val)->getElementCount());
2797 if (Opcode == Instruction::ExtractElement) {
2798 InstructionCost ExtendCost
2799 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2801 InstructionCost ExtractCost
2802 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2803 return ExtendCost + ExtractCost;
2804 }
2805 InstructionCost ExtendCost
2806 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2808 InstructionCost InsertCost
2809 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2810 InstructionCost TruncCost
2811 = getCastInstrCost(Instruction::Trunc, Val, WideTy,
2813 return ExtendCost + InsertCost + TruncCost;
2814 }
2815
2816
2817 // In RVV, we could use vslidedown + vmv.x.s to extract element from vector
2818 // and vslideup + vmv.s.x to insert element to vector.
2819 unsigned MoveOpc;
2820 if (LT.second.isFloatingPoint())
2821 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2822 : RISCV::VFMV_F_S;
2823 else
2824 MoveOpc =
2825 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2826 InstructionCost BaseCost =
2827 getRISCVInstructionCost(MoveOpc, LT.second, CostKind);
2828 // When insertelement we should add the index with 1 as the input of vslideup.
2829 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2830
2831 if (Index != -1U) {
2832 // The type may be split. For fixed-width vectors we can normalize the
2833 // index to the new type.
2834 if (LT.second.isFixedLengthVector()) {
2835 unsigned Width = LT.second.getVectorNumElements();
2836 Index = Index % Width;
2837 }
2838
2839 // If exact VLEN is known, we will insert/extract into the appropriate
2840 // subvector with no additional subvector insert/extract cost.
2841 if (auto VLEN = ST->getRealVLen()) {
2842 unsigned EltSize = LT.second.getScalarSizeInBits();
2843 unsigned M1Max = *VLEN / EltSize;
2844 Index = Index % M1Max;
2845 }
2846
2847 if (Index == 0)
2848 // We can extract/insert the first element without vslidedown/vslideup.
2849 SlideCost = 0;
2850 else if (Opcode == Instruction::InsertElement)
2851 SlideCost = 1; // With a constant index, we do not need to use addi.
2852 }
2853
2854 // When the vector needs to split into multiple register groups and the index
2855 // exceeds single vector register group, we need to insert/extract the element
2856 // via stack.
2857 if (LT.first > 1 &&
2858 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2859 LT.second.isScalableVector()))) {
2860 Type *ScalarType = Val->getScalarType();
2861 Align VecAlign = DL.getPrefTypeAlign(Val);
2862 Align SclAlign = DL.getPrefTypeAlign(ScalarType);
2863 // Extra addi for unknown index.
2864 InstructionCost IdxCost = Index == -1U ? 1 : 0;
2865
2866 // Store all split vectors into stack and load the target element.
2867 if (Opcode == Instruction::ExtractElement)
2868 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2869 getMemoryOpCost(Instruction::Load, ScalarType, SclAlign, 0,
2870 CostKind) +
2871 IdxCost;
2872
2873 // Store all split vectors into stack and store the target element and load
2874 // vectors back.
2875 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2876 getMemoryOpCost(Instruction::Load, Val, VecAlign, 0, CostKind) +
2877 getMemoryOpCost(Instruction::Store, ScalarType, SclAlign, 0,
2878 CostKind) +
2879 IdxCost;
2880 }
2881
2882 // Extract i64 in the target that has XLEN=32 need more instruction.
2883 if (Val->getScalarType()->isIntegerTy() &&
2884 ST->getXLen() < Val->getScalarSizeInBits()) {
2885 // For extractelement, we need the following instructions:
2886 // vsetivli zero, 1, e64, m1, ta, mu (not count)
2887 // vslidedown.vx v8, v8, a0
2888 // vmv.x.s a0, v8
2889 // li a1, 32
2890 // vsrl.vx v8, v8, a1
2891 // vmv.x.s a1, v8
2892
2893 // For insertelement, we need the following instructions:
2894 // vsetivli zero, 2, e32, m4, ta, ma (don't count)
2895 // vslide1down.vx v12, v8, a0
2896 // vslide1down.vx v12, v12, a1
2897 // addi a0, a2, 1
2898 // vsetvli zero, a0, e64, m4, tu, ma (don't count)
2899 // vslideup.vx v8, v12, a2
2900
2901 // TODO: should we count these special vsetvlis?
2902 BaseCost =
2903 Opcode == Instruction::InsertElement
2904 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2905 RISCV::VSLIDE1DOWN_VX,
2906 RISCV::VSLIDEUP_VX},
2907 LT.second, CostKind)
2908 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2909 RISCV::VSRL_VX, RISCV::VMV_X_S},
2910 LT.second, CostKind);
2911 }
2912 return BaseCost + SlideCost;
2913}
2914
2918 unsigned Index) const {
2919 if (isa<FixedVectorType>(Val))
2921 Index);
2922
2923 // TODO: This code replicates what LoopVectorize.cpp used to do when asking
2924 // for the cost of extracting the last lane of a scalable vector. It probably
2925 // needs a more accurate cost.
2926 ElementCount EC = cast<VectorType>(Val)->getElementCount();
2927 assert(Index < EC.getKnownMinValue() && "Unexpected reverse index");
2928 return getVectorInstrCost(Opcode, Val, CostKind,
2929 EC.getKnownMinValue() - 1 - Index, nullptr,
2930 nullptr);
2931}
2932
2933/// Check to see if this instruction is expected to be combined to a simpler
2934/// operation during/before lowering. If so return the cost of the combined
2935/// operation rather than provided one. For instance, `udiv i16 %X, 2` is likely
2936/// to be combined to `lshr i16 %X, 1`, so return the cost of a `lshr` rather
2937/// than the cost of a `udiv`
2938std::optional<InstructionCost>
2940 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2942 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
2943 // Vector unsigned division/remainder will be simplified to shifts/masks.
2944 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2945 Opd2Info.isConstant() && Opd2Info.isPowerOf2()) {
2946 if (Opcode == Instruction::UDiv)
2947 return getArithmeticInstrCost(Instruction::LShr, Ty, CostKind, Opd1Info,
2948 Opd2Info.getNoProps());
2949 // UREM
2950 return getArithmeticInstrCost(Instruction::And, Ty, CostKind, Opd1Info,
2951 Opd2Info.getNoProps());
2952 }
2953 return std::nullopt;
2954}
2955
2957 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2959 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
2960
2961 // TODO: Handle more cost kinds.
2963 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2964 Args, CtxI);
2965
2966 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2967 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2968 Args, CtxI);
2969
2970 // Skip if scalar size of Ty is bigger than ELEN.
2971 if (isa<VectorType>(Ty) && Ty->getScalarSizeInBits() > ST->getELen())
2972 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2973 Args, CtxI);
2974
2975 if (std::optional<InstructionCost> CombinedCost =
2977 Op2Info, Args, CtxI))
2978 return *CombinedCost;
2979
2980 // Legalize the type.
2981 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2982 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2983
2984 // TODO: Handle scalar type.
2985 if (!LT.second.isVector()) {
2986 static const CostTblEntry DivTbl[]{
2987 {ISD::UDIV, MVT::i32, TTI::TCC_Expensive},
2988 {ISD::UDIV, MVT::i64, TTI::TCC_Expensive},
2989 {ISD::SDIV, MVT::i32, TTI::TCC_Expensive},
2990 {ISD::SDIV, MVT::i64, TTI::TCC_Expensive},
2991 {ISD::UREM, MVT::i32, TTI::TCC_Expensive},
2992 {ISD::UREM, MVT::i64, TTI::TCC_Expensive},
2993 {ISD::SREM, MVT::i32, TTI::TCC_Expensive},
2994 {ISD::SREM, MVT::i64, TTI::TCC_Expensive}};
2995 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
2996 if (const auto *Entry = CostTableLookup(DivTbl, ISDOpcode, LT.second))
2997 return Entry->Cost * LT.first;
2998
2999 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
3000 Args, CtxI);
3001 }
3002
3003 // f16 with zvfhmin and bf16 will be promoted to f32.
3004 // FIXME: nxv32[b]f16 will be custom lowered and split.
3005 InstructionCost CastCost = 0;
3006 if ((LT.second.getVectorElementType() == MVT::f16 ||
3007 LT.second.getVectorElementType() == MVT::bf16) &&
3008 TLI->getOperationAction(ISDOpcode, LT.second) ==
3010 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
3011 Type *PromotedTy = EVT(PromotedVT).getTypeForEVT(Ty->getContext());
3012 Type *LegalTy = EVT(LT.second).getTypeForEVT(Ty->getContext());
3013 // Add cost of extending arguments
3014 CastCost += LT.first * Args.size() *
3015 getCastInstrCost(Instruction::FPExt, PromotedTy, LegalTy,
3017 // Add cost of truncating result
3018 CastCost +=
3019 LT.first * getCastInstrCost(Instruction::FPTrunc, LegalTy, PromotedTy,
3021 // Compute cost of op in promoted type
3022 LT.second = PromotedVT;
3023 }
3024
3025 auto getConstantMatCost =
3026 [&](unsigned Operand, TTI::OperandValueInfo OpInfo) -> InstructionCost {
3027 if (OpInfo.isUniform() && canSplatOperand(Opcode, Operand))
3028 // Two sub-cases:
3029 // * Has a 5 bit immediate operand which can be splatted.
3030 // * Has a larger immediate which must be materialized in scalar register
3031 // We return 0 for both as we currently ignore the cost of materializing
3032 // scalar constants in GPRs.
3033 return 0;
3034
3035 return getConstantPoolLoadCost(Ty, CostKind);
3036 };
3037
3038 // Add the cost of materializing any constant vectors required.
3039 InstructionCost ConstantMatCost = 0;
3040 if (Op1Info.isConstant())
3041 ConstantMatCost += getConstantMatCost(0, Op1Info);
3042 if (Op2Info.isConstant())
3043 ConstantMatCost += getConstantMatCost(1, Op2Info);
3044
3045 unsigned Op;
3046 switch (ISDOpcode) {
3047 case ISD::ADD:
3048 case ISD::SUB:
3049 Op = RISCV::VADD_VV;
3050 break;
3051 case ISD::SHL:
3052 case ISD::SRL:
3053 case ISD::SRA:
3054 Op = RISCV::VSLL_VV;
3055 break;
3056 case ISD::AND:
3057 case ISD::OR:
3058 case ISD::XOR:
3059 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
3060 break;
3061 case ISD::MUL:
3062 case ISD::MULHS:
3063 case ISD::MULHU:
3064 Op = RISCV::VMUL_VV;
3065 break;
3066 case ISD::SDIV:
3067 case ISD::UDIV:
3068 Op = RISCV::VDIV_VV;
3069 break;
3070 case ISD::SREM:
3071 case ISD::UREM:
3072 Op = RISCV::VREM_VV;
3073 break;
3074 case ISD::FADD:
3075 case ISD::FSUB:
3076 Op = RISCV::VFADD_VV;
3077 break;
3078 case ISD::FMUL:
3079 Op = RISCV::VFMUL_VV;
3080 break;
3081 case ISD::FDIV:
3082 Op = RISCV::VFDIV_VV;
3083 break;
3084 case ISD::FNEG:
3085 Op = RISCV::VFSGNJN_VV;
3086 break;
3087 default:
3088 // Assuming all other instructions have the same cost until a need arises to
3089 // differentiate them.
3090 return CastCost + ConstantMatCost +
3091 BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
3092 Args, CtxI);
3093 }
3094
3095 InstructionCost InstrCost = getRISCVInstructionCost(Op, LT.second, CostKind);
3096 // We use BasicTTIImpl to calculate scalar costs, which assumes floating point
3097 // ops are twice as expensive as integer ops. Do the same for vectors so
3098 // scalar floating point ops aren't cheaper than their vector equivalents.
3099 if (Ty->isFPOrFPVectorTy())
3100 InstrCost *= 2;
3101 return CastCost + ConstantMatCost + LT.first * InstrCost;
3102}
3103
3104// TODO: Deduplicate from TargetTransformInfoImplCRTPBase.
3106 ArrayRef<const Value *> Ptrs, const Value *Base,
3107 const TTI::PointersChainInfo &Info, Type *AccessTy,
3108 const TTI::TargetCostKind CostKind) const {
3110 // In the basic model we take into account GEP instructions only
3111 // (although here can come alloca instruction, a value, constants and/or
3112 // constant expressions, PHIs, bitcasts ... whatever allowed to be used as a
3113 // pointer). Typically, if Base is a not a GEP-instruction and all the
3114 // pointers are relative to the same base address, all the rest are
3115 // either GEP instructions, PHIs, bitcasts or constants. When we have same
3116 // base, we just calculate cost of each non-Base GEP as an ADD operation if
3117 // any their index is a non-const.
3118 // If no known dependencies between the pointers cost is calculated as a sum
3119 // of costs of GEP instructions.
3120 for (auto [I, V] : enumerate(Ptrs)) {
3121 const auto *GEP = dyn_cast<GetElementPtrInst>(V);
3122 if (!GEP)
3123 continue;
3124 if (Info.isSameBase() && V != Base) {
3125 if (GEP->hasAllConstantIndices())
3126 continue;
3127 // If the chain is unit-stride and BaseReg + stride*i is a legal
3128 // addressing mode, then presume the base GEP is sitting around in a
3129 // register somewhere and check if we can fold the offset relative to
3130 // it.
3131 unsigned Stride = DL.getTypeStoreSize(AccessTy);
3132 if (Info.isUnitStride() &&
3133 isLegalAddressingMode(AccessTy,
3134 /* BaseGV */ nullptr,
3135 /* BaseOffset */ Stride * I,
3136 /* HasBaseReg */ true,
3137 /* Scale */ 0,
3138 GEP->getType()->getPointerAddressSpace()))
3139 continue;
3140 Cost += getArithmeticInstrCost(Instruction::Add, GEP->getType(), CostKind,
3141 {TTI::OK_AnyValue, TTI::OP_None},
3142 {TTI::OK_AnyValue, TTI::OP_None}, {});
3143 } else {
3144 SmallVector<const Value *> Indices(GEP->indices());
3145 Cost += getGEPCost(GEP->getSourceElementType(), GEP->getPointerOperand(),
3146 Indices, CostKind, AccessTy);
3147 }
3148 }
3149 return Cost;
3150}
3151
3154 OptimizationRemarkEmitter *ORE) const {
3155 // TODO: More tuning on benchmarks and metrics with changes as needed
3156 // would apply to all settings below to enable performance.
3157
3158
3159 if (ST->enableDefaultUnroll())
3160 return BasicTTIImplBase::getUnrollingPreferences(L, SE, UP, ORE);
3161
3162 // Enable Upper bound unrolling universally, not dependent upon the conditions
3163 // below.
3164 UP.UpperBound = true;
3165
3166 // Disable loop unrolling for Oz and Os.
3167 UP.OptSizeThreshold = 0;
3169 if (L->getHeader()->getParent()->hasOptSize())
3170 return;
3171
3172 SmallVector<BasicBlock *, 4> ExitingBlocks;
3173 L->getExitingBlocks(ExitingBlocks);
3174 LLVM_DEBUG(dbgs() << "Loop has:\n"
3175 << "Blocks: " << L->getNumBlocks() << "\n"
3176 << "Exit blocks: " << ExitingBlocks.size() << "\n");
3177
3178 // Only allow another exit other than the latch. This acts as an early exit
3179 // as it mirrors the profitability calculation of the runtime unroller.
3180 if (ExitingBlocks.size() > 2)
3181 return;
3182
3183 // Limit the CFG of the loop body for targets with a branch predictor.
3184 // Allowing 4 blocks permits if-then-else diamonds in the body.
3185 if (L->getNumBlocks() > 4)
3186 return;
3187
3188 // Scan the loop: don't unroll loops with calls as this could prevent
3189 // inlining. Don't unroll auto-vectorized loops either, though do allow
3190 // unrolling of the scalar remainder.
3191 bool IsVectorized = getBooleanLoopAttribute(L, "llvm.loop.isvectorized");
3193 for (auto *BB : L->getBlocks()) {
3194 for (auto &I : *BB) {
3195 // Both auto-vectorized loops and the scalar remainder have the
3196 // isvectorized attribute, so differentiate between them by the presence
3197 // of vector instructions.
3198 if (IsVectorized && (I.getType()->isVectorTy() ||
3199 llvm::any_of(I.operand_values(), [](Value *V) {
3200 return V->getType()->isVectorTy();
3201 })))
3202 return;
3203
3204 if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
3205 const Function *F = cast<CallBase>(I).getCalledFunction();
3206 if (!F || isLoweredToCall(F))
3207 return;
3208 }
3209
3210 SmallVector<const Value *> Operands(I.operand_values());
3213 }
3214 }
3215
3216 LLVM_DEBUG(dbgs() << "Cost of loop: " << Cost << "\n");
3217
3218 UP.Partial = true;
3219 UP.Runtime = true;
3220 UP.UnrollRemainder = true;
3221 UP.UnrollAndJam = true;
3222
3223 // Force unrolling small loops can be very useful because of the branch
3224 // taken cost of the backedge.
3225 if (Cost < 12)
3226 UP.Force = true;
3227}
3228
3233
3235 MemIntrinsicInfo &Info) const {
3236 const DataLayout &DL = getDataLayout();
3237 Intrinsic::ID IID = Inst->getIntrinsicID();
3238 LLVMContext &C = Inst->getContext();
3239 bool HasMask = false;
3240
3241 auto getSegNum = [](const IntrinsicInst *II, unsigned PtrOperandNo,
3242 bool IsWrite) -> int64_t {
3243 if (auto *TarExtTy =
3244 dyn_cast<TargetExtType>(II->getArgOperand(0)->getType()))
3245 return TarExtTy->getIntParameter(0);
3246
3247 return 1;
3248 };
3249
3250 switch (IID) {
3251 case Intrinsic::riscv_vle_mask:
3252 case Intrinsic::riscv_vse_mask:
3253 case Intrinsic::riscv_vlseg2_mask:
3254 case Intrinsic::riscv_vlseg3_mask:
3255 case Intrinsic::riscv_vlseg4_mask:
3256 case Intrinsic::riscv_vlseg5_mask:
3257 case Intrinsic::riscv_vlseg6_mask:
3258 case Intrinsic::riscv_vlseg7_mask:
3259 case Intrinsic::riscv_vlseg8_mask:
3260 case Intrinsic::riscv_vsseg2_mask:
3261 case Intrinsic::riscv_vsseg3_mask:
3262 case Intrinsic::riscv_vsseg4_mask:
3263 case Intrinsic::riscv_vsseg5_mask:
3264 case Intrinsic::riscv_vsseg6_mask:
3265 case Intrinsic::riscv_vsseg7_mask:
3266 case Intrinsic::riscv_vsseg8_mask:
3267 HasMask = true;
3268 [[fallthrough]];
3269 case Intrinsic::riscv_vle:
3270 case Intrinsic::riscv_vse:
3271 case Intrinsic::riscv_vlseg2:
3272 case Intrinsic::riscv_vlseg3:
3273 case Intrinsic::riscv_vlseg4:
3274 case Intrinsic::riscv_vlseg5:
3275 case Intrinsic::riscv_vlseg6:
3276 case Intrinsic::riscv_vlseg7:
3277 case Intrinsic::riscv_vlseg8:
3278 case Intrinsic::riscv_vsseg2:
3279 case Intrinsic::riscv_vsseg3:
3280 case Intrinsic::riscv_vsseg4:
3281 case Intrinsic::riscv_vsseg5:
3282 case Intrinsic::riscv_vsseg6:
3283 case Intrinsic::riscv_vsseg7:
3284 case Intrinsic::riscv_vsseg8: {
3285 // Intrinsic interface:
3286 // riscv_vle(merge, ptr, vl)
3287 // riscv_vle_mask(merge, ptr, mask, vl, policy)
3288 // riscv_vse(val, ptr, vl)
3289 // riscv_vse_mask(val, ptr, mask, vl, policy)
3290 // riscv_vlseg#(merge, ptr, vl, sew)
3291 // riscv_vlseg#_mask(merge, ptr, mask, vl, policy, sew)
3292 // riscv_vsseg#(val, ptr, vl, sew)
3293 // riscv_vsseg#_mask(val, ptr, mask, vl, sew)
3294 bool IsWrite = Inst->getType()->isVoidTy();
3295 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3296 // The results of segment loads are TargetExtType.
3297 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3298 unsigned SEW =
3299 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3300 ->getZExtValue();
3301 Ty = TarExtTy->getTypeParameter(0U);
3303 IntegerType::get(C, SEW),
3304 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3305 }
3306 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3307 unsigned VLIndex = RVVIInfo->VLOperand;
3308 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3309 MaybeAlign Alignment =
3310 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3311 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3312 Value *Mask = ConstantInt::getTrue(MaskType);
3313 if (HasMask)
3314 Mask = Inst->getArgOperand(VLIndex - 1);
3315 Value *EVL = Inst->getArgOperand(VLIndex);
3316 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3317 // RVV uses contiguous elements as a segment.
3318 if (SegNum > 1) {
3319 unsigned ElemSize = Ty->getScalarSizeInBits();
3320 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3321 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3322 }
3323 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3324 Alignment, Mask, EVL);
3325 return true;
3326 }
3327 case Intrinsic::riscv_vlse_mask:
3328 case Intrinsic::riscv_vsse_mask:
3329 case Intrinsic::riscv_vlsseg2_mask:
3330 case Intrinsic::riscv_vlsseg3_mask:
3331 case Intrinsic::riscv_vlsseg4_mask:
3332 case Intrinsic::riscv_vlsseg5_mask:
3333 case Intrinsic::riscv_vlsseg6_mask:
3334 case Intrinsic::riscv_vlsseg7_mask:
3335 case Intrinsic::riscv_vlsseg8_mask:
3336 case Intrinsic::riscv_vssseg2_mask:
3337 case Intrinsic::riscv_vssseg3_mask:
3338 case Intrinsic::riscv_vssseg4_mask:
3339 case Intrinsic::riscv_vssseg5_mask:
3340 case Intrinsic::riscv_vssseg6_mask:
3341 case Intrinsic::riscv_vssseg7_mask:
3342 case Intrinsic::riscv_vssseg8_mask:
3343 HasMask = true;
3344 [[fallthrough]];
3345 case Intrinsic::riscv_vlse:
3346 case Intrinsic::riscv_vsse:
3347 case Intrinsic::riscv_vlsseg2:
3348 case Intrinsic::riscv_vlsseg3:
3349 case Intrinsic::riscv_vlsseg4:
3350 case Intrinsic::riscv_vlsseg5:
3351 case Intrinsic::riscv_vlsseg6:
3352 case Intrinsic::riscv_vlsseg7:
3353 case Intrinsic::riscv_vlsseg8:
3354 case Intrinsic::riscv_vssseg2:
3355 case Intrinsic::riscv_vssseg3:
3356 case Intrinsic::riscv_vssseg4:
3357 case Intrinsic::riscv_vssseg5:
3358 case Intrinsic::riscv_vssseg6:
3359 case Intrinsic::riscv_vssseg7:
3360 case Intrinsic::riscv_vssseg8: {
3361 // Intrinsic interface:
3362 // riscv_vlse(merge, ptr, stride, vl)
3363 // riscv_vlse_mask(merge, ptr, stride, mask, vl, policy)
3364 // riscv_vsse(val, ptr, stride, vl)
3365 // riscv_vsse_mask(val, ptr, stride, mask, vl, policy)
3366 // riscv_vlsseg#(merge, ptr, offset, vl, sew)
3367 // riscv_vlsseg#_mask(merge, ptr, offset, mask, vl, policy, sew)
3368 // riscv_vssseg#(val, ptr, offset, vl, sew)
3369 // riscv_vssseg#_mask(val, ptr, offset, mask, vl, sew)
3370 bool IsWrite = Inst->getType()->isVoidTy();
3371 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3372 // The results of segment loads are TargetExtType.
3373 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3374 unsigned SEW =
3375 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3376 ->getZExtValue();
3377 Ty = TarExtTy->getTypeParameter(0U);
3379 IntegerType::get(C, SEW),
3380 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3381 }
3382 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3383 unsigned VLIndex = RVVIInfo->VLOperand;
3384 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3385 MaybeAlign Alignment =
3386 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3387
3388 Value *Stride = Inst->getArgOperand(PtrOperandNo + 1);
3389 // Use the pointer alignment as the element alignment if the stride is a
3390 // multiple of the pointer alignment. Otherwise, the element alignment
3391 // should be the greatest common divisor of pointer alignment and stride.
3392 // For simplicity, just consider unalignment for elements.
3393 unsigned PointerAlign = Alignment.valueOrOne().value();
3394 if (!isa<ConstantInt>(Stride) ||
3395 cast<ConstantInt>(Stride)->getZExtValue() % PointerAlign != 0)
3396 Alignment = Align(1);
3397
3398 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3399 Value *Mask = ConstantInt::getTrue(MaskType);
3400 if (HasMask)
3401 Mask = Inst->getArgOperand(VLIndex - 1);
3402 Value *EVL = Inst->getArgOperand(VLIndex);
3403 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3404 // RVV uses contiguous elements as a segment.
3405 if (SegNum > 1) {
3406 unsigned ElemSize = Ty->getScalarSizeInBits();
3407 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3408 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3409 }
3410 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3411 Alignment, Mask, EVL, Stride);
3412 return true;
3413 }
3414 case Intrinsic::riscv_vloxei_mask:
3415 case Intrinsic::riscv_vluxei_mask:
3416 case Intrinsic::riscv_vsoxei_mask:
3417 case Intrinsic::riscv_vsuxei_mask:
3418 case Intrinsic::riscv_vloxseg2_mask:
3419 case Intrinsic::riscv_vloxseg3_mask:
3420 case Intrinsic::riscv_vloxseg4_mask:
3421 case Intrinsic::riscv_vloxseg5_mask:
3422 case Intrinsic::riscv_vloxseg6_mask:
3423 case Intrinsic::riscv_vloxseg7_mask:
3424 case Intrinsic::riscv_vloxseg8_mask:
3425 case Intrinsic::riscv_vluxseg2_mask:
3426 case Intrinsic::riscv_vluxseg3_mask:
3427 case Intrinsic::riscv_vluxseg4_mask:
3428 case Intrinsic::riscv_vluxseg5_mask:
3429 case Intrinsic::riscv_vluxseg6_mask:
3430 case Intrinsic::riscv_vluxseg7_mask:
3431 case Intrinsic::riscv_vluxseg8_mask:
3432 case Intrinsic::riscv_vsoxseg2_mask:
3433 case Intrinsic::riscv_vsoxseg3_mask:
3434 case Intrinsic::riscv_vsoxseg4_mask:
3435 case Intrinsic::riscv_vsoxseg5_mask:
3436 case Intrinsic::riscv_vsoxseg6_mask:
3437 case Intrinsic::riscv_vsoxseg7_mask:
3438 case Intrinsic::riscv_vsoxseg8_mask:
3439 case Intrinsic::riscv_vsuxseg2_mask:
3440 case Intrinsic::riscv_vsuxseg3_mask:
3441 case Intrinsic::riscv_vsuxseg4_mask:
3442 case Intrinsic::riscv_vsuxseg5_mask:
3443 case Intrinsic::riscv_vsuxseg6_mask:
3444 case Intrinsic::riscv_vsuxseg7_mask:
3445 case Intrinsic::riscv_vsuxseg8_mask:
3446 HasMask = true;
3447 [[fallthrough]];
3448 case Intrinsic::riscv_vloxei:
3449 case Intrinsic::riscv_vluxei:
3450 case Intrinsic::riscv_vsoxei:
3451 case Intrinsic::riscv_vsuxei:
3452 case Intrinsic::riscv_vloxseg2:
3453 case Intrinsic::riscv_vloxseg3:
3454 case Intrinsic::riscv_vloxseg4:
3455 case Intrinsic::riscv_vloxseg5:
3456 case Intrinsic::riscv_vloxseg6:
3457 case Intrinsic::riscv_vloxseg7:
3458 case Intrinsic::riscv_vloxseg8:
3459 case Intrinsic::riscv_vluxseg2:
3460 case Intrinsic::riscv_vluxseg3:
3461 case Intrinsic::riscv_vluxseg4:
3462 case Intrinsic::riscv_vluxseg5:
3463 case Intrinsic::riscv_vluxseg6:
3464 case Intrinsic::riscv_vluxseg7:
3465 case Intrinsic::riscv_vluxseg8:
3466 case Intrinsic::riscv_vsoxseg2:
3467 case Intrinsic::riscv_vsoxseg3:
3468 case Intrinsic::riscv_vsoxseg4:
3469 case Intrinsic::riscv_vsoxseg5:
3470 case Intrinsic::riscv_vsoxseg6:
3471 case Intrinsic::riscv_vsoxseg7:
3472 case Intrinsic::riscv_vsoxseg8:
3473 case Intrinsic::riscv_vsuxseg2:
3474 case Intrinsic::riscv_vsuxseg3:
3475 case Intrinsic::riscv_vsuxseg4:
3476 case Intrinsic::riscv_vsuxseg5:
3477 case Intrinsic::riscv_vsuxseg6:
3478 case Intrinsic::riscv_vsuxseg7:
3479 case Intrinsic::riscv_vsuxseg8: {
3480 // Intrinsic interface (only listed ordered version):
3481 // riscv_vloxei(merge, ptr, index, vl)
3482 // riscv_vloxei_mask(merge, ptr, index, mask, vl, policy)
3483 // riscv_vsoxei(val, ptr, index, vl)
3484 // riscv_vsoxei_mask(val, ptr, index, mask, vl, policy)
3485 // riscv_vloxseg#(merge, ptr, index, vl, sew)
3486 // riscv_vloxseg#_mask(merge, ptr, index, mask, vl, policy, sew)
3487 // riscv_vsoxseg#(val, ptr, index, vl, sew)
3488 // riscv_vsoxseg#_mask(val, ptr, index, mask, vl, sew)
3489 bool IsWrite = Inst->getType()->isVoidTy();
3490 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3491 // The results of segment loads are TargetExtType.
3492 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3493 unsigned SEW =
3494 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3495 ->getZExtValue();
3496 Ty = TarExtTy->getTypeParameter(0U);
3498 IntegerType::get(C, SEW),
3499 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3500 }
3501 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3502 unsigned VLIndex = RVVIInfo->VLOperand;
3503 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3504 Value *Mask;
3505 if (HasMask) {
3506 Mask = Inst->getArgOperand(VLIndex - 1);
3507 } else {
3508 // Mask cannot be nullptr here: vector GEP produces <vscale x N x ptr>,
3509 // and casting that to scalar i64 triggers a vector/scalar mismatch
3510 // assertion in CreatePointerCast. Use an all-true mask so ASan lowers it
3511 // via extractelement instead.
3512 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3513 Mask = ConstantInt::getTrue(MaskType);
3514 }
3515 Value *EVL = Inst->getArgOperand(VLIndex);
3516 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3517 // RVV uses contiguous elements as a segment.
3518 if (SegNum > 1) {
3519 unsigned ElemSize = Ty->getScalarSizeInBits();
3520 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3521 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3522 }
3523 Value *OffsetOp = Inst->getArgOperand(PtrOperandNo + 1);
3524 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3525 Align(1), Mask, EVL,
3526 /* Stride */ nullptr, OffsetOp);
3527 return true;
3528 }
3529 }
3530 return false;
3531}
3532
3534 if (Ty->isVectorTy()) {
3535 // f16 with only zvfhmin and bf16 will be promoted to f32
3536 Type *EltTy = cast<VectorType>(Ty)->getElementType();
3537 if ((EltTy->isHalfTy() && !ST->hasVInstructionsF16()) ||
3538 EltTy->isBFloatTy())
3539 Ty = VectorType::get(Type::getFloatTy(Ty->getContext()),
3540 cast<VectorType>(Ty));
3541
3542 TypeSize Size = DL.getTypeSizeInBits(Ty);
3543 if (Size.isScalable() && ST->hasVInstructions())
3544 return divideCeil(Size.getKnownMinValue(), RISCV::RVVBitsPerBlock);
3545
3546 if (ST->useRVVForFixedLengthVectors())
3547 return divideCeil(Size, ST->getRealMinVLen());
3548 }
3549
3550 return BaseT::getRegUsageForType(Ty);
3551}
3552
3553unsigned RISCVTTIImpl::getMaximumVF(unsigned ElemWidth, unsigned Opcode) const {
3554 if (SLPMaxVF.getNumOccurrences())
3555 return SLPMaxVF;
3556
3557 // Return how many elements can fit in getRegisterBitwidth. This is the
3558 // same routine as used in LoopVectorizer. We should probably be
3559 // accounting for whether we actually have instructions with the right
3560 // lane type, but we don't have enough information to do that without
3561 // some additional plumbing which hasn't been justified yet.
3562 TypeSize RegWidth =
3564 // If no vector registers, or absurd element widths, disable
3565 // vectorization by returning 1.
3566 return std::max<unsigned>(1U, RegWidth.getFixedValue() / ElemWidth);
3567}
3568
3572
3574 return ST->enableUnalignedVectorMem();
3575}
3576
3579 ScalarEvolution *SE) const {
3580 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3581 return TTI::AMK_PostIndexed;
3582
3584}
3585
3587 const TargetTransformInfo::LSRCost &C2) const {
3588 // RISC-V specific here are "instruction number 1st priority".
3589 // If we need to emit adds inside the loop to add up base registers, then
3590 // we need at least one extra temporary register.
3591 unsigned C1NumRegs = C1.NumRegs + (C1.NumBaseAdds != 0);
3592 unsigned C2NumRegs = C2.NumRegs + (C2.NumBaseAdds != 0);
3593 return std::tie(C1.Insns, C1NumRegs, C1.AddRecCost,
3594 C1.NumIVMuls, C1.NumBaseAdds,
3595 C1.ScaleCost, C1.ImmCost, C1.SetupCost) <
3596 std::tie(C2.Insns, C2NumRegs, C2.AddRecCost,
3597 C2.NumIVMuls, C2.NumBaseAdds,
3598 C2.ScaleCost, C2.ImmCost, C2.SetupCost);
3599}
3600
3602 Align Alignment) const {
3603 auto *VTy = dyn_cast<VectorType>(DataTy);
3604 if (!VTy)
3605 return false;
3606
3607 if (!isLegalMaskedLoadStore(DataTy, Alignment))
3608 return false;
3609
3610 // FIXME: If it is an i8 vector and the element count exceeds 256, we should
3611 // scalarize these types with LMUL >= maximum fixed-length LMUL.
3612 if (VTy->getElementType()->isIntegerTy(8)) {
3613 uint64_t MaxEltCount = VTy->getElementCount().getKnownMinValue();
3614 if (VTy->isScalableTy())
3615 MaxEltCount *= ST->getRealMaxVLen() / RISCV::RVVBitsPerBlock;
3616 // We can't yet split any widened indices type.
3617 if (MaxEltCount > 256)
3619 VTy->getWithNewType(Type::getInt16Ty(VTy->getContext())))
3620 .first == 1;
3621 }
3622 return true;
3623}
3624
3626 Align Alignment) const {
3627 return isLegalMaskedLoadStore(DataTy, Alignment);
3628}
3629
3631 ElementCount NumElements) const {
3632 // Optimized zero-stride loads can be treated as broadcasts.
3633 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3634 return false;
3635
3636 return TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, ElementTy));
3637}
3638
3639/// See if \p I should be considered for address type promotion. We check if \p
3640/// I is a sext with right type and used in memory accesses. If it used in a
3641/// "complex" getelementptr, we allow it to be promoted without finding other
3642/// sext instructions that sign extended the same initial value. A getelementptr
3643/// is considered as "complex" if it has more than 2 operands.
3645 const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const {
3646 bool Considerable = false;
3647 AllowPromotionWithoutCommonHeader = false;
3648 if (!isa<SExtInst>(&I))
3649 return false;
3650 Type *ConsideredSExtType =
3651 Type::getInt64Ty(I.getParent()->getParent()->getContext());
3652 if (I.getType() != ConsideredSExtType)
3653 return false;
3654 // See if the sext is the one with the right type and used in at least one
3655 // GetElementPtrInst.
3656 for (const User *U : I.users()) {
3657 if (const GetElementPtrInst *GEPInst = dyn_cast<GetElementPtrInst>(U)) {
3658 Considerable = true;
3659 // A getelementptr is considered as "complex" if it has more than 2
3660 // operands. We will promote a SExt used in such complex GEP as we
3661 // expect some computation to be merged if they are done on 64 bits.
3662 if (GEPInst->getNumOperands() > 2) {
3663 AllowPromotionWithoutCommonHeader = true;
3664 break;
3665 }
3666 }
3667 }
3668 return Considerable;
3669}
3670
3671bool RISCVTTIImpl::canSplatOperand(unsigned Opcode, int Operand) const {
3672 switch (Opcode) {
3673 case Instruction::Add:
3674 case Instruction::Sub:
3675 case Instruction::Mul:
3676 case Instruction::And:
3677 case Instruction::Or:
3678 case Instruction::Xor:
3679 case Instruction::FAdd:
3680 case Instruction::FSub:
3681 case Instruction::FMul:
3682 case Instruction::FDiv:
3683 case Instruction::ICmp:
3684 case Instruction::FCmp:
3685 return true;
3686 case Instruction::Shl:
3687 case Instruction::LShr:
3688 case Instruction::AShr:
3689 case Instruction::UDiv:
3690 case Instruction::SDiv:
3691 case Instruction::URem:
3692 case Instruction::SRem:
3693 case Instruction::Select:
3694 return Operand == 1;
3695 default:
3696 return false;
3697 }
3698}
3699
3701 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3702 return false;
3703
3704 if (canSplatOperand(I->getOpcode(), Operand))
3705 return true;
3706
3707 auto *II = dyn_cast<IntrinsicInst>(I);
3708 if (!II)
3709 return false;
3710
3711 switch (II->getIntrinsicID()) {
3712 case Intrinsic::fma:
3713 case Intrinsic::fmuladd:
3714 return Operand == 0 || Operand == 1;
3715 case Intrinsic::vp_udiv:
3716 case Intrinsic::vp_sdiv:
3717 case Intrinsic::vp_urem:
3718 case Intrinsic::vp_srem:
3719 case Intrinsic::ssub_sat:
3720 case Intrinsic::usub_sat:
3721 return Operand == 1;
3722 // These intrinsics are commutative.
3723 case Intrinsic::smin:
3724 case Intrinsic::umin:
3725 case Intrinsic::smax:
3726 case Intrinsic::umax:
3727 case Intrinsic::sadd_sat:
3728 case Intrinsic::uadd_sat:
3729 return Operand == 0 || Operand == 1;
3730 default:
3731 return false;
3732 }
3733}
3734
3736 ArrayRef<int> Mask, ArrayRef<Value *> Scalars,
3738 GatherUseOps) const {
3739 if (Scalars.empty() || !ST->hasVInstructions() || !ST->sinkSplatOperands() ||
3740 !ShuffleVectorInst::isZeroEltSplatMask(Mask, Mask.size()))
3742
3743 const auto *SplatIt = find_if_not(Scalars, IsaPred<UndefValue>);
3744 if (SplatIt == Scalars.end() || (*SplatIt)->getType()->isIntegerTy(1) ||
3745 isa<VectorType>((*SplatIt)->getType()) ||
3746 isa<ExtractElementInst>(*SplatIt))
3748
3750 if (!GatherUseOps(UserOps) || UserOps.empty())
3752
3753 if (all_of(UserOps,
3754 [this](const TargetTransformInfo::BuildVectorUseOp &UserOp) {
3755 return canSplatOperand(UserOp.Opcode, UserOp.OperandIndex);
3756 }))
3758
3760}
3761
3762/// Check if sinking \p I's operands to I's basic block is profitable, because
3763/// the operands can be folded into a target instruction, e.g.
3764/// splats of scalars can fold into vector instructions.
3767 using namespace llvm::PatternMatch;
3768
3769 if (I->isBitwiseLogicOp()) {
3770 if (!I->getType()->isVectorTy()) {
3771 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3772 for (auto &Op : I->operands()) {
3773 // (and/or/xor X, (not Y)) -> (andn/orn/xnor X, Y)
3774 if (match(Op.get(), m_Not(m_Value()))) {
3775 Ops.push_back(&Op);
3776 return true;
3777 }
3778 }
3779 }
3780 } else if (I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3781 for (auto &Op : I->operands()) {
3782 // (and X, (not Y)) -> (vandn.vv X, Y)
3783 if (match(Op.get(), m_Not(m_Value()))) {
3784 Ops.push_back(&Op);
3785 return true;
3786 }
3787 // (and X, (splat (not Y))) -> (vandn.vx X, Y)
3789 m_ZeroInt()),
3790 m_Value(), m_ZeroMask()))) {
3791 Use &InsertElt = cast<Instruction>(Op)->getOperandUse(0);
3792 Use &Not = cast<Instruction>(InsertElt)->getOperandUse(1);
3793 Ops.push_back(&Not);
3794 Ops.push_back(&InsertElt);
3795 Ops.push_back(&Op);
3796 return true;
3797 }
3798 }
3799 }
3800 }
3801
3802 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3803 return false;
3804
3805 // Don't sink splat operands if the target prefers it. Some targets requires
3806 // S2V transfer buffers and we can run out of them copying the same value
3807 // repeatedly.
3808 // FIXME: It could still be worth doing if it would improve vector register
3809 // pressure and prevent a vector spill.
3810 if (!ST->sinkSplatOperands())
3811 return false;
3812
3813 for (auto OpIdx : enumerate(I->operands())) {
3814 if (!canSplatOperand(I, OpIdx.index()))
3815 continue;
3816
3817 Instruction *Op = dyn_cast<Instruction>(OpIdx.value().get());
3818 // Make sure we are not already sinking this operand
3819 if (!Op || any_of(Ops, [&](Use *U) { return U->get() == Op; }))
3820 continue;
3821
3822 // We are looking for a splat that can be sunk.
3824 m_Value(), m_ZeroMask())))
3825 continue;
3826
3827 // Don't sink i1 splats.
3828 if (cast<VectorType>(Op->getType())->getElementType()->isIntegerTy(1))
3829 continue;
3830
3831 // All uses of the shuffle should be sunk to avoid duplicating it across gpr
3832 // and vector registers
3833 for (Use &U : Op->uses()) {
3834 Instruction *Insn = cast<Instruction>(U.getUser());
3835 if (!canSplatOperand(Insn, U.getOperandNo()))
3836 return false;
3837 }
3838
3839 // Sink any fpexts since they might be used in a widening fp pattern.
3840 Use *InsertEltUse = &Op->getOperandUse(0);
3841 auto *InsertElt = cast<InsertElementInst>(InsertEltUse);
3842 if (isa<FPExtInst>(InsertElt->getOperand(1)))
3843 Ops.push_back(&InsertElt->getOperandUse(1));
3844 Ops.push_back(InsertEltUse);
3845 Ops.push_back(&OpIdx.value());
3846 }
3847 return true;
3848}
3849
3851RISCVTTIImpl::enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
3853
3854 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3855 return Options;
3856
3857 Options.AllowOverlappingLoads = true;
3858 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3859 Options.NumLoadsPerBlock = IsZeroCmp ? Options.MaxNumLoads : 1;
3860 if (ST->is64Bit()) {
3861 Options.LoadSizes = {8, 4, 2, 1};
3862 Options.AllowedTailExpansions = {3, 5, 6};
3863 } else {
3864 Options.LoadSizes = {4, 2, 1};
3865 Options.AllowedTailExpansions = {3};
3866 }
3867
3868 if (IsZeroCmp && ST->hasVInstructions()) {
3869 unsigned VLenB = ST->getRealMinVLen() / 8;
3870 // The minimum size should be `XLen / 8 + 1`, and the maxinum size should be
3871 // `VLenB * MaxLMUL` so that it fits in a single register group.
3872 unsigned MinSize = ST->getXLen() / 8 + 1;
3873 unsigned MaxSize = VLenB * 8;
3874 for (unsigned Size = MinSize; Size <= MaxSize; Size++)
3875 Options.LoadSizes.insert(Options.LoadSizes.begin(), Size);
3876 }
3877 return Options;
3878}
3879
3881 const Instruction *I) const {
3883 // For the binary operators (e.g. or) we need to be more careful than
3884 // selects, here we only transform them if they are already at a natural
3885 // break point in the code - the end of a block with an unconditional
3886 // terminator.
3887 if (I->getOpcode() == Instruction::Or &&
3888 isa<UncondBrInst>(I->getNextNode()))
3889 return true;
3890
3891 if (I->getOpcode() == Instruction::Add ||
3892 I->getOpcode() == Instruction::Sub)
3893 return true;
3894 }
3896}
3897
3899 const Function *Caller, const Attribute &Attr) const {
3900 // "interrupt" controls the prolog/epilog of interrupt handlers (and includes
3901 // restrictions on their signatures). We can outline from the bodies of these
3902 // handlers, but when we do we need to make sure we don't mark the outlined
3903 // function as an interrupt handler too.
3904 if (Attr.isStringAttribute() && Attr.getKindAsString() == "interrupt")
3905 return false;
3906
3908}
3909
3910std::optional<Instruction *>
3912 // Attach a range return attribute describing the result of vsetvli/vsetvlimax
3913 // so generic value analyses can reason about it. The verifier guarantees an
3914 // XLen result and constant VSEW/VLMUL encoding a valid vtype, so no defensive
3915 // validation is needed here.
3916 if (is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3917 II.getIntrinsicID())) {
3918 // These intrinsics require the V extension; without it the VLEN queries
3919 // below would assert. Such IR would fail isel anyway, so just bail out.
3920 if (!ST->hasVInstructions())
3921 return {};
3922
3923 bool HasAVL = II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3924 unsigned Offset = HasAVL ? 1 : 0;
3925 unsigned BitWidth = II.getType()->getIntegerBitWidth();
3926 ConstantRange VLenRange(APInt(BitWidth, ST->getRealMinVLen()),
3927 APInt(BitWidth, ST->getRealMaxVLen()) + 1);
3928
3929 uint64_t VSEW = cast<ConstantInt>(II.getArgOperand(Offset))->getZExtValue();
3930 auto VLMUL = static_cast<RISCVVType::VLMUL>(
3931 cast<ConstantInt>(II.getArgOperand(Offset + 1))->getZExtValue());
3932 unsigned SEW = RISCVVType::decodeVSEW(VSEW);
3933 unsigned Ratio = RISCVVType::getSEWLMULRatio(SEW, VLMUL);
3934
3935 // VLMAX = VLEN / (SEW / LMUL), clamped to >= 1 for any usable vtype.
3936 ConstantRange VLMAXRange =
3937 VLenRange.udiv(ConstantRange(APInt(BitWidth, Ratio)))
3939
3940 // vsetvlimax returns exactly VLMAX; vsetvli returns vl with
3941 // 0 <= vl <= min(AVL, VLMAX). vl == AVL only when AVL <= the smallest
3942 // possible VLMAX; otherwise vl can shrink below VLMAX (to 0 at runtime), so
3943 // only the VLMAX upper bound is sound.
3944 ConstantRange VLRange = VLMAXRange;
3945 if (HasAVL) {
3946 // vl ≤ VLMAX
3947 VLRange =
3949
3950 Value *AVL = II.getArgOperand(0);
3952 AVL, /*ForSigned=*/false,
3954
3955 // vl = AVL if AVL ≤ VLMAX
3956 if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
3957 return IC.replaceInstUsesWith(II, AVL);
3958
3959 // vl ≤ AVL
3960 VLRange = VLRange.umin(AVLRange.getUnsignedMax());
3961
3962 // vl > 0 if AVL > 0
3964 VLRange = VLRange.umax(APInt(BitWidth, 1));
3965
3966 // vl = VLMAX if AVL ≥ (2 * VLMAX)
3967 ConstantRange TwoVLMAX = VLMAXRange.multiply(APInt(BitWidth, 2));
3968 if (AVLRange.icmp(CmpInst::ICMP_UGE, TwoVLMAX))
3969 VLRange = VLRange.intersectWith(VLMAXRange);
3970
3971 // ceil(AVL / 2) ≤ vl ≤ VLMAX if AVL < (2 * VLMAX)
3972 if (AVLRange.icmp(CmpInst::ICMP_ULT, TwoVLMAX))
3973 VLRange = VLRange.umax(APIntOps::RoundingUDiv(AVLRange.getUnsignedMin(),
3974 APInt(BitWidth, 2),
3976 }
3977
3978 ConstantRange OldRange =
3979 II.getRange().value_or(ConstantRange::getFull(BitWidth));
3980 ConstantRange NewRange = VLRange.intersectWith(OldRange);
3981 if (NewRange != OldRange) {
3982 II.addRangeRetAttr(NewRange);
3983 return &II;
3984 }
3985 return {};
3986 }
3987
3988 // If all operands of a vmv.v.x are constant, fold a bitcast(vmv.v.x) to scale
3989 // the vmv.v.x, enabling removal of the bitcast. The transform helps avoid
3990 // creating redundant masks.
3991 const DataLayout &DL = IC.getDataLayout();
3992 if (II.user_empty())
3993 return {};
3994 auto *TargetVecTy = dyn_cast<ScalableVectorType>(II.user_back()->getType());
3995 if (!TargetVecTy)
3996 return {};
3997 const APInt *Scalar;
3998 uint64_t VL;
4000 m_Poison(), m_APInt(Scalar), m_ConstantInt(VL))) ||
4001 !all_of(II.users(), [TargetVecTy](User *U) {
4002 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
4003 }))
4004 return {};
4005 auto *SourceVecTy = cast<ScalableVectorType>(II.getType());
4006 unsigned TargetEltBW = DL.getTypeSizeInBits(TargetVecTy->getElementType());
4007 unsigned SourceEltBW = DL.getTypeSizeInBits(SourceVecTy->getElementType());
4008 if (TargetEltBW % SourceEltBW)
4009 return {};
4010 unsigned TargetScale = TargetEltBW / SourceEltBW;
4011 if (VL % TargetScale || TargetScale == 1)
4012 return {};
4013 Type *VLTy = II.getOperand(2)->getType();
4014 ElementCount SourceEC = SourceVecTy->getElementCount();
4015 unsigned NewEltBW = SourceEltBW * TargetScale;
4016 if (!SourceEC.isKnownMultipleOf(TargetScale) ||
4017 !DL.fitsInLegalInteger(NewEltBW))
4018 return {};
4019 auto *NewEltTy = IntegerType::get(II.getContext(), NewEltBW);
4020 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, NewEltTy)))
4021 return {};
4022 ElementCount NewEC = SourceEC.divideCoefficientBy(TargetScale);
4023 Type *RetTy = VectorType::get(NewEltTy, NewEC);
4024 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
4025 "Lossless bitcast between types expected");
4026 APInt NewScalar = APInt::getSplat(NewEltBW, *Scalar);
4027 return IC.replaceInstUsesWith(
4028 II,
4031 RetTy, Intrinsic::riscv_vmv_v_x,
4032 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
4033 ConstantInt::get(VLTy, VL / TargetScale)}),
4034 SourceVecTy));
4035}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > EnableOrLikeSelectOpt("enable-aarch64-or-like-select", cl::init(true), cl::Hidden)
unsigned Imm
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
Hexagon Common GEP
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static LVOptions Options
Definition LVOptions.cpp:25
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
uint64_t IntrinsicInst * II
static InstructionCost costShuffleViaVRegSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, std::optional< unsigned > VLen, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Try to perform better estimation of the permutation.
static InstructionCost costShuffleViaSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Attempt to approximate the cost of a shuffle which will require splitting during legalization.
static bool isRepeatedConcatMask(ArrayRef< int > Mask, int &SubVectorSize)
static unsigned isM1OrSmaller(MVT VT)
static cl::opt< bool > EnableOrLikeSelectOpt("enable-riscv-or-like-select", cl::init(true), cl::Hidden)
static cl::opt< unsigned > SLPMaxVF("riscv-v-slp-max-vf", cl::desc("Overrides result used for getMaximumVF query which is used " "exclusively by SLP vectorizer."), cl::Hidden)
static cl::opt< unsigned > RVVRegisterWidthLMUL("riscv-v-register-bit-width-lmul", cl::desc("The LMUL to use for getRegisterBitWidth queries. Affects LMUL used " "by autovectorized code. Fractional LMULs are not supported."), cl::init(2), cl::Hidden)
static cl::opt< unsigned > RVVMinTripCount("riscv-v-min-trip-count", cl::desc("Set the lower bound of a trip count to decide on " "vectorization while tail-folding."), cl::init(5), cl::Hidden)
static InstructionCost getIntImmCostImpl(const DataLayout &DL, const RISCVSubtarget *ST, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, bool FreeZeroes)
static VectorType * getVRGatherIndexType(MVT DataVT, const RISCVSubtarget &ST, LLVMContext &C)
static const CostTblEntry VectorIntrinsicCostTable[]
static bool canUseShiftPair(Instruction *Inst, const APInt &Imm)
static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm)
This file defines a TargetTransformInfoImplBase conforming object specific to the RISC-V target machi...
SI Fold Operands
This file contains some templates that are useful if you are working with the STL at all.
#define LLVM_DEBUG(...)
Definition Debug.h:119
This file describes how to lower LLVM code to machine code.
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
iterator end() const
Definition ArrayRef.h:130
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
Definition InstrTypes.h:757
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
Definition InstrTypes.h:742
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static bool isFPPredicate(Predicate P)
Definition InstrTypes.h:833
static bool isIntPredicate(Predicate P)
Definition InstrTypes.h:839
This is the shared class of boolean and integer constants.
Definition Constants.h:87
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool noNaNs() const
Definition FMF.h:65
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
static FixedVectorType * getHalfElementsVectorType(FixedVectorType *VTy)
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2235
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
user_iterator user_begin()
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
const SmallVectorImpl< Type * > & getArgTypes() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Machine Value Type.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector value type where the runtime length is machine dependent.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
ElementCount getVectorElementCount() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getHalfNumVectorElementsVT() const
Return a VT for a vector type with the same element type but half the number of elements.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
The optimization diagnostic interface.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
TargetTransformInfo::VectorInstrContext getBuildVectorContextHint(ArrayRef< int > Mask, ArrayRef< Value * > Scalars, function_ref< bool(SmallVectorImpl< TargetTransformInfo::BuildVectorUseOp > &)> GatherUseOps) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CtxI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
Definition Type.cpp:865
The main scalar evolution driver.
static LLVM_ABI bool isZeroEltSplatMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses all elements with the same value as the first element of exa...
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
Definition DenseSet.h:293
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
virtual const DataLayout & getDataLayout() const
virtual bool shouldTreatInstructionLikeSelect(const Instruction *I) const
virtual TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const
virtual bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const
virtual bool isLoweredToCall(const Function *F) const
InstructionCost getInstructionCost(const User *U, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind) const override
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
PopcntSupportKind
Flags indicating the kind of support for population count.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Expensive
The cost of a 'div' instruction on x86.
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
AddressingModeKind
Which addressing mode Loop Strength Reduction will try to generate.
@ AMK_PostIndexed
Prefer post-indexed addressing mode.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
Definition Type.cpp:298
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
Definition Type.cpp:61
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:276
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Definition Value.cpp:1002
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
Definition TypeSize.h:180
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:230
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isKnownEven() const
A return value of true indicates we know at compile time that the number of elements (vscale * Min) i...
Definition TypeSize.h:176
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
An efficient, type-erasing, non-owning reference to a callable.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
Definition APInt.cpp:2801
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
Definition ISDOpcodes.h:26
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
Definition CostTable.h:36
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
InstructionCost Cost
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ None
The instruction is not folded.
@ BinaryOp
One of the operands is a binary op.
@ SplatOpFolded
All of the value's users support splatting the value.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
Definition STLExtras.h:1834
bool isPairEven(const std::array< std::pair< int, int >, 2 > &SrcInfo, ArrayRef< int > Mask, unsigned &Factor)
Given a shuffle which can be represented as a pair of two slides, see if it is a pair-even idiom.
bool isPairOdd(const std::array< std::pair< int, int >, 2 > &SrcInfo, ArrayRef< int > Mask, unsigned &Factor)
Given a shuffle which can be represented as a pair of two slides, see if it is a pair-odd idiom.
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
auto find_if_not(R &&Range, UnaryPredicate P)
Definition STLExtras.h:1793
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
Definition STLExtras.h:1986
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
TargetTransformInfo TTI
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
Definition STLExtras.h:1901
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
Definition CostTable.h:31
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2162
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const
Stores information about the uses of a build vector.
unsigned Insns
TODO: Some of these could be merged.
Returns options for expansion of memcmp. IsZeroCmp is.
Describe known properties for a set of pointers.
Parameters that control the generic loop unrolling transformation.
bool UpperBound
Allow using trip count upper bound to unroll loops.
bool Force
Apply loop unroll on any kind of loop (mainly to loops that fail runtime unrolling).
unsigned PartialOptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size, like OptSizeThreshold,...
bool UnrollAndJam
Allow unroll and jam. Used to enable unroll and jam for the target.
bool UnrollRemainder
Allow unrolling of all the iterations of the runtime loop remainder.
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
unsigned OptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size (set to UINT_MAX to disable).