LLVM 24.0.0git
RISCVTargetTransformInfo.cpp
Go to the documentation of this file.
1//===-- RISCVTargetTransformInfo.cpp - RISC-V specific TTI ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
11#include "llvm/ADT/STLExtras.h"
18#include "llvm/IR/IntrinsicsRISCV.h"
21#include <cmath>
22#include <optional>
23using namespace llvm;
24using namespace llvm::PatternMatch;
25
26#define DEBUG_TYPE "riscvtti"
27
29 "riscv-v-register-bit-width-lmul",
31 "The LMUL to use for getRegisterBitWidth queries. Affects LMUL used "
32 "by autovectorized code. Fractional LMULs are not supported."),
34
36 "riscv-v-slp-max-vf",
38 "Overrides result used for getMaximumVF query which is used "
39 "exclusively by SLP vectorizer."),
41
43 RVVMinTripCount("riscv-v-min-trip-count",
44 cl::desc("Set the lower bound of a trip count to decide on "
45 "vectorization while tail-folding."),
47
48static cl::opt<bool> EnableOrLikeSelectOpt("enable-riscv-or-like-select",
49 cl::init(true), cl::Hidden);
50
52RISCVTTIImpl::getRISCVInstructionCost(ArrayRef<unsigned> OpCodes, MVT VT,
54 // Check if the type is valid for all CostKind
55 if (!VT.isVector())
57 size_t NumInstr = OpCodes.size();
59 return NumInstr;
60 InstructionCost LMULCost = TLI->getLMULCost(VT);
62 return LMULCost * NumInstr;
63 InstructionCost Cost = 0;
64 for (auto Op : OpCodes) {
65 switch (Op) {
66 case RISCV::VRGATHER_VI:
67 Cost += TLI->getVRGatherVICost(VT);
68 break;
69 case RISCV::VRGATHER_VV:
70 Cost += TLI->getVRGatherVVCost(VT);
71 break;
72 case RISCV::VSLIDEUP_VI:
73 case RISCV::VSLIDEDOWN_VI:
74 Cost += TLI->getVSlideVICost(VT);
75 break;
76 case RISCV::VSLIDEUP_VX:
77 case RISCV::VSLIDEDOWN_VX:
78 Cost += TLI->getVSlideVXCost(VT);
79 break;
80 case RISCV::VREDMAX_VS:
81 case RISCV::VREDMIN_VS:
82 case RISCV::VREDMAXU_VS:
83 case RISCV::VREDMINU_VS:
84 case RISCV::VREDSUM_VS:
85 case RISCV::VREDAND_VS:
86 case RISCV::VREDOR_VS:
87 case RISCV::VREDXOR_VS:
88 case RISCV::VFREDMAX_VS:
89 case RISCV::VFREDMIN_VS:
90 case RISCV::VFREDUSUM_VS: {
91 unsigned VL = VT.getVectorMinNumElements();
92 if (!VT.isFixedLengthVector())
93 VL *= *getVScaleForTuning();
94 Cost += Log2_32_Ceil(VL);
95 break;
96 }
97 case RISCV::VFREDOSUM_VS: {
98 unsigned VL = VT.getVectorMinNumElements();
99 if (!VT.isFixedLengthVector())
100 VL *= *getVScaleForTuning();
101 Cost += VL;
102 break;
103 }
104 case RISCV::VMV_X_S:
105 case RISCV::VFMV_F_S:
106 // Domain crossings from vector -> scalar are usually more expensive.
107 Cost += 2;
108 break;
109 case RISCV::VMV_S_X:
110 case RISCV::VFMV_S_F:
111 case RISCV::VMOR_MM:
112 case RISCV::VMXOR_MM:
113 case RISCV::VMAND_MM:
114 case RISCV::VMANDN_MM:
115 case RISCV::VMNAND_MM:
116 case RISCV::VCPOP_M:
117 case RISCV::VFIRST_M:
118 Cost += 1;
119 break;
120 case RISCV::VDIV_VV:
121 case RISCV::VREM_VV:
122 Cost += LMULCost * TTI::TCC_Expensive;
123 break;
124 default:
125 Cost += LMULCost;
126 }
127 }
128 return Cost;
129}
130
132 const RISCVSubtarget *ST,
133 const APInt &Imm, Type *Ty,
135 bool FreeZeroes) {
136 assert(Ty->isIntegerTy() &&
137 "getIntImmCost can only estimate cost of materialising integers");
138
139 // We have a Zero register, so 0 is always free.
140 if (Imm == 0)
141 return TTI::TCC_Free;
142
143 // Otherwise, we check how many instructions it will take to materialise.
144 return RISCVMatInt::getIntMatCost(Imm, DL.getTypeSizeInBits(Ty), *ST,
145 /*CompressionCost=*/false, FreeZeroes);
146}
147
151 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind, false);
152}
153
154// Look for patterns of shift followed by AND that can be turned into a pair of
155// shifts. We won't need to materialize an immediate for the AND so these can
156// be considered free.
157static bool canUseShiftPair(Instruction *Inst, const APInt &Imm) {
158 uint64_t Mask = Imm.getZExtValue();
159 auto *BO = dyn_cast<BinaryOperator>(Inst->getOperand(0));
160 if (!BO || !BO->hasOneUse())
161 return false;
162
163 if (BO->getOpcode() != Instruction::Shl)
164 return false;
165
166 if (!isa<ConstantInt>(BO->getOperand(1)))
167 return false;
168
169 unsigned ShAmt = cast<ConstantInt>(BO->getOperand(1))->getZExtValue();
170 // (and (shl x, c2), c1) will be matched to (srli (slli x, c2+c3), c3) if c1
171 // is a mask shifted by c2 bits with c3 leading zeros.
172 if (isShiftedMask_64(Mask)) {
173 unsigned Trailing = llvm::countr_zero(Mask);
174 if (ShAmt == Trailing)
175 return true;
176 }
177
178 return false;
179}
180
181// If this is i64 AND is part of (X & -(1 << C1) & 0xffffffff) == C2 << C1),
182// DAGCombiner can convert this to (sraiw X, C1) == sext(C2) for RV64. On RV32,
183// the type will be split so only the lower 32 bits need to be compared using
184// (srai/srli X, C) == C2.
185static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm) {
186 if (!Inst->hasOneUse())
187 return false;
188
189 // Look for equality comparison.
190 auto *Cmp = dyn_cast<ICmpInst>(*Inst->user_begin());
191 if (!Cmp || !Cmp->isEquality())
192 return false;
193
194 // Right hand side of comparison should be a constant.
195 auto *C = dyn_cast<ConstantInt>(Cmp->getOperand(1));
196 if (!C)
197 return false;
198
199 uint64_t Mask = Imm.getZExtValue();
200
201 // Mask should be of the form -(1 << C) in the lower 32 bits.
202 if (!isUInt<32>(Mask) || !isPowerOf2_32(-uint32_t(Mask)))
203 return false;
204
205 // Comparison constant should be a subset of Mask.
206 uint64_t CmpC = C->getZExtValue();
207 if ((CmpC & Mask) != CmpC)
208 return false;
209
210 // We'll need to sign extend the comparison constant and shift it right. Make
211 // sure the new constant can use addi/xori+seqz/snez.
212 unsigned ShiftBits = llvm::countr_zero(Mask);
213 int64_t NewCmpC = SignExtend64<32>(CmpC) >> ShiftBits;
214 return NewCmpC >= -2048 && NewCmpC <= 2048;
215}
216
218 const APInt &Imm, Type *Ty,
220 Instruction *Inst) const {
221 assert(Ty->isIntegerTy() &&
222 "getIntImmCost can only estimate cost of materialising integers");
223
224 // We have a Zero register, so 0 is always free.
225 if (Imm == 0)
226 return TTI::TCC_Free;
227
228 // Some instructions in RISC-V can take a 12-bit immediate. Some of these are
229 // commutative, in others the immediate comes from a specific argument index.
230 bool Takes12BitImm = false;
231 unsigned ImmArgIdx = ~0U;
232
233 switch (Opcode) {
234 case Instruction::GetElementPtr:
235 // Never hoist any arguments to a GetElementPtr. CodeGenPrepare will
236 // split up large offsets in GEP into better parts than ConstantHoisting
237 // can.
238 return TTI::TCC_Free;
239 case Instruction::Store: {
240 // Use the materialization cost regardless of if it's the address or the
241 // value that is constant, except for if the store is misaligned and
242 // misaligned accesses are not legal (experience shows constant hoisting
243 // can sometimes be harmful in such cases).
244 if (Idx == 1 || !Inst)
245 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
246 /*FreeZeroes=*/true);
247
248 StoreInst *ST = cast<StoreInst>(Inst);
249 if (!getTLI()->allowsMemoryAccessForAlignment(
250 Ty->getContext(), DL, getTLI()->getValueType(DL, Ty),
251 ST->getPointerAddressSpace(), ST->getAlign()))
252 return TTI::TCC_Free;
253
254 return getIntImmCostImpl(getDataLayout(), getST(), Imm, Ty, CostKind,
255 /*FreeZeroes=*/true);
256 }
257 case Instruction::Load:
258 // If the address is a constant, use the materialization cost.
259 return getIntImmCost(Imm, Ty, CostKind);
260 case Instruction::And:
261 // zext.h
262 if (Imm == UINT64_C(0xffff) && ST->hasStdExtZbb())
263 return TTI::TCC_Free;
264 // zext.w
265 if (Imm == UINT64_C(0xffffffff) && (!ST->is64Bit() || ST->hasStdExtZba()))
266 return TTI::TCC_Free;
267 // bclri
268 if (ST->hasStdExtZbs() && (~Imm).isPowerOf2())
269 return TTI::TCC_Free;
270 if (Inst && Idx == 1 && Imm.getBitWidth() <= ST->getXLen() &&
271 canUseShiftPair(Inst, Imm))
272 return TTI::TCC_Free;
273 if (Inst && Idx == 1 && Imm.getBitWidth() == 64 &&
274 canUseShiftCmp(Inst, Imm))
275 return TTI::TCC_Free;
276 Takes12BitImm = true;
277 break;
278 case Instruction::Add:
279 Takes12BitImm = true;
280 break;
281 case Instruction::Or:
282 case Instruction::Xor:
283 // bseti/binvi
284 if (ST->hasStdExtZbs() && Imm.isPowerOf2())
285 return TTI::TCC_Free;
286 Takes12BitImm = true;
287 break;
288 case Instruction::Mul:
289 // Power of 2 is a shift. Negated power of 2 is a shift and a negate.
290 if (Imm.isPowerOf2() || Imm.isNegatedPowerOf2())
291 return TTI::TCC_Free;
292 // One more or less than a power of 2 can use SLLI+ADD/SUB.
293 if ((Imm + 1).isPowerOf2() || (Imm - 1).isPowerOf2())
294 return TTI::TCC_Free;
295 // FIXME: There is no MULI instruction.
296 Takes12BitImm = true;
297 break;
298 case Instruction::Sub:
299 case Instruction::Shl:
300 case Instruction::LShr:
301 case Instruction::AShr:
302 Takes12BitImm = true;
303 ImmArgIdx = 1;
304 break;
305 default:
306 break;
307 }
308
309 if (Takes12BitImm) {
310 // Check immediate is the correct argument...
311 if (Instruction::isCommutative(Opcode) || Idx == ImmArgIdx) {
312 // ... and fits into the 12-bit immediate.
313 if (Imm.getSignificantBits() <= 64 &&
314 getTLI()->isLegalAddImmediate(Imm.getSExtValue())) {
315 return TTI::TCC_Free;
316 }
317 }
318
319 // Otherwise, use the full materialisation cost.
320 return getIntImmCost(Imm, Ty, CostKind);
321 }
322
323 // By default, prevent hoisting.
324 return TTI::TCC_Free;
325}
326
329 const APInt &Imm, Type *Ty,
331 // Prevent hoisting in unknown cases.
332 return TTI::TCC_Free;
333}
334
336 return ST->hasVInstructions();
337}
338
340RISCVTTIImpl::getPopcntSupport(unsigned TyWidth) const {
341 assert(isPowerOf2_32(TyWidth) && "Ty width must be power of 2");
342 return ST->hasCPOPLike() ? TTI::PSK_FastHardware : TTI::PSK_Software;
343}
344
346 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
348 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
349 TTI::TargetCostKind CostKind, std::optional<FastMathFlags> FMF) const {
350 if (Opcode == Instruction::FAdd)
352
353 // zve32x is broken for partial_reduce_umla, but let's make sure we
354 // don't generate them.
355 // vdot4a* reduces four i8 products into an i32 result; an i64 accumulator is
356 // additionally supported by widening the i32 partial sums to i64 (see
357 // lowerPARTIAL_REDUCE_MLA). VF is the number of i8 input elements, so the
358 // reduction factor is AccumBits / 8 (4 for i32, 8 for i64).
359 if (!ST->hasStdExtZvdot4a8i() || ST->getELen() < 64 ||
360 Opcode != Instruction::Add || !BinOp || *BinOp != Instruction::Mul ||
361 InputTypeA != InputTypeB || !InputTypeA->isIntegerTy(8) ||
362 (!AccumType->isIntegerTy(32) && !AccumType->isIntegerTy(64)))
364
365 unsigned Ratio = AccumType->getScalarSizeInBits() / 8;
366 if (!VF.isKnownMultipleOf(Ratio))
368
369 // Cost of the vdot4a* itself, which operates on the i32 intermediate type
370 // holding VF/4 elements.
371 Type *DotTp = VectorType::get(Type::getInt32Ty(AccumType->getContext()),
372 VF.divideCoefficientBy(4));
373 std::pair<InstructionCost, MVT> DotLT = getTypeLegalizationCost(DotTp);
374 // Note: Asuming all vdot4a* variants are equal cost
376 DotLT.first *
377 getRISCVInstructionCost(RISCV::VDOT4A_VV, DotLT.second, CostKind);
378
379 // Account for reducing the i32 partial sums down to the i64 accumulator's
380 // element count and accumulating into it (see lowerPARTIAL_REDUCE_MLA), which
381 // has two shapes depending on the accumulator's LMUL.
382 if (AccumType->isIntegerTy(64)) {
383 LLVMContext &Ctx = AccumType->getContext();
384 Type *I32Ty = Type::getInt32Ty(Ctx);
385 ElementCount AccVF = VF.divideCoefficientBy(Ratio);
386 std::pair<InstructionCost, MVT> AccLT =
387 getTypeLegalizationCost(VectorType::get(AccumType, AccVF));
388
389 // When the i32 subvectors of a single-vector scalable accumulator are a
390 // fractional LMUL, extracting the high subvector would need a vslidedown,
391 // so instead the i32 dot result is widened to i64 first (vsext.vf2 /
392 // vzext.vf2) and then reduced and accumulated with register-aligned i64
393 // vadd.vv.
394 bool WidenFirst = false;
395 if (VF.isScalable() && AccLT.second.isScalableVector()) {
396 MVT NarrowMVT = AccLT.second.changeVectorElementType(MVT::i32);
397 WidenFirst =
399 .second;
400 }
401
402 if (WidenFirst) {
403 // The widened i64 dot result has VF/4 elements, i.e. twice the
404 // accumulator's element count, so the reduction plus the accumulate are
405 // two i64 vadd.vv.
406 std::pair<InstructionCost, MVT> WideLT = getTypeLegalizationCost(
407 VectorType::get(AccumType, VF.divideCoefficientBy(4)));
408 Cost +=
409 WideLT.first * getRISCVInstructionCost(RISCV::VSEXT_VF2,
410 WideLT.second, CostKind) +
411 2 * AccLT.first *
412 getRISCVInstructionCost(RISCV::VADD_VV, AccLT.second, CostKind);
413 } else {
414 // Otherwise the scale-4 i32 sums are halved with a single i32 vadd.vv,
415 // then widened and added into the i64 result with a vwadd.wv.
416 std::pair<InstructionCost, MVT> RedLT =
418 Cost += RedLT.first * getRISCVInstructionCost(RISCV::VADD_VV,
419 RedLT.second, CostKind) +
420 AccLT.first * getRISCVInstructionCost(RISCV::VWADD_WV,
421 AccLT.second, CostKind);
422 // Fixed-length vectors extract the high i32 subvector with a vslidedown.
423 if (VF.isFixed())
424 Cost += DotLT.first * getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI,
425 DotLT.second, CostKind);
426 }
427 }
428
429 return Cost;
430}
431
433 // Currently, the ExpandReductions pass can't expand scalable-vector
434 // reductions, but we still request expansion as RVV doesn't support certain
435 // reductions and the SelectionDAG can't legalize them either.
436 switch (II->getIntrinsicID()) {
437 default:
438 return false;
439 // These reductions have no equivalent in RVV
440 case Intrinsic::vector_reduce_mul:
441 case Intrinsic::vector_reduce_fmul:
442 return true;
443 }
444}
445
446std::optional<unsigned> RISCVTTIImpl::getVScaleForTuning() const {
447 if (ST->hasVInstructions())
448 if (unsigned MinVLen = ST->getRealMinVLen();
449 MinVLen >= RISCV::RVVBitsPerBlock)
450 return MinVLen / RISCV::RVVBitsPerBlock;
452}
453
456 unsigned LMUL =
457 llvm::bit_floor(std::clamp<unsigned>(RVVRegisterWidthLMUL, 1, 8));
458 switch (K) {
460 return TypeSize::getFixed(ST->getXLen());
462 return TypeSize::getFixed(
463 ST->useRVVForFixedLengthVectors() ? LMUL * ST->getRealMinVLen() : 0);
466 (ST->hasVInstructions() &&
467 ST->getRealMinVLen() >= RISCV::RVVBitsPerBlock)
469 : 0);
470 }
471
472 llvm_unreachable("Unsupported register kind");
473}
474
475InstructionCost RISCVTTIImpl::getStaticDataAddrGenerationCost(
476 const TTI::TargetCostKind CostKind) const {
477 switch (CostKind) {
480 // Always 2 instructions
481 return 2;
482 case TTI::TCK_Latency:
484 // Depending on the memory model the address generation will
485 // require AUIPC + ADDI (medany) or LUI + ADDI (medlow). Don't
486 // have a way of getting this information here, so conservatively
487 // require both.
488 // In practice, these are generally implemented together.
489 return (ST->hasAUIPCADDIFusion() && ST->hasLUIADDIFusion()) ? 1 : 2;
490 }
491 llvm_unreachable("Unsupported cost kind");
492}
493
495RISCVTTIImpl::getConstantPoolLoadCost(Type *Ty,
497 // Add a cost of address generation + the cost of the load. The address
498 // is expected to be a PC relative offset to a constant pool entry
499 // using auipc/addi.
500 return getStaticDataAddrGenerationCost(CostKind) +
501 getMemoryOpCost(Instruction::Load, Ty, DL.getABITypeAlign(Ty),
502 /*AddressSpace=*/0, CostKind);
503}
504
505static bool isRepeatedConcatMask(ArrayRef<int> Mask, int &SubVectorSize) {
506 unsigned Size = Mask.size();
507 if (!isPowerOf2_32(Size))
508 return false;
509 for (unsigned I = 0; I != Size; ++I) {
510 if (static_cast<unsigned>(Mask[I]) == I)
511 continue;
512 if (Mask[I] != 0)
513 return false;
514 if (Size % I != 0)
515 return false;
516 for (unsigned J = I + 1; J != Size; ++J)
517 // Check the pattern is repeated.
518 if (static_cast<unsigned>(Mask[J]) != J % I)
519 return false;
520 SubVectorSize = I;
521 return true;
522 }
523 // That means Mask is <0, 1, 2, 3>. This is not a concatenation.
524 return false;
525}
526
528 LLVMContext &C) {
529 assert((DataVT.getScalarSizeInBits() != 8 ||
530 DataVT.getVectorNumElements() <= 256) && "unhandled case in lowering");
531 MVT IndexVT = DataVT.changeTypeToInteger();
532 if (IndexVT.getScalarType().bitsGT(ST.getXLenVT()))
533 IndexVT = IndexVT.changeVectorElementType(MVT::i16);
534 return cast<VectorType>(EVT(IndexVT).getTypeForEVT(C));
535}
536
537/// Attempt to approximate the cost of a shuffle which will require splitting
538/// during legalization. Note that processShuffleMasks is not an exact proxy
539/// for the algorithm used in LegalizeVectorTypes, but hopefully it's a
540/// reasonably close upperbound.
542 MVT LegalVT, VectorType *Tp,
543 ArrayRef<int> Mask,
545 assert(LegalVT.isFixedLengthVector() && !Mask.empty() &&
546 "Expected fixed vector type and non-empty mask");
547 unsigned LegalNumElts = LegalVT.getVectorNumElements();
548 // Number of destination vectors after legalization:
549 unsigned NumOfDests = divideCeil(Mask.size(), LegalNumElts);
550 // We are going to permute multiple sources and the result will be in
551 // multiple destinations. Providing an accurate cost only for splits where
552 // the element type remains the same.
553 if (NumOfDests <= 1 ||
555 Tp->getElementType()->getPrimitiveSizeInBits() ||
556 LegalNumElts >= Tp->getElementCount().getFixedValue())
558
559 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
560 unsigned LegalVTSize = LegalVT.getStoreSize();
561 // Number of source vectors after legalization:
562 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
563
564 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(), LegalNumElts);
565
566 unsigned NormalizedVF = LegalNumElts * std::max(NumOfSrcs, NumOfDests);
567 unsigned NumOfSrcRegs = NormalizedVF / LegalNumElts;
568 unsigned NumOfDestRegs = NormalizedVF / LegalNumElts;
569 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
570 assert(NormalizedVF >= Mask.size() &&
571 "Normalized mask expected to be not shorter than original mask.");
572 copy(Mask, NormalizedMask.begin());
573 InstructionCost Cost = 0;
574 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
576 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
577 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
578 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
579 return;
580 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
581 .second)
582 return;
583 Cost += TTI.getShuffleCost(
585 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
586 SingleOpTy, CostKind, RegMask, 0, nullptr);
587 },
588 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
589 Cost += TTI.getShuffleCost(
591 FixedVectorType::get(SingleOpTy->getElementType(), RegMask.size()),
592 SingleOpTy, CostKind, RegMask, 0, nullptr);
593 });
594 return Cost;
595}
596
597/// Try to perform better estimation of the permutation.
598/// 1. Split the source/destination vectors into real registers.
599/// 2. Do the mask analysis to identify which real registers are
600/// permuted. If more than 1 source registers are used for the
601/// destination register building, the cost for this destination register
602/// is (Number_of_source_register - 1) * Cost_PermuteTwoSrc. If only one
603/// source register is used, build mask and calculate the cost as a cost
604/// of PermuteSingleSrc.
605/// Also, for the single register permute we try to identify if the
606/// destination register is just a copy of the source register or the
607/// copy of the previous destination register (the cost is
608/// TTI::TCC_Basic). If the source register is just reused, the cost for
609/// this operation is 0.
610static InstructionCost
612 std::optional<unsigned> VLen, VectorType *Tp,
614 assert(LegalVT.isFixedLengthVector());
615 if (!VLen || Mask.empty())
617 MVT ElemVT = LegalVT.getVectorElementType();
618 unsigned ElemsPerVReg = *VLen / ElemVT.getFixedSizeInBits();
619 LegalVT = TTI.getTypeLegalizationCost(
620 FixedVectorType::get(Tp->getElementType(), ElemsPerVReg))
621 .second;
622 // Number of destination vectors after legalization:
623 InstructionCost NumOfDests =
624 divideCeil(Mask.size(), LegalVT.getVectorNumElements());
625 if (NumOfDests <= 1 ||
627 Tp->getElementType()->getPrimitiveSizeInBits() ||
628 LegalVT.getVectorNumElements() >= Tp->getElementCount().getFixedValue())
630
631 unsigned VecTySize = TTI.getDataLayout().getTypeStoreSize(Tp);
632 unsigned LegalVTSize = LegalVT.getStoreSize();
633 // Number of source vectors after legalization:
634 unsigned NumOfSrcs = divideCeil(VecTySize, LegalVTSize);
635
636 auto *SingleOpTy = FixedVectorType::get(Tp->getElementType(),
637 LegalVT.getVectorNumElements());
638
639 unsigned E = NumOfDests.getValue();
640 unsigned NormalizedVF =
641 LegalVT.getVectorNumElements() * std::max(NumOfSrcs, E);
642 unsigned NumOfSrcRegs = NormalizedVF / LegalVT.getVectorNumElements();
643 unsigned NumOfDestRegs = NormalizedVF / LegalVT.getVectorNumElements();
644 SmallVector<int> NormalizedMask(NormalizedVF, PoisonMaskElem);
645 assert(NormalizedVF >= Mask.size() &&
646 "Normalized mask expected to be not shorter than original mask.");
647 copy(Mask, NormalizedMask.begin());
648 InstructionCost Cost = 0;
649 int NumShuffles = 0;
650 SmallDenseSet<std::pair<ArrayRef<int>, unsigned>> ReusedSingleSrcShuffles;
652 NormalizedMask, NumOfSrcRegs, NumOfDestRegs, NumOfDestRegs, []() {},
653 [&](ArrayRef<int> RegMask, unsigned SrcReg, unsigned DestReg) {
654 if (ShuffleVectorInst::isIdentityMask(RegMask, RegMask.size()))
655 return;
656 if (!ReusedSingleSrcShuffles.insert(std::make_pair(RegMask, SrcReg))
657 .second)
658 return;
659 ++NumShuffles;
660 Cost += TTI.getShuffleCost(TTI::SK_PermuteSingleSrc, SingleOpTy,
661 SingleOpTy, CostKind, RegMask, 0, nullptr);
662 },
663 [&](ArrayRef<int> RegMask, unsigned Idx1, unsigned Idx2, bool NewReg) {
664 Cost += TTI.getShuffleCost(TTI::SK_PermuteTwoSrc, SingleOpTy,
665 SingleOpTy, CostKind, RegMask, 0, nullptr);
666 NumShuffles += 2;
667 });
668 // Note: check that we do not emit too many shuffles here to prevent code
669 // size explosion.
670 // TODO: investigate, if it can be improved by extra analysis of the masks
671 // to check if the code is more profitable.
672 if ((NumOfDestRegs > 2 && NumShuffles <= static_cast<int>(NumOfDestRegs)) ||
673 (NumOfDestRegs <= 2 && NumShuffles < 4))
674 return Cost;
676}
677
678InstructionCost RISCVTTIImpl::getSlideCost(FixedVectorType *Tp,
679 ArrayRef<int> Mask,
681 // Avoid missing masks and length changing shuffles
682 if (Mask.size() <= 2 || Mask.size() != Tp->getNumElements())
684
685 int NumElts = Tp->getNumElements();
686 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
687 // Avoid scalarization cases
688 if (!LT.second.isFixedLengthVector())
690
691 // Requires moving elements between parts, which requires additional
692 // unmodeled instructions.
693 if (LT.first != 1)
695
696 auto GetSlideOpcode = [&](int SlideAmt) {
697 assert(SlideAmt != 0);
698 bool IsVI = isUInt<5>(std::abs(SlideAmt));
699 if (SlideAmt < 0)
700 return IsVI ? RISCV::VSLIDEDOWN_VI : RISCV::VSLIDEDOWN_VX;
701 return IsVI ? RISCV::VSLIDEUP_VI : RISCV::VSLIDEUP_VX;
702 };
703
704 std::array<std::pair<int, int>, 2> SrcInfo;
705 if (!isMaskedSlidePair(Mask, NumElts, SrcInfo))
707
708 if (SrcInfo[1].second == 0)
709 std::swap(SrcInfo[0], SrcInfo[1]);
710
711 InstructionCost FirstSlideCost = 0;
712 if (SrcInfo[0].second != 0) {
713 unsigned Opcode = GetSlideOpcode(SrcInfo[0].second);
714 FirstSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
715 }
716
717 if (SrcInfo[1].first == -1)
718 return FirstSlideCost;
719
720 InstructionCost SecondSlideCost = 0;
721 if (SrcInfo[1].second != 0) {
722 unsigned Opcode = GetSlideOpcode(SrcInfo[1].second);
723 SecondSlideCost = getRISCVInstructionCost(Opcode, LT.second, CostKind);
724 } else {
725 SecondSlideCost =
726 getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second, CostKind);
727 }
728
729 auto EC = Tp->getElementCount();
730 VectorType *MaskTy =
732 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
733 return FirstSlideCost + SecondSlideCost + MaskCost;
734}
735
739 ArrayRef<int> Mask, int Index, VectorType *SubTp,
741 const Instruction *CxtI) const {
742 assert((Mask.empty() || DstTy->isScalableTy() ||
743 Mask.size() == DstTy->getElementCount().getKnownMinValue()) &&
744 "Expected the Mask to match the return size if given");
745 assert(SrcTy->getScalarType() == DstTy->getScalarType() &&
746 "Expected the same scalar types");
747
748 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
749
750 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
751 // For now, skip all fixed vector cost analysis when P extension is available
752 // to avoid crashes in getMinRVVVectorSizeInBits()
753 if (ST->hasStdExtP() && isa<FixedVectorType>(SrcTy))
754 return 1;
755
756 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(SrcTy);
757
758 // First, handle cases where having a fixed length vector enables us to
759 // give a more accurate cost than falling back to generic scalable codegen.
760 // TODO: Each of these cases hints at a modeling gap around scalable vectors.
761 if (auto *FVTp = dyn_cast<FixedVectorType>(SrcTy);
762 FVTp && ST->hasVInstructions() && LT.second.isFixedLengthVector()) {
764 *this, LT.second, ST->getRealVLen(),
765 Kind == TTI::SK_InsertSubvector ? DstTy : SrcTy, Mask, CostKind);
766 if (VRegSplittingCost.isValid())
767 return VRegSplittingCost;
768 switch (Kind) {
769 default:
770 break;
772 if (Mask.size() >= 2) {
773 MVT EltTp = LT.second.getVectorElementType();
774 // If the size of the element is < ELEN then shuffles of interleaves and
775 // deinterleaves of 2 vectors can be lowered into the following
776 // sequences
777 if (EltTp.getScalarSizeInBits() < ST->getELen()) {
778 // Example sequence:
779 // vsetivli zero, 4, e8, mf4, ta, ma (ignored)
780 // vwaddu.vv v10, v8, v9
781 // li a0, -1 (ignored)
782 // vwmaccu.vx v10, a0, v9
783 if (ShuffleVectorInst::isInterleaveMask(Mask, 2, Mask.size()))
784 return 2 * LT.first * TLI->getLMULCost(LT.second);
785
786 if (Mask[0] == 0 || Mask[0] == 1) {
787 auto DeinterleaveMask = createStrideMask(Mask[0], 2, Mask.size());
788 // Example sequence:
789 // vnsrl.wi v10, v8, 0
790 if (equal(DeinterleaveMask, Mask))
791 return LT.first * getRISCVInstructionCost(RISCV::VNSRL_WI,
792 LT.second, CostKind);
793 }
794 }
795 int SubVectorSize;
796 if (LT.second.getScalarSizeInBits() != 1 &&
797 isRepeatedConcatMask(Mask, SubVectorSize)) {
799 unsigned NumSlides = Log2_32(Mask.size() / SubVectorSize);
800 // The cost of extraction from a subvector is 0 if the index is 0.
801 for (unsigned I = 0; I != NumSlides; ++I) {
802 unsigned InsertIndex = SubVectorSize * (1 << I);
803 FixedVectorType *SubTp =
804 FixedVectorType::get(SrcTy->getElementType(), InsertIndex);
805 FixedVectorType *DestTp =
807 std::pair<InstructionCost, MVT> DestLT =
809 // Add the cost of whole vector register move because the
810 // destination vector register group for vslideup cannot overlap the
811 // source.
812 Cost += DestLT.first * TLI->getLMULCost(DestLT.second);
814 CostKind, {}, InsertIndex, SubTp);
815 }
816 return Cost;
817 }
818 }
819
820 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
821 SlideCost.isValid())
822 return SlideCost;
823
824 // vrgather + cost of generating the mask constant.
825 // We model this for an unknown mask with a single vrgather.
826 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
827 LT.second.getVectorNumElements() <= 256)) {
828 VectorType *IdxTy =
829 getVRGatherIndexType(LT.second, *ST, SrcTy->getContext());
830 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
831 return IndexCost +
832 getRISCVInstructionCost(RISCV::VRGATHER_VV, LT.second, CostKind);
833 }
834 break;
835 }
838
839 if (InstructionCost SlideCost = getSlideCost(FVTp, Mask, CostKind);
840 SlideCost.isValid())
841 return SlideCost;
842
843 // 2 x (vrgather + cost of generating the mask constant) + cost of mask
844 // register for the second vrgather. We model this for an unknown
845 // (shuffle) mask.
846 if (LT.first == 1 && (LT.second.getScalarSizeInBits() != 8 ||
847 LT.second.getVectorNumElements() <= 256)) {
848 auto &C = SrcTy->getContext();
849 auto EC = SrcTy->getElementCount();
850 VectorType *IdxTy = getVRGatherIndexType(LT.second, *ST, C);
852 InstructionCost IndexCost = getConstantPoolLoadCost(IdxTy, CostKind);
853 InstructionCost MaskCost = getConstantPoolLoadCost(MaskTy, CostKind);
854 return 2 * IndexCost +
855 getRISCVInstructionCost({RISCV::VRGATHER_VV, RISCV::VRGATHER_VV},
856 LT.second, CostKind) +
857 MaskCost;
858 }
859 break;
860 }
861 }
862
863 auto shouldSplit = [](TTI::ShuffleKind Kind) {
864 switch (Kind) {
865 default:
866 return false;
870 return true;
871 }
872 };
873
874 if (!Mask.empty() && LT.first.isValid() && LT.first != 1 &&
875 shouldSplit(Kind)) {
876 InstructionCost SplitCost =
877 costShuffleViaSplitting(*this, LT.second, FVTp, Mask, CostKind);
878 if (SplitCost.isValid())
879 return SplitCost;
880 }
881 }
882
883 // Handle scalable vectors (and fixed vectors legalized to scalable vectors).
884 switch (Kind) {
885 default:
886 // Fallthrough to generic handling.
887 // TODO: Most of these cases will return getInvalid in generic code, and
888 // must be implemented here.
889 break;
891 // Extract at zero is always a subregister extract
892 if (Index == 0)
893 return TTI::TCC_Free;
894
895 // If we're extracting a subvector of at most m1 size at a sub-register
896 // boundary - which unfortunately we need exact vlen to identify - this is
897 // a subregister extract at worst and thus won't require a vslidedown.
898 // TODO: Extend for aligned m2, m4 subvector extracts
899 // TODO: Extend for misalgined (but contained) extracts
900 // TODO: Extend for scalable subvector types
901 if (std::pair<InstructionCost, MVT> SubLT = getTypeLegalizationCost(SubTp);
902 SubLT.second.isValid() && SubLT.second.isFixedLengthVector()) {
903 if (std::optional<unsigned> VLen = ST->getRealVLen();
904 VLen && SubLT.second.getScalarSizeInBits() * Index % *VLen == 0 &&
905 SubLT.second.getSizeInBits() <= *VLen)
906 return TTI::TCC_Free;
907 }
908
909 // Example sequence:
910 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
911 // vslidedown.vi v8, v9, 2
912 return LT.first *
913 getRISCVInstructionCost(RISCV::VSLIDEDOWN_VI, LT.second, CostKind);
915 // Example sequence:
916 // vsetivli zero, 4, e8, mf2, tu, ma (ignored)
917 // vslideup.vi v8, v9, 2
918 LT = getTypeLegalizationCost(DstTy);
919 return LT.first *
920 getRISCVInstructionCost(RISCV::VSLIDEUP_VI, LT.second, CostKind);
921 case TTI::SK_Select: {
922 // Example sequence:
923 // li a0, 90
924 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
925 // vmv.s.x v0, a0
926 // vmerge.vvm v8, v9, v8, v0
927 // We use 2 for the cost of the mask materialization as this is the true
928 // cost for small masks and most shuffles are small. At worst, this cost
929 // should be a very small constant for the constant pool load. As such,
930 // we may bias towards large selects slightly more than truly warranted.
931 return LT.first *
932 (1 + getRISCVInstructionCost({RISCV::VMV_S_X, RISCV::VMERGE_VVM},
933 LT.second, CostKind));
934 }
935 case TTI::SK_Broadcast: {
936 // Check for broadcast loads, which are synthesized by optimized zero-stride
937 // loads (this is checked in RISCVTTIImpl::isLegalBroadcastLoad).
938 bool IsLoad = !Args.empty() && isa<LoadInst>(Args[0]);
939 if (IsLoad && LT.second.isVector() &&
940 isLegalBroadcastLoad(SrcTy->getElementType(),
941 LT.second.getVectorElementCount()))
942 return 0;
943
944 bool HasScalar = (Args.size() > 0) && (Operator::getOpcode(Args[0]) ==
945 Instruction::InsertElement);
946 if (LT.second.getScalarSizeInBits() == 1) {
947 if (HasScalar) {
948 // Example sequence:
949 // andi a0, a0, 1
950 // vsetivli zero, 2, e8, mf8, ta, ma (ignored)
951 // vmv.v.x v8, a0
952 // vmsne.vi v0, v8, 0
953 return LT.first *
954 (1 + getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
955 LT.second, CostKind));
956 }
957 // Example sequence:
958 // vsetivli zero, 2, e8, mf8, ta, mu (ignored)
959 // vmv.v.i v8, 0
960 // vmerge.vim v8, v8, 1, v0
961 // vmv.x.s a0, v8
962 // andi a0, a0, 1
963 // vmv.v.x v8, a0
964 // vmsne.vi v0, v8, 0
965
966 return LT.first *
967 (1 + getRISCVInstructionCost({RISCV::VMV_V_I, RISCV::VMERGE_VIM,
968 RISCV::VMV_X_S, RISCV::VMV_V_X,
969 RISCV::VMSNE_VI},
970 LT.second, CostKind));
971 }
972
973 if (HasScalar) {
974 // Example sequence:
975 // vmv.v.x v8, a0
976 return LT.first *
977 getRISCVInstructionCost(RISCV::VMV_V_X, LT.second, CostKind);
978 }
979
980 // Example sequence:
981 // vrgather.vi v9, v8, 0
982 return LT.first *
983 getRISCVInstructionCost(RISCV::VRGATHER_VI, LT.second, CostKind);
984 }
985 case TTI::SK_Splice: {
986 // vslidedown+vslideup.
987 // TODO: Multiplying by LT.first implies this legalizes into multiple copies
988 // of similar code, but I think we expand through memory.
989 unsigned Opcodes[2] = {RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX};
990 if (Index >= 0 && Index < 32)
991 Opcodes[0] = RISCV::VSLIDEDOWN_VI;
992 else if (Index < 0 && Index > -32)
993 Opcodes[1] = RISCV::VSLIDEUP_VI;
994 return LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
995 }
996 case TTI::SK_Reverse: {
997
998 if (!LT.second.isVector())
1000
1001 // TODO: Cases to improve here:
1002 // * Illegal vector types
1003 // * i64 on RV32
1004 if (SrcTy->getElementType()->isIntegerTy(1)) {
1005 VectorType *WideTy =
1006 VectorType::get(IntegerType::get(SrcTy->getContext(), 8),
1007 cast<VectorType>(SrcTy)->getElementCount());
1008 return getCastInstrCost(Instruction::ZExt, WideTy, SrcTy,
1010 getShuffleCost(TTI::SK_Reverse, WideTy, WideTy, CostKind, {}, 0,
1011 nullptr) +
1012 getCastInstrCost(Instruction::Trunc, SrcTy, WideTy,
1014 }
1015
1016 MVT ContainerVT = LT.second;
1017 if (LT.second.isFixedLengthVector())
1018 ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1019 MVT M1VT = RISCVTargetLowering::getM1VT(ContainerVT);
1020 if (ContainerVT.bitsLE(M1VT)) {
1021 // Example sequence:
1022 // csrr a0, vlenb
1023 // srli a0, a0, 3
1024 // addi a0, a0, -1
1025 // vsetvli a1, zero, e8, mf8, ta, mu (ignored)
1026 // vid.v v9
1027 // vrsub.vx v10, v9, a0
1028 // vrgather.vv v9, v8, v10
1029 InstructionCost LenCost = 3;
1030 if (LT.second.isFixedLengthVector())
1031 // vrsub.vi has a 5 bit immediate field, otherwise an li suffices
1032 LenCost = isInt<5>(LT.second.getVectorNumElements() - 1) ? 0 : 1;
1033 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX, RISCV::VRGATHER_VV};
1034 if (LT.second.isFixedLengthVector() &&
1035 isInt<5>(LT.second.getVectorNumElements() - 1))
1036 Opcodes[1] = RISCV::VRSUB_VI;
1037 InstructionCost GatherCost =
1038 getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1039 return LT.first * (LenCost + GatherCost);
1040 }
1041
1042 // At high LMUL, we split into a series of M1 reverses (see
1043 // lowerVECTOR_REVERSE) and then do a single slide at the end to eliminate
1044 // the resulting gap at the bottom (for fixed vectors only). The important
1045 // bit is that the cost scales linearly, not quadratically with LMUL.
1046 unsigned M1Opcodes[] = {RISCV::VID_V, RISCV::VRSUB_VX};
1047 InstructionCost FixedCost =
1048 getRISCVInstructionCost(M1Opcodes, M1VT, CostKind) + 3;
1049 unsigned Ratio =
1050 ContainerVT.getVectorMinNumElements() / M1VT.getVectorMinNumElements();
1051 InstructionCost GatherCost =
1052 getRISCVInstructionCost({RISCV::VRGATHER_VV}, M1VT, CostKind) * Ratio;
1053 InstructionCost SlideCost = !LT.second.isFixedLengthVector() ? 0 :
1054 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX}, LT.second, CostKind);
1055 return FixedCost + LT.first * (GatherCost + SlideCost);
1056 }
1057 }
1058 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
1059 SubTp);
1060}
1061
1062static unsigned isM1OrSmaller(MVT VT) {
1064 return (LMUL == RISCVVType::VLMUL::LMUL_F8 ||
1068}
1069
1071 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
1072 TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef<Value *> VL,
1073 TTI::VectorInstrContext VIC) const {
1076
1077 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
1078 // For now, skip all fixed vector cost analysis when P extension is available
1079 // to avoid crashes in getMinRVVVectorSizeInBits()
1080 if (ST->hasStdExtP() && isa<FixedVectorType>(Ty)) {
1081 return 1; // Treat as single instruction cost for now
1082 }
1083
1084 // A build_vector (which is m1 sized or smaller) can be done in no
1085 // worse than one vslide1down.vx per element in the type. We could
1086 // in theory do an explode_vector in the inverse manner, but our
1087 // lowering today does not have a first class node for this pattern.
1089 Ty, DemandedElts, Insert, Extract, CostKind);
1090 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1091 if (Insert && !Extract && LT.first.isValid() && LT.second.isVector()) {
1092 if (Ty->getScalarSizeInBits() == 1) {
1093 auto *WideVecTy = cast<VectorType>(Ty->getWithNewBitWidth(8));
1094 // Note: Implicit scalar anyextend is assumed to be free since the i1
1095 // must be stored in a GPR.
1096 return getScalarizationOverhead(WideVecTy, DemandedElts, Insert, Extract,
1097 CostKind) +
1098 getCastInstrCost(Instruction::Trunc, Ty, WideVecTy,
1100 }
1101
1102 assert(LT.second.isFixedLengthVector());
1103 MVT ContainerVT = TLI->getContainerForFixedLengthVector(LT.second);
1104 if (isM1OrSmaller(ContainerVT)) {
1105 InstructionCost BV =
1106 cast<FixedVectorType>(Ty)->getNumElements() *
1107 getRISCVInstructionCost(RISCV::VSLIDE1DOWN_VX, LT.second, CostKind);
1108 if (BV < Cost)
1109 Cost = BV;
1110 }
1111 }
1112 return Cost;
1113}
1114
1118 Type *DataTy = MICA.getDataType();
1119 Align Alignment = MICA.getAlignment();
1120 switch (MICA.getID()) {
1121 case Intrinsic::vp_load_ff: {
1122 EVT DataTypeVT = TLI->getValueType(DL, DataTy);
1123 if (!TLI->isLegalFirstFaultLoad(DataTypeVT, Alignment))
1125
1126 unsigned AS = MICA.getAddressSpace();
1127 return getMemoryOpCost(Instruction::Load, DataTy, Alignment, AS, CostKind,
1128 {TTI::OK_AnyValue, TTI::OP_None}, nullptr);
1129 }
1130 case Intrinsic::experimental_vp_strided_load:
1131 case Intrinsic::experimental_vp_strided_store:
1132 return getStridedMemoryOpCost(MICA, CostKind);
1133 case Intrinsic::masked_compressstore:
1134 case Intrinsic::masked_expandload:
1136 case Intrinsic::vp_scatter:
1137 case Intrinsic::vp_gather:
1138 case Intrinsic::masked_scatter:
1139 case Intrinsic::masked_gather:
1140 return getGatherScatterOpCost(MICA, CostKind);
1141 case Intrinsic::vp_load:
1142 case Intrinsic::vp_store:
1143 case Intrinsic::masked_load:
1144 case Intrinsic::masked_store:
1145 return getMaskedMemoryOpCost(MICA, CostKind);
1146 }
1148}
1149
1153 unsigned Opcode = MICA.getID() == Intrinsic::masked_load ? Instruction::Load
1154 : Instruction::Store;
1155 Type *Src = MICA.getDataType();
1156 Align Alignment = MICA.getAlignment();
1157 unsigned AddressSpace = MICA.getAddressSpace();
1158
1159 if (!isLegalMaskedLoadStore(Src, Alignment) ||
1162
1163 return getMemoryOpCost(Opcode, Src, Alignment, AddressSpace, CostKind);
1164}
1165
1167 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
1168 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
1169 bool UseMaskForCond, bool UseMaskForGaps) const {
1170
1171 // The interleaved memory access pass will lower (de)interleave ops combined
1172 // with an adjacent appropriate memory to vlseg/vsseg intrinsics. vlseg/vsseg
1173 // only support masking per-iteration (i.e. condition), not per-segment (i.e.
1174 // gap).
1175 if (!UseMaskForGaps && Factor <= TLI->getMaxSupportedInterleaveFactor()) {
1176 auto *VTy = cast<VectorType>(VecTy);
1177 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(VTy);
1178 // Need to make sure type has't been scalarized
1179 if (LT.second.isVector()) {
1181 return LT.first * TTI::TCC_Basic;
1182
1183 auto *SubVecTy =
1184 VectorType::get(VTy->getElementType(),
1185 VTy->getElementCount().divideCoefficientBy(Factor));
1186 if (VTy->getElementCount().isKnownMultipleOf(Factor) &&
1187 TLI->isLegalInterleavedAccessType(SubVecTy, Factor, Alignment,
1188 AddressSpace, DL)) {
1189
1190 // Some processors optimize segment loads/stores as N * DLEN sized
1191 // load ops + Factor * LMUL shuffle ops.
1192 if (ST->hasOptimizedSegmentLoadStore(Factor)) {
1193 unsigned VecSizeInBits =
1194 getEstimatedVLFor(VTy) * VTy->getScalarSizeInBits();
1195 unsigned VLENForTuning =
1197 unsigned DLENForTuning = VLENForTuning / ST->getDLenFactor();
1198 InstructionCost Cost = divideCeil(VecSizeInBits, DLENForTuning);
1199 MVT SubVecVT = getTLI()->getValueType(DL, SubVecTy).getSimpleVT();
1200 Cost += Factor * TLI->getLMULCost(SubVecVT);
1201 return Cost;
1202 }
1203
1204 // Otherwise, the cost is proportional to the number of elements (VL *
1205 // Factor ops).
1206 unsigned NumLoads = getEstimatedVLFor(VTy);
1207 return NumLoads * TTI::TCC_Basic;
1208 }
1209 }
1210 }
1211
1212 // TODO: Return the cost of interleaved accesses for scalable vector when
1213 // unable to convert to segment accesses instructions.
1214 if (isa<ScalableVectorType>(VecTy))
1216
1217 auto *FVTy = cast<FixedVectorType>(VecTy);
1218 // When gaps are only at the tail, for interleaved load, we can emit a wide
1219 // masked load and shufflevectors. For interleaved store, we can emit
1220 // shufflevectors and a wide masked store. The interleaved memory access pass
1221 // will lower them into vlsseg/vssseg intrinsics.
1222 if (UseMaskForGaps) {
1223 assert(llvm::is_sorted(Indices) && "Indices must be sorted");
1224 assert(llvm::adjacent_find(Indices) == Indices.end() &&
1225 "Indices should not contain duplicate elements");
1226 unsigned NumOfFields = Indices.size();
1227 bool IsTailGapOnly = NumOfFields > 1 && (NumOfFields == Indices.back() + 1);
1228 if (IsTailGapOnly &&
1229 NumOfFields <= TLI->getMaxSupportedInterleaveFactor()) {
1230 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(FVTy);
1231 if (LT.second.isVector() &&
1232 FVTy->getElementCount().isKnownMultipleOf(Factor)) {
1233 auto *SubVecTy = VectorType::get(
1234 FVTy->getElementType(),
1235 FVTy->getElementCount().divideCoefficientBy(Factor));
1236 if (TLI->isLegalInterleavedAccessType(SubVecTy, NumOfFields, Alignment,
1237 AddressSpace, DL)) {
1238 // The cost is proportional to the total number of element accesses.
1239 unsigned NumAccesses = getEstimatedVLFor(FVTy);
1240 return NumAccesses * TTI::TCC_Basic;
1241 }
1242 }
1243 }
1244 }
1245
1246 InstructionCost MemCost =
1247 getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace, CostKind);
1248 unsigned VF = FVTy->getNumElements() / Factor;
1249
1250 // An interleaved load will look like this for Factor=3:
1251 // %wide.vec = load <12 x i32>, ptr %3, align 4
1252 // %strided.vec = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1253 // %strided.vec1 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1254 // %strided.vec2 = shufflevector %wide.vec, poison, <4 x i32> <stride mask>
1255 if (Opcode == Instruction::Load) {
1256 InstructionCost Cost = MemCost;
1257 for (unsigned Index : Indices) {
1258 FixedVectorType *VecTy =
1259 FixedVectorType::get(FVTy->getElementType(), VF * Factor);
1260 auto Mask = createStrideMask(Index, Factor, VF);
1261 Mask.resize(VF * Factor, -1);
1262 InstructionCost ShuffleCost =
1264 CostKind, Mask, 0, nullptr, {});
1265 Cost += ShuffleCost;
1266 }
1267 return Cost;
1268 }
1269
1270 // TODO: Model for NF > 2
1271 // We'll need to enhance getShuffleCost to model shuffles that are just
1272 // inserts and extracts into subvectors, since they won't have the full cost
1273 // of a vrgather.
1274 // An interleaved store for 3 vectors of 4 lanes will look like
1275 // %11 = shufflevector <4 x i32> %4, <4 x i32> %6, <8 x i32> <0...7>
1276 // %12 = shufflevector <4 x i32> %9, <4 x i32> poison, <8 x i32> <0...3>
1277 // %13 = shufflevector <8 x i32> %11, <8 x i32> %12, <12 x i32> <0...11>
1278 // %interleaved.vec = shufflevector %13, poison, <12 x i32> <interleave mask>
1279 // store <12 x i32> %interleaved.vec, ptr %10, align 4
1280 if (Factor != 2)
1281 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
1282 Alignment, AddressSpace, CostKind,
1283 UseMaskForCond, UseMaskForGaps);
1284
1285 assert(Opcode == Instruction::Store && "Opcode must be a store");
1286 // For an interleaving store of 2 vectors, we perform one large interleaving
1287 // shuffle that goes into the wide store
1288 auto Mask = createInterleaveMask(VF, Factor);
1289 InstructionCost ShuffleCost =
1291 CostKind, Mask, 0, nullptr, {});
1292 return MemCost + ShuffleCost;
1293}
1294
1298
1299 bool IsLoad = MICA.getID() == Intrinsic::masked_gather ||
1300 MICA.getID() == Intrinsic::vp_gather;
1301 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
1302 Type *DataTy = MICA.getDataType();
1303 Align Alignment = MICA.getAlignment();
1306
1307 if ((Opcode == Instruction::Load &&
1308 !isLegalMaskedGather(DataTy, Align(Alignment))) ||
1309 (Opcode == Instruction::Store &&
1310 !isLegalMaskedScatter(DataTy, Align(Alignment))))
1312
1313 // Cost is proportional to the number of memory operations implied. For
1314 // scalable vectors, we use an estimate on that number since we don't
1315 // know exactly what VL will be.
1316 auto &VTy = *cast<VectorType>(DataTy);
1317 unsigned NumLoads = getEstimatedVLFor(&VTy);
1318 return NumLoads * TTI::TCC_Basic;
1319}
1320
1322 const MemIntrinsicCostAttributes &MICA,
1324 unsigned Opcode = MICA.getID() == Intrinsic::masked_expandload
1325 ? Instruction::Load
1326 : Instruction::Store;
1327 Type *DataTy = MICA.getDataType();
1328 bool VariableMask = MICA.getVariableMask();
1329 Align Alignment = MICA.getAlignment();
1330 bool IsLegal = (Opcode == Instruction::Store &&
1331 isLegalMaskedCompressStore(DataTy, Alignment)) ||
1332 (Opcode == Instruction::Load &&
1333 isLegalMaskedExpandLoad(DataTy, Alignment));
1334 if (!IsLegal || CostKind != TTI::TCK_RecipThroughput)
1336 // Example compressstore sequence:
1337 // vsetivli zero, 8, e32, m2, ta, ma (ignored)
1338 // vcompress.vm v10, v8, v0
1339 // vcpop.m a1, v0
1340 // vsetvli zero, a1, e32, m2, ta, ma
1341 // vse32.v v10, (a0)
1342 // Example expandload sequence:
1343 // vsetivli zero, 8, e8, mf2, ta, ma (ignored)
1344 // vcpop.m a1, v0
1345 // vsetvli zero, a1, e32, m2, ta, ma
1346 // vle32.v v10, (a0)
1347 // vsetivli zero, 8, e32, m2, ta, ma
1348 // viota.m v12, v0
1349 // vrgather.vv v8, v10, v12, v0.t
1350 auto MemOpCost =
1351 getMemoryOpCost(Opcode, DataTy, Alignment, /*AddressSpace*/ 0, CostKind);
1352 auto LT = getTypeLegalizationCost(DataTy);
1353 SmallVector<unsigned, 4> Opcodes{RISCV::VSETVLI};
1354 if (VariableMask)
1355 Opcodes.push_back(RISCV::VCPOP_M);
1356 if (Opcode == Instruction::Store)
1357 Opcodes.append({RISCV::VCOMPRESS_VM});
1358 else
1359 Opcodes.append({RISCV::VSETIVLI, RISCV::VIOTA_M, RISCV::VRGATHER_VV});
1360 return MemOpCost +
1361 LT.first * getRISCVInstructionCost(Opcodes, LT.second, CostKind);
1362}
1363
1367 Type *DataTy = MICA.getDataType();
1368 Align Alignment = MICA.getAlignment();
1369
1370 if (!isLegalStridedLoadStore(DataTy, Alignment))
1372
1374 return TTI::TCC_Basic;
1375
1376 // Cost is proportional to the number of memory operations implied. For
1377 // scalable vectors, we use an estimate on that number since we don't
1378 // know exactly what VL will be.
1379 auto &VTy = *cast<VectorType>(DataTy);
1380 unsigned NumLoads = getEstimatedVLFor(&VTy);
1381 return NumLoads * TTI::TCC_Basic;
1382}
1383
1386 // FIXME: This is a property of the default vector convention, not
1387 // all possible calling conventions. Fixing that will require
1388 // some TTI API and SLP rework.
1391 for (auto *Ty : Tys) {
1392 if (!Ty->isVectorTy())
1393 continue;
1394 Align A = DL.getPrefTypeAlign(Ty);
1395 Cost += getMemoryOpCost(Instruction::Store, Ty, A, 0, CostKind) +
1396 getMemoryOpCost(Instruction::Load, Ty, A, 0, CostKind);
1397 }
1398 return Cost;
1399}
1400
1401// Currently, these represent both throughput and codesize costs
1402// for the respective intrinsics. The costs in this table are simply
1403// instruction counts with the following adjustments made:
1404// * One vsetvli is considered free.
1406 {Intrinsic::floor, MVT::f32, 9},
1407 {Intrinsic::floor, MVT::f64, 9},
1408 {Intrinsic::ceil, MVT::f32, 9},
1409 {Intrinsic::ceil, MVT::f64, 9},
1410 {Intrinsic::trunc, MVT::f32, 7},
1411 {Intrinsic::trunc, MVT::f64, 7},
1412 {Intrinsic::round, MVT::f32, 9},
1413 {Intrinsic::round, MVT::f64, 9},
1414 {Intrinsic::roundeven, MVT::f32, 9},
1415 {Intrinsic::roundeven, MVT::f64, 9},
1416 {Intrinsic::rint, MVT::f32, 7},
1417 {Intrinsic::rint, MVT::f64, 7},
1418 {Intrinsic::nearbyint, MVT::f32, 9},
1419 {Intrinsic::nearbyint, MVT::f64, 9},
1420 {Intrinsic::bswap, MVT::i16, 3},
1421 {Intrinsic::bswap, MVT::i32, 12},
1422 {Intrinsic::bswap, MVT::i64, 31},
1423 {Intrinsic::bitreverse, MVT::i8, 17},
1424 {Intrinsic::bitreverse, MVT::i16, 24},
1425 {Intrinsic::bitreverse, MVT::i32, 33},
1426 {Intrinsic::bitreverse, MVT::i64, 52},
1427 {Intrinsic::ctpop, MVT::i8, 12},
1428 {Intrinsic::ctpop, MVT::i16, 19},
1429 {Intrinsic::ctpop, MVT::i32, 20},
1430 {Intrinsic::ctpop, MVT::i64, 21},
1431 {Intrinsic::ctlz, MVT::i8, 19},
1432 {Intrinsic::ctlz, MVT::i16, 28},
1433 {Intrinsic::ctlz, MVT::i32, 31},
1434 {Intrinsic::ctlz, MVT::i64, 35},
1435 {Intrinsic::cttz, MVT::i8, 16},
1436 {Intrinsic::cttz, MVT::i16, 23},
1437 {Intrinsic::cttz, MVT::i32, 24},
1438 {Intrinsic::cttz, MVT::i64, 25},
1439};
1440
1444 auto *RetTy = ICA.getReturnType();
1445 switch (ICA.getID()) {
1446 case Intrinsic::lrint:
1447 case Intrinsic::llrint:
1448 case Intrinsic::lround:
1449 case Intrinsic::llround: {
1450 auto LT = getTypeLegalizationCost(RetTy);
1451 Type *SrcTy = ICA.getArgTypes().front();
1452 auto SrcLT = getTypeLegalizationCost(SrcTy);
1453 if (ST->hasVInstructions() && LT.second.isVector()) {
1455 unsigned SrcEltSz = DL.getTypeSizeInBits(SrcTy->getScalarType());
1456 unsigned DstEltSz = DL.getTypeSizeInBits(RetTy->getScalarType());
1457 if (LT.second.getVectorElementType() == MVT::bf16) {
1458 if (!ST->hasVInstructionsBF16Minimal())
1460 if (DstEltSz == 32)
1461 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFCVT_X_F_V};
1462 else
1463 Ops = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVT_X_F_V};
1464 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1465 !ST->hasVInstructionsF16()) {
1466 if (!ST->hasVInstructionsF16Minimal())
1468 if (DstEltSz == 32)
1469 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFCVT_X_F_V};
1470 else
1471 Ops = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_X_F_V};
1472
1473 } else if (SrcEltSz > DstEltSz) {
1474 Ops = {RISCV::VFNCVT_X_F_W};
1475 } else if (SrcEltSz < DstEltSz) {
1476 Ops = {RISCV::VFWCVT_X_F_V};
1477 } else {
1478 Ops = {RISCV::VFCVT_X_F_V};
1479 }
1480
1481 // We need to use the source LMUL in the case of a narrowing op, and the
1482 // destination LMUL otherwise.
1483 if (SrcEltSz > DstEltSz)
1484 return SrcLT.first *
1485 getRISCVInstructionCost(Ops, SrcLT.second, CostKind);
1486 return LT.first * getRISCVInstructionCost(Ops, LT.second, CostKind);
1487 }
1488 break;
1489 }
1490 case Intrinsic::ceil:
1491 case Intrinsic::floor:
1492 case Intrinsic::trunc:
1493 case Intrinsic::rint:
1494 case Intrinsic::round:
1495 case Intrinsic::roundeven: {
1496 // These all use the same code.
1497 auto LT = getTypeLegalizationCost(RetTy);
1498 if (!LT.second.isVector() && TLI->isOperationCustom(ISD::FCEIL, LT.second))
1499 return LT.first * 8;
1500 break;
1501 }
1502 case Intrinsic::umin:
1503 case Intrinsic::umax:
1504 case Intrinsic::smin:
1505 case Intrinsic::smax: {
1506 auto LT = getTypeLegalizationCost(RetTy);
1507 if (LT.second.isScalarInteger() && ST->hasStdExtZbb())
1508 return LT.first;
1509
1510 if (ST->hasVInstructions() && LT.second.isVector()) {
1511 unsigned Op;
1512 switch (ICA.getID()) {
1513 case Intrinsic::umin:
1514 Op = RISCV::VMINU_VV;
1515 break;
1516 case Intrinsic::umax:
1517 Op = RISCV::VMAXU_VV;
1518 break;
1519 case Intrinsic::smin:
1520 Op = RISCV::VMIN_VV;
1521 break;
1522 case Intrinsic::smax:
1523 Op = RISCV::VMAX_VV;
1524 break;
1525 }
1526 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1527 }
1528 break;
1529 }
1530 case Intrinsic::sadd_sat:
1531 case Intrinsic::ssub_sat:
1532 case Intrinsic::uadd_sat:
1533 case Intrinsic::usub_sat: {
1534 auto LT = getTypeLegalizationCost(RetTy);
1535 if (ST->hasVInstructions() && LT.second.isVector()) {
1536 unsigned Op;
1537 switch (ICA.getID()) {
1538 case Intrinsic::sadd_sat:
1539 Op = RISCV::VSADD_VV;
1540 break;
1541 case Intrinsic::ssub_sat:
1542 Op = RISCV::VSSUB_VV;
1543 break;
1544 case Intrinsic::uadd_sat:
1545 Op = RISCV::VSADDU_VV;
1546 break;
1547 case Intrinsic::usub_sat:
1548 Op = RISCV::VSSUBU_VV;
1549 break;
1550 }
1551 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1552 }
1553 break;
1554 }
1555 case Intrinsic::fma:
1556 case Intrinsic::fmuladd: {
1557 // TODO: handle promotion with f16/bf16 with zvfhmin/zvfbfmin
1558 auto LT = getTypeLegalizationCost(RetTy);
1559 if (ST->hasVInstructions() && LT.second.isVector())
1560 return LT.first *
1561 getRISCVInstructionCost(RISCV::VFMADD_VV, LT.second, CostKind);
1562 break;
1563 }
1564 case Intrinsic::fabs: {
1565 auto LT = getTypeLegalizationCost(RetTy);
1566 if (ST->hasVInstructions() && LT.second.isVector()) {
1567 // lui a0, 8
1568 // addi a0, a0, -1
1569 // vsetvli a1, zero, e16, m1, ta, ma
1570 // vand.vx v8, v8, a0
1571 // f16 with zvfhmin and bf16 with zvfhbmin
1572 if (LT.second.getVectorElementType() == MVT::bf16 ||
1573 (LT.second.getVectorElementType() == MVT::f16 &&
1574 !ST->hasVInstructionsF16()))
1575 return LT.first * getRISCVInstructionCost(RISCV::VAND_VX, LT.second,
1576 CostKind) +
1577 2;
1578 else
1579 return LT.first *
1580 getRISCVInstructionCost(RISCV::VFSGNJX_VV, LT.second, CostKind);
1581 }
1582 break;
1583 }
1584 case Intrinsic::sqrt: {
1585 auto LT = getTypeLegalizationCost(RetTy);
1586 if (ST->hasVInstructions() && LT.second.isVector()) {
1589 MVT ConvType = LT.second;
1590 MVT FsqrtType = LT.second;
1591 // f16 with zvfhmin and bf16 with zvfbfmin and the type of nxv32[b]f16
1592 // will be spilt.
1593 if (LT.second.getVectorElementType() == MVT::bf16) {
1594 if (LT.second == MVT::nxv32bf16) {
1595 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFWCVTBF16_F_F_V,
1596 RISCV::VFNCVTBF16_F_F_W, RISCV::VFNCVTBF16_F_F_W};
1597 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1598 ConvType = MVT::nxv16f16;
1599 FsqrtType = MVT::nxv16f32;
1600 } else {
1601 ConvOp = {RISCV::VFWCVTBF16_F_F_V, RISCV::VFNCVTBF16_F_F_W};
1602 FsqrtOp = {RISCV::VFSQRT_V};
1603 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1604 }
1605 } else if (LT.second.getVectorElementType() == MVT::f16 &&
1606 !ST->hasVInstructionsF16()) {
1607 if (LT.second == MVT::nxv32f16) {
1608 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFWCVT_F_F_V,
1609 RISCV::VFNCVT_F_F_W, RISCV::VFNCVT_F_F_W};
1610 FsqrtOp = {RISCV::VFSQRT_V, RISCV::VFSQRT_V};
1611 ConvType = MVT::nxv16f16;
1612 FsqrtType = MVT::nxv16f32;
1613 } else {
1614 ConvOp = {RISCV::VFWCVT_F_F_V, RISCV::VFNCVT_F_F_W};
1615 FsqrtOp = {RISCV::VFSQRT_V};
1616 FsqrtType = TLI->getTypeToPromoteTo(ISD::FSQRT, FsqrtType);
1617 }
1618 } else {
1619 FsqrtOp = {RISCV::VFSQRT_V};
1620 }
1621
1622 return LT.first * (getRISCVInstructionCost(FsqrtOp, FsqrtType, CostKind) +
1623 getRISCVInstructionCost(ConvOp, ConvType, CostKind));
1624 }
1625 break;
1626 }
1627 case Intrinsic::cttz:
1628 case Intrinsic::ctlz:
1629 case Intrinsic::ctpop: {
1630 auto LT = getTypeLegalizationCost(RetTy);
1631 if (ST->hasStdExtZvbb() && LT.second.isVector()) {
1632 unsigned Op;
1633 switch (ICA.getID()) {
1634 case Intrinsic::cttz:
1635 Op = RISCV::VCTZ_V;
1636 break;
1637 case Intrinsic::ctlz:
1638 Op = RISCV::VCLZ_V;
1639 break;
1640 case Intrinsic::ctpop:
1641 Op = RISCV::VCPOP_V;
1642 break;
1643 }
1644 return LT.first * getRISCVInstructionCost(Op, LT.second, CostKind);
1645 }
1646 break;
1647 }
1648 case Intrinsic::abs: {
1649 auto LT = getTypeLegalizationCost(RetTy);
1650 if (ST->hasVInstructions() && LT.second.isVector()) {
1651 // vabs.v v10, v8 (alias for vabd.vx v10, v8, zero)
1652 if (ST->hasStdExtZvabd())
1653 return LT.first *
1654 getRISCVInstructionCost({RISCV::VABD_VX}, LT.second, CostKind);
1655
1656 // vrsub.vi v10, v8, 0
1657 // vmax.vv v8, v8, v10
1658 return LT.first *
1659 getRISCVInstructionCost({RISCV::VRSUB_VI, RISCV::VMAX_VV},
1660 LT.second, CostKind);
1661 }
1662 break;
1663 }
1664 case Intrinsic::fshl:
1665 case Intrinsic::fshr: {
1666 if (ICA.getArgs().empty())
1667 break;
1668
1669 // Funnel-shifts are ROTL/ROTR when the first and second operand are equal.
1670 // When Zbb/Zbkb is enabled we can use a single ROL(W)/ROR(I)(W)
1671 // instruction.
1672 if ((ST->hasStdExtZbb() || ST->hasStdExtZbkb()) && RetTy->isIntegerTy() &&
1673 ICA.getArgs()[0] == ICA.getArgs()[1] &&
1674 (RetTy->getIntegerBitWidth() == 32 ||
1675 RetTy->getIntegerBitWidth() == 64) &&
1676 RetTy->getIntegerBitWidth() <= ST->getXLen()) {
1677 return 1;
1678 }
1679 break;
1680 }
1681 case Intrinsic::clmul: {
1682 auto LT = getTypeLegalizationCost(RetTy);
1683 if (!LT.second.isVector() && ST->hasStdExtZvbc() && !ST->hasStdExtZbc() &&
1684 !ST->hasStdExtZbkc()) {
1685 // TODO: Once custom lowering in this case for RV32 is added, this guard
1686 // should be removed and the cost model should be updated.
1687 if (!ST->is64Bit() || LT.second != MVT::i64)
1688 break;
1689 // vmv.s.x v8, a0
1690 // vclmul.vx v8, v8, a1
1691 // vmv.x.s a0, v8
1692 MVT VecVT = MVT::getScalableVectorVT(LT.second, 1);
1693 return LT.first * getRISCVInstructionCost(
1694 {RISCV::VMV_S_X, RISCV::VCLMUL_VX, RISCV::VMV_X_S},
1695 VecVT, CostKind);
1696 }
1697 break;
1698 }
1699 case Intrinsic::masked_udiv:
1700 return getArithmeticInstrCost(Instruction::UDiv, ICA.getReturnType(),
1701 CostKind);
1702 case Intrinsic::masked_sdiv:
1703 return getArithmeticInstrCost(Instruction::SDiv, ICA.getReturnType(),
1704 CostKind);
1705 case Intrinsic::masked_urem:
1706 return getArithmeticInstrCost(Instruction::URem, ICA.getReturnType(),
1707 CostKind);
1708 case Intrinsic::masked_srem:
1709 return getArithmeticInstrCost(Instruction::SRem, ICA.getReturnType(),
1710 CostKind);
1711 case Intrinsic::get_active_lane_mask: {
1712 if (ST->hasVInstructions()) {
1713 Type *ExpRetTy = VectorType::get(
1714 ICA.getArgTypes()[0], cast<VectorType>(RetTy)->getElementCount());
1715 auto LT = getTypeLegalizationCost(ExpRetTy);
1716
1717 // vid.v v8 // considered hoisted
1718 // vsaddu.vx v8, v8, a0
1719 // vmsltu.vx v0, v8, a1
1720 return LT.first *
1721 getRISCVInstructionCost({RISCV::VSADDU_VX, RISCV::VMSLTU_VX},
1722 LT.second, CostKind);
1723 }
1724 break;
1725 }
1726 // TODO: add more intrinsic
1727 case Intrinsic::stepvector: {
1728 auto LT = getTypeLegalizationCost(RetTy);
1729 // Legalisation of illegal types involves an `index' instruction plus
1730 // (LT.first - 1) vector adds.
1731 if (ST->hasVInstructions())
1732 return getRISCVInstructionCost(RISCV::VID_V, LT.second, CostKind) +
1733 (LT.first - 1) *
1734 getRISCVInstructionCost(RISCV::VADD_VX, LT.second, CostKind);
1735 return 1 + (LT.first - 1);
1736 }
1737 case Intrinsic::vector_splice_left:
1738 case Intrinsic::vector_splice_right: {
1739 auto LT = getTypeLegalizationCost(RetTy);
1740 // Constant offsets fall through to getShuffleCost.
1741 if (!ICA.isTypeBasedOnly() && isa<ConstantInt>(ICA.getArgs()[2]))
1742 break;
1743 if (ST->hasVInstructions() && LT.second.isVector()) {
1744 return LT.first *
1745 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VSLIDEUP_VX},
1746 LT.second, CostKind);
1747 }
1748 break;
1749 }
1750 case Intrinsic::experimental_cttz_elts: {
1751 if (!ST->hasVInstructions())
1752 break;
1754 Type *ArgTy = ICA.getArgTypes()[0];
1755 auto LT = getTypeLegalizationCost(ArgTy);
1756
1757 // If the element type is not i1, do a comparison with all-zeros.
1758 if (LT.second.getVectorElementType() != MVT::i1)
1759 Cost += getRISCVInstructionCost(RISCV::VMSNE_VI, LT.second, CostKind);
1760
1761 Cost += getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind);
1762
1763 // If zero_is_poison is false, then we will generate additional
1764 // cmp + select instructions to convert -1 to EVL.
1765 Type *BoolTy = Type::getInt1Ty(RetTy->getContext());
1766 if (ICA.getArgs().size() > 1 &&
1767 cast<ConstantInt>(ICA.getArgs()[1])->isZero())
1768 Cost += getCmpSelInstrCost(Instruction::ICmp, BoolTy, RetTy,
1770 getCmpSelInstrCost(Instruction::Select, RetTy, BoolTy,
1772
1773 return LT.first * Cost;
1774 }
1775 case Intrinsic::experimental_vp_splice: {
1776 // To support type-based query from vectorizer, set the index to 0.
1777 // Note that index only change the cost from vslide.vx to vslide.vi and in
1778 // current implementations they have same costs.
1780 cast<VectorType>(ICA.getArgTypes()[0]), CostKind, {},
1782 }
1783 case Intrinsic::vp_merge: {
1784 // If an operand is a binary op and the type is legal, RISCVVectorPeephole
1785 // will likely fold the resulting vmerge.vvm away.
1787 getTypeLegalizationCost(RetTy).first == 1)
1788 return TTI::TCC_Free;
1789 break;
1790 }
1791 case Intrinsic::fptoui_sat:
1792 case Intrinsic::fptosi_sat: {
1794 bool IsSigned = ICA.getID() == Intrinsic::fptosi_sat;
1795 Type *SrcTy = ICA.getArgTypes()[0];
1796
1797 auto SrcLT = getTypeLegalizationCost(SrcTy);
1798 auto DstLT = getTypeLegalizationCost(RetTy);
1799 if (!SrcTy->isVectorTy())
1800 break;
1801
1802 if (!SrcLT.first.isValid() || !DstLT.first.isValid())
1804
1805 Cost +=
1806 getCastInstrCost(IsSigned ? Instruction::FPToSI : Instruction::FPToUI,
1807 RetTy, SrcTy, TTI::CastContextHint::None, CostKind);
1808
1809 // Handle NaN.
1810 // vmfne v0, v8, v8 # If v8[i] is NaN set v0[i] to 1.
1811 // vmerge.vim v8, v8, 0, v0 # Convert NaN to 0.
1812 Type *CondTy = RetTy->getWithNewBitWidth(1);
1813 Cost += getCmpSelInstrCost(BinaryOperator::FCmp, SrcTy, CondTy,
1815 Cost += getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
1817 return Cost;
1818 }
1819 case Intrinsic::experimental_vector_extract_last_active: {
1820 auto *ValTy = cast<VectorType>(ICA.getArgTypes()[0]);
1821 auto *MaskTy = cast<VectorType>(ICA.getArgTypes()[1]);
1822
1823 auto ValLT = getTypeLegalizationCost(ValTy);
1824 auto MaskLT = getTypeLegalizationCost(MaskTy);
1825
1826 // TODO: Return cheaper cost when the entire lane is inactive.
1827 // The expected asm sequence is:
1828 // vcpop.m a0, v0
1829 // beqz a0, exit # Return passthru when the entire lane is inactive.
1830 // vid v10, v0.t
1831 // vredmaxu.vs v10, v10, v10
1832 // vmv.x.s a0, v10
1833 // zext.b a0, a0
1834 // vslidedown.vx v8, v8, a0
1835 // vmv.x.s a0, v8
1836 // exit:
1837 // ...
1838
1839 // Find a suitable type for a stepvector.
1840 ConstantRange VScaleRange(APInt(64, 1), APInt::getZero(64));
1841 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
1842 TLI->getVectorIdxTy(getDataLayout()), MaskTy->getElementCount(),
1843 /*ZeroIsPoison=*/true, &VScaleRange);
1844 EltWidth = std::max(EltWidth, MaskTy->getScalarSizeInBits());
1845 Type *StepTy = Type::getIntNTy(MaskTy->getContext(), EltWidth);
1846 auto *StepVecTy = VectorType::get(StepTy, ValTy->getElementCount());
1847 auto StepLT = getTypeLegalizationCost(StepVecTy);
1848
1849 // Currently expandVectorFindLastActive cannot handle step vector split.
1850 // So return invalid when the type needs split.
1851 // FIXME: Remove this if expandVectorFindLastActive supports split vector.
1852 if (StepLT.first > 1)
1854
1856 unsigned Opcodes[] = {RISCV::VID_V, RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
1857
1858 Cost += MaskLT.first *
1859 getRISCVInstructionCost(RISCV::VCPOP_M, MaskLT.second, CostKind);
1860 Cost += getCFInstrCost(Instruction::CondBr, CostKind, nullptr);
1861 Cost += StepLT.first *
1862 getRISCVInstructionCost(Opcodes, StepLT.second, CostKind);
1863 Cost += getCastInstrCost(Instruction::ZExt,
1864 Type::getInt64Ty(ValTy->getContext()), StepTy,
1866 Cost += ValLT.first *
1867 getRISCVInstructionCost({RISCV::VSLIDEDOWN_VI, RISCV::VMV_X_S},
1868 ValLT.second, CostKind);
1869 return Cost;
1870 }
1871 }
1872
1873 if (ST->hasVInstructions() && RetTy->isVectorTy()) {
1874 if (auto LT = getTypeLegalizationCost(RetTy);
1875 LT.second.isVector()) {
1876 MVT EltTy = LT.second.getVectorElementType();
1877 if (const auto *Entry = CostTableLookup(VectorIntrinsicCostTable,
1878 ICA.getID(), EltTy))
1879 return LT.first * Entry->Cost;
1880 }
1881 }
1882
1884}
1885
1888 const SCEV *Ptr,
1890 // Address computations for vector indexed load/store likely require an offset
1891 // and/or scaling.
1892 if (ST->hasVInstructions() && PtrTy->isVectorTy())
1893 return getArithmeticInstrCost(Instruction::Add, PtrTy, CostKind);
1894
1895 return BaseT::getAddressComputationCost(PtrTy, SE, Ptr, CostKind);
1896}
1897
1899 Type *Src,
1902 const Instruction *I) const {
1903 bool IsVectorType = isa<VectorType>(Dst) && isa<VectorType>(Src);
1904 if (!IsVectorType)
1905 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1906
1907 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
1908 // For now, skip all fixed vector cost analysis when P extension is available
1909 // to avoid crashes in getMinRVVVectorSizeInBits()
1910 if (ST->hasStdExtP() &&
1912 return 1; // Treat as single instruction cost for now
1913 }
1914
1915 // FIXME: Need to compute legalizing cost for illegal types. The current
1916 // code handles only legal types and those which can be trivially
1917 // promoted to legal.
1918 if (!ST->hasVInstructions() || Src->getScalarSizeInBits() > ST->getELen() ||
1919 Dst->getScalarSizeInBits() > ST->getELen())
1920 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1921
1922 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1923 assert(ISD && "Invalid opcode");
1924 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Src);
1925 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Dst);
1926
1927 // Handle i1 source and dest cases *before* calling logic in BasicTTI.
1928 // The shared implementation doesn't model vector widening during legalization
1929 // and instead assumes scalarization. In order to scalarize an <N x i1>
1930 // vector, we need to extend/trunc to/from i8. If we don't special case
1931 // this, we can get an infinite recursion cycle.
1932 switch (ISD) {
1933 default:
1934 break;
1935 case ISD::SIGN_EXTEND:
1936 case ISD::ZERO_EXTEND:
1937 if (Src->getScalarSizeInBits() == 1) {
1938 // We do not use vsext/vzext to extend from mask vector.
1939 // Instead we use the following instructions to extend from mask vector:
1940 // vmv.v.i v8, 0
1941 // vmerge.vim v8, v8, -1, v0 (repeated per split)
1942 return getRISCVInstructionCost(RISCV::VMV_V_I, DstLT.second, CostKind) +
1943 DstLT.first * getRISCVInstructionCost(RISCV::VMERGE_VIM,
1944 DstLT.second, CostKind) +
1945 DstLT.first - 1;
1946 }
1947 break;
1948 case ISD::TRUNCATE:
1949 if (Dst->getScalarSizeInBits() == 1) {
1950 // We do not use several vncvt to truncate to mask vector. So we could
1951 // not use PowDiff to calculate it.
1952 // Instead we use the following instructions to truncate to mask vector:
1953 // vand.vi v8, v8, 1
1954 // vmsne.vi v0, v8, 0
1955 return SrcLT.first *
1956 getRISCVInstructionCost({RISCV::VAND_VI, RISCV::VMSNE_VI},
1957 SrcLT.second, CostKind) +
1958 SrcLT.first - 1;
1959 }
1960 break;
1961 };
1962
1963 // Our actual lowering for the case where a wider legal type is available
1964 // uses promotion to the wider type. This is reflected in the result of
1965 // getTypeLegalizationCost, but BasicTTI assumes the widened cases are
1966 // scalarized if the legalized Src and Dst are not equal sized.
1967 const DataLayout &DL = this->getDataLayout();
1968 if (!SrcLT.second.isVector() || !DstLT.second.isVector() ||
1969 !SrcLT.first.isValid() || !DstLT.first.isValid() ||
1970 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Src),
1971 SrcLT.second.getSizeInBits()) ||
1972 !TypeSize::isKnownLE(DL.getTypeSizeInBits(Dst),
1973 DstLT.second.getSizeInBits()) ||
1974 SrcLT.first > 1 || DstLT.first > 1)
1975 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1976
1977 // The split cost is handled by the base getCastInstrCost
1978 assert((SrcLT.first == 1) && (DstLT.first == 1) && "Illegal type");
1979
1980 int PowDiff = (int)Log2_32(DstLT.second.getScalarSizeInBits()) -
1981 (int)Log2_32(SrcLT.second.getScalarSizeInBits());
1982 switch (ISD) {
1983 case ISD::SIGN_EXTEND:
1984 case ISD::ZERO_EXTEND: {
1985 if ((PowDiff < 1) || (PowDiff > 3))
1986 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1987 unsigned SExtOp[] = {RISCV::VSEXT_VF2, RISCV::VSEXT_VF4, RISCV::VSEXT_VF8};
1988 unsigned ZExtOp[] = {RISCV::VZEXT_VF2, RISCV::VZEXT_VF4, RISCV::VZEXT_VF8};
1989 unsigned Op =
1990 (ISD == ISD::SIGN_EXTEND) ? SExtOp[PowDiff - 1] : ZExtOp[PowDiff - 1];
1991 return getRISCVInstructionCost(Op, DstLT.second, CostKind);
1992 }
1993 case ISD::TRUNCATE:
1994 case ISD::FP_EXTEND:
1995 case ISD::FP_ROUND: {
1996 // Counts of narrow/widen instructions.
1997 unsigned SrcEltSize = SrcLT.second.getScalarSizeInBits();
1998 unsigned DstEltSize = DstLT.second.getScalarSizeInBits();
1999
2000 unsigned Op = (ISD == ISD::TRUNCATE) ? RISCV::VNSRL_WI
2001 : (ISD == ISD::FP_EXTEND) ? RISCV::VFWCVT_F_F_V
2002 : RISCV::VFNCVT_F_F_W;
2004 for (; SrcEltSize != DstEltSize;) {
2005 MVT ElementMVT = (ISD == ISD::TRUNCATE)
2006 ? MVT::getIntegerVT(DstEltSize)
2007 : MVT::getFloatingPointVT(DstEltSize);
2008 MVT DstMVT = DstLT.second.changeVectorElementType(ElementMVT);
2009 DstEltSize =
2010 (DstEltSize > SrcEltSize) ? DstEltSize >> 1 : DstEltSize << 1;
2011 Cost += getRISCVInstructionCost(Op, DstMVT, CostKind);
2012 }
2013 return Cost;
2014 }
2015 case ISD::FP_TO_SINT:
2016 case ISD::FP_TO_UINT: {
2017 unsigned IsSigned = ISD == ISD::FP_TO_SINT;
2018 unsigned FCVT = IsSigned ? RISCV::VFCVT_RTZ_X_F_V : RISCV::VFCVT_RTZ_XU_F_V;
2019 unsigned FWCVT =
2020 IsSigned ? RISCV::VFWCVT_RTZ_X_F_V : RISCV::VFWCVT_RTZ_XU_F_V;
2021 unsigned FNCVT =
2022 IsSigned ? RISCV::VFNCVT_RTZ_X_F_W : RISCV::VFNCVT_RTZ_XU_F_W;
2023 unsigned SrcEltSize = Src->getScalarSizeInBits();
2024 unsigned DstEltSize = Dst->getScalarSizeInBits();
2026 if ((SrcEltSize == 16) &&
2027 (!ST->hasVInstructionsF16() || ((DstEltSize / 2) > SrcEltSize))) {
2028 // If the target only supports zvfhmin or it is fp16-to-i64 conversion
2029 // pre-widening to f32 and then convert f32 to integer
2030 VectorType *VecF32Ty =
2031 VectorType::get(Type::getFloatTy(Dst->getContext()),
2032 cast<VectorType>(Dst)->getElementCount());
2033 std::pair<InstructionCost, MVT> VecF32LT =
2034 getTypeLegalizationCost(VecF32Ty);
2035 Cost +=
2036 VecF32LT.first * getRISCVInstructionCost(RISCV::VFWCVT_F_F_V,
2037 VecF32LT.second, CostKind);
2038 Cost += getCastInstrCost(Opcode, Dst, VecF32Ty, CCH, CostKind, I);
2039 return Cost;
2040 }
2041 if (DstEltSize == SrcEltSize)
2042 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2043 else if (DstEltSize > SrcEltSize)
2044 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2045 else { // (SrcEltSize > DstEltSize)
2046 // First do a narrowing conversion to an integer half the size, then
2047 // truncate if needed.
2048 MVT ElementVT = MVT::getIntegerVT(SrcEltSize / 2);
2049 MVT VecVT = DstLT.second.changeVectorElementType(ElementVT);
2050 Cost += getRISCVInstructionCost(FNCVT, VecVT, CostKind);
2051 if ((SrcEltSize / 2) > DstEltSize) {
2052 Type *VecTy = EVT(VecVT).getTypeForEVT(Dst->getContext());
2053 Cost +=
2054 getCastInstrCost(Instruction::Trunc, Dst, VecTy, CCH, CostKind, I);
2055 }
2056 }
2057 return Cost;
2058 }
2059 case ISD::SINT_TO_FP:
2060 case ISD::UINT_TO_FP: {
2061 unsigned IsSigned = ISD == ISD::SINT_TO_FP;
2062 unsigned FCVT = IsSigned ? RISCV::VFCVT_F_X_V : RISCV::VFCVT_F_XU_V;
2063 unsigned FWCVT = IsSigned ? RISCV::VFWCVT_F_X_V : RISCV::VFWCVT_F_XU_V;
2064 unsigned FNCVT = IsSigned ? RISCV::VFNCVT_F_X_W : RISCV::VFNCVT_F_XU_W;
2065 unsigned SrcEltSize = Src->getScalarSizeInBits();
2066 unsigned DstEltSize = Dst->getScalarSizeInBits();
2067
2069 if ((DstEltSize == 16) &&
2070 (!ST->hasVInstructionsF16() || ((SrcEltSize / 2) > DstEltSize))) {
2071 // If the target only supports zvfhmin or it is i64-to-fp16 conversion
2072 // it is converted to f32 and then converted to f16
2073 VectorType *VecF32Ty =
2074 VectorType::get(Type::getFloatTy(Dst->getContext()),
2075 cast<VectorType>(Dst)->getElementCount());
2076 std::pair<InstructionCost, MVT> VecF32LT =
2077 getTypeLegalizationCost(VecF32Ty);
2078 Cost += getCastInstrCost(Opcode, VecF32Ty, Src, CCH, CostKind, I);
2079 Cost += VecF32LT.first * getRISCVInstructionCost(RISCV::VFNCVT_F_F_W,
2080 DstLT.second, CostKind);
2081 return Cost;
2082 }
2083
2084 if (DstEltSize == SrcEltSize)
2085 Cost += getRISCVInstructionCost(FCVT, DstLT.second, CostKind);
2086 else if (DstEltSize > SrcEltSize) {
2087 if ((DstEltSize / 2) > SrcEltSize) {
2088 VectorType *VecTy =
2089 VectorType::get(IntegerType::get(Dst->getContext(), DstEltSize / 2),
2090 cast<VectorType>(Dst)->getElementCount());
2091 unsigned Op = IsSigned ? Instruction::SExt : Instruction::ZExt;
2092 Cost += getCastInstrCost(Op, VecTy, Src, CCH, CostKind, I);
2093 }
2094 Cost += getRISCVInstructionCost(FWCVT, DstLT.second, CostKind);
2095 } else
2096 Cost += getRISCVInstructionCost(FNCVT, DstLT.second, CostKind);
2097 return Cost;
2098 }
2099 }
2100 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
2101}
2102
2103unsigned RISCVTTIImpl::getEstimatedVLFor(VectorType *Ty) const {
2104 if (isa<ScalableVectorType>(Ty)) {
2105 const unsigned EltSize = DL.getTypeSizeInBits(Ty->getElementType());
2106 const unsigned MinSize = DL.getTypeSizeInBits(Ty).getKnownMinValue();
2107 const unsigned VectorBits = *getVScaleForTuning() * RISCV::RVVBitsPerBlock;
2108 return RISCVTargetLowering::computeVLMAX(VectorBits, EltSize, MinSize);
2109 }
2110 return cast<FixedVectorType>(Ty)->getNumElements();
2111}
2112
2115 FastMathFlags FMF,
2117 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2118 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2119
2120 // Skip if scalar size of Ty is bigger than ELEN.
2121 if (Ty->getScalarSizeInBits() > ST->getELen())
2122 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
2123
2124 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2125 if (Ty->getElementType()->isIntegerTy(1)) {
2126 // SelectionDAGBuilder does following transforms:
2127 // vector_reduce_{smin,umax}(<n x i1>) --> vector_reduce_or(<n x i1>)
2128 // vector_reduce_{smax,umin}(<n x i1>) --> vector_reduce_and(<n x i1>)
2129 if (IID == Intrinsic::umax || IID == Intrinsic::smin)
2130 return getArithmeticReductionCost(Instruction::Or, Ty, FMF, CostKind);
2131 else
2132 return getArithmeticReductionCost(Instruction::And, Ty, FMF, CostKind);
2133 }
2134
2135 if (IID == Intrinsic::maximum || IID == Intrinsic::minimum) {
2137 InstructionCost ExtraCost = 0;
2138 switch (IID) {
2139 case Intrinsic::maximum:
2140 if (FMF.noNaNs()) {
2141 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2142 } else {
2143 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMAX_VS,
2144 RISCV::VFMV_F_S};
2145 // Cost of Canonical Nan + branch
2146 // lui a0, 523264
2147 // fmv.w.x fa0, a0
2148 Type *DstTy = Ty->getScalarType();
2149 const unsigned EltTyBits = DstTy->getScalarSizeInBits();
2150 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2151 ExtraCost = 1 +
2152 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2154 getCFInstrCost(Instruction::CondBr, CostKind);
2155 }
2156 break;
2157
2158 case Intrinsic::minimum:
2159 if (FMF.noNaNs()) {
2160 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2161 } else {
2162 Opcodes = {RISCV::VMFNE_VV, RISCV::VCPOP_M, RISCV::VFREDMIN_VS,
2163 RISCV::VFMV_F_S};
2164 // Cost of Canonical Nan + branch
2165 // lui a0, 523264
2166 // fmv.w.x fa0, a0
2167 Type *DstTy = Ty->getScalarType();
2168 const unsigned EltTyBits = DL.getTypeSizeInBits(DstTy);
2169 Type *SrcTy = IntegerType::getIntNTy(DstTy->getContext(), EltTyBits);
2170 ExtraCost = 1 +
2171 getCastInstrCost(Instruction::UIToFP, DstTy, SrcTy,
2173 getCFInstrCost(Instruction::CondBr, CostKind);
2174 }
2175 break;
2176 }
2177 return ExtraCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2178 }
2179
2180 // IR Reduction is composed by one rvv reduction instruction and vmv
2181 unsigned SplitOp;
2183 switch (IID) {
2184 default:
2185 llvm_unreachable("Unsupported intrinsic");
2186 case Intrinsic::smax:
2187 SplitOp = RISCV::VMAX_VV;
2188 Opcodes = {RISCV::VREDMAX_VS, RISCV::VMV_X_S};
2189 break;
2190 case Intrinsic::smin:
2191 SplitOp = RISCV::VMIN_VV;
2192 Opcodes = {RISCV::VREDMIN_VS, RISCV::VMV_X_S};
2193 break;
2194 case Intrinsic::umax:
2195 SplitOp = RISCV::VMAXU_VV;
2196 Opcodes = {RISCV::VREDMAXU_VS, RISCV::VMV_X_S};
2197 break;
2198 case Intrinsic::umin:
2199 SplitOp = RISCV::VMINU_VV;
2200 Opcodes = {RISCV::VREDMINU_VS, RISCV::VMV_X_S};
2201 break;
2202 case Intrinsic::maxnum:
2203 SplitOp = RISCV::VFMAX_VV;
2204 Opcodes = {RISCV::VFREDMAX_VS, RISCV::VFMV_F_S};
2205 break;
2206 case Intrinsic::minnum:
2207 SplitOp = RISCV::VFMIN_VV;
2208 Opcodes = {RISCV::VFREDMIN_VS, RISCV::VFMV_F_S};
2209 break;
2210 }
2211 // Add a cost for data larger than LMUL8
2212 InstructionCost SplitCost =
2213 (LT.first > 1) ? (LT.first - 1) *
2214 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2215 : 0;
2216 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2217}
2218
2221 std::optional<FastMathFlags> FMF,
2223 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2224 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2225
2226 // Skip if scalar size of Ty is bigger than ELEN.
2227 if (Ty->getScalarSizeInBits() > ST->getELen())
2228 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2229
2230 int ISD = TLI->InstructionOpcodeToISD(Opcode);
2231 assert(ISD && "Invalid opcode");
2232
2233 if (ISD != ISD::ADD && ISD != ISD::OR && ISD != ISD::XOR && ISD != ISD::AND &&
2234 ISD != ISD::FADD)
2235 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2236
2237 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2238 Type *ElementTy = Ty->getElementType();
2239 if (ElementTy->isIntegerTy(1)) {
2240 // Example sequences:
2241 // vfirst.m a0, v0
2242 // seqz a0, a0
2243 if (LT.second == MVT::v1i1)
2244 return getRISCVInstructionCost(RISCV::VFIRST_M, LT.second, CostKind) +
2245 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2247
2248 if (ISD == ISD::AND) {
2249 // Example sequences:
2250 // vmand.mm v8, v9, v8 ; needed every time type is split
2251 // vmnot.m v8, v0 ; alias for vmnand
2252 // vcpop.m a0, v8
2253 // seqz a0, a0
2254
2255 // See the discussion: https://github.com/llvm/llvm-project/pull/119160
2256 // For LMUL <= 8, there is no splitting,
2257 // the sequences are vmnot, vcpop and seqz.
2258 // When LMUL > 8 and split = 1,
2259 // the sequences are vmnand, vcpop and seqz.
2260 // When LMUL > 8 and split > 1,
2261 // the sequences are (LT.first-2) * vmand, vmnand, vcpop and seqz.
2262 return ((LT.first > 2) ? (LT.first - 2) : 0) *
2263 getRISCVInstructionCost(RISCV::VMAND_MM, LT.second, CostKind) +
2264 getRISCVInstructionCost(RISCV::VMNAND_MM, LT.second, CostKind) +
2265 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2266 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2268 } else if (ISD == ISD::XOR || ISD == ISD::ADD) {
2269 // Example sequences:
2270 // vsetvli a0, zero, e8, mf8, ta, ma
2271 // vmxor.mm v8, v0, v8 ; needed every time type is split
2272 // vcpop.m a0, v8
2273 // andi a0, a0, 1
2274 return (LT.first - 1) *
2275 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind) +
2276 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) + 1;
2277 } else {
2278 assert(ISD == ISD::OR);
2279 // Example sequences:
2280 // vsetvli a0, zero, e8, mf8, ta, ma
2281 // vmor.mm v8, v9, v8 ; needed every time type is split
2282 // vcpop.m a0, v0
2283 // snez a0, a0
2284 return (LT.first - 1) *
2285 getRISCVInstructionCost(RISCV::VMOR_MM, LT.second, CostKind) +
2286 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind) +
2287 getCmpSelInstrCost(Instruction::ICmp, ElementTy, ElementTy,
2289 }
2290 }
2291
2292 // IR Reduction of or/and is composed by one vmv and one rvv reduction
2293 // instruction, and others is composed by two vmv and one rvv reduction
2294 // instruction
2295 unsigned SplitOp;
2297 switch (ISD) {
2298 case ISD::ADD:
2299 SplitOp = RISCV::VADD_VV;
2300 Opcodes = {RISCV::VMV_S_X, RISCV::VREDSUM_VS, RISCV::VMV_X_S};
2301 break;
2302 case ISD::OR:
2303 SplitOp = RISCV::VOR_VV;
2304 Opcodes = {RISCV::VREDOR_VS, RISCV::VMV_X_S};
2305 break;
2306 case ISD::XOR:
2307 SplitOp = RISCV::VXOR_VV;
2308 Opcodes = {RISCV::VMV_S_X, RISCV::VREDXOR_VS, RISCV::VMV_X_S};
2309 break;
2310 case ISD::AND:
2311 SplitOp = RISCV::VAND_VV;
2312 Opcodes = {RISCV::VREDAND_VS, RISCV::VMV_X_S};
2313 break;
2314 case ISD::FADD:
2315 // We can't promote f16/bf16 fadd reductions.
2316 if ((LT.second.getScalarType() == MVT::f16 && !ST->hasVInstructionsF16()) ||
2317 LT.second.getScalarType() == MVT::bf16)
2318 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
2320 Opcodes.push_back(RISCV::VFMV_S_F);
2321 for (unsigned i = 0; i < LT.first.getValue(); i++)
2322 Opcodes.push_back(RISCV::VFREDOSUM_VS);
2323 Opcodes.push_back(RISCV::VFMV_F_S);
2324 return getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2325 }
2326 SplitOp = RISCV::VFADD_VV;
2327 Opcodes = {RISCV::VFMV_S_F, RISCV::VFREDUSUM_VS, RISCV::VFMV_F_S};
2328 break;
2329 }
2330 // Add a cost for data larger than LMUL8
2331 InstructionCost SplitCost =
2332 (LT.first > 1) ? (LT.first - 1) *
2333 getRISCVInstructionCost(SplitOp, LT.second, CostKind)
2334 : 0;
2335 return SplitCost + getRISCVInstructionCost(Opcodes, LT.second, CostKind);
2336}
2337
2339 unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy,
2340 std::optional<FastMathFlags> FMF, TTI::TargetCostKind CostKind) const {
2341 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2342 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2343 FMF, CostKind);
2344
2345 // Skip if scalar size of ResTy is bigger than ELEN.
2346 if (ResTy->getScalarSizeInBits() > ST->getELen())
2347 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2348 FMF, CostKind);
2349
2350 if (Opcode != Instruction::Add && Opcode != Instruction::FAdd)
2351 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2352 FMF, CostKind);
2353
2354 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2355
2356 if (IsUnsigned && Opcode == Instruction::Add &&
2357 LT.second.isFixedLengthVectorOf(MVT::i1)) {
2358 // Represent vector_reduce_add(ZExt(<n x i1>)) as
2359 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
2360 return LT.first *
2361 getRISCVInstructionCost(RISCV::VCPOP_M, LT.second, CostKind);
2362 }
2363
2364 if (ResTy->getScalarSizeInBits() != 2 * LT.second.getScalarSizeInBits())
2365 return BaseT::getExtendedReductionCost(Opcode, IsUnsigned, ResTy, ValTy,
2366 FMF, CostKind);
2367
2368 return (LT.first - 1) +
2369 getArithmeticReductionCost(Opcode, ValTy, FMF, CostKind);
2370}
2371
2375 assert(OpInfo.isConstant() && "non constant operand?");
2376 if (!isa<VectorType>(Ty))
2377 // FIXME: We need to account for immediate materialization here, but doing
2378 // a decent job requires more knowledge about the immediate than we
2379 // currently have here.
2380 return 0;
2381
2382 if (OpInfo.isUniform())
2383 // vmv.v.i, vmv.v.x, or vfmv.v.f
2384 // We ignore the cost of the scalar constant materialization to be consistent
2385 // with how we treat scalar constants themselves just above.
2386 return 1;
2387
2388 return getConstantPoolLoadCost(Ty, CostKind);
2389}
2390
2392 Align Alignment,
2393 unsigned AddressSpace,
2395 TTI::OperandValueInfo OpInfo,
2396 const Instruction *I) const {
2397 EVT VT = TLI->getValueType(DL, Src, true);
2398 // Type legalization can't handle structs, and load latency isn't handled here
2399 if (VT == MVT::Other ||
2400 (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency))
2401 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2402 CostKind, OpInfo, I);
2403
2405 if (Opcode == Instruction::Store && OpInfo.isConstant())
2406 Cost += getStoreImmCost(Src, OpInfo, CostKind);
2407
2408 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Src);
2409
2410 InstructionCost BaseCost = [&]() {
2411 InstructionCost Cost = LT.first;
2413 return Cost;
2414
2415 // Our actual lowering for the case where a wider legal type is available
2416 // uses the a VL predicated load on the wider type. This is reflected in
2417 // the result of getTypeLegalizationCost, but BasicTTI assumes the
2418 // widened cases are scalarized.
2419 const DataLayout &DL = this->getDataLayout();
2420 if (Src->isVectorTy() && LT.second.isVector() &&
2421 TypeSize::isKnownLT(DL.getTypeStoreSizeInBits(Src),
2422 LT.second.getSizeInBits()))
2423 return Cost;
2424
2425 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
2426 CostKind, OpInfo, I);
2427 }();
2428
2429 // Assume memory ops cost scale with the number of vector registers
2430 // possible accessed by the instruction. Note that BasicTTI already
2431 // handles the LT.first term for us.
2432 if (ST->hasVInstructions() && LT.second.isVector() &&
2434 BaseCost *= TLI->getLMULCost(LT.second);
2435 return Cost + BaseCost;
2436}
2437
2439 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
2441 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
2443 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2444 Op1Info, Op2Info, I);
2445
2446 if (isa<FixedVectorType>(ValTy) && !ST->useRVVForFixedLengthVectors())
2447 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2448 Op1Info, Op2Info, I);
2449
2450 // Skip if scalar size of ValTy is bigger than ELEN.
2451 if (ValTy->isVectorTy() && ValTy->getScalarSizeInBits() > ST->getELen())
2452 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2453 Op1Info, Op2Info, I);
2454
2455 auto GetConstantMatCost =
2456 [&](TTI::OperandValueInfo OpInfo) -> InstructionCost {
2457 if (OpInfo.isUniform())
2458 // We return 0 we currently ignore the cost of materializing scalar
2459 // constants in GPRs.
2460 return 0;
2461
2462 return getConstantPoolLoadCost(ValTy, CostKind);
2463 };
2464
2465 InstructionCost ConstantMatCost;
2466 if (Op1Info.isConstant())
2467 ConstantMatCost += GetConstantMatCost(Op1Info);
2468 if (Op2Info.isConstant())
2469 ConstantMatCost += GetConstantMatCost(Op2Info);
2470
2471 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
2472 if (Opcode == Instruction::Select && LT.second.isVector()) {
2473 if (CondTy->isVectorTy()) {
2474 if (ValTy->getScalarSizeInBits() == 1) {
2475 // vmandn.mm v8, v8, v9
2476 // vmand.mm v9, v0, v9
2477 // vmor.mm v0, v9, v8
2478 return ConstantMatCost +
2479 LT.first *
2480 getRISCVInstructionCost(
2481 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2482 LT.second, CostKind);
2483 }
2484 // vselect and max/min are supported natively.
2485 return ConstantMatCost +
2486 LT.first * getRISCVInstructionCost(RISCV::VMERGE_VVM, LT.second,
2487 CostKind);
2488 }
2489
2490 if (ValTy->getScalarSizeInBits() == 1) {
2491 // vmv.v.x v9, a0
2492 // vmsne.vi v9, v9, 0
2493 // vmandn.mm v8, v8, v9
2494 // vmand.mm v9, v0, v9
2495 // vmor.mm v0, v9, v8
2496 MVT InterimVT = LT.second.changeVectorElementType(MVT::i8);
2497 return ConstantMatCost +
2498 LT.first *
2499 getRISCVInstructionCost({RISCV::VMV_V_X, RISCV::VMSNE_VI},
2500 InterimVT, CostKind) +
2501 LT.first * getRISCVInstructionCost(
2502 {RISCV::VMANDN_MM, RISCV::VMAND_MM, RISCV::VMOR_MM},
2503 LT.second, CostKind);
2504 }
2505
2506 // vmv.v.x v10, a0
2507 // vmsne.vi v0, v10, 0
2508 // vmerge.vvm v8, v9, v8, v0
2509 return ConstantMatCost +
2510 LT.first * getRISCVInstructionCost(
2511 {RISCV::VMV_V_X, RISCV::VMSNE_VI, RISCV::VMERGE_VVM},
2512 LT.second, CostKind);
2513 }
2514
2515 if ((Opcode == Instruction::ICmp) && ValTy->isVectorTy() &&
2516 CmpInst::isIntPredicate(VecPred)) {
2517 // Use VMSLT_VV to represent VMSEQ, VMSNE, VMSLTU, VMSLEU, VMSLT, VMSLE
2518 // provided they incur the same cost across all implementations
2519 return ConstantMatCost + LT.first * getRISCVInstructionCost(RISCV::VMSLT_VV,
2520 LT.second,
2521 CostKind);
2522 }
2523
2524 if ((Opcode == Instruction::FCmp) && ValTy->isVectorTy() &&
2525 CmpInst::isFPPredicate(VecPred)) {
2526
2527 // Use VMXOR_MM and VMXNOR_MM to generate all true/false mask
2528 if ((VecPred == CmpInst::FCMP_FALSE) || (VecPred == CmpInst::FCMP_TRUE))
2529 return ConstantMatCost +
2530 getRISCVInstructionCost(RISCV::VMXOR_MM, LT.second, CostKind);
2531
2532 // If we do not support the input floating point vector type, use the base
2533 // one which will calculate as:
2534 // ScalarizeCost + Num * Cost for fixed vector,
2535 // InvalidCost for scalable vector.
2536 if ((ValTy->getScalarSizeInBits() == 16 && !ST->hasVInstructionsF16()) ||
2537 (ValTy->getScalarSizeInBits() == 32 && !ST->hasVInstructionsF32()) ||
2538 (ValTy->getScalarSizeInBits() == 64 && !ST->hasVInstructionsF64()))
2539 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2540 Op1Info, Op2Info, I);
2541
2542 // Assuming vector fp compare and mask instructions are all the same cost
2543 // until a need arises to differentiate them.
2544 switch (VecPred) {
2545 case CmpInst::FCMP_ONE: // vmflt.vv + vmflt.vv + vmor.mm
2546 case CmpInst::FCMP_ORD: // vmfeq.vv + vmfeq.vv + vmand.mm
2547 case CmpInst::FCMP_UNO: // vmfne.vv + vmfne.vv + vmor.mm
2548 case CmpInst::FCMP_UEQ: // vmflt.vv + vmflt.vv + vmnor.mm
2549 return ConstantMatCost +
2550 LT.first * getRISCVInstructionCost(
2551 {RISCV::VMFLT_VV, RISCV::VMFLT_VV, RISCV::VMOR_MM},
2552 LT.second, CostKind);
2553
2554 case CmpInst::FCMP_UGT: // vmfle.vv + vmnot.m
2555 case CmpInst::FCMP_UGE: // vmflt.vv + vmnot.m
2556 case CmpInst::FCMP_ULT: // vmfle.vv + vmnot.m
2557 case CmpInst::FCMP_ULE: // vmflt.vv + vmnot.m
2558 return ConstantMatCost +
2559 LT.first *
2560 getRISCVInstructionCost({RISCV::VMFLT_VV, RISCV::VMNAND_MM},
2561 LT.second, CostKind);
2562
2563 case CmpInst::FCMP_OEQ: // vmfeq.vv
2564 case CmpInst::FCMP_OGT: // vmflt.vv
2565 case CmpInst::FCMP_OGE: // vmfle.vv
2566 case CmpInst::FCMP_OLT: // vmflt.vv
2567 case CmpInst::FCMP_OLE: // vmfle.vv
2568 case CmpInst::FCMP_UNE: // vmfne.vv
2569 return ConstantMatCost +
2570 LT.first *
2571 getRISCVInstructionCost(RISCV::VMFLT_VV, LT.second, CostKind);
2572 default:
2573 break;
2574 }
2575 }
2576
2577 // With ShortForwardBranchOpt or ConditionalMoveFusion, scalar icmp + select
2578 // instructions will lower to SELECT_CC and lower to PseudoCCMOVGPR which will
2579 // generate a conditional branch + mv. The cost of scalar (icmp + select) will
2580 // be (0 + select instr cost).
2581 if (ST->hasConditionalMoveFusion() && I && isa<ICmpInst>(I) &&
2582 ValTy->isIntegerTy() && !I->user_empty()) {
2583 if (all_of(I->users(), [&](const User *U) {
2584 return match(U, m_Select(m_Specific(I), m_Value(), m_Value())) &&
2585 U->getType()->isIntegerTy() &&
2586 !isa<ConstantData>(U->getOperand(1)) &&
2587 !isa<ConstantData>(U->getOperand(2));
2588 }))
2589 return 0;
2590 }
2591
2592 // TODO: Add cost for scalar type.
2593
2594 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
2595 Op1Info, Op2Info, I);
2596}
2597
2600 const Instruction *I) const {
2602 return Opcode == Instruction::PHI ? 0 : 1;
2603 // Branches are assumed to be predicted.
2604 return 0;
2605}
2606
2608 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
2609 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
2610 assert(Val->isVectorTy() && "This must be a vector type");
2611
2612 // TODO: Add proper cost model for P extension fixed vectors (e.g., v4i16)
2613 // For now, skip all fixed vector cost analysis when P extension is available
2614 // to avoid crashes in getMinRVVVectorSizeInBits()
2615 if (ST->hasStdExtP() && isa<FixedVectorType>(Val)) {
2616 return 1; // Treat as single instruction cost for now
2617 }
2618
2619 if (Opcode != Instruction::ExtractElement &&
2620 Opcode != Instruction::InsertElement)
2621 return BaseT::getVectorInstrCost(Opcode, Val, CostKind, Index, Op0, Op1,
2622 VIC);
2623
2624 // Legalize the type.
2625 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Val);
2626
2627 // This type is legalized to a scalar type.
2628 if (!LT.second.isVector()) {
2629 auto *FixedVecTy = cast<FixedVectorType>(Val);
2630 // If Index is a known constant, cost is zero.
2631 if (Index != -1U)
2632 return 0;
2633 // Extract/InsertElement with non-constant index is very costly when
2634 // scalarized; estimate cost of loads/stores sequence via the stack:
2635 // ExtractElement cost: store vector to stack, load scalar;
2636 // InsertElement cost: store vector to stack, store scalar, load vector.
2637 Type *ElemTy = FixedVecTy->getElementType();
2638 auto NumElems = FixedVecTy->getNumElements();
2639 auto Align = DL.getPrefTypeAlign(ElemTy);
2640 InstructionCost LoadCost =
2641 getMemoryOpCost(Instruction::Load, ElemTy, Align, 0, CostKind);
2642 InstructionCost StoreCost =
2643 getMemoryOpCost(Instruction::Store, ElemTy, Align, 0, CostKind);
2644 return Opcode == Instruction::ExtractElement
2645 ? StoreCost * NumElems + LoadCost
2646 : (StoreCost + LoadCost) * NumElems + StoreCost;
2647 }
2648
2649 // For unsupported scalable vector.
2650 if (LT.second.isScalableVector() && !LT.first.isValid())
2651 return LT.first;
2652
2653 // Mask vector extract/insert is expanded via e8.
2654 if (Val->getScalarSizeInBits() == 1) {
2655 VectorType *WideTy =
2657 cast<VectorType>(Val)->getElementCount());
2658 if (Opcode == Instruction::ExtractElement) {
2659 InstructionCost ExtendCost
2660 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2662 InstructionCost ExtractCost
2663 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2664 return ExtendCost + ExtractCost;
2665 }
2666 InstructionCost ExtendCost
2667 = getCastInstrCost(Instruction::ZExt, WideTy, Val,
2669 InstructionCost InsertCost
2670 = getVectorInstrCost(Opcode, WideTy, CostKind, Index, nullptr, nullptr);
2671 InstructionCost TruncCost
2672 = getCastInstrCost(Instruction::Trunc, Val, WideTy,
2674 return ExtendCost + InsertCost + TruncCost;
2675 }
2676
2677
2678 // In RVV, we could use vslidedown + vmv.x.s to extract element from vector
2679 // and vslideup + vmv.s.x to insert element to vector.
2680 unsigned MoveOpc;
2681 if (LT.second.isFloatingPoint())
2682 MoveOpc = Opcode == Instruction::InsertElement ? RISCV::VFMV_S_F
2683 : RISCV::VFMV_F_S;
2684 else
2685 MoveOpc =
2686 Opcode == Instruction::InsertElement ? RISCV::VMV_S_X : RISCV::VMV_X_S;
2687 InstructionCost BaseCost =
2688 getRISCVInstructionCost(MoveOpc, LT.second, CostKind);
2689 // When insertelement we should add the index with 1 as the input of vslideup.
2690 InstructionCost SlideCost = Opcode == Instruction::InsertElement ? 2 : 1;
2691
2692 if (Index != -1U) {
2693 // The type may be split. For fixed-width vectors we can normalize the
2694 // index to the new type.
2695 if (LT.second.isFixedLengthVector()) {
2696 unsigned Width = LT.second.getVectorNumElements();
2697 Index = Index % Width;
2698 }
2699
2700 // If exact VLEN is known, we will insert/extract into the appropriate
2701 // subvector with no additional subvector insert/extract cost.
2702 if (auto VLEN = ST->getRealVLen()) {
2703 unsigned EltSize = LT.second.getScalarSizeInBits();
2704 unsigned M1Max = *VLEN / EltSize;
2705 Index = Index % M1Max;
2706 }
2707
2708 if (Index == 0)
2709 // We can extract/insert the first element without vslidedown/vslideup.
2710 SlideCost = 0;
2711 else if (Opcode == Instruction::InsertElement)
2712 SlideCost = 1; // With a constant index, we do not need to use addi.
2713 }
2714
2715 // When the vector needs to split into multiple register groups and the index
2716 // exceeds single vector register group, we need to insert/extract the element
2717 // via stack.
2718 if (LT.first > 1 &&
2719 ((Index == -1U) || (Index >= LT.second.getVectorMinNumElements() &&
2720 LT.second.isScalableVector()))) {
2721 Type *ScalarType = Val->getScalarType();
2722 Align VecAlign = DL.getPrefTypeAlign(Val);
2723 Align SclAlign = DL.getPrefTypeAlign(ScalarType);
2724 // Extra addi for unknown index.
2725 InstructionCost IdxCost = Index == -1U ? 1 : 0;
2726
2727 // Store all split vectors into stack and load the target element.
2728 if (Opcode == Instruction::ExtractElement)
2729 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2730 getMemoryOpCost(Instruction::Load, ScalarType, SclAlign, 0,
2731 CostKind) +
2732 IdxCost;
2733
2734 // Store all split vectors into stack and store the target element and load
2735 // vectors back.
2736 return getMemoryOpCost(Instruction::Store, Val, VecAlign, 0, CostKind) +
2737 getMemoryOpCost(Instruction::Load, Val, VecAlign, 0, CostKind) +
2738 getMemoryOpCost(Instruction::Store, ScalarType, SclAlign, 0,
2739 CostKind) +
2740 IdxCost;
2741 }
2742
2743 // Extract i64 in the target that has XLEN=32 need more instruction.
2744 if (Val->getScalarType()->isIntegerTy() &&
2745 ST->getXLen() < Val->getScalarSizeInBits()) {
2746 // For extractelement, we need the following instructions:
2747 // vsetivli zero, 1, e64, m1, ta, mu (not count)
2748 // vslidedown.vx v8, v8, a0
2749 // vmv.x.s a0, v8
2750 // li a1, 32
2751 // vsrl.vx v8, v8, a1
2752 // vmv.x.s a1, v8
2753
2754 // For insertelement, we need the following instructions:
2755 // vsetivli zero, 2, e32, m4, ta, ma (don't count)
2756 // vslide1down.vx v12, v8, a0
2757 // vslide1down.vx v12, v12, a1
2758 // addi a0, a2, 1
2759 // vsetvli zero, a0, e64, m4, tu, ma (don't count)
2760 // vslideup.vx v8, v12, a2
2761
2762 // TODO: should we count these special vsetvlis?
2763 BaseCost =
2764 Opcode == Instruction::InsertElement
2765 ? getRISCVInstructionCost({RISCV::VSLIDE1DOWN_VX,
2766 RISCV::VSLIDE1DOWN_VX,
2767 RISCV::VSLIDEUP_VX},
2768 LT.second, CostKind)
2769 : getRISCVInstructionCost({RISCV::VSLIDEDOWN_VX, RISCV::VMV_X_S,
2770 RISCV::VSRL_VX, RISCV::VMV_X_S},
2771 LT.second, CostKind);
2772 }
2773 return BaseCost + SlideCost;
2774}
2775
2779 unsigned Index) const {
2780 if (isa<FixedVectorType>(Val))
2782 Index);
2783
2784 // TODO: This code replicates what LoopVectorize.cpp used to do when asking
2785 // for the cost of extracting the last lane of a scalable vector. It probably
2786 // needs a more accurate cost.
2787 ElementCount EC = cast<VectorType>(Val)->getElementCount();
2788 assert(Index < EC.getKnownMinValue() && "Unexpected reverse index");
2789 return getVectorInstrCost(Opcode, Val, CostKind,
2790 EC.getKnownMinValue() - 1 - Index, nullptr,
2791 nullptr);
2792}
2793
2794/// Check to see if this instruction is expected to be combined to a simpler
2795/// operation during/before lowering. If so return the cost of the combined
2796/// operation rather than provided one. For instance, `udiv i16 %X, 2` is likely
2797/// to be combined to `lshr i16 %X, 1`, so return the cost of a `lshr` rather
2798/// than the cost of a `udiv`
2799std::optional<InstructionCost>
2801 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2803 ArrayRef<const Value *> Args, const Instruction *CxtI) const {
2804 // Vector unsigned division/remainder will be simplified to shifts/masks.
2805 if ((Opcode == Instruction::UDiv || Opcode == Instruction::URem) &&
2806 Opd2Info.isConstant() && Opd2Info.isPowerOf2()) {
2807 if (Opcode == Instruction::UDiv)
2808 return getArithmeticInstrCost(Instruction::LShr, Ty, CostKind, Opd1Info,
2809 Opd2Info.getNoProps());
2810 // UREM
2811 return getArithmeticInstrCost(Instruction::And, Ty, CostKind, Opd1Info,
2812 Opd2Info.getNoProps());
2813 }
2814 return std::nullopt;
2815}
2816
2818 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
2820 ArrayRef<const Value *> Args, const Instruction *CxtI) const {
2821
2822 // TODO: Handle more cost kinds.
2824 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2825 Args, CxtI);
2826
2827 if (isa<FixedVectorType>(Ty) && !ST->useRVVForFixedLengthVectors())
2828 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2829 Args, CxtI);
2830
2831 // Skip if scalar size of Ty is bigger than ELEN.
2832 if (isa<VectorType>(Ty) && Ty->getScalarSizeInBits() > ST->getELen())
2833 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2834 Args, CxtI);
2835
2836 if (std::optional<InstructionCost> CombinedCost =
2838 Op2Info, Args, CxtI))
2839 return *CombinedCost;
2840
2841 // Legalize the type.
2842 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
2843 unsigned ISDOpcode = TLI->InstructionOpcodeToISD(Opcode);
2844
2845 // TODO: Handle scalar type.
2846 if (!LT.second.isVector()) {
2847 static const CostTblEntry DivTbl[]{
2848 {ISD::UDIV, MVT::i32, TTI::TCC_Expensive},
2849 {ISD::UDIV, MVT::i64, TTI::TCC_Expensive},
2850 {ISD::SDIV, MVT::i32, TTI::TCC_Expensive},
2851 {ISD::SDIV, MVT::i64, TTI::TCC_Expensive},
2852 {ISD::UREM, MVT::i32, TTI::TCC_Expensive},
2853 {ISD::UREM, MVT::i64, TTI::TCC_Expensive},
2854 {ISD::SREM, MVT::i32, TTI::TCC_Expensive},
2855 {ISD::SREM, MVT::i64, TTI::TCC_Expensive}};
2856 if (TLI->isOperationLegalOrPromote(ISDOpcode, LT.second))
2857 if (const auto *Entry = CostTableLookup(DivTbl, ISDOpcode, LT.second))
2858 return Entry->Cost * LT.first;
2859
2860 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2861 Args, CxtI);
2862 }
2863
2864 // f16 with zvfhmin and bf16 will be promoted to f32.
2865 // FIXME: nxv32[b]f16 will be custom lowered and split.
2866 InstructionCost CastCost = 0;
2867 if ((LT.second.getVectorElementType() == MVT::f16 ||
2868 LT.second.getVectorElementType() == MVT::bf16) &&
2869 TLI->getOperationAction(ISDOpcode, LT.second) ==
2871 MVT PromotedVT = TLI->getTypeToPromoteTo(ISDOpcode, LT.second);
2872 Type *PromotedTy = EVT(PromotedVT).getTypeForEVT(Ty->getContext());
2873 Type *LegalTy = EVT(LT.second).getTypeForEVT(Ty->getContext());
2874 // Add cost of extending arguments
2875 CastCost += LT.first * Args.size() *
2876 getCastInstrCost(Instruction::FPExt, PromotedTy, LegalTy,
2878 // Add cost of truncating result
2879 CastCost +=
2880 LT.first * getCastInstrCost(Instruction::FPTrunc, LegalTy, PromotedTy,
2882 // Compute cost of op in promoted type
2883 LT.second = PromotedVT;
2884 }
2885
2886 auto getConstantMatCost =
2887 [&](unsigned Operand, TTI::OperandValueInfo OpInfo) -> InstructionCost {
2888 if (OpInfo.isUniform() && canSplatOperand(Opcode, Operand))
2889 // Two sub-cases:
2890 // * Has a 5 bit immediate operand which can be splatted.
2891 // * Has a larger immediate which must be materialized in scalar register
2892 // We return 0 for both as we currently ignore the cost of materializing
2893 // scalar constants in GPRs.
2894 return 0;
2895
2896 return getConstantPoolLoadCost(Ty, CostKind);
2897 };
2898
2899 // Add the cost of materializing any constant vectors required.
2900 InstructionCost ConstantMatCost = 0;
2901 if (Op1Info.isConstant())
2902 ConstantMatCost += getConstantMatCost(0, Op1Info);
2903 if (Op2Info.isConstant())
2904 ConstantMatCost += getConstantMatCost(1, Op2Info);
2905
2906 unsigned Op;
2907 switch (ISDOpcode) {
2908 case ISD::ADD:
2909 case ISD::SUB:
2910 Op = RISCV::VADD_VV;
2911 break;
2912 case ISD::SHL:
2913 case ISD::SRL:
2914 case ISD::SRA:
2915 Op = RISCV::VSLL_VV;
2916 break;
2917 case ISD::AND:
2918 case ISD::OR:
2919 case ISD::XOR:
2920 Op = (Ty->getScalarSizeInBits() == 1) ? RISCV::VMAND_MM : RISCV::VAND_VV;
2921 break;
2922 case ISD::MUL:
2923 case ISD::MULHS:
2924 case ISD::MULHU:
2925 Op = RISCV::VMUL_VV;
2926 break;
2927 case ISD::SDIV:
2928 case ISD::UDIV:
2929 Op = RISCV::VDIV_VV;
2930 break;
2931 case ISD::SREM:
2932 case ISD::UREM:
2933 Op = RISCV::VREM_VV;
2934 break;
2935 case ISD::FADD:
2936 case ISD::FSUB:
2937 Op = RISCV::VFADD_VV;
2938 break;
2939 case ISD::FMUL:
2940 Op = RISCV::VFMUL_VV;
2941 break;
2942 case ISD::FDIV:
2943 Op = RISCV::VFDIV_VV;
2944 break;
2945 case ISD::FNEG:
2946 Op = RISCV::VFSGNJN_VV;
2947 break;
2948 default:
2949 // Assuming all other instructions have the same cost until a need arises to
2950 // differentiate them.
2951 return CastCost + ConstantMatCost +
2952 BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
2953 Args, CxtI);
2954 }
2955
2956 InstructionCost InstrCost = getRISCVInstructionCost(Op, LT.second, CostKind);
2957 // We use BasicTTIImpl to calculate scalar costs, which assumes floating point
2958 // ops are twice as expensive as integer ops. Do the same for vectors so
2959 // scalar floating point ops aren't cheaper than their vector equivalents.
2960 if (Ty->isFPOrFPVectorTy())
2961 InstrCost *= 2;
2962 return CastCost + ConstantMatCost + LT.first * InstrCost;
2963}
2964
2965// TODO: Deduplicate from TargetTransformInfoImplCRTPBase.
2967 ArrayRef<const Value *> Ptrs, const Value *Base,
2968 const TTI::PointersChainInfo &Info, Type *AccessTy,
2969 const TTI::TargetCostKind CostKind) const {
2971 // In the basic model we take into account GEP instructions only
2972 // (although here can come alloca instruction, a value, constants and/or
2973 // constant expressions, PHIs, bitcasts ... whatever allowed to be used as a
2974 // pointer). Typically, if Base is a not a GEP-instruction and all the
2975 // pointers are relative to the same base address, all the rest are
2976 // either GEP instructions, PHIs, bitcasts or constants. When we have same
2977 // base, we just calculate cost of each non-Base GEP as an ADD operation if
2978 // any their index is a non-const.
2979 // If no known dependencies between the pointers cost is calculated as a sum
2980 // of costs of GEP instructions.
2981 for (auto [I, V] : enumerate(Ptrs)) {
2982 const auto *GEP = dyn_cast<GetElementPtrInst>(V);
2983 if (!GEP)
2984 continue;
2985 if (Info.isSameBase() && V != Base) {
2986 if (GEP->hasAllConstantIndices())
2987 continue;
2988 // If the chain is unit-stride and BaseReg + stride*i is a legal
2989 // addressing mode, then presume the base GEP is sitting around in a
2990 // register somewhere and check if we can fold the offset relative to
2991 // it.
2992 unsigned Stride = DL.getTypeStoreSize(AccessTy);
2993 if (Info.isUnitStride() &&
2994 isLegalAddressingMode(AccessTy,
2995 /* BaseGV */ nullptr,
2996 /* BaseOffset */ Stride * I,
2997 /* HasBaseReg */ true,
2998 /* Scale */ 0,
2999 GEP->getType()->getPointerAddressSpace()))
3000 continue;
3001 Cost += getArithmeticInstrCost(Instruction::Add, GEP->getType(), CostKind,
3002 {TTI::OK_AnyValue, TTI::OP_None},
3003 {TTI::OK_AnyValue, TTI::OP_None}, {});
3004 } else {
3005 SmallVector<const Value *> Indices(GEP->indices());
3006 Cost += getGEPCost(GEP->getSourceElementType(), GEP->getPointerOperand(),
3007 Indices, CostKind, AccessTy);
3008 }
3009 }
3010 return Cost;
3011}
3012
3015 OptimizationRemarkEmitter *ORE) const {
3016 // TODO: More tuning on benchmarks and metrics with changes as needed
3017 // would apply to all settings below to enable performance.
3018
3019
3020 if (ST->enableDefaultUnroll())
3021 return BasicTTIImplBase::getUnrollingPreferences(L, SE, UP, ORE);
3022
3023 // Enable Upper bound unrolling universally, not dependent upon the conditions
3024 // below.
3025 UP.UpperBound = true;
3026
3027 // Disable loop unrolling for Oz and Os.
3028 UP.OptSizeThreshold = 0;
3030 if (L->getHeader()->getParent()->hasOptSize())
3031 return;
3032
3033 SmallVector<BasicBlock *, 4> ExitingBlocks;
3034 L->getExitingBlocks(ExitingBlocks);
3035 LLVM_DEBUG(dbgs() << "Loop has:\n"
3036 << "Blocks: " << L->getNumBlocks() << "\n"
3037 << "Exit blocks: " << ExitingBlocks.size() << "\n");
3038
3039 // Only allow another exit other than the latch. This acts as an early exit
3040 // as it mirrors the profitability calculation of the runtime unroller.
3041 if (ExitingBlocks.size() > 2)
3042 return;
3043
3044 // Limit the CFG of the loop body for targets with a branch predictor.
3045 // Allowing 4 blocks permits if-then-else diamonds in the body.
3046 if (L->getNumBlocks() > 4)
3047 return;
3048
3049 // Scan the loop: don't unroll loops with calls as this could prevent
3050 // inlining. Don't unroll auto-vectorized loops either, though do allow
3051 // unrolling of the scalar remainder.
3052 bool IsVectorized = getBooleanLoopAttribute(L, "llvm.loop.isvectorized");
3054 for (auto *BB : L->getBlocks()) {
3055 for (auto &I : *BB) {
3056 // Both auto-vectorized loops and the scalar remainder have the
3057 // isvectorized attribute, so differentiate between them by the presence
3058 // of vector instructions.
3059 if (IsVectorized && (I.getType()->isVectorTy() ||
3060 llvm::any_of(I.operand_values(), [](Value *V) {
3061 return V->getType()->isVectorTy();
3062 })))
3063 return;
3064
3065 if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
3066 const Function *F = cast<CallBase>(I).getCalledFunction();
3067 if (!F || isLoweredToCall(F))
3068 return;
3069 }
3070
3071 SmallVector<const Value *> Operands(I.operand_values());
3074 }
3075 }
3076
3077 LLVM_DEBUG(dbgs() << "Cost of loop: " << Cost << "\n");
3078
3079 UP.Partial = true;
3080 UP.Runtime = true;
3081 UP.UnrollRemainder = true;
3082 UP.UnrollAndJam = true;
3083
3084 // Force unrolling small loops can be very useful because of the branch
3085 // taken cost of the backedge.
3086 if (Cost < 12)
3087 UP.Force = true;
3088}
3089
3094
3096 MemIntrinsicInfo &Info) const {
3097 const DataLayout &DL = getDataLayout();
3098 Intrinsic::ID IID = Inst->getIntrinsicID();
3099 LLVMContext &C = Inst->getContext();
3100 bool HasMask = false;
3101
3102 auto getSegNum = [](const IntrinsicInst *II, unsigned PtrOperandNo,
3103 bool IsWrite) -> int64_t {
3104 if (auto *TarExtTy =
3105 dyn_cast<TargetExtType>(II->getArgOperand(0)->getType()))
3106 return TarExtTy->getIntParameter(0);
3107
3108 return 1;
3109 };
3110
3111 switch (IID) {
3112 case Intrinsic::riscv_vle_mask:
3113 case Intrinsic::riscv_vse_mask:
3114 case Intrinsic::riscv_vlseg2_mask:
3115 case Intrinsic::riscv_vlseg3_mask:
3116 case Intrinsic::riscv_vlseg4_mask:
3117 case Intrinsic::riscv_vlseg5_mask:
3118 case Intrinsic::riscv_vlseg6_mask:
3119 case Intrinsic::riscv_vlseg7_mask:
3120 case Intrinsic::riscv_vlseg8_mask:
3121 case Intrinsic::riscv_vsseg2_mask:
3122 case Intrinsic::riscv_vsseg3_mask:
3123 case Intrinsic::riscv_vsseg4_mask:
3124 case Intrinsic::riscv_vsseg5_mask:
3125 case Intrinsic::riscv_vsseg6_mask:
3126 case Intrinsic::riscv_vsseg7_mask:
3127 case Intrinsic::riscv_vsseg8_mask:
3128 HasMask = true;
3129 [[fallthrough]];
3130 case Intrinsic::riscv_vle:
3131 case Intrinsic::riscv_vse:
3132 case Intrinsic::riscv_vlseg2:
3133 case Intrinsic::riscv_vlseg3:
3134 case Intrinsic::riscv_vlseg4:
3135 case Intrinsic::riscv_vlseg5:
3136 case Intrinsic::riscv_vlseg6:
3137 case Intrinsic::riscv_vlseg7:
3138 case Intrinsic::riscv_vlseg8:
3139 case Intrinsic::riscv_vsseg2:
3140 case Intrinsic::riscv_vsseg3:
3141 case Intrinsic::riscv_vsseg4:
3142 case Intrinsic::riscv_vsseg5:
3143 case Intrinsic::riscv_vsseg6:
3144 case Intrinsic::riscv_vsseg7:
3145 case Intrinsic::riscv_vsseg8: {
3146 // Intrinsic interface:
3147 // riscv_vle(merge, ptr, vl)
3148 // riscv_vle_mask(merge, ptr, mask, vl, policy)
3149 // riscv_vse(val, ptr, vl)
3150 // riscv_vse_mask(val, ptr, mask, vl, policy)
3151 // riscv_vlseg#(merge, ptr, vl, sew)
3152 // riscv_vlseg#_mask(merge, ptr, mask, vl, policy, sew)
3153 // riscv_vsseg#(val, ptr, vl, sew)
3154 // riscv_vsseg#_mask(val, ptr, mask, vl, sew)
3155 bool IsWrite = Inst->getType()->isVoidTy();
3156 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3157 // The results of segment loads are TargetExtType.
3158 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3159 unsigned SEW =
3160 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3161 ->getZExtValue();
3162 Ty = TarExtTy->getTypeParameter(0U);
3164 IntegerType::get(C, SEW),
3165 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3166 }
3167 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3168 unsigned VLIndex = RVVIInfo->VLOperand;
3169 unsigned PtrOperandNo = VLIndex - 1 - HasMask;
3170 MaybeAlign Alignment =
3171 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3172 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3173 Value *Mask = ConstantInt::getTrue(MaskType);
3174 if (HasMask)
3175 Mask = Inst->getArgOperand(VLIndex - 1);
3176 Value *EVL = Inst->getArgOperand(VLIndex);
3177 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3178 // RVV uses contiguous elements as a segment.
3179 if (SegNum > 1) {
3180 unsigned ElemSize = Ty->getScalarSizeInBits();
3181 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3182 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3183 }
3184 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3185 Alignment, Mask, EVL);
3186 return true;
3187 }
3188 case Intrinsic::riscv_vlse_mask:
3189 case Intrinsic::riscv_vsse_mask:
3190 case Intrinsic::riscv_vlsseg2_mask:
3191 case Intrinsic::riscv_vlsseg3_mask:
3192 case Intrinsic::riscv_vlsseg4_mask:
3193 case Intrinsic::riscv_vlsseg5_mask:
3194 case Intrinsic::riscv_vlsseg6_mask:
3195 case Intrinsic::riscv_vlsseg7_mask:
3196 case Intrinsic::riscv_vlsseg8_mask:
3197 case Intrinsic::riscv_vssseg2_mask:
3198 case Intrinsic::riscv_vssseg3_mask:
3199 case Intrinsic::riscv_vssseg4_mask:
3200 case Intrinsic::riscv_vssseg5_mask:
3201 case Intrinsic::riscv_vssseg6_mask:
3202 case Intrinsic::riscv_vssseg7_mask:
3203 case Intrinsic::riscv_vssseg8_mask:
3204 HasMask = true;
3205 [[fallthrough]];
3206 case Intrinsic::riscv_vlse:
3207 case Intrinsic::riscv_vsse:
3208 case Intrinsic::riscv_vlsseg2:
3209 case Intrinsic::riscv_vlsseg3:
3210 case Intrinsic::riscv_vlsseg4:
3211 case Intrinsic::riscv_vlsseg5:
3212 case Intrinsic::riscv_vlsseg6:
3213 case Intrinsic::riscv_vlsseg7:
3214 case Intrinsic::riscv_vlsseg8:
3215 case Intrinsic::riscv_vssseg2:
3216 case Intrinsic::riscv_vssseg3:
3217 case Intrinsic::riscv_vssseg4:
3218 case Intrinsic::riscv_vssseg5:
3219 case Intrinsic::riscv_vssseg6:
3220 case Intrinsic::riscv_vssseg7:
3221 case Intrinsic::riscv_vssseg8: {
3222 // Intrinsic interface:
3223 // riscv_vlse(merge, ptr, stride, vl)
3224 // riscv_vlse_mask(merge, ptr, stride, mask, vl, policy)
3225 // riscv_vsse(val, ptr, stride, vl)
3226 // riscv_vsse_mask(val, ptr, stride, mask, vl, policy)
3227 // riscv_vlsseg#(merge, ptr, offset, vl, sew)
3228 // riscv_vlsseg#_mask(merge, ptr, offset, mask, vl, policy, sew)
3229 // riscv_vssseg#(val, ptr, offset, vl, sew)
3230 // riscv_vssseg#_mask(val, ptr, offset, mask, vl, sew)
3231 bool IsWrite = Inst->getType()->isVoidTy();
3232 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3233 // The results of segment loads are TargetExtType.
3234 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3235 unsigned SEW =
3236 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3237 ->getZExtValue();
3238 Ty = TarExtTy->getTypeParameter(0U);
3240 IntegerType::get(C, SEW),
3241 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3242 }
3243 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3244 unsigned VLIndex = RVVIInfo->VLOperand;
3245 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3246 MaybeAlign Alignment =
3247 Inst->getArgOperand(PtrOperandNo)->getPointerAlignment(DL);
3248
3249 Value *Stride = Inst->getArgOperand(PtrOperandNo + 1);
3250 // Use the pointer alignment as the element alignment if the stride is a
3251 // multiple of the pointer alignment. Otherwise, the element alignment
3252 // should be the greatest common divisor of pointer alignment and stride.
3253 // For simplicity, just consider unalignment for elements.
3254 unsigned PointerAlign = Alignment.valueOrOne().value();
3255 if (!isa<ConstantInt>(Stride) ||
3256 cast<ConstantInt>(Stride)->getZExtValue() % PointerAlign != 0)
3257 Alignment = Align(1);
3258
3259 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3260 Value *Mask = ConstantInt::getTrue(MaskType);
3261 if (HasMask)
3262 Mask = Inst->getArgOperand(VLIndex - 1);
3263 Value *EVL = Inst->getArgOperand(VLIndex);
3264 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3265 // RVV uses contiguous elements as a segment.
3266 if (SegNum > 1) {
3267 unsigned ElemSize = Ty->getScalarSizeInBits();
3268 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3269 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3270 }
3271 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3272 Alignment, Mask, EVL, Stride);
3273 return true;
3274 }
3275 case Intrinsic::riscv_vloxei_mask:
3276 case Intrinsic::riscv_vluxei_mask:
3277 case Intrinsic::riscv_vsoxei_mask:
3278 case Intrinsic::riscv_vsuxei_mask:
3279 case Intrinsic::riscv_vloxseg2_mask:
3280 case Intrinsic::riscv_vloxseg3_mask:
3281 case Intrinsic::riscv_vloxseg4_mask:
3282 case Intrinsic::riscv_vloxseg5_mask:
3283 case Intrinsic::riscv_vloxseg6_mask:
3284 case Intrinsic::riscv_vloxseg7_mask:
3285 case Intrinsic::riscv_vloxseg8_mask:
3286 case Intrinsic::riscv_vluxseg2_mask:
3287 case Intrinsic::riscv_vluxseg3_mask:
3288 case Intrinsic::riscv_vluxseg4_mask:
3289 case Intrinsic::riscv_vluxseg5_mask:
3290 case Intrinsic::riscv_vluxseg6_mask:
3291 case Intrinsic::riscv_vluxseg7_mask:
3292 case Intrinsic::riscv_vluxseg8_mask:
3293 case Intrinsic::riscv_vsoxseg2_mask:
3294 case Intrinsic::riscv_vsoxseg3_mask:
3295 case Intrinsic::riscv_vsoxseg4_mask:
3296 case Intrinsic::riscv_vsoxseg5_mask:
3297 case Intrinsic::riscv_vsoxseg6_mask:
3298 case Intrinsic::riscv_vsoxseg7_mask:
3299 case Intrinsic::riscv_vsoxseg8_mask:
3300 case Intrinsic::riscv_vsuxseg2_mask:
3301 case Intrinsic::riscv_vsuxseg3_mask:
3302 case Intrinsic::riscv_vsuxseg4_mask:
3303 case Intrinsic::riscv_vsuxseg5_mask:
3304 case Intrinsic::riscv_vsuxseg6_mask:
3305 case Intrinsic::riscv_vsuxseg7_mask:
3306 case Intrinsic::riscv_vsuxseg8_mask:
3307 HasMask = true;
3308 [[fallthrough]];
3309 case Intrinsic::riscv_vloxei:
3310 case Intrinsic::riscv_vluxei:
3311 case Intrinsic::riscv_vsoxei:
3312 case Intrinsic::riscv_vsuxei:
3313 case Intrinsic::riscv_vloxseg2:
3314 case Intrinsic::riscv_vloxseg3:
3315 case Intrinsic::riscv_vloxseg4:
3316 case Intrinsic::riscv_vloxseg5:
3317 case Intrinsic::riscv_vloxseg6:
3318 case Intrinsic::riscv_vloxseg7:
3319 case Intrinsic::riscv_vloxseg8:
3320 case Intrinsic::riscv_vluxseg2:
3321 case Intrinsic::riscv_vluxseg3:
3322 case Intrinsic::riscv_vluxseg4:
3323 case Intrinsic::riscv_vluxseg5:
3324 case Intrinsic::riscv_vluxseg6:
3325 case Intrinsic::riscv_vluxseg7:
3326 case Intrinsic::riscv_vluxseg8:
3327 case Intrinsic::riscv_vsoxseg2:
3328 case Intrinsic::riscv_vsoxseg3:
3329 case Intrinsic::riscv_vsoxseg4:
3330 case Intrinsic::riscv_vsoxseg5:
3331 case Intrinsic::riscv_vsoxseg6:
3332 case Intrinsic::riscv_vsoxseg7:
3333 case Intrinsic::riscv_vsoxseg8:
3334 case Intrinsic::riscv_vsuxseg2:
3335 case Intrinsic::riscv_vsuxseg3:
3336 case Intrinsic::riscv_vsuxseg4:
3337 case Intrinsic::riscv_vsuxseg5:
3338 case Intrinsic::riscv_vsuxseg6:
3339 case Intrinsic::riscv_vsuxseg7:
3340 case Intrinsic::riscv_vsuxseg8: {
3341 // Intrinsic interface (only listed ordered version):
3342 // riscv_vloxei(merge, ptr, index, vl)
3343 // riscv_vloxei_mask(merge, ptr, index, mask, vl, policy)
3344 // riscv_vsoxei(val, ptr, index, vl)
3345 // riscv_vsoxei_mask(val, ptr, index, mask, vl, policy)
3346 // riscv_vloxseg#(merge, ptr, index, vl, sew)
3347 // riscv_vloxseg#_mask(merge, ptr, index, mask, vl, policy, sew)
3348 // riscv_vsoxseg#(val, ptr, index, vl, sew)
3349 // riscv_vsoxseg#_mask(val, ptr, index, mask, vl, sew)
3350 bool IsWrite = Inst->getType()->isVoidTy();
3351 Type *Ty = IsWrite ? Inst->getArgOperand(0)->getType() : Inst->getType();
3352 // The results of segment loads are TargetExtType.
3353 if (auto *TarExtTy = dyn_cast<TargetExtType>(Ty)) {
3354 unsigned SEW =
3355 1 << cast<ConstantInt>(Inst->getArgOperand(Inst->arg_size() - 1))
3356 ->getZExtValue();
3357 Ty = TarExtTy->getTypeParameter(0U);
3359 IntegerType::get(C, SEW),
3360 cast<ScalableVectorType>(Ty)->getMinNumElements() * 8 / SEW);
3361 }
3362 const auto *RVVIInfo = RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IID);
3363 unsigned VLIndex = RVVIInfo->VLOperand;
3364 unsigned PtrOperandNo = VLIndex - 2 - HasMask;
3365 Value *Mask;
3366 if (HasMask) {
3367 Mask = Inst->getArgOperand(VLIndex - 1);
3368 } else {
3369 // Mask cannot be nullptr here: vector GEP produces <vscale x N x ptr>,
3370 // and casting that to scalar i64 triggers a vector/scalar mismatch
3371 // assertion in CreatePointerCast. Use an all-true mask so ASan lowers it
3372 // via extractelement instead.
3373 Type *MaskType = Ty->getWithNewType(Type::getInt1Ty(C));
3374 Mask = ConstantInt::getTrue(MaskType);
3375 }
3376 Value *EVL = Inst->getArgOperand(VLIndex);
3377 unsigned SegNum = getSegNum(Inst, PtrOperandNo, IsWrite);
3378 // RVV uses contiguous elements as a segment.
3379 if (SegNum > 1) {
3380 unsigned ElemSize = Ty->getScalarSizeInBits();
3381 auto *SegTy = IntegerType::get(C, ElemSize * SegNum);
3382 Ty = VectorType::get(SegTy, cast<VectorType>(Ty));
3383 }
3384 Value *OffsetOp = Inst->getArgOperand(PtrOperandNo + 1);
3385 Info.InterestingOperands.emplace_back(Inst, PtrOperandNo, IsWrite, Ty,
3386 Align(1), Mask, EVL,
3387 /* Stride */ nullptr, OffsetOp);
3388 return true;
3389 }
3390 }
3391 return false;
3392}
3393
3395 if (Ty->isVectorTy()) {
3396 // f16 with only zvfhmin and bf16 will be promoted to f32
3397 Type *EltTy = cast<VectorType>(Ty)->getElementType();
3398 if ((EltTy->isHalfTy() && !ST->hasVInstructionsF16()) ||
3399 EltTy->isBFloatTy())
3400 Ty = VectorType::get(Type::getFloatTy(Ty->getContext()),
3401 cast<VectorType>(Ty));
3402
3403 TypeSize Size = DL.getTypeSizeInBits(Ty);
3404 if (Size.isScalable() && ST->hasVInstructions())
3405 return divideCeil(Size.getKnownMinValue(), RISCV::RVVBitsPerBlock);
3406
3407 if (ST->useRVVForFixedLengthVectors())
3408 return divideCeil(Size, ST->getRealMinVLen());
3409 }
3410
3411 return BaseT::getRegUsageForType(Ty);
3412}
3413
3414unsigned RISCVTTIImpl::getMaximumVF(unsigned ElemWidth, unsigned Opcode) const {
3415 if (SLPMaxVF.getNumOccurrences())
3416 return SLPMaxVF;
3417
3418 // Return how many elements can fit in getRegisterBitwidth. This is the
3419 // same routine as used in LoopVectorizer. We should probably be
3420 // accounting for whether we actually have instructions with the right
3421 // lane type, but we don't have enough information to do that without
3422 // some additional plumbing which hasn't been justified yet.
3423 TypeSize RegWidth =
3425 // If no vector registers, or absurd element widths, disable
3426 // vectorization by returning 1.
3427 return std::max<unsigned>(1U, RegWidth.getFixedValue() / ElemWidth);
3428}
3429
3433
3435 return ST->enableUnalignedVectorMem();
3436}
3437
3440 ScalarEvolution *SE) const {
3441 if (ST->hasVendorXCVmem() && !ST->is64Bit())
3442 return TTI::AMK_PostIndexed;
3443
3445}
3446
3448 const TargetTransformInfo::LSRCost &C2) const {
3449 // RISC-V specific here are "instruction number 1st priority".
3450 // If we need to emit adds inside the loop to add up base registers, then
3451 // we need at least one extra temporary register.
3452 unsigned C1NumRegs = C1.NumRegs + (C1.NumBaseAdds != 0);
3453 unsigned C2NumRegs = C2.NumRegs + (C2.NumBaseAdds != 0);
3454 return std::tie(C1.Insns, C1NumRegs, C1.AddRecCost,
3455 C1.NumIVMuls, C1.NumBaseAdds,
3456 C1.ScaleCost, C1.ImmCost, C1.SetupCost) <
3457 std::tie(C2.Insns, C2NumRegs, C2.AddRecCost,
3458 C2.NumIVMuls, C2.NumBaseAdds,
3459 C2.ScaleCost, C2.ImmCost, C2.SetupCost);
3460}
3461
3463 Align Alignment) const {
3464 auto *VTy = dyn_cast<VectorType>(DataTy);
3465 if (!VTy || VTy->isScalableTy())
3466 return false;
3467
3468 if (!isLegalMaskedLoadStore(DataTy, Alignment))
3469 return false;
3470
3471 // FIXME: If it is an i8 vector and the element count exceeds 256, we should
3472 // scalarize these types with LMUL >= maximum fixed-length LMUL.
3473 if (VTy->getElementType()->isIntegerTy(8))
3474 if (VTy->getElementCount().getFixedValue() > 256)
3475 return VTy->getPrimitiveSizeInBits() / ST->getRealMinVLen() <
3476 ST->getMaxLMULForFixedLengthVectors();
3477 return true;
3478}
3479
3481 Align Alignment) const {
3482 auto *VTy = dyn_cast<VectorType>(DataTy);
3483 if (!VTy || VTy->isScalableTy())
3484 return false;
3485
3486 if (!isLegalMaskedLoadStore(DataTy, Alignment))
3487 return false;
3488 return true;
3489}
3490
3492 ElementCount NumElements) const {
3493 // Optimized zero-stride loads can be treated as broadcasts.
3494 if (!ST->hasVInstructions() || !ST->hasOptimizedZeroStrideLoad())
3495 return false;
3496
3497 return TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, ElementTy));
3498}
3499
3500/// See if \p I should be considered for address type promotion. We check if \p
3501/// I is a sext with right type and used in memory accesses. If it used in a
3502/// "complex" getelementptr, we allow it to be promoted without finding other
3503/// sext instructions that sign extended the same initial value. A getelementptr
3504/// is considered as "complex" if it has more than 2 operands.
3506 const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const {
3507 bool Considerable = false;
3508 AllowPromotionWithoutCommonHeader = false;
3509 if (!isa<SExtInst>(&I))
3510 return false;
3511 Type *ConsideredSExtType =
3512 Type::getInt64Ty(I.getParent()->getParent()->getContext());
3513 if (I.getType() != ConsideredSExtType)
3514 return false;
3515 // See if the sext is the one with the right type and used in at least one
3516 // GetElementPtrInst.
3517 for (const User *U : I.users()) {
3518 if (const GetElementPtrInst *GEPInst = dyn_cast<GetElementPtrInst>(U)) {
3519 Considerable = true;
3520 // A getelementptr is considered as "complex" if it has more than 2
3521 // operands. We will promote a SExt used in such complex GEP as we
3522 // expect some computation to be merged if they are done on 64 bits.
3523 if (GEPInst->getNumOperands() > 2) {
3524 AllowPromotionWithoutCommonHeader = true;
3525 break;
3526 }
3527 }
3528 }
3529 return Considerable;
3530}
3531
3532bool RISCVTTIImpl::canSplatOperand(unsigned Opcode, int Operand) const {
3533 switch (Opcode) {
3534 case Instruction::Add:
3535 case Instruction::Sub:
3536 case Instruction::Mul:
3537 case Instruction::And:
3538 case Instruction::Or:
3539 case Instruction::Xor:
3540 case Instruction::FAdd:
3541 case Instruction::FSub:
3542 case Instruction::FMul:
3543 case Instruction::FDiv:
3544 case Instruction::ICmp:
3545 case Instruction::FCmp:
3546 return true;
3547 case Instruction::Shl:
3548 case Instruction::LShr:
3549 case Instruction::AShr:
3550 case Instruction::UDiv:
3551 case Instruction::SDiv:
3552 case Instruction::URem:
3553 case Instruction::SRem:
3554 case Instruction::Select:
3555 return Operand == 1;
3556 default:
3557 return false;
3558 }
3559}
3560
3562 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3563 return false;
3564
3565 if (canSplatOperand(I->getOpcode(), Operand))
3566 return true;
3567
3568 auto *II = dyn_cast<IntrinsicInst>(I);
3569 if (!II)
3570 return false;
3571
3572 switch (II->getIntrinsicID()) {
3573 case Intrinsic::fma:
3574 case Intrinsic::fmuladd:
3575 return Operand == 0 || Operand == 1;
3576 case Intrinsic::vp_udiv:
3577 case Intrinsic::vp_sdiv:
3578 case Intrinsic::vp_urem:
3579 case Intrinsic::vp_srem:
3580 case Intrinsic::ssub_sat:
3581 case Intrinsic::usub_sat:
3582 return Operand == 1;
3583 // These intrinsics are commutative.
3584 case Intrinsic::smin:
3585 case Intrinsic::umin:
3586 case Intrinsic::smax:
3587 case Intrinsic::umax:
3588 case Intrinsic::sadd_sat:
3589 case Intrinsic::uadd_sat:
3590 return Operand == 0 || Operand == 1;
3591 default:
3592 return false;
3593 }
3594}
3595
3596/// Check if sinking \p I's operands to I's basic block is profitable, because
3597/// the operands can be folded into a target instruction, e.g.
3598/// splats of scalars can fold into vector instructions.
3601 using namespace llvm::PatternMatch;
3602
3603 if (I->isBitwiseLogicOp()) {
3604 if (!I->getType()->isVectorTy()) {
3605 if (ST->hasStdExtZbb() || ST->hasStdExtZbkb()) {
3606 for (auto &Op : I->operands()) {
3607 // (and/or/xor X, (not Y)) -> (andn/orn/xnor X, Y)
3608 if (match(Op.get(), m_Not(m_Value()))) {
3609 Ops.push_back(&Op);
3610 return true;
3611 }
3612 }
3613 }
3614 } else if (I->getOpcode() == Instruction::And && ST->hasStdExtZvkb()) {
3615 for (auto &Op : I->operands()) {
3616 // (and X, (not Y)) -> (vandn.vv X, Y)
3617 if (match(Op.get(), m_Not(m_Value()))) {
3618 Ops.push_back(&Op);
3619 return true;
3620 }
3621 // (and X, (splat (not Y))) -> (vandn.vx X, Y)
3623 m_ZeroInt()),
3624 m_Value(), m_ZeroMask()))) {
3625 Use &InsertElt = cast<Instruction>(Op)->getOperandUse(0);
3626 Use &Not = cast<Instruction>(InsertElt)->getOperandUse(1);
3627 Ops.push_back(&Not);
3628 Ops.push_back(&InsertElt);
3629 Ops.push_back(&Op);
3630 return true;
3631 }
3632 }
3633 }
3634 }
3635
3636 if (!I->getType()->isVectorTy() || !ST->hasVInstructions())
3637 return false;
3638
3639 // Don't sink splat operands if the target prefers it. Some targets requires
3640 // S2V transfer buffers and we can run out of them copying the same value
3641 // repeatedly.
3642 // FIXME: It could still be worth doing if it would improve vector register
3643 // pressure and prevent a vector spill.
3644 if (!ST->sinkSplatOperands())
3645 return false;
3646
3647 for (auto OpIdx : enumerate(I->operands())) {
3648 if (!canSplatOperand(I, OpIdx.index()))
3649 continue;
3650
3651 Instruction *Op = dyn_cast<Instruction>(OpIdx.value().get());
3652 // Make sure we are not already sinking this operand
3653 if (!Op || any_of(Ops, [&](Use *U) { return U->get() == Op; }))
3654 continue;
3655
3656 // We are looking for a splat that can be sunk.
3658 m_Value(), m_ZeroMask())))
3659 continue;
3660
3661 // Don't sink i1 splats.
3662 if (cast<VectorType>(Op->getType())->getElementType()->isIntegerTy(1))
3663 continue;
3664
3665 // All uses of the shuffle should be sunk to avoid duplicating it across gpr
3666 // and vector registers
3667 for (Use &U : Op->uses()) {
3668 Instruction *Insn = cast<Instruction>(U.getUser());
3669 if (!canSplatOperand(Insn, U.getOperandNo()))
3670 return false;
3671 }
3672
3673 // Sink any fpexts since they might be used in a widening fp pattern.
3674 Use *InsertEltUse = &Op->getOperandUse(0);
3675 auto *InsertElt = cast<InsertElementInst>(InsertEltUse);
3676 if (isa<FPExtInst>(InsertElt->getOperand(1)))
3677 Ops.push_back(&InsertElt->getOperandUse(1));
3678 Ops.push_back(InsertEltUse);
3679 Ops.push_back(&OpIdx.value());
3680 }
3681 return true;
3682}
3683
3685RISCVTTIImpl::enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
3687
3688 if (!ST->hasStdExtZbb() && !ST->hasStdExtZbkb() && !IsZeroCmp)
3689 return Options;
3690
3691 Options.AllowOverlappingLoads = true;
3692 Options.MaxNumLoads = TLI->getMaxExpandSizeMemcmp(OptSize);
3693 Options.NumLoadsPerBlock = IsZeroCmp ? Options.MaxNumLoads : 1;
3694 if (ST->is64Bit()) {
3695 Options.LoadSizes = {8, 4, 2, 1};
3696 Options.AllowedTailExpansions = {3, 5, 6};
3697 } else {
3698 Options.LoadSizes = {4, 2, 1};
3699 Options.AllowedTailExpansions = {3};
3700 }
3701
3702 if (IsZeroCmp && ST->hasVInstructions()) {
3703 unsigned VLenB = ST->getRealMinVLen() / 8;
3704 // The minimum size should be `XLen / 8 + 1`, and the maxinum size should be
3705 // `VLenB * MaxLMUL` so that it fits in a single register group.
3706 unsigned MinSize = ST->getXLen() / 8 + 1;
3707 unsigned MaxSize = VLenB * ST->getMaxLMULForFixedLengthVectors();
3708 for (unsigned Size = MinSize; Size <= MaxSize; Size++)
3709 Options.LoadSizes.insert(Options.LoadSizes.begin(), Size);
3710 }
3711 return Options;
3712}
3713
3715 const Instruction *I) const {
3717 // For the binary operators (e.g. or) we need to be more careful than
3718 // selects, here we only transform them if they are already at a natural
3719 // break point in the code - the end of a block with an unconditional
3720 // terminator.
3721 if (I->getOpcode() == Instruction::Or &&
3722 isa<UncondBrInst>(I->getNextNode()))
3723 return true;
3724
3725 if (I->getOpcode() == Instruction::Add ||
3726 I->getOpcode() == Instruction::Sub)
3727 return true;
3728 }
3730}
3731
3733 const Function *Caller, const Attribute &Attr) const {
3734 // "interrupt" controls the prolog/epilog of interrupt handlers (and includes
3735 // restrictions on their signatures). We can outline from the bodies of these
3736 // handlers, but when we do we need to make sure we don't mark the outlined
3737 // function as an interrupt handler too.
3738 if (Attr.isStringAttribute() && Attr.getKindAsString() == "interrupt")
3739 return false;
3740
3742}
3743
3744std::optional<Instruction *>
3746 // Attach a range return attribute describing the result of vsetvli/vsetvlimax
3747 // so generic value analyses can reason about it. The verifier guarantees an
3748 // XLen result and constant VSEW/VLMUL encoding a valid vtype, so no defensive
3749 // validation is needed here.
3750 if (is_contained({Intrinsic::riscv_vsetvli, Intrinsic::riscv_vsetvlimax},
3751 II.getIntrinsicID())) {
3752 // These intrinsics require the V extension; without it the VLEN queries
3753 // below would assert. Such IR would fail isel anyway, so just bail out.
3754 if (!ST->hasVInstructions())
3755 return {};
3756
3757 bool HasAVL = II.getIntrinsicID() == Intrinsic::riscv_vsetvli;
3758 unsigned Offset = HasAVL ? 1 : 0;
3759 unsigned BitWidth = II.getType()->getIntegerBitWidth();
3760 ConstantRange VLenRange(APInt(BitWidth, ST->getRealMinVLen()),
3761 APInt(BitWidth, ST->getRealMaxVLen()) + 1);
3762
3763 uint64_t VSEW = cast<ConstantInt>(II.getArgOperand(Offset))->getZExtValue();
3764 auto VLMUL = static_cast<RISCVVType::VLMUL>(
3765 cast<ConstantInt>(II.getArgOperand(Offset + 1))->getZExtValue());
3766 unsigned SEW = RISCVVType::decodeVSEW(VSEW);
3767 unsigned Ratio = RISCVVType::getSEWLMULRatio(SEW, VLMUL);
3768
3769 // VLMAX = VLEN / (SEW / LMUL), clamped to >= 1 for any usable vtype.
3770 ConstantRange VLMAXRange =
3771 VLenRange.udiv(ConstantRange(APInt(BitWidth, Ratio)))
3773
3774 // vsetvlimax returns exactly VLMAX; vsetvli returns vl with
3775 // 0 <= vl <= min(AVL, VLMAX). vl == AVL only when AVL <= the smallest
3776 // possible VLMAX; otherwise vl can shrink below VLMAX (to 0 at runtime), so
3777 // only the VLMAX upper bound is sound.
3778 ConstantRange VLRange = VLMAXRange;
3779 if (HasAVL) {
3780 // vl ≤ VLMAX
3781 VLRange =
3783
3784 Value *AVL = II.getArgOperand(0);
3786 AVL, /*ForSigned=*/false,
3788
3789 // vl = AVL if AVL ≤ VLMAX
3790 if (AVLRange.icmp(CmpInst::ICMP_ULE, VLMAXRange))
3791 return IC.replaceInstUsesWith(II, AVL);
3792
3793 // vl ≤ AVL
3794 VLRange = VLRange.umin(AVLRange.getUnsignedMax());
3795
3796 // vl > 0 if AVL > 0
3798 VLRange = VLRange.umax(APInt(BitWidth, 1));
3799
3800 // vl = VLMAX if AVL ≥ (2 * VLMAX)
3801 ConstantRange TwoVLMAX = VLMAXRange.multiply(APInt(BitWidth, 2));
3802 if (AVLRange.icmp(CmpInst::ICMP_UGE, TwoVLMAX))
3803 VLRange = VLRange.intersectWith(VLMAXRange);
3804
3805 // ceil(AVL / 2) ≤ vl ≤ VLMAX if AVL < (2 * VLMAX)
3806 if (AVLRange.icmp(CmpInst::ICMP_ULT, TwoVLMAX))
3807 VLRange = VLRange.umax(APIntOps::RoundingUDiv(AVLRange.getUnsignedMin(),
3808 APInt(BitWidth, 2),
3810 }
3811
3812 ConstantRange OldRange =
3813 II.getRange().value_or(ConstantRange::getFull(BitWidth));
3814 ConstantRange NewRange = VLRange.intersectWith(OldRange);
3815 if (NewRange != OldRange) {
3816 II.addRangeRetAttr(NewRange);
3817 return &II;
3818 }
3819 return {};
3820 }
3821
3822 // If all operands of a vmv.v.x are constant, fold a bitcast(vmv.v.x) to scale
3823 // the vmv.v.x, enabling removal of the bitcast. The transform helps avoid
3824 // creating redundant masks.
3825 const DataLayout &DL = IC.getDataLayout();
3826 if (II.user_empty())
3827 return {};
3828 auto *TargetVecTy = dyn_cast<ScalableVectorType>(II.user_back()->getType());
3829 if (!TargetVecTy)
3830 return {};
3831 const APInt *Scalar;
3832 uint64_t VL;
3834 m_Poison(), m_APInt(Scalar), m_ConstantInt(VL))) ||
3835 !all_of(II.users(), [TargetVecTy](User *U) {
3836 return U->getType() == TargetVecTy && match(U, m_BitCast(m_Value()));
3837 }))
3838 return {};
3839 auto *SourceVecTy = cast<ScalableVectorType>(II.getType());
3840 unsigned TargetEltBW = DL.getTypeSizeInBits(TargetVecTy->getElementType());
3841 unsigned SourceEltBW = DL.getTypeSizeInBits(SourceVecTy->getElementType());
3842 if (TargetEltBW % SourceEltBW)
3843 return {};
3844 unsigned TargetScale = TargetEltBW / SourceEltBW;
3845 if (VL % TargetScale || TargetScale == 1)
3846 return {};
3847 Type *VLTy = II.getOperand(2)->getType();
3848 ElementCount SourceEC = SourceVecTy->getElementCount();
3849 unsigned NewEltBW = SourceEltBW * TargetScale;
3850 if (!SourceEC.isKnownMultipleOf(TargetScale) ||
3851 !DL.fitsInLegalInteger(NewEltBW))
3852 return {};
3853 auto *NewEltTy = IntegerType::get(II.getContext(), NewEltBW);
3854 if (!TLI->isLegalElementTypeForRVV(TLI->getValueType(DL, NewEltTy)))
3855 return {};
3856 ElementCount NewEC = SourceEC.divideCoefficientBy(TargetScale);
3857 Type *RetTy = VectorType::get(NewEltTy, NewEC);
3858 assert(SourceVecTy->canLosslesslyBitCastTo(RetTy) &&
3859 "Lossless bitcast between types expected");
3860 APInt NewScalar = APInt::getSplat(NewEltBW, *Scalar);
3861 return IC.replaceInstUsesWith(
3862 II,
3865 RetTy, Intrinsic::riscv_vmv_v_x,
3866 {PoisonValue::get(RetTy), ConstantInt::get(NewEltTy, NewScalar),
3867 ConstantInt::get(VLTy, VL / TargetScale)}),
3868 SourceVecTy));
3869}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > EnableOrLikeSelectOpt("enable-aarch64-or-like-select", cl::init(true), cl::Hidden)
unsigned Imm
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static bool shouldSplit(Instruction *InsertPoint, DenseSet< Value * > &PrevConditionValues, DenseSet< Value * > &ConditionValues, DominatorTree &DT, DenseSet< Instruction * > &Unhoistables)
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
Cost tables and simple lookup functions.
Hexagon Common GEP
static cl::opt< int > InstrCost("inline-instr-cost", cl::Hidden, cl::init(5), cl::desc("Cost of a single instruction when inlining"))
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static LVOptions Options
Definition LVOptions.cpp:25
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
uint64_t IntrinsicInst * II
static InstructionCost costShuffleViaVRegSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, std::optional< unsigned > VLen, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Try to perform better estimation of the permutation.
static InstructionCost costShuffleViaSplitting(const RISCVTTIImpl &TTI, MVT LegalVT, VectorType *Tp, ArrayRef< int > Mask, TTI::TargetCostKind CostKind)
Attempt to approximate the cost of a shuffle which will require splitting during legalization.
static bool isRepeatedConcatMask(ArrayRef< int > Mask, int &SubVectorSize)
static unsigned isM1OrSmaller(MVT VT)
static cl::opt< bool > EnableOrLikeSelectOpt("enable-riscv-or-like-select", cl::init(true), cl::Hidden)
static cl::opt< unsigned > SLPMaxVF("riscv-v-slp-max-vf", cl::desc("Overrides result used for getMaximumVF query which is used " "exclusively by SLP vectorizer."), cl::Hidden)
static cl::opt< unsigned > RVVRegisterWidthLMUL("riscv-v-register-bit-width-lmul", cl::desc("The LMUL to use for getRegisterBitWidth queries. Affects LMUL used " "by autovectorized code. Fractional LMULs are not supported."), cl::init(2), cl::Hidden)
static cl::opt< unsigned > RVVMinTripCount("riscv-v-min-trip-count", cl::desc("Set the lower bound of a trip count to decide on " "vectorization while tail-folding."), cl::init(5), cl::Hidden)
static InstructionCost getIntImmCostImpl(const DataLayout &DL, const RISCVSubtarget *ST, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, bool FreeZeroes)
static VectorType * getVRGatherIndexType(MVT DataVT, const RISCVSubtarget &ST, LLVMContext &C)
static const CostTblEntry VectorIntrinsicCostTable[]
static bool canUseShiftPair(Instruction *Inst, const APInt &Imm)
static bool canUseShiftCmp(Instruction *Inst, const APInt &Imm)
This file defines a TargetTransformInfoImplBase conforming object specific to the RISC-V target machi...
SI Fold Operands
This file contains some templates that are useful if you are working with the STL at all.
#define LLVM_DEBUG(...)
Definition Debug.h:119
This file describes how to lower LLVM code to machine code.
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:197
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
iterator end() const
Definition ArrayRef.h:130
size_t size() const
Get the array size.
Definition ArrayRef.h:141
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
LLVM_ABI bool isStringAttribute() const
Return true if the attribute is a string (target-dependent) attribute.
LLVM_ABI StringRef getKindAsString() const
Return the attribute's kind as a string.
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
Definition InstrTypes.h:757
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
Definition InstrTypes.h:742
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static bool isFPPredicate(Predicate P)
Definition InstrTypes.h:833
static bool isIntPredicate(Predicate P)
Definition InstrTypes.h:839
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
This class represents a range of values.
LLVM_ABI ConstantRange umin(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned minimum of a value in ...
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange umax(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned maximum of a value in ...
static LLVM_ABI ConstantRange makeAllowedICmpRegion(CmpInst::Predicate Pred, const ConstantRange &Other)
Produce the smallest range such that all values that may satisfy the given predicate with any value c...
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
LLVM_ABI ConstantRange udiv(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an unsigned division of a value in...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool noNaNs() const
Definition FMF.h:65
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static FixedVectorType * getDoubleElementsVectorType(FixedVectorType *VTy)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2251
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
The core instruction combiner logic.
const DataLayout & getDataLayout() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
const SimplifyQuery & getSimplifyQuery() const
static InstructionCost getInvalid(CostType Val=0)
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
user_iterator user_begin()
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
const SmallVectorImpl< Type * > & getArgTypes() const
const SmallVectorImpl< const Value * > & getArgs() const
VectorInstrContext getVectorInstrContext() const
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Machine Value Type.
static MVT getFloatingPointVT(unsigned BitWidth)
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
uint64_t getScalarSizeInBits() const
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
static MVT getScalableVectorVT(MVT VT, unsigned NumElements)
MVT changeTypeToInteger()
Return the type converted to an equivalently sized integer or vector with integer element type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool bitsGT(MVT VT) const
Return true if this has more bits than VT.
bool isFixedLengthVector() const
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
static MVT getIntegerVT(unsigned BitWidth)
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Information for memory intrinsic cost model.
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
The optimization diagnostic interface.
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *ValTy, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const override
InstructionCost getStridedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const override
bool isLegalMaskedLoadStore(Type *DataType, Align Alignment) const
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMinTripCountTailFoldingThreshold() const override
TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const override
InstructionCost getAddressComputationCost(Type *PTy, ScalarEvolution *SE, const SCEV *Ptr, TTI::TargetCostKind CostKind) const override
InstructionCost getStoreImmCost(Type *VecTy, TTI::OperandValueInfo OpInfo, TTI::TargetCostKind CostKind) const
Return the cost of materializing an immediate for a value operand of a store instruction.
bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const override
InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const override
std::optional< InstructionCost > getCombinedArithmeticInstructionCost(unsigned ISDOpcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CxtI) const
Check to see if this instruction is expected to be combined to a simpler operation during/before lowe...
bool hasActiveVectorLength() const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const override
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool canSplatOperand(Instruction *I, int Operand) const
Return true if the (vector) instruction I will be lowered to an instruction with a scalar splat opera...
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
bool isLegalMaskedScatter(Type *DataType, Align Alignment) const override
bool isLegalMaskedCompressStore(Type *DataTy, Align Alignment) const override
InstructionCost getGatherScatterOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool shouldTreatInstructionLikeSelect(const Instruction *I) const override
InstructionCost getExpandCompressMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
bool preferAlternateOpcodeVectorization() const override
bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const override
Check if sinking I's operands to I's basic block is profitable, because the operands can be folded in...
bool shouldExpandReduction(const IntrinsicInst *II) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
bool isLegalMaskedGather(Type *DataType, Align Alignment) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const override
unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const override
TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const override
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpdInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
InstructionCost getMaskedMemoryOpCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const override
See if I should be considered for address type promotion.
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
TargetTransformInfo::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
static MVT getM1VT(MVT VT)
Given a vector (either fixed or scalable), return the scalable vector corresponding to a vector regis...
InstructionCost getVRGatherVVCost(MVT VT) const
Return the cost of a vrgather.vv instruction for the type VT.
InstructionCost getVRGatherVICost(MVT VT) const
Return the cost of a vrgather.vi (or vx) instruction for the type VT.
static unsigned computeVLMAX(unsigned VectorBits, unsigned EltSize, unsigned MinSize)
InstructionCost getLMULCost(MVT VT) const
Return the cost of LMUL for linear operations.
InstructionCost getVSlideVICost(MVT VT) const
Return the cost of a vslidedown.vi or vslideup.vi instruction for the type VT.
InstructionCost getVSlideVXCost(MVT VT) const
Return the cost of a vslidedown.vx or vslideup.vx instruction for the type VT.
static RISCVVType::VLMUL getLMUL(MVT VT)
This class represents an analyzed expression in the program.
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
Definition Type.cpp:865
The main scalar evolution driver.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
static LLVM_ABI bool isInterleaveMask(ArrayRef< int > Mask, unsigned Factor, unsigned NumInputElts, SmallVectorImpl< unsigned > &StartIndexes)
Return true if the mask interleaves one or more input vectors together.
Implements a dense probed hash-table based set with some number of buckets stored inline.
Definition DenseSet.h:293
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
virtual const DataLayout & getDataLayout() const
virtual bool shouldTreatInstructionLikeSelect(const Instruction *I) const
virtual TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const
virtual bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const
virtual bool isLoweredToCall(const Function *F) const
InstructionCost getInstructionCost(const User *U, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind) const override
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
PopcntSupportKind
Flags indicating the kind of support for population count.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Expensive
The cost of a 'div' instruction on x86.
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
AddressingModeKind
Which addressing mode Loop Strength Reduction will try to generate.
@ AMK_PostIndexed
Prefer post-indexed addressing mode.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
Definition Type.cpp:61
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:276
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Definition Value.cpp:1002
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
constexpr bool isKnownMultipleOf(ScalarTy RHS) const
This function tells the caller whether the element count is known at compile time to be a multiple of...
Definition TypeSize.h:180
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:230
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
Definition APInt.cpp:2801
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
Definition ISDOpcodes.h:24
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:890
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:417
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:706
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:771
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:860
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:988
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:936
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:741
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
ThreeOps_match< Val_t, Elt_t, Idx_t, Instruction::InsertElement > m_InsertElt(const Val_t &Val, const Elt_t &Elt, const Idx_t &Idx)
Matches InsertElementInst.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
int getIntMatCost(const APInt &Val, unsigned Size, const MCSubtargetInfo &STI, bool CompressionCost, bool FreeZeroes)
static unsigned decodeVSEW(unsigned VSEW)
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
LLVM_ABI unsigned getSEWLMULRatio(unsigned SEW, VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
const CostTblEntryT< CostType > * CostTableLookup(ArrayRef< CostTblEntryT< CostType > > Tbl, int ISD, MVT Ty)
Find in cost table.
Definition CostTable.h:36
LLVM_ABI bool getBooleanLoopAttribute(const Loop *TheLoop, StringRef Name)
Returns true if Name is applied to TheLoop and enabled.
InstructionCost Cost
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ BinaryOp
One of the operands is a binary op.
auto adjacent_find(R &&Range)
Provide wrappers to std::adjacent_find which finds the first pair of adjacent elements that are equal...
Definition STLExtras.h:1818
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
Definition STLExtras.h:1970
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
TargetTransformInfo TTI
LLVM_ABI bool isMaskedSlidePair(ArrayRef< int > Mask, int NumElts, std::array< std::pair< int, int >, 2 > &SrcInfo)
Does this shuffle mask represent either one slide shuffle or a pair of two slide shuffles,...
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
OutputIt copy(R &&Range, OutputIt Out)
Definition STLExtras.h:1885
constexpr unsigned BitWidth
CostTblEntryT< uint16_t > CostTblEntry
Definition CostTable.h:31
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1947
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
LLVM_ABI void processShuffleMasks(ArrayRef< int > Mask, unsigned NumOfSrcRegs, unsigned NumOfDestRegs, unsigned NumOfUsedRegs, function_ref< void()> NoInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned)> SingleInputAction, function_ref< void(ArrayRef< int >, unsigned, unsigned, bool)> ManyInputsAction)
Splits and processes shuffle mask depending on the number of input and output registers.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2146
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Information about a load/store intrinsic defined by the target.
SimplifyQuery getWithInstruction(const Instruction *I) const
unsigned Insns
TODO: Some of these could be merged.
Returns options for expansion of memcmp. IsZeroCmp is.
Describe known properties for a set of pointers.
Parameters that control the generic loop unrolling transformation.
bool UpperBound
Allow using trip count upper bound to unroll loops.
bool Force
Apply loop unroll on any kind of loop (mainly to loops that fail runtime unrolling).
unsigned PartialOptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size, like OptSizeThreshold,...
bool UnrollAndJam
Allow unroll and jam. Used to enable unroll and jam for the target.
bool UnrollRemainder
Allow unrolling of all the iterations of the runtime loop remainder.
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
unsigned OptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size (set to UINT_MAX to disable).