LLVM 24.0.0git
BasicTTIImpl.h
Go to the documentation of this file.
1//===- BasicTTIImpl.h -------------------------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file provides a helper that implements much of the TTI interface in
11/// terms of the target-independent code generator and TargetLowering
12/// interfaces.
13//
14//===----------------------------------------------------------------------===//
15
16#ifndef LLVM_CODEGEN_BASICTTIIMPL_H
17#define LLVM_CODEGEN_BASICTTIIMPL_H
18
19#include "llvm/ADT/APInt.h"
20#include "llvm/ADT/BitVector.h"
21#include "llvm/ADT/STLExtras.h"
35#include "llvm/IR/BasicBlock.h"
36#include "llvm/IR/Constant.h"
37#include "llvm/IR/Constants.h"
38#include "llvm/IR/DataLayout.h"
40#include "llvm/IR/InstrTypes.h"
41#include "llvm/IR/Instruction.h"
43#include "llvm/IR/Intrinsics.h"
44#include "llvm/IR/Operator.h"
45#include "llvm/IR/Type.h"
46#include "llvm/IR/Value.h"
55#include <algorithm>
56#include <cassert>
57#include <cstdint>
58#include <limits>
59#include <optional>
60#include <utility>
61
62namespace llvm {
63
64class Function;
65class GlobalValue;
66class LLVMContext;
67class ScalarEvolution;
68class SCEV;
69class TargetMachine;
70
71/// Returns -partial-unrolling-threshold if specified.
72LLVM_ABI std::optional<unsigned> getPartialUnrollingThreshold();
73
74/// Base class which can be used to help build a TTI implementation.
75///
76/// This class provides as much implementation of the TTI interface as is
77/// possible using the target independent parts of the code generator.
78///
79/// In order to subclass it, your class must implement a getST() method to
80/// return the subtarget, and a getTLI() method to return the target lowering.
81/// We need these methods implemented in the derived class so that this class
82/// doesn't have to duplicate storage for them.
83template <typename T>
85private:
87 using TTI = TargetTransformInfo;
88
89 /// Helper function to access this as a T.
90 const T *thisT() const { return static_cast<const T *>(this); }
91
92 /// Estimate a cost of Broadcast as an extract and sequence of insert
93 /// operations.
95 getBroadcastShuffleOverhead(FixedVectorType *VTy,
98 // Broadcast cost is equal to the cost of extracting the zero'th element
99 // plus the cost of inserting it into every element of the result vector.
100 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, VTy,
101 CostKind, 0, nullptr, nullptr);
102
103 for (int i = 0, e = VTy->getNumElements(); i < e; ++i) {
104 Cost += thisT()->getVectorInstrCost(Instruction::InsertElement, VTy,
105 CostKind, i, nullptr, nullptr);
106 }
107 return Cost;
108 }
109
110 /// Estimate a cost of shuffle as a sequence of extract and insert
111 /// operations.
113 getPermuteShuffleOverhead(FixedVectorType *VTy,
116 // Shuffle cost is equal to the cost of extracting element from its argument
117 // plus the cost of inserting them onto the result vector.
118
119 // e.g. <4 x float> has a mask of <0,5,2,7> i.e we need to extract from
120 // index 0 of first vector, index 1 of second vector,index 2 of first
121 // vector and finally index 3 of second vector and insert them at index
122 // <0,1,2,3> of result vector.
123 for (int i = 0, e = VTy->getNumElements(); i < e; ++i) {
124 Cost += thisT()->getVectorInstrCost(Instruction::InsertElement, VTy,
125 CostKind, i, nullptr, nullptr);
126 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, VTy,
127 CostKind, i, nullptr, nullptr);
128 }
129 return Cost;
130 }
131
132 /// Estimate a cost of subvector extraction as a sequence of extract and
133 /// insert operations.
134 InstructionCost getExtractSubvectorOverhead(VectorType *VTy,
136 int Index,
137 FixedVectorType *SubVTy) const {
138 assert(VTy && SubVTy &&
139 "Can only extract subvectors from vectors");
140 int NumSubElts = SubVTy->getNumElements();
142 (Index + NumSubElts) <=
143 (int)cast<FixedVectorType>(VTy)->getNumElements()) &&
144 "SK_ExtractSubvector index out of range");
145
147 // Subvector extraction cost is equal to the cost of extracting element from
148 // the source type plus the cost of inserting them into the result vector
149 // type.
150 for (int i = 0; i != NumSubElts; ++i) {
151 Cost +=
152 thisT()->getVectorInstrCost(Instruction::ExtractElement, VTy,
153 CostKind, i + Index, nullptr, nullptr);
154 Cost += thisT()->getVectorInstrCost(Instruction::InsertElement, SubVTy,
155 CostKind, i, nullptr, nullptr);
156 }
157 return Cost;
158 }
159
160 /// Estimate a cost of subvector insertion as a sequence of extract and
161 /// insert operations.
162 InstructionCost getInsertSubvectorOverhead(VectorType *VTy,
164 int Index,
165 FixedVectorType *SubVTy) const {
166 assert(VTy && SubVTy &&
167 "Can only insert subvectors into vectors");
168 int NumSubElts = SubVTy->getNumElements();
170 (Index + NumSubElts) <=
171 (int)cast<FixedVectorType>(VTy)->getNumElements()) &&
172 "SK_InsertSubvector index out of range");
173
175 // Subvector insertion cost is equal to the cost of extracting element from
176 // the source type plus the cost of inserting them into the result vector
177 // type.
178 for (int i = 0; i != NumSubElts; ++i) {
179 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, SubVTy,
180 CostKind, i, nullptr, nullptr);
181 Cost +=
182 thisT()->getVectorInstrCost(Instruction::InsertElement, VTy, CostKind,
183 i + Index, nullptr, nullptr);
184 }
185 return Cost;
186 }
187
188 /// Local query method delegates up to T which *must* implement this!
189 const TargetSubtargetInfo *getST() const {
190 return static_cast<const T *>(this)->getST();
191 }
192
193 /// Local query method delegates up to T which *must* implement this!
194 const TargetLoweringBase *getTLI() const {
195 return static_cast<const T *>(this)->getTLI();
196 }
197
198 static ISD::MemIndexedMode getISDIndexedMode(TTI::MemIndexedMode M) {
199 switch (M) {
201 return ISD::UNINDEXED;
202 case TTI::MIM_PreInc:
203 return ISD::PRE_INC;
204 case TTI::MIM_PreDec:
205 return ISD::PRE_DEC;
206 case TTI::MIM_PostInc:
207 return ISD::POST_INC;
208 case TTI::MIM_PostDec:
209 return ISD::POST_DEC;
210 }
211 llvm_unreachable("Unexpected MemIndexedMode");
212 }
213
214 InstructionCost getCommonMaskedMemoryOpCost(unsigned Opcode, Type *DataTy,
215 Align Alignment,
216 bool VariableMask,
217 bool IsGatherScatter,
219 unsigned AddressSpace = 0) const {
220 // We cannot scalarize scalable vectors, so return Invalid.
221 if (isa<ScalableVectorType>(DataTy))
223
224 auto *VT = cast<FixedVectorType>(DataTy);
225 unsigned VF = VT->getNumElements();
226
227 // Assume the target does not have support for gather/scatter operations
228 // and provide a rough estimate.
229 //
230 // First, compute the cost of the individual memory operations.
231 InstructionCost AddrExtractCost =
232 IsGatherScatter ? getScalarizationOverhead(
234 PointerType::get(VT->getContext(), 0), VF),
235 /*Insert=*/false, /*Extract=*/true, CostKind)
236 : 0;
237
238 // The cost of the scalar loads/stores.
239 InstructionCost MemoryOpCost =
240 VF * thisT()->getMemoryOpCost(Opcode, VT->getElementType(), Alignment,
242
243 // Next, compute the cost of packing the result in a vector.
244 InstructionCost PackingCost =
245 getScalarizationOverhead(VT, Opcode != Instruction::Store,
246 Opcode == Instruction::Store, CostKind);
247
248 InstructionCost ConditionalCost = 0;
249 if (VariableMask) {
250 // Compute the cost of conditionally executing the memory operations with
251 // variable masks. This includes extracting the individual conditions, a
252 // branches and PHIs to combine the results.
253 // NOTE: Estimating the cost of conditionally executing the memory
254 // operations accurately is quite difficult and the current solution
255 // provides a very rough estimate only.
256 ConditionalCost =
259 /*Insert=*/false, /*Extract=*/true, CostKind) +
260 VF * (thisT()->getCFInstrCost(Instruction::CondBr, CostKind) +
261 thisT()->getCFInstrCost(Instruction::PHI, CostKind));
262 }
263
264 return AddrExtractCost + MemoryOpCost + PackingCost + ConditionalCost;
265 }
266
267 /// Checks if the provided mask \p is a splat mask, i.e. it contains only -1
268 /// or same non -1 index value and this index value contained at least twice.
269 /// So, mask <0, -1,-1, -1> is not considered splat (it is just identity),
270 /// same for <-1, 0, -1, -1> (just a slide), while <2, -1, 2, -1> is a splat
271 /// with \p Index=2.
272 static bool isSplatMask(ArrayRef<int> Mask, unsigned NumSrcElts, int &Index) {
273 // Check that the broadcast index meets at least twice.
274 bool IsCompared = false;
275 if (int SplatIdx = PoisonMaskElem;
276 all_of(enumerate(Mask), [&](const auto &P) {
277 if (P.value() == PoisonMaskElem)
278 return P.index() != Mask.size() - 1 || IsCompared;
279 if (static_cast<unsigned>(P.value()) >= NumSrcElts * 2)
280 return false;
281 if (SplatIdx == PoisonMaskElem) {
282 SplatIdx = P.value();
283 return P.index() != Mask.size() - 1;
284 }
285 IsCompared = true;
286 return SplatIdx == P.value();
287 })) {
288 Index = SplatIdx;
289 return true;
290 }
291 return false;
292 }
293
294 /// Several intrinsics that return structs (including llvm.sincos[pi] and
295 /// llvm.modf) can be lowered to a vector library call (for certain VFs). The
296 /// vector library functions correspond to the scalar calls (e.g. sincos or
297 /// modf), which unlike the intrinsic return values via output pointers. This
298 /// helper checks if a vector call exists for the given intrinsic, and returns
299 /// the cost, which includes the cost of the mask (if required), and the loads
300 /// for values returned via output pointers. \p LC is the scalar libcall and
301 /// \p CallRetElementIndex (optional) is the struct element which is mapped to
302 /// the call return value. If std::nullopt is returned, then no vector library
303 /// call is available, so the intrinsic should be assigned the default cost
304 /// (e.g. scalarization).
305 std::optional<InstructionCost> getMultipleResultIntrinsicVectorLibCallCost(
307 std::optional<unsigned> CallRetElementIndex = {}) const {
308 Type *RetTy = ICA.getReturnType();
309 // Vector variants of the intrinsic can be mapped to a vector library call.
310 if (!isa<StructType>(RetTy) ||
312 return std::nullopt;
313
314 Type *Ty = getContainedTypes(RetTy).front();
315 EVT VT = getTLI()->getValueType(DL, Ty);
316
317 RTLIB::Libcall LC = RTLIB::UNKNOWN_LIBCALL;
318
319 switch (ICA.getID()) {
320 case Intrinsic::modf:
321 LC = RTLIB::getMODF(VT);
322 break;
323 case Intrinsic::sincospi:
324 LC = RTLIB::getSINCOSPI(VT);
325 break;
326 case Intrinsic::sincos:
327 LC = RTLIB::getSINCOS(VT);
328 break;
329 default:
330 return std::nullopt;
331 }
332
333 // Find associated libcall.
334 RTLIB::LibcallImpl LibcallImpl = getTLI()->getLibcallImpl(LC);
335 if (LibcallImpl == RTLIB::Unsupported)
336 return std::nullopt;
337
338 LLVMContext &Ctx = RetTy->getContext();
339
340 // Cost the call + mask.
341 auto Cost =
342 thisT()->getCallInstrCost(nullptr, RetTy, ICA.getArgTypes(), CostKind);
343
346 auto VecTy = VectorType::get(IntegerType::getInt1Ty(Ctx), VF);
347 Cost += thisT()->getShuffleCost(TargetTransformInfo::SK_Broadcast, VecTy,
348 VecTy, CostKind, {}, 0, nullptr, {});
349 }
350
351 // Lowering to a library call (with output pointers) may require us to emit
352 // reloads for the results.
353 for (auto [Idx, VectorTy] : enumerate(getContainedTypes(RetTy))) {
354 if (Idx == CallRetElementIndex)
355 continue;
356 Cost += thisT()->getMemoryOpCost(
357 Instruction::Load, VectorTy,
358 thisT()->getDataLayout().getABITypeAlign(VectorTy), 0, CostKind);
359 }
360 return Cost;
361 }
362
363 /// Filter out constant and duplicated entries in \p Ops and return a vector
364 /// containing the types from \p Tys corresponding to the remaining operands.
366 filterConstantAndDuplicatedOperands(ArrayRef<const Value *> Ops,
367 ArrayRef<Type *> Tys) {
368 SmallPtrSet<const Value *, 4> UniqueOperands;
369 SmallVector<Type *, 4> FilteredTys;
370 for (const auto &[Op, Ty] : zip_equal(Ops, Tys)) {
371 if (isa<Constant>(Op) || !UniqueOperands.insert(Op).second)
372 continue;
373 FilteredTys.push_back(Ty);
374 }
375 return FilteredTys;
376 }
377
378protected:
379 explicit BasicTTIImplBase(const TargetMachine *TM, const DataLayout &DL)
380 : BaseT(DL) {}
381 ~BasicTTIImplBase() override = default;
382
385
386public:
387 /// \name Scalar TTI Implementations
388 /// @{
390 unsigned AddressSpace, Align Alignment,
391 unsigned *Fast) const override {
392 EVT E = EVT::getIntegerVT(Context, BitWidth);
393 return getTLI()->allowsMisalignedMemoryAccesses(
395 }
396
397 bool areInlineCompatible(const Function *Caller,
398 const Function *Callee) const override {
399 const TargetMachine &TM = getTLI()->getTargetMachine();
400
401 const TargetSubtargetInfo *CallerSTI = TM.getSubtargetImpl(*Caller);
402 const TargetSubtargetInfo *CalleeSTI = TM.getSubtargetImpl(*Callee);
403 FeatureBitset InlineIgnoreFeatures = CallerSTI->getInlineIgnoreFeatures();
404 FeatureBitset InlineInverseFeatures = CallerSTI->getInlineInverseFeatures();
405 FeatureBitset InlineMustMatchFeatures =
406 CallerSTI->getInlineMustMatchFeatures();
407
408 FeatureBitset CallerBits =
409 (CallerSTI->getFeatureBits() ^ InlineInverseFeatures) &
410 ~InlineIgnoreFeatures;
411 FeatureBitset CalleeBits =
412 (CalleeSTI->getFeatureBits() ^ InlineInverseFeatures) &
413 ~InlineIgnoreFeatures;
414
415 if ((CallerBits & InlineMustMatchFeatures) !=
416 (CalleeBits & InlineMustMatchFeatures))
417 return false;
418
419 // Inline a callee if its target-features are a subset of the callers
420 // target-features.
421 return (CallerBits & CalleeBits) == CalleeBits;
422 }
423
424 bool hasBranchDivergence(const Function *F = nullptr) const override {
425 return false;
426 }
427
428 bool isValidAddrSpaceCast(unsigned FromAS, unsigned ToAS) const override {
429 return false;
430 }
431
432 bool addrspacesMayAlias(unsigned AS0, unsigned AS1) const override {
433 return true;
434 }
435
436 unsigned getFlatAddressSpace() const override {
437 // Return an invalid address space.
438 return -1;
439 }
440
442 Intrinsic::ID IID) const override {
443 return false;
444 }
445
446 bool isNoopAddrSpaceCast(unsigned FromAS, unsigned ToAS) const override {
447 return getTLI()->getTargetMachine().isNoopAddrSpaceCast(DL, FromAS, ToAS);
448 }
449
450 unsigned getAssumedAddrSpace(const Value *V) const override {
451 return getTLI()->getTargetMachine().getAssumedAddrSpace(V);
452 }
453
454 std::pair<const Value *, unsigned>
455 getPredicatedAddrSpace(const Value *V) const override {
456 return getTLI()->getTargetMachine().getPredicatedAddrSpace(V);
457 }
458
460 Value *NewV) const override {
461 return nullptr;
462 }
463
464 bool isLegalAddImmediate(int64_t imm) const override {
465 return getTLI()->isLegalAddImmediate(imm);
466 }
467
468 bool isLegalAddScalableImmediate(int64_t Imm) const override {
469 return getTLI()->isLegalAddScalableImmediate(Imm);
470 }
471
472 bool isLegalICmpImmediate(int64_t imm) const override {
473 return getTLI()->isLegalICmpImmediate(imm);
474 }
475
476 bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset,
477 bool HasBaseReg, int64_t Scale, unsigned AddrSpace,
478 Instruction *I = nullptr,
479 int64_t ScalableOffset = 0) const override {
481 AM.BaseGV = BaseGV;
482 AM.BaseOffs = BaseOffset;
483 AM.HasBaseReg = HasBaseReg;
484 AM.Scale = Scale;
485 AM.ScalableOffset = ScalableOffset;
486 return getTLI()->isLegalAddressingMode(DL, AM, Ty, AddrSpace, I);
487 }
488
489 int64_t getPreferredLargeGEPBaseOffset(int64_t MinOffset, int64_t MaxOffset) {
490 return getTLI()->getPreferredLargeGEPBaseOffset(MinOffset, MaxOffset);
491 }
492
493 unsigned getStoreMinimumVF(unsigned VF, Type *ScalarMemTy, Type *ScalarValTy,
494 Align Alignment,
495 unsigned AddrSpace) const override {
496 auto &&IsSupportedByTarget = [this, ScalarMemTy, ScalarValTy, Alignment,
497 AddrSpace](unsigned VF) {
498 auto *SrcTy = FixedVectorType::get(ScalarMemTy, VF / 2);
499 EVT VT = getTLI()->getValueType(DL, SrcTy);
500 if (getTLI()->isOperationLegal(ISD::STORE, VT) ||
501 getTLI()->isOperationCustom(ISD::STORE, VT))
502 return true;
503
504 EVT ValVT =
505 getTLI()->getValueType(DL, FixedVectorType::get(ScalarValTy, VF / 2));
506 EVT LegalizedVT =
507 getTLI()->getTypeToTransformTo(ScalarMemTy->getContext(), VT);
508 return getTLI()->isTruncStoreLegal(LegalizedVT, ValVT, Alignment,
509 AddrSpace);
510 };
511 while (VF > 2 && IsSupportedByTarget(VF))
512 VF /= 2;
513 return VF;
514 }
515
516 bool isIndexedLoadLegal(TTI::MemIndexedMode M, Type *Ty) const override {
517 EVT VT = getTLI()->getValueType(DL, Ty, /*AllowUnknown=*/true);
518 return getTLI()->isIndexedLoadLegal(getISDIndexedMode(M), VT);
519 }
520
521 bool isIndexedStoreLegal(TTI::MemIndexedMode M, Type *Ty) const override {
522 EVT VT = getTLI()->getValueType(DL, Ty, /*AllowUnknown=*/true);
523 return getTLI()->isIndexedStoreLegal(getISDIndexedMode(M), VT);
524 }
525
527 const TTI::LSRCost &C2) const override {
529 }
530
534
538
542
544 StackOffset BaseOffset, bool HasBaseReg,
545 int64_t Scale,
546 unsigned AddrSpace) const override {
548 AM.BaseGV = BaseGV;
549 AM.BaseOffs = BaseOffset.getFixed();
550 AM.HasBaseReg = HasBaseReg;
551 AM.Scale = Scale;
552 AM.ScalableOffset = BaseOffset.getScalable();
553 if (getTLI()->isLegalAddressingMode(DL, AM, Ty, AddrSpace))
554 return 0;
556 }
557
558 bool isTruncateFree(Type *Ty1, Type *Ty2) const override {
559 return getTLI()->isTruncateFree(Ty1, Ty2);
560 }
561
562 bool isProfitableToHoist(Instruction *I) const override {
563 return getTLI()->isProfitableToHoist(I);
564 }
565
566 bool useAA() const override { return getST()->useAA(); }
567
568 bool isTypeLegal(Type *Ty) const override {
569 EVT VT = getTLI()->getValueType(DL, Ty, /*AllowUnknown=*/true);
570 return getTLI()->isTypeLegal(VT);
571 }
572
573 unsigned getRegUsageForType(Type *Ty) const override {
574 EVT ETy = getTLI()->getValueType(DL, Ty);
575 return getTLI()->getNumRegisters(Ty->getContext(), ETy);
576 }
577
578 InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr,
581 Type *AccessType) const override {
582 return BaseT::getGEPCost(PointeeType, Ptr, Operands, CostKind, AccessType);
583 }
584
586 const SwitchInst &SI, unsigned &JumpTableSize, ProfileSummaryInfo *PSI,
587 BlockFrequencyInfo *BFI) const override {
588 /// Try to find the estimated number of clusters. Note that the number of
589 /// clusters identified in this function could be different from the actual
590 /// numbers found in lowering. This function ignore switches that are
591 /// lowered with a mix of jump table / bit test / BTree. This function was
592 /// initially intended to be used when estimating the cost of switch in
593 /// inline cost heuristic, but it's a generic cost model to be used in other
594 /// places (e.g., in loop unrolling).
595 unsigned N = SI.getNumCases();
596 const TargetLoweringBase *TLI = getTLI();
597 const DataLayout &DL = this->getDataLayout();
598
599 JumpTableSize = 0;
600 bool IsJTAllowed = TLI->areJTsAllowed(SI.getParent()->getParent());
601
602 // Early exit if both a jump table and bit test are not allowed.
603 if (N < 1 || (!IsJTAllowed && DL.getIndexSizeInBits(0u) < N))
604 return N;
605
606 APInt MaxCaseVal = SI.case_begin()->getCaseValue()->getValue();
607 APInt MinCaseVal = MaxCaseVal;
608 for (auto CI : SI.cases()) {
609 const APInt &CaseVal = CI.getCaseValue()->getValue();
610 if (CaseVal.sgt(MaxCaseVal))
611 MaxCaseVal = CaseVal;
612 if (CaseVal.slt(MinCaseVal))
613 MinCaseVal = CaseVal;
614 }
615
616 // Check if suitable for a bit test
617 if (N <= DL.getIndexSizeInBits(0u)) {
619 for (auto I : SI.cases()) {
620 const BasicBlock *BB = I.getCaseSuccessor();
621 ++DestMap[BB];
622 }
623
624 if (TLI->isSuitableForBitTests(DestMap, MinCaseVal, MaxCaseVal, DL))
625 return 1;
626 }
627
628 // Check if suitable for a jump table.
629 if (IsJTAllowed) {
630 if (N < 2 || N < TLI->getMinimumJumpTableEntries())
631 return N;
633 (MaxCaseVal - MinCaseVal)
634 .getLimitedValue(std::numeric_limits<uint64_t>::max() - 1) + 1;
635 // Check whether a range of clusters is dense enough for a jump table
636 if (TLI->isSuitableForJumpTable(&SI, N, Range, PSI, BFI)) {
637 JumpTableSize = Range;
638 return 1;
639 }
640 }
641 return N;
642 }
643
644 bool shouldBuildLookupTables() const override {
645 const TargetLoweringBase *TLI = getTLI();
646 return TLI->isOperationLegalOrCustom(ISD::BR_JT, MVT::Other) ||
647 TLI->isOperationLegalOrCustom(ISD::BRIND, MVT::Other);
648 }
649
650 bool shouldBuildRelLookupTables() const override {
651 const TargetMachine &TM = getTLI()->getTargetMachine();
652 // If non-PIC mode, do not generate a relative lookup table.
653 if (!TM.isPositionIndependent())
654 return false;
655
656 /// Relative lookup table entries consist of 32-bit offsets.
657 /// Do not generate relative lookup tables for large code models
658 /// in 64-bit achitectures where 32-bit offsets might not be enough.
659 if (TM.getCodeModel() == CodeModel::Medium ||
661 return false;
662
663 const Triple &TargetTriple = TM.getTargetTriple();
664 if (!TargetTriple.isArch64Bit())
665 return false;
666
667 // TODO: Triggers issues on aarch64 on darwin, so temporarily disable it
668 // there.
669 if (TargetTriple.getArch() == Triple::aarch64 && TargetTriple.isOSDarwin())
670 return false;
671
672 return true;
673 }
674
675 bool haveFastSqrt(Type *Ty) const override {
676 const TargetLoweringBase *TLI = getTLI();
677 EVT VT = TLI->getValueType(DL, Ty);
678 return TLI->isTypeLegal(VT) &&
680 }
681
682 bool haveFastClmul(IntegerType *Ty) const override {
683 // FIXME: clmul should really be Promote for any bitwidth under the largest
684 // legal bitwidth for clmul. Using IndexTy instead of Ty is a hack to get
685 // around that shortcoming.
686 const DataLayout &DL = thisT()->DL;
687 IntegerType *IndexTy =
688 DL.getIndexType(Ty->getContext(), DL.getAllocaAddrSpace());
689 if (Ty->getBitWidth() > IndexTy->getBitWidth())
690 return false;
691
692 const TargetLoweringBase *TLI = getTLI();
693 EVT VT = TLI->getValueType(DL, IndexTy);
694 return TLI->isOperationLegalOrCustom(ISD::CLMUL, VT);
695 }
696
697 bool isFCmpOrdCheaperThanFCmpZero(Type *Ty) const override { return true; }
698
699 InstructionCost getFPOpCost(Type *Ty) const override {
700 // Check whether FADD is available, as a proxy for floating-point in
701 // general.
702 const TargetLoweringBase *TLI = getTLI();
703 EVT VT = TLI->getValueType(DL, Ty);
707 }
708
710 const Function &Fn) const override {
711 switch (Inst.getOpcode()) {
712 default:
713 break;
714 case Instruction::SDiv:
715 case Instruction::SRem:
716 case Instruction::UDiv:
717 case Instruction::URem: {
718 if (!isa<ConstantInt>(Inst.getOperand(1)))
719 return false;
720 EVT VT = getTLI()->getValueType(DL, Inst.getType());
721 return !getTLI()->isIntDivCheap(VT, Fn.getAttributes());
722 }
723 };
724
725 return false;
726 }
727
728 unsigned getInliningThresholdMultiplier() const override { return 1; }
729 unsigned adjustInliningThreshold(const CallBase *CB) const override {
730 return 0;
731 }
732 unsigned getCallerAllocaCost(const CallBase *CB,
733 const AllocaInst *AI) const override {
734 return 0;
735 }
736
737 int getInlinerVectorBonusPercent() const override { return 150; }
738
741 OptimizationRemarkEmitter *ORE) const override {
742 // This unrolling functionality is target independent, but to provide some
743 // motivation for its intended use, for x86:
744
745 // According to the Intel 64 and IA-32 Architectures Optimization Reference
746 // Manual, Intel Core models and later have a loop stream detector (and
747 // associated uop queue) that can benefit from partial unrolling.
748 // The relevant requirements are:
749 // - The loop must have no more than 4 (8 for Nehalem and later) branches
750 // taken, and none of them may be calls.
751 // - The loop can have no more than 18 (28 for Nehalem and later) uops.
752
753 // According to the Software Optimization Guide for AMD Family 15h
754 // Processors, models 30h-4fh (Steamroller and later) have a loop predictor
755 // and loop buffer which can benefit from partial unrolling.
756 // The relevant requirements are:
757 // - The loop must have fewer than 16 branches
758 // - The loop must have less than 40 uops in all executed loop branches
759
760 // The number of taken branches in a loop is hard to estimate here, and
761 // benchmarking has revealed that it is better not to be conservative when
762 // estimating the branch count. As a result, we'll ignore the branch limits
763 // until someone finds a case where it matters in practice.
764
765 unsigned MaxOps;
766 const TargetSubtargetInfo *ST = getST();
767 if (std::optional<unsigned> Threshold = getPartialUnrollingThreshold())
768 MaxOps = *Threshold;
769 else if (ST->getSchedModel().LoopMicroOpBufferSize > 0)
770 MaxOps = ST->getSchedModel().LoopMicroOpBufferSize;
771 else
772 return;
773
774 // Scan the loop: don't unroll loops with calls.
775 for (BasicBlock *BB : L->blocks()) {
776 for (Instruction &I : *BB) {
777 if (isa<CallInst>(I) || isa<InvokeInst>(I)) {
778 if (const Function *F = cast<CallBase>(I).getCalledFunction()) {
779 if (!thisT()->isLoweredToCall(F))
780 continue;
781 }
782
783 if (ORE) {
784 ORE->emit([&]() {
785 return OptimizationRemark("TTI", "DontUnroll", L->getStartLoc(),
786 L->getHeader())
787 << "advising against unrolling the loop because it "
788 "contains a "
789 << ore::NV("Call", &I);
790 });
791 }
792 return;
793 }
794 }
795 }
796
797 // Enable runtime and partial unrolling up to the specified size.
798 // Enable using trip count upper bound to unroll loops.
799 UP.Partial = UP.Runtime = UP.UpperBound = true;
800 UP.PartialThreshold = MaxOps;
801
802 // Avoid unrolling when optimizing for size.
803 UP.OptSizeThreshold = 0;
805
806 // Set number of instructions optimized when "back edge"
807 // becomes "fall through" to default value of 2.
808 UP.BEInsns = 2;
809 }
810
812 TTI::PeelingPreferences &PP) const override {
813 PP.PeelCount = 0;
814 PP.AllowPeeling = true;
815 PP.AllowLoopNestsPeeling = false;
816 PP.PeelProfiledIterations = true;
817 }
818
821 HardwareLoopInfo &HWLoopInfo) const override {
822 return BaseT::isHardwareLoopProfitable(L, SE, AC, LibInfo, HWLoopInfo);
823 }
824
825 unsigned getEpilogueVectorizationMinVF() const override {
827 }
828
832
836
837 std::optional<Instruction *>
840 }
841
842 std::optional<Value *>
844 APInt DemandedMask, KnownBits &Known,
845 bool &KnownBitsComputed) const override {
846 return BaseT::simplifyDemandedUseBitsIntrinsic(IC, II, DemandedMask, Known,
847 KnownBitsComputed);
848 }
849
851 InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts,
852 APInt &UndefElts2, APInt &UndefElts3,
853 std::function<void(Instruction *, unsigned, APInt, APInt &)>
854 SimplifyAndSetOp) const override {
856 IC, II, DemandedElts, UndefElts, UndefElts2, UndefElts3,
857 SimplifyAndSetOp);
858 }
859
861 return getST()->getMispredictionPenalty();
862 }
863
864 std::optional<unsigned>
866 return std::optional<unsigned>(
867 getST()->getCacheSize(static_cast<unsigned>(Level)));
868 }
869
870 std::optional<unsigned>
872 std::optional<unsigned> TargetResult =
873 getST()->getCacheAssociativity(static_cast<unsigned>(Level));
874
875 if (TargetResult)
876 return TargetResult;
877
878 return BaseT::getCacheAssociativity(Level);
879 }
880
881 unsigned getCacheLineSize() const override {
882 return getST()->getCacheLineSize();
883 }
884
885 unsigned getPrefetchDistance() const override {
886 return getST()->getPrefetchDistance();
887 }
888
889 unsigned getMinPrefetchStride(unsigned NumMemAccesses,
890 unsigned NumStridedMemAccesses,
891 unsigned NumPrefetches,
892 bool HasCall) const override {
893 return getST()->getMinPrefetchStride(NumMemAccesses, NumStridedMemAccesses,
894 NumPrefetches, HasCall);
895 }
896
897 unsigned getMaxPrefetchIterationsAhead() const override {
898 return getST()->getMaxPrefetchIterationsAhead();
899 }
900
901 bool enableWritePrefetching() const override {
902 return getST()->enableWritePrefetching();
903 }
904
905 bool shouldPrefetchAddressSpace(unsigned AS) const override {
906 return getST()->shouldPrefetchAddressSpace(AS);
907 }
908
909 /// @}
910
911 /// \name Vector TTI Implementations
912 /// @{
913
918
919 std::optional<unsigned> getVScaleForTuning() const override {
920 return std::nullopt;
921 }
922
923 /// Estimate the overhead of scalarizing an instruction. Insert and Extract
924 /// are set if the demanded result elements need to be inserted and/or
925 /// extracted from vectors.
927 getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts,
928 bool Insert, bool Extract,
930 bool ForPoisonSrc = true, ArrayRef<Value *> VL = {},
932 TTI::VectorInstrContext::None) const override {
933 /// FIXME: a bitfield is not a reasonable abstraction for talking about
934 /// which elements are needed from a scalable vector
935 if (isa<ScalableVectorType>(InTy))
937 auto *Ty = cast<FixedVectorType>(InTy);
938
939 assert(DemandedElts.getBitWidth() == Ty->getNumElements() &&
940 (VL.empty() || VL.size() == Ty->getNumElements()) &&
941 "Vector size mismatch");
942
944
945 for (int i = 0, e = Ty->getNumElements(); i < e; ++i) {
946 if (!DemandedElts[i])
947 continue;
948 if (Insert) {
949 Value *InsertedVal = VL.empty() ? nullptr : VL[i];
950 Cost +=
951 thisT()->getVectorInstrCost(Instruction::InsertElement, Ty,
952 CostKind, i, nullptr, InsertedVal, VIC);
953 }
954 if (Extract)
955 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, Ty,
956 CostKind, i, nullptr, nullptr, VIC);
957 }
958
959 return Cost;
960 }
961
962 bool
964 unsigned ScalarOpdIdx) const override {
965 return false;
966 }
967
969 int OpdIdx) const override {
970 return OpdIdx == -1;
971 }
972
973 bool
975 int RetIdx) const override {
976 return RetIdx == 0;
977 }
978
979 /// Helper wrapper for the DemandedElts variant of getScalarizationOverhead.
981 VectorType *InTy, bool Insert, bool Extract, TTI::TargetCostKind CostKind,
982 bool ForPoisonSrc = true, ArrayRef<Value *> VL = {},
984 if (isa<ScalableVectorType>(InTy))
986 auto *Ty = cast<FixedVectorType>(InTy);
987
988 APInt DemandedElts = APInt::getAllOnes(Ty->getNumElements());
989 // Use CRTP to allow target overrides
990 return thisT()->getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
991 CostKind, ForPoisonSrc, VL, VIC);
992 }
993
994 /// Estimate the overhead of scalarizing an instruction's
995 /// operands. The (potentially vector) types to use for each of
996 /// argument are passes via Tys.
1000 TTI::VectorInstrContext::None) const override {
1002 for (Type *Ty : Tys) {
1003 // Disregard things like metadata arguments.
1004 if (!Ty->isIntOrIntVectorTy() && !Ty->isFPOrFPVectorTy() &&
1005 !Ty->isPtrOrPtrVectorTy())
1006 continue;
1007
1008 if (auto *VecTy = dyn_cast<VectorType>(Ty))
1009 Cost += getScalarizationOverhead(VecTy, /*Insert*/ false,
1010 /*Extract*/ true, CostKind,
1011 /*ForPoisonSrc=*/true, {}, VIC);
1012 }
1013
1014 return Cost;
1015 }
1016
1017 /// Estimate the overhead of scalarizing the inputs and outputs of an
1018 /// instruction, with return type RetTy and arguments Args of type Tys. If
1019 /// Args are unknown (empty), then the cost associated with one argument is
1020 /// added as a heuristic.
1023 ArrayRef<Type *> Tys,
1026 RetTy, /*Insert*/ true, /*Extract*/ false, CostKind);
1027 if (!Args.empty())
1029 filterConstantAndDuplicatedOperands(Args, Tys), CostKind);
1030 else
1031 // When no information on arguments is provided, we add the cost
1032 // associated with one argument as a heuristic.
1033 Cost += getScalarizationOverhead(RetTy, /*Insert*/ false,
1034 /*Extract*/ true, CostKind);
1035
1036 return Cost;
1037 }
1038
1039 /// Estimate the cost of type-legalization and the legalized type.
1040 std::pair<InstructionCost, MVT> getTypeLegalizationCost(Type *Ty) const {
1041 auto [It, Inserted] = TypeLegalizationCostCache.try_emplace(Ty);
1042 if (Inserted)
1043 It->second = computeTypeLegalizationCost(Ty);
1044 return It->second;
1045 }
1046
1047private:
1048 std::pair<InstructionCost, MVT> computeTypeLegalizationCost(Type *Ty) const {
1049 LLVMContext &C = Ty->getContext();
1050 EVT MTy = getTLI()->getValueType(DL, Ty);
1051
1053 // We keep legalizing the type until we find a legal kind. We assume that
1054 // the only operation that costs anything is the split. After splitting
1055 // we need to handle two types.
1056 while (true) {
1058
1060 // Ensure we return a sensible simple VT here, since many callers of
1061 // this function require it.
1062 MVT VT = MTy.isSimple() ? MTy.getSimpleVT() : MVT::i64;
1063 return std::make_pair(InstructionCost::getInvalid(), VT);
1064 }
1065
1066 if (LK.first == TargetLoweringBase::TypeLegal)
1067 return std::make_pair(Cost, MTy.getSimpleVT());
1068
1069 if (LK.first == TargetLoweringBase::TypeSplitVector ||
1071 Cost *= 2;
1072
1073 // Do not loop with f128 type.
1074 if (MTy == LK.second)
1075 return std::make_pair(Cost, MTy.getSimpleVT());
1076
1077 // Keep legalizing the type.
1078 MTy = LK.second;
1079 }
1080 }
1081
1082 /// Memoizes type legalization cost. The mapping does not depend on the IR, so
1083 /// entries stay valid for the lifetime of this object.
1084 mutable DenseMap<Type *, std::pair<InstructionCost, MVT>>
1085 TypeLegalizationCostCache;
1086
1087public:
1089 bool HasUnorderedReductions) const override {
1090 return 1;
1091 }
1092
1094 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
1097 ArrayRef<const Value *> Args = {},
1098 const Instruction *CtxI = nullptr) const override {
1099 // Check if any of the operands are vector operands.
1100 const TargetLoweringBase *TLI = getTLI();
1101 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1102 assert(ISD && "Invalid opcode");
1103
1104 // TODO: Handle more cost kinds.
1106 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind,
1107 Opd1Info, Opd2Info,
1108 Args, CtxI);
1109
1110 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Ty);
1111
1112 bool IsFloat = Ty->isFPOrFPVectorTy();
1113 // Assume that floating point arithmetic operations cost twice as much as
1114 // integer operations.
1115 InstructionCost OpCost = (IsFloat ? 2 : 1);
1116
1117 if (TLI->isOperationLegalOrPromote(ISD, LT.second)) {
1118 // The operation is legal. Assume it costs 1.
1119 // TODO: Once we have extract/insert subvector cost we need to use them.
1120 return LT.first * OpCost;
1121 }
1122
1123 if (!TLI->isOperationExpand(ISD, LT.second)) {
1124 // If the operation is custom lowered, then assume that the code is twice
1125 // as expensive.
1126 return LT.first * 2 * OpCost;
1127 }
1128
1129 // An 'Expand' of URem and SRem is special because it may default
1130 // to expanding the operation into a sequence of sub-operations
1131 // i.e. X % Y -> X-(X/Y)*Y.
1132 if (ISD == ISD::UREM || ISD == ISD::SREM) {
1133 bool IsSigned = ISD == ISD::SREM;
1134 if (TLI->isOperationLegalOrCustom(IsSigned ? ISD::SDIVREM : ISD::UDIVREM,
1135 LT.second) ||
1136 TLI->isOperationLegalOrCustom(IsSigned ? ISD::SDIV : ISD::UDIV,
1137 LT.second)) {
1138 unsigned DivOpc = IsSigned ? Instruction::SDiv : Instruction::UDiv;
1139 InstructionCost DivCost = thisT()->getArithmeticInstrCost(
1140 DivOpc, Ty, CostKind, Opd1Info, Opd2Info);
1141 InstructionCost MulCost =
1142 thisT()->getArithmeticInstrCost(Instruction::Mul, Ty, CostKind);
1143 InstructionCost SubCost =
1144 thisT()->getArithmeticInstrCost(Instruction::Sub, Ty, CostKind);
1145 return DivCost + MulCost + SubCost;
1146 }
1147 }
1148
1149 // We cannot scalarize scalable vectors, so return Invalid.
1152
1153 // Else, assume that we need to scalarize this op.
1154 // TODO: If one of the types get legalized by splitting, handle this
1155 // similarly to what getCastInstrCost() does.
1156 if (auto *VTy = dyn_cast<FixedVectorType>(Ty)) {
1157 InstructionCost Cost = thisT()->getArithmeticInstrCost(
1158 Opcode, VTy->getScalarType(), CostKind, Opd1Info, Opd2Info,
1159 Args, CtxI);
1160 // Return the cost of multiple scalar invocation plus the cost of
1161 // inserting and extracting the values.
1162 SmallVector<Type *> Tys(Args.size(), Ty);
1163 return getScalarizationOverhead(VTy, Args, Tys, CostKind) +
1164 VTy->getNumElements() * Cost;
1165 }
1166
1167 // We don't know anything about this scalar instruction.
1168 return OpCost;
1169 }
1170
1172 ArrayRef<int> Mask,
1173 VectorType *SrcTy, int &Index,
1174 VectorType *&SubTy) const {
1175 if (Mask.empty())
1176 return Kind;
1177 int NumDstElts = Mask.size();
1178 int NumSrcElts = SrcTy->getElementCount().getKnownMinValue();
1179 switch (Kind) {
1181 if (ShuffleVectorInst::isReverseMask(Mask, NumSrcElts))
1182 return TTI::SK_Reverse;
1183 if (ShuffleVectorInst::isZeroEltSplatMask(Mask, NumSrcElts))
1184 return TTI::SK_Broadcast;
1185 if (isSplatMask(Mask, NumSrcElts, Index))
1186 return TTI::SK_Broadcast;
1187 if (ShuffleVectorInst::isExtractSubvectorMask(Mask, NumSrcElts, Index) &&
1188 (Index + NumDstElts) <= NumSrcElts) {
1189 SubTy = FixedVectorType::get(SrcTy->getElementType(), NumDstElts);
1191 }
1192 break;
1193 }
1194 case TTI::SK_PermuteTwoSrc: {
1195 if (all_of(Mask, [NumSrcElts](int M) { return M < NumSrcElts; }))
1197 Index, SubTy);
1198 int NumSubElts;
1199 if (NumDstElts > 2 && ShuffleVectorInst::isInsertSubvectorMask(
1200 Mask, NumSrcElts, NumSubElts, Index)) {
1201 if (Index + NumSubElts > NumSrcElts)
1202 return Kind;
1203 SubTy = FixedVectorType::get(SrcTy->getElementType(), NumSubElts);
1205 }
1206 if (ShuffleVectorInst::isSelectMask(Mask, NumSrcElts))
1207 return TTI::SK_Select;
1208 if (ShuffleVectorInst::isTransposeMask(Mask, NumSrcElts))
1209 return TTI::SK_Transpose;
1210 if (ShuffleVectorInst::isSpliceMask(Mask, NumSrcElts, Index))
1211 return TTI::SK_Splice;
1212 break;
1213 }
1214 case TTI::SK_Select:
1215 case TTI::SK_Reverse:
1216 case TTI::SK_Broadcast:
1217 case TTI::SK_Transpose:
1220 case TTI::SK_Splice:
1221 break;
1222 }
1223 return Kind;
1224 }
1225
1229 VectorType *SubTp, ArrayRef<const Value *> Args = {},
1230 const Instruction *CtxI = nullptr,
1232 TTI::VectorInstrContext::None) const override {
1233 switch (improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp)) {
1234 case TTI::SK_Broadcast:
1235 if (auto *FVT = dyn_cast<FixedVectorType>(SrcTy))
1236 return getBroadcastShuffleOverhead(FVT, CostKind);
1238 case TTI::SK_Select:
1239 case TTI::SK_Splice:
1240 case TTI::SK_Reverse:
1241 case TTI::SK_Transpose:
1244 if (auto *FVT = dyn_cast<FixedVectorType>(SrcTy))
1245 return getPermuteShuffleOverhead(FVT, CostKind);
1248 return getExtractSubvectorOverhead(SrcTy, CostKind, Index,
1249 cast<FixedVectorType>(SubTp));
1251 return getInsertSubvectorOverhead(DstTy, CostKind, Index,
1252 cast<FixedVectorType>(SubTp));
1253 }
1254 llvm_unreachable("Unknown TTI::ShuffleKind");
1255 }
1256
1258 getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src,
1260 const Instruction *I = nullptr) const override {
1261 if (BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I) == 0)
1262 return 0;
1263
1264 const TargetLoweringBase *TLI = getTLI();
1265 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1266 assert(ISD && "Invalid opcode");
1267 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Src);
1268 std::pair<InstructionCost, MVT> DstLT = getTypeLegalizationCost(Dst);
1269
1270 TypeSize SrcSize = SrcLT.second.getSizeInBits();
1271 TypeSize DstSize = DstLT.second.getSizeInBits();
1272 bool IntOrPtrSrc = Src->isIntegerTy() || Src->isPointerTy();
1273 bool IntOrPtrDst = Dst->isIntegerTy() || Dst->isPointerTy();
1274
1275 switch (Opcode) {
1276 default:
1277 break;
1278 case Instruction::Trunc:
1279 // Check for NOOP conversions.
1280 if (TLI->isTruncateFree(SrcLT.second, DstLT.second))
1281 return 0;
1282 [[fallthrough]];
1283 case Instruction::BitCast:
1284 // Bitcast between types that are legalized to the same type are free and
1285 // assume int to/from ptr of the same size is also free.
1286 if (SrcLT.first == DstLT.first && IntOrPtrSrc == IntOrPtrDst &&
1287 SrcSize == DstSize)
1288 return 0;
1289 break;
1290 case Instruction::FPExt:
1291 if (I && getTLI()->isExtFree(I))
1292 return 0;
1293 break;
1294 case Instruction::ZExt:
1295 if (TLI->isZExtFree(SrcLT.second, DstLT.second))
1296 return 0;
1297 [[fallthrough]];
1298 case Instruction::SExt:
1299 if (I && getTLI()->isExtFree(I))
1300 return 0;
1301
1302 // If this is a zext/sext of a load, return 0 if the corresponding
1303 // extending load exists on target and the result type is legal.
1304 if (CCH == TTI::CastContextHint::Normal) {
1305 EVT ExtVT = EVT::getEVT(Dst);
1306 EVT LoadVT = EVT::getEVT(Src);
1307 unsigned LType =
1308 Opcode == Instruction::ZExt ? ISD::ZEXTLOAD : ISD::SEXTLOAD;
1309 if (I) {
1310 if (auto *LI = dyn_cast<LoadInst>(I->getOperand(0))) {
1311 if (DstLT.first == SrcLT.first &&
1312 TLI->isLoadLegal(ExtVT, LoadVT, LI->getAlign(),
1313 LI->getPointerAddressSpace(), LType, false))
1314 return 0;
1315 } else if (auto *II = dyn_cast<IntrinsicInst>(I->getOperand(0))) {
1316 switch (II->getIntrinsicID()) {
1317 case Intrinsic::masked_load: {
1318 Type *PtrType = II->getArgOperand(0)->getType();
1319 assert(PtrType->isPointerTy());
1320
1321 if (DstLT.first == SrcLT.first &&
1322 TLI->isLoadLegal(
1323 ExtVT, LoadVT, II->getParamAlign(0).valueOrOne(),
1324 PtrType->getPointerAddressSpace(), LType, false))
1325 return 0;
1326
1327 break;
1328 }
1329 default:
1330 break;
1331 }
1332 }
1333 }
1334 }
1335 break;
1336 case Instruction::AddrSpaceCast:
1337 if (TLI->isFreeAddrSpaceCast(DL, Src->getPointerAddressSpace(),
1338 Dst->getPointerAddressSpace()))
1339 return 0;
1340 break;
1341 }
1342
1343 auto *SrcVTy = dyn_cast<VectorType>(Src);
1344 auto *DstVTy = dyn_cast<VectorType>(Dst);
1345
1346 // If the cast is marked as legal (or promote) then assume low cost.
1347 if (SrcLT.first == DstLT.first &&
1348 TLI->isOperationLegalOrPromote(ISD, DstLT.second))
1349 return SrcLT.first;
1350
1351 // Handle scalar conversions.
1352 if (!SrcVTy && !DstVTy) {
1353 // Just check the op cost. If the operation is legal then assume it costs
1354 // 1.
1355 if (!TLI->isOperationExpand(ISD, DstLT.second))
1356 return 1;
1357
1358 // Assume that illegal scalar instruction are expensive.
1359 return 4;
1360 }
1361
1362 // Check vector-to-vector casts.
1363 if (DstVTy && SrcVTy) {
1364 // If the cast is between same-sized registers, then the check is simple.
1365 if (SrcLT.first == DstLT.first && SrcSize == DstSize) {
1366
1367 // Assume that Zext is done using AND.
1368 if (Opcode == Instruction::ZExt)
1369 return SrcLT.first;
1370
1371 // Assume that sext is done using SHL and SRA.
1372 if (Opcode == Instruction::SExt)
1373 return SrcLT.first * 2;
1374
1375 // Just check the op cost. If the operation is legal then assume it
1376 // costs
1377 // 1 and multiply by the type-legalization overhead.
1378 if (!TLI->isOperationExpand(ISD, DstLT.second))
1379 return SrcLT.first * 1;
1380 }
1381
1382 // If we are legalizing by splitting, query the concrete TTI for the cost
1383 // of casting the original vector twice. We also need to factor in the
1384 // cost of the split itself. Count that as 1, to be consistent with
1385 // getTypeLegalizationCost().
1386 bool SplitSrc =
1387 TLI->getTypeAction(Src->getContext(), TLI->getValueType(DL, Src)) ==
1389 bool SplitDst =
1390 TLI->getTypeAction(Dst->getContext(), TLI->getValueType(DL, Dst)) ==
1392 if ((SplitSrc || SplitDst) && SrcVTy->getElementCount().isKnownEven() &&
1393 DstVTy->getElementCount().isKnownEven()) {
1394 Type *SplitDstTy = VectorType::getHalfElementsVectorType(DstVTy);
1395 Type *SplitSrcTy = VectorType::getHalfElementsVectorType(SrcVTy);
1396 const T *TTI = thisT();
1397 // If both types need to be split then the split is free.
1398 InstructionCost SplitCost =
1399 (!SplitSrc || !SplitDst) ? TTI->getVectorSplitCost() : 0;
1400 return SplitCost +
1401 (2 * TTI->getCastInstrCost(Opcode, SplitDstTy, SplitSrcTy, CCH,
1402 CostKind, I));
1403 }
1404
1405 // Scalarization cost is Invalid, can't assume any num elements.
1406 if (isa<ScalableVectorType>(DstVTy))
1408
1409 // In other cases where the source or destination are illegal, assume
1410 // the operation will get scalarized.
1411 unsigned Num = cast<FixedVectorType>(DstVTy)->getNumElements();
1412 InstructionCost Cost = thisT()->getCastInstrCost(
1413 Opcode, Dst->getScalarType(), Src->getScalarType(), CCH, CostKind, I);
1414
1415 // Return the cost of multiple scalar invocation plus the cost of
1416 // inserting and extracting the values.
1417 return getScalarizationOverhead(DstVTy, /*Insert*/ true, /*Extract*/ true,
1418 CostKind) +
1419 Num * Cost;
1420 }
1421
1422 // We already handled vector-to-vector and scalar-to-scalar conversions.
1423 // This
1424 // is where we handle bitcast between vectors and scalars. We need to assume
1425 // that the conversion is scalarized in one way or another.
1426 if (Opcode == Instruction::BitCast) {
1427 // Illegal bitcasts are done by storing and loading from a stack slot.
1428 return (SrcVTy ? getScalarizationOverhead(SrcVTy, /*Insert*/ false,
1429 /*Extract*/ true, CostKind)
1430 : 0) +
1431 (DstVTy ? getScalarizationOverhead(DstVTy, /*Insert*/ true,
1432 /*Extract*/ false, CostKind)
1433 : 0);
1434 }
1435
1436 llvm_unreachable("Unhandled cast");
1437 }
1438
1440 getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy,
1441 unsigned Index,
1442 TTI::TargetCostKind CostKind) const override {
1443 return thisT()->getVectorInstrCost(Instruction::ExtractElement, VecTy,
1444 CostKind, Index, nullptr, nullptr) +
1445 thisT()->getCastInstrCost(Opcode, Dst, VecTy->getElementType(),
1447 }
1448
1451 const Instruction *I = nullptr) const override {
1452 return BaseT::getCFInstrCost(Opcode, CostKind, I);
1453 }
1454
1456 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
1460 const Instruction *I = nullptr) const override {
1461 const TargetLoweringBase *TLI = getTLI();
1462 int ISD = TLI->InstructionOpcodeToISD(Opcode);
1463 assert(ISD && "Invalid opcode");
1464
1465 if (getTLI()->getValueType(DL, ValTy, true) == MVT::Other)
1466 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
1467 Op1Info, Op2Info, I);
1468
1469 // Selects on vectors are actually vector selects.
1470 if (ISD == ISD::SELECT) {
1471 assert(CondTy && "CondTy must exist");
1472 if (CondTy->isVectorTy())
1473 ISD = ISD::VSELECT;
1474 }
1475 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(ValTy);
1476
1477 if (!(ValTy->isVectorTy() && !LT.second.isVector()) &&
1478 !TLI->isOperationExpand(ISD, LT.second)) {
1479 // The operation is legal. Assume it costs 1. Multiply
1480 // by the type-legalization overhead.
1481 return LT.first * 1;
1482 }
1483
1484 // Otherwise, assume that the cast is scalarized.
1485 // TODO: If one of the types get legalized by splitting, handle this
1486 // similarly to what getCastInstrCost() does.
1487 if (auto *ValVTy = dyn_cast<VectorType>(ValTy)) {
1488 if (isa<ScalableVectorType>(ValTy))
1490
1491 unsigned Num = cast<FixedVectorType>(ValVTy)->getNumElements();
1492 InstructionCost Cost = thisT()->getCmpSelInstrCost(
1493 Opcode, ValVTy->getScalarType(), CondTy->getScalarType(), VecPred,
1494 CostKind, Op1Info, Op2Info, I);
1495
1496 // Return the cost of multiple scalar invocation plus the cost of
1497 // inserting and extracting the values.
1498 return getScalarizationOverhead(ValVTy, /*Insert*/ true,
1499 /*Extract*/ false, CostKind) +
1500 Num * Cost;
1501 }
1502
1503 // Unknown scalar opcode.
1504 return 1;
1505 }
1506
1509 unsigned Index, const Value *Op0, const Value *Op1,
1511 TTI::VectorInstrContext::None) const override {
1512 return getRegUsageForType(Val->getScalarType());
1513 }
1514
1515 /// \param ScalarUserAndIdx encodes the information about extracts from a
1516 /// vector with 'Scalar' being the value being extracted,'User' being the user
1517 /// of the extract(nullptr if user is not known before vectorization) and
1518 /// 'Idx' being the extract lane.
1520 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
1521 Value *Scalar,
1522 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
1524 TTI::VectorInstrContext::None) const override {
1525 return getVectorInstrCost(Opcode, Val, CostKind, Index, nullptr, nullptr,
1526 VIC);
1527 }
1528
1531 TTI::TargetCostKind CostKind, unsigned Index,
1533 TTI::VectorInstrContext::None) const override {
1534 Value *Op0 = nullptr;
1535 Value *Op1 = nullptr;
1536 if (auto *IE = dyn_cast<InsertElementInst>(&I)) {
1537 Op0 = IE->getOperand(0);
1538 Op1 = IE->getOperand(1);
1539 }
1540 // If VIC is None, compute it from the instruction
1543 return thisT()->getVectorInstrCost(I.getOpcode(), Val, CostKind, Index, Op0,
1544 Op1, VIC);
1545 }
1546
1550 unsigned Index) const override {
1551 unsigned NewIndex = -1;
1552 if (auto *FVTy = dyn_cast<FixedVectorType>(Val)) {
1553 assert(Index < FVTy->getNumElements() &&
1554 "Unexpected index from end of vector");
1555 NewIndex = FVTy->getNumElements() - 1 - Index;
1556 }
1557 return thisT()->getVectorInstrCost(Opcode, Val, CostKind, NewIndex, nullptr,
1558 nullptr);
1559 }
1560
1562 getReplicationShuffleCost(Type *EltTy, int ReplicationFactor, int VF,
1563 const APInt &DemandedDstElts,
1564 TTI::TargetCostKind CostKind) const override {
1565 assert(DemandedDstElts.getBitWidth() == (unsigned)VF * ReplicationFactor &&
1566 "Unexpected size of DemandedDstElts.");
1567
1569
1570 auto *SrcVT = FixedVectorType::get(EltTy, VF);
1571 auto *ReplicatedVT = FixedVectorType::get(EltTy, VF * ReplicationFactor);
1572
1573 // The Mask shuffling cost is extract all the elements of the Mask
1574 // and insert each of them Factor times into the wide vector:
1575 //
1576 // E.g. an interleaved group with factor 3:
1577 // %mask = icmp ult <8 x i32> %vec1, %vec2
1578 // %interleaved.mask = shufflevector <8 x i1> %mask, <8 x i1> undef,
1579 // <24 x i32> <0,0,0,1,1,1,2,2,2,3,3,3,4,4,4,5,5,5,6,6,6,7,7,7>
1580 // The cost is estimated as extract all mask elements from the <8xi1> mask
1581 // vector and insert them factor times into the <24xi1> shuffled mask
1582 // vector.
1583 APInt DemandedSrcElts = APIntOps::ScaleBitMask(DemandedDstElts, VF);
1584 Cost += thisT()->getScalarizationOverhead(SrcVT, DemandedSrcElts,
1585 /*Insert*/ false,
1586 /*Extract*/ true, CostKind);
1587 Cost += thisT()->getScalarizationOverhead(ReplicatedVT, DemandedDstElts,
1588 /*Insert*/ true,
1589 /*Extract*/ false, CostKind);
1590
1591 return Cost;
1592 }
1593
1595 unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace,
1598 const Instruction *I = nullptr) const override {
1599 assert(!Src->isVoidTy() && "Invalid type");
1600 // Assume types, such as structs, are expensive.
1601 if (getTLI()->getValueType(DL, Src, true) == MVT::Other)
1602 return 4;
1603 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Src);
1604
1605 // FIXME: Arbitrary cost
1606 if (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency)
1607 return 4;
1608
1609 // Assuming that all loads of legal types cost 1.
1610 InstructionCost Cost = LT.first;
1612 return Cost;
1613
1614 const DataLayout &DL = this->getDataLayout();
1615 if (Src->isVectorTy() &&
1616 // In practice it's not currently possible to have a change in lane
1617 // length for extending loads or truncating stores so both types should
1618 // have the same scalable property.
1619 TypeSize::isKnownLT(DL.getTypeStoreSizeInBits(Src),
1620 LT.second.getSizeInBits())) {
1621 // This is a vector load that legalizes to a larger type than the vector
1622 // itself. Unless the corresponding extending load or truncating store is
1623 // legal, then this will scalarize.
1625 EVT MemVT = getTLI()->getValueType(DL, Src);
1626 if (Opcode == Instruction::Store)
1627 LA = getTLI()->getTruncStoreAction(LT.second, MemVT, Alignment,
1628 AddressSpace);
1629 else
1630 LA = getTLI()->getLoadAction(LT.second, MemVT, Alignment, AddressSpace,
1631 ISD::EXTLOAD, false);
1632
1633 if (LA != TargetLowering::Legal && LA != TargetLowering::Custom) {
1634 // This is a vector load/store for some illegal type that is scalarized.
1635 // We must account for the cost of building or decomposing the vector.
1637 cast<VectorType>(Src), Opcode != Instruction::Store,
1638 Opcode == Instruction::Store, CostKind);
1639 }
1640 }
1641
1642 return Cost;
1643 }
1644
1646 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
1647 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
1648 bool UseMaskForCond = false, bool UseMaskForGaps = false) const override {
1649
1650 // We cannot scalarize scalable vectors, so return Invalid.
1651 if (isa<ScalableVectorType>(VecTy))
1653
1654 auto *VT = cast<FixedVectorType>(VecTy);
1655
1656 unsigned NumElts = VT->getNumElements();
1657 assert(Factor > 1 && NumElts % Factor == 0 && "Invalid interleave factor");
1658
1659 unsigned NumSubElts = NumElts / Factor;
1660 auto *SubVT = FixedVectorType::get(VT->getElementType(), NumSubElts);
1661
1662 // Firstly, the cost of load/store operation.
1664 if (UseMaskForCond || UseMaskForGaps) {
1665 unsigned IID = Opcode == Instruction::Load ? Intrinsic::masked_load
1666 : Intrinsic::masked_store;
1667 Cost = thisT()->getMemIntrinsicInstrCost(
1668 MemIntrinsicCostAttributes(IID, VecTy, Alignment, AddressSpace),
1669 CostKind);
1670 } else
1671 Cost = thisT()->getMemoryOpCost(Opcode, VecTy, Alignment, AddressSpace,
1672 CostKind);
1673
1674 // Legalize the vector type, and get the legalized and unlegalized type
1675 // sizes.
1676 MVT VecTyLT = getTypeLegalizationCost(VecTy).second;
1677 unsigned VecTySize = thisT()->getDataLayout().getTypeStoreSize(VecTy);
1678 unsigned VecTyLTSize = VecTyLT.getStoreSize();
1679
1680 // Scale the cost of the memory operation by the fraction of legalized
1681 // instructions that will actually be used. We shouldn't account for the
1682 // cost of dead instructions since they will be removed.
1683 //
1684 // E.g., An interleaved load of factor 8:
1685 // %vec = load <16 x i64>, <16 x i64>* %ptr
1686 // %v0 = shufflevector %vec, undef, <0, 8>
1687 //
1688 // If <16 x i64> is legalized to 8 v2i64 loads, only 2 of the loads will be
1689 // used (those corresponding to elements [0:1] and [8:9] of the unlegalized
1690 // type). The other loads are unused.
1691 //
1692 // TODO: Note that legalization can turn masked loads/stores into unmasked
1693 // (legalized) loads/stores. This can be reflected in the cost.
1694 if (Cost.isValid() && VecTySize > VecTyLTSize) {
1695 // The number of loads of a legal type it will take to represent a load
1696 // of the unlegalized vector type.
1697 unsigned NumLegalInsts = divideCeil(VecTySize, VecTyLTSize);
1698
1699 // The number of elements of the unlegalized type that correspond to a
1700 // single legal instruction.
1701 unsigned NumEltsPerLegalInst = divideCeil(NumElts, NumLegalInsts);
1702
1703 // Determine which legal instructions will be used.
1704 BitVector UsedInsts(NumLegalInsts, false);
1705 for (unsigned Index : Indices)
1706 for (unsigned Elt = 0; Elt < NumSubElts; ++Elt)
1707 UsedInsts.set((Index + Elt * Factor) / NumEltsPerLegalInst);
1708
1709 // Scale the cost of the load by the fraction of legal instructions that
1710 // will be used.
1711 Cost = divideCeil(UsedInsts.count() * Cost.getValue(), NumLegalInsts);
1712 }
1713
1714 // Then plus the cost of interleave operation.
1715 assert(Indices.size() <= Factor &&
1716 "Interleaved memory op has too many members");
1717
1718 const APInt DemandedAllSubElts = APInt::getAllOnes(NumSubElts);
1719 const APInt DemandedAllResultElts = APInt::getAllOnes(NumElts);
1720
1721 APInt DemandedLoadStoreElts = APInt::getZero(NumElts);
1722 for (unsigned Index : Indices) {
1723 assert(Index < Factor && "Invalid index for interleaved memory op");
1724 for (unsigned Elm = 0; Elm < NumSubElts; Elm++)
1725 DemandedLoadStoreElts.setBit(Index + Elm * Factor);
1726 }
1727
1728 if (Opcode == Instruction::Load) {
1729 // The interleave cost is similar to extract sub vectors' elements
1730 // from the wide vector, and insert them into sub vectors.
1731 //
1732 // E.g. An interleaved load of factor 2 (with one member of index 0):
1733 // %vec = load <8 x i32>, <8 x i32>* %ptr
1734 // %v0 = shuffle %vec, undef, <0, 2, 4, 6> ; Index 0
1735 // The cost is estimated as extract elements at 0, 2, 4, 6 from the
1736 // <8 x i32> vector and insert them into a <4 x i32> vector.
1737 InstructionCost InsSubCost = thisT()->getScalarizationOverhead(
1738 SubVT, DemandedAllSubElts,
1739 /*Insert*/ true, /*Extract*/ false, CostKind);
1740 Cost += Indices.size() * InsSubCost;
1741 Cost += thisT()->getScalarizationOverhead(VT, DemandedLoadStoreElts,
1742 /*Insert*/ false,
1743 /*Extract*/ true, CostKind);
1744 } else {
1745 // The interleave cost is extract elements from sub vectors, and
1746 // insert them into the wide vector.
1747 //
1748 // E.g. An interleaved store of factor 3 with 2 members at indices 0,1:
1749 // (using VF=4):
1750 // %v0_v1 = shuffle %v0, %v1, <0,4,undef,1,5,undef,2,6,undef,3,7,undef>
1751 // %gaps.mask = <true, true, false, true, true, false,
1752 // true, true, false, true, true, false>
1753 // call llvm.masked.store <12 x i32> %v0_v1, <12 x i32>* %ptr,
1754 // i32 Align, <12 x i1> %gaps.mask
1755 // The cost is estimated as extract all elements (of actual members,
1756 // excluding gaps) from both <4 x i32> vectors and insert into the <12 x
1757 // i32> vector.
1758 InstructionCost ExtSubCost = thisT()->getScalarizationOverhead(
1759 SubVT, DemandedAllSubElts,
1760 /*Insert*/ false, /*Extract*/ true, CostKind);
1761 Cost += ExtSubCost * Indices.size();
1762 Cost += thisT()->getScalarizationOverhead(VT, DemandedLoadStoreElts,
1763 /*Insert*/ true,
1764 /*Extract*/ false, CostKind);
1765 }
1766
1767 if (!UseMaskForCond)
1768 return Cost;
1769
1770 Type *I8Type = Type::getInt8Ty(VT->getContext());
1771
1772 Cost += thisT()->getReplicationShuffleCost(
1773 I8Type, Factor, NumSubElts,
1774 UseMaskForGaps ? DemandedLoadStoreElts : DemandedAllResultElts,
1775 CostKind);
1776
1777 // The Gaps mask is invariant and created outside the loop, therefore the
1778 // cost of creating it is not accounted for here. However if we have both
1779 // a MaskForGaps and some other mask that guards the execution of the
1780 // memory access, we need to account for the cost of And-ing the two masks
1781 // inside the loop.
1782 if (UseMaskForGaps) {
1783 auto *MaskVT = FixedVectorType::get(I8Type, NumElts);
1784 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::And, MaskVT,
1785 CostKind);
1786 }
1787
1788 return Cost;
1789 }
1790
1791 /// Get intrinsic cost based on arguments.
1794 TTI::TargetCostKind CostKind) const override {
1795 // Check for generically free intrinsics.
1797 return 0;
1798
1799 // Assume that target intrinsics are cheap.
1800 Intrinsic::ID IID = ICA.getID();
1803
1804 // VP Intrinsics should have the same cost as their non-vp counterpart.
1805 // TODO: Adjust the cost to make the vp intrinsic cheaper than its non-vp
1806 // counterpart when the vector length argument is smaller than the maximum
1807 // vector length.
1808 // TODO: Support other kinds of VPIntrinsics
1809 if (VPIntrinsic::isVPIntrinsic(ICA.getID())) {
1810 std::optional<unsigned> FOp =
1812 if (FOp) {
1813 if (ICA.getID() == Intrinsic::vp_load) {
1814 Align Alignment;
1815 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1816 Alignment = VPI->getPointerAlignment().valueOrOne();
1817 unsigned AS = 0;
1818 if (ICA.getArgTypes().size() > 1)
1819 if (auto *PtrTy = dyn_cast<PointerType>(ICA.getArgTypes()[0]))
1820 AS = PtrTy->getAddressSpace();
1821 return thisT()->getMemoryOpCost(*FOp, ICA.getReturnType(), Alignment,
1822 AS, CostKind);
1823 }
1824 if (ICA.getID() == Intrinsic::vp_store) {
1825 Align Alignment;
1826 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1827 Alignment = VPI->getPointerAlignment().valueOrOne();
1828 unsigned AS = 0;
1829 if (ICA.getArgTypes().size() >= 2)
1830 if (auto *PtrTy = dyn_cast<PointerType>(ICA.getArgTypes()[1]))
1831 AS = PtrTy->getAddressSpace();
1832 return thisT()->getMemoryOpCost(*FOp, ICA.getArgTypes()[0], Alignment,
1833 AS, CostKind);
1834 }
1835 if (ICA.getID() == Intrinsic::vp_udiv ||
1836 ICA.getID() == Intrinsic::vp_sdiv ||
1837 ICA.getID() == Intrinsic::vp_urem ||
1838 ICA.getID() == Intrinsic::vp_srem) {
1839 return thisT()->getArithmeticInstrCost(*FOp, ICA.getReturnType(),
1840 CostKind);
1841 }
1842 }
1843 if (ICA.getID() == Intrinsic::vp_load_ff) {
1844 Type *RetTy = ICA.getReturnType();
1845 Type *DataTy = cast<StructType>(RetTy)->getElementType(0);
1846 Align Alignment;
1847 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1848 Alignment = VPI->getPointerAlignment().valueOrOne();
1849 return thisT()->getMemIntrinsicInstrCost(
1850 MemIntrinsicCostAttributes(ICA.getID(), DataTy, Alignment),
1851 CostKind);
1852 }
1853 if (ICA.getID() == Intrinsic::vp_scatter) {
1854 if (ICA.isTypeBasedOnly()) {
1855 IntrinsicCostAttributes MaskedScatter(
1858 ICA.getFlags());
1859 return getTypeBasedIntrinsicInstrCost(MaskedScatter, CostKind);
1860 }
1861 Align Alignment;
1862 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1863 Alignment = VPI->getPointerAlignment().valueOrOne();
1864 bool VarMask = isa<Constant>(ICA.getArgs()[2]);
1865 return thisT()->getMemIntrinsicInstrCost(
1866 MemIntrinsicCostAttributes(Intrinsic::vp_scatter,
1867 ICA.getArgTypes()[0], ICA.getArgs()[1],
1868 VarMask, Alignment, nullptr),
1869 CostKind);
1870 }
1871 if (ICA.getID() == Intrinsic::vp_gather) {
1872 if (ICA.isTypeBasedOnly()) {
1873 IntrinsicCostAttributes MaskedGather(
1876 ICA.getFlags());
1877 return getTypeBasedIntrinsicInstrCost(MaskedGather, CostKind);
1878 }
1879 Align Alignment;
1880 if (auto *VPI = dyn_cast_or_null<VPIntrinsic>(ICA.getInst()))
1881 Alignment = VPI->getPointerAlignment().valueOrOne();
1882 bool VarMask = isa<Constant>(ICA.getArgs()[1]);
1883 return thisT()->getMemIntrinsicInstrCost(
1884 MemIntrinsicCostAttributes(Intrinsic::vp_gather,
1885 ICA.getReturnType(), ICA.getArgs()[0],
1886 VarMask, Alignment, nullptr),
1887 CostKind);
1888 }
1889
1890 if (ICA.getID() == Intrinsic::vp_merge) {
1891 TTI::OperandValueInfo OpInfoX, OpInfoY;
1892 if (!ICA.isTypeBasedOnly()) {
1893 OpInfoX = TTI::getOperandInfo(ICA.getArgs()[0]);
1894 OpInfoY = TTI::getOperandInfo(ICA.getArgs()[1]);
1895 }
1896 return getCmpSelInstrCost(
1897 Instruction::Select, ICA.getReturnType(), ICA.getArgTypes()[0],
1898 CmpInst::BAD_ICMP_PREDICATE, CostKind, OpInfoX, OpInfoY);
1899 }
1900
1901 std::optional<Intrinsic::ID> FID =
1903
1904 // Not functionally equivalent but close enough for cost modelling.
1905 if (ICA.getID() == Intrinsic::experimental_vp_reverse)
1906 FID = Intrinsic::vector_reverse;
1907
1908 if (FID) {
1909 // Non-vp version will have same arg types except mask and vector
1910 // length.
1911 assert(ICA.getArgTypes().size() >= 2 &&
1912 "Expected VPIntrinsic to have Mask and Vector Length args and "
1913 "types");
1914
1915 ArrayRef<const Value *> NewArgs = ArrayRef(ICA.getArgs());
1916 if (!ICA.isTypeBasedOnly())
1917 NewArgs = NewArgs.drop_back(2);
1919
1920 // VPReduction intrinsics have a start value argument that their non-vp
1921 // counterparts do not have, except for the fadd and fmul non-vp
1922 // counterpart.
1924 *FID != Intrinsic::vector_reduce_fadd &&
1925 *FID != Intrinsic::vector_reduce_fmul) {
1926 if (!ICA.isTypeBasedOnly())
1927 NewArgs = NewArgs.drop_front();
1928 NewTys = NewTys.drop_front();
1929 }
1930
1931 IntrinsicCostAttributes NewICA(*FID, ICA.getReturnType(), NewArgs,
1932 NewTys, ICA.getFlags());
1933 return thisT()->getIntrinsicInstrCost(NewICA, CostKind);
1934 }
1935 }
1936
1937 if (ICA.isTypeBasedOnly())
1939
1940 Type *RetTy = ICA.getReturnType();
1941
1942 ElementCount RetVF = isVectorizedTy(RetTy) ? getVectorizedTypeVF(RetTy)
1944
1945 const IntrinsicInst *I = ICA.getInst();
1946 const SmallVectorImpl<const Value *> &Args = ICA.getArgs();
1947 FastMathFlags FMF = ICA.getFlags();
1948 switch (IID) {
1949 default:
1950 break;
1951
1952 case Intrinsic::powi:
1953 if (auto *RHSC = dyn_cast<ConstantInt>(Args[1])) {
1954 bool ShouldOptForSize = I->getParent()->getParent()->hasOptSize();
1955 if (getTLI()->isBeneficialToExpandPowI(RHSC->getSExtValue(),
1956 ShouldOptForSize)) {
1957 // The cost is modeled on the expansion performed by ExpandPowI in
1958 // SelectionDAGBuilder.
1959 APInt Exponent = RHSC->getValue().abs();
1960 unsigned ActiveBits = Exponent.getActiveBits();
1961 unsigned PopCount = Exponent.popcount();
1962 InstructionCost Cost = (ActiveBits + PopCount - 2) *
1963 thisT()->getArithmeticInstrCost(
1964 Instruction::FMul, RetTy, CostKind);
1965 if (RHSC->isNegative())
1966 Cost += thisT()->getArithmeticInstrCost(Instruction::FDiv, RetTy,
1967 CostKind);
1968 return Cost;
1969 }
1970 }
1971 break;
1972 case Intrinsic::cttz:
1973 // FIXME: If necessary, this should go in target-specific overrides.
1974 if (RetVF.isScalar() && getTLI()->isCheapToSpeculateCttz(RetTy))
1976 break;
1977
1978 case Intrinsic::ctlz:
1979 // FIXME: If necessary, this should go in target-specific overrides.
1980 if (RetVF.isScalar() && getTLI()->isCheapToSpeculateCtlz(RetTy))
1982 break;
1983
1984 case Intrinsic::memcpy:
1985 return thisT()->getMemcpyCost(ICA.getInst());
1986
1987 case Intrinsic::masked_scatter: {
1988 const Value *Mask = Args[2];
1989 bool VarMask = !isa<Constant>(Mask);
1990 Align Alignment = I->getParamAlign(1).valueOrOne();
1991 return thisT()->getMemIntrinsicInstrCost(
1992 MemIntrinsicCostAttributes(Intrinsic::masked_scatter,
1993 ICA.getArgTypes()[0], Args[1], VarMask,
1994 Alignment, I),
1995 CostKind);
1996 }
1997 case Intrinsic::masked_gather: {
1998 const Value *Mask = Args[1];
1999 bool VarMask = !isa<Constant>(Mask);
2000 Align Alignment = I->getParamAlign(0).valueOrOne();
2001 return thisT()->getMemIntrinsicInstrCost(
2002 MemIntrinsicCostAttributes(Intrinsic::masked_gather, RetTy, Args[0],
2003 VarMask, Alignment, I),
2004 CostKind);
2005 }
2006 case Intrinsic::masked_compressstore: {
2007 const Value *Data = Args[0];
2008 const Value *Mask = Args[2];
2009 Align Alignment = I->getParamAlign(1).valueOrOne();
2010 return thisT()->getMemIntrinsicInstrCost(
2011 MemIntrinsicCostAttributes(IID, Data->getType(), !isa<Constant>(Mask),
2012 Alignment, I),
2013 CostKind);
2014 }
2015 case Intrinsic::masked_expandload: {
2016 const Value *Mask = Args[1];
2017 Align Alignment = I->getParamAlign(0).valueOrOne();
2018 return thisT()->getMemIntrinsicInstrCost(
2019 MemIntrinsicCostAttributes(IID, RetTy, !isa<Constant>(Mask),
2020 Alignment, I),
2021 CostKind);
2022 }
2023 case Intrinsic::experimental_vp_strided_store: {
2024 const Value *Data = Args[0];
2025 const Value *Ptr = Args[1];
2026 const Value *Stride = Args[2];
2027 const Value *Mask = Args[3];
2028 const Value *EVL = Args[4];
2029 bool VarMask = !isa<Constant>(Mask) || !isa<Constant>(EVL);
2030 Type *EltTy = cast<VectorType>(Data->getType())->getElementType();
2031 Align Alignment =
2032 I->getParamAlign(1).value_or(thisT()->DL.getABITypeAlign(EltTy));
2033 return thisT()->getMemIntrinsicInstrCost(
2034 MemIntrinsicCostAttributes(IID, Data->getType(), Ptr, VarMask,
2035 Alignment, I, Stride),
2036 CostKind);
2037 }
2038 case Intrinsic::experimental_vp_strided_load: {
2039 const Value *Ptr = Args[0];
2040 const Value *Stride = Args[1];
2041 const Value *Mask = Args[2];
2042 const Value *EVL = Args[3];
2043 bool VarMask = !isa<Constant>(Mask) || !isa<Constant>(EVL);
2044 Type *EltTy = cast<VectorType>(RetTy)->getElementType();
2045 Align Alignment =
2046 I->getParamAlign(0).value_or(thisT()->DL.getABITypeAlign(EltTy));
2047 return thisT()->getMemIntrinsicInstrCost(
2048 MemIntrinsicCostAttributes(IID, RetTy, Ptr, VarMask, Alignment, I,
2049 Stride),
2050 CostKind);
2051 }
2052 case Intrinsic::stepvector: {
2053 if (isa<ScalableVectorType>(RetTy))
2055 // The cost of materialising a constant integer vector.
2057 }
2058 case Intrinsic::vector_extract: {
2059 // FIXME: Handle case where a scalable vector is extracted from a scalable
2060 // vector
2061 if (isa<ScalableVectorType>(RetTy))
2063 unsigned Index = cast<ConstantInt>(Args[1])->getZExtValue();
2064 return thisT()->getShuffleCost(
2066 cast<VectorType>(Args[0]->getType()), CostKind, {}, Index,
2067 cast<VectorType>(RetTy));
2068 }
2069 case Intrinsic::vector_insert: {
2070 // FIXME: Handle case where a scalable vector is inserted into a scalable
2071 // vector
2072 if (isa<ScalableVectorType>(Args[1]->getType()))
2074 unsigned Index = cast<ConstantInt>(Args[2])->getZExtValue();
2075 return thisT()->getShuffleCost(
2077 cast<VectorType>(Args[0]->getType()), CostKind, {}, Index,
2078 cast<VectorType>(Args[1]->getType()));
2079 }
2080 case Intrinsic::vector_splice_left:
2081 case Intrinsic::vector_splice_right: {
2082 auto *COffset = dyn_cast<ConstantInt>(Args[2]);
2083 if (!COffset)
2084 break;
2085 unsigned Index = COffset->getZExtValue();
2086 return thisT()->getShuffleCost(
2088 cast<VectorType>(Args[0]->getType()), CostKind, {},
2089 IID == Intrinsic::vector_splice_left ? Index : -Index,
2090 cast<VectorType>(RetTy));
2091 }
2092 case Intrinsic::vector_reduce_add:
2093 case Intrinsic::vector_reduce_mul:
2094 case Intrinsic::vector_reduce_and:
2095 case Intrinsic::vector_reduce_or:
2096 case Intrinsic::vector_reduce_xor:
2097 case Intrinsic::vector_reduce_smax:
2098 case Intrinsic::vector_reduce_smin:
2099 case Intrinsic::vector_reduce_fmax:
2100 case Intrinsic::vector_reduce_fmin:
2101 case Intrinsic::vector_reduce_fmaximum:
2102 case Intrinsic::vector_reduce_fminimum:
2103 case Intrinsic::vector_reduce_fmaximumnum:
2104 case Intrinsic::vector_reduce_fminimumnum:
2105 case Intrinsic::vector_reduce_umax:
2106 case Intrinsic::vector_reduce_umin: {
2107 IntrinsicCostAttributes Attrs(IID, RetTy, Args[0]->getType(), FMF, I, 1);
2109 }
2110 case Intrinsic::vector_reduce_fadd:
2111 case Intrinsic::vector_reduce_fmul: {
2113 IID, RetTy, {Args[0]->getType(), Args[1]->getType()}, FMF, I, 1);
2115 }
2116 case Intrinsic::fshl:
2117 case Intrinsic::fshr: {
2118 const Value *X = Args[0];
2119 const Value *Y = Args[1];
2120 const Value *Z = Args[2];
2123 const TTI::OperandValueInfo OpInfoZ = TTI::getOperandInfo(Z);
2124
2125 // fshl: (X << (Z % BW)) | (Y >> (BW - (Z % BW)))
2126 // fshr: (X << (BW - (Z % BW))) | (Y >> (Z % BW))
2128 Cost +=
2129 thisT()->getArithmeticInstrCost(BinaryOperator::Or, RetTy, CostKind);
2130 Cost += thisT()->getArithmeticInstrCost(
2131 BinaryOperator::Shl, RetTy, CostKind, OpInfoX,
2132 {OpInfoZ.Kind, TTI::OP_None});
2133 Cost += thisT()->getArithmeticInstrCost(
2134 BinaryOperator::LShr, RetTy, CostKind, OpInfoY,
2135 {OpInfoZ.Kind, TTI::OP_None});
2136
2137 if (!OpInfoZ.isConstant()) {
2138 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::Sub, RetTy,
2139 CostKind);
2140 // Non-constant shift amounts requires a modulo. If the typesize is a
2141 // power-2 then this will be converted to an and, otherwise it will use
2142 // a urem.
2143 Cost += thisT()->getArithmeticInstrCost(
2144 isPowerOf2_32(RetTy->getScalarSizeInBits()) ? BinaryOperator::And
2145 : BinaryOperator::URem,
2146 RetTy, CostKind, OpInfoZ,
2147 {TTI::OK_UniformConstantValue, TTI::OP_None});
2148 // For non-rotates (X != Y) we must add shift-by-zero handling costs.
2149 if (X != Y) {
2150 Type *CondTy = RetTy->getWithNewBitWidth(1);
2151 Cost += thisT()->getCmpSelInstrCost(
2152 BinaryOperator::ICmp, RetTy, CondTy, CmpInst::ICMP_EQ, CostKind);
2153 Cost +=
2154 thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
2156 }
2157 }
2158 return Cost;
2159 }
2160 case Intrinsic::experimental_cttz_elts: {
2161 EVT ArgType = getTLI()->getValueType(DL, ICA.getArgTypes()[0], true);
2162
2163 // TODO: The costs below reflect the expansion code in
2164 // TargetLowering::expandCttzElts, but we may want to sacrifice some
2165 // accuracy in favour of compile time.
2166
2167 // Find the smallest "sensible" element type to use for the expansion.
2168 bool ZeroIsPoison = !cast<ConstantInt>(Args[1])->isZero();
2169 ConstantRange VScaleRange(APInt(64, 1), APInt::getZero(64));
2170 if (isa<ScalableVectorType>(ICA.getArgTypes()[0]) && I && I->getCaller())
2171 VScaleRange = getVScaleRange(I->getCaller(), 64);
2172
2173 unsigned EltWidth = getTLI()->getBitWidthForCttzElements(
2174 getTLI()->getValueType(DL, RetTy), ArgType.getVectorElementCount(),
2175 ZeroIsPoison, &VScaleRange);
2176 Type *NewEltTy = IntegerType::getIntNTy(RetTy->getContext(), EltWidth);
2177
2178 // Create the new vector type & get the vector length
2179 Type *NewVecTy = VectorType::get(
2180 NewEltTy, cast<VectorType>(Args[0]->getType())->getElementCount());
2181
2182 IntrinsicCostAttributes StepVecAttrs(Intrinsic::stepvector, NewVecTy, {},
2183 FMF);
2185 thisT()->getIntrinsicInstrCost(StepVecAttrs, CostKind);
2186
2187 Cost +=
2188 thisT()->getArithmeticInstrCost(Instruction::Sub, NewVecTy, CostKind);
2189 Cost += thisT()->getCastInstrCost(Instruction::SExt, NewVecTy,
2190 Args[0]->getType(),
2192 Cost +=
2193 thisT()->getArithmeticInstrCost(Instruction::And, NewVecTy, CostKind);
2194
2195 IntrinsicCostAttributes ReducAttrs(Intrinsic::vector_reduce_umax,
2196 NewEltTy, NewVecTy, FMF, I, 1);
2197 Cost += thisT()->getTypeBasedIntrinsicInstrCost(ReducAttrs, CostKind);
2198 Cost +=
2199 thisT()->getArithmeticInstrCost(Instruction::Sub, NewEltTy, CostKind);
2200
2201 return Cost;
2202 }
2203 case Intrinsic::get_active_lane_mask:
2204 case Intrinsic::experimental_vector_match:
2205 case Intrinsic::experimental_vector_histogram_add:
2206 case Intrinsic::experimental_vector_histogram_uadd_sat:
2207 case Intrinsic::experimental_vector_histogram_umax:
2208 case Intrinsic::experimental_vector_histogram_umin:
2209 case Intrinsic::masked_udiv:
2210 case Intrinsic::masked_sdiv:
2211 case Intrinsic::masked_urem:
2212 case Intrinsic::masked_srem:
2213 return thisT()->getTypeBasedIntrinsicInstrCost(ICA, CostKind);
2214 case Intrinsic::modf:
2215 case Intrinsic::sincos:
2216 case Intrinsic::sincospi: {
2217 std::optional<unsigned> CallRetElementIndex;
2218 // The first element of the modf result is returned by value in the
2219 // libcall.
2220 if (ICA.getID() == Intrinsic::modf)
2221 CallRetElementIndex = 0;
2222
2223 if (auto Cost = getMultipleResultIntrinsicVectorLibCallCost(
2224 ICA, CostKind, CallRetElementIndex))
2225 return *Cost;
2226 // Otherwise, fallback to default scalarization cost.
2227 break;
2228 }
2229 case Intrinsic::loop_dependence_war_mask:
2230 case Intrinsic::loop_dependence_raw_mask: {
2231 // Compute the cost of the expanded version of these intrinsics:
2232 //
2233 // The possible expansions are...
2234 //
2235 // loop_dependence_war_mask:
2236 // diff = (addrB - addrA) / eltSize
2237 // cmp = icmp sle diff, 0
2238 // upper_bound = select cmp, -1, diff
2239 // mask = get_active_lane_mask 0, upper_bound
2240 //
2241 // loop_dependence_raw_mask:
2242 // diff = (abs(addrB - addrA)) / eltSize
2243 // cmp = icmp eq diff, 0
2244 // upper_bound = select cmp, -1, diff
2245 // mask = get_active_lane_mask 0, upper_bound
2246 //
2247 Type *AddrTy = ICA.getArgTypes()[0];
2248 bool IsReadAfterWrite = IID == Intrinsic::loop_dependence_raw_mask;
2249
2251 thisT()->getArithmeticInstrCost(Instruction::Sub, AddrTy, CostKind);
2252 if (IsReadAfterWrite) {
2253 IntrinsicCostAttributes AbsAttrs(Intrinsic::abs, AddrTy, {AddrTy}, {});
2254 Cost += thisT()->getIntrinsicInstrCost(AbsAttrs, CostKind);
2255 }
2256
2257 TTI::OperandValueInfo EltSizeOpInfo =
2258 TTI::getOperandInfo(ICA.getArgs()[2]);
2259 Cost += thisT()->getArithmeticInstrCost(Instruction::SDiv, AddrTy,
2260 CostKind, {}, EltSizeOpInfo);
2261
2262 Type *CondTy = IntegerType::getInt1Ty(RetTy->getContext());
2263 CmpInst::Predicate Pred =
2264 IsReadAfterWrite ? CmpInst::ICMP_EQ : CmpInst::ICMP_SLE;
2265 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, CondTy, AddrTy,
2266 Pred, CostKind);
2267 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::Select, AddrTy,
2268 CondTy, Pred, CostKind);
2269
2270 IntrinsicCostAttributes Attrs(Intrinsic::get_active_lane_mask, RetTy,
2271 {AddrTy, AddrTy}, FMF);
2272 Cost += thisT()->getIntrinsicInstrCost(Attrs, CostKind);
2273 return Cost;
2274 }
2275 }
2276
2277 // Assume that we need to scalarize this intrinsic.)
2278 // Compute the scalarization overhead based on Args for a vector
2279 // intrinsic.
2280 InstructionCost ScalarizationCost = InstructionCost::getInvalid();
2281 if (RetVF.isVector() && !RetVF.isScalable()) {
2282 ScalarizationCost = 0;
2283 if (!RetTy->isVoidTy()) {
2284 for (Type *VectorTy : getContainedTypes(RetTy)) {
2285 ScalarizationCost += getScalarizationOverhead(
2286 cast<VectorType>(VectorTy),
2287 /*Insert=*/true, /*Extract=*/false, CostKind);
2288 }
2289 }
2290 ScalarizationCost += getOperandsScalarizationOverhead(
2291 filterConstantAndDuplicatedOperands(Args, ICA.getArgTypes()),
2292 CostKind);
2293 }
2294
2295 IntrinsicCostAttributes Attrs(IID, RetTy, ICA.getArgTypes(), FMF, I,
2296 ScalarizationCost);
2297 return thisT()->getTypeBasedIntrinsicInstrCost(Attrs, CostKind);
2298 }
2299
2300 /// Get intrinsic cost based on argument types.
2301 /// If ScalarizationCostPassed is std::numeric_limits<unsigned>::max(), the
2302 /// cost of scalarizing the arguments and the return value will be computed
2303 /// based on types.
2307 Intrinsic::ID IID = ICA.getID();
2308 Type *RetTy = ICA.getReturnType();
2309 const SmallVectorImpl<Type *> &Tys = ICA.getArgTypes();
2310 FastMathFlags FMF = ICA.getFlags();
2311 InstructionCost ScalarizationCostPassed = ICA.getScalarizationCost();
2312 bool SkipScalarizationCost = ICA.skipScalarizationCost();
2313
2314 VectorType *VecOpTy = nullptr;
2315 if (!Tys.empty()) {
2316 // The vector reduction operand is operand 0 except for fadd/fmul.
2317 // Their operand 0 is a scalar start value, so the vector op is operand 1.
2318 unsigned VecTyIndex = 0;
2319 if (IID == Intrinsic::vector_reduce_fadd ||
2320 IID == Intrinsic::vector_reduce_fmul)
2321 VecTyIndex = 1;
2322 assert(Tys.size() > VecTyIndex && "Unexpected IntrinsicCostAttributes");
2323 VecOpTy = dyn_cast<VectorType>(Tys[VecTyIndex]);
2324 }
2325
2326 // Library call cost - other than size, make it expensive.
2327 unsigned SingleCallCost = CostKind == TTI::TCK_CodeSize ? 1 : 10;
2328 unsigned ISD = 0;
2329 switch (IID) {
2330 default: {
2331 // Scalable vectors cannot be scalarized, so return Invalid.
2332 if (isa<ScalableVectorType>(RetTy) || any_of(Tys, [](const Type *Ty) {
2333 return isa<ScalableVectorType>(Ty);
2334 }))
2336
2337 // Assume that we need to scalarize this intrinsic.
2338 InstructionCost ScalarizationCost =
2339 SkipScalarizationCost ? ScalarizationCostPassed : 0;
2340 unsigned ScalarCalls = 1;
2341 Type *ScalarRetTy = RetTy;
2342 if (auto *RetVTy = dyn_cast<VectorType>(RetTy)) {
2343 if (!SkipScalarizationCost)
2344 ScalarizationCost = getScalarizationOverhead(
2345 RetVTy, /*Insert*/ true, /*Extract*/ false, CostKind);
2346 ScalarCalls = std::max(ScalarCalls,
2347 cast<FixedVectorType>(RetVTy)->getNumElements());
2348 ScalarRetTy = RetTy->getScalarType();
2349 }
2350 SmallVector<Type *, 4> ScalarTys;
2351 for (Type *Ty : Tys) {
2352 if (auto *VTy = dyn_cast<VectorType>(Ty)) {
2353 if (!SkipScalarizationCost)
2354 ScalarizationCost += getScalarizationOverhead(
2355 VTy, /*Insert*/ false, /*Extract*/ true, CostKind);
2356 ScalarCalls = std::max(ScalarCalls,
2357 cast<FixedVectorType>(VTy)->getNumElements());
2358 Ty = Ty->getScalarType();
2359 }
2360 ScalarTys.push_back(Ty);
2361 }
2362 if (ScalarCalls == 1)
2363 return 1; // Return cost of a scalar intrinsic. Assume it to be cheap.
2364
2365 IntrinsicCostAttributes ScalarAttrs(IID, ScalarRetTy, ScalarTys, FMF);
2366 InstructionCost ScalarCost =
2367 thisT()->getIntrinsicInstrCost(ScalarAttrs, CostKind);
2368
2369 return ScalarCalls * ScalarCost + ScalarizationCost;
2370 }
2371 // Look for intrinsics that can be lowered directly or turned into a scalar
2372 // intrinsic call.
2373 case Intrinsic::sqrt:
2374 ISD = ISD::FSQRT;
2375 break;
2376 case Intrinsic::sin:
2377 ISD = ISD::FSIN;
2378 break;
2379 case Intrinsic::cos:
2380 ISD = ISD::FCOS;
2381 break;
2382 case Intrinsic::sincos:
2383 ISD = ISD::FSINCOS;
2384 break;
2385 case Intrinsic::sincospi:
2387 break;
2388 case Intrinsic::modf:
2389 ISD = ISD::FMODF;
2390 break;
2391 case Intrinsic::tan:
2392 ISD = ISD::FTAN;
2393 break;
2394 case Intrinsic::asin:
2395 ISD = ISD::FASIN;
2396 break;
2397 case Intrinsic::acos:
2398 ISD = ISD::FACOS;
2399 break;
2400 case Intrinsic::atan:
2401 ISD = ISD::FATAN;
2402 break;
2403 case Intrinsic::atan2:
2404 ISD = ISD::FATAN2;
2405 break;
2406 case Intrinsic::sinh:
2407 ISD = ISD::FSINH;
2408 break;
2409 case Intrinsic::cosh:
2410 ISD = ISD::FCOSH;
2411 break;
2412 case Intrinsic::tanh:
2413 ISD = ISD::FTANH;
2414 break;
2415 case Intrinsic::exp:
2416 ISD = ISD::FEXP;
2417 break;
2418 case Intrinsic::exp2:
2419 ISD = ISD::FEXP2;
2420 break;
2421 case Intrinsic::exp10:
2422 ISD = ISD::FEXP10;
2423 break;
2424 case Intrinsic::log:
2425 ISD = ISD::FLOG;
2426 break;
2427 case Intrinsic::log10:
2428 ISD = ISD::FLOG10;
2429 break;
2430 case Intrinsic::log2:
2431 ISD = ISD::FLOG2;
2432 break;
2433 case Intrinsic::ldexp:
2434 ISD = ISD::FLDEXP;
2435 break;
2436 case Intrinsic::fabs:
2437 ISD = ISD::FABS;
2438 break;
2439 case Intrinsic::canonicalize:
2441 break;
2442 case Intrinsic::minnum:
2443 ISD = ISD::FMINNUM;
2444 break;
2445 case Intrinsic::maxnum:
2446 ISD = ISD::FMAXNUM;
2447 break;
2448 case Intrinsic::minimum:
2450 break;
2451 case Intrinsic::maximum:
2453 break;
2454 case Intrinsic::minimumnum:
2456 break;
2457 case Intrinsic::maximumnum:
2459 break;
2460 case Intrinsic::copysign:
2462 break;
2463 case Intrinsic::floor:
2464 ISD = ISD::FFLOOR;
2465 break;
2466 case Intrinsic::ceil:
2467 ISD = ISD::FCEIL;
2468 break;
2469 case Intrinsic::trunc:
2470 ISD = ISD::FTRUNC;
2471 break;
2472 case Intrinsic::nearbyint:
2474 break;
2475 case Intrinsic::rint:
2476 ISD = ISD::FRINT;
2477 break;
2478 case Intrinsic::lrint:
2479 ISD = ISD::LRINT;
2480 break;
2481 case Intrinsic::llrint:
2482 ISD = ISD::LLRINT;
2483 break;
2484 case Intrinsic::round:
2485 ISD = ISD::FROUND;
2486 break;
2487 case Intrinsic::roundeven:
2489 break;
2490 case Intrinsic::lround:
2491 ISD = ISD::LROUND;
2492 break;
2493 case Intrinsic::llround:
2494 ISD = ISD::LLROUND;
2495 break;
2496 case Intrinsic::pow:
2497 ISD = ISD::FPOW;
2498 break;
2499 case Intrinsic::fma:
2500 ISD = ISD::FMA;
2501 break;
2502 case Intrinsic::fmuladd:
2503 ISD = ISD::FMA;
2504 break;
2505 case Intrinsic::experimental_constrained_fmuladd:
2507 break;
2508 // FIXME: We should return 0 whenever getIntrinsicCost == TCC_Free.
2509 case Intrinsic::lifetime_start:
2510 case Intrinsic::lifetime_end:
2511 case Intrinsic::sideeffect:
2512 case Intrinsic::pseudoprobe:
2513 case Intrinsic::arithmetic_fence:
2514 return 0;
2515 case Intrinsic::masked_store: {
2516 Type *Ty = Tys[0];
2517 Align TyAlign = thisT()->DL.getABITypeAlign(Ty);
2518 return thisT()->getMemIntrinsicInstrCost(
2519 MemIntrinsicCostAttributes(IID, Ty, TyAlign, 0), CostKind);
2520 }
2521 case Intrinsic::masked_load: {
2522 Type *Ty = RetTy;
2523 Align TyAlign = thisT()->DL.getABITypeAlign(Ty);
2524 return thisT()->getMemIntrinsicInstrCost(
2525 MemIntrinsicCostAttributes(IID, Ty, TyAlign, 0), CostKind);
2526 }
2527 case Intrinsic::speculative_load: {
2528 const IntrinsicInst *I = ICA.getInst();
2529 Align Alignment = I ? I->getParamAlign(0).valueOrOne() : Align(1);
2530 unsigned AS = Tys[0]->getPointerAddressSpace();
2531 return thisT()->getMemIntrinsicInstrCost(
2532 MemIntrinsicCostAttributes(IID, RetTy, Alignment, AS), CostKind);
2533 }
2534 case Intrinsic::experimental_vp_strided_store: {
2535 auto *Ty = cast<VectorType>(ICA.getArgTypes()[0]);
2536 Align Alignment = thisT()->DL.getABITypeAlign(Ty->getElementType());
2537 return thisT()->getMemIntrinsicInstrCost(
2538 MemIntrinsicCostAttributes(IID, Ty, /*Ptr=*/nullptr,
2539 /*VariableMask=*/true, Alignment,
2540 ICA.getInst()),
2541 CostKind);
2542 }
2543 case Intrinsic::experimental_vp_strided_load: {
2544 auto *Ty = cast<VectorType>(ICA.getReturnType());
2545 Align Alignment = thisT()->DL.getABITypeAlign(Ty->getElementType());
2546 return thisT()->getMemIntrinsicInstrCost(
2547 MemIntrinsicCostAttributes(IID, Ty, /*Ptr=*/nullptr,
2548 /*VariableMask=*/true, Alignment,
2549 ICA.getInst()),
2550 CostKind);
2551 }
2552 case Intrinsic::vector_reduce_add:
2553 case Intrinsic::vector_reduce_mul:
2554 case Intrinsic::vector_reduce_and:
2555 case Intrinsic::vector_reduce_or:
2556 case Intrinsic::vector_reduce_xor:
2557 return thisT()->getArithmeticReductionCost(
2558 getArithmeticReductionInstruction(IID), VecOpTy, std::nullopt,
2559 CostKind);
2560 case Intrinsic::vector_reduce_fadd:
2561 case Intrinsic::vector_reduce_fmul:
2562 return thisT()->getArithmeticReductionCost(
2563 getArithmeticReductionInstruction(IID), VecOpTy, FMF, CostKind);
2564 case Intrinsic::vector_reduce_smax:
2565 case Intrinsic::vector_reduce_smin:
2566 case Intrinsic::vector_reduce_umax:
2567 case Intrinsic::vector_reduce_umin:
2568 case Intrinsic::vector_reduce_fmax:
2569 case Intrinsic::vector_reduce_fmin:
2570 case Intrinsic::vector_reduce_fmaximum:
2571 case Intrinsic::vector_reduce_fminimum:
2572 case Intrinsic::vector_reduce_fmaximumnum:
2573 case Intrinsic::vector_reduce_fminimumnum:
2574 return thisT()->getMinMaxReductionCost(getMinMaxReductionIntrinsicOp(IID),
2575 VecOpTy, ICA.getFlags(), CostKind);
2576 case Intrinsic::experimental_vector_match: {
2577 auto *SearchTy = cast<VectorType>(ICA.getArgTypes()[0]);
2578 auto *NeedleTy = cast<FixedVectorType>(ICA.getArgTypes()[1]);
2579 unsigned SearchSize = NeedleTy->getNumElements();
2580
2581 // Approximate the cost based on the expansion code in
2582 // TargetLowering::expandVectorMatch.
2584 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, NeedleTy,
2585 CostKind, 1, nullptr, nullptr);
2586 Cost += thisT()->getVectorInstrCost(Instruction::InsertElement, SearchTy,
2587 CostKind, 0, nullptr, nullptr);
2588 Cost += thisT()->getShuffleCost(TTI::SK_Broadcast, SearchTy, SearchTy,
2589 CostKind, {}, 0, nullptr);
2590 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, SearchTy, RetTy,
2592 Cost +=
2593 thisT()->getArithmeticInstrCost(BinaryOperator::Or, RetTy, CostKind);
2594 Cost *= SearchSize;
2595 Cost +=
2596 thisT()->getArithmeticInstrCost(BinaryOperator::And, RetTy, CostKind);
2597 return Cost;
2598 }
2599 case Intrinsic::vector_reverse:
2600 return thisT()->getShuffleCost(TTI::SK_Reverse, cast<VectorType>(RetTy),
2601 cast<VectorType>(ICA.getArgTypes()[0]),
2602 CostKind, {}, 0, cast<VectorType>(RetTy));
2603 case Intrinsic::experimental_vector_histogram_add:
2604 case Intrinsic::experimental_vector_histogram_uadd_sat:
2605 case Intrinsic::experimental_vector_histogram_umax:
2606 case Intrinsic::experimental_vector_histogram_umin: {
2608 Type *EltTy = ICA.getArgTypes()[1];
2609
2610 // Targets with scalable vectors must handle this on their own.
2611 if (!PtrsTy)
2613
2614 Align Alignment = thisT()->DL.getABITypeAlign(EltTy);
2616 Cost += thisT()->getVectorInstrCost(Instruction::ExtractElement, PtrsTy,
2617 CostKind, 1, nullptr, nullptr);
2618 Cost += thisT()->getMemoryOpCost(Instruction::Load, EltTy, Alignment, 0,
2619 CostKind);
2620 switch (IID) {
2621 default:
2622 llvm_unreachable("Unhandled histogram update operation.");
2623 case Intrinsic::experimental_vector_histogram_add:
2624 Cost +=
2625 thisT()->getArithmeticInstrCost(Instruction::Add, EltTy, CostKind);
2626 break;
2627 case Intrinsic::experimental_vector_histogram_uadd_sat: {
2628 IntrinsicCostAttributes UAddSat(Intrinsic::uadd_sat, EltTy, {EltTy});
2629 Cost += thisT()->getIntrinsicInstrCost(UAddSat, CostKind);
2630 break;
2631 }
2632 case Intrinsic::experimental_vector_histogram_umax: {
2633 IntrinsicCostAttributes UMax(Intrinsic::umax, EltTy, {EltTy});
2634 Cost += thisT()->getIntrinsicInstrCost(UMax, CostKind);
2635 break;
2636 }
2637 case Intrinsic::experimental_vector_histogram_umin: {
2638 IntrinsicCostAttributes UMin(Intrinsic::umin, EltTy, {EltTy});
2639 Cost += thisT()->getIntrinsicInstrCost(UMin, CostKind);
2640 break;
2641 }
2642 }
2643 Cost += thisT()->getMemoryOpCost(Instruction::Store, EltTy, Alignment, 0,
2644 CostKind);
2645 Cost *= PtrsTy->getNumElements();
2646 return Cost;
2647 }
2648 case Intrinsic::get_active_lane_mask: {
2649 Type *ArgTy = ICA.getArgTypes()[0];
2650 EVT ResVT = getTLI()->getValueType(DL, RetTy, true);
2651 EVT ArgVT = getTLI()->getValueType(DL, ArgTy, true);
2652
2653 // If we're not expanding the intrinsic then we assume this is cheap
2654 // to implement.
2655 if (!getTLI()->shouldExpandGetActiveLaneMask(ResVT, ArgVT))
2656 return getTypeLegalizationCost(RetTy).first;
2657
2658 // Create the expanded types that will be used to calculate the uadd_sat
2659 // operation.
2660 Type *ExpRetTy =
2661 VectorType::get(ArgTy, cast<VectorType>(RetTy)->getElementCount());
2662 IntrinsicCostAttributes Attrs(Intrinsic::uadd_sat, ExpRetTy, {}, FMF);
2664 thisT()->getTypeBasedIntrinsicInstrCost(Attrs, CostKind);
2665 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, ExpRetTy, RetTy,
2667 return Cost;
2668 }
2669 case Intrinsic::experimental_memset_pattern:
2670 // This cost is set to match the cost of the memset_pattern16 libcall.
2671 // It should likely be re-evaluated after migration to this intrinsic
2672 // is complete.
2673 return TTI::TCC_Basic * 4;
2674 case Intrinsic::abs:
2675 ISD = ISD::ABS;
2676 break;
2677 case Intrinsic::fshl:
2678 ISD = ISD::FSHL;
2679 break;
2680 case Intrinsic::fshr:
2681 ISD = ISD::FSHR;
2682 break;
2683 case Intrinsic::smax:
2684 ISD = ISD::SMAX;
2685 break;
2686 case Intrinsic::smin:
2687 ISD = ISD::SMIN;
2688 break;
2689 case Intrinsic::umax:
2690 ISD = ISD::UMAX;
2691 break;
2692 case Intrinsic::umin:
2693 ISD = ISD::UMIN;
2694 break;
2695 case Intrinsic::sadd_sat:
2696 ISD = ISD::SADDSAT;
2697 break;
2698 case Intrinsic::ssub_sat:
2699 ISD = ISD::SSUBSAT;
2700 break;
2701 case Intrinsic::uadd_sat:
2702 ISD = ISD::UADDSAT;
2703 break;
2704 case Intrinsic::usub_sat:
2705 ISD = ISD::USUBSAT;
2706 break;
2707 case Intrinsic::smul_fix:
2708 ISD = ISD::SMULFIX;
2709 break;
2710 case Intrinsic::umul_fix:
2711 ISD = ISD::UMULFIX;
2712 break;
2713 case Intrinsic::sadd_with_overflow:
2714 ISD = ISD::SADDO;
2715 break;
2716 case Intrinsic::ssub_with_overflow:
2717 ISD = ISD::SSUBO;
2718 break;
2719 case Intrinsic::uadd_with_overflow:
2720 ISD = ISD::UADDO;
2721 break;
2722 case Intrinsic::usub_with_overflow:
2723 ISD = ISD::USUBO;
2724 break;
2725 case Intrinsic::smul_with_overflow:
2726 ISD = ISD::SMULO;
2727 break;
2728 case Intrinsic::umul_with_overflow:
2729 ISD = ISD::UMULO;
2730 break;
2731 case Intrinsic::fptosi_sat:
2732 case Intrinsic::fptoui_sat: {
2733 std::pair<InstructionCost, MVT> SrcLT = getTypeLegalizationCost(Tys[0]);
2734 std::pair<InstructionCost, MVT> RetLT = getTypeLegalizationCost(RetTy);
2735
2736 // For cast instructions, types are different between source and
2737 // destination. Also need to check if the source type can be legalize.
2738 if (!SrcLT.first.isValid() || !RetLT.first.isValid())
2740 ISD = IID == Intrinsic::fptosi_sat ? ISD::FP_TO_SINT_SAT
2742 break;
2743 }
2744 case Intrinsic::ctpop:
2745 ISD = ISD::CTPOP;
2746 // In case of legalization use TCC_Expensive. This is cheaper than a
2747 // library call but still not a cheap instruction.
2748 SingleCallCost = TargetTransformInfo::TCC_Expensive;
2749 break;
2750 case Intrinsic::ctlz:
2751 ISD = ISD::CTLZ;
2752 break;
2753 case Intrinsic::cttz:
2754 ISD = ISD::CTTZ;
2755 break;
2756 case Intrinsic::bswap:
2757 ISD = ISD::BSWAP;
2758 break;
2759 case Intrinsic::bitreverse:
2761 break;
2762 case Intrinsic::ucmp:
2763 ISD = ISD::UCMP;
2764 break;
2765 case Intrinsic::scmp:
2766 ISD = ISD::SCMP;
2767 break;
2768 case Intrinsic::clmul:
2769 ISD = ISD::CLMUL;
2770 break;
2771 case Intrinsic::smulh:
2772 ISD = ISD::MULHS;
2773 break;
2774 case Intrinsic::umulh:
2775 ISD = ISD::MULHU;
2776 break;
2777 case Intrinsic::masked_udiv:
2778 case Intrinsic::masked_sdiv:
2779 case Intrinsic::masked_urem:
2780 case Intrinsic::masked_srem: {
2781 unsigned UnmaskedOpc;
2782 switch (IID) {
2783 case Intrinsic::masked_udiv:
2785 UnmaskedOpc = Instruction::UDiv;
2786 break;
2787 case Intrinsic::masked_sdiv:
2789 UnmaskedOpc = Instruction::SDiv;
2790 break;
2791 case Intrinsic::masked_urem:
2793 UnmaskedOpc = Instruction::URem;
2794 break;
2795 case Intrinsic::masked_srem:
2797 UnmaskedOpc = Instruction::SRem;
2798 break;
2799 default:
2800 llvm_unreachable("Unexpected intrinsic ID");
2801 }
2803 thisT()->getArithmeticInstrCost(UnmaskedOpc, RetTy, CostKind);
2804
2805 // Expansion generates a (select %mask, %rhs, 1) for the divisor.
2806 MVT LT = getTypeLegalizationCost(RetTy).second;
2807 if (!getTLI()->isOperationLegalOrCustom(ISD, LT)) {
2808 Type *CondTy = cast<VectorType>(RetTy)->getWithNewType(
2810 Cost += thisT()->getCmpSelInstrCost(
2811 BinaryOperator::Select, RetTy, CondTy, CmpInst::BAD_ICMP_PREDICATE,
2813 }
2814
2815 return Cost;
2816 }
2817 }
2818
2819 auto *ST = dyn_cast<StructType>(RetTy);
2820 Type *LegalizeTy = ST ? ST->getContainedType(0) : RetTy;
2821 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(LegalizeTy);
2822
2823 const TargetLoweringBase *TLI = getTLI();
2824
2825 if (TLI->isOperationLegalOrPromote(ISD, LT.second)) {
2826 if (IID == Intrinsic::fabs && LT.second.isFloatingPoint() &&
2827 TLI->isFAbsFree(LT.second)) {
2828 return 0;
2829 }
2830
2831 // The operation is legal. Assume it costs 1.
2832 // If the type is split to multiple registers, assume that there is some
2833 // overhead to this.
2834 // TODO: Once we have extract/insert subvector cost we need to use them.
2835 if (LT.first > 1)
2836 return (LT.first * 2);
2837 else
2838 return (LT.first * 1);
2839 } else if (TLI->isOperationCustom(ISD, LT.second)) {
2840 // If the operation is custom lowered then assume
2841 // that the code is twice as expensive.
2842 return (LT.first * 2);
2843 }
2844
2845 switch (IID) {
2846 case Intrinsic::fmuladd: {
2847 // If we can't lower fmuladd into an FMA estimate the cost as a floating
2848 // point mul followed by an add.
2849
2850 return thisT()->getArithmeticInstrCost(BinaryOperator::FMul, RetTy,
2851 CostKind) +
2852 thisT()->getArithmeticInstrCost(BinaryOperator::FAdd, RetTy,
2853 CostKind);
2854 }
2855 case Intrinsic::experimental_constrained_fmuladd: {
2856 IntrinsicCostAttributes FMulAttrs(
2857 Intrinsic::experimental_constrained_fmul, RetTy, Tys);
2858 IntrinsicCostAttributes FAddAttrs(
2859 Intrinsic::experimental_constrained_fadd, RetTy, Tys);
2860 return thisT()->getIntrinsicInstrCost(FMulAttrs, CostKind) +
2861 thisT()->getIntrinsicInstrCost(FAddAttrs, CostKind);
2862 }
2863 case Intrinsic::smin:
2864 case Intrinsic::smax:
2865 case Intrinsic::umin:
2866 case Intrinsic::umax: {
2867 // minmax(X,Y) = select(icmp(X,Y),X,Y)
2868 Type *CondTy = RetTy->getWithNewBitWidth(1);
2869 bool IsUnsigned = IID == Intrinsic::umax || IID == Intrinsic::umin;
2870 CmpInst::Predicate Pred =
2871 IsUnsigned ? CmpInst::ICMP_UGT : CmpInst::ICMP_SGT;
2873 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
2874 Pred, CostKind);
2875 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
2876 Pred, CostKind);
2877 return Cost;
2878 }
2879 case Intrinsic::sadd_with_overflow:
2880 case Intrinsic::ssub_with_overflow: {
2881 Type *SumTy = RetTy->getContainedType(0);
2882 Type *OverflowTy = RetTy->getContainedType(1);
2883 unsigned Opcode = IID == Intrinsic::sadd_with_overflow
2884 ? BinaryOperator::Add
2885 : BinaryOperator::Sub;
2886
2887 // Add:
2888 // Overflow -> (Result < LHS) ^ (RHS < 0)
2889 // Sub:
2890 // Overflow -> (Result < LHS) ^ (RHS > 0)
2892 Cost += thisT()->getArithmeticInstrCost(Opcode, SumTy, CostKind);
2893 Cost +=
2894 2 * thisT()->getCmpSelInstrCost(Instruction::ICmp, SumTy, OverflowTy,
2896 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::Xor, OverflowTy,
2897 CostKind);
2898 return Cost;
2899 }
2900 case Intrinsic::uadd_with_overflow:
2901 case Intrinsic::usub_with_overflow: {
2902 Type *SumTy = RetTy->getContainedType(0);
2903 Type *OverflowTy = RetTy->getContainedType(1);
2904 unsigned Opcode = IID == Intrinsic::uadd_with_overflow
2905 ? BinaryOperator::Add
2906 : BinaryOperator::Sub;
2907 CmpInst::Predicate Pred = IID == Intrinsic::uadd_with_overflow
2910
2912 Cost += thisT()->getArithmeticInstrCost(Opcode, SumTy, CostKind);
2913 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, SumTy,
2914 OverflowTy, Pred, CostKind);
2915 return Cost;
2916 }
2917 case Intrinsic::smul_with_overflow:
2918 case Intrinsic::umul_with_overflow: {
2919 Type *MulTy = RetTy->getContainedType(0);
2920 Type *OverflowTy = RetTy->getContainedType(1);
2921 unsigned ExtSize = MulTy->getScalarSizeInBits() * 2;
2922 Type *ExtTy = MulTy->getWithNewBitWidth(ExtSize);
2923 bool IsSigned = IID == Intrinsic::smul_with_overflow;
2924
2925 unsigned ExtOp = IsSigned ? Instruction::SExt : Instruction::ZExt;
2927
2929 Cost += 2 * thisT()->getCastInstrCost(ExtOp, ExtTy, MulTy, CCH, CostKind);
2930 Cost +=
2931 thisT()->getArithmeticInstrCost(Instruction::Mul, ExtTy, CostKind);
2932 Cost += 2 * thisT()->getCastInstrCost(Instruction::Trunc, MulTy, ExtTy,
2933 CCH, CostKind);
2934 Cost += thisT()->getArithmeticInstrCost(
2935 Instruction::LShr, ExtTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
2937
2938 if (IsSigned)
2939 Cost += thisT()->getArithmeticInstrCost(
2940 Instruction::AShr, MulTy, CostKind,
2943
2944 Cost += thisT()->getCmpSelInstrCost(
2945 BinaryOperator::ICmp, MulTy, OverflowTy, CmpInst::ICMP_NE, CostKind);
2946 return Cost;
2947 }
2948 case Intrinsic::sadd_sat:
2949 case Intrinsic::ssub_sat: {
2950 // Assume a default expansion.
2951 Type *CondTy = RetTy->getWithNewBitWidth(1);
2952
2953 Type *OpTy = StructType::create({RetTy, CondTy});
2954 Intrinsic::ID OverflowOp = IID == Intrinsic::sadd_sat
2955 ? Intrinsic::sadd_with_overflow
2956 : Intrinsic::ssub_with_overflow;
2958
2959 // SatMax -> Overflow && SumDiff < 0
2960 // SatMin -> Overflow && SumDiff >= 0
2962 IntrinsicCostAttributes Attrs(OverflowOp, OpTy, {RetTy, RetTy}, FMF,
2963 nullptr, ScalarizationCostPassed);
2964 Cost += thisT()->getIntrinsicInstrCost(Attrs, CostKind);
2965 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
2966 Pred, CostKind);
2967 Cost += 2 * thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy,
2968 CondTy, Pred, CostKind);
2969 return Cost;
2970 }
2971 case Intrinsic::uadd_sat:
2972 case Intrinsic::usub_sat: {
2973 Type *CondTy = RetTy->getWithNewBitWidth(1);
2974
2975 Type *OpTy = StructType::create({RetTy, CondTy});
2976 Intrinsic::ID OverflowOp = IID == Intrinsic::uadd_sat
2977 ? Intrinsic::uadd_with_overflow
2978 : Intrinsic::usub_with_overflow;
2979
2981 IntrinsicCostAttributes Attrs(OverflowOp, OpTy, {RetTy, RetTy}, FMF,
2982 nullptr, ScalarizationCostPassed);
2983 Cost += thisT()->getIntrinsicInstrCost(Attrs, CostKind);
2984 Cost +=
2985 thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
2987 return Cost;
2988 }
2989 case Intrinsic::smul_fix:
2990 case Intrinsic::umul_fix: {
2991 unsigned ExtSize = RetTy->getScalarSizeInBits() * 2;
2992 Type *ExtTy = RetTy->getWithNewBitWidth(ExtSize);
2993
2994 unsigned ExtOp =
2995 IID == Intrinsic::smul_fix ? Instruction::SExt : Instruction::ZExt;
2997
2999 Cost += 2 * thisT()->getCastInstrCost(ExtOp, ExtTy, RetTy, CCH, CostKind);
3000 Cost +=
3001 thisT()->getArithmeticInstrCost(Instruction::Mul, ExtTy, CostKind);
3002 Cost += 2 * thisT()->getCastInstrCost(Instruction::Trunc, RetTy, ExtTy,
3003 CCH, CostKind);
3004 Cost += thisT()->getArithmeticInstrCost(
3005 Instruction::LShr, RetTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
3007 Cost += thisT()->getArithmeticInstrCost(
3008 Instruction::Shl, RetTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
3010 Cost += thisT()->getArithmeticInstrCost(Instruction::Or, RetTy, CostKind);
3011 return Cost;
3012 }
3013 case Intrinsic::abs: {
3014 // abs(X) = select(icmp(X,0),X,sub(0,X))
3015 Type *CondTy = RetTy->getWithNewBitWidth(1);
3018 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
3019 Pred, CostKind);
3020 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
3021 Pred, CostKind);
3022 // TODO: Should we add an OperandValueProperties::OP_Zero property?
3023 Cost += thisT()->getArithmeticInstrCost(
3024 BinaryOperator::Sub, RetTy, CostKind,
3026 return Cost;
3027 }
3028 case Intrinsic::fshl:
3029 case Intrinsic::fshr: {
3030 // fshl: (X << (Z % BW)) | (Y >> (BW - (Z % BW)))
3031 // fshr: (X << (BW - (Z % BW))) | (Y >> (Z % BW))
3032 Type *CondTy = RetTy->getWithNewBitWidth(1);
3034 Cost +=
3035 thisT()->getArithmeticInstrCost(BinaryOperator::Or, RetTy, CostKind);
3036 Cost +=
3037 thisT()->getArithmeticInstrCost(BinaryOperator::Sub, RetTy, CostKind);
3038 Cost +=
3039 thisT()->getArithmeticInstrCost(BinaryOperator::Shl, RetTy, CostKind);
3040 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::LShr, RetTy,
3041 CostKind);
3042 // Non-constant shift amounts requires a modulo. If the typesize is a
3043 // power-2 then this will be converted to an and, otherwise it will use a
3044 // urem.
3045 Cost += thisT()->getArithmeticInstrCost(
3046 isPowerOf2_32(RetTy->getScalarSizeInBits()) ? BinaryOperator::And
3047 : BinaryOperator::URem,
3048 RetTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
3049 {TTI::OK_UniformConstantValue, TTI::OP_None});
3050 // Shift-by-zero handling.
3051 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, RetTy, CondTy,
3053 Cost += thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, CondTy,
3055 return Cost;
3056 }
3057 case Intrinsic::fptosi_sat:
3058 case Intrinsic::fptoui_sat: {
3059 if (Tys.empty())
3060 break;
3061 Type *FromTy = Tys[0];
3062 bool IsSigned = IID == Intrinsic::fptosi_sat;
3063
3065 IntrinsicCostAttributes Attrs1(Intrinsic::minnum, FromTy,
3066 {FromTy, FromTy});
3067 Cost += thisT()->getIntrinsicInstrCost(Attrs1, CostKind);
3068 IntrinsicCostAttributes Attrs2(Intrinsic::maxnum, FromTy,
3069 {FromTy, FromTy});
3070 Cost += thisT()->getIntrinsicInstrCost(Attrs2, CostKind);
3071 Cost += thisT()->getCastInstrCost(
3072 IsSigned ? Instruction::FPToSI : Instruction::FPToUI, RetTy, FromTy,
3074 if (IsSigned) {
3075 Type *CondTy = RetTy->getWithNewBitWidth(1);
3076 Cost += thisT()->getCmpSelInstrCost(
3077 BinaryOperator::FCmp, FromTy, CondTy, CmpInst::FCMP_UNO, CostKind);
3078 Cost += thisT()->getCmpSelInstrCost(
3079 BinaryOperator::Select, RetTy, CondTy, CmpInst::FCMP_UNO, CostKind);
3080 }
3081 return Cost;
3082 }
3083 case Intrinsic::ucmp:
3084 case Intrinsic::scmp: {
3085 Type *CmpTy = Tys[0];
3086 Type *CondTy = RetTy->getWithNewBitWidth(1);
3088 thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, CmpTy, CondTy,
3090 CostKind) +
3091 thisT()->getCmpSelInstrCost(BinaryOperator::ICmp, CmpTy, CondTy,
3093 CostKind);
3094
3095 EVT VT = TLI->getValueType(DL, CmpTy, true);
3097 // x < y ? -1 : (x > y ? 1 : 0)
3098 Cost += 2 * thisT()->getCmpSelInstrCost(
3099 BinaryOperator::Select, RetTy, CondTy,
3101 } else {
3102 // zext(x > y) - zext(x < y)
3103 Cost +=
3104 2 * thisT()->getCastInstrCost(CastInst::ZExt, RetTy, CondTy,
3106 Cost += thisT()->getArithmeticInstrCost(BinaryOperator::Sub, RetTy,
3107 CostKind);
3108 }
3109 return Cost;
3110 }
3111 case Intrinsic::maximumnum:
3112 case Intrinsic::minimumnum: {
3113 // On platform that support FMAXNUM_IEEE/FMINNUM_IEEE, we expand
3114 // maximumnum/minimumnum to
3115 // ARG0 = fcanonicalize ARG0, ARG0 // to quiet ARG0
3116 // ARG1 = fcanonicalize ARG1, ARG1 // to quiet ARG1
3117 // RESULT = MAXNUM_IEEE ARG0, ARG1 // or MINNUM_IEEE
3118 // FIXME: In LangRef, we claimed FMAXNUM has the same behaviour of
3119 // FMAXNUM_IEEE, while the backend hasn't migrated the code yet.
3120 // Finally, we will remove FMAXNUM_IEEE and FMINNUM_IEEE.
3121 int IeeeISD =
3122 IID == Intrinsic::maximumnum ? ISD::FMAXNUM_IEEE : ISD::FMINNUM_IEEE;
3123 if (TLI->isOperationLegal(IeeeISD, LT.second)) {
3124 IntrinsicCostAttributes FCanonicalizeAttrs(Intrinsic::canonicalize,
3125 RetTy, Tys[0]);
3126 InstructionCost FCanonicalizeCost =
3127 thisT()->getIntrinsicInstrCost(FCanonicalizeAttrs, CostKind);
3128 return LT.first + FCanonicalizeCost * 2;
3129 }
3130 break;
3131 }
3132 case Intrinsic::clmul: {
3133 // This cost model should match the expansion in
3134 // TargetLowering::expandCLMUL.
3135 unsigned BW = RetTy->getScalarSizeInBits();
3136 InstructionCost AndCost =
3137 thisT()->getArithmeticInstrCost(Instruction::And, RetTy, CostKind);
3138 InstructionCost OrCost =
3139 thisT()->getArithmeticInstrCost(Instruction::Or, RetTy, CostKind);
3140 InstructionCost XorCost =
3141 thisT()->getArithmeticInstrCost(Instruction::Xor, RetTy, CostKind);
3142 InstructionCost MulCost =
3143 thisT()->getArithmeticInstrCost(Instruction::Mul, RetTy, CostKind);
3144
3145 // When the multiplication with holes approach is used, it splits the
3146 // operands into S phases (the smallest stride with ceil(BW/S) <= 2^S) and
3147 // emits S*S MULs, 3*S ANDs, S*(S-1) XORs and S-1 ORs.
3148 //
3149 // * BW <= 8 uses S = 2
3150 // * BW <= 24 uses S = 3
3151 // * BW <= 64 uses S = 4
3152 // * BW <= 160 uses S = 5
3153 // * BW <= 384 uses S = 6
3154 unsigned S = 1;
3155 while (S < 32 && divideCeil(BW, S) > (1u << S))
3156 ++S;
3157
3158 // The naive algorithm usually uses AND+MUL+XOR per bit.
3159 unsigned NaiveCost = 3 * BW;
3160 unsigned HolesCost = S * S + 3 * S + S * (S - 1) + (S - 1);
3161
3162 if (HolesCost < NaiveCost &&
3164 TLI->getValueType(DL, RetTy))) {
3165 return S * S * MulCost + 3 * S * AndCost + S * (S - 1) * XorCost +
3166 (S - 1) * OrCost;
3167 }
3168
3169 InstructionCost PerBitCostMul = AndCost + MulCost + XorCost;
3170 InstructionCost PerBitCostBittest =
3171 AndCost +
3172 thisT()->getCmpSelInstrCost(BinaryOperator::Select, RetTy, RetTy,
3174 thisT()->getCmpSelInstrCost(Instruction::ICmp, RetTy, RetTy,
3176 InstructionCost PerBitCost = std::min(PerBitCostMul, PerBitCostBittest);
3177 return BW * PerBitCost;
3178 }
3179 case Intrinsic::smulh:
3180 case Intrinsic::umulh: {
3181 unsigned BW = RetTy->getScalarSizeInBits();
3182 Type *WideTy = RetTy->getWithNewBitWidth(BW * 2);
3183 bool IsSigned = IID == Intrinsic::smulh;
3184 unsigned ExtOp = IsSigned ? Instruction::SExt : Instruction::ZExt;
3186 Cost +=
3187 2 * thisT()->getCastInstrCost(ExtOp, WideTy, RetTy,
3189 Cost +=
3190 thisT()->getArithmeticInstrCost(Instruction::Mul, WideTy, CostKind);
3191 Cost += thisT()->getArithmeticInstrCost(
3192 Instruction::LShr, WideTy, CostKind, {TTI::OK_AnyValue, TTI::OP_None},
3194 Cost += thisT()->getCastInstrCost(Instruction::Trunc, RetTy, WideTy,
3196 return Cost;
3197 }
3198 default:
3199 break;
3200 }
3201
3202 // Else, assume that we need to scalarize this intrinsic. For math builtins
3203 // this will emit a costly libcall, adding call overhead and spills. Make it
3204 // very expensive.
3205 if (isVectorizedTy(RetTy)) {
3206 ArrayRef<Type *> RetVTys = getContainedTypes(RetTy);
3207
3208 // Scalable vectors cannot be scalarized, so return Invalid.
3209 if (any_of(concat<Type *const>(RetVTys, Tys),
3210 [](Type *Ty) { return isa<ScalableVectorType>(Ty); }))
3212
3213 InstructionCost ScalarizationCost = ScalarizationCostPassed;
3214 if (!SkipScalarizationCost) {
3215 ScalarizationCost = 0;
3216 for (Type *RetVTy : RetVTys) {
3217 ScalarizationCost += getScalarizationOverhead(
3218 cast<VectorType>(RetVTy), /*Insert=*/true,
3219 /*Extract=*/false, CostKind);
3220 }
3221 }
3222
3223 unsigned ScalarCalls = getVectorizedTypeVF(RetTy).getFixedValue();
3224 SmallVector<Type *, 4> ScalarTys;
3225 for (Type *Ty : Tys) {
3226 if (Ty->isVectorTy())
3227 Ty = Ty->getScalarType();
3228 ScalarTys.push_back(Ty);
3229 }
3230 IntrinsicCostAttributes Attrs(IID, toScalarizedTy(RetTy), ScalarTys, FMF);
3231 InstructionCost ScalarCost =
3232 thisT()->getIntrinsicInstrCost(Attrs, CostKind);
3233 for (Type *Ty : Tys) {
3234 if (auto *VTy = dyn_cast<VectorType>(Ty)) {
3235 if (!ICA.skipScalarizationCost())
3236 ScalarizationCost += getScalarizationOverhead(
3237 VTy, /*Insert*/ false, /*Extract*/ true, CostKind);
3238 ScalarCalls = std::max(ScalarCalls,
3239 cast<FixedVectorType>(VTy)->getNumElements());
3240 }
3241 }
3242 return ScalarCalls * ScalarCost + ScalarizationCost;
3243 }
3244
3245 // This is going to be turned into a library call, make it expensive.
3246 return SingleCallCost;
3247 }
3248
3249 /// Get memory intrinsic cost based on arguments.
3252 TTI::TargetCostKind CostKind) const override {
3253 unsigned Id = MICA.getID();
3254 Type *DataTy = MICA.getDataType();
3255 bool VariableMask = MICA.getVariableMask();
3256 Align Alignment = MICA.getAlignment();
3257
3258 switch (Id) {
3259 case Intrinsic::experimental_vp_strided_load:
3260 case Intrinsic::experimental_vp_strided_store: {
3261 unsigned Opcode = Id == Intrinsic::experimental_vp_strided_load
3262 ? Instruction::Load
3263 : Instruction::Store;
3264 // For a target without strided memory operations (or for an illegal
3265 // operation type on one which does), assume we lower to a gather/scatter
3266 // operation. (Which may in turn be scalarized.)
3267 return getCommonMaskedMemoryOpCost(Opcode, DataTy, Alignment,
3268 VariableMask, true, CostKind);
3269 }
3270 case Intrinsic::masked_scatter:
3271 case Intrinsic::masked_gather:
3272 case Intrinsic::vp_scatter:
3273 case Intrinsic::vp_gather: {
3274 unsigned Opcode = (MICA.getID() == Intrinsic::masked_gather ||
3275 MICA.getID() == Intrinsic::vp_gather)
3276 ? Instruction::Load
3277 : Instruction::Store;
3278
3279 return getCommonMaskedMemoryOpCost(Opcode, DataTy, Alignment,
3280 VariableMask, true, CostKind);
3281 }
3282 case Intrinsic::vp_load:
3283 case Intrinsic::vp_store:
3285 case Intrinsic::masked_load:
3286 case Intrinsic::masked_store: {
3287 unsigned Opcode =
3288 Id == Intrinsic::masked_load ? Instruction::Load : Instruction::Store;
3289 // TODO: Pass on AddressSpace when we have test coverage.
3290 return getCommonMaskedMemoryOpCost(Opcode, DataTy, Alignment, true, false,
3291 CostKind);
3292 }
3293 case Intrinsic::masked_compressstore:
3294 case Intrinsic::masked_expandload: {
3295 unsigned Opcode = MICA.getID() == Intrinsic::masked_expandload
3296 ? Instruction::Load
3297 : Instruction::Store;
3298 // Treat expand load/compress store as gather/scatter operation.
3299 // TODO: implement more precise cost estimation for these intrinsics.
3300 return getCommonMaskedMemoryOpCost(Opcode, DataTy, Alignment,
3301 VariableMask,
3302 /*IsGatherScatter*/ true, CostKind);
3303 }
3304 case Intrinsic::vp_load_ff:
3306 case Intrinsic::speculative_load:
3307 // Speculative loads are lowered to regular loads of the full type.
3308 return thisT()->getMemoryOpCost(Instruction::Load, DataTy, Alignment,
3309 MICA.getAddressSpace(), CostKind);
3310 default:
3311 llvm_unreachable("unexpected intrinsic");
3312 }
3313 }
3314
3315 /// Compute a cost of the given call instruction.
3316 ///
3317 /// Compute the cost of calling function F with return type RetTy and
3318 /// argument types Tys. F might be nullptr, in this case the cost of an
3319 /// arbitrary call with the specified signature will be returned.
3320 /// This is used, for instance, when we estimate call of a vector
3321 /// counterpart of the given function.
3322 /// \param F Called function, might be nullptr.
3323 /// \param RetTy Return value types.
3324 /// \param Tys Argument types.
3325 /// \returns The cost of Call instruction.
3328 TTI::TargetCostKind CostKind) const override {
3329 return 10;
3330 }
3331
3332 unsigned getNumberOfParts(Type *Tp) const override {
3333 std::pair<InstructionCost, MVT> LT = getTypeLegalizationCost(Tp);
3334 if (!LT.first.isValid())
3335 return 0;
3336 // Try to find actual number of parts for non-power-of-2 elements as
3337 // ceil(num-of-elements/num-of-subtype-elements).
3338 if (auto *FTp = dyn_cast<FixedVectorType>(Tp);
3339 FTp && LT.second.isFixedLengthVector() &&
3340 !has_single_bit(FTp->getNumElements())) {
3341 if (auto *SubTp = dyn_cast_if_present<FixedVectorType>(
3342 EVT(LT.second).getTypeForEVT(Tp->getContext()));
3343 SubTp && SubTp->getElementType() == FTp->getElementType())
3344 return divideCeil(FTp->getNumElements(), SubTp->getNumElements());
3345 }
3346 return LT.first.getValue();
3347 }
3348
3351 TTI::TargetCostKind) const override {
3352 return 0;
3353 }
3354
3355 /// Try to calculate arithmetic and shuffle op costs for reduction intrinsics.
3356 /// We're assuming that reduction operation are performing the following way:
3357 ///
3358 /// %val1 = shufflevector<n x t> %val, <n x t> %undef,
3359 /// <n x i32> <i32 n/2, i32 n/2 + 1, ..., i32 n, i32 undef, ..., i32 undef>
3360 /// \----------------v-------------/ \----------v------------/
3361 /// n/2 elements n/2 elements
3362 /// %red1 = op <n x t> %val, <n x t> val1
3363 /// After this operation we have a vector %red1 where only the first n/2
3364 /// elements are meaningful, the second n/2 elements are undefined and can be
3365 /// dropped. All other operations are actually working with the vector of
3366 /// length n/2, not n, though the real vector length is still n.
3367 /// %val2 = shufflevector<n x t> %red1, <n x t> %undef,
3368 /// <n x i32> <i32 n/4, i32 n/4 + 1, ..., i32 n/2, i32 undef, ..., i32 undef>
3369 /// \----------------v-------------/ \----------v------------/
3370 /// n/4 elements 3*n/4 elements
3371 /// %red2 = op <n x t> %red1, <n x t> val2 - working with the vector of
3372 /// length n/2, the resulting vector has length n/4 etc.
3373 ///
3374 /// The cost model should take into account that the actual length of the
3375 /// vector is reduced on each iteration.
3378 // Targets must implement a default value for the scalable case, since
3379 // we don't know how many lanes the vector has.
3382
3383 Type *ScalarTy = Ty->getElementType();
3384 unsigned NumVecElts = cast<FixedVectorType>(Ty)->getNumElements();
3385 if ((Opcode == Instruction::Or || Opcode == Instruction::And) &&
3386 ScalarTy == IntegerType::getInt1Ty(Ty->getContext()) &&
3387 NumVecElts >= 2) {
3388 // Or reduction for i1 is represented as:
3389 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
3390 // %res = cmp ne iReduxWidth %val, 0
3391 // And reduction for i1 is represented as:
3392 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
3393 // %res = cmp eq iReduxWidth %val, 11111
3394 Type *ValTy = IntegerType::get(Ty->getContext(), NumVecElts);
3395 return thisT()->getCastInstrCost(Instruction::BitCast, ValTy, Ty,
3397 thisT()->getCmpSelInstrCost(Instruction::ICmp, ValTy,
3400 }
3401 unsigned NumReduxLevels = Log2_32(NumVecElts);
3402 InstructionCost ArithCost = 0;
3403 InstructionCost ShuffleCost = 0;
3404 std::pair<InstructionCost, MVT> LT = thisT()->getTypeLegalizationCost(Ty);
3405 unsigned LongVectorCount = 0;
3406 unsigned MVTLen =
3407 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
3408 while (NumVecElts > MVTLen) {
3409 NumVecElts /= 2;
3410 VectorType *SubTy = FixedVectorType::get(ScalarTy, NumVecElts);
3411 ShuffleCost += thisT()->getShuffleCost(
3412 TTI::SK_ExtractSubvector, SubTy, Ty, CostKind, {}, NumVecElts, SubTy);
3413 ArithCost += thisT()->getArithmeticInstrCost(Opcode, SubTy, CostKind);
3414 Ty = SubTy;
3415 ++LongVectorCount;
3416 }
3417
3418 NumReduxLevels -= LongVectorCount;
3419
3420 // The minimal length of the vector is limited by the real length of vector
3421 // operations performed on the current platform. That's why several final
3422 // reduction operations are performed on the vectors with the same
3423 // architecture-dependent length.
3424
3425 // By default reductions need one shuffle per reduction level.
3426 ShuffleCost +=
3427 NumReduxLevels * thisT()->getShuffleCost(TTI::SK_PermuteSingleSrc, Ty,
3428 Ty, CostKind, {}, 0, Ty);
3429 ArithCost +=
3430 NumReduxLevels * thisT()->getArithmeticInstrCost(Opcode, Ty, CostKind);
3431 return ShuffleCost + ArithCost +
3432 thisT()->getVectorInstrCost(Instruction::ExtractElement, Ty,
3433 CostKind, 0, nullptr, nullptr);
3434 }
3435
3436 /// Try to calculate the cost of performing strict (in-order) reductions,
3437 /// which involves doing a sequence of floating point additions in lane
3438 /// order, starting with an initial value. For example, consider a scalar
3439 /// initial value 'InitVal' of type float and a vector of type <4 x float>:
3440 ///
3441 /// Vector = <float %v0, float %v1, float %v2, float %v3>
3442 ///
3443 /// %add1 = %InitVal + %v0
3444 /// %add2 = %add1 + %v1
3445 /// %add3 = %add2 + %v2
3446 /// %add4 = %add3 + %v3
3447 ///
3448 /// As a simple estimate we can say the cost of such a reduction is 4 times
3449 /// the cost of a scalar FP addition. We can only estimate the costs for
3450 /// fixed-width vectors here because for scalable vectors we do not know the
3451 /// runtime number of operations.
3454 // Targets must implement a default value for the scalable case, since
3455 // we don't know how many lanes the vector has.
3458
3459 auto *VTy = cast<FixedVectorType>(Ty);
3461 VTy, /*Insert=*/false, /*Extract=*/true, CostKind);
3462 InstructionCost ArithCost = thisT()->getArithmeticInstrCost(
3463 Opcode, VTy->getElementType(), CostKind);
3464 ArithCost *= VTy->getNumElements();
3465
3466 return ExtractCost + ArithCost;
3467 }
3468
3471 std::optional<FastMathFlags> FMF,
3472 TTI::TargetCostKind CostKind) const override {
3473 assert(Ty && "Unknown reduction vector type");
3475 return getOrderedReductionCost(Opcode, Ty, CostKind);
3476 return getTreeReductionCost(Opcode, Ty, CostKind);
3477 }
3478
3479 /// Try to calculate op costs for min/max reduction operations.
3480 /// \param CondTy Conditional type for the Select instruction.
3483 TTI::TargetCostKind CostKind) const override {
3484 // Targets must implement a default value for the scalable case, since
3485 // we don't know how many lanes the vector has.
3488
3489 Type *ScalarTy = Ty->getElementType();
3490 unsigned NumVecElts = cast<FixedVectorType>(Ty)->getNumElements();
3491 unsigned NumReduxLevels = Log2_32(NumVecElts);
3492 InstructionCost MinMaxCost = 0;
3493 InstructionCost ShuffleCost = 0;
3494 std::pair<InstructionCost, MVT> LT = thisT()->getTypeLegalizationCost(Ty);
3495 unsigned LongVectorCount = 0;
3496 unsigned MVTLen =
3497 LT.second.isVector() ? LT.second.getVectorNumElements() : 1;
3498 while (NumVecElts > MVTLen) {
3499 NumVecElts /= 2;
3500 auto *SubTy = FixedVectorType::get(ScalarTy, NumVecElts);
3501
3502 ShuffleCost += thisT()->getShuffleCost(
3503 TTI::SK_ExtractSubvector, SubTy, Ty, CostKind, {}, NumVecElts, SubTy);
3504
3505 IntrinsicCostAttributes Attrs(IID, SubTy, {SubTy, SubTy}, FMF);
3506 MinMaxCost += getIntrinsicInstrCost(Attrs, CostKind);
3507 Ty = SubTy;
3508 ++LongVectorCount;
3509 }
3510
3511 NumReduxLevels -= LongVectorCount;
3512
3513 // The minimal length of the vector is limited by the real length of vector
3514 // operations performed on the current platform. That's why several final
3515 // reduction opertions are perfomed on the vectors with the same
3516 // architecture-dependent length.
3517 ShuffleCost +=
3518 NumReduxLevels * thisT()->getShuffleCost(TTI::SK_PermuteSingleSrc, Ty,
3519 Ty, CostKind, {}, 0, Ty);
3520 IntrinsicCostAttributes Attrs(IID, Ty, {Ty, Ty}, FMF);
3521 MinMaxCost += NumReduxLevels * getIntrinsicInstrCost(Attrs, CostKind);
3522 // The last min/max should be in vector registers and we counted it above.
3523 // So just need a single extractelement.
3524 return ShuffleCost + MinMaxCost +
3525 thisT()->getVectorInstrCost(Instruction::ExtractElement, Ty,
3526 CostKind, 0, nullptr, nullptr);
3527 }
3528
3530 getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy,
3531 VectorType *Ty, std::optional<FastMathFlags> FMF,
3532 TTI::TargetCostKind CostKind) const override {
3533 if (auto *FTy = dyn_cast<FixedVectorType>(Ty);
3534 FTy && IsUnsigned && Opcode == Instruction::Add &&
3535 FTy->getElementType() == IntegerType::getInt1Ty(Ty->getContext())) {
3536 // Represent vector_reduce_add(ZExt(<n x i1>)) as
3537 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
3538 auto *IntTy =
3539 IntegerType::get(ResTy->getContext(), FTy->getNumElements());
3540 IntrinsicCostAttributes ICA(Intrinsic::ctpop, IntTy, {IntTy},
3541 FMF ? *FMF : FastMathFlags());
3542 return thisT()->getCastInstrCost(Instruction::BitCast, IntTy, FTy,
3544 thisT()->getIntrinsicInstrCost(ICA, CostKind);
3545 }
3546 // Without any native support, this is equivalent to the cost of
3547 // vecreduce.opcode(ext(Ty A)).
3548 VectorType *ExtTy = VectorType::get(ResTy, Ty);
3549 InstructionCost RedCost =
3550 thisT()->getArithmeticReductionCost(Opcode, ExtTy, FMF, CostKind);
3551 InstructionCost ExtCost = thisT()->getCastInstrCost(
3552 IsUnsigned ? Instruction::ZExt : Instruction::SExt, ExtTy, Ty,
3554
3555 return RedCost + ExtCost;
3556 }
3557
3559 getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy,
3560 VectorType *Ty,
3561 TTI::TargetCostKind CostKind) const override {
3562 // Without any native support, this is equivalent to the cost of
3563 // vecreduce.add(mul(ext(Ty A), ext(Ty B))) or
3564 // vecreduce.add(mul(A, B)).
3565 assert((RedOpcode == Instruction::Add || RedOpcode == Instruction::Sub) &&
3566 "The reduction opcode is expected to be Add or Sub.");
3567 VectorType *ExtTy = VectorType::get(ResTy, Ty);
3568 InstructionCost RedCost = thisT()->getArithmeticReductionCost(
3569 RedOpcode, ExtTy, std::nullopt, CostKind);
3570 InstructionCost ExtCost = thisT()->getCastInstrCost(
3571 IsUnsigned ? Instruction::ZExt : Instruction::SExt, ExtTy, Ty,
3573
3574 InstructionCost MulCost =
3575 thisT()->getArithmeticInstrCost(Instruction::Mul, ExtTy, CostKind);
3576
3577 return RedCost + MulCost + 2 * ExtCost;
3578 }
3579
3581 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
3583 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
3585 std::optional<FastMathFlags> FMF) const override {
3586 unsigned EltSizeAcc = AccumType->getScalarSizeInBits();
3587 unsigned EltSizeInA = InputTypeA->getScalarSizeInBits();
3588 unsigned Ratio = EltSizeAcc / EltSizeInA;
3589 if (VF.getKnownMinValue() <= Ratio || VF.getKnownMinValue() % Ratio != 0 ||
3590 EltSizeAcc % EltSizeInA != 0 || (BinOp && InputTypeA != InputTypeB))
3592
3593 Type *InputVectorType = VectorType::get(InputTypeA, VF);
3594 Type *ExtInputVectorType = VectorType::get(AccumType, VF);
3595 Type *AccumVectorType =
3596 VectorType::get(AccumType, VF.divideCoefficientBy(Ratio));
3597
3598 InstructionCost ExtendCostA = 0;
3600 ExtendCostA = getCastInstrCost(
3602 ExtInputVectorType, InputVectorType, TTI::CastContextHint::None,
3603 CostKind);
3604
3605 // TODO: add cost of extracting subvectors from the source vector that
3606 // is to be partially reduced.
3607 InstructionCost ReductionOpCost =
3608 Ratio * getArithmeticInstrCost(Opcode, AccumVectorType, CostKind);
3609
3610 if (!BinOp)
3611 return ExtendCostA + ReductionOpCost;
3612
3613 InstructionCost ExtendCostB = 0;
3615 ExtendCostB = getCastInstrCost(
3617 ExtInputVectorType, InputVectorType, TTI::CastContextHint::None,
3618 CostKind);
3619 return ExtendCostA + ExtendCostB + ReductionOpCost +
3620 getArithmeticInstrCost(*BinOp, ExtInputVectorType, CostKind);
3621 }
3622
3624
3625 /// @}
3626};
3627
3628/// Concrete BasicTTIImpl that can be used if no further customization
3629/// is needed.
3630class BasicTTIImpl : public BasicTTIImplBase<BasicTTIImpl> {
3631 using BaseT = BasicTTIImplBase<BasicTTIImpl>;
3632
3633 friend class BasicTTIImplBase<BasicTTIImpl>;
3634
3635 const TargetSubtargetInfo *ST;
3636 const TargetLoweringBase *TLI;
3637
3638 const TargetSubtargetInfo *getST() const { return ST; }
3639 const TargetLoweringBase *getTLI() const { return TLI; }
3640
3641public:
3642 LLVM_ABI explicit BasicTTIImpl(const TargetMachine *TM, const Function &F);
3643};
3644
3645} // end namespace llvm
3646
3647#endif // LLVM_CODEGEN_BASICTTIIMPL_H
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
This file implements the BitVector class.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
#define LLVM_ABI
Definition Compiler.h:215
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
#define T
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
#define P(N)
SI Fold Operands
This file contains some templates that are useful if you are working with the STL at all.
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
This file describes how to lower LLVM code to machine code.
This file provides helpers for the implementation of a TargetTransformInfo-conforming class.
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1350
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1205
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
bool slt(const APInt &RHS) const
Signed less than comparison.
Definition APInt.h:1134
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
an instruction to allocate memory on the stack
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
size_t size() const
Get the array size.
Definition ArrayRef.h:141
ArrayRef< T > drop_back(size_t N=1) const
Drop the last N elements of the array.
Definition ArrayRef.h:200
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
InstructionCost getFPOpCost(Type *Ty) const override
bool preferToKeepConstantsAttached(const Instruction &Inst, const Function &Fn) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
bool isIndexedLoadLegal(TTI::MemIndexedMode M, Type *Ty) const override
unsigned getCallerAllocaCost(const CallBase *CB, const AllocaInst *AI) const override
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
bool shouldBuildLookupTables() const override
bool isNoopAddrSpaceCast(unsigned FromAS, unsigned ToAS) const override
bool isProfitableToHoist(Instruction *I) const override
unsigned getNumberOfParts(Type *Tp) const override
unsigned getMinPrefetchStride(unsigned NumMemAccesses, unsigned NumStridedMemAccesses, unsigned NumPrefetches, bool HasCall) const override
bool useAA() const override
unsigned getPrefetchDistance() const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
InstructionCost getOperandsScalarizationOverhead(ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction's operands.
bool isLegalAddScalableImmediate(int64_t Imm) const override
bool haveFastClmul(IntegerType *Ty) const override
unsigned getAssumedAddrSpace(const Value *V) const override
std::optional< Value * > simplifyDemandedUseBitsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedMask, KnownBits &Known, bool &KnownBitsComputed) const override
bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const override
bool addrspacesMayAlias(unsigned AS0, unsigned AS1) const override
bool areInlineCompatible(const Function *Caller, const Function *Callee) const override
bool isIndexedStoreLegal(TTI::MemIndexedMode M, Type *Ty) const override
bool haveFastSqrt(Type *Ty) const override
bool collectFlatAddressOperands(SmallVectorImpl< int > &OpIndexes, Intrinsic::ID IID) const override
unsigned getEstimatedNumberOfCaseClusters(const SwitchInst &SI, unsigned &JumpTableSize, ProfileSummaryInfo *PSI, BlockFrequencyInfo *BFI) const override
unsigned getStoreMinimumVF(unsigned VF, Type *ScalarMemTy, Type *ScalarValTy, Align Alignment, unsigned AddrSpace) const override
Value * rewriteIntrinsicWithAddressSpace(IntrinsicInst *II, Value *OldV, Value *NewV) const override
unsigned adjustInliningThreshold(const CallBase *CB) const override
unsigned getInliningThresholdMultiplier() const override
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, Value *Scalar, ArrayRef< std::tuple< Value *, User *, int > > ScalarUserAndIdx, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
int64_t getPreferredLargeGEPBaseOffset(int64_t MinOffset, int64_t MaxOffset)
bool shouldBuildRelLookupTables() const override
bool isTargetIntrinsicWithStructReturnOverloadAtField(Intrinsic::ID ID, int RetIdx) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
InstructionCost getVectorInstrCost(const Instruction &I, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const override
unsigned getEpilogueVectorizationMinVF() const override
InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const override
InstructionCost getVectorSplitCost() const
bool isTruncateFree(Type *Ty1, Type *Ty2) const override
unsigned getFlatAddressSpace() const override
InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const override
Compute a cost of the given call instruction.
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
InstructionCost getTreeReductionCost(unsigned Opcode, VectorType *Ty, TTI::TargetCostKind CostKind) const
Try to calculate arithmetic and shuffle op costs for reduction intrinsics.
~BasicTTIImplBase() override=default
std::pair< const Value *, unsigned > getPredicatedAddrSpace(const Value *V) const override
unsigned getMaxPrefetchIterationsAhead() const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getTypeBasedIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const
Get intrinsic cost based on argument types.
bool hasBranchDivergence(const Function *F=nullptr) const override
InstructionCost getOrderedReductionCost(unsigned Opcode, VectorType *Ty, TTI::TargetCostKind CostKind) const
Try to calculate the cost of performing strict (in-order) reductions, which involves doing a sequence...
std::optional< unsigned > getCacheAssociativity(TargetTransformInfo::CacheLevel Level) const override
bool shouldPrefetchAddressSpace(unsigned AS) const override
bool allowsMisalignedMemoryAccesses(LLVMContext &Context, unsigned BitWidth, unsigned AddressSpace, Align Alignment, unsigned *Fast) const override
unsigned getCacheLineSize() const override
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
bool shouldDropLSRSolutionIfLessProfitable() const override
int getInlinerVectorBonusPercent() const override
InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
std::pair< InstructionCost, MVT > getTypeLegalizationCost(Type *Ty) const
Estimate the cost of type-legalization and the legalized type.
InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const override
bool isLegalAddImmediate(int64_t imm) const override
InstructionCost getReplicationShuffleCost(Type *EltTy, int ReplicationFactor, int VF, const APInt &DemandedDstElts, TTI::TargetCostKind CostKind) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
bool isProfitableLSRChainElement(Instruction *I) const override
bool isValidAddrSpaceCast(unsigned FromAS, unsigned ToAS) const override
bool isTargetIntrinsicWithOverloadTypeAtArg(Intrinsic::ID ID, int OpdIdx) const override
bool isTargetIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx) const override
std::optional< unsigned > getVScaleForTuning() const override
InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const override
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const override
bool isFCmpOrdCheaperThanFCmpZero(Type *Ty) const override
InstructionCost getScalarizationOverhead(VectorType *RetTy, ArrayRef< const Value * > Args, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const
Estimate the overhead of scalarizing the inputs and outputs of an instruction, with return type RetTy...
TailFoldingStyle getPreferredTailFoldingStyle() const override
std::optional< unsigned > getCacheSize(TargetTransformInfo::CacheLevel Level) const override
bool isLegalICmpImmediate(int64_t imm) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE, AssumptionCache &AC, TargetLibraryInfo *LibInfo, HardwareLoopInfo &HWLoopInfo) const override
unsigned getRegUsageForType(Type *Ty) const override
InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const override
Get memory intrinsic cost based on arguments.
BasicTTIImplBase(const TargetMachine *TM, const DataLayout &DL)
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
bool isTypeLegal(Type *Ty) const override
bool enableWritePrefetching() const override
bool isLSRCostLess(const TTI::LSRCost &C1, const TTI::LSRCost &C2) const override
InstructionCost getScalarizationOverhead(VectorType *InTy, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
Helper wrapper for the DemandedElts variant of getScalarizationOverhead.
InstructionCost getBranchMispredictPenalty() const override
bool isNumRegsMajorCostOfLSR() const override
LLVM_ABI BasicTTIImpl(const TargetMachine *TM, const Function &F)
size_type count() const
Returns the number of bits which are set.
Definition BitVector.h:181
BitVector & set()
Set all bits in the bitvector.
Definition BitVector.h:366
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
static Type * makeCmpResultType(Type *opnd_type)
Create a result type for fcmp/icmp.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static CmpInst::Predicate getGTPredicate(Intrinsic::ID ID)
static CmpInst::Predicate getLTPredicate(Intrinsic::ID ID)
This class represents a range of values.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
constexpr bool isVector() const
One or more elements.
Definition TypeSize.h:320
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:305
constexpr bool isScalar() const
Exactly one element.
Definition TypeSize.h:316
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
Container class for subtarget features.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
AttributeList getAttributes() const
Return the attribute list for this Function.
Definition Function.h:329
The core instruction combiner logic.
static InstructionCost getInvalid(CostType Val=0)
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
unsigned getBitWidth() const
Get the number of bits in this IntegerType.
const SmallVectorImpl< Type * > & getArgTypes() const
const SmallVectorImpl< const Value * > & getArgs() const
InstructionCost getScalarizationCost() const
const IntrinsicInst * getInst() const
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
const FeatureBitset & getFeatureBits() const
Machine Value Type.
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Information for memory intrinsic cost model.
The optimization diagnostic interface.
LLVM_ABI void emit(DiagnosticInfoOptimizationBase &OptDiag)
Output the remark via the diagnostic handler and to the optimization record file.
Diagnostic information for applied optimization remarks.
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
Definition Type.cpp:887
Analysis providing profile information.
This class represents an analyzed expression in the program.
The main scalar evolution driver.
static LLVM_ABI bool isZeroEltSplatMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses all elements with the same value as the first element of exa...
static LLVM_ABI bool isSpliceMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is a splice mask, concatenating the two inputs together and then ext...
static LLVM_ABI bool isSelectMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from its source vectors without lane crossings.
static LLVM_ABI bool isExtractSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &Index)
Return true if this shuffle mask is an extract subvector mask.
static LLVM_ABI bool isReverseMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask swaps the order of elements from exactly one source vector.
static LLVM_ABI bool isTransposeMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask is a transpose mask.
static LLVM_ABI bool isInsertSubvectorMask(ArrayRef< int > Mask, int NumSrcElts, int &NumSubElts, int &Index)
Return true if this shuffle mask is an insert subvector mask.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
static StackOffset getScalable(int64_t Scalable)
Definition TypeSize.h:40
static StackOffset getFixed(int64_t Fixed)
Definition TypeSize.h:39
static LLVM_ABI StructType * create(LLVMContext &Context, StringRef Name)
This creates an identified struct.
Definition Type.cpp:662
Multiway switch.
Provides information about what library functions are available for the current target.
This base class for TargetLowering contains the SelectionDAG-independent parts that can be used from ...
bool isOperationExpand(unsigned Op, EVT VT) const
Return true if the specified operation is illegal on this target or unlikely to be made legal with cu...
int InstructionOpcodeToISD(unsigned Opcode) const
Get the ISD node that corresponds to the Instruction class opcode.
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
LegalizeAction
This enum indicates whether operations are valid for a target, and if not, what action should be used...
virtual bool preferSelectsOverBooleanArithmetic(EVT VT) const
Should we prefer selects to doing arithmetic on boolean types.
virtual bool isFreeAddrSpaceCast(const DataLayout &DL, unsigned SrcAS, unsigned DestAS) const
Returns true if a cast from SrcAS to DestAS is "cheap", such that e.g.
virtual bool isZExtFree(Type *FromTy, Type *ToTy) const
Return true if any actual instruction that defines a value of type FromTy implicitly zero-extends the...
virtual bool isSuitableForJumpTable(const SwitchInst *SI, uint64_t NumCases, uint64_t Range, ProfileSummaryInfo *PSI, BlockFrequencyInfo *BFI) const
Return true if lowering to a jump table is suitable for a set of case clusters which may contain NumC...
virtual bool areJTsAllowed(const Function *Fn) const
Return true if lowering to a jump table is allowed.
bool isOperationLegalOrPromote(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal using promotion.
LegalizeAction getTruncStoreAction(EVT ValVT, EVT MemVT, Align Alignment, unsigned AddrSpace) const
Return how this store with truncation should be treated: either it is legal, needs to be promoted to ...
bool isOperationCustom(unsigned Op, EVT VT) const
Return true if the operation uses custom lowering, regardless of whether the type is legal or not.
bool isSuitableForBitTests(const DenseMap< const BasicBlock *, unsigned int > &DestCmps, const APInt &Low, const APInt &High, const DataLayout &DL) const
Return true if lowering to a bit test is suitable for a set of case clusters which contains NumDests ...
virtual bool isTruncateFree(Type *FromTy, Type *ToTy) const
Return true if it's free to truncate a value of type FromTy to type ToTy.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
bool isOperationLegalOrCustom(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
LegalizeAction getLoadAction(EVT ValVT, EVT MemVT, Align Alignment, unsigned AddrSpace, unsigned ExtType, bool Atomic) const
Return how this load with extension should be treated: either it is legal, needs to be promoted to a ...
LegalizeKind getTypeConversion(LLVMContext &Context, EVT VT) const
Return pair that represents the legalization kind (first) that needs to happen to EVT (second) in ord...
LegalizeTypeAction getTypeAction(LLVMContext &Context, EVT VT) const
Return how we should legalize values of this type, either it is already legal (return 'Legal') or we ...
bool isLoadLegal(EVT ValVT, EVT MemVT, Align Alignment, unsigned AddrSpace, unsigned ExtType, bool Atomic) const
Return true if the specified load with extension is legal on this target.
virtual bool isFAbsFree(EVT VT) const
Return true if an fabs operation is free to the point where it is never worthwhile to replace it with...
bool isOperationLegalOrCustomOrPromote(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
std::pair< LegalizeTypeAction, EVT > LegalizeKind
LegalizeKind holds the legalization kind that needs to happen to EVT in order to type-legalize it.
Primary interface to the complete machine description for the target machine.
bool isPositionIndependent() const
const Triple & getTargetTriple() const
virtual const TargetSubtargetInfo * getSubtargetImpl(const Function &) const
Virtual method implemented by subclasses that returns a reference to that target's TargetSubtargetInf...
CodeModel::Model getCodeModel() const
Returns the code model.
TargetSubtargetInfo - Generic base class for all target subtargets.
virtual const FeatureBitset & getInlineMustMatchFeatures() const =0
Target features where all mismatches prevent inlining.
virtual const FeatureBitset & getInlineInverseFeatures() const =0
Target features where the callee may have an additional feature, instead of the caller.
virtual const FeatureBitset & getInlineIgnoreFeatures() const =0
Target features to ignore for inline compatibility check.
virtual bool isProfitableLSRChainElement(Instruction *I) const
virtual TailFoldingStyle getPreferredTailFoldingStyle() const
virtual const DataLayout & getDataLayout() const
virtual std::optional< unsigned > getCacheAssociativity(TargetTransformInfo::CacheLevel Level) const
virtual InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CtxI=nullptr) const
virtual std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const
virtual bool shouldDropLSRSolutionIfLessProfitable() const
virtual bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE, AssumptionCache &AC, TargetLibraryInfo *LibInfo, HardwareLoopInfo &HWLoopInfo) const
virtual std::optional< Value * > simplifyDemandedUseBitsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedMask, KnownBits &Known, bool &KnownBitsComputed) const
virtual bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const
virtual std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const
virtual unsigned getEpilogueVectorizationMinVF() const
virtual InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
virtual bool isLoweredToCall(const Function *F) const
virtual InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const
virtual bool isLSRCostLess(const TTI::LSRCost &C1, const TTI::LSRCost &C2) const
virtual InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I) const
virtual InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const
virtual InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, const Instruction *I) const
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
static LLVM_ABI OperandValueInfo getOperandInfo(const Value *V)
Collect properties of V used in cost analysis, e.g. OP_PowerOf2.
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
llvm::VectorInstrContext VectorInstrContext
@ TCC_Expensive
The cost of a 'div' instruction on x86.
@ TCC_Basic
The cost of a typical 'add' instruction.
static LLVM_ABI Instruction::CastOps getOpcodeForPartialReductionExtendKind(PartialReductionExtendKind Kind)
Get the cast opcode for an extension kind.
MemIndexedMode
The type of load/store indexing.
static LLVM_ABI VectorInstrContext getVectorInstrContextHint(const Instruction *I)
Calculates a VectorInstrContext from I.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
@ Normal
The cast is used with a normal load/store.
CacheLevel
The possible cache levels.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
ArchType getArch() const
Get the parsed architecture type of this triple.
Definition Triple.h:514
LLVM_ABI bool isArch64Bit() const
Test whether the architecture is 64-bit.
Definition Triple.cpp:1827
bool isOSDarwin() const
Is this a "Darwin" OS (macOS, iOS, tvOS, watchOS, DriverKit, XROS, or bridgeOS).
Definition Triple.h:723
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
bool isFPOrFPVectorTy() const
Return true if this is a FP type or a vector of FP.
Definition Type.h:222
Type * getContainedType(unsigned i) const
This method is used to implement the type iterator (defined at the end of the file).
Definition Type.h:392
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
Value * getOperand(unsigned i) const
Definition User.h:207
static LLVM_ABI std::optional< unsigned > getFunctionalOpcodeForVP(Intrinsic::ID ID)
static LLVM_ABI std::optional< Intrinsic::ID > getFunctionalIntrinsicIDForVP(Intrinsic::ID ID)
static LLVM_ABI bool isVPIntrinsic(Intrinsic::ID)
static LLVM_ABI bool isVPReduction(Intrinsic::ID ID)
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
Base class of all SIMD vector types.
static VectorType * getHalfElementsVectorType(VectorType *VTy)
This static method returns a VectorType with half as many elements as the input type and the same ele...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
LLVM_ABI APInt ScaleBitMask(const APInt &A, unsigned NewBitWidth, bool MatchAllBits=false)
Splat/Merge neighboring bits to widen/narrow the bitmask represented by.
Definition APInt.cpp:3043
ISD namespace - This namespace contains an enum which represents all of the SelectionDAG node types a...
Definition ISDOpcodes.h:26
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ SMULFIX
RESULT = [US]MULFIX(LHS, RHS, SCALE) - Perform fixed point multiplication on 2 integers with the same...
Definition ISDOpcodes.h:397
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ FMODF
FMODF - Decomposes the operand into integral and fractional parts, each having the same type and sign...
@ FATAN2
FATAN2 - atan2, inspired by libm.
@ FSINCOSPI
FSINCOSPI - Compute both the sine and cosine times pi more accurately than FSINCOS(pi*x),...
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:757
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:282
@ CLMUL
Carry-less multiplication operations.
Definition ISDOpcodes.h:788
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ SSUBO
Same for subtraction.
Definition ISDOpcodes.h:355
@ BRIND
BRIND - Indirect branch.
@ BR_JT
BR_JT - Jumptable branch.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
Definition ISDOpcodes.h:546
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
Definition ISDOpcodes.h:377
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:814
@ SADDO
RESULT, BOOL = [SU]ADDO(LHS, RHS) - Overflow-aware nodes for addition.
Definition ISDOpcodes.h:351
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ FMINNUM_IEEE
FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or maximumNumber on two values,...
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ SMULO
Same for multiplication.
Definition ISDOpcodes.h:359
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ MASKED_UDIV
Masked vector arithmetic that returns poison on disabled lanes.
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:823
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ SCMP
[US]CMP - 3-way comparison of signed or unsigned integers.
Definition ISDOpcodes.h:745
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:963
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:539
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
Definition ISDOpcodes.h:368
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
MemIndexedMode
MemIndexedMode enum - This enum defines the load / store indexed addressing modes.
LLVM_ABI bool isTargetIntrinsic(ID IID)
isTargetIntrinsic - Returns true if IID is an intrinsic specific to a certain target.
DiagnosticInfoOptimizationBase::Argument NV
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicOp(Intrinsic::ID RdxID)
Returns the min/max intrinsic used when expanding a min/max reduction.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
Definition STLExtras.h:856
InstructionCost Cost
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
Type * toScalarizedTy(Type *Ty)
A helper for converting vectorized types to scalarized (non-vector) types.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
LLVM_ABI unsigned getArithmeticReductionInstruction(Intrinsic::ID RdxID)
Returns the arithmetic instruction opcode used when expanding a reduction.
bool isVectorizedTy(Type *Ty)
Returns true if Ty is a vector type or a struct of vector types where all vector types share the same...
detail::concat_range< ValueT, RangeTs... > concat(RangeTs &&...Ranges)
Returns a concatenated range across two or more ranges.
Definition STLExtras.h:1167
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
LLVM_ABI std::optional< unsigned > getPartialUnrollingThreshold()
Returns -partial-unrolling-threshold if specified.
constexpr bool has_single_bit(T Value) noexcept
Definition bit.h:149
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
ElementCount getVectorizedTypeVF(Type *Ty)
Returns the number of vector elements for a vectorized type.
LLVM_ABI ConstantRange getVScaleRange(const Function *F, unsigned BitWidth)
Determine the possible constant range of vscale with the given bit width, based on the vscale_range f...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
@ Fast
Assign the register banks as fast as possible (default).
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
ArrayRef< Type * > getContainedTypes(Type *const &Ty)
Returns the types contained in Ty.
LLVM_ABI bool isVectorizedStructTy(StructType *StructTy)
Returns true if StructTy is an unpacked literal struct where all elements are vectors of matching ele...
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
ElementCount getVectorElementCount() const
Definition ValueTypes.h:373
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
Attributes of a target dependent hardware loop.
static LLVM_ABI bool hasVectorMaskArgument(RTLIB::LibcallImpl Impl)
Returns true if the function has a vector mask argument, which is assumed to be the last argument.
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
bool AllowPeeling
Allow peeling off loop iterations.
bool AllowLoopNestsPeeling
Allow peeling off loop iterations for loop nests.
bool PeelProfiledIterations
Allow peeling basing on profile.
unsigned PeelCount
A forced peeling factor (the number of bodied of the original loop that should be peeled off before t...
Parameters that control the generic loop unrolling transformation.
bool UpperBound
Allow using trip count upper bound to unroll loops.
unsigned PartialOptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size, like OptSizeThreshold,...
unsigned PartialThreshold
The cost threshold for the unrolled loop, like Threshold, but used for partial/runtime unrolling (set...
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
unsigned OptSizeThreshold
The cost threshold for the unrolled loop when optimizing for size (set to UINT_MAX to disable).