LLVM 24.0.0git
TargetTransformInfoImpl.h
Go to the documentation of this file.
1//===- TargetTransformInfoImpl.h --------------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file provides helpers for the implementation of
10/// a TargetTransformInfo-conforming class.
11///
12//===----------------------------------------------------------------------===//
13
14#ifndef LLVM_ANALYSIS_TARGETTRANSFORMINFOIMPL_H
15#define LLVM_ANALYSIS_TARGETTRANSFORMINFOIMPL_H
16
21#include "llvm/IR/DataLayout.h"
24#include "llvm/IR/Operator.h"
26#include <optional>
27#include <utility>
28
29namespace llvm {
30
31class Function;
32
33/// Base class for use as a mix-in that aids implementing
34/// a TargetTransformInfo-compatible class.
36
37protected:
39
40 const DataLayout &DL;
41
43
44public:
46
47 // Provide value semantics. MSVC requires that we spell all of these out.
50
51 virtual const DataLayout &getDataLayout() const { return DL; }
52
53 // FIXME: It looks like this implementation is dead. All clients appear to
54 // use the (non-const) version from `TargetTransformInfoImplCRTPBase`.
55 virtual InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr,
58 Type *AccessType) const {
59 // In the basic model, we just assume that all-constant GEPs will be folded
60 // into their uses via addressing modes.
61 for (const Value *Operand : Operands)
62 if (!isa<Constant>(Operand))
63 return TTI::TCC_Basic;
64
65 return TTI::TCC_Free;
66 }
67
68 virtual InstructionCost
70 const TTI::PointersChainInfo &Info, Type *AccessTy,
71 const TTI::TargetCostKind CostKind) const {
72 llvm_unreachable("Not implemented");
73 }
74
75 virtual unsigned
78 BlockFrequencyInfo *BFI) const {
79 (void)PSI;
80 (void)BFI;
81 JTSize = 0;
82 return SI.getNumCases();
83 }
84
85 virtual InstructionCost
90
91 virtual unsigned getInliningThresholdMultiplier() const { return 1; }
93 return 8;
94 }
96 return 8;
97 }
99 // This is the value of InlineConstants::LastCallToStaticBonus before it was
100 // removed along with the introduction of this function.
101 return 15000;
102 }
103 virtual unsigned adjustInliningThreshold(const CallBase *CB) const {
104 return 0;
105 }
106 virtual unsigned getCallerAllocaCost(const CallBase *CB,
107 const AllocaInst *AI) const {
108 return 0;
109 };
110
111 virtual int getInlinerVectorBonusPercent() const { return 150; }
112
114 return TTI::TCC_Expensive;
115 }
116
117 virtual uint64_t getMaxMemIntrinsicInlineSizeThreshold() const { return 64; }
118
119 // Although this default value is arbitrary, it is not random. It is assumed
120 // that a condition that evaluates the same way by a higher percentage than
121 // this is best represented as control flow. Therefore, the default value N
122 // should be set such that the win from N% correct executions is greater than
123 // the loss from (100 - N)% mispredicted executions for the majority of
124 // intended targets.
126 return BranchProbability(99, 100);
127 }
128
129 virtual InstructionCost getBranchMispredictPenalty() const { return 0; }
130
131 virtual bool hasBranchDivergence(const Function *F = nullptr) const {
132 return false;
133 }
134
135 virtual ValueUniformity getValueUniformity(const Value *V) const {
137 }
138
139 virtual bool isValidAddrSpaceCast(unsigned FromAS, unsigned ToAS) const {
140 return false;
141 }
142
143 virtual bool addrspacesMayAlias(unsigned AS0, unsigned AS1) const {
144 return true;
145 }
146
147 virtual unsigned getFlatAddressSpace() const { return -1; }
148
150 Intrinsic::ID IID) const {
151 return false;
152 }
153
154 virtual bool isNoopAddrSpaceCast(unsigned, unsigned) const { return false; }
155
156 virtual std::pair<KnownBits, KnownBits>
157 computeKnownBitsAddrSpaceCast(unsigned ToAS, const Value &PtrOp) const {
158 const Type *PtrTy = PtrOp.getType();
159 assert(PtrTy->isPtrOrPtrVectorTy() &&
160 "expected pointer or pointer vector type");
161 unsigned FromAS = PtrTy->getPointerAddressSpace();
162
163 if (DL.isNonIntegralAddressSpace(FromAS))
164 return std::pair(KnownBits(DL.getPointerSizeInBits(FromAS)),
165 KnownBits(DL.getPointerSizeInBits(ToAS)));
166
167 KnownBits FromPtrBits;
168 if (const AddrSpaceCastInst *CastI = dyn_cast<AddrSpaceCastInst>(&PtrOp)) {
169 std::pair<KnownBits, KnownBits> KB = computeKnownBitsAddrSpaceCast(
170 CastI->getDestAddressSpace(), *CastI->getPointerOperand());
171 FromPtrBits = KB.second;
172 } else {
173 FromPtrBits = computeKnownBits(&PtrOp, DL, nullptr);
174 }
175
176 KnownBits ToPtrBits =
177 computeKnownBitsAddrSpaceCast(FromAS, ToAS, FromPtrBits);
178
179 return {FromPtrBits, ToPtrBits};
180 }
181
182 virtual KnownBits
183 computeKnownBitsAddrSpaceCast(unsigned FromAS, unsigned ToAS,
184 const KnownBits &FromPtrBits) const {
185 unsigned ToASBitSize = DL.getPointerSizeInBits(ToAS);
186
187 if (DL.isNonIntegralAddressSpace(FromAS))
188 return KnownBits(ToASBitSize);
189
190 // By default, we assume that all valid "larger" (e.g. 64-bit) to "smaller"
191 // (e.g. 32-bit) casts work by chopping off the high bits.
192 // By default, we do not assume that null results in null again.
193 return FromPtrBits.anyextOrTrunc(ToASBitSize);
194 }
195
197 unsigned DstAS) const {
198 return {DL.getPointerSizeInBits(SrcAS), 0};
199 }
200
201 virtual bool
203 return AS == 0;
204 };
205
206 virtual unsigned getAssumedAddrSpace(const Value *V) const { return -1; }
207
208 virtual std::pair<const Value *, unsigned>
210 return std::make_pair(nullptr, -1);
211 }
212
214 Value *OldV,
215 Value *NewV) const {
216 return nullptr;
217 }
218
219 virtual bool isLoweredToCall(const Function *F) const {
220 assert(F && "A concrete function must be provided to this routine.");
221
222 // FIXME: These should almost certainly not be handled here, and instead
223 // handled with the help of TLI or the target itself. This was largely
224 // ported from existing analysis heuristics here so that such refactorings
225 // can take place in the future.
226
227 if (F->isIntrinsic())
228 return false;
229
230 if (F->hasLocalLinkage() || !F->hasName())
231 return true;
232
233 StringRef Name = F->getName();
234
235 // These will all likely lower to a single selection DAG node.
236 // clang-format off
237 if (Name == "copysign" || Name == "copysignf" || Name == "copysignl" ||
238 Name == "fabs" || Name == "fabsf" || Name == "fabsl" ||
239 Name == "fmin" || Name == "fminf" || Name == "fminl" ||
240 Name == "fmax" || Name == "fmaxf" || Name == "fmaxl" ||
241 Name == "sin" || Name == "sinf" || Name == "sinl" ||
242 Name == "cos" || Name == "cosf" || Name == "cosl" ||
243 Name == "tan" || Name == "tanf" || Name == "tanl" ||
244 Name == "asin" || Name == "asinf" || Name == "asinl" ||
245 Name == "acos" || Name == "acosf" || Name == "acosl" ||
246 Name == "atan" || Name == "atanf" || Name == "atanl" ||
247 Name == "atan2" || Name == "atan2f" || Name == "atan2l"||
248 Name == "sinh" || Name == "sinhf" || Name == "sinhl" ||
249 Name == "cosh" || Name == "coshf" || Name == "coshl" ||
250 Name == "tanh" || Name == "tanhf" || Name == "tanhl" ||
251 Name == "sqrt" || Name == "sqrtf" || Name == "sqrtl" ||
252 Name == "exp10" || Name == "exp10l" || Name == "exp10f")
253 return false;
254 // clang-format on
255 // These are all likely to be optimized into something smaller.
256 if (Name == "pow" || Name == "powf" || Name == "powl" || Name == "exp2" ||
257 Name == "exp2l" || Name == "exp2f" || Name == "floor" ||
258 Name == "floorf" || Name == "ceil" || Name == "round" ||
259 Name == "ffs" || Name == "ffsl" || Name == "abs" || Name == "labs" ||
260 Name == "llabs")
261 return false;
262
263 return true;
264 }
265
267 AssumptionCache &AC,
268 TargetLibraryInfo *LibInfo,
269 HardwareLoopInfo &HWLoopInfo) const {
270 return false;
271 }
272
273 virtual unsigned getEpilogueVectorizationMinVF() const { return 16; }
274
276 return false;
277 }
278
282
283 virtual std::optional<Instruction *>
285 return std::nullopt;
286 }
287
288 virtual std::optional<Value *>
290 APInt DemandedMask, KnownBits &Known,
291 bool &KnownBitsComputed) const {
292 return std::nullopt;
293 }
294
295 virtual std::optional<Value *> simplifyDemandedVectorEltsIntrinsic(
296 InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts,
297 APInt &UndefElts2, APInt &UndefElts3,
298 std::function<void(Instruction *, unsigned, APInt, APInt &)>
299 SimplifyAndSetOp) const {
300 return std::nullopt;
301 }
302
306
309
310 virtual bool isLegalAddImmediate(int64_t Imm) const { return false; }
311
312 virtual bool isLegalAddScalableImmediate(int64_t Imm) const { return false; }
313
314 virtual bool isLegalICmpImmediate(int64_t Imm) const { return false; }
315
316 virtual bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV,
317 int64_t BaseOffset, bool HasBaseReg,
318 int64_t Scale, unsigned AddrSpace,
319 Instruction *I = nullptr,
320 int64_t ScalableOffset = 0) const {
321 // Guess that only reg and reg+reg addressing is allowed. This heuristic is
322 // taken from the implementation of LSR.
323 return !BaseGV && BaseOffset == 0 && (Scale == 0 || Scale == 1);
324 }
325
326 virtual bool isLSRCostLess(const TTI::LSRCost &C1,
327 const TTI::LSRCost &C2) const {
328 return std::tie(C1.NumRegs, C1.AddRecCost, C1.NumIVMuls, C1.NumBaseAdds,
329 C1.ScaleCost, C1.ImmCost, C1.SetupCost) <
330 std::tie(C2.NumRegs, C2.AddRecCost, C2.NumIVMuls, C2.NumBaseAdds,
331 C2.ScaleCost, C2.ImmCost, C2.SetupCost);
332 }
333
334 virtual bool isNumRegsMajorCostOfLSR() const { return true; }
335
336 virtual bool shouldDropLSRSolutionIfLessProfitable() const { return false; }
337
339 return false;
340 }
341
342 virtual bool canMacroFuseCmp() const { return false; }
343
344 virtual bool canSaveCmp(Loop *L, CondBrInst **BI, ScalarEvolution *SE,
346 TargetLibraryInfo *LibInfo) const {
347 return false;
348 }
349
352 return TTI::AMK_None;
353 }
354
355 virtual bool isLegalMaskedStore(Type *DataType, Align Alignment,
356 unsigned AddressSpace,
357 TTI::MaskKind MaskKind) const {
358 return false;
359 }
360
361 virtual bool isLegalMaskedLoad(Type *DataType, Align Alignment,
362 unsigned AddressSpace,
363 TTI::MaskKind MaskKind) const {
364 return false;
365 }
366
367 virtual bool isLegalNTStore(Type *DataType, Align Alignment) const {
368 // By default, assume nontemporal memory stores are available for stores
369 // that are aligned and have a size that is a power of 2.
370 unsigned DataSize = DL.getTypeStoreSize(DataType);
371 return Alignment >= DataSize && isPowerOf2_32(DataSize);
372 }
373
374 virtual bool isLegalNTLoad(Type *DataType, Align Alignment) const {
375 // By default, assume nontemporal memory loads are available for loads that
376 // are aligned and have a size that is a power of 2.
377 unsigned DataSize = DL.getTypeStoreSize(DataType);
378 return Alignment >= DataSize && isPowerOf2_32(DataSize);
379 }
380
381 virtual bool isLegalBroadcastLoad(Type *ElementTy,
382 ElementCount NumElements) const {
383 return false;
384 }
385
386 virtual bool isLegalMaskedScatter(Type *DataType, Align Alignment) const {
387 return false;
388 }
389
390 virtual bool isLegalMaskedGather(Type *DataType, Align Alignment) const {
391 return false;
392 }
393
395 Align Alignment) const {
396 return false;
397 }
398
400 Align Alignment) const {
401 return false;
402 }
403
404 virtual bool isLegalMaskedCompressStore(Type *DataType,
405 Align Alignment) const {
406 return false;
407 }
408
409 virtual bool isLegalAltInstr(VectorType *VecTy, unsigned Opcode0,
410 unsigned Opcode1,
411 const SmallBitVector &OpcodeMask) const {
412 return false;
413 }
414
415 virtual bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const {
416 return false;
417 }
418
419 virtual bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const {
420 return false;
421 }
422
423 virtual bool isLegalInterleavedAccessType(VectorType *VTy, unsigned Factor,
424 Align Alignment,
425 unsigned AddrSpace) const {
426 return false;
427 }
428
429 virtual bool isLegalMaskedVectorHistogram(Type *AddrType,
430 Type *DataType) const {
431 return false;
432 }
433
434 virtual bool enableOrderedReductions() const { return false; }
435
436 virtual bool hasDivRemOp(Type *DataType, bool IsSigned) const {
437 return false;
438 }
439
440 virtual bool hasVolatileVariant(Instruction *I, unsigned AddrSpace) const {
441 return false;
442 }
443
444 virtual bool prefersVectorizedAddressing() const { return true; }
445
447 StackOffset BaseOffset,
448 bool HasBaseReg, int64_t Scale,
449 unsigned AddrSpace) const {
450 // Guess that all legal addressing mode are free.
451 if (isLegalAddressingMode(Ty, BaseGV, BaseOffset.getFixed(), HasBaseReg,
452 Scale, AddrSpace, /*I=*/nullptr,
453 BaseOffset.getScalable()))
454 return 0;
456 }
457
458 virtual bool LSRWithInstrQueries() const { return false; }
459
460 virtual bool isTruncateFree(Type *Ty1, Type *Ty2) const { return false; }
461
462 virtual bool isProfitableToHoist(Instruction *I) const { return true; }
463
464 virtual bool useAA() const { return false; }
465
466 virtual bool isTypeLegal(Type *Ty) const { return false; }
467
468 virtual unsigned getRegUsageForType(Type *Ty) const { return 1; }
469
470 virtual bool shouldBuildLookupTables() const { return true; }
471
473 return true;
474 }
475
476 virtual unsigned getMinimumLookupTableEntryBitWidth() const { return 8; }
477
478 virtual bool shouldBuildRelLookupTables() const { return false; }
479
480 virtual bool useColdCCForColdCall(Function &F) const { return false; }
481
482 virtual bool useFastCCForInternalCall(Function &F) const { return true; }
483
485 unsigned ScalarOpdIdx) const {
486 return false;
487 }
488
490 int OpdIdx) const {
491 return OpdIdx == -1;
492 }
493
494 virtual bool
496 int RetIdx) const {
497 return RetIdx == 0;
498 }
499
501 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
502 TTI::TargetCostKind CostKind, bool ForPoisonSrc = true,
503 ArrayRef<Value *> VL = {},
505 // Default implementation returns 0.
506 // BasicTTIImpl provides the actual implementation.
507 return 0;
508 }
509
515
516 virtual bool supportsEfficientVectorElementLoadStore() const { return false; }
517
518 virtual bool supportsTailCalls() const { return true; }
519
520 virtual bool supportsTailCallFor(const CallBase *CB) const {
521 llvm_unreachable("Not implemented");
522 }
523
524 virtual bool enableAggressiveInterleaving(bool LoopHasReductions) const {
525 return false;
526 }
527
529 enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const {
530 return {};
531 }
532
533 virtual bool enableSelectOptimize() const { return true; }
534
535 virtual bool shouldTreatInstructionLikeSelect(const Instruction *I) const {
536 // A select with two constant operands will usually be better left as a
537 // select.
538 using namespace llvm::PatternMatch;
540 return false;
541 // If the select is a logical-and/logical-or then it is better treated as a
542 // and/or by the backend.
543 return isa<SelectInst>(I) &&
546 }
547
548 virtual bool enableInterleavedAccessVectorization() const { return false; }
549
551 return false;
552 }
553
554 virtual bool isFPVectorizationPotentiallyUnsafe() const { return false; }
555
557 unsigned BitWidth,
558 unsigned AddressSpace,
559 Align Alignment,
560 unsigned *Fast) const {
561 return false;
562 }
563
565 getPopcntSupport(unsigned IntTyWidthInBit) const {
566 return TTI::PSK_Software;
567 }
568
569 virtual bool haveFastSqrt(Type *Ty) const { return false; }
570
571 virtual bool haveFastClmul(IntegerType *Ty) const { return false; }
572
574 return true;
575 }
576
577 virtual bool isFCmpOrdCheaperThanFCmpZero(Type *Ty) const { return true; }
578
579 virtual InstructionCost getFPOpCost(Type *Ty) const {
581 }
582
583 virtual InstructionCost getIntImmCodeSizeCost(unsigned Opcode, unsigned Idx,
584 const APInt &Imm,
585 Type *Ty) const {
586 return 0;
587 }
588
591 return TTI::TCC_Basic;
592 }
593
594 virtual InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx,
595 const APInt &Imm, Type *Ty,
597 Instruction *Inst = nullptr) const {
598 return TTI::TCC_Free;
599 }
600
601 virtual InstructionCost
602 getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm,
603 Type *Ty, TTI::TargetCostKind CostKind) const {
604 return TTI::TCC_Free;
605 }
606
608 const Function &Fn) const {
609 return false;
610 }
611
612 virtual unsigned getNumberOfRegisters(unsigned ClassID) const { return 8; }
613 virtual bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const {
614 return false;
615 }
616
617 virtual unsigned getRegisterClassForType(bool Vector,
618 Type *Ty = nullptr) const {
619 return Vector ? 1 : 0;
620 }
621
622 virtual const char *getRegisterClassName(unsigned ClassID) const {
623 switch (ClassID) {
624 default:
625 return "Generic::Unknown Register Class";
626 case 0:
627 return "Generic::ScalarRC";
628 case 1:
629 return "Generic::VectorRC";
630 }
631 }
632
633 virtual InstructionCost
636 return TTI::TCC_Basic;
637 }
638
639 virtual InstructionCost
642 return TTI::TCC_Basic;
643 }
644
645 virtual TypeSize
649
650 virtual unsigned getMinVectorRegisterBitWidth() const { return 128; }
651
652 virtual std::optional<unsigned> getVScaleForTuning() const {
653 return std::nullopt;
654 }
655
656 virtual bool
660
661 virtual ElementCount getMinimumVF(unsigned ElemWidth, bool IsScalable) const {
662 return ElementCount::get(0, IsScalable);
663 }
664
665 virtual unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const {
666 return 0;
667 }
668 virtual unsigned getStoreMinimumVF(unsigned VF, Type *, Type *, Align,
669 unsigned) const {
670 return VF;
671 }
672
674 const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const {
675 AllowPromotionWithoutCommonHeader = false;
676 return false;
677 }
678
679 virtual unsigned getCacheLineSize() const { return 0; }
680 virtual std::optional<unsigned>
682 switch (Level) {
684 [[fallthrough]];
686 return std::nullopt;
687 }
688 llvm_unreachable("Unknown TargetTransformInfo::CacheLevel");
689 }
690
691 virtual std::optional<unsigned>
693 switch (Level) {
695 [[fallthrough]];
697 return std::nullopt;
698 }
699
700 llvm_unreachable("Unknown TargetTransformInfo::CacheLevel");
701 }
702
703 virtual std::optional<unsigned> getMinPageSize() const { return {}; }
704
705 virtual unsigned getPrefetchDistance() const { return 0; }
706 virtual unsigned getMinPrefetchStride(unsigned NumMemAccesses,
707 unsigned NumStridedMemAccesses,
708 unsigned NumPrefetches,
709 bool HasCall) const {
710 return 1;
711 }
712 virtual unsigned getMaxPrefetchIterationsAhead() const { return UINT_MAX; }
713 virtual bool enableWritePrefetching() const { return false; }
714 virtual bool shouldPrefetchAddressSpace(unsigned AS) const { return !AS; }
715
717 unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType,
719 TTI::PartialReductionExtendKind OpBExtend, std::optional<unsigned> BinOp,
720 TTI::TargetCostKind CostKind, std::optional<FastMathFlags> FMF) const {
722 }
723
725 bool HasUnorderedReductions) const {
726 return 1;
727 }
728
730 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
732 ArrayRef<const Value *> Args, const Instruction *CxtI = nullptr) const {
733 // Widenable conditions will eventually lower into constants, so some
734 // operations with them will be trivially optimized away.
735 auto IsWidenableCondition = [](const Value *V) {
736 if (auto *II = dyn_cast<IntrinsicInst>(V))
737 if (II->getIntrinsicID() == Intrinsic::experimental_widenable_condition)
738 return true;
739 return false;
740 };
741 // FIXME: A number of transformation tests seem to require these values
742 // which seems a little odd for how arbitary there are.
743 switch (Opcode) {
744 default:
745 break;
746 case Instruction::FDiv:
747 case Instruction::FRem:
748 case Instruction::SDiv:
749 case Instruction::SRem:
750 case Instruction::UDiv:
751 case Instruction::URem:
752 // FIXME: Unlikely to be true for CodeSize.
753 return TTI::TCC_Expensive;
754 case Instruction::And:
755 case Instruction::Or:
756 if (any_of(Args, IsWidenableCondition))
757 return TTI::TCC_Free;
758 break;
759 }
760
761 // Assume a 3cy latency for fp arithmetic ops.
763 if (Ty->getScalarType()->isFloatingPointTy())
764 return 3;
765
766 return 1;
767 }
768
769 virtual InstructionCost getAltInstrCost(VectorType *VecTy, unsigned Opcode0,
770 unsigned Opcode1,
771 const SmallBitVector &OpcodeMask,
774 }
775
776 virtual InstructionCost
779 VectorType *SubTp, ArrayRef<const Value *> Args = {},
780 const Instruction *CxtI = nullptr) const {
781 return 1;
782 }
783
784 virtual InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst,
785 Type *Src, TTI::CastContextHint CCH,
787 const Instruction *I) const {
788 switch (Opcode) {
789 default:
790 break;
791 case Instruction::IntToPtr: {
792 unsigned SrcSize = Src->getScalarSizeInBits();
793 if (DL.isLegalInteger(SrcSize) &&
794 SrcSize <= DL.getPointerTypeSizeInBits(Dst))
795 return 0;
796 break;
797 }
798 case Instruction::PtrToAddr: {
799 unsigned DstSize = Dst->getScalarSizeInBits();
800 assert(DstSize == DL.getAddressSizeInBits(Src));
801 if (DL.isLegalInteger(DstSize))
802 return 0;
803 break;
804 }
805 case Instruction::PtrToInt: {
806 unsigned DstSize = Dst->getScalarSizeInBits();
807 if (DL.isLegalInteger(DstSize) &&
808 DstSize >= DL.getPointerTypeSizeInBits(Src))
809 return 0;
810 break;
811 }
812 case Instruction::BitCast:
813 if (Dst == Src || (Dst->isPointerTy() && Src->isPointerTy()))
814 // Identity and pointer-to-pointer casts are free.
815 return 0;
816 break;
817 case Instruction::Trunc: {
818 // trunc to a native type is free (assuming the target has compare and
819 // shift-right of the same width).
820 TypeSize DstSize = DL.getTypeSizeInBits(Dst);
821 if (!DstSize.isScalable() && DL.isLegalInteger(DstSize.getFixedValue()))
822 return 0;
823 break;
824 }
825 }
826 return 1;
827 }
828
829 virtual InstructionCost
830 getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy,
831 unsigned Index, TTI::TargetCostKind CostKind) const {
832 return 1;
833 }
834
835 virtual InstructionCost getCFInstrCost(unsigned Opcode,
837 const Instruction *I = nullptr) const {
838 // A phi would be free, unless we're costing the throughput because it
839 // will require a register.
840 if (Opcode == Instruction::PHI && CostKind != TTI::TCK_RecipThroughput)
841 return 0;
842 return 1;
843 }
844
846 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
848 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
849 return 1;
850 }
851
853 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
854 const Value *Op0, const Value *Op1,
856 return 1;
857 }
858
859 /// \param ScalarUserAndIdx encodes the information about extracts from a
860 /// vector with 'Scalar' being the value being extracted,'User' being the user
861 /// of the extract(nullptr if user is not known before vectorization) and
862 /// 'Idx' being the extract lane.
864 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
865 Value *Scalar,
866 ArrayRef<std::tuple<Value *, User *, int>> ScalarUserAndIdx,
868 return 1;
869 }
870
873 unsigned Index,
875 return 1;
876 }
877
878 virtual InstructionCost
881 unsigned Index) const {
882 return 1;
883 }
884
885 virtual InstructionCost
886 getReplicationShuffleCost(Type *EltTy, int ReplicationFactor, int VF,
887 const APInt &DemandedDstElts,
889 return 1;
890 }
891
892 virtual InstructionCost
895 // Note: The `insertvalue` cost here is chosen to match the default case of
896 // getInstructionCost() -- as prior to adding this helper `insertvalue` was
897 // not handled.
898 if (Opcode == Instruction::InsertValue &&
900 return TTI::TCC_Basic;
901 return TTI::TCC_Free;
902 }
903
904 virtual InstructionCost
905 getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment,
907 TTI::OperandValueInfo OpInfo, const Instruction *I) const {
908 return 1;
909 }
910
912 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
913 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
914 bool UseMaskForCond, bool UseMaskForGaps) const {
915 return 1;
916 }
917
918 virtual InstructionCost
921 switch (ICA.getID()) {
922 default:
923 break;
924 case Intrinsic::allow_runtime_check:
925 case Intrinsic::allow_ubsan_check:
926 case Intrinsic::annotation:
927 case Intrinsic::assume:
928 case Intrinsic::sideeffect:
929 case Intrinsic::pseudoprobe:
930 case Intrinsic::arithmetic_fence:
931 case Intrinsic::dbg_assign:
932 case Intrinsic::dbg_declare:
933 case Intrinsic::dbg_value:
934 case Intrinsic::dbg_label:
935 case Intrinsic::invariant_start:
936 case Intrinsic::invariant_end:
937 case Intrinsic::launder_invariant_group:
938 case Intrinsic::strip_invariant_group:
939 case Intrinsic::is_constant:
940 case Intrinsic::lifetime_start:
941 case Intrinsic::lifetime_end:
942 case Intrinsic::experimental_noalias_scope_decl:
943 case Intrinsic::objectsize:
944 case Intrinsic::ptr_annotation:
945 case Intrinsic::var_annotation:
946 case Intrinsic::experimental_gc_result:
947 case Intrinsic::experimental_gc_relocate:
948 case Intrinsic::coro_alloc:
949 case Intrinsic::coro_begin:
950 case Intrinsic::coro_begin_custom_abi:
951 case Intrinsic::coro_dead:
952 case Intrinsic::coro_id:
953 case Intrinsic::coro_id_async:
954 case Intrinsic::coro_id_retcon:
955 case Intrinsic::coro_id_retcon_once:
956 case Intrinsic::coro_noop:
957 case Intrinsic::coro_free:
958 case Intrinsic::coro_end:
959 case Intrinsic::coro_frame:
960 case Intrinsic::coro_size:
961 case Intrinsic::coro_align:
962 case Intrinsic::coro_suspend:
963 case Intrinsic::coro_subfn_addr:
964 case Intrinsic::threadlocal_address:
965 case Intrinsic::experimental_widenable_condition:
966 case Intrinsic::ssa_copy:
967 // These intrinsics don't actually represent code after lowering.
968 return 0;
969 case Intrinsic::bswap:
970 if (!ICA.getReturnType()->isVectorTy() &&
971 !isPowerOf2_64(DL.getTypeSizeInBits(ICA.getReturnType())))
973 }
974 return 1;
975 }
976
977 virtual InstructionCost
980 switch (MICA.getID()) {
981 case Intrinsic::masked_scatter:
982 case Intrinsic::masked_gather:
983 case Intrinsic::masked_load:
984 case Intrinsic::masked_store:
985 case Intrinsic::vp_scatter:
986 case Intrinsic::vp_gather:
987 case Intrinsic::masked_compressstore:
988 case Intrinsic::masked_expandload:
989 return 1;
990 }
992 }
993
997 return 1;
998 }
999
1000 // Assume that we have a register of the right size for the type.
1001 virtual unsigned getNumberOfParts(Type *Tp) const { return 1; }
1002
1005 const SCEV *,
1006 TTI::TargetCostKind) const {
1007 return 0;
1008 }
1009
1010 virtual InstructionCost
1012 std::optional<FastMathFlags> FMF,
1013 TTI::TargetCostKind) const {
1014 return 1;
1015 }
1016
1019 TTI::TargetCostKind) const {
1020 return 1;
1021 }
1022
1023 virtual InstructionCost
1024 getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy,
1025 VectorType *Ty, std::optional<FastMathFlags> FMF,
1027 return 1;
1028 }
1029
1030 virtual InstructionCost
1031 getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy,
1033 return 1;
1034 }
1035
1036 virtual InstructionCost
1038 return 0;
1039 }
1040
1042 MemIntrinsicInfo &Info) const {
1043 return false;
1044 }
1045
1046 virtual unsigned getAtomicMemIntrinsicMaxElementSize() const {
1047 // Note for overrides: You must ensure for all element unordered-atomic
1048 // memory intrinsics that all power-of-2 element sizes up to, and
1049 // including, the return value of this method have a corresponding
1050 // runtime lib call. These runtime lib call definitions can be found
1051 // in RuntimeLibcalls.h
1052 return 0;
1053 }
1054
1055 virtual Value *
1057 bool CanCreate = true) const {
1058 return nullptr;
1059 }
1060
1061 virtual Type *
1063 unsigned SrcAddrSpace, unsigned DestAddrSpace,
1064 Align SrcAlign, Align DestAlign,
1065 std::optional<uint32_t> AtomicElementSize) const {
1066 return AtomicElementSize ? Type::getIntNTy(Context, *AtomicElementSize * 8)
1067 : Type::getInt8Ty(Context);
1068 }
1069
1071 SmallVectorImpl<Type *> &OpsOut, LLVMContext &Context,
1072 unsigned RemainingBytes, unsigned SrcAddrSpace, unsigned DestAddrSpace,
1073 Align SrcAlign, Align DestAlign,
1074 std::optional<uint32_t> AtomicCpySize) const {
1075 unsigned OpSizeInBytes = AtomicCpySize.value_or(1);
1076 Type *OpType = Type::getIntNTy(Context, OpSizeInBytes * 8);
1077 for (unsigned i = 0; i != RemainingBytes; i += OpSizeInBytes)
1078 OpsOut.push_back(OpType);
1079 }
1080
1081 virtual bool areInlineCompatible(const Function *Caller,
1082 const Function *Callee) const {
1083 return (Caller->getFnAttribute("target-cpu") ==
1084 Callee->getFnAttribute("target-cpu")) &&
1085 (Caller->getFnAttribute("target-features") ==
1086 Callee->getFnAttribute("target-features"));
1087 }
1088
1089 virtual unsigned getInlineCallPenalty(const Function *F, const CallBase &Call,
1090 unsigned DefaultCallPenalty) const {
1091 return DefaultCallPenalty;
1092 }
1093
1094 virtual bool
1096 const Attribute &Attr) const {
1097 // Copy attributes by default
1098 return true;
1099 }
1100
1101 virtual bool areTypesABICompatible(const Function *Caller,
1102 const Function *Callee,
1103 ArrayRef<Type *> Types) const {
1104 return (Caller->getFnAttribute("target-cpu") ==
1105 Callee->getFnAttribute("target-cpu")) &&
1106 (Caller->getFnAttribute("target-features") ==
1107 Callee->getFnAttribute("target-features"));
1108 }
1109
1111 return false;
1112 }
1113
1115 return false;
1116 }
1117
1118 virtual unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const {
1119 return 128;
1120 }
1121
1122 virtual bool isLegalToVectorizeLoad(LoadInst *LI) const { return true; }
1123
1124 virtual bool isLegalToVectorizeStore(StoreInst *SI) const { return true; }
1125
1126 virtual bool isLegalToVectorizeLoadChain(unsigned ChainSizeInBytes,
1127 Align Alignment,
1128 unsigned AddrSpace) const {
1129 return true;
1130 }
1131
1132 virtual bool isLegalToVectorizeStoreChain(unsigned ChainSizeInBytes,
1133 Align Alignment,
1134 unsigned AddrSpace) const {
1135 return true;
1136 }
1137
1139 ElementCount VF) const {
1140 return true;
1141 }
1142
1144 return true;
1145 }
1146
1147 virtual unsigned getLoadVectorFactor(unsigned VF, unsigned LoadSize,
1148 unsigned ChainSizeInBytes,
1149 VectorType *VecTy) const {
1150 return VF;
1151 }
1152
1153 virtual unsigned getStoreVectorFactor(unsigned VF, unsigned StoreSize,
1154 unsigned ChainSizeInBytes,
1155 VectorType *VecTy) const {
1156 return VF;
1157 }
1158
1159 virtual bool preferFixedOverScalableIfEqualCost() const { return false; }
1160
1161 virtual bool preferInLoopReduction(RecurKind Kind, Type *Ty) const {
1162 return false;
1163 }
1164 virtual bool preferAlternateOpcodeVectorization() const { return true; }
1165
1166 virtual bool preferSLPInstCountCheck() const { return true; }
1167
1168 virtual bool preferPredicatedReductionSelect() const { return false; }
1169
1170 virtual bool preferEpilogueVectorization(ElementCount Iters) const {
1171 // We consider epilogue vectorization unprofitable for targets that
1172 // don't consider interleaving beneficial (eg. MVE).
1173 return getMaxInterleaveFactor(Iters, false) > 1;
1174 }
1175
1176 virtual bool shouldConsiderVectorizationRegPressure() const { return false; }
1177
1178 virtual bool shouldExpandReduction(const IntrinsicInst *II) const {
1179 return true;
1180 }
1181
1182 virtual TTI::ReductionShuffle
1186
1187 virtual unsigned getGISelRematGlobalCost() const { return 1; }
1188
1189 virtual unsigned getMinTripCountTailFoldingThreshold() const { return 0; }
1190
1191 virtual bool supportsScalableVectors() const { return false; }
1192
1193 virtual bool enableScalableVectorization() const { return false; }
1194
1195 virtual bool hasActiveVectorLength() const { return false; }
1196
1198 SmallVectorImpl<Use *> &Ops) const {
1199 return false;
1200 }
1201
1202 virtual bool isVectorShiftByScalarCheap(Type *Ty) const { return false; }
1203
1210
1211 virtual bool hasArmWideBranch(bool) const { return false; }
1212
1213 virtual APInt getFeatureMask(const Function &F) const {
1214 return APInt::getZero(32);
1215 }
1216
1217 virtual APInt getPriorityMask(const Function &F) const {
1218 return APInt::getZero(32);
1219 }
1220
1221 virtual bool isMultiversionedFunction(const Function &F) const {
1222 return false;
1223 }
1224
1225 virtual unsigned getMaxNumArgs() const { return UINT_MAX; }
1226
1227 virtual unsigned getNumBytesToPadGlobalArray(unsigned Size,
1228 Type *ArrayType) const {
1229 return 0;
1230 }
1231
1233 const Function &F,
1234 SmallVectorImpl<std::pair<StringRef, int64_t>> &LB) const {}
1235
1236 virtual bool allowVectorElementIndexingUsingGEP() const { return true; }
1237
1238 virtual bool isUniform(const Instruction *I,
1239 const SmallBitVector &UniformArgs) const {
1240 llvm_unreachable("target must implement isUniform for Custom uniformity");
1241 }
1242
1243protected:
1244 // Obtain the minimum required size to hold the value (without the sign)
1245 // In case of a vector it returns the min required size for one element.
1246 unsigned minRequiredElementSize(const Value *Val, bool &isSigned) const {
1248 const auto *VectorValue = cast<Constant>(Val);
1249
1250 // In case of a vector need to pick the max between the min
1251 // required size for each element
1252 auto *VT = cast<FixedVectorType>(Val->getType());
1253
1254 // Assume unsigned elements
1255 isSigned = false;
1256
1257 // The max required size is the size of the vector element type
1258 unsigned MaxRequiredSize =
1259 VT->getElementType()->getPrimitiveSizeInBits().getFixedValue();
1260
1261 unsigned MinRequiredSize = 0;
1262 for (unsigned i = 0, e = VT->getNumElements(); i < e; ++i) {
1263 if (auto *IntElement =
1264 dyn_cast<ConstantInt>(VectorValue->getAggregateElement(i))) {
1265 bool signedElement = IntElement->getValue().isNegative();
1266 // Get the element min required size.
1267 unsigned ElementMinRequiredSize =
1268 IntElement->getValue().getSignificantBits() - 1;
1269 // In case one element is signed then all the vector is signed.
1270 isSigned |= signedElement;
1271 // Save the max required bit size between all the elements.
1272 MinRequiredSize = std::max(MinRequiredSize, ElementMinRequiredSize);
1273 } else {
1274 // not an int constant element
1275 return MaxRequiredSize;
1276 }
1277 }
1278 return MinRequiredSize;
1279 }
1280
1281 if (const auto *CI = dyn_cast<ConstantInt>(Val)) {
1282 isSigned = CI->getValue().isNegative();
1283 return CI->getValue().getSignificantBits() - 1;
1284 }
1285
1286 if (const auto *Cast = dyn_cast<SExtInst>(Val)) {
1287 isSigned = true;
1288 return Cast->getSrcTy()->getScalarSizeInBits() - 1;
1289 }
1290
1291 if (const auto *Cast = dyn_cast<ZExtInst>(Val)) {
1292 isSigned = false;
1293 return Cast->getSrcTy()->getScalarSizeInBits();
1294 }
1295
1296 isSigned = false;
1297 return Val->getType()->getScalarSizeInBits();
1298 }
1299
1300 bool isStridedAccess(const SCEV *Ptr) const {
1301 return Ptr && isa<SCEVAddRecExpr>(Ptr);
1302 }
1303
1305 const SCEV *Ptr) const {
1306 if (!isStridedAccess(Ptr))
1307 return nullptr;
1308 const SCEVAddRecExpr *AddRec = cast<SCEVAddRecExpr>(Ptr);
1309 return dyn_cast<SCEVConstant>(AddRec->getStepRecurrence(*SE));
1310 }
1311
1313 int64_t MergeDistance) const {
1314 const SCEVConstant *Step = getConstantStrideStep(SE, Ptr);
1315 if (!Step)
1316 return false;
1317 APInt StrideVal = Step->getAPInt();
1318 if (StrideVal.getBitWidth() > 64)
1319 return false;
1320 // FIXME: Need to take absolute value for negative stride case.
1321 return StrideVal.getSExtValue() < MergeDistance;
1322 }
1323};
1324
1325/// CRTP base class for use as a mix-in that aids implementing
1326/// a TargetTransformInfo-compatible class.
1327template <typename T>
1329private:
1330 typedef TargetTransformInfoImplBase BaseT;
1331
1332protected:
1334
1335public:
1336 InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr,
1339 Type *AccessType) const override {
1340 assert(PointeeType && Ptr && "can't get GEPCost of nullptr");
1341 auto *BaseGV = dyn_cast<GlobalValue>(Ptr->stripPointerCasts());
1342 bool HasBaseReg = (BaseGV == nullptr);
1343
1344 auto PtrSizeBits = DL.getPointerTypeSizeInBits(Ptr->getType());
1345 APInt BaseOffset(PtrSizeBits, 0);
1346 int64_t Scale = 0;
1347
1348 auto GTI = gep_type_begin(PointeeType, Operands);
1349 Type *TargetType = nullptr;
1350
1351 // Handle the case where the GEP instruction has a single operand,
1352 // the basis, therefore TargetType is a nullptr.
1353 if (Operands.empty())
1354 return !BaseGV ? TTI::TCC_Free : TTI::TCC_Basic;
1355
1356 for (auto I = Operands.begin(); I != Operands.end(); ++I, ++GTI) {
1357 TargetType = GTI.getIndexedType();
1358 // We assume that the cost of Scalar GEP with constant index and the
1359 // cost of Vector GEP with splat constant index are the same.
1360 const ConstantInt *ConstIdx = dyn_cast<ConstantInt>(*I);
1361 if (!ConstIdx)
1362 if (auto Splat = getSplatValue(*I))
1363 ConstIdx = dyn_cast<ConstantInt>(Splat);
1364 if (StructType *STy = GTI.getStructTypeOrNull()) {
1365 // For structures the index is always splat or scalar constant
1366 assert(ConstIdx && "Unexpected GEP index");
1367 uint64_t Field = ConstIdx->getZExtValue();
1368 BaseOffset += DL.getStructLayout(STy)->getElementOffset(Field);
1369 } else {
1370 // If this operand is a scalable type, bail out early.
1371 // TODO: Make isLegalAddressingMode TypeSize aware.
1372 if (TargetType->isScalableTy())
1373 return TTI::TCC_Basic;
1374 int64_t ElementSize =
1375 GTI.getSequentialElementStride(DL).getFixedValue();
1376 if (ConstIdx) {
1377 BaseOffset +=
1378 ConstIdx->getValue().sextOrTrunc(PtrSizeBits) * ElementSize;
1379 } else {
1380 // Needs scale register.
1381 if (Scale != 0)
1382 // No addressing mode takes two scale registers.
1383 return TTI::TCC_Basic;
1384 Scale = ElementSize;
1385 }
1386 }
1387 }
1388
1389 // If we haven't been provided a hint, use the target type for now.
1390 //
1391 // TODO: Take a look at potentially removing this: This is *slightly* wrong
1392 // as it's possible to have a GEP with a foldable target type but a memory
1393 // access that isn't foldable. For example, this load isn't foldable on
1394 // RISC-V:
1395 //
1396 // %p = getelementptr i32, ptr %base, i32 42
1397 // %x = load <2 x i32>, ptr %p
1398 if (!AccessType)
1399 AccessType = TargetType;
1400
1401 // If the final address of the GEP is a legal addressing mode for the given
1402 // access type, then we can fold it into its users.
1403 if (static_cast<const T *>(this)->isLegalAddressingMode(
1404 AccessType, const_cast<GlobalValue *>(BaseGV),
1405 BaseOffset.sextOrTrunc(64).getSExtValue(), HasBaseReg, Scale,
1407 return TTI::TCC_Free;
1408
1409 // TODO: Instead of returning TCC_Basic here, we should use
1410 // getArithmeticInstrCost. Or better yet, provide a hook to let the target
1411 // model it.
1412 return TTI::TCC_Basic;
1413 }
1414
1417 const TTI::PointersChainInfo &Info, Type *AccessTy,
1418 TTI::TargetCostKind CostKind) const override {
1420 // In the basic model we take into account GEP instructions only
1421 // (although here can come alloca instruction, a value, constants and/or
1422 // constant expressions, PHIs, bitcasts ... whatever allowed to be used as a
1423 // pointer). Typically, if Base is a not a GEP-instruction and all the
1424 // pointers are relative to the same base address, all the rest are
1425 // either GEP instructions, PHIs, bitcasts or constants. When we have same
1426 // base, we just calculate cost of each non-Base GEP as an ADD operation if
1427 // any their index is a non-const.
1428 // If no known dependecies between the pointers cost is calculated as a sum
1429 // of costs of GEP instructions.
1430 for (const Value *V : Ptrs) {
1431 const auto *GEP = dyn_cast<GetElementPtrInst>(V);
1432 if (!GEP)
1433 continue;
1434 if (Info.isSameBase() && V != Base) {
1435 if (GEP->hasAllConstantIndices())
1436 continue;
1437 Cost += static_cast<const T *>(this)->getArithmeticInstrCost(
1438 Instruction::Add, GEP->getType(), CostKind,
1439 {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None},
1440 {});
1441 } else {
1442 SmallVector<const Value *> Indices(GEP->indices());
1443 Cost += static_cast<const T *>(this)->getGEPCost(
1444 GEP->getSourceElementType(), GEP->getPointerOperand(), Indices,
1445 CostKind, AccessTy);
1446 }
1447 }
1448 return Cost;
1449 }
1450
1453 TTI::TargetCostKind CostKind) const override {
1454 using namespace llvm::PatternMatch;
1455
1456 auto *TargetTTI = static_cast<const T *>(this);
1457 // Handle non-intrinsic calls, invokes, and callbr.
1458 // FIXME: Unlikely to be true for anything but CodeSize.
1459 auto *CB = dyn_cast<CallBase>(U);
1460 if (CB && !isa<IntrinsicInst>(U)) {
1461 if (const Function *F = CB->getCalledFunction()) {
1462 if (!TargetTTI->isLoweredToCall(F))
1463 return TTI::TCC_Basic; // Give a basic cost if it will be lowered
1464
1465 return TTI::TCC_Basic * (F->getFunctionType()->getNumParams() + 1);
1466 }
1467 // For indirect or other calls, scale cost by number of arguments.
1468 return TTI::TCC_Basic * (CB->arg_size() + 1);
1469 }
1470
1471 Type *Ty = U->getType();
1472 unsigned Opcode = Operator::getOpcode(U);
1473 auto *I = dyn_cast<Instruction>(U);
1474 switch (Opcode) {
1475 default:
1476 break;
1477 case Instruction::Call: {
1478 assert(isa<IntrinsicInst>(U) && "Unexpected non-intrinsic call");
1479 auto *Intrinsic = cast<IntrinsicInst>(U);
1480 IntrinsicCostAttributes CostAttrs(Intrinsic->getIntrinsicID(), *CB);
1481 return TargetTTI->getIntrinsicInstrCost(CostAttrs, CostKind);
1482 }
1483 case Instruction::UncondBr:
1484 case Instruction::CondBr:
1485 case Instruction::Ret:
1486 case Instruction::PHI:
1487 case Instruction::Switch:
1488 return TargetTTI->getCFInstrCost(Opcode, CostKind, I);
1489 case Instruction::Freeze:
1490 return TTI::TCC_Free;
1491 case Instruction::ExtractValue:
1492 case Instruction::InsertValue:
1493 return TargetTTI->getInsertExtractValueCost(Opcode, CostKind);
1494 case Instruction::Alloca:
1495 if (cast<AllocaInst>(U)->isStaticAlloca())
1496 return TTI::TCC_Free;
1497 break;
1498 case Instruction::GetElementPtr: {
1499 const auto *GEP = cast<GEPOperator>(U);
1500 Type *AccessType = nullptr;
1501 // For now, only provide the AccessType in the simple case where the GEP
1502 // only has one user.
1503 if (GEP->hasOneUser() && I)
1504 AccessType = I->user_back()->getAccessType();
1505
1506 return TargetTTI->getGEPCost(GEP->getSourceElementType(),
1507 Operands.front(), Operands.drop_front(),
1508 CostKind, AccessType);
1509 }
1510 case Instruction::Add:
1511 case Instruction::FAdd:
1512 case Instruction::Sub:
1513 case Instruction::FSub:
1514 case Instruction::Mul:
1515 case Instruction::FMul:
1516 case Instruction::UDiv:
1517 case Instruction::SDiv:
1518 case Instruction::FDiv:
1519 case Instruction::URem:
1520 case Instruction::SRem:
1521 case Instruction::FRem:
1522 case Instruction::Shl:
1523 case Instruction::LShr:
1524 case Instruction::AShr:
1525 case Instruction::And:
1526 case Instruction::Or:
1527 case Instruction::Xor:
1528 case Instruction::FNeg: {
1530 TTI::OperandValueInfo Op2Info;
1531 if (Opcode != Instruction::FNeg)
1532 Op2Info = TTI::getOperandInfo(Operands[1]);
1533 return TargetTTI->getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info,
1534 Op2Info, Operands, I);
1535 }
1536 case Instruction::IntToPtr:
1537 case Instruction::PtrToAddr:
1538 case Instruction::PtrToInt:
1539 case Instruction::SIToFP:
1540 case Instruction::UIToFP:
1541 case Instruction::FPToUI:
1542 case Instruction::FPToSI:
1543 case Instruction::Trunc:
1544 case Instruction::FPTrunc:
1545 case Instruction::BitCast:
1546 case Instruction::FPExt:
1547 case Instruction::SExt:
1548 case Instruction::ZExt:
1549 case Instruction::AddrSpaceCast: {
1550 Type *OpTy = Operands[0]->getType();
1551 return TargetTTI->getCastInstrCost(
1552 Opcode, Ty, OpTy, TTI::getCastContextHint(I), CostKind, I);
1553 }
1554 case Instruction::Store: {
1555 auto *SI = cast<StoreInst>(U);
1556 Type *ValTy = Operands[0]->getType();
1558 return TargetTTI->getMemoryOpCost(Opcode, ValTy, SI->getAlign(),
1559 SI->getPointerAddressSpace(), CostKind,
1560 OpInfo, I);
1561 }
1562 case Instruction::Load: {
1563 auto *LI = cast<LoadInst>(U);
1564 Type *LoadType = U->getType();
1565 // If there is a non-register sized type, the cost estimation may expand
1566 // it to be several instructions to load into multiple registers on the
1567 // target. But, if the only use of the load is a trunc instruction to a
1568 // register sized type, the instruction selector can combine these
1569 // instructions to be a single load. So, in this case, we use the
1570 // destination type of the trunc instruction rather than the load to
1571 // accurately estimate the cost of this load instruction.
1572 if (CostKind == TTI::TCK_CodeSize && LI->hasOneUse() &&
1573 !LoadType->isVectorTy()) {
1574 if (const TruncInst *TI = dyn_cast<TruncInst>(*LI->user_begin()))
1575 LoadType = TI->getDestTy();
1576 }
1577 return TargetTTI->getMemoryOpCost(Opcode, LoadType, LI->getAlign(),
1579 {TTI::OK_AnyValue, TTI::OP_None}, I);
1580 }
1581 case Instruction::Select: {
1582 const Value *Op0, *Op1;
1583 if (match(U, m_LogicalAnd(m_Value(Op0), m_Value(Op1))) ||
1584 match(U, m_LogicalOr(m_Value(Op0), m_Value(Op1)))) {
1585 // select x, y, false --> x & y
1586 // select x, true, y --> x | y
1587 const auto Op1Info = TTI::getOperandInfo(Op0);
1588 const auto Op2Info = TTI::getOperandInfo(Op1);
1589 assert(Op0->getType()->getScalarSizeInBits() == 1 &&
1590 Op1->getType()->getScalarSizeInBits() == 1);
1591
1593 return TargetTTI->getArithmeticInstrCost(
1594 match(U, m_LogicalOr()) ? Instruction::Or : Instruction::And, Ty,
1595 CostKind, Op1Info, Op2Info, Operands, I);
1596 }
1597 const auto Op1Info = TTI::getOperandInfo(Operands[1]);
1598 const auto Op2Info = TTI::getOperandInfo(Operands[2]);
1599 Type *CondTy = Operands[0]->getType();
1600 return TargetTTI->getCmpSelInstrCost(Opcode, U->getType(), CondTy,
1602 CostKind, Op1Info, Op2Info, I);
1603 }
1604 case Instruction::ICmp:
1605 case Instruction::FCmp: {
1606 const auto Op1Info = TTI::getOperandInfo(Operands[0]);
1607 const auto Op2Info = TTI::getOperandInfo(Operands[1]);
1608 Type *ValTy = Operands[0]->getType();
1609 // TODO: Also handle ICmp/FCmp constant expressions.
1610 return TargetTTI->getCmpSelInstrCost(Opcode, ValTy, U->getType(),
1611 I ? cast<CmpInst>(I)->getPredicate()
1613 CostKind, Op1Info, Op2Info, I);
1614 }
1615 case Instruction::InsertElement: {
1616 auto *IE = dyn_cast<InsertElementInst>(U);
1617 if (!IE)
1618 return TTI::TCC_Basic; // FIXME
1619 unsigned Idx = -1;
1620 if (auto *CI = dyn_cast<ConstantInt>(Operands[2]))
1621 if (CI->getValue().getActiveBits() <= 32)
1622 Idx = CI->getZExtValue();
1623 return TargetTTI->getVectorInstrCost(*IE, Ty, CostKind, Idx,
1625 }
1626 case Instruction::ShuffleVector: {
1627 auto *Shuffle = dyn_cast<ShuffleVectorInst>(U);
1628 if (!Shuffle)
1629 return TTI::TCC_Basic; // FIXME
1630
1631 auto *VecTy = cast<VectorType>(U->getType());
1632 auto *VecSrcTy = cast<VectorType>(Operands[0]->getType());
1633 ArrayRef<int> Mask = Shuffle->getShuffleMask();
1634 int NumSubElts, SubIndex;
1635
1636 // Treat undef/poison mask as free (no matter the length).
1637 if (all_of(Mask, [](int M) { return M < 0; }))
1638 return TTI::TCC_Free;
1639
1640 // TODO: move more of this inside improveShuffleKindFromMask.
1641 if (Shuffle->changesLength()) {
1642 // Treat a 'subvector widening' as a free shuffle.
1643 if (Shuffle->increasesLength() && Shuffle->isIdentityWithPadding())
1644 return TTI::TCC_Free;
1645
1646 if (Shuffle->isExtractSubvectorMask(SubIndex))
1647 return TargetTTI->getShuffleCost(TTI::SK_ExtractSubvector, VecTy,
1648 VecSrcTy, CostKind, Mask, SubIndex,
1649 VecTy, Operands, Shuffle);
1650
1651 if (Shuffle->isInsertSubvectorMask(NumSubElts, SubIndex))
1652 return TargetTTI->getShuffleCost(
1653 TTI::SK_InsertSubvector, VecTy, VecSrcTy, CostKind, Mask,
1654 SubIndex,
1655 FixedVectorType::get(VecTy->getScalarType(), NumSubElts),
1656 Operands, Shuffle);
1657
1658 int ReplicationFactor, VF;
1659 if (Shuffle->isReplicationMask(ReplicationFactor, VF)) {
1660 APInt DemandedDstElts = APInt::getZero(Mask.size());
1661 for (auto I : enumerate(Mask)) {
1662 if (I.value() != PoisonMaskElem)
1663 DemandedDstElts.setBit(I.index());
1664 }
1665 return TargetTTI->getReplicationShuffleCost(
1666 VecSrcTy->getElementType(), ReplicationFactor, VF,
1667 DemandedDstElts, CostKind);
1668 }
1669
1670 bool IsUnary = isa<UndefValue>(Operands[1]);
1671 NumSubElts = VecSrcTy->getElementCount().getKnownMinValue();
1672 SmallVector<int, 16> AdjustMask(Mask);
1673
1674 // Widening shuffle - widening the source(s) to the new length
1675 // (treated as free - see above), and then perform the adjusted
1676 // shuffle at that width.
1677 if (Shuffle->increasesLength()) {
1678 for (int &M : AdjustMask)
1679 M = M >= NumSubElts ? (M + (Mask.size() - NumSubElts)) : M;
1680
1681 return TargetTTI->getShuffleCost(
1683 VecTy, CostKind, AdjustMask, 0, nullptr, Operands, Shuffle);
1684 }
1685
1686 // Narrowing shuffle - perform shuffle at original wider width and
1687 // then extract the lower elements.
1688 // FIXME: This can assume widening, which is not true of all vector
1689 // architectures (and is not even the default).
1690 AdjustMask.append(NumSubElts - Mask.size(), PoisonMaskElem);
1691
1692 InstructionCost ShuffleCost = TargetTTI->getShuffleCost(
1694 VecSrcTy, VecSrcTy, CostKind, AdjustMask, 0, nullptr, Operands,
1695 Shuffle);
1696
1697 SmallVector<int, 16> ExtractMask(Mask.size());
1698 std::iota(ExtractMask.begin(), ExtractMask.end(), 0);
1699 return ShuffleCost + TargetTTI->getShuffleCost(
1700 TTI::SK_ExtractSubvector, VecTy, VecSrcTy,
1701 CostKind, ExtractMask, 0, VecTy, {}, Shuffle);
1702 }
1703
1704 if (Shuffle->isIdentity())
1705 return TTI::TCC_Free;
1706
1707 if (Shuffle->isReverse())
1708 return TargetTTI->getShuffleCost(TTI::SK_Reverse, VecTy, VecSrcTy,
1709 CostKind, Mask, 0, nullptr, Operands,
1710 Shuffle);
1711
1712 if (Shuffle->isTranspose())
1713 return TargetTTI->getShuffleCost(TTI::SK_Transpose, VecTy, VecSrcTy,
1714 CostKind, Mask, 0, nullptr, Operands,
1715 Shuffle);
1716
1717 if (Shuffle->isZeroEltSplat())
1718 return TargetTTI->getShuffleCost(TTI::SK_Broadcast, VecTy, VecSrcTy,
1719 CostKind, Mask, 0, nullptr, Operands,
1720 Shuffle);
1721
1722 if (Shuffle->isSingleSource())
1723 return TargetTTI->getShuffleCost(TTI::SK_PermuteSingleSrc, VecTy,
1724 VecSrcTy, CostKind, Mask, 0, nullptr,
1725 Operands, Shuffle);
1726
1727 if (Shuffle->isInsertSubvectorMask(NumSubElts, SubIndex))
1728 return TargetTTI->getShuffleCost(
1729 TTI::SK_InsertSubvector, VecTy, VecSrcTy, CostKind, Mask, SubIndex,
1730 FixedVectorType::get(VecTy->getScalarType(), NumSubElts), Operands,
1731 Shuffle);
1732
1733 if (Shuffle->isSelect())
1734 return TargetTTI->getShuffleCost(TTI::SK_Select, VecTy, VecSrcTy,
1735 CostKind, Mask, 0, nullptr, Operands,
1736 Shuffle);
1737
1738 if (Shuffle->isSplice(SubIndex))
1739 return TargetTTI->getShuffleCost(TTI::SK_Splice, VecTy, VecSrcTy,
1740 CostKind, Mask, SubIndex, nullptr,
1741 Operands, Shuffle);
1742
1743 return TargetTTI->getShuffleCost(TTI::SK_PermuteTwoSrc, VecTy, VecSrcTy,
1744 CostKind, Mask, 0, nullptr, Operands,
1745 Shuffle);
1746 }
1747 case Instruction::ExtractElement: {
1748 auto *EEI = dyn_cast<ExtractElementInst>(U);
1749 if (!EEI)
1750 return TTI::TCC_Basic; // FIXME
1751 unsigned Idx = -1;
1752 if (auto *CI = dyn_cast<ConstantInt>(Operands[1]))
1753 if (CI->getValue().getActiveBits() <= 32)
1754 Idx = CI->getZExtValue();
1755 Type *DstTy = Operands[0]->getType();
1756 return TargetTTI->getVectorInstrCost(*EEI, DstTy, CostKind, Idx);
1757 }
1758 }
1759
1760 // By default, just classify everything remaining as 'basic'.
1761 return TTI::TCC_Basic;
1762 }
1763
1765 auto *TargetTTI = static_cast<const T *>(this);
1766 SmallVector<const Value *, 4> Ops(I->operand_values());
1767 InstructionCost Cost = TargetTTI->getInstructionCost(
1770 }
1771
1772 bool supportsTailCallFor(const CallBase *CB) const override {
1773 return static_cast<const T *>(this)->supportsTailCalls();
1774 }
1775};
1776} // namespace llvm
1777
1778#endif
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
#define LLVM_ABI
Definition Compiler.h:215
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static bool isSigned(unsigned Opcode)
Hexagon Common GEP
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define T
uint64_t IntrinsicInst * II
OptimizedStructLayoutField Field
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
SI Fold Operands
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
This pass exposes codegen information to IR-level passes.
static void computeKnownBits(const Value *V, const APInt &DemandedElts, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth)
Determine which bits of V are known to be either zero or one and return them in the Known bit set.
Class for arbitrary precision integers.
Definition APInt.h:78
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1350
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1086
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
This class represents a conversion between pointers from one address space to another.
an instruction to allocate memory on the stack
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Class to represent array types.
A cache of @llvm.assume calls within a function.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
Conditional Branch instruction.
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
This is an important base class in LLVM.
Definition Constant.h:43
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
static constexpr ElementCount get(ScalarTy MinVal, bool Scalable)
Definition TypeSize.h:311
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
The core instruction combiner logic.
static InstructionCost getInvalid(CostType Val=0)
Class to represent integer types.
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Information for memory intrinsic cost model.
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
The optimization diagnostic interface.
Analysis providing profile information.
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
This node represents a polynomial recurrence on the trip count of the specified loop.
SCEVUse getStepRecurrence(ScalarEvolution &SE) const
Constructs and returns the recurrence indicating how much this expression steps by.
This class represents a constant integer value.
const APInt & getAPInt() const
This class represents an analyzed expression in the program.
The main scalar evolution driver.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
static StackOffset getScalable(int64_t Scalable)
Definition TypeSize.h:40
static StackOffset getFixed(int64_t Fixed)
Definition TypeSize.h:39
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Class to represent struct types.
Multiway switch.
Provides information about what library functions are available for the current target.
virtual bool preferAlternateOpcodeVectorization() const
virtual bool isProfitableLSRChainElement(Instruction *I) const
virtual unsigned getCallerAllocaCost(const CallBase *CB, const AllocaInst *AI) const
virtual unsigned getMinimumLookupTableEntryBitWidth() const
virtual bool getTgtMemIntrinsic(IntrinsicInst *Inst, MemIntrinsicInfo &Info) const
virtual InstructionCost getCostOfKeepingLiveOverCall(ArrayRef< Type * > Tys) const
virtual TailFoldingStyle getPreferredTailFoldingStyle() const
virtual unsigned getMaximumVF(unsigned ElemWidth, unsigned Opcode) const
virtual bool haveFastClmul(IntegerType *Ty) const
virtual bool preferFixedOverScalableIfEqualCost() const
virtual InstructionCost getMulAccReductionCost(bool IsUnsigned, unsigned RedOpcode, Type *ResTy, VectorType *Ty, TTI::TargetCostKind CostKind) const
virtual const DataLayout & getDataLayout() const
virtual std::optional< unsigned > getCacheAssociativity(TargetTransformInfo::CacheLevel Level) const
virtual InstructionCost getCallInstrCost(Function *F, Type *RetTy, ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind) const
virtual bool enableInterleavedAccessVectorization() const
virtual InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, TTI::PartialReductionExtendKind OpAExtend, TTI::PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const
virtual InstructionCost getOperandsScalarizationOverhead(ArrayRef< Type * > Tys, TTI::TargetCostKind CostKind, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
virtual InstructionCost getFPOpCost(Type *Ty) const
virtual bool isLegalMaskedExpandLoad(Type *DataType, Align Alignment) const
virtual TTI::MemCmpExpansionOptions enableMemCmpExpansion(bool OptSize, bool IsZeroCmp) const
virtual bool isLegalToVectorizeLoadChain(unsigned ChainSizeInBytes, Align Alignment, unsigned AddrSpace) const
bool isStridedAccess(const SCEV *Ptr) const
virtual unsigned getAtomicMemIntrinsicMaxElementSize() const
virtual Value * rewriteIntrinsicWithAddressSpace(IntrinsicInst *II, Value *OldV, Value *NewV) const
virtual TargetTransformInfo::VPLegalization getVPLegalizationStrategy(const VPIntrinsic &PI) const
virtual bool enableAggressiveInterleaving(bool LoopHasReductions) const
virtual std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const
virtual bool isLegalMaskedStore(Type *DataType, Align Alignment, unsigned AddressSpace, TTI::MaskKind MaskKind) const
virtual InstructionCost getAddressComputationCost(Type *PtrTy, ScalarEvolution *, const SCEV *, TTI::TargetCostKind) const
virtual bool isLegalBroadcastLoad(Type *ElementTy, ElementCount NumElements) const
virtual InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CxtI=nullptr) const
virtual bool isIndexedLoadLegal(TTI::MemIndexedMode Mode, Type *Ty) const
virtual unsigned adjustInliningThreshold(const CallBase *CB) const
virtual unsigned getLoadVectorFactor(unsigned VF, unsigned LoadSize, unsigned ChainSizeInBytes, VectorType *VecTy) const
virtual bool shouldDropLSRSolutionIfLessProfitable() const
virtual bool hasVolatileVariant(Instruction *I, unsigned AddrSpace) const
virtual bool isLegalMaskedLoad(Type *DataType, Align Alignment, unsigned AddressSpace, TTI::MaskKind MaskKind) const
virtual bool hasDivRemOp(Type *DataType, bool IsSigned) const
virtual bool isLegalStridedLoadStore(Type *DataType, Align Alignment) const
virtual bool isLegalICmpImmediate(int64_t Imm) const
virtual InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo, const Instruction *I) const
virtual bool haveFastSqrt(Type *Ty) const
virtual ElementCount getMinimumVF(unsigned ElemWidth, bool IsScalable) const
virtual bool collectFlatAddressOperands(SmallVectorImpl< int > &OpIndexes, Intrinsic::ID IID) const
virtual bool addrspacesMayAlias(unsigned AS0, unsigned AS1) const
virtual unsigned getRegisterClassForType(bool Vector, Type *Ty=nullptr) const
virtual std::optional< unsigned > getVScaleForTuning() const
virtual InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const
virtual InstructionCost getScalingFactorCost(Type *Ty, GlobalValue *BaseGV, StackOffset BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace) const
virtual unsigned getNumberOfParts(Type *Tp) const
virtual bool isLegalMaskedCompressStore(Type *DataType, Align Alignment) const
virtual bool isHardwareLoopProfitable(Loop *L, ScalarEvolution &SE, AssumptionCache &AC, TargetLibraryInfo *LibInfo, HardwareLoopInfo &HWLoopInfo) const
virtual void getPeelingPreferences(Loop *, ScalarEvolution &, TTI::PeelingPreferences &) const
virtual std::optional< Value * > simplifyDemandedUseBitsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedMask, KnownBits &Known, bool &KnownBitsComputed) const
virtual bool useColdCCForColdCall(Function &F) const
virtual unsigned getNumberOfRegisters(unsigned ClassID) const
virtual bool canHaveNonUndefGlobalInitializerInAddressSpace(unsigned AS) const
virtual APInt getAddrSpaceCastPreservedPtrMask(unsigned SrcAS, unsigned DstAS) const
virtual bool isLegalAddScalableImmediate(int64_t Imm) const
virtual bool isLegalInterleavedAccessType(VectorType *VTy, unsigned Factor, Align Alignment, unsigned AddrSpace) const
virtual bool preferTailFoldingOverEpilogue(TailFoldingInfo *TFI) const
TargetTransformInfoImplBase(TargetTransformInfoImplBase &&Arg)
virtual bool shouldPrefetchAddressSpace(unsigned AS) const
virtual bool forceScalarizeMaskedScatter(VectorType *DataType, Align Alignment) const
virtual uint64_t getMaxMemIntrinsicInlineSizeThreshold() const
virtual KnownBits computeKnownBitsAddrSpaceCast(unsigned FromAS, unsigned ToAS, const KnownBits &FromPtrBits) const
virtual unsigned getMinVectorRegisterBitWidth() const
unsigned minRequiredElementSize(const Value *Val, bool &isSigned) const
virtual bool shouldBuildLookupTablesForConstant(Constant *C) const
virtual bool isFPVectorizationPotentiallyUnsafe() const
virtual bool isLegalToVectorizeReduction(const RecurrenceDescriptor &RdxDesc, ElementCount VF) const
virtual InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const
virtual bool isLegalAltInstr(VectorType *VecTy, unsigned Opcode0, unsigned Opcode1, const SmallBitVector &OpcodeMask) const
virtual InstructionCost getIndexedVectorInstrCostFromEnd(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index) const
virtual std::optional< unsigned > getCacheSize(TargetTransformInfo::CacheLevel Level) const
virtual InstructionCost getExtractWithExtendCost(unsigned Opcode, Type *Dst, VectorType *VecTy, unsigned Index, TTI::TargetCostKind CostKind) const
virtual bool shouldTreatInstructionLikeSelect(const Instruction *I) const
virtual std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const
virtual unsigned getEpilogueVectorizationMinVF() const
virtual std::pair< const Value *, unsigned > getPredicatedAddrSpace(const Value *V) const
virtual bool shouldMaximizeVectorBandwidth(TargetTransformInfo::RegisterKind K) const
virtual void getMemcpyLoopResidualLoweringType(SmallVectorImpl< Type * > &OpsOut, LLVMContext &Context, unsigned RemainingBytes, unsigned SrcAddrSpace, unsigned DestAddrSpace, Align SrcAlign, Align DestAlign, std::optional< uint32_t > AtomicCpySize) const
virtual unsigned getStoreMinimumVF(unsigned VF, Type *, Type *, Align, unsigned) const
virtual InstructionCost getRegisterClassReloadCost(unsigned ClassID, TTI::TargetCostKind CostKind) const
virtual TTI::PopcntSupportKind getPopcntSupport(unsigned IntTyWidthInBit) const
virtual TTI::AddressingModeKind getPreferredAddressingMode(const Loop *L, ScalarEvolution *SE) const
virtual bool forceScalarizeMaskedGather(VectorType *DataType, Align Alignment) const
virtual unsigned getMaxPrefetchIterationsAhead() const
virtual bool allowVectorElementIndexingUsingGEP() const
virtual bool isUniform(const Instruction *I, const SmallBitVector &UniformArgs) const
virtual InstructionCost getInstructionCost(const User *U, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind) const
virtual TTI::ReductionShuffle getPreferredExpandedReductionShuffle(const IntrinsicInst *II) const
const SCEVConstant * getConstantStrideStep(ScalarEvolution *SE, const SCEV *Ptr) const
virtual bool hasBranchDivergence(const Function *F=nullptr) const
virtual InstructionCost getArithmeticReductionCost(unsigned, VectorType *, std::optional< FastMathFlags > FMF, TTI::TargetCostKind) const
virtual bool isProfitableToHoist(Instruction *I) const
virtual const char * getRegisterClassName(unsigned ClassID) const
virtual InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *, FastMathFlags, TTI::TargetCostKind) const
virtual bool isLegalToVectorizeLoad(LoadInst *LI) const
virtual unsigned getLoadStoreVecRegBitWidth(unsigned AddrSpace) const
virtual InstructionCost getAltInstrCost(VectorType *VecTy, unsigned Opcode0, unsigned Opcode1, const SmallBitVector &OpcodeMask, TTI::TargetCostKind CostKind) const
virtual unsigned getInlineCallPenalty(const Function *F, const CallBase &Call, unsigned DefaultCallPenalty) const
virtual unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const
virtual InstructionCost getVectorInstrCost(const Instruction &I, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
virtual bool isVectorShiftByScalarCheap(Type *Ty) const
virtual bool isLegalNTStore(Type *DataType, Align Alignment) const
virtual APInt getFeatureMask(const Function &F) const
virtual InstructionCost getMemIntrinsicInstrCost(const MemIntrinsicCostAttributes &MICA, TTI::TargetCostKind CostKind) const
virtual std::optional< unsigned > getMinPageSize() const
virtual bool shouldCopyAttributeWhenOutliningFrom(const Function *Caller, const Attribute &Attr) const
virtual unsigned getRegUsageForType(Type *Ty) const
virtual bool isLegalAddressingMode(Type *Ty, GlobalValue *BaseGV, int64_t BaseOffset, bool HasBaseReg, int64_t Scale, unsigned AddrSpace, Instruction *I=nullptr, int64_t ScalableOffset=0) const
virtual InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
virtual bool isElementTypeLegalForScalableVector(Type *Ty) const
virtual bool isLoweredToCall(const Function *F) const
virtual bool isLegalMaskedScatter(Type *DataType, Align Alignment) const
virtual bool isTruncateFree(Type *Ty1, Type *Ty2) const
virtual InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, Value *Scalar, ArrayRef< std::tuple< Value *, User *, int > > ScalarUserAndIdx, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
virtual InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info, TTI::OperandValueInfo Opd2Info, ArrayRef< const Value * > Args, const Instruction *CxtI=nullptr) const
virtual InstructionCost getRegisterClassSpillCost(unsigned ClassID, TTI::TargetCostKind CostKind) const
virtual bool isIndexedStoreLegal(TTI::MemIndexedMode Mode, Type *Ty) const
virtual BranchProbability getPredictableBranchThreshold() const
virtual InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const
virtual bool isValidAddrSpaceCast(unsigned FromAS, unsigned ToAS) const
virtual InstructionCost getReplicationShuffleCost(Type *EltTy, int ReplicationFactor, int VF, const APInt &DemandedDstElts, TTI::TargetCostKind CostKind) const
virtual bool isLegalToVectorizeStore(StoreInst *SI) const
virtual bool areInlineCompatible(const Function *Caller, const Function *Callee) const
virtual bool isTargetIntrinsicWithStructReturnOverloadAtField(Intrinsic::ID ID, int RetIdx) const
virtual bool hasConditionalLoadStoreForType(Type *Ty, bool IsStore) const
virtual bool canSaveCmp(Loop *L, CondBrInst **BI, ScalarEvolution *SE, LoopInfo *LI, DominatorTree *DT, AssumptionCache *AC, TargetLibraryInfo *LibInfo) const
virtual bool preferInLoopReduction(RecurKind Kind, Type *Ty) const
virtual bool isMultiversionedFunction(const Function &F) const
virtual InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const
virtual bool isNoopAddrSpaceCast(unsigned, unsigned) const
virtual bool isExpensiveToSpeculativelyExecute(const Instruction *I) const
virtual bool isLSRCostLess(const TTI::LSRCost &C1, const TTI::LSRCost &C2) const
virtual bool isLegalMaskedVectorHistogram(Type *AddrType, Type *DataType) const
virtual bool isLegalMaskedGather(Type *DataType, Align Alignment) const
virtual unsigned getEstimatedNumberOfCaseClusters(const SwitchInst &SI, unsigned &JTSize, ProfileSummaryInfo *PSI, BlockFrequencyInfo *BFI) const
virtual bool isLegalAddImmediate(int64_t Imm) const
virtual InstructionCost getInsertExtractValueCost(unsigned Opcode, TTI::TargetCostKind CostKind) const
virtual InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I) const
virtual ValueUniformity getValueUniformity(const Value *V) const
virtual bool isLegalNTLoad(Type *DataType, Align Alignment) const
virtual InstructionCost getBranchMispredictPenalty() const
virtual bool isTargetIntrinsicWithOverloadTypeAtArg(Intrinsic::ID ID, int OpdIdx) const
virtual InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const
virtual InstructionCost getIntImmCodeSizeCost(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty) const
bool isConstantStridedAccessLessThan(ScalarEvolution *SE, const SCEV *Ptr, int64_t MergeDistance) const
virtual Value * getOrCreateResultFromMemIntrinsic(IntrinsicInst *Inst, Type *ExpectedType, bool CanCreate=true) const
virtual bool enableMaskedInterleavedAccessVectorization() const
virtual std::pair< KnownBits, KnownBits > computeKnownBitsAddrSpaceCast(unsigned ToAS, const Value &PtrOp) const
virtual Type * getMemcpyLoopLoweringType(LLVMContext &Context, Value *Length, unsigned SrcAddrSpace, unsigned DestAddrSpace, Align SrcAlign, Align DestAlign, std::optional< uint32_t > AtomicElementSize) const
virtual unsigned getInliningThresholdMultiplier() const
TargetTransformInfoImplBase(const DataLayout &DL)
virtual InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const
virtual InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info, TTI::OperandValueInfo Op2Info, const Instruction *I) const
virtual bool shouldExpandReduction(const IntrinsicInst *II) const
virtual bool isLegalToVectorizeStoreChain(unsigned ChainSizeInBytes, Align Alignment, unsigned AddrSpace) const
virtual unsigned getGISelRematGlobalCost() const
virtual InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond, bool UseMaskForGaps) const
virtual bool isTypeLegal(Type *Ty) const
virtual unsigned getAssumedAddrSpace(const Value *V) const
virtual bool allowsMisalignedMemoryAccesses(LLVMContext &Context, unsigned BitWidth, unsigned AddressSpace, Align Alignment, unsigned *Fast) const
virtual unsigned getStoreVectorFactor(unsigned VF, unsigned StoreSize, unsigned ChainSizeInBytes, VectorType *VecTy) const
virtual InstructionCost getExtendedReductionCost(unsigned Opcode, bool IsUnsigned, Type *ResTy, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const
virtual unsigned getInliningCostBenefitAnalysisSavingsMultiplier() const
virtual bool areTypesABICompatible(const Function *Caller, const Function *Callee, ArrayRef< Type * > Types) const
virtual unsigned getNumBytesToPadGlobalArray(unsigned Size, Type *ArrayType) const
virtual bool preferToKeepConstantsAttached(const Instruction &Inst, const Function &Fn) const
virtual bool isFCmpOrdCheaperThanFCmpZero(Type *Ty) const
virtual bool supportsTailCallFor(const CallBase *CB) const
virtual bool shouldConsiderAddressTypePromotion(const Instruction &I, bool &AllowPromotionWithoutCommonHeader) const
virtual InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, const TTI::TargetCostKind CostKind) const
virtual InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const
virtual bool isTargetIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx) const
virtual bool shouldConsiderVectorizationRegPressure() const
virtual InstructionCost getMemcpyCost(const Instruction *I) const
virtual unsigned getInliningCostBenefitAnalysisProfitableMultiplier() const
virtual bool useFastCCForInternalCall(Function &F) const
virtual bool preferEpilogueVectorization(ElementCount Iters) const
virtual void getUnrollingPreferences(Loop *, ScalarEvolution &, TTI::UnrollingPreferences &, OptimizationRemarkEmitter *) const
TargetTransformInfoImplBase(const TargetTransformInfoImplBase &Arg)=default
virtual bool isProfitableToSinkOperands(Instruction *I, SmallVectorImpl< Use * > &Ops) const
virtual bool supportsEfficientVectorElementLoadStore() const
virtual unsigned getMinPrefetchStride(unsigned NumMemAccesses, unsigned NumStridedMemAccesses, unsigned NumPrefetches, bool HasCall) const
virtual APInt getPriorityMask(const Function &F) const
virtual unsigned getMinTripCountTailFoldingThreshold() const
virtual TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const
virtual void collectKernelLaunchBounds(const Function &F, SmallVectorImpl< std::pair< StringRef, int64_t > > &LB) const
bool supportsTailCallFor(const CallBase *CB) const override
bool isExpensiveToSpeculativelyExecute(const Instruction *I) const override
InstructionCost getInstructionCost(const User *U, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind) const override
InstructionCost getPointersChainCost(ArrayRef< const Value * > Ptrs, const Value *Base, const TTI::PointersChainInfo &Info, Type *AccessTy, TTI::TargetCostKind CostKind) const override
InstructionCost getGEPCost(Type *PointeeType, const Value *Ptr, ArrayRef< const Value * > Operands, TTI::TargetCostKind CostKind, Type *AccessType) const override
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
static LLVM_ABI CastContextHint getCastContextHint(const Instruction *I)
Calculates a CastContextHint from I.
MaskKind
Some targets only support masked load/store with a constant mask.
static LLVM_ABI OperandValueInfo getOperandInfo(const Value *V)
Collect properties of V used in cost analysis, e.g. OP_PowerOf2.
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
PopcntSupportKind
Flags indicating the kind of support for population count.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Expensive
The cost of a 'div' instruction on x86.
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
MemIndexedMode
The type of load/store indexing.
AddressingModeKind
Which addressing mode Loop Strength Reduction will try to generate.
@ AMK_None
Don't prefer any addressing mode.
static LLVM_ABI VectorInstrContext getVectorInstrContextHint(const Instruction *I)
Calculates a VectorInstrContext from I.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_InsertSubvector
InsertSubvector. Index indicates start offset.
@ SK_Select
Selects elements from the corresponding lane of either source operand.
@ SK_PermuteSingleSrc
Shuffle elements of single source vector with any shuffle mask.
@ SK_Transpose
Transpose two vectors.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_PermuteTwoSrc
Merge elements from two source vectors into one with any shuffle mask.
@ SK_Reverse
Reverse the order of the vector.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
@ None
The cast is not used with a load/store of any kind.
CacheLevel
The possible cache levels.
This class represents a truncation of integer types.
static constexpr TypeSize get(ScalarTy Quantity, bool Scalable)
Definition TypeSize.h:336
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
Definition Type.h:280
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
This is the common base class for vector predication intrinsics.
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:713
Base class of all SIMD vector types.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
bool match(Val *V, const Pattern &P)
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_Value()
Match an arbitrary value and ignore it.
auto m_Constant()
Match an arbitrary Constant and ignore it.
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
LogicalOp_match< LHS, RHS, Instruction::Or > m_LogicalOr(const LHS &L, const RHS &R)
Matches L || R either in the form of L | R or L ?
This is an optimization pass for GlobalISel generic memory operations.
@ Length
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
InstructionCost Cost
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr int PoisonMaskElem
RecurKind
These are the kinds of recurrences that we support.
@ Fast
Assign the register banks as fast as possible (default).
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
gep_type_iterator gep_type_begin(const User *GEP)
@ DataWithoutLaneMask
Same as Data, but avoids using the get.active.lane.mask intrinsic to calculate the mask and instead i...
ValueUniformity
Enum describing how values behave with respect to uniformity and divergence, to answer the question: ...
Definition Uniformity.h:18
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Attributes of a target dependent hardware loop.
KnownBits anyextOrTrunc(unsigned BitWidth) const
Return known bits for an "any" extension or truncation of the value we're tracking.
Definition KnownBits.h:190
Information about a load/store intrinsic defined by the target.
Returns options for expansion of memcmp. IsZeroCmp is.
Describe known properties for a set of pointers.
Parameters that control the generic loop unrolling transformation.