LLVM 24.0.0git
SystemZTargetTransformInfo.cpp
Go to the documentation of this file.
1//===-- SystemZTargetTransformInfo.cpp - SystemZ-specific TTI -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements a TargetTransformInfo analysis pass specific to the
10// SystemZ target machine. It uses the target's detailed information to provide
11// more precise answers to certain TTI queries, while letting the target
12// independent and default TTI implementations handle the rest.
13//
14//===----------------------------------------------------------------------===//
15
23#include "llvm/IR/Intrinsics.h"
24#include "llvm/Support/Debug.h"
27
28using namespace llvm;
29
30#define DEBUG_TYPE "systemztti"
31
32//===----------------------------------------------------------------------===//
33//
34// SystemZ cost model.
35//
36//===----------------------------------------------------------------------===//
37
38static bool isUsedAsMemCpySource(const Value *V, bool &OtherUse) {
39 bool UsedAsMemCpySource = false;
40 for (const User *U : V->users())
41 if (const Instruction *User = dyn_cast<Instruction>(U)) {
43 UsedAsMemCpySource |= isUsedAsMemCpySource(User, OtherUse);
44 continue;
45 }
46 if (const MemCpyInst *Memcpy = dyn_cast<MemCpyInst>(User)) {
47 if (Memcpy->getOperand(1) == V && !Memcpy->isVolatile()) {
48 UsedAsMemCpySource = true;
49 continue;
50 }
51 }
52 OtherUse = true;
53 }
54 return UsedAsMemCpySource;
55}
56
57static void countNumMemAccesses(const Value *Ptr, unsigned &NumStores,
58 unsigned &NumLoads, const Function *F) {
59 if (!isa<PointerType>(Ptr->getType()))
60 return;
61 for (const User *U : Ptr->users())
62 if (const Instruction *User = dyn_cast<Instruction>(U)) {
63 if (User->getParent()->getParent() == F) {
64 if (const auto *SI = dyn_cast<StoreInst>(User)) {
65 if (SI->getPointerOperand() == Ptr && !SI->isVolatile())
66 NumStores++;
67 } else if (const auto *LI = dyn_cast<LoadInst>(User)) {
68 if (LI->getPointerOperand() == Ptr && !LI->isVolatile())
69 NumLoads++;
70 } else if (const auto *GEP = dyn_cast<GetElementPtrInst>(User)) {
71 if (GEP->getPointerOperand() == Ptr)
72 countNumMemAccesses(GEP, NumStores, NumLoads, F);
73 }
74 }
75 }
76}
77
79 unsigned Bonus = 0;
80 const Function *Caller = CB->getParent()->getParent();
81 const Function *Callee = CB->getCalledFunction();
82 if (!Callee)
83 return 0;
84
85 // Increase the threshold if an incoming argument is used only as a memcpy
86 // source.
87 for (const Argument &Arg : Callee->args()) {
88 bool OtherUse = false;
89 if (isUsedAsMemCpySource(&Arg, OtherUse) && !OtherUse) {
90 Bonus = 1000;
91 break;
92 }
93 }
94
95 // Give bonus for globals used much in both caller and a relatively small
96 // callee.
97 unsigned InstrCount = 0;
99 for (auto &I : instructions(Callee)) {
100 if (++InstrCount == 200) {
101 Ptr2NumUses.clear();
102 break;
103 }
104 if (const auto *SI = dyn_cast<StoreInst>(&I)) {
105 if (!SI->isVolatile())
106 if (auto *GV = dyn_cast<GlobalVariable>(SI->getPointerOperand()))
107 Ptr2NumUses[GV]++;
108 } else if (const auto *LI = dyn_cast<LoadInst>(&I)) {
109 if (!LI->isVolatile())
110 if (auto *GV = dyn_cast<GlobalVariable>(LI->getPointerOperand()))
111 Ptr2NumUses[GV]++;
112 } else if (const auto *GEP = dyn_cast<GetElementPtrInst>(&I)) {
113 if (auto *GV = dyn_cast<GlobalVariable>(GEP->getPointerOperand())) {
114 unsigned NumStores = 0, NumLoads = 0;
115 countNumMemAccesses(GEP, NumStores, NumLoads, Callee);
116 Ptr2NumUses[GV] += NumLoads + NumStores;
117 }
118 }
119 }
120
121 for (auto [Ptr, NumCalleeUses] : Ptr2NumUses)
122 if (NumCalleeUses > 10) {
123 unsigned CallerStores = 0, CallerLoads = 0;
124 countNumMemAccesses(Ptr, CallerStores, CallerLoads, Caller);
125 if (CallerStores + CallerLoads > 10) {
126 Bonus = 1000;
127 break;
128 }
129 }
130
131 // Give bonus when Callee accesses an Alloca of Caller heavily.
132 unsigned NumStores = 0;
133 unsigned NumLoads = 0;
134 for (unsigned OpIdx = 0; OpIdx != Callee->arg_size(); ++OpIdx) {
135 Value *CallerArg = CB->getArgOperand(OpIdx);
136 Argument *CalleeArg = Callee->getArg(OpIdx);
137 if (isa<AllocaInst>(CallerArg))
138 countNumMemAccesses(CalleeArg, NumStores, NumLoads, Callee);
139 }
140 if (NumLoads > 10)
141 Bonus += NumLoads * 50;
142 if (NumStores > 10)
143 Bonus += NumStores * 50;
144 Bonus = std::min(Bonus, unsigned(1000));
145
146 LLVM_DEBUG(if (Bonus)
147 dbgs() << "++ SZTTI Adding inlining bonus: " << Bonus << "\n";);
148 return Bonus;
149}
150
154 assert(Ty->isIntegerTy());
155
156 unsigned BitSize = Ty->getPrimitiveSizeInBits();
157 // There is no cost model for constants with a bit size of 0. Return TCC_Free
158 // here, so that constant hoisting will ignore this constant.
159 if (BitSize == 0)
160 return TTI::TCC_Free;
161 // No cost model for operations on integers larger than 128 bit implemented yet.
162 if ((!ST->hasVector() && BitSize > 64) || BitSize > 128)
163 return TTI::TCC_Free;
164
165 if (Imm == 0)
166 return TTI::TCC_Free;
167
168 if (Imm.getBitWidth() <= 64) {
169 // Constants loaded via lgfi.
170 if (isInt<32>(Imm.getSExtValue()))
171 return TTI::TCC_Basic;
172 // Constants loaded via llilf.
173 if (isUInt<32>(Imm.getZExtValue()))
174 return TTI::TCC_Basic;
175 // Constants loaded via llihf:
176 if ((Imm.getZExtValue() & 0xffffffff) == 0)
177 return TTI::TCC_Basic;
178
179 return 2 * TTI::TCC_Basic;
180 }
181
182 // i128 immediates loads from Constant Pool
183 return 2 * TTI::TCC_Basic;
184}
185
187 const APInt &Imm, Type *Ty,
189 Instruction *Inst) const {
190 assert(Ty->isIntegerTy());
191
192 unsigned BitSize = Ty->getPrimitiveSizeInBits();
193 // There is no cost model for constants with a bit size of 0. Return TCC_Free
194 // here, so that constant hoisting will ignore this constant.
195 if (BitSize == 0)
196 return TTI::TCC_Free;
197 // No cost model for operations on integers larger than 64 bit implemented yet.
198 if (BitSize > 64)
199 return TTI::TCC_Free;
200
201 switch (Opcode) {
202 default:
203 return TTI::TCC_Free;
204 case Instruction::GetElementPtr:
205 // Always hoist the base address of a GetElementPtr. This prevents the
206 // creation of new constants for every base constant that gets constant
207 // folded with the offset.
208 if (Idx == 0)
209 return 2 * TTI::TCC_Basic;
210 return TTI::TCC_Free;
211 case Instruction::Store:
212 if (Idx == 0 && Imm.getBitWidth() <= 64) {
213 // Any 8-bit immediate store can by implemented via mvi.
214 if (BitSize == 8)
215 return TTI::TCC_Free;
216 // 16-bit immediate values can be stored via mvhhi/mvhi/mvghi.
217 if (isInt<16>(Imm.getSExtValue()))
218 return TTI::TCC_Free;
219 }
220 break;
221 case Instruction::ICmp:
222 if (Idx == 1 && Imm.getBitWidth() <= 64) {
223 // Comparisons against signed 32-bit immediates implemented via cgfi.
224 if (isInt<32>(Imm.getSExtValue()))
225 return TTI::TCC_Free;
226 // Comparisons against unsigned 32-bit immediates implemented via clgfi.
227 if (isUInt<32>(Imm.getZExtValue()))
228 return TTI::TCC_Free;
229 }
230 break;
231 case Instruction::Add:
232 case Instruction::Sub:
233 if (Idx == 1 && Imm.getBitWidth() <= 64) {
234 // We use algfi/slgfi to add/subtract 32-bit unsigned immediates.
235 if (isUInt<32>(Imm.getZExtValue()))
236 return TTI::TCC_Free;
237 // Or their negation, by swapping addition vs. subtraction.
238 if (isUInt<32>(-Imm.getSExtValue()))
239 return TTI::TCC_Free;
240 }
241 break;
242 case Instruction::Mul:
243 if (Idx == 1 && Imm.getBitWidth() <= 64) {
244 // We use msgfi to multiply by 32-bit signed immediates.
245 if (isInt<32>(Imm.getSExtValue()))
246 return TTI::TCC_Free;
247 }
248 break;
249 case Instruction::Or:
250 case Instruction::Xor:
251 if (Idx == 1 && Imm.getBitWidth() <= 64) {
252 // Masks supported by oilf/xilf.
253 if (isUInt<32>(Imm.getZExtValue()))
254 return TTI::TCC_Free;
255 // Masks supported by oihf/xihf.
256 if ((Imm.getZExtValue() & 0xffffffff) == 0)
257 return TTI::TCC_Free;
258 }
259 break;
260 case Instruction::And:
261 if (Idx == 1 && Imm.getBitWidth() <= 64) {
262 // Any 32-bit AND operation can by implemented via nilf.
263 if (BitSize <= 32)
264 return TTI::TCC_Free;
265 // 64-bit masks supported by nilf.
266 if (isUInt<32>(~Imm.getZExtValue()))
267 return TTI::TCC_Free;
268 // 64-bit masks supported by nilh.
269 if ((Imm.getZExtValue() & 0xffffffff) == 0xffffffff)
270 return TTI::TCC_Free;
271 // Some 64-bit AND operations can be implemented via risbg.
272 const SystemZInstrInfo *TII = ST->getInstrInfo();
273 unsigned Start, End;
274 if (TII->isRxSBGMask(Imm.getZExtValue(), BitSize, Start, End))
275 return TTI::TCC_Free;
276 }
277 break;
278 case Instruction::Shl:
279 case Instruction::LShr:
280 case Instruction::AShr:
281 // Always return TCC_Free for the shift value of a shift instruction.
282 if (Idx == 1)
283 return TTI::TCC_Free;
284 break;
285 case Instruction::UDiv:
286 case Instruction::SDiv:
287 case Instruction::URem:
288 case Instruction::SRem:
289 case Instruction::Trunc:
290 case Instruction::ZExt:
291 case Instruction::SExt:
292 case Instruction::IntToPtr:
293 case Instruction::PtrToInt:
294 case Instruction::BitCast:
295 case Instruction::PHI:
296 case Instruction::Call:
297 case Instruction::Select:
298 case Instruction::Ret:
299 case Instruction::Load:
300 break;
301 }
302
304}
305
308 const APInt &Imm, Type *Ty,
310 assert(Ty->isIntegerTy());
311
312 unsigned BitSize = Ty->getPrimitiveSizeInBits();
313 // There is no cost model for constants with a bit size of 0. Return TCC_Free
314 // here, so that constant hoisting will ignore this constant.
315 if (BitSize == 0)
316 return TTI::TCC_Free;
317 // No cost model for operations on integers larger than 64 bit implemented yet.
318 if (BitSize > 64)
319 return TTI::TCC_Free;
320
321 switch (IID) {
322 default:
323 return TTI::TCC_Free;
324 case Intrinsic::sadd_with_overflow:
325 case Intrinsic::uadd_with_overflow:
326 case Intrinsic::ssub_with_overflow:
327 case Intrinsic::usub_with_overflow:
328 // These get expanded to include a normal addition/subtraction.
329 if (Idx == 1 && Imm.getBitWidth() <= 64) {
330 if (isUInt<32>(Imm.getZExtValue()))
331 return TTI::TCC_Free;
332 if (isUInt<32>(-Imm.getSExtValue()))
333 return TTI::TCC_Free;
334 }
335 break;
336 case Intrinsic::smul_with_overflow:
337 case Intrinsic::umul_with_overflow:
338 // These get expanded to include a normal multiplication.
339 if (Idx == 1 && Imm.getBitWidth() <= 64) {
340 if (isInt<32>(Imm.getSExtValue()))
341 return TTI::TCC_Free;
342 }
343 break;
344 case Intrinsic::experimental_stackmap:
345 if ((Idx < 2) || (Imm.getBitWidth() <= 64 && isInt<64>(Imm.getSExtValue())))
346 return TTI::TCC_Free;
347 break;
348 case Intrinsic::experimental_patchpoint_void:
349 case Intrinsic::experimental_patchpoint:
350 if ((Idx < 4) || (Imm.getBitWidth() <= 64 && isInt<64>(Imm.getSExtValue())))
351 return TTI::TCC_Free;
352 break;
353 }
355}
356
358SystemZTTIImpl::getPopcntSupport(unsigned TyWidth) const {
359 assert(isPowerOf2_32(TyWidth) && "Type width must be power of 2");
360 if (ST->hasPopulationCount() && TyWidth <= 64)
362 return TTI::PSK_Software;
363}
364
367 OptimizationRemarkEmitter *ORE) const {
368 // Find out if L contains a call, what the machine instruction count
369 // estimate is, and how many stores there are.
370 bool HasCall = false;
371 InstructionCost NumStores = 0;
372 for (auto &BB : L->blocks())
373 for (auto &I : *BB) {
374 if (isa<CallInst>(&I) || isa<InvokeInst>(&I)) {
375 if (const Function *F = cast<CallBase>(I).getCalledFunction()) {
376 if (isLoweredToCall(F))
377 HasCall = true;
378 if (F->getIntrinsicID() == Intrinsic::memcpy ||
379 F->getIntrinsicID() == Intrinsic::memset)
380 NumStores++;
381 } else { // indirect call.
382 HasCall = true;
383 }
384 }
385 if (isa<StoreInst>(&I)) {
386 Type *MemAccessTy = I.getOperand(0)->getType();
387 NumStores += getMemoryOpCost(Instruction::Store, MemAccessTy, Align(),
389 }
390 }
391
392 // The z13 processor will run out of store tags if too many stores
393 // are fed into it too quickly. Therefore make sure there are not
394 // too many stores in the resulting unrolled loop.
395 unsigned const NumStoresVal = NumStores.getValue();
396 unsigned const Max = (NumStoresVal ? (12 / NumStoresVal) : UINT_MAX);
397
398 if (HasCall) {
399 // Only allow full unrolling if loop has any calls.
400 UP.FullUnrollMaxCount = Max;
401 UP.MaxCount = 1;
402 return;
403 }
404
405 UP.MaxCount = Max;
406 if (UP.MaxCount <= 1)
407 return;
408
409 // Allow partial and runtime trip count unrolling.
410 UP.Partial = UP.Runtime = true;
411
412 UP.PartialThreshold = 75;
414
415 // Allow expensive instructions in the pre-header of the loop.
416 UP.AllowExpensiveTripCount = true;
417
418 UP.Force = true;
419}
420
425
428 const TargetTransformInfo::LSRCost &C2) const {
429 // SystemZ specific: check instruction count (first), and don't care about
430 // ImmCost, since offsets are checked explicitly.
431 return std::tie(C1.Insns, C1.NumRegs, C1.AddRecCost,
432 C1.NumIVMuls, C1.NumBaseAdds,
433 C1.ScaleCost, C1.SetupCost) <
434 std::tie(C2.Insns, C2.NumRegs, C2.AddRecCost,
435 C2.NumIVMuls, C2.NumBaseAdds,
436 C2.ScaleCost, C2.SetupCost);
437}
438
439unsigned SystemZTTIImpl::getNumberOfRegisters(unsigned ClassID) const {
440 bool Vector = (ClassID == 1);
441 if (!Vector)
442 // Discount the stack pointer. Also leave out %r0, since it can't
443 // be used in an address.
444 return 14;
445 if (ST->hasVector())
446 return 32;
447 return 0;
448}
449
452 switch (K) {
454 return TypeSize::getFixed(64);
456 return TypeSize::getFixed(ST->hasVector() ? 128 : 0);
458 return TypeSize::getScalable(0);
459 }
460
461 llvm_unreachable("Unsupported register kind");
462}
463
464unsigned SystemZTTIImpl::getMinPrefetchStride(unsigned NumMemAccesses,
465 unsigned NumStridedMemAccesses,
466 unsigned NumPrefetches,
467 bool HasCall) const {
468 // Don't prefetch a loop with many far apart accesses.
469 if (NumPrefetches > 16)
470 return UINT_MAX;
471
472 // Emit prefetch instructions for smaller strides in cases where we think
473 // the hardware prefetcher might not be able to keep up.
474 if (NumStridedMemAccesses > 32 && !HasCall &&
475 (NumMemAccesses - NumStridedMemAccesses) * 32 <= NumStridedMemAccesses)
476 return 1;
477
478 return ST->hasMiscellaneousExtensions3() ? 8192 : 2048;
479}
480
481unsigned
483 bool HasUnorderedReductions) const {
484 return VF.isVector() ? 8 : 1;
485}
486
487bool SystemZTTIImpl::hasDivRemOp(Type *DataType, bool IsSigned) const {
488 EVT VT = TLI->getValueType(DL, DataType);
489 return (VT.isScalarInteger() && TLI->isTypeLegal(VT));
490}
491
492static bool isFreeEltLoad(const Value *Op) {
493 if (isa<LoadInst>(Op) && Op->hasOneUse()) {
494 const Instruction *UserI = cast<Instruction>(*Op->user_begin());
495 return !isa<StoreInst>(UserI); // Prefer MVC
496 }
497 return false;
498}
499
501 VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract,
502 TTI::TargetCostKind CostKind, bool ForPoisonSrc, ArrayRef<Value *> VL,
503 TTI::VectorInstrContext VIC) const {
504 unsigned NumElts = cast<FixedVectorType>(Ty)->getNumElements();
506
507 if (Insert && Ty->isIntOrIntVectorTy(64)) {
508 // VLVGP will insert two GPRs with one instruction, while VLE will load
509 // an element directly with no extra cost
510 assert((VL.empty() || VL.size() == NumElts) &&
511 "Type does not match the number of values.");
512 InstructionCost CurrVectorCost = 0;
513 for (unsigned Idx = 0; Idx < NumElts; ++Idx) {
514 if (DemandedElts[Idx] && !(VL.size() && isFreeEltLoad(VL[Idx])))
515 ++CurrVectorCost;
516 if (Idx % 2 == 1) {
517 Cost += std::min(InstructionCost(1), CurrVectorCost);
518 CurrVectorCost = 0;
519 }
520 }
521 Insert = false;
522 }
523
524 Cost += BaseT::getScalarizationOverhead(Ty, DemandedElts, Insert, Extract,
525 CostKind, ForPoisonSrc, VL);
526 return Cost;
527}
528
529// Return the bit size for the scalar type or vector element
530// type. getScalarSizeInBits() returns 0 for a pointer type.
531static unsigned getScalarSizeInBits(Type *Ty) {
532 unsigned Size =
533 (Ty->isPtrOrPtrVectorTy() ? 64U : Ty->getScalarSizeInBits());
534 assert(Size > 0 && "Element must have non-zero size.");
535 return Size;
536}
537
538// getNumberOfParts() calls getTypeLegalizationCost() which splits the vector
539// type until it is legal. This would e.g. return 4 for <6 x i64>, instead of
540// 3.
541static unsigned getNumVectorRegs(Type *Ty) {
542 auto *VTy = cast<FixedVectorType>(Ty);
543 unsigned WideBits = getScalarSizeInBits(Ty) * VTy->getNumElements();
544 assert(WideBits > 0 && "Could not compute size of vector");
545 return ((WideBits % 128U) ? ((WideBits / 128U) + 1) : (WideBits / 128U));
546}
547
548static bool isFoldableRMW(const Instruction *I, Type *Ty) {
550 if (!BI || !BI->hasOneUse())
551 return false;
552
553 unsigned Opcode = BI->getOpcode();
554 unsigned BitWidth = Ty->getScalarSizeInBits();
555
556 switch (Opcode) {
557 case Instruction::And:
558 case Instruction::Or:
559 case Instruction::Xor: {
560 if (BitWidth == 8)
561 break;
562 if (BitWidth != 16 && BitWidth != 32 && BitWidth != 64)
563 return false;
564
565 auto *CI = dyn_cast<ConstantInt>(I->getOperand(1));
566 if (!CI)
567 return false;
568
569 uint64_t Val = CI->getZExtValue();
570 if (Opcode == Instruction::And) {
571 if (BitWidth == 16 && (Val & 0xff00ULL) != 0xff00ULL)
572 return false;
573 if (BitWidth == 32 && (Val & 0xffffff00ULL) != 0xffffff00ULL)
574 return false;
575 if (BitWidth == 64 &&
576 (Val & 0xffffffffffffff00ULL) != 0xffffffffffffff00ULL)
577 return false;
578 } else {
579 if (CI->getValue().getActiveBits() > 8) {
580 return false;
581 }
582 }
583 break;
584 }
585 case Instruction::Add:
586 case Instruction::Sub:
587 if (BitWidth != 32 && BitWidth != 64)
588 return false;
589 break;
590 default:
591 return false;
592 }
593
594 Value *Op0 = BI->getOperand(0), *Op1 = BI->getOperand(1);
595 if (!isa<ConstantInt>(Op0) && !isa<ConstantInt>(Op1))
596 return false;
597
598 Value *V =
599 (Opcode == Instruction::Sub) ? Op0 : (isa<ConstantInt>(Op0) ? Op1 : Op0);
600 if (Opcode == Instruction::Sub && !isa<ConstantInt>(Op1))
601 return false;
602
603 auto *LI = dyn_cast_or_null<LoadInst>(V);
604 // Already checked BI hasOneUse.
605 auto *SI = dyn_cast<StoreInst>(BI->user_back());
606
607 return LI && SI && !LI->isVolatile() && !SI->isVolatile() &&
608 LI->hasOneUse() && LI->getPointerOperand() == SI->getPointerOperand();
609}
610
612 unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind,
614 ArrayRef<const Value *> Args, const Instruction *CtxI) const {
615
616 // TODO: Handle more cost kinds.
618 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info,
619 Op2Info, Args, CtxI);
620 if (CtxI && Ty && !Ty->isVectorTy() && isFoldableRMW(CtxI, Ty))
621 return TTI::TCC_Free;
622 // TODO: return a good value for BB-VECTORIZER that includes the
623 // immediate loads, which we do not want to count for the loop
624 // vectorizer, since they are hopefully hoisted out of the loop. This
625 // would require a new parameter 'InLoop', but not sure if constant
626 // args are common enough to motivate this.
627
628 unsigned ScalarBits = Ty->getScalarSizeInBits();
629
630 // There are thre cases of division and remainder: Dividing with a register
631 // needs a divide instruction. A divisor which is a power of two constant
632 // can be implemented with a sequence of shifts. Any other constant needs a
633 // multiply and shifts.
634 const unsigned DivInstrCost = 20;
635 const unsigned DivMulSeqCost = 10;
636 const unsigned SDivPow2Cost = 4;
637
638 bool SignedDivRem =
639 Opcode == Instruction::SDiv || Opcode == Instruction::SRem;
640 bool UnsignedDivRem =
641 Opcode == Instruction::UDiv || Opcode == Instruction::URem;
642
643 // Check for a constant divisor.
644 bool DivRemConst = false;
645 bool DivRemConstPow2 = false;
646 if ((SignedDivRem || UnsignedDivRem) && Args.size() == 2) {
647 if (const Constant *C = dyn_cast<Constant>(Args[1])) {
648 const ConstantInt *CVal =
649 (C->getType()->isVectorTy()
650 ? dyn_cast_or_null<const ConstantInt>(C->getSplatValue())
652 if (CVal && (CVal->getValue().isPowerOf2() ||
653 CVal->getValue().isNegatedPowerOf2()))
654 DivRemConstPow2 = true;
655 else
656 DivRemConst = true;
657 }
658 }
659
660 if (!Ty->isVectorTy()) {
661 // These FP operations are supported with a dedicated instruction for
662 // float, double and fp128 (base implementation assumes float generally
663 // costs 2).
664 if (Opcode == Instruction::FAdd || Opcode == Instruction::FSub ||
665 Opcode == Instruction::FMul || Opcode == Instruction::FDiv)
666 return 1;
667
668 // There is no native support for FRem.
669 if (Opcode == Instruction::FRem)
670 return LIBCALL_COST;
671
672 // Give discount for some combined logical operations if supported.
673 if (Args.size() == 2) {
674 if (Opcode == Instruction::Xor) {
675 for (const Value *A : Args) {
676 if (const Instruction *I = dyn_cast<Instruction>(A))
677 if (I->hasOneUse() &&
678 (I->getOpcode() == Instruction::Or ||
679 I->getOpcode() == Instruction::And ||
680 I->getOpcode() == Instruction::Xor))
681 if ((ScalarBits <= 64 && ST->hasMiscellaneousExtensions3()) ||
682 (isInt128InVR(Ty) &&
683 (I->getOpcode() == Instruction::Or || ST->hasVectorEnhancements1())))
684 return 0;
685 }
686 }
687 else if (Opcode == Instruction::And || Opcode == Instruction::Or) {
688 for (const Value *A : Args) {
689 if (const Instruction *I = dyn_cast<Instruction>(A))
690 if ((I->hasOneUse() && I->getOpcode() == Instruction::Xor) &&
691 ((ScalarBits <= 64 && ST->hasMiscellaneousExtensions3()) ||
692 (isInt128InVR(Ty) &&
693 (Opcode == Instruction::And || ST->hasVectorEnhancements1()))))
694 return 0;
695 }
696 }
697 }
698
699 // Or requires one instruction, although it has custom handling for i64.
700 if (Opcode == Instruction::Or)
701 return 1;
702
703 if (Opcode == Instruction::Xor && ScalarBits == 1) {
704 if (ST->hasLoadStoreOnCond2())
705 return 5; // 2 * (li 0; loc 1); xor
706 return 7; // 2 * ipm sequences ; xor ; shift ; compare
707 }
708
709 if (DivRemConstPow2)
710 return (SignedDivRem ? SDivPow2Cost : 1);
711 if (DivRemConst)
712 return DivMulSeqCost;
713 if (SignedDivRem || UnsignedDivRem)
714 return DivInstrCost;
715 }
716 else if (ST->hasVector()) {
717 auto *VTy = cast<FixedVectorType>(Ty);
718 unsigned VF = VTy->getNumElements();
719 unsigned NumVectors = getNumVectorRegs(Ty);
720
721 // These vector operations are custom handled, but are still supported
722 // with one instruction per vector, regardless of element size.
723 if (Opcode == Instruction::Shl || Opcode == Instruction::LShr ||
724 Opcode == Instruction::AShr) {
725 return NumVectors;
726 }
727
728 if (DivRemConstPow2)
729 return (NumVectors * (SignedDivRem ? SDivPow2Cost : 1));
730 if (DivRemConst) {
731 SmallVector<Type *> Tys(Args.size(), Ty);
732 return VF * DivMulSeqCost +
734 }
735 if (SignedDivRem || UnsignedDivRem) {
736 if (ST->hasVectorEnhancements3() && ScalarBits >= 32)
737 return NumVectors * DivInstrCost;
738 else if (VF > 4)
739 // Temporary hack: disable high vectorization factors with integer
740 // division/remainder, which will get scalarized and handled with
741 // GR128 registers. The mischeduler is not clever enough to avoid
742 // spilling yet.
743 return 1000;
744 }
745
746 // These FP operations are supported with a single vector instruction for
747 // double (base implementation assumes float generally costs 2). For
748 // FP128, the scalar cost is 1, and there is no overhead since the values
749 // are already in scalar registers.
750 if (Opcode == Instruction::FAdd || Opcode == Instruction::FSub ||
751 Opcode == Instruction::FMul || Opcode == Instruction::FDiv) {
752 switch (ScalarBits) {
753 case 32: {
754 // The vector enhancements facility 1 provides v4f32 instructions.
755 if (ST->hasVectorEnhancements1())
756 return NumVectors;
757 // Return the cost of multiple scalar invocation plus the cost of
758 // inserting and extracting the values.
759 InstructionCost ScalarCost =
760 getArithmeticInstrCost(Opcode, Ty->getScalarType(), CostKind);
761 SmallVector<Type *> Tys(Args.size(), Ty);
763 (VF * ScalarCost) +
765 // FIXME: VF 2 for these FP operations are currently just as
766 // expensive as for VF 4.
767 if (VF == 2)
768 Cost *= 2;
769 return Cost;
770 }
771 case 64:
772 case 128:
773 return NumVectors;
774 default:
775 break;
776 }
777 }
778
779 // There is no native support for FRem.
780 if (Opcode == Instruction::FRem) {
781 SmallVector<Type *> Tys(Args.size(), Ty);
783 (VF * LIBCALL_COST) +
785 // FIXME: VF 2 for float is currently just as expensive as for VF 4.
786 if (VF == 2 && ScalarBits == 32)
787 Cost *= 2;
788 return Cost;
789 }
790 }
791
792 // Fallback to the default implementation.
793 return BaseT::getArithmeticInstrCost(Opcode, Ty, CostKind, Op1Info, Op2Info,
794 Args, CtxI);
795}
796
798 TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy,
800 VectorType *SubTp, ArrayRef<const Value *> Args, const Instruction *CtxI,
801 TTI::VectorInstrContext VIC) const {
802 Kind = improveShuffleKindFromMask(Kind, Mask, SrcTy, Index, SubTp);
803 if (ST->hasVector()) {
804 unsigned NumVectors = getNumVectorRegs(SrcTy);
805
806 // TODO: Since fp32 is expanded, the shuffle cost should always be 0.
807
808 // FP128 values are always in scalar registers, so there is no work
809 // involved with a shuffle, except for broadcast. In that case register
810 // moves are done with a single instruction per element.
811 if (SrcTy->getScalarType()->isFP128Ty())
812 return (Kind == TargetTransformInfo::SK_Broadcast ? NumVectors - 1 : 0);
813
814 switch (Kind) {
816 // ExtractSubvector Index indicates start offset.
817
818 // Extracting a subvector from first index is a noop.
819 return (Index == 0 ? 0 : NumVectors);
820
822 // Loop vectorizer calls here to figure out the extra cost of
823 // broadcasting a loaded value to all elements of a vector. Since vlrep
824 // loads and replicates with a single instruction, adjust the returned
825 // value.
826 return NumVectors - 1;
827
828 default:
829
830 // SystemZ supports single instruction permutation / replication.
831 return NumVectors;
832 }
833 }
834
835 return BaseT::getShuffleCost(Kind, DstTy, SrcTy, CostKind, Mask, Index,
836 SubTp);
837}
838
839// Return the log2 difference of the element sizes of the two vector types.
840static unsigned getElSizeLog2Diff(Type *Ty0, Type *Ty1) {
841 unsigned Bits0 = getScalarSizeInBits(Ty0);
842 unsigned Bits1 = getScalarSizeInBits(Ty1);
843
844 if (Bits1 > Bits0)
845 return (Log2_32(Bits1) - Log2_32(Bits0));
846
847 return (Log2_32(Bits0) - Log2_32(Bits1));
848}
849
850// Return the number of instructions needed to truncate SrcTy to DstTy.
851unsigned SystemZTTIImpl::getVectorTruncCost(Type *SrcTy, Type *DstTy) const {
852 assert (SrcTy->isVectorTy() && DstTy->isVectorTy());
854 "Packing must reduce size of vector type.");
855 assert(cast<FixedVectorType>(SrcTy)->getNumElements() ==
856 cast<FixedVectorType>(DstTy)->getNumElements() &&
857 "Packing should not change number of elements.");
858
859 // TODO: Since fp32 is expanded, the extract cost should always be 0.
860
861 unsigned NumParts = getNumVectorRegs(SrcTy);
862 if (NumParts <= 2)
863 // Up to 2 vector registers can be truncated efficiently with pack or
864 // permute. The latter requires an immediate mask to be loaded, which
865 // typically gets hoisted out of a loop. TODO: return a good value for
866 // BB-VECTORIZER that includes the immediate loads, which we do not want
867 // to count for the loop vectorizer.
868 return 1;
869
870 unsigned Cost = 0;
871 unsigned Log2Diff = getElSizeLog2Diff(SrcTy, DstTy);
872 unsigned VF = cast<FixedVectorType>(SrcTy)->getNumElements();
873 for (unsigned P = 0; P < Log2Diff; ++P) {
874 if (NumParts > 1)
875 NumParts /= 2;
876 Cost += NumParts;
877 }
878
879 // Currently, a general mix of permutes and pack instructions is output by
880 // isel, which follow the cost computation above except for this case which
881 // is one instruction less:
882 if (VF == 8 && SrcTy->getScalarSizeInBits() == 64 &&
883 DstTy->getScalarSizeInBits() == 8)
884 Cost--;
885
886 return Cost;
887}
888
889// Return the cost of converting a vector bitmask produced by a compare
890// (SrcTy), to the type of the select or extend instruction (DstTy).
892 Type *DstTy) const {
893 assert (SrcTy->isVectorTy() && DstTy->isVectorTy() &&
894 "Should only be called with vector types.");
895
896 unsigned PackCost = 0;
897 unsigned SrcScalarBits = getScalarSizeInBits(SrcTy);
898 unsigned DstScalarBits = getScalarSizeInBits(DstTy);
899 unsigned Log2Diff = getElSizeLog2Diff(SrcTy, DstTy);
900 if (SrcScalarBits > DstScalarBits)
901 // The bitmask will be truncated.
902 PackCost = getVectorTruncCost(SrcTy, DstTy);
903 else if (SrcScalarBits < DstScalarBits) {
904 unsigned DstNumParts = getNumVectorRegs(DstTy);
905 // Each vector select needs its part of the bitmask unpacked.
906 PackCost = Log2Diff * DstNumParts;
907 // Extra cost for moving part of mask before unpacking.
908 PackCost += DstNumParts - 1;
909 }
910
911 return PackCost;
912}
913
914// Return the type of the compared operands. This is needed to compute the
915// cost for a Select / ZExt or SExt instruction.
916static Type *getCmpOpsType(const Instruction *I, unsigned VF = 1) {
917 Type *OpTy = nullptr;
918 if (CmpInst *CI = dyn_cast<CmpInst>(I->getOperand(0)))
919 OpTy = CI->getOperand(0)->getType();
920 else if (Instruction *LogicI = dyn_cast<Instruction>(I->getOperand(0)))
921 if (LogicI->getNumOperands() == 2)
922 if (CmpInst *CI0 = dyn_cast<CmpInst>(LogicI->getOperand(0)))
923 if (isa<CmpInst>(LogicI->getOperand(1)))
924 OpTy = CI0->getOperand(0)->getType();
925
926 if (OpTy != nullptr) {
927 if (VF == 1) {
928 assert (!OpTy->isVectorTy() && "Expected scalar type");
929 return OpTy;
930 }
931 // Return the potentially vectorized type based on 'I' and 'VF'. 'I' may
932 // be either scalar or already vectorized with a same or lesser VF.
933 Type *ElTy = OpTy->getScalarType();
934 return FixedVectorType::get(ElTy, VF);
935 }
936
937 return nullptr;
938}
939
940// Get the cost of converting a boolean vector to a vector with same width
941// and element size as Dst, plus the cost of zero extending if needed.
942unsigned
944 const Instruction *I) const {
945 auto *DstVTy = cast<FixedVectorType>(Dst);
946 unsigned VF = DstVTy->getNumElements();
947 unsigned Cost = 0;
948 // If we know what the widths of the compared operands, get any cost of
949 // converting it to match Dst. Otherwise assume same widths.
950 Type *CmpOpTy = ((I != nullptr) ? getCmpOpsType(I, VF) : nullptr);
951 if (CmpOpTy != nullptr)
952 Cost = getVectorBitmaskConversionCost(CmpOpTy, Dst);
953 if (Opcode == Instruction::ZExt || Opcode == Instruction::UIToFP)
954 // One 'vn' per dst vector with an immediate mask.
955 Cost += getNumVectorRegs(Dst);
956 return Cost;
957}
958
960 Type *Src,
963 const Instruction *I) const {
964 // FIXME: Can the logic below also be used for these cost kinds?
966 auto BaseCost = BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
967 return BaseCost == 0 ? BaseCost : 1;
968 }
969
970 unsigned DstScalarBits = Dst->getScalarSizeInBits();
971 unsigned SrcScalarBits = Src->getScalarSizeInBits();
972
973 if (!Src->isVectorTy()) {
974 if (Dst->isVectorTy())
975 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
976
977 if (Opcode == Instruction::SIToFP || Opcode == Instruction::UIToFP) {
978 if (Src->isIntegerTy(128))
979 return LIBCALL_COST;
980 if (SrcScalarBits >= 32 ||
981 (I != nullptr && isa<LoadInst>(I->getOperand(0))))
982 return 1;
983 return SrcScalarBits > 1 ? 2 /*i8/i16 extend*/ : 5 /*branch seq.*/;
984 }
985
986 if ((Opcode == Instruction::FPToSI || Opcode == Instruction::FPToUI) &&
987 Dst->isIntegerTy(128))
988 return LIBCALL_COST;
989
990 if ((Opcode == Instruction::ZExt || Opcode == Instruction::SExt)) {
991 if (Src->isIntegerTy(1)) {
992 if (DstScalarBits == 128) {
993 if (Opcode == Instruction::SExt && ST->hasVectorEnhancements3())
994 return 0;/*VCEQQ*/
995 return 5 /*branch seq.*/;
996 }
997
998 if (ST->hasLoadStoreOnCond2())
999 return 2; // li 0; loc 1
1000
1001 // This should be extension of a compare i1 result, which is done with
1002 // ipm and a varying sequence of instructions.
1003 unsigned Cost = 0;
1004 if (Opcode == Instruction::SExt)
1005 Cost = (DstScalarBits < 64 ? 3 : 4);
1006 if (Opcode == Instruction::ZExt)
1007 Cost = 3;
1008 Type *CmpOpTy = ((I != nullptr) ? getCmpOpsType(I) : nullptr);
1009 if (CmpOpTy != nullptr && CmpOpTy->isFloatingPointTy())
1010 // If operands of an fp-type was compared, this costs +1.
1011 Cost++;
1012 return Cost;
1013 }
1014 else if (isInt128InVR(Dst)) {
1015 // Extensions from GPR to i128 (in VR) typically costs two instructions,
1016 // but a zero-extending load would be just one extra instruction.
1017 if (Opcode == Instruction::ZExt && I != nullptr)
1018 if (LoadInst *Ld = dyn_cast<LoadInst>(I->getOperand(0)))
1019 if (Ld->hasOneUse())
1020 return 1;
1021 return 2;
1022 }
1023 }
1024
1025 if (Opcode == Instruction::Trunc && isInt128InVR(Src) && I != nullptr) {
1026 if (LoadInst *Ld = dyn_cast<LoadInst>(I->getOperand(0)))
1027 if (Ld->hasOneUse())
1028 return 0; // Will be converted to GPR load.
1029 bool OnlyTruncatingStores = true;
1030 for (const User *U : I->users())
1031 if (!isa<StoreInst>(U)) {
1032 OnlyTruncatingStores = false;
1033 break;
1034 }
1035 if (OnlyTruncatingStores)
1036 return 0;
1037 return 2; // Vector element extraction.
1038 }
1039 }
1040 else if (ST->hasVector()) {
1041 // Vector to scalar cast.
1042 auto *SrcVecTy = cast<FixedVectorType>(Src);
1043 auto *DstVecTy = dyn_cast<FixedVectorType>(Dst);
1044 if (!DstVecTy) {
1045 // TODO: tune vector-to-scalar cast.
1046 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1047 }
1048 unsigned VF = SrcVecTy->getNumElements();
1049 unsigned NumDstVectors = getNumVectorRegs(Dst);
1050 unsigned NumSrcVectors = getNumVectorRegs(Src);
1051
1052 if (Opcode == Instruction::Trunc) {
1053 if (Src->getScalarSizeInBits() == Dst->getScalarSizeInBits())
1054 return 0; // Check for NOOP conversions.
1055 return getVectorTruncCost(Src, Dst);
1056 }
1057
1058 if (Opcode == Instruction::ZExt || Opcode == Instruction::SExt) {
1059 if (SrcScalarBits >= 8) {
1060 // ZExt will use either a single unpack or a vector permute.
1061 if (Opcode == Instruction::ZExt)
1062 return NumDstVectors;
1063
1064 // SExt will be handled with one unpack per doubling of width.
1065 unsigned NumUnpacks = getElSizeLog2Diff(Src, Dst);
1066
1067 // For types that spans multiple vector registers, some additional
1068 // instructions are used to setup the unpacking.
1069 unsigned NumSrcVectorOps =
1070 (NumUnpacks > 1 ? (NumDstVectors - NumSrcVectors)
1071 : (NumDstVectors / 2));
1072
1073 return (NumUnpacks * NumDstVectors) + NumSrcVectorOps;
1074 }
1075 else if (SrcScalarBits == 1)
1076 return getBoolVecToIntConversionCost(Opcode, Dst, I);
1077 }
1078
1079 if (Opcode == Instruction::SIToFP || Opcode == Instruction::UIToFP ||
1080 Opcode == Instruction::FPToSI || Opcode == Instruction::FPToUI) {
1081 // TODO: Fix base implementation which could simplify things a bit here
1082 // (seems to miss on differentiating on scalar/vector types).
1083
1084 // Only 64 bit vector conversions are natively supported before z15.
1085 if (DstScalarBits == 64 || ST->hasVectorEnhancements2()) {
1086 if (SrcScalarBits == DstScalarBits)
1087 return NumDstVectors;
1088
1089 if (SrcScalarBits == 1)
1090 return getBoolVecToIntConversionCost(Opcode, Dst, I) + NumDstVectors;
1091 }
1092
1093 // Return the cost of multiple scalar invocation plus the cost of
1094 // inserting and extracting the values. Base implementation does not
1095 // realize float->int gets scalarized.
1096 InstructionCost ScalarCost = getCastInstrCost(
1097 Opcode, Dst->getScalarType(), Src->getScalarType(), CCH, CostKind);
1098 InstructionCost TotCost = VF * ScalarCost;
1099 bool NeedsInserts = true, NeedsExtracts = true;
1100 // FP128 registers do not get inserted or extracted.
1101 if (DstScalarBits == 128 &&
1102 (Opcode == Instruction::SIToFP || Opcode == Instruction::UIToFP))
1103 NeedsInserts = false;
1104 if (SrcScalarBits == 128 &&
1105 (Opcode == Instruction::FPToSI || Opcode == Instruction::FPToUI))
1106 NeedsExtracts = false;
1107
1108 TotCost += BaseT::getScalarizationOverhead(SrcVecTy, /*Insert*/ false,
1109 NeedsExtracts, CostKind);
1110 TotCost += BaseT::getScalarizationOverhead(DstVecTy, NeedsInserts,
1111 /*Extract*/ false, CostKind);
1112
1113 // FIXME: VF 2 for float<->i32 is currently just as expensive as for VF 4.
1114 if (VF == 2 && SrcScalarBits == 32 && DstScalarBits == 32)
1115 TotCost *= 2;
1116
1117 return TotCost;
1118 }
1119
1120 if (Opcode == Instruction::FPTrunc) {
1121 if (SrcScalarBits == 128) // fp128 -> double/float + inserts of elements.
1122 return VF /*ldxbr/lexbr*/ +
1123 BaseT::getScalarizationOverhead(DstVecTy, /*Insert*/ true,
1124 /*Extract*/ false, CostKind);
1125 else // double -> float
1126 return VF / 2 /*vledb*/ + std::max(1U, VF / 4 /*vperm*/);
1127 }
1128
1129 if (Opcode == Instruction::FPExt) {
1130 if (SrcScalarBits == 32 && DstScalarBits == 64) {
1131 // float -> double is very rare and currently unoptimized. Instead of
1132 // using vldeb, which can do two at a time, all conversions are
1133 // scalarized.
1134 return VF * 2;
1135 }
1136 // -> fp128. VF * lxdb/lxeb + extraction of elements.
1137 return VF + BaseT::getScalarizationOverhead(SrcVecTy, /*Insert*/ false,
1138 /*Extract*/ true, CostKind);
1139 }
1140 }
1141
1142 return BaseT::getCastInstrCost(Opcode, Dst, Src, CCH, CostKind, I);
1143}
1144
1145// Scalar i8 / i16 operations will typically be made after first extending
1146// the operands to i32.
1147static unsigned getOperandsExtensionCost(const Instruction *I) {
1148 unsigned ExtCost = 0;
1149 for (Value *Op : I->operands())
1150 // A load of i8 or i16 sign/zero extends to i32.
1152 ExtCost++;
1153
1154 return ExtCost;
1155}
1156
1159 const Instruction *I) const {
1161 return Opcode == Instruction::PHI ? TTI::TCC_Free : TTI::TCC_Basic;
1162 // Branches are assumed to be predicted.
1163 return TTI::TCC_Free;
1164}
1165
1167 unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred,
1169 TTI::OperandValueInfo Op2Info, const Instruction *I) const {
1171 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
1172 Op1Info, Op2Info);
1173
1174 if (!ValTy->isVectorTy()) {
1175 switch (Opcode) {
1176 case Instruction::ICmp: {
1177 // A loaded value compared with 0 with multiple users becomes Load and
1178 // Test. The load is then not foldable, so return 0 cost for the ICmp.
1179 unsigned ScalarBits = ValTy->getScalarSizeInBits();
1180 if (I != nullptr && (ScalarBits == 32 || ScalarBits == 64))
1181 if (LoadInst *Ld = dyn_cast<LoadInst>(I->getOperand(0)))
1182 if (const ConstantInt *C = dyn_cast<ConstantInt>(I->getOperand(1)))
1183 if (!Ld->hasOneUse() && Ld->getParent() == I->getParent() &&
1184 C->isZero())
1185 return 0;
1186
1187 unsigned Cost = 1;
1188 if (ValTy->isIntegerTy() && ValTy->getScalarSizeInBits() <= 16)
1189 Cost += (I != nullptr ? getOperandsExtensionCost(I) : 2);
1190 return Cost;
1191 }
1192 case Instruction::Select:
1193 if (ValTy->isFloatingPointTy())
1194 return 4; // No LOC for FP - costs a conditional jump.
1195
1196 // When selecting based on an i128 comparison, LOC / VSEL is possible
1197 // if i128 comparisons are directly supported.
1198 if (I != nullptr)
1199 if (ICmpInst *CI = dyn_cast<ICmpInst>(I->getOperand(0)))
1200 if (CI->getOperand(0)->getType()->isIntegerTy(128))
1201 return ST->hasVectorEnhancements3() ? 1 : 4;
1202
1203 // Load On Condition / Select Register available, except for i128.
1204 return !isInt128InVR(ValTy) ? 1 : 4;
1205 }
1206 }
1207 else if (ST->hasVector()) {
1208 unsigned VF = cast<FixedVectorType>(ValTy)->getNumElements();
1209
1210 // Called with a compare instruction.
1211 if (Opcode == Instruction::ICmp || Opcode == Instruction::FCmp) {
1212 unsigned PredicateExtraCost = 0;
1213 if (I != nullptr) {
1214 // Some predicates cost one or two extra instructions.
1215 switch (cast<CmpInst>(I)->getPredicate()) {
1221 PredicateExtraCost = 1;
1222 break;
1227 PredicateExtraCost = 2;
1228 break;
1229 default:
1230 break;
1231 }
1232 }
1233
1234 // Float is handled with 2*vmr[lh]f + 2*vldeb + vfchdb for each pair of
1235 // floats. FIXME: <2 x float> generates same code as <4 x float>.
1236 unsigned CmpCostPerVector = (ValTy->getScalarType()->isFloatTy() ? 10 : 1);
1237 unsigned NumVecs_cmp = getNumVectorRegs(ValTy);
1238
1239 unsigned Cost = (NumVecs_cmp * (CmpCostPerVector + PredicateExtraCost));
1240 return Cost;
1241 }
1242 else { // Called with a select instruction.
1243 assert (Opcode == Instruction::Select);
1244
1245 // We can figure out the extra cost of packing / unpacking if the
1246 // instruction was passed and the compare instruction is found.
1247 unsigned PackCost = 0;
1248 Type *CmpOpTy = ((I != nullptr) ? getCmpOpsType(I, VF) : nullptr);
1249 if (CmpOpTy != nullptr)
1250 PackCost =
1251 getVectorBitmaskConversionCost(CmpOpTy, ValTy);
1252
1253 return getNumVectorRegs(ValTy) /*vsel*/ + PackCost;
1254 }
1255 }
1256
1257 return BaseT::getCmpSelInstrCost(Opcode, ValTy, CondTy, VecPred, CostKind,
1258 Op1Info, Op2Info);
1259}
1260
1262 unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index,
1263 const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC) const {
1264 if (Opcode == Instruction::InsertElement) {
1265 // Vector Element Load.
1266 if (Op1 != nullptr && isFreeEltLoad(Op1))
1267 return 0;
1268
1269 // vlvgp will insert two grs into a vector register, so count half the
1270 // number of instructions as an estimate when we don't have the full
1271 // picture (as in getScalarizationOverhead()).
1272 if (Val->isIntOrIntVectorTy(64))
1273 return ((Index % 2 == 0) ? 1 : 0);
1274 }
1275
1276 if (Opcode == Instruction::ExtractElement) {
1277 int Cost = ((getScalarSizeInBits(Val) == 1) ? 2 /*+test-under-mask*/ : 1);
1278
1279 // Give a slight penalty for moving out of vector pipeline to FXU unit.
1280 if (Index == 0 && Val->isIntOrIntVectorTy())
1281 Cost += 1;
1282
1283 return Cost;
1284 }
1285
1286 return BaseT::getVectorInstrCost(Opcode, Val, CostKind, Index, Op0, Op1, VIC);
1287}
1288
1289// Check if a load may be folded as a memory operand in its user.
1291 const Instruction *&FoldedValue) const {
1292 if (!Ld->hasOneUse())
1293 return false;
1294 FoldedValue = Ld;
1295 const Instruction *UserI = cast<Instruction>(*Ld->user_begin());
1296 unsigned LoadedBits = getScalarSizeInBits(Ld->getType());
1297 unsigned TruncBits = 0;
1298 unsigned SExtBits = 0;
1299 unsigned ZExtBits = 0;
1300 if (UserI->hasOneUse()) {
1301 unsigned UserBits = UserI->getType()->getScalarSizeInBits();
1302 if (isa<TruncInst>(UserI))
1303 TruncBits = UserBits;
1304 else if (isa<SExtInst>(UserI))
1305 SExtBits = UserBits;
1306 else if (isa<ZExtInst>(UserI))
1307 ZExtBits = UserBits;
1308 }
1309 if (TruncBits || SExtBits || ZExtBits) {
1310 FoldedValue = UserI;
1311 UserI = cast<Instruction>(*UserI->user_begin());
1312 // Load (single use) -> trunc/extend (single use) -> UserI
1313 }
1314 if ((UserI->getOpcode() == Instruction::Sub ||
1315 UserI->getOpcode() == Instruction::SDiv ||
1316 UserI->getOpcode() == Instruction::UDiv) &&
1317 UserI->getOperand(1) != FoldedValue)
1318 return false; // Not commutative, only RHS foldable.
1319 // LoadOrTruncBits holds the number of effectively loaded bits, but 0 if an
1320 // extension was made of the load.
1321 unsigned LoadOrTruncBits =
1322 ((SExtBits || ZExtBits) ? 0 : (TruncBits ? TruncBits : LoadedBits));
1323 switch (UserI->getOpcode()) {
1324 case Instruction::Add: // SE: 16->32, 16/32->64, z14:16->64. ZE: 32->64
1325 case Instruction::Sub:
1326 case Instruction::ICmp:
1327 if (LoadedBits == 32 && ZExtBits == 64)
1328 return true;
1329 [[fallthrough]];
1330 case Instruction::Mul: // SE: 16->32, 32->64, z14:16->64
1331 if (UserI->getOpcode() != Instruction::ICmp) {
1332 if (LoadedBits == 16 &&
1333 (SExtBits == 32 ||
1334 (SExtBits == 64 && ST->hasMiscellaneousExtensions2())))
1335 return true;
1336 if (LoadOrTruncBits == 16)
1337 return true;
1338 }
1339 [[fallthrough]];
1340 case Instruction::SDiv:// SE: 32->64
1341 if (LoadedBits == 32 && SExtBits == 64)
1342 return true;
1343 [[fallthrough]];
1344 case Instruction::UDiv:
1345 case Instruction::And:
1346 case Instruction::Or:
1347 case Instruction::Xor:
1348 // This also makes sense for float operations, but disabled for now due
1349 // to regressions.
1350 // case Instruction::FCmp:
1351 // case Instruction::FAdd:
1352 // case Instruction::FSub:
1353 // case Instruction::FMul:
1354 // case Instruction::FDiv:
1355
1356 // All possible extensions of memory checked above.
1357
1358 // Comparison between memory and immediate.
1359 if (UserI->getOpcode() == Instruction::ICmp)
1360 if (ConstantInt *CI = dyn_cast<ConstantInt>(UserI->getOperand(1)))
1361 if (CI->getValue().isIntN(16))
1362 return true;
1363 return (LoadOrTruncBits == 32 || LoadOrTruncBits == 64);
1364 break;
1365 }
1366 return false;
1367}
1368
1369static bool isBswapIntrinsicCall(const Value *V) {
1370 if (const Instruction *I = dyn_cast<Instruction>(V))
1371 if (auto *CI = dyn_cast<CallInst>(I))
1372 if (auto *F = CI->getCalledFunction())
1373 if (F->getIntrinsicID() == Intrinsic::bswap)
1374 return true;
1375 return false;
1376}
1377
1379 Align Alignment,
1380 unsigned AddressSpace,
1382 TTI::OperandValueInfo OpInfo,
1383 const Instruction *I) const {
1384 assert(!Src->isVoidTy() && "Invalid type");
1385
1386 // FIXME: Load latency isn't handled here
1387 if (Opcode == Instruction::Load && CostKind == TTI::TCK_Latency)
1388 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
1389 CostKind, OpInfo, I);
1390
1391 // TODO: Handle other cost kinds.
1393 return 1;
1394
1395 if (I && Opcode == Instruction::Store && !Src->isVectorTy()) {
1396 if (isFoldableRMW(dyn_cast<Instruction>(I->getOperand(0)), Src))
1397 return TTI::TCC_Free;
1398 }
1399
1400 if (!Src->isVectorTy() && Opcode == Instruction::Load && I != nullptr) {
1401 // Store the load or its truncated or extended value in FoldedValue.
1402 const Instruction *FoldedValue = nullptr;
1403 if (isFoldableLoad(cast<LoadInst>(I), FoldedValue)) {
1404 const Instruction *UserI = cast<Instruction>(*FoldedValue->user_begin());
1405 assert (UserI->getNumOperands() == 2 && "Expected a binop.");
1406
1407 // UserI can't fold two loads, so in that case return 0 cost only
1408 // half of the time.
1409 for (unsigned i = 0; i < 2; ++i) {
1410 if (UserI->getOperand(i) == FoldedValue)
1411 continue;
1412
1413 if (Instruction *OtherOp = dyn_cast<Instruction>(UserI->getOperand(i))){
1414 LoadInst *OtherLoad = dyn_cast<LoadInst>(OtherOp);
1415 if (!OtherLoad &&
1416 (isa<TruncInst>(OtherOp) || isa<SExtInst>(OtherOp) ||
1417 isa<ZExtInst>(OtherOp)))
1418 OtherLoad = dyn_cast<LoadInst>(OtherOp->getOperand(0));
1419 if (OtherLoad && isFoldableLoad(OtherLoad, FoldedValue/*dummy*/))
1420 return i == 0; // Both operands foldable.
1421 }
1422 }
1423
1424 return 0; // Only I is foldable in user.
1425 }
1426 }
1427
1428 // Type legalization (via getNumberOfParts) can't handle structs
1429 if (TLI->getValueType(DL, Src, true) == MVT::Other)
1430 return BaseT::getMemoryOpCost(Opcode, Src, Alignment, AddressSpace,
1431 CostKind);
1432
1433 // FP128 is a legal type but kept in a register pair on older CPUs.
1434 if (Src->isFP128Ty() && !ST->hasVectorEnhancements1())
1435 return 2;
1436
1437 unsigned NumOps =
1438 (Src->isVectorTy() ? getNumVectorRegs(Src) : getNumberOfParts(Src));
1439
1440 // Store/Load reversed saves one instruction.
1441 if (((!Src->isVectorTy() && NumOps == 1) || ST->hasVectorEnhancements2()) &&
1442 I != nullptr) {
1443 if (Opcode == Instruction::Load && I->hasOneUse()) {
1444 const Instruction *LdUser = cast<Instruction>(*I->user_begin());
1445 // In case of load -> bswap -> store, return normal cost for the load.
1446 if (isBswapIntrinsicCall(LdUser) &&
1447 (!LdUser->hasOneUse() || !isa<StoreInst>(*LdUser->user_begin())))
1448 return 0;
1449 }
1450 else if (const StoreInst *SI = dyn_cast<StoreInst>(I)) {
1451 const Value *StoredVal = SI->getValueOperand();
1452 if (StoredVal->hasOneUse() && isBswapIntrinsicCall(StoredVal))
1453 return 0;
1454 }
1455 }
1456
1457 return NumOps;
1458}
1459
1460// The generic implementation of getInterleavedMemoryOpCost() is based on
1461// adding costs of the memory operations plus all the extracts and inserts
1462// needed for using / defining the vector operands. The SystemZ version does
1463// roughly the same but bases the computations on vector permutations
1464// instead.
1466 unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef<unsigned> Indices,
1467 Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind,
1468 bool UseMaskForCond, bool UseMaskForGaps) const {
1469 if (UseMaskForCond || UseMaskForGaps)
1470 return BaseT::getInterleavedMemoryOpCost(Opcode, VecTy, Factor, Indices,
1471 Alignment, AddressSpace, CostKind,
1472 UseMaskForCond, UseMaskForGaps);
1473 assert(isa<VectorType>(VecTy) &&
1474 "Expect a vector type for interleaved memory op");
1475
1476 unsigned NumElts = cast<FixedVectorType>(VecTy)->getNumElements();
1477 assert(Factor > 1 && NumElts % Factor == 0 && "Invalid interleave factor");
1478 unsigned VF = NumElts / Factor;
1479 unsigned NumEltsPerVecReg = (128U / getScalarSizeInBits(VecTy));
1480 unsigned NumVectorMemOps = getNumVectorRegs(VecTy);
1481 unsigned NumPermutes = 0;
1482
1483 if (Opcode == Instruction::Load) {
1484 // Loading interleave groups may have gaps, which may mean fewer
1485 // loads. Find out how many vectors will be loaded in total, and in how
1486 // many of them each value will be in.
1487 BitVector UsedInsts(NumVectorMemOps, false);
1488 std::vector<BitVector> ValueVecs(Factor, BitVector(NumVectorMemOps, false));
1489 for (unsigned Index : Indices)
1490 for (unsigned Elt = 0; Elt < VF; ++Elt) {
1491 unsigned Vec = (Index + Elt * Factor) / NumEltsPerVecReg;
1492 UsedInsts.set(Vec);
1493 ValueVecs[Index].set(Vec);
1494 }
1495 NumVectorMemOps = UsedInsts.count();
1496
1497 for (unsigned Index : Indices) {
1498 // Estimate that each loaded source vector containing this Index
1499 // requires one operation, except that vperm can handle two input
1500 // registers first time for each dst vector.
1501 unsigned NumSrcVecs = ValueVecs[Index].count();
1502 unsigned NumDstVecs = divideCeil(VF * getScalarSizeInBits(VecTy), 128U);
1503 assert (NumSrcVecs >= NumDstVecs && "Expected at least as many sources");
1504 NumPermutes += std::max(1U, NumSrcVecs - NumDstVecs);
1505 }
1506 } else {
1507 // Estimate the permutes for each stored vector as the smaller of the
1508 // number of elements and the number of source vectors. Subtract one per
1509 // dst vector for vperm (S.A.).
1510 unsigned NumSrcVecs = std::min(NumEltsPerVecReg, Factor);
1511 unsigned NumDstVecs = NumVectorMemOps;
1512 NumPermutes += (NumDstVecs * NumSrcVecs) - NumDstVecs;
1513 }
1514
1515 // Cost of load/store operations and the permutations needed.
1516 return NumVectorMemOps + NumPermutes;
1517}
1518
1519InstructionCost getIntAddReductionCost(unsigned NumVec, unsigned ScalarBits) {
1520 InstructionCost Cost = 0;
1521 // Binary Tree of N/2 + N/4 + ... operations yields N - 1 operations total.
1522 Cost += NumVec - 1;
1523 // For integer adds, VSUM creates shorter reductions on the final vector.
1524 Cost += (ScalarBits < 32) ? 3 : 2;
1525 return Cost;
1526}
1527
1528InstructionCost getFastReductionCost(unsigned NumVec, unsigned NumElems,
1529 unsigned ScalarBits) {
1530 unsigned NumEltsPerVecReg = (SystemZ::VectorBits / ScalarBits);
1531 InstructionCost Cost = 0;
1532 // Binary Tree of N/2 + N/4 + ... operations yields N - 1 operations total.
1533 Cost += NumVec - 1;
1534 // For each shuffle / arithmetic layer, we need 2 instructions, and we need
1535 // log2(Elements in Last Vector) layers.
1536 Cost += 2 * Log2_32_Ceil(std::min(NumElems, NumEltsPerVecReg));
1537 return Cost;
1538}
1539
1540inline bool customCostReductions(unsigned Opcode) {
1541 return Opcode == Instruction::FAdd || Opcode == Instruction::FMul ||
1542 Opcode == Instruction::Add || Opcode == Instruction::Mul;
1543}
1544
1547 std::optional<FastMathFlags> FMF,
1549 unsigned ScalarBits = Ty->getScalarSizeInBits();
1550 // The following is only for subtargets with vector math, non-ordered
1551 // reductions, and reasonable scalar sizes for int and fp add/mul.
1552 if (customCostReductions(Opcode) && ST->hasVector() &&
1554 ScalarBits <= SystemZ::VectorBits) {
1555 unsigned NumVectors = getNumVectorRegs(Ty);
1556 unsigned NumElems = ((FixedVectorType *)Ty)->getNumElements();
1557 // Integer Add is using custom code gen, that needs to be accounted for.
1558 if (Opcode == Instruction::Add)
1559 return getIntAddReductionCost(NumVectors, ScalarBits);
1560 // The base cost is the same across all other arithmetic instructions
1562 getFastReductionCost(NumVectors, NumElems, ScalarBits);
1563 // But we need to account for the final op involving the scalar operand.
1564 if ((Opcode == Instruction::FAdd) || (Opcode == Instruction::FMul))
1565 Cost += 1;
1566 return Cost;
1567 }
1568 // otherwise, fall back to the standard implementation
1569 return BaseT::getArithmeticReductionCost(Opcode, Ty, FMF, CostKind);
1570}
1571
1574 FastMathFlags FMF,
1576 // Return custom costs only on subtargets with vector enhancements.
1577 if (ST->hasVectorEnhancements1()) {
1578 unsigned NumVectors = getNumVectorRegs(Ty);
1579 unsigned NumElems = ((FixedVectorType *)Ty)->getNumElements();
1580 unsigned ScalarBits = Ty->getScalarSizeInBits();
1582 // Binary Tree of N/2 + N/4 + ... operations yields N - 1 operations total.
1583 Cost += NumVectors - 1;
1584 // For the final vector, we need shuffle + min/max operations, and
1585 // we need #Elements - 1 of them.
1586 Cost += 2 * (std::min(NumElems, SystemZ::VectorBits / ScalarBits) - 1);
1587 return Cost;
1588 }
1589 // For other targets, fall back to the standard implementation
1590 return BaseT::getMinMaxReductionCost(IID, Ty, FMF, CostKind);
1591}
1592
1593static int
1595 const SmallVectorImpl<Type *> &ParamTys) {
1596 if (RetTy->isVectorTy() && ID == Intrinsic::bswap)
1597 return getNumVectorRegs(RetTy); // VPERM
1598
1599 return -1;
1600}
1601
1611
1613 // Always expand on Subtargets without vector instructions.
1614 if (!ST->hasVector())
1615 return true;
1616
1617 // Whether or not to expand is a per-intrinsic decision.
1618 switch (II->getIntrinsicID()) {
1619 default:
1620 return true;
1621 // Do not expand vector.reduce.add...
1622 case Intrinsic::vector_reduce_add:
1623 auto *VType = cast<FixedVectorType>(II->getOperand(0)->getType());
1624 // ...unless the scalar size is i64 or larger,
1625 // or the operand vector is not full, since the
1626 // performance benefit is dubious in those cases.
1627 return VType->getScalarSizeInBits() >= 64 ||
1628 VType->getPrimitiveSizeInBits() < SystemZ::VectorBits;
1629 }
1630}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
Expand Atomic instructions
This file provides a helper that implements much of the TTI interface in terms of the target-independ...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static unsigned InstrCount
Hexagon Common GEP
const HexagonInstrInfo * TII
This file defines an InstructionCost class that is used when calculating the cost of an instruction,...
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
uint64_t IntrinsicInst * II
#define P(N)
#define LLVM_DEBUG(...)
Definition Debug.h:119
bool customCostReductions(unsigned Opcode)
static unsigned getElSizeLog2Diff(Type *Ty0, Type *Ty1)
static bool isBswapIntrinsicCall(const Value *V)
InstructionCost getIntAddReductionCost(unsigned NumVec, unsigned ScalarBits)
static void countNumMemAccesses(const Value *Ptr, unsigned &NumStores, unsigned &NumLoads, const Function *F)
static unsigned getOperandsExtensionCost(const Instruction *I)
static Type * getCmpOpsType(const Instruction *I, unsigned VF=1)
static unsigned getScalarSizeInBits(Type *Ty)
static bool isFoldableRMW(const Instruction *I, Type *Ty)
static bool isFreeEltLoad(const Value *Op)
InstructionCost getFastReductionCost(unsigned NumVec, unsigned NumElems, unsigned ScalarBits)
static int getVectorIntrinsicInstrCost(Intrinsic::ID ID, Type *RetTy, const SmallVectorImpl< Type * > &ParamTys)
static bool isUsedAsMemCpySource(const Value *V, bool &OtherUse)
static unsigned getNumVectorRegs(Type *Ty)
This file describes how to lower LLVM code to machine code.
This pass exposes codegen information to IR-level passes.
Class for arbitrary precision integers.
Definition APInt.h:78
bool isNegatedPowerOf2() const
Check if this APInt's negated value is a power of two greater than zero.
Definition APInt.h:445
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
Definition APInt.h:436
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
unsigned getNumberOfParts(Type *Tp) const override
TTI::ShuffleKind improveShuffleKindFromMask(TTI::ShuffleKind Kind, ArrayRef< int > Mask, VectorType *SrcTy, int &Index, VectorType *&SubTy) const
InstructionCost getScalarizationOverhead(VectorType *InTy, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Opd1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Opd2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
size_type count() const
Returns the number of bits which are set.
Definition BitVector.h:181
BitVector & set()
Set all bits in the bitvector.
Definition BitVector.h:366
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
Value * getArgOperand(unsigned i) const
This class is the base class for the comparison instructions.
Definition InstrTypes.h:728
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_SGE
signed greater or equal
Definition InstrTypes.h:768
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
This is the shared class of boolean and integer constants.
Definition Constants.h:87
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
This is an important base class in LLVM.
Definition Constant.h:43
constexpr bool isVector() const
One or more elements.
Definition TypeSize.h:320
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
Class to represent fixed width SIMD vectors.
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
This instruction compares its operands according to the predicate given to the constructor.
CostType getValue() const
This function is intended to be used as sparingly as possible, since the class provides the full rang...
user_iterator user_begin()
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
const SmallVectorImpl< Type * > & getArgTypes() const
A wrapper class for inspecting calls to intrinsic functions.
An instruction for reading from memory.
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
This class wraps the llvm.memcpy intrinsic.
The optimization diagnostic interface.
The main scalar evolution driver.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
InstructionCost getScalarizationOverhead(VectorType *Ty, const APInt &DemandedElts, bool Insert, bool Extract, TTI::TargetCostKind CostKind, bool ForPoisonSrc=true, ArrayRef< Value * > VL={}, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
Estimate the overhead of scalarizing an instruction.
bool isFoldableLoad(const LoadInst *Ld, const Instruction *&FoldedValue) const
bool isLSRCostLess(const TargetTransformInfo::LSRCost &C1, const TargetTransformInfo::LSRCost &C2) const override
InstructionCost getMinMaxReductionCost(Intrinsic::ID IID, VectorType *Ty, FastMathFlags FMF, TTI::TargetCostKind CostKind) const override
Try to calculate op costs for min/max reduction operations.
InstructionCost getCFInstrCost(unsigned Opcode, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
unsigned getNumberOfRegisters(unsigned ClassID) const override
void getPeelingPreferences(Loop *L, ScalarEvolution &SE, TTI::PeelingPreferences &PP) const override
InstructionCost getShuffleCost(TTI::ShuffleKind Kind, VectorType *DstTy, VectorType *SrcTy, TTI::TargetCostKind CostKind, ArrayRef< int > Mask, int Index, VectorType *SubTp, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntrinsicInstrCost(const IntrinsicCostAttributes &ICA, TTI::TargetCostKind CostKind) const override
Get intrinsic cost based on arguments.
unsigned getMinPrefetchStride(unsigned NumMemAccesses, unsigned NumStridedMemAccesses, unsigned NumPrefetches, bool HasCall) const override
InstructionCost getVectorInstrCost(unsigned Opcode, Type *Val, TTI::TargetCostKind CostKind, unsigned Index, const Value *Op0, const Value *Op1, TTI::VectorInstrContext VIC=TTI::VectorInstrContext::None) const override
InstructionCost getIntImmCostIntrin(Intrinsic::ID IID, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getVectorBitmaskConversionCost(Type *SrcTy, Type *DstTy) const
unsigned getBoolVecToIntConversionCost(unsigned Opcode, Type *Dst, const Instruction *I) const
InstructionCost getIntImmCostInst(unsigned Opcode, unsigned Idx, const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind, Instruction *Inst=nullptr) const override
bool shouldExpandReduction(const IntrinsicInst *II) const override
TTI::PopcntSupportKind getPopcntSupport(unsigned TyWidth) const override
InstructionCost getInterleavedMemoryOpCost(unsigned Opcode, Type *VecTy, unsigned Factor, ArrayRef< unsigned > Indices, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, bool UseMaskForCond=false, bool UseMaskForGaps=false) const override
InstructionCost getArithmeticInstrCost(unsigned Opcode, Type *Ty, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, ArrayRef< const Value * > Args={}, const Instruction *CtxI=nullptr) const override
InstructionCost getArithmeticReductionCost(unsigned Opcode, VectorType *Ty, std::optional< FastMathFlags > FMF, TTI::TargetCostKind CostKind) const override
bool hasDivRemOp(Type *DataType, bool IsSigned) const override
unsigned getVectorTruncCost(Type *SrcTy, Type *DstTy) const
void getUnrollingPreferences(Loop *L, ScalarEvolution &SE, TTI::UnrollingPreferences &UP, OptimizationRemarkEmitter *ORE) const override
unsigned adjustInliningThreshold(const CallBase *CB) const override
InstructionCost getMemoryOpCost(unsigned Opcode, Type *Src, Align Alignment, unsigned AddressSpace, TTI::TargetCostKind CostKind, TTI::OperandValueInfo OpInfo={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
InstructionCost getCmpSelInstrCost(unsigned Opcode, Type *ValTy, Type *CondTy, CmpInst::Predicate VecPred, TTI::TargetCostKind CostKind, TTI::OperandValueInfo Op1Info={TTI::OK_AnyValue, TTI::OP_None}, TTI::OperandValueInfo Op2Info={TTI::OK_AnyValue, TTI::OP_None}, const Instruction *I=nullptr) const override
TypeSize getRegisterBitWidth(TargetTransformInfo::RegisterKind K) const override
InstructionCost getIntImmCost(const APInt &Imm, Type *Ty, TTI::TargetCostKind CostKind) const override
unsigned getMaxInterleaveFactor(ElementCount VF, bool HasUnorderedReductions) const override
InstructionCost getCastInstrCost(unsigned Opcode, Type *Dst, Type *Src, TTI::CastContextHint CCH, TTI::TargetCostKind CostKind, const Instruction *I=nullptr) const override
virtual bool isLoweredToCall(const Function *F) const
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCK_Latency
The latency of instruction.
static bool requiresOrderedReduction(std::optional< FastMathFlags > FMF)
A helper function to determine the type of reduction algorithm used for a given Opcode and set of Fas...
PopcntSupportKind
Flags indicating the kind of support for population count.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
ShuffleKind
The various kinds of shuffle patterns for vector queries.
@ SK_Broadcast
Broadcast element 0 to all other elements.
@ SK_ExtractSubvector
ExtractSubvector Index indicates start offset.
CastContextHint
Represents a hint about the context in which a cast is used.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:258
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
Value * getOperand(unsigned i) const
Definition User.h:207
unsigned getNumOperands() const
Definition User.h:229
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
iterator_range< user_iterator > users()
Definition Value.h:428
Base class of all SIMD vector types.
const ParentTy * getParent() const
Definition ilist_node.h:34
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
const unsigned VectorBits
Definition SystemZ.h:156
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
InstructionCost Cost
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
DWARFExpression::Operation Op
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
unsigned Insns
TODO: Some of these could be merged.
Parameters that control the generic loop unrolling transformation.
bool Force
Apply loop unroll on any kind of loop (mainly to loops that fail runtime unrolling).
unsigned DefaultUnrollRuntimeCount
Default unroll count for loops with run-time trip count.
unsigned FullUnrollMaxCount
Set the maximum unrolling factor for full unrolling.
unsigned PartialThreshold
The cost threshold for the unrolled loop, like Threshold, but used for partial/runtime unrolling (set...
bool Runtime
Allow runtime unrolling (unrolling of loops to expand the size of the loop body even when the number ...
bool Partial
Allow partial unrolling (unrolling of loops to expand the size of the loop body, not only to eliminat...
bool AllowExpensiveTripCount
Allow emitting expensive instructions (such as divisions) when computing the trip count of a loop for...