LLVM 24.0.0git
AMDGPULibCalls.cpp
Go to the documentation of this file.
1//===- AMDGPULibCalls.cpp -------------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file does AMD library function optimizations.
11//
12//===----------------------------------------------------------------------===//
13
14#include "AMDGPU.h"
15#include "AMDGPULibFunc.h"
20#include "llvm/IR/Dominators.h"
21#include "llvm/IR/IRBuilder.h"
22#include "llvm/IR/IntrinsicsAMDGPU.h"
23#include "llvm/IR/MDBuilder.h"
25#include <cmath>
26
27#define DEBUG_TYPE "amdgpu-simplifylib"
28
29using namespace llvm;
30using namespace llvm::PatternMatch;
31
32static cl::opt<bool> EnablePreLink("amdgpu-prelink",
33 cl::desc("Enable pre-link mode optimizations"),
34 cl::init(false),
36
37static cl::list<std::string> UseNative("amdgpu-use-native",
38 cl::desc("Comma separated list of functions to replace with native, or all"),
41
42#define MATH_PI numbers::pi
43#define MATH_E numbers::e
44#define MATH_SQRT2 numbers::sqrt2
45#define MATH_SQRT1_2 numbers::inv_sqrt2
46
47enum class PowKind { Pow, PowR, PowN, RootN };
48
49namespace llvm {
50
52private:
54
55 using FuncInfo = llvm::AMDGPULibFunc;
56
57 // -fuse-native.
58 bool AllNative = false;
59
60 bool useNativeFunc(const StringRef F) const;
61
62 // Return a pointer (pointer expr) to the function if function definition with
63 // "FuncName" exists. It may create a new function prototype in pre-link mode.
64 FunctionCallee getFunction(Module *M, const FuncInfo &fInfo);
65
66 /// Wrapper around getFunction which tries to use a faster variant if
67 /// available, and falls back to a less fast option.
68 ///
69 /// Return a replacement function for \p fInfo that has float-typed fast
70 /// variants. \p NewFunc is a base replacement function to use. \p
71 /// NewFuncFastVariant is a faster version to use if the calling context knows
72 /// it's legal. If there is no fast variant to use, \p NewFuncFastVariant
73 /// should be EI_NONE.
74 FunctionCallee getFloatFastVariant(Module *M, const FuncInfo &fInfo,
75 FuncInfo &newInfo,
77 AMDGPULibFunc::EFuncId NewFuncFastVariant);
78
79 bool parseFunctionName(const StringRef &FMangledName, FuncInfo &FInfo);
80
81 bool TDOFold(CallInst *CI, const FuncInfo &FInfo);
82
83 /* Specialized optimizations */
84
85 // pow/powr/pown
86 bool fold_pow(FPMathOperator *FPOp, IRBuilder<> &B, const FuncInfo &FInfo);
87
88 /// Peform a fast math expansion of pow, powr, pown or rootn.
89 bool expandFastPow(FPMathOperator *FPOp, IRBuilder<> &B, PowKind Kind);
90
91 bool tryOptimizePow(FPMathOperator *FPOp, IRBuilder<> &B,
92 const FuncInfo &FInfo);
93
94 // rootn
95 bool fold_rootn(FPMathOperator *FPOp, IRBuilder<> &B, const FuncInfo &FInfo);
96
97 // -fuse-native for sincos
98 bool sincosUseNative(CallInst *aCI, const FuncInfo &FInfo);
99
100 // evaluate calls if calls' arguments are constants.
101 bool evaluateScalarMathFunc(const FuncInfo &FInfo, APFloat &Res0,
102 APFloat &Res1, Constant *copr0, Constant *copr1);
103 bool evaluateCall(CallInst *aCI, const FuncInfo &FInfo);
104
105 /// Insert a value to sincos function \p Fsincos. Returns (value of sin, value
106 /// of cos, sincos call).
107 std::tuple<Value *, Value *, Value *> insertSinCos(Value *Arg,
108 FastMathFlags FMF,
109 IRBuilder<> &B,
110 FunctionCallee Fsincos);
111
112 // sin/cos
113 bool fold_sincos(FPMathOperator *FPOp, IRBuilder<> &B, const FuncInfo &FInfo);
114
115 // __read_pipe/__write_pipe
116 bool fold_read_write_pipe(CallInst *CI, IRBuilder<> &B,
117 const FuncInfo &FInfo);
118
119 // Get a scalar native builtin single argument FP function
120 FunctionCallee getNativeFunction(Module *M, const FuncInfo &FInfo);
121
122 /// Substitute a call to a known libcall with an intrinsic call. If \p
123 /// AllowMinSize is true, allow the replacement in a minsize function.
124 bool shouldReplaceLibcallWithIntrinsic(const CallInst *CI,
125 bool AllowMinSizeF32 = false,
126 bool AllowF64 = false,
127 bool AllowStrictFP = false);
128 void replaceLibCallWithSimpleIntrinsic(IRBuilder<> &B, CallInst *CI,
129 Intrinsic::ID IntrID);
130
131 bool tryReplaceLibcallWithSimpleIntrinsic(IRBuilder<> &B, CallInst *CI,
132 Intrinsic::ID IntrID,
133 bool AllowMinSizeF32 = false,
134 bool AllowF64 = false,
135 bool AllowStrictFP = false);
136
137protected:
138 bool isUnsafeFiniteOnlyMath(const FPMathOperator *FPOp) const;
139
141
142 static void replaceCall(Instruction *I, Value *With) {
143 I->replaceAllUsesWith(With);
144 I->eraseFromParent();
145 }
146
147 static void replaceCall(FPMathOperator *I, Value *With) {
149 }
150
151public:
153
154 bool fold(CallInst *CI);
155
156 void initNativeFuncs();
157
158 // Replace a normal math function call with that native version
159 bool useNative(CallInst *CI);
160};
161
162} // end namespace llvm
163
164template <typename IRB>
165static CallInst *CreateCallEx(IRB &B, FunctionCallee Callee, Value *Arg,
166 const Twine &Name = "") {
167 CallInst *R = B.CreateCall(Callee, Arg, Name);
168 if (Function *F = dyn_cast<Function>(Callee.getCallee()))
169 R->setCallingConv(F->getCallingConv());
170 return R;
171}
172
173template <typename IRB>
174static CallInst *CreateCallEx2(IRB &B, FunctionCallee Callee, Value *Arg1,
175 Value *Arg2, const Twine &Name = "") {
176 CallInst *R = B.CreateCall(Callee, {Arg1, Arg2}, Name);
177 if (Function *F = dyn_cast<Function>(Callee.getCallee()))
178 R->setCallingConv(F->getCallingConv());
179 return R;
180}
181
183 Type *PowNExpTy = Type::getInt32Ty(FT->getContext());
184 if (VectorType *VecTy = dyn_cast<VectorType>(FT->getReturnType()))
185 PowNExpTy = VectorType::get(PowNExpTy, VecTy->getElementCount());
186
187 return FunctionType::get(FT->getReturnType(),
188 {FT->getParamType(0), PowNExpTy}, false);
189}
190
191// Data structures for table-driven optimizations.
192// FuncTbl works for both f32 and f64 functions with 1 input argument
193
195 double result;
196 double input;
197};
198
199/* a list of {result, input} */
200static const TableEntry tbl_acos[] = {
201 {MATH_PI / 2.0, 0.0},
202 {MATH_PI / 2.0, -0.0},
203 {0.0, 1.0},
204 {MATH_PI, -1.0}
205};
206static const TableEntry tbl_acosh[] = {
207 {0.0, 1.0}
208};
209static const TableEntry tbl_acospi[] = {
210 {0.5, 0.0},
211 {0.5, -0.0},
212 {0.0, 1.0},
213 {1.0, -1.0}
214};
215static const TableEntry tbl_asin[] = {
216 {0.0, 0.0},
217 {-0.0, -0.0},
218 {MATH_PI / 2.0, 1.0},
219 {-MATH_PI / 2.0, -1.0}
220};
221static const TableEntry tbl_asinh[] = {
222 {0.0, 0.0},
223 {-0.0, -0.0}
224};
225static const TableEntry tbl_asinpi[] = {
226 {0.0, 0.0},
227 {-0.0, -0.0},
228 {0.5, 1.0},
229 {-0.5, -1.0}
230};
231static const TableEntry tbl_atan[] = {
232 {0.0, 0.0},
233 {-0.0, -0.0},
234 {MATH_PI / 4.0, 1.0},
235 {-MATH_PI / 4.0, -1.0}
236};
237static const TableEntry tbl_atanh[] = {
238 {0.0, 0.0},
239 {-0.0, -0.0}
240};
241static const TableEntry tbl_atanpi[] = {
242 {0.0, 0.0},
243 {-0.0, -0.0},
244 {0.25, 1.0},
245 {-0.25, -1.0}
246};
247static const TableEntry tbl_cbrt[] = {
248 {0.0, 0.0},
249 {-0.0, -0.0},
250 {1.0, 1.0},
251 {-1.0, -1.0},
252};
253static const TableEntry tbl_cos[] = {
254 {1.0, 0.0},
255 {1.0, -0.0}
256};
257static const TableEntry tbl_cosh[] = {
258 {1.0, 0.0},
259 {1.0, -0.0}
260};
261static const TableEntry tbl_cospi[] = {
262 {1.0, 0.0},
263 {1.0, -0.0}
264};
265static const TableEntry tbl_erfc[] = {
266 {1.0, 0.0},
267 {1.0, -0.0}
268};
269static const TableEntry tbl_erf[] = {
270 {0.0, 0.0},
271 {-0.0, -0.0}
272};
273static const TableEntry tbl_exp[] = {
274 {1.0, 0.0},
275 {1.0, -0.0},
276 {MATH_E, 1.0}
277};
278static const TableEntry tbl_exp2[] = {
279 {1.0, 0.0},
280 {1.0, -0.0},
281 {2.0, 1.0}
282};
283static const TableEntry tbl_exp10[] = {
284 {1.0, 0.0},
285 {1.0, -0.0},
286 {10.0, 1.0}
287};
288static const TableEntry tbl_expm1[] = {
289 {0.0, 0.0},
290 {-0.0, -0.0}
291};
292static const TableEntry tbl_log[] = {
293 {0.0, 1.0},
294 {1.0, MATH_E}
295};
296static const TableEntry tbl_log2[] = {
297 {0.0, 1.0},
298 {1.0, 2.0}
299};
300static const TableEntry tbl_log10[] = {
301 {0.0, 1.0},
302 {1.0, 10.0}
303};
304static const TableEntry tbl_rsqrt[] = {
305 {1.0, 1.0},
306 {MATH_SQRT1_2, 2.0}
307};
308static const TableEntry tbl_sin[] = {
309 {0.0, 0.0},
310 {-0.0, -0.0}
311};
312static const TableEntry tbl_sinh[] = {
313 {0.0, 0.0},
314 {-0.0, -0.0}
315};
316static const TableEntry tbl_sinpi[] = {
317 {0.0, 0.0},
318 {-0.0, -0.0}
319};
320static const TableEntry tbl_sqrt[] = {
321 {0.0, 0.0},
322 {1.0, 1.0},
323 {MATH_SQRT2, 2.0}
324};
325static const TableEntry tbl_tan[] = {
326 {0.0, 0.0},
327 {-0.0, -0.0}
328};
329static const TableEntry tbl_tanh[] = {
330 {0.0, 0.0},
331 {-0.0, -0.0}
332};
333static const TableEntry tbl_tanpi[] = {
334 {0.0, 0.0},
335 {-0.0, -0.0}
336};
337static const TableEntry tbl_tgamma[] = {
338 {1.0, 1.0},
339 {1.0, 2.0},
340 {2.0, 3.0},
341 {6.0, 4.0}
342};
343
345 switch(id) {
361 return true;
362 default:;
363 }
364 return false;
365}
366
368
370 switch(id) {
408 default:;
409 }
410 return TableRef();
411}
412
413static inline int getVecSize(const AMDGPULibFunc& FInfo) {
414 return FInfo.getLeads()[0].VectorSize;
415}
416
417static inline AMDGPULibFunc::EType getArgType(const AMDGPULibFunc& FInfo) {
418 return (AMDGPULibFunc::EType)FInfo.getLeads()[0].ArgType;
419}
420
421FunctionCallee AMDGPULibCalls::getFunction(Module *M, const FuncInfo &fInfo) {
422 // If we are doing PreLinkOpt, the function is external. So it is safe to
423 // use getOrInsertFunction() at this stage.
424
426 : AMDGPULibFunc::getFunction(M, fInfo);
427}
428
429FunctionCallee AMDGPULibCalls::getFloatFastVariant(
430 Module *M, const FuncInfo &fInfo, FuncInfo &newInfo,
431 AMDGPULibFunc::EFuncId NewFunc, AMDGPULibFunc::EFuncId FastVariant) {
432 assert(NewFunc != FastVariant);
433
434 if (FastVariant != AMDGPULibFunc::EI_NONE &&
435 getArgType(fInfo) == AMDGPULibFunc::F32) {
436 newInfo = AMDGPULibFunc(FastVariant, fInfo);
437 if (FunctionCallee NewCallee = getFunction(M, newInfo))
438 return NewCallee;
439 }
440
441 newInfo = AMDGPULibFunc(NewFunc, fInfo);
442 return getFunction(M, newInfo);
443}
444
445bool AMDGPULibCalls::parseFunctionName(const StringRef &FMangledName,
446 FuncInfo &FInfo) {
447 return AMDGPULibFunc::parse(FMangledName, FInfo);
448}
449
451 return FPOp->hasApproxFunc() && FPOp->hasNoNaNs() && FPOp->hasNoInfs();
452}
453
455 const FPMathOperator *FPOp) const {
456 // TODO: Refine to approxFunc or contract
457 return FPOp->isFast();
458}
459
461 : SQ(F.getParent()->getDataLayout(),
462 &FAM.getResult<TargetLibraryAnalysis>(F),
463 FAM.getCachedResult<DominatorTreeAnalysis>(F),
464 &FAM.getResult<AssumptionAnalysis>(F)) {}
465
466bool AMDGPULibCalls::useNativeFunc(const StringRef F) const {
467 return AllNative || llvm::is_contained(UseNative, F);
468}
469
471 AllNative = useNativeFunc("all") ||
472 (UseNative.getNumOccurrences() && UseNative.size() == 1 &&
473 UseNative.begin()->empty());
474}
475
476bool AMDGPULibCalls::sincosUseNative(CallInst *aCI, const FuncInfo &FInfo) {
477 bool native_sin = useNativeFunc("sin");
478 bool native_cos = useNativeFunc("cos");
479
480 if (native_sin && native_cos) {
481 Module *M = aCI->getModule();
482 Value *opr0 = aCI->getArgOperand(0);
483
484 AMDGPULibFunc nf;
485 nf.getLeads()[0].ArgType = FInfo.getLeads()[0].ArgType;
486 nf.getLeads()[0].VectorSize = FInfo.getLeads()[0].VectorSize;
487
490 FunctionCallee sinExpr = getFunction(M, nf);
491
494 FunctionCallee cosExpr = getFunction(M, nf);
495 if (sinExpr && cosExpr) {
496 Value *sinval =
497 CallInst::Create(sinExpr, opr0, "splitsin", aCI->getIterator());
498 Value *cosval =
499 CallInst::Create(cosExpr, opr0, "splitcos", aCI->getIterator());
500 new StoreInst(cosval, aCI->getArgOperand(1), aCI->getIterator());
501
502 DEBUG_WITH_TYPE("usenative", dbgs() << "<useNative> replace " << *aCI
503 << " with native version of sin/cos");
504
505 replaceCall(aCI, sinval);
506 return true;
507 }
508 }
509 return false;
510}
511
513 Function *Callee = aCI->getCalledFunction();
514 if (!Callee || aCI->isNoBuiltin())
515 return false;
516
517 FuncInfo FInfo;
518 if (!parseFunctionName(Callee->getName(), FInfo) || !FInfo.isMangled() ||
519 FInfo.getPrefix() != AMDGPULibFunc::NOPFX ||
520 getArgType(FInfo) == AMDGPULibFunc::F64 || !HasNative(FInfo.getId()) ||
521 !(AllNative || useNativeFunc(FInfo.getName()))) {
522 return false;
523 }
524
525 if (FInfo.getId() == AMDGPULibFunc::EI_SINCOS)
526 return sincosUseNative(aCI, FInfo);
527
529 FunctionCallee F = getFunction(aCI->getModule(), FInfo);
530 if (!F)
531 return false;
532
533 aCI->setCalledFunction(F);
534 DEBUG_WITH_TYPE("usenative", dbgs() << "<useNative> replace " << *aCI
535 << " with native version");
536 return true;
537}
538
539// Clang emits call of __read_pipe_2 or __read_pipe_4 for OpenCL read_pipe
540// builtin, with appended type size and alignment arguments, where 2 or 4
541// indicates the original number of arguments. The library has optimized version
542// of __read_pipe_2/__read_pipe_4 when the type size and alignment has the same
543// power of 2 value. This function transforms __read_pipe_2 to __read_pipe_2_N
544// for such cases where N is the size in bytes of the type (N = 1, 2, 4, 8, ...,
545// 128). The same for __read_pipe_4, write_pipe_2, and write_pipe_4.
546bool AMDGPULibCalls::fold_read_write_pipe(CallInst *CI, IRBuilder<> &B,
547 const FuncInfo &FInfo) {
548 auto *Callee = CI->getCalledFunction();
549 if (!Callee->isDeclaration())
550 return false;
551
552 assert(Callee->hasName() && "Invalid read_pipe/write_pipe function");
553 auto *M = Callee->getParent();
554 std::string Name = std::string(Callee->getName());
555 auto NumArg = CI->arg_size();
556 if (NumArg != 4 && NumArg != 6)
557 return false;
558 ConstantInt *PacketSize =
559 dyn_cast<ConstantInt>(CI->getArgOperand(NumArg - 2));
560 ConstantInt *PacketAlign =
561 dyn_cast<ConstantInt>(CI->getArgOperand(NumArg - 1));
562 if (!PacketSize || !PacketAlign)
563 return false;
564
565 unsigned Size = PacketSize->getZExtValue();
566 Align Alignment = PacketAlign->getAlignValue();
567 if (Alignment != Size)
568 return false;
569
570 unsigned PtrArgLoc = CI->arg_size() - 3;
571 Value *PtrArg = CI->getArgOperand(PtrArgLoc);
572 Type *PtrTy = PtrArg->getType();
573
575 for (unsigned I = 0; I != PtrArgLoc; ++I)
576 ArgTys.push_back(CI->getArgOperand(I)->getType());
577 ArgTys.push_back(PtrTy);
578
579 Name = Name + "_" + std::to_string(Size);
580 auto *FTy = FunctionType::get(Callee->getReturnType(),
581 ArrayRef<Type *>(ArgTys), false);
582 AMDGPULibFunc NewLibFunc(Name, FTy);
584 if (!F)
585 return false;
586
588 for (unsigned I = 0; I != PtrArgLoc; ++I)
589 Args.push_back(CI->getArgOperand(I));
590 Args.push_back(PtrArg);
591
592 auto *NCI = B.CreateCall(F, Args);
593 NCI->setAttributes(CI->getAttributes());
594 CI->replaceAllUsesWith(NCI);
595 CI->dropAllReferences();
596 CI->eraseFromParent();
597
598 return true;
599}
600
601// This function returns false if no change; return true otherwise.
603 Function *Callee = CI->getCalledFunction();
604 // Ignore indirect calls.
605 if (!Callee || Callee->isIntrinsic() || CI->isNoBuiltin())
606 return false;
607
608 FuncInfo FInfo;
609 if (!parseFunctionName(Callee->getName(), FInfo))
610 return false;
611
612 // Further check the number of arguments to see if they match.
613 // TODO: Check calling convention matches too
614 if (!FInfo.isCompatibleSignature(*Callee->getParent(), CI->getFunctionType()))
615 return false;
616
617 LLVM_DEBUG(dbgs() << "AMDIC: try folding " << *CI << '\n');
618
619 if (TDOFold(CI, FInfo))
620 return true;
621
622 IRBuilder<> B(CI);
623 if (CI->isStrictFP())
624 B.setIsFPConstrained(true);
625
627 // Under unsafe-math, evaluate calls if possible.
628 // According to Brian Sumner, we can do this for all f32 function calls
629 // using host's double function calls.
630 if (canIncreasePrecisionOfConstantFold(FPOp) && evaluateCall(CI, FInfo))
631 return true;
632
633 // Copy fast flags from the original call.
634 FastMathFlags FMF = FPOp->getFastMathFlags();
635 B.setFastMathFlags(FMF);
636
637 // Specialized optimizations for each function call.
638 //
639 // TODO: Handle native functions
640 switch (FInfo.getId()) {
642 if (FMF.none())
643 return false;
644 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::exp,
645 FMF.approxFunc());
647 if (FMF.none())
648 return false;
649 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::exp2,
650 FMF.approxFunc());
652 if (FMF.none())
653 return false;
654 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::log,
655 FMF.approxFunc());
657 if (FMF.none())
658 return false;
659 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::log2,
660 FMF.approxFunc());
662 if (FMF.none())
663 return false;
664 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::log10,
665 FMF.approxFunc());
667 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::minnum,
668 true, true);
670 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::maxnum,
671 true, true);
673 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::fma, true,
674 true);
676 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::fmuladd,
677 true, true);
679 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::fabs, true,
680 true, true);
682 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::copysign,
683 true, true, true);
685 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::floor, true,
686 true);
688 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::ceil, true,
689 true);
691 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::trunc, true,
692 true);
694 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::rint, true,
695 true);
697 return tryReplaceLibcallWithSimpleIntrinsic(B, CI, Intrinsic::round, true,
698 true);
700 if (!shouldReplaceLibcallWithIntrinsic(CI, true, true))
701 return false;
702
703 Value *Arg1 = CI->getArgOperand(1);
704 if (VectorType *VecTy = dyn_cast<VectorType>(CI->getType());
705 VecTy && !isa<VectorType>(Arg1->getType())) {
706 Value *SplatArg1 = B.CreateVectorSplat(VecTy->getElementCount(), Arg1);
707 CI->setArgOperand(1, SplatArg1);
708 }
709
711 CI->getModule(), Intrinsic::ldexp,
712 {CI->getType(), CI->getArgOperand(1)->getType()}));
714 return true;
715 }
718 return tryOptimizePow(FPOp, B, FInfo);
721 if (fold_pow(FPOp, B, FInfo))
722 return true;
723 if (!FMF.approxFunc())
724 return false;
725
726 if (FInfo.getId() == AMDGPULibFunc::EI_POWR && FMF.approxFunc() &&
727 getArgType(FInfo) == AMDGPULibFunc::F32) {
728 Module *M = Callee->getParent();
729 AMDGPULibFunc PowrFastInfo(AMDGPULibFunc::EI_POWR_FAST, FInfo);
730 if (FunctionCallee PowrFastFunc = getFunction(M, PowrFastInfo)) {
731 CI->setCalledFunction(PowrFastFunc);
732 return true;
733 }
734 }
735
736 if (!shouldReplaceLibcallWithIntrinsic(CI))
737 return false;
738 return expandFastPow(FPOp, B, PowKind::PowR);
739 }
742 if (fold_pow(FPOp, B, FInfo))
743 return true;
744 if (!FMF.approxFunc())
745 return false;
746
747 if (FInfo.getId() == AMDGPULibFunc::EI_POWN &&
748 getArgType(FInfo) == AMDGPULibFunc::F32) {
749 Module *M = Callee->getParent();
750 AMDGPULibFunc PownFastInfo(AMDGPULibFunc::EI_POWN_FAST, FInfo);
751 if (FunctionCallee PownFastFunc = getFunction(M, PownFastInfo)) {
752 CI->setCalledFunction(PownFastFunc);
753 return true;
754 }
755 }
756
757 if (!shouldReplaceLibcallWithIntrinsic(CI))
758 return false;
759 return expandFastPow(FPOp, B, PowKind::PowN);
760 }
763 if (fold_rootn(FPOp, B, FInfo))
764 return true;
765 if (!FMF.approxFunc())
766 return false;
767
768 if (getArgType(FInfo) == AMDGPULibFunc::F32) {
769 Module *M = Callee->getParent();
770 AMDGPULibFunc RootnFastInfo(AMDGPULibFunc::EI_ROOTN_FAST, FInfo);
771 if (FunctionCallee RootnFastFunc = getFunction(M, RootnFastInfo)) {
772 CI->setCalledFunction(RootnFastFunc);
773 return true;
774 }
775 }
776
777 return expandFastPow(FPOp, B, PowKind::RootN);
778 }
780 // TODO: Allow with strictfp + constrained intrinsic
781 return tryReplaceLibcallWithSimpleIntrinsic(
782 B, CI, Intrinsic::sqrt, true, true, /*AllowStrictFP=*/false);
785 return fold_sincos(FPOp, B, FInfo);
786 default:
787 break;
788 }
789 } else {
790 // Specialized optimizations for each function call
791 switch (FInfo.getId()) {
796 return fold_read_write_pipe(CI, B, FInfo);
797 default:
798 break;
799 }
800 }
801
802 return false;
803}
804
806 const Type *Ty) {
807
808 assert(Ty->isSingleValueType() &&
809 "Type must either be a scalar or a vector.");
810 assert((!Ty->isVectorTy() || Ty->isScalableTy() ||
811 Values.size() == cast<FixedVectorType>(Ty)->getNumElements()) &&
812 "Unexpected number of constant values.");
813 assert((Ty->isVectorTy() || Values.size() == 1) &&
814 "Expected exactly one constant value");
815
816 Type *ElemTy = Ty->getScalarType();
817 const fltSemantics &FltSem = ElemTy->getFltSemantics();
818
819 SmallVector<Constant *, 4> ConstValues;
820 ConstValues.reserve(Values.size());
821 for (APFloat APF : Values) {
822 bool Unused;
823 APF.convert(FltSem, APFloat::rmNearestTiesToEven, &Unused);
824 ConstValues.push_back(ConstantFP::get(ElemTy, APF));
825 }
826
827 return Ty->isVectorTy() ? ConstantVector::get(ConstValues) : ConstValues[0];
828}
829
830bool AMDGPULibCalls::TDOFold(CallInst *CI, const FuncInfo &FInfo) {
831 // Table-Driven optimization
832 const TableRef tr = getOptTable(FInfo.getId());
833 if (tr.empty())
834 return false;
835
836 int const sz = (int)tr.size();
837 Value *opr0 = CI->getArgOperand(0);
838
839 int vecSize = getVecSize(FInfo);
840 if (vecSize > 1) {
841 // Vector version
842 Constant *CV = dyn_cast<Constant>(opr0);
843 if (CV && CV->getType()->isVectorTy()) {
845 Values.reserve(vecSize);
846 for (int eltNo = 0; eltNo < vecSize; ++eltNo) {
847 // A lane may be undef or poison, in which case there is nothing to
848 // look up in the table.
850 CV->getAggregateElement((unsigned)eltNo));
851 if (!eltval)
852 return false;
853 auto MatchingRow = llvm::find_if(tr, [eltval](const TableEntry &entry) {
854 return eltval->isExactlyValue(entry.input);
855 });
856 if (MatchingRow == tr.end())
857 return false;
858 Values.push_back(APFloat(MatchingRow->result));
859 }
860 Constant *NewValues = getConstantFloat(Values, CI->getType());
861 LLVM_DEBUG(errs() << "AMDIC: " << *CI << " ---> " << *NewValues << "\n");
862 replaceCall(CI, NewValues);
863 return true;
864 }
865 } else {
866 // Scalar version
867 if (ConstantFP *CF = dyn_cast<ConstantFP>(opr0)) {
868 for (int i = 0; i < sz; ++i) {
869 if (CF->isExactlyValue(tr[i].input)) {
870 Value *nval = ConstantFP::get(CF->getType(), tr[i].result);
871 LLVM_DEBUG(errs() << "AMDIC: " << *CI << " ---> " << *nval << "\n");
872 replaceCall(CI, nval);
873 return true;
874 }
875 }
876 }
877 }
878
879 return false;
880}
881
882namespace llvm {
883static double log2(double V) {
884#if _XOPEN_SOURCE >= 600 || defined(_ISOC99_SOURCE) || _POSIX_C_SOURCE >= 200112L
885 return ::log2(V);
886#else
887 return log(V) / numbers::ln2;
888#endif
889}
890} // namespace llvm
891
892bool AMDGPULibCalls::fold_pow(FPMathOperator *FPOp, IRBuilder<> &B,
893 const FuncInfo &FInfo) {
894 assert((FInfo.getId() == AMDGPULibFunc::EI_POW ||
895 FInfo.getId() == AMDGPULibFunc::EI_POW_FAST ||
896 FInfo.getId() == AMDGPULibFunc::EI_POWR ||
897 FInfo.getId() == AMDGPULibFunc::EI_POWR_FAST ||
898 FInfo.getId() == AMDGPULibFunc::EI_POWN ||
899 FInfo.getId() == AMDGPULibFunc::EI_POWN_FAST) &&
900 "fold_pow: encounter a wrong function call");
901
902 Module *M = B.GetInsertBlock()->getModule();
903 Type *eltType = FPOp->getType()->getScalarType();
904 Value *opr0 = FPOp->getOperand(0);
905 Value *opr1 = FPOp->getOperand(1);
906
907 const APFloat *CF = nullptr;
908 const APInt *CINT = nullptr;
909 if (!match(opr1, m_APFloatAllowPoison(CF)))
910 match(opr1, m_APIntAllowPoison(CINT));
911
912 // 0x1111111 means that we don't do anything for this call.
913 int ci_opr1 = (CINT ? (int)CINT->getSExtValue() : 0x1111111);
914
915 // OpenCL powr(x<0, y) = NaN, but the folds below would turn it into a
916 // finite number. Skip them unless NaNs are ignored or the base is known
917 // non-negative.
918 bool IsPowr = FInfo.getId() == AMDGPULibFunc::EI_POWR ||
919 FInfo.getId() == AMDGPULibFunc::EI_POWR_FAST;
920 bool SkipConstantFolds =
921 IsPowr && !FPOp->hasNoNaNs() &&
923 opr0, SQ.getWithInstruction(cast<Instruction>(FPOp)));
924
925 if (CF && (CF->isExactlyValue(0.5) || CF->isExactlyValue(-0.5))) {
926 // pow[r](x, [-]0.5) = sqrt(x) / rsqrt(x)
927 //
928 // sqrt/rsqrt and pow disagree on two negative inputs:
929 // pow(-Inf, 0.5) == +Inf but sqrt(-Inf) == NaN (ninf case)
930 // pow(-0.0, 0.5) == +0.0 but sqrt(-0.0) == -0.0 (nsz case)
931 // powr requires x >= 0 by the OpenCL spec, so -Inf is undefined behaviour
932 // and the ninf check can be skipped for powr/powr_fast. -0.0 is a valid
933 // input for powr since -0.0 >= 0 by IEEE comparison, so nsz is still
934 // required for all variants. sqrt/rsqrt already return NaN for a
935 // negative base like powr does, so this fold skips the base-sign check.
936 if (FPOp->hasNoSignedZeros() && (IsPowr || FPOp->hasNoInfs())) {
937 bool issqrt = CF->isExactlyValue(0.5);
938 if (FunctionCallee FPExpr =
939 getFunction(M, AMDGPULibFunc(issqrt ? AMDGPULibFunc::EI_SQRT
941 FInfo))) {
942 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> " << FInfo.getName()
943 << '(' << *opr0 << ")\n");
944 Value *nval = CreateCallEx(B, FPExpr, opr0,
945 issqrt ? "__pow2sqrt" : "__pow2rsqrt");
946 replaceCall(FPOp, nval);
947 return true;
948 }
949 }
950 }
951
952 if (!SkipConstantFolds) {
953 if ((CF && CF->isZero()) || (CINT && ci_opr1 == 0)) {
954 // pow/powr/pown(x, 0) == 1
955 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> 1\n");
956 Constant *cnval = ConstantFP::get(eltType, 1.0);
957 if (getVecSize(FInfo) > 1) {
958 cnval = ConstantDataVector::getSplat(getVecSize(FInfo), cnval);
959 }
960 replaceCall(FPOp, cnval);
961 return true;
962 }
963 if ((CF && CF->isOne()) || (CINT && ci_opr1 == 1)) {
964 // pow/powr/pown(x, 1.0) = x
965 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> " << *opr0 << "\n");
966 replaceCall(FPOp, opr0);
967 return true;
968 }
969 if ((CF && CF->isExactlyValue(2.0)) || (CINT && ci_opr1 == 2)) {
970 // pow/powr/pown(x, 2.0) = x*x
971 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> " << *opr0 << " * "
972 << *opr0 << "\n");
973 Value *nval = B.CreateFMul(opr0, opr0, "__pow2");
974 replaceCall(FPOp, nval);
975 return true;
976 }
977 if ((CF && CF->isMinusOne()) || (CINT && ci_opr1 == -1)) {
978 // pow/powr/pown(x, -1.0) = 1.0/x
979 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> 1 / " << *opr0 << "\n");
980 Constant *cnval = ConstantFP::get(eltType, 1.0);
981 if (getVecSize(FInfo) > 1) {
982 cnval = ConstantDataVector::getSplat(getVecSize(FInfo), cnval);
983 }
984 Value *nval = B.CreateFDiv(cnval, opr0, "__powrecip");
985 replaceCall(FPOp, nval);
986 return true;
987 }
988 }
989
990 if (!isUnsafeFiniteOnlyMath(FPOp))
991 return false;
992
993 // Unsafe Math optimization
994
995 // Remember that ci_opr1 is set if opr1 is integral
996 if (CF) {
997 double dval = (getArgType(FInfo) == AMDGPULibFunc::F32)
998 ? (double)CF->convertToFloat()
999 : CF->convertToDouble();
1000 int ival = (int)dval;
1001 if ((double)ival == dval) {
1002 ci_opr1 = ival;
1003 } else
1004 ci_opr1 = 0x11111111;
1005 }
1006
1007 // pow/powr/pown(x, c) = [1/](x*x*..x); where
1008 // trunc(c) == c && the number of x == c && |c| <= 12
1009 unsigned abs_opr1 = (ci_opr1 < 0) ? -ci_opr1 : ci_opr1;
1010 if (abs_opr1 <= 12) {
1011 Constant *cnval;
1012 Value *nval;
1013 if (abs_opr1 == 0) {
1014 cnval = ConstantFP::get(eltType, 1.0);
1015 if (getVecSize(FInfo) > 1) {
1016 cnval = ConstantDataVector::getSplat(getVecSize(FInfo), cnval);
1017 }
1018 nval = cnval;
1019 } else {
1020 Value *valx2 = nullptr;
1021 nval = nullptr;
1022 while (abs_opr1 > 0) {
1023 valx2 = valx2 ? B.CreateFMul(valx2, valx2, "__powx2") : opr0;
1024 if (abs_opr1 & 1) {
1025 nval = nval ? B.CreateFMul(nval, valx2, "__powprod") : valx2;
1026 }
1027 abs_opr1 >>= 1;
1028 }
1029 }
1030
1031 if (ci_opr1 < 0) {
1032 cnval = ConstantFP::get(eltType, 1.0);
1033 if (getVecSize(FInfo) > 1) {
1034 cnval = ConstantDataVector::getSplat(getVecSize(FInfo), cnval);
1035 }
1036 nval = B.CreateFDiv(cnval, nval, "__1powprod");
1037 }
1038 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> "
1039 << ((ci_opr1 < 0) ? "1/prod(" : "prod(") << *opr0
1040 << ")\n");
1041 replaceCall(FPOp, nval);
1042 return true;
1043 }
1044
1045 // If we should use the generic intrinsic instead of emitting a libcall
1046 const bool ShouldUseIntrinsic = eltType->isFloatTy() || eltType->isHalfTy();
1047
1048 // powr ---> exp2(y * log2(x))
1049 // pown/pow ---> powr(fabs(x), y) | (x & ((int)y << 31))
1050 FunctionCallee ExpExpr;
1051 if (ShouldUseIntrinsic)
1052 ExpExpr = Intrinsic::getOrInsertDeclaration(M, Intrinsic::exp2,
1053 {FPOp->getType()});
1054 else {
1055 ExpExpr = getFunction(M, AMDGPULibFunc(AMDGPULibFunc::EI_EXP2, FInfo));
1056 if (!ExpExpr)
1057 return false;
1058 }
1059
1060 bool needlog = false;
1061 bool needabs = false;
1062 bool needcopysign = false;
1063 Constant *cnval = nullptr;
1064 if (getVecSize(FInfo) == 1) {
1065 CF = nullptr;
1066 match(opr0, m_APFloatAllowPoison(CF));
1067
1068 if (CF) {
1069 double V = (getArgType(FInfo) == AMDGPULibFunc::F32)
1070 ? (double)CF->convertToFloat()
1071 : CF->convertToDouble();
1072
1073 V = log2(std::abs(V));
1074 cnval = ConstantFP::get(eltType, V);
1075 needcopysign = (FInfo.getId() != AMDGPULibFunc::EI_POWR &&
1076 FInfo.getId() != AMDGPULibFunc::EI_POWR_FAST) &&
1077 CF->isNegative();
1078 } else {
1079 needlog = true;
1080 needcopysign = needabs = FInfo.getId() != AMDGPULibFunc::EI_POWR &&
1081 FInfo.getId() != AMDGPULibFunc::EI_POWR_FAST;
1082 }
1083 } else {
1084 ConstantDataVector *CDV = dyn_cast<ConstantDataVector>(opr0);
1085
1086 if (!CDV) {
1087 needlog = true;
1088 needcopysign = needabs = FInfo.getId() != AMDGPULibFunc::EI_POWR &&
1089 FInfo.getId() != AMDGPULibFunc::EI_POWR_FAST;
1090 } else {
1091 assert ((int)CDV->getNumElements() == getVecSize(FInfo) &&
1092 "Wrong vector size detected");
1093
1095 for (int i=0; i < getVecSize(FInfo); ++i) {
1096 double V = CDV->getElementAsAPFloat(i).convertToDouble();
1097 if (V < 0.0) needcopysign = true;
1098 V = log2(std::abs(V));
1099 DVal.push_back(V);
1100 }
1101 if (getArgType(FInfo) == AMDGPULibFunc::F32) {
1103 for (double D : DVal)
1104 FVal.push_back((float)D);
1105 ArrayRef<float> tmp(FVal);
1106 cnval = ConstantDataVector::get(M->getContext(), tmp);
1107 } else {
1108 ArrayRef<double> tmp(DVal);
1109 cnval = ConstantDataVector::get(M->getContext(), tmp);
1110 }
1111 }
1112 }
1113
1114 if (needcopysign && (FInfo.getId() == AMDGPULibFunc::EI_POW ||
1115 FInfo.getId() == AMDGPULibFunc::EI_POW_FAST)) {
1116 // We cannot handle corner cases for a general pow() function, give up
1117 // unless y is a constant integral value. Then proceed as if it were pown.
1118 if (!isKnownIntegral(opr1, SQ.getWithInstruction(cast<Instruction>(FPOp)),
1119 FPOp->getFastMathFlags()))
1120 return false;
1121 }
1122
1123 Value *nval;
1124 if (needabs) {
1125 nval = B.CreateFAbs(opr0, nullptr, "__fabs");
1126 } else {
1127 nval = cnval ? cnval : opr0;
1128 }
1129 if (needlog) {
1130 FunctionCallee LogExpr;
1131 if (ShouldUseIntrinsic) {
1132 LogExpr = Intrinsic::getOrInsertDeclaration(M, Intrinsic::log2,
1133 {FPOp->getType()});
1134 } else {
1135 LogExpr = getFunction(M, AMDGPULibFunc(AMDGPULibFunc::EI_LOG2, FInfo));
1136 if (!LogExpr)
1137 return false;
1138 }
1139
1140 nval = CreateCallEx(B,LogExpr, nval, "__log2");
1141 }
1142
1143 if (FInfo.getId() == AMDGPULibFunc::EI_POWN ||
1144 FInfo.getId() == AMDGPULibFunc::EI_POWN_FAST) {
1145 // convert int(32) to fp(f32 or f64)
1146 opr1 = B.CreateSIToFP(opr1, nval->getType(), "pownI2F");
1147 }
1148 nval = B.CreateFMul(opr1, nval, "__ylogx");
1149
1150 CallInst *Exp2Call = CreateCallEx(B, ExpExpr, nval, "__exp2");
1151
1152 // TODO: Generalized fpclass logic for pow
1154 if (FPOp->hasNoNaNs())
1155 KnownNot |= FPClassTest::fcNan;
1156
1157 Exp2Call->addRetAttr(
1158 Attribute::getWithNoFPClass(Exp2Call->getContext(), KnownNot));
1159 nval = Exp2Call;
1160
1161 if (needcopysign) {
1162 Type* nTyS = B.getIntNTy(eltType->getPrimitiveSizeInBits());
1163 Type *nTy = FPOp->getType()->getWithNewType(nTyS);
1164 Value *opr_n = FPOp->getOperand(1);
1165 if (opr_n->getType()->getScalarType()->isIntegerTy())
1166 opr_n = B.CreateZExtOrTrunc(opr_n, nTy, "__ytou");
1167 else
1168 opr_n = B.CreateFPToSI(opr1, nTy, "__ytou");
1169
1170 unsigned size = nTy->getScalarSizeInBits();
1171 Value *sign = B.CreateShl(opr_n, size-1, "__yeven");
1172 sign = B.CreateAnd(B.CreateBitCast(opr0, nTy), sign, "__pow_sign");
1173
1174 nval = B.CreateCopySign(nval, B.CreateBitCast(sign, nval->getType()),
1175 nullptr, "__pow_sign");
1176 }
1177
1178 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> "
1179 << "exp2(" << *opr1 << " * log2(" << *opr0 << "))\n");
1180 replaceCall(FPOp, nval);
1181
1182 return true;
1183}
1184
1185bool AMDGPULibCalls::fold_rootn(FPMathOperator *FPOp, IRBuilder<> &B,
1186 const FuncInfo &FInfo) {
1187 Value *opr0 = FPOp->getOperand(0);
1188 Value *opr1 = FPOp->getOperand(1);
1189
1190 const APInt *CINT = nullptr;
1191 if (!match(opr1, m_APIntAllowPoison(CINT)))
1192 return false;
1193
1194 Function *Parent = B.GetInsertBlock()->getParent();
1195
1196 int ci_opr1 = (int)CINT->getSExtValue();
1197 if (ci_opr1 == 1 && !Parent->hasFnAttribute(Attribute::StrictFP)) {
1198 // rootn(x, 1) = x
1199 //
1200 // TODO: Insert constrained canonicalize for strictfp case.
1201 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> " << *opr0 << '\n');
1202 replaceCall(FPOp, opr0);
1203 return true;
1204 }
1205
1206 Module *M = B.GetInsertBlock()->getModule();
1207
1208 CallInst *CI = cast<CallInst>(FPOp);
1209
1210 // rootn and sqrt disagree on signed-zero / -Inf inputs (e.g. rootn(-0.0, 2)
1211 // is +0.0, sqrt(-0.0) is -0.0), so require nsz/ninf.
1212 bool FMFOkForSqrt = FPOp->hasNoSignedZeros() && FPOp->hasNoInfs();
1213
1214 if (ci_opr1 == 2 && FMFOkForSqrt &&
1215 shouldReplaceLibcallWithIntrinsic(CI,
1216 /*AllowMinSizeF32=*/true,
1217 /*AllowF64=*/true)) {
1218 // rootn(x, 2) = sqrt(x)
1219 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> sqrt(" << *opr0 << ")\n");
1220
1221 Value *NewCall = B.CreateUnaryIntrinsic(Intrinsic::sqrt, opr0, CI);
1222 NewCall->takeName(CI);
1223
1224 // OpenCL rootn has a looser ulp of 2 requirement than sqrt, so add some
1225 // metadata.
1226 MDBuilder MDHelper(M->getContext());
1227 MDNode *FPMD = MDHelper.createFPMath(std::max(FPOp->getFPAccuracy(), 2.0f));
1228 if (auto *NewCallI = dyn_cast<Instruction>(NewCall))
1229 NewCallI->setMetadata(LLVMContext::MD_fpmath, FPMD);
1230
1231 replaceCall(CI, NewCall);
1232 return true;
1233 }
1234
1235 if (ci_opr1 == 3) { // rootn(x, 3) = cbrt(x)
1236 if (FunctionCallee FPExpr =
1237 getFunction(M, AMDGPULibFunc(AMDGPULibFunc::EI_CBRT, FInfo))) {
1238 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> cbrt(" << *opr0
1239 << ")\n");
1240 Value *nval = CreateCallEx(B,FPExpr, opr0, "__rootn2cbrt");
1241 replaceCall(FPOp, nval);
1242 return true;
1243 }
1244 } else if (ci_opr1 == -1) { // rootn(x, -1) = 1.0/x
1245 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> 1.0 / " << *opr0 << "\n");
1246 Value *nval = B.CreateFDiv(ConstantFP::get(opr0->getType(), 1.0),
1247 opr0,
1248 "__rootn2div");
1249 replaceCall(FPOp, nval);
1250 return true;
1251 }
1252
1253 if (ci_opr1 == -2 && FMFOkForSqrt &&
1254 shouldReplaceLibcallWithIntrinsic(CI,
1255 /*AllowMinSizeF32=*/true,
1256 /*AllowF64=*/true)) {
1257 // rootn(x, -2) = rsqrt(x)
1258
1259 // The original rootn had looser ulp requirements than the resultant sqrt
1260 // and fdiv.
1261 MDBuilder MDHelper(M->getContext());
1262 MDNode *FPMD = MDHelper.createFPMath(std::max(FPOp->getFPAccuracy(), 2.0f));
1263
1264 // TODO: Could handle strictfp but need to fix strict sqrt emission
1265 FastMathFlags FMF = FPOp->getFastMathFlags();
1266 FMF.setAllowContract(true);
1267
1268 Value *Sqrt = B.CreateUnaryIntrinsic(Intrinsic::sqrt, opr0, CI);
1270 B.CreateFDiv(ConstantFP::get(opr0->getType(), 1.0), Sqrt));
1271 if (auto *SqrtI = dyn_cast<Instruction>(Sqrt))
1272 SqrtI->setFastMathFlags(FMF);
1273 RSqrt->setFastMathFlags(FMF);
1274 RSqrt->setMetadata(LLVMContext::MD_fpmath, FPMD);
1275
1276 LLVM_DEBUG(errs() << "AMDIC: " << *FPOp << " ---> rsqrt(" << *opr0
1277 << ")\n");
1278 replaceCall(CI, RSqrt);
1279 return true;
1280 }
1281
1282 return false;
1283}
1284
1285// is_integer(y) => trunc(y) == y
1287 Value *TruncY = B.CreateUnaryIntrinsic(Intrinsic::trunc, Y);
1288 return B.CreateFCmpOEQ(TruncY, Y);
1289}
1290
1292 // Even integers are still integers after division by 2.
1293 auto *HalfY = B.CreateFMul(Y, ConstantFP::get(Y->getType(), 0.5));
1294 return emitIsInteger(B, HalfY);
1295}
1296
1297// is_odd_integer(y) => is_integer(y) && !is_even_integer(y)
1299 Value *IsIntY = emitIsInteger(B, Y);
1300 Value *IsEvenY = emitIsEvenInteger(B, Y);
1301 Value *NotEvenY = B.CreateNot(IsEvenY);
1302 return B.CreateAnd(IsIntY, NotEvenY);
1303}
1304
1305// isinf(val) => fabs(val) == +inf
1307 auto *fabsVal = B.CreateFAbs(val);
1308 return B.CreateFCmpOEQ(fabsVal, ConstantFP::getInfinity(val->getType()));
1309}
1310
1311// y * log2(fabs(x))
1313 Value *AbsX = B.CreateFAbs(X);
1314 Value *LogAbsX = B.CreateUnaryIntrinsic(Intrinsic::log2, AbsX);
1315 Value *YTimesLogX = B.CreateFMul(Y, LogAbsX);
1316 return B.CreateUnaryIntrinsic(Intrinsic::exp2, YTimesLogX);
1317}
1318
1319/// Emit special case management epilog code for fast pow, powr, pown, and rootn
1320/// expansions. \p x and \p y should be the arguments to the library call
1321/// (possibly with some values clamped). \p expylnx should be the result to use
1322/// in normal circumstances.
1324 PowKind Kind) {
1325 Constant *Zero = ConstantFP::getZero(X->getType());
1326 Constant *One = ConstantFP::get(X->getType(), 1.0);
1327 Constant *QNaN = ConstantFP::getQNaN(X->getType());
1328 Constant *PInf = ConstantFP::getInfinity(X->getType());
1329
1330 switch (Kind) {
1331 case PowKind::Pow: {
1332 // is_odd_integer(y)
1333 Value *IsOddY = emitIsOddInteger(B, Y);
1334
1335 // ret = copysign(expylnx, is_odd_y ? x : 1.0f)
1336 Value *SelSign = B.CreateSelect(IsOddY, X, One);
1337 Value *Ret = B.CreateCopySign(ExpYLnX, SelSign);
1338
1339 // if (x < 0 && !is_integer(y)) ret = QNAN
1340 Value *IsIntY = emitIsInteger(B, Y);
1341 Value *condNegX = B.CreateFCmpOLT(X, Zero);
1342 Value *condNotIntY = B.CreateNot(IsIntY);
1343 Value *condNaN = B.CreateAnd(condNegX, condNotIntY);
1344 Ret = B.CreateSelect(condNaN, QNaN, Ret);
1345
1346 // if (isinf(ay)) { ... }
1347
1348 // FIXME: Missing backend optimization to save on materialization cost of
1349 // mixed sign constant infinities.
1350 Value *YIsInf = emitIsInf(B, Y);
1351
1352 Value *AY = B.CreateFAbs(Y);
1353 Value *YIsNegInf = B.CreateFCmpUNE(Y, AY);
1354
1355 Value *AX = B.CreateFAbs(X);
1356 Value *AxEqOne = B.CreateFCmpOEQ(AX, One);
1357 Value *AxLtOne = B.CreateFCmpOLT(AX, One);
1358 Value *XorCond = B.CreateXor(AxLtOne, YIsNegInf);
1359 Value *SelInf =
1360 B.CreateSelect(AxEqOne, AX, B.CreateSelect(XorCond, Zero, AY));
1361 Ret = B.CreateSelect(YIsInf, SelInf, Ret);
1362
1363 // if (isinf(ax) || x == 0.0f) { ... }
1364 Value *XIsInf = emitIsInf(B, X);
1365 Value *XEqZero = B.CreateFCmpOEQ(X, Zero);
1366 Value *AxInfOrZero = B.CreateOr(XIsInf, XEqZero);
1367 Value *YLtZero = B.CreateFCmpOLT(Y, Zero);
1368 Value *XorZeroInf = B.CreateXor(XEqZero, YLtZero);
1369 Value *SelVal = B.CreateSelect(XorZeroInf, Zero, PInf);
1370 Value *SelSign2 = B.CreateSelect(IsOddY, X, Zero);
1371 Value *Copysign = B.CreateCopySign(SelVal, SelSign2);
1372 Ret = B.CreateSelect(AxInfOrZero, Copysign, Ret);
1373
1374 // if (isunordered(x, y)) ret = QNAN
1375 Value *isUnordered = B.CreateFCmpUNO(X, Y);
1376 return B.CreateSelect(isUnordered, QNaN, Ret);
1377 }
1378 case PowKind::PowR: {
1379 Value *YIsNeg = B.CreateFCmpOLT(Y, Zero);
1380 Value *IZ = B.CreateSelect(YIsNeg, PInf, Zero);
1381 Value *ZI = B.CreateSelect(YIsNeg, Zero, PInf);
1382
1383 Value *YEqZero = B.CreateFCmpOEQ(Y, Zero);
1384 Value *SelZeroCase = B.CreateSelect(YEqZero, QNaN, IZ);
1385 Value *XEqZero = B.CreateFCmpOEQ(X, Zero);
1386 Value *Ret = B.CreateSelect(XEqZero, SelZeroCase, ExpYLnX);
1387
1388 Value *XEqInf = B.CreateFCmpOEQ(X, PInf);
1389 Value *YNeZero = B.CreateFCmpUNE(Y, Zero);
1390 Value *CondInfCase = B.CreateAnd(XEqInf, YNeZero);
1391 Ret = B.CreateSelect(CondInfCase, ZI, Ret);
1392
1393 Value *IsInfY = emitIsInf(B, Y);
1394 Value *XNeOne = B.CreateFCmpUNE(X, One);
1395 Value *CondInfY = B.CreateAnd(IsInfY, XNeOne);
1396 Value *XLtOne = B.CreateFCmpOLT(X, One);
1397 Value *SelInfYCase = B.CreateSelect(XLtOne, IZ, ZI);
1398 Ret = B.CreateSelect(CondInfY, SelInfYCase, Ret);
1399
1400 Value *IsUnordered = B.CreateFCmpUNO(X, Y);
1401 return B.CreateSelect(IsUnordered, QNaN, Ret);
1402 }
1403 case PowKind::PowN: {
1404 Constant *ZeroI = ConstantInt::get(Y->getType(), 0);
1405
1406 // is_odd_y = (ny & 1) != 0
1407 Value *OneI = ConstantInt::get(Y->getType(), 1);
1408 Value *YAnd1 = B.CreateAnd(Y, OneI);
1409 Value *IsOddY = B.CreateICmpNE(YAnd1, ZeroI);
1410
1411 // ret = copysign(expylnx, is_odd_y ? x : 1.0f)
1412 Value *SelSign = B.CreateSelect(IsOddY, X, One);
1413 Value *Ret = B.CreateCopySign(ExpYLnX, SelSign);
1414
1415 // if (isinf(x) || x == 0.0f)
1416 Value *FabsX = B.CreateFAbs(X);
1417 Value *XIsInf = B.CreateFCmpOEQ(FabsX, PInf);
1418 Value *XEqZero = B.CreateFCmpOEQ(X, Zero);
1419 Value *InfOrZero = B.CreateOr(XIsInf, XEqZero);
1420
1421 // (x == 0.0f) ^ (ny < 0) ? 0.0f : +inf
1422 Value *YLtZero = B.CreateICmpSLT(Y, ZeroI);
1423 Value *XorZeroInf = B.CreateXor(XEqZero, YLtZero);
1424 Value *SelVal = B.CreateSelect(XorZeroInf, Zero, PInf);
1425
1426 // copysign(selVal, is_odd_y ? x : 0.0f)
1427 Value *SelSign2 = B.CreateSelect(IsOddY, X, Zero);
1428 Value *Copysign = B.CreateCopySign(SelVal, SelSign2);
1429
1430 return B.CreateSelect(InfOrZero, Copysign, Ret);
1431 }
1432 case PowKind::RootN: {
1433 Constant *ZeroI = ConstantInt::get(Y->getType(), 0);
1434
1435 // is_odd_y = (ny & 1) != 0
1436 Value *YAnd1 = B.CreateAnd(Y, ConstantInt::get(Y->getType(), 1));
1437 Value *IsOddY = B.CreateICmpNE(YAnd1, ZeroI);
1438
1439 // ret = copysign(expylnx, is_odd_y ? x : 1.0f)
1440 Value *SelSign = B.CreateSelect(IsOddY, X, One);
1441 Value *Ret = B.CreateCopySign(ExpYLnX, SelSign);
1442
1443 // if (isinf(x) || x == 0.0f)
1444 Value *FabsX = B.CreateFAbs(X);
1445 Value *IsInfX = B.CreateFCmpOEQ(FabsX, PInf);
1446 Value *XEqZero = B.CreateFCmpOEQ(X, Zero);
1447 Value *CondInfOrZero = B.CreateOr(IsInfX, XEqZero);
1448
1449 // (x == 0.0f) ^ (ny < 0) ? 0.0f : +inf
1450 Value *YLtZero = B.CreateICmpSLT(Y, ZeroI);
1451 Value *XorZeroInf = B.CreateXor(XEqZero, YLtZero);
1452 Value *SelVal = B.CreateSelect(XorZeroInf, Zero, PInf);
1453
1454 // copysign(selVal, is_odd_y ? x : 0.0f)
1455 Value *SelSign2 = B.CreateSelect(IsOddY, X, Zero);
1456 Value *Copysign = B.CreateCopySign(SelVal, SelSign2);
1457
1458 Ret = B.CreateSelect(CondInfOrZero, Copysign, Ret);
1459
1460 // if ((x < 0.0f && !is_odd_y) || ny == 0) ret = QNAN
1461 Value *XIsNeg = B.CreateFCmpOLT(X, Zero);
1462 Value *NotOddY = B.CreateNot(IsOddY);
1463 Value *CondNegAndNotOdd = B.CreateAnd(XIsNeg, NotOddY);
1464 Value *YEqZero = B.CreateICmpEQ(Y, ZeroI);
1465 Value *CondBad = B.CreateOr(CondNegAndNotOdd, YEqZero);
1466 return B.CreateSelect(CondBad, QNaN, Ret);
1467 }
1468 }
1469
1470 llvm_unreachable("covered switch");
1471}
1472
1473// TODO: Move the fold_pow folding to sqrt/fdiv here
1474bool AMDGPULibCalls::expandFastPow(FPMathOperator *FPOp, IRBuilder<> &B,
1475 PowKind Kind) {
1476 Type *Ty = FPOp->getType();
1477
1478 // There's currently no reason to do this for half. The correct path is
1479 // promote to float and use the fast float expansion.
1480 //
1481 // TODO: We could move this expansion to lowering to get half pow to work.
1482 if (!Ty->getScalarType()->isFloatTy())
1483 return false;
1484
1485 // TODO: Verify optimization for double and bfloat.
1486 Value *X = FPOp->getOperand(0);
1487 Value *Y = FPOp->getOperand(1);
1488
1489 switch (Kind) {
1490 case PowKind::Pow: {
1491 Constant *One = ConstantFP::get(X->getType(), 1.0);
1492
1493 // if (x == 1.0f) y = 1.0f;
1494 Value *XEqOne = B.CreateFCmpOEQ(X, One);
1495 Y = B.CreateSelect(XEqOne, One, Y);
1496
1497 // if (y == 0.0f) x = 1.0f;
1498 Value *YEqZero = B.CreateFCmpOEQ(Y, ConstantFP::getZero(X->getType()));
1499 X = B.CreateSelect(YEqZero, One, X);
1500
1501 Value *ExpYLnX = emitFastExpYLnx(B, X, Y);
1502 Value *Fixed = emitPowFixup(B, X, Y, ExpYLnX, Kind);
1503 replaceCall(FPOp, Fixed);
1504 return true;
1505 }
1506 case PowKind::PowR: {
1507 Value *NegX = B.CreateFCmpOLT(X, ConstantFP::getZero(X->getType()));
1508 X = B.CreateSelect(NegX, ConstantFP::getQNaN(X->getType()), X);
1509
1510 Value *ExpYLnX = emitFastExpYLnx(B, X, Y);
1511 Value *Fixed = emitPowFixup(B, X, Y, ExpYLnX, Kind);
1512 replaceCall(FPOp, Fixed);
1513 return true;
1514 }
1515 case PowKind::PowN: {
1516 // ny == 0
1517 Value *YEqZero = B.CreateICmpEQ(Y, ConstantInt::get(Y->getType(), 0));
1518
1519 // x = (ny == 0 ? 1.0f : x)
1520 X = B.CreateSelect(YEqZero, ConstantFP::get(X->getType(), 1.0), X);
1521
1522 Value *CastY = B.CreateSIToFP(Y, X->getType());
1523 Value *ExpYLnX = emitFastExpYLnx(B, X, CastY);
1524 Value *Fixed = emitPowFixup(B, X, Y, ExpYLnX, Kind);
1525 replaceCall(FPOp, Fixed);
1526 return true;
1527 }
1528 case PowKind::RootN: {
1529 Value *CastY = B.CreateSIToFP(Y, X->getType());
1530
1531 // This is afn anyway, so we will turn into rcp.
1532 Value *RcpY = B.CreateFDiv(ConstantFP::get(X->getType(), 1.0), CastY);
1533
1534 Value *ExpYLnX = emitFastExpYLnx(B, X, RcpY);
1535 Value *Fixed = emitPowFixup(B, X, Y, ExpYLnX, Kind);
1536 replaceCall(FPOp, Fixed);
1537 return true;
1538 }
1539 }
1540 llvm_unreachable("Unhandled PowKind enum");
1541}
1542
1543bool AMDGPULibCalls::tryOptimizePow(FPMathOperator *FPOp, IRBuilder<> &B,
1544 const FuncInfo &FInfo) {
1545 FastMathFlags FMF = FPOp->getFastMathFlags();
1546 CallInst *Call = cast<CallInst>(FPOp);
1547 Module *M = Call->getModule();
1548
1549 FuncInfo PowrInfo;
1550 AMDGPULibFunc::EFuncId FastPowrFuncId =
1551 FMF.approxFunc() || FInfo.getId() == AMDGPULibFunc::EI_POW_FAST
1554 FunctionCallee PowrFunc = getFloatFastVariant(
1555 M, FInfo, PowrInfo, AMDGPULibFunc::EI_POWR, FastPowrFuncId);
1556
1557 // TODO: Prefer fast pown to fast powr, but slow powr to slow pown.
1558
1559 // pow(x, y) -> powr(x, y) for x >= -0.0
1560 // TODO: Account for flags on current call
1561 if (PowrFunc && cannotBeOrderedLessThanZero(FPOp->getOperand(0),
1562 SQ.getWithInstruction(Call))) {
1563 Call->setCalledFunction(PowrFunc);
1564 return fold_pow(FPOp, B, PowrInfo) || true;
1565 }
1566
1567 // pow(x, y) -> pown(x, y) for known integral y
1568 if (isKnownIntegral(FPOp->getOperand(1), SQ.getWithInstruction(Call),
1569 FPOp->getFastMathFlags())) {
1570 FunctionType *PownType = getPownType(Call->getFunctionType());
1571
1572 FuncInfo PownInfo;
1573 AMDGPULibFunc::EFuncId FastPownFuncId =
1574 FMF.approxFunc() || FInfo.getId() == AMDGPULibFunc::EI_POW_FAST
1577 FunctionCallee PownFunc = getFloatFastVariant(
1578 M, FInfo, PownInfo, AMDGPULibFunc::EI_POWN, FastPownFuncId);
1579
1580 if (PownFunc) {
1581 // TODO: If the incoming integral value is an sitofp/uitofp, it won't
1582 // fold out without a known range. We can probably take the source
1583 // value directly.
1584 Value *CastedArg =
1585 B.CreateFPToSI(FPOp->getOperand(1), PownType->getParamType(1));
1586 // Have to drop any nofpclass attributes on the original call site.
1588 1, AttributeFuncs::typeIncompatible(CastedArg->getType(),
1590 Call->setCalledFunction(PownFunc);
1591 Call->setArgOperand(1, CastedArg);
1592 return fold_pow(FPOp, B, PownInfo) || true;
1593 }
1594 }
1595
1596 if (fold_pow(FPOp, B, FInfo))
1597 return true;
1598
1599 if (!FMF.approxFunc())
1600 return false;
1601
1602 if (FInfo.getId() == AMDGPULibFunc::EI_POW && FMF.approxFunc() &&
1603 getArgType(FInfo) == AMDGPULibFunc::F32) {
1604 AMDGPULibFunc PowFastInfo(AMDGPULibFunc::EI_POW_FAST, FInfo);
1605 if (FunctionCallee PowFastFunc = getFunction(M, PowFastInfo)) {
1606 Call->setCalledFunction(PowFastFunc);
1607 return fold_pow(FPOp, B, PowFastInfo) || true;
1608 }
1609 }
1610
1611 return expandFastPow(FPOp, B, PowKind::Pow);
1612}
1613
1614// Get a scalar native builtin single argument FP function
1615FunctionCallee AMDGPULibCalls::getNativeFunction(Module *M,
1616 const FuncInfo &FInfo) {
1617 if (getArgType(FInfo) == AMDGPULibFunc::F64 || !HasNative(FInfo.getId()))
1618 return nullptr;
1619 FuncInfo nf = FInfo;
1621 return getFunction(M, nf);
1622}
1623
1624// Some library calls are just wrappers around llvm intrinsics, but compiled
1625// conservatively. Preserve the flags from the original call site by
1626// substituting them with direct calls with all the flags.
1627bool AMDGPULibCalls::shouldReplaceLibcallWithIntrinsic(const CallInst *CI,
1628 bool AllowMinSizeF32,
1629 bool AllowF64,
1630 bool AllowStrictFP) {
1631 Type *FltTy = CI->getType()->getScalarType();
1632 const bool IsF32 = FltTy->isFloatTy();
1633
1634 // f64 intrinsics aren't implemented for most operations.
1635 if (!IsF32 && !FltTy->isHalfTy() && (!AllowF64 || !FltTy->isDoubleTy()))
1636 return false;
1637
1638 // We're implicitly inlining by replacing the libcall with the intrinsic, so
1639 // don't do it for noinline call sites.
1640 if (CI->isNoInline())
1641 return false;
1642
1643 const Function *ParentF = CI->getFunction();
1644 // TODO: Handle strictfp
1645 if (!AllowStrictFP && ParentF->hasFnAttribute(Attribute::StrictFP))
1646 return false;
1647
1648 if (IsF32 && !AllowMinSizeF32 && ParentF->hasMinSize())
1649 return false;
1650 return true;
1651}
1652
1653void AMDGPULibCalls::replaceLibCallWithSimpleIntrinsic(IRBuilder<> &B,
1654 CallInst *CI,
1655 Intrinsic::ID IntrID) {
1656 if (CI->arg_size() == 2) {
1657 Value *Arg0 = CI->getArgOperand(0);
1658 Value *Arg1 = CI->getArgOperand(1);
1659 VectorType *Arg0VecTy = dyn_cast<VectorType>(Arg0->getType());
1660 VectorType *Arg1VecTy = dyn_cast<VectorType>(Arg1->getType());
1661 if (Arg0VecTy && !Arg1VecTy) {
1662 Value *SplatRHS = B.CreateVectorSplat(Arg0VecTy->getElementCount(), Arg1);
1663 CI->setArgOperand(1, SplatRHS);
1664 } else if (!Arg0VecTy && Arg1VecTy) {
1665 Value *SplatLHS = B.CreateVectorSplat(Arg1VecTy->getElementCount(), Arg0);
1666 CI->setArgOperand(0, SplatLHS);
1667 }
1668 }
1669
1671 CI->getModule(), IntrID, {CI->getType()}));
1673}
1674
1675bool AMDGPULibCalls::tryReplaceLibcallWithSimpleIntrinsic(
1676 IRBuilder<> &B, CallInst *CI, Intrinsic::ID IntrID, bool AllowMinSizeF32,
1677 bool AllowF64, bool AllowStrictFP) {
1678 if (!shouldReplaceLibcallWithIntrinsic(CI, AllowMinSizeF32, AllowF64,
1679 AllowStrictFP))
1680 return false;
1681 replaceLibCallWithSimpleIntrinsic(B, CI, IntrID);
1682 return true;
1683}
1684
1685std::tuple<Value *, Value *, Value *>
1686AMDGPULibCalls::insertSinCos(Value *Arg, FastMathFlags FMF, IRBuilder<> &B,
1687 FunctionCallee Fsincos) {
1688 DebugLoc DL = B.getCurrentDebugLocation();
1689 Function *F = B.GetInsertBlock()->getParent();
1690 B.SetInsertPointPastAllocas(F);
1691
1692 AllocaInst *Alloc = B.CreateAlloca(Arg->getType(), nullptr, "__sincos_");
1693
1694 if (Instruction *ArgInst = dyn_cast<Instruction>(Arg)) {
1695 // If the argument is an instruction, it must dominate all uses so put our
1696 // sincos call there. Otherwise, right after the allocas works well enough
1697 // if it's an argument or constant.
1698
1699 B.SetInsertPoint(ArgInst->getParent(), ++ArgInst->getIterator());
1700
1701 // SetInsertPoint unwelcomely always tries to set the debug loc.
1702 B.SetCurrentDebugLocation(DL);
1703 }
1704
1705 Type *CosPtrTy = Fsincos.getFunctionType()->getParamType(1);
1706
1707 // The allocaInst allocates the memory in private address space. This need
1708 // to be addrspacecasted to point to the address space of cos pointer type.
1709 // In OpenCL 2.0 this is generic, while in 1.2 that is private.
1710 Value *CastAlloc = B.CreateAddrSpaceCast(Alloc, CosPtrTy);
1711
1712 CallInst *SinCos = CreateCallEx2(B, Fsincos, Arg, CastAlloc);
1713
1714 // TODO: Is it worth trying to preserve the location for the cos calls for the
1715 // load?
1716
1717 LoadInst *LoadCos = B.CreateLoad(Arg->getType(), Alloc);
1718 return {SinCos, LoadCos, SinCos};
1719}
1720
1721// fold sin, cos -> sincos.
1722bool AMDGPULibCalls::fold_sincos(FPMathOperator *FPOp, IRBuilder<> &B,
1723 const FuncInfo &fInfo) {
1724 assert(fInfo.getId() == AMDGPULibFunc::EI_SIN ||
1725 fInfo.getId() == AMDGPULibFunc::EI_COS);
1726
1727 if ((getArgType(fInfo) != AMDGPULibFunc::F32 &&
1728 getArgType(fInfo) != AMDGPULibFunc::F64) ||
1729 fInfo.getPrefix() != AMDGPULibFunc::NOPFX)
1730 return false;
1731
1732 bool const isSin = fInfo.getId() == AMDGPULibFunc::EI_SIN;
1733
1734 Value *CArgVal = FPOp->getOperand(0);
1735
1736 // TODO: Constant fold the call
1737 if (isa<ConstantData>(CArgVal))
1738 return false;
1739
1740 CallInst *CI = cast<CallInst>(FPOp);
1741
1742 Function *F = B.GetInsertBlock()->getParent();
1743 Module *M = F->getParent();
1744
1745 // Merge the sin and cos. For OpenCL 2.0, there may only be a generic pointer
1746 // implementation. Prefer the private form if available.
1747 AMDGPULibFunc SinCosLibFuncPrivate(AMDGPULibFunc::EI_SINCOS, fInfo);
1748 SinCosLibFuncPrivate.getLeads()[0].PtrKind =
1750
1751 AMDGPULibFunc SinCosLibFuncGeneric(AMDGPULibFunc::EI_SINCOS, fInfo);
1752 SinCosLibFuncGeneric.getLeads()[0].PtrKind =
1754
1755 FunctionCallee FSinCosPrivate = getFunction(M, SinCosLibFuncPrivate);
1756 FunctionCallee FSinCosGeneric = getFunction(M, SinCosLibFuncGeneric);
1757 FunctionCallee FSinCos = FSinCosPrivate ? FSinCosPrivate : FSinCosGeneric;
1758 if (!FSinCos)
1759 return false;
1760
1761 SmallVector<CallInst *> SinCalls;
1762 SmallVector<CallInst *> CosCalls;
1763 SmallVector<CallInst *> SinCosCalls;
1764 FuncInfo PartnerInfo(isSin ? AMDGPULibFunc::EI_COS : AMDGPULibFunc::EI_SIN,
1765 fInfo);
1766 const std::string PairName = PartnerInfo.mangle();
1767
1768 StringRef SinName = isSin ? CI->getCalledFunction()->getName() : PairName;
1769 StringRef CosName = isSin ? PairName : CI->getCalledFunction()->getName();
1770 const std::string SinCosPrivateName = SinCosLibFuncPrivate.mangle();
1771 const std::string SinCosGenericName = SinCosLibFuncGeneric.mangle();
1772
1773 // Intersect the two sets of flags.
1774 FastMathFlags FMF = FPOp->getFastMathFlags();
1775 MDNode *FPMath = CI->getMetadata(LLVMContext::MD_fpmath);
1776
1777 SmallVector<DILocation *> MergeDbgLocs = {CI->getDebugLoc()};
1778
1779 for (User* U : CArgVal->users()) {
1780 CallInst *XI = dyn_cast<CallInst>(U);
1781 if (!XI || XI->getFunction() != F || XI->isNoBuiltin())
1782 continue;
1783
1784 Function *UCallee = XI->getCalledFunction();
1785 if (!UCallee)
1786 continue;
1787
1788 bool Handled = true;
1789
1790 if (UCallee->getName() == SinName)
1791 SinCalls.push_back(XI);
1792 else if (UCallee->getName() == CosName)
1793 CosCalls.push_back(XI);
1794 else if (UCallee->getName() == SinCosPrivateName ||
1795 UCallee->getName() == SinCosGenericName)
1796 SinCosCalls.push_back(XI);
1797 else
1798 Handled = false;
1799
1800 if (Handled) {
1801 MergeDbgLocs.push_back(XI->getDebugLoc());
1802 auto *OtherOp = cast<FPMathOperator>(XI);
1803 FMF &= OtherOp->getFastMathFlags();
1805 FPMath, XI->getMetadata(LLVMContext::MD_fpmath));
1806 }
1807 }
1808
1809 if (SinCalls.empty() || CosCalls.empty())
1810 return false;
1811
1812 B.setFastMathFlags(FMF);
1813 B.setDefaultFPMathTag(FPMath);
1814 DILocation *DbgLoc = DILocation::getMergedLocations(MergeDbgLocs);
1815 B.SetCurrentDebugLocation(DbgLoc);
1816
1817 auto [Sin, Cos, SinCos] = insertSinCos(CArgVal, FMF, B, FSinCos);
1818
1819 auto replaceTrigInsts = [](ArrayRef<CallInst *> Calls, Value *Res) {
1820 for (CallInst *C : Calls)
1821 C->replaceAllUsesWith(Res);
1822
1823 // Leave the other dead instructions to avoid clobbering iterators.
1824 };
1825
1826 replaceTrigInsts(SinCalls, Sin);
1827 replaceTrigInsts(CosCalls, Cos);
1828 replaceTrigInsts(SinCosCalls, SinCos);
1829
1830 // It's safe to delete the original now.
1831 CI->eraseFromParent();
1832 return true;
1833}
1834
1835bool AMDGPULibCalls::evaluateScalarMathFunc(const FuncInfo &FInfo,
1836 APFloat &Res0, APFloat &Res1,
1837 Constant *copr0, Constant *copr1) {
1838 // Every function handled below reads its first operand as a floating-point
1839 // value. Refuse anything else, e.g. a poison vector lane: silently treating
1840 // it as 0.0 misfolds the whole call.
1842 if (!fpopr0)
1843 return false;
1844
1845 double opr0 = (getArgType(FInfo) == AMDGPULibFunc::F64)
1846 ? fpopr0->getValueAPF().convertToDouble()
1847 : (double)fpopr0->getValueAPF().convertToFloat();
1848
1849 switch (FInfo.getId()) {
1850 default:
1851 return false;
1852
1854 Res0 = APFloat{acos(opr0)};
1855 return true;
1856
1858 // acosh(x) == log(x + sqrt(x*x - 1))
1859 Res0 = APFloat{log(opr0 + sqrt(opr0 * opr0 - 1.0))};
1860 return true;
1861
1863 Res0 = APFloat{acos(opr0) / MATH_PI};
1864 return true;
1865
1867 Res0 = APFloat{asin(opr0)};
1868 return true;
1869
1871 // asinh(x) == log(x + sqrt(x*x + 1))
1872 Res0 = APFloat{log(opr0 + sqrt(opr0 * opr0 + 1.0))};
1873 return true;
1874
1876 Res0 = APFloat{asin(opr0) / MATH_PI};
1877 return true;
1878
1880 Res0 = APFloat{atan(opr0)};
1881 return true;
1882
1884 // atanh(x) == (log(x+1) - log(x-1))/2;
1885 Res0 = APFloat{(log(opr0 + 1.0) - log(opr0 - 1.0)) / 2.0};
1886 return true;
1887
1889 Res0 = APFloat{atan(opr0) / MATH_PI};
1890 return true;
1891
1893 Res0 =
1894 APFloat{(opr0 < 0.0) ? -pow(-opr0, 1.0 / 3.0) : pow(opr0, 1.0 / 3.0)};
1895 return true;
1896
1898 Res0 = APFloat{cos(opr0)};
1899 return true;
1900
1902 Res0 = APFloat{cosh(opr0)};
1903 return true;
1904
1906 Res0 = APFloat{cos(MATH_PI * opr0)};
1907 return true;
1908
1910 Res0 = APFloat{std::exp(opr0)};
1911 return true;
1912
1914 Res0 = APFloat{pow(2.0, opr0)};
1915 return true;
1916
1918 Res0 = APFloat{pow(10.0, opr0)};
1919 return true;
1920
1922 Res0 = APFloat{log(opr0)};
1923 return true;
1924
1926 Res0 = APFloat{log(opr0) / log(2.0)};
1927 return true;
1928
1930 Res0 = APFloat{log(opr0) / log(10.0)};
1931 return true;
1932
1934 Res0 = APFloat{1.0 / sqrt(opr0)};
1935 return true;
1936
1938 Res0 = APFloat{sin(opr0)};
1939 return true;
1940
1942 Res0 = APFloat{sinh(opr0)};
1943 return true;
1944
1946 Res0 = APFloat{sin(MATH_PI * opr0)};
1947 return true;
1948
1950 Res0 = APFloat{tan(opr0)};
1951 return true;
1952
1954 Res0 = APFloat{tanh(opr0)};
1955 return true;
1956
1958 Res0 = APFloat{tan(MATH_PI * opr0)};
1959 return true;
1960
1961 // two-arg functions
1965 if (!fpopr1)
1966 return false;
1967 double opr1 = (getArgType(FInfo) == AMDGPULibFunc::F64)
1968 ? fpopr1->getValueAPF().convertToDouble()
1969 : (double)fpopr1->getValueAPF().convertToFloat();
1970 Res0 = APFloat{pow(opr0, opr1)};
1971 return true;
1972 }
1973
1975 if (ConstantInt *iopr1 = dyn_cast_or_null<ConstantInt>(copr1)) {
1976 double val = (double)iopr1->getSExtValue();
1977 Res0 = APFloat{pow(opr0, val)};
1978 return true;
1979 }
1980 return false;
1981 }
1982
1984 if (ConstantInt *iopr1 = dyn_cast_or_null<ConstantInt>(copr1)) {
1985 double val = (double)iopr1->getSExtValue();
1986 Res0 = APFloat{pow(opr0, 1.0 / val)};
1987 return true;
1988 }
1989 return false;
1990 }
1991
1992 // with ptr arg
1994 Res0 = APFloat{sin(opr0)};
1995 Res1 = APFloat{cos(opr0)};
1996 return true;
1997 }
1998
1999 return false;
2000}
2001
2002bool AMDGPULibCalls::evaluateCall(CallInst *aCI, const FuncInfo &FInfo) {
2003 int numArgs = (int)aCI->arg_size();
2004 if (numArgs > 3)
2005 return false;
2006
2007 Constant *copr0 = nullptr;
2008 Constant *copr1 = nullptr;
2009 if (numArgs > 0) {
2010 if ((copr0 = dyn_cast<Constant>(aCI->getArgOperand(0))) == nullptr)
2011 return false;
2012 }
2013
2014 if (numArgs > 1) {
2015 if ((copr1 = dyn_cast<Constant>(aCI->getArgOperand(1))) == nullptr) {
2016 if (FInfo.getId() != AMDGPULibFunc::EI_SINCOS)
2017 return false;
2018 }
2019 }
2020
2021 // At this point, all arguments to aCI are constants.
2022
2023 // max vector size is 16, and sincos will generate two results.
2024 SmallVector<APFloat, 16> Val0, Val1;
2025 int FuncVecSize = getVecSize(FInfo);
2026 if (FuncVecSize == 1) {
2027 if (!evaluateScalarMathFunc(FInfo, Val0.emplace_back(0.0),
2028 Val1.emplace_back(0.0), copr0, copr1)) {
2029 return false;
2030 }
2031 } else {
2032 // An operand of a vector variant is not necessarily a vector: sincos takes
2033 // a pointer as its second operand, and fmin/fmax/ldexp accept an
2034 // implicitly splatted scalar. Only index into actual vectors.
2035 Constant *CV0 = copr0 && copr0->getType()->isVectorTy() ? copr0 : nullptr;
2036 Constant *CV1 = copr1 && copr1->getType()->isVectorTy() ? copr1 : nullptr;
2037 for (int i = 0; i < FuncVecSize; ++i) {
2038 Constant *celt0 = CV0 ? CV0->getAggregateElement((unsigned)i) : nullptr;
2039 Constant *celt1 = CV1 ? CV1->getAggregateElement((unsigned)i) : nullptr;
2040 if (!evaluateScalarMathFunc(FInfo, Val0.emplace_back(0.0),
2041 Val1.emplace_back(0.0), celt0, celt1)) {
2042 return false;
2043 }
2044 }
2045 }
2046
2047 Constant *nval0 = getConstantFloat(Val0, aCI->getType());
2048
2049 // sincos
2050 if (FInfo.getId() == AMDGPULibFunc::EI_SINCOS) {
2051 Constant *nval1 = getConstantFloat(Val1, aCI->getType());
2052 new StoreInst(nval1, aCI->getArgOperand(1), aCI->getIterator());
2053 }
2054
2055 replaceCall(aCI, nval0);
2056 return true;
2057}
2058
2061 AMDGPULibCalls Simplifier(F, AM);
2062 Simplifier.initNativeFuncs();
2063
2064 bool Changed = false;
2065
2066 LLVM_DEBUG(dbgs() << "AMDIC: process function ";
2067 F.printAsOperand(dbgs(), false, F.getParent()); dbgs() << '\n';);
2068
2069 for (auto &BB : F) {
2070 for (BasicBlock::iterator I = BB.begin(), E = BB.end(); I != E;) {
2071 // Ignore non-calls.
2073 ++I;
2074
2075 if (CI) {
2076 if (Simplifier.fold(CI))
2077 Changed = true;
2078 }
2079 }
2080 }
2082}
2083
2086 if (UseNative.empty())
2087 return PreservedAnalyses::all();
2088
2089 AMDGPULibCalls Simplifier(F, AM);
2090 Simplifier.initNativeFuncs();
2091
2092 bool Changed = false;
2093 for (auto &BB : F) {
2094 for (BasicBlock::iterator I = BB.begin(), E = BB.end(); I != E;) {
2095 // Ignore non-calls.
2097 ++I;
2098 if (CI && Simplifier.useNative(CI))
2099 Changed = true;
2100 }
2101 }
2103}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static const TableEntry tbl_log[]
static const TableEntry tbl_tgamma[]
static AMDGPULibFunc::EType getArgType(const AMDGPULibFunc &FInfo)
static const TableEntry tbl_expm1[]
static const TableEntry tbl_asinpi[]
static const TableEntry tbl_cos[]
#define MATH_SQRT2
static const TableEntry tbl_exp10[]
static CallInst * CreateCallEx(IRB &B, FunctionCallee Callee, Value *Arg, const Twine &Name="")
static CallInst * CreateCallEx2(IRB &B, FunctionCallee Callee, Value *Arg1, Value *Arg2, const Twine &Name="")
static const TableEntry tbl_rsqrt[]
static const TableEntry tbl_atanh[]
static const TableEntry tbl_cosh[]
static const TableEntry tbl_asin[]
static const TableEntry tbl_sinh[]
static const TableEntry tbl_acos[]
static const TableEntry tbl_tan[]
static const TableEntry tbl_cospi[]
static const TableEntry tbl_tanpi[]
static cl::opt< bool > EnablePreLink("amdgpu-prelink", cl::desc("Enable pre-link mode optimizations"), cl::init(false), cl::Hidden)
static bool HasNative(AMDGPULibFunc::EFuncId id)
static Value * emitIsInf(IRBuilder<> &B, Value *val)
ArrayRef< TableEntry > TableRef
static int getVecSize(const AMDGPULibFunc &FInfo)
static Value * emitFastExpYLnx(IRBuilder<> &B, Value *X, Value *Y)
static Value * emitIsInteger(IRBuilder<> &B, Value *Y)
static Value * emitIsEvenInteger(IRBuilder<> &B, Value *Y)
static const TableEntry tbl_sin[]
static const TableEntry tbl_atan[]
static const TableEntry tbl_log2[]
static Constant * getConstantFloat(const ArrayRef< APFloat > Values, const Type *Ty)
static const TableEntry tbl_acospi[]
static Value * emitPowFixup(IRBuilder<> &B, Value *X, Value *Y, Value *ExpYLnX, PowKind Kind)
Emit special case management epilog code for fast pow, powr, pown, and rootn expansions.
static const TableEntry tbl_sqrt[]
static const TableEntry tbl_asinh[]
#define MATH_E
static TableRef getOptTable(AMDGPULibFunc::EFuncId id)
static const TableEntry tbl_acosh[]
static const TableEntry tbl_exp[]
static const TableEntry tbl_cbrt[]
static const TableEntry tbl_sinpi[]
static const TableEntry tbl_atanpi[]
#define MATH_PI
static FunctionType * getPownType(FunctionType *FT)
static const TableEntry tbl_erf[]
static const TableEntry tbl_log10[]
#define MATH_SQRT1_2
static const TableEntry tbl_erfc[]
static cl::list< std::string > UseNative("amdgpu-use-native", cl::desc("Comma separated list of functions to replace with native, or all"), cl::CommaSeparated, cl::ValueOptional, cl::Hidden)
static const TableEntry tbl_tanh[]
static Value * emitIsOddInteger(IRBuilder<> &B, Value *Y)
static const TableEntry tbl_exp2[]
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static const Function * getParent(const Value *V)
#define X(NUM, ENUM, NAME)
Definition ELF.h:856
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
loop term fold
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
FunctionAnalysisManager FAM
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static Function * getFunction(FunctionType *Ty, const Twine &Name, Module *M)
static void replaceCall(FPMathOperator *I, Value *With)
bool isUnsafeFiniteOnlyMath(const FPMathOperator *FPOp) const
bool canIncreasePrecisionOfConstantFold(const FPMathOperator *FPOp) const
bool fold(CallInst *CI)
static void replaceCall(Instruction *I, Value *With)
AMDGPULibCalls(Function &F, FunctionAnalysisManager &FAM)
bool useNative(CallInst *CI)
static unsigned getEPtrKindFromAddrSpace(unsigned AS)
Wrapper class for AMDGPULIbFuncImpl.
static bool parse(StringRef MangledName, AMDGPULibFunc &Ptr)
std::string getName() const
Get unmangled name for mangled library function and name for unmangled library function.
static FunctionCallee getOrInsertFunction(llvm::Module *M, const AMDGPULibFunc &fInfo)
void setPrefix(ENamePrefix PFX)
bool isCompatibleSignature(const Module &M, const FunctionType *FuncTy) const
EFuncId getId() const
bool isMangled() const
Param * getLeads()
Get leading parameters for mangled lib functions.
void setId(EFuncId Id)
ENamePrefix getPrefix() const
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:353
bool isNegative() const
Definition APFloat.h:1575
LLVM_ABI double convertToDouble() const
Converts this APFloat to host double value.
Definition APFloat.cpp:6005
bool isExactlyValue(double V) const
We don't rely on operator== working on double values, as it returns true for things that are clearly ...
Definition APFloat.h:1558
LLVM_ABI float convertToFloat() const
Converts this APFloat to host float value.
Definition APFloat.cpp:6033
bool isZero() const
Definition APFloat.h:1571
LLVM_READONLY bool isOne() const
Definition APFloat.h:1653
LLVM_READONLY bool isMinusOne() const
Definition APFloat.h:1656
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1583
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
iterator end() const
Definition ArrayRef.h:130
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
A function analysis which provides an AssumptionCache.
static LLVM_ABI Attribute getWithNoFPClass(LLVMContext &Context, FPClassTest Mask)
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
void setCallingConv(CallingConv::ID CC)
void removeParamAttrs(unsigned ArgNo, const AttributeMask &AttrsToRemove)
Removes the attributes from the given argument.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
bool isStrictFP() const
Determine if the call requires strict floating point semantics.
AttributeSet getParamAttributes(unsigned ArgNo) const
Return the param attributes for this call.
bool isNoInline() const
Return true if the call should not be inlined.
void addRetAttr(Attribute::AttrKind Kind)
Adds the attribute to the return value.
Value * getArgOperand(unsigned i) const
void setArgOperand(unsigned i, Value *v)
FunctionType * getFunctionType() const
unsigned arg_size() const
AttributeList getAttributes() const
Return the attributes for this call.
void setCalledFunction(Function *Fn)
Sets the function called, including updating the function type.
This class represents a function call, abstracting a target machine's calling convention.
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
LLVM_ABI APFloat getElementAsAPFloat(uint64_t i) const
If this is a sequential container of floating point type, return the specified element as an APFloat.
LLVM_ABI uint64_t getNumElements() const
Return the number of elements in the array or vector.
static LLVM_ABI Constant * getSplat(unsigned NumElts, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(LLVMContext &Context, ArrayRef< uint8_t > Elts)
get() constructors - Return a constant with vector type with an element count and element type matchi...
const APFloat & getValueAPF() const
Definition Constants.h:463
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getQNaN(Type *Ty, bool Negative=false, APInt *Payload=nullptr)
LLVM_ABI bool isExactlyValue(const APFloat &V) const
We don't rely on operator== working on double values, as it returns true for things that are clearly ...
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
Align getAlignValue() const
Return the constant as an llvm::Align, interpreting 0 as Align(1).
Definition Constants.h:186
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
LLVM_ABI Constant * getAggregateElement(unsigned Elt) const
For aggregates (struct/array/vector) return the constant that corresponds to the specified element if...
static LLVM_ABI DILocation * getMergedLocations(ArrayRef< DILocation * > Locs)
Try to combine the vector of locations passed as input in a single one.
Analysis pass which computes a DominatorTree.
Definition Dominators.h:241
Utility class for floating point operations which can have information about relaxed accuracy require...
Definition Operator.h:202
bool isFast() const
Test if this operation allows all non-strict floating-point transforms.
Definition Operator.h:264
bool hasNoNaNs() const
Test if this operation's arguments and results are assumed not-NaN.
Definition Operator.h:270
FastMathFlags getFastMathFlags() const
Convenience function for getting all the fast-math flags.
Definition Operator.h:291
bool hasNoSignedZeros() const
Test if this operation can ignore the sign of zero.
Definition Operator.h:276
bool hasNoInfs() const
Test if this operation's arguments and results are assumed not-infinite.
Definition Operator.h:273
bool hasApproxFunc() const
Test if this operation allows approximations of math library functions or intrinsics.
Definition Operator.h:288
LLVM_ABI float getFPAccuracy() const
Get the maximum error permitted by this operation in ULPs.
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
void setAllowContract(bool B=true)
Definition FMF.h:90
bool none() const
Definition FMF.h:57
bool approxFunc() const
Definition FMF.h:70
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
FunctionType * getFunctionType()
Type * getParamType(unsigned i) const
Parameter type accessors.
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
bool hasMinSize() const
Optimize this function for minimum size (-Oz).
Definition Function.h:695
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:727
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2893
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setFastMathFlags(FastMathFlags FMF)
Convenience function for setting multiple fast-math flags on this instruction, which must be an opera...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
static LLVM_ABI MDNode * getMostGenericFPMath(MDNode *A, MDNode *B)
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses none()
Convenience factory function for the empty preserved set.
Definition Analysis.h:115
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Analysis pass providing the TargetLibraryInfo.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:288
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
Definition Type.h:155
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:197
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
Definition Type.h:158
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
void dropAllReferences()
Drop all references to operands.
Definition User.h:324
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
iterator_range< user_iterator > users()
Definition Value.h:426
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
self_iterator getIterator()
Definition ilist_node.h:123
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ FLAT_ADDRESS
Address space for flat memory.
@ PRIVATE_ADDRESS
Address space for private memory.
LLVM_ABI APInt pow(const APInt &X, int64_t N)
Compute X^N for N>=0.
Definition APInt.cpp:3187
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
ap_match< APInt > m_APIntAllowPoison(const APInt *&Res)
Match APInt while allowing poison in splat vector constants.
bool match(Val *V, const Pattern &P)
ap_match< APFloat > m_APFloatAllowPoison(const APFloat *&Res)
Match APFloat while allowing poison in splat vector constants.
initializer< Ty > init(const Ty &Val)
constexpr double ln2
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
static double log2(double V)
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1669
RelativeUniformCounterPtr Values
Definition InstrProf.h:91
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1772
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1947
LLVM_ABI bool isKnownIntegral(const Value *V, const SimplifyQuery &SQ, FastMathFlags FMF)
Return true if the floating-point value V is known to be an integer value.
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI bool cannotBeOrderedLessThanZero(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Return true if we can prove that the specified FP value is either NaN or never less than -0....
PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)
PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39