LLVM 24.0.0git
InstCombineCalls.cpp
Go to the documentation of this file.
1//===- InstCombineCalls.cpp -----------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the visitCall, visitInvoke, and visitCallBr functions.
10//
11//===----------------------------------------------------------------------===//
12
13#include "InstCombineInternal.h"
14#include "llvm/ADT/APFloat.h"
15#include "llvm/ADT/APInt.h"
16#include "llvm/ADT/APSInt.h"
17#include "llvm/ADT/ArrayRef.h"
18#include "llvm/ADT/Bitset.h"
22#include "llvm/ADT/Statistic.h"
28#include "llvm/Analysis/Loads.h"
33#include "llvm/IR/Attributes.h"
34#include "llvm/IR/BasicBlock.h"
36#include "llvm/IR/Constant.h"
37#include "llvm/IR/Constants.h"
38#include "llvm/IR/DataLayout.h"
39#include "llvm/IR/DebugInfo.h"
41#include "llvm/IR/Function.h"
43#include "llvm/IR/InlineAsm.h"
44#include "llvm/IR/InstrTypes.h"
45#include "llvm/IR/Instruction.h"
48#include "llvm/IR/Intrinsics.h"
49#include "llvm/IR/IntrinsicsAArch64.h"
50#include "llvm/IR/IntrinsicsAMDGPU.h"
51#include "llvm/IR/IntrinsicsARM.h"
52#include "llvm/IR/IntrinsicsHexagon.h"
53#include "llvm/IR/LLVMContext.h"
54#include "llvm/IR/Metadata.h"
57#include "llvm/IR/Statepoint.h"
58#include "llvm/IR/Type.h"
59#include "llvm/IR/User.h"
60#include "llvm/IR/Value.h"
61#include "llvm/IR/ValueHandle.h"
65#include "llvm/Support/Debug.h"
76#include <algorithm>
77#include <cassert>
78#include <cstdint>
79#include <optional>
80#include <utility>
81#include <vector>
82
83#define DEBUG_TYPE "instcombine"
85
86using namespace llvm;
87using namespace PatternMatch;
88
89STATISTIC(NumSimplified, "Number of library calls simplified");
90
91/// Return the specified type promoted as it would be to pass though a va_arg
92/// area.
94 if (IntegerType* ITy = dyn_cast<IntegerType>(Ty)) {
95 if (ITy->getBitWidth() < 32)
96 return Type::getInt32Ty(Ty->getContext());
97 }
98 return Ty;
99}
100
101/// Recognize a memcpy/memmove from a trivially otherwise unused alloca.
102/// TODO: This should probably be integrated with visitAllocSites, but that
103/// requires a deeper change to allow either unread or unwritten objects.
105 auto *Src = MI->getRawSource();
106 while (isa<GetElementPtrInst>(Src)) {
107 if (!Src->hasOneUse())
108 return false;
109 Src = cast<Instruction>(Src)->getOperand(0);
110 }
111 return isa<AllocaInst>(Src) && Src->hasOneUse();
112}
113
115 Align DstAlign = getKnownAlignment(MI->getRawDest(), DL, MI, &AC, &DT);
116 MaybeAlign CopyDstAlign = MI->getDestAlign();
117 if (!CopyDstAlign || *CopyDstAlign < DstAlign) {
118 MI->setDestAlignment(DstAlign);
119 return MI;
120 }
121
122 Align SrcAlign = getKnownAlignment(MI->getRawSource(), DL, MI, &AC, &DT);
123 MaybeAlign CopySrcAlign = MI->getSourceAlign();
124 if (!CopySrcAlign || *CopySrcAlign < SrcAlign) {
125 MI->setSourceAlignment(SrcAlign);
126 return MI;
127 }
128
129 // If we have a store to a location which is known constant, we can conclude
130 // that the store must be storing the constant value (else the memory
131 // wouldn't be constant), and this must be a noop.
132 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
133 // Set the size of the copy to 0, it will be deleted on the next iteration.
134 MI->setLength((uint64_t)0);
135 return MI;
136 }
137
138 // If the source is provably undef, the memcpy/memmove doesn't do anything
139 // (unless the transfer is volatile).
140 if (hasUndefSource(MI) && !MI->isVolatile()) {
141 // Set the size of the copy to 0, it will be deleted on the next iteration.
142 MI->setLength((uint64_t)0);
143 return MI;
144 }
145
146 // If MemCpyInst length is 1/2/4/8 bytes then replace memcpy with
147 // load/store.
148 ConstantInt *MemOpLength = dyn_cast<ConstantInt>(MI->getLength());
149 if (!MemOpLength) return nullptr;
150
151 // Source and destination pointer types are always "i8*" for intrinsic. See
152 // if the size is something we can handle with a single primitive load/store.
153 // A single load+store correctly handles overlapping memory in the memmove
154 // case.
155 uint64_t Size = MemOpLength->getLimitedValue();
156 assert(Size && "0-sized memory transferring should be removed already.");
157
158 if (Size > 8 || (Size&(Size-1)))
159 return nullptr; // If not 1/2/4/8 bytes, exit.
160
161 // If it is an atomic and alignment is less than the size then we will
162 // introduce the unaligned memory access which will be later transformed
163 // into libcall in CodeGen. This is not evident performance gain so disable
164 // it now.
165 if (MI->isAtomic())
166 if (*CopyDstAlign < Size || *CopySrcAlign < Size)
167 return nullptr;
168
169 // Use an integer load+store unless we can find something better.
170 IntegerType* IntType = IntegerType::get(MI->getContext(), Size<<3);
171
172 // If the memcpy has metadata describing the members, see if we can get the
173 // TBAA, scope and noalias tags describing our copy.
174 AAMDNodes AACopyMD = MI->getAAMetadata().adjustForAccess(Size);
175
176 Value *Src = MI->getArgOperand(1);
177 Value *Dest = MI->getArgOperand(0);
178 LoadInst *L = Builder.CreateLoad(IntType, Src);
179 // Alignment from the mem intrinsic will be better, so use it.
180 L->setAlignment(*CopySrcAlign);
181 L->setAAMetadata(AACopyMD);
182 MDNode *LoopMemParallelMD =
183 MI->getMetadata(LLVMContext::MD_mem_parallel_loop_access);
184 if (LoopMemParallelMD)
185 L->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
186 MDNode *AccessGroupMD = MI->getMetadata(LLVMContext::MD_access_group);
187 if (AccessGroupMD)
188 L->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
189
190 StoreInst *S = Builder.CreateStore(L, Dest);
191 // Alignment from the mem intrinsic will be better, so use it.
192 S->setAlignment(*CopyDstAlign);
193 S->setAAMetadata(AACopyMD);
194 if (LoopMemParallelMD)
195 S->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
196 if (AccessGroupMD)
197 S->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
198 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
199
200 if (auto *MT = dyn_cast<MemTransferInst>(MI)) {
201 // non-atomics can be volatile
202 L->setVolatile(MT->isVolatile());
203 S->setVolatile(MT->isVolatile());
204 }
205 if (MI->isAtomic()) {
206 // atomics have to be unordered
207 L->setOrdering(AtomicOrdering::Unordered);
209 }
210
211 // Set the size of the copy to 0, it will be deleted on the next iteration.
212 MI->setLength((uint64_t)0);
213 return MI;
214}
215
217 const Align KnownAlignment =
218 getKnownAlignment(MI->getDest(), DL, MI, &AC, &DT);
219 MaybeAlign MemSetAlign = MI->getDestAlign();
220 if (!MemSetAlign || *MemSetAlign < KnownAlignment) {
221 MI->setDestAlignment(KnownAlignment);
222 return MI;
223 }
224
225 // If we have a store to a location which is known constant, we can conclude
226 // that the store must be storing the constant value (else the memory
227 // wouldn't be constant), and this must be a noop.
228 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
229 // Set the size of the copy to 0, it will be deleted on the next iteration.
230 MI->setLength((uint64_t)0);
231 return MI;
232 }
233
234 // Remove memset with an undef value.
235 // FIXME: This is technically incorrect because it might overwrite a poison
236 // value. Change to PoisonValue once #52930 is resolved.
237 if (isa<UndefValue>(MI->getValue())) {
238 // Set the size of the copy to 0, it will be deleted on the next iteration.
239 MI->setLength((uint64_t)0);
240 return MI;
241 }
242
243 // Extract the length and validate the fill type.
244 ConstantInt *LenC = dyn_cast<ConstantInt>(MI->getLength());
245 Value *Fill = MI->getValue();
246 if (!LenC || !Fill->getType()->isIntegerTy(8))
247 return nullptr;
248 const uint64_t Len = LenC->getLimitedValue();
249 assert(Len && "0-sized memory setting should be removed already.");
250 const Align Alignment = MI->getDestAlign().valueOrOne();
251
252 // If it is an atomic and alignment is less than the size then we will
253 // introduce the unaligned memory access which will be later transformed
254 // into libcall in CodeGen. This is not evident performance gain so disable
255 // it now.
256 if (MI->isAtomic() && Alignment < Len)
257 return nullptr;
258
259 // memset(s,c,n) -> store s, c (for n=1,2,4,8)
260 if (Len <= 8 && isPowerOf2_32((uint32_t)Len)) {
261 Value *Dest = MI->getDest();
262
263 // Extract the fill value and store. A one-byte memset does not need
264 // replication so a nonconstant i8 fill can be stored directly.
265 Value *FillVal;
266 if (auto *FillC = dyn_cast<ConstantInt>(Fill))
267 FillVal = ConstantInt::get(MI->getContext(),
268 APInt::getSplat(Len * 8, FillC->getValue()));
269 else if (Len == 1)
270 FillVal = Fill;
271 else
272 return nullptr;
273
274 StoreInst *S = Builder.CreateStore(FillVal, Dest, MI->isVolatile());
275 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
276 for (DbgVariableRecord *DbgAssign : at::getDVRAssignmentMarkers(S)) {
277 if (llvm::is_contained(DbgAssign->location_ops(), Fill))
278 DbgAssign->replaceVariableLocationOp(Fill, FillVal);
279 }
280
281 S->setAlignment(Alignment);
282 if (MI->isAtomic())
284
285 // Set the size of the copy to 0, it will be deleted on the next iteration.
286 MI->setLength((uint64_t)0);
287 return MI;
288 }
289
290 return nullptr;
291}
292
293// TODO, Obvious Missing Transforms:
294// * Narrow width by halfs excluding zero/undef lanes
295Value *InstCombinerImpl::simplifyMaskedLoad(IntrinsicInst &II) {
296 Value *LoadPtr = II.getArgOperand(0);
297 const Align Alignment = II.getParamAlign(0).valueOrOne();
298 Value *Mask = II.getArgOperand(1);
299
300 // If the mask is all ones or poison, this is a plain vector load of the 1st
301 // argument.
302 if (match(Mask, m_AllOnesOrPoison())) {
303 LoadInst *L = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
304 "unmaskedload");
305 L->copyMetadata(II);
306 return L;
307 }
308
309 // If we can unconditionally load from this address, replace with a
310 // load/select idiom.
311 if (isDereferenceablePointer(LoadPtr, II.getType(),
313 LoadInst *LI = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
314 "unmaskedload");
315 LI->copyMetadata(II);
316 return Builder.CreateSelect(II.getArgOperand(1), LI, II.getArgOperand(2));
317 }
318
319 return nullptr;
320}
321
322// TODO, Obvious Missing Transforms:
323// * Single constant active lane -> store
324// * Narrow width by halfs excluding zero/undef lanes
325Instruction *InstCombinerImpl::simplifyMaskedStore(IntrinsicInst &II) {
326 Value *StorePtr = II.getArgOperand(1);
327 Align Alignment = II.getParamAlign(1).valueOrOne();
328 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
329 if (!ConstMask)
330 return nullptr;
331
332 // If the mask is all zeros or poison, this instruction does nothing.
333 if (match(ConstMask, m_ZeroOrPoison()))
335
336 // If the mask is all ones or poison, this is a plain vector store of the 1st
337 // argument.
338 if (match(ConstMask, m_AllOnesOrPoison())) {
339 StoreInst *S =
340 new StoreInst(II.getArgOperand(0), StorePtr, false, Alignment);
341 S->copyMetadata(II);
342 return S;
343 }
344
345 if (isa<ScalableVectorType>(ConstMask->getType()))
346 return nullptr;
347
348 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
349 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
350 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
351 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
352 PoisonElts))
353 return replaceOperand(II, 0, V);
354
355 return nullptr;
356}
357
358// TODO, Obvious Missing Transforms:
359// * Single constant active lane load -> load
360// * Dereferenceable address & few lanes -> scalarize speculative load/selects
361// * Adjacent vector addresses -> masked.load
362// * Narrow width by halfs excluding zero/undef lanes
363// * Vector incrementing address -> vector masked load
364Instruction *InstCombinerImpl::simplifyMaskedGather(IntrinsicInst &II) {
365 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(1));
366 if (!ConstMask)
367 return nullptr;
368
369 // Vector splat address w/known mask -> scalar load
370 // Fold the gather to load the source vector first lane
371 // because it is reloading the same value each time
372 if (ConstMask->isAllOnesValue())
373 if (auto *SplatPtr = getSplatValue(II.getArgOperand(0))) {
374 auto *VecTy = cast<VectorType>(II.getType());
375 const Align Alignment = II.getParamAlign(0).valueOrOne();
376 LoadInst *L = Builder.CreateAlignedLoad(VecTy->getElementType(), SplatPtr,
377 Alignment, "load.scalar");
378 Value *Shuf =
379 Builder.CreateVectorSplat(VecTy->getElementCount(), L, "broadcast");
381 }
382
383 return nullptr;
384}
385
386// TODO, Obvious Missing Transforms:
387// * Single constant active lane -> store
388// * Adjacent vector addresses -> masked.store
389// * Narrow store width by halfs excluding zero/undef lanes
390// * Vector incrementing address -> vector masked store
391Instruction *InstCombinerImpl::simplifyMaskedScatter(IntrinsicInst &II) {
392 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
393 if (!ConstMask)
394 return nullptr;
395
396 // If the mask is all zeros or poison, a scatter does nothing.
397 if (match(ConstMask, m_ZeroOrPoison()))
399
400 // Vector splat address -> scalar store
401 if (auto *SplatPtr = getSplatValue(II.getArgOperand(1))) {
402 // scatter(splat(value), splat(ptr), non-zero-mask) -> store value, ptr
403 if (auto *SplatValue = getSplatValue(II.getArgOperand(0))) {
404 if (maskContainsAllOneOrUndef(ConstMask)) {
405 Align Alignment = II.getParamAlign(1).valueOrOne();
406 StoreInst *S = new StoreInst(SplatValue, SplatPtr, /*IsVolatile=*/false,
407 Alignment);
408 S->copyMetadata(II);
409 return S;
410 }
411 }
412 // scatter(vector, splat(ptr), splat(true)) -> store extract(vector,
413 // lastlane), ptr
414 if (ConstMask->isAllOnesValue()) {
415 Align Alignment = II.getParamAlign(1).valueOrOne();
416 VectorType *WideLoadTy = cast<VectorType>(II.getArgOperand(1)->getType());
417 ElementCount VF = WideLoadTy->getElementCount();
418 Value *RunTimeVF = Builder.CreateElementCount(Builder.getInt32Ty(), VF);
419 Value *LastLane = Builder.CreateSub(RunTimeVF, Builder.getInt32(1));
420 Value *Extract =
421 Builder.CreateExtractElement(II.getArgOperand(0), LastLane);
422 StoreInst *S =
423 new StoreInst(Extract, SplatPtr, /*IsVolatile=*/false, Alignment);
424 S->copyMetadata(II);
425 return S;
426 }
427 }
428 if (isa<ScalableVectorType>(ConstMask->getType()))
429 return nullptr;
430
431 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
432 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
433 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
434 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
435 PoisonElts))
436 return replaceOperand(II, 0, V);
437 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(1), DemandedElts,
438 PoisonElts))
439 return replaceOperand(II, 1, V);
440
441 return nullptr;
442}
443
444/// This function transforms launder.invariant.group like:
445/// launder(launder(%x)) -> launder(%x) (the result is not the argument)
446/// This is legal because it preserves the most recent information about
447/// the presence or absence of invariant.group.
449 InstCombinerImpl &IC) {
450 auto *Arg = II.getArgOperand(0);
451 auto *StrippedArg = Arg->stripPointerCasts();
452 auto *StrippedInvariantGroupsArg = StrippedArg;
453 while (auto *Intr = dyn_cast<IntrinsicInst>(StrippedInvariantGroupsArg)) {
454 if (Intr->getIntrinsicID() != Intrinsic::launder_invariant_group)
455 break;
456 StrippedInvariantGroupsArg = Intr->getArgOperand(0)->stripPointerCasts();
457 }
458 if (StrippedArg == StrippedInvariantGroupsArg)
459 return nullptr; // No launders to remove.
460
461 Value *Result =
462 IC.Builder.CreateLaunderInvariantGroup(StrippedInvariantGroupsArg);
463 if (Result->getType()->getPointerAddressSpace() !=
464 II.getType()->getPointerAddressSpace())
465 Result = IC.Builder.CreateAddrSpaceCast(Result, II.getType());
466
467 return cast<Instruction>(Result);
468}
469
471 assert((II.getIntrinsicID() == Intrinsic::cttz ||
472 II.getIntrinsicID() == Intrinsic::ctlz) &&
473 "Expected cttz or ctlz intrinsic");
474 bool IsTZ = II.getIntrinsicID() == Intrinsic::cttz;
475 Value *Op0 = II.getArgOperand(0);
476 Value *Op1 = II.getArgOperand(1);
477 Value *X;
478 // ctlz(bitreverse(x)) -> cttz(x)
479 // cttz(bitreverse(x)) -> ctlz(x)
480 if (match(Op0, m_BitReverse(m_Value(X)))) {
481 Intrinsic::ID ID = IsTZ ? Intrinsic::ctlz : Intrinsic::cttz;
482 Function *F =
483 Intrinsic::getOrInsertDeclaration(II.getModule(), ID, II.getType());
484 return CallInst::Create(F, {X, II.getArgOperand(1)});
485 }
486
487 if (II.getType()->isIntOrIntVectorTy(1)) {
488 // ctlz/cttz i1 Op0 --> not Op0
489 if (match(Op1, m_Zero()))
490 return BinaryOperator::CreateNot(Op0);
491 // If zero is poison, then the input can be assumed to be "true", so the
492 // instruction simplifies to "false".
493 assert(match(Op1, m_One()) && "Expected ctlz/cttz operand to be 0 or 1");
494 return IC.replaceInstUsesWith(II, ConstantInt::getNullValue(II.getType()));
495 }
496
497 // If ctlz/cttz is only used as a shift amount, set is_zero_poison to true.
498 if (II.hasOneUse() && match(Op1, m_Zero()) &&
499 match(II.user_back(), m_Shift(m_Value(), m_Specific(&II))))
500 return CallInst::Create(II.getCalledFunction(),
501 {Op0, IC.Builder.getTrue()});
502
503 Constant *C;
504
505 if (IsTZ) {
506 // cttz(-x) -> cttz(x)
507 if (match(Op0, m_Neg(m_Value(X))))
508 return CallInst::Create(II.getCalledFunction(), {X, Op1});
509
510 // cttz(-x & x) -> cttz(x)
511 if (match(Op0, m_c_And(m_Neg(m_Value(X)), m_Deferred(X))))
512 return CallInst::Create(II.getCalledFunction(), {X, Op1});
513
514 // cttz(mul(X, OddC)) -> cttz(X)
515 if (match(Op0, m_Mul(m_Value(X),
516 m_CheckedInt([](const APInt &C) { return C[0]; }))))
517 return CallInst::Create(II.getCalledFunction(), {X, Op1});
518
519 // cttz(sext(x)) -> cttz(zext(x))
520 if (match(Op0, m_OneUse(m_SExt(m_Value(X))))) {
521 auto *Zext = IC.Builder.CreateZExt(X, II.getType());
522 auto *CttzZext =
523 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, Zext, Op1);
524 return IC.replaceInstUsesWith(II, CttzZext);
525 }
526
527 // Zext doesn't change the number of trailing zeros, so narrow:
528 // cttz(zext(x)) -> zext(cttz(x)) if the 'ZeroIsPoison' parameter is 'true'.
529 if (match(Op0, m_OneUse(m_ZExt(m_Value(X)))) && match(Op1, m_One())) {
530 auto *Cttz = IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, X,
531 IC.Builder.getTrue());
532 auto *ZextCttz = IC.Builder.CreateZExt(Cttz, II.getType());
533 return IC.replaceInstUsesWith(II, ZextCttz);
534 }
535
536 // cttz(abs(x)) -> cttz(x)
537 // cttz(nabs(x)) -> cttz(x)
538 Value *Y;
540 if (SPF == SPF_ABS || SPF == SPF_NABS)
541 return CallInst::Create(II.getCalledFunction(), {X, Op1});
542
544 return CallInst::Create(II.getCalledFunction(), {X, Op1});
545
546 // cttz(shl(%const, %val), 1) --> add(cttz(%const, 1), %val)
547 if (match(Op0, m_Shl(m_ImmConstant(C), m_Value(X))) &&
548 match(Op1, m_One())) {
549 Value *ConstCttz =
550 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
551 return BinaryOperator::CreateAdd(ConstCttz, X);
552 }
553
554 // cttz(lshr exact (%const, %val), 1) --> sub(cttz(%const, 1), %val)
555 if (match(Op0, m_Exact(m_LShr(m_ImmConstant(C), m_Value(X)))) &&
556 match(Op1, m_One())) {
557 Value *ConstCttz =
558 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
559 return BinaryOperator::CreateSub(ConstCttz, X);
560 }
561
562 // cttz(add(lshr(UINT_MAX, %val), 1)) --> sub(width, %val)
563 if (match(Op0, m_Add(m_LShr(m_AllOnes(), m_Value(X)), m_One()))) {
564 Value *Width =
565 ConstantInt::get(II.getType(), II.getType()->getScalarSizeInBits());
566 return BinaryOperator::CreateSub(Width, X);
567 }
568 } else {
569 // ctlz(lshr(%const, %val), 1) --> add(ctlz(%const, 1), %val)
570 if (match(Op0, m_LShr(m_ImmConstant(C), m_Value(X))) &&
571 match(Op1, m_One())) {
572 Value *ConstCtlz =
573 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
574 return BinaryOperator::CreateAdd(ConstCtlz, X);
575 }
576
577 // ctlz(shl nuw (%const, %val), 1) --> sub(ctlz(%const, 1), %val)
578 if (match(Op0, m_NUWShl(m_ImmConstant(C), m_Value(X))) &&
579 match(Op1, m_One())) {
580 Value *ConstCtlz =
581 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
582 return BinaryOperator::CreateSub(ConstCtlz, X);
583 }
584
585 // ctlz(~x & (x - 1)) -> bitwidth - cttz(x, false)
586 if (Op0->hasOneUse() &&
587 match(Op0,
589 Type *Ty = II.getType();
590 unsigned BitWidth = Ty->getScalarSizeInBits();
591 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
592 {X, IC.Builder.getFalse()});
593 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
594 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
595 }
596 }
597
598 // cttz(Pow2) -> Log2(Pow2)
599 // ctlz(Pow2) -> BitWidth - 1 - Log2(Pow2)
600 if (auto *R = IC.tryGetLog2(Op0, match(Op1, m_One()))) {
601 if (IsTZ)
602 return IC.replaceInstUsesWith(II, R);
603 BinaryOperator *BO = BinaryOperator::CreateSub(
604 ConstantInt::get(R->getType(), R->getType()->getScalarSizeInBits() - 1),
605 R);
606 BO->setHasNoSignedWrap();
608 return BO;
609 }
610
612
613 // Create a mask for bits above (ctlz) or below (cttz) the first known one.
614 unsigned PossibleZeros = IsTZ ? Known.countMaxTrailingZeros()
615 : Known.countMaxLeadingZeros();
616 unsigned DefiniteZeros = IsTZ ? Known.countMinTrailingZeros()
617 : Known.countMinLeadingZeros();
618
619 // If all bits above (ctlz) or below (cttz) the first known one are known
620 // zero, this value is constant.
621 // FIXME: This should be in InstSimplify because we're replacing an
622 // instruction with a constant.
623 if (PossibleZeros == DefiniteZeros) {
624 auto *C = ConstantInt::get(Op0->getType(), DefiniteZeros);
625 return IC.replaceInstUsesWith(II, C);
626 }
627
628 // If the input to cttz/ctlz is known to be non-zero,
629 // then change the 'ZeroIsPoison' parameter to 'true'
630 // because we know the zero behavior can't affect the result.
631 if (!Known.One.isZero() ||
633 if (!match(II.getArgOperand(1), m_One()))
634 return CallInst::Create(II.getCalledFunction(),
635 {Op0, IC.Builder.getTrue()});
636 }
637
638 // Add range attribute since known bits can't completely reflect what we know.
639 unsigned BitWidth = Op0->getType()->getScalarSizeInBits();
640 if (BitWidth != 1 && !II.hasRetAttr(Attribute::Range) &&
641 !II.getMetadata(LLVMContext::MD_range)) {
642 ConstantRange Range(APInt(BitWidth, DefiniteZeros),
643 APInt(BitWidth, PossibleZeros + 1));
644 II.addRangeRetAttr(Range);
645 return &II;
646 }
647
648 return nullptr;
649}
650
652 assert(II.getIntrinsicID() == Intrinsic::ctpop &&
653 "Expected ctpop intrinsic");
654 Type *Ty = II.getType();
655 unsigned BitWidth = Ty->getScalarSizeInBits();
656 Value *Op0 = II.getArgOperand(0);
657 Value *X, *Y;
658
659 // ctpop(bitreverse(x)) -> ctpop(x)
660 // ctpop(bswap(x)) -> ctpop(x)
661 if (match(Op0, m_BitReverse(m_Value(X))) || match(Op0, m_BSwap(m_Value(X))))
662 return CallInst::Create(II.getCalledFunction(), X);
663
664 // ctpop(rot(x)) -> ctpop(x)
665 if ((match(Op0, m_FShl(m_Value(X), m_Value(Y), m_Value())) ||
666 match(Op0, m_FShr(m_Value(X), m_Value(Y), m_Value()))) &&
667 X == Y)
668 return CallInst::Create(II.getCalledFunction(), X);
669
670 // ctpop(x | -x) -> bitwidth - cttz(x, false)
671 if (Op0->hasOneUse() &&
672 match(Op0, m_c_Or(m_Value(X), m_Neg(m_Deferred(X))))) {
673 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
674 {X, IC.Builder.getFalse()});
675 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
676 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
677 }
678
679 // ctpop(~x & (x - 1)) -> cttz(x, false)
680 if (match(Op0,
682 Function *F =
683 Intrinsic::getOrInsertDeclaration(II.getModule(), Intrinsic::cttz, Ty);
684 return CallInst::Create(F, {X, IC.Builder.getFalse()});
685 }
686
687 // Zext doesn't change the number of set bits, so narrow:
688 // ctpop (zext X) --> zext (ctpop X)
689 if (match(Op0, m_OneUse(m_ZExt(m_Value(X))))) {
690 Value *NarrowPop = IC.Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, X);
691 return CastInst::Create(Instruction::ZExt, NarrowPop, Ty);
692 }
693
695 IC.computeKnownBits(Op0, Known, &II);
696
697 // If all bits are zero except for exactly one fixed bit, then the result
698 // must be 0 or 1, and we can get that answer by shifting to LSB:
699 // ctpop (X & 32) --> (X & 32) >> 5
700 // TODO: Investigate removing this as its likely unnecessary given the below
701 // `isKnownToBeAPowerOfTwo` check.
702 if ((~Known.Zero).isPowerOf2())
703 return BinaryOperator::CreateLShr(
704 Op0, ConstantInt::get(Ty, (~Known.Zero).exactLogBase2()));
705
706 // More generally we can also handle non-constant power of 2 patterns such as
707 // shl/shr(Pow2, X), (X & -X), etc... by transforming:
708 // ctpop(Pow2OrZero) --> icmp ne X, 0
709 if (IC.isKnownToBeAPowerOfTwo(Op0, /* OrZero */ true))
710 return CastInst::Create(Instruction::ZExt,
713 Ty);
714
715 // Add range attribute since known bits can't completely reflect what we know.
716 if (BitWidth != 1) {
717 ConstantRange OldRange =
718 II.getRange().value_or(ConstantRange::getFull(BitWidth));
719
720 unsigned Lower = Known.countMinPopulation();
721 unsigned Upper = Known.countMaxPopulation() + 1;
722
723 if (Lower == 0 && OldRange.contains(APInt::getZero(BitWidth)) &&
725 Lower = 1;
726
728 Range = Range.intersectWith(OldRange, ConstantRange::Unsigned);
729
730 if (Range != OldRange) {
731 II.addRangeRetAttr(Range);
732 return &II;
733 }
734 }
735
736 return nullptr;
737}
738
739/// Convert `tbl`/`tbx` intrinsics to shufflevector if the mask is constant, and
740/// at most two source operands are actually referenced.
742 bool IsExtension) {
743 // Bail out if the mask is not a constant.
744 auto *C = dyn_cast<Constant>(II.getArgOperand(II.arg_size() - 1));
745 if (!C)
746 return nullptr;
747
748 auto *RetTy = cast<FixedVectorType>(II.getType());
749 unsigned NumIndexes = RetTy->getNumElements();
750
751 // Only perform this transformation for <8 x i8> and <16 x i8> vector types.
752 if (!RetTy->getElementType()->isIntegerTy(8) ||
753 (NumIndexes != 8 && NumIndexes != 16))
754 return nullptr;
755
756 // For tbx instructions, the first argument is the "fallback" vector, which
757 // has the same length as the mask and return type.
758 unsigned int StartIndex = (unsigned)IsExtension;
759 auto *SourceTy =
760 cast<FixedVectorType>(II.getArgOperand(StartIndex)->getType());
761 // Note that the element count of each source vector does *not* need to be the
762 // same as the element count of the return type and mask! All source vectors
763 // must have the same element count as each other, though.
764 unsigned NumElementsPerSource = SourceTy->getNumElements();
765
766 // There are no tbl/tbx intrinsics for which the destination size exceeds the
767 // source size. However, our definitions of the intrinsics, at least in
768 // IntrinsicsAArch64.td, allow for arbitrary destination vector sizes, so it
769 // *could* technically happen.
770 if (NumIndexes > NumElementsPerSource)
771 return nullptr;
772
773 // The tbl/tbx intrinsics take several source operands followed by a mask
774 // operand.
775 unsigned int NumSourceOperands = II.arg_size() - 1 - (unsigned)IsExtension;
776
777 // Map input operands to shuffle indices. This also helpfully deduplicates the
778 // input arguments, in case the same value is passed as an argument multiple
779 // times.
780 SmallDenseMap<Value *, unsigned, 2> ValueToShuffleSlot;
781 Value *ShuffleOperands[2] = {PoisonValue::get(SourceTy),
782 PoisonValue::get(SourceTy)};
783
784 int Indexes[16];
785 for (unsigned I = 0; I < NumIndexes; ++I) {
786 Constant *COp = C->getAggregateElement(I);
787
788 if (!COp || (!isa<UndefValue>(COp) && !isa<ConstantInt>(COp)))
789 return nullptr;
790
791 if (isa<UndefValue>(COp)) {
792 Indexes[I] = -1;
793 continue;
794 }
795
796 uint64_t Index = cast<ConstantInt>(COp)->getZExtValue();
797 // The index of the input argument that this index references (0 = first
798 // source argument, etc).
799 unsigned SourceOperandIndex = Index / NumElementsPerSource;
800 // The index of the element at that source operand.
801 unsigned SourceOperandElementIndex = Index % NumElementsPerSource;
802
803 Value *SourceOperand;
804 if (SourceOperandIndex >= NumSourceOperands) {
805 // This index is out of bounds. Map it to index into either the fallback
806 // vector (tbx) or vector of zeroes (tbl).
807 SourceOperandIndex = NumSourceOperands;
808 if (IsExtension) {
809 // For out-of-bounds indices in tbx, choose the `I`th element of the
810 // fallback.
811 SourceOperand = II.getArgOperand(0);
812 SourceOperandElementIndex = I;
813 } else {
814 // Otherwise, choose some element from the dummy vector of zeroes (we'll
815 // always choose the first).
816 SourceOperand = Constant::getNullValue(SourceTy);
817 SourceOperandElementIndex = 0;
818 }
819 } else {
820 SourceOperand = II.getArgOperand(SourceOperandIndex + StartIndex);
821 }
822
823 // The source operand may be the fallback vector, which may not have the
824 // same number of elements as the source vector. In that case, we *could*
825 // choose to extend its length with another shufflevector, but it's simpler
826 // to just bail instead.
827 if (cast<FixedVectorType>(SourceOperand->getType())->getNumElements() !=
828 NumElementsPerSource)
829 return nullptr;
830
831 // We now know the source operand referenced by this index. Make it a
832 // shufflevector operand, if it isn't already.
833 unsigned NumSlots = ValueToShuffleSlot.size();
834 // This shuffle references more than two sources, and hence cannot be
835 // represented as a shufflevector.
836 if (NumSlots == 2 && !ValueToShuffleSlot.contains(SourceOperand))
837 return nullptr;
838
839 auto [It, Inserted] =
840 ValueToShuffleSlot.try_emplace(SourceOperand, NumSlots);
841 if (Inserted)
842 ShuffleOperands[It->getSecond()] = SourceOperand;
843
844 unsigned RemappedIndex =
845 (It->getSecond() * NumElementsPerSource) + SourceOperandElementIndex;
846 Indexes[I] = RemappedIndex;
847 }
848
850 ShuffleOperands[0], ShuffleOperands[1], ArrayRef(Indexes, NumIndexes));
851 return IC.replaceInstUsesWith(II, Shuf);
852}
853
854// Returns true iff the 2 intrinsics have the same operands, limiting the
855// comparison to the first NumOperands.
856static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E,
857 unsigned NumOperands) {
858 assert(I.arg_size() >= NumOperands && "Not enough operands");
859 assert(E.arg_size() >= NumOperands && "Not enough operands");
860 for (unsigned i = 0; i < NumOperands; i++)
861 if (I.getArgOperand(i) != E.getArgOperand(i))
862 return false;
863 return true;
864}
865
866// Remove trivially empty start/end intrinsic ranges, i.e. a start
867// immediately followed by an end (ignoring debuginfo or other
868// start/end intrinsics in between). As this handles only the most trivial
869// cases, tracking the nesting level is not needed:
870//
871// call @llvm.foo.start(i1 0)
872// call @llvm.foo.start(i1 0) ; This one won't be skipped: it will be removed
873// call @llvm.foo.end(i1 0)
874// call @llvm.foo.end(i1 0) ; &I
875static bool
877 std::function<bool(const IntrinsicInst &)> IsStart) {
878 // We start from the end intrinsic and scan backwards, so that InstCombine
879 // has already processed (and potentially removed) all the instructions
880 // before the end intrinsic.
881 BasicBlock::reverse_iterator BI(EndI), BE(EndI.getParent()->rend());
882 for (; BI != BE; ++BI) {
883 if (auto *I = dyn_cast<IntrinsicInst>(&*BI)) {
884 if (I->isDebugOrPseudoInst() ||
885 I->getIntrinsicID() == EndI.getIntrinsicID())
886 continue;
887 if (IsStart(*I)) {
888 if (haveSameOperands(EndI, *I, EndI.arg_size())) {
890 IC.eraseInstFromFunction(EndI);
891 return true;
892 }
893 // Skip start intrinsics that don't pair with this end intrinsic.
894 continue;
895 }
896 }
897 break;
898 }
899
900 return false;
901}
902
904 removeTriviallyEmptyRange(I, *this, [&I](const IntrinsicInst &II) {
905 // Bail out on the case where the source va_list of a va_copy is destroyed
906 // immediately by a follow-up va_end.
907 return II.getIntrinsicID() == Intrinsic::vastart ||
908 (II.getIntrinsicID() == Intrinsic::vacopy &&
909 I.getArgOperand(0) != II.getArgOperand(1));
910 });
911 return nullptr;
912}
913
915 assert(Call.arg_size() > 1 && "Need at least 2 args to swap");
916 Value *Arg0 = Call.getArgOperand(0), *Arg1 = Call.getArgOperand(1);
917 if (isa<Constant>(Arg0) && !isa<Constant>(Arg1)) {
918 Call.setArgOperand(0, Arg1);
919 Call.setArgOperand(1, Arg0);
920 AttributeList CallAttr = Call.getAttributes();
921 AttributeSet LHSAttr = CallAttr.getParamAttrs(0);
922 AttributeSet RHSAttr = CallAttr.getParamAttrs(1);
923 LLVMContext &Ctx = Call.getContext();
924 Call.setAttributes(CallAttr
925 .setAttributesAtIndex(
926 Ctx, AttributeList::FirstArgIndex + 0, RHSAttr)
927 .setAttributesAtIndex(
928 Ctx, AttributeList::FirstArgIndex + 1, LHSAttr));
929 return &Call;
930 }
931 return nullptr;
932}
933
934/// Creates a result tuple for an overflow intrinsic \p II with a given
935/// \p Result and a constant \p Overflow value.
937 Constant *Overflow) {
938 Constant *V[] = {PoisonValue::get(Result->getType()), Overflow};
939 StructType *ST = cast<StructType>(II->getType());
940 Constant *Struct = ConstantStruct::get(ST, V);
941 return InsertValueInst::Create(Struct, Result, 0);
942}
943
945InstCombinerImpl::foldIntrinsicWithOverflowCommon(IntrinsicInst *II) {
946 WithOverflowInst *WO = cast<WithOverflowInst>(II);
947 Value *OperationResult = nullptr;
948 Constant *OverflowResult = nullptr;
949 if (OptimizeOverflowCheck(WO->getBinaryOp(), WO->isSigned(), WO->getLHS(),
950 WO->getRHS(), *WO, OperationResult, OverflowResult))
951 return createOverflowTuple(WO, OperationResult, OverflowResult);
952
953 // See whether we can optimize the overflow check with assumption information.
954 for (User *U : WO->users()) {
955 if (!match(U, m_ExtractValue<1>(m_Value())))
956 continue;
957
958 for (auto &AssumeVH : AC.assumptionsFor(U)) {
959 if (!AssumeVH)
960 continue;
961 CallInst *I = cast<CallInst>(AssumeVH);
962 if (!match(I->getArgOperand(0), m_Not(m_Specific(U))))
963 continue;
964 if (!isValidAssumeForContext(I, II, /*DT=*/nullptr,
965 /*AllowEphemerals=*/true))
966 continue;
967 Value *Result =
968 Builder.CreateBinOp(WO->getBinaryOp(), WO->getLHS(), WO->getRHS());
969 Result->takeName(WO);
970 if (auto *Inst = dyn_cast<Instruction>(Result)) {
971 if (WO->isSigned())
972 Inst->setHasNoSignedWrap();
973 else
974 Inst->setHasNoUnsignedWrap();
975 }
976 return createOverflowTuple(WO, Result,
977 ConstantInt::getFalse(U->getType()));
978 }
979 }
980
981 return nullptr;
982}
983
984static bool inputDenormalIsIEEE(const Function &F, const Type *Ty) {
985 Ty = Ty->getScalarType();
986 return F.getDenormalMode(Ty->getFltSemantics()).Input == DenormalMode::IEEE;
987}
988
989static bool inputDenormalIsDAZ(const Function &F, const Type *Ty) {
990 Ty = Ty->getScalarType();
991 return F.getDenormalMode(Ty->getFltSemantics()).inputsAreZero();
992}
993
994/// Flushing a denormal to +0.0 breaks f(-x) = -f(x) for odd f.
998 return Mode.inputsMayBePositiveZero() || Mode.outputsMayBePositiveZero();
999}
1000
1001/// \returns the compare predicate type if the test performed by
1002/// llvm.is.fpclass(x, \p Mask) is equivalent to fcmp o__ x, 0.0 with the
1003/// floating-point environment assumed for \p F for type \p Ty
1005 const Function &F, Type *Ty) {
1006 switch (static_cast<unsigned>(Mask)) {
1007 case fcZero:
1008 if (inputDenormalIsIEEE(F, Ty))
1009 return FCmpInst::FCMP_OEQ;
1010 break;
1011 case fcZero | fcSubnormal:
1012 if (inputDenormalIsDAZ(F, Ty))
1013 return FCmpInst::FCMP_OEQ;
1014 break;
1015 case fcPositive | fcNegZero:
1016 if (inputDenormalIsIEEE(F, Ty))
1017 return FCmpInst::FCMP_OGE;
1018 break;
1020 if (inputDenormalIsDAZ(F, Ty))
1021 return FCmpInst::FCMP_OGE;
1022 break;
1024 if (inputDenormalIsIEEE(F, Ty))
1025 return FCmpInst::FCMP_OGT;
1026 break;
1027 case fcNegative | fcPosZero:
1028 if (inputDenormalIsIEEE(F, Ty))
1029 return FCmpInst::FCMP_OLE;
1030 break;
1032 if (inputDenormalIsDAZ(F, Ty))
1033 return FCmpInst::FCMP_OLE;
1034 break;
1036 if (inputDenormalIsIEEE(F, Ty))
1037 return FCmpInst::FCMP_OLT;
1038 break;
1039 case fcPosNormal | fcPosInf:
1040 if (inputDenormalIsDAZ(F, Ty))
1041 return FCmpInst::FCMP_OGT;
1042 break;
1043 case fcNegNormal | fcNegInf:
1044 if (inputDenormalIsDAZ(F, Ty))
1045 return FCmpInst::FCMP_OLT;
1046 break;
1047 case ~fcZero & ~fcNan:
1048 if (inputDenormalIsIEEE(F, Ty))
1049 return FCmpInst::FCMP_ONE;
1050 break;
1051 case ~(fcZero | fcSubnormal) & ~fcNan:
1052 if (inputDenormalIsDAZ(F, Ty))
1053 return FCmpInst::FCMP_ONE;
1054 break;
1055 default:
1056 break;
1057 }
1058
1060}
1061
1062Instruction *InstCombinerImpl::foldIntrinsicIsFPClass(IntrinsicInst &II) {
1063 Value *Src0 = II.getArgOperand(0);
1064 Value *Src1 = II.getArgOperand(1);
1065 const ConstantInt *CMask = cast<ConstantInt>(Src1);
1066 FPClassTest Mask = static_cast<FPClassTest>(CMask->getZExtValue());
1067 const bool IsUnordered = (Mask & fcNan) == fcNan;
1068 const bool IsOrdered = (Mask & fcNan) == fcNone;
1069 const FPClassTest OrderedMask = Mask & ~fcNan;
1070 const FPClassTest OrderedInvertedMask = ~OrderedMask & ~fcNan;
1071
1072 const bool IsStrict =
1073 II.getFunction()->getAttributes().hasFnAttr(Attribute::StrictFP);
1074
1075 Value *FNegSrc;
1076 // is.fpclass (fneg x), mask -> is.fpclass x, (fneg mask)
1077 if (match(Src0, m_FNeg(m_Value(FNegSrc))))
1078 return CallInst::Create(
1079 II.getCalledFunction(),
1080 {FNegSrc, ConstantInt::get(Src1->getType(), fneg(Mask))});
1081
1082 Value *FAbsSrc;
1083 if (match(Src0, m_FAbs(m_Value(FAbsSrc))))
1084 return CallInst::Create(
1085 II.getCalledFunction(),
1086 {FAbsSrc, ConstantInt::get(Src1->getType(), inverse_fabs(Mask))});
1087
1088 if ((OrderedMask == fcInf || OrderedInvertedMask == fcInf) &&
1089 (IsOrdered || IsUnordered) && !IsStrict) {
1090 // is.fpclass(x, fcInf) -> fcmp oeq fabs(x), +inf
1091 // is.fpclass(x, ~fcInf) -> fcmp one fabs(x), +inf
1092 // is.fpclass(x, fcInf|fcNan) -> fcmp ueq fabs(x), +inf
1093 // is.fpclass(x, ~(fcInf|fcNan)) -> fcmp une fabs(x), +inf
1095 FCmpInst::Predicate Pred =
1096 IsUnordered ? FCmpInst::FCMP_UEQ : FCmpInst::FCMP_OEQ;
1097 if (OrderedInvertedMask == fcInf)
1098 Pred = IsUnordered ? FCmpInst::FCMP_UNE : FCmpInst::FCMP_ONE;
1099
1100 Value *Fabs = Builder.CreateFAbs(Src0);
1101 Value *CmpInf = Builder.CreateFCmp(Pred, Fabs, Inf);
1102 CmpInf->takeName(&II);
1103 return replaceInstUsesWith(II, CmpInf);
1104 }
1105
1106 if ((OrderedMask == fcPosInf || OrderedMask == fcNegInf) &&
1107 (IsOrdered || IsUnordered) && !IsStrict) {
1108 // is.fpclass(x, fcPosInf) -> fcmp oeq x, +inf
1109 // is.fpclass(x, fcNegInf) -> fcmp oeq x, -inf
1110 // is.fpclass(x, fcPosInf|fcNan) -> fcmp ueq x, +inf
1111 // is.fpclass(x, fcNegInf|fcNan) -> fcmp ueq x, -inf
1112 Constant *Inf =
1113 ConstantFP::getInfinity(Src0->getType(), OrderedMask == fcNegInf);
1114 Value *EqInf = IsUnordered ? Builder.CreateFCmpUEQ(Src0, Inf)
1115 : Builder.CreateFCmpOEQ(Src0, Inf);
1116
1117 EqInf->takeName(&II);
1118 return replaceInstUsesWith(II, EqInf);
1119 }
1120
1121 if ((OrderedInvertedMask == fcPosInf || OrderedInvertedMask == fcNegInf) &&
1122 (IsOrdered || IsUnordered) && !IsStrict) {
1123 // is.fpclass(x, ~fcPosInf) -> fcmp one x, +inf
1124 // is.fpclass(x, ~fcNegInf) -> fcmp one x, -inf
1125 // is.fpclass(x, ~fcPosInf|fcNan) -> fcmp une x, +inf
1126 // is.fpclass(x, ~fcNegInf|fcNan) -> fcmp une x, -inf
1128 OrderedInvertedMask == fcNegInf);
1129 Value *NeInf = IsUnordered ? Builder.CreateFCmpUNE(Src0, Inf)
1130 : Builder.CreateFCmpONE(Src0, Inf);
1131 NeInf->takeName(&II);
1132 return replaceInstUsesWith(II, NeInf);
1133 }
1134
1135 if (Mask == fcNan && !IsStrict) {
1136 // Equivalent of isnan. Replace with standard fcmp if we don't care about FP
1137 // exceptions.
1138 Value *IsNan =
1139 Builder.CreateFCmpUNO(Src0, ConstantFP::getZero(Src0->getType()));
1140 IsNan->takeName(&II);
1141 return replaceInstUsesWith(II, IsNan);
1142 }
1143
1144 if (Mask == (~fcNan & fcAllFlags) && !IsStrict) {
1145 // Equivalent of !isnan. Replace with standard fcmp.
1146 Value *FCmp =
1147 Builder.CreateFCmpORD(Src0, ConstantFP::getZero(Src0->getType()));
1148 FCmp->takeName(&II);
1149 return replaceInstUsesWith(II, FCmp);
1150 }
1151
1153
1154 // Try to replace with an fcmp with 0
1155 //
1156 // is.fpclass(x, fcZero) -> fcmp oeq x, 0.0
1157 // is.fpclass(x, fcZero | fcNan) -> fcmp ueq x, 0.0
1158 // is.fpclass(x, ~fcZero & ~fcNan) -> fcmp one x, 0.0
1159 // is.fpclass(x, ~fcZero) -> fcmp une x, 0.0
1160 //
1161 // is.fpclass(x, fcPosSubnormal | fcPosNormal | fcPosInf) -> fcmp ogt x, 0.0
1162 // is.fpclass(x, fcPositive | fcNegZero) -> fcmp oge x, 0.0
1163 //
1164 // is.fpclass(x, fcNegSubnormal | fcNegNormal | fcNegInf) -> fcmp olt x, 0.0
1165 // is.fpclass(x, fcNegative | fcPosZero) -> fcmp ole x, 0.0
1166 //
1167 if (!IsStrict && (IsOrdered || IsUnordered) &&
1168 (PredType = fpclassTestIsFCmp0(OrderedMask, *II.getFunction(),
1169 Src0->getType())) !=
1172 // Equivalent of == 0.
1173 Value *FCmp = Builder.CreateFCmp(
1174 IsUnordered ? FCmpInst::getUnorderedPredicate(PredType) : PredType,
1175 Src0, Zero);
1176
1177 FCmp->takeName(&II);
1178 return replaceInstUsesWith(II, FCmp);
1179 }
1180
1181 KnownFPClass Known =
1182 computeKnownFPClass(Src0, Mask, SQ.getWithInstruction(&II));
1183
1184 // If none of the tests which can return false are possible, fold to true.
1185 // fp_class (nnan x), ~(qnan|snan) -> true
1186 // fp_class (ninf x), ~(ninf|pinf) -> true
1187 if (Known.isKnownAlways(Mask))
1188 return replaceInstUsesWith(II, ConstantInt::get(II.getType(), true));
1189
1190 // Clear test bits we know must be false from the source value.
1191 // fp_class (nnan x), qnan|snan|other -> fp_class (nnan x), other
1192 // fp_class (ninf x), ninf|pinf|other -> fp_class (ninf x), other
1193 if ((Mask & Known.getKnownFPClasses()) != Mask) {
1194 II.setArgOperand(
1195 1, ConstantInt::get(Src1->getType(), Mask & Known.getKnownFPClasses()));
1196 return &II;
1197 }
1198
1199 return nullptr;
1200}
1201
1202static std::optional<bool> getKnownSign(Value *Op, const SimplifyQuery &SQ) {
1204 if (Known.isNonNegative())
1205 return false;
1206 if (Known.isNegative())
1207 return true;
1208
1209 Value *X, *Y;
1210 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1212
1213 return std::nullopt;
1214}
1215
1216static std::optional<bool> getKnownSignOrZero(Value *Op,
1217 const SimplifyQuery &SQ) {
1218 if (std::optional<bool> Sign = getKnownSign(Op, SQ))
1219 return Sign;
1220
1221 Value *X, *Y;
1222 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1224
1225 return std::nullopt;
1226}
1227
1228/// Return true if two values \p Op0 and \p Op1 are known to have the same sign.
1229static bool signBitMustBeTheSame(Value *Op0, Value *Op1,
1230 const SimplifyQuery &SQ) {
1231 std::optional<bool> Known1 = getKnownSign(Op1, SQ);
1232 if (!Known1)
1233 return false;
1234 std::optional<bool> Known0 = getKnownSign(Op0, SQ);
1235 if (!Known0)
1236 return false;
1237 return *Known0 == *Known1;
1238}
1239
1240// Determines if ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b)) is safe.
1241//
1242// This is true if, when the add saturates, the resulting ldexp is guaranteed to
1243// produce 0 or inf.
1244static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy) {
1245 const fltSemantics &FltSem = FpTy->getScalarType()->getFltSemantics();
1246 if (!APFloat::semanticsHasInf(FltSem))
1247 return false;
1248
1249 // Cap ExpBits at 32 because scalbn takes an int. This is sufficient for any
1250 // reasonable fp type (for example, `double` only has 11 exponent bits).
1251 unsigned ExpBits = std::min(ExpTy->getScalarSizeInBits(), 32u);
1252 int SignedMax = static_cast<int>(maxIntN(ExpBits));
1253 int SignedMin = static_cast<int>(minIntN(ExpBits));
1254 APFloat ScaledUp = scalbn(APFloat::getSmallest(FltSem), SignedMax,
1256 APFloat ScaledDown = scalbn(APFloat::getLargest(FltSem), SignedMin,
1258 return ScaledUp.isInfinity() && ScaledDown.isZero();
1259}
1260
1261/// Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0. This
1262/// can trigger other combines.
1264 InstCombiner::BuilderTy &Builder) {
1265 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1266 assert((MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin ||
1267 MinMaxID == Intrinsic::umax || MinMaxID == Intrinsic::umin) &&
1268 "Expected a min or max intrinsic");
1269
1270 // TODO: Match vectors with undef elements, but undef may not propagate.
1271 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
1272 Value *X;
1273 const APInt *C0, *C1;
1274 if (!match(Op0, m_OneUse(m_Add(m_Value(X), m_APInt(C0)))) ||
1275 !match(Op1, m_APInt(C1)))
1276 return nullptr;
1277
1278 // Check for necessary no-wrap and overflow constraints.
1279 bool IsSigned = MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin;
1280 auto *Add = cast<BinaryOperator>(Op0);
1281 if ((IsSigned && !Add->hasNoSignedWrap()) ||
1282 (!IsSigned && !Add->hasNoUnsignedWrap()))
1283 return nullptr;
1284
1285 // If the constant difference overflows, then instsimplify should reduce the
1286 // min/max to the add or C1.
1287 bool Overflow;
1288 APInt CDiff =
1289 IsSigned ? C1->ssub_ov(*C0, Overflow) : C1->usub_ov(*C0, Overflow);
1290 assert(!Overflow && "Expected simplify of min/max");
1291
1292 // min/max (add X, C0), C1 --> add (min/max X, C1 - C0), C0
1293 // Note: the "mismatched" no-overflow setting does not propagate.
1294 Constant *NewMinMaxC = ConstantInt::get(II->getType(), CDiff);
1295 Value *NewMinMax = Builder.CreateBinaryIntrinsic(MinMaxID, X, NewMinMaxC);
1296 return IsSigned ? BinaryOperator::CreateNSWAdd(NewMinMax, Add->getOperand(1))
1297 : BinaryOperator::CreateNUWAdd(NewMinMax, Add->getOperand(1));
1298}
1299/// Match a sadd_sat or ssub_sat which is using min/max to clamp the value.
1300Instruction *InstCombinerImpl::matchSAddSubSat(IntrinsicInst &MinMax1) {
1301 Type *Ty = MinMax1.getType();
1302
1303 // We are looking for a tree of:
1304 // max(INT_MIN, min(INT_MAX, add(sext(A), sext(B))))
1305 // Where the min and max could be reversed
1306 Instruction *MinMax2;
1307 BinaryOperator *AddSub;
1308 const APInt *MinValue, *MaxValue;
1309 if (match(&MinMax1, m_SMin(m_Instruction(MinMax2), m_APInt(MaxValue)))) {
1310 if (!match(MinMax2, m_SMax(m_BinOp(AddSub), m_APInt(MinValue))))
1311 return nullptr;
1312 } else if (match(&MinMax1,
1313 m_SMax(m_Instruction(MinMax2), m_APInt(MinValue)))) {
1314 if (!match(MinMax2, m_SMin(m_BinOp(AddSub), m_APInt(MaxValue))))
1315 return nullptr;
1316 } else
1317 return nullptr;
1318
1319 // Check that the constants clamp a saturate, and that the new type would be
1320 // sensible to convert to.
1321 if (!(*MaxValue + 1).isPowerOf2() || -*MinValue != *MaxValue + 1)
1322 return nullptr;
1323 // In what bitwidth can this be treated as saturating arithmetics?
1324 unsigned NewBitWidth = (*MaxValue + 1).logBase2() + 1;
1325 // FIXME: This isn't quite right for vectors, but using the scalar type is a
1326 // good first approximation for what should be done there.
1327 if (!shouldChangeType(Ty->getScalarType()->getIntegerBitWidth(), NewBitWidth))
1328 return nullptr;
1329
1330 // Also make sure that the inner min/max and the add/sub have one use.
1331 if (!MinMax2->hasOneUse() || !AddSub->hasOneUse())
1332 return nullptr;
1333
1334 // Create the new type (which can be a vector type)
1335 Type *NewTy = Ty->getWithNewBitWidth(NewBitWidth);
1336
1337 Intrinsic::ID IntrinsicID;
1338 if (AddSub->getOpcode() == Instruction::Add)
1339 IntrinsicID = Intrinsic::sadd_sat;
1340 else if (AddSub->getOpcode() == Instruction::Sub)
1341 IntrinsicID = Intrinsic::ssub_sat;
1342 else
1343 return nullptr;
1344
1345 // The two operands of the add/sub must be nsw-truncatable to the NewTy. This
1346 // is usually achieved via a sext from a smaller type.
1347 if (ComputeMaxSignificantBits(AddSub->getOperand(0), AddSub) > NewBitWidth ||
1348 ComputeMaxSignificantBits(AddSub->getOperand(1), AddSub) > NewBitWidth)
1349 return nullptr;
1350
1351 // Finally create and return the sat intrinsic, truncated to the new type
1352 Value *AT = Builder.CreateTrunc(AddSub->getOperand(0), NewTy);
1353 Value *BT = Builder.CreateTrunc(AddSub->getOperand(1), NewTy);
1354 Value *Sat = Builder.CreateIntrinsic(IntrinsicID, NewTy, {AT, BT});
1355 return CastInst::Create(Instruction::SExt, Sat, Ty);
1356}
1357
1358
1359/// If we have a clamp pattern like max (min X, 42), 41 -- where the output
1360/// can only be one of two possible constant values -- turn that into a select
1361/// of constants.
1363 InstCombiner::BuilderTy &Builder) {
1364 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
1365 Value *X;
1366 const APInt *C0, *C1;
1367 if (!match(I1, m_APInt(C1)) || !I0->hasOneUse())
1368 return nullptr;
1369
1371 switch (II->getIntrinsicID()) {
1372 case Intrinsic::smax:
1373 if (match(I0, m_SMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1374 Pred = ICmpInst::ICMP_SGT;
1375 break;
1376 case Intrinsic::smin:
1377 if (match(I0, m_SMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1378 Pred = ICmpInst::ICMP_SLT;
1379 break;
1380 case Intrinsic::umax:
1381 if (match(I0, m_UMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1382 Pred = ICmpInst::ICMP_UGT;
1383 break;
1384 case Intrinsic::umin:
1385 if (match(I0, m_UMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1386 Pred = ICmpInst::ICMP_ULT;
1387 break;
1388 default:
1389 llvm_unreachable("Expected min/max intrinsic");
1390 }
1391 if (Pred == CmpInst::BAD_ICMP_PREDICATE)
1392 return nullptr;
1393
1394 // max (min X, 42), 41 --> X > 41 ? 42 : 41
1395 // min (max X, 42), 43 --> X < 43 ? 42 : 43
1396 Value *Cmp = Builder.CreateICmp(Pred, X, I1);
1397 return SelectInst::Create(Cmp, ConstantInt::get(II->getType(), *C0), I1);
1398}
1399
1400/// If this min/max has a constant operand and an operand that is a matching
1401/// min/max with a constant operand, constant-fold the 2 constant operands.
1403 IRBuilderBase &Builder,
1404 const SimplifyQuery &SQ) {
1405 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1406 auto *LHS = dyn_cast<MinMaxIntrinsic>(II->getArgOperand(0));
1407 if (!LHS)
1408 return nullptr;
1409
1410 Constant *C0, *C1;
1411 if (!match(LHS->getArgOperand(1), m_ImmConstant(C0)) ||
1412 !match(II->getArgOperand(1), m_ImmConstant(C1)))
1413 return nullptr;
1414
1415 // max (max X, C0), C1 --> max X, (max C0, C1)
1416 // min (min X, C0), C1 --> min X, (min C0, C1)
1417 // umax (smax X, nneg C0), nneg C1 --> smax X, (umax C0, C1)
1418 // smin (umin X, nneg C0), nneg C1 --> umin X, (smin C0, C1)
1419 Intrinsic::ID InnerMinMaxID = LHS->getIntrinsicID();
1420 if (InnerMinMaxID != MinMaxID &&
1421 !(((MinMaxID == Intrinsic::umax && InnerMinMaxID == Intrinsic::smax) ||
1422 (MinMaxID == Intrinsic::smin && InnerMinMaxID == Intrinsic::umin)) &&
1423 isKnownNonNegative(C0, SQ) && isKnownNonNegative(C1, SQ)))
1424 return nullptr;
1425
1427 Value *CondC = Builder.CreateICmp(Pred, C0, C1);
1428 Value *NewC = Builder.CreateSelect(CondC, C0, C1);
1429 return Builder.CreateIntrinsic(InnerMinMaxID, II->getType(),
1430 {LHS->getArgOperand(0), NewC});
1431}
1432
1433/// If this min/max has a matching min/max operand with a constant, try to push
1434/// the constant operand into this instruction. This can enable more folds.
1435static Instruction *
1437 InstCombiner::BuilderTy &Builder) {
1438 // Match and capture a min/max operand candidate.
1439 Value *X, *Y;
1440 Constant *C;
1441 Instruction *Inner;
1443 m_Instruction(Inner),
1445 m_Value(Y))))
1446 return nullptr;
1447
1448 // The inner op must match. Check for constants to avoid infinite loops.
1449 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1450 auto *InnerMM = dyn_cast<IntrinsicInst>(Inner);
1451 if (!InnerMM || InnerMM->getIntrinsicID() != MinMaxID ||
1453 return nullptr;
1454
1455 // max (max X, C), Y --> max (max X, Y), C
1457 MinMaxID, II->getType());
1458 Value *NewInner = Builder.CreateBinaryIntrinsic(MinMaxID, X, Y);
1459 NewInner->takeName(Inner);
1460 return CallInst::Create(MinMax, {NewInner, C});
1461}
1462
1463/// Reduce a sequence of min/max intrinsics with a common operand.
1465 // Match 3 of the same min/max ops. Example: umin(umin(), umin()).
1466 auto *LHS = dyn_cast<IntrinsicInst>(II->getArgOperand(0));
1467 auto *RHS = dyn_cast<IntrinsicInst>(II->getArgOperand(1));
1468 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1469 if (!LHS || !RHS || LHS->getIntrinsicID() != MinMaxID ||
1470 RHS->getIntrinsicID() != MinMaxID ||
1471 (!LHS->hasOneUse() && !RHS->hasOneUse()))
1472 return nullptr;
1473
1474 Value *A = LHS->getArgOperand(0);
1475 Value *B = LHS->getArgOperand(1);
1476 Value *C = RHS->getArgOperand(0);
1477 Value *D = RHS->getArgOperand(1);
1478
1479 // Look for a common operand.
1480 Value *MinMaxOp = nullptr;
1481 Value *ThirdOp = nullptr;
1482 if (LHS->hasOneUse()) {
1483 // If the LHS is only used in this chain and the RHS is used outside of it,
1484 // reuse the RHS min/max because that will eliminate the LHS.
1485 if (D == A || C == A) {
1486 // min(min(a, b), min(c, a)) --> min(min(c, a), b)
1487 // min(min(a, b), min(a, d)) --> min(min(a, d), b)
1488 MinMaxOp = RHS;
1489 ThirdOp = B;
1490 } else if (D == B || C == B) {
1491 // min(min(a, b), min(c, b)) --> min(min(c, b), a)
1492 // min(min(a, b), min(b, d)) --> min(min(b, d), a)
1493 MinMaxOp = RHS;
1494 ThirdOp = A;
1495 }
1496 } else {
1497 assert(RHS->hasOneUse() && "Expected one-use operand");
1498 // Reuse the LHS. This will eliminate the RHS.
1499 if (D == A || D == B) {
1500 // min(min(a, b), min(c, a)) --> min(min(a, b), c)
1501 // min(min(a, b), min(c, b)) --> min(min(a, b), c)
1502 MinMaxOp = LHS;
1503 ThirdOp = C;
1504 } else if (C == A || C == B) {
1505 // min(min(a, b), min(b, d)) --> min(min(a, b), d)
1506 // min(min(a, b), min(c, b)) --> min(min(a, b), d)
1507 MinMaxOp = LHS;
1508 ThirdOp = D;
1509 }
1510 }
1511
1512 if (!MinMaxOp || !ThirdOp)
1513 return nullptr;
1514
1515 Module *Mod = II->getModule();
1516 Function *MinMax =
1517 Intrinsic::getOrInsertDeclaration(Mod, MinMaxID, II->getType());
1518 return CallInst::Create(MinMax, { MinMaxOp, ThirdOp });
1519}
1520
1521/// If all arguments of the intrinsic are unary shuffles with the same mask,
1522/// try to shuffle after the intrinsic.
1525 if (!II->getType()->isVectorTy() ||
1526 !isTriviallyVectorizable(II->getIntrinsicID()) ||
1527 !II->getCalledFunction()->isSpeculatable())
1528 return nullptr;
1529
1530 Value *X;
1531 Constant *C;
1532 ArrayRef<int> Mask;
1533 auto *NonConstArg = find_if_not(II->args(), [&II](Use &Arg) {
1534 return isa<Constant>(Arg.get()) ||
1535 isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1536 Arg.getOperandNo(), nullptr);
1537 });
1538 if (!NonConstArg ||
1539 !match(NonConstArg, m_Shuffle(m_Value(X), m_Poison(), m_Mask(Mask))))
1540 return nullptr;
1541
1542 // At least 1 operand must be a shuffle with 1 use because we are creating 2
1543 // instructions.
1544 if (none_of(II->args(), match_fn(m_OneUse(m_Shuffle(m_Value(), m_Value())))))
1545 return nullptr;
1546
1547 // See if all arguments are shuffled with the same mask.
1549 Type *SrcTy = X->getType();
1550 for (Use &Arg : II->args()) {
1551 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1552 Arg.getOperandNo(), nullptr))
1553 NewArgs.push_back(Arg);
1554 else if (match(&Arg,
1555 m_Shuffle(m_Value(X), m_Poison(), m_SpecificMask(Mask))) &&
1556 X->getType() == SrcTy)
1557 NewArgs.push_back(X);
1558 else if (match(&Arg, m_ImmConstant(C))) {
1559 // If it's a constant, try find the constant that would be shuffled to C.
1560 if (Constant *ShuffledC =
1561 unshuffleConstant(Mask, C, cast<VectorType>(SrcTy)))
1562 NewArgs.push_back(ShuffledC);
1563 else
1564 return nullptr;
1565 } else
1566 return nullptr;
1567 }
1568
1569 // intrinsic (shuf X, M), (shuf Y, M), ... --> shuf (intrinsic X, Y, ...), M
1570 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1571 // Result type might be a different vector width.
1572 // TODO: Check that the result type isn't widened?
1573 VectorType *ResTy =
1574 VectorType::get(II->getType()->getScalarType(), cast<VectorType>(SrcTy));
1575 Value *NewIntrinsic =
1576 Builder.CreateIntrinsic(ResTy, II->getIntrinsicID(), NewArgs, FPI);
1577 return new ShuffleVectorInst(NewIntrinsic, Mask);
1578}
1579
1580/// If all arguments of the intrinsic are reverses, try to pull the reverse
1581/// after the intrinsic.
1583 if (!II->getType()->isVectorTy() ||
1584 !isTriviallyVectorizable(II->getIntrinsicID()))
1585 return nullptr;
1586
1587 // At least 1 operand must be a reverse with 1 use because we are creating 2
1588 // instructions.
1589 if (none_of(II->args(), [](Value *V) {
1590 return match(V, m_OneUse(m_VecReverse(m_Value())));
1591 }))
1592 return nullptr;
1593
1594 Value *X;
1595 Constant *C;
1596 SmallVector<Value *> NewArgs;
1597 for (Use &Arg : II->args()) {
1598 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1599 Arg.getOperandNo(), nullptr))
1600 NewArgs.push_back(Arg);
1601 else if (match(&Arg, m_VecReverse(m_Value(X))))
1602 NewArgs.push_back(X);
1603 else if (isSplatValue(Arg))
1604 NewArgs.push_back(Arg);
1605 else if (match(&Arg, m_ImmConstant(C)))
1606 NewArgs.push_back(Builder.CreateVectorReverse(C));
1607 else
1608 return nullptr;
1609 }
1610
1611 // intrinsic (reverse X), (reverse Y), ... --> reverse (intrinsic X, Y, ...)
1612 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1613 Value *NewIntrinsic = Builder.CreateIntrinsic(
1614 II->getType(), II->getIntrinsicID(), NewArgs, FPI);
1615 return Builder.CreateVectorReverse(NewIntrinsic);
1616}
1617
1618/// Fold the following cases and accepts bswap and bitreverse intrinsics:
1619/// bswap(logic_op(bswap(x), y)) --> logic_op(x, bswap(y))
1620/// bswap(logic_op(bswap(x), bswap(y))) --> logic_op(x, y) (ignores multiuse)
1621template <Intrinsic::ID IntrID>
1623 InstCombiner::BuilderTy &Builder) {
1624 static_assert(IntrID == Intrinsic::bswap || IntrID == Intrinsic::bitreverse,
1625 "This helper only supports BSWAP and BITREVERSE intrinsics");
1626
1627 Value *X, *Y;
1628 // Find bitwise logic op. Check that it is a BinaryOperator explicitly so we
1629 // don't match ConstantExpr that aren't meaningful for this transform.
1632 Value *OldReorderX, *OldReorderY;
1634
1635 // If both X and Y are bswap/bitreverse, the transform reduces the number
1636 // of instructions even if there's multiuse.
1637 // If only one operand is bswap/bitreverse, we need to ensure the operand
1638 // have only one use.
1639 if (match(X, m_Intrinsic<IntrID>(m_Value(OldReorderX))) &&
1640 match(Y, m_Intrinsic<IntrID>(m_Value(OldReorderY)))) {
1641 return BinaryOperator::Create(Op, OldReorderX, OldReorderY);
1642 }
1643
1644 if (match(X, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderX))))) {
1645 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, Y);
1646 return BinaryOperator::Create(Op, OldReorderX, NewReorder);
1647 }
1648
1649 if (match(Y, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderY))))) {
1650 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, X);
1651 return BinaryOperator::Create(Op, NewReorder, OldReorderY);
1652 }
1653 }
1654 return nullptr;
1655}
1656
1657/// Helper to match idempotent binary intrinsics, namely, intrinsics where
1658/// `f(f(x, y), y) == f(x, y)` holds.
1660 switch (IID) {
1661 case Intrinsic::smax:
1662 case Intrinsic::smin:
1663 case Intrinsic::umax:
1664 case Intrinsic::umin:
1665 case Intrinsic::maximum:
1666 case Intrinsic::minimum:
1667 case Intrinsic::maximumnum:
1668 case Intrinsic::minimumnum:
1669 case Intrinsic::maxnum:
1670 case Intrinsic::minnum:
1671 return true;
1672 default:
1673 return false;
1674 }
1675}
1676
1677/// Attempt to simplify value-accumulating recurrences of kind:
1678/// %umax.acc = phi i8 [ %umax, %backedge ], [ %a, %entry ]
1679/// %umax = call i8 @llvm.umax.i8(i8 %umax.acc, i8 %b)
1680/// And let the idempotent binary intrinsic be hoisted, when the operands are
1681/// known to be loop-invariant.
1683 IntrinsicInst *II) {
1684 PHINode *PN;
1685 Value *Init, *OtherOp;
1686
1687 // A binary intrinsic recurrence with loop-invariant operands is equivalent to
1688 // `call @llvm.binary.intrinsic(Init, OtherOp)`.
1689 auto IID = II->getIntrinsicID();
1690 if (!isIdempotentBinaryIntrinsic(IID) ||
1692 !IC.getDominatorTree().dominates(OtherOp, PN))
1693 return nullptr;
1694
1695 auto *InvariantBinaryInst =
1696 IC.Builder.CreateBinaryIntrinsic(IID, Init, OtherOp);
1697 if (isa<FPMathOperator>(InvariantBinaryInst))
1698 cast<Instruction>(InvariantBinaryInst)->copyFastMathFlags(II);
1699 return InvariantBinaryInst;
1700}
1701
1702static Value *simplifyReductionOperand(Value *Arg, bool CanReorderLanes) {
1703 if (!CanReorderLanes)
1704 return nullptr;
1705
1706 Value *V;
1707 if (match(Arg, m_VecReverse(m_Value(V))))
1708 return V;
1709
1710 ArrayRef<int> Mask;
1711 if (!isa<FixedVectorType>(Arg->getType()) ||
1712 !match(Arg, m_Shuffle(m_Value(V), m_Undef(), m_Mask(Mask))) ||
1713 !cast<ShuffleVectorInst>(Arg)->isSingleSource())
1714 return nullptr;
1715
1716 int Sz = Mask.size();
1717 SmallBitVector UsedIndices(Sz);
1718 for (int Idx : Mask) {
1719 if (Idx == PoisonMaskElem || UsedIndices.test(Idx))
1720 return nullptr;
1721 UsedIndices.set(Idx);
1722 }
1723
1724 // Can remove shuffle iff just shuffled elements, no repeats, undefs, or
1725 // other changes.
1726 return UsedIndices.all() ? V : nullptr;
1727}
1728
1729/// Fold an unsigned minimum of trailing or leading zero bits counts:
1730/// umin(cttz(CtOp1, ZeroUndef), ConstOp) --> cttz(CtOp1 | (1 << ConstOp))
1731/// umin(ctlz(CtOp1, ZeroUndef), ConstOp) --> ctlz(CtOp1 | (SignedMin
1732/// >> ConstOp))
1733/// umin(cttz(CtOp1), cttz(CtOp2)) --> cttz(CtOp1 | CtOp2)
1734/// umin(ctlz(CtOp1), ctlz(CtOp2)) --> ctlz(CtOp1 | CtOp2)
1735template <Intrinsic::ID IntrID>
1736static Value *
1738 const DataLayout &DL,
1739 InstCombiner::BuilderTy &Builder) {
1740 static_assert(IntrID == Intrinsic::cttz || IntrID == Intrinsic::ctlz,
1741 "This helper only supports cttz and ctlz intrinsics");
1742
1743 Value *CtOp1, *CtOp2;
1744 Value *ZeroUndef1, *ZeroUndef2;
1745 if (!match(I0, m_OneUse(
1746 m_Intrinsic<IntrID>(m_Value(CtOp1), m_Value(ZeroUndef1)))))
1747 return nullptr;
1748
1749 if (match(I1,
1750 m_OneUse(m_Intrinsic<IntrID>(m_Value(CtOp2), m_Value(ZeroUndef2)))))
1751 return Builder.CreateBinaryIntrinsic(
1752 IntrID, Builder.CreateOr(CtOp1, CtOp2),
1753 Builder.CreateOr(ZeroUndef1, ZeroUndef2));
1754
1755 unsigned BitWidth = I1->getType()->getScalarSizeInBits();
1756 auto LessBitWidth = [BitWidth](auto &C) { return C.ult(BitWidth); };
1757 if (!match(I1, m_CheckedInt(LessBitWidth)))
1758 // We have a constant >= BitWidth (which can be handled by CVP)
1759 // or a non-splat vector with elements < and >= BitWidth
1760 return nullptr;
1761
1762 Type *Ty = I1->getType();
1764 IntrID == Intrinsic::cttz ? Instruction::Shl : Instruction::LShr,
1765 IntrID == Intrinsic::cttz
1766 ? ConstantInt::get(Ty, 1)
1767 : ConstantInt::get(Ty, APInt::getSignedMinValue(BitWidth)),
1768 cast<Constant>(I1), DL);
1769 return Builder.CreateBinaryIntrinsic(
1770 IntrID, Builder.CreateOr(CtOp1, NewConst),
1771 ConstantInt::getTrue(ZeroUndef1->getType()));
1772}
1773
1774/// Return whether "X LOp (Y ROp Z)" is always equal to
1775/// "(X LOp Y) ROp (X LOp Z)".
1777 bool HasNSW, Intrinsic::ID ROp) {
1778 switch (ROp) {
1779 case Intrinsic::umax:
1780 case Intrinsic::umin:
1781 if (HasNUW && LOp == Instruction::Add)
1782 return true;
1783 if (HasNUW && LOp == Instruction::Shl)
1784 return true;
1785 return false;
1786 case Intrinsic::smax:
1787 case Intrinsic::smin:
1788 return HasNSW && LOp == Instruction::Add;
1789 default:
1790 return false;
1791 }
1792}
1793
1794/// Return whether "(X ROp Y) LOp Z" is always equal to
1795/// "(X LOp Z) ROp (Y LOp Z)".
1797 bool HasNSW, Intrinsic::ID ROp) {
1798 if (Instruction::isCommutative(LOp) || LOp == Instruction::Shl)
1799 return leftDistributesOverRight(LOp, HasNUW, HasNSW, ROp);
1800 switch (ROp) {
1801 case Intrinsic::umax:
1802 case Intrinsic::umin:
1803 return HasNUW && LOp == Instruction::Sub;
1804 case Intrinsic::smax:
1805 case Intrinsic::smin:
1806 return HasNSW && LOp == Instruction::Sub;
1807 default:
1808 return false;
1809 }
1810}
1811
1812// Attempts to factorise a common term
1813// in an instruction that has the form "(A op' B) op (C op' D)
1814// where op is an intrinsic and op' is a binop
1815static Value *
1817 InstCombiner::BuilderTy &Builder) {
1818 Value *LHS = II->getOperand(0), *RHS = II->getOperand(1);
1819 Intrinsic::ID TopLevelOpcode = II->getIntrinsicID();
1820
1823
1824 if (!Op0 || !Op1)
1825 return nullptr;
1826
1827 if (Op0->getOpcode() != Op1->getOpcode())
1828 return nullptr;
1829
1830 if (!Op0->hasOneUse() || !Op1->hasOneUse())
1831 return nullptr;
1832
1833 Instruction::BinaryOps InnerOpcode =
1834 static_cast<Instruction::BinaryOps>(Op0->getOpcode());
1835 bool HasNUW = Op0->hasNoUnsignedWrap() && Op1->hasNoUnsignedWrap();
1836 bool HasNSW = Op0->hasNoSignedWrap() && Op1->hasNoSignedWrap();
1837
1838 Value *A = Op0->getOperand(0);
1839 Value *B = Op0->getOperand(1);
1840 Value *C = Op1->getOperand(0);
1841 Value *D = Op1->getOperand(1);
1842
1843 // Attempts to swap variables such that A equals C or B equals D,
1844 // if the inner operation is commutative.
1845 if (Op0->isCommutative() && A != C && B != D) {
1846 if (A == D || B == C)
1847 std::swap(C, D);
1848 else
1849 return nullptr;
1850 }
1851
1852 if (A == C &&
1853 leftDistributesOverRight(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1854 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, B, D);
1855 return Builder.CreateNoWrapBinOp(InnerOpcode, A, NewIntrinsic, HasNUW,
1856 HasNSW);
1857 }
1858 if (B == D &&
1859 rightDistributesOverLeft(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1860 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, A, C);
1861 return Builder.CreateNoWrapBinOp(InnerOpcode, NewIntrinsic, B, HasNUW,
1862 HasNSW);
1863 }
1864 return nullptr;
1865}
1866
1868 Value *Arg0 = II->getArgOperand(0);
1869 auto *ShiftConst = dyn_cast<Constant>(II->getArgOperand(1));
1870 if (!ShiftConst)
1871 return nullptr;
1872
1873 int ElemBits = Arg0->getType()->getScalarSizeInBits();
1874 bool AllPositive = true;
1875 bool AllNegative = true;
1876
1877 auto Check = [&](Constant *C) -> bool {
1878 if (auto *CI = dyn_cast_or_null<ConstantInt>(C)) {
1879 const APInt &V = CI->getValue();
1880 if (V.isNonNegative()) {
1881 AllNegative = false;
1882 return AllPositive && V.ult(ElemBits);
1883 }
1884 AllPositive = false;
1885 return AllNegative && V.sgt(-ElemBits);
1886 }
1887 return false;
1888 };
1889
1890 if (auto *VTy = dyn_cast<FixedVectorType>(Arg0->getType())) {
1891 for (unsigned I = 0, E = VTy->getNumElements(); I < E; ++I) {
1892 if (!Check(ShiftConst->getAggregateElement(I)))
1893 return nullptr;
1894 }
1895
1896 } else if (!Check(ShiftConst))
1897 return nullptr;
1898
1899 IRBuilderBase &B = IC.Builder;
1900 if (AllPositive)
1901 return IC.replaceInstUsesWith(*II, B.CreateShl(Arg0, ShiftConst));
1902
1903 Value *NegAmt = B.CreateNeg(ShiftConst);
1904 Intrinsic::ID IID = II->getIntrinsicID();
1905 const bool IsSigned =
1906 IID == Intrinsic::arm_neon_vshifts || IID == Intrinsic::aarch64_neon_sshl;
1907 Value *Result =
1908 IsSigned ? B.CreateAShr(Arg0, NegAmt) : B.CreateLShr(Arg0, NegAmt);
1909 return IC.replaceInstUsesWith(*II, Result);
1910}
1911
1912// If II is llvm.sin(x) or llvm.cos(x), and there is a matching
1913// llvm.cos(x) or llvm.sin(x) using the same argument, combine them
1914// into a single llvm.sincos(x) call. Returns the result for II
1915// extracted from sincos, or nullptr if no match is found.
1917 InstCombinerImpl &IC) {
1918 Intrinsic::ID IID = II->getIntrinsicID();
1919 bool IsSin = IID == Intrinsic::sin;
1920 Intrinsic::ID MatchID = IsSin ? Intrinsic::cos : Intrinsic::sin;
1921
1922 Value *Arg = II->getArgOperand(0);
1923
1924 // Don't bother looking through uses of constants.
1925 if (isa<Constant>(Arg))
1926 return nullptr;
1927
1928 // Look for a matching cos/sin intrinsic with the same argument.
1929 IntrinsicInst *Match = nullptr;
1930 for (User *U : Arg->users()) {
1931 if (auto *Cand = dyn_cast<IntrinsicInst>(U)) {
1932 if (Cand != II && !Cand->use_empty() &&
1933 Cand->getIntrinsicID() == MatchID) {
1934 Match = Cand;
1935 break;
1936 }
1937 }
1938 }
1939
1940 if (!Match)
1941 return nullptr;
1942
1943 // Insert sincos right after the argument definition.
1945 if (auto *ArgInst = dyn_cast<Instruction>(Arg)) {
1946 std::optional<BasicBlock::iterator> InsertPt =
1947 ArgInst->getInsertionPointAfterDef();
1948 if (!InsertPt)
1949 return nullptr;
1950 B.SetInsertPoint(*InsertPt);
1951 } else {
1952 BasicBlock &EntryBB = II->getFunction()->getEntryBlock();
1953 B.SetInsertPoint(&EntryBB, EntryBB.begin());
1954 }
1955
1957 II->getModule(), Intrinsic::sincos, Arg->getType());
1958 CallInst *SinCos = B.CreateCall(SinCosFunc, Arg, "sincos");
1959 // Intersect fast-math flags from the two calls.
1960 SinCos->setFastMathFlags(II->getFastMathFlags() & Match->getFastMathFlags());
1961 // Propagate the most-generic fpmath metadata from the two original calls.
1963 II->getMetadata(LLVMContext::MD_fpmath),
1964 Match->getMetadata(LLVMContext::MD_fpmath)))
1965 SinCos->setMetadata(LLVMContext::MD_fpmath, MD);
1966 Value *Sin = B.CreateExtractValue(SinCos, 0, "sin");
1967 Value *Cos = B.CreateExtractValue(SinCos, 1, "cos");
1968
1969 // Replace the matching call and erase it.
1970 IC.replaceInstUsesWith(*Match, IsSin ? Cos : Sin);
1971 IC.eraseInstFromFunction(*Match);
1972 return IsSin ? Sin : Cos;
1973}
1974
1975/// Fold an scmp/ucmp intrinsic whose operands are extended from a narrower
1976/// type:
1977/// scmp (sext X), (sext Y) --> scmp X, Y
1978/// scmp (zext X), (zext Y) --> ucmp X, Y
1979/// ucmp (ext X), (ext Y) --> ucmp X, Y
1980/// Both operands must use the same extend opcode and source type. A constant
1981/// operand is narrowed instead, if truncating and re-extending it gives back
1982/// the same constant.
1984 InstCombiner::BuilderTy &Builder,
1985 const DataLayout &DL) {
1986 // scmp/ucmp are not commutative, so the extend may be on either side.
1987 unsigned ExtIdx = 0;
1988 Value *X;
1989 if (!match(II->getArgOperand(0), m_ZExtOrSExt(m_Value(X)))) {
1990 ExtIdx = 1;
1991 if (!match(II->getArgOperand(1), m_ZExtOrSExt(m_Value(X))))
1992 return nullptr;
1993 }
1994
1995 auto CastOpc = static_cast<Instruction::CastOps>(
1996 cast<Operator>(II->getArgOperand(ExtIdx))->getOpcode());
1997 Type *NarrowTy = X->getType();
1998
1999 // The other operand must be the same kind of extend from the same type, or a
2000 // constant that can be narrowed losslessly.
2001 Value *OtherOp = II->getArgOperand(1 - ExtIdx);
2002 Value *Y;
2003 Constant *WideC;
2004 if (match(OtherOp, m_ZExtOrSExt(m_Value(Y)))) {
2005 if (cast<Operator>(OtherOp)->getOpcode() != CastOpc ||
2006 Y->getType() != NarrowTy)
2007 return nullptr;
2008 } else if (match(OtherOp, m_ImmConstant(WideC))) {
2009 Y = getLosslessInvCast(WideC, NarrowTy, CastOpc, DL);
2010 if (!Y)
2011 return nullptr;
2012 } else {
2013 return nullptr;
2014 }
2015
2016 // Both extends preserve the unsigned order, so an unsigned compare of the
2017 // narrow operands is always equivalent. The signed order is only preserved by
2018 // sext; zero extended values are non-negative, so a signed compare of those
2019 // is an unsigned compare of the narrow operands.
2020 Intrinsic::ID NewIID =
2021 II->getIntrinsicID() == Intrinsic::scmp && CastOpc == Instruction::SExt
2022 ? Intrinsic::scmp
2023 : Intrinsic::ucmp;
2024 if (ExtIdx != 0)
2025 std::swap(X, Y);
2026 return Builder.CreateIntrinsic(II->getType(), NewIID, {X, Y});
2027}
2028
2029/// CallInst simplification. This mostly only handles folding of intrinsic
2030/// instructions. For normal calls, it allows visitCallBase to do the heavy
2031/// lifting.
2033 // Don't try to simplify calls without uses. It will not do anything useful,
2034 // but will result in the following folds being skipped.
2035 if (!CI.use_empty()) {
2036 SmallVector<Value *, 8> Args(CI.args());
2037 if (Value *V = simplifyCall(&CI, CI.getCalledOperand(), Args,
2038 SQ.getWithInstruction(&CI)))
2039 return replaceInstUsesWith(CI, V);
2040 }
2041
2042 if (Value *FreedOp = getFreedOperand(&CI, &TLI))
2043 return visitFree(CI, FreedOp);
2044
2045 // If the caller function (i.e. us, the function that contains this CallInst)
2046 // is nounwind, mark the call as nounwind, even if the callee isn't.
2047 if (CI.getFunction()->doesNotThrow() && !CI.doesNotThrow()) {
2048 CI.setDoesNotThrow();
2049 return &CI;
2050 }
2051
2053 if (!II)
2054 return visitCallBase(CI);
2055
2056 // Intrinsics cannot occur in an invoke or a callbr, so handle them here
2057 // instead of in visitCallBase.
2058 if (auto *MI = dyn_cast<AnyMemIntrinsic>(II)) {
2059 if (auto NumBytes = MI->getLengthInBytes()) {
2060 // memmove/cpy/set of zero bytes is a noop.
2061 if (NumBytes->isZero())
2062 return eraseInstFromFunction(CI);
2063
2064 // For atomic unordered mem intrinsics if len is not a positive or
2065 // not a multiple of element size then behavior is undefined.
2066 if (MI->isAtomic() &&
2067 (NumBytes->isNegative() ||
2068 (NumBytes->getZExtValue() % MI->getElementSizeInBytes() != 0))) {
2070 assert(MI->getType()->isVoidTy() &&
2071 "non void atomic unordered mem intrinsic");
2072 return eraseInstFromFunction(*MI);
2073 }
2074 }
2075
2076 // No other transformations apply to volatile transfers.
2077 if (MI->isVolatile())
2078 return nullptr;
2079
2081 // memmove(x,x,size) -> noop.
2082 if (MTI->getSource() == MTI->getDest())
2083 return eraseInstFromFunction(CI);
2084 }
2085
2086 auto IsPointerUndefined = [MI](Value *Ptr) {
2087 return isa<ConstantPointerNull>(Ptr) &&
2089 MI->getFunction(),
2090 cast<PointerType>(Ptr->getType())->getAddressSpace());
2091 };
2092 bool SrcIsUndefined = false;
2093 // If we can determine a pointer alignment that is bigger than currently
2094 // set, update the alignment.
2095 if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI)) {
2097 return I;
2098 SrcIsUndefined = IsPointerUndefined(MTI->getRawSource());
2099 } else if (auto *MSI = dyn_cast<AnyMemSetInst>(MI)) {
2100 if (Instruction *I = SimplifyAnyMemSet(MSI))
2101 return I;
2102 }
2103
2104 // If src/dest is null, this memory intrinsic must be a noop.
2105 if (SrcIsUndefined || IsPointerUndefined(MI->getRawDest())) {
2106 Builder.CreateAssumption(Builder.CreateIsNull(MI->getLength()));
2107 return eraseInstFromFunction(CI);
2108 }
2109
2110 // If we have a memmove and the source operation is a constant global,
2111 // then the source and dest pointers can't alias, so we can change this
2112 // into a call to memcpy.
2113 if (auto *MMI = dyn_cast<AnyMemMoveInst>(MI)) {
2114 if (GlobalVariable *GVSrc = dyn_cast<GlobalVariable>(MMI->getSource()))
2115 if (GVSrc->isConstant()) {
2116 Module *M = CI.getModule();
2117 Intrinsic::ID MemCpyID =
2118 MMI->isAtomic()
2119 ? Intrinsic::memcpy_element_unordered_atomic
2120 : Intrinsic::memcpy;
2121 Type *Tys[3] = { CI.getArgOperand(0)->getType(),
2122 CI.getArgOperand(1)->getType(),
2123 CI.getArgOperand(2)->getType() };
2125 Intrinsic::getOrInsertDeclaration(M, MemCpyID, Tys));
2126 return II;
2127 }
2128 }
2129 }
2130
2131 // For fixed width vector result intrinsics, use the generic demanded vector
2132 // support.
2133 if (auto *IIFVTy = dyn_cast<FixedVectorType>(II->getType())) {
2134 auto VWidth = IIFVTy->getNumElements();
2135 APInt PoisonElts(VWidth, 0);
2136 APInt AllOnesEltMask(APInt::getAllOnes(VWidth));
2137 if (Value *V = SimplifyDemandedVectorElts(II, AllOnesEltMask, PoisonElts)) {
2138 if (V != II)
2139 return replaceInstUsesWith(*II, V);
2140 return II;
2141 }
2142 }
2143
2144 if (II->isCommutative()) {
2145 if (auto Pair = matchSymmetricPair(II->getOperand(0), II->getOperand(1))) {
2146 replaceOperand(*II, 0, Pair->first);
2147 replaceOperand(*II, 1, Pair->second);
2148 II->dropPoisonGeneratingAnnotations();
2149 II->dropUBImplyingAttrsAndMetadata();
2150 return II;
2151 }
2152
2153 if (CallInst *NewCall = canonicalizeConstantArg0ToArg1(CI))
2154 return NewCall;
2155 }
2156
2157 // Unused constrained FP intrinsic calls may have declared side effect, which
2158 // prevents it from being removed. In some cases however the side effect is
2159 // actually absent. To detect this case, call SimplifyConstrainedFPCall. If it
2160 // returns a replacement, the call may be removed.
2161 if (CI.use_empty() && isa<ConstrainedFPIntrinsic>(CI)) {
2162 if (simplifyConstrainedFPCall(&CI, SQ.getWithInstruction(&CI)))
2163 return eraseInstFromFunction(CI);
2164 }
2165
2166 Intrinsic::ID IID = II->getIntrinsicID();
2167 switch (IID) {
2168 case Intrinsic::objectsize: {
2169 SmallVector<Instruction *> InsertedInstructions;
2170 if (Value *V = lowerObjectSizeCall(II, DL, &TLI, AA, /*MustSucceed=*/false,
2171 &InsertedInstructions)) {
2172 for (Instruction *Inserted : InsertedInstructions)
2173 Worklist.add(Inserted);
2174 return replaceInstUsesWith(CI, V);
2175 }
2176 return nullptr;
2177 }
2178 case Intrinsic::abs: {
2179 Value *IIOperand = II->getArgOperand(0);
2180 bool IntMinIsPoison = cast<Constant>(II->getArgOperand(1))->isOneValue();
2181
2182 // abs(-x) -> abs(x)
2183 Value *X;
2184 if (match(IIOperand, m_Neg(m_Value(X))))
2185 return CallInst::Create(
2186 II->getCalledFunction(),
2187 {X,
2188 Builder.getInt1(IntMinIsPoison ||
2189 cast<Instruction>(IIOperand)->hasNoSignedWrap())});
2190
2191 if (match(IIOperand, m_c_Select(m_Neg(m_Value(X)), m_Deferred(X))))
2192 return CallInst::Create(II->getCalledFunction(),
2193 {X, II->getArgOperand(1)});
2194
2195 Value *Y;
2196 // abs(a * abs(b)) -> abs(a * b)
2197 if (match(IIOperand,
2200 bool NSW =
2201 cast<Instruction>(IIOperand)->hasNoSignedWrap() && IntMinIsPoison;
2202 auto *XY = NSW ? Builder.CreateNSWMul(X, Y) : Builder.CreateMul(X, Y);
2203 return CallInst::Create(II->getCalledFunction(),
2204 {XY, II->getArgOperand(1)});
2205 }
2206
2207 if (std::optional<bool> Known =
2208 getKnownSignOrZero(IIOperand, SQ.getWithInstruction(II))) {
2209 // abs(x) -> x if x >= 0 (include abs(x-y) --> x - y where x >= y)
2210 // abs(x) -> x if x > 0 (include abs(x-y) --> x - y where x > y)
2211 if (!*Known)
2212 return replaceInstUsesWith(*II, IIOperand);
2213
2214 // abs(x) -> -x if x < 0
2215 // abs(x) -> -x if x < = 0 (include abs(x-y) --> y - x where x <= y)
2216 if (IntMinIsPoison)
2217 return BinaryOperator::CreateNSWNeg(IIOperand);
2218 return BinaryOperator::CreateNeg(IIOperand);
2219 }
2220
2221 // abs (sext X) --> zext (abs X*)
2222 // Clear the IsIntMin (nsw) bit on the abs to allow narrowing.
2223 if (match(IIOperand, m_OneUse(m_SExt(m_Value(X))))) {
2224 Value *NarrowAbs =
2225 Builder.CreateBinaryIntrinsic(Intrinsic::abs, X, Builder.getFalse());
2226 return CastInst::Create(Instruction::ZExt, NarrowAbs, II->getType());
2227 }
2228
2229 // Match a complicated way to check if a number is odd/even:
2230 // abs (srem X, 2) --> and X, 1
2231 const APInt *C;
2232 if (match(IIOperand, m_SRem(m_Value(X), m_APInt(C))) && *C == 2)
2233 return BinaryOperator::CreateAnd(X, ConstantInt::get(II->getType(), 1));
2234
2235 break;
2236 }
2237 case Intrinsic::umin: {
2238 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2239 // umin(x, 1) == zext(x != 0)
2240 if (match(I1, m_One())) {
2241 assert(II->getType()->getScalarSizeInBits() != 1 &&
2242 "Expected simplify of umin with max constant");
2243 Value *Zero = Constant::getNullValue(I0->getType());
2244 Value *Cmp = Builder.CreateICmpNE(I0, Zero);
2245 return CastInst::Create(Instruction::ZExt, Cmp, II->getType());
2246 }
2247 // umin(cttz(x), const) --> cttz(x | (1 << const))
2248 if (Value *FoldedCttz =
2250 I0, I1, DL, Builder))
2251 return replaceInstUsesWith(*II, FoldedCttz);
2252 // umin(ctlz(x), const) --> ctlz(x | (SignedMin >> const))
2253 if (Value *FoldedCtlz =
2255 I0, I1, DL, Builder))
2256 return replaceInstUsesWith(*II, FoldedCtlz);
2257 [[fallthrough]];
2258 }
2259 case Intrinsic::umax: {
2260 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2261 Value *X, *Y;
2262 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_ZExt(m_Value(Y))) &&
2263 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2264 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2265 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2266 }
2267 Constant *C;
2268 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2269 I0->hasOneUse()) {
2270 if (Constant *NarrowC = getLosslessUnsignedTrunc(C, X->getType(), DL)) {
2271 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2272 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2273 }
2274 }
2275 // If C is not 0:
2276 // umax(nuw_shl(x, C), x + 1) -> x == 0 ? 1 : nuw_shl(x, C)
2277 // If C is not 0 or 1:
2278 // umax(nuw_mul(x, C), x + 1) -> x == 0 ? 1 : nuw_mul(x, C)
2279 auto foldMaxMulShift = [&](Value *A, Value *B) -> Instruction * {
2280 const APInt *C;
2281 Value *X;
2282 if (!match(A, m_NUWShl(m_Value(X), m_APInt(C))) &&
2283 !(match(A, m_NUWMul(m_Value(X), m_APInt(C))) && !C->isOne()))
2284 return nullptr;
2285 if (C->isZero())
2286 return nullptr;
2287 if (!match(B, m_OneUse(m_Add(m_Specific(X), m_One()))))
2288 return nullptr;
2289
2290 Value *Cmp = Builder.CreateICmpEQ(X, ConstantInt::get(X->getType(), 0));
2291 Value *NewSelect = nullptr;
2292 NewSelect = Builder.CreateSelectWithUnknownProfile(
2293 Cmp, ConstantInt::get(X->getType(), 1), A, DEBUG_TYPE);
2294 return replaceInstUsesWith(*II, NewSelect);
2295 };
2296
2297 if (IID == Intrinsic::umax) {
2298 if (Instruction *I = foldMaxMulShift(I0, I1))
2299 return I;
2300 if (Instruction *I = foldMaxMulShift(I1, I0))
2301 return I;
2302 }
2303
2304 // If both operands of unsigned min/max are sign-extended, it is still ok
2305 // to narrow the operation.
2306 [[fallthrough]];
2307 }
2308 case Intrinsic::smax:
2309 case Intrinsic::smin: {
2310 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2311 Value *X, *Y;
2312 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_SExt(m_Value(Y))) &&
2313 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2314 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2315 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2316 }
2317
2318 Constant *C;
2319 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2320 I0->hasOneUse()) {
2321 if (Constant *NarrowC = getLosslessSignedTrunc(C, X->getType(), DL)) {
2322 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2323 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2324 }
2325 }
2326
2327 // smax(smin(X, MinC), MaxC) -> smin(smax(X, MaxC), MinC) if MinC s>= MaxC
2328 // umax(umin(X, MinC), MaxC) -> umin(umax(X, MaxC), MinC) if MinC u>= MaxC
2329 const APInt *MinC, *MaxC;
2330 auto CreateCanonicalClampForm = [&](bool IsSigned) {
2331 auto MaxIID = IsSigned ? Intrinsic::smax : Intrinsic::umax;
2332 auto MinIID = IsSigned ? Intrinsic::smin : Intrinsic::umin;
2333 Value *NewMax = Builder.CreateBinaryIntrinsic(
2334 MaxIID, X, ConstantInt::get(X->getType(), *MaxC));
2335 return replaceInstUsesWith(
2336 *II, Builder.CreateBinaryIntrinsic(
2337 MinIID, NewMax, ConstantInt::get(X->getType(), *MinC)));
2338 };
2339 if (IID == Intrinsic::smax &&
2341 m_APInt(MinC)))) &&
2342 match(I1, m_APInt(MaxC)) && MinC->sgt(*MaxC))
2343 return CreateCanonicalClampForm(true);
2344 if (IID == Intrinsic::umax &&
2346 m_APInt(MinC)))) &&
2347 match(I1, m_APInt(MaxC)) && MinC->ugt(*MaxC))
2348 return CreateCanonicalClampForm(false);
2349
2350 // umin(i1 X, i1 Y) -> and i1 X, Y
2351 // smax(i1 X, i1 Y) -> and i1 X, Y
2352 if ((IID == Intrinsic::umin || IID == Intrinsic::smax) &&
2353 II->getType()->isIntOrIntVectorTy(1)) {
2354 return BinaryOperator::CreateAnd(I0, I1);
2355 }
2356
2357 // umax(i1 X, i1 Y) -> or i1 X, Y
2358 // smin(i1 X, i1 Y) -> or i1 X, Y
2359 if ((IID == Intrinsic::umax || IID == Intrinsic::smin) &&
2360 II->getType()->isIntOrIntVectorTy(1)) {
2361 return BinaryOperator::CreateOr(I0, I1);
2362 }
2363
2364 // smin(smax(X, -1), 1) -> scmp(X, 0)
2365 // smax(smin(X, 1), -1) -> scmp(X, 0)
2366 // At this point, smax(smin(X, 1), -1) is changed to smin(smax(X, -1)
2367 // And i1's have been changed to and/ors
2368 // So we only need to check for smin
2369 if (IID == Intrinsic::smin) {
2370 if (match(I0, m_OneUse(m_SMax(m_Value(X), m_AllOnes()))) &&
2371 match(I1, m_One())) {
2372 Value *Zero = ConstantInt::get(X->getType(), 0);
2373 return replaceInstUsesWith(
2374 CI,
2375 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {X, Zero}));
2376 }
2377 }
2378
2379 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2380 // smax (neg nsw X), (neg nsw Y) --> neg nsw (smin X, Y)
2381 // smin (neg nsw X), (neg nsw Y) --> neg nsw (smax X, Y)
2382 // TODO: Canonicalize neg after min/max if I1 is constant.
2383 if (match(I0, m_NSWNeg(m_Value(X))) && match(I1, m_NSWNeg(m_Value(Y))) &&
2384 (I0->hasOneUse() || I1->hasOneUse())) {
2386 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, X, Y);
2387 return BinaryOperator::CreateNSWNeg(InvMaxMin);
2388 }
2389 }
2390
2391 // (umax X, (xor X, Pow2))
2392 // -> (or X, Pow2)
2393 // (umin X, (xor X, Pow2))
2394 // -> (and X, ~Pow2)
2395 // (smax X, (xor X, Pos_Pow2))
2396 // -> (or X, Pos_Pow2)
2397 // (smin X, (xor X, Pos_Pow2))
2398 // -> (and X, ~Pos_Pow2)
2399 // (smax X, (xor X, Neg_Pow2))
2400 // -> (and X, ~Neg_Pow2)
2401 // (smin X, (xor X, Neg_Pow2))
2402 // -> (or X, Neg_Pow2)
2403 if ((match(I0, m_c_Xor(m_Specific(I1), m_Value(X))) ||
2404 match(I1, m_c_Xor(m_Specific(I0), m_Value(X)))) &&
2405 isKnownToBeAPowerOfTwo(X, /* OrZero */ true)) {
2406 bool UseOr = IID == Intrinsic::smax || IID == Intrinsic::umax;
2407 bool UseAndN = IID == Intrinsic::smin || IID == Intrinsic::umin;
2408
2409 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2410 auto KnownSign = getKnownSign(X, SQ.getWithInstruction(II));
2411 if (KnownSign == std::nullopt) {
2412 UseOr = false;
2413 UseAndN = false;
2414 } else if (*KnownSign /* true is Signed. */) {
2415 UseOr ^= true;
2416 UseAndN ^= true;
2417 Type *Ty = I0->getType();
2418 // Negative power of 2 must be IntMin. It's possible to be able to
2419 // prove negative / power of 2 without actually having known bits, so
2420 // just get the value by hand.
2422 Ty, APInt::getSignedMinValue(Ty->getScalarSizeInBits()));
2423 }
2424 }
2425 if (UseOr)
2426 return BinaryOperator::CreateOr(I0, X);
2427 else if (UseAndN)
2428 return BinaryOperator::CreateAnd(I0, Builder.CreateNot(X));
2429 }
2430
2431 // If we can eliminate ~A and Y is free to invert:
2432 // max ~A, Y --> ~(min A, ~Y)
2433 //
2434 // Examples:
2435 // max ~A, ~Y --> ~(min A, Y)
2436 // max ~A, C --> ~(min A, ~C)
2437 // max ~A, (max ~Y, ~Z) --> ~min( A, (min Y, Z))
2438 auto moveNotAfterMinMax = [&](Value *X, Value *Y) -> Instruction * {
2439 Value *A;
2440 if (match(X, m_OneUse(m_Not(m_Value(A)))) &&
2441 !isFreeToInvert(A, A->hasOneUse())) {
2442 if (Value *NotY = getFreelyInverted(Y, Y->hasOneUse(), &Builder)) {
2444 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, A, NotY);
2445 return BinaryOperator::CreateNot(InvMaxMin);
2446 }
2447 }
2448 return nullptr;
2449 };
2450
2451 if (Instruction *I = moveNotAfterMinMax(I0, I1))
2452 return I;
2453 if (Instruction *I = moveNotAfterMinMax(I1, I0))
2454 return I;
2455
2457 return I;
2458
2459 // minmax (X & NegPow2C, Y & NegPow2C) --> minmax(X, Y) & NegPow2C
2460 const APInt *RHSC;
2461 if (match(I0, m_OneUse(m_And(m_Value(X), m_NegatedPower2(RHSC)))) &&
2462 match(I1, m_OneUse(m_And(m_Value(Y), m_SpecificInt(*RHSC)))))
2463 return BinaryOperator::CreateAnd(Builder.CreateBinaryIntrinsic(IID, X, Y),
2464 ConstantInt::get(II->getType(), *RHSC));
2465
2466 // smax(X, -X) --> abs(X)
2467 // smin(X, -X) --> -abs(X)
2468 // umax(X, -X) --> -abs(X)
2469 // umin(X, -X) --> abs(X)
2470 if (isKnownNegation(I0, I1)) {
2471 // We can choose either operand as the input to abs(), but if we can
2472 // eliminate the only use of a value, that's better for subsequent
2473 // transforms/analysis.
2474 if (I0->hasOneUse() && !I1->hasOneUse())
2475 std::swap(I0, I1);
2476
2477 // This is some variant of abs(). See if we can propagate 'nsw' to the abs
2478 // operation and potentially its negation.
2479 bool IntMinIsPoison = isKnownNegation(I0, I1, /* NeedNSW */ true);
2480 Value *Abs = Builder.CreateBinaryIntrinsic(
2481 Intrinsic::abs, I0,
2482 ConstantInt::getBool(II->getContext(), IntMinIsPoison));
2483
2484 // We don't have a "nabs" intrinsic, so negate if needed based on the
2485 // max/min operation.
2486 if (IID == Intrinsic::smin || IID == Intrinsic::umax)
2487 Abs = Builder.CreateNeg(Abs, "nabs", IntMinIsPoison);
2488 return replaceInstUsesWith(CI, Abs);
2489 }
2490
2492 return Sel;
2493
2494 if (Instruction *SAdd = matchSAddSubSat(*II))
2495 return SAdd;
2496
2497 if (Value *NewMinMax = reassociateMinMaxWithConstants(II, Builder, SQ))
2498 return replaceInstUsesWith(*II, NewMinMax);
2499
2501 return R;
2502
2503 if (Instruction *NewMinMax = factorizeMinMaxTree(II))
2504 return NewMinMax;
2505
2506 // Try to fold minmax with constant RHS based on range information
2507 if (match(I1, m_APIntAllowPoison(RHSC))) {
2508 ICmpInst::Predicate Pred =
2510 bool IsSigned = MinMaxIntrinsic::isSigned(IID);
2512 I0, IsSigned, SQ.getWithInstruction(II));
2513 if (!LHS_CR.isFullSet()) {
2514 if (LHS_CR.icmp(Pred, *RHSC))
2515 return replaceInstUsesWith(*II, I0);
2516 if (LHS_CR.icmp(ICmpInst::getSwappedPredicate(Pred), *RHSC))
2517 return replaceInstUsesWith(*II,
2518 ConstantInt::get(II->getType(), *RHSC));
2519 }
2520 }
2521
2523 return replaceInstUsesWith(*II, V);
2524
2525 break;
2526 }
2527 case Intrinsic::scmp:
2528 case Intrinsic::ucmp: {
2530 return replaceInstUsesWith(CI, V);
2531
2532 if (IID == Intrinsic::ucmp)
2533 break;
2534
2535 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2536
2537 // scmp(X, 0) -> sext_or_trunc(X) if X is known to be one of -1, 0, 1.
2538 if (match(I1, m_Zero())) {
2539 ConstantRange Range = computeConstantRange(I0, /*ForSigned=*/true,
2540 SQ.getWithInstruction(II));
2541 if (Range.getSignedMin().sge(-1) && Range.getSignedMax().sle(1))
2542 return replaceInstUsesWith(
2543 CI, Builder.CreateSExtOrTrunc(I0, II->getType()));
2544 }
2545 Value *LHS, *RHS;
2546 if (match(I0, m_NSWSub(m_Value(LHS), m_Value(RHS))) && match(I1, m_Zero()))
2547 return replaceInstUsesWith(
2548 CI,
2549 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {LHS, RHS}));
2550 break;
2551 }
2552 case Intrinsic::bitreverse: {
2553 Value *IIOperand = II->getArgOperand(0);
2554 // bitrev (zext i1 X to ?) --> X ? SignBitC : 0
2555 Value *X;
2556 if (match(IIOperand, m_ZExt(m_Value(X))) &&
2557 X->getType()->isIntOrIntVectorTy(1)) {
2558 Type *Ty = II->getType();
2559 APInt SignBit = APInt::getSignMask(Ty->getScalarSizeInBits());
2560 SelectInst *SI = SelectInst::Create(X, ConstantInt::get(Ty, SignBit),
2562 // Mark the branch weights explicitly unknown as in the general case we
2563 // cannot infer the probability of the condition without additional value
2564 // profiling.
2566 return SI;
2567 }
2568
2569 if (Instruction *crossLogicOpFold =
2571 return crossLogicOpFold;
2572
2573 break;
2574 }
2575 case Intrinsic::bswap: {
2576 Value *IIOperand = II->getArgOperand(0);
2577
2578 // Try to canonicalize bswap-of-logical-shift-by-8-bit-multiple as
2579 // inverse-shift-of-bswap:
2580 // bswap (shl X, Y) --> lshr (bswap X), Y
2581 // bswap (lshr X, Y) --> shl (bswap X), Y
2582 Value *X, *Y;
2583 if (match(IIOperand, m_OneUse(m_LogicalShift(m_Value(X), m_Value(Y))))) {
2584 unsigned BitWidth = IIOperand->getType()->getScalarSizeInBits();
2586 Value *NewSwap = Builder.CreateUnaryIntrinsic(Intrinsic::bswap, X);
2587 BinaryOperator::BinaryOps InverseShift =
2588 cast<BinaryOperator>(IIOperand)->getOpcode() == Instruction::Shl
2589 ? Instruction::LShr
2590 : Instruction::Shl;
2591 return BinaryOperator::Create(InverseShift, NewSwap, Y);
2592 }
2593 }
2594
2595 KnownBits Known = computeKnownBits(IIOperand, II);
2596 uint64_t LZ = alignDown(Known.countMinLeadingZeros(), 8);
2597 uint64_t TZ = alignDown(Known.countMinTrailingZeros(), 8);
2598 unsigned BW = Known.getBitWidth();
2599
2600 // bswap(x) -> shift(x) if x has exactly one "active byte"
2601 if (BW - LZ - TZ == 8) {
2602 assert(LZ != TZ && "active byte cannot be in the middle");
2603 if (LZ > TZ) // -> shl(x) if the "active byte" is in the low part of x
2604 return BinaryOperator::CreateNUWShl(
2605 IIOperand, ConstantInt::get(IIOperand->getType(), LZ - TZ));
2606 // -> lshr(x) if the "active byte" is in the high part of x
2607 return BinaryOperator::CreateExactLShr(
2608 IIOperand, ConstantInt::get(IIOperand->getType(), TZ - LZ));
2609 }
2610
2611 // bswap(trunc(bswap(x))) -> trunc(lshr(x, c))
2612 if (match(IIOperand, m_Trunc(m_BSwap(m_Value(X))))) {
2613 unsigned C = X->getType()->getScalarSizeInBits() - BW;
2614 Value *CV = ConstantInt::get(X->getType(), C);
2615 Value *V = Builder.CreateLShr(X, CV);
2616 return new TruncInst(V, IIOperand->getType());
2617 }
2618
2619 if (Instruction *crossLogicOpFold =
2621 return crossLogicOpFold;
2622 }
2623
2624 // Try to fold into bitreverse if bswap is the root of the expression tree.
2625 if (Instruction *BitOp = matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ false,
2626 /*MatchBitReversals*/ true))
2627 return BitOp;
2628 break;
2629 }
2630 case Intrinsic::masked_load:
2631 if (Value *SimplifiedMaskedOp = simplifyMaskedLoad(*II))
2632 return replaceInstUsesWith(CI, SimplifiedMaskedOp);
2633 break;
2634 case Intrinsic::masked_store:
2635 return simplifyMaskedStore(*II);
2636 case Intrinsic::masked_gather:
2637 return simplifyMaskedGather(*II);
2638 case Intrinsic::masked_scatter:
2639 return simplifyMaskedScatter(*II);
2640 case Intrinsic::launder_invariant_group:
2641 if (auto *SkippedBarrier = simplifyInvariantGroupIntrinsic(*II, *this))
2642 return replaceInstUsesWith(*II, SkippedBarrier);
2643 break;
2644 case Intrinsic::powi: {
2645 if (ConstantInt *Power = dyn_cast<ConstantInt>(II->getArgOperand(1))) {
2646 // 0 and 1 are handled in instsimplify
2647 // powi(x, -1) -> 1/x
2648 if (Power->isMinusOne())
2649 return BinaryOperator::CreateFDivFMF(ConstantFP::get(CI.getType(), 1.0),
2650 II->getArgOperand(0), II);
2651 // powi(x, 2) -> x*x
2652 if (Power->equalsInt(2))
2653 return BinaryOperator::CreateFMulFMF(II->getArgOperand(0),
2654 II->getArgOperand(0), II);
2655
2656 if (!Power->getValue()[0]) {
2657 Value *X;
2658 // If power is even:
2659 // powi(-x, p) -> powi(x, p)
2660 // powi(fabs(x), p) -> powi(x, p)
2661 // powi(copysign(x, y), p) -> powi(x, p)
2662 if (match(II->getArgOperand(0), m_FNeg(m_Value(X))) ||
2663 match(II->getArgOperand(0), m_FAbs(m_Value(X))) ||
2664 match(II->getArgOperand(0),
2666 return CallInst::Create(II->getCalledFunction(), {X, Power});
2667 }
2668 }
2669 if (ConstantFP *Base = dyn_cast<ConstantFP>(II->getArgOperand(0))) {
2670 Value *Exp = II->getArgOperand(1);
2671 Type *Ty = Base->getType();
2672 // powi(2.0, p) -> ldexp(1.0, p)
2673 if (II->hasApproxFunc() && Base->isExactlyValue(2.0)) {
2674 ConstantFP *One = ConstantFP::get(Ty, 1.0);
2675 if (auto *VTy = dyn_cast<VectorType>(Ty))
2676 Exp = Builder.CreateVectorSplat(VTy->getElementCount(), Exp);
2677 Value *Ldexp = Builder.CreateLdexp(One, Exp, II);
2678 return replaceInstUsesWith(*II, Ldexp);
2679 }
2680 }
2681 break;
2682 }
2683
2684 case Intrinsic::cttz:
2685 case Intrinsic::ctlz:
2686 if (auto *I = foldCttzCtlz(*II, *this))
2687 return I;
2688 break;
2689
2690 case Intrinsic::ctpop:
2691 if (auto *I = foldCtpop(*II, *this))
2692 return I;
2693 break;
2694
2695 case Intrinsic::fshl:
2696 case Intrinsic::fshr: {
2697 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
2698 Type *Ty = II->getType();
2699 unsigned BitWidth = Ty->getScalarSizeInBits();
2700 Constant *ShAmtC;
2701 if (match(II->getArgOperand(2), m_ImmConstant(ShAmtC))) {
2702 // Canonicalize a shift amount constant operand to modulo the bit-width.
2703 Constant *WidthC = ConstantInt::get(Ty, BitWidth);
2704 Constant *ModuloC =
2705 ConstantFoldBinaryOpOperands(Instruction::URem, ShAmtC, WidthC, DL);
2706 if (!ModuloC)
2707 return nullptr;
2708 if (ModuloC != ShAmtC)
2709 return CallInst::Create(II->getCalledFunction(), {Op0, Op1, ModuloC});
2710
2712 ShAmtC, DL),
2713 m_One()) &&
2714 "Shift amount expected to be modulo bitwidth");
2715
2716 // Canonicalize funnel shift right by constant to funnel shift left. This
2717 // is not entirely arbitrary. For historical reasons, the backend may
2718 // recognize rotate left patterns but miss rotate right patterns.
2719 if (IID == Intrinsic::fshr) {
2720 // fshr X, Y, C --> fshl X, Y, (BitWidth - C) if C is not zero.
2721 if (!isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2722 return nullptr;
2723
2724 Constant *LeftShiftC = ConstantExpr::getSub(WidthC, ShAmtC);
2725 Module *Mod = II->getModule();
2726 Function *Fshl =
2727 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::fshl, Ty);
2728 return CallInst::Create(Fshl, { Op0, Op1, LeftShiftC });
2729 }
2730 assert(IID == Intrinsic::fshl &&
2731 "All funnel shifts by simple constants should go left");
2732
2733 // fshl(X, 0, C) --> shl X, C
2734 // fshl(X, undef, C) --> shl X, C
2735 if (match(Op1, m_ZeroInt()) || match(Op1, m_Undef()))
2736 return BinaryOperator::CreateShl(Op0, ShAmtC);
2737
2738 // fshl(0, X, C) --> lshr X, (BW-C)
2739 // fshl(undef, X, C) --> lshr X, (BW-C)
2740 // Similar to fshr -> fshl fold above, this is only valid if C is not zero
2741 if ((match(Op0, m_ZeroInt()) || match(Op0, m_Undef())) &&
2742 isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2743 return BinaryOperator::CreateLShr(Op1,
2744 ConstantExpr::getSub(WidthC, ShAmtC));
2745
2746 // fshl i16 X, X, 8 --> bswap i16 X (reduce to more-specific form)
2747 if (Op0 == Op1 && BitWidth == 16 && match(ShAmtC, m_SpecificInt(8))) {
2748 Module *Mod = II->getModule();
2749 Function *Bswap =
2750 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::bswap, Ty);
2751 return CallInst::Create(Bswap, { Op0 });
2752 }
2753 if (Instruction *BitOp =
2754 matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ true,
2755 /*MatchBitReversals*/ true))
2756 return BitOp;
2757
2758 // R = fshl(X, X, C2)
2759 // fshl(R, R, C1) --> fshl(X, X, (C1 + C2) % bitsize)
2760 Value *InnerOp;
2761 const APInt *ShAmtInnerC, *ShAmtOuterC;
2762 if (match(Op0, m_FShl(m_Value(InnerOp), m_Deferred(InnerOp),
2763 m_APInt(ShAmtInnerC))) &&
2764 match(ShAmtC, m_APInt(ShAmtOuterC)) && Op0 == Op1) {
2765 APInt Sum = *ShAmtOuterC + *ShAmtInnerC;
2766 APInt Modulo = Sum.urem(APInt(Sum.getBitWidth(), BitWidth));
2767 if (Modulo.isZero())
2768 return replaceInstUsesWith(*II, InnerOp);
2769 Constant *ModuloC = ConstantInt::get(Ty, Modulo);
2771 {InnerOp, InnerOp, ModuloC});
2772 }
2773 }
2774
2775 // fshl(X, X, Neg(Y)) --> fshr(X, X, Y)
2776 // fshr(X, X, Neg(Y)) --> fshl(X, X, Y)
2777 // if BitWidth is a power-of-2
2778 Value *Y;
2779 if (Op0 == Op1 && isPowerOf2_32(BitWidth) &&
2780 match(II->getArgOperand(2), m_Neg(m_Value(Y)))) {
2781 Module *Mod = II->getModule();
2783 Mod, IID == Intrinsic::fshl ? Intrinsic::fshr : Intrinsic::fshl, Ty);
2784 return CallInst::Create(OppositeShift, {Op0, Op1, Y});
2785 }
2786
2787 // fshl(X, 0, Y) --> shl(X, and(Y, BitWidth - 1)) if bitwidth is a
2788 // power-of-2
2789 if (IID == Intrinsic::fshl && isPowerOf2_32(BitWidth) &&
2790 match(Op1, m_ZeroInt())) {
2791 Value *Op2 = II->getArgOperand(2);
2792 Value *And = Builder.CreateAnd(Op2, ConstantInt::get(Ty, BitWidth - 1));
2793 return BinaryOperator::CreateShl(Op0, And);
2794 }
2795
2796 // Left or right might be masked.
2798 return &CI;
2799
2800 // The shift amount (operand 2) of a funnel shift is modulo the bitwidth,
2801 // so only the low bits of the shift amount are demanded if the bitwidth is
2802 // a power-of-2.
2803 if (!isPowerOf2_32(BitWidth))
2804 break;
2806 KnownBits Op2Known(BitWidth);
2807 if (SimplifyDemandedBits(II, 2, Op2Demanded, Op2Known))
2808 return &CI;
2809 break;
2810 }
2811 case Intrinsic::pdep: {
2812 const APInt *MaskC;
2813 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2814 unsigned MaskIdx, MaskLen;
2815 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2816 // any single contiguous sequence of 1s anywhere in the mask simply
2817 // describes a subset of the input bits shifted to the appropriate
2818 // position. Replace with the straight forward IR.
2819 Value *Input = II->getArgOperand(0);
2820 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2821 Value *Shifted = Builder.CreateShl(Input, ShiftAmt);
2822 Value *Masked = Builder.CreateAnd(Shifted, II->getArgOperand(1));
2823 return replaceInstUsesWith(*II, Masked);
2824 }
2825 }
2826 break;
2827 }
2828 case Intrinsic::pext: {
2829 const APInt *MaskC;
2830 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2831 unsigned MaskIdx, MaskLen;
2832 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2833 // any single contiguous sequence of 1s anywhere in the mask simply
2834 // describes a subset of the input bits shifted to the appropriate
2835 // position. Replace with the straight forward IR.
2836 Value *Input = II->getArgOperand(0);
2837 Value *Masked = Builder.CreateAnd(Input, II->getArgOperand(1));
2838 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2839 Value *Shifted = Builder.CreateLShr(Masked, ShiftAmt);
2840 return replaceInstUsesWith(*II, Shifted);
2841 }
2842 }
2843 break;
2844 }
2845 case Intrinsic::ptrmask: {
2846 unsigned BitWidth = DL.getPointerTypeSizeInBits(II->getType());
2849 return II;
2850
2851 Value *InnerPtr, *InnerMask;
2852 bool Changed = false;
2853 // Combine:
2854 // (ptrmask (ptrmask p, A), B)
2855 // -> (ptrmask p, (and A, B))
2856 if (match(II->getArgOperand(0),
2858 m_Value(InnerMask))))) {
2859 assert(II->getArgOperand(1)->getType() == InnerMask->getType() &&
2860 "Mask types must match");
2861 // TODO: If InnerMask == Op1, we could copy attributes from inner
2862 // callsite -> outer callsite.
2863 Value *NewMask = Builder.CreateAnd(II->getArgOperand(1), InnerMask);
2864 replaceOperand(CI, 0, InnerPtr);
2865 replaceOperand(CI, 1, NewMask);
2866 Changed = true;
2867 }
2868
2869 // See if we can deduce non-null.
2870 if (!CI.hasRetAttr(Attribute::NonNull) &&
2871 (Known.isNonZero() ||
2872 isKnownNonZero(II, getSimplifyQuery().getWithInstruction(II)))) {
2873 CI.addRetAttr(Attribute::NonNull);
2874 Changed = true;
2875 }
2876
2877 unsigned NewAlignmentLog =
2879 std::min(BitWidth - 1, Known.countMinTrailingZeros()));
2880 // Known bits will capture if we had alignment information associated with
2881 // the pointer argument.
2882 if (NewAlignmentLog > Log2(CI.getRetAlign().valueOrOne())) {
2884 CI.getContext(), Align(uint64_t(1) << NewAlignmentLog)));
2885 Changed = true;
2886 }
2887 if (Changed)
2888 return &CI;
2889 break;
2890 }
2891
2892 case Intrinsic::smulh: {
2893 Value *Arg0 = II->getArgOperand(0);
2894 Value *Arg1 = II->getArgOperand(1);
2895 unsigned BitWidth = II->getType()->getScalarSizeInBits();
2896
2897 // Multiply by one.
2898 if (match(Arg1, m_One()))
2899 return replaceInstUsesWith(CI, Builder.CreateAShr(Arg0, BitWidth - 1));
2900 break;
2901 }
2902
2903 case Intrinsic::uadd_with_overflow:
2904 case Intrinsic::sadd_with_overflow: {
2905 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2906 return I;
2907
2908 // Given 2 constant operands whose sum does not overflow:
2909 // uaddo (X +nuw C0), C1 -> uaddo X, C0 + C1
2910 // saddo (X +nsw C0), C1 -> saddo X, C0 + C1
2911 Value *X;
2912 const APInt *C0, *C1;
2913 Value *Arg0 = II->getArgOperand(0);
2914 Value *Arg1 = II->getArgOperand(1);
2915 bool IsSigned = IID == Intrinsic::sadd_with_overflow;
2916 bool HasNWAdd = IsSigned
2917 ? match(Arg0, m_NSWAddLike(m_Value(X), m_APInt(C0)))
2918 : match(Arg0, m_NUWAddLike(m_Value(X), m_APInt(C0)));
2919 if (HasNWAdd && match(Arg1, m_APInt(C1))) {
2920 bool Overflow;
2921 APInt NewC =
2922 IsSigned ? C1->sadd_ov(*C0, Overflow) : C1->uadd_ov(*C0, Overflow);
2923 if (!Overflow)
2924 return replaceInstUsesWith(
2925 *II, Builder.CreateBinaryIntrinsic(
2926 IID, X, ConstantInt::get(Arg1->getType(), NewC)));
2927 }
2928 break;
2929 }
2930
2931 case Intrinsic::umul_with_overflow:
2932 case Intrinsic::smul_with_overflow:
2933 case Intrinsic::usub_with_overflow:
2934 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2935 return I;
2936 break;
2937
2938 case Intrinsic::ssub_with_overflow: {
2939 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2940 return I;
2941
2942 Constant *C;
2943 Value *Arg0 = II->getArgOperand(0);
2944 Value *Arg1 = II->getArgOperand(1);
2945 // Given a constant C that is not the minimum signed value
2946 // for an integer of a given bit width:
2947 //
2948 // ssubo X, C -> saddo X, -C
2949 if (match(Arg1, m_Constant(C)) && C->isNotMinSignedValue()) {
2950 Value *NegVal = ConstantExpr::getNeg(C);
2951 // Build a saddo call that is equivalent to the discovered
2952 // ssubo call.
2953 return replaceInstUsesWith(
2954 *II, Builder.CreateBinaryIntrinsic(Intrinsic::sadd_with_overflow,
2955 Arg0, NegVal));
2956 }
2957
2958 break;
2959 }
2960
2961 case Intrinsic::uadd_sat:
2962 case Intrinsic::sadd_sat:
2963 case Intrinsic::usub_sat:
2964 case Intrinsic::ssub_sat: {
2966 Type *Ty = SI->getType();
2967 Value *Arg0 = SI->getLHS();
2968 Value *Arg1 = SI->getRHS();
2969
2970 // Make use of known overflow information.
2971 OverflowResult OR = computeOverflow(SI->getBinaryOp(), SI->isSigned(),
2972 Arg0, Arg1, SI);
2973 switch (OR) {
2975 break;
2977 if (SI->isSigned())
2978 return BinaryOperator::CreateNSW(SI->getBinaryOp(), Arg0, Arg1);
2979 else
2980 return BinaryOperator::CreateNUW(SI->getBinaryOp(), Arg0, Arg1);
2982 unsigned BitWidth = Ty->getScalarSizeInBits();
2983 APInt Min = APSInt::getMinValue(BitWidth, !SI->isSigned());
2984 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Min));
2985 }
2987 unsigned BitWidth = Ty->getScalarSizeInBits();
2988 APInt Max = APSInt::getMaxValue(BitWidth, !SI->isSigned());
2989 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Max));
2990 }
2991 }
2992
2993 // usub_sat((sub nuw C, A), C1) -> usub_sat(usub_sat(C, C1), A)
2994 // which after that:
2995 // usub_sat((sub nuw C, A), C1) -> usub_sat(C - C1, A) if C1 u< C
2996 // usub_sat((sub nuw C, A), C1) -> 0 otherwise
2997 Constant *C, *C1;
2998 Value *A;
2999 if (IID == Intrinsic::usub_sat &&
3000 match(Arg0, m_NUWSub(m_ImmConstant(C), m_Value(A))) &&
3001 match(Arg1, m_ImmConstant(C1))) {
3002 auto *NewC = Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, C, C1);
3003 auto *NewSub =
3004 Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, NewC, A);
3005 return replaceInstUsesWith(*SI, NewSub);
3006 }
3007
3008 // ssub.sat(X, C) -> sadd.sat(X, -C) if C != MIN
3009 if (IID == Intrinsic::ssub_sat && match(Arg1, m_Constant(C)) &&
3010 C->isNotMinSignedValue()) {
3011 Value *NegVal = ConstantExpr::getNeg(C);
3012 return replaceInstUsesWith(
3013 *II, Builder.CreateBinaryIntrinsic(
3014 Intrinsic::sadd_sat, Arg0, NegVal));
3015 }
3016
3017 // sat(sat(X + Val2) + Val) -> sat(X + (Val+Val2))
3018 // sat(sat(X - Val2) - Val) -> sat(X - (Val+Val2))
3019 // if Val and Val2 have the same sign
3020 if (auto *Other = dyn_cast<IntrinsicInst>(Arg0)) {
3021 Value *X;
3022 const APInt *Val, *Val2;
3023 APInt NewVal;
3024 bool IsUnsigned =
3025 IID == Intrinsic::uadd_sat || IID == Intrinsic::usub_sat;
3026 if (Other->getIntrinsicID() == IID &&
3027 match(Arg1, m_APInt(Val)) &&
3028 match(Other->getArgOperand(0), m_Value(X)) &&
3029 match(Other->getArgOperand(1), m_APInt(Val2))) {
3030 if (IsUnsigned)
3031 NewVal = Val->uadd_sat(*Val2);
3032 else if (Val->isNonNegative() == Val2->isNonNegative()) {
3033 bool Overflow;
3034 NewVal = Val->sadd_ov(*Val2, Overflow);
3035 if (Overflow) {
3036 // Both adds together may add more than SignedMaxValue
3037 // without saturating the final result.
3038 break;
3039 }
3040 } else {
3041 // Cannot fold saturated addition with different signs.
3042 break;
3043 }
3044
3045 return replaceInstUsesWith(
3046 *II, Builder.CreateBinaryIntrinsic(
3047 IID, X, ConstantInt::get(II->getType(), NewVal)));
3048 }
3049 }
3050 break;
3051 }
3052
3053 case Intrinsic::minnum:
3054 case Intrinsic::maxnum:
3055 case Intrinsic::minimumnum:
3056 case Intrinsic::maximumnum:
3057 case Intrinsic::minimum:
3058 case Intrinsic::maximum: {
3059 Value *Arg0 = II->getArgOperand(0);
3060 Value *Arg1 = II->getArgOperand(1);
3061 Value *X, *Y;
3062 if (match(Arg0, m_FNeg(m_Value(X))) && match(Arg1, m_FNeg(m_Value(Y))) &&
3063 (Arg0->hasOneUse() || Arg1->hasOneUse())) {
3064 // If both operands are negated, invert the call and negate the result:
3065 // min(-X, -Y) --> -(max(X, Y))
3066 // max(-X, -Y) --> -(min(X, Y))
3067 Intrinsic::ID NewIID;
3068 switch (IID) {
3069 case Intrinsic::maxnum:
3070 NewIID = Intrinsic::minnum;
3071 break;
3072 case Intrinsic::minnum:
3073 NewIID = Intrinsic::maxnum;
3074 break;
3075 case Intrinsic::maximumnum:
3076 NewIID = Intrinsic::minimumnum;
3077 break;
3078 case Intrinsic::minimumnum:
3079 NewIID = Intrinsic::maximumnum;
3080 break;
3081 case Intrinsic::maximum:
3082 NewIID = Intrinsic::minimum;
3083 break;
3084 case Intrinsic::minimum:
3085 NewIID = Intrinsic::maximum;
3086 break;
3087 default:
3088 llvm_unreachable("unexpected intrinsic ID");
3089 }
3090 Value *NewCall = Builder.CreateBinaryIntrinsic(NewIID, X, Y, II);
3091 Instruction *FNeg = UnaryOperator::CreateFNeg(NewCall);
3092 FNeg->copyIRFlags(II);
3093 return FNeg;
3094 }
3095
3096 // m(m(X, C2), C1) -> m(X, C)
3097 const APFloat *C1, *C2;
3098 if (auto *M = dyn_cast<IntrinsicInst>(Arg0)) {
3099 if (M->getIntrinsicID() == IID && match(Arg1, m_APFloat(C1)) &&
3100 ((match(M->getArgOperand(0), m_Value(X)) &&
3101 match(M->getArgOperand(1), m_APFloat(C2))) ||
3102 (match(M->getArgOperand(1), m_Value(X)) &&
3103 match(M->getArgOperand(0), m_APFloat(C2))))) {
3104 APFloat Res(0.0);
3105 switch (IID) {
3106 case Intrinsic::maxnum:
3107 Res = maxnum(*C1, *C2);
3108 break;
3109 case Intrinsic::minnum:
3110 Res = minnum(*C1, *C2);
3111 break;
3112 case Intrinsic::maximumnum:
3113 Res = maximumnum(*C1, *C2);
3114 break;
3115 case Intrinsic::minimumnum:
3116 Res = minimumnum(*C1, *C2);
3117 break;
3118 case Intrinsic::maximum:
3119 Res = maximum(*C1, *C2);
3120 break;
3121 case Intrinsic::minimum:
3122 Res = minimum(*C1, *C2);
3123 break;
3124 default:
3125 llvm_unreachable("unexpected intrinsic ID");
3126 }
3127 // TODO: Conservatively intersecting FMF. If Res == C2, the transform
3128 // was a simplification (so Arg0 and its original flags could
3129 // propagate?)
3130 Value *V = Builder.CreateBinaryIntrinsic(
3131 IID, X, ConstantFP::get(Arg0->getType(), Res),
3133 return replaceInstUsesWith(*II, V);
3134 }
3135 }
3136
3137 // m((fpext X), (fpext Y)) -> fpext (m(X, Y))
3138 if (match(Arg0, m_FPExt(m_Value(X))) && match(Arg1, m_FPExt(m_Value(Y))) &&
3139 (Arg0->hasOneUse() || Arg1->hasOneUse()) &&
3140 X->getType() == Y->getType()) {
3141 Value *NewCall =
3142 Builder.CreateBinaryIntrinsic(IID, X, Y, II, II->getName());
3143 return new FPExtInst(NewCall, II->getType());
3144 }
3145
3146 // m(fpext X, C) -> fpext m(X, TruncC) if C can be losslessly truncated.
3147 Constant *C;
3148 if (match(Arg0, m_OneUse(m_FPExt(m_Value(X)))) &&
3149 match(Arg1, m_ImmConstant(C))) {
3150 if (Constant *TruncC =
3151 getLosslessInvCast(C, X->getType(), Instruction::FPExt, DL)) {
3152 Value *NewCall =
3153 Builder.CreateBinaryIntrinsic(IID, X, TruncC, II, II->getName());
3154 return new FPExtInst(NewCall, II->getType());
3155 }
3156 }
3157
3158 // max X, -X --> fabs X
3159 // min X, -X --> -(fabs X)
3160 // TODO: Remove one-use limitation? That is obviously better for max,
3161 // hence why we don't check for one-use for that. However,
3162 // it would be an extra instruction for min (fnabs), but
3163 // that is still likely better for analysis and codegen.
3164 auto IsMinMaxOrXNegX = [IID, &X](Value *Op0, Value *Op1) {
3165 if (match(Op0, m_FNeg(m_Value(X))) && match(Op1, m_Specific(X)))
3166 return Op0->hasOneUse() ||
3167 (IID != Intrinsic::minimum && IID != Intrinsic::minnum &&
3168 IID != Intrinsic::minimumnum);
3169 return false;
3170 };
3171
3172 if (IsMinMaxOrXNegX(Arg0, Arg1) || IsMinMaxOrXNegX(Arg1, Arg0)) {
3173 Value *R = Builder.CreateFAbs(X, II);
3174 if (IID == Intrinsic::minimum || IID == Intrinsic::minnum ||
3175 IID == Intrinsic::minimumnum)
3176 R = Builder.CreateFNegFMF(R, II);
3177 return replaceInstUsesWith(*II, R);
3178 }
3179
3180 break;
3181 }
3182 case Intrinsic::matrix_multiply: {
3183 // Optimize negation in matrix multiplication.
3184
3185 // -A * -B -> A * B
3186 Value *A, *B;
3187 if (match(II->getArgOperand(0), m_FNeg(m_Value(A))) &&
3188 match(II->getArgOperand(1), m_FNeg(m_Value(B)))) {
3189 replaceOperand(*II, 0, A);
3190 replaceOperand(*II, 1, B);
3191 return II;
3192 }
3193
3194 Value *Op0 = II->getOperand(0);
3195 Value *Op1 = II->getOperand(1);
3196 Value *OpNotNeg, *NegatedOp;
3197 unsigned NegatedOpArg, OtherOpArg;
3198 if (match(Op0, m_FNeg(m_Value(OpNotNeg)))) {
3199 NegatedOp = Op0;
3200 NegatedOpArg = 0;
3201 OtherOpArg = 1;
3202 } else if (match(Op1, m_FNeg(m_Value(OpNotNeg)))) {
3203 NegatedOp = Op1;
3204 NegatedOpArg = 1;
3205 OtherOpArg = 0;
3206 } else
3207 // Multiplication doesn't have a negated operand.
3208 break;
3209
3210 // Only optimize if the negated operand has only one use.
3211 if (!NegatedOp->hasOneUse())
3212 break;
3213
3214 Value *OtherOp = II->getOperand(OtherOpArg);
3215 VectorType *RetTy = cast<VectorType>(II->getType());
3216 VectorType *NegatedOpTy = cast<VectorType>(NegatedOp->getType());
3217 VectorType *OtherOpTy = cast<VectorType>(OtherOp->getType());
3218 ElementCount NegatedCount = NegatedOpTy->getElementCount();
3219 ElementCount OtherCount = OtherOpTy->getElementCount();
3220 ElementCount RetCount = RetTy->getElementCount();
3221 // (-A) * B -> A * (-B), if it is cheaper to negate B and vice versa.
3222 if (ElementCount::isKnownGT(NegatedCount, OtherCount) &&
3223 ElementCount::isKnownLT(OtherCount, RetCount)) {
3224 Value *InverseOtherOp = Builder.CreateFNeg(OtherOp);
3225 replaceOperand(*II, NegatedOpArg, OpNotNeg);
3226 replaceOperand(*II, OtherOpArg, InverseOtherOp);
3227 return II;
3228 }
3229 // (-A) * B -> -(A * B), if it is cheaper to negate the result
3230 if (ElementCount::isKnownGT(NegatedCount, RetCount)) {
3231 SmallVector<Value *, 5> NewArgs(II->args());
3232 NewArgs[NegatedOpArg] = OpNotNeg;
3233 Value *NewMul = Builder.CreateIntrinsic(II->getType(), IID, NewArgs, II);
3234 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(NewMul, II));
3235 }
3236 break;
3237 }
3238 case Intrinsic::fmuladd: {
3239 // Try to simplify the underlying FMul.
3240 if (Value *V =
3241 simplifyFMulInst(II->getArgOperand(0), II->getArgOperand(1),
3242 II->getFastMathFlags(), SQ.getWithInstruction(II)))
3243 return BinaryOperator::CreateFAddFMF(V, II->getArgOperand(2),
3244 II->getFastMathFlags());
3245
3246 [[fallthrough]];
3247 }
3248 case Intrinsic::fma: {
3249 // fma fneg(x), fneg(y), z -> fma x, y, z
3250 Value *Src0 = II->getArgOperand(0);
3251 Value *Src1 = II->getArgOperand(1);
3252 Value *Src2 = II->getArgOperand(2);
3253 Value *X, *Y;
3254 if (match(Src0, m_FNeg(m_Value(X))) && match(Src1, m_FNeg(m_Value(Y))))
3255 return replaceInstUsesWith(
3256 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, Y, Src2}, II));
3257
3258 // fma fabs(x), fabs(x), z -> fma x, x, z
3259 if (match(Src0, m_FAbs(m_Value(X))) && match(Src1, m_FAbs(m_Specific(X))))
3260 return replaceInstUsesWith(
3261 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, X, Src2}, II));
3262
3263 // Try to simplify the underlying FMul. We can only apply simplifications
3264 // that do not require rounding.
3265 if (Value *V = simplifyFMAFMul(Src0, Src1, II->getFastMathFlags(),
3266 SQ.getWithInstruction(II)))
3267 return BinaryOperator::CreateFAddFMF(V, Src2, II->getFastMathFlags());
3268
3269 // fma x, y, 0 -> fmul x, y
3270 // This is always valid for -0.0, but requires nsz for +0.0 as
3271 // -0.0 + 0.0 = 0.0, which would not be the same as the fmul on its own.
3272 if (match(Src2, m_NegZeroFP()) ||
3273 (match(Src2, m_PosZeroFP()) && II->getFastMathFlags().noSignedZeros()))
3274 return BinaryOperator::CreateFMulFMF(Src0, Src1, II);
3275
3276 // fma x, -1.0, y -> fsub y, x
3277 if (match(Src1, m_SpecificFP(-1.0)))
3278 return BinaryOperator::CreateFSubFMF(Src2, Src0, II);
3279
3280 break;
3281 }
3282 case Intrinsic::copysign: {
3283 Value *Mag = II->getArgOperand(0), *Sign = II->getArgOperand(1);
3284 if (std::optional<bool> KnownSignBit = computeKnownFPSignBit(
3285 Sign, getSimplifyQuery().getWithInstruction(II))) {
3286 if (*KnownSignBit) {
3287 // If we know that the sign argument is negative, reduce to FNABS:
3288 // copysign Mag, -Sign --> fneg (fabs Mag)
3289 Value *Fabs = Builder.CreateFAbs(Mag, II);
3290 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(Fabs, II));
3291 }
3292
3293 // If we know that the sign argument is positive, reduce to FABS:
3294 // copysign Mag, +Sign --> fabs Mag
3295 Value *Fabs = Builder.CreateFAbs(Mag, II);
3296 return replaceInstUsesWith(*II, Fabs);
3297 }
3298
3299 // Propagate sign argument through nested calls:
3300 // copysign Mag, (copysign ?, X) --> copysign Mag, X
3301 Value *X;
3303 Value *CopySign =
3304 Builder.CreateCopySign(Mag, X, FMFSource::intersect(II, Sign));
3305 return replaceInstUsesWith(*II, CopySign);
3306 }
3307
3308 // Clear sign-bit of constant magnitude:
3309 // copysign -MagC, X --> copysign MagC, X
3310 // TODO: Support constant folding for fabs
3311 const APFloat *MagC;
3312 if (match(Mag, m_APFloat(MagC)) && MagC->isNegative()) {
3313 APFloat PosMagC = *MagC;
3314 PosMagC.clearSign();
3315 return replaceInstUsesWith(
3316 *II, Builder.CreateCopySign(ConstantFP::get(Mag->getType(), PosMagC),
3317 Sign, II));
3318 }
3319
3320 // Peek through changes of magnitude's sign-bit. This call rewrites those:
3321 // copysign (fabs X), Sign --> copysign X, Sign
3322 // copysign (fneg X), Sign --> copysign X, Sign
3323 if (match(Mag, m_FAbs(m_Value(X))) || match(Mag, m_FNeg(m_Value(X))))
3324 return replaceInstUsesWith(*II, Builder.CreateCopySign(X, Sign, II));
3325
3326 // copysign(floor(fabs(X)), X) --> copysign(trunc(X), X)
3327 // copysign ignores the sign bit of its magnitude argument (implicit fabs),
3328 // so replacing floor(fabs(X)) with trunc(X) is correct for all inputs
3329 // including NaN without requiring nnan. The m_FAbs match also ensures
3330 // the floor argument is non-negative, so floor == trunc.
3331 Value *FAbsArg;
3332 if (match(Mag, m_Intrinsic<Intrinsic::floor>(m_FAbs(m_Value(FAbsArg)))) &&
3333 FAbsArg == Sign) {
3334 Value *Trunc = Builder.CreateUnaryIntrinsic(Intrinsic::trunc, Sign, II);
3335 return replaceInstUsesWith(*II, Builder.CreateCopySign(Trunc, Sign, II));
3336 }
3337
3338 Type *SignEltTy = Sign->getType()->getScalarType();
3339
3340 Value *CastSrc;
3341 if (match(Sign,
3343 CastSrc->getType()->isIntOrIntVectorTy() &&
3347 APInt::getSignMask(Known.getBitWidth()), Known,
3348 SQ))
3349 return II;
3350 }
3351
3352 break;
3353 }
3354 case Intrinsic::fabs: {
3355 Value *Cond, *TVal, *FVal;
3356 Value *Arg = II->getArgOperand(0);
3357 Value *X;
3358 // fabs (-X) --> fabs (X)
3359 if (match(Arg, m_FNeg(m_Value(X)))) {
3360 Value *Fabs = Builder.CreateFAbs(X, II);
3361 return replaceInstUsesWith(CI, Fabs);
3362 }
3363
3364 if (match(Arg, m_Select(m_Value(Cond), m_Value(TVal), m_Value(FVal)))) {
3365 // fabs (select Cond, TrueC, FalseC) --> select Cond, AbsT, AbsF
3366 if (Arg->hasOneUse() ? (isa<Constant>(TVal) || isa<Constant>(FVal))
3367 : (isa<Constant>(TVal) && isa<Constant>(FVal))) {
3368 CallInst *AbsT = Builder.CreateCall(II->getCalledFunction(), {TVal});
3369 CallInst *AbsF = Builder.CreateCall(II->getCalledFunction(), {FVal});
3370 // Given the condition is the same, we pull metadata (particularly
3371 // profile metadata) from the original select instruction.
3373 Cond, AbsT, AbsF, "", nullptr,
3375 SI->setFastMathFlags(II->getFastMathFlags() |
3376 cast<SelectInst>(Arg)->getFastMathFlags());
3377 // Can't copy nsz to select, as even with the nsz flag the fabs result
3378 // always has the sign bit unset.
3379 SI->setHasNoSignedZeros(false);
3380 return SI;
3381 }
3382 // fabs (select Cond, -FVal, FVal) --> fabs FVal
3383 if (match(TVal, m_FNeg(m_Specific(FVal))))
3384 return replaceInstUsesWith(*II, Builder.CreateFAbs(FVal, II));
3385 // fabs (select Cond, TVal, -TVal) --> fabs TVal
3386 if (match(FVal, m_FNeg(m_Specific(TVal))))
3387 return replaceInstUsesWith(*II, Builder.CreateFAbs(TVal, II));
3388 }
3389
3390 Value *Magnitude, *Sign;
3391 if (match(II->getArgOperand(0),
3392 m_CopySign(m_Value(Magnitude), m_Value(Sign)))) {
3393 // fabs (copysign x, y) -> (fabs x)
3394 Value *AbsSign = Builder.CreateFAbs(Magnitude, II);
3395 return replaceInstUsesWith(*II, AbsSign);
3396 }
3397
3398 [[fallthrough]];
3399 }
3400 case Intrinsic::ceil:
3401 case Intrinsic::floor:
3402 case Intrinsic::round:
3403 case Intrinsic::roundeven:
3404 case Intrinsic::nearbyint:
3405 case Intrinsic::rint:
3406 case Intrinsic::trunc: {
3407 Value *ExtSrc;
3408 if (match(II->getArgOperand(0), m_OneUse(m_FPExt(m_Value(ExtSrc))))) {
3409 // Narrow the call: intrinsic (fpext x) -> fpext (intrinsic x)
3410 Value *NarrowII = Builder.CreateUnaryIntrinsic(IID, ExtSrc, II);
3411 return new FPExtInst(NarrowII, II->getType());
3412 }
3413 break;
3414 }
3415 case Intrinsic::cos:
3416 case Intrinsic::amdgcn_cos:
3417 case Intrinsic::cosh: {
3418 Value *X, *Sign;
3419 Value *Src = II->getArgOperand(0);
3420 if (match(Src, m_FNeg(m_Value(X))) || match(Src, m_FAbs(m_Value(X))) ||
3421 match(Src, m_CopySign(m_Value(X), m_Value(Sign)))) {
3422 // f(-x) --> f(x)
3423 // f(fabs(x)) --> f(x)
3424 // f(copysign(x, y)) --> f(x)
3425 // for f in {cos, cosh}
3426 return replaceInstUsesWith(*II, Builder.CreateUnaryIntrinsic(IID, X, II));
3427 }
3428 if (IID == Intrinsic::cos) {
3429 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3430 return replaceInstUsesWith(*II, Result);
3431 }
3432 break;
3433 }
3434 case Intrinsic::sin:
3435 case Intrinsic::amdgcn_sin:
3436 case Intrinsic::sinh:
3437 case Intrinsic::tan:
3438 case Intrinsic::tanh: {
3439 Value *X;
3440 if (match(II->getArgOperand(0), m_OneUse(m_FNeg(m_Value(X)))) &&
3442 // f(-x) --> -f(x)
3443 // for f in {sin, sinh, tan, tanh}
3444 Value *NewFunc = Builder.CreateUnaryIntrinsic(IID, X, II);
3445 return UnaryOperator::CreateFNegFMF(NewFunc, II);
3446 }
3447 if (IID == Intrinsic::sin) {
3448 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3449 return replaceInstUsesWith(*II, Result);
3450 }
3451 break;
3452 }
3453 case Intrinsic::ldexp: {
3454 Value *Src = II->getArgOperand(0);
3455 Value *Exp = II->getArgOperand(1);
3456
3457 // ldexp(x, K) -> fmul x, 2^K
3458 uint64_t ConstExp;
3459 if (match(Exp, m_ConstantInt(ConstExp))) {
3460 const fltSemantics &FPTy =
3461 Src->getType()->getScalarType()->getFltSemantics();
3462
3463 APFloat Scaled = scalbn(APFloat::getOne(FPTy), static_cast<int>(ConstExp),
3465 if (!Scaled.isZero() && !Scaled.isInfinity()) {
3466 // Skip overflow and underflow cases.
3467 Constant *FPConst = ConstantFP::get(Src->getType(), Scaled);
3468 return BinaryOperator::CreateFMulFMF(Src, FPConst, II);
3469 }
3470 }
3471
3472 // ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b))
3473 //
3474 // A danger is if the first ldexp would overflow to infinity or underflow to
3475 // zero, but the combined exponent avoids it.
3476 //
3477 // We ignore this with reassoc, or if we know both exponents have the same
3478 // sign (since then we'd just double down on the over/underflow which would
3479 // occur anyway).
3480 //
3481 // ldexp can take arbitrary integer types, so we also need to ensure that
3482 // our exponent type is wide enough so that if sadd.sat(a, b) saturates,
3483 // then ldexp at the saturated exponent saturates to inf or zero as well.
3484 //
3485 // TODO: Could do better if we had range tracking for the input value
3486 // exponent. Also could broaden sign check to cover == 0 case.
3487 Value *InnerSrc;
3488 Value *InnerExp;
3490 m_Value(InnerSrc), m_Value(InnerExp)))) &&
3491 Exp->getType() == InnerExp->getType()) {
3492 FastMathFlags FMF = II->getFastMathFlags();
3493 FastMathFlags InnerFlags = cast<FPMathOperator>(Src)->getFastMathFlags();
3494
3495 if (ldexpSaturatingAddIsSafe(II->getType(), Exp->getType()) &&
3496 ((FMF.allowReassoc() && InnerFlags.allowReassoc()) ||
3497 signBitMustBeTheSame(Exp, InnerExp, SQ.getWithInstruction(II)))) {
3498 Value *NewExp =
3499 Builder.CreateBinaryIntrinsic(Intrinsic::sadd_sat, InnerExp, Exp);
3500 return replaceInstUsesWith(
3501 *II, Builder.CreateLdexp(InnerSrc, NewExp, FMF | InnerFlags));
3502 }
3503 }
3504
3505 // ldexp(x, zext(i1 y)) -> fmul x, (select y, 2.0, 1.0)
3506 // ldexp(x, sext(i1 y)) -> fmul x, (select y, 0.5, 1.0)
3507 Value *ExtSrc;
3508 if (match(Exp, m_ZExt(m_Value(ExtSrc))) &&
3509 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3510 Value *Select =
3511 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 2.0),
3512 ConstantFP::get(II->getType(), 1.0));
3514 }
3515 if (match(Exp, m_SExt(m_Value(ExtSrc))) &&
3516 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3517 Value *Select =
3518 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 0.5),
3519 ConstantFP::get(II->getType(), 1.0));
3521 }
3522
3523 // ldexp(x, c ? exp : 0) -> c ? ldexp(x, exp) : x
3524 // ldexp(x, c ? 0 : exp) -> c ? x : ldexp(x, exp)
3525 ///
3526 // TODO: If we cared, should insert a canonicalize for x
3527 Value *SelectCond, *SelectLHS, *SelectRHS;
3528 if (match(II->getArgOperand(1),
3529 m_OneUse(m_Select(m_Value(SelectCond), m_Value(SelectLHS),
3530 m_Value(SelectRHS))))) {
3531 Value *NewLdexp = nullptr;
3532 Value *Select = nullptr;
3533 if (match(SelectRHS, m_ZeroInt())) {
3534 NewLdexp = Builder.CreateLdexp(Src, SelectLHS, II);
3535 Select = Builder.CreateSelect(SelectCond, NewLdexp, Src);
3536 } else if (match(SelectLHS, m_ZeroInt())) {
3537 NewLdexp = Builder.CreateLdexp(Src, SelectRHS, II);
3538 Select = Builder.CreateSelect(SelectCond, Src, NewLdexp);
3539 }
3540
3541 if (NewLdexp) {
3542 Select->takeName(II);
3543 return replaceInstUsesWith(*II, Select);
3544 }
3545 }
3546
3547 break;
3548 }
3549 case Intrinsic::ptrauth_auth:
3550 case Intrinsic::ptrauth_resign: {
3551 // (sign|resign) + (auth|resign) can be folded by omitting the middle
3552 // sign+auth component if the key and discriminator match.
3553 bool NeedSign = II->getIntrinsicID() == Intrinsic::ptrauth_resign;
3554 Value *Ptr = II->getArgOperand(0);
3555 Value *Key = II->getArgOperand(1);
3556 Value *Disc = II->getArgOperand(2);
3557 Value *DS = nullptr;
3558 if (auto Bundle = II->getOperandBundle(LLVMContext::OB_deactivation_symbol))
3559 DS = Bundle->Inputs[0];
3560
3561 // AuthKey will be the key we need to end up authenticating against in
3562 // whatever we replace this sequence with.
3563 Value *AuthKey = nullptr, *AuthDisc = nullptr, *BasePtr;
3564 if (const auto *CI = dyn_cast<CallBase>(Ptr)) {
3565 Value *OtherDS = nullptr;
3566 if (auto Bundle =
3568 OtherDS = Bundle->Inputs[0];
3569 if (DS != OtherDS)
3570 break;
3571
3572 if (CI->getIntrinsicID() == Intrinsic::ptrauth_sign) {
3573 if (CI->getArgOperand(1) != Key || CI->getArgOperand(2) != Disc)
3574 break;
3575 } else if (CI->getIntrinsicID() == Intrinsic::ptrauth_resign) {
3576 // The resign intrinsic does not support deactivation symbols.
3577 assert(!DS);
3578 if (CI->getArgOperand(3) != Key || CI->getArgOperand(4) != Disc)
3579 break;
3580 AuthKey = CI->getArgOperand(1);
3581 AuthDisc = CI->getArgOperand(2);
3582 } else
3583 break;
3584 BasePtr = CI->getArgOperand(0);
3585 } else if (const auto *PtrToInt = dyn_cast<PtrToIntOperator>(Ptr)) {
3586 // ptrauth constants are equivalent to a call to @llvm.ptrauth.sign for
3587 // our purposes, so check for that too.
3588 const auto *CPA = dyn_cast<ConstantPtrAuth>(PtrToInt->getOperand(0));
3589 if (!CPA || DS || !CPA->isKnownCompatibleWith(Key, Disc, DL))
3590 break;
3591
3592 // resign(ptrauth(p,ks,ds),ks,ds,kr,dr) -> ptrauth(p,kr,dr)
3593 if (NeedSign && isa<ConstantInt>(II->getArgOperand(4))) {
3594 auto *SignKey = cast<ConstantInt>(II->getArgOperand(3));
3595 auto *SignDisc = cast<ConstantInt>(II->getArgOperand(4));
3596 auto *Null = ConstantPointerNull::get(Builder.getPtrTy());
3597 auto *NewCPA = ConstantPtrAuth::get(CPA->getPointer(), SignKey,
3598 SignDisc, /*AddrDisc=*/Null,
3599 /*DeactivationSymbol=*/Null);
3601 *II, ConstantExpr::getPointerCast(NewCPA, II->getType()));
3602 return eraseInstFromFunction(*II);
3603 }
3604
3605 // auth(ptrauth(p,k,d),k,d) -> p
3606 BasePtr = Builder.CreatePtrToInt(CPA->getPointer(), II->getType());
3607 } else
3608 break;
3609
3610 unsigned NewIntrin;
3611 if (AuthKey && NeedSign) {
3612 // resign(0,1) + resign(1,2) = resign(0, 2)
3613 NewIntrin = Intrinsic::ptrauth_resign;
3614 } else if (AuthKey) {
3615 // resign(0,1) + auth(1) = auth(0)
3616 NewIntrin = Intrinsic::ptrauth_auth;
3617 } else if (NeedSign) {
3618 // sign(0) + resign(0, 1) = sign(1)
3619 NewIntrin = Intrinsic::ptrauth_sign;
3620 } else {
3621 // sign(0) + auth(0) = nop
3622 replaceInstUsesWith(*II, BasePtr);
3623 return eraseInstFromFunction(*II);
3624 }
3625
3626 SmallVector<Value *, 4> CallArgs;
3627 CallArgs.push_back(BasePtr);
3628 if (AuthKey) {
3629 CallArgs.push_back(AuthKey);
3630 CallArgs.push_back(AuthDisc);
3631 }
3632
3633 if (NeedSign) {
3634 CallArgs.push_back(II->getArgOperand(3));
3635 CallArgs.push_back(II->getArgOperand(4));
3636 }
3637
3638 std::vector<OperandBundleDef> Bundles;
3639 if (DS)
3640 Bundles.push_back(OperandBundleDef("deactivation-symbol", DS));
3641
3642 Function *NewFn =
3643 Intrinsic::getOrInsertDeclaration(II->getModule(), NewIntrin);
3644 return CallInst::Create(NewFn, CallArgs, Bundles);
3645 }
3646 case Intrinsic::arm_neon_vtbl1:
3647 case Intrinsic::arm_neon_vtbl2:
3648 case Intrinsic::arm_neon_vtbl3:
3649 case Intrinsic::arm_neon_vtbl4:
3650 case Intrinsic::aarch64_neon_tbl1:
3651 case Intrinsic::aarch64_neon_tbl2:
3652 case Intrinsic::aarch64_neon_tbl3:
3653 case Intrinsic::aarch64_neon_tbl4:
3654 return simplifyNeonTbl(*II, *this, /*IsExtension=*/false);
3655 case Intrinsic::arm_neon_vtbx1:
3656 case Intrinsic::arm_neon_vtbx2:
3657 case Intrinsic::arm_neon_vtbx3:
3658 case Intrinsic::arm_neon_vtbx4:
3659 case Intrinsic::aarch64_neon_tbx1:
3660 case Intrinsic::aarch64_neon_tbx2:
3661 case Intrinsic::aarch64_neon_tbx3:
3662 case Intrinsic::aarch64_neon_tbx4:
3663 return simplifyNeonTbl(*II, *this, /*IsExtension=*/true);
3664
3665 case Intrinsic::arm_neon_vmulls:
3666 case Intrinsic::arm_neon_vmullu:
3667 case Intrinsic::aarch64_neon_smull:
3668 case Intrinsic::aarch64_neon_umull: {
3669 Value *Arg0 = II->getArgOperand(0);
3670 Value *Arg1 = II->getArgOperand(1);
3671
3672 // Handle mul by zero first:
3674 return replaceInstUsesWith(CI, ConstantAggregateZero::get(II->getType()));
3675 }
3676
3677 // Check for constant LHS & RHS - in this case we just simplify.
3678 bool Zext = (IID == Intrinsic::arm_neon_vmullu ||
3679 IID == Intrinsic::aarch64_neon_umull);
3680 VectorType *NewVT = cast<VectorType>(II->getType());
3681 if (Constant *CV0 = dyn_cast<Constant>(Arg0)) {
3682 if (Constant *CV1 = dyn_cast<Constant>(Arg1)) {
3683 Value *V0 = Builder.CreateIntCast(CV0, NewVT, /*isSigned=*/!Zext);
3684 Value *V1 = Builder.CreateIntCast(CV1, NewVT, /*isSigned=*/!Zext);
3685 return replaceInstUsesWith(CI, Builder.CreateMul(V0, V1));
3686 }
3687
3688 // Couldn't simplify - canonicalize constant to the RHS.
3689 std::swap(Arg0, Arg1);
3690 }
3691
3692 // Handle mul by one:
3693 if (Constant *CV1 = dyn_cast<Constant>(Arg1))
3694 if (ConstantInt *Splat =
3695 dyn_cast_or_null<ConstantInt>(CV1->getSplatValue()))
3696 if (Splat->isOne())
3697 return CastInst::CreateIntegerCast(Arg0, II->getType(),
3698 /*isSigned=*/!Zext);
3699
3700 break;
3701 }
3702 case Intrinsic::arm_neon_aesd:
3703 case Intrinsic::arm_neon_aese:
3704 case Intrinsic::aarch64_crypto_aesd:
3705 case Intrinsic::aarch64_crypto_aese:
3706 case Intrinsic::aarch64_sve_aesd:
3707 case Intrinsic::aarch64_sve_aese: {
3708 Value *DataArg = II->getArgOperand(0);
3709 Value *KeyArg = II->getArgOperand(1);
3710
3711 // Accept zero on either operand.
3712 if (!match(KeyArg, m_ZeroInt()))
3713 std::swap(KeyArg, DataArg);
3714
3715 // Try to use the builtin XOR in AESE and AESD to eliminate a prior XOR
3716 Value *Data, *Key;
3717 if (match(KeyArg, m_ZeroInt()) &&
3718 match(DataArg, m_Xor(m_Value(Data), m_Value(Key)))) {
3719 replaceOperand(*II, 0, Data);
3720 replaceOperand(*II, 1, Key);
3721 return II;
3722 }
3723 break;
3724 }
3725 case Intrinsic::arm_neon_vshifts:
3726 case Intrinsic::arm_neon_vshiftu:
3727 case Intrinsic::aarch64_neon_sshl:
3728 case Intrinsic::aarch64_neon_ushl:
3729 return foldNeonShift(II, *this);
3730 case Intrinsic::hexagon_V6_vandvrt:
3731 case Intrinsic::hexagon_V6_vandvrt_128B: {
3732 // Simplify Q -> V -> Q conversion.
3733 if (auto Op0 = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3734 Intrinsic::ID ID0 = Op0->getIntrinsicID();
3735 if (ID0 != Intrinsic::hexagon_V6_vandqrt &&
3736 ID0 != Intrinsic::hexagon_V6_vandqrt_128B)
3737 break;
3738 Value *Bytes = Op0->getArgOperand(1), *Mask = II->getArgOperand(1);
3739 uint64_t Bytes1 = computeKnownBits(Bytes, Op0).One.getZExtValue();
3740 uint64_t Mask1 = computeKnownBits(Mask, II).One.getZExtValue();
3741 // Check if every byte has common bits in Bytes and Mask.
3742 uint64_t C = Bytes1 & Mask1;
3743 if ((C & 0xFF) && (C & 0xFF00) && (C & 0xFF0000) && (C & 0xFF000000))
3744 return replaceInstUsesWith(*II, Op0->getArgOperand(0));
3745 }
3746 break;
3747 }
3748 case Intrinsic::stackrestore: {
3749 enum class ClassifyResult {
3750 None,
3751 Alloca,
3752 StackRestore,
3753 CallWithSideEffects,
3754 };
3755 auto Classify = [](const Instruction *I) {
3756 if (isa<AllocaInst>(I))
3757 return ClassifyResult::Alloca;
3758
3759 if (auto *CI = dyn_cast<CallInst>(I)) {
3760 if (auto *II = dyn_cast<IntrinsicInst>(CI)) {
3761 if (II->getIntrinsicID() == Intrinsic::stackrestore)
3762 return ClassifyResult::StackRestore;
3763
3764 if (II->mayHaveSideEffects())
3765 return ClassifyResult::CallWithSideEffects;
3766 } else {
3767 // Consider all non-intrinsic calls to be side effects
3768 return ClassifyResult::CallWithSideEffects;
3769 }
3770 }
3771
3772 return ClassifyResult::None;
3773 };
3774
3775 // If the stacksave and the stackrestore are in the same BB, and there is
3776 // no intervening call, alloca, or stackrestore of a different stacksave,
3777 // remove the restore. This can happen when variable allocas are DCE'd.
3778 if (IntrinsicInst *SS = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3779 if (SS->getIntrinsicID() == Intrinsic::stacksave &&
3780 SS->getParent() == II->getParent()) {
3781 BasicBlock::iterator BI(SS);
3782 bool CannotRemove = false;
3783 for (++BI; &*BI != II; ++BI) {
3784 switch (Classify(&*BI)) {
3785 case ClassifyResult::None:
3786 // So far so good, look at next instructions.
3787 break;
3788
3789 case ClassifyResult::StackRestore:
3790 // If we found an intervening stackrestore for a different
3791 // stacksave, we can't remove the stackrestore. Otherwise, continue.
3792 if (cast<IntrinsicInst>(*BI).getArgOperand(0) != SS)
3793 CannotRemove = true;
3794 break;
3795
3796 case ClassifyResult::Alloca:
3797 case ClassifyResult::CallWithSideEffects:
3798 // If we found an alloca, a non-intrinsic call, or an intrinsic
3799 // call with side effects, we can't remove the stackrestore.
3800 CannotRemove = true;
3801 break;
3802 }
3803 if (CannotRemove)
3804 break;
3805 }
3806
3807 if (!CannotRemove)
3808 return eraseInstFromFunction(CI);
3809 }
3810 }
3811
3812 // Scan down this block to see if there is another stack restore in the
3813 // same block without an intervening call/alloca.
3815 Instruction *TI = II->getParent()->getTerminator();
3816 bool CannotRemove = false;
3817 for (++BI; &*BI != TI; ++BI) {
3818 switch (Classify(&*BI)) {
3819 case ClassifyResult::None:
3820 // So far so good, look at next instructions.
3821 break;
3822
3823 case ClassifyResult::StackRestore:
3824 // If there is a stackrestore below this one, remove this one.
3825 return eraseInstFromFunction(CI);
3826
3827 case ClassifyResult::Alloca:
3828 case ClassifyResult::CallWithSideEffects:
3829 // If we found an alloca, a non-intrinsic call, or an intrinsic call
3830 // with side effects (such as llvm.stacksave and llvm.read_register),
3831 // we can't remove the stack restore.
3832 CannotRemove = true;
3833 break;
3834 }
3835 if (CannotRemove)
3836 break;
3837 }
3838
3839 // If the stack restore is in a return, resume, or unwind block and if there
3840 // are no allocas or calls between the restore and the return, nuke the
3841 // restore.
3842 if (!CannotRemove && (isa<ReturnInst>(TI) || isa<ResumeInst>(TI)))
3843 return eraseInstFromFunction(CI);
3844 break;
3845 }
3846 case Intrinsic::lifetime_end:
3847 // Asan needs to poison memory to detect invalid access which is possible
3848 // even for empty lifetime range.
3849 if (II->getFunction()->hasFnAttribute(Attribute::SanitizeAddress) ||
3850 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemory) ||
3851 II->getFunction()->hasFnAttribute(Attribute::SanitizeHWAddress) ||
3852 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemTag))
3853 break;
3854
3855 if (removeTriviallyEmptyRange(*II, *this, [](const IntrinsicInst &I) {
3856 return I.getIntrinsicID() == Intrinsic::lifetime_start;
3857 }))
3858 return nullptr;
3859 break;
3860 case Intrinsic::assume: {
3861 for (auto [Idx, OBU] : llvm::enumerate(II->operand_bundles())) {
3862 auto RemoveBundle = [&, Idx = Idx]() -> Instruction * {
3863 if (II->getNumOperandBundles() == 1)
3864 return eraseInstFromFunction(*II);
3866 };
3867
3868 switch (getBundleAttrFromOBU(OBU)) {
3869 case BundleAttr::None:
3870 llvm_unreachable("Unexpected Attribute");
3871 case BundleAttr::Align: {
3872 // Try to remove redundant alignment assumptions.
3873 auto [Ptr, _, OffsetPtr, Alignment, Offset] = getAssumeAlignInfo(OBU);
3874
3875 if (!Alignment)
3876 break;
3877
3878 // Remove align 1 and non-power-of-two bundles; they don't add any
3879 // useful information.
3880 if (*Alignment == 1 || !isPowerOf2_64(*Alignment))
3881 return RemoveBundle();
3882
3883 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3884 GEP &&
3885 GEP->getMaxPreservedAlignment(getDataLayout()) >= *Alignment) {
3886 Builder.CreateAlignmentAssumption(
3887 getDataLayout(), GEP->getPointerOperand(), *Alignment,
3888 OffsetPtr ? const_cast<Value *>(OffsetPtr->get()) : nullptr);
3889 return RemoveBundle();
3890 }
3891
3892 if (!Offset)
3893 break;
3894
3895 Value *BasePtr;
3896 const APInt *PtrOffset;
3897 if (match(Ptr.get(), m_PtrAdd(m_Value(BasePtr), m_APInt(PtrOffset)))) {
3898 auto PtrOffsetVal =
3899 PtrOffset->sextOrTrunc(DL.getIndexTypeSizeInBits(Ptr->getType()))
3900 .trySExtValue();
3901 if (!PtrOffsetVal)
3902 break;
3903 Builder.CreateAlignmentAssumption(
3904 DL, BasePtr, *Alignment,
3905 Builder.getInt64(*Offset - *PtrOffsetVal));
3906 return RemoveBundle();
3907 }
3908
3909 // Don't try to remove align assumptions for pointers derived from
3910 // arguments. We might lose information if the function gets inline and
3911 // the align argument attribute disappears.
3912 Value *UO = getUnderlyingObject(Ptr);
3913 if (!UO || isa<Argument>(UO))
3914 break;
3915
3916 // Compute known bits for the pointer and drop the assume if the
3917 // known alignment isn't increased by it.
3918 auto AlignMask = (*Alignment - 1);
3919 if (KnownBits KB = computeKnownBits(Ptr, II);
3920 (KB.Zero & AlignMask) == (~*Offset & AlignMask) &&
3921 (KB.One & AlignMask) == (*Offset & AlignMask))
3922 return RemoveBundle();
3923 break;
3924 }
3925
3926 case BundleAttr::Dereferenceable: {
3927 auto [Ptr, _, Count] = getAssumeDereferenceableInfo(OBU);
3928
3929 if (!Count)
3930 break;
3931
3932 if (*Count == 0 ||
3934 getSimplifyQuery().getWithInstruction(II)))
3935 return RemoveBundle();
3936
3937 break;
3938 }
3939
3940 case BundleAttr::Ignore:
3941 return RemoveBundle();
3942
3943 case BundleAttr::NonNull: {
3944 auto [Ptr] = llvm::getAssumeNonNullInfo(OBU);
3945
3946 // Drop assume if we can prove nonnull without it
3947 if (isKnownNonZero(Ptr, getSimplifyQuery().getWithInstruction(II)))
3948 return RemoveBundle();
3949
3950 // Fold the assume into metadata if it's valid at the load
3951 if (auto *LI = dyn_cast<LoadInst>(Ptr);
3952 LI &&
3953 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3954 MDNode *MD = MDNode::get(II->getContext(), {});
3955 LI->setMetadata(LLVMContext::MD_nonnull, MD);
3956 LI->setMetadata(LLVMContext::MD_noundef, MD);
3957 return RemoveBundle();
3958 }
3959
3960 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3961 GEP && GEP->isInBounds() &&
3962 !NullPointerIsDefined(II->getFunction(),
3963 Ptr->getType()->getPointerAddressSpace())) {
3964 Builder.CreateNonnullAssumption(GEP->stripInBoundsOffsets());
3965 return RemoveBundle();
3966 }
3967
3968 // TODO: apply nonnull return attributes to calls and invokes
3969 break;
3970 }
3971
3972 case BundleAttr::NoUndef: {
3973 auto [Val] = getAssumeNoUndefInfo(OBU);
3974
3976 return RemoveBundle();
3977
3978 if (auto *LI = dyn_cast<LoadInst>(Val);
3979 LI &&
3980 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3981 LI->setMetadata(LLVMContext::MD_noundef,
3982 MDNode::get(II->getContext(), {}));
3983 return RemoveBundle();
3984 }
3985
3986 } break;
3987
3988 case BundleAttr::SeparateStorage: {
3989 auto [Ptr1, Ptr2] = getAssumeSeparateStorageInfo(OBU);
3990 // Separate storage assumptions apply to the underlying allocations, not
3991 // any particular pointer within them. When evaluating the hints for AA
3992 // purposes we getUnderlyingObject them; by precomputing the answers
3993 // here we can avoid having to do so repeatedly there.
3994 auto MaybeSimplifyHint = [&](const Use &U) {
3995 Value *Hint = U.get();
3996 // Not having a limit is safe because InstCombine removes unreachable
3997 // code.
3998 Value *UnderlyingObject = getUnderlyingObject(Hint, /*MaxLookup*/ 0);
3999 if (Hint != UnderlyingObject)
4000 replaceUse(const_cast<Use &>(U), UnderlyingObject);
4001 };
4002 MaybeSimplifyHint(Ptr1);
4003 MaybeSimplifyHint(Ptr2);
4004 } break;
4005
4006 // TODO: Drop these assumes when they are redundant
4007 case BundleAttr::DereferenceableOrNull:
4008 break;
4009
4010 // This cannot be simplified
4011 case BundleAttr::Cold:
4012 break;
4013 }
4014 }
4015
4016 // If the assume has operand bundles, the folds below will never work, so
4017 // don't bother trying.
4018 if (II->hasOperandBundles())
4019 break;
4020
4021 Value *IIOperand = II->getArgOperand(0);
4022
4023 // Canonicalize assume(a && b) -> assume(a); assume(b);
4024 // Note: New assumption intrinsics created here are registered by
4025 // the InstCombineIRInserter object.
4026 Value *A, *B;
4027 if (match(IIOperand, m_LogicalAnd(m_Value(A), m_Value(B)))) {
4028 Builder.CreateAssumption(A);
4029 Builder.CreateAssumption(B);
4030 return eraseInstFromFunction(*II);
4031 }
4032 // assume(!(a || b)) -> assume(!a); assume(!b);
4033 if (match(IIOperand, m_Not(m_LogicalOr(m_Value(A), m_Value(B))))) {
4034 Builder.CreateAssumption(Builder.CreateNot(A));
4035 Builder.CreateAssumption(Builder.CreateNot(B));
4036 return eraseInstFromFunction(*II);
4037 }
4038
4039 // Convert nonnull assume like:
4040 // %A = icmp ne i32* %PTR, null
4041 // call void @llvm.assume(i1 %A)
4042 // into
4043 // call void @llvm.assume(i1 true) [ "nonnull"(i32* %PTR) ]
4044 if (match(
4045 IIOperand,
4048 m_Zero())))) &&
4049 A->getType()->isPointerTy()) {
4050 Builder.CreateNonnullAssumption(A);
4051 return eraseInstFromFunction(*II);
4052 }
4053
4054 // Convert alignment assume like:
4055 // %B = ptrtoint ptr %A to i64
4056 // %C = and i64 %B, Constant
4057 // %D = icmp eq i64 %C, 0
4058 // call void @llvm.assume(i1 %D)
4059 // into
4060 // call void @llvm.assume(i1 true) [ "align"(ptr [[A]], i64 Constant + 1)]
4061 uint64_t AlignMask = 1;
4062 if ((match(IIOperand, m_Not(m_Trunc(m_Value(A)))) ||
4063 match(IIOperand,
4065 m_And(m_Value(A), m_ConstantInt(AlignMask)),
4066 m_Zero())))) {
4067 if (isPowerOf2_64(AlignMask + 1) &&
4069 Builder.CreateAlignmentAssumption(getDataLayout(), A, AlignMask + 1);
4070 return eraseInstFromFunction(*II);
4071 }
4072 }
4073
4074 // Remove assumes on true/false
4075 if (auto *CI = dyn_cast<ConstantInt>(IIOperand);
4076 CI || isa<UndefValue, PoisonValue>(IIOperand)) {
4077 if (!CI || CI->isZero())
4079 return eraseInstFromFunction(*II);
4080 }
4081
4082 // Update the cache of affected values for this assumption (we might be
4083 // here because we just simplified the condition).
4084 AC.updateAffectedValues(cast<AssumeInst>(II));
4085 break;
4086 }
4087 case Intrinsic::experimental_guard: {
4088 // Is this guard followed by another guard? We scan forward over a small
4089 // fixed window of instructions to handle common cases with conditions
4090 // computed between guards.
4091 Instruction *NextInst = II->getNextNode();
4092 for (unsigned i = 0; i < CLOpts.guard_widening_window; i++) {
4093 // Note: Using context-free form to avoid compile time blow up
4094 if (!isSafeToSpeculativelyExecute(NextInst))
4095 break;
4096 NextInst = NextInst->getNextNode();
4097 }
4098 Value *NextCond = nullptr;
4099 if (match(NextInst,
4101 Value *CurrCond = II->getArgOperand(0);
4102
4103 // Remove a guard that it is immediately preceded by an identical guard.
4104 // Otherwise canonicalize guard(a); guard(b) -> guard(a & b).
4105 if (CurrCond != NextCond) {
4106 Instruction *MoveI = II->getNextNode();
4107 while (MoveI != NextInst) {
4108 auto *Temp = MoveI;
4109 MoveI = MoveI->getNextNode();
4110 Temp->moveBefore(II->getIterator());
4111 }
4112 replaceOperand(*II, 0, Builder.CreateAnd(CurrCond, NextCond));
4113 }
4114 eraseInstFromFunction(*NextInst);
4115 return II;
4116 }
4117 break;
4118 }
4119 case Intrinsic::vector_insert: {
4120 Value *Vec = II->getArgOperand(0);
4121 Value *SubVec = II->getArgOperand(1);
4122 Value *Idx = II->getArgOperand(2);
4123 auto *DstTy = dyn_cast<FixedVectorType>(II->getType());
4124 auto *VecTy = dyn_cast<FixedVectorType>(Vec->getType());
4125 auto *SubVecTy = dyn_cast<FixedVectorType>(SubVec->getType());
4126
4127 // Only canonicalize if the destination vector, Vec, and SubVec are all
4128 // fixed vectors.
4129 if (DstTy && VecTy && SubVecTy) {
4130 unsigned DstNumElts = DstTy->getNumElements();
4131 unsigned VecNumElts = VecTy->getNumElements();
4132 unsigned SubVecNumElts = SubVecTy->getNumElements();
4133 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4134
4135 // An insert that entirely overwrites Vec with SubVec is a nop.
4136 if (VecNumElts == SubVecNumElts)
4137 return replaceInstUsesWith(CI, SubVec);
4138
4139 // Widen SubVec into a vector of the same width as Vec, since
4140 // shufflevector requires the two input vectors to be the same width.
4141 // Elements beyond the bounds of SubVec within the widened vector are
4142 // undefined.
4143 SmallVector<int, 8> WidenMask;
4144 unsigned i;
4145 for (i = 0; i != SubVecNumElts; ++i)
4146 WidenMask.push_back(i);
4147 for (; i != VecNumElts; ++i)
4148 WidenMask.push_back(PoisonMaskElem);
4149
4150 Value *WidenShuffle = Builder.CreateShuffleVector(SubVec, WidenMask);
4151
4153 for (unsigned i = 0; i != IdxN; ++i)
4154 Mask.push_back(i);
4155 for (unsigned i = DstNumElts; i != DstNumElts + SubVecNumElts; ++i)
4156 Mask.push_back(i);
4157 for (unsigned i = IdxN + SubVecNumElts; i != DstNumElts; ++i)
4158 Mask.push_back(i);
4159
4160 Value *Shuffle = Builder.CreateShuffleVector(Vec, WidenShuffle, Mask);
4161 return replaceInstUsesWith(CI, Shuffle);
4162 }
4163 break;
4164 }
4165 case Intrinsic::vector_extract: {
4166 Value *Vec = II->getArgOperand(0);
4167 Value *Idx = II->getArgOperand(1);
4168
4169 Type *ReturnType = II->getType();
4170 // (extract_vector (insert_vector InsertTuple, InsertValue, InsertIdx),
4171 // ExtractIdx)
4172 uint64_t ExtractIdx = cast<ConstantInt>(Idx)->getZExtValue();
4173 Value *InsertTuple, *InsertIdx, *InsertValue;
4175 m_Value(InsertValue),
4176 m_Value(InsertIdx))) &&
4177 InsertValue->getType() == ReturnType) {
4178 uint64_t Index = cast<ConstantInt>(InsertIdx)->getZExtValue();
4179 // Case where we get the same index right after setting it.
4180 // extract.vector(insert.vector(InsertTuple, InsertValue, Idx), Idx) -->
4181 // InsertValue
4182 if (ExtractIdx == Index)
4183 return replaceInstUsesWith(CI, InsertValue);
4184 // If we are getting a different index than what was set in the
4185 // insert.vector intrinsic. We can just set the input tuple to the one up
4186 // in the chain. extract.vector(insert.vector(InsertTuple, InsertValue,
4187 // InsertIndex), ExtractIndex)
4188 // --> extract.vector(InsertTuple, ExtractIndex)
4189 else
4190 return replaceOperand(CI, 0, InsertTuple);
4191 }
4192
4193 ConstantInt *ALMUpperBound;
4195 m_Value(), m_ConstantInt(ALMUpperBound)))) {
4196 const auto &Attrs = II->getFunction()->getAttributes().getFnAttrs();
4197 unsigned VScaleMin = Attrs.getVScaleRangeMin();
4198 unsigned ScaleFactor =
4199 cast<VectorType>(ReturnType)->isScalableTy() ? VScaleMin : 1;
4200 if (ExtractIdx * ScaleFactor >= ALMUpperBound->getZExtValue())
4201 return replaceInstUsesWith(CI,
4202 ConstantVector::getNullValue(ReturnType));
4203 }
4204
4205 auto *DstTy = dyn_cast<VectorType>(ReturnType);
4206 auto *VecTy = dyn_cast<VectorType>(Vec->getType());
4207
4208 if (DstTy && VecTy) {
4209 auto DstEltCnt = DstTy->getElementCount();
4210 auto VecEltCnt = VecTy->getElementCount();
4211 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4212
4213 // Extracting the entirety of Vec is a nop.
4214 if (DstEltCnt == VecTy->getElementCount()) {
4215 replaceInstUsesWith(CI, Vec);
4216 return eraseInstFromFunction(CI);
4217 }
4218
4219 // Only canonicalize to shufflevector if the destination vector and
4220 // Vec are fixed vectors.
4221 if (VecEltCnt.isScalable() || DstEltCnt.isScalable())
4222 break;
4223
4225 for (unsigned i = 0; i != DstEltCnt.getKnownMinValue(); ++i)
4226 Mask.push_back(IdxN + i);
4227
4228 Value *Shuffle = Builder.CreateShuffleVector(Vec, Mask);
4229 return replaceInstUsesWith(CI, Shuffle);
4230 }
4231 break;
4232 }
4233 case Intrinsic::experimental_vp_reverse: {
4234 Value *X;
4235 Value *Vec = II->getArgOperand(0);
4236 Value *Mask = II->getArgOperand(1);
4237 if (!match(Mask, m_AllOnes()))
4238 break;
4239 Value *EVL = II->getArgOperand(2);
4240 // TODO: Canonicalize experimental.vp.reverse after unop/binops?
4241 // rev(unop rev(X)) --> unop X
4242 if (match(Vec,
4244 m_Value(X), m_AllOnes(), m_Specific(EVL)))))) {
4245 auto *OldUnOp = cast<UnaryOperator>(Vec);
4247 OldUnOp->getOpcode(), X, OldUnOp, OldUnOp->getName(),
4248 II->getIterator());
4249 return replaceInstUsesWith(CI, NewUnOp);
4250 }
4251 break;
4252 }
4253 case Intrinsic::vector_reduce_or:
4254 case Intrinsic::vector_reduce_and: {
4255 // Canonicalize logical or/and reductions:
4256 // Or reduction for i1 is represented as:
4257 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4258 // %res = cmp ne iReduxWidth %val, 0
4259 // And reduction for i1 is represented as:
4260 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4261 // %res = cmp eq iReduxWidth %val, 11111
4262 Value *Arg = II->getArgOperand(0);
4263 Value *Vect;
4264
4265 if (Value *NewOp =
4266 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4267 replaceUse(II->getOperandUse(0), NewOp);
4268 return II;
4269 }
4270
4271 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4272 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4273 if (FTy->getElementType() == Builder.getInt1Ty()) {
4274 Value *Res = Builder.CreateBitCast(
4275 Vect, Builder.getIntNTy(FTy->getNumElements()));
4276 if (IID == Intrinsic::vector_reduce_and) {
4277 Res = Builder.CreateICmpEQ(
4279 } else {
4280 assert(IID == Intrinsic::vector_reduce_or &&
4281 "Expected or reduction.");
4282 Res = Builder.CreateIsNotNull(Res);
4283 }
4284 if (Arg != Vect)
4285 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4286 II->getType());
4287 return replaceInstUsesWith(CI, Res);
4288 }
4289 }
4290 [[fallthrough]];
4291 }
4292 case Intrinsic::vector_reduce_add: {
4293 if (IID == Intrinsic::vector_reduce_add) {
4294 // Convert vector_reduce_add(ZExt(<n x i1>)) to
4295 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4296 // Convert vector_reduce_add(SExt(<n x i1>)) to
4297 // -ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4298 // Convert vector_reduce_add(<n x i1>) to
4299 // Trunc(ctpop(bitcast <n x i1> to in)).
4300 Value *Arg = II->getArgOperand(0);
4301 Value *Vect;
4302
4303 if (Value *NewOp =
4304 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4305 replaceUse(II->getOperandUse(0), NewOp);
4306 return II;
4307 }
4308
4309 // vector.reduce.add.vNiM(splat(%x)) -> mul(%x, N)
4310 if (Value *Splat = getSplatValue(Arg)) {
4311 ElementCount VecToReduceCount =
4312 cast<VectorType>(Arg->getType())->getElementCount();
4313 if (VecToReduceCount.isFixed()) {
4314 unsigned VectorSize = VecToReduceCount.getFixedValue();
4315 return BinaryOperator::CreateMul(
4316 Splat,
4317 ConstantInt::get(Splat->getType(), VectorSize, /*IsSigned=*/false,
4318 /*ImplicitTrunc=*/true));
4319 }
4320 }
4321
4322 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4323 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4324 if (FTy->getElementType() == Builder.getInt1Ty()) {
4325 Value *V = Builder.CreateBitCast(
4326 Vect, Builder.getIntNTy(FTy->getNumElements()));
4327 Value *Res = Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
4328 Res = Builder.CreateZExtOrTrunc(Res, II->getType());
4329 if (Arg != Vect &&
4330 cast<Instruction>(Arg)->getOpcode() == Instruction::SExt)
4331 Res = Builder.CreateNeg(Res);
4332 return replaceInstUsesWith(CI, Res);
4333 }
4334 }
4335 }
4336 [[fallthrough]];
4337 }
4338 case Intrinsic::vector_reduce_xor: {
4339 if (IID == Intrinsic::vector_reduce_xor) {
4340 // Exclusive disjunction reduction over the vector with
4341 // (potentially-extended) i1 element type is actually a
4342 // (potentially-extended) arithmetic `add` reduction over the original
4343 // non-extended value:
4344 // vector_reduce_xor(?ext(<n x i1>))
4345 // -->
4346 // ?ext(vector_reduce_add(<n x i1>))
4347 Value *Arg = II->getArgOperand(0);
4348 Value *Vect;
4349
4350 if (Value *NewOp =
4351 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4352 replaceUse(II->getOperandUse(0), NewOp);
4353 return II;
4354 }
4355
4356 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4357 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4358 if (VTy->getElementType() == Builder.getInt1Ty()) {
4359 Value *Res = Builder.CreateAddReduce(Vect);
4360 if (Arg != Vect)
4361 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4362 II->getType());
4363 return replaceInstUsesWith(CI, Res);
4364 }
4365 }
4366 }
4367 [[fallthrough]];
4368 }
4369 case Intrinsic::vector_reduce_mul: {
4370 if (IID == Intrinsic::vector_reduce_mul) {
4371 Value *Arg = II->getArgOperand(0);
4372
4373 if (Value *NewOp =
4374 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4375 replaceUse(II->getOperandUse(0), NewOp);
4376 return II;
4377 }
4378
4379 // vector_reduce_mul(zext(<n x i1>)), or
4380 // vector_reduce_mul(sext(<n x i1>)) (if n is even) -->
4381 // zext(vector_reduce_and(<n x i1>)).
4382 // (The sext case doesn't work if n is odd because multiplying an odd
4383 // number of -1's produces -1, not 1.)
4384 Value *Vect;
4385 bool IsZext = match(Arg, m_ZExt(m_Value(Vect))) &&
4386 Vect->getType()->isIntOrIntVectorTy(1);
4387 bool IsSext =
4388 match(Arg, m_SExt(m_Value(Vect))) &&
4389 Vect->getType()->isIntOrIntVectorTy(1) &&
4390 cast<VectorType>(Vect->getType())->getElementCount().isKnownEven();
4391 if (IsZext || IsSext) {
4392 Value *Res = Builder.CreateAndReduce(Vect);
4393 return CastInst::Create(Instruction::ZExt, Res, II->getType());
4394 }
4395
4396 // vector_reduce_mul(<n x i1>) --> vector_reduce_and(<n x i1>)
4397 if (Arg->getType()->isIntOrIntVectorTy(1))
4398 return replaceInstUsesWith(CI, Builder.CreateAndReduce(Arg));
4399 }
4400 [[fallthrough]];
4401 }
4402 case Intrinsic::vector_reduce_umin:
4403 case Intrinsic::vector_reduce_umax: {
4404 if (IID == Intrinsic::vector_reduce_umin ||
4405 IID == Intrinsic::vector_reduce_umax) {
4406 // UMin/UMax reduction over the vector with (potentially-extended)
4407 // i1 element type is actually a (potentially-extended)
4408 // logical `and`/`or` reduction over the original non-extended value:
4409 // vector_reduce_u{min,max}(?ext(<n x i1>))
4410 // -->
4411 // ?ext(vector_reduce_{and,or}(<n x i1>))
4412 Value *Arg = II->getArgOperand(0);
4413 Value *Vect;
4414
4415 if (Value *NewOp =
4416 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4417 replaceUse(II->getOperandUse(0), NewOp);
4418 return II;
4419 }
4420
4421 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4422 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4423 if (VTy->getElementType() == Builder.getInt1Ty()) {
4424 Value *Res = IID == Intrinsic::vector_reduce_umin
4425 ? Builder.CreateAndReduce(Vect)
4426 : Builder.CreateOrReduce(Vect);
4427 if (Arg != Vect)
4428 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4429 II->getType());
4430 return replaceInstUsesWith(CI, Res);
4431 }
4432 }
4433 }
4434 [[fallthrough]];
4435 }
4436 case Intrinsic::vector_reduce_smin:
4437 case Intrinsic::vector_reduce_smax: {
4438 if (IID == Intrinsic::vector_reduce_smin ||
4439 IID == Intrinsic::vector_reduce_smax) {
4440 // SMin/SMax reduction over the vector with (potentially-extended)
4441 // i1 element type is actually a (potentially-extended)
4442 // logical `and`/`or` reduction over the original non-extended value:
4443 // vector_reduce_s{min,max}(<n x i1>)
4444 // -->
4445 // vector_reduce_{or,and}(<n x i1>)
4446 // and
4447 // vector_reduce_s{min,max}(sext(<n x i1>))
4448 // -->
4449 // sext(vector_reduce_{or,and}(<n x i1>))
4450 // and
4451 // vector_reduce_s{min,max}(zext(<n x i1>))
4452 // -->
4453 // zext(vector_reduce_{and,or}(<n x i1>))
4454 Value *Arg = II->getArgOperand(0);
4455 Value *Vect;
4456
4457 if (Value *NewOp =
4458 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4459 replaceUse(II->getOperandUse(0), NewOp);
4460 return II;
4461 }
4462
4463 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4464 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4465 if (VTy->getElementType() == Builder.getInt1Ty()) {
4466 Instruction::CastOps ExtOpc = Instruction::CastOps::CastOpsEnd;
4467 if (Arg != Vect)
4468 ExtOpc = cast<CastInst>(Arg)->getOpcode();
4469 Value *Res = ((IID == Intrinsic::vector_reduce_smin) ==
4470 (ExtOpc == Instruction::CastOps::ZExt))
4471 ? Builder.CreateAndReduce(Vect)
4472 : Builder.CreateOrReduce(Vect);
4473 if (Arg != Vect)
4474 Res = Builder.CreateCast(ExtOpc, Res, II->getType());
4475 return replaceInstUsesWith(CI, Res);
4476 }
4477 }
4478 }
4479 [[fallthrough]];
4480 }
4481 case Intrinsic::vector_reduce_fmax:
4482 case Intrinsic::vector_reduce_fmin:
4483 case Intrinsic::vector_reduce_fadd:
4484 case Intrinsic::vector_reduce_fmul: {
4485 bool CanReorderLanes = (IID != Intrinsic::vector_reduce_fadd &&
4486 IID != Intrinsic::vector_reduce_fmul) ||
4487 II->hasAllowReassoc();
4488 const unsigned ArgIdx = (IID == Intrinsic::vector_reduce_fadd ||
4489 IID == Intrinsic::vector_reduce_fmul)
4490 ? 1
4491 : 0;
4492 Value *Arg = II->getArgOperand(ArgIdx);
4493 if (Value *NewOp = simplifyReductionOperand(Arg, CanReorderLanes)) {
4494 replaceUse(II->getOperandUse(ArgIdx), NewOp);
4495 return nullptr;
4496 }
4497 break;
4498 }
4499 case Intrinsic::is_fpclass: {
4500 if (Instruction *I = foldIntrinsicIsFPClass(*II))
4501 return I;
4502 break;
4503 }
4504 case Intrinsic::threadlocal_address: {
4505 Align MinAlign = getKnownAlignment(II->getArgOperand(0), DL, II, &AC, &DT);
4506 MaybeAlign Align = II->getRetAlign();
4507 if (MinAlign > Align.valueOrOne()) {
4508 II->addRetAttr(Attribute::getWithAlignment(II->getContext(), MinAlign));
4509 return II;
4510 }
4511 break;
4512 }
4513 case Intrinsic::fptoui_sat:
4514 case Intrinsic::fptosi_sat:
4515 if (Instruction *I = foldItoFPtoI(*II))
4516 return I;
4517 break;
4518 case Intrinsic::frexp: {
4519 // frexp(frexp(x).fract) -> { frexp(x).fract, 0 }: the fraction operand is
4520 // already normalized, so the first result is idempotent and the second is
4521 // zero.
4522 if (match(II->getArgOperand(0),
4524 Value *Res = Builder.CreateInsertValue(PoisonValue::get(II->getType()),
4525 II->getArgOperand(0), 0);
4526 Res = Builder.CreateInsertValue(
4527 Res, Constant::getNullValue(II->getType()->getStructElementType(1)),
4528 1);
4529 return replaceInstUsesWith(*II, Res);
4530 }
4531 break;
4532 }
4533 case Intrinsic::get_active_lane_mask: {
4534 const APInt *Op0, *Op1;
4535 if (match(II->getOperand(0), m_StrictlyPositive(Op0)) &&
4536 match(II->getOperand(1), m_APInt(Op1))) {
4537 Type *OpTy = II->getOperand(0)->getType();
4538 return replaceInstUsesWith(
4539 *II, Builder.CreateIntrinsic(
4540 II->getType(), Intrinsic::get_active_lane_mask,
4541 {Constant::getNullValue(OpTy),
4542 ConstantInt::get(OpTy, Op1->usub_sat(*Op0))}));
4543 }
4544 break;
4545 }
4546 case Intrinsic::experimental_get_vector_length: {
4547 // get.vector.length(Cnt, MaxLanes) --> Cnt when Cnt <= MaxLanes
4548 unsigned BitWidth =
4549 std::max(II->getArgOperand(0)->getType()->getScalarSizeInBits(),
4550 II->getType()->getScalarSizeInBits());
4551 ConstantRange Cnt =
4552 computeConstantRangeIncludingKnownBits(II->getArgOperand(0), false,
4553 SQ.getWithInstruction(II))
4555 ConstantRange MaxLanes = cast<ConstantInt>(II->getArgOperand(1))
4556 ->getValue()
4557 .zextOrTrunc(Cnt.getBitWidth());
4558 if (cast<ConstantInt>(II->getArgOperand(2))->isOne())
4559 MaxLanes = MaxLanes.multiply(
4560 getVScaleRange(II->getFunction(), Cnt.getBitWidth()));
4561
4562 if (Cnt.icmp(CmpInst::ICMP_ULE, MaxLanes))
4563 return replaceInstUsesWith(
4564 *II, Builder.CreateZExtOrTrunc(II->getArgOperand(0), II->getType()));
4565 return nullptr;
4566 }
4567 default: {
4568 // Handle target specific intrinsics
4569 std::optional<Instruction *> V = targetInstCombineIntrinsic(*II);
4570 if (V)
4571 return *V;
4572 break;
4573 }
4574 }
4575
4576 // Try to fold intrinsic into select/phi operands. This is legal if:
4577 // * The intrinsic is speculatable.
4578 // * The operand is one of the following:
4579 // - a phi.
4580 // - a select with a scalar condition.
4581 // - a select with a vector condition and II is not a cross lane operation.
4583 for (Value *Op : II->args()) {
4584 if (auto *Sel = dyn_cast<SelectInst>(Op)) {
4585 bool IsVectorCond = Sel->getCondition()->getType()->isVectorTy();
4586 if (IsVectorCond &&
4587 (!isNotCrossLaneOperation(II) || !II->getType()->isVectorTy()))
4588 continue;
4589 // Don't replace a scalar select with a more expensive vector select if
4590 // we can't simplify both arms of the select.
4591 bool SimplifyBothArms =
4592 !Op->getType()->isVectorTy() && II->getType()->isVectorTy();
4594 *II, Sel, /*FoldWithMultiUse=*/false, SimplifyBothArms))
4595 return R;
4596 }
4597 if (auto *Phi = dyn_cast<PHINode>(Op))
4598 if (Instruction *R = foldOpIntoPhi(*II, Phi))
4599 return R;
4600 }
4601 }
4602
4604 return Shuf;
4605
4607 return replaceInstUsesWith(*II, Reverse);
4608
4610 return replaceInstUsesWith(*II, Res);
4611
4612 // Some intrinsics (like experimental_gc_statepoint) can be used in invoke
4613 // context, so it is handled in visitCallBase and we should trigger it.
4614 return visitCallBase(*II);
4615}
4616
4617// Fence instruction simplification
4619 auto *NFI = dyn_cast<FenceInst>(FI.getNextNode());
4620 // This check is solely here to handle arbitrary target-dependent syncscopes.
4621 // TODO: Can remove if does not matter in practice.
4622 if (NFI && FI.isIdenticalTo(NFI))
4623 return eraseInstFromFunction(FI);
4624
4625 // Returns true if FI1 is identical or stronger fence than FI2.
4626 auto isIdenticalOrStrongerFence = [](FenceInst *FI1, FenceInst *FI2) {
4627 auto FI1SyncScope = FI1->getSyncScopeID();
4628 // Consider same scope, where scope is global or single-thread.
4629 if (FI1SyncScope != FI2->getSyncScopeID() ||
4630 (FI1SyncScope != SyncScope::System &&
4631 FI1SyncScope != SyncScope::SingleThread))
4632 return false;
4633
4634 return isAtLeastOrStrongerThan(FI1->getOrdering(), FI2->getOrdering());
4635 };
4636 if (NFI && isIdenticalOrStrongerFence(NFI, &FI))
4637 return eraseInstFromFunction(FI);
4638
4639 if (auto *PFI = dyn_cast_or_null<FenceInst>(FI.getPrevNode()))
4640 if (isIdenticalOrStrongerFence(PFI, &FI))
4641 return eraseInstFromFunction(FI);
4642 return nullptr;
4643}
4644
4645// InvokeInst simplification
4647 return visitCallBase(II);
4648}
4649
4650// CallBrInst simplification
4652 return visitCallBase(CBI);
4653}
4654
4655// A simple parser for format string specifiers for the purposes of the
4656// modular-format attribute. In the case of malformed format strings this might
4657// under or over report the specifiers present, but such cases are undefined
4658// behavior.
4660 Bitset<256> Specifiers;
4661 for (size_t I = 0; I < FormatStr.size(); ++I) {
4662 if (FormatStr[I] != '%')
4663 continue;
4664
4665 // Check for escaped '%'.
4666 if (I + 1 < FormatStr.size() && FormatStr[I + 1] == '%') {
4667 ++I; // Skip the second '%'.
4668 continue;
4669 }
4670
4671 // Scan past allowed prefix characters.
4672 size_t J =
4673 FormatStr.find_first_not_of("0123456789-+ #0$.*'hlLjztqwvI", I + 1);
4674 if (J == StringRef::npos)
4675 break;
4676
4677 Specifiers.set(static_cast<unsigned char>(FormatStr[J]));
4678 I = J; // Resume search from after the specifier.
4679 }
4680 return Specifiers;
4681}
4682
4683static bool isAspectNeeded(StringRef Aspect, CallInst *CI,
4684 std::optional<unsigned> FirstArgIdx,
4685 const std::optional<Bitset<256>> &Specifiers) {
4686 if (Aspect == "float") {
4687 if (Specifiers) {
4688 static constexpr Bitset<256> FloatSpecifiers{'f', 'F', 'e', 'E',
4689 'g', 'G', 'a', 'A'};
4690 return (*Specifiers & FloatSpecifiers).any();
4691 }
4692 // Fallback to type-based check for dynamic format string.
4693 if (!FirstArgIdx)
4694 return true;
4695 return llvm::any_of(
4696 llvm::make_range(std::next(CI->arg_begin(), *FirstArgIdx),
4697 CI->arg_end()),
4698 [](Value *V) { return V->getType()->isFloatingPointTy(); });
4699 }
4700 if (Aspect == "fixed") {
4701 if (Specifiers) {
4702 static constexpr Bitset<256> FixedSpecifiers{'r', 'R', 'k', 'K'};
4703 return (*Specifiers & FixedSpecifiers).any();
4704 }
4705 // Fallback for fixed-point: assume needed if format is dynamic.
4706 return true;
4707 }
4708 // Unknown aspects are always considered to be needed.
4709 return true;
4710}
4711
4712static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M,
4713 IRBuilderBase &B) {
4714 SmallString<20> Name = ImplName;
4715 Name += '_';
4716 Name += Aspect;
4717 LLVMContext &Ctx = M->getContext();
4718 Function *RelocNoneFn =
4719 Intrinsic::getOrInsertDeclaration(M, Intrinsic::reloc_none);
4720 B.CreateCall(RelocNoneFn,
4721 {MetadataAsValue::get(Ctx, MDString::get(Ctx, Name))});
4722}
4723
4725 if (!CI->hasFnAttr("modular-format"))
4726 return nullptr;
4727
4729 llvm::split(CI->getFnAttr("modular-format").getValueAsString(), ','));
4730 if (Args.size() < 5)
4731 return nullptr;
4732
4733 StringRef FormatIdxStr = Args[1];
4734 StringRef FirstArgIdxStr = Args[2];
4735 StringRef FnName = Args[3];
4736 StringRef ImplName = Args[4];
4738
4739 unsigned FormatIdx;
4740 std::optional<unsigned> FirstArgIdx;
4741 [[maybe_unused]] bool Error;
4742 Error = FormatIdxStr.getAsInteger(10, FormatIdx);
4743 assert(!Error && "invalid format arg index");
4744 --FormatIdx; // 1-based to 0-based
4745
4746 FirstArgIdx.emplace();
4747 Error = FirstArgIdxStr.getAsInteger(10, *FirstArgIdx);
4748 assert(!Error && "invalid first arg index");
4749 if (*FirstArgIdx > 0)
4750 --*FirstArgIdx; // 1-based to 0-based
4751 else
4752 FirstArgIdx.reset();
4753
4754 if (AllAspects.empty())
4755 return nullptr;
4756
4757 Value *FormatVal = CI->getArgOperand(FormatIdx);
4758 StringRef FormatStr;
4759
4760 std::optional<Bitset<256>> Specifiers;
4761 if (getConstantStringInfo(FormatVal, FormatStr))
4762 Specifiers = parseFormatStringSpecifiers(FormatStr);
4763
4764 SmallVector<StringRef> NeededAspects;
4765 for (StringRef Aspect : AllAspects)
4766 if (isAspectNeeded(Aspect, CI, FirstArgIdx, Specifiers))
4767 NeededAspects.push_back(Aspect);
4768
4769 if (NeededAspects.size() == AllAspects.size())
4770 return nullptr;
4771
4772 Module *M = CI->getModule();
4773 LLVMContext &Ctx = M->getContext();
4774 Function *Callee = CI->getCalledFunction();
4775 FunctionCallee ModularFn = M->getOrInsertFunction(
4776 FnName, Callee->getFunctionType(),
4777 Callee->getAttributes().removeFnAttribute(Ctx, "modular-format"));
4778 CallInst *New = cast<CallInst>(CI->clone());
4779 New->setCalledFunction(ModularFn);
4780 New->removeFnAttr("modular-format");
4781 B.Insert(New);
4782
4783 llvm::sort(NeededAspects);
4784 for (StringRef Request : NeededAspects)
4785 referenceAspect(Request, ImplName, M, B);
4786
4787 return New;
4788}
4789
4790Instruction *InstCombinerImpl::tryOptimizeCall(CallInst *CI) {
4791 if (!CI->getCalledFunction()) return nullptr;
4792
4793 // Skip optimizing notail and musttail calls so
4794 // LibCallSimplifier::optimizeCall doesn't have to preserve those invariants.
4795 // LibCallSimplifier::optimizeCall should try to preserve tail calls though.
4796 if (CI->isMustTailCall() || CI->isNoTailCall())
4797 return nullptr;
4798
4799 auto InstCombineRAUW = [this](Instruction *From, Value *With) {
4800 replaceInstUsesWith(*From, With);
4801 };
4802 auto InstCombineErase = [this](Instruction *I) {
4804 };
4805 LibCallSimplifier Simplifier(DL, &TLI, &DT, &DC, &AC, ORE, BFI, PSI,
4806 InstCombineRAUW, InstCombineErase);
4807 if (Value *With = Simplifier.optimizeCall(CI, Builder)) {
4808 ++NumSimplified;
4809 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4810 }
4811 if (Value *With = optimizeModularFormat(CI, Builder)) {
4812 ++NumSimplified;
4813 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4814 }
4815
4816 return nullptr;
4817}
4818
4820 // Strip off at most one level of pointer casts, looking for an alloca. This
4821 // is good enough in practice and simpler than handling any number of casts.
4822 Value *Underlying = TrampMem->stripPointerCasts();
4823 if (Underlying != TrampMem &&
4824 (!Underlying->hasOneUse() || Underlying->user_back() != TrampMem))
4825 return nullptr;
4826 if (!isa<AllocaInst>(Underlying))
4827 return nullptr;
4828
4829 IntrinsicInst *InitTrampoline = nullptr;
4830 for (User *U : TrampMem->users()) {
4832 if (!II)
4833 return nullptr;
4834 if (II->getIntrinsicID() == Intrinsic::init_trampoline) {
4835 if (InitTrampoline)
4836 // More than one init_trampoline writes to this value. Give up.
4837 return nullptr;
4838 InitTrampoline = II;
4839 continue;
4840 }
4841 if (II->getIntrinsicID() == Intrinsic::adjust_trampoline)
4842 // Allow any number of calls to adjust.trampoline.
4843 continue;
4844 return nullptr;
4845 }
4846
4847 // No call to init.trampoline found.
4848 if (!InitTrampoline)
4849 return nullptr;
4850
4851 // Check that the alloca is being used in the expected way.
4852 if (InitTrampoline->getOperand(0) != TrampMem)
4853 return nullptr;
4854
4855 return InitTrampoline;
4856}
4857
4859 Value *TrampMem) {
4860 // Visit all the previous instructions in the basic block, and try to find a
4861 // init.trampoline which has a direct path to the adjust.trampoline.
4862 for (BasicBlock::iterator I = AdjustTramp->getIterator(),
4863 E = AdjustTramp->getParent()->begin();
4864 I != E;) {
4865 Instruction *Inst = &*--I;
4867 if (II->getIntrinsicID() == Intrinsic::init_trampoline &&
4868 II->getOperand(0) == TrampMem)
4869 return II;
4870 if (Inst->mayWriteToMemory())
4871 return nullptr;
4872 }
4873 return nullptr;
4874}
4875
4876// Given a call to llvm.adjust.trampoline, find and return the corresponding
4877// call to llvm.init.trampoline if the call to the trampoline can be optimized
4878// to a direct call to a function. Otherwise return NULL.
4880 Callee = Callee->stripPointerCasts();
4881 IntrinsicInst *AdjustTramp = dyn_cast<IntrinsicInst>(Callee);
4882 if (!AdjustTramp ||
4883 AdjustTramp->getIntrinsicID() != Intrinsic::adjust_trampoline)
4884 return nullptr;
4885
4886 Value *TrampMem = AdjustTramp->getOperand(0);
4887
4889 return IT;
4890 if (IntrinsicInst *IT = findInitTrampolineFromBB(AdjustTramp, TrampMem))
4891 return IT;
4892 return nullptr;
4893}
4894
4895Instruction *InstCombinerImpl::foldPtrAuthIntrinsicCallee(CallBase &Call) {
4896 const Value *Callee = Call.getCalledOperand();
4897 const auto *IPC = dyn_cast<IntToPtrInst>(Callee);
4898 if (!IPC || !IPC->isNoopCast(DL))
4899 return nullptr;
4900
4901 const auto *II = dyn_cast<IntrinsicInst>(IPC->getOperand(0));
4902 if (!II)
4903 return nullptr;
4904
4905 Intrinsic::ID IIID = II->getIntrinsicID();
4906 if (IIID != Intrinsic::ptrauth_resign && IIID != Intrinsic::ptrauth_sign)
4907 return nullptr;
4908
4909 // Isolate the ptrauth bundle from the others.
4910 std::optional<OperandBundleUse> PtrAuthBundleOrNone;
4912 for (unsigned BI = 0, BE = Call.getNumOperandBundles(); BI != BE; ++BI) {
4913 OperandBundleUse Bundle = Call.getOperandBundleAt(BI);
4914 if (Bundle.getTagID() == LLVMContext::OB_ptrauth)
4915 PtrAuthBundleOrNone = Bundle;
4916 else
4917 NewBundles.emplace_back(Bundle);
4918 }
4919
4920 if (!PtrAuthBundleOrNone)
4921 return nullptr;
4922
4923 Value *NewCallee = nullptr;
4924 switch (IIID) {
4925 // call(ptrauth.resign(p)), ["ptrauth"()] -> call p, ["ptrauth"()]
4926 // assuming the call bundle and the sign operands match.
4927 case Intrinsic::ptrauth_resign: {
4928 // Resign result key should match bundle.
4929 if (II->getOperand(3) != PtrAuthBundleOrNone->Inputs[0])
4930 return nullptr;
4931 // Resign result discriminator should match bundle.
4932 if (II->getOperand(4) != PtrAuthBundleOrNone->Inputs[1])
4933 return nullptr;
4934
4935 // Resign input (auth) key should also match: we can't change the key on
4936 // the new call we're generating, because we don't know what keys are valid.
4937 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4938 return nullptr;
4939
4940 Value *NewBundleOps[] = {II->getOperand(1), II->getOperand(2)};
4941 NewBundles.emplace_back("ptrauth", NewBundleOps);
4942 NewCallee = II->getOperand(0);
4943 break;
4944 }
4945
4946 // call(ptrauth.sign(p)), ["ptrauth"()] -> call p
4947 // assuming the call bundle and the sign operands match.
4948 // Non-ptrauth indirect calls are undesirable, but so is ptrauth.sign.
4949 case Intrinsic::ptrauth_sign: {
4950 // Sign key should match bundle.
4951 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4952 return nullptr;
4953 // Sign discriminator should match bundle.
4954 if (II->getOperand(2) != PtrAuthBundleOrNone->Inputs[1])
4955 return nullptr;
4956 NewCallee = II->getOperand(0);
4957 break;
4958 }
4959 default:
4960 llvm_unreachable("unexpected intrinsic ID");
4961 }
4962
4963 if (!NewCallee)
4964 return nullptr;
4965
4966 NewCallee = Builder.CreateBitOrPointerCast(NewCallee, Callee->getType());
4967 CallBase *NewCall = CallBase::Create(&Call, NewBundles);
4968 NewCall->setCalledOperand(NewCallee);
4969 return NewCall;
4970}
4971
4972Instruction *InstCombinerImpl::foldPtrAuthConstantCallee(CallBase &Call) {
4974 if (!CPA)
4975 return nullptr;
4976
4977 auto *CalleeF = dyn_cast<Function>(CPA->getPointer());
4978 // If the ptrauth constant isn't based on a function pointer, bail out.
4979 if (!CalleeF)
4980 return nullptr;
4981
4982 // Inspect the call ptrauth bundle to check it matches the ptrauth constant.
4984 if (!PAB)
4985 return nullptr;
4986
4987 auto *Key = cast<ConstantInt>(PAB->Inputs[0]);
4988 Value *Discriminator = PAB->Inputs[1];
4989
4990 // If the bundle doesn't match, this is probably going to fail to auth.
4991 if (!CPA->isKnownCompatibleWith(Key, Discriminator, DL))
4992 return nullptr;
4993
4994 // If the bundle matches the constant, proceed in making this a direct call.
4996 NewCall->setCalledOperand(CalleeF);
4997 return NewCall;
4998}
4999
5000bool InstCombinerImpl::annotateAnyAllocSite(CallBase &Call,
5001 const TargetLibraryInfo *TLI) {
5002 // Note: We only handle cases which can't be driven from generic attributes
5003 // here. So, for example, nonnull and noalias (which are common properties
5004 // of some allocation functions) are expected to be handled via annotation
5005 // of the respective allocator declaration with generic attributes.
5006 bool Changed = false;
5007
5008 if (!Call.getType()->isPointerTy())
5009 return Changed;
5010
5011 std::optional<APInt> Size = getAllocSize(&Call, TLI);
5012 if (Size && *Size != 0) {
5013 // TODO: We really should just emit deref_or_null here and then
5014 // let the generic inference code combine that with nonnull.
5015 if (Call.hasRetAttr(Attribute::NonNull)) {
5016 Changed = !Call.hasRetAttr(Attribute::Dereferenceable);
5018 Call.getContext(), Size->getLimitedValue()));
5019 } else {
5020 Changed = !Call.hasRetAttr(Attribute::DereferenceableOrNull);
5022 Call.getContext(), Size->getLimitedValue()));
5023 }
5024 }
5025
5026 // Add alignment attribute if alignment is a power of two constant.
5028 if (!Alignment)
5029 return Changed;
5030
5031 ConstantInt *AlignOpC = dyn_cast<ConstantInt>(Alignment);
5032 if (AlignOpC && AlignOpC->getValue().ult(llvm::Value::MaximumAlignment)) {
5033 uint64_t AlignmentVal = AlignOpC->getZExtValue();
5034 if (llvm::isPowerOf2_64(AlignmentVal)) {
5035 Align ExistingAlign = Call.getRetAlign().valueOrOne();
5036 Align NewAlign = Align(AlignmentVal);
5037 if (NewAlign > ExistingAlign) {
5040 Changed = true;
5041 }
5042 }
5043 }
5044 return Changed;
5045}
5046
5047/// Improvements for call, callbr and invoke instructions.
5048Instruction *InstCombinerImpl::visitCallBase(CallBase &Call) {
5049 bool Changed = annotateAnyAllocSite(Call, &TLI);
5050
5051 // Mark any parameters that are known to be non-null with the nonnull
5052 // attribute. This is helpful for inlining calls to functions with null
5053 // checks on their arguments.
5054 SmallVector<unsigned, 4> ArgNos;
5055 unsigned ArgNo = 0;
5056
5057 for (Value *V : Call.args()) {
5058 if (V->getType()->isPointerTy()) {
5059 // Simplify the nonnull operand if the parameter is known to be nonnull.
5060 // Otherwise, try to infer nonnull for it.
5061 bool UseProvenance =
5064 V->getType()->getPointerAddressSpace());
5065 if (Call.paramHasAttr(ArgNo, Attribute::NonNull) || UseProvenance) {
5066 if (Value *Res = simplifyNonNullOperand(V, UseProvenance)) {
5067 replaceOperand(Call, ArgNo, Res);
5068 Changed = true;
5069 }
5070 } else if (isKnownNonZero(V,
5071 getSimplifyQuery().getWithInstruction(&Call))) {
5072 ArgNos.push_back(ArgNo);
5073 }
5074 }
5075 ArgNo++;
5076 }
5077
5078 assert(ArgNo == Call.arg_size() && "Call arguments not processed correctly.");
5079
5080 if (!ArgNos.empty()) {
5081 AttributeList AS = Call.getAttributes();
5082 LLVMContext &Ctx = Call.getContext();
5083 AS = AS.addParamAttribute(Ctx, ArgNos,
5084 Attribute::get(Ctx, Attribute::NonNull));
5085 Call.setAttributes(AS);
5086 Changed = true;
5087 }
5088
5089 // If the callee is a pointer to a function, attempt to move any casts to the
5090 // arguments of the call/callbr/invoke.
5092 Function *CalleeF = dyn_cast<Function>(Callee);
5093 if ((!CalleeF || CalleeF->getFunctionType() != Call.getFunctionType()) &&
5094 transformConstExprCastCall(Call))
5095 return nullptr;
5096
5097 if (CalleeF) {
5098 // Remove the convergent attr on calls when the callee is not convergent.
5099 if (Call.isConvergent() && !CalleeF->isConvergent() &&
5100 !CalleeF->isIntrinsic()) {
5101 LLVM_DEBUG(dbgs() << "Removing convergent attr from instr " << Call
5102 << "\n");
5104 return &Call;
5105 }
5106
5107 // If the call and callee calling conventions don't match, and neither one
5108 // of the calling conventions is compatible with C calling convention
5109 // this call must be unreachable, as the call is undefined.
5110 if ((CalleeF->getCallingConv() != Call.getCallingConv() &&
5111 !(CalleeF->getCallingConv() == llvm::CallingConv::C &&
5115 // Only do this for calls to a function with a body. A prototype may
5116 // not actually end up matching the implementation's calling conv for a
5117 // variety of reasons (e.g. it may be written in assembly).
5118 !CalleeF->isDeclaration()) {
5119 Instruction *OldCall = &Call;
5121 // If OldCall does not return void then replaceInstUsesWith poison.
5122 // This allows ValueHandlers and custom metadata to adjust itself.
5123 if (!OldCall->getType()->isVoidTy())
5124 replaceInstUsesWith(*OldCall, PoisonValue::get(OldCall->getType()));
5125 if (isa<CallInst>(OldCall))
5126 return eraseInstFromFunction(*OldCall);
5127
5128 // We cannot remove an invoke or a callbr, because it would change thexi
5129 // CFG, just change the callee to a null pointer.
5130 cast<CallBase>(OldCall)->setCalledFunction(
5131 CalleeF->getFunctionType(),
5132 Constant::getNullValue(CalleeF->getType()));
5133 return nullptr;
5134 }
5135 }
5136
5137 // Calling a null function pointer is undefined if a null address isn't
5138 // dereferenceable.
5139 if ((isa<ConstantPointerNull>(Callee) &&
5141 isa<UndefValue>(Callee)) {
5142 // If Call does not return void then replaceInstUsesWith poison.
5143 // This allows ValueHandlers and custom metadata to adjust itself.
5144 if (!Call.getType()->isVoidTy())
5146
5147 if (Call.isTerminator()) {
5148 // Can't remove an invoke or callbr because we cannot change the CFG.
5149 return nullptr;
5150 }
5151
5152 // This instruction is not reachable, just remove it.
5155 }
5156
5157 if (IntrinsicInst *II = findInitTrampoline(Callee))
5158 return transformCallThroughTrampoline(Call, *II);
5159
5160 // Combine calls involving pointer authentication intrinsics.
5161 if (Instruction *NewCall = foldPtrAuthIntrinsicCallee(Call))
5162 return NewCall;
5163
5164 // Combine calls to ptrauth constants.
5165 if (Instruction *NewCall = foldPtrAuthConstantCallee(Call))
5166 return NewCall;
5167
5168 if (isa<InlineAsm>(Callee) && !Call.doesNotThrow()) {
5169 InlineAsm *IA = cast<InlineAsm>(Callee);
5170 if (!IA->canThrow()) {
5171 // Normal inline asm calls cannot throw - mark them
5172 // 'nounwind'.
5174 Changed = true;
5175 }
5176 }
5177
5178 // Try to optimize the call if possible, we require DataLayout for most of
5179 // this. None of these calls are seen as possibly dead so go ahead and
5180 // delete the instruction now.
5181 if (CallInst *CI = dyn_cast<CallInst>(&Call)) {
5182 Instruction *I = tryOptimizeCall(CI);
5183 // If we changed something return the result, etc. Otherwise let
5184 // the fallthrough check.
5185 if (I) return eraseInstFromFunction(*I);
5186 }
5187
5188 if (!Call.use_empty() && !Call.isMustTailCall())
5189 if (Value *ReturnedArg = Call.getReturnedArgOperand()) {
5190 Type *CallTy = Call.getType();
5191 Type *RetArgTy = ReturnedArg->getType();
5192 if (RetArgTy->canLosslesslyBitCastTo(CallTy))
5193 return replaceInstUsesWith(
5194 Call, Builder.CreateBitOrPointerCast(ReturnedArg, CallTy));
5195 }
5196
5197 // Drop unnecessary callee_type metadata from calls that were converted
5198 // into direct calls.
5199 if (Call.getMetadata(LLVMContext::MD_callee_type) && !Call.isIndirectCall()) {
5200 Call.setMetadata(LLVMContext::MD_callee_type, nullptr);
5201 Changed = true;
5202 }
5203
5204 // Drop unnecessary kcfi operand bundles from calls that were converted
5205 // into direct calls.
5207 if (Bundle && !Call.isIndirectCall()) {
5208 DEBUG_WITH_TYPE(DEBUG_TYPE "-kcfi", {
5209 if (CalleeF) {
5210 ConstantInt *FunctionType = nullptr;
5211 ConstantInt *ExpectedType = cast<ConstantInt>(Bundle->Inputs[0]);
5212
5213 if (MDNode *MD = CalleeF->getMetadata(LLVMContext::MD_kcfi_type))
5214 FunctionType = mdconst::extract<ConstantInt>(MD->getOperand(0));
5215
5216 if (FunctionType &&
5217 FunctionType->getZExtValue() != ExpectedType->getZExtValue())
5218 dbgs() << Call.getModule()->getName()
5219 << ": warning: kcfi: " << Call.getCaller()->getName()
5220 << ": call to " << CalleeF->getName()
5221 << " using a mismatching function pointer type\n";
5222 }
5223 });
5224
5226 }
5227
5228 if (isRemovableAlloc(&Call, &TLI))
5229 return visitAllocSite(Call);
5230
5231 // Handle intrinsics which can be used in both call and invoke context.
5232 switch (Call.getIntrinsicID()) {
5233 case Intrinsic::experimental_gc_statepoint: {
5234 GCStatepointInst &GCSP = *cast<GCStatepointInst>(&Call);
5235 SmallPtrSet<Value *, 32> LiveGcValues;
5236 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5237 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5238
5239 // Remove the relocation if unused.
5240 if (GCR.use_empty()) {
5242 continue;
5243 }
5244
5245 Value *DerivedPtr = GCR.getDerivedPtr();
5246 Value *BasePtr = GCR.getBasePtr();
5247
5248 // Undef is undef, even after relocation.
5249 if (isa<UndefValue>(DerivedPtr) || isa<UndefValue>(BasePtr)) {
5252 continue;
5253 }
5254
5255 if (auto *PT = dyn_cast<PointerType>(GCR.getType())) {
5256 // The relocation of null will be null for most any collector.
5257 // TODO: provide a hook for this in GCStrategy. There might be some
5258 // weird collector this property does not hold for.
5259 if (isa<ConstantPointerNull>(DerivedPtr)) {
5260 // Use null-pointer of gc_relocate's type to replace it.
5263 continue;
5264 }
5265
5266 // isKnownNonNull -> nonnull attribute
5267 if (!GCR.hasRetAttr(Attribute::NonNull) &&
5268 isKnownNonZero(DerivedPtr,
5269 getSimplifyQuery().getWithInstruction(&Call))) {
5270 GCR.addRetAttr(Attribute::NonNull);
5271 // We discovered new fact, re-check users.
5272 Worklist.pushUsersToWorkList(GCR);
5273 }
5274 }
5275
5276 // If we have two copies of the same pointer in the statepoint argument
5277 // list, canonicalize to one. This may let us common gc.relocates.
5278 if (GCR.getBasePtr() == GCR.getDerivedPtr() &&
5279 GCR.getBasePtrIndex() != GCR.getDerivedPtrIndex()) {
5280 auto *OpIntTy = GCR.getOperand(2)->getType();
5281 GCR.setOperand(2, ConstantInt::get(OpIntTy, GCR.getBasePtrIndex()));
5282 }
5283
5284 // TODO: bitcast(relocate(p)) -> relocate(bitcast(p))
5285 // Canonicalize on the type from the uses to the defs
5286
5287 // TODO: relocate((gep p, C, C2, ...)) -> gep(relocate(p), C, C2, ...)
5288 LiveGcValues.insert(BasePtr);
5289 LiveGcValues.insert(DerivedPtr);
5290 }
5291 std::optional<OperandBundleUse> Bundle =
5293 unsigned NumOfGCLives = LiveGcValues.size();
5294 if (!Bundle || NumOfGCLives == Bundle->Inputs.size())
5295 break;
5296 // We can reduce the size of gc live bundle.
5297 DenseMap<Value *, unsigned> Val2Idx;
5298 std::vector<Value *> NewLiveGc;
5299 for (Value *V : Bundle->Inputs) {
5300 auto [It, Inserted] = Val2Idx.try_emplace(V);
5301 if (!Inserted)
5302 continue;
5303 if (LiveGcValues.count(V)) {
5304 It->second = NewLiveGc.size();
5305 NewLiveGc.push_back(V);
5306 } else
5307 It->second = NumOfGCLives;
5308 }
5309 // Update all gc.relocates
5310 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5311 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5312 Value *BasePtr = GCR.getBasePtr();
5313 assert(Val2Idx.count(BasePtr) && Val2Idx[BasePtr] != NumOfGCLives &&
5314 "Missed live gc for base pointer");
5315 auto *OpIntTy1 = GCR.getOperand(1)->getType();
5316 GCR.setOperand(1, ConstantInt::get(OpIntTy1, Val2Idx[BasePtr]));
5317 Value *DerivedPtr = GCR.getDerivedPtr();
5318 assert(Val2Idx.count(DerivedPtr) && Val2Idx[DerivedPtr] != NumOfGCLives &&
5319 "Missed live gc for derived pointer");
5320 auto *OpIntTy2 = GCR.getOperand(2)->getType();
5321 GCR.setOperand(2, ConstantInt::get(OpIntTy2, Val2Idx[DerivedPtr]));
5322 }
5323 // Create new statepoint instruction.
5324 OperandBundleDef NewBundle("gc-live", std::move(NewLiveGc));
5325 return CallBase::Create(&Call, NewBundle);
5326 }
5327 default: { break; }
5328 }
5329
5330 return Changed ? &Call : nullptr;
5331}
5332
5333/// If the callee is a constexpr cast of a function, attempt to move the cast to
5334/// the arguments of the call/invoke.
5335/// CallBrInst is not supported.
5336bool InstCombinerImpl::transformConstExprCastCall(CallBase &Call) {
5337 auto *Callee =
5339 if (!Callee)
5340 return false;
5341
5343 "CallBr's don't have a single point after a def to insert at");
5344
5345 // Don't perform the transform for declarations, which may not be fully
5346 // accurate. For example, void @foo() is commonly used as a placeholder for
5347 // unknown prototypes.
5348 if (Callee->isDeclaration())
5349 return false;
5350
5351 // If this is a call to a thunk function, don't remove the cast. Thunks are
5352 // used to transparently forward all incoming parameters and outgoing return
5353 // values, so it's important to leave the cast in place.
5354 if (Callee->hasFnAttribute("thunk"))
5355 return false;
5356
5357 // If this is a call to a naked function, the assembly might be
5358 // using an argument, or otherwise rely on the frame layout,
5359 // the function prototype will mismatch.
5360 if (Callee->hasFnAttribute(Attribute::Naked))
5361 return false;
5362
5363 // If this is a musttail call, the callee's prototype must match the caller's
5364 // prototype with the exception of pointee types. The code below doesn't
5365 // implement that, so we can't do this transform.
5366 // TODO: Do the transform if it only requires adding pointer casts.
5367 if (Call.isMustTailCall())
5368 return false;
5369
5371 const AttributeList &CallerPAL = Call.getAttributes();
5372
5373 // Okay, this is a cast from a function to a different type. Unless doing so
5374 // would cause a type conversion of one of our arguments, change this call to
5375 // be a direct call with arguments casted to the appropriate types.
5376 FunctionType *FT = Callee->getFunctionType();
5377 Type *OldRetTy = Caller->getType();
5378 Type *NewRetTy = FT->getReturnType();
5379
5380 // Check to see if we are changing the return type...
5381 if (OldRetTy != NewRetTy) {
5382
5383 if (NewRetTy->isStructTy())
5384 return false; // TODO: Handle multiple return values.
5385
5386 if (!CastInst::isBitOrNoopPointerCastable(NewRetTy, OldRetTy, DL)) {
5387 if (!Caller->use_empty())
5388 return false; // Cannot transform this return value.
5389 }
5390
5391 if (!CallerPAL.isEmpty() && !Caller->use_empty()) {
5392 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5393 if (RAttrs.overlaps(AttributeFuncs::typeIncompatible(
5394 NewRetTy, CallerPAL.getRetAttrs())))
5395 return false; // Attribute not compatible with transformed value.
5396 }
5397
5398 // If the callbase is an invoke instruction, and the return value is
5399 // used by a PHI node in a successor, we cannot change the return type of
5400 // the call because there is no place to put the cast instruction (without
5401 // breaking the critical edge). Bail out in this case.
5402 if (!Caller->use_empty()) {
5403 BasicBlock *PhisNotSupportedBlock = nullptr;
5404 if (auto *II = dyn_cast<InvokeInst>(Caller))
5405 PhisNotSupportedBlock = II->getNormalDest();
5406 if (PhisNotSupportedBlock)
5407 for (User *U : Caller->users())
5408 if (PHINode *PN = dyn_cast<PHINode>(U))
5409 if (PN->getParent() == PhisNotSupportedBlock)
5410 return false;
5411 }
5412 }
5413
5414 unsigned NumActualArgs = Call.arg_size();
5415 unsigned NumCommonArgs = std::min(FT->getNumParams(), NumActualArgs);
5416
5417 // Prevent us turning:
5418 // declare void @takes_i32_inalloca(i32* inalloca)
5419 // call void bitcast (void (i32*)* @takes_i32_inalloca to void (i32)*)(i32 0)
5420 //
5421 // into:
5422 // call void @takes_i32_inalloca(i32* null)
5423 //
5424 // Similarly, avoid folding away bitcasts of byval calls.
5425 if (Callee->getAttributes().hasAttrSomewhere(Attribute::InAlloca) ||
5426 Callee->getAttributes().hasAttrSomewhere(Attribute::Preallocated))
5427 return false;
5428
5429 auto AI = Call.arg_begin();
5430 for (unsigned i = 0, e = NumCommonArgs; i != e; ++i, ++AI) {
5431 Type *ParamTy = FT->getParamType(i);
5432 Type *ActTy = (*AI)->getType();
5433
5434 if (!CastInst::isBitOrNoopPointerCastable(ActTy, ParamTy, DL))
5435 return false; // Cannot transform this parameter value.
5436
5437 // Check if there are any incompatible attributes we cannot drop safely.
5438 if (AttrBuilder(FT->getContext(), CallerPAL.getParamAttrs(i))
5439 .overlaps(AttributeFuncs::typeIncompatible(
5440 ParamTy, CallerPAL.getParamAttrs(i),
5441 AttributeFuncs::ASK_UNSAFE_TO_DROP)))
5442 return false; // Attribute not compatible with transformed value.
5443
5444 if (Call.isInAllocaArgument(i) ||
5445 CallerPAL.hasParamAttr(i, Attribute::Preallocated))
5446 return false; // Cannot transform to and from inalloca/preallocated.
5447
5448 if (CallerPAL.hasParamAttr(i, Attribute::SwiftError))
5449 return false;
5450
5451 if (CallerPAL.hasParamAttr(i, Attribute::ByVal) !=
5452 Callee->getAttributes().hasParamAttr(i, Attribute::ByVal))
5453 return false; // Cannot transform to or from byval.
5454 }
5455
5456 if (FT->getNumParams() < NumActualArgs && FT->isVarArg() &&
5457 !CallerPAL.isEmpty()) {
5458 // In this case we have more arguments than the new function type, but we
5459 // won't be dropping them. Check that these extra arguments have attributes
5460 // that are compatible with being a vararg call argument.
5461 unsigned SRetIdx;
5462 if (CallerPAL.hasAttrSomewhere(Attribute::StructRet, &SRetIdx) &&
5463 SRetIdx - AttributeList::FirstArgIndex >= FT->getNumParams())
5464 return false;
5465 }
5466
5467 // Okay, we decided that this is a safe thing to do: go ahead and start
5468 // inserting cast instructions as necessary.
5469 SmallVector<Value *, 8> Args;
5471 Args.reserve(NumActualArgs);
5472 ArgAttrs.reserve(NumActualArgs);
5473
5474 // Get any return attributes.
5475 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5476
5477 // If the return value is not being used, the type may not be compatible
5478 // with the existing attributes. Wipe out any problematic attributes.
5479 RAttrs.remove(
5480 AttributeFuncs::typeIncompatible(NewRetTy, CallerPAL.getRetAttrs()));
5481
5482 LLVMContext &Ctx = Call.getContext();
5483 AI = Call.arg_begin();
5484 for (unsigned i = 0; i != NumCommonArgs; ++i, ++AI) {
5485 Type *ParamTy = FT->getParamType(i);
5486
5487 Value *NewArg = *AI;
5488 if ((*AI)->getType() != ParamTy)
5489 NewArg = Builder.CreateBitOrPointerCast(*AI, ParamTy);
5490 Args.push_back(NewArg);
5491
5492 // Add any parameter attributes except the ones incompatible with the new
5493 // type. Note that we made sure all incompatible ones are safe to drop.
5494 AttributeMask IncompatibleAttrs = AttributeFuncs::typeIncompatible(
5495 ParamTy, CallerPAL.getParamAttrs(i), AttributeFuncs::ASK_SAFE_TO_DROP);
5496 ArgAttrs.push_back(
5497 CallerPAL.getParamAttrs(i).removeAttributes(Ctx, IncompatibleAttrs));
5498 }
5499
5500 // If the function takes more arguments than the call was taking, add them
5501 // now.
5502 for (unsigned i = NumCommonArgs; i != FT->getNumParams(); ++i) {
5503 Args.push_back(Constant::getNullValue(FT->getParamType(i)));
5504 ArgAttrs.push_back(AttributeSet());
5505 }
5506
5507 // If we are removing arguments to the function, emit an obnoxious warning.
5508 if (FT->getNumParams() < NumActualArgs) {
5509 // TODO: if (!FT->isVarArg()) this call may be unreachable. PR14722
5510 if (FT->isVarArg()) {
5511 // Add all of the arguments in their promoted form to the arg list.
5512 for (unsigned i = FT->getNumParams(); i != NumActualArgs; ++i, ++AI) {
5513 Type *PTy = getPromotedType((*AI)->getType());
5514 Value *NewArg = *AI;
5515 if (PTy != (*AI)->getType()) {
5516 // Must promote to pass through va_arg area!
5517 Instruction::CastOps opcode =
5518 CastInst::getCastOpcode(*AI, false, PTy, false);
5519 NewArg = Builder.CreateCast(opcode, *AI, PTy);
5520 }
5521 Args.push_back(NewArg);
5522
5523 // Add any parameter attributes.
5524 ArgAttrs.push_back(CallerPAL.getParamAttrs(i));
5525 }
5526 }
5527 }
5528
5529 AttributeSet FnAttrs = CallerPAL.getFnAttrs();
5530
5531 if (NewRetTy->isVoidTy())
5532 Caller->setName(""); // Void type should not have a name.
5533
5534 assert((ArgAttrs.size() == FT->getNumParams() || FT->isVarArg()) &&
5535 "missing argument attributes");
5536 AttributeList NewCallerPAL = AttributeList::get(
5537 Ctx, FnAttrs, AttributeSet::get(Ctx, RAttrs), ArgAttrs);
5538
5540 Call.getOperandBundlesAsDefs(OpBundles);
5541
5542 CallBase *NewCall;
5543 if (InvokeInst *II = dyn_cast<InvokeInst>(Caller)) {
5544 NewCall = Builder.CreateInvoke(Callee, II->getNormalDest(),
5545 II->getUnwindDest(), Args, OpBundles);
5546 } else {
5547 NewCall = Builder.CreateCall(Callee, Args, OpBundles);
5548 cast<CallInst>(NewCall)->setTailCallKind(
5549 cast<CallInst>(Caller)->getTailCallKind());
5550 }
5551 NewCall->takeName(Caller);
5553 NewCall->setAttributes(NewCallerPAL);
5554
5555 // Preserve prof metadata if any.
5556 NewCall->copyMetadata(*Caller, {LLVMContext::MD_prof});
5557
5558 // Insert a cast of the return type as necessary.
5559 Instruction *NC = NewCall;
5560 Value *NV = NC;
5561 if (OldRetTy != NV->getType() && !Caller->use_empty()) {
5562 assert(!NV->getType()->isVoidTy());
5564 NC->setDebugLoc(Caller->getDebugLoc());
5565
5566 auto OptInsertPt = NewCall->getInsertionPointAfterDef();
5567 assert(OptInsertPt && "No place to insert cast");
5568 InsertNewInstBefore(NC, *OptInsertPt);
5569 Worklist.pushUsersToWorkList(*Caller);
5570 }
5571
5572 if (!Caller->use_empty())
5573 replaceInstUsesWith(*Caller, NV);
5574 else if (Caller->hasValueHandle()) {
5575 if (OldRetTy == NV->getType())
5577 else
5578 // We cannot call ValueIsRAUWd with a different type, and the
5579 // actual tracked value will disappear.
5581 }
5582
5583 eraseInstFromFunction(*Caller);
5584 return true;
5585}
5586
5587/// Turn a call to a function created by init_trampoline / adjust_trampoline
5588/// intrinsic pair into a direct call to the underlying function.
5590InstCombinerImpl::transformCallThroughTrampoline(CallBase &Call,
5591 IntrinsicInst &Tramp) {
5592 FunctionType *FTy = Call.getFunctionType();
5593 AttributeList Attrs = Call.getAttributes();
5594
5595 // If the call already has the 'nest' attribute somewhere then give up -
5596 // otherwise 'nest' would occur twice after splicing in the chain.
5597 if (Attrs.hasAttrSomewhere(Attribute::Nest))
5598 return nullptr;
5599
5601 FunctionType *NestFTy = NestF->getFunctionType();
5602
5603 AttributeList NestAttrs = NestF->getAttributes();
5604 if (!NestAttrs.isEmpty()) {
5605 unsigned NestArgNo = 0;
5606 Type *NestTy = nullptr;
5607 AttributeSet NestAttr;
5608
5609 // Look for a parameter marked with the 'nest' attribute.
5610 for (FunctionType::param_iterator I = NestFTy->param_begin(),
5611 E = NestFTy->param_end();
5612 I != E; ++NestArgNo, ++I) {
5613 AttributeSet AS = NestAttrs.getParamAttrs(NestArgNo);
5614 if (AS.hasAttribute(Attribute::Nest)) {
5615 // Record the parameter type and any other attributes.
5616 NestTy = *I;
5617 NestAttr = AS;
5618 break;
5619 }
5620 }
5621
5622 if (NestTy) {
5623 std::vector<Value*> NewArgs;
5624 std::vector<AttributeSet> NewArgAttrs;
5625 NewArgs.reserve(Call.arg_size() + 1);
5626 NewArgAttrs.reserve(Call.arg_size());
5627
5628 // Insert the nest argument into the call argument list, which may
5629 // mean appending it. Likewise for attributes.
5630
5631 {
5632 unsigned ArgNo = 0;
5633 auto I = Call.arg_begin(), E = Call.arg_end();
5634 do {
5635 if (ArgNo == NestArgNo) {
5636 // Add the chain argument and attributes.
5637 Value *NestVal = Tramp.getArgOperand(2);
5638 if (NestVal->getType() != NestTy)
5639 NestVal = Builder.CreateBitCast(NestVal, NestTy, "nest");
5640 NewArgs.push_back(NestVal);
5641 NewArgAttrs.push_back(NestAttr);
5642 }
5643
5644 if (I == E)
5645 break;
5646
5647 // Add the original argument and attributes.
5648 NewArgs.push_back(*I);
5649 NewArgAttrs.push_back(Attrs.getParamAttrs(ArgNo));
5650
5651 ++ArgNo;
5652 ++I;
5653 } while (true);
5654 }
5655
5656 // The trampoline may have been bitcast to a bogus type (FTy).
5657 // Handle this by synthesizing a new function type, equal to FTy
5658 // with the chain parameter inserted.
5659
5660 std::vector<Type*> NewTypes;
5661 NewTypes.reserve(FTy->getNumParams()+1);
5662
5663 // Insert the chain's type into the list of parameter types, which may
5664 // mean appending it.
5665 {
5666 unsigned ArgNo = 0;
5667 FunctionType::param_iterator I = FTy->param_begin(),
5668 E = FTy->param_end();
5669
5670 do {
5671 if (ArgNo == NestArgNo)
5672 // Add the chain's type.
5673 NewTypes.push_back(NestTy);
5674
5675 if (I == E)
5676 break;
5677
5678 // Add the original type.
5679 NewTypes.push_back(*I);
5680
5681 ++ArgNo;
5682 ++I;
5683 } while (true);
5684 }
5685
5686 // Replace the trampoline call with a direct call. Let the generic
5687 // code sort out any function type mismatches.
5688 FunctionType *NewFTy =
5689 FunctionType::get(FTy->getReturnType(), NewTypes, FTy->isVarArg());
5690 AttributeList NewPAL =
5691 AttributeList::get(FTy->getContext(), Attrs.getFnAttrs(),
5692 Attrs.getRetAttrs(), NewArgAttrs);
5693
5695 Call.getOperandBundlesAsDefs(OpBundles);
5696
5697 Instruction *NewCaller;
5698 if (InvokeInst *II = dyn_cast<InvokeInst>(&Call)) {
5699 NewCaller = InvokeInst::Create(NewFTy, NestF, II->getNormalDest(),
5700 II->getUnwindDest(), NewArgs, OpBundles);
5701 cast<InvokeInst>(NewCaller)->setCallingConv(II->getCallingConv());
5702 cast<InvokeInst>(NewCaller)->setAttributes(NewPAL);
5703 } else if (CallBrInst *CBI = dyn_cast<CallBrInst>(&Call)) {
5704 NewCaller =
5705 CallBrInst::Create(NewFTy, NestF, CBI->getDefaultDest(),
5706 CBI->getIndirectDests(), NewArgs, OpBundles);
5707 cast<CallBrInst>(NewCaller)->setCallingConv(CBI->getCallingConv());
5708 cast<CallBrInst>(NewCaller)->setAttributes(NewPAL);
5709 } else {
5710 NewCaller = CallInst::Create(NewFTy, NestF, NewArgs, OpBundles);
5711 cast<CallInst>(NewCaller)->setTailCallKind(
5712 cast<CallInst>(Call).getTailCallKind());
5713 cast<CallInst>(NewCaller)->setCallingConv(
5714 cast<CallInst>(Call).getCallingConv());
5715 cast<CallInst>(NewCaller)->setAttributes(NewPAL);
5716 }
5717 NewCaller->setDebugLoc(Call.getDebugLoc());
5718
5719 return NewCaller;
5720 }
5721 }
5722
5723 // Replace the trampoline call with a direct call. Since there is no 'nest'
5724 // parameter, there is no need to adjust the argument list. Let the generic
5725 // code sort out any function type mismatches.
5726 Call.setCalledFunction(FTy, NestF);
5727 return &Call;
5728}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
AMDGPU Register Bank Select
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
@ Scaled
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Atomic ordering constants.
This file contains the simple types necessary to represent the attributes associated with functions a...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
BitTracker BT
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static SDValue foldBitOrderCrossLogicOp(SDNode *N, SelectionDAG &DAG)
#define Check(C,...)
#define DEBUG_TYPE
Hexagon Common GEP
#define _
IRTranslator LLVM IR MI
static Type * getPromotedType(Type *Ty)
Return the specified type promoted as it would be to pass though a va_arg area.
static Instruction * createOverflowTuple(IntrinsicInst *II, Value *Result, Constant *Overflow)
Creates a result tuple for an overflow intrinsic II with a given Result and a constant Overflow value...
static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M, IRBuilderBase &B)
static IntrinsicInst * findInitTrampolineFromAlloca(Value *TrampMem)
static bool removeTriviallyEmptyRange(IntrinsicInst &EndI, InstCombinerImpl &IC, std::function< bool(const IntrinsicInst &)> IsStart)
static bool inputDenormalIsDAZ(const Function &F, const Type *Ty)
static Instruction * reassociateMinMaxWithConstantInOperand(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If this min/max has a matching min/max operand with a constant, try to push the constant operand into...
static bool isIdempotentBinaryIntrinsic(Intrinsic::ID IID)
Helper to match idempotent binary intrinsics, namely, intrinsics where f(f(x, y), y) == f(x,...
static bool signBitMustBeTheSame(Value *Op0, Value *Op1, const SimplifyQuery &SQ)
Return true if two values Op0 and Op1 are known to have the same sign.
static Value * optimizeModularFormat(CallInst *CI, IRBuilderBase &B)
static Instruction * moveAddAfterMinMax(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0.
static Instruction * simplifyInvariantGroupIntrinsic(IntrinsicInst &II, InstCombinerImpl &IC)
This function transforms launder.invariant.group like: launder(launder(x)) -> launder(x) (the result ...
static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E, unsigned NumOperands)
static std::optional< bool > getKnownSign(Value *Op, const SimplifyQuery &SQ)
static bool hasUndefSource(AnyMemTransferInst *MI)
Recognize a memcpy/memmove from a trivially otherwise unused alloca.
static Instruction * factorizeMinMaxTree(IntrinsicInst *II)
Reduce a sequence of min/max intrinsics with a common operand.
static Instruction * foldClampRangeOfTwo(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If we have a clamp pattern like max (min X, 42), 41 – where the output can only be one of two possibl...
static Value * simplifyReductionOperand(Value *Arg, bool CanReorderLanes)
static IntrinsicInst * findInitTrampolineFromBB(IntrinsicInst *AdjustTramp, Value *TrampMem)
static bool isAspectNeeded(StringRef Aspect, CallInst *CI, std::optional< unsigned > FirstArgIdx, const std::optional< Bitset< 256 > > &Specifiers)
static Value * foldIntrinsicUsingDistributiveLaws(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
static std::optional< bool > getKnownSignOrZero(Value *Op, const SimplifyQuery &SQ)
static Value * foldMinimumOverTrailingOrLeadingZeroCount(Value *I0, Value *I1, const DataLayout &DL, InstCombiner::BuilderTy &Builder)
Fold an unsigned minimum of trailing or leading zero bits counts: umin(cttz(CtOp1,...
static bool rightDistributesOverLeft(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "(X ROp Y) LOp Z" is always equal to "(X LOp Z) ROp (Y LOp Z)".
static Value * foldIdempotentBinaryIntrinsicRecurrence(InstCombinerImpl &IC, IntrinsicInst *II)
Attempt to simplify value-accumulating recurrences of kind: umax.acc = phi i8 [ umax,...
static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy)
static Instruction * foldCtpop(IntrinsicInst &II, InstCombinerImpl &IC)
static Instruction * simplifyNeonTbl(IntrinsicInst &II, InstCombiner &IC, bool IsExtension)
Convert tbl/tbx intrinsics to shufflevector if the mask is constant, and at most two source operands ...
static Instruction * foldCttzCtlz(IntrinsicInst &II, InstCombinerImpl &IC)
static IntrinsicInst * findInitTrampoline(Value *Callee)
static Value * foldCmpIntrinsicOfExtended(IntrinsicInst *II, InstCombiner::BuilderTy &Builder, const DataLayout &DL)
Fold an scmp/ucmp intrinsic whose operands are extended from a narrower type: scmp (sext X),...
static Bitset< 256 > parseFormatStringSpecifiers(StringRef FormatStr)
static FCmpInst::Predicate fpclassTestIsFCmp0(FPClassTest Mask, const Function &F, Type *Ty)
static bool leftDistributesOverRight(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "X LOp (Y ROp Z)" is always equal to "(X LOp Y) ROp (X LOp Z)".
static Value * reassociateMinMaxWithConstants(IntrinsicInst *II, IRBuilderBase &Builder, const SimplifyQuery &SQ)
If this min/max has a constant operand and an operand that is a matching min/max with a constant oper...
static Value * foldSinAndCosToSinCos(IntrinsicInst *II, IRBuilderBase &B, InstCombinerImpl &IC)
static bool mayFlushDenormalsToPositiveZero(const CallInst *CI)
Flushing a denormal to +0.0 breaks f(-x) = -f(x) for odd f.
static CallInst * canonicalizeConstantArg0ToArg1(CallInst &Call)
static Instruction * foldNeonShift(IntrinsicInst *II, InstCombinerImpl &IC)
This file provides internal interfaces used to implement the InstCombine.
This file provides the interface for the instcombine pass implementation.
static bool inputDenormalIsIEEE(DenormalMode Mode)
Return true if it's possible to assume IEEE treatment of input denormals in F for Val.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
This file contains the declarations for metadata subclasses.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
This file implements the SmallBitVector class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
This file contains some functions that are useful when dealing with strings.
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
The Input class is used to parse a yaml document into in-memory structs and vectors.
static LLVM_ABI bool semanticsHasInf(const fltSemantics &)
Definition APFloat.cpp:362
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static LLVM_ABI bool hasSignBitInMSB(const fltSemantics &)
Definition APFloat.cpp:375
bool isNegative() const
Definition APFloat.h:1583
void clearSign()
Definition APFloat.h:1402
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1192
bool isZero() const
Definition APFloat.h:1579
static APFloat getLargest(const fltSemantics &Sem, bool Negative=false)
Returns the largest finite number in the given semantics.
Definition APFloat.h:1242
static APFloat getSmallest(const fltSemantics &Sem, bool Negative=false)
Returns the smallest (by magnitude) finite number in the given semantics.
Definition APFloat.h:1252
bool isInfinity() const
Definition APFloat.h:1580
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:225
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1205
LLVM_ABI APInt usub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1986
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
Definition APInt.h:1186
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1695
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1115
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1966
LLVM_ABI APInt uadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1973
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:215
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1086
bool isShiftedMask() const
Return true if this APInt value contains a non-empty sequence of ones with the remainder zero.
Definition APInt.h:506
LLVM_ABI APInt uadd_sat(const APInt &RHS) const
Definition APInt.cpp:2074
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:330
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
std::optional< int64_t > trySExtValue() const
Get sign extended value if possible.
Definition APInt.h:1594
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1979
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
Definition APSInt.h:310
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
Definition APSInt.h:302
This class represents any memset intrinsic.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:410
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Return true if the attribute exists in this set.
static LLVM_ABI AttributeSet get(LLVMContext &C, const AttrBuilder &B)
static LLVM_ABI Attribute get(LLVMContext &Context, AttrKind Kind, uint64_t Val=0)
Return a uniquified Attribute object.
static LLVM_ABI Attribute getWithDereferenceableBytes(LLVMContext &Context, uint64_t Bytes)
static LLVM_ABI Attribute getWithDereferenceableOrNullBytes(LLVMContext &Context, uint64_t Bytes)
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator begin()
Instruction iterator methods.
Definition BasicBlock.h:446
InstListType::reverse_iterator reverse_iterator
Definition BasicBlock.h:172
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
LLVM_ABI bool isSigned() const
Whether the intrinsic is signed or unsigned.
LLVM_ABI Instruction::BinaryOps getBinaryOp() const
Returns the binary operation underlying the intrinsic.
static BinaryOperator * CreateFAddFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:271
static LLVM_ABI BinaryOperator * CreateNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Helper functions to construct and inspect unary operations (NEG and NOT) via binary operators SUB and...
static BinaryOperator * CreateNSW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:314
static LLVM_ABI BinaryOperator * CreateNot(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
static LLVM_ABI BinaryOperator * Create(BinaryOps Op, Value *S1, Value *S2, const Twine &Name=Twine(), InsertPosition InsertBefore=nullptr)
Construct a binary instruction, given the opcode and the two operands.
static BinaryOperator * CreateNUW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:329
static BinaryOperator * CreateFMulFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:279
static BinaryOperator * CreateFDivFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:283
static BinaryOperator * CreateFSubFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:275
static LLVM_ABI BinaryOperator * CreateNSWNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
This is a constexpr reimplementation of a subset of std::bitset.
Definition Bitset.h:30
constexpr bool any() const
Definition Bitset.h:113
constexpr Bitset & set()
Definition Bitset.h:81
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
void setCallingConv(CallingConv::ID CC)
void setDoesNotThrow()
MaybeAlign getRetAlign() const
Extract the alignment of the return value.
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
OperandBundleUse getOperandBundleAt(unsigned Index) const
Return the operand bundle at a specific index.
std::optional< OperandBundleUse > getOperandBundle(StringRef Name) const
Return an operand bundle by name, if present.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
bool isInAllocaArgument(unsigned ArgNo) const
Determine whether this argument is passed in an alloca.
bool hasFnAttr(Attribute::AttrKind Kind) const
Determine whether this call has the given attribute.
bool hasRetAttr(Attribute::AttrKind Kind) const
Determine whether the return value has the given attribute.
unsigned getNumOperandBundles() const
Return the number of operand bundles associated with this User.
uint64_t getParamDereferenceableBytes(unsigned i) const
Extract the number of dereferenceable bytes for a call or parameter (0=unknown).
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
LLVM_ABI bool isIndirectCall() const
Return true if the callsite is an indirect call.
static LLVM_ABI CallBase * removeOperandBundleAt(CallBase *CB, size_t Offset, InsertPosition InsertPtr=nullptr)
void setNotConvergent()
Value * getCalledOperand() const
void setAttributes(AttributeList A)
Set the attributes for this call.
Attribute getFnAttr(StringRef Kind) const
Get the attribute of a given kind for the function.
bool doesNotThrow() const
Determine if the call cannot unwind.
void addRetAttr(Attribute::AttrKind Kind)
Adds the attribute to the return value.
Value * getArgOperand(unsigned i) const
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
bool isConvergent() const
Determine if the invoke is convergent.
FunctionType * getFunctionType() const
LLVM_ABI Intrinsic::ID getIntrinsicID() const
Returns the intrinsic ID of the intrinsic called or Intrinsic::not_intrinsic if the called function i...
Value * getReturnedArgOperand() const
If one of the arguments has the 'returned' attribute, returns its operand value.
static LLVM_ABI CallBase * Create(CallBase *CB, ArrayRef< OperandBundleDef > Bundles, InsertPosition InsertPt=nullptr)
Create a clone of CB with a different set of operand bundles and insert it before InsertPt.
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
void setCalledOperand(Value *V)
static LLVM_ABI CallBase * removeOperandBundle(CallBase *CB, uint32_t ID, InsertPosition InsertPt=nullptr)
Create a clone of CB with operand bundle ID removed.
unsigned arg_size() const
AttributeList getAttributes() const
Return the attributes for this call.
void setCalledFunction(Function *Fn)
Sets the function called, including updating the function type.
LLVM_ABI Function * getCaller()
Helper to get the caller (the parent function).
CallBr instruction, tracking function calls that may not return control but instead transfer it to a ...
static CallBrInst * Create(FunctionType *Ty, Value *Func, BasicBlock *DefaultDest, ArrayRef< BasicBlock * > IndirectDests, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This class represents a function call, abstracting a target machine's calling convention.
bool isNoTailCall() const
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
bool isMustTailCall() const
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI CastInst * CreateIntegerCast(Value *S, Type *Ty, bool isSigned, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a ZExt, BitCast, or Trunc for int -> int casts.
static LLVM_ABI bool isBitOrNoopPointerCastable(Type *SrcTy, Type *DestTy, const DataLayout &DL)
Check whether a bitcast, inttoptr, or ptrtoint cast between these types is valid and a no-op.
static LLVM_ABI CastInst * CreateBitOrPointerCast(Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a BitCast, a PtrToInt, or an IntToPTr cast instruction.
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
Predicate getNonStrictPredicate() const
For example, SGT -> SGE, SLT -> SLE, ULT -> ULE, UGT -> UGE.
Definition InstrTypes.h:934
Predicate getUnorderedPredicate() const
Definition InstrTypes.h:874
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static LLVM_ABI Constant * getNeg(Constant *C, bool HasNSW=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getLimitedValue(uint64_t Limit=~0ULL) const
getLimitedValue - If the value is smaller than the specified limit, return it, otherwise return the l...
Definition Constants.h:269
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI ConstantPtrAuth * get(Constant *Ptr, ConstantInt *Key, ConstantInt *Disc, Constant *AddrDisc, Constant *DeactivationSymbol)
Return a pointer signed with the specified parameters.
This class represents a range of values.
LLVM_ABI ConstantRange zextOrTrunc(uint32_t BitWidth) const
Make this range have the bit width given by BitWidth.
LLVM_ABI bool isFullSet() const
Return true if this set contains all of the elements possible for this data-type.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
uint32_t getBitWidth() const
Get the bit width of this ConstantRange.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getIntegerValue(Type *Ty, const APInt &V)
Return the value for an integer or pointer constant, or a vector thereof, with the given scalar value...
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Record of a variable value-assignment, aka a non instruction representation of the dbg....
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Definition DenseMap.h:773
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
Definition DenseMap.h:778
unsigned size() const
Definition DenseMap.h:733
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:872
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
static FMFSource intersect(Value *A, Value *B)
Intersect the FMF from two instructions.
Definition IRBuilder.h:107
This class represents an extension of floating point types.
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool allowReassoc() const
Flag queries.
Definition FMF.h:64
An instruction for ordering other memory operations.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this fence instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this fence instruction.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
Type::subtype_iterator param_iterator
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
bool isConvergent() const
Determine if the call is convergent.
Definition Function.h:593
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
AttributeList getAttributes() const
Return the attribute list for this Function.
Definition Function.h:329
bool doesNotThrow() const
Determine if the function cannot unwind.
Definition Function.h:577
bool isIntrinsic() const
isIntrinsic - Returns true if the function's name starts with "llvm.".
Definition Function.h:252
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
Definition Function.cpp:806
LLVM_ABI Value * getBasePtr() const
unsigned getBasePtrIndex() const
The index into the associate statepoint's argument list which contains the base pointer of the pointe...
LLVM_ABI Value * getDerivedPtr() const
unsigned getDerivedPtrIndex() const
The index into the associate statepoint's argument list which contains the pointer whose relocation t...
std::vector< const GCRelocateInst * > getGCRelocates() const
Get list of all gc reloactes linked to this statepoint May contain several relocations for the same b...
Definition Statepoint.h:206
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this GlobalObject.
LLVM_ABI bool isDeclaration() const
Return true if the primary definition of this global value is outside of the current translation unit...
Definition Globals.cpp:408
PointerType * getType() const
Global values are always pointers.
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
Value * CreateAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNull=false)
Definition IRBuilder.h:2240
LLVM_ABI Value * CreateLaunderInvariantGroup(Value *Ptr)
Create a launder.invariant.group intrinsic call.
ConstantInt * getTrue()
Get the constant value for i1 true.
Definition IRBuilder.h:436
LLVM_ABI Value * CreateBinaryIntrinsic(Intrinsic::ID ID, Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 2 operands which is mangled on the first type.
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1426
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2113
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
Definition IRBuilder.h:2683
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
ConstantInt * getFalse()
Get the constant value for i1 false.
Definition IRBuilder.h:441
Value * CreateICmp(CmpInst::Predicate P, Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2484
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
static InsertValueInst * Create(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Instruction * foldOpIntoPhi(Instruction &I, PHINode *PN, bool AllowMultipleUses=false)
Given a binary operator, cast instruction, or select which has a PHI node as operand #0,...
Value * SimplifyDemandedVectorElts(Value *V, APInt DemandedElts, APInt &PoisonElts, unsigned Depth=0, bool AllowMultipleUsers=false) override
The specified value produces a vector with any number of elements.
bool SimplifyDemandedBits(Instruction *I, unsigned Op, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0) override
This form of SimplifyDemandedBits simplifies the specified instruction operand if possible,...
Instruction * FoldOpIntoSelect(Instruction &Op, SelectInst *SI, bool FoldWithMultiUse=false, bool SimplifyBothArms=false)
Given an instruction with a select as one operand and a constant as the other operand,...
Instruction * SimplifyAnyMemSet(AnyMemSetInst *MI)
Instruction * foldItoFPtoI(FPToIntTy &FI)
fpto{s/u}i.sat --> X or zext(X) or sext(X) or trunc(X) This is safe if the intermediate type has enou...
Instruction * visitFree(CallInst &FI, Value *FreedOp)
Instruction * visitCallBrInst(CallBrInst &CBI)
OverflowResult computeOverflow(Instruction::BinaryOps BinaryOp, bool IsSigned, Value *LHS, Value *RHS, Instruction *CtxI) const
Instruction * eraseInstFromFunction(Instruction &I) override
Combiner aware instruction erasure.
Value * foldReversedIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are reverses, try to pull the reverse after the intrinsic.
const InstCombineCLOptions & CLOpts
Value * tryGetLog2(Value *Op, bool AssumeNonZero)
Instruction * visitFenceInst(FenceInst &FI)
Instruction * foldShuffledIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are unary shuffles with the same mask, try to shuffle after the int...
Instruction * visitInvokeInst(InvokeInst &II)
bool SimplifyDemandedInstructionBits(Instruction &Inst)
Tries to simplify operands to an integer instruction based on its demanded bits.
void CreateNonTerminatorUnreachable(Instruction *InsertAt)
Create and insert the idiom we use to indicate a block is unreachable without having to rewrite the C...
Instruction * visitVAEndInst(VAEndInst &I)
Instruction * matchBSwapOrBitReverse(Instruction &I, bool MatchBSwaps, bool MatchBitReversals)
Given an initial instruction, check to see if it is the root of a bswap/bitreverse idiom.
Constant * unshuffleConstant(ArrayRef< int > ShMask, Constant *C, VectorType *NewCTy)
Find a constant NewC that has property: shuffle(NewC, poison, ShMask) = C for lanes that select NewC.
Instruction * visitAllocSite(Instruction &FI)
Instruction * SimplifyAnyMemTransfer(AnyMemTransferInst *MI)
Instruction * visitCallInst(CallInst &CI)
CallInst simplification.
The core instruction combiner logic.
SimplifyQuery SQ
const DataLayout & getDataLayout() const
bool isFreeToInvert(Value *V, bool WillInvertAllUses, bool &DoesConsume)
Return true if the specified value is free to invert (apply ~ to).
DominatorTree & getDominatorTree() const
BlockFrequencyInfo * BFI
unsigned ComputeMaxSignificantBits(const Value *Op, const Instruction *CtxI=nullptr, unsigned Depth=0) const
bool isKnownToBeAPowerOfTwo(const Value *V, bool OrZero=false, const Instruction *CtxI=nullptr, unsigned Depth=0)
TargetLibraryInfo & TLI
Instruction * InsertNewInstBefore(Instruction *New, BasicBlock::iterator Old)
Inserts an instruction New before instruction Old.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
void replaceUse(Use &U, Value *NewValue)
Replace use and add the previously used value to the worklist.
InstructionWorklist & Worklist
A worklist of the instructions that need to be simplified.
const DataLayout & DL
DomConditionCache DC
bool MaskedValueIsZero(const Value *V, const APInt &Mask, const Instruction *CtxI=nullptr, unsigned Depth=0) const
IRBuilder< TargetFolder, IRBuilderInstCombineInserter > BuilderTy
An IRBuilder that automatically inserts new instructions into the worklist.
LLVM_ABI std::optional< Instruction * > targetInstCombineIntrinsic(IntrinsicInst &II)
AssumptionCache & AC
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
DominatorTree & DT
ProfileSummaryInfo * PSI
OptimizationRemarkEmitter & ORE
void computeKnownBits(const Value *V, KnownBits &Known, const Instruction *CtxI, unsigned Depth=0) const
Value * getFreelyInverted(Value *V, bool WillInvertAllUses, BuilderTy *Builder, bool &DoesConsume)
const SimplifyQuery & getSimplifyQuery() const
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void setHasNoUnsignedWrap(bool b=true)
Set or clear the nuw flag on this instruction, which must be an operator which supports this flag.
LLVM_ABI bool mayWriteToMemory() const LLVM_READONLY
Return true if this instruction may modify memory.
LLVM_ABI void copyIRFlags(const Value *V, bool IncludeWrapFlags=true)
Convenience method to copy supported exact, fast-math, and (optionally) wrapping flags from V to this...
LLVM_ABI void setHasNoSignedWrap(bool b=true)
Set or clear the nsw flag on this instruction, which must be an operator which supports this flag.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI void setFastMathFlags(FastMathFlags FMF)
Convenience function for setting multiple fast-math flags on this instruction, which must be an opera...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
bool isTerminator() const
iterator_range< user_iterator > users()
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
LLVM_ABI std::optional< InstListType::iterator > getInsertionPointAfterDef()
Get the first insertion point at which the result of this instruction is defined.
LLVM_ABI bool isIdenticalTo(const Instruction *I) const LLVM_READONLY
Return true if the specified instruction is exactly identical to the current one.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
Invoke instruction.
static InvokeInst * Create(FunctionType *Ty, Value *Func, BasicBlock *IfNormal, BasicBlock *IfException, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
Metadata node.
Definition Metadata.h:1081
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1579
static LLVM_ABI MDNode * getMostGenericFPMath(MDNode *A, MDNode *B)
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
Definition Metadata.cpp:597
static LLVM_ABI MetadataAsValue * get(LLVMContext &Context, Metadata *MD)
Definition Metadata.cpp:107
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
ICmpInst::Predicate getPredicate() const
Returns the comparison predicate underlying the intrinsic.
bool isSigned() const
Whether the intrinsic is signed or unsigned.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
StringRef getName() const
Get a short "name" for the module.
Definition Module.h:316
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
Utility class for integer operators which may exhibit overflow - Add, Sub, Mul, and Shl.
Definition Operator.h:78
bool hasNoSignedWrap() const
Test whether this operation is known to never undergo signed overflow, aka the nsw property.
Definition Operator.h:113
bool hasNoUnsignedWrap() const
Test whether this operation is known to never undergo unsigned overflow, aka the nuw property.
Definition Operator.h:107
bool isCommutative() const
Return true if the instruction is commutative.
Definition Operator.h:130
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Represents a saturating add/sub intrinsic.
This class represents the LLVM 'select' instruction.
static SelectInst * Create(Value *C, Value *S1, Value *S2, const Twine &NameStr="", InsertPosition InsertBefore=nullptr, const Instruction *MDFrom=nullptr)
This instruction constructs a fixed permutation of two input vectors.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool all() const
Returns true if all bits are set.
size_type size() const
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setVolatile(bool V)
Specify whether this is a volatile store or not.
void setAlignment(Align Align)
void setOrdering(AtomicOrdering Ordering)
Sets the ordering constraint of this store instruction.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static constexpr size_t npos
Definition StringRef.h:58
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
LLVM_ABI size_t find_first_not_of(char C, size_t From=0) const
Find the first character in the string that is not C or npos if not found.
Class to represent struct types.
static LLVM_ABI bool isCallingConvCCompatible(CallBase *CI)
Returns true if call site / callee has cdecl-compatible calling conventions.
Provides information about what library functions are available for the current target.
This class represents a truncation of integer types.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getIntegerBitWidth() const
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:258
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
LLVM_ABI bool canLosslesslyBitCastTo(Type *Ty) const
Return true if this type could be converted with a lossless BitCast to type 'Ty'.
Definition Type.cpp:143
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:271
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:96
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
static UnaryOperator * CreateWithCopiedFlags(UnaryOps Opc, Value *V, Instruction *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:148
static UnaryOperator * CreateFNegFMF(Value *Op, Instruction *FMFSource, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:156
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
Definition Use.cpp:35
void setOperand(unsigned i, Value *Val)
Definition User.h:212
Value * getOperand(unsigned i) const
Definition User.h:207
This represents the llvm.va_end intrinsic.
static LLVM_ABI void ValueIsDeleted(Value *V)
Definition Value.cpp:1241
static LLVM_ABI void ValueIsRAUWd(Value *Old, Value *New)
Definition Value.cpp:1294
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
static constexpr uint64_t MaximumAlignment
Definition Value.h:801
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
iterator_range< user_iterator > users()
Definition Value.h:428
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:712
bool use_empty() const
Definition Value.h:348
static constexpr unsigned MaxAlignmentExponent
The maximum alignment for instructions.
Definition Value.h:800
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
static constexpr bool isKnownGT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:223
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
Definition ilist_node.h:348
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr char Attrs[]
Key for Kernel::Metadata::mAttrs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
auto m_PosZeroFP()
Matches a floating-point positive zero.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_BSwap(const Opnd0 &Op0)
PtrAdd_match< PointerOpTy, OffsetOpTy > m_PtrAdd(const PointerOpTy &PointerOp, const OffsetOpTy &OffsetOp)
Matches GEP with i8 source element type.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
auto m_BitReverse(const Opnd0 &Op0)
auto m_PtrToIntOrAddr(const OpTy &Op)
Matches PtrToInt or PtrToAddr.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
BinaryOp_match< LHS, RHS, Instruction::Xor > m_Xor(const LHS &L, const RHS &R)
ap_match< APInt > m_APIntAllowPoison(const APInt *&Res)
Match APInt while allowing poison in splat vector constants.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWSub(const LHS &L, const RHS &R)
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
auto m_UMin(const Opnd0 &Op0, const Opnd1 &Op1)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
ap_match< APFloat > m_APFloat(const APFloat *&Res)
Match a ConstantFP or splatted ConstantVector, binding the specified pointer to the contained APFloat...
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
OverflowingBinaryOp_match< cst_pred_ty< is_zero_int >, ValTy, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWNeg(const ValTy &V)
Matches a 'Neg' as 'sub nsw 0, V'.
auto m_SMax(const Opnd0 &Op0, const Opnd1 &Op1)
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
cstfp_pred_ty< is_neg_zero_fp > m_NegZeroFP()
Match a floating-point negative zero.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_UMax(const Opnd0 &Op0, const Opnd1 &Op1)
specific_fpval m_SpecificFP(double V)
Match a specific floating point value or vector with all elements equal to the value.
auto m_CopySign(const Opnd0 &Op0, const Opnd1 &Op1)
ExtractValue_match< Ind, Val_t > m_ExtractValue(const Val_t &V)
Match a single index ExtractValue instruction.
BinOpPred_match< LHS, RHS, is_logical_shift_op > m_LogicalShift(const LHS &L, const RHS &R)
Matches logical shift operations.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_Constant()
Match an arbitrary Constant and ignore it.
match_combine_or< match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > >, OpTy > m_ZExtOrSExtOrSelf(const OpTy &Op)
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
cst_pred_ty< is_strictlypositive > m_StrictlyPositive()
Match an integer or vector of strictly positive values.
ThreeOps_match< decltype(m_Value()), LHS, RHS, Instruction::Select, true > m_c_Select(const LHS &L, const RHS &R)
Match Select(C, LHS, RHS) or Select(C, RHS, LHS)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Shl, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWShl(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Mul, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWMul(const LHS &L, const RHS &R)
auto m_FShl(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
cst_pred_ty< is_negated_power2 > m_NegatedPower2()
Match a integer or vector negated power-of-2.
match_immconstant_ty m_ImmConstant()
Match an arbitrary immediate Constant and ignore it.
cst_pred_ty< custom_checkfn< APInt > > m_CheckedInt(function_ref< bool(const APInt &)> CheckFn)
Match an integer or vector where CheckFn(ele) for each element is true.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_c_MaxOrMin(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWSub(const LHS &L, const RHS &R)
auto m_SMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_FAbs(const Opnd0 &Op0)
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoSignedWrap >, DisjointOr_match< LHS, RHS > > m_NSWAddLike(const LHS &L, const RHS &R)
Match either "add nsw" or "or disjoint".
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
Exact_match< T > m_Exact(const T &SubPattern)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
auto m_UnOp()
Match an arbitrary unary operation and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_MaxOrMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
BinaryOp_match< LHS, RHS, Instruction::SRem > m_SRem(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
auto m_VecReverse(const Opnd0 &Op0)
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoUnsignedWrap >, DisjointOr_match< LHS, RHS > > m_NUWAddLike(const LHS &L, const RHS &R)
Match either "add nuw" or "or disjoint".
BinOpPred_match< LHS, RHS, is_bitwiselogic_op > m_BitwiseLogic(const LHS &L, const RHS &R)
Matches bitwise logic operations.
ElementWiseBitCast_match< OpTy > m_ElementWiseBitCast(const OpTy &Op)
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
auto m_FShr(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
@ SingleThread
Synchronized with respect to signal handlers executing in the same thread.
Definition LLVMContext.h:55
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
Definition DebugInfo.h:212
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:679
constexpr double e
DiagnosticInfoOptimizationBase::Argument NV
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Intrinsic::ID getInverseMinMaxIntrinsic(Intrinsic::ID MinMaxID)
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
@ NeverOverflows
Never overflows.
@ AlwaysOverflowsHigh
Always overflows in the direction of signed/unsigned max value.
@ AlwaysOverflowsLow
Always overflows in the direction of signed/unsigned min value.
@ MayOverflow
May or may not overflow.
LLVM_ABI KnownFPClass computeKnownFPClass(const Value *V, const APInt &DemandedElts, FPClassTest InterestedClasses, const SimplifyQuery &SQ, unsigned Depth=0)
Determine which floating-point classes are valid for V, and return them in KnownFPClass bit sets.
LLVM_ABI cl::opt< bool > ProfcheckDisableMetadataFixes
Definition LoopInfo.cpp:60
LLVM_ABI Value * simplifyFMulInst(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for an FMul, fold the result or return null.
LLVM_ABI APInt possiblyDemandedEltsInMask(Value *Mask)
Given a mask vector of the form <Y x i1>, return an APInt (of bitwidth Y) for each lane which may be ...
BundleAttr getBundleAttrFromOBU(OperandBundleUse OBU)
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
LLVM_ABI void setExplicitlyUnknownBranchWeightsIfProfiled(Instruction &I, StringRef PassName, const Function *F=nullptr)
Like setExplicitlyUnknownBranchWeights(...), but only sets unknown branch weights in the new instruct...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI bool isRemovableAlloc(const CallBase *V, const TargetLibraryInfo *TLI)
Return true if this is a call to an allocation function that does not have side effects that we are r...
LLVM_ABI bool getConstantStringInfo(const Value *V, StringRef &Str, bool TrimAtNul=true)
This function computes the length of a null-terminated C string pointed to by V.
constexpr int64_t minIntN(int64_t N)
Gets the minimum value for a N-bit signed integer.
Definition MathExtras.h:224
LLVM_ABI Value * lowerObjectSizeCall(IntrinsicInst *ObjectSize, const DataLayout &DL, const TargetLibraryInfo *TLI, bool MustSucceed)
Try to turn a call to @llvm.objectsize into an integer value of the given Type.
LLVM_ABI AssumeSeparateStorageInfo getAssumeSeparateStorageInfo(OperandBundleUse)
LLVM_ABI Value * getAllocAlignment(const CallBase *V, const TargetLibraryInfo *TLI)
Gets the alignment argument for an aligned_alloc-like function, using either built-in knowledge based...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_READONLY APFloat maximum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximum semantics.
Definition APFloat.h:1801
LLVM_ABI Value * simplifyCall(CallBase *Call, Value *Callee, ArrayRef< Value * > Args, const SimplifyQuery &Q)
Given a callsite, callee, and arguments, fold the result or return null.
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
constexpr T MinAlign(U A, V B)
A and B are either alignments or offsets.
Definition MathExtras.h:352
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Function *CtxF=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
LLVM_ABI bool isValidAssumeForContext(const Instruction *I, const Instruction *CtxI, const DominatorTree *DT=nullptr, bool AllowEphemerals=false)
Return true if it is valid to use the assumptions provided by an assume intrinsic,...
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
Definition APFloat.h:1756
SelectPatternFlavor
Specific patterns of select instructions we can match.
@ SPF_ABS
Floating point maxnum.
@ SPF_NABS
Absolute value.
LLVM_ABI Constant * getLosslessUnsignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool isModSet(const ModRefInfo MRI)
Definition ModRef.h:49
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
LLVM_READONLY APFloat minimumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimumNumber semantics.
Definition APFloat.h:1787
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
Definition APFloat.h:1701
LLVM_ABI SelectPatternResult matchSelectPattern(Value *V, Value *&LHS, Value *&RHS, Instruction::CastOps *CastOp=nullptr, unsigned Depth=0)
Pattern match integer [SU]MIN, [SU]MAX and ABS idioms, returning the kind and providing the out param...
LLVM_ABI bool matchSimpleBinaryIntrinsicRecurrence(const IntrinsicInst *I, PHINode *&P, Value *&Init, Value *&OtherOp)
Attempt to match a simple value-accumulating recurrence of the form: llvm.intrinsic....
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
auto find_if_not(R &&Range, UnaryPredicate P)
Definition STLExtras.h:1793
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
bool isAtLeastOrStrongerThan(AtomicOrdering AO, AtomicOrdering Other)
LLVM_ABI Constant * getLosslessSignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_ABI ConstantRange getVScaleRange(const Function *F, unsigned BitWidth)
Determine the possible constant range of vscale with the given bit width, based on the vscale_range f...
iterator_range< SplittingIterator > split(StringRef Str, StringRef Separator)
Split the specified string over a separator and return a range-compatible iterable over its partition...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI bool isNotCrossLaneOperation(const Instruction *I)
Return true if the instruction doesn't potentially cross vector lanes.
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
LLVM_ABI bool isKnownNonZero(const Value *V, const SimplifyQuery &Q, unsigned Depth=0)
Return true if the given value is known to be non-zero when defined.
constexpr int PoisonMaskElem
@ Mod
The access may modify the value stored in memory.
Definition ModRef.h:34
LLVM_ABI Value * simplifyFMAFMul(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for the multiplication of a FMA, fold the result or return null.
@ Other
Any other memory.
Definition ModRef.h:68
LLVM_ABI Value * simplifyConstrainedFPCall(CallBase *Call, const SimplifyQuery &Q)
Given a constrained FP intrinsic call, tries to compute its simplified version.
LLVM_READONLY APFloat minnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 minNum semantics.
Definition APFloat.h:1737
OperandBundleDefT< Value * > OperandBundleDef
Definition AutoUpgrade.h:34
LLVM_ABI AssumeNonNullInfo getAssumeNonNullInfo(OperandBundleUse)
@ Add
Sum of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
bool isSafeToSpeculativelyExecuteWithVariableReplaced(const Instruction *I, bool IgnoreUBImplyingAttrs=true)
Don't use information from its non-constant operands.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI Value * getFreedOperand(const CallBase *CB, const TargetLibraryInfo *TLI)
If this if a call to a free function, return the freed operand.
constexpr int64_t maxIntN(int64_t N)
Gets the maximum value for a N-bit signed integer.
Definition MathExtras.h:233
constexpr unsigned BitWidth
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
LLVM_ABI std::optional< APInt > getAllocSize(const CallBase *CB, const TargetLibraryInfo *TLI, function_ref< const Value *(const Value *)> Mapper=[](const Value *V) { return V;})
Return the size of the requested allocation.
Align getKnownAlignment(Value *V, const DataLayout &DL, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr)
Try to infer an alignment for the specified pointer.
Definition Local.h:240
LLVM_ABI AssumeAlignInfo getAssumeAlignInfo(OperandBundleUse)
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI bool maskContainsAllOneOrUndef(Value *Mask)
Given a mask vector of i1, Return true if any of the elements of this predicate mask are known to be ...
LLVM_ABI std::optional< bool > isImpliedByDomCondition(const Value *Cond, const Instruction *ContextI, const DataLayout &DL)
Return the boolean condition value in the context of the given instruction if it is known based on do...
LLVM_ABI bool isDereferenceablePointer(const Value *V, Type *Ty, const SimplifyQuery &Q, bool IgnoreFree=false)
Equivalent to isDereferenceableAndAlignedPointer with an alignment of 1.
Definition Loads.cpp:264
LLVM_READONLY APFloat minimum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimum semantics.
Definition APFloat.h:1774
LLVM_ABI bool isKnownNegation(const Value *X, const Value *Y, bool NeedNSW=false, bool AllowPoison=true)
Return true if the two given values are negation.
LLVM_READONLY APFloat maximumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximumNumber semantics.
Definition APFloat.h:1814
LLVM_ABI AssumeDereferenceableInfo getAssumeDereferenceableInfo(OperandBundleUse)
LLVM_ABI bool isKnownNonNegative(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Returns true if the give value is known to be non-negative.
LLVM_ABI AssumeNoUndefInfo getAssumeNoUndefInfo(OperandBundleUse)
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
LLVM_ABI std::optional< bool > computeKnownFPSignBit(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Return false if we can prove that the specified FP value's sign bit is 0.
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define NC
Definition regutils.h:42
A collection of metadata nodes that might be associated with a memory access used by the alias-analys...
Definition Metadata.h:774
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Represent subnormal handling kind for floating point instruction inputs and outputs.
@ IEEE
IEEE-754 denormal numbers preserved.
Matching combinators.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Align valueOrOne() const
For convenience, returns a valid alignment or 1 if undefined.
Definition Alignment.h:130
uint32_t getTagID() const
Return the tag of this operand bundle as an integer.
ArrayRef< Use > Inputs
SelectPatternFlavor Flavor
const DataLayout & DL
SimplifyQuery getWithInstruction(const Instruction *I) const
const Instruction * CtxI