LLVM 24.0.0git
InstructionCombining.cpp
Go to the documentation of this file.
1//===- InstructionCombining.cpp - Combine multiple instructions -----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// InstructionCombining - Combine instructions to form fewer, simple
10// instructions. This pass does not modify the CFG. This pass is where
11// algebraic simplification happens.
12//
13// This pass combines things like:
14// %Y = add i32 %X, 1
15// %Z = add i32 %Y, 1
16// into:
17// %Z = add i32 %X, 2
18//
19// This is a simple worklist driven algorithm.
20//
21// This pass guarantees that the following canonicalizations are performed on
22// the program:
23// 1. If a binary operator has a constant operand, it is moved to the RHS
24// 2. Bitwise operators with constant operands are always grouped so that
25// shifts are performed first, then or's, then and's, then xor's.
26// 3. Compare instructions are converted from <,>,<=,>= to ==,!= if possible
27// 4. All cmp instructions on boolean values are replaced with logical ops
28// 5. add X, X is represented as (X*2) => (X << 1)
29// 6. Multiplies with a power-of-two constant argument are transformed into
30// shifts.
31// ... etc.
32//
33//===----------------------------------------------------------------------===//
34
35#include "InstCombineInternal.h"
36#include "llvm/ADT/APFloat.h"
37#include "llvm/ADT/APInt.h"
38#include "llvm/ADT/ArrayRef.h"
39#include "llvm/ADT/DenseMap.h"
43#include "llvm/ADT/Statistic.h"
48#include "llvm/Analysis/CFG.h"
63#include "llvm/IR/BasicBlock.h"
64#include "llvm/IR/CFG.h"
65#include "llvm/IR/Constant.h"
66#include "llvm/IR/Constants.h"
67#include "llvm/IR/DIBuilder.h"
68#include "llvm/IR/DataLayout.h"
69#include "llvm/IR/DebugInfo.h"
71#include "llvm/IR/Dominators.h"
73#include "llvm/IR/Function.h"
75#include "llvm/IR/IRBuilder.h"
76#include "llvm/IR/InstrTypes.h"
77#include "llvm/IR/Instruction.h"
80#include "llvm/IR/Intrinsics.h"
81#include "llvm/IR/LLVMContext.h"
82#include "llvm/IR/Metadata.h"
83#include "llvm/IR/Operator.h"
84#include "llvm/IR/PassManager.h"
86#include "llvm/IR/Type.h"
87#include "llvm/IR/Use.h"
88#include "llvm/IR/User.h"
89#include "llvm/IR/Value.h"
90#include "llvm/IR/ValueHandle.h"
94#include "llvm/Support/Debug.h"
103#include <algorithm>
104#include <cassert>
105#include <cstdint>
106#include <memory>
107#include <optional>
108#include <string>
109#include <utility>
110
111#define DEBUG_TYPE "instcombine"
113#include <optional>
114
115using namespace llvm;
116using namespace llvm::PatternMatch;
117
118STATISTIC(NumWorklistIterations,
119 "Number of instruction combining iterations performed");
120STATISTIC(NumOneIteration, "Number of functions with one iteration");
121STATISTIC(NumTwoIterations, "Number of functions with two iterations");
122STATISTIC(NumThreeIterations, "Number of functions with three iterations");
123STATISTIC(NumFourOrMoreIterations,
124 "Number of functions with four or more iterations");
125
126STATISTIC(NumCombined , "Number of insts combined");
127STATISTIC(NumConstProp, "Number of constant folds");
128STATISTIC(NumDeadInst , "Number of dead inst eliminated");
129STATISTIC(NumSunkInst , "Number of instructions sunk");
130STATISTIC(NumExpand, "Number of expansions");
131STATISTIC(NumFactor , "Number of factorizations");
132STATISTIC(NumReassoc , "Number of reassociations");
133DEBUG_COUNTER(VisitCounter, "instcombine-visit",
134 "Controls which instructions are visited");
135
136InstCombiner::IRBuilderInstCombineInserter::~IRBuilderInstCombineInserter() =
137 default;
138
139void InstCombiner::IRBuilderInstCombineInserter::InsertHelper(
140 Instruction *I, const Twine &Name, BasicBlock::iterator InsertPt) const {
142 IC.Worklist.add(I);
143 if (auto *Assume = dyn_cast<AssumeInst>(I))
144 IC.AC.registerAssumption(Assume);
145 if (IC.AnnotationMetadataSource)
146 I->copyMetadata(*IC.AnnotationMetadataSource, LLVMContext::MD_annotation);
147}
148
149std::optional<Instruction *>
151 // Handle target specific intrinsics
152 if (II.getCalledFunction()->isTargetIntrinsic()) {
153 return TTIForTargetIntrinsicsOnly.instCombineIntrinsic(*this, II);
154 }
155 return std::nullopt;
156}
157
159 IntrinsicInst &II, APInt DemandedMask, KnownBits &Known,
160 bool &KnownBitsComputed) {
161 // Handle target specific intrinsics
162 if (II.getCalledFunction()->isTargetIntrinsic()) {
163 return TTIForTargetIntrinsicsOnly.simplifyDemandedUseBitsIntrinsic(
164 *this, II, DemandedMask, Known, KnownBitsComputed);
165 }
166 return std::nullopt;
167}
168
170 IntrinsicInst &II, APInt DemandedElts, APInt &PoisonElts,
171 APInt &PoisonElts2, APInt &PoisonElts3,
172 std::function<void(Instruction *, unsigned, APInt, APInt &)>
173 SimplifyAndSetOp) {
174 // Handle target specific intrinsics
175 if (II.getCalledFunction()->isTargetIntrinsic()) {
176 return TTIForTargetIntrinsicsOnly.simplifyDemandedVectorEltsIntrinsic(
177 *this, II, DemandedElts, PoisonElts, PoisonElts2, PoisonElts3,
178 SimplifyAndSetOp);
179 }
180 return std::nullopt;
181}
182
183bool InstCombiner::isValidAddrSpaceCast(unsigned FromAS, unsigned ToAS) const {
184 // Approved exception for TTI use: This queries a legality property of the
185 // target, not an profitability heuristic. Ideally this should be part of
186 // DataLayout instead.
187 return TTIForTargetIntrinsicsOnly.isValidAddrSpaceCast(FromAS, ToAS);
188}
189
190Value *InstCombinerImpl::EmitGEPOffset(GEPOperator *GEP, bool RewriteGEP) {
191 if (!RewriteGEP)
192 return llvm::emitGEPOffset(&Builder, DL, GEP);
193
194 IRBuilderBase::InsertPointGuard Guard(Builder);
195 auto *Inst = dyn_cast<Instruction>(GEP);
196 if (Inst)
197 Builder.SetInsertPoint(Inst);
198
199 Value *Offset = EmitGEPOffset(GEP);
200 // Rewrite non-trivial GEPs to avoid duplicating the offset arithmetic.
201 if (Inst && !GEP->hasAllConstantIndices() &&
202 !GEP->getSourceElementType()->isIntegerTy(8)) {
204 *Inst, Builder.CreateGEP(Builder.getInt8Ty(), GEP->getPointerOperand(),
205 Offset, "", GEP->getNoWrapFlags()));
207 }
208 return Offset;
209}
210
211Value *InstCombinerImpl::EmitGEPOffsets(ArrayRef<GEPOperator *> GEPs,
212 GEPNoWrapFlags NW, Type *IdxTy,
213 bool RewriteGEPs) {
214 auto Add = [&](Value *Sum, Value *Offset) -> Value * {
215 if (Sum)
216 return Builder.CreateAdd(Sum, Offset, "", NW.hasNoUnsignedWrap(),
217 NW.isInBounds());
218 else
219 return Offset;
220 };
221
222 Value *Sum = nullptr;
223 Value *OneUseSum = nullptr;
224 Value *OneUseBase = nullptr;
225 GEPNoWrapFlags OneUseFlags = GEPNoWrapFlags::all();
226 for (GEPOperator *GEP : reverse(GEPs)) {
227 Value *Offset;
228 {
229 // Expand the offset at the point of the previous GEP to enable rewriting.
230 // However, use the original insertion point for calculating Sum.
231 IRBuilderBase::InsertPointGuard Guard(Builder);
232 auto *Inst = dyn_cast<Instruction>(GEP);
233 if (RewriteGEPs && Inst)
234 Builder.SetInsertPoint(Inst);
235
237 if (Offset->getType() != IdxTy)
238 Offset = Builder.CreateVectorSplat(
239 cast<VectorType>(IdxTy)->getElementCount(), Offset);
240 if (GEP->hasOneUse()) {
241 // Offsets of one-use GEPs will be merged into the next multi-use GEP.
242 OneUseSum = Add(OneUseSum, Offset);
243 OneUseFlags = OneUseFlags.intersectForOffsetAdd(GEP->getNoWrapFlags());
244 if (!OneUseBase)
245 OneUseBase = GEP->getPointerOperand();
246 continue;
247 }
248
249 if (OneUseSum)
250 Offset = Add(OneUseSum, Offset);
251
252 // Rewrite the GEP to reuse the computed offset. This also includes
253 // offsets from preceding one-use GEPs of matched type.
254 if (RewriteGEPs && Inst &&
255 Offset->getType()->isVectorTy() == GEP->getType()->isVectorTy() &&
256 !(GEP->getSourceElementType()->isIntegerTy(8) &&
257 GEP->getOperand(1) == Offset)) {
259 *Inst,
260 Builder.CreatePtrAdd(
261 OneUseBase ? OneUseBase : GEP->getPointerOperand(), Offset, "",
262 OneUseFlags.intersectForOffsetAdd(GEP->getNoWrapFlags())));
264 }
265 }
266
267 Sum = Add(Sum, Offset);
268 OneUseSum = OneUseBase = nullptr;
269 OneUseFlags = GEPNoWrapFlags::all();
270 }
271 if (OneUseSum)
272 Sum = Add(Sum, OneUseSum);
273 if (!Sum)
274 return Constant::getNullValue(IdxTy);
275 return Sum;
276}
277
278/// Legal integers and common types are considered desirable. This is used to
279/// avoid creating instructions with types that may not be supported well by the
280/// the backend.
281/// NOTE: This treats i8, i16 and i32 specially because they are common
282/// types in frontend languages.
283bool InstCombinerImpl::isDesirableIntType(unsigned BitWidth) const {
284 switch (BitWidth) {
285 case 8:
286 case 16:
287 case 32:
288 return true;
289 default:
290 return DL.isLegalInteger(BitWidth);
291 }
292}
293
294/// Return true if it is desirable to convert an integer computation from a
295/// given bit width to a new bit width.
296/// We don't want to convert from a legal or desirable type (like i8) to an
297/// illegal type or from a smaller to a larger illegal type. A width of '1'
298/// is always treated as a desirable type because i1 is a fundamental type in
299/// IR, and there are many specialized optimizations for i1 types.
300/// Common/desirable widths are equally treated as legal to convert to, in
301/// order to open up more combining opportunities.
302bool InstCombinerImpl::shouldChangeType(unsigned FromWidth,
303 unsigned ToWidth) const {
304 bool FromLegal = FromWidth == 1 || DL.isLegalInteger(FromWidth);
305 bool ToLegal = ToWidth == 1 || DL.isLegalInteger(ToWidth);
306
307 // Convert to desirable widths even if they are not legal types.
308 // Only shrink types, to prevent infinite loops.
309 if (ToWidth < FromWidth && isDesirableIntType(ToWidth))
310 return true;
311
312 // If this is a legal or desiable integer from type, and the result would be
313 // an illegal type, don't do the transformation.
314 if ((FromLegal || isDesirableIntType(FromWidth)) && !ToLegal)
315 return false;
316
317 // Otherwise, if both are illegal, do not increase the size of the result. We
318 // do allow things like i160 -> i64, but not i64 -> i160.
319 if (!FromLegal && !ToLegal && ToWidth > FromWidth)
320 return false;
321
322 return true;
323}
324
325/// Return true if it is desirable to convert a computation from 'From' to 'To'.
326/// We don't want to convert from a legal to an illegal type or from a smaller
327/// to a larger illegal type. i1 is always treated as a legal type because it is
328/// a fundamental type in IR, and there are many specialized optimizations for
329/// i1 types.
330bool InstCombinerImpl::shouldChangeType(Type *From, Type *To) const {
331 // TODO: This could be extended to allow vectors. Datalayout changes might be
332 // needed to properly support that.
333 if (!From->isIntegerTy() || !To->isIntegerTy())
334 return false;
335
336 unsigned FromWidth = From->getPrimitiveSizeInBits();
337 unsigned ToWidth = To->getPrimitiveSizeInBits();
338 return shouldChangeType(FromWidth, ToWidth);
339}
340
341// Return true, if No Signed Wrap should be maintained for I.
342// The No Signed Wrap flag can be kept if the operation "B (I.getOpcode) C",
343// where both B and C should be ConstantInts, results in a constant that does
344// not overflow. This function only handles the Add/Sub/Mul opcodes. For
345// all other opcodes, the function conservatively returns false.
348 if (!OBO || !OBO->hasNoSignedWrap())
349 return false;
350
351 const APInt *BVal, *CVal;
352 if (!match(B, m_APInt(BVal)) || !match(C, m_APInt(CVal)))
353 return false;
354
355 // We reason about Add/Sub/Mul Only.
356 bool Overflow = false;
357 switch (I.getOpcode()) {
358 case Instruction::Add:
359 (void)BVal->sadd_ov(*CVal, Overflow);
360 break;
361 case Instruction::Sub:
362 (void)BVal->ssub_ov(*CVal, Overflow);
363 break;
364 case Instruction::Mul:
365 (void)BVal->smul_ov(*CVal, Overflow);
366 break;
367 default:
368 // Conservatively return false for other opcodes.
369 return false;
370 }
371 return !Overflow;
372}
373
376 return OBO && OBO->hasNoUnsignedWrap();
377}
378
381 return OBO && OBO->hasNoSignedWrap();
382}
383
384/// Combine constant operands of associative operations either before or after a
385/// cast to eliminate one of the associative operations:
386/// (op (cast (op X, C2)), C1) --> (cast (op X, op (C1, C2)))
387/// (op (cast (op X, C2)), C1) --> (op (cast X), op (C1, C2))
389 InstCombinerImpl &IC) {
390 auto *Cast = dyn_cast<CastInst>(BinOp1->getOperand(0));
391 if (!Cast || !Cast->hasOneUse())
392 return false;
393
394 // TODO: Enhance logic for other casts and remove this check.
395 auto CastOpcode = Cast->getOpcode();
396 if (CastOpcode != Instruction::ZExt)
397 return false;
398
399 // TODO: Enhance logic for other BinOps and remove this check.
400 if (!BinOp1->isBitwiseLogicOp())
401 return false;
402
403 auto AssocOpcode = BinOp1->getOpcode();
404 auto *BinOp2 = dyn_cast<BinaryOperator>(Cast->getOperand(0));
405 if (!BinOp2 || !BinOp2->hasOneUse() || BinOp2->getOpcode() != AssocOpcode)
406 return false;
407
408 Constant *C1, *C2;
409 if (!match(BinOp1->getOperand(1), m_Constant(C1)) ||
410 !match(BinOp2->getOperand(1), m_Constant(C2)))
411 return false;
412
413 // TODO: This assumes a zext cast.
414 // Eg, if it was a trunc, we'd cast C1 to the source type because casting C2
415 // to the destination type might lose bits.
416
417 // Fold the constants together in the destination type:
418 // (op (cast (op X, C2)), C1) --> (op (cast X), FoldedC)
419 const DataLayout &DL = IC.getDataLayout();
420 Type *DestTy = C1->getType();
421 Constant *CastC2 = ConstantFoldCastOperand(CastOpcode, C2, DestTy, DL);
422 if (!CastC2)
423 return false;
424 Constant *FoldedC = ConstantFoldBinaryOpOperands(AssocOpcode, C1, CastC2, DL);
425 if (!FoldedC)
426 return false;
427
428 IC.replaceOperand(*Cast, 0, BinOp2->getOperand(0));
429 IC.replaceOperand(*BinOp1, 1, FoldedC);
431 Cast->dropPoisonGeneratingFlags();
432 return true;
433}
434
435// Simplifies IntToPtr/PtrToInt RoundTrip Cast.
436// inttoptr ( ptrtoint (x) ) --> x
437Value *InstCombinerImpl::simplifyIntToPtrRoundTripCast(Value *Val) {
438 auto *IntToPtr = dyn_cast<IntToPtrInst>(Val);
439 if (IntToPtr && DL.getTypeSizeInBits(IntToPtr->getDestTy()) ==
440 DL.getTypeSizeInBits(IntToPtr->getSrcTy())) {
441 auto *PtrToInt = dyn_cast<PtrToIntInst>(IntToPtr->getOperand(0));
442 Type *CastTy = IntToPtr->getDestTy();
443 if (PtrToInt &&
444 CastTy->getPointerAddressSpace() ==
445 PtrToInt->getSrcTy()->getPointerAddressSpace() &&
446 DL.getTypeSizeInBits(PtrToInt->getSrcTy()) ==
447 DL.getTypeSizeInBits(PtrToInt->getDestTy()))
448 return PtrToInt->getOperand(0);
449 }
450 return nullptr;
451}
452
453/// This performs a few simplifications for operators that are associative or
454/// commutative:
455///
456/// Commutative operators:
457///
458/// 1. Order operands such that they are listed from right (least complex) to
459/// left (most complex). This puts constants before unary operators before
460/// binary operators.
461///
462/// Associative operators:
463///
464/// 2. Transform: "(A op B) op C" ==> "A op (B op C)" if "B op C" simplifies.
465/// 3. Transform: "A op (B op C)" ==> "(A op B) op C" if "A op B" simplifies.
466///
467/// Associative and commutative operators:
468///
469/// 4. Transform: "(A op B) op C" ==> "(C op A) op B" if "C op A" simplifies.
470/// 5. Transform: "A op (B op C)" ==> "B op (C op A)" if "C op A" simplifies.
471/// 6. Transform: "(A op C1) op (B op C2)" ==> "(A op B) op (C1 op C2)"
472/// if C1 and C2 are constants.
474 Instruction::BinaryOps Opcode = I.getOpcode();
475 bool Changed = false;
476
477 do {
478 // Order operands such that they are listed from right (least complex) to
479 // left (most complex). This puts constants before unary operators before
480 // binary operators.
481 if (I.isCommutative() && getComplexity(I.getOperand(0)) <
482 getComplexity(I.getOperand(1)))
483 Changed = !I.swapOperands();
484
485 if (I.isCommutative()) {
486 if (auto Pair = matchSymmetricPair(I.getOperand(0), I.getOperand(1))) {
487 replaceOperand(I, 0, Pair->first);
488 replaceOperand(I, 1, Pair->second);
489 Changed = true;
490 }
491 }
492
493 BinaryOperator *Op0 = dyn_cast<BinaryOperator>(I.getOperand(0));
494 BinaryOperator *Op1 = dyn_cast<BinaryOperator>(I.getOperand(1));
495
496 if (I.isAssociative()) {
497 // Transform: "(A op B) op C" ==> "A op (B op C)" if "B op C" simplifies.
498 if (Op0 && Op0->getOpcode() == Opcode) {
499 Value *A = Op0->getOperand(0);
500 Value *B = Op0->getOperand(1);
501 Value *C = I.getOperand(1);
502
503 // Does "B op C" simplify?
504 if (Value *V = simplifyBinOp(Opcode, B, C, SQ.getWithInstruction(&I))) {
505 // It simplifies to V. Form "A op V".
506 replaceOperand(I, 0, A);
507 replaceOperand(I, 1, V);
508 bool IsNUW = hasNoUnsignedWrap(I) && hasNoUnsignedWrap(*Op0);
509 bool IsNSW = maintainNoSignedWrap(I, B, C) && hasNoSignedWrap(*Op0);
510
511 // Conservatively clear all optional flags since they may not be
512 // preserved by the reassociation. Reset nsw/nuw based on the above
513 // analysis.
514 if (auto *PDI = dyn_cast<PossiblyDisjointInst>(&I))
515 PDI->setIsDisjoint(false);
516
517 // Note: this is only valid because SimplifyBinOp doesn't look at
518 // the operands to Op0.
520 I.setHasNoUnsignedWrap(IsNUW);
521 I.setHasNoSignedWrap(IsNSW);
522 }
523
524 Changed = true;
525 ++NumReassoc;
526 continue;
527 }
528 }
529
530 // Transform: "A op (B op C)" ==> "(A op B) op C" if "A op B" simplifies.
531 if (Op1 && Op1->getOpcode() == Opcode) {
532 Value *A = I.getOperand(0);
533 Value *B = Op1->getOperand(0);
534 Value *C = Op1->getOperand(1);
535
536 // Does "A op B" simplify?
537 if (Value *V = simplifyBinOp(Opcode, A, B, SQ.getWithInstruction(&I))) {
538 // It simplifies to V. Form "V op C".
539 replaceOperand(I, 0, V);
540 replaceOperand(I, 1, C);
541 // Conservatively clear the optional flags, since they may not be
542 // preserved by the reassociation.
544 I.dropPoisonGeneratingFlags();
545 Changed = true;
546 ++NumReassoc;
547 continue;
548 }
549 }
550 }
551
552 if (I.isAssociative() && I.isCommutative()) {
553 if (simplifyAssocCastAssoc(&I, *this)) {
554 Changed = true;
555 ++NumReassoc;
556 continue;
557 }
558
559 // Transform: "(A op B) op C" ==> "(C op A) op B" if "C op A" simplifies.
560 if (Op0 && Op0->getOpcode() == Opcode) {
561 Value *A = Op0->getOperand(0);
562 Value *B = Op0->getOperand(1);
563 Value *C = I.getOperand(1);
564
565 // Does "C op A" simplify?
566 if (Value *V = simplifyBinOp(Opcode, C, A, SQ.getWithInstruction(&I))) {
567 // It simplifies to V. Form "V op B".
568 replaceOperand(I, 0, V);
569 replaceOperand(I, 1, B);
570 // Conservatively clear the optional flags, since they may not be
571 // preserved by the reassociation.
573 I.dropPoisonGeneratingFlags();
574 Changed = true;
575 ++NumReassoc;
576 continue;
577 }
578 }
579
580 // Transform: "A op (B op C)" ==> "B op (C op A)" if "C op A" simplifies.
581 if (Op1 && Op1->getOpcode() == Opcode) {
582 Value *A = I.getOperand(0);
583 Value *B = Op1->getOperand(0);
584 Value *C = Op1->getOperand(1);
585
586 // Does "C op A" simplify?
587 if (Value *V = simplifyBinOp(Opcode, C, A, SQ.getWithInstruction(&I))) {
588 // It simplifies to V. Form "B op V".
589 replaceOperand(I, 0, B);
590 replaceOperand(I, 1, V);
591 // Conservatively clear the optional flags, since they may not be
592 // preserved by the reassociation.
594 I.dropPoisonGeneratingFlags();
595 Changed = true;
596 ++NumReassoc;
597 continue;
598 }
599 }
600
601 // Transform: "(A op C1) op (B op C2)" ==> "(A op B) op (C1 op C2)"
602 // if C1 and C2 are constants.
603 Value *A, *B;
604 Constant *C1, *C2, *CRes;
605 if (Op0 && Op1 &&
606 Op0->getOpcode() == Opcode && Op1->getOpcode() == Opcode &&
607 match(Op0, m_OneUse(m_BinOp(m_Value(A), m_Constant(C1)))) &&
608 match(Op1, m_OneUse(m_BinOp(m_Value(B), m_Constant(C2)))) &&
609 (CRes = ConstantFoldBinaryOpOperands(Opcode, C1, C2, DL))) {
610 bool IsNUW = hasNoUnsignedWrap(I) &&
611 hasNoUnsignedWrap(*Op0) &&
612 hasNoUnsignedWrap(*Op1);
613 BinaryOperator *NewBO = (IsNUW && Opcode == Instruction::Add) ?
614 BinaryOperator::CreateNUW(Opcode, A, B) :
615 BinaryOperator::Create(Opcode, A, B);
616
617 if (isa<FPMathOperator>(NewBO)) {
618 FastMathFlags Flags = I.getFastMathFlags() &
619 Op0->getFastMathFlags() &
620 Op1->getFastMathFlags();
621 NewBO->setFastMathFlags(Flags);
622 }
623 InsertNewInstWith(NewBO, I.getIterator());
624 NewBO->takeName(Op1);
625 replaceOperand(I, 0, NewBO);
626 replaceOperand(I, 1, CRes);
627 // Conservatively clear the optional flags, since they may not be
628 // preserved by the reassociation.
630 I.dropPoisonGeneratingFlags();
631 if (IsNUW)
632 I.setHasNoUnsignedWrap(true);
633
634 Changed = true;
635 continue;
636 }
637 }
638
639 // No further simplifications.
640 return Changed;
641 } while (true);
642}
643
644/// Return whether "X LOp (Y ROp Z)" is always equal to
645/// "(X LOp Y) ROp (X LOp Z)".
648 // X & (Y | Z) <--> (X & Y) | (X & Z)
649 // X & (Y ^ Z) <--> (X & Y) ^ (X & Z)
650 if (LOp == Instruction::And)
651 return ROp == Instruction::Or || ROp == Instruction::Xor;
652
653 // X | (Y & Z) <--> (X | Y) & (X | Z)
654 if (LOp == Instruction::Or)
655 return ROp == Instruction::And;
656
657 // X * (Y + Z) <--> (X * Y) + (X * Z)
658 // X * (Y - Z) <--> (X * Y) - (X * Z)
659 if (LOp == Instruction::Mul)
660 return ROp == Instruction::Add || ROp == Instruction::Sub;
661
662 return false;
663}
664
665/// Return whether "(X LOp Y) ROp Z" is always equal to
666/// "(X ROp Z) LOp (Y ROp Z)".
670 return leftDistributesOverRight(ROp, LOp);
671
672 // (X {&|^} Y) >> Z <--> (X >> Z) {&|^} (Y >> Z) for all shifts.
674
675 // TODO: It would be nice to handle division, aka "(X + Y)/Z = X/Z + Y/Z",
676 // but this requires knowing that the addition does not overflow and other
677 // such subtleties.
678}
679
680/// This function returns identity value for given opcode, which can be used to
681/// factor patterns like (X * 2) + X ==> (X * 2) + (X * 1) ==> X * (2 + 1).
683 if (isa<Constant>(V))
684 return nullptr;
685
686 return ConstantExpr::getBinOpIdentity(Opcode, V->getType());
687}
688
689/// This function predicates factorization using distributive laws. By default,
690/// it just returns the 'Op' inputs. But for special-cases like
691/// 'add(shl(X, 5), ...)', this function will have TopOpcode == Instruction::Add
692/// and Op = shl(X, 5). The 'shl' is treated as the more general 'mul X, 32' to
693/// allow more factorization opportunities.
696 Value *&LHS, Value *&RHS, BinaryOperator *OtherOp) {
697 assert(Op && "Expected a binary operator");
698 LHS = Op->getOperand(0);
699 RHS = Op->getOperand(1);
700 if (TopOpcode == Instruction::Add || TopOpcode == Instruction::Sub) {
701 Constant *C;
702 if (match(Op, m_Shl(m_Value(), m_ImmConstant(C)))) {
703 // X << C --> X * (1 << C)
705 Instruction::Shl, ConstantInt::get(Op->getType(), 1), C);
706 assert(RHS && "Constant folding of immediate constants failed");
707 return Instruction::Mul;
708 }
709 // TODO: We can add other conversions e.g. shr => div etc.
710 }
711 if (Instruction::isBitwiseLogicOp(TopOpcode)) {
712 if (OtherOp && OtherOp->getOpcode() == Instruction::AShr &&
714 // lshr nneg C, X --> ashr nneg C, X
715 return Instruction::AShr;
716 }
717 }
718 return Op->getOpcode();
719}
720
721/// This tries to simplify binary operations by factorizing out common terms
722/// (e. g. "(A*B)+(A*C)" -> "A*(B+C)").
725 Instruction::BinaryOps InnerOpcode, Value *A,
726 Value *B, Value *C, Value *D) {
727 assert(A && B && C && D && "All values must be provided");
728
729 Value *V = nullptr;
730 Value *RetVal = nullptr;
731 Value *LHS = I.getOperand(0), *RHS = I.getOperand(1);
732 Instruction::BinaryOps TopLevelOpcode = I.getOpcode();
733
734 // Does "X op' Y" always equal "Y op' X"?
735 bool InnerCommutative = Instruction::isCommutative(InnerOpcode);
736
737 // Does "X op' (Y op Z)" always equal "(X op' Y) op (X op' Z)"?
738 if (leftDistributesOverRight(InnerOpcode, TopLevelOpcode)) {
739 // Does the instruction have the form "(A op' B) op (A op' D)" or, in the
740 // commutative case, "(A op' B) op (C op' A)"?
741 if (A == C || (InnerCommutative && A == D)) {
742 if (A != C)
743 std::swap(C, D);
744 // Consider forming "A op' (B op D)".
745 // If "B op D" simplifies then it can be formed with no cost.
746 V = simplifyBinOp(TopLevelOpcode, B, D, SQ.getWithInstruction(&I));
747
748 // If "B op D" doesn't simplify then only go on if one of the existing
749 // operations "A op' B" and "C op' D" will be zapped as no longer used.
750 if (!V && (LHS->hasOneUse() || RHS->hasOneUse()))
751 V = Builder.CreateBinOp(TopLevelOpcode, B, D, RHS->getName());
752 if (V)
753 RetVal = Builder.CreateBinOp(InnerOpcode, A, V);
754 }
755 }
756
757 // Does "(X op Y) op' Z" always equal "(X op' Z) op (Y op' Z)"?
758 if (!RetVal && rightDistributesOverLeft(TopLevelOpcode, InnerOpcode)) {
759 // Does the instruction have the form "(A op' B) op (C op' B)" or, in the
760 // commutative case, "(A op' B) op (B op' D)"?
761 if (B == D || (InnerCommutative && B == C)) {
762 if (B != D)
763 std::swap(C, D);
764 // Consider forming "(A op C) op' B".
765 // If "A op C" simplifies then it can be formed with no cost.
766 V = simplifyBinOp(TopLevelOpcode, A, C, SQ.getWithInstruction(&I));
767
768 // If "A op C" doesn't simplify then only go on if one of the existing
769 // operations "A op' B" and "C op' D" will be zapped as no longer used.
770 if (!V && (LHS->hasOneUse() || RHS->hasOneUse()))
771 V = Builder.CreateBinOp(TopLevelOpcode, A, C, LHS->getName());
772 if (V)
773 RetVal = Builder.CreateBinOp(InnerOpcode, V, B);
774 }
775 }
776
777 if (!RetVal)
778 return nullptr;
779
780 ++NumFactor;
781 RetVal->takeName(&I);
782
783 // Try to add no-overflow flags to the final value.
784 if (isa<BinaryOperator>(RetVal)) {
785 bool HasNSW = false;
786 bool HasNUW = false;
788 HasNSW = I.hasNoSignedWrap();
789 HasNUW = I.hasNoUnsignedWrap();
790 }
791 if (auto *LOBO = dyn_cast<OverflowingBinaryOperator>(LHS)) {
792 HasNSW &= LOBO->hasNoSignedWrap();
793 HasNUW &= LOBO->hasNoUnsignedWrap();
794 }
795
796 if (auto *ROBO = dyn_cast<OverflowingBinaryOperator>(RHS)) {
797 HasNSW &= ROBO->hasNoSignedWrap();
798 HasNUW &= ROBO->hasNoUnsignedWrap();
799 }
800
801 if (TopLevelOpcode == Instruction::Add && InnerOpcode == Instruction::Mul) {
802 // We can propagate 'nsw' if we know that
803 // %Y = mul nsw i16 %X, C
804 // %Z = add nsw i16 %Y, %X
805 // =>
806 // %Z = mul nsw i16 %X, C+1
807 //
808 // iff C+1 isn't INT_MIN
809 const APInt *CInt;
810 if (match(V, m_APInt(CInt)) && !CInt->isMinSignedValue())
811 cast<Instruction>(RetVal)->setHasNoSignedWrap(HasNSW);
812
813 // nuw can be propagated with any constant or nuw value.
814 cast<Instruction>(RetVal)->setHasNoUnsignedWrap(HasNUW);
815 }
816 }
817 return RetVal;
818}
819
820// If `I` has one Const operand and the other matches `(ctpop (not x))`,
821// replace `(ctpop (not x))` with `(sub nuw nsw BitWidth(x), (ctpop x))`.
822// This is only useful is the new subtract can fold so we only handle the
823// following cases:
824// 1) (add/sub/disjoint_or C, (ctpop (not x))
825// -> (add/sub/disjoint_or C', (ctpop x))
826// 1) (cmp pred C, (ctpop (not x))
827// -> (cmp pred C', (ctpop x))
829 unsigned Opc = I->getOpcode();
830 unsigned ConstIdx = 1;
831 switch (Opc) {
832 default:
833 return nullptr;
834 // (ctpop (not x)) <-> (sub nuw nsw BitWidth(x) - (ctpop x))
835 // We can fold the BitWidth(x) with add/sub/icmp as long the other operand
836 // is constant.
837 case Instruction::Sub:
838 ConstIdx = 0;
839 break;
840 case Instruction::ICmp:
841 // Signed predicates aren't correct in some edge cases like for i2 types, as
842 // well since (ctpop x) is known [0, log2(BitWidth(x))] almost all signed
843 // comparisons against it are simplfied to unsigned.
844 if (cast<ICmpInst>(I)->isSigned())
845 return nullptr;
846 break;
847 case Instruction::Or:
848 if (!match(I, m_DisjointOr(m_Value(), m_Value())))
849 return nullptr;
850 [[fallthrough]];
851 case Instruction::Add:
852 break;
853 }
854
855 Value *Op;
856 // Find ctpop.
857 if (!match(I->getOperand(1 - ConstIdx), m_OneUse(m_Ctpop(m_Value(Op)))))
858 return nullptr;
859
860 Constant *C;
861 // Check other operand is ImmConstant.
862 if (!match(I->getOperand(ConstIdx), m_ImmConstant(C)))
863 return nullptr;
864
865 Type *Ty = Op->getType();
866 Constant *BitWidthC = ConstantInt::get(Ty, Ty->getScalarSizeInBits());
867 // Need extra check for icmp. Note if this check is true, it generally means
868 // the icmp will simplify to true/false.
869 if (Opc == Instruction::ICmp && !cast<ICmpInst>(I)->isEquality()) {
870 Constant *Cmp =
872 if (!Cmp || !Cmp->isNullValue())
873 return nullptr;
874 }
875
876 // Check we can invert `(not x)` for free.
877 bool Consumes = false;
878 if (!isFreeToInvert(Op, Op->hasOneUse(), Consumes) || !Consumes)
879 return nullptr;
880 Value *NotOp = getFreelyInverted(Op, Op->hasOneUse(), &Builder);
881 assert(NotOp != nullptr &&
882 "Desync between isFreeToInvert and getFreelyInverted");
883
884 Value *CtpopOfNotOp = Builder.CreateIntrinsic(Ty, Intrinsic::ctpop, NotOp);
885
886 Value *R = nullptr;
887
888 // Do the transformation here to avoid potentially introducing an infinite
889 // loop.
890 switch (Opc) {
891 case Instruction::Sub:
892 R = Builder.CreateAdd(CtpopOfNotOp, ConstantExpr::getSub(C, BitWidthC));
893 break;
894 case Instruction::Or:
895 case Instruction::Add:
896 R = Builder.CreateSub(ConstantExpr::getAdd(C, BitWidthC), CtpopOfNotOp);
897 break;
898 case Instruction::ICmp:
899 R = Builder.CreateICmp(cast<ICmpInst>(I)->getSwappedPredicate(),
900 CtpopOfNotOp, ConstantExpr::getSub(BitWidthC, C));
901 break;
902 default:
903 llvm_unreachable("Unhandled Opcode");
904 }
905 assert(R != nullptr);
906 return replaceInstUsesWith(*I, R);
907}
908
909// (Binop1 (Binop2 (logic_shift X, C), C1), (logic_shift Y, C))
910// IFF
911// 1) the logic_shifts match
912// 2) either both binops are binops and one is `and` or
913// BinOp1 is `and`
914// (logic_shift (inv_logic_shift C1, C), C) == C1 or
915//
916// -> (logic_shift (Binop1 (Binop2 X, inv_logic_shift(C1, C)), Y), C)
917//
918// (Binop1 (Binop2 (logic_shift X, Amt), Mask), (logic_shift Y, Amt))
919// IFF
920// 1) the logic_shifts match
921// 2) BinOp1 == BinOp2 (if BinOp == `add`, then also requires `shl`).
922//
923// -> (BinOp (logic_shift (BinOp X, Y)), Mask)
924//
925// (Binop1 (Binop2 (arithmetic_shift X, Amt), Mask), (arithmetic_shift Y, Amt))
926// IFF
927// 1) Binop1 is bitwise logical operator `and`, `or` or `xor`
928// 2) Binop2 is `not`
929//
930// -> (arithmetic_shift Binop1((not X), Y), Amt)
931
933 const DataLayout &DL = I.getDataLayout();
934 auto IsValidBinOpc = [](unsigned Opc) {
935 switch (Opc) {
936 default:
937 return false;
938 case Instruction::And:
939 case Instruction::Or:
940 case Instruction::Xor:
941 case Instruction::Add:
942 // Skip Sub as we only match constant masks which will canonicalize to use
943 // add.
944 return true;
945 }
946 };
947
948 // Check if we can distribute binop arbitrarily. `add` + `lshr` has extra
949 // constraints.
950 auto IsCompletelyDistributable = [](unsigned BinOpc1, unsigned BinOpc2,
951 unsigned ShOpc) {
952 assert(ShOpc != Instruction::AShr);
953 return (BinOpc1 != Instruction::Add && BinOpc2 != Instruction::Add) ||
954 ShOpc == Instruction::Shl;
955 };
956
957 auto GetInvShift = [](unsigned ShOpc) {
958 assert(ShOpc != Instruction::AShr);
959 return ShOpc == Instruction::LShr ? Instruction::Shl : Instruction::LShr;
960 };
961
962 auto CanDistributeBinops = [&](unsigned BinOpc1, unsigned BinOpc2,
963 unsigned ShOpc, Constant *CMask,
964 Constant *CShift) {
965 // If the BinOp1 is `and` we don't need to check the mask.
966 if (BinOpc1 == Instruction::And)
967 return true;
968
969 // For all other possible transfers we need complete distributable
970 // binop/shift (anything but `add` + `lshr`).
971 if (!IsCompletelyDistributable(BinOpc1, BinOpc2, ShOpc))
972 return false;
973
974 // If BinOp2 is `and`, any mask works (this only really helps for non-splat
975 // vecs, otherwise the mask will be simplified and the following check will
976 // handle it).
977 if (BinOpc2 == Instruction::And)
978 return true;
979
980 // Otherwise, need mask that meets the below requirement.
981 // (logic_shift (inv_logic_shift Mask, ShAmt), ShAmt) == Mask
982 Constant *MaskInvShift =
983 ConstantFoldBinaryOpOperands(GetInvShift(ShOpc), CMask, CShift, DL);
984 return ConstantFoldBinaryOpOperands(ShOpc, MaskInvShift, CShift, DL) ==
985 CMask;
986 };
987
988 auto MatchBinOp = [&](unsigned ShOpnum) -> Instruction * {
989 Constant *CMask, *CShift;
990 Value *X, *Y, *ShiftedX, *Mask, *Shift;
991 if (!match(I.getOperand(ShOpnum),
992 m_OneUse(m_Shift(m_Value(Y), m_Value(Shift)))))
993 return nullptr;
994 if (!match(
995 I.getOperand(1 - ShOpnum),
998 m_Value(ShiftedX)),
999 m_Value(Mask)))))
1000 return nullptr;
1001 // Make sure we are matching instruction shifts and not ConstantExpr
1002 auto *IY = dyn_cast<Instruction>(I.getOperand(ShOpnum));
1003 auto *IX = dyn_cast<Instruction>(ShiftedX);
1004 if (!IY || !IX)
1005 return nullptr;
1006
1007 // LHS and RHS need same shift opcode
1008 unsigned ShOpc = IY->getOpcode();
1009 if (ShOpc != IX->getOpcode())
1010 return nullptr;
1011
1012 // Make sure binop is real instruction and not ConstantExpr
1013 auto *BO2 = dyn_cast<Instruction>(I.getOperand(1 - ShOpnum));
1014 if (!BO2)
1015 return nullptr;
1016
1017 unsigned BinOpc = BO2->getOpcode();
1018 // Make sure we have valid binops.
1019 if (!IsValidBinOpc(I.getOpcode()) || !IsValidBinOpc(BinOpc))
1020 return nullptr;
1021
1022 if (ShOpc == Instruction::AShr) {
1023 if (Instruction::isBitwiseLogicOp(I.getOpcode()) &&
1024 BinOpc == Instruction::Xor && match(Mask, m_AllOnes())) {
1025 Value *NotX = Builder.CreateNot(X);
1026 Value *NewBinOp = Builder.CreateBinOp(I.getOpcode(), Y, NotX);
1028 static_cast<Instruction::BinaryOps>(ShOpc), NewBinOp, Shift);
1029 }
1030
1031 return nullptr;
1032 }
1033
1034 // If BinOp1 == BinOp2 and it's bitwise or shl with add, then just
1035 // distribute to drop the shift irrelevant of constants.
1036 if (BinOpc == I.getOpcode() &&
1037 IsCompletelyDistributable(I.getOpcode(), BinOpc, ShOpc)) {
1038 Value *NewBinOp2 = Builder.CreateBinOp(I.getOpcode(), X, Y);
1039 Value *NewBinOp1 = Builder.CreateBinOp(
1040 static_cast<Instruction::BinaryOps>(ShOpc), NewBinOp2, Shift);
1041 return BinaryOperator::Create(I.getOpcode(), NewBinOp1, Mask);
1042 }
1043
1044 // Otherwise we can only distribute by constant shifting the mask, so
1045 // ensure we have constants.
1046 if (!match(Shift, m_ImmConstant(CShift)))
1047 return nullptr;
1048 if (!match(Mask, m_ImmConstant(CMask)))
1049 return nullptr;
1050
1051 // Check if we can distribute the binops.
1052 if (!CanDistributeBinops(I.getOpcode(), BinOpc, ShOpc, CMask, CShift))
1053 return nullptr;
1054
1055 Constant *NewCMask =
1056 ConstantFoldBinaryOpOperands(GetInvShift(ShOpc), CMask, CShift, DL);
1057 Value *NewBinOp2 = Builder.CreateBinOp(
1058 static_cast<Instruction::BinaryOps>(BinOpc), X, NewCMask);
1059 Value *NewBinOp1 = Builder.CreateBinOp(I.getOpcode(), Y, NewBinOp2);
1060 return BinaryOperator::Create(static_cast<Instruction::BinaryOps>(ShOpc),
1061 NewBinOp1, CShift);
1062 };
1063
1064 if (Instruction *R = MatchBinOp(0))
1065 return R;
1066 return MatchBinOp(1);
1067}
1068
1069// (Binop (zext C), (select C, T, F))
1070// -> (select C, (binop 1, T), (binop 0, F))
1071//
1072// (Binop (sext C), (select C, T, F))
1073// -> (select C, (binop -1, T), (binop 0, F))
1074//
1075// Attempt to simplify binary operations into a select with folded args, when
1076// one operand of the binop is a select instruction and the other operand is a
1077// zext/sext extension, whose value is the select condition.
1080 // TODO: this simplification may be extended to any speculatable instruction,
1081 // not just binops, and would possibly be handled better in FoldOpIntoSelect.
1082 Instruction::BinaryOps Opc = I.getOpcode();
1083 Value *LHS = I.getOperand(0), *RHS = I.getOperand(1);
1084 Value *A, *CondVal, *TrueVal, *FalseVal;
1085 Value *CastOp;
1086 Constant *CastTrueVal, *CastFalseVal;
1087
1088 auto MatchSelectAndCast = [&](Value *CastOp, Value *SelectOp) {
1089 return match(CastOp, m_SelectLike(m_Value(A), m_Constant(CastTrueVal),
1090 m_Constant(CastFalseVal))) &&
1091 match(SelectOp, m_Select(m_Value(CondVal), m_Value(TrueVal),
1092 m_Value(FalseVal)));
1093 };
1094
1095 // Make sure one side of the binop is a select instruction, and the other is a
1096 // zero/sign extension operating on a i1.
1097 if (MatchSelectAndCast(LHS, RHS))
1098 CastOp = LHS;
1099 else if (MatchSelectAndCast(RHS, LHS))
1100 CastOp = RHS;
1101 else
1102 return nullptr;
1103
1104 SelectInst *SI = cast<SelectInst>(CastOp == LHS ? RHS : LHS);
1105
1106 auto NewFoldedConst = [&](bool IsTrueArm, Value *V) {
1107 bool IsCastOpRHS = (CastOp == RHS);
1108 Value *CastVal = IsTrueArm ? CastFalseVal : CastTrueVal;
1109
1110 return IsCastOpRHS ? Builder.CreateBinOp(Opc, V, CastVal)
1111 : Builder.CreateBinOp(Opc, CastVal, V);
1112 };
1113
1114 // If the value used in the zext/sext is the select condition, or the negated
1115 // of the select condition, the binop can be simplified.
1116 if (CondVal == A) {
1117 Value *NewTrueVal = NewFoldedConst(false, TrueVal);
1118 return SelectInst::Create(CondVal, NewTrueVal,
1119 NewFoldedConst(true, FalseVal), "", nullptr, SI);
1120 }
1121 if (match(A, m_Not(m_Specific(CondVal)))) {
1122 Value *NewTrueVal = NewFoldedConst(true, TrueVal);
1123 return SelectInst::Create(CondVal, NewTrueVal,
1124 NewFoldedConst(false, FalseVal), "", nullptr, SI);
1125 }
1126
1127 return nullptr;
1128}
1129
1131 Value *LHS = I.getOperand(0), *RHS = I.getOperand(1);
1134 Instruction::BinaryOps TopLevelOpcode = I.getOpcode();
1135 Value *A, *B, *C, *D;
1136 Instruction::BinaryOps LHSOpcode, RHSOpcode;
1137
1138 if (Op0)
1139 LHSOpcode = getBinOpsForFactorization(TopLevelOpcode, Op0, A, B, Op1);
1140 if (Op1)
1141 RHSOpcode = getBinOpsForFactorization(TopLevelOpcode, Op1, C, D, Op0);
1142
1143 // The instruction has the form "(A op' B) op (C op' D)". Try to factorize
1144 // a common term.
1145 if (Op0 && Op1 && LHSOpcode == RHSOpcode)
1146 if (Value *V = tryFactorization(I, SQ, Builder, LHSOpcode, A, B, C, D))
1147 return V;
1148
1149 // The instruction has the form "(A op' B) op (C)". Try to factorize common
1150 // term.
1151 if (Op0)
1152 if (Value *Ident = getIdentityValue(LHSOpcode, RHS))
1153 if (Value *V =
1154 tryFactorization(I, SQ, Builder, LHSOpcode, A, B, RHS, Ident))
1155 return V;
1156
1157 // The instruction has the form "(B) op (C op' D)". Try to factorize common
1158 // term.
1159 if (Op1)
1160 if (Value *Ident = getIdentityValue(RHSOpcode, LHS))
1161 if (Value *V =
1162 tryFactorization(I, SQ, Builder, RHSOpcode, LHS, Ident, C, D))
1163 return V;
1164
1165 return nullptr;
1166}
1167
1168/// This tries to simplify binary operations which some other binary operation
1169/// distributes over either by factorizing out common terms
1170/// (eg "(A*B)+(A*C)" -> "A*(B+C)") or expanding out if this results in
1171/// simplifications (eg: "A & (B | C) -> (A&B) | (A&C)" if this is a win).
1172/// Returns the simplified value, or null if it didn't simplify.
1174 Value *LHS = I.getOperand(0), *RHS = I.getOperand(1);
1177 Instruction::BinaryOps TopLevelOpcode = I.getOpcode();
1178
1179 // Factorization.
1180 if (Value *R = tryFactorizationFolds(I))
1181 return R;
1182
1183 // Expansion.
1184 if (Op0 && rightDistributesOverLeft(Op0->getOpcode(), TopLevelOpcode)) {
1185 // The instruction has the form "(A op' B) op C". See if expanding it out
1186 // to "(A op C) op' (B op C)" results in simplifications.
1187 Value *A = Op0->getOperand(0), *B = Op0->getOperand(1), *C = RHS;
1188 Instruction::BinaryOps InnerOpcode = Op0->getOpcode(); // op'
1189
1190 // Disable the use of undef because it's not safe to distribute undef.
1191 auto SQDistributive = SQ.getWithInstruction(&I).getWithoutUndef();
1192 Value *L = simplifyBinOp(TopLevelOpcode, A, C, SQDistributive);
1193 Value *R = simplifyBinOp(TopLevelOpcode, B, C, SQDistributive);
1194
1195 // Do "A op C" and "B op C" both simplify?
1196 if (L && R) {
1197 // They do! Return "L op' R".
1198 ++NumExpand;
1199 C = Builder.CreateBinOp(InnerOpcode, L, R);
1200 C->takeName(&I);
1201 return C;
1202 }
1203
1204 // Does "A op C" simplify to the identity value for the inner opcode?
1205 if (L && L == ConstantExpr::getBinOpIdentity(InnerOpcode, L->getType())) {
1206 // They do! Return "B op C".
1207 ++NumExpand;
1208 C = Builder.CreateBinOp(TopLevelOpcode, B, C);
1209 C->takeName(&I);
1210 return C;
1211 }
1212
1213 // Does "B op C" simplify to the identity value for the inner opcode?
1214 if (R && R == ConstantExpr::getBinOpIdentity(InnerOpcode, R->getType())) {
1215 // They do! Return "A op C".
1216 ++NumExpand;
1217 C = Builder.CreateBinOp(TopLevelOpcode, A, C);
1218 C->takeName(&I);
1219 return C;
1220 }
1221 }
1222
1223 if (Op1 && leftDistributesOverRight(TopLevelOpcode, Op1->getOpcode())) {
1224 // The instruction has the form "A op (B op' C)". See if expanding it out
1225 // to "(A op B) op' (A op C)" results in simplifications.
1226 Value *A = LHS, *B = Op1->getOperand(0), *C = Op1->getOperand(1);
1227 Instruction::BinaryOps InnerOpcode = Op1->getOpcode(); // op'
1228
1229 // Disable the use of undef because it's not safe to distribute undef.
1230 auto SQDistributive = SQ.getWithInstruction(&I).getWithoutUndef();
1231 Value *L = simplifyBinOp(TopLevelOpcode, A, B, SQDistributive);
1232 Value *R = simplifyBinOp(TopLevelOpcode, A, C, SQDistributive);
1233
1234 // Do "A op B" and "A op C" both simplify?
1235 if (L && R) {
1236 // They do! Return "L op' R".
1237 ++NumExpand;
1238 A = Builder.CreateBinOp(InnerOpcode, L, R);
1239 A->takeName(&I);
1240 return A;
1241 }
1242
1243 // Does "A op B" simplify to the identity value for the inner opcode?
1244 if (L && L == ConstantExpr::getBinOpIdentity(InnerOpcode, L->getType())) {
1245 // They do! Return "A op C".
1246 ++NumExpand;
1247 A = Builder.CreateBinOp(TopLevelOpcode, A, C);
1248 A->takeName(&I);
1249 return A;
1250 }
1251
1252 // Does "A op C" simplify to the identity value for the inner opcode?
1253 if (R && R == ConstantExpr::getBinOpIdentity(InnerOpcode, R->getType())) {
1254 // They do! Return "A op B".
1255 ++NumExpand;
1256 A = Builder.CreateBinOp(TopLevelOpcode, A, B);
1257 A->takeName(&I);
1258 return A;
1259 }
1260 }
1261
1262 return SimplifySelectsFeedingBinaryOp(I, LHS, RHS);
1263}
1264
1265static std::optional<std::pair<Value *, Value *>>
1267 if (LHS->getParent() != RHS->getParent())
1268 return std::nullopt;
1269
1270 if (LHS->getNumIncomingValues() < 2)
1271 return std::nullopt;
1272
1273 if (!equal(LHS->blocks(), RHS->blocks()))
1274 return std::nullopt;
1275
1276 Value *L0 = LHS->getIncomingValue(0);
1277 Value *R0 = RHS->getIncomingValue(0);
1278
1279 for (unsigned I = 1, E = LHS->getNumIncomingValues(); I != E; ++I) {
1280 Value *L1 = LHS->getIncomingValue(I);
1281 Value *R1 = RHS->getIncomingValue(I);
1282
1283 if ((L0 == L1 && R0 == R1) || (L0 == R1 && R0 == L1))
1284 continue;
1285
1286 return std::nullopt;
1287 }
1288
1289 return std::optional(std::pair(L0, R0));
1290}
1291
1292std::optional<std::pair<Value *, Value *>>
1293InstCombinerImpl::matchSymmetricPair(Value *LHS, Value *RHS) {
1296 if (!LHSInst || !RHSInst || LHSInst->getOpcode() != RHSInst->getOpcode())
1297 return std::nullopt;
1298 switch (LHSInst->getOpcode()) {
1299 case Instruction::PHI:
1301 case Instruction::Select: {
1302 Value *Cond = LHSInst->getOperand(0);
1303 Value *TrueVal = LHSInst->getOperand(1);
1304 Value *FalseVal = LHSInst->getOperand(2);
1305 if (Cond == RHSInst->getOperand(0) && TrueVal == RHSInst->getOperand(2) &&
1306 FalseVal == RHSInst->getOperand(1))
1307 return std::pair(TrueVal, FalseVal);
1308 return std::nullopt;
1309 }
1310 case Instruction::Call: {
1311 // Match min(a, b) and max(a, b)
1312 MinMaxIntrinsic *LHSMinMax = dyn_cast<MinMaxIntrinsic>(LHSInst);
1313 MinMaxIntrinsic *RHSMinMax = dyn_cast<MinMaxIntrinsic>(RHSInst);
1314 if (LHSMinMax && RHSMinMax &&
1315 LHSMinMax->getPredicate() ==
1317 ((LHSMinMax->getLHS() == RHSMinMax->getLHS() &&
1318 LHSMinMax->getRHS() == RHSMinMax->getRHS()) ||
1319 (LHSMinMax->getLHS() == RHSMinMax->getRHS() &&
1320 LHSMinMax->getRHS() == RHSMinMax->getLHS())))
1321 return std::pair(LHSMinMax->getLHS(), LHSMinMax->getRHS());
1322 return std::nullopt;
1323 }
1324 default:
1325 return std::nullopt;
1326 }
1327}
1328
1330 Value *LHS,
1331 Value *RHS) {
1332 Value *A, *B, *C, *D, *E, *F;
1333 bool LHSIsSelect = match(LHS, m_Select(m_Value(A), m_Value(B), m_Value(C)));
1334 bool RHSIsSelect = match(RHS, m_Select(m_Value(D), m_Value(E), m_Value(F)));
1335 if (!LHSIsSelect && !RHSIsSelect)
1336 return nullptr;
1337
1338 SelectInst *SI = cast<SelectInst>(LHSIsSelect ? LHS : RHS);
1339
1340 FastMathFlags FMF;
1342 if (const auto *FPOp = dyn_cast<FPMathOperator>(&I)) {
1343 FMF = FPOp->getFastMathFlags();
1344 Builder.setFastMathFlags(FMF);
1345 }
1346
1347 Instruction::BinaryOps Opcode = I.getOpcode();
1348 SimplifyQuery Q = SQ.getWithInstruction(&I);
1349
1350 Value *Cond, *True = nullptr, *False = nullptr;
1351
1352 // If V is a select whose condition is implied by Cond, resolve it to the
1353 // appropriate arm for this value of Cond.
1354 auto simplifySelectWithImpliedCond = [&](Value *V, Value *Cond,
1355 bool CondIsTrue) -> Value * {
1356 auto *InnerSI = dyn_cast<SelectInst>(V);
1357 if (!InnerSI || Cond->getType() != InnerSI->getCondition()->getType())
1358 return V;
1359
1360 if (std::optional<bool> Implied =
1361 isImpliedCondition(Cond, InnerSI->getCondition(), DL, CondIsTrue))
1362 return InnerSI->getOperand(*Implied ? 1 : 2);
1363 return V;
1364 };
1365
1366 // Special-case for add/negate combination. Replace the zero in the negation
1367 // with the trailing add operand:
1368 // (Cond ? TVal : -N) + Z --> Cond ? True : (Z - N)
1369 // (Cond ? -N : FVal) + Z --> Cond ? (Z - N) : False
1370 auto foldAddNegate = [&](Value *TVal, Value *FVal, Value *Z) -> Value * {
1371 // We need an 'add' and exactly 1 arm of the select to have been simplified.
1372 if (Opcode != Instruction::Add || (!True && !False) || (True && False))
1373 return nullptr;
1374 Value *N;
1375 if (True && match(FVal, m_Neg(m_Value(N)))) {
1376 Value *Sub = Builder.CreateSub(Z, N);
1377 return Builder.CreateSelect(Cond, True, Sub, I.getName(), SI);
1378 }
1379 if (False && match(TVal, m_Neg(m_Value(N)))) {
1380 Value *Sub = Builder.CreateSub(Z, N);
1381 return Builder.CreateSelect(Cond, Sub, False, I.getName(), SI);
1382 }
1383 return nullptr;
1384 };
1385
1386 if (LHSIsSelect && RHSIsSelect && A == D) {
1387 // (A ? B : C) op (A ? E : F) -> A ? (B op E) : (C op F)
1388 Cond = A;
1389 True = simplifyBinOp(Opcode, B, E, FMF, Q);
1390 False = simplifyBinOp(Opcode, C, F, FMF, Q);
1391
1392 if (LHS->hasOneUse() && RHS->hasOneUse()) {
1393 if (False && !True)
1394 True = Builder.CreateBinOp(Opcode, B, E);
1395 else if (True && !False)
1396 False = Builder.CreateBinOp(Opcode, C, F);
1397 }
1398 } else if (LHSIsSelect && LHS->hasOneUse()) {
1399 // (A ? B : C) op Y -> A ? (B op Y) : (C op Y)
1400 Cond = A;
1401 Value *TrueRHS = simplifySelectWithImpliedCond(RHS, Cond, true);
1402 Value *FalseRHS = simplifySelectWithImpliedCond(RHS, Cond, false);
1403 True = simplifyBinOp(Opcode, B, TrueRHS, FMF, Q);
1404 False = simplifyBinOp(Opcode, C, FalseRHS, FMF, Q);
1405 if (Value *NewSel = foldAddNegate(B, C, RHS))
1406 return NewSel;
1407 } else if (RHSIsSelect && RHS->hasOneUse()) {
1408 // X op (D ? E : F) -> D ? (X op E) : (X op F)
1409 Cond = D;
1410 Value *TrueLHS = simplifySelectWithImpliedCond(LHS, Cond, true);
1411 Value *FalseLHS = simplifySelectWithImpliedCond(LHS, Cond, false);
1412 True = simplifyBinOp(Opcode, TrueLHS, E, FMF, Q);
1413 False = simplifyBinOp(Opcode, FalseLHS, F, FMF, Q);
1414 if (Value *NewSel = foldAddNegate(E, F, LHS))
1415 return NewSel;
1416 }
1417
1418 if (!True || !False)
1419 return nullptr;
1420
1421 Value *NewSI = Builder.CreateSelect(Cond, True, False, I.getName(), SI);
1422 NewSI->takeName(&I);
1423 return NewSI;
1424}
1425
1426/// Freely adapt every user of V as-if V was changed to !V.
1427/// WARNING: only if canFreelyInvertAllUsersOf() said this can be done.
1429 assert(!isa<Constant>(I) && "Shouldn't invert users of constant");
1430 for (User *U : make_early_inc_range(I->users())) {
1431 if (U == IgnoredUser)
1432 continue; // Don't consider this user.
1433 switch (cast<Instruction>(U)->getOpcode()) {
1434 case Instruction::Select: {
1435 auto *SI = cast<SelectInst>(U);
1436 SI->swapValues();
1437 SI->swapProfMetadata();
1438 break;
1439 }
1440 case Instruction::CondBr: {
1442 BI->swapSuccessors(); // swaps prof metadata too
1443 if (BPI)
1444 BPI->swapSuccEdgesProbabilities(BI->getParent());
1445 break;
1446 }
1447 case Instruction::Xor:
1449 // Add to worklist for DCE.
1451 break;
1452 default:
1453 llvm_unreachable("Got unexpected user - out of sync with "
1454 "canFreelyInvertAllUsersOf() ?");
1455 }
1456 }
1457
1458 // Update pre-existing debug value uses.
1459 SmallVector<DbgVariableRecord *, 4> DbgVariableRecords;
1460 llvm::findDbgValues(I, DbgVariableRecords);
1461
1462 for (DbgVariableRecord *DbgVal : DbgVariableRecords) {
1463 SmallVector<uint64_t, 1> Ops = {dwarf::DW_OP_not};
1464 for (unsigned Idx = 0, End = DbgVal->getNumVariableLocationOps();
1465 Idx != End; ++Idx)
1466 if (DbgVal->getVariableLocationOp(Idx) == I)
1467 DbgVal->setExpression(
1468 DIExpression::appendOpsToArg(DbgVal->getExpression(), Ops, Idx));
1469 }
1470}
1471
1472/// Given a 'sub' instruction, return the RHS of the instruction if the LHS is a
1473/// constant zero (which is the 'negate' form).
1474Value *InstCombinerImpl::dyn_castNegVal(Value *V) const {
1475 Value *NegV;
1476 if (match(V, m_Neg(m_Value(NegV))))
1477 return NegV;
1478
1479 // Constants can be considered to be negated values if they can be folded.
1481 return ConstantExpr::getNeg(C);
1482
1484 if (C->getType()->getElementType()->isIntegerTy())
1485 return ConstantExpr::getNeg(C);
1486
1488 for (unsigned i = 0, e = CV->getNumOperands(); i != e; ++i) {
1489 Constant *Elt = CV->getAggregateElement(i);
1490 if (!Elt)
1491 return nullptr;
1492
1493 if (isa<UndefValue>(Elt))
1494 continue;
1495
1496 if (!isa<ConstantInt>(Elt))
1497 return nullptr;
1498 }
1499 return ConstantExpr::getNeg(CV);
1500 }
1501
1502 // Negate integer vector splats.
1503 if (auto *CV = dyn_cast<Constant>(V))
1504 if (CV->getType()->isVectorTy() &&
1505 CV->getType()->getScalarType()->isIntegerTy() && CV->getSplatValue())
1506 return ConstantExpr::getNeg(CV);
1507
1508 return nullptr;
1509}
1510
1511// Try to fold:
1512// 1) (fp_binop ({s|u}itofp x), ({s|u}itofp y))
1513// -> ({s|u}itofp (int_binop x, y))
1514// 2) (fp_binop ({s|u}itofp x), FpC)
1515// -> ({s|u}itofp (int_binop x, (fpto{s|u}i FpC)))
1516//
1517// Assuming the sign of the cast for x/y is `OpsFromSigned`.
1518Instruction *InstCombinerImpl::foldFBinOpOfIntCastsFromSign(
1519 BinaryOperator &BO, bool OpsFromSigned, std::array<Value *, 2> IntOps,
1521
1522 Type *FPTy = BO.getType();
1523 Type *IntTy = IntOps[0]->getType();
1524
1525 unsigned IntSz = IntTy->getScalarSizeInBits();
1526 // This is the maximum number of inuse bits by the integer where the int -> fp
1527 // casts are exact.
1528 unsigned MaxRepresentableBits =
1530
1531 // Preserve known number of leading bits. This can allow us to trivial nsw/nuw
1532 // checks later on.
1533 unsigned NumUsedLeadingBits[2] = {IntSz, IntSz};
1534
1535 // NB: This only comes up if OpsFromSigned is true, so there is no need to
1536 // cache if between calls to `foldFBinOpOfIntCastsFromSign`.
1537 auto IsNonZero = [&](unsigned OpNo) -> bool {
1538 if (OpsKnown[OpNo].hasKnownBits() &&
1539 OpsKnown[OpNo].getKnownBits(SQ).isNonZero())
1540 return true;
1541 return isKnownNonZero(IntOps[OpNo], SQ);
1542 };
1543
1544 auto IsNonNeg = [&](unsigned OpNo) -> bool {
1545 // NB: This matches the impl in ValueTracking, we just try to use cached
1546 // knownbits here. If we ever start supporting WithCache for
1547 // `isKnownNonNegative`, change this to an explicit call.
1548 return OpsKnown[OpNo].getKnownBits(SQ).isNonNegative();
1549 };
1550
1551 // Check if we know for certain that ({s|u}itofp op) is exact.
1552 auto IsValidPromotion = [&](unsigned OpNo) -> bool {
1553 // Can we treat this operand as the desired sign?
1554 if (OpsFromSigned != isa<SIToFPInst>(BO.getOperand(OpNo)) &&
1555 !IsNonNeg(OpNo))
1556 return false;
1557
1558 // If fp precision >= bitwidth(op) then its exact.
1559 // NB: This is slightly conservative for `sitofp`. For signed conversion, we
1560 // can handle `MaxRepresentableBits == IntSz - 1` as the sign bit will be
1561 // handled specially. We can't, however, increase the bound arbitrarily for
1562 // `sitofp` as for larger sizes, it won't sign extend.
1563 if (MaxRepresentableBits < IntSz) {
1564 // Otherwise if its signed cast check that fp precisions >= bitwidth(op) -
1565 // numSignBits(op).
1566 // TODO: If we add support for `WithCache` in `ComputeNumSignBits`, change
1567 // `IntOps[OpNo]` arguments to `KnownOps[OpNo]`.
1568 if (OpsFromSigned)
1569 NumUsedLeadingBits[OpNo] = IntSz - ComputeNumSignBits(IntOps[OpNo]);
1570 // Finally for unsigned check that fp precision >= bitwidth(op) -
1571 // numLeadingZeros(op).
1572 else {
1573 NumUsedLeadingBits[OpNo] =
1574 IntSz - OpsKnown[OpNo].getKnownBits(SQ).countMinLeadingZeros();
1575 }
1576 }
1577 // NB: We could also check if op is known to be a power of 2 or zero (which
1578 // will always be representable). Its unlikely, however, that is we are
1579 // unable to bound op in any way we will be able to pass the overflow checks
1580 // later on.
1581
1582 if (MaxRepresentableBits < NumUsedLeadingBits[OpNo])
1583 return false;
1584 // Signed + Mul also requires that op is non-zero to avoid -0 cases.
1585 return !OpsFromSigned || BO.getOpcode() != Instruction::FMul ||
1586 IsNonZero(OpNo);
1587 };
1588
1589 // If we have a constant rhs, see if we can losslessly convert it to an int.
1590 if (Op1FpC != nullptr) {
1591 // Signed + Mul req non-zero
1592 if (OpsFromSigned && BO.getOpcode() == Instruction::FMul &&
1593 !match(Op1FpC, m_NonZeroFP()))
1594 return nullptr;
1595
1597 OpsFromSigned ? Instruction::FPToSI : Instruction::FPToUI, Op1FpC,
1598 IntTy, DL);
1599 if (Op1IntC == nullptr)
1600 return nullptr;
1601 if (ConstantFoldCastOperand(OpsFromSigned ? Instruction::SIToFP
1602 : Instruction::UIToFP,
1603 Op1IntC, FPTy, DL) != Op1FpC)
1604 return nullptr;
1605
1606 // First try to keep sign of cast the same.
1607 IntOps[1] = Op1IntC;
1608 }
1609
1610 // Ensure lhs/rhs integer types match.
1611 if (IntTy != IntOps[1]->getType())
1612 return nullptr;
1613
1614 if (Op1FpC == nullptr) {
1615 if (!IsValidPromotion(1))
1616 return nullptr;
1617 }
1618 if (!IsValidPromotion(0))
1619 return nullptr;
1620
1621 // Final we check if the integer version of the binop will not overflow.
1623 // Because of the precision check, we can often rule out overflows.
1624 bool NeedsOverflowCheck = true;
1625 // Try to conservatively rule out overflow based on the already done precision
1626 // checks.
1627 unsigned OverflowMaxOutputBits = OpsFromSigned ? 2 : 1;
1628 unsigned OverflowMaxCurBits =
1629 std::max(NumUsedLeadingBits[0], NumUsedLeadingBits[1]);
1630 bool OutputSigned = OpsFromSigned;
1631 switch (BO.getOpcode()) {
1632 case Instruction::FAdd:
1633 IntOpc = Instruction::Add;
1634 OverflowMaxOutputBits += OverflowMaxCurBits;
1635 break;
1636 case Instruction::FSub:
1637 IntOpc = Instruction::Sub;
1638 OverflowMaxOutputBits += OverflowMaxCurBits;
1639 break;
1640 case Instruction::FMul:
1641 IntOpc = Instruction::Mul;
1642 OverflowMaxOutputBits += OverflowMaxCurBits * 2;
1643 break;
1644 default:
1645 llvm_unreachable("Unsupported binop");
1646 }
1647 // The precision check may have already ruled out overflow.
1648 if (OverflowMaxOutputBits < IntSz) {
1649 NeedsOverflowCheck = false;
1650 // We can bound unsigned overflow from sub to in range signed value (this is
1651 // what allows us to avoid the overflow check for sub).
1652 if (IntOpc == Instruction::Sub)
1653 OutputSigned = true;
1654 }
1655
1656 // Precision check did not rule out overflow, so need to check.
1657 // TODO: If we add support for `WithCache` in `willNotOverflow`, change
1658 // `IntOps[...]` arguments to `KnownOps[...]`.
1659 if (NeedsOverflowCheck &&
1660 !willNotOverflow(IntOpc, IntOps[0], IntOps[1], BO, OutputSigned))
1661 return nullptr;
1662
1663 Value *IntBinOp = Builder.CreateBinOp(IntOpc, IntOps[0], IntOps[1]);
1664 if (auto *IntBO = dyn_cast<BinaryOperator>(IntBinOp)) {
1665 IntBO->setHasNoSignedWrap(OutputSigned);
1666 IntBO->setHasNoUnsignedWrap(!OutputSigned);
1667 }
1668 if (OutputSigned)
1669 return new SIToFPInst(IntBinOp, FPTy);
1670 return new UIToFPInst(IntBinOp, FPTy);
1671}
1672
1673// Try to fold:
1674// 1) (fp_binop ({s|u}itofp x), ({s|u}itofp y))
1675// -> ({s|u}itofp (int_binop x, y))
1676// 2) (fp_binop ({s|u}itofp x), FpC)
1677// -> ({s|u}itofp (int_binop x, (fpto{s|u}i FpC)))
1678Instruction *InstCombinerImpl::foldFBinOpOfIntCasts(BinaryOperator &BO) {
1679 // Don't perform the fold on vectors, as the integer operation may be much
1680 // more expensive than the float operation in that case.
1681 if (BO.getType()->isVectorTy())
1682 return nullptr;
1683
1684 std::array<Value *, 2> IntOps = {nullptr, nullptr};
1685 Constant *Op1FpC = nullptr;
1686 // Check for:
1687 // 1) (binop ({s|u}itofp x), ({s|u}itofp y))
1688 // 2) (binop ({s|u}itofp x), FpC)
1689 if (!match(BO.getOperand(0), m_IToFP(m_Value(IntOps[0]))))
1690 return nullptr;
1691
1692 if (!match(BO.getOperand(1), m_Constant(Op1FpC)) &&
1693 !match(BO.getOperand(1), m_IToFP(m_Value(IntOps[1]))))
1694 return nullptr;
1695
1696 // Cache KnownBits a bit to potentially save some analysis.
1697 SmallVector<WithCache<const Value *>, 2> OpsKnown = {IntOps[0], IntOps[1]};
1698
1699 // Try treating x/y as coming from both `uitofp` and `sitofp`. There are
1700 // different constraints depending on the sign of the cast.
1701 // NB: `(uitofp nneg X)` == `(sitofp nneg X)`.
1702 if (Instruction *R = foldFBinOpOfIntCastsFromSign(BO, /*OpsFromSigned=*/false,
1703 IntOps, Op1FpC, OpsKnown))
1704 return R;
1705 return foldFBinOpOfIntCastsFromSign(BO, /*OpsFromSigned=*/true, IntOps,
1706 Op1FpC, OpsKnown);
1707}
1708
1709/// A binop with a constant operand and a sign-extended boolean operand may be
1710/// converted into a select of constants by applying the binary operation to
1711/// the constant with the two possible values of the extended boolean (0 or -1).
1712Instruction *InstCombinerImpl::foldBinopOfSextBoolToSelect(BinaryOperator &BO) {
1713 // TODO: Handle non-commutative binop (constant is operand 0).
1714 // TODO: Handle zext.
1715 // TODO: Peek through 'not' of cast.
1716 Value *BO0 = BO.getOperand(0);
1717 Value *BO1 = BO.getOperand(1);
1718 Value *X;
1719 Constant *C;
1720 if (!match(BO0, m_SExt(m_Value(X))) || !match(BO1, m_ImmConstant(C)) ||
1721 !X->getType()->isIntOrIntVectorTy(1))
1722 return nullptr;
1723
1724 // bo (sext i1 X), C --> select X, (bo -1, C), (bo 0, C)
1727 Value *TVal = Builder.CreateBinOp(BO.getOpcode(), Ones, C);
1728 Value *FVal = Builder.CreateBinOp(BO.getOpcode(), Zero, C);
1729 return createSelectInstWithUnknownProfile(X, TVal, FVal);
1730}
1731
1733 bool IsTrueArm) {
1735 for (Value *Op : I.operands()) {
1736 Value *V = nullptr;
1737 if (Op == SI) {
1738 V = IsTrueArm ? SI->getTrueValue() : SI->getFalseValue();
1739 } else if (match(SI->getCondition(),
1742 m_Specific(Op), m_Value(V))) &&
1744 // Pass
1745 } else if (match(Op, m_ZExt(m_Specific(SI->getCondition())))) {
1746 V = IsTrueArm ? ConstantInt::get(Op->getType(), 1)
1747 : ConstantInt::getNullValue(Op->getType());
1748 } else {
1749 V = Op;
1750 }
1751 Ops.push_back(V);
1752 }
1753
1754 return simplifyInstructionWithOperands(&I, Ops, I.getDataLayout());
1755}
1756
1758 Value *NewOp, InstCombiner &IC) {
1759 Instruction *Clone = I.clone();
1760 Clone->replaceUsesOfWith(SI, NewOp);
1762 IC.InsertNewInstBefore(Clone, I.getIterator());
1763 return Clone;
1764}
1765
1767 bool FoldWithMultiUse,
1768 bool SimplifyBothArms) {
1769 // Don't modify shared select instructions unless set FoldWithMultiUse
1770 if (!SI->hasOneUser() && !FoldWithMultiUse)
1771 return nullptr;
1772
1773 Value *TV = SI->getTrueValue();
1774 Value *FV = SI->getFalseValue();
1775
1776 // Bool selects with constant operands can be folded to logical ops.
1777 if (SI->getType()->isIntOrIntVectorTy(1))
1778 return nullptr;
1779
1780 // Avoid breaking min/max reduction pattern,
1781 // which is necessary for vectorization later.
1783 for (Value *IntrinOp : Op.operands())
1784 if (auto *PN = dyn_cast<PHINode>(IntrinOp))
1785 for (Value *PhiOp : PN->operands())
1786 if (PhiOp == &Op)
1787 return nullptr;
1788
1789 // Test if a FCmpInst instruction is used exclusively by a select as
1790 // part of a minimum or maximum operation. If so, refrain from doing
1791 // any other folding. This helps out other analyses which understand
1792 // non-obfuscated minimum and maximum idioms. And in this case, at
1793 // least one of the comparison operands has at least one user besides
1794 // the compare (the select), which would often largely negate the
1795 // benefit of folding anyway.
1796 if (auto *CI = dyn_cast<FCmpInst>(SI->getCondition())) {
1797 if (CI->hasOneUse()) {
1798 Value *Op0 = CI->getOperand(0), *Op1 = CI->getOperand(1);
1799 if (((TV == Op0 && FV == Op1) || (FV == Op0 && TV == Op1)) &&
1800 !CI->isCommutative())
1801 return nullptr;
1802 }
1803 }
1804
1805 // Make sure that one of the select arms folds successfully.
1806 Value *NewTV = simplifyOperationIntoSelectOperand(Op, SI, /*IsTrueArm=*/true);
1807 Value *NewFV =
1808 simplifyOperationIntoSelectOperand(Op, SI, /*IsTrueArm=*/false);
1809 if (!NewTV && !NewFV)
1810 return nullptr;
1811
1812 if (SimplifyBothArms && !(NewTV && NewFV))
1813 return nullptr;
1814
1815 // Create an instruction for the arm that did not fold.
1816 if (!NewTV)
1817 NewTV = foldOperationIntoSelectOperand(Op, SI, TV, *this);
1818 if (!NewFV)
1819 NewFV = foldOperationIntoSelectOperand(Op, SI, FV, *this);
1820
1821 SelectInst *NewSel = SelectInst::Create(SI->getCondition(), NewTV, NewFV);
1822
1823 // Preserve metadata that remains valid for the transformed select including
1824 // source location information.
1825 NewSel->copyMetadata(*SI,
1826 {LLVMContext::MD_prof, LLVMContext::MD_unpredictable,
1827 LLVMContext::MD_dbg});
1828
1829 return NewSel;
1830}
1831
1833 Value *InValue, BasicBlock *InBB,
1834 const DataLayout &DL,
1835 const SimplifyQuery SQ) {
1836 // NB: It is a precondition of this transform that the operands be
1837 // phi translatable!
1839 for (Value *Op : I.operands()) {
1840 if (Op == PN)
1841 Ops.push_back(InValue);
1842 else
1843 Ops.push_back(Op->DoPHITranslation(PN->getParent(), InBB));
1844 }
1845
1846 // Don't consider the simplification successful if we get back a constant
1847 // expression. That's just an instruction in hiding.
1848 // Also reject the case where we simplify back to the phi node. We wouldn't
1849 // be able to remove it in that case.
1851 &I, Ops, SQ.getWithInstruction(InBB->getTerminator()));
1852 if (NewVal && NewVal != PN && !match(NewVal, m_ConstantExpr()))
1853 return NewVal;
1854
1855 // Check if incoming PHI value can be replaced with constant
1856 // based on implied condition.
1857 CondBrInst *TerminatorBI = dyn_cast<CondBrInst>(InBB->getTerminator());
1858 const ICmpInst *ICmp = dyn_cast<ICmpInst>(&I);
1859 if (TerminatorBI &&
1860 TerminatorBI->getSuccessor(0) != TerminatorBI->getSuccessor(1) && ICmp) {
1861 bool LHSIsTrue = TerminatorBI->getSuccessor(0) == PN->getParent();
1862 std::optional<bool> ImpliedCond = isImpliedCondition(
1863 TerminatorBI->getCondition(), ICmp->getCmpPredicate(), Ops[0], Ops[1],
1864 DL, LHSIsTrue);
1865 if (ImpliedCond)
1866 return ConstantInt::getBool(I.getType(), ImpliedCond.value());
1867 }
1868
1869 return nullptr;
1870}
1871
1872/// In some cases it is beneficial to fold a select into a binary operator.
1873/// For example:
1874/// %1 = or %in, 4
1875/// %2 = select %cond, %1, %in
1876/// %3 = or %2, 1
1877/// =>
1878/// %1 = select i1 %cond, 5, 1
1879/// %2 = or %1, %in
1881 assert(Op.isAssociative() && "The operation must be associative!");
1882
1883 SelectInst *SI = dyn_cast<SelectInst>(Op.getOperand(0));
1884
1885 Constant *Const;
1886 if (!SI || !match(Op.getOperand(1), m_ImmConstant(Const)) ||
1887 !Op.hasOneUse() || !SI->hasOneUse())
1888 return nullptr;
1889
1890 Value *TV = SI->getTrueValue();
1891 Value *FV = SI->getFalseValue();
1892 Value *Input, *NewTV, *NewFV;
1893 Constant *Const2;
1894
1895 if (TV->hasOneUse() && match(TV, m_BinOp(Op.getOpcode(), m_Specific(FV),
1896 m_ImmConstant(Const2)))) {
1897 NewTV = ConstantFoldBinaryInstruction(Op.getOpcode(), Const, Const2);
1898 NewFV = Const;
1899 Input = FV;
1900 } else if (FV->hasOneUse() &&
1901 match(FV, m_BinOp(Op.getOpcode(), m_Specific(TV),
1902 m_ImmConstant(Const2)))) {
1903 NewTV = Const;
1904 NewFV = ConstantFoldBinaryInstruction(Op.getOpcode(), Const, Const2);
1905 Input = TV;
1906 } else
1907 return nullptr;
1908
1909 if (!NewTV || !NewFV)
1910 return nullptr;
1911
1912 Value *NewSI = Builder.CreateSelect(SI->getCondition(), NewTV, NewFV, "", SI);
1913 return BinaryOperator::Create(Op.getOpcode(), NewSI, Input);
1914}
1915
1917 bool AllowMultipleUses) {
1918 unsigned NumPHIValues = PN->getNumIncomingValues();
1919 if (NumPHIValues == 0)
1920 return nullptr;
1921
1922 // We normally only transform phis with a single use. However, if a PHI has
1923 // multiple uses and they are all the same operation, we can fold *all* of the
1924 // uses into the PHI.
1925 bool OneUse = PN->hasOneUse();
1926 bool IdenticalUsers = false;
1927 if (!AllowMultipleUses && !OneUse) {
1928 // Walk the use list for the instruction, comparing them to I.
1929 for (User *U : PN->users()) {
1931 if (UI != &I && !I.isIdenticalTo(UI))
1932 return nullptr;
1933 }
1934 // Otherwise, we can replace *all* users with the new PHI we form.
1935 IdenticalUsers = true;
1936 }
1937
1938 // Check that all operands are phi-translatable.
1939 for (Value *Op : I.operands()) {
1940 if (Op == PN)
1941 continue;
1942
1943 // Non-instructions never require phi-translation.
1944 auto *I = dyn_cast<Instruction>(Op);
1945 if (!I)
1946 continue;
1947
1948 // Phi-translate can handle phi nodes in the same block.
1949 if (isa<PHINode>(I))
1950 if (I->getParent() == PN->getParent())
1951 continue;
1952
1953 // Operand dominates the block, no phi-translation necessary.
1954 if (DT.dominates(I, PN->getParent()))
1955 continue;
1956
1957 // Not phi-translatable, bail out.
1958 return nullptr;
1959 }
1960
1961 // Check to see whether the instruction can be folded into each phi operand.
1962 // If there is one operand that does not fold, remember the BB it is in.
1963 SmallVector<Value *> NewPhiValues;
1964 SmallVector<unsigned int> OpsToMoveUseToIncomingBB;
1965 bool SeenNonSimplifiedInVal = false;
1966 for (unsigned i = 0; i != NumPHIValues; ++i) {
1967 Value *InVal = PN->getIncomingValue(i);
1968 BasicBlock *InBB = PN->getIncomingBlock(i);
1969
1970 if (auto *NewVal = simplifyInstructionWithPHI(I, PN, InVal, InBB, DL, SQ)) {
1971 NewPhiValues.push_back(NewVal);
1972 continue;
1973 }
1974
1975 // Handle some cases that can't be fully simplified, but where we know that
1976 // the two instructions will fold into one.
1977 auto WillFold = [&]() {
1978 if (!InVal->hasUseList() || !InVal->hasOneUser())
1979 return false;
1980
1981 // icmp of ucmp/scmp with constant will fold to icmp.
1982 const APInt *Ignored;
1983 if (isa<CmpIntrinsic>(InVal) &&
1984 match(&I, m_ICmp(m_Specific(PN), m_APInt(Ignored))))
1985 return true;
1986
1987 // icmp eq zext(bool), 0 will fold to !bool.
1988 if (isa<ZExtInst>(InVal) &&
1989 cast<ZExtInst>(InVal)->getSrcTy()->isIntOrIntVectorTy(1) &&
1990 match(&I,
1992 return true;
1993
1994 return false;
1995 };
1996
1997 if (WillFold()) {
1998 OpsToMoveUseToIncomingBB.push_back(i);
1999 NewPhiValues.push_back(nullptr);
2000 continue;
2001 }
2002
2003 if (!OneUse && !IdenticalUsers)
2004 return nullptr;
2005
2006 if (SeenNonSimplifiedInVal)
2007 return nullptr; // More than one non-simplified value.
2008 SeenNonSimplifiedInVal = true;
2009
2010 // If there is exactly one non-simplified value, we can insert a copy of the
2011 // operation in that block. However, if this is a critical edge, we would
2012 // be inserting the computation on some other paths (e.g. inside a loop).
2013 // Only do this if the pred block is unconditionally branching into the phi
2014 // block. Also, make sure that the pred block is not dead code.
2016 if (!BI || !DT.isReachableFromEntry(InBB))
2017 return nullptr;
2018
2019 NewPhiValues.push_back(nullptr);
2020 OpsToMoveUseToIncomingBB.push_back(i);
2021
2022 // Do not push the operation across a loop backedge. This could result in
2023 // an infinite combine loop, and is generally non-profitable (especially
2024 // if the operation was originally outside the loop).
2025 if (isBackEdge(InBB, PN->getParent()))
2026 return nullptr;
2027 }
2028
2029 // Clone the instruction that uses the phi node and move it into the incoming
2030 // BB because we know that the next iteration of InstCombine will simplify it.
2032 for (auto OpIndex : OpsToMoveUseToIncomingBB) {
2033 Value *Op = PN->getIncomingValue(OpIndex);
2034 BasicBlock *OpBB = PN->getIncomingBlock(OpIndex);
2035
2036 Instruction *Clone = Clones.lookup(OpBB);
2037 if (!Clone) {
2038 Clone = I.clone();
2039 for (Use &U : Clone->operands()) {
2040 if (U == PN)
2041 U = Op;
2042 else
2043 U = U->DoPHITranslation(PN->getParent(), OpBB);
2044 }
2045 Clone = InsertNewInstBefore(Clone, OpBB->getTerminator()->getIterator());
2046 Clones.insert({OpBB, Clone});
2047 // We may have speculated the instruction.
2049 }
2050
2051 NewPhiValues[OpIndex] = Clone;
2052 }
2053
2054 // Okay, we can do the transformation: create the new PHI node.
2055 PHINode *NewPN = PHINode::Create(I.getType(), PN->getNumIncomingValues());
2056 InsertNewInstBefore(NewPN, PN->getIterator());
2057 NewPN->takeName(PN);
2058 NewPN->setDebugLoc(PN->getDebugLoc());
2059
2060 for (unsigned i = 0; i != NumPHIValues; ++i)
2061 NewPN->addIncoming(NewPhiValues[i], PN->getIncomingBlock(i));
2062
2063 if (IdenticalUsers) {
2064 // Collect and deduplicate users up-front to avoid iterator invalidation.
2066 for (User *U : PN->users()) {
2068 if (User == &I)
2069 continue;
2070 ToReplace.insert(User);
2071 }
2072 for (Instruction *I : ToReplace) {
2073 replaceInstUsesWith(*I, NewPN);
2075 }
2076 OneUse = true;
2077 }
2078
2079 if (OneUse) {
2080 replaceAllDbgUsesWith(*PN, *NewPN, *PN, DT);
2081 }
2082 return replaceInstUsesWith(I, NewPN);
2083}
2084
2086 if (!BO.isAssociative())
2087 return nullptr;
2088
2089 // Find the interleaved binary ops.
2090 auto Opc = BO.getOpcode();
2091 auto *BO0 = dyn_cast<BinaryOperator>(BO.getOperand(0));
2092 auto *BO1 = dyn_cast<BinaryOperator>(BO.getOperand(1));
2093 if (!BO0 || !BO1 || !BO0->hasNUses(2) || !BO1->hasNUses(2) ||
2094 BO0->getOpcode() != Opc || BO1->getOpcode() != Opc ||
2095 !BO0->isAssociative() || !BO1->isAssociative() ||
2096 BO0->getParent() != BO1->getParent())
2097 return nullptr;
2098
2099 assert(BO.isCommutative() && BO0->isCommutative() && BO1->isCommutative() &&
2100 "Expected commutative instructions!");
2101
2102 // Find the matching phis, forming the recurrences.
2103 PHINode *PN0, *PN1;
2104 Value *Start0, *Step0, *Start1, *Step1;
2105 if (!matchSimpleRecurrence(BO0, PN0, Start0, Step0) || !PN0->hasOneUse() ||
2106 !matchSimpleRecurrence(BO1, PN1, Start1, Step1) || !PN1->hasOneUse() ||
2107 PN0->getParent() != PN1->getParent())
2108 return nullptr;
2109
2110 assert(PN0->getNumIncomingValues() == 2 && PN1->getNumIncomingValues() == 2 &&
2111 "Expected PHIs with two incoming values!");
2112
2113 // Convert the start and step values to constants.
2114 auto *Init0 = dyn_cast<Constant>(Start0);
2115 auto *Init1 = dyn_cast<Constant>(Start1);
2116 auto *C0 = dyn_cast<Constant>(Step0);
2117 auto *C1 = dyn_cast<Constant>(Step1);
2118 if (!Init0 || !Init1 || !C0 || !C1)
2119 return nullptr;
2120
2121 // Fold the recurrence constants.
2122 auto *Init = ConstantFoldBinaryInstruction(Opc, Init0, Init1);
2123 auto *C = ConstantFoldBinaryInstruction(Opc, C0, C1);
2124 if (!Init || !C)
2125 return nullptr;
2126
2127 // Create the reduced PHI.
2128 auto *NewPN = PHINode::Create(PN0->getType(), PN0->getNumIncomingValues(),
2129 "reduced.phi");
2130
2131 // Create the new binary op.
2132 auto *NewBO = BinaryOperator::Create(Opc, NewPN, C);
2133 if (Opc == Instruction::FAdd || Opc == Instruction::FMul) {
2134 // Intersect FMF flags for FADD and FMUL.
2135 FastMathFlags Intersect = BO0->getFastMathFlags() &
2136 BO1->getFastMathFlags() & BO.getFastMathFlags();
2137 NewBO->setFastMathFlags(Intersect);
2138 } else {
2139 OverflowTracking Flags;
2140 Flags.AllKnownNonNegative = false;
2141 Flags.AllKnownNonZero = false;
2142 Flags.mergeFlags(*BO0);
2143 Flags.mergeFlags(*BO1);
2144 Flags.mergeFlags(BO);
2145 Flags.applyFlags(*NewBO);
2146 }
2147 NewBO->takeName(&BO);
2148
2149 for (unsigned I = 0, E = PN0->getNumIncomingValues(); I != E; ++I) {
2150 auto *V = PN0->getIncomingValue(I);
2151 auto *BB = PN0->getIncomingBlock(I);
2152 if (V == Init0) {
2153 assert(((PN1->getIncomingValue(0) == Init1 &&
2154 PN1->getIncomingBlock(0) == BB) ||
2155 (PN1->getIncomingValue(1) == Init1 &&
2156 PN1->getIncomingBlock(1) == BB)) &&
2157 "Invalid incoming block!");
2158 NewPN->addIncoming(Init, BB);
2159 } else if (V == BO0) {
2160 assert(((PN1->getIncomingValue(0) == BO1 &&
2161 PN1->getIncomingBlock(0) == BB) ||
2162 (PN1->getIncomingValue(1) == BO1 &&
2163 PN1->getIncomingBlock(1) == BB)) &&
2164 "Invalid incoming block!");
2165 NewPN->addIncoming(NewBO, BB);
2166 } else
2167 llvm_unreachable("Unexpected incoming value!");
2168 }
2169
2170 LLVM_DEBUG(dbgs() << " Combined " << *PN0 << "\n " << *BO0
2171 << "\n with " << *PN1 << "\n " << *BO1
2172 << '\n');
2173
2174 // Insert the new recurrence and remove the old (dead) ones.
2175 InsertNewInstWith(NewPN, PN0->getIterator());
2176 InsertNewInstWith(NewBO, BO0->getIterator());
2177
2184
2185 return replaceInstUsesWith(BO, NewBO);
2186}
2187
2189 // Attempt to fold binary operators whose operands are simple recurrences.
2190 if (auto *NewBO = foldBinopWithRecurrence(BO))
2191 return NewBO;
2192
2193 // TODO: This should be similar to the incoming values check in foldOpIntoPhi:
2194 // we are guarding against replicating the binop in >1 predecessor.
2195 // This could miss matching a phi with 2 constant incoming values.
2196 auto *Phi0 = dyn_cast<PHINode>(BO.getOperand(0));
2197 auto *Phi1 = dyn_cast<PHINode>(BO.getOperand(1));
2198 if (!Phi0 || !Phi1 || !Phi0->hasOneUse() || !Phi1->hasOneUse() ||
2199 Phi0->getNumOperands() != Phi1->getNumOperands())
2200 return nullptr;
2201
2202 // TODO: Remove the restriction for binop being in the same block as the phis.
2203 if (BO.getParent() != Phi0->getParent() ||
2204 BO.getParent() != Phi1->getParent())
2205 return nullptr;
2206
2207 // Fold if there is at least one specific constant value in phi0 or phi1's
2208 // incoming values that comes from the same block and this specific constant
2209 // value can be used to do optimization for specific binary operator.
2210 // For example:
2211 // %phi0 = phi i32 [0, %bb0], [%i, %bb1]
2212 // %phi1 = phi i32 [%j, %bb0], [0, %bb1]
2213 // %add = add i32 %phi0, %phi1
2214 // ==>
2215 // %add = phi i32 [%j, %bb0], [%i, %bb1]
2217 /*AllowRHSConstant*/ false);
2218 if (C) {
2219 SmallVector<Value *, 4> NewIncomingValues;
2220 auto CanFoldIncomingValuePair = [&](std::tuple<Use &, Use &> T) {
2221 auto &Phi0Use = std::get<0>(T);
2222 auto &Phi1Use = std::get<1>(T);
2223 if (Phi0->getIncomingBlock(Phi0Use) != Phi1->getIncomingBlock(Phi1Use))
2224 return false;
2225 Value *Phi0UseV = Phi0Use.get();
2226 Value *Phi1UseV = Phi1Use.get();
2227 if (Phi0UseV == C)
2228 NewIncomingValues.push_back(Phi1UseV);
2229 else if (Phi1UseV == C)
2230 NewIncomingValues.push_back(Phi0UseV);
2231 else
2232 return false;
2233 return true;
2234 };
2235
2236 if (all_of(zip(Phi0->operands(), Phi1->operands()),
2237 CanFoldIncomingValuePair)) {
2238 PHINode *NewPhi =
2239 PHINode::Create(Phi0->getType(), Phi0->getNumOperands());
2240 assert(NewIncomingValues.size() == Phi0->getNumOperands() &&
2241 "The number of collected incoming values should equal the number "
2242 "of the original PHINode operands!");
2243 for (unsigned I = 0; I < Phi0->getNumOperands(); I++)
2244 NewPhi->addIncoming(NewIncomingValues[I], Phi0->getIncomingBlock(I));
2245 return NewPhi;
2246 }
2247 }
2248
2249 if (Phi0->getNumOperands() != 2 || Phi1->getNumOperands() != 2)
2250 return nullptr;
2251
2252 // Match a pair of incoming constants for one of the predecessor blocks.
2253 BasicBlock *ConstBB, *OtherBB;
2254 Constant *C0, *C1;
2255 if (match(Phi0->getIncomingValue(0), m_ImmConstant(C0))) {
2256 ConstBB = Phi0->getIncomingBlock(0);
2257 OtherBB = Phi0->getIncomingBlock(1);
2258 } else if (match(Phi0->getIncomingValue(1), m_ImmConstant(C0))) {
2259 ConstBB = Phi0->getIncomingBlock(1);
2260 OtherBB = Phi0->getIncomingBlock(0);
2261 } else {
2262 return nullptr;
2263 }
2264 if (!match(Phi1->getIncomingValueForBlock(ConstBB), m_ImmConstant(C1)))
2265 return nullptr;
2266
2267 // The block that we are hoisting to must reach here unconditionally.
2268 // Otherwise, we could be speculatively executing an expensive or
2269 // non-speculative op.
2270 auto *PredBlockBranch = dyn_cast<UncondBrInst>(OtherBB->getTerminator());
2271 if (!PredBlockBranch || !DT.isReachableFromEntry(OtherBB))
2272 return nullptr;
2273
2274 // TODO: This check could be tightened to only apply to binops (div/rem) that
2275 // are not safe to speculatively execute. But that could allow hoisting
2276 // potentially expensive instructions (fdiv for example).
2277 for (auto BBIter = BO.getParent()->begin(); &*BBIter != &BO; ++BBIter)
2279 return nullptr;
2280
2281 // Fold constants for the predecessor block with constant incoming values.
2282 Constant *NewC = ConstantFoldBinaryOpOperands(BO.getOpcode(), C0, C1, DL);
2283 if (!NewC)
2284 return nullptr;
2285
2286 // Make a new binop in the predecessor block with the non-constant incoming
2287 // values.
2288 Builder.SetInsertPoint(PredBlockBranch);
2289 Value *NewBO = Builder.CreateBinOp(BO.getOpcode(),
2290 Phi0->getIncomingValueForBlock(OtherBB),
2291 Phi1->getIncomingValueForBlock(OtherBB));
2292 if (auto *NotFoldedNewBO = dyn_cast<BinaryOperator>(NewBO))
2293 NotFoldedNewBO->copyIRFlags(&BO);
2294
2295 // Replace the binop with a phi of the new values. The old phis are dead.
2296 PHINode *NewPhi = PHINode::Create(BO.getType(), 2);
2297 NewPhi->addIncoming(NewBO, OtherBB);
2298 NewPhi->addIncoming(NewC, ConstBB);
2299 return NewPhi;
2300}
2301
2303 auto TryFoldOperand = [&](unsigned OpIdx,
2304 bool IsOtherParamConst) -> Instruction * {
2305 if (auto *Sel = dyn_cast<SelectInst>(I.getOperand(OpIdx)))
2306 return FoldOpIntoSelect(I, Sel, false, !IsOtherParamConst);
2307 if (auto *PN = dyn_cast<PHINode>(I.getOperand(OpIdx)))
2308 return foldOpIntoPhi(I, PN);
2309 return nullptr;
2310 };
2311
2312 if (Instruction *NewI =
2313 TryFoldOperand(/*OpIdx=*/0, isa<Constant>(I.getOperand(1))))
2314 return NewI;
2315 return TryFoldOperand(/*OpIdx=*/1, isa<Constant>(I.getOperand(0)));
2316}
2317
2319 // If this GEP has only 0 indices, it is the same pointer as
2320 // Src. If Src is not a trivial GEP too, don't combine
2321 // the indices.
2322 if (GEP.hasAllZeroIndices() && !Src.hasAllZeroIndices() &&
2323 !Src.hasOneUse())
2324 return false;
2325 return true;
2326}
2327
2328/// Find a constant NewC that has property:
2329/// shuffle(NewC, poison, ShMask) = C
2330/// for lanes that select NewC. Lanes that select the poison operand are not
2331/// constrained.
2332/// Returns nullptr if such a constant does not exist e.g. ShMask=<0,0> C=<1,2>
2333///
2334/// A 1-to-1 mapping is not required. Example:
2335/// ShMask = <1,1,2,2> and C = <5,5,6,6> --> NewC = <poison,5,6,poison>
2337 VectorType *NewCTy) {
2338 if (isa<ScalableVectorType>(NewCTy)) {
2339 Constant *Splat = C->getSplatValue();
2340 if (!Splat)
2341 return nullptr;
2343 }
2344
2345 if (cast<FixedVectorType>(NewCTy)->getNumElements() >
2346 cast<FixedVectorType>(C->getType())->getNumElements())
2347 return nullptr;
2348
2349 unsigned NewCNumElts = cast<FixedVectorType>(NewCTy)->getNumElements();
2350 PoisonValue *PoisonScalar = PoisonValue::get(C->getType()->getScalarType());
2351 SmallVector<Constant *, 16> NewVecC(NewCNumElts, PoisonScalar);
2352 unsigned NumElts = cast<FixedVectorType>(C->getType())->getNumElements();
2353 for (unsigned I = 0; I < NumElts; ++I) {
2354 Constant *CElt = C->getAggregateElement(I);
2355 if (ShMask[I] >= 0) {
2356 int MaskElt = ShMask[I];
2357 if (MaskElt >= (int)NewCNumElts)
2358 continue;
2359
2360 Constant *NewCElt = NewVecC[MaskElt];
2361 // Bail out if:
2362 // 1. The constant vector contains a constant expression.
2363 // 2. The shuffle needs an element of the constant vector that can't
2364 // be mapped to a new constant vector.
2365 // 3. This is a widening shuffle that copies elements of V1 into the
2366 // extended elements (extending with poison is allowed).
2367 if (!CElt || (!isa<PoisonValue>(NewCElt) && NewCElt != CElt) ||
2368 I >= NewCNumElts)
2369 return nullptr;
2370 NewVecC[MaskElt] = CElt;
2371 }
2372 }
2373 return ConstantVector::get(NewVecC);
2374}
2375
2376// Get the result of `Vector Op Splat` (or Splat Op Vector if \p SplatLHS).
2378 Constant *Splat, bool SplatLHS,
2379 const DataLayout &DL) {
2380 ElementCount EC = cast<VectorType>(Vector->getType())->getElementCount();
2382 Constant *RHS = Vector;
2383 if (!SplatLHS)
2384 std::swap(LHS, RHS);
2385 return ConstantFoldBinaryOpOperands(Opcode, LHS, RHS, DL);
2386}
2387
2388template <Intrinsic::ID SpliceID>
2390 InstCombiner::BuilderTy &Builder) {
2391 Value *LHS = Inst.getOperand(0), *RHS = Inst.getOperand(1);
2392 auto CreateBinOpSplice = [&](Value *X, Value *Y, Value *Offset) {
2393 Value *V = Builder.CreateBinOp(Inst.getOpcode(), X, Y, Inst.getName());
2394 if (auto *BO = dyn_cast<BinaryOperator>(V))
2395 BO->copyIRFlags(&Inst);
2396 Module *M = Inst.getModule();
2397 Function *F = Intrinsic::getOrInsertDeclaration(M, SpliceID, V->getType());
2398 return CallInst::Create(F, {V, PoisonValue::get(V->getType()), Offset});
2399 };
2400 Value *V1, *V2, *Offset;
2401 if (match(LHS,
2403 // Op(splice(V1, poison, offset), splice(V2, poison, offset))
2404 // -> splice(Op(V1, V2), poison, offset)
2406 m_Specific(Offset))) &&
2407 (LHS->hasOneUse() || RHS->hasOneUse() ||
2408 (LHS == RHS && LHS->hasNUses(2))))
2409 return CreateBinOpSplice(V1, V2, Offset);
2410
2411 // Op(splice(V1, poison, offset), RHSSplat)
2412 // -> splice(Op(V1, RHSSplat), poison, offset)
2413 if (LHS->hasOneUse() && isSplatValue(RHS))
2414 return CreateBinOpSplice(V1, RHS, Offset);
2415 }
2416 // Op(LHSSplat, splice(V2, poison, offset))
2417 // -> splice(Op(LHSSplat, V2), poison, offset)
2418 else if (isSplatValue(LHS) &&
2420 m_Value(Offset)))))
2421 return CreateBinOpSplice(LHS, V2, Offset);
2422
2423 // TODO: Fold binops of the form
2424 // Op(splice(poison, V1, offset), splice(poison, V2, offset))
2425 // -> splice(poison, Op(V1, V2), offset)
2426
2427 return nullptr;
2428}
2429
2431 if (!isa<VectorType>(Inst.getType()))
2432 return nullptr;
2433
2434 BinaryOperator::BinaryOps Opcode = Inst.getOpcode();
2435 Value *LHS = Inst.getOperand(0), *RHS = Inst.getOperand(1);
2436 assert(cast<VectorType>(LHS->getType())->getElementCount() ==
2437 cast<VectorType>(Inst.getType())->getElementCount());
2438 assert(cast<VectorType>(RHS->getType())->getElementCount() ==
2439 cast<VectorType>(Inst.getType())->getElementCount());
2440
2441 auto foldConstantsThroughSubVectorInsertSplat =
2442 [&](Value *MaybeSubVector, Value *MaybeSplat,
2443 bool SplatLHS) -> Instruction * {
2444 Value *Idx;
2445 Constant *Splat, *SubVector, *Dest;
2446 if (!match(MaybeSplat, m_Splat(m_Constant(Splat))) ||
2447 !match(MaybeSubVector,
2448 m_VectorInsert(m_Constant(Dest), m_Constant(SubVector),
2449 m_Value(Idx))))
2450 return nullptr;
2451 SubVector =
2452 constantFoldBinOpWithSplat(Opcode, SubVector, Splat, SplatLHS, DL);
2453 Dest = constantFoldBinOpWithSplat(Opcode, Dest, Splat, SplatLHS, DL);
2454 if (!SubVector || !Dest)
2455 return nullptr;
2456 auto *InsertVector =
2457 Builder.CreateInsertVector(Dest->getType(), Dest, SubVector, Idx);
2458 return replaceInstUsesWith(Inst, InsertVector);
2459 };
2460
2461 // If one operand is a constant splat and the other operand is a
2462 // `vector.insert` where both the destination and subvector are constant,
2463 // apply the operation to both the destination and subvector, returning a new
2464 // constant `vector.insert`. This helps constant folding for scalable vectors.
2465 if (Instruction *Folded = foldConstantsThroughSubVectorInsertSplat(
2466 /*MaybeSubVector=*/LHS, /*MaybeSplat=*/RHS, /*SplatLHS=*/false))
2467 return Folded;
2468 if (Instruction *Folded = foldConstantsThroughSubVectorInsertSplat(
2469 /*MaybeSubVector=*/RHS, /*MaybeSplat=*/LHS, /*SplatLHS=*/true))
2470 return Folded;
2471
2472 auto createBinOpReverse = [&](Value *X, Value *Y) {
2473 Value *V = Builder.CreateBinOp(Opcode, X, Y, Inst.getName());
2474 if (auto *BO = dyn_cast<BinaryOperator>(V))
2475 BO->copyIRFlags(&Inst);
2476 Module *M = Inst.getModule();
2478 M, Intrinsic::vector_reverse, V->getType());
2479 return CallInst::Create(F, V);
2480 };
2481
2482 // NOTE: Reverse shuffles don't require the speculative execution protection
2483 // below because they don't affect which lanes take part in the computation.
2484
2485 Value *V1, *V2;
2486 if (match(LHS, m_VecReverse(m_Value(V1)))) {
2487 // Op(rev(V1), rev(V2)) -> rev(Op(V1, V2))
2488 if (match(RHS, m_VecReverse(m_Value(V2))) &&
2489 (LHS->hasOneUse() || RHS->hasOneUse() ||
2490 (LHS == RHS && LHS->hasNUses(2))))
2491 return createBinOpReverse(V1, V2);
2492
2493 // Op(rev(V1), RHSSplat)) -> rev(Op(V1, RHSSplat))
2494 if (LHS->hasOneUse() && isSplatValue(RHS))
2495 return createBinOpReverse(V1, RHS);
2496 }
2497 // Op(LHSSplat, rev(V2)) -> rev(Op(LHSSplat, V2))
2498 else if (isSplatValue(LHS) && match(RHS, m_OneUse(m_VecReverse(m_Value(V2)))))
2499 return createBinOpReverse(LHS, V2);
2500
2501 auto createBinOpVPReverse = [&](Value *X, Value *Y, Value *EVL) {
2502 Value *V = Builder.CreateBinOp(Opcode, X, Y, Inst.getName());
2503 if (auto *BO = dyn_cast<BinaryOperator>(V))
2504 BO->copyIRFlags(&Inst);
2505
2506 ElementCount EC = cast<VectorType>(V->getType())->getElementCount();
2507 Value *AllTrueMask = Builder.CreateVectorSplat(EC, Builder.getTrue());
2508 Module *M = Inst.getModule();
2510 M, Intrinsic::experimental_vp_reverse, V->getType());
2511 return CallInst::Create(F, {V, AllTrueMask, EVL});
2512 };
2513
2514 Value *EVL;
2516 m_Value(V1), m_AllOnes(), m_Value(EVL)))) {
2517 // Op(rev(V1), rev(V2)) -> rev(Op(V1, V2))
2519 m_Value(V2), m_AllOnes(), m_Specific(EVL))) &&
2520 (LHS->hasOneUse() || RHS->hasOneUse() ||
2521 (LHS == RHS && LHS->hasNUses(2))))
2522 return createBinOpVPReverse(V1, V2, EVL);
2523
2524 // Op(rev(V1), RHSSplat)) -> rev(Op(V1, RHSSplat))
2525 if (LHS->hasOneUse() && isSplatValue(RHS))
2526 return createBinOpVPReverse(V1, RHS, EVL);
2527 }
2528 // Op(LHSSplat, rev(V2)) -> rev(Op(LHSSplat, V2))
2529 else if (isSplatValue(LHS) &&
2531 m_Value(V2), m_AllOnes(), m_Value(EVL))))
2532 return createBinOpVPReverse(LHS, V2, EVL);
2533
2534 if (Instruction *Folded =
2536 return Folded;
2537 if (Instruction *Folded =
2539 return Folded;
2540
2541 // It may not be safe to reorder shuffles and things like div, urem, etc.
2542 // because we may trap when executing those ops on unknown vector elements.
2543 // See PR20059.
2545 return nullptr;
2546
2547 auto createBinOpShuffle = [&](Value *X, Value *Y, ArrayRef<int> M) {
2548 Value *XY = Builder.CreateBinOp(Opcode, X, Y);
2549 if (auto *BO = dyn_cast<BinaryOperator>(XY))
2550 BO->copyIRFlags(&Inst);
2551 return new ShuffleVectorInst(XY, M);
2552 };
2553
2554 // If both arguments of the binary operation are shuffles that use the same
2555 // mask and shuffle within a single vector, move the shuffle after the binop.
2556 ArrayRef<int> Mask;
2557 if (match(LHS, m_Shuffle(m_Value(V1), m_Poison(), m_Mask(Mask))) &&
2558 match(RHS, m_Shuffle(m_Value(V2), m_Poison(), m_SpecificMask(Mask))) &&
2559 Inst.getType() == V1->getType() && V1->getType() == V2->getType() &&
2560 (LHS->hasOneUse() || RHS->hasOneUse() || LHS == RHS)) {
2561 // Op(shuffle(V1, Mask), shuffle(V2, Mask)) -> shuffle(Op(V1, V2), Mask)
2562 return createBinOpShuffle(V1, V2, Mask);
2563 }
2564
2565 // If both arguments of a commutative binop are select-shuffles that use the
2566 // same mask with commuted operands, the shuffles are unnecessary.
2567 if (Inst.isCommutative() &&
2568 match(LHS, m_Shuffle(m_Value(V1), m_Value(V2), m_Mask(Mask))) &&
2569 match(RHS,
2571 auto *LShuf = cast<ShuffleVectorInst>(LHS);
2572 auto *RShuf = cast<ShuffleVectorInst>(RHS);
2573 // TODO: Allow shuffles that contain undefs in the mask?
2574 // That is legal, but it reduces undef knowledge.
2575 // TODO: Allow arbitrary shuffles by shuffling after binop?
2576 // That might be legal, but we have to deal with poison.
2577 if (LShuf->isSelect() &&
2578 !is_contained(LShuf->getShuffleMask(), PoisonMaskElem) &&
2579 RShuf->isSelect() &&
2580 !is_contained(RShuf->getShuffleMask(), PoisonMaskElem)) {
2581 // Example:
2582 // LHS = shuffle V1, V2, <0, 5, 6, 3>
2583 // RHS = shuffle V2, V1, <0, 5, 6, 3>
2584 // LHS + RHS --> (V10+V20, V21+V11, V22+V12, V13+V23) --> V1 + V2
2585 Instruction *NewBO = BinaryOperator::Create(Opcode, V1, V2);
2586 NewBO->copyIRFlags(&Inst);
2587 return NewBO;
2588 }
2589 }
2590
2591 // If one argument is a shuffle within one vector and the other is a constant,
2592 // try moving the shuffle after the binary operation. This canonicalization
2593 // intends to move shuffles closer to other shuffles and binops closer to
2594 // other binops, so they can be folded. It may also enable demanded elements
2595 // transforms.
2596 Constant *C;
2598 m_Mask(Mask))),
2599 m_ImmConstant(C)))) {
2600 assert(Inst.getType()->getScalarType() == V1->getType()->getScalarType() &&
2601 "Shuffle should not change scalar type");
2602
2603 bool ConstOp1 = isa<Constant>(RHS);
2604 if (Constant *NewC =
2605 unshuffleConstant(Mask, C, cast<VectorType>(V1->getType()))) {
2606 // For fixed vectors, lanes of NewC not used by the shuffle will be poison
2607 // which will cause UB for div/rem. Mask them with a safe constant.
2608 if (isa<FixedVectorType>(V1->getType()) && Inst.isIntDivRem())
2609 NewC = getSafeVectorConstantForBinop(Opcode, NewC, ConstOp1);
2610
2611 // Op(shuffle(V1, Mask), C) -> shuffle(Op(V1, NewC), Mask)
2612 // Op(C, shuffle(V1, Mask)) -> shuffle(Op(NewC, V1), Mask)
2613 Value *NewLHS = ConstOp1 ? V1 : NewC;
2614 Value *NewRHS = ConstOp1 ? NewC : V1;
2615 return createBinOpShuffle(NewLHS, NewRHS, Mask);
2616 }
2617 }
2618
2619 // Try to reassociate to sink a splat shuffle after a binary operation.
2620 if (Inst.isAssociative() && Inst.isCommutative()) {
2621 // Canonicalize shuffle operand as LHS.
2622 if (isa<ShuffleVectorInst>(RHS))
2623 std::swap(LHS, RHS);
2624
2625 Value *X;
2626 ArrayRef<int> MaskC;
2627 int SplatIndex;
2628 Value *Y, *OtherOp;
2629 if (!match(LHS,
2630 m_OneUse(m_Shuffle(m_Value(X), m_Undef(), m_Mask(MaskC)))) ||
2631 !match(MaskC, m_SplatOrPoisonMask(SplatIndex)) ||
2632 X->getType() != Inst.getType() ||
2633 !match(RHS, m_OneUse(m_BinOp(Opcode, m_Value(Y), m_Value(OtherOp)))))
2634 return nullptr;
2635
2636 // FIXME: This may not be safe if the analysis allows undef elements. By
2637 // moving 'Y' before the splat shuffle, we are implicitly assuming
2638 // that it is not undef/poison at the splat index.
2639 if (isSplatValue(OtherOp, SplatIndex)) {
2640 std::swap(Y, OtherOp);
2641 } else if (!isSplatValue(Y, SplatIndex)) {
2642 return nullptr;
2643 }
2644
2645 // X and Y are splatted values, so perform the binary operation on those
2646 // values followed by a splat followed by the 2nd binary operation:
2647 // bo (splat X), (bo Y, OtherOp) --> bo (splat (bo X, Y)), OtherOp
2648 Value *NewBO = Builder.CreateBinOp(Opcode, X, Y);
2649 SmallVector<int, 8> NewMask(MaskC.size(), SplatIndex);
2650 Value *NewSplat = Builder.CreateShuffleVector(NewBO, NewMask);
2651 Instruction *R = BinaryOperator::Create(Opcode, NewSplat, OtherOp);
2652
2653 // Intersect FMF on both new binops. Other (poison-generating) flags are
2654 // dropped to be safe.
2655 if (isa<FPMathOperator>(R)) {
2656 R->copyFastMathFlags(&Inst);
2657 R->andIRFlags(RHS);
2658 }
2659 if (auto *NewInstBO = dyn_cast<BinaryOperator>(NewBO))
2660 NewInstBO->copyIRFlags(R);
2661 return R;
2662 }
2663
2664 return nullptr;
2665}
2666
2667/// Try to narrow the width of a binop if at least 1 operand is an extend of
2668/// of a value. This requires a potentially expensive known bits check to make
2669/// sure the narrow op does not overflow.
2670Instruction *InstCombinerImpl::narrowMathIfNoOverflow(BinaryOperator &BO) {
2671 // We need at least one extended operand.
2672 Value *Op0 = BO.getOperand(0), *Op1 = BO.getOperand(1);
2673
2674 // If this is a sub, we swap the operands since we always want an extension
2675 // on the RHS. The LHS can be an extension or a constant.
2676 if (BO.getOpcode() == Instruction::Sub)
2677 std::swap(Op0, Op1);
2678
2679 Value *X;
2680 bool IsSext = match(Op0, m_SExt(m_Value(X)));
2681 if (!IsSext && !match(Op0, m_ZExt(m_Value(X))))
2682 return nullptr;
2683
2684 // If both operands are the same extension from the same source type and we
2685 // can eliminate at least one (hasOneUse), this might work.
2686 CastInst::CastOps CastOpc = IsSext ? Instruction::SExt : Instruction::ZExt;
2687 Value *Y;
2688 if (!(match(Op1, m_ZExtOrSExt(m_Value(Y))) && X->getType() == Y->getType() &&
2689 cast<Operator>(Op1)->getOpcode() == CastOpc &&
2690 (Op0->hasOneUse() || Op1->hasOneUse()))) {
2691 // If that did not match, see if we have a suitable constant operand.
2692 // Truncating and extending must produce the same constant.
2693 Constant *WideC;
2694 if (!Op0->hasOneUse() || !match(Op1, m_Constant(WideC)))
2695 return nullptr;
2696 Constant *NarrowC = getLosslessInvCast(WideC, X->getType(), CastOpc, DL);
2697 if (!NarrowC)
2698 return nullptr;
2699 Y = NarrowC;
2700 }
2701
2702 // Swap back now that we found our operands.
2703 if (BO.getOpcode() == Instruction::Sub)
2704 std::swap(X, Y);
2705
2706 // Both operands have narrow versions. Last step: the math must not overflow
2707 // in the narrow width.
2708 if (!willNotOverflow(BO.getOpcode(), X, Y, BO, IsSext))
2709 return nullptr;
2710
2711 // bo (ext X), (ext Y) --> ext (bo X, Y)
2712 // bo (ext X), C --> ext (bo X, C')
2713 Value *NarrowBO = Builder.CreateBinOp(BO.getOpcode(), X, Y, "narrow");
2714 if (auto *NewBinOp = dyn_cast<BinaryOperator>(NarrowBO)) {
2715 if (IsSext)
2716 NewBinOp->setHasNoSignedWrap();
2717 else
2718 NewBinOp->setHasNoUnsignedWrap();
2719 }
2720 return CastInst::Create(CastOpc, NarrowBO, BO.getType());
2721}
2722
2723/// Determine nowrap flags for (gep (gep p, x), y) to (gep p, (x + y))
2724/// transform.
2729
2730/// Thread a GEP operation with constant indices through the constant true/false
2731/// arms of a select.
2733 InstCombiner::BuilderTy &Builder) {
2734 if (!GEP.hasAllConstantIndices())
2735 return nullptr;
2736
2737 Instruction *Sel;
2738 Value *Cond;
2739 Constant *TrueC, *FalseC;
2740 if (!match(GEP.getPointerOperand(), m_Instruction(Sel)) ||
2741 !match(Sel,
2742 m_Select(m_Value(Cond), m_Constant(TrueC), m_Constant(FalseC))))
2743 return nullptr;
2744
2745 // gep (select Cond, TrueC, FalseC), IndexC --> select Cond, TrueC', FalseC'
2746 // Propagate 'inbounds' and metadata from existing instructions.
2747 // Note: using IRBuilder to create the constants for efficiency.
2748 SmallVector<Value *, 4> IndexC(GEP.indices());
2749 GEPNoWrapFlags NW = GEP.getNoWrapFlags();
2750 Type *Ty = GEP.getSourceElementType();
2751 Value *NewTrueC = Builder.CreateGEP(Ty, TrueC, IndexC, "", NW);
2752 Value *NewFalseC = Builder.CreateGEP(Ty, FalseC, IndexC, "", NW);
2753 return SelectInst::Create(Cond, NewTrueC, NewFalseC, "", nullptr, Sel);
2754}
2755
2756// Canonicalization:
2757// gep T, (gep i8, base, C1), (Index + C2) into
2758// gep T, (gep i8, base, C1 + C2 * sizeof(T)), Index
2760 GEPOperator *Src,
2761 InstCombinerImpl &IC) {
2762 if (GEP.getNumIndices() != 1)
2763 return nullptr;
2764 auto &DL = IC.getDataLayout();
2765 Value *Base;
2766 const APInt *C1;
2767 if (!match(Src, m_PtrAdd(m_Value(Base), m_APInt(C1))))
2768 return nullptr;
2769 Value *VarIndex;
2770 const APInt *C2;
2771 Type *PtrTy = Src->getType()->getScalarType();
2772 unsigned IndexSizeInBits = DL.getIndexTypeSizeInBits(PtrTy);
2773 if (!match(GEP.getOperand(1), m_AddLike(m_Value(VarIndex), m_APInt(C2))))
2774 return nullptr;
2775 if (C1->getBitWidth() != IndexSizeInBits ||
2776 C2->getBitWidth() != IndexSizeInBits)
2777 return nullptr;
2778 Type *BaseType = GEP.getSourceElementType();
2780 return nullptr;
2781 APInt TypeSize(IndexSizeInBits, DL.getTypeAllocSize(BaseType));
2782 APInt NewOffset = TypeSize * *C2 + *C1;
2783 if (NewOffset.isZero() ||
2784 (Src->hasOneUse() && GEP.getOperand(1)->hasOneUse())) {
2786 if (GEP.hasNoUnsignedWrap() &&
2787 cast<GEPOperator>(Src)->hasNoUnsignedWrap() &&
2788 match(GEP.getOperand(1), m_NUWAddLike(m_Value(), m_Value()))) {
2790 if (GEP.isInBounds() && cast<GEPOperator>(Src)->isInBounds())
2791 Flags |= GEPNoWrapFlags::inBounds();
2792 }
2793
2794 Value *GEPConst =
2795 IC.Builder.CreatePtrAdd(Base, IC.Builder.getInt(NewOffset), "", Flags);
2796 return GetElementPtrInst::Create(BaseType, GEPConst, VarIndex, Flags);
2797 }
2798
2799 return nullptr;
2800}
2801
2802/// Combine constant offsets separated by variable offsets.
2803/// ptradd (ptradd (ptradd p, C1), x), C2 -> ptradd (ptradd p, x), C1+C2
2805 InstCombinerImpl &IC) {
2806 if (!GEP.hasAllConstantIndices())
2807 return nullptr;
2808
2811 auto *InnerGEP = dyn_cast<GetElementPtrInst>(GEP.getPointerOperand());
2812 while (true) {
2813 if (!InnerGEP)
2814 return nullptr;
2815
2816 NW = NW.intersectForReassociate(InnerGEP->getNoWrapFlags());
2817 if (InnerGEP->hasAllConstantIndices())
2818 break;
2819
2820 if (!InnerGEP->hasOneUse())
2821 return nullptr;
2822
2823 Skipped.push_back(InnerGEP);
2824 InnerGEP = dyn_cast<GetElementPtrInst>(InnerGEP->getPointerOperand());
2825 }
2826
2827 // The two constant offset GEPs are directly adjacent: Let normal offset
2828 // merging handle it.
2829 if (Skipped.empty())
2830 return nullptr;
2831
2832 // FIXME: This one-use check is not strictly necessary. Consider relaxing it
2833 // if profitable.
2834 if (!InnerGEP->hasOneUse())
2835 return nullptr;
2836
2837 // Don't bother with vector splats.
2838 Type *Ty = GEP.getType();
2839 if (InnerGEP->getType() != Ty)
2840 return nullptr;
2841
2842 const DataLayout &DL = IC.getDataLayout();
2843 APInt Offset(DL.getIndexTypeSizeInBits(Ty), 0);
2844 if (!GEP.accumulateConstantOffset(DL, Offset) ||
2845 !InnerGEP->accumulateConstantOffset(DL, Offset))
2846 return nullptr;
2847
2848 IC.replaceOperand(*Skipped.back(), 0, InnerGEP->getPointerOperand());
2849 for (GetElementPtrInst *SkippedGEP : Skipped)
2850 SkippedGEP->setNoWrapFlags(NW);
2851
2852 return IC.replaceInstUsesWith(
2853 GEP,
2854 IC.Builder.CreatePtrAdd(Skipped.front(), IC.Builder.getInt(Offset), "",
2855 NW.intersectForOffsetAdd(GEP.getNoWrapFlags())));
2856}
2857
2859 GEPOperator *Src) {
2860 // Combine Indices - If the source pointer to this getelementptr instruction
2861 // is a getelementptr instruction with matching element type, combine the
2862 // indices of the two getelementptr instructions into a single instruction.
2863 if (!shouldMergeGEPs(*cast<GEPOperator>(&GEP), *Src))
2864 return nullptr;
2865
2866 if (auto *I = canonicalizeGEPOfConstGEPI8(GEP, Src, *this))
2867 return I;
2868
2869 if (auto *I = combineConstantOffsets(GEP, *this))
2870 return I;
2871
2872 if (Src->getResultElementType() != GEP.getSourceElementType())
2873 return nullptr;
2874
2875 // Fold chained GEP with constant base into single GEP:
2876 // gep i8, (gep i8, %base, C1), (select Cond, C2, C3)
2877 // -> gep i8, %base, (select Cond, C1+C2, C1+C3)
2878 if (Src->hasOneUse() && GEP.getNumIndices() == 1 &&
2879 Src->getNumIndices() == 1) {
2880 Value *SrcIdx = *Src->idx_begin();
2881 Value *GEPIdx = *GEP.idx_begin();
2882 const APInt *ConstOffset, *TrueVal, *FalseVal;
2883 Value *Cond;
2884
2885 if ((match(SrcIdx, m_APInt(ConstOffset)) &&
2886 match(GEPIdx,
2887 m_Select(m_Value(Cond), m_APInt(TrueVal), m_APInt(FalseVal)))) ||
2888 (match(GEPIdx, m_APInt(ConstOffset)) &&
2889 match(SrcIdx,
2890 m_Select(m_Value(Cond), m_APInt(TrueVal), m_APInt(FalseVal))))) {
2891 auto *Select = isa<SelectInst>(GEPIdx) ? cast<SelectInst>(GEPIdx)
2892 : cast<SelectInst>(SrcIdx);
2893
2894 // Make sure the select has only one use.
2895 if (!Select->hasOneUse())
2896 return nullptr;
2897
2898 if (TrueVal->getBitWidth() != ConstOffset->getBitWidth() ||
2899 FalseVal->getBitWidth() != ConstOffset->getBitWidth())
2900 return nullptr;
2901
2902 APInt NewTrueVal = *ConstOffset + *TrueVal;
2903 APInt NewFalseVal = *ConstOffset + *FalseVal;
2904 Constant *NewTrue = ConstantInt::get(Select->getType(), NewTrueVal);
2905 Constant *NewFalse = ConstantInt::get(Select->getType(), NewFalseVal);
2906 Value *NewSelect =
2907 Builder.CreateSelect(Cond, NewTrue, NewFalse, /*Name=*/"",
2908 /*MDFrom=*/Select);
2909 GEPNoWrapFlags Flags =
2911 return replaceInstUsesWith(GEP,
2912 Builder.CreateGEP(GEP.getResultElementType(),
2913 Src->getPointerOperand(),
2914 NewSelect, "", Flags));
2915 }
2916 }
2917
2918 // Find out whether the last index in the source GEP is a sequential idx.
2919 bool EndsWithSequential = false;
2920 for (gep_type_iterator I = gep_type_begin(*Src), E = gep_type_end(*Src);
2921 I != E; ++I)
2922 EndsWithSequential = I.isSequential();
2923 if (!EndsWithSequential)
2924 return nullptr;
2925
2926 // Replace: gep (gep %P, long B), long A, ...
2927 // With: T = long A+B; gep %P, T, ...
2928 Value *SO1 = Src->getOperand(Src->getNumOperands() - 1);
2929 Value *GO1 = GEP.getOperand(1);
2930
2931 // If they aren't the same type, then the input hasn't been processed
2932 // by the loop above yet (which canonicalizes sequential index types to
2933 // intptr_t). Just avoid transforming this until the input has been
2934 // normalized.
2935 if (SO1->getType() != GO1->getType())
2936 return nullptr;
2937
2938 Value *Sum =
2939 simplifyAddInst(GO1, SO1, false, false, SQ.getWithInstruction(&GEP));
2940 // Only do the combine when we are sure the cost after the
2941 // merge is never more than that before the merge.
2942 if (Sum == nullptr)
2943 return nullptr;
2944
2946 Indices.append(Src->op_begin() + 1, Src->op_end() - 1);
2947 Indices.push_back(Sum);
2948 Indices.append(GEP.op_begin() + 2, GEP.op_end());
2949
2950 // Don't create GEPs with more than one non-zero index.
2951 unsigned NumNonZeroIndices = count_if(Indices, [](Value *Idx) {
2952 auto *C = dyn_cast<Constant>(Idx);
2953 return !C || !C->isNullValue();
2954 });
2955 if (NumNonZeroIndices > 1)
2956 return nullptr;
2957
2958 return replaceInstUsesWith(
2959 GEP, Builder.CreateGEP(
2960 Src->getSourceElementType(), Src->getOperand(0), Indices, "",
2962}
2963
2966 bool &DoesConsume, unsigned Depth) {
2967 static Value *const NonNull = reinterpret_cast<Value *>(uintptr_t(1));
2968 // ~(~(X)) -> X.
2969 Value *A, *B;
2970 if (match(V, m_Not(m_Value(A)))) {
2971 DoesConsume = true;
2972 return A;
2973 }
2974
2975 Constant *C;
2976 // Constants can be considered to be not'ed values.
2977 if (match(V, m_ImmConstant(C)))
2978 return ConstantExpr::getNot(C);
2979
2981 return nullptr;
2982
2983 // The rest of the cases require that we invert all uses so don't bother
2984 // doing the analysis if we know we can't use the result.
2985 if (!WillInvertAllUses)
2986 return nullptr;
2987
2988 // Compares can be inverted if all of their uses are being modified to use
2989 // the ~V.
2990 if (auto *I = dyn_cast<CmpInst>(V)) {
2991 if (Builder != nullptr)
2992 return Builder->CreateCmp(I->getInversePredicate(), I->getOperand(0),
2993 I->getOperand(1));
2994 return NonNull;
2995 }
2996
2997 // If `V` is of the form `A + B` then `-1 - V` can be folded into
2998 // `(-1 - B) - A` if we are willing to invert all of the uses.
2999 if (match(V, m_Add(m_Value(A), m_Value(B)))) {
3000 if (auto *BV = getFreelyInvertedImpl(B, B->hasOneUse(), Builder,
3001 DoesConsume, Depth))
3002 return Builder ? Builder->CreateSub(BV, A) : NonNull;
3003 if (auto *AV = getFreelyInvertedImpl(A, A->hasOneUse(), Builder,
3004 DoesConsume, Depth))
3005 return Builder ? Builder->CreateSub(AV, B) : NonNull;
3006 return nullptr;
3007 }
3008
3009 // If `V` is of the form `A ^ ~B` then `~(A ^ ~B)` can be folded
3010 // into `A ^ B` if we are willing to invert all of the uses.
3011 if (match(V, m_Xor(m_Value(A), m_Value(B)))) {
3012 if (auto *BV = getFreelyInvertedImpl(B, B->hasOneUse(), Builder,
3013 DoesConsume, Depth))
3014 return Builder ? Builder->CreateXor(A, BV) : NonNull;
3015 if (auto *AV = getFreelyInvertedImpl(A, A->hasOneUse(), Builder,
3016 DoesConsume, Depth))
3017 return Builder ? Builder->CreateXor(AV, B) : NonNull;
3018 return nullptr;
3019 }
3020
3021 // If `V` is of the form `B - A` then `-1 - V` can be folded into
3022 // `A + (-1 - B)` if we are willing to invert all of the uses.
3023 if (match(V, m_Sub(m_Value(A), m_Value(B)))) {
3024 if (auto *AV = getFreelyInvertedImpl(A, A->hasOneUse(), Builder,
3025 DoesConsume, Depth))
3026 return Builder ? Builder->CreateAdd(AV, B) : NonNull;
3027 return nullptr;
3028 }
3029
3030 // If `V` is of the form `(~A) s>> B` then `~((~A) s>> B)` can be folded
3031 // into `A s>> B` if we are willing to invert all of the uses.
3032 if (match(V, m_AShr(m_Value(A), m_Value(B)))) {
3033 if (auto *AV = getFreelyInvertedImpl(A, A->hasOneUse(), Builder,
3034 DoesConsume, Depth))
3035 return Builder ? Builder->CreateAShr(AV, B) : NonNull;
3036 return nullptr;
3037 }
3038
3039 Value *Cond;
3040 // LogicOps are special in that we canonicalize them at the cost of an
3041 // instruction.
3042 bool IsSelect = match(V, m_Select(m_Value(Cond), m_Value(A), m_Value(B))) &&
3044 // Selects/min/max with invertible operands are freely invertible
3045 if (IsSelect || match(V, m_MaxOrMin(m_Value(A), m_Value(B)))) {
3046 bool LocalDoesConsume = DoesConsume;
3047 if (!getFreelyInvertedImpl(B, B->hasOneUse(), /*Builder*/ nullptr,
3048 LocalDoesConsume, Depth))
3049 return nullptr;
3050 if (Value *NotA = getFreelyInvertedImpl(A, A->hasOneUse(), Builder,
3051 LocalDoesConsume, Depth)) {
3052 DoesConsume = LocalDoesConsume;
3053 if (Builder != nullptr) {
3054 Value *NotB = getFreelyInvertedImpl(B, B->hasOneUse(), Builder,
3055 DoesConsume, Depth);
3056 assert(NotB != nullptr &&
3057 "Unable to build inverted value for known freely invertable op");
3058 if (auto *II = dyn_cast<IntrinsicInst>(V))
3059 return Builder->CreateBinaryIntrinsic(
3060 getInverseMinMaxIntrinsic(II->getIntrinsicID()), NotA, NotB);
3061 return Builder->CreateSelect(Cond, NotA, NotB, "",
3063 }
3064 return NonNull;
3065 }
3066 }
3067
3068 if (PHINode *PN = dyn_cast<PHINode>(V)) {
3069 bool LocalDoesConsume = DoesConsume;
3071 for (Use &U : PN->operands()) {
3072 BasicBlock *IncomingBlock = PN->getIncomingBlock(U);
3073 Value *NewIncomingVal = getFreelyInvertedImpl(
3074 U.get(), /*WillInvertAllUses=*/false,
3075 /*Builder=*/nullptr, LocalDoesConsume, MaxAnalysisRecursionDepth - 1);
3076 if (NewIncomingVal == nullptr)
3077 return nullptr;
3078 // Make sure that we can safely erase the original PHI node.
3079 if (NewIncomingVal == V)
3080 return nullptr;
3081 if (Builder != nullptr)
3082 IncomingValues.emplace_back(NewIncomingVal, IncomingBlock);
3083 }
3084
3085 DoesConsume = LocalDoesConsume;
3086 if (Builder != nullptr) {
3088 Builder->SetInsertPoint(PN);
3089 PHINode *NewPN =
3090 Builder->CreatePHI(PN->getType(), PN->getNumIncomingValues());
3091 for (auto [Val, Pred] : IncomingValues)
3092 NewPN->addIncoming(Val, Pred);
3093 return NewPN;
3094 }
3095 return NonNull;
3096 }
3097
3098 if (match(V, m_SExtLike(m_Value(A)))) {
3099 if (auto *AV = getFreelyInvertedImpl(A, A->hasOneUse(), Builder,
3100 DoesConsume, Depth))
3101 return Builder ? Builder->CreateSExt(AV, V->getType()) : NonNull;
3102 return nullptr;
3103 }
3104
3105 if (match(V, m_Trunc(m_Value(A)))) {
3106 if (auto *AV = getFreelyInvertedImpl(A, A->hasOneUse(), Builder,
3107 DoesConsume, Depth))
3108 return Builder ? Builder->CreateTrunc(AV, V->getType()) : NonNull;
3109 return nullptr;
3110 }
3111
3112 // De Morgan's Laws:
3113 // (~(A | B)) -> (~A & ~B)
3114 // (~(A & B)) -> (~A | ~B)
3115 auto TryInvertAndOrUsingDeMorgan = [&](Instruction::BinaryOps Opcode,
3116 bool IsLogical, Value *A,
3117 Value *B) -> Value * {
3118 bool LocalDoesConsume = DoesConsume;
3119 if (!getFreelyInvertedImpl(B, B->hasOneUse(), /*Builder=*/nullptr,
3120 LocalDoesConsume, Depth))
3121 return nullptr;
3122 if (auto *NotA = getFreelyInvertedImpl(A, A->hasOneUse(), Builder,
3123 LocalDoesConsume, Depth)) {
3124 auto *NotB = getFreelyInvertedImpl(B, B->hasOneUse(), Builder,
3125 LocalDoesConsume, Depth);
3126 DoesConsume = LocalDoesConsume;
3127 if (IsLogical)
3128 return Builder ? Builder->CreateLogicalOp(Opcode, NotA, NotB) : NonNull;
3129 return Builder ? Builder->CreateBinOp(Opcode, NotA, NotB) : NonNull;
3130 }
3131
3132 return nullptr;
3133 };
3134
3135 if (match(V, m_Or(m_Value(A), m_Value(B))))
3136 return TryInvertAndOrUsingDeMorgan(Instruction::And, /*IsLogical=*/false, A,
3137 B);
3138
3139 if (match(V, m_And(m_Value(A), m_Value(B))))
3140 return TryInvertAndOrUsingDeMorgan(Instruction::Or, /*IsLogical=*/false, A,
3141 B);
3142
3143 if (match(V, m_LogicalOr(m_Value(A), m_Value(B))))
3144 return TryInvertAndOrUsingDeMorgan(Instruction::And, /*IsLogical=*/true, A,
3145 B);
3146
3147 if (match(V, m_LogicalAnd(m_Value(A), m_Value(B))))
3148 return TryInvertAndOrUsingDeMorgan(Instruction::Or, /*IsLogical=*/true, A,
3149 B);
3150
3151 return nullptr;
3152}
3153
3154/// Return true if we should canonicalize the gep to an i8 ptradd.
3156 Value *PtrOp = GEP.getOperand(0);
3157 Type *GEPEltType = GEP.getSourceElementType();
3158 if (GEPEltType->isIntegerTy(8))
3159 return false;
3160
3161 // Canonicalize scalable GEPs to an explicit offset using the llvm.vscale
3162 // intrinsic. This has better support in BasicAA.
3163 if (GEPEltType->isScalableTy())
3164 return true;
3165
3166 // gep i32 p, mul(O, C) -> gep i8, p, mul(O, C*4) to fold the two multiplies
3167 // together.
3168 if (GEP.getNumIndices() == 1 &&
3169 match(GEP.getOperand(1),
3171 m_Shl(m_Value(), m_ConstantInt())))))
3172 return true;
3173
3174 // gep (gep %p, C1), %x, C2 is expanded so the two constants can
3175 // possibly be merged together.
3176 auto PtrOpGep = dyn_cast<GEPOperator>(PtrOp);
3177 return PtrOpGep && PtrOpGep->hasAllConstantIndices() &&
3178 any_of(GEP.indices(), [](Value *V) {
3179 const APInt *C;
3180 return match(V, m_APInt(C)) && !C->isZero();
3181 });
3182}
3183
3185 IRBuilderBase &Builder) {
3186 auto *Op1 = dyn_cast<GetElementPtrInst>(PN->getOperand(0));
3187 if (!Op1)
3188 return nullptr;
3189
3190 // Don't fold a GEP into itself through a PHI node. This can only happen
3191 // through the back-edge of a loop. Folding a GEP into itself means that
3192 // the value of the previous iteration needs to be stored in the meantime,
3193 // thus requiring an additional register variable to be live, but not
3194 // actually achieving anything (the GEP still needs to be executed once per
3195 // loop iteration).
3196 if (Op1 == &GEP)
3197 return nullptr;
3198 GEPNoWrapFlags NW = Op1->getNoWrapFlags();
3199
3200 int DI = -1;
3201
3202 for (auto I = PN->op_begin()+1, E = PN->op_end(); I !=E; ++I) {
3203 auto *Op2 = dyn_cast<GetElementPtrInst>(*I);
3204 if (!Op2 || Op1->getNumOperands() != Op2->getNumOperands() ||
3205 Op1->getSourceElementType() != Op2->getSourceElementType())
3206 return nullptr;
3207
3208 // As for Op1 above, don't try to fold a GEP into itself.
3209 if (Op2 == &GEP)
3210 return nullptr;
3211
3212 // Keep track of the type as we walk the GEP.
3213 Type *CurTy = nullptr;
3214
3215 for (unsigned J = 0, F = Op1->getNumOperands(); J != F; ++J) {
3216 if (Op1->getOperand(J)->getType() != Op2->getOperand(J)->getType())
3217 return nullptr;
3218
3219 if (Op1->getOperand(J) != Op2->getOperand(J)) {
3220 if (DI == -1) {
3221 // We have not seen any differences yet in the GEPs feeding the
3222 // PHI yet, so we record this one if it is allowed to be a
3223 // variable.
3224
3225 // The first two arguments can vary for any GEP, the rest have to be
3226 // static for struct slots
3227 if (J > 1) {
3228 assert(CurTy && "No current type?");
3229 if (CurTy->isStructTy())
3230 return nullptr;
3231 }
3232
3233 DI = J;
3234 } else {
3235 // The GEP is different by more than one input. While this could be
3236 // extended to support GEPs that vary by more than one variable it
3237 // doesn't make sense since it greatly increases the complexity and
3238 // would result in an R+R+R addressing mode which no backend
3239 // directly supports and would need to be broken into several
3240 // simpler instructions anyway.
3241 return nullptr;
3242 }
3243 }
3244
3245 // Sink down a layer of the type for the next iteration.
3246 if (J > 0) {
3247 if (J == 1) {
3248 CurTy = Op1->getSourceElementType();
3249 } else {
3250 CurTy =
3251 GetElementPtrInst::getTypeAtIndex(CurTy, Op1->getOperand(J));
3252 }
3253 }
3254 }
3255
3256 NW &= Op2->getNoWrapFlags();
3257 }
3258
3259 // If not all GEPs are identical we'll have to create a new PHI node.
3260 // Check that the old PHI node has only one use so that it will get
3261 // removed.
3262 if (DI != -1 && !PN->hasOneUse())
3263 return nullptr;
3264
3265 auto *NewGEP = cast<GetElementPtrInst>(Op1->clone());
3266 NewGEP->setNoWrapFlags(NW);
3267
3268 if (DI == -1) {
3269 // All the GEPs feeding the PHI are identical. Clone one down into our
3270 // BB so that it can be merged with the current GEP.
3271 } else {
3272 // All the GEPs feeding the PHI differ at a single offset. Clone a GEP
3273 // into the current block so it can be merged, and create a new PHI to
3274 // set that index.
3275 PHINode *NewPN;
3276 {
3277 IRBuilderBase::InsertPointGuard Guard(Builder);
3278 Builder.SetInsertPoint(PN);
3279 NewPN = Builder.CreatePHI(Op1->getOperand(DI)->getType(),
3280 PN->getNumOperands());
3281 }
3282
3283 for (auto &I : PN->operands())
3284 NewPN->addIncoming(cast<GEPOperator>(I)->getOperand(DI),
3285 PN->getIncomingBlock(I));
3286
3287 NewGEP->setOperand(DI, NewPN);
3288 }
3289
3290 NewGEP->insertBefore(*GEP.getParent(), GEP.getParent()->getFirstInsertionPt());
3291 return NewGEP;
3292}
3293
3295 Value *PtrOp = GEP.getOperand(0);
3296 SmallVector<Value *, 8> Indices(GEP.indices());
3297 Type *GEPType = GEP.getType();
3298 Type *GEPEltType = GEP.getSourceElementType();
3299 if (Value *V =
3300 simplifyGEPInst(GEPEltType, PtrOp, Indices, GEP.getNoWrapFlags(),
3301 SQ.getWithInstruction(&GEP)))
3302 return replaceInstUsesWith(GEP, V);
3303
3304 // For vector geps, use the generic demanded vector support.
3305 // Skip if GEP return type is scalable. The number of elements is unknown at
3306 // compile-time.
3307 if (auto *GEPFVTy = dyn_cast<FixedVectorType>(GEPType)) {
3308 auto VWidth = GEPFVTy->getNumElements();
3309 APInt PoisonElts(VWidth, 0);
3310 APInt AllOnesEltMask(APInt::getAllOnes(VWidth));
3311 if (Value *V = SimplifyDemandedVectorElts(&GEP, AllOnesEltMask,
3312 PoisonElts)) {
3313 if (V != &GEP)
3314 return replaceInstUsesWith(GEP, V);
3315 return &GEP;
3316 }
3317 }
3318
3319 // Eliminate unneeded casts for indices, and replace indices which displace
3320 // by multiples of a zero size type with zero.
3321 bool MadeChange = false;
3322
3323 // Index width may not be the same width as pointer width.
3324 // Data layout chooses the right type based on supported integer types.
3325 Type *NewScalarIndexTy =
3326 DL.getIndexType(GEP.getPointerOperandType()->getScalarType());
3327
3329 for (User::op_iterator I = GEP.op_begin() + 1, E = GEP.op_end(); I != E;
3330 ++I, ++GTI) {
3331 // Skip indices into struct types.
3332 if (GTI.isStruct())
3333 continue;
3334
3335 Type *IndexTy = (*I)->getType();
3336 Type *NewIndexType =
3337 IndexTy->isVectorTy()
3338 ? VectorType::get(NewScalarIndexTy,
3339 cast<VectorType>(IndexTy)->getElementCount())
3340 : NewScalarIndexTy;
3341
3342 // If the element type has zero size then any index over it is equivalent
3343 // to an index of zero, so replace it with zero if it is not zero already.
3344 Type *EltTy = GTI.getIndexedType();
3345 if (EltTy->isSized() && DL.getTypeAllocSize(EltTy).isZero())
3346 if (!isa<Constant>(*I) || !match(I->get(), m_Zero())) {
3347 *I = Constant::getNullValue(NewIndexType);
3348 MadeChange = true;
3349 }
3350
3351 if (IndexTy != NewIndexType) {
3352 // If we are using a wider index than needed for this platform, shrink
3353 // it to what we need. If narrower, sign-extend it to what we need.
3354 // This explicit cast can make subsequent optimizations more obvious.
3355 if (IndexTy->getScalarSizeInBits() <
3356 NewIndexType->getScalarSizeInBits()) {
3357 if (GEP.hasNoUnsignedWrap() && GEP.hasNoUnsignedSignedWrap())
3358 *I = Builder.CreateZExt(*I, NewIndexType, "", /*IsNonNeg=*/true);
3359 else
3360 *I = Builder.CreateSExt(*I, NewIndexType);
3361 } else {
3362 *I = Builder.CreateTrunc(*I, NewIndexType, "", GEP.hasNoUnsignedWrap(),
3363 GEP.hasNoUnsignedSignedWrap());
3364 }
3365 MadeChange = true;
3366 }
3367 }
3368 if (MadeChange)
3369 return &GEP;
3370
3371 // Canonicalize constant GEPs to i8 type.
3372 if (!GEPEltType->isIntegerTy(8) && GEP.hasAllConstantIndices()) {
3373 APInt Offset(DL.getIndexTypeSizeInBits(GEPType), 0);
3374 if (GEP.accumulateConstantOffset(DL, Offset))
3375 return replaceInstUsesWith(
3376 GEP, Builder.CreatePtrAdd(PtrOp, Builder.getInt(Offset), "",
3377 GEP.getNoWrapFlags()));
3378 }
3379
3381 Value *Offset = EmitGEPOffset(cast<GEPOperator>(&GEP));
3382 Value *NewGEP =
3383 Builder.CreatePtrAdd(PtrOp, Offset, "", GEP.getNoWrapFlags());
3384 return replaceInstUsesWith(GEP, NewGEP);
3385 }
3386
3387 // Strip trailing zero indices.
3388 auto *LastIdx = dyn_cast<Constant>(Indices.back());
3389 if (LastIdx && LastIdx->isNullValue() && !LastIdx->getType()->isVectorTy()) {
3390 return replaceInstUsesWith(
3391 GEP, Builder.CreateGEP(GEP.getSourceElementType(), PtrOp,
3392 drop_end(Indices), "", GEP.getNoWrapFlags()));
3393 }
3394
3395 // Strip leading zero indices.
3396 auto *FirstIdx = dyn_cast<Constant>(Indices.front());
3397 if (FirstIdx && FirstIdx->isNullValue() &&
3398 !FirstIdx->getType()->isVectorTy()) {
3400 ++GTI;
3401 if (!GTI.isStruct() && GTI.getSequentialElementStride(DL) ==
3402 DL.getTypeAllocSize(GTI.getIndexedType()))
3403 return replaceInstUsesWith(GEP, Builder.CreateGEP(GTI.getIndexedType(),
3404 GEP.getPointerOperand(),
3405 drop_begin(Indices), "",
3406 GEP.getNoWrapFlags()));
3407 }
3408
3409 // Scalarize vector operands; prefer splat-of-gep.as canonical form.
3410 // Note that this looses information about undef lanes; we run it after
3411 // demanded bits to partially mitigate that loss.
3412 if (GEPType->isVectorTy() && llvm::any_of(GEP.operands(), [](Value *Op) {
3413 return Op->getType()->isVectorTy() && getSplatValue(Op);
3414 })) {
3415 SmallVector<Value *> NewOps;
3416 for (auto &Op : GEP.operands()) {
3417 if (Op->getType()->isVectorTy())
3418 if (Value *Scalar = getSplatValue(Op)) {
3419 NewOps.push_back(Scalar);
3420 continue;
3421 }
3422 NewOps.push_back(Op);
3423 }
3424
3425 Value *Res = Builder.CreateGEP(GEP.getSourceElementType(), NewOps[0],
3426 ArrayRef(NewOps).drop_front(), GEP.getName(),
3427 GEP.getNoWrapFlags());
3428 if (!Res->getType()->isVectorTy()) {
3429 ElementCount EC = cast<VectorType>(GEPType)->getElementCount();
3430 Res = Builder.CreateVectorSplat(EC, Res);
3431 }
3432 return replaceInstUsesWith(GEP, Res);
3433 }
3434
3435 bool SeenNonZeroIndex = false;
3436 for (auto [IdxNum, Idx] : enumerate(Indices)) {
3437 // Ignore one leading zero index.
3438 auto *C = dyn_cast<Constant>(Idx);
3439 if (C && C->isNullValue() && IdxNum == 0)
3440 continue;
3441
3442 if (!SeenNonZeroIndex) {
3443 SeenNonZeroIndex = true;
3444 continue;
3445 }
3446
3447 // GEP has multiple non-zero indices: Split it.
3448 ArrayRef<Value *> FrontIndices = ArrayRef(Indices).take_front(IdxNum);
3449 Value *FrontGEP =
3450 Builder.CreateGEP(GEPEltType, PtrOp, FrontIndices,
3451 GEP.getName() + ".split", GEP.getNoWrapFlags());
3452
3453 SmallVector<Value *> BackIndices;
3454 BackIndices.push_back(Constant::getNullValue(NewScalarIndexTy));
3455 append_range(BackIndices, drop_begin(Indices, IdxNum));
3457 GetElementPtrInst::getIndexedType(GEPEltType, FrontIndices), FrontGEP,
3458 BackIndices, GEP.getNoWrapFlags());
3459 }
3460
3461 // Canonicalize gep %T to gep [sizeof(%T) x i8]:
3462 auto IsCanonicalType = [](Type *Ty) {
3463 if (auto *AT = dyn_cast<ArrayType>(Ty))
3464 Ty = AT->getElementType();
3465 return Ty->isIntegerTy(8);
3466 };
3467 if (Indices.size() == 1 && !IsCanonicalType(GEPEltType)) {
3468 TypeSize Scale = DL.getTypeAllocSize(GEPEltType);
3469 assert(!Scale.isScalable() && "Should have been handled earlier");
3470 Type *NewElemTy = Builder.getInt8Ty();
3471 if (Scale.getFixedValue() != 1)
3472 NewElemTy = ArrayType::get(NewElemTy, Scale.getFixedValue());
3473 GEP.setSourceElementType(NewElemTy);
3474 GEP.setResultElementType(NewElemTy);
3475 // Don't bother revisiting the GEP after this change.
3476 MadeIRChange = true;
3477 }
3478
3479 // Check to see if the inputs to the PHI node are getelementptr instructions.
3480 if (auto *PN = dyn_cast<PHINode>(PtrOp)) {
3481 if (Value *NewPtrOp = foldGEPOfPhi(GEP, PN, Builder))
3482 return replaceOperand(GEP, 0, NewPtrOp);
3483 }
3484
3485 if (auto *Src = dyn_cast<GEPOperator>(PtrOp))
3486 if (Instruction *I = visitGEPOfGEP(GEP, Src))
3487 return I;
3488
3489 if (GEP.getNumIndices() == 1) {
3490 unsigned AS = GEP.getPointerAddressSpace();
3491 if (GEP.getOperand(1)->getType()->getScalarSizeInBits() ==
3492 DL.getIndexSizeInBits(AS)) {
3493 uint64_t TyAllocSize = DL.getTypeAllocSize(GEPEltType).getFixedValue();
3494
3495 if (TyAllocSize == 1) {
3496 // Canonicalize (gep i8* X, (ptrtoint Y)-(ptrtoint X)) to (bitcast Y),
3497 // but only if the result pointer is only used as if it were an integer.
3498 // (The case where the underlying object is the same is handled by
3499 // InstSimplify.)
3500 Value *X = GEP.getPointerOperand();
3501 Value *Y;
3502 if (match(GEP.getOperand(1), m_Sub(m_PtrToIntOrAddr(m_Value(Y)),
3504 GEPType == Y->getType()) {
3505 bool HasNonAddressBits =
3506 DL.getAddressSizeInBits(AS) != DL.getPointerSizeInBits(AS);
3507 bool Changed = GEP.replaceUsesWithIf(Y, [&](Use &U) {
3508 return isa<PtrToAddrInst, ICmpInst>(U.getUser()) ||
3509 (!HasNonAddressBits && isa<PtrToIntInst>(U.getUser()));
3510 });
3511 return Changed ? &GEP : nullptr;
3512 }
3513 } else if (auto *ExactIns =
3514 dyn_cast<PossiblyExactOperator>(GEP.getOperand(1))) {
3515 // Canonicalize (gep T* X, V / sizeof(T)) to (gep i8* X, V)
3516 Value *V;
3517 if (ExactIns->isExact()) {
3518 if ((has_single_bit(TyAllocSize) &&
3519 match(GEP.getOperand(1),
3520 m_Shr(m_Value(V),
3521 m_SpecificInt(countr_zero(TyAllocSize))))) ||
3522 match(GEP.getOperand(1),
3523 m_IDiv(m_Value(V), m_SpecificInt(TyAllocSize)))) {
3524 return GetElementPtrInst::Create(Builder.getInt8Ty(),
3525 GEP.getPointerOperand(), V,
3526 GEP.getNoWrapFlags());
3527 }
3528 }
3529 if (ExactIns->isExact() && ExactIns->hasOneUse()) {
3530 // Try to canonicalize non-i8 element type to i8 if the index is an
3531 // exact instruction. If the index is an exact instruction (div/shr)
3532 // with a constant RHS, we can fold the non-i8 element scale into the
3533 // div/shr (similiar to the mul case, just inverted).
3534 const APInt *C;
3535 std::optional<APInt> NewC;
3536 if (has_single_bit(TyAllocSize) &&
3537 match(ExactIns, m_Shr(m_Value(V), m_APInt(C))) &&
3538 C->uge(countr_zero(TyAllocSize)))
3539 NewC = *C - countr_zero(TyAllocSize);
3540 else if (match(ExactIns, m_UDiv(m_Value(V), m_APInt(C)))) {
3541 APInt Quot;
3542 uint64_t Rem;
3543 APInt::udivrem(*C, TyAllocSize, Quot, Rem);
3544 if (Rem == 0)
3545 NewC = Quot;
3546 } else if (match(ExactIns, m_SDiv(m_Value(V), m_APInt(C)))) {
3547 APInt Quot;
3548 int64_t Rem;
3549 APInt::sdivrem(*C, TyAllocSize, Quot, Rem);
3550 // For sdiv we need to make sure we arent creating INT_MIN / -1.
3551 if (!Quot.isAllOnes() && Rem == 0)
3552 NewC = Quot;
3553 }
3554
3555 if (NewC.has_value()) {
3556 Value *NewOp = Builder.CreateExactBinOp(
3557 static_cast<Instruction::BinaryOps>(ExactIns->getOpcode()), V,
3558 ConstantInt::get(V->getType(), *NewC), /*IsExact=*/true);
3559 return GetElementPtrInst::Create(Builder.getInt8Ty(),
3560 GEP.getPointerOperand(), NewOp,
3561 GEP.getNoWrapFlags());
3562 }
3563 }
3564 }
3565 }
3566 }
3567 // We do not handle pointer-vector geps here.
3568 if (GEPType->isVectorTy())
3569 return nullptr;
3570
3571 if (!GEP.isInBounds()) {
3572 unsigned IdxWidth =
3573 DL.getIndexSizeInBits(PtrOp->getType()->getPointerAddressSpace());
3574 APInt BasePtrOffset(IdxWidth, 0);
3575 Value *UnderlyingPtrOp =
3576 PtrOp->stripAndAccumulateInBoundsConstantOffsets(DL, BasePtrOffset);
3577 bool CanBeNull;
3578 uint64_t DerefBytes = UnderlyingPtrOp->getPointerDereferenceableBytes(
3579 DL, CanBeNull, /*CanBeFreed=*/nullptr);
3580 // We can ignore CanBeFreed here, because inbounds is explicitly allowed to
3581 // refer to a deallocated object.
3582 if (!CanBeNull && DerefBytes != 0) {
3583 if (GEP.accumulateConstantOffset(DL, BasePtrOffset) &&
3584 BasePtrOffset.isNonNegative()) {
3585 APInt AllocSize(IdxWidth, DerefBytes);
3586 if (BasePtrOffset.ule(AllocSize)) {
3588 GEP.getSourceElementType(), PtrOp, Indices, GEP.getName());
3589 }
3590 }
3591 }
3592 }
3593
3594 // nusw + nneg -> nuw
3595 if (GEP.hasNoUnsignedSignedWrap() && !GEP.hasNoUnsignedWrap() &&
3596 all_of(GEP.indices(), [&](Value *Idx) {
3597 return isKnownNonNegative(Idx, SQ.getWithInstruction(&GEP));
3598 })) {
3599 GEP.setNoWrapFlags(GEP.getNoWrapFlags() | GEPNoWrapFlags::noUnsignedWrap());
3600 return &GEP;
3601 }
3602
3603 // These rewrites are trying to preserve inbounds/nuw attributes. So we want
3604 // to do this after having tried to derive "nuw" above.
3605 if (GEP.getNumIndices() == 1) {
3606 // Given (gep p, x+y) we want to determine the common nowrap flags for both
3607 // geps if transforming into (gep (gep p, x), y).
3608 auto GetPreservedNoWrapFlags = [&](bool AddIsNUW) {
3609 // We can preserve both "inbounds nuw", "nusw nuw" and "nuw" if we know
3610 // that x + y does not have unsigned wrap.
3611 if (GEP.hasNoUnsignedWrap() && AddIsNUW)
3612 return GEP.getNoWrapFlags();
3613 return GEPNoWrapFlags::none();
3614 };
3615
3616 // Try to replace ADD + GEP with GEP + GEP.
3617 Value *Idx1, *Idx2;
3618 if (match(GEP.getOperand(1),
3619 m_OneUse(m_AddLike(m_Value(Idx1), m_Value(Idx2))))) {
3620 // %idx = add i64 %idx1, %idx2
3621 // %gep = getelementptr i32, ptr %ptr, i64 %idx
3622 // as:
3623 // %newptr = getelementptr i32, ptr %ptr, i64 %idx1
3624 // %newgep = getelementptr i32, ptr %newptr, i64 %idx2
3625 bool NUW = match(GEP.getOperand(1), m_NUWAddLike(m_Value(), m_Value()));
3626 GEPNoWrapFlags NWFlags = GetPreservedNoWrapFlags(NUW);
3627 auto *NewPtr =
3628 Builder.CreateGEP(GEP.getSourceElementType(), GEP.getPointerOperand(),
3629 Idx1, "", NWFlags);
3630 return replaceInstUsesWith(GEP,
3631 Builder.CreateGEP(GEP.getSourceElementType(),
3632 NewPtr, Idx2, "", NWFlags));
3633 }
3634 ConstantInt *C;
3635 if (match(GEP.getOperand(1), m_OneUse(m_SExtLike(m_OneUse(m_NSWAddLike(
3636 m_Value(Idx1), m_ConstantInt(C))))))) {
3637 // %add = add nsw i32 %idx1, idx2
3638 // %sidx = sext i32 %add to i64
3639 // %gep = getelementptr i32, ptr %ptr, i64 %sidx
3640 // as:
3641 // %newptr = getelementptr i32, ptr %ptr, i32 %idx1
3642 // %newgep = getelementptr i32, ptr %newptr, i32 idx2
3643 bool NUW = match(GEP.getOperand(1),
3645 GEPNoWrapFlags NWFlags = GetPreservedNoWrapFlags(NUW);
3646 auto *NewPtr = Builder.CreateGEP(
3647 GEP.getSourceElementType(), GEP.getPointerOperand(),
3648 Builder.CreateSExt(Idx1, GEP.getOperand(1)->getType()), "", NWFlags);
3649 return replaceInstUsesWith(
3650 GEP,
3651 Builder.CreateGEP(GEP.getSourceElementType(), NewPtr,
3652 Builder.CreateSExt(C, GEP.getOperand(1)->getType()),
3653 "", NWFlags));
3654 }
3655 }
3656
3658 return R;
3659
3660 // srem -> (and/urem) for inbounds+nuw GEP
3661 if (Indices.size() == 1 && GEP.isInBounds() && GEP.hasNoUnsignedWrap()) {
3662 Value *X, *Y;
3663
3664 // Match: idx = srem X, Y -- where Y is a power-of-two value.
3665 if (match(Indices[0], m_OneUse(m_SRem(m_Value(X), m_Value(Y)))) &&
3666 isKnownToBeAPowerOfTwo(Y, /*OrZero=*/true, &GEP)) {
3667 // If GEP is inbounds+nuw, the offset cannot be negative
3668 // -> srem by power-of-two can be treated as urem,
3669 // and urem by power-of-two folds to 'and' later.
3670 // OrZero=true is fine here because division by zero is UB.
3671 Instruction *OldIdxI = cast<Instruction>(Indices[0]);
3672 Value *NewIdx = Builder.CreateURem(X, Y, OldIdxI->getName());
3673
3674 return GetElementPtrInst::Create(GEPEltType, PtrOp, {NewIdx},
3675 GEP.getNoWrapFlags());
3676 }
3677 }
3678
3679 return nullptr;
3680}
3681
3683 Instruction *AI) {
3685 return true;
3686 if (auto *LI = dyn_cast<LoadInst>(V))
3687 return isa<GlobalVariable>(LI->getPointerOperand());
3688 // Two distinct allocations will never be equal.
3689 return isAllocLikeFn(V, &TLI) && V != AI;
3690}
3691
3692/// Given a call CB which uses an address UsedV, return true if we can prove the
3693/// call's only possible effect is storing to V.
3694static bool isRemovableWrite(CallBase &CB, Value *UsedV,
3695 const TargetLibraryInfo &TLI) {
3696 if (!CB.use_empty())
3697 // TODO: add recursion if returned attribute is present
3698 return false;
3699
3700 if (CB.isTerminator())
3701 // TODO: remove implementation restriction
3702 return false;
3703
3704 if (!CB.willReturn() || !CB.doesNotThrow())
3705 return false;
3706
3707 // If the only possible side effect of the call is writing to the alloca,
3708 // and the result isn't used, we can safely remove any reads implied by the
3709 // call including those which might read the alloca itself.
3710 std::optional<MemoryLocation> Dest = MemoryLocation::getForDest(&CB, TLI);
3711 return Dest && Dest->Ptr == UsedV;
3712}
3713
3714static std::optional<ModRefInfo>
3716 const TargetLibraryInfo &TLI, bool KnowInit,
3717 unsigned MaxUsers) {
3719 const std::optional<StringRef> Family = getAllocationFamily(AI, &TLI);
3720 Worklist.push_back(AI);
3722
3723 do {
3724 Instruction *PI = Worklist.pop_back_val();
3725 for (User *U : PI->users()) {
3727 if (Users.size() >= MaxUsers)
3728 return std::nullopt;
3729 switch (I->getOpcode()) {
3730 default:
3731 // Give up the moment we see something we can't handle.
3732 return std::nullopt;
3733
3734 case Instruction::AddrSpaceCast:
3735 case Instruction::BitCast:
3736 case Instruction::GetElementPtr:
3737 Users.emplace_back(I);
3738 Worklist.push_back(I);
3739 continue;
3740
3741 case Instruction::ICmp: {
3742 ICmpInst *ICI = cast<ICmpInst>(I);
3743 // We can fold eq/ne comparisons with null to false/true, respectively.
3744 // We also fold comparisons in some conditions provided the alloc has
3745 // not escaped (see isNeverEqualToUnescapedAlloc).
3746 if (!ICI->isEquality())
3747 return std::nullopt;
3748 unsigned OtherIndex = (ICI->getOperand(0) == PI) ? 1 : 0;
3749 if (!isNeverEqualToUnescapedAlloc(ICI->getOperand(OtherIndex), TLI, AI))
3750 return std::nullopt;
3751
3752 // Do not fold compares to aligned_alloc calls, as they may have to
3753 // return null in case the required alignment cannot be satisfied,
3754 // unless we can prove that both alignment and size are valid.
3755 auto AlignmentAndSizeKnownValid = [](CallBase *CB) {
3756 // Check if alignment and size of a call to aligned_alloc is valid,
3757 // that is alignment is a power-of-2 and the size is a multiple of the
3758 // alignment.
3759 const APInt *Alignment;
3760 const APInt *Size;
3761 return match(CB->getArgOperand(0), m_APInt(Alignment)) &&
3762 match(CB->getArgOperand(1), m_APInt(Size)) &&
3763 Alignment->isPowerOf2() && Size->urem(*Alignment).isZero();
3764 };
3765 auto *CB = dyn_cast<CallBase>(AI);
3766 if (CB &&
3767 TLI.getLibFunc(*CB->getCalledFunction()) == LibFunc_aligned_alloc &&
3768 TLI.has(LibFunc_aligned_alloc) && !AlignmentAndSizeKnownValid(CB))
3769 return std::nullopt;
3770 Users.emplace_back(I);
3771 continue;
3772 }
3773
3774 case Instruction::Call:
3775 // Ignore no-op and store intrinsics.
3777 switch (II->getIntrinsicID()) {
3778 default:
3779 return std::nullopt;
3780
3781 case Intrinsic::memmove:
3782 case Intrinsic::memcpy:
3783 case Intrinsic::memset: {
3785 if (MI->isVolatile())
3786 return std::nullopt;
3787 // Note: this could also be ModRef, but we can still interpret that
3788 // as just Mod in that case.
3789 ModRefInfo NewAccess =
3790 MI->getRawDest() == PI ? ModRefInfo::Mod : ModRefInfo::Ref;
3791 if ((Access & ~NewAccess) != ModRefInfo::NoModRef)
3792 return std::nullopt;
3793 Access |= NewAccess;
3794 [[fallthrough]];
3795 }
3796 case Intrinsic::assume:
3797 case Intrinsic::invariant_start:
3798 case Intrinsic::invariant_end:
3799 case Intrinsic::lifetime_start:
3800 case Intrinsic::lifetime_end:
3801 case Intrinsic::objectsize:
3802 Users.emplace_back(I);
3803 continue;
3804 case Intrinsic::launder_invariant_group:
3805 Users.emplace_back(I);
3806 Worklist.push_back(I);
3807 continue;
3808 }
3809 }
3810
3811 if (Family && getFreedOperand(cast<CallBase>(I), &TLI) == PI &&
3812 getAllocationFamily(I, &TLI) == Family) {
3813 Users.emplace_back(I);
3814 continue;
3815 }
3816
3817 if (Family && getReallocatedOperand(cast<CallBase>(I)) == PI &&
3818 getAllocationFamily(I, &TLI) == Family) {
3819 Users.emplace_back(I);
3820 Worklist.push_back(I);
3821 continue;
3822 }
3823
3824 if (!isRefSet(Access) &&
3825 isRemovableWrite(*cast<CallBase>(I), PI, TLI)) {
3827 Users.emplace_back(I);
3828 continue;
3829 }
3830
3831 return std::nullopt;
3832
3833 case Instruction::Store: {
3835 if (SI->isVolatile() || SI->getPointerOperand() != PI)
3836 return std::nullopt;
3837 if (isRefSet(Access))
3838 return std::nullopt;
3840 Users.emplace_back(I);
3841 continue;
3842 }
3843
3844 case Instruction::Load: {
3845 LoadInst *LI = cast<LoadInst>(I);
3846 if (LI->isVolatile() || LI->getPointerOperand() != PI)
3847 return std::nullopt;
3848 if (isModSet(Access))
3849 return std::nullopt;
3851 Users.emplace_back(I);
3852 continue;
3853 }
3854 }
3855 llvm_unreachable("missing a return?");
3856 }
3857 } while (!Worklist.empty());
3858
3860 return Access;
3861}
3862
3865
3866 // If we have a malloc call which is only used in any amount of comparisons to
3867 // null and free calls, delete the calls and replace the comparisons with true
3868 // or false as appropriate.
3869
3870 // This is based on the principle that we can substitute our own allocation
3871 // function (which will never return null) rather than knowledge of the
3872 // specific function being called. In some sense this can change the permitted
3873 // outputs of a program (when we convert a malloc to an alloca, the fact that
3874 // the allocation is now on the stack is potentially visible, for example),
3875 // but we believe in a permissible manner.
3876 //
3877 // Collect into Instruction* first to avoid expensive WeakTrackingVH
3878 // register/unregister overhead; convert to WeakTrackingVH only when the
3879 // site is actually removable.
3881
3882 // If we are removing an alloca with a dbg.declare, insert dbg.value calls
3883 // before each store.
3885 std::unique_ptr<DIBuilder> DIB;
3886 if (isa<AllocaInst>(MI)) {
3887 findDbgUsers(&MI, DVRs);
3888 DIB.reset(new DIBuilder(*MI.getModule(), /*AllowUnresolved=*/false));
3889 }
3890
3891 // Determine what getInitialValueOfAllocation would return without actually
3892 // allocating the result.
3893 bool KnowInitUndef = false;
3894 bool KnowInitZero = false;
3895 Constant *Init =
3897 if (Init) {
3898 if (isa<UndefValue>(Init))
3899 KnowInitUndef = true;
3900 else if (Init->isNullValue())
3901 KnowInitZero = true;
3902 }
3903 // The various sanitizers don't actually return undef memory, but rather
3904 // memory initialized with special forms of runtime poison
3905 auto &F = *MI.getFunction();
3906 if (F.hasFnAttribute(Attribute::SanitizeMemory) ||
3907 F.hasFnAttribute(Attribute::SanitizeAddress))
3908 KnowInitUndef = false;
3909
3910 auto Removable =
3911 isAllocSiteRemovable(&MI, RawUsers, TLI, KnowInitZero | KnowInitUndef,
3912 CLOpts.max_allocsite_removable_users);
3913 if (Removable) {
3914 SmallVector<WeakTrackingVH, 64> Users(RawUsers.begin(), RawUsers.end());
3915 for (WeakTrackingVH &User : Users) {
3916 // Lowering all @llvm.objectsize and MTI calls first because they may use
3917 // a bitcast/GEP of the alloca we are removing.
3918 if (!User)
3919 continue;
3920
3922
3924 if (II->getIntrinsicID() == Intrinsic::objectsize) {
3925 SmallVector<Instruction *> InsertedInstructions;
3926 Value *Result = lowerObjectSizeCall(
3927 II, DL, &TLI, AA, /*MustSucceed=*/true, &InsertedInstructions);
3928 for (Instruction *Inserted : InsertedInstructions)
3929 Worklist.add(Inserted);
3930 replaceInstUsesWith(*I, Result);
3932 User = nullptr; // Skip examining in the next loop.
3933 continue;
3934 }
3935 if (auto *MTI = dyn_cast<MemTransferInst>(I)) {
3936 if (KnowInitZero && isRefSet(*Removable)) {
3938 Builder.SetInsertPoint(MTI);
3939 auto *M = Builder.CreateMemSet(
3940 MTI->getRawDest(),
3941 ConstantInt::get(Type::getInt8Ty(MI.getContext()), 0),
3942 MTI->getLength(), MTI->getDestAlign());
3943 M->copyMetadata(*MTI);
3944 }
3945 }
3946 }
3947 }
3948 for (WeakTrackingVH &User : Users) {
3949 if (!User)
3950 continue;
3951
3953
3954 if (ICmpInst *C = dyn_cast<ICmpInst>(I)) {
3956 *C, ConstantInt::get(C->getType(), C->isFalseWhenEqual()));
3957 } else if (auto *SI = dyn_cast<StoreInst>(I)) {
3958 for (auto *DVR : DVRs)
3959 if (DVR->isAddressOfVariable())
3961 } else {
3962 // Casts, GEP, or anything else: we're about to delete this instruction,
3963 // so it can not have any valid uses.
3965 if (isa<LoadInst>(I)) {
3966 assert(KnowInitZero || KnowInitUndef);
3967 Replace = KnowInitUndef ? UndefValue::get(I->getType())
3968 : Constant::getNullValue(I->getType());
3969 } else
3970 Replace = PoisonValue::get(I->getType());
3972 }
3974 }
3975
3977 // Replace invoke with a NOP intrinsic to maintain the original CFG
3978 Module *M = II->getModule();
3979 Function *F = Intrinsic::getOrInsertDeclaration(M, Intrinsic::donothing);
3980 auto *NewII = InvokeInst::Create(
3981 F, II->getNormalDest(), II->getUnwindDest(), {}, "", II->getParent());
3982 NewII->setDebugLoc(II->getDebugLoc());
3983 }
3984
3985 // Remove debug intrinsics which describe the value contained within the
3986 // alloca. In addition to removing dbg.{declare,addr} which simply point to
3987 // the alloca, remove dbg.value(<alloca>, ..., DW_OP_deref)'s as well, e.g.:
3988 //
3989 // ```
3990 // define void @foo(i32 %0) {
3991 // %a = alloca i32 ; Deleted.
3992 // store i32 %0, i32* %a
3993 // dbg.value(i32 %0, "arg0") ; Not deleted.
3994 // dbg.value(i32* %a, "arg0", DW_OP_deref) ; Deleted.
3995 // call void @trivially_inlinable_no_op(i32* %a)
3996 // ret void
3997 // }
3998 // ```
3999 //
4000 // This may not be required if we stop describing the contents of allocas
4001 // using dbg.value(<alloca>, ..., DW_OP_deref), but we currently do this in
4002 // the LowerDbgDeclare utility.
4003 //
4004 // If there is a dead store to `%a` in @trivially_inlinable_no_op, the
4005 // "arg0" dbg.value may be stale after the call. However, failing to remove
4006 // the DW_OP_deref dbg.value causes large gaps in location coverage.
4007 //
4008 // FIXME: the Assignment Tracking project has now likely made this
4009 // redundant (and it's sometimes harmful).
4010 for (auto *DVR : DVRs)
4011 if (DVR->isAddressOfVariable() || DVR->getExpression()->startsWithDeref())
4012 DVR->eraseFromParent();
4013
4014 return eraseInstFromFunction(MI);
4015 }
4016 return nullptr;
4017}
4018
4019/// Move the call to free before a NULL test.
4020///
4021/// Check if this free is accessed after its argument has been test
4022/// against NULL (property 0).
4023/// If yes, it is legal to move this call in its predecessor block.
4024///
4025/// The move is performed only if the block containing the call to free
4026/// will be removed, i.e.:
4027/// 1. it has only one predecessor P, and P has two successors
4028/// 2. it contains the call, noops, and an unconditional branch
4029/// 3. its successor is the same as its predecessor's successor
4030///
4031/// The profitability is out-of concern here and this function should
4032/// be called only if the caller knows this transformation would be
4033/// profitable (e.g., for code size).
4035 const DataLayout &DL) {
4036 Value *Op = FI.getArgOperand(0);
4037 BasicBlock *FreeInstrBB = FI.getParent();
4038 BasicBlock *PredBB = FreeInstrBB->getSinglePredecessor();
4039
4040 // Validate part of constraint #1: Only one predecessor
4041 // FIXME: We can extend the number of predecessor, but in that case, we
4042 // would duplicate the call to free in each predecessor and it may
4043 // not be profitable even for code size.
4044 if (!PredBB)
4045 return nullptr;
4046
4047 // Validate constraint #2: Does this block contains only the call to
4048 // free, noops, and an unconditional branch?
4049 BasicBlock *SuccBB;
4050 Instruction *FreeInstrBBTerminator = FreeInstrBB->getTerminator();
4051 if (!match(FreeInstrBBTerminator, m_UnconditionalBr(SuccBB)))
4052 return nullptr;
4053
4054 // If there are only 2 instructions in the block, at this point,
4055 // this is the call to free and unconditional.
4056 // If there are more than 2 instructions, check that they are noops
4057 // i.e., they won't hurt the performance of the generated code.
4058 if (FreeInstrBB->size() != 2) {
4059 for (const Instruction &Inst : *FreeInstrBB) {
4060 if (&Inst == &FI || &Inst == FreeInstrBBTerminator ||
4062 continue;
4063 auto *Cast = dyn_cast<CastInst>(&Inst);
4064 if (!Cast || !Cast->isNoopCast(DL))
4065 return nullptr;
4066 }
4067 }
4068 // Validate the rest of constraint #1 by matching on the pred branch.
4069 Instruction *TI = PredBB->getTerminator();
4070 BasicBlock *TrueBB, *FalseBB;
4071 CmpPredicate Pred;
4072 if (!match(TI, m_Br(m_ICmp(Pred,
4074 m_Specific(Op->stripPointerCasts())),
4075 m_Zero()),
4076 TrueBB, FalseBB)))
4077 return nullptr;
4078 if (Pred != ICmpInst::ICMP_EQ && Pred != ICmpInst::ICMP_NE)
4079 return nullptr;
4080
4081 // Validate constraint #3: Ensure the null case just falls through.
4082 if (SuccBB != (Pred == ICmpInst::ICMP_EQ ? TrueBB : FalseBB))
4083 return nullptr;
4084 assert(FreeInstrBB == (Pred == ICmpInst::ICMP_EQ ? FalseBB : TrueBB) &&
4085 "Broken CFG: missing edge from predecessor to successor");
4086
4087 // At this point, we know that everything in FreeInstrBB can be moved
4088 // before TI.
4089 for (Instruction &Instr : llvm::make_early_inc_range(*FreeInstrBB)) {
4090 if (&Instr == FreeInstrBBTerminator)
4091 break;
4092 Instr.moveBeforePreserving(TI->getIterator());
4093 }
4094 assert(FreeInstrBB->size() == 1 &&
4095 "Only the branch instruction should remain");
4096
4097 // Now that we've moved the call to free before the NULL check, we have to
4098 // remove any attributes on its parameter that imply it's non-null, because
4099 // those attributes might have only been valid because of the NULL check, and
4100 // we can get miscompiles if we keep them. This is conservative if non-null is
4101 // also implied by something other than the NULL check, but it's guaranteed to
4102 // be correct, and the conservativeness won't matter in practice, since the
4103 // attributes are irrelevant for the call to free itself and the pointer
4104 // shouldn't be used after the call.
4105 AttributeList Attrs = FI.getAttributes();
4106 Attrs = Attrs.removeParamAttribute(FI.getContext(), 0, Attribute::NonNull);
4107 Attribute Dereferenceable = Attrs.getParamAttr(0, Attribute::Dereferenceable);
4108 if (Dereferenceable.isValid()) {
4109 uint64_t Bytes = Dereferenceable.getDereferenceableBytes();
4110 Attrs = Attrs.removeParamAttribute(FI.getContext(), 0,
4111 Attribute::Dereferenceable);
4112 Attrs = Attrs.addDereferenceableOrNullParamAttr(FI.getContext(), 0, Bytes);
4113 }
4114 FI.setAttributes(Attrs);
4115
4116 return &FI;
4117}
4118
4120 // free undef -> unreachable.
4121 if (isa<UndefValue>(Op)) {
4122 // Leave a marker since we can't modify the CFG here.
4124 return eraseInstFromFunction(FI);
4125 }
4126
4127 // If we have 'free null' delete the instruction. This can happen in stl code
4128 // when lots of inlining happens.
4130 return eraseInstFromFunction(FI);
4131
4132 // If we had free(realloc(...)) with no intervening uses, then eliminate the
4133 // realloc() entirely.
4135 if (CI && CI->hasOneUse())
4136 if (Value *ReallocatedOp = getReallocatedOperand(CI))
4137 return eraseInstFromFunction(*replaceInstUsesWith(*CI, ReallocatedOp));
4138
4139 // If we optimize for code size, try to move the call to free before the null
4140 // test so that simplify cfg can remove the empty block and dead code
4141 // elimination the branch. I.e., helps to turn something like:
4142 // if (foo) free(foo);
4143 // into
4144 // free(foo);
4145 //
4146 // Note that we can only do this for 'free' and not for any flavor of
4147 // 'operator delete'; there is no 'operator delete' symbol for which we are
4148 // permitted to invent a call, even if we're passing in a null pointer.
4149 if (MinimizeSize) {
4150 if (TLI.getLibFunc(FI) == LibFunc_free && TLI.has(LibFunc_free))
4152 return I;
4153 }
4154
4155 return nullptr;
4156}
4157
4159 Value *RetVal = RI.getReturnValue();
4160 if (!RetVal)
4161 return nullptr;
4162
4163 Function *F = RI.getFunction();
4164 Type *RetTy = RetVal->getType();
4165 if (RetTy->isPointerTy()) {
4166 bool UseProvenance =
4167 F->getAttributes().getRetDereferenceableBytes() > 0 &&
4169 if (F->hasRetAttribute(Attribute::NonNull) || UseProvenance) {
4170 if (Value *V = simplifyNonNullOperand(RetVal, UseProvenance))
4171 return replaceOperand(RI, 0, V);
4172 }
4173 }
4174
4175 if (!AttributeFuncs::isNoFPClassCompatibleType(RetTy))
4176 return nullptr;
4177
4178 FPClassTest ReturnClass = F->getAttributes().getRetNoFPClass();
4179 if (ReturnClass == fcNone)
4180 return nullptr;
4181
4182 KnownFPClass KnownClass;
4183 if (SimplifyDemandedFPClass(&RI, 0, ~ReturnClass, KnownClass,
4184 SQ.getWithInstruction(&RI)))
4185 return &RI;
4186
4187 return nullptr;
4188}
4189
4190// WARNING: keep in sync with SimplifyCFGOpt::simplifyUnreachable()!
4192 // Try to remove the previous instruction if it must lead to unreachable.
4193 // This includes instructions like stores and "llvm.assume" that may not get
4194 // removed by simple dead code elimination.
4195 bool Changed = false;
4196 while (Instruction *Prev = I.getPrevNode()) {
4197 // While we theoretically can erase EH, that would result in a block that
4198 // used to start with an EH no longer starting with EH, which is invalid.
4199 // To make it valid, we'd need to fixup predecessors to no longer refer to
4200 // this block, but that changes CFG, which is not allowed in InstCombine.
4201 if (Prev->isEHPad())
4202 break; // Can not drop any more instructions. We're done here.
4203
4205 break; // Can not drop any more instructions. We're done here.
4206 // Otherwise, this instruction can be freely erased,
4207 // even if it is not side-effect free.
4208
4209 // A value may still have uses before we process it here (for example, in
4210 // another unreachable block), so convert those to poison.
4211 replaceInstUsesWith(*Prev, PoisonValue::get(Prev->getType()));
4212 eraseInstFromFunction(*Prev);
4213 Changed = true;
4214 }
4215 return Changed;
4216}
4217
4222
4224 // If this store is the second-to-last instruction in the basic block
4225 // (excluding debug info) and if the block ends with
4226 // an unconditional branch, try to move the store to the successor block.
4227
4228 auto GetLastSinkableStore = [](BasicBlock::iterator BBI) {
4229 BasicBlock::iterator FirstInstr = BBI->getParent()->begin();
4230 do {
4231 if (BBI != FirstInstr)
4232 --BBI;
4233 } while (BBI != FirstInstr && BBI->isDebugOrPseudoInst());
4234
4235 return dyn_cast<StoreInst>(BBI);
4236 };
4237
4238 if (StoreInst *SI = GetLastSinkableStore(BasicBlock::iterator(BI)))
4240 return &BI;
4241
4242 return nullptr;
4243}
4244
4247 if (!DeadEdges.insert({From, To}).second)
4248 return;
4249
4250 // Replace phi node operands in successor with poison.
4251 for (PHINode &PN : To->phis())
4252 for (Use &U : PN.incoming_values())
4253 if (PN.getIncomingBlock(U) == From && !isa<PoisonValue>(U)) {
4254 replaceUse(U, PoisonValue::get(PN.getType()));
4255 addToWorklist(&PN);
4256 MadeIRChange = true;
4257 }
4258
4259 Worklist.push_back(To);
4260}
4261
4262// Under the assumption that I is unreachable, remove it and following
4263// instructions. Changes are reported directly to MadeIRChange.
4266 BasicBlock *BB = I->getParent();
4267 for (Instruction &Inst : make_early_inc_range(
4268 make_range(std::next(BB->getTerminator()->getReverseIterator()),
4269 std::next(I->getReverseIterator())))) {
4270 if (!Inst.use_empty() && !Inst.getType()->isTokenTy()) {
4271 replaceInstUsesWith(Inst, PoisonValue::get(Inst.getType()));
4272 MadeIRChange = true;
4273 }
4274 if (Inst.isEHPad() || Inst.getType()->isTokenTy())
4275 continue;
4276 // RemoveDIs: erase debug-info on this instruction manually.
4277 Inst.dropDbgRecords();
4279 MadeIRChange = true;
4280 }
4281
4284 MadeIRChange = true;
4285 for (Value *V : Changed)
4287 }
4288
4289 // Handle potentially dead successors.
4290 for (BasicBlock *Succ : successors(BB))
4291 addDeadEdge(BB, Succ, Worklist);
4292}
4293
4296 while (!Worklist.empty()) {
4297 BasicBlock *BB = Worklist.pop_back_val();
4298 if (!all_of(predecessors(BB), [&](BasicBlock *Pred) {
4299 return DeadEdges.contains({Pred, BB}) || DT.dominates(BB, Pred);
4300 }))
4301 continue;
4302
4304 }
4305}
4306
4308 BasicBlock *LiveSucc) {
4310 for (BasicBlock *Succ : successors(BB)) {
4311 // The live successor isn't dead.
4312 if (Succ == LiveSucc)
4313 continue;
4314
4315 addDeadEdge(BB, Succ, Worklist);
4316 }
4317
4319}
4320
4322 // Change br (not X), label True, label False to: br X, label False, True
4323 Value *Cond = BI.getCondition();
4324 Value *X;
4325 if (match(Cond, m_Not(m_Value(X))) && !isa<Constant>(X)) {
4326 // Swap Destinations and condition...
4327 BI.swapSuccessors();
4328 if (BPI)
4329 BPI->swapSuccEdgesProbabilities(BI.getParent());
4330 return replaceOperand(BI, 0, X);
4331 }
4332
4333 // Canonicalize logical-and-with-invert as logical-or-with-invert.
4334 // This is done by inverting the condition and swapping successors:
4335 // br (X && !Y), T, F --> br !(X && !Y), F, T --> br (!X || Y), F, T
4336 Value *Y;
4337 if (isa<SelectInst>(Cond) &&
4338 match(Cond,
4340 Value *NotX = Builder.CreateNot(X, "not." + X->getName());
4341 Value *Or = Builder.CreateLogicalOr(NotX, Y);
4342
4343 // Set weights for the new OR select instruction too.
4344 if (auto *OrInst = dyn_cast<Instruction>(Or)) {
4345 if (auto *CondInst = dyn_cast<Instruction>(Cond)) {
4346 SmallVector<uint32_t> Weights;
4347 if (extractBranchWeights(*CondInst, Weights)) {
4348 assert(Weights.size() == 2 && "Unexpected number of branch weights!");
4349 std::swap(Weights[0], Weights[1]);
4350 setBranchWeights(*OrInst, Weights, /*IsExpected=*/false);
4351 }
4352 }
4353 }
4354 BI.swapSuccessors();
4355 if (BPI)
4356 BPI->swapSuccEdgesProbabilities(BI.getParent());
4357 return replaceOperand(BI, 0, Or);
4358 }
4359
4360 // If the condition is irrelevant, remove the use so that other
4361 // transforms on the condition become more effective.
4362 if (!isa<ConstantInt>(Cond) && BI.getSuccessor(0) == BI.getSuccessor(1))
4363 return replaceOperand(BI, 0, ConstantInt::getFalse(Cond->getType()));
4364
4365 // Canonicalize, for example, fcmp_one -> fcmp_oeq.
4366 CmpPredicate Pred;
4367 if (match(Cond, m_OneUse(m_FCmp(Pred, m_Value(), m_Value()))) &&
4368 !isCanonicalPredicate(Pred)) {
4369 // Swap destinations and condition.
4370 auto *Cmp = cast<CmpInst>(Cond);
4371 Cmp->setPredicate(CmpInst::getInversePredicate(Pred));
4372 BI.swapSuccessors();
4373 if (BPI)
4374 BPI->swapSuccEdgesProbabilities(BI.getParent());
4375 Worklist.push(Cmp);
4376 return &BI;
4377 }
4378
4379 if (isa<UndefValue>(Cond)) {
4380 handlePotentiallyDeadSuccessors(BI.getParent(), /*LiveSucc*/ nullptr);
4381 return nullptr;
4382 }
4383 if (auto *CI = dyn_cast<ConstantInt>(Cond)) {
4385 BI.getSuccessor(!CI->getZExtValue()));
4386 return nullptr;
4387 }
4388
4389 // Replace all dominated uses of the condition with true/false
4390 // Ignore constant expressions to avoid iterating over uses on other
4391 // functions.
4392 if (!isa<Constant>(Cond) && BI.getSuccessor(0) != BI.getSuccessor(1)) {
4393 for (auto &U : make_early_inc_range(Cond->uses())) {
4394 BasicBlockEdge Edge0(BI.getParent(), BI.getSuccessor(0));
4395 if (DT.dominates(Edge0, U)) {
4396 replaceUse(U, ConstantInt::getTrue(Cond->getType()));
4397 addToWorklist(cast<Instruction>(U.getUser()));
4398 continue;
4399 }
4400 BasicBlockEdge Edge1(BI.getParent(), BI.getSuccessor(1));
4401 if (DT.dominates(Edge1, U)) {
4402 replaceUse(U, ConstantInt::getFalse(Cond->getType()));
4403 addToWorklist(cast<Instruction>(U.getUser()));
4404 }
4405 }
4406 }
4407
4408 DC.registerBranch(&BI);
4409 return nullptr;
4410}
4411
4412// Replaces (switch (select cond, X, C)/(select cond, C, X)) with (switch X) if
4413// we can prove that both (switch C) and (switch X) go to the default when cond
4414// is false/true.
4417 bool IsTrueArm) {
4418 unsigned CstOpIdx = IsTrueArm ? 1 : 2;
4419 auto *C = dyn_cast<ConstantInt>(Select->getOperand(CstOpIdx));
4420 if (!C)
4421 return nullptr;
4422
4423 BasicBlock *CstBB = SI.findCaseValue(C)->getCaseSuccessor();
4424 if (CstBB != SI.getDefaultDest())
4425 return nullptr;
4426 Value *X = Select->getOperand(3 - CstOpIdx);
4427 CmpPredicate Pred;
4428 const APInt *RHSC;
4429 if (!match(Select->getCondition(),
4430 m_ICmp(Pred, m_Specific(X), m_APInt(RHSC))))
4431 return nullptr;
4432 if (IsTrueArm)
4433 Pred = ICmpInst::getInversePredicate(Pred);
4434
4435 // See whether we can replace the select with X
4437 for (auto Case : SI.cases())
4438 if (!CR.contains(Case.getCaseValue()->getValue()))
4439 return nullptr;
4440
4441 return X;
4442}
4443
4445 Value *Cond = SI.getCondition();
4446 Value *Op0;
4447 const APInt *CondOpC;
4448 using InvertFn = std::function<APInt(const APInt &Case, const APInt &C)>;
4449
4450 auto MaybeInvertible = [&](Value *Cond) -> InvertFn {
4451 if (match(Cond, m_Add(m_Value(Op0), m_APInt(CondOpC))))
4452 // Change 'switch (X+C) case Case:' into 'switch (X) case Case-C'.
4453 return [](const APInt &Case, const APInt &C) { return Case - C; };
4454
4455 if (match(Cond, m_Sub(m_APInt(CondOpC), m_Value(Op0))))
4456 // Change 'switch (C-X) case Case:' into 'switch (X) case C-Case'.
4457 return [](const APInt &Case, const APInt &C) { return C - Case; };
4458
4459 if (match(Cond, m_Xor(m_Value(Op0), m_APInt(CondOpC))) &&
4460 !CondOpC->isMinSignedValue() && !CondOpC->isMaxSignedValue())
4461 // Change 'switch (X^C) case Case:' into 'switch (X) case Case^C'.
4462 // Prevent creation of large case values by excluding extremes.
4463 return [](const APInt &Case, const APInt &C) { return Case ^ C; };
4464
4465 return nullptr;
4466 };
4467
4468 // Attempt to invert and simplify the switch condition, as long as the
4469 // condition is not used further, as it may not be profitable otherwise.
4470 if (auto InvertFn = MaybeInvertible(Cond); InvertFn && Cond->hasOneUse()) {
4471 for (auto &Case : SI.cases()) {
4472 const APInt &New = InvertFn(Case.getCaseValue()->getValue(), *CondOpC);
4473 Case.setValue(ConstantInt::get(SI.getContext(), New));
4474 }
4475 return replaceOperand(SI, 0, Op0);
4476 }
4477
4478 uint64_t ShiftAmt;
4479 if (match(Cond, m_Shl(m_Value(Op0), m_ConstantInt(ShiftAmt))) &&
4480 ShiftAmt < Op0->getType()->getScalarSizeInBits() &&
4481 all_of(SI.cases(), [&](const auto &Case) {
4482 return Case.getCaseValue()->getValue().countr_zero() >= ShiftAmt;
4483 })) {
4484 // Change 'switch (X << 2) case 4:' into 'switch (X) case 1:'.
4486 if (Shl->hasNoUnsignedWrap() || Shl->hasNoSignedWrap() ||
4487 Shl->hasOneUse()) {
4488 Value *NewCond = Op0;
4489 if (!Shl->hasNoUnsignedWrap() && !Shl->hasNoSignedWrap()) {
4490 // If the shift may wrap, we need to mask off the shifted bits.
4491 unsigned BitWidth = Op0->getType()->getScalarSizeInBits();
4492 NewCond = Builder.CreateAnd(
4493 Op0, APInt::getLowBitsSet(BitWidth, BitWidth - ShiftAmt));
4494 }
4495 for (auto Case : SI.cases()) {
4496 const APInt &CaseVal = Case.getCaseValue()->getValue();
4497 APInt ShiftedCase = Shl->hasNoSignedWrap() ? CaseVal.ashr(ShiftAmt)
4498 : CaseVal.lshr(ShiftAmt);
4499 Case.setValue(ConstantInt::get(SI.getContext(), ShiftedCase));
4500 }
4501 return replaceOperand(SI, 0, NewCond);
4502 }
4503 }
4504
4505 // Fold switch(zext/sext(X)) into switch(X) if possible.
4506 if (match(Cond, m_ZExtOrSExt(m_Value(Op0)))) {
4507 bool IsZExt = isa<ZExtInst>(Cond);
4508 Type *SrcTy = Op0->getType();
4509 unsigned NewWidth = SrcTy->getScalarSizeInBits();
4510
4511 if (all_of(SI.cases(), [&](const auto &Case) {
4512 const APInt &CaseVal = Case.getCaseValue()->getValue();
4513 return IsZExt ? CaseVal.isIntN(NewWidth)
4514 : CaseVal.isSignedIntN(NewWidth);
4515 })) {
4516 for (auto &Case : SI.cases()) {
4517 APInt TruncatedCase = Case.getCaseValue()->getValue().trunc(NewWidth);
4518 Case.setValue(ConstantInt::get(SI.getContext(), TruncatedCase));
4519 }
4520 return replaceOperand(SI, 0, Op0);
4521 }
4522 }
4523
4524 // Fold switch(select cond, X, Y) into switch(X/Y) if possible
4525 if (auto *Select = dyn_cast<SelectInst>(Cond)) {
4526 if (Value *V =
4527 simplifySwitchOnSelectUsingRanges(SI, Select, /*IsTrueArm=*/true))
4528 return replaceOperand(SI, 0, V);
4529 if (Value *V =
4530 simplifySwitchOnSelectUsingRanges(SI, Select, /*IsTrueArm=*/false))
4531 return replaceOperand(SI, 0, V);
4532 }
4533
4535 unsigned LeadingKnownZeros = Known.countMinLeadingZeros();
4536 unsigned LeadingKnownOnes = Known.countMinLeadingOnes();
4537
4538 // Compute the number of leading bits we can ignore.
4539 // TODO: A better way to determine this would use ComputeNumSignBits().
4540 for (const auto &C : SI.cases()) {
4541 LeadingKnownZeros =
4542 std::min(LeadingKnownZeros, C.getCaseValue()->getValue().countl_zero());
4543 LeadingKnownOnes =
4544 std::min(LeadingKnownOnes, C.getCaseValue()->getValue().countl_one());
4545 }
4546
4547 unsigned NewWidth = Known.getBitWidth() - std::max(LeadingKnownZeros, LeadingKnownOnes);
4548
4549 // Shrink the condition operand if the new type is smaller than the old type.
4550 // But do not shrink to a non-standard type, because backend can't generate
4551 // good code for that yet.
4552 // TODO: We can make it aggressive again after fixing PR39569.
4553 if (NewWidth > 0 && NewWidth < Known.getBitWidth() &&
4554 shouldChangeType(Known.getBitWidth(), NewWidth)) {
4555 IntegerType *Ty = IntegerType::get(SI.getContext(), NewWidth);
4556 Builder.SetInsertPoint(&SI);
4557 Value *NewCond = Builder.CreateTrunc(Cond, Ty, "trunc");
4558
4559 for (auto Case : SI.cases()) {
4560 APInt TruncatedCase = Case.getCaseValue()->getValue().trunc(NewWidth);
4561 Case.setValue(ConstantInt::get(SI.getContext(), TruncatedCase));
4562 }
4563 return replaceOperand(SI, 0, NewCond);
4564 }
4565
4566 if (isa<UndefValue>(Cond)) {
4567 handlePotentiallyDeadSuccessors(SI.getParent(), /*LiveSucc*/ nullptr);
4568 return nullptr;
4569 }
4570 if (auto *CI = dyn_cast<ConstantInt>(Cond)) {
4572 SI.findCaseValue(CI)->getCaseSuccessor());
4573 return nullptr;
4574 }
4575
4576 return nullptr;
4577}
4578
4580InstCombinerImpl::foldExtractOfOverflowIntrinsic(ExtractValueInst &EV) {
4582 if (!WO)
4583 return nullptr;
4584
4585 Intrinsic::ID OvID = WO->getIntrinsicID();
4586 const APInt *C = nullptr;
4587 if (match(WO->getRHS(), m_APIntAllowPoison(C))) {
4588 if (*EV.idx_begin() == 0 && (OvID == Intrinsic::smul_with_overflow ||
4589 OvID == Intrinsic::umul_with_overflow)) {
4590 // extractvalue (any_mul_with_overflow X, -1), 0 --> -X
4591 if (C->isAllOnes())
4592 return BinaryOperator::CreateNeg(WO->getLHS());
4593 // extractvalue (any_mul_with_overflow X, 2^n), 0 --> X << n
4594 if (C->isPowerOf2()) {
4595 return BinaryOperator::CreateShl(
4596 WO->getLHS(),
4597 ConstantInt::get(WO->getLHS()->getType(), C->logBase2()));
4598 }
4599 }
4600 }
4601
4602 // We're extracting from an overflow intrinsic. See if we're the only user.
4603 // That allows us to simplify multiple result intrinsics to simpler things
4604 // that just get one value.
4605 if (!WO->hasOneUse())
4606 return nullptr;
4607
4608 // Check if we're grabbing only the result of a 'with overflow' intrinsic
4609 // and replace it with a traditional binary instruction.
4610 if (*EV.idx_begin() == 0) {
4611 Instruction::BinaryOps BinOp = WO->getBinaryOp();
4612 Value *LHS = WO->getLHS(), *RHS = WO->getRHS();
4613 // Replace the old instruction's uses with poison.
4614 replaceInstUsesWith(*WO, PoisonValue::get(WO->getType()));
4616 return BinaryOperator::Create(BinOp, LHS, RHS);
4617 }
4618
4619 assert(*EV.idx_begin() == 1 && "Unexpected extract index for overflow inst");
4620
4621 // (usub LHS, RHS) overflows when LHS is unsigned-less-than RHS.
4622 if (OvID == Intrinsic::usub_with_overflow)
4623 return new ICmpInst(ICmpInst::ICMP_ULT, WO->getLHS(), WO->getRHS());
4624
4625 // smul with i1 types overflows when both sides are set: -1 * -1 == +1, but
4626 // +1 is not possible because we assume signed values.
4627 if (OvID == Intrinsic::smul_with_overflow &&
4628 WO->getLHS()->getType()->isIntOrIntVectorTy(1))
4629 return BinaryOperator::CreateAnd(WO->getLHS(), WO->getRHS());
4630
4631 // extractvalue (umul_with_overflow X, X), 1 -> X u> 2^(N/2)-1
4632 if (OvID == Intrinsic::umul_with_overflow && WO->getLHS() == WO->getRHS()) {
4633 unsigned BitWidth = WO->getLHS()->getType()->getScalarSizeInBits();
4634 // Only handle even bitwidths for performance reasons.
4635 if (BitWidth % 2 == 0)
4636 return new ICmpInst(
4637 ICmpInst::ICMP_UGT, WO->getLHS(),
4638 ConstantInt::get(WO->getLHS()->getType(),
4640 }
4641
4642 // If only the overflow result is used, and the right hand side is a
4643 // constant (or constant splat), we can remove the intrinsic by directly
4644 // checking for overflow.
4645 if (C) {
4646 // Compute the no-wrap range for LHS given RHS=C, then construct an
4647 // equivalent icmp, potentially using an offset.
4648 ConstantRange NWR = ConstantRange::makeExactNoWrapRegion(
4649 WO->getBinaryOp(), *C, WO->getNoWrapKind());
4650
4651 CmpInst::Predicate Pred;
4652 APInt NewRHSC, Offset;
4653 NWR.getEquivalentICmp(Pred, NewRHSC, Offset);
4654 auto *OpTy = WO->getRHS()->getType();
4655 auto *NewLHS = WO->getLHS();
4656 if (Offset != 0)
4657 NewLHS = Builder.CreateAdd(NewLHS, ConstantInt::get(OpTy, Offset));
4658 return new ICmpInst(ICmpInst::getInversePredicate(Pred), NewLHS,
4659 ConstantInt::get(OpTy, NewRHSC));
4660 }
4661
4662 return nullptr;
4663}
4664
4667 InstCombiner::BuilderTy &Builder) {
4668 // Helper to fold frexp of select to select of frexp.
4669
4670 if (!SelectInst->hasOneUse() || !FrexpCall->hasOneUse())
4671 return nullptr;
4673 Value *TrueVal = SelectInst->getTrueValue();
4674 Value *FalseVal = SelectInst->getFalseValue();
4675
4676 const APFloat *ConstVal = nullptr;
4677 Value *VarOp = nullptr;
4678 bool ConstIsTrue = false;
4679
4680 if (match(TrueVal, m_APFloat(ConstVal))) {
4681 VarOp = FalseVal;
4682 ConstIsTrue = true;
4683 } else if (match(FalseVal, m_APFloat(ConstVal))) {
4684 VarOp = TrueVal;
4685 ConstIsTrue = false;
4686 } else {
4687 return nullptr;
4688 }
4689
4690 Builder.SetInsertPoint(&EV);
4691
4692 CallInst *NewFrexp =
4693 Builder.CreateCall(FrexpCall->getCalledFunction(), {VarOp}, "frexp");
4694 NewFrexp->copyIRFlags(FrexpCall);
4695
4696 Value *NewEV = Builder.CreateExtractValue(NewFrexp, 0, "mantissa");
4697
4698 int Exp;
4699 APFloat Mantissa = frexp(*ConstVal, Exp, APFloat::rmNearestTiesToEven);
4700
4701 Constant *ConstantMantissa = ConstantFP::get(TrueVal->getType(), Mantissa);
4702
4703 Value *NewSel = Builder.CreateSelectFMF(
4704 Cond, ConstIsTrue ? ConstantMantissa : NewEV,
4705 ConstIsTrue ? NewEV : ConstantMantissa, SelectInst, "select.frexp");
4706 return NewSel;
4707}
4709 Value *Agg = EV.getAggregateOperand();
4710
4711 if (!EV.hasIndices())
4712 return replaceInstUsesWith(EV, Agg);
4713
4714 if (Value *V = simplifyExtractValueInst(Agg, EV.getIndices(),
4715 SQ.getWithInstruction(&EV)))
4716 return replaceInstUsesWith(EV, V);
4717
4718 Value *Cond, *TrueVal, *FalseVal;
4720 m_Value(Cond), m_Value(TrueVal), m_Value(FalseVal)))))) {
4721 auto *SelInst =
4722 cast<SelectInst>(cast<IntrinsicInst>(Agg)->getArgOperand(0));
4723 if (Value *Result =
4724 foldFrexpOfSelect(EV, cast<IntrinsicInst>(Agg), SelInst, Builder))
4725 return replaceInstUsesWith(EV, Result);
4726 }
4728 // We're extracting from an insertvalue instruction, compare the indices
4729 const unsigned *exti, *exte, *insi, *inse;
4730 for (exti = EV.idx_begin(), insi = IV->idx_begin(),
4731 exte = EV.idx_end(), inse = IV->idx_end();
4732 exti != exte && insi != inse;
4733 ++exti, ++insi) {
4734 if (*insi != *exti)
4735 // The insert and extract both reference distinctly different elements.
4736 // This means the extract is not influenced by the insert, and we can
4737 // replace the aggregate operand of the extract with the aggregate
4738 // operand of the insert. i.e., replace
4739 // %I = insertvalue { i32, { i32 } } %A, { i32 } { i32 42 }, 1
4740 // %E = extractvalue { i32, { i32 } } %I, 0
4741 // with
4742 // %E = extractvalue { i32, { i32 } } %A, 0
4743 return ExtractValueInst::Create(IV->getAggregateOperand(),
4744 EV.getIndices());
4745 }
4746 if (exti == exte && insi == inse)
4747 // Both iterators are at the end: Index lists are identical. Replace
4748 // %B = insertvalue { i32, { i32 } } %A, i32 42, 1, 0
4749 // %C = extractvalue { i32, { i32 } } %B, 1, 0
4750 // with "i32 42"
4751 return replaceInstUsesWith(EV, IV->getInsertedValueOperand());
4752 if (exti == exte) {
4753 // The extract list is a prefix of the insert list. i.e. replace
4754 // %I = insertvalue { i32, { i32 } } %A, i32 42, 1, 0
4755 // %E = extractvalue { i32, { i32 } } %I, 1
4756 // with
4757 // %X = extractvalue { i32, { i32 } } %A, 1
4758 // %E = insertvalue { i32 } %X, i32 42, 0
4759 // by switching the order of the insert and extract (though the
4760 // insertvalue should be left in, since it may have other uses).
4761 Value *NewEV = Builder.CreateExtractValue(IV->getAggregateOperand(),
4762 EV.getIndices());
4763 return InsertValueInst::Create(NewEV, IV->getInsertedValueOperand(),
4764 ArrayRef(insi, inse));
4765 }
4766 if (insi == inse)
4767 // The insert list is a prefix of the extract list
4768 // We can simply remove the common indices from the extract and make it
4769 // operate on the inserted value instead of the insertvalue result.
4770 // i.e., replace
4771 // %I = insertvalue { i32, { i32 } } %A, { i32 } { i32 42 }, 1
4772 // %E = extractvalue { i32, { i32 } } %I, 1, 0
4773 // with
4774 // %E extractvalue { i32 } { i32 42 }, 0
4775 return ExtractValueInst::Create(IV->getInsertedValueOperand(),
4776 ArrayRef(exti, exte));
4777 }
4778
4779 if (Instruction *R = foldExtractOfOverflowIntrinsic(EV))
4780 return R;
4781
4782 if (LoadInst *L = dyn_cast<LoadInst>(Agg)) {
4783 // Bail out if the aggregate contains scalable vector type
4784 if (auto *STy = dyn_cast<StructType>(Agg->getType());
4785 STy && STy->isScalableTy())
4786 return nullptr;
4787
4788 // If the (non-volatile) load only has one use, we can rewrite this to a
4789 // load from a GEP. This reduces the size of the load. If a load is used
4790 // only by extractvalue instructions then this either must have been
4791 // optimized before, or it is a struct with padding, in which case we
4792 // don't want to do the transformation as it loses padding knowledge.
4793 if (L->isSimple() && L->hasOneUse()) {
4794 // extractvalue has integer indices, getelementptr has Value*s. Convert.
4795 SmallVector<Value*, 4> Indices;
4796 // Prefix an i32 0 since we need the first element.
4797 Indices.push_back(Builder.getInt32(0));
4798 for (unsigned Idx : EV.indices())
4799 Indices.push_back(Builder.getInt32(Idx));
4800
4801 // We need to insert these at the location of the old load, not at that of
4802 // the extractvalue.
4803 Builder.SetInsertPoint(L);
4804 Value *GEP = Builder.CreateInBoundsGEP(L->getType(),
4805 L->getPointerOperand(), Indices);
4806 Instruction *NL = Builder.CreateLoad(EV.getType(), GEP);
4807 // Whatever aliasing information we had for the orignal load must also
4808 // hold for the smaller load, so propagate the annotations.
4809 NL->setAAMetadata(L->getAAMetadata());
4810 // Returning the load directly will cause the main loop to insert it in
4811 // the wrong spot, so use replaceInstUsesWith().
4812 return replaceInstUsesWith(EV, NL);
4813 }
4814 }
4815
4816 if (auto *PN = dyn_cast<PHINode>(Agg))
4817 if (Instruction *Res = foldOpIntoPhi(EV, PN))
4818 return Res;
4819
4820 // Canonicalize extract (select Cond, TV, FV)
4821 // -> select cond, (extract TV), (extract FV)
4822 if (auto *SI = dyn_cast<SelectInst>(Agg))
4823 if (Instruction *R = FoldOpIntoSelect(EV, SI, /*FoldWithMultiUse=*/true))
4824 return R;
4825
4826 // We could simplify extracts from other values. Note that nested extracts may
4827 // already be simplified implicitly by the above: extract (extract (insert) )
4828 // will be translated into extract ( insert ( extract ) ) first and then just
4829 // the value inserted, if appropriate. Similarly for extracts from single-use
4830 // loads: extract (extract (load)) will be translated to extract (load (gep))
4831 // and if again single-use then via load (gep (gep)) to load (gep).
4832 // However, double extracts from e.g. function arguments or return values
4833 // aren't handled yet.
4834 return nullptr;
4835}
4836
4837/// Return 'true' if the given typeinfo will match anything.
4838static bool isCatchAll(EHPersonality Personality, Constant *TypeInfo) {
4839 switch (Personality) {
4843 // The GCC C EH and Rust personality only exists to support cleanups, so
4844 // it's not clear what the semantics of catch clauses are.
4845 return false;
4847 return false;
4849 // While __gnat_all_others_value will match any Ada exception, it doesn't
4850 // match foreign exceptions (or didn't, before gcc-4.7).
4851 return false;
4863 return isa<ConstantPointerNull>(TypeInfo);
4864 }
4865 llvm_unreachable("invalid enum");
4866}
4867
4868static bool shorter_filter(const Value *LHS, const Value *RHS) {
4869 return
4870 cast<ArrayType>(LHS->getType())->getNumElements()
4871 <
4872 cast<ArrayType>(RHS->getType())->getNumElements();
4873}
4874
4876 // The logic here should be correct for any real-world personality function.
4877 // However if that turns out not to be true, the offending logic can always
4878 // be conditioned on the personality function, like the catch-all logic is.
4879 EHPersonality Personality =
4880 classifyEHPersonality(LI.getParent()->getParent()->getPersonalityFn());
4881
4882 // Simplify the list of clauses, eg by removing repeated catch clauses
4883 // (these are often created by inlining).
4884 bool MakeNewInstruction = false; // If true, recreate using the following:
4885 SmallVector<Constant *, 16> NewClauses; // - Clauses for the new instruction;
4886 bool CleanupFlag = LI.isCleanup(); // - The new instruction is a cleanup.
4887
4888 SmallPtrSet<Value *, 16> AlreadyCaught; // Typeinfos known caught already.
4889 for (unsigned i = 0, e = LI.getNumClauses(); i != e; ++i) {
4890 bool isLastClause = i + 1 == e;
4891 if (LI.isCatch(i)) {
4892 // A catch clause.
4893 Constant *CatchClause = LI.getClause(i);
4894 Constant *TypeInfo = CatchClause->stripPointerCasts();
4895
4896 // If we already saw this clause, there is no point in having a second
4897 // copy of it.
4898 if (AlreadyCaught.insert(TypeInfo).second) {
4899 // This catch clause was not already seen.
4900 NewClauses.push_back(CatchClause);
4901 } else {
4902 // Repeated catch clause - drop the redundant copy.
4903 MakeNewInstruction = true;
4904 }
4905
4906 // If this is a catch-all then there is no point in keeping any following
4907 // clauses or marking the landingpad as having a cleanup.
4908 if (isCatchAll(Personality, TypeInfo)) {
4909 if (!isLastClause)
4910 MakeNewInstruction = true;
4911 CleanupFlag = false;
4912 break;
4913 }
4914 } else {
4915 // A filter clause. If any of the filter elements were already caught
4916 // then they can be dropped from the filter. It is tempting to try to
4917 // exploit the filter further by saying that any typeinfo that does not
4918 // occur in the filter can't be caught later (and thus can be dropped).
4919 // However this would be wrong, since typeinfos can match without being
4920 // equal (for example if one represents a C++ class, and the other some
4921 // class derived from it).
4922 assert(LI.isFilter(i) && "Unsupported landingpad clause!");
4923 Constant *FilterClause = LI.getClause(i);
4924 ArrayType *FilterType = cast<ArrayType>(FilterClause->getType());
4925 unsigned NumTypeInfos = FilterType->getNumElements();
4926
4927 // An empty filter catches everything, so there is no point in keeping any
4928 // following clauses or marking the landingpad as having a cleanup. By
4929 // dealing with this case here the following code is made a bit simpler.
4930 if (!NumTypeInfos) {
4931 NewClauses.push_back(FilterClause);
4932 if (!isLastClause)
4933 MakeNewInstruction = true;
4934 CleanupFlag = false;
4935 break;
4936 }
4937
4938 bool MakeNewFilter = false; // If true, make a new filter.
4939 SmallVector<Constant *, 16> NewFilterElts; // New elements.
4940 if (isa<ConstantAggregateZero>(FilterClause)) {
4941 // Not an empty filter - it contains at least one null typeinfo.
4942 assert(NumTypeInfos > 0 && "Should have handled empty filter already!");
4943 Constant *TypeInfo =
4945 // If this typeinfo is a catch-all then the filter can never match.
4946 if (isCatchAll(Personality, TypeInfo)) {
4947 // Throw the filter away.
4948 MakeNewInstruction = true;
4949 continue;
4950 }
4951
4952 // There is no point in having multiple copies of this typeinfo, so
4953 // discard all but the first copy if there is more than one.
4954 NewFilterElts.push_back(TypeInfo);
4955 if (NumTypeInfos > 1)
4956 MakeNewFilter = true;
4957 } else {
4958 ConstantArray *Filter = cast<ConstantArray>(FilterClause);
4959 SmallPtrSet<Value *, 16> SeenInFilter; // For uniquing the elements.
4960 NewFilterElts.reserve(NumTypeInfos);
4961
4962 // Remove any filter elements that were already caught or that already
4963 // occurred in the filter. While there, see if any of the elements are
4964 // catch-alls. If so, the filter can be discarded.
4965 bool SawCatchAll = false;
4966 for (unsigned j = 0; j != NumTypeInfos; ++j) {
4967 Constant *Elt = Filter->getOperand(j);
4968 Constant *TypeInfo = Elt->stripPointerCasts();
4969 if (isCatchAll(Personality, TypeInfo)) {
4970 // This element is a catch-all. Bail out, noting this fact.
4971 SawCatchAll = true;
4972 break;
4973 }
4974
4975 // Even if we've seen a type in a catch clause, we don't want to
4976 // remove it from the filter. An unexpected type handler may be
4977 // set up for a call site which throws an exception of the same
4978 // type caught. In order for the exception thrown by the unexpected
4979 // handler to propagate correctly, the filter must be correctly
4980 // described for the call site.
4981 //
4982 // Example:
4983 //
4984 // void unexpected() { throw 1;}
4985 // void foo() throw (int) {
4986 // std::set_unexpected(unexpected);
4987 // try {
4988 // throw 2.0;
4989 // } catch (int i) {}
4990 // }
4991
4992 // There is no point in having multiple copies of the same typeinfo in
4993 // a filter, so only add it if we didn't already.
4994 if (SeenInFilter.insert(TypeInfo).second)
4995 NewFilterElts.push_back(cast<Constant>(Elt));
4996 }
4997 // A filter containing a catch-all cannot match anything by definition.
4998 if (SawCatchAll) {
4999 // Throw the filter away.
5000 MakeNewInstruction = true;
5001 continue;
5002 }
5003
5004 // If we dropped something from the filter, make a new one.
5005 if (NewFilterElts.size() < NumTypeInfos)
5006 MakeNewFilter = true;
5007 }
5008 if (MakeNewFilter) {
5009 FilterType = ArrayType::get(FilterType->getElementType(),
5010 NewFilterElts.size());
5011 FilterClause = ConstantArray::get(FilterType, NewFilterElts);
5012 MakeNewInstruction = true;
5013 }
5014
5015 NewClauses.push_back(FilterClause);
5016
5017 // If the new filter is empty then it will catch everything so there is
5018 // no point in keeping any following clauses or marking the landingpad
5019 // as having a cleanup. The case of the original filter being empty was
5020 // already handled above.
5021 if (MakeNewFilter && !NewFilterElts.size()) {
5022 assert(MakeNewInstruction && "New filter but not a new instruction!");
5023 CleanupFlag = false;
5024 break;
5025 }
5026 }
5027 }
5028
5029 // If several filters occur in a row then reorder them so that the shortest
5030 // filters come first (those with the smallest number of elements). This is
5031 // advantageous because shorter filters are more likely to match, speeding up
5032 // unwinding, but mostly because it increases the effectiveness of the other
5033 // filter optimizations below.
5034 for (unsigned i = 0, e = NewClauses.size(); i + 1 < e; ) {
5035 unsigned j;
5036 // Find the maximal 'j' s.t. the range [i, j) consists entirely of filters.
5037 for (j = i; j != e; ++j)
5038 if (!isa<ArrayType>(NewClauses[j]->getType()))
5039 break;
5040
5041 // Check whether the filters are already sorted by length. We need to know
5042 // if sorting them is actually going to do anything so that we only make a
5043 // new landingpad instruction if it does.
5044 for (unsigned k = i; k + 1 < j; ++k)
5045 if (shorter_filter(NewClauses[k+1], NewClauses[k])) {
5046 // Not sorted, so sort the filters now. Doing an unstable sort would be
5047 // correct too but reordering filters pointlessly might confuse users.
5048 std::stable_sort(NewClauses.begin() + i, NewClauses.begin() + j,
5050 MakeNewInstruction = true;
5051 break;
5052 }
5053
5054 // Look for the next batch of filters.
5055 i = j + 1;
5056 }
5057
5058 // If typeinfos matched if and only if equal, then the elements of a filter L
5059 // that occurs later than a filter F could be replaced by the intersection of
5060 // the elements of F and L. In reality two typeinfos can match without being
5061 // equal (for example if one represents a C++ class, and the other some class
5062 // derived from it) so it would be wrong to perform this transform in general.
5063 // However the transform is correct and useful if F is a subset of L. In that
5064 // case L can be replaced by F, and thus removed altogether since repeating a
5065 // filter is pointless. So here we look at all pairs of filters F and L where
5066 // L follows F in the list of clauses, and remove L if every element of F is
5067 // an element of L. This can occur when inlining C++ functions with exception
5068 // specifications.
5069 for (unsigned i = 0; i + 1 < NewClauses.size(); ++i) {
5070 // Examine each filter in turn.
5071 Value *Filter = NewClauses[i];
5072 ArrayType *FTy = dyn_cast<ArrayType>(Filter->getType());
5073 if (!FTy)
5074 // Not a filter - skip it.
5075 continue;
5076 unsigned FElts = FTy->getNumElements();
5077 // Examine each filter following this one. Doing this backwards means that
5078 // we don't have to worry about filters disappearing under us when removed.
5079 for (unsigned j = NewClauses.size() - 1; j != i; --j) {
5080 Value *LFilter = NewClauses[j];
5081 ArrayType *LTy = dyn_cast<ArrayType>(LFilter->getType());
5082 if (!LTy)
5083 // Not a filter - skip it.
5084 continue;
5085 // If Filter is a subset of LFilter, i.e. every element of Filter is also
5086 // an element of LFilter, then discard LFilter.
5087 SmallVectorImpl<Constant *>::iterator J = NewClauses.begin() + j;
5088 // If Filter is empty then it is a subset of LFilter.
5089 if (!FElts) {
5090 // Discard LFilter.
5091 NewClauses.erase(J);
5092 MakeNewInstruction = true;
5093 // Move on to the next filter.
5094 continue;
5095 }
5096 unsigned LElts = LTy->getNumElements();
5097 // If Filter is longer than LFilter then it cannot be a subset of it.
5098 if (FElts > LElts)
5099 // Move on to the next filter.
5100 continue;
5101 // At this point we know that LFilter has at least one element.
5102 if (isa<ConstantAggregateZero>(LFilter)) { // LFilter only contains zeros.
5103 // Filter is a subset of LFilter iff Filter contains only zeros (as we
5104 // already know that Filter is not longer than LFilter).
5106 assert(FElts <= LElts && "Should have handled this case earlier!");
5107 // Discard LFilter.
5108 NewClauses.erase(J);
5109 MakeNewInstruction = true;
5110 }
5111 // Move on to the next filter.
5112 continue;
5113 }
5114 ConstantArray *LArray = cast<ConstantArray>(LFilter);
5115 if (isa<ConstantAggregateZero>(Filter)) { // Filter only contains zeros.
5116 // Since Filter is non-empty and contains only zeros, it is a subset of
5117 // LFilter iff LFilter contains a zero.
5118 assert(FElts > 0 && "Should have eliminated the empty filter earlier!");
5119 for (unsigned l = 0; l != LElts; ++l)
5120 if (isa<ConstantPointerNull>(LArray->getOperand(l))) {
5121 // LFilter contains a zero - discard it.
5122 NewClauses.erase(J);
5123 MakeNewInstruction = true;
5124 break;
5125 }
5126 // Move on to the next filter.
5127 continue;
5128 }
5129 // At this point we know that both filters are ConstantArrays. Loop over
5130 // operands to see whether every element of Filter is also an element of
5131 // LFilter. Since filters tend to be short this is probably faster than
5132 // using a method that scales nicely.
5134 bool AllFound = true;
5135 for (unsigned f = 0; f != FElts; ++f) {
5136 Value *FTypeInfo = FArray->getOperand(f)->stripPointerCasts();
5137 AllFound = false;
5138 for (unsigned l = 0; l != LElts; ++l) {
5139 Value *LTypeInfo = LArray->getOperand(l)->stripPointerCasts();
5140 if (LTypeInfo == FTypeInfo) {
5141 AllFound = true;
5142 break;
5143 }
5144 }
5145 if (!AllFound)
5146 break;
5147 }
5148 if (AllFound) {
5149 // Discard LFilter.
5150 NewClauses.erase(J);
5151 MakeNewInstruction = true;
5152 }
5153 // Move on to the next filter.
5154 }
5155 }
5156
5157 // If we changed any of the clauses, replace the old landingpad instruction
5158 // with a new one.
5159 if (MakeNewInstruction) {
5161 NewClauses.size());
5162 for (Constant *C : NewClauses)
5163 NLI->addClause(C);
5164 // A landing pad with no clauses must have the cleanup flag set. It is
5165 // theoretically possible, though highly unlikely, that we eliminated all
5166 // clauses. If so, force the cleanup flag to true.
5167 if (NewClauses.empty())
5168 CleanupFlag = true;
5169 NLI->setCleanup(CleanupFlag);
5170 return NLI;
5171 }
5172
5173 // Even if none of the clauses changed, we may nonetheless have understood
5174 // that the cleanup flag is pointless. Clear it if so.
5175 if (LI.isCleanup() != CleanupFlag) {
5176 assert(!CleanupFlag && "Adding a cleanup, not removing one?!");
5177 LI.setCleanup(CleanupFlag);
5178 return &LI;
5179 }
5180
5181 return nullptr;
5182}
5183
5184Value *
5186 // Try to push freeze through instructions that propagate but don't produce
5187 // poison as far as possible. If an operand of freeze follows three
5188 // conditions 1) one-use, 2) does not produce poison, and 3) has all but one
5189 // guaranteed-non-poison operands then push the freeze through to the one
5190 // operand that is not guaranteed non-poison. The actual transform is as
5191 // follows.
5192 // Op1 = ... ; Op1 can be posion
5193 // Op0 = Inst(Op1, NonPoisonOps...) ; Op0 has only one use and only have
5194 // ; single guaranteed-non-poison operands
5195 // ... = Freeze(Op0)
5196 // =>
5197 // Op1 = ...
5198 // Op1.fr = Freeze(Op1)
5199 // ... = Inst(Op1.fr, NonPoisonOps...)
5200 auto *OrigOp = OrigFI.getOperand(0);
5201 auto *OrigOpInst = dyn_cast<Instruction>(OrigOp);
5202
5203 // While we could change the other users of OrigOp to use freeze(OrigOp), that
5204 // potentially reduces their optimization potential, so let's only do this iff
5205 // the OrigOp is only used by the freeze.
5206 if (!OrigOpInst || !OrigOpInst->hasOneUse() || isa<PHINode>(OrigOp))
5207 return nullptr;
5208
5209 // We can't push the freeze through an instruction which can itself create
5210 // poison. If the only source of new poison is flags, we can simply
5211 // strip them (since we know the only use is the freeze and nothing can
5212 // benefit from them.)
5214 /*ConsiderFlagsAndMetadata*/ false))
5215 return nullptr;
5216
5217 // If operand is guaranteed not to be poison, there is no need to add freeze
5218 // to the operand. So we first find the operand that is not guaranteed to be
5219 // poison.
5220 Value *MaybePoisonOperand = nullptr;
5221 for (Value *V : OrigOpInst->operands()) {
5223 // Treat identical operands as a single operand.
5224 (MaybePoisonOperand && MaybePoisonOperand == V))
5225 continue;
5226 if (!MaybePoisonOperand)
5227 MaybePoisonOperand = V;
5228 else
5229 return nullptr;
5230 }
5231
5232 OrigOpInst->dropPoisonGeneratingAnnotations();
5233
5234 // If all operands are guaranteed to be non-poison, we can drop freeze.
5235 if (!MaybePoisonOperand)
5236 return OrigOp;
5237
5238 Builder.SetInsertPoint(OrigOpInst);
5239 Value *FrozenMaybePoisonOperand = Builder.CreateFreeze(
5240 MaybePoisonOperand, MaybePoisonOperand->getName() + ".fr");
5241
5242 OrigOpInst->replaceUsesOfWith(MaybePoisonOperand, FrozenMaybePoisonOperand);
5243 return OrigOp;
5244}
5245
5247 PHINode *PN) {
5248 // Detect whether this is a recurrence with a start value and some number of
5249 // backedge values. We'll check whether we can push the freeze through the
5250 // backedge values (possibly dropping poison flags along the way) until we
5251 // reach the phi again. In that case, we can move the freeze to the start
5252 // value.
5253 Use *StartU = nullptr;
5255 for (Use &U : PN->incoming_values()) {
5256 if (DT.dominates(PN->getParent(), PN->getIncomingBlock(U))) {
5257 // Add backedge value to worklist.
5258 Worklist.push_back(U.get());
5259 continue;
5260 }
5261
5262 // Don't bother handling multiple start values.
5263 if (StartU)
5264 return nullptr;
5265 StartU = &U;
5266 }
5267
5268 if (!StartU || Worklist.empty())
5269 return nullptr; // Not a recurrence.
5270
5271 Value *StartV = StartU->get();
5272 BasicBlock *StartBB = PN->getIncomingBlock(*StartU);
5273 bool StartNeedsFreeze = !isGuaranteedNotToBeUndefOrPoison(StartV);
5274 // We can't insert freeze if the start value is the result of the
5275 // terminator (e.g. an invoke).
5276 if (StartNeedsFreeze && StartBB->getTerminator() == StartV)
5277 return nullptr;
5278
5281 while (!Worklist.empty()) {
5282 Value *V = Worklist.pop_back_val();
5283 if (!Visited.insert(V).second)
5284 continue;
5285
5286 if (Visited.size() > 32)
5287 return nullptr; // Limit the total number of values we inspect.
5288
5289 // Assume that PN is non-poison, because it will be after the transform.
5290 if (V == PN || isGuaranteedNotToBeUndefOrPoison(V))
5291 continue;
5292
5295 /*ConsiderFlagsAndMetadata*/ false))
5296 return nullptr;
5297
5298 DropFlags.push_back(I);
5299 append_range(Worklist, I->operands());
5300 }
5301
5302 for (Instruction *I : DropFlags)
5303 I->dropPoisonGeneratingAnnotations();
5304
5305 if (StartNeedsFreeze) {
5306 Builder.SetInsertPoint(StartBB->getTerminator());
5307 Value *FrozenStartV = Builder.CreateFreeze(StartV,
5308 StartV->getName() + ".fr");
5309 replaceUse(*StartU, FrozenStartV);
5310 }
5311 return replaceInstUsesWith(FI, PN);
5312}
5313
5315 Value *Op = FI.getOperand(0);
5316
5317 if (isa<Constant>(Op) || Op->hasOneUse())
5318 return false;
5319
5320 // Move the freeze directly after the definition of its operand, so that
5321 // it dominates the maximum number of uses. Note that it may not dominate
5322 // *all* uses if the operand is an invoke/callbr and the use is in a phi on
5323 // the normal/default destination. This is why the domination check in the
5324 // replacement below is still necessary.
5325 BasicBlock::iterator MoveBefore;
5326 if (isa<Argument>(Op)) {
5327 MoveBefore =
5329 } else {
5330 auto MoveBeforeOpt = cast<Instruction>(Op)->getInsertionPointAfterDef();
5331 if (!MoveBeforeOpt)
5332 return false;
5333 MoveBefore = *MoveBeforeOpt;
5334 }
5335
5336 // Re-point iterator to come after any debug-info records.
5337 MoveBefore.setHeadBit(false);
5338
5339 bool Changed = false;
5340 if (&FI != &*MoveBefore) {
5341 FI.moveBefore(*MoveBefore->getParent(), MoveBefore);
5342 Changed = true;
5343 }
5344
5346 Changed |= Op->replaceUsesWithIf(&FI, [&](Use &U) -> bool {
5347 if (!DT.dominates(&FI, U))
5348 return false;
5349
5350 Users.push_back(U.getUser());
5351 return true;
5352 });
5353
5354 for (auto *U : Users) {
5355 // Re-queue U and its users: freezing U's operand can expose a fold on a
5356 // user of U (e.g. a freeze of U can now be pushed through it) that would
5357 // otherwise only fire on a later iteration, tripping the fixpoint verifier.
5358 auto *UI = cast<Instruction>(U);
5359 Worklist.pushUsersToWorkList(*UI);
5360 Worklist.push(UI);
5361 }
5362
5363 return Changed;
5364}
5365
5366// Check if any direct or bitcast user of this value is a shuffle instruction.
5368 for (auto *U : V->users()) {
5370 return true;
5371 else if (match(U, m_BitCast(m_Specific(V))) && isUsedWithinShuffleVector(U))
5372 return true;
5373 }
5374 return false;
5375}
5376
5378 Value *Op0 = I.getOperand(0);
5379
5380 if (Value *V = simplifyFreezeInst(Op0, SQ.getWithInstruction(&I)))
5381 return replaceInstUsesWith(I, V);
5382
5383 // freeze (phi const, x) --> phi const, (freeze x)
5384 if (auto *PN = dyn_cast<PHINode>(Op0)) {
5385 if (Instruction *NV = foldOpIntoPhi(I, PN))
5386 return NV;
5387 if (Instruction *NV = foldFreezeIntoRecurrence(I, PN))
5388 return NV;
5389 }
5390
5392 return replaceInstUsesWith(I, NI);
5393
5394 // If I is freeze(undef), check its uses and fold it to a fixed constant.
5395 // - or: pick -1
5396 // - select's condition: if the true value is constant, choose it by making
5397 // the condition true.
5398 // - phi: pick the common constant across operands
5399 // - default: pick 0
5400 //
5401 // Note that this transform is intentionally done here rather than
5402 // via an analysis in InstSimplify or at individual user sites. That is
5403 // because we must produce the same value for all uses of the freeze -
5404 // it's the reason "freeze" exists!
5405 //
5406 // TODO: This could use getBinopAbsorber() / getBinopIdentity() to avoid
5407 // duplicating logic for binops at least.
5408 auto getUndefReplacement = [&](Type *Ty) {
5409 auto pickCommonConstantFromPHI = [](PHINode &PN) -> Value * {
5410 // phi(freeze(undef), C, C). Choose C for freeze so the PHI can be
5411 // removed.
5412 Constant *BestValue = nullptr;
5413 for (Value *V : PN.incoming_values()) {
5414 if (match(V, m_Freeze(m_Undef())))
5415 continue;
5416
5418 if (!C)
5419 return nullptr;
5420
5422 return nullptr;
5423
5424 if (BestValue && BestValue != C)
5425 return nullptr;
5426
5427 BestValue = C;
5428 }
5429 return BestValue;
5430 };
5431
5432 Value *NullValue = Constant::getNullValue(Ty);
5433 Value *BestValue = nullptr;
5434 for (auto *U : I.users()) {
5435 Value *V = NullValue;
5436 if (match(U, m_Or(m_Value(), m_Value())))
5438 else if (match(U, m_Select(m_Specific(&I), m_Constant(), m_Value())))
5439 V = ConstantInt::getTrue(Ty);
5440 else if (match(U, m_c_Select(m_Specific(&I), m_Value(V)))) {
5441 if (V == &I || !isGuaranteedNotToBeUndefOrPoison(V, &AC, &I, &DT))
5442 V = NullValue;
5443 } else if (auto *PHI = dyn_cast<PHINode>(U)) {
5444 if (Value *MaybeV = pickCommonConstantFromPHI(*PHI))
5445 V = MaybeV;
5446 }
5447
5448 if (!BestValue)
5449 BestValue = V;
5450 else if (BestValue != V)
5451 BestValue = NullValue;
5452 }
5453 assert(BestValue && "Must have at least one use");
5454 assert(BestValue != &I && "Cannot replace with itself");
5455 return BestValue;
5456 };
5457
5458 if (match(Op0, m_Undef())) {
5459 // Don't fold freeze(undef/poison) if it's used as a vector operand in
5460 // a shuffle. This may improve codegen for shuffles that allow
5461 // unspecified inputs.
5463 return nullptr;
5464 return replaceInstUsesWith(I, getUndefReplacement(I.getType()));
5465 }
5466
5467 auto getFreezeVectorReplacement = [](Constant *C) -> Constant * {
5468 Type *Ty = C->getType();
5469 auto *VTy = dyn_cast<FixedVectorType>(Ty);
5470 if (!VTy)
5471 return nullptr;
5472 Constant *BestValue;
5474 m_Unless(m_Undef()), m_Constant(BestValue)))))
5475 BestValue = Constant::getNullValue(VTy->getScalarType());
5476 return Constant::replaceUndefsWith(C, BestValue);
5477 };
5478
5479 Constant *C;
5480 if (match(Op0, m_Constant(C)) && C->containsUndefOrPoisonElement() &&
5481 !C->containsConstantExpression()) {
5482 if (Constant *Repl = getFreezeVectorReplacement(C))
5483 return replaceInstUsesWith(I, Repl);
5484 }
5485
5486 // Replace uses of Op with freeze(Op).
5487 if (freezeOtherUses(I))
5488 return &I;
5489
5490 return nullptr;
5491}
5492
5493/// Check for case where the call writes to an otherwise dead alloca. This
5494/// shows up for unused out-params in idiomatic C/C++ code. Note that this
5495/// helper *only* analyzes the write; doesn't check any other legality aspect.
5497 auto *CB = dyn_cast<CallBase>(I);
5498 if (!CB)
5499 // TODO: handle e.g. store to alloca here - only worth doing if we extend
5500 // to allow reload along used path as described below. Otherwise, this
5501 // is simply a store to a dead allocation which will be removed.
5502 return false;
5503 std::optional<MemoryLocation> Dest = MemoryLocation::getForDest(CB, TLI);
5504 if (!Dest)
5505 return false;
5506 auto *AI = dyn_cast<AllocaInst>(getUnderlyingObject(Dest->Ptr));
5507 if (!AI)
5508 // TODO: allow malloc?
5509 return false;
5510 // TODO: allow memory access dominated by move point? Note that since AI
5511 // could have a reference to itself captured by the call, we would need to
5512 // account for cycles in doing so.
5513 SmallVector<const User *> AllocaUsers;
5515 auto pushUsers = [&](const Instruction &I) {
5516 for (const User *U : I.users()) {
5517 if (Visited.insert(U).second)
5518 AllocaUsers.push_back(U);
5519 }
5520 };
5521 pushUsers(*AI);
5522 while (!AllocaUsers.empty()) {
5523 auto *UserI = cast<Instruction>(AllocaUsers.pop_back_val());
5524 if (isa<GetElementPtrInst>(UserI) || isa<AddrSpaceCastInst>(UserI)) {
5525 pushUsers(*UserI);
5526 continue;
5527 }
5528 if (UserI == CB)
5529 continue;
5530 // TODO: support lifetime.start/end here
5531 return false;
5532 }
5533 return true;
5534}
5535
5536/// Try to move the specified instruction from its current block into the
5537/// beginning of DestBlock, which can only happen if it's safe to move the
5538/// instruction past all of the instructions between it and the end of its
5539/// block.
5541 BasicBlock *DestBlock) {
5542 BasicBlock *SrcBlock = I->getParent();
5543
5544 // Cannot move control-flow-involving, volatile loads, vaarg, etc.
5545 if (isa<PHINode>(I) || I->isEHPad() || I->mayThrow() || !I->willReturn() ||
5546 I->isTerminator())
5547 return false;
5548
5549 // Do not sink static or dynamic alloca instructions. Static allocas must
5550 // remain in the entry block, and dynamic allocas must not be sunk in between
5551 // a stacksave / stackrestore pair, which would incorrectly shorten its
5552 // lifetime.
5553 if (isa<AllocaInst>(I))
5554 return false;
5555
5556 // Do not sink into catchswitch blocks.
5557 if (isa<CatchSwitchInst>(DestBlock->getTerminator()))
5558 return false;
5559
5560 // Do not sink convergent call instructions.
5561 if (auto *CI = dyn_cast<CallInst>(I)) {
5562 if (CI->isConvergent())
5563 return false;
5564 }
5565
5566 // Unless we can prove that the memory write isn't visibile except on the
5567 // path we're sinking to, we must bail.
5568 if (I->mayWriteToMemory()) {
5569 if (!SoleWriteToDeadLocal(I, TLI))
5570 return false;
5571 }
5572
5573 // We can only sink load instructions if there is nothing between the load and
5574 // the end of block that could change the value.
5575 if (I->mayReadFromMemory() &&
5576 !I->hasMetadata(LLVMContext::MD_invariant_load)) {
5577 // We don't want to do any sophisticated alias analysis, so we only check
5578 // the instructions after I in I's parent block if we try to sink to its
5579 // successor block.
5580 if (DestBlock->getUniquePredecessor() != I->getParent())
5581 return false;
5582 for (BasicBlock::iterator Scan = std::next(I->getIterator()),
5583 E = I->getParent()->end();
5584 Scan != E; ++Scan)
5585 if (Scan->mayWriteToMemory() && !isa<AssumeInst>(Scan))
5586 return false;
5587 }
5588
5589 I->dropDroppableUses([&](const Use *U) {
5590 auto *I = dyn_cast<Instruction>(U->getUser());
5591 if (I && I->getParent() != DestBlock) {
5592 Worklist.add(I);
5593 return true;
5594 }
5595 return false;
5596 });
5597 /// FIXME: We could remove droppable uses that are not dominated by
5598 /// the new position.
5599
5600 BasicBlock::iterator InsertPos = DestBlock->getFirstInsertionPt();
5601 I->moveBefore(*DestBlock, InsertPos);
5602 ++NumSunkInst;
5603
5604 // Also sink all related debug uses from the source basic block. Otherwise we
5605 // get debug use before the def. Attempt to salvage debug uses first, to
5606 // maximise the range variables have location for. If we cannot salvage, then
5607 // mark the location undef: we know it was supposed to receive a new location
5608 // here, but that computation has been sunk.
5609 SmallVector<DbgVariableRecord *, 2> DbgVariableRecords;
5610 findDbgUsers(I, DbgVariableRecords);
5611 if (!DbgVariableRecords.empty())
5612 tryToSinkInstructionDbgVariableRecords(I, InsertPos, SrcBlock, DestBlock,
5613 DbgVariableRecords);
5614
5615 // PS: there are numerous flaws with this behaviour, not least that right now
5616 // assignments can be re-ordered past other assignments to the same variable
5617 // if they use different Values. Creating more undef assignements can never be
5618 // undone. And salvaging all users outside of this block can un-necessarily
5619 // alter the lifetime of the live-value that the variable refers to.
5620 // Some of these things can be resolved by tolerating debug use-before-defs in
5621 // LLVM-IR, however it depends on the instruction-referencing CodeGen backend
5622 // being used for more architectures.
5623
5624 return true;
5625}
5626
5628 Instruction *I, BasicBlock::iterator InsertPos, BasicBlock *SrcBlock,
5629 BasicBlock *DestBlock,
5630 SmallVectorImpl<DbgVariableRecord *> &DbgVariableRecords) {
5631 // For all debug values in the destination block, the sunk instruction
5632 // will still be available, so they do not need to be dropped.
5633
5634 // Fetch all DbgVariableRecords not already in the destination.
5635 SmallVector<DbgVariableRecord *, 2> DbgVariableRecordsToSalvage;
5636 for (auto &DVR : DbgVariableRecords)
5637 if (DVR->getParent() != DestBlock)
5638 DbgVariableRecordsToSalvage.push_back(DVR);
5639
5640 // Fetch a second collection, of DbgVariableRecords in the source block that
5641 // we're going to sink.
5642 SmallVector<DbgVariableRecord *> DbgVariableRecordsToSink;
5643 for (DbgVariableRecord *DVR : DbgVariableRecordsToSalvage)
5644 if (DVR->getParent() == SrcBlock)
5645 DbgVariableRecordsToSink.push_back(DVR);
5646
5647 // Sort DbgVariableRecords according to their position in the block. This is a
5648 // partial order: DbgVariableRecords attached to different instructions will
5649 // be ordered by the instruction order, but DbgVariableRecords attached to the
5650 // same instruction won't have an order.
5651 auto Order = [](DbgVariableRecord *A, DbgVariableRecord *B) -> bool {
5652 return B->getInstruction()->comesBefore(A->getInstruction());
5653 };
5654 llvm::stable_sort(DbgVariableRecordsToSink, Order);
5655
5656 // If there are two assignments to the same variable attached to the same
5657 // instruction, the ordering between the two assignments is important. Scan
5658 // for this (rare) case and establish which is the last assignment.
5659 using InstVarPair = std::pair<const Instruction *, DebugVariable>;
5661 if (DbgVariableRecordsToSink.size() > 1) {
5663 // Count how many assignments to each variable there is per instruction.
5664 for (DbgVariableRecord *DVR : DbgVariableRecordsToSink) {
5665 DebugVariable DbgUserVariable =
5666 DebugVariable(DVR->getVariable(), DVR->getExpression(),
5667 DVR->getDebugLoc()->getInlinedAt());
5668 CountMap[std::make_pair(DVR->getInstruction(), DbgUserVariable)] += 1;
5669 }
5670
5671 // If there are any instructions with two assignments, add them to the
5672 // FilterOutMap to record that they need extra filtering.
5674 for (auto It : CountMap) {
5675 if (It.second > 1) {
5676 FilterOutMap[It.first] = nullptr;
5677 DupSet.insert(It.first.first);
5678 }
5679 }
5680
5681 // For all instruction/variable pairs needing extra filtering, find the
5682 // latest assignment.
5683 for (const Instruction *Inst : DupSet) {
5684 for (DbgVariableRecord &DVR :
5685 llvm::reverse(filterDbgVars(Inst->getDbgRecordRange()))) {
5686 DebugVariable DbgUserVariable =
5687 DebugVariable(DVR.getVariable(), DVR.getExpression(),
5688 DVR.getDebugLoc()->getInlinedAt());
5689 auto FilterIt =
5690 FilterOutMap.find(std::make_pair(Inst, DbgUserVariable));
5691 if (FilterIt == FilterOutMap.end())
5692 continue;
5693 if (FilterIt->second != nullptr)
5694 continue;
5695 FilterIt->second = &DVR;
5696 }
5697 }
5698 }
5699
5700 // Perform cloning of the DbgVariableRecords that we plan on sinking, filter
5701 // out any duplicate assignments identified above.
5703 SmallSet<DebugVariable, 4> SunkVariables;
5704 for (DbgVariableRecord *DVR : DbgVariableRecordsToSink) {
5706 continue;
5707
5708 DebugVariable DbgUserVariable =
5709 DebugVariable(DVR->getVariable(), DVR->getExpression(),
5710 DVR->getDebugLoc()->getInlinedAt());
5711
5712 // For any variable where there were multiple assignments in the same place,
5713 // ignore all but the last assignment.
5714 if (!FilterOutMap.empty()) {
5715 InstVarPair IVP = std::make_pair(DVR->getInstruction(), DbgUserVariable);
5716 auto It = FilterOutMap.find(IVP);
5717
5718 // Filter out.
5719 if (It != FilterOutMap.end() && It->second != DVR)
5720 continue;
5721 }
5722
5723 if (!SunkVariables.insert(DbgUserVariable).second)
5724 continue;
5725
5726 if (DVR->isDbgAssign())
5727 continue;
5728
5729 DVRClones.emplace_back(DVR->clone());
5730 LLVM_DEBUG(dbgs() << "CLONE: " << *DVRClones.back() << '\n');
5731 }
5732
5733 // Perform salvaging without the clones, then sink the clones.
5734 if (DVRClones.empty())
5735 return;
5736
5737 salvageDebugInfoForDbgValues(*I, DbgVariableRecordsToSalvage);
5738
5739 // The clones are in reverse order of original appearance. Assert that the
5740 // head bit is set on the iterator as we _should_ have received it via
5741 // getFirstInsertionPt. Inserting like this will reverse the clone order as
5742 // we'll repeatedly insert at the head, such as:
5743 // DVR-3 (third insertion goes here)
5744 // DVR-2 (second insertion goes here)
5745 // DVR-1 (first insertion goes here)
5746 // Any-Prior-DVRs
5747 // InsertPtInst
5748 assert(InsertPos.getHeadBit());
5749 for (DbgVariableRecord *DVRClone : DVRClones) {
5750 InsertPos->getParent()->insertDbgRecordBefore(DVRClone, InsertPos);
5751 LLVM_DEBUG(dbgs() << "SINK: " << *DVRClone << '\n');
5752 }
5753}
5754
5756 while (!Worklist.isEmpty()) {
5757 // Walk deferred instructions in reverse order, and push them to the
5758 // worklist, which means they'll end up popped from the worklist in-order.
5759 while (Instruction *I = Worklist.popDeferred()) {
5760 // Check to see if we can DCE the instruction. We do this already here to
5761 // reduce the number of uses and thus allow other folds to trigger.
5762 // Note that eraseInstFromFunction() may push additional instructions on
5763 // the deferred worklist, so this will DCE whole instruction chains.
5766 ++NumDeadInst;
5767 continue;
5768 }
5769
5770 Worklist.push(I);
5771 }
5772
5773 Instruction *I = Worklist.removeOne();
5774 if (I == nullptr) continue; // skip null values.
5775
5776 // Check to see if we can DCE the instruction.
5779 ++NumDeadInst;
5780 continue;
5781 }
5782
5783 if (!DebugCounter::shouldExecute(VisitCounter))
5784 continue;
5785
5786 // See if we can trivially sink this instruction to its user if we can
5787 // prove that the successor is not executed more frequently than our block.
5788 // Return the UserBlock if successful.
5789 auto getOptionalSinkBlockForInst =
5790 [this](Instruction *I) -> std::optional<BasicBlock *> {
5791 if (!CLOpts.code_sinking)
5792 return std::nullopt;
5793
5794 BasicBlock *BB = I->getParent();
5795 BasicBlock *UserParent = nullptr;
5796 unsigned NumUsers = 0;
5797
5798 for (Use &U : I->uses()) {
5799 User *User = U.getUser();
5800 if (User->isDroppable()) {
5801 // Do not sink if there are dereferenceable assumes that would be
5802 // removed.
5804 if (II->getIntrinsicID() != Intrinsic::assume ||
5805 !II->getOperandBundle("dereferenceable"))
5806 continue;
5807 }
5808
5809 if (NumUsers > CLOpts.max_sink_users)
5810 return std::nullopt;
5811
5812 Instruction *UserInst = cast<Instruction>(User);
5813 // Special handling for Phi nodes - get the block the use occurs in.
5814 BasicBlock *UserBB = UserInst->getParent();
5815 if (PHINode *PN = dyn_cast<PHINode>(UserInst))
5816 UserBB = PN->getIncomingBlock(U);
5817 // Bail out if we have uses in different blocks. We don't do any
5818 // sophisticated analysis (i.e finding NearestCommonDominator of these
5819 // use blocks).
5820 if (UserParent && UserParent != UserBB)
5821 return std::nullopt;
5822 UserParent = UserBB;
5823
5824 // Make sure these checks are done only once, naturally we do the checks
5825 // the first time we get the userparent, this will save compile time.
5826 if (NumUsers == 0) {
5827 // Try sinking to another block. If that block is unreachable, then do
5828 // not bother. SimplifyCFG should handle it.
5829 if (UserParent == BB || !DT.isReachableFromEntry(UserParent))
5830 return std::nullopt;
5831
5832 auto *Term = UserParent->getTerminator();
5833 // See if the user is one of our successors that has only one
5834 // predecessor, so that we don't have to split the critical edge.
5835 // Another option where we can sink is a block that ends with a
5836 // terminator that does not pass control to other block (such as
5837 // return or unreachable or resume). In this case:
5838 // - I dominates the User (by SSA form);
5839 // - the User will be executed at most once.
5840 // So sinking I down to User is always profitable or neutral.
5841 if (UserParent->getUniquePredecessor() != BB && !succ_empty(Term))
5842 return std::nullopt;
5843
5844 assert(DT.dominates(BB, UserParent) && "Dominance relation broken?");
5845 }
5846
5847 NumUsers++;
5848 }
5849
5850 // No user or only has droppable users.
5851 if (!UserParent)
5852 return std::nullopt;
5853
5854 return UserParent;
5855 };
5856
5857 auto OptBB = getOptionalSinkBlockForInst(I);
5858 if (OptBB) {
5859 auto *UserParent = *OptBB;
5860 // Okay, the CFG is simple enough, try to sink this instruction.
5861 if (tryToSinkInstruction(I, UserParent)) {
5862 LLVM_DEBUG(dbgs() << "IC: Sink: " << *I << '\n');
5863 MadeIRChange = true;
5864 // We'll add uses of the sunk instruction below, but since
5865 // sinking can expose opportunities for it's *operands* add
5866 // them to the worklist
5867 for (Use &U : I->operands())
5868 if (Instruction *OpI = dyn_cast<Instruction>(U.get()))
5869 Worklist.push(OpI);
5870 }
5871 }
5872
5873 // Now that we have an instruction, try combining it to simplify it.
5874 Builder.SetInsertPoint(I);
5875 Builder.SetCurrentDebugLocation(I->getDebugLoc());
5876 // Used by our IRBuilder inserter to copy annotation metadata.
5878
5879#ifndef NDEBUG
5880 std::string OrigI;
5881#endif
5882 LLVM_DEBUG(raw_string_ostream SS(OrigI); I->print(SS););
5883 LLVM_DEBUG(dbgs() << "IC: Visiting: " << OrigI << '\n');
5884
5885 if (Instruction *Result = visit(*I)) {
5886 ++NumCombined;
5887 // Should we replace the old instruction with a new one?
5888 if (Result != I) {
5889 LLVM_DEBUG(dbgs() << "IC: Old = " << *I << '\n'
5890 << " New = " << *Result << '\n');
5891
5892 // We copy the old instruction's DebugLoc to the new instruction, unless
5893 // InstCombine already assigned a DebugLoc to it, in which case we
5894 // should trust the more specifically selected DebugLoc.
5895 Result->setDebugLoc(Result->getDebugLoc().orElse(I->getDebugLoc()));
5896 // We also copy annotation metadata to the new instruction.
5897 Result->copyMetadata(*I, LLVMContext::MD_annotation);
5898 // Everything uses the new instruction now.
5899 I->replaceAllUsesWith(Result);
5900
5901 // Move the name to the new instruction first.
5902 Result->takeName(I);
5903
5904 // Insert the new instruction into the basic block...
5905 BasicBlock *InstParent = I->getParent();
5906 BasicBlock::iterator InsertPos = I->getIterator();
5907
5908 // Are we replace a PHI with something that isn't a PHI, or vice versa?
5909 if (isa<PHINode>(Result) != isa<PHINode>(I)) {
5910 // We need to fix up the insertion point.
5911 if (isa<PHINode>(I)) // PHI -> Non-PHI
5912 InsertPos = InstParent->getFirstInsertionPt();
5913 else // Non-PHI -> PHI
5914 InsertPos = InstParent->getFirstNonPHIIt();
5915 }
5916
5917 Result->insertInto(InstParent, InsertPos);
5918
5919 // Register newly created assumptions.
5920 if (auto *Assume = dyn_cast<AssumeInst>(Result))
5921 AC.registerAssumption(Assume);
5922
5923 // Push the new instruction and any users onto the worklist.
5924 Worklist.pushUsersToWorkList(*Result);
5925 Worklist.push(Result);
5926
5928 } else {
5929 LLVM_DEBUG(dbgs() << "IC: Mod = " << OrigI << '\n'
5930 << " New = " << *I << '\n');
5931
5932 // If the instruction was modified, it's possible that it is now dead.
5933 // if so, remove it.
5936 } else {
5937 Worklist.pushUsersToWorkList(*I);
5938 Worklist.push(I);
5939 }
5940 }
5941 MadeIRChange = true;
5942 }
5943 }
5944
5945 Worklist.zap();
5946 return MadeIRChange;
5947}
5948
5949// Track the scopes used by !alias.scope and !noalias. In a function, a
5950// @llvm.experimental.noalias.scope.decl is only useful if that scope is used
5951// by both sets. If not, the declaration of the scope can be safely omitted.
5952// The MDNode of the scope can be omitted as well for the instructions that are
5953// part of this function. We do not do that at this point, as this might become
5954// too time consuming to do.
5956 SmallPtrSet<const MDNode *, 8> UsedAliasScopesAndLists;
5957 SmallPtrSet<const MDNode *, 8> UsedNoAliasScopesAndLists;
5958 // Scopes used by every !alias.scope list that scopes from a disjoint-scope
5959 // domain appears in. This is used to catch scopes that don't actually make
5960 // anything noalias.
5962 CommonScopesOfDisjointDomain;
5963
5964 // Record, for each disjoint-scope domain \p ScopeList uses, which of its
5965 // scopes are used by \p ScopeList, adding to a running intersection.
5966 void recordDisjointDomainScopes(const MDNode *ScopeList) {
5968 for (const MDOperand &MDOperand : ScopeList->operands()) {
5969 const auto *MDScope = cast<MDNode>(MDOperand);
5970 const MDNode *Domain = AliasScopeNode(MDScope).getDomain();
5971 if (AliasScopeDomainNode(Domain).hasDisjointScopes())
5972 UsedScopes[Domain].insert(MDScope);
5973 }
5974
5975 for (auto &[Domain, Scopes] : UsedScopes) {
5976 auto [It, Inserted] =
5977 CommonScopesOfDisjointDomain.try_emplace(Domain, Scopes);
5978 if (!Inserted)
5979 llvm::set_intersect(It->second, Scopes);
5980 }
5981 }
5982
5983 // Return true if \p Scope is on the implicit !noalias list of one of the
5984 // analysed accesses, that is, if it belongs to a disjoint-scope domain and
5985 // some access uses that domain without using \p Scope.
5986 bool isImplicitlyNoAlias(const MDNode *Scope) const {
5987 auto It =
5988 CommonScopesOfDisjointDomain.find(AliasScopeNode(Scope).getDomain());
5989 return It != CommonScopesOfDisjointDomain.end() &&
5990 !It->second.contains(Scope);
5991 }
5992
5993public:
5995 // This seems to be faster than checking 'mayReadOrWriteMemory()'.
5996 if (!I->hasMetadataOtherThanDebugLoc())
5997 return;
5998
5999 auto Track = [](Metadata *ScopeList, auto &Container) -> const MDNode * {
6000 const auto *MDScopeList = dyn_cast_or_null<MDNode>(ScopeList);
6001 if (!MDScopeList || !Container.insert(MDScopeList).second)
6002 return nullptr;
6003 for (const auto &MDOperand : MDScopeList->operands())
6004 if (auto *MDScope = dyn_cast<MDNode>(MDOperand))
6005 Container.insert(MDScope);
6006 return MDScopeList;
6007 };
6008
6009 if (const MDNode *AliasScopeList =
6010 Track(I->getMetadata(LLVMContext::MD_alias_scope),
6011 UsedAliasScopesAndLists))
6012 recordDisjointDomainScopes(AliasScopeList);
6013 Track(I->getMetadata(LLVMContext::MD_noalias), UsedNoAliasScopesAndLists);
6014 }
6015
6018 if (!Decl)
6019 return false;
6020
6021 assert(Decl->use_empty() &&
6022 "llvm.experimental.noalias.scope.decl in use ?");
6023 const MDNode *MDSL = Decl->getScopeList();
6024 assert(MDSL->getNumOperands() == 1 &&
6025 "llvm.experimental.noalias.scope should refer to a single scope");
6026 auto &MDOperand = MDSL->getOperand(0);
6027 // A scope is relevant if it appears in an !alias.scope list, and either it
6028 // appears in a !noalias list, or it is on the implicit !noalias list of
6029 // some access using its disjoint-scope domain.
6030 if (auto *MD = dyn_cast<MDNode>(MDOperand))
6031 return !UsedAliasScopesAndLists.contains(MD) ||
6032 (!UsedNoAliasScopesAndLists.contains(MD) &&
6033 !isImplicitlyNoAlias(MD));
6034
6035 // Not an MDNode ? throw away.
6036 return true;
6037 }
6038};
6039
6040/// Populate the IC worklist from a function, by walking it in reverse
6041/// post-order and adding all reachable code to the worklist.
6042///
6043/// This has a couple of tricks to make the code faster and more powerful. In
6044/// particular, we constant fold and DCE instructions as we go, to avoid adding
6045/// them to the worklist (this significantly speeds up instcombine on code where
6046/// many instructions are dead or constant). Additionally, if we find a branch
6047/// whose condition is a known constant, we only visit the reachable successors.
6049 bool MadeIRChange = false;
6051 SmallVector<Instruction *, 128> InstrsForInstructionWorklist;
6052 DenseMap<Constant *, Constant *> FoldedConstants;
6053 AliasScopeTracker SeenAliasScopes;
6054
6055 auto HandleOnlyLiveSuccessor = [&](BasicBlock *BB, BasicBlock *LiveSucc) {
6056 for (BasicBlock *Succ : successors(BB))
6057 if (Succ != LiveSucc && DeadEdges.insert({BB, Succ}).second)
6058 for (PHINode &PN : Succ->phis())
6059 for (Use &U : PN.incoming_values())
6060 if (PN.getIncomingBlock(U) == BB && !isa<PoisonValue>(U)) {
6061 U.set(PoisonValue::get(PN.getType()));
6062 MadeIRChange = true;
6063 }
6064 };
6065
6066 for (BasicBlock *BB : RPOT) {
6067 if (!BB->isEntryBlock() && all_of(predecessors(BB), [&](BasicBlock *Pred) {
6068 return DeadEdges.contains({Pred, BB}) || DT.dominates(BB, Pred);
6069 })) {
6070 HandleOnlyLiveSuccessor(BB, nullptr);
6071 continue;
6072 }
6073 LiveBlocks.insert(BB);
6074
6075 for (Instruction &Inst : llvm::make_early_inc_range(*BB)) {
6076 // ConstantProp instruction if trivially constant.
6077 if (!Inst.use_empty() &&
6078 (Inst.getNumOperands() == 0 || isa<Constant>(Inst.getOperand(0))))
6079 if (Constant *C = ConstantFoldInstruction(&Inst, DL, &TLI)) {
6080 LLVM_DEBUG(dbgs() << "IC: ConstFold to: " << *C << " from: " << Inst
6081 << '\n');
6082 Inst.replaceAllUsesWith(C);
6083 ++NumConstProp;
6084 if (isInstructionTriviallyDead(&Inst, &TLI))
6085 Inst.eraseFromParent();
6086 MadeIRChange = true;
6087 continue;
6088 }
6089
6090 // See if we can constant fold its operands.
6091 for (Use &U : Inst.operands()) {
6093 continue;
6094
6095 auto *C = cast<Constant>(U);
6096 Constant *&FoldRes = FoldedConstants[C];
6097 if (!FoldRes)
6098 FoldRes = ConstantFoldConstant(C, DL, &TLI);
6099
6100 if (FoldRes != C) {
6101 LLVM_DEBUG(dbgs() << "IC: ConstFold operand of: " << Inst
6102 << "\n Old = " << *C
6103 << "\n New = " << *FoldRes << '\n');
6104 U = FoldRes;
6105 MadeIRChange = true;
6106 }
6107 }
6108
6109 // Skip processing debug and pseudo intrinsics in InstCombine. Processing
6110 // these call instructions consumes non-trivial amount of time and
6111 // provides no value for the optimization.
6112 if (!Inst.isDebugOrPseudoInst()) {
6113 InstrsForInstructionWorklist.push_back(&Inst);
6114 SeenAliasScopes.analyse(&Inst);
6115 }
6116 }
6117
6118 // If this is a branch or switch on a constant, mark only the single
6119 // live successor. Otherwise assume all successors are live.
6120 Instruction *TI = BB->getTerminator();
6121 if (CondBrInst *BI = dyn_cast<CondBrInst>(TI)) {
6122 if (isa<UndefValue>(BI->getCondition())) {
6123 // Branch on undef is UB.
6124 HandleOnlyLiveSuccessor(BB, nullptr);
6125 continue;
6126 }
6127 if (auto *Cond = dyn_cast<ConstantInt>(BI->getCondition())) {
6128 bool CondVal = Cond->getZExtValue();
6129 HandleOnlyLiveSuccessor(BB, BI->getSuccessor(!CondVal));
6130 continue;
6131 }
6132 } else if (SwitchInst *SI = dyn_cast<SwitchInst>(TI)) {
6133 if (isa<UndefValue>(SI->getCondition())) {
6134 // Switch on undef is UB.
6135 HandleOnlyLiveSuccessor(BB, nullptr);
6136 continue;
6137 }
6138 if (auto *Cond = dyn_cast<ConstantInt>(SI->getCondition())) {
6139 HandleOnlyLiveSuccessor(BB,
6140 SI->findCaseValue(Cond)->getCaseSuccessor());
6141 continue;
6142 }
6143 }
6144 }
6145
6146 // Remove instructions inside unreachable blocks. This prevents the
6147 // instcombine code from having to deal with some bad special cases, and
6148 // reduces use counts of instructions.
6149 for (BasicBlock &BB : F) {
6150 if (LiveBlocks.count(&BB))
6151 continue;
6152
6153 unsigned NumDeadInstInBB;
6154 NumDeadInstInBB = removeAllNonTerminatorAndEHPadInstructions(&BB);
6155
6156 MadeIRChange |= NumDeadInstInBB != 0;
6157 NumDeadInst += NumDeadInstInBB;
6158 }
6159
6160 // Once we've found all of the instructions to add to instcombine's worklist,
6161 // add them in reverse order. This way instcombine will visit from the top
6162 // of the function down. This jives well with the way that it adds all uses
6163 // of instructions to the worklist after doing a transformation, thus avoiding
6164 // some N^2 behavior in pathological cases.
6165 Worklist.reserve(InstrsForInstructionWorklist.size());
6166 for (Instruction *Inst : reverse(InstrsForInstructionWorklist)) {
6167 // DCE instruction if trivially dead. As we iterate in reverse program
6168 // order here, we will clean up whole chains of dead instructions.
6169 if (isInstructionTriviallyDead(Inst, &TLI) ||
6170 SeenAliasScopes.isNoAliasScopeDeclDead(Inst)) {
6171 ++NumDeadInst;
6172 LLVM_DEBUG(dbgs() << "IC: DCE: " << *Inst << '\n');
6173 salvageDebugInfo(*Inst);
6174 Inst->eraseFromParent();
6175 MadeIRChange = true;
6176 continue;
6177 }
6178
6179 Worklist.push(Inst);
6180 }
6181
6182 return MadeIRChange;
6183}
6184
6186 // Collect backedges.
6187 SmallVector<bool> Visited(F.getMaxBlockNumber());
6188 for (BasicBlock *BB : RPOT) {
6189 Visited[BB->getNumber()] = true;
6190 for (BasicBlock *Succ : successors(BB))
6191 if (Visited[Succ->getNumber()])
6192 BackEdges.insert({BB, Succ});
6193 }
6194 ComputedBackEdges = true;
6195}
6196
6202 const InstCombineOptions &Opts) {
6203 auto &DL = F.getDataLayout();
6204 bool VerifyFixpoint = Opts.VerifyFixpoint &&
6205 !F.hasFnAttribute("instcombine-no-verify-fixpoint");
6206
6208
6209 // Lower dbg.declare intrinsics otherwise their value may be clobbered
6210 // by instcombiner.
6211 const InstCombineCLOptions &CLOpts = InstCombineCLOptions::Global;
6212 bool MadeIRChange = false;
6213 if (CLOpts.lower_dbg_declare)
6214 MadeIRChange = LowerDbgDeclare(F);
6215
6216 // Iterate while there is work to do.
6217 unsigned Iteration = 0;
6218 while (true) {
6219 if (Iteration >= Opts.MaxIterations && !VerifyFixpoint) {
6220 LLVM_DEBUG(dbgs() << "\n\n[IC] Iteration limit #" << Opts.MaxIterations
6221 << " on " << F.getName()
6222 << " reached; stopping without verifying fixpoint\n");
6223 break;
6224 }
6225
6226 ++Iteration;
6227 ++NumWorklistIterations;
6228 LLVM_DEBUG(dbgs() << "\n\nINSTCOMBINE ITERATION #" << Iteration << " on "
6229 << F.getName() << "\n");
6230
6231 InstCombinerImpl IC(Worklist, F, AA, AC, TLI, TTI, DT, ORE, BFI, BPI, PSI,
6232 DL, RPOT, CLOpts);
6233 bool MadeChangeInThisIteration = IC.prepareWorklist(F);
6234 MadeChangeInThisIteration |= IC.run();
6235 if (!MadeChangeInThisIteration)
6236 break;
6237
6238 MadeIRChange = true;
6239 if (Iteration > Opts.MaxIterations) {
6241 "Instruction Combining on " + Twine(F.getName()) +
6242 " did not reach a fixpoint after " + Twine(Opts.MaxIterations) +
6243 " iterations. " +
6244 "Use 'instcombine<no-verify-fixpoint>' or function attribute "
6245 "'instcombine-no-verify-fixpoint' to suppress this error.");
6246 }
6247 }
6248
6249 if (Iteration == 1)
6250 ++NumOneIteration;
6251 else if (Iteration == 2)
6252 ++NumTwoIterations;
6253 else if (Iteration == 3)
6254 ++NumThreeIterations;
6255 else
6256 ++NumFourOrMoreIterations;
6257
6258 return MadeIRChange;
6259}
6260
6262
6264 raw_ostream &OS, function_ref<StringRef(StringRef)> MapClassName2PassName) {
6265 static_cast<PassInfoMixin<InstCombinePass> *>(this)->printPipeline(
6266 OS, MapClassName2PassName);
6267 OS << '<';
6268 OS << "max-iterations=" << Options.MaxIterations << ";";
6269 OS << (Options.VerifyFixpoint ? "" : "no-") << "verify-fixpoint";
6270 OS << '>';
6271}
6272
6273char InstCombinePass::ID = 0;
6274
6277 auto &LRT = AM.getResult<LastRunTrackingAnalysis>(F);
6278 // No changes since last InstCombine pass, exit early.
6279 if (LRT.shouldSkip(&ID))
6280 return PreservedAnalyses::all();
6281
6282 auto &AC = AM.getResult<AssumptionAnalysis>(F);
6283 auto &DT = AM.getResult<DominatorTreeAnalysis>(F);
6284 auto &TLI = AM.getResult<TargetLibraryAnalysis>(F);
6286 auto &TTI = AM.getResult<TargetIRAnalysis>(F);
6287
6288 auto *AA = &AM.getResult<AAManager>(F);
6289 auto &MAMProxy = AM.getResult<ModuleAnalysisManagerFunctionProxy>(F);
6290 ProfileSummaryInfo *PSI =
6291 MAMProxy.getCachedResult<ProfileSummaryAnalysis>(*F.getParent());
6292 auto *BFI = (PSI && PSI->hasProfileSummary()) ?
6293 &AM.getResult<BlockFrequencyAnalysis>(F) : nullptr;
6295
6296 if (!combineInstructionsOverFunction(F, Worklist, AA, AC, TLI, TTI, DT, ORE,
6297 BFI, BPI, PSI, Options)) {
6298 // No changes, all analyses are preserved.
6299 LRT.update(&ID, /*Changed=*/false);
6300 return PreservedAnalyses::all();
6301 }
6302
6303 // Mark all the analyses that instcombine updates as preserved.
6305 LRT.update(&ID, /*Changed=*/true);
6308 return PA;
6309}
6310
6324
6326 if (skipFunction(F))
6327 return false;
6328
6329 // Required analyses.
6330 auto AA = &getAnalysis<AAResultsWrapperPass>().getAAResults();
6331 auto &AC = getAnalysis<AssumptionCacheTracker>().getAssumptionCache(F);
6332 auto &TLI = getAnalysis<TargetLibraryInfoWrapperPass>().getTLI(F);
6334 auto &DT = getAnalysis<DominatorTreeWrapperPass>().getDomTree();
6336
6337 // Optional analyses.
6338 ProfileSummaryInfo *PSI =
6340 BlockFrequencyInfo *BFI =
6341 (PSI && PSI->hasProfileSummary()) ?
6343 nullptr;
6344 BranchProbabilityInfo *BPI = nullptr;
6345 if (auto *WrapperPass =
6347 BPI = &WrapperPass->getBPI();
6348
6349 return combineInstructionsOverFunction(F, Worklist, AA, AC, TLI, TTI, DT, ORE,
6350 BFI, BPI, PSI, InstCombineOptions());
6351}
6352
6354
6356
6358 "Combine redundant instructions", false, false)
6369 "Combine redundant instructions", false, false)
6370
6371// Initialization Routines.
6375
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
AMDGPU Register Bank Select
Rewrite undef for PHI
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This is the interface for LLVM's primary stateless and local alias analysis.
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static bool willNotOverflow(BinaryOpIntrinsic *BO, LazyValueInfo *LVI)
static Domain getDomain(const ConstantRange &CR)
DXIL Resource Access
This file provides an implementation of debug counters.
#define DEBUG_COUNTER(VARNAME, COUNTERNAME, DESC)
This file defines the DenseMap class.
static bool isSigned(unsigned Opcode)
This is the interface for a simple mod/ref and alias analysis over globals.
Hexagon Common GEP
IRTranslator LLVM IR MI
This file provides various utilities for inspecting and working with the control flow graph in LLVM I...
This header defines various interfaces for pass management in LLVM.
This defines the Use class.
iv Induction Variable Users
Definition IVUsers.cpp:48
static bool rightDistributesOverLeft(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "(X ROp Y) LOp Z" is always equal to "(X LOp Z) ROp (Y LOp Z)".
static bool leftDistributesOverRight(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "X LOp (Y ROp Z)" is always equal to "(X LOp Y) ROp (X LOp Z)".
This file provides internal interfaces used to implement the InstCombine.
This file provides the primary interface to the instcombine pass.
static Value * simplifySwitchOnSelectUsingRanges(SwitchInst &SI, SelectInst *Select, bool IsTrueArm)
static bool isUsedWithinShuffleVector(Value *V)
static bool isNeverEqualToUnescapedAlloc(Value *V, const TargetLibraryInfo &TLI, Instruction *AI)
static Constant * constantFoldBinOpWithSplat(unsigned Opcode, Constant *Vector, Constant *Splat, bool SplatLHS, const DataLayout &DL)
static bool shorter_filter(const Value *LHS, const Value *RHS)
static Instruction * combineConstantOffsets(GetElementPtrInst &GEP, InstCombinerImpl &IC)
Combine constant offsets separated by variable offsets.
static std::optional< ModRefInfo > isAllocSiteRemovable(Instruction *AI, SmallVectorImpl< Instruction * > &Users, const TargetLibraryInfo &TLI, bool KnowInit, unsigned MaxUsers)
static Instruction * foldSelectGEP(GetElementPtrInst &GEP, InstCombiner::BuilderTy &Builder)
Thread a GEP operation with constant indices through the constant true/false arms of a select.
static bool shouldMergeGEPs(GEPOperator &GEP, GEPOperator &Src)
static Instruction * foldSpliceBinOp(BinaryOperator &Inst, InstCombiner::BuilderTy &Builder)
static bool hasNoSignedWrap(BinaryOperator &I)
static bool simplifyAssocCastAssoc(BinaryOperator *BinOp1, InstCombinerImpl &IC)
Combine constant operands of associative operations either before or after a cast to eliminate one of...
static bool combineInstructionsOverFunction(Function &F, InstructionWorklist &Worklist, AliasAnalysis *AA, AssumptionCache &AC, TargetLibraryInfo &TLI, TargetTransformInfo &TTI, DominatorTree &DT, OptimizationRemarkEmitter &ORE, BlockFrequencyInfo *BFI, BranchProbabilityInfo *BPI, ProfileSummaryInfo *PSI, const InstCombineOptions &Opts)
static Value * simplifyInstructionWithPHI(Instruction &I, PHINode *PN, Value *InValue, BasicBlock *InBB, const DataLayout &DL, const SimplifyQuery SQ)
static bool shouldCanonicalizeGEPToPtrAdd(GetElementPtrInst &GEP)
Return true if we should canonicalize the gep to an i8 ptradd.
static Value * getIdentityValue(Instruction::BinaryOps Opcode, Value *V)
This function returns identity value for given opcode, which can be used to factor patterns like (X *...
static Value * foldFrexpOfSelect(ExtractValueInst &EV, IntrinsicInst *FrexpCall, SelectInst *SelectInst, InstCombiner::BuilderTy &Builder)
static std::optional< std::pair< Value *, Value * > > matchSymmetricPhiNodesPair(PHINode *LHS, PHINode *RHS)
static Value * foldOperationIntoSelectOperand(Instruction &I, SelectInst *SI, Value *NewOp, InstCombiner &IC)
static Instruction * canonicalizeGEPOfConstGEPI8(GetElementPtrInst &GEP, GEPOperator *Src, InstCombinerImpl &IC)
static Instruction * tryToMoveFreeBeforeNullTest(CallInst &FI, const DataLayout &DL)
Move the call to free before a NULL test.
static Value * simplifyOperationIntoSelectOperand(Instruction &I, SelectInst *SI, bool IsTrueArm)
static Value * tryFactorization(BinaryOperator &I, const SimplifyQuery &SQ, InstCombiner::BuilderTy &Builder, Instruction::BinaryOps InnerOpcode, Value *A, Value *B, Value *C, Value *D)
This tries to simplify binary operations by factorizing out common terms (e.
static bool isRemovableWrite(CallBase &CB, Value *UsedV, const TargetLibraryInfo &TLI)
Given a call CB which uses an address UsedV, return true if we can prove the call's only possible eff...
static Instruction::BinaryOps getBinOpsForFactorization(Instruction::BinaryOps TopOpcode, BinaryOperator *Op, Value *&LHS, Value *&RHS, BinaryOperator *OtherOp)
This function predicates factorization using distributive laws.
static bool hasNoUnsignedWrap(BinaryOperator &I)
static bool SoleWriteToDeadLocal(Instruction *I, TargetLibraryInfo &TLI)
Check for case where the call writes to an otherwise dead alloca.
static Instruction * foldGEPOfPhi(GetElementPtrInst &GEP, PHINode *PN, IRBuilderBase &Builder)
static bool isCatchAll(EHPersonality Personality, Constant *TypeInfo)
Return 'true' if the given typeinfo will match anything.
static bool maintainNoSignedWrap(BinaryOperator &I, Value *B, Value *C)
static GEPNoWrapFlags getMergedGEPNoWrapFlags(GEPOperator &GEP1, GEPOperator &GEP2)
Determine nowrap flags for (gep (gep p, x), y) to (gep p, (x + y)) transform.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
This file contains the declarations for metadata subclasses.
#define T
uint64_t IntrinsicInst * II
static bool IsSelect(unsigned Opcode, bool CheckOnlyCC=false)
Check if the opcode is a SELECT or SELECT_CC variant.
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
const SmallVectorImpl< MachineOperand > & Cond
BaseType
A given derived pointer can have multiple base pointers through phi/selects.
This file defines generic set operations that may be used on set's of different types,...
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
static unsigned getScalarSizeInBits(Type *Ty)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
This pass exposes codegen information to IR-level passes.
Value * RHS
Value * LHS
static const uint32_t IV[8]
Definition blake3_impl.h:83
bool isNoAliasScopeDeclDead(Instruction *Inst)
void analyse(Instruction *I)
The Input class is used to parse a yaml document into in-memory structs and vectors.
A manager for alias analyses.
A wrapper pass to provide the legacy pass manager access to a suitably prepared AAResults object.
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static LLVM_ABI unsigned int semanticsPrecision(const fltSemantics &)
Definition APFloat.cpp:329
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
static LLVM_ABI void udivrem(const APInt &LHS, const APInt &RHS, APInt &Quotient, APInt &Remainder)
Dual division/remainder interface.
Definition APInt.cpp:1796
bool isMinSignedValue() const
Determine if this is the smallest signed value.
Definition APInt.h:419
static LLVM_ABI void sdivrem(const APInt &LHS, const APInt &RHS, APInt &Quotient, APInt &Remainder)
Definition APInt.cpp:1928
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
bool isAllOnes() const
Determine if all bits are set. This is true for zero-width values.
Definition APInt.h:367
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1966
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
Definition APInt.h:829
LLVM_ABI APInt smul_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1998
bool isMaxSignedValue() const
Determine if this is the largest signed value.
Definition APInt.h:401
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:330
bool ule(const APInt &RHS) const
Unsigned less or equal comparison.
Definition APInt.h:1154
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1979
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:853
Wrapper around alias scope domain metedata to allow accessing their fields, including surfacing the o...
Definition Metadata.h:1603
This is a simple wrapper around an MDNode which provides a higher-level interface by hiding the detai...
Definition Metadata.h:1633
const MDNode * getDomain() const
Get the MDNode for this AliasScopeNode's domain.
Definition Metadata.h:1644
PassT::Result * getCachedResult(IRUnitT &IR) const
Get the cached result of an analysis pass for a given IR unit.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Definition Pass.cpp:278
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > take_front(size_t N=1) const
Return a copy of *this with only the first N elements.
Definition ArrayRef.h:218
size_t size() const
Get the array size.
Definition ArrayRef.h:141
Class to represent array types.
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
uint64_t getNumElements() const
Type * getElementType() const
A function analysis which provides an AssumptionCache.
An immutable pass that tracks lazily created AssumptionCache objects.
A cache of @llvm.assume calls within a function.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
LLVM_ABI uint64_t getDereferenceableBytes() const
Returns the number of dereferenceable bytes from the dereferenceable attribute.
bool isValid() const
Return true if the attribute is any kind of attribute.
Definition Attributes.h:266
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator_range< const_phi_iterator > phis() const
Returns a range that iterates over the phis in the basic block.
Definition BasicBlock.h:515
LLVM_ABI const_iterator getFirstInsertionPt() const
Returns an iterator to the first instruction in this block that is suitable for inserting a non-PHI i...
LLVM_ABI InstListType::const_iterator getFirstNonPHIIt() const
Returns an iterator to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI bool isEntryBlock() const
Return true if this is the entry block of the containing function.
LLVM_ABI const BasicBlock * getSinglePredecessor() const
Return the predecessor of this block if it has a single predecessor block.
const Instruction & front() const
Definition BasicBlock.h:469
LLVM_ABI const BasicBlock * getUniquePredecessor() const
Return the predecessor of this block if it has a unique predecessor block.
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
LLVM_ABI const_iterator getFirstNonPHIOrDbgOrAlloca() const
Returns an iterator to the first instruction in this block that is not a PHINode, a debug intrinsic,...
size_t size() const
Definition BasicBlock.h:467
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Definition BasicBlock.h:237
static LLVM_ABI BinaryOperator * CreateNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Helper functions to construct and inspect unary operations (NEG and NOT) via binary operators SUB and...
BinaryOps getOpcode() const
Definition InstrTypes.h:409
static LLVM_ABI BinaryOperator * Create(BinaryOps Op, Value *S1, Value *S2, const Twine &Name=Twine(), InsertPosition InsertBefore=nullptr)
Construct a binary instruction, given the opcode and the two operands.
static BinaryOperator * CreateNUW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:329
Analysis pass which computes BlockFrequencyInfo.
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
Analysis pass which computes BranchProbabilityInfo.
Analysis providing branch probability information.
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
void setAttributes(AttributeList A)
Set the attributes for this call.
bool doesNotThrow() const
Determine if the call cannot unwind.
Value * getArgOperand(unsigned i) const
AttributeList getAttributes() const
Return the attributes for this call.
This class represents a function call, abstracting a target machine's calling convention.
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ ICMP_NE
not equal
Definition InstrTypes.h:762
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Definition InstrTypes.h:852
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
Conditional Branch instruction.
LLVM_ABI void swapSuccessors()
Swap the successors of this branch instruction.
Value * getCondition() const
BasicBlock * getSuccessor(unsigned i) const
ConstantArray - Constant Array Declarations.
Definition Constants.h:590
static LLVM_ABI Constant * get(ArrayType *T, ArrayRef< Constant * > V)
A vector constant whose element type is a simple 1/2/4/8-byte integer or float/double,...
Definition Constants.h:951
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static LLVM_ABI Constant * getNot(Constant *C)
static LLVM_ABI Constant * getAdd(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static LLVM_ABI Constant * getBinOpIdentity(unsigned Opcode, Type *Ty, bool AllowRHSConstant=false, bool NSZ=false)
Return the identity constant for a binary opcode.
static LLVM_ABI Constant * getNeg(Constant *C, bool HasNSW=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
This class represents a range of values.
LLVM_ABI bool getEquivalentICmp(CmpInst::Predicate &Pred, APInt &RHS) const
Set up Pred and RHS such that ConstantRange::makeExactICmpRegion(Pred, RHS) == *this.
static LLVM_ABI ConstantRange makeExactICmpRegion(CmpInst::Predicate Pred, const APInt &Other)
Produce the exact range such that all values in the returned range satisfy the given predicate with a...
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
static LLVM_ABI ConstantRange makeExactNoWrapRegion(Instruction::BinaryOps BinOp, const APInt &Other, unsigned NoWrapKind)
Produce the range that contains X if and only if "X BinOp Other" does not wrap.
Constant Vector Declarations.
Definition Constants.h:674
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * replaceUndefsWith(Constant *C, Constant *Replacement)
Try to replace undefined constant C or undefined elements in C with Replacement.
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
const Constant * stripPointerCasts() const
Definition Constant.h:237
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
LLVM_ABI Constant * getAggregateElement(unsigned Elt) const
For aggregates (struct/array/vector) return the constant that corresponds to the specified element if...
static LLVM_ABI DIExpression * appendOpsToArg(const DIExpression *Expr, ArrayRef< uint64_t > Ops, unsigned ArgNo, bool StackValue=false)
Create a copy of Expr by appending the given list of Ops to each instance of the operand DW_OP_LLVM_a...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Record of a variable value-assignment, aka a non instruction representation of the dbg....
static bool shouldExecute(CounterInfo &Counter)
Identifies a unique instance of a variable.
bool empty() const
Definition DenseMap.h:732
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:782
iterator end()
Definition DenseMap.h:702
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
Definition DenseMap.h:809
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition DenseMap.h:843
Analysis pass which computes a DominatorTree.
Definition Dominators.h:241
Legacy analysis pass which computes a DominatorTree.
Definition Dominators.h:277
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
This instruction extracts a struct member or array element value from an aggregate value.
ArrayRef< unsigned > getIndices() const
iterator_range< idx_iterator > indices() const
idx_iterator idx_end() const
static ExtractValueInst * Create(Value *Agg, ArrayRef< unsigned > Idxs, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
idx_iterator idx_begin() const
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
This class represents a freeze function that returns random concrete value if an operand is either a ...
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
FunctionPass(char &pid)
Definition Pass.h:316
bool skipFunction(const Function &F) const
Optional passes call this function to check whether the pass should be skipped.
Definition Pass.cpp:196
const BasicBlock & getEntryBlock() const
Definition Function.h:794
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags inBounds()
static GEPNoWrapFlags all()
static GEPNoWrapFlags noUnsignedWrap()
GEPNoWrapFlags intersectForReassociate(GEPNoWrapFlags Other) const
Given (gep (gep p, x), y), determine the nowrap flags for (gep (gep, p, y), x).
bool hasNoUnsignedWrap() const
bool isInBounds() const
GEPNoWrapFlags intersectForOffsetAdd(GEPNoWrapFlags Other) const
Given (gep (gep p, x), y), determine the nowrap flags for (gep p, x+y).
static GEPNoWrapFlags none()
GEPNoWrapFlags getNoWrapFlags() const
Definition Operator.h:385
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
static LLVM_ABI Type * getTypeAtIndex(Type *Ty, Value *Idx)
Return the type of the element at the given index of an indexable type.
static GetElementPtrInst * Create(Type *PointeeType, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
static LLVM_ABI Type * getIndexedType(Type *Ty, ArrayRef< Value * > IdxList)
Returns the result type of a getelementptr with the given source element type and indexes.
static GetElementPtrInst * CreateInBounds(Type *PointeeType, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Create an "inbounds" getelementptr.
Legacy wrapper pass to provide the GlobalsAAResult object.
This instruction compares its operands according to the predicate given to the constructor.
CmpPredicate getCmpPredicate() const
static bool isEquality(Predicate P)
Return true if this predicate is either EQ or NE.
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
Value * CreatePtrAdd(Value *Ptr, Value *Offset, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
Definition IRBuilder.h:2084
ConstantInt * getInt(const APInt &AI)
Get a constant integer value.
Definition IRBuilder.h:471
virtual void InsertHelper(Instruction *I, const Twine &Name, BasicBlock::iterator InsertPt) const
Definition IRBuilder.h:65
This instruction inserts a struct field of array element value into an aggregate value.
static InsertValueInst * Create(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
LLVM_ABI InstCombinePass(InstCombineOptions Opts={})
LLVM_ABI void printPipeline(raw_ostream &OS, function_ref< StringRef(StringRef)> MapClassName2PassName)
LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)
Instruction * foldBinOpOfSelectAndCastOfSelectCondition(BinaryOperator &I)
Tries to simplify binops of select and cast of the select condition.
Instruction * visitCondBrInst(CondBrInst &BI)
Instruction * foldBinOpIntoSelectOrPhi(BinaryOperator &I)
This is a convenience wrapper function for the above two functions.
bool SimplifyAssociativeOrCommutative(BinaryOperator &I)
Performs a few simplifications for operators which are associative or commutative.
Instruction * visitGEPOfGEP(GetElementPtrInst &GEP, GEPOperator *Src)
Value * foldUsingDistributiveLaws(BinaryOperator &I)
Tries to simplify binary operations which some other binary operation distributes over.
Instruction * foldBinOpShiftWithShift(BinaryOperator &I)
Instruction * visitUnreachableInst(UnreachableInst &I)
Instruction * foldOpIntoPhi(Instruction &I, PHINode *PN, bool AllowMultipleUses=false)
Given a binary operator, cast instruction, or select which has a PHI node as operand #0,...
void handleUnreachableFrom(Instruction *I, SmallVectorImpl< BasicBlock * > &Worklist)
Value * SimplifyDemandedVectorElts(Value *V, APInt DemandedElts, APInt &PoisonElts, unsigned Depth=0, bool AllowMultipleUsers=false) override
The specified value produces a vector with any number of elements.
Instruction * visitFreeze(FreezeInst &I)
Instruction * foldBinOpSelectBinOp(BinaryOperator &Op)
In some cases it is beneficial to fold a select into a binary operator.
void handlePotentiallyDeadBlocks(SmallVectorImpl< BasicBlock * > &Worklist)
bool prepareWorklist(Function &F)
Perform early cleanup and prepare the InstCombine worklist.
Instruction * FoldOpIntoSelect(Instruction &Op, SelectInst *SI, bool FoldWithMultiUse=false, bool SimplifyBothArms=false)
Given an instruction with a select as one operand and a constant as the other operand,...
Instruction * visitFree(CallInst &FI, Value *FreedOp)
Instruction * visitExtractValueInst(ExtractValueInst &EV)
void handlePotentiallyDeadSuccessors(BasicBlock *BB, BasicBlock *LiveSucc)
Instruction * foldBinopWithRecurrence(BinaryOperator &BO)
Try to fold binary operators whose operands are simple interleaved recurrences to a single recurrence...
Instruction * eraseInstFromFunction(Instruction &I) override
Combiner aware instruction erasure.
Instruction * visitLandingPadInst(LandingPadInst &LI)
const InstCombineCLOptions & CLOpts
Instruction * visitReturnInst(ReturnInst &RI)
Instruction * visitSwitchInst(SwitchInst &SI)
Instruction * foldBinopWithPhiOperands(BinaryOperator &BO)
For a binary operator with 2 phi operands, try to hoist the binary operation before the phi.
bool SimplifyDemandedFPClass(Instruction *I, unsigned Op, FPClassTest DemandedMask, KnownFPClass &Known, const SimplifyQuery &Q, unsigned Depth=0)
bool mergeStoreIntoSuccessor(StoreInst &SI)
Try to transform: if () { *P = v1; } else { *P = v2 } or: *P = v1; if () { *P = v2; }...
Instruction * tryFoldInstWithCtpopWithNot(Instruction *I)
Instruction * visitUncondBrInst(UncondBrInst &BI)
void CreateNonTerminatorUnreachable(Instruction *InsertAt)
Create and insert the idiom we use to indicate a block is unreachable without having to rewrite the C...
Value * pushFreezeToPreventPoisonFromPropagating(FreezeInst &FI)
bool run()
Run the combiner over the entire worklist until it is empty.
Instruction * foldVectorBinop(BinaryOperator &Inst)
Canonicalize the position of binops relative to shufflevector.
bool removeInstructionsBeforeUnreachable(Instruction &I)
Value * SimplifySelectsFeedingBinaryOp(BinaryOperator &I, Value *LHS, Value *RHS)
void tryToSinkInstructionDbgVariableRecords(Instruction *I, BasicBlock::iterator InsertPos, BasicBlock *SrcBlock, BasicBlock *DestBlock, SmallVectorImpl< DbgVariableRecord * > &DPUsers)
void addDeadEdge(BasicBlock *From, BasicBlock *To, SmallVectorImpl< BasicBlock * > &Worklist)
Constant * unshuffleConstant(ArrayRef< int > ShMask, Constant *C, VectorType *NewCTy)
Find a constant NewC that has property: shuffle(NewC, poison, ShMask) = C for lanes that select NewC.
Instruction * visitAllocSite(Instruction &FI)
Instruction * visitGetElementPtrInst(GetElementPtrInst &GEP)
Value * tryFactorizationFolds(BinaryOperator &I)
This tries to simplify binary operations by factorizing out common terms (e.
Instruction * foldFreezeIntoRecurrence(FreezeInst &I, PHINode *PN)
bool tryToSinkInstruction(Instruction *I, BasicBlock *DestBlock)
Try to move the specified instruction from its current block into the beginning of DestBlock,...
bool freezeOtherUses(FreezeInst &FI)
void freelyInvertAllUsersOf(Value *V, Value *IgnoredUser=nullptr)
Freely adapt every user of V as-if V was changed to !V.
The core instruction combiner logic.
SimplifyQuery SQ
const DataLayout & getDataLayout() const
bool isFreeToInvert(Value *V, bool WillInvertAllUses, bool &DoesConsume)
Return true if the specified value is free to invert (apply ~ to).
static unsigned getComplexity(Value *V)
Assign a complexity or rank value to LLVM Values.
bool isKnownToBeAPowerOfTwo(const Value *V, bool OrZero=false, const Instruction *CtxI=nullptr, unsigned Depth=0)
TargetLibraryInfo & TLI
Instruction * InsertNewInstBefore(Instruction *New, BasicBlock::iterator Old)
Inserts an instruction New before instruction Old.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
static bool shouldAvoidAbsorbingNotIntoSelect(const SelectInst &SI)
void replaceUse(Use &U, Value *NewValue)
Replace use and add the previously used value to the worklist.
static bool isCanonicalPredicate(CmpPredicate Pred)
Predicate canonicalization reduces the number of patterns that need to be matched by other transforms...
Instruction * AnnotationMetadataSource
Source for annotation metadata, used by the IRBuilder inserter.
InstructionWorklist & Worklist
A worklist of the instructions that need to be simplified.
Instruction * InsertNewInstWith(Instruction *New, BasicBlock::iterator Old)
Same as InsertNewInstBefore, but also sets the debug loc.
BranchProbabilityInfo * BPI
ReversePostOrderTraversal< BasicBlock * > & RPOT
const DataLayout & DL
DomConditionCache DC
unsigned ComputeNumSignBits(const Value *Op, const Instruction *CtxI=nullptr, unsigned Depth=0) const
const bool MinimizeSize
IRBuilder< TargetFolder, IRBuilderInstCombineInserter > BuilderTy
An IRBuilder that automatically inserts new instructions into the worklist.
LLVM_ABI std::optional< Instruction * > targetInstCombineIntrinsic(IntrinsicInst &II)
AssumptionCache & AC
void addToWorklist(Instruction *I)
LLVM_ABI Value * getFreelyInvertedImpl(Value *V, bool WillInvertAllUses, BuilderTy *Builder, bool &DoesConsume, unsigned Depth)
Return nonnull value if V is free to invert under the condition of WillInvertAllUses.
SmallDenseSet< std::pair< const BasicBlock *, const BasicBlock * >, 8 > BackEdges
Backedges, used to avoid pushing instructions across backedges in cases where this may result in infi...
LLVM_ABI std::optional< Value * > targetSimplifyDemandedVectorEltsIntrinsic(IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp)
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
DominatorTree & DT
static Constant * getSafeVectorConstantForBinop(BinaryOperator::BinaryOps Opcode, Constant *In, bool IsRHSConstant)
Some binary operators require special handling to avoid poison and undefined behavior.
SmallDenseSet< std::pair< BasicBlock *, BasicBlock * >, 8 > DeadEdges
Edges that are known to never be taken.
LLVM_ABI std::optional< Value * > targetSimplifyDemandedUseBitsIntrinsic(IntrinsicInst &II, APInt DemandedMask, KnownBits &Known, bool &KnownBitsComputed)
LLVM_ABI bool isValidAddrSpaceCast(unsigned FromAS, unsigned ToAS) const
void computeKnownBits(const Value *V, KnownBits &Known, const Instruction *CtxI, unsigned Depth=0) const
Value * getFreelyInverted(Value *V, bool WillInvertAllUses, BuilderTy *Builder, bool &DoesConsume)
bool isBackEdge(const BasicBlock *From, const BasicBlock *To)
void visit(Iterator Start, Iterator End)
Definition InstVisitor.h:87
The legacy pass manager's instcombine pass.
Definition InstCombine.h:68
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - This function should be overriden by passes that need analysis information to do t...
bool runOnFunction(Function &F) override
runOnFunction - Virtual method overriden by subclasses to do the per-function processing of the pass.
InstructionWorklist - This is the worklist management logic for InstCombine and other simplification ...
LLVM_ABI void dropUBImplyingAttrsAndMetadata(ArrayRef< unsigned > Keep={})
Drop any attributes or metadata that can cause immediate undefined behavior.
static bool isBitwiseLogicOp(unsigned Opcode)
Determine if the Opcode is and/or/xor.
LLVM_ABI void copyIRFlags(const Value *V, bool IncludeWrapFlags=true)
Convenience method to copy supported exact, fast-math, and (optionally) wrapping flags from V to this...
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI bool isAssociative() const LLVM_READONLY
Return true if the instruction is associative:
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI void setFastMathFlags(FastMathFlags FMF)
Convenience function for setting multiple fast-math flags on this instruction, which must be an opera...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
bool isTerminator() const
iterator_range< user_iterator > users()
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
LLVM_ABI bool willReturn() const LLVM_READONLY
Return true if the instruction will return (unwinding is considered as a form of returning control fl...
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
bool isBitwiseLogicOp() const
Return true if this is and/or/xor.
bool isShift() const
LLVM_ABI void dropPoisonGeneratingFlags()
Drops flags that may cause this instruction to evaluate to poison despite having non-poison inputs.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
bool isIntDivRem() const
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
A wrapper class for inspecting calls to intrinsic functions.
Invoke instruction.
static InvokeInst * Create(FunctionType *Ty, Value *Func, BasicBlock *IfNormal, BasicBlock *IfException, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
The landingpad instruction holds all of the information necessary to generate correct exception handl...
bool isCleanup() const
Return 'true' if this landingpad instruction is a cleanup.
unsigned getNumClauses() const
Get the number of clauses for this landing pad.
static LLVM_ABI LandingPadInst * Create(Type *RetTy, unsigned NumReservedClauses, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedClauses is a hint for the number of incoming clauses that this landingpad w...
LLVM_ABI void addClause(Constant *ClauseVal)
Add a catch or filter clause to the landing pad.
bool isCatch(unsigned Idx) const
Return 'true' if the clause and index Idx is a catch clause.
bool isFilter(unsigned Idx) const
Return 'true' if the clause and index Idx is a filter clause.
Constant * getClause(unsigned Idx) const
Get the value of the clause at index Idx.
void setCleanup(bool V)
Indicate that this landingpad instruction is a cleanup.
A function/module analysis which provides an empty LastRunTrackingInfo.
This is an alternative analysis pass to BlockFrequencyInfoWrapperPass.
static void getLazyBFIAnalysisUsage(AnalysisUsage &AU)
Helper for client passes to set up the analysis usage on behalf of this pass.
An instruction for reading from memory.
Value * getPointerOperand()
bool isVolatile() const
Return true if this is a load from a volatile memory location.
Metadata node.
Definition Metadata.h:1081
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1437
ArrayRef< MDOperand > operands() const
Definition Metadata.h:1435
unsigned getNumOperands() const
Return number of MDNode operands.
Definition Metadata.h:1443
Tracking metadata reference owned by Metadata.
Definition Metadata.h:902
This is the common base class for memset/memcpy/memmove.
static LLVM_ABI MemoryLocation getForDest(const MemIntrinsic *MI)
Return a location representing the destination of a memory set or transfer.
Root of the metadata hierarchy.
Definition Metadata.h:64
Value * getLHS() const
Value * getRHS() const
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
MDNode * getScopeList() const
OptimizationRemarkEmitter legacy analysis pass.
The optimization diagnostic interface.
Utility class for integer operators which may exhibit overflow - Add, Sub, Mul, and Shl.
Definition Operator.h:78
bool hasNoSignedWrap() const
Test whether this operation is known to never undergo signed overflow, aka the nsw property.
Definition Operator.h:113
bool hasNoUnsignedWrap() const
Test whether this operation is known to never undergo unsigned overflow, aka the nuw property.
Definition Operator.h:107
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
op_range incoming_values()
BasicBlock * getIncomingBlock(unsigned i) const
Return incoming basic block number i.
Value * getIncomingValue(unsigned i) const
Return incoming value number x.
unsigned getNumIncomingValues() const
Return the number of incoming edges.
static PHINode * Create(Type *Ty, unsigned NumReservedValues, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedValues is a hint for the number of incoming edges that this phi node will h...
PassRegistry - This class manages the registration and intitialization of the pass subsystem as appli...
AnalysisType & getAnalysis() const
getAnalysis<AnalysisType>() - This function is used by subclasses to get to the analysis information ...
AnalysisType * getAnalysisIfAvailable() const
getAnalysisIfAvailable<AnalysisType>() - Subclasses use this function to get analysis information tha...
In order to facilitate speculative execution, many instructions do not invoke immediate undefined beh...
Definition Constants.h:1705
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Definition Analysis.h:151
PreservedAnalyses & preserve()
Mark an analysis as preserved.
Definition Analysis.h:132
An analysis pass based on the new PM to deliver ProfileSummaryInfo.
An analysis pass based on legacy pass manager to deliver ProfileSummaryInfo.
Analysis providing profile information.
bool hasProfileSummary() const
Returns true if profile summary is available.
A global registry used in conjunction with static constructors to make pluggable components (like tar...
Definition Registry.h:116
Return a value (possibly void), from a function.
Value * getReturnValue() const
Convenience accessor. Returns null if there is no return value.
This class represents the LLVM 'select' instruction.
const Value * getFalseValue() const
const Value * getCondition() const
static SelectInst * Create(Value *C, Value *S1, Value *S2, const Twine &NameStr="", InsertPosition InsertBefore=nullptr, const Instruction *MDFrom=nullptr)
const Value * getTrueValue() const
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
This instruction constructs a fixed permutation of two input vectors.
size_type size() const
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
Definition SmallSet.h:134
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
iterator erase(const_iterator CI)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
typename SuperClass::iterator iterator
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Multiway switch.
Analysis pass providing the TargetTransformInfo.
Analysis pass providing the TargetLibraryInfo.
Provides information about what library functions are available for the current target.
bool has(LibFunc F) const
Tests whether a library function is available.
LibFunc getLibFunc(StringRef funcName) const
Searches for a particular function name.
Wrapper pass for TargetTransformInfo.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
bool isSized() const
Return true if it makes sense to take the size of this type.
Definition Type.h:321
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:271
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
LLVM_ABI bool isScalableTy() const
Return true if this is a type whose size is a known multiple of vscale.
Definition Type.cpp:61
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:96
Unconditional Branch instruction.
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
This function has undefined behavior.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
Use * op_iterator
Definition User.h:254
op_range operands()
Definition User.h:267
op_iterator op_begin()
Definition User.h:259
LLVM_ABI bool isDroppable() const
A droppable user is a user for which uses can be dropped without affecting correctness and should be ...
Definition User.cpp:119
LLVM_ABI bool replaceUsesOfWith(Value *From, Value *To)
Replace uses of one Value with another.
Definition User.cpp:25
Value * getOperand(unsigned i) const
Definition User.h:207
unsigned getNumOperands() const
Definition User.h:229
op_iterator op_end()
Definition User.h:261
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
const Value * stripAndAccumulateInBoundsConstantOffsets(const DataLayout &DL, APInt &Offset) const
This is a wrapper around stripAndAccumulateConstantOffsets with the in-bounds requirement set to fals...
Definition Value.h:729
LLVM_ABI bool hasOneUser() const
Return true if there is exactly one user of this value.
Definition Value.cpp:163
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
bool hasUseList() const
Check if this Value has a use-list.
Definition Value.h:346
LLVM_ABI bool hasNUses(unsigned N) const
Return true if this Value has exactly N uses.
Definition Value.cpp:147
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:712
bool use_empty() const
Definition Value.h:348
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
LLVM_ABI uint64_t getPointerDereferenceableBytes(const DataLayout &DL, bool &CanBeNull, bool *CanBeFreed) const
Returns the number of bytes known to be dereferenceable for the pointer value.
Definition Value.cpp:918
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Value handle that is nullable, but tries to track the Value.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
An efficient, type-erasing, non-owning reference to a callable.
TypeSize getSequentialElementStride(const DataLayout &DL) const
const ParentTy * getParent() const
Definition ilist_node.h:34
reverse_self_iterator getReverseIterator()
Definition ilist_node.h:126
self_iterator getIterator()
Definition ilist_node.h:123
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an std::string.
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
Definition Attributor.h:165
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
PtrAdd_match< PointerOpTy, OffsetOpTy > m_PtrAdd(const PointerOpTy &PointerOp, const OffsetOpTy &OffsetOp)
Matches GEP with i8 source element type.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, FCmpInst > m_FCmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::AShr > m_AShr(const LHS &L, const RHS &R)
auto m_PtrToIntOrAddr(const OpTy &Op)
Matches PtrToInt or PtrToAddr.
OneOps_match< OpTy, Instruction::Freeze > m_Freeze(const OpTy &Op)
Matches FreezeInst.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
BinaryOp_match< LHS, RHS, Instruction::Xor > m_Xor(const LHS &L, const RHS &R)
br_match m_UnconditionalBr(BasicBlock *&Succ)
ap_match< APInt > m_APIntAllowPoison(const APInt *&Res)
Match APInt while allowing poison in splat vector constants.
auto m_ConstantExpr()
Match a constant expression or a constant that contains a constant expression.
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
BinOpPred_match< LHS, RHS, is_idiv_op > m_IDiv(const LHS &L, const RHS &R)
Matches integer division operations.
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
DisjointOr_match< LHS, RHS > m_DisjointOr(const LHS &L, const RHS &R)
BinOpPred_match< LHS, RHS, is_right_shift_op > m_Shr(const LHS &L, const RHS &R)
Matches logical shift operations.
ap_match< APFloat > m_APFloat(const APFloat *&Res)
Match a ConstantFP or splatted ConstantVector, binding the specified pointer to the contained APFloat...
cst_pred_ty< is_nonnegative > m_NonNegative()
Match an integer or vector of non-negative values.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
ExtractValue_match< Ind, Val_t > m_ExtractValue(const Val_t &V)
Match a single index ExtractValue instruction.
match_combine_or< CastInst_match< OpTy, UIToFPInst >, CastInst_match< OpTy, SIToFPInst > > m_IToFP(const OpTy &Op)
auto m_Value()
Match an arbitrary value and ignore it.
auto m_Ctpop(const Opnd0 &Op0)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_Constant()
Match an arbitrary Constant and ignore it.
ContainsMatchingVectorElement_match< SPTy > m_ContainsMatchingVectorElement(const SPTy &SubPattern)
Match a vector constant where at least one of its elements matches the subpattern.
NNegZExt_match< OpTy > m_NNegZExt(const OpTy &Op)
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
ThreeOps_match< decltype(m_Value()), LHS, RHS, Instruction::Select, true > m_c_Select(const LHS &L, const RHS &R)
Match Select(C, LHS, RHS) or Select(C, RHS, LHS)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
Splat_match< T > m_Splat(const T &SubPattern)
Match a vector splat.
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
match_immconstant_ty m_ImmConstant()
Match an arbitrary immediate Constant and ignore it.
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
match_combine_or< BinaryOp_match< LHS, RHS, Instruction::Add >, DisjointOr_match< LHS, RHS > > m_AddLike(const LHS &L, const RHS &R)
Match either "add" or "or disjoint".
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
match_combine_or< CastInst_match< OpTy, SExtInst >, NNegZExt_match< OpTy > > m_SExtLike(const OpTy &Op)
Match either "sext" or "zext nneg".
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
BinaryOp_match< LHS, RHS, Instruction::SDiv > m_SDiv(const LHS &L, const RHS &R)
auto m_VectorInsert(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoSignedWrap >, DisjointOr_match< LHS, RHS > > m_NSWAddLike(const LHS &L, const RHS &R)
Match either "add nsw" or "or disjoint".
AnyBinaryOp_match< LHS, RHS, true > m_c_BinOp(const LHS &L, const RHS &R)
Matches a BinaryOperator with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
cstfp_pred_ty< is_non_zero_fp > m_NonZeroFP()
Match a floating-point non-zero.
auto m_MaxOrMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
brc_match< Cond_t, match_bind< BasicBlock >, match_bind< BasicBlock > > m_Br(const Cond_t &C, BasicBlock *&T, BasicBlock *&F)
BinaryOp_match< LHS, RHS, Instruction::SRem > m_SRem(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
auto m_VecReverse(const Opnd0 &Op0)
BinaryOp_match< LHS, RHS, Instruction::Or > m_Or(const LHS &L, const RHS &R)
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoUnsignedWrap >, DisjointOr_match< LHS, RHS > > m_NUWAddLike(const LHS &L, const RHS &R)
Match either "add nuw" or "or disjoint".
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
LLVM_ABI Intrinsic::ID getInverseMinMaxIntrinsic(Intrinsic::ID MinMaxID)
@ Offset
Definition DWP.cpp:577
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
Definition STLExtras.h:846
void stable_sort(R &&Range)
Definition STLExtras.h:2132
LLVM_ABI void initializeInstructionCombiningPassPass(PassRegistry &)
LLVM_ABI unsigned removeAllNonTerminatorAndEHPadInstructions(BasicBlock *BB)
Remove all instructions from a basic block other than its terminator and any present EH pad instructi...
Definition Local.cpp:2515
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
LLVM_ABI Value * simplifyGEPInst(Type *SrcTy, Value *Ptr, ArrayRef< Value * > Indices, GEPNoWrapFlags NW, const SimplifyQuery &Q)
Given operands for a GetElementPtrInst, fold the result or return null.
LLVM_ABI Constant * getInitialValueOfAllocation(const Value *V, const TargetLibraryInfo *TLI, Type *Ty)
If this is a call to an allocation function that initializes memory to a fixed value,...
bool succ_empty(const Instruction *I)
Definition CFG.h:141
LLVM_ABI Value * simplifyFreezeInst(Value *Op, const SimplifyQuery &Q)
Given an operand for a Freeze, see if we can fold the result.
LLVM_ABI FunctionPass * createInstructionCombiningPass()
LLVM_ABI void findDbgValues(Value *V, SmallVectorImpl< DbgVariableRecord * > &DbgVariableRecords)
Finds the dbg.values describing a value.
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
void set_intersect(S1Ty &S1, const S2Ty &S2)
set_intersect(A, B) - Compute A := A ^ B Identical to set_intersection, except that it works on set<>...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI void salvageDebugInfo(const MachineRegisterInfo &MRI, MachineInstr &MI)
Assuming the instruction MI is going to be deleted, attempt to salvage debug users of MI by writing t...
Definition Utils.cpp:1676
auto successors(const MachineBasicBlock *BB)
LLVM_ABI Constant * ConstantFoldInstruction(const Instruction *I, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr)
ConstantFoldInstruction - Try to constant fold the specified instruction.
LLVM_ABI bool isRemovableAlloc(const CallBase *V, const TargetLibraryInfo *TLI)
Return true if this is a call to an allocation function that does not have side effects that we are r...
LLVM_ABI std::optional< StringRef > getAllocationFamily(const Value *I, const TargetLibraryInfo *TLI)
If a function is part of an allocation family (e.g.
OuterAnalysisManagerProxy< ModuleAnalysisManager, Function > ModuleAnalysisManagerFunctionProxy
Provide the ModuleAnalysisManager to Function proxy.
LLVM_ABI Value * lowerObjectSizeCall(IntrinsicInst *ObjectSize, const DataLayout &DL, const TargetLibraryInfo *TLI, bool MustSucceed)
Try to turn a call to @llvm.objectsize into an integer value of the given Type.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ABI Value * simplifyInstructionWithOperands(Instruction *I, ArrayRef< Value * > NewOps, const SimplifyQuery &Q)
Like simplifyInstruction but the operands of I are replaced with NewOps.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
gep_type_iterator gep_type_end(const User *GEP)
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
LLVM_ABI Value * getReallocatedOperand(const CallBase *CB)
If this is a call to a realloc function, return the reallocated operand.
APFloat frexp(const APFloat &X, int &Exp, APFloat::roundingMode RM)
Equivalent of C standard library function.
Definition APFloat.h:1713
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI bool isAllocLikeFn(const Value *V, const TargetLibraryInfo *TLI)
Tests if a value is a call or invoke to a library function that allocates memory (either malloc,...
LLVM_ABI bool handleUnreachableTerminator(Instruction *I, SmallVectorImpl< Value * > &PoisonedValues)
If a terminator in an unreachable basic block has an operand of type Instruction, transform it into p...
Definition Local.cpp:2498
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
LLVM_ABI void setBranchWeights(Instruction &I, ArrayRef< uint32_t > Weights, bool IsExpected, bool ElideAllZero=false)
Create a new branch_weights metadata node and add or overwrite a prof metadata reference to instructi...
LLVM_ABI bool matchSimpleRecurrence(const PHINode *P, BinaryOperator *&BO, Value *&Start, Value *&Step)
Attempt to match a simple first order recurrence cycle of the form: iv = phi Ty [Start,...
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Function *CtxF=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
LLVM_ABI Value * simplifyAddInst(Value *LHS, Value *RHS, bool IsNSW, bool IsNUW, const SimplifyQuery &Q)
Given operands for an Add, fold the result or return null.
LLVM_ABI Constant * ConstantFoldConstant(const Constant *C, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr)
ConstantFoldConstant - Fold the constant using the specified DataLayout.
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
constexpr bool has_single_bit(T Value) noexcept
Definition bit.h:149
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
LLVM_ABI bool isInstructionTriviallyDead(Instruction *I, const TargetLibraryInfo *TLI=nullptr)
Return true if the result produced by the instruction is not used, and the instruction will return.
Definition Local.cpp:402
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
LLVM_ABI Value * emitGEPOffset(IRBuilderBase *Builder, const DataLayout &DL, User *GEP, bool NoAssumptions=false)
Given a getelementptr instruction/constantexpr, emit the code necessary to compute the offset from th...
Definition Local.cpp:22
constexpr unsigned MaxAnalysisRecursionDepth
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
bool isModSet(const ModRefInfo MRI)
Definition ModRef.h:49
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
LLVM_ABI bool LowerDbgDeclare(Function &F)
Lowers dbg.declare records into appropriate set of dbg.value records.
Definition Local.cpp:1813
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI void salvageDebugInfoForDbgValues(Instruction &I, ArrayRef< DbgVariableRecord * > DbgRecords)
Salvage only the records in DbgRecords instead of finding every debug user of I.
Definition Local.cpp:2121
generic_gep_type_iterator<> gep_type_iterator
LLVM_ABI void ConvertDebugDeclareToDebugValue(DbgVariableRecord *DVR, StoreInst *SI, DIBuilder &Builder)
Inserts a dbg.value record before a store to an alloca'd value that has an associated dbg....
Definition Local.cpp:1654
LLVM_ABI Constant * ConstantFoldCastOperand(unsigned Opcode, Constant *C, Type *DestTy, const DataLayout &DL)
Attempt to constant fold a cast with the specified operand.
LLVM_ABI bool canCreateUndefOrPoison(const Operator *Op, bool ConsiderFlagsAndMetadata=true)
canCreateUndefOrPoison returns true if Op can create undef or poison from non-undef & non-poison oper...
LLVM_ABI EHPersonality classifyEHPersonality(const Value *Pers)
See if the given exception handling personality function is one that we understand.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI Value * simplifyExtractValueInst(Value *Agg, ArrayRef< unsigned > Idxs, const SimplifyQuery &Q)
Given operands for an ExtractValueInst, fold the result or return null.
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
LLVM_ABI bool replaceAllDbgUsesWith(Instruction &From, Value &To, Instruction &DomPoint, DominatorTree &DT)
Point debug users of From to To or salvage them.
Definition Local.cpp:2444
LLVM_ABI bool isKnownNonZero(const Value *V, const SimplifyQuery &Q, unsigned Depth=0)
Return true if the given value is known to be non-zero when defined.
constexpr int PoisonMaskElem
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
ModRefInfo
Flags indicating whether a memory access modifies or references memory.
Definition ModRef.h:28
@ Ref
The access may reference the value stored in memory.
Definition ModRef.h:32
@ ModRef
The access may reference and may modify the value stored in memory.
Definition ModRef.h:36
@ Mod
The access may modify the value stored in memory.
Definition ModRef.h:34
@ NoModRef
The access neither references nor modifies the value stored in memory.
Definition ModRef.h:30
TargetTransformInfo TTI
LLVM_ABI Value * simplifyBinOp(unsigned Opcode, Value *LHS, Value *RHS, const SimplifyQuery &Q)
Given operands for a BinaryOperator, fold the result or return null.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
DWARFExpression::Operation Op
bool isSafeToSpeculativelyExecuteWithVariableReplaced(const Instruction *I, bool IgnoreUBImplyingAttrs=true)
Don't use information from its non-constant operands.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI Value * getFreedOperand(const CallBase *CB, const TargetLibraryInfo *TLI)
If this if a call to a free function, return the freed operand.
constexpr unsigned BitWidth
LLVM_ABI bool isGuaranteedToTransferExecutionToSuccessor(const Instruction *I)
Return true if this function can prove that the instruction I will always transfer execution to one o...
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
LLVM_ABI bool extractBranchWeights(const MDNode *ProfileData, SmallVectorImpl< uint32_t > &Weights)
Extract branch weights from MD_prof metadata.
auto count_if(R &&Range, UnaryPredicate P)
Wrapper function around std::count_if to count the number of times an element satisfying a given pred...
Definition STLExtras.h:2035
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
gep_type_iterator gep_type_begin(const User *GEP)
auto predecessors(const MachineBasicBlock *BB)
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2162
AAResults AliasAnalysis
Temporary typedef for legacy code that uses a generic AliasAnalysis pointer or reference.
static auto filterDbgVars(iterator_range< simple_ilist< DbgRecord >::iterator > R)
Filter the DbgRecord range to DbgVariableRecord types only and downcast.
LLVM_ABI void initializeInstCombine(PassRegistry &)
Initialize all passes linked into the InstCombine library.
LLVM_ABI void findDbgUsers(Value *V, SmallVectorImpl< DbgVariableRecord * > &DbgVariableRecords)
Finds the debug info records describing a value.
LLVM_ABI Constant * ConstantFoldBinaryInstruction(unsigned Opcode, Constant *V1, Constant *V2)
bool isRefSet(const ModRefInfo MRI)
Definition ModRef.h:52
LLVM_ABI std::optional< bool > isImpliedCondition(const Value *LHS, const Value *RHS, const DataLayout &DL, bool LHSIsTrue=true, unsigned Depth=0)
Return true if RHS is known to be implied true by LHS.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
SimplifyQuery getWithInstruction(const Instruction *I) const