LLVM 24.0.0git
SimplifyCFG.cpp
Go to the documentation of this file.
1//===- SimplifyCFG.cpp - Code to perform CFG simplification ---------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// Peephole optimize the CFG.
10//
11//===----------------------------------------------------------------------===//
12
13#include "llvm/ADT/APInt.h"
14#include "llvm/ADT/ArrayRef.h"
15#include "llvm/ADT/DenseMap.h"
16#include "llvm/ADT/MapVector.h"
17#include "llvm/ADT/STLExtras.h"
18#include "llvm/ADT/Sequence.h"
20#include "llvm/ADT/SetVector.h"
23#include "llvm/ADT/Statistic.h"
24#include "llvm/ADT/StringRef.h"
31#include "llvm/Analysis/Loads.h"
36#include "llvm/IR/Attributes.h"
37#include "llvm/IR/BasicBlock.h"
38#include "llvm/IR/CFG.h"
39#include "llvm/IR/Constant.h"
41#include "llvm/IR/Constants.h"
42#include "llvm/IR/DataLayout.h"
43#include "llvm/IR/DebugInfo.h"
45#include "llvm/IR/Function.h"
46#include "llvm/IR/GlobalValue.h"
48#include "llvm/IR/IRBuilder.h"
49#include "llvm/IR/InstrTypes.h"
50#include "llvm/IR/Instruction.h"
53#include "llvm/IR/LLVMContext.h"
54#include "llvm/IR/MDBuilder.h"
56#include "llvm/IR/Metadata.h"
57#include "llvm/IR/Module.h"
58#include "llvm/IR/NoFolder.h"
59#include "llvm/IR/Operator.h"
62#include "llvm/IR/Type.h"
63#include "llvm/IR/Use.h"
64#include "llvm/IR/User.h"
65#include "llvm/IR/Value.h"
66#include "llvm/IR/ValueHandle.h"
70#include "llvm/Support/Debug.h"
80#include <algorithm>
81#include <cassert>
82#include <climits>
83#include <cstddef>
84#include <cstdint>
85#include <iterator>
86#include <map>
87#include <optional>
88#include <set>
89#include <tuple>
90#include <utility>
91#include <vector>
92
93using namespace llvm;
94using namespace PatternMatch;
95
96#define DEBUG_TYPE "simplifycfg"
97
98namespace llvm {
99
101 "simplifycfg-require-and-preserve-domtree", cl::Hidden,
102
103 cl::desc(
104 "Temporary development switch used to gradually uplift SimplifyCFG "
105 "into preserving DomTree,"));
106
107// Chosen as 2 so as to be cheap, but still to have enough power to fold
108// a select, so the "clamp" idiom (of a min followed by a max) will be caught.
109// To catch this, we need to fold a compare and a select, hence '2' being the
110// minimum reasonable default.
112 "phi-node-folding-threshold", cl::Hidden, cl::init(2),
113 cl::desc(
114 "Control the amount of phi node folding to perform (default = 2)"));
115
117 "two-entry-phi-node-folding-threshold", cl::Hidden, cl::init(4),
118 cl::desc("Control the maximal total instruction cost that we are willing "
119 "to speculatively execute to fold a 2-entry PHI node into a "
120 "select (default = 4)"));
121
122static cl::opt<bool>
123 HoistCommon("simplifycfg-hoist-common", cl::Hidden, cl::init(true),
124 cl::desc("Hoist common instructions up to the parent block"));
125
127 "simplifycfg-hoist-loads-with-cond-faulting", cl::Hidden, cl::init(true),
128 cl::desc("Hoist loads if the target supports conditional faulting"));
129
131 "simplifycfg-hoist-stores-with-cond-faulting", cl::Hidden, cl::init(true),
132 cl::desc("Hoist stores if the target supports conditional faulting"));
133
135 "hoist-loads-stores-with-cond-faulting-threshold", cl::Hidden, cl::init(6),
136 cl::desc("Control the maximal conditional load/store that we are willing "
137 "to speculatively execute to eliminate conditional branch "
138 "(default = 6)"));
139
141 HoistCommonSkipLimit("simplifycfg-hoist-common-skip-limit", cl::Hidden,
142 cl::init(20),
143 cl::desc("Allow reordering across at most this many "
144 "instructions when hoisting"));
145
146static cl::opt<bool>
147 SinkCommon("simplifycfg-sink-common", cl::Hidden, cl::init(true),
148 cl::desc("Sink common instructions down to the end block"));
149
151 "simplifycfg-hoist-cond-stores", cl::Hidden, cl::init(true),
152 cl::desc("Hoist conditional stores if an unconditional store precedes"));
153
155 "simplifycfg-merge-cond-stores", cl::Hidden, cl::init(true),
156 cl::desc("Hoist conditional stores even if an unconditional store does not "
157 "precede - hoist multiple conditional stores into a single "
158 "predicated store"));
159
161 "simplifycfg-merge-cond-stores-aggressively", cl::Hidden, cl::init(false),
162 cl::desc("When merging conditional stores, do so even if the resultant "
163 "basic blocks are unlikely to be if-converted as a result"));
164
166 "speculate-one-expensive-inst", cl::Hidden, cl::init(true),
167 cl::desc("Allow exactly one expensive instruction to be speculatively "
168 "executed"));
169
171 "max-speculation-depth", cl::Hidden, cl::init(10),
172 cl::desc("Limit maximum recursion depth when calculating costs of "
173 "speculatively executed instructions"));
174
175static cl::opt<int>
176 MaxSmallBlockSize("simplifycfg-max-small-block-size", cl::Hidden,
177 cl::init(10),
178 cl::desc("Max size of a block which is still considered "
179 "small enough to thread through"));
180
181// Two is chosen to allow one negation and a logical combine.
183 BranchFoldThreshold("simplifycfg-branch-fold-threshold", cl::Hidden,
184 cl::init(2),
185 cl::desc("Maximum cost of combining conditions when "
186 "folding branches"));
187
189 "simplifycfg-branch-fold-common-dest-vector-multiplier", cl::Hidden,
190 cl::init(2),
191 cl::desc("Multiplier to apply to threshold when determining whether or not "
192 "to fold branch to common destination when vector operations are "
193 "present"));
194
196 "simplifycfg-merge-compatible-invokes", cl::Hidden, cl::init(true),
197 cl::desc("Allow SimplifyCFG to merge invokes together when appropriate"));
198
200 "max-switch-cases-per-result", cl::Hidden, cl::init(16),
201 cl::desc("Limit cases to analyze when converting a switch to select"));
202
204 "max-jump-threading-live-blocks", cl::Hidden, cl::init(24),
205 cl::desc("Limit number of blocks a define in a threaded block is allowed "
206 "to be live in"));
207
209
210} // end namespace llvm
211
212STATISTIC(NumBitMaps, "Number of switch instructions turned into bitmaps");
213STATISTIC(NumLinearMaps,
214 "Number of switch instructions turned into linear mapping");
215STATISTIC(NumLookupTables,
216 "Number of switch instructions turned into lookup tables");
218 NumLookupTablesHoles,
219 "Number of switch instructions turned into lookup tables (holes checked)");
220STATISTIC(NumTableCmpReuses, "Number of reused switch table lookup compares");
221STATISTIC(NumFoldValueComparisonIntoPredecessors,
222 "Number of value comparisons folded into predecessor basic blocks");
223STATISTIC(NumFoldBranchToCommonDest,
224 "Number of branches folded into predecessor basic block");
226 NumHoistCommonCode,
227 "Number of common instruction 'blocks' hoisted up to the begin block");
228STATISTIC(NumHoistCommonInstrs,
229 "Number of common instructions hoisted up to the begin block");
230STATISTIC(NumSinkCommonCode,
231 "Number of common instruction 'blocks' sunk down to the end block");
232STATISTIC(NumSinkCommonInstrs,
233 "Number of common instructions sunk down to the end block");
234STATISTIC(NumSpeculations, "Number of speculative executed instructions");
235STATISTIC(NumInvokes,
236 "Number of invokes with empty resume blocks simplified into calls");
237STATISTIC(NumInvokesMerged, "Number of invokes that were merged together");
238STATISTIC(NumInvokeSetsFormed, "Number of invoke sets that were formed");
239
240namespace {
241
242// The first field contains the value that the switch produces when a certain
243// case group is selected, and the second field is a vector containing the
244// cases composing the case group.
245using SwitchCaseResultVectorTy =
247
248// The first field contains the phi node that generates a result of the switch
249// and the second field contains the value generated for a certain case in the
250// switch for that PHI.
251using SwitchCaseResultsTy = SmallVector<std::pair<PHINode *, Constant *>, 4>;
252
253/// ValueEqualityComparisonCase - Represents a case of a switch.
254struct ValueEqualityComparisonCase {
256 BasicBlock *Dest;
257
258 ValueEqualityComparisonCase(ConstantInt *Value, BasicBlock *Dest)
259 : Value(Value), Dest(Dest) {}
260
261 bool operator<(ValueEqualityComparisonCase RHS) const {
262 // Comparing pointers is ok as we only rely on the order for uniquing.
263 return Value < RHS.Value;
264 }
265
266 bool operator==(BasicBlock *RHSDest) const { return Dest == RHSDest; }
267};
268
269class SimplifyCFGOpt {
270 const TargetTransformInfo &TTI;
271 DomTreeUpdater *DTU;
272 const DataLayout &DL;
273 ArrayRef<WeakVH> LoopHeaders;
274 const SimplifyCFGOptions &Options;
275 bool Resimplify;
276
277 Value *isValueEqualityComparison(Instruction *TI);
278 BasicBlock *getValueEqualityComparisonCases(
279 Instruction *TI, std::vector<ValueEqualityComparisonCase> &Cases);
280 bool simplifyEqualityComparisonWithOnlyPredecessor(Instruction *TI,
281 BasicBlock *Pred,
282 IRBuilder<> &Builder);
283 bool performValueComparisonIntoPredecessorFolding(Instruction *TI, Value *&CV,
284 Instruction *PTI,
285 IRBuilder<> &Builder);
286 bool foldValueComparisonIntoPredecessors(Instruction *TI,
287 IRBuilder<> &Builder);
288
289 bool simplifyResume(ResumeInst *RI, IRBuilder<> &Builder);
290 bool simplifySingleResume(ResumeInst *RI);
291 bool simplifyCommonResume(ResumeInst *RI);
292 bool simplifyCleanupReturn(CleanupReturnInst *RI);
293 bool simplifyUnreachable(UnreachableInst *UI);
294 bool simplifySwitch(SwitchInst *SI, IRBuilder<> &Builder);
295 bool simplifyDuplicateSwitchArms(SwitchInst *SI, DomTreeUpdater *DTU);
296 bool simplifyIndirectBr(IndirectBrInst *IBI);
297 bool simplifyUncondBranch(UncondBrInst *BI, IRBuilder<> &Builder);
298 bool simplifyCondBranch(CondBrInst *BI, IRBuilder<> &Builder);
299 bool foldCondBranchOnValueKnownInPredecessor(CondBrInst *BI);
300
301 bool tryToSimplifyUncondBranchWithICmpInIt(ICmpInst *ICI,
302 IRBuilder<> &Builder);
303 bool tryToSimplifyUncondBranchWithICmpSelectInIt(ICmpInst *ICI,
304 SelectInst *Select,
305 IRBuilder<> &Builder);
306 bool hoistCommonCodeFromSuccessors(Instruction *TI, bool AllInstsEqOnly);
307 bool hoistSuccIdenticalTerminatorToSwitchOrIf(
308 Instruction *TI, Instruction *I1,
309 SmallVectorImpl<Instruction *> &OtherSuccTIs,
310 ArrayRef<BasicBlock *> UniqueSuccessors);
311 bool speculativelyExecuteBB(CondBrInst *BI, BasicBlock *ThenBB);
312 bool simplifyTerminatorOnSelect(Instruction *OldTerm, Value *Cond,
313 BasicBlock *TrueBB, BasicBlock *FalseBB,
314 uint32_t TrueWeight, uint32_t FalseWeight);
315 bool simplifyBranchOnICmpChain(CondBrInst *BI, IRBuilder<> &Builder,
316 const DataLayout &DL);
317 bool simplifySwitchOnSelect(SwitchInst *SI, SelectInst *Select);
318 bool simplifySwitchOnSelectRemap(SwitchInst *SI, SelectInst *Select, Value *X,
319 ConstantInt *C, bool Negate);
320 bool simplifyIndirectBrOnSelect(IndirectBrInst *IBI, SelectInst *SI);
321 bool turnSwitchRangeIntoICmp(SwitchInst *SI, IRBuilder<> &Builder);
322 bool simplifyDuplicatePredecessors(BasicBlock *Succ, DomTreeUpdater *DTU);
323
324public:
325 SimplifyCFGOpt(const TargetTransformInfo &TTI, DomTreeUpdater *DTU,
326 const DataLayout &DL, ArrayRef<WeakVH> LoopHeaders,
327 const SimplifyCFGOptions &Opts)
328 : TTI(TTI), DTU(DTU), DL(DL), LoopHeaders(LoopHeaders), Options(Opts) {
329 assert((!DTU || !DTU->hasPostDomTree()) &&
330 "SimplifyCFG is not yet capable of maintaining validity of a "
331 "PostDomTree, so don't ask for it.");
332 }
333
334 bool simplifyOnce(BasicBlock *BB);
335 bool run(BasicBlock *BB);
336
337 // Helper to set Resimplify and return change indication.
338 bool requestResimplify() {
339 Resimplify = true;
340 return true;
341 }
342};
343
344// we synthesize a || b as select a, true, b
345// we synthesize a && b as select a, b, false
346// this function determines if SI is playing one of those roles.
347[[maybe_unused]] bool
348isSelectInRoleOfConjunctionOrDisjunction(const SelectInst *SI) {
349 return ((isa<ConstantInt>(SI->getTrueValue()) &&
350 (dyn_cast<ConstantInt>(SI->getTrueValue())->isOne())) ||
351 (isa<ConstantInt>(SI->getFalseValue()) &&
352 (dyn_cast<ConstantInt>(SI->getFalseValue())->isNullValue())));
353}
354
355} // end anonymous namespace
356
357/// Return true if all the PHI nodes in the basic block \p BB
358/// receive compatible (identical) incoming values when coming from
359/// all of the predecessor blocks that are specified in \p IncomingBlocks.
360///
361/// Note that if the values aren't exactly identical, but \p EquivalenceSet
362/// is provided, and *both* of the values are present in the set,
363/// then they are considered equal.
365 BasicBlock *BB, ArrayRef<BasicBlock *> IncomingBlocks,
366 SmallPtrSetImpl<Value *> *EquivalenceSet = nullptr) {
367 assert(IncomingBlocks.size() == 2 &&
368 "Only for a pair of incoming blocks at the time!");
369
370 // FIXME: it is okay if one of the incoming values is an `undef` value,
371 // iff the other incoming value is guaranteed to be a non-poison value.
372 // FIXME: it is okay if one of the incoming values is a `poison` value.
373 return all_of(BB->phis(), [IncomingBlocks, EquivalenceSet](PHINode &PN) {
374 Value *IV0 = PN.getIncomingValueForBlock(IncomingBlocks[0]);
375 Value *IV1 = PN.getIncomingValueForBlock(IncomingBlocks[1]);
376 if (IV0 == IV1)
377 return true;
378 if (EquivalenceSet && EquivalenceSet->contains(IV0) &&
379 EquivalenceSet->contains(IV1))
380 return true;
381 return false;
382 });
383}
384
385/// Return true if it is safe to merge these two
386/// terminator instructions together.
387static bool
389 SmallSetVector<BasicBlock *, 4> *FailBlocks = nullptr) {
390 if (SI1 == SI2)
391 return false; // Can't merge with self!
392
393 // It is not safe to merge these two switch instructions if they have a common
394 // successor, and if that successor has a PHI node, and if *that* PHI node has
395 // conflicting incoming values from the two switch blocks.
396 BasicBlock *SI1BB = SI1->getParent();
397 BasicBlock *SI2BB = SI2->getParent();
398
400 bool Fail = false;
401 for (BasicBlock *Succ : successors(SI2BB)) {
402 if (!SI1Succs.count(Succ))
403 continue;
404 if (incomingValuesAreCompatible(Succ, {SI1BB, SI2BB}))
405 continue;
406 Fail = true;
407 if (FailBlocks)
408 FailBlocks->insert(Succ);
409 else
410 break;
411 }
412
413 return !Fail;
414}
415
416/// Update PHI nodes in Succ to indicate that there will now be entries in it
417/// from the 'NewPred' block. The values that will be flowing into the PHI nodes
418/// will be the same as those coming in from ExistPred, an existing predecessor
419/// of Succ.
420static void addPredecessorToBlock(BasicBlock *Succ, BasicBlock *NewPred,
421 BasicBlock *ExistPred,
422 MemorySSAUpdater *MSSAU = nullptr) {
423 for (PHINode &PN : Succ->phis())
424 PN.addIncoming(PN.getIncomingValueForBlock(ExistPred), NewPred);
425 if (MSSAU)
426 if (auto *MPhi = MSSAU->getMemorySSA()->getMemoryAccess(Succ))
427 MPhi->addIncoming(MPhi->getIncomingValueForBlock(ExistPred), NewPred);
428}
429
430/// Compute an abstract "cost" of speculating the given instruction,
431/// which is assumed to be safe to speculate. TCC_Free means cheap,
432/// TCC_Basic means less cheap, and TCC_Expensive means prohibitively
433/// expensive.
435 const TargetTransformInfo &TTI) {
436 return TTI.getInstructionCost(I, TargetTransformInfo::TCK_SizeAndLatency);
437}
438
439/// If we have a merge point of an "if condition" as accepted above,
440/// return true if the specified value dominates the block. We don't handle
441/// the true generality of domination here, just a special case which works
442/// well enough for us.
443///
444/// If AggressiveInsts is non-null, and if V does not dominate BB, we check to
445/// see if V (which must be an instruction) and its recursive operands
446/// that do not dominate BB have a combined cost lower than Budget and
447/// are non-trapping. If both are true, the instruction is inserted into the
448/// set and true is returned.
449///
450/// The cost for most non-trapping instructions is defined as 1 except for
451/// Select whose cost is 2.
452///
453/// After this function returns, Cost is increased by the cost of
454/// V plus its non-dominating operands. If that cost is greater than
455/// Budget, false is returned and Cost is undefined.
457 Value *V, BasicBlock *BB, Instruction *InsertPt,
458 SmallPtrSetImpl<Instruction *> &AggressiveInsts, InstructionCost &Cost,
460 SmallPtrSetImpl<Instruction *> &ZeroCostInstructions, unsigned Depth = 0) {
461 // It is possible to hit a zero-cost cycle (phi/gep instructions for example),
462 // so limit the recursion depth.
463 // TODO: While this recursion limit does prevent pathological behavior, it
464 // would be better to track visited instructions to avoid cycles.
466 return false;
467
469 if (!I) {
470 // Non-instructions dominate all instructions and can be executed
471 // unconditionally.
472 return true;
473 }
474 BasicBlock *PBB = I->getParent();
475
476 // We don't want to allow weird loops that might have the "if condition" in
477 // the bottom of this block.
478 if (PBB == BB)
479 return false;
480
481 // If this instruction is defined in a block that contains an unconditional
482 // branch to BB, then it must be in the 'conditional' part of the "if
483 // statement". If not, it definitely dominates the region.
485 if (!BI || BI->getSuccessor() != BB)
486 return true;
487
488 // If we have seen this instruction before, don't count it again.
489 if (AggressiveInsts.count(I))
490 return true;
491
492 // Okay, it looks like the instruction IS in the "condition". Check to
493 // see if it's a cheap instruction to unconditionally compute, and if it
494 // only uses stuff defined outside of the condition. If so, hoist it out.
495 if (!isSafeToSpeculativelyExecute(I, InsertPt, AC))
496 return false;
497
498 // Overflow arithmetic instruction plus extract value are usually generated
499 // when a division is being replaced. But, in this case, the zero check may
500 // still be kept in the code. In that case it would be worth to hoist these
501 // two instruction out of the basic block. Let's treat this pattern as one
502 // single cheap instruction here!
503 WithOverflowInst *OverflowInst;
504 if (match(I, m_ExtractValue<1>(m_OneUse(m_WithOverflowInst(OverflowInst))))) {
505 ZeroCostInstructions.insert(OverflowInst);
506 Cost += 1;
507 } else if (!ZeroCostInstructions.contains(I))
508 Cost += computeSpeculationCost(I, TTI);
509
510 // Allow exactly one instruction to be speculated regardless of its cost
511 // (as long as it is safe to do so).
512 // This is intended to flatten the CFG even if the instruction is a division
513 // or other expensive operation. The speculation of an expensive instruction
514 // is expected to be undone in CodeGenPrepare if the speculation has not
515 // enabled further IR optimizations.
516 if (Cost > Budget &&
517 (!SpeculateOneExpensiveInst || !AggressiveInsts.empty() || Depth > 0 ||
518 !Cost.isValid()))
519 return false;
520
521 // Okay, we can only really hoist these out if their operands do
522 // not take us over the cost threshold.
523 for (Use &Op : I->operands())
524 if (!dominatesMergePoint(Op, BB, InsertPt, AggressiveInsts, Cost, Budget,
525 TTI, AC, ZeroCostInstructions, Depth + 1))
526 return false;
527 // Okay, it's safe to do this! Remember this instruction.
528 AggressiveInsts.insert(I);
529 return true;
530}
531
532/// Extract ConstantInt from value, looking through IntToPtr
533/// and PointerNullValue. Return NULL if value is not a constant int.
535 // Normal constant int.
537 if (CI || !isa<Constant>(V) || !V->getType()->isPointerTy())
538 return CI;
539
540 // It is not safe to look through inttoptr or ptrtoint when using unstable
541 // pointer types.
542 if (DL.hasUnstableRepresentation(V->getType()))
543 return nullptr;
544
545 // This is some kind of pointer constant. Turn it into a pointer-sized
546 // ConstantInt if possible.
547 IntegerType *IntPtrTy = cast<IntegerType>(DL.getIntPtrType(V->getType()));
548
549 // Null pointer means 0, see SelectionDAGBuilder::getValue(const Value*).
551 return ConstantInt::get(IntPtrTy, 0);
552
553 // IntToPtr const int, we can look through this if the semantics of
554 // inttoptr for this address space are a simple (truncating) bitcast.
556 if (CE->getOpcode() == Instruction::IntToPtr)
557 if (ConstantInt *CI = dyn_cast<ConstantInt>(CE->getOperand(0))) {
558 // The constant is very likely to have the right type already.
559 if (CI->getType() == IntPtrTy)
560 return CI;
561 else
562 return cast<ConstantInt>(
563 ConstantFoldIntegerCast(CI, IntPtrTy, /*isSigned=*/false, DL));
564 }
565 return nullptr;
566}
567
568namespace {
569
570/// Given a chain of or (||) or and (&&) comparison of a value against a
571/// constant, this will try to recover the information required for a switch
572/// structure.
573/// It will depth-first traverse the chain of comparison, seeking for patterns
574/// like %a == 12 or %a < 4 and combine them to produce a set of integer
575/// representing the different cases for the switch.
576/// Note that if the chain is composed of '||' it will build the set of elements
577/// that matches the comparisons (i.e. any of this value validate the chain)
578/// while for a chain of '&&' it will build the set elements that make the test
579/// fail.
580struct ConstantComparesGatherer {
581 const DataLayout &DL;
582
583 /// Value found for the switch comparison
584 Value *CompValue = nullptr;
585
586 /// Extra clause to be checked before the switch
587 Value *Extra = nullptr;
588
589 /// Set of integers to match in switch
591
592 /// Number of comparisons matched in the and/or chain
593 unsigned UsedICmps = 0;
594
595 /// If the elements in Vals matches the comparisons
596 bool IsEq = false;
597
598 // Used to check if the first matched CompValue shall be the Extra check.
599 bool IgnoreFirstMatch = false;
600 bool MultipleMatches = false;
601
602 /// Construct and compute the result for the comparison instruction Cond
603 ConstantComparesGatherer(Instruction *Cond, const DataLayout &DL) : DL(DL) {
604 gather(Cond);
605 if (CompValue || !MultipleMatches)
606 return;
607 Extra = nullptr;
608 Vals.clear();
609 UsedICmps = 0;
610 IgnoreFirstMatch = true;
611 gather(Cond);
612 }
613
614 ConstantComparesGatherer(const ConstantComparesGatherer &) = delete;
615 ConstantComparesGatherer &
616 operator=(const ConstantComparesGatherer &) = delete;
617
618private:
619 /// Try to set the current value used for the comparison, it succeeds only if
620 /// it wasn't set before or if the new value is the same as the old one
621 bool setValueOnce(Value *NewVal) {
622 if (IgnoreFirstMatch) {
623 IgnoreFirstMatch = false;
624 return false;
625 }
626 if (CompValue && CompValue != NewVal) {
627 MultipleMatches = true;
628 return false;
629 }
630 CompValue = NewVal;
631 return true;
632 }
633
634 /// Try to match Instruction "I" as a comparison against a constant and
635 /// populates the array Vals with the set of values that match (or do not
636 /// match depending on isEQ).
637 /// Return false on failure. On success, the Value the comparison matched
638 /// against is placed in CompValue.
639 /// If CompValue is already set, the function is expected to fail if a match
640 /// is found but the value compared to is different.
641 bool matchInstruction(Instruction *I, bool isEQ) {
642 if (match(I, m_Not(m_Instruction(I))))
643 isEQ = !isEQ;
644
645 Value *Val;
646 if (match(I, m_NUWTrunc(m_Value(Val)))) {
647 // If we already have a value for the switch, it has to match!
648 if (!setValueOnce(Val))
649 return false;
650 UsedICmps++;
651 Vals.push_back(ConstantInt::get(cast<IntegerType>(Val->getType()), isEQ));
652 return true;
653 }
654 // If this is an icmp against a constant, handle this as one of the cases.
655 ICmpInst *ICI;
656 ConstantInt *C;
657 if (!((ICI = dyn_cast<ICmpInst>(I)) &&
658 (C = getConstantInt(I->getOperand(1), DL)))) {
659 return false;
660 }
661
662 Value *RHSVal;
663 const APInt *RHSC;
664
665 // Pattern match a special case
666 // (x & ~2^z) == y --> x == y || x == y|2^z
667 // This undoes a transformation done by instcombine to fuse 2 compares.
668 if (ICI->getPredicate() == (isEQ ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_NE)) {
669 // It's a little bit hard to see why the following transformations are
670 // correct. Here is a CVC3 program to verify them for 64-bit values:
671
672 /*
673 ONE : BITVECTOR(64) = BVZEROEXTEND(0bin1, 63);
674 x : BITVECTOR(64);
675 y : BITVECTOR(64);
676 z : BITVECTOR(64);
677 mask : BITVECTOR(64) = BVSHL(ONE, z);
678 QUERY( (y & ~mask = y) =>
679 ((x & ~mask = y) <=> (x = y OR x = (y | mask)))
680 );
681 QUERY( (y | mask = y) =>
682 ((x | mask = y) <=> (x = y OR x = (y & ~mask)))
683 );
684 */
685
686 // Please note that each pattern must be a dual implication (<--> or
687 // iff). One directional implication can create spurious matches. If the
688 // implication is only one-way, an unsatisfiable condition on the left
689 // side can imply a satisfiable condition on the right side. Dual
690 // implication ensures that satisfiable conditions are transformed to
691 // other satisfiable conditions and unsatisfiable conditions are
692 // transformed to other unsatisfiable conditions.
693
694 // Here is a concrete example of a unsatisfiable condition on the left
695 // implying a satisfiable condition on the right:
696 //
697 // mask = (1 << z)
698 // (x & ~mask) == y --> (x == y || x == (y | mask))
699 //
700 // Substituting y = 3, z = 0 yields:
701 // (x & -2) == 3 --> (x == 3 || x == 2)
702
703 // Pattern match a special case:
704 /*
705 QUERY( (y & ~mask = y) =>
706 ((x & ~mask = y) <=> (x = y OR x = (y | mask)))
707 );
708 */
709 if (match(ICI->getOperand(0),
710 m_And(m_Value(RHSVal), m_APInt(RHSC)))) {
711 APInt Mask = ~*RHSC;
712 if (Mask.isPowerOf2() && (C->getValue() & ~Mask) == C->getValue()) {
713 // If we already have a value for the switch, it has to match!
714 if (!setValueOnce(RHSVal))
715 return false;
716
717 Vals.push_back(C);
718 Vals.push_back(
719 ConstantInt::get(C->getContext(),
720 C->getValue() | Mask));
721 UsedICmps++;
722 return true;
723 }
724 }
725
726 // Pattern match a special case:
727 /*
728 QUERY( (y | mask = y) =>
729 ((x | mask = y) <=> (x = y OR x = (y & ~mask)))
730 );
731 */
732 if (match(ICI->getOperand(0),
733 m_Or(m_Value(RHSVal), m_APInt(RHSC)))) {
734 APInt Mask = *RHSC;
735 if (Mask.isPowerOf2() && (C->getValue() | Mask) == C->getValue()) {
736 // If we already have a value for the switch, it has to match!
737 if (!setValueOnce(RHSVal))
738 return false;
739
740 Vals.push_back(C);
741 Vals.push_back(ConstantInt::get(C->getContext(),
742 C->getValue() & ~Mask));
743 UsedICmps++;
744 return true;
745 }
746 }
747
748 // If we already have a value for the switch, it has to match!
749 if (!setValueOnce(ICI->getOperand(0)))
750 return false;
751
752 UsedICmps++;
753 Vals.push_back(C);
754 return true;
755 }
756
757 // If we have "x ult 3", for example, then we can add 0,1,2 to the set.
758 ConstantRange Span =
760
761 // Shift the range if the compare is fed by an add. This is the range
762 // compare idiom as emitted by instcombine.
763 Value *CandidateVal = I->getOperand(0);
764 if (match(I->getOperand(0), m_Add(m_Value(RHSVal), m_APInt(RHSC)))) {
765 Span = Span.subtract(*RHSC);
766 CandidateVal = RHSVal;
767 }
768
769 // If this is an and/!= check, then we are looking to build the set of
770 // value that *don't* pass the and chain. I.e. to turn "x ugt 2" into
771 // x != 0 && x != 1.
772 if (!isEQ)
773 Span = Span.inverse();
774
775 // If there are a ton of values, we don't want to make a ginormous switch.
776 if (Span.isSizeLargerThan(8) || Span.isEmptySet()) {
777 return false;
778 }
779
780 // If we already have a value for the switch, it has to match!
781 if (!setValueOnce(CandidateVal))
782 return false;
783
784 // Add all values from the range to the set
785 APInt Tmp = Span.getLower();
786 do
787 Vals.push_back(ConstantInt::get(I->getContext(), Tmp));
788 while (++Tmp != Span.getUpper());
789
790 UsedICmps++;
791 return true;
792 }
793
794 /// Given a potentially 'or'd or 'and'd together collection of icmp
795 /// eq/ne/lt/gt instructions that compare a value against a constant, extract
796 /// the value being compared, and stick the list constants into the Vals
797 /// vector.
798 /// One "Extra" case is allowed to differ from the other.
799 void gather(Value *V) {
800 Value *Op0, *Op1;
801 if (match(V, m_LogicalOr(m_Value(Op0), m_Value(Op1))))
802 IsEq = true;
803 else if (match(V, m_LogicalAnd(m_Value(Op0), m_Value(Op1))))
804 IsEq = false;
805 else
806 return;
807 // Keep a stack (SmallVector for efficiency) for depth-first traversal
808 SmallVector<Value *, 8> DFT{Op0, Op1};
809 SmallPtrSet<Value *, 8> Visited{V, Op0, Op1};
810
811 while (!DFT.empty()) {
812 V = DFT.pop_back_val();
813
814 if (Instruction *I = dyn_cast<Instruction>(V)) {
815 // If it is a || (or && depending on isEQ), process the operands.
816 if (IsEq ? match(I, m_LogicalOr(m_Value(Op0), m_Value(Op1)))
817 : match(I, m_LogicalAnd(m_Value(Op0), m_Value(Op1)))) {
818 if (Visited.insert(Op1).second)
819 DFT.push_back(Op1);
820 if (Visited.insert(Op0).second)
821 DFT.push_back(Op0);
822
823 continue;
824 }
825
826 // Try to match the current instruction
827 if (matchInstruction(I, IsEq))
828 // Match succeed, continue the loop
829 continue;
830 }
831
832 // One element of the sequence of || (or &&) could not be match as a
833 // comparison against the same value as the others.
834 // We allow only one "Extra" case to be checked before the switch
835 if (!Extra) {
836 Extra = V;
837 continue;
838 }
839 // Failed to parse a proper sequence, abort now
840 CompValue = nullptr;
841 break;
842 }
843 }
844};
845
846} // end anonymous namespace
847
849 MemorySSAUpdater *MSSAU = nullptr) {
850 Instruction *Cond = nullptr;
852 Cond = dyn_cast<Instruction>(SI->getCondition());
853 } else if (CondBrInst *BI = dyn_cast<CondBrInst>(TI)) {
854 Cond = dyn_cast<Instruction>(BI->getCondition());
855 } else if (IndirectBrInst *IBI = dyn_cast<IndirectBrInst>(TI)) {
856 Cond = dyn_cast<Instruction>(IBI->getAddress());
857 }
858
859 TI->eraseFromParent();
860 if (Cond)
862}
863
864/// Return true if the specified terminator checks
865/// to see if a value is equal to constant integer value.
866Value *SimplifyCFGOpt::isValueEqualityComparison(Instruction *TI) {
867 Value *CV = nullptr;
868 if (SwitchInst *SI = dyn_cast<SwitchInst>(TI)) {
869 // Do not permit merging of large switch instructions into their
870 // predecessors unless there is only one predecessor.
871 if (!SI->getParent()->hasNPredecessorsOrMore(128 / SI->getNumSuccessors()))
872 CV = SI->getCondition();
873 } else if (CondBrInst *BI = dyn_cast<CondBrInst>(TI))
874 if (BI->getCondition()->hasOneUse()) {
875 if (ICmpInst *ICI = dyn_cast<ICmpInst>(BI->getCondition())) {
876 if (ICI->isEquality() && getConstantInt(ICI->getOperand(1), DL))
877 CV = ICI->getOperand(0);
878 } else if (auto *Trunc = dyn_cast<TruncInst>(BI->getCondition())) {
879 if (Trunc->hasNoUnsignedWrap())
880 CV = Trunc->getOperand(0);
881 }
882 }
883
884 // Unwrap any lossless ptrtoint cast (except for unstable pointers).
885 if (CV) {
886 if (PtrToIntInst *PTII = dyn_cast<PtrToIntInst>(CV)) {
887 Value *Ptr = PTII->getPointerOperand();
888 if (DL.hasUnstableRepresentation(Ptr->getType()))
889 return CV;
890 if (PTII->getType() == DL.getIntPtrType(Ptr->getType()))
891 CV = Ptr;
892 }
893 }
894 return CV;
895}
896
897/// Given a value comparison instruction,
898/// decode all of the 'cases' that it represents and return the 'default' block.
899BasicBlock *SimplifyCFGOpt::getValueEqualityComparisonCases(
900 Instruction *TI, std::vector<ValueEqualityComparisonCase> &Cases) {
901 if (SwitchInst *SI = dyn_cast<SwitchInst>(TI)) {
902 Cases.reserve(SI->getNumCases());
903 for (auto Case : SI->cases())
904 Cases.push_back(ValueEqualityComparisonCase(Case.getCaseValue(),
905 Case.getCaseSuccessor()));
906 return SI->getDefaultDest();
907 }
908
909 CondBrInst *BI = cast<CondBrInst>(TI);
910 Value *Cond = BI->getCondition();
911 ICmpInst::Predicate Pred;
912 ConstantInt *C;
913 if (auto *ICI = dyn_cast<ICmpInst>(Cond)) {
914 Pred = ICI->getPredicate();
915 C = getConstantInt(ICI->getOperand(1), DL);
916 } else {
917 Pred = ICmpInst::ICMP_NE;
918 auto *Trunc = cast<TruncInst>(Cond);
919 C = ConstantInt::get(cast<IntegerType>(Trunc->getOperand(0)->getType()), 0);
920 }
921 BasicBlock *Succ = BI->getSuccessor(Pred == ICmpInst::ICMP_NE);
922 Cases.push_back(ValueEqualityComparisonCase(C, Succ));
923 return BI->getSuccessor(Pred == ICmpInst::ICMP_EQ);
924}
925
926/// Given a vector of bb/value pairs, remove any entries
927/// in the list that match the specified block.
928static void
930 std::vector<ValueEqualityComparisonCase> &Cases) {
931 llvm::erase(Cases, BB);
932}
933
934/// Return true if there are any keys in C1 that exist in C2 as well.
935static bool valuesOverlap(std::vector<ValueEqualityComparisonCase> &C1,
936 std::vector<ValueEqualityComparisonCase> &C2) {
937 std::vector<ValueEqualityComparisonCase> *V1 = &C1, *V2 = &C2;
938
939 // Make V1 be smaller than V2.
940 if (V1->size() > V2->size())
941 std::swap(V1, V2);
942
943 if (V1->empty())
944 return false;
945 if (V1->size() == 1) {
946 // Just scan V2.
947 ConstantInt *TheVal = (*V1)[0].Value;
948 for (const ValueEqualityComparisonCase &VECC : *V2)
949 if (TheVal == VECC.Value)
950 return true;
951 }
952
953 // Otherwise, just sort both lists and compare element by element.
954 array_pod_sort(V1->begin(), V1->end());
955 array_pod_sort(V2->begin(), V2->end());
956 unsigned i1 = 0, i2 = 0, e1 = V1->size(), e2 = V2->size();
957 while (i1 != e1 && i2 != e2) {
958 if ((*V1)[i1].Value == (*V2)[i2].Value)
959 return true;
960 if ((*V1)[i1].Value < (*V2)[i2].Value)
961 ++i1;
962 else
963 ++i2;
964 }
965 return false;
966}
967
968/// If TI is known to be a terminator instruction and its block is known to
969/// only have a single predecessor block, check to see if that predecessor is
970/// also a value comparison with the same value, and if that comparison
971/// determines the outcome of this comparison. If so, simplify TI. This does a
972/// very limited form of jump threading.
973bool SimplifyCFGOpt::simplifyEqualityComparisonWithOnlyPredecessor(
974 Instruction *TI, BasicBlock *Pred, IRBuilder<> &Builder) {
975 Value *PredVal = isValueEqualityComparison(Pred->getTerminator());
976 if (!PredVal)
977 return false; // Not a value comparison in predecessor.
978
979 Value *ThisVal = isValueEqualityComparison(TI);
980 assert(ThisVal && "This isn't a value comparison!!");
981 if (ThisVal != PredVal)
982 return false; // Different predicates.
983
984 // TODO: Preserve branch weight metadata, similarly to how
985 // foldValueComparisonIntoPredecessors preserves it.
986
987 // Find out information about when control will move from Pred to TI's block.
988 std::vector<ValueEqualityComparisonCase> PredCases;
989 BasicBlock *PredDef =
990 getValueEqualityComparisonCases(Pred->getTerminator(), PredCases);
991 eliminateBlockCases(PredDef, PredCases); // Remove default from cases.
992
993 // Find information about how control leaves this block.
994 std::vector<ValueEqualityComparisonCase> ThisCases;
995 BasicBlock *ThisDef = getValueEqualityComparisonCases(TI, ThisCases);
996 eliminateBlockCases(ThisDef, ThisCases); // Remove default from cases.
997
998 // If TI's block is the default block from Pred's comparison, potentially
999 // simplify TI based on this knowledge.
1000 if (PredDef == TI->getParent()) {
1001 // If we are here, we know that the value is none of those cases listed in
1002 // PredCases. If there are any cases in ThisCases that are in PredCases, we
1003 // can simplify TI.
1004 if (!valuesOverlap(PredCases, ThisCases))
1005 return false;
1006
1007 if (isa<CondBrInst>(TI)) {
1008 // Okay, one of the successors of this condbr is dead. Convert it to a
1009 // uncond br.
1010 assert(ThisCases.size() == 1 && "Branch can only have one case!");
1011 // Insert the new branch.
1012 Instruction *NI = Builder.CreateBr(ThisDef);
1013 (void)NI;
1014
1015 // Remove PHI node entries for the dead edge.
1016 ThisCases[0].Dest->removePredecessor(PredDef);
1017
1018 LLVM_DEBUG(dbgs() << "Threading pred instr: " << *Pred->getTerminator()
1019 << "Through successor TI: " << *TI << "Leaving: " << *NI
1020 << "\n");
1021
1023
1024 if (DTU)
1025 DTU->applyUpdates(
1026 {{DominatorTree::Delete, PredDef, ThisCases[0].Dest}});
1027
1028 return true;
1029 }
1030
1031 SwitchInstProfUpdateWrapper SI = *cast<SwitchInst>(TI);
1032 // Okay, TI has cases that are statically dead, prune them away.
1033 SmallPtrSet<Constant *, 16> DeadCases;
1034 for (const ValueEqualityComparisonCase &Case : PredCases)
1035 DeadCases.insert(Case.Value);
1036
1037 LLVM_DEBUG(dbgs() << "Threading pred instr: " << *Pred->getTerminator()
1038 << "Through successor TI: " << *TI);
1039
1040 SmallDenseMap<BasicBlock *, int, 8> NumPerSuccessorCases;
1041 for (SwitchInst::CaseIt i = SI->case_end(), e = SI->case_begin(); i != e;) {
1042 --i;
1043 auto *Successor = i->getCaseSuccessor();
1044 if (DTU)
1045 ++NumPerSuccessorCases[Successor];
1046 if (DeadCases.count(i->getCaseValue())) {
1047 Successor->removePredecessor(PredDef);
1048 SI.removeCase(i);
1049 if (DTU)
1050 --NumPerSuccessorCases[Successor];
1051 }
1052 }
1053
1054 if (DTU) {
1055 std::vector<DominatorTree::UpdateType> Updates;
1056 for (const auto &I : NumPerSuccessorCases)
1057 if (I.second == 0)
1058 Updates.push_back({DominatorTree::Delete, PredDef, I.first});
1059 DTU->applyUpdates(Updates);
1060 }
1061
1062 LLVM_DEBUG(dbgs() << "Leaving: " << *TI << "\n");
1063 return true;
1064 }
1065
1066 // Otherwise, TI's block must correspond to some matched value. Find out
1067 // which value (or set of values) this is.
1068 ConstantInt *TIV = nullptr;
1069 BasicBlock *TIBB = TI->getParent();
1070 for (const auto &[Value, Dest] : PredCases)
1071 if (Dest == TIBB) {
1072 if (TIV)
1073 return false; // Cannot handle multiple values coming to this block.
1074 TIV = Value;
1075 }
1076 assert(TIV && "No edge from pred to succ?");
1077
1078 // Okay, we found the one constant that our value can be if we get into TI's
1079 // BB. Find out which successor will unconditionally be branched to.
1080 BasicBlock *TheRealDest = nullptr;
1081 for (const auto &[Value, Dest] : ThisCases)
1082 if (Value == TIV) {
1083 TheRealDest = Dest;
1084 break;
1085 }
1086
1087 // If not handled by any explicit cases, it is handled by the default case.
1088 if (!TheRealDest)
1089 TheRealDest = ThisDef;
1090
1091 SmallPtrSet<BasicBlock *, 2> RemovedSuccs;
1092
1093 // Remove PHI node entries for dead edges.
1094 BasicBlock *CheckEdge = TheRealDest;
1095 for (BasicBlock *Succ : successors(TIBB))
1096 if (Succ != CheckEdge) {
1097 if (Succ != TheRealDest)
1098 RemovedSuccs.insert(Succ);
1099 Succ->removePredecessor(TIBB);
1100 } else
1101 CheckEdge = nullptr;
1102
1103 // Insert the new branch.
1104 Instruction *NI = Builder.CreateBr(TheRealDest);
1105 (void)NI;
1106
1107 LLVM_DEBUG(dbgs() << "Threading pred instr: " << *Pred->getTerminator()
1108 << "Through successor TI: " << *TI << "Leaving: " << *NI
1109 << "\n");
1110
1112 if (DTU) {
1113 SmallVector<DominatorTree::UpdateType, 2> Updates;
1114 Updates.reserve(RemovedSuccs.size());
1115 for (auto *RemovedSucc : RemovedSuccs)
1116 Updates.push_back({DominatorTree::Delete, TIBB, RemovedSucc});
1117 DTU->applyUpdates(Updates);
1118 }
1119 return true;
1120}
1121
1122namespace {
1123
1124/// This class implements a stable ordering of constant
1125/// integers that does not depend on their address. This is important for
1126/// applications that sort ConstantInt's to ensure uniqueness.
1127struct ConstantIntOrdering {
1128 bool operator()(const ConstantInt *LHS, const ConstantInt *RHS) const {
1129 return LHS->getValue().ult(RHS->getValue());
1130 }
1131};
1132
1133} // end anonymous namespace
1134
1136 ConstantInt *const *P2) {
1137 const ConstantInt *LHS = *P1;
1138 const ConstantInt *RHS = *P2;
1139 if (LHS == RHS)
1140 return 0;
1141 return LHS->getValue().ult(RHS->getValue()) ? 1 : -1;
1142}
1143
1144/// Get Weights of a given terminator, the default weight is at the front
1145/// of the vector. If TI is a conditional eq, we need to swap the branch-weight
1146/// metadata.
1148 SmallVectorImpl<uint64_t> &Weights) {
1149 MDNode *MD = TI->getMetadata(LLVMContext::MD_prof);
1150 assert(MD && "Invalid branch-weight metadata");
1151 extractFromBranchWeightMD64(MD, Weights);
1152
1153 // If TI is a conditional eq, the default case is the false case,
1154 // and the corresponding branch-weight data is at index 2. We swap the
1155 // default weight to be the first entry.
1156 if (CondBrInst *BI = dyn_cast<CondBrInst>(TI)) {
1157 assert(Weights.size() == 2);
1158 auto *ICI = dyn_cast<ICmpInst>(BI->getCondition());
1159 if (!ICI)
1160 return;
1161
1162 if (ICI->getPredicate() == ICmpInst::ICMP_EQ)
1163 std::swap(Weights.front(), Weights.back());
1164 }
1165}
1166
1168 BasicBlock *BB, BasicBlock *PredBlock, ValueToValueMapTy &VMap) {
1169 Instruction *PTI = PredBlock->getTerminator();
1170
1171 // If we have bonus instructions, clone them into the predecessor block.
1172 // Note that there may be multiple predecessor blocks, so we cannot move
1173 // bonus instructions to a predecessor block.
1174 for (Instruction &BonusInst : *BB) {
1175 if (BonusInst.isTerminator())
1176 continue;
1177
1178 // Skip cloning pseudo probes into the predecessor, as it would overcount
1179 // otherwise.
1180 if (isa<PseudoProbeInst>(BonusInst))
1181 continue;
1182
1183 Instruction *NewBonusInst = BonusInst.clone();
1184 NewBonusInst->insertInto(PredBlock, PTI->getIterator());
1185
1186 if (!NewBonusInst->getDebugLoc().isSameSourceLocation(PTI->getDebugLoc())) {
1187 // Unless the instruction has the same !dbg location as the original
1188 // branch, drop it. When we fold the bonus instructions we want to make
1189 // sure we reset their debug locations in order to avoid stepping on
1190 // dead code caused by folding dead branches.
1191 NewBonusInst->dropLocation();
1192 } else if (const DebugLoc &DL = NewBonusInst->getDebugLoc()) {
1193 mapAtomInstance(DL, VMap);
1194 }
1195
1196 RemapInstruction(NewBonusInst, VMap,
1198
1199 // If we speculated an instruction, we need to drop any metadata that may
1200 // result in undefined behavior, as the metadata might have been valid
1201 // only given the branch precondition.
1202 // Similarly strip attributes on call parameters that may cause UB in
1203 // location the call is moved to.
1204 NewBonusInst->dropUBImplyingAttrsAndMetadata();
1205
1206 auto Range = NewBonusInst->cloneDebugInfoFrom(&BonusInst);
1207 RemapDbgRecordRange(NewBonusInst->getModule(), Range, VMap,
1209
1210 NewBonusInst->takeName(&BonusInst);
1211 BonusInst.setName(NewBonusInst->getName() + ".old");
1212 VMap[&BonusInst] = NewBonusInst;
1213
1214 // Update (liveout) uses of bonus instructions,
1215 // now that the bonus instruction has been cloned into predecessor.
1216 // Note that we expect to be in a block-closed SSA form for this to work!
1217 for (Use &U : make_early_inc_range(BonusInst.uses())) {
1218 auto *UI = cast<Instruction>(U.getUser());
1219 auto *PN = dyn_cast<PHINode>(UI);
1220 if (!PN) {
1221 assert(UI->getParent() == BB && BonusInst.comesBefore(UI) &&
1222 "If the user is not a PHI node, then it should be in the same "
1223 "block as, and come after, the original bonus instruction.");
1224 continue; // Keep using the original bonus instruction.
1225 }
1226 // Is this the block-closed SSA form PHI node?
1227 if (PN->getIncomingBlock(U) == BB)
1228 continue; // Great, keep using the original bonus instruction.
1229 // The only other alternative is an "use" when coming from
1230 // the predecessor block - here we should refer to the cloned bonus instr.
1231 assert(PN->getIncomingBlock(U) == PredBlock &&
1232 "Not in block-closed SSA form?");
1233 U.set(NewBonusInst);
1234 }
1235 }
1236
1237 // Key Instructions: We may have propagated atom info into the pred. If the
1238 // pred's terminator already has atom info do nothing as merging would drop
1239 // one atom group anyway. If it doesn't, propagte the remapped atom group
1240 // from BB's terminator.
1241 if (auto &PredDL = PTI->getDebugLoc()) {
1242 auto &DL = BB->getTerminator()->getDebugLoc();
1243 if (!PredDL->getAtomGroup() && DL && DL->getAtomGroup() &&
1244 PredDL.isSameSourceLocation(DL)) {
1245 PTI->setDebugLoc(DL);
1246 RemapSourceAtom(PTI, VMap);
1247 }
1248 }
1249}
1250
1251bool SimplifyCFGOpt::performValueComparisonIntoPredecessorFolding(
1252 Instruction *TI, Value *&CV, Instruction *PTI, IRBuilder<> &Builder) {
1253 BasicBlock *BB = TI->getParent();
1254 BasicBlock *Pred = PTI->getParent();
1255
1257
1258 // Figure out which 'cases' to copy from SI to PSI.
1259 std::vector<ValueEqualityComparisonCase> BBCases;
1260 BasicBlock *BBDefault = getValueEqualityComparisonCases(TI, BBCases);
1261
1262 std::vector<ValueEqualityComparisonCase> PredCases;
1263 BasicBlock *PredDefault = getValueEqualityComparisonCases(PTI, PredCases);
1264
1265 // Based on whether the default edge from PTI goes to BB or not, fill in
1266 // PredCases and PredDefault with the new switch cases we would like to
1267 // build.
1268 SmallMapVector<BasicBlock *, int, 8> NewSuccessors;
1269
1270 // Update the branch weight metadata along the way
1271 SmallVector<uint64_t, 8> Weights;
1272 bool PredHasWeights = hasBranchWeightMD(*PTI);
1273 bool SuccHasWeights = hasBranchWeightMD(*TI);
1274
1275 if (PredHasWeights) {
1276 getBranchWeights(PTI, Weights);
1277 // branch-weight metadata is inconsistent here.
1278 if (Weights.size() != 1 + PredCases.size())
1279 PredHasWeights = SuccHasWeights = false;
1280 } else if (SuccHasWeights)
1281 // If there are no predecessor weights but there are successor weights,
1282 // populate Weights with 1, which will later be scaled to the sum of
1283 // successor's weights
1284 Weights.assign(1 + PredCases.size(), 1);
1285
1286 SmallVector<uint64_t, 8> SuccWeights;
1287 if (SuccHasWeights) {
1288 getBranchWeights(TI, SuccWeights);
1289 // branch-weight metadata is inconsistent here.
1290 if (SuccWeights.size() != 1 + BBCases.size())
1291 PredHasWeights = SuccHasWeights = false;
1292 } else if (PredHasWeights)
1293 SuccWeights.assign(1 + BBCases.size(), 1);
1294
1295 if (PredDefault == BB) {
1296 // If this is the default destination from PTI, only the edges in TI
1297 // that don't occur in PTI, or that branch to BB will be activated.
1298 std::set<ConstantInt *, ConstantIntOrdering> PTIHandled;
1299 for (unsigned i = 0, e = PredCases.size(); i != e; ++i)
1300 if (PredCases[i].Dest != BB)
1301 PTIHandled.insert(PredCases[i].Value);
1302 else {
1303 // The default destination is BB, we don't need explicit targets.
1304 std::swap(PredCases[i], PredCases.back());
1305
1306 if (PredHasWeights || SuccHasWeights) {
1307 // Increase weight for the default case.
1308 Weights[0] += Weights[i + 1];
1309 std::swap(Weights[i + 1], Weights.back());
1310 Weights.pop_back();
1311 }
1312
1313 PredCases.pop_back();
1314 --i;
1315 --e;
1316 }
1317
1318 // Reconstruct the new switch statement we will be building.
1319 if (PredDefault != BBDefault) {
1320 PredDefault->removePredecessor(Pred);
1321 if (DTU && PredDefault != BB)
1322 Updates.push_back({DominatorTree::Delete, Pred, PredDefault});
1323 PredDefault = BBDefault;
1324 ++NewSuccessors[BBDefault];
1325 }
1326
1327 unsigned CasesFromPred = Weights.size();
1328 uint64_t ValidTotalSuccWeight = 0;
1329 for (unsigned i = 0, e = BBCases.size(); i != e; ++i)
1330 if (!PTIHandled.count(BBCases[i].Value) && BBCases[i].Dest != BBDefault) {
1331 PredCases.push_back(BBCases[i]);
1332 ++NewSuccessors[BBCases[i].Dest];
1333 if (SuccHasWeights || PredHasWeights) {
1334 // The default weight is at index 0, so weight for the ith case
1335 // should be at index i+1. Scale the cases from successor by
1336 // PredDefaultWeight (Weights[0]).
1337 Weights.push_back(Weights[0] * SuccWeights[i + 1]);
1338 ValidTotalSuccWeight += SuccWeights[i + 1];
1339 }
1340 }
1341
1342 if (SuccHasWeights || PredHasWeights) {
1343 ValidTotalSuccWeight += SuccWeights[0];
1344 // Scale the cases from predecessor by ValidTotalSuccWeight.
1345 for (unsigned i = 1; i < CasesFromPred; ++i)
1346 Weights[i] *= ValidTotalSuccWeight;
1347 // Scale the default weight by SuccDefaultWeight (SuccWeights[0]).
1348 Weights[0] *= SuccWeights[0];
1349 }
1350 } else {
1351 // If this is not the default destination from PSI, only the edges
1352 // in SI that occur in PSI with a destination of BB will be
1353 // activated.
1354 std::set<ConstantInt *, ConstantIntOrdering> PTIHandled;
1355 std::map<ConstantInt *, uint64_t> WeightsForHandled;
1356 for (unsigned i = 0, e = PredCases.size(); i != e; ++i)
1357 if (PredCases[i].Dest == BB) {
1358 PTIHandled.insert(PredCases[i].Value);
1359
1360 if (PredHasWeights || SuccHasWeights) {
1361 WeightsForHandled[PredCases[i].Value] = Weights[i + 1];
1362 std::swap(Weights[i + 1], Weights.back());
1363 Weights.pop_back();
1364 }
1365
1366 std::swap(PredCases[i], PredCases.back());
1367 PredCases.pop_back();
1368 --i;
1369 --e;
1370 }
1371
1372 // Okay, now we know which constants were sent to BB from the
1373 // predecessor. Figure out where they will all go now.
1374 for (const ValueEqualityComparisonCase &Case : BBCases)
1375 if (PTIHandled.count(Case.Value)) {
1376 // If this is one we are capable of getting...
1377 if (PredHasWeights || SuccHasWeights)
1378 Weights.push_back(WeightsForHandled[Case.Value]);
1379 PredCases.push_back(Case);
1380 ++NewSuccessors[Case.Dest];
1381 PTIHandled.erase(Case.Value); // This constant is taken care of
1382 }
1383
1384 // If there are any constants vectored to BB that TI doesn't handle,
1385 // they must go to the default destination of TI.
1386 for (ConstantInt *I : PTIHandled) {
1387 if (PredHasWeights || SuccHasWeights)
1388 Weights.push_back(WeightsForHandled[I]);
1389 PredCases.push_back(ValueEqualityComparisonCase(I, BBDefault));
1390 ++NewSuccessors[BBDefault];
1391 }
1392 }
1393
1394 // Okay, at this point, we know which new successor Pred will get. Make
1395 // sure we update the number of entries in the PHI nodes for these
1396 // successors.
1397 SmallPtrSet<BasicBlock *, 2> SuccsOfPred;
1398 if (DTU) {
1399 SuccsOfPred = {llvm::from_range, successors(Pred)};
1400 Updates.reserve(Updates.size() + NewSuccessors.size());
1401 }
1402 for (const std::pair<BasicBlock *, int /*Num*/> &NewSuccessor :
1403 NewSuccessors) {
1404 for (auto I : seq(NewSuccessor.second)) {
1405 (void)I;
1406 addPredecessorToBlock(NewSuccessor.first, Pred, BB);
1407 }
1408 if (DTU && !SuccsOfPred.contains(NewSuccessor.first))
1409 Updates.push_back({DominatorTree::Insert, Pred, NewSuccessor.first});
1410 }
1411
1412 Builder.SetInsertPoint(PTI);
1413 // Convert pointer to int before we switch.
1414 if (CV->getType()->isPointerTy()) {
1415 assert(!DL.hasUnstableRepresentation(CV->getType()) &&
1416 "Should not end up here with unstable pointers");
1417 CV =
1418 Builder.CreatePtrToInt(CV, DL.getIntPtrType(CV->getType()), "magicptr");
1419 }
1420
1421 // Now that the successors are updated, create the new Switch instruction.
1422 SwitchInst *NewSI = Builder.CreateSwitch(CV, PredDefault, PredCases.size());
1423 NewSI->setDebugLoc(PTI->getDebugLoc());
1424 for (ValueEqualityComparisonCase &V : PredCases)
1425 NewSI->addCase(V.Value, V.Dest);
1426
1427 if (PredHasWeights || SuccHasWeights)
1428 setFittedBranchWeights(*NewSI, Weights, /*IsExpected=*/false,
1429 /*ElideAllZero=*/true);
1430
1431 // The new switch is only known to be unpredictable if both of the comparisons
1432 // it was built from were unpredictable.
1433 if (MDNode *Unpredictable = PTI->getMetadata(LLVMContext::MD_unpredictable))
1434 if (TI->hasMetadata(LLVMContext::MD_unpredictable))
1435 NewSI->setMetadata(LLVMContext::MD_unpredictable, Unpredictable);
1436
1438
1439 // Okay, last check. If BB is still a successor of PSI, then we must
1440 // have an infinite loop case. If so, add an infinitely looping block
1441 // to handle the case to preserve the behavior of the code.
1442 BasicBlock *InfLoopBlock = nullptr;
1443 for (unsigned i = 0, e = NewSI->getNumSuccessors(); i != e; ++i)
1444 if (NewSI->getSuccessor(i) == BB) {
1445 if (!InfLoopBlock) {
1446 // Insert it at the end of the function, because it's either code,
1447 // or it won't matter if it's hot. :)
1448 InfLoopBlock =
1449 BasicBlock::Create(BB->getContext(), "infloop", BB->getParent());
1450 UncondBrInst::Create(InfLoopBlock, InfLoopBlock);
1451 if (DTU)
1452 Updates.push_back(
1453 {DominatorTree::Insert, InfLoopBlock, InfLoopBlock});
1454 }
1455 NewSI->setSuccessor(i, InfLoopBlock);
1456 }
1457
1458 if (DTU) {
1459 if (InfLoopBlock)
1460 Updates.push_back({DominatorTree::Insert, Pred, InfLoopBlock});
1461
1462 Updates.push_back({DominatorTree::Delete, Pred, BB});
1463
1464 DTU->applyUpdates(Updates);
1465 }
1466
1467 ++NumFoldValueComparisonIntoPredecessors;
1468 return true;
1469}
1470
1471/// The specified terminator is a value equality comparison instruction
1472/// (either a switch or a branch on "X == c").
1473/// See if any of the predecessors of the terminator block are value comparisons
1474/// on the same value. If so, and if safe to do so, fold them together.
1475bool SimplifyCFGOpt::foldValueComparisonIntoPredecessors(Instruction *TI,
1476 IRBuilder<> &Builder) {
1477 BasicBlock *BB = TI->getParent();
1478 Value *CV = isValueEqualityComparison(TI); // CondVal
1479 assert(CV && "Not a comparison?");
1480
1481 bool Changed = false;
1482
1483 SmallSetVector<BasicBlock *, 16> Preds(pred_begin(BB), pred_end(BB));
1484 while (!Preds.empty()) {
1485 BasicBlock *Pred = Preds.pop_back_val();
1486 Instruction *PTI = Pred->getTerminator();
1487
1488 // Don't try to fold into itself.
1489 if (Pred == BB)
1490 continue;
1491
1492 // See if the predecessor is a comparison with the same value.
1493 Value *PCV = isValueEqualityComparison(PTI); // PredCondVal
1494 if (PCV != CV)
1495 continue;
1496
1497 SmallSetVector<BasicBlock *, 4> FailBlocks;
1498 if (!safeToMergeTerminators(TI, PTI, &FailBlocks)) {
1499 for (auto *Succ : FailBlocks) {
1500 if (!SplitBlockPredecessors(Succ, TI->getParent(), ".fold.split", DTU))
1501 return false;
1502 }
1503 }
1504
1505 performValueComparisonIntoPredecessorFolding(TI, CV, PTI, Builder);
1506 Changed = true;
1507 }
1508 return Changed;
1509}
1510
1511// If we would need to insert a select that uses the value of this invoke
1512// (comments in hoistSuccIdenticalTerminatorToSwitchOrIf explain why we would
1513// need to do this), we can't hoist the invoke, as there is nowhere to put the
1514// select in this case.
1516 Instruction *I1, Instruction *I2) {
1517 for (BasicBlock *Succ : successors(BB1)) {
1518 for (const PHINode &PN : Succ->phis()) {
1519 Value *BB1V = PN.getIncomingValueForBlock(BB1);
1520 Value *BB2V = PN.getIncomingValueForBlock(BB2);
1521 if (BB1V != BB2V && (BB1V == I1 || BB2V == I2)) {
1522 return false;
1523 }
1524 }
1525 }
1526 return true;
1527}
1528
1529// Get interesting characteristics of instructions that
1530// `hoistCommonCodeFromSuccessors` didn't hoist. They restrict what kind of
1531// instructions can be reordered across.
1537
1539 // Pseudo probes don't constrain reordering of other instructions.
1541 return 0;
1542 unsigned Flags = 0;
1543 if (I->mayReadFromMemory())
1544 Flags |= SkipReadMem;
1545 // We can't arbitrarily move around allocas, e.g. moving allocas (especially
1546 // inalloca) across stacksave/stackrestore boundaries.
1547 if (I->mayHaveSideEffects() || isa<AllocaInst>(I))
1548 Flags |= SkipSideEffect;
1550 Flags |= SkipImplicitControlFlow;
1551 return Flags;
1552}
1553
1554// Returns true if it is safe to reorder an instruction across preceding
1555// instructions in a basic block.
1556static bool isSafeToHoistInstr(Instruction *I, unsigned Flags) {
1557 // Don't reorder a store over a load.
1558 if ((Flags & SkipReadMem) && I->mayWriteToMemory())
1559 return false;
1560
1561 // If we have seen an instruction with side effects, it's unsafe to reorder an
1562 // instruction which reads memory or itself has side effects.
1563 if ((Flags & SkipSideEffect) &&
1564 (I->mayReadFromMemory() || I->mayHaveSideEffects() || isa<AllocaInst>(I)))
1565 return false;
1566
1567 // Reordering across an instruction which does not necessarily transfer
1568 // control to the next instruction is speculation.
1570 return false;
1571
1572 // Hoisting of llvm.deoptimize is only legal together with the next return
1573 // instruction, which this pass is not always able to do.
1574 if (auto *CB = dyn_cast<CallBase>(I))
1575 if (CB->getIntrinsicID() == Intrinsic::experimental_deoptimize)
1576 return false;
1577
1578 // It's also unsafe/illegal to hoist an instruction above its instruction
1579 // operands
1580 BasicBlock *BB = I->getParent();
1581 for (Value *Op : I->operands()) {
1582 if (auto *J = dyn_cast<Instruction>(Op))
1583 if (J->getParent() == BB)
1584 return false;
1585 }
1586
1587 return true;
1588}
1589
1590static bool passingValueIsAlwaysUndefined(Value *V, Instruction *I, bool PtrValueMayBeModified = false);
1591
1592/// Helper function for hoistCommonCodeFromSuccessors. Return true if identical
1593/// instructions \p I1 and \p I2 can and should be hoisted.
1595 const TargetTransformInfo &TTI) {
1596 // If we're going to hoist a call, make sure that the two instructions
1597 // we're commoning/hoisting are both marked with musttail, or neither of
1598 // them is marked as such. Otherwise, we might end up in a situation where
1599 // we hoist from a block where the terminator is a `ret` to a block where
1600 // the terminator is a `br`, and `musttail` calls expect to be followed by
1601 // a return.
1602 auto *C1 = dyn_cast<CallInst>(I1);
1603 auto *C2 = dyn_cast<CallInst>(I2);
1604 if (C1 && C2)
1605 if (C1->isMustTailCall() != C2->isMustTailCall())
1606 return false;
1607
1608 if (!TTI.isProfitableToHoist(I1) || !TTI.isProfitableToHoist(I2))
1609 return false;
1610
1611 // If any of the two call sites has nomerge or convergent attribute, stop
1612 // hoisting.
1613 if (const auto *CB1 = dyn_cast<CallBase>(I1))
1614 if (CB1->cannotMerge() || CB1->isConvergent())
1615 return false;
1616 if (const auto *CB2 = dyn_cast<CallBase>(I2))
1617 if (CB2->cannotMerge() || CB2->isConvergent())
1618 return false;
1619
1620 return true;
1621}
1622
1623/// Hoists DbgVariableRecords from \p I1 and \p OtherInstrs that are identical
1624/// in lock-step to \p TI. This matches how dbg.* intrinsics are hoisting in
1625/// hoistCommonCodeFromSuccessors. e.g. The input:
1626/// I1 DVRs: { x, z },
1627/// OtherInsts: { I2 DVRs: { x, y, z } }
1628/// would result in hoisting only DbgVariableRecord x.
1630 Instruction *TI, Instruction *I1,
1631 SmallVectorImpl<Instruction *> &OtherInsts) {
1632 if (!I1->hasDbgRecords())
1633 return;
1634 using CurrentAndEndIt =
1635 std::pair<DbgRecord::self_iterator, DbgRecord::self_iterator>;
1636 // Vector of {Current, End} iterators.
1638 Itrs.reserve(OtherInsts.size() + 1);
1639 // Helper lambdas for lock-step checks:
1640 // Return true if this Current == End.
1641 auto atEnd = [](const CurrentAndEndIt &Pair) {
1642 return Pair.first == Pair.second;
1643 };
1644 // Return true if all Current are identical.
1645 auto allIdentical = [](const SmallVector<CurrentAndEndIt> &Itrs) {
1646 return all_of(make_first_range(ArrayRef(Itrs).drop_front()),
1648 return Itrs[0].first->isIdenticalToWhenDefined(*I);
1649 });
1650 };
1651
1652 // Collect the iterators.
1653 Itrs.push_back(
1654 {I1->getDbgRecordRange().begin(), I1->getDbgRecordRange().end()});
1655 for (Instruction *Other : OtherInsts) {
1656 if (!Other->hasDbgRecords())
1657 return;
1658 Itrs.push_back(
1659 {Other->getDbgRecordRange().begin(), Other->getDbgRecordRange().end()});
1660 }
1661
1662 // Iterate in lock-step until any of the DbgRecord lists are exausted. If
1663 // the lock-step DbgRecord are identical, hoist all of them to TI.
1664 // This replicates the dbg.* intrinsic behaviour in
1665 // hoistCommonCodeFromSuccessors.
1666 while (none_of(Itrs, atEnd)) {
1667 bool HoistDVRs = allIdentical(Itrs);
1668 for (CurrentAndEndIt &Pair : Itrs) {
1669 // Increment Current iterator now as we may be about to move the
1670 // DbgRecord.
1671 DbgRecord &DR = *Pair.first++;
1672 if (HoistDVRs) {
1673 DR.removeFromParent();
1674 TI->getParent()->insertDbgRecordBefore(&DR, TI->getIterator());
1675 }
1676 }
1677 }
1678}
1679
1681 const Instruction *I2) {
1682 if (I1->isIdenticalToWhenDefined(I2, /*IntersectAttrs=*/true))
1683 return true;
1684
1685 if (auto *Cmp1 = dyn_cast<CmpInst>(I1))
1686 if (auto *Cmp2 = dyn_cast<CmpInst>(I2))
1687 return Cmp1->getPredicate() == Cmp2->getSwappedPredicate() &&
1688 Cmp1->getOperand(0) == Cmp2->getOperand(1) &&
1689 Cmp1->getOperand(1) == Cmp2->getOperand(0);
1690
1691 if (I1->isCommutative() && I1->isSameOperationAs(I2)) {
1692 return I1->getOperand(0) == I2->getOperand(1) &&
1693 I1->getOperand(1) == I2->getOperand(0) &&
1694 equal(drop_begin(I1->operands(), 2), drop_begin(I2->operands(), 2));
1695 }
1696
1697 return false;
1698}
1699
1700/// If the target supports conditional faulting,
1701/// we look for the following pattern:
1702/// \code
1703/// BB:
1704/// ...
1705/// %cond = icmp ult %x, %y
1706/// br i1 %cond, label %TrueBB, label %FalseBB
1707/// FalseBB:
1708/// store i32 1, ptr %q, align 4
1709/// ...
1710/// TrueBB:
1711/// %maskedloadstore = load i32, ptr %b, align 4
1712/// store i32 %maskedloadstore, ptr %p, align 4
1713/// ...
1714/// \endcode
1715///
1716/// and transform it into:
1717///
1718/// \code
1719/// BB:
1720/// ...
1721/// %cond = icmp ult %x, %y
1722/// %maskedloadstore = cload i32, ptr %b, %cond
1723/// cstore i32 %maskedloadstore, ptr %p, %cond
1724/// cstore i32 1, ptr %q, ~%cond
1725/// br i1 %cond, label %TrueBB, label %FalseBB
1726/// FalseBB:
1727/// ...
1728/// TrueBB:
1729/// ...
1730/// \endcode
1731///
1732/// where cload/cstore are represented by llvm.masked.load/store intrinsics,
1733/// e.g.
1734///
1735/// \code
1736/// %vcond = bitcast i1 %cond to <1 x i1>
1737/// %v0 = call <1 x i32> @llvm.masked.load.v1i32.p0
1738/// (ptr %b, i32 4, <1 x i1> %vcond, <1 x i32> poison)
1739/// %maskedloadstore = bitcast <1 x i32> %v0 to i32
1740/// call void @llvm.masked.store.v1i32.p0
1741/// (<1 x i32> %v0, ptr %p, i32 4, <1 x i1> %vcond)
1742/// %cond.not = xor i1 %cond, true
1743/// %vcond.not = bitcast i1 %cond.not to <1 x i>
1744/// call void @llvm.masked.store.v1i32.p0
1745/// (<1 x i32> <i32 1>, ptr %q, i32 4, <1x i1> %vcond.not)
1746/// \endcode
1747///
1748/// So we need to turn hoisted load/store into cload/cstore.
1749///
1750/// \param BI The branch instruction.
1751/// \param SpeculatedConditionalLoadsStores The load/store instructions that
1752/// will be speculated.
1753/// \param Invert indicates if speculates FalseBB. Only used in triangle CFG.
1755 CondBrInst *BI,
1756 SmallVectorImpl<Instruction *> &SpeculatedConditionalLoadsStores,
1757 std::optional<bool> Invert, Instruction *Sel) {
1758 auto &Context = BI->getParent()->getContext();
1759 auto *VCondTy = FixedVectorType::get(Type::getInt1Ty(Context), 1);
1760 auto *Cond = BI->getCondition();
1761 // Construct the condition if needed.
1762 BasicBlock *BB = BI->getParent();
1763 Value *Mask = nullptr;
1764 Value *MaskFalse = nullptr;
1765 Value *MaskTrue = nullptr;
1766 if (Invert.has_value()) {
1767 IRBuilder<> Builder(Sel ? Sel : SpeculatedConditionalLoadsStores.back());
1768 Mask = Builder.CreateBitCast(
1769 *Invert ? Builder.CreateXor(Cond, ConstantInt::getTrue(Context)) : Cond,
1770 VCondTy);
1771 } else {
1772 IRBuilder<> Builder(BI);
1773 MaskFalse = Builder.CreateBitCast(
1774 Builder.CreateXor(Cond, ConstantInt::getTrue(Context)), VCondTy);
1775 MaskTrue = Builder.CreateBitCast(Cond, VCondTy);
1776 }
1777 auto PeekThroughBitcasts = [](Value *V) {
1778 while (auto *BitCast = dyn_cast<BitCastInst>(V))
1779 V = BitCast->getOperand(0);
1780 return V;
1781 };
1782 for (auto *I : SpeculatedConditionalLoadsStores) {
1783 IRBuilder<> Builder(Invert.has_value() ? I : BI);
1784 if (!Invert.has_value())
1785 Mask = I->getParent() == BI->getSuccessor(0) ? MaskTrue : MaskFalse;
1786 // We currently assume conditional faulting load/store is supported for
1787 // scalar types only when creating new instructions. This can be easily
1788 // extended for vector types in the future.
1789 assert(!getLoadStoreType(I)->isVectorTy() && "not implemented");
1790 auto *Op0 = I->getOperand(0);
1791 CallInst *MaskedLoadStore = nullptr;
1792 if (auto *LI = dyn_cast<LoadInst>(I)) {
1793 // Handle Load.
1794 auto *Ty = I->getType();
1795 PHINode *PN = nullptr;
1796 Value *PassThru = nullptr;
1797 if (Invert.has_value())
1798 for (User *U : I->users()) {
1799 if ((PN = dyn_cast<PHINode>(U))) {
1800 PassThru = Builder.CreateBitCast(
1801 PeekThroughBitcasts(PN->getIncomingValueForBlock(BB)),
1802 FixedVectorType::get(Ty, 1));
1803 } else if (auto *Ins = cast<Instruction>(U);
1804 Sel && Ins->getParent() == BB) {
1805 // This happens when store or/and a speculative instruction between
1806 // load and store were hoisted to the BB. Make sure the masked load
1807 // inserted before its use.
1808 // We assume there's one of such use.
1809 Builder.SetInsertPoint(Ins);
1810 }
1811 }
1812 MaskedLoadStore = Builder.CreateMaskedLoad(
1813 FixedVectorType::get(Ty, 1), Op0, LI->getAlign(), Mask, PassThru);
1814 Value *NewLoadStore = Builder.CreateBitCast(MaskedLoadStore, Ty);
1815 if (PN)
1816 PN->setIncomingValue(PN->getBasicBlockIndex(BB), NewLoadStore);
1817 I->replaceAllUsesWith(NewLoadStore);
1818 } else {
1819 // Handle Store.
1820 auto *StoredVal = Builder.CreateBitCast(
1821 PeekThroughBitcasts(Op0), FixedVectorType::get(Op0->getType(), 1));
1822 MaskedLoadStore = Builder.CreateMaskedStore(
1823 StoredVal, I->getOperand(1), cast<StoreInst>(I)->getAlign(), Mask);
1824 }
1825 // For non-debug metadata, only !annotation, !range, !nonnull and !align are
1826 // kept when hoisting (see Instruction::dropUBImplyingAttrsAndMetadata).
1827 //
1828 // !nonnull, !align : Not support pointer type, no need to keep.
1829 // !range: Load type is changed from scalar to vector, but the metadata on
1830 // vector specifies a per-element range, so the semantics stay the
1831 // same. Keep it.
1832 // !annotation: Not impact semantics. Keep it.
1833 if (const MDNode *Ranges = I->getMetadata(LLVMContext::MD_range))
1834 MaskedLoadStore->addRangeRetAttr(getConstantRangeFromMetadata(*Ranges));
1835 I->dropUBImplyingAttrsAndUnknownMetadata({LLVMContext::MD_annotation});
1836 // FIXME: DIAssignID is not supported for masked store yet.
1837 // (Verifier::visitDIAssignIDMetadata)
1839 I->eraseMetadataIf([](unsigned MDKind, MDNode *Node) {
1840 return Node->getMetadataID() == Metadata::DIAssignIDKind;
1841 });
1842 MaskedLoadStore->copyMetadata(*I);
1843 I->eraseFromParent();
1844 }
1845}
1846
1848 const TargetTransformInfo &TTI) {
1849 // Not handle volatile or atomic.
1850 bool IsStore = false;
1851 if (auto *L = dyn_cast<LoadInst>(I)) {
1852 if (!L->isSimple() || !HoistLoadsWithCondFaulting)
1853 return false;
1854 } else if (auto *S = dyn_cast<StoreInst>(I)) {
1855 if (!S->isSimple() || !HoistStoresWithCondFaulting)
1856 return false;
1857 IsStore = true;
1858 } else
1859 return false;
1860
1861 // llvm.masked.load/store use i32 for alignment while load/store use i64.
1862 // That's why we have the alignment limitation.
1863 // FIXME: Update the prototype of the intrinsics?
1864 return TTI.hasConditionalLoadStoreForType(getLoadStoreType(I), IsStore) &&
1866}
1867
1868/// Hoist any common code in the successor blocks up into the block. This
1869/// function guarantees that BB dominates all successors. If AllInstsEqOnly is
1870/// given, only perform hoisting in case all successors blocks contain matching
1871/// instructions only. In that case, all instructions can be hoisted and the
1872/// original branch will be replaced and selects for PHIs are added.
1873bool SimplifyCFGOpt::hoistCommonCodeFromSuccessors(Instruction *TI,
1874 bool AllInstsEqOnly) {
1875 // This does very trivial matching, with limited scanning, to find identical
1876 // instructions in the two blocks. In particular, we don't want to get into
1877 // O(N1*N2*...) situations here where Ni are the sizes of these successors. As
1878 // such, we currently just scan for obviously identical instructions in an
1879 // identical order, possibly separated by the same number of non-identical
1880 // instructions.
1881 BasicBlock *BB = TI->getParent();
1882 unsigned int SuccSize = succ_size(BB);
1883 if (SuccSize < 2)
1884 return false;
1885
1886 // If either of the blocks has it's address taken, then we can't do this fold,
1887 // because the code we'd hoist would no longer run when we jump into the block
1888 // by it's address.
1889 SmallSetVector<BasicBlock *, 4> UniqueSuccessors(from_range, successors(BB));
1890 for (auto *Succ : UniqueSuccessors) {
1891 if (Succ->hasAddressTaken())
1892 return false;
1893 // Use getUniquePredecessor instead of getSinglePredecessor to support
1894 // multi-cases successors in switch.
1895 if (Succ->getUniquePredecessor())
1896 continue;
1897 // If Succ has >1 predecessors, continue to check if the Succ contains only
1898 // one `unreachable` inst. Since executing `unreachable` inst is an UB, we
1899 // can relax the condition based on the assumptiom that the program would
1900 // never enter Succ and trigger such an UB.
1901 if (isa<UnreachableInst>(*Succ->begin()))
1902 continue;
1903 return false;
1904 }
1905 // The second of pair is a SkipFlags bitmask.
1906 using SuccIterPair = std::pair<BasicBlock::iterator, unsigned>;
1907 SmallVector<SuccIterPair, 8> SuccIterPairs;
1908 for (auto *Succ : UniqueSuccessors) {
1909 BasicBlock::iterator SuccItr = Succ->begin();
1910 if (isa<PHINode>(*SuccItr))
1911 return false;
1912 SuccIterPairs.push_back(SuccIterPair(SuccItr, 0));
1913 }
1914
1915 if (AllInstsEqOnly) {
1916 // Check if all instructions in the successor blocks match. This allows
1917 // hoisting all instructions and removing the blocks we are hoisting from,
1918 // so does not add any new instructions.
1919
1920 // Check if sizes and terminators of all successors match.
1921 unsigned Size0 = UniqueSuccessors[0]->size();
1922 Instruction *Term0 = UniqueSuccessors[0]->getTerminator();
1923 bool AllSame =
1924 all_of(drop_begin(UniqueSuccessors), [Term0, Size0](BasicBlock *Succ) {
1925 return Succ->getTerminator()->isIdenticalTo(Term0) &&
1926 Succ->size() == Size0;
1927 });
1928 if (!AllSame)
1929 return false;
1930 LockstepReverseIterator<true> LRI(UniqueSuccessors.getArrayRef());
1931 while (LRI.isValid()) {
1932 Instruction *I0 = (*LRI)[0];
1933 if (any_of(*LRI, [I0](Instruction *I) {
1934 return !areIdenticalUpToCommutativity(I0, I);
1935 })) {
1936 return false;
1937 }
1938 --LRI;
1939 }
1940 // Now we know that all instructions in all successors can be hoisted. Let
1941 // the loop below handle the hoisting.
1942 }
1943
1944 // Count how many instructions were not hoisted so far. There's a limit on how
1945 // many instructions we skip, serving as a compilation time control as well as
1946 // preventing excessive increase of life ranges.
1947 unsigned NumSkipped = 0;
1948 // If we find an unreachable instruction at the beginning of a basic block, we
1949 // can still hoist instructions from the rest of the basic blocks.
1950 if (SuccIterPairs.size() > 2) {
1951 erase_if(SuccIterPairs,
1952 [](const auto &Pair) { return isa<UnreachableInst>(Pair.first); });
1953 if (SuccIterPairs.size() < 2)
1954 return false;
1955 }
1956
1957 bool Changed = false;
1958
1959 for (;;) {
1960 auto *SuccIterPairBegin = SuccIterPairs.begin();
1961 auto &BB1ItrPair = *SuccIterPairBegin++;
1962 auto OtherSuccIterPairRange =
1963 iterator_range(SuccIterPairBegin, SuccIterPairs.end());
1964 auto OtherSuccIterRange = make_first_range(OtherSuccIterPairRange);
1965
1966 Instruction *I1 = &*BB1ItrPair.first;
1967
1968 bool AllInstsAreIdentical = true;
1969 bool HasTerminator = I1->isTerminator();
1970 for (auto &SuccIter : OtherSuccIterRange) {
1971 Instruction *I2 = &*SuccIter;
1972 HasTerminator |= I2->isTerminator();
1973 if (AllInstsAreIdentical && (!areIdenticalUpToCommutativity(I1, I2) ||
1974 MMRAMetadata(*I1) != MMRAMetadata(*I2)))
1975 AllInstsAreIdentical = false;
1976 }
1977
1978 SmallVector<Instruction *, 8> OtherInsts;
1979 for (auto &SuccIter : OtherSuccIterRange)
1980 OtherInsts.push_back(&*SuccIter);
1981
1982 // If we are hoisting the terminator instruction, don't move one (making a
1983 // broken BB), instead clone it, and remove BI.
1984 if (HasTerminator) {
1985 // Even if BB, which contains only one unreachable instruction, is ignored
1986 // at the beginning of the loop, we can hoist the terminator instruction.
1987 // If any instructions remain in the block, we cannot hoist terminators.
1988 if (NumSkipped || !AllInstsAreIdentical) {
1989 hoistLockstepIdenticalDbgVariableRecords(TI, I1, OtherInsts);
1990 return Changed;
1991 }
1992
1993 return hoistSuccIdenticalTerminatorToSwitchOrIf(
1994 TI, I1, OtherInsts, UniqueSuccessors.getArrayRef()) ||
1995 Changed;
1996 }
1997
1998 if (AllInstsAreIdentical) {
1999 unsigned SkipFlagsBB1 = BB1ItrPair.second;
2000 AllInstsAreIdentical =
2001 isSafeToHoistInstr(I1, SkipFlagsBB1) &&
2002 all_of(OtherSuccIterPairRange, [=](const auto &Pair) {
2003 Instruction *I2 = &*Pair.first;
2004 unsigned SkipFlagsBB2 = Pair.second;
2005 // Even if the instructions are identical, it may not
2006 // be safe to hoist them if we have skipped over
2007 // instructions with side effects or their operands
2008 // weren't hoisted.
2009 return isSafeToHoistInstr(I2, SkipFlagsBB2) &&
2011 });
2012 }
2013
2014 // A musttail call must be immediately followed by a ret, so hoisting is
2015 // only legal if its ret is hoisted with it on the next iteration. That is,
2016 // no instruction has been skipped (the entire successor can be hoisted into
2017 // the predecessor) and the call is directly followed by a ret.
2018 if (auto *CI = dyn_cast<CallInst>(I1);
2019 AllInstsAreIdentical && CI && CI->isMustTailCall()) {
2020 AllInstsAreIdentical =
2021 NumSkipped == 0 && all_of(SuccIterPairs, [](const SuccIterPair &P) {
2022 return isa<ReturnInst>(*std::next(P.first));
2023 });
2024 }
2025
2026 if (AllInstsAreIdentical) {
2027 BB1ItrPair.first++;
2028 // For a normal instruction, we just move one to right before the
2029 // branch, then replace all uses of the other with the first. Finally,
2030 // we remove the now redundant second instruction.
2031 hoistLockstepIdenticalDbgVariableRecords(TI, I1, OtherInsts);
2032 // We've just hoisted DbgVariableRecords; move I1 after them (before TI)
2033 // and leave any that were not hoisted behind (by calling moveBefore
2034 // rather than moveBeforePreserving).
2035 I1->moveBefore(TI->getIterator());
2036 for (auto &SuccIter : OtherSuccIterRange) {
2037 Instruction *I2 = &*SuccIter++;
2038 assert(I2 != I1);
2039 if (!I2->use_empty())
2040 I2->replaceAllUsesWith(I1);
2041 I1->andIRFlags(I2);
2042 if (auto *CB = dyn_cast<CallBase>(I1)) {
2043 bool Success = CB->tryIntersectAttributes(cast<CallBase>(I2));
2044 assert(Success && "We should not be trying to hoist callbases "
2045 "with non-intersectable attributes");
2046 // For NDEBUG Compile.
2047 (void)Success;
2048 }
2049
2050 combineMetadataForCSE(I1, I2, true);
2051 // I1 and I2 are being combined into a single instruction. Its debug
2052 // location is the merged locations of the original instructions.
2053 I1->applyMergedLocation(I1->getDebugLoc(), I2->getDebugLoc());
2054 I2->eraseFromParent();
2055 }
2056 // I1 now executes before the instructions we skipped.
2057 unsigned SkippedFlags = 0;
2058 for (const SuccIterPair &P : SuccIterPairs)
2059 SkippedFlags |= P.second;
2060 if (SkippedFlags & SkipImplicitControlFlow) {
2061 // One of them may throw or not return, so I1 is speculated.
2062 I1->dropUBImplyingAttrsAndMetadata();
2063 }
2064 if (!Changed)
2065 NumHoistCommonCode += SuccIterPairs.size();
2066 Changed = true;
2067 NumHoistCommonInstrs += SuccIterPairs.size();
2068 } else {
2069 if (NumSkipped >= HoistCommonSkipLimit) {
2070 hoistLockstepIdenticalDbgVariableRecords(TI, I1, OtherInsts);
2071 return Changed;
2072 }
2073 // We are about to skip over a pair of non-identical instructions. Record
2074 // if any have characteristics that would prevent reordering instructions
2075 // across them.
2076 for (auto &SuccIterPair : SuccIterPairs) {
2077 Instruction *I = &*SuccIterPair.first++;
2078 SuccIterPair.second |= skippedInstrFlags(I);
2079 }
2080 ++NumSkipped;
2081 }
2082 }
2083}
2084
2085bool SimplifyCFGOpt::hoistSuccIdenticalTerminatorToSwitchOrIf(
2086 Instruction *TI, Instruction *I1,
2087 SmallVectorImpl<Instruction *> &OtherSuccTIs,
2088 ArrayRef<BasicBlock *> UniqueSuccessors) {
2089
2090 auto *BI = dyn_cast<CondBrInst>(TI);
2091
2092 bool Changed = false;
2093 BasicBlock *TIParent = TI->getParent();
2094 BasicBlock *BB1 = I1->getParent();
2095
2096 // Use only for an if statement.
2097 auto *I2 = *OtherSuccTIs.begin();
2098 auto *BB2 = I2->getParent();
2099 if (BI) {
2100 assert(OtherSuccTIs.size() == 1);
2101 assert(BI->getSuccessor(0) == I1->getParent());
2102 assert(BI->getSuccessor(1) == I2->getParent());
2103 }
2104
2105 // In the case of an if statement, we try to hoist an invoke.
2106 // FIXME: Can we define a safety predicate for CallBr?
2107 // FIXME: Test case llvm/test/Transforms/SimplifyCFG/2009-06-15-InvokeCrash.ll
2108 // removed in 4c923b3b3fd0ac1edebf0603265ca3ba51724937 commit?
2109 if (isa<InvokeInst>(I1) && (!BI || !isSafeToHoistInvoke(BB1, BB2, I1, I2)))
2110 return false;
2111
2112 // TODO: callbr hoisting currently disabled pending further study.
2113 if (isa<CallBrInst>(I1))
2114 return false;
2115
2116 for (BasicBlock *Succ : successors(BB1)) {
2117 for (PHINode &PN : Succ->phis()) {
2118 Value *BB1V = PN.getIncomingValueForBlock(BB1);
2119 for (Instruction *OtherSuccTI : OtherSuccTIs) {
2120 Value *BB2V = PN.getIncomingValueForBlock(OtherSuccTI->getParent());
2121 if (BB1V == BB2V)
2122 continue;
2123
2124 // In the case of an if statement, check for
2125 // passingValueIsAlwaysUndefined here because we would rather eliminate
2126 // undefined control flow then converting it to a select.
2127 if (!BI || passingValueIsAlwaysUndefined(BB1V, &PN) ||
2129 return false;
2130 }
2131 }
2132 }
2133
2134 // Hoist DbgVariableRecords attached to the terminator to match dbg.*
2135 // intrinsic hoisting behaviour in hoistCommonCodeFromSuccessors.
2136 hoistLockstepIdenticalDbgVariableRecords(TI, I1, OtherSuccTIs);
2137 // Clone the terminator and hoist it into the pred, without any debug info.
2138 Instruction *NT = I1->clone();
2139 NT->insertInto(TIParent, TI->getIterator());
2140 if (!NT->getType()->isVoidTy()) {
2141 I1->replaceAllUsesWith(NT);
2142 for (Instruction *OtherSuccTI : OtherSuccTIs)
2143 OtherSuccTI->replaceAllUsesWith(NT);
2144 NT->takeName(I1);
2145 }
2146 Changed = true;
2147 NumHoistCommonInstrs += OtherSuccTIs.size() + 1;
2148
2149 // Ensure terminator gets a debug location, even an unknown one, in case
2150 // it involves inlinable calls.
2152 Locs.push_back(I1->getDebugLoc());
2153 for (auto *OtherSuccTI : OtherSuccTIs)
2154 Locs.push_back(OtherSuccTI->getDebugLoc());
2155 NT->setDebugLoc(DebugLoc::getMergedLocations(Locs));
2156
2157 // PHIs created below will adopt NT's merged DebugLoc.
2158 IRBuilder<NoFolder> Builder(NT);
2159
2160 // In the case of an if statement, hoisting one of the terminators from our
2161 // successor is a great thing. Unfortunately, the successors of the if/else
2162 // blocks may have PHI nodes in them. If they do, all PHI entries for BB1/BB2
2163 // must agree for all PHI nodes, so we insert select instruction to compute
2164 // the final result.
2165 if (BI) {
2166 std::map<std::pair<Value *, Value *>, SelectInst *> InsertedSelects;
2167 for (BasicBlock *Succ : successors(BB1)) {
2168 for (PHINode &PN : Succ->phis()) {
2169 Value *BB1V = PN.getIncomingValueForBlock(BB1);
2170 Value *BB2V = PN.getIncomingValueForBlock(BB2);
2171 if (BB1V == BB2V)
2172 continue;
2173
2174 // These values do not agree. Insert a select instruction before NT
2175 // that determines the right value.
2176 SelectInst *&SI = InsertedSelects[std::make_pair(BB1V, BB2V)];
2177 if (!SI) {
2178 // Propagate fast-math-flags from phi node to its replacement select.
2180 BI->getCondition(), BB1V, BB2V,
2181 isa<FPMathOperator>(PN) ? &PN : nullptr,
2182 BB1V->getName() + "." + BB2V->getName(), BI));
2183 }
2184
2185 // Make the PHI node use the select for all incoming values for BB1/BB2
2186 for (unsigned i = 0, e = PN.getNumIncomingValues(); i != e; ++i)
2187 if (PN.getIncomingBlock(i) == BB1 || PN.getIncomingBlock(i) == BB2)
2188 PN.setIncomingValue(i, SI);
2189 }
2190 }
2191 }
2192
2194
2195 // Update any PHI nodes in our new successors.
2196 SmallPtrSet<BasicBlock *, 8> VisitedSuccs;
2197 for (BasicBlock *Succ : successors(BB1)) {
2198 addPredecessorToBlock(Succ, TIParent, BB1);
2199
2200 if (DTU && VisitedSuccs.insert(Succ).second)
2201 Updates.push_back({DominatorTree::Insert, TIParent, Succ});
2202 }
2203
2204 if (DTU) {
2205 // TI might be a switch with multi-cases destination, so we need to care for
2206 // the duplication of successors.
2207 for (BasicBlock *Succ : UniqueSuccessors)
2208 Updates.push_back({DominatorTree::Delete, TIParent, Succ});
2209 }
2210
2212 if (DTU)
2213 DTU->applyUpdates(Updates);
2214 return Changed;
2215}
2216
2217// TODO: Refine this. This should avoid cases like turning constant memcpy sizes
2218// into variables.
2220 int OpIdx) {
2221 // Divide/Remainder by constant is typically much cheaper than by variable.
2222 if (I->isIntDivRem())
2223 return OpIdx != 1;
2224 return !isa<IntrinsicInst>(I);
2225}
2226
2227// All instructions in Insts belong to different blocks that all unconditionally
2228// branch to a common successor. Analyze each instruction and return true if it
2229// would be possible to sink them into their successor, creating one common
2230// instruction instead. For every value that would be required to be provided by
2231// PHI node (because an operand varies in each input block), add to PHIOperands.
2234 DenseMap<const Use *, SmallVector<Value *, 4>> &PHIOperands) {
2235 // Prune out obviously bad instructions to move. Each instruction must have
2236 // the same number of uses, and we check later that the uses are consistent.
2237 std::optional<unsigned> NumUses;
2238 for (auto *I : Insts) {
2239 // These instructions may change or break semantics if moved.
2240 if (isa<PHINode>(I) || I->isEHPad() || isa<AllocaInst>(I) ||
2241 I->getType()->isTokenTy())
2242 return false;
2243
2244 // Do not try to sink an instruction in an infinite loop - it can cause
2245 // this algorithm to infinite loop.
2246 if (I->getParent()->getSingleSuccessor() == I->getParent())
2247 return false;
2248
2249 // Conservatively return false if I is an inline-asm instruction. Sinking
2250 // and merging inline-asm instructions can potentially create arguments
2251 // that cannot satisfy the inline-asm constraints.
2252 // If the instruction has nomerge or convergent attribute, return false.
2253 if (const auto *C = dyn_cast<CallBase>(I))
2254 if (C->isInlineAsm() || C->cannotMerge() || C->isConvergent())
2255 return false;
2256
2257 if (!NumUses)
2258 NumUses = I->getNumUses();
2259 else if (NumUses != I->getNumUses())
2260 return false;
2261 }
2262
2263 const Instruction *I0 = Insts.front();
2264 const auto I0MMRA = MMRAMetadata(*I0);
2265 for (auto *I : Insts) {
2266 if (!I->isSameOperationAs(I0, Instruction::CompareUsingIntersectedAttrs))
2267 return false;
2268
2269 // Treat MMRAs conservatively. This pass can be quite aggressive and
2270 // could drop a lot of MMRAs otherwise.
2271 if (MMRAMetadata(*I) != I0MMRA)
2272 return false;
2273 }
2274
2275 // Uses must be consistent: If I0 is used in a phi node in the sink target,
2276 // then the other phi operands must match the instructions from Insts. This
2277 // also has to hold true for any phi nodes that would be created as a result
2278 // of sinking. Both of these cases are represented by PhiOperands.
2279 for (const Use &U : I0->uses()) {
2280 auto It = PHIOperands.find(&U);
2281 if (It == PHIOperands.end())
2282 // There may be uses in other blocks when sinking into a loop header.
2283 return false;
2284 if (!equal(Insts, It->second))
2285 return false;
2286 }
2287
2288 // For calls to be sinkable, they must all be indirect, or have same callee.
2289 // I.e. if we have two direct calls to different callees, we don't want to
2290 // turn that into an indirect call. Likewise, if we have an indirect call,
2291 // and a direct call, we don't actually want to have a single indirect call.
2292 if (isa<CallBase>(I0)) {
2293 auto IsIndirectCall = [](const Instruction *I) {
2294 return cast<CallBase>(I)->isIndirectCall();
2295 };
2296 bool HaveIndirectCalls = any_of(Insts, IsIndirectCall);
2297 bool AllCallsAreIndirect = all_of(Insts, IsIndirectCall);
2298 if (HaveIndirectCalls) {
2299 if (!AllCallsAreIndirect)
2300 return false;
2301 } else {
2302 // All callees must be identical.
2303 Value *Callee = nullptr;
2304 for (const Instruction *I : Insts) {
2305 Value *CurrCallee = cast<CallBase>(I)->getCalledOperand();
2306 if (!Callee)
2307 Callee = CurrCallee;
2308 else if (Callee != CurrCallee)
2309 return false;
2310 }
2311 }
2312 }
2313
2314 for (unsigned OI = 0, OE = I0->getNumOperands(); OI != OE; ++OI) {
2315 Value *Op = I0->getOperand(OI);
2316 auto SameAsI0 = [&I0, OI](const Instruction *I) {
2317 assert(I->getNumOperands() == I0->getNumOperands());
2318 return I->getOperand(OI) == I0->getOperand(OI);
2319 };
2320 if (!all_of(Insts, SameAsI0)) {
2321 auto CanReplaceOperand = [OI](const Instruction *I) {
2322 return canReplaceOperandWithVariable(I, OI);
2323 };
2325 !all_of(Insts, CanReplaceOperand))
2326 // We can't create a PHI from this operand.
2327 return false;
2328 auto &Ops = PHIOperands[&I0->getOperandUse(OI)];
2329 for (auto *I : Insts)
2330 Ops.push_back(I->getOperand(OI));
2331 }
2332 }
2333 return true;
2334}
2335
2336// Assuming canSinkInstructions(Blocks) has returned true, sink the last
2337// instruction of every block in Blocks to their common successor, commoning
2338// into one instruction.
2340 auto *BBEnd = Blocks[0]->getTerminator()->getSuccessor(0);
2341
2342 // canSinkInstructions returning true guarantees that every block has at
2343 // least one non-terminator instruction.
2345 for (auto *BB : Blocks) {
2346 Instruction *I = BB->getTerminator();
2347 I = I->getPrevNode();
2348 Insts.push_back(I);
2349 }
2350
2351 // We don't need to do any more checking here; canSinkInstructions should
2352 // have done it all for us.
2353 SmallVector<Value*, 4> NewOperands;
2354 Instruction *I0 = Insts.front();
2355 for (unsigned O = 0, E = I0->getNumOperands(); O != E; ++O) {
2356 // This check is different to that in canSinkInstructions. There, we
2357 // cared about the global view once simplifycfg (and instcombine) have
2358 // completed - it takes into account PHIs that become trivially
2359 // simplifiable. However here we need a more local view; if an operand
2360 // differs we create a PHI and rely on instcombine to clean up the very
2361 // small mess we may make.
2362 bool NeedPHI = any_of(Insts, [&I0, O](const Instruction *I) {
2363 return I->getOperand(O) != I0->getOperand(O);
2364 });
2365 if (!NeedPHI) {
2366 NewOperands.push_back(I0->getOperand(O));
2367 continue;
2368 }
2369
2370 // Create a new PHI in the successor block and populate it.
2371 auto *Op = I0->getOperand(O);
2372 assert(!Op->getType()->isTokenTy() && "Can't PHI tokens!");
2373 auto *PN =
2374 PHINode::Create(Op->getType(), Insts.size(), Op->getName() + ".sink");
2375 PN->insertBefore(BBEnd->begin());
2376 for (auto *I : Insts)
2377 PN->addIncoming(I->getOperand(O), I->getParent());
2378 NewOperands.push_back(PN);
2379 }
2380
2381 // Arbitrarily use I0 as the new "common" instruction; remap its operands
2382 // and move it to the start of the successor block.
2383 for (unsigned O = 0, E = I0->getNumOperands(); O != E; ++O)
2384 I0->getOperandUse(O).set(NewOperands[O]);
2385
2386 I0->moveBefore(*BBEnd, BBEnd->getFirstInsertionPt());
2387
2388 // Update metadata and IR flags, and merge debug locations.
2389 for (auto *I : Insts)
2390 if (I != I0) {
2391 // The debug location for the "common" instruction is the merged locations
2392 // of all the commoned instructions. We start with the original location
2393 // of the "common" instruction and iteratively merge each location in the
2394 // loop below.
2395 // This is an N-way merge, which will be inefficient if I0 is a CallInst.
2396 // However, as N-way merge for CallInst is rare, so we use simplified API
2397 // instead of using complex API for N-way merge.
2398 I0->applyMergedLocation(I0->getDebugLoc(), I->getDebugLoc());
2399 combineMetadataForCSE(I0, I, true);
2400 I0->andIRFlags(I);
2401 if (auto *CB = dyn_cast<CallBase>(I0)) {
2402 bool Success = CB->tryIntersectAttributes(cast<CallBase>(I));
2403 assert(Success && "We should not be trying to sink callbases "
2404 "with non-intersectable attributes");
2405 // For NDEBUG Compile.
2406 (void)Success;
2407 }
2408 }
2409
2410 for (User *U : make_early_inc_range(I0->users())) {
2411 // canSinkLastInstruction checked that all instructions are only used by
2412 // phi nodes in a way that allows replacing the phi node with the common
2413 // instruction.
2414 auto *PN = cast<PHINode>(U);
2415 PN->replaceAllUsesWith(I0);
2416 PN->eraseFromParent();
2417 }
2418
2419 // Finally nuke all instructions apart from the common instruction.
2420 for (auto *I : Insts) {
2421 if (I == I0)
2422 continue;
2423 // The remaining uses are debug users, replace those with the common inst.
2424 // In most (all?) cases this just introduces a use-before-def.
2425 assert(I->user_empty() && "Inst unexpectedly still has non-dbg users");
2426 I->replaceAllUsesWith(I0);
2427 I->eraseFromParent();
2428 }
2429}
2430
2431/// Check whether BB's predecessors end with unconditional branches. If it is
2432/// true, sink any common code from the predecessors to BB.
2434 DomTreeUpdater *DTU) {
2435 // We support two situations:
2436 // (1) all incoming arcs are unconditional
2437 // (2) there are non-unconditional incoming arcs
2438 //
2439 // (2) is very common in switch defaults and
2440 // else-if patterns;
2441 //
2442 // if (a) f(1);
2443 // else if (b) f(2);
2444 //
2445 // produces:
2446 //
2447 // [if]
2448 // / \
2449 // [f(1)] [if]
2450 // | | \
2451 // | | |
2452 // | [f(2)]|
2453 // \ | /
2454 // [ end ]
2455 //
2456 // [end] has two unconditional predecessor arcs and one conditional. The
2457 // conditional refers to the implicit empty 'else' arc. This conditional
2458 // arc can also be caused by an empty default block in a switch.
2459 //
2460 // In this case, we attempt to sink code from all *unconditional* arcs.
2461 // If we can sink instructions from these arcs (determined during the scan
2462 // phase below) we insert a common successor for all unconditional arcs and
2463 // connect that to [end], to enable sinking:
2464 //
2465 // [if]
2466 // / \
2467 // [x(1)] [if]
2468 // | | \
2469 // | | \
2470 // | [x(2)] |
2471 // \ / |
2472 // [sink.split] |
2473 // \ /
2474 // [ end ]
2475 //
2476 SmallVector<BasicBlock*,4> UnconditionalPreds;
2477 bool HaveNonUnconditionalPredecessors = false;
2478 for (auto *PredBB : predecessors(BB)) {
2479 auto *PredBr = dyn_cast<UncondBrInst>(PredBB->getTerminator());
2480 if (PredBr)
2481 UnconditionalPreds.push_back(PredBB);
2482 else
2483 HaveNonUnconditionalPredecessors = true;
2484 }
2485 if (UnconditionalPreds.size() < 2)
2486 return false;
2487
2488 // We take a two-step approach to tail sinking. First we scan from the end of
2489 // each block upwards in lockstep. If the n'th instruction from the end of each
2490 // block can be sunk, those instructions are added to ValuesToSink and we
2491 // carry on. If we can sink an instruction but need to PHI-merge some operands
2492 // (because they're not identical in each instruction) we add these to
2493 // PHIOperands.
2494 // We prepopulate PHIOperands with the phis that already exist in BB.
2496 for (PHINode &PN : BB->phis()) {
2498 for (const Use &U : PN.incoming_values())
2499 IncomingVals.insert({PN.getIncomingBlock(U), &U});
2500 auto &Ops = PHIOperands[IncomingVals[UnconditionalPreds[0]]];
2501 for (BasicBlock *Pred : UnconditionalPreds)
2502 Ops.push_back(*IncomingVals[Pred]);
2503 }
2504
2505 int ScanIdx = 0;
2506 SmallPtrSet<Value*,4> InstructionsToSink;
2507 LockstepReverseIterator<true> LRI(UnconditionalPreds);
2508 while (LRI.isValid() &&
2509 canSinkInstructions(*LRI, PHIOperands)) {
2510 LLVM_DEBUG(dbgs() << "SINK: instruction can be sunk: " << *(*LRI)[0]
2511 << "\n");
2512 InstructionsToSink.insert_range(*LRI);
2513 ++ScanIdx;
2514 --LRI;
2515 }
2516
2517 // If no instructions can be sunk, early-return.
2518 if (ScanIdx == 0)
2519 return false;
2520
2521 bool followedByDeoptOrUnreachable = IsBlockFollowedByDeoptOrUnreachable(BB);
2522
2523 if (!followedByDeoptOrUnreachable) {
2524 // Check whether this is the pointer operand of a load/store.
2525 auto IsMemOperand = [](Use &U) {
2526 auto *I = cast<Instruction>(U.getUser());
2527 if (isa<LoadInst>(I))
2528 return U.getOperandNo() == LoadInst::getPointerOperandIndex();
2529 if (isa<StoreInst>(I))
2530 return U.getOperandNo() == StoreInst::getPointerOperandIndex();
2531 return false;
2532 };
2533
2534 // Okay, we *could* sink last ScanIdx instructions. But how many can we
2535 // actually sink before encountering instruction that is unprofitable to
2536 // sink?
2537 auto ProfitableToSinkInstruction = [&](LockstepReverseIterator<true> &LRI) {
2538 unsigned NumPHIInsts = 0;
2539 for (Use &U : (*LRI)[0]->operands()) {
2540 auto It = PHIOperands.find(&U);
2541 if (It != PHIOperands.end() && !all_of(It->second, [&](Value *V) {
2542 return InstructionsToSink.contains(V);
2543 })) {
2544 ++NumPHIInsts;
2545 // Do not separate a load/store from the gep producing the address.
2546 // The gep can likely be folded into the load/store as an addressing
2547 // mode. Additionally, a load of a gep is easier to analyze than a
2548 // load of a phi.
2549 if (IsMemOperand(U) &&
2550 any_of(It->second, [](Value *V) { return isa<GEPOperator>(V); }))
2551 return false;
2552 // FIXME: this check is overly optimistic. We may end up not sinking
2553 // said instruction, due to the very same profitability check.
2554 // See @creating_too_many_phis in sink-common-code.ll.
2555 }
2556 }
2557 LLVM_DEBUG(dbgs() << "SINK: #phi insts: " << NumPHIInsts << "\n");
2558 return NumPHIInsts <= 1;
2559 };
2560
2561 // We've determined that we are going to sink last ScanIdx instructions,
2562 // and recorded them in InstructionsToSink. Now, some instructions may be
2563 // unprofitable to sink. But that determination depends on the instructions
2564 // that we are going to sink.
2565
2566 // First, forward scan: find the first instruction unprofitable to sink,
2567 // recording all the ones that are profitable to sink.
2568 // FIXME: would it be better, after we detect that not all are profitable.
2569 // to either record the profitable ones, or erase the unprofitable ones?
2570 // Maybe we need to choose (at runtime) the one that will touch least
2571 // instrs?
2572 LRI.reset();
2573 int Idx = 0;
2574 SmallPtrSet<Value *, 4> InstructionsProfitableToSink;
2575 while (Idx < ScanIdx) {
2576 if (!ProfitableToSinkInstruction(LRI)) {
2577 // Too many PHIs would be created.
2578 LLVM_DEBUG(
2579 dbgs() << "SINK: stopping here, too many PHIs would be created!\n");
2580 break;
2581 }
2582 InstructionsProfitableToSink.insert_range(*LRI);
2583 --LRI;
2584 ++Idx;
2585 }
2586
2587 // If no instructions can be sunk, early-return.
2588 if (Idx == 0)
2589 return false;
2590
2591 // Did we determine that (only) some instructions are unprofitable to sink?
2592 if (Idx < ScanIdx) {
2593 // Okay, some instructions are unprofitable.
2594 ScanIdx = Idx;
2595 InstructionsToSink = InstructionsProfitableToSink;
2596
2597 // But, that may make other instructions unprofitable, too.
2598 // So, do a backward scan, do any earlier instructions become
2599 // unprofitable?
2600 assert(
2601 !ProfitableToSinkInstruction(LRI) &&
2602 "We already know that the last instruction is unprofitable to sink");
2603 ++LRI;
2604 --Idx;
2605 while (Idx >= 0) {
2606 // If we detect that an instruction becomes unprofitable to sink,
2607 // all earlier instructions won't be sunk either,
2608 // so preemptively keep InstructionsProfitableToSink in sync.
2609 // FIXME: is this the most performant approach?
2610 for (auto *I : *LRI)
2611 InstructionsProfitableToSink.erase(I);
2612 if (!ProfitableToSinkInstruction(LRI)) {
2613 // Everything starting with this instruction won't be sunk.
2614 ScanIdx = Idx;
2615 InstructionsToSink = InstructionsProfitableToSink;
2616 }
2617 ++LRI;
2618 --Idx;
2619 }
2620 }
2621
2622 // If no instructions can be sunk, early-return.
2623 if (ScanIdx == 0)
2624 return false;
2625 }
2626
2627 bool Changed = false;
2628
2629 if (HaveNonUnconditionalPredecessors) {
2630 if (!followedByDeoptOrUnreachable) {
2631 // It is always legal to sink common instructions from unconditional
2632 // predecessors. However, if not all predecessors are unconditional,
2633 // this transformation might be pessimizing. So as a rule of thumb,
2634 // don't do it unless we'd sink at least one non-speculatable instruction.
2635 // See https://bugs.llvm.org/show_bug.cgi?id=30244
2636 LRI.reset();
2637 int Idx = 0;
2638 bool Profitable = false;
2639 while (Idx < ScanIdx) {
2640 if (!isSafeToSpeculativelyExecute((*LRI)[0])) {
2641 Profitable = true;
2642 break;
2643 }
2644 --LRI;
2645 ++Idx;
2646 }
2647 if (!Profitable)
2648 return false;
2649 }
2650
2651 LLVM_DEBUG(dbgs() << "SINK: Splitting edge\n");
2652 // We have a conditional edge and we're going to sink some instructions.
2653 // Insert a new block postdominating all blocks we're going to sink from.
2654 if (!SplitBlockPredecessors(BB, UnconditionalPreds, ".sink.split", DTU))
2655 // Edges couldn't be split.
2656 return false;
2657 Changed = true;
2658 }
2659
2660 // Now that we've analyzed all potential sinking candidates, perform the
2661 // actual sink. We iteratively sink the last non-terminator of the source
2662 // blocks into their common successor unless doing so would require too
2663 // many PHI instructions to be generated (currently only one PHI is allowed
2664 // per sunk instruction).
2665 //
2666 // We can use InstructionsToSink to discount values needing PHI-merging that will
2667 // actually be sunk in a later iteration. This allows us to be more
2668 // aggressive in what we sink. This does allow a false positive where we
2669 // sink presuming a later value will also be sunk, but stop half way through
2670 // and never actually sink it which means we produce more PHIs than intended.
2671 // This is unlikely in practice though.
2672 int SinkIdx = 0;
2673 for (; SinkIdx != ScanIdx; ++SinkIdx) {
2674 LLVM_DEBUG(dbgs() << "SINK: Sink: "
2675 << *UnconditionalPreds[0]->getTerminator()->getPrevNode()
2676 << "\n");
2677
2678 // Because we've sunk every instruction in turn, the current instruction to
2679 // sink is always at index 0.
2680 LRI.reset();
2681
2682 sinkLastInstruction(UnconditionalPreds);
2683 NumSinkCommonInstrs++;
2684 Changed = true;
2685 }
2686 if (SinkIdx != 0)
2687 ++NumSinkCommonCode;
2688 return Changed;
2689}
2690
2691namespace {
2692
2693struct CompatibleSets {
2694 using SetTy = SmallVector<InvokeInst *, 2>;
2695
2697
2698 static bool shouldBelongToSameSet(ArrayRef<InvokeInst *> Invokes);
2699
2700 SetTy &getCompatibleSet(InvokeInst *II);
2701
2702 void insert(InvokeInst *II);
2703};
2704
2705CompatibleSets::SetTy &CompatibleSets::getCompatibleSet(InvokeInst *II) {
2706 // Perform a linear scan over all the existing sets, see if the new `invoke`
2707 // is compatible with any particular set. Since we know that all the `invokes`
2708 // within a set are compatible, only check the first `invoke` in each set.
2709 // WARNING: at worst, this has quadratic complexity.
2710 for (CompatibleSets::SetTy &Set : Sets) {
2711 if (CompatibleSets::shouldBelongToSameSet({Set.front(), II}))
2712 return Set;
2713 }
2714
2715 // Otherwise, we either had no sets yet, or this invoke forms a new set.
2716 return Sets.emplace_back();
2717}
2718
2719void CompatibleSets::insert(InvokeInst *II) {
2720 getCompatibleSet(II).emplace_back(II);
2721}
2722
2723bool CompatibleSets::shouldBelongToSameSet(ArrayRef<InvokeInst *> Invokes) {
2724 assert(Invokes.size() == 2 && "Always called with exactly two candidates.");
2725
2726 // Can we theoretically merge these `invoke`s?
2727 auto IsIllegalToMerge = [](InvokeInst *II) {
2728 return II->cannotMerge() || II->isInlineAsm();
2729 };
2730 if (any_of(Invokes, IsIllegalToMerge))
2731 return false;
2732
2733 // Either both `invoke`s must be direct,
2734 // or both `invoke`s must be indirect.
2735 auto IsIndirectCall = [](InvokeInst *II) { return II->isIndirectCall(); };
2736 bool HaveIndirectCalls = any_of(Invokes, IsIndirectCall);
2737 bool AllCallsAreIndirect = all_of(Invokes, IsIndirectCall);
2738 if (HaveIndirectCalls) {
2739 if (!AllCallsAreIndirect)
2740 return false;
2741 } else {
2742 // All callees must be identical.
2743 Value *Callee = nullptr;
2744 for (InvokeInst *II : Invokes) {
2745 Value *CurrCallee = II->getCalledOperand();
2746 assert(CurrCallee && "There is always a called operand.");
2747 if (!Callee)
2748 Callee = CurrCallee;
2749 else if (Callee != CurrCallee)
2750 return false;
2751 }
2752 }
2753
2754 // Either both `invoke`s must not have a normal destination,
2755 // or both `invoke`s must have a normal destination,
2756 auto HasNormalDest = [](InvokeInst *II) {
2757 return !isa<UnreachableInst>(II->getNormalDest()->getFirstNonPHIOrDbg());
2758 };
2759 if (any_of(Invokes, HasNormalDest)) {
2760 // Do not merge `invoke` that does not have a normal destination with one
2761 // that does have a normal destination, even though doing so would be legal.
2762 if (!all_of(Invokes, HasNormalDest))
2763 return false;
2764
2765 // All normal destinations must be identical.
2766 BasicBlock *NormalBB = nullptr;
2767 for (InvokeInst *II : Invokes) {
2768 BasicBlock *CurrNormalBB = II->getNormalDest();
2769 assert(CurrNormalBB && "There is always a 'continue to' basic block.");
2770 if (!NormalBB)
2771 NormalBB = CurrNormalBB;
2772 else if (NormalBB != CurrNormalBB)
2773 return false;
2774 }
2775
2776 // In the normal destination, the incoming values for these two `invoke`s
2777 // must be compatible.
2778 SmallPtrSet<Value *, 16> EquivalenceSet(llvm::from_range, Invokes);
2780 NormalBB, {Invokes[0]->getParent(), Invokes[1]->getParent()},
2781 &EquivalenceSet))
2782 return false;
2783 }
2784
2785#ifndef NDEBUG
2786 // All unwind destinations must be identical.
2787 // We know that because we have started from said unwind destination.
2788 BasicBlock *UnwindBB = nullptr;
2789 for (InvokeInst *II : Invokes) {
2790 BasicBlock *CurrUnwindBB = II->getUnwindDest();
2791 assert(CurrUnwindBB && "There is always an 'unwind to' basic block.");
2792 if (!UnwindBB)
2793 UnwindBB = CurrUnwindBB;
2794 else
2795 assert(UnwindBB == CurrUnwindBB && "Unexpected unwind destination.");
2796 }
2797#endif
2798
2799 // In the unwind destination, the incoming values for these two `invoke`s
2800 // must be compatible.
2802 Invokes.front()->getUnwindDest(),
2803 {Invokes[0]->getParent(), Invokes[1]->getParent()}))
2804 return false;
2805
2806 // Ignoring arguments, these `invoke`s must be identical,
2807 // including operand bundles.
2808 const InvokeInst *II0 = Invokes.front();
2809 for (auto *II : Invokes.drop_front())
2810 if (!II->isSameOperationAs(II0, Instruction::CompareUsingIntersectedAttrs))
2811 return false;
2812
2813 // Can we theoretically form the data operands for the merged `invoke`?
2814 auto IsIllegalToMergeArguments = [](auto Ops) {
2815 Use &U0 = std::get<0>(Ops);
2816 Use &U1 = std::get<1>(Ops);
2817 if (U0 == U1)
2818 return false;
2820 U0.getOperandNo());
2821 };
2822 assert(Invokes.size() == 2 && "Always called with exactly two candidates.");
2823 if (any_of(zip(Invokes[0]->data_ops(), Invokes[1]->data_ops()),
2824 IsIllegalToMergeArguments))
2825 return false;
2826
2827 return true;
2828}
2829
2830} // namespace
2831
2832// Merge all invokes in the provided set, all of which are compatible
2833// as per the `CompatibleSets::shouldBelongToSameSet()`.
2835 DomTreeUpdater *DTU) {
2836 assert(Invokes.size() >= 2 && "Must have at least two invokes to merge.");
2837
2839 if (DTU)
2840 Updates.reserve(2 + 3 * Invokes.size());
2841
2842 bool HasNormalDest =
2843 !isa<UnreachableInst>(Invokes[0]->getNormalDest()->getFirstNonPHIOrDbg());
2844
2845 // Clone one of the invokes into a new basic block.
2846 // Since they are all compatible, it doesn't matter which invoke is cloned.
2847 InvokeInst *MergedInvoke = [&Invokes, HasNormalDest]() {
2848 InvokeInst *II0 = Invokes.front();
2849 BasicBlock *II0BB = II0->getParent();
2850 BasicBlock *InsertBeforeBlock =
2851 II0->getParent()->getIterator()->getNextNode();
2852 Function *Func = II0BB->getParent();
2853 LLVMContext &Ctx = II0->getContext();
2854
2855 BasicBlock *MergedInvokeBB = BasicBlock::Create(
2856 Ctx, II0BB->getName() + ".invoke", Func, InsertBeforeBlock);
2857
2858 auto *MergedInvoke = cast<InvokeInst>(II0->clone());
2859 // NOTE: all invokes have the same attributes, so no handling needed.
2860 MergedInvoke->insertInto(MergedInvokeBB, MergedInvokeBB->end());
2861
2862 if (!HasNormalDest) {
2863 // This set does not have a normal destination,
2864 // so just form a new block with unreachable terminator.
2865 BasicBlock *MergedNormalDest = BasicBlock::Create(
2866 Ctx, II0BB->getName() + ".cont", Func, InsertBeforeBlock);
2867 auto *UI = new UnreachableInst(Ctx, MergedNormalDest);
2868 UI->setDebugLoc(DebugLoc::getTemporary());
2869 MergedInvoke->setNormalDest(MergedNormalDest);
2870 }
2871
2872 // The unwind destination, however, remainds identical for all invokes here.
2873
2874 return MergedInvoke;
2875 }();
2876
2877 if (DTU) {
2878 // Predecessor blocks that contained these invokes will now branch to
2879 // the new block that contains the merged invoke, ...
2880 for (InvokeInst *II : Invokes)
2881 Updates.push_back(
2882 {DominatorTree::Insert, II->getParent(), MergedInvoke->getParent()});
2883
2884 // ... which has the new `unreachable` block as normal destination,
2885 // or unwinds to the (same for all `invoke`s in this set) `landingpad`,
2886 for (BasicBlock *SuccBBOfMergedInvoke : successors(MergedInvoke))
2887 Updates.push_back({DominatorTree::Insert, MergedInvoke->getParent(),
2888 SuccBBOfMergedInvoke});
2889
2890 // Since predecessor blocks now unconditionally branch to a new block,
2891 // they no longer branch to their original successors.
2892 for (InvokeInst *II : Invokes)
2893 for (BasicBlock *SuccOfPredBB : successors(II->getParent()))
2894 Updates.push_back(
2895 {DominatorTree::Delete, II->getParent(), SuccOfPredBB});
2896 }
2897
2898 bool IsIndirectCall = Invokes[0]->isIndirectCall();
2899
2900 // Form the merged operands for the merged invoke.
2901 for (Use &U : MergedInvoke->operands()) {
2902 // Only PHI together the indirect callees and data operands.
2903 if (MergedInvoke->isCallee(&U)) {
2904 if (!IsIndirectCall)
2905 continue;
2906 } else if (!MergedInvoke->isDataOperand(&U))
2907 continue;
2908
2909 // Don't create trivial PHI's with all-identical incoming values.
2910 bool NeedPHI = any_of(Invokes, [&U](InvokeInst *II) {
2911 return II->getOperand(U.getOperandNo()) != U.get();
2912 });
2913 if (!NeedPHI)
2914 continue;
2915
2916 // Form a PHI out of all the data ops under this index.
2918 U->getType(), /*NumReservedValues=*/Invokes.size(), "", MergedInvoke->getIterator());
2919 for (InvokeInst *II : Invokes)
2920 PN->addIncoming(II->getOperand(U.getOperandNo()), II->getParent());
2921
2922 U.set(PN);
2923 }
2924
2925 // We've ensured that each PHI node has compatible (identical) incoming values
2926 // when coming from each of the `invoke`s in the current merge set,
2927 // so update the PHI nodes accordingly.
2928 for (BasicBlock *Succ : successors(MergedInvoke))
2929 addPredecessorToBlock(Succ, /*NewPred=*/MergedInvoke->getParent(),
2930 /*ExistPred=*/Invokes.front()->getParent());
2931
2932 // And finally, replace the original `invoke`s with an unconditional branch
2933 // to the block with the merged `invoke`. Also, give that merged `invoke`
2934 // the merged debugloc of all the original `invoke`s.
2935 DILocation *MergedDebugLoc = nullptr;
2936 for (InvokeInst *II : Invokes) {
2937 // Compute the debug location common to all the original `invoke`s.
2938 if (!MergedDebugLoc)
2939 MergedDebugLoc = II->getDebugLoc();
2940 else
2941 MergedDebugLoc =
2942 DebugLoc::getMergedLocation(MergedDebugLoc, II->getDebugLoc());
2943
2944 // And replace the old `invoke` with an unconditionally branch
2945 // to the block with the merged `invoke`.
2946 for (BasicBlock *OrigSuccBB : successors(II->getParent()))
2947 OrigSuccBB->removePredecessor(II->getParent());
2948 auto *BI = UncondBrInst::Create(MergedInvoke->getParent(), II->getParent());
2949 // The unconditional branch is part of the replacement for the original
2950 // invoke, so should use its DebugLoc.
2951 BI->setDebugLoc(II->getDebugLoc());
2952 bool Success = MergedInvoke->tryIntersectAttributes(II);
2953 assert(Success && "Merged invokes with incompatible attributes");
2954 // For NDEBUG Compile
2955 (void)Success;
2956 II->replaceAllUsesWith(MergedInvoke);
2957 II->eraseFromParent();
2958 ++NumInvokesMerged;
2959 }
2960 MergedInvoke->setDebugLoc(MergedDebugLoc);
2961 ++NumInvokeSetsFormed;
2962
2963 if (DTU)
2964 DTU->applyUpdates(Updates);
2965}
2966
2967/// If this block is a `landingpad` exception handling block, categorize all
2968/// the predecessor `invoke`s into sets, with all `invoke`s in each set
2969/// being "mergeable" together, and then merge invokes in each set together.
2970///
2971/// This is a weird mix of hoisting and sinking. Visually, it goes from:
2972/// [...] [...]
2973/// | |
2974/// [invoke0] [invoke1]
2975/// / \ / \
2976/// [cont0] [landingpad] [cont1]
2977/// to:
2978/// [...] [...]
2979/// \ /
2980/// [invoke]
2981/// / \
2982/// [cont] [landingpad]
2983///
2984/// But of course we can only do that if the invokes share the `landingpad`,
2985/// edges invoke0->cont0 and invoke1->cont1 are "compatible",
2986/// and the invoked functions are "compatible".
2989 return false;
2990
2991 bool Changed = false;
2992
2993 // FIXME: generalize to all exception handling blocks?
2994 if (!BB->isLandingPad())
2995 return Changed;
2996
2997 CompatibleSets Grouper;
2998
2999 // Record all the predecessors of this `landingpad`. As per verifier,
3000 // the only allowed predecessor is the unwind edge of an `invoke`.
3001 // We want to group "compatible" `invokes` into the same set to be merged.
3002 for (BasicBlock *PredBB : predecessors(BB))
3003 Grouper.insert(cast<InvokeInst>(PredBB->getTerminator()));
3004
3005 // And now, merge `invoke`s that were grouped togeter.
3006 for (ArrayRef<InvokeInst *> Invokes : Grouper.Sets) {
3007 if (Invokes.size() < 2)
3008 continue;
3009 Changed = true;
3010 mergeCompatibleInvokesImpl(Invokes, DTU);
3011 }
3012
3013 return Changed;
3014}
3015
3016namespace {
3017/// Track ephemeral values, which should be ignored for cost-modelling
3018/// purposes. Requires walking instructions in reverse order.
3019class EphemeralValueTracker {
3020 SmallPtrSet<const Instruction *, 32> EphValues;
3021
3022 bool isEphemeral(const Instruction *I) {
3023 if (isa<AssumeInst>(I))
3024 return true;
3025 return !I->mayHaveSideEffects() && !I->isTerminator() &&
3026 all_of(I->users(), [&](const User *U) {
3027 return EphValues.count(cast<Instruction>(U));
3028 });
3029 }
3030
3031public:
3032 bool track(const Instruction *I) {
3033 if (isEphemeral(I)) {
3034 EphValues.insert(I);
3035 return true;
3036 }
3037 return false;
3038 }
3039
3040 bool contains(const Instruction *I) const { return EphValues.contains(I); }
3041};
3042} // namespace
3043
3044/// Determine if we can hoist sink a sole store instruction out of a
3045/// conditional block.
3046///
3047/// We are looking for code like the following:
3048/// BrBB:
3049/// store i32 %add, i32* %arrayidx2
3050/// ... // No other stores or function calls (we could be calling a memory
3051/// ... // function).
3052/// %cmp = icmp ult %x, %y
3053/// br i1 %cmp, label %EndBB, label %ThenBB
3054/// ThenBB:
3055/// store i32 %add5, i32* %arrayidx2
3056/// br label EndBB
3057/// EndBB:
3058/// ...
3059/// We are going to transform this into:
3060/// BrBB:
3061/// store i32 %add, i32* %arrayidx2
3062/// ... //
3063/// %cmp = icmp ult %x, %y
3064/// %add.add5 = select i1 %cmp, i32 %add, %add5
3065/// store i32 %add.add5, i32* %arrayidx2
3066/// ...
3067///
3068/// \return The pointer to the value of the previous store if the store can be
3069/// hoisted into the predecessor block. 0 otherwise.
3071 BasicBlock *StoreBB, BasicBlock *EndBB) {
3072 StoreInst *StoreToHoist = dyn_cast<StoreInst>(I);
3073 if (!StoreToHoist)
3074 return nullptr;
3075
3076 // Volatile or atomic.
3077 if (!StoreToHoist->isSimple())
3078 return nullptr;
3079
3080 Value *StorePtr = StoreToHoist->getPointerOperand();
3081 Type *StoreTy = StoreToHoist->getValueOperand()->getType();
3082
3083 // Look for a store to the same pointer in BrBB.
3084 unsigned MaxNumInstToLookAt = 9;
3085 // Skip pseudo probe intrinsic calls which are not really killing any memory
3086 // accesses.
3087 for (Instruction &CurI : reverse(*BrBB)) {
3088 if (!MaxNumInstToLookAt)
3089 break;
3090 --MaxNumInstToLookAt;
3091
3092 if (isa<PseudoProbeInst>(CurI))
3093 continue;
3094
3095 // Could be calling an instruction that affects memory like free().
3096 if (CurI.mayWriteToMemory() && !isa<StoreInst>(CurI))
3097 return nullptr;
3098
3099 if (auto *SI = dyn_cast<StoreInst>(&CurI)) {
3100 // Found the previous store to same location and type. Make sure it is
3101 // simple, to avoid introducing a spurious non-atomic write after an
3102 // atomic write.
3103 if (SI->getPointerOperand() == StorePtr &&
3104 SI->getValueOperand()->getType() == StoreTy && SI->isSimple() &&
3105 SI->getAlign() >= StoreToHoist->getAlign())
3106 // Found the previous store, return its value operand.
3107 return SI->getValueOperand();
3108 return nullptr; // Unknown store.
3109 }
3110
3111 if (auto *LI = dyn_cast<LoadInst>(&CurI)) {
3112 if (LI->getPointerOperand() == StorePtr && LI->getType() == StoreTy &&
3113 LI->isSimple() && LI->getAlign() >= StoreToHoist->getAlign()) {
3114 Value *Obj = getUnderlyingObject(StorePtr);
3115 bool ExplicitlyDereferenceableOnly;
3116 // The dereferenceability query here is only required to satisfy the
3117 // writable contract, actual dereferenceability is proven by the
3118 // presence of an access. As such, we can ignore frees.
3119 if (isWritableObject(Obj, ExplicitlyDereferenceableOnly) &&
3122 .WithoutRet) &&
3123 (!ExplicitlyDereferenceableOnly ||
3124 isDereferenceablePointer(StorePtr, StoreTy, LI->getDataLayout(),
3125 /*IgnoreFree=*/true))) {
3126 // Found a previous load, return it.
3127 return LI;
3128 }
3129 }
3130 // The load didn't work out, but we may still find a store.
3131 }
3132 }
3133
3134 return nullptr;
3135}
3136
3137/// Estimate the cost of the insertion(s) and check that the PHI nodes can be
3138/// converted to selects.
3140 BasicBlock *EndBB,
3141 unsigned &SpeculatedInstructions,
3142 InstructionCost &Cost,
3143 const TargetTransformInfo &TTI) {
3145 BB->getParent()->hasMinSize()
3148
3149 bool HaveRewritablePHIs = false;
3150 for (PHINode &PN : EndBB->phis()) {
3151 Value *OrigV = PN.getIncomingValueForBlock(BB);
3152 Value *ThenV = PN.getIncomingValueForBlock(ThenBB);
3153
3154 // FIXME: Try to remove some of the duplication with
3155 // hoistCommonCodeFromSuccessors. Skip PHIs which are trivial.
3156 if (ThenV == OrigV)
3157 continue;
3158
3159 Cost += TTI.getCmpSelInstrCost(Instruction::Select, PN.getType(),
3160 Type::getInt1Ty(PN.getContext()),
3162
3163 // Don't convert to selects if we could remove undefined behavior instead.
3164 if (passingValueIsAlwaysUndefined(OrigV, &PN) ||
3166 return false;
3167
3168 HaveRewritablePHIs = true;
3169 ConstantExpr *OrigCE = dyn_cast<ConstantExpr>(OrigV);
3170 ConstantExpr *ThenCE = dyn_cast<ConstantExpr>(ThenV);
3171 if (!OrigCE && !ThenCE)
3172 continue; // Known cheap (FIXME: Maybe not true for aggregates).
3173
3174 InstructionCost OrigCost = OrigCE ? computeSpeculationCost(OrigCE, TTI) : 0;
3175 InstructionCost ThenCost = ThenCE ? computeSpeculationCost(ThenCE, TTI) : 0;
3176 InstructionCost MaxCost =
3178 if (OrigCost + ThenCost > MaxCost)
3179 return false;
3180
3181 // Account for the cost of an unfolded ConstantExpr which could end up
3182 // getting expanded into Instructions.
3183 // FIXME: This doesn't account for how many operations are combined in the
3184 // constant expression.
3185 ++SpeculatedInstructions;
3186 if (SpeculatedInstructions > 1)
3187 return false;
3188 }
3189
3190 return HaveRewritablePHIs;
3191}
3192
3194 std::optional<bool> Invert,
3195 const TargetTransformInfo &TTI) {
3196 // If the branch is non-unpredictable, and is predicted to *not* branch to
3197 // the `then` block, then avoid speculating it.
3198 if (BI->getMetadata(LLVMContext::MD_unpredictable))
3199 return true;
3200
3201 uint64_t TWeight, FWeight;
3202 if (!extractBranchWeights(*BI, TWeight, FWeight) || (TWeight + FWeight) == 0)
3203 return true;
3204
3205 if (!Invert.has_value())
3206 return false;
3207
3208 uint64_t EndWeight = *Invert ? TWeight : FWeight;
3209 BranchProbability BIEndProb =
3210 BranchProbability::getBranchProbability(EndWeight, TWeight + FWeight);
3211 BranchProbability Likely = TTI.getPredictableBranchThreshold();
3212 return BIEndProb < Likely;
3213}
3214
3215/// Speculate a conditional basic block flattening the CFG.
3216///
3217/// Note that this is a very risky transform currently. Speculating
3218/// instructions like this is most often not desirable. Instead, there is an MI
3219/// pass which can do it with full awareness of the resource constraints.
3220/// However, some cases are "obvious" and we should do directly. An example of
3221/// this is speculating a single, reasonably cheap instruction.
3222///
3223/// There is only one distinct advantage to flattening the CFG at the IR level:
3224/// it makes very common but simplistic optimizations such as are common in
3225/// instcombine and the DAG combiner more powerful by removing CFG edges and
3226/// modeling their effects with easier to reason about SSA value graphs.
3227///
3228///
3229/// An illustration of this transform is turning this IR:
3230/// \code
3231/// BB:
3232/// %cmp = icmp ult %x, %y
3233/// br i1 %cmp, label %EndBB, label %ThenBB
3234/// ThenBB:
3235/// %sub = sub %x, %y
3236/// br label BB2
3237/// EndBB:
3238/// %phi = phi [ %sub, %ThenBB ], [ 0, %BB ]
3239/// ...
3240/// \endcode
3241///
3242/// Into this IR:
3243/// \code
3244/// BB:
3245/// %cmp = icmp ult %x, %y
3246/// %sub = sub %x, %y
3247/// %cond = select i1 %cmp, 0, %sub
3248/// ...
3249/// \endcode
3250///
3251/// \returns true if the conditional block is removed.
3252bool SimplifyCFGOpt::speculativelyExecuteBB(CondBrInst *BI,
3253 BasicBlock *ThenBB) {
3254 if (!Options.SpeculateBlocks)
3255 return false;
3256
3257 BasicBlock *BB = BI->getParent();
3258 BasicBlock *EndBB = ThenBB->getTerminator()->getSuccessor(0);
3259 InstructionCost Budget =
3261
3262 // If ThenBB is actually on the false edge of the conditional branch, remember
3263 // to swap the select operands later.
3264 bool Invert = false;
3265 if (ThenBB != BI->getSuccessor(0)) {
3266 assert(ThenBB == BI->getSuccessor(1) && "No edge from 'if' block?");
3267 Invert = true;
3268 }
3269 assert(EndBB == BI->getSuccessor(!Invert) && "No edge from to end block");
3270
3271 if (!isProfitableToSpeculate(BI, Invert, TTI))
3272 return false;
3273
3274 // Keep a count of how many times instructions are used within ThenBB when
3275 // they are candidates for sinking into ThenBB. Specifically:
3276 // - They are defined in BB, and
3277 // - They have no side effects, and
3278 // - All of their uses are in ThenBB.
3279 SmallDenseMap<Instruction *, unsigned, 4> SinkCandidateUseCounts;
3280
3281 SmallVector<Instruction *, 4> SpeculatedPseudoProbes;
3282
3283 unsigned SpeculatedInstructions = 0;
3284 bool HoistLoadsStores = Options.HoistLoadsStoresWithCondFaulting;
3285 SmallVector<Instruction *, 2> SpeculatedConditionalLoadsStores;
3286 Value *SpeculatedStoreValue = nullptr;
3287 StoreInst *SpeculatedStore = nullptr;
3288 EphemeralValueTracker EphTracker;
3289 for (Instruction &I : reverse(drop_end(*ThenBB))) {
3290 // Skip pseudo probes. The consequence is we lose track of the branch
3291 // probability for ThenBB, which is fine since the optimization here takes
3292 // place regardless of the branch probability.
3293 if (isa<PseudoProbeInst>(I)) {
3294 // The probe should be deleted so that it will not be over-counted when
3295 // the samples collected on the non-conditional path are counted towards
3296 // the conditional path. We leave it for the counts inference algorithm to
3297 // figure out a proper count for an unknown probe.
3298 SpeculatedPseudoProbes.push_back(&I);
3299 continue;
3300 }
3301
3302 // Ignore ephemeral values, they will be dropped by the transform.
3303 if (EphTracker.track(&I))
3304 continue;
3305
3306 // Only speculatively execute a single instruction (not counting the
3307 // terminator) for now.
3308 bool IsSafeCheapLoadStore = HoistLoadsStores &&
3310 SpeculatedConditionalLoadsStores.size() <
3312 // Not count load/store into cost if target supports conditional faulting
3313 // b/c it's cheap to speculate it.
3314 if (IsSafeCheapLoadStore)
3315 SpeculatedConditionalLoadsStores.push_back(&I);
3316 else
3317 ++SpeculatedInstructions;
3318
3319 if (SpeculatedInstructions > 1)
3320 return false;
3321
3322 // Don't hoist the instruction if it's unsafe or expensive.
3323 if (!IsSafeCheapLoadStore &&
3325 !(HoistCondStores && !SpeculatedStoreValue &&
3326 (SpeculatedStoreValue =
3327 isSafeToSpeculateStore(&I, BB, ThenBB, EndBB))))
3328 return false;
3329 if (!IsSafeCheapLoadStore && !SpeculatedStoreValue &&
3332 return false;
3333
3334 // Store the store speculation candidate.
3335 if (!SpeculatedStore && SpeculatedStoreValue)
3336 SpeculatedStore = cast<StoreInst>(&I);
3337
3338 // Do not hoist the instruction if any of its operands are defined but not
3339 // used in BB. The transformation will prevent the operand from
3340 // being sunk into the use block.
3341 for (Use &Op : I.operands()) {
3343 if (!OpI || OpI->getParent() != BB || OpI->mayHaveSideEffects())
3344 continue; // Not a candidate for sinking.
3345
3346 ++SinkCandidateUseCounts[OpI];
3347 }
3348 }
3349
3350 // Consider any sink candidates which are only used in ThenBB as costs for
3351 // speculation. Note, while we iterate over a DenseMap here, we are summing
3352 // and so iteration order isn't significant.
3353 for (const auto &[Inst, Count] : SinkCandidateUseCounts)
3354 if (Inst->hasNUses(Count)) {
3355 ++SpeculatedInstructions;
3356 if (SpeculatedInstructions > 1)
3357 return false;
3358 }
3359
3360 // Check that we can insert the selects and that it's not too expensive to do
3361 // so.
3362 bool Convert =
3363 SpeculatedStore != nullptr || !SpeculatedConditionalLoadsStores.empty();
3365 Convert |= validateAndCostRequiredSelects(BB, ThenBB, EndBB,
3366 SpeculatedInstructions, Cost, TTI);
3367 if (!Convert || Cost > Budget)
3368 return false;
3369
3370 // If we get here, we can hoist the instruction and if-convert.
3371 LLVM_DEBUG(dbgs() << "SPECULATIVELY EXECUTING BB" << *ThenBB << "\n";);
3372
3373 Instruction *Sel = nullptr;
3374 Value *BrCond = BI->getCondition();
3375 // Insert a select of the value of the speculated store.
3376 if (SpeculatedStoreValue) {
3377 IRBuilder<NoFolder> Builder(BI);
3378 Value *OrigV = SpeculatedStore->getValueOperand();
3379 Value *TrueV = SpeculatedStore->getValueOperand();
3380 Value *FalseV = SpeculatedStoreValue;
3381 if (Invert)
3382 std::swap(TrueV, FalseV);
3383 Value *S = Builder.CreateSelect(
3384 BrCond, TrueV, FalseV, "spec.store.select", BI);
3385 Sel = cast<Instruction>(S);
3386 SpeculatedStore->setOperand(0, S);
3387 SpeculatedStore->applyMergedLocation(BI->getDebugLoc(),
3388 SpeculatedStore->getDebugLoc());
3389 // The value stored is still conditional, but the store itself is now
3390 // unconditionally executed, so we must be sure that any linked dbg.assign
3391 // intrinsics are tracking the new stored value (the result of the
3392 // select). If we don't, and the store were to be removed by another pass
3393 // (e.g. DSE), then we'd eventually end up emitting a location describing
3394 // the conditional value, unconditionally.
3395 //
3396 // === Before this transformation ===
3397 // pred:
3398 // store %one, %x.dest, !DIAssignID !1
3399 // dbg.assign %one, "x", ..., !1, ...
3400 // br %cond if.then
3401 //
3402 // if.then:
3403 // store %two, %x.dest, !DIAssignID !2
3404 // dbg.assign %two, "x", ..., !2, ...
3405 //
3406 // === After this transformation ===
3407 // pred:
3408 // store %one, %x.dest, !DIAssignID !1
3409 // dbg.assign %one, "x", ..., !1
3410 /// ...
3411 // %merge = select %cond, %two, %one
3412 // store %merge, %x.dest, !DIAssignID !2
3413 // dbg.assign %merge, "x", ..., !2
3414 for (DbgVariableRecord *DbgAssign :
3415 at::getDVRAssignmentMarkers(SpeculatedStore))
3416 if (llvm::is_contained(DbgAssign->location_ops(), OrigV))
3417 DbgAssign->replaceVariableLocationOp(OrigV, S);
3418 }
3419
3420 // Metadata can be dependent on the condition we are hoisting above.
3421 // Strip all UB-implying metadata on the instruction. Drop the debug loc
3422 // to avoid making it appear as if the condition is a constant, which would
3423 // be misleading while debugging.
3424 // Similarly strip attributes that maybe dependent on condition we are
3425 // hoisting above.
3426 for (auto &I : make_early_inc_range(*ThenBB)) {
3427 if (!SpeculatedStoreValue || &I != SpeculatedStore) {
3428 I.dropLocation();
3429 }
3430 I.dropUBImplyingAttrsAndMetadata();
3431
3432 // Drop ephemeral values.
3433 if (EphTracker.contains(&I)) {
3434 I.replaceAllUsesWith(PoisonValue::get(I.getType()));
3435 I.eraseFromParent();
3436 }
3437 }
3438
3439 // Hoist the instructions.
3440 // Drop DbgVariableRecords attached to these instructions.
3441 for (auto &It : *ThenBB)
3442 for (DbgRecord &DR : make_early_inc_range(It.getDbgRecordRange()))
3443 // Drop all records except assign-kind DbgVariableRecords (dbg.assign
3444 // equivalent).
3445 if (DbgVariableRecord *DVR = dyn_cast<DbgVariableRecord>(&DR);
3446 !DVR || !DVR->isDbgAssign())
3447 It.dropOneDbgRecord(&DR);
3448 BB->splice(BI->getIterator(), ThenBB, ThenBB->begin(),
3449 std::prev(ThenBB->end()));
3450
3451 if (!SpeculatedConditionalLoadsStores.empty())
3452 hoistConditionalLoadsStores(BI, SpeculatedConditionalLoadsStores, Invert,
3453 Sel);
3454
3455 // Insert selects and rewrite the PHI operands.
3456 IRBuilder<NoFolder> Builder(BI);
3457 for (PHINode &PN : EndBB->phis()) {
3458 unsigned OrigI = PN.getBasicBlockIndex(BB);
3459 unsigned ThenI = PN.getBasicBlockIndex(ThenBB);
3460 Value *OrigV = PN.getIncomingValue(OrigI);
3461 Value *ThenV = PN.getIncomingValue(ThenI);
3462
3463 // Skip PHIs which are trivial.
3464 if (OrigV == ThenV)
3465 continue;
3466
3467 // Create a select whose true value is the speculatively executed value and
3468 // false value is the pre-existing value. Swap them if the branch
3469 // destinations were inverted.
3470 Value *TrueV = ThenV, *FalseV = OrigV;
3471 if (Invert)
3472 std::swap(TrueV, FalseV);
3473 // Propagate fast-math flags from the phi node to the replacement select.
3474 Value *V = Builder.CreateSelectFMF(
3475 BrCond, TrueV, FalseV, PN.getFastMathFlagsOrNone(), "spec.select", BI);
3476 PN.setIncomingValue(OrigI, V);
3477 PN.setIncomingValue(ThenI, V);
3478 }
3479
3480 // Remove speculated pseudo probes.
3481 for (Instruction *I : SpeculatedPseudoProbes)
3482 I->eraseFromParent();
3483
3484 ++NumSpeculations;
3485 return true;
3486}
3487
3489
3490// Return false if number of blocks searched is too much.
3491static bool findReaching(BasicBlock *BB, BasicBlock *DefBB,
3492 BlocksSet &ReachesNonLocalUses) {
3493 if (BB == DefBB)
3494 return true;
3495 if (!ReachesNonLocalUses.insert(BB).second)
3496 return true;
3497
3498 if (ReachesNonLocalUses.size() > MaxJumpThreadingLiveBlocks)
3499 return false;
3500 for (BasicBlock *Pred : predecessors(BB))
3501 if (!findReaching(Pred, DefBB, ReachesNonLocalUses))
3502 return false;
3503 return true;
3504}
3505
3506/// Return true if we can thread a branch across this block.
3508 BlocksSet &NonLocalUseBlocks) {
3509 int Size = 0;
3510 EphemeralValueTracker EphTracker;
3511
3512 // Walk the loop in reverse so that we can identify ephemeral values properly
3513 // (values only feeding assumes).
3514 for (Instruction &I : reverse(*BB)) {
3515 // Can't fold blocks that contain noduplicate or convergent calls.
3516 if (CallInst *CI = dyn_cast<CallInst>(&I))
3517 if (CI->cannotDuplicate() || CI->isConvergent())
3518 return false;
3519
3520 // Ignore ephemeral values which are deleted during codegen.
3521 // We will delete Phis while threading, so Phis should not be accounted in
3522 // block's size.
3523 if (!EphTracker.track(&I) && !isa<PHINode>(I)) {
3524 if (Size++ > MaxSmallBlockSize)
3525 return false; // Don't clone large BB's.
3526 }
3527
3528 // Record blocks with non-local uses of values defined in the current basic
3529 // block.
3530 for (User *U : I.users()) {
3532 BasicBlock *UsedInBB = UI->getParent();
3533 if (UsedInBB == BB) {
3534 if (isa<PHINode>(UI))
3535 return false;
3536 } else
3537 NonLocalUseBlocks.insert(UsedInBB);
3538 }
3539
3540 // Looks ok, continue checking.
3541 }
3542
3543 return true;
3544}
3545
3547 BasicBlock *To) {
3548 // Don't look past the block defining the value, we might get the value from
3549 // a previous loop iteration.
3550 auto *I = dyn_cast<Instruction>(V);
3551 if (I && I->getParent() == To)
3552 return nullptr;
3553
3554 // We know the value if the From block branches on it.
3555 auto *BI = dyn_cast<CondBrInst>(From->getTerminator());
3556 if (BI && BI->getCondition() == V &&
3557 BI->getSuccessor(0) != BI->getSuccessor(1))
3558 return BI->getSuccessor(0) == To ? ConstantInt::getTrue(BI->getContext())
3560
3561 return nullptr;
3562}
3563
3565 return CB->isConvergent() && !isa<ConvergenceControlInst>(CB) &&
3567}
3568
3570 BasicBlock *StopBB) {
3571 static constexpr unsigned MaxInstructionsToScan = 512;
3572
3573 // Walk predecessors of StopBB to find blocks that can reach it. Only
3574 // convergent calls on a cycle with StopBB matter - a convergent call on a
3575 // path to function exit cannot have its dynamic instance changed by
3576 // threading.
3577 SmallPtrSet<BasicBlock *, 8> CanReachStop;
3578 SmallPtrSet<BasicBlock *, 8> BlocksWithUncontrolledConvergentCalls;
3580 for (BasicBlock *Pred : predecessors(StopBB))
3581 Worklist.push_back(Pred);
3582
3583 // Cache blocks with relevant calls while building CanReachStop. This keeps
3584 // the instruction scan bounded without a separate block limit.
3585 unsigned NumScannedInstructions = 0;
3586 while (!Worklist.empty()) {
3587 BasicBlock *BB = Worklist.pop_back_val();
3588 if (BB == StopBB)
3589 continue;
3590 if (!CanReachStop.insert(BB).second)
3591 continue;
3592
3593 for (Instruction &I : *BB) {
3594 if (++NumScannedInstructions > MaxInstructionsToScan)
3595 return true;
3596 auto *CB = dyn_cast<CallBase>(&I);
3597 if (CB && isUncontrolledConvergentCall(CB)) {
3598 BlocksWithUncontrolledConvergentCalls.insert(BB);
3599 break;
3600 }
3601 }
3602
3603 append_range(Worklist, predecessors(BB));
3604 }
3605
3606 if (!CanReachStop.contains(From))
3607 return false;
3608
3610 Worklist.push_back(From);
3611
3612 while (!Worklist.empty()) {
3613 BasicBlock *BB = Worklist.pop_back_val();
3614 if (BB == StopBB || !CanReachStop.contains(BB))
3615 continue;
3616
3617 if (!Visited.insert(BB).second)
3618 continue;
3619
3620 if (BlocksWithUncontrolledConvergentCalls.contains(BB))
3621 return true;
3622
3623 append_range(Worklist, successors(BB));
3624 }
3625
3626 return false;
3627}
3628
3629/// If we have a conditional branch on something for which we know the constant
3630/// value in predecessors (e.g. a phi node in the current block), thread edges
3631/// from the predecessor to their ultimate destination.
3634 AssumptionCache *AC, const DataLayout &DL) {
3636 BasicBlock *BB = BI->getParent();
3637 Value *Cond = BI->getCondition();
3639 if (PN && PN->getParent() == BB) {
3640 // Degenerate case of a single entry PHI.
3641 if (PN->getNumIncomingValues() == 1) {
3643 return true;
3644 }
3645
3646 for (Use &U : PN->incoming_values())
3647 if (auto *CB = dyn_cast<ConstantInt>(U))
3648 KnownValues[CB].insert(PN->getIncomingBlock(U));
3649 } else {
3650 for (BasicBlock *Pred : predecessors(BB)) {
3651 if (ConstantInt *CB = getKnownValueOnEdge(Cond, Pred, BB))
3652 KnownValues[CB].insert(Pred);
3653 }
3654 }
3655
3656 if (KnownValues.empty())
3657 return false;
3658
3659 // Now we know that this block has multiple preds and two succs.
3660 // Check that the block is small enough and record which non-local blocks use
3661 // values defined in the block.
3662
3663 BlocksSet NonLocalUseBlocks;
3664 BlocksSet ReachesNonLocalUseBlocks;
3665 if (!blockIsSimpleEnoughToThreadThrough(BB, NonLocalUseBlocks))
3666 return false;
3667
3668 // Jump-threading can only be done to destinations where no values defined
3669 // in BB are live.
3670
3671 // Quickly check if both destinations have uses. If so, jump-threading cannot
3672 // be done.
3673 if (NonLocalUseBlocks.contains(BI->getSuccessor(0)) &&
3674 NonLocalUseBlocks.contains(BI->getSuccessor(1)))
3675 return false;
3676
3677 // Search backward from NonLocalUseBlocks to find which blocks
3678 // reach non-local uses.
3679 for (BasicBlock *UseBB : NonLocalUseBlocks)
3680 // Give up if too many blocks are searched.
3681 if (!findReaching(UseBB, BB, ReachesNonLocalUseBlocks))
3682 return false;
3683
3684 for (const auto &Pair : KnownValues) {
3685 ConstantInt *CB = Pair.first;
3686 ArrayRef<BasicBlock *> PredBBs = Pair.second.getArrayRef();
3687 BasicBlock *RealDest = BI->getSuccessor(!CB->getZExtValue());
3688
3689 // Okay, we now know that all edges from PredBB should be revectored to
3690 // branch to RealDest.
3691 if (RealDest == BB)
3692 continue; // Skip self loops.
3693
3694 // Skip if the predecessor's terminator is an indirect branch.
3695 if (any_of(PredBBs, [](BasicBlock *PredBB) {
3696 return isa<IndirectBrInst>(PredBB->getTerminator());
3697 }))
3698 continue;
3699
3700 // Only revector to RealDest if no values defined in BB are live.
3701 if (ReachesNonLocalUseBlocks.contains(RealDest))
3702 continue;
3703
3704 // Threading through a branch can bypass a reconvergence point. If the
3705 // destination can execute an uncontrolled convergent operation before
3706 // returning to this block, this may change the dynamic instance of that
3707 // operation.
3708 if (TTI.hasBranchDivergence(BB->getParent()) &&
3710 continue;
3711
3712 LLVM_DEBUG({
3713 dbgs() << "Condition " << *Cond << " in " << BB->getName()
3714 << " has value " << *Pair.first << " in predecessors:\n";
3715 for (const BasicBlock *PredBB : Pair.second)
3716 dbgs() << " " << PredBB->getName() << "\n";
3717 dbgs() << "Threading to destination " << RealDest->getName() << ".\n";
3718 });
3719
3720 // Split the predecessors we are threading into a new edge block. We'll
3721 // clone the instructions into this block, and then redirect it to RealDest.
3722 BasicBlock *EdgeBB = SplitBlockPredecessors(BB, PredBBs, ".critedge", DTU);
3723 if (!EdgeBB)
3724 continue;
3725
3726 // TODO: These just exist to reduce test diff, we can drop them if we like.
3727 EdgeBB->setName(RealDest->getName() + ".critedge");
3728 EdgeBB->moveBefore(RealDest);
3729
3730 // Update PHI nodes.
3731 addPredecessorToBlock(RealDest, EdgeBB, BB);
3732
3733 // BB may have instructions that are being threaded over. Clone these
3734 // instructions into EdgeBB. We know that there will be no uses of the
3735 // cloned instructions outside of EdgeBB.
3736 BasicBlock::iterator InsertPt = EdgeBB->getFirstInsertionPt();
3737 ValueToValueMapTy TranslateMap; // Track translated values.
3738 TranslateMap[Cond] = CB;
3739
3740 // RemoveDIs: track instructions that we optimise away while folding, so
3741 // that we can copy DbgVariableRecords from them later.
3742 BasicBlock::iterator SrcDbgCursor = BB->begin();
3743 for (BasicBlock::iterator BBI = BB->begin(); &*BBI != BI; ++BBI) {
3744 if (PHINode *PN = dyn_cast<PHINode>(BBI)) {
3745 TranslateMap[PN] = PN->getIncomingValueForBlock(EdgeBB);
3746 continue;
3747 }
3748 // Clone the instruction.
3749 Instruction *N = BBI->clone();
3750 // Insert the new instruction into its new home.
3751 N->insertInto(EdgeBB, InsertPt);
3752
3753 if (BBI->hasName())
3754 N->setName(BBI->getName() + ".c");
3755
3756 // Update operands due to translation.
3757 // Key Instructions: Remap all the atom groups.
3758 if (const DebugLoc &DL = BBI->getDebugLoc())
3759 mapAtomInstance(DL, TranslateMap);
3760 RemapInstruction(N, TranslateMap,
3762
3763 // Check for trivial simplification.
3764 if (Value *V = simplifyInstruction(N, {DL, nullptr, nullptr, AC})) {
3765 if (!BBI->use_empty())
3766 TranslateMap[&*BBI] = V;
3767 if (!N->mayHaveSideEffects()) {
3768 N->eraseFromParent(); // Instruction folded away, don't need actual
3769 // inst
3770 N = nullptr;
3771 }
3772 } else {
3773 if (!BBI->use_empty())
3774 TranslateMap[&*BBI] = N;
3775 }
3776 if (N) {
3777 // Copy all debug-info attached to instructions from the last we
3778 // successfully clone, up to this instruction (they might have been
3779 // folded away).
3780 for (; SrcDbgCursor != BBI; ++SrcDbgCursor)
3781 N->cloneDebugInfoFrom(&*SrcDbgCursor);
3782 SrcDbgCursor = std::next(BBI);
3783 // Clone debug-info on this instruction too.
3784 N->cloneDebugInfoFrom(&*BBI);
3785
3786 // Register the new instruction with the assumption cache if necessary.
3787 if (auto *Assume = dyn_cast<AssumeInst>(N))
3788 if (AC)
3789 AC->registerAssumption(Assume);
3790 }
3791 }
3792
3793 for (; &*SrcDbgCursor != BI; ++SrcDbgCursor)
3794 InsertPt->cloneDebugInfoFrom(&*SrcDbgCursor);
3795 InsertPt->cloneDebugInfoFrom(BI);
3796
3797 BB->removePredecessor(EdgeBB);
3798 UncondBrInst *EdgeBI = cast<UncondBrInst>(EdgeBB->getTerminator());
3799 EdgeBI->setSuccessor(0, RealDest);
3800 EdgeBI->setDebugLoc(BI->getDebugLoc());
3801
3802 if (DTU) {
3804 Updates.push_back({DominatorTree::Delete, EdgeBB, BB});
3805 Updates.push_back({DominatorTree::Insert, EdgeBB, RealDest});
3806 DTU->applyUpdates(Updates);
3807 }
3808
3809 // For simplicity, we created a separate basic block for the edge. Merge
3810 // it back into the predecessor if possible. This not only avoids
3811 // unnecessary SimplifyCFG iterations, but also makes sure that we don't
3812 // bypass the check for trivial cycles above.
3813 MergeBlockIntoPredecessor(EdgeBB, DTU);
3814
3815 // Signal repeat, simplifying any other constants.
3816 return std::nullopt;
3817 }
3818
3819 return false;
3820}
3821
3822bool SimplifyCFGOpt::foldCondBranchOnValueKnownInPredecessor(CondBrInst *BI) {
3823 // Note: If BB is a loop header then there is a risk that threading introduces
3824 // a non-canonical loop by moving a back edge. So we avoid this optimization
3825 // for loop headers if NeedCanonicalLoop is set.
3826 if (Options.NeedCanonicalLoop && is_contained(LoopHeaders, BI->getParent()))
3827 return false;
3828
3829 std::optional<bool> Result;
3830 bool EverChanged = false;
3831 do {
3832 // Note that None means "we changed things, but recurse further."
3834 Options.AC, DL);
3835 EverChanged |= Result == std::nullopt || *Result;
3836 } while (Result == std::nullopt);
3837 return EverChanged;
3838}
3839
3840/// Given a BB that starts with the specified two-entry PHI node,
3841/// see if we can eliminate it.
3844 const DataLayout &DL,
3845 bool SpeculateUnpredictables) {
3846 // Ok, this is a two entry PHI node. Check to see if this is a simple "if
3847 // statement", which has a very simple dominance structure. Basically, we
3848 // are trying to find the condition that is being branched on, which
3849 // subsequently causes this merge to happen. We really want control
3850 // dependence information for this check, but simplifycfg can't keep it up
3851 // to date, and this catches most of the cases we care about anyway.
3852 BasicBlock *BB = PN->getParent();
3853
3854 BasicBlock *IfTrue, *IfFalse;
3855 CondBrInst *DomBI = GetIfCondition(BB, IfTrue, IfFalse);
3856 if (!DomBI)
3857 return false;
3858 Value *IfCond = DomBI->getCondition();
3859 // Don't bother if the branch will be constant folded trivially.
3860 if (isa<ConstantInt>(IfCond))
3861 return false;
3862
3863 BasicBlock *DomBlock = DomBI->getParent();
3865 llvm::copy_if(PN->blocks(), std::back_inserter(IfBlocks),
3866 [](BasicBlock *IfBlock) {
3867 return isa<UncondBrInst>(IfBlock->getTerminator());
3868 });
3869 assert((IfBlocks.size() == 1 || IfBlocks.size() == 2) &&
3870 "Will have either one or two blocks to speculate.");
3871
3872 // If the branch is non-unpredictable, see if we either predictably jump to
3873 // the merge bb (if we have only a single 'then' block), or if we predictably
3874 // jump to one specific 'then' block (if we have two of them).
3875 // It isn't beneficial to speculatively execute the code
3876 // from the block that we know is predictably not entered.
3877 bool IsUnpredictable = DomBI->getMetadata(LLVMContext::MD_unpredictable);
3878 if (!IsUnpredictable) {
3879 uint64_t TWeight, FWeight;
3880 if (extractBranchWeights(*DomBI, TWeight, FWeight) &&
3881 (TWeight + FWeight) != 0) {
3882 BranchProbability BITrueProb =
3883 BranchProbability::getBranchProbability(TWeight, TWeight + FWeight);
3884 BranchProbability Likely = TTI.getPredictableBranchThreshold();
3885 BranchProbability BIFalseProb = BITrueProb.getCompl();
3886 if (IfBlocks.size() == 1) {
3887 BranchProbability BIBBProb =
3888 DomBI->getSuccessor(0) == BB ? BITrueProb : BIFalseProb;
3889 if (BIBBProb >= Likely)
3890 return false;
3891 } else {
3892 if (BITrueProb >= Likely || BIFalseProb >= Likely)
3893 return false;
3894 }
3895 }
3896 }
3897
3898 // Don't try to fold an unreachable block. For example, the phi node itself
3899 // can't be the candidate if-condition for a select that we want to form.
3900 if (auto *IfCondPhiInst = dyn_cast<PHINode>(IfCond))
3901 if (IfCondPhiInst->getParent() == BB)
3902 return false;
3903
3904 // Okay, we found that we can merge this two-entry phi node into a select.
3905 // Doing so would require us to fold *all* two entry phi nodes in this block.
3906 // At some point this becomes non-profitable (particularly if the target
3907 // doesn't support cmov's). Only do this transformation if there are two or
3908 // fewer PHI nodes in this block.
3909 unsigned NumPhis = 0;
3910 for (BasicBlock::iterator I = BB->begin(); isa<PHINode>(I); ++NumPhis, ++I)
3911 if (NumPhis > 2)
3912 return false;
3913
3914 // Loop over the PHI's seeing if we can promote them all to select
3915 // instructions. While we are at it, keep track of the instructions
3916 // that need to be moved to the dominating block.
3917 SmallPtrSet<Instruction *, 4> AggressiveInsts;
3918 SmallPtrSet<Instruction *, 2> ZeroCostInstructions;
3919 InstructionCost Cost = 0;
3920 InstructionCost Budget =
3922 if (SpeculateUnpredictables && IsUnpredictable)
3923 Budget += TTI.getBranchMispredictPenalty();
3924
3925 bool Changed = false;
3926 for (BasicBlock::iterator II = BB->begin(); isa<PHINode>(II);) {
3927 PHINode *PN = cast<PHINode>(II++);
3928 if (Value *V = simplifyInstruction(PN, {DL, PN})) {
3929 PN->replaceAllUsesWith(V);
3930 PN->eraseFromParent();
3931 Changed = true;
3932 continue;
3933 }
3934
3935 if (!dominatesMergePoint(PN->getIncomingValue(0), BB, DomBI,
3936 AggressiveInsts, Cost, Budget, TTI, AC,
3937 ZeroCostInstructions) ||
3938 !dominatesMergePoint(PN->getIncomingValue(1), BB, DomBI,
3939 AggressiveInsts, Cost, Budget, TTI, AC,
3940 ZeroCostInstructions))
3941 return Changed;
3942 }
3943
3944 // If we folded the first phi, PN dangles at this point. Refresh it. If
3945 // we ran out of PHIs then we simplified them all.
3946 PN = dyn_cast<PHINode>(BB->begin());
3947 if (!PN)
3948 return true;
3949
3950 // Don't fold i1 branches on PHIs which contain binary operators or
3951 // (possibly inverted) select form of or/ands if their parameters are
3952 // an equality test.
3953 auto IsBinOpOrAndEq = [](Value *V) {
3954 CmpPredicate Pred;
3955 if (match(V, m_CombineOr(
3957 m_BinOp(m_Cmp(Pred, m_Value(), m_Value()), m_Value()),
3958 m_BinOp(m_Value(), m_Cmp(Pred, m_Value(), m_Value()))),
3960 m_Cmp(Pred, m_Value(), m_Value()))))) {
3961 return CmpInst::isEquality(Pred);
3962 }
3963 return false;
3964 };
3965 if (PN->getType()->isIntegerTy(1) &&
3966 (IsBinOpOrAndEq(PN->getIncomingValue(0)) ||
3967 IsBinOpOrAndEq(PN->getIncomingValue(1)) || IsBinOpOrAndEq(IfCond)))
3968 return Changed;
3969
3970 // If all PHI nodes are promotable, check to make sure that all instructions
3971 // in the predecessor blocks can be promoted as well. If not, we won't be able
3972 // to get rid of the control flow, so it's not worth promoting to select
3973 // instructions.
3974 for (BasicBlock *IfBlock : IfBlocks)
3975 for (BasicBlock::iterator I = IfBlock->begin(); !I->isTerminator(); ++I)
3976 if (!AggressiveInsts.count(&*I) && !I->isDebugOrPseudoInst()) {
3977 // This is not an aggressive instruction that we can promote.
3978 // Because of this, we won't be able to get rid of the control flow, so
3979 // the xform is not worth it.
3980 return Changed;
3981 }
3982
3983 // If either of the blocks has it's address taken, we can't do this fold.
3984 if (any_of(IfBlocks,
3985 [](BasicBlock *IfBlock) { return IfBlock->hasAddressTaken(); }))
3986 return Changed;
3987
3988 LLVM_DEBUG(dbgs() << "FOUND IF CONDITION! " << *IfCond;
3989 if (IsUnpredictable) dbgs() << " (unpredictable)";
3990 dbgs() << " T: " << IfTrue->getName()
3991 << " F: " << IfFalse->getName() << "\n");
3992
3993 // If we can still promote the PHI nodes after this gauntlet of tests,
3994 // do all of the PHI's now.
3995
3996 // Move all 'aggressive' instructions, which are defined in the
3997 // conditional parts of the if's up to the dominating block.
3998 for (BasicBlock *IfBlock : IfBlocks)
3999 hoistAllInstructionsInto(DomBlock, DomBI, IfBlock);
4000
4001 IRBuilder<NoFolder> Builder(DomBI);
4002 // Propagate fast-math-flags from phi nodes to replacement selects.
4003 while (PHINode *PN = dyn_cast<PHINode>(BB->begin())) {
4004 // Change the PHI node into a select instruction.
4005 Value *TrueVal = PN->getIncomingValueForBlock(IfTrue);
4006 Value *FalseVal = PN->getIncomingValueForBlock(IfFalse);
4007
4008 Value *Sel = Builder.CreateSelectFMF(IfCond, TrueVal, FalseVal,
4009 isa<FPMathOperator>(PN) ? PN : nullptr,
4010 "", DomBI);
4011 PN->replaceAllUsesWith(Sel);
4012 Sel->takeName(PN);
4013 PN->eraseFromParent();
4014 }
4015
4016 // At this point, all IfBlocks are empty, so our if statement
4017 // has been flattened. Change DomBlock to jump directly to our new block to
4018 // avoid other simplifycfg's kicking in on the diamond.
4019 Builder.CreateBr(BB);
4020
4022 if (DTU) {
4023 Updates.push_back({DominatorTree::Insert, DomBlock, BB});
4024 for (auto *Successor : successors(DomBlock))
4025 Updates.push_back({DominatorTree::Delete, DomBlock, Successor});
4026 }
4027
4028 DomBI->eraseFromParent();
4029 if (DTU)
4030 DTU->applyUpdates(Updates);
4031
4032 return true;
4033}
4034
4037 Value *RHS, const Twine &Name = "") {
4038 // Try to relax logical op to binary op.
4039 if (impliesPoison(RHS, LHS))
4040 return Builder.CreateBinOp(Opc, LHS, RHS, Name);
4041 if (Opc == Instruction::And)
4042 return Builder.CreateLogicalAnd(LHS, RHS, Name);
4043 if (Opc == Instruction::Or)
4044 return Builder.CreateLogicalOr(LHS, RHS, Name);
4045 llvm_unreachable("Invalid logical opcode");
4046}
4047
4048/// Return true if either PBI or BI has branch weight available, and store
4049/// the weights in {Pred|Succ}{True|False}Weight. If one of PBI and BI does
4050/// not have branch weight, use 1:1 as its weight.
4052 uint64_t &PredTrueWeight,
4053 uint64_t &PredFalseWeight,
4054 uint64_t &SuccTrueWeight,
4055 uint64_t &SuccFalseWeight) {
4056 bool PredHasWeights =
4057 extractBranchWeights(*PBI, PredTrueWeight, PredFalseWeight);
4058 bool SuccHasWeights =
4059 extractBranchWeights(*BI, SuccTrueWeight, SuccFalseWeight);
4060 if (PredHasWeights || SuccHasWeights) {
4061 if (!PredHasWeights)
4062 PredTrueWeight = PredFalseWeight = 1;
4063 if (!SuccHasWeights)
4064 SuccTrueWeight = SuccFalseWeight = 1;
4065 return true;
4066 } else {
4067 return false;
4068 }
4069}
4070
4071/// Determine if the two branches share a common destination and deduce a glue
4072/// that joins the branches' conditions to arrive at the common destination if
4073/// that would be profitable.
4074static std::optional<std::tuple<BasicBlock *, Instruction::BinaryOps, bool>>
4076 const TargetTransformInfo *TTI) {
4077 assert(BI && PBI && "Both blocks must end with a conditional branches.");
4079 "PredBB must be a predecessor of BB.");
4080
4081 // We have the potential to fold the conditions together, but if the
4082 // predecessor branch is predictable, we may not want to merge them.
4083 uint64_t PTWeight, PFWeight;
4084 BranchProbability PBITrueProb, Likely;
4085 if (TTI && !PBI->getMetadata(LLVMContext::MD_unpredictable) &&
4086 extractBranchWeights(*PBI, PTWeight, PFWeight) &&
4087 (PTWeight + PFWeight) != 0) {
4088 PBITrueProb =
4089 BranchProbability::getBranchProbability(PTWeight, PTWeight + PFWeight);
4090 Likely = TTI->getPredictableBranchThreshold();
4091 }
4092
4093 if (PBI->getSuccessor(0) == BI->getSuccessor(0)) {
4094 // Speculate the 2nd condition unless the 1st is probably true.
4095 if (PBITrueProb.isUnknown() || PBITrueProb < Likely)
4096 return {{BI->getSuccessor(0), Instruction::Or, false}};
4097 } else if (PBI->getSuccessor(1) == BI->getSuccessor(1)) {
4098 // Speculate the 2nd condition unless the 1st is probably false.
4099 if (PBITrueProb.isUnknown() || PBITrueProb.getCompl() < Likely)
4100 return {{BI->getSuccessor(1), Instruction::And, false}};
4101 } else if (PBI->getSuccessor(0) == BI->getSuccessor(1)) {
4102 // Speculate the 2nd condition unless the 1st is probably true.
4103 if (PBITrueProb.isUnknown() || PBITrueProb < Likely)
4104 return {{BI->getSuccessor(1), Instruction::And, true}};
4105 } else if (PBI->getSuccessor(1) == BI->getSuccessor(0)) {
4106 // Speculate the 2nd condition unless the 1st is probably false.
4107 if (PBITrueProb.isUnknown() || PBITrueProb.getCompl() < Likely)
4108 return {{BI->getSuccessor(0), Instruction::Or, true}};
4109 }
4110 return std::nullopt;
4111}
4112
4114 DomTreeUpdater *DTU,
4115 MemorySSAUpdater *MSSAU,
4116 const TargetTransformInfo *TTI) {
4117 BasicBlock *BB = BI->getParent();
4118 BasicBlock *PredBlock = PBI->getParent();
4119
4120 // Determine if the two branches share a common destination.
4121 BasicBlock *CommonSucc;
4123 bool InvertPredCond;
4124 std::tie(CommonSucc, Opc, InvertPredCond) =
4126
4127 LLVM_DEBUG(dbgs() << "FOLDING BRANCH TO COMMON DEST:\n" << *PBI << *BB);
4128
4130 BB->getContext(), ConstantFolder{},
4132 // The builder is used to create instructions to eliminate the branch in
4133 // BB. If BB's terminator has !annotation metadata, add it to the new
4134 // instructions.
4135 I->copyMetadata(*BB->getTerminator(), LLVMContext::MD_annotation);
4136 }));
4137 Builder.SetInsertPoint(PBI);
4138
4139 // If we need to invert the condition in the pred block to match, do so now.
4140 if (InvertPredCond) {
4141 InvertBranch(PBI, Builder);
4142 }
4143
4144 BasicBlock *UniqueSucc =
4145 PBI->getSuccessor(0) == BB ? BI->getSuccessor(0) : BI->getSuccessor(1);
4146
4147 // Before cloning instructions, notify the successor basic block that it
4148 // is about to have a new predecessor. This will update PHI nodes,
4149 // which will allow us to update live-out uses of bonus instructions.
4150 addPredecessorToBlock(UniqueSucc, PredBlock, BB, MSSAU);
4151
4152 // Try to update branch weights.
4153 uint64_t PredTrueWeight, PredFalseWeight, SuccTrueWeight, SuccFalseWeight;
4154 SmallVector<uint64_t, 2> MDWeights;
4155 if (extractPredSuccWeights(PBI, BI, PredTrueWeight, PredFalseWeight,
4156 SuccTrueWeight, SuccFalseWeight)) {
4157
4158 if (PBI->getSuccessor(0) == BB) {
4159 // PBI: br i1 %x, BB, FalseDest
4160 // BI: br i1 %y, UniqueSucc, FalseDest
4161 // TrueWeight is TrueWeight for PBI * TrueWeight for BI.
4162 MDWeights.push_back(PredTrueWeight * SuccTrueWeight);
4163 // FalseWeight is FalseWeight for PBI * TotalWeight for BI +
4164 // TrueWeight for PBI * FalseWeight for BI.
4165 // We assume that total weights of a CondBrInst can fit into 32 bits.
4166 // Therefore, we will not have overflow using 64-bit arithmetic.
4167 MDWeights.push_back(PredFalseWeight * (SuccFalseWeight + SuccTrueWeight) +
4168 PredTrueWeight * SuccFalseWeight);
4169 } else {
4170 // PBI: br i1 %x, TrueDest, BB
4171 // BI: br i1 %y, TrueDest, UniqueSucc
4172 // TrueWeight is TrueWeight for PBI * TotalWeight for BI +
4173 // FalseWeight for PBI * TrueWeight for BI.
4174 MDWeights.push_back(PredTrueWeight * (SuccFalseWeight + SuccTrueWeight) +
4175 PredFalseWeight * SuccTrueWeight);
4176 // FalseWeight is FalseWeight for PBI * FalseWeight for BI.
4177 MDWeights.push_back(PredFalseWeight * SuccFalseWeight);
4178 }
4179
4180 setFittedBranchWeights(*PBI, MDWeights, /*IsExpected=*/false,
4181 /*ElideAllZero=*/true);
4182
4183 // TODO: If BB is reachable from all paths through PredBlock, then we
4184 // could replace PBI's branch probabilities with BI's.
4185 } else
4186 PBI->setMetadata(LLVMContext::MD_prof, nullptr);
4187
4188 // Now, update the CFG.
4189 PBI->setSuccessor(PBI->getSuccessor(0) != BB, UniqueSucc);
4190
4191 if (DTU)
4192 DTU->applyUpdates({{DominatorTree::Insert, PredBlock, UniqueSucc},
4193 {DominatorTree::Delete, PredBlock, BB}});
4194
4195 // If BI was a loop latch, it may have had associated loop metadata.
4196 // We need to copy it to the new latch, that is, PBI.
4197 if (MDNode *LoopMD = BI->getMetadata(LLVMContext::MD_loop))
4198 PBI->setMetadata(LLVMContext::MD_loop, LoopMD);
4199
4200 ValueToValueMapTy VMap; // maps original values to cloned values
4202
4203 Module *M = BB->getModule();
4204
4205 PredBlock->getTerminator()->cloneDebugInfoFrom(BB->getTerminator());
4206 for (DbgVariableRecord &DVR :
4208 RemapDbgRecord(M, &DVR, VMap,
4210 }
4211
4212 // Now that the Cond was cloned into the predecessor basic block,
4213 // or/and the two conditions together.
4214 Value *BICond = VMap[BI->getCondition()];
4215 PBI->setCondition(
4216 createLogicalOp(Builder, Opc, PBI->getCondition(), BICond, "or.cond"));
4217 if (auto *SI = dyn_cast<SelectInst>(PBI->getCondition()))
4218 if (!MDWeights.empty()) {
4219 assert(isSelectInRoleOfConjunctionOrDisjunction(SI));
4220 setFittedBranchWeights(*SI, {MDWeights[0], MDWeights[1]},
4221 /*IsExpected=*/false, /*ElideAllZero=*/true);
4222 }
4223
4224 ++NumFoldBranchToCommonDest;
4225 return true;
4226}
4227
4228/// Return if an instruction's type or any of its operands' types are a vector
4229/// type.
4230static bool isVectorOp(Instruction &I) {
4231 return I.getType()->isVectorTy() || any_of(I.operands(), [](Use &U) {
4232 return U->getType()->isVectorTy();
4233 });
4234}
4235
4236/// If this basic block is simple enough, and if a predecessor branches to us
4237/// and one of our successors, fold the block into the predecessor and use
4238/// logical operations to pick the right destination.
4240 MemorySSAUpdater *MSSAU,
4241 const TargetTransformInfo *TTI,
4242 AssumptionCache *AC,
4243 unsigned BonusInstThreshold) {
4244 BasicBlock *BB = BI->getParent();
4248
4250
4252 Cond->getParent() != BB || !Cond->hasOneUse())
4253 return false;
4254
4255 // Finally, don't infinitely unroll conditional loops.
4256 if (is_contained(successors(BB), BB))
4257 return false;
4258
4259 // With which predecessors will we want to deal with?
4261 for (BasicBlock *PredBlock : predecessors(BB)) {
4262 CondBrInst *PBI = dyn_cast<CondBrInst>(PredBlock->getTerminator());
4263
4264 // Check that we have two conditional branches. If there is a PHI node in
4265 // the common successor, verify that the same value flows in from both
4266 // blocks.
4267 if (!PBI || !safeToMergeTerminators(BI, PBI))
4268 continue;
4269
4270 // Determine if the two branches share a common destination.
4271 BasicBlock *CommonSucc;
4273 bool InvertPredCond;
4274 if (auto Recipe = shouldFoldCondBranchesToCommonDestination(BI, PBI, TTI))
4275 std::tie(CommonSucc, Opc, InvertPredCond) = *Recipe;
4276 else
4277 continue;
4278
4279 // Check the cost of inserting the necessary logic before performing the
4280 // transformation.
4281 if (TTI) {
4282 Type *Ty = BI->getCondition()->getType();
4283 InstructionCost Cost = TTI->getArithmeticInstrCost(Opc, Ty, CostKind);
4284 if (InvertPredCond && (!PBI->getCondition()->hasOneUse() ||
4285 !isa<CmpInst>(PBI->getCondition())))
4286 Cost += TTI->getArithmeticInstrCost(Instruction::Xor, Ty, CostKind);
4287
4289 continue;
4290 }
4291
4292 // Ok, we do want to deal with this predecessor. Record it.
4293 Preds.emplace_back(PredBlock);
4294 }
4295
4296 // If there aren't any predecessors into which we can fold,
4297 // don't bother checking the cost.
4298 if (Preds.empty())
4299 return false;
4300
4301 // Only allow this transformation if computing the condition doesn't involve
4302 // too many instructions and these involved instructions can be executed
4303 // unconditionally. We denote all involved instructions except the condition
4304 // as "bonus instructions", and only allow this transformation when the
4305 // number of the bonus instructions we'll need to create when cloning into
4306 // each predecessor does not exceed a certain threshold.
4307 unsigned NumBonusInsts = 0;
4308 bool SawVectorOp = false;
4309 const unsigned PredCount = Preds.size();
4310 // Speculated instructions will be inserted before the terminator of the
4311 // predecessor. Only handle the simple case of one predecessor.
4312 const Instruction *CtxI =
4313 PredCount == 1 ? Preds[0]->getTerminator() : nullptr;
4314 for (Instruction &I : *BB) {
4315 // Don't check the branch condition comparison itself.
4316 if (&I == Cond)
4317 continue;
4318 // Ignore the terminator.
4320 continue;
4321 // Pseudo probes aren't speculatable but can be dropped on fold.
4323 continue;
4324 // I must be safe to execute unconditionally.
4325 if (!isSafeToSpeculativelyExecute(&I, CtxI, AC))
4326 return false;
4327 SawVectorOp |= isVectorOp(I);
4328
4329 // Account for the cost of duplicating this instruction into each
4330 // predecessor. Ignore free instructions.
4331 if (!TTI || TTI->getInstructionCost(&I, CostKind) !=
4333 NumBonusInsts += PredCount;
4334
4335 // Early exits once we reach the limit.
4336 if (NumBonusInsts >
4337 BonusInstThreshold * BranchFoldToCommonDestVectorMultiplier)
4338 return false;
4339 }
4340
4341 auto IsBCSSAUse = [BB, &I](Use &U) {
4342 auto *UI = cast<Instruction>(U.getUser());
4343 if (auto *PN = dyn_cast<PHINode>(UI))
4344 return PN->getIncomingBlock(U) == BB;
4345 return UI->getParent() == BB && I.comesBefore(UI);
4346 };
4347
4348 // Does this instruction require rewriting of uses?
4349 if (!all_of(I.uses(), IsBCSSAUse))
4350 return false;
4351 }
4352 if (NumBonusInsts >
4353 BonusInstThreshold *
4354 (SawVectorOp ? BranchFoldToCommonDestVectorMultiplier : 1))
4355 return false;
4356
4357 // Ok, we have the budget. Perform the transformation.
4358 for (BasicBlock *PredBlock : Preds) {
4359 auto *PBI = cast<CondBrInst>(PredBlock->getTerminator());
4360 return performBranchToCommonDestFolding(BI, PBI, DTU, MSSAU, TTI);
4361 }
4362 return false;
4363}
4364
4365// If there is only one store in BB1 and BB2, return it, otherwise return
4366// nullptr.
4368 StoreInst *S = nullptr;
4369 for (auto *BB : {BB1, BB2}) {
4370 if (!BB)
4371 continue;
4372 for (auto &I : *BB)
4373 if (auto *SI = dyn_cast<StoreInst>(&I)) {
4374 if (S)
4375 // Multiple stores seen.
4376 return nullptr;
4377 else
4378 S = SI;
4379 }
4380 }
4381 return S;
4382}
4383
4385 Value *AlternativeV = nullptr) {
4386 // PHI is going to be a PHI node that allows the value V that is defined in
4387 // BB to be referenced in BB's only successor.
4388 //
4389 // If AlternativeV is nullptr, the only value we care about in PHI is V. It
4390 // doesn't matter to us what the other operand is (it'll never get used). We
4391 // could just create a new PHI with an undef incoming value, but that could
4392 // increase register pressure if EarlyCSE/InstCombine can't fold it with some
4393 // other PHI. So here we directly look for some PHI in BB's successor with V
4394 // as an incoming operand. If we find one, we use it, else we create a new
4395 // one.
4396 //
4397 // If AlternativeV is not nullptr, we care about both incoming values in PHI.
4398 // PHI must be exactly: phi <ty> [ %BB, %V ], [ %OtherBB, %AlternativeV]
4399 // where OtherBB is the single other predecessor of BB's only successor.
4400 PHINode *PHI = nullptr;
4401 BasicBlock *Succ = BB->getSingleSuccessor();
4402
4403 for (auto I = Succ->begin(); isa<PHINode>(I); ++I)
4404 if (cast<PHINode>(I)->getIncomingValueForBlock(BB) == V) {
4405 PHI = cast<PHINode>(I);
4406 if (!AlternativeV)
4407 break;
4408
4409 assert(Succ->hasNPredecessors(2));
4410 auto PredI = pred_begin(Succ);
4411 BasicBlock *OtherPredBB = *PredI == BB ? *++PredI : *PredI;
4412 if (PHI->getIncomingValueForBlock(OtherPredBB) == AlternativeV)
4413 break;
4414 PHI = nullptr;
4415 }
4416 if (PHI)
4417 return PHI;
4418
4419 // If V is not an instruction defined in BB, just return it.
4420 if (!AlternativeV &&
4421 (!isa<Instruction>(V) || cast<Instruction>(V)->getParent() != BB))
4422 return V;
4423
4424 PHI = PHINode::Create(V->getType(), 2, "simplifycfg.merge");
4425 PHI->insertBefore(Succ->begin());
4426 PHI->addIncoming(V, BB);
4427 for (BasicBlock *PredBB : predecessors(Succ))
4428 if (PredBB != BB)
4429 PHI->addIncoming(
4430 AlternativeV ? AlternativeV : PoisonValue::get(V->getType()), PredBB);
4431 return PHI;
4432}
4433
4435 BasicBlock *PTB, BasicBlock *PFB, BasicBlock *QTB, BasicBlock *QFB,
4436 BasicBlock *PostBB, Value *Address, bool InvertPCond, bool InvertQCond,
4437 DomTreeUpdater *DTU, const DataLayout &DL, const TargetTransformInfo &TTI) {
4438 // For every pointer, there must be exactly two stores, one coming from
4439 // PTB or PFB, and the other from QTB or QFB. We don't support more than one
4440 // store (to any address) in PTB,PFB or QTB,QFB.
4441 // FIXME: We could relax this restriction with a bit more work and performance
4442 // testing.
4443 StoreInst *PStore = findUniqueStoreInBlocks(PTB, PFB);
4444 StoreInst *QStore = findUniqueStoreInBlocks(QTB, QFB);
4445 if (!PStore || !QStore)
4446 return false;
4447
4448 // Now check the stores are compatible.
4449 if (!QStore->isUnordered() || !PStore->isUnordered() ||
4450 PStore->getOrdering() != QStore->getOrdering() ||
4451 PStore->getSyncScopeID() != QStore->getSyncScopeID() ||
4452 PStore->getValueOperand()->getType() !=
4453 QStore->getValueOperand()->getType())
4454 return false;
4455
4456 // Check that sinking the store won't cause program behavior changes. Sinking
4457 // the store out of the Q blocks won't change any behavior as we're sinking
4458 // from a block to its unconditional successor. But we're moving a store from
4459 // the P blocks down through the middle block (QBI) and past both QFB and QTB.
4460 // So we need to check that there are no aliasing loads or stores in
4461 // QBI, QTB and QFB. We also need to check there are no conflicting memory
4462 // operations between PStore and the end of its parent block.
4463 //
4464 // The ideal way to do this is to query AliasAnalysis, but we don't
4465 // preserve AA currently so that is dangerous. Be super safe and just
4466 // check there are no other memory operations at all.
4467 for (auto &I : *QFB->getSinglePredecessor())
4468 if (I.mayReadOrWriteMemory())
4469 return false;
4470 for (auto &I : *QFB)
4471 if (&I != QStore && I.mayReadOrWriteMemory())
4472 return false;
4473 if (QTB)
4474 for (auto &I : *QTB)
4475 if (&I != QStore && I.mayReadOrWriteMemory())
4476 return false;
4477 for (auto I = BasicBlock::iterator(PStore), E = PStore->getParent()->end();
4478 I != E; ++I)
4479 if (&*I != PStore && I->mayReadOrWriteMemory())
4480 return false;
4481
4482 // If we're not in aggressive mode, we only optimize if we have some
4483 // confidence that by optimizing we'll allow P and/or Q to be if-converted.
4484 auto IsWorthwhile = [&](BasicBlock *BB, ArrayRef<StoreInst *> FreeStores) {
4485 if (!BB)
4486 return true;
4487 // Heuristic: if the block can be if-converted/phi-folded and the
4488 // instructions inside are all cheap (arithmetic/GEPs), it's worthwhile to
4489 // thread this store.
4490 InstructionCost Cost = 0;
4491 InstructionCost Budget =
4493 for (auto &I : *BB) {
4494 // Consider terminator instruction to be free.
4495 if (I.isTerminator())
4496 continue;
4497 // If this is one the stores that we want to speculate out of this BB,
4498 // then don't count it's cost, consider it to be free.
4499 if (auto *S = dyn_cast<StoreInst>(&I))
4500 if (llvm::find(FreeStores, S))
4501 continue;
4502 // Else, we have a white-list of instructions that we are ak speculating.
4504 return false; // Not in white-list - not worthwhile folding.
4505 // And finally, if this is a non-free instruction that we are okay
4506 // speculating, ensure that we consider the speculation budget.
4507 Cost +=
4508 TTI.getInstructionCost(&I, TargetTransformInfo::TCK_SizeAndLatency);
4509 if (Cost > Budget)
4510 return false; // Eagerly refuse to fold as soon as we're out of budget.
4511 }
4512 assert(Cost <= Budget &&
4513 "When we run out of budget we will eagerly return from within the "
4514 "per-instruction loop.");
4515 return true;
4516 };
4517
4518 const std::array<StoreInst *, 2> FreeStores = {PStore, QStore};
4520 (!IsWorthwhile(PTB, FreeStores) || !IsWorthwhile(PFB, FreeStores) ||
4521 !IsWorthwhile(QTB, FreeStores) || !IsWorthwhile(QFB, FreeStores)))
4522 return false;
4523
4524 // If PostBB has more than two predecessors, we need to split it so we can
4525 // sink the store.
4526 if (std::next(pred_begin(PostBB), 2) != pred_end(PostBB)) {
4527 // We know that QFB's only successor is PostBB. And QFB has a single
4528 // predecessor. If QTB exists, then its only successor is also PostBB.
4529 // If QTB does not exist, then QFB's only predecessor has a conditional
4530 // branch to QFB and PostBB.
4531 BasicBlock *TruePred = QTB ? QTB : QFB->getSinglePredecessor();
4532 BasicBlock *NewBB =
4533 SplitBlockPredecessors(PostBB, {QFB, TruePred}, "condstore.split", DTU);
4534 if (!NewBB)
4535 return false;
4536 PostBB = NewBB;
4537 }
4538
4539 // OK, we're going to sink the stores to PostBB. The store has to be
4540 // conditional though, so first create the predicate.
4541 CondBrInst *PBranch =
4543 CondBrInst *QBranch =
4545 Value *PCond = PBranch->getCondition();
4546 Value *QCond = QBranch->getCondition();
4547
4549 PStore->getParent());
4551 QStore->getParent(), PPHI);
4552
4553 BasicBlock::iterator PostBBFirst = PostBB->getFirstInsertionPt();
4554 IRBuilder<> QB(PostBB, PostBBFirst);
4555 QB.SetCurrentDebugLocation(PostBBFirst->getStableDebugLoc());
4556
4557 InvertPCond ^= (PStore->getParent() != PTB);
4558 InvertQCond ^= (QStore->getParent() != QTB);
4559 Value *PPred = InvertPCond ? QB.CreateNot(PCond) : PCond;
4560 Value *QPred = InvertQCond ? QB.CreateNot(QCond) : QCond;
4561
4562 Value *CombinedPred = QB.CreateOr(PPred, QPred);
4563
4564 BasicBlock::iterator InsertPt = QB.GetInsertPoint();
4565 auto *T = SplitBlockAndInsertIfThen(CombinedPred, InsertPt,
4566 /*Unreachable=*/false,
4567 /*BranchWeights=*/nullptr, DTU);
4568 if (hasBranchWeightMD(*PBranch) && hasBranchWeightMD(*QBranch)) {
4569 SmallVector<uint32_t, 2> PWeights, QWeights;
4570 extractBranchWeights(*PBranch, PWeights);
4571 extractBranchWeights(*QBranch, QWeights);
4572 if (InvertPCond)
4573 std::swap(PWeights[0], PWeights[1]);
4574 if (InvertQCond)
4575 std::swap(QWeights[0], QWeights[1]);
4576 auto CombinedWeights = getDisjunctionWeights(PWeights, QWeights);
4578 {CombinedWeights[0], CombinedWeights[1]},
4579 /*IsExpected=*/false, /*ElideAllZero=*/true);
4580 }
4581
4582 QB.SetInsertPoint(T);
4583 StoreInst *SI = cast<StoreInst>(QB.CreateStore(QPHI, Address));
4584 combineMetadataForCSE(QStore, PStore, true);
4585 SI->copyMetadata(*QStore);
4586 // Update any dbg.assign intrinsics to track the merged value (QPHI) instead
4587 // of the original constant values, likely making these identical.
4588 for (auto *DbgAssign : at::getDVRAssignmentMarkers(SI)) {
4589 if (llvm::is_contained(DbgAssign->location_ops(),
4590 PStore->getValueOperand()))
4591 DbgAssign->replaceVariableLocationOp(PStore->getValueOperand(), QPHI);
4592 if (llvm::is_contained(DbgAssign->location_ops(),
4593 QStore->getValueOperand()))
4594 DbgAssign->replaceVariableLocationOp(QStore->getValueOperand(), QPHI);
4595 }
4596
4597 // Choose the minimum alignment. If we could prove both stores execute, we
4598 // could use biggest one. In this case, though, we only know that one of the
4599 // stores executes. And we don't know it's safe to take the alignment from a
4600 // store that doesn't execute.
4601 SI->setAlignment(std::min(PStore->getAlign(), QStore->getAlign()));
4602
4603 if (QStore->isAtomic())
4604 SI->setAtomic(QStore->getOrdering(), QStore->getSyncScopeID());
4605
4606 QStore->eraseFromParent();
4607 PStore->eraseFromParent();
4608
4609 return true;
4610}
4611
4613 DomTreeUpdater *DTU, const DataLayout &DL,
4614 const TargetTransformInfo &TTI) {
4615 // The intention here is to find diamonds or triangles (see below) where each
4616 // conditional block contains a store to the same address. Both of these
4617 // stores are conditional, so they can't be unconditionally sunk. But it may
4618 // be profitable to speculatively sink the stores into one merged store at the
4619 // end, and predicate the merged store on the union of the two conditions of
4620 // PBI and QBI.
4621 //
4622 // This can reduce the number of stores executed if both of the conditions are
4623 // true, and can allow the blocks to become small enough to be if-converted.
4624 // This optimization will also chain, so that ladders of test-and-set
4625 // sequences can be if-converted away.
4626 //
4627 // We only deal with simple diamonds or triangles:
4628 //
4629 // PBI or PBI or a combination of the two
4630 // / \ | \
4631 // PTB PFB | PFB
4632 // \ / | /
4633 // QBI QBI
4634 // / \ | \
4635 // QTB QFB | QFB
4636 // \ / | /
4637 // PostBB PostBB
4638 //
4639 // We model triangles as a type of diamond with a nullptr "true" block.
4640 // Triangles are canonicalized so that the fallthrough edge is represented by
4641 // a true condition, as in the diagram above.
4642 BasicBlock *PTB = PBI->getSuccessor(0);
4643 BasicBlock *PFB = PBI->getSuccessor(1);
4644 BasicBlock *QTB = QBI->getSuccessor(0);
4645 BasicBlock *QFB = QBI->getSuccessor(1);
4646 BasicBlock *PostBB = QFB->getSingleSuccessor();
4647
4648 // Make sure we have a good guess for PostBB. If QTB's only successor is
4649 // QFB, then QFB is a better PostBB.
4650 if (QTB->getSingleSuccessor() == QFB)
4651 PostBB = QFB;
4652
4653 // If we couldn't find a good PostBB, stop.
4654 if (!PostBB)
4655 return false;
4656
4657 bool InvertPCond = false, InvertQCond = false;
4658 // Canonicalize fallthroughs to the true branches.
4659 if (PFB == QBI->getParent()) {
4660 std::swap(PFB, PTB);
4661 InvertPCond = true;
4662 }
4663 if (QFB == PostBB) {
4664 std::swap(QFB, QTB);
4665 InvertQCond = true;
4666 }
4667
4668 // From this point on we can assume PTB or QTB may be fallthroughs but PFB
4669 // and QFB may not. Model fallthroughs as a nullptr block.
4670 if (PTB == QBI->getParent())
4671 PTB = nullptr;
4672 if (QTB == PostBB)
4673 QTB = nullptr;
4674
4675 // Legality bailouts. We must have at least the non-fallthrough blocks and
4676 // the post-dominating block, and the non-fallthroughs must only have one
4677 // predecessor.
4678 auto HasOnePredAndOneSucc = [](BasicBlock *BB, BasicBlock *P, BasicBlock *S) {
4679 return BB->getSinglePredecessor() == P && BB->getSingleSuccessor() == S;
4680 };
4681 if (!HasOnePredAndOneSucc(PFB, PBI->getParent(), QBI->getParent()) ||
4682 !HasOnePredAndOneSucc(QFB, QBI->getParent(), PostBB))
4683 return false;
4684 if ((PTB && !HasOnePredAndOneSucc(PTB, PBI->getParent(), QBI->getParent())) ||
4685 (QTB && !HasOnePredAndOneSucc(QTB, QBI->getParent(), PostBB)))
4686 return false;
4687 if (!QBI->getParent()->hasNUses(2))
4688 return false;
4689
4690 // OK, this is a sequence of two diamonds or triangles.
4691 // Check if there are stores in PTB or PFB that are repeated in QTB or QFB.
4692 SmallPtrSet<Value *, 4> PStoreAddresses, QStoreAddresses;
4693 for (auto *BB : {PTB, PFB}) {
4694 if (!BB)
4695 continue;
4696 for (auto &I : *BB)
4698 PStoreAddresses.insert(SI->getPointerOperand());
4699 }
4700 for (auto *BB : {QTB, QFB}) {
4701 if (!BB)
4702 continue;
4703 for (auto &I : *BB)
4705 QStoreAddresses.insert(SI->getPointerOperand());
4706 }
4707
4708 set_intersect(PStoreAddresses, QStoreAddresses);
4709 // set_intersect mutates PStoreAddresses in place. Rename it here to make it
4710 // clear what it contains.
4711 auto &CommonAddresses = PStoreAddresses;
4712
4713 bool Changed = false;
4714 for (auto *Address : CommonAddresses)
4715 Changed |=
4716 mergeConditionalStoreToAddress(PTB, PFB, QTB, QFB, PostBB, Address,
4717 InvertPCond, InvertQCond, DTU, DL, TTI);
4718 return Changed;
4719}
4720
4721/// If the previous block ended with a widenable branch, determine if reusing
4722/// the target block is profitable and legal. This will have the effect of
4723/// "widening" PBI, but doesn't require us to reason about hosting safety.
4725 DomTreeUpdater *DTU) {
4726 // TODO: This can be generalized in two important ways:
4727 // 1) We can allow phi nodes in IfFalseBB and simply reuse all the input
4728 // values from the PBI edge.
4729 // 2) We can sink side effecting instructions into BI's fallthrough
4730 // successor provided they doesn't contribute to computation of
4731 // BI's condition.
4732 BasicBlock *IfTrueBB = PBI->getSuccessor(0);
4733 BasicBlock *IfFalseBB = PBI->getSuccessor(1);
4734 if (!isWidenableBranch(PBI) || IfTrueBB != BI->getParent() ||
4735 !BI->getParent()->getSinglePredecessor())
4736 return false;
4737 if (!IfFalseBB->phis().empty())
4738 return false; // TODO
4739 // This helps avoid infinite loop with SimplifyCondBranchToCondBranch which
4740 // may undo the transform done here.
4741 // TODO: There might be a more fine-grained solution to this.
4742 if (!llvm::succ_empty(IfFalseBB))
4743 return false;
4744 // Use lambda to lazily compute expensive condition after cheap ones.
4745 auto NoSideEffects = [](BasicBlock &BB) {
4746 return llvm::none_of(BB, [](const Instruction &I) {
4747 return I.mayWriteToMemory() || I.mayHaveSideEffects();
4748 });
4749 };
4750 if (BI->getSuccessor(1) != IfFalseBB && // no inf looping
4751 BI->getSuccessor(1)->getTerminatingDeoptimizeCall() && // profitability
4752 NoSideEffects(*BI->getParent())) {
4753 auto *OldSuccessor = BI->getSuccessor(1);
4754 OldSuccessor->removePredecessor(BI->getParent());
4755 BI->setSuccessor(1, IfFalseBB);
4756 if (DTU)
4757 DTU->applyUpdates(
4758 {{DominatorTree::Insert, BI->getParent(), IfFalseBB},
4759 {DominatorTree::Delete, BI->getParent(), OldSuccessor}});
4760 return true;
4761 }
4762 if (BI->getSuccessor(0) != IfFalseBB && // no inf looping
4763 BI->getSuccessor(0)->getTerminatingDeoptimizeCall() && // profitability
4764 NoSideEffects(*BI->getParent())) {
4765 auto *OldSuccessor = BI->getSuccessor(0);
4766 OldSuccessor->removePredecessor(BI->getParent());
4767 BI->setSuccessor(0, IfFalseBB);
4768 if (DTU)
4769 DTU->applyUpdates(
4770 {{DominatorTree::Insert, BI->getParent(), IfFalseBB},
4771 {DominatorTree::Delete, BI->getParent(), OldSuccessor}});
4772 return true;
4773 }
4774 return false;
4775}
4776
4777/// If we have a conditional branch as a predecessor of another block,
4778/// this function tries to simplify it. We know
4779/// that PBI and BI are both conditional branches, and BI is in one of the
4780/// successor blocks of PBI - PBI branches to BI.
4782 DomTreeUpdater *DTU,
4783 const DataLayout &DL,
4784 const TargetTransformInfo &TTI) {
4785 BasicBlock *BB = BI->getParent();
4786
4787 // If this block ends with a branch instruction, and if there is a
4788 // predecessor that ends on a branch of the same condition, make
4789 // this conditional branch redundant.
4790 if (PBI->getCondition() == BI->getCondition() &&
4791 PBI->getSuccessor(0) != PBI->getSuccessor(1)) {
4792 // Okay, the outcome of this conditional branch is statically
4793 // knowable. If this block had a single pred, handle specially, otherwise
4794 // foldCondBranchOnValueKnownInPredecessor() will handle it.
4795 if (BB->getSinglePredecessor()) {
4796 // Turn this into a branch on constant.
4797 bool CondIsTrue = PBI->getSuccessor(0) == BB;
4798 BI->setCondition(
4799 ConstantInt::get(Type::getInt1Ty(BB->getContext()), CondIsTrue));
4800 return true; // Nuke the branch on constant.
4801 }
4802 }
4803
4804 // If the previous block ended with a widenable branch, determine if reusing
4805 // the target block is profitable and legal. This will have the effect of
4806 // "widening" PBI, but doesn't require us to reason about hosting safety.
4807 if (tryWidenCondBranchToCondBranch(PBI, BI, DTU))
4808 return true;
4809
4810 // If both branches are conditional and both contain stores to the same
4811 // address, remove the stores from the conditionals and create a conditional
4812 // merged store at the end.
4813 if (MergeCondStores && mergeConditionalStores(PBI, BI, DTU, DL, TTI))
4814 return true;
4815
4816 // If this is a conditional branch in an empty block, and if any
4817 // predecessors are a conditional branch to one of our destinations,
4818 // fold the conditions into logical ops and one cond br.
4819
4820 // Ignore dbg intrinsics.
4821 if (&*BB->begin() != BI)
4822 return false;
4823
4824 int PBIOp, BIOp;
4825 if (PBI->getSuccessor(0) == BI->getSuccessor(0)) {
4826 PBIOp = 0;
4827 BIOp = 0;
4828 } else if (PBI->getSuccessor(0) == BI->getSuccessor(1)) {
4829 PBIOp = 0;
4830 BIOp = 1;
4831 } else if (PBI->getSuccessor(1) == BI->getSuccessor(0)) {
4832 PBIOp = 1;
4833 BIOp = 0;
4834 } else if (PBI->getSuccessor(1) == BI->getSuccessor(1)) {
4835 PBIOp = 1;
4836 BIOp = 1;
4837 } else {
4838 return false;
4839 }
4840
4841 // Check to make sure that the other destination of this branch
4842 // isn't BB itself. If so, this is an infinite loop that will
4843 // keep getting unwound.
4844 if (PBI->getSuccessor(PBIOp) == BB)
4845 return false;
4846
4847 // If predecessor's branch probability to BB is too low don't merge branches.
4848 SmallVector<uint32_t, 2> PredWeights;
4849 if (!PBI->getMetadata(LLVMContext::MD_unpredictable) &&
4850 extractBranchWeights(*PBI, PredWeights) &&
4851 (static_cast<uint64_t>(PredWeights[0]) + PredWeights[1]) != 0) {
4852
4854 PredWeights[PBIOp],
4855 static_cast<uint64_t>(PredWeights[0]) + PredWeights[1]);
4856
4857 BranchProbability Likely = TTI.getPredictableBranchThreshold();
4858 if (CommonDestProb >= Likely)
4859 return false;
4860 }
4861
4862 // Do not perform this transformation if it would require
4863 // insertion of a large number of select instructions. For targets
4864 // without predication/cmovs, this is a big pessimization.
4865
4866 BasicBlock *CommonDest = PBI->getSuccessor(PBIOp);
4867 BasicBlock *RemovedDest = PBI->getSuccessor(PBIOp ^ 1);
4868 unsigned NumPhis = 0;
4869 for (BasicBlock::iterator II = CommonDest->begin(); isa<PHINode>(II);
4870 ++II, ++NumPhis) {
4871 if (NumPhis > 2) // Disable this xform.
4872 return false;
4873 }
4874
4875 // Finally, if everything is ok, fold the branches to logical ops.
4876 BasicBlock *OtherDest = BI->getSuccessor(BIOp ^ 1);
4877
4878 LLVM_DEBUG(dbgs() << "FOLDING BRs:" << *PBI->getParent()
4879 << "AND: " << *BI->getParent());
4880
4882
4883 // If OtherDest *is* BB, then BB is a basic block with a single conditional
4884 // branch in it, where one edge (OtherDest) goes back to itself but the other
4885 // exits. We don't *know* that the program avoids the infinite loop
4886 // (even though that seems likely). If we do this xform naively, we'll end up
4887 // recursively unpeeling the loop. Since we know that (after the xform is
4888 // done) that the block *is* infinite if reached, we just make it an obviously
4889 // infinite loop with no cond branch.
4890 if (OtherDest == BB) {
4891 // Insert it at the end of the function, because it's either code,
4892 // or it won't matter if it's hot. :)
4893 BasicBlock *InfLoopBlock =
4894 BasicBlock::Create(BB->getContext(), "infloop", BB->getParent());
4895 UncondBrInst::Create(InfLoopBlock, InfLoopBlock);
4896 if (DTU)
4897 Updates.push_back({DominatorTree::Insert, InfLoopBlock, InfLoopBlock});
4898 OtherDest = InfLoopBlock;
4899 }
4900
4901 LLVM_DEBUG(dbgs() << *PBI->getParent()->getParent());
4902
4903 // BI may have other predecessors. Because of this, we leave
4904 // it alone, but modify PBI.
4905
4906 // Make sure we get to CommonDest on True&True directions.
4907 Value *PBICond = PBI->getCondition();
4908 IRBuilder<NoFolder> Builder(PBI);
4909 if (PBIOp)
4910 PBICond = Builder.CreateNot(PBICond, PBICond->getName() + ".not");
4911
4912 Value *BICond = BI->getCondition();
4913 if (BIOp)
4914 BICond = Builder.CreateNot(BICond, BICond->getName() + ".not");
4915
4916 // Merge the conditions.
4917 Value *Cond =
4918 createLogicalOp(Builder, Instruction::Or, PBICond, BICond, "brmerge");
4919
4920 // Modify PBI to branch on the new condition to the new dests.
4921 PBI->setCondition(Cond);
4922 PBI->setSuccessor(0, CommonDest);
4923 PBI->setSuccessor(1, OtherDest);
4924
4925 if (DTU) {
4926 Updates.push_back({DominatorTree::Insert, PBI->getParent(), OtherDest});
4927 Updates.push_back({DominatorTree::Delete, PBI->getParent(), RemovedDest});
4928
4929 DTU->applyUpdates(Updates);
4930 }
4931
4932 // Update branch weight for PBI.
4933 uint64_t PredTrueWeight, PredFalseWeight, SuccTrueWeight, SuccFalseWeight;
4934 uint64_t PredCommon, PredOther, SuccCommon, SuccOther;
4935 bool HasWeights =
4936 extractPredSuccWeights(PBI, BI, PredTrueWeight, PredFalseWeight,
4937 SuccTrueWeight, SuccFalseWeight);
4938 if (HasWeights) {
4939 PredCommon = PBIOp ? PredFalseWeight : PredTrueWeight;
4940 PredOther = PBIOp ? PredTrueWeight : PredFalseWeight;
4941 SuccCommon = BIOp ? SuccFalseWeight : SuccTrueWeight;
4942 SuccOther = BIOp ? SuccTrueWeight : SuccFalseWeight;
4943 // The weight to CommonDest should be PredCommon * SuccTotal +
4944 // PredOther * SuccCommon.
4945 // The weight to OtherDest should be PredOther * SuccOther.
4946 uint64_t NewWeights[2] = {PredCommon * (SuccCommon + SuccOther) +
4947 PredOther * SuccCommon,
4948 PredOther * SuccOther};
4949
4950 setFittedBranchWeights(*PBI, NewWeights, /*IsExpected=*/false,
4951 /*ElideAllZero=*/true);
4952 // Cond may be a select instruction with the first operand set to "true", or
4953 // the second to "false" (see how createLogicalOp works for `and` and `or`)
4954 if (auto *SI = dyn_cast<SelectInst>(Cond)) {
4955 assert(isSelectInRoleOfConjunctionOrDisjunction(SI));
4956 // The select is predicated on PBICond
4957 assert(SI->getCondition() == PBICond);
4958 // The corresponding probabilities are what was referred to above as
4959 // PredCommon and PredOther.
4960 setFittedBranchWeights(*SI, {PredCommon, PredOther},
4961 /*IsExpected=*/false, /*ElideAllZero=*/true);
4962 }
4963 }
4964
4965 // OtherDest may have phi nodes. If so, add an entry from PBI's
4966 // block that are identical to the entries for BI's block.
4967 addPredecessorToBlock(OtherDest, PBI->getParent(), BB);
4968
4969 // We know that the CommonDest already had an edge from PBI to
4970 // it. If it has PHIs though, the PHIs may have different
4971 // entries for BB and PBI's BB. If so, insert a select to make
4972 // them agree.
4973 for (PHINode &PN : CommonDest->phis()) {
4974 Value *BIV = PN.getIncomingValueForBlock(BB);
4975 unsigned PBBIdx = PN.getBasicBlockIndex(PBI->getParent());
4976 Value *PBIV = PN.getIncomingValue(PBBIdx);
4977 if (BIV != PBIV) {
4978 // Insert a select in PBI to pick the right value.
4980 Builder.CreateSelect(PBICond, PBIV, BIV, PBIV->getName() + ".mux"));
4981 PN.setIncomingValue(PBBIdx, NV);
4982 // The select has the same condition as PBI, in the same BB. The
4983 // probabilities don't change.
4984 if (HasWeights) {
4985 uint64_t TrueWeight = PBIOp ? PredFalseWeight : PredTrueWeight;
4986 uint64_t FalseWeight = PBIOp ? PredTrueWeight : PredFalseWeight;
4987 setFittedBranchWeights(*NV, {TrueWeight, FalseWeight},
4988 /*IsExpected=*/false, /*ElideAllZero=*/true);
4989 }
4990 }
4991 }
4992
4993 LLVM_DEBUG(dbgs() << "INTO: " << *PBI->getParent());
4994 LLVM_DEBUG(dbgs() << *PBI->getParent()->getParent());
4995
4996 // This basic block is probably dead. We know it has at least
4997 // one fewer predecessor.
4998 return true;
4999}
5000
5001// Simplifies a terminator by replacing it with a branch to TrueBB if Cond is
5002// true or to FalseBB if Cond is false.
5003// Takes care of updating the successors and removing the old terminator.
5004// Also makes sure not to introduce new successors by assuming that edges to
5005// non-successor TrueBBs and FalseBBs aren't reachable.
5006bool SimplifyCFGOpt::simplifyTerminatorOnSelect(Instruction *OldTerm,
5007 Value *Cond, BasicBlock *TrueBB,
5008 BasicBlock *FalseBB,
5009 uint32_t TrueWeight,
5010 uint32_t FalseWeight) {
5011 auto *BB = OldTerm->getParent();
5012 // Remove any superfluous successor edges from the CFG.
5013 // First, figure out which successors to preserve.
5014 // If TrueBB and FalseBB are equal, only try to preserve one copy of that
5015 // successor.
5016 BasicBlock *KeepEdge1 = TrueBB;
5017 BasicBlock *KeepEdge2 = TrueBB != FalseBB ? FalseBB : nullptr;
5018
5019 SmallSetVector<BasicBlock *, 2> RemovedSuccessors;
5020
5021 // Then remove the rest.
5022 for (BasicBlock *Succ : successors(OldTerm)) {
5023 // Make sure only to keep exactly one copy of each edge.
5024 if (Succ == KeepEdge1)
5025 KeepEdge1 = nullptr;
5026 else if (Succ == KeepEdge2)
5027 KeepEdge2 = nullptr;
5028 else {
5029 Succ->removePredecessor(BB,
5030 /*KeepOneInputPHIs=*/true);
5031
5032 if (Succ != TrueBB && Succ != FalseBB)
5033 RemovedSuccessors.insert(Succ);
5034 }
5035 }
5036
5037 IRBuilder<> Builder(OldTerm);
5038 Builder.SetCurrentDebugLocation(OldTerm->getDebugLoc());
5039
5040 // Insert an appropriate new terminator.
5041 if (!KeepEdge1 && !KeepEdge2) {
5042 if (TrueBB == FalseBB) {
5043 // We were only looking for one successor, and it was present.
5044 // Create an unconditional branch to it.
5045 Builder.CreateBr(TrueBB);
5046 } else {
5047 // We found both of the successors we were looking for.
5048 // Create a conditional branch sharing the condition of the select.
5049 CondBrInst *NewBI = Builder.CreateCondBr(Cond, TrueBB, FalseBB);
5050 setBranchWeights(*NewBI, {TrueWeight, FalseWeight},
5051 /*IsExpected=*/false, /*ElideAllZero=*/true);
5052 }
5053 } else if (KeepEdge1 && (KeepEdge2 || TrueBB == FalseBB)) {
5054 // Neither of the selected blocks were successors, so this
5055 // terminator must be unreachable.
5056 new UnreachableInst(OldTerm->getContext(), OldTerm->getIterator());
5057 } else {
5058 // One of the selected values was a successor, but the other wasn't.
5059 // Insert an unconditional branch to the one that was found;
5060 // the edge to the one that wasn't must be unreachable.
5061 if (!KeepEdge1) {
5062 // Only TrueBB was found.
5063 Builder.CreateBr(TrueBB);
5064 } else {
5065 // Only FalseBB was found.
5066 Builder.CreateBr(FalseBB);
5067 }
5068 }
5069
5071
5072 if (DTU) {
5073 SmallVector<DominatorTree::UpdateType, 2> Updates;
5074 Updates.reserve(RemovedSuccessors.size());
5075 for (auto *RemovedSuccessor : RemovedSuccessors)
5076 Updates.push_back({DominatorTree::Delete, BB, RemovedSuccessor});
5077 DTU->applyUpdates(Updates);
5078 }
5079
5080 return true;
5081}
5082
5083// Folds switch(select(icmp eq X, C, K, X)) into switch(X), retargeting
5084// (or adding) the case for C to wherever K currently dispatches to:
5085// %cmp = icmp eq T %x, C
5086// %key = select i1 %cmp, T K, T %x
5087// switch T %key, label %default [ T K, label %case_k ... ]
5088// becomes
5089// switch T %x, label %default [ T C, label %case_k
5090// T K, label %case_k ... ]
5091bool SimplifyCFGOpt::simplifySwitchOnSelectRemap(SwitchInst *SI,
5092 SelectInst *Select, Value *X,
5093 ConstantInt *C, bool Negate) {
5094 Value *TrueVal = Select->getTrueValue();
5095 Value *FalseVal = Select->getFalseValue();
5096 if (Negate)
5097 std::swap(TrueVal, FalseVal);
5098 if (FalseVal != X)
5099 return false;
5100 auto *K = dyn_cast<ConstantInt>(TrueVal);
5101 if (!K)
5102 return false;
5103
5104 BasicBlock *DestFork = SI->findCaseValue(K)->getCaseSuccessor();
5105 auto CaseC = SI->findCaseValue(C);
5106 bool IsDefault = CaseC == SI->case_default();
5107 // Save before setSuccessor()/addCase() change it.
5108 BasicBlock *OldDest = CaseC->getCaseSuccessor();
5109 BasicBlock *BB = SI->getParent();
5110
5111 // Switch on X first: removePredecessor() below may fold X away if it is a
5112 // PHI in OldDest, and the RAUW must update the switch condition too.
5113 SI->setCondition(X);
5114
5115 if (OldDest != DestFork) {
5116 if (!IsDefault)
5117 OldDest->removePredecessor(BB);
5118 if (IsDefault)
5119 SI->addCase(C, DestFork);
5120 else
5121 CaseC->setSuccessor(DestFork);
5122 // Not a new edge (BB->DestFork exists via K), just adding the PHI
5123 // entry.
5124 addPredecessorToBlock(DestFork, BB, BB);
5125
5126 if (!IsDefault) {
5127 // Edge to OldDest is gone only if nothing else still uses it.
5128 bool OldDestStillTargeted = any_of(
5129 successors(SI), [&](BasicBlock *Succ) { return Succ == OldDest; });
5130 if (DTU && !OldDestStillTargeted)
5131 DTU->applyUpdates({{DominatorTree::Delete, BB, OldDest}});
5132 }
5133
5134 // Update the profile information on the switch if we had a profile
5135 // for both it and the select instruction.
5136 SmallVector<uint32_t> SwitchWeights;
5137 bool SwitchHasBranchWeights = extractBranchWeights(*SI, SwitchWeights);
5138 // If we add a case, ensure the length of the branch weights list matches
5139 // to make iterating over them easier later.
5140 if (IsDefault)
5141 SwitchWeights.push_back(0);
5144 bool SelectHasBranchWeights =
5146 uint64_t SelectTotalWeight = SelectTrueWeight + SelectFalseWeight;
5147 if (Negate)
5149 if (SwitchHasBranchWeights && SelectHasBranchWeights &&
5151 // We update the branch weights by subtracting P(x=C) from the probability
5152 // of case K in the switch (what C redirect to before the transformation),
5153 // plugging the probability for case C into the switch (which we derive
5154 // from the select), and ensuring everything is scaled to have a common
5155 // denominator.
5156 uint64_t SwitchTotalWeight = sum_of(SwitchWeights, uint64_t{0});
5157 SmallVector<uint64_t> NewSwitchWeights;
5158 NewSwitchWeights.reserve(SwitchWeights.size());
5159 NewSwitchWeights.push_back(SwitchWeights[0] * SelectTotalWeight);
5160 for (const auto &[SwitchCase, SwitchWeight] :
5161 zip(SI->cases(), drop_begin(SwitchWeights))) {
5162 if (SwitchCase.getCaseValue() == C) {
5163 NewSwitchWeights.push_back(SwitchTotalWeight * SelectTrueWeight);
5164 } else if (SwitchCase.getCaseValue() == K) {
5165 // In reality, P(key=K) > P(x=C) should always hold, but explicitly
5166 // guard against bad profiles here to prevent underflow by saturating
5167 // to zero.
5168 uint64_t ProbabilityKeyEqualsK = SwitchWeight * SelectTotalWeight;
5169 uint64_t ProbabilityXEqualsC = SelectTrueWeight * SwitchTotalWeight;
5170 uint64_t ProbabilityXEqualsK =
5171 ProbabilityKeyEqualsK > ProbabilityXEqualsC
5172 ? ProbabilityKeyEqualsK - ProbabilityXEqualsC
5173 : 0;
5174 NewSwitchWeights.push_back(ProbabilityXEqualsK);
5175 } else {
5176 NewSwitchWeights.push_back(SwitchWeight * SelectTotalWeight);
5177 }
5178 }
5179 setFittedBranchWeights(*SI, NewSwitchWeights, /*IsExpected=*/false);
5180 } else if (SwitchHasBranchWeights) {
5181 // If we only have branch weights on the switch, we cannot reconstruct
5182 // branch weights correctly, so mark them as unknown if the function has
5183 // a profile count. Reset the branch weights first to ensure we remove
5184 // the now invalid branch weights if the function is not otherwise
5185 // profiled.
5186 SI->setMetadata(LLVMContext::MD_prof, nullptr);
5188 }
5189 }
5190
5191 // The compare/select are now dead.
5193 return true;
5194}
5195
5196// Replaces
5197// (switch (select cond, X, Y)) on constant X, Y
5198// with a branch - conditional if X and Y lead to distinct BBs,
5199// unconditional otherwise.
5200bool SimplifyCFGOpt::simplifySwitchOnSelect(SwitchInst *SI,
5201 SelectInst *Select) {
5202 CmpPredicate Pred;
5203 Value *X;
5204 ConstantInt *C;
5205 if (Select->hasOneUse() &&
5206 match(Select->getCondition(),
5207 m_ICmp(Pred, m_Value(X), m_ConstantInt(C))) &&
5208 ICmpInst::isEquality(Pred) &&
5209 simplifySwitchOnSelectRemap(SI, Select, X, C, Pred == ICmpInst::ICMP_NE))
5210 return true;
5211
5212 // Check for constant integer values in the select.
5213 ConstantInt *TrueVal = dyn_cast<ConstantInt>(Select->getTrueValue());
5214 ConstantInt *FalseVal = dyn_cast<ConstantInt>(Select->getFalseValue());
5215 if (!TrueVal || !FalseVal)
5216 return false;
5217
5218 // Find the relevant condition and destinations.
5219 Value *Condition = Select->getCondition();
5220 BasicBlock *TrueBB = SI->findCaseValue(TrueVal)->getCaseSuccessor();
5221 BasicBlock *FalseBB = SI->findCaseValue(FalseVal)->getCaseSuccessor();
5222
5223 // Get weight for TrueBB and FalseBB.
5224 uint32_t TrueWeight = 0, FalseWeight = 0;
5225 SmallVector<uint64_t, 8> Weights;
5226 bool HasWeights = hasBranchWeightMD(*SI);
5227 if (HasWeights) {
5228 getBranchWeights(SI, Weights);
5229 if (Weights.size() == 1 + SI->getNumCases()) {
5230 TrueWeight =
5231 (uint32_t)Weights[SI->findCaseValue(TrueVal)->getSuccessorIndex()];
5232 FalseWeight =
5233 (uint32_t)Weights[SI->findCaseValue(FalseVal)->getSuccessorIndex()];
5234 }
5235 }
5236
5237 // Perform the actual simplification.
5238 return simplifyTerminatorOnSelect(SI, Condition, TrueBB, FalseBB, TrueWeight,
5239 FalseWeight);
5240}
5241
5242// Replaces
5243// (indirectbr (select cond, blockaddress(@fn, BlockA),
5244// blockaddress(@fn, BlockB)))
5245// with
5246// (br cond, BlockA, BlockB).
5247bool SimplifyCFGOpt::simplifyIndirectBrOnSelect(IndirectBrInst *IBI,
5248 SelectInst *SI) {
5249 // Check that both operands of the select are block addresses.
5250 BlockAddress *TBA = dyn_cast<BlockAddress>(SI->getTrueValue());
5251 BlockAddress *FBA = dyn_cast<BlockAddress>(SI->getFalseValue());
5252 if (!TBA || !FBA)
5253 return false;
5254
5255 // Extract the actual blocks.
5256 BasicBlock *TrueBB = TBA->getBasicBlock();
5257 BasicBlock *FalseBB = FBA->getBasicBlock();
5258
5259 // The select's profile becomes the profile of the conditional branch that
5260 // replaces the indirect branch.
5261 SmallVector<uint32_t> SelectBranchWeights(2);
5262 extractBranchWeights(*SI, SelectBranchWeights);
5263 // Perform the actual simplification.
5264 return simplifyTerminatorOnSelect(IBI, SI->getCondition(), TrueBB, FalseBB,
5265 SelectBranchWeights[0],
5266 SelectBranchWeights[1]);
5267}
5268
5269/// This is called when we find an icmp instruction
5270/// (a seteq/setne with a constant) as the only instruction in a
5271/// block that ends with an uncond branch. We are looking for a very specific
5272/// pattern that occurs when "A == 1 || A == 2 || A == 3" gets simplified. In
5273/// this case, we merge the first two "or's of icmp" into a switch, but then the
5274/// default value goes to an uncond block with a seteq in it, we get something
5275/// like:
5276///
5277/// switch i8 %A, label %DEFAULT [ i8 1, label %end i8 2, label %end ]
5278/// DEFAULT:
5279/// %tmp = icmp eq i8 %A, 92
5280/// br label %end
5281/// end:
5282/// ... = phi i1 [ true, %entry ], [ %tmp, %DEFAULT ], [ true, %entry ]
5283///
5284/// We prefer to split the edge to 'end' so that there is a true/false entry to
5285/// the PHI, merging the third icmp into the switch.
5286bool SimplifyCFGOpt::tryToSimplifyUncondBranchWithICmpInIt(
5287 ICmpInst *ICI, IRBuilder<> &Builder) {
5288 // Select == nullptr means we assume that there is a hidden no-op select
5289 // instruction of `_ = select %icmp, true, false` after `%icmp = icmp ...`
5290 return tryToSimplifyUncondBranchWithICmpSelectInIt(ICI, nullptr, Builder);
5291}
5292
5293/// Similar to tryToSimplifyUncondBranchWithICmpInIt, but handle a more generic
5294/// case. This is called when we find an icmp instruction (a seteq/setne with a
5295/// constant) and its following select instruction as the only TWO instructions
5296/// in a block that ends with an uncond branch. We are looking for a very
5297/// specific pattern that occurs when "
5298/// if (A == 1) return C1;
5299/// if (A == 2) return C2;
5300/// if (A < 3) return C3;
5301/// return C4;
5302/// " gets simplified. In this case, we merge the first two "branches of icmp"
5303/// into a switch, but then the default value goes to an uncond block with a lt
5304/// icmp and select in it, as InstCombine can not simplify "A < 3" as "A == 2".
5305/// After SimplifyCFG and other subsequent optimizations (e.g., SCCP), we might
5306/// get something like:
5307///
5308/// case1:
5309/// switch i8 %A, label %DEFAULT [ i8 0, label %end i8 1, label %case2 ]
5310/// case2:
5311/// br label %end
5312/// DEFAULT:
5313/// %tmp = icmp eq i8 %A, 2
5314/// %val = select i1 %tmp, i8 C3, i8 C4
5315/// br label %end
5316/// end:
5317/// _ = phi i8 [ C1, %case1 ], [ C2, %case2 ], [ %val, %DEFAULT ]
5318///
5319/// We prefer to split the edge to 'end' so that there are TWO entries of V3/V4
5320/// to the PHI, merging the icmp & select into the switch, as follows:
5321///
5322/// case1:
5323/// switch i8 %A, label %DEFAULT [
5324/// i8 0, label %end
5325/// i8 1, label %case2
5326/// i8 2, label %case3
5327/// ]
5328/// case2:
5329/// br label %end
5330/// case3:
5331/// br label %end
5332/// DEFAULT:
5333/// br label %end
5334/// end:
5335/// _ = phi i8 [ C1, %case1 ], [ C2, %case2 ], [ C3, %case2 ], [ C4, %DEFAULT]
5336bool SimplifyCFGOpt::tryToSimplifyUncondBranchWithICmpSelectInIt(
5337 ICmpInst *ICI, SelectInst *Select, IRBuilder<> &Builder) {
5338 BasicBlock *BB = ICI->getParent();
5339
5340 // If the block has any PHIs in it or the icmp/select has multiple uses, it is
5341 // too complex.
5342 /// TODO: support multi-phis in succ BB of select's BB.
5343 if (isa<PHINode>(BB->begin()) || !ICI->hasOneUse() ||
5344 (Select && !Select->hasOneUse()))
5345 return false;
5346
5347 // The pattern we're looking for is where our only predecessor is a switch on
5348 // 'V' and this block is the default case for the switch. In this case we can
5349 // fold the compared value into the switch to simplify things.
5350 BasicBlock *Pred = BB->getSinglePredecessor();
5351 if (!Pred || !isa<SwitchInst>(Pred->getTerminator()))
5352 return false;
5353
5354 Value *IcmpCond;
5355 ConstantInt *NewCaseVal;
5356 CmpPredicate Predicate;
5357
5358 // Match icmp X, C
5359 if (!match(ICI,
5360 m_ICmp(Predicate, m_Value(IcmpCond), m_ConstantInt(NewCaseVal))))
5361 return false;
5362
5363 Value *SelectCond, *SelectTrueVal, *SelectFalseVal;
5365 if (!Select) {
5366 // If Select == nullptr, we can assume that there is a hidden no-op select
5367 // just after icmp
5368 SelectCond = ICI;
5369 SelectTrueVal = Builder.getTrue();
5370 SelectFalseVal = Builder.getFalse();
5371 User = ICI->user_back();
5372 } else {
5373 SelectCond = Select->getCondition();
5374 // Check if the select condition is the same as the icmp condition.
5375 if (SelectCond != ICI)
5376 return false;
5377 SelectTrueVal = Select->getTrueValue();
5378 SelectFalseVal = Select->getFalseValue();
5379 User = Select->user_back();
5380 }
5381
5382 SwitchInst *SI = cast<SwitchInst>(Pred->getTerminator());
5383 if (SI->getCondition() != IcmpCond)
5384 return false;
5385
5386 // If BB is reachable on a non-default case, then we simply know the value of
5387 // V in this block. Substitute it and constant fold the icmp instruction
5388 // away.
5389 if (SI->getDefaultDest() != BB) {
5390 ConstantInt *VVal = SI->findCaseDest(BB);
5391 assert(VVal && "Should have a unique destination value");
5392 ICI->setOperand(0, VVal);
5393
5394 if (Value *V = simplifyInstruction(ICI, {DL, ICI})) {
5395 ICI->replaceAllUsesWith(V);
5396 ICI->eraseFromParent();
5397 }
5398 // BB is now empty, so it is likely to simplify away.
5399 return requestResimplify();
5400 }
5401
5402 // Ok, the block is reachable from the default dest. If the constant we're
5403 // comparing exists in one of the other edges, then we can constant fold ICI
5404 // and zap it.
5405 if (SI->findCaseValue(NewCaseVal) != SI->case_default()) {
5406 Value *V;
5407 if (Predicate == ICmpInst::ICMP_EQ)
5409 else
5411
5412 ICI->replaceAllUsesWith(V);
5413 ICI->eraseFromParent();
5414 // BB is now empty, so it is likely to simplify away.
5415 return requestResimplify();
5416 }
5417
5418 // The use of the select has to be in the 'end' block, by the only PHI node in
5419 // the block.
5420 BasicBlock *SuccBlock = BB->getTerminator()->getSuccessor(0);
5421 PHINode *PHIUse = dyn_cast<PHINode>(User);
5422 if (PHIUse == nullptr || PHIUse != &SuccBlock->front() ||
5424 return false;
5425
5426 // If the icmp is a SETEQ, then the default dest gets SelectFalseVal, the new
5427 // edge gets SelectTrueVal in the PHI.
5428 Value *DefaultCst = SelectFalseVal;
5429 Value *NewCst = SelectTrueVal;
5430
5431 if (ICI->getPredicate() == ICmpInst::ICMP_NE)
5432 std::swap(DefaultCst, NewCst);
5433
5434 // Replace Select (which is used by the PHI for the default value) with
5435 // SelectFalseVal or SelectTrueVal depending on if ICI is EQ or NE.
5436 if (Select) {
5437 Select->replaceAllUsesWith(DefaultCst);
5438 Select->eraseFromParent();
5439 } else {
5440 ICI->replaceAllUsesWith(DefaultCst);
5441 }
5442 ICI->eraseFromParent();
5443
5444 SmallVector<DominatorTree::UpdateType, 2> Updates;
5445
5446 // Okay, the switch goes to this block on a default value. Add an edge from
5447 // the switch to the merge point on the compared value.
5448 BasicBlock *NewBB =
5449 BasicBlock::Create(BB->getContext(), "switch.edge", BB->getParent(), BB);
5450 {
5451 SwitchInstProfUpdateWrapper SIW(*SI);
5452 auto W0 = SIW.getSuccessorWeight(0);
5454 if (W0) {
5455 NewW = ((uint64_t(*W0) + 1) >> 1);
5456 SIW.setSuccessorWeight(0, *NewW);
5457 }
5458 SIW.addCase(NewCaseVal, NewBB, NewW);
5459 if (DTU)
5460 Updates.push_back({DominatorTree::Insert, Pred, NewBB});
5461 }
5462
5463 // NewBB branches to the phi block, add the uncond branch and the phi entry.
5464 Builder.SetInsertPoint(NewBB);
5465 Builder.SetCurrentDebugLocation(SI->getDebugLoc());
5466 Builder.CreateBr(SuccBlock);
5467 PHIUse->addIncoming(NewCst, NewBB);
5468 if (DTU) {
5469 Updates.push_back({DominatorTree::Insert, NewBB, SuccBlock});
5470 DTU->applyUpdates(Updates);
5471 }
5472 return true;
5473}
5474
5475/// Check to see if it is branching on an or/and chain of icmp instructions, and
5476/// fold it into a switch instruction if so.
5477bool SimplifyCFGOpt::simplifyBranchOnICmpChain(CondBrInst *BI,
5478 IRBuilder<> &Builder,
5479 const DataLayout &DL) {
5481 if (!Cond)
5482 return false;
5483
5484 // Change br (X == 0 | X == 1), T, F into a switch instruction.
5485 // If this is a bunch of seteq's or'd together, or if it's a bunch of
5486 // 'setne's and'ed together, collect them.
5487
5488 // Try to gather values from a chain of and/or to be turned into a switch
5489 ConstantComparesGatherer ConstantCompare(Cond, DL);
5490 // Unpack the result
5491 SmallVectorImpl<ConstantInt *> &Values = ConstantCompare.Vals;
5492 Value *CompVal = ConstantCompare.CompValue;
5493 unsigned UsedICmps = ConstantCompare.UsedICmps;
5494 Value *ExtraCase = ConstantCompare.Extra;
5495 bool TrueWhenEqual = ConstantCompare.IsEq;
5496
5497 // If we didn't have a multiply compared value, fail.
5498 if (!CompVal)
5499 return false;
5500
5501 // Avoid turning single icmps into a switch.
5502 if (UsedICmps <= 1)
5503 return false;
5504
5505 // There might be duplicate constants in the list, which the switch
5506 // instruction can't handle, remove them now.
5508 Values.erase(llvm::unique(Values), Values.end());
5509
5510 // If Extra was used, we require at least two switch values to do the
5511 // transformation. A switch with one value is just a conditional branch.
5512 if (ExtraCase && Values.size() < 2)
5513 return false;
5514
5515 SmallVector<uint32_t> BranchWeights;
5516 const bool HasProfile = extractBranchWeights(*BI, BranchWeights);
5517
5518 // Figure out which block is which destination.
5519 BasicBlock *DefaultBB = BI->getSuccessor(1);
5520 BasicBlock *EdgeBB = BI->getSuccessor(0);
5521 if (!TrueWhenEqual) {
5522 std::swap(DefaultBB, EdgeBB);
5523 if (HasProfile)
5524 std::swap(BranchWeights[0], BranchWeights[1]);
5525 }
5526
5527 BasicBlock *BB = BI->getParent();
5528
5529 LLVM_DEBUG(dbgs() << "Converting 'icmp' chain with " << Values.size()
5530 << " cases into SWITCH. BB is:\n"
5531 << *BB);
5532
5533 SmallVector<DominatorTree::UpdateType, 2> Updates;
5534
5535 // If there are any extra values that couldn't be folded into the switch
5536 // then we evaluate them with an explicit branch first. Split the block
5537 // right before the condbr to handle it.
5538 if (ExtraCase) {
5539 BasicBlock *NewBB = SplitBlock(BB, BI, DTU, /*LI=*/nullptr,
5540 /*MSSAU=*/nullptr, "switch.early.test");
5541
5542 // Remove the uncond branch added to the old block.
5543 Instruction *OldTI = BB->getTerminator();
5544 Builder.SetInsertPoint(OldTI);
5545
5546 // There can be an unintended UB if extra values are Poison. Before the
5547 // transformation, extra values may not be evaluated according to the
5548 // condition, and it will not raise UB. But after transformation, we are
5549 // evaluating extra values before checking the condition, and it will raise
5550 // UB. It can be solved by adding freeze instruction to extra values.
5551 AssumptionCache *AC = Options.AC;
5552
5553 if (!isGuaranteedNotToBeUndefOrPoison(ExtraCase, AC, BI, nullptr))
5554 ExtraCase = Builder.CreateFreeze(ExtraCase);
5555
5556 // We don't have any info about this condition.
5557 auto *Br = TrueWhenEqual ? Builder.CreateCondBr(ExtraCase, EdgeBB, NewBB)
5558 : Builder.CreateCondBr(ExtraCase, NewBB, EdgeBB);
5560
5561 OldTI->eraseFromParent();
5562
5563 if (DTU)
5564 Updates.push_back({DominatorTree::Insert, BB, EdgeBB});
5565
5566 // If there are PHI nodes in EdgeBB, then we need to add a new entry to them
5567 // for the edge we just added.
5568 addPredecessorToBlock(EdgeBB, BB, NewBB);
5569
5570 LLVM_DEBUG(dbgs() << " ** 'icmp' chain unhandled condition: " << *ExtraCase
5571 << "\nEXTRABB = " << *BB);
5572 BB = NewBB;
5573 }
5574
5575 Builder.SetInsertPoint(BI);
5576 // Convert pointer to int before we switch.
5577 if (CompVal->getType()->isPointerTy()) {
5578 assert(!DL.hasUnstableRepresentation(CompVal->getType()) &&
5579 "Should not end up here with unstable pointers");
5580 CompVal = Builder.CreatePtrToInt(
5581 CompVal, DL.getIntPtrType(CompVal->getType()), "magicptr");
5582 }
5583
5584 // Check if we can represent the values as a contiguous range. If so, we use a
5585 // range check + conditional branch instead of a switch.
5586 if (Values.front()->getValue() - Values.back()->getValue() ==
5587 Values.size() - 1) {
5588 ConstantRange RangeToCheck = ConstantRange::getNonEmpty(
5589 Values.back()->getValue(), Values.front()->getValue() + 1);
5590 APInt Offset, RHS;
5591 ICmpInst::Predicate Pred;
5592 RangeToCheck.getEquivalentICmp(Pred, RHS, Offset);
5593 Value *X = CompVal;
5594 if (!Offset.isZero())
5595 X = Builder.CreateAdd(X, ConstantInt::get(CompVal->getType(), Offset));
5596 Value *Cond =
5597 Builder.CreateICmp(Pred, X, ConstantInt::get(CompVal->getType(), RHS));
5598 CondBrInst *NewBI = Builder.CreateCondBr(Cond, EdgeBB, DefaultBB);
5599 if (HasProfile)
5600 setBranchWeights(*NewBI, BranchWeights, /*IsExpected=*/false);
5601 if (MDNode *Unpredictable = BI->getMetadata(LLVMContext::MD_unpredictable))
5602 NewBI->setMetadata(LLVMContext::MD_unpredictable, Unpredictable);
5603 // We don't need to update PHI nodes since we don't add any new edges.
5604 } else {
5605 // Create the new switch instruction now.
5606 SwitchInst *New = Builder.CreateSwitch(CompVal, DefaultBB, Values.size());
5607 if (MDNode *Unpredictable = BI->getMetadata(LLVMContext::MD_unpredictable))
5608 New->setMetadata(LLVMContext::MD_unpredictable, Unpredictable);
5609 if (HasProfile) {
5610 // We know the weight of the default case. We don't know the weight of the
5611 // other cases, but rather than completely lose profiling info, we split
5612 // the remaining probability equally over them.
5613 SmallVector<uint32_t> NewWeights(Values.size() + 1);
5614 NewWeights[0] = BranchWeights[1]; // this is the default, and we swapped
5615 // if TrueWhenEqual.
5616 for (auto &V : drop_begin(NewWeights))
5617 V = BranchWeights[0] / Values.size();
5618 setBranchWeights(*New, NewWeights, /*IsExpected=*/false);
5619 }
5620
5621 // Add all of the 'cases' to the switch instruction.
5622 for (ConstantInt *Val : Values)
5623 New->addCase(Val, EdgeBB);
5624
5625 // We added edges from PI to the EdgeBB. As such, if there were any
5626 // PHI nodes in EdgeBB, they need entries to be added corresponding to
5627 // the number of edges added.
5628 for (BasicBlock::iterator BBI = EdgeBB->begin(); isa<PHINode>(BBI); ++BBI) {
5629 PHINode *PN = cast<PHINode>(BBI);
5630 Value *InVal = PN->getIncomingValueForBlock(BB);
5631 for (unsigned i = 0, e = Values.size() - 1; i != e; ++i)
5632 PN->addIncoming(InVal, BB);
5633 }
5634 }
5635
5636 // Erase the old branch instruction.
5638 if (DTU)
5639 DTU->applyUpdates(Updates);
5640
5641 LLVM_DEBUG(dbgs() << " ** 'icmp' chain result is:\n" << *BB << '\n');
5642 return true;
5643}
5644
5645bool SimplifyCFGOpt::simplifyResume(ResumeInst *RI, IRBuilder<> &Builder) {
5646 if (isa<PHINode>(RI->getValue()))
5647 return simplifyCommonResume(RI);
5648 else if (isa<LandingPadInst>(RI->getParent()->getFirstNonPHIIt()) &&
5649 RI->getValue() == &*RI->getParent()->getFirstNonPHIIt())
5650 // The resume must unwind the exception that caused control to branch here.
5651 return simplifySingleResume(RI);
5652
5653 return false;
5654}
5655
5656// Check if cleanup block is empty
5658 for (Instruction &I : R) {
5659 auto *II = dyn_cast<IntrinsicInst>(&I);
5660 if (!II)
5661 return false;
5662
5663 Intrinsic::ID IntrinsicID = II->getIntrinsicID();
5664 switch (IntrinsicID) {
5665 case Intrinsic::dbg_declare:
5666 case Intrinsic::dbg_value:
5667 case Intrinsic::dbg_label:
5668 case Intrinsic::lifetime_end:
5669 break;
5670 default:
5671 return false;
5672 }
5673 }
5674 return true;
5675}
5676
5677// Simplify resume that is shared by several landing pads (phi of landing pad).
5678bool SimplifyCFGOpt::simplifyCommonResume(ResumeInst *RI) {
5679 BasicBlock *BB = RI->getParent();
5680
5681 // Check that there are no other instructions except for debug and lifetime
5682 // intrinsics between the phi's and resume instruction.
5683 if (!isCleanupBlockEmpty(make_range(RI->getParent()->getFirstNonPHIIt(),
5684 BB->getTerminator()->getIterator())))
5685 return false;
5686
5687 SmallSetVector<BasicBlock *, 4> TrivialUnwindBlocks;
5688 auto *PhiLPInst = cast<PHINode>(RI->getValue());
5689
5690 // Check incoming blocks to see if any of them are trivial.
5691 for (unsigned Idx = 0, End = PhiLPInst->getNumIncomingValues(); Idx != End;
5692 Idx++) {
5693 auto *IncomingBB = PhiLPInst->getIncomingBlock(Idx);
5694 auto *IncomingValue = PhiLPInst->getIncomingValue(Idx);
5695
5696 // If the block has other successors, we can not delete it because
5697 // it has other dependents.
5698 if (IncomingBB->getUniqueSuccessor() != BB)
5699 continue;
5700
5701 auto *LandingPad = dyn_cast<LandingPadInst>(IncomingBB->getFirstNonPHIIt());
5702 // Not the landing pad that caused the control to branch here.
5703 if (IncomingValue != LandingPad)
5704 continue;
5705
5707 make_range(LandingPad->getNextNode(), IncomingBB->getTerminator())))
5708 TrivialUnwindBlocks.insert(IncomingBB);
5709 }
5710
5711 // If no trivial unwind blocks, don't do any simplifications.
5712 if (TrivialUnwindBlocks.empty())
5713 return false;
5714
5715 // Turn all invokes that unwind here into calls.
5716 for (auto *TrivialBB : TrivialUnwindBlocks) {
5717 // Blocks that will be simplified should be removed from the phi node.
5718 // Note there could be multiple edges to the resume block, and we need
5719 // to remove them all.
5720 while (PhiLPInst->getBasicBlockIndex(TrivialBB) != -1)
5721 BB->removePredecessor(TrivialBB, true);
5722
5723 for (BasicBlock *Pred :
5725 removeUnwindEdge(Pred, DTU);
5726 ++NumInvokes;
5727 }
5728
5729 // In each SimplifyCFG run, only the current processed block can be erased.
5730 // Otherwise, it will break the iteration of SimplifyCFG pass. So instead
5731 // of erasing TrivialBB, we only remove the branch to the common resume
5732 // block so that we can later erase the resume block since it has no
5733 // predecessors.
5734 TrivialBB->getTerminator()->eraseFromParent();
5735 new UnreachableInst(RI->getContext(), TrivialBB);
5736 if (DTU)
5737 DTU->applyUpdates({{DominatorTree::Delete, TrivialBB, BB}});
5738 }
5739
5740 // Delete the resume block if all its predecessors have been removed.
5741 if (pred_empty(BB))
5742 DeleteDeadBlock(BB, DTU);
5743
5744 return !TrivialUnwindBlocks.empty();
5745}
5746
5747// Simplify resume that is only used by a single (non-phi) landing pad.
5748bool SimplifyCFGOpt::simplifySingleResume(ResumeInst *RI) {
5749 BasicBlock *BB = RI->getParent();
5750 auto *LPInst = cast<LandingPadInst>(BB->getFirstNonPHIIt());
5751 assert(RI->getValue() == LPInst &&
5752 "Resume must unwind the exception that caused control to here");
5753
5754 // Check that there are no other instructions except for debug intrinsics.
5756 make_range<Instruction *>(LPInst->getNextNode(), RI)))
5757 return false;
5758
5759 // Turn all invokes that unwind here into calls and delete the basic block.
5760 for (BasicBlock *Pred : llvm::make_early_inc_range(predecessors(BB))) {
5761 removeUnwindEdge(Pred, DTU);
5762 ++NumInvokes;
5763 }
5764
5765 // The landingpad is now unreachable. Zap it.
5766 DeleteDeadBlock(BB, DTU);
5767 return true;
5768}
5769
5771 // If this is a trivial cleanup pad that executes no instructions, it can be
5772 // eliminated. If the cleanup pad continues to the caller, any predecessor
5773 // that is an EH pad will be updated to continue to the caller and any
5774 // predecessor that terminates with an invoke instruction will have its invoke
5775 // instruction converted to a call instruction. If the cleanup pad being
5776 // simplified does not continue to the caller, each predecessor will be
5777 // updated to continue to the unwind destination of the cleanup pad being
5778 // simplified.
5779 BasicBlock *BB = RI->getParent();
5780 CleanupPadInst *CPInst = RI->getCleanupPad();
5781 if (CPInst->getParent() != BB)
5782 // This isn't an empty cleanup.
5783 return false;
5784
5785 // We cannot kill the pad if it has multiple uses. This typically arises
5786 // from unreachable basic blocks.
5787 if (!CPInst->hasOneUse())
5788 return false;
5789
5790 // Check that there are no other instructions except for benign intrinsics.
5792 make_range<Instruction *>(CPInst->getNextNode(), RI)))
5793 return false;
5794
5795 // If the cleanup return we are simplifying unwinds to the caller, this will
5796 // set UnwindDest to nullptr.
5797 BasicBlock *UnwindDest = RI->getUnwindDest();
5798
5799 // We're about to remove BB from the control flow. Before we do, sink any
5800 // PHINodes into the unwind destination. Doing this before changing the
5801 // control flow avoids some potentially slow checks, since we can currently
5802 // be certain that UnwindDest and BB have no common predecessors (since they
5803 // are both EH pads).
5804 if (UnwindDest) {
5805 // First, go through the PHI nodes in UnwindDest and update any nodes that
5806 // reference the block we are removing
5807 for (PHINode &DestPN : UnwindDest->phis()) {
5808 int Idx = DestPN.getBasicBlockIndex(BB);
5809 // Since BB unwinds to UnwindDest, it has to be in the PHI node.
5810 assert(Idx != -1);
5811 // This PHI node has an incoming value that corresponds to a control
5812 // path through the cleanup pad we are removing. If the incoming
5813 // value is in the cleanup pad, it must be a PHINode (because we
5814 // verified above that the block is otherwise empty). Otherwise, the
5815 // value is either a constant or a value that dominates the cleanup
5816 // pad being removed.
5817 //
5818 // Because BB and UnwindDest are both EH pads, all of their
5819 // predecessors must unwind to these blocks, and since no instruction
5820 // can have multiple unwind destinations, there will be no overlap in
5821 // incoming blocks between SrcPN and DestPN.
5822 Value *SrcVal = DestPN.getIncomingValue(Idx);
5823 PHINode *SrcPN = dyn_cast<PHINode>(SrcVal);
5824
5825 bool NeedPHITranslation = SrcPN && SrcPN->getParent() == BB;
5826 for (auto *Pred : predecessors(BB)) {
5827 Value *Incoming =
5828 NeedPHITranslation ? SrcPN->getIncomingValueForBlock(Pred) : SrcVal;
5829 DestPN.addIncoming(Incoming, Pred);
5830 }
5831 }
5832
5833 // Sink any remaining PHI nodes directly into UnwindDest.
5834 BasicBlock::iterator InsertPt = UnwindDest->getFirstNonPHIIt();
5835 for (PHINode &PN : make_early_inc_range(BB->phis())) {
5836 if (PN.use_empty() || !PN.isUsedOutsideOfBlock(BB))
5837 // If the PHI node has no uses or all of its uses are in this basic
5838 // block (meaning they are debug or lifetime intrinsics), just leave
5839 // it. It will be erased when we erase BB below.
5840 continue;
5841
5842 // Otherwise, sink this PHI node into UnwindDest.
5843 // Any predecessors to UnwindDest which are not already represented
5844 // must be back edges which inherit the value from the path through
5845 // BB. In this case, the PHI value must reference itself.
5846 for (auto *pred : predecessors(UnwindDest))
5847 if (pred != BB)
5848 PN.addIncoming(&PN, pred);
5849 PN.moveBefore(InsertPt);
5850 // Also, add a dummy incoming value for the original BB itself,
5851 // so that the PHI is well-formed until we drop said predecessor.
5852 PN.addIncoming(PoisonValue::get(PN.getType()), BB);
5853 }
5854 }
5855
5856 std::vector<DominatorTree::UpdateType> Updates;
5857
5858 // We use make_early_inc_range here because we will remove all predecessors.
5860 if (UnwindDest == nullptr) {
5861 if (DTU) {
5862 DTU->applyUpdates(Updates);
5863 Updates.clear();
5864 }
5865 removeUnwindEdge(PredBB, DTU);
5866 ++NumInvokes;
5867 } else {
5868 BB->removePredecessor(PredBB);
5869 Instruction *TI = PredBB->getTerminator();
5870 TI->replaceUsesOfWith(BB, UnwindDest);
5871 if (DTU) {
5872 Updates.push_back({DominatorTree::Insert, PredBB, UnwindDest});
5873 Updates.push_back({DominatorTree::Delete, PredBB, BB});
5874 }
5875 }
5876 }
5877
5878 if (DTU)
5879 DTU->applyUpdates(Updates);
5880
5881 DeleteDeadBlock(BB, DTU);
5882
5883 return true;
5884}
5885
5886// Try to merge two cleanuppads together.
5888 // Skip any cleanuprets which unwind to caller, there is nothing to merge
5889 // with.
5890 BasicBlock *UnwindDest = RI->getUnwindDest();
5891 if (!UnwindDest)
5892 return false;
5893
5894 // This cleanupret isn't the only predecessor of this cleanuppad, it wouldn't
5895 // be safe to merge without code duplication.
5896 if (UnwindDest->getSinglePredecessor() != RI->getParent())
5897 return false;
5898
5899 // Verify that our cleanuppad's unwind destination is another cleanuppad.
5900 auto *SuccessorCleanupPad = dyn_cast<CleanupPadInst>(&UnwindDest->front());
5901 if (!SuccessorCleanupPad)
5902 return false;
5903
5904 CleanupPadInst *PredecessorCleanupPad = RI->getCleanupPad();
5905 // Replace any uses of the successor cleanupad with the predecessor pad
5906 // The only cleanuppad uses should be this cleanupret, it's cleanupret and
5907 // funclet bundle operands.
5908 SuccessorCleanupPad->replaceAllUsesWith(PredecessorCleanupPad);
5909 // Remove the old cleanuppad.
5910 SuccessorCleanupPad->eraseFromParent();
5911 // Now, we simply replace the cleanupret with a branch to the unwind
5912 // destination.
5913 UncondBrInst::Create(UnwindDest, RI->getParent());
5914 RI->eraseFromParent();
5915
5916 return true;
5917}
5918
5919bool SimplifyCFGOpt::simplifyCleanupReturn(CleanupReturnInst *RI) {
5920 // It is possible to transiantly have an undef cleanuppad operand because we
5921 // have deleted some, but not all, dead blocks.
5922 // Eventually, this block will be deleted.
5923 if (isa<UndefValue>(RI->getOperand(0)))
5924 return false;
5925
5926 if (mergeCleanupPad(RI))
5927 return true;
5928
5929 if (removeEmptyCleanup(RI, DTU))
5930 return true;
5931
5932 return false;
5933}
5934
5935// WARNING: keep in sync with InstCombinerImpl::visitUnreachableInst()!
5936bool SimplifyCFGOpt::simplifyUnreachable(UnreachableInst *UI) {
5937 BasicBlock *BB = UI->getParent();
5938
5939 bool Changed = false;
5940
5941 // Ensure that any debug-info records that used to occur after the Unreachable
5942 // are moved to in front of it -- otherwise they'll "dangle" at the end of
5943 // the block.
5945
5946 // Debug-info records on the unreachable inst itself should be deleted, as
5947 // below we delete everything past the final executable instruction.
5948 UI->dropDbgRecords();
5949
5950 // If there are any instructions immediately before the unreachable that can
5951 // be removed, do so.
5952 while (UI->getIterator() != BB->begin()) {
5954 --BBI;
5955
5957 break; // Can not drop any more instructions. We're done here.
5958 // Otherwise, this instruction can be freely erased,
5959 // even if it is not side-effect free.
5960
5961 // Note that deleting EH's here is in fact okay, although it involves a bit
5962 // of subtle reasoning. If this inst is an EH, all the predecessors of this
5963 // block will be the unwind edges of Invoke/CatchSwitch/CleanupReturn,
5964 // and we can therefore guarantee this block will be erased.
5965
5966 // If we're deleting this, we're deleting any subsequent debug info, so
5967 // delete DbgRecords.
5968 BBI->dropDbgRecords();
5969
5970 // Delete this instruction (any uses are guaranteed to be dead)
5971 BBI->replaceAllUsesWith(PoisonValue::get(BBI->getType()));
5972 BBI->eraseFromParent();
5973 Changed = true;
5974 }
5975
5976 // If the unreachable instruction is the first in the block, take a gander
5977 // at all of the predecessors of this instruction, and simplify them.
5978 if (&BB->front() != UI)
5979 return Changed;
5980
5981 std::vector<DominatorTree::UpdateType> Updates;
5982
5983 SmallSetVector<BasicBlock *, 8> Preds(pred_begin(BB), pred_end(BB));
5984 for (BasicBlock *Predecessor : Preds) {
5985 Instruction *TI = Predecessor->getTerminator();
5986 IRBuilder<> Builder(TI);
5987 if (isa<UncondBrInst>(TI)) {
5988 new UnreachableInst(TI->getContext(), TI->getIterator());
5989 TI->eraseFromParent();
5990 Changed = true;
5991 if (DTU)
5992 Updates.push_back({DominatorTree::Delete, Predecessor, BB});
5993 } else if (auto *BI = dyn_cast<CondBrInst>(TI)) {
5994 // We could either have a proper unconditional branch,
5995 // or a degenerate conditional branch with matching destinations.
5996 if (BI->getSuccessor(0) == BI->getSuccessor(1)) {
5997 new UnreachableInst(TI->getContext(), TI->getIterator());
5998 TI->eraseFromParent();
5999 Changed = true;
6000 } else {
6001 Value* Cond = BI->getCondition();
6002 assert(BI->getSuccessor(0) != BI->getSuccessor(1) &&
6003 "The destinations are guaranteed to be different here.");
6004 CallInst *Assumption;
6005 if (BI->getSuccessor(0) == BB) {
6006 Assumption = Builder.CreateAssumption(Builder.CreateNot(Cond));
6007 Builder.CreateBr(BI->getSuccessor(1));
6008 } else {
6009 assert(BI->getSuccessor(1) == BB && "Incorrect CFG");
6010 Assumption = Builder.CreateAssumption(Cond);
6011 Builder.CreateBr(BI->getSuccessor(0));
6012 }
6013 if (Options.AC)
6014 Options.AC->registerAssumption(cast<AssumeInst>(Assumption));
6015
6017 Changed = true;
6018 }
6019 if (DTU)
6020 Updates.push_back({DominatorTree::Delete, Predecessor, BB});
6021 } else if (auto *SI = dyn_cast<SwitchInst>(TI)) {
6022 SwitchInstProfUpdateWrapper SU(*SI);
6023 for (auto i = SU->case_begin(), e = SU->case_end(); i != e;) {
6024 if (i->getCaseSuccessor() != BB) {
6025 ++i;
6026 continue;
6027 }
6028 BB->removePredecessor(SU->getParent());
6029 i = SU.removeCase(i);
6030 e = SU->case_end();
6031 Changed = true;
6032 }
6033 // Note that the default destination can't be removed!
6034 if (DTU && SI->getDefaultDest() != BB)
6035 Updates.push_back({DominatorTree::Delete, Predecessor, BB});
6036 } else if (auto *II = dyn_cast<InvokeInst>(TI)) {
6037 if (II->getUnwindDest() == BB) {
6038 if (DTU) {
6039 DTU->applyUpdates(Updates);
6040 Updates.clear();
6041 }
6042 auto *CI = cast<CallInst>(removeUnwindEdge(TI->getParent(), DTU));
6043 if (!CI->doesNotThrow())
6044 CI->setDoesNotThrow();
6045 Changed = true;
6046 }
6047 } else if (auto *CSI = dyn_cast<CatchSwitchInst>(TI)) {
6048 if (CSI->getUnwindDest() == BB) {
6049 if (DTU) {
6050 DTU->applyUpdates(Updates);
6051 Updates.clear();
6052 }
6053 removeUnwindEdge(TI->getParent(), DTU);
6054 Changed = true;
6055 continue;
6056 }
6057
6058 for (CatchSwitchInst::handler_iterator I = CSI->handler_begin(),
6059 E = CSI->handler_end();
6060 I != E; ++I) {
6061 if (*I == BB) {
6062 CSI->removeHandler(I);
6063 --I;
6064 --E;
6065 Changed = true;
6066 }
6067 }
6068 if (DTU)
6069 Updates.push_back({DominatorTree::Delete, Predecessor, BB});
6070 if (CSI->getNumHandlers() == 0) {
6071 if (CSI->hasUnwindDest()) {
6072 // Redirect all predecessors of the block containing CatchSwitchInst
6073 // to instead branch to the CatchSwitchInst's unwind destination.
6074 if (DTU) {
6075 for (auto *PredecessorOfPredecessor : predecessors(Predecessor)) {
6076 Updates.push_back({DominatorTree::Insert,
6077 PredecessorOfPredecessor,
6078 CSI->getUnwindDest()});
6079 Updates.push_back({DominatorTree::Delete,
6080 PredecessorOfPredecessor, Predecessor});
6081 }
6082 }
6083 Predecessor->replaceAllUsesWith(CSI->getUnwindDest());
6084 } else {
6085 // Rewrite all preds to unwind to caller (or from invoke to call).
6086 if (DTU) {
6087 DTU->applyUpdates(Updates);
6088 Updates.clear();
6089 }
6090 SmallVector<BasicBlock *, 8> EHPreds(predecessors(Predecessor));
6091 for (BasicBlock *EHPred : EHPreds)
6092 removeUnwindEdge(EHPred, DTU);
6093 }
6094 // The catchswitch is no longer reachable.
6095 new UnreachableInst(CSI->getContext(), CSI->getIterator());
6096 CSI->eraseFromParent();
6097 Changed = true;
6098 }
6099 } else if (auto *CRI = dyn_cast<CleanupReturnInst>(TI)) {
6100 (void)CRI;
6101 assert(CRI->hasUnwindDest() && CRI->getUnwindDest() == BB &&
6102 "Expected to always have an unwind to BB.");
6103 if (DTU)
6104 Updates.push_back({DominatorTree::Delete, Predecessor, BB});
6105 new UnreachableInst(TI->getContext(), TI->getIterator());
6106 TI->eraseFromParent();
6107 Changed = true;
6108 }
6109 }
6110
6111 if (DTU)
6112 DTU->applyUpdates(Updates);
6113
6114 // If this block is now dead, remove it.
6115 if (pred_empty(BB) && BB != &BB->getParent()->getEntryBlock()) {
6116 DeleteDeadBlock(BB, DTU);
6117 return true;
6118 }
6119
6120 return Changed;
6121}
6122
6131
6132static std::optional<ContiguousCasesResult>
6135 BasicBlock *Dest, BasicBlock *OtherDest) {
6136 assert(Cases.size() >= 1);
6137
6139 const APInt &Min = Cases.back()->getValue();
6140 const APInt &Max = Cases.front()->getValue();
6141 APInt Offset = Max - Min;
6142 size_t ContiguousOffset = Cases.size() - 1;
6143 if (Offset == ContiguousOffset) {
6144 return ContiguousCasesResult{
6145 /*Min=*/Cases.back(),
6146 /*Max=*/Cases.front(),
6147 /*Dest=*/Dest,
6148 /*OtherDest=*/OtherDest,
6149 /*Cases=*/&Cases,
6150 /*OtherCases=*/&OtherCases,
6151 };
6152 }
6153 ConstantRange CR = computeConstantRange(Condition, /*ForSigned=*/false,
6154 SimplifyQuery(Dest->getDataLayout()));
6155 // If this is a wrapping contiguous range, that is, [Min, OtherMin] +
6156 // [OtherMax, Max] (also [OtherMax, OtherMin]), [OtherMin+1, OtherMax-1] is a
6157 // contiguous range for the other destination. N.B. If CR is not a full range,
6158 // Max+1 is not equal to Min. It's not continuous in arithmetic.
6159 if (Max == CR.getUnsignedMax() && Min == CR.getUnsignedMin()) {
6160 assert(Cases.size() >= 2);
6161 auto *It =
6162 std::adjacent_find(Cases.begin(), Cases.end(), [](auto L, auto R) {
6163 return L->getValue() != R->getValue() + 1;
6164 });
6165 if (It == Cases.end())
6166 return std::nullopt;
6167 auto [OtherMax, OtherMin] = std::make_pair(*It, *std::next(It));
6168 if ((Max - OtherMax->getValue()) + (OtherMin->getValue() - Min) ==
6169 Cases.size() - 2) {
6170 return ContiguousCasesResult{
6171 /*Min=*/cast<ConstantInt>(
6172 ConstantInt::get(OtherMin->getType(), OtherMin->getValue() + 1)),
6173 /*Max=*/
6175 ConstantInt::get(OtherMax->getType(), OtherMax->getValue() - 1)),
6176 /*Dest=*/OtherDest,
6177 /*OtherDest=*/Dest,
6178 /*Cases=*/&OtherCases,
6179 /*OtherCases=*/&Cases,
6180 };
6181 }
6182 }
6183 return std::nullopt;
6184}
6185
6187 DomTreeUpdater *DTU,
6188 bool RemoveOrigDefaultBlock = true) {
6189 LLVM_DEBUG(dbgs() << "SimplifyCFG: switch default is dead.\n");
6190 auto *BB = Switch->getParent();
6191 auto *OrigDefaultBlock = Switch->getDefaultDest();
6192 if (RemoveOrigDefaultBlock)
6193 OrigDefaultBlock->removePredecessor(BB);
6194 BasicBlock *NewDefaultBlock = BasicBlock::Create(
6195 BB->getContext(), BB->getName() + ".unreachabledefault", BB->getParent(),
6196 OrigDefaultBlock);
6197 auto *UI = new UnreachableInst(Switch->getContext(), NewDefaultBlock);
6199 Switch->setDefaultDest(&*NewDefaultBlock);
6200 if (DTU) {
6202 Updates.push_back({DominatorTree::Insert, BB, &*NewDefaultBlock});
6203 if (RemoveOrigDefaultBlock &&
6204 !is_contained(successors(BB), OrigDefaultBlock))
6205 Updates.push_back({DominatorTree::Delete, BB, &*OrigDefaultBlock});
6206 DTU->applyUpdates(Updates);
6207 }
6208}
6209
6210/// Turn a switch into an integer range comparison and branch.
6211/// Switches with more than 2 destinations are ignored.
6212/// Switches with 1 destination are also ignored.
6213bool SimplifyCFGOpt::turnSwitchRangeIntoICmp(SwitchInst *SI,
6214 IRBuilder<> &Builder) {
6215 assert(SI->getNumCases() > 1 && "Degenerate switch?");
6216
6217 bool HasDefault = !SI->defaultDestUnreachable();
6218
6219 auto *BB = SI->getParent();
6220 // Partition the cases into two sets with different destinations.
6221 BasicBlock *DestA = HasDefault ? SI->getDefaultDest() : nullptr;
6222 BasicBlock *DestB = nullptr;
6225
6226 for (auto Case : SI->cases()) {
6227 BasicBlock *Dest = Case.getCaseSuccessor();
6228 if (!DestA)
6229 DestA = Dest;
6230 if (Dest == DestA) {
6231 CasesA.push_back(Case.getCaseValue());
6232 continue;
6233 }
6234 if (!DestB)
6235 DestB = Dest;
6236 if (Dest == DestB) {
6237 CasesB.push_back(Case.getCaseValue());
6238 continue;
6239 }
6240 return false; // More than two destinations.
6241 }
6242 if (!DestB)
6243 return false; // All destinations are the same and the default is unreachable
6244
6245 assert(DestA && DestB &&
6246 "Single-destination switch should have been folded.");
6247 assert(DestA != DestB);
6248 assert(DestB != SI->getDefaultDest());
6249 assert(!CasesB.empty() && "There must be non-default cases.");
6250 assert(!CasesA.empty() || HasDefault);
6251
6252 // Figure out if one of the sets of cases form a contiguous range.
6253 std::optional<ContiguousCasesResult> ContiguousCases;
6254
6255 // Only one icmp is needed when there is only one case.
6256 if (!HasDefault && CasesA.size() == 1)
6257 ContiguousCases = ContiguousCasesResult{
6258 /*Min=*/CasesA[0],
6259 /*Max=*/CasesA[0],
6260 /*Dest=*/DestA,
6261 /*OtherDest=*/DestB,
6262 /*Cases=*/&CasesA,
6263 /*OtherCases=*/&CasesB,
6264 };
6265 else if (CasesB.size() == 1)
6266 ContiguousCases = ContiguousCasesResult{
6267 /*Min=*/CasesB[0],
6268 /*Max=*/CasesB[0],
6269 /*Dest=*/DestB,
6270 /*OtherDest=*/DestA,
6271 /*Cases=*/&CasesB,
6272 /*OtherCases=*/&CasesA,
6273 };
6274 // Correctness: Cases to the default destination cannot be contiguous cases.
6275 else if (!HasDefault)
6276 ContiguousCases =
6277 findContiguousCases(SI->getCondition(), CasesA, CasesB, DestA, DestB);
6278
6279 if (!ContiguousCases)
6280 ContiguousCases =
6281 findContiguousCases(SI->getCondition(), CasesB, CasesA, DestB, DestA);
6282
6283 if (!ContiguousCases)
6284 return false;
6285
6286 auto [Min, Max, Dest, OtherDest, Cases, OtherCases] = *ContiguousCases;
6287
6288 // Start building the compare and branch.
6289
6291 Constant *NumCases = ConstantInt::get(Offset->getType(),
6292 Max->getValue() - Min->getValue() + 1);
6293 Instruction *NewBI;
6294 if (NumCases->isOneValue()) {
6295 assert(Max->getValue() == Min->getValue());
6296 Value *Cmp = Builder.CreateICmpEQ(SI->getCondition(), Min);
6297 NewBI = Builder.CreateCondBr(Cmp, Dest, OtherDest);
6298 }
6299 // If NumCases overflowed, then all possible values jump to the successor.
6300 else if (NumCases->isNullValue() && !Cases->empty()) {
6301 NewBI = Builder.CreateBr(Dest);
6302 } else {
6303 Value *Sub = SI->getCondition();
6304 if (!Offset->isNullValue())
6305 Sub = Builder.CreateAdd(Sub, Offset, Sub->getName() + ".off");
6306 Value *Cmp = Builder.CreateICmpULT(Sub, NumCases, "switch");
6307 NewBI = Builder.CreateCondBr(Cmp, Dest, OtherDest);
6308 }
6309
6310 // Update weight for the newly-created conditional branch.
6311 if (hasBranchWeightMD(*SI) && isa<CondBrInst>(NewBI)) {
6312 SmallVector<uint64_t, 8> Weights;
6313 getBranchWeights(SI, Weights);
6314 if (Weights.size() == 1 + SI->getNumCases()) {
6315 uint64_t TrueWeight = 0;
6316 uint64_t FalseWeight = 0;
6317 for (size_t I = 0, E = Weights.size(); I != E; ++I) {
6318 if (SI->getSuccessor(I) == Dest)
6319 TrueWeight += Weights[I];
6320 else
6321 FalseWeight += Weights[I];
6322 }
6323 while (TrueWeight > UINT32_MAX || FalseWeight > UINT32_MAX) {
6324 TrueWeight /= 2;
6325 FalseWeight /= 2;
6326 }
6327 setFittedBranchWeights(*NewBI, {TrueWeight, FalseWeight},
6328 /*IsExpected=*/false, /*ElideAllZero=*/true);
6329 }
6330 }
6331
6332 // Prune obsolete incoming values off the successors' PHI nodes.
6333 for (auto &PHI : make_early_inc_range(Dest->phis())) {
6334 unsigned PreviousEdges = Cases->size();
6335 if (Dest == SI->getDefaultDest())
6336 ++PreviousEdges;
6337 for (unsigned I = 0, E = PreviousEdges - 1; I != E; ++I)
6338 PHI.removeIncomingValue(SI->getParent());
6339 }
6340 for (auto &PHI : make_early_inc_range(OtherDest->phis())) {
6341 unsigned PreviousEdges = OtherCases->size();
6342 if (OtherDest == SI->getDefaultDest())
6343 ++PreviousEdges;
6344 unsigned E = PreviousEdges - 1;
6345 // Remove all incoming values from OtherDest if OtherDest is unreachable.
6346 if (isa<UncondBrInst>(NewBI))
6347 ++E;
6348 for (unsigned I = 0; I != E; ++I)
6349 PHI.removeIncomingValue(SI->getParent());
6350 }
6351
6352 // Clean up the default block.
6353 SmallVector<DominatorTree::UpdateType, 2> Updates;
6354 if (!HasDefault) {
6355 BasicBlock *OrigDefaultBlock = SI->getDefaultDest();
6356 OrigDefaultBlock->removePredecessor(BB);
6357 Updates.push_back({DominatorTree::Delete, BB, OrigDefaultBlock});
6358 }
6359
6360 // Drop the switch.
6361 SI->eraseFromParent();
6362
6363 if (isa<UncondBrInst>(NewBI))
6364 Updates.push_back({DominatorTree::Delete, BB, OtherDest});
6365
6366 if (DTU)
6367 DTU->applyUpdates(Updates);
6368 return true;
6369}
6370
6371/// Compute masked bits for the condition of a switch
6372/// and use it to remove dead cases.
6374 AssumptionCache *AC,
6375 const DataLayout &DL) {
6376 Value *Cond = SI->getCondition();
6379 bool IsKnownValuesValid = collectPossibleValues(Cond, KnownValues, 4);
6380
6381 // We can also eliminate cases by determining that their values are outside of
6382 // the limited range of the condition based on how many significant (non-sign)
6383 // bits are in the condition value.
6384 unsigned MaxSignificantBitsInCond =
6386
6387 // Gather dead cases.
6389 SmallDenseMap<BasicBlock *, int, 8> NumPerSuccessorCases;
6390 SmallVector<BasicBlock *, 8> UniqueSuccessors;
6391 for (const auto &Case : SI->cases()) {
6392 auto *Successor = Case.getCaseSuccessor();
6393 if (DTU) {
6394 auto [It, Inserted] = NumPerSuccessorCases.try_emplace(Successor);
6395 if (Inserted)
6396 UniqueSuccessors.push_back(Successor);
6397 ++It->second;
6398 }
6399 ConstantInt *CaseC = Case.getCaseValue();
6400 const APInt &CaseVal = CaseC->getValue();
6401 if (Known.Zero.intersects(CaseVal) || !Known.One.isSubsetOf(CaseVal) ||
6402 (CaseVal.getSignificantBits() > MaxSignificantBitsInCond) ||
6403 (IsKnownValuesValid && !KnownValues.contains(CaseC))) {
6404 DeadCases.push_back(CaseC);
6405 if (DTU)
6406 --NumPerSuccessorCases[Successor];
6407 LLVM_DEBUG(dbgs() << "SimplifyCFG: switch case " << CaseVal
6408 << " is dead.\n");
6409 } else if (IsKnownValuesValid)
6410 KnownValues.erase(CaseC);
6411 }
6412
6413 // If we can prove that the cases must cover all possible values, the
6414 // default destination becomes dead and we can remove it. If we know some
6415 // of the bits in the value, we can use that to more precisely compute the
6416 // number of possible unique case values.
6417 bool HasDefault = !SI->defaultDestUnreachable();
6418 const unsigned NumUnknownBits =
6419 Known.getBitWidth() - (Known.Zero | Known.One).popcount();
6420 assert(NumUnknownBits <= Known.getBitWidth());
6421 if (HasDefault && DeadCases.empty()) {
6422 if (IsKnownValuesValid && all_of(KnownValues, IsaPred<UndefValue>)) {
6424 return true;
6425 }
6426
6427 if (NumUnknownBits < 64 /* avoid overflow */) {
6428 uint64_t AllNumCases = 1ULL << NumUnknownBits;
6429 if (SI->getNumCases() == AllNumCases) {
6431 return true;
6432 }
6433 // When only one case value is missing, replace default with that case.
6434 // Eliminating the default branch will provide more opportunities for
6435 // optimization, such as lookup tables.
6436 if (SI->getNumCases() == AllNumCases - 1) {
6437 assert(NumUnknownBits > 1 && "Should be canonicalized to a branch");
6438 IntegerType *CondTy = cast<IntegerType>(Cond->getType());
6439 if (CondTy->getIntegerBitWidth() > 64 ||
6440 !DL.fitsInLegalInteger(CondTy->getIntegerBitWidth()))
6441 return false;
6442
6443 uint64_t MissingCaseVal = 0;
6444 for (const auto &Case : SI->cases())
6445 MissingCaseVal ^= Case.getCaseValue()->getValue().getLimitedValue();
6446 auto *MissingCase = cast<ConstantInt>(
6447 ConstantInt::get(Cond->getType(), MissingCaseVal));
6449 SIW.addCase(MissingCase, SI->getDefaultDest(),
6450 SIW.getSuccessorWeight(0));
6452 /*RemoveOrigDefaultBlock*/ false);
6453 SIW.setSuccessorWeight(0, 0);
6454 return true;
6455 }
6456 }
6457 }
6458
6459 if (DeadCases.empty())
6460 return false;
6461
6463 for (ConstantInt *DeadCase : DeadCases) {
6464 SwitchInst::CaseIt CaseI = SI->findCaseValue(DeadCase);
6465 assert(CaseI != SI->case_default() &&
6466 "Case was not found. Probably mistake in DeadCases forming.");
6467 // Prune unused values from PHI nodes.
6468 CaseI->getCaseSuccessor()->removePredecessor(SI->getParent());
6469 SIW.removeCase(CaseI);
6470 }
6471
6472 if (DTU) {
6473 std::vector<DominatorTree::UpdateType> Updates;
6474 for (auto *Successor : UniqueSuccessors)
6475 if (NumPerSuccessorCases[Successor] == 0)
6476 Updates.push_back({DominatorTree::Delete, SI->getParent(), Successor});
6477 DTU->applyUpdates(Updates);
6478 }
6479
6480 return true;
6481}
6482
6483/// If BB would be eligible for simplification by
6484/// TryToSimplifyUncondBranchFromEmptyBlock (i.e. it is empty and terminated
6485/// by an unconditional branch), look at the phi node for BB in the successor
6486/// block and see if the incoming value is equal to CaseValue. If so, return
6487/// the phi node, and set PhiIndex to BB's index in the phi node.
6489 BasicBlock *BB, int *PhiIndex) {
6490 if (&*BB->getFirstNonPHIIt() != BB->getTerminator())
6491 return nullptr; // BB must be empty to be a candidate for simplification.
6492 if (!BB->getSinglePredecessor())
6493 return nullptr; // BB must be dominated by the switch.
6494
6496 if (!Branch)
6497 return nullptr; // Terminator must be unconditional branch.
6498
6499 BasicBlock *Succ = Branch->getSuccessor();
6500
6501 for (PHINode &PHI : Succ->phis()) {
6502 int Idx = PHI.getBasicBlockIndex(BB);
6503 assert(Idx >= 0 && "PHI has no entry for predecessor?");
6504
6505 Value *InValue = PHI.getIncomingValue(Idx);
6506 if (InValue != CaseValue)
6507 continue;
6508
6509 *PhiIndex = Idx;
6510 return &PHI;
6511 }
6512
6513 return nullptr;
6514}
6515
6516/// Try to forward the condition of a switch instruction to a phi node
6517/// dominated by the switch, if that would mean that some of the destination
6518/// blocks of the switch can be folded away. Return true if a change is made.
6520 using ForwardingNodesMap = DenseMap<PHINode *, SmallVector<int, 4>>;
6521
6522 ForwardingNodesMap ForwardingNodes;
6523 BasicBlock *SwitchBlock = SI->getParent();
6524 bool Changed = false;
6525 for (const auto &Case : SI->cases()) {
6526 ConstantInt *CaseValue = Case.getCaseValue();
6527 BasicBlock *CaseDest = Case.getCaseSuccessor();
6528
6529 // Replace phi operands in successor blocks that are using the constant case
6530 // value rather than the switch condition variable:
6531 // switchbb:
6532 // switch i32 %x, label %default [
6533 // i32 17, label %succ
6534 // ...
6535 // succ:
6536 // %r = phi i32 ... [ 17, %switchbb ] ...
6537 // -->
6538 // %r = phi i32 ... [ %x, %switchbb ] ...
6539
6540 for (PHINode &Phi : CaseDest->phis()) {
6541 // This only works if there is exactly 1 incoming edge from the switch to
6542 // a phi. If there is >1, that means multiple cases of the switch map to 1
6543 // value in the phi, and that phi value is not the switch condition. Thus,
6544 // this transform would not make sense (the phi would be invalid because
6545 // a phi can't have different incoming values from the same block).
6546 int SwitchBBIdx = Phi.getBasicBlockIndex(SwitchBlock);
6547 if (Phi.getIncomingValue(SwitchBBIdx) == CaseValue &&
6548 count(Phi.blocks(), SwitchBlock) == 1) {
6549 Phi.setIncomingValue(SwitchBBIdx, SI->getCondition());
6550 Changed = true;
6551 }
6552 }
6553
6554 // Collect phi nodes that are indirectly using this switch's case constants.
6555 int PhiIdx;
6556 if (auto *Phi = findPHIForConditionForwarding(CaseValue, CaseDest, &PhiIdx))
6557 ForwardingNodes[Phi].push_back(PhiIdx);
6558 }
6559
6560 for (auto &ForwardingNode : ForwardingNodes) {
6561 PHINode *Phi = ForwardingNode.first;
6562 SmallVectorImpl<int> &Indexes = ForwardingNode.second;
6563 // Check if it helps to fold PHI.
6564 if (Indexes.size() < 2 && !llvm::is_contained(Phi->incoming_values(), SI->getCondition()))
6565 continue;
6566
6567 for (int Index : Indexes)
6568 Phi->setIncomingValue(Index, SI->getCondition());
6569 Changed = true;
6570 }
6571
6572 return Changed;
6573}
6574
6575/// Return true if the backend will be able to handle
6576/// initializing an array of constants like C.
6578 if (C->isThreadDependent())
6579 return false;
6580 if (C->isDLLImportDependent())
6581 return false;
6582
6585 return false;
6586
6587 // Globals cannot contain scalable types.
6588 if (C->getType()->isScalableTy())
6589 return false;
6590
6592 // Pointer casts and in-bounds GEPs will not prohibit the backend from
6593 // materializing the array of constants.
6594 Constant *StrippedC = cast<Constant>(CE->stripInBoundsConstantOffsets());
6595 if (StrippedC == C || !validLookupTableConstant(StrippedC, TTI))
6596 return false;
6597 }
6598
6599 if (!TTI.shouldBuildLookupTablesForConstant(C))
6600 return false;
6601
6602 return true;
6603}
6604
6605/// If V is a Constant, return it. Otherwise, try to look up
6606/// its constant value in ConstantPool, returning 0 if it's not there.
6607static Constant *
6610 if (Constant *C = dyn_cast<Constant>(V))
6611 return C;
6612 return ConstantPool.lookup(V);
6613}
6614
6615/// Try to fold instruction I into a constant. This works for
6616/// simple instructions such as binary operations where both operands are
6617/// constant or can be replaced by constants from the ConstantPool. Returns the
6618/// resulting constant on success, 0 otherwise.
6619static Constant *
6623 Constant *A = lookupConstant(Select->getCondition(), ConstantPool);
6624 if (!A)
6625 return nullptr;
6626 if (A->isAllOnesValue())
6627 return lookupConstant(Select->getTrueValue(), ConstantPool);
6628 if (A->isNullValue())
6629 return lookupConstant(Select->getFalseValue(), ConstantPool);
6630 return nullptr;
6631 }
6632
6634 for (unsigned N = 0, E = I->getNumOperands(); N != E; ++N) {
6635 if (Constant *A = lookupConstant(I->getOperand(N), ConstantPool))
6636 COps.push_back(A);
6637 else
6638 return nullptr;
6639 }
6640
6641 return ConstantFoldInstOperands(I, COps, DL);
6642}
6643
6644/// Try to determine the resulting constant values in phi nodes
6645/// at the common destination basic block, *CommonDest, for one of the case
6646/// destinations CaseDest corresponding to value CaseVal (nullptr for the
6647/// default case), of a switch instruction SI.
6648static bool
6650 BasicBlock **CommonDest,
6651 SmallVectorImpl<std::pair<PHINode *, Constant *>> &Res,
6652 const DataLayout &DL) {
6653 // The block from which we enter the common destination.
6654 BasicBlock *Pred = SI->getParent();
6655
6656 // If CaseDest is empty except for some side-effect free instructions through
6657 // which we can constant-propagate the CaseVal, continue to its successor.
6659 ConstantPool.insert(std::make_pair(SI->getCondition(), CaseVal));
6660 for (Instruction &I : *CaseDest) {
6661 if (I.isTerminator()) {
6662 // If the terminator is a simple branch, continue to the next block.
6663 if (I.getNumSuccessors() != 1 || I.isSpecialTerminator())
6664 return false;
6665 Pred = CaseDest;
6666 CaseDest = I.getSuccessor(0);
6667 } else if (Constant *C = constantFold(&I, DL, ConstantPool)) {
6668 // Instruction is side-effect free and constant.
6669
6670 // If the instruction has uses outside this block or a phi node slot for
6671 // the block, it is not safe to bypass the instruction since it would then
6672 // no longer dominate all its uses.
6673 for (auto &Use : I.uses()) {
6674 User *User = Use.getUser();
6676 if (I->getParent() == CaseDest)
6677 continue;
6678 if (PHINode *Phi = dyn_cast<PHINode>(User))
6679 if (Phi->getIncomingBlock(Use) == CaseDest)
6680 continue;
6681 return false;
6682 }
6683
6684 ConstantPool.insert(std::make_pair(&I, C));
6685 } else {
6686 break;
6687 }
6688 }
6689
6690 // If we did not have a CommonDest before, use the current one.
6691 if (!*CommonDest)
6692 *CommonDest = CaseDest;
6693 // If the destination isn't the common one, abort.
6694 if (CaseDest != *CommonDest)
6695 return false;
6696
6697 // Get the values for this case from phi nodes in the destination block.
6698 for (PHINode &PHI : (*CommonDest)->phis()) {
6699 int Idx = PHI.getBasicBlockIndex(Pred);
6700 if (Idx == -1)
6701 continue;
6702
6703 Constant *ConstVal =
6704 lookupConstant(PHI.getIncomingValue(Idx), ConstantPool);
6705 if (!ConstVal)
6706 return false;
6707
6708 Res.push_back(std::make_pair(&PHI, ConstVal));
6709 }
6710
6711 return Res.size() > 0;
6712}
6713
6714// Helper function used to add CaseVal to the list of cases that generate
6715// Result. Returns the updated number of cases that generate this result.
6716static size_t mapCaseToResult(ConstantInt *CaseVal,
6717 SwitchCaseResultVectorTy &UniqueResults,
6718 Constant *Result) {
6719 for (auto &I : UniqueResults) {
6720 if (I.first == Result) {
6721 I.second.push_back(CaseVal);
6722 return I.second.size();
6723 }
6724 }
6725 UniqueResults.push_back(
6726 std::make_pair(Result, SmallVector<ConstantInt *, 4>(1, CaseVal)));
6727 return 1;
6728}
6729
6730// Helper function that initializes a map containing
6731// results for the PHI node of the common destination block for a switch
6732// instruction. Returns false if multiple PHI nodes have been found or if
6733// there is not a common destination block for the switch.
6735 BasicBlock *&CommonDest,
6736 SwitchCaseResultVectorTy &UniqueResults,
6737 Constant *&DefaultResult,
6738 const DataLayout &DL,
6739 uintptr_t MaxUniqueResults) {
6740 for (const auto &I : SI->cases()) {
6741 ConstantInt *CaseVal = I.getCaseValue();
6742
6743 // Resulting value at phi nodes for this case value.
6744 SwitchCaseResultsTy Results;
6745 if (!getCaseResults(SI, CaseVal, I.getCaseSuccessor(), &CommonDest, Results,
6746 DL))
6747 return false;
6748
6749 // Only one value per case is permitted.
6750 if (Results.size() > 1)
6751 return false;
6752
6753 // Add the case->result mapping to UniqueResults.
6754 const size_t NumCasesForResult =
6755 mapCaseToResult(CaseVal, UniqueResults, Results.begin()->second);
6756
6757 // Early out if there are too many cases for this result.
6758 if (NumCasesForResult > MaxSwitchCasesPerResult)
6759 return false;
6760
6761 // Early out if there are too many unique results.
6762 if (UniqueResults.size() > MaxUniqueResults)
6763 return false;
6764
6765 // Check the PHI consistency.
6766 if (!PHI)
6767 PHI = Results[0].first;
6768 else if (PHI != Results[0].first)
6769 return false;
6770 }
6771 // Find the default result value.
6773 getCaseResults(SI, nullptr, SI->getDefaultDest(), &CommonDest, DefaultResults,
6774 DL);
6775 // If the default value is not found abort unless the default destination
6776 // is unreachable.
6777 DefaultResult =
6778 DefaultResults.size() == 1 ? DefaultResults.begin()->second : nullptr;
6779
6780 return DefaultResult || SI->defaultDestUnreachable();
6781}
6782
6783// Helper function that checks if it is possible to transform a switch with only
6784// two cases (or two cases + default) that produces a result into a select.
6785// TODO: Handle switches with more than 2 cases that map to the same result.
6786// The branch weights correspond to the provided Condition (i.e. if Condition is
6787// modified from the original SwitchInst, the caller must adjust the weights)
6788static Value *foldSwitchToSelect(const SwitchCaseResultVectorTy &ResultVector,
6789 Constant *DefaultResult, Value *Condition,
6790 IRBuilder<> &Builder, const DataLayout &DL,
6791 ArrayRef<uint32_t> BranchWeights) {
6792 // If we are selecting between only two cases transform into a simple
6793 // select or a two-way select if default is possible.
6794 // Example:
6795 // switch (a) { %0 = icmp eq i32 %a, 10
6796 // case 10: return 42; %1 = select i1 %0, i32 42, i32 4
6797 // case 20: return 2; ----> %2 = icmp eq i32 %a, 20
6798 // default: return 4; %3 = select i1 %2, i32 2, i32 %1
6799 // }
6800
6801 const bool HasBranchWeights = !BranchWeights.empty();
6802
6803 if (ResultVector.size() == 2 && ResultVector[0].second.size() == 1 &&
6804 ResultVector[1].second.size() == 1) {
6805 ConstantInt *FirstCase = ResultVector[0].second[0];
6806 ConstantInt *SecondCase = ResultVector[1].second[0];
6807 Value *SelectValue = ResultVector[1].first;
6808 if (DefaultResult) {
6809 Value *ValueCompare =
6810 Builder.CreateICmpEQ(Condition, SecondCase, "switch.selectcmp");
6811 SelectValue = Builder.CreateSelect(ValueCompare, ResultVector[1].first,
6812 DefaultResult, "switch.select");
6813 if (auto *SI = dyn_cast<SelectInst>(SelectValue);
6814 SI && HasBranchWeights) {
6815 // We start with 3 probabilities, where the numerator is the
6816 // corresponding BranchWeights[i], and the denominator is the sum over
6817 // BranchWeights. We want the probability and negative probability of
6818 // Condition == SecondCase.
6819 assert(BranchWeights.size() == 3);
6821 *SI, {BranchWeights[2], BranchWeights[0] + BranchWeights[1]},
6822 /*IsExpected=*/false, /*ElideAllZero=*/true);
6823 }
6824 }
6825 Value *ValueCompare =
6826 Builder.CreateICmpEQ(Condition, FirstCase, "switch.selectcmp");
6827 Value *Ret = Builder.CreateSelect(ValueCompare, ResultVector[0].first,
6828 SelectValue, "switch.select");
6829 if (auto *SI = dyn_cast<SelectInst>(Ret); SI && HasBranchWeights) {
6830 // We may have had a DefaultResult. Base the position of the first and
6831 // second's branch weights accordingly. Also the proability that Condition
6832 // != FirstCase needs to take that into account.
6833 assert(BranchWeights.size() >= 2);
6834 size_t FirstCasePos = (Condition != nullptr);
6835 size_t SecondCasePos = FirstCasePos + 1;
6836 uint32_t DefaultCase = (Condition != nullptr) ? BranchWeights[0] : 0;
6838 {BranchWeights[FirstCasePos],
6839 DefaultCase + BranchWeights[SecondCasePos]},
6840 /*IsExpected=*/false, /*ElideAllZero=*/true);
6841 }
6842 return Ret;
6843 }
6844
6845 // Handle the degenerate case where two cases have the same result value.
6846 if (ResultVector.size() == 1 && DefaultResult) {
6847 ArrayRef<ConstantInt *> CaseValues = ResultVector[0].second;
6848 unsigned CaseCount = CaseValues.size();
6849 // n bits group cases map to the same result:
6850 // case 0,4 -> Cond & 0b1..1011 == 0 ? result : default
6851 // case 0,2,4,6 -> Cond & 0b1..1001 == 0 ? result : default
6852 // case 0,2,8,10 -> Cond & 0b1..0101 == 0 ? result : default
6853 if (isPowerOf2_32(CaseCount)) {
6854 ConstantInt *MinCaseVal = CaseValues[0];
6855 // If there are bits that are set exclusively by CaseValues, we
6856 // can transform the switch into a select if the conjunction of
6857 // all the values uniquely identify CaseValues.
6858 APInt AndMask = APInt::getAllOnes(MinCaseVal->getBitWidth());
6859
6860 // Find the minimum value and compute the and of all the case values.
6861 for (auto *Case : CaseValues) {
6862 if (Case->getValue().slt(MinCaseVal->getValue()))
6863 MinCaseVal = Case;
6864 AndMask &= Case->getValue();
6865 }
6866 KnownBits Known = computeKnownBits(Condition, DL);
6867
6868 if (!AndMask.isZero() && Known.getMaxValue().uge(AndMask)) {
6869 // Compute the number of bits that are free to vary.
6870 unsigned FreeBits = Known.countMaxActiveBits() - AndMask.popcount();
6871
6872 // Check if the number of values covered by the mask is equal
6873 // to the number of cases.
6874 if (FreeBits == Log2_32(CaseCount)) {
6875 Value *And = Builder.CreateAnd(Condition, AndMask);
6876 Value *Cmp = Builder.CreateICmpEQ(
6877 And, Constant::getIntegerValue(And->getType(), AndMask));
6878 Value *Ret =
6879 Builder.CreateSelect(Cmp, ResultVector[0].first, DefaultResult);
6880 if (auto *SI = dyn_cast<SelectInst>(Ret); SI && HasBranchWeights) {
6881 // We know there's a Default case. We base the resulting branch
6882 // weights off its probability.
6883 assert(BranchWeights.size() >= 2);
6885 *SI,
6886 {accumulate(drop_begin(BranchWeights), 0U), BranchWeights[0]},
6887 /*IsExpected=*/false, /*ElideAllZero=*/true);
6888 }
6889 return Ret;
6890 }
6891 }
6892
6893 // Mark the bits case number touched.
6894 APInt BitMask = APInt::getZero(MinCaseVal->getBitWidth());
6895 for (auto *Case : CaseValues)
6896 BitMask |= (Case->getValue() - MinCaseVal->getValue());
6897
6898 // Check if cases with the same result can cover all number
6899 // in touched bits.
6900 if (BitMask.popcount() == Log2_32(CaseCount)) {
6901 if (!MinCaseVal->isNullValue())
6902 Condition = Builder.CreateSub(Condition, MinCaseVal);
6903 Value *And = Builder.CreateAnd(Condition, ~BitMask, "switch.and");
6904 Value *Cmp = Builder.CreateICmpEQ(
6905 And, Constant::getNullValue(And->getType()), "switch.selectcmp");
6906 Value *Ret =
6907 Builder.CreateSelect(Cmp, ResultVector[0].first, DefaultResult);
6908 if (auto *SI = dyn_cast<SelectInst>(Ret); SI && HasBranchWeights) {
6909 assert(BranchWeights.size() >= 2);
6911 *SI,
6912 {accumulate(drop_begin(BranchWeights), 0U), BranchWeights[0]},
6913 /*IsExpected=*/false, /*ElideAllZero=*/true);
6914 }
6915 return Ret;
6916 }
6917 }
6918
6919 // Handle the degenerate case where two cases have the same value.
6920 if (CaseValues.size() == 2) {
6921 Value *Cmp1 = Builder.CreateICmpEQ(Condition, CaseValues[0],
6922 "switch.selectcmp.case1");
6923 Value *Cmp2 = Builder.CreateICmpEQ(Condition, CaseValues[1],
6924 "switch.selectcmp.case2");
6925 Value *Cmp = Builder.CreateOr(Cmp1, Cmp2, "switch.selectcmp");
6926 Value *Ret =
6927 Builder.CreateSelect(Cmp, ResultVector[0].first, DefaultResult);
6928 if (auto *SI = dyn_cast<SelectInst>(Ret); SI && HasBranchWeights) {
6929 assert(BranchWeights.size() >= 2);
6931 *SI, {accumulate(drop_begin(BranchWeights), 0U), BranchWeights[0]},
6932 /*IsExpected=*/false, /*ElideAllZero=*/true);
6933 }
6934 return Ret;
6935 }
6936 }
6937
6938 return nullptr;
6939}
6940
6941// Helper function to cleanup a switch instruction that has been converted into
6942// a select, fixing up PHI nodes and basic blocks.
6944 Value *SelectValue,
6945 IRBuilder<> &Builder,
6946 DomTreeUpdater *DTU) {
6947 std::vector<DominatorTree::UpdateType> Updates;
6948
6949 BasicBlock *SelectBB = SI->getParent();
6950 BasicBlock *DestBB = PHI->getParent();
6951
6952 if (DTU && !is_contained(predecessors(DestBB), SelectBB))
6953 Updates.push_back({DominatorTree::Insert, SelectBB, DestBB});
6954 Builder.CreateBr(DestBB);
6955
6956 // Remove the switch.
6957
6958 PHI->removeIncomingValueIf(
6959 [&](unsigned Idx) { return PHI->getIncomingBlock(Idx) == SelectBB; });
6960 PHI->addIncoming(SelectValue, SelectBB);
6961
6962 SmallPtrSet<BasicBlock *, 4> RemovedSuccessors;
6963 for (unsigned i = 0, e = SI->getNumSuccessors(); i < e; ++i) {
6964 BasicBlock *Succ = SI->getSuccessor(i);
6965
6966 if (Succ == DestBB)
6967 continue;
6968 Succ->removePredecessor(SelectBB);
6969 if (DTU && RemovedSuccessors.insert(Succ).second)
6970 Updates.push_back({DominatorTree::Delete, SelectBB, Succ});
6971 }
6972 SI->eraseFromParent();
6973 if (DTU)
6974 DTU->applyUpdates(Updates);
6975}
6976
6977/// If a switch is only used to initialize one or more phi nodes in a common
6978/// successor block with only two different constant values, try to replace the
6979/// switch with a select. Returns true if the fold was made.
6981 DomTreeUpdater *DTU, const DataLayout &DL) {
6982 Value *const Cond = SI->getCondition();
6983 PHINode *PHI = nullptr;
6984 BasicBlock *CommonDest = nullptr;
6985 Constant *DefaultResult;
6986 SwitchCaseResultVectorTy UniqueResults;
6987 // Collect all the cases that will deliver the same value from the switch.
6988 if (!initializeUniqueCases(SI, PHI, CommonDest, UniqueResults, DefaultResult,
6989 DL, /*MaxUniqueResults*/ 2))
6990 return false;
6991
6992 assert(PHI != nullptr && "PHI for value select not found");
6993 Builder.SetInsertPoint(SI);
6994 SmallVector<uint32_t, 4> BranchWeights;
6995 [[maybe_unused]] auto HasWeights =
6997 assert(!HasWeights == (BranchWeights.empty()));
6998 assert(BranchWeights.empty() ||
6999 (BranchWeights.size() >=
7000 UniqueResults.size() + (DefaultResult != nullptr)));
7001
7002 Value *SelectValue = foldSwitchToSelect(UniqueResults, DefaultResult, Cond,
7003 Builder, DL, BranchWeights);
7004 if (!SelectValue)
7005 return false;
7006
7007 removeSwitchAfterSelectFold(SI, PHI, SelectValue, Builder, DTU);
7008 return true;
7009}
7010
7011namespace {
7012
7013/// This class finds alternatives for switches to ultimately
7014/// replace the switch.
7015class SwitchReplacement {
7016public:
7017 /// Create a helper for optimizations to use as a switch replacement.
7018 /// Find a better representation for the content of Values,
7019 /// using DefaultValue to fill any holes in the table.
7020 /// If no representation is possible, isValid() returns false.
7021 SwitchReplacement(
7022 Module &M, uint64_t TableSize, ConstantInt *Offset,
7023 const SmallVectorImpl<std::pair<ConstantInt *, Constant *>> &Values,
7024 Constant *DefaultValue, const DataLayout &DL,
7025 const TargetTransformInfo &TTI, const StringRef &FuncName);
7026
7027 /// Build instructions with Builder to retrieve values using Index
7028 /// and replace the switch.
7029 Value *replaceSwitch(Value *Index, IRBuilder<> &Builder, const DataLayout &DL,
7030 Function *Func);
7031
7032 /// Return true if a table with TableSize elements of
7033 /// type ElementType would fit in a target-legal register.
7034 static bool wouldFitInRegister(const DataLayout &DL, uint64_t TableSize,
7035 Type *ElementType);
7036
7037 /// Return the default value of the switch.
7038 Constant *getDefaultValue();
7039
7040 /// Return true if the replacement is a lookup table.
7041 bool isLookupTable();
7042
7043 /// Return true if the replacement is a bit map.
7044 bool isBitMap();
7045
7046 /// Return true if a suitable switch replacement was found.
7047 bool isValid() const { return Kind != InvalidKind; }
7048
7049private:
7050 // Depending on the switch, there are different alternatives.
7051 enum {
7052 // No suitable replacement was found.
7053 InvalidKind,
7054
7055 // For switches where each case contains the same value, we just have to
7056 // store that single value and return it for each lookup.
7057 SingleValueKind,
7058
7059 // For switches where there is a linear relationship between table index
7060 // and values. We calculate the result with a simple multiplication
7061 // and addition instead of a table lookup.
7062 LinearMapKind,
7063
7064 // For small tables with integer elements, we can pack them into a bitmap
7065 // that fits into a target-legal register. Values are retrieved by
7066 // shift and mask operations.
7067 BitMapKind,
7068
7069 // The table is stored as an array of values. Values are retrieved by load
7070 // instructions from the table.
7071 LookupTableKind
7072 } Kind;
7073
7074 // The default value of the switch.
7075 Constant *DefaultValue;
7076
7077 // The type of the output values.
7078 Type *ValueType;
7079
7080 // For SingleValueKind, this is the single value.
7081 Constant *SingleValue = nullptr;
7082
7083 // For BitMapKind, this is the bitmap.
7084 ConstantInt *BitMap = nullptr;
7085 IntegerType *BitMapElementTy = nullptr;
7086
7087 // For LinearMapKind, these are the constants used to derive the value.
7088 ConstantInt *LinearOffset = nullptr;
7089 ConstantInt *LinearMultiplier = nullptr;
7090 bool LinearMapValWrapped = false;
7091
7092 // For LookupTableKind, this is the table.
7093 Constant *Initializer = nullptr;
7094};
7095
7096} // end anonymous namespace
7097
7098SwitchReplacement::SwitchReplacement(
7099 Module &M, uint64_t TableSize, ConstantInt *Offset,
7100 const SmallVectorImpl<std::pair<ConstantInt *, Constant *>> &Values,
7101 Constant *DefaultValue, const DataLayout &DL,
7102 const TargetTransformInfo &TTI, const StringRef &FuncName)
7103 : DefaultValue(DefaultValue) {
7104 assert(Values.size() && "Can't build lookup table without values!");
7105 assert(TableSize >= Values.size() && "Can't fit values in table!");
7106
7107 // If all values in the table are equal, this is that value.
7108 SingleValue = Values.begin()->second;
7109
7110 ValueType = Values.begin()->second->getType();
7111
7112 // Build up the table contents.
7113 SmallVector<Constant *, 64> TableContents(TableSize);
7114 for (const auto &[CaseVal, CaseRes] : Values) {
7115 assert(CaseRes->getType() == ValueType);
7116
7117 uint64_t Idx = (CaseVal->getValue() - Offset->getValue()).getLimitedValue();
7118 TableContents[Idx] = CaseRes;
7119
7120 if (SingleValue && !isa<PoisonValue>(CaseRes) && CaseRes != SingleValue)
7121 SingleValue = isa<PoisonValue>(SingleValue) ? CaseRes : nullptr;
7122 }
7123
7124 // Fill in any holes in the table with the default result.
7125 if (Values.size() < TableSize) {
7126 assert(DefaultValue &&
7127 "Need a default value to fill the lookup table holes.");
7128 assert(DefaultValue->getType() == ValueType);
7129 for (uint64_t I = 0; I < TableSize; ++I) {
7130 if (!TableContents[I])
7131 TableContents[I] = DefaultValue;
7132 }
7133
7134 // If the default value is poison, all the holes are poison.
7135 bool DefaultValueIsPoison = isa<PoisonValue>(DefaultValue);
7136
7137 if (DefaultValue != SingleValue && !DefaultValueIsPoison)
7138 SingleValue = nullptr;
7139 }
7140
7141 // If each element in the table contains the same value, we only need to store
7142 // that single value.
7143 if (SingleValue) {
7144 Kind = SingleValueKind;
7145 return;
7146 }
7147
7148 // Check if we can derive the value with a linear transformation from the
7149 // table index.
7151 bool LinearMappingPossible = true;
7152 APInt PrevVal;
7153 APInt DistToPrev;
7154 // When linear map is monotonic and signed overflow doesn't happen on
7155 // maximum index, we can attach nsw on Add and Mul.
7156 bool NonMonotonic = false;
7157 assert(TableSize >= 2 && "Should be a SingleValue table.");
7158 // Check if there is the same distance between two consecutive values.
7159 for (uint64_t I = 0; I < TableSize; ++I) {
7160 ConstantInt *ConstVal = dyn_cast<ConstantInt>(TableContents[I]);
7161
7162 if (!ConstVal && isa<PoisonValue>(TableContents[I])) {
7163 // This is an poison, so it's (probably) a lookup table hole.
7164 // To prevent any regressions from before we switched to using poison as
7165 // the default value, holes will fall back to using the first value.
7166 // This can be removed once we add proper handling for poisons in lookup
7167 // tables.
7168 ConstVal = dyn_cast<ConstantInt>(Values[0].second);
7169 }
7170
7171 if (!ConstVal) {
7172 // We only handle literal integers when checking for a linear mapping.
7173 LinearMappingPossible = false;
7174 break;
7175 }
7176 const APInt &Val = ConstVal->getValue();
7177 if (I != 0) {
7178 APInt Dist = Val - PrevVal;
7179 if (I == 1) {
7180 DistToPrev = Dist;
7181 } else if (Dist != DistToPrev) {
7182 LinearMappingPossible = false;
7183 break;
7184 }
7185 NonMonotonic |=
7186 Dist.isStrictlyPositive() ? Val.sle(PrevVal) : Val.sgt(PrevVal);
7187 }
7188 PrevVal = Val;
7189 }
7190 if (LinearMappingPossible) {
7191 LinearOffset = cast<ConstantInt>(TableContents[0]);
7192 LinearMultiplier = ConstantInt::get(M.getContext(), DistToPrev);
7193 APInt M = LinearMultiplier->getValue();
7194 bool MayWrap = true;
7195 if (isIntN(M.getBitWidth(), TableSize - 1))
7196 (void)M.smul_ov(APInt(M.getBitWidth(), TableSize - 1), MayWrap);
7197 LinearMapValWrapped = NonMonotonic || MayWrap;
7198 Kind = LinearMapKind;
7199 return;
7200 }
7201 }
7202
7203 // If the values are integer constants and the table fits in a register,
7204 // build a bitmap.
7205 if (wouldFitInRegister(DL, TableSize, ValueType) &&
7206 all_of(TableContents, IsaPred<ConstantInt, UndefValue>)) {
7208 APInt TableInt(TableSize * IT->getBitWidth(), 0);
7209 for (uint64_t I = TableSize; I > 0; --I) {
7210 TableInt <<= IT->getBitWidth();
7211 // Insert values into the bitmap. Undef values are set to zero.
7212 if (!isa<UndefValue>(TableContents[I - 1])) {
7213 ConstantInt *Val = cast<ConstantInt>(TableContents[I - 1]);
7214 TableInt |= Val->getValue().zext(TableInt.getBitWidth());
7215 }
7216 }
7217 BitMap = ConstantInt::get(M.getContext(), TableInt);
7218 BitMapElementTy = IT;
7219 Kind = BitMapKind;
7220 return;
7221 }
7222
7223 // The remaining representation needs a global initializer.
7224 if (!all_of(TableContents,
7225 [&](Constant *C) { return validLookupTableConstant(C, TTI); })) {
7226 Kind = InvalidKind;
7227 return;
7228 }
7229
7230 if (auto *IT = dyn_cast<IntegerType>(ValueType)) {
7231 ConstantRange Range(IT->getBitWidth(), false);
7232 for (Constant *Value : TableContents)
7233 if (!isa<UndefValue>(Value))
7234 Range = Range.unionWith(cast<ConstantInt>(Value)->getValue());
7235 // TODO: handle sign extension as well?
7236 unsigned NeededBitWidth =
7237 std::max(TTI.getMinimumLookupTableEntryBitWidth(),
7238 unsigned(PowerOf2Ceil(Range.getActiveBits())));
7239 if (NeededBitWidth < IT->getBitWidth()) {
7240 IntegerType *DstTy = IntegerType::get(IT->getContext(), NeededBitWidth);
7241 for (Constant *&Value : TableContents)
7242 Value = ConstantFoldCastInstruction(Instruction::Trunc, Value, DstTy);
7243 }
7244 }
7245
7246 // Store the table in an array.
7247 auto *TableTy = ArrayType::get(TableContents[0]->getType(), TableSize);
7248 Initializer = ConstantArray::get(TableTy, TableContents);
7249
7250 Kind = LookupTableKind;
7251}
7252
7253Value *SwitchReplacement::replaceSwitch(Value *Index, IRBuilder<> &Builder,
7254 const DataLayout &DL, Function *Func) {
7255 switch (Kind) {
7256 case InvalidKind:
7257 llvm_unreachable("Cannot use an invalid switch replacement");
7258 case SingleValueKind:
7259 return SingleValue;
7260 case LinearMapKind: {
7261 ++NumLinearMaps;
7262 // Derive the result value from the input value.
7263 Value *Result = Builder.CreateIntCast(Index, LinearMultiplier->getType(),
7264 false, "switch.idx.cast");
7265 if (!LinearMultiplier->isOne())
7266 Result = Builder.CreateMul(Result, LinearMultiplier, "switch.idx.mult",
7267 /*HasNUW = */ false,
7268 /*HasNSW = */ !LinearMapValWrapped);
7269
7270 if (!LinearOffset->isZero())
7271 Result = Builder.CreateAdd(Result, LinearOffset, "switch.offset",
7272 /*HasNUW = */ false,
7273 /*HasNSW = */ !LinearMapValWrapped);
7274 return Result;
7275 }
7276 case BitMapKind: {
7277 ++NumBitMaps;
7278 // Type of the bitmap (e.g. i59).
7279 IntegerType *MapTy = BitMap->getIntegerType();
7280
7281 // Cast Index to the same type as the bitmap.
7282 // Note: The Index is <= the number of elements in the table, so
7283 // truncating it to the width of the bitmask is safe.
7284 Value *ShiftAmt = Builder.CreateZExtOrTrunc(Index, MapTy, "switch.cast");
7285
7286 // Multiply the shift amount by the element width. NUW/NSW can always be
7287 // set, because wouldFitInRegister guarantees Index * ShiftAmt is in
7288 // BitMap's bit width.
7289 ShiftAmt = Builder.CreateMul(
7290 ShiftAmt, ConstantInt::get(MapTy, BitMapElementTy->getBitWidth()),
7291 "switch.shiftamt",/*HasNUW =*/true,/*HasNSW =*/true);
7292
7293 // Shift down.
7294 Value *DownShifted =
7295 Builder.CreateLShr(BitMap, ShiftAmt, "switch.downshift");
7296 // Mask off.
7297 return Builder.CreateTrunc(DownShifted, BitMapElementTy, "switch.masked");
7298 }
7299 case LookupTableKind: {
7300 ++NumLookupTables;
7301 auto *Table =
7302 new GlobalVariable(*Func->getParent(), Initializer->getType(),
7303 /*isConstant=*/true, GlobalVariable::PrivateLinkage,
7304 Initializer, "switch.table." + Func->getName());
7305 Table->setUnnamedAddr(GlobalValue::UnnamedAddr::Global);
7306 // Set the alignment to that of an array items. We will be only loading one
7307 // value out of it.
7308 Table->setAlignment(DL.getPrefTypeAlign(ValueType));
7309 Type *IndexTy = DL.getIndexType(Table->getType());
7310 auto *ArrayTy = cast<ArrayType>(Table->getValueType());
7311
7312 if (Index->getType() != IndexTy) {
7313 unsigned OldBitWidth = Index->getType()->getIntegerBitWidth();
7314 Index = Builder.CreateZExtOrTrunc(Index, IndexTy);
7315 if (auto *Zext = dyn_cast<ZExtInst>(Index))
7316 Zext->setNonNeg(
7317 isUIntN(OldBitWidth - 1, ArrayTy->getNumElements() - 1));
7318 }
7319
7320 Value *GEPIndices[] = {ConstantInt::get(IndexTy, 0), Index};
7321 Value *GEP =
7322 Builder.CreateInBoundsGEP(ArrayTy, Table, GEPIndices, "switch.gep");
7323 Value *Load =
7324 Builder.CreateLoad(ArrayTy->getElementType(), GEP, "switch.load");
7325 if (Load->getType() == ValueType)
7326 return Load;
7327 return Builder.CreateZExt(Load, ValueType, "switch.ext");
7328 }
7329 }
7330 llvm_unreachable("Unknown helper kind!");
7331}
7332
7333bool SwitchReplacement::wouldFitInRegister(const DataLayout &DL,
7334 uint64_t TableSize,
7335 Type *ElementType) {
7336 auto *IT = dyn_cast<IntegerType>(ElementType);
7337 if (!IT)
7338 return false;
7339 // FIXME: If the type is wider than it needs to be, e.g. i8 but all values
7340 // are <= 15, we could try to narrow the type.
7341
7342 // Avoid overflow, fitsInLegalInteger uses unsigned int for the width.
7343 if (TableSize >= UINT_MAX / IT->getBitWidth())
7344 return false;
7345 return DL.fitsInLegalInteger(TableSize * IT->getBitWidth());
7346}
7347
7349 const DataLayout &DL) {
7350 // Allow any legal type.
7351 if (TTI.isTypeLegal(Ty))
7352 return true;
7353
7354 auto *IT = dyn_cast<IntegerType>(Ty);
7355 if (!IT)
7356 return false;
7357
7358 // Also allow power of 2 integer types that have at least 8 bits and fit in
7359 // a register. These types are common in frontend languages and targets
7360 // usually support loads of these types.
7361 // TODO: We could relax this to any integer that fits in a register and rely
7362 // on ABI alignment and padding in the table to allow the load to be widened.
7363 // Or we could widen the constants and truncate the load.
7364 unsigned BitWidth = IT->getBitWidth();
7365 return BitWidth >= 8 && isPowerOf2_32(BitWidth) &&
7366 DL.fitsInLegalInteger(IT->getBitWidth());
7367}
7368
7369Constant *SwitchReplacement::getDefaultValue() { return DefaultValue; }
7370
7371bool SwitchReplacement::isLookupTable() { return Kind == LookupTableKind; }
7372
7373bool SwitchReplacement::isBitMap() { return Kind == BitMapKind; }
7374
7375static bool isSwitchDense(uint64_t NumCases, uint64_t CaseRange, bool OptSize) {
7376 // 40% is the default density for building a jump table in optsize/minsize
7377 // mode, 10% is the default density for jump tables. See also
7378 // TargetLoweringBase::isSuitableForJumpTable(), which this function was based
7379 // on.
7380 const uint64_t MinDensity = OptSize ? 40 : 10;
7381
7382 if (CaseRange >= UINT64_MAX / 100)
7383 return false; // Avoid multiplication overflows below.
7384
7385 return NumCases * 100 >= CaseRange * MinDensity;
7386}
7387
7388static bool isSwitchDense(ArrayRef<int64_t> Values, bool OptSize) {
7389 uint64_t Diff = (uint64_t)Values.back() - (uint64_t)Values.front();
7390 uint64_t Range = Diff + 1;
7391 if (Range < Diff)
7392 return false; // Overflow.
7393
7394 return isSwitchDense(Values.size(), Range, OptSize);
7395}
7396
7397static std::optional<unsigned>
7399 bool OptSize) {
7400 assert(Values.size() > 1 && "expected multiple switch cases");
7401 if (!llvm::all_of(Values, [Base](int64_t V) { return V >= Base; }))
7402 return std::nullopt;
7403
7404 // First, transform the values by subtracting Base.
7405 SmallVector<int64_t, 4> ReducedValues(Values);
7406 uint64_t ReducedValuesOr = 0;
7407 for (auto &V : ReducedValues) {
7408 uint64_t Reduced = (uint64_t)V - (uint64_t)Base;
7409 ReducedValuesOr |= Reduced;
7410 V = (int64_t)Reduced;
7411 }
7412
7413 // Conceptually, the reduced values are non-negative distances from Base.
7414 // Since the rest of the transform is bitwise only, treat them as unsigned
7415 // bit patterns from here.
7416
7417 // countr_zero(0) returns 64. As Values is guaranteed to have more than
7418 // one element and LLVM disallows duplicate cases, ReducedValuesOr will
7419 // have at least one bit set, so Shift will be less than 64.
7420 unsigned Shift = llvm::countr_zero(ReducedValuesOr);
7421 assert(Shift < 64);
7422 if (Shift > 0)
7423 for (auto &V : ReducedValues)
7424 V = (int64_t)((uint64_t)V >> Shift);
7425
7426 if (!isSwitchDense(ReducedValues, OptSize))
7427 return std::nullopt;
7428
7429 return Shift;
7430}
7431
7432/// Determine whether a lookup table should be built for this switch, based on
7433/// the number of cases, size of the table, and the types of the results.
7434// TODO: We could support larger than legal types by limiting based on the
7435// number of loads required and/or table size. If the constants are small we
7436// could use smaller table entries and extend after the load.
7438 const TargetTransformInfo &TTI,
7439 const DataLayout &DL,
7440 const SmallVector<Type *> &ResultTypes) {
7441 if (SI->getNumCases() > TableSize)
7442 return false; // TableSize overflowed.
7443
7444 bool AllTablesFitInRegister = true;
7445 bool HasIllegalType = false;
7446 for (const auto &Ty : ResultTypes) {
7447 // Saturate this flag to true.
7448 HasIllegalType = HasIllegalType || !isTypeLegalForLookupTable(Ty, TTI, DL);
7449
7450 // Saturate this flag to false.
7451 AllTablesFitInRegister =
7452 AllTablesFitInRegister &&
7453 SwitchReplacement::wouldFitInRegister(DL, TableSize, Ty);
7454
7455 // If both flags saturate, we're done. NOTE: This *only* works with
7456 // saturating flags, and all flags have to saturate first due to the
7457 // non-deterministic behavior of iterating over a dense map.
7458 if (HasIllegalType && !AllTablesFitInRegister)
7459 break;
7460 }
7461
7462 // If each table would fit in a register, we should build it anyway.
7463 if (AllTablesFitInRegister)
7464 return true;
7465
7466 // Don't build a table that doesn't fit in-register if it has illegal types.
7467 if (HasIllegalType)
7468 return false;
7469
7470 return isSwitchDense(SI->getNumCases(), TableSize,
7471 SI->getFunction()->hasOptSize());
7472}
7473
7475 ConstantInt &MinCaseVal, const ConstantInt &MaxCaseVal,
7476 bool HasDefaultResults, const SmallVector<Type *> &ResultTypes,
7477 const DataLayout &DL, const TargetTransformInfo &TTI) {
7478 if (MinCaseVal.isNullValue())
7479 return true;
7480 if (MinCaseVal.isNegative() ||
7481 MaxCaseVal.getLimitedValue() == std::numeric_limits<uint64_t>::max() ||
7482 !HasDefaultResults)
7483 return false;
7484 return all_of(ResultTypes, [&](const auto &ResultType) {
7485 return SwitchReplacement::wouldFitInRegister(
7486 DL, MaxCaseVal.getLimitedValue() + 1 /* TableSize */, ResultType);
7487 });
7488}
7489
7490/// Try to reuse the switch table index compare. Following pattern:
7491/// \code
7492/// if (idx < tablesize)
7493/// r = table[idx]; // table does not contain default_value
7494/// else
7495/// r = default_value;
7496/// if (r != default_value)
7497/// ...
7498/// \endcode
7499/// Is optimized to:
7500/// \code
7501/// cond = idx < tablesize;
7502/// if (cond)
7503/// r = table[idx];
7504/// else
7505/// r = default_value;
7506/// if (cond)
7507/// ...
7508/// \endcode
7509/// Jump threading will then eliminate the second if(cond).
7511 User *PhiUser, BasicBlock *PhiBlock, CondBrInst *RangeCheckBranch,
7512 Constant *DefaultValue,
7513 const SmallVectorImpl<std::pair<ConstantInt *, Constant *>> &Values) {
7515 if (!CmpInst)
7516 return;
7517
7518 // We require that the compare is in the same block as the phi so that jump
7519 // threading can do its work afterwards.
7520 if (CmpInst->getParent() != PhiBlock)
7521 return;
7522
7524 if (!CmpOp1)
7525 return;
7526
7527 Value *RangeCmp = RangeCheckBranch->getCondition();
7528 Constant *TrueConst = ConstantInt::getTrue(RangeCmp->getType());
7529 Constant *FalseConst = ConstantInt::getFalse(RangeCmp->getType());
7530
7531 // Check if the compare with the default value is constant true or false.
7532 const DataLayout &DL = PhiBlock->getDataLayout();
7534 CmpInst->getPredicate(), DefaultValue, CmpOp1, DL);
7535 if (DefaultConst != TrueConst && DefaultConst != FalseConst)
7536 return;
7537
7538 // Check if the compare with the case values is distinct from the default
7539 // compare result.
7540 for (auto ValuePair : Values) {
7542 CmpInst->getPredicate(), ValuePair.second, CmpOp1, DL);
7543 if (!CaseConst || CaseConst == DefaultConst ||
7544 (CaseConst != TrueConst && CaseConst != FalseConst))
7545 return;
7546 }
7547
7548 // Check if the branch instruction dominates the phi node. It's a simple
7549 // dominance check, but sufficient for our needs.
7550 // Although this check is invariant in the calling loops, it's better to do it
7551 // at this late stage. Practically we do it at most once for a switch.
7552 BasicBlock *BranchBlock = RangeCheckBranch->getParent();
7553 for (BasicBlock *Pred : predecessors(PhiBlock)) {
7554 if (Pred != BranchBlock && Pred->getUniquePredecessor() != BranchBlock)
7555 return;
7556 }
7557
7558 if (DefaultConst == FalseConst) {
7559 // The compare yields the same result. We can replace it.
7560 CmpInst->replaceAllUsesWith(RangeCmp);
7561 ++NumTableCmpReuses;
7562 } else {
7563 // The compare yields the same result, just inverted. We can replace it.
7564 Value *InvertedTableCmp = BinaryOperator::CreateXor(
7565 RangeCmp, ConstantInt::get(RangeCmp->getType(), 1), "inverted.cmp",
7566 RangeCheckBranch->getIterator());
7567 CmpInst->replaceAllUsesWith(InvertedTableCmp);
7568 ++NumTableCmpReuses;
7569 }
7570}
7571
7572/// If the switch is only used to initialize one or more phi nodes in a common
7573/// successor block with different constant values, replace the switch with
7574/// lookup tables.
7576 DomTreeUpdater *DTU, const DataLayout &DL,
7577 const TargetTransformInfo &TTI,
7578 bool ConvertSwitchToLookupTable) {
7579 assert(SI->getNumCases() > 1 && "Degenerate switch?");
7580
7581 BasicBlock *BB = SI->getParent();
7582 Function *Fn = BB->getParent();
7583
7584 // FIXME: If the switch is too sparse for a lookup table, perhaps we could
7585 // split off a dense part and build a lookup table for that.
7586
7587 // FIXME: This creates arrays of GEPs to constant strings, which means each
7588 // GEP needs a runtime relocation in PIC code. We should just build one big
7589 // string and lookup indices into that.
7590
7591 // Ignore switches with less than three cases. Lookup tables will not make
7592 // them faster, so we don't analyze them.
7593 if (SI->getNumCases() < 3)
7594 return false;
7595
7596 // Figure out the corresponding result for each case value and phi node in the
7597 // common destination, as well as the min and max case values.
7598 assert(!SI->cases().empty());
7599 SwitchInst::CaseIt CI = SI->case_begin();
7600 ConstantInt *MinCaseVal = CI->getCaseValue();
7601 ConstantInt *MaxCaseVal = CI->getCaseValue();
7602
7603 BasicBlock *CommonDest = nullptr;
7604
7605 using ResultListTy = SmallVector<std::pair<ConstantInt *, Constant *>, 4>;
7607
7609 SmallVector<Type *> ResultTypes;
7611
7612 for (SwitchInst::CaseIt E = SI->case_end(); CI != E; ++CI) {
7613 ConstantInt *CaseVal = CI->getCaseValue();
7614 if (CaseVal->getValue().slt(MinCaseVal->getValue()))
7615 MinCaseVal = CaseVal;
7616 if (CaseVal->getValue().sgt(MaxCaseVal->getValue()))
7617 MaxCaseVal = CaseVal;
7618
7619 // Resulting value at phi nodes for this case value.
7621 ResultsTy Results;
7622 if (!getCaseResults(SI, CaseVal, CI->getCaseSuccessor(), &CommonDest,
7623 Results, DL))
7624 return false;
7625
7626 // Append the result and result types from this case to the list for each
7627 // phi.
7628 for (const auto &I : Results) {
7629 PHINode *PHI = I.first;
7630 Constant *Value = I.second;
7631 auto [It, Inserted] = ResultLists.try_emplace(PHI);
7632 if (Inserted)
7633 PHIs.push_back(PHI);
7634 It->second.push_back(std::make_pair(CaseVal, Value));
7635 ResultTypes.push_back(PHI->getType());
7636 }
7637 }
7638
7639 // If the table has holes, we need a constant result for the default case
7640 // or a bitmask that fits in a register.
7641 SmallVector<std::pair<PHINode *, Constant *>, 4> DefaultResultsList;
7642 bool HasDefaultResults = getCaseResults(SI, nullptr, SI->getDefaultDest(),
7643 &CommonDest, DefaultResultsList, DL);
7644 for (const auto &I : DefaultResultsList) {
7645 PHINode *PHI = I.first;
7646 Constant *Result = I.second;
7647 DefaultResults[PHI] = Result;
7648 }
7649
7650 bool UseSwitchConditionAsTableIndex = shouldUseSwitchConditionAsTableIndex(
7651 *MinCaseVal, *MaxCaseVal, HasDefaultResults, ResultTypes, DL, TTI);
7652 uint64_t TableSize;
7653 ConstantInt *TableIndexOffset;
7654 if (UseSwitchConditionAsTableIndex) {
7655 TableSize = MaxCaseVal->getLimitedValue() + 1;
7656 TableIndexOffset = ConstantInt::get(MaxCaseVal->getIntegerType(), 0);
7657 } else {
7658 TableSize =
7659 (MaxCaseVal->getValue() - MinCaseVal->getValue()).getLimitedValue() + 1;
7660
7661 TableIndexOffset = MinCaseVal;
7662 }
7663
7664 // If the default destination is unreachable, or if the lookup table covers
7665 // all values of the conditional variable, branch directly to the lookup table
7666 // BB. Otherwise, check that the condition is within the case range.
7667 uint64_t NumResults = ResultLists[PHIs[0]].size();
7668 bool DefaultIsReachable = !SI->defaultDestUnreachable();
7669
7670 bool TableHasHoles = (NumResults < TableSize);
7671
7672 // If the table has holes but the default destination doesn't produce any
7673 // constant results, the lookup table entries corresponding to the holes will
7674 // contain poison.
7675 bool AllHolesArePoison = TableHasHoles && !HasDefaultResults;
7676
7677 // If the default destination doesn't produce a constant result but is still
7678 // reachable, and the lookup table has holes, we need to use a mask to
7679 // determine if the current index should load from the lookup table or jump
7680 // to the default case.
7681 // The mask is unnecessary if the table has holes but the default destination
7682 // is unreachable, as in that case the holes must also be unreachable.
7683 bool NeedMask = AllHolesArePoison && DefaultIsReachable;
7684 if (NeedMask) {
7685 // As an extra penalty for the validity test we require more cases.
7686 if (SI->getNumCases() < 4) // FIXME: Find best threshold value (benchmark).
7687 return false;
7688 if (!DL.fitsInLegalInteger(TableSize))
7689 return false;
7690 }
7691
7692 if (!shouldBuildLookupTable(SI, TableSize, TTI, DL, ResultTypes))
7693 return false;
7694
7695 // Compute the table index value.
7696 Value *TableIndex;
7697 if (UseSwitchConditionAsTableIndex) {
7698 TableIndex = SI->getCondition();
7699 if (HasDefaultResults) {
7700 // Grow the table to cover all possible index values to avoid the range
7701 // check. It will use the default result to fill in the table hole later,
7702 // so make sure it exist.
7703 ConstantRange CR = computeConstantRange(TableIndex, /*ForSigned=*/false,
7704 SimplifyQuery(DL));
7705 // Grow the table shouldn't have any size impact by checking
7706 // wouldFitInRegister.
7707 // TODO: Consider growing the table also when it doesn't fit in a register
7708 // if no optsize is specified.
7709 const uint64_t UpperBound = CR.getUpper().getLimitedValue();
7710 if (!CR.isUpperWrapped() &&
7711 all_of(ResultTypes, [&](const auto &ResultType) {
7712 return SwitchReplacement::wouldFitInRegister(DL, UpperBound,
7713 ResultType);
7714 })) {
7715 // There may be some case index larger than the UpperBound (unreachable
7716 // case), so make sure the table size does not get smaller.
7717 TableSize = std::max(UpperBound, TableSize);
7718 // The default branch is unreachable after we enlarge the lookup table.
7719 // Adjust DefaultIsReachable to reuse code path.
7720 DefaultIsReachable = false;
7721 }
7722 }
7723 }
7724
7725 // Keep track of the switch replacement for each phi
7727 for (PHINode *PHI : PHIs) {
7728 const auto &ResultList = ResultLists[PHI];
7729
7730 Type *ResultType = ResultList.begin()->second->getType();
7731 // Use any value to fill the lookup table holes.
7732 Constant *DefaultVal =
7733 AllHolesArePoison ? PoisonValue::get(ResultType) : DefaultResults[PHI];
7734 StringRef FuncName = Fn->getName();
7735 SwitchReplacement Replacement(*Fn->getParent(), TableSize, TableIndexOffset,
7736 ResultList, DefaultVal, DL, TTI, FuncName);
7737 if (!Replacement.isValid())
7738 return false;
7739 PhiToReplacementMap.insert({PHI, Replacement});
7740 }
7741
7742 bool AnyLookupTables = any_of(
7743 PhiToReplacementMap, [](auto &KV) { return KV.second.isLookupTable(); });
7744 bool AnyBitMaps = any_of(PhiToReplacementMap,
7745 [](auto &KV) { return KV.second.isBitMap(); });
7746
7747 // A few conditions prevent the generation of lookup tables:
7748 // 1. The target does not support lookup tables.
7749 // 2. The "no-jump-tables" function attribute is set.
7750 // However, these objections do not apply to other switch replacements, like
7751 // the bitmap, so we only stop here if any of these conditions are met and we
7752 // want to create a LUT. Otherwise, continue with the switch replacement.
7753 if (AnyLookupTables &&
7754 (!TTI.shouldBuildLookupTables() ||
7755 Fn->getFnAttribute("no-jump-tables").getValueAsBool()))
7756 return false;
7757
7758 // In the early optimization pipeline, disable formation of lookup tables,
7759 // bit maps and mask checks, as they may inhibit further optimization.
7760 if (!ConvertSwitchToLookupTable &&
7761 (AnyLookupTables || AnyBitMaps || NeedMask))
7762 return false;
7763
7764 Builder.SetInsertPoint(SI);
7765 // TableIndex is the switch condition - TableIndexOffset if we don't
7766 // use the condition directly
7767 if (!UseSwitchConditionAsTableIndex) {
7768 // If the default is unreachable, all case values are s>= MinCaseVal. Then
7769 // we can try to attach nsw.
7770 bool MayWrap = true;
7771 if (!DefaultIsReachable) {
7772 APInt Res =
7773 MaxCaseVal->getValue().ssub_ov(MinCaseVal->getValue(), MayWrap);
7774 (void)Res;
7775 }
7776 TableIndex = Builder.CreateSub(SI->getCondition(), TableIndexOffset,
7777 "switch.tableidx", /*HasNUW =*/false,
7778 /*HasNSW =*/!MayWrap);
7779 }
7780
7781 std::vector<DominatorTree::UpdateType> Updates;
7782
7783 // Compute the maximum table size representable by the integer type we are
7784 // switching upon.
7785 unsigned CaseSize = MinCaseVal->getType()->getPrimitiveSizeInBits();
7786 uint64_t MaxTableSize = CaseSize > 63 ? UINT64_MAX : 1ULL << CaseSize;
7787 assert(MaxTableSize >= TableSize &&
7788 "It is impossible for a switch to have more entries than the max "
7789 "representable value of its input integer type's size.");
7790
7791 // Create the BB that does the lookups.
7792 Module &Mod = *CommonDest->getParent()->getParent();
7793 BasicBlock *LookupBB = BasicBlock::Create(
7794 Mod.getContext(), "switch.lookup", CommonDest->getParent(), CommonDest);
7795
7796 CondBrInst *RangeCheckBranch = nullptr;
7797 CondBrInst *CondBranch = nullptr;
7798
7799 Builder.SetInsertPoint(SI);
7800 const bool GeneratingCoveredLookupTable = (MaxTableSize == TableSize);
7801 if (!DefaultIsReachable || GeneratingCoveredLookupTable) {
7802 Builder.CreateBr(LookupBB);
7803 if (DTU)
7804 Updates.push_back({DominatorTree::Insert, BB, LookupBB});
7805 // Note: We call removeProdecessor later since we need to be able to get the
7806 // PHI value for the default case in case we're using a bit mask.
7807 } else {
7808 Value *Cmp = Builder.CreateICmpULT(
7809 TableIndex, ConstantInt::get(MinCaseVal->getType(), TableSize));
7810 RangeCheckBranch =
7811 Builder.CreateCondBr(Cmp, LookupBB, SI->getDefaultDest());
7812 CondBranch = RangeCheckBranch;
7813 if (DTU)
7814 Updates.push_back({DominatorTree::Insert, BB, LookupBB});
7815 }
7816
7817 // Populate the BB that does the lookups.
7818 Builder.SetInsertPoint(LookupBB);
7819
7820 if (NeedMask) {
7821 // Before doing the lookup, we do the hole check. The LookupBB is therefore
7822 // re-purposed to do the hole check, and we create a new LookupBB.
7823 BasicBlock *MaskBB = LookupBB;
7824 MaskBB->setName("switch.hole_check");
7825 LookupBB = BasicBlock::Create(Mod.getContext(), "switch.lookup",
7826 CommonDest->getParent(), CommonDest);
7827
7828 // Make the mask's bitwidth at least 8-bit and a power-of-2 to avoid
7829 // unnecessary illegal types.
7830 uint64_t TableSizePowOf2 = NextPowerOf2(std::max(7ULL, TableSize - 1ULL));
7831 APInt MaskInt(TableSizePowOf2, 0);
7832 APInt One(TableSizePowOf2, 1);
7833 // Build bitmask; fill in a 1 bit for every case.
7834 const ResultListTy &ResultList = ResultLists[PHIs[0]];
7835 for (const auto &Result : ResultList) {
7836 uint64_t Idx = (Result.first->getValue() - TableIndexOffset->getValue())
7837 .getLimitedValue();
7838 MaskInt |= One << Idx;
7839 }
7840 ConstantInt *TableMask = ConstantInt::get(Mod.getContext(), MaskInt);
7841
7842 // Get the TableIndex'th bit of the bitmask.
7843 // If this bit is 0 (meaning hole) jump to the default destination,
7844 // else continue with table lookup.
7845 IntegerType *MapTy = TableMask->getIntegerType();
7846 Value *MaskIndex =
7847 Builder.CreateZExtOrTrunc(TableIndex, MapTy, "switch.maskindex");
7848 Value *Shifted = Builder.CreateLShr(TableMask, MaskIndex, "switch.shifted");
7849 Value *LoBit = Builder.CreateTrunc(
7850 Shifted, Type::getInt1Ty(Mod.getContext()), "switch.lobit");
7851 CondBranch = Builder.CreateCondBr(LoBit, LookupBB, SI->getDefaultDest());
7852 if (DTU) {
7853 Updates.push_back({DominatorTree::Insert, MaskBB, LookupBB});
7854 Updates.push_back({DominatorTree::Insert, MaskBB, SI->getDefaultDest()});
7855 }
7856 Builder.SetInsertPoint(LookupBB);
7857 addPredecessorToBlock(SI->getDefaultDest(), MaskBB, BB);
7858 }
7859
7860 if (!DefaultIsReachable || GeneratingCoveredLookupTable) {
7861 // We cached PHINodes in PHIs. To avoid accessing deleted PHINodes later,
7862 // do not delete PHINodes here.
7863 SI->getDefaultDest()->removePredecessor(BB,
7864 /*KeepOneInputPHIs=*/true);
7865 if (DTU)
7866 Updates.push_back({DominatorTree::Delete, BB, SI->getDefaultDest()});
7867 }
7868
7869 for (PHINode *PHI : PHIs) {
7870 const ResultListTy &ResultList = ResultLists[PHI];
7871 auto Replacement = PhiToReplacementMap.at(PHI);
7872 auto *Result = Replacement.replaceSwitch(TableIndex, Builder, DL, Fn);
7873 // Do a small peephole optimization: re-use the switch table compare if
7874 // possible.
7875 if (!TableHasHoles && HasDefaultResults && RangeCheckBranch) {
7876 BasicBlock *PhiBlock = PHI->getParent();
7877 // Search for compare instructions which use the phi.
7878 for (auto *User : PHI->users()) {
7879 reuseTableCompare(User, PhiBlock, RangeCheckBranch,
7880 Replacement.getDefaultValue(), ResultList);
7881 }
7882 }
7883
7884 PHI->addIncoming(Result, LookupBB);
7885 }
7886
7887 Builder.CreateBr(CommonDest);
7888 if (DTU)
7889 Updates.push_back({DominatorTree::Insert, LookupBB, CommonDest});
7890
7891 SmallVector<uint32_t> BranchWeights;
7892 const bool HasBranchWeights =
7893 CondBranch && extractBranchWeights(*SI, BranchWeights);
7894 uint64_t ToLookupWeight = 0;
7895 uint64_t ToDefaultWeight = 0;
7896
7897 // Remove the switch.
7898 SmallPtrSet<BasicBlock *, 8> RemovedSuccessors;
7899 for (unsigned I = 0, E = SI->getNumSuccessors(); I < E; ++I) {
7900 BasicBlock *Succ = SI->getSuccessor(I);
7901
7902 if (Succ == SI->getDefaultDest()) {
7903 if (HasBranchWeights)
7904 ToDefaultWeight += BranchWeights[I];
7905 continue;
7906 }
7907 Succ->removePredecessor(BB);
7908 if (DTU && RemovedSuccessors.insert(Succ).second)
7909 Updates.push_back({DominatorTree::Delete, BB, Succ});
7910 if (HasBranchWeights)
7911 ToLookupWeight += BranchWeights[I];
7912 }
7913 SI->eraseFromParent();
7914 if (HasBranchWeights)
7915 setFittedBranchWeights(*CondBranch, {ToLookupWeight, ToDefaultWeight},
7916 /*IsExpected=*/false);
7917 if (DTU)
7918 DTU->applyUpdates(Updates);
7919
7920 if (NeedMask)
7921 ++NumLookupTablesHoles;
7922 return true;
7923}
7924
7925/// Try to transform a switch that has "holes" in it to a contiguous sequence
7926/// of cases.
7927///
7928/// A switch such as: switch(i) {case 5: case 9: case 13: case 17:} can be
7929/// range-reduced to: switch ((i-5) / 4) {case 0: case 1: case 2: case 3:}.
7930///
7931/// This converts a sparse switch into a dense switch which allows better
7932/// lowering and could also allow transforming into a lookup table.
7934 const DataLayout &DL,
7935 const TargetTransformInfo &TTI) {
7936 auto *CondTy = cast<IntegerType>(SI->getCondition()->getType());
7937 if (CondTy->getIntegerBitWidth() > 64 ||
7938 !DL.fitsInLegalInteger(CondTy->getIntegerBitWidth()))
7939 return false;
7940 // Only bother with this optimization if there are more than 3 switch cases;
7941 // SDAG will only bother creating jump tables for 4 or more cases.
7942 if (SI->getNumCases() < 4)
7943 return false;
7944
7945 // This transform is agnostic to the signedness of the input or case values. We
7946 // can treat the case values as signed or unsigned. We can optimize more common
7947 // cases such as a sequence crossing zero {-4,0,4,8} if we interpret case values
7948 // as signed.
7950 for (const auto &C : SI->cases())
7951 Values.push_back(C.getCaseValue()->getValue().getSExtValue());
7953
7954 // If the switch is already dense, there's nothing useful to do here.
7955 bool OptSize = SI->getFunction()->hasOptSize();
7956 if (isSwitchDense(Values, OptSize))
7957 return false;
7958
7959 // Find a Base and corresponding Shift that results in a dense switch range.
7960 // Values[0] is the local minimum.
7961 int64_t Base = Values[0];
7962 std::optional<unsigned> Shift;
7963 // Prefer Base=0 when shifting out common low zero bits still produces a dense
7964 // range, as this avoids an unnecessary `(condition - local_min)` expression.
7965 // However, avoiding the subtract can leave a wider reduced range than using
7966 // the local minimum, so require Base=0 to satisfy the stricter optsize
7967 // density threshold before falling back to the normal density policy for
7968 // local-min.
7969 if ((Shift = getDenseSwitchRangeReductionShift(Values, /*Base=*/0,
7970 /*OptSize=*/true)))
7971 Base = 0;
7972 else if (Base != 0)
7974
7975 if (!Shift)
7976 return false;
7977
7978 // The obvious transform is to shift the switch condition right and emit a
7979 // check that the condition actually cleanly divided by GCD, i.e.
7980 // C & (1 << Shift - 1) == 0
7981 // inserting a new CFG edge to handle the case where it didn't divide cleanly.
7982 //
7983 // A cheaper way of doing this is a simple ROTR(C, Shift). This performs the
7984 // shift and puts the shifted-off bits in the uppermost bits. If any of these
7985 // are nonzero then the switch condition will be very large and will hit the
7986 // default case.
7987 //
7988 // This transform can be done speculatively because it is so cheap - it
7989 // results in a single rotate operation being inserted.
7990
7991 auto *Ty = cast<IntegerType>(SI->getCondition()->getType());
7992 Builder.SetInsertPoint(SI);
7993 Value *Sub = SI->getCondition();
7994 if (Base != 0)
7995 Sub = Builder.CreateSub(Sub, ConstantInt::getSigned(Ty, Base));
7996 Value *Rot = Builder.CreateIntrinsic(
7997 Ty, Intrinsic::fshl,
7998 {Sub, Sub, ConstantInt::get(Ty, Ty->getBitWidth() - *Shift)});
7999 SI->replaceUsesOfWith(SI->getCondition(), Rot);
8000
8001 for (auto Case : SI->cases()) {
8002 auto *Orig = Case.getCaseValue();
8003 auto Sub = Orig->getValue() - APInt(Ty->getBitWidth(), Base, true);
8004 Case.setValue(cast<ConstantInt>(ConstantInt::get(Ty, Sub.lshr(*Shift))));
8005 }
8006 return true;
8007}
8008
8009/// Tries to transform the switch when the condition is umin with a constant.
8010/// In that case, the default branch can be replaced by the constant's branch.
8011/// This method also removes dead cases when the simplification cannot replace
8012/// the default branch.
8013///
8014/// For example:
8015/// switch(umin(a, 3)) {
8016/// case 0:
8017/// case 1:
8018/// case 2:
8019/// case 3:
8020/// case 4:
8021/// // ...
8022/// default:
8023/// unreachable
8024/// }
8025///
8026/// Transforms into:
8027///
8028/// switch(a) {
8029/// case 0:
8030/// case 1:
8031/// case 2:
8032/// default:
8033/// // This is case 3
8034/// }
8036 Value *A;
8038
8039 if (!match(SI->getCondition(), m_UMin(m_Value(A), m_ConstantInt(Constant))))
8040 return false;
8041
8044 BasicBlock *BB = SIW->getParent();
8045
8046 // Dead cases are removed even when the simplification fails.
8047 // A case is dead when its value is higher than the Constant.
8048 for (auto I = SI->case_begin(), E = SI->case_end(); I != E;) {
8049 if (!I->getCaseValue()->getValue().ugt(Constant->getValue())) {
8050 ++I;
8051 continue;
8052 }
8053 BasicBlock *DeadCaseBB = I->getCaseSuccessor();
8054 DeadCaseBB->removePredecessor(BB);
8055 I = SIW.removeCase(I);
8056 E = SIW->case_end();
8057 if (!is_contained(successors(BB), DeadCaseBB))
8058 Updates.push_back({DominatorTree::Delete, BB, DeadCaseBB});
8059 }
8060
8061 auto Case = SI->findCaseValue(Constant);
8062 // If the case value is not found, `findCaseValue` returns the default case.
8063 // In this scenario, since there is no explicit `case 3:`, the simplification
8064 // fails. The simplification also fails when the switch’s default destination
8065 // is reachable.
8066 if (!SI->defaultDestUnreachable() || Case == SI->case_default()) {
8067 if (DTU)
8068 DTU->applyUpdates(Updates);
8069 return !Updates.empty();
8070 }
8071
8072 BasicBlock *Unreachable = SI->getDefaultDest();
8073 SIW.replaceDefaultDest(Case);
8074 SIW.removeCase(Case);
8075 SIW->setCondition(A);
8076
8077 Updates.push_back({DominatorTree::Delete, BB, Unreachable});
8078
8079 if (DTU)
8080 DTU->applyUpdates(Updates);
8081
8082 return true;
8083}
8084
8086 const DataLayout &DL,
8087 AssumptionCache *AC) {
8088 assert(SI);
8089 if (SI->defaultDestUnreachable())
8090 return false;
8091
8092 // If it can be proved that the switch condition takes some concrete value
8093 // in the default block, we can make some nice simplifications to the
8094 // switch.
8095 BasicBlock *Default = SI->getDefaultDest();
8096 const Instruction *CtxI = &*Default->getFirstNonPHIIt();
8098 SI->getCondition(),
8099 SimplifyQuery(DL, /*DT=*/nullptr, AC, CtxI).allowEphemerals(true));
8100 if (!Known.isConstant())
8101 return false;
8102
8103 // At this point, we know that only one value can be mapped to the
8104 // default block. So, if a case doesn't exist for it already, we
8105 // can create one pointing to the default block.
8106 ConstantInt *CaseVal =
8107 ConstantInt::get(SI->getContext(), Known.getConstant());
8108 const llvm::SwitchInst::CaseIt CaseIt = SI->findCaseValue(CaseVal);
8109 if (CaseIt == SI->case_default()) {
8111 SIW.addCase(CaseVal, Default, SIW.getSuccessorWeight(0));
8112 SIW.setSuccessorWeight(0, 0);
8113 }
8114 // If there is a pre-existing case for the constant, the default branch
8115 // will be removed rather than being moved. Thus, we are removing an edge
8116 // in the CFG, and need to update any PHIs in the default block.
8117 createUnreachableSwitchDefault(SI, DTU, /*RemoveOrigDefaultBlock=*/CaseIt !=
8118 SI->case_default());
8119
8120 assert(SI->getNumCases() > 0 && "Switch should have at least one case");
8121 assert(SI->findCaseValue(CaseVal) != SI->case_default() &&
8122 "Proven value should have a dedicated case");
8123 assert(SI->defaultDestUnreachable());
8124 return true;
8125}
8126
8127/// Tries to transform switch of powers of two to reduce switch range.
8128/// For example, switch like:
8129/// switch (C) { case 1: case 2: case 64: case 128: }
8130/// will be transformed to:
8131/// switch (count_trailing_zeros(C)) { case 0: case 1: case 6: case 7: }
8132///
8133/// This transformation allows better lowering and may transform the switch
8134/// instruction into a sequence of bit manipulation and a smaller
8135/// log2(C)-indexed value table (instead of traditionally emitting a load of the
8136/// address of the jump target, and indirectly jump to it).
8138 DomTreeUpdater *DTU,
8139 const DataLayout &DL,
8140 const TargetTransformInfo &TTI) {
8141 Value *Condition = SI->getCondition();
8142 LLVMContext &Context = SI->getContext();
8143 auto *CondTy = cast<IntegerType>(Condition->getType());
8144
8145 if (CondTy->getIntegerBitWidth() > 64 ||
8146 !DL.fitsInLegalInteger(CondTy->getIntegerBitWidth()))
8147 return false;
8148
8149 // Ensure trailing zeroes count intrinsic emission is not too expensive.
8150 IntrinsicCostAttributes Attrs(Intrinsic::cttz, CondTy,
8151 {Condition, ConstantInt::getTrue(Context)});
8152 if (TTI.getIntrinsicInstrCost(Attrs, TTI::TCK_SizeAndLatency) >
8153 TTI::TCC_Basic * 2)
8154 return false;
8155
8156 // Only bother with this optimization if there are more than 3 switch cases.
8157 // SDAG will start emitting jump tables for 4 or more cases.
8158 if (SI->getNumCases() < 4)
8159 return false;
8160
8161 // Check that switch cases are powers of two.
8163 for (const auto &Case : SI->cases()) {
8164 uint64_t CaseValue = Case.getCaseValue()->getValue().getZExtValue();
8165 if (llvm::has_single_bit(CaseValue))
8166 Values.push_back(CaseValue);
8167 else
8168 return false;
8169 }
8170
8171 // isSwichDense requires case values to be sorted.
8173 if (!isSwitchDense(Values.size(),
8174 llvm::countr_zero(Values.back()) -
8175 llvm::countr_zero(Values.front()) + 1,
8176 SI->getFunction()->hasOptSize()))
8177 // Transform is unable to generate dense switch.
8178 return false;
8179
8180 Builder.SetInsertPoint(SI);
8181
8182 if (!SI->defaultDestUnreachable()) {
8183 // Let non-power-of-two inputs jump to the default case, when the latter is
8184 // reachable.
8185 auto *PopC = Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, Condition);
8186 auto *IsPow2 = Builder.CreateICmpEQ(PopC, ConstantInt::get(CondTy, 1));
8187
8188 auto *OrigBB = SI->getParent();
8189 auto *DefaultCaseBB = SI->getDefaultDest();
8190 BasicBlock *SplitBB = SplitBlock(OrigBB, SI, DTU);
8191 auto It = OrigBB->getTerminator()->getIterator();
8192 SmallVector<uint32_t> Weights;
8193 auto HasWeights = extractBranchWeights(*SI, Weights);
8194 auto *BI = CondBrInst::Create(IsPow2, SplitBB, DefaultCaseBB, It);
8195 if (HasWeights && any_of(Weights, not_equal_to(0))) {
8196 // IsPow2 covers a subset of the cases in which we'd go to the default
8197 // label. The other is those powers of 2 that don't appear in the case
8198 // statement. We don't know the distribution of the values coming in, so
8199 // the safest is to split 50-50 the original probability to `default`.
8200 uint64_t OrigDenominator =
8202 SmallVector<uint64_t> NewWeights(2);
8203 NewWeights[1] = Weights[0] / 2;
8204 NewWeights[0] = OrigDenominator - NewWeights[1];
8205 setFittedBranchWeights(*BI, NewWeights, /*IsExpected=*/false);
8206 // The probability of executing the default block stays constant. It was
8207 // p_d = Weights[0] / OrigDenominator
8208 // we rewrite as W/D
8209 // We want to find the probability of the default branch of the switch
8210 // statement. Let's call it X. We have W/D = W/2D + X * (1-W/2D)
8211 // i.e. the original probability is the probability we go to the default
8212 // branch from the BI branch, or we take the default branch on the SI.
8213 // Meaning X = W / (2D - W), or (W/2) / (D - W/2)
8214 // This matches using W/2 for the default branch probability numerator and
8215 // D-W/2 as the denominator.
8216 Weights[0] = NewWeights[1];
8217 uint64_t CasesDenominator = OrigDenominator - Weights[0];
8218 for (auto &W : drop_begin(Weights))
8219 W = NewWeights[0] * static_cast<double>(W) / CasesDenominator;
8220
8221 setBranchWeights(*SI, Weights, /*IsExpected=*/false);
8222 }
8223 // BI is handling the default case for SI, and so should share its DebugLoc.
8224 BI->setDebugLoc(SI->getDebugLoc());
8225 It->eraseFromParent();
8226
8227 addPredecessorToBlock(DefaultCaseBB, OrigBB, SplitBB);
8228 if (DTU)
8229 DTU->applyUpdates({{DominatorTree::Insert, OrigBB, DefaultCaseBB}});
8230 }
8231
8232 // Replace each case with its trailing zeros number.
8233 for (auto &Case : SI->cases()) {
8234 auto *OrigValue = Case.getCaseValue();
8235 Case.setValue(ConstantInt::get(OrigValue->getIntegerType(),
8236 OrigValue->getValue().countr_zero()));
8237 }
8238
8239 // Replace condition with its trailing zeros number.
8240 auto *ConditionTrailingZeros = Builder.CreateIntrinsic(
8241 Intrinsic::cttz, {CondTy}, {Condition, ConstantInt::getTrue(Context)});
8242
8243 SI->setCondition(ConditionTrailingZeros);
8244
8245 return true;
8246}
8247
8248/// Fold switch over ucmp/scmp intrinsic to br if two of the switch arms have
8249/// the same destination.
8251 DomTreeUpdater *DTU) {
8252 auto *Cmp = dyn_cast<CmpIntrinsic>(SI->getCondition());
8253 if (!Cmp || !Cmp->hasOneUse())
8254 return false;
8255
8257 bool HasWeights = extractBranchWeights(getBranchWeightMDNode(*SI), Weights);
8258 if (!HasWeights)
8259 Weights.resize(4); // Avoid checking HasWeights everywhere.
8260
8261 // Normalize to [us]cmp == Res ? Succ : OtherSucc.
8262 int64_t Res;
8263 BasicBlock *Succ, *OtherSucc;
8264 uint32_t SuccWeight = 0, OtherSuccWeight = 0;
8265 BasicBlock *Unreachable = nullptr;
8266
8267 if (SI->getNumCases() == 2) {
8268 // Find which of 1, 0 or -1 is missing (handled by default dest).
8269 SmallSet<int64_t, 3> Missing;
8270 Missing.insert(1);
8271 Missing.insert(0);
8272 Missing.insert(-1);
8273
8274 Succ = SI->getDefaultDest();
8275 SuccWeight = Weights[0];
8276 OtherSucc = nullptr;
8277 for (auto &Case : SI->cases()) {
8278 std::optional<int64_t> Val =
8279 Case.getCaseValue()->getValue().trySExtValue();
8280 if (!Val)
8281 return false;
8282 if (!Missing.erase(*Val))
8283 return false;
8284 if (OtherSucc && OtherSucc != Case.getCaseSuccessor())
8285 return false;
8286 OtherSucc = Case.getCaseSuccessor();
8287 OtherSuccWeight += Weights[Case.getSuccessorIndex()];
8288 }
8289
8290 assert(Missing.size() == 1 && "Should have one case left");
8291 Res = *Missing.begin();
8292 } else if (SI->getNumCases() == 3 && SI->defaultDestUnreachable()) {
8293 // Normalize so that Succ is taken once and OtherSucc twice.
8294 Unreachable = SI->getDefaultDest();
8295 Succ = OtherSucc = nullptr;
8296 for (auto &Case : SI->cases()) {
8297 BasicBlock *NewSucc = Case.getCaseSuccessor();
8298 uint32_t Weight = Weights[Case.getSuccessorIndex()];
8299 if (!OtherSucc || OtherSucc == NewSucc) {
8300 OtherSucc = NewSucc;
8301 OtherSuccWeight += Weight;
8302 } else if (!Succ) {
8303 Succ = NewSucc;
8304 SuccWeight = Weight;
8305 } else if (Succ == NewSucc) {
8306 std::swap(Succ, OtherSucc);
8307 std::swap(SuccWeight, OtherSuccWeight);
8308 } else
8309 return false;
8310 }
8311 for (auto &Case : SI->cases()) {
8312 std::optional<int64_t> Val =
8313 Case.getCaseValue()->getValue().trySExtValue();
8314 if (!Val || (Val != 1 && Val != 0 && Val != -1))
8315 return false;
8316 if (Case.getCaseSuccessor() == Succ) {
8317 Res = *Val;
8318 break;
8319 }
8320 }
8321 } else {
8322 return false;
8323 }
8324
8325 // Determine predicate for the missing case.
8327 switch (Res) {
8328 case 1:
8329 Pred = ICmpInst::ICMP_UGT;
8330 break;
8331 case 0:
8332 Pred = ICmpInst::ICMP_EQ;
8333 break;
8334 case -1:
8335 Pred = ICmpInst::ICMP_ULT;
8336 break;
8337 }
8338 if (Cmp->isSigned())
8339 Pred = ICmpInst::getSignedPredicate(Pred);
8340
8341 MDNode *NewWeights = nullptr;
8342 if (HasWeights)
8343 NewWeights = MDBuilder(SI->getContext())
8344 .createBranchWeights(SuccWeight, OtherSuccWeight);
8345
8346 BasicBlock *BB = SI->getParent();
8347 Builder.SetInsertPoint(SI->getIterator());
8348 Value *ICmp = Builder.CreateICmp(Pred, Cmp->getLHS(), Cmp->getRHS());
8349 Builder.CreateCondBr(ICmp, Succ, OtherSucc, NewWeights,
8350 SI->getMetadata(LLVMContext::MD_unpredictable));
8351 OtherSucc->removePredecessor(BB);
8352 if (Unreachable)
8353 Unreachable->removePredecessor(BB);
8354 SI->eraseFromParent();
8355 Cmp->eraseFromParent();
8356 if (DTU && Unreachable)
8357 DTU->applyUpdates({{DominatorTree::Delete, BB, Unreachable}});
8358 return true;
8359}
8360
8361/// Checking whether two BBs are equal depends on the contents of the
8362/// BasicBlock and the incoming values of their successor PHINodes.
8363/// PHINode::getIncomingValueForBlock is O(|Preds|), so we'd like to avoid
8364/// calling this function on each BasicBlock every time isEqual is called,
8365/// especially since the same BasicBlock may be passed as an argument multiple
8366/// times. To do this, we can precompute a map of PHINode -> Pred BasicBlock ->
8367/// IncomingValue and add it in the Wrapper so isEqual can do O(1) checking
8368/// of the incoming values.
8371
8372 // One Phi usually has < 8 incoming values.
8376
8377 // We only merge the identical non-entry BBs with
8378 // - terminator unconditional br to Succ (pending relaxation),
8379 // - does not have address taken / weird control.
8380 static bool canBeMerged(const BasicBlock *BB) {
8381 assert(BB && "Expected non-null BB");
8382 // Entry block cannot be eliminated or have predecessors.
8383 if (BB->isEntryBlock())
8384 return false;
8385
8386 // Single successor and must be Succ.
8387 // FIXME: Relax that the terminator is a BranchInst by checking for equality
8388 // on other kinds of terminators. We decide to only support unconditional
8389 // branches for now for compile time reasons.
8390 auto *BI = dyn_cast<UncondBrInst>(BB->getTerminator());
8391 if (!BI)
8392 return false;
8393
8394 // Avoid blocks that are "address-taken" (blockaddress) or have unusual
8395 // uses.
8396 if (BB->hasAddressTaken() || BB->isEHPad())
8397 return false;
8398
8399 // TODO: relax this condition to merge equal blocks with >1 instructions?
8400 // Here, we use a O(1) form of the O(n) comparison of `size() != 1`.
8401 if (&BB->front() != &BB->back())
8402 return false;
8403
8404 // The BB must have at least one predecessor.
8405 if (pred_empty(BB))
8406 return false;
8407
8408 return true;
8409 }
8410};
8411
8413 static unsigned getHashValue(const EqualBBWrapper *EBW) {
8414 BasicBlock *BB = EBW->BB;
8416 assert(BB->size() == 1 && "Expected just a single branch in the BB");
8417
8418 // Since we assume the BB is just a single UncondBrInst with a single
8419 // successor, we hash as the BB and the incoming Values of its successor
8420 // PHIs. Initially, we tried to just use the successor BB as the hash, but
8421 // including the incoming PHI values leads to better performance.
8422 // We also tried to build a map from BB -> Succs.IncomingValues ahead of
8423 // time and passing it in EqualBBWrapper, but this slowed down the average
8424 // compile time without having any impact on the worst case compile time.
8425 BasicBlock *Succ = BI->getSuccessor();
8426 auto PhiValsForBB = map_range(Succ->phis(), [&](PHINode &Phi) {
8427 return (*EBW->PhiPredIVs)[&Phi][BB];
8428 });
8429 return hash_combine(Succ, hash_combine_range(PhiValsForBB));
8430 }
8431 static bool isEqual(const EqualBBWrapper *LHS, const EqualBBWrapper *RHS) {
8432 BasicBlock *A = LHS->BB;
8433 BasicBlock *B = RHS->BB;
8434
8435 // FIXME: we checked that the size of A and B are both 1 in
8436 // mergeIdenticalUncondBBs to make the Case list smaller to
8437 // improve performance. If we decide to support BasicBlocks with more
8438 // than just a single instruction, we need to check that A.size() ==
8439 // B.size() here, and we need to check more than just the BranchInsts
8440 // for equality.
8441
8442 UncondBrInst *ABI = cast<UncondBrInst>(A->getTerminator());
8443 UncondBrInst *BBI = cast<UncondBrInst>(B->getTerminator());
8444 if (ABI->getSuccessor() != BBI->getSuccessor())
8445 return false;
8446
8447 // Need to check that PHIs in successor have matching values.
8448 BasicBlock *Succ = ABI->getSuccessor();
8449 auto IfPhiIVMatch = [&](PHINode &Phi) {
8450 // Replace O(|Pred|) Phi.getIncomingValueForBlock with this O(1) hashmap
8451 // query.
8452 auto &PredIVs = (*LHS->PhiPredIVs)[&Phi];
8453 return PredIVs[A] == PredIVs[B];
8454 };
8455 return all_of(Succ->phis(), IfPhiIVMatch);
8456 }
8457};
8458
8459// Merge identical BBs into one of them.
8461 DomTreeUpdater *DTU) {
8462 if (Candidates.size() < 2)
8463 return false;
8464
8465 // Build Cases. Skip BBs that are not candidates for simplification. Mark
8466 // PHINodes which need to be processed into PhiPredIVs. We decide to process
8467 // an entire PHI at once after the loop, opposed to calling
8468 // getIncomingValueForBlock inside this loop, since each call to
8469 // getIncomingValueForBlock is O(|Preds|).
8470 EqualBBWrapper::Phi2IVsMap PhiPredIVs;
8472 BBs2Merge.reserve(Candidates.size());
8474
8475 for (BasicBlock *BB : Candidates) {
8476 BasicBlock *Succ = BB->getSingleSuccessor();
8477 assert(Succ && "Expected unconditional BB");
8478 BBs2Merge.emplace_back(EqualBBWrapper{BB, &PhiPredIVs});
8479 Phis.insert_range(make_pointer_range(Succ->phis()));
8480 }
8481
8482 // Precompute a data structure to improve performance of isEqual for
8483 // EqualBBWrapper.
8484 PhiPredIVs.reserve(Phis.size());
8485 for (PHINode *Phi : Phis) {
8486 auto &IVs =
8487 PhiPredIVs.try_emplace(Phi, Phi->getNumIncomingValues()).first->second;
8488 // Pre-fill all incoming for O(1) lookup as Phi.getIncomingValueForBlock is
8489 // O(|Pred|).
8490 for (auto &IV : Phi->incoming_values())
8491 IVs.insert({Phi->getIncomingBlock(IV), IV.get()});
8492 }
8493
8494 // Group duplicates using DenseSet with custom equality/hashing.
8495 // Build a set such that if the EqualBBWrapper exists in the set and another
8496 // EqualBBWrapper isEqual, then the equivalent EqualBBWrapper which is not in
8497 // the set should be replaced with the one in the set. If the EqualBBWrapper
8498 // is not in the set, then it should be added to the set so other
8499 // EqualBBWrapper can check against it in the same manner. We use
8500 // EqualBBWrapper instead of just BasicBlock because we'd like to pass around
8501 // information to isEquality, getHashValue, and when doing the replacement
8502 // with better performance.
8504 Keep.reserve(BBs2Merge.size());
8505
8507 Updates.reserve(BBs2Merge.size() * 2);
8508
8509 bool MadeChange = false;
8510
8511 // Helper: redirect all edges X -> DeadPred to X -> LivePred.
8512 auto RedirectIncomingEdges = [&](BasicBlock *Dead, BasicBlock *Live) {
8515 if (DTU) {
8516 // All predecessors of DeadPred (except the common predecessor) will be
8517 // moved to LivePred.
8518 Updates.reserve(Updates.size() + DeadPreds.size() * 2);
8520 predecessors(Live));
8521 for (BasicBlock *PredOfDead : DeadPreds) {
8522 // Do not modify those common predecessors of DeadPred and LivePred.
8523 if (!LivePreds.contains(PredOfDead))
8524 Updates.push_back({DominatorTree::Insert, PredOfDead, Live});
8525 Updates.push_back({DominatorTree::Delete, PredOfDead, Dead});
8526 }
8527 }
8528 LLVM_DEBUG(dbgs() << "Replacing duplicate pred BB ";
8529 Dead->printAsOperand(dbgs()); dbgs() << " with pred ";
8530 Live->printAsOperand(dbgs()); dbgs() << " for ";
8531 Live->getSingleSuccessor()->printAsOperand(dbgs());
8532 dbgs() << "\n");
8533 // Replace successors in all predecessors of DeadPred.
8534 for (BasicBlock *PredOfDead : DeadPreds) {
8535 Instruction *T = PredOfDead->getTerminator();
8536 T->replaceSuccessorWith(Dead, Live);
8537 }
8538 };
8539
8540 // Try to eliminate duplicate predecessors.
8541 for (const auto &EBW : BBs2Merge) {
8542 // EBW is a candidate for simplification. If we find a duplicate BB,
8543 // replace it.
8544 const auto &[It, Inserted] = Keep.insert(&EBW);
8545 if (Inserted)
8546 continue;
8547
8548 // Found duplicate: merge P into canonical predecessor It->Pred.
8549 BasicBlock *KeepBB = (*It)->BB;
8550 BasicBlock *DeadBB = EBW.BB;
8551
8552 // Avoid merging a BB with itself.
8553 if (KeepBB == DeadBB)
8554 continue;
8555
8556 // Redirect all edges into DeadPred to KeepPred.
8557 RedirectIncomingEdges(DeadBB, KeepBB);
8558
8559 // Now DeadBB should become unreachable; leave DCE to later,
8560 // but we can try to simplify it if it only branches to Succ.
8561 // (We won't erase here to keep the routine simple and DT-safe.)
8562 assert(pred_empty(DeadBB) && "DeadBB should be unreachable.");
8563 MadeChange = true;
8564 }
8565
8566 if (DTU && !Updates.empty())
8567 DTU->applyUpdates(Updates);
8568
8569 return MadeChange;
8570}
8571
8572bool SimplifyCFGOpt::simplifyDuplicateSwitchArms(SwitchInst *SI,
8573 DomTreeUpdater *DTU) {
8574 // Collect candidate switch-arms top-down.
8575 SmallSetVector<BasicBlock *, 16> FilteredArms(
8578 return mergeIdenticalBBs(FilteredArms.getArrayRef(), DTU);
8579}
8580
8581bool SimplifyCFGOpt::simplifyDuplicatePredecessors(BasicBlock *BB,
8582 DomTreeUpdater *DTU) {
8583 // Need at least 2 predecessors to do anything.
8584 if (!BB || !BB->hasNPredecessorsOrMore(2))
8585 return false;
8586
8587 // Compilation time consideration: retain the canonical loop, otherwise, we
8588 // require more time in the later loop canonicalization.
8589 if (Options.NeedCanonicalLoop && is_contained(LoopHeaders, BB))
8590 return false;
8591
8592 // Collect candidate predecessors bottom-up.
8593 SmallSetVector<BasicBlock *, 8> FilteredPreds(
8596 return mergeIdenticalBBs(FilteredPreds.getArrayRef(), DTU);
8597}
8598
8599bool SimplifyCFGOpt::simplifySwitch(SwitchInst *SI, IRBuilder<> &Builder) {
8600 BasicBlock *BB = SI->getParent();
8601
8602 if (isValueEqualityComparison(SI)) {
8603 // If we only have one predecessor, and if it is a branch on this value,
8604 // see if that predecessor totally determines the outcome of this switch.
8605 if (BasicBlock *OnlyPred = BB->getSinglePredecessor())
8606 if (simplifyEqualityComparisonWithOnlyPredecessor(SI, OnlyPred, Builder))
8607 return requestResimplify();
8608
8609 Value *Cond = SI->getCondition();
8610 if (SelectInst *Select = dyn_cast<SelectInst>(Cond))
8611 if (simplifySwitchOnSelect(SI, Select))
8612 return requestResimplify();
8613
8614 // If the block only contains the switch, see if we can fold the block
8615 // away into any preds.
8616 if (SI == &*BB->begin())
8617 if (foldValueComparisonIntoPredecessors(SI, Builder))
8618 return requestResimplify();
8619 }
8620
8621 // Try to transform the switch into an icmp and a branch.
8622 // The conversion from switch to comparison may lose information on
8623 // impossible switch values, so disable it early in the pipeline.
8624 if (Options.ConvertSwitchRangeToICmp && turnSwitchRangeIntoICmp(SI, Builder))
8625 return requestResimplify();
8626
8627 // Remove unreachable cases.
8628 if (eliminateDeadSwitchCases(SI, DTU, Options.AC, DL))
8629 return requestResimplify();
8630
8631 if (simplifySwitchOfCmpIntrinsic(SI, Builder, DTU))
8632 return requestResimplify();
8633
8634 if (trySwitchToSelect(SI, Builder, DTU, DL))
8635 return requestResimplify();
8636
8637 if (Options.ForwardSwitchCondToPhi && forwardSwitchConditionToPHI(SI))
8638 return requestResimplify();
8639
8640 // The conversion of switches to arithmetic or lookup table is disabled in
8641 // the early optimization pipeline, as it may lose information or make the
8642 // resulting code harder to analyze.
8643 if (Options.ConvertSwitchToArithmetic || Options.ConvertSwitchToLookupTable)
8644 if (simplifySwitchLookup(SI, Builder, DTU, DL, TTI,
8645 Options.ConvertSwitchToLookupTable))
8646 return requestResimplify();
8647
8648 if (simplifySwitchOfPowersOfTwo(SI, Builder, DTU, DL, TTI))
8649 return requestResimplify();
8650
8651 if (reduceSwitchRange(SI, Builder, DL, TTI))
8652 return requestResimplify();
8653
8654 if (HoistCommon &&
8655 hoistCommonCodeFromSuccessors(SI, !Options.HoistCommonInsts))
8656 return requestResimplify();
8657
8658 // We can merge identical switch arms early to enhance more aggressive
8659 // optimization on switch.
8660 if (simplifyDuplicateSwitchArms(SI, DTU))
8661 return requestResimplify();
8662
8663 if (simplifySwitchWhenUMin(SI, DTU))
8664 return requestResimplify();
8665
8666 if (simplifySwitchDefaultBranch(SI, DTU, DL, Options.AC))
8667 return requestResimplify();
8668
8669 return false;
8670}
8671
8672bool SimplifyCFGOpt::simplifyIndirectBr(IndirectBrInst *IBI) {
8673 BasicBlock *BB = IBI->getParent();
8674 bool Changed = false;
8675 SmallVector<uint32_t> BranchWeights;
8676 const bool HasBranchWeights = extractBranchWeights(*IBI, BranchWeights);
8677
8678 DenseMap<const BasicBlock *, uint64_t> TargetWeight;
8679 if (HasBranchWeights)
8680 for (size_t I = 0, E = IBI->getNumDestinations(); I < E; ++I)
8681 TargetWeight[IBI->getDestination(I)] += BranchWeights[I];
8682
8683 // Eliminate redundant destinations.
8684 SmallPtrSet<Value *, 8> Succs;
8685 SmallSetVector<BasicBlock *, 8> RemovedSuccs;
8686 for (unsigned I = 0, E = IBI->getNumDestinations(); I != E; ++I) {
8687 BasicBlock *Dest = IBI->getDestination(I);
8688 if (!Dest->hasAddressTaken() || !Succs.insert(Dest).second) {
8689 if (!Dest->hasAddressTaken())
8690 RemovedSuccs.insert(Dest);
8691 Dest->removePredecessor(BB);
8692 IBI->removeDestination(I);
8693 --I;
8694 --E;
8695 Changed = true;
8696 }
8697 }
8698
8699 if (DTU) {
8700 std::vector<DominatorTree::UpdateType> Updates;
8701 Updates.reserve(RemovedSuccs.size());
8702 for (auto *RemovedSucc : RemovedSuccs)
8703 Updates.push_back({DominatorTree::Delete, BB, RemovedSucc});
8704 DTU->applyUpdates(Updates);
8705 }
8706
8707 if (IBI->getNumDestinations() == 0) {
8708 // If the indirectbr has no successors, change it to unreachable.
8709 new UnreachableInst(IBI->getContext(), IBI->getIterator());
8711 return true;
8712 }
8713
8714 if (IBI->getNumDestinations() == 1) {
8715 // If the indirectbr has one successor, change it to a direct branch.
8718 return true;
8719 }
8720 if (HasBranchWeights) {
8721 SmallVector<uint64_t> NewBranchWeights(IBI->getNumDestinations());
8722 for (size_t I = 0, E = IBI->getNumDestinations(); I < E; ++I)
8723 NewBranchWeights[I] += TargetWeight.find(IBI->getDestination(I))->second;
8724 setFittedBranchWeights(*IBI, NewBranchWeights, /*IsExpected=*/false);
8725 }
8726 if (SelectInst *SI = dyn_cast<SelectInst>(IBI->getAddress())) {
8727 if (simplifyIndirectBrOnSelect(IBI, SI))
8728 return requestResimplify();
8729 }
8730 return Changed;
8731}
8732
8733/// Given an block with only a single landing pad and a unconditional branch
8734/// try to find another basic block which this one can be merged with. This
8735/// handles cases where we have multiple invokes with unique landing pads, but
8736/// a shared handler.
8737///
8738/// We specifically choose to not worry about merging non-empty blocks
8739/// here. That is a PRE/scheduling problem and is best solved elsewhere. In
8740/// practice, the optimizer produces empty landing pad blocks quite frequently
8741/// when dealing with exception dense code. (see: instcombine, gvn, if-else
8742/// sinking in this file)
8743///
8744/// This is primarily a code size optimization. We need to avoid performing
8745/// any transform which might inhibit optimization (such as our ability to
8746/// specialize a particular handler via tail commoning). We do this by not
8747/// merging any blocks which require us to introduce a phi. Since the same
8748/// values are flowing through both blocks, we don't lose any ability to
8749/// specialize. If anything, we make such specialization more likely.
8750///
8751/// TODO - This transformation could remove entries from a phi in the target
8752/// block when the inputs in the phi are the same for the two blocks being
8753/// merged. In some cases, this could result in removal of the PHI entirely.
8755 BasicBlock *BB, DomTreeUpdater *DTU) {
8756 auto Succ = BB->getUniqueSuccessor();
8757 assert(Succ);
8758 // If there's a phi in the successor block, we'd likely have to introduce
8759 // a phi into the merged landing pad block.
8760 if (isa<PHINode>(*Succ->begin()))
8761 return false;
8762
8763 for (BasicBlock *OtherPred : predecessors(Succ)) {
8764 if (BB == OtherPred)
8765 continue;
8766 BasicBlock::iterator I = OtherPred->begin();
8768 if (!LPad2 || !LPad2->isIdenticalTo(LPad))
8769 continue;
8770 ++I;
8772 if (!BI2 || !BI2->isIdenticalTo(BI))
8773 continue;
8774
8775 std::vector<DominatorTree::UpdateType> Updates;
8776
8777 // We've found an identical block. Update our predecessors to take that
8778 // path instead and make ourselves dead.
8780 for (BasicBlock *Pred : UniquePreds) {
8781 InvokeInst *II = cast<InvokeInst>(Pred->getTerminator());
8782 assert(II->getNormalDest() != BB && II->getUnwindDest() == BB &&
8783 "unexpected successor");
8784 II->setUnwindDest(OtherPred);
8785 if (DTU) {
8786 Updates.push_back({DominatorTree::Insert, Pred, OtherPred});
8787 Updates.push_back({DominatorTree::Delete, Pred, BB});
8788 }
8789 }
8790
8792 for (BasicBlock *Succ : UniqueSuccs) {
8793 Succ->removePredecessor(BB);
8794 if (DTU)
8795 Updates.push_back({DominatorTree::Delete, BB, Succ});
8796 }
8797
8798 IRBuilder<> Builder(BI);
8799 Builder.CreateUnreachable();
8800 BI->eraseFromParent();
8801 if (DTU)
8802 DTU->applyUpdates(Updates);
8803 return true;
8804 }
8805 return false;
8806}
8807
8808bool SimplifyCFGOpt::simplifyUncondBranch(UncondBrInst *BI,
8809 IRBuilder<> &Builder) {
8810 BasicBlock *BB = BI->getParent();
8811 BasicBlock *Succ = BI->getSuccessor(0);
8812
8813 // If the Terminator is the only non-phi instruction, simplify the block.
8814 // If LoopHeader is provided, check if the block or its successor is a loop
8815 // header. (This is for early invocations before loop simplify and
8816 // vectorization to keep canonical loop forms for nested loops. These blocks
8817 // can be eliminated when the pass is invoked later in the back-end.)
8818 // Note that if BB has only one predecessor then we do not introduce new
8819 // backedge, so we can eliminate BB.
8820 bool NeedCanonicalLoop =
8821 Options.NeedCanonicalLoop &&
8822 (!LoopHeaders.empty() && BB->hasNPredecessorsOrMore(2) &&
8823 (is_contained(LoopHeaders, BB) || is_contained(LoopHeaders, Succ)));
8825 if (I->isTerminator() && BB != &BB->getParent()->getEntryBlock() &&
8826 !NeedCanonicalLoop && TryToSimplifyUncondBranchFromEmptyBlock(BB, DTU))
8827 return true;
8828
8829 // If the only instruction in the block is a seteq/setne comparison against a
8830 // constant, try to simplify the block.
8831 if (ICmpInst *ICI = dyn_cast<ICmpInst>(I)) {
8832 if (ICI->isEquality() && isa<ConstantInt>(ICI->getOperand(1))) {
8833 ++I;
8834 if (I->isTerminator() &&
8835 tryToSimplifyUncondBranchWithICmpInIt(ICI, Builder))
8836 return true;
8837 if (isa<SelectInst>(I) && I->getNextNode()->isTerminator() &&
8838 tryToSimplifyUncondBranchWithICmpSelectInIt(ICI, cast<SelectInst>(I),
8839 Builder))
8840 return true;
8841 }
8842 }
8843
8844 // See if we can merge an empty landing pad block with another which is
8845 // equivalent.
8846 if (LandingPadInst *LPad = dyn_cast<LandingPadInst>(I)) {
8847 ++I;
8848 if (I->isTerminator() && tryToMergeLandingPad(LPad, BI, BB, DTU))
8849 return true;
8850 }
8851
8852 return false;
8853}
8854
8856 BasicBlock *PredPred = nullptr;
8857 for (auto *P : predecessors(BB)) {
8858 BasicBlock *PPred = P->getSinglePredecessor();
8859 if (!PPred || (PredPred && PredPred != PPred))
8860 return nullptr;
8861 PredPred = PPred;
8862 }
8863 return PredPred;
8864}
8865
8866/// Fold the following pattern:
8867/// bb0:
8868/// br i1 %cond1, label %bb1, label %bb2
8869/// bb1:
8870/// br i1 %cond2, label %bb3, label %bb4
8871/// bb2:
8872/// br i1 %cond2, label %bb4, label %bb3
8873/// bb3:
8874/// ...
8875/// bb4:
8876/// ...
8877/// into
8878/// bb0:
8879/// %cond = xor i1 %cond1, %cond2
8880/// br i1 %cond, label %bb4, label %bb3
8881/// bb3:
8882/// ...
8883/// bb4:
8884/// ...
8885/// NOTE: %cond2 always dominates the terminator of bb0.
8887 BasicBlock *BB = BI->getParent();
8888 BasicBlock *BB1 = BI->getSuccessor(0);
8889 BasicBlock *BB2 = BI->getSuccessor(1);
8890 auto IsSimpleSuccessor = [BB](BasicBlock *Succ, CondBrInst *&SuccBI) {
8891 if (Succ == BB)
8892 return false;
8893 if (&Succ->front() != Succ->getTerminator())
8894 return false;
8895 SuccBI = dyn_cast<CondBrInst>(Succ->getTerminator());
8896 if (!SuccBI)
8897 return false;
8898 BasicBlock *Succ1 = SuccBI->getSuccessor(0);
8899 BasicBlock *Succ2 = SuccBI->getSuccessor(1);
8900 return Succ1 != Succ && Succ2 != Succ && Succ1 != BB && Succ2 != BB &&
8901 !isa<PHINode>(Succ1->front()) && !isa<PHINode>(Succ2->front());
8902 };
8903 CondBrInst *BB1BI, *BB2BI;
8904 if (!IsSimpleSuccessor(BB1, BB1BI) || !IsSimpleSuccessor(BB2, BB2BI))
8905 return false;
8906
8907 if (BB1BI->getCondition() != BB2BI->getCondition() ||
8908 BB1BI->getSuccessor(0) != BB2BI->getSuccessor(1) ||
8909 BB1BI->getSuccessor(1) != BB2BI->getSuccessor(0))
8910 return false;
8911
8912 BasicBlock *BB3 = BB1BI->getSuccessor(0);
8913 BasicBlock *BB4 = BB1BI->getSuccessor(1);
8914 // Bail out on trivial cases to avoid bothering to handle the special case in
8915 // the code below.
8916 if (BB3 == BB4)
8917 return false;
8918 IRBuilder<> Builder(BI);
8919 BI->setCondition(
8920 Builder.CreateXor(BI->getCondition(), BB1BI->getCondition()));
8921 BB1->removePredecessor(BB);
8922 BI->setSuccessor(0, BB4);
8923 BB2->removePredecessor(BB);
8924 BI->setSuccessor(1, BB3);
8925 if (DTU) {
8927 Updates.push_back({DominatorTree::Delete, BB, BB1});
8928 Updates.push_back({DominatorTree::Insert, BB, BB4});
8929 Updates.push_back({DominatorTree::Delete, BB, BB2});
8930 Updates.push_back({DominatorTree::Insert, BB, BB3});
8931
8932 DTU->applyUpdates(Updates);
8933 }
8934 bool HasWeight = false;
8935 uint64_t BBTWeight, BBFWeight;
8936 if (extractBranchWeights(*BI, BBTWeight, BBFWeight))
8937 HasWeight = true;
8938 else
8939 BBTWeight = BBFWeight = 1;
8940 uint64_t BB1TWeight, BB1FWeight;
8941 if (extractBranchWeights(*BB1BI, BB1TWeight, BB1FWeight))
8942 HasWeight = true;
8943 else
8944 BB1TWeight = BB1FWeight = 1;
8945 uint64_t BB2TWeight, BB2FWeight;
8946 if (extractBranchWeights(*BB2BI, BB2TWeight, BB2FWeight))
8947 HasWeight = true;
8948 else
8949 BB2TWeight = BB2FWeight = 1;
8950 if (HasWeight) {
8951 uint64_t Weights[2] = {BBTWeight * BB1FWeight + BBFWeight * BB2TWeight,
8952 BBTWeight * BB1TWeight + BBFWeight * BB2FWeight};
8953 setFittedBranchWeights(*BI, Weights, /*IsExpected=*/false,
8954 /*ElideAllZero=*/true);
8955 }
8956 return true;
8957}
8958
8959bool SimplifyCFGOpt::simplifyCondBranch(CondBrInst *BI, IRBuilder<> &Builder) {
8960 assert(
8962 BI->getSuccessor(0) != BI->getSuccessor(1) &&
8963 "Tautological conditional branch should have been eliminated already.");
8964
8965 BasicBlock *BB = BI->getParent();
8966 if (!Options.SimplifyCondBranch ||
8967 BI->getFunction()->hasFnAttribute(Attribute::OptForFuzzing))
8968 return false;
8969
8970 // Conditional branch
8971 if (isValueEqualityComparison(BI)) {
8972 // If we only have one predecessor, and if it is a branch on this value,
8973 // see if that predecessor totally determines the outcome of this
8974 // switch.
8975 if (BasicBlock *OnlyPred = BB->getSinglePredecessor())
8976 if (simplifyEqualityComparisonWithOnlyPredecessor(BI, OnlyPred, Builder))
8977 return requestResimplify();
8978
8979 // This block must be empty, except for the setcond inst, if it exists.
8980 // Ignore pseudo intrinsics.
8981 for (auto &I : *BB) {
8982 if (isa<PseudoProbeInst>(I) ||
8983 &I == cast<Instruction>(BI->getCondition()))
8984 continue;
8985 if (&I == BI)
8986 if (foldValueComparisonIntoPredecessors(BI, Builder))
8987 return requestResimplify();
8988 break;
8989 }
8990 }
8991
8992 // Try to turn "br (X == 0 | X == 1), T, F" into a switch instruction.
8993 if (simplifyBranchOnICmpChain(BI, Builder, DL))
8994 return true;
8995
8996 // If this basic block has dominating predecessor blocks and the dominating
8997 // blocks' conditions imply BI's condition, we know the direction of BI.
8998 std::optional<bool> Imp = isImpliedByDomCondition(BI->getCondition(), BI, DL);
8999 if (Imp) {
9000 // Turn this into a branch on constant.
9001 auto *OldCond = BI->getCondition();
9002 ConstantInt *TorF = *Imp ? ConstantInt::getTrue(BB->getContext())
9003 : ConstantInt::getFalse(BB->getContext());
9004 BI->setCondition(TorF);
9006 return requestResimplify();
9007 }
9008
9009 // If this basic block is ONLY a compare and a branch, and if a predecessor
9010 // branches to us and one of our successors, fold the comparison into the
9011 // predecessor and use logical operations to pick the right destination.
9012 if (Options.SpeculateBlocks &&
9013 foldBranchToCommonDest(BI, DTU, /*MSSAU=*/nullptr, &TTI, Options.AC,
9014 Options.BonusInstThreshold))
9015 return requestResimplify();
9016
9017 // We have a conditional branch to two blocks that are only reachable
9018 // from BI. We know that the condbr dominates the two blocks, so see if
9019 // there is any identical code in the "then" and "else" blocks. If so, we
9020 // can hoist it up to the branching block.
9021 if (BI->getSuccessor(0)->getSinglePredecessor()) {
9022 if (BI->getSuccessor(1)->getSinglePredecessor()) {
9023 if (HoistCommon &&
9024 hoistCommonCodeFromSuccessors(BI, !Options.HoistCommonInsts))
9025 return requestResimplify();
9026
9027 if (BI && Options.HoistLoadsStoresWithCondFaulting &&
9028 isProfitableToSpeculate(BI, std::nullopt, TTI)) {
9029 SmallVector<Instruction *, 2> SpeculatedConditionalLoadsStores;
9030 auto CanSpeculateConditionalLoadsStores = [&]() {
9031 for (auto *Succ : successors(BB)) {
9032 for (Instruction &I : *Succ) {
9033 if (I.isTerminator()) {
9034 if (I.getNumSuccessors() > 1)
9035 return false;
9036 continue;
9037 } else if (!isSafeCheapLoadStore(&I, TTI) ||
9038 SpeculatedConditionalLoadsStores.size() ==
9040 return false;
9041 }
9042 SpeculatedConditionalLoadsStores.push_back(&I);
9043 }
9044 }
9045 return !SpeculatedConditionalLoadsStores.empty();
9046 };
9047
9048 if (CanSpeculateConditionalLoadsStores()) {
9049 hoistConditionalLoadsStores(BI, SpeculatedConditionalLoadsStores,
9050 std::nullopt, nullptr);
9051 return requestResimplify();
9052 }
9053 }
9054 } else {
9055 // If Successor #1 has multiple preds, we may be able to conditionally
9056 // execute Successor #0 if it branches to Successor #1.
9057 Instruction *Succ0TI = BI->getSuccessor(0)->getTerminator();
9058 if (Succ0TI->getNumSuccessors() == 1 &&
9059 Succ0TI->getSuccessor(0) == BI->getSuccessor(1))
9060 if (speculativelyExecuteBB(BI, BI->getSuccessor(0)))
9061 return requestResimplify();
9062 }
9063 } else if (BI->getSuccessor(1)->getSinglePredecessor()) {
9064 // If Successor #0 has multiple preds, we may be able to conditionally
9065 // execute Successor #1 if it branches to Successor #0.
9066 Instruction *Succ1TI = BI->getSuccessor(1)->getTerminator();
9067 if (Succ1TI->getNumSuccessors() == 1 &&
9068 Succ1TI->getSuccessor(0) == BI->getSuccessor(0))
9069 if (speculativelyExecuteBB(BI, BI->getSuccessor(1)))
9070 return requestResimplify();
9071 }
9072
9073 // If this is a branch on something for which we know the constant value in
9074 // predecessors (e.g. a phi node in the current block), thread control
9075 // through this block.
9076 if (foldCondBranchOnValueKnownInPredecessor(BI))
9077 return requestResimplify();
9078
9079 // Scan predecessor blocks for conditional branches.
9080 for (BasicBlock *Pred : predecessors(BB))
9081 if (CondBrInst *PBI = dyn_cast<CondBrInst>(Pred->getTerminator()))
9082 if (PBI != BI)
9083 if (SimplifyCondBranchToCondBranch(PBI, BI, DTU, DL, TTI))
9084 return requestResimplify();
9085
9086 // Look for diamond patterns.
9087 if (MergeCondStores)
9088 if (BasicBlock *PrevBB = allPredecessorsComeFromSameSource(BB))
9089 if (CondBrInst *PBI = dyn_cast<CondBrInst>(PrevBB->getTerminator()))
9090 if (PBI != BI)
9091 if (mergeConditionalStores(PBI, BI, DTU, DL, TTI))
9092 return requestResimplify();
9093
9094 // Look for nested conditional branches.
9095 if (mergeNestedCondBranch(BI, DTU))
9096 return requestResimplify();
9097
9098 return false;
9099}
9100
9101/// Check if passing a value to an instruction will cause undefined behavior.
9102static bool passingValueIsAlwaysUndefined(Value *V, Instruction *I, bool PtrValueMayBeModified) {
9103 assert(V->getType() == I->getType() && "Mismatched types");
9105 if (!C)
9106 return false;
9107
9108 if (I->use_empty())
9109 return false;
9110
9111 if (C->isNullValue() || isa<UndefValue>(C)) {
9112 // Find the first same-block use with a UB-triggering opcode, skipping
9113 // cross-block or before-I uses.
9114 auto FindUse = llvm::find_if(I->uses(), [I](auto &U) {
9115 auto *Use = cast<Instruction>(U.getUser());
9116 // Only same-block uses after I can witness UB at I's program point.
9117 // Self-uses and before-I uses can occur when I is a PHI node.
9118 if (Use->getParent() != I->getParent() || Use == I || Use->comesBefore(I))
9119 return false;
9120 // Change this list when we want to add new instructions.
9121 switch (Use->getOpcode()) {
9122 default:
9123 return false;
9124 case Instruction::GetElementPtr:
9125 case Instruction::Ret:
9126 case Instruction::BitCast:
9127 case Instruction::Load:
9128 case Instruction::Store:
9129 case Instruction::Call:
9130 case Instruction::CallBr:
9131 case Instruction::Invoke:
9132 case Instruction::UDiv:
9133 case Instruction::URem:
9134 // Note: signed div/rem of INT_MIN / -1 is also immediate UB, not
9135 // implemented to avoid code complexity as it is unclear how useful such
9136 // logic is.
9137 case Instruction::SDiv:
9138 case Instruction::SRem:
9139 return true;
9140 }
9141 });
9142 if (FindUse == I->use_end())
9143 return false;
9144 auto &Use = *FindUse;
9145 auto *User = cast<Instruction>(Use.getUser());
9146
9147 // Now make sure that there are no instructions in between that can alter
9148 // control flow (eg. calls)
9149 auto InstrRange =
9150 make_range(std::next(I->getIterator()), User->getIterator());
9151 if (any_of(InstrRange, [](Instruction &I) {
9153 }))
9154 return false;
9155
9156 // Look through GEPs. A load from a GEP derived from NULL is still undefined
9158 if (GEP->getPointerOperand() == I) {
9159 // The type of GEP may differ from the type of base pointer.
9160 // Bail out on vector GEPs, as they are not handled by other checks.
9161 if (GEP->getType()->isVectorTy())
9162 return false;
9163 // The current base address is null, there are four cases to consider:
9164 // getelementptr (TY, null, 0) -> null
9165 // getelementptr (TY, null, not zero) -> may be modified
9166 // getelementptr inbounds (TY, null, 0) -> null
9167 // getelementptr inbounds (TY, null, not zero) -> poison iff null is
9168 // undefined?
9169 if (!GEP->hasAllZeroIndices() &&
9170 (!GEP->isInBounds() ||
9171 NullPointerIsDefined(GEP->getFunction(),
9172 GEP->getPointerAddressSpace())))
9173 PtrValueMayBeModified = true;
9174 return passingValueIsAlwaysUndefined(V, GEP, PtrValueMayBeModified);
9175 }
9176
9177 // Look through return.
9178 if (ReturnInst *Ret = dyn_cast<ReturnInst>(User)) {
9179 bool HasNoUndefAttr =
9180 Ret->getFunction()->hasRetAttribute(Attribute::NoUndef);
9181 // Return undefined to a noundef return value is undefined.
9182 if (isa<UndefValue>(C) && HasNoUndefAttr)
9183 return true;
9184 // Return null to a nonnull+noundef return value is undefined.
9185 if (C->isNullValue() && HasNoUndefAttr &&
9186 Ret->getFunction()->hasRetAttribute(Attribute::NonNull)) {
9187 return !PtrValueMayBeModified;
9188 }
9189 }
9190
9191 // Load from null is undefined.
9192 if (LoadInst *LI = dyn_cast<LoadInst>(User))
9193 if (!LI->isVolatile())
9194 return !NullPointerIsDefined(LI->getFunction(),
9195 LI->getPointerAddressSpace());
9196
9197 // Store to null is undefined.
9199 if (!SI->isVolatile())
9200 return (!NullPointerIsDefined(SI->getFunction(),
9201 SI->getPointerAddressSpace())) &&
9202 SI->getPointerOperand() == I;
9203
9204 // llvm.assume(false/undef) always triggers immediate UB.
9205 if (auto *Assume = dyn_cast<AssumeInst>(User)) {
9206 // Ignore assume operand bundles.
9207 if (I == Assume->getArgOperand(0))
9208 return true;
9209 }
9210
9211 if (auto *CB = dyn_cast<CallBase>(User)) {
9212 if (C->isNullValue() && NullPointerIsDefined(CB->getFunction()))
9213 return false;
9214 // A call to null is undefined.
9215 if (CB->getCalledOperand() == I)
9216 return true;
9217
9218 if (CB->isArgOperand(&Use)) {
9219 unsigned ArgIdx = CB->getArgOperandNo(&Use);
9220 // Passing null to a nonnnull+noundef argument is undefined.
9221 if (isa<ConstantPointerNull>(C) && C->getType()->isPointerTy() &&
9222 CB->paramHasNonNullAttr(ArgIdx, /*AllowUndefOrPoison=*/false))
9223 return !PtrValueMayBeModified;
9224 // Passing undef to a noundef argument is undefined.
9225 if (isa<UndefValue>(C) && CB->isPassingUndefUB(ArgIdx))
9226 return true;
9227 }
9228 }
9229 // Div/Rem by zero is immediate UB
9230 if (match(User, m_BinOp(m_Value(), m_Specific(I))) && User->isIntDivRem())
9231 return true;
9232 }
9233 return false;
9234}
9235
9236/// If BB has an incoming value that will always trigger undefined behavior
9237/// (eg. null pointer dereference), remove the branch leading here.
9239 DomTreeUpdater *DTU,
9240 AssumptionCache *AC) {
9241 for (PHINode &PHI : BB->phis())
9242 for (unsigned i = 0, e = PHI.getNumIncomingValues(); i != e; ++i)
9243 if (passingValueIsAlwaysUndefined(PHI.getIncomingValue(i), &PHI)) {
9244 BasicBlock *Predecessor = PHI.getIncomingBlock(i);
9245 Instruction *T = Predecessor->getTerminator();
9246 IRBuilder<> Builder(T);
9247 if (isa<UncondBrInst>(T)) {
9248 BB->removePredecessor(Predecessor);
9249 // Turn unconditional branches into unreachables.
9250 Builder.CreateUnreachable();
9251 T->eraseFromParent();
9252 if (DTU)
9253 DTU->applyUpdates({{DominatorTree::Delete, Predecessor, BB}});
9254 return true;
9255 } else if (CondBrInst *BI = dyn_cast<CondBrInst>(T)) {
9256 BB->removePredecessor(Predecessor);
9257 // Handle degenerate conditional branches.
9258 if (BI->getSuccessor(0) == BI->getSuccessor(1)) {
9259 // The only difference from the UncondBrInst path above is that it
9260 // has two edges in CFG.
9261 BB->removePredecessor(Predecessor);
9262 // Turn unconditional branches into unreachables.
9263 Builder.CreateUnreachable();
9264 } else {
9265 // Preserve guarding condition in assume, because it might not be
9266 // inferrable from any dominating condition.
9267 Value *Cond = BI->getCondition();
9268 CallInst *Assumption;
9269 if (BI->getSuccessor(0) == BB)
9270 Assumption = Builder.CreateAssumption(Builder.CreateNot(Cond));
9271 else
9272 Assumption = Builder.CreateAssumption(Cond);
9273 if (AC)
9274 AC->registerAssumption(cast<AssumeInst>(Assumption));
9275 Builder.CreateBr(BI->getSuccessor(0) == BB ? BI->getSuccessor(1)
9276 : BI->getSuccessor(0));
9277 }
9278 BI->eraseFromParent();
9279 if (DTU)
9280 DTU->applyUpdates({{DominatorTree::Delete, Predecessor, BB}});
9281 return true;
9282 } else if (SwitchInst *SI = dyn_cast<SwitchInst>(T)) {
9283 // Redirect all branches leading to UB into
9284 // a newly created unreachable block.
9285 BasicBlock *Unreachable = BasicBlock::Create(
9286 Predecessor->getContext(), "unreachable", BB->getParent(), BB);
9287 Builder.SetInsertPoint(Unreachable);
9288 // The new block contains only one instruction: Unreachable
9289 Builder.CreateUnreachable();
9290 for (const auto &Case : SI->cases())
9291 if (Case.getCaseSuccessor() == BB) {
9292 BB->removePredecessor(Predecessor);
9293 Case.setSuccessor(Unreachable);
9294 }
9295 if (SI->getDefaultDest() == BB) {
9296 BB->removePredecessor(Predecessor);
9297 SI->setDefaultDest(Unreachable);
9298 }
9299
9300 if (DTU)
9301 DTU->applyUpdates(
9302 { { DominatorTree::Insert, Predecessor, Unreachable },
9303 { DominatorTree::Delete, Predecessor, BB } });
9304 return true;
9305 }
9306 }
9307
9308 return false;
9309}
9310
9311bool SimplifyCFGOpt::simplifyOnce(BasicBlock *BB) {
9312 bool Changed = false;
9313
9314 assert(BB && BB->getParent() && "Block not embedded in function!");
9315 assert(BB->getTerminator() && "Degenerate basic block encountered!");
9316
9317 // Remove basic blocks that have no predecessors (except the entry block)...
9318 // or that just have themself as a predecessor. These are unreachable.
9319 if ((pred_empty(BB) && BB != &BB->getParent()->getEntryBlock()) ||
9320 BB->getSinglePredecessor() == BB) {
9321 LLVM_DEBUG(dbgs() << "Removing BB: \n" << *BB);
9322 DeleteDeadBlock(BB, DTU);
9323 return true;
9324 }
9325
9326 // Check to see if we can constant propagate this terminator instruction
9327 // away...
9328 Changed |= ConstantFoldTerminator(BB, /*DeleteDeadConditions=*/true,
9329 /*TLI=*/nullptr, DTU);
9330
9331 // Check for and eliminate duplicate PHI nodes in this block.
9333
9334 // Check for and remove branches that will always cause undefined behavior.
9336 return requestResimplify();
9337
9338 // Merge basic blocks into their predecessor if there is only one distinct
9339 // pred, and if there is only one distinct successor of the predecessor, and
9340 // if there are no PHI nodes.
9341 if (MergeBlockIntoPredecessor(BB, DTU))
9342 return true;
9343
9344 if (SinkCommon && Options.SinkCommonInsts) {
9345 if (sinkCommonCodeFromPredecessors(BB, DTU) ||
9346 mergeCompatibleInvokes(BB, DTU)) {
9347 // sinkCommonCodeFromPredecessors() does not automatically CSE PHI's,
9348 // so we may now how duplicate PHI's.
9349 // Let's rerun EliminateDuplicatePHINodes() first,
9350 // before foldTwoEntryPHINode() potentially converts them into select's,
9351 // after which we'd need a whole EarlyCSE pass run to cleanup them.
9352 return true;
9353 }
9354 // Merge identical predecessors of this block.
9355 if (simplifyDuplicatePredecessors(BB, DTU))
9356 return true;
9357 }
9358
9359 if (Options.SpeculateBlocks &&
9360 !BB->getParent()->hasFnAttribute(Attribute::OptForFuzzing)) {
9361 // If there is a trivial two-entry PHI node in this basic block, and we can
9362 // eliminate it, do so now.
9363 if (auto *PN = dyn_cast<PHINode>(BB->begin()))
9364 if (PN->getNumIncomingValues() == 2)
9365 if (foldTwoEntryPHINode(PN, TTI, DTU, Options.AC, DL,
9366 Options.SpeculateUnpredictables))
9367 return true;
9368 }
9369
9370 IRBuilder<> Builder(BB);
9372 Builder.SetInsertPoint(Terminator);
9373 switch (Terminator->getOpcode()) {
9374 case Instruction::UncondBr:
9375 Changed |= simplifyUncondBranch(cast<UncondBrInst>(Terminator), Builder);
9376 break;
9377 case Instruction::CondBr:
9378 Changed |= simplifyCondBranch(cast<CondBrInst>(Terminator), Builder);
9379 break;
9380 case Instruction::Resume:
9381 Changed |= simplifyResume(cast<ResumeInst>(Terminator), Builder);
9382 break;
9383 case Instruction::CleanupRet:
9384 Changed |= simplifyCleanupReturn(cast<CleanupReturnInst>(Terminator));
9385 break;
9386 case Instruction::Switch:
9387 Changed |= simplifySwitch(cast<SwitchInst>(Terminator), Builder);
9388 break;
9389 case Instruction::Unreachable:
9390 Changed |= simplifyUnreachable(cast<UnreachableInst>(Terminator));
9391 break;
9392 case Instruction::IndirectBr:
9393 Changed |= simplifyIndirectBr(cast<IndirectBrInst>(Terminator));
9394 break;
9395 }
9396
9397 return Changed;
9398}
9399
9400bool SimplifyCFGOpt::run(BasicBlock *BB) {
9401 bool Changed = false;
9402
9403 // Repeated simplify BB as long as resimplification is requested.
9404 do {
9405 Resimplify = false;
9406
9407 // Perform one round of simplifcation. Resimplify flag will be set if
9408 // another iteration is requested.
9409 Changed |= simplifyOnce(BB);
9410 } while (Resimplify);
9411
9412 return Changed;
9413}
9414
9417 ArrayRef<WeakVH> LoopHeaders) {
9418 return SimplifyCFGOpt(TTI, DTU, BB->getDataLayout(), LoopHeaders,
9419 Options)
9420 .run(BB);
9421}
#define Fail
#define Success
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
aarch64 promote const
unsigned uint64_t
AMDGPU Register Bank Select
Rewrite undef for PHI
This file implements a class to represent arbitrary precision integral constant values and operations...
static MachineBasicBlock * OtherSucc(MachineBasicBlock *MBB, MachineBasicBlock *Succ)
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Function Alias Analysis Results
This file contains the simple types necessary to represent the attributes associated with functions a...
static const Function * getParent(const Value *V)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
This file defines the DenseMap class.
@ Default
#define DEBUG_TYPE
Hexagon Common GEP
static bool IsIndirectCall(const MachineInstr *MI)
This file provides various utilities for inspecting and working with the control flow graph in LLVM I...
Module.h This file contains the declarations for the Module class.
This defines the Use class.
static Constant * getFalse(Type *Ty)
For a boolean type or a vector of boolean type, return false or a vector with every element false.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static LVOptions Options
Definition LVOptions.cpp:25
#define I(x, y, z)
Definition MD5.cpp:57
Machine Check Debug Module
This file implements a map that provides insertion order iteration.
This file provides utility for Memory Model Relaxation Annotations (MMRAs).
This file exposes an interface to building/using memory SSA to walk memory instructions using a use/d...
This file contains the declarations for metadata subclasses.
#define T
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
#define P(N)
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
This file contains the declarations for profiling metadata utility functions.
static cl::opt< uint32_t > SelectFalseWeight("profcheck-default-select-false-weight", cl::init(3U), cl::desc("When annotating `select` instructions, this value will be used " "for the second ('false') case."))
static cl::opt< uint32_t > SelectTrueWeight("profcheck-default-select-true-weight", cl::init(2U), cl::desc("When annotating `select` instructions, this value will be used " "for the first ('true') case."))
const SmallVectorImpl< MachineOperand > & Cond
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
This file contains some templates that are useful if you are working with the STL at all.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
Provides some synthesis utilities to produce sequences of values.
This file defines generic set operations that may be used on set's of different types,...
This file implements a set that has insertion order iteration characteristics.
static std::optional< ContiguousCasesResult > findContiguousCases(Value *Condition, SmallVectorImpl< ConstantInt * > &Cases, SmallVectorImpl< ConstantInt * > &OtherCases, BasicBlock *Dest, BasicBlock *OtherDest)
static void addPredecessorToBlock(BasicBlock *Succ, BasicBlock *NewPred, BasicBlock *ExistPred, MemorySSAUpdater *MSSAU=nullptr)
Update PHI nodes in Succ to indicate that there will now be entries in it from the 'NewPred' block.
static bool validLookupTableConstant(Constant *C, const TargetTransformInfo &TTI)
Return true if the backend will be able to handle initializing an array of constants like C.
static StoreInst * findUniqueStoreInBlocks(BasicBlock *BB1, BasicBlock *BB2)
static bool isSwitchDense(uint64_t NumCases, uint64_t CaseRange, bool OptSize)
static bool validateAndCostRequiredSelects(BasicBlock *BB, BasicBlock *ThenBB, BasicBlock *EndBB, unsigned &SpeculatedInstructions, InstructionCost &Cost, const TargetTransformInfo &TTI)
Estimate the cost of the insertion(s) and check that the PHI nodes can be converted to selects.
static bool simplifySwitchLookup(SwitchInst *SI, IRBuilder<> &Builder, DomTreeUpdater *DTU, const DataLayout &DL, const TargetTransformInfo &TTI, bool ConvertSwitchToLookupTable)
If the switch is only used to initialize one or more phi nodes in a common successor block with diffe...
static void removeSwitchAfterSelectFold(SwitchInst *SI, PHINode *PHI, Value *SelectValue, IRBuilder<> &Builder, DomTreeUpdater *DTU)
static bool valuesOverlap(std::vector< ValueEqualityComparisonCase > &C1, std::vector< ValueEqualityComparisonCase > &C2)
Return true if there are any keys in C1 that exist in C2 as well.
static bool isProfitableToSpeculate(const CondBrInst *BI, std::optional< bool > Invert, const TargetTransformInfo &TTI)
static bool mergeConditionalStoreToAddress(BasicBlock *PTB, BasicBlock *PFB, BasicBlock *QTB, BasicBlock *QFB, BasicBlock *PostBB, Value *Address, bool InvertPCond, bool InvertQCond, DomTreeUpdater *DTU, const DataLayout &DL, const TargetTransformInfo &TTI)
static bool mergeCleanupPad(CleanupReturnInst *RI)
static bool isVectorOp(Instruction &I)
Return if an instruction's type or any of its operands' types are a vector type.
static BasicBlock * allPredecessorsComeFromSameSource(BasicBlock *BB)
static void cloneInstructionsIntoPredecessorBlockAndUpdateSSAUses(BasicBlock *BB, BasicBlock *PredBlock, ValueToValueMapTy &VMap)
static int constantIntSortPredicate(ConstantInt *const *P1, ConstantInt *const *P2)
static bool passingValueIsAlwaysUndefined(Value *V, Instruction *I, bool PtrValueMayBeModified=false)
Check if passing a value to an instruction will cause undefined behavior.
static std::optional< std::tuple< BasicBlock *, Instruction::BinaryOps, bool > > shouldFoldCondBranchesToCommonDestination(CondBrInst *BI, CondBrInst *PBI, const TargetTransformInfo *TTI)
Determine if the two branches share a common destination and deduce a glue that joins the branches' c...
static bool isSafeToHoistInstr(Instruction *I, unsigned Flags)
static std::optional< bool > foldCondBranchOnValueKnownInPredecessorImpl(CondBrInst *BI, const TargetTransformInfo &TTI, DomTreeUpdater *DTU, AssumptionCache *AC, const DataLayout &DL)
If we have a conditional branch on something for which we know the constant value in predecessors (e....
static bool isSafeToHoistInvoke(BasicBlock *BB1, BasicBlock *BB2, Instruction *I1, Instruction *I2)
static ConstantInt * getConstantInt(Value *V, const DataLayout &DL)
Extract ConstantInt from value, looking through IntToPtr and PointerNullValue.
static bool simplifySwitchOfCmpIntrinsic(SwitchInst *SI, IRBuilderBase &Builder, DomTreeUpdater *DTU)
Fold switch over ucmp/scmp intrinsic to br if two of the switch arms have the same destination.
static bool shouldBuildLookupTable(SwitchInst *SI, uint64_t TableSize, const TargetTransformInfo &TTI, const DataLayout &DL, const SmallVector< Type * > &ResultTypes)
Determine whether a lookup table should be built for this switch, based on the number of cases,...
static Constant * constantFold(Instruction *I, const DataLayout &DL, const SmallDenseMap< Value *, Constant * > &ConstantPool)
Try to fold instruction I into a constant.
static bool getCaseResults(SwitchInst *SI, ConstantInt *CaseVal, BasicBlock *CaseDest, BasicBlock **CommonDest, SmallVectorImpl< std::pair< PHINode *, Constant * > > &Res, const DataLayout &DL)
Try to determine the resulting constant values in phi nodes at the common destination basic block,...
static bool areIdenticalUpToCommutativity(const Instruction *I1, const Instruction *I2)
static bool forwardSwitchConditionToPHI(SwitchInst *SI)
Try to forward the condition of a switch instruction to a phi node dominated by the switch,...
static PHINode * findPHIForConditionForwarding(ConstantInt *CaseValue, BasicBlock *BB, int *PhiIndex)
If BB would be eligible for simplification by TryToSimplifyUncondBranchFromEmptyBlock (i....
static bool reachesUncontrolledConvergentCallBeforeBlock(BasicBlock *From, BasicBlock *StopBB)
static bool simplifySwitchOfPowersOfTwo(SwitchInst *SI, IRBuilder<> &Builder, DomTreeUpdater *DTU, const DataLayout &DL, const TargetTransformInfo &TTI)
Tries to transform switch of powers of two to reduce switch range.
static bool isCleanupBlockEmpty(iterator_range< BasicBlock::iterator > R)
static Value * ensureValueAvailableInSuccessor(Value *V, BasicBlock *BB, Value *AlternativeV=nullptr)
static Value * createLogicalOp(IRBuilderBase &Builder, Instruction::BinaryOps Opc, Value *LHS, Value *RHS, const Twine &Name="")
static void hoistConditionalLoadsStores(CondBrInst *BI, SmallVectorImpl< Instruction * > &SpeculatedConditionalLoadsStores, std::optional< bool > Invert, Instruction *Sel)
If the target supports conditional faulting, we look for the following pattern:
static bool shouldHoistCommonInstructions(Instruction *I1, Instruction *I2, const TargetTransformInfo &TTI)
Helper function for hoistCommonCodeFromSuccessors.
static bool reduceSwitchRange(SwitchInst *SI, IRBuilder<> &Builder, const DataLayout &DL, const TargetTransformInfo &TTI)
Try to transform a switch that has "holes" in it to a contiguous sequence of cases.
static bool safeToMergeTerminators(Instruction *SI1, Instruction *SI2, SmallSetVector< BasicBlock *, 4 > *FailBlocks=nullptr)
Return true if it is safe to merge these two terminator instructions together.
SkipFlags
@ SkipReadMem
@ SkipSideEffect
@ SkipImplicitControlFlow
static bool simplifySwitchDefaultBranch(SwitchInst *SI, DomTreeUpdater *DTU, const DataLayout &DL, AssumptionCache *AC)
static bool incomingValuesAreCompatible(BasicBlock *BB, ArrayRef< BasicBlock * > IncomingBlocks, SmallPtrSetImpl< Value * > *EquivalenceSet=nullptr)
Return true if all the PHI nodes in the basic block BB receive compatible (identical) incoming values...
static void createUnreachableSwitchDefault(SwitchInst *Switch, DomTreeUpdater *DTU, bool RemoveOrigDefaultBlock=true)
static Value * foldSwitchToSelect(const SwitchCaseResultVectorTy &ResultVector, Constant *DefaultResult, Value *Condition, IRBuilder<> &Builder, const DataLayout &DL, ArrayRef< uint32_t > BranchWeights)
static bool sinkCommonCodeFromPredecessors(BasicBlock *BB, DomTreeUpdater *DTU)
Check whether BB's predecessors end with unconditional branches.
static bool isTypeLegalForLookupTable(Type *Ty, const TargetTransformInfo &TTI, const DataLayout &DL)
static bool eliminateDeadSwitchCases(SwitchInst *SI, DomTreeUpdater *DTU, AssumptionCache *AC, const DataLayout &DL)
Compute masked bits for the condition of a switch and use it to remove dead cases.
static bool blockIsSimpleEnoughToThreadThrough(BasicBlock *BB, BlocksSet &NonLocalUseBlocks)
Return true if we can thread a branch across this block.
static Value * isSafeToSpeculateStore(Instruction *I, BasicBlock *BrBB, BasicBlock *StoreBB, BasicBlock *EndBB)
Determine if we can hoist sink a sole store instruction out of a conditional block.
static bool foldTwoEntryPHINode(PHINode *PN, const TargetTransformInfo &TTI, DomTreeUpdater *DTU, AssumptionCache *AC, const DataLayout &DL, bool SpeculateUnpredictables)
Given a BB that starts with the specified two-entry PHI node, see if we can eliminate it.
static bool findReaching(BasicBlock *BB, BasicBlock *DefBB, BlocksSet &ReachesNonLocalUses)
static bool extractPredSuccWeights(CondBrInst *PBI, CondBrInst *BI, uint64_t &PredTrueWeight, uint64_t &PredFalseWeight, uint64_t &SuccTrueWeight, uint64_t &SuccFalseWeight)
Return true if either PBI or BI has branch weight available, and store the weights in {Pred|Succ}...
static bool shouldUseSwitchConditionAsTableIndex(ConstantInt &MinCaseVal, const ConstantInt &MaxCaseVal, bool HasDefaultResults, const SmallVector< Type * > &ResultTypes, const DataLayout &DL, const TargetTransformInfo &TTI)
static InstructionCost computeSpeculationCost(const User *I, const TargetTransformInfo &TTI)
Compute an abstract "cost" of speculating the given instruction, which is assumed to be safe to specu...
static bool performBranchToCommonDestFolding(CondBrInst *BI, CondBrInst *PBI, DomTreeUpdater *DTU, MemorySSAUpdater *MSSAU, const TargetTransformInfo *TTI)
static std::optional< unsigned > getDenseSwitchRangeReductionShift(ArrayRef< int64_t > Values, int64_t Base, bool OptSize)
SmallPtrSet< BasicBlock *, 8 > BlocksSet
static unsigned skippedInstrFlags(Instruction *I)
static bool mergeCompatibleInvokes(BasicBlock *BB, DomTreeUpdater *DTU)
If this block is a landingpad exception handling block, categorize all the predecessor invokes into s...
static bool replacingOperandWithVariableIsCheap(const Instruction *I, int OpIdx)
static void eraseTerminatorAndDCECond(Instruction *TI, MemorySSAUpdater *MSSAU=nullptr)
static void eliminateBlockCases(BasicBlock *BB, std::vector< ValueEqualityComparisonCase > &Cases)
Given a vector of bb/value pairs, remove any entries in the list that match the specified block.
static bool mergeConditionalStores(CondBrInst *PBI, CondBrInst *QBI, DomTreeUpdater *DTU, const DataLayout &DL, const TargetTransformInfo &TTI)
static bool mergeNestedCondBranch(CondBrInst *BI, DomTreeUpdater *DTU)
Fold the following pattern: bb0: br i1 cond1, label bb1, label bb2 bb1: br i1 cond2,...
static void sinkLastInstruction(ArrayRef< BasicBlock * > Blocks)
static size_t mapCaseToResult(ConstantInt *CaseVal, SwitchCaseResultVectorTy &UniqueResults, Constant *Result)
static bool tryWidenCondBranchToCondBranch(CondBrInst *PBI, CondBrInst *BI, DomTreeUpdater *DTU)
If the previous block ended with a widenable branch, determine if reusing the target block is profita...
static void mergeCompatibleInvokesImpl(ArrayRef< InvokeInst * > Invokes, DomTreeUpdater *DTU)
static bool mergeIdenticalBBs(ArrayRef< BasicBlock * > Candidates, DomTreeUpdater *DTU)
static void getBranchWeights(Instruction *TI, SmallVectorImpl< uint64_t > &Weights)
Get Weights of a given terminator, the default weight is at the front of the vector.
static bool tryToMergeLandingPad(LandingPadInst *LPad, UncondBrInst *BI, BasicBlock *BB, DomTreeUpdater *DTU)
Given an block with only a single landing pad and a unconditional branch try to find another basic bl...
static bool initializeUniqueCases(SwitchInst *SI, PHINode *&PHI, BasicBlock *&CommonDest, SwitchCaseResultVectorTy &UniqueResults, Constant *&DefaultResult, const DataLayout &DL, uintptr_t MaxUniqueResults)
static Constant * lookupConstant(Value *V, const SmallDenseMap< Value *, Constant * > &ConstantPool)
If V is a Constant, return it.
static bool SimplifyCondBranchToCondBranch(CondBrInst *PBI, CondBrInst *BI, DomTreeUpdater *DTU, const DataLayout &DL, const TargetTransformInfo &TTI)
If we have a conditional branch as a predecessor of another block, this function tries to simplify it...
static bool canSinkInstructions(ArrayRef< Instruction * > Insts, DenseMap< const Use *, SmallVector< Value *, 4 > > &PHIOperands)
static void hoistLockstepIdenticalDbgVariableRecords(Instruction *TI, Instruction *I1, SmallVectorImpl< Instruction * > &OtherInsts)
Hoists DbgVariableRecords from I1 and OtherInstrs that are identical in lock-step to TI.
static bool removeEmptyCleanup(CleanupReturnInst *RI, DomTreeUpdater *DTU)
static bool trySwitchToSelect(SwitchInst *SI, IRBuilder<> &Builder, DomTreeUpdater *DTU, const DataLayout &DL)
If a switch is only used to initialize one or more phi nodes in a common successor block with only tw...
static bool removeUndefIntroducingPredecessor(BasicBlock *BB, DomTreeUpdater *DTU, AssumptionCache *AC)
If BB has an incoming value that will always trigger undefined behavior (eg.
static bool isUncontrolledConvergentCall(CallBase *CB)
static bool simplifySwitchWhenUMin(SwitchInst *SI, DomTreeUpdater *DTU)
Tries to transform the switch when the condition is umin with a constant.
static bool isSafeCheapLoadStore(const Instruction *I, const TargetTransformInfo &TTI)
static ConstantInt * getKnownValueOnEdge(Value *V, BasicBlock *From, BasicBlock *To)
static bool dominatesMergePoint(Value *V, BasicBlock *BB, Instruction *InsertPt, SmallPtrSetImpl< Instruction * > &AggressiveInsts, InstructionCost &Cost, InstructionCost Budget, const TargetTransformInfo &TTI, AssumptionCache *AC, SmallPtrSetImpl< Instruction * > &ZeroCostInstructions, unsigned Depth=0)
If we have a merge point of an "if condition" as accepted above, return true if the specified value d...
static void reuseTableCompare(User *PhiUser, BasicBlock *PhiBlock, CondBrInst *RangeCheckBranch, Constant *DefaultValue, const SmallVectorImpl< std::pair< ConstantInt *, Constant * > > &Values)
Try to reuse the switch table index compare.
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
This pass exposes codegen information to IR-level passes.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
Value * RHS
Value * LHS
static const uint32_t IV[8]
Definition blake3_impl.h:83
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
unsigned popcount() const
Count the number of bits set.
Definition APInt.h:1690
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1205
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
bool sle(const APInt &RHS) const
Signed less or equal comparison.
Definition APInt.h:1170
unsigned getSignificantBits() const
Get the minimum bit size for this signed APInt.
Definition APInt.h:1551
bool isStrictlyPositive() const
Determine if this APInt Value is positive.
Definition APInt.h:352
uint64_t getLimitedValue(uint64_t Limit=UINT64_MAX) const
If this value is smaller than the specified limit, return it, otherwise return the limit value.
Definition APInt.h:471
LLVM_ABI APInt smul_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1998
bool slt(const APInt &RHS) const
Signed less than comparison.
Definition APInt.h:1134
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
std::optional< int64_t > trySExtValue() const
Get sign extended value if possible.
Definition APInt.h:1594
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1979
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & front() const
Get the first element.
Definition ArrayRef.h:144
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
A cache of @llvm.assume calls within a function.
LLVM_ABI void registerAssumption(AssumeInst *CI)
Add an @llvm.assume intrinsic to this function's cache.
LLVM_ABI bool getValueAsBool() const
Return the attribute's value as a boolean.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator end()
Definition BasicBlock.h:459
iterator begin()
Instruction iterator methods.
Definition BasicBlock.h:446
iterator_range< const_phi_iterator > phis() const
Returns a range that iterates over the phis in the basic block.
Definition BasicBlock.h:515
LLVM_ABI const_iterator getFirstInsertionPt() const
Returns an iterator to the first instruction in this block that is suitable for inserting a non-PHI i...
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
bool hasAddressTaken() const
Returns true if there are any uses of this basic block other than direct branches,...
Definition BasicBlock.h:672
LLVM_ABI InstListType::const_iterator getFirstNonPHIIt() const
Returns an iterator to the first instruction in this block that is not a PHINode instruction.
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
Definition BasicBlock.h:206
LLVM_ABI InstListType::const_iterator getFirstNonPHIOrDbg(bool SkipPseudoOp=true) const
Returns a pointer to the first instruction in this block that is not a PHINode or a debug intrinsic,...
LLVM_ABI bool hasNPredecessors(unsigned N) const
Return true if this block has exactly N predecessors.
LLVM_ABI const BasicBlock * getUniqueSuccessor() const
Return the successor of this block if it has a unique successor.
LLVM_ABI const BasicBlock * getSinglePredecessor() const
Return the predecessor of this block if it has a single predecessor block.
const Instruction & front() const
Definition BasicBlock.h:469
LLVM_ABI const CallInst * getTerminatingDeoptimizeCall() const
Returns the call instruction calling @llvm.experimental.deoptimize prior to the terminating return in...
LLVM_ABI const BasicBlock * getUniquePredecessor() const
Return the predecessor of this block if it has a unique predecessor block.
LLVM_ABI const BasicBlock * getSingleSuccessor() const
Return the successor of this block if it has a single successor.
LLVM_ABI void flushTerminatorDbgRecords()
Eject any debug-info trailing at the end of a block.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this basic block belongs to.
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
LLVM_ABI LLVMContext & getContext() const
Get the context in which this basic block lives.
size_t size() const
Definition BasicBlock.h:467
LLVM_ABI bool isLandingPad() const
Return true if this basic block is a landing pad.
LLVM_ABI bool hasNPredecessorsOrMore(unsigned N) const
Return true if this block has N predecessors or more.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Definition BasicBlock.h:237
void splice(BasicBlock::iterator ToIt, BasicBlock *FromBB)
Transfer all instructions from FromBB to this basic block at ToIt.
Definition BasicBlock.h:644
LLVM_ABI const Module * getModule() const
Return the module owning the function this basic block belongs to, or nullptr if the function does no...
LLVM_ABI void removePredecessor(BasicBlock *Pred, bool KeepOneInputPHIs=false)
Update PHI nodes in this BasicBlock before removal of predecessor Pred.
BasicBlock * getBasicBlock() const
Definition Constants.h:1125
static LLVM_ABI BranchProbability getBranchProbability(uint64_t Numerator, uint64_t Denominator)
BranchProbability getCompl() const
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
void addRangeRetAttr(const ConstantRange &CR)
adds the range attribute to the list of attributes.
bool isCallee(Value::const_user_iterator UI) const
Determine whether the passed iterator points to the callee operand's Use.
bool isConvergent() const
Determine if the invoke is convergent.
Value * getConvergenceControlToken() const
Return the convergence control token for this call, if it exists.
bool isDataOperand(const Use *U) const
bool tryIntersectAttributes(const CallBase *Other)
Try to intersect the attributes from 'this' CallBase and the 'Other' CallBase.
This class represents a function call, abstracting a target machine's calling convention.
mapped_iterator< op_iterator, DerefFnTy > handler_iterator
CleanupPadInst * getCleanupPad() const
Convenience accessor.
BasicBlock * getUnwindDest() const
This class is the base class for the comparison instructions.
Definition InstrTypes.h:728
bool isEquality() const
Determine if this is an equals/not equals predicate.
Definition InstrTypes.h:978
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
Predicate getPredicate() const
Return the predicate for this instruction.
Definition InstrTypes.h:828
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
Conditional Branch instruction.
static CondBrInst * Create(Value *Cond, BasicBlock *IfTrue, BasicBlock *IfFalse, InsertPosition InsertBefore=nullptr)
void setSuccessor(unsigned idx, BasicBlock *NewSucc)
void setCondition(Value *V)
Value * getCondition() const
BasicBlock * getSuccessor(unsigned i) const
static LLVM_ABI Constant * get(ArrayType *T, ArrayRef< Constant * > V)
A vector constant whose element type is a simple 1/2/4/8-byte integer or float/double,...
Definition Constants.h:951
A constant value that is initialized with an expression using other constant values.
Definition Constants.h:1316
static LLVM_ABI Constant * getNeg(Constant *C, bool HasNSW=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
ConstantFolder - Create constants with minimum, target independent, folding.
This is the shared class of boolean and integer constants.
Definition Constants.h:87
bool isOne() const
This is just a convenience method to make client code smaller for a common case.
Definition Constants.h:225
bool isNegative() const
Definition Constants.h:214
uint64_t getLimitedValue(uint64_t Limit=~0ULL) const
getLimitedValue - If the value is smaller than the specified limit, return it, otherwise return the l...
Definition Constants.h:269
IntegerType * getIntegerType() const
Variant of the getType() method to always return an IntegerType, which reduces the amount of casting ...
Definition Constants.h:198
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
Definition Constants.h:135
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
Definition Constants.h:219
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
unsigned getBitWidth() const
getBitWidth - Return the scalar bitwidth of this constant.
Definition Constants.h:162
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
A constant pointer value that points to null.
Definition Constants.h:716
This class represents a range of values.
LLVM_ABI bool getEquivalentICmp(CmpInst::Predicate &Pred, APInt &RHS) const
Set up Pred and RHS such that ConstantRange::makeExactICmpRegion(Pred, RHS) == *this.
LLVM_ABI ConstantRange subtract(const APInt &CI) const
Subtract the specified constant from the endpoints of this constant range.
const APInt & getLower() const
Return the lower value for this range.
LLVM_ABI APInt getUnsignedMin() const
Return the smallest unsigned value contained in the ConstantRange.
LLVM_ABI bool isEmptySet() const
Return true if this set contains no members.
LLVM_ABI bool isSizeLargerThan(uint64_t MaxSize) const
Compare set size of this range with Value.
const APInt & getUpper() const
Return the upper value for this range.
LLVM_ABI bool isUpperWrapped() const
Return true if the exclusive upper bound wraps around the unsigned domain.
static LLVM_ABI ConstantRange makeExactICmpRegion(CmpInst::Predicate Pred, const APInt &Other)
Produce the exact range such that all values in the returned range satisfy the given predicate with a...
LLVM_ABI ConstantRange inverse() const
Return a new range that is the logical not of the current set.
LLVM_ABI APInt getUnsignedMax() const
Return the largest unsigned value contained in the ConstantRange.
static ConstantRange getNonEmpty(APInt Lower, APInt Upper)
Create non-empty constant range with the given bounds.
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getIntegerValue(Type *Ty, const APInt &V)
Return the value for an integer or pointer constant, or a vector thereof, with the given scalar value...
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
Definition Constant.h:64
LLVM_ABI bool isOneValue() const
Returns true if the value is one.
Definition Constants.cpp:89
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Base class for non-instruction debug metadata records that have positions within IR.
LLVM_ABI void removeFromParent()
simple_ilist< DbgRecord >::iterator self_iterator
Record of a variable value-assignment, aka a non instruction representation of the dbg....
A debug info location.
Definition DebugLoc.h:126
bool isSameSourceLocation(const DebugLoc &Other) const
Return true if the source locations match, ignoring isImplicitCode and source atom info.
Definition DebugLoc.h:244
static DebugLoc getTemporary()
Definition DebugLoc.h:152
static LLVM_ABI DebugLoc getMergedLocation(DebugLoc LocA, DebugLoc LocB)
When two instructions are combined into a single instruction we also need to combine the original loc...
Definition DebugLoc.cpp:173
static LLVM_ABI DebugLoc getMergedLocations(ArrayRef< DebugLoc > Locs)
Try to combine the vector of locations passed as input in a single one.
Definition DebugLoc.cpp:160
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:782
ValueT & at(const_arg_type_t< KeyT > Val)
Return the entry for the specified key, or abort if no such entry exists.
Definition DenseMap.h:827
void reserve(size_type NumEntries)
Grow the densemap so that it can contain at least NumEntries items before resizing again.
Definition DenseMap.h:737
iterator end()
Definition DenseMap.h:702
unsigned size() const
Definition DenseMap.h:733
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition DenseMap.h:843
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:872
Implements a dense probed hash-table based set.
Definition DenseSet.h:281
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
const BasicBlock & getEntryBlock() const
Definition Function.h:794
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
Definition Function.cpp:765
bool hasMinSize() const
Optimize this function for minimum size (-Oz).
Definition Function.h:696
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:730
void applyUpdates(ArrayRef< UpdateT > Updates)
Submit updates to all available trees.
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Module * getParent()
Get the module that this global value is contained inside of...
This instruction compares its operands according to the predicate given to the constructor.
Predicate getSignedPredicate() const
For example, EQ->EQ, SLE->SLE, UGT->SGT, etc.
bool isEquality() const
Return true if this predicate is either EQ or NE.
static bool isEquality(Predicate P)
Return true if this predicate is either EQ or NE.
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
Value * CreateICmpULT(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2390
Value * CreateZExtOrTrunc(Value *V, Type *DestTy, const Twine &Name="")
Create a ZExt or Trunc from the integer value V to DestTy.
Definition IRBuilder.h:2131
CondBrInst * CreateCondBr(Value *Cond, BasicBlock *True, BasicBlock *False, MDNode *BranchWeights=nullptr, MDNode *Unpredictable=nullptr)
Create a conditional 'br Cond, TrueDest, FalseDest' instruction.
Definition IRBuilder.h:1203
LLVM_ABI Value * CreateSelectFMF(Value *C, Value *True, Value *False, FMFSource FMFSource, const Twine &Name="", Instruction *MDFrom=nullptr)
ConstantInt * getTrue()
Get the constant value for i1 true.
Definition IRBuilder.h:436
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
BasicBlock::iterator GetInsertPoint() const
Definition IRBuilder.h:176
Value * CreateFreeze(Value *V, const Twine &Name="")
Definition IRBuilder.h:2727
void SetCurrentDebugLocation(const DebugLoc &L)
Set location information used by debugging information.
Definition IRBuilder.h:220
Value * CreateLShr(Value *LHS, Value *RHS, const Twine &Name="", bool isExact=false)
Definition IRBuilder.h:1519
LLVM_ABI CallInst * CreateAssumption(Value *Cond)
Create an assume intrinsic call that allows the optimizer to assume that the provided condition will ...
Value * CreateInBoundsGEP(Type *Ty, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &Name="")
Definition IRBuilder.h:2011
UncondBrInst * CreateBr(BasicBlock *Dest)
Create an unconditional 'br label X' instruction.
Definition IRBuilder.h:1197
Value * CreateNot(Value *V, const Twine &Name="")
Definition IRBuilder.h:1841
SwitchInst * CreateSwitch(Value *V, BasicBlock *Dest, unsigned NumCases=10, MDNode *BranchWeights=nullptr, MDNode *Unpredictable=nullptr)
Create a switch instruction with the specified value, default dest, and with a hint for the number of...
Definition IRBuilder.h:1226
Value * CreateICmpEQ(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2374
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Definition IRBuilder.h:1898
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2113
StoreInst * CreateStore(Value *Val, Value *Ptr, bool isVolatile=false)
Definition IRBuilder.h:1917
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1409
Value * CreatePtrToInt(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2225
ConstantInt * getFalse()
Get the constant value for i1 false.
Definition IRBuilder.h:441
Value * CreateTrunc(Value *V, Type *DestTy, const Twine &Name="", bool IsNUW=false, bool IsNSW=false)
Definition IRBuilder.h:2099
Value * CreateIntCast(Value *V, Type *DestTy, bool isSigned, const Twine &Name="")
Definition IRBuilder.h:2315
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Definition IRBuilder.h:181
Value * CreateICmp(CmpInst::Predicate P, Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2484
Value * CreateOr(Value *LHS, Value *RHS, const Twine &Name="", bool IsDisjoint=false)
Definition IRBuilder.h:1579
Value * CreateMul(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1443
Provides an 'InsertHelper' that calls a user-provided callback after performing the default insertion...
Definition IRBuilder.h:75
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2901
Indirect Branch Instruction.
BasicBlock * getDestination(unsigned i)
Return the specified destination.
unsigned getNumDestinations() const
return the number of possible destinations in this indirectbr instruction.
LLVM_ABI void removeDestination(unsigned i)
This method removes the specified successor from the indirectbr instruction.
LLVM_ABI void dropUBImplyingAttrsAndMetadata(ArrayRef< unsigned > Keep={})
Drop any attributes or metadata that can cause immediate undefined behavior.
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI iterator_range< simple_ilist< DbgRecord >::iterator > cloneDebugInfoFrom(const Instruction *From, std::optional< simple_ilist< DbgRecord >::iterator > FromHere=std::nullopt, bool InsertAtHead=false)
Clone any debug-info attached to From onto this instruction.
LLVM_ABI unsigned getNumSuccessors() const LLVM_READONLY
Return the number of successors that this instruction has.
iterator_range< simple_ilist< DbgRecord >::iterator > getDbgRecordRange() const
Return a range over the DbgRecords attached to this instruction.
LLVM_ABI void dropLocation()
Drop the instruction's debug location.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void andIRFlags(const Value *V)
Logical 'and' of any supported wrapping, exact, and fast-math flags of V and this instruction.
bool hasMetadata() const
Return true if this instruction has any metadata attached to it.
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI bool isAtomic() const LLVM_READONLY
Return true if this instruction has an AtomicOrdering of unordered or higher.
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Instruction * user_back()
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
LLVM_ABI BasicBlock * getSuccessor(unsigned Idx) const LLVM_READONLY
Return the specified successor. This instruction must be a terminator.
LLVM_ABI bool mayHaveSideEffects() const LLVM_READONLY
Return true if the instruction may have side effects.
bool isTerminator() const
iterator_range< user_iterator > users()
LLVM_ABI bool isUsedOutsideOfBlock(const BasicBlock *BB) const LLVM_READONLY
Return true if there are any uses of this instruction in blocks other than the specified block.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
@ CompareUsingIntersectedAttrs
Check for equivalence with intersected callbase attrs.
LLVM_ABI bool isIdenticalTo(const Instruction *I) const LLVM_READONLY
Return true if the specified instruction is exactly identical to the current one.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
LLVM_ABI void applyMergedLocation(DebugLoc LocA, DebugLoc LocB)
Merge 2 debug locations and apply it to the Instruction.
LLVM_ABI void dropDbgRecords()
Erase any DbgRecords attached to this instruction.
LLVM_ABI InstListType::iterator insertInto(BasicBlock *ParentBB, InstListType::iterator It)
Inserts an unlinked instruction into ParentBB at position It and returns the iterator of the inserted...
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
unsigned getBitWidth() const
Get the number of bits in this IntegerType.
Invoke instruction.
void setNormalDest(BasicBlock *B)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
The landingpad instruction holds all of the information necessary to generate correct exception handl...
An instruction for reading from memory.
static unsigned getPointerOperandIndex()
Iterates through instructions in a set of blocks in reverse order from the first non-terminator.
LLVM_ABI MDNode * createBranchWeights(uint32_t TrueWeight, uint32_t FalseWeight, bool IsExpected=false)
Return metadata containing two branch weights.
Definition MDBuilder.cpp:38
Metadata node.
Definition Metadata.h:1081
Helper class to manipulate !mmra metadata nodes.
bool empty() const
Definition MapVector.h:79
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition MapVector.h:126
size_type size() const
Definition MapVector.h:58
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
iterator_range< const_block_iterator > blocks() const
op_range incoming_values()
void setIncomingValue(unsigned i, Value *V)
Value * getIncomingValueForBlock(const BasicBlock *BB) const
BasicBlock * getIncomingBlock(unsigned i) const
Return incoming basic block number i.
Value * getIncomingValue(unsigned i) const
Return incoming value number x.
int getBasicBlockIndex(const BasicBlock *BB) const
Return the first index of the specified basic block in the value list for this PHI.
unsigned getNumIncomingValues() const
Return the number of incoming edges.
static PHINode * Create(Type *Ty, unsigned NumReservedValues, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedValues is a hint for the number of incoming edges that this phi node will h...
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Value * getValue() const
Convenience accessor.
Return a value (possibly void), from a function.
This class represents the LLVM 'select' instruction.
size_type size() const
Determine the number of elements in the SetVector.
Definition SetVector.h:103
void insert_range(Range &&R)
Definition SetVector.h:182
bool empty() const
Determine if the SetVector is empty or not.
Definition SetVector.h:100
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
size_type size() const
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
bool erase(PtrType Ptr)
Remove pointer from the set.
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
void insert_range(Range &&R)
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
Definition SmallSet.h:134
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void assign(size_type NumElts, ValueParamT Elt)
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void resize(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this store instruction.
Align getAlign() const
bool isSimple() const
Value * getValueOperand()
bool isUnordered() const
static unsigned getPointerOperandIndex()
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this store instruction.
Value * getPointerOperand()
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
A wrapper class to simplify modification of SwitchInst cases along with their prof branch_weights met...
LLVM_ABI void setSuccessorWeight(unsigned idx, CaseWeightOpt W)
LLVM_ABI void addCase(ConstantInt *OnVal, BasicBlock *Dest, CaseWeightOpt W)
Delegate the call to the underlying SwitchInst::addCase() and set the specified branch weight for the...
LLVM_ABI CaseWeightOpt getSuccessorWeight(unsigned idx)
LLVM_ABI void replaceDefaultDest(SwitchInst::CaseIt I)
Replace the default destination by given case.
std::optional< uint32_t > CaseWeightOpt
LLVM_ABI SwitchInst::CaseIt removeCase(SwitchInst::CaseIt I)
Delegate the call to the underlying SwitchInst::removeCase() and remove correspondent branch weight.
Multiway switch.
CaseIt case_end()
Returns a read/write iterator that points one past the last in the SwitchInst.
BasicBlock * getSuccessor(unsigned idx) const
void setCondition(Value *V)
LLVM_ABI void addCase(ConstantInt *OnVal, BasicBlock *Dest)
Add an entry to the switch instruction.
CaseIteratorImpl< CaseHandle > CaseIt
void setSuccessor(unsigned idx, BasicBlock *NewSucc)
unsigned getNumSuccessors() const
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
TargetCostKind
The kind of cost model.
@ TCK_CodeSize
Instruction code size.
@ TCK_SizeAndLatency
The weighted sum of size and latency.
@ TCC_Free
Expected to fold away in lowering.
@ TCC_Basic
The cost of a typical 'add' instruction.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
Unconditional Branch instruction.
void setSuccessor(BasicBlock *NewSucc)
static UncondBrInst * Create(BasicBlock *Target, InsertPosition InsertBefore=nullptr)
BasicBlock * getSuccessor(unsigned i=0) const
'undef' values are things that do not have specified contents.
Definition Constants.h:1657
This function has undefined behavior.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
Definition Use.cpp:35
LLVM_ABI void set(Value *Val)
Definition Value.h:876
User * getUser() const
Returns the User that contains this Use.
Definition Use.h:61
op_range operands()
Definition User.h:267
const Use & getOperandUse(unsigned i) const
Definition User.h:220
void setOperand(unsigned i, Value *Val)
Definition User.h:212
LLVM_ABI bool replaceUsesOfWith(Value *From, Value *To)
Replace uses of one Value with another.
Definition User.cpp:25
Value * getOperand(unsigned i) const
Definition User.h:207
unsigned getNumOperands() const
Definition User.h:229
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
static constexpr uint64_t MaximumAlignment
Definition Value.h:801
LLVM_ABI Value(Type *Ty, unsigned scid)
Definition Value.cpp:54
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
Definition Value.cpp:394
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
bool use_empty() const
Definition Value.h:348
iterator_range< use_iterator > uses()
Definition Value.h:382
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Represents an op.with.overflow intrinsic.
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
Definition ilist_node.h:348
A range adaptor for a pair of iterators.
Changed
#define UINT64_MAX
Definition DataTypes.h:77
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
Predicate
Predicate - These are "(BI << 5) | BO" for various predicates.
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
auto m_UMin(const Opnd0 &Op0, const Opnd1 &Op1)
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
ExtractValue_match< Ind, Val_t > m_ExtractValue(const Val_t &V)
Match a single index ExtractValue instruction.
auto m_Value()
Match an arbitrary value and ignore it.
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
ThreeOps_match< decltype(m_Value()), LHS, RHS, Instruction::Select, true > m_c_Select(const LHS &L, const RHS &R)
Match Select(C, LHS, RHS) or Select(C, RHS, LHS)
match_bind< WithOverflowInst > m_WithOverflowInst(WithOverflowInst *&I)
Match a with overflow intrinsic, capturing it if we match.
match_immconstant_ty m_ImmConstant()
Match an arbitrary immediate Constant and ignore it.
NoWrapTrunc_match< OpTy, TruncInst::NoUnsignedWrap > m_NUWTrunc(const OpTy &Op)
Matches trunc nuw.
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
BinaryOp_match< LHS, RHS, Instruction::Or > m_Or(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
Definition DebugInfo.h:212
LLVM_ABI void deleteAssignmentMarkers(const Instruction *Inst)
Delete the llvm.dbg.assign intrinsics linked to Inst.
initializer< Ty > init(const Ty &Val)
PointerTypeMap run(const Module &M)
Compute the PointerTypeMap for the module M.
constexpr double e
@ User
could "use" a pointer
NodeAddr< UseNode * > Use
Definition RDFGraph.h:385
NodeAddr< FuncNode * > Func
Definition RDFGraph.h:393
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Offset
Definition DWP.cpp:577
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
Definition STLExtras.h:846
bool operator<(int64_t V1, const APSInt &V2)
Definition APSInt.h:360
constexpr auto not_equal_to(T &&Arg)
Functor variant of std::not_equal_to that can be used as a UnaryPredicate in functional algorithms li...
Definition STLExtras.h:2196
LLVM_ABI bool foldBranchToCommonDest(CondBrInst *BI, llvm::DomTreeUpdater *DTU=nullptr, MemorySSAUpdater *MSSAU=nullptr, const TargetTransformInfo *TTI=nullptr, AssumptionCache *AC=nullptr, unsigned BonusInstThreshold=1)
If this basic block is ONLY a setcc and a branch, and if a predecessor branches to us and one of our ...
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1781
LLVM_ABI cl::opt< bool > ProfcheckDisableMetadataFixes
Definition LoopInfo.cpp:60
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
LLVM_ABI bool RecursivelyDeleteTriviallyDeadInstructions(Value *V, const TargetLibraryInfo *TLI=nullptr, MemorySSAUpdater *MSSAU=nullptr, std::function< void(Value *)> AboutToDeleteCallback=std::function< void(Value *)>())
If the specified value is a trivially dead instruction, delete it.
Definition Local.cpp:522
bool succ_empty(const Instruction *I)
Definition CFG.h:141
LLVM_ABI bool IsBlockFollowedByDeoptOrUnreachable(const BasicBlock *BB)
Check if we can prove that all paths starting from this block converge to a block that either has a @...
LLVM_ABI bool ConstantFoldTerminator(BasicBlock *BB, bool DeleteDeadConditions=false, const TargetLibraryInfo *TLI=nullptr, DomTreeUpdater *DTU=nullptr)
If a terminator instruction is predicated on a constant value, convert it into an unconditional branc...
Definition Local.cpp:133
static cl::opt< unsigned > MaxSwitchCasesPerResult("max-switch-cases-per-result", cl::Hidden, cl::init(16), cl::desc("Limit cases to analyze when converting a switch to select"))
InstructionCost Cost
RelativeUniformCounterPtr Values
Definition InstrProf.h:91
static cl::opt< bool > SpeculateOneExpensiveInst("speculate-one-expensive-inst", cl::Hidden, cl::init(true), cl::desc("Allow exactly one expensive instruction to be speculatively " "executed"))
@ Known
Known to have no common set bits.
@ Dead
Unused definition.
auto pred_end(const MachineBasicBlock *BB)
void set_intersect(S1Ty &S1, const S2Ty &S2)
set_intersect(A, B) - Compute A := A ^ B Identical to set_intersection, except that it works on set<>...
LLVM_ABI void setExplicitlyUnknownBranchWeightsIfProfiled(Instruction &I, StringRef PassName, const Function *F=nullptr)
Like setExplicitlyUnknownBranchWeights(...), but only sets unknown branch weights in the new instruct...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
auto successors(const MachineBasicBlock *BB)
@ Load
The value being inserted comes from a load (InsertElement only).
auto accumulate(R &&Range, E &&Init)
Wrapper for std::accumulate.
Definition STLExtras.h:1718
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ABI MDNode * getBranchWeightMDNode(const Instruction &I)
Get the branch weights metadata node.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
Definition MathExtras.h:244
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
LLVM_ABI void DeleteDeadBlock(BasicBlock *BB, DomTreeUpdater *DTU=nullptr, bool KeepOneInputPHIs=false)
Delete the specified block, which must have no predecessors.
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
auto unique(Range &&R, Predicate P)
Definition STLExtras.h:2150
static cl::opt< unsigned > MaxSpeculationDepth("max-speculation-depth", cl::Hidden, cl::init(10), cl::desc("Limit maximum recursion depth when calculating costs of " "speculatively executed instructions"))
OutputIt copy_if(R &&Range, OutputIt Out, UnaryPredicate P)
Provide wrappers to std::copy_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1807
static cl::opt< unsigned > PHINodeFoldingThreshold("phi-node-folding-threshold", cl::Hidden, cl::init(2), cl::desc("Control the amount of phi node folding to perform (default = 2)"))
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
static cl::opt< bool > MergeCondStoresAggressively("simplifycfg-merge-cond-stores-aggressively", cl::Hidden, cl::init(false), cl::desc("When merging conditional stores, do so even if the resultant " "basic blocks are unlikely to be if-converted as a result"))
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
LLVM_ABI ConstantRange getConstantRangeFromMetadata(const MDNode &RangeMD)
Parse out a conservative ConstantRange from !range metadata.
LLVM_ABI unsigned ComputeMaxSignificantBits(const Value *Op, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Get the upper bound on bit size for this Value Op as a signed integer.
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
Definition STLExtras.h:366
static cl::opt< unsigned > BranchFoldThreshold("simplifycfg-branch-fold-threshold", cl::Hidden, cl::init(2), cl::desc("Maximum cost of combining conditions when " "folding branches"))
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
LLVM_ABI Value * simplifyInstruction(Instruction *I, const SimplifyQuery &Q)
See if we can compute a simplified version of this instruction.
LLVM_ABI void setBranchWeights(Instruction &I, ArrayRef< uint32_t > Weights, bool IsExpected, bool ElideAllZero=false)
Create a new branch_weights metadata node and add or overwrite a prof metadata reference to instructi...
static cl::opt< bool > SinkCommon("simplifycfg-sink-common", cl::Hidden, cl::init(true), cl::desc("Sink common instructions down to the end block"))
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Function *CtxF=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
Definition STLExtras.h:2216
constexpr bool has_single_bit(T Value) noexcept
Definition bit.h:149
static cl::opt< bool > HoistStoresWithCondFaulting("simplifycfg-hoist-stores-with-cond-faulting", cl::Hidden, cl::init(true), cl::desc("Hoist stores if the target supports conditional faulting"))
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
constexpr detail::StaticCastFunc< To > StaticCastTo
Function objects corresponding to the Cast types defined above.
Definition Casting.h:882
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI CondBrInst * GetIfCondition(BasicBlock *BB, BasicBlock *&IfTrue, BasicBlock *&IfFalse)
Check whether BB is the merge point of a if-region.
LLVM_ABI bool TryToSimplifyUncondBranchFromEmptyBlock(BasicBlock *BB, DomTreeUpdater *DTU=nullptr)
BB is known to contain an unconditional branch, and contains no instructions other than PHI nodes,...
Definition Local.cpp:1147
void RemapDbgRecordRange(Module *M, iterator_range< DbgRecordIterator > Range, ValueToValueMapTy &VM, RemapFlags Flags=RF_None, ValueMapTypeRemapper *TypeMapper=nullptr, ValueMaterializer *Materializer=nullptr, const MetadataPredicate *IdentityMD=nullptr)
Remap the Values used in the DbgRecords Range using the value map VM.
LLVM_ABI void InvertBranch(CondBrInst *PBI, IRBuilderBase &Builder)
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI bool impliesPoison(const Value *ValAssumedPoison, const Value *V)
Return true if V is poison given that ValAssumedPoison is already poison.
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
static cl::opt< bool > EnableMergeCompatibleInvokes("simplifycfg-merge-compatible-invokes", cl::Hidden, cl::init(true), cl::desc("Allow SimplifyCFG to merge invokes together when appropriate"))
@ RF_IgnoreMissingLocals
If this flag is set, the remapper ignores missing function-local entries (Argument,...
Definition ValueMapper.h:98
@ RF_NoModuleLevelChanges
If this flag is set, the remapper knows that only local values within a function (such as an instruct...
Definition ValueMapper.h:80
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
IRBuilder(LLVMContext &, FolderTy, InserterTy) -> IRBuilder< FolderTy, InserterTy >
auto make_first_range(ContainerTy &&c)
Given a container of pairs, return a range over the first elements.
Definition STLExtras.h:1415
LLVM_ABI bool collectPossibleValues(const Value *V, SmallPtrSetImpl< const Constant * > &Constants, unsigned MaxCount, bool AllowUndefOrPoison=true)
Enumerates all possible immediate values of V and inserts them into the set Constants.
LLVM_ABI Instruction * removeUnwindEdge(BasicBlock *BB, DomTreeUpdater *DTU=nullptr)
Replace 'BB's terminator with one that does not have an unwind successor block.
Definition Local.cpp:2874
auto succ_size(const MachineBasicBlock *BB)
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
Definition STLExtras.h:552
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
static cl::opt< unsigned > MaxJumpThreadingLiveBlocks("max-jump-threading-live-blocks", cl::Hidden, cl::init(24), cl::desc("Limit number of blocks a define in a threaded block is allowed " "to be live in"))
RNSuccIterator< NodeRef, BlockT, RegionT > succ_begin(NodeRef Node)
LLVM_ABI void combineMetadataForCSE(Instruction *K, const Instruction *J, bool DoesKMove)
Combine the metadata of two instructions so that K can replace J.
Definition Local.cpp:3122
iterator_range(Container &&) -> iterator_range< llvm::detail::IterOfRange< Container > >
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
static cl::opt< int > MaxSmallBlockSize("simplifycfg-max-small-block-size", cl::Hidden, cl::init(10), cl::desc("Max size of a block which is still considered " "small enough to thread through"))
LLVM_ABI BasicBlock * SplitBlockPredecessors(BasicBlock *BB, ArrayRef< BasicBlock * > Preds, const char *Suffix, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, bool PreserveLCSSA=false)
This method introduces at least one new basic block into the function and moves some of the predecess...
LLVM_ABI bool isWidenableBranch(const User *U)
Returns true iff U is a widenable branch (that is, extractWidenableCondition returns widenable condit...
@ Other
Any other memory.
Definition ModRef.h:68
TargetTransformInfo TTI
static cl::opt< unsigned > HoistCommonSkipLimit("simplifycfg-hoist-common-skip-limit", cl::Hidden, cl::init(20), cl::desc("Allow reordering across at most this many " "instructions when hoisting"))
LLVM_ABI cl::opt< bool > RequireAndPreserveDomTree
This function is used to do simplification of a CFG.
static cl::opt< bool > MergeCondStores("simplifycfg-merge-cond-stores", cl::Hidden, cl::init(true), cl::desc("Hoist conditional stores even if an unconditional store does not " "precede - hoist multiple conditional stores into a single " "predicated store"))
static cl::opt< unsigned > BranchFoldToCommonDestVectorMultiplier("simplifycfg-branch-fold-common-dest-vector-multiplier", cl::Hidden, cl::init(2), cl::desc("Multiplier to apply to threshold when determining whether or not " "to fold branch to common destination when vector operations are " "present"))
RNSuccIterator< NodeRef, BlockT, RegionT > succ_end(NodeRef Node)
LLVM_ABI bool MergeBlockIntoPredecessor(BasicBlock *BB, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, MemoryDependenceResults *MemDep=nullptr, bool PredecessorWithTwoSuccessors=false, DominatorTree *DT=nullptr)
Attempts to merge a block into its predecessor, if possible.
LLVM_ABI void hoistAllInstructionsInto(BasicBlock *DomBlock, Instruction *InsertPt, BasicBlock *BB)
Hoist all of the instructions in the IfBlock to the dominant block DomBlock, by moving its instructio...
Definition Local.cpp:3401
@ Sub
Subtraction of integers.
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
Definition STLExtras.h:2028
IntPtrTy
Definition InstrProf.h:82
void RemapInstruction(Instruction *I, ValueToValueMapTy &VM, RemapFlags Flags=RF_None, ValueMapTypeRemapper *TypeMapper=nullptr, ValueMaterializer *Materializer=nullptr, const MetadataPredicate *IdentityMD=nullptr)
Convert the instruction operands from referencing the current values into those specified by VM.
LLVM_ABI bool canReplaceOperandWithVariable(const Instruction *I, unsigned OpIdx)
Given an instruction, is it legal to set operand OpIdx to a non-constant value?
Definition Local.cpp:3907
DWARFExpression::Operation Op
LLVM_ABI bool PointerMayBeCaptured(const Value *V, bool ReturnCaptures, unsigned MaxUsesToExplore=0)
PointerMayBeCaptured - Return true if this pointer value may be captured by the enclosing function (w...
LLVM_ABI bool FoldSingleEntryPHINodes(BasicBlock *BB, MemoryDependenceResults *MemDep=nullptr)
We know that BB has one predecessor.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
void RemapDbgRecord(Module *M, DbgRecord *DR, ValueToValueMapTy &VM, RemapFlags Flags=RF_None, ValueMapTypeRemapper *TypeMapper=nullptr, ValueMaterializer *Materializer=nullptr, const MetadataPredicate *IdentityMD=nullptr)
Remap the Values used in the DbgRecord DR using the value map VM.
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned BitWidth
auto sum_of(R &&Range, E Init=E{0})
Returns the sum of all values in Range with Init initial value.
Definition STLExtras.h:1733
ValueMap< const Value *, WeakTrackingVH > ValueToValueMapTy
LLVM_ABI bool isGuaranteedToTransferExecutionToSuccessor(const Instruction *I)
Return true if this function can prove that the instruction I will always transfer execution to one o...
static cl::opt< bool > HoistCondStores("simplifycfg-hoist-cond-stores", cl::Hidden, cl::init(true), cl::desc("Hoist conditional stores if an unconditional store precedes"))
LLVM_ABI bool extractBranchWeights(const MDNode *ProfileData, SmallVectorImpl< uint32_t > &Weights)
Extract branch weights from MD_prof metadata.
LLVM_ABI bool simplifyCFG(BasicBlock *BB, const TargetTransformInfo &TTI, DomTreeUpdater *DTU=nullptr, const SimplifyCFGOptions &Options={}, ArrayRef< WeakVH > LoopHeaders={})
auto pred_begin(const MachineBasicBlock *BB)
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
Definition Sequence.h:341
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Definition MathExtras.h:249
auto predecessors(const MachineBasicBlock *BB)
static cl::opt< unsigned > HoistLoadsStoresWithCondFaultingThreshold("hoist-loads-stores-with-cond-faulting-threshold", cl::Hidden, cl::init(6), cl::desc("Control the maximal conditional load/store that we are willing " "to speculatively execute to eliminate conditional branch " "(default = 6)"))
static cl::opt< bool > HoistCommon("simplifycfg-hoist-common", cl::Hidden, cl::init(true), cl::desc("Hoist common instructions up to the parent block"))
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
Definition iterator.h:368
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
static cl::opt< unsigned > TwoEntryPHINodeFoldingThreshold("two-entry-phi-node-folding-threshold", cl::Hidden, cl::init(4), cl::desc("Control the maximal total instruction cost that we are willing " "to speculatively execute to fold a 2-entry PHI node into a " "select (default = 4)"))
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
PointerUnion< const Value *, const PseudoSourceValue * > ValueType
SmallVector< uint64_t, 2 > getDisjunctionWeights(const SmallVector< T1, 2 > &B1, const SmallVector< T2, 2 > &B2)
Get the branch weights of a branch conditioned on b1 || b2, where b1 and b2 are 2 booleans that are t...
bool pred_empty(const BasicBlock *BB)
Definition CFG.h:107
LLVM_ABI Constant * ConstantFoldCastInstruction(unsigned opcode, Constant *V, Type *DestTy)
LLVM_ABI Instruction * SplitBlockAndInsertIfThen(Value *Cond, BasicBlock::iterator SplitBefore, bool Unreachable, MDNode *BranchWeights=nullptr, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr, BasicBlock *ThenBlock=nullptr)
Split the containing block at the specified instruction - everything before SplitBefore stays in the ...
LLVM_ABI std::optional< bool > isImpliedByDomCondition(const Value *Cond, const Instruction *ContextI, const DataLayout &DL)
Return the boolean condition value in the context of the given instruction if it is known based on do...
void array_pod_sort(IteratorTy Start, IteratorTy End)
array_pod_sort - This sorts an array with the specified start and end extent.
Definition STLExtras.h:1612
LLVM_ABI bool hasBranchWeightMD(const Instruction &I)
Checks if an instructions has Branch Weight Metadata.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
Definition Hashing.h:307
LLVM_ABI bool isDereferenceablePointer(const Value *V, Type *Ty, const SimplifyQuery &Q, bool IgnoreFree=false)
Equivalent to isDereferenceableAndAlignedPointer with an alignment of 1.
Definition Loads.cpp:264
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2162
static cl::opt< bool > HoistLoadsWithCondFaulting("simplifycfg-hoist-loads-with-cond-faulting", cl::Hidden, cl::init(true), cl::desc("Hoist loads if the target supports conditional faulting"))
LLVM_ABI Constant * ConstantFoldInstOperands(const Instruction *I, ArrayRef< Constant * > Ops, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldInstOperands - Attempt to constant fold an instruction with the specified operands.
LLVM_ABI void setFittedBranchWeights(Instruction &I, ArrayRef< uint64_t > Weights, bool IsExpected, bool ElideAllZero=false)
Variant of setBranchWeights where the Weights will be fit first to uint32_t by shifting right.
LLVM_ABI Constant * ConstantFoldIntegerCast(Constant *C, Type *DestTy, bool IsSigned, const DataLayout &DL)
Constant fold a zext, sext or trunc, depending on IsSigned and whether the DestTy is wider or narrowe...
bool capturesNothing(CaptureComponents CC)
Definition ModRef.h:375
static auto filterDbgVars(iterator_range< simple_ilist< DbgRecord >::iterator > R)
Filter the DbgRecord range to DbgVariableRecord types only and downcast.
LLVM_ABI bool EliminateDuplicatePHINodes(BasicBlock *BB)
Check for and eliminate duplicate PHI nodes in this block.
Definition Local.cpp:1501
@ Keep
No function return thunk.
Definition CodeGen.h:307
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
LLVM_ABI void RemapSourceAtom(Instruction *I, ValueToValueMapTy &VM)
Remap source location atom.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
Definition Hashing.h:287
LLVM_ABI bool isWritableObject(const Value *Object, bool &ExplicitlyDereferenceableOnly)
Return true if the Object is writable, in the sense that any location based on this pointer that can ...
LLVM_ABI void mapAtomInstance(const DebugLoc &DL, ValueToValueMapTy &VMap)
Mark a cloned instruction as a new instance so that its source loc can be updated when remapped.
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
Definition MathExtras.h:368
LLVM_ABI void extractFromBranchWeightMD64(const MDNode *ProfileData, SmallVectorImpl< uint64_t > &Weights)
Faster version of extractBranchWeights() that skips checks and must only be called with "branch_weigh...
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
SmallVectorImpl< ConstantInt * > * Cases
SmallVectorImpl< ConstantInt * > * OtherCases
Checking whether two BBs are equal depends on the contents of the BasicBlock and the incoming values ...
SmallDenseMap< BasicBlock *, Value *, 8 > BB2ValueMap
Phi2IVsMap * PhiPredIVs
DenseMap< PHINode *, BB2ValueMap > Phi2IVsMap
static bool canBeMerged(const BasicBlock *BB)
BasicBlock * BB
static bool isEqual(const EqualBBWrapper *LHS, const EqualBBWrapper *RHS)
static unsigned getHashValue(const EqualBBWrapper *EBW)
An information struct used to provide DenseMap with the various necessary components for a given valu...
Matching combinators.
A MapVector that performs no allocations if smaller than a certain size.
Definition MapVector.h:342