65#define DEBUG_TYPE "si-lower-control-flow"
73class SILowerControlFlow {
89 bool EnableOptimizeEndCf =
false;
124 while (
I != End && !
I->isUnconditionalBranch())
130 void optimizeEndCf();
133 SILowerControlFlow(
const GCNSubtarget *ST, LiveIntervals *LIS,
134 MachineDominatorTree *MDT, MachinePostDominatorTree *PDT)
135 : LIS(LIS), MDT(MDT), PDT(PDT), LMC(AMDGPU::LaneMaskConstants::
get(*
ST)) {
144 SILowerControlFlowLegacy() : MachineFunctionPass(ID) {}
148 StringRef getPassName()
const override {
149 return "SI Lower control flow pseudo instructions";
152 void getAnalysisUsage(AnalysisUsage &AU)
const override {
160 AU.
addPreserved<MachineBlockFrequencyInfoWrapperPass>();
167char SILowerControlFlowLegacy::ID = 0;
181 setImpSCCDefDead(
MI, OrigSCCDef.
isDead());
188 DenseSet<const MachineBasicBlock*> Visited;
189 SmallVector<MachineBasicBlock *, 4> Worklist(Begin->
successors());
191 while (!Worklist.empty()) {
192 MachineBasicBlock *
MBB = Worklist.pop_back_val();
206 Register SaveExecReg =
MI.getOperand(0).getReg();
217void SILowerControlFlow::emitIf(MachineInstr &
MI) {
221 Register SaveExecReg =
MI.getOperand(0).getReg();
222 MachineOperand&
Cond =
MI.getOperand(1);
223 assert(
Cond.getSubReg() == AMDGPU::NoSubRegister);
239 Register CopyReg = SimpleIf ? SaveExecReg
241 MachineInstr *CopyExec =
BuildMI(
MBB,
I,
DL,
TII->get(AMDGPU::COPY), CopyReg)
244 LoweredIf.
insert(CopyReg);
250 setImpSCCDefDead(*
And);
252 MachineInstr *
Xor =
nullptr;
257 copySCCDefDead(*
Xor,
MI.getOperand(4));
262 MachineInstr *SetExec =
264 .
addReg(Tmp, RegState::Kill);
268 I = skipToUncondBrOrEnd(
MBB,
I);
272 MachineInstr *NewBr =
BuildMI(
MBB,
I,
DL,
TII->get(AMDGPU::S_CBRANCH_EXECZ))
273 .
add(
MI.getOperand(2));
276 MI.eraseFromParent();
291 MI.eraseFromParent();
296 RecomputeRegs.
insert(SaveExecReg);
302void SILowerControlFlow::emitElse(MachineInstr &
MI) {
314 MachineInstr *OrSaveExec =
316 .
add(
MI.getOperand(1));
317 setImpSCCDefDead(*OrSaveExec,
true);
319 MachineBasicBlock *DestBB =
MI.getOperand(2).getMBB();
328 setImpSCCDefDead(*
And,
true);
334 copySCCDefDead(*
Xor,
MI.getOperand(4));
338 ElsePt = skipToUncondBrOrEnd(
MBB, ElsePt);
345 MI.eraseFromParent();
350 MI.eraseFromParent();
358 RecomputeRegs.
insert(SrcReg);
359 RecomputeRegs.
insert(DstReg);
363void SILowerControlFlow::emitIfBreak(MachineInstr &
MI) {
366 auto Dst =
MI.getOperand(0).getReg();
372 bool SkipAnding =
false;
373 if (
MI.getOperand(1).isReg()) {
375 SkipAnding =
Def->getParent() ==
MI.getParent() &&
382 MachineInstr *
And =
nullptr, *
Or =
nullptr;
388 .
add(
MI.getOperand(1));
389 setImpSCCDefDead(*
And,
true);
392 .
add(
MI.getOperand(2));
395 .
add(
MI.getOperand(1))
396 .
add(
MI.getOperand(2));
399 copySCCDefDead(*
Or,
MI.getOperand(3));
405 RecomputeRegs.
insert(
And->getOperand(2).getReg());
411 MI.eraseFromParent();
414void SILowerControlFlow::emitLoop(MachineInstr &
MI) {
418 MachineInstr *AndN2 =
421 .
add(
MI.getOperand(0));
422 copySCCDefDead(*AndN2,
MI.getOperand(3));
427 .
add(
MI.getOperand(1));
430 RecomputeRegs.
insert(
MI.getOperand(0).getReg());
435 MI.eraseFromParent();
439SILowerControlFlow::skipIgnoreExecInstsTrivialSucc(
442 SmallPtrSet<const MachineBasicBlock *, 4> Visited;
443 MachineBasicBlock *
B = &
MBB;
449 for ( ; It !=
E; ++It) {
450 if (
TII->mayReadEXEC(*MRI, *It))
457 if (
B->succ_size() != 1)
461 MachineBasicBlock *Succ = *
B->succ_begin();
468MachineBasicBlock *SILowerControlFlow::emitEndCf(MachineInstr &
MI) {
477 bool NeedBlockSplit =
false;
481 if (
I->modifiesRegister(DataReg,
TRI)) {
482 NeedBlockSplit =
true;
487 unsigned Opcode = LMC.
OrOpc;
488 MachineBasicBlock *SplitBB = &
MBB;
489 if (NeedBlockSplit) {
491 if (SplitBB != &
MBB && (MDT || PDT)) {
494 for (MachineBasicBlock *Succ : SplitBB->
successors()) {
495 DTUpdates.
push_back({DomTreeT::Insert, SplitBB, Succ});
510 .
add(
MI.getOperand(0));
511 copySCCDefDead(*NewMI,
MI.getOperand(2));
513 LoweredEndCf.
insert(NewMI);
518 MI.eraseFromParent();
527void SILowerControlFlow::findMaskOperands(
528 MachineInstr &
MI,
unsigned OpNo,
529 SmallVectorImpl<MachineOperand *> &Src)
const {
530 MachineOperand &
Op =
MI.getOperand(OpNo);
531 if (!
Op.isReg() || !
Op.getReg().isVirtual()) {
537 if (!Def ||
Def->getParent() !=
MI.getParent() ||
538 !(
Def->isFullCopy() || (
Def->getOpcode() ==
MI.getOpcode())))
544 for (
auto I =
Def->getIterator();
I !=
MI.getIterator(); ++
I)
545 if (
I->modifiesRegister(AMDGPU::EXEC,
TRI) &&
546 !(
I->isCopy() &&
I->getOperand(0).getReg() != LMC.
ExecReg))
549 for (MachineOperand &SrcOp :
Def->explicit_operands())
550 if (SrcOp.isReg() && SrcOp.isUse() &&
551 (SrcOp.getReg().isVirtual() || SrcOp.getReg() == LMC.
ExecReg))
552 Src.push_back(&SrcOp);
559void SILowerControlFlow::combineMasks(MachineInstr &
MI) {
560 assert(
MI.getNumExplicitOperands() == 3);
562 findMaskOperands(
MI, 1, Src1);
563 findMaskOperands(
MI, 2, Src2);
567 unsigned OpToReplace;
568 MachineOperand *Leaf, *NestedLHS, *NestedRHS;
569 if (Src1.
size() == 2 && Src2.
size() == 1) {
574 }
else if (Src1.
size() == 1 && Src2.
size() == 2) {
584 MachineOperand *KeepOp;
594 MI.removeOperand(OpToReplace);
595 MI.addOperand(*KeepOp);
600void SILowerControlFlow::optimizeEndCf() {
603 if (!EnableOptimizeEndCf)
606 for (MachineInstr *
MI :
reverse(LoweredEndCf)) {
609 skipIgnoreExecInstsTrivialSucc(
MBB, std::next(
MI->getIterator()));
615 =
TII->getNamedOperand(*
Next, AMDGPU::OpName::src1)->getReg();
619 if (Def && LoweredIf.
count(SavedExec)) {
623 MI->eraseFromParent();
624 removeMBBifRedundant(
MBB);
629MachineBasicBlock *SILowerControlFlow::process(MachineInstr &
MI) {
632 MachineInstr *Prev = (
I !=
MBB.
begin()) ? &*(std::prev(
I)) : nullptr;
634 MachineBasicBlock *SplitBB = &
MBB;
636 switch (
MI.getOpcode()) {
641 case AMDGPU::SI_ELSE:
645 case AMDGPU::SI_IF_BREAK:
649 case AMDGPU::SI_LOOP:
653 case AMDGPU::SI_WATERFALL_LOOP:
654 MI.setDesc(
TII->get(AMDGPU::S_CBRANCH_EXECNZ));
657 case AMDGPU::SI_END_CF:
658 SplitBB = emitEndCf(
MI);
662 assert(
false &&
"Attempt to process unsupported instruction");
669 MachineInstr &MaskMI = *
I;
671 case AMDGPU::S_AND_B64:
672 case AMDGPU::S_OR_B64:
673 case AMDGPU::S_AND_B32:
674 case AMDGPU::S_OR_B32:
676 combineMasks(MaskMI);
687bool SILowerControlFlow::removeMBBifRedundant(MachineBasicBlock &
MBB) {
689 if (!
I.isDebugInstr() && !
I.isUnconditionalBranch())
696 MachineBasicBlock *FallThrough =
nullptr;
703 if (
P->getFallThrough(
false) == &
MBB)
706 DTUpdates.
push_back({DomTreeT::Insert,
P, Succ});
745 MachineInstr *BranchMI =
BuildMI(*FallThrough, FallThrough->
end(),
757 TII =
ST.getInstrInfo();
763 BoolRC =
TRI->getBoolRC();
766 const bool CanDemote =
768 for (
auto &
MBB : MF) {
769 bool IsKillBlock =
false;
771 if (
TII->isKillTerminator(
Term.getOpcode())) {
777 if (CanDemote && !IsKillBlock) {
778 for (
auto &
MI :
MBB) {
779 if (
MI.getOpcode() == AMDGPU::SI_DEMOTE_I1) {
790 BI != MF.end(); BI = NextBB) {
791 NextBB = std::next(BI);
792 MachineBasicBlock *
MBB = &*BI;
798 MachineInstr &
MI = *
I;
799 MachineBasicBlock *SplitMBB =
MBB;
801 switch (
MI.getOpcode()) {
803 case AMDGPU::SI_ELSE:
804 case AMDGPU::SI_IF_BREAK:
805 case AMDGPU::SI_WATERFALL_LOOP:
806 case AMDGPU::SI_LOOP:
807 case AMDGPU::SI_END_CF:
808 SplitMBB = process(
MI);
813 if (SplitMBB !=
MBB) {
832 RecomputeRegs.clear();
833 LoweredEndCf.
clear();
840bool SILowerControlFlowLegacy::runOnMachineFunction(
MachineFunction &MF) {
843 auto *LISWrapper = getAnalysisIfAvailable<LiveIntervalsWrapperPass>();
844 LiveIntervals *LIS = LISWrapper ? &LISWrapper->getLIS() :
nullptr;
845 auto *MDTWrapper = getAnalysisIfAvailable<MachineDominatorTreeWrapperPass>();
846 MachineDominatorTree *MDT = MDTWrapper ? &MDTWrapper->getDomTree() :
nullptr;
848 getAnalysisIfAvailable<MachinePostDominatorTreeWrapperPass>();
849 MachinePostDominatorTree *PDT =
850 PDTWrapper ? &PDTWrapper->getPostDomTree() :
nullptr;
851 return SILowerControlFlow(ST, LIS, MDT, PDT).run(MF);
864 bool Changed = SILowerControlFlow(ST, LIS, MDT, PDT).run(MF);
MachineInstrBuilder & UseMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
Register const TargetRegisterInfo * TRI
Promote Memory to Register
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< bool > RemoveRedundantEndcf("amdgpu-remove-redundant-endcf", cl::init(true), cl::ReallyHidden)
static bool isSimpleIf(const MachineInstr &MI, const MachineRegisterInfo *MRI)
const unsigned XorTermOpc
const unsigned MovTermOpc
const unsigned OrSaveExecOpc
const unsigned AndN2TermOpc
PassT::Result * getCachedResult(IRUnitT &IR) const
Get the cached result of an analysis pass for a given IR unit.
AnalysisUsage & addUsedIfAvailable()
Add the specified Pass class to the set of analyses used by this pass.
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
Implements a dense probed hash-table based set.
void applyUpdates(ArrayRef< UpdateType > Updates)
Inform the dominator tree about a sequence of CFG edge insertions and deletions and perform a batch u...
void eraseNode(NodeT *BB)
eraseNode - Removes a node from the dominator tree.
DomTreeNodeBase< NodeT > * getNode(const NodeT *BB) const
getNode - return the (Post)DominatorTree node for the specified basic block.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
const HexagonRegisterInfo & getRegisterInfo() const
void removeAllRegUnitsForPhysReg(MCRegister Reg)
Remove associated live ranges for the register units associated with Reg.
bool hasInterval(Register Reg) const
SlotIndex getMBBStartIdx(const MachineBasicBlock *mbb) const
Return the first index in the given basic block.
SlotIndex InsertMachineInstrInMaps(MachineInstr &MI)
LLVM_ABI void handleMove(MachineInstr &MI, bool UpdateFlags=false)
Call this method to notify LiveIntervals that instruction MI has been moved within a basic block.
SlotIndexes * getSlotIndexes() const
void RemoveMachineInstrFromMaps(MachineInstr &MI)
LiveInterval & getInterval(Register Reg)
void removeInterval(Register Reg)
Interval removal.
LiveInterval & createAndComputeVirtRegInterval(Register Reg)
SlotIndex ReplaceMachineInstrInMaps(MachineInstr &MI, MachineInstr &NewMI)
bool liveAt(SlotIndex index) const
succ_iterator succ_begin()
unsigned succ_size() const
LLVM_ABI void removeSuccessor(MachineBasicBlock *Succ, bool NormalizeSuccProbs=false)
Remove successor from the successors list of this MachineBasicBlock.
pred_iterator pred_begin()
LLVM_ABI void ReplaceUsesOfBlockWith(MachineBasicBlock *Old, MachineBasicBlock *New)
Given a machine basic block that branched to 'Old', change the code and CFG so that it branches to 'N...
LLVM_ABI bool isLayoutSuccessor(const MachineBasicBlock *MBB) const
Return true if the specified MBB will be emitted immediately after this block, such that if this bloc...
LLVM_ABI MachineBasicBlock * splitAt(MachineInstr &SplitInst, bool UpdateLiveIns=true, LiveIntervals *LIS=nullptr)
Split a basic block into 2 pieces at SplitPoint.
LLVM_ABI void eraseFromParent()
This method unlinks 'this' from the containing function and deletes it.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator_range< iterator > terminators()
LLVM_ABI DebugLoc findBranchDebugLoc()
Find and return the merged DebugLoc of the branch instructions of the block.
iterator_range< succ_iterator > successors()
iterator_range< pred_iterator > predecessors()
MachineInstrBundleIterator< MachineInstr > iterator
Analysis pass which computes a MachineDominatorTree.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand class - Representation of each machine instruction operand.
void setIsDead(bool Val=true)
Register getReg() const
getReg - Returns the register number.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
MachinePostDominatorTree - an analysis pass wrapper for DominatorTree used to compute the post-domina...
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
use_instr_nodbg_iterator use_instr_nodbg_begin(Register RegNo) const
unsigned getNumVirtRegs() const
getNumVirtRegs - Return the number of virtual registers created.
bool use_empty(Register RegNo) const
use_empty - Return true if there are no instructions using the specified register.
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
static use_instr_nodbg_iterator use_instr_nodbg_end()
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
static bool isVALU(const MachineInstr &MI, bool AllowLDSDMA)
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
A vector that has set insertion semantics.
size_type count(const_arg_type key) const
Count the number of elements of a given key in the SetVector.
void clear()
Completely clear the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
SlotIndex getPrevSlot() const
Returns the previous slot in the index list.
LLVM_ABI void removeMBBFromMaps(MachineBasicBlock &MBB)
Inverse of insertMBBInMaps: merge MBB's slot range into its layout predecessor and drop it from the m...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
std::pair< iterator, bool > insert(const ValueT &V)
size_type count(const_arg_type_t< ValueT > V) const
Return 1 if the specified key is in the set, 0 otherwise.
self_iterator getIterator()
initializer< Ty > init(const Ty &Val)
PointerTypeMap run(const Module &M)
Compute the PointerTypeMap for the module M.
NodeAddr< DefNode * > Def
This is an optimization pass for GlobalISel generic memory operations.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
char & SILowerControlFlowLegacyID
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
auto reverse(ContainerTy &&C)
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
DominatorTreeBase< T, false > DomTreeBase
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
@ Or
Bitwise or logical OR of integers.
@ Xor
Bitwise or logical XOR of integers.
@ And
Bitwise or logical AND of integers.
DWARFExpression::Operation Op
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
MCRegisterClass TargetRegisterClass