LLVM 24.0.0git
AArch64CodeLayoutOpt.cpp
Go to the documentation of this file.
1//===-- AArch64CodeLayoutOpt.cpp - Code Layout Optimizations --===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This pass runs after instruction scheduling and employs code layout
10// optimizations for certain patterns.
11//
12// Option -aarch64-code-layout-opt-enable selects instruction pairs to optimize:
13// cmp-csel: Enable CMP/CMN-CSEL code layout optimization
14// fcmp-fcsel: Enable FCMP-FCSEL code layout optimization
15//
16// The initial implementation induces function alignment when a supported
17// pattern is detected, and possibly instruction-alignment when a pair would
18// straddle cache-lines.
19//===----------------------------------------------------------------------===//
20
21#include "AArch64.h"
22#include "AArch64InstrInfo.h"
23#include "AArch64Subtarget.h"
26#include "llvm/ADT/Statistic.h"
30#include "llvm/Support/Debug.h"
33
34using namespace llvm;
35
36#define DEBUG_TYPE "aarch64-code-layout-opt"
37#define DBG(...) LLVM_DEBUG(dbgs() << DEBUG_TYPE ": " << __VA_ARGS__)
38#define AARCH64_CODE_LAYOUT_OPT_NAME "AArch64 Code Layout Optimization"
39
41 None = 0,
42 CmpCsel = 1 << 0, // Align CMP/CMN-CSEL pairs
43 FcmpFcsel = 1 << 1, // Align FCMP-FCSEL pairs
45};
46
48 "aarch64-code-layout-opt-enable", cl::Hidden, cl::CommaSeparated,
49 cl::desc("Enable code alignment optimization for instruction pairs"),
51 clEnumValN(None, "none", "Disable the code alignment pass"),
52 clEnumValN(CmpCsel, "cmp-csel", "CMP/CMN-CSEL pair alignment (32-bit)"),
53 clEnumValN(FcmpFcsel, "fcmp-fcsel", "FCMP-FCSEL pair alignment")));
54
56 "aarch64-code-layout-opt-align-functions", cl::Hidden,
57 cl::desc("Function alignment in bytes for code layout optimization "
58 "(must be a power of 2)"),
59 cl::init(64));
60
61STATISTIC(NumFunctionsAligned,
62 "Number of functions with aligned (to 64-bytes by default)");
63STATISTIC(NumCmpCselPairsDetected,
64 "Number of CMP/CMN-CSEL pairs detected for alignment");
65STATISTIC(NumFcmpFcselPairsDetected,
66 "Number of FCMP-FCSEL pairs detected for alignment");
67
68namespace {
69
70class AArch64CodeLayoutOpt : public MachineFunctionPass {
71public:
72 static char ID;
73 AArch64CodeLayoutOpt() : MachineFunctionPass(ID) {}
74 void getAnalysisUsage(AnalysisUsage &AU) const override;
75 bool runOnMachineFunction(MachineFunction &MF) override;
76 StringRef getPassName() const override {
78 }
79
80private:
81 const AArch64InstrInfo *TII = nullptr;
82
83 /// Align each fusible CMP/CMN-CSEL or FCMP-FCSEL pair in \p MBB by emitting
84 /// .p2align before the lead instruction (splitting the block if needed).
85 /// \returns true iff at least one pair was found and aligned.
86 bool alignLayoutSensitivePatterns(MachineBasicBlock *MBB, CodeLayoutOpt CLO);
87
88 /// Emit .p2align before MI. Splits the block if MI is not at its start.
89 void emitP2Align(MachineInstr &MI, Align DesiredAlign,
90 unsigned MaxSkipBytes = 4);
91
92 bool optimizeForCodeLayout(MachineFunction &MF, CodeLayoutOpt CLO);
93};
94
95} // end anonymous namespace
96
97char AArch64CodeLayoutOpt::ID = 0;
98
99INITIALIZE_PASS(AArch64CodeLayoutOpt, "aarch64-code-layout-opt",
100 AARCH64_CODE_LAYOUT_OPT_NAME, false, false)
101
102void AArch64CodeLayoutOpt::getAnalysisUsage(AnalysisUsage &AU) const {
103 AU.setPreservesAll();
105}
106
108 return new AArch64CodeLayoutOpt();
109}
110
111/// \returns true iff Opc is a floating-point comparison (FCMP/FCMPE).
112static bool isFloatingPointCompare(unsigned Opc) {
113 switch (Opc) {
114 case AArch64::FCMPSrr:
115 case AArch64::FCMPDrr:
116 case AArch64::FCMPESrr:
117 case AArch64::FCMPEDrr:
118 case AArch64::FCMPHrr:
119 case AArch64::FCMPEHrr:
120 return true;
121 default:
122 return false;
123 }
124}
125
126/// \returns true iff Opc is a floating-point conditional select (FCSEL).
128 switch (Opc) {
129 case AArch64::FCSELSrrr:
130 case AArch64::FCSELDrrr:
131 case AArch64::FCSELHrrr:
132 return true;
133 default:
134 return false;
135 }
136}
137
138/// \returns true if MI is a qualifying 32-bit CMP or CMN instruction.
139/// CMP is encoded as SUBS with WZR destination, CMN as ADDS with WZR.
140/// Only simple variants (no shifted/extended reg) qualify, and immediate
141/// variants require no LSL shift and small immediates (<=15).
143 switch (MI.getOpcode()) {
144 case AArch64::SUBSWrr:
145 case AArch64::ADDSWrr:
146 return MI.definesRegister(AArch64::WZR, /*TRI=*/nullptr);
147 case AArch64::SUBSWri:
148 case AArch64::ADDSWri:
149 return MI.definesRegister(AArch64::WZR, /*TRI=*/nullptr) &&
150 MI.getOperand(3).getImm() == 0 && MI.getOperand(2).getImm() <= 15;
151 case AArch64::SUBSWrs:
152 case AArch64::ADDSWrs:
153 return MI.definesRegister(AArch64::WZR, /*TRI=*/nullptr) &&
154 !AArch64InstrInfo::hasShiftedReg(MI);
155 case AArch64::SUBSWrx:
156 return MI.definesRegister(AArch64::WZR, /*TRI=*/nullptr) &&
157 !AArch64InstrInfo::hasExtendedReg(MI);
158 default:
159 return false;
160 }
161}
162
163bool AArch64CodeLayoutOpt::runOnMachineFunction(MachineFunction &MF) {
164 const Function &F = MF.getFunction();
165 // hasOptSize() returns true for both -Os and -Oz.
166 if (F.hasOptSize())
167 return false;
168
169 const auto *Subtarget = &MF.getSubtarget<AArch64Subtarget>();
170
171 // Aligning basic blocks currently isn't compatible with Windows unwind info.
172 if (Subtarget->isTargetWindows())
173 return false;
174
175 TII = Subtarget->getInstrInfo();
176
177 CodeLayoutOpt CLO = None;
178 if (EnableCodeAlignment.getNumOccurrences()) {
183 } else {
184 // Default: enable when the subtarget opts in via FeatureAlignCmpCSelPairs.
185 if (Subtarget->hasAlignCmpCSelPairs()) {
186 if (Subtarget->hasFuseCmpCSel())
188 if (Subtarget->hasFuseFCmpFCSel())
190 }
191 }
192
193 if (CLO == None)
194 return false;
195
196 return optimizeForCodeLayout(MF, CLO);
197}
198
199void AArch64CodeLayoutOpt::emitP2Align(MachineInstr &MI, Align DesiredAlign,
200 unsigned MaxSkipBytes) {
201 MachineBasicBlock *MBB = MI.getParent();
202
203 auto FirstReal =
205 if (&*FirstReal != &MI) {
206 auto PrevIt = prev_nodbg(MI.getIterator(), MBB->instr_begin());
207 MBB = MBB->splitAt(*PrevIt, /*UpdateLiveIns=*/true);
208 }
209
210 MBB->setAlignment(DesiredAlign);
211 MBB->setMaxBytesForAlignment(MaxSkipBytes);
212}
213
214// Align each fusible CMP/CMN-CSEL or FCMP-FCSEL pair in MBB by emitting
215// .p2align before the lead instruction (splitting the block if needed).
216// A pair is: a qualifying lead instruction immediately followed by its
217// consumer (CMP/CMN→CSEL or FCMP→FCSEL), with no intervening instructions.
218// Returns true iff at least one pair was found and aligned.
219bool AArch64CodeLayoutOpt::alignLayoutSensitivePatterns(MachineBasicBlock *MBB,
220 CodeLayoutOpt CLO) {
221 auto End = MBB->instr_end();
223
224 for (auto &MI : instructionsWithoutDebug(MBB->begin(), MBB->end())) {
225 auto NextIt =
226 skipDebugInstructionsForward(std::next(MI.getIterator()), End);
227 if (NextIt == End)
228 break;
229
230 // --- CMP/CMN-CSEL detection ---
232 NextIt->getOpcode() == AArch64::CSELWr) {
233 Pairs.push_back({&MI, true});
234 continue;
235 }
236
237 // --- FCMP-FCSEL detection ---
238 if ((CLO & CodeLayoutOpt::FcmpFcsel) &&
239 isFloatingPointCompare(MI.getOpcode()) &&
240 isFloatingPointConditionalSelect(NextIt->getOpcode())) {
241 Pairs.push_back({&MI, false});
242 continue;
243 }
244 }
245
246 for (auto &[MI, IsCmpCsel] : Pairs) {
247 emitP2Align(*MI, Align(64));
248 DBG(".p2align 6, , 4 before " << *MI);
249 ++(IsCmpCsel ? NumCmpCselPairsDetected : NumFcmpFcselPairsDetected);
250 }
251
252 return !Pairs.empty();
253}
254
255bool AArch64CodeLayoutOpt::optimizeForCodeLayout(MachineFunction &MF,
256 CodeLayoutOpt CLO) {
257 DBG("optimizeForCodeLayout: " << MF.getName() << "\n");
258
259 bool Changed = false;
260 for (auto &MBB : MF)
261 Changed |= alignLayoutSensitivePatterns(&MBB, CLO);
262
263 if (!Changed)
264 return false;
265
268 "aarch64-code-layout-opt-align-functions must be a power of 2");
269 if (MF.getAlignment() < Align(FunctionAlignBytes)) {
270 MF.setAlignment(Align(FunctionAlignBytes));
271 ++NumFunctionsAligned;
272 DBG("Set " << FunctionAlignBytes << "-byte alignment for function "
273 << MF.getName() << "\n");
274 } else {
275 DBG("Function " << MF.getName() << " already has sufficient alignment\n");
276 }
277 return true;
278}
static bool isFloatingPointConditionalSelect(unsigned Opc)
#define AARCH64_CODE_LAYOUT_OPT_NAME
static cl::list< CodeLayoutOpt > EnableCodeAlignment("aarch64-code-layout-opt-enable", cl::Hidden, cl::CommaSeparated, cl::desc("Enable code alignment optimization for instruction pairs"), cl::values(clEnumValN(None, "none", "Disable the code alignment pass"), clEnumValN(CmpCsel, "cmp-csel", "CMP/CMN-CSEL pair alignment (32-bit)"), clEnumValN(FcmpFcsel, "fcmp-fcsel", "FCMP-FCSEL pair alignment")))
static bool isFloatingPointCompare(unsigned Opc)
#define DBG(...)
static bool isQualifyingIntCompare(const MachineInstr &MI)
static cl::opt< unsigned > FunctionAlignBytes("aarch64-code-layout-opt-align-functions", cl::Hidden, cl::desc("Function alignment in bytes for code layout optimization " "(must be a power of 2)"), cl::init(64))
MachineBasicBlock & MBB
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
#define F(x, y, z)
Definition MD5.cpp:54
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
Represent the analysis usage information of a pass.
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
void setMaxBytesForAlignment(unsigned MaxBytes)
Set the maximum amount of padding allowed for aligning the basic block.
void setAlignment(Align A)
Set alignment of the basic block.
LLVM_ABI MachineBasicBlock * splitAt(MachineInstr &SplitInst, bool UpdateLiveIns=true, LiveIntervals *LIS=nullptr)
Split a basic block into 2 pieces at SplitPoint.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
StringRef getName() const
getName - Return the name of the corresponding LLVM function.
Function & getFunction()
Return the LLVM function that this machine code represents.
Representation of each machine instruction.
void push_back(const T &Elt)
Changed
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
IterT skipDebugInstructionsForward(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It until it points to a non-debug instruction or to End and return the resulting iterator.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
auto instructionsWithoutDebug(IterT It, IterT End, bool SkipPseudoOp=true)
Construct a range iterator which begins at It and moves forwards until End is reached,...
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
FunctionPass * createAArch64CodeLayoutOptPass()
IterT prev_nodbg(IterT It, IterT Begin, bool SkipPseudoOp=true)
Decrement It, then continue decrementing it while it points to a debug instruction.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177