LLVM 24.0.0git
X86FrameLowering.cpp
Go to the documentation of this file.
1//===-- X86FrameLowering.cpp - X86 Frame Information ----------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains the X86 implementation of TargetFrameLowering class.
10//
11//===----------------------------------------------------------------------===//
12
13#include "X86FrameLowering.h"
15#include "X86.h"
16#include "X86InstrBuilder.h"
17#include "X86InstrInfo.h"
19#include "X86Subtarget.h"
20#include "X86TargetMachine.h"
21#include "llvm/ADT/Statistic.h"
30#include "llvm/IR/DataLayout.h"
32#include "llvm/IR/Function.h"
33#include "llvm/IR/Module.h"
34#include "llvm/MC/MCAsmInfo.h"
36#include "llvm/MC/MCSymbol.h"
37#include "llvm/Support/LEB128.h"
39#include <cstdlib>
40
41#define DEBUG_TYPE "x86-fl"
42
43STATISTIC(NumFrameLoopProbe, "Number of loop stack probes used in prologue");
44STATISTIC(NumFrameExtraProbe,
45 "Number of extra stack probes generated in prologue");
46STATISTIC(NumFunctionUsingPush2Pop2, "Number of functions using push2/pop2");
47
48using namespace llvm;
49
51 const Function &Fn = MF.getFunction();
52
53 // Whole module is in V3 mode.
55 return true;
56
57 // Otherwise promote a function that may use EGPR (R16-R31), which V1/V2
58 // unwind codes cannot encode. The per-function "+egpr" feature is the signal,
59 // so an auto-dispatch APX clone gets V3 while the baseline clone stays on the
60 // module default. We conservatively promote any egpr function rather than
61 // checking for an actual EGPR save, keeping this a cheap query. (PUSH2/POP2
62 // does not need V3: V1/V2 describe a PUSH2 as two SEH_PushReg codes.)
63 return Fn.needsUnwindTableEntry() &&
64 MF.getSubtarget<X86Subtarget>().hasEGPR();
65}
66
67static const TargetRegisterClass *
69 const TargetRegisterInfo &TRI) {
70 if (X86::VK16RegClass.contains(Reg))
71 return STI.hasBWI() ? &X86::VK64RegClass : &X86::VK16RegClass;
72 return TRI.getMinimalPhysRegClass(Reg);
73}
74
76 MaybeAlign StackAlignOverride)
77 : TargetFrameLowering(StackGrowsDown, StackAlignOverride.valueOrOne(),
78 STI.is64Bit() ? -8 : -4),
79 STI(STI), TII(*STI.getInstrInfo()), TRI(STI.getRegisterInfo()) {
80 // Cache a bunch of frame-related predicates for this subtarget.
81 SlotSize = TRI->getSlotSize();
82 assert(SlotSize == 4 || SlotSize == 8);
83 Is64Bit = STI.is64Bit();
84 IsLP64 = STI.isTarget64BitLP64();
85 // standard x86_64 uses 64-bit frame/stack pointers, x32 - 32-bit.
86 Uses64BitFramePtr = STI.isTarget64BitLP64();
87 StackPtr = TRI->getStackRegister();
88}
89
91 return !MF.getFrameInfo().hasVarSizedObjects() &&
92 !MF.getInfo<X86MachineFunctionInfo>()->getHasPushSequences() &&
93 !MF.getInfo<X86MachineFunctionInfo>()->hasPreallocatedCall();
94}
95
96/// canSimplifyCallFramePseudos - If there is a reserved call frame, the
97/// call frame pseudos can be simplified. Having a FP, as in the default
98/// implementation, is not sufficient here since we can't always use it.
99/// Use a more nuanced condition.
101 const MachineFunction &MF) const {
102 return hasReservedCallFrame(MF) ||
103 MF.getInfo<X86MachineFunctionInfo>()->hasPreallocatedCall() ||
104 (hasFP(MF) && !TRI->hasStackRealignment(MF)) ||
105 TRI->hasBasePointer(MF);
106}
107
108// needsFrameIndexResolution - Do we need to perform FI resolution for
109// this function. Normally, this is required only when the function
110// has any stack objects. However, FI resolution actually has another job,
111// not apparent from the title - it resolves callframesetup/destroy
112// that were not simplified earlier.
113// So, this is required for x86 functions that have push sequences even
114// when there are no stack objects.
116 const MachineFunction &MF) const {
117 return MF.getFrameInfo().hasStackObjects() ||
118 MF.getInfo<X86MachineFunctionInfo>()->getHasPushSequences();
119}
120
121/// hasFPImpl - Return true if the specified function should have a dedicated
122/// frame pointer register. This is true if the function has variable sized
123/// allocas or if frame pointer elimination is disabled.
125 const MachineFrameInfo &MFI = MF.getFrameInfo();
126 return (MF.disableFramePointerElim() || TRI->hasStackRealignment(MF) ||
128 MFI.hasOpaqueSPAdjustment() ||
131 MF.callsUnwindInit() || MF.hasEHFunclets() || MF.callsEHReturn() ||
132 MFI.hasStackMap() || MFI.hasPatchPoint() ||
133 (isWin64Prologue(MF) && MFI.hasCopyImplyingStackAdjustment()));
134}
135
136static unsigned getSUBriOpcode(bool IsLP64) {
137 return IsLP64 ? X86::SUB64ri32 : X86::SUB32ri;
138}
139
140static unsigned getADDriOpcode(bool IsLP64) {
141 return IsLP64 ? X86::ADD64ri32 : X86::ADD32ri;
142}
143
144static unsigned getSUBrrOpcode(bool IsLP64) {
145 return IsLP64 ? X86::SUB64rr : X86::SUB32rr;
146}
147
148static unsigned getADDrrOpcode(bool IsLP64) {
149 return IsLP64 ? X86::ADD64rr : X86::ADD32rr;
150}
151
152static unsigned getANDriOpcode(bool IsLP64, int64_t Imm) {
153 return IsLP64 ? X86::AND64ri32 : X86::AND32ri;
154}
155
156static unsigned getLEArOpcode(bool IsLP64) {
157 return IsLP64 ? X86::LEA64r : X86::LEA32r;
158}
159
160// Push-Pop Acceleration (PPX) hint is used to indicate that the POP reads the
161// value written by the PUSH from the stack. The processor tracks these marked
162// instructions internally and fast-forwards register data between matching PUSH
163// and POP instructions, without going through memory or through the training
164// loop of the Fast Store Forwarding Predictor (FSFP). Instead, a more efficient
165// memory-renaming optimization can be used.
166//
167// The PPX hint is purely a performance hint. Instructions with this hint have
168// the same functional semantics as those without. PPX hints set by the
169// compiler that violate the balancing rule may turn off the PPX optimization,
170// but they will not affect program semantics.
171//
172// Hence, PPX is used for balanced spill/reloads (Exceptions and setjmp/longjmp
173// are not considered).
174//
175// PUSH2 and POP2 are instructions for (respectively) pushing/popping 2
176// GPRs at a time to/from the stack.
177static unsigned getPUSHOpcode(const X86Subtarget &ST) {
178 return ST.is64Bit() ? (ST.hasPPX() ? X86::PUSHP64r : X86::PUSH64r)
179 : X86::PUSH32r;
180}
181static unsigned getPOPOpcode(const X86Subtarget &ST) {
182 return ST.is64Bit() ? (ST.hasPPX() ? X86::POPP64r : X86::POP64r)
183 : X86::POP32r;
184}
185static unsigned getPUSH2Opcode(const X86Subtarget &ST) {
186 return ST.hasPPX() ? X86::PUSH2P : X86::PUSH2;
187}
188static unsigned getPOP2Opcode(const X86Subtarget &ST) {
189 return ST.hasPPX() ? X86::POP2P : X86::POP2;
190}
191
193 for (MachineBasicBlock::RegisterMaskPair RegMask : MBB.liveins()) {
194 MCRegister Reg = RegMask.PhysReg;
195
196 if (Reg == X86::RAX || Reg == X86::EAX || Reg == X86::AX ||
197 Reg == X86::AH || Reg == X86::AL)
198 return true;
199 }
200
201 return false;
202}
203
204/// Check if the flags need to be preserved before the terminators.
205/// This would be the case, if the eflags is live-in of the region
206/// composed by the terminators or live-out of that region, without
207/// being defined by a terminator.
208static bool
210 for (const MachineInstr &MI : MBB.terminators()) {
211 bool BreakNext = false;
212 for (const MachineOperand &MO : MI.operands()) {
213 if (!MO.isReg())
214 continue;
215 Register Reg = MO.getReg();
216 if (Reg != X86::EFLAGS)
217 continue;
218
219 // This terminator needs an eflags that is not defined
220 // by a previous another terminator:
221 // EFLAGS is live-in of the region composed by the terminators.
222 if (!MO.isDef())
223 return true;
224 // This terminator defines the eflags, i.e., we don't need to preserve it.
225 // However, we still need to check this specific terminator does not
226 // read a live-in value.
227 BreakNext = true;
228 }
229 // We found a definition of the eflags, no need to preserve them.
230 if (BreakNext)
231 return false;
232 }
233
234 // None of the terminators use or define the eflags.
235 // Check if they are live-out, that would imply we need to preserve them.
236 for (const MachineBasicBlock *Succ : MBB.successors())
237 if (Succ->isLiveIn(X86::EFLAGS))
238 return true;
239
240 return false;
241}
242
243constexpr uint64_t MaxSPChunk = (1ULL << 31) - 1;
244
245/// emitSPUpdate - Emit a series of instructions to increment / decrement the
246/// stack pointer by a constant value.
249 const DebugLoc &DL, int64_t NumBytes,
250 bool InEpilogue) const {
251 bool isSub = NumBytes < 0;
252 uint64_t Offset = isSub ? -NumBytes : NumBytes;
255
257 // We're being asked to adjust a 32-bit stack pointer by 4 GiB or more.
258 // This might be unreachable code, so don't complain now; just trap if
259 // it's reached at runtime.
260 BuildMI(MBB, MBBI, DL, TII.get(X86::TRAP));
261 return;
262 }
263
264 MachineFunction &MF = *MBB.getParent();
266 const X86TargetLowering &TLI = *STI.getTargetLowering();
267 const bool EmitInlineStackProbe = TLI.hasInlineStackProbe(MF);
268
269 // It's ok to not take into account large chunks when probing, as the
270 // allocation is split in smaller chunks anyway.
271 if (EmitInlineStackProbe && !InEpilogue) {
272
273 // This pseudo-instruction is going to be expanded, potentially using a
274 // loop, by inlineStackProbe().
275 BuildMI(MBB, MBBI, DL, TII.get(X86::STACKALLOC_W_PROBING)).addImm(Offset);
276 return;
277 } else if (Offset > MaxSPChunk) {
278 // Rather than emit a long series of instructions for large offsets,
279 // load the offset into a register and do one sub/add
280 unsigned Reg = 0;
281 unsigned Rax = (unsigned)(Uses64BitFramePtr ? X86::RAX : X86::EAX);
282
283 if (isSub && !isEAXLiveIn(MBB))
284 Reg = Rax;
285 else
286 Reg = getX86SubSuperRegister(TRI->findDeadCallerSavedReg(MBB, MBBI),
287 Uses64BitFramePtr ? 64 : 32);
288
289 unsigned AddSubRROpc = isSub ? getSUBrrOpcode(Uses64BitFramePtr)
291 if (Reg) {
292 BuildMI(MBB, MBBI, DL,
294 .addImm(Offset)
295 .setMIFlag(Flag);
296 MachineInstr *MI = BuildMI(MBB, MBBI, DL, TII.get(AddSubRROpc), StackPtr)
298 .addReg(Reg);
299 MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
300 return;
301 } else if (Offset > 8 * MaxSPChunk) {
302 // If we would need more than 8 add or sub instructions (a >16GB stack
303 // frame), it's worth spilling RAX to materialize this immediate.
304 // pushq %rax
305 // movabsq +-$Offset+-SlotSize, %rax
306 // addq %rsp, %rax
307 // xchg %rax, (%rsp)
308 // movq (%rsp), %rsp
309 assert(Uses64BitFramePtr && "can't have 32-bit 16GB stack frame");
310 BuildMI(MBB, MBBI, DL, TII.get(X86::PUSH64r))
312 .setMIFlag(Flag);
313 // Subtract is not commutative, so negate the offset and always use add.
314 // Subtract 8 less and add 8 more to account for the PUSH we just did.
315 if (isSub)
316 Offset = -(Offset - SlotSize);
317 else
319 BuildMI(MBB, MBBI, DL,
321 .addImm(Offset)
322 .setMIFlag(Flag);
323 MachineInstr *MI = BuildMI(MBB, MBBI, DL, TII.get(X86::ADD64rr), Rax)
324 .addReg(Rax)
326 MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
327 // Exchange the new SP in RAX with the top of the stack.
329 BuildMI(MBB, MBBI, DL, TII.get(X86::XCHG64rm), Rax).addReg(Rax),
330 StackPtr, false, 0);
331 // Load new SP from the top of the stack into RSP.
332 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::MOV64rm), StackPtr),
333 StackPtr, false, 0);
334 return;
335 }
336 }
337
338 while (Offset) {
339 if (Offset == SlotSize) {
340 // Use push / pop for slot sized adjustments as a size optimization. We
341 // need to find a dead register when using pop.
342 unsigned Reg = isSub ? (unsigned)(Is64Bit ? X86::RAX : X86::EAX)
343 : TRI->findDeadCallerSavedReg(MBB, MBBI);
344 if (Reg) {
345 unsigned Opc = isSub ? (Is64Bit ? X86::PUSH64r : X86::PUSH32r)
346 : (Is64Bit ? X86::POP64r : X86::POP32r);
347 BuildMI(MBB, MBBI, DL, TII.get(Opc))
348 .addReg(Reg, getDefRegState(!isSub) | getUndefRegState(isSub))
349 .setMIFlag(Flag);
350 return;
351 }
352 }
353
354 uint64_t ThisVal = std::min(Offset, MaxSPChunk);
355
356 BuildStackAdjustment(MBB, MBBI, DL, isSub ? -ThisVal : ThisVal, InEpilogue)
357 .setMIFlag(Flag);
358
359 Offset -= ThisVal;
360 }
361}
362
363MachineInstrBuilder X86FrameLowering::BuildStackAdjustment(
365 const DebugLoc &DL, int64_t Offset, bool InEpilogue) const {
366 assert(Offset != 0 && "zero offset stack adjustment requested");
367
368 // On Atom, using LEA to adjust SP is preferred, but using it in the epilogue
369 // is tricky.
370 bool UseLEA;
371 if (!InEpilogue) {
372 // Check if inserting the prologue at the beginning
373 // of MBB would require to use LEA operations.
374 // We need to use LEA operations if EFLAGS is live in, because
375 // it means an instruction will read it before it gets defined.
376 UseLEA = STI.useLeaForSP() || MBB.isLiveIn(X86::EFLAGS);
377 } else {
378 // If we can use LEA for SP but we shouldn't, check that none
379 // of the terminators uses the eflags. Otherwise we will insert
380 // a ADD that will redefine the eflags and break the condition.
381 // Alternatively, we could move the ADD, but this may not be possible
382 // and is an optimization anyway.
383 UseLEA = canUseLEAForSPInEpilogue(*MBB.getParent());
384 if (UseLEA && !STI.useLeaForSP())
386 // If that assert breaks, that means we do not do the right thing
387 // in canUseAsEpilogue.
389 "We shouldn't have allowed this insertion point");
390 }
391
392 MachineInstrBuilder MI;
393 // Use an NF (no-flags) variant as a smaller replacement for LEA when EFLAGS
394 // must be preserved (i.e. only when we would otherwise emit LEA). If EFLAGS
395 // is dead we prefer the plain SUB/ADD, which is shorter than the EVEX-encoded
396 // NF form. The NF stack-adjust opcodes below are 64-bit (SUB64ri32_NF/
397 // ADD64ri32_NF), so don't use them for the x32 ABI where the stack pointer is
398 // 32-bit. NF cannot reach a Win64 epilogue (which never uses LEA for the SP
399 // adjustment unless it has a frame pointer, and that path doesn't go through
400 // here), so the Windows epilogue unwinder never sees an undisassemblable NF
401 // add/sub.
402 bool UseNF = UseLEA && STI.hasNF() && Uses64BitFramePtr;
403 bool IsSub = Offset < 0;
404 uint64_t AbsOffset = IsSub ? -Offset : Offset;
405 if (UseNF) {
406 const unsigned Opc = IsSub ? X86::SUB64ri32_NF : X86::ADD64ri32_NF;
407 MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
409 .addImm(AbsOffset);
410 // NF instructions define no EFLAGS, so there is nothing to mark dead.
411 } else if (UseLEA) {
414 StackPtr),
415 StackPtr, false, Offset);
416 } else {
417 unsigned Opc = IsSub ? getSUBriOpcode(Uses64BitFramePtr)
419 int64_t Imm = AbsOffset;
420 // Prefer `add rsp, -128` over `sub rsp, 128` (and vice versa in the
421 // epilogue): 128 is the one magnitude whose negation fits the
422 // sign-extended 8-bit immediate while the value itself does not, so the
423 // flipped operation is three bytes shorter. EFLAGS is dead here (this
424 // branch clobbers it anyway). Windows CFI epilogues keep the canonical
425 // ADD: v1 unwind info describes no epilogues, so the unwinder detects
426 // one by disassembling forward for `add rsp, imm` (prologues are
427 // delimited by SizeOfProlog and never disassembled). Unwind v2/v3 do
428 // describe epilogues, but X86WinEHUnwindV2 expects the ADD spelling
429 // too.
430 if (AbsOffset == 128 &&
431 !(InEpilogue &&
433 Opc = IsSub ? getADDriOpcode(Uses64BitFramePtr)
434 : getSUBriOpcode(Uses64BitFramePtr);
435 Imm = -128;
436 }
437 MI = BuildMI(MBB, MBBI, DL, TII.get(Opc), StackPtr)
439 .addImm(Imm);
440 MI->getOperand(3).setIsDead(); // The EFLAGS implicit def is dead.
441 }
442 return MI;
443}
444
445template <typename FoundT, typename CalcT>
446int64_t X86FrameLowering::mergeSPUpdates(MachineBasicBlock &MBB,
448 FoundT FoundStackAdjust,
449 CalcT CalcNewOffset,
450 bool doMergeWithPrevious) const {
451 if ((doMergeWithPrevious && MBBI == MBB.begin()) ||
452 (!doMergeWithPrevious && MBBI == MBB.end()))
453 return CalcNewOffset(0);
454
455 MachineBasicBlock::iterator PI = doMergeWithPrevious ? std::prev(MBBI) : MBBI;
456
458 // It is assumed that ADD/SUB/LEA instruction is succeded by one CFI
459 // instruction, and that there are no DBG_VALUE or other instructions between
460 // ADD/SUB/LEA and its corresponding CFI instruction.
461 /* TODO: Add support for the case where there are multiple CFI instructions
462 below the ADD/SUB/LEA, e.g.:
463 ...
464 add
465 cfi_def_cfa_offset
466 cfi_offset
467 ...
468 */
469 if (doMergeWithPrevious && PI != MBB.begin() && PI->isCFIInstruction())
470 PI = std::prev(PI);
471
472 int64_t Offset = 0;
473 for (;;) {
474 unsigned Opc = PI->getOpcode();
475
476 if ((Opc == X86::ADD64ri32 || Opc == X86::ADD32ri ||
477 Opc == X86::ADD64ri32_NF) &&
478 PI->getOperand(0).getReg() == StackPtr) {
479 assert(PI->getOperand(1).getReg() == StackPtr);
480 Offset = PI->getOperand(2).getImm();
481 } else if ((Opc == X86::LEA32r || Opc == X86::LEA64_32r) &&
482 PI->getOperand(0).getReg() == StackPtr &&
483 PI->getOperand(1).getReg() == StackPtr &&
484 PI->getOperand(2).getImm() == 1 &&
485 PI->getOperand(3).getReg() == X86::NoRegister &&
486 PI->getOperand(5).getReg() == X86::NoRegister) {
487 // For LEAs we have: def = lea SP, FI, noreg, Offset, noreg.
488 Offset = PI->getOperand(4).getImm();
489 } else if ((Opc == X86::SUB64ri32 || Opc == X86::SUB32ri ||
490 Opc == X86::SUB64ri32_NF) &&
491 PI->getOperand(0).getReg() == StackPtr) {
492 assert(PI->getOperand(1).getReg() == StackPtr);
493 Offset = -PI->getOperand(2).getImm();
494 } else
495 return CalcNewOffset(0);
496
497 FoundStackAdjust(PI, Offset);
498 if ((uint64_t)std::abs((int64_t)CalcNewOffset(Offset)) < MaxSPChunk)
499 break;
500
501 if (doMergeWithPrevious ? (PI == MBB.begin()) : (PI == MBB.end()))
502 return CalcNewOffset(0);
503
504 PI = doMergeWithPrevious ? std::prev(PI) : std::next(PI);
505 }
506
507 PI = MBB.erase(PI);
508 if (PI != MBB.end() && PI->isCFIInstruction()) {
509 auto CIs = MBB.getParent()->getFrameInstructions();
510 MCCFIInstruction CI = CIs[PI->getOperand(0).getCFIIndex()];
513 PI = MBB.erase(PI);
514 }
515 if (!doMergeWithPrevious)
517
518 return CalcNewOffset(Offset);
519}
520
523 int64_t AddOffset,
524 bool doMergeWithPrevious) const {
525 return mergeSPUpdates(
526 MBB, MBBI, [AddOffset](int64_t Offset) { return AddOffset + Offset; },
527 doMergeWithPrevious);
528}
529
532 const DebugLoc &DL,
533 const MCCFIInstruction &CFIInst,
534 MachineInstr::MIFlag Flag) const {
535 MachineFunction &MF = *MBB.getParent();
536 unsigned CFIIndex = MF.addFrameInst(CFIInst);
537
539 MF.getInfo<X86MachineFunctionInfo>()->setHasCFIAdjustCfa(true);
540
541 BuildMI(MBB, MBBI, DL, TII.get(TargetOpcode::CFI_INSTRUCTION))
542 .addCFIIndex(CFIIndex)
543 .setMIFlag(Flag);
544}
545
546/// Emits Dwarf Info specifying offsets of callee saved registers and
547/// frame pointer. This is called only when basic block sections are enabled.
550 MachineFunction &MF = *MBB.getParent();
551 if (!hasFP(MF)) {
553 return;
554 }
555 const MCRegisterInfo *MRI = MF.getContext().getRegisterInfo();
556 const Register FramePtr = TRI->getFrameRegister(MF);
557 const Register MachineFramePtr =
558 STI.isTarget64BitILP32() ? Register(getX86SubSuperRegister(FramePtr, 64))
559 : FramePtr;
560 unsigned DwarfReg = MRI->getDwarfRegNum(MachineFramePtr, true);
561 // Offset = space for return address + size of the frame pointer itself.
562 int64_t Offset = (Is64Bit ? 8 : 4) + (Uses64BitFramePtr ? 8 : 4);
564 MCCFIInstruction::createOffset(nullptr, DwarfReg, -Offset));
566}
567
570 const DebugLoc &DL, bool IsPrologue) const {
571 MachineFunction &MF = *MBB.getParent();
572 MachineFrameInfo &MFI = MF.getFrameInfo();
573 const MCRegisterInfo *MRI = MF.getContext().getRegisterInfo();
575
576 // Add callee saved registers to move list.
577 const std::vector<CalleeSavedInfo> &CSI = MFI.getCalleeSavedInfo();
578
579 // Calculate offsets.
580 for (const CalleeSavedInfo &I : CSI) {
581 int64_t Offset = MFI.getObjectOffset(I.getFrameIdx());
582 MCRegister Reg = I.getReg();
583 unsigned DwarfReg = MRI->getDwarfRegNum(Reg, true);
584
585 if (IsPrologue) {
586 if (X86FI->getStackPtrSaveMI()) {
587 // +2*SlotSize because there is return address and ebp at the bottom
588 // of the stack.
589 // | retaddr |
590 // | ebp |
591 // | |<--ebp
592 Offset += 2 * SlotSize;
593 SmallString<64> CfaExpr;
594 CfaExpr.push_back(dwarf::DW_CFA_expression);
595 uint8_t buffer[16];
596 CfaExpr.append(buffer, buffer + encodeULEB128(DwarfReg, buffer));
597 CfaExpr.push_back(2);
598 Register FramePtr = TRI->getFrameRegister(MF);
599 const Register MachineFramePtr =
600 STI.isTarget64BitILP32()
602 : FramePtr;
603 unsigned DwarfFramePtr = MRI->getDwarfRegNum(MachineFramePtr, true);
604 CfaExpr.push_back((uint8_t)(dwarf::DW_OP_breg0 + DwarfFramePtr));
605 CfaExpr.append(buffer, buffer + encodeSLEB128(Offset, buffer));
607 MCCFIInstruction::createEscape(nullptr, CfaExpr.str()),
609 } else {
611 MCCFIInstruction::createOffset(nullptr, DwarfReg, Offset));
612 }
613 } else {
615 MCCFIInstruction::createRestore(nullptr, DwarfReg));
616 }
617 }
618 if (auto *MI = X86FI->getStackPtrSaveMI()) {
619 int FI = MI->getOperand(1).getIndex();
620 int64_t Offset = MFI.getObjectOffset(FI) + 2 * SlotSize;
621 SmallString<64> CfaExpr;
622 Register FramePtr = TRI->getFrameRegister(MF);
623 const Register MachineFramePtr =
624 STI.isTarget64BitILP32()
626 : FramePtr;
627 unsigned DwarfFramePtr = MRI->getDwarfRegNum(MachineFramePtr, true);
628 CfaExpr.push_back((uint8_t)(dwarf::DW_OP_breg0 + DwarfFramePtr));
629 uint8_t buffer[16];
630 CfaExpr.append(buffer, buffer + encodeSLEB128(Offset, buffer));
631 CfaExpr.push_back(dwarf::DW_OP_deref);
632
633 SmallString<64> DefCfaExpr;
634 DefCfaExpr.push_back(dwarf::DW_CFA_def_cfa_expression);
635 DefCfaExpr.append(buffer, buffer + encodeSLEB128(CfaExpr.size(), buffer));
636 DefCfaExpr.append(CfaExpr.str());
637 // DW_CFA_def_cfa_expression: DW_OP_breg5 offset, DW_OP_deref
639 MCCFIInstruction::createEscape(nullptr, DefCfaExpr.str()),
641 }
642}
643
644void X86FrameLowering::emitZeroCallUsedRegs(BitVector RegsToZero,
646 RegScavenger *) const {
647 const MachineFunction &MF = *MBB.getParent();
648
649 // Insertion point.
650 MachineBasicBlock::iterator MBBI = MBB.getFirstTerminator();
651
652 // Fake a debug loc.
653 DebugLoc DL;
654 if (MBBI != MBB.end())
655 DL = MBBI->getDebugLoc();
656
657 // Zero out FP stack if referenced. Do this outside of the loop below so that
658 // it's done only once.
659 for (MCRegister Reg : RegsToZero.set_bits()) {
660 if (!X86::RFP80RegClass.contains(Reg))
661 continue;
662
663 // Do not push zeros over x87 return values. X86FloatingPoint records
664 // returned values as implicit ST0/ST1 uses on the return instruction.
665 unsigned NumFPRegs = 8;
666 if (MBBI->hasRegisterImplicitUseOperand(X86::ST0))
667 --NumFPRegs;
668 if (MBBI->hasRegisterImplicitUseOperand(X86::ST1))
669 --NumFPRegs;
670
671 for (unsigned i = 0; i != NumFPRegs; ++i)
672 BuildMI(MBB, MBBI, DL, TII.get(X86::LD_F0));
673
674 for (unsigned i = 0; i != NumFPRegs; ++i)
675 BuildMI(MBB, MBBI, DL, TII.get(X86::ST_FPrr)).addReg(X86::ST0);
676 break;
677 }
678
679 // For GPRs, we only care to clear out the 32-bit register.
680 BitVector GPRsToZero(TRI->getNumRegs());
681 for (MCRegister Reg : RegsToZero.set_bits())
682 if (TRI->isGeneralPurposeRegister(MF, Reg)) {
683 GPRsToZero.set(getX86SubSuperRegister(Reg, 32));
684 RegsToZero.reset(Reg);
685 }
686
687 // Zero out the GPRs first.
688 for (MCRegister Reg : GPRsToZero.set_bits())
689 TII.buildClearRegister(Reg, MBB, MBBI, DL);
690
691 // Coalesce the aliasing XMM/YMM/ZMM views of each vector register so a lane
692 // is cleared only once, mirroring the GPR handling above.
693 auto getVectorClearReg = [&](MCRegister Reg) -> MCRegister {
694 if (!X86::VR128RegClass.contains(Reg) &&
695 !X86::VR128XRegClass.contains(Reg) &&
696 !X86::VR256RegClass.contains(Reg) &&
697 !X86::VR256XRegClass.contains(Reg) && !X86::VR512RegClass.contains(Reg))
698 return MCRegister();
699
700 // Clearing the XMM zeroes the whole lane. XMM0-15 use the compact VEX form;
701 // XMM16-31 are EVEX-only, reachable only via the ZMM form.
702 MCRegister Xmm = TRI->getSubReg(Reg, X86::sub_xmm);
703 if (!Xmm)
704 Xmm = Reg;
705 if (X86::VR128RegClass.contains(Xmm))
706 return Xmm;
707 MCRegister Zmm =
708 TRI->getMatchingSuperReg(Xmm, X86::sub_xmm, &X86::VR512RegClass);
709 assert(Zmm && "XMM16-31 must have an enclosing ZMM to clear through");
710 return Zmm;
711 };
712
713 BitVector VecRegsToZero(TRI->getNumRegs());
714 for (MCRegister Reg : RegsToZero.set_bits())
715 if (MCRegister Clear = getVectorClearReg(Reg)) {
716 VecRegsToZero.set(Clear.id());
717 RegsToZero.reset(Reg);
718 }
719
720 for (MCRegister Reg : VecRegsToZero.set_bits())
721 TII.buildClearRegister(Reg, MBB, MBBI, DL);
722
723 // Zero out the remaining registers (e.g. mask registers).
724 for (MCRegister Reg : RegsToZero.set_bits())
725 TII.buildClearRegister(Reg, MBB, MBBI, DL);
726}
727
730 MachineBasicBlock::iterator MBBI, const DebugLoc &DL, bool InProlog,
731 std::optional<MachineFunction::DebugInstrOperandPair> InstrNum) const {
733 if (STI.isTargetWindowsCoreCLR()) {
734 if (InProlog) {
735 BuildMI(MBB, MBBI, DL, TII.get(X86::STACKALLOC_W_PROBING))
736 .addImm(0 /* no explicit stack size */);
737 } else {
738 emitStackProbeInline(MF, MBB, MBBI, DL, false);
739 }
740 } else {
741 emitStackProbeCall(MF, MBB, MBBI, DL, InProlog, InstrNum);
742 }
743}
744
746 return STI.isOSWindows() && !STI.isTargetWin64();
747}
748
750 MachineBasicBlock &PrologMBB) const {
751 auto Where = llvm::find_if(PrologMBB, [](MachineInstr &MI) {
752 return MI.getOpcode() == X86::STACKALLOC_W_PROBING;
753 });
754 if (Where != PrologMBB.end()) {
755 DebugLoc DL = PrologMBB.findDebugLoc(Where);
756 emitStackProbeInline(MF, PrologMBB, Where, DL, true);
757 Where->eraseFromParent();
758 }
759}
760
761void X86FrameLowering::emitStackProbeInline(MachineFunction &MF,
764 const DebugLoc &DL,
765 bool InProlog) const {
767 if (STI.isTargetWindowsCoreCLR() && STI.is64Bit())
768 emitStackProbeInlineWindowsCoreCLR64(MF, MBB, MBBI, DL, InProlog);
769 else
770 emitStackProbeInlineGeneric(MF, MBB, MBBI, DL, InProlog);
771}
772
773void X86FrameLowering::emitStackProbeInlineGeneric(
775 MachineBasicBlock::iterator MBBI, const DebugLoc &DL, bool InProlog) const {
776 MachineInstr &AllocWithProbe = *MBBI;
777 uint64_t Offset = AllocWithProbe.getOperand(0).getImm();
778
781 assert(!(STI.is64Bit() && STI.isTargetWindowsCoreCLR()) &&
782 "different expansion expected for CoreCLR 64 bit");
783
784 const uint64_t StackProbeSize = TLI.getStackProbeSize(MF);
785 uint64_t ProbeChunk = StackProbeSize * 8;
786
787 uint64_t MaxAlign =
788 TRI->hasStackRealignment(MF) ? calculateMaxStackAlign(MF) : 0;
789
790 // Synthesize a loop or unroll it, depending on the number of iterations.
791 // BuildStackAlignAND ensures that only MaxAlign % StackProbeSize bits left
792 // between the unaligned rsp and current rsp.
793 if (Offset > ProbeChunk) {
794 emitStackProbeInlineGenericLoop(MF, MBB, MBBI, DL, Offset,
795 MaxAlign % StackProbeSize);
796 } else {
797 emitStackProbeInlineGenericBlock(MF, MBB, MBBI, DL, Offset,
798 MaxAlign % StackProbeSize);
799 }
800}
801
802void X86FrameLowering::emitStackProbeInlineGenericBlock(
805 uint64_t AlignOffset) const {
806
807 const bool NeedsDwarfCFI = needsDwarfCFI(MF);
808 const bool HasFP = hasFP(MF);
809 const X86Subtarget &STI = MF.getSubtarget<X86Subtarget>();
810 const X86TargetLowering &TLI = *STI.getTargetLowering();
811 const unsigned MovMIOpc = Is64Bit ? X86::MOV64mi32 : X86::MOV32mi;
812 const uint64_t StackProbeSize = TLI.getStackProbeSize(MF);
813
814 uint64_t CurrentOffset = 0;
815
816 assert(AlignOffset < StackProbeSize);
817
818 // If the offset is so small it fits within a page, there's nothing to do.
819 if (StackProbeSize < Offset + AlignOffset) {
820
821 uint64_t StackAdjustment = StackProbeSize - AlignOffset;
822 BuildStackAdjustment(MBB, MBBI, DL, -StackAdjustment, /*InEpilogue=*/false)
823 .setMIFlag(MachineInstr::FrameSetup);
824 if (!HasFP && NeedsDwarfCFI) {
825 BuildCFI(
826 MBB, MBBI, DL,
827 MCCFIInstruction::createAdjustCfaOffset(nullptr, StackAdjustment));
828 }
829
830 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(MovMIOpc))
832 StackPtr, false, 0)
833 .addImm(0)
835 NumFrameExtraProbe++;
836 CurrentOffset = StackProbeSize - AlignOffset;
837 }
838
839 // For the next N - 1 pages, just probe. I tried to take advantage of
840 // natural probes but it implies much more logic and there was very few
841 // interesting natural probes to interleave.
842 while (CurrentOffset + StackProbeSize < Offset) {
843 BuildStackAdjustment(MBB, MBBI, DL, -StackProbeSize, /*InEpilogue=*/false)
844 .setMIFlag(MachineInstr::FrameSetup);
845
846 if (!HasFP && NeedsDwarfCFI) {
847 BuildCFI(
848 MBB, MBBI, DL,
849 MCCFIInstruction::createAdjustCfaOffset(nullptr, StackProbeSize));
850 }
851 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(MovMIOpc))
853 StackPtr, false, 0)
854 .addImm(0)
856 NumFrameExtraProbe++;
857 CurrentOffset += StackProbeSize;
858 }
859
860 // No need to probe the tail, it is smaller than a Page.
861 uint64_t ChunkSize = Offset - CurrentOffset;
862 if (ChunkSize == SlotSize) {
863 // Use push for slot sized adjustments as a size optimization,
864 // like emitSPUpdate does when not probing.
865 unsigned Reg = Is64Bit ? X86::RAX : X86::EAX;
866 unsigned Opc = Is64Bit ? X86::PUSH64r : X86::PUSH32r;
867 BuildMI(MBB, MBBI, DL, TII.get(Opc))
870 } else {
871 BuildStackAdjustment(MBB, MBBI, DL, -ChunkSize, /*InEpilogue=*/false)
872 .setMIFlag(MachineInstr::FrameSetup);
873 }
874 // No need to adjust Dwarf CFA offset here, the last position of the stack has
875 // been defined
876}
877
878void X86FrameLowering::emitStackProbeInlineGenericLoop(
881 uint64_t AlignOffset) const {
882 assert(Offset && "null offset");
883
884 assert(MBB.computeRegisterLiveness(TRI, X86::EFLAGS, MBBI) !=
886 "Inline stack probe loop will clobber live EFLAGS.");
887
888 const bool NeedsDwarfCFI = needsDwarfCFI(MF);
889 const bool HasFP = hasFP(MF);
890 const X86Subtarget &STI = MF.getSubtarget<X86Subtarget>();
891 const X86TargetLowering &TLI = *STI.getTargetLowering();
892 const unsigned MovMIOpc = Is64Bit ? X86::MOV64mi32 : X86::MOV32mi;
893 const uint64_t StackProbeSize = TLI.getStackProbeSize(MF);
894
895 if (AlignOffset) {
896 if (AlignOffset < StackProbeSize) {
897 // Perform a first smaller allocation followed by a probe.
898 BuildStackAdjustment(MBB, MBBI, DL, -AlignOffset, /*InEpilogue=*/false)
899 .setMIFlag(MachineInstr::FrameSetup);
900
901 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(MovMIOpc))
903 StackPtr, false, 0)
904 .addImm(0)
906 NumFrameExtraProbe++;
907 Offset -= AlignOffset;
908 }
909 }
910
911 // Synthesize a loop
912 NumFrameLoopProbe++;
913 const BasicBlock *LLVM_BB = MBB.getBasicBlock();
914
915 MachineBasicBlock *testMBB = MF.CreateMachineBasicBlock(LLVM_BB);
916 MachineBasicBlock *tailMBB = MF.CreateMachineBasicBlock(LLVM_BB);
917
919 MF.insert(MBBIter, testMBB);
920 MF.insert(MBBIter, tailMBB);
921
922 Register FinalStackProbed = Uses64BitFramePtr ? X86::R11
923 : Is64Bit ? X86::R11D
924 : X86::EAX;
925
926 // save loop bound
927 {
928 const uint64_t BoundOffset = alignDown(Offset, StackProbeSize);
929
930 // Can we calculate the loop bound using SUB with a 32-bit immediate?
931 // Note that the immediate gets sign-extended when used with a 64-bit
932 // register, so in that case we only have 31 bits to work with.
933 bool canUseSub =
934 Uses64BitFramePtr ? isUInt<31>(BoundOffset) : isUInt<32>(BoundOffset);
935
936 if (canUseSub) {
937 const unsigned SUBOpc = getSUBriOpcode(Uses64BitFramePtr);
938
939 BuildMI(MBB, MBBI, DL, TII.get(TargetOpcode::COPY), FinalStackProbed)
942 BuildMI(MBB, MBBI, DL, TII.get(SUBOpc), FinalStackProbed)
943 .addReg(FinalStackProbed)
944 .addImm(BoundOffset)
946 } else if (Uses64BitFramePtr) {
947 BuildMI(MBB, MBBI, DL, TII.get(X86::MOV64ri), FinalStackProbed)
948 .addImm(-BoundOffset)
950 BuildMI(MBB, MBBI, DL, TII.get(X86::ADD64rr), FinalStackProbed)
951 .addReg(FinalStackProbed)
954 } else {
955 llvm_unreachable("Offset too large for 32-bit stack pointer");
956 }
957
958 // while in the loop, use loop-invariant reg for CFI,
959 // instead of the stack pointer, which changes during the loop
960 if (!HasFP && NeedsDwarfCFI) {
961 // x32 uses the same DWARF register numbers as x86-64,
962 // so there isn't a register number for r11d, we must use r11 instead
963 const Register DwarfFinalStackProbed =
964 STI.isTarget64BitILP32()
965 ? Register(getX86SubSuperRegister(FinalStackProbed, 64))
966 : FinalStackProbed;
967
970 nullptr, TRI->getDwarfRegNum(DwarfFinalStackProbed, true)));
972 MCCFIInstruction::createAdjustCfaOffset(nullptr, BoundOffset));
973 }
974 }
975
976 // allocate a page
977 BuildStackAdjustment(*testMBB, testMBB->end(), DL, -StackProbeSize,
978 /*InEpilogue=*/false)
979 .setMIFlag(MachineInstr::FrameSetup);
980
981 // touch the page
982 addRegOffset(BuildMI(testMBB, DL, TII.get(MovMIOpc))
984 StackPtr, false, 0)
985 .addImm(0)
987
988 // cmp with stack pointer bound
989 BuildMI(testMBB, DL, TII.get(Uses64BitFramePtr ? X86::CMP64rr : X86::CMP32rr))
991 .addReg(FinalStackProbed)
993
994 // jump
995 BuildMI(testMBB, DL, TII.get(X86::JCC_1))
996 .addMBB(testMBB)
999 testMBB->addSuccessor(testMBB);
1000 testMBB->addSuccessor(tailMBB);
1001
1002 // BB management
1003 tailMBB->splice(tailMBB->end(), &MBB, MBBI, MBB.end());
1005 MBB.addSuccessor(testMBB);
1006
1007 // handle tail
1008 const uint64_t TailOffset = Offset % StackProbeSize;
1009 MachineBasicBlock::iterator TailMBBIter = tailMBB->begin();
1010 if (TailOffset) {
1011 BuildStackAdjustment(*tailMBB, TailMBBIter, DL, -TailOffset,
1012 /*InEpilogue=*/false)
1013 .setMIFlag(MachineInstr::FrameSetup);
1014 }
1015
1016 // after the loop, switch back to stack pointer for CFI
1017 if (!HasFP && NeedsDwarfCFI) {
1018 // x32 uses the same DWARF register numbers as x86-64,
1019 // so there isn't a register number for esp, we must use rsp instead
1020 const Register DwarfStackPtr =
1021 STI.isTarget64BitILP32()
1023 : Register(StackPtr);
1024
1025 BuildCFI(*tailMBB, TailMBBIter, DL,
1027 nullptr, TRI->getDwarfRegNum(DwarfStackPtr, true)));
1028 }
1029
1030 // Update Live In information
1031 fullyRecomputeLiveIns({tailMBB, testMBB});
1032}
1033
1034void X86FrameLowering::emitStackProbeInlineWindowsCoreCLR64(
1036 MachineBasicBlock::iterator MBBI, const DebugLoc &DL, bool InProlog) const {
1037 const X86Subtarget &STI = MF.getSubtarget<X86Subtarget>();
1038 assert(STI.is64Bit() && "different expansion needed for 32 bit");
1039 assert(STI.isTargetWindowsCoreCLR() && "custom expansion expects CoreCLR");
1040 const TargetInstrInfo &TII = *STI.getInstrInfo();
1041 const BasicBlock *LLVM_BB = MBB.getBasicBlock();
1042
1043 assert(MBB.computeRegisterLiveness(TRI, X86::EFLAGS, MBBI) !=
1045 "Inline stack probe loop will clobber live EFLAGS.");
1046
1047 // RAX contains the number of bytes of desired stack adjustment.
1048 // The handling here assumes this value has already been updated so as to
1049 // maintain stack alignment.
1050 //
1051 // We need to exit with RSP modified by this amount and execute suitable
1052 // page touches to notify the OS that we're growing the stack responsibly.
1053 // All stack probing must be done without modifying RSP.
1054 //
1055 // MBB:
1056 // SizeReg = RAX;
1057 // ZeroReg = 0
1058 // CopyReg = RSP
1059 // Flags, TestReg = CopyReg - SizeReg
1060 // FinalReg = !Flags.Ovf ? TestReg : ZeroReg
1061 // LimitReg = gs magic thread env access
1062 // if FinalReg >= LimitReg goto ContinueMBB
1063 // RoundBB:
1064 // RoundReg = page address of FinalReg
1065 // LoopMBB:
1066 // LoopReg = PHI(LimitReg,ProbeReg)
1067 // ProbeReg = LoopReg - PageSize
1068 // [ProbeReg] = 0
1069 // if (ProbeReg > RoundReg) goto LoopMBB
1070 // ContinueMBB:
1071 // RSP = RSP - RAX
1072 // [rest of original MBB]
1073
1074 // Set up the new basic blocks
1075 MachineBasicBlock *RoundMBB = MF.CreateMachineBasicBlock(LLVM_BB);
1076 MachineBasicBlock *LoopMBB = MF.CreateMachineBasicBlock(LLVM_BB);
1077 MachineBasicBlock *ContinueMBB = MF.CreateMachineBasicBlock(LLVM_BB);
1078
1079 MachineFunction::iterator MBBIter = std::next(MBB.getIterator());
1080 MF.insert(MBBIter, RoundMBB);
1081 MF.insert(MBBIter, LoopMBB);
1082 MF.insert(MBBIter, ContinueMBB);
1083
1084 // Split MBB and move the tail portion down to ContinueMBB.
1085 MachineBasicBlock::iterator BeforeMBBI = std::prev(MBBI);
1086 ContinueMBB->splice(ContinueMBB->begin(), &MBB, MBBI, MBB.end());
1087 ContinueMBB->transferSuccessorsAndUpdatePHIs(&MBB);
1088
1089 // Some useful constants
1090 const int64_t ThreadEnvironmentStackLimit = 0x10;
1091 const int64_t PageSize = 0x1000;
1092 const int64_t PageMask = ~(PageSize - 1);
1093
1094 // Registers we need. For the normal case we use virtual
1095 // registers. For the prolog expansion we use RAX, RCX and RDX.
1096 MachineRegisterInfo &MRI = MF.getRegInfo();
1097 const TargetRegisterClass *RegClass = &X86::GR64RegClass;
1098 const Register
1099 SizeReg = InProlog ? X86::RAX : MRI.createVirtualRegister(RegClass),
1100 ZeroReg = InProlog ? X86::RCX : MRI.createVirtualRegister(RegClass),
1101 CopyReg = InProlog ? X86::RDX : MRI.createVirtualRegister(RegClass),
1102 TestReg = InProlog ? X86::RDX : MRI.createVirtualRegister(RegClass),
1103 FinalReg = InProlog ? X86::RDX : MRI.createVirtualRegister(RegClass),
1104 RoundedReg = InProlog ? X86::RDX : MRI.createVirtualRegister(RegClass),
1105 LimitReg = InProlog ? X86::RCX : MRI.createVirtualRegister(RegClass),
1106 JoinReg = InProlog ? X86::RCX : MRI.createVirtualRegister(RegClass),
1107 ProbeReg = InProlog ? X86::RCX : MRI.createVirtualRegister(RegClass);
1108
1109 // SP-relative offsets where we can save RCX and RDX.
1110 int64_t RCXShadowSlot = 0;
1111 int64_t RDXShadowSlot = 0;
1112
1113 // If inlining in the prolog, save RCX and RDX.
1114 if (InProlog) {
1115 // Compute the offsets. We need to account for things already
1116 // pushed onto the stack at this point: return address, frame
1117 // pointer (if used), and callee saves.
1118 X86MachineFunctionInfo *X86FI = MF.getInfo<X86MachineFunctionInfo>();
1119 const int64_t CalleeSaveSize = X86FI->getCalleeSavedFrameSize();
1120 const bool HasFP = hasFP(MF);
1121
1122 // Check if we need to spill RCX and/or RDX.
1123 // Here we assume that no earlier prologue instruction changes RCX and/or
1124 // RDX, so checking the block live-ins is enough.
1125 const bool IsRCXLiveIn = MBB.isLiveIn(X86::RCX);
1126 const bool IsRDXLiveIn = MBB.isLiveIn(X86::RDX);
1127 int64_t InitSlot = 8 + CalleeSaveSize + (HasFP ? 8 : 0);
1128 // Assign the initial slot to both registers, then change RDX's slot if both
1129 // need to be spilled.
1130 if (IsRCXLiveIn)
1131 RCXShadowSlot = InitSlot;
1132 if (IsRDXLiveIn)
1133 RDXShadowSlot = InitSlot;
1134 if (IsRDXLiveIn && IsRCXLiveIn)
1135 RDXShadowSlot += 8;
1136 // Emit the saves if needed.
1137 if (IsRCXLiveIn)
1138 addRegOffset(BuildMI(&MBB, DL, TII.get(X86::MOV64mr)), X86::RSP, false,
1139 RCXShadowSlot)
1140 .addReg(X86::RCX);
1141 if (IsRDXLiveIn)
1142 addRegOffset(BuildMI(&MBB, DL, TII.get(X86::MOV64mr)), X86::RSP, false,
1143 RDXShadowSlot)
1144 .addReg(X86::RDX);
1145 } else {
1146 // Not in the prolog. Copy RAX to a virtual reg.
1147 BuildMI(&MBB, DL, TII.get(X86::MOV64rr), SizeReg).addReg(X86::RAX);
1148 }
1149
1150 // Add code to MBB to check for overflow and set the new target stack pointer
1151 // to zero if so.
1152 BuildMI(&MBB, DL, TII.get(X86::XOR64rr), ZeroReg)
1153 .addReg(ZeroReg, RegState::Undef)
1154 .addReg(ZeroReg, RegState::Undef);
1155 BuildMI(&MBB, DL, TII.get(X86::MOV64rr), CopyReg).addReg(X86::RSP);
1156 BuildMI(&MBB, DL, TII.get(X86::SUB64rr), TestReg)
1157 .addReg(CopyReg)
1158 .addReg(SizeReg);
1159 BuildMI(&MBB, DL, TII.get(X86::CMOV64rr), FinalReg)
1160 .addReg(TestReg)
1161 .addReg(ZeroReg)
1163
1164 // FinalReg now holds final stack pointer value, or zero if
1165 // allocation would overflow. Compare against the current stack
1166 // limit from the thread environment block. Note this limit is the
1167 // lowest touched page on the stack, not the point at which the OS
1168 // will cause an overflow exception, so this is just an optimization
1169 // to avoid unnecessarily touching pages that are below the current
1170 // SP but already committed to the stack by the OS.
1171 BuildMI(&MBB, DL, TII.get(X86::MOV64rm), LimitReg)
1172 .addReg(0)
1173 .addImm(1)
1174 .addReg(0)
1175 .addImm(ThreadEnvironmentStackLimit)
1176 .addReg(X86::GS);
1177 BuildMI(&MBB, DL, TII.get(X86::CMP64rr)).addReg(FinalReg).addReg(LimitReg);
1178 // Jump if the desired stack pointer is at or above the stack limit.
1179 BuildMI(&MBB, DL, TII.get(X86::JCC_1))
1180 .addMBB(ContinueMBB)
1182
1183 // Add code to roundMBB to round the final stack pointer to a page boundary.
1184 if (InProlog)
1185 RoundMBB->addLiveIn(FinalReg);
1186 BuildMI(RoundMBB, DL, TII.get(X86::AND64ri32), RoundedReg)
1187 .addReg(FinalReg)
1188 .addImm(PageMask);
1189 BuildMI(RoundMBB, DL, TII.get(X86::JMP_1)).addMBB(LoopMBB);
1190
1191 // LimitReg now holds the current stack limit, RoundedReg page-rounded
1192 // final RSP value. Add code to loopMBB to decrement LimitReg page-by-page
1193 // and probe until we reach RoundedReg.
1194 if (!InProlog) {
1195 BuildMI(LoopMBB, DL, TII.get(X86::PHI), JoinReg)
1196 .addReg(LimitReg)
1197 .addMBB(RoundMBB)
1198 .addReg(ProbeReg)
1199 .addMBB(LoopMBB);
1200 }
1201
1202 if (InProlog)
1203 LoopMBB->addLiveIn(JoinReg);
1204 addRegOffset(BuildMI(LoopMBB, DL, TII.get(X86::LEA64r), ProbeReg), JoinReg,
1205 false, -PageSize);
1206
1207 // Probe by storing a byte onto the stack.
1208 BuildMI(LoopMBB, DL, TII.get(X86::MOV8mi))
1209 .addReg(ProbeReg)
1210 .addImm(1)
1211 .addReg(0)
1212 .addImm(0)
1213 .addReg(0)
1214 .addImm(0);
1215
1216 if (InProlog)
1217 LoopMBB->addLiveIn(RoundedReg);
1218 BuildMI(LoopMBB, DL, TII.get(X86::CMP64rr))
1219 .addReg(RoundedReg)
1220 .addReg(ProbeReg);
1221 BuildMI(LoopMBB, DL, TII.get(X86::JCC_1))
1222 .addMBB(LoopMBB)
1224
1225 MachineBasicBlock::iterator ContinueMBBI = ContinueMBB->getFirstNonPHI();
1226
1227 // If in prolog, restore RDX and RCX.
1228 if (InProlog) {
1229 if (RCXShadowSlot) // It means we spilled RCX in the prologue.
1230 addRegOffset(BuildMI(*ContinueMBB, ContinueMBBI, DL,
1231 TII.get(X86::MOV64rm), X86::RCX),
1232 X86::RSP, false, RCXShadowSlot);
1233 if (RDXShadowSlot) // It means we spilled RDX in the prologue.
1234 addRegOffset(BuildMI(*ContinueMBB, ContinueMBBI, DL,
1235 TII.get(X86::MOV64rm), X86::RDX),
1236 X86::RSP, false, RDXShadowSlot);
1237 }
1238
1239 // Now that the probing is done, add code to continueMBB to update
1240 // the stack pointer for real.
1241 BuildMI(*ContinueMBB, ContinueMBBI, DL, TII.get(X86::SUB64rr), X86::RSP)
1242 .addReg(X86::RSP)
1243 .addReg(SizeReg);
1244
1245 // Add the control flow edges we need.
1246 MBB.addSuccessor(ContinueMBB);
1247 MBB.addSuccessor(RoundMBB);
1248 RoundMBB->addSuccessor(LoopMBB);
1249 LoopMBB->addSuccessor(ContinueMBB);
1250 LoopMBB->addSuccessor(LoopMBB);
1251
1252 if (InProlog) {
1253 LivePhysRegs LiveRegs;
1254 computeAndAddLiveIns(LiveRegs, *ContinueMBB);
1255 }
1256
1257 // Mark all the instructions added to the prolog as frame setup.
1258 if (InProlog) {
1259 for (++BeforeMBBI; BeforeMBBI != MBB.end(); ++BeforeMBBI) {
1260 BeforeMBBI->setFlag(MachineInstr::FrameSetup);
1261 }
1262 for (MachineInstr &MI : *RoundMBB) {
1264 }
1265 for (MachineInstr &MI : *LoopMBB) {
1267 }
1268 for (MachineInstr &MI :
1269 llvm::make_range(ContinueMBB->begin(), ContinueMBBI)) {
1271 }
1272 }
1273}
1274
1275void X86FrameLowering::emitStackProbeCall(
1277 MachineBasicBlock::iterator MBBI, const DebugLoc &DL, bool InProlog,
1278 std::optional<MachineFunction::DebugInstrOperandPair> InstrNum) const {
1279 bool IsLargeCodeModel = MF.getTarget().getCodeModel() == CodeModel::Large;
1280
1281 // FIXME: Add indirect thunk support and remove this.
1282 if (Is64Bit && IsLargeCodeModel && STI.useIndirectThunkCalls())
1283 report_fatal_error("Emitting stack probe calls on 64-bit with the large "
1284 "code model and indirect thunks not yet implemented.");
1285
1286 assert(MBB.computeRegisterLiveness(TRI, X86::EFLAGS, MBBI) !=
1288 "Stack probe calls will clobber live EFLAGS.");
1289
1290 unsigned CallOp;
1291 if (Is64Bit)
1292 CallOp = IsLargeCodeModel ? X86::CALL64r : X86::CALL64pcrel32;
1293 else
1294 CallOp = X86::CALLpcrel32;
1295
1296 StringRef Symbol = STI.getTargetLowering()->getStackProbeSymbolName(MF);
1297
1298 MachineInstrBuilder CI;
1299 MachineBasicBlock::iterator ExpansionMBBI = std::prev(MBBI);
1300
1301 // All current stack probes take AX and SP as input, clobber flags, and
1302 // preserve all registers. x86_64 probes leave RSP unmodified.
1304 // For the large code model, we have to call through a register. Use R11,
1305 // as it is scratch in all supported calling conventions.
1306 BuildMI(MBB, MBBI, DL, TII.get(X86::MOV64ri), X86::R11)
1308 CI = BuildMI(MBB, MBBI, DL, TII.get(CallOp)).addReg(X86::R11);
1309 } else {
1310 CI = BuildMI(MBB, MBBI, DL, TII.get(CallOp))
1312 }
1313
1314 unsigned AX = Uses64BitFramePtr ? X86::RAX : X86::EAX;
1315 unsigned SP = Uses64BitFramePtr ? X86::RSP : X86::ESP;
1321
1322 MachineInstr *ModInst = CI;
1323 if (STI.isTargetWin64() || !STI.isOSWindows()) {
1324 // MSVC x32's _chkstk and cygwin/mingw's _alloca adjust %esp themselves.
1325 // MSVC x64's __chkstk and cygwin/mingw's ___chkstk_ms do not adjust %rsp
1326 // themselves. They also does not clobber %rax so we can reuse it when
1327 // adjusting %rsp.
1328 // All other platforms do not specify a particular ABI for the stack probe
1329 // function, so we arbitrarily define it to not adjust %esp/%rsp itself.
1330 ModInst =
1332 .addReg(SP)
1333 .addReg(AX);
1334 }
1335
1336 // DebugInfo variable locations -- if there's an instruction number for the
1337 // allocation (i.e., DYN_ALLOC_*), substitute it for the instruction that
1338 // modifies SP.
1339 if (InstrNum) {
1340 if (STI.isTargetWin64() || !STI.isOSWindows()) {
1341 // Label destination operand of the subtract.
1342 MF.makeDebugValueSubstitution(*InstrNum,
1343 {ModInst->getDebugInstrNum(), 0});
1344 } else {
1345 // Label the call. The operand number is the penultimate operand, zero
1346 // based.
1347 unsigned SPDefOperand = ModInst->getNumOperands() - 2;
1349 *InstrNum, {ModInst->getDebugInstrNum(), SPDefOperand});
1350 }
1351 }
1352
1353 if (InProlog) {
1354 // Apply the frame setup flag to all inserted instrs.
1355 for (++ExpansionMBBI; ExpansionMBBI != MBBI; ++ExpansionMBBI)
1356 ExpansionMBBI->setFlag(MachineInstr::FrameSetup);
1357 }
1358}
1359
1360static unsigned calculateSetFPREG(uint64_t SPAdjust) {
1361 // Win64 ABI has a less restrictive limitation of 240; 128 works equally well
1362 // and might require smaller successive adjustments.
1363 const uint64_t Win64MaxSEHOffset = 128;
1364 uint64_t SEHFrameOffset = std::min(SPAdjust, Win64MaxSEHOffset);
1365 // Win64 ABI requires 16-byte alignment for the UWOP_SET_FPREG opcode.
1366 return SEHFrameOffset & -16;
1367}
1368
1369// If we're forcing a stack realignment we can't rely on just the frame
1370// info, we need to know the ABI stack alignment as well in case we
1371// have a call out. Otherwise just make sure we have some alignment - we'll
1372// go with the minimum SlotSize.
1374X86FrameLowering::calculateMaxStackAlign(const MachineFunction &MF) const {
1375 const MachineFrameInfo &MFI = MF.getFrameInfo();
1376 Align MaxAlign = MFI.getMaxAlign(); // Desired stack alignment.
1377 Align StackAlign = getStackAlign();
1378 bool HasRealign = MF.getFunction().hasFnAttribute("stackrealign");
1379 if (HasRealign) {
1380 if (MFI.hasCalls())
1381 MaxAlign = (StackAlign > MaxAlign) ? StackAlign : MaxAlign;
1382 else if (MaxAlign < SlotSize)
1383 MaxAlign = Align(SlotSize);
1384 }
1385
1387 if (HasRealign)
1388 MaxAlign = (MaxAlign > 16) ? MaxAlign : Align(16);
1389 else
1390 MaxAlign = Align(16);
1391 }
1392 return MaxAlign.value();
1393}
1394
1395void X86FrameLowering::BuildStackAlignAND(MachineBasicBlock &MBB,
1397 const DebugLoc &DL, Register Reg,
1398 uint64_t MaxAlign) const {
1399 uint64_t Val = -MaxAlign;
1400 unsigned AndOp = getANDriOpcode(Uses64BitFramePtr, Val);
1401
1402 MachineFunction &MF = *MBB.getParent();
1403 const X86Subtarget &STI = MF.getSubtarget<X86Subtarget>();
1404 const X86TargetLowering &TLI = *STI.getTargetLowering();
1405 const uint64_t StackProbeSize = TLI.getStackProbeSize(MF);
1406 const bool EmitInlineStackProbe = TLI.hasInlineStackProbe(MF);
1407
1408 // We want to make sure that (in worst case) less than StackProbeSize bytes
1409 // are not probed after the AND. This assumption is used in
1410 // emitStackProbeInlineGeneric.
1411 if (Reg == StackPtr && EmitInlineStackProbe && MaxAlign >= StackProbeSize) {
1412 {
1413 NumFrameLoopProbe++;
1414 MachineBasicBlock *entryMBB =
1416 MachineBasicBlock *headMBB =
1418 MachineBasicBlock *bodyMBB =
1420 MachineBasicBlock *footMBB =
1422
1424 MF.insert(MBBIter, entryMBB);
1425 MF.insert(MBBIter, headMBB);
1426 MF.insert(MBBIter, bodyMBB);
1427 MF.insert(MBBIter, footMBB);
1428 const unsigned MovMIOpc = Is64Bit ? X86::MOV64mi32 : X86::MOV32mi;
1429 Register FinalStackProbed = Uses64BitFramePtr ? X86::R11
1430 : Is64Bit ? X86::R11D
1431 : X86::EAX;
1432
1433 // Setup entry block
1434 {
1435
1436 entryMBB->splice(entryMBB->end(), &MBB, MBB.begin(), MBBI);
1437 BuildMI(entryMBB, DL, TII.get(TargetOpcode::COPY), FinalStackProbed)
1440 MachineInstr *MI =
1441 BuildMI(entryMBB, DL, TII.get(AndOp), FinalStackProbed)
1442 .addReg(FinalStackProbed)
1443 .addImm(Val)
1445
1446 // The EFLAGS implicit def is dead.
1447 MI->getOperand(3).setIsDead();
1448
1449 BuildMI(entryMBB, DL,
1450 TII.get(Uses64BitFramePtr ? X86::CMP64rr : X86::CMP32rr))
1451 .addReg(FinalStackProbed)
1454 BuildMI(entryMBB, DL, TII.get(X86::JCC_1))
1455 .addMBB(&MBB)
1458 entryMBB->addSuccessor(headMBB);
1459 entryMBB->addSuccessor(&MBB);
1460 }
1461
1462 // Loop entry block
1463
1464 {
1465 const unsigned SUBOpc = getSUBriOpcode(Uses64BitFramePtr);
1466 BuildMI(headMBB, DL, TII.get(SUBOpc), StackPtr)
1468 .addImm(StackProbeSize)
1470
1471 BuildMI(headMBB, DL,
1472 TII.get(Uses64BitFramePtr ? X86::CMP64rr : X86::CMP32rr))
1474 .addReg(FinalStackProbed)
1476
1477 // jump to the footer if StackPtr < FinalStackProbed
1478 BuildMI(headMBB, DL, TII.get(X86::JCC_1))
1479 .addMBB(footMBB)
1482
1483 headMBB->addSuccessor(bodyMBB);
1484 headMBB->addSuccessor(footMBB);
1485 }
1486
1487 // setup loop body
1488 {
1489 addRegOffset(BuildMI(bodyMBB, DL, TII.get(MovMIOpc))
1491 StackPtr, false, 0)
1492 .addImm(0)
1494
1495 const unsigned SUBOpc = getSUBriOpcode(Uses64BitFramePtr);
1496 BuildMI(bodyMBB, DL, TII.get(SUBOpc), StackPtr)
1498 .addImm(StackProbeSize)
1500
1501 // cmp with stack pointer bound
1502 BuildMI(bodyMBB, DL,
1503 TII.get(Uses64BitFramePtr ? X86::CMP64rr : X86::CMP32rr))
1504 .addReg(FinalStackProbed)
1507
1508 // jump back while FinalStackProbed < StackPtr
1509 BuildMI(bodyMBB, DL, TII.get(X86::JCC_1))
1510 .addMBB(bodyMBB)
1513 bodyMBB->addSuccessor(bodyMBB);
1514 bodyMBB->addSuccessor(footMBB);
1515 }
1516
1517 // setup loop footer
1518 {
1519 BuildMI(footMBB, DL, TII.get(TargetOpcode::COPY), StackPtr)
1520 .addReg(FinalStackProbed)
1522 addRegOffset(BuildMI(footMBB, DL, TII.get(MovMIOpc))
1524 StackPtr, false, 0)
1525 .addImm(0)
1527 footMBB->addSuccessor(&MBB);
1528 }
1529
1530 fullyRecomputeLiveIns({footMBB, bodyMBB, headMBB, &MBB});
1531 }
1532 } else {
1533 MachineInstr *MI = BuildMI(MBB, MBBI, DL, TII.get(AndOp), Reg)
1534 .addReg(Reg)
1535 .addImm(Val)
1537
1538 // The EFLAGS implicit def is dead.
1539 MI->getOperand(3).setIsDead();
1540 }
1541}
1542
1544 // x86-64 (non Win64) has a 128 byte red zone which is guaranteed not to be
1545 // clobbered by any interrupt handler.
1546 assert(&STI == &MF.getSubtarget<X86Subtarget>() &&
1547 "MF used frame lowering for wrong subtarget");
1548 const Function &Fn = MF.getFunction();
1549 const bool IsWin64CC = STI.isCallingConvWin64(Fn.getCallingConv());
1550 return Is64Bit && !IsWin64CC && !Fn.hasFnAttribute(Attribute::NoRedZone);
1551}
1552
1553/// Return true if we need to use the restricted Windows x64 prologue and
1554/// epilogue code patterns that can be described with WinCFI (.seh_*
1555/// directives).
1556bool X86FrameLowering::isWin64Prologue(const MachineFunction &MF) const {
1557 return MF.getTarget().getMCAsmInfo().usesWindowsCFI();
1558}
1559
1560bool X86FrameLowering::needsDwarfCFI(const MachineFunction &MF) const {
1561 return !isWin64Prologue(MF) && MF.needsFrameMoves();
1562}
1563
1564/// Return true if an opcode is part of the REP group of instructions
1565static bool isOpcodeRep(unsigned Opcode) {
1566 switch (Opcode) {
1567 case X86::REPNE_PREFIX:
1568 case X86::REP_MOVSB_32:
1569 case X86::REP_MOVSB_64:
1570 case X86::REP_MOVSD_32:
1571 case X86::REP_MOVSD_64:
1572 case X86::REP_MOVSQ_32:
1573 case X86::REP_MOVSQ_64:
1574 case X86::REP_MOVSW_32:
1575 case X86::REP_MOVSW_64:
1576 case X86::REP_PREFIX:
1577 case X86::REP_STOSB_32:
1578 case X86::REP_STOSB_64:
1579 case X86::REP_STOSD_32:
1580 case X86::REP_STOSD_64:
1581 case X86::REP_STOSQ_32:
1582 case X86::REP_STOSQ_64:
1583 case X86::REP_STOSW_32:
1584 case X86::REP_STOSW_64:
1585 return true;
1586 default:
1587 break;
1588 }
1589 return false;
1590}
1591
1592/// Returns the number of bytes between the end of the fixed and callee-save
1593/// area and the first local object, which PEI leaves as padding when it aligns
1594/// the local objects relative to the incoming stack pointer.
1596 int64_t FixedEnd = 0;
1597 int64_t LocalsTop = std::numeric_limits<int64_t>::max();
1598 for (int I : seq(MFI.getObjectIndexBegin(), MFI.getObjectIndexEnd())) {
1601 continue;
1602
1603 int64_t ObjOffset = MFI.getObjectOffset(I);
1604 int64_t ObjSize = MFI.getObjectSize(I);
1605 // Offsets are negative, measured from the incoming stack pointer.
1606 if (MFI.isFixedObjectIndex(I))
1607 FixedEnd = std::max(FixedEnd, -ObjOffset);
1608 else
1609 LocalsTop = std::min<int64_t>(LocalsTop, -ObjOffset - ObjSize);
1610 }
1611 if (LocalsTop == std::numeric_limits<int64_t>::max())
1612 return 0;
1613
1614 assert(LocalsTop >= FixedEnd && "Local object overlaps the fixed area");
1615 return LocalsTop - FixedEnd;
1616}
1617
1618/// emitPrologue - Push callee-saved registers onto the stack, which
1619/// automatically adjust the stack pointer. Adjust the stack pointer to allocate
1620/// space for local variables. Also emit labels used by the exception handler to
1621/// generate the exception handling frames.
1622
1623/*
1624 Here's a gist of what gets emitted:
1625
1626 ; Establish frame pointer, if needed
1627 [if needs FP]
1628 push %rbp
1629 .cfi_def_cfa_offset 16
1630 .cfi_offset %rbp, -16
1631 .seh_pushreg %rpb
1632 mov %rsp, %rbp
1633 .cfi_def_cfa_register %rbp
1634
1635 ; Spill general-purpose registers
1636 [for all callee-saved GPRs]
1637 pushq %<reg>
1638 [if not needs FP]
1639 .cfi_def_cfa_offset (offset from RETADDR)
1640 .seh_pushreg %<reg>
1641
1642 ; If the required stack alignment > default stack alignment
1643 ; rsp needs to be re-aligned. This creates a "re-alignment gap"
1644 ; of unknown size in the stack frame.
1645 [if stack needs re-alignment]
1646 and $MASK, %rsp
1647
1648 ; Allocate space for locals
1649 [if target is Windows and allocated space > 4096 bytes]
1650 ; Windows needs special care for allocations larger
1651 ; than one page.
1652 mov $NNN, %rax
1653 call ___chkstk_ms/___chkstk
1654 sub %rax, %rsp
1655 [else]
1656 sub $NNN, %rsp
1657
1658 [if needs FP]
1659 .seh_stackalloc (size of XMM spill slots)
1660 .seh_setframe %rbp, SEHFrameOffset ; = size of all spill slots
1661 [else]
1662 .seh_stackalloc NNN
1663
1664 ; Spill XMMs
1665 ; Note, that while only Windows 64 ABI specifies XMMs as callee-preserved,
1666 ; they may get spilled on any platform, if the current function
1667 ; calls @llvm.eh.unwind.init
1668 [if needs FP]
1669 [for all callee-saved XMM registers]
1670 movaps %<xmm reg>, -MMM(%rbp)
1671 [for all callee-saved XMM registers]
1672 .seh_savexmm %<xmm reg>, (-MMM + SEHFrameOffset)
1673 ; i.e. the offset relative to (%rbp - SEHFrameOffset)
1674 [else]
1675 [for all callee-saved XMM registers]
1676 movaps %<xmm reg>, KKK(%rsp)
1677 [for all callee-saved XMM registers]
1678 .seh_savexmm %<xmm reg>, KKK
1679
1680 .seh_endprologue
1681
1682 [if needs base pointer]
1683 mov %rsp, %rbx
1684 [if needs to restore base pointer]
1685 mov %rsp, -MMM(%rbp)
1686
1687 ; Emit CFI info
1688 [if needs FP]
1689 [for all callee-saved registers]
1690 .cfi_offset %<reg>, (offset from %rbp)
1691 [else]
1692 .cfi_def_cfa_offset (offset from RETADDR)
1693 [for all callee-saved registers]
1694 .cfi_offset %<reg>, (offset from %rsp)
1695
1696 Notes:
1697 - .seh directives are emitted only for Windows 64 ABI
1698 - .cv_fpo directives are emitted on win32 when emitting CodeView
1699 - .cfi directives are emitted for all other ABIs
1700 - for 32-bit code, substitute %e?? registers for %r??
1701*/
1702
1704 MachineBasicBlock &MBB) const {
1705 assert(&STI == &MF.getSubtarget<X86Subtarget>() &&
1706 "MF used frame lowering for wrong subtarget");
1708 MachineFrameInfo &MFI = MF.getFrameInfo();
1709 const Function &Fn = MF.getFunction();
1711 uint64_t MaxAlign = calculateMaxStackAlign(MF); // Desired stack alignment.
1712 uint64_t StackSize = MFI.getStackSize(); // Number of bytes to allocate.
1713 bool IsFunclet = MBB.isEHFuncletEntry();
1715 if (Fn.hasPersonalityFn())
1716 Personality = classifyEHPersonality(Fn.getPersonalityFn());
1717 bool FnHasClrFunclet =
1718 MF.hasEHFunclets() && Personality == EHPersonality::CoreCLR;
1719 bool IsClrFunclet = IsFunclet && FnHasClrFunclet;
1720 bool HasFP = hasFP(MF);
1721 bool IsWin64Prologue = isWin64Prologue(MF);
1722 bool NeedsWin64CFI = IsWin64Prologue && Fn.needsUnwindTableEntry();
1723 // FIXME: Emit FPO data for EH funclets.
1724 bool NeedsWinFPO = !IsFunclet && STI.isTargetWin32() &&
1726 bool NeedsWinCFI = NeedsWin64CFI || NeedsWinFPO;
1727 bool NeedsDwarfCFI = needsDwarfCFI(MF);
1728 bool IsWin64UnwindV3 = NeedsWin64CFI && requireWinX64UnwindV3(MF);
1729 Register FramePtr = TRI->getFrameRegister(MF);
1730 const Register MachineFramePtr =
1731 STI.isTarget64BitILP32() ? Register(getX86SubSuperRegister(FramePtr, 64))
1732 : FramePtr;
1733 Register BasePtr = TRI->getBaseRegister();
1734 bool HasWinCFI = false;
1735
1736 // Helpers to emit Windows x64 unwind SEH pseudos with the correct placement.
1737 // V1/V2: pseudo goes after the real instruction.
1738 // V3: pseudo goes before the real instruction.
1739 // Usage:
1740 // EmitSEHBefore([&]{ BuildMI(...SEH_PushReg...); });
1741 // BuildMI(... real instruction ...);
1742 // EmitSEHAfter([&]{ BuildMI(...SEH_PushReg...); });
1743 auto EmitSEHBefore = [&](auto EmitFn) {
1744 if (NeedsWinCFI && IsWin64UnwindV3) {
1745 HasWinCFI = true;
1746 EmitFn();
1747 }
1748 };
1749 auto EmitSEHAfter = [&](auto EmitFn) {
1750 if (NeedsWinCFI && !IsWin64UnwindV3) {
1751 HasWinCFI = true;
1752 EmitFn();
1753 }
1754 };
1755
1756 // Debug location must be unknown since the first debug location is used
1757 // to determine the end of the prologue.
1758 DebugLoc DL;
1759 Register ArgBaseReg;
1760
1761 // Emit extra prolog for argument stack slot reference.
1762 if (auto *MI = X86FI->getStackPtrSaveMI()) {
1763 // MI is lea instruction that created in X86ArgumentStackSlotPass.
1764 // Creat extra prolog for stack realignment.
1765 ArgBaseReg = MI->getOperand(0).getReg();
1766 // leal 4(%esp), %basereg
1767 // .cfi_def_cfa %basereg, 0
1768 // andl $-128, %esp
1769 // pushl -4(%basereg)
1770 BuildMI(MBB, MBBI, DL, TII.get(Is64Bit ? X86::LEA64r : X86::LEA32r),
1771 ArgBaseReg)
1773 .addImm(1)
1774 .addUse(X86::NoRegister)
1776 .addUse(X86::NoRegister)
1778 if (NeedsDwarfCFI) {
1779 // .cfi_def_cfa %basereg, 0
1780 unsigned DwarfStackPtr = TRI->getDwarfRegNum(ArgBaseReg, true);
1781 BuildCFI(MBB, MBBI, DL,
1782 MCCFIInstruction::cfiDefCfa(nullptr, DwarfStackPtr, 0),
1784 }
1785 BuildStackAlignAND(MBB, MBBI, DL, StackPtr, MaxAlign);
1786 int64_t Offset = -(int64_t)SlotSize;
1787 BuildMI(MBB, MBBI, DL, TII.get(Is64Bit ? X86::PUSH64rmm : X86::PUSH32rmm))
1788 .addReg(ArgBaseReg)
1789 .addImm(1)
1790 .addReg(X86::NoRegister)
1791 .addImm(Offset)
1792 .addReg(X86::NoRegister)
1794 }
1795
1796 // Space reserved for stack-based arguments when making a (ABI-guaranteed)
1797 // tail call.
1798 unsigned TailCallArgReserveSize = -X86FI->getTCReturnAddrDelta();
1799 if (TailCallArgReserveSize && IsWin64Prologue)
1800 report_fatal_error("Can't handle guaranteed tail call under win64 yet");
1801
1802 const bool EmitStackProbeCall =
1803 STI.getTargetLowering()->hasStackProbeSymbol(MF);
1804 unsigned StackProbeSize = STI.getTargetLowering()->getStackProbeSize(MF);
1805
1806 if (HasFP && X86FI->hasSwiftAsyncContext()) {
1809 if (STI.swiftAsyncContextIsDynamicallySet()) {
1810 // The special symbol below is absolute and has a *value* suitable to be
1811 // combined with the frame pointer directly.
1812 BuildMI(MBB, MBBI, DL, TII.get(X86::OR64rm), MachineFramePtr)
1813 .addUse(MachineFramePtr)
1814 .addUse(X86::RIP)
1815 .addImm(1)
1816 .addUse(X86::NoRegister)
1817 .addExternalSymbol("swift_async_extendedFramePointerFlags",
1819 .addUse(X86::NoRegister);
1820 break;
1821 }
1822 [[fallthrough]];
1823
1825 assert(
1826 !IsWin64Prologue &&
1827 "win64 prologue does not set the bit 60 in the saved frame pointer");
1828 BuildMI(MBB, MBBI, DL, TII.get(X86::BTS64ri8), MachineFramePtr)
1829 .addUse(MachineFramePtr)
1830 .addImm(60)
1832 break;
1833
1835 break;
1836 }
1837 }
1838
1839 // Re-align the stack on 64-bit if the x86-interrupt calling convention is
1840 // used and an error code was pushed, since the x86-64 ABI requires a 16-byte
1841 // stack alignment.
1843 Fn.arg_size() == 2) {
1844 // Update the stack pointer by pushing a register. This is the instruction
1845 // emitted that would be end up being emitted by a call to `emitSPUpdate`.
1846 // Hard-coding the update to a push avoids emitting a second
1847 // `STACKALLOC_W_PROBING` instruction in the save block: We know that stack
1848 // probing isn't needed anyways for an 8-byte update.
1849 // Pushing a register leaves us in a similar situation to a regular
1850 // function call where we know that the address at (rsp-8) is writeable.
1851 // That way we avoid any off-by-ones with stack probing for additional
1852 // stack pointer updates later on.
1853 BuildMI(MBB, MBBI, DL, TII.get(X86::PUSH64r))
1854 .addReg(X86::RAX, RegState::Undef)
1856 }
1857
1858 // If this is x86-64 and the Red Zone is not disabled, if we are a leaf
1859 // function, and use up to 128 bytes of stack space, don't have a frame
1860 // pointer, calls, or dynamic alloca then we do not need to adjust the
1861 // stack pointer (we fit in the Red Zone). We also check that we don't
1862 // push and pop from the stack.
1863 if (has128ByteRedZone(MF) && !TRI->hasStackRealignment(MF) &&
1864 !MFI.hasVarSizedObjects() && // No dynamic alloca.
1865 !MFI.adjustsStack() && // No calls.
1866 !EmitStackProbeCall && // No stack probes.
1867 !MFI.hasCopyImplyingStackAdjustment() && // Don't push and pop.
1868 !MF.shouldSplitStack()) { // Regular stack
1869 uint64_t MinSize =
1871 if (HasFP)
1872 MinSize += SlotSize;
1873 X86FI->setUsesRedZone(MinSize > 0 || StackSize > 0);
1874 StackSize = std::max(MinSize, StackSize > 128 ? StackSize - 128 : 0);
1875 MFI.setStackSize(StackSize);
1876 }
1877
1878 // Insert stack pointer adjustment for later moving of return addr. Only
1879 // applies to tail call optimized functions where the callee argument stack
1880 // size is bigger than the callers.
1881 if (TailCallArgReserveSize != 0) {
1882 BuildStackAdjustment(MBB, MBBI, DL, -(int)TailCallArgReserveSize,
1883 /*InEpilogue=*/false)
1884 .setMIFlag(MachineInstr::FrameSetup);
1885 }
1886
1887 // Mapping for machine moves:
1888 //
1889 // DST: VirtualFP AND
1890 // SRC: VirtualFP => DW_CFA_def_cfa_offset
1891 // ELSE => DW_CFA_def_cfa
1892 //
1893 // SRC: VirtualFP AND
1894 // DST: Register => DW_CFA_def_cfa_register
1895 //
1896 // ELSE
1897 // OFFSET < 0 => DW_CFA_offset_extended_sf
1898 // REG < 64 => DW_CFA_offset + Reg
1899 // ELSE => DW_CFA_offset_extended
1900
1901 uint64_t NumBytes = 0;
1902 int stackGrowth = -SlotSize;
1903
1904 // Find the funclet establisher parameter
1905 MCRegister Establisher;
1906 if (IsClrFunclet)
1907 Establisher = Uses64BitFramePtr ? X86::RCX : X86::ECX;
1908 else if (IsFunclet)
1909 Establisher = Uses64BitFramePtr ? X86::RDX : X86::EDX;
1910
1911 if (IsWin64Prologue && IsFunclet && !IsClrFunclet) {
1912 // Immediately spill establisher into the home slot.
1913 // The runtime cares about this.
1914 // MOV64mr %rdx, 16(%rsp)
1915 unsigned MOVmr = Uses64BitFramePtr ? X86::MOV64mr : X86::MOV32mr;
1916 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(MOVmr)), StackPtr, true, 16)
1917 .addReg(Establisher)
1919 MBB.addLiveIn(Establisher);
1920 }
1921
1922 if (HasFP) {
1923 assert(MF.getRegInfo().isReserved(MachineFramePtr) && "FP reserved");
1924
1925 // Calculate required stack adjustment.
1926 uint64_t FrameSize = StackSize - SlotSize;
1927 NumBytes =
1928 FrameSize - (X86FI->getCalleeSavedFrameSize() + TailCallArgReserveSize);
1929
1930 // Callee-saved registers are pushed on stack before the stack is realigned,
1931 // and the realignment itself already provides the local objects' alignment,
1932 // so leave out the padding PEI put in front of them for it.
1933 if (TRI->hasStackRealignment(MF) && !IsWin64Prologue) {
1934 uint64_t Padding = getUnusedLocalAreaPadding(MFI);
1935 assert(Padding <= NumBytes && "Padding exceeds the local area");
1936 NumBytes = alignTo(NumBytes - Padding, MaxAlign);
1937 }
1938
1939 // Save EBP/RBP into the appropriate stack slot.
1940 auto EmitSEHPushFramePtr = [&]() {
1941 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_PushReg))
1944 };
1945 EmitSEHBefore(EmitSEHPushFramePtr);
1946 BuildMI(MBB, MBBI, DL,
1948 .addReg(MachineFramePtr, RegState::Kill)
1950 EmitSEHAfter(EmitSEHPushFramePtr);
1951
1952 if (NeedsDwarfCFI && !ArgBaseReg.isValid()) {
1953 // Mark the place where EBP/RBP was saved.
1954 // Define the current CFA rule to use the provided offset.
1955 assert(StackSize);
1956 BuildCFI(MBB, MBBI, DL,
1958 nullptr, -2 * stackGrowth + (int)TailCallArgReserveSize),
1960
1961 // Change the rule for the FramePtr to be an "offset" rule.
1962 unsigned DwarfFramePtr = TRI->getDwarfRegNum(MachineFramePtr, true);
1963 BuildCFI(MBB, MBBI, DL,
1964 MCCFIInstruction::createOffset(nullptr, DwarfFramePtr,
1965 2 * stackGrowth -
1966 (int)TailCallArgReserveSize),
1968 }
1969
1970 if (!IsFunclet) {
1971 if (X86FI->hasSwiftAsyncContext()) {
1972 assert(!IsWin64Prologue &&
1973 "win64 prologue does not store async context right below rbp");
1974 const auto &Attrs = MF.getFunction().getAttributes();
1975
1976 // Before we update the live frame pointer we have to ensure there's a
1977 // valid (or null) asynchronous context in its slot just before FP in
1978 // the frame record, so store it now.
1979 auto EmitSEHPushR14 = [&]() {
1980 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_PushReg))
1981 .addImm(X86::R14)
1983 };
1984 EmitSEHBefore(EmitSEHPushR14);
1985 if (Attrs.hasAttrSomewhere(Attribute::SwiftAsync)) {
1986 // We have an initial context in r14, store it just before the frame
1987 // pointer.
1988 MBB.addLiveIn(X86::R14);
1989 BuildMI(MBB, MBBI, DL, TII.get(X86::PUSH64r))
1990 .addReg(X86::R14)
1992 } else {
1993 // No initial context, store null so that there's no pointer that
1994 // could be misused.
1995 BuildMI(MBB, MBBI, DL, TII.get(X86::PUSH64i32))
1996 .addImm(0)
1998 }
1999
2000 // Update CFA offset for the async-context push.
2001 if (NeedsDwarfCFI && !ArgBaseReg.isValid()) {
2002 BuildCFI(
2003 MBB, MBBI, DL,
2004 MCCFIInstruction::createAdjustCfaOffset(nullptr, -stackGrowth),
2006 }
2007
2008 EmitSEHAfter(EmitSEHPushR14);
2009
2010 BuildMI(MBB, MBBI, DL, TII.get(X86::LEA64r), FramePtr)
2011 .addUse(X86::RSP)
2012 .addImm(1)
2013 .addUse(X86::NoRegister)
2014 .addImm(8)
2015 .addUse(X86::NoRegister)
2017
2018 // Switch to an FP-relative CFA before adjusting RSP below.
2019 if (NeedsDwarfCFI && !ArgBaseReg.isValid()) {
2020 unsigned DwarfFramePtr = TRI->getDwarfRegNum(MachineFramePtr, true);
2021 BuildCFI(MBB, MBBI, DL,
2022 MCCFIInstruction::cfiDefCfa(nullptr, DwarfFramePtr,
2023 -2 * stackGrowth +
2024 (int)TailCallArgReserveSize),
2026 }
2027
2028 BuildMI(MBB, MBBI, DL, TII.get(X86::SUB64ri32), X86::RSP)
2029 .addUse(X86::RSP)
2030 .addImm(8)
2032 }
2033
2034 if (!IsWin64Prologue && !IsFunclet) {
2035 // Update EBP with the new base value.
2036 if (!X86FI->hasSwiftAsyncContext())
2037 BuildMI(MBB, MBBI, DL,
2038 TII.get(Uses64BitFramePtr ? X86::MOV64rr : X86::MOV32rr),
2039 FramePtr)
2042
2043 if (NeedsDwarfCFI) {
2044 if (ArgBaseReg.isValid()) {
2045 SmallString<64> CfaExpr;
2046 CfaExpr.push_back(dwarf::DW_CFA_expression);
2047 uint8_t buffer[16];
2048 unsigned DwarfReg = TRI->getDwarfRegNum(MachineFramePtr, true);
2049 CfaExpr.append(buffer, buffer + encodeULEB128(DwarfReg, buffer));
2050 CfaExpr.push_back(2);
2051 CfaExpr.push_back((uint8_t)(dwarf::DW_OP_breg0 + DwarfReg));
2052 CfaExpr.push_back(0);
2053 // DW_CFA_expression: reg5 DW_OP_breg5 +0
2054 BuildCFI(MBB, MBBI, DL,
2055 MCCFIInstruction::createEscape(nullptr, CfaExpr.str()),
2057 } else if (!X86FI->hasSwiftAsyncContext()) {
2058 // Mark effective beginning of when frame pointer becomes valid.
2059 // Define the current CFA to use the EBP/RBP register.
2060 unsigned DwarfFramePtr = TRI->getDwarfRegNum(MachineFramePtr, true);
2061 BuildCFI(
2062 MBB, MBBI, DL,
2063 MCCFIInstruction::createDefCfaRegister(nullptr, DwarfFramePtr),
2065 }
2066 }
2067
2068 if (NeedsWinFPO) {
2069 // .cv_fpo_setframe $FramePtr
2070 // NeedsWinFPO is Win32 only, so we're never using Unwind v3, hence it
2071 // is always inserted afterwards.
2072 assert(!IsWin64UnwindV3);
2073 HasWinCFI = true;
2074 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_SetFrame))
2076 .addImm(0)
2078 }
2079 }
2080 }
2081 } else {
2082 assert(!IsFunclet && "funclets without FPs not yet implemented");
2083 NumBytes =
2084 StackSize - (X86FI->getCalleeSavedFrameSize() + TailCallArgReserveSize);
2085 }
2086
2087 // Update the offset adjustment, which is mainly used by codeview to translate
2088 // from ESP to VFRAME relative local variable offsets.
2089 if (!IsFunclet) {
2090 if (HasFP && TRI->hasStackRealignment(MF))
2091 MFI.setOffsetAdjustment(-NumBytes);
2092 else
2093 MFI.setOffsetAdjustment(-StackSize);
2094 }
2095
2096 // For EH funclets, only allocate enough space for outgoing calls. Save the
2097 // NumBytes value that we would've used for the parent frame.
2098 unsigned ParentFrameNumBytes = NumBytes;
2099 if (IsFunclet)
2100 NumBytes = getWinEHFuncletFrameSize(MF);
2101
2102 // Skip the callee-saved push instructions.
2103 bool PushedRegs = false;
2104 int StackOffset = 2 * stackGrowth;
2106 auto IsCSPush = [&](const MachineBasicBlock::iterator &MBBI) {
2107 if (MBBI == MBB.end() || !MBBI->getFlag(MachineInstr::FrameSetup))
2108 return false;
2109 unsigned Opc = MBBI->getOpcode();
2110 return Opc == X86::PUSH32r || Opc == X86::PUSH64r || Opc == X86::PUSHP64r ||
2111 Opc == X86::PUSH2 || Opc == X86::PUSH2P;
2112 };
2113
2114 while (IsCSPush(MBBI)) {
2115 PushedRegs = true;
2116 Register Reg = MBBI->getOperand(0).getReg();
2117 LastCSPush = MBBI;
2118 unsigned Opc = LastCSPush->getOpcode();
2119 bool IsPush2 = Opc == X86::PUSH2 || Opc == X86::PUSH2P;
2120
2121 // V3: emit SEH pseudo before the real instruction.
2122 EmitSEHBefore([&]() {
2123 if (IsPush2) {
2124 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_Push2Regs))
2125 .addImm(Reg)
2126 .addImm(LastCSPush->getOperand(1).getReg())
2128 } else {
2129 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_PushReg))
2130 .addImm(Reg)
2132 }
2133 });
2134 ++MBBI;
2135
2136 if (!HasFP && NeedsDwarfCFI) {
2137 // Mark callee-saved push instruction.
2138 // Define the current CFA rule to use the provided offset.
2139 assert(StackSize);
2140 // Compared to push, push2 introduces more stack offset (one more
2141 // register).
2142 if (IsPush2)
2143 StackOffset += stackGrowth;
2144 BuildCFI(MBB, MBBI, DL,
2147 StackOffset += stackGrowth;
2148 }
2149
2150 // V1/V2: emit SEH pseudo after the real instruction.
2151 EmitSEHAfter([&]() {
2152 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_PushReg))
2153 .addImm(Reg)
2155 if (IsPush2)
2156 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_PushReg))
2157 .addImm(LastCSPush->getOperand(1).getReg())
2159 });
2160 }
2161
2162 // Realign stack after we pushed callee-saved registers (so that we'll be
2163 // able to calculate their offsets from the frame pointer).
2164 // Don't do this for Win64, it needs to realign the stack after the prologue.
2165 if (!IsWin64Prologue && !IsFunclet && TRI->hasStackRealignment(MF) &&
2166 !ArgBaseReg.isValid()) {
2167 assert(HasFP && "There should be a frame pointer if stack is realigned.");
2168 auto EmitSEHStackAlign = [&]() {
2169 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_StackAlign))
2170 .addImm(MaxAlign)
2172 };
2173 EmitSEHBefore(EmitSEHStackAlign);
2174 BuildStackAlignAND(MBB, MBBI, DL, StackPtr, MaxAlign);
2175 EmitSEHAfter(EmitSEHStackAlign);
2176 }
2177
2178 // If there is an SUB32ri of ESP immediately before this instruction, merge
2179 // the two. This can be the case when tail call elimination is enabled and
2180 // the callee has more arguments than the caller.
2181 NumBytes = mergeSPUpdates(
2182 MBB, MBBI, [NumBytes](int64_t Offset) { return NumBytes - Offset; },
2183 true);
2184
2185 // Adjust stack pointer: ESP -= numbytes.
2186
2187 // Windows and cygwin/mingw require a prologue helper routine when allocating
2188 // more than 4K bytes on the stack. Windows uses __chkstk and cygwin/mingw
2189 // uses __alloca. __alloca and the 32-bit version of __chkstk will probe the
2190 // stack and adjust the stack pointer in one go. The 64-bit version of
2191 // __chkstk is only responsible for probing the stack. The 64-bit prologue is
2192 // responsible for adjusting the stack pointer. Touching the stack at 4K
2193 // increments is necessary to ensure that the guard pages used by the OS
2194 // virtual memory manager are allocated in correct sequence.
2195 uint64_t AlignedNumBytes = NumBytes;
2196 if (IsWin64Prologue && !IsFunclet && TRI->hasStackRealignment(MF))
2197 AlignedNumBytes = alignTo(AlignedNumBytes, MaxAlign);
2198
2199 auto EmitSEHStackAlloc = [&]() {
2200 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_StackAlloc))
2201 .addImm(NumBytes)
2203 };
2204 if (NumBytes)
2205 EmitSEHBefore(EmitSEHStackAlloc);
2206
2207 if (AlignedNumBytes >= StackProbeSize && EmitStackProbeCall) {
2208 assert(!X86FI->getUsesRedZone() &&
2209 "The Red Zone is not accounted for in stack probes");
2210
2211 // Check whether EAX is livein for this block.
2212 bool isEAXAlive = isEAXLiveIn(MBB);
2213
2214 if (isEAXAlive) {
2215 if (Is64Bit) {
2216 // Save RAX
2217 BuildMI(MBB, MBBI, DL, TII.get(X86::PUSH64r))
2218 .addReg(X86::RAX, RegState::Kill)
2220 } else {
2221 // Save EAX
2222 BuildMI(MBB, MBBI, DL, TII.get(X86::PUSH32r))
2223 .addReg(X86::EAX, RegState::Kill)
2225 }
2226 }
2227
2228 if (Is64Bit) {
2229 // Handle the 64-bit Windows ABI case where we need to call __chkstk.
2230 // Function prologue is responsible for adjusting the stack pointer.
2231 int64_t Alloc = isEAXAlive ? NumBytes - 8 : NumBytes;
2233 X86::RAX)
2234 .addImm(Alloc)
2236 } else {
2237 // Allocate NumBytes-4 bytes on stack in case of isEAXAlive.
2238 // We'll also use 4 already allocated bytes for EAX.
2239 BuildMI(MBB, MBBI, DL, TII.get(X86::MOV32ri), X86::EAX)
2240 .addImm(isEAXAlive ? NumBytes - 4 : NumBytes)
2242 }
2243
2244 // Call __chkstk, __chkstk_ms, or __alloca.
2245 emitStackProbe(MF, MBB, MBBI, DL, true);
2246
2247 if (isEAXAlive) {
2248 // Restore RAX/EAX
2250 if (Is64Bit)
2251 MI = addRegOffset(BuildMI(MF, DL, TII.get(X86::MOV64rm), X86::RAX),
2252 StackPtr, false, NumBytes - 8);
2253 else
2254 MI = addRegOffset(BuildMI(MF, DL, TII.get(X86::MOV32rm), X86::EAX),
2255 StackPtr, false, NumBytes - 4);
2256 MI->setFlag(MachineInstr::FrameSetup);
2257 MBB.insert(MBBI, MI);
2258 }
2259 } else if (NumBytes) {
2260 emitSPUpdate(MBB, MBBI, DL, -(int64_t)NumBytes, /*InEpilogue=*/false);
2261 }
2262
2263 if (NumBytes)
2264 EmitSEHAfter(EmitSEHStackAlloc);
2265
2266 int SEHFrameOffset = 0;
2267 Register SPOrEstablisher;
2268 if (IsFunclet) {
2269 if (IsClrFunclet) {
2270 // The establisher parameter passed to a CLR funclet is actually a pointer
2271 // to the (mostly empty) frame of its nearest enclosing funclet; we have
2272 // to find the root function establisher frame by loading the PSPSym from
2273 // the intermediate frame.
2274 unsigned PSPSlotOffset = getPSPSlotOffsetFromSP(MF);
2275 MachinePointerInfo NoInfo;
2276 MBB.addLiveIn(Establisher);
2277 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::MOV64rm), Establisher),
2278 Establisher, false, PSPSlotOffset)
2281 ;
2282 // Save the root establisher back into the current funclet's (mostly
2283 // empty) frame, in case a sub-funclet or the GC needs it.
2284 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::MOV64mr)), StackPtr,
2285 false, PSPSlotOffset)
2286 .addReg(Establisher)
2288 NoInfo,
2291 }
2292 SPOrEstablisher = Establisher;
2293 } else {
2294 SPOrEstablisher = StackPtr;
2295 }
2296
2297 if (IsWin64Prologue && HasFP) {
2298 // Set RBP to a small fixed offset from RSP. In the funclet case, we base
2299 // this calculation on the incoming establisher, which holds the value of
2300 // RSP from the parent frame at the end of the prologue.
2301 SEHFrameOffset = calculateSetFPREG(ParentFrameNumBytes);
2302
2303 // If this is not a funclet, emit the CFI describing our frame pointer.
2304 if (NeedsWinCFI && !IsFunclet) {
2305 assert(!NeedsWinFPO && "this setframe incompatible with FPO data");
2306 HasWinCFI = true;
2307 if (isAsynchronousEHPersonality(Personality) || MF.hasEHFunclets()) {
2308 if (TRI->hasBasePointer(MF))
2311 else
2312 MF.getWinEHFuncInfo()->SEHSetFrameOffset = SEHFrameOffset;
2313 }
2314 }
2315
2316 auto EmitSEHSetFrame = [&]() {
2317 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_SetFrame))
2319 .addImm(SEHFrameOffset)
2321 };
2322
2323 if (!IsFunclet)
2324 EmitSEHBefore(EmitSEHSetFrame);
2325
2326 if (SEHFrameOffset)
2327 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::LEA64r), FramePtr),
2328 SPOrEstablisher, false, SEHFrameOffset);
2329 else
2330 BuildMI(MBB, MBBI, DL, TII.get(X86::MOV64rr), FramePtr)
2331 .addReg(SPOrEstablisher);
2332
2333 if (!IsFunclet)
2334 EmitSEHAfter(EmitSEHSetFrame);
2335 } else if (IsFunclet && STI.is32Bit()) {
2336 // Reset EBP / ESI to something good for funclets.
2338 // If we're a catch funclet, we can be returned to via catchret. Save ESP
2339 // into the registration node so that the runtime will restore it for us.
2340 if (!MBB.isCleanupFuncletEntry()) {
2341 assert(Personality == EHPersonality::MSVC_CXX);
2342 Register FrameReg;
2344 int64_t EHRegOffset = getFrameIndexReference(MF, FI, FrameReg).getFixed();
2345 // ESP is the first field, so no extra displacement is needed.
2346 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::MOV32mr)), FrameReg,
2347 false, EHRegOffset)
2348 .addReg(X86::ESP);
2349 }
2350 }
2351
2352 while (MBBI != MBB.end() && MBBI->getFlag(MachineInstr::FrameSetup)) {
2353 const MachineInstr &FrameInstr = *MBBI;
2354
2355 if (NeedsWinCFI) {
2356 int FI;
2357 if (Register Reg = TII.isStoreToStackSlot(FrameInstr, FI)) {
2358 if (X86::FR64RegClass.contains(Reg)) {
2359 int Offset;
2360 Register IgnoredFrameReg;
2361 if (IsWin64Prologue && IsFunclet)
2362 Offset = getWin64EHFrameIndexRef(MF, FI, IgnoredFrameReg);
2363 else
2364 Offset =
2365 getFrameIndexReference(MF, FI, IgnoredFrameReg).getFixed() +
2366 SEHFrameOffset;
2367
2368 assert(!NeedsWinFPO && "SEH_SaveXMM incompatible with FPO data");
2369 auto EmitSEHSaveXMM = [&]() {
2370 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_SaveXMM))
2371 .addImm(Reg)
2372 .addImm(Offset)
2374 };
2375 EmitSEHBefore(EmitSEHSaveXMM);
2376 ++MBBI;
2377 EmitSEHAfter(EmitSEHSaveXMM);
2378 continue;
2379 }
2380 }
2381 }
2382 ++MBBI;
2383 }
2384
2385 if (NeedsWinCFI && HasWinCFI) {
2386 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_EndPrologue))
2388 }
2389
2390 if (FnHasClrFunclet && !IsFunclet) {
2391 // Save the so-called Initial-SP (i.e. the value of the stack pointer
2392 // immediately after the prolog) into the PSPSlot so that funclets
2393 // and the GC can recover it.
2394 unsigned PSPSlotOffset = getPSPSlotOffsetFromSP(MF);
2395 auto PSPInfo = MachinePointerInfo::getFixedStack(
2397 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::MOV64mr)), StackPtr, false,
2398 PSPSlotOffset)
2403 }
2404
2405 // Realign stack after we spilled callee-saved registers (so that we'll be
2406 // able to calculate their offsets from the frame pointer).
2407 // Win64 requires aligning the stack after the prologue.
2408 if (IsWin64Prologue && TRI->hasStackRealignment(MF)) {
2409 assert(HasFP && "There should be a frame pointer if stack is realigned.");
2410 BuildStackAlignAND(MBB, MBBI, DL, SPOrEstablisher, MaxAlign);
2411 }
2412
2413 // We already dealt with stack realignment and funclets above.
2414 if (IsFunclet && STI.is32Bit())
2415 return;
2416
2417 // If we need a base pointer, set it up here. It's whatever the value
2418 // of the stack pointer is at this point. Any variable size objects
2419 // will be allocated after this, so we can still use the base pointer
2420 // to reference locals.
2421 if (TRI->hasBasePointer(MF)) {
2422 // Update the base pointer with the current stack pointer.
2423 unsigned Opc = Uses64BitFramePtr ? X86::MOV64rr : X86::MOV32rr;
2424 BuildMI(MBB, MBBI, DL, TII.get(Opc), BasePtr)
2425 .addReg(SPOrEstablisher)
2427 if (X86FI->getRestoreBasePointer()) {
2428 // Stash value of base pointer. Saving RSP instead of EBP shortens
2429 // dependence chain. Used by SjLj EH.
2430 unsigned Opm = Uses64BitFramePtr ? X86::MOV64mr : X86::MOV32mr;
2431 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(Opm)), FramePtr, true,
2433 .addReg(SPOrEstablisher)
2435 }
2436
2437 if (X86FI->getHasSEHFramePtrSave() && !IsFunclet) {
2438 // Stash the value of the frame pointer relative to the base pointer for
2439 // Win32 EH. This supports Win32 EH, which does the inverse of the above:
2440 // it recovers the frame pointer from the base pointer rather than the
2441 // other way around.
2442 unsigned Opm = Uses64BitFramePtr ? X86::MOV64mr : X86::MOV32mr;
2443 Register UsedReg;
2444 int Offset =
2445 getFrameIndexReference(MF, X86FI->getSEHFramePtrSaveIndex(), UsedReg)
2446 .getFixed();
2447 assert(UsedReg == BasePtr);
2448 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(Opm)), UsedReg, true, Offset)
2451 }
2452 }
2453 if (ArgBaseReg.isValid()) {
2454 // Save argument base pointer.
2455 auto *MI = X86FI->getStackPtrSaveMI();
2456 int FI = MI->getOperand(1).getIndex();
2457 unsigned MOVmr = Is64Bit ? X86::MOV64mr : X86::MOV32mr;
2458 // movl %basereg, offset(%ebp)
2459 addFrameReference(BuildMI(MBB, MBBI, DL, TII.get(MOVmr)), FI)
2460 .addReg(ArgBaseReg)
2462 }
2463
2464 if (((!HasFP && NumBytes) || PushedRegs) && NeedsDwarfCFI) {
2465 // Mark end of stack pointer adjustment.
2466 if (!HasFP && NumBytes) {
2467 // Define the current CFA rule to use the provided offset.
2468 assert(StackSize);
2469 BuildCFI(
2470 MBB, MBBI, DL,
2471 MCCFIInstruction::cfiDefCfaOffset(nullptr, StackSize - stackGrowth),
2473 }
2474
2475 // Emit DWARF info specifying the offsets of the callee-saved registers.
2477 }
2478
2479 // X86 Interrupt handling function cannot assume anything about the direction
2480 // flag (DF in EFLAGS register). Clear this flag by creating "cld" instruction
2481 // in each prologue of interrupt handler function.
2482 //
2483 // Create "cld" instruction only in these cases:
2484 // 1. The interrupt handling function uses any of the "rep" instructions.
2485 // 2. Interrupt handling function calls another function.
2486 // 3. If there are any inline asm blocks, as we do not know what they do
2487 //
2488 // TODO: We should also emit cld if we detect the use of std, but as of now,
2489 // the compiler does not even emit that instruction or even define it, so in
2490 // practice, this would only happen with inline asm, which we cover anyway.
2492 bool NeedsCLD = false;
2493
2494 for (const MachineBasicBlock &B : MF) {
2495 for (const MachineInstr &MI : B) {
2496 if (MI.isCall()) {
2497 NeedsCLD = true;
2498 break;
2499 }
2500
2501 if (isOpcodeRep(MI.getOpcode())) {
2502 NeedsCLD = true;
2503 break;
2504 }
2505
2506 if (MI.isInlineAsm()) {
2507 // TODO: Parse asm for rep instructions or call sites?
2508 // For now, let's play it safe and emit a cld instruction
2509 // just in case.
2510 NeedsCLD = true;
2511 break;
2512 }
2513 }
2514 }
2515
2516 if (NeedsCLD) {
2517 BuildMI(MBB, MBBI, DL, TII.get(X86::CLD))
2519 }
2520 }
2521
2522 // At this point we know if the function has WinCFI or not.
2523 MF.setHasWinCFI(HasWinCFI);
2524}
2525
2527 const MachineFunction &MF) const {
2528 // We can't use LEA instructions for adjusting the stack pointer if we don't
2529 // have a frame pointer in the Win64 ABI. Only ADD instructions may be used
2530 // to deallocate the stack.
2531 // This means that we can use LEA for SP in two situations:
2532 // 1. We *aren't* using the Win64 ABI which means we are free to use LEA.
2533 // 2. We *have* a frame pointer which means we are permitted to use LEA.
2534 return !MF.getTarget().getMCAsmInfo().usesWindowsCFI() || hasFP(MF);
2535}
2536
2538 switch (MI.getOpcode()) {
2539 case X86::CATCHRET:
2540 case X86::CLEANUPRET:
2541 return true;
2542 default:
2543 return false;
2544 }
2545 llvm_unreachable("impossible");
2546}
2547
2548// CLR funclets use a special "Previous Stack Pointer Symbol" slot on the
2549// stack. It holds a pointer to the bottom of the root function frame. The
2550// establisher frame pointer passed to a nested funclet may point to the
2551// (mostly empty) frame of its parent funclet, but it will need to find
2552// the frame of the root function to access locals. To facilitate this,
2553// every funclet copies the pointer to the bottom of the root function
2554// frame into a PSPSym slot in its own (mostly empty) stack frame. Using the
2555// same offset for the PSPSym in the root function frame that's used in the
2556// funclets' frames allows each funclet to dynamically accept any ancestor
2557// frame as its establisher argument (the runtime doesn't guarantee the
2558// immediate parent for some reason lost to history), and also allows the GC,
2559// which uses the PSPSym for some bookkeeping, to find it in any funclet's
2560// frame with only a single offset reported for the entire method.
2561unsigned
2562X86FrameLowering::getPSPSlotOffsetFromSP(const MachineFunction &MF) const {
2563 const WinEHFuncInfo &Info = *MF.getWinEHFuncInfo();
2565 int Offset = getFrameIndexReferencePreferSP(MF, Info.PSPSymFrameIdx, SPReg,
2566 /*IgnoreSPUpdates*/ true)
2567 .getFixed();
2568 assert(Offset >= 0 && SPReg == TRI->getStackRegister());
2569 return static_cast<unsigned>(Offset);
2570}
2571
2572unsigned
2573X86FrameLowering::getWinEHFuncletFrameSize(const MachineFunction &MF) const {
2574 const X86MachineFunctionInfo *X86FI = MF.getInfo<X86MachineFunctionInfo>();
2575 // This is the size of the pushed CSRs.
2576 unsigned CSSize = X86FI->getCalleeSavedFrameSize();
2577 // This is the size of callee saved XMMs.
2578 const auto &WinEHXMMSlotInfo = X86FI->getWinEHXMMSlotInfo();
2579 unsigned XMMSize =
2580 WinEHXMMSlotInfo.size() * TRI->getSpillSize(X86::VR128RegClass);
2581 // This is the amount of stack a funclet needs to allocate.
2582 unsigned UsedSize;
2583 EHPersonality Personality =
2585 if (Personality == EHPersonality::CoreCLR) {
2586 // CLR funclets need to hold enough space to include the PSPSym, at the
2587 // same offset from the stack pointer (immediately after the prolog) as it
2588 // resides at in the main function.
2589 UsedSize = getPSPSlotOffsetFromSP(MF) + SlotSize;
2590 } else {
2591 // Other funclets just need enough stack for outgoing call arguments.
2592 UsedSize = MF.getFrameInfo().getMaxCallFrameSize();
2593 }
2594 // RBP is not included in the callee saved register block. After pushing RBP,
2595 // everything is 16 byte aligned. Everything we allocate before an outgoing
2596 // call must also be 16 byte aligned.
2597 unsigned FrameSizeMinusRBP = alignTo(CSSize + UsedSize, getStackAlign());
2598 // Subtract out the size of the callee saved registers. This is how much stack
2599 // each funclet will allocate.
2600 return FrameSizeMinusRBP + XMMSize - CSSize;
2601}
2602
2603static bool isTailCallOpcode(unsigned Opc) {
2604 return Opc == X86::TCRETURNri || Opc == X86::TCRETURN_WIN64ri ||
2605 Opc == X86::TCRETURN_HIPE32ri || Opc == X86::TCRETURNdi ||
2606 Opc == X86::TCRETURNmi || Opc == X86::TCRETURNri64 ||
2607 Opc == X86::TCRETURNri64_ImpCall || Opc == X86::TCRETURNdi64 ||
2608 Opc == X86::TCRETURNmi64 || Opc == X86::TCRETURN_WINmi64;
2609}
2610
2612 MachineBasicBlock &MBB) const {
2613 const MachineFrameInfo &MFI = MF.getFrameInfo();
2615 MachineBasicBlock::iterator Terminator = MBB.getFirstTerminator();
2616 MachineBasicBlock::iterator MBBI = Terminator;
2617 DebugLoc DL;
2618 if (MBBI != MBB.end())
2619 DL = MBBI->getDebugLoc();
2620 // standard x86_64 uses 64-bit frame/stack pointers, x32 - 32-bit.
2621 const bool Is64BitILP32 = STI.isTarget64BitILP32();
2622 Register FramePtr = TRI->getFrameRegister(MF);
2623 Register MachineFramePtr =
2624 Is64BitILP32 ? Register(getX86SubSuperRegister(FramePtr, 64)) : FramePtr;
2625
2626 bool IsWin64Prologue = MF.getTarget().getMCAsmInfo().usesWindowsCFI();
2627 bool NeedsWin64CFI =
2628 IsWin64Prologue && MF.getFunction().needsUnwindTableEntry();
2629 // For V3 unwind, epilog SEH pseudos are emitted inline before each
2630 // unwind-effecting instruction.
2631 bool IsWin64UnwindV3 =
2632 NeedsWin64CFI && MF.hasWinCFI() && requireWinX64UnwindV3(MF);
2633 bool IsFunclet = MBBI == MBB.end() ? false : isFuncletReturnInstr(*MBBI);
2634
2635 // Get the number of bytes to allocate from the FrameInfo.
2636 uint64_t StackSize = MFI.getStackSize();
2637 uint64_t MaxAlign = calculateMaxStackAlign(MF);
2638 unsigned CSSize = X86FI->getCalleeSavedFrameSize();
2639 unsigned TailCallArgReserveSize = -X86FI->getTCReturnAddrDelta();
2640 bool HasFP = hasFP(MF);
2641 uint64_t NumBytes = 0;
2642
2643 bool NeedsDwarfCFI = (!MF.getTarget().getTargetTriple().isOSDarwin() &&
2645 !MF.getTarget().getTargetTriple().isUEFI()) &&
2646 MF.needsFrameMoves();
2647
2648 Register ArgBaseReg;
2649 if (auto *MI = X86FI->getStackPtrSaveMI()) {
2650 unsigned Opc = X86::LEA32r;
2651 Register StackReg = X86::ESP;
2652 ArgBaseReg = MI->getOperand(0).getReg();
2653 if (STI.is64Bit()) {
2654 Opc = X86::LEA64r;
2655 StackReg = X86::RSP;
2656 }
2657 // leal -4(%basereg), %esp
2658 // .cfi_def_cfa %esp, 4
2659 BuildMI(MBB, MBBI, DL, TII.get(Opc), StackReg)
2660 .addUse(ArgBaseReg)
2661 .addImm(1)
2662 .addUse(X86::NoRegister)
2663 .addImm(-(int64_t)SlotSize)
2664 .addUse(X86::NoRegister)
2666 if (NeedsDwarfCFI) {
2667 unsigned DwarfStackPtr = TRI->getDwarfRegNum(StackReg, true);
2668 BuildCFI(MBB, MBBI, DL,
2669 MCCFIInstruction::cfiDefCfa(nullptr, DwarfStackPtr, SlotSize),
2671 --MBBI;
2672 }
2673 --MBBI;
2674 }
2675
2676 if (IsFunclet) {
2677 assert(HasFP && "EH funclets without FP not yet implemented");
2678 NumBytes = getWinEHFuncletFrameSize(MF);
2679 } else if (HasFP) {
2680 // Calculate required stack adjustment.
2681 uint64_t FrameSize = StackSize - SlotSize;
2682 NumBytes = FrameSize - CSSize - TailCallArgReserveSize;
2683
2684 // Callee-saved registers were pushed on stack before the stack was
2685 // realigned.
2686 if (TRI->hasStackRealignment(MF) && !IsWin64Prologue)
2687 NumBytes = alignTo(FrameSize, MaxAlign);
2688 } else {
2689 NumBytes = StackSize - CSSize - TailCallArgReserveSize;
2690 }
2691 uint64_t SEHStackAllocAmt = NumBytes;
2692
2693 unsigned SEHFrameOffset = 0;
2694 if (IsWin64Prologue && HasFP)
2695 SEHFrameOffset = calculateSetFPREG(SEHStackAllocAmt);
2696
2697 // AfterPop is the position to insert .cfi_restore.
2699 if (HasFP) {
2700 if (X86FI->hasSwiftAsyncContext()) {
2701 // Discard the context.
2702 int64_t Offset = mergeSPAdd(MBB, MBBI, 16, true);
2703 emitSPUpdate(MBB, MBBI, DL, Offset, /*InEpilogue*/ true);
2704 }
2705 // Pop EBP.
2706 if (IsWin64UnwindV3)
2707 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_PushReg))
2710 BuildMI(MBB, MBBI, DL,
2712 MachineFramePtr)
2714
2715 // We need to reset FP to its untagged state on return. Bit 60 is currently
2716 // used to show the presence of an extended frame.
2717 if (X86FI->hasSwiftAsyncContext()) {
2718 BuildMI(MBB, MBBI, DL, TII.get(X86::BTR64ri8), MachineFramePtr)
2719 .addUse(MachineFramePtr)
2720 .addImm(60)
2722 }
2723
2724 if (NeedsDwarfCFI) {
2725 if (!ArgBaseReg.isValid()) {
2726 unsigned DwarfStackPtr =
2727 TRI->getDwarfRegNum(Is64Bit ? X86::RSP : X86::ESP, true);
2728 BuildCFI(MBB, MBBI, DL,
2729 MCCFIInstruction::cfiDefCfa(nullptr, DwarfStackPtr, SlotSize),
2731 }
2732 if (!MBB.succ_empty() && !MBB.isReturnBlock()) {
2733 unsigned DwarfFramePtr = TRI->getDwarfRegNum(MachineFramePtr, true);
2734 BuildCFI(MBB, AfterPop, DL,
2735 MCCFIInstruction::createRestore(nullptr, DwarfFramePtr),
2737 --MBBI;
2738 --AfterPop;
2739 }
2740 --MBBI;
2741 }
2742 }
2743
2744 MachineBasicBlock::iterator FirstCSPop = MBBI;
2745 // Skip the callee-saved pop instructions.
2746 while (MBBI != MBB.begin()) {
2747 MachineBasicBlock::iterator PI = std::prev(MBBI);
2748 unsigned Opc = PI->getOpcode();
2749
2750 if (Opc != X86::DBG_VALUE && !PI->isTerminator()) {
2751 if (!PI->getFlag(MachineInstr::FrameDestroy) ||
2752 (Opc != X86::POP32r && Opc != X86::POP64r && Opc != X86::BTR64ri8 &&
2753 Opc != X86::ADD64ri32 && Opc != X86::POPP64r && Opc != X86::POP2 &&
2754 Opc != X86::POP2P && Opc != X86::LEA64r && Opc != X86::SEH_PushReg &&
2755 Opc != X86::SEH_Push2Regs && Opc != X86::SEH_StackAlloc &&
2756 Opc != X86::ADD64ri32_NF))
2757 break;
2758 FirstCSPop = PI;
2759 }
2760
2761 --MBBI;
2762 }
2763 if (ArgBaseReg.isValid()) {
2764 // Restore argument base pointer.
2765 auto *MI = X86FI->getStackPtrSaveMI();
2766 int FI = MI->getOperand(1).getIndex();
2767 unsigned MOVrm = Is64Bit ? X86::MOV64rm : X86::MOV32rm;
2768 // movl offset(%ebp), %basereg
2769 addFrameReference(BuildMI(MBB, MBBI, DL, TII.get(MOVrm), ArgBaseReg), FI)
2771 }
2772 MBBI = FirstCSPop;
2773
2774 if (IsFunclet && Terminator->getOpcode() == X86::CATCHRET)
2775 emitCatchRetReturnValue(MBB, FirstCSPop, &*Terminator);
2776
2777 if (MBBI != MBB.end())
2778 DL = MBBI->getDebugLoc();
2779 // If there is an ADD32ri or SUB32ri of ESP immediately before this
2780 // instruction, merge the two instructions.
2781 if (NumBytes || MFI.hasVarSizedObjects())
2782 NumBytes = mergeSPAdd(MBB, MBBI, NumBytes, true);
2783
2784 if (IsWin64UnwindV3 && NeedsWin64CFI && MF.hasWinCFI()) {
2785 // Find the XMM restores that were tagged with FrameDestroy, now that we
2786 // know the offset we can emit the SEH pseudos for them.
2787 auto EpilogStart = MBBI;
2788 {
2789 auto ScanIt = MBBI;
2790 while (ScanIt != MBB.begin()) {
2791 auto PI = std::prev(ScanIt);
2792 int FI;
2793 if (PI->getFlag(MachineInstr::FrameDestroy) &&
2794 TII.isLoadFromStackSlot(*PI, FI)) {
2795 Register Reg = PI->getOperand(0).getReg();
2796 if (X86::FR64RegClass.contains(Reg)) {
2797 Register IgnoredFrameReg;
2798 int Offset =
2799 getFrameIndexReference(MF, FI, IgnoredFrameReg).getFixed() +
2800 SEHFrameOffset;
2801 BuildMI(MBB, PI, DL, TII.get(X86::SEH_SaveXMM))
2802 .addImm(Reg)
2803 .addImm(Offset)
2805 // std::prev(PI) is the SEH_SaveXMM we just inserted (before PI).
2806 // We start ScanIt from that point so that the next
2807 // std::prev(ScanIt) will examine the instruction before the pseudo,
2808 // i.e. the next potential XMM restore further up the block.
2809 EpilogStart = std::prev(PI);
2810 ScanIt = EpilogStart;
2811 continue;
2812 }
2813 }
2814 break;
2815 }
2816 }
2817
2818 // For V3, SEH_BeginEpilogue must be emitted before any epilog SEH pseudos.
2819 BuildMI(MBB, EpilogStart, DL, TII.get(X86::SEH_BeginEpilogue))
2821 }
2822
2823 // If dynamic alloca is used, then reset esp to point to the last callee-saved
2824 // slot before popping them off! Same applies for the case, when stack was
2825 // realigned. Don't do this if this was a funclet epilogue, since the funclets
2826 // will not do realignment or dynamic stack allocation.
2827 if (((TRI->hasStackRealignment(MF)) || MFI.hasVarSizedObjects()) &&
2828 !IsFunclet) {
2829 if (TRI->hasStackRealignment(MF))
2830 MBBI = FirstCSPop;
2831 uint64_t LEAAmount =
2832 IsWin64Prologue ? SEHStackAllocAmt - SEHFrameOffset : -CSSize;
2833
2834 if (X86FI->hasSwiftAsyncContext())
2835 LEAAmount -= 16;
2836
2837 // There are only two legal forms of epilogue:
2838 // - add SEHAllocationSize, %rsp
2839 // - lea SEHAllocationSize(%FramePtr), %rsp
2840 //
2841 // 'mov %FramePtr, %rsp' will not be recognized as an epilogue sequence.
2842 // However, we may use this sequence if we have a frame pointer because the
2843 // effects of the prologue can safely be undone.
2844 if (IsWin64UnwindV3) {
2845 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_SetFrame))
2847 .addImm(SEHFrameOffset)
2849 if (SEHStackAllocAmt)
2850 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_StackAlloc))
2851 .addImm(SEHStackAllocAmt)
2853 }
2854 if (LEAAmount != 0) {
2857 false, LEAAmount);
2858 --MBBI;
2859 } else {
2860 unsigned Opc = (Uses64BitFramePtr ? X86::MOV64rr : X86::MOV32rr);
2862 --MBBI;
2863 }
2864 } else if (NumBytes) {
2865 // Adjust stack pointer back: ESP += numbytes.
2866 if (IsWin64UnwindV3)
2867 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_StackAlloc))
2868 .addImm(NumBytes)
2870 emitSPUpdate(MBB, MBBI, DL, NumBytes, /*InEpilogue=*/true);
2871 if (!HasFP && NeedsDwarfCFI) {
2872 // Define the current CFA rule to use the provided offset.
2873 BuildCFI(MBB, MBBI, DL,
2875 nullptr, CSSize + TailCallArgReserveSize + SlotSize),
2877 }
2878 --MBBI;
2879 }
2880
2881 // For V1/V2, emit SEH_BeginEpilogue after stack restore code.
2882 if (!IsWin64UnwindV3 && NeedsWin64CFI && MF.hasWinCFI())
2883 BuildMI(MBB, MBBI, DL, TII.get(X86::SEH_BeginEpilogue))
2885
2886 if (!HasFP && NeedsDwarfCFI) {
2887 MBBI = FirstCSPop;
2888 int64_t Offset = -(int64_t)CSSize - SlotSize;
2889 // Mark callee-saved pop instruction.
2890 // Define the current CFA rule to use the provided offset.
2891 while (MBBI != MBB.end()) {
2893 unsigned Opc = PI->getOpcode();
2894 ++MBBI;
2895 if (Opc == X86::POP32r || Opc == X86::POP64r || Opc == X86::POPP64r ||
2896 Opc == X86::POP2 || Opc == X86::POP2P) {
2897 Offset += SlotSize;
2898 // Compared to pop, pop2 introduces more stack offset (one more
2899 // register).
2900 if (Opc == X86::POP2 || Opc == X86::POP2P)
2901 Offset += SlotSize;
2902 BuildCFI(MBB, MBBI, DL,
2905 }
2906 }
2907 }
2908
2909 // Emit DWARF info specifying the restores of the callee-saved registers.
2910 // For epilogue with return inside or being other block without successor,
2911 // no need to generate .cfi_restore for callee-saved registers.
2912 if (NeedsDwarfCFI && !MBB.succ_empty())
2913 emitCalleeSavedFrameMoves(MBB, AfterPop, DL, false);
2914
2915 if (Terminator == MBB.end() || !isTailCallOpcode(Terminator->getOpcode())) {
2916 // Add the return addr area delta back since we are not tail calling.
2917 int64_t Delta = X86FI->getTCReturnAddrDelta();
2918 assert(Delta <= 0 && "TCDelta should never be positive");
2919 if (Delta) {
2920 // Check for possible merge with preceding ADD instruction.
2921 int64_t Offset = mergeSPAdd(MBB, Terminator, -Delta, true);
2922 emitSPUpdate(MBB, Terminator, DL, Offset, /*InEpilogue=*/true);
2923 }
2924 }
2925
2926 // Emit tilerelease for AMX kernel.
2928 BuildMI(MBB, Terminator, DL, TII.get(X86::TILERELEASE));
2929
2930 if (NeedsWin64CFI && MF.hasWinCFI())
2931 BuildMI(MBB, Terminator, DL, TII.get(X86::SEH_EndEpilogue))
2933}
2934
2936 int FI,
2937 Register &FrameReg) const {
2938 const MachineFrameInfo &MFI = MF.getFrameInfo();
2939
2940 bool IsFixed = MFI.isFixedObjectIndex(FI);
2941 // We can't calculate offset from frame pointer if the stack is realigned,
2942 // so enforce usage of stack/base pointer. The base pointer is used when we
2943 // have dynamic allocas in addition to dynamic realignment.
2944 if (TRI->hasBasePointer(MF))
2945 FrameReg = IsFixed ? TRI->getFramePtr() : TRI->getBaseRegister();
2946 else if (TRI->hasStackRealignment(MF))
2947 FrameReg = IsFixed ? TRI->getFramePtr() : TRI->getStackRegister();
2948 else
2949 FrameReg = TRI->getFrameRegister(MF);
2950
2951 // Offset will hold the offset from the stack pointer at function entry to the
2952 // object.
2953 // We need to factor in additional offsets applied during the prologue to the
2954 // frame, base, and stack pointer depending on which is used.
2955 int64_t Offset = MFI.getObjectOffset(FI) - getOffsetOfLocalArea();
2957 unsigned CSSize = X86FI->getCalleeSavedFrameSize();
2958 uint64_t StackSize = MFI.getStackSize();
2959 bool IsWin64Prologue = MF.getTarget().getMCAsmInfo().usesWindowsCFI();
2960 int64_t FPDelta = 0;
2961
2962 // In an x86 interrupt, remove the offset we added to account for the return
2963 // address from any stack object allocated in the caller's frame. Interrupts
2964 // do not have a standard return address. Fixed objects in the current frame,
2965 // such as SSE register spills, should not get this treatment.
2967 Offset >= 0) {
2969 }
2970
2971 if (IsWin64Prologue) {
2972 assert(!MFI.hasCalls() || (StackSize % 16) == 8);
2973
2974 // Calculate required stack adjustment.
2975 uint64_t FrameSize = StackSize - SlotSize;
2976 // If required, include space for extra hidden slot for stashing base
2977 // pointer.
2978 if (X86FI->getRestoreBasePointer())
2979 FrameSize += SlotSize;
2980 uint64_t NumBytes = FrameSize - CSSize;
2981
2982 uint64_t SEHFrameOffset = calculateSetFPREG(NumBytes);
2983 if (FI && FI == X86FI->getFAIndex())
2984 return StackOffset::getFixed(-SEHFrameOffset);
2985
2986 // FPDelta is the offset from the "traditional" FP location of the old base
2987 // pointer followed by return address and the location required by the
2988 // restricted Win64 prologue.
2989 // Add FPDelta to all offsets below that go through the frame pointer.
2990 FPDelta = FrameSize - SEHFrameOffset;
2991 assert((!MFI.hasCalls() || (FPDelta % 16) == 0) &&
2992 "FPDelta isn't aligned per the Win64 ABI!");
2993 }
2994
2995 if (FrameReg == TRI->getFramePtr()) {
2996 // Skip saved EBP/RBP
2997 Offset += SlotSize;
2998
2999 // Account for restricted Windows prologue.
3000 Offset += FPDelta;
3001
3002 // Skip the RETADDR move area
3003 int TailCallReturnAddrDelta = X86FI->getTCReturnAddrDelta();
3004 if (TailCallReturnAddrDelta < 0)
3005 Offset -= TailCallReturnAddrDelta;
3006
3008 }
3009
3010 // FrameReg is either the stack pointer or a base pointer. But the base is
3011 // located at the end of the statically known StackSize so the distinction
3012 // doesn't really matter.
3013 if (TRI->hasStackRealignment(MF) || TRI->hasBasePointer(MF))
3014 assert(isAligned(MFI.getObjectAlign(FI), -(Offset + StackSize)));
3015 return StackOffset::getFixed(Offset + StackSize);
3016}
3017
3019 Register &FrameReg) const {
3020 const MachineFrameInfo &MFI = MF.getFrameInfo();
3022 const auto &WinEHXMMSlotInfo = X86FI->getWinEHXMMSlotInfo();
3023 const auto it = WinEHXMMSlotInfo.find(FI);
3024
3025 if (it == WinEHXMMSlotInfo.end())
3026 return getFrameIndexReference(MF, FI, FrameReg).getFixed();
3027
3028 FrameReg = TRI->getStackRegister();
3029 return alignDown(MFI.getMaxCallFrameSize(), getStackAlign().value()) +
3030 it->second;
3031}
3032
3035 Register &FrameReg,
3036 int Adjustment) const {
3037 const MachineFrameInfo &MFI = MF.getFrameInfo();
3038 FrameReg = TRI->getStackRegister();
3039 return StackOffset::getFixed(MFI.getObjectOffset(FI) -
3040 getOffsetOfLocalArea() + Adjustment);
3041}
3042
3045 int FI, Register &FrameReg,
3046 bool IgnoreSPUpdates) const {
3047
3048 const MachineFrameInfo &MFI = MF.getFrameInfo();
3049 // Does not include any dynamic realign.
3050 const uint64_t StackSize = MFI.getStackSize();
3051 // LLVM arranges the stack as follows:
3052 // ...
3053 // ARG2
3054 // ARG1
3055 // RETADDR
3056 // PUSH RBP <-- RBP points here
3057 // PUSH CSRs
3058 // ~~~~~~~ <-- possible stack realignment (non-win64)
3059 // ...
3060 // STACK OBJECTS
3061 // ... <-- RSP after prologue points here
3062 // ~~~~~~~ <-- possible stack realignment (win64)
3063 //
3064 // if (hasVarSizedObjects()):
3065 // ... <-- "base pointer" (ESI/RBX) points here
3066 // DYNAMIC ALLOCAS
3067 // ... <-- RSP points here
3068 //
3069 // Case 1: In the simple case of no stack realignment and no dynamic
3070 // allocas, both "fixed" stack objects (arguments and CSRs) are addressable
3071 // with fixed offsets from RSP.
3072 //
3073 // Case 2: In the case of stack realignment with no dynamic allocas, fixed
3074 // stack objects are addressed with RBP and regular stack objects with RSP.
3075 //
3076 // Case 3: In the case of dynamic allocas and stack realignment, RSP is used
3077 // to address stack arguments for outgoing calls and nothing else. The "base
3078 // pointer" points to local variables, and RBP points to fixed objects.
3079 //
3080 // In cases 2 and 3, we can only answer for non-fixed stack objects, and the
3081 // answer we give is relative to the SP after the prologue, and not the
3082 // SP in the middle of the function.
3083
3084 if (MFI.isFixedObjectIndex(FI) && TRI->hasStackRealignment(MF) &&
3085 !STI.isTargetWin64())
3086 return getFrameIndexReference(MF, FI, FrameReg);
3087
3088 // If !hasReservedCallFrame the function might have SP adjustement in the
3089 // body. So, even though the offset is statically known, it depends on where
3090 // we are in the function.
3091 if (!IgnoreSPUpdates && !hasReservedCallFrame(MF))
3092 return getFrameIndexReference(MF, FI, FrameReg);
3093
3094 // We don't handle tail calls, and shouldn't be seeing them either.
3096 "we don't handle this case!");
3097
3098 // This is how the math works out:
3099 //
3100 // %rsp grows (i.e. gets lower) left to right. Each box below is
3101 // one word (eight bytes). Obj0 is the stack slot we're trying to
3102 // get to.
3103 //
3104 // ----------------------------------
3105 // | BP | Obj0 | Obj1 | ... | ObjN |
3106 // ----------------------------------
3107 // ^ ^ ^ ^
3108 // A B C E
3109 //
3110 // A is the incoming stack pointer.
3111 // (B - A) is the local area offset (-8 for x86-64) [1]
3112 // (C - A) is the Offset returned by MFI.getObjectOffset for Obj0 [2]
3113 //
3114 // |(E - B)| is the StackSize (absolute value, positive). For a
3115 // stack that grown down, this works out to be (B - E). [3]
3116 //
3117 // E is also the value of %rsp after stack has been set up, and we
3118 // want (C - E) -- the value we can add to %rsp to get to Obj0. Now
3119 // (C - E) == (C - A) - (B - A) + (B - E)
3120 // { Using [1], [2] and [3] above }
3121 // == getObjectOffset - LocalAreaOffset + StackSize
3122
3123 return getFrameIndexReferenceSP(MF, FI, FrameReg, StackSize);
3124}
3125
3128 std::vector<CalleeSavedInfo> &CSI) const {
3129 MachineFrameInfo &MFI = MF.getFrameInfo();
3131
3132 unsigned CalleeSavedFrameSize = 0;
3133 unsigned XMMCalleeSavedFrameSize = 0;
3134 auto &WinEHXMMSlotInfo = X86FI->getWinEHXMMSlotInfo();
3135 int SpillSlotOffset = getOffsetOfLocalArea() + X86FI->getTCReturnAddrDelta();
3136
3137 int64_t TailCallReturnAddrDelta = X86FI->getTCReturnAddrDelta();
3138
3139 if (TailCallReturnAddrDelta < 0) {
3140 // create RETURNADDR area
3141 // arg
3142 // arg
3143 // RETADDR
3144 // { ...
3145 // RETADDR area
3146 // ...
3147 // }
3148 // [EBP]
3149 MFI.CreateFixedObject(-TailCallReturnAddrDelta,
3150 TailCallReturnAddrDelta - SlotSize, true);
3151 }
3152
3153 // Spill the BasePtr if it's used.
3154 if (this->TRI->hasBasePointer(MF)) {
3155 // Allocate a spill slot for EBP if we have a base pointer and EH funclets.
3156 if (MF.hasEHFunclets()) {
3158 X86FI->setHasSEHFramePtrSave(true);
3159 X86FI->setSEHFramePtrSaveIndex(FI);
3160 }
3161 }
3162
3163 bool IsFPRemovedFromCSI = false;
3164 if (hasFP(MF)) {
3165 // emitPrologue always spills frame register the first thing.
3166 SpillSlotOffset -= SlotSize;
3167 MFI.CreateFixedSpillStackObject(SlotSize, SpillSlotOffset);
3168
3169 // The async context lives directly before the frame pointer, and we
3170 // allocate a second slot to preserve stack alignment.
3171 if (X86FI->hasSwiftAsyncContext()) {
3172 SpillSlotOffset -= SlotSize;
3173 MFI.CreateFixedSpillStackObject(SlotSize, SpillSlotOffset);
3174 SpillSlotOffset -= SlotSize;
3175 }
3176
3177 // Since emitPrologue and emitEpilogue will handle spilling and restoring of
3178 // the frame register, we can delete it from CSI list and not have to worry
3179 // about avoiding it later.
3180 Register FPReg = TRI->getFrameRegister(MF);
3181 for (unsigned i = 0; i < CSI.size(); ++i) {
3182 if (TRI->regsOverlap(CSI[i].getReg(), FPReg)) {
3183 CSI.erase(CSI.begin() + i);
3184 IsFPRemovedFromCSI = true;
3185 break;
3186 }
3187 }
3188 }
3189
3190 // Strategy:
3191 // 1. Use push2 when
3192 // a) number of CSR > 1 if no need padding
3193 // b) number of CSR > 2 if need padding
3194 // c) stack alignment >= 16 bytes
3195 // 2. When the number of CSR push is odd
3196 // a. Start to use push2 from the 1st push if stack is 16B aligned.
3197 // b. Start to use push2 from the 2nd push if stack is not 16B aligned.
3198 // 3. When the number of CSR push is even, start to use push2 from the 1st
3199 // push and make the stack 16B aligned before the push
3200 unsigned NumRegsForPush2 = 0;
3201 if (STI.hasPush2Pop2() && getStackAlignment() >= 16) {
3202 unsigned NumCSGPR = llvm::count_if(CSI, [](const CalleeSavedInfo &I) {
3203 return X86::GR64RegClass.contains(I.getReg());
3204 });
3205 bool UsePush2Pop2 = !IsFPRemovedFromCSI ? NumCSGPR > 2 : NumCSGPR > 1;
3206 NumRegsForPush2 =
3207 UsePush2Pop2
3208 ? alignDown(IsFPRemovedFromCSI ? NumCSGPR : NumCSGPR - 1, 2)
3209 : 0;
3210 }
3211
3212 // Assign slots for GPRs. It increases frame size.
3213 for (CalleeSavedInfo &I : llvm::reverse(CSI)) {
3214 MCRegister Reg = I.getReg();
3215
3216 if (!X86::GR64RegClass.contains(Reg) && !X86::GR32RegClass.contains(Reg))
3217 continue;
3218
3219 // A CSR is a candidate for push2/pop2 when it's slot offset is 16B aligned
3220 // or only an odd number of registers in the candidates.
3221 if (X86FI->getNumCandidatesForPush2Pop2() < NumRegsForPush2 &&
3222 (SpillSlotOffset % 16 == 0 ||
3223 X86FI->getNumCandidatesForPush2Pop2() % 2))
3224 X86FI->addCandidateForPush2Pop2(Reg);
3225
3226 SpillSlotOffset -= SlotSize;
3227 CalleeSavedFrameSize += SlotSize;
3228
3229 int SlotIndex = MFI.CreateFixedSpillStackObject(SlotSize, SpillSlotOffset);
3230 I.setFrameIdx(SlotIndex);
3231 }
3232
3233 // Adjust the offset of spill slot as we know the accurate callee saved frame
3234 // size.
3235 if (X86FI->getRestoreBasePointer()) {
3236 SpillSlotOffset -= SlotSize;
3237 CalleeSavedFrameSize += SlotSize;
3238
3239 MFI.CreateFixedSpillStackObject(SlotSize, SpillSlotOffset);
3240 // TODO: saving the slot index is better?
3241 X86FI->setRestoreBasePointer(CalleeSavedFrameSize);
3242 }
3243 assert(X86FI->getNumCandidatesForPush2Pop2() % 2 == 0 &&
3244 "Expect even candidates for push2/pop2");
3245 if (X86FI->getNumCandidatesForPush2Pop2())
3246 ++NumFunctionUsingPush2Pop2;
3247 X86FI->setCalleeSavedFrameSize(CalleeSavedFrameSize);
3248 MFI.setCVBytesOfCalleeSavedRegisters(CalleeSavedFrameSize);
3249
3250 // Assign slots for XMMs.
3251 for (CalleeSavedInfo &I : llvm::reverse(CSI)) {
3252 MCRegister Reg = I.getReg();
3253 if (X86::GR64RegClass.contains(Reg) || X86::GR32RegClass.contains(Reg))
3254 continue;
3255
3257 unsigned Size = TRI->getSpillSize(*RC);
3258 Align Alignment = TRI->getSpillAlign(*RC);
3259 // ensure alignment
3260 assert(SpillSlotOffset < 0 && "SpillSlotOffset should always < 0 on X86");
3261 SpillSlotOffset = -alignTo(-SpillSlotOffset, Alignment);
3262
3263 // spill into slot
3264 SpillSlotOffset -= Size;
3265 int SlotIndex = MFI.CreateFixedSpillStackObject(Size, SpillSlotOffset);
3266 I.setFrameIdx(SlotIndex);
3267 MFI.ensureMaxAlignment(Alignment);
3268
3269 // Save the start offset and size of XMM in stack frame for funclets.
3270 if (X86::VR128RegClass.contains(Reg)) {
3271 WinEHXMMSlotInfo[SlotIndex] = XMMCalleeSavedFrameSize;
3272 XMMCalleeSavedFrameSize += Size;
3273 }
3274 }
3275
3276 return true;
3277}
3278
3282 DebugLoc DL = MBB.findDebugLoc(MI);
3283
3284 // Don't save CSRs in 32-bit EH funclets. The caller saves EBX, EBP, ESI, EDI
3285 // for us, and there are no XMM CSRs on Win32.
3286 if (MBB.isEHFuncletEntry() && STI.is32Bit() && STI.isOSWindows())
3287 return true;
3288
3289 // Push GPRs. It increases frame size.
3290 const MachineFunction &MF = *MBB.getParent();
3292
3293 // Update LiveIn of the basic block and decide whether we can add a kill flag
3294 // to the use.
3295 auto UpdateLiveInCheckCanKill = [&](Register Reg) {
3296 const MachineRegisterInfo &MRI = MF.getRegInfo();
3297 // Do not set a kill flag on values that are also marked as live-in. This
3298 // happens with the @llvm-returnaddress intrinsic and with arguments
3299 // passed in callee saved registers.
3300 // Omitting the kill flags is conservatively correct even if the live-in
3301 // is not used after all.
3302 if (MRI.isLiveIn(Reg))
3303 return false;
3304 MBB.addLiveIn(Reg);
3305 // Check if any subregister is live-in
3306 for (MCRegAliasIterator AReg(Reg, TRI, false); AReg.isValid(); ++AReg)
3307 if (MRI.isLiveIn(*AReg))
3308 return false;
3309 return true;
3310 };
3311 auto UpdateLiveInGetKillRegState = [&](Register Reg) {
3312 return getKillRegState(UpdateLiveInCheckCanKill(Reg));
3313 };
3314
3315 for (auto RI = CSI.rbegin(), RE = CSI.rend(); RI != RE; ++RI) {
3316 MCRegister Reg = RI->getReg();
3317 if (!X86::GR64RegClass.contains(Reg) && !X86::GR32RegClass.contains(Reg))
3318 continue;
3319
3320 if (X86FI->isCandidateForPush2Pop2(Reg)) {
3321 MCRegister Reg2 = (++RI)->getReg();
3323 .addReg(Reg, UpdateLiveInGetKillRegState(Reg))
3324 .addReg(Reg2, UpdateLiveInGetKillRegState(Reg2))
3326 } else {
3327 BuildMI(MBB, MI, DL, TII.get(getPUSHOpcode(STI)))
3328 .addReg(Reg, UpdateLiveInGetKillRegState(Reg))
3330 }
3331 }
3332
3333 if (X86FI->getRestoreBasePointer()) {
3334 unsigned Opc = STI.is64Bit() ? X86::PUSH64r : X86::PUSH32r;
3335 Register BaseReg = this->TRI->getBaseRegister();
3336 BuildMI(MBB, MI, DL, TII.get(Opc))
3337 .addReg(BaseReg, getKillRegState(true))
3339 }
3340
3341 // Make XMM regs spilled. X86 does not have ability of push/pop XMM.
3342 // It can be done by spilling XMMs to stack frame.
3343 for (const CalleeSavedInfo &I : llvm::reverse(CSI)) {
3344 MCRegister Reg = I.getReg();
3345 if (X86::GR64RegClass.contains(Reg) || X86::GR32RegClass.contains(Reg))
3346 continue;
3347
3348 // Add the callee-saved register as live-in. It's killed at the spill.
3349 MBB.addLiveIn(Reg);
3351
3352 TII.storeRegToStackSlot(MBB, MI, Reg, true, I.getFrameIdx(), RC, Register(),
3354 }
3355
3356 return true;
3357}
3358
3359void X86FrameLowering::emitCatchRetReturnValue(MachineBasicBlock &MBB,
3361 MachineInstr *CatchRet) const {
3362 // SEH shouldn't use catchret.
3364 MBB.getParent()->getFunction().getPersonalityFn())) &&
3365 "SEH should not use CATCHRET");
3366 const DebugLoc &DL = CatchRet->getDebugLoc();
3367 MachineBasicBlock *CatchRetTarget = CatchRet->getOperand(0).getMBB();
3368
3369 // Fill EAX/RAX with the address of the target block.
3370 if (STI.is64Bit()) {
3371 // LEA64r CatchRetTarget(%rip), %rax
3372 BuildMI(MBB, MBBI, DL, TII.get(X86::LEA64r), X86::RAX)
3373 .addReg(X86::RIP)
3374 .addImm(0)
3375 .addReg(0)
3376 .addMBB(CatchRetTarget)
3377 .addReg(0);
3378 } else {
3379 // MOV32ri $CatchRetTarget, %eax
3380 BuildMI(MBB, MBBI, DL, TII.get(X86::MOV32ri), X86::EAX)
3381 .addMBB(CatchRetTarget);
3382 }
3383
3384 // Record that we've taken the address of CatchRetTarget and no longer just
3385 // reference it in a terminator.
3386 CatchRetTarget->setMachineBlockAddressTaken();
3387}
3388
3392 if (CSI.empty())
3393 return false;
3394
3395 if (MI != MBB.end() && isFuncletReturnInstr(*MI) && STI.isOSWindows()) {
3396 // Don't restore CSRs in 32-bit EH funclets. Matches
3397 // spillCalleeSavedRegisters.
3398 if (STI.is32Bit())
3399 return true;
3400 // Don't restore CSRs before an SEH catchret. SEH except blocks do not form
3401 // funclets. emitEpilogue transforms these to normal jumps.
3402 if (MI->getOpcode() == X86::CATCHRET) {
3403 const Function &F = MBB.getParent()->getFunction();
3404 bool IsSEH = isAsynchronousEHPersonality(
3405 classifyEHPersonality(F.getPersonalityFn()));
3406 if (IsSEH)
3407 return true;
3408 }
3409 }
3410
3411 DebugLoc DL = MBB.findDebugLoc(MI);
3412 MachineFunction &MF = *MBB.getParent();
3414
3415 bool NeedsWin64CFI =
3416 isWin64Prologue(MF) && MF.getFunction().needsUnwindTableEntry();
3417 bool IsWin64UnwindV3 = NeedsWin64CFI && requireWinX64UnwindV3(MF);
3418
3419 // Reload XMMs from stack frame.
3420 for (const CalleeSavedInfo &I : CSI) {
3421 MCRegister Reg = I.getReg();
3422 if (X86::GR64RegClass.contains(Reg) || X86::GR32RegClass.contains(Reg))
3423 continue;
3424
3426 TII.loadRegFromStackSlot(MBB, MI, Reg, I.getFrameIdx(), RC, Register(), 0,
3428 }
3429
3430 // Clear the stack slot for spill base pointer register.
3431 if (X86FI->getRestoreBasePointer()) {
3432 if (IsWin64UnwindV3)
3433 BuildMI(MBB, MI, DL, TII.get(X86::SEH_PushReg))
3434 .addImm(this->TRI->getBaseRegister())
3436 unsigned Opc = STI.is64Bit() ? X86::POP64r : X86::POP32r;
3437 Register BaseReg = this->TRI->getBaseRegister();
3438 BuildMI(MBB, MI, DL, TII.get(Opc), BaseReg)
3440 }
3441
3442 // POP GPRs.
3443 for (auto I = CSI.begin(), E = CSI.end(); I != E; ++I) {
3444 MCRegister Reg = I->getReg();
3445 if (!X86::GR64RegClass.contains(Reg) && !X86::GR32RegClass.contains(Reg))
3446 continue;
3447
3448 if (X86FI->isCandidateForPush2Pop2(Reg)) {
3449 MCRegister Reg2 = (++I)->getReg();
3450 if (IsWin64UnwindV3) {
3451 BuildMI(MBB, MI, DL, TII.get(X86::SEH_Push2Regs))
3452 .addImm(Reg)
3453 .addImm(Reg2)
3455 }
3456 BuildMI(MBB, MI, DL, TII.get(getPOP2Opcode(STI)), Reg)
3457 .addReg(Reg2, RegState::Define)
3459 } else {
3460 if (IsWin64UnwindV3)
3461 BuildMI(MBB, MI, DL, TII.get(X86::SEH_PushReg))
3462 .addImm(Reg)
3464 BuildMI(MBB, MI, DL, TII.get(getPOPOpcode(STI)), Reg)
3466 }
3467 }
3468
3469 return true;
3470}
3471
3473 BitVector &SavedRegs,
3474 RegScavenger *RS) const {
3476
3477 // Spill the BasePtr if it's used.
3478 if (TRI->hasBasePointer(MF)) {
3479 Register BasePtr = TRI->getBaseRegister();
3480 if (STI.isTarget64BitILP32())
3481 BasePtr = getX86SubSuperRegister(BasePtr, 64);
3482 SavedRegs.set(BasePtr);
3483 }
3484 if (STI.hasUserReservedRegisters()) {
3485 for (int Reg = SavedRegs.find_first(); Reg != -1;
3486 Reg = SavedRegs.find_next(Reg)) {
3487 if (STI.isRegisterReservedByUser(Reg)) {
3488 SavedRegs.reset(Reg);
3489 }
3490 }
3491 }
3492}
3493
3494static bool HasNestArgument(const MachineFunction *MF) {
3495 const Function &F = MF->getFunction();
3496 for (Function::const_arg_iterator I = F.arg_begin(), E = F.arg_end(); I != E;
3497 I++) {
3498 if (I->hasNestAttr() && !I->use_empty())
3499 return true;
3500 }
3501 return false;
3502}
3503
3504/// GetScratchRegister - Get a temp register for performing work in the
3505/// segmented stack and the Erlang/HiPE stack prologue. Depending on platform
3506/// and the properties of the function either one or two registers will be
3507/// needed. Set primary to true for the first register, false for the second.
3508static unsigned GetScratchRegister(bool Is64Bit, bool IsLP64,
3509 const MachineFunction &MF, bool Primary) {
3510 CallingConv::ID CallingConvention = MF.getFunction().getCallingConv();
3511
3512 // Erlang stuff.
3513 if (CallingConvention == CallingConv::HiPE) {
3514 if (Is64Bit)
3515 return Primary ? X86::R14 : X86::R13;
3516 else
3517 return Primary ? X86::EBX : X86::EDI;
3518 }
3519
3520 if (Is64Bit) {
3521 if (IsLP64)
3522 return Primary ? X86::R11 : X86::R12;
3523 else
3524 return Primary ? X86::R11D : X86::R12D;
3525 }
3526
3527 bool IsNested = HasNestArgument(&MF);
3528
3529 if (CallingConvention == CallingConv::X86_FastCall ||
3530 CallingConvention == CallingConv::Fast ||
3531 CallingConvention == CallingConv::Tail) {
3532 if (IsNested)
3533 report_fatal_error("Segmented stacks does not support fastcall with "
3534 "nested function.");
3535 return Primary ? X86::EAX : X86::ECX;
3536 }
3537 if (IsNested)
3538 return Primary ? X86::EDX : X86::EAX;
3539 return Primary ? X86::ECX : X86::EAX;
3540}
3541
3542// The stack limit in the TCB is set to this many bytes above the actual stack
3543// limit.
3545
3547 MachineFunction &MF, MachineBasicBlock &PrologueMBB) const {
3548 MachineFrameInfo &MFI = MF.getFrameInfo();
3549 uint64_t StackSize;
3550 unsigned TlsReg, TlsOffset;
3551 DebugLoc DL;
3552
3553 // To support shrink-wrapping we would need to insert the new blocks
3554 // at the right place and update the branches to PrologueMBB.
3555 assert(&(*MF.begin()) == &PrologueMBB && "Shrink-wrapping not supported yet");
3556
3557 unsigned ScratchReg = GetScratchRegister(Is64Bit, IsLP64, MF, true);
3558 assert(!MF.getRegInfo().isLiveIn(ScratchReg) &&
3559 "Scratch register is live-in");
3560
3561 if (MF.getFunction().isVarArg())
3562 report_fatal_error("Segmented stacks do not support vararg functions.");
3563 if (!STI.isTargetLinux() && !STI.isTargetDarwin() && !STI.isTargetWin32() &&
3564 !STI.isTargetWin64() && !STI.isTargetFreeBSD() &&
3565 !STI.isTargetDragonFly())
3566 report_fatal_error("Segmented stacks not supported on this platform.");
3567
3568 // Eventually StackSize will be calculated by a link-time pass; which will
3569 // also decide whether checking code needs to be injected into this particular
3570 // prologue.
3571 StackSize = MFI.getStackSize();
3572
3573 if (!MFI.needsSplitStackProlog())
3574 return;
3575
3579 bool IsNested = false;
3580
3581 // We need to know if the function has a nest argument only in 64 bit mode.
3582 if (Is64Bit)
3583 IsNested = HasNestArgument(&MF);
3584
3585 // The MOV R10, RAX needs to be in a different block, since the RET we emit in
3586 // allocMBB needs to be last (terminating) instruction.
3587
3588 for (const auto &LI : PrologueMBB.liveins()) {
3589 allocMBB->addLiveIn(LI);
3590 checkMBB->addLiveIn(LI);
3591 }
3592
3593 if (IsNested)
3594 allocMBB->addLiveIn(IsLP64 ? X86::R10 : X86::R10D);
3595
3596 MF.push_front(allocMBB);
3597 MF.push_front(checkMBB);
3598
3599 // When the frame size is less than 256 we just compare the stack
3600 // boundary directly to the value of the stack pointer, per gcc.
3601 bool CompareStackPointer = StackSize < kSplitStackAvailable;
3602
3603 // Read the limit off the current stacklet off the stack_guard location.
3604 if (Is64Bit) {
3605 if (STI.isTargetLinux()) {
3606 TlsReg = X86::FS;
3607 TlsOffset = IsLP64 ? 0x70 : 0x40;
3608 } else if (STI.isTargetDarwin()) {
3609 TlsReg = X86::GS;
3610 TlsOffset = 0x60 + 90 * 8; // See pthread_machdep.h. Steal TLS slot 90.
3611 } else if (STI.isTargetWin64()) {
3612 TlsReg = X86::GS;
3613 TlsOffset = 0x28; // pvArbitrary, reserved for application use
3614 } else if (STI.isTargetFreeBSD()) {
3615 TlsReg = X86::FS;
3616 TlsOffset = 0x18;
3617 } else if (STI.isTargetDragonFly()) {
3618 TlsReg = X86::FS;
3619 TlsOffset = 0x20; // use tls_tcb.tcb_segstack
3620 } else {
3621 report_fatal_error("Segmented stacks not supported on this platform.");
3622 }
3623
3624 if (CompareStackPointer)
3625 ScratchReg = IsLP64 ? X86::RSP : X86::ESP;
3626 else
3627 BuildMI(checkMBB, DL, TII.get(IsLP64 ? X86::LEA64r : X86::LEA64_32r),
3628 ScratchReg)
3629 .addReg(X86::RSP)
3630 .addImm(1)
3631 .addReg(0)
3632 .addImm(-StackSize)
3633 .addReg(0);
3634
3635 BuildMI(checkMBB, DL, TII.get(IsLP64 ? X86::CMP64rm : X86::CMP32rm))
3636 .addReg(ScratchReg)
3637 .addReg(0)
3638 .addImm(1)
3639 .addReg(0)
3640 .addImm(TlsOffset)
3641 .addReg(TlsReg);
3642 } else {
3643 if (STI.isTargetLinux()) {
3644 TlsReg = X86::GS;
3645 TlsOffset = 0x30;
3646 } else if (STI.isTargetDarwin()) {
3647 TlsReg = X86::GS;
3648 TlsOffset = 0x48 + 90 * 4;
3649 } else if (STI.isTargetWin32()) {
3650 TlsReg = X86::FS;
3651 TlsOffset = 0x14; // pvArbitrary, reserved for application use
3652 } else if (STI.isTargetDragonFly()) {
3653 TlsReg = X86::FS;
3654 TlsOffset = 0x10; // use tls_tcb.tcb_segstack
3655 } else if (STI.isTargetFreeBSD()) {
3656 report_fatal_error("Segmented stacks not supported on FreeBSD i386.");
3657 } else {
3658 report_fatal_error("Segmented stacks not supported on this platform.");
3659 }
3660
3661 if (CompareStackPointer)
3662 ScratchReg = X86::ESP;
3663 else
3664 BuildMI(checkMBB, DL, TII.get(X86::LEA32r), ScratchReg)
3665 .addReg(X86::ESP)
3666 .addImm(1)
3667 .addReg(0)
3668 .addImm(-StackSize)
3669 .addReg(0);
3670
3671 if (STI.isTargetLinux() || STI.isTargetWin32() || STI.isTargetWin64() ||
3672 STI.isTargetDragonFly()) {
3673 BuildMI(checkMBB, DL, TII.get(X86::CMP32rm))
3674 .addReg(ScratchReg)
3675 .addReg(0)
3676 .addImm(0)
3677 .addReg(0)
3678 .addImm(TlsOffset)
3679 .addReg(TlsReg);
3680 } else if (STI.isTargetDarwin()) {
3681
3682 // TlsOffset doesn't fit into a mod r/m byte so we need an extra register.
3683 unsigned ScratchReg2;
3684 bool SaveScratch2;
3685 if (CompareStackPointer) {
3686 // The primary scratch register is available for holding the TLS offset.
3687 ScratchReg2 = GetScratchRegister(Is64Bit, IsLP64, MF, true);
3688 SaveScratch2 = false;
3689 } else {
3690 // Need to use a second register to hold the TLS offset
3691 ScratchReg2 = GetScratchRegister(Is64Bit, IsLP64, MF, false);
3692
3693 // Unfortunately, with fastcc the second scratch register may hold an
3694 // argument.
3695 SaveScratch2 = MF.getRegInfo().isLiveIn(ScratchReg2);
3696 }
3697
3698 // If Scratch2 is live-in then it needs to be saved.
3699 assert((!MF.getRegInfo().isLiveIn(ScratchReg2) || SaveScratch2) &&
3700 "Scratch register is live-in and not saved");
3701
3702 if (SaveScratch2)
3703 BuildMI(checkMBB, DL, TII.get(X86::PUSH32r))
3704 .addReg(ScratchReg2, RegState::Kill);
3705
3706 BuildMI(checkMBB, DL, TII.get(X86::MOV32ri), ScratchReg2)
3707 .addImm(TlsOffset);
3708 BuildMI(checkMBB, DL, TII.get(X86::CMP32rm))
3709 .addReg(ScratchReg)
3710 .addReg(ScratchReg2)
3711 .addImm(1)
3712 .addReg(0)
3713 .addImm(0)
3714 .addReg(TlsReg);
3715
3716 if (SaveScratch2)
3717 BuildMI(checkMBB, DL, TII.get(X86::POP32r), ScratchReg2);
3718 }
3719 }
3720
3721 // This jump is taken if SP >= (Stacklet Limit + Stack Space required).
3722 // It jumps to normal execution of the function body.
3723 BuildMI(checkMBB, DL, TII.get(X86::JCC_1))
3724 .addMBB(&PrologueMBB)
3726
3727 // On 32 bit we first push the arguments size and then the frame size. On 64
3728 // bit, we pass the stack frame size in r10 and the argument size in r11.
3729 if (Is64Bit) {
3730 // Functions with nested arguments use R10, so it needs to be saved across
3731 // the call to _morestack
3732
3733 const unsigned RegAX = IsLP64 ? X86::RAX : X86::EAX;
3734 const unsigned Reg10 = IsLP64 ? X86::R10 : X86::R10D;
3735 const unsigned Reg11 = IsLP64 ? X86::R11 : X86::R11D;
3736 const unsigned MOVrr = IsLP64 ? X86::MOV64rr : X86::MOV32rr;
3737
3738 if (IsNested)
3739 BuildMI(allocMBB, DL, TII.get(MOVrr), RegAX).addReg(Reg10);
3740
3741 BuildMI(allocMBB, DL, TII.get(X86::getMOVriOpcode(IsLP64, StackSize)),
3742 Reg10)
3743 .addImm(StackSize);
3744 BuildMI(allocMBB, DL,
3746 Reg11)
3747 .addImm(X86FI->getArgumentStackSize());
3748 } else {
3749 BuildMI(allocMBB, DL, TII.get(X86::PUSH32i))
3750 .addImm(X86FI->getArgumentStackSize());
3751 BuildMI(allocMBB, DL, TII.get(X86::PUSH32i)).addImm(StackSize);
3752 }
3753
3754 // __morestack is in libgcc
3756 // Under the large code model, we cannot assume that __morestack lives
3757 // within 2^31 bytes of the call site, so we cannot use pc-relative
3758 // addressing. We cannot perform the call via a temporary register,
3759 // as the rax register may be used to store the static chain, and all
3760 // other suitable registers may be either callee-save or used for
3761 // parameter passing. We cannot use the stack at this point either
3762 // because __morestack manipulates the stack directly.
3763 //
3764 // To avoid these issues, perform an indirect call via a read-only memory
3765 // location containing the address.
3766 //
3767 // This solution is not perfect, as it assumes that the .rodata section
3768 // is laid out within 2^31 bytes of each function body, but this seems
3769 // to be sufficient for JIT.
3770 // FIXME: Add retpoline support and remove the error here..
3771 if (STI.useIndirectThunkCalls())
3772 report_fatal_error("Emitting morestack calls on 64-bit with the large "
3773 "code model and thunks not yet implemented.");
3774 BuildMI(allocMBB, DL, TII.get(X86::CALL64m))
3775 .addReg(X86::RIP)
3776 .addImm(0)
3777 .addReg(0)
3778 .addExternalSymbol("__morestack_addr")
3779 .addReg(0);
3780 } else {
3781 if (Is64Bit)
3782 BuildMI(allocMBB, DL, TII.get(X86::CALL64pcrel32))
3783 .addExternalSymbol("__morestack");
3784 else
3785 BuildMI(allocMBB, DL, TII.get(X86::CALLpcrel32))
3786 .addExternalSymbol("__morestack");
3787 }
3788
3789 if (IsNested)
3790 BuildMI(allocMBB, DL, TII.get(X86::MORESTACK_RET_RESTORE_R10));
3791 else
3792 BuildMI(allocMBB, DL, TII.get(X86::MORESTACK_RET));
3793
3794 allocMBB->addSuccessor(&PrologueMBB);
3795
3796 checkMBB->addSuccessor(allocMBB, BranchProbability::getZero());
3797 checkMBB->addSuccessor(&PrologueMBB, BranchProbability::getOne());
3798
3799#ifdef EXPENSIVE_CHECKS
3800 MF.verify();
3801#endif
3802}
3803
3804/// Lookup an ERTS parameter in the !hipe.literals named metadata node.
3805/// HiPE provides Erlang Runtime System-internal parameters, such as PCB offsets
3806/// to fields it needs, through a named metadata node "hipe.literals" containing
3807/// name-value pairs.
3808static unsigned getHiPELiteral(NamedMDNode *HiPELiteralsMD,
3809 const StringRef LiteralName) {
3810 for (int i = 0, e = HiPELiteralsMD->getNumOperands(); i != e; ++i) {
3811 MDNode *Node = HiPELiteralsMD->getOperand(i);
3812 if (Node->getNumOperands() != 2)
3813 continue;
3814 MDString *NodeName = dyn_cast<MDString>(Node->getOperand(0));
3815 ValueAsMetadata *NodeVal = dyn_cast<ValueAsMetadata>(Node->getOperand(1));
3816 if (!NodeName || !NodeVal)
3817 continue;
3818 ConstantInt *ValConst = dyn_cast_or_null<ConstantInt>(NodeVal->getValue());
3819 if (ValConst && NodeName->getString() == LiteralName) {
3820 return ValConst->getZExtValue();
3821 }
3822 }
3823
3824 report_fatal_error("HiPE literal " + LiteralName +
3825 " required but not provided");
3826}
3827
3828// Return true if there are no non-ehpad successors to MBB and there are no
3829// non-meta instructions between MBBI and MBB.end().
3832 return llvm::all_of(
3833 MBB.successors(),
3834 [](const MachineBasicBlock *Succ) { return Succ->isEHPad(); }) &&
3835 std::all_of(MBBI, MBB.end(), [](const MachineInstr &MI) {
3836 return MI.isMetaInstruction();
3837 });
3838}
3839
3840/// Erlang programs may need a special prologue to handle the stack size they
3841/// might need at runtime. That is because Erlang/OTP does not implement a C
3842/// stack but uses a custom implementation of hybrid stack/heap architecture.
3843/// (for more information see Eric Stenman's Ph.D. thesis:
3844/// http://publications.uu.se/uu/fulltext/nbn_se_uu_diva-2688.pdf)
3845///
3846/// CheckStack:
3847/// temp0 = sp - MaxStack
3848/// if( temp0 < SP_LIMIT(P) ) goto IncStack else goto OldStart
3849/// OldStart:
3850/// ...
3851/// IncStack:
3852/// call inc_stack # doubles the stack space
3853/// temp0 = sp - MaxStack
3854/// if( temp0 < SP_LIMIT(P) ) goto IncStack else goto OldStart
3856 MachineFunction &MF, MachineBasicBlock &PrologueMBB) const {
3857 MachineFrameInfo &MFI = MF.getFrameInfo();
3858 DebugLoc DL;
3859
3860 // To support shrink-wrapping we would need to insert the new blocks
3861 // at the right place and update the branches to PrologueMBB.
3862 assert(&(*MF.begin()) == &PrologueMBB && "Shrink-wrapping not supported yet");
3863
3864 // HiPE-specific values
3865 NamedMDNode *HiPELiteralsMD =
3866 MF.getFunction().getParent()->getNamedMetadata("hipe.literals");
3867 if (!HiPELiteralsMD)
3869 "Can't generate HiPE prologue without runtime parameters");
3870 const unsigned HipeLeafWords = getHiPELiteral(
3871 HiPELiteralsMD, Is64Bit ? "AMD64_LEAF_WORDS" : "X86_LEAF_WORDS");
3872 const unsigned CCRegisteredArgs = Is64Bit ? 6 : 5;
3873 const unsigned Guaranteed = HipeLeafWords * SlotSize;
3874 unsigned CallerStkArity = MF.getFunction().arg_size() > CCRegisteredArgs
3875 ? MF.getFunction().arg_size() - CCRegisteredArgs
3876 : 0;
3877 unsigned MaxStack = MFI.getStackSize() + CallerStkArity * SlotSize + SlotSize;
3878
3879 assert(STI.isTargetLinux() &&
3880 "HiPE prologue is only supported on Linux operating systems.");
3881
3882 // Compute the largest caller's frame that is needed to fit the callees'
3883 // frames. This 'MaxStack' is computed from:
3884 //
3885 // a) the fixed frame size, which is the space needed for all spilled temps,
3886 // b) outgoing on-stack parameter areas, and
3887 // c) the minimum stack space this function needs to make available for the
3888 // functions it calls (a tunable ABI property).
3889 if (MFI.hasCalls()) {
3890 unsigned MoreStackForCalls = 0;
3891
3892 for (auto &MBB : MF) {
3893 for (auto &MI : MBB) {
3894 if (!MI.isCall())
3895 continue;
3896
3897 // Get callee operand.
3898 const MachineOperand &MO = MI.getOperand(0);
3899
3900 // Only take account of global function calls (no closures etc.).
3901 if (!MO.isGlobal())
3902 continue;
3903
3904 const Function *F = dyn_cast<Function>(MO.getGlobal());
3905 if (!F)
3906 continue;
3907
3908 // Do not update 'MaxStack' for primitive and built-in functions
3909 // (encoded with names either starting with "erlang."/"bif_" or not
3910 // having a ".", such as a simple <Module>.<Function>.<Arity>, or an
3911 // "_", such as the BIF "suspend_0") as they are executed on another
3912 // stack.
3913 if (F->getName().contains("erlang.") || F->getName().contains("bif_") ||
3914 F->getName().find_first_of("._") == StringRef::npos)
3915 continue;
3916
3917 unsigned CalleeStkArity = F->arg_size() > CCRegisteredArgs
3918 ? F->arg_size() - CCRegisteredArgs
3919 : 0;
3920 if (HipeLeafWords - 1 > CalleeStkArity)
3921 MoreStackForCalls =
3922 std::max(MoreStackForCalls,
3923 (HipeLeafWords - 1 - CalleeStkArity) * SlotSize);
3924 }
3925 }
3926 MaxStack += MoreStackForCalls;
3927 }
3928
3929 // If the stack frame needed is larger than the guaranteed then runtime checks
3930 // and calls to "inc_stack_0" BIF should be inserted in the assembly prologue.
3931 if (MaxStack > Guaranteed) {
3932 MachineBasicBlock *stackCheckMBB = MF.CreateMachineBasicBlock();
3933 MachineBasicBlock *incStackMBB = MF.CreateMachineBasicBlock();
3934
3935 for (const auto &LI : PrologueMBB.liveins()) {
3936 stackCheckMBB->addLiveIn(LI);
3937 incStackMBB->addLiveIn(LI);
3938 }
3939
3940 MF.push_front(incStackMBB);
3941 MF.push_front(stackCheckMBB);
3942
3943 unsigned ScratchReg, SPReg, PReg, SPLimitOffset;
3944 unsigned LEAop, CMPop, CALLop;
3945 SPLimitOffset = getHiPELiteral(HiPELiteralsMD, "P_NSP_LIMIT");
3946 if (Is64Bit) {
3947 SPReg = X86::RSP;
3948 PReg = X86::RBP;
3949 LEAop = X86::LEA64r;
3950 CMPop = X86::CMP64rm;
3951 CALLop = X86::CALL64pcrel32;
3952 } else {
3953 SPReg = X86::ESP;
3954 PReg = X86::EBP;
3955 LEAop = X86::LEA32r;
3956 CMPop = X86::CMP32rm;
3957 CALLop = X86::CALLpcrel32;
3958 }
3959
3960 ScratchReg = GetScratchRegister(Is64Bit, IsLP64, MF, true);
3961 assert(!MF.getRegInfo().isLiveIn(ScratchReg) &&
3962 "HiPE prologue scratch register is live-in");
3963
3964 // Create new MBB for StackCheck:
3965 addRegOffset(BuildMI(stackCheckMBB, DL, TII.get(LEAop), ScratchReg), SPReg,
3966 false, -MaxStack);
3967 // SPLimitOffset is in a fixed heap location (pointed by BP).
3968 addRegOffset(BuildMI(stackCheckMBB, DL, TII.get(CMPop)).addReg(ScratchReg),
3969 PReg, false, SPLimitOffset);
3970 BuildMI(stackCheckMBB, DL, TII.get(X86::JCC_1))
3971 .addMBB(&PrologueMBB)
3973
3974 // Create new MBB for IncStack:
3975 BuildMI(incStackMBB, DL, TII.get(CALLop)).addExternalSymbol("inc_stack_0");
3976 addRegOffset(BuildMI(incStackMBB, DL, TII.get(LEAop), ScratchReg), SPReg,
3977 false, -MaxStack);
3978 addRegOffset(BuildMI(incStackMBB, DL, TII.get(CMPop)).addReg(ScratchReg),
3979 PReg, false, SPLimitOffset);
3980 BuildMI(incStackMBB, DL, TII.get(X86::JCC_1))
3981 .addMBB(incStackMBB)
3983
3984 stackCheckMBB->addSuccessor(&PrologueMBB, {99, 100});
3985 stackCheckMBB->addSuccessor(incStackMBB, {1, 100});
3986 incStackMBB->addSuccessor(&PrologueMBB, {99, 100});
3987 incStackMBB->addSuccessor(incStackMBB, {1, 100});
3988 }
3989#ifdef EXPENSIVE_CHECKS
3990 MF.verify();
3991#endif
3992}
3993
3994bool X86FrameLowering::adjustStackWithPops(MachineBasicBlock &MBB,
3996 const DebugLoc &DL,
3997 int Offset) const {
3998 if (Offset <= 0)
3999 return false;
4000
4001 if (Offset % SlotSize)
4002 return false;
4003
4004 int NumPops = Offset / SlotSize;
4005 // This is only worth it if we have at most 2 pops.
4006 if (NumPops != 1 && NumPops != 2)
4007 return false;
4008
4009 // Handle only the trivial case where the adjustment directly follows
4010 // a call. This is the most common one, anyway.
4011 if (MBBI == MBB.begin())
4012 return false;
4013 MachineBasicBlock::iterator Prev = std::prev(MBBI);
4014 if (!Prev->isCall() || !Prev->getOperand(1).isRegMask())
4015 return false;
4016
4017 unsigned Regs[2];
4018 unsigned FoundRegs = 0;
4019
4020 const MachineRegisterInfo &MRI = MBB.getParent()->getRegInfo();
4021 const MachineOperand &RegMask = Prev->getOperand(1);
4022
4023 auto &RegClass =
4024 Is64Bit ? X86::GR64_NOREX_NOSPRegClass : X86::GR32_NOREX_NOSPRegClass;
4025 // Try to find up to NumPops free registers.
4026 for (auto Candidate : RegClass) {
4027 // Poor man's liveness:
4028 // Since we're immediately after a call, any register that is clobbered
4029 // by the call and not defined by it can be considered dead.
4030 if (!RegMask.clobbersPhysReg(Candidate))
4031 continue;
4032
4033 // Don't clobber reserved registers
4034 if (MRI.isReserved(Candidate))
4035 continue;
4036
4037 bool IsDef = false;
4038 for (const MachineOperand &MO : Prev->implicit_operands()) {
4039 if (MO.isReg() && MO.isDef() &&
4040 TRI->isSuperOrSubRegisterEq(MO.getReg(), Candidate)) {
4041 IsDef = true;
4042 break;
4043 }
4044 }
4045
4046 if (IsDef)
4047 continue;
4048
4049 Regs[FoundRegs++] = Candidate;
4050 if (FoundRegs == (unsigned)NumPops)
4051 break;
4052 }
4053
4054 if (FoundRegs == 0)
4055 return false;
4056
4057 // If we found only one free register, but need two, reuse the same one twice.
4058 while (FoundRegs < (unsigned)NumPops)
4059 Regs[FoundRegs++] = Regs[0];
4060
4061 for (int i = 0; i < NumPops; ++i)
4062 BuildMI(MBB, MBBI, DL, TII.get(STI.is64Bit() ? X86::POP64r : X86::POP32r),
4063 Regs[i]);
4064
4065 return true;
4066}
4067
4071 bool reserveCallFrame = hasReservedCallFrame(MF);
4072 unsigned Opcode = I->getOpcode();
4073 bool isDestroy = Opcode == TII.getCallFrameDestroyOpcode();
4074 DebugLoc DL = I->getDebugLoc(); // copy DebugLoc as I will be erased.
4075 uint64_t Amount = TII.getFrameSize(*I);
4076 uint64_t InternalAmt = (isDestroy || Amount) ? TII.getFrameAdjustment(*I) : 0;
4077 I = MBB.erase(I);
4078 auto InsertPos = skipDebugInstructionsForward(I, MBB.end());
4079
4080 // Try to avoid emitting dead SP adjustments if the block end is unreachable,
4081 // typically because the function is marked noreturn (abort, throw,
4082 // assert_fail, etc).
4083 if (isDestroy && blockEndIsUnreachable(MBB, I))
4084 return I;
4085
4086 if (!reserveCallFrame) {
4087 // If the stack pointer can be changed after prologue, turn the
4088 // adjcallstackup instruction into a 'sub ESP, <amt>' and the
4089 // adjcallstackdown instruction into 'add ESP, <amt>'
4090
4091 // We need to keep the stack aligned properly. To do this, we round the
4092 // amount of space needed for the outgoing arguments up to the next
4093 // alignment boundary.
4094 Amount = alignTo(Amount, getStackAlign());
4095
4096 const Function &F = MF.getFunction();
4097 bool WindowsCFI = MF.getTarget().getMCAsmInfo().usesWindowsCFI();
4098 bool DwarfCFI = !WindowsCFI && MF.needsFrameMoves();
4099
4100 // If we have any exception handlers in this function, and we adjust
4101 // the SP before calls, we may need to indicate this to the unwinder
4102 // using GNU_ARGS_SIZE. Note that this may be necessary even when
4103 // Amount == 0, because the preceding function may have set a non-0
4104 // GNU_ARGS_SIZE.
4105 // TODO: We don't need to reset this between subsequent functions,
4106 // if it didn't change.
4107 bool HasDwarfEHHandlers = !WindowsCFI && !MF.getLandingPads().empty();
4108
4109 if (HasDwarfEHHandlers && !isDestroy &&
4111 BuildCFI(MBB, InsertPos, DL,
4112 MCCFIInstruction::createGnuArgsSize(nullptr, Amount));
4113
4114 if (Amount == 0)
4115 return I;
4116
4117 // Factor out the amount that gets handled inside the sequence
4118 // (Pushes of argument for frame setup, callee pops for frame destroy)
4119 Amount -= InternalAmt;
4120
4121 // TODO: This is needed only if we require precise CFA.
4122 // If this is a callee-pop calling convention, emit a CFA adjust for
4123 // the amount the callee popped.
4124 if (isDestroy && InternalAmt && DwarfCFI && !hasFP(MF))
4125 BuildCFI(MBB, InsertPos, DL,
4126 MCCFIInstruction::createAdjustCfaOffset(nullptr, -InternalAmt));
4127
4128 // Add Amount to SP to destroy a frame, or subtract to setup.
4129 int64_t StackAdjustment = isDestroy ? Amount : -Amount;
4130 int64_t CfaAdjustment = StackAdjustment;
4131
4132 if (StackAdjustment) {
4133 // Merge with any previous or following adjustment instruction. Note: the
4134 // instructions merged with here do not have CFI, so their stack
4135 // adjustments do not feed into CfaAdjustment
4136
4137 auto CalcCfaAdjust = [&CfaAdjustment](MachineBasicBlock::iterator PI,
4138 int64_t Offset) {
4139 CfaAdjustment += Offset;
4140 };
4141 auto CalcNewOffset = [&StackAdjustment](int64_t Offset) {
4142 return StackAdjustment + Offset;
4143 };
4144 StackAdjustment =
4145 mergeSPUpdates(MBB, InsertPos, CalcCfaAdjust, CalcNewOffset, true);
4146 StackAdjustment =
4147 mergeSPUpdates(MBB, InsertPos, CalcCfaAdjust, CalcNewOffset, false);
4148
4149 if (StackAdjustment) {
4150 if (!(F.hasMinSize() &&
4151 adjustStackWithPops(MBB, InsertPos, DL, StackAdjustment)))
4152 BuildStackAdjustment(MBB, InsertPos, DL, StackAdjustment,
4153 /*InEpilogue=*/false);
4154 }
4155 }
4156
4157 if (DwarfCFI && !hasFP(MF) && CfaAdjustment) {
4158 // If we don't have FP, but need to generate unwind information,
4159 // we need to set the correct CFA offset after the stack adjustment.
4160 // How much we adjust the CFA offset depends on whether we're emitting
4161 // CFI only for EH purposes or for debugging. EH only requires the CFA
4162 // offset to be correct at each call site, while for debugging we want
4163 // it to be more precise.
4164
4165 // TODO: When not using precise CFA, we also need to adjust for the
4166 // InternalAmt here.
4167 BuildCFI(
4168 MBB, InsertPos, DL,
4169 MCCFIInstruction::createAdjustCfaOffset(nullptr, -CfaAdjustment));
4170 }
4171
4172 return I;
4173 }
4174
4175 if (InternalAmt) {
4178 while (CI != B && !std::prev(CI)->isCall())
4179 --CI;
4180 BuildStackAdjustment(MBB, CI, DL, -InternalAmt, /*InEpilogue=*/false);
4181 }
4182
4183 return I;
4184}
4185
4187 assert(MBB.getParent() && "Block is not attached to a function!");
4188 const MachineFunction &MF = *MBB.getParent();
4189 if (!MBB.isLiveIn(X86::EFLAGS))
4190 return true;
4191
4192 // If stack probes have to loop inline or call, that will clobber EFLAGS.
4193 // FIXME: we could allow cases that will use emitStackProbeInlineGenericBlock.
4195 const X86TargetLowering &TLI = *STI.getTargetLowering();
4196 if (TLI.hasInlineStackProbe(MF) || TLI.hasStackProbeSymbol(MF))
4197 return false;
4198
4200 return !TRI->hasStackRealignment(MF) && !X86FI->hasSwiftAsyncContext();
4201}
4202
4204 assert(MBB.getParent() && "Block is not attached to a function!");
4205
4206 // Win64 has strict requirements in terms of epilogue and we are
4207 // not taking a chance at messing with them.
4208 // I.e., unless this block is already an exit block, we can't use
4209 // it as an epilogue.
4210 if (STI.isTargetWin64() && !MBB.succ_empty() && !MBB.isReturnBlock())
4211 return false;
4212
4213 // Swift async context epilogue has a BTR instruction that clobbers parts of
4214 // EFLAGS.
4215 const MachineFunction &MF = *MBB.getParent();
4218
4219 if (canUseLEAForSPInEpilogue(*MBB.getParent()))
4220 return true;
4221
4222 // If we cannot use LEA to adjust SP, we may need to use ADD, which
4223 // clobbers the EFLAGS. Check that we do not need to preserve it,
4224 // otherwise, conservatively assume this is not
4225 // safe to insert the epilogue here.
4227}
4228
4230 // If we may need to emit frameless compact unwind information, give
4231 // up as this is currently broken: PR25614.
4232 bool CompactUnwind =
4234 return (MF.getFunction().hasFnAttribute(Attribute::NoUnwind) || hasFP(MF) ||
4235 !CompactUnwind) &&
4236 // The lowering of segmented stack and HiPE only support entry
4237 // blocks as prologue blocks: PR26107. This limitation may be
4238 // lifted if we fix:
4239 // - adjustForSegmentedStacks
4240 // - adjustForHiPEPrologue
4242 !MF.shouldSplitStack();
4243}
4244
4247 const DebugLoc &DL, bool RestoreSP) const {
4248 assert(STI.isTargetWindowsMSVC() && "funclets only supported in MSVC env");
4249 assert(STI.isTargetWin32() && "EBP/ESI restoration only required on win32");
4250 assert(STI.is32Bit() && !Uses64BitFramePtr &&
4251 "restoring EBP/ESI on non-32-bit target");
4252
4253 MachineFunction &MF = *MBB.getParent();
4254 Register FramePtr = TRI->getFrameRegister(MF);
4255 Register BasePtr = TRI->getBaseRegister();
4256 WinEHFuncInfo &FuncInfo = *MF.getWinEHFuncInfo();
4258 MachineFrameInfo &MFI = MF.getFrameInfo();
4259
4260 // FIXME: Don't set FrameSetup flag in catchret case.
4261
4262 int FI = FuncInfo.EHRegNodeFrameIndex;
4263 int EHRegSize = MFI.getObjectSize(FI);
4264
4265 if (RestoreSP) {
4266 // MOV32rm -EHRegSize(%ebp), %esp
4267 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::MOV32rm), X86::ESP),
4268 X86::EBP, true, -EHRegSize)
4270 }
4271
4272 Register UsedReg;
4273 int EHRegOffset = getFrameIndexReference(MF, FI, UsedReg).getFixed();
4274 int EndOffset = -EHRegOffset - EHRegSize;
4275 FuncInfo.EHRegNodeEndOffset = EndOffset;
4276
4277 if (UsedReg == FramePtr) {
4278 // ADD $offset, %ebp
4279 unsigned ADDri = getADDriOpcode(false);
4280 BuildMI(MBB, MBBI, DL, TII.get(ADDri), FramePtr)
4282 .addImm(EndOffset)
4284 ->getOperand(3)
4285 .setIsDead();
4286 assert(EndOffset >= 0 &&
4287 "end of registration object above normal EBP position!");
4288 } else if (UsedReg == BasePtr) {
4289 // LEA offset(%ebp), %esi
4290 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::LEA32r), BasePtr),
4291 FramePtr, false, EndOffset)
4293 // MOV32rm SavedEBPOffset(%esi), %ebp
4294 assert(X86FI->getHasSEHFramePtrSave());
4295 int Offset =
4296 getFrameIndexReference(MF, X86FI->getSEHFramePtrSaveIndex(), UsedReg)
4297 .getFixed();
4298 assert(UsedReg == BasePtr);
4299 addRegOffset(BuildMI(MBB, MBBI, DL, TII.get(X86::MOV32rm), FramePtr),
4300 UsedReg, true, Offset)
4302 } else {
4303 llvm_unreachable("32-bit frames with WinEH must use FramePtr or BasePtr");
4304 }
4305 return MBBI;
4306}
4307
4309 return TRI->getSlotSize();
4310}
4311
4316
4320 Register FrameRegister = RI->getFrameRegister(MF);
4321 if (getInitialCFARegister(MF) == FrameRegister &&
4323 DwarfFrameBase FrameBase;
4324 FrameBase.Kind = DwarfFrameBase::CFA;
4325 FrameBase.Location.Offset =
4327 return FrameBase;
4328 }
4329
4330 return DwarfFrameBase{DwarfFrameBase::Register, {FrameRegister}};
4331}
4332
4333namespace {
4334// Struct used by orderFrameObjects to help sort the stack objects.
4335struct X86FrameSortingObject {
4336 bool IsValid = false; // true if we care about this Object.
4337 unsigned ObjectIndex = 0; // Index of Object into MFI list.
4338 unsigned ObjectSize = 0; // Size of Object in bytes.
4339 Align ObjectAlignment = Align(1); // Alignment of Object in bytes.
4340 unsigned ObjectNumUses = 0; // Object static number of uses.
4341};
4342
4343// The comparison function we use for std::sort to order our local
4344// stack symbols. The current algorithm is to use an estimated
4345// "density". This takes into consideration the size and number of
4346// uses each object has in order to roughly minimize code size.
4347// So, for example, an object of size 16B that is referenced 5 times
4348// will get higher priority than 4 4B objects referenced 1 time each.
4349// It's not perfect and we may be able to squeeze a few more bytes out of
4350// it (for example : 0(esp) requires fewer bytes, symbols allocated at the
4351// fringe end can have special consideration, given their size is less
4352// important, etc.), but the algorithmic complexity grows too much to be
4353// worth the extra gains we get. This gets us pretty close.
4354// The final order leaves us with objects with highest priority going
4355// at the end of our list.
4356struct X86FrameSortingComparator {
4357 inline bool operator()(const X86FrameSortingObject &A,
4358 const X86FrameSortingObject &B) const {
4359 uint64_t DensityAScaled, DensityBScaled;
4360
4361 // For consistency in our comparison, all invalid objects are placed
4362 // at the end. This also allows us to stop walking when we hit the
4363 // first invalid item after it's all sorted.
4364 if (!A.IsValid)
4365 return false;
4366 if (!B.IsValid)
4367 return true;
4368
4369 // The density is calculated by doing :
4370 // (double)DensityA = A.ObjectNumUses / A.ObjectSize
4371 // (double)DensityB = B.ObjectNumUses / B.ObjectSize
4372 // Since this approach may cause inconsistencies in
4373 // the floating point <, >, == comparisons, depending on the floating
4374 // point model with which the compiler was built, we're going
4375 // to scale both sides by multiplying with
4376 // A.ObjectSize * B.ObjectSize. This ends up factoring away
4377 // the division and, with it, the need for any floating point
4378 // arithmetic.
4379 DensityAScaled = static_cast<uint64_t>(A.ObjectNumUses) *
4380 static_cast<uint64_t>(B.ObjectSize);
4381 DensityBScaled = static_cast<uint64_t>(B.ObjectNumUses) *
4382 static_cast<uint64_t>(A.ObjectSize);
4383
4384 // If the two densities are equal, prioritize highest alignment
4385 // objects. This allows for similar alignment objects
4386 // to be packed together (given the same density).
4387 // There's room for improvement here, also, since we can pack
4388 // similar alignment (different density) objects next to each
4389 // other to save padding. This will also require further
4390 // complexity/iterations, and the overall gain isn't worth it,
4391 // in general. Something to keep in mind, though.
4392 if (DensityAScaled == DensityBScaled)
4393 return A.ObjectAlignment < B.ObjectAlignment;
4394
4395 return DensityAScaled < DensityBScaled;
4396 }
4397};
4398} // namespace
4399
4400// Order the symbols in the local stack.
4401// We want to place the local stack objects in some sort of sensible order.
4402// The heuristic we use is to try and pack them according to static number
4403// of uses and size of object in order to minimize code size.
4405 const MachineFunction &MF, SmallVectorImpl<int> &ObjectsToAllocate) const {
4406 const MachineFrameInfo &MFI = MF.getFrameInfo();
4407
4408 // Don't waste time if there's nothing to do.
4409 if (ObjectsToAllocate.empty())
4410 return;
4411
4412 // Create an array of all MFI objects. We won't need all of these
4413 // objects, but we're going to create a full array of them to make
4414 // it easier to index into when we're counting "uses" down below.
4415 // We want to be able to easily/cheaply access an object by simply
4416 // indexing into it, instead of having to search for it every time.
4417 std::vector<X86FrameSortingObject> SortingObjects(MFI.getObjectIndexEnd());
4418
4419 // Walk the objects we care about and mark them as such in our working
4420 // struct.
4421 for (auto &Obj : ObjectsToAllocate) {
4422 SortingObjects[Obj].IsValid = true;
4423 SortingObjects[Obj].ObjectIndex = Obj;
4424 SortingObjects[Obj].ObjectAlignment = MFI.getObjectAlign(Obj);
4425 // Set the size.
4426 int ObjectSize = MFI.getObjectSize(Obj);
4427 if (ObjectSize == 0)
4428 // Variable size. Just use 4.
4429 SortingObjects[Obj].ObjectSize = 4;
4430 else
4431 SortingObjects[Obj].ObjectSize = ObjectSize;
4432 }
4433
4434 // Count the number of uses for each object.
4435 for (auto &MBB : MF) {
4436 for (auto &MI : MBB) {
4437 if (MI.isDebugInstr())
4438 continue;
4439 for (const MachineOperand &MO : MI.operands()) {
4440 // Check to see if it's a local stack symbol.
4441 if (!MO.isFI())
4442 continue;
4443 int Index = MO.getIndex();
4444 // Check to see if it falls within our range, and is tagged
4445 // to require ordering.
4446 if (Index >= 0 && Index < MFI.getObjectIndexEnd() &&
4447 SortingObjects[Index].IsValid)
4448 SortingObjects[Index].ObjectNumUses++;
4449 }
4450 }
4451 }
4452
4453 // Sort the objects using X86FrameSortingAlgorithm (see its comment for
4454 // info).
4455 llvm::stable_sort(SortingObjects, X86FrameSortingComparator());
4456
4457 // Now modify the original list to represent the final order that
4458 // we want. The order will depend on whether we're going to access them
4459 // from the stack pointer or the frame pointer. For SP, the list should
4460 // end up with the END containing objects that we want with smaller offsets.
4461 // For FP, it should be flipped.
4462 int i = 0;
4463 for (auto &Obj : SortingObjects) {
4464 // All invalid items are sorted at the end, so it's safe to stop.
4465 if (!Obj.IsValid)
4466 break;
4467 ObjectsToAllocate[i++] = Obj.ObjectIndex;
4468 }
4469
4470 // Flip it if we're accessing off of the FP.
4471 if (!TRI->hasStackRealignment(MF) && hasFP(MF))
4472 std::reverse(ObjectsToAllocate.begin(), ObjectsToAllocate.end());
4473}
4474
4475unsigned
4477 // RDX, the parent frame pointer, is homed into 16(%rsp) in the prologue.
4478 unsigned Offset = 16;
4479 // RBP is immediately pushed.
4480 Offset += SlotSize;
4481 // All callee-saved registers are then pushed.
4482 Offset += MF.getInfo<X86MachineFunctionInfo>()->getCalleeSavedFrameSize();
4483 // Every funclet allocates enough stack space for the largest outgoing call.
4484 Offset += getWinEHFuncletFrameSize(MF);
4485 return Offset;
4486}
4487
4489 MachineFunction &MF, RegScavenger *RS) const {
4490 // Mark the function as not having WinCFI. We will set it back to true in
4491 // emitPrologue if it gets called and emits CFI.
4492 MF.setHasWinCFI(false);
4493
4494 MachineFrameInfo &MFI = MF.getFrameInfo();
4495 // If the frame is big enough that we might need to scavenge a register to
4496 // handle huge offsets, reserve a stack slot for that now.
4497 if (!isInt<32>(MFI.estimateStackSize(MF))) {
4498 int FI = MFI.CreateStackObject(SlotSize, Align(SlotSize), false);
4499 RS->addScavengingFrameIndex(FI);
4500 }
4501
4502 // If we are using Windows x64 CFI, ensure that the stack is always 8 byte
4503 // aligned. The format doesn't support misaligned stack adjustments.
4506
4507 // If this function isn't doing Win64-style C++ EH, we don't need to do
4508 // anything.
4509 if (STI.is64Bit() && MF.hasEHFunclets() &&
4512 adjustFrameForMsvcCxxEh(MF);
4513 }
4514}
4515
4516void X86FrameLowering::adjustFrameForMsvcCxxEh(MachineFunction &MF) const {
4517 // Win64 C++ EH needs to allocate the UnwindHelp object at some fixed offset
4518 // relative to RSP after the prologue. Find the offset of the last fixed
4519 // object, so that we can allocate a slot immediately following it. If there
4520 // were no fixed objects, use offset -SlotSize, which is immediately after the
4521 // return address. Fixed objects have negative frame indices.
4522 MachineFrameInfo &MFI = MF.getFrameInfo();
4523 WinEHFuncInfo &EHInfo = *MF.getWinEHFuncInfo();
4524 int64_t MinFixedObjOffset = -SlotSize;
4525 for (int I = MFI.getObjectIndexBegin(); I < 0; ++I)
4526 MinFixedObjOffset = std::min(MinFixedObjOffset, MFI.getObjectOffset(I));
4527
4528 for (WinEHTryBlockMapEntry &TBME : EHInfo.TryBlockMap) {
4529 for (WinEHHandlerType &H : TBME.HandlerArray) {
4530 int FrameIndex = H.CatchObj.FrameIndex;
4531 if ((FrameIndex != INT_MAX) && MFI.getObjectOffset(FrameIndex) == 0) {
4532 // Ensure alignment.
4533 unsigned Align = MFI.getObjectAlign(FrameIndex).value();
4534 MinFixedObjOffset -= std::abs(MinFixedObjOffset) % Align;
4535 MinFixedObjOffset -= MFI.getObjectSize(FrameIndex);
4536 MFI.setObjectOffset(FrameIndex, MinFixedObjOffset);
4537 }
4538 }
4539 }
4540
4541 // Ensure alignment.
4542 MinFixedObjOffset -= std::abs(MinFixedObjOffset) % 8;
4543 int64_t UnwindHelpOffset = MinFixedObjOffset - SlotSize;
4544 int UnwindHelpFI =
4545 MFI.CreateFixedObject(SlotSize, UnwindHelpOffset, /*IsImmutable=*/false);
4546 EHInfo.UnwindHelpFrameIdx = UnwindHelpFI;
4547
4548 // Store -2 into UnwindHelp on function entry. We have to scan forwards past
4549 // other frame setup instructions.
4550 MachineBasicBlock &MBB = MF.front();
4551 auto MBBI = MBB.begin();
4552 while (MBBI != MBB.end() && MBBI->getFlag(MachineInstr::FrameSetup))
4553 ++MBBI;
4554
4556 addFrameReference(BuildMI(MBB, MBBI, DL, TII.get(X86::MOV64mi32)),
4557 UnwindHelpFI)
4558 .addImm(-2);
4559}
4560
4562 MachineFunction &MF, RegScavenger *RS) const {
4563 auto *X86FI = MF.getInfo<X86MachineFunctionInfo>();
4564
4565 if (STI.is32Bit() && MF.hasEHFunclets())
4567 // We have emitted prolog and epilog. Don't need stack pointer saving
4568 // instruction any more.
4569 if (MachineInstr *MI = X86FI->getStackPtrSaveMI()) {
4570 MI->eraseFromParent();
4571 X86FI->setStackPtrSaveMI(nullptr);
4572 }
4573}
4574
4576 MachineFunction &MF) const {
4577 // 32-bit functions have to restore stack pointers when control is transferred
4578 // back to the parent function. These blocks are identified as eh pads that
4579 // are not funclet entries.
4580 bool IsSEH = isAsynchronousEHPersonality(
4582 for (MachineBasicBlock &MBB : MF) {
4583 bool NeedsRestore = MBB.isEHPad() && !MBB.isEHFuncletEntry();
4584 if (NeedsRestore)
4586 /*RestoreSP=*/IsSEH);
4587 }
4588}
4589
4590// Compute the alignment gap between current SP after spilling FP/BP and the
4591// next properly aligned stack offset.
4593 const TargetRegisterClass *RC,
4594 unsigned NumSpilledRegs) {
4596 unsigned AllocSize = TRI->getSpillSize(*RC) * NumSpilledRegs;
4597 Align StackAlign = MF.getSubtarget().getFrameLowering()->getStackAlign();
4598 unsigned AlignedSize = alignTo(AllocSize, StackAlign);
4599 return AlignedSize - AllocSize;
4600}
4601
4602void X86FrameLowering::spillFPBPUsingSP(MachineFunction &MF,
4604 Register FP, Register BP,
4605 int SPAdjust) const {
4606 assert(FP.isValid() || BP.isValid());
4607
4608 MachineBasicBlock *MBB = BeforeMI->getParent();
4609 DebugLoc DL = BeforeMI->getDebugLoc();
4610
4611 // Spill FP.
4612 if (FP.isValid()) {
4613 BuildMI(*MBB, BeforeMI, DL,
4614 TII.get(getPUSHOpcode(MF.getSubtarget<X86Subtarget>())))
4615 .addReg(FP);
4616 }
4617
4618 // Spill BP.
4619 if (BP.isValid()) {
4620 BuildMI(*MBB, BeforeMI, DL,
4621 TII.get(getPUSHOpcode(MF.getSubtarget<X86Subtarget>())))
4622 .addReg(BP);
4623 }
4624
4625 // Make sure SP is aligned.
4626 if (SPAdjust)
4627 emitSPUpdate(*MBB, BeforeMI, DL, -SPAdjust, false);
4628
4629 // Emit unwinding information.
4630 if (FP.isValid() && needsDwarfCFI(MF)) {
4631 // Emit .cfi_remember_state to remember old frame.
4632 unsigned CFIIndex =
4634 BuildMI(*MBB, BeforeMI, DL, TII.get(TargetOpcode::CFI_INSTRUCTION))
4635 .addCFIIndex(CFIIndex);
4636
4637 // Setup new CFA value with DW_CFA_def_cfa_expression:
4638 // DW_OP_breg7+offset, DW_OP_deref, DW_OP_consts 16, DW_OP_plus
4639 SmallString<64> CfaExpr;
4640 uint8_t buffer[16];
4641 int Offset = SPAdjust;
4642 if (BP.isValid())
4643 Offset += TRI->getSpillSize(*TRI->getMinimalPhysRegClass(BP));
4644 // If BeforeMI is a frame setup instruction, we need to adjust the position
4645 // and offset of the new cfi instruction.
4646 if (TII.isFrameSetup(*BeforeMI)) {
4647 Offset += alignTo(TII.getFrameSize(*BeforeMI), getStackAlign());
4648 BeforeMI = std::next(BeforeMI);
4649 }
4650 Register StackPtr = TRI->getStackRegister();
4651 if (STI.isTarget64BitILP32())
4653 unsigned DwarfStackPtr = TRI->getDwarfRegNum(StackPtr, true);
4654 CfaExpr.push_back((uint8_t)(dwarf::DW_OP_breg0 + DwarfStackPtr));
4655 CfaExpr.append(buffer, buffer + encodeSLEB128(Offset, buffer));
4656 CfaExpr.push_back(dwarf::DW_OP_deref);
4657 CfaExpr.push_back(dwarf::DW_OP_consts);
4658 CfaExpr.append(buffer, buffer + encodeSLEB128(SlotSize * 2, buffer));
4659 CfaExpr.push_back((uint8_t)dwarf::DW_OP_plus);
4660
4661 SmallString<64> DefCfaExpr;
4662 DefCfaExpr.push_back(dwarf::DW_CFA_def_cfa_expression);
4663 DefCfaExpr.append(buffer, buffer + encodeSLEB128(CfaExpr.size(), buffer));
4664 DefCfaExpr.append(CfaExpr.str());
4665 BuildCFI(*MBB, BeforeMI, DL,
4666 MCCFIInstruction::createEscape(nullptr, DefCfaExpr.str()),
4668 }
4669}
4670
4671void X86FrameLowering::restoreFPBPUsingSP(MachineFunction &MF,
4673 Register FP, Register BP,
4674 int SPAdjust) const {
4675 assert(FP.isValid() || BP.isValid());
4676
4677 // Adjust SP so it points to spilled FP or BP.
4678 MachineBasicBlock *MBB = AfterMI->getParent();
4679 MachineBasicBlock::iterator Pos = std::next(AfterMI);
4680 DebugLoc DL = AfterMI->getDebugLoc();
4681 if (SPAdjust)
4682 emitSPUpdate(*MBB, Pos, DL, SPAdjust, false);
4683
4684 // Restore BP.
4685 if (BP.isValid()) {
4686 BuildMI(*MBB, Pos, DL,
4687 TII.get(getPOPOpcode(MF.getSubtarget<X86Subtarget>())), BP);
4688 }
4689
4690 // Restore FP.
4691 if (FP.isValid()) {
4692 BuildMI(*MBB, Pos, DL,
4693 TII.get(getPOPOpcode(MF.getSubtarget<X86Subtarget>())), FP);
4694
4695 // Emit unwinding information.
4696 if (needsDwarfCFI(MF)) {
4697 // Restore original frame with .cfi_restore_state.
4698 unsigned CFIIndex =
4700 BuildMI(*MBB, Pos, DL, TII.get(TargetOpcode::CFI_INSTRUCTION))
4701 .addCFIIndex(CFIIndex);
4702 }
4703 }
4704}
4705
4706void X86FrameLowering::saveAndRestoreFPBPUsingSP(
4708 MachineBasicBlock::iterator AfterMI, bool SpillFP, bool SpillBP) const {
4709 assert(SpillFP || SpillBP);
4710
4711 Register FP, BP;
4712 const TargetRegisterClass *RC;
4713 unsigned NumRegs = 0;
4714
4715 if (SpillFP) {
4716 FP = TRI->getFrameRegister(MF);
4717 if (STI.isTarget64BitILP32())
4719 RC = TRI->getMinimalPhysRegClass(FP);
4720 ++NumRegs;
4721 }
4722 if (SpillBP) {
4723 BP = TRI->getBaseRegister();
4724 if (STI.isTarget64BitILP32())
4725 BP = Register(getX86SubSuperRegister(BP, 64));
4726 RC = TRI->getMinimalPhysRegClass(BP);
4727 ++NumRegs;
4728 }
4729 int SPAdjust = computeFPBPAlignmentGap(MF, RC, NumRegs);
4730
4731 spillFPBPUsingSP(MF, BeforeMI, FP, BP, SPAdjust);
4732 restoreFPBPUsingSP(MF, AfterMI, FP, BP, SPAdjust);
4733}
4734
4735bool X86FrameLowering::skipSpillFPBP(
4737 if (MI->getOpcode() == X86::LCMPXCHG16B_SAVE_RBX) {
4738 // The pseudo instruction LCMPXCHG16B_SAVE_RBX is generated in the form
4739 // SaveRbx = COPY RBX
4740 // SaveRbx = LCMPXCHG16B_SAVE_RBX ..., SaveRbx, implicit-def rbx
4741 // And later LCMPXCHG16B_SAVE_RBX is expanded to restore RBX from SaveRbx.
4742 // We should skip this instruction sequence.
4743 int FI;
4744 Register Reg;
4745 while (!(MI->getOpcode() == TargetOpcode::COPY &&
4746 MI->getOperand(1).getReg() == X86::RBX) &&
4747 !((Reg = TII.isStoreToStackSlot(*MI, FI)) && Reg == X86::RBX))
4748 ++MI;
4749 return true;
4750 }
4751 return false;
4752}
4753
4755 const TargetRegisterInfo *TRI, bool &AccessFP,
4756 bool &AccessBP) {
4757 AccessFP = AccessBP = false;
4758 if (FP) {
4759 if (MI.findRegisterUseOperandIdx(FP, TRI, false) != -1 ||
4760 MI.findRegisterDefOperandIdx(FP, TRI, false, true) != -1)
4761 AccessFP = true;
4762 }
4763 if (BP) {
4764 if (MI.findRegisterUseOperandIdx(BP, TRI, false) != -1 ||
4765 MI.findRegisterDefOperandIdx(BP, TRI, false, true) != -1)
4766 AccessBP = true;
4767 }
4768 return AccessFP || AccessBP;
4769}
4770
4771// Invoke instruction has been lowered to normal function call. We try to figure
4772// out if MI comes from Invoke.
4773// Do we have any better method?
4774static bool isInvoke(const MachineInstr &MI, bool InsideEHLabels) {
4775 if (!MI.isCall())
4776 return false;
4777 if (InsideEHLabels)
4778 return true;
4779
4780 const MachineBasicBlock *MBB = MI.getParent();
4781 if (!MBB->hasEHPadSuccessor())
4782 return false;
4783
4784 // Check if there is another call instruction from MI to the end of MBB.
4786 for (++MBBI; MBBI != ME; ++MBBI)
4787 if (MBBI->isCall())
4788 return false;
4789 return true;
4790}
4791
4792/// Given the live range of FP or BP (DefMI, KillMI), check if there is any
4793/// interfered stack access in the range, usually generated by register spill.
4794void X86FrameLowering::checkInterferedAccess(
4796 MachineBasicBlock::reverse_iterator KillMI, bool SpillFP,
4797 bool SpillBP) const {
4798 if (DefMI == KillMI)
4799 return;
4800 if (TRI->hasBasePointer(MF)) {
4801 if (!SpillBP)
4802 return;
4803 } else {
4804 if (!SpillFP)
4805 return;
4806 }
4807
4808 auto MI = KillMI;
4809 while (MI != DefMI) {
4810 if (any_of(MI->operands(),
4811 [](const MachineOperand &MO) { return MO.isFI(); }))
4812 MF.getContext().reportError(SMLoc(),
4813 "Interference usage of base pointer/frame "
4814 "pointer.");
4815 MI++;
4816 }
4817}
4818
4819/// If a function uses base pointer and the base pointer is clobbered by inline
4820/// asm, RA doesn't detect this case, and after the inline asm, the base pointer
4821/// contains garbage value.
4822/// For example if a 32b x86 function uses base pointer esi, and esi is
4823/// clobbered by following inline asm
4824/// asm("rep movsb" : "+D"(ptr), "+S"(x), "+c"(c)::"memory");
4825/// We need to save esi before the asm and restore it after the asm.
4826///
4827/// The problem can also occur to frame pointer if there is a function call, and
4828/// the callee uses a different calling convention and clobbers the fp.
4829///
4830/// Because normal frame objects (spill slots) are accessed through fp/bp
4831/// register, so we can't spill fp/bp to normal spill slots.
4832///
4833/// FIXME: There are 2 possible enhancements:
4834/// 1. In many cases there are different physical registers not clobbered by
4835/// inline asm, we can use one of them as base pointer. Or use a virtual
4836/// register as base pointer and let RA allocate a physical register to it.
4837/// 2. If there is no other instructions access stack with fp/bp from the
4838/// inline asm to the epilog, and no cfi requirement for a correct fp, we can
4839/// skip the save and restore operations.
4841 Register FP, BP;
4843 if (TFI.hasFP(MF))
4844 FP = TRI->getFrameRegister(MF);
4845 if (TRI->hasBasePointer(MF))
4846 BP = TRI->getBaseRegister();
4847
4848 // Currently only inline asm and function call can clobbers fp/bp. So we can
4849 // do some quick test and return early.
4850 if (!MF.hasInlineAsm()) {
4852 if (!X86FI->getFPClobberedByCall())
4853 FP = 0;
4854 if (!X86FI->getBPClobberedByCall())
4855 BP = 0;
4856 }
4857 if (!FP && !BP)
4858 return;
4859
4860 for (MachineBasicBlock &MBB : MF) {
4861 bool InsideEHLabels = false;
4862 auto MI = MBB.rbegin(), ME = MBB.rend();
4863 auto TermMI = MBB.getFirstTerminator();
4864 if (TermMI == MBB.begin())
4865 continue;
4866 MI = *(std::prev(TermMI));
4867
4868 while (MI != ME) {
4869 // Skip frame setup/destroy instructions.
4870 // Skip Invoke (call inside try block) instructions.
4871 // Skip instructions handled by target.
4872 if (MI->getFlag(MachineInstr::MIFlag::FrameSetup) ||
4874 isInvoke(*MI, InsideEHLabels) || skipSpillFPBP(MF, MI)) {
4875 ++MI;
4876 continue;
4877 }
4878
4879 if (MI->getOpcode() == TargetOpcode::EH_LABEL) {
4880 InsideEHLabels = !InsideEHLabels;
4881 ++MI;
4882 continue;
4883 }
4884
4885 bool AccessFP, AccessBP;
4886 // Check if fp or bp is used in MI.
4887 if (!isFPBPAccess(*MI, FP, BP, TRI, AccessFP, AccessBP)) {
4888 ++MI;
4889 continue;
4890 }
4891
4892 // Look for the range [DefMI, KillMI] in which fp or bp is defined and
4893 // used.
4894 bool FPLive = false, BPLive = false;
4895 bool SpillFP = false, SpillBP = false;
4896 auto DefMI = MI, KillMI = MI;
4897 do {
4898 SpillFP |= AccessFP;
4899 SpillBP |= AccessBP;
4900
4901 // Maintain FPLive and BPLive.
4902 if (FPLive && MI->findRegisterDefOperandIdx(FP, TRI, false, true) != -1)
4903 FPLive = false;
4904 if (FP && MI->findRegisterUseOperandIdx(FP, TRI, false) != -1)
4905 FPLive = true;
4906 if (BPLive && MI->findRegisterDefOperandIdx(BP, TRI, false, true) != -1)
4907 BPLive = false;
4908 if (BP && MI->findRegisterUseOperandIdx(BP, TRI, false) != -1)
4909 BPLive = true;
4910
4911 DefMI = MI++;
4912 } while ((MI != ME) &&
4913 (FPLive || BPLive ||
4914 isFPBPAccess(*MI, FP, BP, TRI, AccessFP, AccessBP)));
4915
4916 // Don't need to save/restore if FP is accessed through llvm.frameaddress.
4917 if (FPLive && !SpillBP)
4918 continue;
4919
4920 // If the bp is clobbered by a call, we should save and restore outside of
4921 // the frame setup instructions.
4922 if (KillMI->isCall() && DefMI != ME) {
4923 auto FrameSetup = std::next(DefMI);
4924 // Look for frame setup instruction toward the start of the BB.
4925 // If we reach another call instruction, it means no frame setup
4926 // instruction for the current call instruction.
4927 while (FrameSetup != ME && !TII.isFrameSetup(*FrameSetup) &&
4928 !FrameSetup->isCall())
4929 ++FrameSetup;
4930 // If a frame setup instruction is found, we need to find out the
4931 // corresponding frame destroy instruction.
4932 if (FrameSetup != ME && TII.isFrameSetup(*FrameSetup) &&
4933 (TII.getFrameSize(*FrameSetup) ||
4934 TII.getFrameAdjustment(*FrameSetup))) {
4935 while (!TII.isFrameInstr(*KillMI))
4936 --KillMI;
4937 DefMI = FrameSetup;
4938 MI = DefMI;
4939 ++MI;
4940 }
4941 }
4942
4943 checkInterferedAccess(MF, DefMI, KillMI, SpillFP, SpillBP);
4944
4945 // Call target function to spill and restore FP and BP registers.
4946 saveAndRestoreFPBPUsingSP(MF, &(*DefMI), &(*KillMI), SpillFP, SpillBP);
4947 }
4948 }
4949}
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
static const uint64_t kSplitStackAvailable
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Module.h This file contains the declarations for the Module class.
static cl::opt< int > PageSize("imp-null-check-page-size", cl::desc("The page size of the target in bytes"), cl::init(4096), cl::Hidden)
This file implements the LivePhysRegs utility for tracking liveness of physical registers.
static bool isTailCallOpcode(unsigned Opc)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define H(x, y, z)
Definition MD5.cpp:56
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static constexpr MCPhysReg FPReg
static constexpr MCPhysReg SPReg
This file declares the machine register scavenger class.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
static bool is64Bit(const char *name)
static unsigned calculateSetFPREG(uint64_t SPAdjust)
static unsigned GetScratchRegister(bool Is64Bit, bool IsLP64, const MachineFunction &MF, bool Primary)
GetScratchRegister - Get a temp register for performing work in the segmented stack and the Erlang/Hi...
static unsigned getADDriOpcode(bool IsLP64)
static unsigned getPUSH2Opcode(const X86Subtarget &ST)
static uint64_t getUnusedLocalAreaPadding(const MachineFrameInfo &MFI)
Returns the number of bytes between the end of the fixed and callee-save area and the first local obj...
static unsigned getLEArOpcode(bool IsLP64)
static unsigned getSUBriOpcode(bool IsLP64)
static bool flagsNeedToBePreservedBeforeTheTerminators(const MachineBasicBlock &MBB)
Check if the flags need to be preserved before the terminators.
static bool isFPBPAccess(const MachineInstr &MI, Register FP, Register BP, const TargetRegisterInfo *TRI, bool &AccessFP, bool &AccessBP)
static const TargetRegisterClass * getCalleeSavedSpillRC(MCRegister Reg, const X86Subtarget &STI, const TargetRegisterInfo &TRI)
static bool isOpcodeRep(unsigned Opcode)
Return true if an opcode is part of the REP group of instructions.
static unsigned getANDriOpcode(bool IsLP64, int64_t Imm)
static bool isEAXLiveIn(MachineBasicBlock &MBB)
static int computeFPBPAlignmentGap(MachineFunction &MF, const TargetRegisterClass *RC, unsigned NumSpilledRegs)
static unsigned getADDrrOpcode(bool IsLP64)
static bool HasNestArgument(const MachineFunction *MF)
static unsigned getPOPOpcode(const X86Subtarget &ST)
static bool isInvoke(const MachineInstr &MI, bool InsideEHLabels)
static unsigned getPOP2Opcode(const X86Subtarget &ST)
static unsigned getHiPELiteral(NamedMDNode *HiPELiteralsMD, const StringRef LiteralName)
Lookup an ERTS parameter in the !hipe.literals named metadata node.
static bool blockEndIsUnreachable(const MachineBasicBlock &MBB, MachineBasicBlock::const_iterator MBBI)
static unsigned getSUBrrOpcode(bool IsLP64)
static unsigned getPUSHOpcode(const X86Subtarget &ST)
constexpr uint64_t MaxSPChunk
static const unsigned FramePtr
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
reverse_iterator rend() const
Definition ArrayRef.h:133
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
reverse_iterator rbegin() const
Definition ArrayRef.h:132
BitVector & reset()
Reset all bits in the bitvector.
Definition BitVector.h:409
int find_first() const
Returns the index of the first set bit, -1 if none of the bits are set.
Definition BitVector.h:317
BitVector & set()
Set all bits in the bitvector.
Definition BitVector.h:366
int find_next(unsigned Prev) const
Returns the index of the next set bit following the "Prev" bit.
Definition BitVector.h:324
iterator_range< const_set_bits_iterator > set_bits() const
Definition BitVector.h:159
static constexpr BranchProbability getOne()
static constexpr BranchProbability getZero()
The CalleeSavedInfo class tracks the information need to locate where a callee saved register is in t...
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
A debug info location.
Definition DebugLoc.h:126
unsigned size() const
Definition DenseMap.h:733
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
bool hasPersonalityFn() const
Check whether this function has a personality function.
Definition Function.h:890
Constant * getPersonalityFn() const
Get the personality function associated with this function.
AttributeList getAttributes() const
Return the attribute list for this Function.
Definition Function.h:329
size_t arg_size() const
Definition Function.h:886
bool needsUnwindTableEntry() const
True if this function needs an unwind table.
Definition Function.h:667
const Argument * const_arg_iterator
Definition Function.h:74
bool isVarArg() const
isVarArg - Return true if this function takes a variable number of arguments.
Definition Function.h:230
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:730
Module * getParent()
Get the module that this global value is contained inside of...
bool usesWindowsCFI() const
Definition MCAsmInfo.h:675
static MCCFIInstruction createDefCfaRegister(MCSymbol *L, unsigned Register, SMLoc Loc={})
.cfi_def_cfa_register modifies a rule for computing CFA.
Definition MCDwarf.h:635
static MCCFIInstruction createGnuArgsSize(MCSymbol *L, int64_t Size, SMLoc Loc={})
A special wrapper for .cfi_escape that indicates GNU_ARGS_SIZE.
Definition MCDwarf.h:765
static MCCFIInstruction createRestore(MCSymbol *L, unsigned Register, SMLoc Loc={})
.cfi_restore says that the rule for Register is now the same as it was at the beginning of the functi...
Definition MCDwarf.h:725
static MCCFIInstruction cfiDefCfa(MCSymbol *L, unsigned Register, int64_t Offset, SMLoc Loc={})
.cfi_def_cfa defines a rule for computing CFA as: take address from Register and add Offset to it.
Definition MCDwarf.h:628
static MCCFIInstruction createOffset(MCSymbol *L, unsigned Register, int64_t Offset, SMLoc Loc={})
.cfi_offset Previous value of Register is saved at offset Offset from CFA.
Definition MCDwarf.h:670
static MCCFIInstruction createRememberState(MCSymbol *L, SMLoc Loc={})
.cfi_remember_state Save all current rules for all registers.
Definition MCDwarf.h:745
OpType getOperation() const
Definition MCDwarf.h:833
static MCCFIInstruction cfiDefCfaOffset(MCSymbol *L, int64_t Offset, SMLoc Loc={})
.cfi_def_cfa_offset modifies a rule for computing CFA.
Definition MCDwarf.h:643
static MCCFIInstruction createEscape(MCSymbol *L, StringRef Vals, SMLoc Loc={}, StringRef Comment="")
.cfi_escape Allows the user to add arbitrary bytes to the unwind info.
Definition MCDwarf.h:756
static MCCFIInstruction createAdjustCfaOffset(MCSymbol *L, int64_t Adjustment, SMLoc Loc={})
.cfi_adjust_cfa_offset Same as .cfi_def_cfa_offset, but Offset is a relative value that is added/subt...
Definition MCDwarf.h:651
static MCCFIInstruction createRestoreState(MCSymbol *L, SMLoc Loc={})
.cfi_restore_state Restore the previously saved state.
Definition MCDwarf.h:750
const MCObjectFileInfo * getObjectFileInfo() const
Definition MCContext.h:413
const MCRegisterInfo * getRegisterInfo() const
Definition MCContext.h:411
LLVM_ABI void reportError(SMLoc L, const Twine &Msg)
MCSection * getCompactUnwindSection() const
MCRegAliasIterator enumerates all registers aliasing Reg.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
virtual int64_t getDwarfRegNum(MCRegister Reg, bool isEH) const
Map a target register to an equivalent dwarf register number.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
Metadata node.
Definition Metadata.h:1081
A single uniqued string.
Definition Metadata.h:733
LLVM_ABI StringRef getString() const
Definition Metadata.cpp:615
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
MachineInstrBundleIterator< const MachineInstr > const_iterator
iterator_range< livein_iterator > liveins() const
const BasicBlock * getBasicBlock() const
Return the LLVM basic block that this instance corresponded to originally.
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI DebugLoc findDebugLoc(instr_iterator MBBI)
Find the next valid DebugLoc starting at MBBI, skipping any debug instructions.
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
void addLiveIn(MCRegister PhysReg, LaneBitmask LaneMask=LaneBitmask::getAll())
Adds the specified register as a live in.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LLVM_ABI instr_iterator erase(instr_iterator I)
Remove an instruction from the instruction list and delete it.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
@ LQR_Live
Register is known to be (at least partially) live.
void setMachineBlockAddressTaken()
Set this block to indicate that its address is used as something other than the target of a terminato...
LLVM_ABI bool isLiveIn(MCRegister Reg, LaneBitmask LaneMask=LaneBitmask::getAll()) const
Return true if the specified register is in the live in set.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool needsSplitStackProlog() const
Return true if this function requires a split stack prolog, even if it uses no stack space.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
bool hasVarSizedObjects() const
This method may be called any time after instruction selection is complete to determine if the stack ...
uint64_t getStackSize() const
Return the number of bytes that must be allocated to hold all of the fixed size frame objects.
bool adjustsStack() const
Return true if this function adjusts the stack – e.g., when calling another function.
LLVM_ABI int CreateStackObject(uint64_t Size, Align Alignment, bool isSpillSlot, const AllocaInst *Alloca=nullptr, uint8_t ID=0)
Create a new statically sized stack object, returning a nonnegative identifier to represent it.
LLVM_ABI void ensureMaxAlignment(Align Alignment)
Make sure the function's frame is at least Align bytes aligned.
bool hasCalls() const
Return true if the current function has any function calls.
bool isFrameAddressTaken() const
This method may be called any time after instruction selection is complete to determine if there is a...
Align getMaxAlign() const
Return alignment of this function's frame.
void setObjectOffset(int ObjectIdx, int64_t SPOffset)
Set the stack frame offset of the specified object.
uint64_t getMaxCallFrameSize() const
Return the maximum size of a call frame that must be allocated for an outgoing function call.
bool hasPatchPoint() const
This method may be called any time after instruction selection is complete to determine if there is a...
bool hasOpaqueSPAdjustment() const
Returns true if the function contains opaque dynamic stack adjustments.
void setCVBytesOfCalleeSavedRegisters(unsigned S)
LLVM_ABI uint64_t estimateStackSize(const MachineFunction &MF) const
Estimate and return the size of the stack frame.
Align getObjectAlign(int ObjectIdx) const
Return the alignment of the specified stack object.
int64_t getObjectSize(int ObjectIdx) const
Return the size of the specified object.
bool hasStackMap() const
This method may be called any time after instruction selection is complete to determine if there is a...
LLVM_ABI int CreateSpillStackObject(uint64_t Size, Align Alignment, TargetStackID::Value StackID=TargetStackID::Default)
Create a new statically sized stack object that represents a spill slot, returning a nonnegative iden...
const std::vector< CalleeSavedInfo > & getCalleeSavedInfo() const
Returns a reference to call saved info vector for the current function.
bool isVariableSizedObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a variable sized object.
int getObjectIndexEnd() const
Return one past the maximum frame object index.
bool hasCopyImplyingStackAdjustment() const
Returns true if the function contains operations which will lower down to instructions which manipula...
bool hasStackObjects() const
Return true if there are any stack objects in this function.
LLVM_ABI int CreateFixedSpillStackObject(uint64_t Size, int64_t SPOffset, bool IsImmutable=false)
Create a spill slot at a fixed location on the stack.
uint8_t getStackID(int ObjectIdx) const
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
void setStackSize(uint64_t Size)
Set the size of the stack.
bool isFixedObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a fixed stack object.
int getObjectIndexBegin() const
Return the minimum frame object index.
bool isDeadObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a dead object.
void setOffsetAdjustment(int64_t Adj)
Set the correction for frame offsets.
const WinEHFuncInfo * getWinEHFuncInfo() const
getWinEHFuncInfo - Return information about how the current function uses Windows exception handling.
unsigned addFrameInst(const MCCFIInstruction &Inst)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
const std::vector< MCCFIInstruction > & getFrameInstructions() const
Returns a reference to a list of cfi instructions in the function's prologue.
bool hasInlineAsm() const
Returns true if the function contains any inline assembly.
void makeDebugValueSubstitution(DebugInstrOperandPair, DebugInstrOperandPair, unsigned SubReg=0)
Create a substitution between one <instr,operand> value to a different, new value.
bool needsFrameMoves() const
True if this function needs frame moves for debug or exceptions.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_front(MachineBasicBlock *MBB)
const char * createExternalSymbolName(StringRef Name)
Allocate a string and populate it with the given external symbol name.
MCContext & getContext() const
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
bool verify(Pass *p=nullptr, const char *Banner=nullptr, raw_ostream *OS=nullptr, bool AbortOnError=true) const
Run the current MachineFunction through the machine code verifier, useful for debugger use.
Function & getFunction()
Return the LLVM function that this machine code represents.
const std::vector< LandingPadInfo > & getLandingPads() const
Return a reference to the landing pad info for the current function.
BasicBlockListType::iterator iterator
bool disableFramePointerElim() const
Returns true if frame pointer elimination should be disabled for this function.
bool shouldSplitStack() const
Should we be emitting segmented stack stuff for the function.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const MachineBasicBlock & front() const
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & addExternalSymbol(const char *FnName, unsigned TargetFlags=0) const
const MachineInstrBuilder & addCFIIndex(unsigned CFIIndex) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & setMIFlag(MachineInstr::MIFlag Flag) const
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
unsigned getNumOperands() const
Retuns the total number of operands.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI unsigned getDebugInstrNum()
Fetch the instruction number of this MachineInstr.
const MachineOperand & getOperand(unsigned i) const
@ MOVolatile
The memory access is volatile.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
const GlobalValue * getGlobal() const
int64_t getImm() const
MachineBasicBlock * getMBB() const
void setIsDead(bool Val=true)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
static bool clobbersPhysReg(const uint32_t *RegMask, MCRegister PhysReg)
clobbersPhysReg - Returns true if this RegMask clobbers PhysReg.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
bool isReserved(MCRegister PhysReg) const
isReserved - Returns true when PhysReg is a reserved register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool isLiveIn(Register Reg) const
NamedMDNode * getNamedMetadata(StringRef Name) const
Return the first NamedMDNode in the module with the specified name.
Definition Module.cpp:301
WinX64EHUnwindMode getWinX64EHUnwindMode() const
Get how unwind information should be generated for x64 Windows.
Definition Module.cpp:1012
unsigned getCodeViewFlag() const
Returns the CodeView Version by checking module flags.
Definition Module.cpp:617
Represent a mutable reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:294
iterator end() const
Definition ArrayRef.h:339
iterator begin() const
Definition ArrayRef.h:338
A tuple of MDNodes.
Definition Metadata.h:1797
LLVM_ABI MDNode * getOperand(unsigned i) const
LLVM_ABI unsigned getNumOperands() const
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isValid() const
Definition Register.h:112
SlotIndex - An opaque wrapper around machine indexes.
Definition SlotIndexes.h:66
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
void append(StringRef RHS)
Append from a StringRef.
Definition SmallString.h:68
StringRef str() const
Explicit conversion to StringRef.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
int64_t getFixed() const
Returns the fixed component of the stack.
Definition TypeSize.h:46
static StackOffset getFixed(int64_t Fixed)
Definition TypeSize.h:39
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static constexpr size_t npos
Definition StringRef.h:58
unsigned getStackAlignment() const
getStackAlignment - This method returns the number of bytes to which the stack pointer must be aligne...
bool hasFP(const MachineFunction &MF) const
hasFP - Return true if the specified function should have a dedicated frame pointer register.
virtual void determineCalleeSaves(MachineFunction &MF, BitVector &SavedRegs, RegScavenger *RS=nullptr) const
This method determines which of the registers reported by TargetRegisterInfo::getCalleeSavedRegs() sh...
int getOffsetOfLocalArea() const
getOffsetOfLocalArea - This method returns the offset of the local area from the stack pointer on ent...
TargetFrameLowering(StackDirection D, Align StackAl, int LAO, Align TransAl=Align(1), bool StackReal=true)
Align getStackAlign() const
getStackAlignment - This method returns the number of bytes to which the stack pointer must be aligne...
const Triple & getTargetTriple() const
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
TargetOptions Options
CodeModel::Model getCodeModel() const
Returns the code model.
SwiftAsyncFramePointerMode SwiftAsyncFramePointer
Control when and how the Swift async frame pointer bit should be set.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
virtual Register getFrameRegister(const MachineFunction &MF) const =0
Debug information queries.
virtual const TargetFrameLowering * getFrameLowering() const
virtual const TargetRegisterInfo * getRegisterInfo() const =0
Return the target's register information.
bool isUEFI() const
Tests whether the OS is UEFI.
Definition Triple.h:774
bool isOSWindows() const
Tests whether the OS is Windows.
Definition Triple.h:777
Value wrapper in the Metadata hierarchy.
Definition Metadata.h:471
Value * getValue() const
Definition Metadata.h:510
bool has128ByteRedZone(const MachineFunction &MF) const
Return true if the function has a redzone (accessible bytes past the frame of the top of stack functi...
void spillFPBP(MachineFunction &MF) const override
If a function uses base pointer and the base pointer is clobbered by inline asm, RA doesn't detect th...
bool canSimplifyCallFramePseudos(const MachineFunction &MF) const override
canSimplifyCallFramePseudos - If there is a reserved call frame, the call frame pseudos can be simpli...
bool needsFrameIndexResolution(const MachineFunction &MF) const override
X86FrameLowering(const X86Subtarget &STI, MaybeAlign StackAlignOverride)
const X86RegisterInfo * TRI
void emitEpilogue(MachineFunction &MF, MachineBasicBlock &MBB) const override
bool hasFPImpl(const MachineFunction &MF) const override
hasFPImpl - Return true if the specified function should have a dedicated frame pointer register.
MachineBasicBlock::iterator restoreWin32EHStackPointers(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, bool RestoreSP=false) const
Sets up EBP and optionally ESI based on the incoming EBP value.
int getInitialCFAOffset(const MachineFunction &MF) const override
Return initial CFA offset value i.e.
bool canUseAsPrologue(const MachineBasicBlock &MBB) const override
Check whether or not the given MBB can be used as a prologue for the target.
bool hasReservedCallFrame(const MachineFunction &MF) const override
hasReservedCallFrame - Under normal circumstances, when a frame pointer is not required,...
void emitStackProbe(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, bool InProlog, std::optional< MachineFunction::DebugInstrOperandPair > InstrNum=std::nullopt) const
Emit target stack probe code.
void processFunctionBeforeFrameFinalized(MachineFunction &MF, RegScavenger *RS) const override
processFunctionBeforeFrameFinalized - This method is called immediately before the specified function...
void emitCalleeSavedFrameMoves(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, bool IsPrologue) const
void determineCalleeSaves(MachineFunction &MF, BitVector &SavedRegs, RegScavenger *RS=nullptr) const override
This method determines which of the registers reported by TargetRegisterInfo::getCalleeSavedRegs() sh...
int64_t mergeSPAdd(MachineBasicBlock &MBB, MachineBasicBlock::iterator &MBBI, int64_t AddOffset, bool doMergeWithPrevious) const
Equivalent to: mergeSPUpdates(MBB, MBBI, [AddOffset](int64_t Offset) { return AddOffset + Offset; }...
StackOffset getFrameIndexReferenceSP(const MachineFunction &MF, int FI, Register &SPReg, int Adjustment) const
bool assignCalleeSavedSpillSlots(MachineFunction &MF, const TargetRegisterInfo *TRI, std::vector< CalleeSavedInfo > &CSI) const override
assignCalleeSavedSpillSlots - Allows target to override spill slot assignment logic.
bool enableShrinkWrapping(const MachineFunction &MF) const override
Returns true if the target will correctly handle shrink wrapping.
StackOffset getFrameIndexReference(const MachineFunction &MF, int FI, Register &FrameReg) const override
getFrameIndexReference - This method should return the base register and offset used to reference a f...
void inlineStackProbe(MachineFunction &MF, MachineBasicBlock &PrologMBB) const override
Replace a StackProbe inline-stub with the actual probe code inline.
bool restoreCalleeSavedRegisters(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, MutableArrayRef< CalleeSavedInfo > CSI, const TargetRegisterInfo *TRI) const override
restoreCalleeSavedRegisters - Issues instruction(s) to restore all callee saved registers and returns...
const X86InstrInfo & TII
MachineBasicBlock::iterator eliminateCallFramePseudoInstr(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const override
This method is called during prolog/epilog code insertion to eliminate call frame setup and destroy p...
void emitSPUpdate(MachineBasicBlock &MBB, MachineBasicBlock::iterator &MBBI, const DebugLoc &DL, int64_t NumBytes, bool InEpilogue) const
Emit a series of instructions to increment / decrement the stack pointer by a constant value.
bool canUseAsEpilogue(const MachineBasicBlock &MBB) const override
Check whether or not the given MBB can be used as a epilogue for the target.
bool Is64Bit
Is64Bit implies that x86_64 instructions are available.
Register getInitialCFARegister(const MachineFunction &MF) const override
Return initial CFA register value i.e.
bool Uses64BitFramePtr
True if the 64-bit frame or stack pointer should be used.
unsigned getWinEHParentFrameOffset(const MachineFunction &MF) const override
void adjustForSegmentedStacks(MachineFunction &MF, MachineBasicBlock &PrologueMBB) const override
Adjust the prologue to have the function use segmented stacks.
DwarfFrameBase getDwarfFrameBase(const MachineFunction &MF) const override
Return the frame base information to be encoded in the DWARF subprogram debug info.
void emitCalleeSavedFrameMovesFullCFA(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI) const override
Emits Dwarf Info specifying offsets of callee saved registers and frame pointer.
int getWin64EHFrameIndexRef(const MachineFunction &MF, int FI, Register &SPReg) const
bool canUseLEAForSPInEpilogue(const MachineFunction &MF) const
Check that LEA can be used on SP in an epilogue sequence for MF.
bool stackProbeFunctionModifiesSP() const override
Does the stack probe function call return with a modified stack pointer?
void orderFrameObjects(const MachineFunction &MF, SmallVectorImpl< int > &ObjectsToAllocate) const override
Order the symbols in the local stack.
void BuildCFI(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, const MCCFIInstruction &CFIInst, MachineInstr::MIFlag Flag=MachineInstr::NoFlags) const
Wraps up getting a CFI index and building a MachineInstr for it.
void emitPrologue(MachineFunction &MF, MachineBasicBlock &MBB) const override
emitProlog/emitEpilog - These methods insert prolog and epilog code into the function.
void processFunctionBeforeFrameIndicesReplaced(MachineFunction &MF, RegScavenger *RS) const override
processFunctionBeforeFrameIndicesReplaced - This method is called immediately before MO_FrameIndex op...
StackOffset getFrameIndexReferencePreferSP(const MachineFunction &MF, int FI, Register &FrameReg, bool IgnoreSPUpdates) const override
Same as getFrameIndexReference, except that the stack pointer (as opposed to the frame pointer) will ...
void restoreWinEHStackPointersInParent(MachineFunction &MF) const
bool spillCalleeSavedRegisters(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, ArrayRef< CalleeSavedInfo > CSI, const TargetRegisterInfo *TRI) const override
spillCalleeSavedRegisters - Issues instruction(s) to spill all callee saved registers and returns tru...
void adjustForHiPEPrologue(MachineFunction &MF, MachineBasicBlock &PrologueMBB) const override
Erlang programs may need a special prologue to handle the stack size they might need at runtime.
const X86Subtarget & STI
X86MachineFunctionInfo - This class is derived from MachineFunction and contains private X86 target-s...
bool isCandidateForPush2Pop2(Register Reg) const
void setRestoreBasePointer(const MachineFunction *MF)
DenseMap< int, unsigned > & getWinEHXMMSlotInfo()
MachineInstr * getStackPtrSaveMI() const
AMXProgModelEnum getAMXProgModel() const
void setStackPtrSaveMI(MachineInstr *MI)
void setCalleeSavedFrameSize(unsigned bytes)
const X86TargetLowering * getTargetLowering() const override
bool isTargetWindowsCoreCLR() const
self_iterator getIterator()
Definition ilist_node.h:123
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
uint16_t StackAdjustment(const RuntimeFunction &RF)
StackAdjustment - calculated stack adjustment in words.
Definition ARMWinEH.h:200
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ HiPE
Used by the High-Performance Erlang Compiler (HiPE).
Definition CallingConv.h:53
@ X86_INTR
x86 hardware interrupt context.
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ Tail
Attemps to make calls as fast as possible while guaranteeing that tail call optimization can always b...
Definition CallingConv.h:76
@ X86_FastCall
'fast' analog of X86_StdCall.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
@ MO_GOTPCREL
MO_GOTPCREL - On a symbol operand this indicates that the immediate is offset to the GOT entry for th...
unsigned getMOVriOpcode(bool Use64BitReg, int64_t Imm)
Return a MOVri opcode for materializing Imm into a 32- or 64-bit GPR.
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
void stable_sort(R &&Range)
Definition STLExtras.h:2132
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool isAligned(Align Lhs, uint64_t SizeInBytes)
Checks that SizeInBytes is a multiple of the alignment.
Definition Alignment.h:134
MCRegister getX86SubSuperRegister(MCRegister Reg, unsigned Size, bool High=false)
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
static const MachineInstrBuilder & addFrameReference(const MachineInstrBuilder &MIB, int FI, int Offset=0, bool mem=true)
addFrameReference - This function is used to add a reference to the base of an abstract object on the...
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
IterT skipDebugInstructionsForward(IterT It, IterT End, bool SkipPseudoOp=true)
Increment It until it points to a non-debug instruction or to End and return the resulting iterator.
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
static bool isFuncletReturnInstr(const MachineInstr &MI)
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
@ DeploymentBased
Determine whether to set the bit statically or dynamically based on the deployment target.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr RegState getDefRegState(bool B)
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
bool requireWinX64UnwindV3(const MachineFunction &MF)
Returns true when MF must use Windows x64 Unwind V3: the module is in V3 mode, or the function needs ...
LLVM_ABI EHPersonality classifyEHPersonality(const Value *Pers)
See if the given exception handling personality function is one that we understand.
IterT skipDebugInstructionsBackward(IterT It, IterT Begin, bool SkipPseudoOp=true)
Decrement It until it points to a non-debug instruction or to Begin and return the resulting iterator...
bool isAsynchronousEHPersonality(EHPersonality Pers)
Returns true if this personality function catches asynchronous exceptions.
@ DwarfCFI
DWARF-like instruction based exceptions.
Definition CodeGen.h:57
unsigned encodeSLEB128(int64_t Value, raw_ostream &OS, unsigned PadTo=0)
Utility function to encode a SLEB128 value to an output stream.
Definition LEB128.h:24
auto count_if(R &&Range, UnaryPredicate P)
Wrapper function around std::count_if to count the number of times an element satisfying a given pred...
Definition STLExtras.h:2035
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
Definition Sequence.h:341
LLVM_ABI void computeAndAddLiveIns(LivePhysRegs &LiveRegs, MachineBasicBlock &MBB)
Convenience function combining computeLiveIns() and addLiveIns().
unsigned encodeULEB128(uint64_t Value, raw_ostream &OS, unsigned PadTo=0)
Utility function to encode a ULEB128 value to an output stream.
Definition LEB128.h:79
static const MachineInstrBuilder & addRegOffset(const MachineInstrBuilder &MIB, Register Reg, bool isKill, int Offset)
addRegOffset - This function is used to add a memory reference of the form [Reg + Offset],...
constexpr RegState getUndefRegState(bool B)
void fullyRecomputeLiveIns(ArrayRef< MachineBasicBlock * > MBBs)
Convenience function for recomputing live-in's for a set of MBBs until the computation converges.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
Pair of physical register and lane mask.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
union llvm::TargetFrameLowering::DwarfFrameBase::@004076321055032247336074224075335064105264310375 Location
enum llvm::TargetFrameLowering::DwarfFrameBase::FrameBaseKind Kind
SmallVector< WinEHTryBlockMapEntry, 4 > TryBlockMap
SmallVector< WinEHHandlerType, 1 > HandlerArray