LLVM 24.0.0git
ARMSubtarget.cpp
Go to the documentation of this file.
1//===-- ARMSubtarget.cpp - ARM Subtarget Information ----------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the ARM specific subclass of TargetSubtargetInfo.
10//
11//===----------------------------------------------------------------------===//
12
13#include "ARM.h"
14
15#include "ARMCallLowering.h"
16#include "ARMFrameLowering.h"
17#include "ARMInstrInfo.h"
18#include "ARMLegalizerInfo.h"
19#include "ARMRegisterBankInfo.h"
20#include "ARMSubtarget.h"
21#include "ARMTargetMachine.h"
23#include "Thumb1FrameLowering.h"
24#include "Thumb1InstrInfo.h"
25#include "Thumb2InstrInfo.h"
26#include "llvm/ADT/BitVector.h"
27#include "llvm/ADT/StringRef.h"
28#include "llvm/ADT/Twine.h"
32#include "llvm/IR/Function.h"
33#include "llvm/IR/GlobalValue.h"
34#include "llvm/MC/MCAsmInfo.h"
41
42using namespace llvm;
43
44#define DEBUG_TYPE "arm-subtarget"
45
46#define GET_SUBTARGETINFO_TARGET_DESC
47#define GET_SUBTARGETINFO_CTOR
48#include "ARMGenSubtargetInfo.inc"
49
50static cl::opt<bool>
51UseFusedMulOps("arm-use-mulops",
52 cl::init(true), cl::Hidden);
53
58
59static cl::opt<ITMode>
60 IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT),
61 cl::values(clEnumValN(DefaultIT, "arm-default-it",
62 "Generate any type of IT block"),
63 clEnumValN(RestrictedIT, "arm-restrict-it",
64 "Disallow complex IT blocks")));
65
66/// ForceFastISel - Use the fast-isel, even for subtargets where it is not
67/// currently supported (for testing only).
68static cl::opt<bool>
69ForceFastISel("arm-force-fast-isel",
70 cl::init(false), cl::Hidden);
71
72/// initializeSubtargetDependencies - Initializes using a CPU and feature string
73/// so that we can use initializer lists for subtarget initialization.
75 StringRef FS) {
76 initSubtargetFeatures(CPU, FS);
77 return *this;
78}
79
80ARMFrameLowering *ARMSubtarget::initializeFrameLowering(StringRef CPU,
81 StringRef FS) {
83 if (STI.isThumb1Only())
84 return (ARMFrameLowering *)new Thumb1FrameLowering(STI);
85
86 return new ARMFrameLowering(STI);
87}
88
89ARMSubtarget::ARMSubtarget(const Triple &TT, const std::string &CPU,
90 const std::string &FS,
91 const ARMBaseTargetMachine &TM, bool IsLittle,
93 bool MinSize, DenormalMode DM)
94 : ARMGenSubtargetInfo(TT, CPU, /*TuneCPU*/ CPU, FS),
98 FrameLowering(initializeFrameLowering(CPU, FS)),
99 // At this point initializeSubtargetDependencies has been called so
100 // we can query directly.
101 InstrInfo(isThumb1Only() ? (ARMBaseInstrInfo *)new Thumb1InstrInfo(*this)
102 : !isThumb() ? (ARMBaseInstrInfo *)new ARMInstrInfo(*this)
103 : (ARMBaseInstrInfo *)new Thumb2InstrInfo(*this)),
104 TLInfo(TM, *this) {
105
106 CallLoweringInfo.reset(new ARMCallLowering(*getTargetLowering()));
107 Legalizer.reset(new ARMLegalizerInfo(*this));
108
109 auto *RBI = new ARMRegisterBankInfo(*getRegisterInfo());
110
111 // FIXME: At this point, we can't rely on Subtarget having RBI.
112 // It's awkward to mix passing RBI and the Subtarget; should we pass
113 // TII/TRI as well?
114 InstSelector.reset(createARMInstructionSelector(TM, *this, *RBI));
115
116 RegBankInfo.reset(RBI);
117}
118
120 return CallLoweringInfo.get();
121}
122
124 return InstSelector.get();
125}
126
128 return Legalizer.get();
129}
130
132 return RegBankInfo.get();
133}
134
136 const Triple &TT = getTargetTriple();
137 if (TT.isOSBinFormatMachO()) {
138 // Uses VFP for Thumb libfuncs if available.
139 if (isThumb() && hasVFP2Base() && hasARMOps() && !useSoftFloat()) {
140 // clang-format off
141 static const struct {
142 const RTLIB::Libcall Op;
143 const RTLIB::LibcallImpl Impl;
144 } LibraryCalls[] = {
145 // Single-precision floating-point arithmetic.
146 { RTLIB::ADD_F32, RTLIB::impl___addsf3vfp },
147 { RTLIB::SUB_F32, RTLIB::impl___subsf3vfp },
148 { RTLIB::MUL_F32, RTLIB::impl___mulsf3vfp },
149 { RTLIB::DIV_F32, RTLIB::impl___divsf3vfp },
150
151 // Double-precision floating-point arithmetic.
152 { RTLIB::ADD_F64, RTLIB::impl___adddf3vfp },
153 { RTLIB::SUB_F64, RTLIB::impl___subdf3vfp },
154 { RTLIB::MUL_F64, RTLIB::impl___muldf3vfp },
155 { RTLIB::DIV_F64, RTLIB::impl___divdf3vfp },
156
157 // Single-precision comparisons.
158 { RTLIB::OEQ_F32, RTLIB::impl___eqsf2vfp },
159 { RTLIB::UNE_F32, RTLIB::impl___nesf2vfp },
160 { RTLIB::OLT_F32, RTLIB::impl___ltsf2vfp },
161 { RTLIB::OLE_F32, RTLIB::impl___lesf2vfp },
162 { RTLIB::OGE_F32, RTLIB::impl___gesf2vfp },
163 { RTLIB::OGT_F32, RTLIB::impl___gtsf2vfp },
164 { RTLIB::UO_F32, RTLIB::impl___unordsf2vfp },
165
166 // Double-precision comparisons.
167 { RTLIB::OEQ_F64, RTLIB::impl___eqdf2vfp },
168 { RTLIB::UNE_F64, RTLIB::impl___nedf2vfp },
169 { RTLIB::OLT_F64, RTLIB::impl___ltdf2vfp },
170 { RTLIB::OLE_F64, RTLIB::impl___ledf2vfp },
171 { RTLIB::OGE_F64, RTLIB::impl___gedf2vfp },
172 { RTLIB::OGT_F64, RTLIB::impl___gtdf2vfp },
173 { RTLIB::UO_F64, RTLIB::impl___unorddf2vfp },
174
175 // Floating-point to integer conversions.
176 // i64 conversions are done via library routines even when generating VFP
177 // instructions, so use the same ones.
178 { RTLIB::FPTOSINT_F64_I32, RTLIB::impl___fixdfsivfp },
179 { RTLIB::FPTOUINT_F64_I32, RTLIB::impl___fixunsdfsivfp },
180 { RTLIB::FPTOSINT_F32_I32, RTLIB::impl___fixsfsivfp },
181 { RTLIB::FPTOUINT_F32_I32, RTLIB::impl___fixunssfsivfp },
182
183 // Conversions between floating types.
184 { RTLIB::FPROUND_F64_F32, RTLIB::impl___truncdfsf2vfp },
185 { RTLIB::FPEXT_F32_F64, RTLIB::impl___extendsfdf2vfp },
186
187 // Integer to floating-point conversions.
188 // i64 conversions are done via library routines even when generating VFP
189 // instructions, so use the same ones.
190 // FIXME: There appears to be some naming inconsistency in ARM libgcc:
191 // e.g., __floatunsidf vs. __floatunssidfvfp.
192 { RTLIB::SINTTOFP_I32_F64, RTLIB::impl___floatsidfvfp },
193 { RTLIB::UINTTOFP_I32_F64, RTLIB::impl___floatunssidfvfp },
194 { RTLIB::SINTTOFP_I32_F32, RTLIB::impl___floatsisfvfp },
195 { RTLIB::UINTTOFP_I32_F32, RTLIB::impl___floatunssisfvfp },
196 };
197 // clang-format on
198
199 for (const auto &LC : LibraryCalls)
200 Info.setLibcallImpl(LC.Op, LC.Impl);
201 }
202 }
203
204 static const struct {
205 const RTLIB::Libcall Op;
206 const RTLIB::LibcallImpl Impl;
207 } AEABISelected[] = {
208 // Double-precision arithmetic.
209 {RTLIB::ADD_F64, RTLIB::impl___aeabi_dadd},
210 {RTLIB::DIV_F64, RTLIB::impl___aeabi_ddiv},
211 {RTLIB::MUL_F64, RTLIB::impl___aeabi_dmul},
212 {RTLIB::SUB_F64, RTLIB::impl___aeabi_dsub},
213 // Double-precision comparisons.
214 {RTLIB::OEQ_F64, RTLIB::impl___aeabi_dcmpeq},
215 {RTLIB::OLT_F64, RTLIB::impl___aeabi_dcmplt},
216 {RTLIB::OLE_F64, RTLIB::impl___aeabi_dcmple},
217 {RTLIB::OGE_F64, RTLIB::impl___aeabi_dcmpge},
218 {RTLIB::OGT_F64, RTLIB::impl___aeabi_dcmpgt},
219 {RTLIB::UO_F64, RTLIB::impl___aeabi_dcmpun},
220 // Single-precision arithmetic.
221 {RTLIB::ADD_F32, RTLIB::impl___aeabi_fadd},
222 {RTLIB::DIV_F32, RTLIB::impl___aeabi_fdiv},
223 {RTLIB::MUL_F32, RTLIB::impl___aeabi_fmul},
224 {RTLIB::SUB_F32, RTLIB::impl___aeabi_fsub},
225 // Single-precision comparisons.
226 {RTLIB::OEQ_F32, RTLIB::impl___aeabi_fcmpeq},
227 {RTLIB::OLT_F32, RTLIB::impl___aeabi_fcmplt},
228 {RTLIB::OLE_F32, RTLIB::impl___aeabi_fcmple},
229 {RTLIB::OGE_F32, RTLIB::impl___aeabi_fcmpge},
230 {RTLIB::OGT_F32, RTLIB::impl___aeabi_fcmpgt},
231 {RTLIB::UO_F32, RTLIB::impl___aeabi_fcmpun},
232 // Floating-point to integer conversions.
233 {RTLIB::FPTOSINT_F64_I32, RTLIB::impl___aeabi_d2iz},
234 {RTLIB::FPTOUINT_F64_I32, RTLIB::impl___aeabi_d2uiz},
235 {RTLIB::FPTOSINT_F64_I64, RTLIB::impl___aeabi_d2lz},
236 {RTLIB::FPTOUINT_F64_I64, RTLIB::impl___aeabi_d2ulz},
237 {RTLIB::FPTOSINT_F32_I32, RTLIB::impl___aeabi_f2iz},
238 {RTLIB::FPTOUINT_F32_I32, RTLIB::impl___aeabi_f2uiz},
239 {RTLIB::FPTOSINT_F32_I64, RTLIB::impl___aeabi_f2lz},
240 {RTLIB::FPTOUINT_F32_I64, RTLIB::impl___aeabi_f2ulz},
241 // Integer to floating-point conversions.
242 {RTLIB::SINTTOFP_I32_F64, RTLIB::impl___aeabi_i2d},
243 {RTLIB::UINTTOFP_I32_F64, RTLIB::impl___aeabi_ui2d},
244 {RTLIB::SINTTOFP_I64_F64, RTLIB::impl___aeabi_l2d},
245 {RTLIB::UINTTOFP_I64_F64, RTLIB::impl___aeabi_ul2d},
246 {RTLIB::SINTTOFP_I32_F32, RTLIB::impl___aeabi_i2f},
247 {RTLIB::UINTTOFP_I32_F32, RTLIB::impl___aeabi_ui2f},
248 {RTLIB::SINTTOFP_I64_F32, RTLIB::impl___aeabi_l2f},
249 {RTLIB::UINTTOFP_I64_F32, RTLIB::impl___aeabi_ul2f},
250 // Long long helpers.
251 {RTLIB::MUL_I64, RTLIB::impl___aeabi_lmul},
252 {RTLIB::SHL_I64, RTLIB::impl___aeabi_llsl},
253 {RTLIB::SRL_I64, RTLIB::impl___aeabi_llsr},
254 {RTLIB::SRA_I64, RTLIB::impl___aeabi_lasr},
255 // Integer division.
256 {RTLIB::SDIV_I32, RTLIB::impl___aeabi_idiv},
257 {RTLIB::UDIV_I32, RTLIB::impl___aeabi_uidiv},
258 };
259
260 const RTLIB::RuntimeLibcallsInfo &RTLCI = Info.getRuntimeLibcallsInfo();
261 for (const auto &LC : AEABISelected) {
262 if (RTLCI.isAvailable(LC.Impl))
263 Info.setLibcallImpl(LC.Op, LC.Impl);
264 }
265
266 // AEABI provides an ordered-equal compare (__aeabi_{f,d}cmpeq) but no
267 // not-equal compare. Clear the not-equal libcalls so UNE will lower as !OEQ
268 // using the AEABI compare, rather than emitting the generic not-equal helper
269 // which would otherwise be preferred.
270 if (RTLCI.isAvailable(RTLIB::impl___aeabi_fcmpeq)) {
271 Info.setLibcallImpl(RTLIB::UNE_F32, RTLIB::Unsupported);
272 Info.setLibcallImpl(RTLIB::FCMP3_PRED_UNE_F32, RTLIB::Unsupported);
273 }
274
275 if (RTLCI.isAvailable(RTLIB::impl___aeabi_dcmpeq)) {
276 Info.setLibcallImpl(RTLIB::UNE_F64, RTLIB::Unsupported);
277 Info.setLibcallImpl(RTLIB::FCMP3_PRED_UNE_F64, RTLIB::Unsupported);
278 }
279}
280
282 // We don't currently support Thumb, but Windows requires Thumb.
283 return hasV6Ops() && hasARMOps() && !isTargetWindows();
284}
285
286void ARMSubtarget::initSubtargetFeatures(StringRef CPU, StringRef FS) {
287 if (CPUString.empty()) {
288 CPUString = "generic";
289
290 if (isTargetDarwin()) {
292 ARM::ArchKind AK = ARM::parseArch(ArchName);
293 if (AK == ARM::ArchKind::ARMV7S)
294 // Default to the Swift CPU when targeting armv7s/thumbv7s.
295 CPUString = "swift";
296 else if (AK == ARM::ArchKind::ARMV7K)
297 // Default to the Cortex-a7 CPU when targeting armv7k/thumbv7k.
298 // ARMv7k does not use SjLj exception handling.
299 CPUString = "cortex-a7";
300 }
301 }
302
303 // Insert the architecture feature derived from the target triple into the
304 // feature string. This is important for setting features that are implied
305 // based on the architecture version.
306 std::string ArchFS = ARM_MC::ParseARMTriple(TargetTriple, CPUString);
307 if (!FS.empty()) {
308 if (!ArchFS.empty())
309 ArchFS = (Twine(ArchFS) + "," + FS).str();
310 else
311 ArchFS = std::string(FS);
312 }
313 ParseSubtargetFeatures(CPUString, /*TuneCPU*/ CPUString, ArchFS);
314
315 // FIXME: This used enable V6T2 support implicitly for Thumb2 mode.
316 // Assert this for now to make the change obvious.
317 assert(hasV6T2Ops() || !hasThumb2());
318
319 if (genExecuteOnly()) {
320 // Execute only support for >= v8-M Baseline requires movt support
321 if (hasV8MBaselineOps())
322 NoMovt = false;
323 if (!hasV6MOps())
324 report_fatal_error("Cannot generate execute-only code for this target");
325 }
326
327 // Keep a pointer to static instruction cost data for the specified CPU.
328 SchedModel = getSchedModelForCPU(CPUString);
329
330 // Initialize scheduling itinerary for the specified CPU.
331 InstrItins = getInstrItineraryForCPU(CPUString);
332
333 // FIXME: this is invalid for WindowsCE
334 if (isTargetWindows())
335 NoARM = true;
336
337 if (isAAPCS_ABI())
339 if (isAAPCS16_ABI())
340 stackAlignment = Align(16);
341
342 // FIXME: Completely disable sibcall for Thumb1 since ThumbRegisterInfo::
343 // emitEpilogue is not ready for them. Thumb tail calls also use t2B, as
344 // the Thumb1 16-bit unconditional branch doesn't have sufficient relocation
345 // support in the assembler and linker to be used. This would need to be
346 // fixed to fully support tail calls in Thumb1.
347 //
348 // For ARMv8-M, we /do/ implement tail calls. Doing this is tricky for v8-M
349 // baseline, since the LDM/POP instruction on Thumb doesn't take LR. This
350 // means if we need to reload LR, it takes extra instructions, which outweighs
351 // the value of the tail call; but here we don't know yet whether LR is going
352 // to be used. We take the optimistic approach of generating the tail call and
353 // perhaps taking a hit if we need to restore the LR.
354
355 // Thumb1 PIC calls to external symbols use BX, so they can be tail calls,
356 // but we need to make sure there are enough registers; the only valid
357 // registers are the 4 used for parameters. We don't currently do this
358 // case.
359
360 SupportsTailCall = !isThumb1Only() || hasV8MBaselineOps();
361
362 switch (IT) {
363 case DefaultIT:
364 RestrictIT = false;
365 break;
366 case RestrictedIT:
367 RestrictIT = true;
368 break;
369 }
370
371 // NEON f32 ops are non-IEEE 754 compliant. Darwin is ok with it by default.
372 const FeatureBitset &Bits = getFeatureBits();
373 if ((Bits[ARM::ProcA5] || Bits[ARM::ProcA8]) && // Where this matters
375 HasNEONForFP = true;
376
377 const ARM::ArchKind Arch = ARM::parseArch(TargetTriple.getArchName());
378 if (isRWPI() ||
379 (isTargetIOS() &&
380 (Arch == ARM::ArchKind::ARMV6K || Arch == ARM::ArchKind::ARMV6) &&
381 TargetTriple.isOSVersionLT(3, 0)))
382 ReserveR9 = true;
383
384 // If MVEVectorCostFactor is still 0 (has not been set to anything else), default it to 2
385 if (MVEVectorCostFactor == 0)
387
388 // FIXME: Teach TableGen to deal with these instead of doing it manually here.
389 switch (ARMProcFamily) {
390 case Others:
391 case CortexA5:
392 break;
393 case CortexA7:
395 break;
396 case CortexA8:
398 break;
399 case CortexA9:
402 break;
403 case CortexA12:
404 break;
405 case CortexA15:
409 break;
410 case CortexA17:
411 case CortexA32:
412 case CortexA35:
413 case CortexA53:
414 case CortexA55:
415 case CortexA57:
416 case CortexA72:
417 case CortexA73:
418 case CortexA75:
419 case CortexA76:
420 case CortexA77:
421 case CortexA78:
422 case CortexA78AE:
423 case CortexA78C:
424 case CortexA510:
425 case CortexA710:
426 case CortexR4:
427 case CortexR5:
428 case CortexR7:
429 case CortexM3:
430 case CortexM55:
431 case CortexM7:
432 case CortexM85:
433 case CortexR52:
434 case CortexR52plus:
435 case CortexX1:
436 case CortexX1C:
437 break;
438 case Exynos:
441 if (!isThumb())
443 break;
444 case Kryo:
445 break;
446 case Krait:
448 break;
449 case NeoverseV1:
450 break;
451 case Swift:
456 break;
457 }
458}
459
461 // FIXME: This should ideally come from a function attribute, to work
462 // correctly with LTO.
463 return TM.getRelocationModel() == Reloc::ROPI ||
464 TM.getRelocationModel() == Reloc::ROPI_RWPI;
465}
466
468 // FIXME: This should ideally come from a function attribute, to work
469 // correctly with LTO.
470 return TM.getRelocationModel() == Reloc::RWPI ||
471 TM.getRelocationModel() == Reloc::ROPI_RWPI;
472}
473
475 return TM.isGVIndirectSymbol(GV);
476}
477
479 return isTargetELF() && TM.isPositionIndependent() && !GV->isDSOLocal();
480}
481
483 // The MachineScheduler can increase register usage, so we use more high
484 // registers and end up with more T2 instructions that cannot be converted to
485 // T1 instructions. At least until we do better at converting to thumb1
486 // instructions, on cortex-m at Oz where we are size-paranoid, don't use the
487 // Machine scheduler, relying on the DAG register pressure scheduler instead.
488 if (isMClass() && hasMinSize())
489 return false;
490 // Enable the MachineScheduler before register allocation for subtargets
491 // with the use-misched feature.
492 return useMachineScheduler();
493}
494
496 // Enable SubRegLiveness for MVE to better optimize s subregs for mqpr regs
497 // and q subregs for qqqqpr regs.
498 return hasMVEIntegerOps();
499}
500
502 // Enable the MachinePipeliner before register allocation for subtargets
503 // with the use-mipipeliner feature.
504 return getSchedModel().hasInstrSchedModel() && useMachinePipeliner();
505}
506
507bool ARMSubtarget::useDFAforSMS() const { return false; }
508
509// This overrides the PostRAScheduler bit in the SchedModel for any CPU.
512 return false;
513 if (disablePostRAScheduler())
514 return false;
515 // Thumb1 cores will generally not benefit from post-ra scheduling
516 return !isThumb1Only();
517}
518
521 return false;
522 if (disablePostRAScheduler())
523 return false;
524 return !isThumb1Only();
525}
526
528 // For general targets, the prologue can grow when VFPs are allocated with
529 // stride 4 (more vpush instructions). But WatchOS uses a compact unwind
530 // format which it's more important to get right.
531 return isTargetWatchABI() ||
532 (useWideStrideVFP() && !OptMinSize);
533}
534
536 // NOTE Windows on ARM needs to use mov.w/mov.t pairs to materialise 32-bit
537 // immediates as it is inherently position independent, and may be out of
538 // range otherwise.
539 return !NoMovt && hasV8MBaselineOps() &&
540 (isTargetWindows() || !OptMinSize || genExecuteOnly());
541}
542
544 // Enable fast-isel for any target, for testing only.
545 if (ForceFastISel)
546 return true;
547
548 // Limit fast-isel to the targets that are or have been tested.
549 if (!hasV6Ops())
550 return false;
551
552 // Thumb2 support on iOS; ARM support on iOS and Linux.
553 return TM.Options.EnableFastISel && ((isTargetMachO() && !isThumb1Only()) ||
554 (isTargetLinux() && !isThumb()));
555}
556
558 // The GPR register class has multiple possible allocation orders, with
559 // tradeoffs preferred by different sub-architectures and optimisation goals.
560 // The allocation orders are:
561 // 0: (the default tablegen order, not used)
562 // 1: r14, r0-r13
563 // 2: r0-r7
564 // 3: r0-r7, r12, lr, r8-r11
565 // Note that the register allocator will change this order so that
566 // callee-saved registers are used later, as they require extra work in the
567 // prologue/epilogue (though we sometimes override that).
568
569 // For thumb1-only targets, only the low registers are allocatable.
570 if (isThumb1Only())
571 return 2;
572
573 // Allocate low registers first, so we can select more 16-bit instructions.
574 // We also (in getCSRAllocationOrderMask) override the default behaviour
575 // with regards to callee-saved registers, because pushing extra registers is
576 // much cheaper (in terms of code size) than using high registers. After
577 // that, we allocate r12 (doesn't need to be saved), lr (saving it means we
578 // can return with the pop, don't need an extra "bx lr") and then the rest of
579 // the high registers.
580 if (isThumb2() && MF.getFunction().hasMinSize())
581 return 3;
582
583 // Otherwise, allocate in the default order, using LR first because saving it
584 // allows a shorter epilogue sequence.
585 return 1;
586}
587
589 BitVector &Mask) const {
590 // To minimize code size in Thumb2, we prefer the usage of low regs (lower
591 // cost per use) so we can use narrow encoding. By default, caller-saved
592 // registers (e.g. lr, r12) are always allocated first, regardless of
593 // their cost per use. When optForMinSize, we prefer the low regs even if
594 // they are CSR because usually push/pop can be folded into existing ones.
595 if (!isThumb2() || !MF.getFunction().hasMinSize())
596 return;
597
598 Mask.resize(getRegisterInfo()->getNumRegs());
599 for (MCPhysReg Reg : ARM::GPRRegClass)
600 Mask.set(Reg);
601}
602
605 const Function &F = MF.getFunction();
606 const MachineFrameInfo &MFI = MF.getFrameInfo();
607
608 // Thumb1 always splits the pushes at R7, because the Thumb1 push instruction
609 // cannot use high registers except for lr.
610 if (isThumb1Only())
611 return SplitR7;
612
613 // If R7 is the frame pointer, we must split at R7 to ensure that the
614 // previous frame pointer (R7) and return address (LR) are adjacent on the
615 // stack, to form a valid frame record.
616 if (getFramePointerReg() == ARM::R7 &&
618 return SplitR7;
619
620 // Returns SplitR11WindowsSEH when the stack pointer needs to be
621 // restored from the frame pointer r11 + an offset and Windows CFI is enabled.
622 // This stack unwinding cannot be expressed with SEH unwind opcodes when done
623 // with a single push, making it necessary to split the push into r4-r10, and
624 // another containing r11+lr.
626 F.needsUnwindTableEntry() &&
627 (MFI.hasVarSizedObjects() || getRegisterInfo()->hasStackRealignment(MF)))
628 return SplitR11WindowsSEH;
629
630 // Returns SplitR11AAPCSSignRA when the frame pointer is R11, requiring R11
631 // and LR to be adjacent on the stack, and branch signing is enabled,
632 // requiring R12 to be on the stack.
634 getFramePointerReg() == ARM::R11 &&
636 return SplitR11AAPCSSignRA;
637 return NoSplit;
638}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static bool isThumb(const MCSubtargetInfo &STI)
This file describes how to lower LLVM calls to machine code calls.
This file declares the targeting of the Machinelegalizer class for ARM.
This file declares the targeting of the RegisterBankInfo class for ARM.
static cl::opt< bool > UseFusedMulOps("arm-use-mulops", cl::init(true), cl::Hidden)
static cl::opt< bool > ForceFastISel("arm-force-fast-isel", cl::init(false), cl::Hidden)
ForceFastISel - Use the fast-isel, even for subtargets where it is not currently supported (for testi...
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
ITMode
@ RestrictedIT
@ DefaultIT
This file implements the BitVector class.
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
#define F(x, y, z)
Definition MD5.cpp:54
ARMFunctionInfo - This class is derived from MachineFunctionInfo and contains private ARM-specific in...
This class provides the information for the target register banks.
bool useFastISel() const
True if fast-isel is used.
bool isTargetMachO() const
bool IsLittle
IsLittle - The target is Little Endian.
FloatABI::ABIType FloatABIType
The floating-point ABI in effect for this subtarget.
bool enablePostRAScheduler() const override
True for some subtargets at > -O0.
ARMLdStMultipleTiming LdStMultipleTiming
What kind of timing do load multiple/store multiple have (double issue, single issue etc).
bool hasARMOps() const
ARMSubtarget(const Triple &TT, const std::string &CPU, const std::string &FS, const ARMBaseTargetMachine &TM, bool IsLittle, FloatABI::ABIType FloatABI, ARM::ARMABI ABI, bool MinSize=false, DenormalMode DM=DenormalMode::getIEEE())
This constructor initializes the data members to match that of the specified triple.
const Triple & getTargetTriple() const
unsigned getGPRAllocationOrder(const MachineFunction &MF) const
const RegisterBankInfo * getRegBankInfo() const override
unsigned MaxInterleaveFactor
const ARMBaseTargetMachine & TM
bool isThumb1Only() const
ARMProcFamilyEnum ARMProcFamily
ARMProcFamily - ARM processor family: Cortex-A8, Cortex-A9, and others.
void getCSRAllocationOrderMask(const MachineFunction &MF, BitVector &Mask) const override
bool isThumb2() const
bool useDFAforSMS() const override
MCPhysReg getFramePointerReg() const
DenormalMode DM
DM - Denormal mode NEON and VFP RunFast mode are not IEEE 754 compliant, use this field to determine ...
bool isTargetWindows() const
bool enableSubRegLiveness() const override
Check whether this subtarget wants to use subregister liveness.
bool isGVIndirectSymbol(const GlobalValue *GV) const
True if the GV will be accessed via an indirect symbol.
unsigned MVEVectorCostFactor
The cost factor for MVE instructions, representing the multiple beats an.
const ARMTargetLowering * getTargetLowering() const override
MCSchedModel SchedModel
SchedModel - Processor specific instruction costs.
std::string CPUString
CPUString - String name of used CPU.
unsigned PreferBranchLogAlignment
What alignment is preferred for loop bodies and functions, in log2(bytes).
void initLibcallLoweringInfo(LibcallLoweringInfo &Info) const override
Triple TargetTriple
TargetTriple - What processor and OS we're targeting.
bool enableMachineScheduler() const override
Returns true if machine scheduler should be enabled.
bool isTargetDarwin() const
const ARMBaseRegisterInfo * getRegisterInfo() const override
InstrItineraryData InstrItins
Selected instruction itineraries (one entry per itinerary class.)
bool isAAPCS_ABI() const
bool useStride4VFPs() const
bool OptMinSize
OptMinSize - True if we're optimising for minimum code size, equal to the function attribute.
bool RestrictIT
RestrictIT - If true, the subtarget disallows generation of complex IT blocks.
bool hasVFP2Base() const
Align stackAlignment
stackAlignment - The minimum alignment known to hold of the stack frame on entry to the function and ...
unsigned PartialUpdateClearance
Clearance before partial register updates (in number of instructions)
bool enableMachinePipeliner() const override
Returns true if machine pipeliner should be enabled.
bool enablePostRAMachineScheduler() const override
True for some subtargets at > -O0.
InstructionSelector * getInstructionSelector() const override
bool isXRaySupported() const override
const CallLowering * getCallLowering() const override
enum PushPopSplitVariation getPushPopSplitVariation(const MachineFunction &MF) const
bool hasMinSize() const
ARMSubtarget & initializeSubtargetDependencies(StringRef CPU, StringRef FS)
initializeSubtargetDependencies - Initializes using a CPU and feature string so that we can use initi...
PushPopSplitVariation
How the push and pop instructions of callee saved general-purpose registers should be split.
@ SplitR11WindowsSEH
When the stack frame size is not known (because of variable-sized objects or realignment),...
@ SplitR7
R7 and LR must be adjacent, because R7 is the frame pointer, and must point to a frame record consist...
@ SplitR11AAPCSSignRA
When generating AAPCS-compilant frame chains, R11 is the frame pointer, and must be pushed adjacent t...
@ NoSplit
All GPRs can be pushed in a single instruction.
bool isTargetIOS() const
bool isGVInGOT(const GlobalValue *GV) const
Returns the constant pool modifier needed to access the GV.
bool isTargetWatchABI() const
const ARM::ARMABI ABI
The ABI in effect.
bool UseMulOps
UseMulOps - True if non-microcoded fused integer multiply-add and multiply-subtract instructions shou...
const TargetOptions & Options
Options passed via command line that could influence the target.
@ DoubleIssueCheckUnalignedAccess
Can load/store 2 registers/cycle, but needs an extra cycle if the access is not 64-bit aligned.
@ DoubleIssue
Can load/store 2 registers/cycle.
@ SingleIssuePlusExtras
Can load/store 1 register/cycle, but needs an extra cycle for address computation and potentially als...
void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS)
ParseSubtargetFeatures - Parses features string setting specified subtarget options.
bool useMachinePipeliner() const
bool useMachineScheduler() const
const LegalizerInfo * getLegalizerInfo() const override
bool isTargetLinux() const
bool isMClass() const
bool SupportsTailCall
SupportsTailCall - True if the OS supports tail call.
bool isAAPCS16_ABI() const
int PreISelOperandLatencyAdjustment
The adjustment that we need to apply to get the operand latency from the operand cycle returned by th...
bool isTargetELF() const
bool hasMinSize() const
Optimize this function for minimum size (-Oz).
Definition Function.h:695
bool isDSOLocal() const
Tracks which library functions to use for a particular subtarget or function.
bool usesWindowsCFI() const
Definition MCAsmInfo.h:675
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool hasVarSizedObjects() const
This method may be called any time after instruction selection is complete to determine if the stack ...
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Holds all the information related to register banks.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
TargetOptions Options
LLVM_ABI bool FramePointerIsReserved(const MachineFunction &MF) const
FramePointerIsReserved - This returns true if the frame pointer must always either point to a new fra...
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
LLVM_ABI StringRef getArchName() const
Get the architecture (first) component of the triple.
Definition Triple.cpp:1416
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
std::string ParseARMTriple(const Triple &TT, StringRef CPU)
LLVM_ABI ArchKind parseArch(StringRef Arch)
@ Swift
Calling convention for Swift.
Definition CallingConv.h:69
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
InstructionSelector * createARMInstructionSelector(const ARMBaseTargetMachine &TM, const ARMSubtarget &STI, const ARMRegisterBankInfo &RBI)
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
DWARFExpression::Operation Op
Represent subnormal handling kind for floating point instruction inputs and outputs.
static constexpr DenormalMode getPreserveSign()
A simple container for information about the supported runtime calls.
bool isAvailable(RTLIB::LibcallImpl Impl) const