LLVM 24.0.0git
ARMBaseInstrInfo.cpp
Go to the documentation of this file.
1//===-- ARMBaseInstrInfo.cpp - ARM Instruction Information ----------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains the Base ARM implementation of the TargetInstrInfo class.
10//
11//===----------------------------------------------------------------------===//
12
13#include "ARMBaseInstrInfo.h"
14#include "ARMBaseRegisterInfo.h"
16#include "ARMFeatures.h"
17#include "ARMHazardRecognizer.h"
19#include "ARMSubtarget.h"
22#include "MVETailPredUtils.h"
23#include "llvm/ADT/DenseMap.h"
24#include "llvm/ADT/STLExtras.h"
25#include "llvm/ADT/SmallSet.h"
47#include "llvm/IR/Attributes.h"
48#include "llvm/IR/DebugLoc.h"
49#include "llvm/IR/Function.h"
50#include "llvm/IR/GlobalValue.h"
51#include "llvm/IR/Module.h"
52#include "llvm/MC/MCAsmInfo.h"
53#include "llvm/MC/MCInstrDesc.h"
58#include "llvm/Support/Debug.h"
62#include <algorithm>
63#include <cassert>
64#include <cstdint>
65#include <iterator>
66#include <new>
67#include <utility>
68#include <vector>
69
70using namespace llvm;
71
72#define DEBUG_TYPE "arm-instrinfo"
73
74#define GET_INSTRINFO_CTOR_DTOR
75#include "ARMGenInstrInfo.inc"
76
77/// ARM_MLxEntry - Record information about MLA / MLS instructions.
79 uint16_t MLxOpc; // MLA / MLS opcode
80 uint16_t MulOpc; // Expanded multiplication opcode
81 uint16_t AddSubOpc; // Expanded add / sub opcode
82 bool NegAcc; // True if the acc is negated before the add / sub.
83 bool HasLane; // True if instruction has an extra "lane" operand.
84};
85
86static const ARM_MLxEntry ARM_MLxTable[] = {
87 // MLxOpc, MulOpc, AddSubOpc, NegAcc, HasLane
88 // fp scalar ops
89 { ARM::VMLAS, ARM::VMULS, ARM::VADDS, false, false },
90 { ARM::VMLSS, ARM::VMULS, ARM::VSUBS, false, false },
91 { ARM::VMLAD, ARM::VMULD, ARM::VADDD, false, false },
92 { ARM::VMLSD, ARM::VMULD, ARM::VSUBD, false, false },
93 { ARM::VNMLAS, ARM::VNMULS, ARM::VSUBS, true, false },
94 { ARM::VNMLSS, ARM::VMULS, ARM::VSUBS, true, false },
95 { ARM::VNMLAD, ARM::VNMULD, ARM::VSUBD, true, false },
96 { ARM::VNMLSD, ARM::VMULD, ARM::VSUBD, true, false },
97
98 // fp SIMD ops
99 { ARM::VMLAfd, ARM::VMULfd, ARM::VADDfd, false, false },
100 { ARM::VMLSfd, ARM::VMULfd, ARM::VSUBfd, false, false },
101 { ARM::VMLAfq, ARM::VMULfq, ARM::VADDfq, false, false },
102 { ARM::VMLSfq, ARM::VMULfq, ARM::VSUBfq, false, false },
103 { ARM::VMLAslfd, ARM::VMULslfd, ARM::VADDfd, false, true },
104 { ARM::VMLSslfd, ARM::VMULslfd, ARM::VSUBfd, false, true },
105 { ARM::VMLAslfq, ARM::VMULslfq, ARM::VADDfq, false, true },
106 { ARM::VMLSslfq, ARM::VMULslfq, ARM::VSUBfq, false, true },
107};
108
111 : ARMGenInstrInfo(STI, TRI, ARM::ADJCALLSTACKDOWN, ARM::ADJCALLSTACKUP),
112 Subtarget(STI) {
113 for (unsigned i = 0, e = std::size(ARM_MLxTable); i != e; ++i) {
114 if (!MLxEntryMap.insert(std::make_pair(ARM_MLxTable[i].MLxOpc, i)).second)
115 llvm_unreachable("Duplicated entries?");
116 MLxHazardOpcodes.insert(ARM_MLxTable[i].AddSubOpc);
117 MLxHazardOpcodes.insert(ARM_MLxTable[i].MulOpc);
118 }
119}
120
121// Use a ScoreboardHazardRecognizer for prepass ARM scheduling. TargetInstrImpl
122// currently defaults to no prepass hazard recognizer.
125 const ScheduleDAG *DAG) const {
126 if (usePreRAHazardRecognizer()) {
127 const InstrItineraryData *II =
128 static_cast<const ARMSubtarget *>(STI)->getInstrItineraryData();
129 return new ScoreboardHazardRecognizer(II, DAG, "pre-RA-sched");
130 }
132}
133
134// Called during:
135// - pre-RA scheduling
136// - post-RA scheduling when FeatureUseMISched is set
138 const InstrItineraryData *II, const ScheduleDAGMI *DAG) const {
140
141 // We would like to restrict this hazard recognizer to only
142 // post-RA scheduling; we can tell that we're post-RA because we don't
143 // track VRegLiveness.
144 // Cortex-M7: TRM indicates that there is a single ITCM bank and two DTCM
145 // banks banked on bit 2. Assume that TCMs are in use.
146 if (Subtarget.isCortexM7() && !DAG->hasVRegLiveness())
148 std::make_unique<ARMBankConflictHazardRecognizer>(DAG, 0x4, true));
149
150 // Not inserting ARMHazardRecognizerFPMLx because that would change
151 // legacy behavior
152
154 MHR->AddHazardRecognizer(std::unique_ptr<ScheduleHazardRecognizer>(BHR));
155 return MHR;
156}
157
158// Called during post-RA scheduling when FeatureUseMISched is not set
161 const ScheduleDAG *DAG) const {
163
164 if (Subtarget.isThumb2() || Subtarget.hasVFP2Base())
165 MHR->AddHazardRecognizer(std::make_unique<ARMHazardRecognizerFPMLx>());
166
168 if (BHR)
169 MHR->AddHazardRecognizer(std::unique_ptr<ScheduleHazardRecognizer>(BHR));
170 return MHR;
171}
172
173// Branch analysis.
174// Cond vector output format:
175// 0 elements indicates an unconditional branch
176// 2 elements indicates a conditional branch; the elements are
177// the condition to check and the CPSR.
178// 3 elements indicates a hardware loop end; the elements
179// are the opcode, the operand value to test, and a dummy
180// operand used to pad out to 3 operands.
183 MachineBasicBlock *&FBB,
185 bool AllowModify) const {
186 TBB = nullptr;
187 FBB = nullptr;
188
190 if (I == MBB.instr_begin())
191 return false; // Empty blocks are easy.
192 --I;
193
194 // Walk backwards from the end of the basic block until the branch is
195 // analyzed or we give up.
196 while (isPredicated(*I) || I->isTerminator() || I->isDebugValue()) {
197 // Flag to be raised on unanalyzeable instructions. This is useful in cases
198 // where we want to clean up on the end of the basic block before we bail
199 // out.
200 bool CantAnalyze = false;
201
202 // Skip over DEBUG values, predicated nonterminators and speculation
203 // barrier terminators.
204 while (I->isDebugInstr() || !I->isTerminator() ||
205 isSpeculationBarrierEndBBOpcode(I->getOpcode()) ||
206 I->getOpcode() == ARM::t2DoLoopStartTP){
207 if (I == MBB.instr_begin())
208 return false;
209 --I;
210 }
211
212 if (isIndirectBranchOpcode(I->getOpcode()) ||
213 isJumpTableBranchOpcode(I->getOpcode())) {
214 // Indirect branches and jump tables can't be analyzed, but we still want
215 // to clean up any instructions at the tail of the basic block.
216 CantAnalyze = true;
217 } else if (isUncondBranchOpcode(I->getOpcode())) {
218 TBB = I->getOperand(0).getMBB();
219 } else if (isCondBranchOpcode(I->getOpcode())) {
220 // Bail out if we encounter multiple conditional branches.
221 if (!Cond.empty())
222 return true;
223
224 assert(!FBB && "FBB should have been null.");
225 FBB = TBB;
226 TBB = I->getOperand(0).getMBB();
227 Cond.push_back(I->getOperand(1));
228 Cond.push_back(I->getOperand(2));
229 } else if (I->isReturn()) {
230 // Returns can't be analyzed, but we should run cleanup.
231 CantAnalyze = true;
232 } else if (I->getOpcode() == ARM::t2LoopEnd &&
233 MBB.getParent()
234 ->getSubtarget<ARMSubtarget>()
236 if (!Cond.empty())
237 return true;
238 FBB = TBB;
239 TBB = I->getOperand(1).getMBB();
240 Cond.push_back(MachineOperand::CreateImm(I->getOpcode()));
241 Cond.push_back(I->getOperand(0));
242 Cond.push_back(MachineOperand::CreateImm(0));
243 } else {
244 // We encountered other unrecognized terminator. Bail out immediately.
245 return true;
246 }
247
248 // Cleanup code - to be run for unpredicated unconditional branches and
249 // returns.
250 if (!isPredicated(*I) &&
251 (isUncondBranchOpcode(I->getOpcode()) ||
252 isIndirectBranchOpcode(I->getOpcode()) ||
253 isJumpTableBranchOpcode(I->getOpcode()) ||
254 I->isReturn())) {
255 // Forget any previous condition branch information - it no longer applies.
256 Cond.clear();
257 FBB = nullptr;
258
259 // If we can modify the function, delete everything below this
260 // unconditional branch.
261 if (AllowModify) {
262 MachineBasicBlock::iterator DI = std::next(I);
263 while (DI != MBB.instr_end()) {
264 MachineInstr &InstToDelete = *DI;
265 ++DI;
266 // Speculation barriers must not be deleted.
267 if (isSpeculationBarrierEndBBOpcode(InstToDelete.getOpcode()))
268 continue;
269 InstToDelete.eraseFromParent();
270 }
271 }
272 }
273
274 if (CantAnalyze) {
275 // We may not be able to analyze the block, but we could still have
276 // an unconditional branch as the last instruction in the block, which
277 // just branches to layout successor. If this is the case, then just
278 // remove it if we're allowed to make modifications.
279 if (AllowModify && !isPredicated(MBB.back()) &&
280 isUncondBranchOpcode(MBB.back().getOpcode()) &&
281 TBB && MBB.isLayoutSuccessor(TBB))
283 return true;
284 }
285
286 if (I == MBB.instr_begin())
287 return false;
288
289 --I;
290 }
291
292 // We made it past the terminators without bailing out - we must have
293 // analyzed this branch successfully.
294 return false;
295}
296
298 int *BytesRemoved) const {
299 assert(!BytesRemoved && "code size not handled");
300
301 MachineBasicBlock::iterator I = MBB.getLastNonDebugInstr();
302 if (I == MBB.end())
303 return 0;
304
305 if (!isUncondBranchOpcode(I->getOpcode()) &&
306 !isCondBranchOpcode(I->getOpcode()) && I->getOpcode() != ARM::t2LoopEnd)
307 return 0;
308
309 // Remove the branch.
310 I->eraseFromParent();
311
312 I = MBB.end();
313
314 if (I == MBB.begin()) return 1;
315 --I;
316 if (!isCondBranchOpcode(I->getOpcode()) && I->getOpcode() != ARM::t2LoopEnd)
317 return 1;
318
319 // Remove the branch.
320 I->eraseFromParent();
321 return 2;
322}
323
328 const DebugLoc &DL,
329 int *BytesAdded) const {
330 assert(!BytesAdded && "code size not handled");
331 ARMFunctionInfo *AFI = MBB.getParent()->getInfo<ARMFunctionInfo>();
332 int BOpc = !AFI->isThumbFunction()
333 ? ARM::B : (AFI->isThumb2Function() ? ARM::t2B : ARM::tB);
334 int BccOpc = !AFI->isThumbFunction()
335 ? ARM::Bcc : (AFI->isThumb2Function() ? ARM::t2Bcc : ARM::tBcc);
336 bool isThumb = AFI->isThumbFunction() || AFI->isThumb2Function();
337
338 // Shouldn't be a fall through.
339 assert(TBB && "insertBranch must not be told to insert a fallthrough");
340 assert((Cond.size() == 2 || Cond.size() == 0 || Cond.size() == 3) &&
341 "ARM branch conditions have two or three components!");
342
343 // For conditional branches, we use addOperand to preserve CPSR flags.
344
345 if (!FBB) {
346 if (Cond.empty()) { // Unconditional branch?
347 if (isThumb)
349 else
350 BuildMI(&MBB, DL, get(BOpc)).addMBB(TBB);
351 } else if (Cond.size() == 2) {
352 BuildMI(&MBB, DL, get(BccOpc))
353 .addMBB(TBB)
354 .addImm(Cond[0].getImm())
355 .add(Cond[1]);
356 } else
357 BuildMI(&MBB, DL, get(Cond[0].getImm())).add(Cond[1]).addMBB(TBB);
358 return 1;
359 }
360
361 // Two-way conditional branch.
362 if (Cond.size() == 2)
363 BuildMI(&MBB, DL, get(BccOpc))
364 .addMBB(TBB)
365 .addImm(Cond[0].getImm())
366 .add(Cond[1]);
367 else if (Cond.size() == 3)
368 BuildMI(&MBB, DL, get(Cond[0].getImm())).add(Cond[1]).addMBB(TBB);
369 if (isThumb)
370 BuildMI(&MBB, DL, get(BOpc)).addMBB(FBB).add(predOps(ARMCC::AL));
371 else
372 BuildMI(&MBB, DL, get(BOpc)).addMBB(FBB);
373 return 2;
374}
375
378 if (Cond.size() == 2) {
379 ARMCC::CondCodes CC = (ARMCC::CondCodes)(int)Cond[0].getImm();
380 Cond[0].setImm(ARMCC::getOppositeCondition(CC));
381 return false;
382 }
383 return true;
384}
385
387 if (MI.isBundle()) {
389 MachineBasicBlock::const_instr_iterator E = MI.getParent()->instr_end();
390 while (++I != E && I->isInsideBundle()) {
391 int PIdx = I->findFirstPredOperandIdx();
392 if (PIdx != -1 && I->getOperand(PIdx).getImm() != ARMCC::AL)
393 return true;
394 }
395 return false;
396 }
397
398 int PIdx = MI.findFirstPredOperandIdx();
399 return PIdx != -1 && MI.getOperand(PIdx).getImm() != ARMCC::AL;
400}
401
403 const MachineInstr &MI, const MachineOperand &Op, unsigned OpIdx,
404 const TargetRegisterInfo *TRI) const {
405
406 // First, let's see if there is a generic comment for this operand
407 std::string GenericComment =
409 if (!GenericComment.empty())
410 return GenericComment;
411
412 // If not, check if we have an immediate operand.
413 if (!Op.isImm())
414 return std::string();
415
416 // And print its corresponding condition code if the immediate is a
417 // predicate.
418 int FirstPredOp = MI.findFirstPredOperandIdx();
419 if (FirstPredOp != (int) OpIdx)
420 return std::string();
421
422 std::string CC = "CC::";
423 CC += ARMCondCodeToString((ARMCC::CondCodes)Op.getImm());
424 return CC;
425}
426
429 unsigned Opc = MI.getOpcode();
432 MachineInstrBuilder(*MI.getParent()->getParent(), MI)
433 .addImm(Pred[0].getImm())
434 .addReg(Pred[1].getReg());
435 return true;
436 }
437
438 int PIdx = MI.findFirstPredOperandIdx();
439 if (PIdx != -1) {
440 MachineOperand &PMO = MI.getOperand(PIdx);
441 PMO.setImm(Pred[0].getImm());
442 MI.getOperand(PIdx+1).setReg(Pred[1].getReg());
443
444 // Thumb 1 arithmetic instructions do not set CPSR when executed inside an
445 // IT block. This affects how they are printed.
446 const MCInstrDesc &MCID = MI.getDesc();
447 if (MCID.TSFlags & ARMII::ThumbArithFlagSetting) {
448 assert(MCID.operands()[1].isOptionalDef() &&
449 "CPSR def isn't expected operand");
450 assert((MI.getOperand(1).isDead() ||
451 MI.getOperand(1).getReg() != ARM::CPSR) &&
452 "if conversion tried to stop defining used CPSR");
453 MI.getOperand(1).setReg(Register());
454 }
455
456 return true;
457 }
458 return false;
459}
460
462 ArrayRef<MachineOperand> Pred2) const {
463 if (Pred1.size() > 2 || Pred2.size() > 2)
464 return false;
465
466 ARMCC::CondCodes CC1 = (ARMCC::CondCodes)Pred1[0].getImm();
467 ARMCC::CondCodes CC2 = (ARMCC::CondCodes)Pred2[0].getImm();
468 if (CC1 == CC2)
469 return true;
470
471 switch (CC1) {
472 default:
473 return false;
474 case ARMCC::AL:
475 return true;
476 case ARMCC::HS:
477 return CC2 == ARMCC::HI;
478 case ARMCC::LS:
479 return CC2 == ARMCC::LO || CC2 == ARMCC::EQ;
480 case ARMCC::GE:
481 return CC2 == ARMCC::GT;
482 case ARMCC::LE:
483 return CC2 == ARMCC::LT;
484 }
485}
486
488 std::vector<MachineOperand> &Pred,
489 bool SkipDead) const {
490 bool Found = false;
491 for (const MachineOperand &MO : MI.operands()) {
492 bool ClobbersCPSR = MO.isRegMask() && MO.clobbersPhysReg(ARM::CPSR);
493 bool IsCPSR = MO.isReg() && MO.isDef() && MO.getReg() == ARM::CPSR;
494 if (ClobbersCPSR || IsCPSR) {
495
496 // Filter out T1 instructions that have a dead CPSR,
497 // allowing IT blocks to be generated containing T1 instructions
498 const MCInstrDesc &MCID = MI.getDesc();
499 if (MCID.TSFlags & ARMII::ThumbArithFlagSetting && MO.isDead() &&
500 SkipDead)
501 continue;
502
503 Pred.push_back(MO);
504 Found = true;
505 }
506 }
507
508 return Found;
509}
510
512 for (const auto &MO : MI.operands())
513 if (MO.isReg() && MO.getReg() == ARM::CPSR && MO.isDef() && !MO.isDead())
514 return true;
515 return false;
516}
517
519 switch (MI->getOpcode()) {
520 default: return true;
521 case ARM::tADC: // ADC (register) T1
522 case ARM::tADDi3: // ADD (immediate) T1
523 case ARM::tADDi8: // ADD (immediate) T2
524 case ARM::tADDrr: // ADD (register) T1
525 case ARM::tAND: // AND (register) T1
526 case ARM::tASRri: // ASR (immediate) T1
527 case ARM::tASRrr: // ASR (register) T1
528 case ARM::tBIC: // BIC (register) T1
529 case ARM::tEOR: // EOR (register) T1
530 case ARM::tLSLri: // LSL (immediate) T1
531 case ARM::tLSLrr: // LSL (register) T1
532 case ARM::tLSRri: // LSR (immediate) T1
533 case ARM::tLSRrr: // LSR (register) T1
534 case ARM::tMUL: // MUL T1
535 case ARM::tMVN: // MVN (register) T1
536 case ARM::tORR: // ORR (register) T1
537 case ARM::tROR: // ROR (register) T1
538 case ARM::tRSB: // RSB (immediate) T1
539 case ARM::tSBC: // SBC (register) T1
540 case ARM::tSUBi3: // SUB (immediate) T1
541 case ARM::tSUBi8: // SUB (immediate) T2
542 case ARM::tSUBrr: // SUB (register) T1
544 }
545}
546
547/// isPredicable - Return true if the specified instruction can be predicated.
548/// By default, this returns true for every instruction with a
549/// PredicateOperand.
551 if (!MI.isPredicable())
552 return false;
553
554 if (MI.isBundle())
555 return false;
556
558 return false;
559
560 const MachineFunction *MF = MI.getParent()->getParent();
561 const ARMFunctionInfo *AFI =
563
564 // Neon instructions in Thumb2 IT blocks are deprecated, see ARMARM.
565 // In their ARM encoding, they can't be encoded in a conditional form.
566 if ((MI.getDesc().TSFlags & ARMII::DomainMask) == ARMII::DomainNEON)
567 return false;
568
569 // Make indirect control flow changes unpredictable when SLS mitigation is
570 // enabled.
571 const ARMSubtarget &ST = MF->getSubtarget<ARMSubtarget>();
572 if (ST.hardenSlsRetBr() && isIndirectControlFlowNotComingBack(MI))
573 return false;
574 if (ST.hardenSlsBlr() && isIndirectCall(MI))
575 return false;
576
577 if (AFI->isThumb2Function()) {
578 if (getSubtarget().restrictIT())
579 return isV8EligibleForIT(&MI);
580 }
581
582 return true;
583}
584
585namespace llvm {
586
587template <> bool IsCPSRDead<MachineInstr>(const MachineInstr *MI) {
588 for (const MachineOperand &MO : MI->operands()) {
589 if (!MO.isReg() || MO.isUndef() || MO.isUse())
590 continue;
591 if (MO.getReg() != ARM::CPSR)
592 continue;
593 if (!MO.isDead())
594 return false;
595 }
596 // all definitions of CPSR are dead
597 return true;
598}
599
600} // end namespace llvm
601
602/// GetInstSize - Return the size of the specified MachineInstr.
603///
605 const MachineBasicBlock &MBB = *MI.getParent();
606 const MachineFunction *MF = MBB.getParent();
607 const MCAsmInfo &MAI = MF->getTarget().getMCAsmInfo();
608
609 const MCInstrDesc &MCID = MI.getDesc();
610
611 switch (MI.getOpcode()) {
612 default:
613 // Return the size specified in .td file. If there's none, return 0, as we
614 // can't define a default size (Thumb1 instructions are 2 bytes, Thumb2
615 // instructions are 2-4 bytes, and ARM instructions are 4 bytes), in
616 // contrast to AArch64 instructions which have a default size of 4 bytes for
617 // example.
618 return MCID.getSize();
619 case TargetOpcode::BUNDLE:
620 return getInstBundleSize(MI);
621 case TargetOpcode::COPY:
623 return 4;
624 else
625 return 2;
626 case TargetOpcode::PATCHABLE_FUNCTION_ENTER:
627 case TargetOpcode::PATCHABLE_FUNCTION_EXIT:
628 case TargetOpcode::PATCHABLE_TAIL_CALL:
629 // Size of xray sled: Branch + 6 nops.
630 return 28;
631 case ARM::CONSTPOOL_ENTRY:
632 case ARM::JUMPTABLE_INSTS:
633 case ARM::JUMPTABLE_ADDRS:
634 case ARM::JUMPTABLE_TBB:
635 case ARM::JUMPTABLE_TBH:
636 // If this machine instr is a constant pool entry, its size is recorded as
637 // operand #2.
638 return MI.getOperand(2).getImm();
639 case ARM::SPACE:
640 return MI.getOperand(1).getImm();
641 case ARM::INLINEASM:
642 case ARM::INLINEASM_BR: {
643 // If this machine instr is an inline asm, measure it.
644 unsigned Size = getInlineAsmLength(MI.getOperand(0).getSymbolName(), MAI);
646 Size = alignTo(Size, 4);
647 return Size;
648 }
649 case ARM::Int_eh_sjlj_longjmp:
650 return Subtarget.isTargetDarwin() || Subtarget.isTargetWindows() ? 16 : 20;
651 case ARM::tInt_eh_sjlj_longjmp:
652 return Subtarget.isTargetDarwin() || Subtarget.isTargetWindows() ? 10 : 12;
653 }
654}
655
658 MCRegister DestReg, bool KillSrc,
659 const ARMSubtarget &Subtarget) const {
660 unsigned Opc = Subtarget.isThumb()
661 ? (Subtarget.isMClass() ? ARM::t2MRS_M : ARM::t2MRS_AR)
662 : ARM::MRS;
663
665 BuildMI(MBB, I, I->getDebugLoc(), get(Opc), DestReg);
666
667 // There is only 1 A/R class MRS instruction, and it always refers to
668 // APSR. However, there are lots of other possibilities on M-class cores.
669 if (Subtarget.isMClass())
670 MIB.addImm(0x800);
671
672 MIB.add(predOps(ARMCC::AL))
673 .addReg(ARM::CPSR, RegState::Implicit | getKillRegState(KillSrc));
674}
675
678 MCRegister SrcReg, bool KillSrc,
679 const ARMSubtarget &Subtarget) const {
680 unsigned Opc = Subtarget.isThumb()
681 ? (Subtarget.isMClass() ? ARM::t2MSR_M : ARM::t2MSR_AR)
682 : ARM::MSR;
683
684 MachineInstrBuilder MIB = BuildMI(MBB, I, I->getDebugLoc(), get(Opc));
685
686 if (Subtarget.isMClass())
687 MIB.addImm(0x800);
688 else
689 MIB.addImm(8);
690
691 MIB.addReg(SrcReg, getKillRegState(KillSrc))
694}
695
697 MIB.addImm(ARMVCC::None);
698 MIB.addReg(0);
699 MIB.addReg(0); // tp_reg
700}
701
707
709 MIB.addImm(Cond);
710 MIB.addReg(ARM::VPR, RegState::Implicit);
711 MIB.addReg(0); // tp_reg
712}
713
715 unsigned Cond, unsigned Inactive) {
717 MIB.addReg(Inactive);
718}
719
722 const DebugLoc &DL, Register DestReg,
723 Register SrcReg, bool KillSrc,
724 bool RenamableDest,
725 bool RenamableSrc) const {
726 bool GPRDest = ARM::GPRRegClass.contains(DestReg);
727 bool GPRSrc = ARM::GPRRegClass.contains(SrcReg);
728
729 if (GPRDest && GPRSrc) {
730 BuildMI(MBB, I, DL, get(ARM::MOVr), DestReg)
731 .addReg(SrcReg, getKillRegState(KillSrc))
733 .add(condCodeOp());
734 return;
735 }
736
737 bool SPRDest = ARM::SPRRegClass.contains(DestReg);
738 bool SPRSrc = ARM::SPRRegClass.contains(SrcReg);
739
740 unsigned Opc = 0;
741 if (SPRDest && SPRSrc)
742 Opc = ARM::VMOVS;
743 else if (GPRDest && SPRSrc)
744 Opc = ARM::VMOVRS;
745 else if (SPRDest && GPRSrc)
746 Opc = ARM::VMOVSR;
747 else if (ARM::DPRRegClass.contains(DestReg, SrcReg) && Subtarget.hasFP64())
748 Opc = ARM::VMOVD;
749 else if (ARM::QPRRegClass.contains(DestReg, SrcReg))
750 Opc = Subtarget.hasNEON() ? ARM::VORRq : ARM::MQPRCopy;
751
752 if (Opc) {
753 MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(Opc), DestReg);
754 MIB.addReg(SrcReg, getKillRegState(KillSrc));
755 if (Opc == ARM::VORRq || Opc == ARM::MVE_VORR)
756 MIB.addReg(SrcReg, getKillRegState(KillSrc));
757 if (Opc == ARM::MVE_VORR)
758 addUnpredicatedMveVpredROp(MIB, DestReg);
759 else if (Opc != ARM::MQPRCopy)
760 MIB.add(predOps(ARMCC::AL));
761 return;
762 }
763
764 // Handle register classes that require multiple instructions.
765 unsigned BeginIdx = 0;
766 unsigned SubRegs = 0;
767 int Spacing = 1;
768
769 // Use VORRq when possible.
770 if (ARM::QQPRRegClass.contains(DestReg, SrcReg)) {
771 Opc = Subtarget.hasNEON() ? ARM::VORRq : ARM::MVE_VORR;
772 BeginIdx = ARM::qsub_0;
773 SubRegs = 2;
774 } else if (ARM::QQQQPRRegClass.contains(DestReg, SrcReg)) {
775 Opc = Subtarget.hasNEON() ? ARM::VORRq : ARM::MVE_VORR;
776 BeginIdx = ARM::qsub_0;
777 SubRegs = 4;
778 // Fall back to VMOVD.
779 } else if (ARM::DPairRegClass.contains(DestReg, SrcReg)) {
780 Opc = ARM::VMOVD;
781 BeginIdx = ARM::dsub_0;
782 SubRegs = 2;
783 } else if (ARM::DTripleRegClass.contains(DestReg, SrcReg)) {
784 Opc = ARM::VMOVD;
785 BeginIdx = ARM::dsub_0;
786 SubRegs = 3;
787 } else if (ARM::DQuadRegClass.contains(DestReg, SrcReg)) {
788 Opc = ARM::VMOVD;
789 BeginIdx = ARM::dsub_0;
790 SubRegs = 4;
791 } else if (ARM::GPRPairRegClass.contains(DestReg, SrcReg)) {
792 Opc = Subtarget.isThumb2() ? ARM::tMOVr : ARM::MOVr;
793 BeginIdx = ARM::gsub_0;
794 SubRegs = 2;
795 } else if (ARM::DPairSpcRegClass.contains(DestReg, SrcReg)) {
796 Opc = ARM::VMOVD;
797 BeginIdx = ARM::dsub_0;
798 SubRegs = 2;
799 Spacing = 2;
800 } else if (ARM::DTripleSpcRegClass.contains(DestReg, SrcReg)) {
801 Opc = ARM::VMOVD;
802 BeginIdx = ARM::dsub_0;
803 SubRegs = 3;
804 Spacing = 2;
805 } else if (ARM::DQuadSpcRegClass.contains(DestReg, SrcReg)) {
806 Opc = ARM::VMOVD;
807 BeginIdx = ARM::dsub_0;
808 SubRegs = 4;
809 Spacing = 2;
810 } else if (ARM::DPRRegClass.contains(DestReg, SrcReg) &&
811 !Subtarget.hasFP64()) {
812 Opc = ARM::VMOVS;
813 BeginIdx = ARM::ssub_0;
814 SubRegs = 2;
815 } else if (SrcReg == ARM::CPSR) {
816 copyFromCPSR(MBB, I, DestReg, KillSrc, Subtarget);
817 return;
818 } else if (DestReg == ARM::CPSR) {
819 copyToCPSR(MBB, I, SrcReg, KillSrc, Subtarget);
820 return;
821 } else if (DestReg == ARM::VPR) {
822 assert(ARM::GPRRegClass.contains(SrcReg));
823 BuildMI(MBB, I, I->getDebugLoc(), get(ARM::VMSR_P0), DestReg)
824 .addReg(SrcReg, getKillRegState(KillSrc))
826 return;
827 } else if (SrcReg == ARM::VPR) {
828 assert(ARM::GPRRegClass.contains(DestReg));
829 BuildMI(MBB, I, I->getDebugLoc(), get(ARM::VMRS_P0), DestReg)
830 .addReg(SrcReg, getKillRegState(KillSrc))
832 return;
833 } else if (DestReg == ARM::FPSCR_NZCV) {
834 assert(ARM::GPRRegClass.contains(SrcReg));
835 BuildMI(MBB, I, I->getDebugLoc(), get(ARM::VMSR_FPSCR_NZCVQC), DestReg)
836 .addReg(SrcReg, getKillRegState(KillSrc))
838 return;
839 } else if (SrcReg == ARM::FPSCR_NZCV) {
840 assert(ARM::GPRRegClass.contains(DestReg));
841 BuildMI(MBB, I, I->getDebugLoc(), get(ARM::VMRS_FPSCR_NZCVQC), DestReg)
842 .addReg(SrcReg, getKillRegState(KillSrc))
844 return;
845 }
846
847 assert(Opc && "Impossible reg-to-reg copy");
848
851
852 // Copy register tuples backward when the first Dest reg overlaps with SrcReg.
853 if (TRI->regsOverlap(SrcReg, TRI->getSubReg(DestReg, BeginIdx))) {
854 BeginIdx = BeginIdx + ((SubRegs - 1) * Spacing);
855 Spacing = -Spacing;
856 }
857#ifndef NDEBUG
858 SmallSet<unsigned, 4> DstRegs;
859#endif
860 for (unsigned i = 0; i != SubRegs; ++i) {
861 Register Dst = TRI->getSubReg(DestReg, BeginIdx + i * Spacing);
862 Register Src = TRI->getSubReg(SrcReg, BeginIdx + i * Spacing);
863 assert(Dst && Src && "Bad sub-register");
864#ifndef NDEBUG
865 assert(!DstRegs.count(Src) && "destructive vector copy");
866 DstRegs.insert(Dst);
867#endif
868 Mov = BuildMI(MBB, I, I->getDebugLoc(), get(Opc), Dst).addReg(Src);
869 // VORR (NEON or MVE) takes two source operands.
870 if (Opc == ARM::VORRq || Opc == ARM::MVE_VORR) {
871 Mov.addReg(Src);
872 }
873 // MVE VORR takes predicate operands in place of an ordinary condition.
874 if (Opc == ARM::MVE_VORR)
876 else
877 Mov = Mov.add(predOps(ARMCC::AL));
878 // MOVr can set CC.
879 if (Opc == ARM::MOVr)
880 Mov = Mov.add(condCodeOp());
881 }
882 // Add implicit super-register defs and kills to the last instruction.
883 Mov->addRegisterDefined(DestReg, TRI);
884 if (KillSrc)
885 Mov->addRegisterKilled(SrcReg, TRI);
886}
887
888std::optional<DestSourcePair>
890 // VMOVRRD is also a copy instruction but it requires
891 // special way of handling. It is more complex copy version
892 // and since that we are not considering it. For recognition
893 // of such instruction isExtractSubregLike MI interface function
894 // could be used.
895 // VORRq is considered as a move only if two inputs are
896 // the same register.
897 if (!MI.isMoveReg() ||
898 (MI.getOpcode() == ARM::VORRq &&
899 MI.getOperand(1).getReg() != MI.getOperand(2).getReg()))
900 return std::nullopt;
901 return DestSourcePair{MI.getOperand(0), MI.getOperand(1)};
902}
903
904std::optional<ParamLoadedValue>
906 Register Reg) const {
907 if (auto DstSrcPair = isCopyInstrImpl(MI)) {
908 Register DstReg = DstSrcPair->Destination->getReg();
909
910 // TODO: We don't handle cases where the forwarding reg is narrower/wider
911 // than the copy registers. Consider for example:
912 //
913 // s16 = VMOVS s0
914 // s17 = VMOVS s1
915 // call @callee(d0)
916 //
917 // We'd like to describe the call site value of d0 as d8, but this requires
918 // gathering and merging the descriptions for the two VMOVS instructions.
919 //
920 // We also don't handle the reverse situation, where the forwarding reg is
921 // narrower than the copy destination:
922 //
923 // d8 = VMOVD d0
924 // call @callee(s1)
925 //
926 // We need to produce a fragment description (the call site value of s1 is
927 // /not/ just d8).
928 if (DstReg != Reg)
929 return std::nullopt;
930 }
932}
933
934const MachineOperand &
936 assert(MI.isCall());
937
938 switch (MI.getOpcode()) {
939 case ARM::tBL:
940 case ARM::tBLXi:
941 case ARM::tBLXr:
942 case ARM::tBLXr_noip:
943 case ARM::tBLXNSr:
944 return MI.getOperand(2);
945 default:
947 }
948}
949
951 unsigned Reg,
952 unsigned SubIdx,
953 RegState State) const {
954 if (!SubIdx)
955 return MIB.addReg(Reg, State);
956
958 return MIB.addReg(getRegisterInfo().getSubReg(Reg, SubIdx), State);
959 return MIB.addReg(Reg, State, SubIdx);
960}
961
964 Register SrcReg, bool isKill, int FI,
965 const TargetRegisterClass *RC,
966 Register VReg,
967 MachineInstr::MIFlag Flags) const {
968 MachineFunction &MF = *MBB.getParent();
969 MachineFrameInfo &MFI = MF.getFrameInfo();
970 Align Alignment = MFI.getObjectAlign(FI);
972
975 MFI.getObjectSize(FI), Alignment);
976
977 switch (TRI.getSpillSize(*RC)) {
978 case 2:
979 if (ARM::HPRRegClass.hasSubClassEq(RC)) {
980 BuildMI(MBB, I, DebugLoc(), get(ARM::VSTRH))
981 .addReg(SrcReg, getKillRegState(isKill))
982 .addFrameIndex(FI)
983 .addImm(0)
984 .addMemOperand(MMO)
986 } else
987 llvm_unreachable("Unknown reg class!");
988 break;
989 case 4:
990 if (ARM::GPRRegClass.hasSubClassEq(RC)) {
991 BuildMI(MBB, I, DebugLoc(), get(ARM::STRi12))
992 .addReg(SrcReg, getKillRegState(isKill))
993 .addFrameIndex(FI)
994 .addImm(0)
995 .addMemOperand(MMO)
997 } else if (ARM::SPRRegClass.hasSubClassEq(RC)) {
998 BuildMI(MBB, I, DebugLoc(), get(ARM::VSTRS))
999 .addReg(SrcReg, getKillRegState(isKill))
1000 .addFrameIndex(FI)
1001 .addImm(0)
1002 .addMemOperand(MMO)
1004 } else if (ARM::VCCRRegClass.hasSubClassEq(RC)) {
1005 BuildMI(MBB, I, DebugLoc(), get(ARM::VSTR_P0_off))
1006 .addReg(SrcReg, getKillRegState(isKill))
1007 .addFrameIndex(FI)
1008 .addImm(0)
1009 .addMemOperand(MMO)
1011 } else if (ARM::cl_FPSCR_NZCVRegClass.hasSubClassEq(RC)) {
1012 BuildMI(MBB, I, DebugLoc(), get(ARM::VSTR_FPSCR_NZCVQC_off))
1013 .addReg(SrcReg, getKillRegState(isKill))
1014 .addFrameIndex(FI)
1015 .addImm(0)
1016 .addMemOperand(MMO)
1018 } else
1019 llvm_unreachable("Unknown reg class!");
1020 break;
1021 case 8:
1022 if (ARM::DPRRegClass.hasSubClassEq(RC)) {
1023 BuildMI(MBB, I, DebugLoc(), get(ARM::VSTRD))
1024 .addReg(SrcReg, getKillRegState(isKill))
1025 .addFrameIndex(FI)
1026 .addImm(0)
1027 .addMemOperand(MMO)
1029 } else if (ARM::GPRPairRegClass.hasSubClassEq(RC)) {
1030 if (Subtarget.hasV5TEOps()) {
1031 MachineInstrBuilder MIB = BuildMI(MBB, I, DebugLoc(), get(ARM::STRD));
1032 AddDReg(MIB, SrcReg, ARM::gsub_0, getKillRegState(isKill));
1033 AddDReg(MIB, SrcReg, ARM::gsub_1, {});
1034 MIB.addFrameIndex(FI).addReg(0).addImm(0).addMemOperand(MMO)
1036 } else {
1037 // Fallback to STM instruction, which has existed since the dawn of
1038 // time.
1039 MachineInstrBuilder MIB = BuildMI(MBB, I, DebugLoc(), get(ARM::STMIA))
1040 .addFrameIndex(FI)
1041 .addMemOperand(MMO)
1043 AddDReg(MIB, SrcReg, ARM::gsub_0, getKillRegState(isKill));
1044 AddDReg(MIB, SrcReg, ARM::gsub_1, {});
1045 }
1046 } else
1047 llvm_unreachable("Unknown reg class!");
1048 break;
1049 case 16:
1050 if (ARM::DPairRegClass.hasSubClassEq(RC) && Subtarget.hasNEON()) {
1051 // Use aligned spills if the stack can be realigned.
1052 if (Alignment >= 16 && getRegisterInfo().canRealignStack(MF)) {
1053 BuildMI(MBB, I, DebugLoc(), get(ARM::VST1q64))
1054 .addFrameIndex(FI)
1055 .addImm(16)
1056 .addReg(SrcReg, getKillRegState(isKill))
1057 .addMemOperand(MMO)
1059 } else {
1060 BuildMI(MBB, I, DebugLoc(), get(ARM::VSTMQIA))
1061 .addReg(SrcReg, getKillRegState(isKill))
1062 .addFrameIndex(FI)
1063 .addMemOperand(MMO)
1065 }
1066 } else if (ARM::QPRRegClass.hasSubClassEq(RC) &&
1067 Subtarget.hasMVEIntegerOps()) {
1068 auto MIB = BuildMI(MBB, I, DebugLoc(), get(ARM::MVE_VSTRWU32));
1069 MIB.addReg(SrcReg, getKillRegState(isKill))
1070 .addFrameIndex(FI)
1071 .addImm(0)
1072 .addMemOperand(MMO);
1074 } else
1075 llvm_unreachable("Unknown reg class!");
1076 break;
1077 case 24:
1078 if (ARM::DTripleRegClass.hasSubClassEq(RC)) {
1079 // Use aligned spills if the stack can be realigned.
1080 if (Alignment >= 16 && getRegisterInfo().canRealignStack(MF) &&
1081 Subtarget.hasNEON()) {
1082 BuildMI(MBB, I, DebugLoc(), get(ARM::VST1d64TPseudo))
1083 .addFrameIndex(FI)
1084 .addImm(16)
1085 .addReg(SrcReg, getKillRegState(isKill))
1086 .addMemOperand(MMO)
1088 } else {
1090 get(ARM::VSTMDIA))
1091 .addFrameIndex(FI)
1093 .addMemOperand(MMO);
1094 MIB = AddDReg(MIB, SrcReg, ARM::dsub_0, getKillRegState(isKill));
1095 MIB = AddDReg(MIB, SrcReg, ARM::dsub_1, {});
1096 AddDReg(MIB, SrcReg, ARM::dsub_2, {});
1097 }
1098 } else
1099 llvm_unreachable("Unknown reg class!");
1100 break;
1101 case 32:
1102 if (ARM::QQPRRegClass.hasSubClassEq(RC) ||
1103 ARM::MQQPRRegClass.hasSubClassEq(RC) ||
1104 ARM::DQuadRegClass.hasSubClassEq(RC)) {
1105 if (Alignment >= 16 && getRegisterInfo().canRealignStack(MF) &&
1106 Subtarget.hasNEON()) {
1107 // FIXME: It's possible to only store part of the QQ register if the
1108 // spilled def has a sub-register index.
1109 BuildMI(MBB, I, DebugLoc(), get(ARM::VST1d64QPseudo))
1110 .addFrameIndex(FI)
1111 .addImm(16)
1112 .addReg(SrcReg, getKillRegState(isKill))
1113 .addMemOperand(MMO)
1115 } else if (Subtarget.hasMVEIntegerOps()) {
1116 BuildMI(MBB, I, DebugLoc(), get(ARM::MQQPRStore))
1117 .addReg(SrcReg, getKillRegState(isKill))
1118 .addFrameIndex(FI)
1119 .addMemOperand(MMO);
1120 } else {
1122 get(ARM::VSTMDIA))
1123 .addFrameIndex(FI)
1125 .addMemOperand(MMO);
1126 MIB = AddDReg(MIB, SrcReg, ARM::dsub_0, getKillRegState(isKill));
1127 MIB = AddDReg(MIB, SrcReg, ARM::dsub_1, {});
1128 MIB = AddDReg(MIB, SrcReg, ARM::dsub_2, {});
1129 AddDReg(MIB, SrcReg, ARM::dsub_3, {});
1130 }
1131 } else
1132 llvm_unreachable("Unknown reg class!");
1133 break;
1134 case 64:
1135 if (ARM::MQQQQPRRegClass.hasSubClassEq(RC) &&
1136 Subtarget.hasMVEIntegerOps()) {
1137 BuildMI(MBB, I, DebugLoc(), get(ARM::MQQQQPRStore))
1138 .addReg(SrcReg, getKillRegState(isKill))
1139 .addFrameIndex(FI)
1140 .addMemOperand(MMO);
1141 } else if (ARM::QQQQPRRegClass.hasSubClassEq(RC)) {
1142 MachineInstrBuilder MIB = BuildMI(MBB, I, DebugLoc(), get(ARM::VSTMDIA))
1143 .addFrameIndex(FI)
1145 .addMemOperand(MMO);
1146 MIB = AddDReg(MIB, SrcReg, ARM::dsub_0, getKillRegState(isKill));
1147 MIB = AddDReg(MIB, SrcReg, ARM::dsub_1, {});
1148 MIB = AddDReg(MIB, SrcReg, ARM::dsub_2, {});
1149 MIB = AddDReg(MIB, SrcReg, ARM::dsub_3, {});
1150 MIB = AddDReg(MIB, SrcReg, ARM::dsub_4, {});
1151 MIB = AddDReg(MIB, SrcReg, ARM::dsub_5, {});
1152 MIB = AddDReg(MIB, SrcReg, ARM::dsub_6, {});
1153 AddDReg(MIB, SrcReg, ARM::dsub_7, {});
1154 } else
1155 llvm_unreachable("Unknown reg class!");
1156 break;
1157 default:
1158 llvm_unreachable("Unknown reg class!");
1159 }
1160}
1161
1163 int &FrameIndex) const {
1164 switch (MI.getOpcode()) {
1165 default: break;
1166 case ARM::STRrs:
1167 case ARM::t2STRs: // FIXME: don't use t2STRs to access frame.
1168 if (MI.getOperand(1).isFI() && MI.getOperand(2).isReg() &&
1169 MI.getOperand(3).isImm() && MI.getOperand(2).getReg() == 0 &&
1170 MI.getOperand(3).getImm() == 0) {
1171 FrameIndex = MI.getOperand(1).getIndex();
1172 return MI.getOperand(0).getReg();
1173 }
1174 break;
1175 case ARM::STRi12:
1176 case ARM::t2STRi12:
1177 case ARM::tSTRspi:
1178 case ARM::VSTRD:
1179 case ARM::VSTRS:
1180 case ARM::VSTRH:
1181 case ARM::VSTR_P0_off:
1182 case ARM::VSTR_FPSCR_NZCVQC_off:
1183 case ARM::MVE_VSTRWU32:
1184 if (MI.getOperand(1).isFI() && MI.getOperand(2).isImm() &&
1185 MI.getOperand(2).getImm() == 0) {
1186 FrameIndex = MI.getOperand(1).getIndex();
1187 return MI.getOperand(0).getReg();
1188 }
1189 break;
1190 case ARM::VST1q64:
1191 case ARM::VST1d64TPseudo:
1192 case ARM::VST1d64QPseudo:
1193 if (MI.getOperand(0).isFI() && MI.getOperand(2).getSubReg() == 0) {
1194 FrameIndex = MI.getOperand(0).getIndex();
1195 return MI.getOperand(2).getReg();
1196 }
1197 break;
1198 case ARM::VSTMQIA:
1199 if (MI.getOperand(1).isFI() && MI.getOperand(0).getSubReg() == 0) {
1200 FrameIndex = MI.getOperand(1).getIndex();
1201 return MI.getOperand(0).getReg();
1202 }
1203 break;
1204 case ARM::MQQPRStore:
1205 case ARM::MQQQQPRStore:
1206 if (MI.getOperand(1).isFI()) {
1207 FrameIndex = MI.getOperand(1).getIndex();
1208 return MI.getOperand(0).getReg();
1209 }
1210 break;
1211 }
1212
1213 return 0;
1214}
1215
1217 int &FrameIndex) const {
1219 if (MI.mayStore() && hasStoreToStackSlot(MI, Accesses) &&
1220 Accesses.size() == 1) {
1221 FrameIndex =
1222 cast<FixedStackPseudoSourceValue>(Accesses.front()->getPseudoValue())
1223 ->getFrameIndex();
1224 return true;
1225 }
1226 return false;
1227}
1228
1231 Register DestReg, int FI,
1232 const TargetRegisterClass *RC,
1233 Register VReg, unsigned SubReg,
1234 MachineInstr::MIFlag Flags) const {
1235 DebugLoc DL;
1236 if (I != MBB.end()) DL = I->getDebugLoc();
1237 MachineFunction &MF = *MBB.getParent();
1238 MachineFrameInfo &MFI = MF.getFrameInfo();
1239 const Align Alignment = MFI.getObjectAlign(FI);
1242 MFI.getObjectSize(FI), Alignment);
1243
1245 switch (TRI.getSpillSize(*RC)) {
1246 case 2:
1247 if (ARM::HPRRegClass.hasSubClassEq(RC)) {
1248 BuildMI(MBB, I, DL, get(ARM::VLDRH), DestReg)
1249 .addFrameIndex(FI)
1250 .addImm(0)
1251 .addMemOperand(MMO)
1253 } else
1254 llvm_unreachable("Unknown reg class!");
1255 break;
1256 case 4:
1257 if (ARM::GPRRegClass.hasSubClassEq(RC)) {
1258 BuildMI(MBB, I, DL, get(ARM::LDRi12), DestReg)
1259 .addFrameIndex(FI)
1260 .addImm(0)
1261 .addMemOperand(MMO)
1263 } else if (ARM::SPRRegClass.hasSubClassEq(RC)) {
1264 BuildMI(MBB, I, DL, get(ARM::VLDRS), DestReg)
1265 .addFrameIndex(FI)
1266 .addImm(0)
1267 .addMemOperand(MMO)
1269 } else if (ARM::VCCRRegClass.hasSubClassEq(RC)) {
1270 BuildMI(MBB, I, DL, get(ARM::VLDR_P0_off), DestReg)
1271 .addFrameIndex(FI)
1272 .addImm(0)
1273 .addMemOperand(MMO)
1275 } else if (ARM::cl_FPSCR_NZCVRegClass.hasSubClassEq(RC)) {
1276 BuildMI(MBB, I, DL, get(ARM::VLDR_FPSCR_NZCVQC_off), DestReg)
1277 .addFrameIndex(FI)
1278 .addImm(0)
1279 .addMemOperand(MMO)
1281 } else
1282 llvm_unreachable("Unknown reg class!");
1283 break;
1284 case 8:
1285 if (ARM::DPRRegClass.hasSubClassEq(RC)) {
1286 BuildMI(MBB, I, DL, get(ARM::VLDRD), DestReg)
1287 .addFrameIndex(FI)
1288 .addImm(0)
1289 .addMemOperand(MMO)
1291 } else if (ARM::GPRPairRegClass.hasSubClassEq(RC)) {
1293
1294 if (Subtarget.hasV5TEOps()) {
1295 MIB = BuildMI(MBB, I, DL, get(ARM::LDRD));
1296 AddDReg(MIB, DestReg, ARM::gsub_0, RegState::DefineNoRead);
1297 AddDReg(MIB, DestReg, ARM::gsub_1, RegState::DefineNoRead);
1298 MIB.addFrameIndex(FI).addReg(0).addImm(0).addMemOperand(MMO)
1300 } else {
1301 // Fallback to LDM instruction, which has existed since the dawn of
1302 // time.
1303 MIB = BuildMI(MBB, I, DL, get(ARM::LDMIA))
1304 .addFrameIndex(FI)
1305 .addMemOperand(MMO)
1307 MIB = AddDReg(MIB, DestReg, ARM::gsub_0, RegState::DefineNoRead);
1308 MIB = AddDReg(MIB, DestReg, ARM::gsub_1, RegState::DefineNoRead);
1309 }
1310
1311 if (DestReg.isPhysical())
1312 MIB.addReg(DestReg, RegState::ImplicitDefine);
1313 } else
1314 llvm_unreachable("Unknown reg class!");
1315 break;
1316 case 16:
1317 if (ARM::DPairRegClass.hasSubClassEq(RC) && Subtarget.hasNEON()) {
1318 if (Alignment >= 16 && getRegisterInfo().canRealignStack(MF)) {
1319 BuildMI(MBB, I, DL, get(ARM::VLD1q64), DestReg)
1320 .addFrameIndex(FI)
1321 .addImm(16)
1322 .addMemOperand(MMO)
1324 } else {
1325 BuildMI(MBB, I, DL, get(ARM::VLDMQIA), DestReg)
1326 .addFrameIndex(FI)
1327 .addMemOperand(MMO)
1329 }
1330 } else if (ARM::QPRRegClass.hasSubClassEq(RC) &&
1331 Subtarget.hasMVEIntegerOps()) {
1332 auto MIB = BuildMI(MBB, I, DL, get(ARM::MVE_VLDRWU32), DestReg);
1333 MIB.addFrameIndex(FI)
1334 .addImm(0)
1335 .addMemOperand(MMO);
1337 } else
1338 llvm_unreachable("Unknown reg class!");
1339 break;
1340 case 24:
1341 if (ARM::DTripleRegClass.hasSubClassEq(RC)) {
1342 if (Alignment >= 16 && getRegisterInfo().canRealignStack(MF) &&
1343 Subtarget.hasNEON()) {
1344 BuildMI(MBB, I, DL, get(ARM::VLD1d64TPseudo), DestReg)
1345 .addFrameIndex(FI)
1346 .addImm(16)
1347 .addMemOperand(MMO)
1349 } else {
1350 MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(ARM::VLDMDIA))
1351 .addFrameIndex(FI)
1352 .addMemOperand(MMO)
1354 MIB = AddDReg(MIB, DestReg, ARM::dsub_0, RegState::DefineNoRead);
1355 MIB = AddDReg(MIB, DestReg, ARM::dsub_1, RegState::DefineNoRead);
1356 MIB = AddDReg(MIB, DestReg, ARM::dsub_2, RegState::DefineNoRead);
1357 if (DestReg.isPhysical())
1358 MIB.addReg(DestReg, RegState::ImplicitDefine);
1359 }
1360 } else
1361 llvm_unreachable("Unknown reg class!");
1362 break;
1363 case 32:
1364 if (ARM::QQPRRegClass.hasSubClassEq(RC) ||
1365 ARM::MQQPRRegClass.hasSubClassEq(RC) ||
1366 ARM::DQuadRegClass.hasSubClassEq(RC)) {
1367 if (Alignment >= 16 && getRegisterInfo().canRealignStack(MF) &&
1368 Subtarget.hasNEON()) {
1369 BuildMI(MBB, I, DL, get(ARM::VLD1d64QPseudo), DestReg)
1370 .addFrameIndex(FI)
1371 .addImm(16)
1372 .addMemOperand(MMO)
1374 } else if (Subtarget.hasMVEIntegerOps()) {
1375 BuildMI(MBB, I, DL, get(ARM::MQQPRLoad), DestReg)
1376 .addFrameIndex(FI)
1377 .addMemOperand(MMO);
1378 } else {
1379 MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(ARM::VLDMDIA))
1380 .addFrameIndex(FI)
1382 .addMemOperand(MMO);
1383 MIB = AddDReg(MIB, DestReg, ARM::dsub_0, RegState::DefineNoRead);
1384 MIB = AddDReg(MIB, DestReg, ARM::dsub_1, RegState::DefineNoRead);
1385 MIB = AddDReg(MIB, DestReg, ARM::dsub_2, RegState::DefineNoRead);
1386 MIB = AddDReg(MIB, DestReg, ARM::dsub_3, RegState::DefineNoRead);
1387 if (DestReg.isPhysical())
1388 MIB.addReg(DestReg, RegState::ImplicitDefine);
1389 }
1390 } else
1391 llvm_unreachable("Unknown reg class!");
1392 break;
1393 case 64:
1394 if (ARM::MQQQQPRRegClass.hasSubClassEq(RC) &&
1395 Subtarget.hasMVEIntegerOps()) {
1396 BuildMI(MBB, I, DL, get(ARM::MQQQQPRLoad), DestReg)
1397 .addFrameIndex(FI)
1398 .addMemOperand(MMO);
1399 } else if (ARM::QQQQPRRegClass.hasSubClassEq(RC)) {
1400 MachineInstrBuilder MIB = BuildMI(MBB, I, DL, get(ARM::VLDMDIA))
1401 .addFrameIndex(FI)
1403 .addMemOperand(MMO);
1404 MIB = AddDReg(MIB, DestReg, ARM::dsub_0, RegState::DefineNoRead);
1405 MIB = AddDReg(MIB, DestReg, ARM::dsub_1, RegState::DefineNoRead);
1406 MIB = AddDReg(MIB, DestReg, ARM::dsub_2, RegState::DefineNoRead);
1407 MIB = AddDReg(MIB, DestReg, ARM::dsub_3, RegState::DefineNoRead);
1408 MIB = AddDReg(MIB, DestReg, ARM::dsub_4, RegState::DefineNoRead);
1409 MIB = AddDReg(MIB, DestReg, ARM::dsub_5, RegState::DefineNoRead);
1410 MIB = AddDReg(MIB, DestReg, ARM::dsub_6, RegState::DefineNoRead);
1411 MIB = AddDReg(MIB, DestReg, ARM::dsub_7, RegState::DefineNoRead);
1412 if (DestReg.isPhysical())
1413 MIB.addReg(DestReg, RegState::ImplicitDefine);
1414 } else
1415 llvm_unreachable("Unknown reg class!");
1416 break;
1417 default:
1418 llvm_unreachable("Unknown regclass!");
1419 }
1420}
1421
1423 int &FrameIndex) const {
1424 switch (MI.getOpcode()) {
1425 default: break;
1426 case ARM::LDRrs:
1427 case ARM::t2LDRs: // FIXME: don't use t2LDRs to access frame.
1428 if (MI.getOperand(1).isFI() && MI.getOperand(2).isReg() &&
1429 MI.getOperand(3).isImm() && MI.getOperand(2).getReg() == 0 &&
1430 MI.getOperand(3).getImm() == 0) {
1431 FrameIndex = MI.getOperand(1).getIndex();
1432 return MI.getOperand(0).getReg();
1433 }
1434 break;
1435 case ARM::LDRi12:
1436 case ARM::t2LDRi12:
1437 case ARM::tLDRspi:
1438 case ARM::VLDRD:
1439 case ARM::VLDRS:
1440 case ARM::VLDRH:
1441 case ARM::VLDR_P0_off:
1442 case ARM::VLDR_FPSCR_NZCVQC_off:
1443 case ARM::MVE_VLDRWU32:
1444 if (MI.getOperand(1).isFI() && MI.getOperand(2).isImm() &&
1445 MI.getOperand(2).getImm() == 0) {
1446 FrameIndex = MI.getOperand(1).getIndex();
1447 return MI.getOperand(0).getReg();
1448 }
1449 break;
1450 case ARM::VLD1q64:
1451 case ARM::VLD1d8TPseudo:
1452 case ARM::VLD1d16TPseudo:
1453 case ARM::VLD1d32TPseudo:
1454 case ARM::VLD1d64TPseudo:
1455 case ARM::VLD1d8QPseudo:
1456 case ARM::VLD1d16QPseudo:
1457 case ARM::VLD1d32QPseudo:
1458 case ARM::VLD1d64QPseudo:
1459 if (MI.getOperand(1).isFI() && MI.getOperand(0).getSubReg() == 0) {
1460 FrameIndex = MI.getOperand(1).getIndex();
1461 return MI.getOperand(0).getReg();
1462 }
1463 break;
1464 case ARM::VLDMQIA:
1465 if (MI.getOperand(1).isFI() && MI.getOperand(0).getSubReg() == 0) {
1466 FrameIndex = MI.getOperand(1).getIndex();
1467 return MI.getOperand(0).getReg();
1468 }
1469 break;
1470 case ARM::MQQPRLoad:
1471 case ARM::MQQQQPRLoad:
1472 if (MI.getOperand(1).isFI()) {
1473 FrameIndex = MI.getOperand(1).getIndex();
1474 return MI.getOperand(0).getReg();
1475 }
1476 break;
1477 }
1478
1479 return 0;
1480}
1481
1483 int &FrameIndex) const {
1485 if (MI.mayLoad() && hasLoadFromStackSlot(MI, Accesses) &&
1486 Accesses.size() == 1) {
1487 FrameIndex =
1488 cast<FixedStackPseudoSourceValue>(Accesses.front()->getPseudoValue())
1489 ->getFrameIndex();
1490 return true;
1491 }
1492 return false;
1493}
1494
1495/// Expands MEMCPY to either LDMIA/STMIA or LDMIA_UPD/STMID_UPD
1496/// depending on whether the result is used.
1497void ARMBaseInstrInfo::expandMEMCPY(MachineBasicBlock::iterator MI) const {
1498 bool isThumb1 = Subtarget.isThumb1Only();
1499 bool isThumb2 = Subtarget.isThumb2();
1500 const ARMBaseInstrInfo *TII = Subtarget.getInstrInfo();
1501
1502 DebugLoc dl = MI->getDebugLoc();
1503 MachineBasicBlock *BB = MI->getParent();
1504
1505 MachineInstrBuilder LDM, STM;
1506 if (isThumb1 || !MI->getOperand(1).isDead()) {
1507 MachineOperand LDWb(MI->getOperand(1));
1508 LDM = BuildMI(*BB, MI, dl, TII->get(isThumb2 ? ARM::t2LDMIA_UPD
1509 : isThumb1 ? ARM::tLDMIA_UPD
1510 : ARM::LDMIA_UPD))
1511 .add(LDWb);
1512 } else {
1513 LDM = BuildMI(*BB, MI, dl, TII->get(isThumb2 ? ARM::t2LDMIA : ARM::LDMIA));
1514 }
1515
1516 if (isThumb1 || !MI->getOperand(0).isDead()) {
1517 MachineOperand STWb(MI->getOperand(0));
1518 STM = BuildMI(*BB, MI, dl, TII->get(isThumb2 ? ARM::t2STMIA_UPD
1519 : isThumb1 ? ARM::tSTMIA_UPD
1520 : ARM::STMIA_UPD))
1521 .add(STWb);
1522 } else {
1523 STM = BuildMI(*BB, MI, dl, TII->get(isThumb2 ? ARM::t2STMIA : ARM::STMIA));
1524 }
1525
1526 MachineOperand LDBase(MI->getOperand(3));
1527 LDM.add(LDBase).add(predOps(ARMCC::AL));
1528
1529 MachineOperand STBase(MI->getOperand(2));
1530 STM.add(STBase).add(predOps(ARMCC::AL));
1531
1532 // Sort the scratch registers into ascending order.
1533 const TargetRegisterInfo &TRI = getRegisterInfo();
1534 SmallVector<unsigned, 6> ScratchRegs;
1535 for (MachineOperand &MO : llvm::drop_begin(MI->operands(), 5))
1536 ScratchRegs.push_back(MO.getReg());
1537 llvm::sort(ScratchRegs,
1538 [&TRI](const unsigned &Reg1, const unsigned &Reg2) -> bool {
1539 return TRI.getEncodingValue(Reg1) <
1540 TRI.getEncodingValue(Reg2);
1541 });
1542
1543 for (const auto &Reg : ScratchRegs) {
1546 }
1547
1548 BB->erase(MI);
1549}
1550
1552 if (MI.getOpcode() == TargetOpcode::LOAD_STACK_GUARD) {
1553 expandLoadStackGuard(MI);
1554 MI.getParent()->erase(MI);
1555 return true;
1556 }
1557
1558 if (MI.getOpcode() == ARM::MEMCPY) {
1559 expandMEMCPY(MI);
1560 return true;
1561 }
1562
1563 // This hook gets to expand COPY instructions before they become
1564 // copyPhysReg() calls. Look for VMOVS instructions that can legally be
1565 // widened to VMOVD. We prefer the VMOVD when possible because it may be
1566 // changed into a VORR that can go down the NEON pipeline.
1567 if (!MI.isCopy() || Subtarget.dontWidenVMOVS() || !Subtarget.hasFP64())
1568 return false;
1569
1570 // Look for a copy between even S-registers. That is where we keep floats
1571 // when using NEON v2f32 instructions for f32 arithmetic.
1572 Register DstRegS = MI.getOperand(0).getReg();
1573 Register SrcRegS = MI.getOperand(1).getReg();
1574 if (!ARM::SPRRegClass.contains(DstRegS, SrcRegS))
1575 return false;
1576
1578 MCRegister DstRegD =
1579 TRI->getMatchingSuperReg(DstRegS, ARM::ssub_0, &ARM::DPRRegClass);
1580 MCRegister SrcRegD =
1581 TRI->getMatchingSuperReg(SrcRegS, ARM::ssub_0, &ARM::DPRRegClass);
1582 if (!DstRegD || !SrcRegD)
1583 return false;
1584
1585 // We want to widen this into a DstRegD = VMOVD SrcRegD copy. This is only
1586 // legal if the COPY already defines the full DstRegD, and it isn't a
1587 // sub-register insertion.
1588 if (!MI.definesRegister(DstRegD, TRI) || MI.readsRegister(DstRegD, TRI))
1589 return false;
1590
1591 // A dead copy shouldn't show up here, but reject it just in case.
1592 if (MI.getOperand(0).isDead())
1593 return false;
1594
1595 // All clear, widen the COPY.
1596 LLVM_DEBUG(dbgs() << "widening: " << MI);
1597 MachineInstrBuilder MIB(*MI.getParent()->getParent(), MI);
1598
1599 // Get rid of the old implicit-def of DstRegD. Leave it if it defines a Q-reg
1600 // or some other super-register.
1601 int ImpDefIdx = MI.findRegisterDefOperandIdx(DstRegD, /*TRI=*/nullptr);
1602 if (ImpDefIdx != -1)
1603 MI.removeOperand(ImpDefIdx);
1604
1605 // Change the opcode and operands.
1606 MI.setDesc(get(ARM::VMOVD));
1607 MI.getOperand(0).setReg(DstRegD);
1608 MI.getOperand(1).setReg(SrcRegD);
1609 MIB.add(predOps(ARMCC::AL));
1610
1611 // We are now reading SrcRegD instead of SrcRegS. This may upset the
1612 // register scavenger and machine verifier, so we need to indicate that we
1613 // are reading an undefined value from SrcRegD, but a proper value from
1614 // SrcRegS.
1615 MI.getOperand(1).setIsUndef();
1616 MIB.addReg(SrcRegS, RegState::Implicit);
1617
1618 // SrcRegD may actually contain an unrelated value in the ssub_1
1619 // sub-register. Don't kill it. Only kill the ssub_0 sub-register.
1620 if (MI.getOperand(1).isKill()) {
1621 MI.getOperand(1).setIsKill(false);
1622 MI.addRegisterKilled(SrcRegS, TRI, true);
1623 }
1624
1625 LLVM_DEBUG(dbgs() << "replaced by: " << MI);
1626 return true;
1627}
1628
1629/// Create a copy of a const pool value. Update CPI to the new index and return
1630/// the label UID.
1631static unsigned duplicateCPV(MachineFunction &MF, unsigned &CPI) {
1634
1635 const MachineConstantPoolEntry &MCPE = MCP->getConstants()[CPI];
1636 assert(MCPE.isMachineConstantPoolEntry() &&
1637 "Expecting a machine constantpool entry!");
1638 ARMConstantPoolValue *ACPV =
1639 static_cast<ARMConstantPoolValue*>(MCPE.Val.MachineCPVal);
1640
1641 unsigned PCLabelId = AFI->createPICLabelUId();
1642 ARMConstantPoolValue *NewCPV = nullptr;
1643
1644 // FIXME: The below assumes PIC relocation model and that the function
1645 // is Thumb mode (t1 or t2). PCAdjustment would be 8 for ARM mode PIC, and
1646 // zero for non-PIC in ARM or Thumb. The callers are all of thumb LDR
1647 // instructions, so that's probably OK, but is PIC always correct when
1648 // we get here?
1649 if (ACPV->isGlobalValue())
1651 cast<ARMConstantPoolConstant>(ACPV)->getGV(), PCLabelId, ARMCP::CPValue,
1652 4, ACPV->getModifier(), ACPV->mustAddCurrentAddress());
1653 else if (ACPV->isExtSymbol())
1656 cast<ARMConstantPoolSymbol>(ACPV)->getSymbol(), PCLabelId, 4);
1657 else if (ACPV->isBlockAddress())
1659 Create(cast<ARMConstantPoolConstant>(ACPV)->getBlockAddress(), PCLabelId,
1661 else if (ACPV->isLSDA())
1662 NewCPV = ARMConstantPoolConstant::Create(&MF.getFunction(), PCLabelId,
1663 ARMCP::CPLSDA, 4);
1664 else if (ACPV->isMachineBasicBlock())
1665 NewCPV = ARMConstantPoolMBB::
1667 cast<ARMConstantPoolMBB>(ACPV)->getMBB(), PCLabelId, 4);
1668 else
1669 llvm_unreachable("Unexpected ARM constantpool value type!!");
1670 CPI = MCP->getConstantPoolIndex(NewCPV, MCPE.getAlign());
1671 return PCLabelId;
1672}
1673
1676 Register DestReg, unsigned SubIdx,
1677 const MachineInstr &Orig,
1678 LaneBitmask UsedLanes) const {
1679 unsigned Opcode = Orig.getOpcode();
1680 switch (Opcode) {
1681 default: {
1682 MachineInstr *MI = MBB.getParent()->CloneMachineInstr(&Orig);
1683 MI->substituteRegister(Orig.getOperand(0).getReg(), DestReg, SubIdx, TRI);
1684 MBB.insert(I, MI);
1685 break;
1686 }
1687 case ARM::tLDRpci_pic:
1688 case ARM::t2LDRpci_pic: {
1689 MachineFunction &MF = *MBB.getParent();
1690 unsigned CPI = Orig.getOperand(1).getIndex();
1691 unsigned PCLabelId = duplicateCPV(MF, CPI);
1692 BuildMI(MBB, I, Orig.getDebugLoc(), get(Opcode), DestReg)
1694 .addImm(PCLabelId)
1695 .cloneMemRefs(Orig);
1696 break;
1697 }
1698 }
1699}
1700
1703 MachineBasicBlock::iterator InsertBefore,
1704 const MachineInstr &Orig) const {
1705 MachineInstr &Cloned = TargetInstrInfo::duplicate(MBB, InsertBefore, Orig);
1707 for (;;) {
1708 switch (I->getOpcode()) {
1709 case ARM::tLDRpci_pic:
1710 case ARM::t2LDRpci_pic: {
1711 MachineFunction &MF = *MBB.getParent();
1712 unsigned CPI = I->getOperand(1).getIndex();
1713 unsigned PCLabelId = duplicateCPV(MF, CPI);
1714 I->getOperand(1).setIndex(CPI);
1715 I->getOperand(2).setImm(PCLabelId);
1716 break;
1717 }
1718 }
1719 if (!I->isBundledWithSucc())
1720 break;
1721 ++I;
1722 }
1723 return Cloned;
1724}
1725
1727 const MachineInstr &MI1,
1728 const MachineRegisterInfo *MRI) const {
1729 unsigned Opcode = MI0.getOpcode();
1730 if (Opcode == ARM::t2LDRpci || Opcode == ARM::t2LDRpci_pic ||
1731 Opcode == ARM::tLDRpci || Opcode == ARM::tLDRpci_pic ||
1732 Opcode == ARM::LDRLIT_ga_pcrel || Opcode == ARM::LDRLIT_ga_pcrel_ldr ||
1733 Opcode == ARM::tLDRLIT_ga_pcrel || Opcode == ARM::t2LDRLIT_ga_pcrel ||
1734 Opcode == ARM::MOV_ga_pcrel || Opcode == ARM::MOV_ga_pcrel_ldr ||
1735 Opcode == ARM::t2MOV_ga_pcrel) {
1736 if (MI1.getOpcode() != Opcode)
1737 return false;
1738 if (MI0.getNumOperands() != MI1.getNumOperands())
1739 return false;
1740
1741 const MachineOperand &MO0 = MI0.getOperand(1);
1742 const MachineOperand &MO1 = MI1.getOperand(1);
1743 if (MO0.getOffset() != MO1.getOffset())
1744 return false;
1745
1746 if (Opcode == ARM::LDRLIT_ga_pcrel || Opcode == ARM::LDRLIT_ga_pcrel_ldr ||
1747 Opcode == ARM::tLDRLIT_ga_pcrel || Opcode == ARM::t2LDRLIT_ga_pcrel ||
1748 Opcode == ARM::MOV_ga_pcrel || Opcode == ARM::MOV_ga_pcrel_ldr ||
1749 Opcode == ARM::t2MOV_ga_pcrel)
1750 // Ignore the PC labels.
1751 return MO0.getGlobal() == MO1.getGlobal();
1752
1753 const MachineFunction *MF = MI0.getParent()->getParent();
1754 const MachineConstantPool *MCP = MF->getConstantPool();
1755 int CPI0 = MO0.getIndex();
1756 int CPI1 = MO1.getIndex();
1757 const MachineConstantPoolEntry &MCPE0 = MCP->getConstants()[CPI0];
1758 const MachineConstantPoolEntry &MCPE1 = MCP->getConstants()[CPI1];
1759 bool isARMCP0 = MCPE0.isMachineConstantPoolEntry();
1760 bool isARMCP1 = MCPE1.isMachineConstantPoolEntry();
1761 if (isARMCP0 && isARMCP1) {
1762 ARMConstantPoolValue *ACPV0 =
1763 static_cast<ARMConstantPoolValue*>(MCPE0.Val.MachineCPVal);
1764 ARMConstantPoolValue *ACPV1 =
1765 static_cast<ARMConstantPoolValue*>(MCPE1.Val.MachineCPVal);
1766 return ACPV0->hasSameValue(ACPV1);
1767 } else if (!isARMCP0 && !isARMCP1) {
1768 return MCPE0.Val.ConstVal == MCPE1.Val.ConstVal;
1769 }
1770 return false;
1771 } else if (Opcode == ARM::PICLDR) {
1772 if (MI1.getOpcode() != Opcode)
1773 return false;
1774 if (MI0.getNumOperands() != MI1.getNumOperands())
1775 return false;
1776
1777 Register Addr0 = MI0.getOperand(1).getReg();
1778 Register Addr1 = MI1.getOperand(1).getReg();
1779 if (Addr0 != Addr1) {
1780 if (!MRI || !Addr0.isVirtual() || !Addr1.isVirtual())
1781 return false;
1782
1783 // This assumes SSA form.
1784 MachineInstr *Def0 = MRI->getVRegDef(Addr0);
1785 MachineInstr *Def1 = MRI->getVRegDef(Addr1);
1786 // Check if the loaded value, e.g. a constantpool of a global address, are
1787 // the same.
1788 if (!produceSameValue(*Def0, *Def1, MRI))
1789 return false;
1790 }
1791
1792 for (unsigned i = 3, e = MI0.getNumOperands(); i != e; ++i) {
1793 // %12 = PICLDR %11, 0, 14, %noreg
1794 const MachineOperand &MO0 = MI0.getOperand(i);
1795 const MachineOperand &MO1 = MI1.getOperand(i);
1796 if (!MO0.isIdenticalTo(MO1))
1797 return false;
1798 }
1799 return true;
1800 }
1801
1803}
1804
1805/// areLoadsFromSameBasePtr - This is used by the pre-regalloc scheduler to
1806/// determine if two loads are loading from the same base address. It should
1807/// only return true if the base pointers are the same and the only differences
1808/// between the two addresses is the offset. It also returns the offsets by
1809/// reference.
1810///
1811/// FIXME: remove this in favor of the MachineInstr interface once pre-RA-sched
1812/// is permanently disabled.
1814 int64_t &Offset1,
1815 int64_t &Offset2) const {
1816 // Don't worry about Thumb: just ARM and Thumb2.
1817 if (Subtarget.isThumb1Only()) return false;
1818
1819 if (!Load1->isMachineOpcode() || !Load2->isMachineOpcode())
1820 return false;
1821
1822 auto IsLoadOpcode = [&](unsigned Opcode) {
1823 switch (Opcode) {
1824 default:
1825 return false;
1826 case ARM::LDRi12:
1827 case ARM::LDRBi12:
1828 case ARM::LDRD:
1829 case ARM::LDRH:
1830 case ARM::LDRSB:
1831 case ARM::LDRSH:
1832 case ARM::VLDRD:
1833 case ARM::VLDRS:
1834 case ARM::t2LDRi8:
1835 case ARM::t2LDRBi8:
1836 case ARM::t2LDRDi8:
1837 case ARM::t2LDRSHi8:
1838 case ARM::t2LDRi12:
1839 case ARM::t2LDRBi12:
1840 case ARM::t2LDRSHi12:
1841 return true;
1842 }
1843 };
1844
1845 if (!IsLoadOpcode(Load1->getMachineOpcode()) ||
1846 !IsLoadOpcode(Load2->getMachineOpcode()))
1847 return false;
1848
1849 // Check if base addresses and chain operands match.
1850 if (Load1->getOperand(0) != Load2->getOperand(0) ||
1851 Load1->getOperand(4) != Load2->getOperand(4))
1852 return false;
1853
1854 // Index should be Reg0.
1855 if (Load1->getOperand(3) != Load2->getOperand(3))
1856 return false;
1857
1858 // Determine the offsets.
1859 if (isa<ConstantSDNode>(Load1->getOperand(1)) &&
1860 isa<ConstantSDNode>(Load2->getOperand(1))) {
1861 Offset1 = cast<ConstantSDNode>(Load1->getOperand(1))->getSExtValue();
1862 Offset2 = cast<ConstantSDNode>(Load2->getOperand(1))->getSExtValue();
1863 return true;
1864 }
1865
1866 return false;
1867}
1868
1869/// shouldScheduleLoadsNear - This is a used by the pre-regalloc scheduler to
1870/// determine (in conjunction with areLoadsFromSameBasePtr) if two loads should
1871/// be scheduled together. On some targets if two loads are loading from
1872/// addresses in the same cache line, it's better if they are scheduled
1873/// together. This function takes two integers that represent the load offsets
1874/// from the common base address. It returns true if it decides it's desirable
1875/// to schedule the two loads together. "NumLoads" is the number of loads that
1876/// have already been scheduled after Load1.
1877///
1878/// FIXME: remove this in favor of the MachineInstr interface once pre-RA-sched
1879/// is permanently disabled.
1881 int64_t Offset1, int64_t Offset2,
1882 unsigned NumLoads) const {
1883 // Don't worry about Thumb: just ARM and Thumb2.
1884 if (Subtarget.isThumb1Only()) return false;
1885
1886 assert(Offset2 > Offset1);
1887
1888 if ((Offset2 - Offset1) / 8 > 64)
1889 return false;
1890
1891 // Check if the machine opcodes are different. If they are different
1892 // then we consider them to not be of the same base address,
1893 // EXCEPT in the case of Thumb2 byte loads where one is LDRBi8 and the other LDRBi12.
1894 // In this case, they are considered to be the same because they are different
1895 // encoding forms of the same basic instruction.
1896 if ((Load1->getMachineOpcode() != Load2->getMachineOpcode()) &&
1897 !((Load1->getMachineOpcode() == ARM::t2LDRBi8 &&
1898 Load2->getMachineOpcode() == ARM::t2LDRBi12) ||
1899 (Load1->getMachineOpcode() == ARM::t2LDRBi12 &&
1900 Load2->getMachineOpcode() == ARM::t2LDRBi8)))
1901 return false; // FIXME: overly conservative?
1902
1903 // Four loads in a row should be sufficient.
1904 if (NumLoads >= 3)
1905 return false;
1906
1907 return true;
1908}
1909
1911 const MachineBasicBlock *MBB,
1912 const MachineFunction &MF) const {
1913 // Debug info is never a scheduling boundary. It's necessary to be explicit
1914 // due to the special treatment of IT instructions below, otherwise a
1915 // dbg_value followed by an IT will result in the IT instruction being
1916 // considered a scheduling hazard, which is wrong. It should be the actual
1917 // instruction preceding the dbg_value instruction(s), just like it is
1918 // when debug info is not present.
1919 if (MI.isDebugInstr())
1920 return false;
1921
1922 // Terminators and labels can't be scheduled around.
1923 if (MI.isTerminator() || MI.isPosition())
1924 return true;
1925
1926 // INLINEASM_BR can jump to another block
1927 if (MI.getOpcode() == TargetOpcode::INLINEASM_BR)
1928 return true;
1929
1930 if (isSEHInstruction(MI))
1931 return true;
1932
1933 // Treat the start of the IT block as a scheduling boundary, but schedule
1934 // t2IT along with all instructions following it.
1935 // FIXME: This is a big hammer. But the alternative is to add all potential
1936 // true and anti dependencies to IT block instructions as implicit operands
1937 // to the t2IT instruction. The added compile time and complexity does not
1938 // seem worth it.
1940 // Make sure to skip any debug instructions
1941 while (++I != MBB->end() && I->isDebugInstr())
1942 ;
1943 if (I != MBB->end() && I->getOpcode() == ARM::t2IT)
1944 return true;
1945
1946 // Don't attempt to schedule around any instruction that defines
1947 // a stack-oriented pointer, as it's unlikely to be profitable. This
1948 // saves compile time, because it doesn't require every single
1949 // stack slot reference to depend on the instruction that does the
1950 // modification.
1951 // Calls don't actually change the stack pointer, even if they have imp-defs.
1952 // No ARM calling conventions change the stack pointer. (X86 calling
1953 // conventions sometimes do).
1954 if (!MI.isCall() && MI.definesRegister(ARM::SP, /*TRI=*/nullptr))
1955 return true;
1956
1957 return false;
1958}
1959
1962 unsigned NumCycles, unsigned ExtraPredCycles,
1963 BranchProbability Probability) const {
1964 if (!NumCycles)
1965 return false;
1966
1967 // If we are optimizing for size, see if the branch in the predecessor can be
1968 // lowered to cbn?z by the constant island lowering pass, and return false if
1969 // so. This results in a shorter instruction sequence.
1970 if (MBB.getParent()->getFunction().hasOptSize()) {
1971 MachineBasicBlock *Pred = *MBB.pred_begin();
1972 if (!Pred->empty()) {
1973 MachineInstr *LastMI = &*Pred->rbegin();
1974 if (LastMI->getOpcode() == ARM::t2Bcc) {
1976 MachineInstr *CmpMI = findCMPToFoldIntoCBZ(LastMI, TRI);
1977 if (CmpMI)
1978 return false;
1979 }
1980 }
1981 }
1982 return isProfitableToIfCvt(MBB, NumCycles, ExtraPredCycles,
1983 MBB, 0, 0, Probability);
1984}
1985
1988 unsigned TCycles, unsigned TExtra,
1989 MachineBasicBlock &FBB,
1990 unsigned FCycles, unsigned FExtra,
1991 BranchProbability Probability) const {
1992 if (!TCycles)
1993 return false;
1994
1995 // In thumb code we often end up trading one branch for a IT block, and
1996 // if we are cloning the instruction can increase code size. Prevent
1997 // blocks with multiple predecessors from being ifcvted to prevent this
1998 // cloning.
1999 if (Subtarget.isThumb2() && TBB.getParent()->getFunction().hasMinSize()) {
2000 if (TBB.pred_size() != 1 || FBB.pred_size() != 1)
2001 return false;
2002 }
2003
2004 // Attempt to estimate the relative costs of predication versus branching.
2005 // Here we scale up each component of UnpredCost to avoid precision issue when
2006 // scaling TCycles/FCycles by Probability.
2007 const unsigned ScalingUpFactor = 1024;
2008
2009 unsigned PredCost = (TCycles + FCycles + TExtra + FExtra) * ScalingUpFactor;
2010 unsigned UnpredCost;
2011 if (!Subtarget.hasBranchPredictor()) {
2012 // When we don't have a branch predictor it's always cheaper to not take a
2013 // branch than take it, so we have to take that into account.
2014 unsigned NotTakenBranchCost = 1;
2015 unsigned TakenBranchCost = Subtarget.getMispredictionPenalty();
2016 unsigned TUnpredCycles, FUnpredCycles;
2017 if (!FCycles) {
2018 // Triangle: TBB is the fallthrough
2019 TUnpredCycles = TCycles + NotTakenBranchCost;
2020 FUnpredCycles = TakenBranchCost;
2021 } else {
2022 // Diamond: TBB is the block that is branched to, FBB is the fallthrough
2023 TUnpredCycles = TCycles + TakenBranchCost;
2024 FUnpredCycles = FCycles + NotTakenBranchCost;
2025 // The branch at the end of FBB will disappear when it's predicated, so
2026 // discount it from PredCost.
2027 PredCost -= 1 * ScalingUpFactor;
2028 }
2029 // The total cost is the cost of each path scaled by their probabilities
2030 unsigned TUnpredCost = Probability.scale(TUnpredCycles * ScalingUpFactor);
2031 unsigned FUnpredCost = Probability.getCompl().scale(FUnpredCycles * ScalingUpFactor);
2032 UnpredCost = TUnpredCost + FUnpredCost;
2033 // When predicating assume that the first IT can be folded away but later
2034 // ones cost one cycle each
2035 if (Subtarget.isThumb2() && TCycles + FCycles > 4) {
2036 PredCost += ((TCycles + FCycles - 4) / 4) * ScalingUpFactor;
2037 }
2038 } else {
2039 unsigned TUnpredCost = Probability.scale(TCycles * ScalingUpFactor);
2040 unsigned FUnpredCost =
2041 Probability.getCompl().scale(FCycles * ScalingUpFactor);
2042 UnpredCost = TUnpredCost + FUnpredCost;
2043 UnpredCost += 1 * ScalingUpFactor; // The branch itself
2044 UnpredCost += Subtarget.getMispredictionPenalty() * ScalingUpFactor / 10;
2045 }
2046
2047 return PredCost <= UnpredCost;
2048}
2049
2050unsigned
2052 unsigned NumInsts) const {
2053 // Thumb2 needs a 2-byte IT instruction to predicate up to 4 instructions.
2054 // ARM has a condition code field in every predicable instruction, using it
2055 // doesn't change code size.
2056 if (!Subtarget.isThumb2())
2057 return 0;
2058
2059 // It's possible that the size of the IT is restricted to a single block.
2060 unsigned MaxInsts = Subtarget.restrictIT() ? 1 : 4;
2061 return divideCeil(NumInsts, MaxInsts) * 2;
2062}
2063
2064unsigned
2066 // If this branch is likely to be folded into the comparison to form a
2067 // CB(N)Z, then removing it won't reduce code size at all, because that will
2068 // just replace the CB(N)Z with a CMP.
2069 if (MI.getOpcode() == ARM::t2Bcc &&
2071 return 0;
2072
2073 unsigned Size = getInstSizeInBytes(MI);
2074
2075 // For Thumb2, all branches are 32-bit instructions during the if conversion
2076 // pass, but may be replaced with 16-bit instructions during size reduction.
2077 // Since the branches considered by if conversion tend to be forward branches
2078 // over small basic blocks, they are very likely to be in range for the
2079 // narrow instructions, so we assume the final code size will be half what it
2080 // currently is.
2081 if (Subtarget.isThumb2())
2082 Size /= 2;
2083
2084 return Size;
2085}
2086
2087bool
2089 MachineBasicBlock &FMBB) const {
2090 // Reduce false anti-dependencies to let the target's out-of-order execution
2091 // engine do its thing.
2092 return Subtarget.isProfitableToUnpredicate();
2093}
2094
2095/// getInstrPredicate - If instruction is predicated, returns its predicate
2096/// condition, otherwise returns AL. It also returns the condition code
2097/// register by reference.
2099 Register &PredReg) {
2100 int PIdx = MI.findFirstPredOperandIdx();
2101 if (PIdx == -1) {
2102 PredReg = 0;
2103 return ARMCC::AL;
2104 }
2105
2106 PredReg = MI.getOperand(PIdx+1).getReg();
2107 return (ARMCC::CondCodes)MI.getOperand(PIdx).getImm();
2108}
2109
2111 if (Opc == ARM::B)
2112 return ARM::Bcc;
2113 if (Opc == ARM::tB)
2114 return ARM::tBcc;
2115 if (Opc == ARM::t2B)
2116 return ARM::t2Bcc;
2117
2118 llvm_unreachable("Unknown unconditional branch opcode!");
2119}
2120
2122 bool NewMI,
2123 unsigned OpIdx1,
2124 unsigned OpIdx2) const {
2125 switch (MI.getOpcode()) {
2126 case ARM::MOVCCr:
2127 case ARM::t2MOVCCr: {
2128 // MOVCC can be commuted by inverting the condition.
2129 Register PredReg;
2130 ARMCC::CondCodes CC = getInstrPredicate(MI, PredReg);
2131 // MOVCC AL can't be inverted. Shouldn't happen.
2132 if (CC == ARMCC::AL || PredReg != ARM::CPSR)
2133 return nullptr;
2134 MachineInstr *CommutedMI =
2135 TargetInstrInfo::commuteInstructionImpl(MI, NewMI, OpIdx1, OpIdx2);
2136 if (!CommutedMI)
2137 return nullptr;
2138 // After swapping the MOVCC operands, also invert the condition.
2139 CommutedMI->getOperand(CommutedMI->findFirstPredOperandIdx())
2141 return CommutedMI;
2142 }
2143 }
2144 return TargetInstrInfo::commuteInstructionImpl(MI, NewMI, OpIdx1, OpIdx2);
2145}
2146
2147/// Identify instructions that can be folded into a MOVCC instruction, and
2148/// return the defining instruction.
2150ARMBaseInstrInfo::canFoldIntoMOVCC(Register Reg, const MachineRegisterInfo &MRI,
2151 const TargetInstrInfo *TII) const {
2152 if (!Reg.isVirtual())
2153 return nullptr;
2154 if (!MRI.hasOneNonDBGUse(Reg))
2155 return nullptr;
2156 MachineInstr *MI = MRI.getVRegDef(Reg);
2157 if (!MI)
2158 return nullptr;
2159 // Check if MI can be predicated and folded into the MOVCC.
2160 if (!isPredicable(*MI))
2161 return nullptr;
2162 // Check if MI has any non-dead defs or physreg uses. This also detects
2163 // predicated instructions which will be reading CPSR.
2164 for (const MachineOperand &MO : llvm::drop_begin(MI->operands(), 1)) {
2165 // Reject frame index operands, PEI can't handle the predicated pseudos.
2166 if (MO.isFI() || MO.isCPI() || MO.isJTI())
2167 return nullptr;
2168 if (!MO.isReg())
2169 continue;
2170 // MI can't have any tied operands, that would conflict with predication.
2171 if (MO.isTied())
2172 return nullptr;
2173 if (MO.getReg().isPhysical())
2174 return nullptr;
2175 if (MO.isDef() && !MO.isDead())
2176 return nullptr;
2177 }
2178 bool DontMoveAcrossStores = true;
2179 if (!MI->isSafeToMove(DontMoveAcrossStores))
2180 return nullptr;
2181 return MI;
2182}
2183
2187 bool PreferFalse) const {
2188 assert((MI.getOpcode() == ARM::MOVCCr || MI.getOpcode() == ARM::t2MOVCCr) &&
2189 "Unknown select instruction");
2190 MachineRegisterInfo &MRI = MI.getParent()->getParent()->getRegInfo();
2191 MachineInstr *DefMI = canFoldIntoMOVCC(MI.getOperand(2).getReg(), MRI, this);
2192 bool Invert = !DefMI;
2193 if (!DefMI)
2194 DefMI = canFoldIntoMOVCC(MI.getOperand(1).getReg(), MRI, this);
2195 if (!DefMI)
2196 return nullptr;
2197
2198 // Find new register class to use.
2199 MachineOperand FalseReg = MI.getOperand(Invert ? 2 : 1);
2200 MachineOperand TrueReg = MI.getOperand(Invert ? 1 : 2);
2201 Register DestReg = MI.getOperand(0).getReg();
2202 const TargetRegisterClass *FalseClass = MRI.getRegClass(FalseReg.getReg());
2203 const TargetRegisterClass *TrueClass = MRI.getRegClass(TrueReg.getReg());
2204 if (!MRI.constrainRegClass(DestReg, FalseClass))
2205 return nullptr;
2206 if (!MRI.constrainRegClass(DestReg, TrueClass))
2207 return nullptr;
2208
2209 // Create a new predicated version of DefMI.
2210 // Rfalse is the first use.
2211 MachineInstrBuilder NewMI =
2212 BuildMI(*MI.getParent(), MI, MI.getDebugLoc(), DefMI->getDesc(), DestReg);
2213
2214 // Copy all the DefMI operands, excluding its (null) predicate.
2215 const MCInstrDesc &DefDesc = DefMI->getDesc();
2216 for (unsigned i = 1, e = DefDesc.getNumOperands();
2217 i != e && !DefDesc.operands()[i].isPredicate(); ++i)
2218 NewMI.add(DefMI->getOperand(i));
2219
2220 unsigned CondCode = MI.getOperand(3).getImm();
2221 if (Invert)
2223 else
2224 NewMI.addImm(CondCode);
2225 NewMI.add(MI.getOperand(4));
2226
2227 // DefMI is not the -S version that sets CPSR, so add an optional %noreg.
2228 if (NewMI->hasOptionalDef())
2229 NewMI.add(condCodeOp());
2230
2231 // The output register value when the predicate is false is an implicit
2232 // register operand tied to the first def.
2233 // The tie makes the register allocator ensure the FalseReg is allocated the
2234 // same register as operand 0.
2235 FalseReg.setImplicit();
2236 NewMI.add(FalseReg);
2237 NewMI->tieOperands(0, NewMI->getNumOperands() - 1);
2238
2239 // Update SeenMIs set: register newly created MI and erase removed DefMI.
2240 SeenMIs.insert(NewMI);
2241 SeenMIs.erase(DefMI);
2242
2243 // If MI is inside a loop, and DefMI is outside the loop, then kill flags on
2244 // DefMI would be invalid when transferred inside the loop. Checking for a
2245 // loop is expensive, but at least remove kill flags if they are in different
2246 // BBs.
2247 if (DefMI->getParent() != MI.getParent())
2248 NewMI->clearKillInfo();
2249
2250 // The caller will erase MI, but not DefMI.
2251 DefMI->eraseFromParent();
2252 return NewMI;
2253}
2254
2255/// Map pseudo instructions that imply an 'S' bit onto real opcodes. Whether the
2256/// instruction is encoded with an 'S' bit is determined by the optional CPSR
2257/// def operand.
2258///
2259/// This will go away once we can teach tblgen how to set the optional CPSR def
2260/// operand itself.
2262 uint16_t PseudoOpc;
2263 uint16_t MachineOpc;
2264};
2265
2267 {ARM::ADDSri, ARM::ADDri},
2268 {ARM::ADDSrr, ARM::ADDrr},
2269 {ARM::ADDSrsi, ARM::ADDrsi},
2270 {ARM::ADDSrsr, ARM::ADDrsr},
2271
2272 {ARM::SUBSri, ARM::SUBri},
2273 {ARM::SUBSrr, ARM::SUBrr},
2274 {ARM::SUBSrsi, ARM::SUBrsi},
2275 {ARM::SUBSrsr, ARM::SUBrsr},
2276
2277 {ARM::RSBSri, ARM::RSBri},
2278 {ARM::RSBSrsi, ARM::RSBrsi},
2279 {ARM::RSBSrsr, ARM::RSBrsr},
2280
2281 {ARM::tADDSi3, ARM::tADDi3},
2282 {ARM::tADDSi8, ARM::tADDi8},
2283 {ARM::tADDSrr, ARM::tADDrr},
2284 {ARM::tADCS, ARM::tADC},
2285
2286 {ARM::tSUBSi3, ARM::tSUBi3},
2287 {ARM::tSUBSi8, ARM::tSUBi8},
2288 {ARM::tSUBSrr, ARM::tSUBrr},
2289 {ARM::tSBCS, ARM::tSBC},
2290 {ARM::tRSBS, ARM::tRSB},
2291 {ARM::tLSLSri, ARM::tLSLri},
2292
2293 {ARM::t2ADDSri, ARM::t2ADDri},
2294 {ARM::t2ADDSrr, ARM::t2ADDrr},
2295 {ARM::t2ADDSrs, ARM::t2ADDrs},
2296
2297 {ARM::t2SUBSri, ARM::t2SUBri},
2298 {ARM::t2SUBSrr, ARM::t2SUBrr},
2299 {ARM::t2SUBSrs, ARM::t2SUBrs},
2300
2301 {ARM::t2RSBSri, ARM::t2RSBri},
2302 {ARM::t2RSBSrs, ARM::t2RSBrs},
2303};
2304
2305unsigned llvm::convertAddSubFlagsOpcode(unsigned OldOpc) {
2306 for (const auto &Entry : AddSubFlagsOpcodeMap)
2307 if (OldOpc == Entry.PseudoOpc)
2308 return Entry.MachineOpc;
2309 return 0;
2310}
2311
2314 const DebugLoc &dl, Register DestReg,
2315 Register BaseReg, int NumBytes,
2316 ARMCC::CondCodes Pred, Register PredReg,
2317 const ARMBaseInstrInfo &TII,
2318 unsigned MIFlags) {
2319 if (NumBytes == 0 && DestReg != BaseReg) {
2320 BuildMI(MBB, MBBI, dl, TII.get(ARM::MOVr), DestReg)
2321 .addReg(BaseReg, RegState::Kill)
2322 .add(predOps(Pred, PredReg))
2323 .add(condCodeOp())
2324 .setMIFlags(MIFlags);
2325 return;
2326 }
2327
2328 bool isSub = NumBytes < 0;
2329 if (isSub) NumBytes = -NumBytes;
2330
2331 while (NumBytes) {
2332 unsigned RotAmt = ARM_AM::getSOImmValRotate(NumBytes);
2333 unsigned ThisVal = NumBytes & llvm::rotr<uint32_t>(0xFF, RotAmt);
2334 assert(ThisVal && "Didn't extract field correctly");
2335
2336 // We will handle these bits from offset, clear them.
2337 NumBytes &= ~ThisVal;
2338
2339 assert(ARM_AM::getSOImmVal(ThisVal) != -1 && "Bit extraction didn't work?");
2340
2341 // Build the new ADD / SUB.
2342 unsigned Opc = isSub ? ARM::SUBri : ARM::ADDri;
2343 BuildMI(MBB, MBBI, dl, TII.get(Opc), DestReg)
2344 .addReg(BaseReg, RegState::Kill)
2345 .addImm(ThisVal)
2346 .add(predOps(Pred, PredReg))
2347 .add(condCodeOp())
2348 .setMIFlags(MIFlags);
2349 BaseReg = DestReg;
2350 }
2351}
2352
2355 unsigned NumBytes) {
2356 // This optimisation potentially adds lots of load and store
2357 // micro-operations, it's only really a great benefit to code-size.
2358 if (!Subtarget.hasMinSize())
2359 return false;
2360
2361 // If only one register is pushed/popped, LLVM can use an LDR/STR
2362 // instead. We can't modify those so make sure we're dealing with an
2363 // instruction we understand.
2364 bool IsPop = isPopOpcode(MI->getOpcode());
2365 bool IsPush = isPushOpcode(MI->getOpcode());
2366 if (!IsPush && !IsPop)
2367 return false;
2368
2369 bool IsVFPPushPop = MI->getOpcode() == ARM::VSTMDDB_UPD ||
2370 MI->getOpcode() == ARM::VLDMDIA_UPD;
2371 bool IsT1PushPop = MI->getOpcode() == ARM::tPUSH ||
2372 MI->getOpcode() == ARM::tPOP ||
2373 MI->getOpcode() == ARM::tPOP_RET;
2374
2375 assert((IsT1PushPop || (MI->getOperand(0).getReg() == ARM::SP &&
2376 MI->getOperand(1).getReg() == ARM::SP)) &&
2377 "trying to fold sp update into non-sp-updating push/pop");
2378
2379 // The VFP push & pop act on D-registers, so we can only fold an adjustment
2380 // by a multiple of 8 bytes in correctly. Similarly rN is 4-bytes. Don't try
2381 // if this is violated.
2382 if (NumBytes % (IsVFPPushPop ? 8 : 4) != 0)
2383 return false;
2384
2385 // ARM and Thumb2 push/pop insts have explicit "sp, sp" operands (+
2386 // pred) so the list starts at 4. Thumb1 starts after the predicate.
2387 int RegListIdx = IsT1PushPop ? 2 : 4;
2388
2389 // Calculate the space we'll need in terms of registers.
2390 unsigned RegsNeeded;
2391 const TargetRegisterClass *RegClass;
2392 if (IsVFPPushPop) {
2393 RegsNeeded = NumBytes / 8;
2394 RegClass = &ARM::DPRRegClass;
2395 } else {
2396 RegsNeeded = NumBytes / 4;
2397 RegClass = &ARM::GPRRegClass;
2398 }
2399
2400 // We're going to have to strip all list operands off before
2401 // re-adding them since the order matters, so save the existing ones
2402 // for later.
2404
2405 // We're also going to need the first register transferred by this
2406 // instruction, which won't necessarily be the first register in the list.
2407 unsigned FirstRegEnc = -1;
2408
2410 for (int i = MI->getNumOperands() - 1; i >= RegListIdx; --i) {
2411 MachineOperand &MO = MI->getOperand(i);
2412 RegList.push_back(MO);
2413
2414 if (MO.isReg() && !MO.isImplicit() &&
2415 TRI->getEncodingValue(MO.getReg()) < FirstRegEnc)
2416 FirstRegEnc = TRI->getEncodingValue(MO.getReg());
2417 }
2418
2419 const MCPhysReg *CSRegs = TRI->getCalleeSavedRegs(&MF);
2420
2421 // Now try to find enough space in the reglist to allocate NumBytes.
2422 for (int CurRegEnc = FirstRegEnc - 1; CurRegEnc >= 0 && RegsNeeded;
2423 --CurRegEnc) {
2424 MCRegister CurReg = RegClass->getRegister(CurRegEnc);
2425 if (IsT1PushPop && CurRegEnc > TRI->getEncodingValue(ARM::R7))
2426 continue;
2427 if (!IsPop) {
2428 // Pushing any register is completely harmless, mark the register involved
2429 // as undef since we don't care about its value and must not restore it
2430 // during stack unwinding.
2431 RegList.push_back(MachineOperand::CreateReg(CurReg, false, false,
2432 false, false, true));
2433 --RegsNeeded;
2434 continue;
2435 }
2436
2437 // However, we can only pop an extra register if it's not live. For
2438 // registers live within the function we might clobber a return value
2439 // register; the other way a register can be live here is if it's
2440 // callee-saved.
2441 if (isCalleeSavedRegister(CurReg, CSRegs) ||
2442 MI->getParent()->computeRegisterLiveness(TRI, CurReg, MI) !=
2444 // VFP pops don't allow holes in the register list, so any skip is fatal
2445 // for our transformation. GPR pops do, so we should just keep looking.
2446 if (IsVFPPushPop)
2447 return false;
2448 else
2449 continue;
2450 }
2451
2452 // Mark the unimportant registers as <def,dead> in the POP.
2453 RegList.push_back(MachineOperand::CreateReg(CurReg, true, false, false,
2454 true));
2455 --RegsNeeded;
2456 }
2457
2458 if (RegsNeeded > 0)
2459 return false;
2460
2461 // Finally we know we can profitably perform the optimisation so go
2462 // ahead: strip all existing registers off and add them back again
2463 // in the right order.
2464 for (int i = MI->getNumOperands() - 1; i >= RegListIdx; --i)
2465 MI->removeOperand(i);
2466
2467 // Add the complete list back in.
2468 MachineInstrBuilder MIB(MF, &*MI);
2469 for (const MachineOperand &MO : llvm::reverse(RegList))
2470 MIB.add(MO);
2471
2472 return true;
2473}
2474
2475bool llvm::rewriteARMFrameIndex(MachineInstr &MI, unsigned FrameRegIdx,
2476 Register FrameReg, int &Offset,
2477 const ARMBaseInstrInfo &TII) {
2478 unsigned Opcode = MI.getOpcode();
2479 const MCInstrDesc &Desc = MI.getDesc();
2480 unsigned AddrMode = (Desc.TSFlags & ARMII::AddrModeMask);
2481 bool isSub = false;
2482
2483 // Memory operands in inline assembly always use AddrMode2.
2484 if (Opcode == ARM::INLINEASM || Opcode == ARM::INLINEASM_BR)
2486
2487 if (Opcode == ARM::ADDri) {
2488 Offset += MI.getOperand(FrameRegIdx+1).getImm();
2489 if (Offset == 0) {
2490 // Turn it into a move.
2491 MI.setDesc(TII.get(ARM::MOVr));
2492 MI.getOperand(FrameRegIdx).ChangeToRegister(FrameReg, false);
2493 MI.removeOperand(FrameRegIdx+1);
2494 Offset = 0;
2495 return true;
2496 } else if (Offset < 0) {
2497 Offset = -Offset;
2498 isSub = true;
2499 MI.setDesc(TII.get(ARM::SUBri));
2500 }
2501
2502 // Common case: small offset, fits into instruction.
2503 if (ARM_AM::getSOImmVal(Offset) != -1) {
2504 // Replace the FrameIndex with sp / fp
2505 MI.getOperand(FrameRegIdx).ChangeToRegister(FrameReg, false);
2506 MI.getOperand(FrameRegIdx+1).ChangeToImmediate(Offset);
2507 Offset = 0;
2508 return true;
2509 }
2510
2511 // Otherwise, pull as much of the immediate into this ADDri/SUBri
2512 // as possible.
2513 unsigned RotAmt = ARM_AM::getSOImmValRotate(Offset);
2514 unsigned ThisImmVal = Offset & llvm::rotr<uint32_t>(0xFF, RotAmt);
2515
2516 // We will handle these bits from offset, clear them.
2517 Offset &= ~ThisImmVal;
2518
2519 // Get the properly encoded SOImmVal field.
2520 assert(ARM_AM::getSOImmVal(ThisImmVal) != -1 &&
2521 "Bit extraction didn't work?");
2522 MI.getOperand(FrameRegIdx+1).ChangeToImmediate(ThisImmVal);
2523 } else {
2524 unsigned ImmIdx = 0;
2525 int InstrOffs = 0;
2526 unsigned NumBits = 0;
2527 unsigned Scale = 1;
2528 switch (AddrMode) {
2530 ImmIdx = FrameRegIdx + 1;
2531 InstrOffs = MI.getOperand(ImmIdx).getImm();
2532 NumBits = 12;
2533 break;
2534 case ARMII::AddrMode2:
2535 ImmIdx = FrameRegIdx+2;
2536 InstrOffs = ARM_AM::getAM2Offset(MI.getOperand(ImmIdx).getImm());
2537 if (ARM_AM::getAM2Op(MI.getOperand(ImmIdx).getImm()) == ARM_AM::sub)
2538 InstrOffs *= -1;
2539 NumBits = 12;
2540 break;
2541 case ARMII::AddrMode3:
2542 ImmIdx = FrameRegIdx+2;
2543 InstrOffs = ARM_AM::getAM3Offset(MI.getOperand(ImmIdx).getImm());
2544 if (ARM_AM::getAM3Op(MI.getOperand(ImmIdx).getImm()) == ARM_AM::sub)
2545 InstrOffs *= -1;
2546 NumBits = 8;
2547 break;
2548 case ARMII::AddrMode4:
2549 case ARMII::AddrMode6:
2550 // Can't fold any offset even if it's zero.
2551 return false;
2552 case ARMII::AddrMode5:
2553 ImmIdx = FrameRegIdx+1;
2554 InstrOffs = ARM_AM::getAM5Offset(MI.getOperand(ImmIdx).getImm());
2555 if (ARM_AM::getAM5Op(MI.getOperand(ImmIdx).getImm()) == ARM_AM::sub)
2556 InstrOffs *= -1;
2557 NumBits = 8;
2558 Scale = 4;
2559 break;
2561 ImmIdx = FrameRegIdx+1;
2562 InstrOffs = ARM_AM::getAM5Offset(MI.getOperand(ImmIdx).getImm());
2563 if (ARM_AM::getAM5Op(MI.getOperand(ImmIdx).getImm()) == ARM_AM::sub)
2564 InstrOffs *= -1;
2565 NumBits = 8;
2566 Scale = 2;
2567 break;
2571 ImmIdx = FrameRegIdx+1;
2572 InstrOffs = MI.getOperand(ImmIdx).getImm();
2573 NumBits = 7;
2574 Scale = (AddrMode == ARMII::AddrModeT2_i7s2 ? 2 :
2575 AddrMode == ARMII::AddrModeT2_i7s4 ? 4 : 1);
2576 break;
2577 default:
2578 llvm_unreachable("Unsupported addressing mode!");
2579 }
2580
2581 Offset += InstrOffs * Scale;
2582 assert((Offset & (Scale-1)) == 0 && "Can't encode this offset!");
2583 if (Offset < 0) {
2584 Offset = -Offset;
2585 isSub = true;
2586 }
2587
2588 // Attempt to fold address comp. if opcode has offset bits
2589 if (NumBits > 0) {
2590 // Common case: small offset, fits into instruction.
2591 MachineOperand &ImmOp = MI.getOperand(ImmIdx);
2592 int ImmedOffset = Offset / Scale;
2593 unsigned Mask = (1 << NumBits) - 1;
2594 if ((unsigned)Offset <= Mask * Scale) {
2595 // Replace the FrameIndex with sp
2596 MI.getOperand(FrameRegIdx).ChangeToRegister(FrameReg, false);
2597 // FIXME: When addrmode2 goes away, this will simplify (like the
2598 // T2 version), as the LDR.i12 versions don't need the encoding
2599 // tricks for the offset value.
2600 if (isSub) {
2602 ImmedOffset = -ImmedOffset;
2603 else
2604 ImmedOffset |= 1 << NumBits;
2605 }
2606 ImmOp.ChangeToImmediate(ImmedOffset);
2607 Offset = 0;
2608 return true;
2609 }
2610
2611 // Otherwise, it didn't fit. Pull in what we can to simplify the immed.
2612 ImmedOffset = ImmedOffset & Mask;
2613 if (isSub) {
2615 ImmedOffset = -ImmedOffset;
2616 else
2617 ImmedOffset |= 1 << NumBits;
2618 }
2619 ImmOp.ChangeToImmediate(ImmedOffset);
2620 Offset &= ~(Mask*Scale);
2621 }
2622 }
2623
2624 Offset = (isSub) ? -Offset : Offset;
2625 return Offset == 0;
2626}
2627
2628/// analyzeCompare - For a comparison instruction, return the source registers
2629/// in SrcReg and SrcReg2 if having two register operands, and the value it
2630/// compares against in CmpValue. Return true if the comparison instruction
2631/// can be analyzed.
2633 Register &SrcReg2, int64_t &CmpMask,
2634 int64_t &CmpValue) const {
2635 switch (MI.getOpcode()) {
2636 default: break;
2637 case ARM::CMPri:
2638 case ARM::t2CMPri:
2639 case ARM::tCMPi8:
2640 SrcReg = MI.getOperand(0).getReg();
2641 SrcReg2 = 0;
2642 CmpMask = ~0;
2643 CmpValue = MI.getOperand(1).getImm();
2644 return true;
2645 case ARM::CMPrr:
2646 case ARM::t2CMPrr:
2647 case ARM::tCMPr:
2648 SrcReg = MI.getOperand(0).getReg();
2649 SrcReg2 = MI.getOperand(1).getReg();
2650 CmpMask = ~0;
2651 CmpValue = 0;
2652 return true;
2653 case ARM::TSTri:
2654 case ARM::t2TSTri:
2655 SrcReg = MI.getOperand(0).getReg();
2656 SrcReg2 = 0;
2657 CmpMask = MI.getOperand(1).getImm();
2658 CmpValue = 0;
2659 return true;
2660 }
2661
2662 return false;
2663}
2664
2665/// isSuitableForMask - Identify a suitable 'and' instruction that
2666/// operates on the given source register and applies the same mask
2667/// as a 'tst' instruction. Provide a limited look-through for copies.
2668/// When successful, MI will hold the found instruction.
2670 int CmpMask, bool CommonUse) {
2671 switch (MI->getOpcode()) {
2672 case ARM::ANDri:
2673 case ARM::t2ANDri:
2674 if (CmpMask != MI->getOperand(2).getImm())
2675 return false;
2676 if (SrcReg == MI->getOperand(CommonUse ? 1 : 0).getReg())
2677 return true;
2678 break;
2679 }
2680
2681 return false;
2682}
2683
2684/// getCmpToAddCondition - assume the flags are set by CMP(a,b), return
2685/// the condition code if we modify the instructions such that flags are
2686/// set by ADD(a,b,X).
2688 switch (CC) {
2689 default: return ARMCC::AL;
2690 case ARMCC::HS: return ARMCC::LO;
2691 case ARMCC::LO: return ARMCC::HS;
2692 case ARMCC::VS: return ARMCC::VS;
2693 case ARMCC::VC: return ARMCC::VC;
2694 }
2695}
2696
2697/// isRedundantFlagInstr - check whether the first instruction, whose only
2698/// purpose is to update flags, can be made redundant.
2699/// CMPrr can be made redundant by SUBrr if the operands are the same.
2700/// CMPri can be made redundant by SUBri if the operands are the same.
2701/// CMPrr(r0, r1) can be made redundant by ADDr[ri](r0, r1, X).
2702/// This function can be extended later on.
2703inline static bool isRedundantFlagInstr(const MachineInstr *CmpI,
2704 Register SrcReg, Register SrcReg2,
2705 int64_t ImmValue,
2706 const MachineInstr *OI,
2707 bool &IsThumb1) {
2708 if ((CmpI->getOpcode() == ARM::CMPrr || CmpI->getOpcode() == ARM::t2CMPrr) &&
2709 (OI->getOpcode() == ARM::SUBrr || OI->getOpcode() == ARM::t2SUBrr) &&
2710 ((OI->getOperand(1).getReg() == SrcReg &&
2711 OI->getOperand(2).getReg() == SrcReg2) ||
2712 (OI->getOperand(1).getReg() == SrcReg2 &&
2713 OI->getOperand(2).getReg() == SrcReg))) {
2714 IsThumb1 = false;
2715 return true;
2716 }
2717
2718 if (CmpI->getOpcode() == ARM::tCMPr && OI->getOpcode() == ARM::tSUBrr &&
2719 ((OI->getOperand(2).getReg() == SrcReg &&
2720 OI->getOperand(3).getReg() == SrcReg2) ||
2721 (OI->getOperand(2).getReg() == SrcReg2 &&
2722 OI->getOperand(3).getReg() == SrcReg))) {
2723 IsThumb1 = true;
2724 return true;
2725 }
2726
2727 if ((CmpI->getOpcode() == ARM::CMPri || CmpI->getOpcode() == ARM::t2CMPri) &&
2728 (OI->getOpcode() == ARM::SUBri || OI->getOpcode() == ARM::t2SUBri) &&
2729 OI->getOperand(1).getReg() == SrcReg &&
2730 OI->getOperand(2).getImm() == ImmValue) {
2731 IsThumb1 = false;
2732 return true;
2733 }
2734
2735 if (CmpI->getOpcode() == ARM::tCMPi8 &&
2736 (OI->getOpcode() == ARM::tSUBi8 || OI->getOpcode() == ARM::tSUBi3) &&
2737 OI->getOperand(2).getReg() == SrcReg &&
2738 OI->getOperand(3).getImm() == ImmValue) {
2739 IsThumb1 = true;
2740 return true;
2741 }
2742
2743 if ((CmpI->getOpcode() == ARM::CMPrr || CmpI->getOpcode() == ARM::t2CMPrr) &&
2744 (OI->getOpcode() == ARM::ADDrr || OI->getOpcode() == ARM::t2ADDrr ||
2745 OI->getOpcode() == ARM::ADDri || OI->getOpcode() == ARM::t2ADDri) &&
2746 OI->getOperand(0).isReg() && OI->getOperand(1).isReg() &&
2747 OI->getOperand(0).getReg() == SrcReg &&
2748 OI->getOperand(1).getReg() == SrcReg2) {
2749 IsThumb1 = false;
2750 return true;
2751 }
2752
2753 if (CmpI->getOpcode() == ARM::tCMPr &&
2754 (OI->getOpcode() == ARM::tADDi3 || OI->getOpcode() == ARM::tADDi8 ||
2755 OI->getOpcode() == ARM::tADDrr) &&
2756 OI->getOperand(0).getReg() == SrcReg &&
2757 OI->getOperand(2).getReg() == SrcReg2) {
2758 IsThumb1 = true;
2759 return true;
2760 }
2761
2762 return false;
2763}
2764
2765static bool isOptimizeCompareCandidate(MachineInstr *MI, bool &IsThumb1) {
2766 switch (MI->getOpcode()) {
2767 default: return false;
2768 case ARM::tLSLri:
2769 case ARM::tLSRri:
2770 case ARM::tLSLrr:
2771 case ARM::tLSRrr:
2772 case ARM::tSUBrr:
2773 case ARM::tADDrr:
2774 case ARM::tADDi3:
2775 case ARM::tADDi8:
2776 case ARM::tSUBi3:
2777 case ARM::tSUBi8:
2778 case ARM::tMUL:
2779 case ARM::tADC:
2780 case ARM::tSBC:
2781 case ARM::tRSB:
2782 case ARM::tAND:
2783 case ARM::tORR:
2784 case ARM::tEOR:
2785 case ARM::tBIC:
2786 case ARM::tMVN:
2787 case ARM::tASRri:
2788 case ARM::tASRrr:
2789 case ARM::tROR:
2790 IsThumb1 = true;
2791 [[fallthrough]];
2792 case ARM::RSBrr:
2793 case ARM::RSBri:
2794 case ARM::RSCrr:
2795 case ARM::RSCri:
2796 case ARM::ADDrr:
2797 case ARM::ADDri:
2798 case ARM::ADCrr:
2799 case ARM::ADCri:
2800 case ARM::SUBrr:
2801 case ARM::SUBri:
2802 case ARM::SBCrr:
2803 case ARM::SBCri:
2804 case ARM::t2RSBri:
2805 case ARM::t2ADDrr:
2806 case ARM::t2ADDri:
2807 case ARM::t2ADCrr:
2808 case ARM::t2ADCri:
2809 case ARM::t2SUBrr:
2810 case ARM::t2SUBri:
2811 case ARM::t2SBCrr:
2812 case ARM::t2SBCri:
2813 case ARM::ANDrr:
2814 case ARM::ANDri:
2815 case ARM::ANDrsr:
2816 case ARM::ANDrsi:
2817 case ARM::t2ANDrr:
2818 case ARM::t2ANDri:
2819 case ARM::t2ANDrs:
2820 case ARM::ORRrr:
2821 case ARM::ORRri:
2822 case ARM::ORRrsr:
2823 case ARM::ORRrsi:
2824 case ARM::t2ORRrr:
2825 case ARM::t2ORRri:
2826 case ARM::t2ORRrs:
2827 case ARM::EORrr:
2828 case ARM::EORri:
2829 case ARM::EORrsr:
2830 case ARM::EORrsi:
2831 case ARM::t2EORrr:
2832 case ARM::t2EORri:
2833 case ARM::t2EORrs:
2834 case ARM::BICri:
2835 case ARM::BICrr:
2836 case ARM::BICrsi:
2837 case ARM::BICrsr:
2838 case ARM::t2BICri:
2839 case ARM::t2BICrr:
2840 case ARM::t2BICrs:
2841 case ARM::t2LSRri:
2842 case ARM::t2LSRrr:
2843 case ARM::t2LSLri:
2844 case ARM::t2LSLrr:
2845 case ARM::MOVsr:
2846 case ARM::MOVsi:
2847 return true;
2848 }
2849}
2850
2851/// optimizeCompareInstr - Convert the instruction supplying the argument to the
2852/// comparison into one that sets the zero bit in the flags register;
2853/// Remove a redundant Compare instruction if an earlier instruction can set the
2854/// flags in the same way as Compare.
2855/// E.g. SUBrr(r1,r2) and CMPrr(r1,r2). We also handle the case where two
2856/// operands are swapped: SUBrr(r1,r2) and CMPrr(r2,r1), by updating the
2857/// condition code of instructions which use the flags.
2859 MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask,
2860 int64_t CmpValue, const MachineRegisterInfo *MRI) const {
2861 // Get the unique definition of SrcReg.
2862 MachineInstr *MI = MRI->getUniqueVRegDef(SrcReg);
2863 if (!MI) return false;
2864
2865 // Masked compares sometimes use the same register as the corresponding 'and'.
2866 if (CmpMask != ~0) {
2867 if (!isSuitableForMask(MI, SrcReg, CmpMask, false) || isPredicated(*MI)) {
2868 MI = nullptr;
2870 UI = MRI->use_instr_begin(SrcReg), UE = MRI->use_instr_end();
2871 UI != UE; ++UI) {
2872 if (UI->getParent() != CmpInstr.getParent())
2873 continue;
2874 MachineInstr *PotentialAND = &*UI;
2875 if (!isSuitableForMask(PotentialAND, SrcReg, CmpMask, true) ||
2876 isPredicated(*PotentialAND))
2877 continue;
2878 MI = PotentialAND;
2879 break;
2880 }
2881 if (!MI) return false;
2882 }
2883 }
2884
2885 // Get ready to iterate backward from CmpInstr.
2886 MachineBasicBlock::iterator I = CmpInstr, E = MI,
2887 B = CmpInstr.getParent()->begin();
2888
2889 // Early exit if CmpInstr is at the beginning of the BB.
2890 if (I == B) return false;
2891
2892 // There are two possible candidates which can be changed to set CPSR:
2893 // One is MI, the other is a SUB or ADD instruction.
2894 // For CMPrr(r1,r2), we are looking for SUB(r1,r2), SUB(r2,r1), or
2895 // ADDr[ri](r1, r2, X).
2896 // For CMPri(r1, CmpValue), we are looking for SUBri(r1, CmpValue).
2897 MachineInstr *SubAdd = nullptr;
2898 if (SrcReg2 != 0)
2899 // MI is not a candidate for CMPrr.
2900 MI = nullptr;
2901 else if (MI->getParent() != CmpInstr.getParent() || CmpValue != 0) {
2902 // Conservatively refuse to convert an instruction which isn't in the same
2903 // BB as the comparison.
2904 // For CMPri w/ CmpValue != 0, a SubAdd may still be a candidate.
2905 // Thus we cannot return here.
2906 if (CmpInstr.getOpcode() == ARM::CMPri ||
2907 CmpInstr.getOpcode() == ARM::t2CMPri ||
2908 CmpInstr.getOpcode() == ARM::tCMPi8)
2909 MI = nullptr;
2910 else
2911 return false;
2912 }
2913
2914 bool IsThumb1 = false;
2915 if (MI && !isOptimizeCompareCandidate(MI, IsThumb1))
2916 return false;
2917
2918 // We also want to do this peephole for cases like this: if (a*b == 0),
2919 // and optimise away the CMP instruction from the generated code sequence:
2920 // MULS, MOVS, MOVS, CMP. Here the MOVS instructions load the boolean values
2921 // resulting from the select instruction, but these MOVS instructions for
2922 // Thumb1 (V6M) are flag setting and are thus preventing this optimisation.
2923 // However, if we only have MOVS instructions in between the CMP and the
2924 // other instruction (the MULS in this example), then the CPSR is dead so we
2925 // can safely reorder the sequence into: MOVS, MOVS, MULS, CMP. We do this
2926 // reordering and then continue the analysis hoping we can eliminate the
2927 // CMP. This peephole works on the vregs, so is still in SSA form. As a
2928 // consequence, the movs won't redefine/kill the MUL operands which would
2929 // make this reordering illegal.
2931 if (MI && IsThumb1) {
2932 --I;
2933 if (I != E && !MI->readsRegister(ARM::CPSR, TRI)) {
2934 bool CanReorder = true;
2935 for (; I != E; --I) {
2936 if (I->getOpcode() != ARM::tMOVi8) {
2937 CanReorder = false;
2938 break;
2939 }
2940 }
2941 if (CanReorder) {
2942 MI = MI->removeFromParent();
2943 E = CmpInstr;
2944 CmpInstr.getParent()->insert(E, MI);
2945 }
2946 }
2947 I = CmpInstr;
2948 E = MI;
2949 }
2950
2951 // Check that CPSR isn't set between the comparison instruction and the one we
2952 // want to change. At the same time, search for SubAdd.
2953 bool SubAddIsThumb1 = false;
2954 do {
2955 const MachineInstr &Instr = *--I;
2956
2957 // Check whether CmpInstr can be made redundant by the current instruction.
2958 if (isRedundantFlagInstr(&CmpInstr, SrcReg, SrcReg2, CmpValue, &Instr,
2959 SubAddIsThumb1)) {
2960 SubAdd = &*I;
2961 break;
2962 }
2963
2964 // Allow E (which was initially MI) to be SubAdd but do not search before E.
2965 if (I == E)
2966 break;
2967
2968 if (Instr.modifiesRegister(ARM::CPSR, TRI) ||
2969 Instr.readsRegister(ARM::CPSR, TRI))
2970 // This instruction modifies or uses CPSR after the one we want to
2971 // change. We can't do this transformation.
2972 return false;
2973
2974 if (I == B) {
2975 // In some cases, we scan the use-list of an instruction for an AND;
2976 // that AND is in the same BB, but may not be scheduled before the
2977 // corresponding TST. In that case, bail out.
2978 //
2979 // FIXME: We could try to reschedule the AND.
2980 return false;
2981 }
2982 } while (true);
2983
2984 // Return false if no candidates exist.
2985 if (!MI && !SubAdd)
2986 return false;
2987
2988 // If we found a SubAdd, use it as it will be closer to the CMP
2989 if (SubAdd) {
2990 MI = SubAdd;
2991 IsThumb1 = SubAddIsThumb1;
2992 }
2993
2994 // We can't use a predicated instruction - it doesn't always write the flags.
2995 if (isPredicated(*MI))
2996 return false;
2997
2998 // Scan forward for the use of CPSR
2999 // When checking against MI: if it's a conditional code that requires
3000 // checking of the V bit or C bit, then this is not safe to do.
3001 // It is safe to remove CmpInstr if CPSR is redefined or killed.
3002 // If we are done with the basic block, we need to check whether CPSR is
3003 // live-out.
3005 OperandsToUpdate;
3006 bool isSafe = false;
3007 I = CmpInstr;
3008 E = CmpInstr.getParent()->end();
3009 while (!isSafe && ++I != E) {
3010 const MachineInstr &Instr = *I;
3011 for (unsigned IO = 0, EO = Instr.getNumOperands();
3012 !isSafe && IO != EO; ++IO) {
3013 const MachineOperand &MO = Instr.getOperand(IO);
3014 if (MO.isRegMask() && MO.clobbersPhysReg(ARM::CPSR)) {
3015 isSafe = true;
3016 break;
3017 }
3018 if (!MO.isReg() || MO.getReg() != ARM::CPSR)
3019 continue;
3020 if (MO.isDef()) {
3021 isSafe = true;
3022 break;
3023 }
3024 // Condition code is after the operand before CPSR except for VSELs.
3026 bool IsInstrVSel = true;
3027 switch (Instr.getOpcode()) {
3028 default:
3029 IsInstrVSel = false;
3030 CC = (ARMCC::CondCodes)Instr.getOperand(IO - 1).getImm();
3031 break;
3032 case ARM::VSELEQD:
3033 case ARM::VSELEQS:
3034 case ARM::VSELEQH:
3035 CC = ARMCC::EQ;
3036 break;
3037 case ARM::VSELGTD:
3038 case ARM::VSELGTS:
3039 case ARM::VSELGTH:
3040 CC = ARMCC::GT;
3041 break;
3042 case ARM::VSELGED:
3043 case ARM::VSELGES:
3044 case ARM::VSELGEH:
3045 CC = ARMCC::GE;
3046 break;
3047 case ARM::VSELVSD:
3048 case ARM::VSELVSS:
3049 case ARM::VSELVSH:
3050 CC = ARMCC::VS;
3051 break;
3052 }
3053
3054 if (SubAdd) {
3055 // If we have SUB(r1, r2) and CMP(r2, r1), the condition code based
3056 // on CMP needs to be updated to be based on SUB.
3057 // If we have ADD(r1, r2, X) and CMP(r1, r2), the condition code also
3058 // needs to be modified.
3059 // Push the condition code operands to OperandsToUpdate.
3060 // If it is safe to remove CmpInstr, the condition code of these
3061 // operands will be modified.
3062 unsigned Opc = SubAdd->getOpcode();
3063 bool IsSub = Opc == ARM::SUBrr || Opc == ARM::t2SUBrr ||
3064 Opc == ARM::SUBri || Opc == ARM::t2SUBri ||
3065 Opc == ARM::tSUBrr || Opc == ARM::tSUBi3 ||
3066 Opc == ARM::tSUBi8;
3067 unsigned OpI = Opc != ARM::tSUBrr ? 1 : 2;
3068 if (!IsSub ||
3069 (SrcReg2 != 0 && SubAdd->getOperand(OpI).getReg() == SrcReg2 &&
3070 SubAdd->getOperand(OpI + 1).getReg() == SrcReg)) {
3071 // VSel doesn't support condition code update.
3072 if (IsInstrVSel)
3073 return false;
3074 // Ensure we can swap the condition.
3075 ARMCC::CondCodes NewCC = (IsSub ? getSwappedCondition(CC) : getCmpToAddCondition(CC));
3076 if (NewCC == ARMCC::AL)
3077 return false;
3078 OperandsToUpdate.push_back(
3079 std::make_pair(&((*I).getOperand(IO - 1)), NewCC));
3080 }
3081 } else {
3082 // No SubAdd, so this is x = <op> y, z; cmp x, 0.
3083 switch (CC) {
3084 case ARMCC::EQ: // Z
3085 case ARMCC::NE: // Z
3086 case ARMCC::MI: // N
3087 case ARMCC::PL: // N
3088 case ARMCC::AL: // none
3089 // CPSR can be used multiple times, we should continue.
3090 break;
3091 case ARMCC::HS: // C
3092 case ARMCC::LO: // C
3093 case ARMCC::VS: // V
3094 case ARMCC::VC: // V
3095 case ARMCC::HI: // C Z
3096 case ARMCC::LS: // C Z
3097 case ARMCC::GE: // N V
3098 case ARMCC::LT: // N V
3099 case ARMCC::GT: // Z N V
3100 case ARMCC::LE: // Z N V
3101 // The instruction uses the V bit or C bit which is not safe.
3102 return false;
3103 }
3104 }
3105 }
3106 }
3107
3108 // If CPSR is not killed nor re-defined, we should check whether it is
3109 // live-out. If it is live-out, do not optimize.
3110 if (!isSafe) {
3111 MachineBasicBlock *MBB = CmpInstr.getParent();
3112 for (MachineBasicBlock *Succ : MBB->successors())
3113 if (Succ->isLiveIn(ARM::CPSR))
3114 return false;
3115 }
3116
3117 // Toggle the optional operand to CPSR (if it exists - in Thumb1 we always
3118 // set CPSR so this is represented as an explicit output)
3119 if (!IsThumb1) {
3120 unsigned CPSRRegNum = MI->getNumExplicitOperands() - 1;
3121 MI->getOperand(CPSRRegNum).setReg(ARM::CPSR);
3122 MI->getOperand(CPSRRegNum).setIsDef(true);
3123 }
3124 assert(!isPredicated(*MI) && "Can't use flags from predicated instruction");
3125 CmpInstr.eraseFromParent();
3126
3127 // Modify the condition code of operands in OperandsToUpdate.
3128 // Since we have SUB(r1, r2) and CMP(r2, r1), the condition code needs to
3129 // be changed from r2 > r1 to r1 < r2, from r2 < r1 to r1 > r2, etc.
3130 for (auto &[MO, Cond] : OperandsToUpdate)
3131 MO->setImm(Cond);
3132
3133 MI->clearRegisterDeads(ARM::CPSR);
3134
3135 return true;
3136}
3137
3139 // Do not sink MI if it might be used to optimize a redundant compare.
3140 // We heuristically only look at the instruction immediately following MI to
3141 // avoid potentially searching the entire basic block.
3142 if (isPredicated(MI))
3143 return true;
3145 ++Next;
3146 Register SrcReg, SrcReg2;
3147 int64_t CmpMask, CmpValue;
3148 bool IsThumb1;
3149 if (Next != MI.getParent()->end() &&
3150 analyzeCompare(*Next, SrcReg, SrcReg2, CmpMask, CmpValue) &&
3151 isRedundantFlagInstr(&*Next, SrcReg, SrcReg2, CmpValue, &MI, IsThumb1))
3152 return false;
3153 return true;
3154}
3155
3157 Register Reg,
3158 MachineRegisterInfo *MRI) const {
3159 // Fold large immediates into add, sub, or, xor.
3160 unsigned DefOpc = DefMI.getOpcode();
3161 if (DefOpc != ARM::t2MOVi32imm && DefOpc != ARM::MOVi32imm &&
3162 DefOpc != ARM::tMOVi32imm)
3163 return false;
3164 if (!DefMI.getOperand(1).isImm())
3165 // Could be t2MOVi32imm @xx
3166 return false;
3167
3168 if (!MRI->hasOneNonDBGUse(Reg))
3169 return false;
3170
3171 const MCInstrDesc &DefMCID = DefMI.getDesc();
3172 if (DefMCID.hasOptionalDef()) {
3173 unsigned NumOps = DefMCID.getNumOperands();
3174 const MachineOperand &MO = DefMI.getOperand(NumOps - 1);
3175 if (MO.getReg() == ARM::CPSR && !MO.isDead())
3176 // If DefMI defines CPSR and it is not dead, it's obviously not safe
3177 // to delete DefMI.
3178 return false;
3179 }
3180
3181 const MCInstrDesc &UseMCID = UseMI.getDesc();
3182 if (UseMCID.hasOptionalDef()) {
3183 unsigned NumOps = UseMCID.getNumOperands();
3184 if (UseMI.getOperand(NumOps - 1).getReg() == ARM::CPSR)
3185 // If the instruction sets the flag, do not attempt this optimization
3186 // since it may change the semantics of the code.
3187 return false;
3188 }
3189
3190 unsigned UseOpc = UseMI.getOpcode();
3191 unsigned NewUseOpc = 0;
3192 uint32_t ImmVal = (uint32_t)DefMI.getOperand(1).getImm();
3193 uint32_t SOImmValV1 = 0, SOImmValV2 = 0;
3194 bool Commute = false;
3195 switch (UseOpc) {
3196 default: return false;
3197 case ARM::SUBrr:
3198 case ARM::ADDrr:
3199 case ARM::ORRrr:
3200 case ARM::EORrr:
3201 case ARM::t2SUBrr:
3202 case ARM::t2ADDrr:
3203 case ARM::t2ORRrr:
3204 case ARM::t2EORrr: {
3205 Commute = UseMI.getOperand(2).getReg() != Reg;
3206 switch (UseOpc) {
3207 default: break;
3208 case ARM::ADDrr:
3209 case ARM::SUBrr:
3210 if (UseOpc == ARM::SUBrr && Commute)
3211 return false;
3212
3213 // ADD/SUB are special because they're essentially the same operation, so
3214 // we can handle a larger range of immediates.
3215 if (ARM_AM::isSOImmTwoPartVal(ImmVal))
3216 NewUseOpc = UseOpc == ARM::ADDrr ? ARM::ADDri : ARM::SUBri;
3217 else if (ARM_AM::isSOImmTwoPartVal(-ImmVal)) {
3218 ImmVal = -ImmVal;
3219 NewUseOpc = UseOpc == ARM::ADDrr ? ARM::SUBri : ARM::ADDri;
3220 } else
3221 return false;
3222 SOImmValV1 = (uint32_t)ARM_AM::getSOImmTwoPartFirst(ImmVal);
3223 SOImmValV2 = (uint32_t)ARM_AM::getSOImmTwoPartSecond(ImmVal);
3224 break;
3225 case ARM::ORRrr:
3226 case ARM::EORrr:
3227 if (!ARM_AM::isSOImmTwoPartVal(ImmVal))
3228 return false;
3229 SOImmValV1 = (uint32_t)ARM_AM::getSOImmTwoPartFirst(ImmVal);
3230 SOImmValV2 = (uint32_t)ARM_AM::getSOImmTwoPartSecond(ImmVal);
3231 switch (UseOpc) {
3232 default: break;
3233 case ARM::ORRrr: NewUseOpc = ARM::ORRri; break;
3234 case ARM::EORrr: NewUseOpc = ARM::EORri; break;
3235 }
3236 break;
3237 case ARM::t2ADDrr:
3238 case ARM::t2SUBrr: {
3239 if (UseOpc == ARM::t2SUBrr && Commute)
3240 return false;
3241
3242 // ADD/SUB are special because they're essentially the same operation, so
3243 // we can handle a larger range of immediates.
3244 const bool ToSP = DefMI.getOperand(0).getReg() == ARM::SP;
3245 const unsigned t2ADD = ToSP ? ARM::t2ADDspImm : ARM::t2ADDri;
3246 const unsigned t2SUB = ToSP ? ARM::t2SUBspImm : ARM::t2SUBri;
3247 if (ARM_AM::isT2SOImmTwoPartVal(ImmVal))
3248 NewUseOpc = UseOpc == ARM::t2ADDrr ? t2ADD : t2SUB;
3249 else if (ARM_AM::isT2SOImmTwoPartVal(-ImmVal)) {
3250 ImmVal = -ImmVal;
3251 NewUseOpc = UseOpc == ARM::t2ADDrr ? t2SUB : t2ADD;
3252 } else
3253 return false;
3254 SOImmValV1 = (uint32_t)ARM_AM::getT2SOImmTwoPartFirst(ImmVal);
3255 SOImmValV2 = (uint32_t)ARM_AM::getT2SOImmTwoPartSecond(ImmVal);
3256 break;
3257 }
3258 case ARM::t2ORRrr:
3259 case ARM::t2EORrr:
3260 if (!ARM_AM::isT2SOImmTwoPartVal(ImmVal))
3261 return false;
3262 SOImmValV1 = (uint32_t)ARM_AM::getT2SOImmTwoPartFirst(ImmVal);
3263 SOImmValV2 = (uint32_t)ARM_AM::getT2SOImmTwoPartSecond(ImmVal);
3264 switch (UseOpc) {
3265 default: break;
3266 case ARM::t2ORRrr: NewUseOpc = ARM::t2ORRri; break;
3267 case ARM::t2EORrr: NewUseOpc = ARM::t2EORri; break;
3268 }
3269 break;
3270 }
3271 }
3272 }
3273
3274 unsigned OpIdx = Commute ? 2 : 1;
3275 Register Reg1 = UseMI.getOperand(OpIdx).getReg();
3276 bool isKill = UseMI.getOperand(OpIdx).isKill();
3277 const TargetRegisterClass *TRC = MRI->getRegClass(Reg);
3278 Register NewReg = MRI->createVirtualRegister(TRC);
3279 BuildMI(*UseMI.getParent(), UseMI, UseMI.getDebugLoc(), get(NewUseOpc),
3280 NewReg)
3281 .addReg(Reg1, getKillRegState(isKill))
3282 .addImm(SOImmValV1)
3284 .add(condCodeOp());
3285 UseMI.setDesc(get(NewUseOpc));
3286 UseMI.getOperand(1).setReg(NewReg);
3287 UseMI.getOperand(1).setIsKill();
3288 UseMI.getOperand(2).ChangeToImmediate(SOImmValV2);
3289 DefMI.eraseFromParent();
3290 // FIXME: t2ADDrr should be split, as different rulles apply when writing to SP.
3291 // Just as t2ADDri, that was split to [t2ADDri, t2ADDspImm].
3292 // Then the below code will not be needed, as the input/output register
3293 // classes will be rgpr or gprSP.
3294 // For now, we fix the UseMI operand explicitly here:
3295 switch(NewUseOpc){
3296 case ARM::t2ADDspImm:
3297 case ARM::t2SUBspImm:
3298 case ARM::t2ADDri:
3299 case ARM::t2SUBri:
3300 MRI->constrainRegClass(UseMI.getOperand(0).getReg(), TRC);
3301 }
3302 return true;
3303}
3304
3305static unsigned getNumMicroOpsSwiftLdSt(const InstrItineraryData *ItinData,
3306 const MachineInstr &MI) {
3307 switch (MI.getOpcode()) {
3308 default: {
3309 const MCInstrDesc &Desc = MI.getDesc();
3310 int UOps = ItinData->getNumMicroOps(Desc.getSchedClass());
3311 assert(UOps >= 0 && "bad # UOps");
3312 return UOps;
3313 }
3314
3315 case ARM::LDRrs:
3316 case ARM::LDRBrs:
3317 case ARM::STRrs:
3318 case ARM::STRBrs: {
3319 unsigned ShOpVal = MI.getOperand(3).getImm();
3320 bool isSub = ARM_AM::getAM2Op(ShOpVal) == ARM_AM::sub;
3321 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
3322 if (!isSub &&
3323 (ShImm == 0 ||
3324 ((ShImm == 1 || ShImm == 2 || ShImm == 3) &&
3325 ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsl)))
3326 return 1;
3327 return 2;
3328 }
3329
3330 case ARM::LDRH:
3331 case ARM::STRH: {
3332 if (!MI.getOperand(2).getReg())
3333 return 1;
3334
3335 unsigned ShOpVal = MI.getOperand(3).getImm();
3336 bool isSub = ARM_AM::getAM2Op(ShOpVal) == ARM_AM::sub;
3337 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
3338 if (!isSub &&
3339 (ShImm == 0 ||
3340 ((ShImm == 1 || ShImm == 2 || ShImm == 3) &&
3341 ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsl)))
3342 return 1;
3343 return 2;
3344 }
3345
3346 case ARM::LDRSB:
3347 case ARM::LDRSH:
3348 return (ARM_AM::getAM3Op(MI.getOperand(3).getImm()) == ARM_AM::sub) ? 3 : 2;
3349
3350 case ARM::LDRSB_POST:
3351 case ARM::LDRSH_POST: {
3352 Register Rt = MI.getOperand(0).getReg();
3353 Register Rm = MI.getOperand(3).getReg();
3354 return (Rt == Rm) ? 4 : 3;
3355 }
3356
3357 case ARM::LDR_PRE_REG:
3358 case ARM::LDRB_PRE_REG: {
3359 Register Rt = MI.getOperand(0).getReg();
3360 Register Rm = MI.getOperand(3).getReg();
3361 if (Rt == Rm)
3362 return 3;
3363 unsigned ShOpVal = MI.getOperand(4).getImm();
3364 bool isSub = ARM_AM::getAM2Op(ShOpVal) == ARM_AM::sub;
3365 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
3366 if (!isSub &&
3367 (ShImm == 0 ||
3368 ((ShImm == 1 || ShImm == 2 || ShImm == 3) &&
3369 ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsl)))
3370 return 2;
3371 return 3;
3372 }
3373
3374 case ARM::STR_PRE_REG:
3375 case ARM::STRB_PRE_REG: {
3376 unsigned ShOpVal = MI.getOperand(4).getImm();
3377 bool isSub = ARM_AM::getAM2Op(ShOpVal) == ARM_AM::sub;
3378 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
3379 if (!isSub &&
3380 (ShImm == 0 ||
3381 ((ShImm == 1 || ShImm == 2 || ShImm == 3) &&
3382 ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsl)))
3383 return 2;
3384 return 3;
3385 }
3386
3387 case ARM::LDRH_PRE:
3388 case ARM::STRH_PRE: {
3389 Register Rt = MI.getOperand(0).getReg();
3390 Register Rm = MI.getOperand(3).getReg();
3391 if (!Rm)
3392 return 2;
3393 if (Rt == Rm)
3394 return 3;
3395 return (ARM_AM::getAM3Op(MI.getOperand(4).getImm()) == ARM_AM::sub) ? 3 : 2;
3396 }
3397
3398 case ARM::LDR_POST_REG:
3399 case ARM::LDRB_POST_REG:
3400 case ARM::LDRH_POST: {
3401 Register Rt = MI.getOperand(0).getReg();
3402 Register Rm = MI.getOperand(3).getReg();
3403 return (Rt == Rm) ? 3 : 2;
3404 }
3405
3406 case ARM::LDR_PRE_IMM:
3407 case ARM::LDRB_PRE_IMM:
3408 case ARM::LDR_POST_IMM:
3409 case ARM::LDRB_POST_IMM:
3410 case ARM::STRB_POST_IMM:
3411 case ARM::STRB_POST_REG:
3412 case ARM::STRB_PRE_IMM:
3413 case ARM::STRH_POST:
3414 case ARM::STR_POST_IMM:
3415 case ARM::STR_POST_REG:
3416 case ARM::STR_PRE_IMM:
3417 return 2;
3418
3419 case ARM::LDRSB_PRE:
3420 case ARM::LDRSH_PRE: {
3421 Register Rm = MI.getOperand(3).getReg();
3422 if (Rm == 0)
3423 return 3;
3424 Register Rt = MI.getOperand(0).getReg();
3425 if (Rt == Rm)
3426 return 4;
3427 unsigned ShOpVal = MI.getOperand(4).getImm();
3428 bool isSub = ARM_AM::getAM2Op(ShOpVal) == ARM_AM::sub;
3429 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
3430 if (!isSub &&
3431 (ShImm == 0 ||
3432 ((ShImm == 1 || ShImm == 2 || ShImm == 3) &&
3433 ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsl)))
3434 return 3;
3435 return 4;
3436 }
3437
3438 case ARM::LDRD: {
3439 Register Rt = MI.getOperand(0).getReg();
3440 Register Rn = MI.getOperand(2).getReg();
3441 Register Rm = MI.getOperand(3).getReg();
3442 if (Rm)
3443 return (ARM_AM::getAM3Op(MI.getOperand(4).getImm()) == ARM_AM::sub) ? 4
3444 : 3;
3445 return (Rt == Rn) ? 3 : 2;
3446 }
3447
3448 case ARM::STRD: {
3449 Register Rm = MI.getOperand(3).getReg();
3450 if (Rm)
3451 return (ARM_AM::getAM3Op(MI.getOperand(4).getImm()) == ARM_AM::sub) ? 4
3452 : 3;
3453 return 2;
3454 }
3455
3456 case ARM::LDRD_POST:
3457 case ARM::t2LDRD_POST:
3458 return 3;
3459
3460 case ARM::STRD_POST:
3461 case ARM::t2STRD_POST:
3462 return 4;
3463
3464 case ARM::LDRD_PRE: {
3465 Register Rt = MI.getOperand(0).getReg();
3466 Register Rn = MI.getOperand(3).getReg();
3467 Register Rm = MI.getOperand(4).getReg();
3468 if (Rm)
3469 return (ARM_AM::getAM3Op(MI.getOperand(5).getImm()) == ARM_AM::sub) ? 5
3470 : 4;
3471 return (Rt == Rn) ? 4 : 3;
3472 }
3473
3474 case ARM::t2LDRD_PRE: {
3475 Register Rt = MI.getOperand(0).getReg();
3476 Register Rn = MI.getOperand(3).getReg();
3477 return (Rt == Rn) ? 4 : 3;
3478 }
3479
3480 case ARM::STRD_PRE: {
3481 Register Rm = MI.getOperand(4).getReg();
3482 if (Rm)
3483 return (ARM_AM::getAM3Op(MI.getOperand(5).getImm()) == ARM_AM::sub) ? 5
3484 : 4;
3485 return 3;
3486 }
3487
3488 case ARM::t2STRD_PRE:
3489 return 3;
3490
3491 case ARM::t2LDR_POST:
3492 case ARM::t2LDRB_POST:
3493 case ARM::t2LDRB_PRE:
3494 case ARM::t2LDRSBi12:
3495 case ARM::t2LDRSBi8:
3496 case ARM::t2LDRSBpci:
3497 case ARM::t2LDRSBs:
3498 case ARM::t2LDRH_POST:
3499 case ARM::t2LDRH_PRE:
3500 case ARM::t2LDRSBT:
3501 case ARM::t2LDRSB_POST:
3502 case ARM::t2LDRSB_PRE:
3503 case ARM::t2LDRSH_POST:
3504 case ARM::t2LDRSH_PRE:
3505 case ARM::t2LDRSHi12:
3506 case ARM::t2LDRSHi8:
3507 case ARM::t2LDRSHpci:
3508 case ARM::t2LDRSHs:
3509 return 2;
3510
3511 case ARM::t2LDRDi8: {
3512 Register Rt = MI.getOperand(0).getReg();
3513 Register Rn = MI.getOperand(2).getReg();
3514 return (Rt == Rn) ? 3 : 2;
3515 }
3516
3517 case ARM::t2STRB_POST:
3518 case ARM::t2STRB_PRE:
3519 case ARM::t2STRBs:
3520 case ARM::t2STRDi8:
3521 case ARM::t2STRH_POST:
3522 case ARM::t2STRH_PRE:
3523 case ARM::t2STRHs:
3524 case ARM::t2STR_POST:
3525 case ARM::t2STR_PRE:
3526 case ARM::t2STRs:
3527 return 2;
3528 }
3529}
3530
3531// Return the number of 32-bit words loaded by LDM or stored by STM. If this
3532// can't be easily determined return 0 (missing MachineMemOperand).
3533//
3534// FIXME: The current MachineInstr design does not support relying on machine
3535// mem operands to determine the width of a memory access. Instead, we expect
3536// the target to provide this information based on the instruction opcode and
3537// operands. However, using MachineMemOperand is the best solution now for
3538// two reasons:
3539//
3540// 1) getNumMicroOps tries to infer LDM memory width from the total number of MI
3541// operands. This is much more dangerous than using the MachineMemOperand
3542// sizes because CodeGen passes can insert/remove optional machine operands. In
3543// fact, it's totally incorrect for preRA passes and appears to be wrong for
3544// postRA passes as well.
3545//
3546// 2) getNumLDMAddresses is only used by the scheduling machine model and any
3547// machine model that calls this should handle the unknown (zero size) case.
3548//
3549// Long term, we should require a target hook that verifies MachineMemOperand
3550// sizes during MC lowering. That target hook should be local to MC lowering
3551// because we can't ensure that it is aware of other MI forms. Doing this will
3552// ensure that MachineMemOperands are correctly propagated through all passes.
3554 unsigned Size = 0;
3555 for (MachineInstr::mmo_iterator I = MI.memoperands_begin(),
3556 E = MI.memoperands_end();
3557 I != E; ++I) {
3558 Size += (*I)->getSize().getValue();
3559 }
3560 // FIXME: The scheduler currently can't handle values larger than 16. But
3561 // the values can actually go up to 32 for floating-point load/store
3562 // multiple (VLDMIA etc.). Also, the way this code is reasoning about memory
3563 // operations isn't right; we could end up with "extra" memory operands for
3564 // various reasons, like tail merge merging two memory operations.
3565 return std::min(Size / 4, 16U);
3566}
3567
3569 unsigned NumRegs) {
3570 unsigned UOps = 1 + NumRegs; // 1 for address computation.
3571 switch (Opc) {
3572 default:
3573 break;
3574 case ARM::VLDMDIA_UPD:
3575 case ARM::VLDMDDB_UPD:
3576 case ARM::VLDMSIA_UPD:
3577 case ARM::VLDMSDB_UPD:
3578 case ARM::VSTMDIA_UPD:
3579 case ARM::VSTMDDB_UPD:
3580 case ARM::VSTMSIA_UPD:
3581 case ARM::VSTMSDB_UPD:
3582 case ARM::LDMIA_UPD:
3583 case ARM::LDMDA_UPD:
3584 case ARM::LDMDB_UPD:
3585 case ARM::LDMIB_UPD:
3586 case ARM::STMIA_UPD:
3587 case ARM::STMDA_UPD:
3588 case ARM::STMDB_UPD:
3589 case ARM::STMIB_UPD:
3590 case ARM::tLDMIA_UPD:
3591 case ARM::tSTMIA_UPD:
3592 case ARM::t2LDMIA_UPD:
3593 case ARM::t2LDMDB_UPD:
3594 case ARM::t2STMIA_UPD:
3595 case ARM::t2STMDB_UPD:
3596 ++UOps; // One for base register writeback.
3597 break;
3598 case ARM::LDMIA_RET:
3599 case ARM::tPOP_RET:
3600 case ARM::t2LDMIA_RET:
3601 UOps += 2; // One for base reg wb, one for write to pc.
3602 break;
3603 }
3604 return UOps;
3605}
3606
3608 const MachineInstr &MI) const {
3609 if (!ItinData || ItinData->isEmpty())
3610 return 1;
3611
3612 const MCInstrDesc &Desc = MI.getDesc();
3613 unsigned Class = Desc.getSchedClass();
3614 int ItinUOps = ItinData->getNumMicroOps(Class);
3615 if (ItinUOps >= 0) {
3616 if (Subtarget.isSwift() && (Desc.mayLoad() || Desc.mayStore()))
3617 return getNumMicroOpsSwiftLdSt(ItinData, MI);
3618
3619 return ItinUOps;
3620 }
3621
3622 unsigned Opc = MI.getOpcode();
3623 switch (Opc) {
3624 default:
3625 llvm_unreachable("Unexpected multi-uops instruction!");
3626 case ARM::VLDMQIA:
3627 case ARM::VSTMQIA:
3628 return 2;
3629
3630 // The number of uOps for load / store multiple are determined by the number
3631 // registers.
3632 //
3633 // On Cortex-A8, each pair of register loads / stores can be scheduled on the
3634 // same cycle. The scheduling for the first load / store must be done
3635 // separately by assuming the address is not 64-bit aligned.
3636 //
3637 // On Cortex-A9, the formula is simply (#reg / 2) + (#reg % 2). If the address
3638 // is not 64-bit aligned, then AGU would take an extra cycle. For VFP / NEON
3639 // load / store multiple, the formula is (#reg / 2) + (#reg % 2) + 1.
3640 case ARM::VLDMDIA:
3641 case ARM::VLDMDIA_UPD:
3642 case ARM::VLDMDDB_UPD:
3643 case ARM::VLDMSIA:
3644 case ARM::VLDMSIA_UPD:
3645 case ARM::VLDMSDB_UPD:
3646 case ARM::VSTMDIA:
3647 case ARM::VSTMDIA_UPD:
3648 case ARM::VSTMDDB_UPD:
3649 case ARM::VSTMSIA:
3650 case ARM::VSTMSIA_UPD:
3651 case ARM::VSTMSDB_UPD: {
3652 unsigned NumRegs = MI.getNumOperands() - Desc.getNumOperands();
3653 return (NumRegs / 2) + (NumRegs % 2) + 1;
3654 }
3655
3656 case ARM::LDMIA_RET:
3657 case ARM::LDMIA:
3658 case ARM::LDMDA:
3659 case ARM::LDMDB:
3660 case ARM::LDMIB:
3661 case ARM::LDMIA_UPD:
3662 case ARM::LDMDA_UPD:
3663 case ARM::LDMDB_UPD:
3664 case ARM::LDMIB_UPD:
3665 case ARM::STMIA:
3666 case ARM::STMDA:
3667 case ARM::STMDB:
3668 case ARM::STMIB:
3669 case ARM::STMIA_UPD:
3670 case ARM::STMDA_UPD:
3671 case ARM::STMDB_UPD:
3672 case ARM::STMIB_UPD:
3673 case ARM::tLDMIA:
3674 case ARM::tLDMIA_UPD:
3675 case ARM::tSTMIA_UPD:
3676 case ARM::tPOP_RET:
3677 case ARM::tPOP:
3678 case ARM::tPUSH:
3679 case ARM::t2LDMIA_RET:
3680 case ARM::t2LDMIA:
3681 case ARM::t2LDMDB:
3682 case ARM::t2LDMIA_UPD:
3683 case ARM::t2LDMDB_UPD:
3684 case ARM::t2STMIA:
3685 case ARM::t2STMDB:
3686 case ARM::t2STMIA_UPD:
3687 case ARM::t2STMDB_UPD: {
3688 unsigned NumRegs = MI.getNumOperands() - Desc.getNumOperands() + 1;
3689 switch (Subtarget.getLdStMultipleTiming()) {
3693 // Assume the worst.
3694 return NumRegs;
3696 if (NumRegs < 4)
3697 return 2;
3698 // 4 registers would be issued: 2, 2.
3699 // 5 registers would be issued: 2, 2, 1.
3700 unsigned UOps = (NumRegs / 2);
3701 if (NumRegs % 2)
3702 ++UOps;
3703 return UOps;
3704 }
3706 unsigned UOps = (NumRegs / 2);
3707 // If there are odd number of registers or if it's not 64-bit aligned,
3708 // then it takes an extra AGU (Address Generation Unit) cycle.
3709 if ((NumRegs % 2) || !MI.hasOneMemOperand() ||
3710 (*MI.memoperands_begin())->getAlign() < Align(8))
3711 ++UOps;
3712 return UOps;
3713 }
3714 }
3715 }
3716 }
3717 llvm_unreachable("Didn't find the number of microops");
3718}
3719
3720std::optional<unsigned>
3721ARMBaseInstrInfo::getVLDMDefCycle(const InstrItineraryData *ItinData,
3722 const MCInstrDesc &DefMCID, unsigned DefClass,
3723 unsigned DefIdx, unsigned DefAlign) const {
3724 int RegNo = (int)(DefIdx+1) - DefMCID.getNumOperands() + 1;
3725 if (RegNo <= 0)
3726 // Def is the address writeback.
3727 return ItinData->getOperandCycle(DefClass, DefIdx);
3728
3729 unsigned DefCycle;
3730 if (Subtarget.isCortexA8() || Subtarget.isCortexA7()) {
3731 // (regno / 2) + (regno % 2) + 1
3732 DefCycle = RegNo / 2 + 1;
3733 if (RegNo % 2)
3734 ++DefCycle;
3735 } else if (Subtarget.isLikeA9() || Subtarget.isSwift()) {
3736 DefCycle = RegNo;
3737 bool isSLoad = false;
3738
3739 switch (DefMCID.getOpcode()) {
3740 default: break;
3741 case ARM::VLDMSIA:
3742 case ARM::VLDMSIA_UPD:
3743 case ARM::VLDMSDB_UPD:
3744 isSLoad = true;
3745 break;
3746 }
3747
3748 // If there are odd number of 'S' registers or if it's not 64-bit aligned,
3749 // then it takes an extra cycle.
3750 if ((isSLoad && (RegNo % 2)) || DefAlign < 8)
3751 ++DefCycle;
3752 } else {
3753 // Assume the worst.
3754 DefCycle = RegNo + 2;
3755 }
3756
3757 return DefCycle;
3758}
3759
3760std::optional<unsigned>
3761ARMBaseInstrInfo::getLDMDefCycle(const InstrItineraryData *ItinData,
3762 const MCInstrDesc &DefMCID, unsigned DefClass,
3763 unsigned DefIdx, unsigned DefAlign) const {
3764 int RegNo = (int)(DefIdx+1) - DefMCID.getNumOperands() + 1;
3765 if (RegNo <= 0)
3766 // Def is the address writeback.
3767 return ItinData->getOperandCycle(DefClass, DefIdx);
3768
3769 unsigned DefCycle;
3770 if (Subtarget.isCortexA8() || Subtarget.isCortexA7()) {
3771 // 4 registers would be issued: 1, 2, 1.
3772 // 5 registers would be issued: 1, 2, 2.
3773 DefCycle = RegNo / 2;
3774 if (DefCycle < 1)
3775 DefCycle = 1;
3776 // Result latency is issue cycle + 2: E2.
3777 DefCycle += 2;
3778 } else if (Subtarget.isLikeA9() || Subtarget.isSwift()) {
3779 DefCycle = (RegNo / 2);
3780 // If there are odd number of registers or if it's not 64-bit aligned,
3781 // then it takes an extra AGU (Address Generation Unit) cycle.
3782 if ((RegNo % 2) || DefAlign < 8)
3783 ++DefCycle;
3784 // Result latency is AGU cycles + 2.
3785 DefCycle += 2;
3786 } else {
3787 // Assume the worst.
3788 DefCycle = RegNo + 2;
3789 }
3790
3791 return DefCycle;
3792}
3793
3794std::optional<unsigned>
3795ARMBaseInstrInfo::getVSTMUseCycle(const InstrItineraryData *ItinData,
3796 const MCInstrDesc &UseMCID, unsigned UseClass,
3797 unsigned UseIdx, unsigned UseAlign) const {
3798 int RegNo = (int)(UseIdx+1) - UseMCID.getNumOperands() + 1;
3799 if (RegNo <= 0)
3800 return ItinData->getOperandCycle(UseClass, UseIdx);
3801
3802 unsigned UseCycle;
3803 if (Subtarget.isCortexA8() || Subtarget.isCortexA7()) {
3804 // (regno / 2) + (regno % 2) + 1
3805 UseCycle = RegNo / 2 + 1;
3806 if (RegNo % 2)
3807 ++UseCycle;
3808 } else if (Subtarget.isLikeA9() || Subtarget.isSwift()) {
3809 UseCycle = RegNo;
3810 bool isSStore = false;
3811
3812 switch (UseMCID.getOpcode()) {
3813 default: break;
3814 case ARM::VSTMSIA:
3815 case ARM::VSTMSIA_UPD:
3816 case ARM::VSTMSDB_UPD:
3817 isSStore = true;
3818 break;
3819 }
3820
3821 // If there are odd number of 'S' registers or if it's not 64-bit aligned,
3822 // then it takes an extra cycle.
3823 if ((isSStore && (RegNo % 2)) || UseAlign < 8)
3824 ++UseCycle;
3825 } else {
3826 // Assume the worst.
3827 UseCycle = RegNo + 2;
3828 }
3829
3830 return UseCycle;
3831}
3832
3833std::optional<unsigned>
3834ARMBaseInstrInfo::getSTMUseCycle(const InstrItineraryData *ItinData,
3835 const MCInstrDesc &UseMCID, unsigned UseClass,
3836 unsigned UseIdx, unsigned UseAlign) const {
3837 int RegNo = (int)(UseIdx+1) - UseMCID.getNumOperands() + 1;
3838 if (RegNo <= 0)
3839 return ItinData->getOperandCycle(UseClass, UseIdx);
3840
3841 unsigned UseCycle;
3842 if (Subtarget.isCortexA8() || Subtarget.isCortexA7()) {
3843 UseCycle = RegNo / 2;
3844 if (UseCycle < 2)
3845 UseCycle = 2;
3846 // Read in E3.
3847 UseCycle += 2;
3848 } else if (Subtarget.isLikeA9() || Subtarget.isSwift()) {
3849 UseCycle = (RegNo / 2);
3850 // If there are odd number of registers or if it's not 64-bit aligned,
3851 // then it takes an extra AGU (Address Generation Unit) cycle.
3852 if ((RegNo % 2) || UseAlign < 8)
3853 ++UseCycle;
3854 } else {
3855 // Assume the worst.
3856 UseCycle = 1;
3857 }
3858 return UseCycle;
3859}
3860
3861std::optional<unsigned> ARMBaseInstrInfo::getOperandLatency(
3862 const InstrItineraryData *ItinData, const MCInstrDesc &DefMCID,
3863 unsigned DefIdx, unsigned DefAlign, const MCInstrDesc &UseMCID,
3864 unsigned UseIdx, unsigned UseAlign) const {
3865 unsigned DefClass = DefMCID.getSchedClass();
3866 unsigned UseClass = UseMCID.getSchedClass();
3867
3868 if (DefIdx < DefMCID.getNumDefs() && UseIdx < UseMCID.getNumOperands())
3869 return ItinData->getOperandLatency(DefClass, DefIdx, UseClass, UseIdx);
3870
3871 // This may be a def / use of a variable_ops instruction, the operand
3872 // latency might be determinable dynamically. Let the target try to
3873 // figure it out.
3874 std::optional<unsigned> DefCycle;
3875 bool LdmBypass = false;
3876 switch (DefMCID.getOpcode()) {
3877 default:
3878 DefCycle = ItinData->getOperandCycle(DefClass, DefIdx);
3879 break;
3880
3881 case ARM::VLDMDIA:
3882 case ARM::VLDMDIA_UPD:
3883 case ARM::VLDMDDB_UPD:
3884 case ARM::VLDMSIA:
3885 case ARM::VLDMSIA_UPD:
3886 case ARM::VLDMSDB_UPD:
3887 DefCycle = getVLDMDefCycle(ItinData, DefMCID, DefClass, DefIdx, DefAlign);
3888 break;
3889
3890 case ARM::LDMIA_RET:
3891 case ARM::LDMIA:
3892 case ARM::LDMDA:
3893 case ARM::LDMDB:
3894 case ARM::LDMIB:
3895 case ARM::LDMIA_UPD:
3896 case ARM::LDMDA_UPD:
3897 case ARM::LDMDB_UPD:
3898 case ARM::LDMIB_UPD:
3899 case ARM::tLDMIA:
3900 case ARM::tLDMIA_UPD:
3901 case ARM::tPUSH:
3902 case ARM::t2LDMIA_RET:
3903 case ARM::t2LDMIA:
3904 case ARM::t2LDMDB:
3905 case ARM::t2LDMIA_UPD:
3906 case ARM::t2LDMDB_UPD:
3907 LdmBypass = true;
3908 DefCycle = getLDMDefCycle(ItinData, DefMCID, DefClass, DefIdx, DefAlign);
3909 break;
3910 }
3911
3912 if (!DefCycle)
3913 // We can't seem to determine the result latency of the def, assume it's 2.
3914 DefCycle = 2;
3915
3916 std::optional<unsigned> UseCycle;
3917 switch (UseMCID.getOpcode()) {
3918 default:
3919 UseCycle = ItinData->getOperandCycle(UseClass, UseIdx);
3920 break;
3921
3922 case ARM::VSTMDIA:
3923 case ARM::VSTMDIA_UPD:
3924 case ARM::VSTMDDB_UPD:
3925 case ARM::VSTMSIA:
3926 case ARM::VSTMSIA_UPD:
3927 case ARM::VSTMSDB_UPD:
3928 UseCycle = getVSTMUseCycle(ItinData, UseMCID, UseClass, UseIdx, UseAlign);
3929 break;
3930
3931 case ARM::STMIA:
3932 case ARM::STMDA:
3933 case ARM::STMDB:
3934 case ARM::STMIB:
3935 case ARM::STMIA_UPD:
3936 case ARM::STMDA_UPD:
3937 case ARM::STMDB_UPD:
3938 case ARM::STMIB_UPD:
3939 case ARM::tSTMIA_UPD:
3940 case ARM::tPOP_RET:
3941 case ARM::tPOP:
3942 case ARM::t2STMIA:
3943 case ARM::t2STMDB:
3944 case ARM::t2STMIA_UPD:
3945 case ARM::t2STMDB_UPD:
3946 UseCycle = getSTMUseCycle(ItinData, UseMCID, UseClass, UseIdx, UseAlign);
3947 break;
3948 }
3949
3950 if (!UseCycle)
3951 // Assume it's read in the first stage.
3952 UseCycle = 1;
3953
3954 if (UseCycle > *DefCycle + 1)
3955 return std::nullopt;
3956
3957 UseCycle = *DefCycle - *UseCycle + 1;
3958 if (UseCycle > 0u) {
3959 if (LdmBypass) {
3960 // It's a variable_ops instruction so we can't use DefIdx here. Just use
3961 // first def operand.
3962 if (ItinData->hasPipelineForwarding(DefClass, DefMCID.getNumOperands()-1,
3963 UseClass, UseIdx))
3964 UseCycle = *UseCycle - 1;
3965 } else if (ItinData->hasPipelineForwarding(DefClass, DefIdx,
3966 UseClass, UseIdx)) {
3967 UseCycle = *UseCycle - 1;
3968 }
3969 }
3970
3971 return UseCycle;
3972}
3973
3975 const MachineInstr *MI, unsigned Reg,
3976 unsigned &DefIdx, unsigned &Dist) {
3977 Dist = 0;
3978
3980 MachineBasicBlock::const_instr_iterator II = std::prev(I.getInstrIterator());
3981 assert(II->isInsideBundle() && "Empty bundle?");
3982
3983 int Idx = -1;
3984 while (II->isInsideBundle()) {
3985 Idx = II->findRegisterDefOperandIdx(Reg, TRI, false, true);
3986 if (Idx != -1)
3987 break;
3988 --II;
3989 ++Dist;
3990 }
3991
3992 assert(Idx != -1 && "Cannot find bundled definition!");
3993 DefIdx = Idx;
3994 return &*II;
3995}
3996
3998 const MachineInstr &MI, unsigned Reg,
3999 unsigned &UseIdx, unsigned &Dist) {
4000 Dist = 0;
4001
4003 assert(II->isInsideBundle() && "Empty bundle?");
4004 MachineBasicBlock::const_instr_iterator E = MI.getParent()->instr_end();
4005
4006 // FIXME: This doesn't properly handle multiple uses.
4007 int Idx = -1;
4008 while (II != E && II->isInsideBundle()) {
4009 Idx = II->findRegisterUseOperandIdx(Reg, TRI, false);
4010 if (Idx != -1)
4011 break;
4012 if (II->getOpcode() != ARM::t2IT)
4013 ++Dist;
4014 ++II;
4015 }
4016
4017 if (Idx == -1) {
4018 Dist = 0;
4019 return nullptr;
4020 }
4021
4022 UseIdx = Idx;
4023 return &*II;
4024}
4025
4026/// Return the number of cycles to add to (or subtract from) the static
4027/// itinerary based on the def opcode and alignment. The caller will ensure that
4028/// adjusted latency is at least one cycle.
4029static int adjustDefLatency(const ARMSubtarget &Subtarget,
4030 const MachineInstr &DefMI,
4031 const MCInstrDesc &DefMCID, unsigned DefAlign) {
4032 int Adjust = 0;
4033 if (Subtarget.isCortexA8() || Subtarget.isLikeA9() || Subtarget.isCortexA7()) {
4034 // FIXME: Shifter op hack: no shift (i.e. [r +/- r]) or [r + r << 2]
4035 // variants are one cycle cheaper.
4036 switch (DefMCID.getOpcode()) {
4037 default: break;
4038 case ARM::LDRrs:
4039 case ARM::LDRBrs: {
4040 unsigned ShOpVal = DefMI.getOperand(3).getImm();
4041 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
4042 if (ShImm == 0 ||
4043 (ShImm == 2 && ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsl))
4044 --Adjust;
4045 break;
4046 }
4047 case ARM::t2LDRs:
4048 case ARM::t2LDRBs:
4049 case ARM::t2LDRHs:
4050 case ARM::t2LDRSHs: {
4051 // Thumb2 mode: lsl only.
4052 unsigned ShAmt = DefMI.getOperand(3).getImm();
4053 if (ShAmt == 0 || ShAmt == 2)
4054 --Adjust;
4055 break;
4056 }
4057 }
4058 } else if (Subtarget.isSwift()) {
4059 // FIXME: Properly handle all of the latency adjustments for address
4060 // writeback.
4061 switch (DefMCID.getOpcode()) {
4062 default: break;
4063 case ARM::LDRrs:
4064 case ARM::LDRBrs: {
4065 unsigned ShOpVal = DefMI.getOperand(3).getImm();
4066 bool isSub = ARM_AM::getAM2Op(ShOpVal) == ARM_AM::sub;
4067 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
4068 if (!isSub &&
4069 (ShImm == 0 ||
4070 ((ShImm == 1 || ShImm == 2 || ShImm == 3) &&
4071 ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsl)))
4072 Adjust -= 2;
4073 else if (!isSub &&
4074 ShImm == 1 && ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsr)
4075 --Adjust;
4076 break;
4077 }
4078 case ARM::t2LDRs:
4079 case ARM::t2LDRBs:
4080 case ARM::t2LDRHs:
4081 case ARM::t2LDRSHs: {
4082 // Thumb2 mode: lsl only.
4083 unsigned ShAmt = DefMI.getOperand(3).getImm();
4084 if (ShAmt == 0 || ShAmt == 1 || ShAmt == 2 || ShAmt == 3)
4085 Adjust -= 2;
4086 break;
4087 }
4088 }
4089 }
4090
4091 if (DefAlign < 8 && Subtarget.checkVLDnAccessAlignment()) {
4092 switch (DefMCID.getOpcode()) {
4093 default: break;
4094 case ARM::VLD1q8:
4095 case ARM::VLD1q16:
4096 case ARM::VLD1q32:
4097 case ARM::VLD1q64:
4098 case ARM::VLD1q8wb_fixed:
4099 case ARM::VLD1q16wb_fixed:
4100 case ARM::VLD1q32wb_fixed:
4101 case ARM::VLD1q64wb_fixed:
4102 case ARM::VLD1q8wb_register:
4103 case ARM::VLD1q16wb_register:
4104 case ARM::VLD1q32wb_register:
4105 case ARM::VLD1q64wb_register:
4106 case ARM::VLD2d8:
4107 case ARM::VLD2d16:
4108 case ARM::VLD2d32:
4109 case ARM::VLD2q8:
4110 case ARM::VLD2q16:
4111 case ARM::VLD2q32:
4112 case ARM::VLD2d8wb_fixed:
4113 case ARM::VLD2d16wb_fixed:
4114 case ARM::VLD2d32wb_fixed:
4115 case ARM::VLD2q8wb_fixed:
4116 case ARM::VLD2q16wb_fixed:
4117 case ARM::VLD2q32wb_fixed:
4118 case ARM::VLD2d8wb_register:
4119 case ARM::VLD2d16wb_register:
4120 case ARM::VLD2d32wb_register:
4121 case ARM::VLD2q8wb_register:
4122 case ARM::VLD2q16wb_register:
4123 case ARM::VLD2q32wb_register:
4124 case ARM::VLD3d8:
4125 case ARM::VLD3d16:
4126 case ARM::VLD3d32:
4127 case ARM::VLD1d64T:
4128 case ARM::VLD3d8_UPD:
4129 case ARM::VLD3d16_UPD:
4130 case ARM::VLD3d32_UPD:
4131 case ARM::VLD1d64Twb_fixed:
4132 case ARM::VLD1d64Twb_register:
4133 case ARM::VLD3q8_UPD:
4134 case ARM::VLD3q16_UPD:
4135 case ARM::VLD3q32_UPD:
4136 case ARM::VLD4d8:
4137 case ARM::VLD4d16:
4138 case ARM::VLD4d32:
4139 case ARM::VLD1d64Q:
4140 case ARM::VLD4d8_UPD:
4141 case ARM::VLD4d16_UPD:
4142 case ARM::VLD4d32_UPD:
4143 case ARM::VLD1d64Qwb_fixed:
4144 case ARM::VLD1d64Qwb_register:
4145 case ARM::VLD4q8_UPD:
4146 case ARM::VLD4q16_UPD:
4147 case ARM::VLD4q32_UPD:
4148 case ARM::VLD1DUPq8:
4149 case ARM::VLD1DUPq16:
4150 case ARM::VLD1DUPq32:
4151 case ARM::VLD1DUPq8wb_fixed:
4152 case ARM::VLD1DUPq16wb_fixed:
4153 case ARM::VLD1DUPq32wb_fixed:
4154 case ARM::VLD1DUPq8wb_register:
4155 case ARM::VLD1DUPq16wb_register:
4156 case ARM::VLD1DUPq32wb_register:
4157 case ARM::VLD2DUPd8:
4158 case ARM::VLD2DUPd16:
4159 case ARM::VLD2DUPd32:
4160 case ARM::VLD2DUPd8wb_fixed:
4161 case ARM::VLD2DUPd16wb_fixed:
4162 case ARM::VLD2DUPd32wb_fixed:
4163 case ARM::VLD2DUPd8wb_register:
4164 case ARM::VLD2DUPd16wb_register:
4165 case ARM::VLD2DUPd32wb_register:
4166 case ARM::VLD4DUPd8:
4167 case ARM::VLD4DUPd16:
4168 case ARM::VLD4DUPd32:
4169 case ARM::VLD4DUPd8_UPD:
4170 case ARM::VLD4DUPd16_UPD:
4171 case ARM::VLD4DUPd32_UPD:
4172 case ARM::VLD1LNd8:
4173 case ARM::VLD1LNd16:
4174 case ARM::VLD1LNd32:
4175 case ARM::VLD1LNd8_UPD:
4176 case ARM::VLD1LNd16_UPD:
4177 case ARM::VLD1LNd32_UPD:
4178 case ARM::VLD2LNd8:
4179 case ARM::VLD2LNd16:
4180 case ARM::VLD2LNd32:
4181 case ARM::VLD2LNq16:
4182 case ARM::VLD2LNq32:
4183 case ARM::VLD2LNd8_UPD:
4184 case ARM::VLD2LNd16_UPD:
4185 case ARM::VLD2LNd32_UPD:
4186 case ARM::VLD2LNq16_UPD:
4187 case ARM::VLD2LNq32_UPD:
4188 case ARM::VLD4LNd8:
4189 case ARM::VLD4LNd16:
4190 case ARM::VLD4LNd32:
4191 case ARM::VLD4LNq16:
4192 case ARM::VLD4LNq32:
4193 case ARM::VLD4LNd8_UPD:
4194 case ARM::VLD4LNd16_UPD:
4195 case ARM::VLD4LNd32_UPD:
4196 case ARM::VLD4LNq16_UPD:
4197 case ARM::VLD4LNq32_UPD:
4198 // If the address is not 64-bit aligned, the latencies of these
4199 // instructions increases by one.
4200 ++Adjust;
4201 break;
4202 }
4203 }
4204 return Adjust;
4205}
4206
4208 const InstrItineraryData *ItinData, const MachineInstr &DefMI,
4209 unsigned DefIdx, const MachineInstr &UseMI, unsigned UseIdx) const {
4210 // No operand latency. The caller may fall back to getInstrLatency.
4211 if (!ItinData || ItinData->isEmpty())
4212 return std::nullopt;
4213
4214 const MachineOperand &DefMO = DefMI.getOperand(DefIdx);
4215 Register Reg = DefMO.getReg();
4216
4217 const MachineInstr *ResolvedDefMI = &DefMI;
4218 unsigned DefAdj = 0;
4219 if (DefMI.isBundle())
4220 ResolvedDefMI =
4221 getBundledDefMI(&getRegisterInfo(), &DefMI, Reg, DefIdx, DefAdj);
4222 if (ResolvedDefMI->isCopyLike() || ResolvedDefMI->isInsertSubreg() ||
4223 ResolvedDefMI->isRegSequence() || ResolvedDefMI->isImplicitDef()) {
4224 return 1;
4225 }
4226
4227 const MachineInstr *ResolvedUseMI = &UseMI;
4228 unsigned UseAdj = 0;
4229 if (UseMI.isBundle()) {
4230 ResolvedUseMI =
4231 getBundledUseMI(&getRegisterInfo(), UseMI, Reg, UseIdx, UseAdj);
4232 if (!ResolvedUseMI)
4233 return std::nullopt;
4234 }
4235
4236 return getOperandLatencyImpl(
4237 ItinData, *ResolvedDefMI, DefIdx, ResolvedDefMI->getDesc(), DefAdj, DefMO,
4238 Reg, *ResolvedUseMI, UseIdx, ResolvedUseMI->getDesc(), UseAdj);
4239}
4240
4241std::optional<unsigned> ARMBaseInstrInfo::getOperandLatencyImpl(
4242 const InstrItineraryData *ItinData, const MachineInstr &DefMI,
4243 unsigned DefIdx, const MCInstrDesc &DefMCID, unsigned DefAdj,
4244 const MachineOperand &DefMO, unsigned Reg, const MachineInstr &UseMI,
4245 unsigned UseIdx, const MCInstrDesc &UseMCID, unsigned UseAdj) const {
4246 if (Reg == ARM::CPSR) {
4247 if (DefMI.getOpcode() == ARM::FMSTAT) {
4248 // fpscr -> cpsr stalls over 20 cycles on A8 (and earlier?)
4249 return Subtarget.isLikeA9() ? 1 : 20;
4250 }
4251
4252 // CPSR set and branch can be paired in the same cycle.
4253 if (UseMI.isBranch())
4254 return 0;
4255
4256 // Otherwise it takes the instruction latency (generally one).
4257 unsigned Latency = getInstrLatency(ItinData, DefMI);
4258
4259 // For Thumb2 and -Os, prefer scheduling CPSR setting instruction close to
4260 // its uses. Instructions which are otherwise scheduled between them may
4261 // incur a code size penalty (not able to use the CPSR setting 16-bit
4262 // instructions).
4263 if (Latency > 0 && Subtarget.isThumb2()) {
4264 const MachineFunction *MF = DefMI.getParent()->getParent();
4265 if (MF->getFunction().hasOptSize())
4266 --Latency;
4267 }
4268 return Latency;
4269 }
4270
4271 if (DefMO.isImplicit() || UseMI.getOperand(UseIdx).isImplicit())
4272 return std::nullopt;
4273
4274 unsigned DefAlign = DefMI.hasOneMemOperand()
4275 ? (*DefMI.memoperands_begin())->getAlign().value()
4276 : 0;
4277 unsigned UseAlign = UseMI.hasOneMemOperand()
4278 ? (*UseMI.memoperands_begin())->getAlign().value()
4279 : 0;
4280
4281 // Get the itinerary's latency if possible, and handle variable_ops.
4282 std::optional<unsigned> Latency = getOperandLatency(
4283 ItinData, DefMCID, DefIdx, DefAlign, UseMCID, UseIdx, UseAlign);
4284 // Unable to find operand latency. The caller may resort to getInstrLatency.
4285 if (!Latency)
4286 return std::nullopt;
4287
4288 // Adjust for IT block position.
4289 int Adj = DefAdj + UseAdj;
4290
4291 // Adjust for dynamic def-side opcode variants not captured by the itinerary.
4292 Adj += adjustDefLatency(Subtarget, DefMI, DefMCID, DefAlign);
4293 if (Adj >= 0 || (int)*Latency > -Adj) {
4294 return *Latency + Adj;
4295 }
4296 // Return the itinerary latency, which may be zero but not less than zero.
4297 return Latency;
4298}
4299
4300std::optional<unsigned>
4302 SDNode *DefNode, unsigned DefIdx,
4303 SDNode *UseNode, unsigned UseIdx) const {
4304 if (!DefNode->isMachineOpcode())
4305 return 1;
4306
4307 const MCInstrDesc &DefMCID = get(DefNode->getMachineOpcode());
4308
4309 if (isZeroCost(DefMCID.Opcode))
4310 return 0;
4311
4312 if (!ItinData || ItinData->isEmpty())
4313 return DefMCID.mayLoad() ? 3 : 1;
4314
4315 if (!UseNode->isMachineOpcode()) {
4316 std::optional<unsigned> Latency =
4317 ItinData->getOperandCycle(DefMCID.getSchedClass(), DefIdx);
4318 int Adj = Subtarget.getPreISelOperandLatencyAdjustment();
4319 int Threshold = 1 + Adj;
4320 return !Latency || Latency <= (unsigned)Threshold ? 1 : *Latency - Adj;
4321 }
4322
4323 const MCInstrDesc &UseMCID = get(UseNode->getMachineOpcode());
4324 auto *DefMN = cast<MachineSDNode>(DefNode);
4325 unsigned DefAlign = !DefMN->memoperands_empty()
4326 ? (*DefMN->memoperands_begin())->getAlign().value()
4327 : 0;
4328 auto *UseMN = cast<MachineSDNode>(UseNode);
4329 unsigned UseAlign = !UseMN->memoperands_empty()
4330 ? (*UseMN->memoperands_begin())->getAlign().value()
4331 : 0;
4332 std::optional<unsigned> Latency = getOperandLatency(
4333 ItinData, DefMCID, DefIdx, DefAlign, UseMCID, UseIdx, UseAlign);
4334 if (!Latency)
4335 return std::nullopt;
4336
4337 if (Latency > 1U &&
4338 (Subtarget.isCortexA8() || Subtarget.isLikeA9() ||
4339 Subtarget.isCortexA7())) {
4340 // FIXME: Shifter op hack: no shift (i.e. [r +/- r]) or [r + r << 2]
4341 // variants are one cycle cheaper.
4342 switch (DefMCID.getOpcode()) {
4343 default: break;
4344 case ARM::LDRrs:
4345 case ARM::LDRBrs: {
4346 unsigned ShOpVal = DefNode->getConstantOperandVal(2);
4347 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
4348 if (ShImm == 0 ||
4349 (ShImm == 2 && ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsl))
4350 Latency = *Latency - 1;
4351 break;
4352 }
4353 case ARM::t2LDRs:
4354 case ARM::t2LDRBs:
4355 case ARM::t2LDRHs:
4356 case ARM::t2LDRSHs: {
4357 // Thumb2 mode: lsl only.
4358 unsigned ShAmt = DefNode->getConstantOperandVal(2);
4359 if (ShAmt == 0 || ShAmt == 2)
4360 Latency = *Latency - 1;
4361 break;
4362 }
4363 }
4364 } else if (DefIdx == 0 && Latency > 2U && Subtarget.isSwift()) {
4365 // FIXME: Properly handle all of the latency adjustments for address
4366 // writeback.
4367 switch (DefMCID.getOpcode()) {
4368 default: break;
4369 case ARM::LDRrs:
4370 case ARM::LDRBrs: {
4371 unsigned ShOpVal = DefNode->getConstantOperandVal(2);
4372 unsigned ShImm = ARM_AM::getAM2Offset(ShOpVal);
4373 if (ShImm == 0 ||
4374 ((ShImm == 1 || ShImm == 2 || ShImm == 3) &&
4376 Latency = *Latency - 2;
4377 else if (ShImm == 1 && ARM_AM::getAM2ShiftOpc(ShOpVal) == ARM_AM::lsr)
4378 Latency = *Latency - 1;
4379 break;
4380 }
4381 case ARM::t2LDRs:
4382 case ARM::t2LDRBs:
4383 case ARM::t2LDRHs:
4384 case ARM::t2LDRSHs:
4385 // Thumb2 mode: lsl 0-3 only.
4386 Latency = *Latency - 2;
4387 break;
4388 }
4389 }
4390
4391 if (DefAlign < 8 && Subtarget.checkVLDnAccessAlignment())
4392 switch (DefMCID.getOpcode()) {
4393 default: break;
4394 case ARM::VLD1q8:
4395 case ARM::VLD1q16:
4396 case ARM::VLD1q32:
4397 case ARM::VLD1q64:
4398 case ARM::VLD1q8wb_register:
4399 case ARM::VLD1q16wb_register:
4400 case ARM::VLD1q32wb_register:
4401 case ARM::VLD1q64wb_register:
4402 case ARM::VLD1q8wb_fixed:
4403 case ARM::VLD1q16wb_fixed:
4404 case ARM::VLD1q32wb_fixed:
4405 case ARM::VLD1q64wb_fixed:
4406 case ARM::VLD2d8:
4407 case ARM::VLD2d16:
4408 case ARM::VLD2d32:
4409 case ARM::VLD2q8Pseudo:
4410 case ARM::VLD2q16Pseudo:
4411 case ARM::VLD2q32Pseudo:
4412 case ARM::VLD2d8wb_fixed:
4413 case ARM::VLD2d16wb_fixed:
4414 case ARM::VLD2d32wb_fixed:
4415 case ARM::VLD2q8PseudoWB_fixed:
4416 case ARM::VLD2q16PseudoWB_fixed:
4417 case ARM::VLD2q32PseudoWB_fixed:
4418 case ARM::VLD2d8wb_register:
4419 case ARM::VLD2d16wb_register:
4420 case ARM::VLD2d32wb_register:
4421 case ARM::VLD2q8PseudoWB_register:
4422 case ARM::VLD2q16PseudoWB_register:
4423 case ARM::VLD2q32PseudoWB_register:
4424 case ARM::VLD3d8Pseudo:
4425 case ARM::VLD3d16Pseudo:
4426 case ARM::VLD3d32Pseudo:
4427 case ARM::VLD1d8TPseudo:
4428 case ARM::VLD1d16TPseudo:
4429 case ARM::VLD1d32TPseudo:
4430 case ARM::VLD1d64TPseudo:
4431 case ARM::VLD1d64TPseudoWB_fixed:
4432 case ARM::VLD1d64TPseudoWB_register:
4433 case ARM::VLD3d8Pseudo_UPD:
4434 case ARM::VLD3d16Pseudo_UPD:
4435 case ARM::VLD3d32Pseudo_UPD:
4436 case ARM::VLD3q8Pseudo_UPD:
4437 case ARM::VLD3q16Pseudo_UPD:
4438 case ARM::VLD3q32Pseudo_UPD:
4439 case ARM::VLD3q8oddPseudo:
4440 case ARM::VLD3q16oddPseudo:
4441 case ARM::VLD3q32oddPseudo:
4442 case ARM::VLD3q8oddPseudo_UPD:
4443 case ARM::VLD3q16oddPseudo_UPD:
4444 case ARM::VLD3q32oddPseudo_UPD:
4445 case ARM::VLD4d8Pseudo:
4446 case ARM::VLD4d16Pseudo:
4447 case ARM::VLD4d32Pseudo:
4448 case ARM::VLD1d8QPseudo:
4449 case ARM::VLD1d16QPseudo:
4450 case ARM::VLD1d32QPseudo:
4451 case ARM::VLD1d64QPseudo:
4452 case ARM::VLD1d64QPseudoWB_fixed:
4453 case ARM::VLD1d64QPseudoWB_register:
4454 case ARM::VLD1q8HighQPseudo:
4455 case ARM::VLD1q8LowQPseudo_UPD:
4456 case ARM::VLD1q8HighTPseudo:
4457 case ARM::VLD1q8LowTPseudo_UPD:
4458 case ARM::VLD1q16HighQPseudo:
4459 case ARM::VLD1q16LowQPseudo_UPD:
4460 case ARM::VLD1q16HighTPseudo:
4461 case ARM::VLD1q16LowTPseudo_UPD:
4462 case ARM::VLD1q32HighQPseudo:
4463 case ARM::VLD1q32LowQPseudo_UPD:
4464 case ARM::VLD1q32HighTPseudo:
4465 case ARM::VLD1q32LowTPseudo_UPD:
4466 case ARM::VLD1q64HighQPseudo:
4467 case ARM::VLD1q64LowQPseudo_UPD:
4468 case ARM::VLD1q64HighTPseudo:
4469 case ARM::VLD1q64LowTPseudo_UPD:
4470 case ARM::VLD4d8Pseudo_UPD:
4471 case ARM::VLD4d16Pseudo_UPD:
4472 case ARM::VLD4d32Pseudo_UPD:
4473 case ARM::VLD4q8Pseudo_UPD:
4474 case ARM::VLD4q16Pseudo_UPD:
4475 case ARM::VLD4q32Pseudo_UPD:
4476 case ARM::VLD4q8oddPseudo:
4477 case ARM::VLD4q16oddPseudo:
4478 case ARM::VLD4q32oddPseudo:
4479 case ARM::VLD4q8oddPseudo_UPD:
4480 case ARM::VLD4q16oddPseudo_UPD:
4481 case ARM::VLD4q32oddPseudo_UPD:
4482 case ARM::VLD1DUPq8:
4483 case ARM::VLD1DUPq16:
4484 case ARM::VLD1DUPq32:
4485 case ARM::VLD1DUPq8wb_fixed:
4486 case ARM::VLD1DUPq16wb_fixed:
4487 case ARM::VLD1DUPq32wb_fixed:
4488 case ARM::VLD1DUPq8wb_register:
4489 case ARM::VLD1DUPq16wb_register:
4490 case ARM::VLD1DUPq32wb_register:
4491 case ARM::VLD2DUPd8:
4492 case ARM::VLD2DUPd16:
4493 case ARM::VLD2DUPd32:
4494 case ARM::VLD2DUPd8wb_fixed:
4495 case ARM::VLD2DUPd16wb_fixed:
4496 case ARM::VLD2DUPd32wb_fixed:
4497 case ARM::VLD2DUPd8wb_register:
4498 case ARM::VLD2DUPd16wb_register:
4499 case ARM::VLD2DUPd32wb_register:
4500 case ARM::VLD2DUPq8EvenPseudo:
4501 case ARM::VLD2DUPq8OddPseudo:
4502 case ARM::VLD2DUPq16EvenPseudo:
4503 case ARM::VLD2DUPq16OddPseudo:
4504 case ARM::VLD2DUPq32EvenPseudo:
4505 case ARM::VLD2DUPq32OddPseudo:
4506 case ARM::VLD3DUPq8EvenPseudo:
4507 case ARM::VLD3DUPq8OddPseudo:
4508 case ARM::VLD3DUPq16EvenPseudo:
4509 case ARM::VLD3DUPq16OddPseudo:
4510 case ARM::VLD3DUPq32EvenPseudo:
4511 case ARM::VLD3DUPq32OddPseudo:
4512 case ARM::VLD4DUPd8Pseudo:
4513 case ARM::VLD4DUPd16Pseudo:
4514 case ARM::VLD4DUPd32Pseudo:
4515 case ARM::VLD4DUPd8Pseudo_UPD:
4516 case ARM::VLD4DUPd16Pseudo_UPD:
4517 case ARM::VLD4DUPd32Pseudo_UPD:
4518 case ARM::VLD4DUPq8EvenPseudo:
4519 case ARM::VLD4DUPq8OddPseudo:
4520 case ARM::VLD4DUPq16EvenPseudo:
4521 case ARM::VLD4DUPq16OddPseudo:
4522 case ARM::VLD4DUPq32EvenPseudo:
4523 case ARM::VLD4DUPq32OddPseudo:
4524 case ARM::VLD1LNq8Pseudo:
4525 case ARM::VLD1LNq16Pseudo:
4526 case ARM::VLD1LNq32Pseudo:
4527 case ARM::VLD1LNq8Pseudo_UPD:
4528 case ARM::VLD1LNq16Pseudo_UPD:
4529 case ARM::VLD1LNq32Pseudo_UPD:
4530 case ARM::VLD2LNd8Pseudo:
4531 case ARM::VLD2LNd16Pseudo:
4532 case ARM::VLD2LNd32Pseudo:
4533 case ARM::VLD2LNq16Pseudo:
4534 case ARM::VLD2LNq32Pseudo:
4535 case ARM::VLD2LNd8Pseudo_UPD:
4536 case ARM::VLD2LNd16Pseudo_UPD:
4537 case ARM::VLD2LNd32Pseudo_UPD:
4538 case ARM::VLD2LNq16Pseudo_UPD:
4539 case ARM::VLD2LNq32Pseudo_UPD:
4540 case ARM::VLD4LNd8Pseudo:
4541 case ARM::VLD4LNd16Pseudo:
4542 case ARM::VLD4LNd32Pseudo:
4543 case ARM::VLD4LNq16Pseudo:
4544 case ARM::VLD4LNq32Pseudo:
4545 case ARM::VLD4LNd8Pseudo_UPD:
4546 case ARM::VLD4LNd16Pseudo_UPD:
4547 case ARM::VLD4LNd32Pseudo_UPD:
4548 case ARM::VLD4LNq16Pseudo_UPD:
4549 case ARM::VLD4LNq32Pseudo_UPD:
4550 // If the address is not 64-bit aligned, the latencies of these
4551 // instructions increases by one.
4552 Latency = *Latency + 1;
4553 break;
4554 }
4555
4556 return Latency;
4557}
4558
4559unsigned ARMBaseInstrInfo::getPredicationCost(const MachineInstr &MI) const {
4560 if (MI.isCopyLike() || MI.isInsertSubreg() || MI.isRegSequence() ||
4561 MI.isImplicitDef())
4562 return 0;
4563
4564 if (MI.isBundle())
4565 return 0;
4566
4567 const MCInstrDesc &MCID = MI.getDesc();
4568
4569 if (MCID.isCall() || (MCID.hasImplicitDefOfPhysReg(ARM::CPSR) &&
4570 !Subtarget.cheapPredicableCPSRDef())) {
4571 // When predicated, CPSR is an additional source operand for CPSR updating
4572 // instructions, this apparently increases their latencies.
4573 return 1;
4574 }
4575 return 0;
4576}
4577
4578unsigned ARMBaseInstrInfo::getInstrLatency(const InstrItineraryData *ItinData,
4579 const MachineInstr &MI,
4580 unsigned *PredCost) const {
4581 if (MI.isCopyLike() || MI.isInsertSubreg() || MI.isRegSequence() ||
4582 MI.isImplicitDef())
4583 return 1;
4584
4585 // An instruction scheduler typically runs on unbundled instructions, however
4586 // other passes may query the latency of a bundled instruction.
4587 if (MI.isBundle()) {
4588 unsigned Latency = 0;
4590 MachineBasicBlock::const_instr_iterator E = MI.getParent()->instr_end();
4591 while (++I != E && I->isInsideBundle()) {
4592 if (I->getOpcode() != ARM::t2IT)
4593 Latency += getInstrLatency(ItinData, *I, PredCost);
4594 }
4595 return Latency;
4596 }
4597
4598 const MCInstrDesc &MCID = MI.getDesc();
4599 if (PredCost && (MCID.isCall() || (MCID.hasImplicitDefOfPhysReg(ARM::CPSR) &&
4600 !Subtarget.cheapPredicableCPSRDef()))) {
4601 // When predicated, CPSR is an additional source operand for CPSR updating
4602 // instructions, this apparently increases their latencies.
4603 *PredCost = 1;
4604 }
4605 // Be sure to call getStageLatency for an empty itinerary in case it has a
4606 // valid MinLatency property.
4607 if (!ItinData)
4608 return MI.mayLoad() ? 3 : 1;
4609
4610 unsigned Class = MCID.getSchedClass();
4611
4612 // For instructions with variable uops, use uops as latency.
4613 if (!ItinData->isEmpty() && ItinData->getNumMicroOps(Class) < 0)
4614 return getNumMicroOps(ItinData, MI);
4615
4616 // For the common case, fall back on the itinerary's latency.
4617 unsigned Latency = ItinData->getStageLatency(Class);
4618
4619 // Adjust for dynamic def-side opcode variants not captured by the itinerary.
4620 unsigned DefAlign =
4621 MI.hasOneMemOperand() ? (*MI.memoperands_begin())->getAlign().value() : 0;
4622 int Adj = adjustDefLatency(Subtarget, MI, MCID, DefAlign);
4623 if (Adj >= 0 || (int)Latency > -Adj) {
4624 return Latency + Adj;
4625 }
4626 return Latency;
4627}
4628
4629unsigned ARMBaseInstrInfo::getInstrLatency(const InstrItineraryData *ItinData,
4630 SDNode *Node) const {
4631 if (!Node->isMachineOpcode())
4632 return 1;
4633
4634 if (!ItinData || ItinData->isEmpty())
4635 return 1;
4636
4637 unsigned Opcode = Node->getMachineOpcode();
4638 switch (Opcode) {
4639 default:
4640 return ItinData->getStageLatency(get(Opcode).getSchedClass());
4641 case ARM::VLDMQIA:
4642 case ARM::VSTMQIA:
4643 return 2;
4644 }
4645}
4646
4647bool ARMBaseInstrInfo::hasHighOperandLatency(const TargetSchedModel &SchedModel,
4648 const MachineRegisterInfo *MRI,
4649 const MachineInstr &DefMI,
4650 unsigned DefIdx,
4651 const MachineInstr &UseMI,
4652 unsigned UseIdx) const {
4653 unsigned DDomain = DefMI.getDesc().TSFlags & ARMII::DomainMask;
4654 unsigned UDomain = UseMI.getDesc().TSFlags & ARMII::DomainMask;
4655 if (Subtarget.nonpipelinedVFP() &&
4656 (DDomain == ARMII::DomainVFP || UDomain == ARMII::DomainVFP))
4657 return true;
4658
4659 // Hoist VFP / NEON instructions with 4 or higher latency.
4660 unsigned Latency =
4661 SchedModel.computeOperandLatency(&DefMI, DefIdx, &UseMI, UseIdx);
4662 if (Latency <= 3)
4663 return false;
4664 return DDomain == ARMII::DomainVFP || DDomain == ARMII::DomainNEON ||
4665 UDomain == ARMII::DomainVFP || UDomain == ARMII::DomainNEON;
4666}
4667
4668bool ARMBaseInstrInfo::hasLowDefLatency(const TargetSchedModel &SchedModel,
4669 const MachineInstr &DefMI,
4670 unsigned DefIdx) const {
4671 const InstrItineraryData *ItinData = SchedModel.getInstrItineraries();
4672 if (!ItinData || ItinData->isEmpty())
4673 return false;
4674
4675 unsigned DDomain = DefMI.getDesc().TSFlags & ARMII::DomainMask;
4676 if (DDomain == ARMII::DomainGeneral) {
4677 unsigned DefClass = DefMI.getDesc().getSchedClass();
4678 std::optional<unsigned> DefCycle =
4679 ItinData->getOperandCycle(DefClass, DefIdx);
4680 return DefCycle && DefCycle <= 2U;
4681 }
4682 return false;
4683}
4684
4685bool ARMBaseInstrInfo::verifyInstruction(const MachineInstr &MI,
4686 StringRef &ErrInfo) const {
4687 if (convertAddSubFlagsOpcode(MI.getOpcode())) {
4688 ErrInfo = "Pseudo flag setting opcodes only exist in Selection DAG";
4689 return false;
4690 }
4691 if (MI.getOpcode() == ARM::tMOVr && !Subtarget.hasV6Ops()) {
4692 // Make sure we don't generate a lo-lo mov that isn't supported.
4693 if (!ARM::hGPRRegClass.contains(MI.getOperand(0).getReg()) &&
4694 !ARM::hGPRRegClass.contains(MI.getOperand(1).getReg())) {
4695 ErrInfo = "Non-flag-setting Thumb1 mov is v6-only";
4696 return false;
4697 }
4698 }
4699 if (MI.getOpcode() == ARM::tPUSH ||
4700 MI.getOpcode() == ARM::tPOP ||
4701 MI.getOpcode() == ARM::tPOP_RET) {
4702 for (const MachineOperand &MO : llvm::drop_begin(MI.operands(), 2)) {
4703 if (MO.isImplicit() || !MO.isReg())
4704 continue;
4705 Register Reg = MO.getReg();
4706 if (Reg < ARM::R0 || Reg > ARM::R7) {
4707 if (!(MI.getOpcode() == ARM::tPUSH && Reg == ARM::LR) &&
4708 !(MI.getOpcode() == ARM::tPOP_RET && Reg == ARM::PC)) {
4709 ErrInfo = "Unsupported register in Thumb1 push/pop";
4710 return false;
4711 }
4712 }
4713 }
4714 }
4715 if (MI.getOpcode() == ARM::MVE_VMOV_q_rr) {
4716 assert(MI.getOperand(4).isImm() && MI.getOperand(5).isImm());
4717 if ((MI.getOperand(4).getImm() != 2 && MI.getOperand(4).getImm() != 3) ||
4718 MI.getOperand(4).getImm() != MI.getOperand(5).getImm() + 2) {
4719 ErrInfo = "Incorrect array index for MVE_VMOV_q_rr";
4720 return false;
4721 }
4722 }
4723
4724 // Check the address model by taking the first Imm operand and checking it is
4725 // legal for that addressing mode.
4727 (ARMII::AddrMode)(MI.getDesc().TSFlags & ARMII::AddrModeMask);
4728 switch (AddrMode) {
4729 default:
4730 break;
4738 case ARMII::AddrModeT2_i12: {
4739 uint32_t Imm = 0;
4740 for (auto Op : MI.operands()) {
4741 if (Op.isImm()) {
4742 Imm = Op.getImm();
4743 break;
4744 }
4745 }
4746 if (!isLegalAddressImm(MI.getOpcode(), Imm, this)) {
4747 ErrInfo = "Incorrect AddrMode Imm for instruction";
4748 return false;
4749 }
4750 break;
4751 }
4752 }
4753 return true;
4754}
4755
4757 unsigned LoadImmOpc,
4758 unsigned LoadOpc) const {
4759 assert(!Subtarget.isROPI() && !Subtarget.isRWPI() &&
4760 "ROPI/RWPI not currently supported with stack guard");
4761
4762 MachineBasicBlock &MBB = *MI->getParent();
4763 DebugLoc DL = MI->getDebugLoc();
4764 Register Reg = MI->getOperand(0).getReg();
4766 unsigned int Offset = 0;
4767
4768 if (LoadImmOpc == ARM::MRC || LoadImmOpc == ARM::t2MRC) {
4769 assert(!Subtarget.isReadTPSoft() &&
4770 "TLS stack protector requires hardware TLS register");
4771
4772 BuildMI(MBB, MI, DL, get(LoadImmOpc), Reg)
4773 .addImm(15)
4774 .addImm(0)
4775 .addImm(13)
4776 .addImm(0)
4777 .addImm(3)
4779
4780 Module &M = *MBB.getParent()->getFunction().getParent();
4781 Offset = M.getStackProtectorGuardOffset();
4782 if (Offset & ~0xfffU) {
4783 // The offset won't fit in the LDR's 12-bit immediate field, so emit an
4784 // extra ADD to cover the delta. This gives us a guaranteed 8 additional
4785 // bits, resulting in a range of 0 to +1 MiB for the guard offset.
4786 unsigned AddOpc = (LoadImmOpc == ARM::MRC) ? ARM::ADDri : ARM::t2ADDri;
4787 BuildMI(MBB, MI, DL, get(AddOpc), Reg)
4788 .addReg(Reg, RegState::Kill)
4789 .addImm(Offset & ~0xfffU)
4791 .addReg(0);
4792 Offset &= 0xfffU;
4793 }
4794 } else {
4795 const GlobalValue *GV =
4796 cast<GlobalValue>((*MI->memoperands_begin())->getValue());
4797 bool IsIndirect = Subtarget.isGVIndirectSymbol(GV);
4798
4799 unsigned TargetFlags = ARMII::MO_NO_FLAG;
4800 if (Subtarget.isTargetMachO()) {
4801 TargetFlags |= ARMII::MO_NONLAZY;
4802 } else if (Subtarget.isTargetCOFF()) {
4803 if (GV->hasDLLImportStorageClass())
4804 TargetFlags |= ARMII::MO_DLLIMPORT;
4805 else if (IsIndirect)
4806 TargetFlags |= ARMII::MO_COFFSTUB;
4807 } else if (IsIndirect) {
4808 TargetFlags |= ARMII::MO_GOT;
4809 }
4810
4811 if (LoadImmOpc == ARM::tMOVi32imm) { // Thumb-1 execute-only
4812 Register CPSRSaveReg = ARM::R12; // Use R12 as scratch register
4813 auto APSREncoding =
4814 ARMSysReg::lookupMClassSysRegByName("apsr_nzcvq")->Encoding;
4815 BuildMI(MBB, MI, DL, get(ARM::t2MRS_M), CPSRSaveReg)
4816 .addImm(APSREncoding)
4818 BuildMI(MBB, MI, DL, get(LoadImmOpc), Reg)
4819 .addGlobalAddress(GV, 0, TargetFlags);
4820 BuildMI(MBB, MI, DL, get(ARM::t2MSR_M))
4821 .addImm(APSREncoding)
4822 .addReg(CPSRSaveReg, RegState::Kill)
4824 } else {
4825 BuildMI(MBB, MI, DL, get(LoadImmOpc), Reg)
4826 .addGlobalAddress(GV, 0, TargetFlags);
4827 }
4828
4829 if (IsIndirect) {
4830 MIB = BuildMI(MBB, MI, DL, get(LoadOpc), Reg);
4831 MIB.addReg(Reg, RegState::Kill).addImm(0);
4832 auto Flags = MachineMemOperand::MOLoad |
4835 MachineMemOperand *MMO = MBB.getParent()->getMachineMemOperand(
4836 MachinePointerInfo::getGOT(*MBB.getParent()), Flags, 4, Align(4));
4838 }
4839 }
4840
4841 MIB = BuildMI(MBB, MI, DL, get(LoadOpc), Reg);
4842 MIB.addReg(Reg, RegState::Kill)
4843 .addImm(Offset)
4844 .cloneMemRefs(*MI)
4846}
4847
4848bool
4849ARMBaseInstrInfo::isFpMLxInstruction(unsigned Opcode, unsigned &MulOpc,
4850 unsigned &AddSubOpc,
4851 bool &NegAcc, bool &HasLane) const {
4852 auto I = MLxEntryMap.find(Opcode);
4853 if (I == MLxEntryMap.end())
4854 return false;
4855
4856 const ARM_MLxEntry &Entry = ARM_MLxTable[I->second];
4857 MulOpc = Entry.MulOpc;
4858 AddSubOpc = Entry.AddSubOpc;
4859 NegAcc = Entry.NegAcc;
4860 HasLane = Entry.HasLane;
4861 return true;
4862}
4863
4864//===----------------------------------------------------------------------===//
4865// Execution domains.
4866//===----------------------------------------------------------------------===//
4867//
4868// Some instructions go down the NEON pipeline, some go down the VFP pipeline,
4869// and some can go down both. The vmov instructions go down the VFP pipeline,
4870// but they can be changed to vorr equivalents that are executed by the NEON
4871// pipeline.
4872//
4873// We use the following execution domain numbering:
4874//
4880
4881//
4882// Also see ARMInstrFormats.td and Domain* enums in ARMBaseInfo.h
4883//
4884std::pair<uint16_t, uint16_t>
4886 // If we don't have access to NEON instructions then we won't be able
4887 // to swizzle anything to the NEON domain. Check to make sure.
4888 if (Subtarget.hasNEON()) {
4889 // VMOVD, VMOVRS and VMOVSR are VFP instructions, but can be changed to NEON
4890 // if they are not predicated.
4891 if (MI.getOpcode() == ARM::VMOVD && !isPredicated(MI))
4892 return std::make_pair(ExeVFP, (1 << ExeVFP) | (1 << ExeNEON));
4893
4894 // CortexA9 is particularly picky about mixing the two and wants these
4895 // converted.
4896 if (Subtarget.useNEONForFPMovs() && !isPredicated(MI) &&
4897 (MI.getOpcode() == ARM::VMOVRS || MI.getOpcode() == ARM::VMOVSR ||
4898 MI.getOpcode() == ARM::VMOVS))
4899 return std::make_pair(ExeVFP, (1 << ExeVFP) | (1 << ExeNEON));
4900 }
4901 // No other instructions can be swizzled, so just determine their domain.
4902 unsigned Domain = MI.getDesc().TSFlags & ARMII::DomainMask;
4903
4905 return std::make_pair(ExeNEON, 0);
4906
4907 // Certain instructions can go either way on Cortex-A8.
4908 // Treat them as NEON instructions.
4909 if ((Domain & ARMII::DomainNEONA8) && Subtarget.isCortexA8())
4910 return std::make_pair(ExeNEON, 0);
4911
4913 return std::make_pair(ExeVFP, 0);
4914
4915 return std::make_pair(ExeGeneric, 0);
4916}
4917
4919 unsigned SReg, unsigned &Lane) {
4920 MCRegister DReg =
4921 TRI->getMatchingSuperReg(SReg, ARM::ssub_0, &ARM::DPRRegClass);
4922 Lane = 0;
4923
4924 if (DReg)
4925 return DReg;
4926
4927 Lane = 1;
4928 DReg = TRI->getMatchingSuperReg(SReg, ARM::ssub_1, &ARM::DPRRegClass);
4929
4930 assert(DReg && "S-register with no D super-register?");
4931 return DReg;
4932}
4933
4934/// getImplicitSPRUseForDPRUse - Given a use of a DPR register and lane,
4935/// set ImplicitSReg to a register number that must be marked as implicit-use or
4936/// zero if no register needs to be defined as implicit-use.
4937///
4938/// If the function cannot determine if an SPR should be marked implicit use or
4939/// not, it returns false.
4940///
4941/// This function handles cases where an instruction is being modified from taking
4942/// an SPR to a DPR[Lane]. A use of the DPR is being added, which may conflict
4943/// with an earlier def of an SPR corresponding to DPR[Lane^1] (i.e. the other
4944/// lane of the DPR).
4945///
4946/// If the other SPR is defined, an implicit-use of it should be added. Else,
4947/// (including the case where the DPR itself is defined), it should not.
4948///
4950 MachineInstr &MI, MCRegister DReg,
4951 unsigned Lane,
4952 MCRegister &ImplicitSReg) {
4953 // If the DPR is defined or used already, the other SPR lane will be chained
4954 // correctly, so there is nothing to be done.
4955 if (MI.definesRegister(DReg, TRI) || MI.readsRegister(DReg, TRI)) {
4956 ImplicitSReg = MCRegister();
4957 return true;
4958 }
4959
4960 // Otherwise we need to go searching to see if the SPR is set explicitly.
4961 ImplicitSReg = TRI->getSubReg(DReg,
4962 (Lane & 1) ? ARM::ssub_0 : ARM::ssub_1);
4964 MI.getParent()->computeRegisterLiveness(TRI, ImplicitSReg, MI);
4965
4966 if (LQR == MachineBasicBlock::LQR_Live)
4967 return true;
4968 else if (LQR == MachineBasicBlock::LQR_Unknown)
4969 return false;
4970
4971 // If the register is known not to be live, there is no need to add an
4972 // implicit-use.
4973 ImplicitSReg = MCRegister();
4974 return true;
4975}
4976
4978 unsigned Domain) const {
4979 unsigned DstReg, SrcReg;
4980 MCRegister DReg;
4981 unsigned Lane;
4982 MachineInstrBuilder MIB(*MI.getParent()->getParent(), MI);
4984 switch (MI.getOpcode()) {
4985 default:
4986 llvm_unreachable("cannot handle opcode!");
4987 break;
4988 case ARM::VMOVD:
4989 if (Domain != ExeNEON)
4990 break;
4991
4992 // Zap the predicate operands.
4993 assert(!isPredicated(MI) && "Cannot predicate a VORRd");
4994
4995 // Make sure we've got NEON instructions.
4996 assert(Subtarget.hasNEON() && "VORRd requires NEON");
4997
4998 // Source instruction is %DDst = VMOVD %DSrc, 14, %noreg (; implicits)
4999 DstReg = MI.getOperand(0).getReg();
5000 SrcReg = MI.getOperand(1).getReg();
5001
5002 for (unsigned i = MI.getDesc().getNumOperands(); i; --i)
5003 MI.removeOperand(i - 1);
5004
5005 // Change to a %DDst = VORRd %DSrc, %DSrc, 14, %noreg (; implicits)
5006 MI.setDesc(get(ARM::VORRd));
5007 MIB.addReg(DstReg, RegState::Define)
5008 .addReg(SrcReg)
5009 .addReg(SrcReg)
5011 break;
5012 case ARM::VMOVRS:
5013 if (Domain != ExeNEON)
5014 break;
5015 assert(!isPredicated(MI) && "Cannot predicate a VGETLN");
5016
5017 // Source instruction is %RDst = VMOVRS %SSrc, 14, %noreg (; implicits)
5018 DstReg = MI.getOperand(0).getReg();
5019 SrcReg = MI.getOperand(1).getReg();
5020
5021 for (unsigned i = MI.getDesc().getNumOperands(); i; --i)
5022 MI.removeOperand(i - 1);
5023
5024 DReg = getCorrespondingDRegAndLane(TRI, SrcReg, Lane);
5025
5026 // Convert to %RDst = VGETLNi32 %DSrc, Lane, 14, %noreg (; imps)
5027 // Note that DSrc has been widened and the other lane may be undef, which
5028 // contaminates the entire register.
5029 MI.setDesc(get(ARM::VGETLNi32));
5030 MIB.addReg(DstReg, RegState::Define)
5031 .addReg(DReg, RegState::Undef)
5032 .addImm(Lane)
5034
5035 // The old source should be an implicit use, otherwise we might think it
5036 // was dead before here.
5037 MIB.addReg(SrcReg, RegState::Implicit);
5038 break;
5039 case ARM::VMOVSR: {
5040 if (Domain != ExeNEON)
5041 break;
5042 assert(!isPredicated(MI) && "Cannot predicate a VSETLN");
5043
5044 // Source instruction is %SDst = VMOVSR %RSrc, 14, %noreg (; implicits)
5045 DstReg = MI.getOperand(0).getReg();
5046 SrcReg = MI.getOperand(1).getReg();
5047
5048 DReg = getCorrespondingDRegAndLane(TRI, DstReg, Lane);
5049
5050 MCRegister ImplicitSReg;
5051 if (!getImplicitSPRUseForDPRUse(TRI, MI, DReg, Lane, ImplicitSReg))
5052 break;
5053
5054 for (unsigned i = MI.getDesc().getNumOperands(); i; --i)
5055 MI.removeOperand(i - 1);
5056
5057 // Convert to %DDst = VSETLNi32 %DDst, %RSrc, Lane, 14, %noreg (; imps)
5058 // Again DDst may be undefined at the beginning of this instruction.
5059 MI.setDesc(get(ARM::VSETLNi32));
5060 MIB.addReg(DReg, RegState::Define)
5061 .addReg(DReg, getUndefRegState(!MI.readsRegister(DReg, TRI)))
5062 .addReg(SrcReg)
5063 .addImm(Lane)
5065
5066 // The narrower destination must be marked as set to keep previous chains
5067 // in place.
5069 if (ImplicitSReg)
5070 MIB.addReg(ImplicitSReg, RegState::Implicit);
5071 break;
5072 }
5073 case ARM::VMOVS: {
5074 if (Domain != ExeNEON)
5075 break;
5076
5077 // Source instruction is %SDst = VMOVS %SSrc, 14, %noreg (; implicits)
5078 DstReg = MI.getOperand(0).getReg();
5079 SrcReg = MI.getOperand(1).getReg();
5080
5081 unsigned DstLane = 0, SrcLane = 0;
5082 MCRegister DDst, DSrc;
5083 DDst = getCorrespondingDRegAndLane(TRI, DstReg, DstLane);
5084 DSrc = getCorrespondingDRegAndLane(TRI, SrcReg, SrcLane);
5085
5086 MCRegister ImplicitSReg;
5087 if (!getImplicitSPRUseForDPRUse(TRI, MI, DSrc, SrcLane, ImplicitSReg))
5088 break;
5089
5090 for (unsigned i = MI.getDesc().getNumOperands(); i; --i)
5091 MI.removeOperand(i - 1);
5092
5093 if (DSrc == DDst) {
5094 // Destination can be:
5095 // %DDst = VDUPLN32d %DDst, Lane, 14, %noreg (; implicits)
5096 MI.setDesc(get(ARM::VDUPLN32d));
5097 MIB.addReg(DDst, RegState::Define)
5098 .addReg(DDst, getUndefRegState(!MI.readsRegister(DDst, TRI)))
5099 .addImm(SrcLane)
5101
5102 // Neither the source or the destination are naturally represented any
5103 // more, so add them in manually.
5105 MIB.addReg(SrcReg, RegState::Implicit);
5106 if (ImplicitSReg)
5107 MIB.addReg(ImplicitSReg, RegState::Implicit);
5108 break;
5109 }
5110
5111 // In general there's no single instruction that can perform an S <-> S
5112 // move in NEON space, but a pair of VEXT instructions *can* do the
5113 // job. It turns out that the VEXTs needed will only use DSrc once, with
5114 // the position based purely on the combination of lane-0 and lane-1
5115 // involved. For example
5116 // vmov s0, s2 -> vext.32 d0, d0, d1, #1 vext.32 d0, d0, d0, #1
5117 // vmov s1, s3 -> vext.32 d0, d1, d0, #1 vext.32 d0, d0, d0, #1
5118 // vmov s0, s3 -> vext.32 d0, d0, d0, #1 vext.32 d0, d1, d0, #1
5119 // vmov s1, s2 -> vext.32 d0, d0, d0, #1 vext.32 d0, d0, d1, #1
5120 //
5121 // Pattern of the MachineInstrs is:
5122 // %DDst = VEXTd32 %DSrc1, %DSrc2, Lane, 14, %noreg (;implicits)
5123 MachineInstrBuilder NewMIB;
5124 NewMIB = BuildMI(*MI.getParent(), MI, MI.getDebugLoc(), get(ARM::VEXTd32),
5125 DDst);
5126
5127 // On the first instruction, both DSrc and DDst may be undef if present.
5128 // Specifically when the original instruction didn't have them as an
5129 // <imp-use>.
5130 MCRegister CurReg = SrcLane == 1 && DstLane == 1 ? DSrc : DDst;
5131 bool CurUndef = !MI.readsRegister(CurReg, TRI);
5132 NewMIB.addReg(CurReg, getUndefRegState(CurUndef));
5133
5134 CurReg = SrcLane == 0 && DstLane == 0 ? DSrc : DDst;
5135 CurUndef = !MI.readsRegister(CurReg, TRI);
5136 NewMIB.addReg(CurReg, getUndefRegState(CurUndef))
5137 .addImm(1)
5139
5140 if (SrcLane == DstLane)
5141 NewMIB.addReg(SrcReg, RegState::Implicit);
5142
5143 MI.setDesc(get(ARM::VEXTd32));
5144 MIB.addReg(DDst, RegState::Define);
5145
5146 // On the second instruction, DDst has definitely been defined above, so
5147 // it is not undef. DSrc, if present, can be undef as above.
5148 CurReg = SrcLane == 1 && DstLane == 0 ? DSrc : DDst;
5149 CurUndef = CurReg == DSrc && !MI.readsRegister(CurReg, TRI);
5150 MIB.addReg(CurReg, getUndefRegState(CurUndef));
5151
5152 CurReg = SrcLane == 0 && DstLane == 1 ? DSrc : DDst;
5153 CurUndef = CurReg == DSrc && !MI.readsRegister(CurReg, TRI);
5154 MIB.addReg(CurReg, getUndefRegState(CurUndef))
5155 .addImm(1)
5157
5158 if (SrcLane != DstLane)
5159 MIB.addReg(SrcReg, RegState::Implicit);
5160
5161 // As before, the original destination is no longer represented, add it
5162 // implicitly.
5164 if (ImplicitSReg != 0)
5165 MIB.addReg(ImplicitSReg, RegState::Implicit);
5166 break;
5167 }
5168 }
5169}
5170
5171//===----------------------------------------------------------------------===//
5172// Partial register updates
5173//===----------------------------------------------------------------------===//
5174//
5175// Swift renames NEON registers with 64-bit granularity. That means any
5176// instruction writing an S-reg implicitly reads the containing D-reg. The
5177// problem is mostly avoided by translating f32 operations to v2f32 operations
5178// on D-registers, but f32 loads are still a problem.
5179//
5180// These instructions can load an f32 into a NEON register:
5181//
5182// VLDRS - Only writes S, partial D update.
5183// VLD1LNd32 - Writes all D-regs, explicit partial D update, 2 uops.
5184// VLD1DUPd32 - Writes all D-regs, no partial reg update, 2 uops.
5185//
5186// FCONSTD can be used as a dependency-breaking instruction.
5188 const MachineInstr &MI, unsigned OpNum,
5189 const TargetRegisterInfo *TRI) const {
5190 auto PartialUpdateClearance = Subtarget.getPartialUpdateClearance();
5191 if (!PartialUpdateClearance)
5192 return 0;
5193
5194 assert(TRI && "Need TRI instance");
5195
5196 const MachineOperand &MO = MI.getOperand(OpNum);
5197 if (MO.readsReg())
5198 return 0;
5199 Register Reg = MO.getReg();
5200 int UseOp = -1;
5201
5202 switch (MI.getOpcode()) {
5203 // Normal instructions writing only an S-register.
5204 case ARM::VLDRS:
5205 case ARM::FCONSTS:
5206 case ARM::VMOVSR:
5207 case ARM::VMOVv8i8:
5208 case ARM::VMOVv4i16:
5209 case ARM::VMOVv2i32:
5210 case ARM::VMOVv2f32:
5211 case ARM::VMOVv1i64:
5212 UseOp = MI.findRegisterUseOperandIdx(Reg, TRI, false);
5213 break;
5214
5215 // Explicitly reads the dependency.
5216 case ARM::VLD1LNd32:
5217 UseOp = 3;
5218 break;
5219 default:
5220 return 0;
5221 }
5222
5223 // If this instruction actually reads a value from Reg, there is no unwanted
5224 // dependency.
5225 if (UseOp != -1 && MI.getOperand(UseOp).readsReg())
5226 return 0;
5227
5228 // We must be able to clobber the whole D-reg.
5229 if (Reg.isVirtual()) {
5230 // Virtual register must be a def undef foo:ssub_0 operand.
5231 if (!MO.getSubReg() || MI.readsVirtualRegister(Reg))
5232 return 0;
5233 } else if (ARM::SPRRegClass.contains(Reg)) {
5234 // Physical register: MI must define the full D-reg.
5235 MCRegister DReg =
5236 TRI->getMatchingSuperReg(Reg, ARM::ssub_0, &ARM::DPRRegClass);
5237 if (!DReg || !MI.definesRegister(DReg, TRI))
5238 return 0;
5239 }
5240
5241 // MI has an unwanted D-register dependency.
5242 // Avoid defs in the previous N instructrions.
5243 return PartialUpdateClearance;
5244}
5245
5246// Break a partial register dependency after getPartialRegUpdateClearance
5247// returned non-zero.
5249 MachineInstr &MI, unsigned OpNum, const TargetRegisterInfo *TRI) const {
5250 assert(OpNum < MI.getDesc().getNumDefs() && "OpNum is not a def");
5251 assert(TRI && "Need TRI instance");
5252
5253 const MachineOperand &MO = MI.getOperand(OpNum);
5254 Register Reg = MO.getReg();
5255 assert(Reg.isPhysical() && "Can't break virtual register dependencies.");
5256 unsigned DReg = Reg;
5257
5258 // If MI defines an S-reg, find the corresponding D super-register.
5259 if (ARM::SPRRegClass.contains(Reg)) {
5260 DReg = ARM::D0 + (Reg - ARM::S0) / 2;
5261 assert(TRI->isSuperRegister(Reg, DReg) && "Register enums broken");
5262 }
5263
5264 assert(ARM::DPRRegClass.contains(DReg) && "Can only break D-reg deps");
5265 assert(MI.definesRegister(DReg, TRI) && "MI doesn't clobber full D-reg");
5266
5267 // FIXME: In some cases, VLDRS can be changed to a VLD1DUPd32 which defines
5268 // the full D-register by loading the same value to both lanes. The
5269 // instruction is micro-coded with 2 uops, so don't do this until we can
5270 // properly schedule micro-coded instructions. The dispatcher stalls cause
5271 // too big regressions.
5272
5273 // Insert the dependency-breaking FCONSTD before MI.
5274 // 96 is the encoding of 0.5, but the actual value doesn't matter here.
5275 BuildMI(*MI.getParent(), MI, MI.getDebugLoc(), get(ARM::FCONSTD), DReg)
5276 .addImm(96)
5278 MI.addRegisterKilled(DReg, TRI, true);
5279}
5280
5282 return Subtarget.hasFeature(ARM::HasV6KOps);
5283}
5284
5286 if (MI->getNumOperands() < 4)
5287 return true;
5288 unsigned ShOpVal = MI->getOperand(3).getImm();
5289 unsigned ShImm = ARM_AM::getSORegOffset(ShOpVal);
5290 // Swift supports faster shifts for: lsl 2, lsl 1, and lsr 1.
5291 if ((ShImm == 1 && ARM_AM::getSORegShOp(ShOpVal) == ARM_AM::lsr) ||
5292 ((ShImm == 1 || ShImm == 2) &&
5293 ARM_AM::getSORegShOp(ShOpVal) == ARM_AM::lsl))
5294 return true;
5295
5296 return false;
5297}
5298
5300 const MachineInstr &MI, unsigned DefIdx,
5301 SmallVectorImpl<RegSubRegPairAndIdx> &InputRegs) const {
5302 assert(DefIdx < MI.getDesc().getNumDefs() && "Invalid definition index");
5303 assert(MI.isRegSequenceLike() && "Invalid kind of instruction");
5304
5305 switch (MI.getOpcode()) {
5306 case ARM::VMOVDRR:
5307 // dX = VMOVDRR rY, rZ
5308 // is the same as:
5309 // dX = REG_SEQUENCE rY, ssub_0, rZ, ssub_1
5310 // Populate the InputRegs accordingly.
5311 // rY
5312 const MachineOperand *MOReg = &MI.getOperand(1);
5313 if (!MOReg->isUndef())
5314 InputRegs.push_back(RegSubRegPairAndIdx(MOReg->getReg(),
5315 MOReg->getSubReg(), ARM::ssub_0));
5316 // rZ
5317 MOReg = &MI.getOperand(2);
5318 if (!MOReg->isUndef())
5319 InputRegs.push_back(RegSubRegPairAndIdx(MOReg->getReg(),
5320 MOReg->getSubReg(), ARM::ssub_1));
5321 return true;
5322 }
5323 llvm_unreachable("Target dependent opcode missing");
5324}
5325
5327 const MachineInstr &MI, unsigned DefIdx,
5328 RegSubRegPairAndIdx &InputReg) const {
5329 assert(DefIdx < MI.getDesc().getNumDefs() && "Invalid definition index");
5330 assert(MI.isExtractSubregLike() && "Invalid kind of instruction");
5331
5332 switch (MI.getOpcode()) {
5333 case ARM::VMOVRRD:
5334 // rX, rY = VMOVRRD dZ
5335 // is the same as:
5336 // rX = EXTRACT_SUBREG dZ, ssub_0
5337 // rY = EXTRACT_SUBREG dZ, ssub_1
5338 const MachineOperand &MOReg = MI.getOperand(2);
5339 if (MOReg.isUndef())
5340 return false;
5341 InputReg.Reg = MOReg.getReg();
5342 InputReg.SubReg = MOReg.getSubReg();
5343 InputReg.SubIdx = DefIdx == 0 ? ARM::ssub_0 : ARM::ssub_1;
5344 return true;
5345 }
5346 llvm_unreachable("Target dependent opcode missing");
5347}
5348
5350 const MachineInstr &MI, unsigned DefIdx, RegSubRegPair &BaseReg,
5351 RegSubRegPairAndIdx &InsertedReg) const {
5352 assert(DefIdx < MI.getDesc().getNumDefs() && "Invalid definition index");
5353 assert(MI.isInsertSubregLike() && "Invalid kind of instruction");
5354
5355 switch (MI.getOpcode()) {
5356 case ARM::VSETLNi32:
5357 case ARM::MVE_VMOV_to_lane_32:
5358 // dX = VSETLNi32 dY, rZ, imm
5359 // qX = MVE_VMOV_to_lane_32 qY, rZ, imm
5360 const MachineOperand &MOBaseReg = MI.getOperand(1);
5361 const MachineOperand &MOInsertedReg = MI.getOperand(2);
5362 if (MOInsertedReg.isUndef())
5363 return false;
5364 const MachineOperand &MOIndex = MI.getOperand(3);
5365 BaseReg.Reg = MOBaseReg.getReg();
5366 BaseReg.SubReg = MOBaseReg.getSubReg();
5367
5368 InsertedReg.Reg = MOInsertedReg.getReg();
5369 InsertedReg.SubReg = MOInsertedReg.getSubReg();
5370 InsertedReg.SubIdx = ARM::ssub_0 + MOIndex.getImm();
5371 return true;
5372 }
5373 llvm_unreachable("Target dependent opcode missing");
5374}
5375
5376std::pair<unsigned, unsigned>
5378 const unsigned Mask = ARMII::MO_OPTION_MASK;
5379 return std::make_pair(TF & Mask, TF & ~Mask);
5380}
5381
5384 using namespace ARMII;
5385
5386 static const std::pair<unsigned, const char *> TargetFlags[] = {
5387 {MO_LO16, "arm-lo16"}, {MO_HI16, "arm-hi16"},
5388 {MO_LO_0_7, "arm-lo-0-7"}, {MO_HI_0_7, "arm-hi-0-7"},
5389 {MO_LO_8_15, "arm-lo-8-15"}, {MO_HI_8_15, "arm-hi-8-15"},
5390 };
5391 return ArrayRef(TargetFlags);
5392}
5393
5396 using namespace ARMII;
5397
5398 static const std::pair<unsigned, const char *> TargetFlags[] = {
5399 {MO_COFFSTUB, "arm-coffstub"},
5400 {MO_GOT, "arm-got"},
5401 {MO_SBREL, "arm-sbrel"},
5402 {MO_DLLIMPORT, "arm-dllimport"},
5403 {MO_SECREL, "arm-secrel"},
5404 {MO_NONLAZY, "arm-nonlazy"}};
5405 return ArrayRef(TargetFlags);
5406}
5407
5408std::optional<RegImmPair>
5410 int Sign = 1;
5411 unsigned Opcode = MI.getOpcode();
5412 int64_t Offset = 0;
5413
5414 // TODO: Handle cases where Reg is a super- or sub-register of the
5415 // destination register.
5416 const MachineOperand &Op0 = MI.getOperand(0);
5417 if (!Op0.isReg() || Reg != Op0.getReg())
5418 return std::nullopt;
5419
5420 // We describe SUBri or ADDri instructions.
5421 if (Opcode == ARM::SUBri)
5422 Sign = -1;
5423 else if (Opcode != ARM::ADDri)
5424 return std::nullopt;
5425
5426 // TODO: Third operand can be global address (usually some string). Since
5427 // strings can be relocated we cannot calculate their offsets for
5428 // now.
5429 if (!MI.getOperand(1).isReg() || !MI.getOperand(2).isImm())
5430 return std::nullopt;
5431
5432 Offset = MI.getOperand(2).getImm() * Sign;
5433 return RegImmPair{MI.getOperand(1).getReg(), Offset};
5434}
5435
5439 const TargetRegisterInfo *TRI) {
5440 for (auto I = From; I != To; ++I)
5441 if (I->modifiesRegister(Reg, TRI))
5442 return true;
5443 return false;
5444}
5445
5447 const TargetRegisterInfo *TRI) {
5448 // Search backwards to the instruction that defines CSPR. This may or not
5449 // be a CMP, we check that after this loop. If we find another instruction
5450 // that reads cpsr, we return nullptr.
5451 MachineBasicBlock::iterator CmpMI = Br;
5452 while (CmpMI != Br->getParent()->begin()) {
5453 --CmpMI;
5454 if (CmpMI->modifiesRegister(ARM::CPSR, TRI))
5455 break;
5456 if (CmpMI->readsRegister(ARM::CPSR, TRI))
5457 break;
5458 }
5459
5460 // Check that this inst is a CMP r[0-7], #0 and that the register
5461 // is not redefined between the cmp and the br.
5462 if (CmpMI->getOpcode() != ARM::tCMPi8 && CmpMI->getOpcode() != ARM::t2CMPri)
5463 return nullptr;
5464 Register Reg = CmpMI->getOperand(0).getReg();
5465 Register PredReg;
5466 ARMCC::CondCodes Pred = getInstrPredicate(*CmpMI, PredReg);
5467 if (Pred != ARMCC::AL || CmpMI->getOperand(1).getImm() != 0)
5468 return nullptr;
5469 if (!isARMLowRegister(Reg))
5470 return nullptr;
5471 if (registerDefinedBetween(Reg, CmpMI->getNextNode(), Br, TRI))
5472 return nullptr;
5473
5474 return &*CmpMI;
5475}
5476
5478 const ARMSubtarget *Subtarget,
5479 bool ForCodesize) {
5480 if (Subtarget->isThumb()) {
5481 if (Val <= 255) // MOV
5482 return ForCodesize ? 2 : 1;
5483 if (Subtarget->hasV6T2Ops() && (Val <= 0xffff || // MOV
5484 ARM_AM::getT2SOImmVal(Val) != -1 || // MOVW
5485 ARM_AM::getT2SOImmVal(~Val) != -1)) // MVN
5486 return ForCodesize ? 4 : 1;
5487 if (Val <= 510) // MOV + ADDi8
5488 return ForCodesize ? 4 : 2;
5489 if (~Val <= 255) // MOV + MVN
5490 return ForCodesize ? 4 : 2;
5491 if (ARM_AM::isThumbImmShiftedVal(Val)) // MOV + LSL
5492 return ForCodesize ? 4 : 2;
5493 } else {
5494 if (ARM_AM::getSOImmVal(Val) != -1) // MOV
5495 return ForCodesize ? 4 : 1;
5496 if (ARM_AM::getSOImmVal(~Val) != -1) // MVN
5497 return ForCodesize ? 4 : 1;
5498 if (Subtarget->hasV6T2Ops() && Val <= 0xffff) // MOVW
5499 return ForCodesize ? 4 : 1;
5500 if (ARM_AM::isSOImmTwoPartVal(Val)) // two instrs
5501 return ForCodesize ? 8 : 2;
5502 if (ARM_AM::isSOImmTwoPartValNeg(Val)) // two instrs
5503 return ForCodesize ? 8 : 2;
5504 }
5505 if (Subtarget->useMovt()) // MOVW + MOVT
5506 return ForCodesize ? 8 : 2;
5507 return ForCodesize ? 8 : 3; // Literal pool load
5508}
5509
5510bool llvm::HasLowerConstantMaterializationCost(unsigned Val1, unsigned Val2,
5511 const ARMSubtarget *Subtarget,
5512 bool ForCodesize) {
5513 // Check with ForCodesize
5514 unsigned Cost1 = ConstantMaterializationCost(Val1, Subtarget, ForCodesize);
5515 unsigned Cost2 = ConstantMaterializationCost(Val2, Subtarget, ForCodesize);
5516 if (Cost1 < Cost2)
5517 return true;
5518 if (Cost1 > Cost2)
5519 return false;
5520
5521 // If they are equal, try with !ForCodesize
5522 return ConstantMaterializationCost(Val1, Subtarget, !ForCodesize) <
5523 ConstantMaterializationCost(Val2, Subtarget, !ForCodesize);
5524}
5525
5526/// Constants defining how certain sequences should be outlined.
5527/// This encompasses how an outlined function should be called, and what kind of
5528/// frame should be emitted for that outlined function.
5529///
5530/// \p MachineOutlinerTailCall implies that the function is being created from
5531/// a sequence of instructions ending in a return.
5532///
5533/// That is,
5534///
5535/// I1 OUTLINED_FUNCTION:
5536/// I2 --> B OUTLINED_FUNCTION I1
5537/// BX LR I2
5538/// BX LR
5539///
5540/// +-------------------------+--------+-----+
5541/// | | Thumb2 | ARM |
5542/// +-------------------------+--------+-----+
5543/// | Call overhead in Bytes | 4 | 4 |
5544/// | Frame overhead in Bytes | 0 | 0 |
5545/// | Stack fixup required | No | No |
5546/// +-------------------------+--------+-----+
5547///
5548/// \p MachineOutlinerThunk implies that the function is being created from
5549/// a sequence of instructions ending in a call. The outlined function is
5550/// called with a BL instruction, and the outlined function tail-calls the
5551/// original call destination.
5552///
5553/// That is,
5554///
5555/// I1 OUTLINED_FUNCTION:
5556/// I2 --> BL OUTLINED_FUNCTION I1
5557/// BL f I2
5558/// B f
5559///
5560/// +-------------------------+--------+-----+
5561/// | | Thumb2 | ARM |
5562/// +-------------------------+--------+-----+
5563/// | Call overhead in Bytes | 4 | 4 |
5564/// | Frame overhead in Bytes | 0 | 0 |
5565/// | Stack fixup required | No | No |
5566/// +-------------------------+--------+-----+
5567///
5568/// \p MachineOutlinerNoLRSave implies that the function should be called using
5569/// a BL instruction, but doesn't require LR to be saved and restored. This
5570/// happens when LR is known to be dead.
5571///
5572/// That is,
5573///
5574/// I1 OUTLINED_FUNCTION:
5575/// I2 --> BL OUTLINED_FUNCTION I1
5576/// I3 I2
5577/// I3
5578/// BX LR
5579///
5580/// +-------------------------+--------+-----+
5581/// | | Thumb2 | ARM |
5582/// +-------------------------+--------+-----+
5583/// | Call overhead in Bytes | 4 | 4 |
5584/// | Frame overhead in Bytes | 2 | 4 |
5585/// | Stack fixup required | No | No |
5586/// +-------------------------+--------+-----+
5587///
5588/// \p MachineOutlinerRegSave implies that the function should be called with a
5589/// save and restore of LR to an available register. This allows us to avoid
5590/// stack fixups. Note that this outlining variant is compatible with the
5591/// NoLRSave case.
5592///
5593/// That is,
5594///
5595/// I1 Save LR OUTLINED_FUNCTION:
5596/// I2 --> BL OUTLINED_FUNCTION I1
5597/// I3 Restore LR I2
5598/// I3
5599/// BX LR
5600///
5601/// +-------------------------+--------+-----+
5602/// | | Thumb2 | ARM |
5603/// +-------------------------+--------+-----+
5604/// | Call overhead in Bytes | 8 | 12 |
5605/// | Frame overhead in Bytes | 2 | 4 |
5606/// | Stack fixup required | No | No |
5607/// +-------------------------+--------+-----+
5608///
5609/// \p MachineOutlinerDefault implies that the function should be called with
5610/// a save and restore of LR to the stack.
5611///
5612/// That is,
5613///
5614/// I1 Save LR OUTLINED_FUNCTION:
5615/// I2 --> BL OUTLINED_FUNCTION I1
5616/// I3 Restore LR I2
5617/// I3
5618/// BX LR
5619///
5620/// +-------------------------+--------+-----+
5621/// | | Thumb2 | ARM |
5622/// +-------------------------+--------+-----+
5623/// | Call overhead in Bytes | 8 | 12 |
5624/// | Frame overhead in Bytes | 2 | 4 |
5625/// | Stack fixup required | Yes | Yes |
5626/// +-------------------------+--------+-----+
5627
5635
5641
5654
5656 : CallTailCall(target.isThumb() ? 4 : 4),
5657 FrameTailCall(target.isThumb() ? 0 : 0),
5658 CallThunk(target.isThumb() ? 4 : 4),
5659 FrameThunk(target.isThumb() ? 0 : 0),
5660 CallNoLRSave(target.isThumb() ? 4 : 4),
5661 FrameNoLRSave(target.isThumb() ? 2 : 4),
5662 CallRegSave(target.isThumb() ? 8 : 12),
5663 FrameRegSave(target.isThumb() ? 2 : 4),
5664 CallDefault(target.isThumb() ? 8 : 12),
5665 FrameDefault(target.isThumb() ? 2 : 4),
5666 SaveRestoreLROnStack(target.isThumb() ? 8 : 8) {}
5667};
5668
5670ARMBaseInstrInfo::findRegisterToSaveLRTo(outliner::Candidate &C) const {
5671 MachineFunction *MF = C.getMF();
5672 const TargetRegisterInfo &TRI = *MF->getSubtarget().getRegisterInfo();
5673 const ARMBaseRegisterInfo *ARI =
5674 static_cast<const ARMBaseRegisterInfo *>(&TRI);
5675
5676 BitVector regsReserved = ARI->getReservedRegs(*MF);
5677 // Check if there is an available register across the sequence that we can
5678 // use.
5679 for (Register Reg : ARM::rGPRRegClass) {
5680 if (!(Reg < regsReserved.size() && regsReserved.test(Reg)) &&
5681 Reg != ARM::LR && // LR is not reserved, but don't use it.
5682 Reg != ARM::R12 && // R12 is not guaranteed to be preserved.
5683 C.isAvailableAcrossAndOutOfSeq(Reg, TRI) &&
5684 C.isAvailableInsideSeq(Reg, TRI))
5685 return Reg;
5686 }
5687 return Register();
5688}
5689
5690// Compute liveness of LR at the point after the interval [I, E), which
5691// denotes a *backward* iteration through instructions. Used only for return
5692// basic blocks, which do not end with a tail call.
5696 // At the end of the function LR dead.
5697 bool Live = false;
5698 for (; I != E; ++I) {
5699 const MachineInstr &MI = *I;
5700
5701 // Check defs of LR.
5702 if (MI.modifiesRegister(ARM::LR, &TRI))
5703 Live = false;
5704
5705 // Check uses of LR.
5706 unsigned Opcode = MI.getOpcode();
5707 if (Opcode == ARM::BX_RET || Opcode == ARM::MOVPCLR ||
5708 Opcode == ARM::SUBS_PC_LR || Opcode == ARM::tBX_RET ||
5709 Opcode == ARM::tBXNS_RET || Opcode == ARM::t2BXAUT_RET) {
5710 // These instructions use LR, but it's not an (explicit or implicit)
5711 // operand.
5712 Live = true;
5713 continue;
5714 }
5715 if (MI.readsRegister(ARM::LR, &TRI))
5716 Live = true;
5717 }
5718 return !Live;
5719}
5720
5721/// Return true if \p MI is a call instruction that the outliner can rewrite as
5722/// a tail call.
5724 auto Opcode = MI.getOpcode();
5725 return (Opcode == ARM::BL || Opcode == ARM::BLX || Opcode == ARM::BLX_noip ||
5726 Opcode == ARM::tBL || Opcode == ARM::tBLXi || Opcode == ARM::tBLXr ||
5727 Opcode == ARM::tBLXr_noip);
5728}
5729
5730std::optional<std::unique_ptr<outliner::OutlinedFunction>>
5732 const MachineModuleInfo &MMI,
5733 std::vector<outliner::Candidate> &RepeatedSequenceLocs,
5734 unsigned MinRepeats) const {
5735 unsigned SequenceSize = 0;
5736 for (auto &MI : RepeatedSequenceLocs[0])
5737 SequenceSize += getInstSizeInBytes(MI);
5738
5739 // Properties about candidate MBBs that hold for all of them.
5740 unsigned FlagsSetInAll = 0xF;
5741
5742 // Compute liveness information for each candidate, and set FlagsSetInAll.
5744 for (outliner::Candidate &C : RepeatedSequenceLocs)
5745 FlagsSetInAll &= C.Flags;
5746
5747 // According to the ARM Procedure Call Standard, the following are
5748 // undefined on entry/exit from a function call:
5749 //
5750 // * Register R12(IP),
5751 // * Condition codes (and thus the CPSR register)
5752 //
5753 // Since we control the instructions which are part of the outlined regions
5754 // we don't need to be fully compliant with the AAPCS, but we have to
5755 // guarantee that if a veneer is inserted at link time the code is still
5756 // correct. Because of this, we can't outline any sequence of instructions
5757 // where one of these registers is live into/across it. Thus, we need to
5758 // delete those candidates.
5759 auto CantGuaranteeValueAcrossCall = [&TRI](outliner::Candidate &C) {
5760 // If the unsafe registers in this block are all dead, then we don't need
5761 // to compute liveness here.
5762 if (C.Flags & UnsafeRegsDead)
5763 return false;
5764 return C.isAnyUnavailableAcrossOrOutOfSeq({ARM::R12, ARM::CPSR}, TRI);
5765 };
5766
5767 // Are there any candidates where those registers are live?
5768 if (!(FlagsSetInAll & UnsafeRegsDead)) {
5769 // Erase every candidate that violates the restrictions above. (It could be
5770 // true that we have viable candidates, so it's not worth bailing out in
5771 // the case that, say, 1 out of 20 candidates violate the restructions.)
5772 llvm::erase_if(RepeatedSequenceLocs, CantGuaranteeValueAcrossCall);
5773
5774 // If the sequence doesn't have enough candidates left, then we're done.
5775 if (RepeatedSequenceLocs.size() < MinRepeats)
5776 return std::nullopt;
5777 }
5778
5779 // We expect the majority of the outlining candidates to be in consensus with
5780 // regard to return address sign and authentication, and branch target
5781 // enforcement, in other words, partitioning according to all the four
5782 // possible combinations of PAC-RET and BTI is going to yield one big subset
5783 // and three small (likely empty) subsets. That allows us to cull incompatible
5784 // candidates separately for PAC-RET and BTI.
5785
5786 // Partition the candidates in two sets: one with BTI enabled and one with BTI
5787 // disabled. Remove the candidates from the smaller set. If they are the same
5788 // number prefer the non-BTI ones for outlining, since they have less
5789 // overhead.
5790 auto NoBTI =
5791 llvm::partition(RepeatedSequenceLocs, [](const outliner::Candidate &C) {
5792 const ARMFunctionInfo &AFI = *C.getMF()->getInfo<ARMFunctionInfo>();
5793 return AFI.branchTargetEnforcement();
5794 });
5795 if (std::distance(RepeatedSequenceLocs.begin(), NoBTI) >
5796 std::distance(NoBTI, RepeatedSequenceLocs.end()))
5797 RepeatedSequenceLocs.erase(NoBTI, RepeatedSequenceLocs.end());
5798 else
5799 RepeatedSequenceLocs.erase(RepeatedSequenceLocs.begin(), NoBTI);
5800
5801 if (RepeatedSequenceLocs.size() < MinRepeats)
5802 return std::nullopt;
5803
5804 // Likewise, partition the candidates according to PAC-RET enablement.
5805 auto NoPAC =
5806 llvm::partition(RepeatedSequenceLocs, [](const outliner::Candidate &C) {
5807 const ARMFunctionInfo &AFI = *C.getMF()->getInfo<ARMFunctionInfo>();
5808 // If the function happens to not spill the LR, do not disqualify it
5809 // from the outlining.
5810 return AFI.shouldSignReturnAddress(true);
5811 });
5812 if (std::distance(RepeatedSequenceLocs.begin(), NoPAC) >
5813 std::distance(NoPAC, RepeatedSequenceLocs.end()))
5814 RepeatedSequenceLocs.erase(NoPAC, RepeatedSequenceLocs.end());
5815 else
5816 RepeatedSequenceLocs.erase(RepeatedSequenceLocs.begin(), NoPAC);
5817
5818 if (RepeatedSequenceLocs.size() < MinRepeats)
5819 return std::nullopt;
5820
5821 // At this point, we have only "safe" candidates to outline. Figure out
5822 // frame + call instruction information.
5823
5824 // Helper lambda which sets call information for every candidate.
5825 auto SetCandidateCallInfo =
5826 [&RepeatedSequenceLocs](unsigned CallID, unsigned NumBytesForCall) {
5827 for (outliner::Candidate &C : RepeatedSequenceLocs)
5828 C.setCallInfo(CallID, NumBytesForCall);
5829 };
5830
5831 OutlinerCosts Costs(Subtarget);
5832
5833 const auto &SomeMFI =
5834 *RepeatedSequenceLocs.front().getMF()->getInfo<ARMFunctionInfo>();
5835 // Adjust costs to account for the BTI instructions.
5836 if (SomeMFI.branchTargetEnforcement()) {
5837 Costs.FrameDefault += 4;
5838 Costs.FrameNoLRSave += 4;
5839 Costs.FrameRegSave += 4;
5840 Costs.FrameTailCall += 4;
5841 Costs.FrameThunk += 4;
5842 }
5843
5844 // Adjust costs to account for sign and authentication instructions.
5845 if (SomeMFI.shouldSignReturnAddress(true)) {
5846 Costs.CallDefault += 8; // +PAC instr, +AUT instr
5847 Costs.SaveRestoreLROnStack += 8; // +PAC instr, +AUT instr
5848 }
5849
5850 unsigned FrameID = MachineOutlinerDefault;
5851 unsigned NumBytesToCreateFrame = Costs.FrameDefault;
5852
5853 // If the last instruction in any candidate is a terminator, then we should
5854 // tail call all of the candidates.
5855 if (RepeatedSequenceLocs[0].back().isTerminator()) {
5856 FrameID = MachineOutlinerTailCall;
5857 NumBytesToCreateFrame = Costs.FrameTailCall;
5858 SetCandidateCallInfo(MachineOutlinerTailCall, Costs.CallTailCall);
5859 } else if (CanTransformInstrIntoTailCall(RepeatedSequenceLocs[0].back())) {
5860 FrameID = MachineOutlinerThunk;
5861 NumBytesToCreateFrame = Costs.FrameThunk;
5862 SetCandidateCallInfo(MachineOutlinerThunk, Costs.CallThunk);
5863 } else {
5864 // We need to decide how to emit calls + frames. We can always emit the same
5865 // frame if we don't need to save to the stack. If we have to save to the
5866 // stack, then we need a different frame.
5867 unsigned NumBytesNoStackCalls = 0;
5868 std::vector<outliner::Candidate> CandidatesWithoutStackFixups;
5869
5870 for (outliner::Candidate &C : RepeatedSequenceLocs) {
5871 // LR liveness is overestimated in return blocks, unless they end with a
5872 // tail call.
5873 const auto Last = C.getMBB()->rbegin();
5874 const bool LRIsAvailable =
5875 C.getMBB()->isReturnBlock() && !Last->isCall()
5878 : C.isAvailableAcrossAndOutOfSeq(ARM::LR, TRI);
5879 if (LRIsAvailable) {
5880 FrameID = MachineOutlinerNoLRSave;
5881 NumBytesNoStackCalls += Costs.CallNoLRSave;
5882 C.setCallInfo(MachineOutlinerNoLRSave, Costs.CallNoLRSave);
5883 CandidatesWithoutStackFixups.push_back(C);
5884 }
5885
5886 // Is an unused register available? If so, we won't modify the stack, so
5887 // we can outline with the same frame type as those that don't save LR.
5888 else if (findRegisterToSaveLRTo(C)) {
5889 FrameID = MachineOutlinerRegSave;
5890 NumBytesNoStackCalls += Costs.CallRegSave;
5891 C.setCallInfo(MachineOutlinerRegSave, Costs.CallRegSave);
5892 CandidatesWithoutStackFixups.push_back(C);
5893 }
5894
5895 // Is SP used in the sequence at all? If not, we don't have to modify
5896 // the stack, so we are guaranteed to get the same frame.
5897 else if (C.isAvailableInsideSeq(ARM::SP, TRI)) {
5898 NumBytesNoStackCalls += Costs.CallDefault;
5899 C.setCallInfo(MachineOutlinerDefault, Costs.CallDefault);
5900 CandidatesWithoutStackFixups.push_back(C);
5901 }
5902
5903 // If we outline this, we need to modify the stack. Pretend we don't
5904 // outline this by saving all of its bytes.
5905 else
5906 NumBytesNoStackCalls += SequenceSize;
5907 }
5908
5909 // If there are no places where we have to save LR, then note that we don't
5910 // have to update the stack. Otherwise, give every candidate the default
5911 // call type
5912 if (NumBytesNoStackCalls <=
5913 RepeatedSequenceLocs.size() * Costs.CallDefault) {
5914 RepeatedSequenceLocs = CandidatesWithoutStackFixups;
5915 FrameID = MachineOutlinerNoLRSave;
5916 if (RepeatedSequenceLocs.size() < MinRepeats)
5917 return std::nullopt;
5918 } else
5919 SetCandidateCallInfo(MachineOutlinerDefault, Costs.CallDefault);
5920 }
5921
5922 // Does every candidate's MBB contain a call? If so, then we might have a
5923 // call in the range.
5924 if (FlagsSetInAll & MachineOutlinerMBBFlags::HasCalls) {
5925 // check if the range contains a call. These require a save + restore of
5926 // the link register.
5927 outliner::Candidate &FirstCand = RepeatedSequenceLocs[0];
5928 if (any_of(drop_end(FirstCand),
5929 [](const MachineInstr &MI) { return MI.isCall(); }))
5930 NumBytesToCreateFrame += Costs.SaveRestoreLROnStack;
5931
5932 // Handle the last instruction separately. If it is tail call, then the
5933 // last instruction is a call, we don't want to save + restore in this
5934 // case. However, it could be possible that the last instruction is a
5935 // call without it being valid to tail call this sequence. We should
5936 // consider this as well.
5937 else if (FrameID != MachineOutlinerThunk &&
5938 FrameID != MachineOutlinerTailCall && FirstCand.back().isCall())
5939 NumBytesToCreateFrame += Costs.SaveRestoreLROnStack;
5940 }
5941
5942 return std::make_unique<outliner::OutlinedFunction>(
5943 RepeatedSequenceLocs, SequenceSize, NumBytesToCreateFrame, FrameID);
5944}
5945
5946bool ARMBaseInstrInfo::checkAndUpdateStackOffset(MachineInstr *MI,
5947 int64_t Fixup,
5948 bool Updt) const {
5949 int SPIdx = MI->findRegisterUseOperandIdx(ARM::SP, /*TRI=*/nullptr);
5950 unsigned AddrMode = (MI->getDesc().TSFlags & ARMII::AddrModeMask);
5951 if (SPIdx < 0)
5952 // No SP operand
5953 return true;
5954 else if (SPIdx != 1 && (AddrMode != ARMII::AddrModeT2_i8s4 || SPIdx != 2))
5955 // If SP is not the base register we can't do much
5956 return false;
5957
5958 // Stack might be involved but addressing mode doesn't handle any offset.
5959 // Rq: AddrModeT1_[1|2|4] don't operate on SP
5960 if (AddrMode == ARMII::AddrMode1 || // Arithmetic instructions
5961 AddrMode == ARMII::AddrMode4 || // Load/Store Multiple
5962 AddrMode == ARMII::AddrMode6 || // Neon Load/Store Multiple
5963 AddrMode == ARMII::AddrModeT2_so || // SP can't be used as based register
5964 AddrMode == ARMII::AddrModeT2_pc || // PCrel access
5965 AddrMode == ARMII::AddrMode2 || // Used by PRE and POST indexed LD/ST
5966 AddrMode == ARMII::AddrModeT2_i7 || // v8.1-M MVE
5967 AddrMode == ARMII::AddrModeT2_i7s2 || // v8.1-M MVE
5968 AddrMode == ARMII::AddrModeT2_i7s4 || // v8.1-M sys regs VLDR/VSTR
5970 AddrMode == ARMII::AddrModeT2_i8 || // Pre/Post inc instructions
5971 AddrMode == ARMII::AddrModeT2_i8neg) // Always negative imm
5972 return false;
5973
5974 unsigned NumOps = MI->getDesc().getNumOperands();
5975 unsigned ImmIdx = NumOps - 3;
5976
5977 const MachineOperand &Offset = MI->getOperand(ImmIdx);
5978 assert(Offset.isImm() && "Is not an immediate");
5979 int64_t OffVal = Offset.getImm();
5980
5981 if (OffVal < 0)
5982 // Don't override data if the are below SP.
5983 return false;
5984
5985 unsigned NumBits = 0;
5986 unsigned Scale = 1;
5987
5988 switch (AddrMode) {
5989 case ARMII::AddrMode3:
5990 if (ARM_AM::getAM3Op(OffVal) == ARM_AM::sub)
5991 return false;
5992 OffVal = ARM_AM::getAM3Offset(OffVal);
5993 NumBits = 8;
5994 break;
5995 case ARMII::AddrMode5:
5996 if (ARM_AM::getAM5Op(OffVal) == ARM_AM::sub)
5997 return false;
5998 OffVal = ARM_AM::getAM5Offset(OffVal);
5999 NumBits = 8;
6000 Scale = 4;
6001 break;
6003 if (ARM_AM::getAM5FP16Op(OffVal) == ARM_AM::sub)
6004 return false;
6005 OffVal = ARM_AM::getAM5FP16Offset(OffVal);
6006 NumBits = 8;
6007 Scale = 2;
6008 break;
6010 NumBits = 8;
6011 break;
6013 // FIXME: Values are already scaled in this addressing mode.
6014 assert((Fixup & 3) == 0 && "Can't encode this offset!");
6015 NumBits = 10;
6016 break;
6018 NumBits = 8;
6019 Scale = 4;
6020 break;
6023 NumBits = 12;
6024 break;
6025 case ARMII::AddrModeT1_s: // SP-relative LD/ST
6026 NumBits = 8;
6027 Scale = 4;
6028 break;
6029 default:
6030 llvm_unreachable("Unsupported addressing mode!");
6031 }
6032 // Make sure the offset is encodable for instructions that scale the
6033 // immediate.
6034 assert(((OffVal * Scale + Fixup) & (Scale - 1)) == 0 &&
6035 "Can't encode this offset!");
6036 OffVal += Fixup / Scale;
6037
6038 unsigned Mask = (1 << NumBits) - 1;
6039
6040 if (OffVal <= Mask) {
6041 if (Updt)
6042 MI->getOperand(ImmIdx).setImm(OffVal);
6043 return true;
6044 }
6045
6046 return false;
6047}
6048
6050 Function &F, std::vector<outliner::Candidate> &Candidates) const {
6051 outliner::Candidate &C = Candidates.front();
6052 // branch-target-enforcement is guaranteed to be consistent between all
6053 // candidates, so we only need to look at one.
6054 const Function &CFn = C.getMF()->getFunction();
6055 if (CFn.hasFnAttribute("branch-target-enforcement"))
6056 F.addFnAttr(CFn.getFnAttribute("branch-target-enforcement"));
6057
6058 if (CFn.hasFnAttribute("sign-return-address"))
6059 F.addFnAttr(CFn.getFnAttribute("sign-return-address"));
6060
6061 ARMGenInstrInfo::mergeOutliningCandidateAttributes(F, Candidates);
6062}
6063
6065 MachineFunction &MF, bool OutlineFromLinkOnceODRs) const {
6066 const Function &F = MF.getFunction();
6067
6068 // Can F be deduplicated by the linker? If it can, don't outline from it.
6069 if (!OutlineFromLinkOnceODRs && F.hasLinkOnceODRLinkage())
6070 return false;
6071
6072 // Don't outline from functions with section markings; the program could
6073 // expect that all the code is in the named section.
6074 // FIXME: Allow outlining from multiple functions with the same section
6075 // marking.
6076 if (F.hasSection())
6077 return false;
6078
6079 // FIXME: Thumb1 outlining is not handled
6081 return false;
6082
6083 // It's safe to outline from MF.
6084 return true;
6085}
6086
6088 unsigned &Flags) const {
6089 // Check if LR is available through all of the MBB. If it's not, then set
6090 // a flag.
6091 assert(MBB.getParent()->getRegInfo().tracksLiveness() &&
6092 "Suitable Machine Function for outlining must track liveness");
6093
6095
6097 LRU.accumulate(MI);
6098
6099 // Check if each of the unsafe registers are available...
6100 bool R12AvailableInBlock = LRU.available(ARM::R12);
6101 bool CPSRAvailableInBlock = LRU.available(ARM::CPSR);
6102
6103 // If all of these are dead (and not live out), we know we don't have to check
6104 // them later.
6105 if (R12AvailableInBlock && CPSRAvailableInBlock)
6107
6108 // Now, add the live outs to the set.
6109 LRU.addLiveOuts(MBB);
6110
6111 // If any of these registers is available in the MBB, but also a live out of
6112 // the block, then we know outlining is unsafe.
6113 if (R12AvailableInBlock && !LRU.available(ARM::R12))
6114 return false;
6115 if (CPSRAvailableInBlock && !LRU.available(ARM::CPSR))
6116 return false;
6117
6118 // Check if there's a call inside this MachineBasicBlock. If there is, then
6119 // set a flag.
6120 if (any_of(MBB, [](MachineInstr &MI) { return MI.isCall(); }))
6122
6123 // LR liveness is overestimated in return blocks.
6124
6125 bool LRIsAvailable =
6126 MBB.isReturnBlock() && !MBB.back().isCall()
6127 ? isLRAvailable(getRegisterInfo(), MBB.rbegin(), MBB.rend())
6128 : LRU.available(ARM::LR);
6129 if (!LRIsAvailable)
6131
6132 return true;
6133}
6134
6138 unsigned Flags) const {
6139 MachineInstr &MI = *MIT;
6141
6142 // PIC instructions contain labels, outlining them would break offset
6143 // computing. unsigned Opc = MI.getOpcode();
6144 unsigned Opc = MI.getOpcode();
6145 if (Opc == ARM::tPICADD || Opc == ARM::PICADD || Opc == ARM::PICSTR ||
6146 Opc == ARM::PICSTRB || Opc == ARM::PICSTRH || Opc == ARM::PICLDR ||
6147 Opc == ARM::PICLDRB || Opc == ARM::PICLDRH || Opc == ARM::PICLDRSB ||
6148 Opc == ARM::PICLDRSH || Opc == ARM::t2LDRpci_pic ||
6149 Opc == ARM::t2MOVi16_ga_pcrel || Opc == ARM::t2MOVTi16_ga_pcrel ||
6150 Opc == ARM::t2MOV_ga_pcrel)
6152
6153 // Be conservative with ARMv8.1 MVE instructions.
6154 if (Opc == ARM::t2BF_LabelPseudo || Opc == ARM::t2DoLoopStart ||
6155 Opc == ARM::t2DoLoopStartTP || Opc == ARM::t2WhileLoopStart ||
6156 Opc == ARM::t2WhileLoopStartLR || Opc == ARM::t2WhileLoopStartTP ||
6157 Opc == ARM::t2LoopDec || Opc == ARM::t2LoopEnd ||
6158 Opc == ARM::t2LoopEndDec)
6160
6161 const MCInstrDesc &MCID = MI.getDesc();
6162 uint64_t MIFlags = MCID.TSFlags;
6163 if ((MIFlags & ARMII::DomainMask) == ARMII::DomainMVE)
6165
6166 // Is this a terminator for a basic block?
6167 if (MI.isTerminator())
6168 // TargetInstrInfo::getOutliningType has already filtered out anything
6169 // that would break this, so we can allow it here.
6171
6172 // Don't outline if link register or program counter value are used.
6173 if (MI.readsRegister(ARM::LR, TRI) || MI.readsRegister(ARM::PC, TRI))
6175
6176 if (MI.isCall()) {
6177 // Get the function associated with the call. Look at each operand and find
6178 // the one that represents the calle and get its name.
6179 const Function *Callee = nullptr;
6180 for (const MachineOperand &MOP : MI.operands()) {
6181 if (MOP.isGlobal()) {
6182 Callee = dyn_cast<Function>(MOP.getGlobal());
6183 break;
6184 }
6185 }
6186
6187 // Dont't outline calls to "mcount" like functions, in particular Linux
6188 // kernel function tracing relies on it.
6189 if (Callee &&
6190 (Callee->getName() == "\01__gnu_mcount_nc" ||
6191 Callee->getName() == "\01mcount" || Callee->getName() == "__mcount"))
6193
6194 // If we don't know anything about the callee, assume it depends on the
6195 // stack layout of the caller. In that case, it's only legal to outline
6196 // as a tail-call. Explicitly list the call instructions we know about so
6197 // we don't get unexpected results with call pseudo-instructions.
6198 auto UnknownCallOutlineType = outliner::InstrType::Illegal;
6200 UnknownCallOutlineType = outliner::InstrType::LegalTerminator;
6201
6202 if (!Callee)
6203 return UnknownCallOutlineType;
6204
6205 // We have a function we have information about. Check if it's something we
6206 // can safely outline.
6207 MachineFunction *CalleeMF = MMI.getMachineFunction(*Callee);
6208
6209 // We don't know what's going on with the callee at all. Don't touch it.
6210 if (!CalleeMF)
6211 return UnknownCallOutlineType;
6212
6213 // Check if we know anything about the callee saves on the function. If we
6214 // don't, then don't touch it, since that implies that we haven't computed
6215 // anything about its stack frame yet.
6216 MachineFrameInfo &MFI = CalleeMF->getFrameInfo();
6217 if (!MFI.isCalleeSavedInfoValid() || MFI.getStackSize() > 0 ||
6218 MFI.getNumObjects() > 0)
6219 return UnknownCallOutlineType;
6220
6221 // At this point, we can say that CalleeMF ought to not pass anything on the
6222 // stack. Therefore, we can outline it.
6224 }
6225
6226 // Since calls are handled, don't touch LR or PC
6227 if (MI.modifiesRegister(ARM::LR, TRI) || MI.modifiesRegister(ARM::PC, TRI))
6229
6230 // Does this use the stack?
6231 if (MI.modifiesRegister(ARM::SP, TRI) || MI.readsRegister(ARM::SP, TRI)) {
6232 // True if there is no chance that any outlined candidate from this range
6233 // could require stack fixups. That is, both
6234 // * LR is available in the range (No save/restore around call)
6235 // * The range doesn't include calls (No save/restore in outlined frame)
6236 // are true.
6237 // These conditions also ensure correctness of the return address
6238 // authentication - we insert sign and authentication instructions only if
6239 // we save/restore LR on stack, but then this condition ensures that the
6240 // outlined range does not modify the SP, therefore the SP value used for
6241 // signing is the same as the one used for authentication.
6242 // FIXME: This is very restrictive; the flags check the whole block,
6243 // not just the bit we will try to outline.
6244 bool MightNeedStackFixUp =
6247
6248 if (!MightNeedStackFixUp)
6250
6251 // Any modification of SP will break our code to save/restore LR.
6252 // FIXME: We could handle some instructions which add a constant offset to
6253 // SP, with a bit more work.
6254 if (MI.modifiesRegister(ARM::SP, TRI))
6256
6257 // At this point, we have a stack instruction that we might need to fix up.
6258 // up. We'll handle it if it's a load or store.
6259 if (checkAndUpdateStackOffset(&MI, Subtarget.getStackAlignment().value(),
6260 false))
6262
6263 // We can't fix it up, so don't outline it.
6265 }
6266
6267 // Be conservative with IT blocks.
6268 if (MI.readsRegister(ARM::ITSTATE, TRI) ||
6269 MI.modifiesRegister(ARM::ITSTATE, TRI))
6271
6272 // Don't outline CFI instructions.
6273 if (MI.isCFIInstruction())
6275
6277}
6278
6279void ARMBaseInstrInfo::fixupPostOutline(MachineBasicBlock &MBB) const {
6280 for (MachineInstr &MI : MBB) {
6281 checkAndUpdateStackOffset(&MI, Subtarget.getStackAlignment().value(), true);
6282 }
6283}
6284
6285void ARMBaseInstrInfo::saveLROnStack(MachineBasicBlock &MBB,
6286 MachineBasicBlock::iterator It, bool CFI,
6287 bool Auth) const {
6288 int Align = std::max(Subtarget.getStackAlignment().value(), uint64_t(8));
6289 unsigned MIFlags = CFI ? MachineInstr::FrameSetup : 0;
6290 assert(Align >= 8 && Align <= 256);
6291 if (Auth) {
6292 assert(Subtarget.isThumb2());
6293 // Compute PAC in R12. Outlining ensures R12 is dead across the outlined
6294 // sequence.
6295 BuildMI(MBB, It, DebugLoc(), get(ARM::t2PAC)).setMIFlags(MIFlags);
6296 BuildMI(MBB, It, DebugLoc(), get(ARM::t2STRD_PRE), ARM::SP)
6297 .addReg(ARM::R12, RegState::Kill)
6298 .addReg(ARM::LR, RegState::Kill)
6299 .addReg(ARM::SP)
6300 .addImm(-Align)
6302 .setMIFlags(MIFlags);
6303 } else {
6304 unsigned Opc = Subtarget.isThumb() ? ARM::t2STR_PRE : ARM::STR_PRE_IMM;
6305 BuildMI(MBB, It, DebugLoc(), get(Opc), ARM::SP)
6306 .addReg(ARM::LR, RegState::Kill)
6307 .addReg(ARM::SP)
6308 .addImm(-Align)
6310 .setMIFlags(MIFlags);
6311 }
6312
6313 if (!CFI)
6314 return;
6315
6316 // Add a CFI, saying CFA is offset by Align bytes from SP.
6317 CFIInstBuilder CFIBuilder(MBB, It, MachineInstr::FrameSetup);
6318 CFIBuilder.buildDefCFAOffset(Align);
6319
6320 // Add a CFI saying that the LR that we want to find is now higher than
6321 // before.
6322 int LROffset = Auth ? Align - 4 : Align;
6323 CFIBuilder.buildOffset(ARM::LR, -LROffset);
6324 if (Auth) {
6325 // Add a CFI for the location of the return address PAC.
6326 CFIBuilder.buildOffset(ARM::RA_AUTH_CODE, -Align);
6327 }
6328}
6329
6330void ARMBaseInstrInfo::restoreLRFromStack(MachineBasicBlock &MBB,
6332 bool CFI, bool Auth) const {
6333 int Align = Subtarget.getStackAlignment().value();
6334 unsigned MIFlags = CFI ? MachineInstr::FrameDestroy : 0;
6335 if (Auth) {
6336 assert(Subtarget.isThumb2());
6337 // Restore return address PAC and LR.
6338 BuildMI(MBB, It, DebugLoc(), get(ARM::t2LDRD_POST))
6339 .addReg(ARM::R12, RegState::Define)
6340 .addReg(ARM::LR, RegState::Define)
6341 .addReg(ARM::SP, RegState::Define)
6342 .addReg(ARM::SP)
6343 .addImm(Align)
6345 .setMIFlags(MIFlags);
6346 // LR authentication is after the CFI instructions, below.
6347 } else {
6348 unsigned Opc = Subtarget.isThumb() ? ARM::t2LDR_POST : ARM::LDR_POST_IMM;
6349 MachineInstrBuilder MIB = BuildMI(MBB, It, DebugLoc(), get(Opc), ARM::LR)
6350 .addReg(ARM::SP, RegState::Define)
6351 .addReg(ARM::SP);
6352 if (!Subtarget.isThumb())
6353 MIB.addReg(0);
6354 MIB.addImm(Subtarget.getStackAlignment().value())
6356 .setMIFlags(MIFlags);
6357 }
6358
6359 if (CFI) {
6360 // Now stack has moved back up and we have restored LR.
6361 CFIInstBuilder CFIBuilder(MBB, It, MachineInstr::FrameDestroy);
6362 CFIBuilder.buildDefCFAOffset(0);
6363 CFIBuilder.buildRestore(ARM::LR);
6364 if (Auth)
6365 CFIBuilder.buildUndefined(ARM::RA_AUTH_CODE);
6366 }
6367
6368 if (Auth)
6369 BuildMI(MBB, It, DebugLoc(), get(ARM::t2AUT));
6370}
6371
6374 const outliner::OutlinedFunction &OF) const {
6375 // For thunk outlining, rewrite the last instruction from a call to a
6376 // tail-call.
6377 if (OF.FrameConstructionID == MachineOutlinerThunk) {
6378 MachineInstr *Call = &*--MBB.instr_end();
6379 bool isThumb = Subtarget.isThumb();
6380 unsigned FuncOp = isThumb ? 2 : 0;
6381 unsigned Opc = Call->getOperand(FuncOp).isReg()
6382 ? isThumb ? ARM::tTAILJMPr : ARM::TAILJMPr
6383 : isThumb ? Subtarget.isTargetMachO() ? ARM::tTAILJMPd
6384 : ARM::tTAILJMPdND
6385 : ARM::TAILJMPd;
6386 MachineInstrBuilder MIB = BuildMI(MBB, MBB.end(), DebugLoc(), get(Opc))
6387 .add(Call->getOperand(FuncOp));
6388 if (isThumb && !Call->getOperand(FuncOp).isReg())
6389 MIB.add(predOps(ARMCC::AL));
6390 Call->eraseFromParent();
6391 }
6392
6393 // Is there a call in the outlined range?
6394 auto IsNonTailCall = [](MachineInstr &MI) {
6395 return MI.isCall() && !MI.isReturn();
6396 };
6397 if (llvm::any_of(MBB.instrs(), IsNonTailCall)) {
6398 MachineBasicBlock::iterator It = MBB.begin();
6400
6401 if (OF.FrameConstructionID == MachineOutlinerTailCall ||
6402 OF.FrameConstructionID == MachineOutlinerThunk)
6403 Et = std::prev(MBB.end());
6404
6405 // We have to save and restore LR, we need to add it to the liveins if it
6406 // is not already part of the set. This is sufficient since outlined
6407 // functions only have one block.
6408 if (!MBB.isLiveIn(ARM::LR))
6409 MBB.addLiveIn(ARM::LR);
6410
6411 // Insert a save before the outlined region
6412 bool Auth = MF.getInfo<ARMFunctionInfo>()->shouldSignReturnAddress(true);
6413 saveLROnStack(MBB, It, true, Auth);
6414
6415 // Fix up the instructions in the range, since we're going to modify the
6416 // stack.
6417 assert(OF.FrameConstructionID != MachineOutlinerDefault &&
6418 "Can only fix up stack references once");
6419 fixupPostOutline(MBB);
6420
6421 // Insert a restore before the terminator for the function. Restore LR.
6422 restoreLRFromStack(MBB, Et, true, Auth);
6423 }
6424
6425 // If this is a tail call outlined function, then there's already a return.
6426 if (OF.FrameConstructionID == MachineOutlinerTailCall ||
6427 OF.FrameConstructionID == MachineOutlinerThunk)
6428 return;
6429
6430 // Here we have to insert the return ourselves. Get the correct opcode from
6431 // current feature set.
6432 BuildMI(MBB, MBB.end(), DebugLoc(), get(Subtarget.getReturnOpcode()))
6434
6435 // Did we have to modify the stack by saving the link register?
6436 if (OF.FrameConstructionID != MachineOutlinerDefault &&
6437 OF.Candidates[0].CallConstructionID != MachineOutlinerDefault)
6438 return;
6439
6440 // We modified the stack.
6441 // Walk over the basic block and fix up all the stack accesses.
6442 fixupPostOutline(MBB);
6443}
6444
6450 unsigned Opc;
6451 bool isThumb = Subtarget.isThumb();
6452
6453 // Are we tail calling?
6454 if (C.CallConstructionID == MachineOutlinerTailCall) {
6455 // If yes, then we can just branch to the label.
6456 Opc = isThumb
6457 ? Subtarget.isTargetMachO() ? ARM::tTAILJMPd : ARM::tTAILJMPdND
6458 : ARM::TAILJMPd;
6459 MIB = BuildMI(MF, DebugLoc(), get(Opc))
6460 .addGlobalAddress(M.getNamedValue(MF.getName()));
6461 if (isThumb)
6462 MIB.add(predOps(ARMCC::AL));
6463 It = MBB.insert(It, MIB);
6464 return It;
6465 }
6466
6467 // Create the call instruction.
6468 Opc = isThumb ? ARM::tBL : ARM::BL;
6469 MachineInstrBuilder CallMIB = BuildMI(MF, DebugLoc(), get(Opc));
6470 if (isThumb)
6471 CallMIB.add(predOps(ARMCC::AL));
6472 CallMIB.addGlobalAddress(M.getNamedValue(MF.getName()));
6473
6474 if (C.CallConstructionID == MachineOutlinerNoLRSave ||
6475 C.CallConstructionID == MachineOutlinerThunk) {
6476 // No, so just insert the call.
6477 It = MBB.insert(It, CallMIB);
6478 return It;
6479 }
6480
6481 const ARMFunctionInfo &AFI = *C.getMF()->getInfo<ARMFunctionInfo>();
6482 // Can we save to a register?
6483 if (C.CallConstructionID == MachineOutlinerRegSave) {
6484 Register Reg = findRegisterToSaveLRTo(C);
6485 assert(Reg != 0 && "No callee-saved register available?");
6486
6487 // Save and restore LR from that register.
6488 copyPhysReg(MBB, It, DebugLoc(), Reg, ARM::LR, true);
6489 if (!AFI.isLRSpilled())
6491 .buildRegister(ARM::LR, Reg);
6492 CallPt = MBB.insert(It, CallMIB);
6493 copyPhysReg(MBB, It, DebugLoc(), ARM::LR, Reg, true);
6494 if (!AFI.isLRSpilled())
6496 It--;
6497 return CallPt;
6498 }
6499 // We have the default case. Save and restore from SP.
6500 if (!MBB.isLiveIn(ARM::LR))
6501 MBB.addLiveIn(ARM::LR);
6502 bool Auth = !AFI.isLRSpilled() && AFI.shouldSignReturnAddress(true);
6503 saveLROnStack(MBB, It, !AFI.isLRSpilled(), Auth);
6504 CallPt = MBB.insert(It, CallMIB);
6505 restoreLRFromStack(MBB, It, !AFI.isLRSpilled(), Auth);
6506 It--;
6507 return CallPt;
6508}
6509
6511 MachineFunction &MF) const {
6512 return Subtarget.isMClass() && MF.getFunction().hasMinSize();
6513}
6514
6515bool ARMBaseInstrInfo::isReMaterializableImpl(
6516 const MachineInstr &MI) const {
6517 // Try hard to rematerialize any VCTPs because if we spill P0, it will block
6518 // the tail predication conversion. This means that the element count
6519 // register has to be live for longer, but that has to be better than
6520 // spill/restore and VPT predication.
6521 return (isVCTP(&MI) && !isPredicated(MI)) ||
6523}
6524
6526 return (MF.getSubtarget<ARMSubtarget>().hardenSlsBlr()) ? ARM::BLX_noip
6527 : ARM::BLX;
6528}
6529
6531 return (MF.getSubtarget<ARMSubtarget>().hardenSlsBlr()) ? ARM::tBLXr_noip
6532 : ARM::tBLXr;
6533}
6534
6536 return (MF.getSubtarget<ARMSubtarget>().hardenSlsBlr()) ? ARM::BLX_pred_noip
6537 : ARM::BLX_pred;
6538}
6539
6540namespace {
6541class ARMPipelinerLoopInfo : public TargetInstrInfo::PipelinerLoopInfo {
6542 MachineInstr *EndLoop, *LoopCount;
6543 MachineFunction *MF;
6544 const TargetInstrInfo *TII;
6545
6546 // Bitset[0 .. MAX_STAGES-1] ... iterations needed
6547 // [LAST_IS_USE] : last reference to register in schedule is a use
6548 // [SEEN_AS_LIVE] : Normal pressure algorithm believes register is live
6549 static int constexpr MAX_STAGES = 30;
6550 static int constexpr LAST_IS_USE = MAX_STAGES;
6551 static int constexpr SEEN_AS_LIVE = MAX_STAGES + 1;
6552 typedef std::bitset<MAX_STAGES + 2> IterNeed;
6553 typedef std::map<Register, IterNeed> IterNeeds;
6554
6555 void bumpCrossIterationPressure(RegPressureTracker &RPT,
6556 const IterNeeds &CIN);
6557 bool tooMuchRegisterPressure(SwingSchedulerDAG &SSD, SMSchedule &SMS);
6558
6559 // Meanings of the various stuff with loop types:
6560 // t2Bcc:
6561 // EndLoop = branch at end of original BB that will become a kernel
6562 // LoopCount = CC setter live into branch
6563 // t2LoopEnd:
6564 // EndLoop = branch at end of original BB
6565 // LoopCount = t2LoopDec
6566public:
6567 ARMPipelinerLoopInfo(MachineInstr *EndLoop, MachineInstr *LoopCount)
6568 : EndLoop(EndLoop), LoopCount(LoopCount),
6569 MF(EndLoop->getParent()->getParent()),
6570 TII(MF->getSubtarget().getInstrInfo()) {}
6571
6572 bool shouldIgnoreForPipelining(const MachineInstr *MI) const override {
6573 // Only ignore the terminator.
6574 return MI == EndLoop || MI == LoopCount;
6575 }
6576
6577 bool shouldUseSchedule(SwingSchedulerDAG &SSD, SMSchedule &SMS) override {
6578 if (tooMuchRegisterPressure(SSD, SMS))
6579 return false;
6580
6581 return true;
6582 }
6583
6584 std::optional<bool> createTripCountGreaterCondition(
6585 int TC, MachineBasicBlock &MBB,
6586 SmallVectorImpl<MachineOperand> &Cond) override {
6587
6588 if (isCondBranchOpcode(EndLoop->getOpcode())) {
6589 Cond.push_back(EndLoop->getOperand(1));
6590 Cond.push_back(EndLoop->getOperand(2));
6591 if (EndLoop->getOperand(0).getMBB() == EndLoop->getParent()) {
6593 }
6594 return {};
6595 } else if (EndLoop->getOpcode() == ARM::t2LoopEnd) {
6596 // General case just lets the unrolled t2LoopDec do the subtraction and
6597 // therefore just needs to check if zero has been reached.
6598 MachineInstr *LoopDec = nullptr;
6599 for (auto &I : MBB.instrs())
6600 if (I.getOpcode() == ARM::t2LoopDec)
6601 LoopDec = &I;
6602 assert(LoopDec && "Unable to find copied LoopDec");
6603 // Check if we're done with the loop.
6604 BuildMI(&MBB, LoopDec->getDebugLoc(), TII->get(ARM::t2CMPri))
6605 .addReg(LoopDec->getOperand(0).getReg())
6606 .addImm(0)
6608 .addReg(Register());
6610 Cond.push_back(MachineOperand::CreateReg(ARM::CPSR, false));
6611 return {};
6612 } else
6613 llvm_unreachable("Unknown EndLoop");
6614 }
6615
6616 void setPreheader(MachineBasicBlock *NewPreheader) override {}
6617
6618 void adjustTripCount(int TripCountAdjust) override {}
6619};
6620
6621void ARMPipelinerLoopInfo::bumpCrossIterationPressure(RegPressureTracker &RPT,
6622 const IterNeeds &CIN) {
6623 // Increase pressure by the amounts in CrossIterationNeeds
6624 for (const auto &N : CIN) {
6625 int Cnt = N.second.count() - N.second[SEEN_AS_LIVE] * 2;
6626 for (int I = 0; I < Cnt; ++I)
6629 }
6630 // Decrease pressure by the amounts in CrossIterationNeeds
6631 for (const auto &N : CIN) {
6632 int Cnt = N.second.count() - N.second[SEEN_AS_LIVE] * 2;
6633 for (int I = 0; I < Cnt; ++I)
6636 }
6637}
6638
6639bool ARMPipelinerLoopInfo::tooMuchRegisterPressure(SwingSchedulerDAG &SSD,
6640 SMSchedule &SMS) {
6641 IterNeeds CrossIterationNeeds;
6642
6643 // Determine which values will be loop-carried after the schedule is
6644 // applied
6645
6646 for (auto &SU : SSD.SUnits) {
6647 const MachineInstr *MI = SU.getInstr();
6648 int Stg = SMS.stageScheduled(const_cast<SUnit *>(&SU));
6649 for (auto &S : SU.Succs)
6650 if (MI->isPHI() && S.getKind() == SDep::Anti) {
6651 Register Reg = S.getReg();
6652 if (Reg.isVirtual())
6653 CrossIterationNeeds[Reg.id()].set(0);
6654 } else if (S.isAssignedRegDep()) {
6655 int OStg = SMS.stageScheduled(S.getSUnit());
6656 if (OStg >= 0 && OStg != Stg) {
6657 Register Reg = S.getReg();
6658 if (Reg.isVirtual())
6659 CrossIterationNeeds[Reg.id()] |= ((1 << (OStg - Stg)) - 1);
6660 }
6661 }
6662 }
6663
6664 // Determine more-or-less what the proposed schedule (reversed) is going to
6665 // be; it might not be quite the same because the within-cycle ordering
6666 // created by SMSchedule depends upon changes to help with address offsets and
6667 // the like.
6668 std::vector<SUnit *> ProposedSchedule;
6669 for (int Cycle = SMS.getFinalCycle(); Cycle >= SMS.getFirstCycle(); --Cycle)
6670 for (int Stage = 0, StageEnd = SMS.getMaxStageCount(); Stage <= StageEnd;
6671 ++Stage) {
6672 std::deque<SUnit *> Instrs =
6673 SMS.getInstructions(Cycle + Stage * SMS.getInitiationInterval());
6674 std::sort(Instrs.begin(), Instrs.end(),
6675 [](SUnit *A, SUnit *B) { return A->NodeNum > B->NodeNum; });
6676 llvm::append_range(ProposedSchedule, Instrs);
6677 }
6678
6679 // Learn whether the last use/def of each cross-iteration register is a use or
6680 // def. If it is a def, RegisterPressure will implicitly increase max pressure
6681 // and we do not have to add the pressure.
6682 for (auto *SU : ProposedSchedule)
6683 for (ConstMIBundleOperands OperI(*SU->getInstr()); OperI.isValid();
6684 ++OperI) {
6685 auto MO = *OperI;
6686 if (!MO.isReg() || !MO.getReg())
6687 continue;
6688 Register Reg = MO.getReg();
6689 auto CIter = CrossIterationNeeds.find(Reg.id());
6690 if (CIter == CrossIterationNeeds.end() || CIter->second[LAST_IS_USE] ||
6691 CIter->second[SEEN_AS_LIVE])
6692 continue;
6693 if (MO.isDef() && !MO.isDead())
6694 CIter->second.set(SEEN_AS_LIVE);
6695 else if (MO.isUse())
6696 CIter->second.set(LAST_IS_USE);
6697 }
6698 for (auto &CI : CrossIterationNeeds)
6699 CI.second.reset(LAST_IS_USE);
6700
6701 RegionPressure RecRegPressure;
6702 RegPressureTracker RPTracker(RecRegPressure);
6703 RegisterClassInfo RegClassInfo;
6704 RegClassInfo.runOnMachineFunction(*MF);
6705 RPTracker.init(MF, &RegClassInfo, nullptr, EndLoop->getParent(),
6706 EndLoop->getParent()->end(), false, false);
6707
6708 bumpCrossIterationPressure(RPTracker, CrossIterationNeeds);
6709
6710 for (auto *SU : ProposedSchedule) {
6711 MachineBasicBlock::const_iterator CurInstI = SU->getInstr();
6712 RPTracker.setPos(std::next(CurInstI));
6713 RPTracker.recede();
6714
6715 // Track what cross-iteration registers would be seen as live
6716 for (ConstMIBundleOperands OperI(*CurInstI); OperI.isValid(); ++OperI) {
6717 auto MO = *OperI;
6718 if (!MO.isReg() || !MO.getReg())
6719 continue;
6720 Register Reg = MO.getReg();
6721 if (MO.isDef() && !MO.isDead()) {
6722 auto CIter = CrossIterationNeeds.find(Reg.id());
6723 if (CIter != CrossIterationNeeds.end()) {
6724 CIter->second.reset(0);
6725 CIter->second.reset(SEEN_AS_LIVE);
6726 }
6727 }
6728 }
6729 for (auto &S : SU->Preds) {
6730 auto Stg = SMS.stageScheduled(SU);
6731 if (S.isAssignedRegDep()) {
6732 Register Reg = S.getReg();
6733 auto CIter = CrossIterationNeeds.find(Reg.id());
6734 if (CIter != CrossIterationNeeds.end()) {
6735 auto Stg2 = SMS.stageScheduled(S.getSUnit());
6736 assert(Stg2 <= Stg && "Data dependence upon earlier stage");
6737 if (Stg - Stg2 < MAX_STAGES)
6738 CIter->second.set(Stg - Stg2);
6739 CIter->second.set(SEEN_AS_LIVE);
6740 }
6741 }
6742 }
6743
6744 bumpCrossIterationPressure(RPTracker, CrossIterationNeeds);
6745 }
6746
6747 auto &P = RPTracker.getPressure().MaxSetPressure;
6748 for (unsigned I = 0, E = P.size(); I < E; ++I) {
6749 // Exclude some Neon register classes.
6750 if (I == ARM::DQuad_with_ssub_0 || I == ARM::DTripleSpc_with_ssub_0 ||
6751 I == ARM::DTriple_with_qsub_0_in_QPR)
6752 continue;
6753
6754 if (P[I] > RegClassInfo.getRegPressureSetLimit(I)) {
6755 return true;
6756 }
6757 }
6758 return false;
6759}
6760
6761} // namespace
6762
6763std::unique_ptr<TargetInstrInfo::PipelinerLoopInfo>
6766 MachineBasicBlock *Preheader = *LoopBB->pred_begin();
6767 if (Preheader == LoopBB)
6768 Preheader = *std::next(LoopBB->pred_begin());
6769
6770 if (I != LoopBB->end() && I->getOpcode() == ARM::t2Bcc) {
6771 // If the branch is a Bcc, then the CPSR should be set somewhere within the
6772 // block. We need to determine the reaching definition of CPSR so that
6773 // it can be marked as non-pipelineable, allowing the pipeliner to force
6774 // it into stage 0 or give up if it cannot or will not do so.
6775 MachineInstr *CCSetter = nullptr;
6776 for (auto &L : LoopBB->instrs()) {
6777 if (L.isCall())
6778 return nullptr;
6779 if (isCPSRDefined(L))
6780 CCSetter = &L;
6781 }
6782 if (CCSetter)
6783 return std::make_unique<ARMPipelinerLoopInfo>(&*I, CCSetter);
6784 else
6785 return nullptr; // Unable to find the CC setter, so unable to guarantee
6786 // that pipeline will work
6787 }
6788
6789 // Recognize:
6790 // preheader:
6791 // %1 = t2DoopLoopStart %0
6792 // loop:
6793 // %2 = phi %1, <not loop>, %..., %loop
6794 // %3 = t2LoopDec %2, <imm>
6795 // t2LoopEnd %3, %loop
6796
6797 if (I != LoopBB->end() && I->getOpcode() == ARM::t2LoopEnd) {
6798 for (auto &L : LoopBB->instrs())
6799 if (L.isCall())
6800 return nullptr;
6801 else if (isVCTP(&L))
6802 return nullptr;
6803 Register LoopDecResult = I->getOperand(0).getReg();
6804 MachineRegisterInfo &MRI = LoopBB->getParent()->getRegInfo();
6805 MachineInstr *LoopDec = MRI.getUniqueVRegDef(LoopDecResult);
6806 if (!LoopDec || LoopDec->getOpcode() != ARM::t2LoopDec)
6807 return nullptr;
6808 MachineInstr *LoopStart = nullptr;
6809 for (auto &J : Preheader->instrs())
6810 if (J.getOpcode() == ARM::t2DoLoopStart)
6811 LoopStart = &J;
6812 if (!LoopStart)
6813 return nullptr;
6814 return std::make_unique<ARMPipelinerLoopInfo>(&*I, LoopDec);
6815 }
6816 return nullptr;
6817}
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
MachineOutlinerMBBFlags
@ LRUnavailableSomewhere
@ UnsafeRegsDead
MachineOutlinerClass
Constants defining how certain sequences should be outlined.
@ MachineOutlinerTailCall
Emit a save, restore, call, and return.
@ MachineOutlinerRegSave
Emit a call and tail-call.
@ MachineOutlinerNoLRSave
Only emit a branch.
@ MachineOutlinerThunk
Emit a call and return.
@ MachineOutlinerDefault
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
static bool isThumb(const MCSubtargetInfo &STI)
static bool getImplicitSPRUseForDPRUse(const TargetRegisterInfo *TRI, MachineInstr &MI, MCRegister DReg, unsigned Lane, MCRegister &ImplicitSReg)
getImplicitSPRUseForDPRUse - Given a use of a DPR register and lane, set ImplicitSReg to a register n...
static const MachineInstr * getBundledUseMI(const TargetRegisterInfo *TRI, const MachineInstr &MI, unsigned Reg, unsigned &UseIdx, unsigned &Dist)
static unsigned duplicateCPV(MachineFunction &MF, unsigned &CPI)
Create a copy of a const pool value.
static bool isSuitableForMask(MachineInstr *&MI, Register SrcReg, int CmpMask, bool CommonUse)
isSuitableForMask - Identify a suitable 'and' instruction that operates on the given source register ...
static int adjustDefLatency(const ARMSubtarget &Subtarget, const MachineInstr &DefMI, const MCInstrDesc &DefMCID, unsigned DefAlign)
Return the number of cycles to add to (or subtract from) the static itinerary based on the def opcode...
static unsigned getNumMicroOpsSwiftLdSt(const InstrItineraryData *ItinData, const MachineInstr &MI)
static MCRegister getCorrespondingDRegAndLane(const TargetRegisterInfo *TRI, unsigned SReg, unsigned &Lane)
static bool CanTransformInstrIntoTailCall(const MachineInstr &MI)
Return true if MI is a call instruction that the outliner can rewrite as a tail call.
static const AddSubFlagsOpcodePair AddSubFlagsOpcodeMap[]
static bool isEligibleForITBlock(const MachineInstr *MI)
static ARMCC::CondCodes getCmpToAddCondition(ARMCC::CondCodes CC)
getCmpToAddCondition - assume the flags are set by CMP(a,b), return the condition code if we modify t...
static bool isOptimizeCompareCandidate(MachineInstr *MI, bool &IsThumb1)
static bool isLRAvailable(const TargetRegisterInfo &TRI, MachineBasicBlock::reverse_iterator I, MachineBasicBlock::reverse_iterator E)
static const ARM_MLxEntry ARM_MLxTable[]
static bool isRedundantFlagInstr(const MachineInstr *CmpI, Register SrcReg, Register SrcReg2, int64_t ImmValue, const MachineInstr *OI, bool &IsThumb1)
isRedundantFlagInstr - check whether the first instruction, whose only purpose is to update flags,...
static unsigned getNumMicroOpsSingleIssuePlusExtras(unsigned Opc, unsigned NumRegs)
static const MachineInstr * getBundledDefMI(const TargetRegisterInfo *TRI, const MachineInstr *MI, unsigned Reg, unsigned &DefIdx, unsigned &Dist)
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
This file contains the simple types necessary to represent the attributes associated with functions a...
static const Function * getParent(const Value *V)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
DXIL Forward Handle Accesses
This file defines the DenseMap class.
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Module.h This file contains the declarations for the Module class.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
This file declares the MachineConstantPool class which is an abstract constant pool to keep track of ...
TargetInstrInfo::RegSubRegPair RegSubRegPair
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
uint64_t IntrinsicInst * II
#define P(N)
PowerPC TLS Dynamic Call Fixup
TargetInstrInfo::RegSubRegPairAndIdx RegSubRegPairAndIdx
const SmallVectorImpl< MachineOperand > MachineBasicBlock * TBB
const SmallVectorImpl< MachineOperand > & Cond
This file contains some templates that are useful if you are working with the STL at all.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file defines the SmallSet class.
This file defines the SmallVector class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
static X86::CondCode getSwappedCondition(X86::CondCode CC)
Assuming the flags are set by MI(a,b), return the condition code if we modify the instructions such t...
static bool isCPSRDefined(const MachineInstr &MI)
const MachineOperand & getCalleeOperand(const MachineInstr &MI) const override
bool optimizeCompareInstr(MachineInstr &CmpInstr, Register SrcReg, Register SrcReg2, int64_t CmpMask, int64_t CmpValue, const MachineRegisterInfo *MRI) const override
optimizeCompareInstr - Convert the instruction to set the zero flag so that we can remove a "comparis...
bool reverseBranchCondition(SmallVectorImpl< MachineOperand > &Cond) const override
ScheduleHazardRecognizer * CreateTargetHazardRecognizer(const TargetSubtargetInfo *STI, const ScheduleDAG *DAG) const override
void copyPhysReg(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register DestReg, Register SrcReg, bool KillSrc, bool RenamableDest=false, bool RenamableSrc=false) const override
bool foldImmediate(MachineInstr &UseMI, MachineInstr &DefMI, Register Reg, MachineRegisterInfo *MRI) const override
foldImmediate - 'Reg' is known to be defined by a move immediate instruction, try to fold the immedia...
std::pair< unsigned, unsigned > decomposeMachineOperandsTargetFlags(unsigned TF) const override
bool isProfitableToIfCvt(MachineBasicBlock &MBB, unsigned NumCycles, unsigned ExtraPredCycles, BranchProbability Probability) const override
bool ClobbersPredicate(MachineInstr &MI, std::vector< MachineOperand > &Pred, bool SkipDead) const override
void loadRegFromStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register DestReg, int FrameIndex, const TargetRegisterClass *RC, Register VReg, unsigned SubReg=0, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
void copyFromCPSR(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, MCRegister DestReg, bool KillSrc, const ARMSubtarget &Subtarget) const
unsigned getNumMicroOps(const InstrItineraryData *ItinData, const MachineInstr &MI) const override
std::optional< RegImmPair > isAddImmediate(const MachineInstr &MI, Register Reg) const override
unsigned getPartialRegUpdateClearance(const MachineInstr &, unsigned, const TargetRegisterInfo *) const override
unsigned getNumLDMAddresses(const MachineInstr &MI) const
Get the number of addresses by LDM or VLDM or zero for unknown.
MachineInstr * optimizeSelect(MachineInstr &MI, SmallPtrSetImpl< MachineInstr * > &SeenMIs, bool) const override
bool produceSameValue(const MachineInstr &MI0, const MachineInstr &MI1, const MachineRegisterInfo *MRI) const override
void setExecutionDomain(MachineInstr &MI, unsigned Domain) const override
ArrayRef< std::pair< unsigned, const char * > > getSerializableBitmaskMachineOperandTargetFlags() const override
ScheduleHazardRecognizer * CreateTargetPostRAHazardRecognizer(const InstrItineraryData *II, const ScheduleDAG *DAG) const override
std::unique_ptr< TargetInstrInfo::PipelinerLoopInfo > analyzeLoopForPipelining(MachineBasicBlock *LoopBB) const override
Analyze loop L, which must be a single-basic-block loop, and if the conditions can be understood enou...
unsigned getInstSizeInBytes(const MachineInstr &MI) const override
GetInstSize - Returns the size of the specified MachineInstr.
void copyToCPSR(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, MCRegister SrcReg, bool KillSrc, const ARMSubtarget &Subtarget) const
unsigned removeBranch(MachineBasicBlock &MBB, int *BytesRemoved=nullptr) const override
void mergeOutliningCandidateAttributes(Function &F, std::vector< outliner::Candidate > &Candidates) const override
const MachineInstrBuilder & AddDReg(MachineInstrBuilder &MIB, unsigned Reg, unsigned SubIdx, RegState State) const
bool isFunctionSafeToOutlineFrom(MachineFunction &MF, bool OutlineFromLinkOnceODRs) const override
ARM supports the MachineOutliner.
bool shouldOutlineFromFunctionByDefault(MachineFunction &MF) const override
Enable outlining by default at -Oz.
std::optional< DestSourcePair > isCopyInstrImpl(const MachineInstr &MI) const override
If the specific machine instruction is an instruction that moves/copies value from one register to an...
MachineInstr & duplicate(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertBefore, const MachineInstr &Orig) const override
ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *II, const ScheduleDAGMI *DAG) const override
MachineBasicBlock::iterator insertOutlinedCall(Module &M, MachineBasicBlock &MBB, MachineBasicBlock::iterator &It, MachineFunction &MF, outliner::Candidate &C) const override
std::string createMIROperandComment(const MachineInstr &MI, const MachineOperand &Op, unsigned OpIdx, const TargetRegisterInfo *TRI) const override
bool isPredicated(const MachineInstr &MI) const override
bool isSchedulingBoundary(const MachineInstr &MI, const MachineBasicBlock *MBB, const MachineFunction &MF) const override
void expandLoadStackGuardBase(MachineBasicBlock::iterator MI, unsigned LoadImmOpc, unsigned LoadOpc) const
bool isPredicable(const MachineInstr &MI) const override
isPredicable - Return true if the specified instruction can be predicated.
Register isLoadFromStackSlotPostFE(const MachineInstr &MI, int &FrameIndex) const override
std::optional< ParamLoadedValue > describeLoadedValue(const MachineInstr &MI, Register Reg) const override
Specialization of TargetInstrInfo::describeLoadedValue, used to enhance debug entry value description...
std::optional< std::unique_ptr< outliner::OutlinedFunction > > getOutliningCandidateInfo(const MachineModuleInfo &MMI, std::vector< outliner::Candidate > &RepeatedSequenceLocs, unsigned MinRepeats) const override
bool analyzeBranch(MachineBasicBlock &MBB, MachineBasicBlock *&TBB, MachineBasicBlock *&FBB, SmallVectorImpl< MachineOperand > &Cond, bool AllowModify=false) const override
unsigned extraSizeToPredicateInstructions(const MachineFunction &MF, unsigned NumInsts) const override
unsigned insertBranch(MachineBasicBlock &MBB, MachineBasicBlock *TBB, MachineBasicBlock *FBB, ArrayRef< MachineOperand > Cond, const DebugLoc &DL, int *BytesAdded=nullptr) const override
const ARMBaseRegisterInfo & getRegisterInfo() const
bool areLoadsFromSameBasePtr(SDNode *Load1, SDNode *Load2, int64_t &Offset1, int64_t &Offset2) const override
areLoadsFromSameBasePtr - This is used by the pre-regalloc scheduler to determine if two loads are lo...
std::optional< unsigned > getOperandLatency(const InstrItineraryData *ItinData, const MachineInstr &DefMI, unsigned DefIdx, const MachineInstr &UseMI, unsigned UseIdx) const override
bool getRegSequenceLikeInputs(const MachineInstr &MI, unsigned DefIdx, SmallVectorImpl< RegSubRegPairAndIdx > &InputRegs) const override
Build the equivalent inputs of a REG_SEQUENCE for the given MI and DefIdx.
unsigned predictBranchSizeForIfCvt(MachineInstr &MI) const override
bool getInsertSubregLikeInputs(const MachineInstr &MI, unsigned DefIdx, RegSubRegPair &BaseReg, RegSubRegPairAndIdx &InsertedReg) const override
Build the equivalent inputs of a INSERT_SUBREG for the given MI and DefIdx.
bool expandPostRAPseudo(MachineInstr &MI) const override
outliner::InstrType getOutliningTypeImpl(const MachineModuleInfo &MMI, MachineBasicBlock::iterator &MIT, unsigned Flags) const override
bool SubsumesPredicate(ArrayRef< MachineOperand > Pred1, ArrayRef< MachineOperand > Pred2) const override
bool shouldScheduleLoadsNear(SDNode *Load1, SDNode *Load2, int64_t Offset1, int64_t Offset2, unsigned NumLoads) const override
shouldScheduleLoadsNear - This is a used by the pre-regalloc scheduler to determine (in conjunction w...
bool PredicateInstruction(MachineInstr &MI, ArrayRef< MachineOperand > Pred) const override
std::pair< uint16_t, uint16_t > getExecutionDomain(const MachineInstr &MI) const override
VFP/NEON execution domains.
bool isProfitableToUnpredicate(MachineBasicBlock &TMBB, MachineBasicBlock &FMBB) const override
void storeRegToStackSlot(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, Register SrcReg, bool isKill, int FrameIndex, const TargetRegisterClass *RC, Register VReg, MachineInstr::MIFlag Flags=MachineInstr::NoFlags) const override
bool isFpMLxInstruction(unsigned Opcode) const
isFpMLxInstruction - Return true if the specified opcode is a fp MLA / MLS instruction.
bool isSwiftFastImmShift(const MachineInstr *MI) const
Returns true if the instruction has a shift by immediate that can be executed in one cycle less.
void reMaterialize(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, Register DestReg, unsigned SubIdx, const MachineInstr &Orig, LaneBitmask UsedLanes=LaneBitmask::getAll()) const override
ARMBaseInstrInfo(const ARMSubtarget &STI, const ARMBaseRegisterInfo &TRI)
ArrayRef< std::pair< unsigned, const char * > > getSerializableDirectMachineOperandTargetFlags() const override
Register isStoreToStackSlotPostFE(const MachineInstr &MI, int &FrameIndex) const override
bool analyzeCompare(const MachineInstr &MI, Register &SrcReg, Register &SrcReg2, int64_t &CmpMask, int64_t &CmpValue) const override
analyzeCompare - For a comparison instruction, return the source registers in SrcReg and SrcReg2 if h...
Register isStoreToStackSlot(const MachineInstr &MI, int &FrameIndex) const override
void breakPartialRegDependency(MachineInstr &, unsigned, const TargetRegisterInfo *TRI) const override
bool isMBBSafeToOutlineFrom(MachineBasicBlock &MBB, unsigned &Flags) const override
void buildOutlinedFrame(MachineBasicBlock &MBB, MachineFunction &MF, const outliner::OutlinedFunction &OF) const override
Register isLoadFromStackSlot(const MachineInstr &MI, int &FrameIndex) const override
const ARMSubtarget & getSubtarget() const
MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx1, unsigned OpIdx2) const override
Commutes the operands in the given instruction.
bool getExtractSubregLikeInputs(const MachineInstr &MI, unsigned DefIdx, RegSubRegPairAndIdx &InputReg) const override
Build the equivalent inputs of a EXTRACT_SUBREG for the given MI and DefIdx.
bool shouldSink(const MachineInstr &MI) const override
BitVector getReservedRegs(const MachineFunction &MF) const override
static ARMConstantPoolConstant * Create(const Constant *C, unsigned ID)
static ARMConstantPoolMBB * Create(LLVMContext &C, const MachineBasicBlock *mbb, unsigned ID, unsigned char PCAdj)
static ARMConstantPoolSymbol * Create(LLVMContext &C, StringRef s, unsigned ID, unsigned char PCAdj, ARMCP::ARMCPModifier Modifier=ARMCP::no_modifier, bool AddCurrentAddress=false)
ARMConstantPoolValue - ARM specific constantpool value.
ARMCP::ARMCPModifier getModifier() const
virtual bool hasSameValue(ARMConstantPoolValue *ACPV)
hasSameValue - Return true if this ARM constpool value can share the same constantpool entry as anoth...
ARMFunctionInfo - This class is derived from MachineFunctionInfo and contains private ARM-specific in...
bool isCortexA7() const
bool isSwift() const
const ARMBaseInstrInfo * getInstrInfo() const override
bool isThumb1Only() const
bool isThumb2() const
bool isLikeA9() const
Align getStackAlignment() const
getStackAlignment - Returns the minimum alignment known to hold of the stack frame on entry to the fu...
bool enableMachinePipeliner() const override
Returns true if machine pipeliner should be enabled.
bool hasMinSize() const
bool isCortexA8() const
@ DoubleIssueCheckUnalignedAccess
Can load/store 2 registers/cycle, but needs an extra cycle if the access is not 64-bit aligned.
@ SingleIssue
Can load/store 1 register/cycle.
@ DoubleIssue
Can load/store 2 registers/cycle.
@ SingleIssuePlusExtras
Can load/store 1 register/cycle, but needs an extra cycle for address computation and potentially als...
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool test(unsigned Idx) const
Returns true if bit Idx is set.
Definition BitVector.h:482
size_type size() const
Returns the number of bits in this bitvector.
Definition BitVector.h:178
LLVM_ABI uint64_t scale(uint64_t Num) const
Scale a large integer.
BranchProbability getCompl() const
Helper class for creating CFI instructions and inserting them into MIR.
void buildRegister(MCRegister Reg1, MCRegister Reg2) const
void buildRestore(MCRegister Reg) const
ConstMIBundleOperands - Iterate over all operands in a const bundle of machine instructions.
A debug info location.
Definition DebugLoc.h:126
bool hasOptSize() const
Optimize this function for size (-Os) or minimum size (-Oz).
Definition Function.h:699
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
Definition Function.cpp:765
bool hasMinSize() const
Optimize this function for minimum size (-Oz).
Definition Function.h:696
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:730
bool hasDLLImportStorageClass() const
bool reverseBranchCondition(SmallVectorImpl< MachineOperand > &Cond) const override
Reverses the branch condition of the specified condition list, returning false on success and true if...
Itinerary data supplied by a subtarget to be used by a target.
int getNumMicroOps(unsigned ItinClassIndx) const
Return the number of micro-ops that the given class decodes to.
std::optional< unsigned > getOperandCycle(unsigned ItinClassIndx, unsigned OperandIdx) const
Return the cycle for the given class and operand.
unsigned getStageLatency(unsigned ItinClassIndx) const
Return the total stage latency of the given class.
std::optional< unsigned > getOperandLatency(unsigned DefClass, unsigned DefIdx, unsigned UseClass, unsigned UseIdx) const
Compute and return the use operand latency of a given itinerary class and operand index if the value ...
bool hasPipelineForwarding(unsigned DefClass, unsigned DefIdx, unsigned UseClass, unsigned UseIdx) const
Return true if there is a pipeline forwarding between instructions of itinerary classes DefClass and ...
bool isEmpty() const
Returns true if there are no itineraries.
A set of register units used to track register liveness.
bool available(MCRegister Reg) const
Returns true if no part of physical register Reg is live.
LLVM_ABI void addLiveOuts(const MachineBasicBlock &MBB)
Adds registers living out of block MBB.
LLVM_ABI void accumulate(const MachineInstr &MI)
Adds all register units used, defined or clobbered in MI.
This class is intended to be used as a base class for asm properties and features specific to the tar...
Definition MCAsmInfo.h:67
Describe properties that are true of each instruction in the target description file.
unsigned getSchedClass() const
Return the scheduling class for this instruction.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
bool mayLoad() const
Return true if this instruction could possibly read memory.
bool hasOptionalDef() const
Set if this instruction has an optional definition, e.g.
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
bool isCall() const
Return true if the instruction is a call.
unsigned getOpcode() const
Return the opcode number for this descriptor.
LLVM_ABI bool hasImplicitDefOfPhysReg(MCRegister Reg, const MCRegisterInfo *MRI=nullptr) const
Return true if this instruction implicitly defines the specified physical register.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
bool isValid() const
isValid - Returns true until all the operands have been visited.
MachineInstrBundleIterator< const MachineInstr > const_iterator
LLVM_ABI instr_iterator insert(instr_iterator I, MachineInstr *M)
Insert MI into the instruction list before I, possibly inside a bundle.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
Instructions::iterator instr_iterator
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
Instructions::const_iterator const_instr_iterator
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LLVM_ABI instr_iterator erase(instr_iterator I)
Remove an instruction from the instruction list and delete it.
MachineInstrBundleIterator< MachineInstr > iterator
LivenessQueryResult
Possible outcome of a register liveness query to computeRegisterLiveness()
@ LQR_Dead
Register is known to be fully dead.
@ LQR_Live
Register is known to be (at least partially) live.
@ LQR_Unknown
Register liveness not decidable from local neighborhood.
This class is a data container for one entry in a MachineConstantPool.
union llvm::MachineConstantPoolEntry::@004270020304201266316354007027341142157160323045 Val
The constant itself.
bool isMachineConstantPoolEntry() const
isMachineConstantPoolEntry - Return true if the MachineConstantPoolEntry is indeed a target specific ...
MachineConstantPoolValue * MachineCPVal
The MachineConstantPool class keeps track of constants referenced by a function which must be spilled...
const std::vector< MachineConstantPoolEntry > & getConstants() const
LLVM_ABI unsigned getConstantPoolIndex(const Constant *C, Align Alignment)
getConstantPoolIndex - Create a new entry in the constant pool or return an existing one.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
uint64_t getStackSize() const
Return the number of bytes that must be allocated to hold all of the fixed size frame objects.
bool isCalleeSavedInfoValid() const
Has the callee saved info been calculated yet?
Align getObjectAlign(int ObjectIdx) const
Return the alignment of the specified stack object.
int64_t getObjectSize(int ObjectIdx) const
Return the size of the specified object.
unsigned getNumObjects() const
Return the number of objects.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
StringRef getName() const
getName - Return the name of the corresponding LLVM function.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineConstantPool * getConstantPool()
getConstantPool - Return the constant pool object for the current function.
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addConstantPoolIndex(unsigned Idx, int Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
ArrayRef< MachineMemOperand * >::iterator mmo_iterator
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool isImplicitDef() const
const MachineBasicBlock * getParent() const
bool isCopyLike() const
Return true if the instruction behaves like a copy.
bool isCall(QueryType Type=AnyInBundle) const
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI int findFirstPredOperandIdx() const
Find the index of the first operand in the operand list that is used to represent the predicate.
const MCInstrDesc & getDesc() const
Returns the target instruction descriptor of this MachineInstr.
bool isRegSequence() const
bool isInsertSubreg() const
LLVM_ABI void tieOperands(unsigned DefIdx, unsigned UseIdx)
Add a tie between the register operands at DefIdx and UseIdx.
LLVM_ABI bool isIdenticalTo(const MachineInstr &Other, MICheckType Check=CheckDefs) const
Return true if this instruction is identical to Other.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI bool addRegisterKilled(Register IncomingReg, const TargetRegisterInfo *RegInfo, bool AddIfNotFound=false)
We have determined MI kills a register.
bool hasOptionalDef(QueryType Type=IgnoreBundle) const
Set if this instruction has an optional definition, e.g.
LLVM_ABI void addRegisterDefined(Register Reg, const TargetRegisterInfo *RegInfo=nullptr)
We have determined MI defines a register.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI void clearKillInfo()
Clears kill flags on all operands.
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
A description of a memory reference used in the backend.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOLoad
The memory access reads data.
@ MOInvariant
The memory access always returns the same value (or traps).
@ MOStore
The memory access writes data.
This class contains meta information specific to a module.
LLVM_ABI MachineFunction * getMachineFunction(const Function &F) const
Returns the MachineFunction associated to IR function F if there is one, otherwise nullptr.
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
const GlobalValue * getGlobal() const
void setImplicit(bool Val=true)
void setImm(int64_t immVal)
int64_t getImm() const
bool readsReg() const
readsReg - Returns true if this operand reads the previous value of its register.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isRegMask() const
isRegMask - Tests if this is a MO_RegisterMask operand.
MachineBasicBlock * getMBB() const
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
static MachineOperand CreateImm(int64_t Val)
Register getReg() const
getReg - Returns the register number.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
static bool clobbersPhysReg(const uint32_t *RegMask, MCRegister PhysReg)
clobbersPhysReg - Returns true if this RegMask clobbers PhysReg.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
int64_t getOffset() const
Return the offset from the symbol in this operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
defusechain_instr_iterator< true, false, false, true > use_instr_iterator
use_instr_iterator/use_instr_begin/use_instr_end - Walk all uses of the specified register,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
use_instr_iterator use_instr_begin(Register RegNo) const
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
static use_instr_iterator use_instr_end()
const TargetRegisterInfo * getTargetRegisterInfo() const
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI LLVM_READONLY MachineInstr * getUniqueVRegDef(Register Reg) const
getUniqueVRegDef - Return the unique machine instr that defines the specified virtual register or nul...
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
void AddHazardRecognizer(std::unique_ptr< ScheduleHazardRecognizer > &&)
Track the current register pressure at some position in the instruction stream, and remember the high...
LLVM_ABI void increaseRegPressure(VirtRegOrUnit VRegOrUnit, LaneBitmask PreviousMask, LaneBitmask NewMask)
LLVM_ABI void decreaseRegPressure(VirtRegOrUnit VRegOrUnit, LaneBitmask PreviousMask, LaneBitmask NewMask)
unsigned getRegPressureSetLimit(unsigned Idx) const
Get the register unit limit for the given pressure set index.
LLVM_ABI void runOnMachineFunction(const MachineFunction &MF, bool Rev=false)
runOnFunction - Prepare to answer questions about MF.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
static constexpr bool isPhysicalRegister(unsigned Reg)
Return true if the specified register number is in the physical register namespace.
Definition Register.h:60
constexpr unsigned id() const
Definition Register.h:100
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
@ Anti
A register anti-dependence (aka WAR).
Definition ScheduleDAG.h:58
This class represents the scheduled code.
unsigned getMaxStageCount()
Return the maximum stage count needed for this schedule.
int stageScheduled(SUnit *SU) const
Return the stage for a scheduled instruction.
int getInitiationInterval() const
Return the initiation interval for this schedule.
std::deque< SUnit * > & getInstructions(int cycle)
Return the instructions that are scheduled at the specified cycle.
int getFirstCycle() const
Return the first cycle in the completed schedule.
int getFinalCycle() const
Return the last cycle in the finalized schedule.
Scheduling unit. This is a node in the scheduling DAG.
ScheduleDAGMI is an implementation of ScheduleDAGInstrs that simply schedules machine instructions ac...
virtual bool hasVRegLiveness() const
Return true if this DAG supports VReg liveness and RegPressure.
std::vector< SUnit > SUnits
The scheduling units.
HazardRecognizer - This determines whether or not an instruction can be issued this cycle,...
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
bool erase(PtrType Ptr)
Remove pointer from the set.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
Definition SmallSet.h:134
size_type count(const T &V) const
count - Return 1 if the element is in the set, 0 otherwise.
Definition SmallSet.h:176
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
This class builds the dependence graph for the instructions in a loop, and attempts to schedule the i...
Object returned by analyzeLoopForPipelining.
TargetInstrInfo - Interface to description of machine instruction set.
virtual ScheduleHazardRecognizer * CreateTargetPostRAHazardRecognizer(const InstrItineraryData *, const ScheduleDAG *DAG) const
Allocate and return a hazard recognizer to use for this target when scheduling the machine instructio...
virtual ScheduleHazardRecognizer * CreateTargetMIHazardRecognizer(const InstrItineraryData *, const ScheduleDAGMI *DAG) const
Allocate and return a hazard recognizer to use for this target when scheduling the machine instructio...
virtual std::optional< ParamLoadedValue > describeLoadedValue(const MachineInstr &MI, Register Reg) const
Produce the expression describing the MI loading a value into the physical register Reg.
virtual ScheduleHazardRecognizer * CreateTargetHazardRecognizer(const TargetSubtargetInfo *STI, const ScheduleDAG *DAG) const
Allocate and return a hazard recognizer to use for this target when scheduling the machine instructio...
virtual bool isReMaterializableImpl(const MachineInstr &MI) const
For instructions with opcodes for which the M_REMATERIALIZABLE flag is set, this hook lets the target...
virtual MachineInstr & duplicate(MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertBefore, const MachineInstr &Orig) const
Clones instruction or the whole instruction bundle Orig and insert into MBB before InsertBefore.
virtual const MachineOperand & getCalleeOperand(const MachineInstr &MI) const
Returns the callee operand from the given MI.
virtual MachineInstr * commuteInstructionImpl(MachineInstr &MI, bool NewMI, unsigned OpIdx1, unsigned OpIdx2) const
This method commutes the operands of the given machine instruction MI.
virtual std::string createMIROperandComment(const MachineInstr &MI, const MachineOperand &Op, unsigned OpIdx, const TargetRegisterInfo *TRI) const
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
Provide an instruction scheduling machine model to CodeGen passes.
LLVM_ABI unsigned computeOperandLatency(const MachineInstr *DefMI, unsigned DefOperIdx, const MachineInstr *UseMI, unsigned UseOperIdx) const
Compute operand latency based on the available machine model.
const InstrItineraryData * getInstrItineraries() const
TargetSubtargetInfo - Generic base class for all target subtargets.
virtual const TargetRegisterInfo * getRegisterInfo() const =0
Return the target's register information.
Wrapper class representing a virtual register or register unit.
Definition Register.h:175
self_iterator getIterator()
Definition ilist_node.h:123
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
static CondCodes getOppositeCondition(CondCodes CC)
Definition ARMBaseInfo.h:49
ARMII - This namespace holds all of the target specific flags that instruction info tracks.
@ ThumbArithFlagSetting
@ MO_OPTION_MASK
MO_OPTION_MASK - Most flags are mutually exclusive; this mask selects just that part of the flag set.
@ MO_NONLAZY
MO_NONLAZY - This is an independent flag, on a symbol operand "FOO" it represents a symbol which,...
@ MO_DLLIMPORT
MO_DLLIMPORT - On a symbol operand, this represents that the reference to the symbol is for an import...
@ MO_GOT
MO_GOT - On a symbol operand, this represents a GOT relative relocation.
@ MO_COFFSTUB
MO_COFFSTUB - On a symbol operand "FOO", this indicates that the reference is actually to the "....
AddrMode
ARM Addressing Modes.
unsigned char getAM3Offset(unsigned AM3Opc)
unsigned char getAM5FP16Offset(unsigned AM5Opc)
unsigned getSORegOffset(unsigned Op)
int getSOImmVal(unsigned Arg)
getSOImmVal - Given a 32-bit immediate, if it is something that can fit into an shifter_operand immed...
ShiftOpc getAM2ShiftOpc(unsigned AM2Opc)
unsigned getAM2Offset(unsigned AM2Opc)
unsigned getSOImmValRotate(unsigned Imm)
getSOImmValRotate - Try to handle Imm with an immediate shifter operand, computing the rotate amount ...
bool isThumbImmShiftedVal(unsigned V)
isThumbImmShiftedVal - Return true if the specified value can be obtained by left shifting a 8-bit im...
int getT2SOImmVal(unsigned Arg)
getT2SOImmVal - Given a 32-bit immediate, if it is something that can fit into a Thumb-2 shifter_oper...
ShiftOpc getSORegShOp(unsigned Op)
AddrOpc getAM5Op(unsigned AM5Opc)
bool isSOImmTwoPartValNeg(unsigned V)
isSOImmTwoPartValNeg - Return true if the specified value can be obtained by two SOImmVal,...
unsigned getSOImmTwoPartSecond(unsigned V)
getSOImmTwoPartSecond - If V is a value that satisfies isSOImmTwoPartVal, return the second chunk of ...
bool isSOImmTwoPartVal(unsigned V)
isSOImmTwoPartVal - Return true if the specified value can be obtained by or'ing together two SOImmVa...
AddrOpc getAM5FP16Op(unsigned AM5Opc)
unsigned getT2SOImmTwoPartSecond(unsigned Imm)
unsigned getT2SOImmTwoPartFirst(unsigned Imm)
bool isT2SOImmTwoPartVal(unsigned Imm)
unsigned char getAM5Offset(unsigned AM5Opc)
unsigned getSOImmTwoPartFirst(unsigned V)
getSOImmTwoPartFirst - If V is a value that satisfies isSOImmTwoPartVal, return the first chunk of it...
AddrOpc getAM2Op(unsigned AM2Opc)
AddrOpc getAM3Op(unsigned AM3Opc)
Define some predicates that are used for node matching.
Definition ARMEHABI.h:25
InstrType
Represents how an instruction should be mapped by the outliner.
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Offset
Definition DWP.cpp:577
constexpr T rotr(T V, int R)
Definition bit.h:399
static bool isIndirectCall(const MachineInstr &MI)
MachineInstr * findCMPToFoldIntoCBZ(MachineInstr *Br, const TargetRegisterInfo *TRI)
Search backwards from a tBcc to find a tCMPi8 against 0, meaning we can convert them to a tCBZ or tCB...
static bool isCondBranchOpcode(int Opc)
bool HasLowerConstantMaterializationCost(unsigned Val1, unsigned Val2, const ARMSubtarget *Subtarget, bool ForCodesize=false)
Returns true if Val1 has a lower Constant Materialization Cost than Val2.
static bool isPushOpcode(int Opc)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
void addPredicatedMveVpredNOp(MachineInstrBuilder &MIB, unsigned Cond)
static bool isVCTP(const MachineInstr *MI)
RegState
Flags to represent properties of register accesses.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
bool IsCPSRDead< MachineInstr >(const MachineInstr *MI)
constexpr RegState getKillRegState(bool B)
unsigned getBLXpredOpcode(const MachineFunction &MF)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
static bool isARMLowRegister(MCRegister Reg)
isARMLowRegister - Returns true if the register is a low register (r0-r7).
static bool isIndirectBranchOpcode(int Opc)
bool isLegalAddressImm(unsigned Opcode, int Imm, const TargetInstrInfo *TII)
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
bool registerDefinedBetween(unsigned Reg, MachineBasicBlock::iterator From, MachineBasicBlock::iterator To, const TargetRegisterInfo *TRI)
Return true if Reg is defd between From and To.
static std::array< MachineOperand, 2 > predOps(ARMCC::CondCodes Pred, unsigned PredReg=0)
Get the operands corresponding to the given Pred value.
Op::Description Desc
static bool isSEHInstruction(const MachineInstr &MI)
static bool isCalleeSavedRegister(MCRegister Reg, const MCPhysReg *CSRegs)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
bool tryFoldSPUpdateIntoPushPop(const ARMSubtarget &Subtarget, MachineFunction &MF, MachineInstr *MI, unsigned NumBytes)
Tries to add registers to the reglist of a given base-updating push/pop instruction to adjust the sta...
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
static bool isJumpTableBranchOpcode(int Opc)
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
static bool isPopOpcode(int Opc)
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
void addPredicatedMveVpredROp(MachineInstrBuilder &MIB, unsigned Cond, unsigned Inactive)
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
void addUnpredicatedMveVpredROp(MachineInstrBuilder &MIB, Register DestReg)
unsigned ConstantMaterializationCost(unsigned Val, const ARMSubtarget *Subtarget, bool ForCodesize=false)
Returns the number of instructions required to materialize the given constant in a register,...
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
bool rewriteARMFrameIndex(MachineInstr &MI, unsigned FrameRegIdx, Register FrameReg, int &Offset, const ARMBaseInstrInfo &TII)
rewriteARMFrameIndex / rewriteT2FrameIndex - Rewrite MI to access 'Offset' bytes from the FP.
static bool isIndirectControlFlowNotComingBack(const MachineInstr &MI)
ARMCC::CondCodes getInstrPredicate(const MachineInstr &MI, Register &PredReg)
getInstrPredicate - If instruction is predicated, returns its predicate condition,...
unsigned getMatchingCondBranchOpcode(unsigned Opc)
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
static bool isUncondBranchOpcode(int Opc)
auto partition(R &&Range, UnaryPredicate P)
Provide wrappers to std::partition which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:2049
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
static const char * ARMCondCodeToString(ARMCC::CondCodes CC)
static MachineOperand condCodeOp(unsigned CCReg=0)
Get the operand corresponding to the conditional code result.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
unsigned gettBLXrOpcode(const MachineFunction &MF)
static bool isSpeculationBarrierEndBBOpcode(int Opc)
unsigned getBLXOpcode(const MachineFunction &MF)
void addUnpredicatedMveVpredNOp(MachineInstrBuilder &MIB)
bool isV8EligibleForIT(const InstrType *Instr)
Definition ARMFeatures.h:24
void emitARMRegPlusImmediate(MachineBasicBlock &MBB, MachineBasicBlock::iterator &MBBI, const DebugLoc &dl, Register DestReg, Register BaseReg, int NumBytes, ARMCC::CondCodes Pred, Register PredReg, const ARMBaseInstrInfo &TII, unsigned MIFlags=0)
emitARMRegPlusImmediate / emitT2RegPlusImmediate - Emits a series of instructions to materializea des...
constexpr RegState getUndefRegState(bool B)
unsigned convertAddSubFlagsOpcode(unsigned OldOpc)
Map pseudo instructions that imply an 'S' bit onto real opcodes.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
#define N
ARM_MLxEntry - Record information about MLA / MLS instructions.
Map pseudo instructions that imply an 'S' bit onto real opcodes.
OutlinerCosts(const ARMSubtarget &target)
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
static constexpr LaneBitmask getAll()
Definition LaneBitmask.h:82
static constexpr LaneBitmask getNone()
Definition LaneBitmask.h:81
static LLVM_ABI MachinePointerInfo getGOT(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a GOT entry.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
Used to describe a register and immediate addition.
RegisterPressure computed within a region of instructions delimited by TopPos and BottomPos.
An individual sequence of instructions to be replaced with a call to an outlined function.
The information necessary to create an outlined function for some class of candidate.