LLVM 24.0.0git
ARMLoadStoreOptimizer.cpp
Go to the documentation of this file.
1//===- ARMLoadStoreOptimizer.cpp - ARM load / store opt. pass -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file This file contains a pass that performs load / store related peephole
10/// optimizations. This pass should be run after register allocation.
11//
12//===----------------------------------------------------------------------===//
13
14#include "ARM.h"
15#include "ARMBaseInstrInfo.h"
16#include "ARMBaseRegisterInfo.h"
17#include "ARMISelLowering.h"
19#include "ARMSubtarget.h"
22#include "Utils/ARMBaseInfo.h"
23#include "llvm/ADT/ArrayRef.h"
24#include "llvm/ADT/DenseMap.h"
25#include "llvm/ADT/DenseSet.h"
26#include "llvm/ADT/STLExtras.h"
27#include "llvm/ADT/SetVector.h"
29#include "llvm/ADT/SmallSet.h"
31#include "llvm/ADT/Statistic.h"
51#include "llvm/IR/DataLayout.h"
52#include "llvm/IR/DebugLoc.h"
53#include "llvm/IR/Function.h"
54#include "llvm/IR/Type.h"
56#include "llvm/MC/MCInstrDesc.h"
57#include "llvm/Pass.h"
60#include "llvm/Support/Debug.h"
63#include <cassert>
64#include <cstddef>
65#include <cstdlib>
66#include <iterator>
67#include <limits>
68#include <utility>
69
70using namespace llvm;
71
72#define DEBUG_TYPE "arm-ldst-opt"
73
74STATISTIC(NumLDMGened , "Number of ldm instructions generated");
75STATISTIC(NumSTMGened , "Number of stm instructions generated");
76STATISTIC(NumVLDMGened, "Number of vldm instructions generated");
77STATISTIC(NumVSTMGened, "Number of vstm instructions generated");
78STATISTIC(NumLdStMoved, "Number of load / store instructions moved");
79STATISTIC(NumLDRDFormed,"Number of ldrd created before allocation");
80STATISTIC(NumSTRDFormed,"Number of strd created before allocation");
81STATISTIC(NumLDRD2LDM, "Number of ldrd instructions turned back into ldm");
82STATISTIC(NumSTRD2STM, "Number of strd instructions turned back into stm");
83STATISTIC(NumLDRD2LDR, "Number of ldrd instructions turned back into ldr's");
84STATISTIC(NumSTRD2STR, "Number of strd instructions turned back into str's");
85
86/// This switch disables formation of double/multi instructions that could
87/// potentially lead to (new) alignment traps even with CCR.UNALIGN_TRP
88/// disabled. This can be used to create libraries that are robust even when
89/// users provoke undefined behaviour by supplying misaligned pointers.
90/// \see mayCombineMisaligned()
91static cl::opt<bool>
92AssumeMisalignedLoadStores("arm-assume-misaligned-load-store", cl::Hidden,
93 cl::init(false), cl::desc("Be more conservative in ARM load/store opt"));
94
95#define ARM_LOAD_STORE_OPT_NAME "ARM load / store optimization pass"
96
97namespace {
98
99/// Post- register allocation pass the combine load / store instructions to
100/// form ldm / stm instructions.
101struct ARMLoadStoreOpt {
102 const MachineFunction *MF;
103 const TargetInstrInfo *TII;
104 const TargetRegisterInfo *TRI;
105 const ARMSubtarget *STI;
106 const TargetLowering *TL;
107 ARMFunctionInfo *AFI;
109 const RegisterClassInfo *RCI = nullptr;
111 bool LiveRegsValid;
112 bool isThumb1, isThumb2;
113
114 bool runOnMachineFunction(MachineFunction &Fn,
115 const RegisterClassInfo &RegClassInfo);
116
117private:
118 /// A set of load/store MachineInstrs with same base register sorted by
119 /// offset.
120 struct MemOpQueueEntry {
122 int Offset; ///< Load/Store offset.
123 unsigned Position; ///< Position as counted from end of basic block.
124
125 MemOpQueueEntry(MachineInstr &MI, int Offset, unsigned Position)
126 : MI(&MI), Offset(Offset), Position(Position) {}
127 };
128 using MemOpQueue = SmallVector<MemOpQueueEntry, 8>;
129
130 /// A set of MachineInstrs that fulfill (nearly all) conditions to get
131 /// merged into a LDM/STM.
132 struct MergeCandidate {
133 /// List of instructions ordered by load/store offset.
135
136 /// Index in Instrs of the instruction being latest in the schedule.
137 unsigned LatestMIIdx;
138
139 /// Index in Instrs of the instruction being earliest in the schedule.
140 unsigned EarliestMIIdx;
141
142 /// Index into the basic block where the merged instruction will be
143 /// inserted. (See MemOpQueueEntry.Position)
144 unsigned InsertPos;
145
146 /// Whether the instructions can be merged into a ldm/stm instruction.
147 bool CanMergeToLSMulti;
148
149 /// Whether the instructions can be merged into a ldrd/strd instruction.
150 bool CanMergeToLSDouble;
151 };
154 SmallVector<MachineInstr *, 4> MergeBaseCandidates;
155
157
158 void moveLiveRegsBefore(const MachineBasicBlock &MBB,
160 unsigned findFreeReg(const TargetRegisterClass &RegClass);
161 void UpdateBaseRegUses(MachineBasicBlock &MBB,
163 unsigned Base, unsigned WordOffset,
164 ARMCC::CondCodes Pred, unsigned PredReg);
165 MachineInstr *CreateLoadStoreMulti(MachineBasicBlock &MBB,
166 MachineBasicBlock::iterator InsertBefore,
167 int Offset, unsigned Base, bool BaseKill,
168 unsigned Opcode, ARMCC::CondCodes Pred,
169 unsigned PredReg, const DebugLoc &DL,
170 ArrayRef<std::pair<unsigned, bool>> Regs,
172 MachineInstr *CreateLoadStoreDouble(MachineBasicBlock &MBB,
173 MachineBasicBlock::iterator InsertBefore,
174 int Offset, unsigned Base, bool BaseKill,
175 unsigned Opcode, ARMCC::CondCodes Pred,
176 unsigned PredReg, const DebugLoc &DL,
177 ArrayRef<std::pair<unsigned, bool>> Regs,
178 ArrayRef<MachineInstr *> Instrs) const;
179 void FormCandidates(const MemOpQueue &MemOps);
180 MachineInstr *MergeOpsUpdate(const MergeCandidate &Cand);
181 bool FixInvalidRegPairOp(MachineBasicBlock &MBB,
183 bool MergeBaseUpdateLoadStore(MachineInstr *MI);
184 bool MergeBaseUpdateLSMultiple(MachineInstr *MI);
185 bool MergeBaseUpdateLSDouble(MachineInstr &MI);
186 bool LoadStoreMultipleOpti(MachineBasicBlock &MBB);
187 bool MergeReturnIntoLDM(MachineBasicBlock &MBB);
188 bool CombineMovBx(MachineBasicBlock &MBB);
189};
190
191struct ARMLoadStoreOptLegacy : public MachineFunctionPass {
192 static char ID;
193
194 ARMLoadStoreOptLegacy() : MachineFunctionPass(ID) {}
195
196 bool runOnMachineFunction(MachineFunction &Fn) override;
197
198 MachineFunctionProperties getRequiredProperties() const override {
199 return MachineFunctionProperties().setNoVRegs();
200 }
201
202 StringRef getPassName() const override { return ARM_LOAD_STORE_OPT_NAME; }
203
204 void getAnalysisUsage(AnalysisUsage &AU) const override {
208 }
209};
210
211char ARMLoadStoreOptLegacy::ID = 0;
212
213} // end anonymous namespace
214
215INITIALIZE_PASS_BEGIN(ARMLoadStoreOptLegacy, "arm-ldst-opt",
216 ARM_LOAD_STORE_OPT_NAME, false, false)
218INITIALIZE_PASS_END(ARMLoadStoreOptLegacy, "arm-ldst-opt",
220
222 for (const auto &MO : MI.operands()) {
223 if (!MO.isReg())
224 continue;
225 if (MO.isDef() && MO.getReg() == ARM::CPSR && !MO.isDead())
226 // If the instruction has live CPSR def, then it's not safe to fold it
227 // into load / store.
228 return true;
229 }
230
231 return false;
232}
233
235 unsigned Opcode = MI.getOpcode();
236 bool isAM3 = Opcode == ARM::LDRD || Opcode == ARM::STRD;
237 unsigned NumOperands = MI.getDesc().getNumOperands();
238 unsigned OffField = MI.getOperand(NumOperands - 3).getImm();
239
240 if (Opcode == ARM::t2LDRi12 || Opcode == ARM::t2LDRi8 ||
241 Opcode == ARM::t2STRi12 || Opcode == ARM::t2STRi8 ||
242 Opcode == ARM::t2LDRDi8 || Opcode == ARM::t2STRDi8 ||
243 Opcode == ARM::LDRi12 || Opcode == ARM::STRi12)
244 return OffField;
245
246 // Thumb1 immediate offsets are scaled by 4
247 if (Opcode == ARM::tLDRi || Opcode == ARM::tSTRi ||
248 Opcode == ARM::tLDRspi || Opcode == ARM::tSTRspi)
249 return OffField * 4;
250
251 int Offset = isAM3 ? ARM_AM::getAM3Offset(OffField)
252 : ARM_AM::getAM5Offset(OffField) * 4;
253 ARM_AM::AddrOpc Op = isAM3 ? ARM_AM::getAM3Op(OffField)
254 : ARM_AM::getAM5Op(OffField);
255
256 if (Op == ARM_AM::sub)
257 return -Offset;
258
259 return Offset;
260}
261
263 return MI.getOperand(1);
264}
265
267 return MI.getOperand(0);
268}
269
271 switch (Opcode) {
272 default: llvm_unreachable("Unhandled opcode!");
273 case ARM::LDRi12:
274 ++NumLDMGened;
275 switch (Mode) {
276 default: llvm_unreachable("Unhandled submode!");
277 case ARM_AM::ia: return ARM::LDMIA;
278 case ARM_AM::da: return ARM::LDMDA;
279 case ARM_AM::db: return ARM::LDMDB;
280 case ARM_AM::ib: return ARM::LDMIB;
281 }
282 case ARM::STRi12:
283 ++NumSTMGened;
284 switch (Mode) {
285 default: llvm_unreachable("Unhandled submode!");
286 case ARM_AM::ia: return ARM::STMIA;
287 case ARM_AM::da: return ARM::STMDA;
288 case ARM_AM::db: return ARM::STMDB;
289 case ARM_AM::ib: return ARM::STMIB;
290 }
291 case ARM::tLDRi:
292 case ARM::tLDRspi:
293 // tLDMIA is writeback-only - unless the base register is in the input
294 // reglist.
295 ++NumLDMGened;
296 switch (Mode) {
297 default: llvm_unreachable("Unhandled submode!");
298 case ARM_AM::ia: return ARM::tLDMIA;
299 }
300 case ARM::tSTRi:
301 case ARM::tSTRspi:
302 // There is no non-writeback tSTMIA either.
303 ++NumSTMGened;
304 switch (Mode) {
305 default: llvm_unreachable("Unhandled submode!");
306 case ARM_AM::ia: return ARM::tSTMIA_UPD;
307 }
308 case ARM::t2LDRi8:
309 case ARM::t2LDRi12:
310 ++NumLDMGened;
311 switch (Mode) {
312 default: llvm_unreachable("Unhandled submode!");
313 case ARM_AM::ia: return ARM::t2LDMIA;
314 case ARM_AM::db: return ARM::t2LDMDB;
315 }
316 case ARM::t2STRi8:
317 case ARM::t2STRi12:
318 ++NumSTMGened;
319 switch (Mode) {
320 default: llvm_unreachable("Unhandled submode!");
321 case ARM_AM::ia: return ARM::t2STMIA;
322 case ARM_AM::db: return ARM::t2STMDB;
323 }
324 case ARM::VLDRS:
325 ++NumVLDMGened;
326 switch (Mode) {
327 default: llvm_unreachable("Unhandled submode!");
328 case ARM_AM::ia: return ARM::VLDMSIA;
329 case ARM_AM::db: return 0; // Only VLDMSDB_UPD exists.
330 }
331 case ARM::VSTRS:
332 ++NumVSTMGened;
333 switch (Mode) {
334 default: llvm_unreachable("Unhandled submode!");
335 case ARM_AM::ia: return ARM::VSTMSIA;
336 case ARM_AM::db: return 0; // Only VSTMSDB_UPD exists.
337 }
338 case ARM::VLDRD:
339 ++NumVLDMGened;
340 switch (Mode) {
341 default: llvm_unreachable("Unhandled submode!");
342 case ARM_AM::ia: return ARM::VLDMDIA;
343 case ARM_AM::db: return 0; // Only VLDMDDB_UPD exists.
344 }
345 case ARM::VSTRD:
346 ++NumVSTMGened;
347 switch (Mode) {
348 default: llvm_unreachable("Unhandled submode!");
349 case ARM_AM::ia: return ARM::VSTMDIA;
350 case ARM_AM::db: return 0; // Only VSTMDDB_UPD exists.
351 }
352 }
353}
354
356 switch (Opcode) {
357 default: llvm_unreachable("Unhandled opcode!");
358 case ARM::LDMIA_RET:
359 case ARM::LDMIA:
360 case ARM::LDMIA_UPD:
361 case ARM::STMIA:
362 case ARM::STMIA_UPD:
363 case ARM::tLDMIA:
364 case ARM::tLDMIA_UPD:
365 case ARM::tSTMIA_UPD:
366 case ARM::t2LDMIA_RET:
367 case ARM::t2LDMIA:
368 case ARM::t2LDMIA_UPD:
369 case ARM::t2STMIA:
370 case ARM::t2STMIA_UPD:
371 case ARM::VLDMSIA:
372 case ARM::VLDMSIA_UPD:
373 case ARM::VSTMSIA:
374 case ARM::VSTMSIA_UPD:
375 case ARM::VLDMDIA:
376 case ARM::VLDMDIA_UPD:
377 case ARM::VSTMDIA:
378 case ARM::VSTMDIA_UPD:
379 return ARM_AM::ia;
380
381 case ARM::LDMDA:
382 case ARM::LDMDA_UPD:
383 case ARM::STMDA:
384 case ARM::STMDA_UPD:
385 return ARM_AM::da;
386
387 case ARM::LDMDB:
388 case ARM::LDMDB_UPD:
389 case ARM::STMDB:
390 case ARM::STMDB_UPD:
391 case ARM::t2LDMDB:
392 case ARM::t2LDMDB_UPD:
393 case ARM::t2STMDB:
394 case ARM::t2STMDB_UPD:
395 case ARM::VLDMSDB_UPD:
396 case ARM::VSTMSDB_UPD:
397 case ARM::VLDMDDB_UPD:
398 case ARM::VSTMDDB_UPD:
399 return ARM_AM::db;
400
401 case ARM::LDMIB:
402 case ARM::LDMIB_UPD:
403 case ARM::STMIB:
404 case ARM::STMIB_UPD:
405 return ARM_AM::ib;
406 }
407}
408
409static bool isT1i32Load(unsigned Opc) {
410 return Opc == ARM::tLDRi || Opc == ARM::tLDRspi;
411}
412
413static bool isT2i32Load(unsigned Opc) {
414 return Opc == ARM::t2LDRi12 || Opc == ARM::t2LDRi8;
415}
416
417static bool isi32Load(unsigned Opc) {
418 return Opc == ARM::LDRi12 || isT1i32Load(Opc) || isT2i32Load(Opc) ;
419}
420
421static bool isT1i32Store(unsigned Opc) {
422 return Opc == ARM::tSTRi || Opc == ARM::tSTRspi;
423}
424
425static bool isT2i32Store(unsigned Opc) {
426 return Opc == ARM::t2STRi12 || Opc == ARM::t2STRi8;
427}
428
429static bool isi32Store(unsigned Opc) {
430 return Opc == ARM::STRi12 || isT1i32Store(Opc) || isT2i32Store(Opc);
431}
432
433static bool isLoadSingle(unsigned Opc) {
434 return isi32Load(Opc) || Opc == ARM::VLDRS || Opc == ARM::VLDRD;
435}
436
437static unsigned getImmScale(unsigned Opc) {
438 switch (Opc) {
439 default: llvm_unreachable("Unhandled opcode!");
440 case ARM::tLDRi:
441 case ARM::tSTRi:
442 case ARM::tLDRspi:
443 case ARM::tSTRspi:
444 return 1;
445 case ARM::tLDRHi:
446 case ARM::tSTRHi:
447 return 2;
448 case ARM::tLDRBi:
449 case ARM::tSTRBi:
450 return 4;
451 }
452}
453
455 switch (MI->getOpcode()) {
456 default: return 0;
457 case ARM::LDRi12:
458 case ARM::STRi12:
459 case ARM::tLDRi:
460 case ARM::tSTRi:
461 case ARM::tLDRspi:
462 case ARM::tSTRspi:
463 case ARM::t2LDRi8:
464 case ARM::t2LDRi12:
465 case ARM::t2STRi8:
466 case ARM::t2STRi12:
467 case ARM::VLDRS:
468 case ARM::VSTRS:
469 return 4;
470 case ARM::VLDRD:
471 case ARM::VSTRD:
472 return 8;
473 case ARM::LDMIA:
474 case ARM::LDMDA:
475 case ARM::LDMDB:
476 case ARM::LDMIB:
477 case ARM::STMIA:
478 case ARM::STMDA:
479 case ARM::STMDB:
480 case ARM::STMIB:
481 case ARM::tLDMIA:
482 case ARM::tLDMIA_UPD:
483 case ARM::tSTMIA_UPD:
484 case ARM::t2LDMIA:
485 case ARM::t2LDMDB:
486 case ARM::t2STMIA:
487 case ARM::t2STMDB:
488 case ARM::VLDMSIA:
489 case ARM::VSTMSIA:
490 return (MI->getNumOperands() - MI->getDesc().getNumOperands() + 1) * 4;
491 case ARM::VLDMDIA:
492 case ARM::VSTMDIA:
493 return (MI->getNumOperands() - MI->getDesc().getNumOperands() + 1) * 8;
494 }
495}
496
497/// Update future uses of the base register with the offset introduced
498/// due to writeback. This function only works on Thumb1.
499void ARMLoadStoreOpt::UpdateBaseRegUses(MachineBasicBlock &MBB,
501 const DebugLoc &DL, unsigned Base,
502 unsigned WordOffset,
503 ARMCC::CondCodes Pred,
504 unsigned PredReg) {
505 assert(isThumb1 && "Can only update base register uses for Thumb1!");
506 // Start updating any instructions with immediate offsets. Insert a SUB before
507 // the first non-updateable instruction (if any).
508 for (; MBBI != MBB.end(); ++MBBI) {
509 bool InsertSub = false;
510 unsigned Opc = MBBI->getOpcode();
511
512 if (MBBI->readsRegister(Base, /*TRI=*/nullptr)) {
513 int Offset;
514 bool IsLoad =
515 Opc == ARM::tLDRi || Opc == ARM::tLDRHi || Opc == ARM::tLDRBi;
516 bool IsStore =
517 Opc == ARM::tSTRi || Opc == ARM::tSTRHi || Opc == ARM::tSTRBi;
518
519 if (IsLoad || IsStore) {
520 // Loads and stores with immediate offsets can be updated, but only if
521 // the new offset isn't negative.
522 // The MachineOperand containing the offset immediate is the last one
523 // before predicates.
524 MachineOperand &MO =
525 MBBI->getOperand(MBBI->getDesc().getNumOperands() - 3);
526 // The offsets are scaled by 1, 2 or 4 depending on the Opcode.
527 Offset = MO.getImm() - WordOffset * getImmScale(Opc);
528
529 // If storing the base register, it needs to be reset first.
530 Register InstrSrcReg = getLoadStoreRegOp(*MBBI).getReg();
531
532 if (Offset >= 0 && !(IsStore && InstrSrcReg == Base))
533 MO.setImm(Offset);
534 else
535 InsertSub = true;
536 } else if ((Opc == ARM::tSUBi8 || Opc == ARM::tADDi8) &&
537 !definesCPSR(*MBBI)) {
538 // SUBS/ADDS using this register, with a dead def of the CPSR.
539 // Merge it with the update; if the merged offset is too large,
540 // insert a new sub instead.
541 MachineOperand &MO =
542 MBBI->getOperand(MBBI->getDesc().getNumOperands() - 3);
543 Offset = (Opc == ARM::tSUBi8) ?
544 MO.getImm() + WordOffset * 4 :
545 MO.getImm() - WordOffset * 4 ;
546 if (Offset >= 0 && TL->isLegalAddImmediate(Offset)) {
547 // FIXME: Swap ADDS<->SUBS if Offset < 0, erase instruction if
548 // Offset == 0.
549 MO.setImm(Offset);
550 // The base register has now been reset, so exit early.
551 return;
552 } else {
553 InsertSub = true;
554 }
555 } else {
556 // Can't update the instruction.
557 InsertSub = true;
558 }
559 } else if (definesCPSR(*MBBI) || MBBI->isCall() || MBBI->isBranch()) {
560 // Since SUBS sets the condition flags, we can't place the base reset
561 // after an instruction that has a live CPSR def.
562 // The base register might also contain an argument for a function call.
563 InsertSub = true;
564 }
565
566 if (InsertSub) {
567 // An instruction above couldn't be updated, so insert a sub.
568 BuildMI(MBB, MBBI, DL, TII->get(ARM::tSUBi8), Base)
569 .add(t1CondCodeOp(true))
570 .addReg(Base)
571 .addImm(WordOffset * 4)
572 .addImm(Pred)
573 .addReg(PredReg);
574 return;
575 }
576
577 if (MBBI->killsRegister(Base, /*TRI=*/nullptr) ||
578 MBBI->definesRegister(Base, /*TRI=*/nullptr))
579 // Register got killed. Stop updating.
580 return;
581 }
582
583 // End of block was reached.
584 if (!MBB.succ_empty()) {
585 // FIXME: Because of a bug, live registers are sometimes missing from
586 // the successor blocks' live-in sets. This means we can't trust that
587 // information and *always* have to reset at the end of a block.
588 // See PR21029.
589 if (MBBI != MBB.end()) --MBBI;
590 BuildMI(MBB, MBBI, DL, TII->get(ARM::tSUBi8), Base)
591 .add(t1CondCodeOp(true))
592 .addReg(Base)
593 .addImm(WordOffset * 4)
594 .addImm(Pred)
595 .addReg(PredReg);
596 }
597}
598
600ARMLoadStoreOpt::eraseInstr(MachineBasicBlock::iterator MI) {
601 if (LiveRegsValid && LiveRegPos == MI)
602 LiveRegsValid = false;
603 return MI->eraseFromParent();
604}
605
606/// Return the first register of class \p RegClass that is not in \p Regs.
607unsigned ARMLoadStoreOpt::findFreeReg(const TargetRegisterClass &RegClass) {
608 for (unsigned Reg : RCI->getOrder(&RegClass))
609 if (LiveRegs.available(Reg))
610 return Reg;
611 return 0;
612}
613
614/// Compute live registers just before instruction \p Before (in normal schedule
615/// direction). Computes backwards so multiple queries in the same block must
616/// come in reverse order.
617void ARMLoadStoreOpt::moveLiveRegsBefore(const MachineBasicBlock &MBB,
619 // Initialize if we never queried in this block.
620 if (!LiveRegsValid) {
621 LiveRegs.init(*TRI);
622 LiveRegs.addLiveOuts(MBB);
623 LiveRegPos = MBB.end();
624 LiveRegsValid = true;
625 }
626 // Move backward just before the "Before" position.
627 while (LiveRegPos != Before) {
628 --LiveRegPos;
629 if (!LiveRegPos->isDebugInstr())
630 LiveRegs.stepBackward(*LiveRegPos);
631 }
632}
633
634static bool ContainsReg(ArrayRef<std::pair<unsigned, bool>> Regs,
635 unsigned Reg) {
636 for (const std::pair<unsigned, bool> &R : Regs)
637 if (R.first == Reg)
638 return true;
639 return false;
640}
641
642/// Create and insert a LDM or STM with Base as base register and registers in
643/// Regs as the register operands that would be loaded / stored. It returns
644/// true if the transformation is done.
645MachineInstr *ARMLoadStoreOpt::CreateLoadStoreMulti(
646 MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertBefore,
647 int Offset, unsigned Base, bool BaseKill, unsigned Opcode,
648 ARMCC::CondCodes Pred, unsigned PredReg, const DebugLoc &DL,
649 ArrayRef<std::pair<unsigned, bool>> Regs,
651 unsigned NumRegs = Regs.size();
652 assert(NumRegs > 1);
653
654 // For Thumb1 targets, it might be necessary to clobber the CPSR to merge.
655 // Compute liveness information for that register to make the decision.
656 bool SafeToClobberCPSR = !isThumb1 ||
657 (MBB.computeRegisterLiveness(TRI, ARM::CPSR, InsertBefore, 20) ==
659
660 bool Writeback = isThumb1; // Thumb1 LDM/STM have base reg writeback.
661
662 // Exception: If the base register is in the input reglist, Thumb1 LDM is
663 // non-writeback.
664 // It's also not possible to merge an STR of the base register in Thumb1.
665 if (isThumb1 && ContainsReg(Regs, Base)) {
666 assert(Base != ARM::SP && "Thumb1 does not allow SP in register list");
667 if (Opcode == ARM::tLDRi)
668 Writeback = false;
669 else if (Opcode == ARM::tSTRi)
670 return nullptr;
671 }
672
674 // VFP and Thumb2 do not support IB or DA modes. Thumb1 only supports IA.
675 bool isNotVFP = isi32Load(Opcode) || isi32Store(Opcode);
676 bool haveIBAndDA = isNotVFP && !isThumb2 && !isThumb1;
677
678 if (Offset == 4 && haveIBAndDA) {
680 } else if (Offset == -4 * (int)NumRegs + 4 && haveIBAndDA) {
682 } else if (Offset == -4 * (int)NumRegs && isNotVFP && !isThumb1) {
683 // VLDM/VSTM do not support DB mode without also updating the base reg.
685 } else if (Offset != 0 || Opcode == ARM::tLDRspi || Opcode == ARM::tSTRspi) {
686 // Check if this is a supported opcode before inserting instructions to
687 // calculate a new base register.
688 if (!getLoadStoreMultipleOpcode(Opcode, Mode)) return nullptr;
689
690 // If starting offset isn't zero, insert a MI to materialize a new base.
691 // But only do so if it is cost effective, i.e. merging more than two
692 // loads / stores.
693 if (NumRegs <= 2)
694 return nullptr;
695
696 // On Thumb1, it's not worth materializing a new base register without
697 // clobbering the CPSR (i.e. not using ADDS/SUBS).
698 if (!SafeToClobberCPSR)
699 return nullptr;
700
701 unsigned NewBase;
702 if (isi32Load(Opcode)) {
703 // If it is a load, then just use one of the destination registers
704 // as the new base. Will no longer be writeback in Thumb1.
705 NewBase = Regs[NumRegs-1].first;
706 Writeback = false;
707 } else {
708 // Find a free register that we can use as scratch register.
709 moveLiveRegsBefore(MBB, InsertBefore);
710 // The merged instruction does not exist yet but will use several Regs if
711 // it is a Store.
712 if (!isLoadSingle(Opcode))
713 for (const std::pair<unsigned, bool> &R : Regs)
714 LiveRegs.addReg(R.first);
715
716 NewBase = findFreeReg(isThumb1 ? ARM::tGPRRegClass : ARM::GPRRegClass);
717 if (NewBase == 0)
718 return nullptr;
719 }
720
721 int BaseOpc = isThumb2 ? (BaseKill && Base == ARM::SP ? ARM::t2ADDspImm
722 : ARM::t2ADDri)
723 : (isThumb1 && Base == ARM::SP)
724 ? ARM::tADDrSPi
725 : (isThumb1 && Offset < 8)
726 ? ARM::tADDi3
727 : isThumb1 ? ARM::tADDi8 : ARM::ADDri;
728
729 if (Offset < 0) {
730 // FIXME: There are no Thumb1 load/store instructions with negative
731 // offsets. So the Base != ARM::SP might be unnecessary.
732 Offset = -Offset;
733 BaseOpc = isThumb2 ? (BaseKill && Base == ARM::SP ? ARM::t2SUBspImm
734 : ARM::t2SUBri)
735 : (isThumb1 && Offset < 8 && Base != ARM::SP)
736 ? ARM::tSUBi3
737 : isThumb1 ? ARM::tSUBi8 : ARM::SUBri;
738 }
739
740 if (!TL->isLegalAddImmediate(Offset))
741 // FIXME: Try add with register operand?
742 return nullptr; // Probably not worth it then.
743
744 // We can only append a kill flag to the add/sub input if the value is not
745 // used in the register list of the stm as well.
746 bool KillOldBase = BaseKill &&
747 (!isi32Store(Opcode) || !ContainsReg(Regs, Base));
748
749 if (isThumb1) {
750 // Thumb1: depending on immediate size, use either
751 // ADDS NewBase, Base, #imm3
752 // or
753 // MOV NewBase, Base
754 // ADDS NewBase, #imm8.
755 if (Base != NewBase &&
756 (BaseOpc == ARM::tADDi8 || BaseOpc == ARM::tSUBi8)) {
757 // Need to insert a MOV to the new base first.
758 if (isARMLowRegister(NewBase) && isARMLowRegister(Base) &&
759 !STI->hasV6Ops()) {
760 // thumbv4t doesn't have lo->lo copies, and we can't predicate tMOVSr
761 if (Pred != ARMCC::AL)
762 return nullptr;
763 BuildMI(MBB, InsertBefore, DL, TII->get(ARM::tMOVSr), NewBase)
764 .addReg(Base, getKillRegState(KillOldBase));
765 } else
766 BuildMI(MBB, InsertBefore, DL, TII->get(ARM::tMOVr), NewBase)
767 .addReg(Base, getKillRegState(KillOldBase))
768 .add(predOps(Pred, PredReg));
769
770 // The following ADDS/SUBS becomes an update.
771 Base = NewBase;
772 KillOldBase = true;
773 }
774 if (BaseOpc == ARM::tADDrSPi) {
775 assert(Offset % 4 == 0 && "tADDrSPi offset is scaled by 4");
776 BuildMI(MBB, InsertBefore, DL, TII->get(BaseOpc), NewBase)
777 .addReg(Base, getKillRegState(KillOldBase))
778 .addImm(Offset / 4)
779 .add(predOps(Pred, PredReg));
780 } else
781 BuildMI(MBB, InsertBefore, DL, TII->get(BaseOpc), NewBase)
782 .add(t1CondCodeOp(true))
783 .addReg(Base, getKillRegState(KillOldBase))
784 .addImm(Offset)
785 .add(predOps(Pred, PredReg));
786 } else {
787 BuildMI(MBB, InsertBefore, DL, TII->get(BaseOpc), NewBase)
788 .addReg(Base, getKillRegState(KillOldBase))
789 .addImm(Offset)
790 .add(predOps(Pred, PredReg))
791 .add(condCodeOp());
792 }
793 Base = NewBase;
794 BaseKill = true; // New base is always killed straight away.
795 }
796
797 bool isDef = isLoadSingle(Opcode);
798
799 // Get LS multiple opcode. Note that for Thumb1 this might be an opcode with
800 // base register writeback.
801 Opcode = getLoadStoreMultipleOpcode(Opcode, Mode);
802 if (!Opcode)
803 return nullptr;
804
805 // Check if a Thumb1 LDM/STM merge is safe. This is the case if:
806 // - There is no writeback (LDM of base register),
807 // - the base register is killed by the merged instruction,
808 // - or it's safe to overwrite the condition flags, i.e. to insert a SUBS
809 // to reset the base register.
810 // Otherwise, don't merge.
811 // It's safe to return here since the code to materialize a new base register
812 // above is also conditional on SafeToClobberCPSR.
813 if (isThumb1 && !SafeToClobberCPSR && Writeback && !BaseKill)
814 return nullptr;
815
816 MachineInstrBuilder MIB;
817
818 if (Writeback) {
819 assert(isThumb1 && "expected Writeback only inThumb1");
820 if (Opcode == ARM::tLDMIA) {
821 assert(!(ContainsReg(Regs, Base)) && "Thumb1 can't LDM ! with Base in Regs");
822 // Update tLDMIA with writeback if necessary.
823 Opcode = ARM::tLDMIA_UPD;
824 }
825
826 MIB = BuildMI(MBB, InsertBefore, DL, TII->get(Opcode));
827
828 // Thumb1: we might need to set base writeback when building the MI.
829 MIB.addReg(Base, getDefRegState(true))
830 .addReg(Base, getKillRegState(BaseKill));
831
832 // The base isn't dead after a merged instruction with writeback.
833 // Insert a sub instruction after the newly formed instruction to reset.
834 if (!BaseKill)
835 UpdateBaseRegUses(MBB, InsertBefore, DL, Base, NumRegs, Pred, PredReg);
836 } else {
837 // No writeback, simply build the MachineInstr.
838 MIB = BuildMI(MBB, InsertBefore, DL, TII->get(Opcode));
839 MIB.addReg(Base, getKillRegState(BaseKill));
840 }
841
842 MIB.addImm(Pred).addReg(PredReg);
843
844 for (const std::pair<unsigned, bool> &R : Regs)
845 MIB.addReg(R.first, getDefRegState(isDef) | getKillRegState(R.second));
846
847 MIB.cloneMergedMemRefs(Instrs);
848
849 return MIB.getInstr();
850}
851
852MachineInstr *ARMLoadStoreOpt::CreateLoadStoreDouble(
853 MachineBasicBlock &MBB, MachineBasicBlock::iterator InsertBefore,
854 int Offset, unsigned Base, bool BaseKill, unsigned Opcode,
855 ARMCC::CondCodes Pred, unsigned PredReg, const DebugLoc &DL,
856 ArrayRef<std::pair<unsigned, bool>> Regs,
857 ArrayRef<MachineInstr*> Instrs) const {
858 bool IsLoad = isi32Load(Opcode);
859 assert((IsLoad || isi32Store(Opcode)) && "Must have integer load or store");
860 unsigned LoadStoreOpcode = IsLoad ? ARM::t2LDRDi8 : ARM::t2STRDi8;
861
862 assert(Regs.size() == 2);
863 MachineInstrBuilder MIB = BuildMI(MBB, InsertBefore, DL,
864 TII->get(LoadStoreOpcode));
865 if (IsLoad) {
866 MIB.addReg(Regs[0].first, RegState::Define)
867 .addReg(Regs[1].first, RegState::Define);
868 } else {
869 MIB.addReg(Regs[0].first, getKillRegState(Regs[0].second))
870 .addReg(Regs[1].first, getKillRegState(Regs[1].second));
871 }
872 MIB.addReg(Base).addImm(Offset).addImm(Pred).addReg(PredReg);
873 MIB.cloneMergedMemRefs(Instrs);
874 return MIB.getInstr();
875}
876
877/// Call MergeOps and update MemOps and merges accordingly on success.
878MachineInstr *ARMLoadStoreOpt::MergeOpsUpdate(const MergeCandidate &Cand) {
879 const MachineInstr *First = Cand.Instrs.front();
880 unsigned Opcode = First->getOpcode();
881 bool IsLoad = isLoadSingle(Opcode);
883 SmallVector<unsigned, 4> ImpDefs;
884 DenseSet<unsigned> KilledRegs;
885 DenseSet<unsigned> UsedRegs;
886 // Determine list of registers and list of implicit super-register defs.
887 for (const MachineInstr *MI : Cand.Instrs) {
888 const MachineOperand &MO = getLoadStoreRegOp(*MI);
889 Register Reg = MO.getReg();
890 bool IsKill = MO.isKill();
891 if (IsKill)
892 KilledRegs.insert(Reg);
893 Regs.push_back(std::make_pair(Reg, IsKill));
894 UsedRegs.insert(Reg);
895
896 if (IsLoad) {
897 // Collect any implicit defs of super-registers, after merging we can't
898 // be sure anymore that we properly preserved these live ranges and must
899 // removed these implicit operands.
900 for (const MachineOperand &MO : MI->implicit_operands()) {
901 if (!MO.isReg() || !MO.isDef() || MO.isDead())
902 continue;
903 assert(MO.isImplicit());
904 Register DefReg = MO.getReg();
905
906 if (is_contained(ImpDefs, DefReg))
907 continue;
908 // We can ignore cases where the super-reg is read and written.
909 if (MI->readsRegister(DefReg, /*TRI=*/nullptr))
910 continue;
911 ImpDefs.push_back(DefReg);
912 }
913 }
914 }
915
916 // Attempt the merge.
918
919 MachineInstr *LatestMI = Cand.Instrs[Cand.LatestMIIdx];
920 iterator InsertBefore = std::next(iterator(LatestMI));
921 MachineBasicBlock &MBB = *LatestMI->getParent();
922 unsigned Offset = getMemoryOpOffset(*First);
924 bool BaseKill = LatestMI->killsRegister(Base, /*TRI=*/nullptr);
925 Register PredReg;
926 ARMCC::CondCodes Pred = getInstrPredicate(*First, PredReg);
927 DebugLoc DL = First->getDebugLoc();
928 MachineInstr *Merged = nullptr;
929 if (Cand.CanMergeToLSDouble)
930 Merged = CreateLoadStoreDouble(MBB, InsertBefore, Offset, Base, BaseKill,
931 Opcode, Pred, PredReg, DL, Regs,
932 Cand.Instrs);
933 if (!Merged && Cand.CanMergeToLSMulti)
934 Merged = CreateLoadStoreMulti(MBB, InsertBefore, Offset, Base, BaseKill,
935 Opcode, Pred, PredReg, DL, Regs, Cand.Instrs);
936 if (!Merged)
937 return nullptr;
938
939 // Determine earliest instruction that will get removed. We then keep an
940 // iterator just above it so the following erases don't invalidated it.
941 iterator EarliestI(Cand.Instrs[Cand.EarliestMIIdx]);
942 bool EarliestAtBegin = false;
943 if (EarliestI == MBB.begin()) {
944 EarliestAtBegin = true;
945 } else {
946 EarliestI = std::prev(EarliestI);
947 }
948
949 // Remove instructions which have been merged.
950 for (MachineInstr *MI : Cand.Instrs)
951 eraseInstr(MI);
952
953 // Determine range between the earliest removed instruction and the new one.
954 if (EarliestAtBegin)
955 EarliestI = MBB.begin();
956 else
957 EarliestI = std::next(EarliestI);
958 auto FixupRange = make_range(EarliestI, iterator(Merged));
959
960 if (isLoadSingle(Opcode)) {
961 // If the previous loads defined a super-reg, then we have to mark earlier
962 // operands undef; Replicate the super-reg def on the merged instruction.
963 for (MachineInstr &MI : FixupRange) {
964 for (unsigned &ImpDefReg : ImpDefs) {
965 for (MachineOperand &MO : MI.implicit_operands()) {
966 if (!MO.isReg() || MO.getReg() != ImpDefReg)
967 continue;
968 if (MO.readsReg())
969 MO.setIsUndef();
970 else if (MO.isDef())
971 ImpDefReg = 0;
972 }
973 }
974 }
975
976 MachineInstrBuilder MIB(*Merged->getParent()->getParent(), Merged);
977 for (unsigned ImpDef : ImpDefs)
978 MIB.addReg(ImpDef, RegState::ImplicitDefine);
979 } else {
980 // Remove kill flags: We are possibly storing the values later now.
981 assert(isi32Store(Opcode) || Opcode == ARM::VSTRS || Opcode == ARM::VSTRD);
982 for (MachineInstr &MI : FixupRange) {
983 for (MachineOperand &MO : MI.uses()) {
984 if (!MO.isReg() || !MO.isKill())
985 continue;
986 if (UsedRegs.count(MO.getReg()))
987 MO.setIsKill(false);
988 }
989 }
990 assert(ImpDefs.empty());
991 }
992
993 return Merged;
994}
995
997 unsigned Value = abs(Offset);
998 // t2LDRDi8/t2STRDi8 supports an 8 bit immediate which is internally
999 // multiplied by 4.
1000 return (Value % 4) == 0 && Value < 1024;
1001}
1002
1003/// Return true for loads/stores that can be combined to a double/multi
1004/// operation without increasing the requirements for alignment.
1006 const MachineInstr &MI) {
1007 // vldr/vstr trap on misaligned pointers anyway, forming vldm makes no
1008 // difference.
1009 unsigned Opcode = MI.getOpcode();
1010 if (!isi32Load(Opcode) && !isi32Store(Opcode))
1011 return true;
1012
1013 // Stack pointer alignment is out of the programmers control so we can trust
1014 // SP-relative loads/stores.
1015 if (getLoadStoreBaseOp(MI).getReg() == ARM::SP &&
1017 return true;
1018 return false;
1019}
1020
1021/// Find candidates for load/store multiple merge in list of MemOpQueueEntries.
1022void ARMLoadStoreOpt::FormCandidates(const MemOpQueue &MemOps) {
1023 const MachineInstr *FirstMI = MemOps[0].MI;
1024 unsigned Opcode = FirstMI->getOpcode();
1025 bool isNotVFP = isi32Load(Opcode) || isi32Store(Opcode);
1026 unsigned Size = getLSMultipleTransferSize(FirstMI);
1027
1028 unsigned SIndex = 0;
1029 unsigned EIndex = MemOps.size();
1030 do {
1031 // Look at the first instruction.
1032 const MachineInstr *MI = MemOps[SIndex].MI;
1033 int Offset = MemOps[SIndex].Offset;
1034 const MachineOperand &PMO = getLoadStoreRegOp(*MI);
1035 Register PReg = PMO.getReg();
1036 unsigned PRegNum = PMO.isUndef() ? std::numeric_limits<unsigned>::max()
1037 : TRI->getEncodingValue(PReg);
1038 unsigned Latest = SIndex;
1039 unsigned Earliest = SIndex;
1040 unsigned Count = 1;
1041 bool CanMergeToLSDouble =
1042 STI->isThumb2() && isNotVFP && isValidLSDoubleOffset(Offset);
1043 // ARM errata 602117: LDRD with base in list may result in incorrect base
1044 // register when interrupted or faulted.
1045 if (STI->isCortexM3() && isi32Load(Opcode) &&
1046 PReg == getLoadStoreBaseOp(*MI).getReg())
1047 CanMergeToLSDouble = false;
1048
1049 bool CanMergeToLSMulti = true;
1050 // On swift vldm/vstm starting with an odd register number as that needs
1051 // more uops than single vldrs.
1052 if (STI->hasSlowOddRegister() && !isNotVFP && (PRegNum % 2) == 1)
1053 CanMergeToLSMulti = false;
1054
1055 // LDRD/STRD do not allow SP/PC. LDM/STM do not support it or have it
1056 // deprecated; LDM to PC is fine but cannot happen here.
1057 if (PReg == ARM::SP || PReg == ARM::PC)
1058 CanMergeToLSMulti = CanMergeToLSDouble = false;
1059
1060 // Should we be conservative?
1062 CanMergeToLSMulti = CanMergeToLSDouble = false;
1063
1064 // vldm / vstm limit are 32 for S variants, 16 for D variants.
1065 unsigned Limit;
1066 switch (Opcode) {
1067 default:
1068 Limit = UINT_MAX;
1069 break;
1070 case ARM::VLDRD:
1071 case ARM::VSTRD:
1072 Limit = 16;
1073 break;
1074 }
1075
1076 // Merge following instructions where possible.
1077 for (unsigned I = SIndex+1; I < EIndex; ++I, ++Count) {
1078 int NewOffset = MemOps[I].Offset;
1079 if (NewOffset != Offset + (int)Size)
1080 break;
1081 const MachineOperand &MO = getLoadStoreRegOp(*MemOps[I].MI);
1082 Register Reg = MO.getReg();
1083 if (Reg == ARM::SP || Reg == ARM::PC)
1084 break;
1085 if (Count == Limit)
1086 break;
1087
1088 // See if the current load/store may be part of a multi load/store.
1089 unsigned RegNum = MO.isUndef() ? std::numeric_limits<unsigned>::max()
1090 : TRI->getEncodingValue(Reg);
1091 bool PartOfLSMulti = CanMergeToLSMulti;
1092 if (PartOfLSMulti) {
1093 // Register numbers must be in ascending order.
1094 if (RegNum <= PRegNum)
1095 PartOfLSMulti = false;
1096 // For VFP / NEON load/store multiples, the registers must be
1097 // consecutive and within the limit on the number of registers per
1098 // instruction.
1099 else if (!isNotVFP && RegNum != PRegNum+1)
1100 PartOfLSMulti = false;
1101 }
1102 // See if the current load/store may be part of a double load/store.
1103 bool PartOfLSDouble = CanMergeToLSDouble && Count <= 1;
1104
1105 if (!PartOfLSMulti && !PartOfLSDouble)
1106 break;
1107 CanMergeToLSMulti &= PartOfLSMulti;
1108 CanMergeToLSDouble &= PartOfLSDouble;
1109 // Track MemOp with latest and earliest position (Positions are
1110 // counted in reverse).
1111 unsigned Position = MemOps[I].Position;
1112 if (Position < MemOps[Latest].Position)
1113 Latest = I;
1114 else if (Position > MemOps[Earliest].Position)
1115 Earliest = I;
1116 // Prepare for next MemOp.
1117 Offset += Size;
1118 PRegNum = RegNum;
1119 }
1120
1121 // Form a candidate from the Ops collected so far.
1122 MergeCandidate *Candidate = new(Allocator.Allocate()) MergeCandidate;
1123 for (unsigned C = SIndex, CE = SIndex + Count; C < CE; ++C)
1124 Candidate->Instrs.push_back(MemOps[C].MI);
1125 Candidate->LatestMIIdx = Latest - SIndex;
1126 Candidate->EarliestMIIdx = Earliest - SIndex;
1127 Candidate->InsertPos = MemOps[Latest].Position;
1128 if (Count == 1)
1129 CanMergeToLSMulti = CanMergeToLSDouble = false;
1130 Candidate->CanMergeToLSMulti = CanMergeToLSMulti;
1131 Candidate->CanMergeToLSDouble = CanMergeToLSDouble;
1132 Candidates.push_back(Candidate);
1133 // Continue after the chain.
1134 SIndex += Count;
1135 } while (SIndex < EIndex);
1136}
1137
1138static unsigned getUpdatingLSMultipleOpcode(unsigned Opc,
1140 switch (Opc) {
1141 default: llvm_unreachable("Unhandled opcode!");
1142 case ARM::LDMIA:
1143 case ARM::LDMDA:
1144 case ARM::LDMDB:
1145 case ARM::LDMIB:
1146 switch (Mode) {
1147 default: llvm_unreachable("Unhandled submode!");
1148 case ARM_AM::ia: return ARM::LDMIA_UPD;
1149 case ARM_AM::ib: return ARM::LDMIB_UPD;
1150 case ARM_AM::da: return ARM::LDMDA_UPD;
1151 case ARM_AM::db: return ARM::LDMDB_UPD;
1152 }
1153 case ARM::STMIA:
1154 case ARM::STMDA:
1155 case ARM::STMDB:
1156 case ARM::STMIB:
1157 switch (Mode) {
1158 default: llvm_unreachable("Unhandled submode!");
1159 case ARM_AM::ia: return ARM::STMIA_UPD;
1160 case ARM_AM::ib: return ARM::STMIB_UPD;
1161 case ARM_AM::da: return ARM::STMDA_UPD;
1162 case ARM_AM::db: return ARM::STMDB_UPD;
1163 }
1164 case ARM::t2LDMIA:
1165 case ARM::t2LDMDB:
1166 switch (Mode) {
1167 default: llvm_unreachable("Unhandled submode!");
1168 case ARM_AM::ia: return ARM::t2LDMIA_UPD;
1169 case ARM_AM::db: return ARM::t2LDMDB_UPD;
1170 }
1171 case ARM::t2STMIA:
1172 case ARM::t2STMDB:
1173 switch (Mode) {
1174 default: llvm_unreachable("Unhandled submode!");
1175 case ARM_AM::ia: return ARM::t2STMIA_UPD;
1176 case ARM_AM::db: return ARM::t2STMDB_UPD;
1177 }
1178 case ARM::VLDMSIA:
1179 switch (Mode) {
1180 default: llvm_unreachable("Unhandled submode!");
1181 case ARM_AM::ia: return ARM::VLDMSIA_UPD;
1182 case ARM_AM::db: return ARM::VLDMSDB_UPD;
1183 }
1184 case ARM::VLDMDIA:
1185 switch (Mode) {
1186 default: llvm_unreachable("Unhandled submode!");
1187 case ARM_AM::ia: return ARM::VLDMDIA_UPD;
1188 case ARM_AM::db: return ARM::VLDMDDB_UPD;
1189 }
1190 case ARM::VSTMSIA:
1191 switch (Mode) {
1192 default: llvm_unreachable("Unhandled submode!");
1193 case ARM_AM::ia: return ARM::VSTMSIA_UPD;
1194 case ARM_AM::db: return ARM::VSTMSDB_UPD;
1195 }
1196 case ARM::VSTMDIA:
1197 switch (Mode) {
1198 default: llvm_unreachable("Unhandled submode!");
1199 case ARM_AM::ia: return ARM::VSTMDIA_UPD;
1200 case ARM_AM::db: return ARM::VSTMDDB_UPD;
1201 }
1202 }
1203}
1204
1205/// Check if the given instruction increments or decrements a register and
1206/// return the amount it is incremented/decremented. Returns 0 if the CPSR flags
1207/// generated by the instruction are possibly read as well.
1209 ARMCC::CondCodes Pred, Register PredReg) {
1210 bool CheckCPSRDef;
1211 int Scale;
1212 switch (MI.getOpcode()) {
1213 case ARM::tADDi8: Scale = 4; CheckCPSRDef = true; break;
1214 case ARM::tSUBi8: Scale = -4; CheckCPSRDef = true; break;
1215 case ARM::t2SUBri:
1216 case ARM::t2SUBspImm:
1217 case ARM::SUBri: Scale = -1; CheckCPSRDef = true; break;
1218 case ARM::t2ADDri:
1219 case ARM::t2ADDspImm:
1220 case ARM::ADDri: Scale = 1; CheckCPSRDef = true; break;
1221 case ARM::tADDspi: Scale = 4; CheckCPSRDef = false; break;
1222 case ARM::tSUBspi: Scale = -4; CheckCPSRDef = false; break;
1223 default: return 0;
1224 }
1225
1226 Register MIPredReg;
1227 if (MI.getOperand(0).getReg() != Reg ||
1228 MI.getOperand(1).getReg() != Reg ||
1229 getInstrPredicate(MI, MIPredReg) != Pred ||
1230 MIPredReg != PredReg)
1231 return 0;
1232
1233 if (CheckCPSRDef && definesCPSR(MI))
1234 return 0;
1235 return MI.getOperand(2).getImm() * Scale;
1236}
1237
1238/// Searches for an increment or decrement of \p Reg before \p MBBI.
1241 ARMCC::CondCodes Pred, Register PredReg, int &Offset) {
1242 Offset = 0;
1243 MachineBasicBlock &MBB = *MBBI->getParent();
1244 MachineBasicBlock::iterator BeginMBBI = MBB.begin();
1245 MachineBasicBlock::iterator EndMBBI = MBB.end();
1246 if (MBBI == BeginMBBI)
1247 return EndMBBI;
1248
1249 // Skip debug values.
1250 MachineBasicBlock::iterator PrevMBBI = std::prev(MBBI);
1251 while (PrevMBBI->isDebugInstr() && PrevMBBI != BeginMBBI)
1252 --PrevMBBI;
1253
1254 Offset = isIncrementOrDecrement(*PrevMBBI, Reg, Pred, PredReg);
1255 return Offset == 0 ? EndMBBI : PrevMBBI;
1256}
1257
1258/// Searches for a increment or decrement of \p Reg after \p MBBI.
1261 ARMCC::CondCodes Pred, Register PredReg, int &Offset,
1262 const TargetRegisterInfo *TRI) {
1263 Offset = 0;
1264 MachineBasicBlock &MBB = *MBBI->getParent();
1265 MachineBasicBlock::iterator EndMBBI = MBB.end();
1266 MachineBasicBlock::iterator NextMBBI = std::next(MBBI);
1267 while (NextMBBI != EndMBBI) {
1268 // Skip debug values.
1269 while (NextMBBI != EndMBBI && NextMBBI->isDebugInstr())
1270 ++NextMBBI;
1271 if (NextMBBI == EndMBBI)
1272 return EndMBBI;
1273
1274 unsigned Off = isIncrementOrDecrement(*NextMBBI, Reg, Pred, PredReg);
1275 if (Off) {
1276 Offset = Off;
1277 return NextMBBI;
1278 }
1279
1280 // SP can only be combined if it is the next instruction after the original
1281 // MBBI, otherwise we may be incrementing the stack pointer (invalidating
1282 // anything below the new pointer) when its frame elements are still in
1283 // use. Other registers can attempt to look further, until a different use
1284 // or def of the register is found.
1285 if (Reg == ARM::SP || NextMBBI->readsRegister(Reg, TRI) ||
1286 NextMBBI->definesRegister(Reg, TRI))
1287 return EndMBBI;
1288
1289 ++NextMBBI;
1290 }
1291 return EndMBBI;
1292}
1293
1294/// Fold proceeding/trailing inc/dec of base register into the
1295/// LDM/STM/VLDM{D|S}/VSTM{D|S} op when possible:
1296///
1297/// stmia rn, <ra, rb, rc>
1298/// rn := rn + 4 * 3;
1299/// =>
1300/// stmia rn!, <ra, rb, rc>
1301///
1302/// rn := rn - 4 * 3;
1303/// ldmia rn, <ra, rb, rc>
1304/// =>
1305/// ldmdb rn!, <ra, rb, rc>
1306bool ARMLoadStoreOpt::MergeBaseUpdateLSMultiple(MachineInstr *MI) {
1307 // Thumb1 is already using updating loads/stores.
1308 if (isThumb1) return false;
1309 LLVM_DEBUG(dbgs() << "Attempting to merge update of: " << *MI);
1310
1311 const MachineOperand &BaseOP = MI->getOperand(0);
1312 Register Base = BaseOP.getReg();
1313 bool BaseKill = BaseOP.isKill();
1314 Register PredReg;
1315 ARMCC::CondCodes Pred = getInstrPredicate(*MI, PredReg);
1316 unsigned Opcode = MI->getOpcode();
1317 DebugLoc DL = MI->getDebugLoc();
1318
1319 // Can't use an updating ld/st if the base register is also a dest
1320 // register. e.g. ldmdb r0!, {r0, r1, r2}. The behavior is undefined.
1321 for (const MachineOperand &MO : llvm::drop_begin(MI->operands(), 2))
1322 if (MO.getReg() == Base)
1323 return false;
1324
1325 int Bytes = getLSMultipleTransferSize(MI);
1326 MachineBasicBlock &MBB = *MI->getParent();
1328 int Offset;
1330 = findIncDecBefore(MBBI, Base, Pred, PredReg, Offset);
1332 if (Mode == ARM_AM::ia && Offset == -Bytes) {
1333 Mode = ARM_AM::db;
1334 } else if (Mode == ARM_AM::ib && Offset == -Bytes) {
1335 Mode = ARM_AM::da;
1336 } else {
1337 MergeInstr = findIncDecAfter(MBBI, Base, Pred, PredReg, Offset, TRI);
1338 if (((Mode != ARM_AM::ia && Mode != ARM_AM::ib) || Offset != Bytes) &&
1339 ((Mode != ARM_AM::da && Mode != ARM_AM::db) || Offset != -Bytes)) {
1340
1341 // We couldn't find an inc/dec to merge. But if the base is dead, we
1342 // can still change to a writeback form as that will save us 2 bytes
1343 // of code size. It can create WAW hazards though, so only do it if
1344 // we're minimizing code size.
1345 if (!STI->hasMinSize() || !BaseKill)
1346 return false;
1347
1348 bool HighRegsUsed = false;
1349 for (const MachineOperand &MO : llvm::drop_begin(MI->operands(), 2))
1350 if (MO.getReg() >= ARM::R8) {
1351 HighRegsUsed = true;
1352 break;
1353 }
1354
1355 if (!HighRegsUsed)
1356 MergeInstr = MBB.end();
1357 else
1358 return false;
1359 }
1360 }
1361 if (MergeInstr != MBB.end()) {
1362 LLVM_DEBUG(dbgs() << " Erasing old increment: " << *MergeInstr);
1363 eraseInstr(MergeInstr);
1364 }
1365
1366 unsigned NewOpc = getUpdatingLSMultipleOpcode(Opcode, Mode);
1367 MachineInstrBuilder MIB = BuildMI(MBB, MBBI, DL, TII->get(NewOpc))
1368 .addReg(Base, getDefRegState(true)) // WB base register
1369 .addReg(Base, getKillRegState(BaseKill))
1370 .addImm(Pred).addReg(PredReg);
1371
1372 // Transfer the rest of operands.
1373 for (const MachineOperand &MO : llvm::drop_begin(MI->operands(), 3))
1374 MIB.add(MO);
1375
1376 // Transfer memoperands.
1377 MIB.setMemRefs(MI->memoperands());
1378
1379 LLVM_DEBUG(dbgs() << " Added new load/store: " << *MIB);
1381 return true;
1382}
1383
1384static unsigned getPreIndexedLoadStoreOpcode(unsigned Opc,
1386 switch (Opc) {
1387 case ARM::LDRi12:
1388 return ARM::LDR_PRE_IMM;
1389 case ARM::STRi12:
1390 return ARM::STR_PRE_IMM;
1391 case ARM::VLDRS:
1392 return Mode == ARM_AM::add ? ARM::VLDMSIA_UPD : ARM::VLDMSDB_UPD;
1393 case ARM::VLDRD:
1394 return Mode == ARM_AM::add ? ARM::VLDMDIA_UPD : ARM::VLDMDDB_UPD;
1395 case ARM::VSTRS:
1396 return Mode == ARM_AM::add ? ARM::VSTMSIA_UPD : ARM::VSTMSDB_UPD;
1397 case ARM::VSTRD:
1398 return Mode == ARM_AM::add ? ARM::VSTMDIA_UPD : ARM::VSTMDDB_UPD;
1399 case ARM::t2LDRi8:
1400 case ARM::t2LDRi12:
1401 return ARM::t2LDR_PRE;
1402 case ARM::t2STRi8:
1403 case ARM::t2STRi12:
1404 return ARM::t2STR_PRE;
1405 default: llvm_unreachable("Unhandled opcode!");
1406 }
1407}
1408
1409static unsigned getPostIndexedLoadStoreOpcode(unsigned Opc,
1411 switch (Opc) {
1412 case ARM::LDRi12:
1413 return ARM::LDR_POST_IMM;
1414 case ARM::STRi12:
1415 return ARM::STR_POST_IMM;
1416 case ARM::VLDRS:
1417 return Mode == ARM_AM::add ? ARM::VLDMSIA_UPD : ARM::VLDMSDB_UPD;
1418 case ARM::VLDRD:
1419 return Mode == ARM_AM::add ? ARM::VLDMDIA_UPD : ARM::VLDMDDB_UPD;
1420 case ARM::VSTRS:
1421 return Mode == ARM_AM::add ? ARM::VSTMSIA_UPD : ARM::VSTMSDB_UPD;
1422 case ARM::VSTRD:
1423 return Mode == ARM_AM::add ? ARM::VSTMDIA_UPD : ARM::VSTMDDB_UPD;
1424 case ARM::t2LDRi8:
1425 case ARM::t2LDRi12:
1426 return ARM::t2LDR_POST;
1427 case ARM::t2LDRBi8:
1428 case ARM::t2LDRBi12:
1429 return ARM::t2LDRB_POST;
1430 case ARM::t2LDRSBi8:
1431 case ARM::t2LDRSBi12:
1432 return ARM::t2LDRSB_POST;
1433 case ARM::t2LDRHi8:
1434 case ARM::t2LDRHi12:
1435 return ARM::t2LDRH_POST;
1436 case ARM::t2LDRSHi8:
1437 case ARM::t2LDRSHi12:
1438 return ARM::t2LDRSH_POST;
1439 case ARM::t2STRi8:
1440 case ARM::t2STRi12:
1441 return ARM::t2STR_POST;
1442 case ARM::t2STRBi8:
1443 case ARM::t2STRBi12:
1444 return ARM::t2STRB_POST;
1445 case ARM::t2STRHi8:
1446 case ARM::t2STRHi12:
1447 return ARM::t2STRH_POST;
1448
1449 case ARM::MVE_VLDRBS16:
1450 return ARM::MVE_VLDRBS16_post;
1451 case ARM::MVE_VLDRBS32:
1452 return ARM::MVE_VLDRBS32_post;
1453 case ARM::MVE_VLDRBU16:
1454 return ARM::MVE_VLDRBU16_post;
1455 case ARM::MVE_VLDRBU32:
1456 return ARM::MVE_VLDRBU32_post;
1457 case ARM::MVE_VLDRHS32:
1458 return ARM::MVE_VLDRHS32_post;
1459 case ARM::MVE_VLDRHU32:
1460 return ARM::MVE_VLDRHU32_post;
1461 case ARM::MVE_VLDRBU8:
1462 return ARM::MVE_VLDRBU8_post;
1463 case ARM::MVE_VLDRHU16:
1464 return ARM::MVE_VLDRHU16_post;
1465 case ARM::MVE_VLDRWU32:
1466 return ARM::MVE_VLDRWU32_post;
1467 case ARM::MVE_VSTRB16:
1468 return ARM::MVE_VSTRB16_post;
1469 case ARM::MVE_VSTRB32:
1470 return ARM::MVE_VSTRB32_post;
1471 case ARM::MVE_VSTRH32:
1472 return ARM::MVE_VSTRH32_post;
1473 case ARM::MVE_VSTRBU8:
1474 return ARM::MVE_VSTRBU8_post;
1475 case ARM::MVE_VSTRHU16:
1476 return ARM::MVE_VSTRHU16_post;
1477 case ARM::MVE_VSTRWU32:
1478 return ARM::MVE_VSTRWU32_post;
1479
1480 default: llvm_unreachable("Unhandled opcode!");
1481 }
1482}
1483
1484/// Fold proceeding/trailing inc/dec of base register into the
1485/// LDR/STR/FLD{D|S}/FST{D|S} op when possible:
1486bool ARMLoadStoreOpt::MergeBaseUpdateLoadStore(MachineInstr *MI) {
1487 // Thumb1 doesn't have updating LDR/STR.
1488 // FIXME: Use LDM/STM with single register instead.
1489 if (isThumb1) return false;
1490 LLVM_DEBUG(dbgs() << "Attempting to merge update of: " << *MI);
1491
1493 bool BaseKill = getLoadStoreBaseOp(*MI).isKill();
1494 unsigned Opcode = MI->getOpcode();
1495 DebugLoc DL = MI->getDebugLoc();
1496 bool isAM5 = (Opcode == ARM::VLDRD || Opcode == ARM::VLDRS ||
1497 Opcode == ARM::VSTRD || Opcode == ARM::VSTRS);
1498 bool isAM2 = (Opcode == ARM::LDRi12 || Opcode == ARM::STRi12);
1499 if (isi32Load(Opcode) || isi32Store(Opcode))
1500 if (MI->getOperand(2).getImm() != 0)
1501 return false;
1502 if (isAM5 && ARM_AM::getAM5Offset(MI->getOperand(2).getImm()) != 0)
1503 return false;
1504
1505 // Can't do the merge if the destination register is the same as the would-be
1506 // writeback register.
1507 if (MI->getOperand(0).getReg() == Base)
1508 return false;
1509
1510 Register PredReg;
1511 ARMCC::CondCodes Pred = getInstrPredicate(*MI, PredReg);
1512 int Bytes = getLSMultipleTransferSize(MI);
1513 MachineBasicBlock &MBB = *MI->getParent();
1515 int Offset;
1517 = findIncDecBefore(MBBI, Base, Pred, PredReg, Offset);
1518 unsigned NewOpc;
1519 if (!isAM5 && Offset == Bytes) {
1520 NewOpc = getPreIndexedLoadStoreOpcode(Opcode, ARM_AM::add);
1521 } else if (Offset == -Bytes) {
1522 NewOpc = getPreIndexedLoadStoreOpcode(Opcode, ARM_AM::sub);
1523 } else {
1524 MergeInstr = findIncDecAfter(MBBI, Base, Pred, PredReg, Offset, TRI);
1525 if (MergeInstr == MBB.end())
1526 return false;
1527
1529 if ((isAM5 && Offset != Bytes) ||
1530 (!isAM5 && !isLegalAddressImm(NewOpc, Offset, TII))) {
1532 if (isAM5 || !isLegalAddressImm(NewOpc, Offset, TII))
1533 return false;
1534 }
1535 }
1536 LLVM_DEBUG(dbgs() << " Erasing old increment: " << *MergeInstr);
1537 eraseInstr(MergeInstr);
1538
1540
1541 bool isLd = isLoadSingle(Opcode);
1542 if (isAM5) {
1543 // VLDM[SD]_UPD, VSTM[SD]_UPD
1544 // (There are no base-updating versions of VLDR/VSTR instructions, but the
1545 // updating load/store-multiple instructions can be used with only one
1546 // register.)
1547 MachineOperand &MO = MI->getOperand(0);
1548 auto MIB = BuildMI(MBB, MBBI, DL, TII->get(NewOpc))
1549 .addReg(Base, getDefRegState(true)) // WB base register
1550 .addReg(Base, getKillRegState(isLd ? BaseKill : false))
1551 .addImm(Pred)
1552 .addReg(PredReg)
1553 .addReg(MO.getReg(), (isLd ? getDefRegState(true)
1554 : getKillRegState(MO.isKill())))
1555 .cloneMemRefs(*MI);
1556 (void)MIB;
1557 LLVM_DEBUG(dbgs() << " Added new instruction: " << *MIB);
1558 } else if (isLd) {
1559 if (isAM2) {
1560 // LDR_PRE, LDR_POST
1561 if (NewOpc == ARM::LDR_PRE_IMM || NewOpc == ARM::LDRB_PRE_IMM) {
1562 auto MIB =
1563 BuildMI(MBB, MBBI, DL, TII->get(NewOpc), MI->getOperand(0).getReg())
1564 .addReg(Base, RegState::Define)
1565 .addReg(Base)
1566 .addImm(Offset)
1567 .addImm(Pred)
1568 .addReg(PredReg)
1569 .cloneMemRefs(*MI);
1570 (void)MIB;
1571 LLVM_DEBUG(dbgs() << " Added new instruction: " << *MIB);
1572 } else {
1574 auto MIB =
1575 BuildMI(MBB, MBBI, DL, TII->get(NewOpc), MI->getOperand(0).getReg())
1576 .addReg(Base, RegState::Define)
1577 .addReg(Base)
1578 .addReg(0)
1579 .addImm(Imm)
1580 .add(predOps(Pred, PredReg))
1581 .cloneMemRefs(*MI);
1582 (void)MIB;
1583 LLVM_DEBUG(dbgs() << " Added new instruction: " << *MIB);
1584 }
1585 } else {
1586 // t2LDR_PRE, t2LDR_POST
1587 auto MIB =
1588 BuildMI(MBB, MBBI, DL, TII->get(NewOpc), MI->getOperand(0).getReg())
1589 .addReg(Base, RegState::Define)
1590 .addReg(Base)
1591 .addImm(Offset)
1592 .add(predOps(Pred, PredReg))
1593 .cloneMemRefs(*MI);
1594 (void)MIB;
1595 LLVM_DEBUG(dbgs() << " Added new instruction: " << *MIB);
1596 }
1597 } else {
1598 MachineOperand &MO = MI->getOperand(0);
1599 // FIXME: post-indexed stores use am2offset_imm, which still encodes
1600 // the vestigial zero-reg offset register. When that's fixed, this clause
1601 // can be removed entirely.
1602 if (isAM2 && NewOpc == ARM::STR_POST_IMM) {
1604 // STR_PRE, STR_POST
1605 auto MIB = BuildMI(MBB, MBBI, DL, TII->get(NewOpc), Base)
1606 .addReg(MO.getReg(), getKillRegState(MO.isKill()))
1607 .addReg(Base)
1608 .addReg(0)
1609 .addImm(Imm)
1610 .add(predOps(Pred, PredReg))
1611 .cloneMemRefs(*MI);
1612 (void)MIB;
1613 LLVM_DEBUG(dbgs() << " Added new instruction: " << *MIB);
1614 } else {
1615 // t2STR_PRE, t2STR_POST
1616 auto MIB = BuildMI(MBB, MBBI, DL, TII->get(NewOpc), Base)
1617 .addReg(MO.getReg(), getKillRegState(MO.isKill()))
1618 .addReg(Base)
1619 .addImm(Offset)
1620 .add(predOps(Pred, PredReg))
1621 .cloneMemRefs(*MI);
1622 (void)MIB;
1623 LLVM_DEBUG(dbgs() << " Added new instruction: " << *MIB);
1624 }
1625 }
1627
1628 return true;
1629}
1630
1631bool ARMLoadStoreOpt::MergeBaseUpdateLSDouble(MachineInstr &MI) {
1632 unsigned Opcode = MI.getOpcode();
1633 assert((Opcode == ARM::t2LDRDi8 || Opcode == ARM::t2STRDi8) &&
1634 "Must have t2STRDi8 or t2LDRDi8");
1635 if (MI.getOperand(3).getImm() != 0)
1636 return false;
1637 LLVM_DEBUG(dbgs() << "Attempting to merge update of: " << MI);
1638
1639 // Behaviour for writeback is undefined if base register is the same as one
1640 // of the others.
1641 const MachineOperand &BaseOp = MI.getOperand(2);
1642 Register Base = BaseOp.getReg();
1643 const MachineOperand &Reg0Op = MI.getOperand(0);
1644 const MachineOperand &Reg1Op = MI.getOperand(1);
1645 if (Reg0Op.getReg() == Base || Reg1Op.getReg() == Base)
1646 return false;
1647
1648 Register PredReg;
1649 ARMCC::CondCodes Pred = getInstrPredicate(MI, PredReg);
1651 MachineBasicBlock &MBB = *MI.getParent();
1652 int Offset;
1654 PredReg, Offset);
1655 unsigned NewOpc;
1656 if (Offset == 8 || Offset == -8) {
1657 NewOpc = Opcode == ARM::t2LDRDi8 ? ARM::t2LDRD_PRE : ARM::t2STRD_PRE;
1658 } else {
1659 MergeInstr = findIncDecAfter(MBBI, Base, Pred, PredReg, Offset, TRI);
1660 if (MergeInstr == MBB.end())
1661 return false;
1662 NewOpc = Opcode == ARM::t2LDRDi8 ? ARM::t2LDRD_POST : ARM::t2STRD_POST;
1663 if (!isLegalAddressImm(NewOpc, Offset, TII))
1664 return false;
1665 }
1666 LLVM_DEBUG(dbgs() << " Erasing old increment: " << *MergeInstr);
1667 eraseInstr(MergeInstr);
1668
1669 DebugLoc DL = MI.getDebugLoc();
1670 MachineInstrBuilder MIB = BuildMI(MBB, MBBI, DL, TII->get(NewOpc));
1671 if (NewOpc == ARM::t2LDRD_PRE || NewOpc == ARM::t2LDRD_POST) {
1672 MIB.add(Reg0Op).add(Reg1Op).addReg(BaseOp.getReg(), RegState::Define);
1673 } else {
1674 assert(NewOpc == ARM::t2STRD_PRE || NewOpc == ARM::t2STRD_POST);
1675 MIB.addReg(BaseOp.getReg(), RegState::Define).add(Reg0Op).add(Reg1Op);
1676 }
1677 MIB.addReg(BaseOp.getReg(), RegState::Kill)
1678 .addImm(Offset).addImm(Pred).addReg(PredReg);
1679 assert(TII->get(Opcode).getNumOperands() == 6 &&
1680 TII->get(NewOpc).getNumOperands() == 7 &&
1681 "Unexpected number of operands in Opcode specification.");
1682
1683 // Transfer implicit operands.
1684 for (const MachineOperand &MO : MI.implicit_operands())
1685 MIB.add(MO);
1686 MIB.cloneMemRefs(MI);
1687
1688 LLVM_DEBUG(dbgs() << " Added new load/store: " << *MIB);
1690 return true;
1691}
1692
1693/// Returns true if instruction is a memory operation that this pass is capable
1694/// of operating on.
1695static bool isMemoryOp(const MachineInstr &MI) {
1696 unsigned Opcode = MI.getOpcode();
1697 switch (Opcode) {
1698 case ARM::VLDRS:
1699 case ARM::VSTRS:
1700 case ARM::VLDRD:
1701 case ARM::VSTRD:
1702 case ARM::LDRi12:
1703 case ARM::STRi12:
1704 case ARM::tLDRi:
1705 case ARM::tSTRi:
1706 case ARM::tLDRspi:
1707 case ARM::tSTRspi:
1708 case ARM::t2LDRi8:
1709 case ARM::t2LDRi12:
1710 case ARM::t2STRi8:
1711 case ARM::t2STRi12:
1712 break;
1713 default:
1714 return false;
1715 }
1716 if (!MI.getOperand(1).isReg())
1717 return false;
1718
1719 // When no memory operands are present, conservatively assume unaligned,
1720 // volatile, unfoldable.
1721 if (!MI.hasOneMemOperand())
1722 return false;
1723
1724 const MachineMemOperand &MMO = **MI.memoperands_begin();
1725
1726 // Don't touch volatile memory accesses - we may be changing their order.
1727 // TODO: We could allow unordered and monotonic atomics here, but we need to
1728 // make sure the resulting ldm/stm is correctly marked as atomic.
1729 if (MMO.isVolatile() || MMO.isAtomic())
1730 return false;
1731
1732 // Unaligned ldr/str is emulated by some kernels, but unaligned ldm/stm is
1733 // not.
1734 if (MMO.getAlign() < Align(4))
1735 return false;
1736
1737 // str <undef> could probably be eliminated entirely, but for now we just want
1738 // to avoid making a mess of it.
1739 // FIXME: Use str <undef> as a wildcard to enable better stm folding.
1740 if (MI.getOperand(0).isReg() && MI.getOperand(0).isUndef())
1741 return false;
1742
1743 // Likewise don't mess with references to undefined addresses.
1744 if (MI.getOperand(1).isUndef())
1745 return false;
1746
1747 return true;
1748}
1749
1752 bool isDef, unsigned NewOpc, unsigned Reg,
1753 bool RegDeadKill, bool RegUndef, unsigned BaseReg,
1754 bool BaseKill, bool BaseUndef, ARMCC::CondCodes Pred,
1755 unsigned PredReg, const TargetInstrInfo *TII,
1756 MachineInstr *MI) {
1757 if (isDef) {
1758 MachineInstrBuilder MIB = BuildMI(MBB, MBBI, MBBI->getDebugLoc(),
1759 TII->get(NewOpc))
1760 .addReg(Reg, getDefRegState(true) | getDeadRegState(RegDeadKill))
1761 .addReg(BaseReg, getKillRegState(BaseKill)|getUndefRegState(BaseUndef));
1762 MIB.addImm(Offset).addImm(Pred).addReg(PredReg);
1763 // FIXME: This is overly conservative; the new instruction accesses 4
1764 // bytes, not 8.
1765 MIB.cloneMemRefs(*MI);
1766 } else {
1767 MachineInstrBuilder MIB = BuildMI(MBB, MBBI, MBBI->getDebugLoc(),
1768 TII->get(NewOpc))
1769 .addReg(Reg, getKillRegState(RegDeadKill) | getUndefRegState(RegUndef))
1770 .addReg(BaseReg, getKillRegState(BaseKill)|getUndefRegState(BaseUndef));
1771 MIB.addImm(Offset).addImm(Pred).addReg(PredReg);
1772 // FIXME: This is overly conservative; the new instruction accesses 4
1773 // bytes, not 8.
1774 MIB.cloneMemRefs(*MI);
1775 }
1776}
1777
1778bool ARMLoadStoreOpt::FixInvalidRegPairOp(MachineBasicBlock &MBB,
1780 MachineInstr *MI = &*MBBI;
1781 unsigned Opcode = MI->getOpcode();
1782 // FIXME: Code/comments below check Opcode == t2STRDi8, but this check returns
1783 // if we see this opcode.
1784 if (Opcode != ARM::LDRD && Opcode != ARM::STRD && Opcode != ARM::t2LDRDi8)
1785 return false;
1786
1787 const MachineOperand &BaseOp = MI->getOperand(2);
1788 Register BaseReg = BaseOp.getReg();
1789 Register EvenReg = MI->getOperand(0).getReg();
1790 Register OddReg = MI->getOperand(1).getReg();
1791 unsigned EvenRegNum = TRI->getDwarfRegNum(EvenReg, false);
1792 unsigned OddRegNum = TRI->getDwarfRegNum(OddReg, false);
1793
1794 // ARM errata 602117: LDRD with base in list may result in incorrect base
1795 // register when interrupted or faulted.
1796 bool Errata602117 = EvenReg == BaseReg &&
1797 (Opcode == ARM::LDRD || Opcode == ARM::t2LDRDi8) && STI->isCortexM3();
1798 // ARM LDRD/STRD needs consecutive registers.
1799 bool NonConsecutiveRegs = (Opcode == ARM::LDRD || Opcode == ARM::STRD) &&
1800 (EvenRegNum % 2 != 0 || EvenRegNum + 1 != OddRegNum);
1801
1802 if (!Errata602117 && !NonConsecutiveRegs)
1803 return false;
1804
1805 bool isT2 = Opcode == ARM::t2LDRDi8 || Opcode == ARM::t2STRDi8;
1806 bool isLd = Opcode == ARM::LDRD || Opcode == ARM::t2LDRDi8;
1807 bool EvenDeadKill = isLd ?
1808 MI->getOperand(0).isDead() : MI->getOperand(0).isKill();
1809 bool EvenUndef = MI->getOperand(0).isUndef();
1810 bool OddDeadKill = isLd ?
1811 MI->getOperand(1).isDead() : MI->getOperand(1).isKill();
1812 bool OddUndef = MI->getOperand(1).isUndef();
1813 bool BaseKill = BaseOp.isKill();
1814 bool BaseUndef = BaseOp.isUndef();
1815 assert((isT2 || !MI->getOperand(3).getReg().isValid()) &&
1816 "register offset not handled below");
1817 int OffImm = getMemoryOpOffset(*MI);
1818 Register PredReg;
1819 ARMCC::CondCodes Pred = getInstrPredicate(*MI, PredReg);
1820
1821 if (OddRegNum > EvenRegNum && OffImm == 0) {
1822 // Ascending register numbers and no offset. It's safe to change it to a
1823 // ldm or stm.
1824 unsigned NewOpc = (isLd)
1825 ? (isT2 ? ARM::t2LDMIA : ARM::LDMIA)
1826 : (isT2 ? ARM::t2STMIA : ARM::STMIA);
1827 if (isLd) {
1828 BuildMI(MBB, MBBI, MBBI->getDebugLoc(), TII->get(NewOpc))
1829 .add(BaseOp)
1830 .addImm(Pred)
1831 .addReg(PredReg)
1832 .addReg(EvenReg, getDefRegState(isLd) | getDeadRegState(EvenDeadKill))
1833 .addReg(OddReg, getDefRegState(isLd) | getDeadRegState(OddDeadKill))
1834 .cloneMemRefs(*MI);
1835 ++NumLDRD2LDM;
1836 } else {
1837 BuildMI(MBB, MBBI, MBBI->getDebugLoc(), TII->get(NewOpc))
1838 .add(BaseOp)
1839 .addImm(Pred)
1840 .addReg(PredReg)
1841 .addReg(EvenReg,
1842 getKillRegState(EvenDeadKill) | getUndefRegState(EvenUndef))
1843 .addReg(OddReg,
1844 getKillRegState(OddDeadKill) | getUndefRegState(OddUndef))
1845 .cloneMemRefs(*MI);
1846 ++NumSTRD2STM;
1847 }
1848 } else {
1849 // Split into two instructions.
1850 unsigned NewOpc = (isLd)
1851 ? (isT2 ? (OffImm < 0 ? ARM::t2LDRi8 : ARM::t2LDRi12) : ARM::LDRi12)
1852 : (isT2 ? (OffImm < 0 ? ARM::t2STRi8 : ARM::t2STRi12) : ARM::STRi12);
1853 // Be extra careful for thumb2. t2LDRi8 can't reference a zero offset,
1854 // so adjust and use t2LDRi12 here for that.
1855 unsigned NewOpc2 = (isLd)
1856 ? (isT2 ? (OffImm+4 < 0 ? ARM::t2LDRi8 : ARM::t2LDRi12) : ARM::LDRi12)
1857 : (isT2 ? (OffImm+4 < 0 ? ARM::t2STRi8 : ARM::t2STRi12) : ARM::STRi12);
1858 // If this is a load, make sure the first load does not clobber the base
1859 // register before the second load reads it.
1860 if (isLd && TRI->regsOverlap(EvenReg, BaseReg)) {
1861 assert(!TRI->regsOverlap(OddReg, BaseReg));
1862 InsertLDR_STR(MBB, MBBI, OffImm + 4, isLd, NewOpc2, OddReg, OddDeadKill,
1863 false, BaseReg, false, BaseUndef, Pred, PredReg, TII, MI);
1864 InsertLDR_STR(MBB, MBBI, OffImm, isLd, NewOpc, EvenReg, EvenDeadKill,
1865 false, BaseReg, BaseKill, BaseUndef, Pred, PredReg, TII,
1866 MI);
1867 } else {
1868 if (OddReg == EvenReg && EvenDeadKill) {
1869 // If the two source operands are the same, the kill marker is
1870 // probably on the first one. e.g.
1871 // t2STRDi8 killed %r5, %r5, killed %r9, 0, 14, %reg0
1872 EvenDeadKill = false;
1873 OddDeadKill = true;
1874 }
1875 // Never kill the base register in the first instruction.
1876 if (EvenReg == BaseReg)
1877 EvenDeadKill = false;
1878 InsertLDR_STR(MBB, MBBI, OffImm, isLd, NewOpc, EvenReg, EvenDeadKill,
1879 EvenUndef, BaseReg, false, BaseUndef, Pred, PredReg, TII,
1880 MI);
1881 InsertLDR_STR(MBB, MBBI, OffImm + 4, isLd, NewOpc2, OddReg, OddDeadKill,
1882 OddUndef, BaseReg, BaseKill, BaseUndef, Pred, PredReg, TII,
1883 MI);
1884 }
1885 if (isLd)
1886 ++NumLDRD2LDR;
1887 else
1888 ++NumSTRD2STR;
1889 }
1890
1891 MBBI = eraseInstr(MBBI);
1892 return true;
1893}
1894
1895/// An optimization pass to turn multiple LDR / STR ops of the same base and
1896/// incrementing offset into LDM / STM ops.
1897bool ARMLoadStoreOpt::LoadStoreMultipleOpti(MachineBasicBlock &MBB) {
1898 MemOpQueue MemOps;
1899 unsigned CurrBase = 0;
1900 unsigned CurrOpc = ~0u;
1901 ARMCC::CondCodes CurrPred = ARMCC::AL;
1902 unsigned Position = 0;
1903 assert(Candidates.size() == 0);
1904 assert(MergeBaseCandidates.size() == 0);
1905 LiveRegsValid = false;
1906
1908 I = MBBI) {
1909 // The instruction in front of the iterator is the one we look at.
1910 MBBI = std::prev(I);
1911 if (FixInvalidRegPairOp(MBB, MBBI))
1912 continue;
1913 ++Position;
1914
1915 if (isMemoryOp(*MBBI)) {
1916 unsigned Opcode = MBBI->getOpcode();
1917 const MachineOperand &MO = MBBI->getOperand(0);
1918 Register Reg = MO.getReg();
1920 Register PredReg;
1921 ARMCC::CondCodes Pred = getInstrPredicate(*MBBI, PredReg);
1923 if (CurrBase == 0) {
1924 // Start of a new chain.
1925 CurrBase = Base;
1926 CurrOpc = Opcode;
1927 CurrPred = Pred;
1928 MemOps.push_back(MemOpQueueEntry(*MBBI, Offset, Position));
1929 continue;
1930 }
1931 // Note: No need to match PredReg in the next if.
1932 if (CurrOpc == Opcode && CurrBase == Base && CurrPred == Pred) {
1933 // Watch out for:
1934 // r4 := ldr [r0, #8]
1935 // r4 := ldr [r0, #4]
1936 // or
1937 // r0 := ldr [r0]
1938 // If a load overrides the base register or a register loaded by
1939 // another load in our chain, we cannot take this instruction.
1940 bool Overlap = false;
1941 if (isLoadSingle(Opcode)) {
1942 Overlap = (Base == Reg);
1943 if (!Overlap) {
1944 for (const MemOpQueueEntry &E : MemOps) {
1945 if (TRI->regsOverlap(Reg, E.MI->getOperand(0).getReg())) {
1946 Overlap = true;
1947 break;
1948 }
1949 }
1950 }
1951 }
1952
1953 if (!Overlap) {
1954 // Check offset and sort memory operation into the current chain.
1955 if (Offset > MemOps.back().Offset) {
1956 MemOps.push_back(MemOpQueueEntry(*MBBI, Offset, Position));
1957 continue;
1958 } else {
1959 MemOpQueue::iterator MI, ME;
1960 for (MI = MemOps.begin(), ME = MemOps.end(); MI != ME; ++MI) {
1961 if (Offset < MI->Offset) {
1962 // Found a place to insert.
1963 break;
1964 }
1965 if (Offset == MI->Offset) {
1966 // Collision, abort.
1967 MI = ME;
1968 break;
1969 }
1970 }
1971 if (MI != MemOps.end()) {
1972 MemOps.insert(MI, MemOpQueueEntry(*MBBI, Offset, Position));
1973 continue;
1974 }
1975 }
1976 }
1977 }
1978
1979 // Don't advance the iterator; The op will start a new chain next.
1980 MBBI = I;
1981 --Position;
1982 // Fallthrough to look into existing chain.
1983 } else if (MBBI->isDebugInstr()) {
1984 continue;
1985 } else if (MBBI->getOpcode() == ARM::t2LDRDi8 ||
1986 MBBI->getOpcode() == ARM::t2STRDi8) {
1987 // ARMPreAllocLoadStoreOpt has already formed some LDRD/STRD instructions
1988 // remember them because we may still be able to merge add/sub into them.
1989 MergeBaseCandidates.push_back(&*MBBI);
1990 }
1991
1992 // If we are here then the chain is broken; Extract candidates for a merge.
1993 if (MemOps.size() > 0) {
1994 FormCandidates(MemOps);
1995 // Reset for the next chain.
1996 CurrBase = 0;
1997 CurrOpc = ~0u;
1998 CurrPred = ARMCC::AL;
1999 MemOps.clear();
2000 }
2001 }
2002 if (MemOps.size() > 0)
2003 FormCandidates(MemOps);
2004
2005 // Sort candidates so they get processed from end to begin of the basic
2006 // block later; This is necessary for liveness calculation.
2007 auto LessThan = [](const MergeCandidate* M0, const MergeCandidate *M1) {
2008 return M0->InsertPos < M1->InsertPos;
2009 };
2010 llvm::sort(Candidates, LessThan);
2011
2012 // Go through list of candidates and merge.
2013 bool Changed = false;
2014 for (const MergeCandidate *Candidate : Candidates) {
2015 if (Candidate->CanMergeToLSMulti || Candidate->CanMergeToLSDouble) {
2016 MachineInstr *Merged = MergeOpsUpdate(*Candidate);
2017 // Merge preceding/trailing base inc/dec into the merged op.
2018 if (Merged) {
2019 Changed = true;
2020 unsigned Opcode = Merged->getOpcode();
2021 if (Opcode == ARM::t2STRDi8 || Opcode == ARM::t2LDRDi8)
2022 MergeBaseUpdateLSDouble(*Merged);
2023 else
2024 MergeBaseUpdateLSMultiple(Merged);
2025 } else {
2026 for (MachineInstr *MI : Candidate->Instrs) {
2027 if (MergeBaseUpdateLoadStore(MI))
2028 Changed = true;
2029 }
2030 }
2031 } else {
2032 assert(Candidate->Instrs.size() == 1);
2033 if (MergeBaseUpdateLoadStore(Candidate->Instrs.front()))
2034 Changed = true;
2035 }
2036 }
2037 Candidates.clear();
2038 // Try to fold add/sub into the LDRD/STRD formed by ARMPreAllocLoadStoreOpt.
2039 for (MachineInstr *MI : MergeBaseCandidates)
2040 MergeBaseUpdateLSDouble(*MI);
2041 MergeBaseCandidates.clear();
2042
2043 return Changed;
2044}
2045
2046/// If this is a exit BB, try merging the return ops ("bx lr" and "mov pc, lr")
2047/// into the preceding stack restore so it directly restore the value of LR
2048/// into pc.
2049/// ldmfd sp!, {..., lr}
2050/// bx lr
2051/// or
2052/// ldmfd sp!, {..., lr}
2053/// mov pc, lr
2054/// =>
2055/// ldmfd sp!, {..., pc}
2056bool ARMLoadStoreOpt::MergeReturnIntoLDM(MachineBasicBlock &MBB) {
2057 // Thumb1 LDM doesn't allow high registers.
2058 if (isThumb1) return false;
2059 if (MBB.empty()) return false;
2060
2062 if (MBBI != MBB.begin() && MBBI != MBB.end() &&
2063 (MBBI->getOpcode() == ARM::BX_RET ||
2064 MBBI->getOpcode() == ARM::tBX_RET ||
2065 MBBI->getOpcode() == ARM::MOVPCLR)) {
2066 MachineBasicBlock::iterator PrevI = std::prev(MBBI);
2067 // Ignore any debug instructions.
2068 while (PrevI->isDebugInstr() && PrevI != MBB.begin())
2069 --PrevI;
2070 MachineInstr &PrevMI = *PrevI;
2071 unsigned Opcode = PrevMI.getOpcode();
2072 if (Opcode == ARM::LDMIA_UPD || Opcode == ARM::LDMDA_UPD ||
2073 Opcode == ARM::LDMDB_UPD || Opcode == ARM::LDMIB_UPD ||
2074 Opcode == ARM::t2LDMIA_UPD || Opcode == ARM::t2LDMDB_UPD) {
2075 MachineOperand &MO = PrevMI.getOperand(PrevMI.getNumOperands() - 1);
2076 if (MO.getReg() != ARM::LR)
2077 return false;
2078 unsigned NewOpc = (isThumb2 ? ARM::t2LDMIA_RET : ARM::LDMIA_RET);
2079 assert(((isThumb2 && Opcode == ARM::t2LDMIA_UPD) ||
2080 Opcode == ARM::LDMIA_UPD) && "Unsupported multiple load-return!");
2081 PrevMI.setDesc(TII->get(NewOpc));
2082 MO.setReg(ARM::PC);
2083 PrevMI.copyImplicitOps(*MBB.getParent(), *MBBI);
2085 return true;
2086 }
2087 }
2088 return false;
2089}
2090
2091bool ARMLoadStoreOpt::CombineMovBx(MachineBasicBlock &MBB) {
2093 if (MBBI == MBB.begin() || MBBI == MBB.end() ||
2094 MBBI->getOpcode() != ARM::tBX_RET)
2095 return false;
2096
2098 --Prev;
2099 if (Prev->getOpcode() != ARM::tMOVr ||
2100 !Prev->definesRegister(ARM::LR, /*TRI=*/nullptr))
2101 return false;
2102
2103 for (auto Use : Prev->uses())
2104 if (Use.isKill()) {
2105 assert(STI->hasV4TOps());
2106 BuildMI(MBB, MBBI, MBBI->getDebugLoc(), TII->get(ARM::tBX))
2107 .addReg(Use.getReg(), RegState::Kill)
2111 eraseInstr(Prev);
2112 return true;
2113 }
2114
2115 llvm_unreachable("tMOVr doesn't kill a reg before tBX_RET?");
2116}
2117
2118bool ARMLoadStoreOpt::runOnMachineFunction(
2119 MachineFunction &Fn, const RegisterClassInfo &RegClassInfo) {
2120 MF = &Fn;
2121 STI = &Fn.getSubtarget<ARMSubtarget>();
2122 TL = STI->getTargetLowering();
2123 AFI = Fn.getInfo<ARMFunctionInfo>();
2124 TII = STI->getInstrInfo();
2125 TRI = STI->getRegisterInfo();
2126 RCI = &RegClassInfo;
2127
2128 isThumb2 = AFI->isThumb2Function();
2129 isThumb1 = AFI->isThumbFunction() && !isThumb2;
2130
2131 bool Modified = false, ModifiedLDMReturn = false;
2132 for (MachineBasicBlock &MBB : Fn) {
2133 Modified |= LoadStoreMultipleOpti(MBB);
2134 if (STI->hasV5TOps() && !AFI->shouldSignReturnAddress())
2135 ModifiedLDMReturn |= MergeReturnIntoLDM(MBB);
2136 if (isThumb1)
2137 Modified |= CombineMovBx(MBB);
2138 }
2139 Modified |= ModifiedLDMReturn;
2140
2141 // If we merged a BX instruction into an LDM, we need to re-calculate whether
2142 // LR is restored. This check needs to consider the whole function, not just
2143 // the instruction(s) we changed, because there may be other BX returns which
2144 // still need LR to be restored.
2145 if (ModifiedLDMReturn)
2147
2148 Allocator.DestroyAll();
2149 return Modified;
2150}
2151
2152bool ARMLoadStoreOptLegacy::runOnMachineFunction(MachineFunction &MF) {
2153 if (skipFunction(MF.getFunction()))
2154 return false;
2155 ARMLoadStoreOpt Impl;
2156 return Impl.runOnMachineFunction(
2157 MF, getAnalysis<MachineRegisterClassInfoWrapperPass>().getRCI());
2158}
2159
2160#define ARM_PREALLOC_LOAD_STORE_OPT_NAME \
2161 "ARM pre- register allocation load / store optimization pass"
2162
2163namespace {
2164
2165/// Pre- register allocation pass that move load / stores from consecutive
2166/// locations close to make it more likely they will be combined later.
2167struct ARMPreAllocLoadStoreOpt {
2169 const DataLayout *TD;
2170 const TargetInstrInfo *TII;
2171 const TargetRegisterInfo *TRI;
2172 const ARMSubtarget *STI;
2175 MachineFunction *MF;
2176
2177 bool runOnMachineFunction(MachineFunction &Fn, AliasAnalysis *AA,
2179
2180private:
2181 bool CanFormLdStDWord(MachineInstr *Op0, MachineInstr *Op1, DebugLoc &dl,
2182 unsigned &NewOpc, Register &EvenReg, Register &OddReg,
2183 Register &BaseReg, int &Offset, Register &PredReg,
2184 ARMCC::CondCodes &Pred, bool &isT2);
2185 bool RescheduleOps(
2187 unsigned Base, bool isLd, DenseMap<MachineInstr *, unsigned> &MI2LocMap,
2189 bool RescheduleLoadStoreInstrs(MachineBasicBlock *MBB);
2190 bool DistributeIncrements();
2191 bool DistributeIncrements(Register Base);
2192};
2193
2194struct ARMPreAllocLoadStoreOptLegacy : public MachineFunctionPass {
2195 static char ID;
2196
2197 ARMPreAllocLoadStoreOptLegacy() : MachineFunctionPass(ID) {}
2198
2199 bool runOnMachineFunction(MachineFunction &Fn) override;
2200
2201 StringRef getPassName() const override {
2203 }
2204
2205 void getAnalysisUsage(AnalysisUsage &AU) const override {
2211 }
2212};
2213
2214char ARMPreAllocLoadStoreOptLegacy::ID = 0;
2215
2216} // end anonymous namespace
2217
2218INITIALIZE_PASS_BEGIN(ARMPreAllocLoadStoreOptLegacy, "arm-prera-ldst-opt",
2221INITIALIZE_PASS_END(ARMPreAllocLoadStoreOptLegacy, "arm-prera-ldst-opt",
2223
2224// Limit the number of instructions to be rescheduled.
2225// FIXME: tune this limit, and/or come up with some better heuristics.
2226static cl::opt<unsigned> InstReorderLimit("arm-prera-ldst-opt-reorder-limit",
2227 cl::init(8), cl::Hidden);
2228
2229bool ARMPreAllocLoadStoreOpt::runOnMachineFunction(MachineFunction &Fn,
2230 AliasAnalysis *AAIn,
2231 MachineDominatorTree *DTIn) {
2233 return false;
2234
2235 AA = AAIn;
2236 DT = DTIn;
2237 TD = &Fn.getDataLayout();
2238 STI = &Fn.getSubtarget<ARMSubtarget>();
2239 TII = STI->getInstrInfo();
2240 TRI = STI->getRegisterInfo();
2241 MRI = &Fn.getRegInfo();
2242 MF = &Fn;
2243
2244 bool Modified = DistributeIncrements();
2245 for (MachineBasicBlock &MFI : Fn)
2246 Modified |= RescheduleLoadStoreInstrs(&MFI);
2247
2248 return Modified;
2249}
2250
2251bool ARMPreAllocLoadStoreOptLegacy::runOnMachineFunction(MachineFunction &Fn) {
2252 if (skipFunction(Fn.getFunction()))
2253 return false;
2254
2255 ARMPreAllocLoadStoreOpt Impl;
2256 AliasAnalysis *AA = &getAnalysis<AAResultsWrapperPass>().getAAResults();
2257 MachineDominatorTree *DT =
2258 &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
2259 return Impl.runOnMachineFunction(Fn, AA, DT);
2260}
2261
2262static bool IsSafeAndProfitableToMove(bool isLd, unsigned Base,
2266 SmallSet<unsigned, 4> &MemRegs,
2267 const TargetRegisterInfo *TRI,
2268 AliasAnalysis *AA) {
2269 // Are there stores / loads / calls between them?
2270 SmallSet<unsigned, 4> AddedRegPressure;
2271 while (++I != E) {
2272 if (I->isDebugInstr() || MemOps.count(&*I))
2273 continue;
2274 if (I->isCall() || I->isTerminator() || I->hasUnmodeledSideEffects())
2275 return false;
2276 if (I->mayStore() || (!isLd && I->mayLoad()))
2277 for (MachineInstr *MemOp : MemOps)
2278 if (I->mayAlias(AA, *MemOp, /*UseTBAA*/ false))
2279 return false;
2280 for (unsigned j = 0, NumOps = I->getNumOperands(); j != NumOps; ++j) {
2281 MachineOperand &MO = I->getOperand(j);
2282 if (!MO.isReg())
2283 continue;
2284 Register Reg = MO.getReg();
2285 if (MO.isDef() && TRI->regsOverlap(Reg, Base))
2286 return false;
2287 if (Reg != Base && !MemRegs.count(Reg))
2288 AddedRegPressure.insert(Reg);
2289 }
2290 }
2291
2292 // Estimate register pressure increase due to the transformation.
2293 if (MemRegs.size() <= 4)
2294 // Ok if we are moving small number of instructions.
2295 return true;
2296 return AddedRegPressure.size() <= MemRegs.size() * 2;
2297}
2298
2299bool ARMPreAllocLoadStoreOpt::CanFormLdStDWord(
2300 MachineInstr *Op0, MachineInstr *Op1, DebugLoc &dl, unsigned &NewOpc,
2301 Register &FirstReg, Register &SecondReg, Register &BaseReg, int &Offset,
2302 Register &PredReg, ARMCC::CondCodes &Pred, bool &isT2) {
2303 // Make sure we're allowed to generate LDRD/STRD.
2304 if (!STI->hasV5TEOps())
2305 return false;
2306
2307 // FIXME: VLDRS / VSTRS -> VLDRD / VSTRD
2308 unsigned Scale = 1;
2309 unsigned Opcode = Op0->getOpcode();
2310 if (Opcode == ARM::LDRi12) {
2311 NewOpc = ARM::LDRD;
2312 } else if (Opcode == ARM::STRi12) {
2313 NewOpc = ARM::STRD;
2314 } else if (Opcode == ARM::t2LDRi8 || Opcode == ARM::t2LDRi12) {
2315 NewOpc = ARM::t2LDRDi8;
2316 Scale = 4;
2317 isT2 = true;
2318 } else if (Opcode == ARM::t2STRi8 || Opcode == ARM::t2STRi12) {
2319 NewOpc = ARM::t2STRDi8;
2320 Scale = 4;
2321 isT2 = true;
2322 } else {
2323 return false;
2324 }
2325
2326 // Make sure the base address satisfies i64 ld / st alignment requirement.
2327 // At the moment, we ignore the memoryoperand's value.
2328 // If we want to use AliasAnalysis, we should check it accordingly.
2329 if (!Op0->hasOneMemOperand() ||
2330 (*Op0->memoperands_begin())->isVolatile() ||
2331 (*Op0->memoperands_begin())->isAtomic())
2332 return false;
2333
2334 Align Alignment = (*Op0->memoperands_begin())->getAlign();
2335 Align ReqAlign = STI->getDualLoadStoreAlignment();
2336 if (Alignment < ReqAlign)
2337 return false;
2338
2339 // Then make sure the immediate offset fits.
2340 int OffImm = getMemoryOpOffset(*Op0);
2341 if (isT2) {
2342 int Limit = (1 << 8) * Scale;
2343 if (OffImm >= Limit || (OffImm <= -Limit) || (OffImm & (Scale-1)))
2344 return false;
2345 Offset = OffImm;
2346 } else {
2348 if (OffImm < 0) {
2350 OffImm = - OffImm;
2351 }
2352 int Limit = (1 << 8) * Scale;
2353 if (OffImm >= Limit || (OffImm & (Scale-1)))
2354 return false;
2355 Offset = ARM_AM::getAM3Opc(AddSub, OffImm);
2356 }
2357 FirstReg = Op0->getOperand(0).getReg();
2358 SecondReg = Op1->getOperand(0).getReg();
2359 if (FirstReg == SecondReg)
2360 return false;
2361 BaseReg = Op0->getOperand(1).getReg();
2362 Pred = getInstrPredicate(*Op0, PredReg);
2363 dl = Op0->getDebugLoc();
2364 return true;
2365}
2366
2367bool ARMPreAllocLoadStoreOpt::RescheduleOps(
2368 MachineBasicBlock *MBB, SmallVectorImpl<MachineInstr *> &Ops, unsigned Base,
2369 bool isLd, DenseMap<MachineInstr *, unsigned> &MI2LocMap,
2370 SmallDenseMap<Register, SmallVector<MachineInstr *>, 8> &RegisterMap) {
2371 bool RetVal = false;
2372
2373 // Sort by offset (in reverse order).
2374 llvm::sort(Ops, [](const MachineInstr *LHS, const MachineInstr *RHS) {
2375 int LOffset = getMemoryOpOffset(*LHS);
2376 int ROffset = getMemoryOpOffset(*RHS);
2377 assert(LHS == RHS || LOffset != ROffset);
2378 return LOffset > ROffset;
2379 });
2380
2381 // The loads / stores of the same base are in order. Scan them from first to
2382 // last and check for the following:
2383 // 1. Any def of base.
2384 // 2. Any gaps.
2385 while (Ops.size() > 1) {
2386 unsigned FirstLoc = ~0U;
2387 unsigned LastLoc = 0;
2388 MachineInstr *FirstOp = nullptr;
2389 MachineInstr *LastOp = nullptr;
2390 int LastOffset = 0;
2391 unsigned LastOpcode = 0;
2392 unsigned LastBytes = 0;
2393 unsigned NumMove = 0;
2394 for (MachineInstr *Op : llvm::reverse(Ops)) {
2395 // Make sure each operation has the same kind.
2396 unsigned LSMOpcode
2397 = getLoadStoreMultipleOpcode(Op->getOpcode(), ARM_AM::ia);
2398 if (LastOpcode && LSMOpcode != LastOpcode)
2399 break;
2400
2401 // Check that we have a continuous set of offsets.
2402 int Offset = getMemoryOpOffset(*Op);
2403 unsigned Bytes = getLSMultipleTransferSize(Op);
2404 if (LastBytes) {
2405 if (Bytes != LastBytes || Offset != (LastOffset + (int)Bytes))
2406 break;
2407 }
2408
2409 // Don't try to reschedule too many instructions.
2410 if (NumMove == InstReorderLimit)
2411 break;
2412
2413 // Found a mergeable instruction; save information about it.
2414 ++NumMove;
2415 LastOffset = Offset;
2416 LastBytes = Bytes;
2417 LastOpcode = LSMOpcode;
2418
2419 unsigned Loc = MI2LocMap[Op];
2420 if (Loc <= FirstLoc) {
2421 FirstLoc = Loc;
2422 FirstOp = Op;
2423 }
2424 if (Loc >= LastLoc) {
2425 LastLoc = Loc;
2426 LastOp = Op;
2427 }
2428 }
2429
2430 if (NumMove <= 1)
2431 Ops.pop_back();
2432 else {
2433 SmallPtrSet<MachineInstr*, 4> MemOps;
2434 SmallSet<unsigned, 4> MemRegs;
2435 for (size_t i = Ops.size() - NumMove, e = Ops.size(); i != e; ++i) {
2436 MemOps.insert(Ops[i]);
2437 MemRegs.insert(Ops[i]->getOperand(0).getReg());
2438 }
2439
2440 // Be conservative, if the instructions are too far apart, don't
2441 // move them. We want to limit the increase of register pressure.
2442 bool DoMove = (LastLoc - FirstLoc) <= NumMove*4; // FIXME: Tune this.
2443 if (DoMove)
2444 DoMove = IsSafeAndProfitableToMove(isLd, Base, FirstOp, LastOp,
2445 MemOps, MemRegs, TRI, AA);
2446 if (!DoMove) {
2447 for (unsigned i = 0; i != NumMove; ++i)
2448 Ops.pop_back();
2449 } else {
2450 // This is the new location for the loads / stores.
2451 MachineBasicBlock::iterator InsertPos = isLd ? FirstOp : LastOp;
2452 while (InsertPos != MBB->end() &&
2453 (MemOps.count(&*InsertPos) || InsertPos->isDebugInstr()))
2454 ++InsertPos;
2455
2456 // If we are moving a pair of loads / stores, see if it makes sense
2457 // to try to allocate a pair of registers that can form register pairs.
2458 MachineInstr *Op0 = Ops.back();
2459 MachineInstr *Op1 = Ops[Ops.size()-2];
2460 Register FirstReg, SecondReg;
2461 Register BaseReg, PredReg;
2463 bool isT2 = false;
2464 unsigned NewOpc = 0;
2465 int Offset = 0;
2466 DebugLoc dl;
2467 if (NumMove == 2 && CanFormLdStDWord(Op0, Op1, dl, NewOpc,
2468 FirstReg, SecondReg, BaseReg,
2469 Offset, PredReg, Pred, isT2)) {
2470 Ops.pop_back();
2471 Ops.pop_back();
2472
2473 const MCInstrDesc &MCID = TII->get(NewOpc);
2474 const TargetRegisterClass *TRC = TII->getRegClass(MCID, 0);
2475 MRI->constrainRegClass(FirstReg, TRC);
2476 MRI->constrainRegClass(SecondReg, TRC);
2477
2478 // Form the pair instruction.
2479 if (isLd) {
2480 MachineInstrBuilder MIB = BuildMI(*MBB, InsertPos, dl, MCID)
2481 .addReg(FirstReg, RegState::Define)
2482 .addReg(SecondReg, RegState::Define)
2483 .addReg(BaseReg);
2484 // FIXME: We're converting from LDRi12 to an insn that still
2485 // uses addrmode2, so we need an explicit offset reg. It should
2486 // always by reg0 since we're transforming LDRi12s.
2487 if (!isT2)
2488 MIB.addReg(0);
2489 MIB.addImm(Offset).addImm(Pred).addReg(PredReg);
2490 MIB.cloneMergedMemRefs({Op0, Op1});
2491 LLVM_DEBUG(dbgs() << "Formed " << *MIB << "\n");
2492 ++NumLDRDFormed;
2493 } else {
2494 MachineInstrBuilder MIB = BuildMI(*MBB, InsertPos, dl, MCID)
2495 .addReg(FirstReg)
2496 .addReg(SecondReg)
2497 .addReg(BaseReg);
2498 // FIXME: We're converting from LDRi12 to an insn that still
2499 // uses addrmode2, so we need an explicit offset reg. It should
2500 // always by reg0 since we're transforming STRi12s.
2501 if (!isT2)
2502 MIB.addReg(0);
2503 MIB.addImm(Offset).addImm(Pred).addReg(PredReg);
2504 MIB.cloneMergedMemRefs({Op0, Op1});
2505 LLVM_DEBUG(dbgs() << "Formed " << *MIB << "\n");
2506 ++NumSTRDFormed;
2507 }
2508 MBB->erase(Op0);
2509 MBB->erase(Op1);
2510
2511 if (!isT2) {
2512 // Add register allocation hints to form register pairs.
2513 MRI->setRegAllocationHint(FirstReg, ARMRI::RegPairEven, SecondReg);
2514 MRI->setRegAllocationHint(SecondReg, ARMRI::RegPairOdd, FirstReg);
2515 }
2516 } else {
2517 for (unsigned i = 0; i != NumMove; ++i) {
2518 MachineInstr *Op = Ops.pop_back_val();
2519 if (isLd) {
2520 // Populate RegisterMap with all Registers defined by loads.
2521 Register Reg = Op->getOperand(0).getReg();
2522 RegisterMap[Reg];
2523 }
2524
2525 MBB->splice(InsertPos, MBB, Op);
2526 }
2527 }
2528
2529 NumLdStMoved += NumMove;
2530 RetVal = true;
2531 }
2532 }
2533 }
2534
2535 return RetVal;
2536}
2537
2539 std::function<void(MachineOperand &)> Fn) {
2540 if (MI->isNonListDebugValue()) {
2541 auto &Op = MI->getOperand(0);
2542 if (Op.isReg())
2543 Fn(Op);
2544 } else {
2545 for (unsigned I = 2; I < MI->getNumOperands(); I++) {
2546 auto &Op = MI->getOperand(I);
2547 if (Op.isReg())
2548 Fn(Op);
2549 }
2550 }
2551}
2552
2553// Update the RegisterMap with the instruction that was moved because a
2554// DBG_VALUE_LIST may need to be moved again.
2557 MachineInstr *DbgValueListInstr, MachineInstr *InstrToReplace) {
2558
2559 forEachDbgRegOperand(DbgValueListInstr, [&](MachineOperand &Op) {
2560 auto RegIt = RegisterMap.find(Op.getReg());
2561 if (RegIt == RegisterMap.end())
2562 return;
2563 auto &InstrVec = RegIt->getSecond();
2564 llvm::replace(InstrVec, InstrToReplace, DbgValueListInstr);
2565 });
2566}
2567
2569 auto DbgVar = DebugVariable(MI->getDebugVariable(), MI->getDebugExpression(),
2570 MI->getDebugLoc()->getInlinedAt());
2571 return DbgVar;
2572}
2573
2574bool
2575ARMPreAllocLoadStoreOpt::RescheduleLoadStoreInstrs(MachineBasicBlock *MBB) {
2576 bool RetVal = false;
2577
2578 DenseMap<MachineInstr *, unsigned> MI2LocMap;
2579 using Base2InstMap = DenseMap<unsigned, SmallVector<MachineInstr *, 4>>;
2580 using BaseVec = SmallVector<unsigned, 4>;
2581 Base2InstMap Base2LdsMap;
2582 Base2InstMap Base2StsMap;
2583 BaseVec LdBases;
2584 BaseVec StBases;
2585 // This map is used to track the relationship between the virtual
2586 // register that is the result of a load that is moved and the DBG_VALUE
2587 // MachineInstr pointer that uses that virtual register.
2588 SmallDenseMap<Register, SmallVector<MachineInstr *>, 8> RegisterMap;
2589
2590 unsigned Loc = 0;
2593 while (MBBI != E) {
2594 for (; MBBI != E; ++MBBI) {
2595 MachineInstr &MI = *MBBI;
2596 if (MI.isCall() || MI.isTerminator()) {
2597 // Stop at barriers.
2598 ++MBBI;
2599 break;
2600 }
2601
2602 if (!MI.isDebugInstr())
2603 MI2LocMap[&MI] = ++Loc;
2604
2605 if (!isMemoryOp(MI))
2606 continue;
2607 Register PredReg;
2608 if (getInstrPredicate(MI, PredReg) != ARMCC::AL)
2609 continue;
2610
2611 int Opc = MI.getOpcode();
2612 bool isLd = isLoadSingle(Opc);
2613 Register Base = MI.getOperand(1).getReg();
2615 bool StopHere = false;
2616 auto FindBases = [&](Base2InstMap &Base2Ops, BaseVec &Bases) {
2617 auto [BI, Inserted] = Base2Ops.try_emplace(Base);
2618 if (Inserted) {
2619 BI->second.push_back(&MI);
2620 Bases.push_back(Base);
2621 return;
2622 }
2623 for (const MachineInstr *MI : BI->second) {
2624 if (Offset == getMemoryOpOffset(*MI)) {
2625 StopHere = true;
2626 break;
2627 }
2628 }
2629 if (!StopHere)
2630 BI->second.push_back(&MI);
2631 };
2632
2633 if (isLd)
2634 FindBases(Base2LdsMap, LdBases);
2635 else
2636 FindBases(Base2StsMap, StBases);
2637
2638 if (StopHere) {
2639 // Found a duplicate (a base+offset combination that's seen earlier).
2640 // Backtrack.
2641 --Loc;
2642 break;
2643 }
2644 }
2645
2646 // Re-schedule loads.
2647 for (unsigned Base : LdBases) {
2648 SmallVectorImpl<MachineInstr *> &Lds = Base2LdsMap[Base];
2649 if (Lds.size() > 1)
2650 RetVal |= RescheduleOps(MBB, Lds, Base, true, MI2LocMap, RegisterMap);
2651 }
2652
2653 // Re-schedule stores.
2654 for (unsigned Base : StBases) {
2655 SmallVectorImpl<MachineInstr *> &Sts = Base2StsMap[Base];
2656 if (Sts.size() > 1)
2657 RetVal |= RescheduleOps(MBB, Sts, Base, false, MI2LocMap, RegisterMap);
2658 }
2659
2660 if (MBBI != E) {
2661 Base2LdsMap.clear();
2662 Base2StsMap.clear();
2663 LdBases.clear();
2664 StBases.clear();
2665 }
2666 }
2667
2668 // Reschedule DBG_VALUEs to match any loads that were moved. When a load is
2669 // sunk beyond a DBG_VALUE that is referring to it, the DBG_VALUE becomes a
2670 // use-before-def, resulting in a loss of debug info.
2671
2672 // Example:
2673 // Before the Pre Register Allocation Load Store Pass
2674 // inst_a
2675 // %2 = ld ...
2676 // inst_b
2677 // DBG_VALUE %2, "x", ...
2678 // %3 = ld ...
2679
2680 // After the Pass:
2681 // inst_a
2682 // inst_b
2683 // DBG_VALUE %2, "x", ...
2684 // %2 = ld ...
2685 // %3 = ld ...
2686
2687 // The code below addresses this by moving the DBG_VALUE to the position
2688 // immediately after the load.
2689
2690 // Example:
2691 // After the code below:
2692 // inst_a
2693 // inst_b
2694 // %2 = ld ...
2695 // DBG_VALUE %2, "x", ...
2696 // %3 = ld ...
2697
2698 // The algorithm works in two phases: First RescheduleOps() populates the
2699 // RegisterMap with registers that were moved as keys, there is no value
2700 // inserted. In the next phase, every MachineInstr in a basic block is
2701 // iterated over. If it is a valid DBG_VALUE or DBG_VALUE_LIST and it uses one
2702 // or more registers in the RegisterMap, the RegisterMap and InstrMap are
2703 // populated with the MachineInstr. If the DBG_VALUE or DBG_VALUE_LIST
2704 // describes debug information for a variable that already exists in the
2705 // DbgValueSinkCandidates, the MachineInstr in the DbgValueSinkCandidates must
2706 // be set to undef. If the current MachineInstr is a load that was moved,
2707 // undef the corresponding DBG_VALUE or DBG_VALUE_LIST and clone it to below
2708 // the load.
2709
2710 // To illustrate the above algorithm visually let's take this example.
2711
2712 // Before the Pre Register Allocation Load Store Pass:
2713 // %2 = ld ...
2714 // DBG_VALUE %2, A, .... # X
2715 // DBG_VALUE 0, A, ... # Y
2716 // %3 = ld ...
2717 // DBG_VALUE %3, A, ..., # Z
2718 // %4 = ld ...
2719
2720 // After Pre Register Allocation Load Store Pass:
2721 // DBG_VALUE %2, A, .... # X
2722 // DBG_VALUE 0, A, ... # Y
2723 // DBG_VALUE %3, A, ..., # Z
2724 // %2 = ld ...
2725 // %3 = ld ...
2726 // %4 = ld ...
2727
2728 // The algorithm below does the following:
2729
2730 // In the beginning, the RegisterMap will have been populated with the virtual
2731 // registers %2, and %3, the DbgValueSinkCandidates and the InstrMap will be
2732 // empty. DbgValueSinkCandidates = {}, RegisterMap = {2 -> {}, 3 -> {}},
2733 // InstrMap {}
2734 // -> DBG_VALUE %2, A, .... # X
2735 // DBG_VALUE 0, A, ... # Y
2736 // DBG_VALUE %3, A, ..., # Z
2737 // %2 = ld ...
2738 // %3 = ld ...
2739 // %4 = ld ...
2740
2741 // After the first DBG_VALUE (denoted with an X) is processed, the
2742 // DbgValueSinkCandidates and InstrMap will be populated and the RegisterMap
2743 // entry for %2 will be populated as well. DbgValueSinkCandidates = {A -> X},
2744 // RegisterMap = {2 -> {X}, 3 -> {}}, InstrMap {X -> 2}
2745 // DBG_VALUE %2, A, .... # X
2746 // -> DBG_VALUE 0, A, ... # Y
2747 // DBG_VALUE %3, A, ..., # Z
2748 // %2 = ld ...
2749 // %3 = ld ...
2750 // %4 = ld ...
2751
2752 // After the DBG_VALUE Y is processed, the DbgValueSinkCandidates is updated
2753 // to now hold Y for A and the RegisterMap is also updated to remove X from
2754 // %2, this is because both X and Y describe the same debug variable A. X is
2755 // also updated to have a $noreg as the first operand.
2756 // DbgValueSinkCandidates = {A -> {Y}}, RegisterMap = {2 -> {}, 3 -> {}},
2757 // InstrMap = {X-> 2}
2758 // DBG_VALUE $noreg, A, .... # X
2759 // DBG_VALUE 0, A, ... # Y
2760 // -> DBG_VALUE %3, A, ..., # Z
2761 // %2 = ld ...
2762 // %3 = ld ...
2763 // %4 = ld ...
2764
2765 // After DBG_VALUE Z is processed, the DbgValueSinkCandidates is updated to
2766 // hold Z fr A, the RegisterMap is updated to hold Z for %3, and the InstrMap
2767 // is updated to have Z mapped to %3. This is again because Z describes the
2768 // debug variable A, Y is not updated to have $noreg as first operand because
2769 // its first operand is an immediate, not a register.
2770 // DbgValueSinkCandidates = {A -> {Z}}, RegisterMap = {2 -> {}, 3 -> {Z}},
2771 // InstrMap = {X -> 2, Z -> 3}
2772 // DBG_VALUE $noreg, A, .... # X
2773 // DBG_VALUE 0, A, ... # Y
2774 // DBG_VALUE %3, A, ..., # Z
2775 // -> %2 = ld ...
2776 // %3 = ld ...
2777 // %4 = ld ...
2778
2779 // Nothing happens here since the RegisterMap for %2 contains no value.
2780 // DbgValueSinkCandidates = {A -> {Z}}, RegisterMap = {2 -> {}, 3 -> {Z}},
2781 // InstrMap = {X -> 2, Z -> 3}
2782 // DBG_VALUE $noreg, A, .... # X
2783 // DBG_VALUE 0, A, ... # Y
2784 // DBG_VALUE %3, A, ..., # Z
2785 // %2 = ld ...
2786 // -> %3 = ld ...
2787 // %4 = ld ...
2788
2789 // Since the RegisterMap contains Z as a value for %3, the MachineInstr
2790 // pointer Z is copied to come after the load for %3 and the old Z's first
2791 // operand is changed to $noreg the Basic Block iterator is moved to after the
2792 // DBG_VALUE Z's new position.
2793 // DbgValueSinkCandidates = {A -> {Z}}, RegisterMap = {2 -> {}, 3 -> {Z}},
2794 // InstrMap = {X -> 2, Z -> 3}
2795 // DBG_VALUE $noreg, A, .... # X
2796 // DBG_VALUE 0, A, ... # Y
2797 // DBG_VALUE $noreg, A, ..., # Old Z
2798 // %2 = ld ...
2799 // %3 = ld ...
2800 // DBG_VALUE %3, A, ..., # Z
2801 // -> %4 = ld ...
2802
2803 // Nothing happens for %4 and the algorithm exits having processed the entire
2804 // Basic Block.
2805 // DbgValueSinkCandidates = {A -> {Z}}, RegisterMap = {2 -> {}, 3 -> {Z}},
2806 // InstrMap = {X -> 2, Z -> 3}
2807 // DBG_VALUE $noreg, A, .... # X
2808 // DBG_VALUE 0, A, ... # Y
2809 // DBG_VALUE $noreg, A, ..., # Old Z
2810 // %2 = ld ...
2811 // %3 = ld ...
2812 // DBG_VALUE %3, A, ..., # Z
2813 // %4 = ld ...
2814
2815 // This map is used to track the relationship between
2816 // a Debug Variable and the DBG_VALUE MachineInstr pointer that describes the
2817 // debug information for that Debug Variable.
2818 SmallDenseMap<DebugVariable, MachineInstr *, 8> DbgValueSinkCandidates;
2819 // This map is used to track the relationship between a DBG_VALUE or
2820 // DBG_VALUE_LIST MachineInstr pointer and Registers that it uses.
2821 SmallDenseMap<MachineInstr *, SmallVector<Register>, 8> InstrMap;
2822 for (MBBI = MBB->begin(), E = MBB->end(); MBBI != E; ++MBBI) {
2823 MachineInstr &MI = *MBBI;
2824
2825 auto PopulateRegisterAndInstrMapForDebugInstr = [&](Register Reg) {
2826 auto RegIt = RegisterMap.find(Reg);
2827 if (RegIt == RegisterMap.end())
2828 return;
2829 auto &InstrVec = RegIt->getSecond();
2830 InstrVec.push_back(&MI);
2831 InstrMap[&MI].push_back(Reg);
2832 };
2833
2834 if (MI.isDebugValue()) {
2835 assert(MI.getDebugVariable() &&
2836 "DBG_VALUE or DBG_VALUE_LIST must contain a DILocalVariable");
2837
2839 // If the first operand is a register and it exists in the RegisterMap, we
2840 // know this is a DBG_VALUE that uses the result of a load that was moved,
2841 // and is therefore a candidate to also be moved, add it to the
2842 // RegisterMap and InstrMap.
2843 forEachDbgRegOperand(&MI, [&](MachineOperand &Op) {
2844 PopulateRegisterAndInstrMapForDebugInstr(Op.getReg());
2845 });
2846
2847 // If the current DBG_VALUE describes the same variable as one of the
2848 // in-flight DBG_VALUEs, remove the candidate from the list and set it to
2849 // undef. Moving one DBG_VALUE past another would result in the variable's
2850 // value going back in time when stepping through the block in the
2851 // debugger.
2852 auto InstrIt = DbgValueSinkCandidates.find(DbgVar);
2853 if (InstrIt != DbgValueSinkCandidates.end()) {
2854 auto *Instr = InstrIt->getSecond();
2855 auto RegIt = InstrMap.find(Instr);
2856 if (RegIt != InstrMap.end()) {
2857 const auto &RegVec = RegIt->getSecond();
2858 // For every Register in the RegVec, remove the MachineInstr in the
2859 // RegisterMap that describes the DbgVar.
2860 for (auto &Reg : RegVec) {
2861 auto RegIt = RegisterMap.find(Reg);
2862 if (RegIt == RegisterMap.end())
2863 continue;
2864 auto &InstrVec = RegIt->getSecond();
2865 auto IsDbgVar = [&](MachineInstr *I) -> bool {
2867 return Var == DbgVar;
2868 };
2869
2870 llvm::erase_if(InstrVec, IsDbgVar);
2871 }
2873 [&](MachineOperand &Op) { Op.setReg(0); });
2874 }
2875 }
2876 DbgValueSinkCandidates[DbgVar] = &MI;
2877 } else {
2878 // If the first operand of a load matches with a DBG_VALUE in RegisterMap,
2879 // then move that DBG_VALUE to below the load.
2880 auto Opc = MI.getOpcode();
2881 if (!isLoadSingle(Opc))
2882 continue;
2883 auto Reg = MI.getOperand(0).getReg();
2884 auto RegIt = RegisterMap.find(Reg);
2885 if (RegIt == RegisterMap.end())
2886 continue;
2887 auto &DbgInstrVec = RegIt->getSecond();
2888 if (!DbgInstrVec.size())
2889 continue;
2890 for (auto *DbgInstr : DbgInstrVec) {
2891 MachineBasicBlock::iterator InsertPos = std::next(MBBI);
2892 auto *ClonedMI = MI.getMF()->CloneMachineInstr(DbgInstr);
2893 MBB->insert(InsertPos, ClonedMI);
2894 MBBI++;
2895 // Erase the entry into the DbgValueSinkCandidates for the DBG_VALUE
2896 // that was moved.
2897 auto DbgVar = createDebugVariableFromMachineInstr(DbgInstr);
2898 // Erase DbgVar from DbgValueSinkCandidates if still present. If the
2899 // instruction is a DBG_VALUE_LIST, it may have already been erased from
2900 // DbgValueSinkCandidates.
2901 DbgValueSinkCandidates.erase(DbgVar);
2902 // Zero out original dbg instr
2903 forEachDbgRegOperand(DbgInstr,
2904 [&](MachineOperand &Op) { Op.setReg(0); });
2905 // Update RegisterMap with ClonedMI because it might have to be moved
2906 // again.
2907 if (DbgInstr->isDebugValueList())
2908 updateRegisterMapForDbgValueListAfterMove(RegisterMap, ClonedMI,
2909 DbgInstr);
2910 }
2911 }
2912 }
2913 return RetVal;
2914}
2915
2916// Get the Base register operand index from the memory access MachineInst if we
2917// should attempt to distribute postinc on it. Return -1 if not of a valid
2918// instruction type. If it returns an index, it is assumed that instruction is a
2919// r+i indexing mode, and getBaseOperandIndex() + 1 is the Offset index.
2921 switch (MI.getOpcode()) {
2922 case ARM::MVE_VLDRBS16:
2923 case ARM::MVE_VLDRBS32:
2924 case ARM::MVE_VLDRBU16:
2925 case ARM::MVE_VLDRBU32:
2926 case ARM::MVE_VLDRHS32:
2927 case ARM::MVE_VLDRHU32:
2928 case ARM::MVE_VLDRBU8:
2929 case ARM::MVE_VLDRHU16:
2930 case ARM::MVE_VLDRWU32:
2931 case ARM::MVE_VSTRB16:
2932 case ARM::MVE_VSTRB32:
2933 case ARM::MVE_VSTRH32:
2934 case ARM::MVE_VSTRBU8:
2935 case ARM::MVE_VSTRHU16:
2936 case ARM::MVE_VSTRWU32:
2937 case ARM::t2LDRHi8:
2938 case ARM::t2LDRHi12:
2939 case ARM::t2LDRSHi8:
2940 case ARM::t2LDRSHi12:
2941 case ARM::t2LDRBi8:
2942 case ARM::t2LDRBi12:
2943 case ARM::t2LDRSBi8:
2944 case ARM::t2LDRSBi12:
2945 case ARM::t2STRBi8:
2946 case ARM::t2STRBi12:
2947 case ARM::t2STRHi8:
2948 case ARM::t2STRHi12:
2949 return 1;
2950 case ARM::MVE_VLDRBS16_post:
2951 case ARM::MVE_VLDRBS32_post:
2952 case ARM::MVE_VLDRBU16_post:
2953 case ARM::MVE_VLDRBU32_post:
2954 case ARM::MVE_VLDRHS32_post:
2955 case ARM::MVE_VLDRHU32_post:
2956 case ARM::MVE_VLDRBU8_post:
2957 case ARM::MVE_VLDRHU16_post:
2958 case ARM::MVE_VLDRWU32_post:
2959 case ARM::MVE_VSTRB16_post:
2960 case ARM::MVE_VSTRB32_post:
2961 case ARM::MVE_VSTRH32_post:
2962 case ARM::MVE_VSTRBU8_post:
2963 case ARM::MVE_VSTRHU16_post:
2964 case ARM::MVE_VSTRWU32_post:
2965 case ARM::MVE_VLDRBS16_pre:
2966 case ARM::MVE_VLDRBS32_pre:
2967 case ARM::MVE_VLDRBU16_pre:
2968 case ARM::MVE_VLDRBU32_pre:
2969 case ARM::MVE_VLDRHS32_pre:
2970 case ARM::MVE_VLDRHU32_pre:
2971 case ARM::MVE_VLDRBU8_pre:
2972 case ARM::MVE_VLDRHU16_pre:
2973 case ARM::MVE_VLDRWU32_pre:
2974 case ARM::MVE_VSTRB16_pre:
2975 case ARM::MVE_VSTRB32_pre:
2976 case ARM::MVE_VSTRH32_pre:
2977 case ARM::MVE_VSTRBU8_pre:
2978 case ARM::MVE_VSTRHU16_pre:
2979 case ARM::MVE_VSTRWU32_pre:
2980 return 2;
2981 }
2982 return -1;
2983}
2984
2986 switch (MI.getOpcode()) {
2987 case ARM::MVE_VLDRBS16_post:
2988 case ARM::MVE_VLDRBS32_post:
2989 case ARM::MVE_VLDRBU16_post:
2990 case ARM::MVE_VLDRBU32_post:
2991 case ARM::MVE_VLDRHS32_post:
2992 case ARM::MVE_VLDRHU32_post:
2993 case ARM::MVE_VLDRBU8_post:
2994 case ARM::MVE_VLDRHU16_post:
2995 case ARM::MVE_VLDRWU32_post:
2996 case ARM::MVE_VSTRB16_post:
2997 case ARM::MVE_VSTRB32_post:
2998 case ARM::MVE_VSTRH32_post:
2999 case ARM::MVE_VSTRBU8_post:
3000 case ARM::MVE_VSTRHU16_post:
3001 case ARM::MVE_VSTRWU32_post:
3002 return true;
3003 }
3004 return false;
3005}
3006
3008 switch (MI.getOpcode()) {
3009 case ARM::MVE_VLDRBS16_pre:
3010 case ARM::MVE_VLDRBS32_pre:
3011 case ARM::MVE_VLDRBU16_pre:
3012 case ARM::MVE_VLDRBU32_pre:
3013 case ARM::MVE_VLDRHS32_pre:
3014 case ARM::MVE_VLDRHU32_pre:
3015 case ARM::MVE_VLDRBU8_pre:
3016 case ARM::MVE_VLDRHU16_pre:
3017 case ARM::MVE_VLDRWU32_pre:
3018 case ARM::MVE_VSTRB16_pre:
3019 case ARM::MVE_VSTRB32_pre:
3020 case ARM::MVE_VSTRH32_pre:
3021 case ARM::MVE_VSTRBU8_pre:
3022 case ARM::MVE_VSTRHU16_pre:
3023 case ARM::MVE_VSTRWU32_pre:
3024 return true;
3025 }
3026 return false;
3027}
3028
3029// Given a memory access Opcode, check that the give Imm would be a valid Offset
3030// for this instruction (same as isLegalAddressImm), Or if the instruction
3031// could be easily converted to one where that was valid. For example converting
3032// t2LDRi12 to t2LDRi8 for negative offsets. Works in conjunction with
3033// AdjustBaseAndOffset below.
3034static bool isLegalOrConvertibleAddressImm(unsigned Opcode, int Imm,
3035 const TargetInstrInfo *TII,
3036 int &CodesizeEstimate) {
3037 if (isLegalAddressImm(Opcode, Imm, TII))
3038 return true;
3039
3040 // We can convert AddrModeT2_i12 to AddrModeT2_i8neg.
3041 const MCInstrDesc &Desc = TII->get(Opcode);
3042 unsigned AddrMode = (Desc.TSFlags & ARMII::AddrModeMask);
3043 switch (AddrMode) {
3045 CodesizeEstimate += 1;
3046 return Imm < 0 && -Imm < ((1 << 8) * 1);
3047 }
3048 return false;
3049}
3050
3051// Given an MI adjust its address BaseReg to use NewBaseReg and address offset
3052// by -Offset. This can either happen in-place or be a replacement as MI is
3053// converted to another instruction type.
3055 int Offset, const TargetInstrInfo *TII,
3056 const TargetRegisterInfo *TRI) {
3057 // Set the Base reg
3058 unsigned BaseOp = getBaseOperandIndex(*MI);
3059 MI->getOperand(BaseOp).setReg(NewBaseReg);
3060 // and constrain the reg class to that required by the instruction.
3061 MachineFunction *MF = MI->getMF();
3062 MachineRegisterInfo &MRI = MF->getRegInfo();
3063 const MCInstrDesc &MCID = TII->get(MI->getOpcode());
3064 const TargetRegisterClass *TRC = TII->getRegClass(MCID, BaseOp);
3065 MRI.constrainRegClass(NewBaseReg, TRC);
3066
3067 int OldOffset = MI->getOperand(BaseOp + 1).getImm();
3068 if (isLegalAddressImm(MI->getOpcode(), OldOffset - Offset, TII))
3069 MI->getOperand(BaseOp + 1).setImm(OldOffset - Offset);
3070 else {
3071 unsigned ConvOpcode;
3072 switch (MI->getOpcode()) {
3073 case ARM::t2LDRHi12:
3074 ConvOpcode = ARM::t2LDRHi8;
3075 break;
3076 case ARM::t2LDRSHi12:
3077 ConvOpcode = ARM::t2LDRSHi8;
3078 break;
3079 case ARM::t2LDRBi12:
3080 ConvOpcode = ARM::t2LDRBi8;
3081 break;
3082 case ARM::t2LDRSBi12:
3083 ConvOpcode = ARM::t2LDRSBi8;
3084 break;
3085 case ARM::t2STRHi12:
3086 ConvOpcode = ARM::t2STRHi8;
3087 break;
3088 case ARM::t2STRBi12:
3089 ConvOpcode = ARM::t2STRBi8;
3090 break;
3091 default:
3092 llvm_unreachable("Unhandled convertible opcode");
3093 }
3094 assert(isLegalAddressImm(ConvOpcode, OldOffset - Offset, TII) &&
3095 "Illegal Address Immediate after convert!");
3096
3097 const MCInstrDesc &MCID = TII->get(ConvOpcode);
3098 BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), MCID)
3099 .add(MI->getOperand(0))
3100 .add(MI->getOperand(1))
3101 .addImm(OldOffset - Offset)
3102 .add(MI->getOperand(3))
3103 .add(MI->getOperand(4))
3104 .cloneMemRefs(*MI);
3105 MI->eraseFromParent();
3106 }
3107}
3108
3110 Register NewReg,
3111 const TargetInstrInfo *TII,
3112 const TargetRegisterInfo *TRI) {
3113 MachineFunction *MF = MI->getMF();
3114 MachineRegisterInfo &MRI = MF->getRegInfo();
3115
3116 unsigned NewOpcode = getPostIndexedLoadStoreOpcode(
3117 MI->getOpcode(), Offset > 0 ? ARM_AM::add : ARM_AM::sub);
3118
3119 const MCInstrDesc &MCID = TII->get(NewOpcode);
3120 // Constrain the def register class
3121 const TargetRegisterClass *TRC = TII->getRegClass(MCID, 0);
3122 MRI.constrainRegClass(NewReg, TRC);
3123 // And do the same for the base operand
3124 TRC = TII->getRegClass(MCID, 2);
3125 MRI.constrainRegClass(MI->getOperand(1).getReg(), TRC);
3126
3127 unsigned AddrMode = (MCID.TSFlags & ARMII::AddrModeMask);
3128 switch (AddrMode) {
3132 // Any MVE load/store
3133 return BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), MCID)
3134 .addReg(NewReg, RegState::Define)
3135 .add(MI->getOperand(0))
3136 .add(MI->getOperand(1))
3137 .addImm(Offset)
3138 .add(MI->getOperand(3))
3139 .add(MI->getOperand(4))
3140 .add(MI->getOperand(5))
3141 .cloneMemRefs(*MI);
3143 if (MI->mayLoad()) {
3144 return BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), MCID)
3145 .add(MI->getOperand(0))
3146 .addReg(NewReg, RegState::Define)
3147 .add(MI->getOperand(1))
3148 .addImm(Offset)
3149 .add(MI->getOperand(3))
3150 .add(MI->getOperand(4))
3151 .cloneMemRefs(*MI);
3152 } else {
3153 return BuildMI(*MI->getParent(), MI, MI->getDebugLoc(), MCID)
3154 .addReg(NewReg, RegState::Define)
3155 .add(MI->getOperand(0))
3156 .add(MI->getOperand(1))
3157 .addImm(Offset)
3158 .add(MI->getOperand(3))
3159 .add(MI->getOperand(4))
3160 .cloneMemRefs(*MI);
3161 }
3162 default:
3163 llvm_unreachable("Unhandled createPostIncLoadStore");
3164 }
3165}
3166
3167// Given a Base Register, optimise the load/store uses to attempt to create more
3168// post-inc accesses and less register moves. We do this by taking zero offset
3169// loads/stores with an add, and convert them to a postinc load/store of the
3170// same type. Any subsequent accesses will be adjusted to use and account for
3171// the post-inc value.
3172// For example:
3173// LDR #0 LDR_POSTINC #16
3174// LDR #4 LDR #-12
3175// LDR #8 LDR #-8
3176// LDR #12 LDR #-4
3177// ADD #16
3178//
3179// At the same time if we do not find an increment but do find an existing
3180// pre/post inc instruction, we can still adjust the offsets of subsequent
3181// instructions to save the register move that would otherwise be needed for the
3182// in-place increment.
3183bool ARMPreAllocLoadStoreOpt::DistributeIncrements(Register Base) {
3184 // We are looking for:
3185 // One zero offset load/store that can become postinc
3186 MachineInstr *BaseAccess = nullptr;
3187 MachineInstr *PrePostInc = nullptr;
3188 // An increment that can be folded in
3189 MachineInstr *Increment = nullptr;
3190 // Other accesses after BaseAccess that will need to be updated to use the
3191 // postinc value.
3192 SmallPtrSet<MachineInstr *, 8> OtherAccesses;
3193 for (auto &Use : MRI->use_nodbg_instructions(Base)) {
3194 if (!Increment && getAddSubImmediate(Use) != 0) {
3195 Increment = &Use;
3196 continue;
3197 }
3198
3199 int BaseOp = getBaseOperandIndex(Use);
3200 if (BaseOp == -1)
3201 return false;
3202
3203 if (!Use.getOperand(BaseOp).isReg() ||
3204 Use.getOperand(BaseOp).getReg() != Base)
3205 return false;
3206 if (isPreIndex(Use) || isPostIndex(Use))
3207 PrePostInc = &Use;
3208 else if (Use.getOperand(BaseOp + 1).getImm() == 0)
3209 BaseAccess = &Use;
3210 else
3211 OtherAccesses.insert(&Use);
3212 }
3213
3214 int IncrementOffset;
3215 Register NewBaseReg;
3216 if (BaseAccess && Increment) {
3217 if (PrePostInc || BaseAccess->getParent() != Increment->getParent())
3218 return false;
3219 Register PredReg;
3220 if (Increment->definesRegister(ARM::CPSR, /*TRI=*/nullptr) ||
3222 return false;
3223
3224 LLVM_DEBUG(dbgs() << "\nAttempting to distribute increments on VirtualReg "
3225 << Base.virtRegIndex() << "\n");
3226
3227 // Make sure that Increment has no uses before BaseAccess that are not PHI
3228 // uses.
3229 for (MachineInstr &Use :
3230 MRI->use_nodbg_instructions(Increment->getOperand(0).getReg())) {
3231 if (&Use == BaseAccess || (Use.getOpcode() != TargetOpcode::PHI &&
3232 !DT->dominates(BaseAccess, &Use))) {
3233 LLVM_DEBUG(dbgs() << " BaseAccess doesn't dominate use of increment\n");
3234 return false;
3235 }
3236 }
3237
3238 // Make sure that Increment can be folded into Base
3239 IncrementOffset = getAddSubImmediate(*Increment);
3240 unsigned NewPostIncOpcode = getPostIndexedLoadStoreOpcode(
3241 BaseAccess->getOpcode(), IncrementOffset > 0 ? ARM_AM::add : ARM_AM::sub);
3242 if (!isLegalAddressImm(NewPostIncOpcode, IncrementOffset, TII)) {
3243 LLVM_DEBUG(dbgs() << " Illegal addressing mode immediate on postinc\n");
3244 return false;
3245 }
3246 }
3247 else if (PrePostInc) {
3248 // If we already have a pre/post index load/store then set BaseAccess,
3249 // IncrementOffset and NewBaseReg to the values it already produces,
3250 // allowing us to update and subsequent uses of BaseOp reg with the
3251 // incremented value.
3252 if (Increment)
3253 return false;
3254
3255 LLVM_DEBUG(dbgs() << "\nAttempting to distribute increments on already "
3256 << "indexed VirtualReg " << Base.virtRegIndex() << "\n");
3257 int BaseOp = getBaseOperandIndex(*PrePostInc);
3258 IncrementOffset = PrePostInc->getOperand(BaseOp+1).getImm();
3259 BaseAccess = PrePostInc;
3260 NewBaseReg = PrePostInc->getOperand(0).getReg();
3261 }
3262 else
3263 return false;
3264
3265 // And make sure that the negative value of increment can be added to all
3266 // other offsets after the BaseAccess. We rely on either
3267 // dominates(BaseAccess, OtherAccess) or dominates(OtherAccess, BaseAccess)
3268 // to keep things simple.
3269 // This also adds a simple codesize metric, to detect if an instruction (like
3270 // t2LDRBi12) which can often be shrunk to a thumb1 instruction (tLDRBi)
3271 // cannot because it is converted to something else (t2LDRBi8). We start this
3272 // at -1 for the gain from removing the increment.
3273 SmallPtrSet<MachineInstr *, 4> SuccessorAccesses;
3274 int CodesizeEstimate = -1;
3275 for (auto *Use : OtherAccesses) {
3276 if (DT->dominates(BaseAccess, Use)) {
3277 SuccessorAccesses.insert(Use);
3278 unsigned BaseOp = getBaseOperandIndex(*Use);
3279 if (!isLegalOrConvertibleAddressImm(Use->getOpcode(),
3280 Use->getOperand(BaseOp + 1).getImm() -
3281 IncrementOffset,
3282 TII, CodesizeEstimate)) {
3283 LLVM_DEBUG(dbgs() << " Illegal addressing mode immediate on use\n");
3284 return false;
3285 }
3286 } else if (!DT->dominates(Use, BaseAccess)) {
3287 LLVM_DEBUG(
3288 dbgs() << " Unknown dominance relation between Base and Use\n");
3289 return false;
3290 }
3291 }
3292 if (STI->hasMinSize() && CodesizeEstimate > 0) {
3293 LLVM_DEBUG(dbgs() << " Expected to grow instructions under minsize\n");
3294 return false;
3295 }
3296
3297 if (!PrePostInc) {
3298 // Replace BaseAccess with a post inc
3299 LLVM_DEBUG(dbgs() << "Changing: "; BaseAccess->dump());
3300 LLVM_DEBUG(dbgs() << " And : "; Increment->dump());
3301 NewBaseReg = Increment->getOperand(0).getReg();
3302 MachineInstr *BaseAccessPost =
3303 createPostIncLoadStore(BaseAccess, IncrementOffset, NewBaseReg, TII, TRI);
3304 BaseAccess->eraseFromParent();
3305 Increment->eraseFromParent();
3306 (void)BaseAccessPost;
3307 LLVM_DEBUG(dbgs() << " To : "; BaseAccessPost->dump());
3308 }
3309
3310 for (auto *Use : SuccessorAccesses) {
3311 LLVM_DEBUG(dbgs() << "Changing: "; Use->dump());
3312 AdjustBaseAndOffset(Use, NewBaseReg, IncrementOffset, TII, TRI);
3313 LLVM_DEBUG(dbgs() << " To : "; Use->dump());
3314 }
3315
3316 // Remove the kill flag from all uses of NewBaseReg, in case any old uses
3317 // remain.
3318 for (MachineOperand &Op : MRI->use_nodbg_operands(NewBaseReg))
3319 Op.setIsKill(false);
3320 return true;
3321}
3322
3323bool ARMPreAllocLoadStoreOpt::DistributeIncrements() {
3324 bool Changed = false;
3325 SmallSetVector<Register, 4> Visited;
3326 for (auto &MBB : *MF) {
3327 for (auto &MI : MBB) {
3328 int BaseOp = getBaseOperandIndex(MI);
3329 if (BaseOp == -1 || !MI.getOperand(BaseOp).isReg())
3330 continue;
3331
3332 Register Base = MI.getOperand(BaseOp).getReg();
3333 if (!Base.isVirtual())
3334 continue;
3335
3336 Visited.insert(Base);
3337 }
3338 }
3339
3340 for (auto Base : Visited)
3341 Changed |= DistributeIncrements(Base);
3342
3343 return Changed;
3344}
3345
3346/// Returns an instance of the load / store optimization pass.
3348 if (PreAlloc)
3349 return new ARMPreAllocLoadStoreOptLegacy();
3350 return new ARMLoadStoreOptLegacy();
3351}
3352
3356 ARMLoadStoreOpt Impl;
3357 bool Changed = Impl.runOnMachineFunction(
3359 if (!Changed)
3360 return PreservedAnalyses::all();
3364 return PA;
3365}
3366
3370 ARMPreAllocLoadStoreOpt Impl;
3371 AliasAnalysis *AA =
3373 .getManager()
3374 .getResult<AAManager>(MF.getFunction());
3376 bool Changed = Impl.runOnMachineFunction(MF, AA, DT);
3377 if (!Changed)
3378 return PreservedAnalyses::all();
3381 return PA;
3382}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
aarch64 promote const
unsigned Imm
static bool isLoadSingle(unsigned Opc)
static int getMemoryOpOffset(const MachineInstr &MI)
static unsigned getPostIndexedLoadStoreOpcode(unsigned Opc, ARM_AM::AddrOpc Mode)
static bool IsSafeAndProfitableToMove(bool isLd, unsigned Base, MachineBasicBlock::iterator I, MachineBasicBlock::iterator E, SmallPtrSetImpl< MachineInstr * > &MemOps, SmallSet< unsigned, 4 > &MemRegs, const TargetRegisterInfo *TRI, AliasAnalysis *AA)
static bool ContainsReg(ArrayRef< std::pair< unsigned, bool > > Regs, unsigned Reg)
static bool isPreIndex(MachineInstr &MI)
static void forEachDbgRegOperand(MachineInstr *MI, std::function< void(MachineOperand &)> Fn)
static bool isPostIndex(MachineInstr &MI)
static int getLoadStoreMultipleOpcode(unsigned Opcode, ARM_AM::AMSubMode Mode)
static unsigned getLSMultipleTransferSize(const MachineInstr *MI)
static bool isLegalOrConvertibleAddressImm(unsigned Opcode, int Imm, const TargetInstrInfo *TII, int &CodesizeEstimate)
static ARM_AM::AMSubMode getLoadStoreMultipleSubMode(unsigned Opcode)
static bool isT1i32Load(unsigned Opc)
static void AdjustBaseAndOffset(MachineInstr *MI, Register NewBaseReg, int Offset, const TargetInstrInfo *TII, const TargetRegisterInfo *TRI)
arm ldst static false bool definesCPSR(const MachineInstr &MI)
static unsigned getPreIndexedLoadStoreOpcode(unsigned Opc, ARM_AM::AddrOpc Mode)
static MachineInstr * createPostIncLoadStore(MachineInstr *MI, int Offset, Register NewReg, const TargetInstrInfo *TII, const TargetRegisterInfo *TRI)
static bool isi32Store(unsigned Opc)
static MachineBasicBlock::iterator findIncDecAfter(MachineBasicBlock::iterator MBBI, Register Reg, ARMCC::CondCodes Pred, Register PredReg, int &Offset, const TargetRegisterInfo *TRI)
Searches for a increment or decrement of Reg after MBBI.
static MachineBasicBlock::iterator findIncDecBefore(MachineBasicBlock::iterator MBBI, Register Reg, ARMCC::CondCodes Pred, Register PredReg, int &Offset)
Searches for an increment or decrement of Reg before MBBI.
static const MachineOperand & getLoadStoreBaseOp(const MachineInstr &MI)
static void updateRegisterMapForDbgValueListAfterMove(SmallDenseMap< Register, SmallVector< MachineInstr * >, 8 > &RegisterMap, MachineInstr *DbgValueListInstr, MachineInstr *InstrToReplace)
arm prera ldst static false cl::opt< unsigned > InstReorderLimit("arm-prera-ldst-opt-reorder-limit", cl::init(8), cl::Hidden)
static void InsertLDR_STR(MachineBasicBlock &MBB, MachineBasicBlock::iterator &MBBI, int Offset, bool isDef, unsigned NewOpc, unsigned Reg, bool RegDeadKill, bool RegUndef, unsigned BaseReg, bool BaseKill, bool BaseUndef, ARMCC::CondCodes Pred, unsigned PredReg, const TargetInstrInfo *TII, MachineInstr *MI)
static int isIncrementOrDecrement(const MachineInstr &MI, Register Reg, ARMCC::CondCodes Pred, Register PredReg)
Check if the given instruction increments or decrements a register and return the amount it is increm...
static bool isT2i32Store(unsigned Opc)
static bool mayCombineMisaligned(const TargetSubtargetInfo &STI, const MachineInstr &MI)
Return true for loads/stores that can be combined to a double/multi operation without increasing the ...
static int getBaseOperandIndex(MachineInstr &MI)
static bool isT2i32Load(unsigned Opc)
static bool isi32Load(unsigned Opc)
static unsigned getImmScale(unsigned Opc)
static bool isT1i32Store(unsigned Opc)
#define ARM_PREALLOC_LOAD_STORE_OPT_NAME
#define ARM_LOAD_STORE_OPT_NAME
static unsigned getUpdatingLSMultipleOpcode(unsigned Opc, ARM_AM::AMSubMode Mode)
static bool isMemoryOp(const MachineInstr &MI)
Returns true if instruction is a memory operation that this pass is capable of operating on.
static const MachineOperand & getLoadStoreRegOp(const MachineInstr &MI)
static bool isValidLSDoubleOffset(int Offset)
static DebugVariable createDebugVariableFromMachineInstr(MachineInstr *MI)
static cl::opt< bool > AssumeMisalignedLoadStores("arm-assume-misaligned-load-store", cl::Hidden, cl::init(false), cl::desc("Be more conservative in ARM load/store opt"))
This switch disables formation of double/multi instructions that could potentially lead to (new) alig...
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
This file defines the BumpPtrAllocator interface.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
This file defines the DenseMap class.
This file defines the DenseSet and SmallDenseSet classes.
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
A set of register units.
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
if(PassOpts->AAPipeline)
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
Basic Register Allocator
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
This file contains some templates that are useful if you are working with the STL at all.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
This file defines the SmallSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
This file describes how to lower LLVM code to machine code.
Value * RHS
Value * LHS
A manager for alias analyses.
A wrapper pass to provide the legacy pass manager access to a suitably prepared AAResults object.
static void updateLRRestored(MachineFunction &MF)
Update the IsRestored flag on LR if it is spilled, based on the return instructions.
ARMFunctionInfo - This class is derived from MachineFunctionInfo and contains private ARM-specific in...
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
const ARMBaseInstrInfo * getInstrInfo() const override
bool isThumb2() const
const ARMTargetLowering * getTargetLowering() const override
const ARMBaseRegisterInfo * getRegisterInfo() const override
bool hasMinSize() const
bool isCortexM3() const
Align getDualLoadStoreAlignment() const
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
Identifies a unique instance of a variable.
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:782
iterator end()
Definition DenseMap.h:702
bool erase(const KeyT &Val)
Definition DenseMap.h:946
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
A set of register units used to track register liveness.
bool available(MCRegister Reg) const
Returns true if no part of physical register Reg is live.
void init(const TargetRegisterInfo &TRI)
Initialize and clear the set.
void addReg(MCRegister Reg)
Adds register units covered by physical register Reg.
LLVM_ABI void stepBackward(const MachineInstr &MI)
Updates liveness when stepping backwards over the instruction MI.
LLVM_ABI void addLiveOuts(const MachineBasicBlock &MBB)
Adds registers living out of block MBB.
Describe properties that are true of each instruction in the target description file.
MachineInstrBundleIterator< const MachineInstr > const_iterator
LLVM_ABI instr_iterator insert(instr_iterator I, MachineInstr *M)
Insert MI into the instruction list before I, possibly inside a bundle.
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI iterator getLastNonDebugInstr(bool SkipPseudoOp=true)
Returns an iterator to the last non-debug instruction in the basic block, or end().
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LLVM_ABI instr_iterator erase(instr_iterator I)
Remove an instruction from the instruction list and delete it.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
@ LQR_Dead
Register is known to be fully dead.
Analysis pass which computes a MachineDominatorTree.
Analysis pass which computes a MachineDominatorTree.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
bool dominates(const MachineInstr *A, const MachineInstr *B) const
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const MachineInstrBuilder & cloneMergedMemRefs(ArrayRef< const MachineInstr * > OtherMIs) const
const MachineInstrBuilder & setMemRefs(ArrayRef< MachineMemOperand * > MMOs) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & copyImplicitOps(const MachineInstr &OtherMI) const
Copy all the implicit operands from OtherMI onto this one.
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI void copyImplicitOps(MachineFunction &MF, const MachineInstr &MI)
Copy implicit register operands from specified instruction to this instruction.
bool killsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr kills the specified register.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
bool hasOneMemOperand() const
Return true if this instruction has exactly one MachineMemOperand.
mmo_iterator memoperands_begin() const
Access to memory operands of the instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void dump() const
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
A description of a memory reference used in the backend.
bool isAtomic() const
Returns true if this operation has an atomic ordering requirement of unordered or higher,...
LLVM_ABI Align getAlign() const
Return the minimum known alignment in bytes of the actual memory reference.
MachineOperand class - Representation of each machine instruction operand.
void setImm(int64_t immVal)
int64_t getImm() const
bool readsReg() const
readsReg - Returns true if this operand reads the previous value of its register.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
void setIsKill(bool Val=true)
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Definition Analysis.h:151
PreservedAnalyses & preserve()
Mark an analysis as preserved.
Definition Analysis.h:132
ArrayRef< MCPhysReg > getOrder(const TargetRegisterClass *RC) const
getOrder - Returns the preferred allocation order for RC.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
Definition SmallSet.h:134
size_type count(const T &V) const
count - Return 1 if the element is in the set, 0 otherwise.
Definition SmallSet.h:176
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
size_type size() const
Definition SmallSet.h:171
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A BumpPtrAllocator that allows only elements of a specific type to be allocated.
Definition Allocator.h:397
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Align getTransientStackAlign() const
getTransientStackAlignment - This method returns the number of bytes to which the stack pointer must ...
TargetInstrInfo - Interface to description of machine instruction set.
virtual bool isLegalAddImmediate(int64_t) const
Return true if the specified immediate is legal add immediate, that is the target has add instruction...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
TargetSubtargetInfo - Generic base class for all target subtargets.
virtual const TargetFrameLowering * getFrameLowering() const
LLVM Value Representation.
Definition Value.h:75
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
size_type count(const_arg_type_t< ValueT > V) const
Return 1 if the specified key is in the set, 0 otherwise.
Definition DenseSet.h:187
Changed
This provides a very simple, boring adaptor for a begin and end iterator into a range type.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
Abstract Attribute helper functions.
Definition Attributor.h:165
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
unsigned char getAM3Offset(unsigned AM3Opc)
unsigned getAM2Opc(AddrOpc Opc, unsigned Imm12, ShiftOpc SO, unsigned IdxMode=0)
AddrOpc getAM5Op(unsigned AM5Opc)
unsigned getAM3Opc(AddrOpc Opc, unsigned char Offset, unsigned IdxMode=0)
getAM3Opc - This function encodes the addrmode3 opc field.
unsigned char getAM5Offset(unsigned AM5Opc)
AddrOpc getAM3Op(unsigned AM3Opc)
@ ARM
Windows AXP64.
Definition MCAsmInfo.h:50
@ CE
Windows NT (Windows on ARM)
Definition MCAsmInfo.h:51
This namespace contains all of the command line option processing machinery.
Definition MCSchedule.h:35
initializer< Ty > init(const Ty &Val)
NodeAddr< InstrNode * > Instr
Definition RDFGraph.h:389
NodeAddr< UseNode * > Use
Definition RDFGraph.h:385
BBIterator iterator
Definition BasicBlock.h:87
BaseReg
Stack frame base register. Bit 0 of FREInfo.Info.
Definition SFrame.h:77
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Offset
Definition DWP.cpp:577
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
@ Define
Register definition.
constexpr RegState getKillRegState(bool B)
static bool isARMLowRegister(MCRegister Reg)
isARMLowRegister - Returns true if the register is a low register (r0-r7).
APFloat abs(APFloat X)
Returns the absolute value of the argument.
Definition APFloat.h:1721
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
bool isLegalAddressImm(unsigned Opcode, int Imm, const TargetInstrInfo *TII)
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
constexpr RegState getDeadRegState(bool B)
static std::array< MachineOperand, 2 > predOps(ARMCC::CondCodes Pred, unsigned PredReg=0)
Get the operands corresponding to the given Pred value.
Op::Description Desc
unsigned M1(unsigned Val)
Definition VE.h:377
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr RegState getDefRegState(bool B)
FunctionPass * createARMLoadStoreOptLegacyPass(bool PreAlloc=false)
Returns an instance of the load / store optimization pass.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
void replace(R &&Range, const T &OldValue, const T &NewValue)
Provide wrappers to std::replace which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1926
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
ARMCC::CondCodes getInstrPredicate(const MachineInstr &MI, Register &PredReg)
getInstrPredicate - If instruction is predicated, returns its predicate condition,...
DWARFExpression::Operation Op
unsigned M0(unsigned Val)
Definition VE.h:376
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI void eraseInstr(MachineInstr &MI, MachineRegisterInfo &MRI, LostDebugLocObserver *LocObserver=nullptr)
Definition Utils.cpp:1671
static MachineOperand t1CondCodeOp(bool isDead=false)
Get the operand corresponding to the conditional code result for Thumb1.
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
static MachineOperand condCodeOp(unsigned CCReg=0)
Get the operand corresponding to the conditional code result.
@ Increment
Incrementally increasing token ID.
Definition AllocToken.h:26
int getAddSubImmediate(MachineInstr &MI)
AAResults AliasAnalysis
Temporary typedef for legacy code that uses a generic AliasAnalysis pointer or reference.
constexpr RegState getUndefRegState(bool B)
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39