LLVM 24.0.0git
AArch64InstructionSelector.cpp
Go to the documentation of this file.
1//===- AArch64InstructionSelector.cpp ----------------------------*- C++ -*-==//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements the targeting of the InstructionSelector class for
10/// AArch64.
11/// \todo This should be generated by TableGen.
12//===----------------------------------------------------------------------===//
13
15#include "AArch64InstrInfo.h"
18#include "AArch64RegisterInfo.h"
19#include "AArch64Subtarget.h"
42#include "llvm/IR/Constants.h"
45#include "llvm/IR/IntrinsicsAArch64.h"
46#include "llvm/IR/Type.h"
47#include "llvm/Pass.h"
48#include "llvm/Support/Debug.h"
50#include <optional>
51
52#define DEBUG_TYPE "aarch64-isel"
53
54using namespace llvm;
55using namespace MIPatternMatch;
56using namespace AArch64GISelUtils;
57
58namespace llvm {
61}
62
63namespace {
64
65#define GET_GLOBALISEL_PREDICATE_BITSET
66#include "AArch64GenGlobalISel.inc"
67#undef GET_GLOBALISEL_PREDICATE_BITSET
68
69
70class AArch64InstructionSelector : public InstructionSelector {
71public:
72 AArch64InstructionSelector(const AArch64TargetMachine &TM,
73 const AArch64Subtarget &STI,
74 const AArch64RegisterBankInfo &RBI);
75
76 bool select(MachineInstr &I) override;
77 static const char *getName() { return DEBUG_TYPE; }
78
79 void setupMF(MachineFunction &MF, GISelValueTracking *VT,
80 CodeGenCoverage *CoverageInfo, ProfileSummaryInfo *PSI,
81 BlockFrequencyInfo *BFI) override {
82 InstructionSelector::setupMF(MF, VT, CoverageInfo, PSI, BFI);
83 MIB.setMF(MF);
84
85 // hasFnAttribute() is expensive to call on every BRCOND selection, so
86 // cache it here for each run of the selector.
87 ProduceNonFlagSettingCondBr =
88 !MF.getFunction().hasFnAttribute(Attribute::SpeculativeLoadHardening);
89 MFReturnAddr = Register();
90
91 processPHIs(MF);
92 }
93
94private:
95 /// tblgen-erated 'select' implementation, used as the initial selector for
96 /// the patterns that don't require complex C++.
97 bool selectImpl(MachineInstr &I, CodeGenCoverage &CoverageInfo) const;
98
99 // A lowering phase that runs before any selection attempts.
100 // Returns true if the instruction was modified.
101 bool preISelLower(MachineInstr &I);
102
103 // An early selection function that runs before the selectImpl() call.
104 bool earlySelect(MachineInstr &I);
105
106 /// Save state that is shared between select calls, call select on \p I and
107 /// then restore the saved state. This can be used to recursively call select
108 /// within a select call.
109 bool selectAndRestoreState(MachineInstr &I);
110
111 // Do some preprocessing of G_PHIs before we begin selection.
112 void processPHIs(MachineFunction &MF);
113
114 bool earlySelectSHL(MachineInstr &I, MachineRegisterInfo &MRI);
115
116 /// Eliminate same-sized cross-bank copies into stores before selectImpl().
117 bool contractCrossBankCopyIntoStore(MachineInstr &I,
119
120 bool convertPtrAddToAdd(MachineInstr &I, MachineRegisterInfo &MRI);
121
122 bool selectVaStartAAPCS(MachineInstr &I, MachineFunction &MF,
123 MachineRegisterInfo &MRI) const;
124 bool selectVaStartDarwin(MachineInstr &I, MachineFunction &MF,
125 MachineRegisterInfo &MRI) const;
126
127 ///@{
128 /// Helper functions for selectCompareBranch.
129 bool selectCompareBranchFedByFCmp(MachineInstr &I, MachineInstr &FCmp,
130 MachineIRBuilder &MIB) const;
131 bool selectCompareBranchFedByICmp(MachineInstr &I, MachineInstr &ICmp,
132 MachineIRBuilder &MIB) const;
133 bool tryOptCompareBranchFedByICmp(MachineInstr &I, MachineInstr &ICmp,
134 MachineIRBuilder &MIB) const;
135 bool tryOptAndIntoCompareBranch(MachineInstr &AndInst, bool Invert,
136 MachineBasicBlock *DstMBB,
137 MachineIRBuilder &MIB) const;
138 ///@}
139
140 bool selectCompareBranch(MachineInstr &I, MachineFunction &MF,
142
143 bool selectVectorAshrLshr(MachineInstr &I, MachineRegisterInfo &MRI);
144 bool selectVectorSHL(MachineInstr &I, MachineRegisterInfo &MRI);
145
146 // Helper to generate an equivalent of scalar_to_vector into a new register,
147 // returned via 'Dst'.
148 MachineInstr *emitScalarToVector(unsigned EltSize,
149 const TargetRegisterClass *DstRC,
150 Register Scalar,
151 MachineIRBuilder &MIRBuilder) const;
152 /// Helper to narrow vector that was widened by emitScalarToVector.
153 /// Copy lowest part of 128-bit or 64-bit vector to 64-bit or 32-bit
154 /// vector, correspondingly.
155 MachineInstr *emitNarrowVector(Register DstReg, Register SrcReg,
156 MachineIRBuilder &MIRBuilder,
157 MachineRegisterInfo &MRI) const;
158
159 /// Emit a lane insert into \p DstReg, or a new vector register if
160 /// std::nullopt is provided.
161 ///
162 /// The lane inserted into is defined by \p LaneIdx. The vector source
163 /// register is given by \p SrcReg. The register containing the element is
164 /// given by \p EltReg.
165 MachineInstr *emitLaneInsert(std::optional<Register> DstReg, Register SrcReg,
166 Register EltReg, unsigned LaneIdx,
167 const RegisterBank &RB,
168 MachineIRBuilder &MIRBuilder) const;
169
170 /// Emit a sequence of instructions representing a constant \p CV for a
171 /// vector register \p Dst. (E.g. a MOV, or a load from a constant pool.)
172 ///
173 /// \returns the last instruction in the sequence on success, and nullptr
174 /// otherwise.
175 MachineInstr *emitConstantVector(Register Dst, Constant *CV,
176 MachineIRBuilder &MIRBuilder,
178
179 MachineInstr *tryAdvSIMDModImm8(Register Dst, unsigned DstSize, APInt Bits,
180 MachineIRBuilder &MIRBuilder);
181
182 MachineInstr *tryAdvSIMDModImm16(Register Dst, unsigned DstSize, APInt Bits,
183 MachineIRBuilder &MIRBuilder, bool Inv);
184
185 MachineInstr *tryAdvSIMDModImm32(Register Dst, unsigned DstSize, APInt Bits,
186 MachineIRBuilder &MIRBuilder, bool Inv);
187 MachineInstr *tryAdvSIMDModImm64(Register Dst, unsigned DstSize, APInt Bits,
188 MachineIRBuilder &MIRBuilder);
189 MachineInstr *tryAdvSIMDModImm321s(Register Dst, unsigned DstSize, APInt Bits,
190 MachineIRBuilder &MIRBuilder, bool Inv);
191 MachineInstr *tryAdvSIMDModImmFP(Register Dst, unsigned DstSize, APInt Bits,
192 MachineIRBuilder &MIRBuilder);
193
194 bool tryOptConstantBuildVec(MachineInstr &MI, LLT DstTy,
196 /// \returns true if a G_BUILD_VECTOR instruction \p MI can be selected as a
197 /// SUBREG_TO_REG.
198 bool tryOptBuildVecToSubregToReg(MachineInstr &MI, MachineRegisterInfo &MRI);
199 bool selectBuildVector(MachineInstr &I, MachineRegisterInfo &MRI);
202
203 bool selectShuffleVector(MachineInstr &I, MachineRegisterInfo &MRI);
204 bool selectExtractElt(MachineInstr &I, MachineRegisterInfo &MRI);
205 bool selectConcatVectors(MachineInstr &I, MachineRegisterInfo &MRI);
206 bool selectSplitVectorUnmerge(MachineInstr &I, MachineRegisterInfo &MRI);
207
208 /// Helper function to select vector load intrinsics like
209 /// @llvm.aarch64.neon.ld2.*, @llvm.aarch64.neon.ld4.*, etc.
210 /// \p Opc is the opcode that the selected instruction should use.
211 /// \p NumVecs is the number of vector destinations for the instruction.
212 /// \p I is the original G_INTRINSIC_W_SIDE_EFFECTS instruction.
213 bool selectVectorLoadIntrinsic(unsigned Opc, unsigned NumVecs,
214 MachineInstr &I);
215 bool selectVectorLoadLaneIntrinsic(unsigned Opc, unsigned NumVecs,
216 MachineInstr &I);
217 void selectVectorStoreIntrinsic(MachineInstr &I, unsigned NumVecs,
218 unsigned Opc);
219 bool selectVectorStoreLaneIntrinsic(MachineInstr &I, unsigned NumVecs,
220 unsigned Opc);
221 bool selectIntrinsicWithSideEffects(MachineInstr &I,
223 bool selectIntrinsic(MachineInstr &I, MachineRegisterInfo &MRI);
224 bool selectJumpTable(MachineInstr &I, MachineRegisterInfo &MRI);
225 bool selectBrJT(MachineInstr &I, MachineRegisterInfo &MRI);
226 bool selectTLSGlobalValueELF(MachineInstr &I, MachineRegisterInfo &MRI);
227 bool selectTLSLocalExecELF(const GlobalValue *GV, MachineInstr &I,
229 bool selectTLSGlobalValueMachO(MachineInstr &I, MachineRegisterInfo &MRI);
230 bool selectTLSGlobalValue(MachineInstr &I, MachineRegisterInfo &MRI);
231 bool selectPtrAuthGlobalValue(MachineInstr &I,
232 MachineRegisterInfo &MRI) const;
233 bool selectMOPS(MachineInstr &I, MachineRegisterInfo &MRI);
234 bool selectUSMovFromExtend(MachineInstr &I, MachineRegisterInfo &MRI);
235 void SelectTable(MachineInstr &I, MachineRegisterInfo &MRI, unsigned NumVecs,
236 unsigned Opc1, unsigned Opc2, bool isExt);
237
238 bool selectIndexedExtLoad(MachineInstr &I, MachineRegisterInfo &MRI);
239 bool selectIndexedLoad(MachineInstr &I, MachineRegisterInfo &MRI);
240 bool selectIndexedStore(GIndexedStore &I, MachineRegisterInfo &MRI);
241
242 unsigned emitConstantPoolEntry(const Constant *CPVal,
243 MachineFunction &MF) const;
245 MachineIRBuilder &MIRBuilder) const;
246
247 // Emit a vector concat operation.
248 MachineInstr *emitVectorConcat(std::optional<Register> Dst, Register Op1,
249 Register Op2,
250 MachineIRBuilder &MIRBuilder) const;
251
252 // Emit an integer compare between LHS and RHS, which checks for Predicate.
253 MachineInstr *emitIntegerCompare(MachineOperand &LHS, MachineOperand &RHS,
255 MachineIRBuilder &MIRBuilder) const;
256
257 /// Emit a floating point comparison between \p LHS and \p RHS.
258 /// \p Pred if given is the intended predicate to use.
260 emitFPCompare(Register LHS, Register RHS, MachineIRBuilder &MIRBuilder,
261 std::optional<CmpInst::Predicate> = std::nullopt) const;
262
264 emitInstr(unsigned Opcode, std::initializer_list<llvm::DstOp> DstOps,
265 std::initializer_list<llvm::SrcOp> SrcOps,
266 MachineIRBuilder &MIRBuilder,
267 const ComplexRendererFns &RenderFns = std::nullopt) const;
268 /// Helper function to emit an add or sub instruction.
269 ///
270 /// \p AddrModeAndSizeToOpcode must contain each of the opcode variants above
271 /// in a specific order.
272 ///
273 /// Below is an example of the expected input to \p AddrModeAndSizeToOpcode.
274 ///
275 /// \code
276 /// const std::array<std::array<unsigned, 2>, 4> Table {
277 /// {{AArch64::ADDXri, AArch64::ADDWri},
278 /// {AArch64::ADDXrs, AArch64::ADDWrs},
279 /// {AArch64::ADDXrr, AArch64::ADDWrr},
280 /// {AArch64::SUBXri, AArch64::SUBWri},
281 /// {AArch64::ADDXrx, AArch64::ADDWrx}}};
282 /// \endcode
283 ///
284 /// Each row in the table corresponds to a different addressing mode. Each
285 /// column corresponds to a different register size.
286 ///
287 /// \attention Rows must be structured as follows:
288 /// - Row 0: The ri opcode variants
289 /// - Row 1: The rs opcode variants
290 /// - Row 2: The rr opcode variants
291 /// - Row 3: The ri opcode variants for negative immediates
292 /// - Row 4: The rx opcode variants
293 ///
294 /// \attention Columns must be structured as follows:
295 /// - Column 0: The 64-bit opcode variants
296 /// - Column 1: The 32-bit opcode variants
297 ///
298 /// \p Dst is the destination register of the binop to emit.
299 /// \p LHS is the left-hand operand of the binop to emit.
300 /// \p RHS is the right-hand operand of the binop to emit.
301 MachineInstr *emitAddSub(
302 const std::array<std::array<unsigned, 2>, 5> &AddrModeAndSizeToOpcode,
304 MachineIRBuilder &MIRBuilder) const;
305 MachineInstr *emitADD(Register DefReg, MachineOperand &LHS,
307 MachineIRBuilder &MIRBuilder) const;
309 MachineIRBuilder &MIRBuilder) const;
311 MachineIRBuilder &MIRBuilder) const;
313 MachineIRBuilder &MIRBuilder) const;
315 MachineIRBuilder &MIRBuilder) const;
317 MachineIRBuilder &MIRBuilder) const;
319 MachineIRBuilder &MIRBuilder) const;
321 MachineIRBuilder &MIRBuilder) const;
324 MachineIRBuilder &MIRBuilder) const;
325 MachineInstr *emitExtractVectorElt(std::optional<Register> DstReg,
326 const RegisterBank &DstRB, LLT ScalarTy,
327 Register VecReg, unsigned LaneIdx,
328 MachineIRBuilder &MIRBuilder) const;
329 MachineInstr *emitCSINC(Register Dst, Register Src1, Register Src2,
331 MachineIRBuilder &MIRBuilder) const;
332 /// Emit a CSet for a FP compare.
333 ///
334 /// \p Dst is expected to be a 32-bit scalar register.
335 MachineInstr *emitCSetForFCmp(Register Dst, CmpInst::Predicate Pred,
336 MachineIRBuilder &MIRBuilder) const;
337
338 /// Emit an instruction that sets NZCV to the carry-in expected by \p I.
339 /// Might elide the instruction if the previous instruction already sets NZCV
340 /// correctly.
341 MachineInstr *emitCarryIn(MachineInstr &I, Register CarryReg);
342
343 /// Emit the overflow op for \p Opcode.
344 ///
345 /// \p Opcode is expected to be an overflow op's opcode, e.g. G_UADDO,
346 /// G_USUBO, etc.
347 std::pair<MachineInstr *, AArch64CC::CondCode>
348 emitOverflowOp(unsigned Opcode, Register Dst, MachineOperand &LHS,
349 MachineOperand &RHS, MachineIRBuilder &MIRBuilder) const;
350
351 bool selectOverflowOp(MachineInstr &I, MachineRegisterInfo &MRI);
352
353 /// Emit expression as a conjunction (a series of CCMP/CFCMP ops).
354 /// In some cases this is even possible with OR operations in the expression.
356 MachineIRBuilder &MIB) const;
361 MachineIRBuilder &MIB) const;
363 bool Negate, Register CCOp,
365 MachineIRBuilder &MIB) const;
366
367 /// Emit a TB(N)Z instruction which tests \p Bit in \p TestReg.
368 /// \p IsNegative is true if the test should be "not zero".
369 /// This will also optimize the test bit instruction when possible.
370 MachineInstr *emitTestBit(Register TestReg, uint64_t Bit, bool IsNegative,
371 MachineBasicBlock *DstMBB,
372 MachineIRBuilder &MIB) const;
373
374 /// Emit a CB(N)Z instruction which branches to \p DestMBB.
375 MachineInstr *emitCBZ(Register CompareReg, bool IsNegative,
376 MachineBasicBlock *DestMBB,
377 MachineIRBuilder &MIB) const;
378
379 // Equivalent to the i32shift_a and friends from AArch64InstrInfo.td.
380 // We use these manually instead of using the importer since it doesn't
381 // support SDNodeXForm.
382 ComplexRendererFns selectShiftA_32(const MachineOperand &Root) const;
383 ComplexRendererFns selectShiftB_32(const MachineOperand &Root) const;
384 ComplexRendererFns selectShiftA_64(const MachineOperand &Root) const;
385 ComplexRendererFns selectShiftB_64(const MachineOperand &Root) const;
386
387 template <unsigned ShiftWidth>
388 ComplexRendererFns selectShiftMask(MachineOperand &Root) const;
389 ComplexRendererFns select12BitValueWithLeftShift(uint64_t Immed) const;
390 ComplexRendererFns selectArithImmed(MachineOperand &Root) const;
391 ComplexRendererFns selectNegArithImmed(MachineOperand &Root) const;
392
393 ComplexRendererFns selectAddrModeUnscaled(MachineOperand &Root,
394 unsigned Size) const;
395
396 ComplexRendererFns selectAddrModeUnscaled8(MachineOperand &Root) const {
397 return selectAddrModeUnscaled(Root, 1);
398 }
399 ComplexRendererFns selectAddrModeUnscaled16(MachineOperand &Root) const {
400 return selectAddrModeUnscaled(Root, 2);
401 }
402 ComplexRendererFns selectAddrModeUnscaled32(MachineOperand &Root) const {
403 return selectAddrModeUnscaled(Root, 4);
404 }
405 ComplexRendererFns selectAddrModeUnscaled64(MachineOperand &Root) const {
406 return selectAddrModeUnscaled(Root, 8);
407 }
408 ComplexRendererFns selectAddrModeUnscaled128(MachineOperand &Root) const {
409 return selectAddrModeUnscaled(Root, 16);
410 }
411
412 /// Helper to try to fold in a GISEL_ADD_LOW into an immediate, to be used
413 /// from complex pattern matchers like selectAddrModeIndexed().
414 ComplexRendererFns tryFoldAddLowIntoImm(MachineInstr &RootDef, unsigned Size,
415 MachineRegisterInfo &MRI) const;
416
417 ComplexRendererFns selectAddrModeIndexed(MachineOperand &Root,
418 unsigned Size) const;
419 template <int Width>
420 ComplexRendererFns selectAddrModeIndexed(MachineOperand &Root) const {
421 return selectAddrModeIndexed(Root, Width / 8);
422 }
423
424 std::optional<bool>
425 isWorthFoldingIntoAddrMode(const MachineInstr &MI,
426 const MachineRegisterInfo &MRI) const;
427
428 bool isWorthFoldingIntoExtendedReg(const MachineInstr &MI,
429 const MachineRegisterInfo &MRI,
430 bool IsAddrOperand) const;
431 ComplexRendererFns
432 selectAddrModeShiftedExtendXReg(MachineOperand &Root,
433 unsigned SizeInBytes) const;
434
435 /// Returns a \p ComplexRendererFns which contains a base, offset, and whether
436 /// or not a shift + extend should be folded into an addressing mode. Returns
437 /// None when this is not profitable or possible.
438 ComplexRendererFns
439 selectExtendedSHL(MachineOperand &Root, MachineOperand &Base,
440 MachineOperand &Offset, unsigned SizeInBytes,
441 bool WantsExt) const;
442 ComplexRendererFns selectAddrModeRegisterOffset(MachineOperand &Root) const;
443 ComplexRendererFns selectAddrModeXRO(MachineOperand &Root,
444 unsigned SizeInBytes) const;
445 template <int Width>
446 ComplexRendererFns selectAddrModeXRO(MachineOperand &Root) const {
447 return selectAddrModeXRO(Root, Width / 8);
448 }
449
450 ComplexRendererFns selectAddrModeWRO(MachineOperand &Root,
451 unsigned SizeInBytes) const;
452 template <int Width>
453 ComplexRendererFns selectAddrModeWRO(MachineOperand &Root) const {
454 return selectAddrModeWRO(Root, Width / 8);
455 }
456
457 ComplexRendererFns selectShiftedRegister(MachineOperand &Root,
458 bool AllowROR = false) const;
459
460 ComplexRendererFns selectArithShiftedRegister(MachineOperand &Root) const {
461 return selectShiftedRegister(Root);
462 }
463
464 ComplexRendererFns selectLogicalShiftedRegister(MachineOperand &Root) const {
465 return selectShiftedRegister(Root, true);
466 }
467
468 /// Given an extend instruction, determine the correct shift-extend type for
469 /// that instruction.
470 ///
471 /// If the instruction is going to be used in a load or store, pass
472 /// \p IsLoadStore = true.
474 getExtendTypeForInst(MachineInstr &MI, MachineRegisterInfo &MRI,
475 bool IsLoadStore = false) const;
476
477 /// Move \p Reg to \p RC if \p Reg is not already on \p RC.
478 ///
479 /// \returns Either \p Reg if no change was necessary, or the new register
480 /// created by moving \p Reg.
481 ///
482 /// Note: This uses emitCopy right now.
483 Register moveScalarRegClass(Register Reg, const TargetRegisterClass &RC,
484 MachineIRBuilder &MIB) const;
485
486 ComplexRendererFns selectArithExtendedRegister(MachineOperand &Root) const;
487
488 ComplexRendererFns selectExtractHigh(MachineOperand &Root) const;
489 template <unsigned Width>
490 ComplexRendererFns selectCVTFixedPoint(MachineOperand &Root) const;
491 template <unsigned Width>
492 ComplexRendererFns selectCVTFixedPosRecipOperand(MachineOperand &Root) const;
493 ComplexRendererFns selectCVTFixedPointBase(const MachineOperand &Root,
494 unsigned width,
495 bool isReciprocal = false) const;
496 ComplexRendererFns selectCVTFixedPointVec(MachineOperand &Root) const;
497 ComplexRendererFns
498 selectCVTFixedPosRecipOperandVec(MachineOperand &Root) const;
499 void renderFixedPointScalarXForm(MachineInstrBuilder &MIB,
500 const MachineInstr &MI, int OpIdx) const;
501 unsigned getFixedPointWidthFromOperand(const MachineOperand &Root) const;
502 void renderFixedPointXForm(MachineInstrBuilder &MIB, const MachineInstr &MI,
503 int OpIdx = -1) const;
504 void renderFixedPointRecipXForm(MachineInstrBuilder &MIB,
505 const MachineInstr &MI, int OpIdx = -1) const;
506 void renderFixedPointImm(MachineInstrBuilder &MIB, const MachineOperand &Root,
507 unsigned Width, bool isReciprocal) const;
508 void renderTruncImm(MachineInstrBuilder &MIB, const MachineInstr &MI,
509 int OpIdx = -1) const;
510 void renderLogicalImm32(MachineInstrBuilder &MIB, const MachineInstr &I,
511 int OpIdx = -1) const;
512 void renderLogicalImm64(MachineInstrBuilder &MIB, const MachineInstr &I,
513 int OpIdx = -1) const;
514 void renderUbsanTrap(MachineInstrBuilder &MIB, const MachineInstr &MI,
515 int OpIdx) const;
516 void renderFPImm16(MachineInstrBuilder &MIB, const MachineInstr &MI,
517 int OpIdx = -1) const;
518 void renderFPImm32(MachineInstrBuilder &MIB, const MachineInstr &MI,
519 int OpIdx = -1) const;
520 void renderFPImm64(MachineInstrBuilder &MIB, const MachineInstr &MI,
521 int OpIdx = -1) const;
522 void renderFPImm32SIMDModImmType4(MachineInstrBuilder &MIB,
523 const MachineInstr &MI,
524 int OpIdx = -1) const;
525
526 // Materialize a GlobalValue or BlockAddress using a movz+movk sequence.
527 void materializeLargeCMVal(MachineInstr &I, const Value *V, unsigned OpFlags);
528
529 // Optimization methods.
530 bool tryOptSelect(GSelect &Sel);
531 bool tryOptSelectConjunction(GSelect &Sel, MachineInstr &CondMI);
532 MachineInstr *tryFoldIntegerCompare(MachineOperand &LHS, MachineOperand &RHS,
534 MachineIRBuilder &MIRBuilder) const;
535
536 /// Return true if \p MI is a load or store of \p NumBytes bytes.
537 bool isLoadStoreOfNumBytes(const MachineInstr &MI, unsigned NumBytes) const;
538
539 /// Returns true if \p MI is guaranteed to have the high-half of a 64-bit
540 /// register zeroed out. In other words, the result of MI has been explicitly
541 /// zero extended.
542 bool isDef32(const MachineInstr &MI) const;
543
544 const AArch64TargetMachine &TM;
545 const AArch64Subtarget &STI;
546 const AArch64InstrInfo &TII;
548 const AArch64RegisterBankInfo &RBI;
549
550 bool ProduceNonFlagSettingCondBr = false;
551
552 // Some cached values used during selection.
553 // We use LR as a live-in register, and we keep track of it here as it can be
554 // clobbered by calls.
555 Register MFReturnAddr;
556
558
559#define GET_GLOBALISEL_PREDICATES_DECL
560#include "AArch64GenGlobalISel.inc"
561#undef GET_GLOBALISEL_PREDICATES_DECL
562
563// We declare the temporaries used by selectImpl() in the class to minimize the
564// cost of constructing placeholder values.
565#define GET_GLOBALISEL_TEMPORARIES_DECL
566#include "AArch64GenGlobalISel.inc"
567#undef GET_GLOBALISEL_TEMPORARIES_DECL
568};
569
570} // end anonymous namespace
571
572#define GET_GLOBALISEL_IMPL
573#include "AArch64GenGlobalISel.inc"
574#undef GET_GLOBALISEL_IMPL
575
576AArch64InstructionSelector::AArch64InstructionSelector(
577 const AArch64TargetMachine &TM, const AArch64Subtarget &STI,
578 const AArch64RegisterBankInfo &RBI)
579 : TM(TM), STI(STI), TII(*STI.getInstrInfo()), TRI(*STI.getRegisterInfo()),
580 RBI(RBI),
582#include "AArch64GenGlobalISel.inc"
585#include "AArch64GenGlobalISel.inc"
587{
588}
589
590// FIXME: This should be target-independent, inferred from the types declared
591// for each class in the bank.
592//
593/// Given a register bank, and a type, return the smallest register class that
594/// can represent that combination.
595static const TargetRegisterClass *
596getRegClassForTypeOnBank(LLT Ty, const RegisterBank &RB,
597 bool GetAllRegSet = false) {
598 if (RB.getID() == AArch64::GPRRegBankID) {
599 if (Ty.getSizeInBits() <= 32)
600 return GetAllRegSet ? &AArch64::GPR32allRegClass
601 : &AArch64::GPR32RegClass;
602 if (Ty.getSizeInBits() == 64)
603 return GetAllRegSet ? &AArch64::GPR64allRegClass
604 : &AArch64::GPR64RegClass;
605 if (Ty.getSizeInBits() == 128)
606 return &AArch64::XSeqPairsClassRegClass;
607 return nullptr;
608 }
609
610 if (RB.getID() == AArch64::FPRRegBankID) {
611 switch (Ty.getSizeInBits()) {
612 case 8:
613 return &AArch64::FPR8RegClass;
614 case 16:
615 return &AArch64::FPR16RegClass;
616 case 32:
617 return &AArch64::FPR32RegClass;
618 case 64:
619 return &AArch64::FPR64RegClass;
620 case 128:
621 return &AArch64::FPR128RegClass;
622 }
623 return nullptr;
624 }
625
626 return nullptr;
627}
628
629/// Given a register bank, and size in bits, return the smallest register class
630/// that can represent that combination.
631static const TargetRegisterClass *
633 bool GetAllRegSet = false) {
634 if (SizeInBits.isScalable()) {
635 assert(RB.getID() == AArch64::FPRRegBankID &&
636 "Expected FPR regbank for scalable type size");
637 return &AArch64::ZPRRegClass;
638 }
639
640 unsigned RegBankID = RB.getID();
641
642 if (RegBankID == AArch64::GPRRegBankID) {
643 assert(!SizeInBits.isScalable() && "Unexpected scalable register size");
644 if (SizeInBits <= 32)
645 return GetAllRegSet ? &AArch64::GPR32allRegClass
646 : &AArch64::GPR32RegClass;
647 if (SizeInBits == 64)
648 return GetAllRegSet ? &AArch64::GPR64allRegClass
649 : &AArch64::GPR64RegClass;
650 if (SizeInBits == 128)
651 return &AArch64::XSeqPairsClassRegClass;
652 }
653
654 if (RegBankID == AArch64::FPRRegBankID) {
655 if (SizeInBits.isScalable()) {
656 assert(SizeInBits == TypeSize::getScalable(128) &&
657 "Unexpected scalable register size");
658 return &AArch64::ZPRRegClass;
659 }
660
661 switch (SizeInBits) {
662 default:
663 return nullptr;
664 case 8:
665 return &AArch64::FPR8RegClass;
666 case 16:
667 return &AArch64::FPR16RegClass;
668 case 32:
669 return &AArch64::FPR32RegClass;
670 case 64:
671 return &AArch64::FPR64RegClass;
672 case 128:
673 return &AArch64::FPR128RegClass;
674 }
675 }
676
677 return nullptr;
678}
679
680/// Returns the correct subregister to use for a given register class.
682 const TargetRegisterInfo &TRI, unsigned &SubReg) {
683 switch (TRI.getRegSizeInBits(*RC)) {
684 case 8:
685 SubReg = AArch64::bsub;
686 break;
687 case 16:
688 SubReg = AArch64::hsub;
689 break;
690 case 32:
691 if (RC != &AArch64::FPR32RegClass)
692 SubReg = AArch64::sub_32;
693 else
694 SubReg = AArch64::ssub;
695 break;
696 case 64:
697 SubReg = AArch64::dsub;
698 break;
699 default:
701 dbgs() << "Couldn't find appropriate subregister for register class.");
702 return false;
703 }
704
705 return true;
706}
707
708/// Returns the minimum size the given register bank can hold.
709static unsigned getMinSizeForRegBank(const RegisterBank &RB) {
710 switch (RB.getID()) {
711 case AArch64::GPRRegBankID:
712 return 32;
713 case AArch64::FPRRegBankID:
714 return 8;
715 default:
716 llvm_unreachable("Tried to get minimum size for unknown register bank.");
717 }
718}
719
720/// Create a REG_SEQUENCE instruction using the registers in \p Regs.
721/// Helper function for functions like createDTuple and createQTuple.
722///
723/// \p RegClassIDs - The list of register class IDs available for some tuple of
724/// a scalar class. E.g. QQRegClassID, QQQRegClassID, QQQQRegClassID. This is
725/// expected to contain between 2 and 4 tuple classes.
726///
727/// \p SubRegs - The list of subregister classes associated with each register
728/// class ID in \p RegClassIDs. E.g., QQRegClassID should use the qsub0
729/// subregister class. The index of each subregister class is expected to
730/// correspond with the index of each register class.
731///
732/// \returns Either the destination register of REG_SEQUENCE instruction that
733/// was created, or the 0th element of \p Regs if \p Regs contains a single
734/// element.
736 const unsigned RegClassIDs[],
737 const unsigned SubRegs[], MachineIRBuilder &MIB) {
738 unsigned NumRegs = Regs.size();
739 if (NumRegs == 1)
740 return Regs[0];
741 assert(NumRegs >= 2 && NumRegs <= 4 &&
742 "Only support between two and 4 registers in a tuple!");
744 auto *DesiredClass = TRI->getRegClass(RegClassIDs[NumRegs - 2]);
745 auto RegSequence =
746 MIB.buildInstr(TargetOpcode::REG_SEQUENCE, {DesiredClass}, {});
747 for (unsigned I = 0, E = Regs.size(); I < E; ++I) {
748 RegSequence.addUse(Regs[I]);
749 RegSequence.addImm(SubRegs[I]);
750 }
751 return RegSequence.getReg(0);
752}
753
754/// Create a tuple of D-registers using the registers in \p Regs.
756 static const unsigned RegClassIDs[] = {
757 AArch64::DDRegClassID, AArch64::DDDRegClassID, AArch64::DDDDRegClassID};
758 static const unsigned SubRegs[] = {AArch64::dsub0, AArch64::dsub1,
759 AArch64::dsub2, AArch64::dsub3};
760 return createTuple(Regs, RegClassIDs, SubRegs, MIB);
761}
762
763/// Create a tuple of Q-registers using the registers in \p Regs.
765 static const unsigned RegClassIDs[] = {
766 AArch64::QQRegClassID, AArch64::QQQRegClassID, AArch64::QQQQRegClassID};
767 static const unsigned SubRegs[] = {AArch64::qsub0, AArch64::qsub1,
768 AArch64::qsub2, AArch64::qsub3};
769 return createTuple(Regs, RegClassIDs, SubRegs, MIB);
770}
771
772static std::optional<uint64_t> getImmedFromMO(const MachineOperand &Root) {
773 auto &MI = *Root.getParent();
774 auto &MBB = *MI.getParent();
775 auto &MF = *MBB.getParent();
776 auto &MRI = MF.getRegInfo();
777 uint64_t Immed;
778 if (Root.isImm())
779 Immed = Root.getImm();
780 else if (Root.isCImm())
781 Immed = Root.getCImm()->getZExtValue();
782 else if (Root.isReg()) {
783 auto ValAndVReg =
785 if (!ValAndVReg)
786 return std::nullopt;
787 Immed = ValAndVReg->Value.getSExtValue();
788 } else
789 return std::nullopt;
790 return Immed;
791}
792
793/// Select the AArch64 opcode for the basic binary operation \p GenericOpc,
794/// appropriate for the register bank \p RegBankID and of size \p OpSize.
795/// \returns \p GenericOpc if the combination is unsupported.
796static unsigned selectBinaryOp(unsigned GenericOpc, unsigned RegBankID,
797 unsigned OpSize) {
798 if (RegBankID == AArch64::GPRRegBankID) {
799 if (OpSize == 32) {
800 switch (GenericOpc) {
801 case TargetOpcode::G_SHL:
802 return AArch64::LSLVWr;
803 case TargetOpcode::G_LSHR:
804 return AArch64::LSRVWr;
805 case TargetOpcode::G_ASHR:
806 return AArch64::ASRVWr;
807 default:
808 return GenericOpc;
809 }
810 } else if (OpSize == 64) {
811 switch (GenericOpc) {
812 case TargetOpcode::G_SHL:
813 return AArch64::LSLVXr;
814 case TargetOpcode::G_LSHR:
815 return AArch64::LSRVXr;
816 case TargetOpcode::G_ASHR:
817 return AArch64::ASRVXr;
818 default:
819 return GenericOpc;
820 }
821 }
822 }
823 return GenericOpc;
824}
825
826/// Select the AArch64 opcode for the G_LOAD or G_STORE operation \p GenericOpc,
827/// appropriate for the (value) register bank \p RegBankID and of memory access
828/// size \p OpSize. This returns the variant with the base+unsigned-immediate
829/// addressing mode (e.g., LDRXui).
830/// \returns \p GenericOpc if the combination is unsupported.
831static unsigned selectLoadStoreUIOp(unsigned GenericOpc, unsigned RegBankID,
832 unsigned OpSize) {
833 const bool isStore = GenericOpc == TargetOpcode::G_STORE;
834 switch (RegBankID) {
835 case AArch64::GPRRegBankID:
836 switch (OpSize) {
837 case 8:
838 return isStore ? AArch64::STRBBui : AArch64::LDRBBui;
839 case 16:
840 return isStore ? AArch64::STRHHui : AArch64::LDRHHui;
841 case 32:
842 return isStore ? AArch64::STRWui : AArch64::LDRWui;
843 case 64:
844 return isStore ? AArch64::STRXui : AArch64::LDRXui;
845 }
846 break;
847 case AArch64::FPRRegBankID:
848 switch (OpSize) {
849 case 8:
850 return isStore ? AArch64::STRBui : AArch64::LDRBui;
851 case 16:
852 return isStore ? AArch64::STRHui : AArch64::LDRHui;
853 case 32:
854 return isStore ? AArch64::STRSui : AArch64::LDRSui;
855 case 64:
856 return isStore ? AArch64::STRDui : AArch64::LDRDui;
857 case 128:
858 return isStore ? AArch64::STRQui : AArch64::LDRQui;
859 }
860 break;
861 }
862 return GenericOpc;
863}
864
865/// Helper function for selectCopy. Inserts a subregister copy from \p SrcReg
866/// to \p *To.
867///
868/// E.g "To = COPY SrcReg:SubReg"
870 const RegisterBankInfo &RBI, Register SrcReg,
871 const TargetRegisterClass *To, unsigned SubReg) {
872 assert(SrcReg.isValid() && "Expected a valid source register?");
873 assert(To && "Destination register class cannot be null");
874 assert(SubReg && "Expected a valid subregister");
875
876 MachineIRBuilder MIB(I);
877 auto SubRegCopy =
878 MIB.buildInstr(TargetOpcode::COPY, {To}, {}).addReg(SrcReg, {}, SubReg);
879 MachineOperand &RegOp = I.getOperand(1);
880 RegOp.setReg(SubRegCopy.getReg(0));
881
882 // It's possible that the destination register won't be constrained. Make
883 // sure that happens.
884 if (!I.getOperand(0).getReg().isPhysical())
885 RBI.constrainGenericRegister(I.getOperand(0).getReg(), *To, MRI);
886
887 return true;
888}
889
890// FIXME: We need some sort of API in RBI/TRI to allow generic code to
891// constrain operands of simple instructions given a TargetRegisterClass
892// and LLT
894 const RegisterBankInfo &RBI) {
895 for (MachineOperand &MO : I.operands()) {
896 if (!MO.isReg())
897 continue;
898 Register Reg = MO.getReg();
899 if (!Reg)
900 continue;
901 if (Reg.isPhysical())
902 continue;
903 LLT Ty = MRI.getType(Reg);
904 const RegClassOrRegBank &RegClassOrBank = MRI.getRegClassOrRegBank(Reg);
905 const TargetRegisterClass *RC =
907 if (!RC) {
908 const RegisterBank &RB = *cast<const RegisterBank *>(RegClassOrBank);
909 RC = getRegClassForTypeOnBank(Ty, RB);
910 if (!RC) {
912 dbgs() << "Warning: DBG_VALUE operand has unexpected size/bank\n");
913 break;
914 }
915 }
916 RBI.constrainGenericRegister(Reg, *RC, MRI);
917 }
918
919 return true;
920}
921
924 const RegisterBankInfo &RBI) {
925 Register DstReg = I.getOperand(0).getReg();
926 Register SrcReg = I.getOperand(1).getReg();
927 const RegisterBank &DstRegBank = *RBI.getRegBank(DstReg, MRI, TRI);
928 const RegisterBank &SrcRegBank = *RBI.getRegBank(SrcReg, MRI, TRI);
929
930 TypeSize DstRegSize = RBI.getSizeInBits(DstReg, MRI, TRI);
931 TypeSize SrcRegSize = RBI.getSizeInBits(SrcReg, MRI, TRI);
932
933 // Special casing for cross-bank copies of s1s. We can technically represent
934 // a 1-bit value with any size of register. The minimum size for a GPR is 32
935 // bits. So, we need to put the FPR on 32 bits as well.
936 //
937 // FIXME: I'm not sure if this case holds true outside of copies. If it does,
938 // then we can pull it into the helpers that get the appropriate class for a
939 // register bank. Or make a new helper that carries along some constraint
940 // information.
941 if (SrcRegBank != DstRegBank && (DstRegSize == TypeSize::getFixed(1) &&
942 SrcRegSize == TypeSize::getFixed(1)))
943 SrcRegSize = DstRegSize = TypeSize::getFixed(32);
944
945 // Find the correct register classes for the source and destination registers.
946 const TargetRegisterClass *SrcRC =
947 getMinClassForRegBank(SrcRegBank, SrcRegSize, true);
948 const TargetRegisterClass *DstRC =
949 getMinClassForRegBank(DstRegBank, DstRegSize, true);
950
951 if (!DstRC) {
952 LLVM_DEBUG(dbgs() << "Unexpected dest size "
953 << RBI.getSizeInBits(DstReg, MRI, TRI) << '\n');
954 return false;
955 }
956
957 if (I.getOpcode() == TargetOpcode::G_BITCAST &&
958 RBI.getSizeInBits(DstReg, MRI, TRI) == TypeSize::getFixed(16)) {
959 if (DstRegBank.getID() == AArch64::FPRRegBankID &&
960 SrcRegBank.getID() == AArch64::GPRRegBankID) {
961 if (!SrcReg.isPhysical() &&
962 !RBI.constrainGenericRegister(SrcReg, AArch64::GPR32RegClass, MRI))
963 return false;
964 if (!DstReg.isPhysical() &&
965 !RBI.constrainGenericRegister(DstReg, AArch64::FPR16RegClass, MRI))
966 return false;
967
968 Register FPR32 = MRI.createVirtualRegister(&AArch64::FPR32RegClass);
969 BuildMI(*I.getParent(), I, I.getDebugLoc(), TII.get(AArch64::FMOVWSr))
970 .addDef(FPR32)
971 .addUse(SrcReg);
972 I.setDesc(TII.get(TargetOpcode::COPY));
973 I.getOperand(1).setReg(FPR32);
974 I.getOperand(1).setSubReg(AArch64::hsub);
975 return true;
976 }
977
978 if (DstRegBank.getID() == AArch64::GPRRegBankID &&
979 SrcRegBank.getID() == AArch64::FPRRegBankID) {
980 if (!SrcReg.isPhysical() &&
981 !RBI.constrainGenericRegister(SrcReg, AArch64::FPR16RegClass, MRI))
982 return false;
983 if (!DstReg.isPhysical() &&
984 !RBI.constrainGenericRegister(DstReg, AArch64::GPR32RegClass, MRI))
985 return false;
986
987 Register FPR32 = MRI.createVirtualRegister(&AArch64::FPR32RegClass);
988 BuildMI(*I.getParent(), I, I.getDebugLoc(),
989 TII.get(TargetOpcode::SUBREG_TO_REG))
990 .addDef(FPR32)
991 .addUse(SrcReg)
992 .addImm(AArch64::hsub);
993 I.setDesc(TII.get(AArch64::FMOVSWr));
994 I.getOperand(1).setReg(FPR32);
995 return true;
996 }
997 }
998
999 // Is this a copy? If so, then we may need to insert a subregister copy.
1000 if (I.isCopy()) {
1001 // Yes. Check if there's anything to fix up.
1002 if (!SrcRC) {
1003 LLVM_DEBUG(dbgs() << "Couldn't determine source register class\n");
1004 return false;
1005 }
1006
1007 const TypeSize SrcSize = TRI.getRegSizeInBits(*SrcRC);
1008 const TypeSize DstSize = TRI.getRegSizeInBits(*DstRC);
1009 unsigned SrcSubReg = I.getOperand(1).getSubReg();
1010 unsigned SubReg;
1011
1012 if (SrcSubReg)
1013 return RBI.constrainGenericRegister(DstReg, *DstRC, MRI);
1014
1015 // If the source bank doesn't support a subregister copy small enough,
1016 // then we first need to copy to the destination bank.
1017 if (getMinSizeForRegBank(SrcRegBank) > DstSize) {
1018 const TargetRegisterClass *DstTempRC =
1019 getMinClassForRegBank(DstRegBank, SrcSize, /* GetAllRegSet */ true);
1020 getSubRegForClass(DstRC, TRI, SubReg);
1021
1022 MachineIRBuilder MIB(I);
1023 auto Copy = MIB.buildCopy({DstTempRC}, {SrcReg});
1024 copySubReg(I, MRI, RBI, Copy.getReg(0), DstRC, SubReg);
1025 } else if (SrcSize > DstSize) {
1026 // If the source register is bigger than the destination we need to
1027 // perform a subregister copy.
1028 const TargetRegisterClass *SubRegRC =
1029 getMinClassForRegBank(SrcRegBank, DstSize, /* GetAllRegSet */ true);
1030 getSubRegForClass(SubRegRC, TRI, SubReg);
1031 copySubReg(I, MRI, RBI, SrcReg, DstRC, SubReg);
1032 } else if (DstSize > SrcSize) {
1033 // If the destination register is bigger than the source we need to do
1034 // a promotion using SUBREG_TO_REG.
1035 const TargetRegisterClass *PromotionRC =
1036 getMinClassForRegBank(SrcRegBank, DstSize, /* GetAllRegSet */ true);
1037 getSubRegForClass(SrcRC, TRI, SubReg);
1038
1039 Register PromoteReg = MRI.createVirtualRegister(PromotionRC);
1040 BuildMI(*I.getParent(), I, I.getDebugLoc(),
1041 TII.get(AArch64::SUBREG_TO_REG), PromoteReg)
1042 .addUse(SrcReg)
1043 .addImm(SubReg);
1044 MachineOperand &RegOp = I.getOperand(1);
1045 RegOp.setReg(PromoteReg);
1046 }
1047
1048 // If the destination is a physical register, then there's nothing to
1049 // change, so we're done.
1050 if (DstReg.isPhysical())
1051 return true;
1052 }
1053
1054 // No need to constrain SrcReg. It will get constrained when we hit another
1055 // of its use or its defs. Copies do not have constraints.
1056 if (!RBI.constrainGenericRegister(DstReg, *DstRC, MRI)) {
1057 LLVM_DEBUG(dbgs() << "Failed to constrain " << TII.getName(I.getOpcode())
1058 << " operand\n");
1059 return false;
1060 }
1061
1062 // If this a GPR ZEXT that we want to just reduce down into a copy.
1063 // The sizes will be mismatched with the source < 32b but that's ok.
1064 if (I.getOpcode() == TargetOpcode::G_ZEXT) {
1065 I.setDesc(TII.get(AArch64::COPY));
1066 assert(SrcRegBank.getID() == AArch64::GPRRegBankID);
1067 return selectCopy(I, TII, MRI, TRI, RBI);
1068 }
1069
1070 I.setDesc(TII.get(AArch64::COPY));
1071 return true;
1072}
1073
1075AArch64InstructionSelector::emitSelect(Register Dst, Register True,
1076 Register False, AArch64CC::CondCode CC,
1077 MachineIRBuilder &MIB) const {
1078 MachineRegisterInfo &MRI = *MIB.getMRI();
1079 assert(RBI.getRegBank(False, MRI, TRI)->getID() ==
1080 RBI.getRegBank(True, MRI, TRI)->getID() &&
1081 "Expected both select operands to have the same regbank?");
1082 LLT Ty = MRI.getType(True);
1083 if (Ty.isVector())
1084 return nullptr;
1085 const unsigned Size = Ty.getSizeInBits();
1086 assert((Size == 32 || Size == 64) &&
1087 "Expected 32 bit or 64 bit select only?");
1088 const bool Is32Bit = Size == 32;
1089 if (RBI.getRegBank(True, MRI, TRI)->getID() != AArch64::GPRRegBankID) {
1090 unsigned Opc = Is32Bit ? AArch64::FCSELSrrr : AArch64::FCSELDrrr;
1091 auto FCSel = MIB.buildInstr(Opc, {Dst}, {True, False}).addImm(CC);
1093 return &*FCSel;
1094 }
1095
1096 // By default, we'll try and emit a CSEL.
1097 unsigned Opc = Is32Bit ? AArch64::CSELWr : AArch64::CSELXr;
1098 bool Optimized = false;
1099 auto TryFoldBinOpIntoSelect = [&Opc, Is32Bit, &CC, &MRI,
1100 &Optimized](Register &Reg, Register &OtherReg,
1101 bool Invert) {
1102 if (Optimized)
1103 return false;
1104
1105 // Attempt to fold:
1106 //
1107 // %sub = G_SUB 0, %x
1108 // %select = G_SELECT cc, %reg, %sub
1109 //
1110 // Into:
1111 // %select = CSNEG %reg, %x, cc
1112 Register MatchReg;
1113 if (mi_match(Reg, MRI, m_Neg(m_Reg(MatchReg)))) {
1114 Opc = Is32Bit ? AArch64::CSNEGWr : AArch64::CSNEGXr;
1115 Reg = MatchReg;
1116 if (Invert) {
1118 std::swap(Reg, OtherReg);
1119 }
1120 return true;
1121 }
1122
1123 // Attempt to fold:
1124 //
1125 // %xor = G_XOR %x, -1
1126 // %select = G_SELECT cc, %reg, %xor
1127 //
1128 // Into:
1129 // %select = CSINV %reg, %x, cc
1130 if (mi_match(Reg, MRI, m_Not(m_Reg(MatchReg)))) {
1131 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1132 Reg = MatchReg;
1133 if (Invert) {
1135 std::swap(Reg, OtherReg);
1136 }
1137 return true;
1138 }
1139
1140 // Attempt to fold:
1141 //
1142 // %add = G_ADD %x, 1
1143 // %select = G_SELECT cc, %reg, %add
1144 //
1145 // Into:
1146 // %select = CSINC %reg, %x, cc
1147 if (mi_match(Reg, MRI,
1148 m_any_of(m_GAdd(m_Reg(MatchReg), m_SpecificICst(1)),
1149 m_GPtrAdd(m_Reg(MatchReg), m_SpecificICst(1))))) {
1150 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1151 Reg = MatchReg;
1152 if (Invert) {
1154 std::swap(Reg, OtherReg);
1155 }
1156 return true;
1157 }
1158
1159 return false;
1160 };
1161
1162 // Helper lambda which tries to use CSINC/CSINV for the instruction when its
1163 // true/false values are constants.
1164 // FIXME: All of these patterns already exist in tablegen. We should be
1165 // able to import these.
1166 auto TryOptSelectCst = [&Opc, &True, &False, &CC, Is32Bit, &MRI,
1167 &Optimized]() {
1168 if (Optimized)
1169 return false;
1170 auto TrueCst = getIConstantVRegValWithLookThrough(True, MRI);
1171 auto FalseCst = getIConstantVRegValWithLookThrough(False, MRI);
1172 if (!TrueCst && !FalseCst)
1173 return false;
1174
1175 Register ZReg = Is32Bit ? AArch64::WZR : AArch64::XZR;
1176 if (TrueCst && FalseCst) {
1177 int64_t T = TrueCst->Value.getSExtValue();
1178 int64_t F = FalseCst->Value.getSExtValue();
1179
1180 if (T == 0 && F == 1) {
1181 // G_SELECT cc, 0, 1 -> CSINC zreg, zreg, cc
1182 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1183 True = ZReg;
1184 False = ZReg;
1185 return true;
1186 }
1187
1188 if (T == 0 && F == -1) {
1189 // G_SELECT cc 0, -1 -> CSINV zreg, zreg cc
1190 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1191 True = ZReg;
1192 False = ZReg;
1193 return true;
1194 }
1195 }
1196
1197 if (TrueCst) {
1198 int64_t T = TrueCst->Value.getSExtValue();
1199 if (T == 1) {
1200 // G_SELECT cc, 1, f -> CSINC f, zreg, inv_cc
1201 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1202 True = False;
1203 False = ZReg;
1205 return true;
1206 }
1207
1208 if (T == -1) {
1209 // G_SELECT cc, -1, f -> CSINV f, zreg, inv_cc
1210 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1211 True = False;
1212 False = ZReg;
1214 return true;
1215 }
1216 }
1217
1218 if (FalseCst) {
1219 int64_t F = FalseCst->Value.getSExtValue();
1220 if (F == 1) {
1221 // G_SELECT cc, t, 1 -> CSINC t, zreg, cc
1222 Opc = Is32Bit ? AArch64::CSINCWr : AArch64::CSINCXr;
1223 False = ZReg;
1224 return true;
1225 }
1226
1227 if (F == -1) {
1228 // G_SELECT cc, t, -1 -> CSINC t, zreg, cc
1229 Opc = Is32Bit ? AArch64::CSINVWr : AArch64::CSINVXr;
1230 False = ZReg;
1231 return true;
1232 }
1233 }
1234 return false;
1235 };
1236
1237 Optimized |= TryFoldBinOpIntoSelect(False, True, /*Invert = */ false);
1238 Optimized |= TryFoldBinOpIntoSelect(True, False, /*Invert = */ true);
1239 Optimized |= TryOptSelectCst();
1240 auto SelectInst = MIB.buildInstr(Opc, {Dst}, {True, False}).addImm(CC);
1241 constrainSelectedInstRegOperands(*SelectInst, TII, TRI, RBI);
1242 return &*SelectInst;
1243}
1244
1247 MachineRegisterInfo *MRI = nullptr) {
1248 switch (P) {
1249 default:
1250 llvm_unreachable("Unknown condition code!");
1251 case CmpInst::ICMP_NE:
1252 return AArch64CC::NE;
1253 case CmpInst::ICMP_EQ:
1254 return AArch64CC::EQ;
1255 case CmpInst::ICMP_SGT:
1256 return AArch64CC::GT;
1257 case CmpInst::ICMP_SGE:
1258 if (RHS && MRI) {
1259 auto ValAndVReg = getIConstantVRegValWithLookThrough(RHS, *MRI);
1260 if (ValAndVReg && ValAndVReg->Value == 0)
1261 return AArch64CC::PL;
1262 }
1263 return AArch64CC::GE;
1264 case CmpInst::ICMP_SLT:
1265 if (RHS && MRI) {
1266 auto ValAndVReg = getIConstantVRegValWithLookThrough(RHS, *MRI);
1267 if (ValAndVReg && ValAndVReg->Value == 0)
1268 return AArch64CC::MI;
1269 }
1270 return AArch64CC::LT;
1271 case CmpInst::ICMP_SLE:
1272 return AArch64CC::LE;
1273 case CmpInst::ICMP_UGT:
1274 return AArch64CC::HI;
1275 case CmpInst::ICMP_UGE:
1276 return AArch64CC::HS;
1277 case CmpInst::ICMP_ULT:
1278 return AArch64CC::LO;
1279 case CmpInst::ICMP_ULE:
1280 return AArch64CC::LS;
1281 }
1282}
1283
1284/// changeFPCCToORAArch64CC - Convert an IR fp condition code to an AArch64 CC.
1286 AArch64CC::CondCode &CondCode,
1287 AArch64CC::CondCode &CondCode2) {
1288 CondCode2 = AArch64CC::AL;
1289 switch (CC) {
1290 default:
1291 llvm_unreachable("Unknown FP condition!");
1292 case CmpInst::FCMP_OEQ:
1293 CondCode = AArch64CC::EQ;
1294 break;
1295 case CmpInst::FCMP_OGT:
1296 CondCode = AArch64CC::GT;
1297 break;
1298 case CmpInst::FCMP_OGE:
1299 CondCode = AArch64CC::GE;
1300 break;
1301 case CmpInst::FCMP_OLT:
1302 CondCode = AArch64CC::MI;
1303 break;
1304 case CmpInst::FCMP_OLE:
1305 CondCode = AArch64CC::LS;
1306 break;
1307 case CmpInst::FCMP_ONE:
1308 CondCode = AArch64CC::MI;
1309 CondCode2 = AArch64CC::GT;
1310 break;
1311 case CmpInst::FCMP_ORD:
1312 CondCode = AArch64CC::VC;
1313 break;
1314 case CmpInst::FCMP_UNO:
1315 CondCode = AArch64CC::VS;
1316 break;
1317 case CmpInst::FCMP_UEQ:
1318 CondCode = AArch64CC::EQ;
1319 CondCode2 = AArch64CC::VS;
1320 break;
1321 case CmpInst::FCMP_UGT:
1322 CondCode = AArch64CC::HI;
1323 break;
1324 case CmpInst::FCMP_UGE:
1325 CondCode = AArch64CC::PL;
1326 break;
1327 case CmpInst::FCMP_ULT:
1328 CondCode = AArch64CC::LT;
1329 break;
1330 case CmpInst::FCMP_ULE:
1331 CondCode = AArch64CC::LE;
1332 break;
1333 case CmpInst::FCMP_UNE:
1334 CondCode = AArch64CC::NE;
1335 break;
1336 }
1337}
1338
1339/// Convert an IR fp condition code to an AArch64 CC.
1340/// This differs from changeFPCCToAArch64CC in that it returns cond codes that
1341/// should be AND'ed instead of OR'ed.
1343 AArch64CC::CondCode &CondCode,
1344 AArch64CC::CondCode &CondCode2) {
1345 CondCode2 = AArch64CC::AL;
1346 switch (CC) {
1347 default:
1348 changeFPCCToORAArch64CC(CC, CondCode, CondCode2);
1349 assert(CondCode2 == AArch64CC::AL);
1350 break;
1351 case CmpInst::FCMP_ONE:
1352 // (a one b)
1353 // == ((a olt b) || (a ogt b))
1354 // == ((a ord b) && (a une b))
1355 CondCode = AArch64CC::VC;
1356 CondCode2 = AArch64CC::NE;
1357 break;
1358 case CmpInst::FCMP_UEQ:
1359 // (a ueq b)
1360 // == ((a uno b) || (a oeq b))
1361 // == ((a ule b) && (a uge b))
1362 CondCode = AArch64CC::PL;
1363 CondCode2 = AArch64CC::LE;
1364 break;
1365 }
1366}
1367
1368/// Return a register which can be used as a bit to test in a TB(N)Z.
1369static Register getTestBitReg(Register Reg, uint64_t &Bit, bool &Invert,
1370 MachineRegisterInfo &MRI) {
1371 assert(Reg.isValid() && "Expected valid register!");
1372 bool HasZext = false;
1373 while (MachineInstr *MI = getDefIgnoringCopies(Reg, MRI)) {
1374 unsigned Opc = MI->getOpcode();
1375
1376 if (!MI->getOperand(0).isReg() ||
1377 !MRI.hasOneNonDBGUse(MI->getOperand(0).getReg()))
1378 break;
1379
1380 // (tbz (any_ext x), b) -> (tbz x, b) and
1381 // (tbz (zext x), b) -> (tbz x, b) if we don't use the extended bits.
1382 //
1383 // (tbz (trunc x), b) -> (tbz x, b) is always safe, because the bit number
1384 // on the truncated x is the same as the bit number on x.
1385 if (Opc == TargetOpcode::G_ANYEXT || Opc == TargetOpcode::G_ZEXT ||
1386 Opc == TargetOpcode::G_TRUNC) {
1387 if (Opc == TargetOpcode::G_ZEXT)
1388 HasZext = true;
1389
1390 Register NextReg = MI->getOperand(1).getReg();
1391 // Did we find something worth folding?
1392 if (!NextReg.isValid() || !MRI.hasOneNonDBGUse(NextReg))
1393 break;
1394 TypeSize InSize = MRI.getType(NextReg).getSizeInBits();
1395 if (Bit >= InSize)
1396 break;
1397
1398 // NextReg is worth folding. Keep looking.
1399 Reg = NextReg;
1400 continue;
1401 }
1402
1403 // Attempt to find a suitable operation with a constant on one side.
1404 std::optional<uint64_t> C;
1405 Register TestReg;
1406 switch (Opc) {
1407 default:
1408 break;
1409 case TargetOpcode::G_AND:
1410 case TargetOpcode::G_XOR: {
1411 TestReg = MI->getOperand(1).getReg();
1412 Register ConstantReg = MI->getOperand(2).getReg();
1413 auto VRegAndVal = getIConstantVRegValWithLookThrough(ConstantReg, MRI);
1414 if (!VRegAndVal) {
1415 // AND commutes, check the other side for a constant.
1416 // FIXME: Can we canonicalize the constant so that it's always on the
1417 // same side at some point earlier?
1418 std::swap(ConstantReg, TestReg);
1419 VRegAndVal = getIConstantVRegValWithLookThrough(ConstantReg, MRI);
1420 }
1421 if (VRegAndVal) {
1422 if (HasZext)
1423 C = VRegAndVal->Value.getZExtValue();
1424 else
1425 C = VRegAndVal->Value.getSExtValue();
1426 }
1427 break;
1428 }
1429 case TargetOpcode::G_ASHR:
1430 case TargetOpcode::G_LSHR:
1431 case TargetOpcode::G_SHL: {
1432 TestReg = MI->getOperand(1).getReg();
1433 auto VRegAndVal =
1434 getIConstantVRegValWithLookThrough(MI->getOperand(2).getReg(), MRI);
1435 if (VRegAndVal)
1436 C = VRegAndVal->Value.getSExtValue();
1437 break;
1438 }
1439 }
1440
1441 // Didn't find a constant or viable register. Bail out of the loop.
1442 if (!C || !TestReg.isValid())
1443 break;
1444
1445 // We found a suitable instruction with a constant. Check to see if we can
1446 // walk through the instruction.
1447 Register NextReg;
1448 unsigned TestRegSize = MRI.getType(TestReg).getSizeInBits();
1449 switch (Opc) {
1450 default:
1451 break;
1452 case TargetOpcode::G_AND:
1453 // (tbz (and x, m), b) -> (tbz x, b) when the b-th bit of m is set.
1454 if ((*C >> Bit) & 1)
1455 NextReg = TestReg;
1456 break;
1457 case TargetOpcode::G_SHL:
1458 // (tbz (shl x, c), b) -> (tbz x, b-c) when b-c is positive and fits in
1459 // the type of the register.
1460 if (*C <= Bit && (Bit - *C) < TestRegSize) {
1461 NextReg = TestReg;
1462 Bit = Bit - *C;
1463 }
1464 break;
1465 case TargetOpcode::G_ASHR:
1466 // (tbz (ashr x, c), b) -> (tbz x, b+c) or (tbz x, msb) if b+c is > # bits
1467 // in x
1468 NextReg = TestReg;
1469 Bit = Bit + *C;
1470 if (Bit >= TestRegSize)
1471 Bit = TestRegSize - 1;
1472 break;
1473 case TargetOpcode::G_LSHR:
1474 // (tbz (lshr x, c), b) -> (tbz x, b+c) when b + c is < # bits in x
1475 if ((Bit + *C) < TestRegSize) {
1476 NextReg = TestReg;
1477 Bit = Bit + *C;
1478 }
1479 break;
1480 case TargetOpcode::G_XOR:
1481 // We can walk through a G_XOR by inverting whether we use tbz/tbnz when
1482 // appropriate.
1483 //
1484 // e.g. If x' = xor x, c, and the b-th bit is set in c then
1485 //
1486 // tbz x', b -> tbnz x, b
1487 //
1488 // Because x' only has the b-th bit set if x does not.
1489 if ((*C >> Bit) & 1)
1490 Invert = !Invert;
1491 NextReg = TestReg;
1492 break;
1493 }
1494
1495 // Check if we found anything worth folding.
1496 if (!NextReg.isValid())
1497 return Reg;
1498 Reg = NextReg;
1499 }
1500
1501 return Reg;
1502}
1503
1504MachineInstr *AArch64InstructionSelector::emitTestBit(
1505 Register TestReg, uint64_t Bit, bool IsNegative, MachineBasicBlock *DstMBB,
1506 MachineIRBuilder &MIB) const {
1507 assert(TestReg.isValid());
1508 assert(ProduceNonFlagSettingCondBr &&
1509 "Cannot emit TB(N)Z with speculation tracking!");
1510 MachineRegisterInfo &MRI = *MIB.getMRI();
1511
1512 // Attempt to optimize the test bit by walking over instructions.
1513 TestReg = getTestBitReg(TestReg, Bit, IsNegative, MRI);
1514 LLT Ty = MRI.getType(TestReg);
1515 unsigned Size = Ty.getSizeInBits();
1516 assert(!Ty.isVector() && "Expected a scalar!");
1517 assert(Bit < 64 && "Bit is too large!");
1518
1519 // When the test register is a 64-bit register, we have to narrow to make
1520 // TBNZW work.
1521 bool UseWReg = Bit < 32;
1522 unsigned NecessarySize = UseWReg ? 32 : 64;
1523 if (Size != NecessarySize)
1524 TestReg = moveScalarRegClass(
1525 TestReg, UseWReg ? AArch64::GPR32RegClass : AArch64::GPR64RegClass,
1526 MIB);
1527
1528 static const unsigned OpcTable[2][2] = {{AArch64::TBZX, AArch64::TBNZX},
1529 {AArch64::TBZW, AArch64::TBNZW}};
1530 unsigned Opc = OpcTable[UseWReg][IsNegative];
1531 auto TestBitMI =
1532 MIB.buildInstr(Opc).addReg(TestReg).addImm(Bit).addMBB(DstMBB);
1533 constrainSelectedInstRegOperands(*TestBitMI, TII, TRI, RBI);
1534 return &*TestBitMI;
1535}
1536
1537bool AArch64InstructionSelector::tryOptAndIntoCompareBranch(
1538 MachineInstr &AndInst, bool Invert, MachineBasicBlock *DstMBB,
1539 MachineIRBuilder &MIB) const {
1540 assert(AndInst.getOpcode() == TargetOpcode::G_AND && "Expected G_AND only?");
1541 // Given something like this:
1542 //
1543 // %x = ...Something...
1544 // %one = G_CONSTANT i64 1
1545 // %zero = G_CONSTANT i64 0
1546 // %and = G_AND %x, %one
1547 // %cmp = G_ICMP intpred(ne), %and, %zero
1548 // %cmp_trunc = G_TRUNC %cmp
1549 // G_BRCOND %cmp_trunc, %bb.3
1550 //
1551 // We want to try and fold the AND into the G_BRCOND and produce either a
1552 // TBNZ (when we have intpred(ne)) or a TBZ (when we have intpred(eq)).
1553 //
1554 // In this case, we'd get
1555 //
1556 // TBNZ %x %bb.3
1557 //
1558
1559 // Check if the AND has a constant on its RHS which we can use as a mask.
1560 // If it's a power of 2, then it's the same as checking a specific bit.
1561 // (e.g, ANDing with 8 == ANDing with 000...100 == testing if bit 3 is set)
1562 auto MaybeBit = getIConstantVRegValWithLookThrough(
1563 AndInst.getOperand(2).getReg(), *MIB.getMRI());
1564 if (!MaybeBit)
1565 return false;
1566
1567 int32_t Bit = MaybeBit->Value.exactLogBase2();
1568 if (Bit < 0)
1569 return false;
1570
1571 Register TestReg = AndInst.getOperand(1).getReg();
1572
1573 // Emit a TB(N)Z.
1574 emitTestBit(TestReg, Bit, Invert, DstMBB, MIB);
1575 return true;
1576}
1577
1578MachineInstr *AArch64InstructionSelector::emitCBZ(Register CompareReg,
1579 bool IsNegative,
1580 MachineBasicBlock *DestMBB,
1581 MachineIRBuilder &MIB) const {
1582 assert(ProduceNonFlagSettingCondBr && "CBZ does not set flags!");
1583 MachineRegisterInfo &MRI = *MIB.getMRI();
1584 assert(RBI.getRegBank(CompareReg, MRI, TRI)->getID() ==
1585 AArch64::GPRRegBankID &&
1586 "Expected GPRs only?");
1587 auto Ty = MRI.getType(CompareReg);
1588 unsigned Width = Ty.getSizeInBits();
1589 assert(!Ty.isVector() && "Expected scalar only?");
1590 assert(Width <= 64 && "Expected width to be at most 64?");
1591 static const unsigned OpcTable[2][2] = {{AArch64::CBZW, AArch64::CBZX},
1592 {AArch64::CBNZW, AArch64::CBNZX}};
1593 unsigned Opc = OpcTable[IsNegative][Width == 64];
1594 auto BranchMI = MIB.buildInstr(Opc, {}, {CompareReg}).addMBB(DestMBB);
1595 constrainSelectedInstRegOperands(*BranchMI, TII, TRI, RBI);
1596 return &*BranchMI;
1597}
1598
1599bool AArch64InstructionSelector::selectCompareBranchFedByFCmp(
1600 MachineInstr &I, MachineInstr &FCmp, MachineIRBuilder &MIB) const {
1601 assert(FCmp.getOpcode() == TargetOpcode::G_FCMP);
1602 assert(I.getOpcode() == TargetOpcode::G_BRCOND);
1603 // Unfortunately, the mapping of LLVM FP CC's onto AArch64 CC's isn't
1604 // totally clean. Some of them require two branches to implement.
1605 auto Pred = (CmpInst::Predicate)FCmp.getOperand(1).getPredicate();
1606 emitFPCompare(FCmp.getOperand(2).getReg(), FCmp.getOperand(3).getReg(), MIB,
1607 Pred);
1608 AArch64CC::CondCode CC1, CC2;
1609 changeFCMPPredToAArch64CC(Pred, CC1, CC2);
1610 MachineBasicBlock *DestMBB = I.getOperand(1).getMBB();
1611 MIB.buildInstr(AArch64::Bcc, {}, {}).addImm(CC1).addMBB(DestMBB);
1612 if (CC2 != AArch64CC::AL)
1613 MIB.buildInstr(AArch64::Bcc, {}, {}).addImm(CC2).addMBB(DestMBB);
1614 I.eraseFromParent();
1615 return true;
1616}
1617
1618bool AArch64InstructionSelector::tryOptCompareBranchFedByICmp(
1619 MachineInstr &I, MachineInstr &ICmp, MachineIRBuilder &MIB) const {
1620 assert(ICmp.getOpcode() == TargetOpcode::G_ICMP);
1621 assert(I.getOpcode() == TargetOpcode::G_BRCOND);
1622 // Attempt to optimize the G_BRCOND + G_ICMP into a TB(N)Z/CB(N)Z.
1623 //
1624 // Speculation tracking/SLH assumes that optimized TB(N)Z/CB(N)Z
1625 // instructions will not be produced, as they are conditional branch
1626 // instructions that do not set flags.
1627 if (!ProduceNonFlagSettingCondBr)
1628 return false;
1629
1630 MachineRegisterInfo &MRI = *MIB.getMRI();
1631 MachineBasicBlock *DestMBB = I.getOperand(1).getMBB();
1632 auto Pred =
1633 static_cast<CmpInst::Predicate>(ICmp.getOperand(1).getPredicate());
1634 Register LHS = ICmp.getOperand(2).getReg();
1635 Register RHS = ICmp.getOperand(3).getReg();
1636
1637 // We're allowed to emit a TB(N)Z/CB(N)Z. Try to do that.
1638 auto VRegAndVal = getIConstantVRegValWithLookThrough(RHS, MRI);
1639 MachineInstr *AndInst = getOpcodeDef(TargetOpcode::G_AND, LHS, MRI);
1640
1641 // When we can emit a TB(N)Z, prefer that.
1642 //
1643 // Handle non-commutative condition codes first.
1644 // Note that we don't want to do this when we have a G_AND because it can
1645 // become a tst. The tst will make the test bit in the TB(N)Z redundant.
1646 if (VRegAndVal && !AndInst) {
1647 int64_t C = VRegAndVal->Value.getSExtValue();
1648
1649 // When we have a greater-than comparison, we can just test if the msb is
1650 // zero.
1651 if (C == -1 && Pred == CmpInst::ICMP_SGT) {
1652 uint64_t Bit = MRI.getType(LHS).getSizeInBits() - 1;
1653 emitTestBit(LHS, Bit, /*IsNegative = */ false, DestMBB, MIB);
1654 I.eraseFromParent();
1655 return true;
1656 }
1657
1658 // When we have a less than comparison, we can just test if the msb is not
1659 // zero.
1660 if (C == 0 && Pred == CmpInst::ICMP_SLT) {
1661 uint64_t Bit = MRI.getType(LHS).getSizeInBits() - 1;
1662 emitTestBit(LHS, Bit, /*IsNegative = */ true, DestMBB, MIB);
1663 I.eraseFromParent();
1664 return true;
1665 }
1666
1667 // Inversely, if we have a signed greater-than-or-equal comparison to zero,
1668 // we can test if the msb is zero.
1669 if (C == 0 && Pred == CmpInst::ICMP_SGE) {
1670 uint64_t Bit = MRI.getType(LHS).getSizeInBits() - 1;
1671 emitTestBit(LHS, Bit, /*IsNegative = */ false, DestMBB, MIB);
1672 I.eraseFromParent();
1673 return true;
1674 }
1675 }
1676
1677 // Attempt to handle commutative condition codes. Right now, that's only
1678 // eq/ne.
1679 if (ICmpInst::isEquality(Pred)) {
1680 if (!VRegAndVal) {
1681 std::swap(RHS, LHS);
1682 VRegAndVal = getIConstantVRegValWithLookThrough(RHS, MRI);
1683 AndInst = getOpcodeDef(TargetOpcode::G_AND, LHS, MRI);
1684 }
1685
1686 if (VRegAndVal && VRegAndVal->Value == 0) {
1687 // If there's a G_AND feeding into this branch, try to fold it away by
1688 // emitting a TB(N)Z instead.
1689 //
1690 // Note: If we have LT, then it *is* possible to fold, but it wouldn't be
1691 // beneficial. When we have an AND and LT, we need a TST/ANDS, so folding
1692 // would be redundant.
1693 if (AndInst &&
1694 tryOptAndIntoCompareBranch(
1695 *AndInst, /*Invert = */ Pred == CmpInst::ICMP_NE, DestMBB, MIB)) {
1696 I.eraseFromParent();
1697 return true;
1698 }
1699
1700 // Otherwise, try to emit a CB(N)Z instead.
1701 auto LHSTy = MRI.getType(LHS);
1702 if (!LHSTy.isVector() && LHSTy.getSizeInBits() <= 64) {
1703 emitCBZ(LHS, /*IsNegative = */ Pred == CmpInst::ICMP_NE, DestMBB, MIB);
1704 I.eraseFromParent();
1705 return true;
1706 }
1707 }
1708 }
1709
1710 return false;
1711}
1712
1713bool AArch64InstructionSelector::selectCompareBranchFedByICmp(
1714 MachineInstr &I, MachineInstr &ICmp, MachineIRBuilder &MIB) const {
1715 assert(ICmp.getOpcode() == TargetOpcode::G_ICMP);
1716 assert(I.getOpcode() == TargetOpcode::G_BRCOND);
1717 if (tryOptCompareBranchFedByICmp(I, ICmp, MIB))
1718 return true;
1719
1720 // Couldn't optimize. Emit a compare + a Bcc.
1721 MachineBasicBlock *DestMBB = I.getOperand(1).getMBB();
1722 auto &PredOp = ICmp.getOperand(1);
1723 emitIntegerCompare(ICmp.getOperand(2), ICmp.getOperand(3), PredOp, MIB);
1725 static_cast<CmpInst::Predicate>(PredOp.getPredicate()),
1726 ICmp.getOperand(3).getReg(), MIB.getMRI());
1727 MIB.buildInstr(AArch64::Bcc, {}, {}).addImm(CC).addMBB(DestMBB);
1728 I.eraseFromParent();
1729 return true;
1730}
1731
1732bool AArch64InstructionSelector::selectCompareBranch(
1733 MachineInstr &I, MachineFunction &MF, MachineRegisterInfo &MRI) {
1734 Register CondReg = I.getOperand(0).getReg();
1735 MachineInstr *CCMI = MRI.getVRegDef(CondReg);
1736 // Try to select the G_BRCOND using whatever is feeding the condition if
1737 // possible.
1738 unsigned CCMIOpc = CCMI->getOpcode();
1739 if (CCMIOpc == TargetOpcode::G_FCMP)
1740 return selectCompareBranchFedByFCmp(I, *CCMI, MIB);
1741 if (CCMIOpc == TargetOpcode::G_ICMP)
1742 return selectCompareBranchFedByICmp(I, *CCMI, MIB);
1743
1744 // Speculation tracking/SLH assumes that optimized TB(N)Z/CB(N)Z
1745 // instructions will not be produced, as they are conditional branch
1746 // instructions that do not set flags.
1747 if (ProduceNonFlagSettingCondBr) {
1748 emitTestBit(CondReg, /*Bit = */ 0, /*IsNegative = */ true,
1749 I.getOperand(1).getMBB(), MIB);
1750 I.eraseFromParent();
1751 return true;
1752 }
1753
1754 // Can't emit TB(N)Z/CB(N)Z. Emit a tst + bcc instead.
1755 auto TstMI =
1756 MIB.buildInstr(AArch64::ANDSWri, {LLT::scalar(32)}, {CondReg}).addImm(1);
1758 auto Bcc = MIB.buildInstr(AArch64::Bcc)
1760 .addMBB(I.getOperand(1).getMBB());
1761 I.eraseFromParent();
1763 return true;
1764}
1765
1766/// Returns the element immediate value of a vector shift operand if found.
1767/// This needs to detect a splat-like operation, e.g. a G_BUILD_VECTOR.
1768static std::optional<int64_t> getVectorShiftImm(Register Reg,
1769 MachineRegisterInfo &MRI) {
1770 assert(MRI.getType(Reg).isVector() && "Expected a *vector* shift operand");
1771 MachineInstr *OpMI = MRI.getVRegDef(Reg);
1772 return getAArch64VectorSplatScalar(*OpMI, MRI);
1773}
1774
1775/// Matches and returns the shift immediate value for a SHL instruction given
1776/// a shift operand.
1777static std::optional<int64_t> getVectorSHLImm(LLT SrcTy, Register Reg,
1778 MachineRegisterInfo &MRI) {
1779 std::optional<int64_t> ShiftImm = getVectorShiftImm(Reg, MRI);
1780 if (!ShiftImm)
1781 return std::nullopt;
1782 // Check the immediate is in range for a SHL.
1783 int64_t Imm = *ShiftImm;
1784 if (Imm < 0)
1785 return std::nullopt;
1786 switch (SrcTy.getElementType().getSizeInBits()) {
1787 default:
1788 LLVM_DEBUG(dbgs() << "Unhandled element type for vector shift");
1789 return std::nullopt;
1790 case 8:
1791 if (Imm > 7)
1792 return std::nullopt;
1793 break;
1794 case 16:
1795 if (Imm > 15)
1796 return std::nullopt;
1797 break;
1798 case 32:
1799 if (Imm > 31)
1800 return std::nullopt;
1801 break;
1802 case 64:
1803 if (Imm > 63)
1804 return std::nullopt;
1805 break;
1806 }
1807 return Imm;
1808}
1809
1810bool AArch64InstructionSelector::selectVectorSHL(MachineInstr &I,
1811 MachineRegisterInfo &MRI) {
1812 assert(I.getOpcode() == TargetOpcode::G_SHL);
1813 Register DstReg = I.getOperand(0).getReg();
1814 const LLT Ty = MRI.getType(DstReg);
1815 Register Src1Reg = I.getOperand(1).getReg();
1816 Register Src2Reg = I.getOperand(2).getReg();
1817
1818 if (!Ty.isVector())
1819 return false;
1820
1821 // Check if we have a vector of constants on RHS that we can select as the
1822 // immediate form.
1823 std::optional<int64_t> ImmVal = getVectorSHLImm(Ty, Src2Reg, MRI);
1824
1825 unsigned Opc = 0;
1826 if (Ty == LLT::fixed_vector(2, 64)) {
1827 Opc = ImmVal ? AArch64::SHLv2i64_shift : AArch64::USHLv2i64;
1828 } else if (Ty == LLT::fixed_vector(4, 32)) {
1829 Opc = ImmVal ? AArch64::SHLv4i32_shift : AArch64::USHLv4i32;
1830 } else if (Ty == LLT::fixed_vector(2, 32)) {
1831 Opc = ImmVal ? AArch64::SHLv2i32_shift : AArch64::USHLv2i32;
1832 } else if (Ty == LLT::fixed_vector(4, 16)) {
1833 Opc = ImmVal ? AArch64::SHLv4i16_shift : AArch64::USHLv4i16;
1834 } else if (Ty == LLT::fixed_vector(8, 16)) {
1835 Opc = ImmVal ? AArch64::SHLv8i16_shift : AArch64::USHLv8i16;
1836 } else if (Ty == LLT::fixed_vector(16, 8)) {
1837 Opc = ImmVal ? AArch64::SHLv16i8_shift : AArch64::USHLv16i8;
1838 } else if (Ty == LLT::fixed_vector(8, 8)) {
1839 Opc = ImmVal ? AArch64::SHLv8i8_shift : AArch64::USHLv8i8;
1840 } else {
1841 LLVM_DEBUG(dbgs() << "Unhandled G_SHL type");
1842 return false;
1843 }
1844
1845 auto Shl = MIB.buildInstr(Opc, {DstReg}, {Src1Reg});
1846 if (ImmVal)
1847 Shl.addImm(*ImmVal);
1848 else
1849 Shl.addUse(Src2Reg);
1851 I.eraseFromParent();
1852 return true;
1853}
1854
1855bool AArch64InstructionSelector::selectVectorAshrLshr(
1856 MachineInstr &I, MachineRegisterInfo &MRI) {
1857 assert(I.getOpcode() == TargetOpcode::G_ASHR ||
1858 I.getOpcode() == TargetOpcode::G_LSHR);
1859 Register DstReg = I.getOperand(0).getReg();
1860 const LLT Ty = MRI.getType(DstReg);
1861 Register Src1Reg = I.getOperand(1).getReg();
1862 Register Src2Reg = I.getOperand(2).getReg();
1863
1864 if (!Ty.isVector())
1865 return false;
1866
1867 bool IsASHR = I.getOpcode() == TargetOpcode::G_ASHR;
1868
1869 // We expect the immediate case to be lowered in the PostLegalCombiner to
1870 // AArch64ISD::VASHR or AArch64ISD::VLSHR equivalents.
1871
1872 // There is not a shift right register instruction, but the shift left
1873 // register instruction takes a signed value, where negative numbers specify a
1874 // right shift.
1875
1876 unsigned Opc = 0;
1877 unsigned NegOpc = 0;
1878 const TargetRegisterClass *RC =
1879 getRegClassForTypeOnBank(Ty, RBI.getRegBank(AArch64::FPRRegBankID));
1880 if (Ty == LLT::fixed_vector(2, 64)) {
1881 Opc = IsASHR ? AArch64::SSHLv2i64 : AArch64::USHLv2i64;
1882 NegOpc = AArch64::NEGv2i64;
1883 } else if (Ty == LLT::fixed_vector(4, 32)) {
1884 Opc = IsASHR ? AArch64::SSHLv4i32 : AArch64::USHLv4i32;
1885 NegOpc = AArch64::NEGv4i32;
1886 } else if (Ty == LLT::fixed_vector(2, 32)) {
1887 Opc = IsASHR ? AArch64::SSHLv2i32 : AArch64::USHLv2i32;
1888 NegOpc = AArch64::NEGv2i32;
1889 } else if (Ty == LLT::fixed_vector(4, 16)) {
1890 Opc = IsASHR ? AArch64::SSHLv4i16 : AArch64::USHLv4i16;
1891 NegOpc = AArch64::NEGv4i16;
1892 } else if (Ty == LLT::fixed_vector(8, 16)) {
1893 Opc = IsASHR ? AArch64::SSHLv8i16 : AArch64::USHLv8i16;
1894 NegOpc = AArch64::NEGv8i16;
1895 } else if (Ty == LLT::fixed_vector(16, 8)) {
1896 Opc = IsASHR ? AArch64::SSHLv16i8 : AArch64::USHLv16i8;
1897 NegOpc = AArch64::NEGv16i8;
1898 } else if (Ty == LLT::fixed_vector(8, 8)) {
1899 Opc = IsASHR ? AArch64::SSHLv8i8 : AArch64::USHLv8i8;
1900 NegOpc = AArch64::NEGv8i8;
1901 } else {
1902 LLVM_DEBUG(dbgs() << "Unhandled G_ASHR type");
1903 return false;
1904 }
1905
1906 auto Neg = MIB.buildInstr(NegOpc, {RC}, {Src2Reg});
1908 auto SShl = MIB.buildInstr(Opc, {DstReg}, {Src1Reg, Neg});
1910 I.eraseFromParent();
1911 return true;
1912}
1913
1914bool AArch64InstructionSelector::selectVaStartAAPCS(
1915 MachineInstr &I, MachineFunction &MF, MachineRegisterInfo &MRI) const {
1916
1918 MF.getFunction().isVarArg()))
1919 return false;
1920
1921 // The layout of the va_list struct is specified in the AArch64 Procedure Call
1922 // Standard, section 10.1.5.
1923
1924 const AArch64FunctionInfo *FuncInfo = MF.getInfo<AArch64FunctionInfo>();
1925 const unsigned PtrSize = STI.isTargetILP32() ? 4 : 8;
1926 const auto *PtrRegClass =
1927 STI.isTargetILP32() ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
1928
1929 const MCInstrDesc &MCIDAddAddr =
1930 TII.get(STI.isTargetILP32() ? AArch64::ADDWri : AArch64::ADDXri);
1931 const MCInstrDesc &MCIDStoreAddr =
1932 TII.get(STI.isTargetILP32() ? AArch64::STRWui : AArch64::STRXui);
1933
1934 /*
1935 * typedef struct va_list {
1936 * void * stack; // next stack param
1937 * void * gr_top; // end of GP arg reg save area
1938 * void * vr_top; // end of FP/SIMD arg reg save area
1939 * int gr_offs; // offset from gr_top to next GP register arg
1940 * int vr_offs; // offset from vr_top to next FP/SIMD register arg
1941 * } va_list;
1942 */
1943 const auto VAList = I.getOperand(0).getReg();
1944
1945 // Our current offset in bytes from the va_list struct (VAList).
1946 unsigned OffsetBytes = 0;
1947
1948 // Helper function to store (FrameIndex + Imm) to VAList at offset OffsetBytes
1949 // and increment OffsetBytes by PtrSize.
1950 const auto PushAddress = [&](const int FrameIndex, const int64_t Imm) {
1951 const Register Top = MRI.createVirtualRegister(PtrRegClass);
1952 auto MIB = BuildMI(*I.getParent(), I, I.getDebugLoc(), MCIDAddAddr)
1953 .addDef(Top)
1954 .addFrameIndex(FrameIndex)
1955 .addImm(Imm)
1956 .addImm(0);
1958
1959 const auto *MMO = *I.memoperands_begin();
1960 MIB = BuildMI(*I.getParent(), I, I.getDebugLoc(), MCIDStoreAddr)
1961 .addUse(Top)
1962 .addUse(VAList)
1963 .addImm(OffsetBytes / PtrSize)
1965 MMO->getPointerInfo().getWithOffset(OffsetBytes),
1966 MachineMemOperand::MOStore, PtrSize, MMO->getBaseAlign()));
1968
1969 OffsetBytes += PtrSize;
1970 };
1971
1972 // void* stack at offset 0
1973 PushAddress(FuncInfo->getVarArgsStackIndex(), 0);
1974
1975 // void* gr_top at offset 8 (4 on ILP32)
1976 const unsigned GPRSize = FuncInfo->getVarArgsGPRSize();
1977 PushAddress(FuncInfo->getVarArgsGPRIndex(), GPRSize);
1978
1979 // void* vr_top at offset 16 (8 on ILP32)
1980 const unsigned FPRSize = FuncInfo->getVarArgsFPRSize();
1981 PushAddress(FuncInfo->getVarArgsFPRIndex(), FPRSize);
1982
1983 // Helper function to store a 4-byte integer constant to VAList at offset
1984 // OffsetBytes, and increment OffsetBytes by 4.
1985 const auto PushIntConstant = [&](const int32_t Value) {
1986 constexpr int IntSize = 4;
1987 const Register Temp = MRI.createVirtualRegister(&AArch64::GPR32RegClass);
1988 auto MIB =
1989 BuildMI(*I.getParent(), I, I.getDebugLoc(), TII.get(AArch64::MOVi32imm))
1990 .addDef(Temp)
1991 .addImm(Value);
1993
1994 const auto *MMO = *I.memoperands_begin();
1995 MIB = BuildMI(*I.getParent(), I, I.getDebugLoc(), TII.get(AArch64::STRWui))
1996 .addUse(Temp)
1997 .addUse(VAList)
1998 .addImm(OffsetBytes / IntSize)
2000 MMO->getPointerInfo().getWithOffset(OffsetBytes),
2001 MachineMemOperand::MOStore, IntSize, MMO->getBaseAlign()));
2003 OffsetBytes += IntSize;
2004 };
2005
2006 // int gr_offs at offset 24 (12 on ILP32)
2007 PushIntConstant(-static_cast<int32_t>(GPRSize));
2008
2009 // int vr_offs at offset 28 (16 on ILP32)
2010 PushIntConstant(-static_cast<int32_t>(FPRSize));
2011
2012 assert(OffsetBytes == (STI.isTargetILP32() ? 20 : 32) && "Unexpected offset");
2013
2014 I.eraseFromParent();
2015 return true;
2016}
2017
2018bool AArch64InstructionSelector::selectVaStartDarwin(
2019 MachineInstr &I, MachineFunction &MF, MachineRegisterInfo &MRI) const {
2020 AArch64FunctionInfo *FuncInfo = MF.getInfo<AArch64FunctionInfo>();
2021 Register ListReg = I.getOperand(0).getReg();
2022
2023 Register ArgsAddrReg = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
2024
2025 int FrameIdx = FuncInfo->getVarArgsStackIndex();
2026 if (MF.getSubtarget<AArch64Subtarget>().isCallingConvWin64(
2028 FrameIdx = FuncInfo->getVarArgsGPRSize() > 0
2029 ? FuncInfo->getVarArgsGPRIndex()
2030 : FuncInfo->getVarArgsStackIndex();
2031 }
2032
2033 auto MIB =
2034 BuildMI(*I.getParent(), I, I.getDebugLoc(), TII.get(AArch64::ADDXri))
2035 .addDef(ArgsAddrReg)
2036 .addFrameIndex(FrameIdx)
2037 .addImm(0)
2038 .addImm(0);
2039
2041
2042 MIB = BuildMI(*I.getParent(), I, I.getDebugLoc(), TII.get(AArch64::STRXui))
2043 .addUse(ArgsAddrReg)
2044 .addUse(ListReg)
2045 .addImm(0)
2046 .addMemOperand(*I.memoperands_begin());
2047
2049 I.eraseFromParent();
2050 return true;
2051}
2052
2053void AArch64InstructionSelector::materializeLargeCMVal(
2054 MachineInstr &I, const Value *V, unsigned OpFlags) {
2055 MachineBasicBlock &MBB = *I.getParent();
2056 MachineFunction &MF = *MBB.getParent();
2057 MachineRegisterInfo &MRI = MF.getRegInfo();
2058
2059 auto MovZ = MIB.buildInstr(AArch64::MOVZXi, {&AArch64::GPR64RegClass}, {});
2060 MovZ->addOperand(MF, I.getOperand(1));
2061 MovZ->getOperand(1).setTargetFlags(OpFlags | AArch64II::MO_G0 |
2063 MovZ->addOperand(MF, MachineOperand::CreateImm(0));
2065
2066 auto BuildMovK = [&](Register SrcReg, unsigned char Flags, unsigned Offset,
2067 Register ForceDstReg) {
2068 Register DstReg = ForceDstReg
2069 ? ForceDstReg
2070 : MRI.createVirtualRegister(&AArch64::GPR64RegClass);
2071 auto MovI = MIB.buildInstr(AArch64::MOVKXi).addDef(DstReg).addUse(SrcReg);
2072 if (auto *GV = dyn_cast<GlobalValue>(V)) {
2073 MovI->addOperand(MF, MachineOperand::CreateGA(
2074 GV, MovZ->getOperand(1).getOffset(), Flags));
2075 } else {
2076 MovI->addOperand(
2078 MovZ->getOperand(1).getOffset(), Flags));
2079 }
2082 return DstReg;
2083 };
2084 Register DstReg = BuildMovK(MovZ.getReg(0),
2086 DstReg = BuildMovK(DstReg, AArch64II::MO_G2 | AArch64II::MO_NC, 32, 0);
2087 BuildMovK(DstReg, AArch64II::MO_G3, 48, I.getOperand(0).getReg());
2088}
2089
2090bool AArch64InstructionSelector::preISelLower(MachineInstr &I) {
2091 MachineBasicBlock &MBB = *I.getParent();
2092 MachineFunction &MF = *MBB.getParent();
2093 MachineRegisterInfo &MRI = MF.getRegInfo();
2094
2095 switch (I.getOpcode()) {
2096 case TargetOpcode::G_CONSTANT: {
2097 Register DefReg = I.getOperand(0).getReg();
2098 const LLT DefTy = MRI.getType(DefReg);
2099 if (!DefTy.isPointer()) {
2100 if (DefTy.getSizeInBits() >= 32 ||
2101 RBI.getRegBank(DefReg, MRI, TRI)->getID() != AArch64::GPRRegBankID)
2102 return false;
2103 // Widen narrow GPR constants to s32 so imported patterns can match.
2104 APInt Val = I.getOperand(1).getCImm()->getValue().zext(32);
2105 I.getOperand(1).setCImm(
2106 ConstantInt::get(MF.getFunction().getContext(), Val));
2107
2109 MRI.setRegBank(WideReg, RBI.getRegBank(AArch64::GPRRegBankID));
2110 I.getOperand(0).setReg(WideReg);
2111
2112 MIB.setInsertPt(MBB, std::next(I.getIterator()));
2113 auto Copy = MIB.buildCopy(DefReg, WideReg);
2114 selectCopy(*Copy, TII, MRI, TRI, RBI);
2115 MIB.setInstr(I);
2116 return true;
2117 }
2118 const unsigned PtrSize = DefTy.getSizeInBits();
2119 if (PtrSize != 32 && PtrSize != 64)
2120 return false;
2121 // Convert pointer typed constants to integers so TableGen can select.
2122 MRI.setType(DefReg, LLT::integer(PtrSize));
2123 return true;
2124 }
2125 case TargetOpcode::G_STORE: {
2126 bool Changed = contractCrossBankCopyIntoStore(I, MRI);
2127 MachineOperand &SrcOp = I.getOperand(0);
2128 if (MRI.getType(SrcOp.getReg()).isPointer()) {
2129 // Allow matching with imported patterns for stores of pointers. Unlike
2130 // G_LOAD/G_PTR_ADD, we may not have selected all users. So, emit a copy
2131 // and constrain.
2132 auto Copy = MIB.buildCopy(LLT::scalar(64), SrcOp);
2133 Register NewSrc = Copy.getReg(0);
2134 SrcOp.setReg(NewSrc);
2135 RBI.constrainGenericRegister(NewSrc, AArch64::GPR64RegClass, MRI);
2136 Changed = true;
2137 }
2138 return Changed;
2139 }
2140 case TargetOpcode::G_PTR_ADD: {
2141 // If Checked Pointer Arithmetic (FEAT_CPA) is present, preserve the pointer
2142 // arithmetic semantics instead of falling back to regular arithmetic.
2143 const auto &TL = STI.getTargetLowering();
2144 if (TL->shouldPreservePtrArith(MF.getFunction(), EVT()))
2145 return false;
2146 return convertPtrAddToAdd(I, MRI);
2147 }
2148 case TargetOpcode::G_LOAD: {
2149 // For scalar loads of pointers, we try to convert the dest type from p0
2150 // to s64 so that our imported patterns can match. Like with the G_PTR_ADD
2151 // conversion, this should be ok because all users should have been
2152 // selected already, so the type doesn't matter for them.
2153 Register DstReg = I.getOperand(0).getReg();
2154 const LLT DstTy = MRI.getType(DstReg);
2155 if (!DstTy.isPointer())
2156 return false;
2157 MRI.setType(DstReg, LLT::scalar(64));
2158 return true;
2159 }
2160 case TargetOpcode::G_VECREDUCE_ADD:
2161 case TargetOpcode::G_VECREDUCE_SMAX:
2162 case TargetOpcode::G_VECREDUCE_SMIN:
2163 case TargetOpcode::G_VECREDUCE_UMAX:
2164 case TargetOpcode::G_VECREDUCE_UMIN: {
2165 // Imported patterns require an FPR result. For a GPR, use a temporary FPR
2166 // and insert a cross-bank copy.
2167 Register DstReg = I.getOperand(0).getReg();
2168 const RegisterBank &DstRB = *RBI.getRegBank(DstReg, MRI, TRI);
2169 if (DstRB.getID() != AArch64::GPRRegBankID)
2170 return false;
2171
2172 LLT DstTy = MRI.getType(DstReg);
2173 const TargetRegisterClass *DstRC =
2174 getRegClassForTypeOnBank(DstTy, DstRB, /*GetAllRegSet=*/true);
2175 if (!DstRC || !RBI.constrainGenericRegister(DstReg, *DstRC, MRI))
2176 return false;
2177
2178 Register FPRDst = MRI.createGenericVirtualRegister(DstTy);
2179 MRI.setRegBank(FPRDst, RBI.getRegBank(AArch64::FPRRegBankID));
2180 I.getOperand(0).setReg(FPRDst);
2181
2182 BuildMI(MBB, std::next(I.getIterator()), MIMetadata(I),
2183 TII.get(TargetOpcode::COPY), DstReg)
2184 .addReg(FPRDst);
2185 return true;
2186 }
2187 case AArch64::G_DUP: {
2188 // Convert the type from p0 to s64 to help selection.
2189 LLT DstTy = MRI.getType(I.getOperand(0).getReg());
2190 if (!DstTy.isPointerVector())
2191 return false;
2192 auto NewSrc = MIB.buildCopy(LLT::scalar(64), I.getOperand(1).getReg());
2193 MRI.setType(I.getOperand(0).getReg(),
2194 DstTy.changeElementType(LLT::scalar(64)));
2195 MRI.setRegClass(NewSrc.getReg(0), &AArch64::GPR64RegClass);
2196 I.getOperand(1).setReg(NewSrc.getReg(0));
2197 return true;
2198 }
2199 case AArch64::G_INSERT_VECTOR_ELT: {
2200 LLT DstTy = MRI.getType(I.getOperand(0).getReg());
2201 LLT SrcVecTy = MRI.getType(I.getOperand(1).getReg());
2202 if (SrcVecTy.isPointerVector()) {
2203 // Convert the type from p0 to s64 to help selection.
2204 auto NewSrc = MIB.buildCopy(LLT::scalar(64), I.getOperand(2).getReg());
2205 MRI.setType(I.getOperand(1).getReg(),
2206 DstTy.changeElementType(LLT::scalar(64)));
2207 MRI.setType(I.getOperand(0).getReg(),
2208 DstTy.changeElementType(LLT::scalar(64)));
2209 MRI.setRegClass(NewSrc.getReg(0), &AArch64::GPR64RegClass);
2210 I.getOperand(2).setReg(NewSrc.getReg(0));
2211 return true;
2212 }
2213
2214 Register EltReg = I.getOperand(2).getReg();
2215 LLT EltTy = MRI.getType(EltReg);
2216 if (EltTy.isScalar() &&
2217 (EltTy.getSizeInBits() == 8 || EltTy.getSizeInBits() == 16) &&
2218 RBI.getRegBank(EltReg, MRI, TRI)->getID() == AArch64::GPRRegBankID) {
2219 // Convert the type from s8/s16 to s32 to help selection.
2220 auto NewElt = MIB.buildCopy(LLT::scalar(32), EltReg);
2221 MRI.setRegClass(NewElt.getReg(0), &AArch64::GPR32RegClass);
2222 I.getOperand(2).setReg(NewElt.getReg(0));
2223 return true;
2224 }
2225 return false;
2226 }
2227 case TargetOpcode::G_UITOFP:
2228 case TargetOpcode::G_SITOFP: {
2229 // If both source and destination regbanks are FPR, then convert the opcode
2230 // to G_SITOF so that the importer can select it to an fpr variant.
2231 // Otherwise, it ends up matching an fpr/gpr variant and adding a cross-bank
2232 // copy.
2233 Register SrcReg = I.getOperand(1).getReg();
2234 LLT SrcTy = MRI.getType(SrcReg);
2235 LLT DstTy = MRI.getType(I.getOperand(0).getReg());
2236 if (SrcTy.isVector() || SrcTy.getSizeInBits() != DstTy.getSizeInBits())
2237 return false;
2238
2239 if (RBI.getRegBank(SrcReg, MRI, TRI)->getID() == AArch64::FPRRegBankID) {
2240 // Need to add a copy to change the type so that the existing patterns can
2241 // match when there is an integer on an FPR bank.
2242 if (SrcTy.getScalarType().isInteger()) {
2243 auto Copy = MIB.buildCopy(DstTy, SrcReg);
2244 I.getOperand(1).setReg(Copy.getReg(0));
2245 MRI.setRegClass(Copy.getReg(0),
2246 getRegClassForTypeOnBank(
2247 SrcTy, RBI.getRegBank(AArch64::FPRRegBankID)));
2248 }
2249 if (I.getOpcode() == TargetOpcode::G_SITOFP)
2250 I.setDesc(TII.get(AArch64::G_SITOF));
2251 else
2252 I.setDesc(TII.get(AArch64::G_UITOF));
2253 return true;
2254 }
2255 return false;
2256 }
2257 default:
2258 return false;
2259 }
2260}
2261
2262/// This lowering tries to look for G_PTR_ADD instructions and then converts
2263/// them to a standard G_ADD with a COPY on the source.
2264///
2265/// The motivation behind this is to expose the add semantics to the imported
2266/// tablegen patterns. We shouldn't need to check for uses being loads/stores,
2267/// because the selector works bottom up, uses before defs. By the time we
2268/// end up trying to select a G_PTR_ADD, we should have already attempted to
2269/// fold this into addressing modes and were therefore unsuccessful.
2270bool AArch64InstructionSelector::convertPtrAddToAdd(
2271 MachineInstr &I, MachineRegisterInfo &MRI) {
2272 assert(I.getOpcode() == TargetOpcode::G_PTR_ADD && "Expected G_PTR_ADD");
2273 Register DstReg = I.getOperand(0).getReg();
2274 Register AddOp1Reg = I.getOperand(1).getReg();
2275 const LLT PtrTy = MRI.getType(DstReg);
2276 if (PtrTy.getAddressSpace() != 0)
2277 return false;
2278
2279 const LLT CastPtrTy = PtrTy.isVector()
2281 : LLT::integer(64);
2282 auto PtrToInt = MIB.buildPtrToInt(CastPtrTy, AddOp1Reg);
2283 // Set regbanks on the registers.
2284 if (PtrTy.isVector())
2285 MRI.setRegBank(PtrToInt.getReg(0), RBI.getRegBank(AArch64::FPRRegBankID));
2286 else
2287 MRI.setRegBank(PtrToInt.getReg(0), RBI.getRegBank(AArch64::GPRRegBankID));
2288
2289 // Now turn the %dst(p0) = G_PTR_ADD %base, off into:
2290 // %dst(intty) = G_ADD %intbase, off
2291 I.setDesc(TII.get(TargetOpcode::G_ADD));
2292 MRI.setType(DstReg, CastPtrTy);
2293 I.getOperand(1).setReg(PtrToInt.getReg(0));
2294 if (!select(*PtrToInt)) {
2295 LLVM_DEBUG(dbgs() << "Failed to select G_PTRTOINT in convertPtrAddToAdd");
2296 return false;
2297 }
2298
2299 // Also take the opportunity here to try to do some optimization.
2300 // Try to convert this into a G_SUB if the offset is a 0-x negate idiom.
2301 Register NegatedReg;
2302 if (!mi_match(I.getOperand(2).getReg(), MRI, m_Neg(m_Reg(NegatedReg))))
2303 return true;
2304 I.getOperand(2).setReg(NegatedReg);
2305 I.setDesc(TII.get(TargetOpcode::G_SUB));
2306 return true;
2307}
2308
2309bool AArch64InstructionSelector::earlySelectSHL(MachineInstr &I,
2310 MachineRegisterInfo &MRI) {
2311 // We try to match the immediate variant of LSL, which is actually an alias
2312 // for a special case of UBFM. Otherwise, we fall back to the imported
2313 // selector which will match the register variant.
2314 assert(I.getOpcode() == TargetOpcode::G_SHL && "unexpected op");
2315 const auto &MO = I.getOperand(2);
2316 auto VRegAndVal = getIConstantVRegVal(MO.getReg(), MRI);
2317 if (!VRegAndVal)
2318 return false;
2319
2320 const LLT DstTy = MRI.getType(I.getOperand(0).getReg());
2321 if (DstTy.isVector())
2322 return false;
2323 bool Is64Bit = DstTy.getSizeInBits() == 64;
2324 auto Imm1Fn = Is64Bit ? selectShiftA_64(MO) : selectShiftA_32(MO);
2325 auto Imm2Fn = Is64Bit ? selectShiftB_64(MO) : selectShiftB_32(MO);
2326
2327 if (!Imm1Fn || !Imm2Fn)
2328 return false;
2329
2330 auto NewI =
2331 MIB.buildInstr(Is64Bit ? AArch64::UBFMXri : AArch64::UBFMWri,
2332 {I.getOperand(0).getReg()}, {I.getOperand(1).getReg()});
2333
2334 for (auto &RenderFn : *Imm1Fn)
2335 RenderFn(NewI);
2336 for (auto &RenderFn : *Imm2Fn)
2337 RenderFn(NewI);
2338
2339 I.eraseFromParent();
2341 return true;
2342}
2343
2344bool AArch64InstructionSelector::contractCrossBankCopyIntoStore(
2345 MachineInstr &I, MachineRegisterInfo &MRI) {
2346 assert(I.getOpcode() == TargetOpcode::G_STORE && "Expected G_STORE");
2347 // If we're storing a scalar, it doesn't matter what register bank that
2348 // scalar is on. All that matters is the size.
2349 //
2350 // So, if we see something like this (with a 32-bit scalar as an example):
2351 //
2352 // %x:gpr(s32) = ... something ...
2353 // %y:fpr(s32) = COPY %x:gpr(s32)
2354 // G_STORE %y:fpr(s32)
2355 //
2356 // We can fix this up into something like this:
2357 //
2358 // G_STORE %x:gpr(s32)
2359 //
2360 // And then continue the selection process normally.
2361 Register DefDstReg = getSrcRegIgnoringCopies(I.getOperand(0).getReg(), MRI);
2362 if (!DefDstReg.isValid())
2363 return false;
2364 LLT DefDstTy = MRI.getType(DefDstReg);
2365 Register StoreSrcReg = I.getOperand(0).getReg();
2366 LLT StoreSrcTy = MRI.getType(StoreSrcReg);
2367
2368 // If we get something strange like a physical register, then we shouldn't
2369 // go any further.
2370 if (!DefDstTy.isValid())
2371 return false;
2372
2373 // Are the source and dst types the same size?
2374 if (DefDstTy.getSizeInBits() != StoreSrcTy.getSizeInBits())
2375 return false;
2376
2377 if (RBI.getRegBank(StoreSrcReg, MRI, TRI) ==
2378 RBI.getRegBank(DefDstReg, MRI, TRI))
2379 return false;
2380
2381 // We have a cross-bank copy, which is entering a store. Let's fold it.
2382 I.getOperand(0).setReg(DefDstReg);
2383 return true;
2384}
2385
2386bool AArch64InstructionSelector::earlySelect(MachineInstr &I) {
2387 assert(I.getParent() && "Instruction should be in a basic block!");
2388 assert(I.getParent()->getParent() && "Instruction should be in a function!");
2389
2390 MachineBasicBlock &MBB = *I.getParent();
2391 MachineFunction &MF = *MBB.getParent();
2392 MachineRegisterInfo &MRI = MF.getRegInfo();
2393
2394 switch (I.getOpcode()) {
2395 case AArch64::G_DUP: {
2396 // Before selecting a DUP instruction, check if it is better selected as a
2397 // MOV or load from a constant pool.
2398 Register Src = I.getOperand(1).getReg();
2399 auto ValAndVReg = getAnyConstantVRegValWithLookThrough(
2400 Src, MRI, /*LookThroughInstrs=*/true, /*LookThroughAnyExt=*/true);
2401 if (!ValAndVReg)
2402 return false;
2403 LLVMContext &Ctx = MF.getFunction().getContext();
2404 Register Dst = I.getOperand(0).getReg();
2406 MRI.getType(Dst).getNumElements(),
2407 ConstantInt::get(
2408 Type::getIntNTy(Ctx, MRI.getType(Dst).getScalarSizeInBits()),
2409 ValAndVReg->Value.trunc(MRI.getType(Dst).getScalarSizeInBits())));
2410 if (!emitConstantVector(Dst, CV, MIB, MRI))
2411 return false;
2412 I.eraseFromParent();
2413 return true;
2414 }
2415 case TargetOpcode::G_SEXT:
2416 // Check for i64 sext(i32 vector_extract) prior to tablegen to select SMOV
2417 // over a normal extend.
2418 if (selectUSMovFromExtend(I, MRI))
2419 return true;
2420 return false;
2421 case TargetOpcode::G_BR:
2422 return false;
2423 case TargetOpcode::G_SHL:
2424 return earlySelectSHL(I, MRI);
2425 case TargetOpcode::G_CONSTANT: {
2426 bool IsZero = false;
2427 if (I.getOperand(1).isCImm())
2428 IsZero = I.getOperand(1).getCImm()->isZero();
2429 else if (I.getOperand(1).isImm())
2430 IsZero = I.getOperand(1).getImm() == 0;
2431
2432 if (!IsZero)
2433 return false;
2434
2435 Register DefReg = I.getOperand(0).getReg();
2436 LLT Ty = MRI.getType(DefReg);
2437 if (Ty.getSizeInBits() == 64) {
2438 I.getOperand(1).ChangeToRegister(AArch64::XZR, false);
2439 RBI.constrainGenericRegister(DefReg, AArch64::GPR64RegClass, MRI);
2440 } else if (Ty.getSizeInBits() <= 32) {
2441 I.getOperand(1).ChangeToRegister(AArch64::WZR, false);
2442 RBI.constrainGenericRegister(DefReg, AArch64::GPR32RegClass, MRI);
2443 } else
2444 return false;
2445
2446 I.setDesc(TII.get(TargetOpcode::COPY));
2447 return true;
2448 }
2449
2450 case TargetOpcode::G_ADD: {
2451 // Check if this is being fed by a G_ICMP on either side.
2452 //
2453 // (cmp pred, x, y) + z
2454 //
2455 // In the above case, when the cmp is true, we increment z by 1. So, we can
2456 // fold the add into the cset for the cmp by using cinc.
2457 //
2458 // FIXME: This would probably be a lot nicer in PostLegalizerLowering.
2459 Register AddDst = I.getOperand(0).getReg();
2460 Register AddLHS = I.getOperand(1).getReg();
2461 Register AddRHS = I.getOperand(2).getReg();
2462 // Only handle scalars.
2463 LLT Ty = MRI.getType(AddLHS);
2464 if (Ty.isVector())
2465 return false;
2466 // Since G_ICMP is modeled as ADDS/SUBS/ANDS, we can handle 32 bits or 64
2467 // bits.
2468 unsigned Size = Ty.getSizeInBits();
2469 if (Size != 32 && Size != 64)
2470 return false;
2471 auto MatchCmp = [&](Register Reg) -> MachineInstr * {
2472 if (!MRI.hasOneNonDBGUse(Reg))
2473 return nullptr;
2474 // If the LHS of the add is 32 bits, then we want to fold a 32-bit
2475 // compare.
2476 if (Size == 32)
2477 return getOpcodeDef(TargetOpcode::G_ICMP, Reg, MRI);
2478 // We model scalar compares using 32-bit destinations right now.
2479 // If it's a 64-bit compare, it'll have 64-bit sources.
2480 Register ZExt;
2481 if (!mi_match(Reg, MRI,
2483 return nullptr;
2484 auto *Cmp = getOpcodeDef(TargetOpcode::G_ICMP, ZExt, MRI);
2485 if (!Cmp ||
2486 MRI.getType(Cmp->getOperand(2).getReg()).getSizeInBits() != 64)
2487 return nullptr;
2488 return Cmp;
2489 };
2490 // Try to match
2491 // z + (cmp pred, x, y)
2492 MachineInstr *Cmp = MatchCmp(AddRHS);
2493 if (!Cmp) {
2494 // (cmp pred, x, y) + z
2495 std::swap(AddLHS, AddRHS);
2496 Cmp = MatchCmp(AddRHS);
2497 if (!Cmp)
2498 return false;
2499 }
2500 auto &PredOp = Cmp->getOperand(1);
2502 emitIntegerCompare(/*LHS=*/Cmp->getOperand(2),
2503 /*RHS=*/Cmp->getOperand(3), PredOp, MIB);
2504 auto Pred = static_cast<CmpInst::Predicate>(PredOp.getPredicate());
2506 CmpInst::getInversePredicate(Pred), Cmp->getOperand(3).getReg(), &MRI);
2507 emitCSINC(/*Dst=*/AddDst, /*Src =*/AddLHS, /*Src2=*/AddLHS, InvCC, MIB);
2508 I.eraseFromParent();
2509 return true;
2510 }
2511 case TargetOpcode::G_OR: {
2512 // Look for operations that take the lower `Width=Size-ShiftImm` bits of
2513 // `ShiftSrc` and insert them into the upper `Width` bits of `MaskSrc` via
2514 // shifting and masking that we can replace with a BFI (encoded as a BFM).
2515 Register Dst = I.getOperand(0).getReg();
2516 LLT Ty = MRI.getType(Dst);
2517
2518 if (!Ty.isScalar())
2519 return false;
2520
2521 unsigned Size = Ty.getSizeInBits();
2522 if (Size != 32 && Size != 64)
2523 return false;
2524
2525 Register ShiftSrc;
2526 int64_t ShiftImm;
2527 Register MaskSrc;
2528 int64_t MaskImm;
2529 if (!mi_match(
2530 Dst, MRI,
2531 m_GOr(m_OneNonDBGUse(m_GShl(m_Reg(ShiftSrc), m_ICst(ShiftImm))),
2532 m_OneNonDBGUse(m_GAnd(m_Reg(MaskSrc), m_ICst(MaskImm))))))
2533 return false;
2534
2535 if (ShiftImm > Size || ((1ULL << ShiftImm) - 1ULL) != uint64_t(MaskImm))
2536 return false;
2537
2538 int64_t Immr = Size - ShiftImm;
2539 int64_t Imms = Size - ShiftImm - 1;
2540 unsigned Opc = Size == 32 ? AArch64::BFMWri : AArch64::BFMXri;
2541 emitInstr(Opc, {Dst}, {MaskSrc, ShiftSrc, Immr, Imms}, MIB);
2542 I.eraseFromParent();
2543 return true;
2544 }
2545 case TargetOpcode::G_FENCE: {
2546 if (I.getOperand(1).getImm() == 0)
2547 BuildMI(MBB, I, MIMetadata(I), TII.get(TargetOpcode::MEMBARRIER));
2548 else
2549 BuildMI(MBB, I, MIMetadata(I), TII.get(AArch64::DMB))
2550 .addImm(I.getOperand(0).getImm() == 4 ? 0x9 : 0xb);
2551 I.eraseFromParent();
2552 return true;
2553 }
2554 default:
2555 return false;
2556 }
2557}
2558
2559bool AArch64InstructionSelector::select(MachineInstr &I) {
2560 assert(I.getParent() && "Instruction should be in a basic block!");
2561 assert(I.getParent()->getParent() && "Instruction should be in a function!");
2562
2563 MachineBasicBlock &MBB = *I.getParent();
2564 MachineFunction &MF = *MBB.getParent();
2565 MachineRegisterInfo &MRI = MF.getRegInfo();
2566
2567 const AArch64Subtarget *Subtarget = &MF.getSubtarget<AArch64Subtarget>();
2568 if (Subtarget->requiresStrictAlign()) {
2569 // We don't support this feature yet.
2570 LLVM_DEBUG(dbgs() << "AArch64 GISel does not support strict-align yet\n");
2571 return false;
2572 }
2573
2575
2576 unsigned Opcode = I.getOpcode();
2577 // G_PHI requires same handling as PHI
2578 if (!I.isPreISelOpcode() || Opcode == TargetOpcode::G_PHI) {
2579 // Certain non-generic instructions also need some special handling.
2580
2581 if (Opcode == TargetOpcode::LOAD_STACK_GUARD) {
2583 return true;
2584 }
2585
2586 if (Opcode == TargetOpcode::PHI || Opcode == TargetOpcode::G_PHI) {
2587 const Register DefReg = I.getOperand(0).getReg();
2588 const LLT DefTy = MRI.getType(DefReg);
2589
2590 const RegClassOrRegBank &RegClassOrBank =
2591 MRI.getRegClassOrRegBank(DefReg);
2592
2593 const TargetRegisterClass *DefRC =
2595 if (!DefRC) {
2596 if (!DefTy.isValid()) {
2597 LLVM_DEBUG(dbgs() << "PHI operand has no type, not a gvreg?\n");
2598 return false;
2599 }
2600 const RegisterBank &RB = *cast<const RegisterBank *>(RegClassOrBank);
2601 DefRC = getRegClassForTypeOnBank(DefTy, RB);
2602 if (!DefRC) {
2603 LLVM_DEBUG(dbgs() << "PHI operand has unexpected size/bank\n");
2604 return false;
2605 }
2606 }
2607
2608 I.setDesc(TII.get(TargetOpcode::PHI));
2609
2610 return RBI.constrainGenericRegister(DefReg, *DefRC, MRI);
2611 }
2612
2613 if (I.isCopy())
2614 return selectCopy(I, TII, MRI, TRI, RBI);
2615
2616 if (I.isDebugInstr())
2617 return selectDebugInstr(I, MRI, RBI);
2618
2619 return true;
2620 }
2621
2622
2623 if (I.getNumOperands() != I.getNumExplicitOperands()) {
2624 LLVM_DEBUG(
2625 dbgs() << "Generic instruction has unexpected implicit operands\n");
2626 return false;
2627 }
2628
2629 // Try to do some lowering before we start instruction selecting. These
2630 // lowerings are purely transformations on the input G_MIR and so selection
2631 // must continue after any modification of the instruction.
2632 if (preISelLower(I)) {
2633 Opcode = I.getOpcode(); // The opcode may have been modified, refresh it.
2634 }
2635
2636 // There may be patterns where the importer can't deal with them optimally,
2637 // but does select it to a suboptimal sequence so our custom C++ selection
2638 // code later never has a chance to work on it. Therefore, we have an early
2639 // selection attempt here to give priority to certain selection routines
2640 // over the imported ones.
2641 if (earlySelect(I))
2642 return true;
2643
2644 if (selectImpl(I, *CoverageInfo))
2645 return true;
2646
2647 LLT Ty =
2648 I.getOperand(0).isReg() ? MRI.getType(I.getOperand(0).getReg()) : LLT{};
2649
2650 switch (Opcode) {
2651 case TargetOpcode::G_SBFX:
2652 case TargetOpcode::G_UBFX: {
2653 static const unsigned OpcTable[2][2] = {
2654 {AArch64::UBFMWri, AArch64::UBFMXri},
2655 {AArch64::SBFMWri, AArch64::SBFMXri}};
2656 bool IsSigned = Opcode == TargetOpcode::G_SBFX;
2657 unsigned Size = Ty.getSizeInBits();
2658 unsigned Opc = OpcTable[IsSigned][Size == 64];
2659 auto Cst1 =
2660 getIConstantVRegValWithLookThrough(I.getOperand(2).getReg(), MRI);
2661 assert(Cst1 && "Should have gotten a constant for src 1?");
2662 auto Cst2 =
2663 getIConstantVRegValWithLookThrough(I.getOperand(3).getReg(), MRI);
2664 assert(Cst2 && "Should have gotten a constant for src 2?");
2665 auto LSB = Cst1->Value.getZExtValue();
2666 auto Width = Cst2->Value.getZExtValue();
2667 auto BitfieldInst =
2668 MIB.buildInstr(Opc, {I.getOperand(0)}, {I.getOperand(1)})
2669 .addImm(LSB)
2670 .addImm(LSB + Width - 1);
2671 I.eraseFromParent();
2672 constrainSelectedInstRegOperands(*BitfieldInst, TII, TRI, RBI);
2673 return true;
2674 }
2675 case TargetOpcode::G_BRCOND:
2676 return selectCompareBranch(I, MF, MRI);
2677
2678 case TargetOpcode::G_BRINDIRECT: {
2679 const Function &Fn = MF.getFunction();
2680 if (std::optional<uint16_t> BADisc =
2682 auto MI = MIB.buildInstr(AArch64::BRA, {}, {I.getOperand(0).getReg()});
2684 MI.addImm(*BADisc);
2685 MI.addReg(/*AddrDisc=*/AArch64::XZR);
2686 I.eraseFromParent();
2688 return true;
2689 }
2690 I.setDesc(TII.get(AArch64::BR));
2692 return true;
2693 }
2694
2695 case TargetOpcode::G_BRJT:
2696 return selectBrJT(I, MRI);
2697
2698 case AArch64::G_ADD_LOW: {
2699 // This op may have been separated from it's ADRP companion by the localizer
2700 // or some other code motion pass. Given that many CPUs will try to
2701 // macro fuse these operations anyway, select this into a MOVaddr pseudo
2702 // which will later be expanded into an ADRP+ADD pair after scheduling.
2703 MachineInstr *BaseMI = MRI.getVRegDef(I.getOperand(1).getReg());
2704 if (BaseMI->getOpcode() != AArch64::ADRP) {
2705 I.setDesc(TII.get(AArch64::ADDXri));
2706 I.addOperand(MachineOperand::CreateImm(0));
2708 return true;
2709 }
2711 "Expected small code model");
2712 auto Op1 = BaseMI->getOperand(1);
2713 auto Op2 = I.getOperand(2);
2714 auto MovAddr = MIB.buildInstr(AArch64::MOVaddr, {I.getOperand(0)}, {})
2715 .addGlobalAddress(Op1.getGlobal(), Op1.getOffset(),
2716 Op1.getTargetFlags())
2717 .addGlobalAddress(Op2.getGlobal(), Op2.getOffset(),
2718 Op2.getTargetFlags());
2719 I.eraseFromParent();
2720 constrainSelectedInstRegOperands(*MovAddr, TII, TRI, RBI);
2721 return true;
2722 }
2723
2724 case TargetOpcode::G_FCONSTANT: {
2725 const Register DefReg = I.getOperand(0).getReg();
2726 const LLT DefTy = MRI.getType(DefReg);
2727 const unsigned DefSize = DefTy.getSizeInBits();
2728 const RegisterBank &RB = *RBI.getRegBank(DefReg, MRI, TRI);
2729
2730 const TargetRegisterClass &FPRRC = *getRegClassForTypeOnBank(DefTy, RB);
2731 // For 16, 64, and 128b values, emit a constant pool load.
2732 switch (DefSize) {
2733 default:
2734 llvm_unreachable("Unexpected destination size for G_FCONSTANT?");
2735 case 32:
2736 case 64: {
2737 bool OptForSize = shouldOptForSize(&MF);
2738 const auto &TLI = MF.getSubtarget().getTargetLowering();
2739 // If TLI says that this fpimm is illegal, then we'll expand to a
2740 // constant pool load.
2741 if (TLI->isFPImmLegal(I.getOperand(1).getFPImm()->getValueAPF(),
2742 EVT::getFloatingPointVT(DefSize), OptForSize))
2743 break;
2744 [[fallthrough]];
2745 }
2746 case 16:
2747 case 128: {
2748 auto *FPImm = I.getOperand(1).getFPImm();
2749 auto *LoadMI = emitLoadFromConstantPool(FPImm, MIB);
2750 if (!LoadMI) {
2751 LLVM_DEBUG(dbgs() << "Failed to load double constant pool entry\n");
2752 return false;
2753 }
2754 MIB.buildCopy({DefReg}, {LoadMI->getOperand(0).getReg()});
2755 I.eraseFromParent();
2756 return RBI.constrainGenericRegister(DefReg, FPRRC, MRI);
2757 }
2758 }
2759
2760 assert((DefSize == 32 || DefSize == 64) && "Unexpected const def size");
2761 // Either emit a FMOV, or emit a copy to emit a normal mov.
2762 const Register DefGPRReg = MRI.createVirtualRegister(
2763 DefSize == 32 ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass);
2764 MachineOperand &RegOp = I.getOperand(0);
2765 RegOp.setReg(DefGPRReg);
2766 MIB.setInsertPt(MIB.getMBB(), std::next(I.getIterator()));
2767 MIB.buildCopy({DefReg}, {DefGPRReg});
2768
2769 if (!RBI.constrainGenericRegister(DefReg, FPRRC, MRI)) {
2770 LLVM_DEBUG(dbgs() << "Failed to constrain G_FCONSTANT def operand\n");
2771 return false;
2772 }
2773
2774 MachineOperand &ImmOp = I.getOperand(1);
2775 ImmOp.ChangeToImmediate(
2777
2778 const unsigned MovOpc =
2779 DefSize == 64 ? AArch64::MOVi64imm : AArch64::MOVi32imm;
2780 I.setDesc(TII.get(MovOpc));
2782 return true;
2783 }
2784 case TargetOpcode::G_EXTRACT: {
2785 Register DstReg = I.getOperand(0).getReg();
2786 Register SrcReg = I.getOperand(1).getReg();
2787 LLT SrcTy = MRI.getType(SrcReg);
2788 LLT DstTy = MRI.getType(DstReg);
2789 (void)DstTy;
2790 unsigned SrcSize = SrcTy.getSizeInBits();
2791
2792 if (SrcTy.getSizeInBits() > 64) {
2793 // This should be an extract of an s128, which is like a vector extract.
2794 if (SrcTy.getSizeInBits() != 128)
2795 return false;
2796 // Only support extracting 64 bits from an s128 at the moment.
2797 if (DstTy.getSizeInBits() != 64)
2798 return false;
2799
2800 unsigned Offset = I.getOperand(2).getImm();
2801 if (Offset % 64 != 0)
2802 return false;
2803
2804 // Check we have the right regbank always.
2805 const RegisterBank &SrcRB = *RBI.getRegBank(SrcReg, MRI, TRI);
2806 const RegisterBank &DstRB = *RBI.getRegBank(DstReg, MRI, TRI);
2807 assert(SrcRB.getID() == DstRB.getID() && "Wrong extract regbank!");
2808
2809 if (SrcRB.getID() == AArch64::GPRRegBankID) {
2810 auto NewI =
2811 MIB.buildInstr(TargetOpcode::COPY, {DstReg}, {})
2812 .addUse(SrcReg, {},
2813 Offset == 0 ? AArch64::sube64 : AArch64::subo64);
2814 constrainOperandRegClass(MF, TRI, MRI, TII, RBI, *NewI,
2815 AArch64::GPR64RegClass, NewI->getOperand(0));
2816 I.eraseFromParent();
2817 return true;
2818 }
2819
2820 // Emit the same code as a vector extract.
2821 // Offset must be a multiple of 64.
2822 unsigned LaneIdx = Offset / 64;
2823 MachineInstr *Extract = emitExtractVectorElt(
2824 DstReg, DstRB, LLT::scalar(64), SrcReg, LaneIdx, MIB);
2825 if (!Extract)
2826 return false;
2827 I.eraseFromParent();
2828 return true;
2829 }
2830
2831 I.setDesc(TII.get(SrcSize == 64 ? AArch64::UBFMXri : AArch64::UBFMWri));
2832 MachineInstrBuilder(MF, I).addImm(I.getOperand(2).getImm() +
2833 Ty.getSizeInBits() - 1);
2834
2835 if (SrcSize < 64) {
2836 assert(SrcSize == 32 && DstTy.getSizeInBits() == 16 &&
2837 "unexpected G_EXTRACT types");
2839 return true;
2840 }
2841
2842 DstReg = MRI.createGenericVirtualRegister(LLT::scalar(64));
2843 MIB.setInsertPt(MIB.getMBB(), std::next(I.getIterator()));
2844 MIB.buildInstr(TargetOpcode::COPY, {I.getOperand(0).getReg()}, {})
2845 .addReg(DstReg, {}, AArch64::sub_32);
2846 RBI.constrainGenericRegister(I.getOperand(0).getReg(),
2847 AArch64::GPR32RegClass, MRI);
2848 I.getOperand(0).setReg(DstReg);
2849
2851 return true;
2852 }
2853
2854 case TargetOpcode::G_INSERT: {
2855 LLT SrcTy = MRI.getType(I.getOperand(2).getReg());
2856 LLT DstTy = MRI.getType(I.getOperand(0).getReg());
2857 unsigned DstSize = DstTy.getSizeInBits();
2858 // Larger inserts are vectors, same-size ones should be something else by
2859 // now (split up or turned into COPYs).
2860 if (Ty.getSizeInBits() > 64 || SrcTy.getSizeInBits() > 32)
2861 return false;
2862
2863 I.setDesc(TII.get(DstSize == 64 ? AArch64::BFMXri : AArch64::BFMWri));
2864 unsigned LSB = I.getOperand(3).getImm();
2865 unsigned Width = MRI.getType(I.getOperand(2).getReg()).getSizeInBits();
2866 I.getOperand(3).setImm((DstSize - LSB) % DstSize);
2867 MachineInstrBuilder(MF, I).addImm(Width - 1);
2868
2869 if (DstSize < 64) {
2870 assert(DstSize == 32 && SrcTy.getSizeInBits() == 16 &&
2871 "unexpected G_INSERT types");
2873 return true;
2874 }
2875
2877 BuildMI(MBB, I.getIterator(), I.getDebugLoc(),
2878 TII.get(AArch64::SUBREG_TO_REG))
2879 .addDef(SrcReg)
2880 .addUse(I.getOperand(2).getReg())
2881 .addImm(AArch64::sub_32);
2882 RBI.constrainGenericRegister(I.getOperand(2).getReg(),
2883 AArch64::GPR32RegClass, MRI);
2884 I.getOperand(2).setReg(SrcReg);
2885
2887 return true;
2888 }
2889 case TargetOpcode::G_FRAME_INDEX: {
2890 // allocas and G_FRAME_INDEX are only supported in addrspace(0).
2891 if (Ty != LLT::pointer(0, 64)) {
2892 LLVM_DEBUG(dbgs() << "G_FRAME_INDEX pointer has type: " << Ty
2893 << ", expected: " << LLT::pointer(0, 64) << '\n');
2894 return false;
2895 }
2896 I.setDesc(TII.get(AArch64::ADDXri));
2897
2898 // MOs for a #0 shifted immediate.
2899 I.addOperand(MachineOperand::CreateImm(0));
2900 I.addOperand(MachineOperand::CreateImm(0));
2901
2903 return true;
2904 }
2905
2906 case TargetOpcode::G_GLOBAL_VALUE: {
2907 const GlobalValue *GV = nullptr;
2908 unsigned OpFlags;
2909 if (I.getOperand(1).isSymbol()) {
2910 OpFlags = I.getOperand(1).getTargetFlags();
2911 // Currently only used by "RtLibUseGOT".
2912 assert(OpFlags == AArch64II::MO_GOT);
2913 } else {
2914 GV = I.getOperand(1).getGlobal();
2915 if (GV->isThreadLocal())
2916 return selectTLSGlobalValue(I, MRI);
2917
2918 OpFlags = STI.ClassifyGlobalReference(GV, TM);
2919 }
2920
2921 if (OpFlags & AArch64II::MO_GOT) {
2922 bool IsGOTSigned = MF.getInfo<AArch64FunctionInfo>()->hasELFSignedGOT();
2923 I.setDesc(TII.get(IsGOTSigned ? AArch64::LOADgotAUTH : AArch64::LOADgot));
2924 I.getOperand(1).setTargetFlags(OpFlags);
2925 I.addImplicitDefUseOperands(MF);
2926 } else if (TM.getCodeModel() == CodeModel::Large &&
2927 !TM.isPositionIndependent()) {
2928 // Materialize the global using movz/movk instructions.
2929 materializeLargeCMVal(I, GV, OpFlags);
2930 I.eraseFromParent();
2931 return true;
2932 } else if (TM.getCodeModel() == CodeModel::Tiny) {
2933 I.setDesc(TII.get(AArch64::ADR));
2934 I.getOperand(1).setTargetFlags(OpFlags);
2935 } else {
2936 I.setDesc(TII.get(AArch64::MOVaddr));
2937 I.getOperand(1).setTargetFlags(OpFlags | AArch64II::MO_PAGE);
2938 MachineInstrBuilder MIB(MF, I);
2939 MIB.addGlobalAddress(GV, I.getOperand(1).getOffset(),
2941 }
2943 return true;
2944 }
2945
2946 case TargetOpcode::G_PTRAUTH_GLOBAL_VALUE:
2947 return selectPtrAuthGlobalValue(I, MRI);
2948
2949 case TargetOpcode::G_ZEXTLOAD:
2950 case TargetOpcode::G_LOAD:
2951 case TargetOpcode::G_STORE: {
2952 GLoadStore &LdSt = cast<GLoadStore>(I);
2953 bool IsZExtLoad = I.getOpcode() == TargetOpcode::G_ZEXTLOAD;
2954 LLT PtrTy = MRI.getType(LdSt.getPointerReg());
2955
2956 // Can only handle AddressSpace 0, 64-bit pointers.
2957 if (PtrTy != LLT::pointer(0, 64)) {
2958 return false;
2959 }
2960
2961 uint64_t MemSizeInBytes = LdSt.getMemSize().getValue();
2962 unsigned MemSizeInBits = LdSt.getMemSizeInBits().getValue();
2963 AtomicOrdering Order = LdSt.getMMO().getSuccessOrdering();
2964
2965 // Need special instructions for atomics that affect ordering.
2966 if (isStrongerThanMonotonic(Order)) {
2967 assert(!isa<GZExtLoad>(LdSt));
2968 assert(MemSizeInBytes <= 8 &&
2969 "128-bit atomics should already be custom-legalized");
2970
2971 if (isa<GLoad>(LdSt)) {
2972 static constexpr unsigned LDAPROpcodes[] = {
2973 AArch64::LDAPRB, AArch64::LDAPRH, AArch64::LDAPRW, AArch64::LDAPRX};
2974 static constexpr unsigned LDAROpcodes[] = {
2975 AArch64::LDARB, AArch64::LDARH, AArch64::LDARW, AArch64::LDARX};
2976 ArrayRef<unsigned> Opcodes =
2977 STI.hasRCPC() && Order != AtomicOrdering::SequentiallyConsistent
2978 ? LDAPROpcodes
2979 : LDAROpcodes;
2980 I.setDesc(TII.get(Opcodes[Log2_32(MemSizeInBytes)]));
2981 } else {
2982 static constexpr unsigned Opcodes[] = {AArch64::STLRB, AArch64::STLRH,
2983 AArch64::STLRW, AArch64::STLRX};
2984 Register ValReg = LdSt.getReg(0);
2985 if (MRI.getType(ValReg).getSizeInBits() == 64 && MemSizeInBits != 64) {
2986 // Emit a subreg copy of 32 bits.
2987 Register NewVal = MRI.createVirtualRegister(&AArch64::GPR32RegClass);
2988 MIB.buildInstr(TargetOpcode::COPY, {NewVal}, {})
2989 .addReg(I.getOperand(0).getReg(), {}, AArch64::sub_32);
2990 I.getOperand(0).setReg(NewVal);
2991 }
2992 I.setDesc(TII.get(Opcodes[Log2_32(MemSizeInBytes)]));
2993 }
2995 return true;
2996 }
2997
2998#ifndef NDEBUG
2999 const Register PtrReg = LdSt.getPointerReg();
3000 const RegisterBank &PtrRB = *RBI.getRegBank(PtrReg, MRI, TRI);
3001 // Check that the pointer register is valid.
3002 assert(PtrRB.getID() == AArch64::GPRRegBankID &&
3003 "Load/Store pointer operand isn't a GPR");
3004 assert(MRI.getType(PtrReg).isPointer() &&
3005 "Load/Store pointer operand isn't a pointer");
3006#endif
3007
3008 const Register ValReg = LdSt.getReg(0);
3009 const RegisterBank &RB = *RBI.getRegBank(ValReg, MRI, TRI);
3010 LLT ValTy = MRI.getType(ValReg);
3011
3012 // The code below doesn't support truncating stores, so we need to split it
3013 // again.
3014 if (isa<GStore>(LdSt) && ValTy.getSizeInBits() > MemSizeInBits &&
3015 RB.getID() == AArch64::FPRRegBankID) {
3016 unsigned SubReg;
3017 LLT MemTy = LdSt.getMMO().getMemoryType();
3018 auto *RC = getRegClassForTypeOnBank(MemTy, RB);
3019 if (!getSubRegForClass(RC, TRI, SubReg))
3020 return false;
3021
3022 // Generate a subreg copy.
3023 auto Copy = MIB.buildInstr(TargetOpcode::COPY, {MemTy}, {})
3024 .addReg(ValReg, {}, SubReg)
3025 .getReg(0);
3026 RBI.constrainGenericRegister(Copy, *RC, MRI);
3027 LdSt.getOperand(0).setReg(Copy);
3028 } else if (isa<GLoad>(LdSt) && ValTy.getSizeInBits() > MemSizeInBits) {
3029 // If this is an any-extending load from the FPR bank, split it into a regular
3030 // load + extend.
3031 if (RB.getID() == AArch64::FPRRegBankID) {
3032 unsigned SubReg;
3033 LLT MemTy = LdSt.getMMO().getMemoryType();
3034 auto *RC = getRegClassForTypeOnBank(MemTy, RB);
3035 if (!getSubRegForClass(RC, TRI, SubReg))
3036 return false;
3037 Register OldDst = LdSt.getReg(0);
3038 Register NewDst =
3040 LdSt.getOperand(0).setReg(NewDst);
3041 MRI.setRegBank(NewDst, RB);
3042 // Generate a SUBREG_TO_REG to extend it.
3043 MIB.setInsertPt(MIB.getMBB(), std::next(LdSt.getIterator()));
3044 MIB.buildInstr(AArch64::SUBREG_TO_REG, {OldDst}, {})
3045 .addUse(NewDst)
3046 .addImm(SubReg);
3047 auto SubRegRC = getRegClassForTypeOnBank(MRI.getType(OldDst), RB);
3048 RBI.constrainGenericRegister(OldDst, *SubRegRC, MRI);
3049 MIB.setInstr(LdSt);
3050 ValTy = MemTy; // This is no longer an extending load.
3051 }
3052 }
3053
3054 // Helper lambda for partially selecting I. Either returns the original
3055 // instruction with an updated opcode, or a new instruction.
3056 auto SelectLoadStoreAddressingMode = [&]() -> MachineInstr * {
3057 bool IsStore = isa<GStore>(I);
3058 const unsigned NewOpc =
3059 selectLoadStoreUIOp(I.getOpcode(), RB.getID(), MemSizeInBits);
3060 if (NewOpc == I.getOpcode())
3061 return nullptr;
3062 // Check if we can fold anything into the addressing mode.
3063 auto AddrModeFns =
3064 selectAddrModeIndexed(I.getOperand(1), MemSizeInBytes);
3065 if (!AddrModeFns) {
3066 // Can't fold anything. Use the original instruction.
3067 I.setDesc(TII.get(NewOpc));
3068 I.addOperand(MachineOperand::CreateImm(0));
3069 return &I;
3070 }
3071
3072 // Folded something. Create a new instruction and return it.
3073 auto NewInst = MIB.buildInstr(NewOpc, {}, {}, I.getFlags());
3074 Register CurValReg = I.getOperand(0).getReg();
3075 IsStore ? NewInst.addUse(CurValReg) : NewInst.addDef(CurValReg);
3076 NewInst.cloneMemRefs(I);
3077 for (auto &Fn : *AddrModeFns)
3078 Fn(NewInst);
3079 I.eraseFromParent();
3080 return &*NewInst;
3081 };
3082
3083 MachineInstr *LoadStore = SelectLoadStoreAddressingMode();
3084 if (!LoadStore)
3085 return false;
3086
3087 // If we're storing a 0, use WZR/XZR.
3088 if (Opcode == TargetOpcode::G_STORE) {
3090 LoadStore->getOperand(0).getReg(), MRI);
3091 if (CVal && CVal->Value == 0) {
3092 switch (LoadStore->getOpcode()) {
3093 case AArch64::STRWui:
3094 case AArch64::STRHHui:
3095 case AArch64::STRBBui:
3096 LoadStore->getOperand(0).setReg(AArch64::WZR);
3097 break;
3098 case AArch64::STRXui:
3099 LoadStore->getOperand(0).setReg(AArch64::XZR);
3100 break;
3101 }
3102 }
3103 }
3104
3105 if (IsZExtLoad || (Opcode == TargetOpcode::G_LOAD &&
3106 ValTy == LLT::scalar(64) && MemSizeInBits == 32)) {
3107 // The any/zextload from a smaller type to i32 should be handled by the
3108 // importer.
3109 if (MRI.getType(LoadStore->getOperand(0).getReg()).getSizeInBits() != 64)
3110 return false;
3111 // If we have an extending load then change the load's type to be a
3112 // narrower reg and zero_extend with SUBREG_TO_REG.
3113 Register LdReg = MRI.createVirtualRegister(&AArch64::GPR32RegClass);
3114 Register DstReg = LoadStore->getOperand(0).getReg();
3115 LoadStore->getOperand(0).setReg(LdReg);
3116
3117 MIB.setInsertPt(MIB.getMBB(), std::next(LoadStore->getIterator()));
3118 MIB.buildInstr(AArch64::SUBREG_TO_REG, {DstReg}, {})
3119 .addUse(LdReg)
3120 .addImm(AArch64::sub_32);
3121 constrainSelectedInstRegOperands(*LoadStore, TII, TRI, RBI);
3122 return RBI.constrainGenericRegister(DstReg, AArch64::GPR64allRegClass,
3123 MRI);
3124 }
3125 constrainSelectedInstRegOperands(*LoadStore, TII, TRI, RBI);
3126 return true;
3127 }
3128
3129 case TargetOpcode::G_INDEXED_ZEXTLOAD:
3130 case TargetOpcode::G_INDEXED_SEXTLOAD:
3131 return selectIndexedExtLoad(I, MRI);
3132 case TargetOpcode::G_INDEXED_LOAD:
3133 return selectIndexedLoad(I, MRI);
3134 case TargetOpcode::G_INDEXED_STORE:
3135 return selectIndexedStore(cast<GIndexedStore>(I), MRI);
3136
3137 case TargetOpcode::G_LSHR:
3138 case TargetOpcode::G_ASHR:
3139 if (MRI.getType(I.getOperand(0).getReg()).isVector())
3140 return selectVectorAshrLshr(I, MRI);
3141 [[fallthrough]];
3142 case TargetOpcode::G_SHL: {
3143 if (Opcode == TargetOpcode::G_SHL &&
3144 MRI.getType(I.getOperand(0).getReg()).isVector())
3145 return selectVectorSHL(I, MRI);
3146
3147 // These shifts were legalized to have 64 bit shift amounts because we
3148 // want to take advantage of the selection patterns that assume the
3149 // immediates are s64s, however, selectBinaryOp will assume both operands
3150 // will have the same bit size.
3151 {
3152 Register SrcReg = I.getOperand(1).getReg();
3153 Register ShiftReg = I.getOperand(2).getReg();
3154 const LLT ShiftTy = MRI.getType(ShiftReg);
3155 const LLT SrcTy = MRI.getType(SrcReg);
3156 if (!SrcTy.isVector() && SrcTy.getSizeInBits() == 32 &&
3157 ShiftTy.getSizeInBits() == 64) {
3158 assert(!ShiftTy.isVector() && "unexpected vector shift ty");
3159 // Insert a subregister copy to implement a 64->32 trunc
3160 auto Trunc = MIB.buildInstr(TargetOpcode::COPY, {SrcTy}, {})
3161 .addReg(ShiftReg, {}, AArch64::sub_32);
3162 MRI.setRegBank(Trunc.getReg(0), RBI.getRegBank(AArch64::GPRRegBankID));
3163 I.getOperand(2).setReg(Trunc.getReg(0));
3164 }
3165 }
3166
3167 const unsigned OpSize = Ty.getSizeInBits();
3168 const Register DefReg = I.getOperand(0).getReg();
3169 const RegisterBank &RB = *RBI.getRegBank(DefReg, MRI, TRI);
3170
3171 const unsigned NewOpc = selectBinaryOp(I.getOpcode(), RB.getID(), OpSize);
3172 if (NewOpc == I.getOpcode())
3173 return false;
3174
3175 I.setDesc(TII.get(NewOpc));
3176 // FIXME: Should the type be always reset in setDesc?
3177
3178 // Now that we selected an opcode, we need to constrain the register
3179 // operands to use appropriate classes.
3181 return true;
3182 }
3183 case TargetOpcode::G_PTR_ADD: {
3184 emitADD(I.getOperand(0).getReg(), I.getOperand(1), I.getOperand(2), MIB);
3185 I.eraseFromParent();
3186 return true;
3187 }
3188
3189 case TargetOpcode::G_SADDE:
3190 case TargetOpcode::G_UADDE:
3191 case TargetOpcode::G_SSUBE:
3192 case TargetOpcode::G_USUBE:
3193 case TargetOpcode::G_SADDO:
3194 case TargetOpcode::G_UADDO:
3195 case TargetOpcode::G_SSUBO:
3196 case TargetOpcode::G_USUBO:
3197 return selectOverflowOp(I, MRI);
3198
3199 case TargetOpcode::G_PTRMASK: {
3200 Register MaskReg = I.getOperand(2).getReg();
3201 std::optional<int64_t> MaskVal = getIConstantVRegSExtVal(MaskReg, MRI);
3202 // TODO: Implement arbitrary cases
3203 if (!MaskVal || !isShiftedMask_64(*MaskVal))
3204 return false;
3205
3206 uint64_t Mask = *MaskVal;
3207 I.setDesc(TII.get(AArch64::ANDXri));
3208 I.getOperand(2).ChangeToImmediate(
3210
3212 return true;
3213 }
3214 case TargetOpcode::G_PTRTOINT:
3215 case TargetOpcode::G_TRUNC: {
3216 const LLT DstTy = MRI.getType(I.getOperand(0).getReg());
3217 const LLT SrcTy = MRI.getType(I.getOperand(1).getReg());
3218
3219 const Register DstReg = I.getOperand(0).getReg();
3220 const Register SrcReg = I.getOperand(1).getReg();
3221
3222 const RegisterBank &DstRB = *RBI.getRegBank(DstReg, MRI, TRI);
3223 const RegisterBank &SrcRB = *RBI.getRegBank(SrcReg, MRI, TRI);
3224
3225 if (DstRB.getID() != SrcRB.getID()) {
3226 LLVM_DEBUG(
3227 dbgs() << "G_TRUNC/G_PTRTOINT input/output on different banks\n");
3228 return false;
3229 }
3230
3231 if (DstRB.getID() == AArch64::GPRRegBankID) {
3232 const TargetRegisterClass *DstRC = getRegClassForTypeOnBank(DstTy, DstRB);
3233 if (!DstRC)
3234 return false;
3235
3236 const TargetRegisterClass *SrcRC = getRegClassForTypeOnBank(SrcTy, SrcRB);
3237 if (!SrcRC)
3238 return false;
3239
3240 if (!RBI.constrainGenericRegister(SrcReg, *SrcRC, MRI) ||
3241 !RBI.constrainGenericRegister(DstReg, *DstRC, MRI)) {
3242 LLVM_DEBUG(dbgs() << "Failed to constrain G_TRUNC/G_PTRTOINT\n");
3243 return false;
3244 }
3245
3246 if (DstRC == SrcRC) {
3247 // Nothing to be done
3248 } else if (Opcode == TargetOpcode::G_TRUNC && DstTy == LLT::scalar(32) &&
3249 SrcTy == LLT::scalar(64)) {
3250 llvm_unreachable("TableGen can import this case");
3251 return false;
3252 } else if (DstRC == &AArch64::GPR32RegClass &&
3253 SrcRC == &AArch64::GPR64RegClass) {
3254 I.getOperand(1).setSubReg(AArch64::sub_32);
3255 } else {
3256 LLVM_DEBUG(
3257 dbgs() << "Unhandled mismatched classes in G_TRUNC/G_PTRTOINT\n");
3258 return false;
3259 }
3260
3261 I.setDesc(TII.get(TargetOpcode::COPY));
3262 return true;
3263 } else if (DstRB.getID() == AArch64::FPRRegBankID) {
3264 if (DstTy == LLT::fixed_vector(4, 16) &&
3265 SrcTy == LLT::fixed_vector(4, 32)) {
3266 I.setDesc(TII.get(AArch64::XTNv4i16));
3268 return true;
3269 }
3270
3271 if (!SrcTy.isVector() && SrcTy.getSizeInBits() == 128) {
3272 MachineInstr *Extract = emitExtractVectorElt(
3273 DstReg, DstRB, LLT::scalar(DstTy.getSizeInBits()), SrcReg, 0, MIB);
3274 if (!Extract)
3275 return false;
3276 I.eraseFromParent();
3277 return true;
3278 }
3279
3280 // We might have a vector G_PTRTOINT, in which case just emit a COPY.
3281 if (Opcode == TargetOpcode::G_PTRTOINT) {
3282 assert(DstTy.isVector() && "Expected an FPR ptrtoint to be a vector");
3283 I.setDesc(TII.get(TargetOpcode::COPY));
3284 return selectCopy(I, TII, MRI, TRI, RBI);
3285 }
3286 }
3287
3288 return false;
3289 }
3290
3291 case TargetOpcode::G_ANYEXT: {
3292 if (selectUSMovFromExtend(I, MRI))
3293 return true;
3294
3295 const Register DstReg = I.getOperand(0).getReg();
3296 const Register SrcReg = I.getOperand(1).getReg();
3297
3298 const RegisterBank &RBDst = *RBI.getRegBank(DstReg, MRI, TRI);
3299 if (RBDst.getID() != AArch64::GPRRegBankID) {
3300 LLVM_DEBUG(dbgs() << "G_ANYEXT on bank: " << RBDst
3301 << ", expected: GPR\n");
3302 return false;
3303 }
3304
3305 const RegisterBank &RBSrc = *RBI.getRegBank(SrcReg, MRI, TRI);
3306 if (RBSrc.getID() != AArch64::GPRRegBankID) {
3307 LLVM_DEBUG(dbgs() << "G_ANYEXT on bank: " << RBSrc
3308 << ", expected: GPR\n");
3309 return false;
3310 }
3311
3312 const unsigned DstSize = MRI.getType(DstReg).getSizeInBits();
3313
3314 if (DstSize == 0) {
3315 LLVM_DEBUG(dbgs() << "G_ANYEXT operand has no size, not a gvreg?\n");
3316 return false;
3317 }
3318
3319 if (DstSize != 64 && DstSize > 32) {
3320 LLVM_DEBUG(dbgs() << "G_ANYEXT to size: " << DstSize
3321 << ", expected: 32 or 64\n");
3322 return false;
3323 }
3324 // At this point G_ANYEXT is just like a plain COPY, but we need
3325 // to explicitly form the 64-bit value if any.
3326 if (DstSize > 32) {
3327 Register ExtSrc = MRI.createVirtualRegister(&AArch64::GPR64allRegClass);
3328 BuildMI(MBB, I, I.getDebugLoc(), TII.get(AArch64::SUBREG_TO_REG))
3329 .addDef(ExtSrc)
3330 .addUse(SrcReg)
3331 .addImm(AArch64::sub_32);
3332 I.getOperand(1).setReg(ExtSrc);
3333 }
3334 return selectCopy(I, TII, MRI, TRI, RBI);
3335 }
3336
3337 case TargetOpcode::G_ZEXT:
3338 case TargetOpcode::G_SEXT_INREG:
3339 case TargetOpcode::G_SEXT: {
3340 if (selectUSMovFromExtend(I, MRI))
3341 return true;
3342
3343 unsigned Opcode = I.getOpcode();
3344 const bool IsSigned = Opcode != TargetOpcode::G_ZEXT;
3345 const Register DefReg = I.getOperand(0).getReg();
3346 Register SrcReg = I.getOperand(1).getReg();
3347 const LLT DstTy = MRI.getType(DefReg);
3348 const LLT SrcTy = MRI.getType(SrcReg);
3349 unsigned DstSize = DstTy.getSizeInBits();
3350 unsigned SrcSize = SrcTy.getSizeInBits();
3351
3352 // SEXT_INREG has the same src reg size as dst, the size of the value to be
3353 // extended is encoded in the imm.
3354 if (Opcode == TargetOpcode::G_SEXT_INREG)
3355 SrcSize = I.getOperand(2).getImm();
3356
3357 if (DstTy.isVector())
3358 return false; // Should be handled by imported patterns.
3359
3360 assert((*RBI.getRegBank(DefReg, MRI, TRI)).getID() ==
3361 AArch64::GPRRegBankID &&
3362 "Unexpected ext regbank");
3363
3364 MachineInstr *ExtI;
3365
3366 // First check if we're extending the result of a load which has a dest type
3367 // smaller than 32 bits, then this zext is redundant. GPR32 is the smallest
3368 // GPR register on AArch64 and all loads which are smaller automatically
3369 // zero-extend the upper bits. E.g.
3370 // %v(s8) = G_LOAD %p, :: (load 1)
3371 // %v2(s32) = G_ZEXT %v(s8)
3372 if (!IsSigned) {
3373 auto *LoadMI = getOpcodeDef(TargetOpcode::G_LOAD, SrcReg, MRI);
3374 bool IsGPR =
3375 RBI.getRegBank(SrcReg, MRI, TRI)->getID() == AArch64::GPRRegBankID;
3376 if (LoadMI && IsGPR) {
3377 const MachineMemOperand *MemOp = *LoadMI->memoperands_begin();
3378 unsigned BytesLoaded = MemOp->getSize().getValue();
3379 if (BytesLoaded < 4 && SrcTy.getSizeInBytes() == BytesLoaded)
3380 return selectCopy(I, TII, MRI, TRI, RBI);
3381 }
3382
3383 // For the 32-bit -> 64-bit case, we can emit a mov (ORRWrs)
3384 // + SUBREG_TO_REG.
3385 if (IsGPR && SrcSize == 32 && DstSize == 64) {
3386 Register SubregToRegSrc =
3387 MRI.createVirtualRegister(&AArch64::GPR32RegClass);
3388 const Register ZReg = AArch64::WZR;
3389 MIB.buildInstr(AArch64::ORRWrs, {SubregToRegSrc}, {ZReg, SrcReg})
3390 .addImm(0);
3391
3392 MIB.buildInstr(AArch64::SUBREG_TO_REG, {DefReg}, {})
3393 .addUse(SubregToRegSrc)
3394 .addImm(AArch64::sub_32);
3395
3396 if (!RBI.constrainGenericRegister(DefReg, AArch64::GPR64RegClass,
3397 MRI)) {
3398 LLVM_DEBUG(dbgs() << "Failed to constrain G_ZEXT destination\n");
3399 return false;
3400 }
3401
3402 if (!RBI.constrainGenericRegister(SrcReg, AArch64::GPR32RegClass,
3403 MRI)) {
3404 LLVM_DEBUG(dbgs() << "Failed to constrain G_ZEXT source\n");
3405 return false;
3406 }
3407
3408 I.eraseFromParent();
3409 return true;
3410 }
3411 }
3412
3413 if (DstSize == 64) {
3414 if (Opcode != TargetOpcode::G_SEXT_INREG) {
3415 // FIXME: Can we avoid manually doing this?
3416 if (!RBI.constrainGenericRegister(SrcReg, AArch64::GPR32RegClass,
3417 MRI)) {
3418 LLVM_DEBUG(dbgs() << "Failed to constrain " << TII.getName(Opcode)
3419 << " operand\n");
3420 return false;
3421 }
3422 SrcReg = MIB.buildInstr(AArch64::SUBREG_TO_REG,
3423 {&AArch64::GPR64RegClass}, {})
3424 .addUse(SrcReg)
3425 .addImm(AArch64::sub_32)
3426 .getReg(0);
3427 }
3428
3429 ExtI = MIB.buildInstr(IsSigned ? AArch64::SBFMXri : AArch64::UBFMXri,
3430 {DefReg}, {SrcReg})
3431 .addImm(0)
3432 .addImm(SrcSize - 1);
3433 } else if (DstSize <= 32) {
3434 ExtI = MIB.buildInstr(IsSigned ? AArch64::SBFMWri : AArch64::UBFMWri,
3435 {DefReg}, {SrcReg})
3436 .addImm(0)
3437 .addImm(SrcSize - 1);
3438 } else {
3439 return false;
3440 }
3441
3443 I.eraseFromParent();
3444 return true;
3445 }
3446
3447 case TargetOpcode::G_FREEZE:
3448 return selectCopy(I, TII, MRI, TRI, RBI);
3449
3450 case TargetOpcode::G_INTTOPTR:
3451 // The importer is currently unable to import pointer types since they
3452 // didn't exist in SelectionDAG.
3453 return selectCopy(I, TII, MRI, TRI, RBI);
3454
3455 case TargetOpcode::G_BITCAST:
3456 // Imported SelectionDAG rules can handle every bitcast except those that
3457 // bitcast from a type to the same type. Ideally, these shouldn't occur
3458 // but we might not run an optimizer that deletes them. The other exception
3459 // is bitcasts involving pointer types, as SelectionDAG has no knowledge
3460 // of them.
3461 return selectCopy(I, TII, MRI, TRI, RBI);
3462
3463 case TargetOpcode::G_SELECT: {
3464 auto &Sel = cast<GSelect>(I);
3465 const Register CondReg = Sel.getCondReg();
3466 const Register TReg = Sel.getTrueReg();
3467 const Register FReg = Sel.getFalseReg();
3468
3469 if (tryOptSelect(Sel))
3470 return true;
3471
3472 // Make sure to use an unused vreg instead of wzr, so that the peephole
3473 // optimizations will be able to optimize these.
3474 Register DeadVReg = MRI.createVirtualRegister(&AArch64::GPR32RegClass);
3475 auto TstMI = MIB.buildInstr(AArch64::ANDSWri, {DeadVReg}, {CondReg})
3476 .addImm(AArch64_AM::encodeLogicalImmediate(1, 32));
3478 if (!emitSelect(Sel.getReg(0), TReg, FReg, AArch64CC::NE, MIB))
3479 return false;
3480 Sel.eraseFromParent();
3481 return true;
3482 }
3483 case TargetOpcode::G_ICMP: {
3484 if (Ty.isVector())
3485 return false;
3486
3487 if (Ty != LLT::scalar(32)) {
3488 LLVM_DEBUG(dbgs() << "G_ICMP result has type: " << Ty
3489 << ", expected: " << LLT::scalar(32) << '\n');
3490 return false;
3491 }
3492
3493 auto &PredOp = I.getOperand(1);
3494 emitIntegerCompare(I.getOperand(2), I.getOperand(3), PredOp, MIB);
3495 auto Pred = static_cast<CmpInst::Predicate>(PredOp.getPredicate());
3497 CmpInst::getInversePredicate(Pred), I.getOperand(3).getReg(), &MRI);
3498 emitCSINC(/*Dst=*/I.getOperand(0).getReg(), /*Src1=*/AArch64::WZR,
3499 /*Src2=*/AArch64::WZR, InvCC, MIB);
3500 I.eraseFromParent();
3501 return true;
3502 }
3503
3504 case TargetOpcode::G_FCMP: {
3505 CmpInst::Predicate Pred =
3506 static_cast<CmpInst::Predicate>(I.getOperand(1).getPredicate());
3507 if (!emitFPCompare(I.getOperand(2).getReg(), I.getOperand(3).getReg(), MIB,
3508 Pred) ||
3509 !emitCSetForFCmp(I.getOperand(0).getReg(), Pred, MIB))
3510 return false;
3511 I.eraseFromParent();
3512 return true;
3513 }
3514 case TargetOpcode::G_VASTART:
3515 return STI.isTargetDarwin() ? selectVaStartDarwin(I, MF, MRI)
3516 : selectVaStartAAPCS(I, MF, MRI);
3517 case TargetOpcode::G_INTRINSIC:
3518 return selectIntrinsic(I, MRI);
3519 case TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS:
3520 return selectIntrinsicWithSideEffects(I, MRI);
3521 case TargetOpcode::G_IMPLICIT_DEF: {
3522 I.setDesc(TII.get(TargetOpcode::IMPLICIT_DEF));
3523 const LLT DstTy = MRI.getType(I.getOperand(0).getReg());
3524 const Register DstReg = I.getOperand(0).getReg();
3525 const RegisterBank &DstRB = *RBI.getRegBank(DstReg, MRI, TRI);
3526 const TargetRegisterClass *DstRC = getRegClassForTypeOnBank(DstTy, DstRB);
3527 RBI.constrainGenericRegister(DstReg, *DstRC, MRI);
3528 return true;
3529 }
3530 case TargetOpcode::G_BLOCK_ADDR: {
3531 Function *BAFn = I.getOperand(1).getBlockAddress()->getFunction();
3532 if (std::optional<uint16_t> BADisc =
3534 MIB.buildInstr(TargetOpcode::IMPLICIT_DEF, {AArch64::X16}, {});
3535 MIB.buildInstr(TargetOpcode::IMPLICIT_DEF, {AArch64::X17}, {});
3536 MIB.buildInstr(AArch64::MOVaddrPAC)
3537 .addBlockAddress(I.getOperand(1).getBlockAddress())
3539 .addReg(/*AddrDisc=*/AArch64::XZR)
3540 .addImm(*BADisc)
3541 .constrainAllUses(TII, TRI, RBI);
3542 MIB.buildCopy(I.getOperand(0).getReg(), Register(AArch64::X16));
3543 RBI.constrainGenericRegister(I.getOperand(0).getReg(),
3544 AArch64::GPR64RegClass, MRI);
3545 I.eraseFromParent();
3546 return true;
3547 }
3549 materializeLargeCMVal(I, I.getOperand(1).getBlockAddress(), 0);
3550 I.eraseFromParent();
3551 return true;
3552 } else {
3553 I.setDesc(TII.get(AArch64::MOVaddrBA));
3554 auto MovMI = BuildMI(MBB, I, I.getDebugLoc(), TII.get(AArch64::MOVaddrBA),
3555 I.getOperand(0).getReg())
3556 .addBlockAddress(I.getOperand(1).getBlockAddress(),
3557 /* Offset */ 0, AArch64II::MO_PAGE)
3559 I.getOperand(1).getBlockAddress(), /* Offset */ 0,
3561 I.eraseFromParent();
3563 return true;
3564 }
3565 }
3566 case AArch64::G_DUP: {
3567 // When the scalar of G_DUP is an s8/s16 gpr, they can't be selected by
3568 // imported patterns. Do it manually here. Avoiding generating s16 gpr is
3569 // difficult because at RBS we may end up pessimizing the fpr case if we
3570 // decided to add an anyextend to fix this. Manual selection is the most
3571 // robust solution for now.
3572 if (RBI.getRegBank(I.getOperand(1).getReg(), MRI, TRI)->getID() !=
3573 AArch64::GPRRegBankID)
3574 return false; // We expect the fpr regbank case to be imported.
3575 LLT VecTy = MRI.getType(I.getOperand(0).getReg());
3576 if (VecTy == LLT::fixed_vector(8, 8))
3577 I.setDesc(TII.get(AArch64::DUPv8i8gpr));
3578 else if (VecTy == LLT::fixed_vector(16, 8))
3579 I.setDesc(TII.get(AArch64::DUPv16i8gpr));
3580 else if (VecTy == LLT::fixed_vector(4, 16))
3581 I.setDesc(TII.get(AArch64::DUPv4i16gpr));
3582 else if (VecTy == LLT::fixed_vector(8, 16))
3583 I.setDesc(TII.get(AArch64::DUPv8i16gpr));
3584 else
3585 return false;
3587 return true;
3588 }
3589 case TargetOpcode::G_BUILD_VECTOR:
3590 return selectBuildVector(I, MRI);
3591 case TargetOpcode::G_MERGE_VALUES:
3592 return selectMergeValues(I, MRI);
3593 case TargetOpcode::G_UNMERGE_VALUES:
3594 return selectUnmergeValues(I, MRI);
3595 case TargetOpcode::G_SHUFFLE_VECTOR:
3596 return selectShuffleVector(I, MRI);
3597 case TargetOpcode::G_EXTRACT_VECTOR_ELT:
3598 return selectExtractElt(I, MRI);
3599 case TargetOpcode::G_CONCAT_VECTORS:
3600 return selectConcatVectors(I, MRI);
3601 case TargetOpcode::G_JUMP_TABLE:
3602 return selectJumpTable(I, MRI);
3603 case TargetOpcode::G_MEMCPY:
3604 case TargetOpcode::G_MEMCPY_INLINE:
3605 case TargetOpcode::G_MEMMOVE:
3606 case TargetOpcode::G_MEMSET:
3607 case TargetOpcode::G_MEMSET_INLINE:
3608 assert(STI.hasMOPS() && "Shouldn't get here without +mops feature");
3609 return selectMOPS(I, MRI);
3610 }
3611
3612 return false;
3613}
3614
3615bool AArch64InstructionSelector::selectAndRestoreState(MachineInstr &I) {
3616 MachineIRBuilderState OldMIBState = MIB.getState();
3617 bool Success = select(I);
3618 MIB.setState(OldMIBState);
3619 return Success;
3620}
3621
3622bool AArch64InstructionSelector::selectMOPS(MachineInstr &GI,
3623 MachineRegisterInfo &MRI) {
3624 unsigned Mopcode;
3625 switch (GI.getOpcode()) {
3626 case TargetOpcode::G_MEMCPY:
3627 case TargetOpcode::G_MEMCPY_INLINE:
3628 Mopcode = AArch64::MOPSMemoryCopyPseudo;
3629 break;
3630 case TargetOpcode::G_MEMMOVE:
3631 Mopcode = AArch64::MOPSMemoryMovePseudo;
3632 break;
3633 case TargetOpcode::G_MEMSET:
3634 case TargetOpcode::G_MEMSET_INLINE:
3635 // For tagged memset see llvm.aarch64.mops.memset.tag
3636 Mopcode = AArch64::MOPSMemorySetPseudo;
3637 break;
3638 }
3639
3640 auto &DstPtr = GI.getOperand(0);
3641 auto &SrcOrVal = GI.getOperand(1);
3642 auto &Size = GI.getOperand(2);
3643
3644 // Create copies of the registers that can be clobbered.
3645 const Register DstPtrCopy = MRI.cloneVirtualRegister(DstPtr.getReg());
3646 const Register SrcValCopy = MRI.cloneVirtualRegister(SrcOrVal.getReg());
3647 const Register SizeCopy = MRI.cloneVirtualRegister(Size.getReg());
3648
3649 const bool IsSet = Mopcode == AArch64::MOPSMemorySetPseudo;
3650 const auto &SrcValRegClass =
3651 IsSet ? AArch64::GPR64RegClass : AArch64::GPR64commonRegClass;
3652
3653 // Constrain to specific registers
3654 RBI.constrainGenericRegister(DstPtrCopy, AArch64::GPR64commonRegClass, MRI);
3655 RBI.constrainGenericRegister(SrcValCopy, SrcValRegClass, MRI);
3656 RBI.constrainGenericRegister(SizeCopy, AArch64::GPR64RegClass, MRI);
3657
3658 MIB.buildCopy(DstPtrCopy, DstPtr);
3659 MIB.buildCopy(SrcValCopy, SrcOrVal);
3660 MIB.buildCopy(SizeCopy, Size);
3661
3662 // New instruction uses the copied registers because it must update them.
3663 // The defs are not used since they don't exist in G_MEM*. They are still
3664 // tied.
3665 // Note: order of operands is different from G_MEMSET, G_MEMCPY, G_MEMMOVE
3666 Register DefDstPtr = MRI.createVirtualRegister(&AArch64::GPR64commonRegClass);
3667 Register DefSize = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
3668 if (IsSet) {
3669 MIB.buildInstr(Mopcode, {DefDstPtr, DefSize},
3670 {DstPtrCopy, SizeCopy, SrcValCopy})
3671 .setOperandDead(5); // implicit-def $nzcv
3672 } else {
3673 Register DefSrcPtr = MRI.createVirtualRegister(&SrcValRegClass);
3674 MIB.buildInstr(Mopcode, {DefDstPtr, DefSrcPtr, DefSize},
3675 {DstPtrCopy, SrcValCopy, SizeCopy})
3676 .setOperandDead(6); // implicit-def $nzcv
3677 }
3678
3679 GI.eraseFromParent();
3680 return true;
3681}
3682
3683bool AArch64InstructionSelector::selectBrJT(MachineInstr &I,
3684 MachineRegisterInfo &MRI) {
3685 assert(I.getOpcode() == TargetOpcode::G_BRJT && "Expected G_BRJT");
3686 Register JTAddr = I.getOperand(0).getReg();
3687 unsigned JTI = I.getOperand(1).getIndex();
3688 Register Index = I.getOperand(2).getReg();
3689
3690 MF->getInfo<AArch64FunctionInfo>()->setJumpTableEntryInfo(JTI, 4, nullptr);
3691
3692 // With aarch64-jump-table-hardening, we only expand the jump table dispatch
3693 // sequence later, to guarantee the integrity of the intermediate values.
3694 if (MF->getFunction().hasFnAttribute("aarch64-jump-table-hardening")) {
3696 if (STI.isTargetMachO()) {
3697 if (CM != CodeModel::Small && CM != CodeModel::Large)
3698 report_fatal_error("Unsupported code-model for hardened jump-table");
3699 } else {
3700 // Note that COFF support would likely also need JUMP_TABLE_DEBUG_INFO.
3701 assert(STI.isTargetELF() &&
3702 "jump table hardening only supported on MachO/ELF");
3703 if (CM != CodeModel::Small)
3704 report_fatal_error("Unsupported code-model for hardened jump-table");
3705 }
3706
3707 MIB.buildCopy({AArch64::X16}, I.getOperand(2).getReg());
3708 MIB.buildInstr(AArch64::BR_JumpTable)
3709 .addJumpTableIndex(I.getOperand(1).getIndex());
3710 I.eraseFromParent();
3711 return true;
3712 }
3713
3714 Register TargetReg = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
3715 Register ScratchReg = MRI.createVirtualRegister(&AArch64::GPR64spRegClass);
3716
3717 auto JumpTableInst = MIB.buildInstr(AArch64::JumpTableDest32,
3718 {TargetReg, ScratchReg}, {JTAddr, Index})
3719 .addJumpTableIndex(JTI);
3720 // Save the jump table info.
3721 MIB.buildInstr(TargetOpcode::JUMP_TABLE_DEBUG_INFO, {},
3722 {static_cast<int64_t>(JTI)});
3723 // Build the indirect branch.
3724 MIB.buildInstr(AArch64::BR, {}, {TargetReg});
3725 I.eraseFromParent();
3726 constrainSelectedInstRegOperands(*JumpTableInst, TII, TRI, RBI);
3727 return true;
3728}
3729
3730bool AArch64InstructionSelector::selectJumpTable(MachineInstr &I,
3731 MachineRegisterInfo &MRI) {
3732 assert(I.getOpcode() == TargetOpcode::G_JUMP_TABLE && "Expected jump table");
3733 assert(I.getOperand(1).isJTI() && "Jump table op should have a JTI!");
3734
3735 Register DstReg = I.getOperand(0).getReg();
3736 unsigned JTI = I.getOperand(1).getIndex();
3737 // We generate a MOVaddrJT which will get expanded to an ADRP + ADD later.
3738 auto MovMI =
3739 MIB.buildInstr(AArch64::MOVaddrJT, {DstReg}, {})
3740 .addJumpTableIndex(JTI, AArch64II::MO_PAGE)
3742 I.eraseFromParent();
3744 return true;
3745}
3746
3747bool AArch64InstructionSelector::selectTLSLocalExecELF(
3748 const GlobalValue *GV, MachineInstr &I, MachineRegisterInfo &MRI) {
3749 auto ConstrainRegOps = [&](MachineInstrBuilder MIB) {
3751 };
3752 Register ThreadBase = MRI.createGenericVirtualRegister(LLT::pointer(0, 64));
3753 ConstrainRegOps(MIB.buildInstr(AArch64::MOVbaseTLS, {ThreadBase}, {}));
3754
3755 switch (MF->getTarget().Options.TLSSize) {
3756 default:
3757 llvm_unreachable("Unexpected TLS size");
3758 case 12: {
3759 // add x0, x0, :tprel_lo12:a
3760 ConstrainRegOps(
3761 MIB.buildInstr(AArch64::ADDXri, {I.getOperand(0).getReg()},
3762 {ThreadBase})
3764 .addImm(0));
3765 break;
3766 }
3767 case 24: {
3768 // add x0, x0, :tprel_hi12:a
3769 // add x0, x0, :tprel_lo12_nc:a
3770 Register Addr = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
3771 ConstrainRegOps(
3772 MIB.buildInstr(AArch64::ADDXri, {Addr}, {ThreadBase})
3774 .addImm(0));
3775 ConstrainRegOps(
3776 MIB.buildInstr(AArch64::ADDXri, {I.getOperand(0).getReg()}, {Addr})
3777 .addGlobalAddress(GV, 0,
3780 .addImm(0));
3781 break;
3782 }
3783 case 32: {
3784 // movz x0, #:tprel_g1:a
3785 // movk x0, #:tprel_g0_nc:a
3786 // add x0, x1, x0
3787 Register Addr = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
3788 ConstrainRegOps(
3789 MIB.buildInstr(AArch64::MOVZXi, {Addr}, {})
3791 .addImm(16));
3792 Register Addr2 = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
3793 ConstrainRegOps(MIB.buildInstr(AArch64::MOVKXi, {Addr2}, {Addr})
3794 .addGlobalAddress(GV, 0,
3797 .addImm(0));
3798 ConstrainRegOps(MIB.buildInstr(AArch64::ADDXrr, {I.getOperand(0).getReg()},
3799 {ThreadBase, Addr2}));
3800 break;
3801 }
3802 case 48: {
3803 // movz x0, #:tprel_g2:a
3804 // movk x0, #:tprel_g1_nc:a
3805 // movk x0, #:tprel_g0_nc:a
3806 // add x0, x1, x0
3807 Register Addr = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
3808 ConstrainRegOps(
3809 MIB.buildInstr(AArch64::MOVZXi, {Addr}, {})
3811 .addImm(32));
3812 Register Addr2 = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
3813 ConstrainRegOps(MIB.buildInstr(AArch64::MOVKXi, {Addr2}, {Addr})
3814 .addGlobalAddress(GV, 0,
3817 .addImm(16));
3818 Register Addr3 = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
3819 ConstrainRegOps(MIB.buildInstr(AArch64::MOVKXi, {Addr3}, {Addr2})
3820 .addGlobalAddress(GV, 0,
3823 .addImm(0));
3824 ConstrainRegOps(MIB.buildInstr(AArch64::ADDXrr, {I.getOperand(0).getReg()},
3825 {ThreadBase, Addr3}));
3826 break;
3827 }
3828 }
3829 I.eraseFromParent();
3830 return true;
3831}
3832
3833// TLS lowering below mirrors the corresponding DAGISel implementation.
3834// See LowerELFTLSModel() for details.
3835bool AArch64InstructionSelector::selectTLSGlobalValueELF(
3836 MachineInstr &I, MachineRegisterInfo &MRI) {
3837 const GlobalValue *GV = I.getOperand(1).getGlobal();
3838 auto *FuncInfo = MF->getInfo<AArch64FunctionInfo>();
3840 AArch64::getELFTLSModel(GV, TM, FuncInfo->hasELFSignedGOT());
3841
3842 Register TPOff = MRI.createVirtualRegister(&AArch64::GPR64commonRegClass);
3843 switch (Model) {
3845 return selectTLSLocalExecELF(GV, I, MRI);
3847 MIB.buildInstr(AArch64::LOADgot, {TPOff}, {})
3848 .addGlobalAddress(GV, 0, AArch64II::MO_TLS);
3849 break;
3852#ifndef NDEBUG
3853 SMEAttrs Attrs = MF->getInfo<AArch64FunctionInfo>()->getSMEFnAttrs();
3854 assert(!Attrs.hasZAState() && !Attrs.hasStreamingInterfaceOrBody() &&
3855 !Attrs.hasStreamingCompatibleInterface() &&
3856 "unsupported SME features reached GlobalISel TLS lowering");
3857#endif
3858 unsigned Opcode = FuncInfo->hasELFSignedGOT()
3859 ? AArch64::TLSDESC_AUTH_CALLSEQ
3860 : AArch64::TLSDESC_CALLSEQ;
3861
3862 if (TLSModel::GeneralDynamic == Model) {
3863 MIB.buildInstr(Opcode, {}, {}).addGlobalAddress(GV, 0, AArch64II::MO_TLS);
3864 MIB.buildCopy(TPOff, Register(AArch64::X0));
3865 break;
3866 }
3868 // These accesses will need deduplicating if there's more than one.
3870
3871 MIB.buildInstr(Opcode, {}, {})
3872 .addExternalSymbol("_TLS_MODULE_BASE_", AArch64II::MO_TLS);
3873 auto Copy = MIB.buildCopy(LLT::scalar(64), Register(AArch64::X0));
3874 auto Add1 =
3875 MIB.buildInstr(AArch64::ADDXri, {LLT::scalar(64)}, {Copy.getReg(0)})
3876 .addGlobalAddress(GV, 0, AArch64II::MO_TLS | AArch64II::MO_HI12)
3877 .addImm(0);
3878 auto Add2 =
3879 MIB.buildInstr(AArch64::ADDXri, {TPOff}, {Add1.getReg(0)})
3880 .addGlobalAddress(GV, 0,
3883 .addImm(0);
3886 }
3887 }
3888 Register ThreadBase = MRI.createGenericVirtualRegister(LLT::pointer(0, 64));
3889 MIB.buildInstr(AArch64::MOVbaseTLS, {ThreadBase}, {});
3890 auto Add = MIB.buildInstr(AArch64::ADDXrr, {I.getOperand(0).getReg()},
3891 {ThreadBase, TPOff});
3893
3894 I.eraseFromParent();
3895 return true;
3896}
3897
3898bool AArch64InstructionSelector::selectTLSGlobalValueMachO(
3899 MachineInstr &I, MachineRegisterInfo &MRI) {
3900 const auto &GlobalOp = I.getOperand(1);
3901 assert(GlobalOp.getOffset() == 0 &&
3902 "Shouldn't have an offset on TLS globals!");
3903
3904 const GlobalValue &GV = *GlobalOp.getGlobal();
3905 MF->getFrameInfo().setAdjustsStack(true);
3906 auto LoadGOT =
3907 MIB.buildInstr(AArch64::LOADgot, {&AArch64::GPR64commonRegClass}, {})
3908 .addGlobalAddress(&GV, 0, AArch64II::MO_TLS);
3909
3910 auto Load = MIB.buildInstr(AArch64::LDRXui, {&AArch64::GPR64commonRegClass},
3911 {LoadGOT.getReg(0)})
3912 .addImm(0);
3913
3914 MIB.buildCopy(Register(AArch64::X0), LoadGOT.getReg(0));
3915 // TLS calls preserve all registers except those that absolutely must be
3916 // trashed: X0 (it takes an argument), LR (it's a call) and NZCV (let's not be
3917 // silly).
3918 unsigned Opcode = getBLRCallOpcode(*MF);
3919
3920 // With ptrauth-calls, the tlv access thunk pointer is authenticated (IA, 0).
3921 if (MF->getFunction().hasFnAttribute("ptrauth-calls")) {
3922 assert(Opcode == AArch64::BLR);
3923 Opcode = AArch64::BLRAAZ;
3924 }
3925
3926 MIB.buildInstr(Opcode, {}, {Load})
3927 .setOperandDead(1) // implicit-def $lr
3928 .addUse(AArch64::X0, RegState::Implicit)
3929 .addDef(AArch64::X0, RegState::Implicit)
3930 .addRegMask(TRI.getTLSCallPreservedMask());
3931
3932 MIB.buildCopy(I.getOperand(0).getReg(), Register(AArch64::X0));
3933 RBI.constrainGenericRegister(I.getOperand(0).getReg(), AArch64::GPR64RegClass,
3934 MRI);
3935 I.eraseFromParent();
3936 return true;
3937}
3938
3939bool AArch64InstructionSelector::selectTLSGlobalValue(
3940 MachineInstr &I, MachineRegisterInfo &MRI) {
3941 // We don't support instructions with emulated TLS variables yet.
3942 if (TM.useEmulatedTLS())
3943 return false;
3944
3945 if (STI.isTargetELF())
3946 return selectTLSGlobalValueELF(I, MRI);
3947
3948 if (STI.isTargetMachO())
3949 return selectTLSGlobalValueMachO(I, MRI);
3950
3951 return false;
3952}
3953
3954MachineInstr *AArch64InstructionSelector::emitScalarToVector(
3955 unsigned EltSize, const TargetRegisterClass *DstRC, Register Scalar,
3956 MachineIRBuilder &MIRBuilder) const {
3957 auto Undef = MIRBuilder.buildInstr(TargetOpcode::IMPLICIT_DEF, {DstRC}, {});
3958
3959 auto BuildFn = [&](unsigned SubregIndex) {
3960 auto Ins =
3961 MIRBuilder
3962 .buildInstr(TargetOpcode::INSERT_SUBREG, {DstRC}, {Undef, Scalar})
3963 .addImm(SubregIndex);
3966 return &*Ins;
3967 };
3968
3969 switch (EltSize) {
3970 case 8:
3971 return BuildFn(AArch64::bsub);
3972 case 16:
3973 return BuildFn(AArch64::hsub);
3974 case 32:
3975 return BuildFn(AArch64::ssub);
3976 case 64:
3977 return BuildFn(AArch64::dsub);
3978 default:
3979 return nullptr;
3980 }
3981}
3982
3983MachineInstr *
3984AArch64InstructionSelector::emitNarrowVector(Register DstReg, Register SrcReg,
3985 MachineIRBuilder &MIB,
3986 MachineRegisterInfo &MRI) const {
3987 LLT DstTy = MRI.getType(DstReg);
3988 const TargetRegisterClass *RC =
3989 getRegClassForTypeOnBank(DstTy, *RBI.getRegBank(SrcReg, MRI, TRI));
3990 if (RC != &AArch64::FPR32RegClass && RC != &AArch64::FPR64RegClass) {
3991 LLVM_DEBUG(dbgs() << "Unsupported register class!\n");
3992 return nullptr;
3993 }
3994 unsigned SubReg = 0;
3995 if (!getSubRegForClass(RC, TRI, SubReg))
3996 return nullptr;
3997 if (SubReg != AArch64::ssub && SubReg != AArch64::dsub) {
3998 LLVM_DEBUG(dbgs() << "Unsupported destination size! ("
3999 << DstTy.getSizeInBits() << "\n");
4000 return nullptr;
4001 }
4002 auto Copy = MIB.buildInstr(TargetOpcode::COPY, {DstReg}, {})
4003 .addReg(SrcReg, {}, SubReg);
4004 RBI.constrainGenericRegister(DstReg, *RC, MRI);
4005 return Copy;
4006}
4007
4008bool AArch64InstructionSelector::selectMergeValues(
4009 MachineInstr &I, MachineRegisterInfo &MRI) {
4010 assert(I.getOpcode() == TargetOpcode::G_MERGE_VALUES && "unexpected opcode");
4011 const LLT DstTy = MRI.getType(I.getOperand(0).getReg());
4012 const LLT SrcTy = MRI.getType(I.getOperand(1).getReg());
4013 assert(!DstTy.isVector() && !SrcTy.isVector() && "invalid merge operation");
4014 const RegisterBank &RB = *RBI.getRegBank(I.getOperand(1).getReg(), MRI, TRI);
4015
4016 if (I.getNumOperands() != 3)
4017 return false;
4018
4019 // Merging 2 s64s into an s128.
4020 if (DstTy == LLT::scalar(128)) {
4021 if (SrcTy.getSizeInBits() != 64)
4022 return false;
4023 Register DstReg = I.getOperand(0).getReg();
4024 Register Src1Reg = I.getOperand(1).getReg();
4025 Register Src2Reg = I.getOperand(2).getReg();
4026 auto Tmp = MIB.buildInstr(TargetOpcode::IMPLICIT_DEF, {DstTy}, {});
4027 MachineInstr *InsMI = emitLaneInsert(std::nullopt, Tmp.getReg(0), Src1Reg,
4028 /* LaneIdx */ 0, RB, MIB);
4029 if (!InsMI)
4030 return false;
4031 MachineInstr *Ins2MI = emitLaneInsert(DstReg, InsMI->getOperand(0).getReg(),
4032 Src2Reg, /* LaneIdx */ 1, RB, MIB);
4033 if (!Ins2MI)
4034 return false;
4037 I.eraseFromParent();
4038 return true;
4039 }
4040
4041 if (RB.getID() != AArch64::GPRRegBankID)
4042 return false;
4043
4044 if (DstTy.getSizeInBits() != 64 || SrcTy.getSizeInBits() != 32)
4045 return false;
4046
4047 auto *DstRC = &AArch64::GPR64RegClass;
4048 Register SubToRegDef = MRI.createVirtualRegister(DstRC);
4049 MachineInstr &SubRegMI = *BuildMI(*I.getParent(), I, I.getDebugLoc(),
4050 TII.get(TargetOpcode::SUBREG_TO_REG))
4051 .addDef(SubToRegDef)
4052 .addUse(I.getOperand(1).getReg())
4053 .addImm(AArch64::sub_32);
4054 Register SubToRegDef2 = MRI.createVirtualRegister(DstRC);
4055 // Need to anyext the second scalar before we can use bfm
4056 MachineInstr &SubRegMI2 = *BuildMI(*I.getParent(), I, I.getDebugLoc(),
4057 TII.get(TargetOpcode::SUBREG_TO_REG))
4058 .addDef(SubToRegDef2)
4059 .addUse(I.getOperand(2).getReg())
4060 .addImm(AArch64::sub_32);
4061 MachineInstr &BFM =
4062 *BuildMI(*I.getParent(), I, I.getDebugLoc(), TII.get(AArch64::BFMXri))
4063 .addDef(I.getOperand(0).getReg())
4064 .addUse(SubToRegDef)
4065 .addUse(SubToRegDef2)
4066 .addImm(32)
4067 .addImm(31);
4068 constrainSelectedInstRegOperands(SubRegMI, TII, TRI, RBI);
4069 constrainSelectedInstRegOperands(SubRegMI2, TII, TRI, RBI);
4071 I.eraseFromParent();
4072 return true;
4073}
4074
4075static bool getLaneCopyOpcode(unsigned &CopyOpc, unsigned &ExtractSubReg,
4076 const unsigned EltSize) {
4077 // Choose a lane copy opcode and subregister based off of the size of the
4078 // vector's elements.
4079 switch (EltSize) {
4080 case 8:
4081 CopyOpc = AArch64::DUPi8;
4082 ExtractSubReg = AArch64::bsub;
4083 break;
4084 case 16:
4085 CopyOpc = AArch64::DUPi16;
4086 ExtractSubReg = AArch64::hsub;
4087 break;
4088 case 32:
4089 CopyOpc = AArch64::DUPi32;
4090 ExtractSubReg = AArch64::ssub;
4091 break;
4092 case 64:
4093 CopyOpc = AArch64::DUPi64;
4094 ExtractSubReg = AArch64::dsub;
4095 break;
4096 default:
4097 // Unknown size, bail out.
4098 LLVM_DEBUG(dbgs() << "Elt size '" << EltSize << "' unsupported.\n");
4099 return false;
4100 }
4101 return true;
4102}
4103
4104MachineInstr *AArch64InstructionSelector::emitExtractVectorElt(
4105 std::optional<Register> DstReg, const RegisterBank &DstRB, LLT ScalarTy,
4106 Register VecReg, unsigned LaneIdx, MachineIRBuilder &MIRBuilder) const {
4107 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
4108 unsigned CopyOpc = 0;
4109 unsigned ExtractSubReg = 0;
4110 if (!getLaneCopyOpcode(CopyOpc, ExtractSubReg, ScalarTy.getSizeInBits())) {
4111 LLVM_DEBUG(
4112 dbgs() << "Couldn't determine lane copy opcode for instruction.\n");
4113 return nullptr;
4114 }
4115
4116 const TargetRegisterClass *DstRC =
4117 getRegClassForTypeOnBank(ScalarTy, DstRB, true);
4118 if (!DstRC) {
4119 LLVM_DEBUG(dbgs() << "Could not determine destination register class.\n");
4120 return nullptr;
4121 }
4122
4123 const RegisterBank &VecRB = *RBI.getRegBank(VecReg, MRI, TRI);
4124 const LLT &VecTy = MRI.getType(VecReg);
4125 const TargetRegisterClass *VecRC =
4126 getRegClassForTypeOnBank(VecTy, VecRB, true);
4127 if (!VecRC) {
4128 LLVM_DEBUG(dbgs() << "Could not determine source register class.\n");
4129 return nullptr;
4130 }
4131
4132 // The register that we're going to copy into.
4133 Register InsertReg = VecReg;
4134 if (!DstReg)
4135 DstReg = MRI.createVirtualRegister(DstRC);
4136 // If the lane index is 0, we just use a subregister COPY.
4137 if (LaneIdx == 0) {
4138 auto Copy = MIRBuilder.buildInstr(TargetOpcode::COPY, {*DstReg}, {})
4139 .addReg(VecReg, {}, ExtractSubReg);
4140 RBI.constrainGenericRegister(*DstReg, *DstRC, MRI);
4141 return &*Copy;
4142 }
4143
4144 // Lane copies require 128-bit wide registers. If we're dealing with an
4145 // unpacked vector, then we need to move up to that width. Insert an implicit
4146 // def and a subregister insert to get us there.
4147 if (VecTy.getSizeInBits() != 128) {
4148 MachineInstr *ScalarToVector = emitScalarToVector(
4149 VecTy.getSizeInBits(), &AArch64::FPR128RegClass, VecReg, MIRBuilder);
4150 if (!ScalarToVector)
4151 return nullptr;
4152 InsertReg = ScalarToVector->getOperand(0).getReg();
4153 }
4154
4155 MachineInstr *LaneCopyMI =
4156 MIRBuilder.buildInstr(CopyOpc, {*DstReg}, {InsertReg}).addImm(LaneIdx);
4157 constrainSelectedInstRegOperands(*LaneCopyMI, TII, TRI, RBI);
4158
4159 // Make sure that we actually constrain the initial copy.
4160 RBI.constrainGenericRegister(*DstReg, *DstRC, MRI);
4161 return LaneCopyMI;
4162}
4163
4164bool AArch64InstructionSelector::selectExtractElt(
4165 MachineInstr &I, MachineRegisterInfo &MRI) {
4166 assert(I.getOpcode() == TargetOpcode::G_EXTRACT_VECTOR_ELT &&
4167 "unexpected opcode!");
4168 Register DstReg = I.getOperand(0).getReg();
4169 const LLT NarrowTy = MRI.getType(DstReg);
4170 const Register SrcReg = I.getOperand(1).getReg();
4171 const LLT WideTy = MRI.getType(SrcReg);
4172 assert(WideTy.getSizeInBits() >= NarrowTy.getSizeInBits() &&
4173 "source register size too small!");
4174 assert(!NarrowTy.isVector() && "cannot extract vector into vector!");
4175
4176 // Need the lane index to determine the correct copy opcode.
4177 MachineOperand &LaneIdxOp = I.getOperand(2);
4178 assert(LaneIdxOp.isReg() && "Lane index operand was not a register?");
4179
4180 // Find the index to extract from.
4181 auto VRegAndVal = getIConstantVRegValWithLookThrough(LaneIdxOp.getReg(), MRI);
4182 if (!VRegAndVal)
4183 return false;
4184 unsigned LaneIdx = VRegAndVal->Value.getSExtValue();
4185
4186 const RegisterBank &DstRB = *RBI.getRegBank(DstReg, MRI, TRI);
4187 if (DstRB.getID() == AArch64::GPRRegBankID) {
4188 unsigned Opcode;
4189 switch (WideTy.getScalarSizeInBits()) {
4190 case 8:
4191 Opcode = AArch64::UMOVvi8;
4192 break;
4193 case 16:
4194 Opcode = AArch64::UMOVvi16;
4195 break;
4196 case 32:
4197 Opcode = AArch64::UMOVvi32;
4198 break;
4199 default:
4200 return false;
4201 }
4202
4203 if (WideTy.getSizeInBits() != 128) {
4204 MachineInstr *ScalarToVector = emitScalarToVector(
4205 WideTy.getSizeInBits(), &AArch64::FPR128RegClass, SrcReg, MIB);
4206 assert(ScalarToVector && "Didn't expect emitScalarToVector to fail!");
4207 I.getOperand(1).setReg(ScalarToVector->getOperand(0).getReg());
4208 }
4209
4210 I.setDesc(TII.get(Opcode));
4211 I.getOperand(2).ChangeToImmediate(LaneIdx);
4213 return true;
4214 }
4215
4216 MachineInstr *Extract = emitExtractVectorElt(DstReg, DstRB, NarrowTy, SrcReg,
4217 LaneIdx, MIB);
4218 if (!Extract)
4219 return false;
4220
4221 I.eraseFromParent();
4222 return true;
4223}
4224
4225bool AArch64InstructionSelector::selectSplitVectorUnmerge(
4226 MachineInstr &I, MachineRegisterInfo &MRI) {
4227 unsigned NumElts = I.getNumOperands() - 1;
4228 Register SrcReg = I.getOperand(NumElts).getReg();
4229 const LLT NarrowTy = MRI.getType(I.getOperand(0).getReg());
4230 const LLT SrcTy = MRI.getType(SrcReg);
4231
4232 assert(NarrowTy.isVector() && "Expected an unmerge into vectors");
4233 if (SrcTy.getSizeInBits() > 128) {
4234 LLVM_DEBUG(dbgs() << "Unexpected vector type for vec split unmerge");
4235 return false;
4236 }
4237
4238 // We implement a split vector operation by treating the sub-vectors as
4239 // scalars and extracting them.
4240 const RegisterBank &DstRB =
4241 *RBI.getRegBank(I.getOperand(0).getReg(), MRI, TRI);
4242 for (unsigned OpIdx = 0; OpIdx < NumElts; ++OpIdx) {
4243 Register Dst = I.getOperand(OpIdx).getReg();
4244 MachineInstr *Extract =
4245 emitExtractVectorElt(Dst, DstRB, NarrowTy, SrcReg, OpIdx, MIB);
4246 if (!Extract)
4247 return false;
4248 }
4249 I.eraseFromParent();
4250 return true;
4251}
4252
4253bool AArch64InstructionSelector::selectUnmergeValues(MachineInstr &I,
4254 MachineRegisterInfo &MRI) {
4255 assert(I.getOpcode() == TargetOpcode::G_UNMERGE_VALUES &&
4256 "unexpected opcode");
4257
4258 // The last operand is the vector source register, and every other operand is
4259 // a register to unpack into.
4260 unsigned NumElts = I.getNumOperands() - 1;
4261 Register SrcReg = I.getOperand(NumElts).getReg();
4262 Register LoReg = I.getOperand(0).getReg();
4263 Register HiReg = I.getOperand(1).getReg();
4264 const LLT NarrowTy = MRI.getType(LoReg);
4265 const LLT WideTy = MRI.getType(SrcReg);
4266 const RegisterBank &LoRB = *RBI.getRegBank(LoReg, MRI, TRI);
4267 const RegisterBank &HiRB = *RBI.getRegBank(HiReg, MRI, TRI);
4268 const RegisterBank &SrcRB = *RBI.getRegBank(SrcReg, MRI, TRI);
4269
4270 // Handle unmerging a 128-bit FPR value into two 64-bit GPR values.
4271 if (NarrowTy == LLT::scalar(64) && WideTy == LLT::scalar(128) &&
4272 LoRB.getID() == AArch64::GPRRegBankID &&
4273 HiRB.getID() == AArch64::GPRRegBankID &&
4274 SrcRB.getID() == AArch64::FPRRegBankID) {
4275 MachineInstr &Lo = *BuildMI(*I.getParent(), I, I.getDebugLoc(),
4276 TII.get(AArch64::UMOVvi64), LoReg)
4277 .addUse(SrcReg)
4278 .addImm(0);
4279 MachineInstr &Hi = *BuildMI(*I.getParent(), I, I.getDebugLoc(),
4280 TII.get(AArch64::UMOVvi64), HiReg)
4281 .addUse(SrcReg)
4282 .addImm(1);
4285 I.eraseFromParent();
4286 return true;
4287 }
4288
4289 // TODO: Handle other unmerges into GPRs and from scalars to scalars.
4290 if (LoRB.getID() != AArch64::FPRRegBankID ||
4291 HiRB.getID() != AArch64::FPRRegBankID) {
4292 LLVM_DEBUG(dbgs() << "Unmerging vector-to-gpr and scalar-to-scalar "
4293 "currently unsupported.\n");
4294 return false;
4295 }
4296
4297 assert(WideTy.getSizeInBits() > NarrowTy.getSizeInBits() &&
4298 "source register size too small!");
4299
4300 if (!NarrowTy.isScalar())
4301 return selectSplitVectorUnmerge(I, MRI);
4302
4303 // Choose a lane copy opcode and subregister based off of the size of the
4304 // vector's elements.
4305 unsigned CopyOpc = 0;
4306 unsigned ExtractSubReg = 0;
4307 if (!getLaneCopyOpcode(CopyOpc, ExtractSubReg, NarrowTy.getSizeInBits()))
4308 return false;
4309
4310 // Set up for the lane copies.
4311 MachineBasicBlock &MBB = *I.getParent();
4312
4313 // Stores the registers we'll be copying from.
4314 SmallVector<Register, 4> InsertRegs;
4315
4316 // We'll use the first register twice, so we only need NumElts-1 registers.
4317 unsigned NumInsertRegs = NumElts - 1;
4318
4319 // If our elements fit into exactly 128 bits, then we can copy from the source
4320 // directly. Otherwise, we need to do a bit of setup with some subregister
4321 // inserts.
4322 if (NarrowTy.getSizeInBits() * NumElts == 128) {
4323 InsertRegs.assign(NumInsertRegs, SrcReg);
4324 } else {
4325 // No. We have to perform subregister inserts. For each insert, create an
4326 // implicit def and a subregister insert, and save the register we create.
4327 // For scalar sources, treat as a pseudo-vector of NarrowTy elements.
4328 unsigned EltSize = WideTy.isVector() ? WideTy.getScalarSizeInBits()
4329 : NarrowTy.getSizeInBits();
4330 const TargetRegisterClass *RC = getRegClassForTypeOnBank(
4331 LLT::fixed_vector(NumElts, EltSize), *RBI.getRegBank(SrcReg, MRI, TRI));
4332 unsigned SubReg = 0;
4333 bool Found = getSubRegForClass(RC, TRI, SubReg);
4334 (void)Found;
4335 assert(Found && "expected to find last operand's subeg idx");
4336 for (unsigned Idx = 0; Idx < NumInsertRegs; ++Idx) {
4337 Register ImpDefReg = MRI.createVirtualRegister(&AArch64::FPR128RegClass);
4338 MachineInstr &ImpDefMI =
4339 *BuildMI(MBB, I, I.getDebugLoc(), TII.get(TargetOpcode::IMPLICIT_DEF),
4340 ImpDefReg);
4341
4342 // Now, create the subregister insert from SrcReg.
4343 Register InsertReg = MRI.createVirtualRegister(&AArch64::FPR128RegClass);
4344 MachineInstr &InsMI =
4345 *BuildMI(MBB, I, I.getDebugLoc(),
4346 TII.get(TargetOpcode::INSERT_SUBREG), InsertReg)
4347 .addUse(ImpDefReg)
4348 .addUse(SrcReg)
4349 .addImm(SubReg);
4350
4351 constrainSelectedInstRegOperands(ImpDefMI, TII, TRI, RBI);
4353
4354 // Save the register so that we can copy from it after.
4355 InsertRegs.push_back(InsertReg);
4356 }
4357 }
4358
4359 // Now that we've created any necessary subregister inserts, we can
4360 // create the copies.
4361 //
4362 // Perform the first copy separately as a subregister copy.
4363 Register CopyTo = I.getOperand(0).getReg();
4364 auto FirstCopy = MIB.buildInstr(TargetOpcode::COPY, {CopyTo}, {})
4365 .addReg(InsertRegs[0], {}, ExtractSubReg);
4366 constrainSelectedInstRegOperands(*FirstCopy, TII, TRI, RBI);
4367
4368 // Now, perform the remaining copies as vector lane copies.
4369 unsigned LaneIdx = 1;
4370 for (Register InsReg : InsertRegs) {
4371 Register CopyTo = I.getOperand(LaneIdx).getReg();
4372 MachineInstr &CopyInst =
4373 *BuildMI(MBB, I, I.getDebugLoc(), TII.get(CopyOpc), CopyTo)
4374 .addUse(InsReg)
4375 .addImm(LaneIdx);
4376 constrainSelectedInstRegOperands(CopyInst, TII, TRI, RBI);
4377 ++LaneIdx;
4378 }
4379
4380 // Separately constrain the first copy's destination. Because of the
4381 // limitation in constrainOperandRegClass, we can't guarantee that this will
4382 // actually be constrained. So, do it ourselves using the second operand.
4383 const TargetRegisterClass *RC =
4384 MRI.getRegClassOrNull(I.getOperand(1).getReg());
4385 if (!RC) {
4386 LLVM_DEBUG(dbgs() << "Couldn't constrain copy destination.\n");
4387 return false;
4388 }
4389
4390 RBI.constrainGenericRegister(CopyTo, *RC, MRI);
4391 I.eraseFromParent();
4392 return true;
4393}
4394
4395bool AArch64InstructionSelector::selectConcatVectors(
4396 MachineInstr &I, MachineRegisterInfo &MRI) {
4397 assert(I.getOpcode() == TargetOpcode::G_CONCAT_VECTORS &&
4398 "Unexpected opcode");
4399 Register Dst = I.getOperand(0).getReg();
4400 Register Op1 = I.getOperand(1).getReg();
4401 Register Op2 = I.getOperand(2).getReg();
4402 MachineInstr *ConcatMI = emitVectorConcat(Dst, Op1, Op2, MIB);
4403 if (!ConcatMI)
4404 return false;
4405 I.eraseFromParent();
4406 return true;
4407}
4408
4409unsigned
4410AArch64InstructionSelector::emitConstantPoolEntry(const Constant *CPVal,
4411 MachineFunction &MF) const {
4412 Type *CPTy = CPVal->getType();
4414
4415 MachineConstantPool *MCP = MF.getConstantPool();
4416 return MCP->getConstantPoolIndex(CPVal, Alignment);
4417}
4418
4419MachineInstr *AArch64InstructionSelector::emitLoadFromConstantPool(
4420 const Constant *CPVal, MachineIRBuilder &MIRBuilder) const {
4421 const TargetRegisterClass *RC;
4422 unsigned Opc;
4423 bool IsTiny = TM.getCodeModel() == CodeModel::Tiny;
4424 unsigned Size = MIRBuilder.getDataLayout().getTypeStoreSize(CPVal->getType());
4425 switch (Size) {
4426 case 16:
4427 RC = &AArch64::FPR128RegClass;
4428 Opc = IsTiny ? AArch64::LDRQl : AArch64::LDRQui;
4429 break;
4430 case 8:
4431 RC = &AArch64::FPR64RegClass;
4432 Opc = IsTiny ? AArch64::LDRDl : AArch64::LDRDui;
4433 break;
4434 case 4:
4435 RC = &AArch64::FPR32RegClass;
4436 Opc = IsTiny ? AArch64::LDRSl : AArch64::LDRSui;
4437 break;
4438 case 2:
4439 RC = &AArch64::FPR16RegClass;
4440 Opc = AArch64::LDRHui;
4441 break;
4442 default:
4443 LLVM_DEBUG(dbgs() << "Could not load from constant pool of type "
4444 << *CPVal->getType());
4445 return nullptr;
4446 }
4447
4448 MachineInstr *LoadMI = nullptr;
4449 auto &MF = MIRBuilder.getMF();
4450 unsigned CPIdx = emitConstantPoolEntry(CPVal, MF);
4451 if (IsTiny && (Size == 16 || Size == 8 || Size == 4)) {
4452 // Use load(literal) for tiny code model.
4453 LoadMI = &*MIRBuilder.buildInstr(Opc, {RC}, {}).addConstantPoolIndex(CPIdx);
4454 } else {
4455 auto Adrp =
4456 MIRBuilder.buildInstr(AArch64::ADRP, {&AArch64::GPR64RegClass}, {})
4457 .addConstantPoolIndex(CPIdx, 0, AArch64II::MO_PAGE);
4458
4459 LoadMI = &*MIRBuilder.buildInstr(Opc, {RC}, {Adrp})
4460 .addConstantPoolIndex(
4462
4464 }
4465
4466 MachinePointerInfo PtrInfo = MachinePointerInfo::getConstantPool(MF);
4467 LoadMI->addMemOperand(MF, MF.getMachineMemOperand(PtrInfo,
4469 Size, Align(Size)));
4471 return LoadMI;
4472}
4473
4474/// Return an <Opcode, SubregIndex> pair to do an vector elt insert of a given
4475/// size and RB.
4476static std::pair<unsigned, unsigned>
4477getInsertVecEltOpInfo(const RegisterBank &RB, unsigned EltSize) {
4478 unsigned Opc, SubregIdx;
4479 if (RB.getID() == AArch64::GPRRegBankID) {
4480 if (EltSize == 8) {
4481 Opc = AArch64::INSvi8gpr;
4482 SubregIdx = AArch64::bsub;
4483 } else if (EltSize == 16) {
4484 Opc = AArch64::INSvi16gpr;
4485 SubregIdx = AArch64::ssub;
4486 } else if (EltSize == 32) {
4487 Opc = AArch64::INSvi32gpr;
4488 SubregIdx = AArch64::ssub;
4489 } else if (EltSize == 64) {
4490 Opc = AArch64::INSvi64gpr;
4491 SubregIdx = AArch64::dsub;
4492 } else {
4493 llvm_unreachable("invalid elt size!");
4494 }
4495 } else {
4496 if (EltSize == 8) {
4497 Opc = AArch64::INSvi8lane;
4498 SubregIdx = AArch64::bsub;
4499 } else if (EltSize == 16) {
4500 Opc = AArch64::INSvi16lane;
4501 SubregIdx = AArch64::hsub;
4502 } else if (EltSize == 32) {
4503 Opc = AArch64::INSvi32lane;
4504 SubregIdx = AArch64::ssub;
4505 } else if (EltSize == 64) {
4506 Opc = AArch64::INSvi64lane;
4507 SubregIdx = AArch64::dsub;
4508 } else {
4509 llvm_unreachable("invalid elt size!");
4510 }
4511 }
4512 return std::make_pair(Opc, SubregIdx);
4513}
4514
4515MachineInstr *AArch64InstructionSelector::emitInstr(
4516 unsigned Opcode, std::initializer_list<llvm::DstOp> DstOps,
4517 std::initializer_list<llvm::SrcOp> SrcOps, MachineIRBuilder &MIRBuilder,
4518 const ComplexRendererFns &RenderFns) const {
4519 assert(Opcode && "Expected an opcode?");
4520 assert(!isPreISelGenericOpcode(Opcode) &&
4521 "Function should only be used to produce selected instructions!");
4522 auto MI = MIRBuilder.buildInstr(Opcode, DstOps, SrcOps);
4523 if (RenderFns)
4524 for (auto &Fn : *RenderFns)
4525 Fn(MI);
4527 return &*MI;
4528}
4529
4530MachineInstr *AArch64InstructionSelector::emitAddSub(
4531 const std::array<std::array<unsigned, 2>, 5> &AddrModeAndSizeToOpcode,
4532 Register Dst, MachineOperand &LHS, MachineOperand &RHS,
4533 MachineIRBuilder &MIRBuilder) const {
4534 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4535 assert(LHS.isReg() && RHS.isReg() && "Expected register operands?");
4536 auto Ty = MRI.getType(LHS.getReg());
4537 assert(!Ty.isVector() && "Expected a scalar or pointer?");
4538 unsigned Size = Ty.getSizeInBits();
4539 assert((Size == 32 || Size == 64) && "Expected a 32-bit or 64-bit type only");
4540 bool Is32Bit = Size == 32;
4541
4542 // INSTRri form with positive arithmetic immediate.
4543 if (auto Fns = selectArithImmed(RHS))
4544 return emitInstr(AddrModeAndSizeToOpcode[0][Is32Bit], {Dst}, {LHS},
4545 MIRBuilder, Fns);
4546
4547 // INSTRri form with negative arithmetic immediate.
4548 if (auto Fns = selectNegArithImmed(RHS))
4549 return emitInstr(AddrModeAndSizeToOpcode[3][Is32Bit], {Dst}, {LHS},
4550 MIRBuilder, Fns);
4551
4552 // INSTRrx form.
4553 if (auto Fns = selectArithExtendedRegister(RHS))
4554 return emitInstr(AddrModeAndSizeToOpcode[4][Is32Bit], {Dst}, {LHS},
4555 MIRBuilder, Fns);
4556
4557 // INSTRrs form.
4558 if (auto Fns = selectShiftedRegister(RHS))
4559 return emitInstr(AddrModeAndSizeToOpcode[1][Is32Bit], {Dst}, {LHS},
4560 MIRBuilder, Fns);
4561 return emitInstr(AddrModeAndSizeToOpcode[2][Is32Bit], {Dst}, {LHS, RHS},
4562 MIRBuilder);
4563}
4564
4565MachineInstr *
4566AArch64InstructionSelector::emitADD(Register DefReg, MachineOperand &LHS,
4567 MachineOperand &RHS,
4568 MachineIRBuilder &MIRBuilder) const {
4569 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4570 {{AArch64::ADDXri, AArch64::ADDWri},
4571 {AArch64::ADDXrs, AArch64::ADDWrs},
4572 {AArch64::ADDXrr, AArch64::ADDWrr},
4573 {AArch64::SUBXri, AArch64::SUBWri},
4574 {AArch64::ADDXrx, AArch64::ADDWrx}}};
4575 return emitAddSub(OpcTable, DefReg, LHS, RHS, MIRBuilder);
4576}
4577
4578MachineInstr *
4579AArch64InstructionSelector::emitADDS(Register Dst, MachineOperand &LHS,
4580 MachineOperand &RHS,
4581 MachineIRBuilder &MIRBuilder) const {
4582 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4583 {{AArch64::ADDSXri, AArch64::ADDSWri},
4584 {AArch64::ADDSXrs, AArch64::ADDSWrs},
4585 {AArch64::ADDSXrr, AArch64::ADDSWrr},
4586 {AArch64::SUBSXri, AArch64::SUBSWri},
4587 {AArch64::ADDSXrx, AArch64::ADDSWrx}}};
4588 return emitAddSub(OpcTable, Dst, LHS, RHS, MIRBuilder);
4589}
4590
4591MachineInstr *
4592AArch64InstructionSelector::emitSUBS(Register Dst, MachineOperand &LHS,
4593 MachineOperand &RHS,
4594 MachineIRBuilder &MIRBuilder) const {
4595 const std::array<std::array<unsigned, 2>, 5> OpcTable{
4596 {{AArch64::SUBSXri, AArch64::SUBSWri},
4597 {AArch64::SUBSXrs, AArch64::SUBSWrs},
4598 {AArch64::SUBSXrr, AArch64::SUBSWrr},
4599 {AArch64::ADDSXri, AArch64::ADDSWri},
4600 {AArch64::SUBSXrx, AArch64::SUBSWrx}}};
4601 return emitAddSub(OpcTable, Dst, LHS, RHS, MIRBuilder);
4602}
4603
4604MachineInstr *
4605AArch64InstructionSelector::emitADCS(Register Dst, MachineOperand &LHS,
4606 MachineOperand &RHS,
4607 MachineIRBuilder &MIRBuilder) const {
4608 assert(LHS.isReg() && RHS.isReg() && "Expected register operands?");
4609 MachineRegisterInfo *MRI = MIRBuilder.getMRI();
4610 bool Is32Bit = (MRI->getType(LHS.getReg()).getSizeInBits() == 32);
4611 static const unsigned OpcTable[2] = {AArch64::ADCSXr, AArch64::ADCSWr};
4612 return emitInstr(OpcTable[Is32Bit], {Dst}, {LHS, RHS}, MIRBuilder);
4613}
4614
4615MachineInstr *
4616AArch64InstructionSelector::emitSBCS(Register Dst, MachineOperand &LHS,
4617 MachineOperand &RHS,
4618 MachineIRBuilder &MIRBuilder) const {
4619 assert(LHS.isReg() && RHS.isReg() && "Expected register operands?");
4620 MachineRegisterInfo *MRI = MIRBuilder.getMRI();
4621 bool Is32Bit = (MRI->getType(LHS.getReg()).getSizeInBits() == 32);
4622 static const unsigned OpcTable[2] = {AArch64::SBCSXr, AArch64::SBCSWr};
4623 return emitInstr(OpcTable[Is32Bit], {Dst}, {LHS, RHS}, MIRBuilder);
4624}
4625
4626MachineInstr *
4627AArch64InstructionSelector::emitCMP(MachineOperand &LHS, MachineOperand &RHS,
4628 MachineIRBuilder &MIRBuilder) const {
4629 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4630 bool Is32Bit = MRI.getType(LHS.getReg()).getSizeInBits() == 32;
4631 auto RC = Is32Bit ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
4632 return emitSUBS(MRI.createVirtualRegister(RC), LHS, RHS, MIRBuilder);
4633}
4634
4635MachineInstr *
4636AArch64InstructionSelector::emitCMN(MachineOperand &LHS, MachineOperand &RHS,
4637 MachineIRBuilder &MIRBuilder) const {
4638 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4639 bool Is32Bit = (MRI.getType(LHS.getReg()).getSizeInBits() == 32);
4640 auto RC = Is32Bit ? &AArch64::GPR32RegClass : &AArch64::GPR64RegClass;
4641 return emitADDS(MRI.createVirtualRegister(RC), LHS, RHS, MIRBuilder);
4642}
4643
4644MachineInstr *
4645AArch64InstructionSelector::emitTST(MachineOperand &LHS, MachineOperand &RHS,
4646 MachineIRBuilder &MIRBuilder) const {
4647 assert(LHS.isReg() && RHS.isReg() && "Expected register operands?");
4648 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4649 LLT Ty = MRI.getType(LHS.getReg());
4650 unsigned RegSize = Ty.getSizeInBits();
4651 bool Is32Bit = (RegSize == 32);
4652 const unsigned OpcTable[3][2] = {{AArch64::ANDSXri, AArch64::ANDSWri},
4653 {AArch64::ANDSXrs, AArch64::ANDSWrs},
4654 {AArch64::ANDSXrr, AArch64::ANDSWrr}};
4655 // ANDS needs a logical immediate for its immediate form. Check if we can
4656 // fold one in.
4657 if (auto ValAndVReg = getIConstantVRegValWithLookThrough(RHS.getReg(), MRI)) {
4658 int64_t Imm = ValAndVReg->Value.getSExtValue();
4659
4661 auto TstMI = MIRBuilder.buildInstr(OpcTable[0][Is32Bit], {Ty}, {LHS});
4664 return &*TstMI;
4665 }
4666 }
4667
4668 if (auto Fns = selectLogicalShiftedRegister(RHS))
4669 return emitInstr(OpcTable[1][Is32Bit], {Ty}, {LHS}, MIRBuilder, Fns);
4670 return emitInstr(OpcTable[2][Is32Bit], {Ty}, {LHS, RHS}, MIRBuilder);
4671}
4672
4673MachineInstr *AArch64InstructionSelector::emitIntegerCompare(
4674 MachineOperand &LHS, MachineOperand &RHS, MachineOperand &Predicate,
4675 MachineIRBuilder &MIRBuilder) const {
4676 assert(LHS.isReg() && RHS.isReg() && "Expected LHS and RHS to be registers!");
4677 assert(Predicate.isPredicate() && "Expected predicate?");
4678 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4679 LLT CmpTy = MRI.getType(LHS.getReg());
4680 assert(!CmpTy.isVector() && "Expected scalar or pointer");
4681 unsigned Size = CmpTy.getSizeInBits();
4682 (void)Size;
4683 assert((Size == 32 || Size == 64) && "Expected a 32-bit or 64-bit LHS/RHS?");
4684 // Fold the compare into a cmn or tst if possible.
4685 if (auto FoldCmp = tryFoldIntegerCompare(LHS, RHS, Predicate, MIRBuilder))
4686 return FoldCmp;
4687 return emitCMP(LHS, RHS, MIRBuilder);
4688}
4689
4690MachineInstr *AArch64InstructionSelector::emitCSetForFCmp(
4691 Register Dst, CmpInst::Predicate Pred, MachineIRBuilder &MIRBuilder) const {
4692 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
4693#ifndef NDEBUG
4694 LLT Ty = MRI.getType(Dst);
4695 assert(!Ty.isVector() && Ty.getSizeInBits() == 32 &&
4696 "Expected a 32-bit scalar register?");
4697#endif
4698 const Register ZReg = AArch64::WZR;
4699 AArch64CC::CondCode CC1, CC2;
4700 changeFCMPPredToAArch64CC(Pred, CC1, CC2);
4701 auto InvCC1 = AArch64CC::getInvertedCondCode(CC1);
4702 if (CC2 == AArch64CC::AL)
4703 return emitCSINC(/*Dst=*/Dst, /*Src1=*/ZReg, /*Src2=*/ZReg, InvCC1,
4704 MIRBuilder);
4705 const TargetRegisterClass *RC = &AArch64::GPR32RegClass;
4706 Register Def1Reg = MRI.createVirtualRegister(RC);
4707 Register Def2Reg = MRI.createVirtualRegister(RC);
4708 auto InvCC2 = AArch64CC::getInvertedCondCode(CC2);
4709 emitCSINC(/*Dst=*/Def1Reg, /*Src1=*/ZReg, /*Src2=*/ZReg, InvCC1, MIRBuilder);
4710 emitCSINC(/*Dst=*/Def2Reg, /*Src1=*/ZReg, /*Src2=*/ZReg, InvCC2, MIRBuilder);
4711 auto OrMI = MIRBuilder.buildInstr(AArch64::ORRWrr, {Dst}, {Def1Reg, Def2Reg});
4713 return &*OrMI;
4714}
4715
4716MachineInstr *AArch64InstructionSelector::emitFPCompare(
4717 Register LHS, Register RHS, MachineIRBuilder &MIRBuilder,
4718 std::optional<CmpInst::Predicate> Pred) const {
4719 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
4720 LLT Ty = MRI.getType(LHS);
4721 if (Ty.isVector())
4722 return nullptr;
4723 unsigned OpSize = Ty.getSizeInBits();
4724 assert(OpSize == 16 || OpSize == 32 || OpSize == 64);
4725
4726 // If this is a compare against +0.0, then we don't have
4727 // to explicitly materialize a constant.
4728 bool ShouldUseImm = mi_match(RHS, MRI, m_PosZeroFP());
4729
4730 auto IsEqualityPred = [](CmpInst::Predicate P) {
4731 return P == CmpInst::FCMP_OEQ || P == CmpInst::FCMP_ONE ||
4733 };
4734 if (!ShouldUseImm && Pred && IsEqualityPred(*Pred)) {
4735 // Try commuting the operands.
4736 if (mi_match(LHS, MRI, m_PosZeroFP())) {
4737 ShouldUseImm = true;
4738 std::swap(LHS, RHS);
4739 }
4740 }
4741 unsigned CmpOpcTbl[2][3] = {
4742 {AArch64::FCMPHrr, AArch64::FCMPSrr, AArch64::FCMPDrr},
4743 {AArch64::FCMPHri, AArch64::FCMPSri, AArch64::FCMPDri}};
4744 unsigned CmpOpc =
4745 CmpOpcTbl[ShouldUseImm][OpSize == 16 ? 0 : (OpSize == 32 ? 1 : 2)];
4746
4747 // Partially build the compare. Decide if we need to add a use for the
4748 // third operand based off whether or not we're comparing against 0.0.
4749 auto CmpMI = MIRBuilder.buildInstr(CmpOpc).addUse(LHS);
4751 if (!ShouldUseImm)
4752 CmpMI.addUse(RHS);
4754 return &*CmpMI;
4755}
4756
4757MachineInstr *AArch64InstructionSelector::emitVectorConcat(
4758 std::optional<Register> Dst, Register Op1, Register Op2,
4759 MachineIRBuilder &MIRBuilder) const {
4760 // We implement a vector concat by:
4761 // 1. Use scalar_to_vector to insert the lower vector into the larger dest
4762 // 2. Insert the upper vector into the destination's upper element
4763 // TODO: some of this code is common with G_BUILD_VECTOR handling.
4764 MachineRegisterInfo &MRI = MIRBuilder.getMF().getRegInfo();
4765
4766 const LLT Op1Ty = MRI.getType(Op1);
4767 const LLT Op2Ty = MRI.getType(Op2);
4768
4769 if (Op1Ty != Op2Ty) {
4770 LLVM_DEBUG(dbgs() << "Could not do vector concat of differing vector tys");
4771 return nullptr;
4772 }
4773 assert(Op1Ty.isVector() && "Expected a vector for vector concat");
4774
4775 if (Op1Ty.getSizeInBits() >= 128) {
4776 LLVM_DEBUG(dbgs() << "Vector concat not supported for full size vectors");
4777 return nullptr;
4778 }
4779
4780 // At the moment we just support 64 bit vector concats.
4781 if (Op1Ty.getSizeInBits() != 64) {
4782 LLVM_DEBUG(dbgs() << "Vector concat supported for 64b vectors");
4783 return nullptr;
4784 }
4785
4786 const LLT ScalarTy = LLT::scalar(Op1Ty.getSizeInBits());
4787 const RegisterBank &FPRBank = *RBI.getRegBank(Op1, MRI, TRI);
4788 const TargetRegisterClass *DstRC =
4789 getRegClassForTypeOnBank(Op1Ty.multiplyElements(2), FPRBank);
4790
4791 MachineInstr *WidenedOp1 =
4792 emitScalarToVector(ScalarTy.getSizeInBits(), DstRC, Op1, MIRBuilder);
4793 MachineInstr *WidenedOp2 =
4794 emitScalarToVector(ScalarTy.getSizeInBits(), DstRC, Op2, MIRBuilder);
4795 if (!WidenedOp1 || !WidenedOp2) {
4796 LLVM_DEBUG(dbgs() << "Could not emit a vector from scalar value");
4797 return nullptr;
4798 }
4799
4800 // Now do the insert of the upper element.
4801 unsigned InsertOpc, InsSubRegIdx;
4802 std::tie(InsertOpc, InsSubRegIdx) =
4803 getInsertVecEltOpInfo(FPRBank, ScalarTy.getSizeInBits());
4804
4805 if (!Dst)
4806 Dst = MRI.createVirtualRegister(DstRC);
4807 auto InsElt =
4808 MIRBuilder
4809 .buildInstr(InsertOpc, {*Dst}, {WidenedOp1->getOperand(0).getReg()})
4810 .addImm(1) /* Lane index */
4811 .addUse(WidenedOp2->getOperand(0).getReg())
4812 .addImm(0);
4814 return &*InsElt;
4815}
4816
4817MachineInstr *
4818AArch64InstructionSelector::emitCSINC(Register Dst, Register Src1,
4819 Register Src2, AArch64CC::CondCode Pred,
4820 MachineIRBuilder &MIRBuilder) const {
4821 auto &MRI = *MIRBuilder.getMRI();
4822 const RegClassOrRegBank &RegClassOrBank = MRI.getRegClassOrRegBank(Dst);
4823 // If we used a register class, then this won't necessarily have an LLT.
4824 // Compute the size based off whether or not we have a class or bank.
4825 unsigned Size;
4826 if (const auto *RC = dyn_cast<const TargetRegisterClass *>(RegClassOrBank))
4827 Size = TRI.getRegSizeInBits(*RC);
4828 else
4829 Size = MRI.getType(Dst).getSizeInBits();
4830 // Some opcodes use s1.
4831 assert(Size <= 64 && "Expected 64 bits or less only!");
4832 static const unsigned OpcTable[2] = {AArch64::CSINCWr, AArch64::CSINCXr};
4833 unsigned Opc = OpcTable[Size == 64];
4834 auto CSINC = MIRBuilder.buildInstr(Opc, {Dst}, {Src1, Src2}).addImm(Pred);
4836 return &*CSINC;
4837}
4838
4839MachineInstr *AArch64InstructionSelector::emitCarryIn(MachineInstr &I,
4840 Register CarryReg) {
4841 MachineRegisterInfo *MRI = MIB.getMRI();
4842 unsigned Opcode = I.getOpcode();
4843
4844 // If the instruction is a SUB, we need to negate the carry,
4845 // because borrowing is indicated by carry-flag == 0.
4846 bool NeedsNegatedCarry =
4847 (Opcode == TargetOpcode::G_USUBE || Opcode == TargetOpcode::G_SSUBE);
4848
4849 // If the previous instruction will already produce the correct carry, do not
4850 // emit a carry generating instruction. E.g. for G_UADDE/G_USUBE sequences
4851 // generated during legalization of wide add/sub. This optimization depends on
4852 // these sequences not being interrupted by other instructions.
4853 // We have to select the previous instruction before the carry-using
4854 // instruction is deleted by the calling function, otherwise the previous
4855 // instruction might become dead and would get deleted.
4856 MachineInstr *SrcMI = MRI->getVRegDef(CarryReg);
4857 if (SrcMI == I.getPrevNode()) {
4858 if (auto *CarrySrcMI = dyn_cast<GAddSubCarryOut>(SrcMI)) {
4859 bool ProducesNegatedCarry = CarrySrcMI->isSub();
4860 if (NeedsNegatedCarry == ProducesNegatedCarry &&
4861 CarrySrcMI->isUnsigned() &&
4862 CarrySrcMI->getCarryOutReg() == CarryReg &&
4863 selectAndRestoreState(*SrcMI))
4864 return nullptr;
4865 }
4866 }
4867
4868 Register DeadReg = MRI->createVirtualRegister(&AArch64::GPR32RegClass);
4869
4870 if (NeedsNegatedCarry) {
4871 // (0 - Carry) sets !C in NZCV when Carry == 1
4872 Register ZReg = AArch64::WZR;
4873 return emitInstr(AArch64::SUBSWrr, {DeadReg}, {ZReg, CarryReg}, MIB);
4874 }
4875
4876 // (Carry - 1) sets !C in NZCV when Carry == 0
4877 auto Fns = select12BitValueWithLeftShift(1);
4878 return emitInstr(AArch64::SUBSWri, {DeadReg}, {CarryReg}, MIB, Fns);
4879}
4880
4881bool AArch64InstructionSelector::selectOverflowOp(MachineInstr &I,
4882 MachineRegisterInfo &MRI) {
4883 auto &CarryMI = cast<GAddSubCarryOut>(I);
4884
4885 if (auto *CarryInMI = dyn_cast<GAddSubCarryInOut>(&I)) {
4886 // Set NZCV carry according to carry-in VReg
4887 emitCarryIn(I, CarryInMI->getCarryInReg());
4888 }
4889
4890 // Emit the operation and get the correct condition code.
4891 auto OpAndCC = emitOverflowOp(I.getOpcode(), CarryMI.getDstReg(),
4892 CarryMI.getLHS(), CarryMI.getRHS(), MIB);
4893
4894 Register CarryOutReg = CarryMI.getCarryOutReg();
4895
4896 // Don't convert carry-out to VReg if it is never used
4897 if (MRI.use_nodbg_empty(CarryOutReg)) {
4898 OpAndCC.first->addRegisterDead(AArch64::NZCV, &TRI);
4899 } else {
4900 // Now, put the overflow result in the register given by the first operand
4901 // to the overflow op. CSINC increments the result when the predicate is
4902 // false, so to get the increment when it's true, we need to use the
4903 // inverse. In this case, we want to increment when carry is set.
4904 Register ZReg = AArch64::WZR;
4905 emitCSINC(/*Dst=*/CarryOutReg, /*Src1=*/ZReg, /*Src2=*/ZReg,
4906 getInvertedCondCode(OpAndCC.second), MIB);
4907 }
4908
4909 I.eraseFromParent();
4910 return true;
4911}
4912
4913std::pair<MachineInstr *, AArch64CC::CondCode>
4914AArch64InstructionSelector::emitOverflowOp(unsigned Opcode, Register Dst,
4915 MachineOperand &LHS,
4916 MachineOperand &RHS,
4917 MachineIRBuilder &MIRBuilder) const {
4918 switch (Opcode) {
4919 default:
4920 llvm_unreachable("Unexpected opcode!");
4921 case TargetOpcode::G_SADDO:
4922 return std::make_pair(emitADDS(Dst, LHS, RHS, MIRBuilder), AArch64CC::VS);
4923 case TargetOpcode::G_UADDO:
4924 return std::make_pair(emitADDS(Dst, LHS, RHS, MIRBuilder), AArch64CC::HS);
4925 case TargetOpcode::G_SSUBO:
4926 return std::make_pair(emitSUBS(Dst, LHS, RHS, MIRBuilder), AArch64CC::VS);
4927 case TargetOpcode::G_USUBO:
4928 return std::make_pair(emitSUBS(Dst, LHS, RHS, MIRBuilder), AArch64CC::LO);
4929 case TargetOpcode::G_SADDE:
4930 return std::make_pair(emitADCS(Dst, LHS, RHS, MIRBuilder), AArch64CC::VS);
4931 case TargetOpcode::G_UADDE:
4932 return std::make_pair(emitADCS(Dst, LHS, RHS, MIRBuilder), AArch64CC::HS);
4933 case TargetOpcode::G_SSUBE:
4934 return std::make_pair(emitSBCS(Dst, LHS, RHS, MIRBuilder), AArch64CC::VS);
4935 case TargetOpcode::G_USUBE:
4936 return std::make_pair(emitSBCS(Dst, LHS, RHS, MIRBuilder), AArch64CC::LO);
4937 }
4938}
4939
4940/// Returns true if @p Val is a tree of AND/OR/CMP operations that can be
4941/// expressed as a conjunction.
4942/// \param CanNegate Set to true if we can negate the whole sub-tree just by
4943/// changing the conditions on the CMP tests.
4944/// (this means we can call emitConjunctionRec() with
4945/// Negate==true on this sub-tree)
4946/// \param MustBeFirst Set to true if this subtree needs to be negated and we
4947/// cannot do the negation naturally. We are required to
4948/// emit the subtree first in this case.
4949/// \param WillNegate Is true if are called when the result of this
4950/// subexpression must be negated. This happens when the
4951/// outer expression is an OR. We can use this fact to know
4952/// that we have a double negation (or (or ...) ...) that
4953/// can be implemented for free.
4954static bool canEmitConjunction(Register Val, bool &CanNegate, bool &MustBeFirst,
4955 bool WillNegate, MachineRegisterInfo &MRI,
4956 unsigned Depth = 0) {
4957 if (!MRI.hasOneNonDBGUse(Val))
4958 return false;
4959 MachineInstr *ValDef = MRI.getVRegDef(Val);
4960 unsigned Opcode = ValDef->getOpcode();
4961 if (isa<GAnyCmp>(ValDef)) {
4962 CanNegate = true;
4963 MustBeFirst = false;
4964 return true;
4965 }
4966 // Protect against exponential runtime and stack overflow.
4967 if (Depth > 6)
4968 return false;
4969 if (Opcode == TargetOpcode::G_AND || Opcode == TargetOpcode::G_OR) {
4970 bool IsOR = Opcode == TargetOpcode::G_OR;
4971 Register O0 = ValDef->getOperand(1).getReg();
4972 Register O1 = ValDef->getOperand(2).getReg();
4973 bool CanNegateL;
4974 bool MustBeFirstL;
4975 if (!canEmitConjunction(O0, CanNegateL, MustBeFirstL, IsOR, MRI, Depth + 1))
4976 return false;
4977 bool CanNegateR;
4978 bool MustBeFirstR;
4979 if (!canEmitConjunction(O1, CanNegateR, MustBeFirstR, IsOR, MRI, Depth + 1))
4980 return false;
4981
4982 if (MustBeFirstL && MustBeFirstR)
4983 return false;
4984
4985 if (IsOR) {
4986 // For an OR expression we need to be able to naturally negate at least
4987 // one side or we cannot do the transformation at all.
4988 if (!CanNegateL && !CanNegateR)
4989 return false;
4990 // If we the result of the OR will be negated and we can naturally negate
4991 // the leaves, then this sub-tree as a whole negates naturally.
4992 CanNegate = WillNegate && CanNegateL && CanNegateR;
4993 // If we cannot naturally negate the whole sub-tree, then this must be
4994 // emitted first.
4995 MustBeFirst = !CanNegate;
4996 } else {
4997 assert(Opcode == TargetOpcode::G_AND && "Must be G_AND");
4998 // We cannot naturally negate an AND operation.
4999 CanNegate = false;
5000 MustBeFirst = MustBeFirstL || MustBeFirstR;
5001 }
5002 return true;
5003 }
5004 return false;
5005}
5006
5007MachineInstr *AArch64InstructionSelector::emitConditionalComparison(
5010 MachineIRBuilder &MIB) const {
5011 auto &MRI = *MIB.getMRI();
5012 LLT OpTy = MRI.getType(LHS);
5013 unsigned CCmpOpc;
5014 std::optional<ValueAndVReg> C;
5015 if (CmpInst::isIntPredicate(CC)) {
5016 assert(OpTy.getSizeInBits() == 32 || OpTy.getSizeInBits() == 64);
5018 if (!C || C->Value.sgt(31) || C->Value.slt(-31))
5019 CCmpOpc = OpTy.getSizeInBits() == 32 ? AArch64::CCMPWr : AArch64::CCMPXr;
5020 else if (C->Value.ule(31))
5021 CCmpOpc = OpTy.getSizeInBits() == 32 ? AArch64::CCMPWi : AArch64::CCMPXi;
5022 else
5023 CCmpOpc = OpTy.getSizeInBits() == 32 ? AArch64::CCMNWi : AArch64::CCMNXi;
5024 } else {
5025 assert(OpTy.getSizeInBits() == 16 || OpTy.getSizeInBits() == 32 ||
5026 OpTy.getSizeInBits() == 64);
5027 switch (OpTy.getSizeInBits()) {
5028 case 16:
5029 assert(STI.hasFullFP16() && "Expected Full FP16 for fp16 comparisons");
5030 CCmpOpc = AArch64::FCCMPHrr;
5031 break;
5032 case 32:
5033 CCmpOpc = AArch64::FCCMPSrr;
5034 break;
5035 case 64:
5036 CCmpOpc = AArch64::FCCMPDrr;
5037 break;
5038 default:
5039 return nullptr;
5040 }
5041 }
5043 unsigned NZCV = AArch64CC::getNZCVToSatisfyCondCode(InvOutCC);
5044 auto CCmp =
5045 MIB.buildInstr(CCmpOpc, {}, {LHS});
5046 if (CCmpOpc == AArch64::CCMPWi || CCmpOpc == AArch64::CCMPXi)
5047 CCmp.addImm(C->Value.getZExtValue());
5048 else if (CCmpOpc == AArch64::CCMNWi || CCmpOpc == AArch64::CCMNXi)
5049 CCmp.addImm(C->Value.abs().getZExtValue());
5050 else
5051 CCmp.addReg(RHS);
5052 CCmp.addImm(NZCV).addImm(Predicate);
5054 return &*CCmp;
5055}
5056
5057MachineInstr *AArch64InstructionSelector::emitConjunctionRec(
5058 Register Val, AArch64CC::CondCode &OutCC, bool Negate, Register CCOp,
5059 AArch64CC::CondCode Predicate, MachineIRBuilder &MIB) const {
5060 // We're at a tree leaf, produce a conditional comparison operation.
5061 auto &MRI = *MIB.getMRI();
5062 MachineInstr *ValDef = MRI.getVRegDef(Val);
5063 unsigned Opcode = ValDef->getOpcode();
5064 if (auto *Cmp = dyn_cast<GAnyCmp>(ValDef)) {
5065 Register LHS = Cmp->getLHSReg();
5066 Register RHS = Cmp->getRHSReg();
5067 CmpInst::Predicate CC = Cmp->getCond();
5068 if (Negate)
5070 if (isa<GICmp>(Cmp)) {
5071 OutCC = changeICMPPredToAArch64CC(CC, RHS, MIB.getMRI());
5072 } else {
5073 // Handle special FP cases.
5074 AArch64CC::CondCode ExtraCC;
5075 changeFPCCToANDAArch64CC(CC, OutCC, ExtraCC);
5076 // Some floating point conditions can't be tested with a single condition
5077 // code. Construct an additional comparison in this case.
5078 if (ExtraCC != AArch64CC::AL) {
5079 MachineInstr *ExtraCmp;
5080 if (!CCOp)
5081 ExtraCmp = emitFPCompare(LHS, RHS, MIB, CC);
5082 else
5083 ExtraCmp =
5084 emitConditionalComparison(LHS, RHS, CC, Predicate, ExtraCC, MIB);
5085 CCOp = ExtraCmp->getOperand(0).getReg();
5086 Predicate = ExtraCC;
5087 }
5088 }
5089
5090 // Produce a normal comparison if we are first in the chain
5091 if (!CCOp) {
5092 if (isa<GICmp>(Cmp))
5093 return emitCMP(Cmp->getOperand(2), Cmp->getOperand(3), MIB);
5094 return emitFPCompare(Cmp->getOperand(2).getReg(),
5095 Cmp->getOperand(3).getReg(), MIB);
5096 }
5097 // Otherwise produce a ccmp.
5098 return emitConditionalComparison(LHS, RHS, CC, Predicate, OutCC, MIB);
5099 }
5100 assert(MRI.hasOneNonDBGUse(Val) && "Valid conjunction/disjunction tree");
5101
5102 bool IsOR = Opcode == TargetOpcode::G_OR;
5103
5104 Register LHS = ValDef->getOperand(1).getReg();
5105 bool CanNegateL;
5106 bool MustBeFirstL;
5107 bool ValidL = canEmitConjunction(LHS, CanNegateL, MustBeFirstL, IsOR, MRI);
5108 assert(ValidL && "Valid conjunction/disjunction tree");
5109 (void)ValidL;
5110
5111 Register RHS = ValDef->getOperand(2).getReg();
5112 bool CanNegateR;
5113 bool MustBeFirstR;
5114 bool ValidR = canEmitConjunction(RHS, CanNegateR, MustBeFirstR, IsOR, MRI);
5115 assert(ValidR && "Valid conjunction/disjunction tree");
5116 (void)ValidR;
5117
5118 // Swap sub-tree that must come first to the right side.
5119 if (MustBeFirstL) {
5120 assert(!MustBeFirstR && "Valid conjunction/disjunction tree");
5121 std::swap(LHS, RHS);
5122 std::swap(CanNegateL, CanNegateR);
5123 std::swap(MustBeFirstL, MustBeFirstR);
5124 }
5125
5126 bool NegateR;
5127 bool NegateAfterR;
5128 bool NegateL;
5129 bool NegateAfterAll;
5130 if (Opcode == TargetOpcode::G_OR) {
5131 // Swap the sub-tree that we can negate naturally to the left.
5132 if (!CanNegateL) {
5133 assert(CanNegateR && "at least one side must be negatable");
5134 assert(!MustBeFirstR && "invalid conjunction/disjunction tree");
5135 assert(!Negate);
5136 std::swap(LHS, RHS);
5137 NegateR = false;
5138 NegateAfterR = true;
5139 } else {
5140 // Negate the left sub-tree if possible, otherwise negate the result.
5141 NegateR = CanNegateR;
5142 NegateAfterR = !CanNegateR;
5143 }
5144 NegateL = true;
5145 NegateAfterAll = !Negate;
5146 } else {
5147 assert(Opcode == TargetOpcode::G_AND &&
5148 "Valid conjunction/disjunction tree");
5149 assert(!Negate && "Valid conjunction/disjunction tree");
5150
5151 NegateL = false;
5152 NegateR = false;
5153 NegateAfterR = false;
5154 NegateAfterAll = false;
5155 }
5156
5157 // Emit sub-trees.
5158 AArch64CC::CondCode RHSCC;
5159 MachineInstr *CmpR =
5160 emitConjunctionRec(RHS, RHSCC, NegateR, CCOp, Predicate, MIB);
5161 if (NegateAfterR)
5162 RHSCC = AArch64CC::getInvertedCondCode(RHSCC);
5163 MachineInstr *CmpL = emitConjunctionRec(
5164 LHS, OutCC, NegateL, CmpR->getOperand(0).getReg(), RHSCC, MIB);
5165 if (NegateAfterAll)
5166 OutCC = AArch64CC::getInvertedCondCode(OutCC);
5167 return CmpL;
5168}
5169
5170MachineInstr *AArch64InstructionSelector::emitConjunction(
5171 Register Val, AArch64CC::CondCode &OutCC, MachineIRBuilder &MIB) const {
5172 bool DummyCanNegate;
5173 bool DummyMustBeFirst;
5174 if (!canEmitConjunction(Val, DummyCanNegate, DummyMustBeFirst, false,
5175 *MIB.getMRI()))
5176 return nullptr;
5177 return emitConjunctionRec(Val, OutCC, false, Register(), AArch64CC::AL, MIB);
5178}
5179
5180bool AArch64InstructionSelector::tryOptSelectConjunction(GSelect &SelI,
5181 MachineInstr &CondMI) {
5182 AArch64CC::CondCode AArch64CC;
5183 MachineInstr *ConjMI = emitConjunction(SelI.getCondReg(), AArch64CC, MIB);
5184 if (!ConjMI)
5185 return false;
5186
5187 emitSelect(SelI.getReg(0), SelI.getTrueReg(), SelI.getFalseReg(), AArch64CC, MIB);
5188 SelI.eraseFromParent();
5189 return true;
5190}
5191
5192bool AArch64InstructionSelector::tryOptSelect(GSelect &I) {
5193 MachineRegisterInfo &MRI = *MIB.getMRI();
5194 // We want to recognize this pattern:
5195 //
5196 // $z = G_FCMP pred, $x, $y
5197 // ...
5198 // $w = G_SELECT $z, $a, $b
5199 //
5200 // Where the value of $z is *only* ever used by the G_SELECT (possibly with
5201 // some copies/truncs in between.)
5202 //
5203 // If we see this, then we can emit something like this:
5204 //
5205 // fcmp $x, $y
5206 // fcsel $w, $a, $b, pred
5207 //
5208 // Rather than emitting both of the rather long sequences in the standard
5209 // G_FCMP/G_SELECT select methods.
5210
5211 // First, check if the condition is defined by a compare.
5212 MachineInstr *CondDef = MRI.getVRegDef(I.getOperand(1).getReg());
5213
5214 // We can only fold if all of the defs have one use.
5215 Register CondDefReg = CondDef->getOperand(0).getReg();
5216 if (!MRI.hasOneNonDBGUse(CondDefReg)) {
5217 // Unless it's another select.
5218 for (const MachineInstr &UI : MRI.use_nodbg_instructions(CondDefReg)) {
5219 if (CondDef == &UI)
5220 continue;
5221 if (UI.getOpcode() != TargetOpcode::G_SELECT)
5222 return false;
5223 }
5224 }
5225
5226 // Is the condition defined by a compare?
5227 unsigned CondOpc = CondDef->getOpcode();
5228 if (CondOpc != TargetOpcode::G_ICMP && CondOpc != TargetOpcode::G_FCMP) {
5229 if (tryOptSelectConjunction(I, *CondDef))
5230 return true;
5231 return false;
5232 }
5233
5235 if (CondOpc == TargetOpcode::G_ICMP) {
5236 auto &PredOp = CondDef->getOperand(1);
5237 emitIntegerCompare(CondDef->getOperand(2), CondDef->getOperand(3), PredOp,
5238 MIB);
5239 auto Pred = static_cast<CmpInst::Predicate>(PredOp.getPredicate());
5240 CondCode =
5241 changeICMPPredToAArch64CC(Pred, CondDef->getOperand(3).getReg(), &MRI);
5242 } else {
5243 // Get the condition code for the select.
5244 auto Pred =
5245 static_cast<CmpInst::Predicate>(CondDef->getOperand(1).getPredicate());
5246 AArch64CC::CondCode CondCode2;
5247 changeFCMPPredToAArch64CC(Pred, CondCode, CondCode2);
5248
5249 // changeFCMPPredToAArch64CC sets CondCode2 to AL when we require two
5250 // instructions to emit the comparison.
5251 // TODO: Handle FCMP_UEQ and FCMP_ONE. After that, this check will be
5252 // unnecessary.
5253 if (CondCode2 != AArch64CC::AL)
5254 return false;
5255
5256 if (!emitFPCompare(CondDef->getOperand(2).getReg(),
5257 CondDef->getOperand(3).getReg(), MIB)) {
5258 LLVM_DEBUG(dbgs() << "Couldn't emit compare for select!\n");
5259 return false;
5260 }
5261 }
5262
5263 // Emit the select.
5264 emitSelect(I.getOperand(0).getReg(), I.getOperand(2).getReg(),
5265 I.getOperand(3).getReg(), CondCode, MIB);
5266 I.eraseFromParent();
5267 return true;
5268}
5269
5270MachineInstr *AArch64InstructionSelector::tryFoldIntegerCompare(
5271 MachineOperand &LHS, MachineOperand &RHS, MachineOperand &Predicate,
5272 MachineIRBuilder &MIRBuilder) const {
5273 assert(LHS.isReg() && RHS.isReg() && Predicate.isPredicate() &&
5274 "Unexpected MachineOperand");
5275 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
5276 // We want to find this sort of thing:
5277 // x = G_SUB 0, y
5278 // G_ICMP z, x
5279 //
5280 // In this case, we can fold the G_SUB into the G_ICMP using a CMN instead.
5281 // e.g:
5282 //
5283 // cmn z, y
5284
5285 // Check if the RHS or LHS of the G_ICMP is defined by a SUB
5286 MachineInstr *LHSDef = getDefIgnoringCopies(LHS.getReg(), MRI);
5287 MachineInstr *RHSDef = getDefIgnoringCopies(RHS.getReg(), MRI);
5288 auto P = static_cast<CmpInst::Predicate>(Predicate.getPredicate());
5289
5290 // Given this:
5291 //
5292 // x = G_SUB 0, y
5293 // G_ICMP z, x
5294 //
5295 // Produce this:
5296 //
5297 // cmn z, y
5298 if (isCMN(RHSDef, P, MRI))
5299 return emitCMN(LHS, RHSDef->getOperand(2), MIRBuilder);
5300
5301 // Same idea here, but with the LHS of the compare instead:
5302 //
5303 // Given this:
5304 //
5305 // x = G_SUB 0, y
5306 // G_ICMP x, z
5307 //
5308 // Produce this:
5309 //
5310 // cmn y, z
5311 //
5312 // But be careful! We need to swap the predicate!
5313 if (isCMN(LHSDef, P, MRI)) {
5314 if (!CmpInst::isEquality(P)) {
5317 }
5318 return emitCMN(LHSDef->getOperand(2), RHS, MIRBuilder);
5319 }
5320
5321 // Given this:
5322 //
5323 // z = G_AND x, y
5324 // G_ICMP z, 0
5325 //
5326 // Produce this if the compare is signed:
5327 //
5328 // tst x, y
5329 if (!CmpInst::isUnsigned(P) && LHSDef &&
5330 LHSDef->getOpcode() == TargetOpcode::G_AND) {
5331 // Make sure that the RHS is 0.
5332 auto ValAndVReg = getIConstantVRegValWithLookThrough(RHS.getReg(), MRI);
5333 if (!ValAndVReg || ValAndVReg->Value != 0)
5334 return nullptr;
5335
5336 return emitTST(LHSDef->getOperand(1),
5337 LHSDef->getOperand(2), MIRBuilder);
5338 }
5339
5340 return nullptr;
5341}
5342
5343bool AArch64InstructionSelector::selectShuffleVector(
5344 MachineInstr &I, MachineRegisterInfo &MRI) {
5345 const LLT DstTy = MRI.getType(I.getOperand(0).getReg());
5346 Register Src1Reg = I.getOperand(1).getReg();
5347 Register Src2Reg = I.getOperand(2).getReg();
5348 ArrayRef<int> Mask = I.getOperand(3).getShuffleMask();
5349 assert(DstTy == MRI.getType(Src1Reg) &&
5350 "Expected equal shuffle types during selection");
5351
5352 MachineBasicBlock &MBB = *I.getParent();
5353 MachineFunction &MF = *MBB.getParent();
5354 LLVMContext &Ctx = MF.getFunction().getContext();
5355
5356 unsigned BytesPerElt = DstTy.getElementType().getSizeInBits() / 8;
5357 int NumElts = DstTy.getNumElements();
5358
5359 SmallVector<int> NewMask;
5360 bool FirstUsed = false;
5361 bool SecondUsed = false;
5362 for (int M : Mask) {
5363 // Map any undef or zero lanes to 255.
5364 if (M < 0 || VT->getKnownBits(M < NumElts ? Src1Reg : Src2Reg,
5365 APInt::getOneBitSet(NumElts, M % NumElts))
5366 .isZero()) {
5367 for (unsigned Byte = 0; Byte < BytesPerElt; ++Byte)
5368 NewMask.push_back(255);
5369 continue;
5370 }
5371
5372 FirstUsed |= M < NumElts;
5373 SecondUsed |= M >= NumElts;
5374 for (unsigned Byte = 0; Byte < BytesPerElt; ++Byte) {
5375 unsigned Offset = Byte + M * BytesPerElt;
5376 NewMask.push_back(Offset);
5377 }
5378 }
5379
5380 // If the first is unused or all zeros, use the second src in a tbl1.
5381 if (!FirstUsed) {
5382 int ByteLanes = DstTy.getSizeInBits() == 128 ? 16 : 8;
5383 for (int &M : NewMask) {
5384 if (M != 255) {
5385 assert(M >= ByteLanes && M < 2 * ByteLanes);
5386 M -= ByteLanes;
5387 }
5388 }
5389 std::swap(Src1Reg, Src2Reg);
5390 std::swap(FirstUsed, SecondUsed);
5391 }
5392
5393 // Use a constant pool to load the index vector for TBL.
5395 transform(NewMask, std::back_inserter(CstIdxs), [&Ctx](int M) {
5396 return ConstantInt::get(Type::getInt8Ty(Ctx), M);
5397 });
5398 Constant *CPVal = ConstantVector::get(CstIdxs);
5399 MachineInstr *IndexLoad = emitLoadFromConstantPool(CPVal, MIB);
5400 if (!IndexLoad) {
5401 LLVM_DEBUG(dbgs() << "Could not load from a constant pool");
5402 return false;
5403 }
5404
5405 if (DstTy.getSizeInBits() != 128) {
5406 assert(DstTy.getSizeInBits() == 64 && "Unexpected shuffle result ty");
5407 // This case can be done with TBL1.
5408 MachineInstr *Concat =
5409 emitVectorConcat(std::nullopt, Src1Reg, Src2Reg, MIB);
5410 if (!Concat) {
5411 LLVM_DEBUG(dbgs() << "Could not do vector concat for tbl1");
5412 return false;
5413 }
5414
5415 // The constant pool load will be 64 bits, so need to convert to FPR128 reg.
5416 IndexLoad = emitScalarToVector(64, &AArch64::FPR128RegClass,
5417 IndexLoad->getOperand(0).getReg(), MIB);
5418
5419 auto TBL1 = MIB.buildInstr(
5420 AArch64::TBLv16i8One, {&AArch64::FPR128RegClass},
5421 {Concat->getOperand(0).getReg(), IndexLoad->getOperand(0).getReg()});
5423
5424 auto Copy =
5425 MIB.buildInstr(TargetOpcode::COPY, {I.getOperand(0).getReg()}, {})
5426 .addReg(TBL1.getReg(0), {}, AArch64::dsub);
5427 RBI.constrainGenericRegister(Copy.getReg(0), AArch64::FPR64RegClass, MRI);
5428 I.eraseFromParent();
5429 return true;
5430 }
5431
5432 if (!SecondUsed) {
5433 auto TBL1 = MIB.buildInstr(AArch64::TBLv16i8One, {I.getOperand(0)},
5434 {Src1Reg, IndexLoad->getOperand(0)});
5436 I.eraseFromParent();
5437 return true;
5438 }
5439
5440 // For TBL2 we need to emit a REG_SEQUENCE to tie together two consecutive
5441 // Q registers for regalloc.
5442 SmallVector<Register, 2> Regs = {Src1Reg, Src2Reg};
5443 auto RegSeq = createQTuple(Regs, MIB);
5444 auto TBL2 = MIB.buildInstr(AArch64::TBLv16i8Two, {I.getOperand(0)},
5445 {RegSeq, IndexLoad->getOperand(0)});
5447 I.eraseFromParent();
5448 return true;
5449}
5450
5451MachineInstr *AArch64InstructionSelector::emitLaneInsert(
5452 std::optional<Register> DstReg, Register SrcReg, Register EltReg,
5453 unsigned LaneIdx, const RegisterBank &RB,
5454 MachineIRBuilder &MIRBuilder) const {
5455 MachineInstr *InsElt = nullptr;
5456 const TargetRegisterClass *DstRC = &AArch64::FPR128RegClass;
5457 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
5458
5459 // Create a register to define with the insert if one wasn't passed in.
5460 if (!DstReg)
5461 DstReg = MRI.createVirtualRegister(DstRC);
5462
5463 unsigned EltSize = MRI.getType(EltReg).getSizeInBits();
5464 unsigned Opc = getInsertVecEltOpInfo(RB, EltSize).first;
5465
5466 if (RB.getID() == AArch64::FPRRegBankID) {
5467 auto InsSub = emitScalarToVector(EltSize, DstRC, EltReg, MIRBuilder);
5468 InsElt = MIRBuilder.buildInstr(Opc, {*DstReg}, {SrcReg})
5469 .addImm(LaneIdx)
5470 .addUse(InsSub->getOperand(0).getReg())
5471 .addImm(0);
5472 } else {
5473 InsElt = MIRBuilder.buildInstr(Opc, {*DstReg}, {SrcReg})
5474 .addImm(LaneIdx)
5475 .addUse(EltReg);
5476 }
5477
5479 return InsElt;
5480}
5481
5482bool AArch64InstructionSelector::selectUSMovFromExtend(
5483 MachineInstr &MI, MachineRegisterInfo &MRI) {
5484 if (MI.getOpcode() != TargetOpcode::G_SEXT &&
5485 MI.getOpcode() != TargetOpcode::G_ZEXT &&
5486 MI.getOpcode() != TargetOpcode::G_ANYEXT)
5487 return false;
5488 bool IsSigned = MI.getOpcode() == TargetOpcode::G_SEXT;
5489 const Register DefReg = MI.getOperand(0).getReg();
5490 const LLT DstTy = MRI.getType(DefReg);
5491 unsigned DstSize = DstTy.getSizeInBits();
5492
5493 if (DstSize != 32 && DstSize != 64)
5494 return false;
5495
5496 MachineInstr *Extract = getOpcodeDef(TargetOpcode::G_EXTRACT_VECTOR_ELT,
5497 MI.getOperand(1).getReg(), MRI);
5498 int64_t Lane;
5499 if (!Extract || !mi_match(Extract->getOperand(2).getReg(), MRI, m_ICst(Lane)))
5500 return false;
5501 Register Src0 = Extract->getOperand(1).getReg();
5502
5503 const LLT VecTy = MRI.getType(Src0);
5504 if (VecTy.isScalableVector())
5505 return false;
5506
5507 if (VecTy.getSizeInBits() != 128) {
5508 const MachineInstr *ScalarToVector = emitScalarToVector(
5509 VecTy.getSizeInBits(), &AArch64::FPR128RegClass, Src0, MIB);
5510 assert(ScalarToVector && "Didn't expect emitScalarToVector to fail!");
5511 Src0 = ScalarToVector->getOperand(0).getReg();
5512 }
5513
5514 unsigned Opcode;
5515 if (DstSize == 64 && VecTy.getScalarSizeInBits() == 32)
5516 Opcode = IsSigned ? AArch64::SMOVvi32to64 : AArch64::UMOVvi32;
5517 else if (DstSize == 64 && VecTy.getScalarSizeInBits() == 16)
5518 Opcode = IsSigned ? AArch64::SMOVvi16to64 : AArch64::UMOVvi16;
5519 else if (DstSize == 64 && VecTy.getScalarSizeInBits() == 8)
5520 Opcode = IsSigned ? AArch64::SMOVvi8to64 : AArch64::UMOVvi8;
5521 else if (DstSize == 32 && VecTy.getScalarSizeInBits() == 16)
5522 Opcode = IsSigned ? AArch64::SMOVvi16to32 : AArch64::UMOVvi16;
5523 else if (DstSize == 32 && VecTy.getScalarSizeInBits() == 8)
5524 Opcode = IsSigned ? AArch64::SMOVvi8to32 : AArch64::UMOVvi8;
5525 else
5526 llvm_unreachable("Unexpected type combo for S/UMov!");
5527
5528 // We may need to generate one of these, depending on the type and sign of the
5529 // input:
5530 // DstReg = SMOV Src0, Lane;
5531 // NewReg = UMOV Src0, Lane; DstReg = SUBREG_TO_REG NewReg, sub_32;
5532 MachineInstr *ExtI = nullptr;
5533 if (DstSize == 64 && !IsSigned) {
5534 Register NewReg = MRI.createVirtualRegister(&AArch64::GPR32RegClass);
5535 MIB.buildInstr(Opcode, {NewReg}, {Src0}).addImm(Lane);
5536 ExtI = MIB.buildInstr(AArch64::SUBREG_TO_REG, {DefReg}, {})
5537 .addUse(NewReg)
5538 .addImm(AArch64::sub_32);
5539 RBI.constrainGenericRegister(DefReg, AArch64::GPR64RegClass, MRI);
5540 } else
5541 ExtI = MIB.buildInstr(Opcode, {DefReg}, {Src0}).addImm(Lane);
5542
5544 MI.eraseFromParent();
5545 return true;
5546}
5547
5548MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm8(
5549 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5550 unsigned int Op;
5551 if (DstSize == 128) {
5552 if (Bits.getHiBits(64) != Bits.getLoBits(64))
5553 return nullptr;
5554 Op = AArch64::MOVIv16b_ns;
5555 } else {
5556 Op = AArch64::MOVIv8b_ns;
5557 }
5558
5559 uint64_t Val = Bits.zextOrTrunc(64).getZExtValue();
5560
5563 auto Mov = Builder.buildInstr(Op, {Dst}, {}).addImm(Val);
5565 return &*Mov;
5566 }
5567 return nullptr;
5568}
5569
5570MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm16(
5571 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5572 bool Inv) {
5573
5574 unsigned int Op;
5575 if (DstSize == 128) {
5576 if (Bits.getHiBits(64) != Bits.getLoBits(64))
5577 return nullptr;
5578 Op = Inv ? AArch64::MVNIv8i16 : AArch64::MOVIv8i16;
5579 } else {
5580 Op = Inv ? AArch64::MVNIv4i16 : AArch64::MOVIv4i16;
5581 }
5582
5583 uint64_t Val = Bits.zextOrTrunc(64).getZExtValue();
5584 uint64_t Shift;
5585
5588 Shift = 0;
5589 } else if (AArch64_AM::isAdvSIMDModImmType6(Val)) {
5591 Shift = 8;
5592 } else
5593 return nullptr;
5594
5595 auto Mov = Builder.buildInstr(Op, {Dst}, {}).addImm(Val).addImm(Shift);
5597 return &*Mov;
5598}
5599
5600MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm32(
5601 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5602 bool Inv) {
5603
5604 unsigned int Op;
5605 if (DstSize == 128) {
5606 if (Bits.getHiBits(64) != Bits.getLoBits(64))
5607 return nullptr;
5608 Op = Inv ? AArch64::MVNIv4i32 : AArch64::MOVIv4i32;
5609 } else {
5610 Op = Inv ? AArch64::MVNIv2i32 : AArch64::MOVIv2i32;
5611 }
5612
5613 uint64_t Val = Bits.zextOrTrunc(64).getZExtValue();
5614 uint64_t Shift;
5615
5618 Shift = 0;
5619 } else if ((AArch64_AM::isAdvSIMDModImmType2(Val))) {
5621 Shift = 8;
5622 } else if ((AArch64_AM::isAdvSIMDModImmType3(Val))) {
5624 Shift = 16;
5625 } else if ((AArch64_AM::isAdvSIMDModImmType4(Val))) {
5627 Shift = 24;
5628 } else
5629 return nullptr;
5630
5631 auto Mov = Builder.buildInstr(Op, {Dst}, {}).addImm(Val).addImm(Shift);
5633 return &*Mov;
5634}
5635
5636MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm64(
5637 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5638
5639 unsigned int Op;
5640 if (DstSize == 128) {
5641 if (Bits.getHiBits(64) != Bits.getLoBits(64))
5642 return nullptr;
5643 Op = AArch64::MOVIv2d_ns;
5644 } else {
5645 Op = AArch64::MOVID;
5646 }
5647
5648 uint64_t Val = Bits.zextOrTrunc(64).getZExtValue();
5651 auto Mov = Builder.buildInstr(Op, {Dst}, {}).addImm(Val);
5653 return &*Mov;
5654 }
5655 return nullptr;
5656}
5657
5658MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImm321s(
5659 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder,
5660 bool Inv) {
5661
5662 unsigned int Op;
5663 if (DstSize == 128) {
5664 if (Bits.getHiBits(64) != Bits.getLoBits(64))
5665 return nullptr;
5666 Op = Inv ? AArch64::MVNIv4s_msl : AArch64::MOVIv4s_msl;
5667 } else {
5668 Op = Inv ? AArch64::MVNIv2s_msl : AArch64::MOVIv2s_msl;
5669 }
5670
5671 uint64_t Val = Bits.zextOrTrunc(64).getZExtValue();
5672 uint64_t Shift;
5673
5676 Shift = 264;
5677 } else if (AArch64_AM::isAdvSIMDModImmType8(Val)) {
5679 Shift = 272;
5680 } else
5681 return nullptr;
5682
5683 auto Mov = Builder.buildInstr(Op, {Dst}, {}).addImm(Val).addImm(Shift);
5685 return &*Mov;
5686}
5687
5688MachineInstr *AArch64InstructionSelector::tryAdvSIMDModImmFP(
5689 Register Dst, unsigned DstSize, APInt Bits, MachineIRBuilder &Builder) {
5690
5691 unsigned int Op;
5692 bool IsWide = false;
5693 if (DstSize == 128) {
5694 if (Bits.getHiBits(64) != Bits.getLoBits(64))
5695 return nullptr;
5696 Op = AArch64::FMOVv4f32_ns;
5697 IsWide = true;
5698 } else {
5699 Op = AArch64::FMOVv2f32_ns;
5700 }
5701
5702 uint64_t Val = Bits.zextOrTrunc(64).getZExtValue();
5703
5706 } else if (IsWide && AArch64_AM::isAdvSIMDModImmType12(Val)) {
5708 Op = AArch64::FMOVv2f64_ns;
5709 } else
5710 return nullptr;
5711
5712 auto Mov = Builder.buildInstr(Op, {Dst}, {}).addImm(Val);
5714 return &*Mov;
5715}
5716
5717bool AArch64InstructionSelector::selectIndexedExtLoad(
5718 MachineInstr &MI, MachineRegisterInfo &MRI) {
5719 auto &ExtLd = cast<GIndexedAnyExtLoad>(MI);
5720 Register Dst = ExtLd.getDstReg();
5721 Register WriteBack = ExtLd.getWritebackReg();
5722 Register Base = ExtLd.getBaseReg();
5723 Register Offset = ExtLd.getOffsetReg();
5724 LLT Ty = MRI.getType(Dst);
5725 assert(Ty.getSizeInBits() <= 64); // Only for scalar GPRs.
5726 unsigned MemSizeBits = ExtLd.getMMO().getMemoryType().getSizeInBits();
5727 bool IsPre = ExtLd.isPre();
5728 bool IsSExt = isa<GIndexedSExtLoad>(ExtLd);
5729 unsigned InsertIntoSubReg = 0;
5730 bool IsDst64 = Ty.getSizeInBits() == 64;
5731
5732 // ZExt/SExt should be on gpr but can handle extload and zextload of fpr, so
5733 // long as they are scalar.
5734 bool IsFPR = RBI.getRegBank(Dst, MRI, TRI)->getID() == AArch64::FPRRegBankID;
5735 if ((IsSExt && IsFPR) || Ty.isVector())
5736 return false;
5737
5738 unsigned Opc = 0;
5739 LLT NewLdDstTy;
5740 LLT s32 = LLT::scalar(32);
5741 LLT s64 = LLT::scalar(64);
5742
5743 if (MemSizeBits == 8) {
5744 if (IsSExt) {
5745 if (IsDst64)
5746 Opc = IsPre ? AArch64::LDRSBXpre : AArch64::LDRSBXpost;
5747 else
5748 Opc = IsPre ? AArch64::LDRSBWpre : AArch64::LDRSBWpost;
5749 NewLdDstTy = IsDst64 ? s64 : s32;
5750 } else if (IsFPR) {
5751 Opc = IsPre ? AArch64::LDRBpre : AArch64::LDRBpost;
5752 InsertIntoSubReg = AArch64::bsub;
5753 NewLdDstTy = LLT::scalar(MemSizeBits);
5754 } else {
5755 Opc = IsPre ? AArch64::LDRBBpre : AArch64::LDRBBpost;
5756 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5757 NewLdDstTy = s32;
5758 }
5759 } else if (MemSizeBits == 16) {
5760 if (IsSExt) {
5761 if (IsDst64)
5762 Opc = IsPre ? AArch64::LDRSHXpre : AArch64::LDRSHXpost;
5763 else
5764 Opc = IsPre ? AArch64::LDRSHWpre : AArch64::LDRSHWpost;
5765 NewLdDstTy = IsDst64 ? s64 : s32;
5766 } else if (IsFPR) {
5767 Opc = IsPre ? AArch64::LDRHpre : AArch64::LDRHpost;
5768 InsertIntoSubReg = AArch64::hsub;
5769 NewLdDstTy = LLT::scalar(MemSizeBits);
5770 } else {
5771 Opc = IsPre ? AArch64::LDRHHpre : AArch64::LDRHHpost;
5772 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5773 NewLdDstTy = s32;
5774 }
5775 } else if (MemSizeBits == 32) {
5776 if (IsSExt) {
5777 Opc = IsPre ? AArch64::LDRSWpre : AArch64::LDRSWpost;
5778 NewLdDstTy = s64;
5779 } else if (IsFPR) {
5780 Opc = IsPre ? AArch64::LDRSpre : AArch64::LDRSpost;
5781 InsertIntoSubReg = AArch64::ssub;
5782 NewLdDstTy = LLT::scalar(MemSizeBits);
5783 } else {
5784 Opc = IsPre ? AArch64::LDRWpre : AArch64::LDRWpost;
5785 InsertIntoSubReg = IsDst64 ? AArch64::sub_32 : 0;
5786 NewLdDstTy = s32;
5787 }
5788 } else {
5789 llvm_unreachable("Unexpected size for indexed load");
5790 }
5791
5792 auto Cst = getIConstantVRegVal(Offset, MRI);
5793 if (!Cst)
5794 return false; // Shouldn't happen, but just in case.
5795
5796 auto LdMI = MIB.buildInstr(Opc, {WriteBack, NewLdDstTy}, {Base})
5797 .addImm(Cst->getSExtValue());
5798 LdMI.cloneMemRefs(ExtLd);
5800 // Make sure to select the load with the MemTy as the dest type, and then
5801 // insert into a larger reg if needed.
5802 if (InsertIntoSubReg) {
5803 // Generate a SUBREG_TO_REG.
5804 auto SubToReg = MIB.buildInstr(TargetOpcode::SUBREG_TO_REG, {Dst}, {})
5805 .addUse(LdMI.getReg(1))
5806 .addImm(InsertIntoSubReg);
5808 SubToReg.getReg(0),
5809 *getRegClassForTypeOnBank(MRI.getType(Dst),
5810 *RBI.getRegBank(Dst, MRI, TRI)),
5811 MRI);
5812 } else {
5813 auto Copy = MIB.buildCopy(Dst, LdMI.getReg(1));
5814 selectCopy(*Copy, TII, MRI, TRI, RBI);
5815 }
5816 MI.eraseFromParent();
5817
5818 return true;
5819}
5820
5821bool AArch64InstructionSelector::selectIndexedLoad(MachineInstr &MI,
5822 MachineRegisterInfo &MRI) {
5823 auto &Ld = cast<GIndexedLoad>(MI);
5824 Register Dst = Ld.getDstReg();
5825 Register WriteBack = Ld.getWritebackReg();
5826 Register Base = Ld.getBaseReg();
5827 Register Offset = Ld.getOffsetReg();
5828 assert(MRI.getType(Dst).getSizeInBits() <= 128 &&
5829 "Unexpected type for indexed load");
5830 unsigned MemSize = Ld.getMMO().getMemoryType().getSizeInBytes();
5831
5832 if (MemSize < MRI.getType(Dst).getSizeInBytes())
5833 return selectIndexedExtLoad(MI, MRI);
5834
5835 unsigned Opc = 0;
5836 if (Ld.isPre()) {
5837 static constexpr unsigned GPROpcodes[] = {
5838 AArch64::LDRBBpre, AArch64::LDRHHpre, AArch64::LDRWpre,
5839 AArch64::LDRXpre};
5840 static constexpr unsigned FPROpcodes[] = {
5841 AArch64::LDRBpre, AArch64::LDRHpre, AArch64::LDRSpre, AArch64::LDRDpre,
5842 AArch64::LDRQpre};
5843 Opc = (RBI.getRegBank(Dst, MRI, TRI)->getID() == AArch64::FPRRegBankID)
5844 ? FPROpcodes[Log2_32(MemSize)]
5845 : GPROpcodes[Log2_32(MemSize)];
5846 ;
5847 } else {
5848 static constexpr unsigned GPROpcodes[] = {
5849 AArch64::LDRBBpost, AArch64::LDRHHpost, AArch64::LDRWpost,
5850 AArch64::LDRXpost};
5851 static constexpr unsigned FPROpcodes[] = {
5852 AArch64::LDRBpost, AArch64::LDRHpost, AArch64::LDRSpost,
5853 AArch64::LDRDpost, AArch64::LDRQpost};
5854 Opc = (RBI.getRegBank(Dst, MRI, TRI)->getID() == AArch64::FPRRegBankID)
5855 ? FPROpcodes[Log2_32(MemSize)]
5856 : GPROpcodes[Log2_32(MemSize)];
5857 ;
5858 }
5859 auto Cst = getIConstantVRegVal(Offset, MRI);
5860 if (!Cst)
5861 return false; // Shouldn't happen, but just in case.
5862 auto LdMI =
5863 MIB.buildInstr(Opc, {WriteBack, Dst}, {Base}).addImm(Cst->getSExtValue());
5864 LdMI.cloneMemRefs(Ld);
5866 MI.eraseFromParent();
5867 return true;
5868}
5869
5870bool AArch64InstructionSelector::selectIndexedStore(GIndexedStore &I,
5871 MachineRegisterInfo &MRI) {
5872 Register Dst = I.getWritebackReg();
5873 Register Val = I.getValueReg();
5874 Register Base = I.getBaseReg();
5875 Register Offset = I.getOffsetReg();
5876 assert(MRI.getType(Val).getSizeInBits() <= 128 &&
5877 "Unexpected type for indexed store");
5878
5879 LocationSize MemSize = I.getMMO().getSize();
5880 unsigned MemSizeInBytes = MemSize.getValue();
5881
5882 assert(MemSizeInBytes && MemSizeInBytes <= 16 &&
5883 "Unexpected indexed store size");
5884 unsigned MemSizeLog2 = Log2_32(MemSizeInBytes);
5885
5886 unsigned Opc = 0;
5887 if (I.isPre()) {
5888 static constexpr unsigned GPROpcodes[] = {
5889 AArch64::STRBBpre, AArch64::STRHHpre, AArch64::STRWpre,
5890 AArch64::STRXpre};
5891 static constexpr unsigned FPROpcodes[] = {
5892 AArch64::STRBpre, AArch64::STRHpre, AArch64::STRSpre, AArch64::STRDpre,
5893 AArch64::STRQpre};
5894
5895 if (RBI.getRegBank(Val, MRI, TRI)->getID() == AArch64::FPRRegBankID)
5896 Opc = FPROpcodes[MemSizeLog2];
5897 else
5898 Opc = GPROpcodes[MemSizeLog2];
5899 } else {
5900 static constexpr unsigned GPROpcodes[] = {
5901 AArch64::STRBBpost, AArch64::STRHHpost, AArch64::STRWpost,
5902 AArch64::STRXpost};
5903 static constexpr unsigned FPROpcodes[] = {
5904 AArch64::STRBpost, AArch64::STRHpost, AArch64::STRSpost,
5905 AArch64::STRDpost, AArch64::STRQpost};
5906
5907 if (RBI.getRegBank(Val, MRI, TRI)->getID() == AArch64::FPRRegBankID)
5908 Opc = FPROpcodes[MemSizeLog2];
5909 else
5910 Opc = GPROpcodes[MemSizeLog2];
5911 }
5912
5913 auto Cst = getIConstantVRegVal(Offset, MRI);
5914 if (!Cst)
5915 return false; // Shouldn't happen, but just in case.
5916 auto Str =
5917 MIB.buildInstr(Opc, {Dst}, {Val, Base}).addImm(Cst->getSExtValue());
5918 Str.cloneMemRefs(I);
5920 I.eraseFromParent();
5921 return true;
5922}
5923
5924MachineInstr *
5925AArch64InstructionSelector::emitConstantVector(Register Dst, Constant *CV,
5926 MachineIRBuilder &MIRBuilder,
5927 MachineRegisterInfo &MRI) {
5928 LLT DstTy = MRI.getType(Dst);
5929 unsigned DstSize = DstTy.getSizeInBits();
5930 assert((DstSize == 64 || DstSize == 128) &&
5931 "Unexpected vector constant size");
5932
5933 if (CV->isNullValue()) {
5934 if (DstSize == 128) {
5935 auto Mov =
5936 MIRBuilder.buildInstr(AArch64::MOVIv2d_ns, {Dst}, {}).addImm(0);
5938 return &*Mov;
5939 }
5940
5941 if (DstSize == 64) {
5942 auto Mov =
5943 MIRBuilder
5944 .buildInstr(AArch64::MOVIv2d_ns, {&AArch64::FPR128RegClass}, {})
5945 .addImm(0);
5946 auto Copy = MIRBuilder.buildInstr(TargetOpcode::COPY, {Dst}, {})
5947 .addReg(Mov.getReg(0), {}, AArch64::dsub);
5948 RBI.constrainGenericRegister(Dst, AArch64::FPR64RegClass, MRI);
5949 return &*Copy;
5950 }
5951 }
5952
5953 if (Constant *SplatValue = CV->getSplatValue()) {
5954 APInt SplatValueAsInt =
5955 isa<ConstantFP>(SplatValue)
5956 ? cast<ConstantFP>(SplatValue)->getValueAPF().bitcastToAPInt()
5957 : SplatValue->getUniqueInteger();
5958 APInt DefBits = APInt::getSplat(
5959 DstSize, SplatValueAsInt.trunc(DstTy.getScalarSizeInBits()));
5960 auto TryMOVIWithBits = [&](APInt DefBits) -> MachineInstr * {
5961 MachineInstr *NewOp;
5962 bool Inv = false;
5963 if ((NewOp = tryAdvSIMDModImm64(Dst, DstSize, DefBits, MIRBuilder)) ||
5964 (NewOp =
5965 tryAdvSIMDModImm32(Dst, DstSize, DefBits, MIRBuilder, Inv)) ||
5966 (NewOp =
5967 tryAdvSIMDModImm321s(Dst, DstSize, DefBits, MIRBuilder, Inv)) ||
5968 (NewOp =
5969 tryAdvSIMDModImm16(Dst, DstSize, DefBits, MIRBuilder, Inv)) ||
5970 (NewOp = tryAdvSIMDModImm8(Dst, DstSize, DefBits, MIRBuilder)) ||
5971 (NewOp = tryAdvSIMDModImmFP(Dst, DstSize, DefBits, MIRBuilder)))
5972 return NewOp;
5973
5974 DefBits = ~DefBits;
5975 Inv = true;
5976 if ((NewOp =
5977 tryAdvSIMDModImm32(Dst, DstSize, DefBits, MIRBuilder, Inv)) ||
5978 (NewOp =
5979 tryAdvSIMDModImm321s(Dst, DstSize, DefBits, MIRBuilder, Inv)) ||
5980 (NewOp = tryAdvSIMDModImm16(Dst, DstSize, DefBits, MIRBuilder, Inv)))
5981 return NewOp;
5982 return nullptr;
5983 };
5984
5985 if (auto *NewOp = TryMOVIWithBits(DefBits))
5986 return NewOp;
5987
5988 // See if a fneg of the constant can be materialized with a MOVI, etc
5989 auto TryWithFNeg = [&](APInt DefBits, int NumBits,
5990 unsigned NegOpc) -> MachineInstr * {
5991 // FNegate each sub-element of the constant
5992 APInt Neg = APInt::getHighBitsSet(NumBits, 1).zext(DstSize);
5993 APInt NegBits(DstSize, 0);
5994 unsigned NumElts = DstSize / NumBits;
5995 for (unsigned i = 0; i < NumElts; i++)
5996 NegBits |= Neg << (NumBits * i);
5997 NegBits = DefBits ^ NegBits;
5998
5999 // Try to create the new constants with MOVI, and if so generate a fneg
6000 // for it.
6001 if (auto *NewOp = TryMOVIWithBits(NegBits)) {
6002 Register NewDst = MRI.createVirtualRegister(
6003 DstSize == 64 ? &AArch64::FPR64RegClass : &AArch64::FPR128RegClass);
6004 NewOp->getOperand(0).setReg(NewDst);
6005 return MIRBuilder.buildInstr(NegOpc, {Dst}, {NewDst});
6006 }
6007 return nullptr;
6008 };
6009 MachineInstr *R;
6010 if ((R = TryWithFNeg(DefBits, 32,
6011 DstSize == 64 ? AArch64::FNEGv2f32
6012 : AArch64::FNEGv4f32)) ||
6013 (R = TryWithFNeg(DefBits, 64,
6014 DstSize == 64 ? AArch64::FNEGDr
6015 : AArch64::FNEGv2f64)) ||
6016 (STI.hasFullFP16() &&
6017 (R = TryWithFNeg(DefBits, 16,
6018 DstSize == 64 ? AArch64::FNEGv4f16
6019 : AArch64::FNEGv8f16))))
6020 return R;
6021 }
6022
6023 auto *CPLoad = emitLoadFromConstantPool(CV, MIRBuilder);
6024 if (!CPLoad) {
6025 LLVM_DEBUG(dbgs() << "Could not generate cp load for constant vector!");
6026 return nullptr;
6027 }
6028
6029 auto Copy = MIRBuilder.buildCopy(Dst, CPLoad->getOperand(0));
6031 Dst, *MRI.getRegClass(CPLoad->getOperand(0).getReg()), MRI);
6032 return &*Copy;
6033}
6034
6035bool AArch64InstructionSelector::tryOptConstantBuildVec(
6036 MachineInstr &I, LLT DstTy, MachineRegisterInfo &MRI) {
6037 assert(I.getOpcode() == TargetOpcode::G_BUILD_VECTOR);
6038 unsigned DstSize = DstTy.getSizeInBits();
6039 assert(DstSize <= 128 && "Unexpected build_vec type!");
6040 if (DstSize < 32)
6041 return false;
6042 // Check if we're building a constant vector, in which case we want to
6043 // generate a constant pool load instead of a vector insert sequence.
6045 for (unsigned Idx = 1; Idx < I.getNumOperands(); ++Idx) {
6046 Register OpReg = I.getOperand(Idx).getReg();
6047 if (auto AnyConst = getAnyConstantVRegValWithLookThrough(
6048 OpReg, MRI, /*LookThroughInstrs=*/true,
6049 /*LookThroughAnyExt=*/true)) {
6050 MachineInstr *DefMI = MRI.getVRegDef(AnyConst->VReg);
6051
6052 if (DefMI->getOpcode() == TargetOpcode::G_CONSTANT) {
6053 Csts.emplace_back(
6054 ConstantInt::get(MIB.getMF().getFunction().getContext(),
6055 std::move(AnyConst->Value)));
6056 continue;
6057 }
6058
6059 if (DefMI->getOpcode() == TargetOpcode::G_FCONSTANT) {
6060 Csts.emplace_back(
6061 const_cast<ConstantFP *>(DefMI->getOperand(1).getFPImm()));
6062 continue;
6063 }
6064 }
6065 return false;
6066 }
6067 Constant *CV = ConstantVector::get(Csts);
6068 if (!emitConstantVector(I.getOperand(0).getReg(), CV, MIB, MRI))
6069 return false;
6070 I.eraseFromParent();
6071 return true;
6072}
6073
6074bool AArch64InstructionSelector::tryOptBuildVecToSubregToReg(
6075 MachineInstr &I, MachineRegisterInfo &MRI) {
6076 // Given:
6077 // %vec = G_BUILD_VECTOR %elt, %undef, %undef, ... %undef
6078 //
6079 // Select the G_BUILD_VECTOR as a SUBREG_TO_REG from %elt.
6080 Register Dst = I.getOperand(0).getReg();
6081 Register EltReg = I.getOperand(1).getReg();
6082 LLT EltTy = MRI.getType(EltReg);
6083 // If the index isn't on the same bank as its elements, then this can't be a
6084 // SUBREG_TO_REG.
6085 const RegisterBank &EltRB = *RBI.getRegBank(EltReg, MRI, TRI);
6086 const RegisterBank &DstRB = *RBI.getRegBank(Dst, MRI, TRI);
6087 if (EltRB != DstRB)
6088 return false;
6089 if (any_of(drop_begin(I.operands(), 2), [&MRI](const MachineOperand &Op) {
6090 return !getOpcodeDef(TargetOpcode::G_IMPLICIT_DEF, Op.getReg(), MRI);
6091 }))
6092 return false;
6093 unsigned SubReg;
6094 const TargetRegisterClass *EltRC = getRegClassForTypeOnBank(EltTy, EltRB);
6095 if (!EltRC)
6096 return false;
6097 const TargetRegisterClass *DstRC =
6098 getRegClassForTypeOnBank(MRI.getType(Dst), DstRB);
6099 if (!DstRC)
6100 return false;
6101 if (!getSubRegForClass(EltRC, TRI, SubReg))
6102 return false;
6103 auto SubregToReg = MIB.buildInstr(AArch64::SUBREG_TO_REG, {Dst}, {})
6104 .addUse(EltReg)
6105 .addImm(SubReg);
6106 I.eraseFromParent();
6107 constrainSelectedInstRegOperands(*SubregToReg, TII, TRI, RBI);
6108 return RBI.constrainGenericRegister(Dst, *DstRC, MRI);
6109}
6110
6111bool AArch64InstructionSelector::selectBuildVector(MachineInstr &I,
6112 MachineRegisterInfo &MRI) {
6113 assert(I.getOpcode() == TargetOpcode::G_BUILD_VECTOR);
6114 // Until we port more of the optimized selections, for now just use a vector
6115 // insert sequence.
6116 const LLT DstTy = MRI.getType(I.getOperand(0).getReg());
6117 const LLT EltTy = MRI.getType(I.getOperand(1).getReg());
6118 unsigned EltSize = EltTy.getSizeInBits();
6119
6120 if (tryOptConstantBuildVec(I, DstTy, MRI))
6121 return true;
6122 if (tryOptBuildVecToSubregToReg(I, MRI))
6123 return true;
6124
6125 if (EltSize != 8 && EltSize != 16 && EltSize != 32 && EltSize != 64)
6126 return false; // Don't support all element types yet.
6127 const RegisterBank &RB = *RBI.getRegBank(I.getOperand(1).getReg(), MRI, TRI);
6128
6129 const TargetRegisterClass *DstRC = &AArch64::FPR128RegClass;
6130 MachineInstr *ScalarToVec =
6131 emitScalarToVector(DstTy.getElementType().getSizeInBits(), DstRC,
6132 I.getOperand(1).getReg(), MIB);
6133 if (!ScalarToVec)
6134 return false;
6135
6136 Register DstVec = ScalarToVec->getOperand(0).getReg();
6137 unsigned DstSize = DstTy.getSizeInBits();
6138
6139 // Keep track of the last MI we inserted. Later on, we might be able to save
6140 // a copy using it.
6141 MachineInstr *PrevMI = ScalarToVec;
6142 for (unsigned i = 2, e = DstSize / EltSize + 1; i < e; ++i) {
6143 // Note that if we don't do a subregister copy, we can end up making an
6144 // extra register.
6145 Register OpReg = I.getOperand(i).getReg();
6146 // Do not emit inserts for undefs
6147 if (!getOpcodeDef<GImplicitDef>(OpReg, MRI)) {
6148 PrevMI = &*emitLaneInsert(std::nullopt, DstVec, OpReg, i - 1, RB, MIB);
6149 DstVec = PrevMI->getOperand(0).getReg();
6150 }
6151 }
6152
6153 // If DstTy's size in bits is less than 128, then emit a subregister copy
6154 // from DstVec to the last register we've defined.
6155 if (DstSize < 128) {
6156 // Force this to be FPR using the destination vector.
6157 const TargetRegisterClass *RC =
6158 getRegClassForTypeOnBank(DstTy, *RBI.getRegBank(DstVec, MRI, TRI));
6159 if (!RC)
6160 return false;
6161 if (RC != &AArch64::FPR32RegClass && RC != &AArch64::FPR64RegClass) {
6162 LLVM_DEBUG(dbgs() << "Unsupported register class!\n");
6163 return false;
6164 }
6165
6166 unsigned SubReg = 0;
6167 if (!getSubRegForClass(RC, TRI, SubReg))
6168 return false;
6169 if (SubReg != AArch64::ssub && SubReg != AArch64::dsub) {
6170 LLVM_DEBUG(dbgs() << "Unsupported destination size! (" << DstSize
6171 << "\n");
6172 return false;
6173 }
6174
6176 Register DstReg = I.getOperand(0).getReg();
6177
6178 MIB.buildInstr(TargetOpcode::COPY, {DstReg}, {}).addReg(DstVec, {}, SubReg);
6179 MachineOperand &RegOp = I.getOperand(1);
6180 RegOp.setReg(Reg);
6181 RBI.constrainGenericRegister(DstReg, *RC, MRI);
6182 } else {
6183 // We either have a vector with all elements (except the first one) undef or
6184 // at least one non-undef non-first element. In the first case, we need to
6185 // constrain the output register ourselves as we may have generated an
6186 // INSERT_SUBREG operation which is a generic operation for which the
6187 // output regclass cannot be automatically chosen.
6188 //
6189 // In the second case, there is no need to do this as it may generate an
6190 // instruction like INSvi32gpr where the regclass can be automatically
6191 // chosen.
6192 //
6193 // Also, we save a copy by re-using the destination register on the final
6194 // insert.
6195 PrevMI->getOperand(0).setReg(I.getOperand(0).getReg());
6197
6198 Register DstReg = PrevMI->getOperand(0).getReg();
6199 if (PrevMI == ScalarToVec && DstReg.isVirtual()) {
6200 const TargetRegisterClass *RC =
6201 getRegClassForTypeOnBank(DstTy, *RBI.getRegBank(DstVec, MRI, TRI));
6202 RBI.constrainGenericRegister(DstReg, *RC, MRI);
6203 }
6204 }
6205
6207 return true;
6208}
6209
6210bool AArch64InstructionSelector::selectVectorLoadIntrinsic(unsigned Opc,
6211 unsigned NumVecs,
6212 MachineInstr &I) {
6213 assert(I.getOpcode() == TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS);
6214 assert(Opc && "Expected an opcode?");
6215 assert(NumVecs > 1 && NumVecs < 5 && "Only support 2, 3, or 4 vectors");
6216 auto &MRI = *MIB.getMRI();
6217 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6218 unsigned Size = Ty.getSizeInBits();
6219 assert((Size == 64 || Size == 128) &&
6220 "Destination must be 64 bits or 128 bits?");
6221 unsigned SubReg = Size == 64 ? AArch64::dsub0 : AArch64::qsub0;
6222 auto Ptr = I.getOperand(I.getNumOperands() - 1).getReg();
6223 assert(MRI.getType(Ptr).isPointer() && "Expected a pointer type?");
6224 auto Load = MIB.buildInstr(Opc, {Ty}, {Ptr});
6227 Register SelectedLoadDst = Load->getOperand(0).getReg();
6228 for (unsigned Idx = 0; Idx < NumVecs; ++Idx) {
6229 auto Vec = MIB.buildInstr(TargetOpcode::COPY, {I.getOperand(Idx)}, {})
6230 .addReg(SelectedLoadDst, {}, SubReg + Idx);
6231 // Emit the subreg copies and immediately select them.
6232 // FIXME: We should refactor our copy code into an emitCopy helper and
6233 // clean up uses of this pattern elsewhere in the selector.
6234 selectCopy(*Vec, TII, MRI, TRI, RBI);
6235 }
6236 return true;
6237}
6238
6239bool AArch64InstructionSelector::selectVectorLoadLaneIntrinsic(
6240 unsigned Opc, unsigned NumVecs, MachineInstr &I) {
6241 assert(I.getOpcode() == TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS);
6242 assert(Opc && "Expected an opcode?");
6243 assert(NumVecs > 1 && NumVecs < 5 && "Only support 2, 3, or 4 vectors");
6244 auto &MRI = *MIB.getMRI();
6245 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6246 bool Narrow = Ty.getSizeInBits() == 64;
6247
6248 auto FirstSrcRegIt = I.operands_begin() + NumVecs + 1;
6249 SmallVector<Register, 4> Regs(NumVecs);
6250 std::transform(FirstSrcRegIt, FirstSrcRegIt + NumVecs, Regs.begin(),
6251 [](auto MO) { return MO.getReg(); });
6252
6253 if (Narrow) {
6254 transform(Regs, Regs.begin(), [this](Register Reg) {
6255 return emitScalarToVector(64, &AArch64::FPR128RegClass, Reg, MIB)
6256 ->getOperand(0)
6257 .getReg();
6258 });
6259 Ty = Ty.multiplyElements(2);
6260 }
6261
6262 Register Tuple = createQTuple(Regs, MIB);
6263 auto LaneNo = getIConstantVRegVal((FirstSrcRegIt + NumVecs)->getReg(), MRI);
6264 if (!LaneNo)
6265 return false;
6266
6267 Register Ptr = (FirstSrcRegIt + NumVecs + 1)->getReg();
6268 auto Load = MIB.buildInstr(Opc, {Ty}, {})
6269 .addReg(Tuple)
6270 .addImm(LaneNo->getZExtValue())
6271 .addReg(Ptr);
6274 Register SelectedLoadDst = Load->getOperand(0).getReg();
6275 unsigned SubReg = AArch64::qsub0;
6276 for (unsigned Idx = 0; Idx < NumVecs; ++Idx) {
6277 auto Vec = MIB.buildInstr(TargetOpcode::COPY,
6278 {Narrow ? DstOp(&AArch64::FPR128RegClass)
6279 : DstOp(I.getOperand(Idx).getReg())},
6280 {})
6281 .addReg(SelectedLoadDst, {}, SubReg + Idx);
6282 Register WideReg = Vec.getReg(0);
6283 // Emit the subreg copies and immediately select them.
6284 selectCopy(*Vec, TII, MRI, TRI, RBI);
6285 if (Narrow &&
6286 !emitNarrowVector(I.getOperand(Idx).getReg(), WideReg, MIB, MRI))
6287 return false;
6288 }
6289 return true;
6290}
6291
6292void AArch64InstructionSelector::selectVectorStoreIntrinsic(MachineInstr &I,
6293 unsigned NumVecs,
6294 unsigned Opc) {
6295 MachineRegisterInfo &MRI = I.getParent()->getParent()->getRegInfo();
6296 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6297 Register Ptr = I.getOperand(1 + NumVecs).getReg();
6298
6299 SmallVector<Register, 2> Regs(NumVecs);
6300 std::transform(I.operands_begin() + 1, I.operands_begin() + 1 + NumVecs,
6301 Regs.begin(), [](auto MO) { return MO.getReg(); });
6302
6303 Register Tuple = Ty.getSizeInBits() == 128 ? createQTuple(Regs, MIB)
6304 : createDTuple(Regs, MIB);
6305 auto Store = MIB.buildInstr(Opc, {}, {Tuple, Ptr});
6308}
6309
6310bool AArch64InstructionSelector::selectVectorStoreLaneIntrinsic(
6311 MachineInstr &I, unsigned NumVecs, unsigned Opc) {
6312 MachineRegisterInfo &MRI = I.getParent()->getParent()->getRegInfo();
6313 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6314 bool Narrow = Ty.getSizeInBits() == 64;
6315
6316 SmallVector<Register, 2> Regs(NumVecs);
6317 std::transform(I.operands_begin() + 1, I.operands_begin() + 1 + NumVecs,
6318 Regs.begin(), [](auto MO) { return MO.getReg(); });
6319
6320 if (Narrow)
6321 transform(Regs, Regs.begin(), [this](Register Reg) {
6322 return emitScalarToVector(64, &AArch64::FPR128RegClass, Reg, MIB)
6323 ->getOperand(0)
6324 .getReg();
6325 });
6326
6327 Register Tuple = createQTuple(Regs, MIB);
6328
6329 auto LaneNo = getIConstantVRegVal(I.getOperand(1 + NumVecs).getReg(), MRI);
6330 if (!LaneNo)
6331 return false;
6332 Register Ptr = I.getOperand(1 + NumVecs + 1).getReg();
6333 auto Store = MIB.buildInstr(Opc, {}, {})
6334 .addReg(Tuple)
6335 .addImm(LaneNo->getZExtValue())
6336 .addReg(Ptr);
6339 return true;
6340}
6341
6342bool AArch64InstructionSelector::selectIntrinsicWithSideEffects(
6343 MachineInstr &I, MachineRegisterInfo &MRI) {
6344 // Find the intrinsic ID.
6345 unsigned IntrinID = cast<GIntrinsic>(I).getIntrinsicID();
6346
6347 const LLT S8 = LLT::scalar(8);
6348 const LLT S16 = LLT::scalar(16);
6349 const LLT S32 = LLT::scalar(32);
6350 const LLT S64 = LLT::scalar(64);
6351 const LLT P0 = LLT::pointer(0, 64);
6352 // Select the instruction.
6353 switch (IntrinID) {
6354 default:
6355 return false;
6356 case Intrinsic::aarch64_ldxp:
6357 case Intrinsic::aarch64_ldaxp: {
6358 auto NewI = MIB.buildInstr(
6359 IntrinID == Intrinsic::aarch64_ldxp ? AArch64::LDXPX : AArch64::LDAXPX,
6360 {I.getOperand(0).getReg(), I.getOperand(1).getReg()},
6361 {I.getOperand(3)});
6362 NewI.cloneMemRefs(I);
6364 break;
6365 }
6366 case Intrinsic::aarch64_neon_ld1x2: {
6367 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6368 unsigned Opc = 0;
6369 if (Ty == LLT::fixed_vector(8, S8))
6370 Opc = AArch64::LD1Twov8b;
6371 else if (Ty == LLT::fixed_vector(16, S8))
6372 Opc = AArch64::LD1Twov16b;
6373 else if (Ty == LLT::fixed_vector(4, S16))
6374 Opc = AArch64::LD1Twov4h;
6375 else if (Ty == LLT::fixed_vector(8, S16))
6376 Opc = AArch64::LD1Twov8h;
6377 else if (Ty == LLT::fixed_vector(2, S32))
6378 Opc = AArch64::LD1Twov2s;
6379 else if (Ty == LLT::fixed_vector(4, S32))
6380 Opc = AArch64::LD1Twov4s;
6381 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6382 Opc = AArch64::LD1Twov2d;
6383 else if (Ty == S64 || Ty == P0)
6384 Opc = AArch64::LD1Twov1d;
6385 else
6386 llvm_unreachable("Unexpected type for ld1x2!");
6387 selectVectorLoadIntrinsic(Opc, 2, I);
6388 break;
6389 }
6390 case Intrinsic::aarch64_neon_ld1x3: {
6391 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6392 unsigned Opc = 0;
6393 if (Ty == LLT::fixed_vector(8, S8))
6394 Opc = AArch64::LD1Threev8b;
6395 else if (Ty == LLT::fixed_vector(16, S8))
6396 Opc = AArch64::LD1Threev16b;
6397 else if (Ty == LLT::fixed_vector(4, S16))
6398 Opc = AArch64::LD1Threev4h;
6399 else if (Ty == LLT::fixed_vector(8, S16))
6400 Opc = AArch64::LD1Threev8h;
6401 else if (Ty == LLT::fixed_vector(2, S32))
6402 Opc = AArch64::LD1Threev2s;
6403 else if (Ty == LLT::fixed_vector(4, S32))
6404 Opc = AArch64::LD1Threev4s;
6405 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6406 Opc = AArch64::LD1Threev2d;
6407 else if (Ty == S64 || Ty == P0)
6408 Opc = AArch64::LD1Threev1d;
6409 else
6410 llvm_unreachable("Unexpected type for ld1x3!");
6411 selectVectorLoadIntrinsic(Opc, 3, I);
6412 break;
6413 }
6414 case Intrinsic::aarch64_neon_ld1x4: {
6415 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6416 unsigned Opc = 0;
6417 if (Ty == LLT::fixed_vector(8, S8))
6418 Opc = AArch64::LD1Fourv8b;
6419 else if (Ty == LLT::fixed_vector(16, S8))
6420 Opc = AArch64::LD1Fourv16b;
6421 else if (Ty == LLT::fixed_vector(4, S16))
6422 Opc = AArch64::LD1Fourv4h;
6423 else if (Ty == LLT::fixed_vector(8, S16))
6424 Opc = AArch64::LD1Fourv8h;
6425 else if (Ty == LLT::fixed_vector(2, S32))
6426 Opc = AArch64::LD1Fourv2s;
6427 else if (Ty == LLT::fixed_vector(4, S32))
6428 Opc = AArch64::LD1Fourv4s;
6429 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6430 Opc = AArch64::LD1Fourv2d;
6431 else if (Ty == S64 || Ty == P0)
6432 Opc = AArch64::LD1Fourv1d;
6433 else
6434 llvm_unreachable("Unexpected type for ld1x4!");
6435 selectVectorLoadIntrinsic(Opc, 4, I);
6436 break;
6437 }
6438 case Intrinsic::aarch64_neon_ld2: {
6439 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6440 unsigned Opc = 0;
6441 if (Ty == LLT::fixed_vector(8, S8))
6442 Opc = AArch64::LD2Twov8b;
6443 else if (Ty == LLT::fixed_vector(16, S8))
6444 Opc = AArch64::LD2Twov16b;
6445 else if (Ty == LLT::fixed_vector(4, S16))
6446 Opc = AArch64::LD2Twov4h;
6447 else if (Ty == LLT::fixed_vector(8, S16))
6448 Opc = AArch64::LD2Twov8h;
6449 else if (Ty == LLT::fixed_vector(2, S32))
6450 Opc = AArch64::LD2Twov2s;
6451 else if (Ty == LLT::fixed_vector(4, S32))
6452 Opc = AArch64::LD2Twov4s;
6453 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6454 Opc = AArch64::LD2Twov2d;
6455 else if (Ty == S64 || Ty == P0)
6456 Opc = AArch64::LD1Twov1d;
6457 else
6458 llvm_unreachable("Unexpected type for ld2!");
6459 selectVectorLoadIntrinsic(Opc, 2, I);
6460 break;
6461 }
6462 case Intrinsic::aarch64_neon_ld2lane: {
6463 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6464 unsigned Opc;
6465 if (Ty == LLT::fixed_vector(8, S8) || Ty == LLT::fixed_vector(16, S8))
6466 Opc = AArch64::LD2i8;
6467 else if (Ty == LLT::fixed_vector(4, S16) || Ty == LLT::fixed_vector(8, S16))
6468 Opc = AArch64::LD2i16;
6469 else if (Ty == LLT::fixed_vector(2, S32) || Ty == LLT::fixed_vector(4, S32))
6470 Opc = AArch64::LD2i32;
6471 else if (Ty == LLT::fixed_vector(2, S64) ||
6472 Ty == LLT::fixed_vector(2, P0) || Ty == S64 || Ty == P0)
6473 Opc = AArch64::LD2i64;
6474 else
6475 llvm_unreachable("Unexpected type for ld2lane!");
6476 if (!selectVectorLoadLaneIntrinsic(Opc, 2, I))
6477 return false;
6478 break;
6479 }
6480 case Intrinsic::aarch64_neon_ld2r: {
6481 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6482 unsigned Opc = 0;
6483 if (Ty == LLT::fixed_vector(8, S8))
6484 Opc = AArch64::LD2Rv8b;
6485 else if (Ty == LLT::fixed_vector(16, S8))
6486 Opc = AArch64::LD2Rv16b;
6487 else if (Ty == LLT::fixed_vector(4, S16))
6488 Opc = AArch64::LD2Rv4h;
6489 else if (Ty == LLT::fixed_vector(8, S16))
6490 Opc = AArch64::LD2Rv8h;
6491 else if (Ty == LLT::fixed_vector(2, S32))
6492 Opc = AArch64::LD2Rv2s;
6493 else if (Ty == LLT::fixed_vector(4, S32))
6494 Opc = AArch64::LD2Rv4s;
6495 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6496 Opc = AArch64::LD2Rv2d;
6497 else if (Ty == S64 || Ty == P0)
6498 Opc = AArch64::LD2Rv1d;
6499 else
6500 llvm_unreachable("Unexpected type for ld2r!");
6501 selectVectorLoadIntrinsic(Opc, 2, I);
6502 break;
6503 }
6504 case Intrinsic::aarch64_neon_ld3: {
6505 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6506 unsigned Opc = 0;
6507 if (Ty == LLT::fixed_vector(8, S8))
6508 Opc = AArch64::LD3Threev8b;
6509 else if (Ty == LLT::fixed_vector(16, S8))
6510 Opc = AArch64::LD3Threev16b;
6511 else if (Ty == LLT::fixed_vector(4, S16))
6512 Opc = AArch64::LD3Threev4h;
6513 else if (Ty == LLT::fixed_vector(8, S16))
6514 Opc = AArch64::LD3Threev8h;
6515 else if (Ty == LLT::fixed_vector(2, S32))
6516 Opc = AArch64::LD3Threev2s;
6517 else if (Ty == LLT::fixed_vector(4, S32))
6518 Opc = AArch64::LD3Threev4s;
6519 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6520 Opc = AArch64::LD3Threev2d;
6521 else if (Ty == S64 || Ty == P0)
6522 Opc = AArch64::LD1Threev1d;
6523 else
6524 llvm_unreachable("Unexpected type for ld3!");
6525 selectVectorLoadIntrinsic(Opc, 3, I);
6526 break;
6527 }
6528 case Intrinsic::aarch64_neon_ld3lane: {
6529 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6530 unsigned Opc;
6531 if (Ty == LLT::fixed_vector(8, S8) || Ty == LLT::fixed_vector(16, S8))
6532 Opc = AArch64::LD3i8;
6533 else if (Ty == LLT::fixed_vector(4, S16) || Ty == LLT::fixed_vector(8, S16))
6534 Opc = AArch64::LD3i16;
6535 else if (Ty == LLT::fixed_vector(2, S32) || Ty == LLT::fixed_vector(4, S32))
6536 Opc = AArch64::LD3i32;
6537 else if (Ty == LLT::fixed_vector(2, S64) ||
6538 Ty == LLT::fixed_vector(2, P0) || Ty == S64 || Ty == P0)
6539 Opc = AArch64::LD3i64;
6540 else
6541 llvm_unreachable("Unexpected type for ld3lane!");
6542 if (!selectVectorLoadLaneIntrinsic(Opc, 3, I))
6543 return false;
6544 break;
6545 }
6546 case Intrinsic::aarch64_neon_ld3r: {
6547 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6548 unsigned Opc = 0;
6549 if (Ty == LLT::fixed_vector(8, S8))
6550 Opc = AArch64::LD3Rv8b;
6551 else if (Ty == LLT::fixed_vector(16, S8))
6552 Opc = AArch64::LD3Rv16b;
6553 else if (Ty == LLT::fixed_vector(4, S16))
6554 Opc = AArch64::LD3Rv4h;
6555 else if (Ty == LLT::fixed_vector(8, S16))
6556 Opc = AArch64::LD3Rv8h;
6557 else if (Ty == LLT::fixed_vector(2, S32))
6558 Opc = AArch64::LD3Rv2s;
6559 else if (Ty == LLT::fixed_vector(4, S32))
6560 Opc = AArch64::LD3Rv4s;
6561 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6562 Opc = AArch64::LD3Rv2d;
6563 else if (Ty == S64 || Ty == P0)
6564 Opc = AArch64::LD3Rv1d;
6565 else
6566 llvm_unreachable("Unexpected type for ld3r!");
6567 selectVectorLoadIntrinsic(Opc, 3, I);
6568 break;
6569 }
6570 case Intrinsic::aarch64_neon_ld4: {
6571 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6572 unsigned Opc = 0;
6573 if (Ty == LLT::fixed_vector(8, S8))
6574 Opc = AArch64::LD4Fourv8b;
6575 else if (Ty == LLT::fixed_vector(16, S8))
6576 Opc = AArch64::LD4Fourv16b;
6577 else if (Ty == LLT::fixed_vector(4, S16))
6578 Opc = AArch64::LD4Fourv4h;
6579 else if (Ty == LLT::fixed_vector(8, S16))
6580 Opc = AArch64::LD4Fourv8h;
6581 else if (Ty == LLT::fixed_vector(2, S32))
6582 Opc = AArch64::LD4Fourv2s;
6583 else if (Ty == LLT::fixed_vector(4, S32))
6584 Opc = AArch64::LD4Fourv4s;
6585 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6586 Opc = AArch64::LD4Fourv2d;
6587 else if (Ty == S64 || Ty == P0)
6588 Opc = AArch64::LD1Fourv1d;
6589 else
6590 llvm_unreachable("Unexpected type for ld4!");
6591 selectVectorLoadIntrinsic(Opc, 4, I);
6592 break;
6593 }
6594 case Intrinsic::aarch64_neon_ld4lane: {
6595 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6596 unsigned Opc;
6597 if (Ty == LLT::fixed_vector(8, S8) || Ty == LLT::fixed_vector(16, S8))
6598 Opc = AArch64::LD4i8;
6599 else if (Ty == LLT::fixed_vector(4, S16) || Ty == LLT::fixed_vector(8, S16))
6600 Opc = AArch64::LD4i16;
6601 else if (Ty == LLT::fixed_vector(2, S32) || Ty == LLT::fixed_vector(4, S32))
6602 Opc = AArch64::LD4i32;
6603 else if (Ty == LLT::fixed_vector(2, S64) ||
6604 Ty == LLT::fixed_vector(2, P0) || Ty == S64 || Ty == P0)
6605 Opc = AArch64::LD4i64;
6606 else
6607 llvm_unreachable("Unexpected type for ld4lane!");
6608 if (!selectVectorLoadLaneIntrinsic(Opc, 4, I))
6609 return false;
6610 break;
6611 }
6612 case Intrinsic::aarch64_neon_ld4r: {
6613 LLT Ty = MRI.getType(I.getOperand(0).getReg());
6614 unsigned Opc = 0;
6615 if (Ty == LLT::fixed_vector(8, S8))
6616 Opc = AArch64::LD4Rv8b;
6617 else if (Ty == LLT::fixed_vector(16, S8))
6618 Opc = AArch64::LD4Rv16b;
6619 else if (Ty == LLT::fixed_vector(4, S16))
6620 Opc = AArch64::LD4Rv4h;
6621 else if (Ty == LLT::fixed_vector(8, S16))
6622 Opc = AArch64::LD4Rv8h;
6623 else if (Ty == LLT::fixed_vector(2, S32))
6624 Opc = AArch64::LD4Rv2s;
6625 else if (Ty == LLT::fixed_vector(4, S32))
6626 Opc = AArch64::LD4Rv4s;
6627 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6628 Opc = AArch64::LD4Rv2d;
6629 else if (Ty == S64 || Ty == P0)
6630 Opc = AArch64::LD4Rv1d;
6631 else
6632 llvm_unreachable("Unexpected type for ld4r!");
6633 selectVectorLoadIntrinsic(Opc, 4, I);
6634 break;
6635 }
6636 case Intrinsic::aarch64_neon_st1x2: {
6637 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6638 unsigned Opc;
6639 if (Ty == LLT::fixed_vector(8, S8))
6640 Opc = AArch64::ST1Twov8b;
6641 else if (Ty == LLT::fixed_vector(16, S8))
6642 Opc = AArch64::ST1Twov16b;
6643 else if (Ty == LLT::fixed_vector(4, S16))
6644 Opc = AArch64::ST1Twov4h;
6645 else if (Ty == LLT::fixed_vector(8, S16))
6646 Opc = AArch64::ST1Twov8h;
6647 else if (Ty == LLT::fixed_vector(2, S32))
6648 Opc = AArch64::ST1Twov2s;
6649 else if (Ty == LLT::fixed_vector(4, S32))
6650 Opc = AArch64::ST1Twov4s;
6651 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6652 Opc = AArch64::ST1Twov2d;
6653 else if (Ty == S64 || Ty == P0)
6654 Opc = AArch64::ST1Twov1d;
6655 else
6656 llvm_unreachable("Unexpected type for st1x2!");
6657 selectVectorStoreIntrinsic(I, 2, Opc);
6658 break;
6659 }
6660 case Intrinsic::aarch64_neon_st1x3: {
6661 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6662 unsigned Opc;
6663 if (Ty == LLT::fixed_vector(8, S8))
6664 Opc = AArch64::ST1Threev8b;
6665 else if (Ty == LLT::fixed_vector(16, S8))
6666 Opc = AArch64::ST1Threev16b;
6667 else if (Ty == LLT::fixed_vector(4, S16))
6668 Opc = AArch64::ST1Threev4h;
6669 else if (Ty == LLT::fixed_vector(8, S16))
6670 Opc = AArch64::ST1Threev8h;
6671 else if (Ty == LLT::fixed_vector(2, S32))
6672 Opc = AArch64::ST1Threev2s;
6673 else if (Ty == LLT::fixed_vector(4, S32))
6674 Opc = AArch64::ST1Threev4s;
6675 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6676 Opc = AArch64::ST1Threev2d;
6677 else if (Ty == S64 || Ty == P0)
6678 Opc = AArch64::ST1Threev1d;
6679 else
6680 llvm_unreachable("Unexpected type for st1x3!");
6681 selectVectorStoreIntrinsic(I, 3, Opc);
6682 break;
6683 }
6684 case Intrinsic::aarch64_neon_st1x4: {
6685 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6686 unsigned Opc;
6687 if (Ty == LLT::fixed_vector(8, S8))
6688 Opc = AArch64::ST1Fourv8b;
6689 else if (Ty == LLT::fixed_vector(16, S8))
6690 Opc = AArch64::ST1Fourv16b;
6691 else if (Ty == LLT::fixed_vector(4, S16))
6692 Opc = AArch64::ST1Fourv4h;
6693 else if (Ty == LLT::fixed_vector(8, S16))
6694 Opc = AArch64::ST1Fourv8h;
6695 else if (Ty == LLT::fixed_vector(2, S32))
6696 Opc = AArch64::ST1Fourv2s;
6697 else if (Ty == LLT::fixed_vector(4, S32))
6698 Opc = AArch64::ST1Fourv4s;
6699 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6700 Opc = AArch64::ST1Fourv2d;
6701 else if (Ty == S64 || Ty == P0)
6702 Opc = AArch64::ST1Fourv1d;
6703 else
6704 llvm_unreachable("Unexpected type for st1x4!");
6705 selectVectorStoreIntrinsic(I, 4, Opc);
6706 break;
6707 }
6708 case Intrinsic::aarch64_neon_st2: {
6709 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6710 unsigned Opc;
6711 if (Ty == LLT::fixed_vector(8, S8))
6712 Opc = AArch64::ST2Twov8b;
6713 else if (Ty == LLT::fixed_vector(16, S8))
6714 Opc = AArch64::ST2Twov16b;
6715 else if (Ty == LLT::fixed_vector(4, S16))
6716 Opc = AArch64::ST2Twov4h;
6717 else if (Ty == LLT::fixed_vector(8, S16))
6718 Opc = AArch64::ST2Twov8h;
6719 else if (Ty == LLT::fixed_vector(2, S32))
6720 Opc = AArch64::ST2Twov2s;
6721 else if (Ty == LLT::fixed_vector(4, S32))
6722 Opc = AArch64::ST2Twov4s;
6723 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6724 Opc = AArch64::ST2Twov2d;
6725 else if (Ty == S64 || Ty == P0)
6726 Opc = AArch64::ST1Twov1d;
6727 else
6728 llvm_unreachable("Unexpected type for st2!");
6729 selectVectorStoreIntrinsic(I, 2, Opc);
6730 break;
6731 }
6732 case Intrinsic::aarch64_neon_st3: {
6733 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6734 unsigned Opc;
6735 if (Ty == LLT::fixed_vector(8, S8))
6736 Opc = AArch64::ST3Threev8b;
6737 else if (Ty == LLT::fixed_vector(16, S8))
6738 Opc = AArch64::ST3Threev16b;
6739 else if (Ty == LLT::fixed_vector(4, S16))
6740 Opc = AArch64::ST3Threev4h;
6741 else if (Ty == LLT::fixed_vector(8, S16))
6742 Opc = AArch64::ST3Threev8h;
6743 else if (Ty == LLT::fixed_vector(2, S32))
6744 Opc = AArch64::ST3Threev2s;
6745 else if (Ty == LLT::fixed_vector(4, S32))
6746 Opc = AArch64::ST3Threev4s;
6747 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6748 Opc = AArch64::ST3Threev2d;
6749 else if (Ty == S64 || Ty == P0)
6750 Opc = AArch64::ST1Threev1d;
6751 else
6752 llvm_unreachable("Unexpected type for st3!");
6753 selectVectorStoreIntrinsic(I, 3, Opc);
6754 break;
6755 }
6756 case Intrinsic::aarch64_neon_st4: {
6757 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6758 unsigned Opc;
6759 if (Ty == LLT::fixed_vector(8, S8))
6760 Opc = AArch64::ST4Fourv8b;
6761 else if (Ty == LLT::fixed_vector(16, S8))
6762 Opc = AArch64::ST4Fourv16b;
6763 else if (Ty == LLT::fixed_vector(4, S16))
6764 Opc = AArch64::ST4Fourv4h;
6765 else if (Ty == LLT::fixed_vector(8, S16))
6766 Opc = AArch64::ST4Fourv8h;
6767 else if (Ty == LLT::fixed_vector(2, S32))
6768 Opc = AArch64::ST4Fourv2s;
6769 else if (Ty == LLT::fixed_vector(4, S32))
6770 Opc = AArch64::ST4Fourv4s;
6771 else if (Ty == LLT::fixed_vector(2, S64) || Ty == LLT::fixed_vector(2, P0))
6772 Opc = AArch64::ST4Fourv2d;
6773 else if (Ty == S64 || Ty == P0)
6774 Opc = AArch64::ST1Fourv1d;
6775 else
6776 llvm_unreachable("Unexpected type for st4!");
6777 selectVectorStoreIntrinsic(I, 4, Opc);
6778 break;
6779 }
6780 case Intrinsic::aarch64_neon_st2lane: {
6781 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6782 unsigned Opc;
6783 if (Ty == LLT::fixed_vector(8, S8) || Ty == LLT::fixed_vector(16, S8))
6784 Opc = AArch64::ST2i8;
6785 else if (Ty == LLT::fixed_vector(4, S16) || Ty == LLT::fixed_vector(8, S16))
6786 Opc = AArch64::ST2i16;
6787 else if (Ty == LLT::fixed_vector(2, S32) || Ty == LLT::fixed_vector(4, S32))
6788 Opc = AArch64::ST2i32;
6789 else if (Ty == LLT::fixed_vector(2, S64) ||
6790 Ty == LLT::fixed_vector(2, P0) || Ty == S64 || Ty == P0)
6791 Opc = AArch64::ST2i64;
6792 else
6793 llvm_unreachable("Unexpected type for st2lane!");
6794 if (!selectVectorStoreLaneIntrinsic(I, 2, Opc))
6795 return false;
6796 break;
6797 }
6798 case Intrinsic::aarch64_neon_st3lane: {
6799 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6800 unsigned Opc;
6801 if (Ty == LLT::fixed_vector(8, S8) || Ty == LLT::fixed_vector(16, S8))
6802 Opc = AArch64::ST3i8;
6803 else if (Ty == LLT::fixed_vector(4, S16) || Ty == LLT::fixed_vector(8, S16))
6804 Opc = AArch64::ST3i16;
6805 else if (Ty == LLT::fixed_vector(2, S32) || Ty == LLT::fixed_vector(4, S32))
6806 Opc = AArch64::ST3i32;
6807 else if (Ty == LLT::fixed_vector(2, S64) ||
6808 Ty == LLT::fixed_vector(2, P0) || Ty == S64 || Ty == P0)
6809 Opc = AArch64::ST3i64;
6810 else
6811 llvm_unreachable("Unexpected type for st3lane!");
6812 if (!selectVectorStoreLaneIntrinsic(I, 3, Opc))
6813 return false;
6814 break;
6815 }
6816 case Intrinsic::aarch64_neon_st4lane: {
6817 LLT Ty = MRI.getType(I.getOperand(1).getReg());
6818 unsigned Opc;
6819 if (Ty == LLT::fixed_vector(8, S8) || Ty == LLT::fixed_vector(16, S8))
6820 Opc = AArch64::ST4i8;
6821 else if (Ty == LLT::fixed_vector(4, S16) || Ty == LLT::fixed_vector(8, S16))
6822 Opc = AArch64::ST4i16;
6823 else if (Ty == LLT::fixed_vector(2, S32) || Ty == LLT::fixed_vector(4, S32))
6824 Opc = AArch64::ST4i32;
6825 else if (Ty == LLT::fixed_vector(2, S64) ||
6826 Ty == LLT::fixed_vector(2, P0) || Ty == S64 || Ty == P0)
6827 Opc = AArch64::ST4i64;
6828 else
6829 llvm_unreachable("Unexpected type for st4lane!");
6830 if (!selectVectorStoreLaneIntrinsic(I, 4, Opc))
6831 return false;
6832 break;
6833 }
6834 case Intrinsic::aarch64_mops_memset_tag: {
6835 // Transform
6836 // %dst:gpr(p0) = \
6837 // G_INTRINSIC_W_SIDE_EFFECTS intrinsic(@llvm.aarch64.mops.memset.tag),
6838 // \ %dst:gpr(p0), %val:gpr(s64), %n:gpr(s64)
6839 // where %dst is updated, into
6840 // %Rd:GPR64common, %Rn:GPR64) = \
6841 // MOPSMemorySetTaggingPseudo \
6842 // %Rd:GPR64common, %Rn:GPR64, %Rm:GPR64
6843 // where Rd and Rn are tied.
6844 // It is expected that %val has been extended to s64 in legalization.
6845 // Note that the order of the size/value operands are swapped.
6846
6847 Register DstDef = I.getOperand(0).getReg();
6848 // I.getOperand(1) is the intrinsic function
6849 Register DstUse = I.getOperand(2).getReg();
6850 Register ValUse = I.getOperand(3).getReg();
6851 Register SizeUse = I.getOperand(4).getReg();
6852
6853 // MOPSMemorySetTaggingPseudo has two defs; the intrinsic call has only one.
6854 // Therefore an additional virtual register is required for the updated size
6855 // operand. This value is not accessible via the semantics of the intrinsic.
6857
6858 auto Memset = MIB.buildInstr(AArch64::MOPSMemorySetTaggingPseudo,
6859 {DstDef, SizeDef}, {DstUse, SizeUse, ValUse});
6860 Memset.cloneMemRefs(I);
6861 Memset.setOperandDead(5); // implicit-def $nzcv
6863 break;
6864 }
6865 case Intrinsic::ptrauth_resign_load_relative: {
6866 Register DstReg = I.getOperand(0).getReg();
6867 Register ValReg = I.getOperand(2).getReg();
6868 uint64_t AUTKey = I.getOperand(3).getImm();
6869 Register AUTDisc = I.getOperand(4).getReg();
6870 uint64_t PACKey = I.getOperand(5).getImm();
6871 Register PACDisc = I.getOperand(6).getReg();
6872 int64_t Addend = I.getOperand(7).getImm();
6873
6874 Register AUTAddrDisc = AUTDisc;
6875 uint16_t AUTConstDiscC = 0;
6876 std::tie(AUTConstDiscC, AUTAddrDisc) =
6878
6879 Register PACAddrDisc = PACDisc;
6880 uint16_t PACConstDiscC = 0;
6881 std::tie(PACConstDiscC, PACAddrDisc) =
6883
6884 MIB.buildCopy({AArch64::X16}, {ValReg});
6885
6886 MIB.buildInstr(AArch64::AUTRELLOADPAC)
6887 .addImm(AUTKey)
6888 .addImm(AUTConstDiscC)
6889 .addUse(AUTAddrDisc)
6890 .addImm(PACKey)
6891 .addImm(PACConstDiscC)
6892 .addUse(PACAddrDisc)
6893 .addImm(Addend)
6894 .constrainAllUses(TII, TRI, RBI);
6895 MIB.buildCopy({DstReg}, Register(AArch64::X16));
6896
6897 RBI.constrainGenericRegister(DstReg, AArch64::GPR64RegClass, MRI);
6898 I.eraseFromParent();
6899 return true;
6900 }
6901 }
6902
6903 I.eraseFromParent();
6904 return true;
6905}
6906
6907bool AArch64InstructionSelector::selectIntrinsic(MachineInstr &I,
6908 MachineRegisterInfo &MRI) {
6909 unsigned IntrinID = cast<GIntrinsic>(I).getIntrinsicID();
6910
6911 switch (IntrinID) {
6912 default:
6913 break;
6914 case Intrinsic::ptrauth_resign: {
6915 Register DstReg = I.getOperand(0).getReg();
6916 Register ValReg = I.getOperand(2).getReg();
6917 uint64_t AUTKey = I.getOperand(3).getImm();
6918 Register AUTDisc = I.getOperand(4).getReg();
6919 uint64_t PACKey = I.getOperand(5).getImm();
6920 Register PACDisc = I.getOperand(6).getReg();
6921
6922 Register AUTAddrDisc = AUTDisc;
6923 uint16_t AUTConstDiscC = 0;
6924 std::tie(AUTConstDiscC, AUTAddrDisc) =
6926
6927 Register PACAddrDisc = PACDisc;
6928 uint16_t PACConstDiscC = 0;
6929 std::tie(PACConstDiscC, PACAddrDisc) =
6931
6932 MIB.buildCopy({AArch64::X16}, {ValReg});
6933 MIB.buildInstr(TargetOpcode::IMPLICIT_DEF, {AArch64::X17}, {});
6934 MIB.buildInstr(AArch64::AUTPAC)
6935 .addImm(AUTKey)
6936 .addImm(AUTConstDiscC)
6937 .addUse(AUTAddrDisc)
6938 .addImm(PACKey)
6939 .addImm(PACConstDiscC)
6940 .addUse(PACAddrDisc)
6941 .constrainAllUses(TII, TRI, RBI);
6942 MIB.buildCopy({DstReg}, Register(AArch64::X16));
6943
6944 RBI.constrainGenericRegister(DstReg, AArch64::GPR64RegClass, MRI);
6945 I.eraseFromParent();
6946 return true;
6947 }
6948 case Intrinsic::ptrauth_auth_with_pc_and_resign: {
6949 Register DstReg = I.getOperand(0).getReg();
6950 Register ValReg = I.getOperand(2).getReg();
6951 uint64_t AUTKey = I.getOperand(3).getImm();
6952 Register AUTDisc = I.getOperand(4).getReg();
6953 Register AUTPC = I.getOperand(5).getReg();
6954 uint64_t PACKey = I.getOperand(6).getImm();
6955 Register PACDisc = I.getOperand(7).getReg();
6956
6957 assert((AUTKey == AArch64PACKey::IA || AUTKey == AArch64PACKey::IB) &&
6958 "auth_with_pc_and_resign only supports IA and IB keys");
6959
6960 uint16_t PACConstDiscC = 0;
6961 Register PACAddrDisc;
6962 std::tie(PACConstDiscC, PACAddrDisc) =
6964
6965 if (!PACAddrDisc.isValid())
6966 PACAddrDisc = AArch64::XZR;
6967
6968 MIB.buildCopy({AArch64::X17}, {ValReg});
6969 MIB.buildCopy({AArch64::X16}, {AUTDisc});
6970 MIB.buildCopy({AArch64::X15}, {AUTPC});
6971
6972 MIB.buildInstr(AArch64::AUTPCPAC)
6973 .addImm(AUTKey)
6974 .addImm(PACKey)
6975 .addImm(PACConstDiscC)
6976 .addUse(PACAddrDisc)
6977 .constrainAllUses(TII, TRI, RBI);
6978
6979 MIB.buildCopy({DstReg}, Register(AArch64::X17));
6980 RBI.constrainGenericRegister(DstReg, AArch64::GPR64RegClass, MRI);
6981 I.eraseFromParent();
6982 return true;
6983 }
6984 case Intrinsic::ptrauth_auth: {
6985 Register DstReg = I.getOperand(0).getReg();
6986 Register ValReg = I.getOperand(2).getReg();
6987 uint64_t AUTKey = I.getOperand(3).getImm();
6988 Register AUTDisc = I.getOperand(4).getReg();
6989
6990 Register AUTAddrDisc = AUTDisc;
6991 uint16_t AUTConstDiscC = 0;
6992 std::tie(AUTConstDiscC, AUTAddrDisc) =
6994
6995 if (STI.isX16X17Safer()) {
6996 MIB.buildCopy({AArch64::X16}, {ValReg});
6997 MIB.buildInstr(TargetOpcode::IMPLICIT_DEF, {AArch64::X17}, {});
6998 MIB.buildInstr(AArch64::AUTx16x17)
6999 .addImm(AUTKey)
7000 .addImm(AUTConstDiscC)
7001 .addUse(AUTAddrDisc)
7002 .constrainAllUses(TII, TRI, RBI);
7003 MIB.buildCopy({DstReg}, Register(AArch64::X16));
7004 } else {
7005 Register ScratchReg =
7006 MRI.createVirtualRegister(&AArch64::GPR64commonRegClass);
7007 MIB.buildInstr(AArch64::AUTxMxN)
7008 .addDef(DstReg)
7009 .addDef(ScratchReg)
7010 .addUse(ValReg)
7011 .addImm(AUTKey)
7012 .addImm(AUTConstDiscC)
7013 .addUse(AUTAddrDisc)
7014 .constrainAllUses(TII, TRI, RBI);
7015 }
7016
7017 RBI.constrainGenericRegister(DstReg, AArch64::GPR64RegClass, MRI);
7018 I.eraseFromParent();
7019 return true;
7020 }
7021 case Intrinsic::frameaddress:
7022 case Intrinsic::returnaddress: {
7023 MachineFunction &MF = *I.getParent()->getParent();
7024 MachineFrameInfo &MFI = MF.getFrameInfo();
7025
7026 unsigned Depth = I.getOperand(2).getImm();
7027 Register DstReg = I.getOperand(0).getReg();
7028 RBI.constrainGenericRegister(DstReg, AArch64::GPR64RegClass, MRI);
7029
7030 if (Depth == 0 && IntrinID == Intrinsic::returnaddress) {
7031 if (!MFReturnAddr) {
7032 // Insert the copy from LR/X30 into the entry block, before it can be
7033 // clobbered by anything.
7034 MFI.setReturnAddressIsTaken(true);
7035 MFReturnAddr = getFunctionLiveInPhysReg(
7036 MF, TII, AArch64::LR, AArch64::GPR64RegClass, I.getDebugLoc());
7037 }
7038
7039 if (STI.hasPAuth()) {
7040 MIB.buildInstr(AArch64::XPACI, {DstReg}, {MFReturnAddr});
7041 } else {
7042 MIB.buildCopy({Register(AArch64::LR)}, {MFReturnAddr});
7043 MIB.buildInstr(AArch64::XPACLRI);
7044 MIB.buildCopy({DstReg}, {Register(AArch64::LR)});
7045 }
7046
7047 I.eraseFromParent();
7048 return true;
7049 }
7050
7051 MFI.setFrameAddressIsTaken(true);
7052 Register FrameAddr(AArch64::FP);
7053 while (Depth--) {
7054 Register NextFrame = MRI.createVirtualRegister(&AArch64::GPR64spRegClass);
7055 auto Ldr =
7056 MIB.buildInstr(AArch64::LDRXui, {NextFrame}, {FrameAddr}).addImm(0);
7058 FrameAddr = NextFrame;
7059 }
7060
7061 if (IntrinID == Intrinsic::frameaddress)
7062 MIB.buildCopy({DstReg}, {FrameAddr});
7063 else {
7064 MFI.setReturnAddressIsTaken(true);
7065
7066 if (STI.hasPAuth()) {
7067 Register TmpReg = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
7068 MIB.buildInstr(AArch64::LDRXui, {TmpReg}, {FrameAddr}).addImm(1);
7069 MIB.buildInstr(AArch64::XPACI, {DstReg}, {TmpReg});
7070 } else {
7071 MIB.buildInstr(AArch64::LDRXui, {Register(AArch64::LR)}, {FrameAddr})
7072 .addImm(1);
7073 MIB.buildInstr(AArch64::XPACLRI);
7074 MIB.buildCopy({DstReg}, {Register(AArch64::LR)});
7075 }
7076 }
7077
7078 I.eraseFromParent();
7079 return true;
7080 }
7081 case Intrinsic::aarch64_neon_tbl2:
7082 SelectTable(I, MRI, 2, AArch64::TBLv8i8Two, AArch64::TBLv16i8Two, false);
7083 return true;
7084 case Intrinsic::aarch64_neon_tbl3:
7085 SelectTable(I, MRI, 3, AArch64::TBLv8i8Three, AArch64::TBLv16i8Three,
7086 false);
7087 return true;
7088 case Intrinsic::aarch64_neon_tbl4:
7089 SelectTable(I, MRI, 4, AArch64::TBLv8i8Four, AArch64::TBLv16i8Four, false);
7090 return true;
7091 case Intrinsic::aarch64_neon_tbx2:
7092 SelectTable(I, MRI, 2, AArch64::TBXv8i8Two, AArch64::TBXv16i8Two, true);
7093 return true;
7094 case Intrinsic::aarch64_neon_tbx3:
7095 SelectTable(I, MRI, 3, AArch64::TBXv8i8Three, AArch64::TBXv16i8Three, true);
7096 return true;
7097 case Intrinsic::aarch64_neon_tbx4:
7098 SelectTable(I, MRI, 4, AArch64::TBXv8i8Four, AArch64::TBXv16i8Four, true);
7099 return true;
7100 case Intrinsic::swift_async_context_addr:
7101 auto Sub = MIB.buildInstr(AArch64::SUBXri, {I.getOperand(0).getReg()},
7102 {Register(AArch64::FP)})
7103 .addImm(8)
7104 .addImm(0);
7106
7108 MF->getInfo<AArch64FunctionInfo>()->setHasSwiftAsyncContext(true);
7109 I.eraseFromParent();
7110 return true;
7111 }
7112 return false;
7113}
7114
7115// G_PTRAUTH_GLOBAL_VALUE lowering
7116//
7117// We have 3 lowering alternatives to choose from:
7118// - MOVaddrPAC: similar to MOVaddr, with added PAC.
7119// If the GV doesn't need a GOT load (i.e., is locally defined)
7120// materialize the pointer using adrp+add+pac. See LowerMOVaddrPAC.
7121//
7122// - LOADgotPAC: similar to LOADgot, with added PAC.
7123// If the GV needs a GOT load, materialize the pointer using the usual
7124// GOT adrp+ldr, +pac. Pointers in GOT are assumed to be not signed, the GOT
7125// section is assumed to be read-only (for example, via relro mechanism). See
7126// LowerMOVaddrPAC.
7127//
7128// - LOADauthptrstatic: similar to LOADgot, but use a
7129// special stub slot instead of a GOT slot.
7130// Load a signed pointer for symbol 'sym' from a stub slot named
7131// 'sym$auth_ptr$key$disc' filled by dynamic linker during relocation
7132// resolving. This usually lowers to adrp+ldr, but also emits an entry into
7133// .data with an
7134// @AUTH relocation. See LowerLOADauthptrstatic.
7135//
7136// All 3 are pseudos that are expand late to longer sequences: this lets us
7137// provide integrity guarantees on the to-be-signed intermediate values.
7138//
7139// LOADauthptrstatic is undesirable because it requires a large section filled
7140// with often similarly-signed pointers, making it a good harvesting target.
7141// Thus, it's only used for ptrauth references to extern_weak to avoid null
7142// checks.
7143
7144bool AArch64InstructionSelector::selectPtrAuthGlobalValue(
7145 MachineInstr &I, MachineRegisterInfo &MRI) const {
7146 Register DefReg = I.getOperand(0).getReg();
7147 Register Addr = I.getOperand(1).getReg();
7148 uint64_t Key = I.getOperand(2).getImm();
7149 Register AddrDisc = I.getOperand(3).getReg();
7150 uint64_t Disc = I.getOperand(4).getImm();
7151 int64_t Offset = 0;
7152
7154 report_fatal_error("key in ptrauth global out of range [0, " +
7155 Twine((int)AArch64PACKey::LAST) + "]");
7156
7157 // Blend only works if the integer discriminator is 16-bit wide.
7158 if (!isUInt<16>(Disc))
7160 "constant discriminator in ptrauth global out of range [0, 0xffff]");
7161
7162 // Choosing between 3 lowering alternatives is target-specific.
7163 if (!STI.isTargetELF() && !STI.isTargetMachO())
7164 report_fatal_error("ptrauth global lowering only supported on MachO/ELF");
7165
7166 if (!MRI.hasOneDef(Addr))
7167 return false;
7168
7169 // First match any offset we take from the real global.
7170 const MachineInstr *DefMI = &*MRI.def_instr_begin(Addr);
7171 if (DefMI->getOpcode() == TargetOpcode::G_PTR_ADD) {
7172 Register OffsetReg = DefMI->getOperand(2).getReg();
7173 if (!MRI.hasOneDef(OffsetReg))
7174 return false;
7175 const MachineInstr &OffsetMI = *MRI.def_instr_begin(OffsetReg);
7176 if (OffsetMI.getOpcode() != TargetOpcode::G_CONSTANT)
7177 return false;
7178
7179 Addr = DefMI->getOperand(1).getReg();
7180 if (!MRI.hasOneDef(Addr))
7181 return false;
7182
7183 DefMI = &*MRI.def_instr_begin(Addr);
7184 Offset = OffsetMI.getOperand(1).getCImm()->getSExtValue();
7185 }
7186
7187 // We should be left with a genuine unauthenticated GlobalValue.
7188 const GlobalValue *GV;
7189 if (DefMI->getOpcode() == TargetOpcode::G_GLOBAL_VALUE) {
7190 GV = DefMI->getOperand(1).getGlobal();
7192 } else if (DefMI->getOpcode() == AArch64::G_ADD_LOW) {
7193 GV = DefMI->getOperand(2).getGlobal();
7195 } else {
7196 return false;
7197 }
7198
7199 MachineIRBuilder MIB(I);
7200
7201 // Classify the reference to determine whether it needs a GOT load.
7202 unsigned OpFlags = STI.ClassifyGlobalReference(GV, TM);
7203 const bool NeedsGOTLoad = ((OpFlags & AArch64II::MO_GOT) != 0);
7204 assert(((OpFlags & (~AArch64II::MO_GOT)) == 0) &&
7205 "unsupported non-GOT op flags on ptrauth global reference");
7206 assert((!GV->hasExternalWeakLinkage() || NeedsGOTLoad) &&
7207 "unsupported non-GOT reference to weak ptrauth global");
7208
7209 std::optional<APInt> AddrDiscVal = getIConstantVRegVal(AddrDisc, MRI);
7210 bool HasAddrDisc = !AddrDiscVal || *AddrDiscVal != 0;
7211
7212 // Non-extern_weak:
7213 // - No GOT load needed -> MOVaddrPAC
7214 // - GOT load for non-extern_weak -> LOADgotPAC
7215 // Note that we disallow extern_weak refs to avoid null checks later.
7216 if (!GV->hasExternalWeakLinkage()) {
7217 MIB.buildInstr(TargetOpcode::IMPLICIT_DEF, {AArch64::X16}, {});
7218 MIB.buildInstr(TargetOpcode::IMPLICIT_DEF, {AArch64::X17}, {});
7219 MIB.buildInstr(NeedsGOTLoad ? AArch64::LOADgotPAC : AArch64::MOVaddrPAC)
7221 .addImm(Key)
7222 .addReg(HasAddrDisc ? AddrDisc : AArch64::XZR)
7223 .addImm(Disc)
7224 .constrainAllUses(TII, TRI, RBI);
7225 MIB.buildCopy(DefReg, Register(AArch64::X16));
7226 RBI.constrainGenericRegister(DefReg, AArch64::GPR64RegClass, MRI);
7227 I.eraseFromParent();
7228 return true;
7229 }
7230
7231 // extern_weak -> LOADauthptrstatic
7232
7233 // Offsets and extern_weak don't mix well: ptrauth aside, you'd get the
7234 // offset alone as a pointer if the symbol wasn't available, which would
7235 // probably break null checks in users. Ptrauth complicates things further:
7236 // error out.
7237 if (Offset != 0)
7239 "unsupported non-zero offset in weak ptrauth global reference");
7240
7241 if (HasAddrDisc)
7242 report_fatal_error("unsupported weak addr-div ptrauth global");
7243
7244 MIB.buildInstr(AArch64::LOADauthptrstatic, {DefReg}, {})
7245 .addGlobalAddress(GV, Offset)
7246 .addImm(Key)
7247 .addImm(Disc);
7248 RBI.constrainGenericRegister(DefReg, AArch64::GPR64RegClass, MRI);
7249
7250 I.eraseFromParent();
7251 return true;
7252}
7253
7254void AArch64InstructionSelector::SelectTable(MachineInstr &I,
7255 MachineRegisterInfo &MRI,
7256 unsigned NumVec, unsigned Opc1,
7257 unsigned Opc2, bool isExt) {
7258 Register DstReg = I.getOperand(0).getReg();
7259 unsigned Opc = MRI.getType(DstReg) == LLT::fixed_vector(8, 8) ? Opc1 : Opc2;
7260
7261 // Create the REG_SEQUENCE
7263 for (unsigned i = 0; i < NumVec; i++)
7264 Regs.push_back(I.getOperand(i + 2 + isExt).getReg());
7265 Register RegSeq = createQTuple(Regs, MIB);
7266
7267 Register IdxReg = I.getOperand(2 + NumVec + isExt).getReg();
7268 MachineInstrBuilder Instr;
7269 if (isExt) {
7270 Register Reg = I.getOperand(2).getReg();
7271 Instr = MIB.buildInstr(Opc, {DstReg}, {Reg, RegSeq, IdxReg});
7272 } else
7273 Instr = MIB.buildInstr(Opc, {DstReg}, {RegSeq, IdxReg});
7275 I.eraseFromParent();
7276}
7277
7278InstructionSelector::ComplexRendererFns
7279AArch64InstructionSelector::selectShiftA_32(const MachineOperand &Root) const {
7280 auto MaybeImmed = getImmedFromMO(Root);
7281 if (MaybeImmed == std::nullopt || *MaybeImmed > 31)
7282 return std::nullopt;
7283 uint64_t Enc = (32 - *MaybeImmed) & 0x1f;
7284 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Enc); }}};
7285}
7286
7287InstructionSelector::ComplexRendererFns
7288AArch64InstructionSelector::selectShiftB_32(const MachineOperand &Root) const {
7289 auto MaybeImmed = getImmedFromMO(Root);
7290 if (MaybeImmed == std::nullopt || *MaybeImmed > 31)
7291 return std::nullopt;
7292 uint64_t Enc = 31 - *MaybeImmed;
7293 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Enc); }}};
7294}
7295
7296InstructionSelector::ComplexRendererFns
7297AArch64InstructionSelector::selectShiftA_64(const MachineOperand &Root) const {
7298 auto MaybeImmed = getImmedFromMO(Root);
7299 if (MaybeImmed == std::nullopt || *MaybeImmed > 63)
7300 return std::nullopt;
7301 uint64_t Enc = (64 - *MaybeImmed) & 0x3f;
7302 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Enc); }}};
7303}
7304
7305InstructionSelector::ComplexRendererFns
7306AArch64InstructionSelector::selectShiftB_64(const MachineOperand &Root) const {
7307 auto MaybeImmed = getImmedFromMO(Root);
7308 if (MaybeImmed == std::nullopt || *MaybeImmed > 63)
7309 return std::nullopt;
7310 uint64_t Enc = 63 - *MaybeImmed;
7311 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(Enc); }}};
7312}
7313
7314template <unsigned ShiftWidth>
7315InstructionSelector::ComplexRendererFns
7316AArch64InstructionSelector::selectShiftMask(MachineOperand &Root) const {
7317 if (!Root.isReg())
7318 return std::nullopt;
7319
7320 MachineRegisterInfo &MRI =
7321 Root.getParent()->getParent()->getParent()->getRegInfo();
7322
7323 Register ShAmtReg = Root.getReg();
7324
7325 // Peek through zext for i32 shifts only. For i64 shifts the zext case
7326 // is already handled by existing patterns in the Shift multiclass.
7327 if (ShiftWidth == 32) {
7328 Register ZExtSrcReg;
7329 if (mi_match(ShAmtReg, MRI, m_GZExt(m_Reg(ZExtSrcReg))))
7330 ShAmtReg = ZExtSrcReg;
7331 }
7332
7333 // Remove AND if the mask covers at least the low log2(ShiftWidth) bits.
7334 APInt AndMask;
7335 Register AndSrcReg;
7336 if (mi_match(ShAmtReg, MRI, m_GAnd(m_Reg(AndSrcReg), m_ICst(AndMask))) &&
7337 MRI.getType(ShAmtReg).getSizeInBits() == ShiftWidth) {
7338 if (AndMask.countr_one() >= Log2_32(ShiftWidth))
7339 ShAmtReg = AndSrcReg;
7340 }
7341
7342 // Only succeed if we changed the shift amount; otherwise let other
7343 // patterns (e.g. zext GPR32 -> SUBREG_TO_REG) match instead.
7344 if (ShAmtReg == Root.getReg())
7345 return std::nullopt;
7346
7347 return {{[=](MachineInstrBuilder &MIB) { MIB.addReg(ShAmtReg); }}};
7348}
7349
7350/// Helper to select an immediate value that can be represented as a 12-bit
7351/// value shifted left by either 0 or 12. If it is possible to do so, return
7352/// the immediate and shift value. If not, return std::nullopt.
7353///
7354/// Used by selectArithImmed and selectNegArithImmed.
7355InstructionSelector::ComplexRendererFns
7356AArch64InstructionSelector::select12BitValueWithLeftShift(
7357 uint64_t Immed) const {
7358 unsigned ShiftAmt;
7359 if (Immed >> 12 == 0) {
7360 ShiftAmt = 0;
7361 } else if ((Immed & 0xfff) == 0 && Immed >> 24 == 0) {
7362 ShiftAmt = 12;
7363 Immed = Immed >> 12;
7364 } else
7365 return std::nullopt;
7366
7367 unsigned ShVal = AArch64_AM::getShifterImm(AArch64_AM::LSL, ShiftAmt);
7368 return {{
7369 [=](MachineInstrBuilder &MIB) { MIB.addImm(Immed); },
7370 [=](MachineInstrBuilder &MIB) { MIB.addImm(ShVal); },
7371 }};
7372}
7373
7374/// SelectArithImmed - Select an immediate value that can be represented as
7375/// a 12-bit value shifted left by either 0 or 12. If so, return true with
7376/// Val set to the 12-bit value and Shift set to the shifter operand.
7377InstructionSelector::ComplexRendererFns
7378AArch64InstructionSelector::selectArithImmed(MachineOperand &Root) const {
7379 // This function is called from the addsub_shifted_imm ComplexPattern,
7380 // which lists [imm] as the list of opcode it's interested in, however
7381 // we still need to check whether the operand is actually an immediate
7382 // here because the ComplexPattern opcode list is only used in
7383 // root-level opcode matching.
7384 auto MaybeImmed = getImmedFromMO(Root);
7385 if (MaybeImmed == std::nullopt)
7386 return std::nullopt;
7387 return select12BitValueWithLeftShift(*MaybeImmed);
7388}
7389
7390/// SelectNegArithImmed - As above, but negates the value before trying to
7391/// select it.
7392InstructionSelector::ComplexRendererFns
7393AArch64InstructionSelector::selectNegArithImmed(MachineOperand &Root) const {
7394 // We need a register here, because we need to know if we have a 64 or 32
7395 // bit immediate.
7396 if (!Root.isReg())
7397 return std::nullopt;
7398 auto MaybeImmed = getImmedFromMO(Root);
7399 if (MaybeImmed == std::nullopt)
7400 return std::nullopt;
7401 uint64_t Immed = *MaybeImmed;
7402
7403 // This negation is almost always valid, but "cmp wN, #0" and "cmn wN, #0"
7404 // have the opposite effect on the C flag, so this pattern mustn't match under
7405 // those circumstances.
7406 if (Immed == 0)
7407 return std::nullopt;
7408
7409 // Check if we're dealing with a 32-bit type on the root or a 64-bit type on
7410 // the root.
7411 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7412 if (MRI.getType(Root.getReg()).getSizeInBits() == 32)
7413 Immed = ~((uint32_t)Immed) + 1;
7414 else
7415 Immed = ~Immed + 1ULL;
7416
7417 if (Immed & 0xFFFFFFFFFF000000ULL)
7418 return std::nullopt;
7419
7420 Immed &= 0xFFFFFFULL;
7421 return select12BitValueWithLeftShift(Immed);
7422}
7423
7424/// Checks if we are sure that folding MI into load/store addressing mode is
7425/// beneficial or not.
7426///
7427/// Returns:
7428/// - true if folding MI would be beneficial.
7429/// - false if folding MI would be bad.
7430/// - std::nullopt if it is not sure whether folding MI is beneficial.
7431///
7432/// \p MI can be the offset operand of G_PTR_ADD, e.g. G_SHL in the example:
7433///
7434/// %13:gpr(s64) = G_CONSTANT i64 1
7435/// %8:gpr(s64) = G_SHL %6, %13(s64)
7436/// %9:gpr(p0) = G_PTR_ADD %0, %8(s64)
7437/// %12:gpr(s32) = G_LOAD %9(p0) :: (load (s16))
7438std::optional<bool> AArch64InstructionSelector::isWorthFoldingIntoAddrMode(
7439 const MachineInstr &MI, const MachineRegisterInfo &MRI) const {
7440 if (MI.getOpcode() == AArch64::G_SHL) {
7441 // Address operands with shifts are free, except for running on subtargets
7442 // with AddrLSLSlow14.
7443 if (const auto ValAndVeg = getIConstantVRegValWithLookThrough(
7444 MI.getOperand(2).getReg(), MRI)) {
7445 const APInt ShiftVal = ValAndVeg->Value;
7446
7447 // Don't fold if we know this will be slow.
7448 return !(STI.hasAddrLSLSlow14() && (ShiftVal == 1 || ShiftVal == 4));
7449 }
7450 }
7451 return std::nullopt;
7452}
7453
7454/// Return true if it is worth folding MI into an extended register. That is,
7455/// if it's safe to pull it into the addressing mode of a load or store as a
7456/// shift.
7457/// \p IsAddrOperand whether the def of MI is used as an address operand
7458/// (e.g. feeding into an LDR/STR).
7459bool AArch64InstructionSelector::isWorthFoldingIntoExtendedReg(
7460 const MachineInstr &MI, const MachineRegisterInfo &MRI,
7461 bool IsAddrOperand) const {
7462
7463 // Always fold if there is one use, or if we're optimizing for size.
7464 Register DefReg = MI.getOperand(0).getReg();
7465 if (MRI.hasOneNonDBGUse(DefReg) ||
7466 MI.getParent()->getParent()->getFunction().hasOptSize())
7467 return true;
7468
7469 if (IsAddrOperand) {
7470 // If we are already sure that folding MI is good or bad, return the result.
7471 if (const auto Worth = isWorthFoldingIntoAddrMode(MI, MRI))
7472 return *Worth;
7473
7474 // Fold G_PTR_ADD if its offset operand can be folded
7475 if (MI.getOpcode() == AArch64::G_PTR_ADD) {
7476 MachineInstr *OffsetInst =
7477 getDefIgnoringCopies(MI.getOperand(2).getReg(), MRI);
7478
7479 // Note, we already know G_PTR_ADD is used by at least two instructions.
7480 // If we are also sure about whether folding is beneficial or not,
7481 // return the result.
7482 if (const auto Worth = isWorthFoldingIntoAddrMode(*OffsetInst, MRI))
7483 return *Worth;
7484 }
7485 }
7486
7487 // FIXME: Consider checking HasALULSLFast as appropriate.
7488
7489 // We have a fastpath, so folding a shift in and potentially computing it
7490 // many times may be beneficial. Check if this is only used in memory ops.
7491 // If it is, then we should fold.
7492 return all_of(MRI.use_nodbg_instructions(DefReg),
7493 [](MachineInstr &Use) { return Use.mayLoadOrStore(); });
7494}
7495
7496InstructionSelector::ComplexRendererFns
7497AArch64InstructionSelector::selectExtendedSHL(
7498 MachineOperand &Root, MachineOperand &Base, MachineOperand &Offset,
7499 unsigned SizeInBytes, bool WantsExt) const {
7500 assert(Base.isReg() && "Expected base to be a register operand");
7501 assert(Offset.isReg() && "Expected offset to be a register operand");
7502
7503 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7504 MachineInstr *OffsetInst = MRI.getVRegDef(Offset.getReg());
7505
7506 unsigned OffsetOpc = OffsetInst->getOpcode();
7507 bool LookedThroughZExt = false;
7508 if (OffsetOpc != TargetOpcode::G_SHL && OffsetOpc != TargetOpcode::G_MUL) {
7509 // Try to look through a ZEXT.
7510 if (OffsetOpc != TargetOpcode::G_ZEXT || !WantsExt)
7511 return std::nullopt;
7512
7513 OffsetInst = MRI.getVRegDef(OffsetInst->getOperand(1).getReg());
7514 OffsetOpc = OffsetInst->getOpcode();
7515 LookedThroughZExt = true;
7516
7517 if (OffsetOpc != TargetOpcode::G_SHL && OffsetOpc != TargetOpcode::G_MUL)
7518 return std::nullopt;
7519 }
7520 // Make sure that the memory op is a valid size.
7521 int64_t LegalShiftVal = Log2_32(SizeInBytes);
7522 if (LegalShiftVal == 0)
7523 return std::nullopt;
7524 if (!isWorthFoldingIntoExtendedReg(*OffsetInst, MRI, true))
7525 return std::nullopt;
7526
7527 // Now, try to find the specific G_CONSTANT. Start by assuming that the
7528 // register we will offset is the LHS, and the register containing the
7529 // constant is the RHS.
7530 Register OffsetReg = OffsetInst->getOperand(1).getReg();
7531 Register ConstantReg = OffsetInst->getOperand(2).getReg();
7532 auto ValAndVReg = getIConstantVRegValWithLookThrough(ConstantReg, MRI);
7533 if (!ValAndVReg) {
7534 // We didn't get a constant on the RHS. If the opcode is a shift, then
7535 // we're done.
7536 if (OffsetOpc == TargetOpcode::G_SHL)
7537 return std::nullopt;
7538
7539 // If we have a G_MUL, we can use either register. Try looking at the RHS.
7540 std::swap(OffsetReg, ConstantReg);
7541 ValAndVReg = getIConstantVRegValWithLookThrough(ConstantReg, MRI);
7542 if (!ValAndVReg)
7543 return std::nullopt;
7544 }
7545
7546 // The value must fit into 3 bits, and must be positive. Make sure that is
7547 // true.
7548 int64_t ImmVal = ValAndVReg->Value.getSExtValue();
7549
7550 // Since we're going to pull this into a shift, the constant value must be
7551 // a power of 2. If we got a multiply, then we need to check this.
7552 if (OffsetOpc == TargetOpcode::G_MUL) {
7553 if (!llvm::has_single_bit<uint32_t>(ImmVal))
7554 return std::nullopt;
7555
7556 // Got a power of 2. So, the amount we'll shift is the log base-2 of that.
7557 ImmVal = Log2_32(ImmVal);
7558 }
7559
7560 if ((ImmVal & 0x7) != ImmVal)
7561 return std::nullopt;
7562
7563 // We are only allowed to shift by LegalShiftVal. This shift value is built
7564 // into the instruction, so we can't just use whatever we want.
7565 if (ImmVal != LegalShiftVal)
7566 return std::nullopt;
7567
7568 unsigned SignExtend = 0;
7569 if (WantsExt) {
7570 // Check if the offset is defined by an extend, unless we looked through a
7571 // G_ZEXT earlier.
7572 if (!LookedThroughZExt) {
7573 MachineInstr *ExtInst = getDefIgnoringCopies(OffsetReg, MRI);
7574 auto Ext = getExtendTypeForInst(*ExtInst, MRI, true);
7576 return std::nullopt;
7577
7578 SignExtend = AArch64_AM::isSignExtendShiftType(Ext) ? 1 : 0;
7579 // We only support SXTW for signed extension here.
7580 if (SignExtend && Ext != AArch64_AM::SXTW)
7581 return std::nullopt;
7582 OffsetReg = ExtInst->getOperand(1).getReg();
7583 }
7584
7585 // Need a 32-bit wide register here.
7586 MachineIRBuilder MIB(*MRI.getVRegDef(Root.getReg()));
7587 OffsetReg = moveScalarRegClass(OffsetReg, AArch64::GPR32RegClass, MIB);
7588 }
7589
7590 // We can use the LHS of the GEP as the base, and the LHS of the shift as an
7591 // offset. Signify that we are shifting by setting the shift flag to 1.
7592 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(Base.getReg()); },
7593 [=](MachineInstrBuilder &MIB) { MIB.addUse(OffsetReg); },
7594 [=](MachineInstrBuilder &MIB) {
7595 // Need to add both immediates here to make sure that they are both
7596 // added to the instruction.
7597 MIB.addImm(SignExtend);
7598 MIB.addImm(1);
7599 }}};
7600}
7601
7602/// This is used for computing addresses like this:
7603///
7604/// ldr x1, [x2, x3, lsl #3]
7605///
7606/// Where x2 is the base register, and x3 is an offset register. The shift-left
7607/// is a constant value specific to this load instruction. That is, we'll never
7608/// see anything other than a 3 here (which corresponds to the size of the
7609/// element being loaded.)
7610InstructionSelector::ComplexRendererFns
7611AArch64InstructionSelector::selectAddrModeShiftedExtendXReg(
7612 MachineOperand &Root, unsigned SizeInBytes) const {
7613 if (!Root.isReg())
7614 return std::nullopt;
7615 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7616
7617 // We want to find something like this:
7618 //
7619 // val = G_CONSTANT LegalShiftVal
7620 // shift = G_SHL off_reg val
7621 // ptr = G_PTR_ADD base_reg shift
7622 // x = G_LOAD ptr
7623 //
7624 // And fold it into this addressing mode:
7625 //
7626 // ldr x, [base_reg, off_reg, lsl #LegalShiftVal]
7627
7628 // Check if we can find the G_PTR_ADD.
7629 MachineInstr *PtrAdd =
7630 getOpcodeDef(TargetOpcode::G_PTR_ADD, Root.getReg(), MRI);
7631 if (!PtrAdd || !isWorthFoldingIntoExtendedReg(*PtrAdd, MRI, true))
7632 return std::nullopt;
7633
7634 // Now, try to match an opcode which will match our specific offset.
7635 // We want a G_SHL or a G_MUL.
7636 MachineInstr *OffsetInst =
7637 getDefIgnoringCopies(PtrAdd->getOperand(2).getReg(), MRI);
7638 return selectExtendedSHL(Root, PtrAdd->getOperand(1),
7639 OffsetInst->getOperand(0), SizeInBytes,
7640 /*WantsExt=*/false);
7641}
7642
7643/// This is used for computing addresses like this:
7644///
7645/// ldr x1, [x2, x3]
7646///
7647/// Where x2 is the base register, and x3 is an offset register.
7648///
7649/// When possible (or profitable) to fold a G_PTR_ADD into the address
7650/// calculation, this will do so. Otherwise, it will return std::nullopt.
7651InstructionSelector::ComplexRendererFns
7652AArch64InstructionSelector::selectAddrModeRegisterOffset(
7653 MachineOperand &Root) const {
7654 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7655
7656 // We need a GEP.
7658 if (!mi_match(Root.getReg(), MRI, m_GPtrAdd(m_Reg(Base), m_Reg(Offset))))
7659 return std::nullopt;
7660
7661 // If this is used more than once, let's not bother folding.
7662 // TODO: Check if they are memory ops. If they are, then we can still fold
7663 // without having to recompute anything.
7664 if (!MRI.hasOneNonDBGUse(Root.getReg()))
7665 return std::nullopt;
7666
7667 // Base is the GEP's LHS, offset is its RHS.
7668 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(Base); },
7669 [=](MachineInstrBuilder &MIB) { MIB.addUse(Offset); },
7670 [=](MachineInstrBuilder &MIB) {
7671 // Need to add both immediates here to make sure that they are both
7672 // added to the instruction.
7673 MIB.addImm(0);
7674 MIB.addImm(0);
7675 }}};
7676}
7677
7678/// This is intended to be equivalent to selectAddrModeXRO in
7679/// AArch64ISelDAGtoDAG. It's used for selecting X register offset loads.
7680InstructionSelector::ComplexRendererFns
7681AArch64InstructionSelector::selectAddrModeXRO(MachineOperand &Root,
7682 unsigned SizeInBytes) const {
7683 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7684 if (!Root.isReg())
7685 return std::nullopt;
7686 MachineInstr *PtrAdd =
7687 getOpcodeDef(TargetOpcode::G_PTR_ADD, Root.getReg(), MRI);
7688 if (!PtrAdd)
7689 return std::nullopt;
7690
7691 // Check for an immediates which cannot be encoded in the [base + imm]
7692 // addressing mode, and can't be encoded in an add/sub. If this happens, we'll
7693 // end up with code like:
7694 //
7695 // mov x0, wide
7696 // add x1 base, x0
7697 // ldr x2, [x1, x0]
7698 //
7699 // In this situation, we can use the [base, xreg] addressing mode to save an
7700 // add/sub:
7701 //
7702 // mov x0, wide
7703 // ldr x2, [base, x0]
7704 auto ValAndVReg =
7706 if (ValAndVReg) {
7707 unsigned Scale = Log2_32(SizeInBytes);
7708 int64_t ImmOff = ValAndVReg->Value.getSExtValue();
7709
7710 // Skip immediates that can be selected in the load/store addressing
7711 // mode.
7712 if (ImmOff % SizeInBytes == 0 && ImmOff >= 0 &&
7713 ImmOff < (0x1000 << Scale))
7714 return std::nullopt;
7715
7716 // Helper lambda to decide whether or not it is preferable to emit an add.
7717 auto isPreferredADD = [](int64_t ImmOff) {
7718 // Constants in [0x0, 0xfff] can be encoded in an add.
7719 if ((ImmOff & 0xfffffffffffff000LL) == 0x0LL)
7720 return true;
7721
7722 // Can it be encoded in an add lsl #12?
7723 if ((ImmOff & 0xffffffffff000fffLL) != 0x0LL)
7724 return false;
7725
7726 // It can be encoded in an add lsl #12, but we may not want to. If it is
7727 // possible to select this as a single movz, then prefer that. A single
7728 // movz is faster than an add with a shift.
7729 return (ImmOff & 0xffffffffff00ffffLL) != 0x0LL &&
7730 (ImmOff & 0xffffffffffff0fffLL) != 0x0LL;
7731 };
7732
7733 // If the immediate can be encoded in a single add/sub, then bail out.
7734 if (isPreferredADD(ImmOff) || isPreferredADD(-ImmOff))
7735 return std::nullopt;
7736 }
7737
7738 // Try to fold shifts into the addressing mode.
7739 auto AddrModeFns = selectAddrModeShiftedExtendXReg(Root, SizeInBytes);
7740 if (AddrModeFns)
7741 return AddrModeFns;
7742
7743 // If that doesn't work, see if it's possible to fold in registers from
7744 // a GEP.
7745 return selectAddrModeRegisterOffset(Root);
7746}
7747
7748/// This is used for computing addresses like this:
7749///
7750/// ldr x0, [xBase, wOffset, sxtw #LegalShiftVal]
7751///
7752/// Where we have a 64-bit base register, a 32-bit offset register, and an
7753/// extend (which may or may not be signed).
7754InstructionSelector::ComplexRendererFns
7755AArch64InstructionSelector::selectAddrModeWRO(MachineOperand &Root,
7756 unsigned SizeInBytes) const {
7757 MachineRegisterInfo &MRI = Root.getParent()->getMF()->getRegInfo();
7758
7759 MachineInstr *PtrAdd =
7760 getOpcodeDef(TargetOpcode::G_PTR_ADD, Root.getReg(), MRI);
7761 if (!PtrAdd || !isWorthFoldingIntoExtendedReg(*PtrAdd, MRI, true))
7762 return std::nullopt;
7763
7764 MachineOperand &LHS = PtrAdd->getOperand(1);
7765 MachineOperand &RHS = PtrAdd->getOperand(2);
7766 MachineInstr *OffsetInst = getDefIgnoringCopies(RHS.getReg(), MRI);
7767
7768 // The first case is the same as selectAddrModeXRO, except we need an extend.
7769 // In this case, we try to find a shift and extend, and fold them into the
7770 // addressing mode.
7771 //
7772 // E.g.
7773 //
7774 // off_reg = G_Z/S/ANYEXT ext_reg
7775 // val = G_CONSTANT LegalShiftVal
7776 // shift = G_SHL off_reg val
7777 // ptr = G_PTR_ADD base_reg shift
7778 // x = G_LOAD ptr
7779 //
7780 // In this case we can get a load like this:
7781 //
7782 // ldr x0, [base_reg, ext_reg, sxtw #LegalShiftVal]
7783 auto ExtendedShl = selectExtendedSHL(Root, LHS, OffsetInst->getOperand(0),
7784 SizeInBytes, /*WantsExt=*/true);
7785 if (ExtendedShl)
7786 return ExtendedShl;
7787
7788 // There was no shift. We can try and fold a G_Z/S/ANYEXT in alone though.
7789 //
7790 // e.g.
7791 // ldr something, [base_reg, ext_reg, sxtw]
7792 if (!isWorthFoldingIntoExtendedReg(*OffsetInst, MRI, true))
7793 return std::nullopt;
7794
7795 // Check if this is an extend. We'll get an extend type if it is.
7797 getExtendTypeForInst(*OffsetInst, MRI, /*IsLoadStore=*/true);
7799 return std::nullopt;
7800
7801 // Need a 32-bit wide register.
7802 MachineIRBuilder MIB(*PtrAdd);
7803 Register ExtReg = moveScalarRegClass(OffsetInst->getOperand(1).getReg(),
7804 AArch64::GPR32RegClass, MIB);
7805 unsigned SignExtend = Ext == AArch64_AM::SXTW;
7806
7807 // Base is LHS, offset is ExtReg.
7808 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(LHS.getReg()); },
7809 [=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); },
7810 [=](MachineInstrBuilder &MIB) {
7811 MIB.addImm(SignExtend);
7812 MIB.addImm(0);
7813 }}};
7814}
7815
7816/// Select a "register plus unscaled signed 9-bit immediate" address. This
7817/// should only match when there is an offset that is not valid for a scaled
7818/// immediate addressing mode. The "Size" argument is the size in bytes of the
7819/// memory reference, which is needed here to know what is valid for a scaled
7820/// immediate.
7821InstructionSelector::ComplexRendererFns
7822AArch64InstructionSelector::selectAddrModeUnscaled(MachineOperand &Root,
7823 unsigned Size) const {
7824 MachineRegisterInfo &MRI =
7825 Root.getParent()->getParent()->getParent()->getRegInfo();
7826
7827 if (!Root.isReg())
7828 return std::nullopt;
7829
7830 if (!isBaseWithConstantOffset(Root, MRI))
7831 return std::nullopt;
7832
7833 MachineInstr *RootDef = MRI.getVRegDef(Root.getReg());
7834
7835 MachineOperand &OffImm = RootDef->getOperand(2);
7836 if (!OffImm.isReg())
7837 return std::nullopt;
7838 MachineInstr *RHS = MRI.getVRegDef(OffImm.getReg());
7839 if (RHS->getOpcode() != TargetOpcode::G_CONSTANT)
7840 return std::nullopt;
7841 int64_t RHSC;
7842 MachineOperand &RHSOp1 = RHS->getOperand(1);
7843 if (!RHSOp1.isCImm() || RHSOp1.getCImm()->getBitWidth() > 64)
7844 return std::nullopt;
7845 RHSC = RHSOp1.getCImm()->getSExtValue();
7846
7847 if (RHSC >= -256 && RHSC < 256) {
7848 MachineOperand &Base = RootDef->getOperand(1);
7849 return {{
7850 [=](MachineInstrBuilder &MIB) { MIB.add(Base); },
7851 [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC); },
7852 }};
7853 }
7854 return std::nullopt;
7855}
7856
7857InstructionSelector::ComplexRendererFns
7858AArch64InstructionSelector::tryFoldAddLowIntoImm(MachineInstr &RootDef,
7859 unsigned Size,
7860 MachineRegisterInfo &MRI) const {
7861 if (RootDef.getOpcode() != AArch64::G_ADD_LOW)
7862 return std::nullopt;
7863 MachineInstr &Adrp = *MRI.getVRegDef(RootDef.getOperand(1).getReg());
7864 if (Adrp.getOpcode() != AArch64::ADRP)
7865 return std::nullopt;
7866
7867 // TODO: add heuristics like isWorthFoldingADDlow() from SelectionDAG.
7868 auto Offset = Adrp.getOperand(1).getOffset();
7869 if (Offset % Size != 0)
7870 return std::nullopt;
7871
7872 auto GV = Adrp.getOperand(1).getGlobal();
7873 if (GV->isThreadLocal())
7874 return std::nullopt;
7875
7876 auto &MF = *RootDef.getParent()->getParent();
7877 if (GV->getPointerAlignment(MF.getDataLayout()) < Size)
7878 return std::nullopt;
7879
7880 unsigned OpFlags = STI.ClassifyGlobalReference(GV, MF.getTarget());
7881 MachineIRBuilder MIRBuilder(RootDef);
7882 Register AdrpReg = Adrp.getOperand(0).getReg();
7883 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(AdrpReg); },
7884 [=](MachineInstrBuilder &MIB) {
7885 MIB.addGlobalAddress(GV, Offset,
7886 OpFlags | AArch64II::MO_PAGEOFF |
7888 }}};
7889}
7890
7891/// Select a "register plus scaled unsigned 12-bit immediate" address. The
7892/// "Size" argument is the size in bytes of the memory reference, which
7893/// determines the scale.
7894InstructionSelector::ComplexRendererFns
7895AArch64InstructionSelector::selectAddrModeIndexed(MachineOperand &Root,
7896 unsigned Size) const {
7897 MachineFunction &MF = *Root.getParent()->getParent()->getParent();
7898 MachineRegisterInfo &MRI = MF.getRegInfo();
7899
7900 if (!Root.isReg())
7901 return std::nullopt;
7902
7903 MachineInstr *RootDef = MRI.getVRegDef(Root.getReg());
7904 if (RootDef->getOpcode() == TargetOpcode::G_FRAME_INDEX) {
7905 return {{
7906 [=](MachineInstrBuilder &MIB) { MIB.add(RootDef->getOperand(1)); },
7907 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); },
7908 }};
7909 }
7910
7912 // Check if we can fold in the ADD of small code model ADRP + ADD address.
7913 // HACK: ld64 on Darwin doesn't support relocations on PRFM, so we can't fold
7914 // globals into the offset.
7915 MachineInstr *RootParent = Root.getParent();
7916 if (CM == CodeModel::Small &&
7917 !(RootParent->getOpcode() == AArch64::G_AARCH64_PREFETCH &&
7918 STI.isTargetDarwin())) {
7919 auto OpFns = tryFoldAddLowIntoImm(*RootDef, Size, MRI);
7920 if (OpFns)
7921 return OpFns;
7922 }
7923
7924 if (isBaseWithConstantOffset(Root, MRI)) {
7925 MachineOperand &LHS = RootDef->getOperand(1);
7926 MachineOperand &RHS = RootDef->getOperand(2);
7927 MachineInstr *LHSDef = MRI.getVRegDef(LHS.getReg());
7928 MachineInstr *RHSDef = MRI.getVRegDef(RHS.getReg());
7929
7930 int64_t RHSC = (int64_t)RHSDef->getOperand(1).getCImm()->getZExtValue();
7931 unsigned Scale = Log2_32(Size);
7932 if ((RHSC & (Size - 1)) == 0 && RHSC >= 0 && RHSC < (0x1000 << Scale)) {
7933 if (LHSDef->getOpcode() == TargetOpcode::G_FRAME_INDEX)
7934 return {{
7935 [=](MachineInstrBuilder &MIB) { MIB.add(LHSDef->getOperand(1)); },
7936 [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC >> Scale); },
7937 }};
7938
7939 return {{
7940 [=](MachineInstrBuilder &MIB) { MIB.add(LHS); },
7941 [=](MachineInstrBuilder &MIB) { MIB.addImm(RHSC >> Scale); },
7942 }};
7943 }
7944 }
7945
7946 // Before falling back to our general case, check if the unscaled
7947 // instructions can handle this. If so, that's preferable.
7948 if (selectAddrModeUnscaled(Root, Size))
7949 return std::nullopt;
7950
7951 return {{
7952 [=](MachineInstrBuilder &MIB) { MIB.add(Root); },
7953 [=](MachineInstrBuilder &MIB) { MIB.addImm(0); },
7954 }};
7955}
7956
7957/// Given a shift instruction, return the correct shift type for that
7958/// instruction.
7960 switch (MI.getOpcode()) {
7961 default:
7963 case TargetOpcode::G_SHL:
7964 return AArch64_AM::LSL;
7965 case TargetOpcode::G_LSHR:
7966 return AArch64_AM::LSR;
7967 case TargetOpcode::G_ASHR:
7968 return AArch64_AM::ASR;
7969 case TargetOpcode::G_ROTR:
7970 return AArch64_AM::ROR;
7971 }
7972}
7973
7974/// Select a "shifted register" operand. If the value is not shifted, set the
7975/// shift operand to a default value of "lsl 0".
7976InstructionSelector::ComplexRendererFns
7977AArch64InstructionSelector::selectShiftedRegister(MachineOperand &Root,
7978 bool AllowROR) const {
7979 if (!Root.isReg())
7980 return std::nullopt;
7981 MachineRegisterInfo &MRI =
7982 Root.getParent()->getParent()->getParent()->getRegInfo();
7983
7984 // Check if the operand is defined by an instruction which corresponds to
7985 // a ShiftExtendType. E.g. a G_SHL, G_LSHR, etc.
7986 MachineInstr *ShiftInst = MRI.getVRegDef(Root.getReg());
7988 if (ShType == AArch64_AM::InvalidShiftExtend)
7989 return std::nullopt;
7990 if (ShType == AArch64_AM::ROR && !AllowROR)
7991 return std::nullopt;
7992 if (!isWorthFoldingIntoExtendedReg(*ShiftInst, MRI, false))
7993 return std::nullopt;
7994
7995 // Need an immediate on the RHS.
7996 MachineOperand &ShiftRHS = ShiftInst->getOperand(2);
7997 auto Immed = getImmedFromMO(ShiftRHS);
7998 if (!Immed)
7999 return std::nullopt;
8000
8001 // We have something that we can fold. Fold in the shift's LHS and RHS into
8002 // the instruction.
8003 MachineOperand &ShiftLHS = ShiftInst->getOperand(1);
8004 Register ShiftReg = ShiftLHS.getReg();
8005
8006 unsigned NumBits = MRI.getType(ShiftReg).getSizeInBits();
8007 unsigned Val = *Immed & (NumBits - 1);
8008 unsigned ShiftVal = AArch64_AM::getShifterImm(ShType, Val);
8009
8010 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ShiftReg); },
8011 [=](MachineInstrBuilder &MIB) { MIB.addImm(ShiftVal); }}};
8012}
8013
8014AArch64_AM::ShiftExtendType AArch64InstructionSelector::getExtendTypeForInst(
8015 MachineInstr &MI, MachineRegisterInfo &MRI, bool IsLoadStore) const {
8016 unsigned Opc = MI.getOpcode();
8017
8018 // Handle explicit extend instructions first.
8019 if (Opc == TargetOpcode::G_SEXT || Opc == TargetOpcode::G_SEXT_INREG) {
8020 unsigned Size;
8021 if (Opc == TargetOpcode::G_SEXT)
8022 Size = MRI.getType(MI.getOperand(1).getReg()).getSizeInBits();
8023 else
8024 Size = MI.getOperand(2).getImm();
8025 assert(Size != 64 && "Extend from 64 bits?");
8026 switch (Size) {
8027 case 8:
8028 return IsLoadStore ? AArch64_AM::InvalidShiftExtend : AArch64_AM::SXTB;
8029 case 16:
8030 return IsLoadStore ? AArch64_AM::InvalidShiftExtend : AArch64_AM::SXTH;
8031 case 32:
8032 return AArch64_AM::SXTW;
8033 default:
8035 }
8036 }
8037
8038 if (Opc == TargetOpcode::G_ZEXT || Opc == TargetOpcode::G_ANYEXT) {
8039 unsigned Size = MRI.getType(MI.getOperand(1).getReg()).getSizeInBits();
8040 assert(Size != 64 && "Extend from 64 bits?");
8041 switch (Size) {
8042 case 8:
8043 return IsLoadStore ? AArch64_AM::InvalidShiftExtend : AArch64_AM::UXTB;
8044 case 16:
8045 return IsLoadStore ? AArch64_AM::InvalidShiftExtend : AArch64_AM::UXTH;
8046 case 32:
8047 return AArch64_AM::UXTW;
8048 default:
8050 }
8051 }
8052
8053 // Don't have an explicit extend. Try to handle a G_AND with a constant mask
8054 // on the RHS.
8055 if (Opc != TargetOpcode::G_AND)
8057
8058 std::optional<uint64_t> MaybeAndMask = getImmedFromMO(MI.getOperand(2));
8059 if (!MaybeAndMask)
8061 uint64_t AndMask = *MaybeAndMask;
8062 switch (AndMask) {
8063 default:
8065 case 0xFF:
8066 return !IsLoadStore ? AArch64_AM::UXTB : AArch64_AM::InvalidShiftExtend;
8067 case 0xFFFF:
8068 return !IsLoadStore ? AArch64_AM::UXTH : AArch64_AM::InvalidShiftExtend;
8069 case 0xFFFFFFFF:
8070 return AArch64_AM::UXTW;
8071 }
8072}
8073
8074Register AArch64InstructionSelector::moveScalarRegClass(
8075 Register Reg, const TargetRegisterClass &RC, MachineIRBuilder &MIB) const {
8076 MachineRegisterInfo &MRI = *MIB.getMRI();
8077 auto Ty = MRI.getType(Reg);
8078 assert(!Ty.isVector() && "Expected scalars only!");
8079 if (Ty.getSizeInBits() == TRI.getRegSizeInBits(RC))
8080 return Reg;
8081
8082 // Create a copy and immediately select it.
8083 // FIXME: We should have an emitCopy function?
8084 auto Copy = MIB.buildCopy({&RC}, {Reg});
8085 selectCopy(*Copy, TII, MRI, TRI, RBI);
8086 return Copy.getReg(0);
8087}
8088
8089/// Select an "extended register" operand. This operand folds in an extend
8090/// followed by an optional left shift.
8091InstructionSelector::ComplexRendererFns
8092AArch64InstructionSelector::selectArithExtendedRegister(
8093 MachineOperand &Root) const {
8094 if (!Root.isReg())
8095 return std::nullopt;
8096 MachineRegisterInfo &MRI =
8097 Root.getParent()->getParent()->getParent()->getRegInfo();
8098
8099 uint64_t ShiftVal = 0;
8100 Register ExtReg;
8102 MachineInstr *RootDef = getDefIgnoringCopies(Root.getReg(), MRI);
8103 if (!RootDef)
8104 return std::nullopt;
8105
8106 if (!isWorthFoldingIntoExtendedReg(*RootDef, MRI, false))
8107 return std::nullopt;
8108
8109 // Check if we can fold a shift and an extend.
8110 if (RootDef->getOpcode() == TargetOpcode::G_SHL) {
8111 // Look for a constant on the RHS of the shift.
8112 MachineOperand &RHS = RootDef->getOperand(2);
8113 std::optional<uint64_t> MaybeShiftVal = getImmedFromMO(RHS);
8114 if (!MaybeShiftVal)
8115 return std::nullopt;
8116 ShiftVal = *MaybeShiftVal;
8117 if (ShiftVal > 4)
8118 return std::nullopt;
8119 // Look for a valid extend instruction on the LHS of the shift.
8120 MachineOperand &LHS = RootDef->getOperand(1);
8121 MachineInstr *ExtDef = getDefIgnoringCopies(LHS.getReg(), MRI);
8122 if (!ExtDef)
8123 return std::nullopt;
8124 Ext = getExtendTypeForInst(*ExtDef, MRI);
8126 return std::nullopt;
8127 ExtReg = ExtDef->getOperand(1).getReg();
8128 } else {
8129 // Didn't get a shift. Try just folding an extend.
8130 Ext = getExtendTypeForInst(*RootDef, MRI);
8132 return std::nullopt;
8133 ExtReg = RootDef->getOperand(1).getReg();
8134
8135 // If we have a 32 bit instruction which zeroes out the high half of a
8136 // register, we get an implicit zero extend for free. Check if we have one.
8137 // FIXME: We actually emit the extend right now even though we don't have
8138 // to.
8139 if (Ext == AArch64_AM::UXTW && MRI.getType(ExtReg).getSizeInBits() == 32) {
8140 MachineInstr *ExtInst = MRI.getVRegDef(ExtReg);
8141 if (isDef32(*ExtInst))
8142 return std::nullopt;
8143 }
8144 }
8145
8146 // We require a GPR32 here. Narrow the ExtReg if needed using a subregister
8147 // copy.
8148 MachineIRBuilder MIB(*RootDef);
8149 ExtReg = moveScalarRegClass(ExtReg, AArch64::GPR32RegClass, MIB);
8150
8151 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); },
8152 [=](MachineInstrBuilder &MIB) {
8153 MIB.addImm(getArithExtendImm(Ext, ShiftVal));
8154 }}};
8155}
8156
8157InstructionSelector::ComplexRendererFns
8158AArch64InstructionSelector::selectExtractHigh(MachineOperand &Root) const {
8159 if (!Root.isReg())
8160 return std::nullopt;
8161 MachineRegisterInfo &MRI =
8162 Root.getParent()->getParent()->getParent()->getRegInfo();
8163
8164 auto Extract = getDefSrcRegIgnoringCopies(Root.getReg(), MRI);
8165 while (Extract && Extract->MI->getOpcode() == TargetOpcode::G_BITCAST &&
8166 STI.isLittleEndian())
8167 Extract =
8168 getDefSrcRegIgnoringCopies(Extract->MI->getOperand(1).getReg(), MRI);
8169 if (!Extract)
8170 return std::nullopt;
8171
8172 if (auto *Unmerge = dyn_cast<GUnmerge>(Extract->MI)) {
8173 if (Unmerge->getNumDefs() == 2 &&
8174 Extract->Reg == Unmerge->getOperand(1).getReg()) {
8175 Register ExtReg = Unmerge->getSourceReg();
8176 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); }}};
8177 }
8178 }
8179 if (auto *ExtElt = dyn_cast<GExtractVectorElement>(Extract->MI)) {
8180 LLT SrcTy = MRI.getType(ExtElt->getVectorReg());
8181 auto LaneIdx =
8182 getIConstantVRegValWithLookThrough(ExtElt->getIndexReg(), MRI);
8183 if (LaneIdx && SrcTy == LLT::fixed_vector(2, 64) &&
8184 LaneIdx->Value.getSExtValue() == 1) {
8185 Register ExtReg = ExtElt->getVectorReg();
8186 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); }}};
8187 }
8188 }
8189 if (auto *Subvec = dyn_cast<GExtractSubvector>(Extract->MI)) {
8190 LLT SrcTy = MRI.getType(Subvec->getSrcVec());
8191 auto LaneIdx = Subvec->getIndexImm();
8192 if (LaneIdx == SrcTy.getNumElements() / 2) {
8193 Register ExtReg = Subvec->getSrcVec();
8194 return {{[=](MachineInstrBuilder &MIB) { MIB.addUse(ExtReg); }}};
8195 }
8196 }
8197
8198 return std::nullopt;
8199}
8200
8201InstructionSelector::ComplexRendererFns
8202AArch64InstructionSelector::selectCVTFixedPointBase(const MachineOperand &Root,
8203 unsigned DstElemWidth,
8204 bool isReciprocal) const {
8205 if (!Root.isReg())
8206 return std::nullopt;
8207 const MachineRegisterInfo &MRI =
8208 Root.getParent()->getParent()->getParent()->getRegInfo();
8209
8210 Register Reg = Root.getReg();
8211 MachineInstr *Dup = getDefIgnoringCopies(Reg, MRI);
8212
8213 if (Dup && Dup->getOpcode() == AArch64::G_DUP)
8214 Reg = Dup->getOperand(1).getReg();
8215
8216 std::optional<ValueAndVReg> CstVal =
8218
8219 if (!CstVal)
8220 return std::nullopt;
8221
8222 unsigned CstElemWidth = MRI.getType(Reg).getScalarSizeInBits();
8223 APFloat FVal(0.0);
8224 switch (CstElemWidth) {
8225 case 16:
8226 FVal = APFloat(APFloat::IEEEhalf(), CstVal->Value);
8227 break;
8228 case 32:
8229 FVal = APFloat(APFloat::IEEEsingle(), CstVal->Value);
8230 break;
8231 case 64:
8232 FVal = APFloat(APFloat::IEEEdouble(), CstVal->Value);
8233 break;
8234 default:
8235 return std::nullopt;
8236 };
8237 if (unsigned FBits =
8238 CheckFixedPointOperandConstant(FVal, DstElemWidth, isReciprocal))
8239 return {{[=](MachineInstrBuilder &MIB) { MIB.addImm(FBits); }}};
8240
8241 return std::nullopt;
8242}
8243
8244unsigned AArch64InstructionSelector::getFixedPointWidthFromOperand(
8245 const MachineOperand &Root) const {
8246 return Root.getParent()
8247 ->getMF()
8248 ->getRegInfo()
8249 .getType(Root.getReg())
8251}
8252
8253template <unsigned Width>
8254InstructionSelector::ComplexRendererFns
8255AArch64InstructionSelector::selectCVTFixedPoint(MachineOperand &Root) const {
8256 return selectCVTFixedPointBase(Root, Width, /*isReciprocal*/ false);
8257}
8258
8259template <unsigned Width>
8260InstructionSelector::ComplexRendererFns
8261AArch64InstructionSelector::selectCVTFixedPosRecipOperand(
8262 MachineOperand &Root) const {
8263 return selectCVTFixedPointBase(Root, Width, /*isReciprocal*/ true);
8264}
8265
8266InstructionSelector::ComplexRendererFns
8267AArch64InstructionSelector::selectCVTFixedPointVec(MachineOperand &Root) const {
8268 return selectCVTFixedPointBase(Root, getFixedPointWidthFromOperand(Root),
8269 /*isReciprocal*/ false);
8270}
8271
8272InstructionSelector::ComplexRendererFns
8273AArch64InstructionSelector::selectCVTFixedPosRecipOperandVec(
8274 MachineOperand &Root) const {
8275 return selectCVTFixedPointBase(Root, getFixedPointWidthFromOperand(Root),
8276 /*isReciprocal*/ true);
8277}
8278
8279void AArch64InstructionSelector::renderFixedPointScalarXForm(
8280 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
8281 assert(OpIdx == 3 && MI.getOperand(OpIdx).isImm() &&
8282 "Expected vecshift immediate operand");
8283 MIB.addImm(MI.getOperand(OpIdx).getImm());
8284}
8285
8286void AArch64InstructionSelector::renderFixedPointImm(MachineInstrBuilder &MIB,
8287 const MachineOperand &Root,
8288 unsigned Width,
8289 bool isReciprocal) const {
8290 // FIXME: This is only needed to satisfy the type checking in tablegen, and
8291 // should be able to reuse the Renderers already calculated by
8292 // selectCVTFixedPointBase.
8293 InstructionSelector::ComplexRendererFns Renderer =
8294 selectCVTFixedPointBase(Root, Width, isReciprocal);
8295 assert((Renderer && Renderer->size() == 1) &&
8296 "Expected selectCVTFixedPointBase to provide a function\n");
8297 (Renderer->front())(MIB);
8298}
8299
8300void AArch64InstructionSelector::renderFixedPointXForm(MachineInstrBuilder &MIB,
8301 const MachineInstr &MI,
8302 int OpIdx) const {
8303 const MachineOperand &Root = MI.getOperand(OpIdx);
8304 renderFixedPointImm(MIB, Root, getFixedPointWidthFromOperand(Root),
8305 /*isReciprocal*/ false);
8306}
8307
8308void AArch64InstructionSelector::renderFixedPointRecipXForm(
8309 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
8310 const MachineOperand &Root = MI.getOperand(OpIdx);
8311 renderFixedPointImm(MIB, Root, getFixedPointWidthFromOperand(Root),
8312 /*isReciprocal*/ true);
8313}
8314
8315void AArch64InstructionSelector::renderTruncImm(MachineInstrBuilder &MIB,
8316 const MachineInstr &MI,
8317 int OpIdx) const {
8318 const MachineRegisterInfo &MRI = MI.getParent()->getParent()->getRegInfo();
8319 assert(MI.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8320 "Expected G_CONSTANT");
8321 std::optional<int64_t> CstVal =
8322 getIConstantVRegSExtVal(MI.getOperand(0).getReg(), MRI);
8323 assert(CstVal && "Expected constant value");
8324 MIB.addImm(*CstVal);
8325}
8326
8327void AArch64InstructionSelector::renderLogicalImm32(
8328 MachineInstrBuilder &MIB, const MachineInstr &I, int OpIdx) const {
8329 assert(I.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8330 "Expected G_CONSTANT");
8331 uint64_t CstVal = I.getOperand(1).getCImm()->getZExtValue();
8333 MIB.addImm(Enc);
8334}
8335
8336void AArch64InstructionSelector::renderLogicalImm64(
8337 MachineInstrBuilder &MIB, const MachineInstr &I, int OpIdx) const {
8338 assert(I.getOpcode() == TargetOpcode::G_CONSTANT && OpIdx == -1 &&
8339 "Expected G_CONSTANT");
8340 uint64_t CstVal = I.getOperand(1).getCImm()->getZExtValue();
8342 MIB.addImm(Enc);
8343}
8344
8345void AArch64InstructionSelector::renderUbsanTrap(MachineInstrBuilder &MIB,
8346 const MachineInstr &MI,
8347 int OpIdx) const {
8348 assert(MI.getOpcode() == TargetOpcode::G_UBSANTRAP && OpIdx == 0 &&
8349 "Expected G_UBSANTRAP");
8350 MIB.addImm(MI.getOperand(0).getImm() | ('U' << 8));
8351}
8352
8353void AArch64InstructionSelector::renderFPImm16(MachineInstrBuilder &MIB,
8354 const MachineInstr &MI,
8355 int OpIdx) const {
8356 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8357 "Expected G_FCONSTANT");
8358 MIB.addImm(
8359 AArch64_AM::getFP16Imm(MI.getOperand(1).getFPImm()->getValueAPF()));
8360}
8361
8362void AArch64InstructionSelector::renderFPImm32(MachineInstrBuilder &MIB,
8363 const MachineInstr &MI,
8364 int OpIdx) const {
8365 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8366 "Expected G_FCONSTANT");
8367 MIB.addImm(
8368 AArch64_AM::getFP32Imm(MI.getOperand(1).getFPImm()->getValueAPF()));
8369}
8370
8371void AArch64InstructionSelector::renderFPImm64(MachineInstrBuilder &MIB,
8372 const MachineInstr &MI,
8373 int OpIdx) const {
8374 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8375 "Expected G_FCONSTANT");
8376 MIB.addImm(
8377 AArch64_AM::getFP64Imm(MI.getOperand(1).getFPImm()->getValueAPF()));
8378}
8379
8380void AArch64InstructionSelector::renderFPImm32SIMDModImmType4(
8381 MachineInstrBuilder &MIB, const MachineInstr &MI, int OpIdx) const {
8382 assert(MI.getOpcode() == TargetOpcode::G_FCONSTANT && OpIdx == -1 &&
8383 "Expected G_FCONSTANT");
8385 .getFPImm()
8386 ->getValueAPF()
8387 .bitcastToAPInt()
8388 .getZExtValue()));
8389}
8390
8391bool AArch64InstructionSelector::isLoadStoreOfNumBytes(
8392 const MachineInstr &MI, unsigned NumBytes) const {
8393 if (!MI.mayLoadOrStore())
8394 return false;
8395 assert(MI.hasOneMemOperand() &&
8396 "Expected load/store to have only one mem op!");
8397 return (*MI.memoperands_begin())->getSize() == NumBytes;
8398}
8399
8400bool AArch64InstructionSelector::isDef32(const MachineInstr &MI) const {
8401 const MachineRegisterInfo &MRI = MI.getParent()->getParent()->getRegInfo();
8402 if (MRI.getType(MI.getOperand(0).getReg()).getSizeInBits() != 32)
8403 return false;
8404
8405 // Only return true if we know the operation will zero-out the high half of
8406 // the 64-bit register. Truncates can be subregister copies, which don't
8407 // zero out the high bits. Copies and other copy-like instructions can be
8408 // fed by truncates, or could be lowered as subregister copies.
8409 switch (MI.getOpcode()) {
8410 default:
8411 return true;
8412 case TargetOpcode::COPY:
8413 case TargetOpcode::G_BITCAST:
8414 case TargetOpcode::G_TRUNC:
8415 case TargetOpcode::G_PHI:
8416 return false;
8417 }
8418}
8419
8420
8421// Perform fixups on the given PHI instruction's operands to force them all
8422// to be the same as the destination regbank.
8424 const AArch64RegisterBankInfo &RBI) {
8425 assert(MI.getOpcode() == TargetOpcode::G_PHI && "Expected a G_PHI");
8426 Register DstReg = MI.getOperand(0).getReg();
8427 const RegisterBank *DstRB = MRI.getRegBankOrNull(DstReg);
8428 assert(DstRB && "Expected PHI dst to have regbank assigned");
8429 MachineIRBuilder MIB(MI);
8430
8431 // Go through each operand and ensure it has the same regbank.
8432 for (MachineOperand &MO : llvm::drop_begin(MI.operands())) {
8433 if (!MO.isReg())
8434 continue;
8435 Register OpReg = MO.getReg();
8436 const RegisterBank *RB = MRI.getRegBankOrNull(OpReg);
8437 if (RB != DstRB) {
8438 // Insert a cross-bank copy.
8439 auto *OpDef = MRI.getVRegDef(OpReg);
8440 const LLT &Ty = MRI.getType(OpReg);
8441 MachineBasicBlock &OpDefBB = *OpDef->getParent();
8442
8443 // Any instruction we insert must appear after all PHIs in the block
8444 // for the block to be valid MIR.
8445 MachineBasicBlock::iterator InsertPt = std::next(OpDef->getIterator());
8446 if (InsertPt != OpDefBB.end() && InsertPt->isPHI())
8447 InsertPt = OpDefBB.getFirstNonPHI();
8448 MIB.setInsertPt(*OpDef->getParent(), InsertPt);
8449 auto Copy = MIB.buildCopy(Ty, OpReg);
8450 MRI.setRegBank(Copy.getReg(0), *DstRB);
8451 MO.setReg(Copy.getReg(0));
8452 }
8453 }
8454}
8455
8456void AArch64InstructionSelector::processPHIs(MachineFunction &MF) {
8457 // We're looking for PHIs, build a list so we don't invalidate iterators.
8458 MachineRegisterInfo &MRI = MF.getRegInfo();
8460 for (auto &BB : MF) {
8461 for (auto &MI : BB) {
8462 if (MI.getOpcode() == TargetOpcode::G_PHI)
8463 Phis.emplace_back(&MI);
8464 }
8465 }
8466
8467 for (auto *MI : Phis) {
8468 // We need to do some work here if the operand types are < 16 bit and they
8469 // are split across fpr/gpr banks. Since all types <32b on gpr
8470 // end up being assigned gpr32 regclasses, we can end up with PHIs here
8471 // which try to select between a gpr32 and an fpr16. Ideally RBS shouldn't
8472 // be selecting heterogenous regbanks for operands if possible, but we
8473 // still need to be able to deal with it here.
8474 //
8475 // To fix this, if we have a gpr-bank operand < 32b in size and at least
8476 // one other operand is on the fpr bank, then we add cross-bank copies
8477 // to homogenize the operand banks. For simplicity the bank that we choose
8478 // to settle on is whatever bank the def operand has. For example:
8479 //
8480 // %endbb:
8481 // %dst:gpr(s16) = G_PHI %in1:gpr(s16), %bb1, %in2:fpr(s16), %bb2
8482 // =>
8483 // %bb2:
8484 // ...
8485 // %in2_copy:gpr(s16) = COPY %in2:fpr(s16)
8486 // ...
8487 // %endbb:
8488 // %dst:gpr(s16) = G_PHI %in1:gpr(s16), %bb1, %in2_copy:gpr(s16), %bb2
8489 bool HasGPROp = false, HasFPROp = false;
8490 for (const MachineOperand &MO : llvm::drop_begin(MI->operands())) {
8491 if (!MO.isReg())
8492 continue;
8493 const LLT &Ty = MRI.getType(MO.getReg());
8494 if (!Ty.isValid() || !Ty.isScalar())
8495 break;
8496 if (Ty.getSizeInBits() >= 32)
8497 break;
8498 const RegisterBank *RB = MRI.getRegBankOrNull(MO.getReg());
8499 // If for some reason we don't have a regbank yet. Don't try anything.
8500 if (!RB)
8501 break;
8502
8503 if (RB->getID() == AArch64::GPRRegBankID)
8504 HasGPROp = true;
8505 else
8506 HasFPROp = true;
8507 }
8508 // We have heterogenous regbanks, need to fixup.
8509 if (HasGPROp && HasFPROp)
8510 fixupPHIOpBanks(*MI, MRI, RBI);
8511 }
8512}
8513
8514namespace llvm {
8515InstructionSelector *
8517 const AArch64Subtarget &Subtarget,
8518 const AArch64RegisterBankInfo &RBI) {
8519 return new AArch64InstructionSelector(TM, Subtarget, RBI);
8520}
8521}
#define Success
MachineInstrBuilder MachineInstrBuilder & DefMI
static std::tuple< SDValue, SDValue > extractPtrauthBlendDiscriminators(SDValue Disc, SelectionDAG *DAG)
static bool isPreferredADD(int64_t ImmOff)
static SDValue emitConditionalComparison(SDValue LHS, SDValue RHS, ISD::CondCode CC, SDValue CCOp, AArch64CC::CondCode Predicate, AArch64CC::CondCode OutCC, const SDLoc &DL, SelectionDAG &DAG)
can be transformed to: not (and (not (and (setCC (cmp C)) (setCD (cmp D)))) (and (not (setCA (cmp A))...
static SDValue tryAdvSIMDModImm16(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits, const SDValue *LHS=nullptr)
static SDValue tryAdvSIMDModImmFP(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits)
static SDValue tryAdvSIMDModImm64(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits)
static bool isCMN(SDValue Op, ISD::CondCode CC, SelectionDAG &DAG)
static SDValue tryAdvSIMDModImm8(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits)
static SDValue emitConjunctionRec(SelectionDAG &DAG, SDValue Val, AArch64CC::CondCode &OutCC, bool Negate, SDValue CCOp, AArch64CC::CondCode Predicate)
Emit conjunction or disjunction tree with the CMP/FCMP followed by a chain of CCMP/CFCMP ops.
static SDValue tryAdvSIMDModImm321s(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits)
static void changeFPCCToANDAArch64CC(ISD::CondCode CC, AArch64CC::CondCode &CondCode, AArch64CC::CondCode &CondCode2)
Convert a DAG fp condition code to an AArch64 CC.
static bool canEmitConjunction(SelectionDAG &DAG, const SDValue Val, bool &CanNegate, bool &MustBeFirst, bool &PreferFirst, bool WillNegate, unsigned Depth=0)
Returns true if Val is a tree of AND/OR/SETCC operations that can be expressed as a conjunction.
static SDValue tryAdvSIMDModImm32(unsigned NewOp, SDValue Op, SelectionDAG &DAG, const APInt &Bits, const SDValue *LHS=nullptr)
static SDValue emitConjunction(SelectionDAG &DAG, SDValue Val, AArch64CC::CondCode &OutCC)
Emit expression as a conjunction (a series of CCMP/CFCMP ops).
#define GET_GLOBALISEL_PREDICATES_INIT
#define GET_GLOBALISEL_TEMPORARIES_INIT
static Register getTestBitReg(Register Reg, uint64_t &Bit, bool &Invert, MachineRegisterInfo &MRI)
Return a register which can be used as a bit to test in a TB(N)Z.
static unsigned getMinSizeForRegBank(const RegisterBank &RB)
Returns the minimum size the given register bank can hold.
static std::optional< int64_t > getVectorShiftImm(Register Reg, MachineRegisterInfo &MRI)
Returns the element immediate value of a vector shift operand if found.
static unsigned selectLoadStoreUIOp(unsigned GenericOpc, unsigned RegBankID, unsigned OpSize)
Select the AArch64 opcode for the G_LOAD or G_STORE operation GenericOpc, appropriate for the (value)...
static const TargetRegisterClass * getMinClassForRegBank(const RegisterBank &RB, TypeSize SizeInBits, bool GetAllRegSet=false)
Given a register bank, and size in bits, return the smallest register class that can represent that c...
static unsigned selectBinaryOp(unsigned GenericOpc, unsigned RegBankID, unsigned OpSize)
Select the AArch64 opcode for the basic binary operation GenericOpc, appropriate for the register ban...
static bool getSubRegForClass(const TargetRegisterClass *RC, const TargetRegisterInfo &TRI, unsigned &SubReg)
Returns the correct subregister to use for a given register class.
static bool selectCopy(MachineInstr &I, const TargetInstrInfo &TII, MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
static bool copySubReg(MachineInstr &I, MachineRegisterInfo &MRI, const RegisterBankInfo &RBI, Register SrcReg, const TargetRegisterClass *To, unsigned SubReg)
Helper function for selectCopy.
static AArch64CC::CondCode changeICMPPredToAArch64CC(CmpInst::Predicate P, Register RHS={}, MachineRegisterInfo *MRI=nullptr)
static Register createDTuple(ArrayRef< Register > Regs, MachineIRBuilder &MIB)
Create a tuple of D-registers using the registers in Regs.
static void fixupPHIOpBanks(MachineInstr &MI, MachineRegisterInfo &MRI, const AArch64RegisterBankInfo &RBI)
static bool selectDebugInstr(MachineInstr &I, MachineRegisterInfo &MRI, const RegisterBankInfo &RBI)
static AArch64_AM::ShiftExtendType getShiftTypeForInst(MachineInstr &MI)
Given a shift instruction, return the correct shift type for that instruction.
static bool getLaneCopyOpcode(unsigned &CopyOpc, unsigned &ExtractSubReg, const unsigned EltSize)
static Register createQTuple(ArrayRef< Register > Regs, MachineIRBuilder &MIB)
Create a tuple of Q-registers using the registers in Regs.
static std::optional< uint64_t > getImmedFromMO(const MachineOperand &Root)
static std::pair< unsigned, unsigned > getInsertVecEltOpInfo(const RegisterBank &RB, unsigned EltSize)
Return an <Opcode, SubregIndex> pair to do an vector elt insert of a given size and RB.
static Register createTuple(ArrayRef< Register > Regs, const unsigned RegClassIDs[], const unsigned SubRegs[], MachineIRBuilder &MIB)
Create a REG_SEQUENCE instruction using the registers in Regs.
static std::optional< int64_t > getVectorSHLImm(LLT SrcTy, Register Reg, MachineRegisterInfo &MRI)
Matches and returns the shift immediate value for a SHL instruction given a shift operand.
static void changeFPCCToORAArch64CC(CmpInst::Predicate CC, AArch64CC::CondCode &CondCode, AArch64CC::CondCode &CondCode2)
changeFPCCToORAArch64CC - Convert an IR fp condition code to an AArch64 CC.
unsigned RegSize
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file declares the targeting of the RegisterBankInfo class for AArch64.
unsigned Imm
unsigned uint64_t
constexpr LLT S16
constexpr LLT S32
constexpr LLT S64
constexpr LLT S8
static bool isStore(int Opcode)
static bool selectMergeValues(MachineInstrBuilder &MIB, const ARMBaseInstrInfo &TII, MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
static bool selectUnmergeValues(MachineInstrBuilder &MIB, const ARMBaseInstrInfo &TII, MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
MachineBasicBlock & MBB
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
This file contains constants used for implementing Dwarf debug support.
Provides analysis for querying information about KnownBits during GISel passes.
#define DEBUG_TYPE
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
static void emitLoadFromConstantPool(Register DstReg, const Constant *ConstVal, MachineIRBuilder &MIRBuilder)
static bool isZero(Value *V, const DataLayout &DL, DominatorTree *DT, AssumptionCache *AC)
Definition Lint.cpp:540
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineConstantPool class which is an abstract constant pool to keep track of ...
This file declares the MachineIRBuilder class.
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define T
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
#define P(N)
static MachineBasicBlock * emitSelect(MachineInstr &MI, MachineBasicBlock *BB, const TargetInstrInfo *TII, const PPCSubtarget &Subtarget)
Emit SELECT instruction, using ISEL if available, otherwise use branch-based control flow.
if(PassOpts->AAPipeline)
static StringRef getName(Value *V)
#define LLVM_DEBUG(...)
Definition Debug.h:119
static constexpr int Concat[]
Value * RHS
Value * LHS
This class provides the information for the target register banks.
std::optional< uint16_t > getPtrAuthBlockAddressDiscriminatorIfEnabled(const Function &ParentFn) const
Compute the integer discriminator for a given BlockAddress constant, if blockaddress signing is enabl...
const AArch64TargetLowering * getTargetLowering() const override
unsigned ClassifyGlobalReference(const GlobalValue *GV, const TargetMachine &TM) const
ClassifyGlobalReference - Find the target operand flags that describe how a global value should be re...
bool isX16X17Safer() const
Returns whether the operating system makes it safer to store sensitive values in x16 and x17 as oppos...
bool isCallingConvWin64(CallingConv::ID CC, bool IsVarArg) const
APInt bitcastToAPInt() const
Definition APFloat.h:1475
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
static APInt getOneBitSet(unsigned numBits, unsigned BitNo)
Return an APInt with exactly one bit set in the result.
Definition APInt.h:235
unsigned countr_one() const
Count the number of trailing one bits.
Definition APInt.h:1676
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
BlockFrequencyInfo pass uses BlockFrequencyInfoImpl implementation to estimate IR basic block frequen...
bool isEquality() const
Determine if this is an equals/not equals predicate.
Definition InstrTypes.h:978
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_SGE
signed greater or equal
Definition InstrTypes.h:768
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Definition InstrTypes.h:852
bool isIntPredicate() const
Definition InstrTypes.h:846
bool isUnsigned() const
Definition InstrTypes.h:999
static LLVM_ABI Constant * getSplat(unsigned NumElts, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
const APFloat & getValueAPF() const
Definition Constants.h:463
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
unsigned getBitWidth() const
getBitWidth - Return the scalar bitwidth of this constant.
Definition Constants.h:162
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
Definition Constant.h:64
TypeSize getTypeStoreSize(Type *Ty) const
Returns the maximum number of bytes that may be overwritten by storing the specified type.
Definition DataLayout.h:579
LLVM_ABI Align getPrefTypeAlign(Type *Ty) const
Returns the preferred stack/global alignment for the specified type.
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
bool isVarArg() const
isVarArg - Return true if this function takes a variable number of arguments.
Definition Function.h:230
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:730
virtual void setupMF(MachineFunction &mf, GISelValueTracking *vt, CodeGenCoverage *covinfo=nullptr, ProfileSummaryInfo *psi=nullptr, BlockFrequencyInfo *bfi=nullptr)
Setup per-MF executor state.
Represents indexed stores.
Register getPointerReg() const
Get the source register of the pointer value.
MachineMemOperand & getMMO() const
Get the MachineMemOperand on this instruction.
LocationSize getMemSize() const
Returns the size in bytes of the memory access.
LocationSize getMemSizeInBits() const
Returns the size in bits of the memory access.
Represents a G_SELECT.
Register getCondReg() const
Register getFalseReg() const
Register getTrueReg() const
Register getReg(unsigned Idx) const
Access the Idx'th operand as a register and return it.
bool isThreadLocal() const
If the value is "Thread Local", its value isn't shared by the threads.
bool hasExternalWeakLinkage() const
bool isEquality() const
Return true if this predicate is either EQ or NE.
constexpr bool isScalableVector() const
Returns true if the LLT is a scalable vector.
constexpr unsigned getScalarSizeInBits() const
constexpr bool isScalar() const
LLT multiplyElements(int Factor) const
Produce a vector type that is Factor times bigger, preserving the element type.
constexpr LLT changeElementType(LLT NewEltTy) const
If this type is a vector, return a vector with the same number of elements but the new element type.
LLT getScalarType() const
constexpr bool isPointerVector() const
constexpr bool isInteger() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr bool isValid() const
constexpr uint16_t getNumElements() const
Returns the number of elements in a vector LLT.
constexpr bool isVector() const
static constexpr LLT pointer(unsigned AddressSpace, unsigned SizeInBits)
Get a low-level pointer in the given address space.
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr bool isPointer() const
constexpr unsigned getAddressSpace() const
static constexpr LLT fixed_vector(unsigned NumElements, unsigned ScalarSizeInBits)
Get a low-level fixed-width vector of some number of elements and element width.
static LLT integer(unsigned SizeInBits)
constexpr TypeSize getSizeInBytes() const
Returns the total size of the type in bytes, i.e.
LLT getElementType() const
Returns the vector's element type. Only valid for vector types.
TypeSize getValue() const
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
LLVM_ABI unsigned getConstantPoolIndex(const Constant *C, Align Alignment)
getConstantPoolIndex - Create a new entry in the constant pool or return an existing one.
void setFrameAddressIsTaken(bool T)
void setReturnAddressIsTaken(bool s)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineConstantPool * getConstantPool()
getConstantPool - Return the constant pool object for the current function.
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Helper class to build MachineInstr.
void setInsertPt(MachineBasicBlock &MBB, MachineBasicBlock::iterator II)
Set the insertion point before the specified position.
void setInstr(MachineInstr &MI)
Set the insertion point to before MI.
MachineInstrBuilder buildInstr(unsigned Opcode)
Build and insert <empty> = Opcode <empty>.
MachineFunction & getMF()
Getter for the function we currently build.
void setInstrAndDebugLoc(MachineInstr &MI)
Set the insertion point to before MI, and set the debug loc to MI's loc.
const MachineBasicBlock & getMBB() const
Getter for the basic block we currently build.
MachineRegisterInfo * getMRI()
Getter for MRI.
MachineIRBuilderState & getState()
Getter for the State.
MachineInstrBuilder buildCopy(const DstOp &Res, const SrcOp &Op)
Build and insert Res = COPY Op.
const DataLayout & getDataLayout() const
void setState(const MachineIRBuilderState &NewState)
Setter for the State.
MachineInstrBuilder buildPtrToInt(const DstOp &Dst, const SrcOp &Src)
Build and insert a G_PTRTOINT instruction.
Register getReg(unsigned Idx) const
Get the register for the operand index.
void constrainAllUses(const TargetInstrInfo &TII, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI) const
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & addBlockAddress(const BlockAddress *BA, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addRegMask(const uint32_t *Mask) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addJumpTableIndex(unsigned Idx, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
LLVM_ABI const MachineFunction * getMF() const
Return the function that contains the basic block that this instruction belongs to.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
LLVM_ABI void addMemOperand(MachineFunction &MF, MachineMemOperand *MO)
Add a MachineMemOperand to the machine instruction.
LLT getMemoryType() const
Return the memory type of the memory reference.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
AtomicOrdering getSuccessOrdering() const
Return the atomic ordering requirements for this memory operation.
MachineOperand class - Representation of each machine instruction operand.
const GlobalValue * getGlobal() const
const ConstantInt * getCImm() const
bool isCImm() const
isCImm - Test if this is a MO_CImmediate operand.
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
static MachineOperand CreatePredicate(unsigned Pred)
static MachineOperand CreateImm(int64_t Val)
Register getReg() const
getReg - Returns the register number.
static MachineOperand CreateGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
static MachineOperand CreateBA(const BlockAddress *BA, int64_t Offset, unsigned TargetFlags=0)
const ConstantFP * getFPImm() const
unsigned getPredicate() const
int64_t getOffset() const
Return the offset from the symbol in this operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
const RegClassOrRegBank & getRegClassOrRegBank(Register Reg) const
Return the register bank or register class of Reg.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
def_instr_iterator def_instr_begin(Register RegNo) const
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
const RegisterBank * getRegBankOrNull(Register Reg) const
Return the register bank of Reg, or null if Reg has not been assigned a register bank or has been ass...
LLVM_ABI void setRegBank(Register Reg, const RegisterBank &RegBank)
Set the register bank to RegBank for Reg.
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
LLVM_ABI void setType(Register VReg, LLT Ty)
Set the low-level type of VReg to Ty.
bool hasOneDef(Register RegNo) const
Return true if there is exactly one operand defining the specified register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI Register createGenericVirtualRegister(LLT Ty, StringRef Name="")
Create and return a new generic virtual register with low-level type Ty.
const TargetRegisterClass * getRegClassOrNull(Register Reg) const
Return the register class of Reg, or null if Reg has not been assigned a register class yet.
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
Analysis providing profile information.
Holds all the information related to register banks.
static const TargetRegisterClass * constrainGenericRegister(Register Reg, const TargetRegisterClass &RC, MachineRegisterInfo &MRI)
Constrain the (possibly generic) virtual register Reg to RC.
const RegisterBank & getRegBank(unsigned ID)
Get the register bank identified by ID.
TypeSize getSizeInBits(Register Reg, const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI) const
Get the size in bits of Reg.
This class implements the register bank concept.
unsigned getID() const
Get the identifier of this register bank.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isValid() const
Definition Register.h:112
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
void assign(size_type NumElts, ValueParamT Elt)
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
TargetInstrInfo - Interface to description of machine instruction set.
bool isPositionIndependent() const
bool useEmulatedTLS() const
Returns true if this target uses emulated TLS.
TargetOptions Options
CodeModel::Model getCodeModel() const
Returns the code model.
unsigned TLSSize
Bit size of immediate TLS offsets (0 == use the default).
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
virtual const TargetRegisterInfo * getRegisterInfo() const =0
Return the target's register information.
virtual const TargetLowering * getTargetLowering() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Definition Value.cpp:1002
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
self_iterator getIterator()
Definition ilist_node.h:123
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
static CondCode getInvertedCondCode(CondCode Code)
static unsigned getNZCVToSatisfyCondCode(CondCode Code)
Given a condition code, return NZCV flags that would satisfy that condition.
void changeFCMPPredToAArch64CC(const CmpInst::Predicate P, AArch64CC::CondCode &CondCode, AArch64CC::CondCode &CondCode2)
Find the AArch64 condition codes necessary to represent P for a scalar floating point comparison.
std::optional< int64_t > getAArch64VectorSplatScalar(const MachineInstr &MI, const MachineRegisterInfo &MRI)
@ MO_NC
MO_NC - Indicates whether the linker is expected to check the symbol reference for overflow.
@ MO_G1
MO_G1 - A symbol operand with this flag (granule 1) represents the bits 16-31 of a 64-bit address,...
@ MO_PAGEOFF
MO_PAGEOFF - A symbol operand with this flag represents the offset of that symbol within a 4K page.
@ MO_GOT
MO_GOT - This flag indicates that a symbol operand represents the address of the GOT entry for the sy...
@ MO_G0
MO_G0 - A symbol operand with this flag (granule 0) represents the bits 0-15 of a 64-bit address,...
@ MO_PAGE
MO_PAGE - A symbol operand with this flag represents the pc-relative offset of the 4K page containing...
@ MO_HI12
MO_HI12 - This flag indicates that a symbol operand represents the bits 13-24 of a 64-bit address,...
@ MO_TLS
MO_TLS - Indicates that the operand being accessed is some kind of thread-local symbol.
@ MO_G2
MO_G2 - A symbol operand with this flag (granule 2) represents the bits 32-47 of a 64-bit address,...
@ MO_G3
MO_G3 - A symbol operand with this flag (granule 3) represents the high 16-bits of a 64-bit address,...
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
static uint8_t encodeAdvSIMDModImmType2(uint64_t Imm)
static bool isAdvSIMDModImmType9(uint64_t Imm)
static bool isAdvSIMDModImmType4(uint64_t Imm)
static bool isAdvSIMDModImmType5(uint64_t Imm)
static int getFP32Imm(const APInt &Imm)
getFP32Imm - Return an 8-bit floating-point version of the 32-bit floating-point value.
static uint8_t encodeAdvSIMDModImmType7(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType12(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType10(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType9(uint64_t Imm)
static uint64_t encodeLogicalImmediate(uint64_t imm, unsigned regSize)
encodeLogicalImmediate - Return the encoded immediate value for a logical immediate instruction of th...
static bool isAdvSIMDModImmType7(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType5(uint64_t Imm)
static int getFP64Imm(const APInt &Imm)
getFP64Imm - Return an 8-bit floating-point version of the 64-bit floating-point value.
static bool isAdvSIMDModImmType10(uint64_t Imm)
static int getFP16Imm(const APInt &Imm)
getFP16Imm - Return an 8-bit floating-point version of the 16-bit floating-point value.
static uint8_t encodeAdvSIMDModImmType8(uint64_t Imm)
static bool isAdvSIMDModImmType12(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType11(uint64_t Imm)
static bool isAdvSIMDModImmType11(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType6(uint64_t Imm)
static bool isAdvSIMDModImmType8(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType4(uint64_t Imm)
static unsigned getShifterImm(AArch64_AM::ShiftExtendType ST, unsigned Imm)
getShifterImm - Encode the shift type and amount: imm: 6-bit shift amount shifter: 000 ==> lsl 001 ==...
static bool isAdvSIMDModImmType6(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType1(uint64_t Imm)
static uint8_t encodeAdvSIMDModImmType3(uint64_t Imm)
static bool isAdvSIMDModImmType2(uint64_t Imm)
static bool isAdvSIMDModImmType3(uint64_t Imm)
static bool isSignExtendShiftType(AArch64_AM::ShiftExtendType Type)
isSignExtendShiftType - Returns true if Type is sign extending.
static bool isAdvSIMDModImmType1(uint64_t Imm)
TLSModel::Model getELFTLSModel(const GlobalValue *GV, const TargetMachine &TM, bool HasELFSignedGOT)
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Attrs[]
Key for Kernel::Metadata::mAttrs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
operand_type_match m_Reg()
SpecificConstantMatch m_SpecificICst(const APInt &RequestedValue)
Matches a constant equal to RequestedValue.
UnaryOp_match< SrcTy, TargetOpcode::G_ZEXT > m_GZExt(const SrcTy &Src)
ConstantMatch< APInt > m_ICst(APInt &Cst)
BinaryOp_match< LHS, RHS, TargetOpcode::G_ADD, true > m_GAdd(const LHS &L, const RHS &R)
auto m_PosZeroFP()
Matches a floating-point positive zero.
BinaryOp_match< LHS, RHS, TargetOpcode::G_OR, true > m_GOr(const LHS &L, const RHS &R)
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
OneNonDBGUse_match< SubPat > m_OneNonDBGUse(const SubPat &SP)
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_PTR_ADD, false > m_GPtrAdd(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_SHL, false > m_GShl(const LHS &L, const RHS &R)
Or< Preds... > m_any_of(Preds &&... preds)
BinaryOp_match< LHS, RHS, TargetOpcode::G_AND, true > m_GAnd(const LHS &L, const RHS &R)
Predicate
Predicate - These are "(BI << 5) | BO" for various predicates.
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
constexpr double e
NodeAddr< InstrNode * > Instr
Definition RDFGraph.h:389
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Register getFunctionLiveInPhysReg(MachineFunction &MF, const TargetInstrInfo &TII, MCRegister PhysReg, const TargetRegisterClass &RC, const DebugLoc &DL, LLT RegTy=LLT())
Return a virtual register corresponding to the incoming argument register PhysReg.
Definition Utils.cpp:848
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
LLVM_ABI Register constrainOperandRegClass(const MachineFunction &MF, const TargetRegisterInfo &TRI, MachineRegisterInfo &MRI, const TargetInstrInfo &TII, const RegisterBankInfo &RBI, MachineInstr &InsertPt, const TargetRegisterClass &RegClass, MachineOperand &RegMO)
Constrain the Register operand OpIdx, so that it is now constrained to the TargetRegisterClass passed...
Definition Utils.cpp:60
LLVM_ABI MachineInstr * getOpcodeDef(unsigned Opcode, Register Reg, const MachineRegisterInfo &MRI)
See if Reg is defined by an single def instruction that is Opcode.
Definition Utils.cpp:656
PointerUnion< const TargetRegisterClass *, const RegisterBank * > RegClassOrRegBank
Convenient type to represent either a register class or a register bank.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
LLVM_ABI std::optional< APInt > getIConstantVRegVal(Register VReg, const MachineRegisterInfo &MRI)
If VReg is defined by a G_CONSTANT, return the corresponding value.
Definition Utils.cpp:297
unsigned CheckFixedPointOperandConstant(APFloat &FVal, unsigned RegWidth, bool isReciprocal)
@ Undef
Value of the register doesn't matter.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool isStrongerThanMonotonic(AtomicOrdering AO)
LLVM_ABI void constrainSelectedInstRegOperands(MachineInstr &I, const TargetInstrInfo &TII, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
Mutate the newly-selected instruction I to constrain its (possibly generic) virtual register operands...
Definition Utils.cpp:159
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
bool isPreISelGenericOpcode(unsigned Opcode)
Check whether the given Opcode is a generic opcode that is not supposed to appear after ISel.
unsigned getBLRCallOpcode(const MachineFunction &MF)
Return opcode to be used for indirect calls.
@ O1
Optimize quickly without destroying debuggability.
@ O0
Disable as many optimizations as possible.
LLVM_ABI MachineInstr * getDefIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, folding away any trivial copies.
Definition Utils.cpp:497
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI std::optional< int64_t > getIConstantVRegSExtVal(Register VReg, const MachineRegisterInfo &MRI)
If VReg is defined by a G_CONSTANT fits in int64_t returns it.
Definition Utils.cpp:317
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
InstructionSelector * createAArch64InstructionSelector(const AArch64TargetMachine &, const AArch64Subtarget &, const AArch64RegisterBankInfo &)
OutputIt transform(R &&Range, OutputIt d_first, UnaryFunction F)
Wrapper function around std::transform to apply a function to a range and store the result elsewhere.
Definition STLExtras.h:2042
constexpr bool has_single_bit(T Value) noexcept
Definition bit.h:149
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
LLVM_ABI std::optional< ValueAndVReg > getAnyConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true, bool LookThroughAnyExt=false)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT or G_FCONST...
Definition Utils.cpp:442
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
AtomicOrdering
Atomic ordering for LLVM's memory model.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
LLVM_ABI std::optional< DefinitionAndSourceRegister > getDefSrcRegIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, and underlying value Register folding away any copies.
Definition Utils.cpp:472
LLVM_ABI Register getSrcRegIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the source register for Reg, folding away any trivial copies.
Definition Utils.cpp:504
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
static EVT getFloatingPointVT(unsigned BitWidth)
Returns the EVT that represents a floating-point type with the given number of bits.
Definition ValueTypes.h:55
static LLVM_ABI MachinePointerInfo getConstantPool(MachineFunction &MF)
Return a MachinePointerInfo record that refers to the constant pool.