LLVM 24.0.0git
AArch64ISelDAGToDAG.cpp
Go to the documentation of this file.
1//===-- AArch64ISelDAGToDAG.cpp - A dag to dag inst selector for AArch64 --===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines an instruction selector for the AArch64 target.
10//
11//===----------------------------------------------------------------------===//
12
13#include "AArch64.h"
14#include "AArch64ExpandImm.h"
18#include "llvm/ADT/APSInt.h"
22#include "llvm/IR/Function.h" // To access function attributes.
23#include "llvm/IR/GlobalValue.h"
24#include "llvm/IR/Intrinsics.h"
25#include "llvm/IR/IntrinsicsAArch64.h"
27#include "llvm/Support/Debug.h"
32
33using namespace llvm;
34using namespace llvm::SDPatternMatch;
35
36#define DEBUG_TYPE "aarch64-isel"
37#define PASS_NAME "AArch64 Instruction Selection"
38
39// https://github.com/llvm/llvm-project/issues/114425
40#if defined(_MSC_VER) && !defined(__clang__) && !defined(NDEBUG)
41#pragma inline_depth(0)
42#endif
43
44//===--------------------------------------------------------------------===//
45/// AArch64DAGToDAGISel - AArch64 specific code to select AArch64 machine
46/// instructions for SelectionDAG operations.
47///
48namespace {
49
50class AArch64DAGToDAGISel : public SelectionDAGISel {
51
52 /// Subtarget - Keep a pointer to the AArch64Subtarget around so that we can
53 /// make the right decision when generating code for different targets.
54 const AArch64Subtarget *Subtarget;
55
56public:
57 AArch64DAGToDAGISel() = delete;
58
59 explicit AArch64DAGToDAGISel(AArch64TargetMachine &tm,
60 CodeGenOptLevel OptLevel)
61 : SelectionDAGISel(tm, OptLevel), Subtarget(nullptr) {}
62
63 bool runOnMachineFunction(MachineFunction &MF) override {
64 Subtarget = &MF.getSubtarget<AArch64Subtarget>();
66 }
67
68 void Select(SDNode *Node) override;
69 void PreprocessISelDAG() override;
70
71 /// SelectInlineAsmMemoryOperand - Implement addressing mode selection for
72 /// inline asm expressions.
73 bool SelectInlineAsmMemoryOperand(const SDValue &Op,
74 InlineAsm::ConstraintCode ConstraintID,
75 std::vector<SDValue> &OutOps) override;
76
77 template <signed Low, signed High, signed Scale>
78 bool SelectRDVLImm(SDValue N, SDValue &Imm);
79
80 template <signed Low, signed High>
81 bool SelectRDSVLShiftImm(SDValue N, SDValue &Imm);
82
83 bool SelectArithExtendedRegister(SDValue N, SDValue &Reg, SDValue &Shift);
84 bool SelectArithUXTXRegister(SDValue N, SDValue &Reg, SDValue &Shift);
85 bool SelectArithImmed(SDValue N, SDValue &Val, SDValue &Shift);
86 bool SelectNegArithImmed(SDValue N, SDValue &Val, SDValue &Shift);
87 bool SelectArithShiftedRegister(SDValue N, SDValue &Reg, SDValue &Shift) {
88 return SelectShiftedRegister(N, false, Reg, Shift);
89 }
90 bool SelectLogicalShiftedRegister(SDValue N, SDValue &Reg, SDValue &Shift) {
91 return SelectShiftedRegister(N, true, Reg, Shift);
92 }
93 template <unsigned ShiftWidth>
94 bool SelectShiftMask(SDValue N, SDValue &ShAmt);
95
96 bool SelectAddrModeIndexed7S8(SDValue N, SDValue &Base, SDValue &OffImm) {
97 return SelectAddrModeIndexed7S(N, 1, Base, OffImm);
98 }
99 bool SelectAddrModeIndexed7S16(SDValue N, SDValue &Base, SDValue &OffImm) {
100 return SelectAddrModeIndexed7S(N, 2, Base, OffImm);
101 }
102 bool SelectAddrModeIndexed7S32(SDValue N, SDValue &Base, SDValue &OffImm) {
103 return SelectAddrModeIndexed7S(N, 4, Base, OffImm);
104 }
105 bool SelectAddrModeIndexed7S64(SDValue N, SDValue &Base, SDValue &OffImm) {
106 return SelectAddrModeIndexed7S(N, 8, Base, OffImm);
107 }
108 bool SelectAddrModeIndexed7S128(SDValue N, SDValue &Base, SDValue &OffImm) {
109 return SelectAddrModeIndexed7S(N, 16, Base, OffImm);
110 }
111 bool SelectAddrModeIndexedS9S128(SDValue N, SDValue &Base, SDValue &OffImm) {
112 return SelectAddrModeIndexedBitWidth(N, true, 9, 16, Base, OffImm);
113 }
114 bool SelectAddrModeIndexedU6S128(SDValue N, SDValue &Base, SDValue &OffImm) {
115 return SelectAddrModeIndexedBitWidth(N, false, 6, 16, Base, OffImm);
116 }
117 bool SelectAddrModeIndexed8(SDValue N, SDValue &Base, SDValue &OffImm) {
118 return SelectAddrModeIndexed(N, 1, Base, OffImm);
119 }
120 bool SelectAddrModeIndexed16(SDValue N, SDValue &Base, SDValue &OffImm) {
121 return SelectAddrModeIndexed(N, 2, Base, OffImm);
122 }
123 bool SelectAddrModeIndexed32(SDValue N, SDValue &Base, SDValue &OffImm) {
124 return SelectAddrModeIndexed(N, 4, Base, OffImm);
125 }
126 bool SelectAddrModeIndexed64(SDValue N, SDValue &Base, SDValue &OffImm) {
127 return SelectAddrModeIndexed(N, 8, Base, OffImm);
128 }
129 bool SelectAddrModeIndexed128(SDValue N, SDValue &Base, SDValue &OffImm) {
130 return SelectAddrModeIndexed(N, 16, Base, OffImm);
131 }
132 bool SelectAddrModeUnscaled8(SDValue N, SDValue &Base, SDValue &OffImm) {
133 return SelectAddrModeUnscaled(N, 1, Base, OffImm);
134 }
135 bool SelectAddrModeUnscaled16(SDValue N, SDValue &Base, SDValue &OffImm) {
136 return SelectAddrModeUnscaled(N, 2, Base, OffImm);
137 }
138 bool SelectAddrModeUnscaled32(SDValue N, SDValue &Base, SDValue &OffImm) {
139 return SelectAddrModeUnscaled(N, 4, Base, OffImm);
140 }
141 bool SelectAddrModeUnscaled64(SDValue N, SDValue &Base, SDValue &OffImm) {
142 return SelectAddrModeUnscaled(N, 8, Base, OffImm);
143 }
144 bool SelectAddrModeUnscaled128(SDValue N, SDValue &Base, SDValue &OffImm) {
145 return SelectAddrModeUnscaled(N, 16, Base, OffImm);
146 }
147 template <unsigned Size, unsigned Max>
148 bool SelectAddrModeIndexedUImm(SDValue N, SDValue &Base, SDValue &OffImm) {
149 // Test if there is an appropriate addressing mode and check if the
150 // immediate fits.
151 bool Found = SelectAddrModeIndexed(N, Size, Base, OffImm);
152 if (Found) {
153 if (auto *CI = dyn_cast<ConstantSDNode>(OffImm)) {
154 int64_t C = CI->getSExtValue();
155 if (C <= Max)
156 return true;
157 }
158 }
159
160 // Otherwise, base only, materialize address in register.
161 Base = N;
162 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i64);
163 return true;
164 }
165
166 template<int Width>
167 bool SelectAddrModeWRO(SDValue N, SDValue &Base, SDValue &Offset,
168 SDValue &SignExtend, SDValue &DoShift) {
169 return SelectAddrModeWRO(N, Width / 8, Base, Offset, SignExtend, DoShift);
170 }
171
172 template<int Width>
173 bool SelectAddrModeXRO(SDValue N, SDValue &Base, SDValue &Offset,
174 SDValue &SignExtend, SDValue &DoShift) {
175 return SelectAddrModeXRO(N, Width / 8, Base, Offset, SignExtend, DoShift);
176 }
177
178 bool SelectExtractHigh(SDValue N, SDValue &Res) {
179 if (Subtarget->isLittleEndian() && N->getOpcode() == ISD::BITCAST)
180 N = N->getOperand(0);
181 if (N->getOpcode() != ISD::EXTRACT_SUBVECTOR ||
182 !isa<ConstantSDNode>(N->getOperand(1)))
183 return false;
184 EVT VT = N->getValueType(0);
185 EVT LVT = N->getOperand(0).getValueType();
186 unsigned Index = N->getConstantOperandVal(1);
187 if (!VT.is64BitVector() || !LVT.is128BitVector() ||
188 Index != VT.getVectorNumElements())
189 return false;
190 Res = N->getOperand(0);
191 return true;
192 }
193
194 bool SelectRoundingVLShr(SDValue N, SDValue &Res1, SDValue &Res2) {
195 if (N.getOpcode() != AArch64ISD::VLSHR)
196 return false;
197 SDValue Op = N->getOperand(0);
198 EVT VT = Op.getValueType();
199 unsigned ShtAmt = N->getConstantOperandVal(1);
200 if (ShtAmt > VT.getScalarSizeInBits() / 2 || Op.getOpcode() != ISD::ADD)
201 return false;
202
203 APInt Imm;
204 if (Op.getOperand(1).getOpcode() == AArch64ISD::MOVIshift)
206 Op.getOperand(1).getConstantOperandVal(0)
207 << Op.getOperand(1).getConstantOperandVal(1));
208 else if (Op.getOperand(1).getOpcode() == AArch64ISD::DUP &&
209 isa<ConstantSDNode>(Op.getOperand(1).getOperand(0)))
211 Op.getOperand(1).getConstantOperandVal(0));
212 else
213 return false;
214
215 if (Imm != 1ULL << (ShtAmt - 1))
216 return false;
217
218 Res1 = Op.getOperand(0);
219 Res2 = CurDAG->getTargetConstant(ShtAmt, SDLoc(N), MVT::i32);
220 return true;
221 }
222
223 bool SelectDupZeroOrUndef(SDValue N) {
224 switch(N->getOpcode()) {
225 case ISD::UNDEF:
226 case ISD::POISON:
227 return true;
228 case AArch64ISD::DUP:
229 case ISD::SPLAT_VECTOR: {
230 auto Opnd0 = N->getOperand(0);
231 if (isNullConstant(Opnd0))
232 return true;
233 if (isNullFPConstant(Opnd0))
234 return true;
235 break;
236 }
237 default:
238 break;
239 }
240
241 return false;
242 }
243
244 bool SelectAny(SDValue) { return true; }
245
246 bool SelectDupZero(SDValue N) {
247 switch(N->getOpcode()) {
248 case AArch64ISD::DUP:
249 case ISD::SPLAT_VECTOR: {
250 auto Opnd0 = N->getOperand(0);
251 if (isNullConstant(Opnd0))
252 return true;
253 if (isNullFPConstant(Opnd0))
254 return true;
255 break;
256 }
257 }
258
259 return false;
260 }
261
262 template <MVT::SimpleValueType VT, bool Negate>
263 bool SelectSVEAddSubImm(SDValue N, SDValue &Imm, SDValue &Shift) {
264 return SelectSVEAddSubImm(N, VT, Imm, Shift, Negate);
265 }
266
267 template <MVT::SimpleValueType VT, bool Negate>
268 bool SelectSVEAddSubSSatImm(SDValue N, SDValue &Imm, SDValue &Shift) {
269 return SelectSVEAddSubSSatImm(N, VT, Imm, Shift, Negate);
270 }
271
272 template <MVT::SimpleValueType VT>
273 bool SelectSVECpyDupImm(SDValue N, SDValue &Imm, SDValue &Shift) {
274 return SelectSVECpyDupImm(N, VT, Imm, Shift);
275 }
276
277 template <MVT::SimpleValueType VT, bool Invert = false>
278 bool SelectSVELogicalImm(SDValue N, SDValue &Imm) {
279 return SelectSVELogicalImm(N, VT, Imm, Invert);
280 }
281
282 template <MVT::SimpleValueType VT>
283 bool SelectSVEArithImm(SDValue N, SDValue &Imm) {
284 return SelectSVEArithImm(N, VT, Imm);
285 }
286
287 template <unsigned Low, unsigned High, bool AllowSaturation = false>
288 bool SelectSVEShiftImm(SDValue N, SDValue &Imm) {
289 return SelectSVEShiftImm(N, Low, High, AllowSaturation, Imm);
290 }
291
292 bool SelectSVEShiftSplatImmR(SDValue N, SDValue &Imm) {
293 if (N->getOpcode() != ISD::SPLAT_VECTOR)
294 return false;
295
296 EVT EltVT = N->getValueType(0).getVectorElementType();
297 return SelectSVEShiftImm(N->getOperand(0), /* Low */ 1,
298 /* High */ EltVT.getFixedSizeInBits(),
299 /* AllowSaturation */ true, Imm);
300 }
301
302 // Returns a suitable CNT/INC/DEC/RDVL multiplier to calculate VSCALE*N.
303 template<signed Min, signed Max, signed Scale, bool Shift>
304 bool SelectCntImm(SDValue N, SDValue &Imm) {
306 return false;
307
308 int64_t MulImm = cast<ConstantSDNode>(N)->getSExtValue();
309 if (Shift)
310 MulImm = 1LL << MulImm;
311
312 if ((MulImm % std::abs(Scale)) != 0)
313 return false;
314
315 MulImm /= Scale;
316 if ((MulImm >= Min) && (MulImm <= Max)) {
317 Imm = CurDAG->getTargetConstant(MulImm, SDLoc(N), MVT::i32);
318 return true;
319 }
320
321 return false;
322 }
323
324 template <signed Max, signed Scale>
325 bool SelectEXTImm(SDValue N, SDValue &Imm) {
327 return false;
328
329 int64_t MulImm = cast<ConstantSDNode>(N)->getSExtValue();
330
331 if (MulImm >= 0 && MulImm <= Max) {
332 MulImm *= Scale;
333 Imm = CurDAG->getTargetConstant(MulImm, SDLoc(N), MVT::i32);
334 return true;
335 }
336
337 return false;
338 }
339
340 template <unsigned BaseReg, unsigned Max>
341 bool ImmToReg(SDValue N, SDValue &Imm) {
342 if (auto *CI = dyn_cast<ConstantSDNode>(N)) {
343 uint64_t C = CI->getZExtValue();
344
345 if (C > Max)
346 return false;
347
348 Imm = CurDAG->getRegister(BaseReg + C, MVT::Other);
349 return true;
350 }
351 return false;
352 }
353
354 /// Form sequences of consecutive 64/128-bit registers for use in NEON
355 /// instructions making use of a vector-list (e.g. ldN, tbl). Vecs must have
356 /// between 1 and 4 elements. If it contains a single element that is returned
357 /// unchanged; otherwise a REG_SEQUENCE value is returned.
360 // Form a sequence of SVE registers for instructions using list of vectors,
361 // e.g. structured loads and stores (ldN, stN).
362 SDValue createZTuple(ArrayRef<SDValue> Vecs);
363
364 // Similar to above, except the register must start at a multiple of the
365 // tuple, e.g. z2 for a 2-tuple, or z8 for a 4-tuple.
366 SDValue createZMulTuple(ArrayRef<SDValue> Regs);
367
368 /// Generic helper for the createDTuple/createQTuple
369 /// functions. Those should almost always be called instead.
370 SDValue createTuple(ArrayRef<SDValue> Vecs, const unsigned RegClassIDs[],
371 const unsigned SubRegs[]);
372
373 void SelectTable(SDNode *N, unsigned NumVecs, unsigned Opc, bool isExt);
374
375 bool tryIndexedLoad(SDNode *N);
376
377 void SelectPtrauthAuth(SDNode *N);
378 void SelectPtrauthResign(SDNode *N);
379 void SelectPtrauthResignWithPC(SDNode *N);
380
381 bool trySelectStackSlotTagP(SDNode *N);
382 void SelectTagP(SDNode *N);
383
384 void SelectLoad(SDNode *N, unsigned NumVecs, unsigned Opc,
385 unsigned SubRegIdx);
386 void SelectPostLoad(SDNode *N, unsigned NumVecs, unsigned Opc,
387 unsigned SubRegIdx);
388 void SelectLoadLane(SDNode *N, unsigned NumVecs, unsigned Opc);
389 void SelectPostLoadLane(SDNode *N, unsigned NumVecs, unsigned Opc);
390 void SelectPredicatedLoad(SDNode *N, unsigned NumVecs, unsigned Scale,
391 unsigned Opc_rr, unsigned Opc_ri,
392 bool IsIntr = false);
393 void SelectContiguousMultiVectorLoad(SDNode *N, unsigned NumVecs,
394 unsigned Scale, unsigned Opc_ri,
395 unsigned Opc_rr);
396 void SelectDestructiveMultiIntrinsic(SDNode *N, unsigned NumVecs,
397 bool IsZmMulti, unsigned Opcode,
398 bool HasPred = false);
399 void SelectPExtPair(SDNode *N, unsigned Opc);
400 void SelectWhilePair(SDNode *N, unsigned Opc);
401 void SelectCVTIntrinsic(SDNode *N, unsigned NumVecs, unsigned Opcode);
402 void SelectCVTIntrinsicFP8(SDNode *N, unsigned NumVecs, unsigned Opcode);
403 void SelectClamp(SDNode *N, unsigned NumVecs, unsigned Opcode);
404 void SelectUnaryMultiIntrinsic(SDNode *N, unsigned NumOutVecs,
405 bool IsTupleInput, unsigned Opc);
406 void SelectFrintFromVT(SDNode *N, unsigned NumVecs, unsigned Opcode);
407
408 template <unsigned MaxIdx, unsigned Scale>
409 void SelectMultiVectorMove(SDNode *N, unsigned NumVecs, unsigned BaseReg,
410 unsigned Op);
411 void SelectMultiVectorMoveZ(SDNode *N, unsigned NumVecs,
412 unsigned Op, unsigned MaxIdx, unsigned Scale,
413 unsigned BaseReg = 0);
414 /// SVE Reg+Imm addressing mode.
415 template <int64_t Min, int64_t Max>
416 bool SelectAddrModeIndexedSVE(SDNode *Root, SDValue N, SDValue &Base,
417 SDValue &OffImm);
418 /// SVE Reg+Reg address mode.
419 template <unsigned Scale>
420 bool SelectSVERegRegAddrMode(SDValue N, SDValue &Base, SDValue &Offset) {
421 return SelectSVERegRegAddrMode(N, Scale, Base, Offset);
422 }
423
424 void SelectMultiVectorLutiLane(SDNode *Node, unsigned NumOutVecs,
425 unsigned Opc, uint32_t MaxImm);
426 void SelectMultiVectorLuti6LaneX4(SDNode *Node, unsigned NumIndexVecs);
427
428 void SelectMultiVectorLuti(SDNode *Node, unsigned NumOutVecs, unsigned Opc,
429 unsigned NumInVecs);
430
431 template <unsigned MaxIdx, unsigned Scale>
432 bool SelectSMETileSlice(SDValue N, SDValue &Vector, SDValue &Offset) {
433 return SelectSMETileSlice(N, MaxIdx, Vector, Offset, Scale);
434 }
435
436 void SelectStore(SDNode *N, unsigned NumVecs, unsigned Opc);
437 void SelectPostStore(SDNode *N, unsigned NumVecs, unsigned Opc);
438 void SelectStoreLane(SDNode *N, unsigned NumVecs, unsigned Opc);
439 void SelectPostStoreLane(SDNode *N, unsigned NumVecs, unsigned Opc);
440 void SelectPredicatedStore(SDNode *N, unsigned NumVecs, unsigned Scale,
441 unsigned Opc_rr, unsigned Opc_ri);
442 std::tuple<unsigned, SDValue, SDValue>
443 findAddrModeSVELoadStore(SDNode *N, unsigned Opc_rr, unsigned Opc_ri,
444 const SDValue &OldBase, const SDValue &OldOffset,
445 unsigned Scale);
446
447 bool tryBitfieldExtractOp(SDNode *N);
448 bool tryBitfieldExtractOpFromSExt(SDNode *N);
449 bool tryBitfieldInsertOp(SDNode *N);
450 bool tryBitfieldInsertInZeroOp(SDNode *N);
451 bool tryShiftAmountMod(SDNode *N);
452
453 bool tryReadRegister(SDNode *N);
454 bool tryWriteRegister(SDNode *N);
455
456 bool trySelectCastFixedLengthToScalableVector(SDNode *N);
457 bool trySelectCastScalableToFixedLengthVector(SDNode *N);
458
459 bool trySelectXAR(SDNode *N);
460
461 bool tryFoldCselToFMaxMin(SDNode *N);
462
463// Include the pieces autogenerated from the target description.
464#include "AArch64GenDAGISel.inc"
465
466private:
467 bool SelectShiftedRegister(SDValue N, bool AllowROR, SDValue &Reg,
468 SDValue &Shift);
469 bool SelectShiftedRegisterFromAnd(SDValue N, SDValue &Reg, SDValue &Shift);
470 bool SelectAddrModeIndexed7S(SDValue N, unsigned Size, SDValue &Base,
471 SDValue &OffImm) {
472 return SelectAddrModeIndexedBitWidth(N, true, 7, Size, Base, OffImm);
473 }
474 bool SelectAddrModeIndexedBitWidth(SDValue N, bool IsSignedImm, unsigned BW,
475 unsigned Size, SDValue &Base,
476 SDValue &OffImm);
477 bool SelectAddrModeIndexed(SDValue N, unsigned Size, SDValue &Base,
478 SDValue &OffImm);
479 bool SelectAddrModeUnscaled(SDValue N, unsigned Size, SDValue &Base,
480 SDValue &OffImm);
481 bool SelectAddrModeWRO(SDValue N, unsigned Size, SDValue &Base,
482 SDValue &Offset, SDValue &SignExtend,
483 SDValue &DoShift);
484 bool SelectAddrModeXRO(SDValue N, unsigned Size, SDValue &Base,
485 SDValue &Offset, SDValue &SignExtend,
486 SDValue &DoShift);
487 bool isWorthNegatingImm(SDValue V) const;
488 bool isWorthFoldingALU(SDValue V, bool LSL = false) const;
489 bool isWorthFoldingAddr(SDValue V, unsigned Size) const;
490 bool SelectExtendedSHL(SDValue N, unsigned Size, bool WantExtend,
491 SDValue &Offset, SDValue &SignExtend);
492
493 template<unsigned RegWidth>
494 bool SelectCVTFixedPosOperand(SDValue N, SDValue &FixedPos) {
495 return SelectCVTFixedPosOperand(N, FixedPos, RegWidth);
496 }
497 bool SelectCVTFixedPosOperand(SDValue N, SDValue &FixedPos, unsigned Width);
498
499 template <unsigned RegWidth>
500 bool SelectCVTFixedPointVec(SDValue N, SDValue &FixedPos) {
501 return SelectCVTFixedPointVec(N, FixedPos, RegWidth);
502 }
503 bool SelectCVTFixedPointVec(SDValue N, SDValue &FixedPos, unsigned Width);
504
505 template<unsigned RegWidth>
506 bool SelectCVTFixedPosRecipOperand(SDValue N, SDValue &FixedPos) {
507 return SelectCVTFixedPosRecipOperand(N, FixedPos, RegWidth);
508 }
509
510 bool SelectCVTFixedPosRecipOperand(SDValue N, SDValue &FixedPos,
511 unsigned Width);
512
513 template <unsigned FloatWidth>
514 bool SelectCVTFixedPosRecipOperandVec(SDValue N, SDValue &FixedPos) {
515 return SelectCVTFixedPosRecipOperandVec(N, FixedPos, FloatWidth);
516 }
517
518 bool SelectCVTFixedPosRecipOperandVec(SDValue N, SDValue &FixedPos,
519 unsigned Width);
520
521 bool SelectCMP_SWAP(SDNode *N);
522
523 AArch64MemoryHint decodeMemoryHintFlags(MachineMemOperand *MMO) const;
524 bool isAtomicSTSHH_KEEP(SDNode *N) const;
525 bool isAtomicSTSHH_STRM(SDNode *N) const;
526
527 bool SelectSVEAddSubImm(SDValue N, MVT VT, SDValue &Imm, SDValue &Shift,
528 bool Negate);
529 bool SelectSVEAddSubImm(SDLoc DL, APInt Value, MVT VT, SDValue &Imm,
530 SDValue &Shift, bool Negate);
531 bool SelectSVEAddSubSSatImm(SDValue N, MVT VT, SDValue &Imm, SDValue &Shift,
532 bool Negate);
533 bool SelectSVECpyDupImm(SDValue N, MVT VT, SDValue &Imm, SDValue &Shift);
534 bool SelectSVELogicalImm(SDValue N, MVT VT, SDValue &Imm, bool Invert);
535
536 // Match `<NEON Splat> SVEImm` (where <NEON Splat> could be fmov, movi, etc).
537 bool SelectNEONSplatOfSVELogicalImm(SDValue N, SDValue &Imm);
538 bool SelectNEONSplatOfSVEAddSubImm(SDValue N, SDValue &Imm, SDValue &Shift);
539 bool SelectNEONSplatOfSVEArithSImm(SDValue N, SDValue &Imm);
540 bool SelectNEONSplatOfSImm8(SDValue N, SDValue &Imm);
541 bool SelectNEONSplatOfUImm8(SDValue N, SDValue &Imm);
542
543 bool SelectSVESignedArithImm(SDLoc DL, APInt Value, SDValue &Imm);
544 bool SelectSVESignedArithImm(SDValue N, SDValue &Imm);
545 bool SelectSVEShiftImm(SDValue N, uint64_t Low, uint64_t High,
546 bool AllowSaturation, SDValue &Imm);
547
548 bool SelectSVEArithImm(SDValue N, MVT VT, SDValue &Imm);
549 bool SelectSVERegRegAddrMode(SDValue N, unsigned Scale, SDValue &Base,
550 SDValue &Offset);
551 bool SelectSMETileSlice(SDValue N, unsigned MaxSize, SDValue &Vector,
552 SDValue &Offset, unsigned Scale = 1);
553
554 bool SelectAllActivePredicate(SDValue N);
555 bool SelectAnyPredicate(SDValue N);
556
557 bool SelectCmpBranchUImm6Operand(SDNode *P, SDValue N, SDValue &Imm);
558
559 template <bool MatchCBB>
560 bool SelectCmpBranchExtOperand(SDValue N, SDValue &Reg, SDValue &ExtType);
561};
562
563class AArch64DAGToDAGISelLegacy : public SelectionDAGISelLegacy {
564public:
565 static char ID;
566 explicit AArch64DAGToDAGISelLegacy(AArch64TargetMachine &tm,
567 CodeGenOptLevel OptLevel)
569 ID, std::make_unique<AArch64DAGToDAGISel>(tm, OptLevel)) {}
570};
571} // end anonymous namespace
572
573char AArch64DAGToDAGISelLegacy::ID = 0;
574
575INITIALIZE_PASS(AArch64DAGToDAGISelLegacy, DEBUG_TYPE, PASS_NAME, false, false)
576
579 std::make_unique<AArch64DAGToDAGISel>(TM, TM.getOptLevel())) {}
580
581/// addBitcastHints - This method adds bitcast hints to the operands of a node
582/// to help instruction selector determine which operands are in Neon registers.
584 SDLoc DL(&N);
585 auto getFloatVT = [&](EVT VT) {
586 EVT ScalarVT = VT.getScalarType();
587 assert((ScalarVT == MVT::i32 || ScalarVT == MVT::i64) && "Unexpected VT");
588 return VT.changeElementType(*(DAG.getContext()),
589 ScalarVT == MVT::i32 ? MVT::f32 : MVT::f64);
590 };
592 NewOps.reserve(N.getNumOperands());
593
594 for (unsigned I = 0, E = N.getNumOperands(); I < E; ++I) {
595 auto bitcasted = DAG.getBitcast(getFloatVT(N.getOperand(I).getValueType()),
596 N.getOperand(I));
597 NewOps.push_back(bitcasted);
598 }
599 EVT OrigVT = N.getValueType(0);
600 SDValue OpNode = DAG.getNode(N.getOpcode(), DL, getFloatVT(OrigVT), NewOps);
601 return DAG.getBitcast(OrigVT, OpNode);
602}
603
604/// isIntImmediate - This method tests to see if the node is a constant
605/// operand. If so Imm will receive the 64-bit value.
606static bool isIntImmediate(const SDNode *N, uint64_t &Imm) {
608 Imm = C->getZExtValue();
609 return true;
610 }
611 return false;
612}
613
614// isIntImmediate - This method tests to see if a constant operand.
615// If so Imm will receive the value.
617 return isIntImmediate(N.getNode(), Imm);
618}
619
620// isOpcWithIntImmediate - This method tests to see if the node is a specific
621// opcode and that it has a immediate integer right operand.
622// If so Imm will receive the 32 bit value.
623static bool isOpcWithIntImmediate(const SDNode *N, unsigned Opc,
624 uint64_t &Imm) {
625 return N->getOpcode() == Opc &&
626 isIntImmediate(N->getOperand(1).getNode(), Imm);
627}
628
629// isIntImmediateEq - This method tests to see if N is a constant operand that
630// is equivalent to 'ImmExpected'.
631#ifndef NDEBUG
632static bool isIntImmediateEq(SDValue N, const uint64_t ImmExpected) {
634 if (!isIntImmediate(N.getNode(), Imm))
635 return false;
636 return Imm == ImmExpected;
637}
638#endif
639
640static APInt DecodeFMOVImm(uint64_t Imm, unsigned RegWidth) {
641 assert(RegWidth == 32 || RegWidth == 64);
642 if (RegWidth == 32)
643 return APInt(RegWidth,
646}
647
648// Decodes the raw integer splat value from a NEON splat operation.
649static std::optional<APInt> DecodeNEONSplat(SDValue N,
650 const AArch64Subtarget *Subtarget) {
651 assert(N.getValueType().isInteger() && "Only integers are supported");
652 if (N->getOpcode() == AArch64ISD::NVCAST ||
653 (N->getOpcode() == ISD::BITCAST && Subtarget->isLittleEndian()))
654 N = N->getOperand(0);
655 unsigned SplatWidth = N.getScalarValueSizeInBits();
656 if (N.getOpcode() == AArch64ISD::FMOV)
657 return DecodeFMOVImm(N.getConstantOperandVal(0), SplatWidth);
658 if (N->getOpcode() == AArch64ISD::MOVI)
659 return APInt(SplatWidth, N.getConstantOperandVal(0));
660 if (N->getOpcode() == AArch64ISD::MOVIshift)
661 return APInt(SplatWidth, N.getConstantOperandVal(0)
662 << N.getConstantOperandVal(1));
663 if (N->getOpcode() == AArch64ISD::MVNIshift)
664 return ~APInt(SplatWidth, N.getConstantOperandVal(0)
665 << N.getConstantOperandVal(1));
666 if (N->getOpcode() == AArch64ISD::MOVIedit)
668 N.getConstantOperandVal(0)));
669 if (N->getOpcode() == AArch64ISD::DUP)
670 if (auto *Const = dyn_cast<ConstantSDNode>(N->getOperand(0)))
671 return Const->getAPIntValue().trunc(SplatWidth);
672 APInt SplatVal;
673 if (ISD::isConstantSplatVector(N.getNode(), SplatVal))
674 return SplatVal.trunc(SplatWidth);
675 // TODO: Recognize more splat-like NEON operations. See ConstantBuildVector
676 // in AArch64ISelLowering.
677 return std::nullopt;
678}
679
680// If \p N is a NEON splat operation (movi, fmov, etc), return the splat value
681// matching the element size of N.
682static std::optional<APInt>
684 unsigned SplatWidth = N.getScalarValueSizeInBits();
685 if (std::optional<APInt> SplatVal = DecodeNEONSplat(N, Subtarget)) {
686 if (SplatVal->getBitWidth() <= SplatWidth)
687 return APInt::getSplat(SplatWidth, *SplatVal);
688 if (SplatVal->isSplat(SplatWidth))
689 return SplatVal->trunc(SplatWidth);
690 }
691 return std::nullopt;
692}
693
694bool AArch64DAGToDAGISel::SelectNEONSplatOfSVELogicalImm(SDValue N,
695 SDValue &Imm) {
696 std::optional<APInt> ImmVal = GetNEONSplatValue(N, Subtarget);
697 if (!ImmVal)
698 return false;
699 uint64_t Encoding;
700 if (!AArch64_AM::isSVELogicalImm(N.getScalarValueSizeInBits(),
701 ImmVal->getZExtValue(), Encoding))
702 return false;
703
704 Imm = CurDAG->getTargetConstant(Encoding, SDLoc(N), MVT::i64);
705 return true;
706}
707
708bool AArch64DAGToDAGISel::SelectNEONSplatOfSVEAddSubImm(SDValue N, SDValue &Imm,
709 SDValue &Shift) {
710 if (std::optional<APInt> ImmVal = GetNEONSplatValue(N, Subtarget))
711 return SelectSVEAddSubImm(SDLoc(N), *ImmVal,
712 N.getValueType().getScalarType().getSimpleVT(),
713 Imm, Shift,
714 /*Negate=*/false);
715 return false;
716}
717
718bool AArch64DAGToDAGISel::SelectNEONSplatOfSVEArithSImm(SDValue N,
719 SDValue &Imm) {
720 if (std::optional<APInt> ImmVal = GetNEONSplatValue(N, Subtarget))
721 return SelectSVESignedArithImm(SDLoc(N), *ImmVal, Imm);
722 return false;
723}
724
725bool AArch64DAGToDAGISel::SelectNEONSplatOfSImm8(SDValue N, SDValue &Imm) {
726 std::optional<APInt> ImmAPIntVal = GetNEONSplatValue(N, Subtarget);
727 if (!ImmAPIntVal)
728 return false;
729
730 int64_t ImmVal = ImmAPIntVal->getSExtValue();
731 if (ImmVal < -128 || ImmVal > 127)
732 return false;
733
734 Imm = CurDAG->getSignedTargetConstant(ImmVal, SDLoc(N), MVT::i32);
735 return true;
736}
737
738bool AArch64DAGToDAGISel::SelectNEONSplatOfUImm8(SDValue N, SDValue &Imm) {
739 std::optional<APInt> ImmAPIntVal = GetNEONSplatValue(N, Subtarget);
740 if (!ImmAPIntVal)
741 return false;
742
743 uint64_t ImmVal = ImmAPIntVal->getZExtValue();
744 if (ImmVal > 255)
745 return false;
746
747 Imm = CurDAG->getTargetConstant(ImmVal, SDLoc(N), MVT::i32);
748 return true;
749}
750
751bool AArch64DAGToDAGISel::SelectInlineAsmMemoryOperand(
752 const SDValue &Op, const InlineAsm::ConstraintCode ConstraintID,
753 std::vector<SDValue> &OutOps) {
754 switch(ConstraintID) {
755 default:
756 llvm_unreachable("Unexpected asm memory constraint");
757 case InlineAsm::ConstraintCode::m:
758 case InlineAsm::ConstraintCode::o:
759 case InlineAsm::ConstraintCode::Q:
760 // We need to make sure that this one operand does not end up in XZR, thus
761 // require the address to be in a pointer register.
762 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
763 const TargetRegisterClass *TRC =
765 SDLoc dl(Op);
766 SDValue RC = CurDAG->getTargetConstant(TRC->getID(), dl, MVT::i64);
767 SDValue NewOp =
768 SDValue(CurDAG->getMachineNode(TargetOpcode::COPY_TO_REGCLASS,
769 dl, Op.getValueType(),
770 Op, RC), 0);
771 OutOps.push_back(NewOp);
772 return false;
773 }
774 return true;
775}
776
777template <unsigned ShiftWidth>
778bool AArch64DAGToDAGISel::SelectShiftMask(SDValue N, SDValue &ShAmt) {
779 // AArch64 shift instructions only use the low log2(ShiftWidth) bits of the
780 // shift amount. If the shift amount has a redundant AND mask that covers
781 // those bits, we can remove it. Return false if nothing was combined so
782 // other patterns (e.g. zext/sext GPR32 → SUBREG_TO_REG) can match.
783 if (N.getOpcode() == ISD::AND && isa<ConstantSDNode>(N.getOperand(1)) &&
784 N.getValueType() == (ShiftWidth == 32 ? MVT::i32 : MVT::i64)) {
785 uint64_t Mask = N.getConstantOperandVal(1);
786 // Remove AND if the mask covers at least the low log2(ShiftWidth) bits.
787 if ((unsigned)llvm::countr_one(Mask) >= Log2_32(ShiftWidth)) {
788 ShAmt = N.getOperand(0);
789 return true;
790 }
791 }
792 // If shifting by X+/-N where N == 0 mod ShiftWidth, then just shift by X
793 // to avoid the ADD/SUB. The low log2(ShiftWidth) bits are unchanged, so the
794 // shift can use X directly; the original ADD/SUB stays for any other users.
795 if ((N.getOpcode() == ISD::ADD || N.getOpcode() == ISD::SUB) &&
796 N.getValueType() == (ShiftWidth == 32 ? MVT::i32 : MVT::i64)) {
798 if (isIntImmediate(N.getOperand(1).getNode(), Imm) &&
799 (Imm % ShiftWidth == 0)) {
800 ShAmt = N.getOperand(0);
801 return true;
802 }
803 }
804
805 return false;
806}
807
808/// SelectArithImmed - Select an immediate value that can be represented as
809/// a 12-bit value shifted left by either 0 or 12. If so, return true with
810/// Val set to the 12-bit value and Shift set to the shifter operand.
811bool AArch64DAGToDAGISel::SelectArithImmed(SDValue N, SDValue &Val,
812 SDValue &Shift) {
813 // This function is called from the addsub_shifted_imm ComplexPattern,
814 // which lists [imm] as the list of opcode it's interested in, however
815 // we still need to check whether the operand is actually an immediate
816 // here because the ComplexPattern opcode list is only used in
817 // root-level opcode matching.
818 if (!isa<ConstantSDNode>(N.getNode()))
819 return false;
820
821 uint64_t Immed = N.getNode()->getAsZExtVal();
822
824 return false;
825
826 unsigned ShiftAmt = AArch64_AM::getArithImmedShift(Immed);
827 Immed >>= ShiftAmt;
828
829 unsigned ShVal = AArch64_AM::getShifterImm(AArch64_AM::LSL, ShiftAmt);
830 SDLoc dl(N);
831 Val = CurDAG->getTargetConstant(Immed, dl, MVT::i32);
832 Shift = CurDAG->getTargetConstant(ShVal, dl, MVT::i32);
833 return true;
834}
835
836/// SelectNegArithImmed - As above, but negates the value before trying to
837/// select it.
838bool AArch64DAGToDAGISel::SelectNegArithImmed(SDValue N, SDValue &Val,
839 SDValue &Shift) {
840 // This function is called from the addsub_shifted_imm ComplexPattern,
841 // which lists [imm] as the list of opcode it's interested in, however
842 // we still need to check whether the operand is actually an immediate
843 // here because the ComplexPattern opcode list is only used in
844 // root-level opcode matching.
845 if (!isa<ConstantSDNode>(N.getNode()))
846 return false;
847
848 // The immediate operand must be a 24-bit zero-extended immediate.
849 uint64_t Immed = N.getNode()->getAsZExtVal();
850
851 // This negation is almost always valid, but "cmp wN, #0" and "cmn wN, #0"
852 // have the opposite effect on the C flag, so this pattern mustn't match under
853 // those circumstances.
854 if (Immed == 0)
855 return false;
856
857 if (N.getValueType() == MVT::i32)
858 Immed = ~((uint32_t)Immed) + 1;
859 else
860 Immed = ~Immed + 1ULL;
861 if (Immed & 0xFFFFFFFFFF000000ULL)
862 return false;
863
864 Immed &= 0xFFFFFFULL;
865 return SelectArithImmed(CurDAG->getConstant(Immed, SDLoc(N), MVT::i32), Val,
866 Shift);
867}
868
869/// getShiftTypeForNode - Translate a shift node to the corresponding
870/// ShiftType value.
872 switch (N.getOpcode()) {
873 default:
875 case ISD::SHL:
876 return AArch64_AM::LSL;
877 case ISD::SRL:
878 return AArch64_AM::LSR;
879 case ISD::SRA:
880 return AArch64_AM::ASR;
881 case ISD::ROTR:
882 return AArch64_AM::ROR;
883 }
884}
885
887 return isa<MemSDNode>(*N) || N->getOpcode() == AArch64ISD::PREFETCH;
888}
889
890/// Determine whether it is worth it to fold SHL into the addressing
891/// mode.
893 assert(V.getOpcode() == ISD::SHL && "invalid opcode");
894 // It is worth folding logical shift of up to three places.
895 auto *CSD = dyn_cast<ConstantSDNode>(V.getOperand(1));
896 if (!CSD)
897 return false;
898 unsigned ShiftVal = CSD->getZExtValue();
899 if (ShiftVal > 3)
900 return false;
901
902 // Check if this particular node is reused in any non-memory related
903 // operation. If yes, do not try to fold this node into the address
904 // computation, since the computation will be kept.
905 const SDNode *Node = V.getNode();
906 for (SDNode *UI : Node->users())
907 if (!isMemOpOrPrefetch(UI))
908 for (SDNode *UII : UI->users())
909 if (!isMemOpOrPrefetch(UII))
910 return false;
911 return true;
912}
913
914/// Determine whether it is worth to fold V into an extended register addressing
915/// mode.
916bool AArch64DAGToDAGISel::isWorthFoldingAddr(SDValue V, unsigned Size) const {
917 // Trivial if we are optimizing for code size or if there is only
918 // one use of the value.
919 if (CurDAG->shouldOptForSize() || V.hasOneUse())
920 return true;
921
922 // If a subtarget has a slow shift, folding a shift into multiple loads
923 // costs additional micro-ops.
924 if (Subtarget->hasAddrLSLSlow14() && (Size == 2 || Size == 16))
925 return false;
926
927 // Check whether we're going to emit the address arithmetic anyway because
928 // it's used by a non-address operation.
929 if (V.getOpcode() == ISD::SHL && isWorthFoldingSHL(V))
930 return true;
931 if (V.getOpcode() == ISD::ADD) {
932 const SDValue LHS = V.getOperand(0);
933 const SDValue RHS = V.getOperand(1);
934 if (LHS.getOpcode() == ISD::SHL && isWorthFoldingSHL(LHS))
935 return true;
936 if (RHS.getOpcode() == ISD::SHL && isWorthFoldingSHL(RHS))
937 return true;
938 }
939
940 // It hurts otherwise, since the value will be reused.
941 return false;
942}
943
944/// and (shl/srl/sra, x, c), mask --> shl (srl/sra, x, c1), c2
945/// to select more shifted register
946bool AArch64DAGToDAGISel::SelectShiftedRegisterFromAnd(SDValue N, SDValue &Reg,
947 SDValue &Shift) {
948 EVT VT = N.getValueType();
949 if (VT != MVT::i32 && VT != MVT::i64)
950 return false;
951
952 if (N->getOpcode() != ISD::AND || !N->hasOneUse())
953 return false;
954 SDValue LHS = N.getOperand(0);
955 if (!LHS->hasOneUse())
956 return false;
957
958 unsigned LHSOpcode = LHS->getOpcode();
959 if (LHSOpcode != ISD::SHL && LHSOpcode != ISD::SRL && LHSOpcode != ISD::SRA)
960 return false;
961
962 ConstantSDNode *ShiftAmtNode = dyn_cast<ConstantSDNode>(LHS.getOperand(1));
963 if (!ShiftAmtNode)
964 return false;
965
966 uint64_t ShiftAmtC = ShiftAmtNode->getZExtValue();
967 ConstantSDNode *RHSC = dyn_cast<ConstantSDNode>(N.getOperand(1));
968 if (!RHSC)
969 return false;
970
971 APInt AndMask = RHSC->getAPIntValue();
972 unsigned LowZBits, MaskLen;
973 if (!AndMask.isShiftedMask(LowZBits, MaskLen))
974 return false;
975
976 unsigned BitWidth = N.getValueSizeInBits();
977 SDLoc DL(LHS);
978 uint64_t NewShiftC;
979 unsigned NewShiftOp;
980 if (LHSOpcode == ISD::SHL) {
981 // LowZBits <= ShiftAmtC will fall into isBitfieldPositioningOp
982 // BitWidth != LowZBits + MaskLen doesn't match the pattern
983 if (LowZBits <= ShiftAmtC || (BitWidth != LowZBits + MaskLen))
984 return false;
985
986 NewShiftC = LowZBits - ShiftAmtC;
987 NewShiftOp = VT == MVT::i64 ? AArch64::UBFMXri : AArch64::UBFMWri;
988 } else {
989 if (LowZBits == 0)
990 return false;
991
992 // NewShiftC >= BitWidth will fall into isBitfieldExtractOp
993 NewShiftC = LowZBits + ShiftAmtC;
994 if (NewShiftC >= BitWidth)
995 return false;
996
997 // SRA need all high bits
998 if (LHSOpcode == ISD::SRA && (BitWidth != (LowZBits + MaskLen)))
999 return false;
1000
1001 // SRL high bits can be 0 or 1
1002 if (LHSOpcode == ISD::SRL && (BitWidth > (NewShiftC + MaskLen)))
1003 return false;
1004
1005 if (LHSOpcode == ISD::SRL)
1006 NewShiftOp = VT == MVT::i64 ? AArch64::UBFMXri : AArch64::UBFMWri;
1007 else
1008 NewShiftOp = VT == MVT::i64 ? AArch64::SBFMXri : AArch64::SBFMWri;
1009 }
1010
1011 assert(NewShiftC < BitWidth && "Invalid shift amount");
1012 SDValue NewShiftAmt = CurDAG->getTargetConstant(NewShiftC, DL, VT);
1013 SDValue BitWidthMinus1 = CurDAG->getTargetConstant(BitWidth - 1, DL, VT);
1014 Reg = SDValue(CurDAG->getMachineNode(NewShiftOp, DL, VT, LHS->getOperand(0),
1015 NewShiftAmt, BitWidthMinus1),
1016 0);
1017 unsigned ShVal = AArch64_AM::getShifterImm(AArch64_AM::LSL, LowZBits);
1018 Shift = CurDAG->getTargetConstant(ShVal, DL, MVT::i32);
1019 return true;
1020}
1021
1022/// getExtendTypeForNode - Translate an extend node to the corresponding
1023/// ExtendType value.
1025getExtendTypeForNode(SDValue N, bool IsLoadStore = false) {
1026 if (N.getOpcode() == ISD::SIGN_EXTEND ||
1027 N.getOpcode() == ISD::SIGN_EXTEND_INREG) {
1028 EVT SrcVT;
1029 if (N.getOpcode() == ISD::SIGN_EXTEND_INREG)
1030 SrcVT = cast<VTSDNode>(N.getOperand(1))->getVT();
1031 else
1032 SrcVT = N.getOperand(0).getValueType();
1033
1034 if (!IsLoadStore && SrcVT == MVT::i8)
1035 return AArch64_AM::SXTB;
1036 else if (!IsLoadStore && SrcVT == MVT::i16)
1037 return AArch64_AM::SXTH;
1038 else if (SrcVT == MVT::i32)
1039 return AArch64_AM::SXTW;
1040 assert(SrcVT != MVT::i64 && "extend from 64-bits?");
1041
1043 } else if (N.getOpcode() == ISD::ZERO_EXTEND ||
1044 N.getOpcode() == ISD::ANY_EXTEND) {
1045 EVT SrcVT = N.getOperand(0).getValueType();
1046 if (!IsLoadStore && SrcVT == MVT::i8)
1047 return AArch64_AM::UXTB;
1048 else if (!IsLoadStore && SrcVT == MVT::i16)
1049 return AArch64_AM::UXTH;
1050 else if (SrcVT == MVT::i32)
1051 return AArch64_AM::UXTW;
1052 assert(SrcVT != MVT::i64 && "extend from 64-bits?");
1053
1055 } else if (N.getOpcode() == ISD::AND) {
1056 ConstantSDNode *CSD = dyn_cast<ConstantSDNode>(N.getOperand(1));
1057 if (!CSD)
1059 uint64_t AndMask = CSD->getZExtValue();
1060
1061 switch (AndMask) {
1062 default:
1064 case 0xFF:
1065 return !IsLoadStore ? AArch64_AM::UXTB : AArch64_AM::InvalidShiftExtend;
1066 case 0xFFFF:
1067 return !IsLoadStore ? AArch64_AM::UXTH : AArch64_AM::InvalidShiftExtend;
1068 case 0xFFFFFFFF:
1069 return AArch64_AM::UXTW;
1070 }
1071 }
1072
1074}
1075
1076/// Determine whether constant -V is cheaper to materialise than V.
1077bool AArch64DAGToDAGISel::isWorthNegatingImm(SDValue V) const {
1078 assert(isa<ConstantSDNode>(V) && "invalid node");
1079
1080 EVT VT = V.getValueType();
1081 assert((VT == MVT::i32 || VT == MVT::i64) && "invalid type");
1082
1083 // It's only worth negating the constant if it doesn't have other uses.
1084 if (!V.hasOneUse())
1085 return false;
1086
1087 uint64_t Imm = cast<ConstantSDNode>(V)->getZExtValue();
1088 unsigned BitSize = VT.getSizeInBits();
1090 AArch64_IMM::expandMOVImm(Imm, BitSize, OrigCost);
1091 AArch64_IMM::expandMOVImm(-Imm, BitSize, NewCost);
1092 return NewCost.size() < OrigCost.size();
1093}
1094
1095/// Determine whether it is worth to fold V into an extended register of an
1096/// Add/Sub. LSL means we are folding into an `add w0, w1, w2, lsl #N`
1097/// instruction, and the shift should be treated as worth folding even if has
1098/// multiple uses.
1099bool AArch64DAGToDAGISel::isWorthFoldingALU(SDValue V, bool LSL) const {
1100 // Trivial if we are optimizing for code size or if there is only
1101 // one use of the value.
1102 if (CurDAG->shouldOptForSize() || V.hasOneUse())
1103 return true;
1104
1105 // If a subtarget has a fastpath LSL we can fold a logical shift into
1106 // the add/sub and save a cycle.
1107 if (LSL && Subtarget->hasALULSLFast() && V.getOpcode() == ISD::SHL &&
1108 V.getConstantOperandVal(1) <= 4 &&
1110 return true;
1111
1112 // It hurts otherwise, since the value will be reused.
1113 return false;
1114}
1115
1116/// SelectShiftedRegister - Select a "shifted register" operand. If the value
1117/// is not shifted, set the Shift operand to default of "LSL 0". The logical
1118/// instructions allow the shifted register to be rotated, but the arithmetic
1119/// instructions do not. The AllowROR parameter specifies whether ROR is
1120/// supported.
1121bool AArch64DAGToDAGISel::SelectShiftedRegister(SDValue N, bool AllowROR,
1122 SDValue &Reg, SDValue &Shift) {
1123 if (SelectShiftedRegisterFromAnd(N, Reg, Shift))
1124 return true;
1125
1127 if (ShType == AArch64_AM::InvalidShiftExtend)
1128 return false;
1129 if (!AllowROR && ShType == AArch64_AM::ROR)
1130 return false;
1131
1132 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1133 unsigned BitSize = N.getValueSizeInBits();
1134 unsigned Val = RHS->getZExtValue() & (BitSize - 1);
1135 unsigned ShVal = AArch64_AM::getShifterImm(ShType, Val);
1136
1137 Reg = N.getOperand(0);
1138 Shift = CurDAG->getTargetConstant(ShVal, SDLoc(N), MVT::i32);
1139 return isWorthFoldingALU(N, true);
1140 }
1141
1142 return false;
1143}
1144
1145/// Instructions that accept extend modifiers like UXTW expect the register
1146/// being extended to be a GPR32, but the incoming DAG might be acting on a
1147/// GPR64 (either via SEXT_INREG or AND). Extract the appropriate low bits if
1148/// this is the case.
1150 if (N.getValueType() == MVT::i32)
1151 return N;
1152
1153 SDLoc dl(N);
1154 return CurDAG->getTargetExtractSubreg(AArch64::sub_32, dl, MVT::i32, N);
1155}
1156
1157// Returns a suitable CNT/INC/DEC/RDVL multiplier to calculate VSCALE*N.
1158template<signed Low, signed High, signed Scale>
1159bool AArch64DAGToDAGISel::SelectRDVLImm(SDValue N, SDValue &Imm) {
1160 if (!isa<ConstantSDNode>(N))
1161 return false;
1162
1163 int64_t MulImm = cast<ConstantSDNode>(N)->getSExtValue();
1164 if ((MulImm % std::abs(Scale)) == 0) {
1165 int64_t RDVLImm = MulImm / Scale;
1166 if ((RDVLImm >= Low) && (RDVLImm <= High)) {
1167 Imm = CurDAG->getSignedTargetConstant(RDVLImm, SDLoc(N), MVT::i32);
1168 return true;
1169 }
1170 }
1171
1172 return false;
1173}
1174
1175// Returns a suitable RDSVL multiplier from a left shift.
1176template <signed Low, signed High>
1177bool AArch64DAGToDAGISel::SelectRDSVLShiftImm(SDValue N, SDValue &Imm) {
1178 if (!isa<ConstantSDNode>(N))
1179 return false;
1180
1181 int64_t MulImm = 1LL << cast<ConstantSDNode>(N)->getSExtValue();
1182 if (MulImm >= Low && MulImm <= High) {
1183 Imm = CurDAG->getSignedTargetConstant(MulImm, SDLoc(N), MVT::i32);
1184 return true;
1185 }
1186
1187 return false;
1188}
1189
1190/// SelectArithExtendedRegister - Select a "extended register" operand. This
1191/// operand folds in an extend followed by an optional left shift.
1192bool AArch64DAGToDAGISel::SelectArithExtendedRegister(SDValue N, SDValue &Reg,
1193 SDValue &Shift) {
1194 unsigned ShiftVal = 0;
1196
1197 if (N.getOpcode() == ISD::SHL) {
1198 ConstantSDNode *CSD = dyn_cast<ConstantSDNode>(N.getOperand(1));
1199 if (!CSD)
1200 return false;
1201 ShiftVal = CSD->getZExtValue();
1202 if (ShiftVal > 4)
1203 return false;
1204
1205 Ext = getExtendTypeForNode(N.getOperand(0));
1207 return false;
1208
1209 Reg = N.getOperand(0).getOperand(0);
1210 } else {
1211 Ext = getExtendTypeForNode(N);
1213 return false;
1214
1215 // Don't match sext of vector extracts. These can use SMOV, but if we match
1216 // this as an extended register, we'll always fold the extend into an ALU op
1217 // user of the extend (which results in a UMOV).
1219 SDValue Op = N.getOperand(0);
1220 if (Op->getOpcode() == ISD::ANY_EXTEND)
1221 Op = Op->getOperand(0);
1222 if (Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
1223 Op.getOperand(0).getValueType().isFixedLengthVector())
1224 return false;
1225 }
1226
1227 Reg = N.getOperand(0);
1228
1229 // Don't match if free 32-bit -> 64-bit zext can be used instead. Use the
1230 // isDef32 as a heuristic for when the operand is likely to be a 32bit def.
1231 auto isDef32 = [](SDValue N) {
1232 unsigned Opc = N.getOpcode();
1233 return Opc != ISD::TRUNCATE && Opc != TargetOpcode::EXTRACT_SUBREG &&
1236 Opc != ISD::FREEZE;
1237 };
1238 if (Ext == AArch64_AM::UXTW && Reg->getValueType(0).getSizeInBits() == 32 &&
1239 isDef32(Reg))
1240 return false;
1241 }
1242
1243 // Don't match if the sext can be folded with an asr to form an SBFX.
1244 if (Ext == AArch64_AM::SXTW && Reg.getOpcode() == ISD::SRA &&
1245 Reg.getValueType() == MVT::i32 &&
1246 isa<ConstantSDNode>(Reg.getOperand(1)) && Reg.hasOneUse())
1247 return false;
1248
1249 // AArch64 mandates that the RHS of the operation must use the smallest
1250 // register class that could contain the size being extended from. Thus,
1251 // if we're folding a (sext i8), we need the RHS to be a GPR32, even though
1252 // there might not be an actual 32-bit value in the program. We can
1253 // (harmlessly) synthesize one by injected an EXTRACT_SUBREG here.
1254 assert(Ext != AArch64_AM::UXTX && Ext != AArch64_AM::SXTX);
1255 Reg = narrowIfNeeded(CurDAG, Reg);
1256 Shift = CurDAG->getTargetConstant(getArithExtendImm(Ext, ShiftVal), SDLoc(N),
1257 MVT::i32);
1258 return isWorthFoldingALU(N);
1259}
1260
1261/// SelectArithUXTXRegister - Select a "UXTX register" operand. This
1262/// operand is referred by the instructions have SP operand
1263bool AArch64DAGToDAGISel::SelectArithUXTXRegister(SDValue N, SDValue &Reg,
1264 SDValue &Shift) {
1265 unsigned ShiftVal = 0;
1267
1268 if (N.getOpcode() != ISD::SHL)
1269 return false;
1270
1271 ConstantSDNode *CSD = dyn_cast<ConstantSDNode>(N.getOperand(1));
1272 if (!CSD)
1273 return false;
1274 ShiftVal = CSD->getZExtValue();
1275 if (ShiftVal > 4)
1276 return false;
1277
1278 Ext = AArch64_AM::UXTX;
1279 Reg = N.getOperand(0);
1280 Shift = CurDAG->getTargetConstant(getArithExtendImm(Ext, ShiftVal), SDLoc(N),
1281 MVT::i32);
1282 return isWorthFoldingALU(N);
1283}
1284
1285/// If there's a use of this ADDlow that's not itself a load/store then we'll
1286/// need to create a real ADD instruction from it anyway and there's no point in
1287/// folding it into the mem op. Theoretically, it shouldn't matter, but there's
1288/// a single pseudo-instruction for an ADRP/ADD pair so over-aggressive folding
1289/// leads to duplicated ADRP instructions.
1291 for (auto *User : N->users()) {
1292 if (User->getOpcode() != ISD::LOAD && User->getOpcode() != ISD::STORE &&
1293 User->getOpcode() != ISD::ATOMIC_LOAD &&
1294 User->getOpcode() != ISD::ATOMIC_STORE)
1295 return false;
1296
1297 // ldar and stlr have much more restrictive addressing modes (just a
1298 // register).
1299 if (isStrongerThanMonotonic(cast<MemSDNode>(User)->getSuccessOrdering()))
1300 return false;
1301 }
1302
1303 return true;
1304}
1305
1306/// Check if the immediate offset is valid as a scaled immediate.
1307static bool isValidAsScaledImmediate(int64_t Offset, unsigned Range,
1308 unsigned Size) {
1309 if ((Offset & (Size - 1)) == 0 && Offset >= 0 &&
1310 Offset < (Range << Log2_32(Size)))
1311 return true;
1312 return false;
1313}
1314
1315/// SelectAddrModeIndexedBitWidth - Select a "register plus scaled (un)signed BW-bit
1316/// immediate" address. The "Size" argument is the size in bytes of the memory
1317/// reference, which determines the scale.
1318bool AArch64DAGToDAGISel::SelectAddrModeIndexedBitWidth(SDValue N, bool IsSignedImm,
1319 unsigned BW, unsigned Size,
1320 SDValue &Base,
1321 SDValue &OffImm) {
1322 SDLoc dl(N);
1323 const DataLayout &DL = CurDAG->getDataLayout();
1324 const TargetLowering *TLI = getTargetLowering();
1325 if (N.getOpcode() == ISD::FrameIndex) {
1326 int FI = cast<FrameIndexSDNode>(N)->getIndex();
1327 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1328 OffImm = CurDAG->getTargetConstant(0, dl, MVT::i64);
1329 return true;
1330 }
1331
1332 // As opposed to the (12-bit) Indexed addressing mode below, the 7/9-bit signed
1333 // selected here doesn't support labels/immediates, only base+offset.
1334 if (CurDAG->isBaseWithConstantOffset(N)) {
1335 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1336 if (IsSignedImm) {
1337 int64_t RHSC = RHS->getSExtValue();
1338 unsigned Scale = Log2_32(Size);
1339 int64_t Range = 0x1LL << (BW - 1);
1340
1341 if ((RHSC & (Size - 1)) == 0 && RHSC >= -(Range << Scale) &&
1342 RHSC < (Range << Scale)) {
1343 Base = N.getOperand(0);
1344 if (Base.getOpcode() == ISD::FrameIndex) {
1345 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1346 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1347 }
1348 OffImm = CurDAG->getTargetConstant(RHSC >> Scale, dl, MVT::i64);
1349 return true;
1350 }
1351 } else {
1352 // unsigned Immediate
1353 uint64_t RHSC = RHS->getZExtValue();
1354 unsigned Scale = Log2_32(Size);
1355 uint64_t Range = 0x1ULL << BW;
1356
1357 if ((RHSC & (Size - 1)) == 0 && RHSC < (Range << Scale)) {
1358 Base = N.getOperand(0);
1359 if (Base.getOpcode() == ISD::FrameIndex) {
1360 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1361 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1362 }
1363 OffImm = CurDAG->getTargetConstant(RHSC >> Scale, dl, MVT::i64);
1364 return true;
1365 }
1366 }
1367 }
1368 }
1369 // Base only. The address will be materialized into a register before
1370 // the memory is accessed.
1371 // add x0, Xbase, #offset
1372 // stp x1, x2, [x0]
1373 Base = N;
1374 OffImm = CurDAG->getTargetConstant(0, dl, MVT::i64);
1375 return true;
1376}
1377
1378/// SelectAddrModeIndexed - Select a "register plus scaled unsigned 12-bit
1379/// immediate" address. The "Size" argument is the size in bytes of the memory
1380/// reference, which determines the scale.
1381bool AArch64DAGToDAGISel::SelectAddrModeIndexed(SDValue N, unsigned Size,
1382 SDValue &Base, SDValue &OffImm) {
1383 SDLoc dl(N);
1384 const DataLayout &DL = CurDAG->getDataLayout();
1385 const TargetLowering *TLI = getTargetLowering();
1386 if (N.getOpcode() == ISD::FrameIndex) {
1387 int FI = cast<FrameIndexSDNode>(N)->getIndex();
1388 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1389 OffImm = CurDAG->getTargetConstant(0, dl, MVT::i64);
1390 return true;
1391 }
1392
1393 if (N.getOpcode() == AArch64ISD::ADDlow && isWorthFoldingADDlow(N)) {
1394 GlobalAddressSDNode *GAN =
1395 dyn_cast<GlobalAddressSDNode>(N.getOperand(1).getNode());
1396 Base = N.getOperand(0);
1397 OffImm = N.getOperand(1);
1398 if (!GAN)
1399 return true;
1400
1401 if (GAN->getOffset() % Size == 0 &&
1403 return true;
1404 }
1405
1406 if (CurDAG->isBaseWithConstantOffset(N)) {
1407 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1408 int64_t RHSC = (int64_t)RHS->getZExtValue();
1409 unsigned Scale = Log2_32(Size);
1410 if (isValidAsScaledImmediate(RHSC, 0x1000, Size)) {
1411 Base = N.getOperand(0);
1412 if (Base.getOpcode() == ISD::FrameIndex) {
1413 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1414 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1415 }
1416 OffImm = CurDAG->getTargetConstant(RHSC >> Scale, dl, MVT::i64);
1417 return true;
1418 }
1419 }
1420 }
1421
1422 // Before falling back to our general case, check if the unscaled
1423 // instructions can handle this. If so, that's preferable.
1424 if (SelectAddrModeUnscaled(N, Size, Base, OffImm))
1425 return false;
1426
1427 // Base only. The address will be materialized into a register before
1428 // the memory is accessed.
1429 // add x0, Xbase, #offset
1430 // ldr x0, [x0]
1431 Base = N;
1432 OffImm = CurDAG->getTargetConstant(0, dl, MVT::i64);
1433 return true;
1434}
1435
1436/// SelectAddrModeUnscaled - Select a "register plus unscaled signed 9-bit
1437/// immediate" address. This should only match when there is an offset that
1438/// is not valid for a scaled immediate addressing mode. The "Size" argument
1439/// is the size in bytes of the memory reference, which is needed here to know
1440/// what is valid for a scaled immediate.
1441bool AArch64DAGToDAGISel::SelectAddrModeUnscaled(SDValue N, unsigned Size,
1442 SDValue &Base,
1443 SDValue &OffImm) {
1444 if (!CurDAG->isBaseWithConstantOffset(N))
1445 return false;
1446 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1447 int64_t RHSC = RHS->getSExtValue();
1448 if (RHSC >= -256 && RHSC < 256) {
1449 Base = N.getOperand(0);
1450 if (Base.getOpcode() == ISD::FrameIndex) {
1451 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1452 const TargetLowering *TLI = getTargetLowering();
1453 Base = CurDAG->getTargetFrameIndex(
1454 FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1455 }
1456 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i64);
1457 return true;
1458 }
1459 }
1460 return false;
1461}
1462
1464 SDLoc dl(N);
1465 SDValue ImpDef = SDValue(
1466 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, MVT::i64), 0);
1467 return CurDAG->getTargetInsertSubreg(AArch64::sub_32, dl, MVT::i64, ImpDef,
1468 N);
1469}
1470
1471/// Check if the given SHL node (\p N), can be used to form an
1472/// extended register for an addressing mode.
1473bool AArch64DAGToDAGISel::SelectExtendedSHL(SDValue N, unsigned Size,
1474 bool WantExtend, SDValue &Offset,
1475 SDValue &SignExtend) {
1476 assert(N.getOpcode() == ISD::SHL && "Invalid opcode.");
1477 ConstantSDNode *CSD = dyn_cast<ConstantSDNode>(N.getOperand(1));
1478 if (!CSD || (CSD->getZExtValue() & 0x7) != CSD->getZExtValue())
1479 return false;
1480
1481 SDLoc dl(N);
1482 if (WantExtend) {
1484 getExtendTypeForNode(N.getOperand(0), true);
1486 return false;
1487
1488 Offset = narrowIfNeeded(CurDAG, N.getOperand(0).getOperand(0));
1489 SignExtend = CurDAG->getTargetConstant(Ext == AArch64_AM::SXTW, dl,
1490 MVT::i32);
1491 } else {
1492 Offset = N.getOperand(0);
1493 SignExtend = CurDAG->getTargetConstant(0, dl, MVT::i32);
1494 }
1495
1496 unsigned LegalShiftVal = Log2_32(Size);
1497 unsigned ShiftVal = CSD->getZExtValue();
1498
1499 if (ShiftVal != 0 && ShiftVal != LegalShiftVal)
1500 return false;
1501
1502 return isWorthFoldingAddr(N, Size);
1503}
1504
1505bool AArch64DAGToDAGISel::SelectAddrModeWRO(SDValue N, unsigned Size,
1506 SDValue &Base, SDValue &Offset,
1507 SDValue &SignExtend,
1508 SDValue &DoShift) {
1509 if (N.getOpcode() != ISD::ADD)
1510 return false;
1511 SDValue LHS = N.getOperand(0);
1512 SDValue RHS = N.getOperand(1);
1513 SDLoc dl(N);
1514
1515 // We don't want to match immediate adds here, because they are better lowered
1516 // to the register-immediate addressing modes.
1518 return false;
1519
1520 // Check if this particular node is reused in any non-memory related
1521 // operation. If yes, do not try to fold this node into the address
1522 // computation, since the computation will be kept.
1523 const SDNode *Node = N.getNode();
1524 for (SDNode *UI : Node->users()) {
1525 if (!isMemOpOrPrefetch(UI))
1526 return false;
1527 }
1528
1529 // Remember if it is worth folding N when it produces extended register.
1530 bool IsExtendedRegisterWorthFolding = isWorthFoldingAddr(N, Size);
1531
1532 // Try to match a shifted extend on the RHS.
1533 if (IsExtendedRegisterWorthFolding && RHS.getOpcode() == ISD::SHL &&
1534 SelectExtendedSHL(RHS, Size, true, Offset, SignExtend)) {
1535 Base = LHS;
1536 DoShift = CurDAG->getTargetConstant(true, dl, MVT::i32);
1537 return true;
1538 }
1539
1540 // Try to match a shifted extend on the LHS.
1541 if (IsExtendedRegisterWorthFolding && LHS.getOpcode() == ISD::SHL &&
1542 SelectExtendedSHL(LHS, Size, true, Offset, SignExtend)) {
1543 Base = RHS;
1544 DoShift = CurDAG->getTargetConstant(true, dl, MVT::i32);
1545 return true;
1546 }
1547
1548 // There was no shift, whatever else we find.
1549 DoShift = CurDAG->getTargetConstant(false, dl, MVT::i32);
1550
1552 // Try to match an unshifted extend on the LHS.
1553 if (IsExtendedRegisterWorthFolding &&
1554 (Ext = getExtendTypeForNode(LHS, true)) !=
1556 Base = RHS;
1557 Offset = narrowIfNeeded(CurDAG, LHS.getOperand(0));
1558 SignExtend = CurDAG->getTargetConstant(Ext == AArch64_AM::SXTW, dl,
1559 MVT::i32);
1560 if (isWorthFoldingAddr(LHS, Size))
1561 return true;
1562 }
1563
1564 // Try to match an unshifted extend on the RHS.
1565 if (IsExtendedRegisterWorthFolding &&
1566 (Ext = getExtendTypeForNode(RHS, true)) !=
1568 Base = LHS;
1569 Offset = narrowIfNeeded(CurDAG, RHS.getOperand(0));
1570 SignExtend = CurDAG->getTargetConstant(Ext == AArch64_AM::SXTW, dl,
1571 MVT::i32);
1572 if (isWorthFoldingAddr(RHS, Size))
1573 return true;
1574 }
1575
1576 return false;
1577}
1578
1579// Check if the given immediate is preferred by ADD. If an immediate can be
1580// encoded in an ADD, or it can be encoded in an "ADD LSL #12" and can not be
1581// encoded by one MOVZ, return true.
1582static bool isPreferredADD(int64_t ImmOff) {
1583 // Constant in [0x0, 0xfff] can be encoded in ADD.
1584 if ((ImmOff & 0xfffffffffffff000LL) == 0x0LL)
1585 return true;
1586 // Check if it can be encoded in an "ADD LSL #12".
1587 if ((ImmOff & 0xffffffffff000fffLL) == 0x0LL)
1588 // As a single MOVZ is faster than a "ADD of LSL #12", ignore such constant.
1589 return (ImmOff & 0xffffffffff00ffffLL) != 0x0LL &&
1590 (ImmOff & 0xffffffffffff0fffLL) != 0x0LL;
1591 return false;
1592}
1593
1594bool AArch64DAGToDAGISel::SelectAddrModeXRO(SDValue N, unsigned Size,
1595 SDValue &Base, SDValue &Offset,
1596 SDValue &SignExtend,
1597 SDValue &DoShift) {
1598 if (N.getOpcode() != ISD::ADD)
1599 return false;
1600 SDValue LHS = N.getOperand(0);
1601 SDValue RHS = N.getOperand(1);
1602 SDLoc DL(N);
1603
1604 // Check if this particular node is reused in any non-memory related
1605 // operation. If yes, do not try to fold this node into the address
1606 // computation, since the computation will be kept.
1607 const SDNode *Node = N.getNode();
1608 for (SDNode *UI : Node->users()) {
1609 if (!isMemOpOrPrefetch(UI))
1610 return false;
1611 }
1612
1613 // Watch out if RHS is a wide immediate, it can not be selected into
1614 // [BaseReg+Imm] addressing mode. Also it may not be able to be encoded into
1615 // ADD/SUB. Instead it will use [BaseReg + 0] address mode and generate
1616 // instructions like:
1617 // MOV X0, WideImmediate
1618 // ADD X1, BaseReg, X0
1619 // LDR X2, [X1, 0]
1620 // For such situation, using [BaseReg, XReg] addressing mode can save one
1621 // ADD/SUB:
1622 // MOV X0, WideImmediate
1623 // LDR X2, [BaseReg, X0]
1624 if (isa<ConstantSDNode>(RHS)) {
1625 int64_t ImmOff = (int64_t)RHS->getAsZExtVal();
1626 // Skip the immediate can be selected by load/store addressing mode.
1627 // Also skip the immediate can be encoded by a single ADD (SUB is also
1628 // checked by using -ImmOff).
1629 if (isValidAsScaledImmediate(ImmOff, 0x1000, Size) ||
1630 isPreferredADD(ImmOff) || isPreferredADD(-ImmOff))
1631 return false;
1632
1633 SDValue Ops[] = { RHS };
1634 SDNode *MOVI =
1635 CurDAG->getMachineNode(AArch64::MOVi64imm, DL, MVT::i64, Ops);
1636 SDValue MOVIV = SDValue(MOVI, 0);
1637 // This ADD of two X register will be selected into [Reg+Reg] mode.
1638 N = CurDAG->getNode(ISD::ADD, DL, MVT::i64, LHS, MOVIV);
1639 }
1640
1641 // Remember if it is worth folding N when it produces extended register.
1642 bool IsExtendedRegisterWorthFolding = isWorthFoldingAddr(N, Size);
1643
1644 // Try to match a shifted extend on the RHS.
1645 if (IsExtendedRegisterWorthFolding && RHS.getOpcode() == ISD::SHL &&
1646 SelectExtendedSHL(RHS, Size, false, Offset, SignExtend)) {
1647 Base = LHS;
1648 DoShift = CurDAG->getTargetConstant(true, DL, MVT::i32);
1649 return true;
1650 }
1651
1652 // Try to match a shifted extend on the LHS.
1653 if (IsExtendedRegisterWorthFolding && LHS.getOpcode() == ISD::SHL &&
1654 SelectExtendedSHL(LHS, Size, false, Offset, SignExtend)) {
1655 Base = RHS;
1656 DoShift = CurDAG->getTargetConstant(true, DL, MVT::i32);
1657 return true;
1658 }
1659
1660 // Match any non-shifted, non-extend, non-immediate add expression.
1661 Base = LHS;
1662 Offset = RHS;
1663 SignExtend = CurDAG->getTargetConstant(false, DL, MVT::i32);
1664 DoShift = CurDAG->getTargetConstant(false, DL, MVT::i32);
1665 // Reg1 + Reg2 is free: no check needed.
1666 return true;
1667}
1668
1669SDValue AArch64DAGToDAGISel::createDTuple(ArrayRef<SDValue> Regs) {
1670 static const unsigned RegClassIDs[] = {
1671 AArch64::DDRegClassID, AArch64::DDDRegClassID, AArch64::DDDDRegClassID};
1672 static const unsigned SubRegs[] = {AArch64::dsub0, AArch64::dsub1,
1673 AArch64::dsub2, AArch64::dsub3};
1674
1675 return createTuple(Regs, RegClassIDs, SubRegs);
1676}
1677
1678SDValue AArch64DAGToDAGISel::createQTuple(ArrayRef<SDValue> Regs) {
1679 static const unsigned RegClassIDs[] = {
1680 AArch64::QQRegClassID, AArch64::QQQRegClassID, AArch64::QQQQRegClassID};
1681 static const unsigned SubRegs[] = {AArch64::qsub0, AArch64::qsub1,
1682 AArch64::qsub2, AArch64::qsub3};
1683
1684 return createTuple(Regs, RegClassIDs, SubRegs);
1685}
1686
1687SDValue AArch64DAGToDAGISel::createZTuple(ArrayRef<SDValue> Regs) {
1688 static const unsigned RegClassIDs[] = {AArch64::ZPR2RegClassID,
1689 AArch64::ZPR3RegClassID,
1690 AArch64::ZPR4RegClassID};
1691 static const unsigned SubRegs[] = {AArch64::zsub0, AArch64::zsub1,
1692 AArch64::zsub2, AArch64::zsub3};
1693
1694 return createTuple(Regs, RegClassIDs, SubRegs);
1695}
1696
1697SDValue AArch64DAGToDAGISel::createZMulTuple(ArrayRef<SDValue> Regs) {
1698 assert(Regs.size() == 2 || Regs.size() == 4);
1699
1700 // The createTuple interface requires 3 RegClassIDs for each possible
1701 // tuple type even though we only have them for ZPR2 and ZPR4.
1702 static const unsigned RegClassIDs[] = {AArch64::ZPR2Mul2RegClassID, 0,
1703 AArch64::ZPR4Mul4RegClassID};
1704 static const unsigned SubRegs[] = {AArch64::zsub0, AArch64::zsub1,
1705 AArch64::zsub2, AArch64::zsub3};
1706 return createTuple(Regs, RegClassIDs, SubRegs);
1707}
1708
1709SDValue AArch64DAGToDAGISel::createTuple(ArrayRef<SDValue> Regs,
1710 const unsigned RegClassIDs[],
1711 const unsigned SubRegs[]) {
1712 // There's no special register-class for a vector-list of 1 element: it's just
1713 // a vector.
1714 if (Regs.size() == 1)
1715 return Regs[0];
1716
1717 assert(Regs.size() >= 2 && Regs.size() <= 4);
1718
1719 SDLoc DL(Regs[0]);
1720
1722
1723 // First operand of REG_SEQUENCE is the desired RegClass.
1724 Ops.push_back(
1725 CurDAG->getTargetConstant(RegClassIDs[Regs.size() - 2], DL, MVT::i32));
1726
1727 // Then we get pairs of source & subregister-position for the components.
1728 for (unsigned i = 0; i < Regs.size(); ++i) {
1729 Ops.push_back(Regs[i]);
1730 Ops.push_back(CurDAG->getTargetConstant(SubRegs[i], DL, MVT::i32));
1731 }
1732
1733 SDNode *N =
1734 CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, DL, MVT::Untyped, Ops);
1735 return SDValue(N, 0);
1736}
1737
1738void AArch64DAGToDAGISel::SelectTable(SDNode *N, unsigned NumVecs, unsigned Opc,
1739 bool isExt) {
1740 SDLoc dl(N);
1741 EVT VT = N->getValueType(0);
1742
1743 unsigned ExtOff = isExt;
1744
1745 // Form a REG_SEQUENCE to force register allocation.
1746 unsigned Vec0Off = ExtOff + 1;
1747 SmallVector<SDValue, 4> Regs(N->ops().slice(Vec0Off, NumVecs));
1748 SDValue RegSeq = createQTuple(Regs);
1749
1751 if (isExt)
1752 Ops.push_back(N->getOperand(1));
1753 Ops.push_back(RegSeq);
1754 Ops.push_back(N->getOperand(NumVecs + ExtOff + 1));
1755 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, VT, Ops));
1756}
1757
1758static std::tuple<SDValue, SDValue>
1760 SDLoc DL(Disc);
1761 SDValue AddrDisc;
1762 SDValue ConstDisc;
1763
1764 // If this is a blend, remember the constant and address discriminators.
1765 // Otherwise, it's either a constant discriminator, or a non-blended
1766 // address discriminator.
1767 if (Disc->getOpcode() == ISD::INTRINSIC_WO_CHAIN &&
1768 Disc->getConstantOperandVal(0) == Intrinsic::ptrauth_blend) {
1769 AddrDisc = Disc->getOperand(1);
1770 ConstDisc = Disc->getOperand(2);
1771 } else {
1772 ConstDisc = Disc;
1773 }
1774
1775 // If the constant discriminator (either the blend RHS, or the entire
1776 // discriminator value) isn't a 16-bit constant, bail out, and let the
1777 // discriminator be computed separately.
1778 auto *ConstDiscN = dyn_cast<ConstantSDNode>(ConstDisc);
1779 if (!ConstDiscN || !isUInt<16>(ConstDiscN->getZExtValue()))
1780 return std::make_tuple(DAG->getTargetConstant(0, DL, MVT::i64), Disc);
1781
1782 // If there's no address discriminator, use XZR directly.
1783 if (!AddrDisc)
1784 AddrDisc = DAG->getRegister(AArch64::XZR, MVT::i64);
1785
1786 return std::make_tuple(
1787 DAG->getTargetConstant(ConstDiscN->getZExtValue(), DL, MVT::i64),
1788 AddrDisc);
1789}
1790
1791void AArch64DAGToDAGISel::SelectPtrauthAuth(SDNode *N) {
1792 SDLoc DL(N);
1793 // IntrinsicID is operand #0
1794 SDValue Val = N->getOperand(1);
1795 SDValue AUTKey = N->getOperand(2);
1796 SDValue AUTDisc = N->getOperand(3);
1797
1798 unsigned AUTKeyC = cast<ConstantSDNode>(AUTKey)->getZExtValue();
1799 AUTKey = CurDAG->getTargetConstant(AUTKeyC, DL, MVT::i64);
1800
1801 SDValue AUTAddrDisc, AUTConstDisc;
1802 std::tie(AUTConstDisc, AUTAddrDisc) =
1803 extractPtrauthBlendDiscriminators(AUTDisc, CurDAG);
1804
1805 if (!Subtarget->isX16X17Safer()) {
1806 std::vector<SDValue> Ops = {Val, AUTKey, AUTConstDisc, AUTAddrDisc};
1807 // Copy deactivation symbol if present.
1808 if (N->getNumOperands() > 4)
1809 Ops.push_back(N->getOperand(4));
1810
1811 SDNode *AUT =
1812 CurDAG->getMachineNode(AArch64::AUTxMxN, DL, MVT::i64, MVT::i64, Ops);
1813 ReplaceNode(N, AUT);
1814 } else {
1815 SDValue X16Copy = CurDAG->getCopyToReg(CurDAG->getEntryNode(), DL,
1816 AArch64::X16, Val, SDValue());
1817 SDValue Ops[] = {AUTKey, AUTConstDisc, AUTAddrDisc, X16Copy.getValue(1)};
1818
1819 SDNode *AUT = CurDAG->getMachineNode(AArch64::AUTx16x17, DL, MVT::i64, Ops);
1820 ReplaceNode(N, AUT);
1821 }
1822}
1823
1824void AArch64DAGToDAGISel::SelectPtrauthResign(SDNode *N) {
1825 SDLoc DL(N);
1826 // IntrinsicID is operand #0, if W_CHAIN it is #1
1827 int OffsetBase = N->getOpcode() == ISD::INTRINSIC_W_CHAIN ? 1 : 0;
1828 SDValue Val = N->getOperand(OffsetBase + 1);
1829 SDValue AUTKey = N->getOperand(OffsetBase + 2);
1830 SDValue AUTDisc = N->getOperand(OffsetBase + 3);
1831 SDValue PACKey = N->getOperand(OffsetBase + 4);
1832 SDValue PACDisc = N->getOperand(OffsetBase + 5);
1833 uint32_t IntNum = N->getConstantOperandVal(OffsetBase + 0);
1834 bool HasLoad = IntNum == Intrinsic::ptrauth_resign_load_relative;
1835
1836 unsigned AUTKeyC = cast<ConstantSDNode>(AUTKey)->getZExtValue();
1837 unsigned PACKeyC = cast<ConstantSDNode>(PACKey)->getZExtValue();
1838
1839 AUTKey = CurDAG->getTargetConstant(AUTKeyC, DL, MVT::i64);
1840 PACKey = CurDAG->getTargetConstant(PACKeyC, DL, MVT::i64);
1841
1842 SDValue AUTAddrDisc, AUTConstDisc;
1843 std::tie(AUTConstDisc, AUTAddrDisc) =
1844 extractPtrauthBlendDiscriminators(AUTDisc, CurDAG);
1845
1846 SDValue PACAddrDisc, PACConstDisc;
1847 std::tie(PACConstDisc, PACAddrDisc) =
1848 extractPtrauthBlendDiscriminators(PACDisc, CurDAG);
1849
1850 SDValue X16Copy = CurDAG->getCopyToReg(CurDAG->getEntryNode(), DL,
1851 AArch64::X16, Val, SDValue());
1852
1853 if (HasLoad) {
1854 SDValue Addend = N->getOperand(OffsetBase + 6);
1855 SDValue IncomingChain = N->getOperand(0);
1856 SDValue Ops[] = {AUTKey, AUTConstDisc, AUTAddrDisc,
1857 PACKey, PACConstDisc, PACAddrDisc,
1858 Addend, IncomingChain, X16Copy.getValue(1)};
1859
1860 SDNode *AUTRELLOADPAC = CurDAG->getMachineNode(AArch64::AUTRELLOADPAC, DL,
1861 MVT::i64, MVT::Other, Ops);
1862 ReplaceNode(N, AUTRELLOADPAC);
1863 } else {
1864 SDValue Ops[] = {AUTKey, AUTConstDisc, AUTAddrDisc, PACKey,
1865 PACConstDisc, PACAddrDisc, X16Copy.getValue(1)};
1866
1867 SDNode *AUTPAC = CurDAG->getMachineNode(AArch64::AUTPAC, DL, MVT::i64, Ops);
1868 ReplaceNode(N, AUTPAC);
1869 }
1870}
1871
1872void AArch64DAGToDAGISel::SelectPtrauthResignWithPC(SDNode *N) {
1873 SDLoc DL(N);
1874 SDValue Val = N->getOperand(1);
1875 SDValue AUTKey = N->getOperand(2);
1876 SDValue AUTDisc = N->getOperand(3);
1877 SDValue AUTPC = N->getOperand(4);
1878 SDValue PACKey = N->getOperand(5);
1879 SDValue PACDisc = N->getOperand(6);
1880
1881 unsigned AUTKeyC = cast<ConstantSDNode>(AUTKey)->getZExtValue();
1882 unsigned PACKeyC = cast<ConstantSDNode>(PACKey)->getZExtValue();
1883
1884 AUTKey = CurDAG->getTargetConstant(AUTKeyC, DL, MVT::i64);
1885 PACKey = CurDAG->getTargetConstant(PACKeyC, DL, MVT::i64);
1886
1887 SDValue PACAddrDisc, PACConstDisc;
1888 std::tie(PACConstDisc, PACAddrDisc) =
1889 extractPtrauthBlendDiscriminators(PACDisc, CurDAG);
1890
1891 SDValue X17Copy = CurDAG->getCopyToReg(CurDAG->getEntryNode(), DL,
1892 AArch64::X17, Val, SDValue());
1893 SDValue X16Copy = CurDAG->getCopyToReg(
1894 CurDAG->getEntryNode(), DL, AArch64::X16, AUTDisc, X17Copy.getValue(1));
1895 SDValue X15Copy = CurDAG->getCopyToReg(
1896 CurDAG->getEntryNode(), DL, AArch64::X15, AUTPC, X16Copy.getValue(1));
1897
1898 SDValue Ops[] = {AUTKey, PACKey, PACConstDisc, PACAddrDisc,
1899 X15Copy.getValue(1)};
1900 SDNode *AUTPCPAC =
1901 CurDAG->getMachineNode(AArch64::AUTPCPAC, DL, MVT::i64, Ops);
1902 ReplaceNode(N, AUTPCPAC);
1903}
1904
1905bool AArch64DAGToDAGISel::tryIndexedLoad(SDNode *N) {
1906 LoadSDNode *LD = cast<LoadSDNode>(N);
1907 if (LD->isUnindexed())
1908 return false;
1909 EVT VT = LD->getMemoryVT();
1910 EVT DstVT = N->getValueType(0);
1911 ISD::MemIndexedMode AM = LD->getAddressingMode();
1912 bool IsPre = AM == ISD::PRE_INC || AM == ISD::PRE_DEC;
1913 ConstantSDNode *OffsetOp = cast<ConstantSDNode>(LD->getOffset());
1914 int OffsetVal = (int)OffsetOp->getZExtValue();
1915
1916 // We're not doing validity checking here. That was done when checking
1917 // if we should mark the load as indexed or not. We're just selecting
1918 // the right instruction.
1919 unsigned Opcode = 0;
1920
1921 ISD::LoadExtType ExtType = LD->getExtensionType();
1922 bool InsertTo64 = false;
1923 bool UseLd1 =
1924 (VT.is64BitVector() || VT.is128BitVector()) &&
1925 (!Subtarget->isLittleEndian() || (Subtarget->requiresStrictAlign() &&
1926 LD->getAlign() < VT.getStoreSize()));
1927 if (VT == MVT::i64)
1928 Opcode = IsPre ? AArch64::LDRXpre : AArch64::LDRXpost;
1929 else if (VT == MVT::i32) {
1930 if (ExtType == ISD::NON_EXTLOAD)
1931 Opcode = IsPre ? AArch64::LDRWpre : AArch64::LDRWpost;
1932 else if (ExtType == ISD::SEXTLOAD)
1933 Opcode = IsPre ? AArch64::LDRSWpre : AArch64::LDRSWpost;
1934 else {
1935 Opcode = IsPre ? AArch64::LDRWpre : AArch64::LDRWpost;
1936 InsertTo64 = true;
1937 // The result of the load is only i32. It's the subreg_to_reg that makes
1938 // it into an i64.
1939 DstVT = MVT::i32;
1940 }
1941 } else if (VT == MVT::i16) {
1942 if (ExtType == ISD::SEXTLOAD) {
1943 if (DstVT == MVT::i64)
1944 Opcode = IsPre ? AArch64::LDRSHXpre : AArch64::LDRSHXpost;
1945 else
1946 Opcode = IsPre ? AArch64::LDRSHWpre : AArch64::LDRSHWpost;
1947 } else {
1948 Opcode = IsPre ? AArch64::LDRHHpre : AArch64::LDRHHpost;
1949 InsertTo64 = DstVT == MVT::i64;
1950 // The result of the load is only i32. It's the subreg_to_reg that makes
1951 // it into an i64.
1952 DstVT = MVT::i32;
1953 }
1954 } else if (VT == MVT::i8) {
1955 if (ExtType == ISD::SEXTLOAD) {
1956 if (DstVT == MVT::i64)
1957 Opcode = IsPre ? AArch64::LDRSBXpre : AArch64::LDRSBXpost;
1958 else
1959 Opcode = IsPre ? AArch64::LDRSBWpre : AArch64::LDRSBWpost;
1960 } else {
1961 Opcode = IsPre ? AArch64::LDRBBpre : AArch64::LDRBBpost;
1962 InsertTo64 = DstVT == MVT::i64;
1963 // The result of the load is only i32. It's the subreg_to_reg that makes
1964 // it into an i64.
1965 DstVT = MVT::i32;
1966 }
1967 } else if (VT == MVT::f16) {
1968 Opcode = IsPre ? AArch64::LDRHpre : AArch64::LDRHpost;
1969 } else if (VT == MVT::bf16) {
1970 Opcode = IsPre ? AArch64::LDRHpre : AArch64::LDRHpost;
1971 } else if (VT == MVT::f32) {
1972 Opcode = IsPre ? AArch64::LDRSpre : AArch64::LDRSpost;
1973 } else if (VT == MVT::f64 || (VT.is64BitVector() && !UseLd1)) {
1974 Opcode = IsPre ? AArch64::LDRDpre : AArch64::LDRDpost;
1975 } else if (VT.is128BitVector() && !UseLd1) {
1976 Opcode = IsPre ? AArch64::LDRQpre : AArch64::LDRQpost;
1977 } else if (VT.is64BitVector() && UseLd1) {
1978 if (IsPre || OffsetVal != 8)
1979 return false;
1980 switch (VT.getScalarSizeInBits()) {
1981 case 8:
1982 Opcode = AArch64::LD1Onev8b_POST;
1983 break;
1984 case 16:
1985 Opcode = AArch64::LD1Onev4h_POST;
1986 break;
1987 case 32:
1988 Opcode = AArch64::LD1Onev2s_POST;
1989 break;
1990 case 64:
1991 Opcode = AArch64::LD1Onev1d_POST;
1992 break;
1993 default:
1994 llvm_unreachable("Expected vector element to be a power of 2");
1995 }
1996 } else if (VT.is128BitVector() && UseLd1) {
1997 if (IsPre || OffsetVal != 16)
1998 return false;
1999 switch (VT.getScalarSizeInBits()) {
2000 case 8:
2001 Opcode = AArch64::LD1Onev16b_POST;
2002 break;
2003 case 16:
2004 Opcode = AArch64::LD1Onev8h_POST;
2005 break;
2006 case 32:
2007 Opcode = AArch64::LD1Onev4s_POST;
2008 break;
2009 case 64:
2010 Opcode = AArch64::LD1Onev2d_POST;
2011 break;
2012 default:
2013 llvm_unreachable("Expected vector element to be a power of 2");
2014 }
2015 } else
2016 return false;
2017 SDValue Chain = LD->getChain();
2018 SDValue Base = LD->getBasePtr();
2019 SDLoc dl(N);
2020 // LD1 encodes an immediate offset by using XZR as the offset register.
2021 SDValue Offset = UseLd1 ? CurDAG->getRegister(AArch64::XZR, MVT::i64)
2022 : CurDAG->getTargetConstant(OffsetVal, dl, MVT::i64);
2023 SDValue Ops[] = { Base, Offset, Chain };
2024 SDNode *Res = CurDAG->getMachineNode(Opcode, dl, MVT::i64, DstVT,
2025 MVT::Other, Ops);
2026
2027 // Transfer memoperands.
2028 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand();
2029 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Res), {MemOp});
2030
2031 // Either way, we're replacing the node, so tell the caller that.
2032 SDValue LoadedVal = SDValue(Res, 1);
2033 if (InsertTo64) {
2034 SDValue SubReg = CurDAG->getTargetConstant(AArch64::sub_32, dl, MVT::i32);
2035 LoadedVal = SDValue(CurDAG->getMachineNode(AArch64::SUBREG_TO_REG, dl,
2036 MVT::i64, LoadedVal, SubReg),
2037 0);
2038 }
2039
2040 ReplaceUses(SDValue(N, 0), LoadedVal);
2041 ReplaceUses(SDValue(N, 1), SDValue(Res, 0));
2042 ReplaceUses(SDValue(N, 2), SDValue(Res, 2));
2043 CurDAG->RemoveDeadNode(N);
2044 return true;
2045}
2046
2047void AArch64DAGToDAGISel::SelectLoad(SDNode *N, unsigned NumVecs, unsigned Opc,
2048 unsigned SubRegIdx) {
2049 SDLoc dl(N);
2050 EVT VT = N->getValueType(0);
2051 SDValue Chain = N->getOperand(0);
2052
2053 SDValue Ops[] = {N->getOperand(2), // Mem operand;
2054 Chain};
2055
2056 const EVT ResTys[] = {MVT::Untyped, MVT::Other};
2057
2058 SDNode *Ld = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2059 SDValue SuperReg = SDValue(Ld, 0);
2060 for (unsigned i = 0; i < NumVecs; ++i)
2061 ReplaceUses(SDValue(N, i),
2062 CurDAG->getTargetExtractSubreg(SubRegIdx + i, dl, VT, SuperReg));
2063
2064 ReplaceUses(SDValue(N, NumVecs), SDValue(Ld, 1));
2065
2066 // Transfer memoperands. In the case of AArch64::LD64B, there won't be one,
2067 // because it's too simple to have needed special treatment during lowering.
2068 if (auto *MemIntr = dyn_cast<MemIntrinsicSDNode>(N)) {
2069 MachineMemOperand *MemOp = MemIntr->getMemOperand();
2070 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Ld), {MemOp});
2071 }
2072
2073 CurDAG->RemoveDeadNode(N);
2074}
2075
2076void AArch64DAGToDAGISel::SelectPostLoad(SDNode *N, unsigned NumVecs,
2077 unsigned Opc, unsigned SubRegIdx) {
2078 SDLoc dl(N);
2079 EVT VT = N->getValueType(0);
2080 SDValue Chain = N->getOperand(0);
2081
2082 SDValue Ops[] = {N->getOperand(1), // Mem operand
2083 N->getOperand(2), // Incremental
2084 Chain};
2085
2086 const EVT ResTys[] = {MVT::i64, // Type of the write back register
2087 MVT::Untyped, MVT::Other};
2088
2089 SDNode *Ld = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2090
2091 // Update uses of write back register
2092 ReplaceUses(SDValue(N, NumVecs), SDValue(Ld, 0));
2093
2094 // Update uses of vector list
2095 SDValue SuperReg = SDValue(Ld, 1);
2096 if (NumVecs == 1)
2097 ReplaceUses(SDValue(N, 0), SuperReg);
2098 else
2099 for (unsigned i = 0; i < NumVecs; ++i)
2100 ReplaceUses(SDValue(N, i),
2101 CurDAG->getTargetExtractSubreg(SubRegIdx + i, dl, VT, SuperReg));
2102
2103 // Transfer memoperands.
2104 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2105 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Ld), {MemOp});
2106
2107 // Update the chain
2108 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(Ld, 2));
2109 CurDAG->RemoveDeadNode(N);
2110}
2111
2112/// Optimize \param OldBase and \param OldOffset selecting the best addressing
2113/// mode. Returns a tuple consisting of an Opcode, an SDValue representing the
2114/// new Base and an SDValue representing the new offset.
2115std::tuple<unsigned, SDValue, SDValue>
2116AArch64DAGToDAGISel::findAddrModeSVELoadStore(SDNode *N, unsigned Opc_rr,
2117 unsigned Opc_ri,
2118 const SDValue &OldBase,
2119 const SDValue &OldOffset,
2120 unsigned Scale) {
2121 SDValue NewBase = OldBase;
2122 SDValue NewOffset = OldOffset;
2123 // Detect a possible Reg+Imm addressing mode.
2124 const bool IsRegImm = SelectAddrModeIndexedSVE</*Min=*/-8, /*Max=*/7>(
2125 N, OldBase, NewBase, NewOffset);
2126
2127 // Detect a possible reg+reg addressing mode, but only if we haven't already
2128 // detected a Reg+Imm one.
2129 const bool IsRegReg =
2130 !IsRegImm && SelectSVERegRegAddrMode(OldBase, Scale, NewBase, NewOffset);
2131
2132 // Select the instruction.
2133 return std::make_tuple(IsRegReg ? Opc_rr : Opc_ri, NewBase, NewOffset);
2134}
2135
2136enum class SelectTypeKind {
2137 Int1 = 0,
2138 Int = 1,
2139 FP = 2,
2141};
2142
2143/// This function selects an opcode from a list of opcodes, which is
2144/// expected to be the opcode for { 8-bit, 16-bit, 32-bit, 64-bit }
2145/// element types, in this order.
2146template <SelectTypeKind Kind>
2147static unsigned SelectOpcodeFromVT(EVT VT, ArrayRef<unsigned> Opcodes) {
2148 // Only match scalable vector VTs
2149 if (!VT.isScalableVector())
2150 return 0;
2151
2152 EVT EltVT = VT.getVectorElementType();
2153 unsigned Key = VT.getVectorMinNumElements();
2154 switch (Kind) {
2156 break;
2158 if (EltVT != MVT::i8 && EltVT != MVT::i16 && EltVT != MVT::i32 &&
2159 EltVT != MVT::i64)
2160 return 0;
2161 break;
2163 if (EltVT != MVT::i1)
2164 return 0;
2165 break;
2166 case SelectTypeKind::FP:
2167 if (EltVT == MVT::bf16)
2168 Key = 16;
2169 else if (EltVT != MVT::bf16 && EltVT != MVT::f16 && EltVT != MVT::f32 &&
2170 EltVT != MVT::f64)
2171 return 0;
2172 break;
2173 }
2174
2175 unsigned Offset;
2176 switch (Key) {
2177 case 16: // 8-bit or bf16
2178 Offset = 0;
2179 break;
2180 case 8: // 16-bit
2181 Offset = 1;
2182 break;
2183 case 4: // 32-bit
2184 Offset = 2;
2185 break;
2186 case 2: // 64-bit
2187 Offset = 3;
2188 break;
2189 default:
2190 return 0;
2191 }
2192
2193 return (Opcodes.size() <= Offset) ? 0 : Opcodes[Offset];
2194}
2195
2196// This function is almost identical to SelectWhilePair, but has an
2197// extra check on the range of the immediate operand.
2198// TODO: Merge these two functions together at some point?
2199void AArch64DAGToDAGISel::SelectPExtPair(SDNode *N, unsigned Opc) {
2200 // Immediate can be either 0 or 1.
2201 if (ConstantSDNode *Imm = dyn_cast<ConstantSDNode>(N->getOperand(2)))
2202 if (Imm->getZExtValue() > 1)
2203 return;
2204
2205 SDLoc DL(N);
2206 EVT VT = N->getValueType(0);
2207 SDValue Ops[] = {N->getOperand(1), N->getOperand(2)};
2208 SDNode *WhilePair = CurDAG->getMachineNode(Opc, DL, MVT::Untyped, Ops);
2209 SDValue SuperReg = SDValue(WhilePair, 0);
2210
2211 for (unsigned I = 0; I < 2; ++I)
2212 ReplaceUses(SDValue(N, I), CurDAG->getTargetExtractSubreg(
2213 AArch64::psub0 + I, DL, VT, SuperReg));
2214
2215 CurDAG->RemoveDeadNode(N);
2216}
2217
2218void AArch64DAGToDAGISel::SelectWhilePair(SDNode *N, unsigned Opc) {
2219 SDLoc DL(N);
2220 EVT VT = N->getValueType(0);
2221
2222 SDValue Ops[] = {N->getOperand(1), N->getOperand(2)};
2223
2224 SDNode *WhilePair = CurDAG->getMachineNode(Opc, DL, MVT::Untyped, Ops);
2225 SDValue SuperReg = SDValue(WhilePair, 0);
2226
2227 for (unsigned I = 0; I < 2; ++I)
2228 ReplaceUses(SDValue(N, I), CurDAG->getTargetExtractSubreg(
2229 AArch64::psub0 + I, DL, VT, SuperReg));
2230
2231 CurDAG->RemoveDeadNode(N);
2232}
2233
2234void AArch64DAGToDAGISel::SelectCVTIntrinsic(SDNode *N, unsigned NumVecs,
2235 unsigned Opcode) {
2236 EVT VT = N->getValueType(0);
2237 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2238 SDValue Ops = createZTuple(Regs);
2239 SDLoc DL(N);
2240 SDNode *Intrinsic = CurDAG->getMachineNode(Opcode, DL, MVT::Untyped, Ops);
2241 SDValue SuperReg = SDValue(Intrinsic, 0);
2242 for (unsigned i = 0; i < NumVecs; ++i)
2243 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2244 AArch64::zsub0 + i, DL, VT, SuperReg));
2245
2246 CurDAG->RemoveDeadNode(N);
2247}
2248
2249void AArch64DAGToDAGISel::SelectCVTIntrinsicFP8(SDNode *N, unsigned NumVecs,
2250 unsigned Opcode) {
2251 SDLoc DL(N);
2252 EVT VT = N->getValueType(0);
2253 SmallVector<SDValue, 4> Ops(N->op_begin() + 2, N->op_end());
2254 Ops.push_back(/*Chain*/ N->getOperand(0));
2255
2256 SDNode *Instruction =
2257 CurDAG->getMachineNode(Opcode, DL, {MVT::Untyped, MVT::Other}, Ops);
2258 SDValue SuperReg = SDValue(Instruction, 0);
2259
2260 for (unsigned i = 0; i < NumVecs; ++i)
2261 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2262 AArch64::zsub0 + i, DL, VT, SuperReg));
2263
2264 // Copy chain
2265 unsigned ChainIdx = NumVecs;
2266 ReplaceUses(SDValue(N, ChainIdx), SDValue(Instruction, 1));
2267 CurDAG->RemoveDeadNode(N);
2268}
2269
2270void AArch64DAGToDAGISel::SelectDestructiveMultiIntrinsic(SDNode *N,
2271 unsigned NumVecs,
2272 bool IsZmMulti,
2273 unsigned Opcode,
2274 bool HasPred) {
2275 assert(Opcode != 0 && "Unexpected opcode");
2276
2277 SDLoc DL(N);
2278 EVT VT = N->getValueType(0);
2279 SDUse *OpsIter = N->op_begin() + 1; // Skip intrinsic ID
2281
2282 auto GetMultiVecOperand = [&]() {
2283 SmallVector<SDValue, 4> Regs(OpsIter, OpsIter + NumVecs);
2284 OpsIter += NumVecs;
2285 return createZMulTuple(Regs);
2286 };
2287
2288 if (HasPred)
2289 Ops.push_back(*OpsIter++);
2290
2291 Ops.push_back(GetMultiVecOperand());
2292 if (IsZmMulti)
2293 Ops.push_back(GetMultiVecOperand());
2294 else
2295 Ops.push_back(*OpsIter++);
2296
2297 // Append any remaining operands.
2298 Ops.append(OpsIter, N->op_end());
2299 SDNode *Intrinsic;
2300 Intrinsic = CurDAG->getMachineNode(Opcode, DL, MVT::Untyped, Ops);
2301 SDValue SuperReg = SDValue(Intrinsic, 0);
2302 for (unsigned i = 0; i < NumVecs; ++i)
2303 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2304 AArch64::zsub0 + i, DL, VT, SuperReg));
2305
2306 CurDAG->RemoveDeadNode(N);
2307}
2308
2309void AArch64DAGToDAGISel::SelectPredicatedLoad(SDNode *N, unsigned NumVecs,
2310 unsigned Scale, unsigned Opc_ri,
2311 unsigned Opc_rr, bool IsIntr) {
2312 assert(Scale < 5 && "Invalid scaling value.");
2313 SDLoc DL(N);
2314 EVT VT = N->getValueType(0);
2315 SDValue Chain = N->getOperand(0);
2316
2317 // Optimize addressing mode.
2318 SDValue Base, Offset;
2319 unsigned Opc;
2320 std::tie(Opc, Base, Offset) = findAddrModeSVELoadStore(
2321 N, Opc_rr, Opc_ri, N->getOperand(IsIntr ? 3 : 2),
2322 CurDAG->getTargetConstant(0, DL, MVT::i64), Scale);
2323
2324 SDValue Ops[] = {N->getOperand(IsIntr ? 2 : 1), // Predicate
2325 Base, // Memory operand
2326 Offset, Chain};
2327
2328 const EVT ResTys[] = {MVT::Untyped, MVT::Other};
2329
2330 SDNode *Load = CurDAG->getMachineNode(Opc, DL, ResTys, Ops);
2331 SDValue SuperReg = SDValue(Load, 0);
2332 for (unsigned i = 0; i < NumVecs; ++i)
2333 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2334 AArch64::zsub0 + i, DL, VT, SuperReg));
2335
2336 // Copy chain
2337 unsigned ChainIdx = NumVecs;
2338 ReplaceUses(SDValue(N, ChainIdx), SDValue(Load, 1));
2339 CurDAG->RemoveDeadNode(N);
2340}
2341
2342void AArch64DAGToDAGISel::SelectContiguousMultiVectorLoad(SDNode *N,
2343 unsigned NumVecs,
2344 unsigned Scale,
2345 unsigned Opc_ri,
2346 unsigned Opc_rr) {
2347 assert(Scale < 4 && "Invalid scaling value.");
2348 SDLoc DL(N);
2349 EVT VT = N->getValueType(0);
2350 SDValue Chain = N->getOperand(0);
2351
2352 SDValue PNg = N->getOperand(2);
2353 SDValue Base = N->getOperand(3);
2354 SDValue Offset = CurDAG->getTargetConstant(0, DL, MVT::i64);
2355 unsigned Opc;
2356 std::tie(Opc, Base, Offset) =
2357 findAddrModeSVELoadStore(N, Opc_rr, Opc_ri, Base, Offset, Scale);
2358
2359 SDValue Ops[] = {PNg, // Predicate-as-counter
2360 Base, // Memory operand
2361 Offset, Chain};
2362
2363 const EVT ResTys[] = {MVT::Untyped, MVT::Other};
2364
2365 SDNode *Load = CurDAG->getMachineNode(Opc, DL, ResTys, Ops);
2366 SDValue SuperReg = SDValue(Load, 0);
2367 for (unsigned i = 0; i < NumVecs; ++i)
2368 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2369 AArch64::zsub0 + i, DL, VT, SuperReg));
2370
2371 // Copy chain
2372 unsigned ChainIdx = NumVecs;
2373 ReplaceUses(SDValue(N, ChainIdx), SDValue(Load, 1));
2374 CurDAG->RemoveDeadNode(N);
2375}
2376
2377void AArch64DAGToDAGISel::SelectFrintFromVT(SDNode *N, unsigned NumVecs,
2378 unsigned Opcode) {
2379 if (N->getValueType(0) != MVT::nxv4f32)
2380 return;
2381 SelectUnaryMultiIntrinsic(N, NumVecs, true, Opcode);
2382}
2383
2384void AArch64DAGToDAGISel::SelectMultiVectorLutiLane(SDNode *Node,
2385 unsigned NumOutVecs,
2386 unsigned Opc,
2387 uint32_t MaxImm) {
2388 if (ConstantSDNode *Imm = dyn_cast<ConstantSDNode>(Node->getOperand(4)))
2389 if (Imm->getZExtValue() > MaxImm)
2390 return;
2391
2392 SDValue ZtValue;
2393 if (!ImmToReg<AArch64::ZT0, 0>(Node->getOperand(2), ZtValue))
2394 return;
2395
2396 SDValue Chain = Node->getOperand(0);
2397 SDValue Ops[] = {ZtValue, Node->getOperand(3), Node->getOperand(4), Chain};
2398 SDLoc DL(Node);
2399 EVT VT = Node->getValueType(0);
2400
2401 SDNode *Instruction =
2402 CurDAG->getMachineNode(Opc, DL, {MVT::Untyped, MVT::Other}, Ops);
2403 SDValue SuperReg = SDValue(Instruction, 0);
2404
2405 for (unsigned I = 0; I < NumOutVecs; ++I)
2406 ReplaceUses(SDValue(Node, I), CurDAG->getTargetExtractSubreg(
2407 AArch64::zsub0 + I, DL, VT, SuperReg));
2408
2409 // Copy chain
2410 unsigned ChainIdx = NumOutVecs;
2411 ReplaceUses(SDValue(Node, ChainIdx), SDValue(Instruction, 1));
2412 CurDAG->RemoveDeadNode(Node);
2413}
2414
2415void AArch64DAGToDAGISel::SelectMultiVectorLuti6LaneX4(SDNode *Node,
2416 unsigned NumIndexVecs) {
2417 assert((NumIndexVecs == 2 || NumIndexVecs == 3) &&
2418 "unexpected number of index vectors");
2419
2420 constexpr unsigned FirstIndexOp = 3;
2421 unsigned ImmOp = FirstIndexOp + NumIndexVecs;
2422 auto *Imm = dyn_cast<ConstantSDNode>(Node->getOperand(ImmOp));
2423 if (!Imm || Imm->getZExtValue() > 1)
2424 return;
2425
2426 // The luti6 instruction always takes a 2-register Zm index tuple. The x3
2427 // ACLE form provides three index vectors, so the lane selects which adjacent
2428 // pair to use before forming Zm (op 3/4 or op 4/5, with op6 as imm)
2429 unsigned Lane = Imm->getZExtValue();
2430 unsigned IndexOp = FirstIndexOp;
2431 if (NumIndexVecs == 3)
2432 IndexOp += Lane;
2433
2434 SDValue TableTuple = createZTuple({Node->getOperand(1), Node->getOperand(2)});
2435 SDValue IndexTuple =
2436 createZTuple({Node->getOperand(IndexOp), Node->getOperand(IndexOp + 1)});
2437 SDValue Ops[] = {TableTuple, IndexTuple, Node->getOperand(ImmOp)};
2438
2439 SDLoc DL(Node);
2440 EVT VT = Node->getValueType(0);
2441 SDNode *Instruction =
2442 CurDAG->getMachineNode(AArch64::LUTI6_4Z2Z2ZI, DL, MVT::Untyped, Ops);
2443 SDValue SuperReg = SDValue(Instruction, 0);
2444
2445 for (unsigned I = 0; I < 4; ++I)
2446 ReplaceUses(SDValue(Node, I), CurDAG->getTargetExtractSubreg(
2447 AArch64::zsub0 + I, DL, VT, SuperReg));
2448
2449 CurDAG->RemoveDeadNode(Node);
2450}
2451
2452void AArch64DAGToDAGISel::SelectMultiVectorLuti(SDNode *Node,
2453 unsigned NumOutVecs,
2454 unsigned Opc,
2455 unsigned NumInVecs) {
2456 assert((NumInVecs == 2 || NumInVecs == 3) &&
2457 "unexpected number of input vectors");
2458
2459 SDValue ZtValue;
2460 if (!ImmToReg<AArch64::ZT0, 0>(Node->getOperand(2), ZtValue))
2461 return;
2462
2463 SmallVector<SDValue, 4> Regs(Node->ops().slice(3, NumInVecs));
2464 SDValue ZTuple = NumInVecs == 3 ? createZTuple(Regs) : createZMulTuple(Regs);
2465 SDValue Ops[] = {ZtValue, ZTuple, Node->getOperand(0)};
2466
2467 SDLoc DL(Node);
2468 EVT VT = Node->getValueType(0);
2469
2470 SDNode *Instruction =
2471 CurDAG->getMachineNode(Opc, DL, {MVT::Untyped, MVT::Other}, Ops);
2472 SDValue SuperReg = SDValue(Instruction, 0);
2473
2474 for (unsigned I = 0; I < NumOutVecs; ++I)
2475 ReplaceUses(SDValue(Node, I), CurDAG->getTargetExtractSubreg(
2476 AArch64::zsub0 + I, DL, VT, SuperReg));
2477
2478 ReplaceUses(SDValue(Node, NumOutVecs), SDValue(Instruction, 1));
2479 CurDAG->RemoveDeadNode(Node);
2480}
2481
2482void AArch64DAGToDAGISel::SelectClamp(SDNode *N, unsigned NumVecs,
2483 unsigned Op) {
2484 SDLoc DL(N);
2485 EVT VT = N->getValueType(0);
2486
2487 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2488 SDValue Zd = createZMulTuple(Regs);
2489 SDValue Zn = N->getOperand(1 + NumVecs);
2490 SDValue Zm = N->getOperand(2 + NumVecs);
2491
2492 SDValue Ops[] = {Zd, Zn, Zm};
2493
2494 SDNode *Intrinsic = CurDAG->getMachineNode(Op, DL, MVT::Untyped, Ops);
2495 SDValue SuperReg = SDValue(Intrinsic, 0);
2496 for (unsigned i = 0; i < NumVecs; ++i)
2497 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2498 AArch64::zsub0 + i, DL, VT, SuperReg));
2499
2500 CurDAG->RemoveDeadNode(N);
2501}
2502
2503bool SelectSMETile(unsigned &BaseReg, unsigned TileNum) {
2504 switch (BaseReg) {
2505 default:
2506 return false;
2507 case AArch64::ZA:
2508 case AArch64::ZAB0:
2509 if (TileNum == 0)
2510 break;
2511 return false;
2512 case AArch64::ZAH0:
2513 if (TileNum <= 1)
2514 break;
2515 return false;
2516 case AArch64::ZAS0:
2517 if (TileNum <= 3)
2518 break;
2519 return false;
2520 case AArch64::ZAD0:
2521 if (TileNum <= 7)
2522 break;
2523 return false;
2524 }
2525
2526 BaseReg += TileNum;
2527 return true;
2528}
2529
2530template <unsigned MaxIdx, unsigned Scale>
2531void AArch64DAGToDAGISel::SelectMultiVectorMove(SDNode *N, unsigned NumVecs,
2532 unsigned BaseReg, unsigned Op) {
2533 unsigned TileNum = 0;
2534 if (BaseReg != AArch64::ZA)
2535 TileNum = N->getConstantOperandVal(2);
2536
2537 if (!SelectSMETile(BaseReg, TileNum))
2538 return;
2539
2540 SDValue SliceBase, Base, Offset;
2541 if (BaseReg == AArch64::ZA)
2542 SliceBase = N->getOperand(2);
2543 else
2544 SliceBase = N->getOperand(3);
2545
2546 if (!SelectSMETileSlice(SliceBase, MaxIdx, Base, Offset, Scale))
2547 return;
2548
2549 SDLoc DL(N);
2550 SDValue SubReg = CurDAG->getRegister(BaseReg, MVT::Other);
2551 SDValue Ops[] = {SubReg, Base, Offset, /*Chain*/ N->getOperand(0)};
2552 SDNode *Mov = CurDAG->getMachineNode(Op, DL, {MVT::Untyped, MVT::Other}, Ops);
2553
2554 EVT VT = N->getValueType(0);
2555 for (unsigned I = 0; I < NumVecs; ++I)
2556 ReplaceUses(SDValue(N, I),
2557 CurDAG->getTargetExtractSubreg(AArch64::zsub0 + I, DL, VT,
2558 SDValue(Mov, 0)));
2559 // Copy chain
2560 unsigned ChainIdx = NumVecs;
2561 ReplaceUses(SDValue(N, ChainIdx), SDValue(Mov, 1));
2562 CurDAG->RemoveDeadNode(N);
2563}
2564
2565void AArch64DAGToDAGISel::SelectMultiVectorMoveZ(SDNode *N, unsigned NumVecs,
2566 unsigned Op, unsigned MaxIdx,
2567 unsigned Scale, unsigned BaseReg) {
2568 // Slice can be in different positions
2569 // The array to vector: llvm.aarch64.sme.readz.<h/v>.<sz>(slice)
2570 // The tile to vector: llvm.aarch64.sme.readz.<h/v>.<sz>(tile, slice)
2571 SDValue SliceBase = N->getOperand(2);
2572 if (BaseReg != AArch64::ZA)
2573 SliceBase = N->getOperand(3);
2574
2575 SDValue Base, Offset;
2576 if (!SelectSMETileSlice(SliceBase, MaxIdx, Base, Offset, Scale))
2577 return;
2578 // The correct Za tile number is computed in Machine Instruction
2579 // See EmitZAInstr
2580 // DAG cannot select Za tile as an output register with ZReg
2581 SDLoc DL(N);
2583 if (BaseReg != AArch64::ZA )
2584 Ops.push_back(N->getOperand(2));
2585 Ops.push_back(Base);
2586 Ops.push_back(Offset);
2587 Ops.push_back(N->getOperand(0)); //Chain
2588 SDNode *Mov = CurDAG->getMachineNode(Op, DL, {MVT::Untyped, MVT::Other}, Ops);
2589
2590 EVT VT = N->getValueType(0);
2591 for (unsigned I = 0; I < NumVecs; ++I)
2592 ReplaceUses(SDValue(N, I),
2593 CurDAG->getTargetExtractSubreg(AArch64::zsub0 + I, DL, VT,
2594 SDValue(Mov, 0)));
2595
2596 // Copy chain
2597 unsigned ChainIdx = NumVecs;
2598 ReplaceUses(SDValue(N, ChainIdx), SDValue(Mov, 1));
2599 CurDAG->RemoveDeadNode(N);
2600}
2601
2602void AArch64DAGToDAGISel::SelectUnaryMultiIntrinsic(SDNode *N,
2603 unsigned NumOutVecs,
2604 bool IsTupleInput,
2605 unsigned Opc) {
2606 SDLoc DL(N);
2607 EVT VT = N->getValueType(0);
2608 unsigned NumInVecs = N->getNumOperands() - 1;
2609
2611 if (IsTupleInput) {
2612 assert((NumInVecs == 2 || NumInVecs == 4) &&
2613 "Don't know how to handle multi-register input!");
2614 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumInVecs));
2615 Ops.push_back(createZMulTuple(Regs));
2616 } else {
2617 // All intrinsic nodes have the ID as the first operand, hence the "1 + I".
2618 for (unsigned I = 0; I < NumInVecs; I++)
2619 Ops.push_back(N->getOperand(1 + I));
2620 }
2621
2622 SDNode *Res = CurDAG->getMachineNode(Opc, DL, MVT::Untyped, Ops);
2623 SDValue SuperReg = SDValue(Res, 0);
2624
2625 for (unsigned I = 0; I < NumOutVecs; I++)
2626 ReplaceUses(SDValue(N, I), CurDAG->getTargetExtractSubreg(
2627 AArch64::zsub0 + I, DL, VT, SuperReg));
2628 CurDAG->RemoveDeadNode(N);
2629}
2630
2631void AArch64DAGToDAGISel::SelectStore(SDNode *N, unsigned NumVecs,
2632 unsigned Opc) {
2633 SDLoc dl(N);
2634 EVT VT = N->getOperand(2)->getValueType(0);
2635
2636 // Form a REG_SEQUENCE to force register allocation.
2637 bool Is128Bit = VT.getSizeInBits() == 128;
2638 SmallVector<SDValue, 4> Regs(N->ops().slice(2, NumVecs));
2639 SDValue RegSeq = Is128Bit ? createQTuple(Regs) : createDTuple(Regs);
2640
2641 SDValue Ops[] = {RegSeq, N->getOperand(NumVecs + 2), N->getOperand(0)};
2642 SDNode *St = CurDAG->getMachineNode(Opc, dl, N->getValueType(0), Ops);
2643
2644 // Transfer memoperands.
2645 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2646 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2647
2648 ReplaceNode(N, St);
2649}
2650
2651void AArch64DAGToDAGISel::SelectPredicatedStore(SDNode *N, unsigned NumVecs,
2652 unsigned Scale, unsigned Opc_rr,
2653 unsigned Opc_ri) {
2654 SDLoc dl(N);
2655
2656 // Form a REG_SEQUENCE to force register allocation.
2657 SmallVector<SDValue, 4> Regs(N->ops().slice(2, NumVecs));
2658 SDValue RegSeq = createZTuple(Regs);
2659
2660 // Optimize addressing mode.
2661 unsigned Opc;
2662 SDValue Offset, Base;
2663 std::tie(Opc, Base, Offset) = findAddrModeSVELoadStore(
2664 N, Opc_rr, Opc_ri, N->getOperand(NumVecs + 3),
2665 CurDAG->getTargetConstant(0, dl, MVT::i64), Scale);
2666
2667 SDValue Ops[] = {RegSeq, N->getOperand(NumVecs + 2), // predicate
2668 Base, // address
2669 Offset, // offset
2670 N->getOperand(0)}; // chain
2671 SDNode *St = CurDAG->getMachineNode(Opc, dl, N->getValueType(0), Ops);
2672
2673 // Transfer memoperands.
2674 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2675 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2676
2677 ReplaceNode(N, St);
2678}
2679
2680void AArch64DAGToDAGISel::SelectPostStore(SDNode *N, unsigned NumVecs,
2681 unsigned Opc) {
2682 SDLoc dl(N);
2683 EVT VT = N->getOperand(2)->getValueType(0);
2684 const EVT ResTys[] = {MVT::i64, // Type of the write back register
2685 MVT::Other}; // Type for the Chain
2686
2687 // Form a REG_SEQUENCE to force register allocation.
2688 bool Is128Bit = VT.getSizeInBits() == 128;
2689 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2690 SDValue RegSeq = Is128Bit ? createQTuple(Regs) : createDTuple(Regs);
2691
2692 SDValue Ops[] = {RegSeq,
2693 N->getOperand(NumVecs + 1), // base register
2694 N->getOperand(NumVecs + 2), // Incremental
2695 N->getOperand(0)}; // Chain
2696 SDNode *St = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2697
2698 // Transfer memoperands.
2699 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2700 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2701
2702 ReplaceNode(N, St);
2703}
2704
2705namespace {
2706/// WidenVector - Given a value in the V64 register class, produce the
2707/// equivalent value in the V128 register class.
2708class WidenVector {
2709 SelectionDAG &DAG;
2710
2711public:
2712 WidenVector(SelectionDAG &DAG) : DAG(DAG) {}
2713
2714 SDValue operator()(SDValue V64Reg) {
2715 EVT VT = V64Reg.getValueType();
2716 unsigned NarrowSize = VT.getVectorNumElements();
2717 MVT EltTy = VT.getVectorElementType().getSimpleVT();
2718 MVT WideTy = MVT::getVectorVT(EltTy, 2 * NarrowSize);
2719 SDLoc DL(V64Reg);
2720
2721 SDValue Undef =
2722 SDValue(DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, WideTy), 0);
2723 return DAG.getTargetInsertSubreg(AArch64::dsub, DL, WideTy, Undef, V64Reg);
2724 }
2725};
2726} // namespace
2727
2728/// NarrowVector - Given a value in the V128 register class, produce the
2729/// equivalent value in the V64 register class.
2731 EVT VT = V128Reg.getValueType();
2732 unsigned WideSize = VT.getVectorNumElements();
2733 MVT EltTy = VT.getVectorElementType().getSimpleVT();
2734 MVT NarrowTy = MVT::getVectorVT(EltTy, WideSize / 2);
2735
2736 return DAG.getTargetExtractSubreg(AArch64::dsub, SDLoc(V128Reg), NarrowTy,
2737 V128Reg);
2738}
2739
2740void AArch64DAGToDAGISel::SelectLoadLane(SDNode *N, unsigned NumVecs,
2741 unsigned Opc) {
2742 SDLoc dl(N);
2743 EVT VT = N->getValueType(0);
2744 bool Narrow = VT.getSizeInBits() == 64;
2745
2746 // Form a REG_SEQUENCE to force register allocation.
2747 SmallVector<SDValue, 4> Regs(N->ops().slice(2, NumVecs));
2748
2749 if (Narrow)
2750 transform(Regs, Regs.begin(),
2751 WidenVector(*CurDAG));
2752
2753 SDValue RegSeq = createQTuple(Regs);
2754
2755 const EVT ResTys[] = {MVT::Untyped, MVT::Other};
2756
2757 unsigned LaneNo = N->getConstantOperandVal(NumVecs + 2);
2758
2759 SDValue Ops[] = {RegSeq, CurDAG->getTargetConstant(LaneNo, dl, MVT::i64),
2760 N->getOperand(NumVecs + 3), N->getOperand(0)};
2761 SDNode *Ld = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2762 SDValue SuperReg = SDValue(Ld, 0);
2763
2764 EVT WideVT = RegSeq.getOperand(1)->getValueType(0);
2765 static const unsigned QSubs[] = { AArch64::qsub0, AArch64::qsub1,
2766 AArch64::qsub2, AArch64::qsub3 };
2767 for (unsigned i = 0; i < NumVecs; ++i) {
2768 SDValue NV = CurDAG->getTargetExtractSubreg(QSubs[i], dl, WideVT, SuperReg);
2769 if (Narrow)
2770 NV = NarrowVector(NV, *CurDAG);
2771 ReplaceUses(SDValue(N, i), NV);
2772 }
2773
2774 ReplaceUses(SDValue(N, NumVecs), SDValue(Ld, 1));
2775 CurDAG->RemoveDeadNode(N);
2776}
2777
2778void AArch64DAGToDAGISel::SelectPostLoadLane(SDNode *N, unsigned NumVecs,
2779 unsigned Opc) {
2780 SDLoc dl(N);
2781 EVT VT = N->getValueType(0);
2782 bool Narrow = VT.getSizeInBits() == 64;
2783
2784 // Form a REG_SEQUENCE to force register allocation.
2785 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2786
2787 if (Narrow)
2788 transform(Regs, Regs.begin(),
2789 WidenVector(*CurDAG));
2790
2791 SDValue RegSeq = createQTuple(Regs);
2792
2793 const EVT ResTys[] = {MVT::i64, // Type of the write back register
2794 RegSeq->getValueType(0), MVT::Other};
2795
2796 unsigned LaneNo = N->getConstantOperandVal(NumVecs + 1);
2797
2798 SDValue Ops[] = {RegSeq,
2799 CurDAG->getTargetConstant(LaneNo, dl,
2800 MVT::i64), // Lane Number
2801 N->getOperand(NumVecs + 2), // Base register
2802 N->getOperand(NumVecs + 3), // Incremental
2803 N->getOperand(0)};
2804 SDNode *Ld = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2805
2806 // Update uses of the write back register
2807 ReplaceUses(SDValue(N, NumVecs), SDValue(Ld, 0));
2808
2809 // Update uses of the vector list
2810 SDValue SuperReg = SDValue(Ld, 1);
2811 if (NumVecs == 1) {
2812 ReplaceUses(SDValue(N, 0),
2813 Narrow ? NarrowVector(SuperReg, *CurDAG) : SuperReg);
2814 } else {
2815 EVT WideVT = RegSeq.getOperand(1)->getValueType(0);
2816 static const unsigned QSubs[] = { AArch64::qsub0, AArch64::qsub1,
2817 AArch64::qsub2, AArch64::qsub3 };
2818 for (unsigned i = 0; i < NumVecs; ++i) {
2819 SDValue NV = CurDAG->getTargetExtractSubreg(QSubs[i], dl, WideVT,
2820 SuperReg);
2821 if (Narrow)
2822 NV = NarrowVector(NV, *CurDAG);
2823 ReplaceUses(SDValue(N, i), NV);
2824 }
2825 }
2826
2827 // Update the Chain
2828 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(Ld, 2));
2829 CurDAG->RemoveDeadNode(N);
2830}
2831
2832void AArch64DAGToDAGISel::SelectStoreLane(SDNode *N, unsigned NumVecs,
2833 unsigned Opc) {
2834 SDLoc dl(N);
2835 EVT VT = N->getOperand(2)->getValueType(0);
2836 bool Narrow = VT.getSizeInBits() == 64;
2837
2838 // Form a REG_SEQUENCE to force register allocation.
2839 SmallVector<SDValue, 4> Regs(N->ops().slice(2, NumVecs));
2840
2841 if (Narrow)
2842 transform(Regs, Regs.begin(),
2843 WidenVector(*CurDAG));
2844
2845 SDValue RegSeq = createQTuple(Regs);
2846
2847 unsigned LaneNo = N->getConstantOperandVal(NumVecs + 2);
2848
2849 SDValue Ops[] = {RegSeq, CurDAG->getTargetConstant(LaneNo, dl, MVT::i64),
2850 N->getOperand(NumVecs + 3), N->getOperand(0)};
2851 SDNode *St = CurDAG->getMachineNode(Opc, dl, MVT::Other, Ops);
2852
2853 // Transfer memoperands.
2854 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2855 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2856
2857 ReplaceNode(N, St);
2858}
2859
2860void AArch64DAGToDAGISel::SelectPostStoreLane(SDNode *N, unsigned NumVecs,
2861 unsigned Opc) {
2862 SDLoc dl(N);
2863 EVT VT = N->getOperand(2)->getValueType(0);
2864 bool Narrow = VT.getSizeInBits() == 64;
2865
2866 // Form a REG_SEQUENCE to force register allocation.
2867 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2868
2869 if (Narrow)
2870 transform(Regs, Regs.begin(),
2871 WidenVector(*CurDAG));
2872
2873 SDValue RegSeq = createQTuple(Regs);
2874
2875 const EVT ResTys[] = {MVT::i64, // Type of the write back register
2876 MVT::Other};
2877
2878 unsigned LaneNo = N->getConstantOperandVal(NumVecs + 1);
2879
2880 SDValue Ops[] = {RegSeq, CurDAG->getTargetConstant(LaneNo, dl, MVT::i64),
2881 N->getOperand(NumVecs + 2), // Base Register
2882 N->getOperand(NumVecs + 3), // Incremental
2883 N->getOperand(0)};
2884 SDNode *St = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2885
2886 // Transfer memoperands.
2887 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2888 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2889
2890 ReplaceNode(N, St);
2891}
2892
2894 unsigned &Opc, SDValue &Opd0,
2895 unsigned &LSB, unsigned &MSB,
2896 unsigned NumberOfIgnoredLowBits,
2897 bool BiggerPattern) {
2898 assert(N->getOpcode() == ISD::AND &&
2899 "N must be a AND operation to call this function");
2900
2901 EVT VT = N->getValueType(0);
2902
2903 // Here we can test the type of VT and return false when the type does not
2904 // match, but since it is done prior to that call in the current context
2905 // we turned that into an assert to avoid redundant code.
2906 assert((VT == MVT::i32 || VT == MVT::i64) &&
2907 "Type checking must have been done before calling this function");
2908
2909 // FIXME: simplify-demanded-bits in DAGCombine will probably have
2910 // changed the AND node to a 32-bit mask operation. We'll have to
2911 // undo that as part of the transform here if we want to catch all
2912 // the opportunities.
2913 // Currently the NumberOfIgnoredLowBits argument helps to recover
2914 // from these situations when matching bigger pattern (bitfield insert).
2915
2916 // For unsigned extracts, check for a shift right and mask
2917 uint64_t AndImm = 0;
2918 if (!isOpcWithIntImmediate(N, ISD::AND, AndImm))
2919 return false;
2920
2921 const SDNode *Op0 = N->getOperand(0).getNode();
2922
2923 // Because of simplify-demanded-bits in DAGCombine, the mask may have been
2924 // simplified. Try to undo that
2925 AndImm |= maskTrailingOnes<uint64_t>(NumberOfIgnoredLowBits);
2926
2927 // The immediate is a mask of the low bits iff imm & (imm+1) == 0
2928 if (AndImm & (AndImm + 1))
2929 return false;
2930
2931 bool ClampMSB = false;
2932 uint64_t SrlImm = 0;
2933 // Handle the SRL + ANY_EXTEND case.
2934 if (VT == MVT::i64 && Op0->getOpcode() == ISD::ANY_EXTEND &&
2935 isOpcWithIntImmediate(Op0->getOperand(0).getNode(), ISD::SRL, SrlImm)) {
2936 // Extend the incoming operand of the SRL to 64-bit.
2937 Opd0 = Widen(CurDAG, Op0->getOperand(0).getOperand(0));
2938 // Make sure to clamp the MSB so that we preserve the semantics of the
2939 // original operations.
2940 ClampMSB = true;
2941 } else if (VT == MVT::i32 && Op0->getOpcode() == ISD::TRUNCATE &&
2943 SrlImm)) {
2944 // If the shift result was truncated, we can still combine them.
2945 Opd0 = Op0->getOperand(0).getOperand(0);
2946
2947 // Use the type of SRL node.
2948 VT = Opd0->getValueType(0);
2949 } else if (isOpcWithIntImmediate(Op0, ISD::SRL, SrlImm)) {
2950 Opd0 = Op0->getOperand(0);
2951 ClampMSB = (VT == MVT::i32);
2952 } else if (BiggerPattern) {
2953 // Let's pretend a 0 shift right has been performed.
2954 // The resulting code will be at least as good as the original one
2955 // plus it may expose more opportunities for bitfield insert pattern.
2956 // FIXME: Currently we limit this to the bigger pattern, because
2957 // some optimizations expect AND and not UBFM.
2958 Opd0 = N->getOperand(0);
2959 } else
2960 return false;
2961
2962 // Bail out on large immediates. This happens when no proper
2963 // combining/constant folding was performed.
2964 if (!BiggerPattern && (SrlImm <= 0 || SrlImm >= VT.getSizeInBits())) {
2965 LLVM_DEBUG(
2966 (dbgs() << N
2967 << ": Found large shift immediate, this should not happen\n"));
2968 return false;
2969 }
2970
2971 LSB = SrlImm;
2972 MSB = SrlImm +
2973 (VT == MVT::i32 ? llvm::countr_one<uint32_t>(AndImm)
2974 : llvm::countr_one<uint64_t>(AndImm)) -
2975 1;
2976 if (ClampMSB)
2977 // Since we're moving the extend before the right shift operation, we need
2978 // to clamp the MSB to make sure we don't shift in undefined bits instead of
2979 // the zeros which would get shifted in with the original right shift
2980 // operation.
2981 MSB = MSB > 31 ? 31 : MSB;
2982
2983 Opc = VT == MVT::i32 ? AArch64::UBFMWri : AArch64::UBFMXri;
2984 return true;
2985}
2986
2988 SDValue &Opd0, unsigned &Immr,
2989 unsigned &Imms) {
2990 assert(N->getOpcode() == ISD::SIGN_EXTEND_INREG);
2991
2992 EVT VT = N->getValueType(0);
2993 unsigned BitWidth = VT.getSizeInBits();
2994 assert((VT == MVT::i32 || VT == MVT::i64) &&
2995 "Type checking must have been done before calling this function");
2996
2997 SDValue Op = N->getOperand(0);
2998 if (Op->getOpcode() == ISD::TRUNCATE) {
2999 Op = Op->getOperand(0);
3000 VT = Op->getValueType(0);
3001 BitWidth = VT.getSizeInBits();
3002 }
3003
3004 uint64_t ShiftImm;
3005 if (!isOpcWithIntImmediate(Op.getNode(), ISD::SRL, ShiftImm) &&
3006 !isOpcWithIntImmediate(Op.getNode(), ISD::SRA, ShiftImm))
3007 return false;
3008
3009 unsigned Width = cast<VTSDNode>(N->getOperand(1))->getVT().getSizeInBits();
3010 if (ShiftImm + Width > BitWidth)
3011 return false;
3012
3013 Opc = (VT == MVT::i32) ? AArch64::SBFMWri : AArch64::SBFMXri;
3014 Opd0 = Op.getOperand(0);
3015 Immr = ShiftImm;
3016 Imms = ShiftImm + Width - 1;
3017 return true;
3018}
3019
3021 SDValue &Opd0, unsigned &LSB,
3022 unsigned &MSB) {
3023 // We are looking for the following pattern which basically extracts several
3024 // continuous bits from the source value and places it from the LSB of the
3025 // destination value, all other bits of the destination value or set to zero:
3026 //
3027 // Value2 = AND Value, MaskImm
3028 // SRL Value2, ShiftImm
3029 //
3030 // with MaskImm >> ShiftImm to search for the bit width.
3031 //
3032 // This gets selected into a single UBFM:
3033 //
3034 // UBFM Value, ShiftImm, Log2_64(MaskImm)
3035 //
3036
3037 if (N->getOpcode() != ISD::SRL)
3038 return false;
3039
3040 uint64_t AndMask = 0;
3041 if (!isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::AND, AndMask))
3042 return false;
3043
3044 Opd0 = N->getOperand(0).getOperand(0);
3045
3046 uint64_t SrlImm = 0;
3047 if (!isIntImmediate(N->getOperand(1), SrlImm))
3048 return false;
3049
3050 // Check whether we really have several bits extract here.
3051 if (!isMask_64(AndMask >> SrlImm))
3052 return false;
3053
3054 Opc = N->getValueType(0) == MVT::i32 ? AArch64::UBFMWri : AArch64::UBFMXri;
3055 LSB = SrlImm;
3056 MSB = llvm::Log2_64(AndMask);
3057 return true;
3058}
3059
3060static bool isBitfieldExtractOpFromShr(SDNode *N, unsigned &Opc, SDValue &Opd0,
3061 unsigned &Immr, unsigned &Imms,
3062 bool BiggerPattern) {
3063 assert((N->getOpcode() == ISD::SRA || N->getOpcode() == ISD::SRL) &&
3064 "N must be a SHR/SRA operation to call this function");
3065
3066 EVT VT = N->getValueType(0);
3067
3068 // Here we can test the type of VT and return false when the type does not
3069 // match, but since it is done prior to that call in the current context
3070 // we turned that into an assert to avoid redundant code.
3071 assert((VT == MVT::i32 || VT == MVT::i64) &&
3072 "Type checking must have been done before calling this function");
3073
3074 // Check for AND + SRL doing several bits extract.
3075 if (isSeveralBitsExtractOpFromShr(N, Opc, Opd0, Immr, Imms))
3076 return true;
3077
3078 // We're looking for a shift of a shift.
3079 uint64_t ShlImm = 0;
3080 uint64_t TruncBits = 0;
3081 if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SHL, ShlImm)) {
3082 Opd0 = N->getOperand(0).getOperand(0);
3083 } else if (VT == MVT::i32 && N->getOpcode() == ISD::SRL &&
3084 N->getOperand(0).getNode()->getOpcode() == ISD::TRUNCATE) {
3085 // We are looking for a shift of truncate. Truncate from i64 to i32 could
3086 // be considered as setting high 32 bits as zero. Our strategy here is to
3087 // always generate 64bit UBFM. This consistency will help the CSE pass
3088 // later find more redundancy.
3089 Opd0 = N->getOperand(0).getOperand(0);
3090 TruncBits = Opd0->getValueType(0).getSizeInBits() - VT.getSizeInBits();
3091 VT = Opd0.getValueType();
3092 assert(VT == MVT::i64 && "the promoted type should be i64");
3093 } else if (BiggerPattern) {
3094 // Let's pretend a 0 shift left has been performed.
3095 // FIXME: Currently we limit this to the bigger pattern case,
3096 // because some optimizations expect AND and not UBFM
3097 Opd0 = N->getOperand(0);
3098 } else
3099 return false;
3100
3101 // Missing combines/constant folding may have left us with strange
3102 // constants.
3103 if (ShlImm >= VT.getSizeInBits()) {
3104 LLVM_DEBUG(
3105 (dbgs() << N
3106 << ": Found large shift immediate, this should not happen\n"));
3107 return false;
3108 }
3109
3110 uint64_t SrlImm = 0;
3111 if (!isIntImmediate(N->getOperand(1), SrlImm))
3112 return false;
3113
3114 assert(SrlImm > 0 && SrlImm < VT.getSizeInBits() &&
3115 "bad amount in shift node!");
3116 int immr = SrlImm - ShlImm;
3117 Immr = immr < 0 ? immr + VT.getSizeInBits() : immr;
3118 Imms = VT.getSizeInBits() - ShlImm - TruncBits - 1;
3119 // SRA requires a signed extraction
3120 if (VT == MVT::i32)
3121 Opc = N->getOpcode() == ISD::SRA ? AArch64::SBFMWri : AArch64::UBFMWri;
3122 else
3123 Opc = N->getOpcode() == ISD::SRA ? AArch64::SBFMXri : AArch64::UBFMXri;
3124 return true;
3125}
3126
3127bool AArch64DAGToDAGISel::tryBitfieldExtractOpFromSExt(SDNode *N) {
3128 assert(N->getOpcode() == ISD::SIGN_EXTEND);
3129
3130 EVT VT = N->getValueType(0);
3131 EVT NarrowVT = N->getOperand(0)->getValueType(0);
3132 if (VT != MVT::i64 || NarrowVT != MVT::i32)
3133 return false;
3134
3135 uint64_t ShiftImm;
3136 SDValue Op = N->getOperand(0);
3137 if (!isOpcWithIntImmediate(Op.getNode(), ISD::SRA, ShiftImm))
3138 return false;
3139
3140 SDLoc dl(N);
3141 // Extend the incoming operand of the shift to 64-bits.
3142 SDValue Opd0 = Widen(CurDAG, Op.getOperand(0));
3143 unsigned Immr = ShiftImm;
3144 unsigned Imms = NarrowVT.getSizeInBits() - 1;
3145 SDValue Ops[] = {Opd0, CurDAG->getTargetConstant(Immr, dl, VT),
3146 CurDAG->getTargetConstant(Imms, dl, VT)};
3147 CurDAG->SelectNodeTo(N, AArch64::SBFMXri, VT, Ops);
3148 return true;
3149}
3150
3151static bool isBitfieldExtractOp(SelectionDAG *CurDAG, SDNode *N, unsigned &Opc,
3152 SDValue &Opd0, unsigned &Immr, unsigned &Imms,
3153 unsigned NumberOfIgnoredLowBits = 0,
3154 bool BiggerPattern = false) {
3155 if (N->getValueType(0) != MVT::i32 && N->getValueType(0) != MVT::i64)
3156 return false;
3157
3158 switch (N->getOpcode()) {
3159 default:
3160 if (!N->isMachineOpcode())
3161 return false;
3162 break;
3163 case ISD::AND:
3164 return isBitfieldExtractOpFromAnd(CurDAG, N, Opc, Opd0, Immr, Imms,
3165 NumberOfIgnoredLowBits, BiggerPattern);
3166 case ISD::SRL:
3167 case ISD::SRA:
3168 return isBitfieldExtractOpFromShr(N, Opc, Opd0, Immr, Imms, BiggerPattern);
3169
3171 return isBitfieldExtractOpFromSExtInReg(N, Opc, Opd0, Immr, Imms);
3172 }
3173
3174 unsigned NOpc = N->getMachineOpcode();
3175 switch (NOpc) {
3176 default:
3177 return false;
3178 case AArch64::SBFMWri:
3179 case AArch64::UBFMWri:
3180 case AArch64::SBFMXri:
3181 case AArch64::UBFMXri:
3182 Opc = NOpc;
3183 Opd0 = N->getOperand(0);
3184 Immr = N->getConstantOperandVal(1);
3185 Imms = N->getConstantOperandVal(2);
3186 return true;
3187 }
3188 // Unreachable
3189 return false;
3190}
3191
3192bool AArch64DAGToDAGISel::tryBitfieldExtractOp(SDNode *N) {
3193 unsigned Opc, Immr, Imms;
3194 SDValue Opd0;
3195 if (!isBitfieldExtractOp(CurDAG, N, Opc, Opd0, Immr, Imms))
3196 return false;
3197
3198 EVT VT = N->getValueType(0);
3199 SDLoc dl(N);
3200
3201 // If the bit extract operation is 64bit but the original type is 32bit, we
3202 // need to add one EXTRACT_SUBREG.
3203 if ((Opc == AArch64::SBFMXri || Opc == AArch64::UBFMXri) && VT == MVT::i32) {
3204 SDValue Ops64[] = {Opd0, CurDAG->getTargetConstant(Immr, dl, MVT::i64),
3205 CurDAG->getTargetConstant(Imms, dl, MVT::i64)};
3206
3207 SDNode *BFM = CurDAG->getMachineNode(Opc, dl, MVT::i64, Ops64);
3208 SDValue Inner = CurDAG->getTargetExtractSubreg(AArch64::sub_32, dl,
3209 MVT::i32, SDValue(BFM, 0));
3210 ReplaceNode(N, Inner.getNode());
3211 return true;
3212 }
3213
3214 SDValue Ops[] = {Opd0, CurDAG->getTargetConstant(Immr, dl, VT),
3215 CurDAG->getTargetConstant(Imms, dl, VT)};
3216 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
3217 return true;
3218}
3219
3220/// Does DstMask form a complementary pair with the mask provided by
3221/// BitsToBeInserted, suitable for use in a BFI instruction. Roughly speaking,
3222/// this asks whether DstMask zeroes precisely those bits that will be set by
3223/// the other half.
3224static bool isBitfieldDstMask(uint64_t DstMask, const APInt &BitsToBeInserted,
3225 unsigned NumberOfIgnoredHighBits, EVT VT) {
3226 assert((VT == MVT::i32 || VT == MVT::i64) &&
3227 "i32 or i64 mask type expected!");
3228 unsigned BitWidth = VT.getSizeInBits() - NumberOfIgnoredHighBits;
3229
3230 // Enable implicitTrunc as we're intentionally ignoring high bits.
3231 APInt SignificantDstMask =
3232 APInt(BitWidth, DstMask, /*isSigned=*/false, /*implicitTrunc=*/true);
3233 APInt SignificantBitsToBeInserted = BitsToBeInserted.zextOrTrunc(BitWidth);
3234
3235 return (SignificantDstMask & SignificantBitsToBeInserted) == 0 &&
3236 (SignificantDstMask | SignificantBitsToBeInserted).isAllOnes();
3237}
3238
3239// Look for bits that will be useful for later uses.
3240// A bit is consider useless as soon as it is dropped and never used
3241// before it as been dropped.
3242// E.g., looking for useful bit of x
3243// 1. y = x & 0x7
3244// 2. z = y >> 2
3245// After #1, x useful bits are 0x7, then the useful bits of x, live through
3246// y.
3247// After #2, the useful bits of x are 0x4.
3248// However, if x is used on an unpredictable instruction, then all its bits
3249// are useful.
3250// E.g.
3251// 1. y = x & 0x7
3252// 2. z = y >> 2
3253// 3. str x, [@x]
3254static void getUsefulBits(SDValue Op, APInt &UsefulBits, unsigned Depth = 0);
3255
3257 unsigned Depth) {
3258 uint64_t Imm =
3259 cast<const ConstantSDNode>(Op.getOperand(1).getNode())->getZExtValue();
3261 UsefulBits &= APInt(UsefulBits.getBitWidth(), Imm);
3262 getUsefulBits(Op, UsefulBits, Depth + 1);
3263}
3264
3266 uint64_t Imm, uint64_t MSB,
3267 unsigned Depth) {
3268 // inherit the bitwidth value
3269 APInt OpUsefulBits(UsefulBits);
3270 OpUsefulBits = 1;
3271
3272 if (MSB >= Imm) {
3273 OpUsefulBits <<= MSB - Imm + 1;
3274 --OpUsefulBits;
3275 // The interesting part will be in the lower part of the result
3276 getUsefulBits(Op, OpUsefulBits, Depth + 1);
3277 // The interesting part was starting at Imm in the argument
3278 OpUsefulBits <<= Imm;
3279 } else {
3280 OpUsefulBits <<= MSB + 1;
3281 --OpUsefulBits;
3282 // The interesting part will be shifted in the result
3283 OpUsefulBits <<= OpUsefulBits.getBitWidth() - Imm;
3284 getUsefulBits(Op, OpUsefulBits, Depth + 1);
3285 // The interesting part was at zero in the argument
3286 OpUsefulBits.lshrInPlace(OpUsefulBits.getBitWidth() - Imm);
3287 }
3288
3289 UsefulBits &= OpUsefulBits;
3290}
3291
3292static void getUsefulBitsFromUBFM(SDValue Op, APInt &UsefulBits,
3293 unsigned Depth) {
3294 uint64_t Imm =
3295 cast<const ConstantSDNode>(Op.getOperand(1).getNode())->getZExtValue();
3296 uint64_t MSB =
3297 cast<const ConstantSDNode>(Op.getOperand(2).getNode())->getZExtValue();
3298
3299 getUsefulBitsFromBitfieldMoveOpd(Op, UsefulBits, Imm, MSB, Depth);
3300}
3301
3303 unsigned Depth) {
3304 uint64_t ShiftTypeAndValue =
3305 cast<const ConstantSDNode>(Op.getOperand(2).getNode())->getZExtValue();
3306 APInt Mask(UsefulBits);
3307 Mask.clearAllBits();
3308 Mask.flipAllBits();
3309
3310 if (AArch64_AM::getShiftType(ShiftTypeAndValue) == AArch64_AM::LSL) {
3311 // Shift Left
3312 uint64_t ShiftAmt = AArch64_AM::getShiftValue(ShiftTypeAndValue);
3313 Mask <<= ShiftAmt;
3314 getUsefulBits(Op, Mask, Depth + 1);
3315 Mask.lshrInPlace(ShiftAmt);
3316 } else if (AArch64_AM::getShiftType(ShiftTypeAndValue) == AArch64_AM::LSR) {
3317 // Shift Right
3318 // We do not handle AArch64_AM::ASR, because the sign will change the
3319 // number of useful bits
3320 uint64_t ShiftAmt = AArch64_AM::getShiftValue(ShiftTypeAndValue);
3321 Mask.lshrInPlace(ShiftAmt);
3322 getUsefulBits(Op, Mask, Depth + 1);
3323 Mask <<= ShiftAmt;
3324 } else
3325 return;
3326
3327 UsefulBits &= Mask;
3328}
3329
3330static void getUsefulBitsFromBFM(SDValue Op, SDValue Orig, APInt &UsefulBits,
3331 unsigned Depth) {
3332 uint64_t Imm =
3333 cast<const ConstantSDNode>(Op.getOperand(2).getNode())->getZExtValue();
3334 uint64_t MSB =
3335 cast<const ConstantSDNode>(Op.getOperand(3).getNode())->getZExtValue();
3336
3337 APInt OpUsefulBits(UsefulBits);
3338 OpUsefulBits = 1;
3339
3340 APInt ResultUsefulBits(UsefulBits.getBitWidth(), 0);
3341 ResultUsefulBits.flipAllBits();
3342 APInt Mask(UsefulBits.getBitWidth(), 0);
3343
3344 getUsefulBits(Op, ResultUsefulBits, Depth + 1);
3345
3346 if (MSB >= Imm) {
3347 // The instruction is a BFXIL.
3348 uint64_t Width = MSB - Imm + 1;
3349 uint64_t LSB = Imm;
3350
3351 OpUsefulBits <<= Width;
3352 --OpUsefulBits;
3353
3354 if (Op.getOperand(1) == Orig) {
3355 // Copy the low bits from the result to bits starting from LSB.
3356 Mask = ResultUsefulBits & OpUsefulBits;
3357 Mask <<= LSB;
3358 }
3359
3360 if (Op.getOperand(0) == Orig)
3361 // Bits starting from LSB in the input contribute to the result.
3362 Mask |= (ResultUsefulBits & ~OpUsefulBits);
3363 } else {
3364 // The instruction is a BFI.
3365 uint64_t Width = MSB + 1;
3366 uint64_t LSB = UsefulBits.getBitWidth() - Imm;
3367
3368 OpUsefulBits <<= Width;
3369 --OpUsefulBits;
3370 OpUsefulBits <<= LSB;
3371
3372 if (Op.getOperand(1) == Orig) {
3373 // Copy the bits from the result to the zero bits.
3374 Mask = ResultUsefulBits & OpUsefulBits;
3375 Mask.lshrInPlace(LSB);
3376 }
3377
3378 if (Op.getOperand(0) == Orig)
3379 Mask |= (ResultUsefulBits & ~OpUsefulBits);
3380 }
3381
3382 UsefulBits &= Mask;
3383}
3384
3385static void getUsefulBitsForUse(SDNode *UserNode, APInt &UsefulBits,
3386 SDValue Orig, unsigned Depth) {
3387
3388 // Users of this node should have already been instruction selected
3389 // FIXME: Can we turn that into an assert?
3390 if (!UserNode->isMachineOpcode())
3391 return;
3392
3393 switch (UserNode->getMachineOpcode()) {
3394 default:
3395 return;
3396 case AArch64::ANDSWri:
3397 case AArch64::ANDSXri:
3398 case AArch64::ANDWri:
3399 case AArch64::ANDXri:
3400 // We increment Depth only when we call the getUsefulBits
3401 return getUsefulBitsFromAndWithImmediate(SDValue(UserNode, 0), UsefulBits,
3402 Depth);
3403 case AArch64::UBFMWri:
3404 case AArch64::UBFMXri:
3405 return getUsefulBitsFromUBFM(SDValue(UserNode, 0), UsefulBits, Depth);
3406
3407 case AArch64::ORRWrs:
3408 case AArch64::ORRXrs:
3409 if (UserNode->getOperand(0) != Orig && UserNode->getOperand(1) == Orig)
3410 getUsefulBitsFromOrWithShiftedReg(SDValue(UserNode, 0), UsefulBits,
3411 Depth);
3412 return;
3413 case AArch64::BFMWri:
3414 case AArch64::BFMXri:
3415 return getUsefulBitsFromBFM(SDValue(UserNode, 0), Orig, UsefulBits, Depth);
3416
3417 case AArch64::STRBBui:
3418 case AArch64::STURBBi:
3419 if (UserNode->getOperand(0) != Orig)
3420 return;
3421 UsefulBits &= APInt(UsefulBits.getBitWidth(), 0xff);
3422 return;
3423
3424 case AArch64::STRHHui:
3425 case AArch64::STURHHi:
3426 if (UserNode->getOperand(0) != Orig)
3427 return;
3428 UsefulBits &= APInt(UsefulBits.getBitWidth(), 0xffff);
3429 return;
3430 }
3431}
3432
3433static void getUsefulBits(SDValue Op, APInt &UsefulBits, unsigned Depth) {
3435 return;
3436 // Initialize UsefulBits
3437 if (!Depth) {
3438 unsigned Bitwidth = Op.getScalarValueSizeInBits();
3439 // At the beginning, assume every produced bits is useful
3440 UsefulBits = APInt(Bitwidth, 0);
3441 UsefulBits.flipAllBits();
3442 }
3443 APInt UsersUsefulBits(UsefulBits.getBitWidth(), 0);
3444
3445 for (SDNode *Node : Op.getNode()->users()) {
3446 // A use cannot produce useful bits
3447 APInt UsefulBitsForUse = APInt(UsefulBits);
3448 getUsefulBitsForUse(Node, UsefulBitsForUse, Op, Depth);
3449 UsersUsefulBits |= UsefulBitsForUse;
3450 }
3451 // UsefulBits contains the produced bits that are meaningful for the
3452 // current definition, thus a user cannot make a bit meaningful at
3453 // this point
3454 UsefulBits &= UsersUsefulBits;
3455}
3456
3457/// Create a machine node performing a notional SHL of Op by ShlAmount. If
3458/// ShlAmount is negative, do a (logical) right-shift instead. If ShlAmount is
3459/// 0, return Op unchanged.
3460static SDValue getLeftShift(SelectionDAG *CurDAG, SDValue Op, int ShlAmount) {
3461 if (ShlAmount == 0)
3462 return Op;
3463
3464 EVT VT = Op.getValueType();
3465 SDLoc dl(Op);
3466 unsigned BitWidth = VT.getSizeInBits();
3467 unsigned UBFMOpc = BitWidth == 32 ? AArch64::UBFMWri : AArch64::UBFMXri;
3468
3469 SDNode *ShiftNode;
3470 if (ShlAmount > 0) {
3471 // LSL wD, wN, #Amt == UBFM wD, wN, #32-Amt, #31-Amt
3472 ShiftNode = CurDAG->getMachineNode(
3473 UBFMOpc, dl, VT, Op,
3474 CurDAG->getTargetConstant(BitWidth - ShlAmount, dl, VT),
3475 CurDAG->getTargetConstant(BitWidth - 1 - ShlAmount, dl, VT));
3476 } else {
3477 // LSR wD, wN, #Amt == UBFM wD, wN, #Amt, #32-1
3478 assert(ShlAmount < 0 && "expected right shift");
3479 int ShrAmount = -ShlAmount;
3480 ShiftNode = CurDAG->getMachineNode(
3481 UBFMOpc, dl, VT, Op, CurDAG->getTargetConstant(ShrAmount, dl, VT),
3482 CurDAG->getTargetConstant(BitWidth - 1, dl, VT));
3483 }
3484
3485 return SDValue(ShiftNode, 0);
3486}
3487
3488// For bit-field-positioning pattern "(and (shl VAL, N), ShiftedMask)".
3489static bool isBitfieldPositioningOpFromAnd(SelectionDAG *CurDAG, SDValue Op,
3490 bool BiggerPattern,
3491 const uint64_t NonZeroBits,
3492 SDValue &Src, int &DstLSB,
3493 int &Width);
3494
3495// For bit-field-positioning pattern "shl VAL, N)".
3496static bool isBitfieldPositioningOpFromShl(SelectionDAG *CurDAG, SDValue Op,
3497 bool BiggerPattern,
3498 const uint64_t NonZeroBits,
3499 SDValue &Src, int &DstLSB,
3500 int &Width);
3501
3502/// Does this tree qualify as an attempt to move a bitfield into position,
3503/// essentially "(and (shl VAL, N), Mask)" or (shl VAL, N).
3505 bool BiggerPattern, SDValue &Src,
3506 int &DstLSB, int &Width) {
3507 EVT VT = Op.getValueType();
3508 unsigned BitWidth = VT.getSizeInBits();
3509 (void)BitWidth;
3510 assert(BitWidth == 32 || BitWidth == 64);
3511
3513
3514 // Non-zero in the sense that they're not provably zero, which is the key
3515 // point if we want to use this value
3516 const uint64_t NonZeroBits = (~Known.Zero).getZExtValue();
3517 if (!isShiftedMask_64(NonZeroBits))
3518 return false;
3519
3520 switch (Op.getOpcode()) {
3521 default:
3522 break;
3523 case ISD::AND:
3524 return isBitfieldPositioningOpFromAnd(CurDAG, Op, BiggerPattern,
3525 NonZeroBits, Src, DstLSB, Width);
3526 case ISD::SHL:
3527 return isBitfieldPositioningOpFromShl(CurDAG, Op, BiggerPattern,
3528 NonZeroBits, Src, DstLSB, Width);
3529 }
3530
3531 return false;
3532}
3533
3535 bool BiggerPattern,
3536 const uint64_t NonZeroBits,
3537 SDValue &Src, int &DstLSB,
3538 int &Width) {
3539 assert(isShiftedMask_64(NonZeroBits) && "Caller guaranteed");
3540
3541 EVT VT = Op.getValueType();
3542 assert((VT == MVT::i32 || VT == MVT::i64) &&
3543 "Caller guarantees VT is one of i32 or i64");
3544 (void)VT;
3545
3546 uint64_t AndImm;
3547 if (!isOpcWithIntImmediate(Op.getNode(), ISD::AND, AndImm))
3548 return false;
3549
3550 // If (~AndImm & NonZeroBits) is not zero at POS, we know that
3551 // 1) (AndImm & (1 << POS) == 0)
3552 // 2) the result of AND is not zero at POS bit (according to NonZeroBits)
3553 //
3554 // 1) and 2) don't agree so something must be wrong (e.g., in
3555 // 'SelectionDAG::computeKnownBits')
3556 assert((~AndImm & NonZeroBits) == 0 &&
3557 "Something must be wrong (e.g., in SelectionDAG::computeKnownBits)");
3558
3559 SDValue AndOp0 = Op.getOperand(0);
3560
3561 uint64_t ShlImm;
3562 SDValue ShlOp0;
3563 if (isOpcWithIntImmediate(AndOp0.getNode(), ISD::SHL, ShlImm)) {
3564 // For pattern "and(shl(val, N), shifted-mask)", 'ShlOp0' is set to 'val'.
3565 ShlOp0 = AndOp0.getOperand(0);
3566 } else if (VT == MVT::i64 && AndOp0.getOpcode() == ISD::ANY_EXTEND &&
3568 ShlImm)) {
3569 // For pattern "and(any_extend(shl(val, N)), shifted-mask)"
3570
3571 // ShlVal == shl(val, N), which is a left shift on a smaller type.
3572 SDValue ShlVal = AndOp0.getOperand(0);
3573
3574 // Since this is after type legalization and ShlVal is extended to MVT::i64,
3575 // expect VT to be MVT::i32.
3576 assert((ShlVal.getValueType() == MVT::i32) && "Expect VT to be MVT::i32.");
3577
3578 // Widens 'val' to MVT::i64 as the source of bit field positioning.
3579 ShlOp0 = Widen(CurDAG, ShlVal.getOperand(0));
3580 } else
3581 return false;
3582
3583 // For !BiggerPattern, bail out if the AndOp0 has more than one use, since
3584 // then we'll end up generating AndOp0+UBFIZ instead of just keeping
3585 // AndOp0+AND.
3586 if (!BiggerPattern && !AndOp0.hasOneUse())
3587 return false;
3588
3589 DstLSB = llvm::countr_zero(NonZeroBits);
3590 Width = llvm::countr_one(NonZeroBits >> DstLSB);
3591
3592 // Bail out on large Width. This happens when no proper combining / constant
3593 // folding was performed.
3594 if (Width >= (int)VT.getSizeInBits()) {
3595 // If VT is i64, Width > 64 is insensible since NonZeroBits is uint64_t, and
3596 // Width == 64 indicates a missed dag-combine from "(and val, AllOnes)" to
3597 // "val".
3598 // If VT is i32, what Width >= 32 means:
3599 // - For "(and (any_extend(shl val, N)), shifted-mask)", the`and` Op
3600 // demands at least 'Width' bits (after dag-combiner). This together with
3601 // `any_extend` Op (undefined higher bits) indicates missed combination
3602 // when lowering the 'and' IR instruction to an machine IR instruction.
3603 LLVM_DEBUG(
3604 dbgs()
3605 << "Found large Width in bit-field-positioning -- this indicates no "
3606 "proper combining / constant folding was performed\n");
3607 return false;
3608 }
3609
3610 // BFI encompasses sufficiently many nodes that it's worth inserting an extra
3611 // LSL/LSR if the mask in NonZeroBits doesn't quite match up with the ISD::SHL
3612 // amount. BiggerPattern is true when this pattern is being matched for BFI,
3613 // BiggerPattern is false when this pattern is being matched for UBFIZ, in
3614 // which case it is not profitable to insert an extra shift.
3615 if (ShlImm != uint64_t(DstLSB) && !BiggerPattern)
3616 return false;
3617
3618 Src = getLeftShift(CurDAG, ShlOp0, ShlImm - DstLSB);
3619 return true;
3620}
3621
3622// For node (shl (and val, mask), N)), returns true if the node is equivalent to
3623// UBFIZ.
3625 SDValue &Src, int &DstLSB,
3626 int &Width) {
3627 // Caller should have verified that N is a left shift with constant shift
3628 // amount; asserts that.
3629 assert(Op.getOpcode() == ISD::SHL &&
3630 "Op.getNode() should be a SHL node to call this function");
3631 assert(isIntImmediateEq(Op.getOperand(1), ShlImm) &&
3632 "Op.getNode() should shift ShlImm to call this function");
3633
3634 uint64_t AndImm = 0;
3635 SDValue Op0 = Op.getOperand(0);
3636 if (!isOpcWithIntImmediate(Op0.getNode(), ISD::AND, AndImm))
3637 return false;
3638
3639 const uint64_t ShiftedAndImm = ((AndImm << ShlImm) >> ShlImm);
3640 if (isMask_64(ShiftedAndImm)) {
3641 // AndImm is a superset of (AllOnes >> ShlImm); in other words, AndImm
3642 // should end with Mask, and could be prefixed with random bits if those
3643 // bits are shifted out.
3644 //
3645 // For example, xyz11111 (with {x,y,z} being 0 or 1) is fine if ShlImm >= 3;
3646 // the AND result corresponding to those bits are shifted out, so it's fine
3647 // to not extract them.
3648 Width = llvm::countr_one(ShiftedAndImm);
3649 DstLSB = ShlImm;
3650 Src = Op0.getOperand(0);
3651 return true;
3652 }
3653 return false;
3654}
3655
3657 bool BiggerPattern,
3658 const uint64_t NonZeroBits,
3659 SDValue &Src, int &DstLSB,
3660 int &Width) {
3661 assert(isShiftedMask_64(NonZeroBits) && "Caller guaranteed");
3662
3663 EVT VT = Op.getValueType();
3664 assert((VT == MVT::i32 || VT == MVT::i64) &&
3665 "Caller guarantees that type is i32 or i64");
3666 (void)VT;
3667
3668 uint64_t ShlImm;
3669 if (!isOpcWithIntImmediate(Op.getNode(), ISD::SHL, ShlImm))
3670 return false;
3671
3672 if (!BiggerPattern && !Op.hasOneUse())
3673 return false;
3674
3675 if (isSeveralBitsPositioningOpFromShl(ShlImm, Op, Src, DstLSB, Width))
3676 return true;
3677
3678 DstLSB = llvm::countr_zero(NonZeroBits);
3679 Width = llvm::countr_one(NonZeroBits >> DstLSB);
3680
3681 if (ShlImm != uint64_t(DstLSB) && !BiggerPattern)
3682 return false;
3683
3684 Src = getLeftShift(CurDAG, Op.getOperand(0), ShlImm - DstLSB);
3685 return true;
3686}
3687
3688static bool isShiftedMask(uint64_t Mask, EVT VT) {
3689 assert(VT == MVT::i32 || VT == MVT::i64);
3690 if (VT == MVT::i32)
3691 return isShiftedMask_32(Mask);
3692 return isShiftedMask_64(Mask);
3693}
3694
3695// Generate a BFI/BFXIL from 'or (and X, MaskImm), OrImm' iff the value being
3696// inserted only sets known zero bits.
3698 assert(N->getOpcode() == ISD::OR && "Expect a OR operation");
3699
3700 EVT VT = N->getValueType(0);
3701 if (VT != MVT::i32 && VT != MVT::i64)
3702 return false;
3703
3704 unsigned BitWidth = VT.getSizeInBits();
3705
3706 uint64_t OrImm;
3707 if (!isOpcWithIntImmediate(N, ISD::OR, OrImm))
3708 return false;
3709
3710 // Skip this transformation if the ORR immediate can be encoded in the ORR.
3711 // Otherwise, we'll trade an AND+ORR for ORR+BFI/BFXIL, which is most likely
3712 // performance neutral.
3714 return false;
3715
3716 uint64_t MaskImm;
3717 SDValue And = N->getOperand(0);
3718 // Must be a single use AND with an immediate operand.
3719 if (!And.hasOneUse() ||
3720 !isOpcWithIntImmediate(And.getNode(), ISD::AND, MaskImm))
3721 return false;
3722
3723 // Compute the Known Zero for the AND as this allows us to catch more general
3724 // cases than just looking for AND with imm.
3726
3727 // Non-zero in the sense that they're not provably zero, which is the key
3728 // point if we want to use this value.
3729 uint64_t NotKnownZero = (~Known.Zero).getZExtValue();
3730
3731 // The KnownZero mask must be a shifted mask (e.g., 1110..011, 11100..00).
3732 if (!isShiftedMask(Known.Zero.getZExtValue(), VT))
3733 return false;
3734
3735 // The bits being inserted must only set those bits that are known to be zero.
3736 if ((OrImm & NotKnownZero) != 0) {
3737 // FIXME: It's okay if the OrImm sets NotKnownZero bits to 1, but we don't
3738 // currently handle this case.
3739 return false;
3740 }
3741
3742 // BFI/BFXIL dst, src, #lsb, #width.
3743 int LSB = llvm::countr_one(NotKnownZero);
3744 int Width = BitWidth - APInt(BitWidth, NotKnownZero).popcount();
3745
3746 // BFI/BFXIL is an alias of BFM, so translate to BFM operands.
3747 unsigned ImmR = (BitWidth - LSB) % BitWidth;
3748 unsigned ImmS = Width - 1;
3749
3750 // If we're creating a BFI instruction avoid cases where we need more
3751 // instructions to materialize the BFI constant as compared to the original
3752 // ORR. A BFXIL will use the same constant as the original ORR, so the code
3753 // should be no worse in this case.
3754 bool IsBFI = LSB != 0;
3755 uint64_t BFIImm = OrImm >> LSB;
3756 if (IsBFI && !AArch64_AM::isLogicalImmediate(BFIImm, BitWidth)) {
3757 // We have a BFI instruction and we know the constant can't be materialized
3758 // with a ORR-immediate with the zero register.
3759 unsigned OrChunks = 0, BFIChunks = 0;
3760 for (unsigned Shift = 0; Shift < BitWidth; Shift += 16) {
3761 if (((OrImm >> Shift) & 0xFFFF) != 0)
3762 ++OrChunks;
3763 if (((BFIImm >> Shift) & 0xFFFF) != 0)
3764 ++BFIChunks;
3765 }
3766 if (BFIChunks > OrChunks)
3767 return false;
3768 }
3769
3770 // Materialize the constant to be inserted.
3771 SDLoc DL(N);
3772 unsigned MOVIOpc = VT == MVT::i32 ? AArch64::MOVi32imm : AArch64::MOVi64imm;
3773 SDNode *MOVI = CurDAG->getMachineNode(
3774 MOVIOpc, DL, VT, CurDAG->getTargetConstant(BFIImm, DL, VT));
3775
3776 // Create the BFI/BFXIL instruction.
3777 SDValue Ops[] = {And.getOperand(0), SDValue(MOVI, 0),
3778 CurDAG->getTargetConstant(ImmR, DL, VT),
3779 CurDAG->getTargetConstant(ImmS, DL, VT)};
3780 unsigned Opc = (VT == MVT::i32) ? AArch64::BFMWri : AArch64::BFMXri;
3781 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
3782 return true;
3783}
3784
3786 SDValue &ShiftedOperand,
3787 uint64_t &EncodedShiftImm) {
3788 // Avoid folding Dst into ORR-with-shift if Dst has other uses than ORR.
3789 if (!Dst.hasOneUse())
3790 return false;
3791
3792 EVT VT = Dst.getValueType();
3793 assert((VT == MVT::i32 || VT == MVT::i64) &&
3794 "Caller should guarantee that VT is one of i32 or i64");
3795 const unsigned SizeInBits = VT.getSizeInBits();
3796
3797 SDLoc DL(Dst.getNode());
3798 uint64_t AndImm, ShlImm;
3799 if (isOpcWithIntImmediate(Dst.getNode(), ISD::AND, AndImm) &&
3800 isShiftedMask_64(AndImm)) {
3801 // Avoid transforming 'DstOp0' if it has other uses than the AND node.
3802 SDValue DstOp0 = Dst.getOperand(0);
3803 if (!DstOp0.hasOneUse())
3804 return false;
3805
3806 // An example to illustrate the transformation
3807 // From:
3808 // lsr x8, x1, #1
3809 // and x8, x8, #0x3f80
3810 // bfxil x8, x1, #0, #7
3811 // To:
3812 // and x8, x23, #0x7f
3813 // ubfx x9, x23, #8, #7
3814 // orr x23, x8, x9, lsl #7
3815 //
3816 // The number of instructions remains the same, but ORR is faster than BFXIL
3817 // on many AArch64 processors (or as good as BFXIL if not faster). Besides,
3818 // the dependency chain is improved after the transformation.
3819 uint64_t SrlImm;
3820 if (isOpcWithIntImmediate(DstOp0.getNode(), ISD::SRL, SrlImm)) {
3821 uint64_t NumTrailingZeroInShiftedMask = llvm::countr_zero(AndImm);
3822 if ((SrlImm + NumTrailingZeroInShiftedMask) < SizeInBits) {
3823 unsigned MaskWidth =
3824 llvm::countr_one(AndImm >> NumTrailingZeroInShiftedMask);
3825 unsigned UBFMOpc =
3826 (VT == MVT::i32) ? AArch64::UBFMWri : AArch64::UBFMXri;
3827 SDNode *UBFMNode = CurDAG->getMachineNode(
3828 UBFMOpc, DL, VT, DstOp0.getOperand(0),
3829 CurDAG->getTargetConstant(SrlImm + NumTrailingZeroInShiftedMask, DL,
3830 VT),
3831 CurDAG->getTargetConstant(
3832 SrlImm + NumTrailingZeroInShiftedMask + MaskWidth - 1, DL, VT));
3833 ShiftedOperand = SDValue(UBFMNode, 0);
3834 EncodedShiftImm = AArch64_AM::getShifterImm(
3835 AArch64_AM::LSL, NumTrailingZeroInShiftedMask);
3836 return true;
3837 }
3838 }
3839 return false;
3840 }
3841
3842 if (isOpcWithIntImmediate(Dst.getNode(), ISD::SHL, ShlImm)) {
3843 ShiftedOperand = Dst.getOperand(0);
3844 EncodedShiftImm = AArch64_AM::getShifterImm(AArch64_AM::LSL, ShlImm);
3845 return true;
3846 }
3847
3848 uint64_t SrlImm;
3849 if (isOpcWithIntImmediate(Dst.getNode(), ISD::SRL, SrlImm)) {
3850 ShiftedOperand = Dst.getOperand(0);
3851 EncodedShiftImm = AArch64_AM::getShifterImm(AArch64_AM::LSR, SrlImm);
3852 return true;
3853 }
3854 return false;
3855}
3856
3857// Given an 'ISD::OR' node that is going to be selected as BFM, analyze
3858// the operands and select it to AArch64::ORR with shifted registers if
3859// that's more efficient. Returns true iff selection to AArch64::ORR happens.
3860static bool tryOrrWithShift(SDNode *N, SDValue OrOpd0, SDValue OrOpd1,
3861 SDValue Src, SDValue Dst, SelectionDAG *CurDAG,
3862 const bool BiggerPattern) {
3863 EVT VT = N->getValueType(0);
3864 assert(N->getOpcode() == ISD::OR && "Expect N to be an OR node");
3865 assert(((N->getOperand(0) == OrOpd0 && N->getOperand(1) == OrOpd1) ||
3866 (N->getOperand(1) == OrOpd0 && N->getOperand(0) == OrOpd1)) &&
3867 "Expect OrOpd0 and OrOpd1 to be operands of ISD::OR");
3868 assert((VT == MVT::i32 || VT == MVT::i64) &&
3869 "Expect result type to be i32 or i64 since N is combinable to BFM");
3870 SDLoc DL(N);
3871
3872 // Bail out if BFM simplifies away one node in BFM Dst.
3873 if (OrOpd1 != Dst)
3874 return false;
3875
3876 const unsigned OrrOpc = (VT == MVT::i32) ? AArch64::ORRWrs : AArch64::ORRXrs;
3877 // For "BFM Rd, Rn, #immr, #imms", it's known that BFM simplifies away fewer
3878 // nodes from Rn (or inserts additional shift node) if BiggerPattern is true.
3879 if (BiggerPattern) {
3880 uint64_t SrcAndImm;
3881 if (isOpcWithIntImmediate(OrOpd0.getNode(), ISD::AND, SrcAndImm) &&
3882 isMask_64(SrcAndImm) && OrOpd0.getOperand(0) == Src) {
3883 // OrOpd0 = AND Src, #Mask
3884 // So BFM simplifies away one AND node from Src and doesn't simplify away
3885 // nodes from Dst. If ORR with left-shifted operand also simplifies away
3886 // one node (from Rd), ORR is better since it has higher throughput and
3887 // smaller latency than BFM on many AArch64 processors (and for the rest
3888 // ORR is at least as good as BFM).
3889 SDValue ShiftedOperand;
3890 uint64_t EncodedShiftImm;
3891 if (isWorthFoldingIntoOrrWithShift(Dst, CurDAG, ShiftedOperand,
3892 EncodedShiftImm)) {
3893 SDValue Ops[] = {OrOpd0, ShiftedOperand,
3894 CurDAG->getTargetConstant(EncodedShiftImm, DL, VT)};
3895 CurDAG->SelectNodeTo(N, OrrOpc, VT, Ops);
3896 return true;
3897 }
3898 }
3899 return false;
3900 }
3901
3902 assert((!BiggerPattern) && "BiggerPattern should be handled above");
3903
3904 uint64_t ShlImm;
3905 if (isOpcWithIntImmediate(OrOpd0.getNode(), ISD::SHL, ShlImm)) {
3906 if (OrOpd0.getOperand(0) == Src && OrOpd0.hasOneUse()) {
3907 SDValue Ops[] = {
3908 Dst, Src,
3909 CurDAG->getTargetConstant(
3911 CurDAG->SelectNodeTo(N, OrrOpc, VT, Ops);
3912 return true;
3913 }
3914
3915 // Select the following pattern to left-shifted operand rather than BFI.
3916 // %val1 = op ..
3917 // %val2 = shl %val1, #imm
3918 // %res = or %val1, %val2
3919 //
3920 // If N is selected to be BFI, we know that
3921 // 1) OrOpd0 would be the operand from which extract bits (i.e., folded into
3922 // BFI) 2) OrOpd1 would be the destination operand (i.e., preserved)
3923 //
3924 // Instead of selecting N to BFI, fold OrOpd0 as a left shift directly.
3925 if (OrOpd0.getOperand(0) == OrOpd1) {
3926 SDValue Ops[] = {
3927 OrOpd1, OrOpd1,
3928 CurDAG->getTargetConstant(
3930 CurDAG->SelectNodeTo(N, OrrOpc, VT, Ops);
3931 return true;
3932 }
3933 }
3934
3935 uint64_t SrlImm;
3936 if (isOpcWithIntImmediate(OrOpd0.getNode(), ISD::SRL, SrlImm)) {
3937 // Select the following pattern to right-shifted operand rather than BFXIL.
3938 // %val1 = op ..
3939 // %val2 = lshr %val1, #imm
3940 // %res = or %val1, %val2
3941 //
3942 // If N is selected to be BFXIL, we know that
3943 // 1) OrOpd0 would be the operand from which extract bits (i.e., folded into
3944 // BFXIL) 2) OrOpd1 would be the destination operand (i.e., preserved)
3945 //
3946 // Instead of selecting N to BFXIL, fold OrOpd0 as a right shift directly.
3947 if (OrOpd0.getOperand(0) == OrOpd1) {
3948 SDValue Ops[] = {
3949 OrOpd1, OrOpd1,
3950 CurDAG->getTargetConstant(
3952 CurDAG->SelectNodeTo(N, OrrOpc, VT, Ops);
3953 return true;
3954 }
3955 }
3956
3957 return false;
3958}
3959
3960static bool tryBitfieldInsertOpFromOr(SDNode *N, const APInt &UsefulBits,
3961 SelectionDAG *CurDAG) {
3962 assert(N->getOpcode() == ISD::OR && "Expect a OR operation");
3963
3964 EVT VT = N->getValueType(0);
3965 if (VT != MVT::i32 && VT != MVT::i64)
3966 return false;
3967
3968 unsigned BitWidth = VT.getSizeInBits();
3969
3970 // Because of simplify-demanded-bits in DAGCombine, involved masks may not
3971 // have the expected shape. Try to undo that.
3972
3973 unsigned NumberOfIgnoredLowBits = UsefulBits.countr_zero();
3974 unsigned NumberOfIgnoredHighBits = UsefulBits.countl_zero();
3975
3976 // Given a OR operation, check if we have the following pattern
3977 // ubfm c, b, imm, imm2 (or something that does the same jobs, see
3978 // isBitfieldExtractOp)
3979 // d = e & mask2 ; where mask is a binary sequence of 1..10..0 and
3980 // countTrailingZeros(mask2) == imm2 - imm + 1
3981 // f = d | c
3982 // if yes, replace the OR instruction with:
3983 // f = BFM Opd0, Opd1, LSB, MSB ; where LSB = imm, and MSB = imm2
3984
3985 // OR is commutative, check all combinations of operand order and values of
3986 // BiggerPattern, i.e.
3987 // Opd0, Opd1, BiggerPattern=false
3988 // Opd1, Opd0, BiggerPattern=false
3989 // Opd0, Opd1, BiggerPattern=true
3990 // Opd1, Opd0, BiggerPattern=true
3991 // Several of these combinations may match, so check with BiggerPattern=false
3992 // first since that will produce better results by matching more instructions
3993 // and/or inserting fewer extra instructions.
3994 for (int I = 0; I < 4; ++I) {
3995
3996 SDValue Dst, Src;
3997 unsigned ImmR, ImmS;
3998 bool BiggerPattern = I / 2;
3999 SDValue OrOpd0Val = N->getOperand(I % 2);
4000 SDNode *OrOpd0 = OrOpd0Val.getNode();
4001 SDValue OrOpd1Val = N->getOperand((I + 1) % 2);
4002 SDNode *OrOpd1 = OrOpd1Val.getNode();
4003
4004 unsigned BFXOpc;
4005 int DstLSB, Width;
4006 if (isBitfieldExtractOp(CurDAG, OrOpd0, BFXOpc, Src, ImmR, ImmS,
4007 NumberOfIgnoredLowBits, BiggerPattern)) {
4008 // Check that the returned opcode is compatible with the pattern,
4009 // i.e., same type and zero extended (U and not S)
4010 if ((BFXOpc != AArch64::UBFMXri && VT == MVT::i64) ||
4011 (BFXOpc != AArch64::UBFMWri && VT == MVT::i32))
4012 continue;
4013
4014 // Compute the width of the bitfield insertion
4015 DstLSB = 0;
4016 Width = ImmS - ImmR + 1;
4017 // FIXME: This constraint is to catch bitfield insertion we may
4018 // want to widen the pattern if we want to grab general bitfield
4019 // move case
4020 if (Width <= 0)
4021 continue;
4022
4023 // If the mask on the insertee is correct, we have a BFXIL operation. We
4024 // can share the ImmR and ImmS values from the already-computed UBFM.
4025 } else if (isBitfieldPositioningOp(CurDAG, OrOpd0Val,
4026 BiggerPattern,
4027 Src, DstLSB, Width)) {
4028 ImmR = (BitWidth - DstLSB) % BitWidth;
4029 ImmS = Width - 1;
4030 } else
4031 continue;
4032
4033 // Check the second part of the pattern
4034 EVT VT = OrOpd1Val.getValueType();
4035 assert((VT == MVT::i32 || VT == MVT::i64) && "unexpected OR operand");
4036
4037 // Compute the Known Zero for the candidate of the first operand.
4038 // This allows to catch more general case than just looking for
4039 // AND with imm. Indeed, simplify-demanded-bits may have removed
4040 // the AND instruction because it proves it was useless.
4041 KnownBits Known = CurDAG->computeKnownBits(OrOpd1Val);
4042
4043 // Check if there is enough room for the second operand to appear
4044 // in the first one
4045 APInt BitsToBeInserted =
4046 APInt::getBitsSet(Known.getBitWidth(), DstLSB, DstLSB + Width);
4047
4048 if ((BitsToBeInserted & ~Known.Zero) != 0)
4049 continue;
4050
4051 // Set the first operand
4052 uint64_t Imm;
4053 if (isOpcWithIntImmediate(OrOpd1, ISD::AND, Imm) &&
4054 isBitfieldDstMask(Imm, BitsToBeInserted, NumberOfIgnoredHighBits, VT))
4055 // In that case, we can eliminate the AND
4056 Dst = OrOpd1->getOperand(0);
4057 else
4058 // Maybe the AND has been removed by simplify-demanded-bits
4059 // or is useful because it discards more bits
4060 Dst = OrOpd1Val;
4061
4062 // Before selecting ISD::OR node to AArch64::BFM, see if an AArch64::ORR
4063 // with shifted operand is more efficient.
4064 if (tryOrrWithShift(N, OrOpd0Val, OrOpd1Val, Src, Dst, CurDAG,
4065 BiggerPattern))
4066 return true;
4067
4068 // both parts match
4069 SDLoc DL(N);
4070 SDValue Ops[] = {Dst, Src, CurDAG->getTargetConstant(ImmR, DL, VT),
4071 CurDAG->getTargetConstant(ImmS, DL, VT)};
4072 unsigned Opc = (VT == MVT::i32) ? AArch64::BFMWri : AArch64::BFMXri;
4073 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
4074 return true;
4075 }
4076
4077 // Generate a BFXIL from 'or (and X, Mask0Imm), (and Y, Mask1Imm)' iff
4078 // Mask0Imm and ~Mask1Imm are equivalent and one of the MaskImms is a shifted
4079 // mask (e.g., 0x000ffff0).
4080 uint64_t Mask0Imm, Mask1Imm;
4081 SDValue And0 = N->getOperand(0);
4082 SDValue And1 = N->getOperand(1);
4083 if (And0.hasOneUse() && And1.hasOneUse() &&
4084 isOpcWithIntImmediate(And0.getNode(), ISD::AND, Mask0Imm) &&
4085 isOpcWithIntImmediate(And1.getNode(), ISD::AND, Mask1Imm) &&
4086 APInt(BitWidth, Mask0Imm) == ~APInt(BitWidth, Mask1Imm) &&
4087 (isShiftedMask(Mask0Imm, VT) || isShiftedMask(Mask1Imm, VT))) {
4088
4089 // ORR is commutative, so canonicalize to the form 'or (and X, Mask0Imm),
4090 // (and Y, Mask1Imm)' where Mask1Imm is the shifted mask masking off the
4091 // bits to be inserted.
4092 if (isShiftedMask(Mask0Imm, VT)) {
4093 std::swap(And0, And1);
4094 std::swap(Mask0Imm, Mask1Imm);
4095 }
4096
4097 SDValue Src = And1->getOperand(0);
4098 SDValue Dst = And0->getOperand(0);
4099 unsigned LSB = llvm::countr_zero(Mask1Imm);
4100 int Width = BitWidth - APInt(BitWidth, Mask0Imm).popcount();
4101
4102 // The BFXIL inserts the low-order bits from a source register, so right
4103 // shift the needed bits into place.
4104 SDLoc DL(N);
4105 unsigned ShiftOpc = (VT == MVT::i32) ? AArch64::UBFMWri : AArch64::UBFMXri;
4106 uint64_t LsrImm = LSB;
4107 if (Src->hasOneUse() &&
4108 isOpcWithIntImmediate(Src.getNode(), ISD::SRL, LsrImm) &&
4109 (LsrImm + LSB) < BitWidth) {
4110 Src = Src->getOperand(0);
4111 LsrImm += LSB;
4112 }
4113
4114 SDNode *LSR = CurDAG->getMachineNode(
4115 ShiftOpc, DL, VT, Src, CurDAG->getTargetConstant(LsrImm, DL, VT),
4116 CurDAG->getTargetConstant(BitWidth - 1, DL, VT));
4117
4118 // BFXIL is an alias of BFM, so translate to BFM operands.
4119 unsigned ImmR = (BitWidth - LSB) % BitWidth;
4120 unsigned ImmS = Width - 1;
4121
4122 // Create the BFXIL instruction.
4123 SDValue Ops[] = {Dst, SDValue(LSR, 0),
4124 CurDAG->getTargetConstant(ImmR, DL, VT),
4125 CurDAG->getTargetConstant(ImmS, DL, VT)};
4126 unsigned Opc = (VT == MVT::i32) ? AArch64::BFMWri : AArch64::BFMXri;
4127 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
4128 return true;
4129 }
4130
4131 return false;
4132}
4133
4134bool AArch64DAGToDAGISel::tryBitfieldInsertOp(SDNode *N) {
4135 if (N->getOpcode() != ISD::OR)
4136 return false;
4137
4138 APInt NUsefulBits;
4139 getUsefulBits(SDValue(N, 0), NUsefulBits);
4140
4141 // If all bits are not useful, just return UNDEF.
4142 if (!NUsefulBits) {
4143 CurDAG->SelectNodeTo(N, TargetOpcode::IMPLICIT_DEF, N->getValueType(0));
4144 return true;
4145 }
4146
4147 if (tryBitfieldInsertOpFromOr(N, NUsefulBits, CurDAG))
4148 return true;
4149
4150 return tryBitfieldInsertOpFromOrAndImm(N, CurDAG);
4151}
4152
4153/// SelectBitfieldInsertInZeroOp - Match a UBFIZ instruction that is the
4154/// equivalent of a left shift by a constant amount followed by an and masking
4155/// out a contiguous set of bits.
4156bool AArch64DAGToDAGISel::tryBitfieldInsertInZeroOp(SDNode *N) {
4157 if (N->getOpcode() != ISD::AND)
4158 return false;
4159
4160 EVT VT = N->getValueType(0);
4161 if (VT != MVT::i32 && VT != MVT::i64)
4162 return false;
4163
4164 SDValue Op0;
4165 int DstLSB, Width;
4166 if (!isBitfieldPositioningOp(CurDAG, SDValue(N, 0), /*BiggerPattern=*/false,
4167 Op0, DstLSB, Width))
4168 return false;
4169
4170 // ImmR is the rotate right amount.
4171 unsigned ImmR = (VT.getSizeInBits() - DstLSB) % VT.getSizeInBits();
4172 // ImmS is the most significant bit of the source to be moved.
4173 unsigned ImmS = Width - 1;
4174
4175 SDLoc DL(N);
4176 SDValue Ops[] = {Op0, CurDAG->getTargetConstant(ImmR, DL, VT),
4177 CurDAG->getTargetConstant(ImmS, DL, VT)};
4178 unsigned Opc = (VT == MVT::i32) ? AArch64::UBFMWri : AArch64::UBFMXri;
4179 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
4180 return true;
4181}
4182
4183/// tryShiftAmountMod - Take advantage of built-in mod of shift amount in
4184/// variable shift/rotate instructions.
4185bool AArch64DAGToDAGISel::tryShiftAmountMod(SDNode *N) {
4186 EVT VT = N->getValueType(0);
4187
4188 unsigned Opc;
4189 switch (N->getOpcode()) {
4190 case ISD::ROTR:
4191 Opc = (VT == MVT::i32) ? AArch64::RORVWr : AArch64::RORVXr;
4192 break;
4193 case ISD::SHL:
4194 Opc = (VT == MVT::i32) ? AArch64::LSLVWr : AArch64::LSLVXr;
4195 break;
4196 case ISD::SRL:
4197 Opc = (VT == MVT::i32) ? AArch64::LSRVWr : AArch64::LSRVXr;
4198 break;
4199 case ISD::SRA:
4200 Opc = (VT == MVT::i32) ? AArch64::ASRVWr : AArch64::ASRVXr;
4201 break;
4202 default:
4203 return false;
4204 }
4205
4206 uint64_t Size;
4207 uint64_t Bits;
4208 if (VT == MVT::i32) {
4209 Bits = 5;
4210 Size = 32;
4211 } else if (VT == MVT::i64) {
4212 Bits = 6;
4213 Size = 64;
4214 } else
4215 return false;
4216
4217 SDValue ShiftAmt = N->getOperand(1);
4218 SDLoc DL(N);
4219 SDValue NewShiftAmt;
4220
4221 // Skip over an extend of the shift amount.
4222 if (ShiftAmt->getOpcode() == ISD::ZERO_EXTEND ||
4223 ShiftAmt->getOpcode() == ISD::ANY_EXTEND)
4224 ShiftAmt = ShiftAmt->getOperand(0);
4225
4226 if (ShiftAmt->getOpcode() == ISD::ADD || ShiftAmt->getOpcode() == ISD::SUB) {
4227 SDValue Add0 = ShiftAmt->getOperand(0);
4228 SDValue Add1 = ShiftAmt->getOperand(1);
4229 uint64_t Add0Imm;
4230 uint64_t Add1Imm;
4231 if (isIntImmediate(Add1, Add1Imm) && (Add1Imm % Size == 0)) {
4232 // If we are shifting by X+/-N where N == 0 mod Size, then just shift by X
4233 // to avoid the ADD/SUB.
4234 NewShiftAmt = Add0;
4235 } else if (ShiftAmt->getOpcode() == ISD::SUB &&
4236 isIntImmediate(Add0, Add0Imm) && Add0Imm != 0 &&
4237 (Add0Imm % Size == 0)) {
4238 // If we are shifting by N-X where N == 0 mod Size, then just shift by -X
4239 // to generate a NEG instead of a SUB from a constant.
4240 unsigned NegOpc;
4241 unsigned ZeroReg;
4242 EVT SubVT = ShiftAmt->getValueType(0);
4243 if (SubVT == MVT::i32) {
4244 NegOpc = AArch64::SUBWrr;
4245 ZeroReg = AArch64::WZR;
4246 } else {
4247 assert(SubVT == MVT::i64);
4248 NegOpc = AArch64::SUBXrr;
4249 ZeroReg = AArch64::XZR;
4250 }
4251 SDValue Zero =
4252 CurDAG->getCopyFromReg(CurDAG->getEntryNode(), DL, ZeroReg, SubVT);
4253 MachineSDNode *Neg =
4254 CurDAG->getMachineNode(NegOpc, DL, SubVT, Zero, Add1);
4255 NewShiftAmt = SDValue(Neg, 0);
4256 } else if (ShiftAmt->getOpcode() == ISD::SUB &&
4257 isIntImmediate(Add0, Add0Imm) && (Add0Imm % Size == Size - 1)) {
4258 // If we are shifting by N-X where N == -1 mod Size, then just shift by ~X
4259 // to generate a NOT instead of a SUB from a constant.
4260 unsigned NotOpc;
4261 unsigned ZeroReg;
4262 EVT SubVT = ShiftAmt->getValueType(0);
4263 if (SubVT == MVT::i32) {
4264 NotOpc = AArch64::ORNWrr;
4265 ZeroReg = AArch64::WZR;
4266 } else {
4267 assert(SubVT == MVT::i64);
4268 NotOpc = AArch64::ORNXrr;
4269 ZeroReg = AArch64::XZR;
4270 }
4271 SDValue Zero =
4272 CurDAG->getCopyFromReg(CurDAG->getEntryNode(), DL, ZeroReg, SubVT);
4273 MachineSDNode *Not =
4274 CurDAG->getMachineNode(NotOpc, DL, SubVT, Zero, Add1);
4275 NewShiftAmt = SDValue(Not, 0);
4276 } else
4277 return false;
4278 } else {
4279 // If the shift amount is masked with an AND, check that the mask covers the
4280 // bits that are implicitly ANDed off by the above opcodes and if so, skip
4281 // the AND.
4282 uint64_t MaskImm;
4283 if (!isOpcWithIntImmediate(ShiftAmt.getNode(), ISD::AND, MaskImm) &&
4284 !isOpcWithIntImmediate(ShiftAmt.getNode(), AArch64ISD::ANDS, MaskImm))
4285 return false;
4286
4287 if ((unsigned)llvm::countr_one(MaskImm) < Bits)
4288 return false;
4289
4290 NewShiftAmt = ShiftAmt->getOperand(0);
4291 }
4292
4293 // Narrow/widen the shift amount to match the size of the shift operation.
4294 if (VT == MVT::i32)
4295 NewShiftAmt = narrowIfNeeded(CurDAG, NewShiftAmt);
4296 else if (VT == MVT::i64 && NewShiftAmt->getValueType(0) == MVT::i32) {
4297 SDValue SubReg = CurDAG->getTargetConstant(AArch64::sub_32, DL, MVT::i32);
4298 MachineSDNode *Ext = CurDAG->getMachineNode(AArch64::SUBREG_TO_REG, DL, VT,
4299 NewShiftAmt, SubReg);
4300 NewShiftAmt = SDValue(Ext, 0);
4301 }
4302
4303 SDValue Ops[] = {N->getOperand(0), NewShiftAmt};
4304 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
4305 return true;
4306}
4307
4309 SDValue &FixedPos,
4310 unsigned RegWidth,
4311 bool isReciprocal) {
4312 APFloat FVal(0.0);
4314 FVal = CN->getValueAPF();
4315 else if (LoadSDNode *LN = dyn_cast<LoadSDNode>(N)) {
4316 // Some otherwise illegal constants are allowed in this case.
4317 if (LN->getOperand(1).getOpcode() != AArch64ISD::ADDlow ||
4318 !isa<ConstantPoolSDNode>(LN->getOperand(1)->getOperand(1)))
4319 return false;
4320
4321 ConstantPoolSDNode *CN =
4322 dyn_cast<ConstantPoolSDNode>(LN->getOperand(1)->getOperand(1));
4323 FVal = cast<ConstantFP>(CN->getConstVal())->getValueAPF();
4324 } else
4325 return false;
4326
4327 if (unsigned FBits =
4328 CheckFixedPointOperandConstant(FVal, RegWidth, isReciprocal)) {
4329 FixedPos = CurDAG->getTargetConstant(FBits, SDLoc(N), MVT::i32);
4330 return true;
4331 }
4332
4333 return false;
4334}
4335
4337 SDValue N,
4338 SDValue &FixedPos,
4339 unsigned RegWidth,
4340 bool isReciprocal) {
4341 if ((N.getOpcode() == AArch64ISD::NVCAST || N.getOpcode() == ISD::BITCAST) &&
4342 N.getValueType().getScalarSizeInBits() ==
4343 N.getOperand(0).getValueType().getScalarSizeInBits())
4344 N = N.getOperand(0);
4345
4346 auto ImmToFloat = [RegWidth](APInt Imm) {
4347 switch (RegWidth) {
4348 case 16:
4349 return APFloat(APFloat::IEEEhalf(), Imm);
4350 case 32:
4351 return APFloat(APFloat::IEEEsingle(), Imm);
4352 case 64:
4353 return APFloat(APFloat::IEEEdouble(), Imm);
4354 default:
4355 llvm_unreachable("Unexpected RegWidth!");
4356 };
4357 };
4358
4359 APFloat FVal(0.0);
4360 switch (N->getOpcode()) {
4361 case AArch64ISD::MOVIshift:
4362 FVal = ImmToFloat(APInt(RegWidth, N.getConstantOperandVal(0)
4363 << N.getConstantOperandVal(1)));
4364 break;
4365 case AArch64ISD::FMOV:
4366 FVal = ImmToFloat(DecodeFMOVImm(N.getConstantOperandVal(0), RegWidth));
4367 break;
4368 case AArch64ISD::DUP:
4369 if (isa<ConstantSDNode>(N.getOperand(0)))
4370 FVal = ImmToFloat(N.getConstantOperandAPInt(0).trunc(RegWidth));
4371 else
4372 return false;
4373 break;
4374 default:
4375 return false;
4376 }
4377
4378 if (unsigned FBits =
4379 CheckFixedPointOperandConstant(FVal, RegWidth, isReciprocal)) {
4380 FixedPos = CurDAG->getTargetConstant(FBits, SDLoc(N), MVT::i32);
4381 return true;
4382 }
4383
4384 return false;
4385}
4386
4387bool AArch64DAGToDAGISel::SelectCVTFixedPosOperand(SDValue N, SDValue &FixedPos,
4388 unsigned RegWidth) {
4389 return checkCVTFixedPointOperandWithFBits(CurDAG, N, FixedPos, RegWidth,
4390 /*isReciprocal*/ false);
4391}
4392
4393bool AArch64DAGToDAGISel::SelectCVTFixedPointVec(SDValue N, SDValue &FixedPos,
4394 unsigned RegWidth) {
4396 CurDAG, N, FixedPos, RegWidth, /*isReciprocal*/ false);
4397}
4398
4399bool AArch64DAGToDAGISel::SelectCVTFixedPosRecipOperandVec(SDValue N,
4400 SDValue &FixedPos,
4401 unsigned RegWidth) {
4403 CurDAG, N, FixedPos, RegWidth, /*isReciprocal*/ true);
4404}
4405
4406bool AArch64DAGToDAGISel::SelectCVTFixedPosRecipOperand(SDValue N,
4407 SDValue &FixedPos,
4408 unsigned RegWidth) {
4409 return checkCVTFixedPointOperandWithFBits(CurDAG, N, FixedPos, RegWidth,
4410 /*isReciprocal*/ true);
4411}
4412
4413// Inspects a register string of the form o0:op1:CRn:CRm:op2 gets the fields
4414// of the string and obtains the integer values from them and combines these
4415// into a single value to be used in the MRS/MSR instruction.
4418 RegString.split(Fields, ':');
4419
4420 if (Fields.size() == 1)
4421 return -1;
4422
4423 assert(Fields.size() == 5
4424 && "Invalid number of fields in read register string");
4425
4427 bool AllIntFields = true;
4428
4429 for (StringRef Field : Fields) {
4430 unsigned IntField;
4431 AllIntFields &= !Field.getAsInteger(10, IntField);
4432 Ops.push_back(IntField);
4433 }
4434
4435 assert(AllIntFields &&
4436 "Unexpected non-integer value in special register string.");
4437 (void)AllIntFields;
4438
4439 // Need to combine the integer fields of the string into a single value
4440 // based on the bit encoding of MRS/MSR instruction.
4441 return (Ops[0] << 14) | (Ops[1] << 11) | (Ops[2] << 7) | (Ops[3] << 3) |
4442 (Ops[4]);
4443}
4444
4445// Lower the read_register intrinsic to an MRS instruction node if the special
4446// register string argument is either of the form detailed in the ALCE (the
4447// form described in getIntOperandsFromRegisterString) or is a named register
4448// known by the MRS SysReg mapper.
4449bool AArch64DAGToDAGISel::tryReadRegister(SDNode *N) {
4450 const auto *MD = cast<MDNodeSDNode>(N->getOperand(1));
4451 const auto *RegString = cast<MDString>(MD->getMD()->getOperand(0));
4452 SDLoc DL(N);
4453
4454 bool ReadIs128Bit = N->getOpcode() == AArch64ISD::MRRS;
4455
4456 unsigned Opcode64Bit = AArch64::MRS;
4457 int Imm = getIntOperandFromRegisterString(RegString->getString());
4458 if (Imm == -1) {
4459 // No match, Use the sysreg mapper to map the remaining possible strings to
4460 // the value for the register to be used for the instruction operand.
4461 const auto *TheReg =
4462 AArch64SysReg::lookupSysRegByName(RegString->getString());
4463 if (TheReg && TheReg->Readable &&
4464 TheReg->haveFeatures(Subtarget->getFeatureBits()))
4465 Imm = TheReg->Encoding;
4466 else
4467 Imm = AArch64SysReg::parseGenericRegister(RegString->getString());
4468
4469 if (Imm == -1) {
4470 // Still no match, see if this is "pc" or give up.
4471 if (!ReadIs128Bit && RegString->getString() == "pc") {
4472 Opcode64Bit = AArch64::ADR;
4473 Imm = 0;
4474 } else {
4475 // Not a system register. It may name an allocatable 64-bit GPR/FPR read
4476 // by the MSVC __getReg/__getRegFp intrinsics. Emit a pseudo that
4477 // carries the source register as an immediate so the read does not
4478 // reference an undefined physical register (which the machine verifier
4479 // rejects); the AsmPrinter materializes the real mov/fmov.
4480 Register PReg = Subtarget->getTargetLowering()->matchRegisterName(
4481 RegString->getString());
4482 unsigned PseudoOp = 0;
4483 if (AArch64::GPR64RegClass.contains(PReg))
4484 PseudoOp = AArch64::READ_REGISTER_GPR64;
4485 else if (AArch64::FPR64RegClass.contains(PReg))
4486 PseudoOp = AArch64::READ_REGISTER_FPR64;
4487 if (!ReadIs128Bit && PseudoOp && N->getValueType(0) == MVT::i64) {
4488 CurDAG->SelectNodeTo(N, PseudoOp, MVT::i64, MVT::Other,
4489 {CurDAG->getTargetConstant(PReg, DL, MVT::i32),
4490 N->getOperand(0)});
4491 return true;
4492 }
4493 return false;
4494 }
4495 }
4496 }
4497
4498 SDValue InChain = N->getOperand(0);
4499 SDValue SysRegImm = CurDAG->getTargetConstant(Imm, DL, MVT::i32);
4500 if (!ReadIs128Bit) {
4501 CurDAG->SelectNodeTo(N, Opcode64Bit, MVT::i64, MVT::Other /* Chain */,
4502 {SysRegImm, InChain});
4503 } else {
4504 SDNode *MRRS = CurDAG->getMachineNode(
4505 AArch64::MRRS, DL,
4506 {MVT::Untyped /* XSeqPair */, MVT::Other /* Chain */},
4507 {SysRegImm, InChain});
4508
4509 // Sysregs are not endian. The even register always contains the low half
4510 // of the register.
4511 SDValue Lo = CurDAG->getTargetExtractSubreg(AArch64::sube64, DL, MVT::i64,
4512 SDValue(MRRS, 0));
4513 SDValue Hi = CurDAG->getTargetExtractSubreg(AArch64::subo64, DL, MVT::i64,
4514 SDValue(MRRS, 0));
4515 SDValue OutChain = SDValue(MRRS, 1);
4516
4517 ReplaceUses(SDValue(N, 0), Lo);
4518 ReplaceUses(SDValue(N, 1), Hi);
4519 ReplaceUses(SDValue(N, 2), OutChain);
4520 };
4521 return true;
4522}
4523
4524// Lower the write_register intrinsic to an MSR instruction node if the special
4525// register string argument is either of the form detailed in the ALCE (the
4526// form described in getIntOperandsFromRegisterString) or is a named register
4527// known by the MSR SysReg mapper.
4528bool AArch64DAGToDAGISel::tryWriteRegister(SDNode *N) {
4529 const auto *MD = cast<MDNodeSDNode>(N->getOperand(1));
4530 const auto *RegString = cast<MDString>(MD->getMD()->getOperand(0));
4531 SDLoc DL(N);
4532
4533 bool WriteIs128Bit = N->getOpcode() == AArch64ISD::MSRR;
4534
4535 if (!WriteIs128Bit) {
4536 // Check if the register was one of those allowed as the pstatefield value
4537 // in the MSR (immediate) instruction. To accept the values allowed in the
4538 // pstatefield for the MSR (immediate) instruction, we also require that an
4539 // immediate value has been provided as an argument, we know that this is
4540 // the case as it has been ensured by semantic checking.
4541 auto trySelectPState = [&](auto PMapper, unsigned State) {
4542 if (PMapper) {
4543 assert(isa<ConstantSDNode>(N->getOperand(2)) &&
4544 "Expected a constant integer expression.");
4545 unsigned Reg = PMapper->Encoding;
4546 uint64_t Immed = N->getConstantOperandVal(2);
4547 CurDAG->SelectNodeTo(
4548 N, State, MVT::Other, CurDAG->getTargetConstant(Reg, DL, MVT::i32),
4549 CurDAG->getTargetConstant(Immed, DL, MVT::i16), N->getOperand(0));
4550 return true;
4551 }
4552 return false;
4553 };
4554
4555 if (trySelectPState(
4556 AArch64PState::lookupPStateImm0_15ByName(RegString->getString()),
4557 AArch64::MSRpstateImm4))
4558 return true;
4559 if (trySelectPState(
4560 AArch64PState::lookupPStateImm0_1ByName(RegString->getString()),
4561 AArch64::MSRpstateImm1))
4562 return true;
4563 }
4564
4565 int Imm = getIntOperandFromRegisterString(RegString->getString());
4566 if (Imm == -1) {
4567 // Use the sysreg mapper to attempt to map the remaining possible strings
4568 // to the value for the register to be used for the MSR (register)
4569 // instruction operand.
4570 auto TheReg = AArch64SysReg::lookupSysRegByName(RegString->getString());
4571 if (TheReg && TheReg->Writeable &&
4572 TheReg->haveFeatures(Subtarget->getFeatureBits()))
4573 Imm = TheReg->Encoding;
4574 else
4575 Imm = AArch64SysReg::parseGenericRegister(RegString->getString());
4576
4577 if (Imm == -1) {
4578 // Used by the MSVC __setReg/__setRegFp intrinsics. Copy the value into
4579 // the physical register and keep it live with a FAKE_USE so the write is
4580 // not dead-eliminated. (getRegisterByName rejects allocatable registers,
4581 // so the generic write path cannot handle these.)
4582 Register PReg = Subtarget->getTargetLowering()->matchRegisterName(
4583 RegString->getString());
4584 bool IsGPR = AArch64::GPR64RegClass.contains(PReg);
4585 bool IsFPR = AArch64::FPR64RegClass.contains(PReg);
4586 if (!WriteIs128Bit && (IsGPR || IsFPR) &&
4587 N->getOperand(2).getValueType() == MVT::i64) {
4588 SDValue Copy =
4589 CurDAG->getCopyToReg(N->getOperand(0), DL, PReg, N->getOperand(2));
4590 SDValue RegOp = CurDAG->getRegister(PReg, MVT::i64);
4591 SDNode *FakeUse = CurDAG->getMachineNode(TargetOpcode::FAKE_USE, DL,
4592 MVT::Other, {RegOp, Copy});
4593 ReplaceUses(SDValue(N, 0), SDValue(FakeUse, 0));
4594 CurDAG->RemoveDeadNode(N);
4595 return true;
4596 }
4597 return false;
4598 }
4599 }
4600
4601 SDValue InChain = N->getOperand(0);
4602 if (!WriteIs128Bit) {
4603 CurDAG->SelectNodeTo(N, AArch64::MSR, MVT::Other,
4604 CurDAG->getTargetConstant(Imm, DL, MVT::i32),
4605 N->getOperand(2), InChain);
4606 } else {
4607 // No endian swap. The lower half always goes into the even subreg, and the
4608 // higher half always into the odd supreg.
4609 SDNode *Pair = CurDAG->getMachineNode(
4610 TargetOpcode::REG_SEQUENCE, DL, MVT::Untyped /* XSeqPair */,
4611 {CurDAG->getTargetConstant(AArch64::XSeqPairsClassRegClass.getID(), DL,
4612 MVT::i32),
4613 N->getOperand(2),
4614 CurDAG->getTargetConstant(AArch64::sube64, DL, MVT::i32),
4615 N->getOperand(3),
4616 CurDAG->getTargetConstant(AArch64::subo64, DL, MVT::i32)});
4617
4618 CurDAG->SelectNodeTo(N, AArch64::MSRR, MVT::Other,
4619 CurDAG->getTargetConstant(Imm, DL, MVT::i32),
4620 SDValue(Pair, 0), InChain);
4621 }
4622
4623 return true;
4624}
4625
4626/// We've got special pseudo-instructions for these
4627bool AArch64DAGToDAGISel::SelectCMP_SWAP(SDNode *N) {
4628 unsigned Opcode;
4629 EVT MemTy = cast<MemSDNode>(N)->getMemoryVT();
4630
4631 // Leave IR for LSE if subtarget supports it.
4632 if (Subtarget->hasLSE()) return false;
4633
4634 if (MemTy == MVT::i8)
4635 Opcode = AArch64::CMP_SWAP_8;
4636 else if (MemTy == MVT::i16)
4637 Opcode = AArch64::CMP_SWAP_16;
4638 else if (MemTy == MVT::i32)
4639 Opcode = AArch64::CMP_SWAP_32;
4640 else if (MemTy == MVT::i64)
4641 Opcode = AArch64::CMP_SWAP_64;
4642 else
4643 llvm_unreachable("Unknown AtomicCmpSwap type");
4644
4645 MVT RegTy = MemTy == MVT::i64 ? MVT::i64 : MVT::i32;
4646 SDValue Ops[] = {N->getOperand(1), N->getOperand(2), N->getOperand(3),
4647 N->getOperand(0)};
4648 SDNode *CmpSwap = CurDAG->getMachineNode(
4649 Opcode, SDLoc(N),
4650 CurDAG->getVTList(RegTy, MVT::i32, MVT::Other), Ops);
4651
4652 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand();
4653 CurDAG->setNodeMemRefs(cast<MachineSDNode>(CmpSwap), {MemOp});
4654
4655 ReplaceUses(SDValue(N, 0), SDValue(CmpSwap, 0));
4656 ReplaceUses(SDValue(N, 1), SDValue(CmpSwap, 2));
4657 CurDAG->RemoveDeadNode(N);
4658
4659 return true;
4660}
4661
4663AArch64DAGToDAGISel::decodeMemoryHintFlags(MachineMemOperand *MMO) const {
4664 int MemoryHint = -1;
4665 const MDNode *MemCacheHint = MMO->getMemCacheHint();
4666 if (!MemCacheHint)
4667 return AArch64MemoryHint::NONE;
4668
4669 for (unsigned I = 0; I + 1 < MemCacheHint->getNumOperands(); I += 2) {
4670 if (MemCacheHint->getOperand(I).equalsStr("aarch64.mem_hint")) {
4671 const Metadata *Val = MemCacheHint->getOperand(I + 1).get();
4673 ->getZExtValue();
4674 }
4675 }
4676
4677 return toAArch64MemoryHint(MemoryHint);
4678}
4679
4680bool AArch64DAGToDAGISel::isAtomicSTSHH_KEEP(SDNode *N) const {
4681 return decodeMemoryHintFlags(cast<MemSDNode>(N)->getMemOperand()) ==
4682 AArch64MemoryHint::STSHH_KEEP;
4683}
4684
4685bool AArch64DAGToDAGISel::isAtomicSTSHH_STRM(SDNode *N) const {
4686 return decodeMemoryHintFlags(cast<MemSDNode>(N)->getMemOperand()) ==
4687 AArch64MemoryHint::STSHH_STRM;
4688}
4689
4690bool AArch64DAGToDAGISel::SelectSVEAddSubImm(SDValue N, MVT VT, SDValue &Imm,
4691 SDValue &Shift, bool Negate) {
4692 if (!isa<ConstantSDNode>(N))
4693 return false;
4694
4695 APInt Val =
4696 cast<ConstantSDNode>(N)->getAPIntValue().trunc(VT.getFixedSizeInBits());
4697
4698 return SelectSVEAddSubImm(SDLoc(N), Val, VT, Imm, Shift, Negate);
4699}
4700
4701bool AArch64DAGToDAGISel::SelectSVEAddSubImm(SDLoc DL, APInt Val, MVT VT,
4702 SDValue &Imm, SDValue &Shift,
4703 bool Negate) {
4704 if (Negate)
4705 Val = -Val;
4706
4707 switch (VT.SimpleTy) {
4708 case MVT::i8:
4709 // All immediates are supported.
4710 Shift = CurDAG->getTargetConstant(0, DL, MVT::i32);
4711 Imm = CurDAG->getTargetConstant(Val.getZExtValue(), DL, MVT::i32);
4712 return true;
4713 case MVT::i16:
4714 case MVT::i32:
4715 case MVT::i64:
4716 // Support 8bit unsigned immediates.
4717 if ((Val & ~0xff) == 0) {
4718 Shift = CurDAG->getTargetConstant(0, DL, MVT::i32);
4719 Imm = CurDAG->getTargetConstant(Val.getZExtValue(), DL, MVT::i32);
4720 return true;
4721 }
4722 // Support 16bit unsigned immediates that are a multiple of 256.
4723 if ((Val & ~0xff00) == 0) {
4724 Shift = CurDAG->getTargetConstant(8, DL, MVT::i32);
4725 Imm = CurDAG->getTargetConstant(Val.lshr(8).getZExtValue(), DL, MVT::i32);
4726 return true;
4727 }
4728 break;
4729 default:
4730 break;
4731 }
4732
4733 return false;
4734}
4735
4736bool AArch64DAGToDAGISel::SelectSVEAddSubSSatImm(SDValue N, MVT VT,
4737 SDValue &Imm, SDValue &Shift,
4738 bool Negate) {
4739 if (!isa<ConstantSDNode>(N))
4740 return false;
4741
4742 SDLoc DL(N);
4743 int64_t Val = cast<ConstantSDNode>(N)
4744 ->getAPIntValue()
4746 .getSExtValue();
4747
4748 if (Negate)
4749 Val = -Val;
4750
4751 // Signed saturating instructions treat their immediate operand as unsigned,
4752 // whereas the related intrinsics define their operands to be signed. This
4753 // means we can only use the immediate form when the operand is non-negative.
4754 if (Val < 0)
4755 return false;
4756
4757 switch (VT.SimpleTy) {
4758 case MVT::i8:
4759 // All positive immediates are supported.
4760 Shift = CurDAG->getTargetConstant(0, DL, MVT::i32);
4761 Imm = CurDAG->getTargetConstant(Val, DL, MVT::i32);
4762 return true;
4763 case MVT::i16:
4764 case MVT::i32:
4765 case MVT::i64:
4766 // Support 8bit positive immediates.
4767 if (Val <= 255) {
4768 Shift = CurDAG->getTargetConstant(0, DL, MVT::i32);
4769 Imm = CurDAG->getTargetConstant(Val, DL, MVT::i32);
4770 return true;
4771 }
4772 // Support 16bit positive immediates that are a multiple of 256.
4773 if (Val <= 65280 && Val % 256 == 0) {
4774 Shift = CurDAG->getTargetConstant(8, DL, MVT::i32);
4775 Imm = CurDAG->getTargetConstant(Val >> 8, DL, MVT::i32);
4776 return true;
4777 }
4778 break;
4779 default:
4780 break;
4781 }
4782
4783 return false;
4784}
4785
4786bool AArch64DAGToDAGISel::SelectSVECpyDupImm(SDValue N, MVT VT, SDValue &Imm,
4787 SDValue &Shift) {
4788 if (!isa<ConstantSDNode>(N))
4789 return false;
4790
4791 SDLoc DL(N);
4792 int64_t Val = cast<ConstantSDNode>(N)
4793 ->getAPIntValue()
4794 .trunc(VT.getFixedSizeInBits())
4795 .getSExtValue();
4796 int32_t ImmVal, ShiftVal;
4797 if (!AArch64_AM::isSVECpyDupImm(VT.getScalarSizeInBits(), Val, ImmVal,
4798 ShiftVal))
4799 return false;
4800
4801 Shift = CurDAG->getTargetConstant(ShiftVal, DL, MVT::i32);
4802 Imm = CurDAG->getTargetConstant(ImmVal, DL, MVT::i32);
4803 return true;
4804}
4805
4806bool AArch64DAGToDAGISel::SelectSVESignedArithImm(SDValue N, SDValue &Imm) {
4807 if (auto CNode = dyn_cast<ConstantSDNode>(N))
4808 return SelectSVESignedArithImm(SDLoc(N), CNode->getAPIntValue(), Imm);
4809 return false;
4810}
4811
4812bool AArch64DAGToDAGISel::SelectSVESignedArithImm(SDLoc DL, APInt Val,
4813 SDValue &Imm) {
4814 int64_t ImmVal = Val.getSExtValue();
4815 if (ImmVal >= -128 && ImmVal < 128) {
4816 Imm = CurDAG->getSignedTargetConstant(ImmVal, DL, MVT::i32);
4817 return true;
4818 }
4819 return false;
4820}
4821
4822bool AArch64DAGToDAGISel::SelectSVEArithImm(SDValue N, MVT VT, SDValue &Imm) {
4823 if (auto CNode = dyn_cast<ConstantSDNode>(N)) {
4824 uint64_t ImmVal = CNode->getZExtValue();
4825
4826 switch (VT.SimpleTy) {
4827 case MVT::i8:
4828 ImmVal &= 0xFF;
4829 break;
4830 case MVT::i16:
4831 ImmVal &= 0xFFFF;
4832 break;
4833 case MVT::i32:
4834 ImmVal &= 0xFFFFFFFF;
4835 break;
4836 case MVT::i64:
4837 break;
4838 default:
4839 llvm_unreachable("Unexpected type");
4840 }
4841
4842 if (ImmVal < 256) {
4843 Imm = CurDAG->getTargetConstant(ImmVal, SDLoc(N), MVT::i32);
4844 return true;
4845 }
4846 }
4847 return false;
4848}
4849
4850bool AArch64DAGToDAGISel::SelectSVELogicalImm(SDValue N, MVT VT, SDValue &Imm,
4851 bool Invert) {
4852 uint64_t ImmVal;
4853 if (auto CI = dyn_cast<ConstantSDNode>(N))
4854 ImmVal = CI->getZExtValue();
4855 else if (auto CFP = dyn_cast<ConstantFPSDNode>(N))
4856 ImmVal = CFP->getValueAPF().bitcastToAPInt().getZExtValue();
4857 else
4858 return false;
4859
4860 if (Invert)
4861 ImmVal = ~ImmVal;
4862
4863 uint64_t encoding;
4864 if (!AArch64_AM::isSVELogicalImm(VT.getScalarSizeInBits(), ImmVal, encoding))
4865 return false;
4866
4867 Imm = CurDAG->getTargetConstant(encoding, SDLoc(N), MVT::i64);
4868 return true;
4869}
4870
4871// SVE shift intrinsics allow shift amounts larger than the element's bitwidth.
4872// Rather than attempt to normalise everything we can sometimes saturate the
4873// shift amount during selection. This function also allows for consistent
4874// isel patterns by ensuring the resulting "Imm" node is of the i32 type
4875// required by the instructions.
4876bool AArch64DAGToDAGISel::SelectSVEShiftImm(SDValue N, uint64_t Low,
4877 uint64_t High, bool AllowSaturation,
4878 SDValue &Imm) {
4879 if (auto *CN = dyn_cast<ConstantSDNode>(N)) {
4880 uint64_t ImmVal = CN->getZExtValue();
4881
4882 // Reject shift amounts that are too small.
4883 if (ImmVal < Low)
4884 return false;
4885
4886 // Reject or saturate shift amounts that are too big.
4887 if (ImmVal > High) {
4888 if (!AllowSaturation)
4889 return false;
4890 ImmVal = High;
4891 }
4892
4893 Imm = CurDAG->getTargetConstant(ImmVal, SDLoc(N), MVT::i32);
4894 return true;
4895 }
4896
4897 return false;
4898}
4899
4900bool AArch64DAGToDAGISel::trySelectStackSlotTagP(SDNode *N) {
4901 // tagp(FrameIndex, IRGstack, tag_offset):
4902 // since the offset between FrameIndex and IRGstack is a compile-time
4903 // constant, this can be lowered to a single ADDG instruction.
4904 if (!(isa<FrameIndexSDNode>(N->getOperand(1)))) {
4905 return false;
4906 }
4907
4908 SDValue IRG_SP = N->getOperand(2);
4909 if (IRG_SP->getOpcode() != ISD::INTRINSIC_W_CHAIN ||
4910 IRG_SP->getConstantOperandVal(1) != Intrinsic::aarch64_irg_sp) {
4911 return false;
4912 }
4913
4914 const TargetLowering *TLI = getTargetLowering();
4915 SDLoc DL(N);
4916 int FI = cast<FrameIndexSDNode>(N->getOperand(1))->getIndex();
4917 SDValue FiOp = CurDAG->getTargetFrameIndex(
4918 FI, TLI->getPointerTy(CurDAG->getDataLayout()));
4919 int TagOffset = N->getConstantOperandVal(3);
4920
4921 SDNode *Out = CurDAG->getMachineNode(
4922 AArch64::TAGPstack, DL, MVT::i64,
4923 {FiOp, CurDAG->getTargetConstant(0, DL, MVT::i64), N->getOperand(2),
4924 CurDAG->getTargetConstant(TagOffset, DL, MVT::i64)});
4925 ReplaceNode(N, Out);
4926 return true;
4927}
4928
4929void AArch64DAGToDAGISel::SelectTagP(SDNode *N) {
4930 assert(isa<ConstantSDNode>(N->getOperand(3)) &&
4931 "llvm.aarch64.tagp third argument must be an immediate");
4932 if (trySelectStackSlotTagP(N))
4933 return;
4934 // FIXME: above applies in any case when offset between Op1 and Op2 is a
4935 // compile-time constant, not just for stack allocations.
4936
4937 // General case for unrelated pointers in Op1 and Op2.
4938 SDLoc DL(N);
4939 int TagOffset = N->getConstantOperandVal(3);
4940 SDNode *N1 = CurDAG->getMachineNode(AArch64::SUBP, DL, MVT::i64,
4941 {N->getOperand(1), N->getOperand(2)});
4942 SDNode *N2 = CurDAG->getMachineNode(AArch64::ADDXrr, DL, MVT::i64,
4943 {SDValue(N1, 0), N->getOperand(2)});
4944 SDNode *N3 = CurDAG->getMachineNode(
4945 AArch64::ADDG, DL, MVT::i64,
4946 {SDValue(N2, 0), CurDAG->getTargetConstant(0, DL, MVT::i64),
4947 CurDAG->getTargetConstant(TagOffset, DL, MVT::i64)});
4948 ReplaceNode(N, N3);
4949}
4950
4951bool AArch64DAGToDAGISel::trySelectCastFixedLengthToScalableVector(SDNode *N) {
4952 assert(N->getOpcode() == ISD::INSERT_SUBVECTOR && "Invalid Node!");
4953
4954 // Bail when not a "cast" like insert_subvector.
4955 if (N->getConstantOperandVal(2) != 0)
4956 return false;
4957 if (!N->getOperand(0).isUndef())
4958 return false;
4959
4960 // Bail when normal isel should do the job.
4961 EVT VT = N->getValueType(0);
4962 EVT InVT = N->getOperand(1).getValueType();
4963 if (VT.isFixedLengthVector() || InVT.isScalableVector())
4964 return false;
4965 if (InVT.getSizeInBits() <= 128)
4966 return false;
4967
4968 // NOTE: We can only get here when doing fixed length SVE code generation.
4969 // We do manual selection because the types involved are not linked to real
4970 // registers (despite being legal) and must be coerced into SVE registers.
4971
4973 "Expected to insert into a packed scalable vector!");
4974
4975 SDLoc DL(N);
4976 auto RC = CurDAG->getTargetConstant(AArch64::ZPRRegClassID, DL, MVT::i64);
4977 ReplaceNode(N, CurDAG->getMachineNode(TargetOpcode::COPY_TO_REGCLASS, DL, VT,
4978 N->getOperand(1), RC));
4979 return true;
4980}
4981
4982bool AArch64DAGToDAGISel::trySelectCastScalableToFixedLengthVector(SDNode *N) {
4983 assert(N->getOpcode() == ISD::EXTRACT_SUBVECTOR && "Invalid Node!");
4984
4985 // Bail when not a "cast" like extract_subvector.
4986 if (N->getConstantOperandVal(1) != 0)
4987 return false;
4988
4989 // Bail when normal isel can do the job.
4990 EVT VT = N->getValueType(0);
4991 EVT InVT = N->getOperand(0).getValueType();
4992 if (VT.isScalableVector() || InVT.isFixedLengthVector())
4993 return false;
4994 if (VT.getSizeInBits() <= 128)
4995 return false;
4996
4997 // NOTE: We can only get here when doing fixed length SVE code generation.
4998 // We do manual selection because the types involved are not linked to real
4999 // registers (despite being legal) and must be coerced into SVE registers.
5000
5002 "Expected to extract from a packed scalable vector!");
5003
5004 SDLoc DL(N);
5005 auto RC = CurDAG->getTargetConstant(AArch64::ZPRRegClassID, DL, MVT::i64);
5006 ReplaceNode(N, CurDAG->getMachineNode(TargetOpcode::COPY_TO_REGCLASS, DL, VT,
5007 N->getOperand(0), RC));
5008 return true;
5009}
5010
5011bool AArch64DAGToDAGISel::trySelectXAR(SDNode *N) {
5012 assert(N->getOpcode() == ISD::OR && "Expected OR instruction");
5013
5014 SDValue N0 = N->getOperand(0);
5015 SDValue N1 = N->getOperand(1);
5016
5017 EVT VT = N->getValueType(0);
5018 SDLoc DL(N);
5019
5020 // Essentially: rotr (xor(x, y), imm) -> xar (x, y, imm)
5021 // Rotate by a constant is a funnel shift in IR which is expanded to
5022 // an OR with shifted operands.
5023 // We do the following transform:
5024 // OR N0, N1 -> xar (x, y, imm)
5025 // Where:
5026 // N1 = SRL_PRED true, V, splat(imm) --> rotr amount
5027 // N0 = SHL_PRED true, V, splat(bits-imm)
5028 // V = (xor x, y)
5029 if (VT.isScalableVector() &&
5030 (Subtarget->hasSVE2() ||
5031 (Subtarget->hasSME() && Subtarget->isStreaming()))) {
5032 if (N0.getOpcode() != AArch64ISD::SHL_PRED ||
5033 N1.getOpcode() != AArch64ISD::SRL_PRED)
5034 std::swap(N0, N1);
5035 if (N0.getOpcode() != AArch64ISD::SHL_PRED ||
5036 N1.getOpcode() != AArch64ISD::SRL_PRED)
5037 return false;
5038
5039 auto *TLI = static_cast<const AArch64TargetLowering *>(getTargetLowering());
5040 if (!TLI->isAllActivePredicate(*CurDAG, N0.getOperand(0)) ||
5041 !TLI->isAllActivePredicate(*CurDAG, N1.getOperand(0)))
5042 return false;
5043
5044 if (N0.getOperand(1) != N1.getOperand(1))
5045 return false;
5046
5047 SDValue R1, R2;
5048 bool IsXOROperand = true;
5049 if (N0.getOperand(1).getOpcode() != ISD::XOR) {
5050 IsXOROperand = false;
5051 } else {
5052 R1 = N0.getOperand(1).getOperand(0);
5053 R2 = N1.getOperand(1).getOperand(1);
5054 }
5055
5056 APInt ShlAmt, ShrAmt;
5057 if (!ISD::isConstantSplatVector(N0.getOperand(2).getNode(), ShlAmt) ||
5059 return false;
5060
5061 if (ShlAmt + ShrAmt != VT.getScalarSizeInBits())
5062 return false;
5063
5064 if (!IsXOROperand) {
5065 SDValue Zero = CurDAG->getTargetConstant(0, DL, MVT::i64);
5066 SDNode *MOV = CurDAG->getMachineNode(AArch64::MOVIv2d_ns, DL, VT, Zero);
5067 SDValue MOVIV = SDValue(MOV, 0);
5068
5069 SDValue ZSub = CurDAG->getTargetConstant(AArch64::zsub, DL, MVT::i32);
5070 SDNode *SubRegToReg =
5071 CurDAG->getMachineNode(AArch64::SUBREG_TO_REG, DL, VT, MOVIV, ZSub);
5072
5073 R1 = N1->getOperand(1);
5074 R2 = SDValue(SubRegToReg, 0);
5075 }
5076
5077 SDValue Imm =
5078 CurDAG->getTargetConstant(ShrAmt.getZExtValue(), DL, MVT::i32);
5079
5080 SDValue Ops[] = {R1, R2, Imm};
5082 VT, {AArch64::XAR_ZZZI_B, AArch64::XAR_ZZZI_H, AArch64::XAR_ZZZI_S,
5083 AArch64::XAR_ZZZI_D})) {
5084 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
5085 return true;
5086 }
5087 return false;
5088 }
5089
5090 // We have Neon SHA3 XAR operation for v2i64 but for types
5091 // v4i32, v8i16, v16i8 we can use SVE operations when SVE2-SHA3
5092 // is available.
5093 EVT SVT;
5094 switch (VT.getSimpleVT().SimpleTy) {
5095 case MVT::v4i32:
5096 case MVT::v2i32:
5097 SVT = MVT::nxv4i32;
5098 break;
5099 case MVT::v8i16:
5100 case MVT::v4i16:
5101 SVT = MVT::nxv8i16;
5102 break;
5103 case MVT::v16i8:
5104 case MVT::v8i8:
5105 SVT = MVT::nxv16i8;
5106 break;
5107 case MVT::v2i64:
5108 case MVT::v1i64:
5109 SVT = Subtarget->hasSHA3() ? MVT::v2i64 : MVT::nxv2i64;
5110 break;
5111 default:
5112 return false;
5113 }
5114
5115 if ((!SVT.isScalableVector() && !Subtarget->hasSHA3()) ||
5116 (SVT.isScalableVector() && !Subtarget->hasSVE2()))
5117 return false;
5118
5119 if (N0->getOpcode() != AArch64ISD::VSHL ||
5120 N1->getOpcode() != AArch64ISD::VLSHR)
5121 return false;
5122
5123 if (N0->getOperand(0) != N1->getOperand(0))
5124 return false;
5125
5126 SDValue R1, R2;
5127 bool IsXOROperand = true;
5128 if (N1->getOperand(0)->getOpcode() != ISD::XOR) {
5129 IsXOROperand = false;
5130 } else {
5131 SDValue XOR = N0.getOperand(0);
5132 R1 = XOR.getOperand(0);
5133 R2 = XOR.getOperand(1);
5134 }
5135
5136 unsigned HsAmt = N0.getConstantOperandVal(1);
5137 unsigned ShAmt = N1.getConstantOperandVal(1);
5138
5139 SDValue Imm = CurDAG->getTargetConstant(
5140 ShAmt, DL, N0.getOperand(1).getValueType(), false);
5141
5142 unsigned VTSizeInBits = VT.getScalarSizeInBits();
5143 if (ShAmt + HsAmt != VTSizeInBits)
5144 return false;
5145
5146 if (!IsXOROperand) {
5147 SDValue Zero = CurDAG->getTargetConstant(0, DL, MVT::i64);
5148 SDNode *MOV =
5149 CurDAG->getMachineNode(AArch64::MOVIv2d_ns, DL, MVT::v2i64, Zero);
5150 SDValue MOVIV = SDValue(MOV, 0);
5151
5152 R1 = N1->getOperand(0);
5153 R2 = MOVIV;
5154 }
5155
5156 if (SVT != VT) {
5157 SDValue Undef =
5158 SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, SVT), 0);
5159
5160 if (SVT.isScalableVector() && VT.is64BitVector()) {
5161 EVT QVT = VT.getDoubleNumVectorElementsVT(*CurDAG->getContext());
5162
5163 SDValue UndefQ = SDValue(
5164 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, QVT), 0);
5165 SDValue DSub = CurDAG->getTargetConstant(AArch64::dsub, DL, MVT::i32);
5166
5167 R1 = SDValue(CurDAG->getMachineNode(AArch64::INSERT_SUBREG, DL, QVT,
5168 UndefQ, R1, DSub),
5169 0);
5170 if (R2.getValueType() == VT)
5171 R2 = SDValue(CurDAG->getMachineNode(AArch64::INSERT_SUBREG, DL, QVT,
5172 UndefQ, R2, DSub),
5173 0);
5174 }
5175
5176 SDValue SubReg = CurDAG->getTargetConstant(
5177 (SVT.isScalableVector() ? AArch64::zsub : AArch64::dsub), DL, MVT::i32);
5178
5179 R1 = SDValue(CurDAG->getMachineNode(AArch64::INSERT_SUBREG, DL, SVT, Undef,
5180 R1, SubReg),
5181 0);
5182
5183 if (SVT.isScalableVector() || R2.getValueType() != SVT)
5184 R2 = SDValue(CurDAG->getMachineNode(AArch64::INSERT_SUBREG, DL, SVT,
5185 Undef, R2, SubReg),
5186 0);
5187 }
5188
5189 SDValue Ops[] = {R1, R2, Imm};
5190 SDNode *XAR = nullptr;
5191
5192 if (SVT.isScalableVector()) {
5194 SVT, {AArch64::XAR_ZZZI_B, AArch64::XAR_ZZZI_H, AArch64::XAR_ZZZI_S,
5195 AArch64::XAR_ZZZI_D}))
5196 XAR = CurDAG->getMachineNode(Opc, DL, SVT, Ops);
5197 } else {
5198 XAR = CurDAG->getMachineNode(AArch64::XAR, DL, SVT, Ops);
5199 }
5200
5201 assert(XAR && "Unexpected NULL value for XAR instruction in DAG");
5202
5203 if (SVT != VT) {
5204 if (VT.is64BitVector() && SVT.isScalableVector()) {
5205 EVT QVT = VT.getDoubleNumVectorElementsVT(*CurDAG->getContext());
5206
5207 SDValue ZSub = CurDAG->getTargetConstant(AArch64::zsub, DL, MVT::i32);
5208 SDNode *Q = CurDAG->getMachineNode(AArch64::EXTRACT_SUBREG, DL, QVT,
5209 SDValue(XAR, 0), ZSub);
5210
5211 SDValue DSub = CurDAG->getTargetConstant(AArch64::dsub, DL, MVT::i32);
5212 XAR = CurDAG->getMachineNode(AArch64::EXTRACT_SUBREG, DL, VT,
5213 SDValue(Q, 0), DSub);
5214 } else {
5215 SDValue SubReg = CurDAG->getTargetConstant(
5216 (SVT.isScalableVector() ? AArch64::zsub : AArch64::dsub), DL,
5217 MVT::i32);
5218 XAR = CurDAG->getMachineNode(AArch64::EXTRACT_SUBREG, DL, VT,
5219 SDValue(XAR, 0), SubReg);
5220 }
5221 }
5222 ReplaceNode(N, XAR);
5223 return true;
5224}
5225
5226/// Returns a copy from WZR or XZR. This can be used during instruction
5227/// selection (it does not require any further selection/legalization).
5229 assert(VT == MVT::i32 || VT == MVT::i64);
5230 return DAG.getCopyFromReg(DAG.getEntryNode(), DL,
5231 VT == MVT::i32 ? AArch64::WZR : AArch64::XZR, VT);
5232}
5233
5234void AArch64DAGToDAGISel::Select(SDNode *Node) {
5235 // If we have a custom node, we already have selected!
5236 if (Node->isMachineOpcode()) {
5237 LLVM_DEBUG(errs() << "== "; Node->dump(CurDAG); errs() << "\n");
5238 Node->setNodeId(-1);
5239 return;
5240 }
5241
5242 // Few custom selection stuff.
5243 EVT VT = Node->getValueType(0);
5244
5245 switch (Node->getOpcode()) {
5246 default:
5247 break;
5248
5250 if (SelectCMP_SWAP(Node))
5251 return;
5252 break;
5253
5254 case ISD::READ_REGISTER:
5255 case AArch64ISD::MRRS:
5256 if (tryReadRegister(Node))
5257 return;
5258 break;
5259
5261 case AArch64ISD::MSRR:
5262 if (tryWriteRegister(Node))
5263 return;
5264 break;
5265
5266 case ISD::LOAD: {
5267 // Try to select as an indexed load. Fall through to normal processing
5268 // if we can't.
5269 if (tryIndexedLoad(Node))
5270 return;
5271 break;
5272 }
5273
5274 case ISD::SRL:
5275 case ISD::AND:
5276 case ISD::SRA:
5278 if (tryBitfieldExtractOp(Node))
5279 return;
5280 if (tryBitfieldInsertInZeroOp(Node))
5281 return;
5282 [[fallthrough]];
5283 case ISD::ROTR:
5284 case ISD::SHL:
5285 if (tryShiftAmountMod(Node))
5286 return;
5287 break;
5288
5289 case ISD::SIGN_EXTEND:
5290 if (tryBitfieldExtractOpFromSExt(Node))
5291 return;
5292 break;
5293
5294 case ISD::OR:
5295 if (tryBitfieldInsertOp(Node))
5296 return;
5297 if (trySelectXAR(Node))
5298 return;
5299 break;
5300
5302 if (trySelectCastScalableToFixedLengthVector(Node))
5303 return;
5304 break;
5305 }
5306
5307 case ISD::INSERT_SUBVECTOR: {
5308 if (trySelectCastFixedLengthToScalableVector(Node))
5309 return;
5310 break;
5311 }
5312
5313 case AArch64ISD::CSEL:
5314 if (tryFoldCselToFMaxMin(Node))
5315 return;
5316 break;
5317
5318 case ISD::Constant: {
5319 // Materialize zero constants as copies from WZR/XZR. This allows
5320 // the coalescer to propagate these into other instructions.
5321 ConstantSDNode *ConstNode = cast<ConstantSDNode>(Node);
5322 if (ConstNode->isZero() && (VT == MVT::i32 || VT == MVT::i64)) {
5323 ReplaceNode(Node, getZeroRegister(*CurDAG, SDLoc(Node), VT).getNode());
5324 return;
5325 }
5326 break;
5327 }
5328
5329 case ISD::FrameIndex: {
5330 // Selects to ADDXri FI, 0 which in turn will become ADDXri SP, imm.
5331 int FI = cast<FrameIndexSDNode>(Node)->getIndex();
5332 unsigned Shifter = AArch64_AM::getShifterImm(AArch64_AM::LSL, 0);
5333 const TargetLowering *TLI = getTargetLowering();
5334 SDValue TFI = CurDAG->getTargetFrameIndex(
5335 FI, TLI->getPointerTy(CurDAG->getDataLayout()));
5336 SDLoc DL(Node);
5337 SDValue Ops[] = { TFI, CurDAG->getTargetConstant(0, DL, MVT::i32),
5338 CurDAG->getTargetConstant(Shifter, DL, MVT::i32) };
5339 CurDAG->SelectNodeTo(Node, AArch64::ADDXri, MVT::i64, Ops);
5340 return;
5341 }
5343 unsigned IntNo = Node->getConstantOperandVal(1);
5344 switch (IntNo) {
5345 default:
5346 break;
5347 case Intrinsic::aarch64_gcsss: {
5348 SDLoc DL(Node);
5349 SDValue Chain = Node->getOperand(0);
5350 SDValue Val = Node->getOperand(2);
5351 SDValue Zero = CurDAG->getCopyFromReg(Chain, DL, AArch64::XZR, MVT::i64);
5352 SDNode *SS1 =
5353 CurDAG->getMachineNode(AArch64::GCSSS1, DL, MVT::Other, Val, Chain);
5354 SDNode *SS2 = CurDAG->getMachineNode(AArch64::GCSSS2, DL, MVT::i64,
5355 MVT::Other, Zero, SDValue(SS1, 0));
5356 ReplaceNode(Node, SS2);
5357 return;
5358 }
5359 case Intrinsic::aarch64_ldaxp:
5360 case Intrinsic::aarch64_ldxp: {
5361 unsigned Op =
5362 IntNo == Intrinsic::aarch64_ldaxp ? AArch64::LDAXPX : AArch64::LDXPX;
5363 SDValue MemAddr = Node->getOperand(2);
5364 SDLoc DL(Node);
5365 SDValue Chain = Node->getOperand(0);
5366
5367 SDNode *Ld = CurDAG->getMachineNode(Op, DL, MVT::i64, MVT::i64,
5368 MVT::Other, MemAddr, Chain);
5369
5370 // Transfer memoperands.
5371 MachineMemOperand *MemOp =
5372 cast<MemIntrinsicSDNode>(Node)->getMemOperand();
5373 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Ld), {MemOp});
5374 ReplaceNode(Node, Ld);
5375 return;
5376 }
5377 case Intrinsic::aarch64_stlxp:
5378 case Intrinsic::aarch64_stxp: {
5379 unsigned Op =
5380 IntNo == Intrinsic::aarch64_stlxp ? AArch64::STLXPX : AArch64::STXPX;
5381 SDLoc DL(Node);
5382 SDValue Chain = Node->getOperand(0);
5383 SDValue ValLo = Node->getOperand(2);
5384 SDValue ValHi = Node->getOperand(3);
5385 SDValue MemAddr = Node->getOperand(4);
5386
5387 // Place arguments in the right order.
5388 SDValue Ops[] = {ValLo, ValHi, MemAddr, Chain};
5389
5390 SDNode *St = CurDAG->getMachineNode(Op, DL, MVT::i32, MVT::Other, Ops);
5391 // Transfer memoperands.
5392 MachineMemOperand *MemOp =
5393 cast<MemIntrinsicSDNode>(Node)->getMemOperand();
5394 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
5395
5396 ReplaceNode(Node, St);
5397 return;
5398 }
5399 case Intrinsic::aarch64_neon_ld1x2:
5400 if (VT == MVT::v8i8) {
5401 SelectLoad(Node, 2, AArch64::LD1Twov8b, AArch64::dsub0);
5402 return;
5403 } else if (VT == MVT::v16i8) {
5404 SelectLoad(Node, 2, AArch64::LD1Twov16b, AArch64::qsub0);
5405 return;
5406 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5407 SelectLoad(Node, 2, AArch64::LD1Twov4h, AArch64::dsub0);
5408 return;
5409 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5410 SelectLoad(Node, 2, AArch64::LD1Twov8h, AArch64::qsub0);
5411 return;
5412 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5413 SelectLoad(Node, 2, AArch64::LD1Twov2s, AArch64::dsub0);
5414 return;
5415 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5416 SelectLoad(Node, 2, AArch64::LD1Twov4s, AArch64::qsub0);
5417 return;
5418 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5419 SelectLoad(Node, 2, AArch64::LD1Twov1d, AArch64::dsub0);
5420 return;
5421 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5422 SelectLoad(Node, 2, AArch64::LD1Twov2d, AArch64::qsub0);
5423 return;
5424 }
5425 break;
5426 case Intrinsic::aarch64_neon_ld1x3:
5427 if (VT == MVT::v8i8) {
5428 SelectLoad(Node, 3, AArch64::LD1Threev8b, AArch64::dsub0);
5429 return;
5430 } else if (VT == MVT::v16i8) {
5431 SelectLoad(Node, 3, AArch64::LD1Threev16b, AArch64::qsub0);
5432 return;
5433 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5434 SelectLoad(Node, 3, AArch64::LD1Threev4h, AArch64::dsub0);
5435 return;
5436 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5437 SelectLoad(Node, 3, AArch64::LD1Threev8h, AArch64::qsub0);
5438 return;
5439 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5440 SelectLoad(Node, 3, AArch64::LD1Threev2s, AArch64::dsub0);
5441 return;
5442 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5443 SelectLoad(Node, 3, AArch64::LD1Threev4s, AArch64::qsub0);
5444 return;
5445 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5446 SelectLoad(Node, 3, AArch64::LD1Threev1d, AArch64::dsub0);
5447 return;
5448 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5449 SelectLoad(Node, 3, AArch64::LD1Threev2d, AArch64::qsub0);
5450 return;
5451 }
5452 break;
5453 case Intrinsic::aarch64_neon_ld1x4:
5454 if (VT == MVT::v8i8) {
5455 SelectLoad(Node, 4, AArch64::LD1Fourv8b, AArch64::dsub0);
5456 return;
5457 } else if (VT == MVT::v16i8) {
5458 SelectLoad(Node, 4, AArch64::LD1Fourv16b, AArch64::qsub0);
5459 return;
5460 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5461 SelectLoad(Node, 4, AArch64::LD1Fourv4h, AArch64::dsub0);
5462 return;
5463 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5464 SelectLoad(Node, 4, AArch64::LD1Fourv8h, AArch64::qsub0);
5465 return;
5466 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5467 SelectLoad(Node, 4, AArch64::LD1Fourv2s, AArch64::dsub0);
5468 return;
5469 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5470 SelectLoad(Node, 4, AArch64::LD1Fourv4s, AArch64::qsub0);
5471 return;
5472 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5473 SelectLoad(Node, 4, AArch64::LD1Fourv1d, AArch64::dsub0);
5474 return;
5475 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5476 SelectLoad(Node, 4, AArch64::LD1Fourv2d, AArch64::qsub0);
5477 return;
5478 }
5479 break;
5480 case Intrinsic::aarch64_neon_ld2:
5481 if (VT == MVT::v8i8) {
5482 SelectLoad(Node, 2, AArch64::LD2Twov8b, AArch64::dsub0);
5483 return;
5484 } else if (VT == MVT::v16i8) {
5485 SelectLoad(Node, 2, AArch64::LD2Twov16b, AArch64::qsub0);
5486 return;
5487 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5488 SelectLoad(Node, 2, AArch64::LD2Twov4h, AArch64::dsub0);
5489 return;
5490 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5491 SelectLoad(Node, 2, AArch64::LD2Twov8h, AArch64::qsub0);
5492 return;
5493 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5494 SelectLoad(Node, 2, AArch64::LD2Twov2s, AArch64::dsub0);
5495 return;
5496 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5497 SelectLoad(Node, 2, AArch64::LD2Twov4s, AArch64::qsub0);
5498 return;
5499 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5500 SelectLoad(Node, 2, AArch64::LD1Twov1d, AArch64::dsub0);
5501 return;
5502 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5503 SelectLoad(Node, 2, AArch64::LD2Twov2d, AArch64::qsub0);
5504 return;
5505 }
5506 break;
5507 case Intrinsic::aarch64_neon_ld3:
5508 if (VT == MVT::v8i8) {
5509 SelectLoad(Node, 3, AArch64::LD3Threev8b, AArch64::dsub0);
5510 return;
5511 } else if (VT == MVT::v16i8) {
5512 SelectLoad(Node, 3, AArch64::LD3Threev16b, AArch64::qsub0);
5513 return;
5514 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5515 SelectLoad(Node, 3, AArch64::LD3Threev4h, AArch64::dsub0);
5516 return;
5517 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5518 SelectLoad(Node, 3, AArch64::LD3Threev8h, AArch64::qsub0);
5519 return;
5520 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5521 SelectLoad(Node, 3, AArch64::LD3Threev2s, AArch64::dsub0);
5522 return;
5523 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5524 SelectLoad(Node, 3, AArch64::LD3Threev4s, AArch64::qsub0);
5525 return;
5526 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5527 SelectLoad(Node, 3, AArch64::LD1Threev1d, AArch64::dsub0);
5528 return;
5529 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5530 SelectLoad(Node, 3, AArch64::LD3Threev2d, AArch64::qsub0);
5531 return;
5532 }
5533 break;
5534 case Intrinsic::aarch64_neon_ld4:
5535 if (VT == MVT::v8i8) {
5536 SelectLoad(Node, 4, AArch64::LD4Fourv8b, AArch64::dsub0);
5537 return;
5538 } else if (VT == MVT::v16i8) {
5539 SelectLoad(Node, 4, AArch64::LD4Fourv16b, AArch64::qsub0);
5540 return;
5541 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5542 SelectLoad(Node, 4, AArch64::LD4Fourv4h, AArch64::dsub0);
5543 return;
5544 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5545 SelectLoad(Node, 4, AArch64::LD4Fourv8h, AArch64::qsub0);
5546 return;
5547 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5548 SelectLoad(Node, 4, AArch64::LD4Fourv2s, AArch64::dsub0);
5549 return;
5550 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5551 SelectLoad(Node, 4, AArch64::LD4Fourv4s, AArch64::qsub0);
5552 return;
5553 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5554 SelectLoad(Node, 4, AArch64::LD1Fourv1d, AArch64::dsub0);
5555 return;
5556 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5557 SelectLoad(Node, 4, AArch64::LD4Fourv2d, AArch64::qsub0);
5558 return;
5559 }
5560 break;
5561 case Intrinsic::aarch64_neon_ld2r:
5562 if (VT == MVT::v8i8) {
5563 SelectLoad(Node, 2, AArch64::LD2Rv8b, AArch64::dsub0);
5564 return;
5565 } else if (VT == MVT::v16i8) {
5566 SelectLoad(Node, 2, AArch64::LD2Rv16b, AArch64::qsub0);
5567 return;
5568 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5569 SelectLoad(Node, 2, AArch64::LD2Rv4h, AArch64::dsub0);
5570 return;
5571 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5572 SelectLoad(Node, 2, AArch64::LD2Rv8h, AArch64::qsub0);
5573 return;
5574 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5575 SelectLoad(Node, 2, AArch64::LD2Rv2s, AArch64::dsub0);
5576 return;
5577 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5578 SelectLoad(Node, 2, AArch64::LD2Rv4s, AArch64::qsub0);
5579 return;
5580 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5581 SelectLoad(Node, 2, AArch64::LD2Rv1d, AArch64::dsub0);
5582 return;
5583 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5584 SelectLoad(Node, 2, AArch64::LD2Rv2d, AArch64::qsub0);
5585 return;
5586 }
5587 break;
5588 case Intrinsic::aarch64_neon_ld3r:
5589 if (VT == MVT::v8i8) {
5590 SelectLoad(Node, 3, AArch64::LD3Rv8b, AArch64::dsub0);
5591 return;
5592 } else if (VT == MVT::v16i8) {
5593 SelectLoad(Node, 3, AArch64::LD3Rv16b, AArch64::qsub0);
5594 return;
5595 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5596 SelectLoad(Node, 3, AArch64::LD3Rv4h, AArch64::dsub0);
5597 return;
5598 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5599 SelectLoad(Node, 3, AArch64::LD3Rv8h, AArch64::qsub0);
5600 return;
5601 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5602 SelectLoad(Node, 3, AArch64::LD3Rv2s, AArch64::dsub0);
5603 return;
5604 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5605 SelectLoad(Node, 3, AArch64::LD3Rv4s, AArch64::qsub0);
5606 return;
5607 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5608 SelectLoad(Node, 3, AArch64::LD3Rv1d, AArch64::dsub0);
5609 return;
5610 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5611 SelectLoad(Node, 3, AArch64::LD3Rv2d, AArch64::qsub0);
5612 return;
5613 }
5614 break;
5615 case Intrinsic::aarch64_neon_ld4r:
5616 if (VT == MVT::v8i8) {
5617 SelectLoad(Node, 4, AArch64::LD4Rv8b, AArch64::dsub0);
5618 return;
5619 } else if (VT == MVT::v16i8) {
5620 SelectLoad(Node, 4, AArch64::LD4Rv16b, AArch64::qsub0);
5621 return;
5622 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5623 SelectLoad(Node, 4, AArch64::LD4Rv4h, AArch64::dsub0);
5624 return;
5625 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5626 SelectLoad(Node, 4, AArch64::LD4Rv8h, AArch64::qsub0);
5627 return;
5628 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5629 SelectLoad(Node, 4, AArch64::LD4Rv2s, AArch64::dsub0);
5630 return;
5631 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5632 SelectLoad(Node, 4, AArch64::LD4Rv4s, AArch64::qsub0);
5633 return;
5634 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5635 SelectLoad(Node, 4, AArch64::LD4Rv1d, AArch64::dsub0);
5636 return;
5637 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5638 SelectLoad(Node, 4, AArch64::LD4Rv2d, AArch64::qsub0);
5639 return;
5640 }
5641 break;
5642 case Intrinsic::aarch64_neon_ld2lane:
5643 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
5644 SelectLoadLane(Node, 2, AArch64::LD2i8);
5645 return;
5646 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
5647 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
5648 SelectLoadLane(Node, 2, AArch64::LD2i16);
5649 return;
5650 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
5651 VT == MVT::v2f32) {
5652 SelectLoadLane(Node, 2, AArch64::LD2i32);
5653 return;
5654 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
5655 VT == MVT::v1f64) {
5656 SelectLoadLane(Node, 2, AArch64::LD2i64);
5657 return;
5658 }
5659 break;
5660 case Intrinsic::aarch64_neon_ld3lane:
5661 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
5662 SelectLoadLane(Node, 3, AArch64::LD3i8);
5663 return;
5664 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
5665 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
5666 SelectLoadLane(Node, 3, AArch64::LD3i16);
5667 return;
5668 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
5669 VT == MVT::v2f32) {
5670 SelectLoadLane(Node, 3, AArch64::LD3i32);
5671 return;
5672 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
5673 VT == MVT::v1f64) {
5674 SelectLoadLane(Node, 3, AArch64::LD3i64);
5675 return;
5676 }
5677 break;
5678 case Intrinsic::aarch64_neon_ld4lane:
5679 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
5680 SelectLoadLane(Node, 4, AArch64::LD4i8);
5681 return;
5682 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
5683 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
5684 SelectLoadLane(Node, 4, AArch64::LD4i16);
5685 return;
5686 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
5687 VT == MVT::v2f32) {
5688 SelectLoadLane(Node, 4, AArch64::LD4i32);
5689 return;
5690 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
5691 VT == MVT::v1f64) {
5692 SelectLoadLane(Node, 4, AArch64::LD4i64);
5693 return;
5694 }
5695 break;
5696 case Intrinsic::aarch64_ld64b:
5697 SelectLoad(Node, 8, AArch64::LD64B, AArch64::x8sub_0);
5698 return;
5699 case Intrinsic::aarch64_sve_ld2q_sret: {
5700 SelectPredicatedLoad(Node, 2, 4, AArch64::LD2Q_IMM, AArch64::LD2Q, true);
5701 return;
5702 }
5703 case Intrinsic::aarch64_sve_ld3q_sret: {
5704 SelectPredicatedLoad(Node, 3, 4, AArch64::LD3Q_IMM, AArch64::LD3Q, true);
5705 return;
5706 }
5707 case Intrinsic::aarch64_sve_ld4q_sret: {
5708 SelectPredicatedLoad(Node, 4, 4, AArch64::LD4Q_IMM, AArch64::LD4Q, true);
5709 return;
5710 }
5711 case Intrinsic::aarch64_sve_ld2_sret: {
5712 if (VT == MVT::nxv16i8) {
5713 SelectPredicatedLoad(Node, 2, 0, AArch64::LD2B_IMM, AArch64::LD2B,
5714 true);
5715 return;
5716 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5717 VT == MVT::nxv8bf16) {
5718 SelectPredicatedLoad(Node, 2, 1, AArch64::LD2H_IMM, AArch64::LD2H,
5719 true);
5720 return;
5721 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5722 SelectPredicatedLoad(Node, 2, 2, AArch64::LD2W_IMM, AArch64::LD2W,
5723 true);
5724 return;
5725 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5726 SelectPredicatedLoad(Node, 2, 3, AArch64::LD2D_IMM, AArch64::LD2D,
5727 true);
5728 return;
5729 }
5730 break;
5731 }
5732 case Intrinsic::aarch64_sve_ld1_pn_x2: {
5733 if (VT == MVT::nxv16i8) {
5734 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5735 SelectContiguousMultiVectorLoad(
5736 Node, 2, 0, AArch64::LD1B_2Z_IMM_PSEUDO, AArch64::LD1B_2Z_PSEUDO);
5737 else if (Subtarget->hasSVE2p1())
5738 SelectContiguousMultiVectorLoad(Node, 2, 0, AArch64::LD1B_2Z_IMM,
5739 AArch64::LD1B_2Z);
5740 else
5741 break;
5742 return;
5743 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5744 VT == MVT::nxv8bf16) {
5745 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5746 SelectContiguousMultiVectorLoad(
5747 Node, 2, 1, AArch64::LD1H_2Z_IMM_PSEUDO, AArch64::LD1H_2Z_PSEUDO);
5748 else if (Subtarget->hasSVE2p1())
5749 SelectContiguousMultiVectorLoad(Node, 2, 1, AArch64::LD1H_2Z_IMM,
5750 AArch64::LD1H_2Z);
5751 else
5752 break;
5753 return;
5754 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5755 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5756 SelectContiguousMultiVectorLoad(
5757 Node, 2, 2, AArch64::LD1W_2Z_IMM_PSEUDO, AArch64::LD1W_2Z_PSEUDO);
5758 else if (Subtarget->hasSVE2p1())
5759 SelectContiguousMultiVectorLoad(Node, 2, 2, AArch64::LD1W_2Z_IMM,
5760 AArch64::LD1W_2Z);
5761 else
5762 break;
5763 return;
5764 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5765 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5766 SelectContiguousMultiVectorLoad(
5767 Node, 2, 3, AArch64::LD1D_2Z_IMM_PSEUDO, AArch64::LD1D_2Z_PSEUDO);
5768 else if (Subtarget->hasSVE2p1())
5769 SelectContiguousMultiVectorLoad(Node, 2, 3, AArch64::LD1D_2Z_IMM,
5770 AArch64::LD1D_2Z);
5771 else
5772 break;
5773 return;
5774 }
5775 break;
5776 }
5777 case Intrinsic::aarch64_sve_ld1_pn_x4: {
5778 if (VT == MVT::nxv16i8) {
5779 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5780 SelectContiguousMultiVectorLoad(
5781 Node, 4, 0, AArch64::LD1B_4Z_IMM_PSEUDO, AArch64::LD1B_4Z_PSEUDO);
5782 else if (Subtarget->hasSVE2p1())
5783 SelectContiguousMultiVectorLoad(Node, 4, 0, AArch64::LD1B_4Z_IMM,
5784 AArch64::LD1B_4Z);
5785 else
5786 break;
5787 return;
5788 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5789 VT == MVT::nxv8bf16) {
5790 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5791 SelectContiguousMultiVectorLoad(
5792 Node, 4, 1, AArch64::LD1H_4Z_IMM_PSEUDO, AArch64::LD1H_4Z_PSEUDO);
5793 else if (Subtarget->hasSVE2p1())
5794 SelectContiguousMultiVectorLoad(Node, 4, 1, AArch64::LD1H_4Z_IMM,
5795 AArch64::LD1H_4Z);
5796 else
5797 break;
5798 return;
5799 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5800 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5801 SelectContiguousMultiVectorLoad(
5802 Node, 4, 2, AArch64::LD1W_4Z_IMM_PSEUDO, AArch64::LD1W_4Z_PSEUDO);
5803 else if (Subtarget->hasSVE2p1())
5804 SelectContiguousMultiVectorLoad(Node, 4, 2, AArch64::LD1W_4Z_IMM,
5805 AArch64::LD1W_4Z);
5806 else
5807 break;
5808 return;
5809 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5810 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5811 SelectContiguousMultiVectorLoad(
5812 Node, 4, 3, AArch64::LD1D_4Z_IMM_PSEUDO, AArch64::LD1D_4Z_PSEUDO);
5813 else if (Subtarget->hasSVE2p1())
5814 SelectContiguousMultiVectorLoad(Node, 4, 3, AArch64::LD1D_4Z_IMM,
5815 AArch64::LD1D_4Z);
5816 else
5817 break;
5818 return;
5819 }
5820 break;
5821 }
5822 case Intrinsic::aarch64_sve_ldnt1_pn_x2: {
5823 if (VT == MVT::nxv16i8) {
5824 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5825 SelectContiguousMultiVectorLoad(Node, 2, 0,
5826 AArch64::LDNT1B_2Z_IMM_PSEUDO,
5827 AArch64::LDNT1B_2Z_PSEUDO);
5828 else if (Subtarget->hasSVE2p1())
5829 SelectContiguousMultiVectorLoad(Node, 2, 0, AArch64::LDNT1B_2Z_IMM,
5830 AArch64::LDNT1B_2Z);
5831 else
5832 break;
5833 return;
5834 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5835 VT == MVT::nxv8bf16) {
5836 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5837 SelectContiguousMultiVectorLoad(Node, 2, 1,
5838 AArch64::LDNT1H_2Z_IMM_PSEUDO,
5839 AArch64::LDNT1H_2Z_PSEUDO);
5840 else if (Subtarget->hasSVE2p1())
5841 SelectContiguousMultiVectorLoad(Node, 2, 1, AArch64::LDNT1H_2Z_IMM,
5842 AArch64::LDNT1H_2Z);
5843 else
5844 break;
5845 return;
5846 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5847 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5848 SelectContiguousMultiVectorLoad(Node, 2, 2,
5849 AArch64::LDNT1W_2Z_IMM_PSEUDO,
5850 AArch64::LDNT1W_2Z_PSEUDO);
5851 else if (Subtarget->hasSVE2p1())
5852 SelectContiguousMultiVectorLoad(Node, 2, 2, AArch64::LDNT1W_2Z_IMM,
5853 AArch64::LDNT1W_2Z);
5854 else
5855 break;
5856 return;
5857 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5858 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5859 SelectContiguousMultiVectorLoad(Node, 2, 3,
5860 AArch64::LDNT1D_2Z_IMM_PSEUDO,
5861 AArch64::LDNT1D_2Z_PSEUDO);
5862 else if (Subtarget->hasSVE2p1())
5863 SelectContiguousMultiVectorLoad(Node, 2, 3, AArch64::LDNT1D_2Z_IMM,
5864 AArch64::LDNT1D_2Z);
5865 else
5866 break;
5867 return;
5868 }
5869 break;
5870 }
5871 case Intrinsic::aarch64_sve_ldnt1_pn_x4: {
5872 if (VT == MVT::nxv16i8) {
5873 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5874 SelectContiguousMultiVectorLoad(Node, 4, 0,
5875 AArch64::LDNT1B_4Z_IMM_PSEUDO,
5876 AArch64::LDNT1B_4Z_PSEUDO);
5877 else if (Subtarget->hasSVE2p1())
5878 SelectContiguousMultiVectorLoad(Node, 4, 0, AArch64::LDNT1B_4Z_IMM,
5879 AArch64::LDNT1B_4Z);
5880 else
5881 break;
5882 return;
5883 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5884 VT == MVT::nxv8bf16) {
5885 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5886 SelectContiguousMultiVectorLoad(Node, 4, 1,
5887 AArch64::LDNT1H_4Z_IMM_PSEUDO,
5888 AArch64::LDNT1H_4Z_PSEUDO);
5889 else if (Subtarget->hasSVE2p1())
5890 SelectContiguousMultiVectorLoad(Node, 4, 1, AArch64::LDNT1H_4Z_IMM,
5891 AArch64::LDNT1H_4Z);
5892 else
5893 break;
5894 return;
5895 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5896 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5897 SelectContiguousMultiVectorLoad(Node, 4, 2,
5898 AArch64::LDNT1W_4Z_IMM_PSEUDO,
5899 AArch64::LDNT1W_4Z_PSEUDO);
5900 else if (Subtarget->hasSVE2p1())
5901 SelectContiguousMultiVectorLoad(Node, 4, 2, AArch64::LDNT1W_4Z_IMM,
5902 AArch64::LDNT1W_4Z);
5903 else
5904 break;
5905 return;
5906 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5907 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5908 SelectContiguousMultiVectorLoad(Node, 4, 3,
5909 AArch64::LDNT1D_4Z_IMM_PSEUDO,
5910 AArch64::LDNT1D_4Z_PSEUDO);
5911 else if (Subtarget->hasSVE2p1())
5912 SelectContiguousMultiVectorLoad(Node, 4, 3, AArch64::LDNT1D_4Z_IMM,
5913 AArch64::LDNT1D_4Z);
5914 else
5915 break;
5916 return;
5917 }
5918 break;
5919 }
5920 case Intrinsic::aarch64_sve_ld3_sret: {
5921 if (VT == MVT::nxv16i8) {
5922 SelectPredicatedLoad(Node, 3, 0, AArch64::LD3B_IMM, AArch64::LD3B,
5923 true);
5924 return;
5925 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5926 VT == MVT::nxv8bf16) {
5927 SelectPredicatedLoad(Node, 3, 1, AArch64::LD3H_IMM, AArch64::LD3H,
5928 true);
5929 return;
5930 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5931 SelectPredicatedLoad(Node, 3, 2, AArch64::LD3W_IMM, AArch64::LD3W,
5932 true);
5933 return;
5934 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5935 SelectPredicatedLoad(Node, 3, 3, AArch64::LD3D_IMM, AArch64::LD3D,
5936 true);
5937 return;
5938 }
5939 break;
5940 }
5941 case Intrinsic::aarch64_sve_ld4_sret: {
5942 if (VT == MVT::nxv16i8) {
5943 SelectPredicatedLoad(Node, 4, 0, AArch64::LD4B_IMM, AArch64::LD4B,
5944 true);
5945 return;
5946 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5947 VT == MVT::nxv8bf16) {
5948 SelectPredicatedLoad(Node, 4, 1, AArch64::LD4H_IMM, AArch64::LD4H,
5949 true);
5950 return;
5951 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5952 SelectPredicatedLoad(Node, 4, 2, AArch64::LD4W_IMM, AArch64::LD4W,
5953 true);
5954 return;
5955 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5956 SelectPredicatedLoad(Node, 4, 3, AArch64::LD4D_IMM, AArch64::LD4D,
5957 true);
5958 return;
5959 }
5960 break;
5961 }
5962 case Intrinsic::aarch64_sme_read_hor_vg2: {
5963 if (VT == MVT::nxv16i8) {
5964 SelectMultiVectorMove<14, 2>(Node, 2, AArch64::ZAB0,
5965 AArch64::MOVA_2ZMXI_H_B);
5966 return;
5967 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5968 VT == MVT::nxv8bf16) {
5969 SelectMultiVectorMove<6, 2>(Node, 2, AArch64::ZAH0,
5970 AArch64::MOVA_2ZMXI_H_H);
5971 return;
5972 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5973 SelectMultiVectorMove<2, 2>(Node, 2, AArch64::ZAS0,
5974 AArch64::MOVA_2ZMXI_H_S);
5975 return;
5976 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5977 SelectMultiVectorMove<0, 2>(Node, 2, AArch64::ZAD0,
5978 AArch64::MOVA_2ZMXI_H_D);
5979 return;
5980 }
5981 break;
5982 }
5983 case Intrinsic::aarch64_sme_read_ver_vg2: {
5984 if (VT == MVT::nxv16i8) {
5985 SelectMultiVectorMove<14, 2>(Node, 2, AArch64::ZAB0,
5986 AArch64::MOVA_2ZMXI_V_B);
5987 return;
5988 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5989 VT == MVT::nxv8bf16) {
5990 SelectMultiVectorMove<6, 2>(Node, 2, AArch64::ZAH0,
5991 AArch64::MOVA_2ZMXI_V_H);
5992 return;
5993 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5994 SelectMultiVectorMove<2, 2>(Node, 2, AArch64::ZAS0,
5995 AArch64::MOVA_2ZMXI_V_S);
5996 return;
5997 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5998 SelectMultiVectorMove<0, 2>(Node, 2, AArch64::ZAD0,
5999 AArch64::MOVA_2ZMXI_V_D);
6000 return;
6001 }
6002 break;
6003 }
6004 case Intrinsic::aarch64_sme_read_hor_vg4: {
6005 if (VT == MVT::nxv16i8) {
6006 SelectMultiVectorMove<12, 4>(Node, 4, AArch64::ZAB0,
6007 AArch64::MOVA_4ZMXI_H_B);
6008 return;
6009 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6010 VT == MVT::nxv8bf16) {
6011 SelectMultiVectorMove<4, 4>(Node, 4, AArch64::ZAH0,
6012 AArch64::MOVA_4ZMXI_H_H);
6013 return;
6014 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6015 SelectMultiVectorMove<0, 2>(Node, 4, AArch64::ZAS0,
6016 AArch64::MOVA_4ZMXI_H_S);
6017 return;
6018 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6019 SelectMultiVectorMove<0, 2>(Node, 4, AArch64::ZAD0,
6020 AArch64::MOVA_4ZMXI_H_D);
6021 return;
6022 }
6023 break;
6024 }
6025 case Intrinsic::aarch64_sme_read_ver_vg4: {
6026 if (VT == MVT::nxv16i8) {
6027 SelectMultiVectorMove<12, 4>(Node, 4, AArch64::ZAB0,
6028 AArch64::MOVA_4ZMXI_V_B);
6029 return;
6030 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6031 VT == MVT::nxv8bf16) {
6032 SelectMultiVectorMove<4, 4>(Node, 4, AArch64::ZAH0,
6033 AArch64::MOVA_4ZMXI_V_H);
6034 return;
6035 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6036 SelectMultiVectorMove<0, 4>(Node, 4, AArch64::ZAS0,
6037 AArch64::MOVA_4ZMXI_V_S);
6038 return;
6039 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6040 SelectMultiVectorMove<0, 4>(Node, 4, AArch64::ZAD0,
6041 AArch64::MOVA_4ZMXI_V_D);
6042 return;
6043 }
6044 break;
6045 }
6046 case Intrinsic::aarch64_sme_read_vg1x2: {
6047 SelectMultiVectorMove<7, 1>(Node, 2, AArch64::ZA,
6048 AArch64::MOVA_VG2_2ZMXI);
6049 return;
6050 }
6051 case Intrinsic::aarch64_sme_read_vg1x4: {
6052 SelectMultiVectorMove<7, 1>(Node, 4, AArch64::ZA,
6053 AArch64::MOVA_VG4_4ZMXI);
6054 return;
6055 }
6056 case Intrinsic::aarch64_sme_readz_horiz_x2: {
6057 if (VT == MVT::nxv16i8) {
6058 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_H_B_PSEUDO, 14, 2);
6059 return;
6060 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6061 VT == MVT::nxv8bf16) {
6062 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_H_H_PSEUDO, 6, 2);
6063 return;
6064 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6065 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_H_S_PSEUDO, 2, 2);
6066 return;
6067 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6068 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_H_D_PSEUDO, 0, 2);
6069 return;
6070 }
6071 break;
6072 }
6073 case Intrinsic::aarch64_sme_readz_vert_x2: {
6074 if (VT == MVT::nxv16i8) {
6075 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_V_B_PSEUDO, 14, 2);
6076 return;
6077 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6078 VT == MVT::nxv8bf16) {
6079 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_V_H_PSEUDO, 6, 2);
6080 return;
6081 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6082 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_V_S_PSEUDO, 2, 2);
6083 return;
6084 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6085 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_V_D_PSEUDO, 0, 2);
6086 return;
6087 }
6088 break;
6089 }
6090 case Intrinsic::aarch64_sme_readz_horiz_x4: {
6091 if (VT == MVT::nxv16i8) {
6092 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_H_B_PSEUDO, 12, 4);
6093 return;
6094 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6095 VT == MVT::nxv8bf16) {
6096 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_H_H_PSEUDO, 4, 4);
6097 return;
6098 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6099 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_H_S_PSEUDO, 0, 4);
6100 return;
6101 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6102 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_H_D_PSEUDO, 0, 4);
6103 return;
6104 }
6105 break;
6106 }
6107 case Intrinsic::aarch64_sme_readz_vert_x4: {
6108 if (VT == MVT::nxv16i8) {
6109 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_V_B_PSEUDO, 12, 4);
6110 return;
6111 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6112 VT == MVT::nxv8bf16) {
6113 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_V_H_PSEUDO, 4, 4);
6114 return;
6115 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6116 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_V_S_PSEUDO, 0, 4);
6117 return;
6118 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6119 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_V_D_PSEUDO, 0, 4);
6120 return;
6121 }
6122 break;
6123 }
6124 case Intrinsic::aarch64_sme_readz_x2: {
6125 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_VG2_2ZMXI_PSEUDO, 7, 1,
6126 AArch64::ZA);
6127 return;
6128 }
6129 case Intrinsic::aarch64_sme_readz_x4: {
6130 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_VG4_4ZMXI_PSEUDO, 7, 1,
6131 AArch64::ZA);
6132 return;
6133 }
6134 case Intrinsic::swift_async_context_addr: {
6135 SDLoc DL(Node);
6136 SDValue Chain = Node->getOperand(0);
6137 SDValue CopyFP = CurDAG->getCopyFromReg(Chain, DL, AArch64::FP, MVT::i64);
6138 SDValue Res = SDValue(
6139 CurDAG->getMachineNode(AArch64::SUBXri, DL, MVT::i64, CopyFP,
6140 CurDAG->getTargetConstant(8, DL, MVT::i32),
6141 CurDAG->getTargetConstant(0, DL, MVT::i32)),
6142 0);
6143 ReplaceUses(SDValue(Node, 0), Res);
6144 ReplaceUses(SDValue(Node, 1), CopyFP.getValue(1));
6145 CurDAG->RemoveDeadNode(Node);
6146
6147 auto &MF = CurDAG->getMachineFunction();
6148 MF.getFrameInfo().setFrameAddressIsTaken(true);
6149 MF.getInfo<AArch64FunctionInfo>()->setHasSwiftAsyncContext(true);
6150 return;
6151 }
6152 case Intrinsic::aarch64_sme_luti2_lane_zt_x4: {
6154 Node->getValueType(0),
6155 {AArch64::LUTI2_4ZTZI_B, AArch64::LUTI2_4ZTZI_H,
6156 AArch64::LUTI2_4ZTZI_S}))
6157 // Second Immediate must be <= 3:
6158 SelectMultiVectorLutiLane(Node, 4, Opc, 3);
6159 return;
6160 }
6161 case Intrinsic::aarch64_sme_luti4_lane_zt_x4: {
6163 Node->getValueType(0),
6164 {0, AArch64::LUTI4_4ZTZI_H, AArch64::LUTI4_4ZTZI_S}))
6165 // Second Immediate must be <= 1:
6166 SelectMultiVectorLutiLane(Node, 4, Opc, 1);
6167 return;
6168 }
6169 case Intrinsic::aarch64_sme_luti2_lane_zt_x2: {
6171 Node->getValueType(0),
6172 {AArch64::LUTI2_2ZTZI_B, AArch64::LUTI2_2ZTZI_H,
6173 AArch64::LUTI2_2ZTZI_S}))
6174 // Second Immediate must be <= 7:
6175 SelectMultiVectorLutiLane(Node, 2, Opc, 7);
6176 return;
6177 }
6178 case Intrinsic::aarch64_sme_luti4_lane_zt_x2: {
6180 Node->getValueType(0),
6181 {AArch64::LUTI4_2ZTZI_B, AArch64::LUTI4_2ZTZI_H,
6182 AArch64::LUTI4_2ZTZI_S}))
6183 // Second Immediate must be <= 3:
6184 SelectMultiVectorLutiLane(Node, 2, Opc, 3);
6185 return;
6186 }
6187 case Intrinsic::aarch64_sme_luti4_zt_x4: {
6188 SelectMultiVectorLuti(Node, 4, AArch64::LUTI4_4ZZT2Z, 2);
6189 return;
6190 }
6191 case Intrinsic::aarch64_sme_luti6_zt_x4: {
6192 SelectMultiVectorLuti(Node, 4, AArch64::LUTI6_4ZT3Z, 3);
6193 return;
6194 }
6195 case Intrinsic::aarch64_sve_fp8_cvtl1_x2:
6197 Node->getValueType(0),
6198 {AArch64::BF1CVTL_2ZZ_BtoH, AArch64::F1CVTL_2ZZ_BtoH}))
6199 SelectCVTIntrinsicFP8(Node, 2, Opc);
6200 return;
6201 case Intrinsic::aarch64_sve_fp8_cvtl2_x2:
6203 Node->getValueType(0),
6204 {AArch64::BF2CVTL_2ZZ_BtoH, AArch64::F2CVTL_2ZZ_BtoH}))
6205 SelectCVTIntrinsicFP8(Node, 2, Opc);
6206 return;
6207 case Intrinsic::aarch64_sve_fp8_cvt1_x2:
6209 Node->getValueType(0),
6210 {AArch64::BF1CVT_2ZZ_BtoH, AArch64::F1CVT_2ZZ_BtoH}))
6211 SelectCVTIntrinsicFP8(Node, 2, Opc);
6212 return;
6213 case Intrinsic::aarch64_sve_fp8_cvt2_x2:
6215 Node->getValueType(0),
6216 {AArch64::BF2CVT_2ZZ_BtoH, AArch64::F2CVT_2ZZ_BtoH}))
6217 SelectCVTIntrinsicFP8(Node, 2, Opc);
6218 return;
6219 case Intrinsic::ptrauth_resign_load_relative:
6220 SelectPtrauthResign(Node);
6221 return;
6222 }
6223 } break;
6225 unsigned IntNo = Node->getConstantOperandVal(0);
6226 switch (IntNo) {
6227 default:
6228 break;
6229 case Intrinsic::aarch64_tagp:
6230 SelectTagP(Node);
6231 return;
6232
6233 case Intrinsic::ptrauth_auth:
6234 SelectPtrauthAuth(Node);
6235 return;
6236
6237 case Intrinsic::ptrauth_resign:
6238 SelectPtrauthResign(Node);
6239 return;
6240
6241 case Intrinsic::ptrauth_auth_with_pc_and_resign:
6242 SelectPtrauthResignWithPC(Node);
6243 return;
6244
6245 case Intrinsic::aarch64_neon_tbl2:
6246 SelectTable(Node, 2,
6247 VT == MVT::v8i8 ? AArch64::TBLv8i8Two : AArch64::TBLv16i8Two,
6248 false);
6249 return;
6250 case Intrinsic::aarch64_neon_tbl3:
6251 SelectTable(Node, 3, VT == MVT::v8i8 ? AArch64::TBLv8i8Three
6252 : AArch64::TBLv16i8Three,
6253 false);
6254 return;
6255 case Intrinsic::aarch64_neon_tbl4:
6256 SelectTable(Node, 4, VT == MVT::v8i8 ? AArch64::TBLv8i8Four
6257 : AArch64::TBLv16i8Four,
6258 false);
6259 return;
6260 case Intrinsic::aarch64_neon_tbx2:
6261 SelectTable(Node, 2,
6262 VT == MVT::v8i8 ? AArch64::TBXv8i8Two : AArch64::TBXv16i8Two,
6263 true);
6264 return;
6265 case Intrinsic::aarch64_neon_tbx3:
6266 SelectTable(Node, 3, VT == MVT::v8i8 ? AArch64::TBXv8i8Three
6267 : AArch64::TBXv16i8Three,
6268 true);
6269 return;
6270 case Intrinsic::aarch64_neon_tbx4:
6271 SelectTable(Node, 4, VT == MVT::v8i8 ? AArch64::TBXv8i8Four
6272 : AArch64::TBXv16i8Four,
6273 true);
6274 return;
6275 case Intrinsic::aarch64_sve_srshl_single_x2:
6277 Node->getValueType(0),
6278 {AArch64::SRSHL_VG2_2ZZ_B, AArch64::SRSHL_VG2_2ZZ_H,
6279 AArch64::SRSHL_VG2_2ZZ_S, AArch64::SRSHL_VG2_2ZZ_D}))
6280 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6281 return;
6282 case Intrinsic::aarch64_sve_srshl_single_x4:
6284 Node->getValueType(0),
6285 {AArch64::SRSHL_VG4_4ZZ_B, AArch64::SRSHL_VG4_4ZZ_H,
6286 AArch64::SRSHL_VG4_4ZZ_S, AArch64::SRSHL_VG4_4ZZ_D}))
6287 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6288 return;
6289 case Intrinsic::aarch64_sme_luti6_lane_x4_x2:
6290 SelectMultiVectorLuti6LaneX4(Node, 2);
6291 return;
6292 case Intrinsic::aarch64_sme_luti6_lane_x4_x3:
6293 SelectMultiVectorLuti6LaneX4(Node, 3);
6294 return;
6295 case Intrinsic::aarch64_sve_urshl_single_x2:
6297 Node->getValueType(0),
6298 {AArch64::URSHL_VG2_2ZZ_B, AArch64::URSHL_VG2_2ZZ_H,
6299 AArch64::URSHL_VG2_2ZZ_S, AArch64::URSHL_VG2_2ZZ_D}))
6300 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6301 return;
6302 case Intrinsic::aarch64_sve_urshl_single_x4:
6304 Node->getValueType(0),
6305 {AArch64::URSHL_VG4_4ZZ_B, AArch64::URSHL_VG4_4ZZ_H,
6306 AArch64::URSHL_VG4_4ZZ_S, AArch64::URSHL_VG4_4ZZ_D}))
6307 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6308 return;
6309 case Intrinsic::aarch64_sve_srshl_x2:
6311 Node->getValueType(0),
6312 {AArch64::SRSHL_VG2_2Z2Z_B, AArch64::SRSHL_VG2_2Z2Z_H,
6313 AArch64::SRSHL_VG2_2Z2Z_S, AArch64::SRSHL_VG2_2Z2Z_D}))
6314 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6315 return;
6316 case Intrinsic::aarch64_sve_srshl_x4:
6318 Node->getValueType(0),
6319 {AArch64::SRSHL_VG4_4Z4Z_B, AArch64::SRSHL_VG4_4Z4Z_H,
6320 AArch64::SRSHL_VG4_4Z4Z_S, AArch64::SRSHL_VG4_4Z4Z_D}))
6321 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6322 return;
6323 case Intrinsic::aarch64_sve_urshl_x2:
6325 Node->getValueType(0),
6326 {AArch64::URSHL_VG2_2Z2Z_B, AArch64::URSHL_VG2_2Z2Z_H,
6327 AArch64::URSHL_VG2_2Z2Z_S, AArch64::URSHL_VG2_2Z2Z_D}))
6328 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6329 return;
6330 case Intrinsic::aarch64_sve_urshl_x4:
6332 Node->getValueType(0),
6333 {AArch64::URSHL_VG4_4Z4Z_B, AArch64::URSHL_VG4_4Z4Z_H,
6334 AArch64::URSHL_VG4_4Z4Z_S, AArch64::URSHL_VG4_4Z4Z_D}))
6335 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6336 return;
6337 case Intrinsic::aarch64_sve_sqdmulh_single_vgx2:
6339 Node->getValueType(0),
6340 {AArch64::SQDMULH_VG2_2ZZ_B, AArch64::SQDMULH_VG2_2ZZ_H,
6341 AArch64::SQDMULH_VG2_2ZZ_S, AArch64::SQDMULH_VG2_2ZZ_D}))
6342 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6343 return;
6344 case Intrinsic::aarch64_sve_sqdmulh_single_vgx4:
6346 Node->getValueType(0),
6347 {AArch64::SQDMULH_VG4_4ZZ_B, AArch64::SQDMULH_VG4_4ZZ_H,
6348 AArch64::SQDMULH_VG4_4ZZ_S, AArch64::SQDMULH_VG4_4ZZ_D}))
6349 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6350 return;
6351 case Intrinsic::aarch64_sve_sqdmulh_vgx2:
6353 Node->getValueType(0),
6354 {AArch64::SQDMULH_VG2_2Z2Z_B, AArch64::SQDMULH_VG2_2Z2Z_H,
6355 AArch64::SQDMULH_VG2_2Z2Z_S, AArch64::SQDMULH_VG2_2Z2Z_D}))
6356 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6357 return;
6358 case Intrinsic::aarch64_sve_sqdmulh_vgx4:
6360 Node->getValueType(0),
6361 {AArch64::SQDMULH_VG4_4Z4Z_B, AArch64::SQDMULH_VG4_4Z4Z_H,
6362 AArch64::SQDMULH_VG4_4Z4Z_S, AArch64::SQDMULH_VG4_4Z4Z_D}))
6363 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6364 return;
6365 case Intrinsic::aarch64_sme_fp8_scale_single_x2:
6367 Node->getValueType(0),
6368 {0, AArch64::FSCALE_2ZZ_H, AArch64::FSCALE_2ZZ_S,
6369 AArch64::FSCALE_2ZZ_D}))
6370 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6371 return;
6372 case Intrinsic::aarch64_sme_fp8_scale_single_x4:
6374 Node->getValueType(0),
6375 {0, AArch64::FSCALE_4ZZ_H, AArch64::FSCALE_4ZZ_S,
6376 AArch64::FSCALE_4ZZ_D}))
6377 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6378 return;
6379 case Intrinsic::aarch64_sme_fp8_scale_x2:
6381 Node->getValueType(0),
6382 {0, AArch64::FSCALE_2Z2Z_H, AArch64::FSCALE_2Z2Z_S,
6383 AArch64::FSCALE_2Z2Z_D}))
6384 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6385 return;
6386 case Intrinsic::aarch64_sme_fp8_scale_x4:
6388 Node->getValueType(0),
6389 {0, AArch64::FSCALE_4Z4Z_H, AArch64::FSCALE_4Z4Z_S,
6390 AArch64::FSCALE_4Z4Z_D}))
6391 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6392 return;
6393 case Intrinsic::aarch64_sve_whilege_x2:
6395 Node->getValueType(0),
6396 {AArch64::WHILEGE_2PXX_B, AArch64::WHILEGE_2PXX_H,
6397 AArch64::WHILEGE_2PXX_S, AArch64::WHILEGE_2PXX_D}))
6398 SelectWhilePair(Node, Op);
6399 return;
6400 case Intrinsic::aarch64_sve_whilegt_x2:
6402 Node->getValueType(0),
6403 {AArch64::WHILEGT_2PXX_B, AArch64::WHILEGT_2PXX_H,
6404 AArch64::WHILEGT_2PXX_S, AArch64::WHILEGT_2PXX_D}))
6405 SelectWhilePair(Node, Op);
6406 return;
6407 case Intrinsic::aarch64_sve_whilehi_x2:
6409 Node->getValueType(0),
6410 {AArch64::WHILEHI_2PXX_B, AArch64::WHILEHI_2PXX_H,
6411 AArch64::WHILEHI_2PXX_S, AArch64::WHILEHI_2PXX_D}))
6412 SelectWhilePair(Node, Op);
6413 return;
6414 case Intrinsic::aarch64_sve_whilehs_x2:
6416 Node->getValueType(0),
6417 {AArch64::WHILEHS_2PXX_B, AArch64::WHILEHS_2PXX_H,
6418 AArch64::WHILEHS_2PXX_S, AArch64::WHILEHS_2PXX_D}))
6419 SelectWhilePair(Node, Op);
6420 return;
6421 case Intrinsic::aarch64_sve_whilele_x2:
6423 Node->getValueType(0),
6424 {AArch64::WHILELE_2PXX_B, AArch64::WHILELE_2PXX_H,
6425 AArch64::WHILELE_2PXX_S, AArch64::WHILELE_2PXX_D}))
6426 SelectWhilePair(Node, Op);
6427 return;
6428 case Intrinsic::aarch64_sve_whilelo_x2:
6430 Node->getValueType(0),
6431 {AArch64::WHILELO_2PXX_B, AArch64::WHILELO_2PXX_H,
6432 AArch64::WHILELO_2PXX_S, AArch64::WHILELO_2PXX_D}))
6433 SelectWhilePair(Node, Op);
6434 return;
6435 case Intrinsic::aarch64_sve_whilels_x2:
6437 Node->getValueType(0),
6438 {AArch64::WHILELS_2PXX_B, AArch64::WHILELS_2PXX_H,
6439 AArch64::WHILELS_2PXX_S, AArch64::WHILELS_2PXX_D}))
6440 SelectWhilePair(Node, Op);
6441 return;
6442 case Intrinsic::aarch64_sve_whilelt_x2:
6444 Node->getValueType(0),
6445 {AArch64::WHILELT_2PXX_B, AArch64::WHILELT_2PXX_H,
6446 AArch64::WHILELT_2PXX_S, AArch64::WHILELT_2PXX_D}))
6447 SelectWhilePair(Node, Op);
6448 return;
6449 case Intrinsic::aarch64_sve_smax_single_x2:
6451 Node->getValueType(0),
6452 {AArch64::SMAX_VG2_2ZZ_B, AArch64::SMAX_VG2_2ZZ_H,
6453 AArch64::SMAX_VG2_2ZZ_S, AArch64::SMAX_VG2_2ZZ_D}))
6454 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6455 return;
6456 case Intrinsic::aarch64_sve_umax_single_x2:
6458 Node->getValueType(0),
6459 {AArch64::UMAX_VG2_2ZZ_B, AArch64::UMAX_VG2_2ZZ_H,
6460 AArch64::UMAX_VG2_2ZZ_S, AArch64::UMAX_VG2_2ZZ_D}))
6461 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6462 return;
6463 case Intrinsic::aarch64_sve_fmax_single_x2:
6465 Node->getValueType(0),
6466 {AArch64::BFMAX_VG2_2ZZ_H, AArch64::FMAX_VG2_2ZZ_H,
6467 AArch64::FMAX_VG2_2ZZ_S, AArch64::FMAX_VG2_2ZZ_D}))
6468 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6469 return;
6470 case Intrinsic::aarch64_sve_smax_single_x4:
6472 Node->getValueType(0),
6473 {AArch64::SMAX_VG4_4ZZ_B, AArch64::SMAX_VG4_4ZZ_H,
6474 AArch64::SMAX_VG4_4ZZ_S, AArch64::SMAX_VG4_4ZZ_D}))
6475 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6476 return;
6477 case Intrinsic::aarch64_sve_umax_single_x4:
6479 Node->getValueType(0),
6480 {AArch64::UMAX_VG4_4ZZ_B, AArch64::UMAX_VG4_4ZZ_H,
6481 AArch64::UMAX_VG4_4ZZ_S, AArch64::UMAX_VG4_4ZZ_D}))
6482 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6483 return;
6484 case Intrinsic::aarch64_sve_fmax_single_x4:
6486 Node->getValueType(0),
6487 {AArch64::BFMAX_VG4_4ZZ_H, AArch64::FMAX_VG4_4ZZ_H,
6488 AArch64::FMAX_VG4_4ZZ_S, AArch64::FMAX_VG4_4ZZ_D}))
6489 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6490 return;
6491 case Intrinsic::aarch64_sve_smin_single_x2:
6493 Node->getValueType(0),
6494 {AArch64::SMIN_VG2_2ZZ_B, AArch64::SMIN_VG2_2ZZ_H,
6495 AArch64::SMIN_VG2_2ZZ_S, AArch64::SMIN_VG2_2ZZ_D}))
6496 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6497 return;
6498 case Intrinsic::aarch64_sve_umin_single_x2:
6500 Node->getValueType(0),
6501 {AArch64::UMIN_VG2_2ZZ_B, AArch64::UMIN_VG2_2ZZ_H,
6502 AArch64::UMIN_VG2_2ZZ_S, AArch64::UMIN_VG2_2ZZ_D}))
6503 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6504 return;
6505 case Intrinsic::aarch64_sve_fmin_single_x2:
6507 Node->getValueType(0),
6508 {AArch64::BFMIN_VG2_2ZZ_H, AArch64::FMIN_VG2_2ZZ_H,
6509 AArch64::FMIN_VG2_2ZZ_S, AArch64::FMIN_VG2_2ZZ_D}))
6510 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6511 return;
6512 case Intrinsic::aarch64_sve_smin_single_x4:
6514 Node->getValueType(0),
6515 {AArch64::SMIN_VG4_4ZZ_B, AArch64::SMIN_VG4_4ZZ_H,
6516 AArch64::SMIN_VG4_4ZZ_S, AArch64::SMIN_VG4_4ZZ_D}))
6517 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6518 return;
6519 case Intrinsic::aarch64_sve_umin_single_x4:
6521 Node->getValueType(0),
6522 {AArch64::UMIN_VG4_4ZZ_B, AArch64::UMIN_VG4_4ZZ_H,
6523 AArch64::UMIN_VG4_4ZZ_S, AArch64::UMIN_VG4_4ZZ_D}))
6524 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6525 return;
6526 case Intrinsic::aarch64_sve_fmin_single_x4:
6528 Node->getValueType(0),
6529 {AArch64::BFMIN_VG4_4ZZ_H, AArch64::FMIN_VG4_4ZZ_H,
6530 AArch64::FMIN_VG4_4ZZ_S, AArch64::FMIN_VG4_4ZZ_D}))
6531 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6532 return;
6533 case Intrinsic::aarch64_sve_smax_x2:
6535 Node->getValueType(0),
6536 {AArch64::SMAX_VG2_2Z2Z_B, AArch64::SMAX_VG2_2Z2Z_H,
6537 AArch64::SMAX_VG2_2Z2Z_S, AArch64::SMAX_VG2_2Z2Z_D}))
6538 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6539 return;
6540 case Intrinsic::aarch64_sve_umax_x2:
6542 Node->getValueType(0),
6543 {AArch64::UMAX_VG2_2Z2Z_B, AArch64::UMAX_VG2_2Z2Z_H,
6544 AArch64::UMAX_VG2_2Z2Z_S, AArch64::UMAX_VG2_2Z2Z_D}))
6545 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6546 return;
6547 case Intrinsic::aarch64_sve_fmax_x2:
6549 Node->getValueType(0),
6550 {AArch64::BFMAX_VG2_2Z2Z_H, AArch64::FMAX_VG2_2Z2Z_H,
6551 AArch64::FMAX_VG2_2Z2Z_S, AArch64::FMAX_VG2_2Z2Z_D}))
6552 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6553 return;
6554 case Intrinsic::aarch64_sve_smax_x4:
6556 Node->getValueType(0),
6557 {AArch64::SMAX_VG4_4Z4Z_B, AArch64::SMAX_VG4_4Z4Z_H,
6558 AArch64::SMAX_VG4_4Z4Z_S, AArch64::SMAX_VG4_4Z4Z_D}))
6559 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6560 return;
6561 case Intrinsic::aarch64_sve_umax_x4:
6563 Node->getValueType(0),
6564 {AArch64::UMAX_VG4_4Z4Z_B, AArch64::UMAX_VG4_4Z4Z_H,
6565 AArch64::UMAX_VG4_4Z4Z_S, AArch64::UMAX_VG4_4Z4Z_D}))
6566 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6567 return;
6568 case Intrinsic::aarch64_sve_fmax_x4:
6570 Node->getValueType(0),
6571 {AArch64::BFMAX_VG4_4Z2Z_H, AArch64::FMAX_VG4_4Z4Z_H,
6572 AArch64::FMAX_VG4_4Z4Z_S, AArch64::FMAX_VG4_4Z4Z_D}))
6573 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6574 return;
6575 case Intrinsic::aarch64_sme_famax_x2:
6577 Node->getValueType(0),
6578 {0, AArch64::FAMAX_2Z2Z_H, AArch64::FAMAX_2Z2Z_S,
6579 AArch64::FAMAX_2Z2Z_D}))
6580 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6581 return;
6582 case Intrinsic::aarch64_sme_famax_x4:
6584 Node->getValueType(0),
6585 {0, AArch64::FAMAX_4Z4Z_H, AArch64::FAMAX_4Z4Z_S,
6586 AArch64::FAMAX_4Z4Z_D}))
6587 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6588 return;
6589 case Intrinsic::aarch64_sme_famin_x2:
6591 Node->getValueType(0),
6592 {0, AArch64::FAMIN_2Z2Z_H, AArch64::FAMIN_2Z2Z_S,
6593 AArch64::FAMIN_2Z2Z_D}))
6594 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6595 return;
6596 case Intrinsic::aarch64_sme_famin_x4:
6598 Node->getValueType(0),
6599 {0, AArch64::FAMIN_4Z4Z_H, AArch64::FAMIN_4Z4Z_S,
6600 AArch64::FAMIN_4Z4Z_D}))
6601 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6602 return;
6603 case Intrinsic::aarch64_sve_smin_x2:
6605 Node->getValueType(0),
6606 {AArch64::SMIN_VG2_2Z2Z_B, AArch64::SMIN_VG2_2Z2Z_H,
6607 AArch64::SMIN_VG2_2Z2Z_S, AArch64::SMIN_VG2_2Z2Z_D}))
6608 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6609 return;
6610 case Intrinsic::aarch64_sve_umin_x2:
6612 Node->getValueType(0),
6613 {AArch64::UMIN_VG2_2Z2Z_B, AArch64::UMIN_VG2_2Z2Z_H,
6614 AArch64::UMIN_VG2_2Z2Z_S, AArch64::UMIN_VG2_2Z2Z_D}))
6615 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6616 return;
6617 case Intrinsic::aarch64_sve_fmin_x2:
6619 Node->getValueType(0),
6620 {AArch64::BFMIN_VG2_2Z2Z_H, AArch64::FMIN_VG2_2Z2Z_H,
6621 AArch64::FMIN_VG2_2Z2Z_S, AArch64::FMIN_VG2_2Z2Z_D}))
6622 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6623 return;
6624 case Intrinsic::aarch64_sve_smin_x4:
6626 Node->getValueType(0),
6627 {AArch64::SMIN_VG4_4Z4Z_B, AArch64::SMIN_VG4_4Z4Z_H,
6628 AArch64::SMIN_VG4_4Z4Z_S, AArch64::SMIN_VG4_4Z4Z_D}))
6629 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6630 return;
6631 case Intrinsic::aarch64_sve_umin_x4:
6633 Node->getValueType(0),
6634 {AArch64::UMIN_VG4_4Z4Z_B, AArch64::UMIN_VG4_4Z4Z_H,
6635 AArch64::UMIN_VG4_4Z4Z_S, AArch64::UMIN_VG4_4Z4Z_D}))
6636 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6637 return;
6638 case Intrinsic::aarch64_sve_fmin_x4:
6640 Node->getValueType(0),
6641 {AArch64::BFMIN_VG4_4Z2Z_H, AArch64::FMIN_VG4_4Z4Z_H,
6642 AArch64::FMIN_VG4_4Z4Z_S, AArch64::FMIN_VG4_4Z4Z_D}))
6643 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6644 return;
6645 case Intrinsic::aarch64_sve_fmaxnm_single_x2 :
6647 Node->getValueType(0),
6648 {AArch64::BFMAXNM_VG2_2ZZ_H, AArch64::FMAXNM_VG2_2ZZ_H,
6649 AArch64::FMAXNM_VG2_2ZZ_S, AArch64::FMAXNM_VG2_2ZZ_D}))
6650 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6651 return;
6652 case Intrinsic::aarch64_sve_fmaxnm_single_x4 :
6654 Node->getValueType(0),
6655 {AArch64::BFMAXNM_VG4_4ZZ_H, AArch64::FMAXNM_VG4_4ZZ_H,
6656 AArch64::FMAXNM_VG4_4ZZ_S, AArch64::FMAXNM_VG4_4ZZ_D}))
6657 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6658 return;
6659 case Intrinsic::aarch64_sve_fminnm_single_x2:
6661 Node->getValueType(0),
6662 {AArch64::BFMINNM_VG2_2ZZ_H, AArch64::FMINNM_VG2_2ZZ_H,
6663 AArch64::FMINNM_VG2_2ZZ_S, AArch64::FMINNM_VG2_2ZZ_D}))
6664 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6665 return;
6666 case Intrinsic::aarch64_sve_fminnm_single_x4:
6668 Node->getValueType(0),
6669 {AArch64::BFMINNM_VG4_4ZZ_H, AArch64::FMINNM_VG4_4ZZ_H,
6670 AArch64::FMINNM_VG4_4ZZ_S, AArch64::FMINNM_VG4_4ZZ_D}))
6671 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6672 return;
6673 case Intrinsic::aarch64_sve_fscale_single_x4:
6674 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::BFSCALE_4ZZ);
6675 return;
6676 case Intrinsic::aarch64_sve_fscale_single_x2:
6677 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::BFSCALE_2ZZ);
6678 return;
6679 case Intrinsic::aarch64_sve_fmul_single_x4:
6681 Node->getValueType(0),
6682 {AArch64::BFMUL_4ZZ, AArch64::FMUL_4ZZ_H, AArch64::FMUL_4ZZ_S,
6683 AArch64::FMUL_4ZZ_D}))
6684 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6685 return;
6686 case Intrinsic::aarch64_sve_fmul_single_x2:
6688 Node->getValueType(0),
6689 {AArch64::BFMUL_2ZZ, AArch64::FMUL_2ZZ_H, AArch64::FMUL_2ZZ_S,
6690 AArch64::FMUL_2ZZ_D}))
6691 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6692 return;
6693 case Intrinsic::aarch64_sve_fmaxnm_x2:
6695 Node->getValueType(0),
6696 {AArch64::BFMAXNM_VG2_2Z2Z_H, AArch64::FMAXNM_VG2_2Z2Z_H,
6697 AArch64::FMAXNM_VG2_2Z2Z_S, AArch64::FMAXNM_VG2_2Z2Z_D}))
6698 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6699 return;
6700 case Intrinsic::aarch64_sve_fmaxnm_x4:
6702 Node->getValueType(0),
6703 {AArch64::BFMAXNM_VG4_4Z2Z_H, AArch64::FMAXNM_VG4_4Z4Z_H,
6704 AArch64::FMAXNM_VG4_4Z4Z_S, AArch64::FMAXNM_VG4_4Z4Z_D}))
6705 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6706 return;
6707 case Intrinsic::aarch64_sve_fminnm_x2:
6709 Node->getValueType(0),
6710 {AArch64::BFMINNM_VG2_2Z2Z_H, AArch64::FMINNM_VG2_2Z2Z_H,
6711 AArch64::FMINNM_VG2_2Z2Z_S, AArch64::FMINNM_VG2_2Z2Z_D}))
6712 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6713 return;
6714 case Intrinsic::aarch64_sve_fminnm_x4:
6716 Node->getValueType(0),
6717 {AArch64::BFMINNM_VG4_4Z2Z_H, AArch64::FMINNM_VG4_4Z4Z_H,
6718 AArch64::FMINNM_VG4_4Z4Z_S, AArch64::FMINNM_VG4_4Z4Z_D}))
6719 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6720 return;
6721 case Intrinsic::aarch64_sve_aese_lane_x2:
6722 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::AESE_2ZZI_B);
6723 return;
6724 case Intrinsic::aarch64_sve_aesd_lane_x2:
6725 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::AESD_2ZZI_B);
6726 return;
6727 case Intrinsic::aarch64_sve_aesemc_lane_x2:
6728 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::AESEMC_2ZZI_B);
6729 return;
6730 case Intrinsic::aarch64_sve_aesdimc_lane_x2:
6731 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::AESDIMC_2ZZI_B);
6732 return;
6733 case Intrinsic::aarch64_sve_aese_lane_x4:
6734 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::AESE_4ZZI_B);
6735 return;
6736 case Intrinsic::aarch64_sve_aesd_lane_x4:
6737 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::AESD_4ZZI_B);
6738 return;
6739 case Intrinsic::aarch64_sve_aesemc_lane_x4:
6740 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::AESEMC_4ZZI_B);
6741 return;
6742 case Intrinsic::aarch64_sve_aesdimc_lane_x4:
6743 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::AESDIMC_4ZZI_B);
6744 return;
6745 case Intrinsic::aarch64_sve_pmlal_pair_x2:
6746 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::PMLAL_2ZZZ_Q);
6747 return;
6748 case Intrinsic::aarch64_sve_pmull_pair_x2: {
6749 SDLoc DL(Node);
6750 SmallVector<SDValue, 4> Regs(Node->ops().slice(1, 2));
6751 SDNode *Res =
6752 CurDAG->getMachineNode(AArch64::PMULL_2ZZZ_Q, DL, MVT::Untyped, Regs);
6753 SDValue SuperReg = SDValue(Res, 0);
6754 for (unsigned I = 0; I < 2; I++)
6755 ReplaceUses(SDValue(Node, I),
6756 CurDAG->getTargetExtractSubreg(AArch64::zsub0 + I, DL, VT,
6757 SuperReg));
6758 CurDAG->RemoveDeadNode(Node);
6759 return;
6760 }
6761 case Intrinsic::aarch64_sve_fscale_x4:
6762 SelectDestructiveMultiIntrinsic(Node, 4, true, AArch64::BFSCALE_4Z4Z);
6763 return;
6764 case Intrinsic::aarch64_sve_fscale_x2:
6765 SelectDestructiveMultiIntrinsic(Node, 2, true, AArch64::BFSCALE_2Z2Z);
6766 return;
6767 case Intrinsic::aarch64_sve_fmul_x4:
6769 Node->getValueType(0),
6770 {AArch64::BFMUL_4Z4Z, AArch64::FMUL_4Z4Z_H, AArch64::FMUL_4Z4Z_S,
6771 AArch64::FMUL_4Z4Z_D}))
6772 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6773 return;
6774 case Intrinsic::aarch64_sve_fmul_x2:
6776 Node->getValueType(0),
6777 {AArch64::BFMUL_2Z2Z, AArch64::FMUL_2Z2Z_H, AArch64::FMUL_2Z2Z_S,
6778 AArch64::FMUL_2Z2Z_D}))
6779 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6780 return;
6781 case Intrinsic::aarch64_sve_fcvtzs_x2:
6782 SelectCVTIntrinsic(Node, 2, AArch64::FCVTZS_2Z2Z_StoS);
6783 return;
6784 case Intrinsic::aarch64_sve_scvtf_x2:
6785 SelectCVTIntrinsic(Node, 2, AArch64::SCVTF_2Z2Z_StoS);
6786 return;
6787 case Intrinsic::aarch64_sve_fcvtzu_x2:
6788 SelectCVTIntrinsic(Node, 2, AArch64::FCVTZU_2Z2Z_StoS);
6789 return;
6790 case Intrinsic::aarch64_sve_ucvtf_x2:
6791 SelectCVTIntrinsic(Node, 2, AArch64::UCVTF_2Z2Z_StoS);
6792 return;
6793 case Intrinsic::aarch64_sve_fcvtzs_x4:
6794 SelectCVTIntrinsic(Node, 4, AArch64::FCVTZS_4Z4Z_StoS);
6795 return;
6796 case Intrinsic::aarch64_sve_scvtf_x4:
6797 SelectCVTIntrinsic(Node, 4, AArch64::SCVTF_4Z4Z_StoS);
6798 return;
6799 case Intrinsic::aarch64_sve_fcvtzu_x4:
6800 SelectCVTIntrinsic(Node, 4, AArch64::FCVTZU_4Z4Z_StoS);
6801 return;
6802 case Intrinsic::aarch64_sve_ucvtf_x4:
6803 SelectCVTIntrinsic(Node, 4, AArch64::UCVTF_4Z4Z_StoS);
6804 return;
6805 case Intrinsic::aarch64_sve_fcvt_widen_x2:
6806 SelectUnaryMultiIntrinsic(Node, 2, false, AArch64::FCVT_2ZZ_H_S);
6807 return;
6808 case Intrinsic::aarch64_sve_fcvtl_widen_x2:
6809 SelectUnaryMultiIntrinsic(Node, 2, false, AArch64::FCVTL_2ZZ_H_S);
6810 return;
6811 case Intrinsic::aarch64_sve_sclamp_single_x2:
6813 Node->getValueType(0),
6814 {AArch64::SCLAMP_VG2_2Z2Z_B, AArch64::SCLAMP_VG2_2Z2Z_H,
6815 AArch64::SCLAMP_VG2_2Z2Z_S, AArch64::SCLAMP_VG2_2Z2Z_D}))
6816 SelectClamp(Node, 2, Op);
6817 return;
6818 case Intrinsic::aarch64_sve_uclamp_single_x2:
6820 Node->getValueType(0),
6821 {AArch64::UCLAMP_VG2_2Z2Z_B, AArch64::UCLAMP_VG2_2Z2Z_H,
6822 AArch64::UCLAMP_VG2_2Z2Z_S, AArch64::UCLAMP_VG2_2Z2Z_D}))
6823 SelectClamp(Node, 2, Op);
6824 return;
6825 case Intrinsic::aarch64_sve_fclamp_single_x2:
6827 Node->getValueType(0),
6828 {0, AArch64::FCLAMP_VG2_2Z2Z_H, AArch64::FCLAMP_VG2_2Z2Z_S,
6829 AArch64::FCLAMP_VG2_2Z2Z_D}))
6830 SelectClamp(Node, 2, Op);
6831 return;
6832 case Intrinsic::aarch64_sve_bfclamp_single_x2:
6833 SelectClamp(Node, 2, AArch64::BFCLAMP_VG2_2ZZZ_H);
6834 return;
6835 case Intrinsic::aarch64_sve_sclamp_single_x4:
6837 Node->getValueType(0),
6838 {AArch64::SCLAMP_VG4_4Z4Z_B, AArch64::SCLAMP_VG4_4Z4Z_H,
6839 AArch64::SCLAMP_VG4_4Z4Z_S, AArch64::SCLAMP_VG4_4Z4Z_D}))
6840 SelectClamp(Node, 4, Op);
6841 return;
6842 case Intrinsic::aarch64_sve_uclamp_single_x4:
6844 Node->getValueType(0),
6845 {AArch64::UCLAMP_VG4_4Z4Z_B, AArch64::UCLAMP_VG4_4Z4Z_H,
6846 AArch64::UCLAMP_VG4_4Z4Z_S, AArch64::UCLAMP_VG4_4Z4Z_D}))
6847 SelectClamp(Node, 4, Op);
6848 return;
6849 case Intrinsic::aarch64_sve_fclamp_single_x4:
6851 Node->getValueType(0),
6852 {0, AArch64::FCLAMP_VG4_4Z4Z_H, AArch64::FCLAMP_VG4_4Z4Z_S,
6853 AArch64::FCLAMP_VG4_4Z4Z_D}))
6854 SelectClamp(Node, 4, Op);
6855 return;
6856 case Intrinsic::aarch64_sve_bfclamp_single_x4:
6857 SelectClamp(Node, 4, AArch64::BFCLAMP_VG4_4ZZZ_H);
6858 return;
6859 case Intrinsic::aarch64_sve_add_single_x2:
6861 Node->getValueType(0),
6862 {AArch64::ADD_VG2_2ZZ_B, AArch64::ADD_VG2_2ZZ_H,
6863 AArch64::ADD_VG2_2ZZ_S, AArch64::ADD_VG2_2ZZ_D}))
6864 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6865 return;
6866 case Intrinsic::aarch64_sve_add_single_x4:
6868 Node->getValueType(0),
6869 {AArch64::ADD_VG4_4ZZ_B, AArch64::ADD_VG4_4ZZ_H,
6870 AArch64::ADD_VG4_4ZZ_S, AArch64::ADD_VG4_4ZZ_D}))
6871 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6872 return;
6873 case Intrinsic::aarch64_sve_zip_x2:
6875 Node->getValueType(0),
6876 {AArch64::ZIP_VG2_2ZZZ_B, AArch64::ZIP_VG2_2ZZZ_H,
6877 AArch64::ZIP_VG2_2ZZZ_S, AArch64::ZIP_VG2_2ZZZ_D}))
6878 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false, Op);
6879 return;
6880 case Intrinsic::aarch64_sve_zipq_x2:
6881 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false,
6882 AArch64::ZIP_VG2_2ZZZ_Q);
6883 return;
6884 case Intrinsic::aarch64_sve_zip_x4:
6886 Node->getValueType(0),
6887 {AArch64::ZIP_VG4_4Z4Z_B, AArch64::ZIP_VG4_4Z4Z_H,
6888 AArch64::ZIP_VG4_4Z4Z_S, AArch64::ZIP_VG4_4Z4Z_D}))
6889 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true, Op);
6890 return;
6891 case Intrinsic::aarch64_sve_zipq_x4:
6892 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true,
6893 AArch64::ZIP_VG4_4Z4Z_Q);
6894 return;
6895 case Intrinsic::aarch64_sve_uzp_x2:
6897 Node->getValueType(0),
6898 {AArch64::UZP_VG2_2ZZZ_B, AArch64::UZP_VG2_2ZZZ_H,
6899 AArch64::UZP_VG2_2ZZZ_S, AArch64::UZP_VG2_2ZZZ_D}))
6900 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false, Op);
6901 return;
6902 case Intrinsic::aarch64_sve_uzpq_x2:
6903 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false,
6904 AArch64::UZP_VG2_2ZZZ_Q);
6905 return;
6906 case Intrinsic::aarch64_sve_uzp_x4:
6908 Node->getValueType(0),
6909 {AArch64::UZP_VG4_4Z4Z_B, AArch64::UZP_VG4_4Z4Z_H,
6910 AArch64::UZP_VG4_4Z4Z_S, AArch64::UZP_VG4_4Z4Z_D}))
6911 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true, Op);
6912 return;
6913 case Intrinsic::aarch64_sve_uzpq_x4:
6914 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true,
6915 AArch64::UZP_VG4_4Z4Z_Q);
6916 return;
6917 case Intrinsic::aarch64_sve_sel_x2:
6919 Node->getValueType(0),
6920 {AArch64::SEL_VG2_2ZC2Z2Z_B, AArch64::SEL_VG2_2ZC2Z2Z_H,
6921 AArch64::SEL_VG2_2ZC2Z2Z_S, AArch64::SEL_VG2_2ZC2Z2Z_D}))
6922 SelectDestructiveMultiIntrinsic(Node, 2, true, Op, /*HasPred=*/true);
6923 return;
6924 case Intrinsic::aarch64_sve_sel_x4:
6926 Node->getValueType(0),
6927 {AArch64::SEL_VG4_4ZC4Z4Z_B, AArch64::SEL_VG4_4ZC4Z4Z_H,
6928 AArch64::SEL_VG4_4ZC4Z4Z_S, AArch64::SEL_VG4_4ZC4Z4Z_D}))
6929 SelectDestructiveMultiIntrinsic(Node, 4, true, Op, /*HasPred=*/true);
6930 return;
6931 case Intrinsic::aarch64_sve_frinta_x2:
6932 SelectFrintFromVT(Node, 2, AArch64::FRINTA_2Z2Z_S);
6933 return;
6934 case Intrinsic::aarch64_sve_frinta_x4:
6935 SelectFrintFromVT(Node, 4, AArch64::FRINTA_4Z4Z_S);
6936 return;
6937 case Intrinsic::aarch64_sve_frintm_x2:
6938 SelectFrintFromVT(Node, 2, AArch64::FRINTM_2Z2Z_S);
6939 return;
6940 case Intrinsic::aarch64_sve_frintm_x4:
6941 SelectFrintFromVT(Node, 4, AArch64::FRINTM_4Z4Z_S);
6942 return;
6943 case Intrinsic::aarch64_sve_frintn_x2:
6944 SelectFrintFromVT(Node, 2, AArch64::FRINTN_2Z2Z_S);
6945 return;
6946 case Intrinsic::aarch64_sve_frintn_x4:
6947 SelectFrintFromVT(Node, 4, AArch64::FRINTN_4Z4Z_S);
6948 return;
6949 case Intrinsic::aarch64_sve_frintp_x2:
6950 SelectFrintFromVT(Node, 2, AArch64::FRINTP_2Z2Z_S);
6951 return;
6952 case Intrinsic::aarch64_sve_frintp_x4:
6953 SelectFrintFromVT(Node, 4, AArch64::FRINTP_4Z4Z_S);
6954 return;
6955 case Intrinsic::aarch64_sve_sunpk_x2:
6957 Node->getValueType(0),
6958 {0, AArch64::SUNPK_VG2_2ZZ_H, AArch64::SUNPK_VG2_2ZZ_S,
6959 AArch64::SUNPK_VG2_2ZZ_D}))
6960 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false, Op);
6961 return;
6962 case Intrinsic::aarch64_sve_uunpk_x2:
6964 Node->getValueType(0),
6965 {0, AArch64::UUNPK_VG2_2ZZ_H, AArch64::UUNPK_VG2_2ZZ_S,
6966 AArch64::UUNPK_VG2_2ZZ_D}))
6967 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false, Op);
6968 return;
6969 case Intrinsic::aarch64_sve_sunpk_x4:
6971 Node->getValueType(0),
6972 {0, AArch64::SUNPK_VG4_4Z2Z_H, AArch64::SUNPK_VG4_4Z2Z_S,
6973 AArch64::SUNPK_VG4_4Z2Z_D}))
6974 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true, Op);
6975 return;
6976 case Intrinsic::aarch64_sve_uunpk_x4:
6978 Node->getValueType(0),
6979 {0, AArch64::UUNPK_VG4_4Z2Z_H, AArch64::UUNPK_VG4_4Z2Z_S,
6980 AArch64::UUNPK_VG4_4Z2Z_D}))
6981 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true, Op);
6982 return;
6983 case Intrinsic::aarch64_sve_pext_x2: {
6985 Node->getValueType(0),
6986 {AArch64::PEXT_2PCI_B, AArch64::PEXT_2PCI_H, AArch64::PEXT_2PCI_S,
6987 AArch64::PEXT_2PCI_D}))
6988 SelectPExtPair(Node, Op);
6989 return;
6990 }
6991 }
6992 break;
6993 }
6994 case ISD::INTRINSIC_VOID: {
6995 unsigned IntNo = Node->getConstantOperandVal(1);
6996 if (Node->getNumOperands() >= 3)
6997 VT = Node->getOperand(2)->getValueType(0);
6998 switch (IntNo) {
6999 default:
7000 break;
7001 case Intrinsic::aarch64_neon_st1x2: {
7002 if (VT == MVT::v8i8) {
7003 SelectStore(Node, 2, AArch64::ST1Twov8b);
7004 return;
7005 } else if (VT == MVT::v16i8) {
7006 SelectStore(Node, 2, AArch64::ST1Twov16b);
7007 return;
7008 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7009 VT == MVT::v4bf16) {
7010 SelectStore(Node, 2, AArch64::ST1Twov4h);
7011 return;
7012 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7013 VT == MVT::v8bf16) {
7014 SelectStore(Node, 2, AArch64::ST1Twov8h);
7015 return;
7016 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7017 SelectStore(Node, 2, AArch64::ST1Twov2s);
7018 return;
7019 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7020 SelectStore(Node, 2, AArch64::ST1Twov4s);
7021 return;
7022 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7023 SelectStore(Node, 2, AArch64::ST1Twov2d);
7024 return;
7025 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7026 SelectStore(Node, 2, AArch64::ST1Twov1d);
7027 return;
7028 }
7029 break;
7030 }
7031 case Intrinsic::aarch64_neon_st1x3: {
7032 if (VT == MVT::v8i8) {
7033 SelectStore(Node, 3, AArch64::ST1Threev8b);
7034 return;
7035 } else if (VT == MVT::v16i8) {
7036 SelectStore(Node, 3, AArch64::ST1Threev16b);
7037 return;
7038 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7039 VT == MVT::v4bf16) {
7040 SelectStore(Node, 3, AArch64::ST1Threev4h);
7041 return;
7042 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7043 VT == MVT::v8bf16) {
7044 SelectStore(Node, 3, AArch64::ST1Threev8h);
7045 return;
7046 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7047 SelectStore(Node, 3, AArch64::ST1Threev2s);
7048 return;
7049 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7050 SelectStore(Node, 3, AArch64::ST1Threev4s);
7051 return;
7052 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7053 SelectStore(Node, 3, AArch64::ST1Threev2d);
7054 return;
7055 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7056 SelectStore(Node, 3, AArch64::ST1Threev1d);
7057 return;
7058 }
7059 break;
7060 }
7061 case Intrinsic::aarch64_neon_st1x4: {
7062 if (VT == MVT::v8i8) {
7063 SelectStore(Node, 4, AArch64::ST1Fourv8b);
7064 return;
7065 } else if (VT == MVT::v16i8) {
7066 SelectStore(Node, 4, AArch64::ST1Fourv16b);
7067 return;
7068 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7069 VT == MVT::v4bf16) {
7070 SelectStore(Node, 4, AArch64::ST1Fourv4h);
7071 return;
7072 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7073 VT == MVT::v8bf16) {
7074 SelectStore(Node, 4, AArch64::ST1Fourv8h);
7075 return;
7076 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7077 SelectStore(Node, 4, AArch64::ST1Fourv2s);
7078 return;
7079 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7080 SelectStore(Node, 4, AArch64::ST1Fourv4s);
7081 return;
7082 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7083 SelectStore(Node, 4, AArch64::ST1Fourv2d);
7084 return;
7085 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7086 SelectStore(Node, 4, AArch64::ST1Fourv1d);
7087 return;
7088 }
7089 break;
7090 }
7091 case Intrinsic::aarch64_neon_st2: {
7092 if (VT == MVT::v8i8) {
7093 SelectStore(Node, 2, AArch64::ST2Twov8b);
7094 return;
7095 } else if (VT == MVT::v16i8) {
7096 SelectStore(Node, 2, AArch64::ST2Twov16b);
7097 return;
7098 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7099 VT == MVT::v4bf16) {
7100 SelectStore(Node, 2, AArch64::ST2Twov4h);
7101 return;
7102 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7103 VT == MVT::v8bf16) {
7104 SelectStore(Node, 2, AArch64::ST2Twov8h);
7105 return;
7106 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7107 SelectStore(Node, 2, AArch64::ST2Twov2s);
7108 return;
7109 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7110 SelectStore(Node, 2, AArch64::ST2Twov4s);
7111 return;
7112 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7113 SelectStore(Node, 2, AArch64::ST2Twov2d);
7114 return;
7115 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7116 SelectStore(Node, 2, AArch64::ST1Twov1d);
7117 return;
7118 }
7119 break;
7120 }
7121 case Intrinsic::aarch64_neon_st3: {
7122 if (VT == MVT::v8i8) {
7123 SelectStore(Node, 3, AArch64::ST3Threev8b);
7124 return;
7125 } else if (VT == MVT::v16i8) {
7126 SelectStore(Node, 3, AArch64::ST3Threev16b);
7127 return;
7128 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7129 VT == MVT::v4bf16) {
7130 SelectStore(Node, 3, AArch64::ST3Threev4h);
7131 return;
7132 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7133 VT == MVT::v8bf16) {
7134 SelectStore(Node, 3, AArch64::ST3Threev8h);
7135 return;
7136 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7137 SelectStore(Node, 3, AArch64::ST3Threev2s);
7138 return;
7139 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7140 SelectStore(Node, 3, AArch64::ST3Threev4s);
7141 return;
7142 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7143 SelectStore(Node, 3, AArch64::ST3Threev2d);
7144 return;
7145 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7146 SelectStore(Node, 3, AArch64::ST1Threev1d);
7147 return;
7148 }
7149 break;
7150 }
7151 case Intrinsic::aarch64_neon_st4: {
7152 if (VT == MVT::v8i8) {
7153 SelectStore(Node, 4, AArch64::ST4Fourv8b);
7154 return;
7155 } else if (VT == MVT::v16i8) {
7156 SelectStore(Node, 4, AArch64::ST4Fourv16b);
7157 return;
7158 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7159 VT == MVT::v4bf16) {
7160 SelectStore(Node, 4, AArch64::ST4Fourv4h);
7161 return;
7162 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7163 VT == MVT::v8bf16) {
7164 SelectStore(Node, 4, AArch64::ST4Fourv8h);
7165 return;
7166 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7167 SelectStore(Node, 4, AArch64::ST4Fourv2s);
7168 return;
7169 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7170 SelectStore(Node, 4, AArch64::ST4Fourv4s);
7171 return;
7172 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7173 SelectStore(Node, 4, AArch64::ST4Fourv2d);
7174 return;
7175 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7176 SelectStore(Node, 4, AArch64::ST1Fourv1d);
7177 return;
7178 }
7179 break;
7180 }
7181 case Intrinsic::aarch64_neon_st2lane: {
7182 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7183 SelectStoreLane(Node, 2, AArch64::ST2i8);
7184 return;
7185 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7186 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7187 SelectStoreLane(Node, 2, AArch64::ST2i16);
7188 return;
7189 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7190 VT == MVT::v2f32) {
7191 SelectStoreLane(Node, 2, AArch64::ST2i32);
7192 return;
7193 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7194 VT == MVT::v1f64) {
7195 SelectStoreLane(Node, 2, AArch64::ST2i64);
7196 return;
7197 }
7198 break;
7199 }
7200 case Intrinsic::aarch64_neon_st3lane: {
7201 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7202 SelectStoreLane(Node, 3, AArch64::ST3i8);
7203 return;
7204 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7205 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7206 SelectStoreLane(Node, 3, AArch64::ST3i16);
7207 return;
7208 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7209 VT == MVT::v2f32) {
7210 SelectStoreLane(Node, 3, AArch64::ST3i32);
7211 return;
7212 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7213 VT == MVT::v1f64) {
7214 SelectStoreLane(Node, 3, AArch64::ST3i64);
7215 return;
7216 }
7217 break;
7218 }
7219 case Intrinsic::aarch64_neon_st4lane: {
7220 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7221 SelectStoreLane(Node, 4, AArch64::ST4i8);
7222 return;
7223 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7224 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7225 SelectStoreLane(Node, 4, AArch64::ST4i16);
7226 return;
7227 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7228 VT == MVT::v2f32) {
7229 SelectStoreLane(Node, 4, AArch64::ST4i32);
7230 return;
7231 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7232 VT == MVT::v1f64) {
7233 SelectStoreLane(Node, 4, AArch64::ST4i64);
7234 return;
7235 }
7236 break;
7237 }
7238 case Intrinsic::aarch64_sve_st2q: {
7239 SelectPredicatedStore(Node, 2, 4, AArch64::ST2Q, AArch64::ST2Q_IMM);
7240 return;
7241 }
7242 case Intrinsic::aarch64_sve_st3q: {
7243 SelectPredicatedStore(Node, 3, 4, AArch64::ST3Q, AArch64::ST3Q_IMM);
7244 return;
7245 }
7246 case Intrinsic::aarch64_sve_st4q: {
7247 SelectPredicatedStore(Node, 4, 4, AArch64::ST4Q, AArch64::ST4Q_IMM);
7248 return;
7249 }
7250 case Intrinsic::aarch64_sve_st2: {
7251 if (VT == MVT::nxv16i8) {
7252 SelectPredicatedStore(Node, 2, 0, AArch64::ST2B, AArch64::ST2B_IMM);
7253 return;
7254 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
7255 VT == MVT::nxv8bf16) {
7256 SelectPredicatedStore(Node, 2, 1, AArch64::ST2H, AArch64::ST2H_IMM);
7257 return;
7258 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
7259 SelectPredicatedStore(Node, 2, 2, AArch64::ST2W, AArch64::ST2W_IMM);
7260 return;
7261 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
7262 SelectPredicatedStore(Node, 2, 3, AArch64::ST2D, AArch64::ST2D_IMM);
7263 return;
7264 }
7265 break;
7266 }
7267 case Intrinsic::aarch64_sve_st3: {
7268 if (VT == MVT::nxv16i8) {
7269 SelectPredicatedStore(Node, 3, 0, AArch64::ST3B, AArch64::ST3B_IMM);
7270 return;
7271 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
7272 VT == MVT::nxv8bf16) {
7273 SelectPredicatedStore(Node, 3, 1, AArch64::ST3H, AArch64::ST3H_IMM);
7274 return;
7275 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
7276 SelectPredicatedStore(Node, 3, 2, AArch64::ST3W, AArch64::ST3W_IMM);
7277 return;
7278 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
7279 SelectPredicatedStore(Node, 3, 3, AArch64::ST3D, AArch64::ST3D_IMM);
7280 return;
7281 }
7282 break;
7283 }
7284 case Intrinsic::aarch64_sve_st4: {
7285 if (VT == MVT::nxv16i8) {
7286 SelectPredicatedStore(Node, 4, 0, AArch64::ST4B, AArch64::ST4B_IMM);
7287 return;
7288 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
7289 VT == MVT::nxv8bf16) {
7290 SelectPredicatedStore(Node, 4, 1, AArch64::ST4H, AArch64::ST4H_IMM);
7291 return;
7292 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
7293 SelectPredicatedStore(Node, 4, 2, AArch64::ST4W, AArch64::ST4W_IMM);
7294 return;
7295 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
7296 SelectPredicatedStore(Node, 4, 3, AArch64::ST4D, AArch64::ST4D_IMM);
7297 return;
7298 }
7299 break;
7300 }
7301 }
7302 break;
7303 }
7304 case AArch64ISD::LD2post: {
7305 if (VT == MVT::v8i8) {
7306 SelectPostLoad(Node, 2, AArch64::LD2Twov8b_POST, AArch64::dsub0);
7307 return;
7308 } else if (VT == MVT::v16i8) {
7309 SelectPostLoad(Node, 2, AArch64::LD2Twov16b_POST, AArch64::qsub0);
7310 return;
7311 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7312 SelectPostLoad(Node, 2, AArch64::LD2Twov4h_POST, AArch64::dsub0);
7313 return;
7314 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7315 SelectPostLoad(Node, 2, AArch64::LD2Twov8h_POST, AArch64::qsub0);
7316 return;
7317 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7318 SelectPostLoad(Node, 2, AArch64::LD2Twov2s_POST, AArch64::dsub0);
7319 return;
7320 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7321 SelectPostLoad(Node, 2, AArch64::LD2Twov4s_POST, AArch64::qsub0);
7322 return;
7323 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7324 SelectPostLoad(Node, 2, AArch64::LD1Twov1d_POST, AArch64::dsub0);
7325 return;
7326 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7327 SelectPostLoad(Node, 2, AArch64::LD2Twov2d_POST, AArch64::qsub0);
7328 return;
7329 }
7330 break;
7331 }
7332 case AArch64ISD::LD3post: {
7333 if (VT == MVT::v8i8) {
7334 SelectPostLoad(Node, 3, AArch64::LD3Threev8b_POST, AArch64::dsub0);
7335 return;
7336 } else if (VT == MVT::v16i8) {
7337 SelectPostLoad(Node, 3, AArch64::LD3Threev16b_POST, AArch64::qsub0);
7338 return;
7339 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7340 SelectPostLoad(Node, 3, AArch64::LD3Threev4h_POST, AArch64::dsub0);
7341 return;
7342 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7343 SelectPostLoad(Node, 3, AArch64::LD3Threev8h_POST, AArch64::qsub0);
7344 return;
7345 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7346 SelectPostLoad(Node, 3, AArch64::LD3Threev2s_POST, AArch64::dsub0);
7347 return;
7348 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7349 SelectPostLoad(Node, 3, AArch64::LD3Threev4s_POST, AArch64::qsub0);
7350 return;
7351 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7352 SelectPostLoad(Node, 3, AArch64::LD1Threev1d_POST, AArch64::dsub0);
7353 return;
7354 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7355 SelectPostLoad(Node, 3, AArch64::LD3Threev2d_POST, AArch64::qsub0);
7356 return;
7357 }
7358 break;
7359 }
7360 case AArch64ISD::LD4post: {
7361 if (VT == MVT::v8i8) {
7362 SelectPostLoad(Node, 4, AArch64::LD4Fourv8b_POST, AArch64::dsub0);
7363 return;
7364 } else if (VT == MVT::v16i8) {
7365 SelectPostLoad(Node, 4, AArch64::LD4Fourv16b_POST, AArch64::qsub0);
7366 return;
7367 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7368 SelectPostLoad(Node, 4, AArch64::LD4Fourv4h_POST, AArch64::dsub0);
7369 return;
7370 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7371 SelectPostLoad(Node, 4, AArch64::LD4Fourv8h_POST, AArch64::qsub0);
7372 return;
7373 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7374 SelectPostLoad(Node, 4, AArch64::LD4Fourv2s_POST, AArch64::dsub0);
7375 return;
7376 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7377 SelectPostLoad(Node, 4, AArch64::LD4Fourv4s_POST, AArch64::qsub0);
7378 return;
7379 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7380 SelectPostLoad(Node, 4, AArch64::LD1Fourv1d_POST, AArch64::dsub0);
7381 return;
7382 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7383 SelectPostLoad(Node, 4, AArch64::LD4Fourv2d_POST, AArch64::qsub0);
7384 return;
7385 }
7386 break;
7387 }
7388 case AArch64ISD::LD1x2post: {
7389 if (VT == MVT::v8i8) {
7390 SelectPostLoad(Node, 2, AArch64::LD1Twov8b_POST, AArch64::dsub0);
7391 return;
7392 } else if (VT == MVT::v16i8) {
7393 SelectPostLoad(Node, 2, AArch64::LD1Twov16b_POST, AArch64::qsub0);
7394 return;
7395 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7396 SelectPostLoad(Node, 2, AArch64::LD1Twov4h_POST, AArch64::dsub0);
7397 return;
7398 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7399 SelectPostLoad(Node, 2, AArch64::LD1Twov8h_POST, AArch64::qsub0);
7400 return;
7401 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7402 SelectPostLoad(Node, 2, AArch64::LD1Twov2s_POST, AArch64::dsub0);
7403 return;
7404 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7405 SelectPostLoad(Node, 2, AArch64::LD1Twov4s_POST, AArch64::qsub0);
7406 return;
7407 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7408 SelectPostLoad(Node, 2, AArch64::LD1Twov1d_POST, AArch64::dsub0);
7409 return;
7410 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7411 SelectPostLoad(Node, 2, AArch64::LD1Twov2d_POST, AArch64::qsub0);
7412 return;
7413 }
7414 break;
7415 }
7416 case AArch64ISD::LD1x3post: {
7417 if (VT == MVT::v8i8) {
7418 SelectPostLoad(Node, 3, AArch64::LD1Threev8b_POST, AArch64::dsub0);
7419 return;
7420 } else if (VT == MVT::v16i8) {
7421 SelectPostLoad(Node, 3, AArch64::LD1Threev16b_POST, AArch64::qsub0);
7422 return;
7423 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7424 SelectPostLoad(Node, 3, AArch64::LD1Threev4h_POST, AArch64::dsub0);
7425 return;
7426 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7427 SelectPostLoad(Node, 3, AArch64::LD1Threev8h_POST, AArch64::qsub0);
7428 return;
7429 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7430 SelectPostLoad(Node, 3, AArch64::LD1Threev2s_POST, AArch64::dsub0);
7431 return;
7432 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7433 SelectPostLoad(Node, 3, AArch64::LD1Threev4s_POST, AArch64::qsub0);
7434 return;
7435 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7436 SelectPostLoad(Node, 3, AArch64::LD1Threev1d_POST, AArch64::dsub0);
7437 return;
7438 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7439 SelectPostLoad(Node, 3, AArch64::LD1Threev2d_POST, AArch64::qsub0);
7440 return;
7441 }
7442 break;
7443 }
7444 case AArch64ISD::LD1x4post: {
7445 if (VT == MVT::v8i8) {
7446 SelectPostLoad(Node, 4, AArch64::LD1Fourv8b_POST, AArch64::dsub0);
7447 return;
7448 } else if (VT == MVT::v16i8) {
7449 SelectPostLoad(Node, 4, AArch64::LD1Fourv16b_POST, AArch64::qsub0);
7450 return;
7451 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7452 SelectPostLoad(Node, 4, AArch64::LD1Fourv4h_POST, AArch64::dsub0);
7453 return;
7454 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7455 SelectPostLoad(Node, 4, AArch64::LD1Fourv8h_POST, AArch64::qsub0);
7456 return;
7457 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7458 SelectPostLoad(Node, 4, AArch64::LD1Fourv2s_POST, AArch64::dsub0);
7459 return;
7460 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7461 SelectPostLoad(Node, 4, AArch64::LD1Fourv4s_POST, AArch64::qsub0);
7462 return;
7463 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7464 SelectPostLoad(Node, 4, AArch64::LD1Fourv1d_POST, AArch64::dsub0);
7465 return;
7466 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7467 SelectPostLoad(Node, 4, AArch64::LD1Fourv2d_POST, AArch64::qsub0);
7468 return;
7469 }
7470 break;
7471 }
7472 case AArch64ISD::LD1DUPpost: {
7473 if (VT == MVT::v8i8) {
7474 SelectPostLoad(Node, 1, AArch64::LD1Rv8b_POST, AArch64::dsub0);
7475 return;
7476 } else if (VT == MVT::v16i8) {
7477 SelectPostLoad(Node, 1, AArch64::LD1Rv16b_POST, AArch64::qsub0);
7478 return;
7479 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7480 SelectPostLoad(Node, 1, AArch64::LD1Rv4h_POST, AArch64::dsub0);
7481 return;
7482 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7483 SelectPostLoad(Node, 1, AArch64::LD1Rv8h_POST, AArch64::qsub0);
7484 return;
7485 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7486 SelectPostLoad(Node, 1, AArch64::LD1Rv2s_POST, AArch64::dsub0);
7487 return;
7488 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7489 SelectPostLoad(Node, 1, AArch64::LD1Rv4s_POST, AArch64::qsub0);
7490 return;
7491 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7492 SelectPostLoad(Node, 1, AArch64::LD1Rv1d_POST, AArch64::dsub0);
7493 return;
7494 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7495 SelectPostLoad(Node, 1, AArch64::LD1Rv2d_POST, AArch64::qsub0);
7496 return;
7497 }
7498 break;
7499 }
7500 case AArch64ISD::LD2DUPpost: {
7501 if (VT == MVT::v8i8) {
7502 SelectPostLoad(Node, 2, AArch64::LD2Rv8b_POST, AArch64::dsub0);
7503 return;
7504 } else if (VT == MVT::v16i8) {
7505 SelectPostLoad(Node, 2, AArch64::LD2Rv16b_POST, AArch64::qsub0);
7506 return;
7507 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7508 SelectPostLoad(Node, 2, AArch64::LD2Rv4h_POST, AArch64::dsub0);
7509 return;
7510 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7511 SelectPostLoad(Node, 2, AArch64::LD2Rv8h_POST, AArch64::qsub0);
7512 return;
7513 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7514 SelectPostLoad(Node, 2, AArch64::LD2Rv2s_POST, AArch64::dsub0);
7515 return;
7516 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7517 SelectPostLoad(Node, 2, AArch64::LD2Rv4s_POST, AArch64::qsub0);
7518 return;
7519 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7520 SelectPostLoad(Node, 2, AArch64::LD2Rv1d_POST, AArch64::dsub0);
7521 return;
7522 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7523 SelectPostLoad(Node, 2, AArch64::LD2Rv2d_POST, AArch64::qsub0);
7524 return;
7525 }
7526 break;
7527 }
7528 case AArch64ISD::LD3DUPpost: {
7529 if (VT == MVT::v8i8) {
7530 SelectPostLoad(Node, 3, AArch64::LD3Rv8b_POST, AArch64::dsub0);
7531 return;
7532 } else if (VT == MVT::v16i8) {
7533 SelectPostLoad(Node, 3, AArch64::LD3Rv16b_POST, AArch64::qsub0);
7534 return;
7535 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7536 SelectPostLoad(Node, 3, AArch64::LD3Rv4h_POST, AArch64::dsub0);
7537 return;
7538 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7539 SelectPostLoad(Node, 3, AArch64::LD3Rv8h_POST, AArch64::qsub0);
7540 return;
7541 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7542 SelectPostLoad(Node, 3, AArch64::LD3Rv2s_POST, AArch64::dsub0);
7543 return;
7544 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7545 SelectPostLoad(Node, 3, AArch64::LD3Rv4s_POST, AArch64::qsub0);
7546 return;
7547 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7548 SelectPostLoad(Node, 3, AArch64::LD3Rv1d_POST, AArch64::dsub0);
7549 return;
7550 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7551 SelectPostLoad(Node, 3, AArch64::LD3Rv2d_POST, AArch64::qsub0);
7552 return;
7553 }
7554 break;
7555 }
7556 case AArch64ISD::LD4DUPpost: {
7557 if (VT == MVT::v8i8) {
7558 SelectPostLoad(Node, 4, AArch64::LD4Rv8b_POST, AArch64::dsub0);
7559 return;
7560 } else if (VT == MVT::v16i8) {
7561 SelectPostLoad(Node, 4, AArch64::LD4Rv16b_POST, AArch64::qsub0);
7562 return;
7563 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7564 SelectPostLoad(Node, 4, AArch64::LD4Rv4h_POST, AArch64::dsub0);
7565 return;
7566 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7567 SelectPostLoad(Node, 4, AArch64::LD4Rv8h_POST, AArch64::qsub0);
7568 return;
7569 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7570 SelectPostLoad(Node, 4, AArch64::LD4Rv2s_POST, AArch64::dsub0);
7571 return;
7572 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7573 SelectPostLoad(Node, 4, AArch64::LD4Rv4s_POST, AArch64::qsub0);
7574 return;
7575 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7576 SelectPostLoad(Node, 4, AArch64::LD4Rv1d_POST, AArch64::dsub0);
7577 return;
7578 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7579 SelectPostLoad(Node, 4, AArch64::LD4Rv2d_POST, AArch64::qsub0);
7580 return;
7581 }
7582 break;
7583 }
7584 case AArch64ISD::LD1LANEpost: {
7585 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7586 SelectPostLoadLane(Node, 1, AArch64::LD1i8_POST);
7587 return;
7588 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7589 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7590 SelectPostLoadLane(Node, 1, AArch64::LD1i16_POST);
7591 return;
7592 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7593 VT == MVT::v2f32) {
7594 SelectPostLoadLane(Node, 1, AArch64::LD1i32_POST);
7595 return;
7596 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7597 VT == MVT::v1f64) {
7598 SelectPostLoadLane(Node, 1, AArch64::LD1i64_POST);
7599 return;
7600 }
7601 break;
7602 }
7603 case AArch64ISD::LD2LANEpost: {
7604 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7605 SelectPostLoadLane(Node, 2, AArch64::LD2i8_POST);
7606 return;
7607 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7608 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7609 SelectPostLoadLane(Node, 2, AArch64::LD2i16_POST);
7610 return;
7611 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7612 VT == MVT::v2f32) {
7613 SelectPostLoadLane(Node, 2, AArch64::LD2i32_POST);
7614 return;
7615 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7616 VT == MVT::v1f64) {
7617 SelectPostLoadLane(Node, 2, AArch64::LD2i64_POST);
7618 return;
7619 }
7620 break;
7621 }
7622 case AArch64ISD::LD3LANEpost: {
7623 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7624 SelectPostLoadLane(Node, 3, AArch64::LD3i8_POST);
7625 return;
7626 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7627 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7628 SelectPostLoadLane(Node, 3, AArch64::LD3i16_POST);
7629 return;
7630 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7631 VT == MVT::v2f32) {
7632 SelectPostLoadLane(Node, 3, AArch64::LD3i32_POST);
7633 return;
7634 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7635 VT == MVT::v1f64) {
7636 SelectPostLoadLane(Node, 3, AArch64::LD3i64_POST);
7637 return;
7638 }
7639 break;
7640 }
7641 case AArch64ISD::LD4LANEpost: {
7642 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7643 SelectPostLoadLane(Node, 4, AArch64::LD4i8_POST);
7644 return;
7645 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7646 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7647 SelectPostLoadLane(Node, 4, AArch64::LD4i16_POST);
7648 return;
7649 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7650 VT == MVT::v2f32) {
7651 SelectPostLoadLane(Node, 4, AArch64::LD4i32_POST);
7652 return;
7653 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7654 VT == MVT::v1f64) {
7655 SelectPostLoadLane(Node, 4, AArch64::LD4i64_POST);
7656 return;
7657 }
7658 break;
7659 }
7660 case AArch64ISD::ST2post: {
7661 VT = Node->getOperand(1).getValueType();
7662 if (VT == MVT::v8i8) {
7663 SelectPostStore(Node, 2, AArch64::ST2Twov8b_POST);
7664 return;
7665 } else if (VT == MVT::v16i8) {
7666 SelectPostStore(Node, 2, AArch64::ST2Twov16b_POST);
7667 return;
7668 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7669 SelectPostStore(Node, 2, AArch64::ST2Twov4h_POST);
7670 return;
7671 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7672 SelectPostStore(Node, 2, AArch64::ST2Twov8h_POST);
7673 return;
7674 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7675 SelectPostStore(Node, 2, AArch64::ST2Twov2s_POST);
7676 return;
7677 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7678 SelectPostStore(Node, 2, AArch64::ST2Twov4s_POST);
7679 return;
7680 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7681 SelectPostStore(Node, 2, AArch64::ST2Twov2d_POST);
7682 return;
7683 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7684 SelectPostStore(Node, 2, AArch64::ST1Twov1d_POST);
7685 return;
7686 }
7687 break;
7688 }
7689 case AArch64ISD::ST3post: {
7690 VT = Node->getOperand(1).getValueType();
7691 if (VT == MVT::v8i8) {
7692 SelectPostStore(Node, 3, AArch64::ST3Threev8b_POST);
7693 return;
7694 } else if (VT == MVT::v16i8) {
7695 SelectPostStore(Node, 3, AArch64::ST3Threev16b_POST);
7696 return;
7697 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7698 SelectPostStore(Node, 3, AArch64::ST3Threev4h_POST);
7699 return;
7700 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7701 SelectPostStore(Node, 3, AArch64::ST3Threev8h_POST);
7702 return;
7703 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7704 SelectPostStore(Node, 3, AArch64::ST3Threev2s_POST);
7705 return;
7706 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7707 SelectPostStore(Node, 3, AArch64::ST3Threev4s_POST);
7708 return;
7709 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7710 SelectPostStore(Node, 3, AArch64::ST3Threev2d_POST);
7711 return;
7712 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7713 SelectPostStore(Node, 3, AArch64::ST1Threev1d_POST);
7714 return;
7715 }
7716 break;
7717 }
7718 case AArch64ISD::ST4post: {
7719 VT = Node->getOperand(1).getValueType();
7720 if (VT == MVT::v8i8) {
7721 SelectPostStore(Node, 4, AArch64::ST4Fourv8b_POST);
7722 return;
7723 } else if (VT == MVT::v16i8) {
7724 SelectPostStore(Node, 4, AArch64::ST4Fourv16b_POST);
7725 return;
7726 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7727 SelectPostStore(Node, 4, AArch64::ST4Fourv4h_POST);
7728 return;
7729 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7730 SelectPostStore(Node, 4, AArch64::ST4Fourv8h_POST);
7731 return;
7732 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7733 SelectPostStore(Node, 4, AArch64::ST4Fourv2s_POST);
7734 return;
7735 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7736 SelectPostStore(Node, 4, AArch64::ST4Fourv4s_POST);
7737 return;
7738 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7739 SelectPostStore(Node, 4, AArch64::ST4Fourv2d_POST);
7740 return;
7741 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7742 SelectPostStore(Node, 4, AArch64::ST1Fourv1d_POST);
7743 return;
7744 }
7745 break;
7746 }
7747 case AArch64ISD::ST1x2post: {
7748 VT = Node->getOperand(1).getValueType();
7749 if (VT == MVT::v8i8) {
7750 SelectPostStore(Node, 2, AArch64::ST1Twov8b_POST);
7751 return;
7752 } else if (VT == MVT::v16i8) {
7753 SelectPostStore(Node, 2, AArch64::ST1Twov16b_POST);
7754 return;
7755 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7756 SelectPostStore(Node, 2, AArch64::ST1Twov4h_POST);
7757 return;
7758 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7759 SelectPostStore(Node, 2, AArch64::ST1Twov8h_POST);
7760 return;
7761 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7762 SelectPostStore(Node, 2, AArch64::ST1Twov2s_POST);
7763 return;
7764 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7765 SelectPostStore(Node, 2, AArch64::ST1Twov4s_POST);
7766 return;
7767 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7768 SelectPostStore(Node, 2, AArch64::ST1Twov1d_POST);
7769 return;
7770 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7771 SelectPostStore(Node, 2, AArch64::ST1Twov2d_POST);
7772 return;
7773 }
7774 break;
7775 }
7776 case AArch64ISD::ST1x3post: {
7777 VT = Node->getOperand(1).getValueType();
7778 if (VT == MVT::v8i8) {
7779 SelectPostStore(Node, 3, AArch64::ST1Threev8b_POST);
7780 return;
7781 } else if (VT == MVT::v16i8) {
7782 SelectPostStore(Node, 3, AArch64::ST1Threev16b_POST);
7783 return;
7784 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7785 SelectPostStore(Node, 3, AArch64::ST1Threev4h_POST);
7786 return;
7787 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16 ) {
7788 SelectPostStore(Node, 3, AArch64::ST1Threev8h_POST);
7789 return;
7790 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7791 SelectPostStore(Node, 3, AArch64::ST1Threev2s_POST);
7792 return;
7793 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7794 SelectPostStore(Node, 3, AArch64::ST1Threev4s_POST);
7795 return;
7796 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7797 SelectPostStore(Node, 3, AArch64::ST1Threev1d_POST);
7798 return;
7799 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7800 SelectPostStore(Node, 3, AArch64::ST1Threev2d_POST);
7801 return;
7802 }
7803 break;
7804 }
7805 case AArch64ISD::ST1x4post: {
7806 VT = Node->getOperand(1).getValueType();
7807 if (VT == MVT::v8i8) {
7808 SelectPostStore(Node, 4, AArch64::ST1Fourv8b_POST);
7809 return;
7810 } else if (VT == MVT::v16i8) {
7811 SelectPostStore(Node, 4, AArch64::ST1Fourv16b_POST);
7812 return;
7813 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7814 SelectPostStore(Node, 4, AArch64::ST1Fourv4h_POST);
7815 return;
7816 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7817 SelectPostStore(Node, 4, AArch64::ST1Fourv8h_POST);
7818 return;
7819 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7820 SelectPostStore(Node, 4, AArch64::ST1Fourv2s_POST);
7821 return;
7822 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7823 SelectPostStore(Node, 4, AArch64::ST1Fourv4s_POST);
7824 return;
7825 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7826 SelectPostStore(Node, 4, AArch64::ST1Fourv1d_POST);
7827 return;
7828 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7829 SelectPostStore(Node, 4, AArch64::ST1Fourv2d_POST);
7830 return;
7831 }
7832 break;
7833 }
7834 case AArch64ISD::ST2LANEpost: {
7835 VT = Node->getOperand(1).getValueType();
7836 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7837 SelectPostStoreLane(Node, 2, AArch64::ST2i8_POST);
7838 return;
7839 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7840 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7841 SelectPostStoreLane(Node, 2, AArch64::ST2i16_POST);
7842 return;
7843 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7844 VT == MVT::v2f32) {
7845 SelectPostStoreLane(Node, 2, AArch64::ST2i32_POST);
7846 return;
7847 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7848 VT == MVT::v1f64) {
7849 SelectPostStoreLane(Node, 2, AArch64::ST2i64_POST);
7850 return;
7851 }
7852 break;
7853 }
7854 case AArch64ISD::ST3LANEpost: {
7855 VT = Node->getOperand(1).getValueType();
7856 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7857 SelectPostStoreLane(Node, 3, AArch64::ST3i8_POST);
7858 return;
7859 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7860 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7861 SelectPostStoreLane(Node, 3, AArch64::ST3i16_POST);
7862 return;
7863 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7864 VT == MVT::v2f32) {
7865 SelectPostStoreLane(Node, 3, AArch64::ST3i32_POST);
7866 return;
7867 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7868 VT == MVT::v1f64) {
7869 SelectPostStoreLane(Node, 3, AArch64::ST3i64_POST);
7870 return;
7871 }
7872 break;
7873 }
7874 case AArch64ISD::ST4LANEpost: {
7875 VT = Node->getOperand(1).getValueType();
7876 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7877 SelectPostStoreLane(Node, 4, AArch64::ST4i8_POST);
7878 return;
7879 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7880 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7881 SelectPostStoreLane(Node, 4, AArch64::ST4i16_POST);
7882 return;
7883 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7884 VT == MVT::v2f32) {
7885 SelectPostStoreLane(Node, 4, AArch64::ST4i32_POST);
7886 return;
7887 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7888 VT == MVT::v1f64) {
7889 SelectPostStoreLane(Node, 4, AArch64::ST4i64_POST);
7890 return;
7891 }
7892 break;
7893 }
7894 }
7895
7896 // Select the default instruction
7897 SelectCode(Node);
7898}
7899
7900/// createAArch64ISelDag - This pass converts a legalized DAG into a
7901/// AArch64-specific DAG, ready for instruction scheduling.
7903 CodeGenOptLevel OptLevel) {
7904 return new AArch64DAGToDAGISelLegacy(TM, OptLevel);
7905}
7906
7907/// When \p PredVT is a scalable vector predicate in the form
7908/// MVT::nx<M>xi1, it builds the correspondent scalable vector of
7909/// integers MVT::nx<M>xi<bits> s.t. M x bits = 128. When targeting
7910/// structured vectors (NumVec >1), the output data type is
7911/// MVT::nx<M*NumVec>xi<bits> s.t. M x bits = 128. If the input
7912/// PredVT is not in the form MVT::nx<M>xi1, it returns an invalid
7913/// EVT.
7915 unsigned NumVec) {
7916 assert(NumVec > 0 && NumVec < 5 && "Invalid number of vectors.");
7917 if (!PredVT.isScalableVectorOf(MVT::i1))
7918 return EVT();
7919
7920 if (PredVT != MVT::nxv16i1 && PredVT != MVT::nxv8i1 &&
7921 PredVT != MVT::nxv4i1 && PredVT != MVT::nxv2i1)
7922 return EVT();
7923
7924 ElementCount EC = PredVT.getVectorElementCount();
7925 EVT ScalarVT =
7926 EVT::getIntegerVT(Ctx, AArch64::SVEBitsPerBlock / EC.getKnownMinValue());
7927 EVT MemVT = EVT::getVectorVT(Ctx, ScalarVT, EC * NumVec);
7928
7929 return MemVT;
7930}
7931
7932/// Builds an integer vector type large enough to hold \p NumVec instances
7933/// of \p VecVT.
7934static EVT getMultipleVectorType(LLVMContext &Ctx, EVT VecVT, unsigned NumVec) {
7936 VecVT.getVectorElementCount() * NumVec);
7937}
7938
7939/// Return the EVT of the data associated to a memory operation in \p
7940/// Root. If such EVT cannot be retrieved, it returns an invalid EVT.
7942 if (auto *MemIntr = dyn_cast<MemIntrinsicSDNode>(Root))
7943 return MemIntr->getMemoryVT();
7944
7945 if (isa<MemSDNode>(Root)) {
7946 EVT MemVT = cast<MemSDNode>(Root)->getMemoryVT();
7947
7948 EVT DataVT;
7949 if (auto *Load = dyn_cast<LoadSDNode>(Root))
7950 DataVT = Load->getValueType(0);
7951 else if (auto *Load = dyn_cast<MaskedLoadSDNode>(Root))
7952 DataVT = Load->getValueType(0);
7953 else if (auto *Store = dyn_cast<StoreSDNode>(Root))
7954 DataVT = Store->getValue().getValueType();
7955 else if (auto *Store = dyn_cast<MaskedStoreSDNode>(Root))
7956 DataVT = Store->getValue().getValueType();
7957 else
7958 llvm_unreachable("Unexpected MemSDNode!");
7959
7960 return DataVT.changeVectorElementType(Ctx, MemVT.getVectorElementType());
7961 }
7962
7963 const unsigned Opcode = Root->getOpcode();
7964 // For custom ISD nodes, we have to look at them individually to extract the
7965 // type of the data moved to/from memory.
7966 switch (Opcode) {
7967 case AArch64ISD::LD1_MERGE_ZERO:
7968 case AArch64ISD::LD1S_MERGE_ZERO:
7969 case AArch64ISD::LDNF1_MERGE_ZERO:
7970 case AArch64ISD::LDNF1S_MERGE_ZERO:
7971 return cast<VTSDNode>(Root->getOperand(3))->getVT();
7972 case AArch64ISD::ST1_PRED:
7973 return cast<VTSDNode>(Root->getOperand(4))->getVT();
7974 default:
7975 break;
7976 }
7977
7978 if (Opcode != ISD::INTRINSIC_VOID && Opcode != ISD::INTRINSIC_W_CHAIN)
7979 return EVT();
7980
7981 switch (Root->getConstantOperandVal(1)) {
7982 default:
7983 return EVT();
7984 case Intrinsic::aarch64_sme_ldr:
7985 case Intrinsic::aarch64_sme_str:
7986 return MVT::nxv16i8;
7987 case Intrinsic::aarch64_sve_prf:
7988 // We are using an SVE prefetch intrinsic. Type must be inferred from the
7989 // width of the predicate.
7991 Ctx, Root->getOperand(2)->getValueType(0), /*NumVec=*/1);
7992 case Intrinsic::aarch64_sve_ld2_sret:
7993 case Intrinsic::aarch64_sve_ld2q_sret:
7995 Ctx, Root->getOperand(2)->getValueType(0), /*NumVec=*/2);
7996 case Intrinsic::aarch64_sve_st2q:
7998 Ctx, Root->getOperand(4)->getValueType(0), /*NumVec=*/2);
7999 case Intrinsic::aarch64_sve_ld3_sret:
8000 case Intrinsic::aarch64_sve_ld3q_sret:
8002 Ctx, Root->getOperand(2)->getValueType(0), /*NumVec=*/3);
8003 case Intrinsic::aarch64_sve_st3q:
8005 Ctx, Root->getOperand(5)->getValueType(0), /*NumVec=*/3);
8006 case Intrinsic::aarch64_sve_ld4_sret:
8007 case Intrinsic::aarch64_sve_ld4q_sret:
8009 Ctx, Root->getOperand(2)->getValueType(0), /*NumVec=*/4);
8010 case Intrinsic::aarch64_sve_st4q:
8012 Ctx, Root->getOperand(6)->getValueType(0), /*NumVec=*/4);
8013 case Intrinsic::aarch64_sve_ld1_pn_x2:
8014 case Intrinsic::aarch64_sve_ldnt1_pn_x2:
8015 return getMultipleVectorType(Ctx, Root->getValueType(0),
8016 /*NumVec=*/2);
8017 case Intrinsic::aarch64_sve_ld1_pn_x4:
8018 case Intrinsic::aarch64_sve_ldnt1_pn_x4:
8019 return getMultipleVectorType(Ctx, Root->getValueType(0),
8020 /*NumVec=*/4);
8021 case Intrinsic::aarch64_sve_st1_pn_x2:
8022 case Intrinsic::aarch64_sve_stnt1_pn_x2:
8023 return getMultipleVectorType(Ctx, Root->getOperand(2).getValueType(),
8024 /*NumVec=*/2);
8025 case Intrinsic::aarch64_sve_st1_pn_x4:
8026 case Intrinsic::aarch64_sve_stnt1_pn_x4:
8027 return getMultipleVectorType(Ctx, Root->getOperand(2).getValueType(),
8028 /*NumVec=*/4);
8029 case Intrinsic::aarch64_sve_ld1udq:
8030 case Intrinsic::aarch64_sve_st1dq:
8031 return EVT(MVT::nxv1i64);
8032 case Intrinsic::aarch64_sve_ld1uwq:
8033 case Intrinsic::aarch64_sve_st1wq:
8034 return EVT(MVT::nxv1i32);
8035 }
8036}
8037
8038/// SelectAddrModeIndexedSVE - Attempt selection of the addressing mode:
8039/// Base + OffImm * sizeof(MemVT) for Min >= OffImm <= Max
8040/// where Root is the memory access using N for its address.
8041template <int64_t Min, int64_t Max>
8042bool AArch64DAGToDAGISel::SelectAddrModeIndexedSVE(SDNode *Root, SDValue N,
8043 SDValue &Base,
8044 SDValue &OffImm) {
8045 const EVT MemVT = getMemVTFromNode(*(CurDAG->getContext()), Root);
8046 const DataLayout &DL = CurDAG->getDataLayout();
8047 const MachineFrameInfo &MFI = MF->getFrameInfo();
8048
8049 if (N.getOpcode() == ISD::FrameIndex) {
8050 int FI = cast<FrameIndexSDNode>(N)->getIndex();
8051 // We can only encode VL scaled offsets, so only fold in frame indexes
8052 // referencing SVE objects.
8053 if (MFI.hasScalableStackID(FI)) {
8054 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
8055 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i64);
8056 return true;
8057 }
8058
8059 return false;
8060 }
8061
8062 if (MemVT == EVT())
8063 return false;
8064
8065 if (N.getOpcode() != ISD::ADD)
8066 return false;
8067
8068 SDValue VScale = N.getOperand(1);
8069 int64_t MulImm = std::numeric_limits<int64_t>::max();
8070 if (VScale.getOpcode() == ISD::VSCALE) {
8071 MulImm = cast<ConstantSDNode>(VScale.getOperand(0))->getSExtValue();
8072 } else if (auto C = dyn_cast<ConstantSDNode>(VScale)) {
8073 int64_t ByteOffset = C->getSExtValue();
8074 const auto KnownVScale =
8076
8077 if (!KnownVScale || ByteOffset % KnownVScale != 0)
8078 return false;
8079
8080 MulImm = ByteOffset / KnownVScale;
8081 } else
8082 return false;
8083
8084 TypeSize TS = MemVT.getSizeInBits();
8085 int64_t MemWidthBytes = static_cast<int64_t>(TS.getKnownMinValue()) / 8;
8086
8087 if ((MulImm % MemWidthBytes) != 0)
8088 return false;
8089
8090 int64_t Offset = MulImm / MemWidthBytes;
8092 return false;
8093
8094 Base = N.getOperand(0);
8095 if (Base.getOpcode() == ISD::FrameIndex) {
8096 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
8097 // We can only encode VL scaled offsets, so only fold in frame indexes
8098 // referencing SVE objects.
8099 if (MFI.hasScalableStackID(FI))
8100 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
8101 }
8102
8103 OffImm = CurDAG->getTargetConstant(Offset, SDLoc(N), MVT::i64);
8104 return true;
8105}
8106
8107/// Select register plus register addressing mode for SVE, with scaled
8108/// offset.
8109bool AArch64DAGToDAGISel::SelectSVERegRegAddrMode(SDValue N, unsigned Scale,
8110 SDValue &Base,
8111 SDValue &Offset) {
8112 if (N.getOpcode() != ISD::ADD)
8113 return false;
8114
8115 // Process an ADD node.
8116 const SDValue LHS = N.getOperand(0);
8117 const SDValue RHS = N.getOperand(1);
8118
8119 // 8 bit data does not come with the SHL node, so it is treated
8120 // separately.
8121 if (Scale == 0) {
8122 Base = LHS;
8123 Offset = RHS;
8124 return true;
8125 }
8126
8127 if (auto C = dyn_cast<ConstantSDNode>(RHS)) {
8128 int64_t ImmOff = C->getSExtValue();
8129 unsigned Size = 1 << Scale;
8130
8131 // To use the reg+reg addressing mode, the immediate must be a multiple of
8132 // the vector element's byte size.
8133 if (ImmOff % Size)
8134 return false;
8135
8136 SDLoc DL(N);
8137 Base = LHS;
8138 Offset = CurDAG->getTargetConstant(ImmOff >> Scale, DL, MVT::i64);
8139 SDValue Ops[] = {Offset};
8140 SDNode *MI = CurDAG->getMachineNode(AArch64::MOVi64imm, DL, MVT::i64, Ops);
8141 Offset = SDValue(MI, 0);
8142 return true;
8143 }
8144
8145 // Check if the RHS is a shift node with a constant.
8146 if (RHS.getOpcode() != ISD::SHL)
8147 return false;
8148
8149 const SDValue ShiftRHS = RHS.getOperand(1);
8150 if (auto *C = dyn_cast<ConstantSDNode>(ShiftRHS))
8151 if (C->getZExtValue() == Scale) {
8152 Base = LHS;
8153 Offset = RHS.getOperand(0);
8154 return true;
8155 }
8156
8157 return false;
8158}
8159
8160bool AArch64DAGToDAGISel::SelectAllActivePredicate(SDValue N) {
8161 const AArch64TargetLowering *TLI =
8162 static_cast<const AArch64TargetLowering *>(getTargetLowering());
8163
8164 return TLI->isAllActivePredicate(*CurDAG, N);
8165}
8166
8167bool AArch64DAGToDAGISel::SelectAnyPredicate(SDValue N) {
8168 return N.getValueType().isScalableVectorOf(MVT::i1);
8169}
8170
8171bool AArch64DAGToDAGISel::SelectSMETileSlice(SDValue N, unsigned MaxSize,
8172 SDValue &Base, SDValue &Offset,
8173 unsigned Scale) {
8174 auto MatchConstantOffset = [&](SDValue CN) -> SDValue {
8175 if (auto *C = dyn_cast<ConstantSDNode>(CN)) {
8176 int64_t ImmOff = C->getSExtValue();
8177 if ((ImmOff > 0 && ImmOff <= MaxSize && (ImmOff % Scale == 0)))
8178 return CurDAG->getTargetConstant(ImmOff / Scale, SDLoc(N), MVT::i64);
8179 }
8180 return SDValue();
8181 };
8182
8183 if (SDValue C = MatchConstantOffset(N)) {
8184 Base = getZeroRegister(*CurDAG, SDLoc(N), MVT::i32);
8185 Offset = C;
8186 return true;
8187 }
8188
8189 // Try to untangle an ADD node into a 'reg + offset'
8190 if (CurDAG->isBaseWithConstantOffset(N)) {
8191 if (SDValue C = MatchConstantOffset(N.getOperand(1))) {
8192 Base = N.getOperand(0);
8193 Offset = C;
8194 return true;
8195 }
8196 }
8197
8198 // By default, just match reg + 0.
8199 Base = N;
8200 Offset = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i64);
8201 return true;
8202}
8203
8204bool AArch64DAGToDAGISel::SelectCmpBranchUImm6Operand(SDNode *P, SDValue N,
8205 SDValue &Imm) {
8207 static_cast<AArch64CC::CondCode>(P->getConstantOperandVal(1));
8208 if (auto *CN = dyn_cast<ConstantSDNode>(N)) {
8209 // Check conservatively if the immediate fits the valid range [0, 64).
8210 // Immediate variants for GE and HS definitely need to be decremented
8211 // when lowering the pseudos later, so an immediate of 1 would become 0.
8212 // For the inverse conditions LT and LO we don't know for sure if they
8213 // will need a decrement but should the decision be made to reverse the
8214 // branch condition, we again end up with the need to decrement.
8215 // The same argument holds for LE, LS, GT and HI and possibly
8216 // incremented immediates. This can lead to slightly less optimal
8217 // codegen, e.g. we never codegen the legal case
8218 // cblt w0, #63, A
8219 // because we could end up with the illegal case
8220 // cbge w0, #64, B
8221 // should the decision to reverse the branch direction be made. For the
8222 // lower bound cases this is no problem since we can express comparisons
8223 // against 0 with either tbz/tnbz or using wzr/xzr.
8224 uint64_t LowerBound = 0, UpperBound = 64;
8225 switch (CC) {
8226 case AArch64CC::GE:
8227 case AArch64CC::HS:
8228 case AArch64CC::LT:
8229 case AArch64CC::LO:
8230 LowerBound = 1;
8231 break;
8232 case AArch64CC::LE:
8233 case AArch64CC::LS:
8234 case AArch64CC::GT:
8235 case AArch64CC::HI:
8236 UpperBound = 63;
8237 break;
8238 default:
8239 break;
8240 }
8241
8242 if (CN->getAPIntValue().uge(LowerBound) &&
8243 CN->getAPIntValue().ult(UpperBound)) {
8244 SDLoc DL(N);
8245 Imm = CurDAG->getTargetConstant(CN->getZExtValue(), DL, N.getValueType());
8246 return true;
8247 }
8248 }
8249
8250 return false;
8251}
8252
8253template <bool MatchCBB>
8254bool AArch64DAGToDAGISel::SelectCmpBranchExtOperand(SDValue N, SDValue &Reg,
8255 SDValue &ExtType) {
8256
8257 // Use an invalid shift-extend value to indicate we don't need to extend later
8258 if (N.getOpcode() == ISD::AssertZext || N.getOpcode() == ISD::AssertSext) {
8259 EVT Ty = cast<VTSDNode>(N.getOperand(1))->getVT();
8260 if (Ty != (MatchCBB ? MVT::i8 : MVT::i16))
8261 return false;
8262 Reg = N.getOperand(0);
8263 ExtType = CurDAG->getSignedTargetConstant(AArch64_AM::InvalidShiftExtend,
8264 SDLoc(N), MVT::i32);
8265 return true;
8266 }
8267
8269
8270 if ((MatchCBB && (ET == AArch64_AM::UXTB || ET == AArch64_AM::SXTB)) ||
8271 (!MatchCBB && (ET == AArch64_AM::UXTH || ET == AArch64_AM::SXTH))) {
8272 Reg = N.getOperand(0);
8273 ExtType =
8274 CurDAG->getTargetConstant(getExtendEncoding(ET), SDLoc(N), MVT::i32);
8275 return true;
8276 }
8277
8278 return false;
8279}
8280
8281/// Try to fold AArch64 CSEL/FCMP patterns to FMAXNM/FMINNM.
8282///
8283/// This is intentionally done in PreprocessISelDAG rather than DAGCombine:
8284/// doing this earlier based on the defining operation of X can be invalidated
8285/// by later DAG combines. At this point the DAG is being prepared for
8286/// instruction selection, so the use of isKnownNeverSNaN(X) applies to the
8287/// final SDValue being selected.
8288/// Only handles FCMP(X, C) with scalar FP types, where C is a non-NaN constant.
8289/// The nsz requirement is needed only when C is zero, to avoid signed-zero
8290/// mismatches. The never-sNaN check is required because AArch64 FMAXNM/FMINNM
8291/// differ from fcmp+fcsel for signaling NaN inputs.
8292bool AArch64DAGToDAGISel::tryFoldCselToFMaxMin(SDNode *N) {
8293 EVT VT = N->getValueType(0);
8294
8295 // Scalar FP only.
8296 if (!VT.isFloatingPoint() || VT.isVector())
8297 return false;
8298
8299 SDValue TVal = N->getOperand(0);
8300 SDValue FVal = N->getOperand(1);
8301 SDValue CCVal = N->getOperand(2);
8302 SDValue Cmp = N->getOperand(3);
8303
8304 if (Cmp.getOpcode() != AArch64ISD::FCMP)
8305 return false;
8306
8307 auto *CC = dyn_cast<ConstantSDNode>(CCVal);
8308 if (!CC)
8309 return false;
8310
8311 SDValue CmpLHS = Cmp.getOperand(0);
8312 SDValue CmpRHS = Cmp.getOperand(1);
8313 unsigned CondCode = CC->getZExtValue();
8314
8315 // Map VT and operation (max/min) to machine opcode.
8316 auto getOpc = [](EVT VT, bool isMax) -> unsigned {
8317 if (VT == MVT::f16)
8318 return isMax ? AArch64::FMAXNMHrr : AArch64::FMINNMHrr;
8319 else if (VT == MVT::f32)
8320 return isMax ? AArch64::FMAXNMSrr : AArch64::FMINNMSrr;
8321 else if (VT == MVT::f64)
8322 return isMax ? AArch64::FMAXNMDrr : AArch64::FMINNMDrr;
8323 else
8324 return 0; // unsupported
8325 };
8326
8327 // Determine whether to use max or min based on condition code and operands.
8328 bool isMax;
8329 if (CondCode == AArch64CC::GT || CondCode == AArch64CC::GE) {
8330 if (TVal == CmpLHS && FVal == CmpRHS)
8331 isMax = true;
8332 else
8333 return false;
8334 } else if (CondCode == AArch64CC::MI || CondCode == AArch64CC::LS) {
8335 if (TVal == CmpLHS && FVal == CmpRHS)
8336 isMax = false;
8337 else
8338 return false;
8339 } else {
8340 return false;
8341 }
8342
8343 // Get the machine opcode for this VT and operation.
8344 unsigned Opc = getOpc(VT, isMax);
8345 if (!Opc)
8346 return false;
8347
8348 // Constant must be non-NaN.
8349 auto *CFP = dyn_cast<ConstantFPSDNode>(CmpRHS);
8350 if (!CFP || CFP->getValueAPF().isNaN())
8351 return false;
8352
8353 // nsz flag required only when constant is zero: fmaxnm(+0,-0)=+0 differs from
8354 // fcmp+select's -0. For non-zero constants, semantics are identical.
8355 if (CFP->isZero() && !N->getFlags().hasNoSignedZeros())
8356 return false;
8357
8358 // Only fold if variable operand is never sNaN.
8359 // This runs after DAG combines, so later combines cannot remove a defining
8360 // operation used by isKnownNeverSNaN().
8361 if (!CurDAG->isKnownNeverSNaN(CmpLHS))
8362 return false;
8363
8364 CurDAG->SelectNodeTo(N, Opc, VT, CmpLHS, CmpRHS);
8365 return true;
8366}
8367
8368void AArch64DAGToDAGISel::PreprocessISelDAG() {
8369 bool MadeChange = false;
8370 for (SDNode &N : llvm::make_early_inc_range(CurDAG->allnodes())) {
8371 if (N.use_empty())
8372 continue;
8373
8374 SDValue Result;
8375 switch (N.getOpcode()) {
8376 case ISD::SCALAR_TO_VECTOR: {
8377 EVT ScalarTy = N.getValueType(0).getVectorElementType();
8378 if ((ScalarTy == MVT::i32 || ScalarTy == MVT::i64) &&
8379 ScalarTy == N.getOperand(0).getValueType())
8380 Result = addBitcastHints(*CurDAG, N);
8381
8382 break;
8383 }
8384 case AArch64ISD::VSHL: {
8385 // Undo mul(shl(A,C),B) -> shl(mul(A,B),C) canonicalisation when A is an
8386 // extend that can be folded into the shift.
8387 EVT VT = N.getValueType(0);
8388 SDValue A, B, C = N.getOperand(1);
8389 if (sd_match(N.getOperand(0),
8391 m_SExt(m_Value()))),
8392 m_Value(B))))) {
8393 // If both mul operands are extended, preserve the smull/umull idiom.
8394 if (B.getOpcode() == A.getOpcode())
8395 break;
8396 SDLoc DL(&N);
8397 SDValue SHL = CurDAG->getNode(AArch64ISD::VSHL, DL, VT, A, C);
8398 Result = CurDAG->getNode(ISD::MUL, DL, VT, SHL, B);
8399 }
8400 break;
8401 }
8402 default:
8403 break;
8404 }
8405
8406 if (Result) {
8407 LLVM_DEBUG(dbgs() << "AArch64 DAG preprocessing replacing:\nOld: ");
8408 LLVM_DEBUG(N.dump(CurDAG));
8409 LLVM_DEBUG(dbgs() << "\nNew: ");
8410 LLVM_DEBUG(Result.dump(CurDAG));
8411 LLVM_DEBUG(dbgs() << "\n");
8412
8413 CurDAG->ReplaceAllUsesOfValueWith(SDValue(&N, 0), Result);
8414 MadeChange = true;
8415 }
8416 }
8417
8418 if (MadeChange)
8419 CurDAG->RemoveDeadNodes();
8420
8422}
static std::optional< APInt > GetNEONSplatValue(SDValue N, const AArch64Subtarget *Subtarget)
static SDValue Widen(SelectionDAG *CurDAG, SDValue N)
static bool isBitfieldExtractOpFromSExtInReg(SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &Immr, unsigned &Imms)
static int getIntOperandFromRegisterString(StringRef RegString)
static SDValue NarrowVector(SDValue V128Reg, SelectionDAG &DAG)
NarrowVector - Given a value in the V128 register class, produce the equivalent value in the V64 regi...
static bool isBitfieldDstMask(uint64_t DstMask, const APInt &BitsToBeInserted, unsigned NumberOfIgnoredHighBits, EVT VT)
Does DstMask form a complementary pair with the mask provided by BitsToBeInserted,...
static SDValue narrowIfNeeded(SelectionDAG *CurDAG, SDValue N)
Instructions that accept extend modifiers like UXTW expect the register being extended to be a GPR32,...
static bool isSeveralBitsPositioningOpFromShl(const uint64_t ShlImm, SDValue Op, SDValue &Src, int &DstLSB, int &Width)
static bool isBitfieldPositioningOp(SelectionDAG *CurDAG, SDValue Op, bool BiggerPattern, SDValue &Src, int &DstLSB, int &Width)
Does this tree qualify as an attempt to move a bitfield into position, essentially "(and (shl VAL,...
static bool isOpcWithIntImmediate(const SDNode *N, unsigned Opc, uint64_t &Imm)
static bool tryBitfieldInsertOpFromOrAndImm(SDNode *N, SelectionDAG *CurDAG)
static std::tuple< SDValue, SDValue > extractPtrauthBlendDiscriminators(SDValue Disc, SelectionDAG *DAG)
static SDValue addBitcastHints(SelectionDAG &DAG, SDNode &N)
addBitcastHints - This method adds bitcast hints to the operands of a node to help instruction select...
static void getUsefulBitsFromOrWithShiftedReg(SDValue Op, APInt &UsefulBits, unsigned Depth)
static bool isBitfieldExtractOpFromAnd(SelectionDAG *CurDAG, SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &LSB, unsigned &MSB, unsigned NumberOfIgnoredLowBits, bool BiggerPattern)
static bool isBitfieldExtractOp(SelectionDAG *CurDAG, SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &Immr, unsigned &Imms, unsigned NumberOfIgnoredLowBits=0, bool BiggerPattern=false)
static bool isShiftedMask(uint64_t Mask, EVT VT)
bool SelectSMETile(unsigned &BaseReg, unsigned TileNum)
static EVT getMemVTFromNode(LLVMContext &Ctx, SDNode *Root)
Return the EVT of the data associated to a memory operation in Root.
static bool checkCVTFixedPointOperandWithFBits(SelectionDAG *CurDAG, SDValue N, SDValue &FixedPos, unsigned RegWidth, bool isReciprocal)
static bool isWorthFoldingADDlow(SDValue N)
If there's a use of this ADDlow that's not itself a load/store then we'll need to create a real ADD i...
static AArch64_AM::ShiftExtendType getShiftTypeForNode(SDValue N)
getShiftTypeForNode - Translate a shift node to the corresponding ShiftType value.
static bool isSeveralBitsExtractOpFromShr(SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &LSB, unsigned &MSB)
static unsigned SelectOpcodeFromVT(EVT VT, ArrayRef< unsigned > Opcodes)
This function selects an opcode from a list of opcodes, which is expected to be the opcode for { 8-bi...
static EVT getPackedVectorTypeFromPredicateType(LLVMContext &Ctx, EVT PredVT, unsigned NumVec)
When PredVT is a scalable vector predicate in the form MVT::nx<M>xi1, it builds the correspondent sca...
static bool checkCVTFixedPointOperandWithFBitsForVectors(SelectionDAG *CurDAG, SDValue N, SDValue &FixedPos, unsigned RegWidth, bool isReciprocal)
static SDValue getZeroRegister(SelectionDAG &DAG, SDLoc DL, EVT VT)
Returns a copy from WZR or XZR.
static bool isPreferredADD(int64_t ImmOff)
static void getUsefulBitsFromBitfieldMoveOpd(SDValue Op, APInt &UsefulBits, uint64_t Imm, uint64_t MSB, unsigned Depth)
static SDValue getLeftShift(SelectionDAG *CurDAG, SDValue Op, int ShlAmount)
Create a machine node performing a notional SHL of Op by ShlAmount.
static bool isWorthFoldingSHL(SDValue V)
Determine whether it is worth it to fold SHL into the addressing mode.
static bool isBitfieldPositioningOpFromAnd(SelectionDAG *CurDAG, SDValue Op, bool BiggerPattern, const uint64_t NonZeroBits, SDValue &Src, int &DstLSB, int &Width)
static void getUsefulBitsFromBFM(SDValue Op, SDValue Orig, APInt &UsefulBits, unsigned Depth)
static bool isBitfieldExtractOpFromShr(SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &Immr, unsigned &Imms, bool BiggerPattern)
static bool tryOrrWithShift(SDNode *N, SDValue OrOpd0, SDValue OrOpd1, SDValue Src, SDValue Dst, SelectionDAG *CurDAG, const bool BiggerPattern)
static void getUsefulBitsForUse(SDNode *UserNode, APInt &UsefulBits, SDValue Orig, unsigned Depth)
static bool isMemOpOrPrefetch(SDNode *N)
static void getUsefulBitsFromUBFM(SDValue Op, APInt &UsefulBits, unsigned Depth)
static bool tryBitfieldInsertOpFromOr(SDNode *N, const APInt &UsefulBits, SelectionDAG *CurDAG)
static APInt DecodeFMOVImm(uint64_t Imm, unsigned RegWidth)
static void getUsefulBitsFromAndWithImmediate(SDValue Op, APInt &UsefulBits, unsigned Depth)
static std::optional< APInt > DecodeNEONSplat(SDValue N, const AArch64Subtarget *Subtarget)
static void getUsefulBits(SDValue Op, APInt &UsefulBits, unsigned Depth=0)
static bool isIntImmediateEq(SDValue N, const uint64_t ImmExpected)
static EVT getMultipleVectorType(LLVMContext &Ctx, EVT VecVT, unsigned NumVec)
Builds an integer vector type large enough to hold NumVec instances of VecVT.
static AArch64_AM::ShiftExtendType getExtendTypeForNode(SDValue N, bool IsLoadStore=false)
getExtendTypeForNode - Translate an extend node to the corresponding ExtendType value.
static bool isIntImmediate(const SDNode *N, uint64_t &Imm)
isIntImmediate - This method tests to see if the node is a constant operand.
static bool isWorthFoldingIntoOrrWithShift(SDValue Dst, SelectionDAG *CurDAG, SDValue &ShiftedOperand, uint64_t &EncodedShiftImm)
static bool isValidAsScaledImmediate(int64_t Offset, unsigned Range, unsigned Size)
Check if the immediate offset is valid as a scaled immediate.
static bool isBitfieldPositioningOpFromShl(SelectionDAG *CurDAG, SDValue Op, bool BiggerPattern, const uint64_t NonZeroBits, SDValue &Src, int &DstLSB, int &Width)
static SDValue WidenVector(SDValue V64Reg, SelectionDAG &DAG)
WidenVector - Given a value in the V64 register class, produce the equivalent value in the V128 regis...
static Register createDTuple(ArrayRef< Register > Regs, MachineIRBuilder &MIB)
Create a tuple of D-registers using the registers in Regs.
static Register createQTuple(ArrayRef< Register > Regs, MachineIRBuilder &MIB)
Create a tuple of Q-registers using the registers in Regs.
static Register createTuple(ArrayRef< Register > Regs, const unsigned RegClassIDs[], const unsigned SubRegs[], MachineIRBuilder &MIB)
Create a REG_SEQUENCE instruction using the registers in Regs.
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static msgpack::DocNode getNode(msgpack::DocNode DN, msgpack::Type Type, MCValue Val)
unsigned Imm
unsigned uint64_t
AMDGPU Register Bank Select
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
dxil translate DXIL Translate Metadata
#define DEBUG_TYPE
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
#define R2(n)
Promote Memory to Register
Definition Mem2Reg.cpp:110
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t High
OptimizedStructLayoutField Field
#define P(N)
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
Contains matchers for matching SelectionDAG nodes and values.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define PASS_NAME
Value * RHS
Value * LHS
AArch64DAGToDAGISelPass(AArch64TargetMachine &TM)
const AArch64InstrInfo * getInstrInfo() const override
const AArch64TargetLowering * getTargetLowering() const override
bool isStreaming() const
Returns true if the function has a streaming body.
bool isX16X17Safer() const
Returns whether the operating system makes it safer to store sensitive values in x16 and x17 as oppos...
unsigned getSVEVectorSizeInBits() const
bool isAllActivePredicate(const SelectionDAG &DAG, SDValue N) const
Register matchRegisterName(StringRef RegName) const
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
unsigned popcount() const
Count the number of bits set.
Definition APInt.h:1690
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1078
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:254
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1659
unsigned countl_zero() const
The APInt version of std::countl_zero.
Definition APInt.h:1618
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
void flipAllBits()
Toggle every bit to its opposite value.
Definition APInt.h:1472
bool isShiftedMask() const
Return true if this APInt value contains a non-empty sequence of ones with the remainder zero.
Definition APInt.h:506
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
void lshrInPlace(unsigned ShiftAmt)
Logical right-shift this APInt by ShiftAmt in place.
Definition APInt.h:860
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:853
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
iterator begin() const
Definition ArrayRef.h:129
const Constant * getConstVal() const
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
const GlobalValue * getGlobal() const
const TargetRegisterClass * getInlineAsmMemoryOperandRegClass(InlineAsm::ConstraintCode C) const override
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
This class is used to represent ISD::LOAD nodes.
unsigned getID() const
getID() - Return the register class ID number.
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1437
unsigned getNumOperands() const
Return number of MDNode operands.
Definition Metadata.h:1443
bool equalsStr(StringRef Str) const
Definition Metadata.h:924
Metadata * get() const
Definition Metadata.h:931
Machine Value Type.
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
static MVT getVectorVT(MVT VT, unsigned NumElements)
bool hasScalableStackID(int ObjectIdx) const
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
A description of a memory reference used in the backend.
const MDNode * getMemCacheHint() const
Return the cache hint metadata for the memory reference.
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
iterator_range< user_iterator > users()
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
const SDValue & getOperand(unsigned i) const
uint64_t getConstantOperandVal(unsigned i) const
unsigned getOpcode() const
SelectionDAGISelPass(std::unique_ptr< SelectionDAGISel > Selector)
SelectionDAGISel - This is the common base class used for SelectionDAG-based pattern-matching instruc...
virtual void PreprocessISelDAG()
PreprocessISelDAG - This hook allows targets to hack on the graph before instruction selection starts...
virtual bool runOnMachineFunction(MachineFunction &mf)
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDNode * SelectNodeTo(SDNode *N, unsigned MachineOpc, EVT VT)
These are used for target selectors to mutate the specified node to have the specified return type,...
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
static constexpr unsigned MaxRecursionDepth
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
LLVM_ABI SDValue getTargetExtractSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand)
A convenience function for creating TargetInstrInfo::EXTRACT_SUBREG nodes.
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetInsertSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand, SDValue Subreg)
A convenience function for creating TargetInstrInfo::INSERT_SUBREG nodes.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
void reserve(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
LLVM Value Representation.
Definition Value.h:75
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Definition Value.cpp:1002
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
uint32_t parseGenericRegister(StringRef Name)
static uint64_t decodeLogicalImmediate(uint64_t val, unsigned regSize)
decodeLogicalImmediate - Decode a logical immediate value in the form "N:immr:imms" (where the immr a...
static unsigned getShiftValue(unsigned Imm)
getShiftValue - Extract the shift value.
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
static uint64_t decodeAdvSIMDModImmType12(uint8_t Imm)
constexpr bool isLegalArithImmed(const uint64_t C)
isLegalArithImmed -
static uint64_t decodeAdvSIMDModImmType11(uint8_t Imm)
unsigned getExtendEncoding(AArch64_AM::ShiftExtendType ET)
Mapping from extend bits to required operation: shifter: 000 ==> uxtb 001 ==> uxth 010 ==> uxtw 011 =...
static uint64_t decodeAdvSIMDModImmType10(uint8_t Imm)
static bool isSVELogicalImm(unsigned SizeInBits, uint64_t ImmVal, uint64_t &Encoding)
constexpr unsigned getArithImmedShift(const uint64_t C)
getArithImmedShift - assumes C is a legal immediate for arithmetic instructions and
static bool isSVECpyDupImm(int SizeInBits, int64_t Val, int32_t &Imm, int32_t &Shift)
static AArch64_AM::ShiftExtendType getShiftType(unsigned Imm)
getShiftType - Extract the shift type.
static unsigned getShifterImm(AArch64_AM::ShiftExtendType ST, unsigned Imm)
getShifterImm - Encode the shift type and amount: imm: 6-bit shift amount shifter: 000 ==> lsl 001 ==...
static bool isSignExtendShiftType(AArch64_AM::ShiftExtendType Type)
isSignExtendShiftType - Returns true if Type is sign extending.
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
static constexpr unsigned SVEBitsPerBlock
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ POISON
POISON - A poison node.
Definition ISDOpcodes.h:238
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:605
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ SPLAT_VECTOR
SPLAT_VECTOR(VAL) - Returns a vector with the scalar value VAL duplicated in all lanes.
Definition ISDOpcodes.h:682
@ AssertAlign
AssertAlign - These nodes record if a register contains a value that has a known alignment and the tr...
Definition ISDOpcodes.h:71
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
Definition ISDOpcodes.h:232
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:619
@ READ_REGISTER
READ_REGISTER, WRITE_REGISTER - This node represents llvm.register on the DAG, which implements the n...
Definition ISDOpcodes.h:141
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ VSCALE
VSCALE(IMM) - Returns the runtime scaling factor used to calculate the number of elements within a sc...
@ ATOMIC_CMP_SWAP
Val, OUTCHAIN = ATOMIC_CMP_SWAP(INCHAIN, ptr, cmp, swap) For double-word atomic operations: ValLo,...
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ FREEZE
FREEZE - FREEZE(VAL) returns an arbitrary value if VAL is UNDEF (or is evaluated to UNDEF),...
Definition ISDOpcodes.h:243
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:64
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
LLVM_ABI bool isConstantSplatVector(const SDNode *N, APInt &SplatValue)
Node predicates.
MemIndexedMode
MemIndexedMode enum - This enum defines the load / store indexed addressing modes.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
BinaryOpc_match< LHS, RHS, true > m_Mul(const LHS &L, const RHS &R)
Or< Preds... > m_AnyOf(const Preds &...preds)
auto m_SExt(const Opnd &Op)
bool sd_match(SDValue N, Pattern &&P)
UnaryOpc_match< Opnd > m_ZExt(const Opnd &Op)
Value_match m_Value()
Match any valid SDValue.
NUses_match< 1, Value_match > m_OneUse()
Not(const Pred &P) -> Not< Pred >
DiagnosticInfoOptimizationBase::Argument NV
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:577
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
unsigned CheckFixedPointOperandConstant(APFloat &FVal, unsigned RegWidth, bool isReciprocal)
@ Known
Known to have no common set bits.
@ Undef
Value of the register doesn't matter.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool isStrongerThanMonotonic(AtomicOrdering AO)
int countr_one(T Value)
Count the number of ones from the least significant bit to the first zero bit.
Definition bit.h:315
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
constexpr bool isShiftedMask_32(uint32_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (32 bit ver...
Definition MathExtras.h:268
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
OutputIt transform(R &&Range, OutputIt d_first, UnaryFunction F)
Wrapper function around std::transform to apply a function to a range and store the result elsewhere.
Definition STLExtras.h:2042
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr bool isMask_64(uint64_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
Definition MathExtras.h:262
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
CodeGenOptLevel
Code generation optimization level.
Definition CodeGen.h:227
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
FunctionPass * createAArch64ISelDag(AArch64TargetMachine &TM, CodeGenOptLevel OptLevel)
createAArch64ISelDag - This pass converts a legalized DAG into a AArch64-specific DAG,...
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool isNullFPConstant(SDValue V)
Returns true if V is an FP constant with a value of positive zero.
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
AArch64MemoryHint toAArch64MemoryHint(Int I)
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
Extended Value Type.
Definition ValueTypes.h:35
bool isScalableVectorOf(EVT EltVT) const
Return true if this is a scalable vector with matching element type.
Definition ValueTypes.h:192
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
EVT changeTypeToInteger() const
Return the type converted to an equivalently sized integer or vector with integer element type.
Definition ValueTypes.h:129
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
ElementCount getVectorElementCount() const
Definition ValueTypes.h:373
EVT getDoubleNumVectorElementsVT(LLVMContext &Context) const
Definition ValueTypes.h:494
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
Definition ValueTypes.h:382
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
EVT changeVectorElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
Definition ValueTypes.h:98
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool is128BitVector() const
Return true if this is a 128-bit vector type.
Definition ValueTypes.h:230
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
bool isFixedLengthVector() const
Definition ValueTypes.h:199
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
Definition ValueTypes.h:187
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool is64BitVector() const
Return true if this is a 64-bit vector type.
Definition ValueTypes.h:225
Matching combinators.