LLVM 24.0.0git
VEISelLowering.cpp
Go to the documentation of this file.
1//===-- VEISelLowering.cpp - VE DAG Lowering Implementation ---------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the interfaces that VE uses to lower LLVM code into a
10// selection DAG.
11//
12//===----------------------------------------------------------------------===//
13
14#include "VEISelLowering.h"
16#include "VECustomDAG.h"
17#include "VEInstrBuilder.h"
19#include "VERegisterInfo.h"
20#include "VESelectionDAGInfo.h"
21#include "VETargetMachine.h"
33#include "llvm/IR/Function.h"
34#include "llvm/IR/IRBuilder.h"
35#include "llvm/IR/Module.h"
37using namespace llvm;
38
39#define DEBUG_TYPE "ve-lower"
40
41//===----------------------------------------------------------------------===//
42// Calling Convention Implementation
43//===----------------------------------------------------------------------===//
44
45#define GET_CALLING_CONV_IMPL
46#include "VEGenCallingConv.inc"
47
49 switch (CallConv) {
50 default:
51 return RetCC_VE_C;
53 return RetCC_VE_Fast;
54 }
55}
56
57CCAssignFn *getParamCC(CallingConv::ID CallConv, bool IsVarArg) {
58 if (IsVarArg)
59 return CC_VE2;
60 switch (CallConv) {
61 default:
62 return CC_VE_C;
64 return CC_VE_Fast;
65 }
66}
67
69 CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg,
70 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context,
71 const Type *RetTy) const {
72 CCAssignFn *RetCC = getReturnCC(CallConv);
74 CCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context);
75 return CCInfo.CheckReturn(Outs, RetCC);
76}
77
78static const MVT AllVectorVTs[] = {MVT::v256i32, MVT::v512i32, MVT::v256i64,
79 MVT::v256f32, MVT::v512f32, MVT::v256f64};
80
81static const MVT AllMaskVTs[] = {MVT::v256i1, MVT::v512i1};
82
83static const MVT AllPackedVTs[] = {MVT::v512i32, MVT::v512f32};
84
85void VETargetLowering::initRegisterClasses() {
86 // Set up the register classes.
87 addRegisterClass(MVT::i32, &VE::I32RegClass);
88 addRegisterClass(MVT::i64, &VE::I64RegClass);
89 addRegisterClass(MVT::f32, &VE::F32RegClass);
90 addRegisterClass(MVT::f64, &VE::I64RegClass);
91 addRegisterClass(MVT::f128, &VE::F128RegClass);
92
93 if (Subtarget->enableVPU()) {
94 for (MVT VecVT : AllVectorVTs)
95 addRegisterClass(VecVT, &VE::V64RegClass);
96 addRegisterClass(MVT::v256i1, &VE::VMRegClass);
97 addRegisterClass(MVT::v512i1, &VE::VM512RegClass);
98 }
99}
100
101void VETargetLowering::initSPUActions() {
102 /// Load & Store {
103
104 // VE doesn't have i1 sign extending load.
105 for (MVT VT : MVT::integer_valuetypes()) {
109 setTruncStoreAction(VT, MVT::i1, Expand);
110 }
111
112 // VE doesn't have floating point extload/truncstore, so expand them.
113 for (MVT FPVT : MVT::fp_valuetypes()) {
114 for (MVT OtherFPVT : MVT::fp_valuetypes()) {
115 setLoadExtAction(ISD::EXTLOAD, FPVT, OtherFPVT, Expand);
116 setTruncStoreAction(FPVT, OtherFPVT, Expand);
117 }
118 }
119
120 // VE doesn't have fp128 load/store, so expand them in custom lower.
123
124 /// } Load & Store
125
126 // Custom legalize address nodes into LO/HI parts.
127 MVT PtrVT = MVT::i64;
133
134 /// VAARG handling {
136 // VAARG needs to be lowered to access with 8 bytes alignment.
138 // Use the default implementation.
141 /// } VAARG handling
142
143 /// Stack {
146
147 // Use the default implementation.
150 /// } Stack
151
152 /// Branch {
153
154 // VE doesn't have BRCOND
156
157 // BR_JT is not implemented yet.
159
160 /// } Branch
161
162 /// Int Ops {
163 for (MVT IntVT : {MVT::i32, MVT::i64}) {
164 // VE has no REM or DIVREM operations.
169
170 // VE has no SHL_PARTS/SRA_PARTS/SRL_PARTS operations.
174
175 // VE has no MULHU/S or U/SMUL_LOHI operations.
176 // TODO: Use MPD instruction to implement SMUL_LOHI for i32 type.
181
182 // VE has no CTTZ, ROTL, ROTR operations.
186
187 // VE has 64 bits instruction which works as i64 BSWAP operation. This
188 // instruction works fine as i32 BSWAP operation with an additional
189 // parameter. Use isel patterns to lower BSWAP.
191
192 // VE has only 64 bits instructions which work as i64 BITREVERSE/CTLZ/CTPOP
193 // operations. Use isel patterns for i64, promote for i32.
194 LegalizeAction Act = (IntVT == MVT::i32) ? Promote : Legal;
196 setOperationAction(ISD::CTLZ, IntVT, Act);
198 setOperationAction(ISD::CTPOP, IntVT, Act);
199
200 // VE has only 64 bits instructions which work as i64 AND/OR/XOR operations.
201 // Use isel patterns for i64, promote for i32.
202 setOperationAction(ISD::AND, IntVT, Act);
203 setOperationAction(ISD::OR, IntVT, Act);
204 setOperationAction(ISD::XOR, IntVT, Act);
205
206 // Legal smax and smin
209 }
210 /// } Int Ops
211
212 /// Conversion {
213 // VE doesn't have instructions for fp<->uint, so expand them by llvm
214 setOperationAction(ISD::FP_TO_UINT, MVT::i32, Promote); // use i64
215 setOperationAction(ISD::UINT_TO_FP, MVT::i32, Promote); // use i64
218
219 // fp16 not supported
220 for (MVT FPVT : MVT::fp_valuetypes()) {
223 }
224 /// } Conversion
225
226 /// Floating-point Ops {
227 /// Note: Floating-point operations are fneg, fadd, fsub, fmul, fdiv, frem,
228 /// and fcmp.
229
230 // VE doesn't have following floating point operations.
231 for (MVT VT : MVT::fp_valuetypes()) {
234 }
235
236 // VE doesn't have fdiv of f128.
238
239 for (MVT FPVT : {MVT::f32, MVT::f64}) {
240 // f32 and f64 uses ConstantFP. f128 uses ConstantPool.
242 }
243 /// } Floating-point Ops
244
245 /// Floating-point math functions {
246
247 // VE doesn't have following floating point math functions.
248 for (MVT VT : MVT::fp_valuetypes()) {
256 }
257
258 // VE has single and double FMINNUM and FMAXNUM
259 for (MVT VT : {MVT::f32, MVT::f64}) {
261 }
262
263 /// } Floating-point math functions
264
265 /// Atomic instructions {
266
270
271 // Use custom inserter for ATOMIC_FENCE.
273
274 // Other atomic instructions.
275 for (MVT VT : MVT::integer_valuetypes()) {
276 // Support i8/i16 atomic swap.
278
279 // FIXME: Support "atmam" instructions.
284
285 // VE doesn't have follwing instructions.
294 }
295
296 /// } Atomic instructions
297
298 /// SJLJ instructions {
302 /// } SJLJ instructions
303
304 // Intrinsic instructions
306}
307
308void VETargetLowering::initVPUActions() {
309 for (MVT LegalMaskVT : AllMaskVTs)
311
312 for (unsigned Opc : {ISD::AND, ISD::OR, ISD::XOR})
313 setOperationAction(Opc, MVT::v512i1, Custom);
314
315 for (MVT LegalVecVT : AllVectorVTs) {
319 // Translate all vector instructions with legal element types to VVP_*
320 // nodes.
321 // TODO We will custom-widen into VVP_* nodes in the future. While we are
322 // buildling the infrastructure for this, we only do this for legal vector
323 // VTs.
324#define HANDLE_VP_TO_VVP(VP_OPC, VVP_NAME) \
325 setOperationAction(ISD::VP_OPC, LegalVecVT, Custom);
326#define ADD_VVP_OP(VVP_NAME, ISD_NAME) \
327 setOperationAction(ISD::ISD_NAME, LegalVecVT, Custom);
328 setOperationAction(ISD::EXPERIMENTAL_VP_STRIDED_LOAD, LegalVecVT, Custom);
329 setOperationAction(ISD::EXPERIMENTAL_VP_STRIDED_STORE, LegalVecVT, Custom);
330#include "VVPNodes.def"
331 }
332
333 for (MVT LegalPackedVT : AllPackedVTs) {
336 }
337
338 // vNt32, vNt64 ops (legal element types)
339 for (MVT VT : MVT::vector_valuetypes()) {
340 MVT ElemVT = VT.getVectorElementType();
341 unsigned ElemBits = ElemVT.getScalarSizeInBits();
342 if (ElemBits != 32 && ElemBits != 64)
343 continue;
344
345 for (unsigned MemOpc : {ISD::MLOAD, ISD::MSTORE, ISD::LOAD, ISD::STORE})
346 setOperationAction(MemOpc, VT, Custom);
347
348 const ISD::NodeType IntReductionOCs[] = {
352
353 for (unsigned IntRedOpc : IntReductionOCs)
354 setOperationAction(IntRedOpc, VT, Custom);
355 }
356
357 // v256i1 and v512i1 ops
358 for (MVT MaskVT : AllMaskVTs) {
359 // Custom lower mask ops
362 }
363}
364
367 bool IsVarArg,
369 const SmallVectorImpl<SDValue> &OutVals,
370 const SDLoc &DL, SelectionDAG &DAG) const {
371 // CCValAssign - represent the assignment of the return value to locations.
373
374 // CCState - Info about the registers and stack slot.
375 CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), RVLocs,
376 *DAG.getContext());
377
378 // Analyze return values.
379 CCInfo.AnalyzeReturn(Outs, getReturnCC(CallConv));
380
381 SDValue Glue;
382 SmallVector<SDValue, 4> RetOps(1, Chain);
383
384 // Copy the result values into the output registers.
385 for (unsigned i = 0; i != RVLocs.size(); ++i) {
386 CCValAssign &VA = RVLocs[i];
387 assert(VA.isRegLoc() && "Can only return in registers!");
388 assert(!VA.needsCustom() && "Unexpected custom lowering");
389 SDValue OutVal = OutVals[i];
390
391 // Integer return values must be sign or zero extended by the callee.
392 switch (VA.getLocInfo()) {
394 break;
396 OutVal = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), OutVal);
397 break;
399 OutVal = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), OutVal);
400 break;
402 OutVal = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), OutVal);
403 break;
404 case CCValAssign::BCvt: {
405 // Convert a float return value to i64 with padding.
406 // 63 31 0
407 // +------+------+
408 // | float| 0 |
409 // +------+------+
410 assert(VA.getLocVT() == MVT::i64);
411 assert(VA.getValVT() == MVT::f32);
413 DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i64), 0);
414 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32);
415 OutVal = SDValue(DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL,
416 MVT::i64, Undef, OutVal, Sub_f32),
417 0);
418 break;
419 }
420 default:
421 llvm_unreachable("Unknown loc info!");
422 }
423
424 Chain = DAG.getCopyToReg(Chain, DL, VA.getLocReg(), OutVal, Glue);
425
426 // Guarantee that all emitted copies are stuck together with flags.
427 Glue = Chain.getValue(1);
428 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT()));
429 }
430
431 RetOps[0] = Chain; // Update chain.
432
433 // Add the glue if we have it.
434 if (Glue.getNode())
435 RetOps.push_back(Glue);
436
437 return DAG.getNode(VEISD::RET_GLUE, DL, MVT::Other, RetOps);
438}
439
441 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
442 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
443 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
445
446 // Get the base offset of the incoming arguments stack space.
447 unsigned ArgsBaseOffset = Subtarget->getRsaSize();
448 // Get the size of the preserved arguments area
449 unsigned ArgsPreserved = 64;
450
451 // Analyze arguments according to CC_VE.
453 CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), ArgLocs,
454 *DAG.getContext());
455 // Allocate the preserved area first.
456 CCInfo.AllocateStack(ArgsPreserved, Align(8));
457 // We already allocated the preserved area, so the stack offset computed
458 // by CC_VE would be correct now.
459 CCInfo.AnalyzeFormalArguments(Ins, getParamCC(CallConv, false));
460
461 for (const CCValAssign &VA : ArgLocs) {
462 assert(!VA.needsCustom() && "Unexpected custom lowering");
463 if (VA.isRegLoc()) {
464 // This argument is passed in a register.
465 // All integer register arguments are promoted by the caller to i64.
466
467 // Create a virtual register for the promoted live-in value.
468 Register VReg =
469 MF.addLiveIn(VA.getLocReg(), getRegClassFor(VA.getLocVT()));
470 SDValue Arg = DAG.getCopyFromReg(Chain, DL, VReg, VA.getLocVT());
471
472 // The caller promoted the argument, so insert an Assert?ext SDNode so we
473 // won't promote the value again in this function.
474 switch (VA.getLocInfo()) {
476 Arg = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), Arg,
477 DAG.getValueType(VA.getValVT()));
478 break;
480 Arg = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), Arg,
481 DAG.getValueType(VA.getValVT()));
482 break;
483 case CCValAssign::BCvt: {
484 // Extract a float argument from i64 with padding.
485 // 63 31 0
486 // +------+------+
487 // | float| 0 |
488 // +------+------+
489 assert(VA.getLocVT() == MVT::i64);
490 assert(VA.getValVT() == MVT::f32);
491 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32);
492 Arg = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
493 MVT::f32, Arg, Sub_f32),
494 0);
495 break;
496 }
497 default:
498 break;
499 }
500
501 // Truncate the register down to the argument type.
502 if (VA.isExtInLoc())
503 Arg = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Arg);
504
505 InVals.push_back(Arg);
506 continue;
507 }
508
509 // The registers are exhausted. This argument was passed on the stack.
510 assert(VA.isMemLoc());
511 // The CC_VE_Full/Half functions compute stack offsets relative to the
512 // beginning of the arguments area at %fp + the size of reserved area.
513 unsigned Offset = VA.getLocMemOffset() + ArgsBaseOffset;
514 unsigned ValSize = VA.getValVT().getSizeInBits() / 8;
515
516 // Adjust offset for a float argument by adding 4 since the argument is
517 // stored in 8 bytes buffer with offset like below. LLVM generates
518 // 4 bytes load instruction, so need to adjust offset here. This
519 // adjustment is required in only LowerFormalArguments. In LowerCall,
520 // a float argument is converted to i64 first, and stored as 8 bytes
521 // data, which is required by ABI, so no need for adjustment.
522 // 0 4
523 // +------+------+
524 // | empty| float|
525 // +------+------+
526 if (VA.getValVT() == MVT::f32)
527 Offset += 4;
528
529 int FI = MF.getFrameInfo().CreateFixedObject(ValSize, Offset, true);
530 InVals.push_back(
531 DAG.getLoad(VA.getValVT(), DL, Chain,
534 }
535
536 if (!IsVarArg)
537 return Chain;
538
539 // This function takes variable arguments, some of which may have been passed
540 // in registers %s0-%s8.
541 //
542 // The va_start intrinsic needs to know the offset to the first variable
543 // argument.
544 // TODO: need to calculate offset correctly once we support f128.
545 unsigned ArgOffset = ArgLocs.size() * 8;
547 // Skip the reserved area at the top of stack.
548 FuncInfo->setVarArgsFrameOffset(ArgOffset + ArgsBaseOffset);
549
550 return Chain;
551}
552
553// FIXME? Maybe this could be a TableGen attribute on some registers and
554// this table could be generated automatically from RegInfo.
556 const MachineFunction &MF) const {
558 .Case("sp", VE::SX11) // Stack pointer
559 .Case("fp", VE::SX9) // Frame pointer
560 .Case("sl", VE::SX8) // Stack limit
561 .Case("lr", VE::SX10) // Link register
562 .Case("tp", VE::SX14) // Thread pointer
563 .Case("outer", VE::SX12) // Outer regiser
564 .Case("info", VE::SX17) // Info area register
565 .Case("got", VE::SX15) // Global offset table register
566 .Case("plt", VE::SX16) // Procedure linkage table register
567 .Default(Register());
568 return Reg;
569}
570
571//===----------------------------------------------------------------------===//
572// TargetLowering Implementation
573//===----------------------------------------------------------------------===//
574
576 SmallVectorImpl<SDValue> &InVals) const {
577 SelectionDAG &DAG = CLI.DAG;
578 SDLoc DL = CLI.DL;
579 SDValue Chain = CLI.Chain;
580 auto PtrVT = getPointerTy(DAG.getDataLayout());
581
582 // VE target does not yet support tail call optimization.
583 CLI.IsTailCall = false;
584
585 // Get the base offset of the outgoing arguments stack space.
586 unsigned ArgsBaseOffset = Subtarget->getRsaSize();
587 // Get the size of the preserved arguments area
588 unsigned ArgsPreserved = 8 * 8u;
589
590 // Analyze operands of the call, assigning locations to each operand.
592 CCState CCInfo(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), ArgLocs,
593 *DAG.getContext());
594 // Allocate the preserved area first.
595 CCInfo.AllocateStack(ArgsPreserved, Align(8));
596 // We already allocated the preserved area, so the stack offset computed
597 // by CC_VE would be correct now.
598 CCInfo.AnalyzeCallOperands(CLI.Outs, getParamCC(CLI.CallConv, false));
599
600 // VE requires to use both register and stack for varargs or no-prototyped
601 // functions.
602 bool UseBoth = CLI.IsVarArg;
603
604 // Analyze operands again if it is required to store BOTH.
606 CCState CCInfo2(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(),
607 ArgLocs2, *DAG.getContext());
608 if (UseBoth)
609 CCInfo2.AnalyzeCallOperands(CLI.Outs, getParamCC(CLI.CallConv, true));
610
611 // Get the size of the outgoing arguments stack space requirement.
612 unsigned ArgsSize = CCInfo.getStackSize();
613
614 // Keep stack frames 16-byte aligned.
615 ArgsSize = alignTo(ArgsSize, 16);
616
617 // Adjust the stack pointer to make room for the arguments.
618 // FIXME: Use hasReservedCallFrame to avoid %sp adjustments around all calls
619 // with more than 6 arguments.
620 Chain = DAG.getCALLSEQ_START(Chain, ArgsSize, 0, DL);
621
622 // Collect the set of registers to pass to the function and their values.
623 // This will be emitted as a sequence of CopyToReg nodes glued to the call
624 // instruction.
626
627 // Collect chains from all the memory opeations that copy arguments to the
628 // stack. They must follow the stack pointer adjustment above and precede the
629 // call instruction itself.
630 SmallVector<SDValue, 8> MemOpChains;
631
632 // VE needs to get address of callee function in a register
633 // So, prepare to copy it to SX12 here.
634
635 // If the callee is a GlobalAddress node (quite common, every direct call is)
636 // turn it into a TargetGlobalAddress node so that legalize doesn't hack it.
637 // Likewise ExternalSymbol -> TargetExternalSymbol.
638 SDValue Callee = CLI.Callee;
639
640 bool IsPICCall = isPositionIndependent();
641
642 // PC-relative references to external symbols should go through $stub.
643 // If so, we need to prepare GlobalBaseReg first.
644 const TargetMachine &TM = DAG.getTarget();
645 const GlobalValue *GV = nullptr;
646 auto *CalleeG = dyn_cast<GlobalAddressSDNode>(Callee);
647 if (CalleeG)
648 GV = CalleeG->getGlobal();
649 bool Local = TM.shouldAssumeDSOLocal(GV);
650 bool UsePlt = !Local;
652
653 // Turn GlobalAddress/ExternalSymbol node into a value node
654 // containing the address of them here.
655 if (CalleeG) {
656 if (IsPICCall) {
657 if (UsePlt)
658 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF);
659 Callee = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, 0);
660 Callee = DAG.getNode(VEISD::GETFUNPLT, DL, PtrVT, Callee);
661 } else {
662 Callee = makeHiLoPair(Callee, VE::S_HI32, VE::S_LO32, DAG);
663 }
665 if (IsPICCall) {
666 if (UsePlt)
667 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF);
668 Callee = DAG.getTargetExternalSymbol(E->getSymbol(), PtrVT, 0);
669 Callee = DAG.getNode(VEISD::GETFUNPLT, DL, PtrVT, Callee);
670 } else {
671 Callee = makeHiLoPair(Callee, VE::S_HI32, VE::S_LO32, DAG);
672 }
673 }
674
675 RegsToPass.push_back(std::make_pair(VE::SX12, Callee));
676
677 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) {
678 CCValAssign &VA = ArgLocs[i];
679 SDValue Arg = CLI.OutVals[i];
680
681 // Promote the value if needed.
682 switch (VA.getLocInfo()) {
683 default:
684 llvm_unreachable("Unknown location info!");
686 break;
688 Arg = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Arg);
689 break;
691 Arg = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Arg);
692 break;
694 Arg = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Arg);
695 break;
696 case CCValAssign::BCvt: {
697 // Convert a float argument to i64 with padding.
698 // 63 31 0
699 // +------+------+
700 // | float| 0 |
701 // +------+------+
702 assert(VA.getLocVT() == MVT::i64);
703 assert(VA.getValVT() == MVT::f32);
705 DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i64), 0);
706 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32);
707 Arg = SDValue(DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL,
708 MVT::i64, Undef, Arg, Sub_f32),
709 0);
710 break;
711 }
712 }
713
714 if (VA.isRegLoc()) {
715 RegsToPass.push_back(std::make_pair(VA.getLocReg(), Arg));
716 if (!UseBoth)
717 continue;
718 VA = ArgLocs2[i];
719 }
720
721 assert(VA.isMemLoc());
722
723 // Create a store off the stack pointer for this argument.
724 SDValue StackPtr = DAG.getRegister(VE::SX11, PtrVT);
725 // The argument area starts at %fp/%sp + the size of reserved area.
726 SDValue PtrOff =
727 DAG.getIntPtrConstant(VA.getLocMemOffset() + ArgsBaseOffset, DL);
728 PtrOff = DAG.getNode(ISD::ADD, DL, PtrVT, StackPtr, PtrOff);
729 MemOpChains.push_back(
730 DAG.getStore(Chain, DL, Arg, PtrOff, MachinePointerInfo()));
731 }
732
733 // Emit all stores, make sure they occur before the call.
734 if (!MemOpChains.empty())
735 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains);
736
737 // Build a sequence of CopyToReg nodes glued together with token chain and
738 // glue operands which copy the outgoing args into registers. The InGlue is
739 // necessary since all emitted instructions must be stuck together in order
740 // to pass the live physical registers.
741 SDValue InGlue;
742 for (const auto &[Reg, N] : RegsToPass) {
743 Chain = DAG.getCopyToReg(Chain, DL, Reg, N, InGlue);
744 InGlue = Chain.getValue(1);
745 }
746
747 // Build the operands for the call instruction itself.
749 Ops.push_back(Chain);
750 for (const auto &[Reg, N] : RegsToPass)
751 Ops.push_back(DAG.getRegister(Reg, N.getValueType()));
752
753 // Add a register mask operand representing the call-preserved registers.
754 const VERegisterInfo *TRI = Subtarget->getRegisterInfo();
755 const uint32_t *Mask =
756 TRI->getCallPreservedMask(DAG.getMachineFunction(), CLI.CallConv);
757 assert(Mask && "Missing call preserved mask for calling convention");
758 Ops.push_back(DAG.getRegisterMask(Mask));
759
760 // Make sure the CopyToReg nodes are glued to the call instruction which
761 // consumes the registers.
762 if (InGlue.getNode())
763 Ops.push_back(InGlue);
764
765 // Now the call itself.
766 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
767 Chain = DAG.getNode(VEISD::CALL, DL, NodeTys, Ops);
768 InGlue = Chain.getValue(1);
769
770 // Revert the stack pointer immediately after the call.
771 Chain = DAG.getCALLSEQ_END(Chain, ArgsSize, 0, InGlue, DL);
772 InGlue = Chain.getValue(1);
773
774 // Now extract the return values. This is more or less the same as
775 // LowerFormalArguments.
776
777 // Assign locations to each value returned by this call.
779 CCState RVInfo(CLI.CallConv, CLI.IsVarArg, DAG.getMachineFunction(), RVLocs,
780 *DAG.getContext());
781
782 // Set inreg flag manually for codegen generated library calls that
783 // return float.
784 if (CLI.Ins.size() == 1 && CLI.Ins[0].VT == MVT::f32 && !CLI.CB)
785 CLI.Ins[0].Flags.setInReg();
786
787 RVInfo.AnalyzeCallResult(CLI.Ins, getReturnCC(CLI.CallConv));
788
789 // Copy all of the result registers out of their specified physreg.
790 for (unsigned i = 0; i != RVLocs.size(); ++i) {
791 CCValAssign &VA = RVLocs[i];
792 assert(!VA.needsCustom() && "Unexpected custom lowering");
793 Register Reg = VA.getLocReg();
794
795 // When returning 'inreg {i32, i32 }', two consecutive i32 arguments can
796 // reside in the same register in the high and low bits. Reuse the
797 // CopyFromReg previous node to avoid duplicate copies.
798 SDValue RV;
799 if (RegisterSDNode *SrcReg = dyn_cast<RegisterSDNode>(Chain.getOperand(1)))
800 if (SrcReg->getReg() == Reg && Chain->getOpcode() == ISD::CopyFromReg)
801 RV = Chain.getValue(0);
802
803 // But usually we'll create a new CopyFromReg for a different register.
804 if (!RV.getNode()) {
805 RV = DAG.getCopyFromReg(Chain, DL, Reg, RVLocs[i].getLocVT(), InGlue);
806 Chain = RV.getValue(1);
807 InGlue = Chain.getValue(2);
808 }
809
810 // The callee promoted the return value, so insert an Assert?ext SDNode so
811 // we won't promote the value again in this function.
812 switch (VA.getLocInfo()) {
814 RV = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), RV,
815 DAG.getValueType(VA.getValVT()));
816 break;
818 RV = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), RV,
819 DAG.getValueType(VA.getValVT()));
820 break;
821 case CCValAssign::BCvt: {
822 // Extract a float return value from i64 with padding.
823 // 63 31 0
824 // +------+------+
825 // | float| 0 |
826 // +------+------+
827 assert(VA.getLocVT() == MVT::i64);
828 assert(VA.getValVT() == MVT::f32);
829 SDValue Sub_f32 = DAG.getTargetConstant(VE::sub_f32, DL, MVT::i32);
830 RV = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
831 MVT::f32, RV, Sub_f32),
832 0);
833 break;
834 }
835 default:
836 break;
837 }
838
839 // Truncate the register down to the return value type.
840 if (VA.isExtInLoc())
841 RV = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), RV);
842
843 InVals.push_back(RV);
844 }
845
846 return Chain;
847}
848
850 const GlobalAddressSDNode *GA) const {
851 // VE uses 64 bit addressing, so we need multiple instructions to generate
852 // an address. Folding address with offset increases the number of
853 // instructions, so that we disable it here. Offsets will be folded in
854 // the DAG combine later if it worth to do so.
855 return false;
856}
857
858/// isFPImmLegal - Returns true if the target can instruction select the
859/// specified FP immediate natively. If false, the legalizer will
860/// materialize the FP immediate as a load from a constant pool.
862 bool ForCodeSize) const {
863 return VT == MVT::f32 || VT == MVT::f64;
864}
865
866/// Determine if the target supports unaligned memory accesses.
867///
868/// This function returns true if the target allows unaligned memory accesses
869/// of the specified type in the given address space. If true, it also returns
870/// whether the unaligned memory access is "fast" in the last argument by
871/// reference. This is used, for example, in situations where an array
872/// copy/move/set is converted to a sequence of store operations. Its use
873/// helps to ensure that such replacements don't generate code that causes an
874/// alignment error (trap) on the target machine.
876 unsigned AddrSpace,
877 Align A,
879 unsigned *Fast) const {
880 if (Fast) {
881 // It's fast anytime on VE
882 *Fast = 1;
883 }
884 return true;
885}
886
888 const VESubtarget &STI)
889 : TargetLowering(TM, STI), Subtarget(&STI) {
890 // Instructions which use registers as conditionals examine all the
891 // bits (as does the pseudo SELECT_CC expansion). I don't think it
892 // matters much whether it's ZeroOrOneBooleanContent, or
893 // ZeroOrNegativeOneBooleanContent, so, arbitrarily choose the
894 // former.
897
898 initRegisterClasses();
899 initSPUActions();
900 initVPUActions();
901
903
904 // We have target-specific dag combine patterns for the following nodes:
908
909 // Set function alignment to 16 bytes
911
912 // VE stores all argument by 8 bytes alignment
914
915 computeRegisterProperties(Subtarget->getRegisterInfo());
916}
917
919 LLVMContext &Context, EVT VT) const {
920 if (VT.isVector())
921 return VT.changeVectorElementType(Context, MVT::i1);
922 return MVT::i32;
923}
924
925// Convert to a target node and set target flags.
927 SelectionDAG &DAG) const {
929 return DAG.getTargetGlobalAddress(GA->getGlobal(), SDLoc(GA),
930 GA->getValueType(0), GA->getOffset(), TF);
931
933 return DAG.getTargetBlockAddress(BA->getBlockAddress(), Op.getValueType(),
934 0, TF);
935
937 return DAG.getTargetConstantPool(CP->getConstVal(), CP->getValueType(0),
938 CP->getAlign(), CP->getOffset(), TF);
939
941 return DAG.getTargetExternalSymbol(ES->getSymbol(), ES->getValueType(0),
942 TF);
943
945 return DAG.getTargetJumpTable(JT->getIndex(), JT->getValueType(0), TF);
946
947 llvm_unreachable("Unhandled address SDNode");
948}
949
950// Split Op into high and low parts according to HiTF and LoTF.
951// Return an ADD node combining the parts.
952SDValue VETargetLowering::makeHiLoPair(SDValue Op, unsigned HiTF, unsigned LoTF,
953 SelectionDAG &DAG) const {
954 SDLoc DL(Op);
955 EVT VT = Op.getValueType();
956 SDValue Hi = DAG.getNode(VEISD::Hi, DL, VT, withTargetFlags(Op, HiTF, DAG));
957 SDValue Lo = DAG.getNode(VEISD::Lo, DL, VT, withTargetFlags(Op, LoTF, DAG));
958 return DAG.getNode(ISD::ADD, DL, VT, Hi, Lo);
959}
960
961// Build SDNodes for producing an address from a GlobalAddress, ConstantPool,
962// or ExternalSymbol SDNode.
964 SDLoc DL(Op);
965 EVT PtrVT = Op.getValueType();
966
967 // Handle PIC mode first. VE needs a got load for every variable!
968 if (isPositionIndependent()) {
969 auto GlobalN = dyn_cast<GlobalAddressSDNode>(Op);
970
972 (GlobalN && GlobalN->getGlobal()->hasLocalLinkage())) {
973 // Create following instructions for local linkage PIC code.
974 // lea %reg, label@gotoff_lo
975 // and %reg, %reg, (32)0
976 // lea.sl %reg, label@gotoff_hi(%reg, %got)
977 SDValue HiLo =
979 SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrVT);
980 return DAG.getNode(ISD::ADD, DL, PtrVT, GlobalBase, HiLo);
981 }
982 // Create following instructions for not local linkage PIC code.
983 // lea %reg, label@got_lo
984 // and %reg, %reg, (32)0
985 // lea.sl %reg, label@got_hi(%reg)
986 // ld %reg, (%reg, %got)
988 SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrVT);
989 SDValue AbsAddr = DAG.getNode(ISD::ADD, DL, PtrVT, GlobalBase, HiLo);
990 return DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), AbsAddr,
992 }
993
994 // This is one of the absolute code models.
995 switch (getTargetMachine().getCodeModel()) {
996 default:
997 llvm_unreachable("Unsupported absolute code model");
998 case CodeModel::Small:
1000 case CodeModel::Large:
1001 // abs64.
1002 return makeHiLoPair(Op, VE::S_HI32, VE::S_LO32, DAG);
1003 }
1004}
1005
1006/// Custom Lower {
1007
1008// The mappings for emitLeading/TrailingFence for VE is designed by following
1009// http://www.cl.cam.ac.uk/~pes20/cpp/cpp0xmappings.html
1011 Instruction *Inst,
1012 AtomicOrdering Ord) const {
1013 switch (Ord) {
1016 llvm_unreachable("Invalid fence: unordered/non-atomic");
1019 return nullptr; // Nothing to do
1022 return Builder.CreateFence(AtomicOrdering::Release);
1024 if (!Inst->hasAtomicStore())
1025 return nullptr; // Nothing to do
1026 return Builder.CreateFence(AtomicOrdering::SequentiallyConsistent);
1027 }
1028 llvm_unreachable("Unknown fence ordering in emitLeadingFence");
1029}
1030
1032 Instruction *Inst,
1033 AtomicOrdering Ord) const {
1034 switch (Ord) {
1037 llvm_unreachable("Invalid fence: unordered/not-atomic");
1040 return nullptr; // Nothing to do
1043 return Builder.CreateFence(AtomicOrdering::Acquire);
1045 return Builder.CreateFence(AtomicOrdering::SequentiallyConsistent);
1046 }
1047 llvm_unreachable("Unknown fence ordering in emitTrailingFence");
1048}
1049
1051 SelectionDAG &DAG) const {
1052 SDLoc DL(Op);
1053 AtomicOrdering FenceOrdering =
1054 static_cast<AtomicOrdering>(Op.getConstantOperandVal(1));
1055 SyncScope::ID FenceSSID =
1056 static_cast<SyncScope::ID>(Op.getConstantOperandVal(2));
1057
1058 // VE uses Release consistency, so need a fence instruction if it is a
1059 // cross-thread fence.
1060 if (FenceSSID == SyncScope::System) {
1061 switch (FenceOrdering) {
1065 // No need to generate fencem instruction here.
1066 break;
1068 // Generate "fencem 2" as acquire fence.
1069 return SDValue(DAG.getMachineNode(VE::FENCEM, DL, MVT::Other,
1070 DAG.getTargetConstant(2, DL, MVT::i32),
1071 Op.getOperand(0)),
1072 0);
1074 // Generate "fencem 1" as release fence.
1075 return SDValue(DAG.getMachineNode(VE::FENCEM, DL, MVT::Other,
1076 DAG.getTargetConstant(1, DL, MVT::i32),
1077 Op.getOperand(0)),
1078 0);
1081 // Generate "fencem 3" as acq_rel and seq_cst fence.
1082 // FIXME: "fencem 3" doesn't wait for PCIe deveices accesses,
1083 // so seq_cst may require more instruction for them.
1084 return SDValue(DAG.getMachineNode(VE::FENCEM, DL, MVT::Other,
1085 DAG.getTargetConstant(3, DL, MVT::i32),
1086 Op.getOperand(0)),
1087 0);
1088 }
1089 }
1090
1091 // MEMBARRIER is a compiler barrier; it codegens to a no-op.
1092 return DAG.getNode(ISD::MEMBARRIER, DL, MVT::Other, Op.getOperand(0));
1093}
1094
1097 // We have TS1AM implementation for i8/i16/i32/i64, so use it.
1098 if (AI->getOperation() == AtomicRMWInst::Xchg) {
1100 }
1101 // FIXME: Support "ATMAM" instruction for LOAD_ADD/SUB/AND/OR.
1102
1103 // Otherwise, expand it using compare and exchange instruction to not call
1104 // __sync_fetch_and_* functions.
1106}
1107
1109 SDValue &Bits) {
1110 SDLoc DL(Op);
1112 SDValue Ptr = N->getOperand(1);
1113 SDValue Val = N->getOperand(2);
1114 EVT PtrVT = Ptr.getValueType();
1115 bool Byte = N->getMemoryVT() == MVT::i8;
1116 // Remainder = AND Ptr, 3
1117 // Flag = 1 << Remainder ; If Byte is true (1 byte swap flag)
1118 // Flag = 3 << Remainder ; If Byte is false (2 bytes swap flag)
1119 // Bits = Remainder << 3
1120 // NewVal = Val << Bits
1121 SDValue Const3 = DAG.getConstant(3, DL, PtrVT);
1122 SDValue Remainder = DAG.getNode(ISD::AND, DL, PtrVT, {Ptr, Const3});
1123 SDValue Mask = Byte ? DAG.getConstant(1, DL, MVT::i32)
1124 : DAG.getConstant(3, DL, MVT::i32);
1125 Flag = DAG.getNode(ISD::SHL, DL, MVT::i32, {Mask, Remainder});
1126 Bits = DAG.getNode(ISD::SHL, DL, PtrVT, {Remainder, Const3});
1127 return DAG.getNode(ISD::SHL, DL, Val.getValueType(), {Val, Bits});
1128}
1129
1131 SDValue Bits) {
1132 SDLoc DL(Op);
1133 EVT VT = Data.getValueType();
1134 bool Byte = cast<AtomicSDNode>(Op)->getMemoryVT() == MVT::i8;
1135 // NewData = Data >> Bits
1136 // Result = NewData & 0xff ; If Byte is true (1 byte)
1137 // Result = NewData & 0xffff ; If Byte is false (2 bytes)
1138
1139 SDValue NewData = DAG.getNode(ISD::SRL, DL, VT, Data, Bits);
1140 return DAG.getNode(ISD::AND, DL, VT,
1141 {NewData, DAG.getConstant(Byte ? 0xff : 0xffff, DL, VT)});
1142}
1143
1145 SelectionDAG &DAG) const {
1146 SDLoc DL(Op);
1148
1149 if (N->getMemoryVT() == MVT::i8) {
1150 // For i8, use "ts1am"
1151 // Input:
1152 // ATOMIC_SWAP Ptr, Val, Order
1153 //
1154 // Output:
1155 // Remainder = AND Ptr, 3
1156 // Flag = 1 << Remainder ; 1 byte swap flag for TS1AM inst.
1157 // Bits = Remainder << 3
1158 // NewVal = Val << Bits
1159 //
1160 // Aligned = AND Ptr, -4
1161 // Data = TS1AM Aligned, Flag, NewVal
1162 //
1163 // NewData = Data >> Bits
1164 // Result = NewData & 0xff ; 1 byte result
1165 SDValue Flag;
1166 SDValue Bits;
1167 SDValue NewVal = prepareTS1AM(Op, DAG, Flag, Bits);
1168
1169 SDValue Ptr = N->getOperand(1);
1171 DAG.getNode(ISD::AND, DL, Ptr.getValueType(),
1172 {Ptr, DAG.getSignedConstant(-4, DL, MVT::i64)});
1173 SDValue TS1AM =
1174 DAG.getMemIntrinsicNode(VEISD::TS1AM, DL,
1175 DAG.getVTList(Op.getNode()->getValueType(0),
1176 Op.getNode()->getValueType(1)),
1177 {N->getChain(), Aligned, Flag, NewVal},
1178 N->getMemoryVT(), N->getMemOperand());
1179
1180 SDValue Result = finalizeTS1AM(Op, DAG, TS1AM, Bits);
1181 SDValue Chain = TS1AM.getValue(1);
1182 return DAG.getMergeValues({Result, Chain}, DL);
1183 }
1184 if (N->getMemoryVT() == MVT::i16) {
1185 // For i16, use "ts1am"
1186 SDValue Flag;
1187 SDValue Bits;
1188 SDValue NewVal = prepareTS1AM(Op, DAG, Flag, Bits);
1189
1190 SDValue Ptr = N->getOperand(1);
1192 DAG.getNode(ISD::AND, DL, Ptr.getValueType(),
1193 {Ptr, DAG.getSignedConstant(-4, DL, MVT::i64)});
1194 SDValue TS1AM =
1195 DAG.getMemIntrinsicNode(VEISD::TS1AM, DL,
1196 DAG.getVTList(Op.getNode()->getValueType(0),
1197 Op.getNode()->getValueType(1)),
1198 {N->getChain(), Aligned, Flag, NewVal},
1199 N->getMemoryVT(), N->getMemOperand());
1200
1201 SDValue Result = finalizeTS1AM(Op, DAG, TS1AM, Bits);
1202 SDValue Chain = TS1AM.getValue(1);
1203 return DAG.getMergeValues({Result, Chain}, DL);
1204 }
1205 // Otherwise, let llvm legalize it.
1206 return Op;
1207}
1208
1213
1218
1223
1224SDValue
1226 SelectionDAG &DAG) const {
1227 SDLoc DL(Op);
1228
1229 // Generate the following code:
1230 // t1: ch,glue = callseq_start t0, 0, 0
1231 // t2: i64,ch,glue = VEISD::GETTLSADDR t1, label, t1:1
1232 // t3: ch,glue = callseq_end t2, 0, 0, t2:2
1233 // t4: i64,ch,glue = CopyFromReg t3, Register:i64 $sx0, t3:1
1234 SDValue Label = withTargetFlags(Op, 0, DAG);
1235 EVT PtrVT = Op.getValueType();
1236
1237 // Lowering the machine isd will make sure everything is in the right
1238 // location.
1239 SDValue Chain = DAG.getEntryNode();
1240 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
1241 const uint32_t *Mask = Subtarget->getRegisterInfo()->getCallPreservedMask(
1243 Chain = DAG.getCALLSEQ_START(Chain, 64, 0, DL);
1244 SDValue Args[] = {Chain, Label, DAG.getRegisterMask(Mask), Chain.getValue(1)};
1245 Chain = DAG.getNode(VEISD::GETTLSADDR, DL, NodeTys, Args);
1246 Chain = DAG.getCALLSEQ_END(Chain, 64, 0, Chain.getValue(1), DL);
1247 Chain = DAG.getCopyFromReg(Chain, DL, VE::SX0, PtrVT, Chain.getValue(1));
1248
1249 // GETTLSADDR will be codegen'ed as call. Inform MFI that function has calls.
1251 MFI.setHasCalls(true);
1252
1253 // Also generate code to prepare a GOT register if it is PIC.
1254 if (isPositionIndependent()) {
1256 Subtarget->getInstrInfo()->getGlobalBaseReg(&MF);
1257 }
1258
1259 return Chain;
1260}
1261
1263 SelectionDAG &DAG) const {
1264 // The current implementation of nld (2.26) doesn't allow local exec model
1265 // code described in VE-tls_v1.1.pdf (*1) as its input. Instead, we always
1266 // generate the general dynamic model code sequence.
1267 //
1268 // *1: https://www.nec.com/en/global/prod/hpc/aurora/document/VE-tls_v1.1.pdf
1269 return lowerToTLSGeneralDynamicModel(Op, DAG);
1270}
1271
1275
1276// Lower a f128 load into two f64 loads.
1278 SDLoc DL(Op);
1279 LoadSDNode *LdNode = dyn_cast<LoadSDNode>(Op.getNode());
1280 assert(LdNode && LdNode->getOffset().isUndef() && "Unexpected node type");
1281 Align Alignment = LdNode->getAlign();
1282 if (Alignment > 8)
1283 Alignment = Align(8);
1284
1285 SDValue Lo64 =
1286 DAG.getLoad(MVT::f64, DL, LdNode->getChain(), LdNode->getBasePtr(),
1287 LdNode->getPointerInfo(), Alignment,
1290 EVT AddrVT = LdNode->getBasePtr().getValueType();
1291 SDValue HiPtr = DAG.getNode(ISD::ADD, DL, AddrVT, LdNode->getBasePtr(),
1292 DAG.getConstant(8, DL, AddrVT));
1293 SDValue Hi64 =
1294 DAG.getLoad(MVT::f64, DL, LdNode->getChain(), HiPtr,
1295 LdNode->getPointerInfo(), Alignment,
1298
1299 SDValue SubRegEven = DAG.getTargetConstant(VE::sub_even, DL, MVT::i32);
1300 SDValue SubRegOdd = DAG.getTargetConstant(VE::sub_odd, DL, MVT::i32);
1301
1302 // VE stores Hi64 to 8(addr) and Lo64 to 0(addr)
1303 SDNode *InFP128 =
1304 DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::f128);
1305 InFP128 = DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, MVT::f128,
1306 SDValue(InFP128, 0), Hi64, SubRegEven);
1307 InFP128 = DAG.getMachineNode(TargetOpcode::INSERT_SUBREG, DL, MVT::f128,
1308 SDValue(InFP128, 0), Lo64, SubRegOdd);
1309 SDValue OutChains[2] = {SDValue(Lo64.getNode(), 1),
1310 SDValue(Hi64.getNode(), 1)};
1311 SDValue OutChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
1312 SDValue Ops[2] = {SDValue(InFP128, 0), OutChain};
1313 return DAG.getMergeValues(Ops, DL);
1314}
1315
1316// Lower a vXi1 load into following instructions
1317// LDrii %1, (,%addr)
1318// LVMxir %vm, 0, %1
1319// LDrii %2, 8(,%addr)
1320// LVMxir %vm, 0, %2
1321// ...
1323 SDLoc DL(Op);
1324 LoadSDNode *LdNode = dyn_cast<LoadSDNode>(Op.getNode());
1325 assert(LdNode && LdNode->getOffset().isUndef() && "Unexpected node type");
1326
1327 SDValue BasePtr = LdNode->getBasePtr();
1328 Align Alignment = LdNode->getAlign();
1329 if (Alignment > 8)
1330 Alignment = Align(8);
1331
1332 EVT AddrVT = BasePtr.getValueType();
1333 EVT MemVT = LdNode->getMemoryVT();
1334 if (MemVT == MVT::v256i1 || MemVT == MVT::v4i64) {
1335 SDValue OutChains[4];
1336 SDNode *VM = DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MemVT);
1337 for (int i = 0; i < 4; ++i) {
1338 // Generate load dag and prepare chains.
1339 SDValue Addr = DAG.getNode(ISD::ADD, DL, AddrVT, BasePtr,
1340 DAG.getConstant(8 * i, DL, AddrVT));
1341 SDValue Val =
1342 DAG.getLoad(MVT::i64, DL, LdNode->getChain(), Addr,
1343 LdNode->getPointerInfo(), Alignment,
1346 OutChains[i] = SDValue(Val.getNode(), 1);
1347
1348 VM = DAG.getMachineNode(VE::LVMir_m, DL, MVT::i64,
1349 DAG.getTargetConstant(i, DL, MVT::i64), Val,
1350 SDValue(VM, 0));
1351 }
1352 SDValue OutChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
1353 SDValue Ops[2] = {SDValue(VM, 0), OutChain};
1354 return DAG.getMergeValues(Ops, DL);
1355 } else if (MemVT == MVT::v512i1 || MemVT == MVT::v8i64) {
1356 SDValue OutChains[8];
1357 SDNode *VM = DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MemVT);
1358 for (int i = 0; i < 8; ++i) {
1359 // Generate load dag and prepare chains.
1360 SDValue Addr = DAG.getNode(ISD::ADD, DL, AddrVT, BasePtr,
1361 DAG.getConstant(8 * i, DL, AddrVT));
1362 SDValue Val =
1363 DAG.getLoad(MVT::i64, DL, LdNode->getChain(), Addr,
1364 LdNode->getPointerInfo(), Alignment,
1367 OutChains[i] = SDValue(Val.getNode(), 1);
1368
1369 VM = DAG.getMachineNode(VE::LVMyir_y, DL, MVT::i64,
1370 DAG.getTargetConstant(i, DL, MVT::i64), Val,
1371 SDValue(VM, 0));
1372 }
1373 SDValue OutChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
1374 SDValue Ops[2] = {SDValue(VM, 0), OutChain};
1375 return DAG.getMergeValues(Ops, DL);
1376 } else {
1377 // Otherwise, ask llvm to expand it.
1378 return SDValue();
1379 }
1380}
1381
1383 LoadSDNode *LdNode = cast<LoadSDNode>(Op.getNode());
1384 EVT MemVT = LdNode->getMemoryVT();
1385
1386 // If VPU is enabled, always expand non-mask vector loads to VVP
1387 if (Subtarget->enableVPU() && MemVT.isVector() && !isMaskType(MemVT))
1388 return lowerToVVP(Op, DAG);
1389
1390 SDValue BasePtr = LdNode->getBasePtr();
1391 if (isa<FrameIndexSDNode>(BasePtr.getNode())) {
1392 // Do not expand store instruction with frame index here because of
1393 // dependency problems. We expand it later in eliminateFrameIndex().
1394 return Op;
1395 }
1396
1397 if (MemVT == MVT::f128)
1398 return lowerLoadF128(Op, DAG);
1399 if (isMaskType(MemVT))
1400 return lowerLoadI1(Op, DAG);
1401
1402 return Op;
1403}
1404
1405// Lower a f128 store into two f64 stores.
1407 SDLoc DL(Op);
1408 StoreSDNode *StNode = dyn_cast<StoreSDNode>(Op.getNode());
1409 assert(StNode && StNode->getOffset().isUndef() && "Unexpected node type");
1410
1411 SDValue SubRegEven = DAG.getTargetConstant(VE::sub_even, DL, MVT::i32);
1412 SDValue SubRegOdd = DAG.getTargetConstant(VE::sub_odd, DL, MVT::i32);
1413
1414 SDNode *Hi64 = DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, MVT::i64,
1415 StNode->getValue(), SubRegEven);
1416 SDNode *Lo64 = DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, MVT::i64,
1417 StNode->getValue(), SubRegOdd);
1418
1419 Align Alignment = StNode->getAlign();
1420 if (Alignment > 8)
1421 Alignment = Align(8);
1422
1423 // VE stores Hi64 to 8(addr) and Lo64 to 0(addr)
1424 SDValue OutChains[2];
1425 OutChains[0] =
1426 DAG.getStore(StNode->getChain(), DL, SDValue(Lo64, 0),
1427 StNode->getBasePtr(), MachinePointerInfo(), Alignment,
1430 EVT AddrVT = StNode->getBasePtr().getValueType();
1431 SDValue HiPtr = DAG.getNode(ISD::ADD, DL, AddrVT, StNode->getBasePtr(),
1432 DAG.getConstant(8, DL, AddrVT));
1433 OutChains[1] =
1434 DAG.getStore(StNode->getChain(), DL, SDValue(Hi64, 0), HiPtr,
1435 MachinePointerInfo(), Alignment,
1438 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
1439}
1440
1441// Lower a vXi1 store into following instructions
1442// SVMi %1, %vm, 0
1443// STrii %1, (,%addr)
1444// SVMi %2, %vm, 1
1445// STrii %2, 8(,%addr)
1446// ...
1448 SDLoc DL(Op);
1449 StoreSDNode *StNode = dyn_cast<StoreSDNode>(Op.getNode());
1450 assert(StNode && StNode->getOffset().isUndef() && "Unexpected node type");
1451
1452 SDValue BasePtr = StNode->getBasePtr();
1453 Align Alignment = StNode->getAlign();
1454 if (Alignment > 8)
1455 Alignment = Align(8);
1456 EVT AddrVT = BasePtr.getValueType();
1457 EVT MemVT = StNode->getMemoryVT();
1458 if (MemVT == MVT::v256i1 || MemVT == MVT::v4i64) {
1459 SDValue OutChains[4];
1460 for (int i = 0; i < 4; ++i) {
1461 SDNode *V =
1462 DAG.getMachineNode(VE::SVMmi, DL, MVT::i64, StNode->getValue(),
1463 DAG.getTargetConstant(i, DL, MVT::i64));
1464 SDValue Addr = DAG.getNode(ISD::ADD, DL, AddrVT, BasePtr,
1465 DAG.getConstant(8 * i, DL, AddrVT));
1466 OutChains[i] =
1467 DAG.getStore(StNode->getChain(), DL, SDValue(V, 0), Addr,
1468 MachinePointerInfo(), Alignment,
1471 }
1472 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
1473 } else if (MemVT == MVT::v512i1 || MemVT == MVT::v8i64) {
1474 SDValue OutChains[8];
1475 for (int i = 0; i < 8; ++i) {
1476 SDNode *V =
1477 DAG.getMachineNode(VE::SVMyi, DL, MVT::i64, StNode->getValue(),
1478 DAG.getTargetConstant(i, DL, MVT::i64));
1479 SDValue Addr = DAG.getNode(ISD::ADD, DL, AddrVT, BasePtr,
1480 DAG.getConstant(8 * i, DL, AddrVT));
1481 OutChains[i] =
1482 DAG.getStore(StNode->getChain(), DL, SDValue(V, 0), Addr,
1483 MachinePointerInfo(), Alignment,
1486 }
1487 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
1488 } else {
1489 // Otherwise, ask llvm to expand it.
1490 return SDValue();
1491 }
1492}
1493
1495 StoreSDNode *StNode = cast<StoreSDNode>(Op.getNode());
1496 assert(StNode && StNode->getOffset().isUndef() && "Unexpected node type");
1497 EVT MemVT = StNode->getMemoryVT();
1498
1499 // If VPU is enabled, always expand non-mask vector stores to VVP
1500 if (Subtarget->enableVPU() && MemVT.isVector() && !isMaskType(MemVT))
1501 return lowerToVVP(Op, DAG);
1502
1503 SDValue BasePtr = StNode->getBasePtr();
1504 if (isa<FrameIndexSDNode>(BasePtr.getNode())) {
1505 // Do not expand store instruction with frame index here because of
1506 // dependency problems. We expand it later in eliminateFrameIndex().
1507 return Op;
1508 }
1509
1510 if (MemVT == MVT::f128)
1511 return lowerStoreF128(Op, DAG);
1512 if (isMaskType(MemVT))
1513 return lowerStoreI1(Op, DAG);
1514
1515 // Otherwise, ask llvm to expand it.
1516 return SDValue();
1517}
1518
1522 auto PtrVT = getPointerTy(DAG.getDataLayout());
1523
1524 // Need frame address to find the address of VarArgsFrameIndex.
1526
1527 // vastart just stores the address of the VarArgsFrameIndex slot into the
1528 // memory location argument.
1529 SDLoc DL(Op);
1530 SDValue Offset =
1531 DAG.getNode(ISD::ADD, DL, PtrVT, DAG.getRegister(VE::SX9, PtrVT),
1532 DAG.getIntPtrConstant(FuncInfo->getVarArgsFrameOffset(), DL));
1533 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
1534 return DAG.getStore(Op.getOperand(0), DL, Offset, Op.getOperand(1),
1535 MachinePointerInfo(SV));
1536}
1537
1539 SDNode *Node = Op.getNode();
1540 EVT VT = Node->getValueType(0);
1541 SDValue InChain = Node->getOperand(0);
1542 SDValue VAListPtr = Node->getOperand(1);
1543 EVT PtrVT = VAListPtr.getValueType();
1544 const Value *SV = cast<SrcValueSDNode>(Node->getOperand(2))->getValue();
1545 SDLoc DL(Node);
1546 SDValue VAList =
1547 DAG.getLoad(PtrVT, DL, InChain, VAListPtr, MachinePointerInfo(SV));
1548 SDValue Chain = VAList.getValue(1);
1549 SDValue NextPtr;
1550
1551 if (VT == MVT::f128) {
1552 // VE f128 values must be stored with 16 bytes alignment. We don't
1553 // know the actual alignment of VAList, so we take alignment of it
1554 // dynamically.
1555 int Align = 16;
1556 VAList = DAG.getNode(ISD::ADD, DL, PtrVT, VAList,
1557 DAG.getConstant(Align - 1, DL, PtrVT));
1558 VAList = DAG.getNode(ISD::AND, DL, PtrVT, VAList,
1559 DAG.getSignedConstant(-Align, DL, PtrVT));
1560 // Increment the pointer, VAList, by 16 to the next vaarg.
1561 NextPtr =
1562 DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(16, DL));
1563 } else if (VT == MVT::f32) {
1564 // float --> need special handling like below.
1565 // 0 4
1566 // +------+------+
1567 // | empty| float|
1568 // +------+------+
1569 // Increment the pointer, VAList, by 8 to the next vaarg.
1570 NextPtr =
1571 DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(8, DL));
1572 // Then, adjust VAList.
1573 unsigned InternalOffset = 4;
1574 VAList = DAG.getNode(ISD::ADD, DL, PtrVT, VAList,
1575 DAG.getConstant(InternalOffset, DL, PtrVT));
1576 } else {
1577 // Increment the pointer, VAList, by 8 to the next vaarg.
1578 NextPtr =
1579 DAG.getNode(ISD::ADD, DL, PtrVT, VAList, DAG.getIntPtrConstant(8, DL));
1580 }
1581
1582 // Store the incremented VAList to the legalized pointer.
1583 InChain = DAG.getStore(Chain, DL, NextPtr, VAListPtr, MachinePointerInfo(SV));
1584
1585 // Load the actual argument out of the pointer VAList.
1586 // We can't count on greater alignment than the word size.
1587 return DAG.getLoad(
1588 VT, DL, InChain, VAList, MachinePointerInfo(),
1589 Align(std::min(PtrVT.getSizeInBits(), VT.getSizeInBits()) / 8));
1590}
1591
1593 SelectionDAG &DAG) const {
1594 // Generate following code.
1595 // (void)__llvm_grow_stack(size);
1596 // ret = GETSTACKTOP; // pseudo instruction
1597 SDLoc DL(Op);
1598
1599 // Get the inputs.
1600 SDNode *Node = Op.getNode();
1601 SDValue Chain = Op.getOperand(0);
1602 SDValue Size = Op.getOperand(1);
1603 MaybeAlign Alignment(Op.getConstantOperandVal(2));
1604 EVT VT = Node->getValueType(0);
1605
1606 // Chain the dynamic stack allocation so that it doesn't modify the stack
1607 // pointer when other instructions are using the stack.
1608 Chain = DAG.getCALLSEQ_START(Chain, 0, 0, DL);
1609
1610 const TargetFrameLowering &TFI = *Subtarget->getFrameLowering();
1611 Align StackAlign = TFI.getStackAlign();
1612 bool NeedsAlign = Alignment.valueOrOne() > StackAlign;
1613
1614 // Prepare arguments
1616 Args.emplace_back(Size, Size.getValueType().getTypeForEVT(*DAG.getContext()));
1617 if (NeedsAlign) {
1618 SDValue Align = DAG.getConstant(~(Alignment->value() - 1ULL), DL, VT);
1619 Args.emplace_back(Align,
1620 Align.getValueType().getTypeForEVT(*DAG.getContext()));
1621 }
1622 Type *RetTy = Type::getVoidTy(*DAG.getContext());
1623
1624 EVT PtrVT = Op.getValueType();
1625 SDValue Callee;
1626 if (NeedsAlign) {
1627 Callee = DAG.getTargetExternalSymbol("__ve_grow_stack_align", PtrVT, 0);
1628 } else {
1629 Callee = DAG.getTargetExternalSymbol("__ve_grow_stack", PtrVT, 0);
1630 }
1631
1633 CLI.setDebugLoc(DL)
1634 .setChain(Chain)
1635 .setCallee(CallingConv::PreserveAll, RetTy, Callee, std::move(Args))
1636 .setDiscardResult(true);
1637 std::pair<SDValue, SDValue> pair = LowerCallTo(CLI);
1638 Chain = pair.second;
1639 SDValue Result = DAG.getNode(VEISD::GETSTACKTOP, DL, VT, Chain);
1640 if (NeedsAlign) {
1641 Result = DAG.getNode(ISD::ADD, DL, VT, Result,
1642 DAG.getConstant((Alignment->value() - 1ULL), DL, VT));
1643 Result = DAG.getNode(ISD::AND, DL, VT, Result,
1644 DAG.getConstant(~(Alignment->value() - 1ULL), DL, VT));
1645 }
1646 // Chain = Result.getValue(1);
1647 Chain = DAG.getCALLSEQ_END(Chain, 0, 0, SDValue(), DL);
1648
1649 SDValue Ops[2] = {Result, Chain};
1650 return DAG.getMergeValues(Ops, DL);
1651}
1652
1654 SelectionDAG &DAG) const {
1655 SDLoc DL(Op);
1656 return DAG.getNode(VEISD::EH_SJLJ_LONGJMP, DL, MVT::Other, Op.getOperand(0),
1657 Op.getOperand(1));
1658}
1659
1661 SelectionDAG &DAG) const {
1662 SDLoc DL(Op);
1663 return DAG.getNode(VEISD::EH_SJLJ_SETJMP, DL,
1664 DAG.getVTList(MVT::i32, MVT::Other), Op.getOperand(0),
1665 Op.getOperand(1));
1666}
1667
1669 SelectionDAG &DAG) const {
1670 SDLoc DL(Op);
1671 return DAG.getNode(VEISD::EH_SJLJ_SETUP_DISPATCH, DL, MVT::Other,
1672 Op.getOperand(0));
1673}
1674
1676 const VETargetLowering &TLI,
1677 const VESubtarget *Subtarget) {
1678 SDLoc DL(Op);
1680 EVT PtrVT = TLI.getPointerTy(MF.getDataLayout());
1681
1682 MachineFrameInfo &MFI = MF.getFrameInfo();
1683 MFI.setFrameAddressIsTaken(true);
1684
1685 unsigned Depth = Op.getConstantOperandVal(0);
1686 const VERegisterInfo *RegInfo = Subtarget->getRegisterInfo();
1687 Register FrameReg = RegInfo->getFrameRegister(MF);
1688 SDValue FrameAddr =
1689 DAG.getCopyFromReg(DAG.getEntryNode(), DL, FrameReg, PtrVT);
1690 while (Depth--)
1691 FrameAddr = DAG.getLoad(Op.getValueType(), DL, DAG.getEntryNode(),
1692 FrameAddr, MachinePointerInfo());
1693 return FrameAddr;
1694}
1695
1697 const VETargetLowering &TLI,
1698 const VESubtarget *Subtarget) {
1700 MachineFrameInfo &MFI = MF.getFrameInfo();
1701 MFI.setReturnAddressIsTaken(true);
1702
1703 SDValue FrameAddr = lowerFRAMEADDR(Op, DAG, TLI, Subtarget);
1704
1705 SDLoc DL(Op);
1706 EVT VT = Op.getValueType();
1707 SDValue Offset = DAG.getConstant(8, DL, VT);
1708 return DAG.getLoad(VT, DL, DAG.getEntryNode(),
1709 DAG.getNode(ISD::ADD, DL, VT, FrameAddr, Offset),
1711}
1712
1714 SelectionDAG &DAG) const {
1715 SDLoc DL(Op);
1716 unsigned IntNo = Op.getConstantOperandVal(0);
1717 switch (IntNo) {
1718 default: // Don't custom lower most intrinsics.
1719 return SDValue();
1720 case Intrinsic::eh_sjlj_lsda: {
1722 MVT VT = Op.getSimpleValueType();
1723 const VETargetMachine *TM =
1724 static_cast<const VETargetMachine *>(&DAG.getTarget());
1725
1726 // Create GCC_except_tableXX string. The real symbol for that will be
1727 // generated in EHStreamer::emitExceptionTable() later. So, we just
1728 // borrow it's name here.
1729 TM->getStrList()->push_back(std::string(
1730 (Twine("GCC_except_table") + Twine(MF.getFunctionNumber())).str()));
1731 SDValue Addr =
1732 DAG.getTargetExternalSymbol(TM->getStrList()->back().c_str(), VT, 0);
1733 if (isPositionIndependent()) {
1735 SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, VT);
1736 return DAG.getNode(ISD::ADD, DL, VT, GlobalBase, Addr);
1737 }
1738 return makeHiLoPair(Addr, VE::S_HI32, VE::S_LO32, DAG);
1739 }
1740 }
1741}
1742
1743static bool getUniqueInsertion(SDNode *N, unsigned &UniqueIdx) {
1745 return false;
1746 const auto *BVN = cast<BuildVectorSDNode>(N);
1747
1748 // Find first non-undef insertion.
1749 unsigned Idx;
1750 for (Idx = 0; Idx < BVN->getNumOperands(); ++Idx) {
1751 auto ElemV = BVN->getOperand(Idx);
1752 if (!ElemV->isUndef())
1753 break;
1754 }
1755 // Catch the (hypothetical) all-undef case.
1756 if (Idx == BVN->getNumOperands())
1757 return false;
1758 // Remember insertion.
1759 UniqueIdx = Idx++;
1760 // Verify that all other insertions are undef.
1761 for (; Idx < BVN->getNumOperands(); ++Idx) {
1762 auto ElemV = BVN->getOperand(Idx);
1763 if (!ElemV->isUndef())
1764 return false;
1765 }
1766 return true;
1767}
1768
1770 if (auto *BuildVec = dyn_cast<BuildVectorSDNode>(N)) {
1771 return BuildVec->getSplatValue();
1772 }
1773 return SDValue();
1774}
1775
1777 SelectionDAG &DAG) const {
1778 VECustomDAG CDAG(DAG, Op);
1779 MVT ResultVT = Op.getSimpleValueType();
1780
1781 // If there is just one element, expand to INSERT_VECTOR_ELT.
1782 unsigned UniqueIdx;
1783 if (getUniqueInsertion(Op.getNode(), UniqueIdx)) {
1784 SDValue AccuV = CDAG.getUNDEF(Op.getValueType());
1785 auto ElemV = Op->getOperand(UniqueIdx);
1786 SDValue IdxV = CDAG.getConstant(UniqueIdx, MVT::i64);
1787 return CDAG.getNode(ISD::INSERT_VECTOR_ELT, ResultVT, {AccuV, ElemV, IdxV});
1788 }
1789
1790 // Else emit a broadcast.
1791 if (SDValue ScalarV = getSplatValue(Op.getNode())) {
1792 unsigned NumEls = ResultVT.getVectorNumElements();
1793 auto AVL = CDAG.getConstant(NumEls, MVT::i32);
1794 return CDAG.getBroadcast(ResultVT, ScalarV, AVL);
1795 }
1796
1797 // Expand
1798 return SDValue();
1799}
1800
1803 // Custom legalization on VVP_* and VEC_* opcodes is required to pack-legalize
1804 // these operations (transform nodes such that their AVL parameter refers to
1805 // packs of 64bit, instead of number of elements.
1806
1807 // Packing opcodes are created with a pack-legal AVL (LEGALAVL). No need to
1808 // re-visit them.
1809 if (isPackingSupportOpcode(Op.getOpcode()))
1810 return Legal;
1811
1812 // Custom lower to legalize AVL for packed mode.
1813 if (isVVPOrVEC(Op.getOpcode()))
1814 return Custom;
1815 return Legal;
1816}
1817
1819 LLVM_DEBUG(dbgs() << "::LowerOperation "; Op.dump(&DAG));
1820 unsigned Opcode = Op.getOpcode();
1821
1822 /// Scalar isel.
1823 switch (Opcode) {
1824 case ISD::ATOMIC_FENCE:
1825 return lowerATOMIC_FENCE(Op, DAG);
1826 case ISD::ATOMIC_SWAP:
1827 return lowerATOMIC_SWAP(Op, DAG);
1828 case ISD::BlockAddress:
1829 return lowerBlockAddress(Op, DAG);
1830 case ISD::ConstantPool:
1831 return lowerConstantPool(Op, DAG);
1833 return lowerDYNAMIC_STACKALLOC(Op, DAG);
1835 return lowerEH_SJLJ_LONGJMP(Op, DAG);
1837 return lowerEH_SJLJ_SETJMP(Op, DAG);
1839 return lowerEH_SJLJ_SETUP_DISPATCH(Op, DAG);
1840 case ISD::FRAMEADDR:
1841 return lowerFRAMEADDR(Op, DAG, *this, Subtarget);
1842 case ISD::GlobalAddress:
1843 return lowerGlobalAddress(Op, DAG);
1845 return lowerGlobalTLSAddress(Op, DAG);
1847 return lowerINTRINSIC_WO_CHAIN(Op, DAG);
1848 case ISD::JumpTable:
1849 return lowerJumpTable(Op, DAG);
1850 case ISD::LOAD:
1851 return lowerLOAD(Op, DAG);
1852 case ISD::RETURNADDR:
1853 return lowerRETURNADDR(Op, DAG, *this, Subtarget);
1854 case ISD::BUILD_VECTOR:
1855 return lowerBUILD_VECTOR(Op, DAG);
1856 case ISD::STORE:
1857 return lowerSTORE(Op, DAG);
1858 case ISD::VASTART:
1859 return lowerVASTART(Op, DAG);
1860 case ISD::VAARG:
1861 return lowerVAARG(Op, DAG);
1862
1864 return lowerINSERT_VECTOR_ELT(Op, DAG);
1866 return lowerEXTRACT_VECTOR_ELT(Op, DAG);
1867 }
1868
1869 /// Vector isel.
1870 if (ISD::isVPOpcode(Opcode))
1871 return lowerToVVP(Op, DAG);
1872
1873 switch (Opcode) {
1874 default:
1875 llvm_unreachable("Should not custom lower this!");
1876
1877 // Legalize the AVL of this internal node.
1878 case VEISD::VEC_BROADCAST:
1879#define ADD_VVP_OP(VVP_NAME, ...) case VEISD::VVP_NAME:
1880#include "VVPNodes.def"
1881 // AVL already legalized.
1882 if (getAnnotatedNodeAVL(Op).second)
1883 return Op;
1884 return legalizeInternalVectorOp(Op, DAG);
1885
1886 // Translate into a VEC_*/VVP_* layer operation.
1887 case ISD::MLOAD:
1888 case ISD::MSTORE:
1889#define ADD_VVP_OP(VVP_NAME, ISD_NAME) case ISD::ISD_NAME:
1890#include "VVPNodes.def"
1891 if (isMaskArithmetic(Op) && isPackedVectorType(Op.getValueType()))
1892 return splitMaskArithmetic(Op, DAG);
1893 return lowerToVVP(Op, DAG);
1894 }
1895}
1896/// } Custom Lower
1897
1900 SelectionDAG &DAG) const {
1901 switch (N->getOpcode()) {
1902 case ISD::ATOMIC_SWAP:
1903 // Let LLVM expand atomic swap instruction through LowerOperation.
1904 return;
1905 default:
1906 LLVM_DEBUG(N->dumpr(&DAG));
1907 llvm_unreachable("Do not know how to custom type legalize this operation!");
1908 }
1909}
1910
1911/// JumpTable for VE.
1912///
1913/// VE cannot generate relocatable symbol in jump table. VE cannot
1914/// generate expressions using symbols in both text segment and data
1915/// segment like below.
1916/// .4byte .LBB0_2-.LJTI0_0
1917/// So, we generate offset from the top of function like below as
1918/// a custom label.
1919/// .4byte .LBB0_2-<function name>
1920
1922 // Use custom label for PIC.
1925
1926 // Otherwise, use the normal jump table encoding heuristics.
1928}
1929
1931 const MachineJumpTableInfo *MJTI, const MachineBasicBlock *MBB,
1932 unsigned Uid, MCContext &Ctx) const {
1934
1935 // Generate custom label for PIC like below.
1936 // .4bytes .LBB0_2-<function name>
1937 const auto *Value = MCSymbolRefExpr::create(MBB->getSymbol(), Ctx);
1938 MCSymbol *Sym = Ctx.getOrCreateSymbol(MBB->getParent()->getName().data());
1939 const auto *Base = MCSymbolRefExpr::create(Sym, Ctx);
1940 return MCBinaryExpr::createSub(Value, Base, Ctx);
1941}
1942
1944 SelectionDAG &DAG) const {
1946 SDLoc DL(Table);
1948 assert(Function != nullptr);
1949 auto PtrTy = getPointerTy(DAG.getDataLayout(), Function->getAddressSpace());
1950
1951 // In the jump table, we have following values in PIC mode.
1952 // .4bytes .LBB0_2-<function name>
1953 // We need to add this value and the address of this function to generate
1954 // .LBB0_2 label correctly under PIC mode. So, we want to generate following
1955 // instructions:
1956 // lea %reg, fun@gotoff_lo
1957 // and %reg, %reg, (32)0
1958 // lea.sl %reg, fun@gotoff_hi(%reg, %got)
1959 // In order to do so, we need to genarate correctly marked DAG node using
1960 // makeHiLoPair.
1961 SDValue Op = DAG.getGlobalAddress(Function, DL, PtrTy);
1963 SDValue GlobalBase = DAG.getNode(VEISD::GLOBAL_BASE_REG, DL, PtrTy);
1964 return DAG.getNode(ISD::ADD, DL, PtrTy, GlobalBase, HiLo);
1965}
1966
1969 MachineBasicBlock *TargetBB,
1970 const DebugLoc &DL) const {
1971 MachineFunction *MF = MBB.getParent();
1972 MachineRegisterInfo &MRI = MF->getRegInfo();
1973 const VEInstrInfo *TII = Subtarget->getInstrInfo();
1974
1975 const TargetRegisterClass *RC = &VE::I64RegClass;
1976 Register Tmp1 = MRI.createVirtualRegister(RC);
1977 Register Tmp2 = MRI.createVirtualRegister(RC);
1978 Register Result = MRI.createVirtualRegister(RC);
1979
1980 if (isPositionIndependent()) {
1981 // Create following instructions for local linkage PIC code.
1982 // lea %Tmp1, TargetBB@gotoff_lo
1983 // and %Tmp2, %Tmp1, (32)0
1984 // lea.sl %Result, TargetBB@gotoff_hi(%Tmp2, %s15) ; %s15 is GOT
1985 BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
1986 .addImm(0)
1987 .addImm(0)
1988 .addMBB(TargetBB, VE::S_GOTOFF_LO32);
1989 BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
1990 .addReg(Tmp1, getKillRegState(true))
1991 .addImm(M0(32));
1992 BuildMI(MBB, I, DL, TII->get(VE::LEASLrri), Result)
1993 .addReg(VE::SX15)
1994 .addReg(Tmp2, getKillRegState(true))
1995 .addMBB(TargetBB, VE::S_GOTOFF_HI32);
1996 } else {
1997 // Create following instructions for non-PIC code.
1998 // lea %Tmp1, TargetBB@lo
1999 // and %Tmp2, %Tmp1, (32)0
2000 // lea.sl %Result, TargetBB@hi(%Tmp2)
2001 BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
2002 .addImm(0)
2003 .addImm(0)
2004 .addMBB(TargetBB, VE::S_LO32);
2005 BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
2006 .addReg(Tmp1, getKillRegState(true))
2007 .addImm(M0(32));
2008 BuildMI(MBB, I, DL, TII->get(VE::LEASLrii), Result)
2009 .addReg(Tmp2, getKillRegState(true))
2010 .addImm(0)
2011 .addMBB(TargetBB, VE::S_HI32);
2012 }
2013 return Result;
2014}
2015
2018 StringRef Symbol, const DebugLoc &DL,
2019 bool IsLocal = false,
2020 bool IsCall = false) const {
2021 MachineFunction *MF = MBB.getParent();
2022 MachineRegisterInfo &MRI = MF->getRegInfo();
2023 const VEInstrInfo *TII = Subtarget->getInstrInfo();
2024
2025 const TargetRegisterClass *RC = &VE::I64RegClass;
2026 Register Result = MRI.createVirtualRegister(RC);
2027
2028 if (isPositionIndependent()) {
2029 if (IsCall && !IsLocal) {
2030 // Create following instructions for non-local linkage PIC code function
2031 // calls. These instructions uses IC and magic number -24, so we expand
2032 // them in VEAsmPrinter.cpp from GETFUNPLT pseudo instruction.
2033 // lea %Reg, Symbol@plt_lo(-24)
2034 // and %Reg, %Reg, (32)0
2035 // sic %s16
2036 // lea.sl %Result, Symbol@plt_hi(%Reg, %s16) ; %s16 is PLT
2037 BuildMI(MBB, I, DL, TII->get(VE::GETFUNPLT), Result)
2038 .addExternalSymbol("abort");
2039 } else if (IsLocal) {
2040 Register Tmp1 = MRI.createVirtualRegister(RC);
2041 Register Tmp2 = MRI.createVirtualRegister(RC);
2042 // Create following instructions for local linkage PIC code.
2043 // lea %Tmp1, Symbol@gotoff_lo
2044 // and %Tmp2, %Tmp1, (32)0
2045 // lea.sl %Result, Symbol@gotoff_hi(%Tmp2, %s15) ; %s15 is GOT
2046 BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
2047 .addImm(0)
2048 .addImm(0)
2049 .addExternalSymbol(Symbol.data(), VE::S_GOTOFF_LO32);
2050 BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
2051 .addReg(Tmp1, getKillRegState(true))
2052 .addImm(M0(32));
2053 BuildMI(MBB, I, DL, TII->get(VE::LEASLrri), Result)
2054 .addReg(VE::SX15)
2055 .addReg(Tmp2, getKillRegState(true))
2056 .addExternalSymbol(Symbol.data(), VE::S_GOTOFF_HI32);
2057 } else {
2058 Register Tmp1 = MRI.createVirtualRegister(RC);
2059 Register Tmp2 = MRI.createVirtualRegister(RC);
2060 // Create following instructions for not local linkage PIC code.
2061 // lea %Tmp1, Symbol@got_lo
2062 // and %Tmp2, %Tmp1, (32)0
2063 // lea.sl %Tmp3, Symbol@gotoff_hi(%Tmp2, %s15) ; %s15 is GOT
2064 // ld %Result, 0(%Tmp3)
2065 Register Tmp3 = MRI.createVirtualRegister(RC);
2066 BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
2067 .addImm(0)
2068 .addImm(0)
2069 .addExternalSymbol(Symbol.data(), VE::S_GOT_LO32);
2070 BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
2071 .addReg(Tmp1, getKillRegState(true))
2072 .addImm(M0(32));
2073 BuildMI(MBB, I, DL, TII->get(VE::LEASLrri), Tmp3)
2074 .addReg(VE::SX15)
2075 .addReg(Tmp2, getKillRegState(true))
2076 .addExternalSymbol(Symbol.data(), VE::S_GOT_HI32);
2077 BuildMI(MBB, I, DL, TII->get(VE::LDrii), Result)
2078 .addReg(Tmp3, getKillRegState(true))
2079 .addImm(0)
2080 .addImm(0);
2081 }
2082 } else {
2083 Register Tmp1 = MRI.createVirtualRegister(RC);
2084 Register Tmp2 = MRI.createVirtualRegister(RC);
2085 // Create following instructions for non-PIC code.
2086 // lea %Tmp1, Symbol@lo
2087 // and %Tmp2, %Tmp1, (32)0
2088 // lea.sl %Result, Symbol@hi(%Tmp2)
2089 BuildMI(MBB, I, DL, TII->get(VE::LEAzii), Tmp1)
2090 .addImm(0)
2091 .addImm(0)
2092 .addExternalSymbol(Symbol.data(), VE::S_LO32);
2093 BuildMI(MBB, I, DL, TII->get(VE::ANDrm), Tmp2)
2094 .addReg(Tmp1, getKillRegState(true))
2095 .addImm(M0(32));
2096 BuildMI(MBB, I, DL, TII->get(VE::LEASLrii), Result)
2097 .addReg(Tmp2, getKillRegState(true))
2098 .addImm(0)
2099 .addExternalSymbol(Symbol.data(), VE::S_HI32);
2100 }
2101 return Result;
2102}
2103
2106 MachineBasicBlock *DispatchBB,
2107 int FI, int Offset) const {
2108 DebugLoc DL = MI.getDebugLoc();
2109 const VEInstrInfo *TII = Subtarget->getInstrInfo();
2110
2111 Register LabelReg =
2113
2114 // Store an address of DispatchBB to a given jmpbuf[1] where has next IC
2115 // referenced by longjmp (throw) later.
2116 MachineInstrBuilder MIB = BuildMI(*MBB, MI, DL, TII->get(VE::STrii));
2117 addFrameReference(MIB, FI, Offset); // jmpbuf[1]
2118 MIB.addReg(LabelReg, getKillRegState(true));
2119}
2120
2123 MachineBasicBlock *MBB) const {
2124 DebugLoc DL = MI.getDebugLoc();
2125 MachineFunction *MF = MBB->getParent();
2126 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
2127 const TargetRegisterInfo *TRI = Subtarget->getRegisterInfo();
2128 MachineRegisterInfo &MRI = MF->getRegInfo();
2129
2130 const BasicBlock *BB = MBB->getBasicBlock();
2131 MachineFunction::iterator I = ++MBB->getIterator();
2132
2133 // Memory Reference.
2134 SmallVector<MachineMemOperand *, 2> MMOs(MI.memoperands());
2135 Register BufReg = MI.getOperand(1).getReg();
2136
2137 Register DstReg;
2138
2139 DstReg = MI.getOperand(0).getReg();
2140 const TargetRegisterClass *RC = MRI.getRegClass(DstReg);
2141 assert(TRI->isTypeLegalForClass(*RC, MVT::i32) && "Invalid destination!");
2142 (void)TRI;
2143 Register MainDestReg = MRI.createVirtualRegister(RC);
2144 Register RestoreDestReg = MRI.createVirtualRegister(RC);
2145
2146 // For `v = call @llvm.eh.sjlj.setjmp(buf)`, we generate following
2147 // instructions. SP/FP must be saved in jmpbuf before `llvm.eh.sjlj.setjmp`.
2148 //
2149 // ThisMBB:
2150 // buf[3] = %s17 iff %s17 is used as BP
2151 // buf[1] = RestoreMBB as IC after longjmp
2152 // # SjLjSetup RestoreMBB
2153 //
2154 // MainMBB:
2155 // v_main = 0
2156 //
2157 // SinkMBB:
2158 // v = phi(v_main, MainMBB, v_restore, RestoreMBB)
2159 // ...
2160 //
2161 // RestoreMBB:
2162 // %s17 = buf[3] = iff %s17 is used as BP
2163 // v_restore = 1
2164 // goto SinkMBB
2165
2166 MachineBasicBlock *ThisMBB = MBB;
2167 MachineBasicBlock *MainMBB = MF->CreateMachineBasicBlock(BB);
2168 MachineBasicBlock *SinkMBB = MF->CreateMachineBasicBlock(BB);
2169 MachineBasicBlock *RestoreMBB = MF->CreateMachineBasicBlock(BB);
2170 MF->insert(I, MainMBB);
2171 MF->insert(I, SinkMBB);
2172 MF->push_back(RestoreMBB);
2173 RestoreMBB->setMachineBlockAddressTaken();
2174
2175 // Transfer the remainder of BB and its successor edges to SinkMBB.
2176 SinkMBB->splice(SinkMBB->begin(), MBB,
2177 std::next(MachineBasicBlock::iterator(MI)), MBB->end());
2179
2180 // ThisMBB:
2181 Register LabelReg =
2183
2184 // Store BP in buf[3] iff this function is using BP.
2185 const VEFrameLowering *TFI = Subtarget->getFrameLowering();
2186 if (TFI->hasBP(*MF)) {
2187 MachineInstrBuilder MIB = BuildMI(*MBB, MI, DL, TII->get(VE::STrii));
2188 MIB.addReg(BufReg);
2189 MIB.addImm(0);
2190 MIB.addImm(24);
2191 MIB.addReg(VE::SX17);
2192 MIB.setMemRefs(MMOs);
2193 }
2194
2195 // Store IP in buf[1].
2196 MachineInstrBuilder MIB = BuildMI(*MBB, MI, DL, TII->get(VE::STrii));
2197 MIB.add(MI.getOperand(1)); // we can preserve the kill flags here.
2198 MIB.addImm(0);
2199 MIB.addImm(8);
2200 MIB.addReg(LabelReg, getKillRegState(true));
2201 MIB.setMemRefs(MMOs);
2202
2203 // SP/FP are already stored in jmpbuf before `llvm.eh.sjlj.setjmp`.
2204
2205 // Insert setup.
2206 MIB =
2207 BuildMI(*ThisMBB, MI, DL, TII->get(VE::EH_SjLj_Setup)).addMBB(RestoreMBB);
2208
2209 const VERegisterInfo *RegInfo = Subtarget->getRegisterInfo();
2210 MIB.addRegMask(RegInfo->getNoPreservedMask());
2211 ThisMBB->addSuccessor(MainMBB);
2212 ThisMBB->addSuccessor(RestoreMBB);
2213
2214 // MainMBB:
2215 BuildMI(MainMBB, DL, TII->get(VE::LEAzii), MainDestReg)
2216 .addImm(0)
2217 .addImm(0)
2218 .addImm(0);
2219 MainMBB->addSuccessor(SinkMBB);
2220
2221 // SinkMBB:
2222 BuildMI(*SinkMBB, SinkMBB->begin(), DL, TII->get(VE::PHI), DstReg)
2223 .addReg(MainDestReg)
2224 .addMBB(MainMBB)
2225 .addReg(RestoreDestReg)
2226 .addMBB(RestoreMBB);
2227
2228 // RestoreMBB:
2229 // Restore BP from buf[3] iff this function is using BP. The address of
2230 // buf is in SX10.
2231 // FIXME: Better to not use SX10 here
2232 if (TFI->hasBP(*MF)) {
2234 BuildMI(RestoreMBB, DL, TII->get(VE::LDrii), VE::SX17);
2235 MIB.addReg(VE::SX10);
2236 MIB.addImm(0);
2237 MIB.addImm(24);
2238 MIB.setMemRefs(MMOs);
2239 }
2240 BuildMI(RestoreMBB, DL, TII->get(VE::LEAzii), RestoreDestReg)
2241 .addImm(0)
2242 .addImm(0)
2243 .addImm(1);
2244 BuildMI(RestoreMBB, DL, TII->get(VE::BRCFLa_t)).addMBB(SinkMBB);
2245 RestoreMBB->addSuccessor(SinkMBB);
2246
2247 MI.eraseFromParent();
2248 return SinkMBB;
2249}
2250
2253 MachineBasicBlock *MBB) const {
2254 DebugLoc DL = MI.getDebugLoc();
2255 MachineFunction *MF = MBB->getParent();
2256 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
2257 MachineRegisterInfo &MRI = MF->getRegInfo();
2258
2259 // Memory Reference.
2260 SmallVector<MachineMemOperand *, 2> MMOs(MI.memoperands());
2261 Register BufReg = MI.getOperand(0).getReg();
2262
2263 Register Tmp = MRI.createVirtualRegister(&VE::I64RegClass);
2264 // Since FP is only updated here but NOT referenced, it's treated as GPR.
2265 Register FP = VE::SX9;
2266 Register SP = VE::SX11;
2267
2269
2270 MachineBasicBlock *ThisMBB = MBB;
2271
2272 // For `call @llvm.eh.sjlj.longjmp(buf)`, we generate following instructions.
2273 //
2274 // ThisMBB:
2275 // %fp = load buf[0]
2276 // %jmp = load buf[1]
2277 // %s10 = buf ; Store an address of buf to SX10 for RestoreMBB
2278 // %sp = load buf[2] ; generated by llvm.eh.sjlj.setjmp.
2279 // jmp %jmp
2280
2281 // Reload FP.
2282 MIB = BuildMI(*ThisMBB, MI, DL, TII->get(VE::LDrii), FP);
2283 MIB.addReg(BufReg);
2284 MIB.addImm(0);
2285 MIB.addImm(0);
2286 MIB.setMemRefs(MMOs);
2287
2288 // Reload IP.
2289 MIB = BuildMI(*ThisMBB, MI, DL, TII->get(VE::LDrii), Tmp);
2290 MIB.addReg(BufReg);
2291 MIB.addImm(0);
2292 MIB.addImm(8);
2293 MIB.setMemRefs(MMOs);
2294
2295 // Copy BufReg to SX10 for later use in setjmp.
2296 // FIXME: Better to not use SX10 here
2297 BuildMI(*ThisMBB, MI, DL, TII->get(VE::ORri), VE::SX10)
2298 .addReg(BufReg)
2299 .addImm(0);
2300
2301 // Reload SP.
2302 MIB = BuildMI(*ThisMBB, MI, DL, TII->get(VE::LDrii), SP);
2303 MIB.add(MI.getOperand(0)); // we can preserve the kill flags here.
2304 MIB.addImm(0);
2305 MIB.addImm(16);
2306 MIB.setMemRefs(MMOs);
2307
2308 // Jump.
2309 BuildMI(*ThisMBB, MI, DL, TII->get(VE::BCFLari_t))
2310 .addReg(Tmp, getKillRegState(true))
2311 .addImm(0);
2312
2313 MI.eraseFromParent();
2314 return ThisMBB;
2315}
2316
2319 MachineBasicBlock *BB) const {
2320 DebugLoc DL = MI.getDebugLoc();
2321 MachineFunction *MF = BB->getParent();
2322 MachineFrameInfo &MFI = MF->getFrameInfo();
2323 MachineRegisterInfo &MRI = MF->getRegInfo();
2324 const VEInstrInfo *TII = Subtarget->getInstrInfo();
2325 int FI = MFI.getFunctionContextIndex();
2326
2327 // Get a mapping of the call site numbers to all of the landing pads they're
2328 // associated with.
2330 unsigned MaxCSNum = 0;
2331 for (auto &MBB : *MF) {
2332 if (!MBB.isEHPad())
2333 continue;
2334
2335 MCSymbol *Sym = nullptr;
2336 for (const auto &MI : MBB) {
2337 if (MI.isDebugInstr())
2338 continue;
2339
2340 assert(MI.isEHLabel() && "expected EH_LABEL");
2341 Sym = MI.getOperand(0).getMCSymbol();
2342 break;
2343 }
2344
2345 if (!MF->hasCallSiteLandingPad(Sym))
2346 continue;
2347
2348 for (unsigned CSI : MF->getCallSiteLandingPad(Sym)) {
2349 CallSiteNumToLPad[CSI].push_back(&MBB);
2350 MaxCSNum = std::max(MaxCSNum, CSI);
2351 }
2352 }
2353
2354 // Get an ordered list of the machine basic blocks for the jump table.
2355 std::vector<MachineBasicBlock *> LPadList;
2357 LPadList.reserve(CallSiteNumToLPad.size());
2358
2359 for (unsigned CSI = 1; CSI <= MaxCSNum; ++CSI) {
2360 for (auto &LP : CallSiteNumToLPad[CSI]) {
2361 LPadList.push_back(LP);
2362 InvokeBBs.insert_range(LP->predecessors());
2363 }
2364 }
2365
2366 assert(!LPadList.empty() &&
2367 "No landing pad destinations for the dispatch jump table!");
2368
2369 // The %fn_context is allocated like below (from --print-after=sjljehprepare):
2370 // %fn_context = alloca { i8*, i64, [4 x i64], i8*, i8*, [5 x i8*] }
2371 //
2372 // This `[5 x i8*]` is jmpbuf, so jmpbuf[1] is FI+72.
2373 // First `i64` is callsite, so callsite is FI+8.
2374 static const int OffsetIC = 72;
2375 static const int OffsetCS = 8;
2376
2377 // Create the MBBs for the dispatch code like following:
2378 //
2379 // ThisMBB:
2380 // Prepare DispatchBB address and store it to buf[1].
2381 // ...
2382 //
2383 // DispatchBB:
2384 // %s15 = GETGOT iff isPositionIndependent
2385 // %callsite = load callsite
2386 // brgt.l.t #size of callsites, %callsite, DispContBB
2387 //
2388 // TrapBB:
2389 // Call abort.
2390 //
2391 // DispContBB:
2392 // %breg = address of jump table
2393 // %pc = load and calculate next pc from %breg and %callsite
2394 // jmp %pc
2395
2396 // Shove the dispatch's address into the return slot in the function context.
2397 MachineBasicBlock *DispatchBB = MF->CreateMachineBasicBlock();
2398 DispatchBB->setIsEHPad(true);
2399
2400 // Trap BB will causes trap like `assert(0)`.
2402 DispatchBB->addSuccessor(TrapBB);
2403
2404 MachineBasicBlock *DispContBB = MF->CreateMachineBasicBlock();
2405 DispatchBB->addSuccessor(DispContBB);
2406
2407 // Insert MBBs.
2408 MF->push_back(DispatchBB);
2409 MF->push_back(DispContBB);
2410 MF->push_back(TrapBB);
2411
2412 // Insert code to call abort in the TrapBB.
2413 Register Abort = prepareSymbol(*TrapBB, TrapBB->end(), "abort", DL,
2414 /* Local */ false, /* Call */ true);
2415 BuildMI(TrapBB, DL, TII->get(VE::BSICrii), VE::SX10)
2416 .addReg(Abort, getKillRegState(true))
2417 .addImm(0)
2418 .addImm(0);
2419
2420 // Insert code into the entry block that creates and registers the function
2421 // context.
2422 setupEntryBlockForSjLj(MI, BB, DispatchBB, FI, OffsetIC);
2423
2424 // Create the jump table and associated information
2425 unsigned JTE = getJumpTableEncoding();
2427 unsigned MJTI = JTI->createJumpTableIndex(LPadList);
2428
2429 const VERegisterInfo &RI = TII->getRegisterInfo();
2430 // Add a register mask with no preserved registers. This results in all
2431 // registers being marked as clobbered.
2432 BuildMI(DispatchBB, DL, TII->get(VE::NOP))
2434
2435 if (isPositionIndependent()) {
2436 // Force to generate GETGOT, since current implementation doesn't store GOT
2437 // register.
2438 BuildMI(DispatchBB, DL, TII->get(VE::GETGOT), VE::SX15);
2439 }
2440
2441 // IReg is used as an index in a memory operand and therefore can't be SP
2442 const TargetRegisterClass *RC = &VE::I64RegClass;
2443 Register IReg = MRI.createVirtualRegister(RC);
2444 addFrameReference(BuildMI(DispatchBB, DL, TII->get(VE::LDLZXrii), IReg), FI,
2445 OffsetCS);
2446 if (LPadList.size() < 64) {
2447 BuildMI(DispatchBB, DL, TII->get(VE::BRCFLir_t))
2449 .addImm(LPadList.size())
2450 .addReg(IReg)
2451 .addMBB(TrapBB);
2452 } else {
2453 assert(LPadList.size() <= 0x7FFFFFFF && "Too large Landing Pad!");
2454 Register TmpReg = MRI.createVirtualRegister(RC);
2455 BuildMI(DispatchBB, DL, TII->get(VE::LEAzii), TmpReg)
2456 .addImm(0)
2457 .addImm(0)
2458 .addImm(LPadList.size());
2459 BuildMI(DispatchBB, DL, TII->get(VE::BRCFLrr_t))
2461 .addReg(TmpReg, getKillRegState(true))
2462 .addReg(IReg)
2463 .addMBB(TrapBB);
2464 }
2465
2466 Register BReg = MRI.createVirtualRegister(RC);
2467 Register Tmp1 = MRI.createVirtualRegister(RC);
2468 Register Tmp2 = MRI.createVirtualRegister(RC);
2469
2470 if (isPositionIndependent()) {
2471 // Create following instructions for local linkage PIC code.
2472 // lea %Tmp1, .LJTI0_0@gotoff_lo
2473 // and %Tmp2, %Tmp1, (32)0
2474 // lea.sl %BReg, .LJTI0_0@gotoff_hi(%Tmp2, %s15) ; %s15 is GOT
2475 BuildMI(DispContBB, DL, TII->get(VE::LEAzii), Tmp1)
2476 .addImm(0)
2477 .addImm(0)
2479 BuildMI(DispContBB, DL, TII->get(VE::ANDrm), Tmp2)
2480 .addReg(Tmp1, getKillRegState(true))
2481 .addImm(M0(32));
2482 BuildMI(DispContBB, DL, TII->get(VE::LEASLrri), BReg)
2483 .addReg(VE::SX15)
2484 .addReg(Tmp2, getKillRegState(true))
2486 } else {
2487 // Create following instructions for non-PIC code.
2488 // lea %Tmp1, .LJTI0_0@lo
2489 // and %Tmp2, %Tmp1, (32)0
2490 // lea.sl %BReg, .LJTI0_0@hi(%Tmp2)
2491 BuildMI(DispContBB, DL, TII->get(VE::LEAzii), Tmp1)
2492 .addImm(0)
2493 .addImm(0)
2495 BuildMI(DispContBB, DL, TII->get(VE::ANDrm), Tmp2)
2496 .addReg(Tmp1, getKillRegState(true))
2497 .addImm(M0(32));
2498 BuildMI(DispContBB, DL, TII->get(VE::LEASLrii), BReg)
2499 .addReg(Tmp2, getKillRegState(true))
2500 .addImm(0)
2502 }
2503
2504 switch (JTE) {
2506 // Generate simple block address code for no-PIC model.
2507 // sll %Tmp1, %IReg, 3
2508 // lds %TReg, 0(%Tmp1, %BReg)
2509 // bcfla %TReg
2510
2511 Register TReg = MRI.createVirtualRegister(RC);
2512 Register Tmp1 = MRI.createVirtualRegister(RC);
2513
2514 BuildMI(DispContBB, DL, TII->get(VE::SLLri), Tmp1)
2515 .addReg(IReg, getKillRegState(true))
2516 .addImm(3);
2517 BuildMI(DispContBB, DL, TII->get(VE::LDrri), TReg)
2518 .addReg(BReg, getKillRegState(true))
2519 .addReg(Tmp1, getKillRegState(true))
2520 .addImm(0);
2521 BuildMI(DispContBB, DL, TII->get(VE::BCFLari_t))
2522 .addReg(TReg, getKillRegState(true))
2523 .addImm(0);
2524 break;
2525 }
2527 // Generate block address code using differences from the function pointer
2528 // for PIC model.
2529 // sll %Tmp1, %IReg, 2
2530 // ldl.zx %OReg, 0(%Tmp1, %BReg)
2531 // Prepare function address in BReg2.
2532 // adds.l %TReg, %BReg2, %OReg
2533 // bcfla %TReg
2534
2536 Register OReg = MRI.createVirtualRegister(RC);
2537 Register TReg = MRI.createVirtualRegister(RC);
2538 Register Tmp1 = MRI.createVirtualRegister(RC);
2539
2540 BuildMI(DispContBB, DL, TII->get(VE::SLLri), Tmp1)
2541 .addReg(IReg, getKillRegState(true))
2542 .addImm(2);
2543 BuildMI(DispContBB, DL, TII->get(VE::LDLZXrri), OReg)
2544 .addReg(BReg, getKillRegState(true))
2545 .addReg(Tmp1, getKillRegState(true))
2546 .addImm(0);
2547 Register BReg2 =
2548 prepareSymbol(*DispContBB, DispContBB->end(),
2549 DispContBB->getParent()->getName(), DL, /* Local */ true);
2550 BuildMI(DispContBB, DL, TII->get(VE::ADDSLrr), TReg)
2551 .addReg(OReg, getKillRegState(true))
2552 .addReg(BReg2, getKillRegState(true));
2553 BuildMI(DispContBB, DL, TII->get(VE::BCFLari_t))
2554 .addReg(TReg, getKillRegState(true))
2555 .addImm(0);
2556 break;
2557 }
2558 default:
2559 llvm_unreachable("Unexpected jump table encoding");
2560 }
2561
2562 // Add the jump table entries as successors to the MBB.
2564 for (auto &LP : LPadList)
2565 if (SeenMBBs.insert(LP).second)
2566 DispContBB->addSuccessor(LP);
2567
2568 // N.B. the order the invoke BBs are processed in doesn't matter here.
2570 const MCPhysReg *SavedRegs = MF->getRegInfo().getCalleeSavedRegs();
2571 for (MachineBasicBlock *MBB : InvokeBBs) {
2572 // Remove the landing pad successor from the invoke block and replace it
2573 // with the new dispatch block.
2574 // Keep a copy of Successors since it's modified inside the loop.
2575 SmallVector<MachineBasicBlock *, 8> Successors(MBB->succ_rbegin(),
2576 MBB->succ_rend());
2577 // FIXME: Avoid quadratic complexity.
2578 for (auto *MBBS : Successors) {
2579 if (MBBS->isEHPad()) {
2580 MBB->removeSuccessor(MBBS);
2581 MBBLPads.push_back(MBBS);
2582 }
2583 }
2584
2585 MBB->addSuccessor(DispatchBB);
2586
2587 // Find the invoke call and mark all of the callee-saved registers as
2588 // 'implicit defined' so that they're spilled. This prevents code from
2589 // moving instructions to before the EH block, where they will never be
2590 // executed.
2591 for (auto &II : reverse(*MBB)) {
2592 if (!II.isCall())
2593 continue;
2594
2595 DenseSet<Register> DefRegs;
2596 for (auto &MOp : II.operands())
2597 if (MOp.isReg())
2598 DefRegs.insert(MOp.getReg());
2599
2600 MachineInstrBuilder MIB(*MF, &II);
2601 for (unsigned RI = 0; SavedRegs[RI]; ++RI) {
2602 Register Reg = SavedRegs[RI];
2603 if (!DefRegs.contains(Reg))
2605 }
2606
2607 break;
2608 }
2609 }
2610
2611 // Mark all former landing pads as non-landing pads. The dispatch is the only
2612 // landing pad now.
2613 for (auto &LP : MBBLPads)
2614 LP->setIsEHPad(false);
2615
2616 // The instruction is gone now.
2617 MI.eraseFromParent();
2618 return BB;
2619}
2620
2623 MachineBasicBlock *BB) const {
2624 switch (MI.getOpcode()) {
2625 default:
2626 llvm_unreachable("Unknown Custom Instruction!");
2627 case VE::EH_SjLj_LongJmp:
2628 return emitEHSjLjLongJmp(MI, BB);
2629 case VE::EH_SjLj_SetJmp:
2630 return emitEHSjLjSetJmp(MI, BB);
2631 case VE::EH_SjLj_Setup_Dispatch:
2632 return emitSjLjDispatchBlock(MI, BB);
2633 }
2634}
2635
2636static bool isSimm7(SDValue V) {
2637 EVT VT = V.getValueType();
2638 if (VT.isVector())
2639 return false;
2640
2641 if (VT.isInteger()) {
2643 return isInt<7>(C->getSExtValue());
2644 } else if (VT.isFloatingPoint()) {
2646 if (VT == MVT::f32 || VT == MVT::f64) {
2647 const APInt &Imm = C->getValueAPF().bitcastToAPInt();
2648 uint64_t Val = Imm.getSExtValue();
2649 if (Imm.getBitWidth() == 32)
2650 Val <<= 32; // Immediate value of float place at higher bits on VE.
2651 return isInt<7>(Val);
2652 }
2653 }
2654 }
2655 return false;
2656}
2657
2658static bool isMImm(SDValue V) {
2659 EVT VT = V.getValueType();
2660 if (VT.isVector())
2661 return false;
2662
2663 if (VT.isInteger()) {
2665 return isMImmVal(getImmVal(C));
2666 } else if (VT.isFloatingPoint()) {
2668 if (VT == MVT::f32) {
2669 // Float value places at higher bits, so ignore lower 32 bits.
2670 return isMImm32Val(getFpImmVal(C) >> 32);
2671 } else if (VT == MVT::f64) {
2672 return isMImmVal(getFpImmVal(C));
2673 }
2674 }
2675 }
2676 return false;
2677}
2678
2679static unsigned decideComp(EVT SrcVT, ISD::CondCode CC) {
2680 if (SrcVT.isFloatingPoint()) {
2681 if (SrcVT == MVT::f128)
2682 return VEISD::CMPQ;
2683 return VEISD::CMPF;
2684 }
2685 return isSignedIntSetCC(CC) ? VEISD::CMPI : VEISD::CMPU;
2686}
2687
2688static EVT decideCompType(EVT SrcVT) {
2689 if (SrcVT == MVT::f128)
2690 return MVT::f64;
2691 return SrcVT;
2692}
2693
2695 bool WithCMov) {
2696 if (SrcVT.isFloatingPoint()) {
2697 // For the case of floating point setcc, only unordered comparison
2698 // or general comparison with -enable-no-nans-fp-math option reach
2699 // here, so it is safe even if values are NaN. Only f128 doesn't
2700 // safe since VE uses f64 result of f128 comparison.
2701 return SrcVT != MVT::f128;
2702 }
2703 if (isIntEqualitySetCC(CC)) {
2704 // For the case of equal or not equal, it is safe without comparison with 0.
2705 return true;
2706 }
2707 if (WithCMov) {
2708 // For the case of integer setcc with cmov, all signed comparison with 0
2709 // are safe.
2710 return isSignedIntSetCC(CC);
2711 }
2712 // For the case of integer setcc, only signed 64 bits comparison is safe.
2713 // For unsigned, "CMPU 0x80000000, 0" has to be greater than 0, but it becomes
2714 // less than 0 witout CMPU. For 32 bits, other half of 32 bits are
2715 // uncoditional, so it is not safe too without CMPI..
2716 return isSignedIntSetCC(CC) && SrcVT == MVT::i64;
2717}
2718
2720 ISD::CondCode CC, bool WithCMov,
2721 const SDLoc &DL, SelectionDAG &DAG) {
2722 // Compare values. If RHS is 0 and it is safe to calculate without
2723 // comparison, we don't generate an instruction for comparison.
2724 EVT CompVT = decideCompType(VT);
2725 if (CompVT == VT && safeWithoutCompWithNull(VT, CC, WithCMov) &&
2727 return LHS;
2728 }
2729 return DAG.getNode(decideComp(VT, CC), DL, CompVT, LHS, RHS);
2730}
2731
2733 DAGCombinerInfo &DCI) const {
2734 assert(N->getOpcode() == ISD::SELECT &&
2735 "Should be called with a SELECT node");
2737 SDValue Cond = N->getOperand(0);
2738 SDValue True = N->getOperand(1);
2739 SDValue False = N->getOperand(2);
2740
2741 // We handle only scalar SELECT.
2742 EVT VT = N->getValueType(0);
2743 if (VT.isVector())
2744 return SDValue();
2745
2746 // Peform combineSelect after leagalize DAG.
2747 if (!DCI.isAfterLegalizeDAG())
2748 return SDValue();
2749
2750 EVT VT0 = Cond.getValueType();
2751 if (isMImm(True)) {
2752 // VE's condition move can handle MImm in True clause, so nothing to do.
2753 } else if (isMImm(False)) {
2754 // VE's condition move can handle MImm in True clause, so swap True and
2755 // False clauses if False has MImm value. And, update condition code.
2756 std::swap(True, False);
2757 CC = getSetCCInverse(CC, VT0);
2758 }
2759
2760 SDLoc DL(N);
2761 SelectionDAG &DAG = DCI.DAG;
2762 VECC::CondCode VECCVal;
2763 if (VT0.isFloatingPoint()) {
2764 VECCVal = fpCondCode2Fcc(CC);
2765 } else {
2766 VECCVal = intCondCode2Icc(CC);
2767 }
2768 SDValue Ops[] = {Cond, True, False,
2769 DAG.getConstant(VECCVal, DL, MVT::i32)};
2770 return DAG.getNode(VEISD::CMOV, DL, VT, Ops);
2771}
2772
2774 DAGCombinerInfo &DCI) const {
2775 assert(N->getOpcode() == ISD::SELECT_CC &&
2776 "Should be called with a SELECT_CC node");
2777 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(4))->get();
2778 SDValue LHS = N->getOperand(0);
2779 SDValue RHS = N->getOperand(1);
2780 SDValue True = N->getOperand(2);
2781 SDValue False = N->getOperand(3);
2782
2783 // We handle only scalar SELECT_CC.
2784 EVT VT = N->getValueType(0);
2785 if (VT.isVector())
2786 return SDValue();
2787
2788 // Peform combineSelectCC after leagalize DAG.
2789 if (!DCI.isAfterLegalizeDAG())
2790 return SDValue();
2791
2792 // We handle only i32/i64/f32/f64/f128 comparisons.
2793 EVT LHSVT = LHS.getValueType();
2794 assert(LHSVT == RHS.getValueType());
2795 switch (LHSVT.getSimpleVT().SimpleTy) {
2796 case MVT::i32:
2797 case MVT::i64:
2798 case MVT::f32:
2799 case MVT::f64:
2800 case MVT::f128:
2801 break;
2802 default:
2803 // Return SDValue to let llvm handle other types.
2804 return SDValue();
2805 }
2806
2807 if (isMImm(RHS)) {
2808 // VE's comparison can handle MImm in RHS, so nothing to do.
2809 } else if (isSimm7(RHS)) {
2810 // VE's comparison can handle Simm7 in LHS, so swap LHS and RHS, and
2811 // update condition code.
2812 std::swap(LHS, RHS);
2813 CC = getSetCCSwappedOperands(CC);
2814 }
2815 if (isMImm(True)) {
2816 // VE's condition move can handle MImm in True clause, so nothing to do.
2817 } else if (isMImm(False)) {
2818 // VE's condition move can handle MImm in True clause, so swap True and
2819 // False clauses if False has MImm value. And, update condition code.
2820 std::swap(True, False);
2821 CC = getSetCCInverse(CC, LHSVT);
2822 }
2823
2824 SDLoc DL(N);
2825 SelectionDAG &DAG = DCI.DAG;
2826
2827 bool WithCMov = true;
2828 SDValue CompNode = generateComparison(LHSVT, LHS, RHS, CC, WithCMov, DL, DAG);
2829
2830 VECC::CondCode VECCVal;
2831 if (LHSVT.isFloatingPoint()) {
2832 VECCVal = fpCondCode2Fcc(CC);
2833 } else {
2834 VECCVal = intCondCode2Icc(CC);
2835 }
2836 SDValue Ops[] = {CompNode, True, False,
2837 DAG.getConstant(VECCVal, DL, MVT::i32)};
2838 return DAG.getNode(VEISD::CMOV, DL, VT, Ops);
2839}
2840
2841static bool isI32InsnAllUses(const SDNode *User, const SDNode *N);
2842static bool isI32Insn(const SDNode *User, const SDNode *N) {
2843 switch (User->getOpcode()) {
2844 default:
2845 return false;
2846 case ISD::ADD:
2847 case ISD::SUB:
2848 case ISD::MUL:
2849 case ISD::SDIV:
2850 case ISD::UDIV:
2851 case ISD::SETCC:
2852 case ISD::SMIN:
2853 case ISD::SMAX:
2854 case ISD::SHL:
2855 case ISD::SRA:
2856 case ISD::BSWAP:
2857 case ISD::SINT_TO_FP:
2858 case ISD::UINT_TO_FP:
2859 case ISD::BR_CC:
2860 case ISD::BITCAST:
2862 case ISD::ATOMIC_SWAP:
2863 case VEISD::CMPU:
2864 case VEISD::CMPI:
2865 return true;
2866 case ISD::SRL:
2867 if (N->getOperand(0).getOpcode() != ISD::SRL)
2868 return true;
2869 // (srl (trunc (srl ...))) may be optimized by combining srl, so
2870 // doesn't optimize trunc now.
2871 return false;
2872 case ISD::SELECT_CC:
2873 if (User->getOperand(2).getNode() != N &&
2874 User->getOperand(3).getNode() != N)
2875 return true;
2876 return isI32InsnAllUses(User, N);
2877 case VEISD::CMOV:
2878 // CMOV in (cmov (trunc ...), true, false, int-comparison) is safe.
2879 // However, trunc in true or false clauses is not safe.
2880 if (User->getOperand(1).getNode() != N &&
2881 User->getOperand(2).getNode() != N &&
2883 VECC::CondCode VECCVal =
2884 static_cast<VECC::CondCode>(User->getConstantOperandVal(3));
2885 return isIntVECondCode(VECCVal);
2886 }
2887 [[fallthrough]];
2888 case ISD::AND:
2889 case ISD::OR:
2890 case ISD::XOR:
2891 case ISD::SELECT:
2892 case ISD::CopyToReg:
2893 // Check all use of selections, bit operations, and copies. If all of them
2894 // are safe, optimize truncate to extract_subreg.
2895 return isI32InsnAllUses(User, N);
2896 }
2897}
2898
2899static bool isI32InsnAllUses(const SDNode *User, const SDNode *N) {
2900 // Check all use of User node. If all of them are safe, optimize
2901 // truncate to extract_subreg.
2902 for (const SDNode *U : User->users()) {
2903 switch (U->getOpcode()) {
2904 default:
2905 // If the use is an instruction which treats the source operand as i32,
2906 // it is safe to avoid truncate here.
2907 if (isI32Insn(U, N))
2908 continue;
2909 break;
2910 case ISD::ANY_EXTEND:
2911 case ISD::SIGN_EXTEND:
2912 case ISD::ZERO_EXTEND: {
2913 // Special optimizations to the combination of ext and trunc.
2914 // (ext ... (select ... (trunc ...))) is safe to avoid truncate here
2915 // since this truncate instruction clears higher 32 bits which is filled
2916 // by one of ext instructions later.
2917 assert(N->getValueType(0) == MVT::i32 &&
2918 "find truncate to not i32 integer");
2919 if (User->getOpcode() == ISD::SELECT_CC ||
2920 User->getOpcode() == ISD::SELECT || User->getOpcode() == VEISD::CMOV)
2921 continue;
2922 break;
2923 }
2924 }
2925 return false;
2926 }
2927 return true;
2928}
2929
2930// Optimize TRUNCATE in DAG combining. Optimizing it in CUSTOM lower is
2931// sometime too early. Optimizing it in DAG pattern matching in VEInstrInfo.td
2932// is sometime too late. So, doing it at here.
2934 DAGCombinerInfo &DCI) const {
2935 assert(N->getOpcode() == ISD::TRUNCATE &&
2936 "Should be called with a TRUNCATE node");
2937
2938 SelectionDAG &DAG = DCI.DAG;
2939 SDLoc DL(N);
2940 EVT VT = N->getValueType(0);
2941
2942 // We prefer to do this when all types are legal.
2943 if (!DCI.isAfterLegalizeDAG())
2944 return SDValue();
2945
2946 // Skip combine TRUNCATE atm if the operand of TRUNCATE might be a constant.
2947 if (N->getOperand(0)->getOpcode() == ISD::SELECT_CC &&
2948 isa<ConstantSDNode>(N->getOperand(0)->getOperand(0)) &&
2949 isa<ConstantSDNode>(N->getOperand(0)->getOperand(1)))
2950 return SDValue();
2951
2952 // Check all use of this TRUNCATE.
2953 for (const SDNode *User : N->users()) {
2954 // Make sure that we're not going to replace TRUNCATE for non i32
2955 // instructions.
2956 //
2957 // FIXME: Although we could sometimes handle this, and it does occur in
2958 // practice that one of the condition inputs to the select is also one of
2959 // the outputs, we currently can't deal with this.
2960 if (isI32Insn(User, N))
2961 continue;
2962
2963 return SDValue();
2964 }
2965
2966 SDValue SubI32 = DAG.getTargetConstant(VE::sub_i32, DL, MVT::i32);
2967 return SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL, VT,
2968 N->getOperand(0), SubI32),
2969 0);
2970}
2971
2973 DAGCombinerInfo &DCI) const {
2974 switch (N->getOpcode()) {
2975 default:
2976 break;
2977 case ISD::SELECT:
2978 return combineSelect(N, DCI);
2979 case ISD::SELECT_CC:
2980 return combineSelectCC(N, DCI);
2981 case ISD::TRUNCATE:
2982 return combineTRUNCATE(N, DCI);
2983 }
2984
2985 return SDValue();
2986}
2987
2988//===----------------------------------------------------------------------===//
2989// VE Inline Assembly Support
2990//===----------------------------------------------------------------------===//
2991
2994 if (Constraint.size() == 1) {
2995 switch (Constraint[0]) {
2996 default:
2997 break;
2998 case 'v': // vector registers
2999 return C_RegisterClass;
3000 }
3001 }
3002 return TargetLowering::getConstraintType(Constraint);
3003}
3004
3005std::pair<unsigned, const TargetRegisterClass *>
3007 StringRef Constraint,
3008 MVT VT) const {
3009 const TargetRegisterClass *RC = nullptr;
3010 if (Constraint.size() == 1) {
3011 switch (Constraint[0]) {
3012 default:
3013 return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
3014 case 'r':
3015 RC = &VE::I64RegClass;
3016 break;
3017 case 'v':
3018 RC = &VE::V64RegClass;
3019 break;
3020 }
3021 return std::make_pair(0U, RC);
3022 }
3023
3024 return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
3025}
3026
3027//===----------------------------------------------------------------------===//
3028// VE Target Optimization Support
3029//===----------------------------------------------------------------------===//
3030
3032 // Specify 8 for PIC model to relieve the impact of PIC load instructions.
3033 if (isJumpTableRelative())
3034 return 8;
3035
3037}
3038
3040 EVT VT = Y.getValueType();
3041
3042 // VE doesn't have vector and not instruction.
3043 if (VT.isVector())
3044 return false;
3045
3046 // VE allows different immediate values for X and Y where ~X & Y.
3047 // Only simm7 works for X, and only mimm works for Y on VE. However, this
3048 // function is used to check whether an immediate value is OK for and-not
3049 // instruction as both X and Y. Generating additional instruction to
3050 // retrieve an immediate value is no good since the purpose of this
3051 // function is to convert a series of 3 instructions to another series of
3052 // 3 instructions with better parallelism. Therefore, we return false
3053 // for all immediate values now.
3054 // FIXME: Change hasAndNot function to have two operands to make it work
3055 // correctly with Aurora VE.
3057 return false;
3058
3059 // It's ok for generic registers.
3060 return true;
3061}
3062
3064 SelectionDAG &DAG) const {
3065 assert(Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT && "Unknown opcode!");
3066 MVT VT = Op.getOperand(0).getSimpleValueType();
3067
3068 // Special treatment for packed V64 types.
3069 assert(VT == MVT::v512i32 || VT == MVT::v512f32);
3070 (void)VT;
3071 // Example of codes:
3072 // %packed_v = extractelt %vr, %idx / 2
3073 // %v = %packed_v >> (%idx % 2 * 32)
3074 // %res = %v & 0xffffffff
3075
3076 SDValue Vec = Op.getOperand(0);
3077 SDValue Idx = Op.getOperand(1);
3078 SDLoc DL(Op);
3079 SDValue Result = Op;
3080 if (false /* Idx->isConstant() */) {
3081 // TODO: optimized implementation using constant values
3082 } else {
3083 SDValue Const1 = DAG.getConstant(1, DL, MVT::i64);
3084 SDValue HalfIdx = DAG.getNode(ISD::SRL, DL, MVT::i64, {Idx, Const1});
3085 SDValue PackedElt =
3086 SDValue(DAG.getMachineNode(VE::LVSvr, DL, MVT::i64, {Vec, HalfIdx}), 0);
3087 SDValue AndIdx = DAG.getNode(ISD::AND, DL, MVT::i64, {Idx, Const1});
3088 SDValue Shift = DAG.getNode(ISD::XOR, DL, MVT::i64, {AndIdx, Const1});
3089 SDValue Const5 = DAG.getConstant(5, DL, MVT::i64);
3090 Shift = DAG.getNode(ISD::SHL, DL, MVT::i64, {Shift, Const5});
3091 PackedElt = DAG.getNode(ISD::SRL, DL, MVT::i64, {PackedElt, Shift});
3092 SDValue Mask = DAG.getConstant(0xFFFFFFFFL, DL, MVT::i64);
3093 PackedElt = DAG.getNode(ISD::AND, DL, MVT::i64, {PackedElt, Mask});
3094 SDValue SubI32 = DAG.getTargetConstant(VE::sub_i32, DL, MVT::i32);
3095 Result = SDValue(DAG.getMachineNode(TargetOpcode::EXTRACT_SUBREG, DL,
3096 MVT::i32, PackedElt, SubI32),
3097 0);
3098
3099 if (Op.getSimpleValueType() == MVT::f32) {
3100 Result = DAG.getBitcast(MVT::f32, Result);
3101 } else {
3102 assert(Op.getSimpleValueType() == MVT::i32);
3103 }
3104 }
3105 return Result;
3106}
3107
3109 SelectionDAG &DAG) const {
3110 assert(Op.getOpcode() == ISD::INSERT_VECTOR_ELT && "Unknown opcode!");
3111 MVT VT = Op.getOperand(0).getSimpleValueType();
3112
3113 // Special treatment for packed V64 types.
3114 assert(VT == MVT::v512i32 || VT == MVT::v512f32);
3115 (void)VT;
3116 // The v512i32 and v512f32 starts from upper bits (0..31). This "upper
3117 // bits" required `val << 32` from C implementation's point of view.
3118 //
3119 // Example of codes:
3120 // %packed_elt = extractelt %vr, (%idx >> 1)
3121 // %shift = ((%idx & 1) ^ 1) << 5
3122 // %packed_elt &= 0xffffffff00000000 >> shift
3123 // %packed_elt |= (zext %val) << shift
3124 // %vr = insertelt %vr, %packed_elt, (%idx >> 1)
3125
3126 SDLoc DL(Op);
3127 SDValue Vec = Op.getOperand(0);
3128 SDValue Val = Op.getOperand(1);
3129 SDValue Idx = Op.getOperand(2);
3130 if (Idx.getSimpleValueType() == MVT::i32)
3131 Idx = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i64, Idx);
3132 if (Val.getSimpleValueType() == MVT::f32)
3133 Val = DAG.getBitcast(MVT::i32, Val);
3134 assert(Val.getSimpleValueType() == MVT::i32);
3135 Val = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i64, Val);
3136
3137 SDValue Result = Op;
3138 if (false /* Idx->isConstant()*/) {
3139 // TODO: optimized implementation using constant values
3140 } else {
3141 SDValue Const1 = DAG.getConstant(1, DL, MVT::i64);
3142 SDValue HalfIdx = DAG.getNode(ISD::SRL, DL, MVT::i64, {Idx, Const1});
3143 SDValue PackedElt =
3144 SDValue(DAG.getMachineNode(VE::LVSvr, DL, MVT::i64, {Vec, HalfIdx}), 0);
3145 SDValue AndIdx = DAG.getNode(ISD::AND, DL, MVT::i64, {Idx, Const1});
3146 SDValue Shift = DAG.getNode(ISD::XOR, DL, MVT::i64, {AndIdx, Const1});
3147 SDValue Const5 = DAG.getConstant(5, DL, MVT::i64);
3148 Shift = DAG.getNode(ISD::SHL, DL, MVT::i64, {Shift, Const5});
3149 SDValue Mask = DAG.getConstant(0xFFFFFFFF00000000L, DL, MVT::i64);
3150 Mask = DAG.getNode(ISD::SRL, DL, MVT::i64, {Mask, Shift});
3151 PackedElt = DAG.getNode(ISD::AND, DL, MVT::i64, {PackedElt, Mask});
3152 Val = DAG.getNode(ISD::SHL, DL, MVT::i64, {Val, Shift});
3153 PackedElt = DAG.getNode(ISD::OR, DL, MVT::i64, {PackedElt, Val});
3154 Result =
3155 SDValue(DAG.getMachineNode(VE::LSVrr_v, DL, Vec.getSimpleValueType(),
3156 {HalfIdx, PackedElt, Vec}),
3157 0);
3158 }
3159 return Result;
3160}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis Results
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Module.h This file contains the declarations for the Module class.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define RegName(no)
#define I(x, y, z)
Definition MD5.cpp:57
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
uint64_t IntrinsicInst * II
static CodeModel::Model getCodeModel(const PPCSubtarget &S, const TargetMachine &TM, const MachineOperand &MO)
const SmallVectorImpl< MachineOperand > & Cond
This file implements the StringSwitch template, which mimics a switch() statement whose cases are str...
#define LLVM_DEBUG(...)
Definition Debug.h:119
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static unsigned decideComp(EVT SrcVT, ISD::CondCode CC)
static bool isSimm7(SDValue V)
CCAssignFn * getParamCC(CallingConv::ID CallConv, bool IsVarArg)
static SDValue lowerLoadF128(SDValue Op, SelectionDAG &DAG)
static bool isMImm(SDValue V)
static SDValue prepareTS1AM(SDValue Op, SelectionDAG &DAG, SDValue &Flag, SDValue &Bits)
CCAssignFn * getReturnCC(CallingConv::ID CallConv)
static bool safeWithoutCompWithNull(EVT SrcVT, ISD::CondCode CC, bool WithCMov)
static bool isI32InsnAllUses(const SDNode *User, const SDNode *N)
static SDValue lowerLoadI1(SDValue Op, SelectionDAG &DAG)
static SDValue generateComparison(EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode CC, bool WithCMov, const SDLoc &DL, SelectionDAG &DAG)
static EVT decideCompType(EVT SrcVT)
static bool isI32Insn(const SDNode *User, const SDNode *N)
static SDValue lowerFRAMEADDR(SDValue Op, SelectionDAG &DAG, const VETargetLowering &TLI, const VESubtarget *Subtarget)
static const MVT AllMaskVTs[]
static bool getUniqueInsertion(SDNode *N, unsigned &UniqueIdx)
static SDValue lowerRETURNADDR(SDValue Op, SelectionDAG &DAG, const VETargetLowering &TLI, const VESubtarget *Subtarget)
static const MVT AllVectorVTs[]
static const MVT AllPackedVTs[]
static SDValue finalizeTS1AM(SDValue Op, SelectionDAG &DAG, SDValue Data, SDValue Bits)
static SDValue lowerStoreF128(SDValue Op, SelectionDAG &DAG)
static SDValue lowerStoreI1(SDValue Op, SelectionDAG &DAG)
Value * RHS
Value * LHS
Class for arbitrary precision integers.
Definition APInt.h:78
an instruction that atomically reads a memory location, combines it with another value,...
BinOp getOperation() const
This is an SDNode representing atomic operations.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
CCState - This class holds information needed while lowering arguments and return values.
LLVM_ABI void AnalyzeCallResult(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeCallResult - Analyze the return values of a call, incorporating info about the passed values i...
LLVM_ABI bool CheckReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
CheckReturn - Analyze the return values of a function, returning true if the return can be performed ...
LLVM_ABI void AnalyzeReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeReturn - Analyze the returned values of a return, incorporating info about the result values i...
int64_t AllocateStack(unsigned Size, Align Alignment)
AllocateStack - Allocate a chunk of stack space with the specified size and alignment.
LLVM_ABI void AnalyzeCallOperands(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeCallOperands - Analyze the outgoing arguments to a call, incorporating info about the passed v...
uint64_t getStackSize() const
Returns the size of the currently allocated portion of the stack.
LLVM_ABI void AnalyzeFormalArguments(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeFormalArguments - Analyze an array of argument values, incorporating info about the formals in...
CCValAssign - Represent assignment of one arg/retval to a location.
Register getLocReg() const
LocInfo getLocInfo() const
bool needsCustom() const
bool isExtInLoc() const
int64_t getLocMemOffset() const
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
unsigned size() const
Definition DenseMap.h:733
Implements a dense probed hash-table based set.
Definition DenseSet.h:281
unsigned getAddressSpace() const
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
LLVM_ABI bool hasAtomicStore() const LLVM_READONLY
Return true if this atomic instruction stores to memory.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
const SDValue & getOffset() const
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:427
Context object for machine code objects.
Definition MCContext.h:83
Base class for the full range of assembler expressions which are needed for parsing.
Definition MCExpr.h:34
static const MCSymbolRefExpr * create(const MCSymbol *Symbol, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:213
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
Definition MCSymbol.h:42
Machine Value Type.
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
static auto integer_valuetypes()
static auto vector_valuetypes()
static auto fp_valuetypes()
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
void setMachineBlockAddressTaken()
Set this block to indicate that its address is used as something other than the target of a terminato...
void setIsEHPad(bool V=true)
Indicates the block is a landing pad.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
void setFrameAddressIsTaken(bool T)
void setReturnAddressIsTaken(bool s)
int getFunctionContextIndex() const
Return the index for the function context object.
unsigned getFunctionNumber() const
getFunctionNumber - Return a unique ID for the current function.
MachineJumpTableInfo * getOrCreateJumpTableInfo(unsigned JTEntryKind)
getOrCreateJumpTableInfo - Get the JumpTableInfo for this function, if it does already exist,...
StringRef getName() const
getName - Return the name of the corresponding LLVM function.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_back(MachineBasicBlock *MBB)
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
bool hasCallSiteLandingPad(MCSymbol *Sym)
Return true if the landing pad Eh symbol has an associated call site.
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
SmallVectorImpl< unsigned > & getCallSiteLandingPad(MCSymbol *Sym)
Get the call site indexes for a landing pad EH symbol.
const MachineInstrBuilder & setMemRefs(ArrayRef< MachineMemOperand * > MMOs) const
const MachineInstrBuilder & addExternalSymbol(const char *FnName, unsigned TargetFlags=0) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addRegMask(const uint32_t *Mask) const
const MachineInstrBuilder & addJumpTableIndex(unsigned Idx, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
Representation of each machine instruction.
LLVM_ABI unsigned createJumpTableIndex(const std::vector< MachineBasicBlock * > &DestBBs)
createJumpTableIndex - Create a new jump table.
@ EK_Custom32
EK_Custom32 - Each entry is a 32-bit value that is custom lowered by the TargetLowering::LowerCustomJ...
@ EK_BlockAddress
EK_BlockAddress - Each entry is a plain address of block, e.g.: .word LBB123.
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI const MCPhysReg * getCalleeSavedRegs() const
Returns list of callee saved registers.
Align getAlign() const
bool isVolatile() const
const MachinePointerInfo & getPointerInfo() const
const SDValue & getChain() const
EVT getMemoryVT() const
Return the type of the in-memory value.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
const SDValue & getOperand(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
SDValue getTargetJumpTable(int JTI, EVT VT, unsigned TargetFlags=0)
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI SDValue getGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, bool isTargetGA=false, unsigned TargetFlags=0)
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
const TargetMachine & getTarget() const
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
SDValue getTargetBlockAddress(const BlockAddress *BA, EVT VT, int64_t Offset=0, unsigned TargetFlags=0)
MachineFunction & getMachineFunction() const
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI SDValue getRegisterMask(const uint32_t *RegMask)
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetExternalSymbol(const char *Sym, EVT VT, unsigned TargetFlags=0)
SDValue getTargetConstantPool(const Constant *C, EVT VT, MaybeAlign Align=std::nullopt, int Offset=0, unsigned TargetFlags=0)
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
void insert_range(Range &&R)
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
This class is used to represent ISD::STORE nodes.
const SDValue & getBasePtr() const
const SDValue & getOffset() const
const SDValue & getValue() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
Information about stack frame layout on the target.
Align getStackAlign() const
getStackAlignment - This method returns the number of bytes to which the stack pointer must be aligne...
TargetInstrInfo - Interface to description of machine instruction set.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
LegalizeAction
This enum indicates whether operations are valid for a target, and if not, what action should be used...
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
void setMinStackArgumentAlignment(Align Alignment)
Set the minimum stack alignment of an argument.
virtual unsigned getMinimumJumpTableEntries() const
Return lower limit for number of blocks in a jump table.
const TargetMachine & getTargetMachine() const
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
void setMinFunctionAlignment(Align Alignment)
Set the target's minimum function alignment.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
void setSupportsUnalignedAtomics(bool UnalignedSupported)
Sets whether unaligned atomic operations are supported.
virtual bool isJumpTableRelative() const
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
void setMinCmpXchgSizeInBits(unsigned SizeInBits)
Sets the minimum cmpxchg or ll/sc size supported by the backend.
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
std::vector< ArgListEntry > ArgListTy
virtual ConstraintType getConstraintType(StringRef Constraint) const
Given a constraint, return the type of constraint it is for this target.
std::pair< SDValue, SDValue > LowerCallTo(CallLoweringInfo &CLI) const
This function lowers an abstract call to a function into an actual call.
bool isPositionIndependent() const
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
TargetLowering(const TargetLowering &)=delete
virtual unsigned getJumpTableEncoding() const
Return the entry encoding for a jump table in the current function.
Primary interface to the complete machine description for the target machine.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
Definition Type.cpp:272
Value * getOperand(unsigned i) const
Definition User.h:207
SDValue getBroadcast(EVT ResultVT, SDValue Scalar, SDValue AVL) const
SDValue getNode(unsigned OC, SDVTList VTL, ArrayRef< SDValue > OpV, std::optional< SDNodeFlags > Flags=std::nullopt) const
getNode {
SDValue getUNDEF(EVT VT) const
SDValue getConstant(uint64_t Val, EVT VT, bool IsTarget=false, bool IsOpaque=false) const
bool hasBP(const MachineFunction &MF) const
const VERegisterInfo * getRegisterInfo() const override
Definition VESubtarget.h:56
SDValue splitMaskArithmetic(SDValue Op, SelectionDAG &DAG) const
SDValue lowerDYNAMIC_STACKALLOC(SDValue Op, SelectionDAG &DAG) const
std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const override
Given a physical register constraint (e.g.
bool isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const override
Return true if folding a constant offset with the given GlobalAddress is legal.
SDValue lowerToVVP(SDValue Op, SelectionDAG &DAG) const
} Custom Inserter
SDValue lowerJumpTable(SDValue Op, SelectionDAG &DAG) const
SDValue getPICJumpTableRelocBase(SDValue Table, SelectionDAG &DAG) const override
Returns relocation base for the given PIC jumptable.
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const override
getSetCCResultType - Return the ISD::SETCC ValueType
TargetLoweringBase::AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *AI) const override
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
SDValue lowerGlobalTLSAddress(SDValue Op, SelectionDAG &DAG) const
bool hasAndNot(SDValue Y) const override
Return true if the target has a bitwise and-not operation: X = ~A & B This can be used to simplify se...
SDValue lowerVAARG(SDValue Op, SelectionDAG &DAG) const
SDValue combineSelect(SDNode *N, DAGCombinerInfo &DCI) const
bool isFPImmLegal(const APFloat &Imm, EVT VT, bool ForCodeSize) const override
isFPImmLegal - Returns true if the target can instruction select the specified FP immediate natively.
VETargetLowering(const TargetMachine &TM, const VESubtarget &STI)
Instruction * emitLeadingFence(IRBuilderBase &Builder, Instruction *Inst, AtomicOrdering Ord) const override
Custom Lower {.
SDValue lowerATOMIC_FENCE(SDValue Op, SelectionDAG &DAG) const
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
SDValue lowerLOAD(SDValue Op, SelectionDAG &DAG) const
MachineBasicBlock * emitEHSjLjLongJmp(MachineInstr &MI, MachineBasicBlock *MBB) const
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
} VVPLowering
SDValue lowerINSERT_VECTOR_ELT(SDValue Op, SelectionDAG &DAG) const
SDValue combineSelectCC(SDNode *N, DAGCombinerInfo &DCI) const
SDValue lowerEH_SJLJ_SETUP_DISPATCH(SDValue Op, SelectionDAG &DAG) const
unsigned getMinimumJumpTableEntries() const override
} Inline Assembly
SDValue lowerEH_SJLJ_SETJMP(SDValue Op, SelectionDAG &DAG) const
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &dl, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
MachineBasicBlock * emitSjLjDispatchBlock(MachineInstr &MI, MachineBasicBlock *BB) const
Instruction * emitTrailingFence(IRBuilderBase &Builder, Instruction *Inst, AtomicOrdering Ord) const override
Register prepareMBB(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, MachineBasicBlock *TargetBB, const DebugLoc &DL) const
void setupEntryBlockForSjLj(MachineInstr &MI, MachineBasicBlock *MBB, MachineBasicBlock *DispatchBB, int FI, int Offset) const
SDValue lowerGlobalAddress(SDValue Op, SelectionDAG &DAG) const
MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *MBB) const override
Custom Inserter {.
MachineBasicBlock * emitEHSjLjSetJmp(MachineInstr &MI, MachineBasicBlock *MBB) const
bool allowsMisalignedMemoryAccesses(EVT VT, unsigned AS, Align A, MachineMemOperand::Flags Flags, unsigned *Fast) const override
Returns true if the target allows unaligned memory accesses of the specified type.
SDValue lowerEH_SJLJ_LONGJMP(SDValue Op, SelectionDAG &DAG) const
SDValue lowerSTORE(SDValue Op, SelectionDAG &DAG) const
TargetLoweringBase::LegalizeAction getCustomOperationAction(SDNode &) const override
Custom Lower {.
SDValue makeAddress(SDValue Op, SelectionDAG &DAG) const
SDValue lowerConstantPool(SDValue Op, SelectionDAG &DAG) const
SDValue lowerBUILD_VECTOR(SDValue Op, SelectionDAG &DAG) const
SDValue legalizeInternalVectorOp(SDValue Op, SelectionDAG &DAG) const
Register prepareSymbol(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, StringRef Symbol, const DebugLoc &DL, bool IsLocal, bool IsCall) const
SDValue LowerFormalArguments(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &dl, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower the incoming (formal) arguments, described by the Ins array,...
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
} Custom Lower
SDValue lowerEXTRACT_VECTOR_ELT(SDValue Op, SelectionDAG &DAG) const
SDValue LowerCall(TargetLowering::CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
SDValue withTargetFlags(SDValue Op, unsigned TF, SelectionDAG &DAG) const
} Custom DAGCombine
SDValue combineTRUNCATE(SDNode *N, DAGCombinerInfo &DCI) const
const MCExpr * LowerCustomJumpTableEntry(const MachineJumpTableInfo *MJTI, const MachineBasicBlock *MBB, unsigned Uid, MCContext &Ctx) const override
bool CanLowerReturn(CallingConv::ID CallConv, MachineFunction &MF, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &ArgsFlags, LLVMContext &Context, const Type *RetTy) const override
This hook should be implemented to check whether the return values described by the Outs array can fi...
SDValue lowerVASTART(SDValue Op, SelectionDAG &DAG) const
unsigned getJumpTableEncoding() const override
JumpTable for VE.
SDValue lowerATOMIC_SWAP(SDValue Op, SelectionDAG &DAG) const
SDValue lowerBlockAddress(SDValue Op, SelectionDAG &DAG) const
Register getRegisterByName(const char *RegName, LLT VT, const MachineFunction &MF) const override
Return the register ID of the name passed in.
SDValue makeHiLoPair(SDValue Op, unsigned HiTF, unsigned LoTF, SelectionDAG &DAG) const
ConstraintType getConstraintType(StringRef Constraint) const override
Inline Assembly {.
SDValue lowerINTRINSIC_WO_CHAIN(SDValue Op, SelectionDAG &DAG) const
SDValue lowerToTLSGeneralDynamicModel(SDValue Op, SelectionDAG &DAG) const
std::list< std::string > * getStrList() const
LLVM Value Representation.
Definition Value.h:75
iterator_range< user_iterator > users()
Definition Value.h:428
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
Definition DenseSet.h:182
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ PreserveAll
Used for runtime calls that preserves (almost) all registers.
Definition CallingConv.h:66
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:43
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ MLOAD
Masked load and store - consecutive vector load and store operations with additional mask operand tha...
@ EH_SJLJ_LONGJMP
OUTCHAIN = EH_SJLJ_LONGJMP(INCHAIN, buffer) This corresponds to the eh.sjlj.longjmp intrinsic.
Definition ISDOpcodes.h:170
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ VAEND
VAEND, VASTART - VAEND and VASTART have three operands: an input chain, pointer, and a SRCVALUE.
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ EH_SJLJ_SETUP_DISPATCH
OUTCHAIN = EH_SJLJ_SETUP_DISPATCH(INCHAIN) The target initializes the dispatch table here.
Definition ISDOpcodes.h:174
@ GlobalAddress
Definition ISDOpcodes.h:90
@ ATOMIC_CMP_SWAP_WITH_SUCCESS
Val, Success, OUTCHAIN = ATOMIC_CMP_SWAP_WITH_SUCCESS(INCHAIN, ptr, cmp, swap) N.b.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ MEMBARRIER
MEMBARRIER - Compiler barrier only; generate a no-op.
@ ATOMIC_FENCE
OUTCHAIN = ATOMIC_FENCE(INCHAIN, ordering, scope) This corresponds to the fence instruction.
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:282
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ GlobalTLSAddress
Definition ISDOpcodes.h:91
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:806
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BR_CC
BR_CC - Conditional branch.
@ BR_JT
BR_JT - Jumptable branch.
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:814
@ VACOPY
VACOPY - VACOPY has 5 operands: an input chain, a destination pointer, a source pointer,...
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
Definition ISDOpcodes.h:232
@ VECREDUCE_ADD
Integer reductions may have a result type larger than the vector element type.
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:226
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:829
@ ATOMIC_CMP_SWAP
Val, OUTCHAIN = ATOMIC_CMP_SWAP(INCHAIN, ptr, cmp, swap) For double-word atomic operations: ValLo,...
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FRAMEADDR
FRAMEADDR, RETURNADDR - These nodes represent llvm.frameaddress and llvm.returnaddress on the DAG.
Definition ISDOpcodes.h:112
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:570
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:55
@ ATOMIC_SWAP
Val, OUTCHAIN = ATOMIC_SWAP(INCHAIN, ptr, amt) Val, OUTCHAIN = ATOMIC_LOAD_[OpName](INCHAIN,...
@ EH_SJLJ_SETJMP
RESULT, OUTCHAIN = EH_SJLJ_SETJMP(INCHAIN, buffer) This corresponds to the eh.sjlj....
Definition ISDOpcodes.h:164
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ VAARG
VAARG - VAARG has four operands: an input chain, a pointer, a SRCVALUE, and the alignment.
@ BRCOND
BRCOND - Conditional branch.
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:851
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:64
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:539
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:561
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LLVM_ABI bool isVPOpcode(unsigned Opcode)
Whether this is a vector-predicated Opcode.
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
CondCode
Definition VE.h:43
@ CC_ILE
Definition VE.h:50
@ S_GOTOFF_LO32
Definition VEMCAsmInfo.h:48
@ S_GOTOFF_HI32
Definition VEMCAsmInfo.h:47
This is an optimization pass for GlobalISel generic memory operations.
static uint64_t getFpImmVal(const ConstantFPSDNode *N)
getFpImmVal - get immediate representation of floating point value
bool isPackedVectorType(EVT SomeVT)
@ Offset
Definition DWP.cpp:577
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Dead
Unused definition.
@ Undef
Value of the register doesn't matter.
constexpr RegState getKillRegState(bool B)
static bool isIntVECondCode(VECC::CondCode CC)
Definition VE.h:151
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool CCAssignFn(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
CCAssignFn - This function assigns a location for Val, updating State to reflect the change.
static uint64_t getImmVal(const ConstantSDNode *N)
getImmVal - get immediate representation of integer value
static const MachineInstrBuilder & addFrameReference(const MachineInstrBuilder &MIB, int FI, int Offset=0, bool mem=true)
addFrameReference - This function is used to add a reference to the base of an abstract object on the...
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
bool isMaskArithmetic(SDValue Op)
static VECC::CondCode fpCondCode2Fcc(ISD::CondCode CC)
Convert a DAG floating point condition code to a VE FCC condition.
bool isMaskType(EVT SomeVT)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
AtomicOrdering
Atomic ordering for LLVM's memory model.
bool isVVPOrVEC(unsigned Opcode)
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
@ Fast
Assign the register banks as fast as possible (default).
bool isPackingSupportOpcode(unsigned Opc)
std::pair< SDValue, bool > getAnnotatedNodeAVL(SDValue Op)
DWARFExpression::Operation Op
unsigned M0(unsigned Val)
Definition VE.h:376
static VECC::CondCode intCondCode2Icc(ISD::CondCode CC)
Convert a DAG integer condition code to a VE ICC condition.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool isNullFPConstant(SDValue V)
Returns true if V is an FP constant with a value of positive zero.
static bool isMImmVal(uint64_t Val)
Definition VE.h:332
static bool isMImm32Val(uint32_t Val)
Definition VE.h:345
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
EVT changeVectorElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
Definition ValueTypes.h:98
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getGOT(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a GOT entry.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
CallLoweringInfo & setDiscardResult(bool Value=true)
CallLoweringInfo & setDebugLoc(const SDLoc &dl)
SmallVector< ISD::OutputArg, 32 > Outs
CallLoweringInfo & setChain(SDValue InChain)
CallLoweringInfo & setCallee(CallingConv::ID CC, Type *ResultType, SDValue Target, ArgListTy &&ArgsList, AttributeSet ResultAttrs={})
const uint32_t * getNoPreservedMask() const override