LLVM 24.0.0git
SystemZISelLowering.cpp
Go to the documentation of this file.
1//===-- SystemZISelLowering.cpp - SystemZ DAG lowering implementation -----===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the SystemZTargetLowering class.
10//
11//===----------------------------------------------------------------------===//
12
13#include "SystemZISelLowering.h"
14#include "SystemZCallingConv.h"
17#include "llvm/ADT/SmallSet.h"
22#include "llvm/IR/GlobalAlias.h"
24#include "llvm/IR/Intrinsics.h"
25#include "llvm/IR/IntrinsicsS390.h"
26#include "llvm/IR/Module.h"
32#include <cctype>
33#include <optional>
34
35using namespace llvm;
36
37#define DEBUG_TYPE "systemz-lower"
38
39// Temporarily let this be disabled by default until all known problems
40// related to argument extensions are fixed.
42 "argext-abi-check", cl::init(false),
43 cl::desc("Verify that narrow int args are properly extended per the "
44 "SystemZ ABI."));
45
46namespace {
47// Represents information about a comparison.
48struct Comparison {
49 Comparison(SDValue Op0In, SDValue Op1In, SDValue ChainIn)
50 : Op0(Op0In), Op1(Op1In), Chain(ChainIn),
51 Opcode(0), ICmpType(0), CCValid(0), CCMask(0) {}
52
53 // The operands to the comparison.
54 SDValue Op0, Op1;
55
56 // Chain if this is a strict floating-point comparison.
57 SDValue Chain;
58
59 // The opcode that should be used to compare Op0 and Op1.
60 unsigned Opcode;
61
62 // A SystemZICMP value. Only used for integer comparisons.
63 unsigned ICmpType;
64
65 // The mask of CC values that Opcode can produce.
66 unsigned CCValid;
67
68 // The mask of CC values for which the original condition is true.
69 unsigned CCMask;
70};
71} // end anonymous namespace
72
73// Classify VT as either 32 or 64 bit.
74static bool is32Bit(EVT VT) {
75 switch (VT.getSimpleVT().SimpleTy) {
76 case MVT::i32:
77 return true;
78 case MVT::i64:
79 return false;
80 default:
81 llvm_unreachable("Unsupported type");
82 }
83}
84
85// Return a version of MachineOperand that can be safely used before the
86// final use.
88 if (Op.isReg())
89 Op.setIsKill(false);
90 return Op;
91}
92
94 const SystemZSubtarget &STI)
95 : TargetLowering(TM, STI), Subtarget(STI) {
96 MVT PtrVT = MVT::getIntegerVT(TM.getPointerSizeInBits(0));
97
98 auto *Regs = STI.getSpecialRegisters();
99
100 // Set up the register classes.
101 if (Subtarget.hasHighWord())
102 addRegisterClass(MVT::i32, &SystemZ::GRX32BitRegClass);
103 else
104 addRegisterClass(MVT::i32, &SystemZ::GR32BitRegClass);
105 addRegisterClass(MVT::i64, &SystemZ::GR64BitRegClass);
106 if (!useSoftFloat()) {
107 if (Subtarget.hasVector()) {
108 addRegisterClass(MVT::f16, &SystemZ::VR16BitRegClass);
109 addRegisterClass(MVT::f32, &SystemZ::VR32BitRegClass);
110 addRegisterClass(MVT::f64, &SystemZ::VR64BitRegClass);
111 } else {
112 addRegisterClass(MVT::f16, &SystemZ::FP16BitRegClass);
113 addRegisterClass(MVT::f32, &SystemZ::FP32BitRegClass);
114 addRegisterClass(MVT::f64, &SystemZ::FP64BitRegClass);
115 }
116 if (Subtarget.hasVectorEnhancements1())
117 addRegisterClass(MVT::f128, &SystemZ::VR128BitRegClass);
118 else
119 addRegisterClass(MVT::f128, &SystemZ::FP128BitRegClass);
120
121 if (Subtarget.hasVector()) {
122 addRegisterClass(MVT::v16i8, &SystemZ::VR128BitRegClass);
123 addRegisterClass(MVT::v8i16, &SystemZ::VR128BitRegClass);
124 addRegisterClass(MVT::v4i32, &SystemZ::VR128BitRegClass);
125 addRegisterClass(MVT::v2i64, &SystemZ::VR128BitRegClass);
126 addRegisterClass(MVT::v8f16, &SystemZ::VR128BitRegClass);
127 addRegisterClass(MVT::v4f32, &SystemZ::VR128BitRegClass);
128 addRegisterClass(MVT::v2f64, &SystemZ::VR128BitRegClass);
129 }
130
131 if (Subtarget.hasVector())
132 addRegisterClass(MVT::i128, &SystemZ::VR128BitRegClass);
133 }
134
135 // Compute derived properties from the register classes
136 computeRegisterProperties(Subtarget.getRegisterInfo());
137
138 // Set up special registers.
139 setStackPointerRegisterToSaveRestore(Regs->getStackPointerRegister());
140
141 // TODO: It may be better to default to latency-oriented scheduling, however
142 // LLVM's current latency-oriented scheduler can't handle physreg definitions
143 // such as SystemZ has with CC, so set this to the register-pressure
144 // scheduler, because it can.
146
149
151
152 // Instructions are strings of 2-byte aligned 2-byte values.
154 // For performance reasons we prefer 16-byte alignment.
156
157 // Handle operations that are handled in a similar way for all types.
158 for (unsigned I = MVT::FIRST_INTEGER_VALUETYPE;
159 I <= MVT::LAST_FP_VALUETYPE;
160 ++I) {
162 if (isTypeLegal(VT)) {
163 // Lower SET_CC into an IPM-based sequence.
167
168 // Expand SELECT(C, A, B) into SELECT_CC(X, 0, A, B, NE).
170
171 // Lower SELECT_CC and BR_CC into separate comparisons and branches.
174 }
175 }
176
177 // Expand jump table branches as address arithmetic followed by an
178 // indirect jump.
180
181 // Expand BRCOND into a BR_CC (see above).
183
184 // Handle integer types except i128.
185 for (unsigned I = MVT::FIRST_INTEGER_VALUETYPE;
186 I <= MVT::LAST_INTEGER_VALUETYPE;
187 ++I) {
189 if (isTypeLegal(VT) && VT != MVT::i128) {
191
192 // Expand individual DIV and REMs into DIVREMs.
199
200 // Support addition/subtraction with overflow.
203
204 // Support addition/subtraction with carry.
207
208 // Support carry in as value rather than glue.
211
212 // Lower ATOMIC_LOAD_SUB into ATOMIC_LOAD_ADD if LAA and LAAG are
213 // available, or if the operand is constant.
215
216 // Use POPCNT on z196 and above.
217 if (Subtarget.hasPopulationCount())
219 else
221
222 // No special instructions for these.
225
226 // Use *MUL_LOHI where possible instead of MULH*.
231
232 // The fp<=>i32/i64 conversions are all Legal except for f16 and for
233 // unsigned on z10 (only z196 and above have native support for
234 // unsigned conversions).
241 // Handle unsigned 32-bit input types as signed 64-bit types on z10.
242 auto OpAction =
243 (!Subtarget.hasFPExtension() && VT == MVT::i32) ? Promote : Custom;
244 setOperationAction(Op, VT, OpAction);
245 }
246 }
247 }
248
249 // Handle i128 if legal.
250 if (isTypeLegal(MVT::i128)) {
251 // No special instructions for these.
258
259 // We may be able to use VSLDB/VSLD/VSRD for these.
262
263 // No special instructions for these before z17.
264 if (!Subtarget.hasVectorEnhancements3()) {
274 } else {
275 // Even if we do have a legal 128-bit multiply, we do not
276 // want 64-bit multiply-high operations to use it.
279 }
280
281 // Support addition/subtraction with carry.
286
287 // Use VPOPCT and add up partial results.
289
290 // Additional instructions available with z17.
291 if (Subtarget.hasVectorEnhancements3()) {
292 setOperationAction(ISD::ABS, MVT::i128, Legal);
293
295 MVT::i128, Legal);
296 }
297 }
298
299 // These need custom handling in order to handle the f16 conversions.
308
309 // Type legalization will convert 8- and 16-bit atomic operations into
310 // forms that operate on i32s (but still keeping the original memory VT).
311 // Lower them into full i32 operations.
323
324 // Whether or not i128 is not a legal type, we need to custom lower
325 // the atomic operations in order to exploit SystemZ instructions.
330
331 // Mark sign/zero extending atomic loads as legal, which will make
332 // DAGCombiner fold extensions into atomic loads if possible.
334 {MVT::i8, MVT::i16, MVT::i32}, Legal);
336 {MVT::i8, MVT::i16}, Legal);
338 MVT::i8, Legal);
339
340 // We can use the CC result of compare-and-swap to implement
341 // the "success" result of ATOMIC_CMP_SWAP_WITH_SUCCESS.
345
347
348 // Traps are legal, as we will convert them to "j .+2".
349 setOperationAction(ISD::TRAP, MVT::Other, Legal);
350
351 // We have native support for a 64-bit CTLZ, via FLOGR.
355
356 // On z17 we have native support for a 64-bit CTTZ.
357 if (Subtarget.hasMiscellaneousExtensions4()) {
361 }
362
363 // On z15 we have native support for a 64-bit CTPOP.
364 if (Subtarget.hasMiscellaneousExtensions3()) {
367 }
368
369 // Give LowerOperation the chance to replace 64-bit ORs with subregs.
371
372 // Expand 128 bit shifts without using a libcall.
376
377 // Also expand 256 bit shifts if i128 is a legal type.
378 if (isTypeLegal(MVT::i128)) {
382 }
383
384 // Handle bitcast from fp128 to i128.
385 if (!isTypeLegal(MVT::i128))
387
388 // We have native instructions for i8, i16 and i32 extensions, but not i1.
390 for (MVT VT : MVT::integer_valuetypes()) {
394 }
395
396 // Handle the various types of symbolic address.
402
403 // We need to handle dynamic allocations specially because of the
404 // 160-byte area at the bottom of the stack.
407
410
411 // Handle prefetches with PFD or PFDRL.
413
414 // Handle readcyclecounter with STCKF.
416
418 // Assume by default that all vector operations need to be expanded.
419 for (unsigned Opcode = 0; Opcode < ISD::BUILTIN_OP_END; ++Opcode)
420 if (getOperationAction(Opcode, VT) == Legal)
421 setOperationAction(Opcode, VT, Expand);
422
423 // Likewise all truncating stores and extending loads.
424 for (MVT InnerVT : MVT::fixedlen_vector_valuetypes()) {
425 setTruncStoreAction(VT, InnerVT, Expand);
428 setLoadExtAction(ISD::EXTLOAD, VT, InnerVT, Expand);
429 }
430
431 if (isTypeLegal(VT)) {
432 // These operations are legal for anything that can be stored in a
433 // vector register, even if there is no native support for the format
434 // as such. In particular, we can do these for v4f32 even though there
435 // are no specific instructions for that format.
441
442 // Likewise, except that we need to replace the nodes with something
443 // more specific.
446 }
447 }
448
449 // Handle integer vector types.
451 if (isTypeLegal(VT)) {
452 // These operations have direct equivalents.
457 if (VT != MVT::v2i64 || Subtarget.hasVectorEnhancements3()) {
461 }
462 if (Subtarget.hasVectorEnhancements3() &&
463 VT != MVT::v16i8 && VT != MVT::v8i16) {
468 }
473 if (Subtarget.hasVectorEnhancements1())
475 else
479
480 // Convert a GPR scalar to a vector by inserting it into element 0.
482
483 // Use a series of unpacks for extensions.
486
487 // Detect shifts/rotates by a scalar amount and convert them into
488 // V*_BY_SCALAR.
493
494 // Add ISD::VECREDUCE_ADD as custom in order to implement
495 // it with VZERO+VSUM
497
498 // Map SETCCs onto one of VCE, VCH or VCHL, swapping the operands
499 // and inverting the result as necessary.
501
503 Legal);
504 }
505 }
506
507 if (Subtarget.hasVector()) {
508 // There should be no need to check for float types other than v2f64
509 // since <2 x f32> isn't a legal type.
518
527 }
528
529 if (Subtarget.hasVectorEnhancements2()) {
538
547 }
548
549 // Handle floating-point types.
550 if (!useSoftFloat()) {
551 // Promote all f16 operations to float, with some exceptions below.
552 for (unsigned Opc = 0; Opc < ISD::BUILTIN_OP_END; ++Opc)
553 setOperationAction(Opc, MVT::f16, Promote);
555 for (MVT VT : {MVT::f32, MVT::f64, MVT::f128}) {
556 setLoadExtAction(ISD::EXTLOAD, VT, MVT::f16, Expand);
557 setTruncStoreAction(VT, MVT::f16, Expand);
558 }
560 setOperationAction(Op, MVT::f16, Subtarget.hasVector() ? Legal : Custom);
564
565 for (auto Op : {ISD::FNEG, ISD::FABS, ISD::FCOPYSIGN})
566 setOperationAction(Op, MVT::f16, Legal);
567 }
568
569 for (unsigned I = MVT::FIRST_FP_VALUETYPE;
570 I <= MVT::LAST_FP_VALUETYPE;
571 ++I) {
573 if (isTypeLegal(VT) && VT != MVT::f16) {
574 // We can use FI for FRINT.
576
577 // We can use the extended form of FI for other rounding operations.
578 if (Subtarget.hasFPExtension()) {
585 }
586
587 // No special instructions for these.
593
594 // Special treatment.
596
597 // Handle constrained floating-point operations.
606 if (Subtarget.hasFPExtension()) {
613 }
614
615 // Extension from f16 needs libcall.
618 }
619 }
620
621 // Handle floating-point vector types.
622 if (Subtarget.hasVector()) {
623 // Scalar-to-vector conversion is just a subreg.
627
628 // Some insertions and extractions can be done directly but others
629 // need to go via integers.
636
637 // These operations have direct equivalents.
638 setOperationAction(ISD::FADD, MVT::v2f64, Legal);
639 setOperationAction(ISD::FNEG, MVT::v2f64, Legal);
640 setOperationAction(ISD::FSUB, MVT::v2f64, Legal);
641 setOperationAction(ISD::FMUL, MVT::v2f64, Legal);
642 setOperationAction(ISD::FMA, MVT::v2f64, Legal);
643 setOperationAction(ISD::FDIV, MVT::v2f64, Legal);
644 setOperationAction(ISD::FABS, MVT::v2f64, Legal);
645 setOperationAction(ISD::FSQRT, MVT::v2f64, Legal);
646 setOperationAction(ISD::FRINT, MVT::v2f64, Legal);
649 setOperationAction(ISD::FCEIL, MVT::v2f64, Legal);
653
654 // Handle constrained floating-point operations.
668
673 if (Subtarget.hasVectorEnhancements1()) {
676 }
677 }
678
679 // The vector enhancements facility 1 has instructions for these.
680 if (Subtarget.hasVectorEnhancements1()) {
681 setOperationAction(ISD::FADD, MVT::v4f32, Legal);
682 setOperationAction(ISD::FNEG, MVT::v4f32, Legal);
683 setOperationAction(ISD::FSUB, MVT::v4f32, Legal);
684 setOperationAction(ISD::FMUL, MVT::v4f32, Legal);
685 setOperationAction(ISD::FMA, MVT::v4f32, Legal);
686 setOperationAction(ISD::FDIV, MVT::v4f32, Legal);
687 setOperationAction(ISD::FABS, MVT::v4f32, Legal);
688 setOperationAction(ISD::FSQRT, MVT::v4f32, Legal);
689 setOperationAction(ISD::FRINT, MVT::v4f32, Legal);
692 setOperationAction(ISD::FCEIL, MVT::v4f32, Legal);
696
697 for (MVT Type : {MVT::f64, MVT::v2f64, MVT::f32, MVT::v4f32, MVT::f128}) {
706 }
707
708 // Handle constrained floating-point operations.
722 for (auto VT : { MVT::f32, MVT::f64, MVT::f128,
723 MVT::v4f32, MVT::v2f64 }) {
730 }
731 }
732
733 // We only have fused f128 multiply-addition on vector registers.
734 if (!Subtarget.hasVectorEnhancements1()) {
737 }
738
739 // We don't have a copysign instruction on vector registers.
740 if (Subtarget.hasVectorEnhancements1())
742
743 // Needed so that we don't try to implement f128 constant loads using
744 // a load-and-extend of a f80 constant (in cases where the constant
745 // would fit in an f80).
746 for (MVT VT : MVT::fp_valuetypes())
747 setLoadExtAction(ISD::EXTLOAD, VT, MVT::f80, Expand);
748
749 // We don't have extending load instruction on vector registers.
750 if (Subtarget.hasVectorEnhancements1()) {
751 setLoadExtAction(ISD::EXTLOAD, MVT::f128, MVT::f32, Expand);
752 setLoadExtAction(ISD::EXTLOAD, MVT::f128, MVT::f64, Expand);
753 }
754
755 // Floating-point truncation and stores need to be done separately.
756 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
757 setTruncStoreAction(MVT::f128, MVT::f32, Expand);
758 setTruncStoreAction(MVT::f128, MVT::f64, Expand);
759
760 // We have 64-bit FPR<->GPR moves, but need special handling for
761 // 32-bit forms.
762 if (!Subtarget.hasVector()) {
765 }
766
767 // VASTART and VACOPY need to deal with the SystemZ-specific varargs
768 // structure, but VAEND is a no-op.
772
773 if (Subtarget.isTargetzOS()) {
774 // Handle address space casts between mixed sized pointers.
777 }
778
780
781 // Codes for which we want to perform some z-specific combinations.
785 ISD::LOAD,
798 ISD::SRL,
799 ISD::SRA,
800 ISD::MUL,
801 ISD::SDIV,
802 ISD::UDIV,
803 ISD::SREM,
804 ISD::UREM,
807
808 // Handle intrinsics.
811
812 // We're not using SJLJ for exception handling, but they're implemented
813 // solely to support use of __builtin_setjmp / __builtin_longjmp.
816
817 // We want to use MVC in preference to even a single load/store pair.
818 MaxStoresPerMemcpy = Subtarget.hasVector() ? 2 : 0;
820
821 // Same with memmove.
822 MaxStoresPerMemmove = Subtarget.hasVector() ? 2 : 0;
824
825 // The main memset sequence is a byte store followed by an MVC.
826 // Two STC or MV..I stores win over that, but the kind of fused stores
827 // generated by target-independent code don't when the byte value is
828 // variable. E.g. "STC <reg>;MHI <reg>,257;STH <reg>" is not better
829 // than "STC;MVC". Handle the choice in target-specific code instead.
830 MaxStoresPerMemset = Subtarget.hasVector() ? 2 : 0;
832
833 // Default to having -disable-strictnode-mutation on
834 IsStrictFPEnabled = true;
835}
836
838 return Subtarget.hasSoftFloat();
839}
840
842 LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT,
843 unsigned &NumIntermediates, MVT &RegisterVT) const {
844 // Pass fp16 vectors in VR(s).
845 if (Subtarget.hasVector() && VT.isVectorOf(MVT::f16)) {
846 IntermediateVT = RegisterVT = MVT::v8f16;
847 return NumIntermediates =
849 }
851 Context, CC, VT, IntermediateVT, NumIntermediates, RegisterVT);
852}
853
856 EVT VT) const {
857 // 128-bit single-element vector types are passed like other vectors,
858 // not like their element type.
859 if (Subtarget.hasVector() && VT.isVector() && VT.getSizeInBits() == 128 &&
860 VT.getVectorNumElements() == 1)
861 return MVT::v16i8;
862 // Pass fp16 vectors in VR(s).
863 if (Subtarget.hasVector() && VT.isVectorOf(MVT::f16))
864 return MVT::v8f16;
865 return TargetLowering::getRegisterTypeForCallingConv(Context, CC, VT);
866}
867
869 LLVMContext &Context, CallingConv::ID CC, EVT VT) const {
870 // Pass fp16 vectors in VR(s).
871 if (Subtarget.hasVector() && VT.isVectorOf(MVT::f16))
873 return TargetLowering::getNumRegistersForCallingConv(Context, CC, VT);
874}
875
877 LLVMContext &, EVT VT) const {
878 if (!VT.isVector())
879 return MVT::i32;
881}
882
884 const MachineFunction &MF, EVT VT) const {
885 if (useSoftFloat())
886 return false;
887
888 VT = VT.getScalarType();
889
890 if (!VT.isSimple())
891 return false;
892
893 switch (VT.getSimpleVT().SimpleTy) {
894 case MVT::f32:
895 case MVT::f64:
896 return true;
897 case MVT::f128:
898 return Subtarget.hasVectorEnhancements1();
899 default:
900 break;
901 }
902
903 return false;
904}
905
906// Return true if the constant can be generated with a vector instruction,
907// such as VGM, VGMB or VREPI.
909 const SystemZSubtarget &Subtarget) {
910 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
911 if (!Subtarget.hasVector() ||
912 (isFP128 && !Subtarget.hasVectorEnhancements1()))
913 return false;
914
915 // Try using VECTOR GENERATE BYTE MASK. This is the architecturally-
916 // preferred way of creating all-zero and all-one vectors so give it
917 // priority over other methods below.
918 unsigned Mask = 0;
919 unsigned I = 0;
920 for (; I < SystemZ::VectorBytes; ++I) {
921 uint64_t Byte = IntBits.lshr(I * 8).trunc(8).getZExtValue();
922 if (Byte == 0xff)
923 Mask |= 1ULL << I;
924 else if (Byte != 0)
925 break;
926 }
927 if (I == SystemZ::VectorBytes) {
928 Opcode = SystemZISD::BYTE_MASK;
929 OpVals.push_back(Mask);
931 return true;
932 }
933
934 if (SplatBitSize > 64)
935 return false;
936
937 auto TryValue = [&](uint64_t Value) -> bool {
938 // Try VECTOR REPLICATE IMMEDIATE
939 int64_t SignedValue = SignExtend64(Value, SplatBitSize);
940 if (isInt<16>(SignedValue)) {
941 OpVals.push_back(((unsigned) SignedValue));
942 Opcode = SystemZISD::REPLICATE;
944 SystemZ::VectorBits / SplatBitSize);
945 return true;
946 }
947 // Try VECTOR GENERATE MASK
948 unsigned Start, End;
949 if (TII->isRxSBGMask(Value, SplatBitSize, Start, End)) {
950 // isRxSBGMask returns the bit numbers for a full 64-bit value, with 0
951 // denoting 1 << 63 and 63 denoting 1. Convert them to bit numbers for
952 // an SplatBitSize value, so that 0 denotes 1 << (SplatBitSize-1).
953 OpVals.push_back(Start - (64 - SplatBitSize));
954 OpVals.push_back(End - (64 - SplatBitSize));
955 Opcode = SystemZISD::ROTATE_MASK;
957 SystemZ::VectorBits / SplatBitSize);
958 return true;
959 }
960 return false;
961 };
962
963 // First try assuming that any undefined bits above the highest set bit
964 // and below the lowest set bit are 1s. This increases the likelihood of
965 // being able to use a sign-extended element value in VECTOR REPLICATE
966 // IMMEDIATE or a wraparound mask in VECTOR GENERATE MASK.
967 uint64_t SplatBitsZ = SplatBits.getZExtValue();
968 uint64_t SplatUndefZ = SplatUndef.getZExtValue();
969 unsigned LowerBits = llvm::countr_zero(SplatBitsZ);
970 unsigned UpperBits = llvm::countl_zero(SplatBitsZ);
971 uint64_t Lower = SplatUndefZ & maskTrailingOnes<uint64_t>(LowerBits);
972 uint64_t Upper = SplatUndefZ & maskLeadingOnes<uint64_t>(UpperBits);
973 if (TryValue(SplatBitsZ | Upper | Lower))
974 return true;
975
976 // Now try assuming that any undefined bits between the first and
977 // last defined set bits are set. This increases the chances of
978 // using a non-wraparound mask.
979 uint64_t Middle = SplatUndefZ & ~Upper & ~Lower;
980 return TryValue(SplatBitsZ | Middle);
981}
982
984 if (IntImm.isSingleWord()) {
985 IntBits = APInt(128, IntImm.getZExtValue());
986 IntBits <<= (SystemZ::VectorBits - IntImm.getBitWidth());
987 } else
988 IntBits = IntImm;
989 assert(IntBits.getBitWidth() == 128 && "Unsupported APInt.");
990
991 // Find the smallest splat.
992 SplatBits = IntImm;
993 unsigned Width = SplatBits.getBitWidth();
994 while (Width > 8) {
995 unsigned HalfSize = Width / 2;
996 APInt HighValue = SplatBits.lshr(HalfSize).trunc(HalfSize);
997 APInt LowValue = SplatBits.trunc(HalfSize);
998
999 // If the two halves do not match, stop here.
1000 if (HighValue != LowValue || 8 > HalfSize)
1001 break;
1002
1003 SplatBits = HighValue;
1004 Width = HalfSize;
1005 }
1006 SplatUndef = 0;
1007 SplatBitSize = Width;
1008}
1009
1011 assert(BVN->isConstant() && "Expected a constant BUILD_VECTOR");
1012 bool HasAnyUndefs;
1013
1014 // Get IntBits by finding the 128 bit splat.
1015 BVN->isConstantSplat(IntBits, SplatUndef, SplatBitSize, HasAnyUndefs, 128,
1016 true);
1017
1018 // Get SplatBits by finding the 8 bit or greater splat.
1019 BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs, 8,
1020 true);
1021}
1022
1024 bool ForCodeSize) const {
1025 // We can load zero using LZ?R and negative zero using LZ?R;LC?BR.
1026 if (Imm.isZero() || Imm.isNegZero())
1027 return true;
1028
1030}
1031
1034 MachineBasicBlock *MBB) const {
1035 DebugLoc DL = MI.getDebugLoc();
1036 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
1037 const SystemZRegisterInfo *TRI = Subtarget.getRegisterInfo();
1038
1039 MachineFunction *MF = MBB->getParent();
1040 MachineRegisterInfo &MRI = MF->getRegInfo();
1041
1042 const BasicBlock *BB = MBB->getBasicBlock();
1043 MachineFunction::iterator I = ++MBB->getIterator();
1044
1045 Register DstReg = MI.getOperand(0).getReg();
1046 const TargetRegisterClass *RC = MRI.getRegClass(DstReg);
1047 assert(TRI->isTypeLegalForClass(*RC, MVT::i32) && "Invalid destination!");
1048 (void)TRI;
1049 Register MainDstReg = MRI.createVirtualRegister(RC);
1050 Register RestoreDstReg = MRI.createVirtualRegister(RC);
1051
1052 MVT PVT = getPointerTy(MF->getDataLayout());
1053 assert((PVT == MVT::i64 || PVT == MVT::i32) && "Invalid Pointer Size!");
1054 // For v = setjmp(buf), we generate.
1055 // Algorithm:
1056 //
1057 // ---------
1058 // | thisMBB |
1059 // ---------
1060 // |
1061 // ------------------------
1062 // | |
1063 // ---------- ---------------
1064 // | mainMBB | | restoreMBB |
1065 // | v = 0 | | v = 1 |
1066 // ---------- ---------------
1067 // | |
1068 // -------------------------
1069 // |
1070 // -----------------------------
1071 // | sinkMBB |
1072 // | phi(v_mainMBB,v_restoreMBB) |
1073 // -----------------------------
1074 // thisMBB:
1075 // buf[FPOffset] = Frame Pointer if hasFP.
1076 // buf[LabelOffset] = restoreMBB <-- takes address of restoreMBB.
1077 // buf[BCOffset] = Backchain value if building with -mbackchain.
1078 // buf[SPOffset] = Stack Pointer.
1079 // buf[LPOffset] = We never write this slot with R13, gcc stores R13 always.
1080 // SjLjSetup restoreMBB
1081 // mainMBB:
1082 // v_main = 0
1083 // sinkMBB:
1084 // v = phi(v_main, v_restore)
1085 // restoreMBB:
1086 // v_restore = 1
1087
1088 MachineBasicBlock *ThisMBB = MBB;
1089 MachineBasicBlock *MainMBB = MF->CreateMachineBasicBlock(BB);
1090 MachineBasicBlock *SinkMBB = MF->CreateMachineBasicBlock(BB);
1091 MachineBasicBlock *RestoreMBB = MF->CreateMachineBasicBlock(BB);
1092
1093 MF->insert(I, MainMBB);
1094 MF->insert(I, SinkMBB);
1095 MF->push_back(RestoreMBB);
1096 RestoreMBB->setMachineBlockAddressTaken();
1097
1099
1100 // Transfer the remainder of BB and its successor edges to sinkMBB.
1101 SinkMBB->splice(SinkMBB->begin(), MBB,
1102 std::next(MachineBasicBlock::iterator(MI)), MBB->end());
1104
1105 // thisMBB:
1106 const int64_t FPOffset = 0; // Slot 1.
1107 const int64_t LabelOffset = 1 * PVT.getStoreSize(); // Slot 2.
1108 const int64_t BCOffset = 2 * PVT.getStoreSize(); // Slot 3.
1109 const int64_t SPOffset = 3 * PVT.getStoreSize(); // Slot 4.
1110
1111 // Buf address.
1112 Register BufReg = MI.getOperand(1).getReg();
1113
1114 const TargetRegisterClass *PtrRC = getRegClassFor(PVT);
1115 Register LabelReg = MRI.createVirtualRegister(PtrRC);
1116
1117 // Prepare IP for longjmp.
1118 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::LARL), LabelReg)
1119 .addMBB(RestoreMBB);
1120 // Store IP for return from jmp, slot 2, offset = 1.
1121 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::STG))
1122 .addReg(LabelReg)
1123 .addReg(BufReg)
1124 .addImm(LabelOffset)
1125 .addReg(0);
1126
1127 auto *SpecialRegs = Subtarget.getSpecialRegisters();
1128 bool HasFP = Subtarget.getFrameLowering()->hasFP(*MF);
1129 if (HasFP) {
1130 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::STG))
1131 .addReg(SpecialRegs->getFramePointerRegister())
1132 .addReg(BufReg)
1133 .addImm(FPOffset)
1134 .addReg(0);
1135 }
1136
1137 // Store SP.
1138 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::STG))
1139 .addReg(SpecialRegs->getStackPointerRegister())
1140 .addReg(BufReg)
1141 .addImm(SPOffset)
1142 .addReg(0);
1143
1144 // Slot 3(Offset = 2) Backchain value (if building with -mbackchain).
1145 bool BackChain = MF->getSubtarget<SystemZSubtarget>().hasBackChain();
1146 if (BackChain) {
1147 Register BCReg = MRI.createVirtualRegister(PtrRC);
1148 auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
1149 MIB = BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::LG), BCReg)
1150 .addReg(SpecialRegs->getStackPointerRegister())
1151 .addImm(TFL->getBackchainOffset(*MF))
1152 .addReg(0);
1153
1154 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::STG))
1155 .addReg(BCReg)
1156 .addReg(BufReg)
1157 .addImm(BCOffset)
1158 .addReg(0);
1159 }
1160
1161 // Setup.
1162 MIB = BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::EH_SjLj_Setup))
1163 .addMBB(RestoreMBB);
1164
1165 const SystemZRegisterInfo *RegInfo = Subtarget.getRegisterInfo();
1166 MIB.addRegMask(RegInfo->getNoPreservedMask());
1167
1168 ThisMBB->addSuccessor(MainMBB);
1169 ThisMBB->addSuccessor(RestoreMBB);
1170
1171 // mainMBB:
1172 BuildMI(MainMBB, DL, TII->get(SystemZ::LHI), MainDstReg).addImm(0);
1173 MainMBB->addSuccessor(SinkMBB);
1174
1175 // sinkMBB:
1176 BuildMI(*SinkMBB, SinkMBB->begin(), DL, TII->get(SystemZ::PHI), DstReg)
1177 .addReg(MainDstReg)
1178 .addMBB(MainMBB)
1179 .addReg(RestoreDstReg)
1180 .addMBB(RestoreMBB);
1181
1182 // restoreMBB.
1183 BuildMI(RestoreMBB, DL, TII->get(SystemZ::LHI), RestoreDstReg).addImm(1);
1184 BuildMI(RestoreMBB, DL, TII->get(SystemZ::J)).addMBB(SinkMBB);
1185 RestoreMBB->addSuccessor(SinkMBB);
1186
1187 MI.eraseFromParent();
1188
1189 return SinkMBB;
1190}
1191
1194 MachineBasicBlock *MBB) const {
1195
1196 DebugLoc DL = MI.getDebugLoc();
1197 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
1198
1199 MachineFunction *MF = MBB->getParent();
1200 MachineRegisterInfo &MRI = MF->getRegInfo();
1201
1202 MVT PVT = getPointerTy(MF->getDataLayout());
1203 assert((PVT == MVT::i64 || PVT == MVT::i32) && "Invalid Pointer Size!");
1204 Register BufReg = MI.getOperand(0).getReg();
1205 const TargetRegisterClass *RC = MRI.getRegClass(BufReg);
1206 auto *SpecialRegs = Subtarget.getSpecialRegisters();
1207
1208 Register Tmp = MRI.createVirtualRegister(RC);
1209 Register BCReg = MRI.createVirtualRegister(RC);
1210
1212
1213 const int64_t FPOffset = 0;
1214 const int64_t LabelOffset = 1 * PVT.getStoreSize();
1215 const int64_t BCOffset = 2 * PVT.getStoreSize();
1216 const int64_t SPOffset = 3 * PVT.getStoreSize();
1217 const int64_t LPOffset = 4 * PVT.getStoreSize();
1218
1219 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG), Tmp)
1220 .addReg(BufReg)
1221 .addImm(LabelOffset)
1222 .addReg(0);
1223
1224 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG),
1225 SpecialRegs->getFramePointerRegister())
1226 .addReg(BufReg)
1227 .addImm(FPOffset)
1228 .addReg(0);
1229
1230 // We are restoring R13 even though we never stored in setjmp from llvm,
1231 // as gcc always stores R13 in builtin_setjmp. We could have mixed code
1232 // gcc setjmp and llvm longjmp.
1233 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG), SystemZ::R13D)
1234 .addReg(BufReg)
1235 .addImm(LPOffset)
1236 .addReg(0);
1237
1238 bool BackChain = MF->getSubtarget<SystemZSubtarget>().hasBackChain();
1239 if (BackChain) {
1240 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG), BCReg)
1241 .addReg(BufReg)
1242 .addImm(BCOffset)
1243 .addReg(0);
1244 }
1245
1246 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG),
1247 SpecialRegs->getStackPointerRegister())
1248 .addReg(BufReg)
1249 .addImm(SPOffset)
1250 .addReg(0);
1251
1252 if (BackChain) {
1253 auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
1254 BuildMI(*MBB, MI, DL, TII->get(SystemZ::STG))
1255 .addReg(BCReg)
1256 .addReg(SpecialRegs->getStackPointerRegister())
1257 .addImm(TFL->getBackchainOffset(*MF))
1258 .addReg(0);
1259 }
1260
1261 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::BR)).addReg(Tmp);
1262
1263 MI.eraseFromParent();
1264 return MBB;
1265}
1266
1267/// Returns true if stack probing through inline assembly is requested.
1269 // If the function specifically requests inline stack probes, emit them.
1270 if (MF.getFunction().hasFnAttribute("probe-stack"))
1271 return MF.getFunction().getFnAttribute("probe-stack").getValueAsString() ==
1272 "inline-asm";
1273 return false;
1274}
1275
1280
1285
1288 const AtomicRMWInst *RMW) const {
1289 // Don't expand subword operations as they require special treatment.
1290 if (RMW->getType()->isIntegerTy(8) || RMW->getType()->isIntegerTy(16))
1292
1293 // Don't expand if there is a target instruction available.
1294 if (Subtarget.hasInterlockedAccess1() &&
1295 (RMW->getType()->isIntegerTy(32) || RMW->getType()->isIntegerTy(64)) &&
1302
1304}
1305
1307 // We can use CGFI or CLGFI.
1308 return isInt<32>(Imm) || isUInt<32>(Imm);
1309}
1310
1312 // We can use ALGFI or SLGFI.
1313 return isUInt<32>(Imm) || isUInt<32>(-Imm);
1314}
1315
1317 EVT VT, unsigned, Align, MachineMemOperand::Flags, unsigned *Fast) const {
1318 // Unaligned accesses should never be slower than the expanded version.
1319 // We check specifically for aligned accesses in the few cases where
1320 // they are required.
1321 if (Fast)
1322 *Fast = 1;
1323 return true;
1324}
1325
1327 EVT VT = Y.getValueType();
1328
1329 // We can use NC(G)RK for types in GPRs ...
1330 if (VT == MVT::i32 || VT == MVT::i64)
1331 return Subtarget.hasMiscellaneousExtensions3();
1332
1333 // ... or VNC for types in VRs.
1334 if (VT.isVector() || VT == MVT::i128)
1335 return Subtarget.hasVector();
1336
1337 return false;
1338}
1339
1340// Information about the addressing mode for a memory access.
1342 // True if a long displacement is supported.
1344
1345 // True if use of index register is supported.
1347
1348 AddressingMode(bool LongDispl, bool IdxReg) :
1349 LongDisplacement(LongDispl), IndexReg(IdxReg) {}
1350};
1351
1352// Return the desired addressing mode for a Load which has only one use (in
1353// the same block) which is a Store.
1355 Type *Ty) {
1356 // With vector support a Load->Store combination may be combined to either
1357 // an MVC or vector operations and it seems to work best to allow the
1358 // vector addressing mode.
1359 if (HasVector)
1360 return AddressingMode(false/*LongDispl*/, true/*IdxReg*/);
1361
1362 // Otherwise only the MVC case is special.
1363 bool MVC = Ty->isIntegerTy(8);
1364 return AddressingMode(!MVC/*LongDispl*/, !MVC/*IdxReg*/);
1365}
1366
1367// Return the addressing mode which seems most desirable given an LLVM
1368// Instruction pointer.
1369static AddressingMode
1372 switch (II->getIntrinsicID()) {
1373 default: break;
1374 case Intrinsic::memset:
1375 case Intrinsic::memmove:
1376 case Intrinsic::memcpy:
1377 return AddressingMode(false/*LongDispl*/, false/*IdxReg*/);
1378 }
1379 }
1380
1381 if (isa<LoadInst>(I) && I->hasOneUse()) {
1382 auto *SingleUser = cast<Instruction>(*I->user_begin());
1383 if (SingleUser->getParent() == I->getParent()) {
1384 if (isa<ICmpInst>(SingleUser)) {
1385 if (auto *C = dyn_cast<ConstantInt>(SingleUser->getOperand(1)))
1386 if (C->getBitWidth() <= 64 &&
1387 (isInt<16>(C->getSExtValue()) || isUInt<16>(C->getZExtValue())))
1388 // Comparison of memory with 16 bit signed / unsigned immediate
1389 return AddressingMode(false/*LongDispl*/, false/*IdxReg*/);
1390 } else if (isa<StoreInst>(SingleUser))
1391 // Load->Store
1392 return getLoadStoreAddrMode(HasVector, I->getType());
1393 }
1394 } else if (auto *StoreI = dyn_cast<StoreInst>(I)) {
1395 if (auto *LoadI = dyn_cast<LoadInst>(StoreI->getValueOperand()))
1396 if (LoadI->hasOneUse() && LoadI->getParent() == I->getParent())
1397 // Load->Store
1398 return getLoadStoreAddrMode(HasVector, LoadI->getType());
1399 }
1400
1401 if (HasVector && (isa<LoadInst>(I) || isa<StoreInst>(I))) {
1402
1403 // * Use LDE instead of LE/LEY for z13 to avoid partial register
1404 // dependencies (LDE only supports small offsets).
1405 // * Utilize the vector registers to hold floating point
1406 // values (vector load / store instructions only support small
1407 // offsets).
1408
1409 Type *MemAccessTy = (isa<LoadInst>(I) ? I->getType() :
1410 I->getOperand(0)->getType());
1411 bool IsFPAccess = MemAccessTy->isFloatingPointTy();
1412 bool IsVectorAccess = MemAccessTy->isVectorTy();
1413
1414 // A store of an extracted vector element will be combined into a VSTE type
1415 // instruction.
1416 if (!IsVectorAccess && isa<StoreInst>(I)) {
1417 Value *DataOp = I->getOperand(0);
1418 if (isa<ExtractElementInst>(DataOp))
1419 IsVectorAccess = true;
1420 }
1421
1422 // A load which gets inserted into a vector element will be combined into a
1423 // VLE type instruction.
1424 if (!IsVectorAccess && isa<LoadInst>(I) && I->hasOneUse()) {
1425 User *LoadUser = *I->user_begin();
1426 if (isa<InsertElementInst>(LoadUser))
1427 IsVectorAccess = true;
1428 }
1429
1430 if (IsFPAccess || IsVectorAccess)
1431 return AddressingMode(false/*LongDispl*/, true/*IdxReg*/);
1432 }
1433
1434 return AddressingMode(true/*LongDispl*/, true/*IdxReg*/);
1435}
1436
1438 const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I) const {
1439 // Punt on globals for now, although they can be used in limited
1440 // RELATIVE LONG cases.
1441 if (AM.BaseGV)
1442 return false;
1443
1444 // Require a 20-bit signed offset.
1445 if (!isInt<20>(AM.BaseOffs))
1446 return false;
1447
1448 bool RequireD12 =
1449 Subtarget.hasVector() && (Ty->isVectorTy() || Ty->isIntegerTy(128));
1450 AddressingMode SupportedAM(!RequireD12, true);
1451 if (I != nullptr)
1452 SupportedAM = supportedAddressingMode(I, Subtarget.hasVector());
1453
1454 if (!SupportedAM.LongDisplacement && !isUInt<12>(AM.BaseOffs))
1455 return false;
1456
1457 if (!SupportedAM.IndexReg)
1458 // No indexing allowed.
1459 return AM.Scale == 0;
1460 else
1461 // Indexing is OK but no scale factor can be applied.
1462 return AM.Scale == 0 || AM.Scale == 1;
1463}
1464
1466 LLVMContext &Context, std::vector<EVT> &MemOps, unsigned Limit,
1467 const MemOp &Op, unsigned DstAS, unsigned SrcAS,
1468 const AttributeList &FuncAttributes, EVT *LargestVT) const {
1469
1470 assert(Limit != ~0U &&
1471 "Expected EmitTargetCodeForMemXXX() to handle AlwaysInline cases.");
1472
1473 if (Op.isZeroMemset())
1474 return false; // Memset zero: Use XC.
1475
1476 const int MVCFastLen = 16;
1477 // Use MVC up to 16 bytes for memcpy. Small memset uses STC/MVI for first
1478 // byte.
1479 if (Op.isMemcpy() && Op.size() <= MVCFastLen)
1480 return false;
1481 if (Op.isMemset() && Op.size() - 1 <= MVCFastLen)
1482 return false;
1483
1484 // Avoid unaligned VL/VST:s.
1485 if ((Op.size() >= 16 && !Op.isAligned(Align(8))) ||
1486 (Op.size() >= 25 && Op.size() <= 31))
1487 return false;
1488
1490 Context, MemOps, Limit, Op, DstAS, SrcAS, FuncAttributes, LargestVT);
1491}
1492
1494 LLVMContext &Context, const MemOp &Op,
1495 const AttributeList &FuncAttributes) const {
1496 return Subtarget.hasVector() ? MVT::v2i64 : MVT::Other;
1497}
1498
1499bool SystemZTargetLowering::isTruncateFree(Type *FromType, Type *ToType) const {
1500 if (!FromType->isIntegerTy() || !ToType->isIntegerTy())
1501 return false;
1502 unsigned FromBits = FromType->getPrimitiveSizeInBits().getFixedValue();
1503 unsigned ToBits = ToType->getPrimitiveSizeInBits().getFixedValue();
1504 return FromBits > ToBits;
1505}
1506
1508 if (!FromVT.isInteger() || !ToVT.isInteger())
1509 return false;
1510 unsigned FromBits = FromVT.getFixedSizeInBits();
1511 unsigned ToBits = ToVT.getFixedSizeInBits();
1512 return FromBits > ToBits;
1513}
1514
1515//===----------------------------------------------------------------------===//
1516// Inline asm support
1517//===----------------------------------------------------------------------===//
1518
1521 if (Constraint.size() == 1) {
1522 switch (Constraint[0]) {
1523 case 'a': // Address register
1524 case 'd': // Data register (equivalent to 'r')
1525 case 'f': // Floating-point register
1526 case 'h': // High-part register
1527 case 'r': // General-purpose register
1528 case 'v': // Vector register
1529 return C_RegisterClass;
1530
1531 case 'Q': // Memory with base and unsigned 12-bit displacement
1532 case 'R': // Likewise, plus an index
1533 case 'S': // Memory with base and signed 20-bit displacement
1534 case 'T': // Likewise, plus an index
1535 case 'm': // Equivalent to 'T'.
1536 return C_Memory;
1537
1538 case 'I': // Unsigned 8-bit constant
1539 case 'J': // Unsigned 12-bit constant
1540 case 'K': // Signed 16-bit constant
1541 case 'L': // Signed 20-bit displacement (on all targets we support)
1542 case 'M': // 0x7fffffff
1543 return C_Immediate;
1544
1545 default:
1546 break;
1547 }
1548 } else if (Constraint.size() == 2 && Constraint[0] == 'Z') {
1549 switch (Constraint[1]) {
1550 case 'Q': // Address with base and unsigned 12-bit displacement
1551 case 'R': // Likewise, plus an index
1552 case 'S': // Address with base and signed 20-bit displacement
1553 case 'T': // Likewise, plus an index
1554 return C_Address;
1555
1556 default:
1557 break;
1558 }
1559 } else if (Constraint.size() == 5 && Constraint.starts_with("{")) {
1560 if (StringRef("{@cc}").compare(Constraint) == 0)
1561 return C_Other;
1562 }
1563 return TargetLowering::getConstraintType(Constraint);
1564}
1565
1568 AsmOperandInfo &Info, const char *Constraint) const {
1570 Value *CallOperandVal = Info.CallOperandVal;
1571 // If we don't have a value, we can't do a match,
1572 // but allow it at the lowest weight.
1573 if (!CallOperandVal)
1574 return CW_Default;
1575 Type *type = CallOperandVal->getType();
1576 // Look at the constraint type.
1577 switch (*Constraint) {
1578 default:
1579 Weight = TargetLowering::getSingleConstraintMatchWeight(Info, Constraint);
1580 break;
1581
1582 case 'a': // Address register
1583 case 'd': // Data register (equivalent to 'r')
1584 case 'h': // High-part register
1585 case 'r': // General-purpose register
1586 Weight =
1587 CallOperandVal->getType()->isIntegerTy() ? CW_Register : CW_Default;
1588 break;
1589
1590 case 'f': // Floating-point register
1591 if (!useSoftFloat())
1592 Weight = type->isFloatingPointTy() ? CW_Register : CW_Default;
1593 break;
1594
1595 case 'v': // Vector register
1596 if (Subtarget.hasVector())
1597 Weight = (type->isVectorTy() || type->isFloatingPointTy()) ? CW_Register
1598 : CW_Default;
1599 break;
1600
1601 case 'I': // Unsigned 8-bit constant
1602 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1603 if (isUInt<8>(C->getZExtValue()))
1604 Weight = CW_Constant;
1605 break;
1606
1607 case 'J': // Unsigned 12-bit constant
1608 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1609 if (isUInt<12>(C->getZExtValue()))
1610 Weight = CW_Constant;
1611 break;
1612
1613 case 'K': // Signed 16-bit constant
1614 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1615 if (isInt<16>(C->getSExtValue()))
1616 Weight = CW_Constant;
1617 break;
1618
1619 case 'L': // Signed 20-bit displacement (on all targets we support)
1620 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1621 if (isInt<20>(C->getSExtValue()))
1622 Weight = CW_Constant;
1623 break;
1624
1625 case 'M': // 0x7fffffff
1626 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1627 if (C->getZExtValue() == 0x7fffffff)
1628 Weight = CW_Constant;
1629 break;
1630 }
1631 return Weight;
1632}
1633
1634// Parse a "{tNNN}" register constraint for which the register type "t"
1635// has already been verified. MC is the class associated with "t" and
1636// Map maps 0-based register numbers to LLVM register numbers.
1637static std::pair<unsigned, const TargetRegisterClass *>
1639 const unsigned *Map, unsigned Size) {
1640 assert(*(Constraint.end()-1) == '}' && "Missing '}'");
1641 if (isdigit(Constraint[2])) {
1642 unsigned Index;
1643 bool Failed =
1644 Constraint.slice(2, Constraint.size() - 1).getAsInteger(10, Index);
1645 if (!Failed && Index < Size && Map[Index])
1646 return std::make_pair(Map[Index], RC);
1647 }
1648 return std::make_pair(0U, nullptr);
1649}
1650
1651std::pair<unsigned, const TargetRegisterClass *>
1653 const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const {
1654 if (Constraint.size() == 1) {
1655 // GCC Constraint Letters
1656 switch (Constraint[0]) {
1657 default: break;
1658 case 'd': // Data register (equivalent to 'r')
1659 case 'r': // General-purpose register
1660 if (VT.getSizeInBits() == 64)
1661 return std::make_pair(0U, &SystemZ::GR64BitRegClass);
1662 else if (VT.getSizeInBits() == 128)
1663 return std::make_pair(0U, &SystemZ::GR128BitRegClass);
1664 return std::make_pair(0U, &SystemZ::GR32BitRegClass);
1665
1666 case 'a': // Address register
1667 if (VT == MVT::i64)
1668 return std::make_pair(0U, &SystemZ::ADDR64BitRegClass);
1669 else if (VT == MVT::i128)
1670 return std::make_pair(0U, &SystemZ::ADDR128BitRegClass);
1671 return std::make_pair(0U, &SystemZ::ADDR32BitRegClass);
1672
1673 case 'h': // High-part register (an LLVM extension)
1674 return std::make_pair(0U, &SystemZ::GRH32BitRegClass);
1675
1676 case 'f': // Floating-point register
1677 if (!useSoftFloat()) {
1678 if (VT.getSizeInBits() == 16)
1679 return std::make_pair(0U, &SystemZ::FP16BitRegClass);
1680 else if (VT.getSizeInBits() == 64)
1681 return std::make_pair(0U, &SystemZ::FP64BitRegClass);
1682 else if (VT.getSizeInBits() == 128)
1683 return std::make_pair(0U, &SystemZ::FP128BitRegClass);
1684 return std::make_pair(0U, &SystemZ::FP32BitRegClass);
1685 }
1686 break;
1687
1688 case 'v': // Vector register
1689 if (Subtarget.hasVector()) {
1690 if (VT.getSizeInBits() == 16)
1691 return std::make_pair(0U, &SystemZ::VR16BitRegClass);
1692 if (VT.getSizeInBits() == 32)
1693 return std::make_pair(0U, &SystemZ::VR32BitRegClass);
1694 if (VT.getSizeInBits() == 64)
1695 return std::make_pair(0U, &SystemZ::VR64BitRegClass);
1696 return std::make_pair(0U, &SystemZ::VR128BitRegClass);
1697 }
1698 break;
1699 }
1700 }
1701 if (Constraint.starts_with("{")) {
1702
1703 // A clobber constraint (e.g. ~{f0}) will have MVT::Other which is illegal
1704 // to check the size on.
1705 auto getVTSizeInBits = [&VT]() {
1706 return VT == MVT::Other ? 0 : VT.getSizeInBits();
1707 };
1708
1709 // We need to override the default register parsing for GPRs and FPRs
1710 // because the interpretation depends on VT. The internal names of
1711 // the registers are also different from the external names
1712 // (F0D and F0S instead of F0, etc.).
1713 if (Constraint[1] == 'r') {
1714 if (getVTSizeInBits() == 32)
1715 return parseRegisterNumber(Constraint, &SystemZ::GR32BitRegClass,
1717 if (getVTSizeInBits() == 128)
1718 return parseRegisterNumber(Constraint, &SystemZ::GR128BitRegClass,
1720 return parseRegisterNumber(Constraint, &SystemZ::GR64BitRegClass,
1722 }
1723 if (Constraint[1] == 'f') {
1724 if (useSoftFloat())
1725 return std::make_pair(
1726 0u, static_cast<const TargetRegisterClass *>(nullptr));
1727 if (getVTSizeInBits() == 16)
1728 return parseRegisterNumber(Constraint, &SystemZ::FP16BitRegClass,
1730 if (getVTSizeInBits() == 32)
1731 return parseRegisterNumber(Constraint, &SystemZ::FP32BitRegClass,
1733 if (getVTSizeInBits() == 128)
1734 return parseRegisterNumber(Constraint, &SystemZ::FP128BitRegClass,
1736 return parseRegisterNumber(Constraint, &SystemZ::FP64BitRegClass,
1738 }
1739 if (Constraint[1] == 'v') {
1740 if (!Subtarget.hasVector())
1741 return std::make_pair(
1742 0u, static_cast<const TargetRegisterClass *>(nullptr));
1743 if (getVTSizeInBits() == 16)
1744 return parseRegisterNumber(Constraint, &SystemZ::VR16BitRegClass,
1746 if (getVTSizeInBits() == 32)
1747 return parseRegisterNumber(Constraint, &SystemZ::VR32BitRegClass,
1749 if (getVTSizeInBits() == 64)
1750 return parseRegisterNumber(Constraint, &SystemZ::VR64BitRegClass,
1752 return parseRegisterNumber(Constraint, &SystemZ::VR128BitRegClass,
1754 }
1755 if (Constraint[1] == '@') {
1756 if (StringRef("{@cc}").compare(Constraint) == 0)
1757 return std::make_pair(SystemZ::CC, &SystemZ::CCRRegClass);
1758 }
1759 }
1760 return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
1761}
1762
1763// FIXME? Maybe this could be a TableGen attribute on some registers and
1764// this table could be generated automatically from RegInfo.
1767 const MachineFunction &MF) const {
1768 Register Reg =
1770 .Case("r4", Subtarget.isTargetXPLINK64() ? SystemZ::R4D
1771 : SystemZ::NoRegister)
1772 .Case("r15",
1773 Subtarget.isTargetELF() ? SystemZ::R15D : SystemZ::NoRegister)
1774 .Default(Register());
1775
1776 return Reg;
1777}
1778
1780 ExceptionHandling EH, const Constant *PersonalityFn) const {
1781 return Subtarget.isTargetXPLINK64() ? SystemZ::R1D : SystemZ::R6D;
1782}
1783
1785 ExceptionHandling EH, const Constant *PersonalityFn) const {
1786 return Subtarget.isTargetXPLINK64() ? SystemZ::R2D : SystemZ::R7D;
1787}
1788
1789// Convert condition code in CCReg to an i32 value.
1791 SDLoc DL(CCReg);
1792 SDValue IPM = DAG.getNode(SystemZISD::IPM, DL, MVT::i32, CCReg);
1793 return DAG.getNode(ISD::SRL, DL, MVT::i32, IPM,
1794 DAG.getConstant(SystemZ::IPM_CC, DL, MVT::i32));
1795}
1796
1797// Lower @cc targets via setcc.
1799 SDValue &Chain, SDValue &Glue, const SDLoc &DL,
1800 const AsmOperandInfo &OpInfo, SelectionDAG &DAG) const {
1801 if (StringRef("{@cc}").compare(OpInfo.ConstraintCode) != 0)
1802 return SDValue();
1803
1804 // Check that return type is valid.
1805 if (OpInfo.ConstraintVT.isVector() || !OpInfo.ConstraintVT.isInteger() ||
1806 OpInfo.ConstraintVT.getSizeInBits() < 8)
1807 report_fatal_error("Glue output operand is of invalid type");
1808
1809 if (Glue.getNode()) {
1810 Glue = DAG.getCopyFromReg(Chain, DL, SystemZ::CC, MVT::i32, Glue);
1811 Chain = Glue.getValue(1);
1812 } else
1813 Glue = DAG.getCopyFromReg(Chain, DL, SystemZ::CC, MVT::i32);
1814 return getCCResult(DAG, Glue);
1815}
1816
1818 SDValue Op, StringRef Constraint, std::vector<SDValue> &Ops,
1819 SelectionDAG &DAG) const {
1820 // Only support length 1 constraints for now.
1821 if (Constraint.size() == 1) {
1822 switch (Constraint[0]) {
1823 case 'I': // Unsigned 8-bit constant
1824 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1825 if (isUInt<8>(C->getZExtValue()))
1826 Ops.push_back(DAG.getTargetConstant(C->getZExtValue(), SDLoc(Op),
1827 Op.getValueType()));
1828 return;
1829
1830 case 'J': // Unsigned 12-bit constant
1831 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1832 if (isUInt<12>(C->getZExtValue()))
1833 Ops.push_back(DAG.getTargetConstant(C->getZExtValue(), SDLoc(Op),
1834 Op.getValueType()));
1835 return;
1836
1837 case 'K': // Signed 16-bit constant
1838 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1839 if (isInt<16>(C->getSExtValue()))
1840 Ops.push_back(DAG.getSignedTargetConstant(
1841 C->getSExtValue(), SDLoc(Op), Op.getValueType()));
1842 return;
1843
1844 case 'L': // Signed 20-bit displacement (on all targets we support)
1845 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1846 if (isInt<20>(C->getSExtValue()))
1847 Ops.push_back(DAG.getSignedTargetConstant(
1848 C->getSExtValue(), SDLoc(Op), Op.getValueType()));
1849 return;
1850
1851 case 'M': // 0x7fffffff
1852 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1853 if (C->getZExtValue() == 0x7fffffff)
1854 Ops.push_back(DAG.getTargetConstant(C->getZExtValue(), SDLoc(Op),
1855 Op.getValueType()));
1856 return;
1857 }
1858 }
1860}
1861
1862//===----------------------------------------------------------------------===//
1863// Calling conventions
1864//===----------------------------------------------------------------------===//
1865
1866#define GET_CALLING_CONV_IMPL
1867#include "SystemZGenCallingConv.inc"
1868
1870 CallingConv::ID) const {
1871 static const MCPhysReg ScratchRegs[] = { SystemZ::R0D, SystemZ::R1D,
1872 SystemZ::R14D, 0 };
1873 return ScratchRegs;
1874}
1875
1877 Type *ToType) const {
1878 return isTruncateFree(FromType, ToType);
1879}
1880
1882 return CI->isTailCall();
1883}
1884
1885// Value is a value that has been passed to us in the location described by VA
1886// (and so has type VA.getLocVT()). Convert Value to VA.getValVT(), chaining
1887// any loads onto Chain.
1889 CCValAssign &VA, SDValue Chain,
1890 SDValue Value) {
1891 // If the argument has been promoted from a smaller type, insert an
1892 // assertion to capture this.
1893 if (VA.getLocInfo() == CCValAssign::SExt)
1895 DAG.getValueType(VA.getValVT()));
1896 else if (VA.getLocInfo() == CCValAssign::ZExt)
1898 DAG.getValueType(VA.getValVT()));
1899
1900 if (VA.isExtInLoc())
1901 Value = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Value);
1902 else if (VA.getLocInfo() == CCValAssign::BCvt) {
1903 // If this is a short vector argument loaded from the stack,
1904 // extend from i64 to full vector size and then bitcast.
1905 assert(VA.getLocVT() == MVT::i64);
1906 assert(VA.getValVT().isVector());
1907 Value = DAG.getBuildVector(MVT::v2i64, DL, {Value, DAG.getUNDEF(MVT::i64)});
1908 Value = DAG.getNode(ISD::BITCAST, DL, VA.getValVT(), Value);
1909 } else
1910 assert(VA.getLocInfo() == CCValAssign::Full && "Unsupported getLocInfo");
1911 return Value;
1912}
1913
1914// Value is a value of type VA.getValVT() that we need to copy into
1915// the location described by VA. Return a copy of Value converted to
1916// VA.getValVT(). The caller is responsible for handling indirect values.
1918 CCValAssign &VA, SDValue Value) {
1919 switch (VA.getLocInfo()) {
1920 case CCValAssign::SExt:
1921 return DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Value);
1922 case CCValAssign::ZExt:
1923 return DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Value);
1924 case CCValAssign::AExt:
1925 return DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Value);
1926 case CCValAssign::BCvt: {
1927 assert(VA.getLocVT() == MVT::i64 || VA.getLocVT() == MVT::i128);
1928 assert(VA.getValVT().isVector() || VA.getValVT() == MVT::f32 ||
1929 VA.getValVT() == MVT::f64 || VA.getValVT() == MVT::f128);
1930 // For an f32 vararg we need to first promote it to an f64 and then
1931 // bitcast it to an i64.
1932 if (VA.getValVT() == MVT::f32 && VA.getLocVT() == MVT::i64)
1933 Value = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f64, Value);
1934 MVT BitCastToType = VA.getValVT().isVector() && VA.getLocVT() == MVT::i64
1935 ? MVT::v2i64
1936 : VA.getLocVT();
1937 Value = DAG.getNode(ISD::BITCAST, DL, BitCastToType, Value);
1938 // For ELF, this is a short vector argument to be stored to the stack,
1939 // bitcast to v2i64 and then extract first element.
1940 if (BitCastToType == MVT::v2i64)
1941 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VA.getLocVT(), Value,
1942 DAG.getConstant(0, DL, MVT::i32));
1943 return Value;
1944 }
1945 case CCValAssign::Full:
1946 return Value;
1947 default:
1948 llvm_unreachable("Unhandled getLocInfo()");
1949 }
1950}
1951
1953 SDLoc DL(In);
1954 SDValue Lo, Hi;
1955 if (DAG.getTargetLoweringInfo().isTypeLegal(MVT::i128)) {
1956 Lo = DAG.getNode(ISD::TRUNCATE, DL, MVT::i64, In);
1957 Hi = DAG.getNode(ISD::TRUNCATE, DL, MVT::i64,
1958 DAG.getNode(ISD::SRL, DL, MVT::i128, In,
1959 DAG.getConstant(64, DL, MVT::i32)));
1960 } else {
1961 std::tie(Lo, Hi) = DAG.SplitScalar(In, DL, MVT::i64, MVT::i64);
1962 }
1963
1964 // FIXME: If v2i64 were a legal type, we could use it instead of
1965 // Untyped here. This might enable improved folding.
1966 SDNode *Pair = DAG.getMachineNode(SystemZ::PAIR128, DL,
1967 MVT::Untyped, Hi, Lo);
1968 return SDValue(Pair, 0);
1969}
1970
1972 SDLoc DL(In);
1973 SDValue Hi = DAG.getTargetExtractSubreg(SystemZ::subreg_h64,
1974 DL, MVT::i64, In);
1975 SDValue Lo = DAG.getTargetExtractSubreg(SystemZ::subreg_l64,
1976 DL, MVT::i64, In);
1977
1978 if (DAG.getTargetLoweringInfo().isTypeLegal(MVT::i128)) {
1979 Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i128, Lo);
1980 Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i128, Hi);
1981 Hi = DAG.getNode(ISD::SHL, DL, MVT::i128, Hi,
1982 DAG.getConstant(64, DL, MVT::i32));
1983 return DAG.getNode(ISD::OR, DL, MVT::i128, Lo, Hi);
1984 } else {
1985 return DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i128, Lo, Hi);
1986 }
1987}
1988
1990 SelectionDAG &DAG, const SDLoc &DL, SDValue Val, SDValue *Parts,
1991 unsigned NumParts, MVT PartVT, std::optional<CallingConv::ID> CC) const {
1992 EVT ValueVT = Val.getValueType();
1993 if (ValueVT.getSizeInBits() == 128 && NumParts == 1 && PartVT == MVT::Untyped) {
1994 // Inline assembly operand.
1995 Parts[0] = lowerI128ToGR128(DAG, DAG.getBitcast(MVT::i128, Val));
1996 return true;
1997 }
1998
1999 return false;
2000}
2001
2003 SelectionDAG &DAG, const SDLoc &DL, const SDValue *Parts, unsigned NumParts,
2004 MVT PartVT, EVT ValueVT, std::optional<CallingConv::ID> CC) const {
2005 if (ValueVT.getSizeInBits() == 128 && NumParts == 1 && PartVT == MVT::Untyped) {
2006 // Inline assembly operand.
2007 SDValue Res = lowerGR128ToI128(DAG, Parts[0]);
2008 return DAG.getBitcast(ValueVT, Res);
2009 }
2010
2011 return SDValue();
2012}
2013
2014// The first part of a split stack argument is at index I in Args (and
2015// ArgLocs). Return the type of a part and the number of them by reference.
2016template <class ArgTy>
2018 SmallVector<CCValAssign, 16> &ArgLocs, unsigned I,
2019 MVT &PartVT, unsigned &NumParts) {
2020 if (!Args[I].Flags.isSplit())
2021 return false;
2022 assert(I < ArgLocs.size() && ArgLocs.size() == Args.size() &&
2023 "ArgLocs havoc.");
2024 PartVT = ArgLocs[I].getValVT();
2025 NumParts = 1;
2026 for (unsigned PartIdx = I + 1;; ++PartIdx) {
2027 assert(PartIdx != ArgLocs.size() && "SplitEnd not found.");
2028 assert(ArgLocs[PartIdx].getValVT() == PartVT && "Unsupported split.");
2029 ++NumParts;
2030 if (Args[PartIdx].Flags.isSplitEnd())
2031 break;
2032 }
2033 return true;
2034}
2035
2037 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
2038 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
2039 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
2041 MachineFrameInfo &MFI = MF.getFrameInfo();
2042 MachineRegisterInfo &MRI = MF.getRegInfo();
2043 SystemZMachineFunctionInfo *FuncInfo =
2045 auto *TFL = Subtarget.getFrameLowering<SystemZELFFrameLowering>();
2046 EVT PtrVT = getPointerTy(DAG.getDataLayout());
2047
2048 // Assign locations to all of the incoming arguments.
2050 CCState CCInfo(CallConv, IsVarArg, MF, ArgLocs, *DAG.getContext());
2051 CCInfo.AnalyzeFormalArguments(Ins, CC_SystemZ);
2052 FuncInfo->setSizeOfFnParams(CCInfo.getStackSize());
2053
2054 unsigned NumFixedGPRs = 0;
2055 unsigned NumFixedFPRs = 0;
2056 for (unsigned I = 0, E = ArgLocs.size(); I != E; ++I) {
2057 SDValue ArgValue;
2058 CCValAssign &VA = ArgLocs[I];
2059 EVT LocVT = VA.getLocVT();
2060 if (VA.isRegLoc()) {
2061 // Arguments passed in registers
2062 const TargetRegisterClass *RC;
2063 switch (LocVT.getSimpleVT().SimpleTy) {
2064 default:
2065 // Integers smaller than i64 should be promoted to i64.
2066 llvm_unreachable("Unexpected argument type");
2067 case MVT::i32:
2068 NumFixedGPRs += 1;
2069 RC = &SystemZ::GR32BitRegClass;
2070 break;
2071 case MVT::i64:
2072 NumFixedGPRs += 1;
2073 RC = &SystemZ::GR64BitRegClass;
2074 break;
2075 case MVT::f16:
2076 NumFixedFPRs += 1;
2077 RC = &SystemZ::FP16BitRegClass;
2078 break;
2079 case MVT::f32:
2080 NumFixedFPRs += 1;
2081 RC = &SystemZ::FP32BitRegClass;
2082 break;
2083 case MVT::f64:
2084 NumFixedFPRs += 1;
2085 RC = &SystemZ::FP64BitRegClass;
2086 break;
2087 case MVT::f128:
2088 NumFixedFPRs += 2;
2089 RC = &SystemZ::FP128BitRegClass;
2090 break;
2091 case MVT::v16i8:
2092 case MVT::v8i16:
2093 case MVT::v4i32:
2094 case MVT::v2i64:
2095 case MVT::v8f16:
2096 case MVT::v4f32:
2097 case MVT::v2f64:
2098 RC = &SystemZ::VR128BitRegClass;
2099 break;
2100 }
2101
2102 Register VReg = MRI.createVirtualRegister(RC);
2103 MRI.addLiveIn(VA.getLocReg(), VReg);
2104 ArgValue = DAG.getCopyFromReg(Chain, DL, VReg, LocVT);
2105 } else {
2106 assert(VA.isMemLoc() && "Argument not register or memory");
2107
2108 // Create the frame index object for this incoming parameter.
2109 // FIXME: Pre-include call frame size in the offset, should not
2110 // need to manually add it here.
2111 int64_t ArgSPOffset = VA.getLocMemOffset();
2112 if (Subtarget.isTargetXPLINK64()) {
2113 auto &XPRegs =
2114 Subtarget.getSpecialRegisters<SystemZXPLINK64Registers>();
2115 ArgSPOffset += XPRegs.getCallFrameSize();
2116 }
2117 int FI =
2118 MFI.CreateFixedObject(LocVT.getSizeInBits() / 8, ArgSPOffset, true);
2119
2120 // Create the SelectionDAG nodes corresponding to a load
2121 // from this parameter. Unpromoted ints and floats are
2122 // passed as right-justified 8-byte values.
2123 SDValue FIN = DAG.getFrameIndex(FI, PtrVT);
2124 if (VA.getLocVT() == MVT::i32 || VA.getLocVT() == MVT::f32 ||
2125 VA.getLocVT() == MVT::f16) {
2126 unsigned SlotOffs = VA.getLocVT() == MVT::f16 ? 6 : 4;
2127 FIN = DAG.getNode(ISD::ADD, DL, PtrVT, FIN,
2128 DAG.getIntPtrConstant(SlotOffs, DL));
2129 }
2130 ArgValue = DAG.getLoad(LocVT, DL, Chain, FIN,
2132 }
2133
2134 // Convert the value of the argument register into the value that's
2135 // being passed.
2136 if (VA.getLocInfo() == CCValAssign::Indirect) {
2137 InVals.push_back(DAG.getLoad(VA.getValVT(), DL, Chain, ArgValue,
2139 // If the original argument was split (e.g. i128), we need
2140 // to load all parts of it here (using the same address).
2141 MVT PartVT;
2142 unsigned NumParts;
2143 if (analyzeArgSplit(Ins, ArgLocs, I, PartVT, NumParts)) {
2144 for (unsigned PartIdx = 1; PartIdx < NumParts; ++PartIdx) {
2145 ++I;
2146 CCValAssign &PartVA = ArgLocs[I];
2147 unsigned PartOffset = Ins[I].PartOffset;
2148 SDValue Address = DAG.getNode(ISD::ADD, DL, PtrVT, ArgValue,
2149 DAG.getIntPtrConstant(PartOffset, DL));
2150 InVals.push_back(DAG.getLoad(PartVA.getValVT(), DL, Chain, Address,
2152 assert(PartOffset && "Offset should be non-zero.");
2153 }
2154 }
2155 } else if (Subtarget.isTargetXPLINK64() &&
2156 (VA.getLocInfo() == CCValAssign::SExt ||
2157 VA.getLocInfo() == CCValAssign::ZExt) &&
2158 Ins[I].ArgVT.isSimple()) {
2159 // Some prior z/OS compilers do not always perform the extension of
2160 // short integer arguments or pointers. To accommodate those, do not
2161 // rely on that extension by avoiding any AssertSext/AssertZext nodes by
2162 // directly truncating ArgValue to the original argument type.
2163 MVT OrigVT = Ins[I].ArgVT.getSimpleVT();
2164 InVals.push_back(DAG.getNode(ISD::TRUNCATE, DL, OrigVT, ArgValue));
2165 } else
2166 InVals.push_back(convertLocVTToValVT(DAG, DL, VA, Chain, ArgValue));
2167 }
2168
2169 if (IsVarArg && Subtarget.isTargetXPLINK64()) {
2170 // Save the number of non-varargs registers for later use by va_start, etc.
2171 FuncInfo->setVarArgsFirstGPR(NumFixedGPRs);
2172 FuncInfo->setVarArgsFirstFPR(NumFixedFPRs);
2173
2174 auto *Regs = static_cast<SystemZXPLINK64Registers *>(
2175 Subtarget.getSpecialRegisters());
2176
2177 // Likewise the address (in the form of a frame index) of where the
2178 // first stack vararg would be. The 1-byte size here is arbitrary.
2179 // FIXME: Pre-include call frame size in the offset, should not
2180 // need to manually add it here.
2181 int64_t VarArgOffset = CCInfo.getStackSize() + Regs->getCallFrameSize();
2182 int FI = MFI.CreateFixedObject(1, VarArgOffset, true);
2183 FuncInfo->setVarArgsFrameIndex(FI);
2184 }
2185
2186 if (IsVarArg && Subtarget.isTargetELF()) {
2187 // Save the number of non-varargs registers for later use by va_start, etc.
2188 FuncInfo->setVarArgsFirstGPR(NumFixedGPRs);
2189 FuncInfo->setVarArgsFirstFPR(NumFixedFPRs);
2190
2191 // Likewise the address (in the form of a frame index) of where the
2192 // first stack vararg would be. The 1-byte size here is arbitrary.
2193 int64_t VarArgsOffset = CCInfo.getStackSize();
2194 FuncInfo->setVarArgsFrameIndex(
2195 MFI.CreateFixedObject(1, VarArgsOffset, true));
2196
2197 // ...and a similar frame index for the caller-allocated save area
2198 // that will be used to store the incoming registers.
2199 int64_t RegSaveOffset =
2200 -SystemZMC::ELFCallFrameSize + TFL->getRegSpillOffset(MF, SystemZ::R2D) - 16;
2201 unsigned RegSaveIndex = MFI.CreateFixedObject(1, RegSaveOffset, true);
2202 FuncInfo->setRegSaveFrameIndex(RegSaveIndex);
2203
2204 // Store the FPR varargs in the reserved frame slots. (We store the
2205 // GPRs as part of the prologue.)
2206 if (NumFixedFPRs < SystemZ::ELFNumArgFPRs && !useSoftFloat()) {
2208 for (unsigned I = NumFixedFPRs; I < SystemZ::ELFNumArgFPRs; ++I) {
2209 unsigned Offset = TFL->getRegSpillOffset(MF, SystemZ::ELFArgFPRs[I]);
2210 int FI =
2212 SDValue FIN = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout()));
2214 &SystemZ::FP64BitRegClass);
2215 SDValue ArgValue = DAG.getCopyFromReg(Chain, DL, VReg, MVT::f64);
2216 MemOps[I] = DAG.getStore(ArgValue.getValue(1), DL, ArgValue, FIN,
2218 }
2219 // Join the stores, which are independent of one another.
2220 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
2221 ArrayRef(&MemOps[NumFixedFPRs],
2222 SystemZ::ELFNumArgFPRs - NumFixedFPRs));
2223 }
2224 }
2225
2226 if (Subtarget.isTargetXPLINK64()) {
2227 // Create virual register for handling incoming "ADA" special register (R5)
2228 const TargetRegisterClass *RC = &SystemZ::ADDR64BitRegClass;
2229 Register ADAvReg = MRI.createVirtualRegister(RC);
2230 auto *Regs = static_cast<SystemZXPLINK64Registers *>(
2231 Subtarget.getSpecialRegisters());
2232 MRI.addLiveIn(Regs->getADARegister(), ADAvReg);
2233 FuncInfo->setADAVirtualRegister(ADAvReg);
2234 }
2235 return Chain;
2236}
2237
2238static bool canUseSiblingCall(const CCState &ArgCCInfo,
2241 // Punt if there are any indirect or stack arguments, or if the call
2242 // needs the callee-saved argument register R6, or if the call uses
2243 // the callee-saved register arguments SwiftSelf and SwiftError.
2244 for (unsigned I = 0, E = ArgLocs.size(); I != E; ++I) {
2245 CCValAssign &VA = ArgLocs[I];
2247 return false;
2248 if (!VA.isRegLoc())
2249 return false;
2250 Register Reg = VA.getLocReg();
2251 if (Reg == SystemZ::R6H || Reg == SystemZ::R6L || Reg == SystemZ::R6D)
2252 return false;
2253 if (Outs[I].Flags.isSwiftSelf() || Outs[I].Flags.isSwiftError())
2254 return false;
2255 }
2256 return true;
2257}
2258
2260 unsigned Offset, bool LoadAdr = false) {
2263 Register ADAvReg = MFI->getADAVirtualRegister();
2265
2266 SDValue Reg = DAG.getRegister(ADAvReg, PtrVT);
2267 SDValue Ofs = DAG.getTargetConstant(Offset, DL, PtrVT);
2268
2269 SDValue Result = DAG.getNode(SystemZISD::ADA_ENTRY, DL, PtrVT, Val, Reg, Ofs);
2270 if (!LoadAdr)
2271 Result = DAG.getLoad(
2272 PtrVT, DL, DAG.getEntryNode(), Result, MachinePointerInfo(), Align(8),
2274
2275 return Result;
2276}
2277
2278// ADA access using Global value
2279// Note: for functions, address of descriptor is returned
2281 EVT PtrVT) {
2282 unsigned ADAtype;
2283 bool LoadAddr = false;
2284 const GlobalAlias *GA = dyn_cast<GlobalAlias>(GV);
2285 bool IsFunction =
2286 (isa<Function>(GV)) || (GA && isa<Function>(GA->getAliaseeObject()));
2287 bool IsInternal = (GV->hasInternalLinkage() || GV->hasPrivateLinkage());
2288
2289 if (IsFunction) {
2290 if (IsInternal) {
2292 LoadAddr = true;
2293 } else
2295 } else {
2297 }
2298 SDValue Val = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, ADAtype);
2299
2300 return getADAEntry(DAG, Val, DL, 0, LoadAddr);
2301}
2302
2303static bool getzOSCalleeAndADA(SelectionDAG &DAG, SDValue &Callee, SDValue &ADA,
2304 SDLoc &DL, SDValue &Chain) {
2305 unsigned ADADelta = 0; // ADA offset in desc.
2306 unsigned EPADelta = 8; // EPA offset in desc.
2309
2310 // XPLink calling convention.
2311 if (auto *G = dyn_cast<GlobalAddressSDNode>(Callee)) {
2312 bool IsInternal = (G->getGlobal()->hasInternalLinkage() ||
2313 G->getGlobal()->hasPrivateLinkage());
2314 if (IsInternal) {
2317 Register ADAvReg = MFI->getADAVirtualRegister();
2318 ADA = DAG.getCopyFromReg(Chain, DL, ADAvReg, PtrVT);
2319 Callee = DAG.getTargetGlobalAddress(G->getGlobal(), DL, PtrVT);
2320 Callee = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Callee);
2321 return true;
2322 } else {
2324 G->getGlobal(), DL, PtrVT, 0, SystemZII::MO_ADA_DIRECT_FUNC_DESC);
2325 ADA = getADAEntry(DAG, GA, DL, ADADelta);
2326 Callee = getADAEntry(DAG, GA, DL, EPADelta);
2327 }
2328 } else if (auto *E = dyn_cast<ExternalSymbolSDNode>(Callee)) {
2330 E->getSymbol(), PtrVT, SystemZII::MO_ADA_DIRECT_FUNC_DESC);
2331 ADA = getADAEntry(DAG, ES, DL, ADADelta);
2332 Callee = getADAEntry(DAG, ES, DL, EPADelta);
2333 } else {
2334 // Function pointer case
2335 ADA = DAG.getNode(ISD::ADD, DL, PtrVT, Callee,
2336 DAG.getConstant(ADADelta, DL, PtrVT));
2337 ADA = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), ADA,
2339 Callee = DAG.getNode(ISD::ADD, DL, PtrVT, Callee,
2340 DAG.getConstant(EPADelta, DL, PtrVT));
2341 Callee = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Callee,
2343 }
2344 return false;
2345}
2346
2347SDValue
2349 SmallVectorImpl<SDValue> &InVals) const {
2350 SelectionDAG &DAG = CLI.DAG;
2351 SDLoc &DL = CLI.DL;
2353 SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
2355 SDValue Chain = CLI.Chain;
2356 SDValue Callee = CLI.Callee;
2357 bool &IsTailCall = CLI.IsTailCall;
2358 CallingConv::ID CallConv = CLI.CallConv;
2359 bool IsVarArg = CLI.IsVarArg;
2361 EVT PtrVT = getPointerTy(MF.getDataLayout());
2362 LLVMContext &Ctx = *DAG.getContext();
2363 SystemZCallingConventionRegisters *Regs = Subtarget.getSpecialRegisters();
2364
2365 // FIXME: z/OS support to be added in later.
2366 if (Subtarget.isTargetXPLINK64())
2367 IsTailCall = false;
2368
2369 // Integer args <=32 bits should have an extension attribute.
2370 verifyNarrowIntegerArgs_Call(Outs, &MF.getFunction(), Callee);
2371
2372 // Analyze the operands of the call, assigning locations to each operand.
2374 CCState ArgCCInfo(CallConv, IsVarArg, MF, ArgLocs, Ctx);
2375 ArgCCInfo.AnalyzeCallOperands(Outs, CC_SystemZ);
2376
2377 // We don't support GuaranteedTailCallOpt, only automatically-detected
2378 // sibling calls.
2379 if (IsTailCall && !canUseSiblingCall(ArgCCInfo, ArgLocs, Outs))
2380 IsTailCall = false;
2381
2382 // Get a count of how many bytes are to be pushed on the stack.
2383 unsigned NumBytes = ArgCCInfo.getStackSize();
2384
2385 // Mark the start of the call.
2386 if (!IsTailCall)
2387 Chain = DAG.getCALLSEQ_START(Chain, NumBytes, 0, DL);
2388
2389 // Copy argument values to their designated locations.
2391 SmallVector<SDValue, 8> MemOpChains;
2392 SDValue StackPtr;
2393 for (unsigned I = 0, E = ArgLocs.size(); I != E; ++I) {
2394 CCValAssign &VA = ArgLocs[I];
2395 SDValue ArgValue = OutVals[I];
2396
2397 if (VA.getLocInfo() == CCValAssign::Indirect) {
2398 // Store the argument in a stack slot and pass its address.
2399 EVT SlotVT;
2400 MVT PartVT;
2401 unsigned NumParts = 1;
2402 if (analyzeArgSplit(Outs, ArgLocs, I, PartVT, NumParts))
2403 SlotVT = EVT::getIntegerVT(Ctx, PartVT.getSizeInBits() * NumParts);
2404 else
2405 SlotVT = Outs[I].VT;
2406 SDValue SpillSlot = DAG.CreateStackTemporary(SlotVT);
2407 int FI = cast<FrameIndexSDNode>(SpillSlot)->getIndex();
2408
2409 MachinePointerInfo StackPtrInfo =
2411 MemOpChains.push_back(
2412 DAG.getStore(Chain, DL, ArgValue, SpillSlot, StackPtrInfo));
2413 // If the original argument was split (e.g. i128), we need
2414 // to store all parts of it here (and pass just one address).
2415 assert(Outs[I].PartOffset == 0);
2416 for (unsigned PartIdx = 1; PartIdx < NumParts; ++PartIdx) {
2417 ++I;
2418 SDValue PartValue = OutVals[I];
2419 unsigned PartOffset = Outs[I].PartOffset;
2420 SDValue Address = DAG.getNode(ISD::ADD, DL, PtrVT, SpillSlot,
2421 DAG.getIntPtrConstant(PartOffset, DL));
2422 MemOpChains.push_back(
2423 DAG.getStore(Chain, DL, PartValue, Address,
2424 StackPtrInfo.getWithOffset(PartOffset)));
2425 assert(PartOffset && "Offset should be non-zero.");
2426 assert((PartOffset + PartValue.getValueType().getStoreSize() <=
2427 SlotVT.getStoreSize()) && "Not enough space for argument part!");
2428 }
2429 ArgValue = SpillSlot;
2430 } else
2431 ArgValue = convertValVTToLocVT(DAG, DL, VA, ArgValue);
2432
2433 if (VA.isRegLoc()) {
2434 // In XPLINK64, for the 128-bit vararg case, ArgValue is bitcasted to a
2435 // MVT::i128 type. We decompose the 128-bit type to a pair of its high
2436 // and low values.
2437 if (VA.getLocVT() == MVT::i128)
2438 ArgValue = lowerI128ToGR128(DAG, ArgValue);
2439 // Queue up the argument copies and emit them at the end.
2440 RegsToPass.push_back(std::make_pair(VA.getLocReg(), ArgValue));
2441 } else {
2442 assert(VA.isMemLoc() && "Argument not register or memory");
2443
2444 // Work out the address of the stack slot. Unpromoted ints and
2445 // floats are passed as right-justified 8-byte values.
2446 if (!StackPtr.getNode())
2447 StackPtr = DAG.getCopyFromReg(Chain, DL,
2448 Regs->getStackPointerRegister(), PtrVT);
2449 unsigned Offset = Regs->getStackPointerBias() + Regs->getCallFrameSize() +
2450 VA.getLocMemOffset();
2451 if (VA.getLocVT() == MVT::i32 || VA.getLocVT() == MVT::f32)
2452 Offset += 4;
2453 else if (VA.getLocVT() == MVT::f16)
2454 Offset += 6;
2455 SDValue Address = DAG.getNode(ISD::ADD, DL, PtrVT, StackPtr,
2457
2458 // Emit the store.
2459 MemOpChains.push_back(
2460 DAG.getStore(Chain, DL, ArgValue, Address, MachinePointerInfo()));
2461
2462 // Although long doubles or vectors are passed through the stack when
2463 // they are vararg (non-fixed arguments), if a long double or vector
2464 // occupies the third and fourth slot of the argument list GPR3 should
2465 // still shadow the third slot of the argument list.
2466 if (Subtarget.isTargetXPLINK64() && VA.needsCustom()) {
2467 SDValue ShadowArgValue =
2468 DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, ArgValue,
2469 DAG.getIntPtrConstant(1, DL));
2470 RegsToPass.push_back(std::make_pair(SystemZ::R3D, ShadowArgValue));
2471 }
2472 }
2473 }
2474
2475 // Join the stores, which are independent of one another.
2476 if (!MemOpChains.empty())
2477 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains);
2478
2479 // Accept direct calls by converting symbolic call addresses to the
2480 // associated Target* opcodes. Force %r1 to be used for indirect
2481 // tail calls.
2482 SDValue Glue;
2483
2484 if (Subtarget.isTargetXPLINK64()) {
2485 SDValue ADA;
2486 bool IsBRASL = getzOSCalleeAndADA(DAG, Callee, ADA, DL, Chain);
2487 if (!IsBRASL) {
2488 unsigned CalleeReg = static_cast<SystemZXPLINK64Registers *>(Regs)
2489 ->getAddressOfCalleeRegister();
2490 Chain = DAG.getCopyToReg(Chain, DL, CalleeReg, Callee, Glue);
2491 Glue = Chain.getValue(1);
2492 Callee = DAG.getRegister(CalleeReg, Callee.getValueType());
2493 }
2494 RegsToPass.push_back(std::make_pair(
2495 static_cast<SystemZXPLINK64Registers *>(Regs)->getADARegister(), ADA));
2496 } else {
2497 if (auto *G = dyn_cast<GlobalAddressSDNode>(Callee)) {
2498 Callee = DAG.getTargetGlobalAddress(G->getGlobal(), DL, PtrVT);
2499 Callee = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Callee);
2500 } else if (auto *E = dyn_cast<ExternalSymbolSDNode>(Callee)) {
2501 Callee = DAG.getTargetExternalSymbol(E->getSymbol(), PtrVT);
2502 Callee = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Callee);
2503 } else if (IsTailCall) {
2504 Chain = DAG.getCopyToReg(Chain, DL, SystemZ::R1D, Callee, Glue);
2505 Glue = Chain.getValue(1);
2506 Callee = DAG.getRegister(SystemZ::R1D, Callee.getValueType());
2507 }
2508 }
2509
2510 // Build a sequence of copy-to-reg nodes, chained and glued together.
2511 for (const auto &[Reg, N] : RegsToPass) {
2512 Chain = DAG.getCopyToReg(Chain, DL, Reg, N, Glue);
2513 Glue = Chain.getValue(1);
2514 }
2515
2516 // The first call operand is the chain and the second is the target address.
2518 Ops.push_back(Chain);
2519 Ops.push_back(Callee);
2520
2521 // Add argument registers to the end of the list so that they are
2522 // known live into the call.
2523 for (const auto &[Reg, N] : RegsToPass)
2524 Ops.push_back(DAG.getRegister(Reg, N.getValueType()));
2525
2526 // Add a register mask operand representing the call-preserved registers.
2527 const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo();
2528 const uint32_t *Mask = TRI->getCallPreservedMask(MF, CallConv);
2529 assert(Mask && "Missing call preserved mask for calling convention");
2530 Ops.push_back(DAG.getRegisterMask(Mask));
2531
2532 // Glue the call to the argument copies, if any.
2533 if (Glue.getNode())
2534 Ops.push_back(Glue);
2535
2536 // Emit the call.
2537 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
2538 if (IsTailCall) {
2539 SDValue Ret = DAG.getNode(SystemZISD::SIBCALL, DL, NodeTys, Ops);
2540 DAG.addNoMergeSiteInfo(Ret.getNode(), CLI.NoMerge);
2541 return Ret;
2542 }
2543 Chain = DAG.getNode(SystemZISD::CALL, DL, NodeTys, Ops);
2544 DAG.addNoMergeSiteInfo(Chain.getNode(), CLI.NoMerge);
2545 Glue = Chain.getValue(1);
2546
2547 // Mark the end of the call, which is glued to the call itself.
2548 Chain = DAG.getCALLSEQ_END(Chain, NumBytes, 0, Glue, DL);
2549 Glue = Chain.getValue(1);
2550
2551 // Assign locations to each value returned by this call.
2553 CCState RetCCInfo(CallConv, IsVarArg, MF, RetLocs, Ctx);
2554 RetCCInfo.AnalyzeCallResult(Ins, RetCC_SystemZ);
2555
2556 // Copy all of the result registers out of their specified physreg.
2557 for (CCValAssign &VA : RetLocs) {
2558 // Copy the value out, gluing the copy to the end of the call sequence.
2559 SDValue RetValue = DAG.getCopyFromReg(Chain, DL, VA.getLocReg(),
2560 VA.getLocVT(), Glue);
2561 Chain = RetValue.getValue(1);
2562 Glue = RetValue.getValue(2);
2563
2564 // Convert the value of the return register into the value that's
2565 // being returned.
2566 InVals.push_back(convertLocVTToValVT(DAG, DL, VA, Chain, RetValue));
2567 }
2568
2569 return Chain;
2570}
2571
2572// Generate a call taking the given operands as arguments and returning a
2573// result of type RetVT.
2575 SDValue Chain, SelectionDAG &DAG, const char *CalleeName, EVT RetVT,
2576 ArrayRef<SDValue> Ops, CallingConv::ID CallConv, bool IsSigned, SDLoc DL,
2577 bool DoesNotReturn, bool IsReturnValueUsed) const {
2579 Args.reserve(Ops.size());
2580
2581 for (SDValue Op : Ops) {
2583 Op, Op.getValueType().getTypeForEVT(*DAG.getContext()));
2584 Entry.IsSExt = shouldSignExtendTypeInLibCall(Entry.Ty, IsSigned);
2585 Entry.IsZExt = !Entry.IsSExt;
2586 Args.push_back(Entry);
2587 }
2588
2589 SDValue Callee =
2590 DAG.getExternalSymbol(CalleeName, getPointerTy(DAG.getDataLayout()));
2591
2592 Type *RetTy = RetVT.getTypeForEVT(*DAG.getContext());
2594 bool SignExtend = shouldSignExtendTypeInLibCall(RetTy, IsSigned);
2595 CLI.setDebugLoc(DL)
2596 .setChain(Chain)
2597 .setCallee(CallConv, RetTy, Callee, std::move(Args))
2598 .setNoReturn(DoesNotReturn)
2599 .setDiscardResult(!IsReturnValueUsed)
2600 .setSExtResult(SignExtend)
2601 .setZExtResult(!SignExtend);
2602 return LowerCallTo(CLI);
2603}
2604
2606 CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg,
2607 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context,
2608 const Type *RetTy) const {
2609 // Special case that we cannot easily detect in RetCC_SystemZ since
2610 // i128 may not be a legal type.
2611 for (auto &Out : Outs)
2612 if (Out.ArgVT.isScalarInteger() && Out.ArgVT.getSizeInBits() > 64)
2613 return false;
2614
2616 CCState RetCCInfo(CallConv, IsVarArg, MF, RetLocs, Context);
2617 return RetCCInfo.CheckReturn(Outs, RetCC_SystemZ);
2618}
2619
2620SDValue
2622 bool IsVarArg,
2624 const SmallVectorImpl<SDValue> &OutVals,
2625 const SDLoc &DL, SelectionDAG &DAG) const {
2627
2628 // Integer args <=32 bits should have an extension attribute.
2629 verifyNarrowIntegerArgs_Ret(Outs, &MF.getFunction());
2630
2631 // Assign locations to each returned value.
2633 CCState RetCCInfo(CallConv, IsVarArg, MF, RetLocs, *DAG.getContext());
2634 RetCCInfo.AnalyzeReturn(Outs, RetCC_SystemZ);
2635
2636 // Quick exit for void returns
2637 if (RetLocs.empty())
2638 return DAG.getNode(SystemZISD::RET_GLUE, DL, MVT::Other, Chain);
2639
2640 if (CallConv == CallingConv::GHC)
2641 report_fatal_error("GHC functions return void only");
2642
2643 // Copy the result values into the output registers.
2644 SDValue Glue;
2646 RetOps.push_back(Chain);
2647 for (unsigned I = 0, E = RetLocs.size(); I != E; ++I) {
2648 CCValAssign &VA = RetLocs[I];
2649 SDValue RetValue = OutVals[I];
2650
2651 // Make the return register live on exit.
2652 assert(VA.isRegLoc() && "Can only return in registers!");
2653
2654 // Promote the value as required.
2655 RetValue = convertValVTToLocVT(DAG, DL, VA, RetValue);
2656
2657 // Chain and glue the copies together.
2658 Register Reg = VA.getLocReg();
2659 Chain = DAG.getCopyToReg(Chain, DL, Reg, RetValue, Glue);
2660 Glue = Chain.getValue(1);
2661 RetOps.push_back(DAG.getRegister(Reg, VA.getLocVT()));
2662 }
2663
2664 // Update chain and glue.
2665 RetOps[0] = Chain;
2666 if (Glue.getNode())
2667 RetOps.push_back(Glue);
2668
2669 return DAG.getNode(SystemZISD::RET_GLUE, DL, MVT::Other, RetOps);
2670}
2671
2672// Return true if Op is an intrinsic node with chain that returns the CC value
2673// as its only (other) argument. Provide the associated SystemZISD opcode and
2674// the mask of valid CC values if so.
2675static bool isIntrinsicWithCCAndChain(SDValue Op, unsigned &Opcode,
2676 unsigned &CCValid) {
2677 unsigned Id = Op.getConstantOperandVal(1);
2678 switch (Id) {
2679 case Intrinsic::s390_tbegin:
2680 Opcode = SystemZISD::TBEGIN;
2681 CCValid = SystemZ::CCMASK_TBEGIN;
2682 return true;
2683
2684 case Intrinsic::s390_tbegin_nofloat:
2685 Opcode = SystemZISD::TBEGIN_NOFLOAT;
2686 CCValid = SystemZ::CCMASK_TBEGIN;
2687 return true;
2688
2689 case Intrinsic::s390_tend:
2690 Opcode = SystemZISD::TEND;
2691 CCValid = SystemZ::CCMASK_TEND;
2692 return true;
2693
2694 default:
2695 return false;
2696 }
2697}
2698
2699// Return true if Op is an intrinsic node without chain that returns the
2700// CC value as its final argument. Provide the associated SystemZISD
2701// opcode and the mask of valid CC values if so.
2702static bool isIntrinsicWithCC(SDValue Op, unsigned &Opcode, unsigned &CCValid) {
2703 unsigned Id = Op.getConstantOperandVal(0);
2704 switch (Id) {
2705 case Intrinsic::s390_vpkshs:
2706 case Intrinsic::s390_vpksfs:
2707 case Intrinsic::s390_vpksgs:
2708 Opcode = SystemZISD::PACKS_CC;
2709 CCValid = SystemZ::CCMASK_VCMP;
2710 return true;
2711
2712 case Intrinsic::s390_vpklshs:
2713 case Intrinsic::s390_vpklsfs:
2714 case Intrinsic::s390_vpklsgs:
2715 Opcode = SystemZISD::PACKLS_CC;
2716 CCValid = SystemZ::CCMASK_VCMP;
2717 return true;
2718
2719 case Intrinsic::s390_vceqbs:
2720 case Intrinsic::s390_vceqhs:
2721 case Intrinsic::s390_vceqfs:
2722 case Intrinsic::s390_vceqgs:
2723 case Intrinsic::s390_vceqqs:
2724 Opcode = SystemZISD::VICMPES;
2725 CCValid = SystemZ::CCMASK_VCMP;
2726 return true;
2727
2728 case Intrinsic::s390_vchbs:
2729 case Intrinsic::s390_vchhs:
2730 case Intrinsic::s390_vchfs:
2731 case Intrinsic::s390_vchgs:
2732 case Intrinsic::s390_vchqs:
2733 Opcode = SystemZISD::VICMPHS;
2734 CCValid = SystemZ::CCMASK_VCMP;
2735 return true;
2736
2737 case Intrinsic::s390_vchlbs:
2738 case Intrinsic::s390_vchlhs:
2739 case Intrinsic::s390_vchlfs:
2740 case Intrinsic::s390_vchlgs:
2741 case Intrinsic::s390_vchlqs:
2742 Opcode = SystemZISD::VICMPHLS;
2743 CCValid = SystemZ::CCMASK_VCMP;
2744 return true;
2745
2746 case Intrinsic::s390_vtm:
2747 Opcode = SystemZISD::VTM;
2748 CCValid = SystemZ::CCMASK_VCMP;
2749 return true;
2750
2751 case Intrinsic::s390_vfaebs:
2752 case Intrinsic::s390_vfaehs:
2753 case Intrinsic::s390_vfaefs:
2754 Opcode = SystemZISD::VFAE_CC;
2755 CCValid = SystemZ::CCMASK_ANY;
2756 return true;
2757
2758 case Intrinsic::s390_vfaezbs:
2759 case Intrinsic::s390_vfaezhs:
2760 case Intrinsic::s390_vfaezfs:
2761 Opcode = SystemZISD::VFAEZ_CC;
2762 CCValid = SystemZ::CCMASK_ANY;
2763 return true;
2764
2765 case Intrinsic::s390_vfeebs:
2766 case Intrinsic::s390_vfeehs:
2767 case Intrinsic::s390_vfeefs:
2768 Opcode = SystemZISD::VFEE_CC;
2769 CCValid = SystemZ::CCMASK_ANY;
2770 return true;
2771
2772 case Intrinsic::s390_vfeezbs:
2773 case Intrinsic::s390_vfeezhs:
2774 case Intrinsic::s390_vfeezfs:
2775 Opcode = SystemZISD::VFEEZ_CC;
2776 CCValid = SystemZ::CCMASK_ANY;
2777 return true;
2778
2779 case Intrinsic::s390_vfenebs:
2780 case Intrinsic::s390_vfenehs:
2781 case Intrinsic::s390_vfenefs:
2782 Opcode = SystemZISD::VFENE_CC;
2783 CCValid = SystemZ::CCMASK_ANY;
2784 return true;
2785
2786 case Intrinsic::s390_vfenezbs:
2787 case Intrinsic::s390_vfenezhs:
2788 case Intrinsic::s390_vfenezfs:
2789 Opcode = SystemZISD::VFENEZ_CC;
2790 CCValid = SystemZ::CCMASK_ANY;
2791 return true;
2792
2793 case Intrinsic::s390_vistrbs:
2794 case Intrinsic::s390_vistrhs:
2795 case Intrinsic::s390_vistrfs:
2796 Opcode = SystemZISD::VISTR_CC;
2798 return true;
2799
2800 case Intrinsic::s390_vstrcbs:
2801 case Intrinsic::s390_vstrchs:
2802 case Intrinsic::s390_vstrcfs:
2803 Opcode = SystemZISD::VSTRC_CC;
2804 CCValid = SystemZ::CCMASK_ANY;
2805 return true;
2806
2807 case Intrinsic::s390_vstrczbs:
2808 case Intrinsic::s390_vstrczhs:
2809 case Intrinsic::s390_vstrczfs:
2810 Opcode = SystemZISD::VSTRCZ_CC;
2811 CCValid = SystemZ::CCMASK_ANY;
2812 return true;
2813
2814 case Intrinsic::s390_vstrsb:
2815 case Intrinsic::s390_vstrsh:
2816 case Intrinsic::s390_vstrsf:
2817 Opcode = SystemZISD::VSTRS_CC;
2818 CCValid = SystemZ::CCMASK_ANY;
2819 return true;
2820
2821 case Intrinsic::s390_vstrszb:
2822 case Intrinsic::s390_vstrszh:
2823 case Intrinsic::s390_vstrszf:
2824 Opcode = SystemZISD::VSTRSZ_CC;
2825 CCValid = SystemZ::CCMASK_ANY;
2826 return true;
2827
2828 case Intrinsic::s390_vfcedbs:
2829 case Intrinsic::s390_vfcesbs:
2830 Opcode = SystemZISD::VFCMPES;
2831 CCValid = SystemZ::CCMASK_VCMP;
2832 return true;
2833
2834 case Intrinsic::s390_vfchdbs:
2835 case Intrinsic::s390_vfchsbs:
2836 Opcode = SystemZISD::VFCMPHS;
2837 CCValid = SystemZ::CCMASK_VCMP;
2838 return true;
2839
2840 case Intrinsic::s390_vfchedbs:
2841 case Intrinsic::s390_vfchesbs:
2842 Opcode = SystemZISD::VFCMPHES;
2843 CCValid = SystemZ::CCMASK_VCMP;
2844 return true;
2845
2846 case Intrinsic::s390_vftcidb:
2847 case Intrinsic::s390_vftcisb:
2848 Opcode = SystemZISD::VFTCI;
2849 CCValid = SystemZ::CCMASK_VCMP;
2850 return true;
2851
2852 case Intrinsic::s390_tdc:
2853 Opcode = SystemZISD::TDC;
2854 CCValid = SystemZ::CCMASK_TDC;
2855 return true;
2856
2857 default:
2858 return false;
2859 }
2860}
2861
2862// Emit an intrinsic with chain and an explicit CC register result.
2864 unsigned Opcode) {
2865 // Copy all operands except the intrinsic ID.
2866 unsigned NumOps = Op.getNumOperands();
2868 Ops.reserve(NumOps - 1);
2869 Ops.push_back(Op.getOperand(0));
2870 for (unsigned I = 2; I < NumOps; ++I)
2871 Ops.push_back(Op.getOperand(I));
2872
2873 assert(Op->getNumValues() == 2 && "Expected only CC result and chain");
2874 SDVTList RawVTs = DAG.getVTList(MVT::i32, MVT::Other);
2875 SDValue Intr = DAG.getNode(Opcode, SDLoc(Op), RawVTs, Ops);
2876 SDValue OldChain = SDValue(Op.getNode(), 1);
2877 SDValue NewChain = SDValue(Intr.getNode(), 1);
2878 DAG.ReplaceAllUsesOfValueWith(OldChain, NewChain);
2879 return Intr.getNode();
2880}
2881
2882// Emit an intrinsic with an explicit CC register result.
2884 unsigned Opcode) {
2885 // Copy all operands except the intrinsic ID.
2886 SDLoc DL(Op);
2887 unsigned NumOps = Op.getNumOperands();
2889 Ops.reserve(NumOps - 1);
2890 for (unsigned I = 1; I < NumOps; ++I) {
2891 SDValue CurrOper = Op.getOperand(I);
2892 if (CurrOper.getValueType() == MVT::f16) {
2893 assert((Op.getConstantOperandVal(0) == Intrinsic::s390_tdc && I == 1) &&
2894 "Unhandled intrinsic with f16 operand.");
2895 CurrOper = DAG.getFPExtendOrRound(CurrOper, DL, MVT::f32);
2896 }
2897 Ops.push_back(CurrOper);
2898 }
2899
2900 SDValue Intr = DAG.getNode(Opcode, DL, Op->getVTList(), Ops);
2901 return Intr.getNode();
2902}
2903
2904// CC is a comparison that will be implemented using an integer or
2905// floating-point comparison. Return the condition code mask for
2906// a branch on true. In the integer case, CCMASK_CMP_UO is set for
2907// unsigned comparisons and clear for signed ones. In the floating-point
2908// case, CCMASK_CMP_UO has its normal mask meaning (unordered).
2910#define CONV(X) \
2911 case ISD::SET##X: return SystemZ::CCMASK_CMP_##X; \
2912 case ISD::SETO##X: return SystemZ::CCMASK_CMP_##X; \
2913 case ISD::SETU##X: return SystemZ::CCMASK_CMP_UO | SystemZ::CCMASK_CMP_##X
2914
2915 switch (CC) {
2916 default:
2917 llvm_unreachable("Invalid integer condition!");
2918
2919 CONV(EQ);
2920 CONV(NE);
2921 CONV(GT);
2922 CONV(GE);
2923 CONV(LT);
2924 CONV(LE);
2925
2926 case ISD::SETO: return SystemZ::CCMASK_CMP_O;
2928 }
2929#undef CONV
2930}
2931
2932// If C can be converted to a comparison against zero, adjust the operands
2933// as necessary.
2934static void adjustZeroCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C) {
2935 if (C.ICmpType == SystemZICMP::UnsignedOnly)
2936 return;
2937
2938 auto *ConstOp1 = dyn_cast<ConstantSDNode>(C.Op1.getNode());
2939 if (!ConstOp1 || ConstOp1->getValueSizeInBits(0) > 64)
2940 return;
2941
2942 int64_t Value = ConstOp1->getSExtValue();
2943 if ((Value == -1 && C.CCMask == SystemZ::CCMASK_CMP_GT) ||
2944 (Value == -1 && C.CCMask == SystemZ::CCMASK_CMP_LE) ||
2945 (Value == 1 && C.CCMask == SystemZ::CCMASK_CMP_LT) ||
2946 (Value == 1 && C.CCMask == SystemZ::CCMASK_CMP_GE)) {
2947 C.CCMask ^= SystemZ::CCMASK_CMP_EQ;
2948 C.Op1 = DAG.getConstant(0, DL, C.Op1.getValueType());
2949 }
2950}
2951
2952// If a comparison described by C is suitable for CLI(Y), CHHSI or CLHHSI,
2953// adjust the operands as necessary.
2954static void adjustSubwordCmp(SelectionDAG &DAG, const SDLoc &DL,
2955 Comparison &C) {
2956 // For us to make any changes, it must a comparison between a single-use
2957 // load and a constant.
2958 if (!C.Op0.hasOneUse() ||
2959 C.Op0.getOpcode() != ISD::LOAD ||
2960 C.Op1.getOpcode() != ISD::Constant)
2961 return;
2962
2963 // We must have an 8- or 16-bit load.
2964 auto *Load = cast<LoadSDNode>(C.Op0);
2965 unsigned NumBits = Load->getMemoryVT().getSizeInBits();
2966 if ((NumBits != 8 && NumBits != 16) ||
2967 NumBits != Load->getMemoryVT().getStoreSizeInBits())
2968 return;
2969
2970 // The load must be an extending one and the constant must be within the
2971 // range of the unextended value.
2972 auto *ConstOp1 = cast<ConstantSDNode>(C.Op1);
2973 if (!ConstOp1 || ConstOp1->getValueSizeInBits(0) > 64)
2974 return;
2975 uint64_t Value = ConstOp1->getZExtValue();
2976 uint64_t Mask = (1 << NumBits) - 1;
2977 if (Load->getExtensionType() == ISD::SEXTLOAD) {
2978 // Make sure that ConstOp1 is in range of C.Op0.
2979 int64_t SignedValue = ConstOp1->getSExtValue();
2980 if (uint64_t(SignedValue) + (uint64_t(1) << (NumBits - 1)) > Mask)
2981 return;
2982 if (C.ICmpType != SystemZICMP::SignedOnly) {
2983 // Unsigned comparison between two sign-extended values is equivalent
2984 // to unsigned comparison between two zero-extended values.
2985 Value &= Mask;
2986 } else if (NumBits == 8) {
2987 // Try to treat the comparison as unsigned, so that we can use CLI.
2988 // Adjust CCMask and Value as necessary.
2989 if (Value == 0 && C.CCMask == SystemZ::CCMASK_CMP_LT)
2990 // Test whether the high bit of the byte is set.
2991 Value = 127, C.CCMask = SystemZ::CCMASK_CMP_GT;
2992 else if (Value == 0 && C.CCMask == SystemZ::CCMASK_CMP_GE)
2993 // Test whether the high bit of the byte is clear.
2994 Value = 128, C.CCMask = SystemZ::CCMASK_CMP_LT;
2995 else
2996 // No instruction exists for this combination.
2997 return;
2998 C.ICmpType = SystemZICMP::UnsignedOnly;
2999 }
3000 } else if (Load->getExtensionType() == ISD::ZEXTLOAD) {
3001 if (Value > Mask)
3002 return;
3003 // If the constant is in range, we can use any comparison.
3004 C.ICmpType = SystemZICMP::Any;
3005 } else
3006 return;
3007
3008 // Make sure that the first operand is an i32 of the right extension type.
3009 ISD::LoadExtType ExtType = (C.ICmpType == SystemZICMP::SignedOnly ?
3012 if (C.Op0.getValueType() != MVT::i32 ||
3013 Load->getExtensionType() != ExtType) {
3014 C.Op0 = DAG.getExtLoad(ExtType, SDLoc(Load), MVT::i32, Load->getChain(),
3015 Load->getBasePtr(), Load->getPointerInfo(),
3016 Load->getMemoryVT(), Load->getAlign(),
3017 Load->getMemOperand()->getFlags());
3018 // Update the chain uses.
3019 DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), C.Op0.getValue(1));
3020 }
3021
3022 // Make sure that the second operand is an i32 with the right value.
3023 if (C.Op1.getValueType() != MVT::i32 ||
3024 Value != ConstOp1->getZExtValue())
3025 C.Op1 = DAG.getConstant((uint32_t)Value, DL, MVT::i32);
3026}
3027
3028// Return true if Op is either an unextended load, or a load suitable
3029// for integer register-memory comparisons of type ICmpType.
3030static bool isNaturalMemoryOperand(SDValue Op, unsigned ICmpType) {
3031 auto *Load = dyn_cast<LoadSDNode>(Op.getNode());
3032 if (Load) {
3033 // There are no instructions to compare a register with a memory byte.
3034 if (Load->getMemoryVT() == MVT::i8)
3035 return false;
3036 // Otherwise decide on extension type.
3037 switch (Load->getExtensionType()) {
3038 case ISD::NON_EXTLOAD:
3039 return true;
3040 case ISD::SEXTLOAD:
3041 return ICmpType != SystemZICMP::UnsignedOnly;
3042 case ISD::ZEXTLOAD:
3043 return ICmpType != SystemZICMP::SignedOnly;
3044 default:
3045 break;
3046 }
3047 }
3048 return false;
3049}
3050
3051// Return true if it is better to swap the operands of C.
3052static bool shouldSwapCmpOperands(const Comparison &C) {
3053 // If one side of the compare is a load of the stackguard reference value,
3054 // then that load should be Op1.
3055 if (C.Op0.isMachineOpcode() &&
3056 (C.Op0.getMachineOpcode() == SystemZ::LOAD_STACK_GUARD))
3057 return true;
3058
3059 // Leave i128 and f128 comparisons alone, since they have no memory forms.
3060 if (C.Op0.getValueType() == MVT::i128)
3061 return false;
3062 if (C.Op0.getValueType() == MVT::f128)
3063 return false;
3064
3065 // Always keep a floating-point constant second, since comparisons with
3066 // zero can use LOAD TEST and comparisons with other constants make a
3067 // natural memory operand.
3068 if (isa<ConstantFPSDNode>(C.Op1))
3069 return false;
3070
3071 // Never swap comparisons with zero since there are many ways to optimize
3072 // those later.
3073 auto *ConstOp1 = dyn_cast<ConstantSDNode>(C.Op1);
3074 if (ConstOp1 && ConstOp1->getZExtValue() == 0)
3075 return false;
3076
3077 // Also keep natural memory operands second if the loaded value is
3078 // only used here. Several comparisons have memory forms.
3079 if (isNaturalMemoryOperand(C.Op1, C.ICmpType) && C.Op1.hasOneUse())
3080 return false;
3081
3082 // Look for cases where Cmp0 is a single-use load and Cmp1 isn't.
3083 // In that case we generally prefer the memory to be second.
3084 if (isNaturalMemoryOperand(C.Op0, C.ICmpType) && C.Op0.hasOneUse()) {
3085 // The only exceptions are when the second operand is a constant and
3086 // we can use things like CHHSI.
3087 if (!ConstOp1)
3088 return true;
3089 // The unsigned memory-immediate instructions can handle 16-bit
3090 // unsigned integers.
3091 if (C.ICmpType != SystemZICMP::SignedOnly &&
3092 isUInt<16>(ConstOp1->getZExtValue()))
3093 return false;
3094 // The signed memory-immediate instructions can handle 16-bit
3095 // signed integers.
3096 if (C.ICmpType != SystemZICMP::UnsignedOnly &&
3097 isInt<16>(ConstOp1->getSExtValue()))
3098 return false;
3099 return true;
3100 }
3101
3102 // Try to promote the use of CGFR and CLGFR.
3103 unsigned Opcode0 = C.Op0.getOpcode();
3104 if (C.ICmpType != SystemZICMP::UnsignedOnly && Opcode0 == ISD::SIGN_EXTEND)
3105 return true;
3106 if (C.ICmpType != SystemZICMP::SignedOnly && Opcode0 == ISD::ZERO_EXTEND)
3107 return true;
3108 if (C.ICmpType != SystemZICMP::SignedOnly && Opcode0 == ISD::AND &&
3109 C.Op0.getOperand(1).getOpcode() == ISD::Constant &&
3110 C.Op0.getConstantOperandVal(1) == 0xffffffff)
3111 return true;
3112
3113 return false;
3114}
3115
3116// Check whether C tests for equality between X and Y and whether X - Y
3117// or Y - X is also computed. In that case it's better to compare the
3118// result of the subtraction against zero.
3120 Comparison &C) {
3121 if (C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3122 C.CCMask == SystemZ::CCMASK_CMP_NE) {
3123 for (SDNode *N : C.Op0->users()) {
3124 if (N->getOpcode() == ISD::SUB &&
3125 ((N->getOperand(0) == C.Op0 && N->getOperand(1) == C.Op1) ||
3126 (N->getOperand(0) == C.Op1 && N->getOperand(1) == C.Op0))) {
3127 // Disable the nsw and nuw flags: the backend needs to handle
3128 // overflow as well during comparison elimination.
3129 N->dropFlags(SDNodeFlags::NoWrap);
3130 C.Op0 = SDValue(N, 0);
3131 C.Op1 = DAG.getConstant(0, DL, N->getValueType(0));
3132 return;
3133 }
3134 }
3135 }
3136}
3137
3138// Check whether C compares a floating-point value with zero and if that
3139// floating-point value is also negated. In this case we can use the
3140// negation to set CC, so avoiding separate LOAD AND TEST and
3141// LOAD (NEGATIVE/COMPLEMENT) instructions.
3142static void adjustForFNeg(Comparison &C) {
3143 // This optimization is invalid for strict comparisons, since FNEG
3144 // does not raise any exceptions.
3145 if (C.Chain)
3146 return;
3147 auto *C1 = dyn_cast<ConstantFPSDNode>(C.Op1);
3148 if (C1 && C1->isZero()) {
3149 for (SDNode *N : C.Op0->users()) {
3150 if (N->getOpcode() == ISD::FNEG) {
3151 C.Op0 = SDValue(N, 0);
3152 C.CCMask = SystemZ::reverseCCMask(C.CCMask);
3153 return;
3154 }
3155 }
3156 }
3157}
3158
3159// Check whether C compares (shl X, 32) with 0 and whether X is
3160// also sign-extended. In that case it is better to test the result
3161// of the sign extension using LTGFR.
3162//
3163// This case is important because InstCombine transforms a comparison
3164// with (sext (trunc X)) into a comparison with (shl X, 32).
3165static void adjustForLTGFR(Comparison &C) {
3166 // Check for a comparison between (shl X, 32) and 0.
3167 if (C.Op0.getOpcode() == ISD::SHL && C.Op0.getValueType() == MVT::i64 &&
3168 C.Op1.getOpcode() == ISD::Constant && C.Op1->getAsZExtVal() == 0) {
3169 auto *C1 = dyn_cast<ConstantSDNode>(C.Op0.getOperand(1));
3170 if (C1 && C1->getZExtValue() == 32) {
3171 SDValue ShlOp0 = C.Op0.getOperand(0);
3172 // See whether X has any SIGN_EXTEND_INREG uses.
3173 for (SDNode *N : ShlOp0->users()) {
3174 if (N->getOpcode() == ISD::SIGN_EXTEND_INREG &&
3175 cast<VTSDNode>(N->getOperand(1))->getVT() == MVT::i32) {
3176 C.Op0 = SDValue(N, 0);
3177 return;
3178 }
3179 }
3180 }
3181 }
3182}
3183
3184// If C compares the truncation of an extending load, try to compare
3185// the untruncated value instead. This exposes more opportunities to
3186// reuse CC.
3187static void adjustICmpTruncate(SelectionDAG &DAG, const SDLoc &DL,
3188 Comparison &C) {
3189 if (C.Op0.getOpcode() == ISD::TRUNCATE &&
3190 C.Op0.getOperand(0).getOpcode() == ISD::LOAD &&
3191 C.Op1.getOpcode() == ISD::Constant &&
3192 cast<ConstantSDNode>(C.Op1)->getValueSizeInBits(0) <= 64 &&
3193 C.Op1->getAsZExtVal() == 0) {
3194 auto *L = cast<LoadSDNode>(C.Op0.getOperand(0));
3195 if (L->getMemoryVT().getStoreSizeInBits().getFixedValue() <=
3196 C.Op0.getValueSizeInBits().getFixedValue()) {
3197 unsigned Type = L->getExtensionType();
3198 if ((Type == ISD::ZEXTLOAD && C.ICmpType != SystemZICMP::SignedOnly) ||
3199 (Type == ISD::SEXTLOAD && C.ICmpType != SystemZICMP::UnsignedOnly)) {
3200 C.Op0 = C.Op0.getOperand(0);
3201 C.Op1 = DAG.getConstant(0, DL, C.Op0.getValueType());
3202 }
3203 }
3204 }
3205}
3206
3207// Adjust if a given Compare is a check of the stack guard against a stack
3208// guard instance on the stack. Specifically, this checks if:
3209// - The operands are a load of the stack guard, and a load from a stack slot
3210// - The original opcode is ICMP
3211// - ICMPType is compatible with unsigned comparison.
3213 Comparison &C) {
3214
3215 // Opcode must be ICMP.
3216 if (C.Opcode != SystemZISD::ICMP)
3217 return;
3218 // ICmpType must be Unsigned or Any.
3219 if (C.ICmpType == SystemZICMP::SignedOnly)
3220 return;
3221 // Op0 must be FrameIndex Load.
3222 if (!(ISD::isNormalLoad(C.Op0.getNode()) &&
3223 dyn_cast<FrameIndexSDNode>(C.Op0.getOperand(1))))
3224 return;
3225 // Op1 must be LOAD_STACK_GUARD.
3226 if (!C.Op1.isMachineOpcode() ||
3227 C.Op1.getMachineOpcode() != SystemZ::LOAD_STACK_GUARD)
3228 return;
3229
3230 // At this point we are sure that this is a proper CMP_STACKGUARD
3231 // case, update the opcode to reflect this.
3232 C.Opcode = SystemZISD::CMP_STACKGUARD;
3233 C.Op1 = SDValue();
3234}
3235
3236// Return true if shift operation N has an in-range constant shift value.
3237// Store it in ShiftVal if so.
3238static bool isSimpleShift(SDValue N, unsigned &ShiftVal) {
3239 auto *Shift = dyn_cast<ConstantSDNode>(N.getOperand(1));
3240 if (!Shift)
3241 return false;
3242
3243 uint64_t Amount = Shift->getZExtValue();
3244 if (Amount >= N.getValueSizeInBits())
3245 return false;
3246
3247 ShiftVal = Amount;
3248 return true;
3249}
3250
3251// Check whether an AND with Mask is suitable for a TEST UNDER MASK
3252// instruction and whether the CC value is descriptive enough to handle
3253// a comparison of type Opcode between the AND result and CmpVal.
3254// CCMask says which comparison result is being tested and BitSize is
3255// the number of bits in the operands. If TEST UNDER MASK can be used,
3256// return the corresponding CC mask, otherwise return 0.
3257static unsigned getTestUnderMaskCond(unsigned BitSize, unsigned CCMask,
3258 uint64_t Mask, uint64_t CmpVal,
3259 unsigned ICmpType) {
3260 assert(Mask != 0 && "ANDs with zero should have been removed by now");
3261
3262 // Check whether the mask is suitable for TMHH, TMHL, TMLH or TMLL.
3263 if (!SystemZ::isImmLL(Mask) && !SystemZ::isImmLH(Mask) &&
3264 !SystemZ::isImmHL(Mask) && !SystemZ::isImmHH(Mask))
3265 return 0;
3266
3267 // Work out the masks for the lowest and highest bits.
3269 uint64_t Low = uint64_t(1) << llvm::countr_zero(Mask);
3270
3271 // Signed ordered comparisons are effectively unsigned if the sign
3272 // bit is dropped.
3273 bool EffectivelyUnsigned = (ICmpType != SystemZICMP::SignedOnly);
3274
3275 // Check for equality comparisons with 0, or the equivalent.
3276 if (CmpVal == 0) {
3277 if (CCMask == SystemZ::CCMASK_CMP_EQ)
3279 if (CCMask == SystemZ::CCMASK_CMP_NE)
3281 }
3282 if (EffectivelyUnsigned && CmpVal > 0 && CmpVal <= Low) {
3283 if (CCMask == SystemZ::CCMASK_CMP_LT)
3285 if (CCMask == SystemZ::CCMASK_CMP_GE)
3287 }
3288 if (EffectivelyUnsigned && CmpVal < Low) {
3289 if (CCMask == SystemZ::CCMASK_CMP_LE)
3291 if (CCMask == SystemZ::CCMASK_CMP_GT)
3293 }
3294
3295 // Check for equality comparisons with the mask, or the equivalent.
3296 if (CmpVal == Mask) {
3297 if (CCMask == SystemZ::CCMASK_CMP_EQ)
3299 if (CCMask == SystemZ::CCMASK_CMP_NE)
3301 }
3302 if (EffectivelyUnsigned && CmpVal >= Mask - Low && CmpVal < Mask) {
3303 if (CCMask == SystemZ::CCMASK_CMP_GT)
3305 if (CCMask == SystemZ::CCMASK_CMP_LE)
3307 }
3308 if (EffectivelyUnsigned && CmpVal > Mask - Low && CmpVal <= Mask) {
3309 if (CCMask == SystemZ::CCMASK_CMP_GE)
3311 if (CCMask == SystemZ::CCMASK_CMP_LT)
3313 }
3314
3315 // Check for ordered comparisons with the top bit.
3316 if (EffectivelyUnsigned && CmpVal >= Mask - High && CmpVal < High) {
3317 if (CCMask == SystemZ::CCMASK_CMP_LE)
3319 if (CCMask == SystemZ::CCMASK_CMP_GT)
3321 }
3322 if (EffectivelyUnsigned && CmpVal > Mask - High && CmpVal <= High) {
3323 if (CCMask == SystemZ::CCMASK_CMP_LT)
3325 if (CCMask == SystemZ::CCMASK_CMP_GE)
3327 }
3328
3329 // If there are just two bits, we can do equality checks for Low and High
3330 // as well.
3331 if (Mask == Low + High) {
3332 if (CCMask == SystemZ::CCMASK_CMP_EQ && CmpVal == Low)
3334 if (CCMask == SystemZ::CCMASK_CMP_NE && CmpVal == Low)
3336 if (CCMask == SystemZ::CCMASK_CMP_EQ && CmpVal == High)
3338 if (CCMask == SystemZ::CCMASK_CMP_NE && CmpVal == High)
3340 }
3341
3342 // Looks like we've exhausted our options.
3343 return 0;
3344}
3345
3346// See whether C can be implemented as a TEST UNDER MASK instruction.
3347// Update the arguments with the TM version if so.
3349 Comparison &C) {
3350 // Use VECTOR TEST UNDER MASK for i128 operations.
3351 if (C.Op0.getValueType() == MVT::i128) {
3352 // We can use VTM for EQ/NE comparisons of x & y against 0.
3353 if (C.Op0.getOpcode() == ISD::AND &&
3354 (C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3355 C.CCMask == SystemZ::CCMASK_CMP_NE)) {
3356 auto *Mask = dyn_cast<ConstantSDNode>(C.Op1);
3357 if (Mask && Mask->getAPIntValue() == 0) {
3358 C.Opcode = SystemZISD::VTM;
3359 C.Op1 = DAG.getNode(ISD::BITCAST, DL, MVT::v16i8, C.Op0.getOperand(1));
3360 C.Op0 = DAG.getNode(ISD::BITCAST, DL, MVT::v16i8, C.Op0.getOperand(0));
3361 C.CCValid = SystemZ::CCMASK_VCMP;
3362 if (C.CCMask == SystemZ::CCMASK_CMP_EQ)
3363 C.CCMask = SystemZ::CCMASK_VCMP_ALL;
3364 else
3365 C.CCMask = SystemZ::CCMASK_VCMP_ALL ^ C.CCValid;
3366 }
3367 }
3368 return;
3369 }
3370
3371 // Check that we have a comparison with a constant.
3372 auto *ConstOp1 = dyn_cast<ConstantSDNode>(C.Op1);
3373 if (!ConstOp1)
3374 return;
3375 uint64_t CmpVal = ConstOp1->getZExtValue();
3376
3377 // Check whether the nonconstant input is an AND with a constant mask.
3378 Comparison NewC(C);
3379 uint64_t MaskVal;
3380 ConstantSDNode *Mask = nullptr;
3381 if (C.Op0.getOpcode() == ISD::AND) {
3382 NewC.Op0 = C.Op0.getOperand(0);
3383 NewC.Op1 = C.Op0.getOperand(1);
3384 Mask = dyn_cast<ConstantSDNode>(NewC.Op1);
3385 if (!Mask)
3386 return;
3387 MaskVal = Mask->getZExtValue();
3388 } else {
3389 // There is no instruction to compare with a 64-bit immediate
3390 // so use TMHH instead if possible. We need an unsigned ordered
3391 // comparison with an i64 immediate.
3392 if (NewC.Op0.getValueType() != MVT::i64 ||
3393 NewC.CCMask == SystemZ::CCMASK_CMP_EQ ||
3394 NewC.CCMask == SystemZ::CCMASK_CMP_NE ||
3395 NewC.ICmpType == SystemZICMP::SignedOnly)
3396 return;
3397 // Convert LE and GT comparisons into LT and GE.
3398 if (NewC.CCMask == SystemZ::CCMASK_CMP_LE ||
3399 NewC.CCMask == SystemZ::CCMASK_CMP_GT) {
3400 if (CmpVal == uint64_t(-1))
3401 return;
3402 CmpVal += 1;
3403 NewC.CCMask ^= SystemZ::CCMASK_CMP_EQ;
3404 }
3405 // If the low N bits of Op1 are zero than the low N bits of Op0 can
3406 // be masked off without changing the result.
3407 MaskVal = -(CmpVal & -CmpVal);
3408 NewC.ICmpType = SystemZICMP::UnsignedOnly;
3409 }
3410 if (!MaskVal)
3411 return;
3412
3413 // Check whether the combination of mask, comparison value and comparison
3414 // type are suitable.
3415 unsigned BitSize = NewC.Op0.getValueSizeInBits();
3416 unsigned NewCCMask, ShiftVal;
3417 if (NewC.ICmpType != SystemZICMP::SignedOnly &&
3418 NewC.Op0.getOpcode() == ISD::SHL &&
3419 isSimpleShift(NewC.Op0, ShiftVal) &&
3420 (MaskVal >> ShiftVal != 0) &&
3421 ((CmpVal >> ShiftVal) << ShiftVal) == CmpVal &&
3422 (NewCCMask = getTestUnderMaskCond(BitSize, NewC.CCMask,
3423 MaskVal >> ShiftVal,
3424 CmpVal >> ShiftVal,
3425 SystemZICMP::Any))) {
3426 NewC.Op0 = NewC.Op0.getOperand(0);
3427 MaskVal >>= ShiftVal;
3428 } else if (NewC.ICmpType != SystemZICMP::SignedOnly &&
3429 NewC.Op0.getOpcode() == ISD::SRL &&
3430 isSimpleShift(NewC.Op0, ShiftVal) &&
3431 (MaskVal << ShiftVal != 0) &&
3432 ((CmpVal << ShiftVal) >> ShiftVal) == CmpVal &&
3433 (NewCCMask = getTestUnderMaskCond(BitSize, NewC.CCMask,
3434 MaskVal << ShiftVal,
3435 CmpVal << ShiftVal,
3437 NewC.Op0 = NewC.Op0.getOperand(0);
3438 MaskVal <<= ShiftVal;
3439 } else {
3440 NewCCMask = getTestUnderMaskCond(BitSize, NewC.CCMask, MaskVal, CmpVal,
3441 NewC.ICmpType);
3442 if (!NewCCMask)
3443 return;
3444 }
3445
3446 // Go ahead and make the change.
3447 C.Opcode = SystemZISD::TM;
3448 C.Op0 = NewC.Op0;
3449 if (Mask && Mask->getZExtValue() == MaskVal)
3450 C.Op1 = SDValue(Mask, 0);
3451 else
3452 C.Op1 = DAG.getConstant(MaskVal, DL, C.Op0.getValueType());
3453 C.CCValid = SystemZ::CCMASK_TM;
3454 C.CCMask = NewCCMask;
3455}
3456
3457// Implement i128 comparison in vector registers.
3458static void adjustICmp128(SelectionDAG &DAG, const SDLoc &DL,
3459 Comparison &C) {
3460 if (C.Opcode != SystemZISD::ICMP)
3461 return;
3462 if (C.Op0.getValueType() != MVT::i128)
3463 return;
3464
3465 // Recognize vector comparison reductions.
3466 if ((C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3467 C.CCMask == SystemZ::CCMASK_CMP_NE) &&
3468 (isNullConstant(C.Op1) || isAllOnesConstant(C.Op1))) {
3469 bool CmpEq = C.CCMask == SystemZ::CCMASK_CMP_EQ;
3470 bool CmpNull = isNullConstant(C.Op1);
3471 SDValue Src = peekThroughBitcasts(C.Op0);
3472 if (Src.hasOneUse() && isBitwiseNot(Src)) {
3473 Src = Src.getOperand(0);
3474 CmpNull = !CmpNull;
3475 }
3476 unsigned Opcode = 0;
3477 if (Src.hasOneUse()) {
3478 switch (Src.getOpcode()) {
3479 case SystemZISD::VICMPE: Opcode = SystemZISD::VICMPES; break;
3480 case SystemZISD::VICMPH: Opcode = SystemZISD::VICMPHS; break;
3481 case SystemZISD::VICMPHL: Opcode = SystemZISD::VICMPHLS; break;
3482 case SystemZISD::VFCMPE: Opcode = SystemZISD::VFCMPES; break;
3483 case SystemZISD::VFCMPH: Opcode = SystemZISD::VFCMPHS; break;
3484 case SystemZISD::VFCMPHE: Opcode = SystemZISD::VFCMPHES; break;
3485 default: break;
3486 }
3487 }
3488 if (Opcode) {
3489 C.Opcode = Opcode;
3490 C.Op0 = Src->getOperand(0);
3491 C.Op1 = Src->getOperand(1);
3492 C.CCValid = SystemZ::CCMASK_VCMP;
3494 if (!CmpEq)
3495 C.CCMask ^= C.CCValid;
3496 return;
3497 }
3498 }
3499
3500 // Everything below here is not useful if we have native i128 compares.
3501 if (DAG.getSubtarget<SystemZSubtarget>().hasVectorEnhancements3())
3502 return;
3503
3504 // (In-)Equality comparisons can be implemented via VCEQGS.
3505 if (C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3506 C.CCMask == SystemZ::CCMASK_CMP_NE) {
3507 C.Opcode = SystemZISD::VICMPES;
3508 C.Op0 = DAG.getNode(ISD::BITCAST, DL, MVT::v2i64, C.Op0);
3509 C.Op1 = DAG.getNode(ISD::BITCAST, DL, MVT::v2i64, C.Op1);
3510 C.CCValid = SystemZ::CCMASK_VCMP;
3511 if (C.CCMask == SystemZ::CCMASK_CMP_EQ)
3512 C.CCMask = SystemZ::CCMASK_VCMP_ALL;
3513 else
3514 C.CCMask = SystemZ::CCMASK_VCMP_ALL ^ C.CCValid;
3515 return;
3516 }
3517
3518 // Normalize other comparisons to GT.
3519 bool Swap = false, Invert = false;
3520 switch (C.CCMask) {
3521 case SystemZ::CCMASK_CMP_GT: break;
3522 case SystemZ::CCMASK_CMP_LT: Swap = true; break;
3523 case SystemZ::CCMASK_CMP_LE: Invert = true; break;
3524 case SystemZ::CCMASK_CMP_GE: Swap = Invert = true; break;
3525 default: llvm_unreachable("Invalid integer condition!");
3526 }
3527 if (Swap)
3528 std::swap(C.Op0, C.Op1);
3529
3530 if (C.ICmpType == SystemZICMP::UnsignedOnly)
3531 C.Opcode = SystemZISD::UCMP128HI;
3532 else
3533 C.Opcode = SystemZISD::SCMP128HI;
3534 C.CCValid = SystemZ::CCMASK_ANY;
3535 C.CCMask = SystemZ::CCMASK_1;
3536
3537 if (Invert)
3538 C.CCMask ^= C.CCValid;
3539}
3540
3541// See whether the comparison argument contains a redundant AND
3542// and remove it if so. This sometimes happens due to the generic
3543// BRCOND expansion.
3545 Comparison &C) {
3546 if (C.Op0.getOpcode() != ISD::AND)
3547 return;
3548 auto *Mask = dyn_cast<ConstantSDNode>(C.Op0.getOperand(1));
3549 if (!Mask || Mask->getValueSizeInBits(0) > 64)
3550 return;
3551 KnownBits Known = DAG.computeKnownBits(C.Op0.getOperand(0));
3552 if ((~Known.Zero).getZExtValue() & ~Mask->getZExtValue())
3553 return;
3554
3555 C.Op0 = C.Op0.getOperand(0);
3556}
3557
3558// Return a Comparison that tests the condition-code result of intrinsic
3559// node Call against constant integer CC using comparison code Cond.
3560// Opcode is the opcode of the SystemZISD operation for the intrinsic
3561// and CCValid is the set of possible condition-code results.
3562static Comparison getIntrinsicCmp(SelectionDAG &DAG, unsigned Opcode,
3563 SDValue Call, unsigned CCValid, uint64_t CC,
3565 Comparison C(Call, SDValue(), SDValue());
3566 C.Opcode = Opcode;
3567 C.CCValid = CCValid;
3568 if (Cond == ISD::SETEQ)
3569 // bit 3 for CC==0, bit 0 for CC==3, always false for CC>3.
3570 C.CCMask = CC < 4 ? 1 << (3 - CC) : 0;
3571 else if (Cond == ISD::SETNE)
3572 // ...and the inverse of that.
3573 C.CCMask = CC < 4 ? ~(1 << (3 - CC)) : -1;
3574 else if (Cond == ISD::SETLT || Cond == ISD::SETULT)
3575 // bits above bit 3 for CC==0 (always false), bits above bit 0 for CC==3,
3576 // always true for CC>3.
3577 C.CCMask = CC < 4 ? ~0U << (4 - CC) : -1;
3578 else if (Cond == ISD::SETGE || Cond == ISD::SETUGE)
3579 // ...and the inverse of that.
3580 C.CCMask = CC < 4 ? ~(~0U << (4 - CC)) : 0;
3581 else if (Cond == ISD::SETLE || Cond == ISD::SETULE)
3582 // bit 3 and above for CC==0, bit 0 and above for CC==3 (always true),
3583 // always true for CC>3.
3584 C.CCMask = CC < 4 ? ~0U << (3 - CC) : -1;
3585 else if (Cond == ISD::SETGT || Cond == ISD::SETUGT)
3586 // ...and the inverse of that.
3587 C.CCMask = CC < 4 ? ~(~0U << (3 - CC)) : 0;
3588 else
3589 llvm_unreachable("Unexpected integer comparison type");
3590 C.CCMask &= CCValid;
3591 return C;
3592}
3593
3594// Decide how to implement a comparison of type Cond between CmpOp0 with CmpOp1.
3595static Comparison getCmp(SelectionDAG &DAG, SDValue CmpOp0, SDValue CmpOp1,
3596 ISD::CondCode Cond, const SDLoc &DL,
3597 SDValue Chain = SDValue(),
3598 bool IsSignaling = false) {
3599 if (CmpOp1.getOpcode() == ISD::Constant) {
3600 assert(!Chain);
3601 unsigned Opcode, CCValid;
3602 if (CmpOp0.getOpcode() == ISD::INTRINSIC_W_CHAIN &&
3603 CmpOp0.getResNo() == 0 && CmpOp0->hasNUsesOfValue(1, 0) &&
3604 isIntrinsicWithCCAndChain(CmpOp0, Opcode, CCValid))
3605 return getIntrinsicCmp(DAG, Opcode, CmpOp0, CCValid,
3606 CmpOp1->getAsZExtVal(), Cond);
3607 if (CmpOp0.getOpcode() == ISD::INTRINSIC_WO_CHAIN &&
3608 CmpOp0.getResNo() == CmpOp0->getNumValues() - 1 &&
3609 isIntrinsicWithCC(CmpOp0, Opcode, CCValid))
3610 return getIntrinsicCmp(DAG, Opcode, CmpOp0, CCValid,
3611 CmpOp1->getAsZExtVal(), Cond);
3612 }
3613 Comparison C(CmpOp0, CmpOp1, Chain);
3614 C.CCMask = CCMaskForCondCode(Cond);
3615 if (C.Op0.getValueType().isFloatingPoint()) {
3616 C.CCValid = SystemZ::CCMASK_FCMP;
3617 if (!C.Chain)
3618 C.Opcode = SystemZISD::FCMP;
3619 else if (!IsSignaling)
3620 C.Opcode = SystemZISD::STRICT_FCMP;
3621 else
3622 C.Opcode = SystemZISD::STRICT_FCMPS;
3624 } else {
3625 assert(!C.Chain);
3626 C.CCValid = SystemZ::CCMASK_ICMP;
3627 C.Opcode = SystemZISD::ICMP;
3628 // Choose the type of comparison. Equality and inequality tests can
3629 // use either signed or unsigned comparisons. The choice also doesn't
3630 // matter if both sign bits are known to be clear. In those cases we
3631 // want to give the main isel code the freedom to choose whichever
3632 // form fits best.
3633 if (C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3634 C.CCMask == SystemZ::CCMASK_CMP_NE ||
3635 (DAG.SignBitIsZero(C.Op0) && DAG.SignBitIsZero(C.Op1)))
3636 C.ICmpType = SystemZICMP::Any;
3637 else if (C.CCMask & SystemZ::CCMASK_CMP_UO)
3638 C.ICmpType = SystemZICMP::UnsignedOnly;
3639 else
3640 C.ICmpType = SystemZICMP::SignedOnly;
3641 C.CCMask &= ~SystemZ::CCMASK_CMP_UO;
3642 adjustForRedundantAnd(DAG, DL, C);
3643 adjustZeroCmp(DAG, DL, C);
3644 adjustSubwordCmp(DAG, DL, C);
3645 adjustForSubtraction(DAG, DL, C);
3647 adjustICmpTruncate(DAG, DL, C);
3648 }
3649
3650 if (shouldSwapCmpOperands(C)) {
3651 std::swap(C.Op0, C.Op1);
3652 C.CCMask = SystemZ::reverseCCMask(C.CCMask);
3653 }
3654
3656 adjustICmp128(DAG, DL, C);
3658 return C;
3659}
3660
3661// Emit the comparison instruction described by C.
3662static SDValue emitCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C) {
3663 if (!C.Op1.getNode()) {
3664 if (C.Opcode == SystemZISD::CMP_STACKGUARD)
3665 return DAG.getNode(SystemZISD::CMP_STACKGUARD, DL, MVT::i32, C.Op0);
3666 SDNode *Node;
3667 switch (C.Op0.getOpcode()) {
3669 Node = emitIntrinsicWithCCAndChain(DAG, C.Op0, C.Opcode);
3670 return SDValue(Node, 0);
3672 Node = emitIntrinsicWithCC(DAG, C.Op0, C.Opcode);
3673 return SDValue(Node, Node->getNumValues() - 1);
3674 default:
3675 llvm_unreachable("Invalid comparison operands");
3676 }
3677 }
3678 if (C.Opcode == SystemZISD::ICMP)
3679 return DAG.getNode(SystemZISD::ICMP, DL, MVT::i32, C.Op0, C.Op1,
3680 DAG.getTargetConstant(C.ICmpType, DL, MVT::i32));
3681 if (C.Opcode == SystemZISD::TM) {
3682 bool RegisterOnly = (bool(C.CCMask & SystemZ::CCMASK_TM_MIXED_MSB_0) !=
3684 return DAG.getNode(SystemZISD::TM, DL, MVT::i32, C.Op0, C.Op1,
3685 DAG.getTargetConstant(RegisterOnly, DL, MVT::i32));
3686 }
3687 if (C.Opcode == SystemZISD::VICMPES ||
3688 C.Opcode == SystemZISD::VICMPHS ||
3689 C.Opcode == SystemZISD::VICMPHLS ||
3690 C.Opcode == SystemZISD::VFCMPES ||
3691 C.Opcode == SystemZISD::VFCMPHS ||
3692 C.Opcode == SystemZISD::VFCMPHES) {
3693 EVT IntVT = C.Op0.getValueType().changeVectorElementTypeToInteger();
3694 SDVTList VTs = DAG.getVTList(IntVT, MVT::i32);
3695 SDValue Val = DAG.getNode(C.Opcode, DL, VTs, C.Op0, C.Op1);
3696 return SDValue(Val.getNode(), 1);
3697 }
3698 if (C.Chain) {
3699 SDVTList VTs = DAG.getVTList(MVT::i32, MVT::Other);
3700 return DAG.getNode(C.Opcode, DL, VTs, C.Chain, C.Op0, C.Op1);
3701 }
3702 return DAG.getNode(C.Opcode, DL, MVT::i32, C.Op0, C.Op1);
3703}
3704
3705// Implement a 32-bit *MUL_LOHI operation by extending both operands to
3706// 64 bits. Extend is the extension type to use. Store the high part
3707// in Hi and the low part in Lo.
3708static void lowerMUL_LOHI32(SelectionDAG &DAG, const SDLoc &DL, unsigned Extend,
3709 SDValue Op0, SDValue Op1, SDValue &Hi,
3710 SDValue &Lo) {
3711 Op0 = DAG.getNode(Extend, DL, MVT::i64, Op0);
3712 Op1 = DAG.getNode(Extend, DL, MVT::i64, Op1);
3713 SDValue Mul = DAG.getNode(ISD::MUL, DL, MVT::i64, Op0, Op1);
3714 Hi = DAG.getNode(ISD::SRL, DL, MVT::i64, Mul,
3715 DAG.getConstant(32, DL, MVT::i64));
3716 Hi = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Hi);
3717 Lo = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Mul);
3718}
3719
3720// Lower a binary operation that produces two VT results, one in each
3721// half of a GR128 pair. Op0 and Op1 are the VT operands to the operation,
3722// and Opcode performs the GR128 operation. Store the even register result
3723// in Even and the odd register result in Odd.
3724static void lowerGR128Binary(SelectionDAG &DAG, const SDLoc &DL, EVT VT,
3725 unsigned Opcode, SDValue Op0, SDValue Op1,
3726 SDValue &Even, SDValue &Odd) {
3727 SDValue Result = DAG.getNode(Opcode, DL, MVT::Untyped, Op0, Op1);
3728 bool Is32Bit = is32Bit(VT);
3729 Even = DAG.getTargetExtractSubreg(SystemZ::even128(Is32Bit), DL, VT, Result);
3730 Odd = DAG.getTargetExtractSubreg(SystemZ::odd128(Is32Bit), DL, VT, Result);
3731}
3732
3733// Return an i32 value that is 1 if the CC value produced by CCReg is
3734// in the mask CCMask and 0 otherwise. CC is known to have a value
3735// in CCValid, so other values can be ignored.
3736static SDValue emitSETCC(SelectionDAG &DAG, const SDLoc &DL, SDValue CCReg,
3737 unsigned CCValid, unsigned CCMask) {
3738 SDValue Ops[] = {DAG.getConstant(1, DL, MVT::i32),
3739 DAG.getConstant(0, DL, MVT::i32),
3740 DAG.getTargetConstant(CCValid, DL, MVT::i32),
3741 DAG.getTargetConstant(CCMask, DL, MVT::i32), CCReg};
3742 return DAG.getNode(SystemZISD::SELECT_CCMASK, DL, MVT::i32, Ops);
3743}
3744
3745// Return the SystemISD vector comparison operation for CC, or 0 if it cannot
3746// be done directly. Mode is CmpMode::Int for integer comparisons, CmpMode::FP
3747// for regular floating-point comparisons, CmpMode::StrictFP for strict (quiet)
3748// floating-point comparisons, and CmpMode::SignalingFP for strict signaling
3749// floating-point comparisons.
3752 switch (CC) {
3753 case ISD::SETOEQ:
3754 case ISD::SETEQ:
3755 switch (Mode) {
3756 case CmpMode::Int: return SystemZISD::VICMPE;
3757 case CmpMode::FP: return SystemZISD::VFCMPE;
3758 case CmpMode::StrictFP: return SystemZISD::STRICT_VFCMPE;
3759 case CmpMode::SignalingFP: return SystemZISD::STRICT_VFCMPES;
3760 }
3761 llvm_unreachable("Bad mode");
3762
3763 case ISD::SETOGE:
3764 case ISD::SETGE:
3765 switch (Mode) {
3766 case CmpMode::Int: return 0;
3767 case CmpMode::FP: return SystemZISD::VFCMPHE;
3768 case CmpMode::StrictFP: return SystemZISD::STRICT_VFCMPHE;
3769 case CmpMode::SignalingFP: return SystemZISD::STRICT_VFCMPHES;
3770 }
3771 llvm_unreachable("Bad mode");
3772
3773 case ISD::SETOGT:
3774 case ISD::SETGT:
3775 switch (Mode) {
3776 case CmpMode::Int: return SystemZISD::VICMPH;
3777 case CmpMode::FP: return SystemZISD::VFCMPH;
3778 case CmpMode::StrictFP: return SystemZISD::STRICT_VFCMPH;
3779 case CmpMode::SignalingFP: return SystemZISD::STRICT_VFCMPHS;
3780 }
3781 llvm_unreachable("Bad mode");
3782
3783 case ISD::SETUGT:
3784 switch (Mode) {
3785 case CmpMode::Int: return SystemZISD::VICMPHL;
3786 case CmpMode::FP: return 0;
3787 case CmpMode::StrictFP: return 0;
3788 case CmpMode::SignalingFP: return 0;
3789 }
3790 llvm_unreachable("Bad mode");
3791
3792 default:
3793 return 0;
3794 }
3795}
3796
3797// Return the SystemZISD vector comparison operation for CC or its inverse,
3798// or 0 if neither can be done directly. Indicate in Invert whether the
3799// result is for the inverse of CC. Mode is as above.
3801 bool &Invert) {
3802 if (unsigned Opcode = getVectorComparison(CC, Mode)) {
3803 Invert = false;
3804 return Opcode;
3805 }
3806
3807 CC = ISD::getSetCCInverse(CC, Mode == CmpMode::Int ? MVT::i32 : MVT::f32);
3808 if (unsigned Opcode = getVectorComparison(CC, Mode)) {
3809 Invert = true;
3810 return Opcode;
3811 }
3812
3813 return 0;
3814}
3815
3816// Return a v2f64 that contains the extended form of elements Start and Start+1
3817// of v4f32 value Op. If Chain is nonnull, return the strict form.
3818static SDValue expandV4F32ToV2F64(SelectionDAG &DAG, int Start, const SDLoc &DL,
3819 SDValue Op, SDValue Chain) {
3820 int Mask[] = { Start, -1, Start + 1, -1 };
3821 Op = DAG.getVectorShuffle(MVT::v4f32, DL, Op, DAG.getUNDEF(MVT::v4f32), Mask);
3822 if (Chain) {
3823 SDVTList VTs = DAG.getVTList(MVT::v2f64, MVT::Other);
3824 return DAG.getNode(SystemZISD::STRICT_VEXTEND, DL, VTs, Chain, Op);
3825 }
3826 return DAG.getNode(SystemZISD::VEXTEND, DL, MVT::v2f64, Op);
3827}
3828
3829// Build a comparison of vectors CmpOp0 and CmpOp1 using opcode Opcode,
3830// producing a result of type VT. If Chain is nonnull, return the strict form.
3831SDValue SystemZTargetLowering::getVectorCmp(SelectionDAG &DAG, unsigned Opcode,
3832 const SDLoc &DL, EVT VT,
3833 SDValue CmpOp0,
3834 SDValue CmpOp1,
3835 SDValue Chain) const {
3836 // There is no hardware support for v4f32 (unless we have the vector
3837 // enhancements facility 1), so extend the vector into two v2f64s
3838 // and compare those.
3839 if (CmpOp0.getValueType() == MVT::v4f32 &&
3840 !Subtarget.hasVectorEnhancements1()) {
3841 SDValue H0 = expandV4F32ToV2F64(DAG, 0, DL, CmpOp0, Chain);
3842 SDValue L0 = expandV4F32ToV2F64(DAG, 2, DL, CmpOp0, Chain);
3843 SDValue H1 = expandV4F32ToV2F64(DAG, 0, DL, CmpOp1, Chain);
3844 SDValue L1 = expandV4F32ToV2F64(DAG, 2, DL, CmpOp1, Chain);
3845 if (Chain) {
3846 SDVTList VTs = DAG.getVTList(MVT::v2i64, MVT::Other);
3847 SDValue HRes = DAG.getNode(Opcode, DL, VTs, Chain, H0, H1);
3848 SDValue LRes = DAG.getNode(Opcode, DL, VTs, Chain, L0, L1);
3849 SDValue Res = DAG.getNode(SystemZISD::PACK, DL, VT, HRes, LRes);
3850 SDValue Chains[6] = { H0.getValue(1), L0.getValue(1),
3851 H1.getValue(1), L1.getValue(1),
3852 HRes.getValue(1), LRes.getValue(1) };
3853 SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
3854 SDValue Ops[2] = { Res, NewChain };
3855 return DAG.getMergeValues(Ops, DL);
3856 }
3857 SDValue HRes = DAG.getNode(Opcode, DL, MVT::v2i64, H0, H1);
3858 SDValue LRes = DAG.getNode(Opcode, DL, MVT::v2i64, L0, L1);
3859 return DAG.getNode(SystemZISD::PACK, DL, VT, HRes, LRes);
3860 }
3861 if (Chain) {
3862 SDVTList VTs = DAG.getVTList(VT, MVT::Other);
3863 return DAG.getNode(Opcode, DL, VTs, Chain, CmpOp0, CmpOp1);
3864 }
3865 return DAG.getNode(Opcode, DL, VT, CmpOp0, CmpOp1);
3866}
3867
3868// Lower a vector comparison of type CC between CmpOp0 and CmpOp1, producing
3869// an integer mask of type VT. If Chain is nonnull, we have a strict
3870// floating-point comparison. If in addition IsSignaling is true, we have
3871// a strict signaling floating-point comparison.
3872SDValue SystemZTargetLowering::lowerVectorSETCC(SelectionDAG &DAG,
3873 const SDLoc &DL, EVT VT,
3874 ISD::CondCode CC,
3875 SDValue CmpOp0,
3876 SDValue CmpOp1,
3877 SDValue Chain,
3878 bool IsSignaling) const {
3879 bool IsFP = CmpOp0.getValueType().isFloatingPoint();
3880 assert (!Chain || IsFP);
3881 assert (!IsSignaling || Chain);
3882 CmpMode Mode = IsSignaling ? CmpMode::SignalingFP :
3883 Chain ? CmpMode::StrictFP : IsFP ? CmpMode::FP : CmpMode::Int;
3884 bool Invert = false;
3885 SDValue Cmp;
3886 switch (CC) {
3887 // Handle tests for order using (or (ogt y x) (oge x y)).
3888 case ISD::SETUO:
3889 Invert = true;
3890 [[fallthrough]];
3891 case ISD::SETO: {
3892 assert(IsFP && "Unexpected integer comparison");
3893 SDValue LT = getVectorCmp(DAG, getVectorComparison(ISD::SETOGT, Mode),
3894 DL, VT, CmpOp1, CmpOp0, Chain);
3895 SDValue GE = getVectorCmp(DAG, getVectorComparison(ISD::SETOGE, Mode),
3896 DL, VT, CmpOp0, CmpOp1, Chain);
3897 Cmp = DAG.getNode(ISD::OR, DL, VT, LT, GE);
3898 if (Chain)
3899 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
3900 LT.getValue(1), GE.getValue(1));
3901 break;
3902 }
3903
3904 // Handle <> tests using (or (ogt y x) (ogt x y)).
3905 case ISD::SETUEQ:
3906 Invert = true;
3907 [[fallthrough]];
3908 case ISD::SETONE: {
3909 assert(IsFP && "Unexpected integer comparison");
3910 SDValue LT = getVectorCmp(DAG, getVectorComparison(ISD::SETOGT, Mode),
3911 DL, VT, CmpOp1, CmpOp0, Chain);
3912 SDValue GT = getVectorCmp(DAG, getVectorComparison(ISD::SETOGT, Mode),
3913 DL, VT, CmpOp0, CmpOp1, Chain);
3914 Cmp = DAG.getNode(ISD::OR, DL, VT, LT, GT);
3915 if (Chain)
3916 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
3917 LT.getValue(1), GT.getValue(1));
3918 break;
3919 }
3920
3921 // Otherwise a single comparison is enough. It doesn't really
3922 // matter whether we try the inversion or the swap first, since
3923 // there are no cases where both work.
3924 default:
3925 // Optimize sign-bit comparisons to signed compares.
3926 if (Mode == CmpMode::Int && (CC == ISD::SETEQ || CC == ISD::SETNE) &&
3928 unsigned EltSize = VT.getVectorElementType().getSizeInBits();
3929 APInt Mask;
3930 if (CmpOp0.getOpcode() == ISD::AND
3931 && ISD::isConstantSplatVector(CmpOp0.getOperand(1).getNode(), Mask)
3932 && Mask == APInt::getSignMask(EltSize)) {
3933 CC = CC == ISD::SETEQ ? ISD::SETGE : ISD::SETLT;
3934 CmpOp0 = CmpOp0.getOperand(0);
3935 }
3936 }
3937 if (unsigned Opcode = getVectorComparisonOrInvert(CC, Mode, Invert))
3938 Cmp = getVectorCmp(DAG, Opcode, DL, VT, CmpOp0, CmpOp1, Chain);
3939 else {
3941 if (unsigned Opcode = getVectorComparisonOrInvert(CC, Mode, Invert))
3942 Cmp = getVectorCmp(DAG, Opcode, DL, VT, CmpOp1, CmpOp0, Chain);
3943 else
3944 llvm_unreachable("Unhandled comparison");
3945 }
3946 if (Chain)
3947 Chain = Cmp.getValue(1);
3948 break;
3949 }
3950 if (Invert) {
3951 SDValue Mask =
3952 DAG.getSplatBuildVector(VT, DL, DAG.getAllOnesConstant(DL, MVT::i64));
3953 Cmp = DAG.getNode(ISD::XOR, DL, VT, Cmp, Mask);
3954 }
3955 if (Chain && Chain.getNode() != Cmp.getNode()) {
3956 SDValue Ops[2] = { Cmp, Chain };
3957 Cmp = DAG.getMergeValues(Ops, DL);
3958 }
3959 return Cmp;
3960}
3961
3962SDValue SystemZTargetLowering::lowerSETCC(SDValue Op,
3963 SelectionDAG &DAG) const {
3964 SDValue CmpOp0 = Op.getOperand(0);
3965 SDValue CmpOp1 = Op.getOperand(1);
3966 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(2))->get();
3967 SDLoc DL(Op);
3968 EVT VT = Op.getValueType();
3969 if (VT.isVector())
3970 return lowerVectorSETCC(DAG, DL, VT, CC, CmpOp0, CmpOp1);
3971
3972 Comparison C(getCmp(DAG, CmpOp0, CmpOp1, CC, DL));
3973 SDValue CCReg = emitCmp(DAG, DL, C);
3974 return emitSETCC(DAG, DL, CCReg, C.CCValid, C.CCMask);
3975}
3976
3977SDValue SystemZTargetLowering::lowerSTRICT_FSETCC(SDValue Op,
3978 SelectionDAG &DAG,
3979 bool IsSignaling) const {
3980 SDValue Chain = Op.getOperand(0);
3981 SDValue CmpOp0 = Op.getOperand(1);
3982 SDValue CmpOp1 = Op.getOperand(2);
3983 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(3))->get();
3984 SDLoc DL(Op);
3985 EVT VT = Op.getNode()->getValueType(0);
3986 if (VT.isVector()) {
3987 SDValue Res = lowerVectorSETCC(DAG, DL, VT, CC, CmpOp0, CmpOp1,
3988 Chain, IsSignaling);
3989 return Res.getValue(Op.getResNo());
3990 }
3991
3992 Comparison C(getCmp(DAG, CmpOp0, CmpOp1, CC, DL, Chain, IsSignaling));
3993 SDValue CCReg = emitCmp(DAG, DL, C);
3994 CCReg->setFlags(Op->getFlags());
3995 SDValue Result = emitSETCC(DAG, DL, CCReg, C.CCValid, C.CCMask);
3996 SDValue Ops[2] = { Result, CCReg.getValue(1) };
3997 return DAG.getMergeValues(Ops, DL);
3998}
3999
4000SDValue SystemZTargetLowering::lowerBR_CC(SDValue Op, SelectionDAG &DAG) const {
4001 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(1))->get();
4002 SDValue CmpOp0 = Op.getOperand(2);
4003 SDValue CmpOp1 = Op.getOperand(3);
4004 SDValue Dest = Op.getOperand(4);
4005 SDLoc DL(Op);
4006
4007 Comparison C(getCmp(DAG, CmpOp0, CmpOp1, CC, DL));
4008 SDValue CCReg = emitCmp(DAG, DL, C);
4009 return DAG.getNode(
4010 SystemZISD::BR_CCMASK, DL, Op.getValueType(), Op.getOperand(0),
4011 DAG.getTargetConstant(C.CCValid, DL, MVT::i32),
4012 DAG.getTargetConstant(C.CCMask, DL, MVT::i32), Dest, CCReg);
4013}
4014
4015// Return true if Pos is CmpOp and Neg is the negative of CmpOp,
4016// allowing Pos and Neg to be wider than CmpOp.
4017static bool isAbsolute(SDValue CmpOp, SDValue Pos, SDValue Neg) {
4018 return (Neg.getOpcode() == ISD::SUB &&
4019 Neg.getOperand(0).getOpcode() == ISD::Constant &&
4020 Neg.getConstantOperandVal(0) == 0 && Neg.getOperand(1) == Pos &&
4021 (Pos == CmpOp || (Pos.getOpcode() == ISD::SIGN_EXTEND &&
4022 Pos.getOperand(0) == CmpOp)));
4023}
4024
4025// Return the absolute or negative absolute of Op; IsNegative decides which.
4027 bool IsNegative) {
4028 Op = DAG.getNode(ISD::ABS, DL, Op.getValueType(), Op);
4029 if (IsNegative)
4030 Op = DAG.getNode(ISD::SUB, DL, Op.getValueType(),
4031 DAG.getConstant(0, DL, Op.getValueType()), Op);
4032 return Op;
4033}
4034
4036 Comparison C, SDValue TrueOp, SDValue FalseOp) {
4037 EVT VT = MVT::i128;
4038 unsigned Op;
4039
4040 if (C.CCMask == SystemZ::CCMASK_CMP_NE ||
4041 C.CCMask == SystemZ::CCMASK_CMP_GE ||
4042 C.CCMask == SystemZ::CCMASK_CMP_LE) {
4043 std::swap(TrueOp, FalseOp);
4044 C.CCMask ^= C.CCValid;
4045 }
4046 if (C.CCMask == SystemZ::CCMASK_CMP_LT) {
4047 std::swap(C.Op0, C.Op1);
4048 C.CCMask = SystemZ::CCMASK_CMP_GT;
4049 }
4050 switch (C.CCMask) {
4052 Op = SystemZISD::VICMPE;
4053 break;
4055 if (C.ICmpType == SystemZICMP::UnsignedOnly)
4056 Op = SystemZISD::VICMPHL;
4057 else
4058 Op = SystemZISD::VICMPH;
4059 break;
4060 default:
4061 llvm_unreachable("Unhandled comparison");
4062 break;
4063 }
4064
4065 SDValue Mask = DAG.getNode(Op, DL, VT, C.Op0, C.Op1);
4066 TrueOp = DAG.getNode(ISD::AND, DL, VT, TrueOp, Mask);
4067 FalseOp = DAG.getNode(ISD::AND, DL, VT, FalseOp, DAG.getNOT(DL, Mask, VT));
4068 return DAG.getNode(ISD::OR, DL, VT, TrueOp, FalseOp);
4069}
4070
4071SDValue SystemZTargetLowering::lowerSELECT_CC(SDValue Op,
4072 SelectionDAG &DAG) const {
4073 SDValue CmpOp0 = Op.getOperand(0);
4074 SDValue CmpOp1 = Op.getOperand(1);
4075 SDValue TrueOp = Op.getOperand(2);
4076 SDValue FalseOp = Op.getOperand(3);
4077 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(4))->get();
4078 SDLoc DL(Op);
4079
4080 // SELECT_CC involving f16 will not have the cmp-ops promoted by the
4081 // legalizer, as it will be handled according to the type of the resulting
4082 // value. Extend them here if needed.
4083 if (CmpOp0.getSimpleValueType() == MVT::f16) {
4084 CmpOp0 = DAG.getFPExtendOrRound(CmpOp0, SDLoc(CmpOp0), MVT::f32);
4085 CmpOp1 = DAG.getFPExtendOrRound(CmpOp1, SDLoc(CmpOp1), MVT::f32);
4086 }
4087
4088 Comparison C(getCmp(DAG, CmpOp0, CmpOp1, CC, DL));
4089
4090 // Check for absolute and negative-absolute selections, including those
4091 // where the comparison value is sign-extended (for LPGFR and LNGFR).
4092 // This check supplements the one in DAGCombiner.
4093 if (C.Opcode == SystemZISD::ICMP && C.CCMask != SystemZ::CCMASK_CMP_EQ &&
4094 C.CCMask != SystemZ::CCMASK_CMP_NE &&
4095 C.Op1.getOpcode() == ISD::Constant &&
4096 cast<ConstantSDNode>(C.Op1)->getValueSizeInBits(0) <= 64 &&
4097 C.Op1->getAsZExtVal() == 0) {
4098 if (isAbsolute(C.Op0, TrueOp, FalseOp))
4099 return getAbsolute(DAG, DL, TrueOp, C.CCMask & SystemZ::CCMASK_CMP_LT);
4100 if (isAbsolute(C.Op0, FalseOp, TrueOp))
4101 return getAbsolute(DAG, DL, FalseOp, C.CCMask & SystemZ::CCMASK_CMP_GT);
4102 }
4103
4104 if (Subtarget.hasVectorEnhancements3() &&
4105 C.Opcode == SystemZISD::ICMP &&
4106 C.Op0.getValueType() == MVT::i128 &&
4107 TrueOp.getValueType() == MVT::i128) {
4108 return getI128Select(DAG, DL, C, TrueOp, FalseOp);
4109 }
4110
4111 SDValue CCReg = emitCmp(DAG, DL, C);
4112 SDValue Ops[] = {TrueOp, FalseOp,
4113 DAG.getTargetConstant(C.CCValid, DL, MVT::i32),
4114 DAG.getTargetConstant(C.CCMask, DL, MVT::i32), CCReg};
4115
4116 return DAG.getNode(SystemZISD::SELECT_CCMASK, DL, Op.getValueType(), Ops);
4117}
4118
4119SDValue SystemZTargetLowering::lowerGlobalAddress(GlobalAddressSDNode *Node,
4120 SelectionDAG &DAG) const {
4121 SDLoc DL(Node);
4122 const GlobalValue *GV = Node->getGlobal();
4123 int64_t Offset = Node->getOffset();
4124 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4126
4128 if (Subtarget.isPC32DBLSymbol(GV, CM)) {
4129 if (isInt<32>(Offset)) {
4130 // Assign anchors at 1<<12 byte boundaries.
4131 uint64_t Anchor = Offset & ~uint64_t(0xfff);
4132 Result = DAG.getTargetGlobalAddress(GV, DL, PtrVT, Anchor);
4133 Result = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4134
4135 // The offset can be folded into the address if it is aligned to a
4136 // halfword.
4137 Offset -= Anchor;
4138 if (Offset != 0 && (Offset & 1) == 0) {
4139 SDValue Full =
4140 DAG.getTargetGlobalAddress(GV, DL, PtrVT, Anchor + Offset);
4141 Result = DAG.getNode(SystemZISD::PCREL_OFFSET, DL, PtrVT, Full, Result);
4142 Offset = 0;
4143 }
4144 } else {
4145 // Conservatively load a constant offset greater than 32 bits into a
4146 // register below.
4147 Result = DAG.getTargetGlobalAddress(GV, DL, PtrVT);
4148 Result = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4149 }
4150 } else if (Subtarget.isTargetELF()) {
4151 Result = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, SystemZII::MO_GOT);
4152 Result = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4153 Result = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Result,
4155 } else if (Subtarget.isTargetzOS()) {
4156 Result = getADAEntry(DAG, GV, DL, PtrVT);
4157 } else
4158 llvm_unreachable("Unexpected Subtarget");
4159
4160 // If there was a non-zero offset that we didn't fold, create an explicit
4161 // addition for it.
4162 if (Offset != 0)
4163 Result = DAG.getNode(ISD::ADD, DL, PtrVT, Result,
4164 DAG.getSignedConstant(Offset, DL, PtrVT));
4165
4166 return Result;
4167}
4168
4169SDValue SystemZTargetLowering::lowerTLSGetOffset(GlobalAddressSDNode *Node,
4170 SelectionDAG &DAG,
4171 unsigned Opcode,
4172 SDValue GOTOffset) const {
4173 SDLoc DL(Node);
4174 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4175 SDValue Chain = DAG.getEntryNode();
4176 SDValue Glue;
4177
4180 report_fatal_error("In GHC calling convention TLS is not supported");
4181
4182 // __tls_get_offset takes the GOT offset in %r2 and the GOT in %r12.
4183 SDValue GOT = DAG.getGLOBAL_OFFSET_TABLE(PtrVT);
4184 Chain = DAG.getCopyToReg(Chain, DL, SystemZ::R12D, GOT, Glue);
4185 Glue = Chain.getValue(1);
4186 Chain = DAG.getCopyToReg(Chain, DL, SystemZ::R2D, GOTOffset, Glue);
4187 Glue = Chain.getValue(1);
4188
4189 // The first call operand is the chain and the second is the TLS symbol.
4191 Ops.push_back(Chain);
4192 Ops.push_back(DAG.getTargetGlobalAddress(Node->getGlobal(), DL,
4193 Node->getValueType(0),
4194 0, 0));
4195
4196 // Add argument registers to the end of the list so that they are
4197 // known live into the call.
4198 Ops.push_back(DAG.getRegister(SystemZ::R2D, PtrVT));
4199 Ops.push_back(DAG.getRegister(SystemZ::R12D, PtrVT));
4200
4201 // Add a register mask operand representing the call-preserved registers.
4202 const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo();
4203 const uint32_t *Mask =
4204 TRI->getCallPreservedMask(DAG.getMachineFunction(), CallingConv::C);
4205 assert(Mask && "Missing call preserved mask for calling convention");
4206 Ops.push_back(DAG.getRegisterMask(Mask));
4207
4208 // Glue the call to the argument copies.
4209 Ops.push_back(Glue);
4210
4211 // Emit the call.
4212 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
4213 Chain = DAG.getNode(Opcode, DL, NodeTys, Ops);
4214 Glue = Chain.getValue(1);
4215
4216 // Copy the return value from %r2.
4217 return DAG.getCopyFromReg(Chain, DL, SystemZ::R2D, PtrVT, Glue);
4218}
4219
4220SDValue SystemZTargetLowering::lowerThreadPointer(const SDLoc &DL,
4221 SelectionDAG &DAG) const {
4222 SDValue Chain = DAG.getEntryNode();
4223 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4224
4225 // The high part of the thread pointer is in access register 0.
4226 SDValue TPHi = DAG.getCopyFromReg(Chain, DL, SystemZ::A0, MVT::i32);
4227 TPHi = DAG.getNode(ISD::ANY_EXTEND, DL, PtrVT, TPHi);
4228
4229 // The low part of the thread pointer is in access register 1.
4230 SDValue TPLo = DAG.getCopyFromReg(Chain, DL, SystemZ::A1, MVT::i32);
4231 TPLo = DAG.getNode(ISD::ZERO_EXTEND, DL, PtrVT, TPLo);
4232
4233 // Merge them into a single 64-bit address.
4234 SDValue TPHiShifted = DAG.getNode(ISD::SHL, DL, PtrVT, TPHi,
4235 DAG.getConstant(32, DL, PtrVT));
4236 return DAG.getNode(ISD::OR, DL, PtrVT, TPHiShifted, TPLo);
4237}
4238
4239SDValue SystemZTargetLowering::lowerGlobalTLSAddress(GlobalAddressSDNode *Node,
4240 SelectionDAG &DAG) const {
4241 if (DAG.getTarget().useEmulatedTLS())
4242 return LowerToTLSEmulatedModel(Node, DAG);
4243 SDLoc DL(Node);
4244 const GlobalValue *GV = Node->getGlobal();
4245 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4246 TLSModel::Model model = DAG.getTarget().getTLSModel(GV);
4247
4250 report_fatal_error("In GHC calling convention TLS is not supported");
4251
4252 SDValue TP = lowerThreadPointer(DL, DAG);
4253
4254 // Get the offset of GA from the thread pointer, based on the TLS model.
4256 switch (model) {
4258 // Load the GOT offset of the tls_index (module ID / per-symbol offset).
4259 SystemZConstantPoolValue *CPV =
4261
4262 Offset = DAG.getConstantPool(CPV, PtrVT, Align(8));
4263 Offset = DAG.getLoad(
4264 PtrVT, DL, DAG.getEntryNode(), Offset,
4266
4267 // Call __tls_get_offset to retrieve the offset.
4268 Offset = lowerTLSGetOffset(Node, DAG, SystemZISD::TLS_GDCALL, Offset);
4269 break;
4270 }
4271
4273 // Load the GOT offset of the module ID.
4274 SystemZConstantPoolValue *CPV =
4276
4277 Offset = DAG.getConstantPool(CPV, PtrVT, Align(8));
4278 Offset = DAG.getLoad(
4279 PtrVT, DL, DAG.getEntryNode(), Offset,
4281
4282 // Call __tls_get_offset to retrieve the module base offset.
4283 Offset = lowerTLSGetOffset(Node, DAG, SystemZISD::TLS_LDCALL, Offset);
4284
4285 // Note: The SystemZLDCleanupPass will remove redundant computations
4286 // of the module base offset. Count total number of local-dynamic
4287 // accesses to trigger execution of that pass.
4288 SystemZMachineFunctionInfo* MFI =
4289 DAG.getMachineFunction().getInfo<SystemZMachineFunctionInfo>();
4291
4292 // Add the per-symbol offset.
4294
4295 SDValue DTPOffset = DAG.getConstantPool(CPV, PtrVT, Align(8));
4296 DTPOffset = DAG.getLoad(
4297 PtrVT, DL, DAG.getEntryNode(), DTPOffset,
4299
4300 Offset = DAG.getNode(ISD::ADD, DL, PtrVT, Offset, DTPOffset);
4301 break;
4302 }
4303
4304 case TLSModel::InitialExec: {
4305 // Load the offset from the GOT.
4306 Offset = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0,
4308 Offset = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Offset);
4309 Offset =
4310 DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Offset,
4312 break;
4313 }
4314
4315 case TLSModel::LocalExec: {
4316 // Force the offset into the constant pool and load it from there.
4317 SystemZConstantPoolValue *CPV =
4319
4320 Offset = DAG.getConstantPool(CPV, PtrVT, Align(8));
4321 Offset = DAG.getLoad(
4322 PtrVT, DL, DAG.getEntryNode(), Offset,
4324 break;
4325 }
4326 }
4327
4328 // Add the base and offset together.
4329 return DAG.getNode(ISD::ADD, DL, PtrVT, TP, Offset);
4330}
4331
4332SDValue SystemZTargetLowering::lowerBlockAddress(BlockAddressSDNode *Node,
4333 SelectionDAG &DAG) const {
4334 SDLoc DL(Node);
4335 const BlockAddress *BA = Node->getBlockAddress();
4336 int64_t Offset = Node->getOffset();
4337 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4338
4339 SDValue Result = DAG.getTargetBlockAddress(BA, PtrVT, Offset);
4340 Result = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4341 return Result;
4342}
4343
4344SDValue SystemZTargetLowering::lowerJumpTable(JumpTableSDNode *JT,
4345 SelectionDAG &DAG) const {
4346 SDLoc DL(JT);
4347 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4348 SDValue Result = DAG.getTargetJumpTable(JT->getIndex(), PtrVT);
4349
4350 // Use LARL to load the address of the table.
4351 return DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4352}
4353
4354SDValue SystemZTargetLowering::lowerConstantPool(ConstantPoolSDNode *CP,
4355 SelectionDAG &DAG) const {
4356 SDLoc DL(CP);
4357 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4358
4361 Result =
4362 DAG.getTargetConstantPool(CP->getMachineCPVal(), PtrVT, CP->getAlign());
4363 else
4364 Result = DAG.getTargetConstantPool(CP->getConstVal(), PtrVT, CP->getAlign(),
4365 CP->getOffset());
4366
4367 // Use LARL to load the address of the constant pool entry.
4368 return DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4369}
4370
4371SDValue SystemZTargetLowering::lowerFRAMEADDR(SDValue Op,
4372 SelectionDAG &DAG) const {
4373 auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
4375 MachineFrameInfo &MFI = MF.getFrameInfo();
4376 MFI.setFrameAddressIsTaken(true);
4377
4378 SDLoc DL(Op);
4379 unsigned Depth = Op.getConstantOperandVal(0);
4380 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4381
4382 // By definition, the frame address is the address of the back chain. (In
4383 // the case of packed stack without backchain, return the address where the
4384 // backchain would have been stored. This will either be an unused space or
4385 // contain a saved register).
4386 int BackChainIdx = TFL->getOrCreateFramePointerSaveIndex(MF);
4387 SDValue BackChain = DAG.getFrameIndex(BackChainIdx, PtrVT);
4388
4389 if (Depth > 0) {
4390 // FIXME The frontend should detect this case.
4391 if (!MF.getSubtarget<SystemZSubtarget>().hasBackChain())
4392 report_fatal_error("Unsupported stack frame traversal count");
4393
4394 SDValue Offset = DAG.getConstant(TFL->getBackchainOffset(MF), DL, PtrVT);
4395 while (Depth--) {
4396 BackChain = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), BackChain,
4397 MachinePointerInfo());
4398 BackChain = DAG.getNode(ISD::ADD, DL, PtrVT, BackChain, Offset);
4399 }
4400 }
4401
4402 return BackChain;
4403}
4404
4405SDValue SystemZTargetLowering::lowerRETURNADDR(SDValue Op,
4406 SelectionDAG &DAG) const {
4408 MachineFrameInfo &MFI = MF.getFrameInfo();
4409 MFI.setReturnAddressIsTaken(true);
4410
4411 SDLoc DL(Op);
4412 unsigned Depth = Op.getConstantOperandVal(0);
4413 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4414
4415 if (Depth > 0) {
4416 // FIXME The frontend should detect this case.
4417 if (!MF.getSubtarget<SystemZSubtarget>().hasBackChain())
4418 report_fatal_error("Unsupported stack frame traversal count");
4419
4420 SDValue FrameAddr = lowerFRAMEADDR(Op, DAG);
4421 const auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
4422 int Offset = TFL->getReturnAddressOffset(MF);
4423 SDValue Ptr = DAG.getNode(ISD::ADD, DL, PtrVT, FrameAddr,
4424 DAG.getSignedConstant(Offset, DL, PtrVT));
4425 return DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Ptr,
4426 MachinePointerInfo());
4427 }
4428
4429 // Return R14D (Elf) / R7D (XPLINK), which has the return address. Mark it an
4430 // implicit live-in.
4431 SystemZCallingConventionRegisters *CCR = Subtarget.getSpecialRegisters();
4433 &SystemZ::GR64BitRegClass);
4434 return DAG.getCopyFromReg(DAG.getEntryNode(), DL, LinkReg, PtrVT);
4435}
4436
4437SDValue SystemZTargetLowering::lowerBITCAST(SDValue Op,
4438 SelectionDAG &DAG) const {
4439 SDLoc DL(Op);
4440 SDValue In = Op.getOperand(0);
4441 EVT InVT = In.getValueType();
4442 EVT ResVT = Op.getValueType();
4443
4444 // Convert loads directly. This is normally done by DAGCombiner,
4445 // but we need this case for bitcasts that are created during lowering
4446 // and which are then lowered themselves.
4447 if (auto *LoadN = dyn_cast<LoadSDNode>(In))
4448 if (ISD::isNormalLoad(LoadN)) {
4449 SDValue NewLoad = DAG.getLoad(ResVT, DL, LoadN->getChain(),
4450 LoadN->getBasePtr(), LoadN->getMemOperand());
4451 // Update the chain uses.
4452 DAG.ReplaceAllUsesOfValueWith(SDValue(LoadN, 1), NewLoad.getValue(1));
4453 return NewLoad;
4454 }
4455
4456 if (InVT == MVT::i32 && ResVT == MVT::f32) {
4457 SDValue In64;
4458 if (Subtarget.hasHighWord()) {
4459 SDNode *U64 = DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL,
4460 MVT::i64);
4461 In64 = DAG.getTargetInsertSubreg(SystemZ::subreg_h32, DL,
4462 MVT::i64, SDValue(U64, 0), In);
4463 } else {
4464 In64 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, In);
4465 In64 = DAG.getNode(ISD::SHL, DL, MVT::i64, In64,
4466 DAG.getConstant(32, DL, MVT::i64));
4467 }
4468 SDValue Out64 = DAG.getNode(ISD::BITCAST, DL, MVT::f64, In64);
4469 return DAG.getTargetExtractSubreg(SystemZ::subreg_h32,
4470 DL, MVT::f32, Out64);
4471 }
4472 if (InVT == MVT::f32 && ResVT == MVT::i32) {
4473 SDNode *U64 = DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::f64);
4474 SDValue In64 = DAG.getTargetInsertSubreg(SystemZ::subreg_h32, DL,
4475 MVT::f64, SDValue(U64, 0), In);
4476 SDValue Out64 = DAG.getNode(ISD::BITCAST, DL, MVT::i64, In64);
4477 if (Subtarget.hasHighWord())
4478 return DAG.getTargetExtractSubreg(SystemZ::subreg_h32, DL,
4479 MVT::i32, Out64);
4480 SDValue Shift = DAG.getNode(ISD::SRL, DL, MVT::i64, Out64,
4481 DAG.getConstant(32, DL, MVT::i64));
4482 return DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Shift);
4483 }
4484 llvm_unreachable("Unexpected bitcast combination");
4485}
4486
4487SDValue SystemZTargetLowering::lowerVASTART(SDValue Op,
4488 SelectionDAG &DAG) const {
4489
4490 if (Subtarget.isTargetXPLINK64())
4491 return lowerVASTART_XPLINK(Op, DAG);
4492 else
4493 return lowerVASTART_ELF(Op, DAG);
4494}
4495
4496SDValue SystemZTargetLowering::lowerVASTART_XPLINK(SDValue Op,
4497 SelectionDAG &DAG) const {
4499 SystemZMachineFunctionInfo *FuncInfo =
4500 MF.getInfo<SystemZMachineFunctionInfo>();
4501
4502 SDLoc DL(Op);
4503
4504 // vastart just stores the address of the VarArgsFrameIndex slot into the
4505 // memory location argument.
4506 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4507 SDValue FR = DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(), PtrVT);
4508 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
4509 return DAG.getStore(Op.getOperand(0), DL, FR, Op.getOperand(1),
4510 MachinePointerInfo(SV));
4511}
4512
4513SDValue SystemZTargetLowering::lowerVASTART_ELF(SDValue Op,
4514 SelectionDAG &DAG) const {
4516 SystemZMachineFunctionInfo *FuncInfo =
4517 MF.getInfo<SystemZMachineFunctionInfo>();
4518 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4519
4520 SDValue Chain = Op.getOperand(0);
4521 SDValue Addr = Op.getOperand(1);
4522 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
4523 SDLoc DL(Op);
4524
4525 // The initial values of each field.
4526 const unsigned NumFields = 4;
4527 SDValue Fields[NumFields] = {
4528 DAG.getConstant(FuncInfo->getVarArgsFirstGPR(), DL, PtrVT),
4529 DAG.getConstant(FuncInfo->getVarArgsFirstFPR(), DL, PtrVT),
4530 DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(), PtrVT),
4531 DAG.getFrameIndex(FuncInfo->getRegSaveFrameIndex(), PtrVT)
4532 };
4533
4534 // Store each field into its respective slot.
4535 SDValue MemOps[NumFields];
4536 unsigned Offset = 0;
4537 for (unsigned I = 0; I < NumFields; ++I) {
4538 SDValue FieldAddr = Addr;
4539 if (Offset != 0)
4540 FieldAddr = DAG.getNode(ISD::ADD, DL, PtrVT, FieldAddr,
4542 MemOps[I] = DAG.getStore(Chain, DL, Fields[I], FieldAddr,
4543 MachinePointerInfo(SV, Offset));
4544 Offset += 8;
4545 }
4546 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOps);
4547}
4548
4549SDValue SystemZTargetLowering::lowerVACOPY(SDValue Op,
4550 SelectionDAG &DAG) const {
4551 SDValue Chain = Op.getOperand(0);
4552 SDValue DstPtr = Op.getOperand(1);
4553 SDValue SrcPtr = Op.getOperand(2);
4554 const Value *DstSV = cast<SrcValueSDNode>(Op.getOperand(3))->getValue();
4555 const Value *SrcSV = cast<SrcValueSDNode>(Op.getOperand(4))->getValue();
4556 SDLoc DL(Op);
4557
4558 uint32_t Sz =
4559 Subtarget.isTargetXPLINK64() ? getTargetMachine().getPointerSize(0) : 32;
4560 return DAG.getMemcpy(Chain, DL, DstPtr, SrcPtr, DAG.getIntPtrConstant(Sz, DL),
4561 Align(8), Align(8), /*isVolatile*/ false,
4562 /*AlwaysInline*/ false,
4563 /*CI=*/nullptr, std::nullopt, MachinePointerInfo(DstSV),
4564 MachinePointerInfo(SrcSV));
4565}
4566
4567SDValue
4568SystemZTargetLowering::lowerDYNAMIC_STACKALLOC(SDValue Op,
4569 SelectionDAG &DAG) const {
4570 if (Subtarget.isTargetXPLINK64())
4571 return lowerDYNAMIC_STACKALLOC_XPLINK(Op, DAG);
4572 else
4573 return lowerDYNAMIC_STACKALLOC_ELF(Op, DAG);
4574}
4575
4576SDValue
4577SystemZTargetLowering::lowerDYNAMIC_STACKALLOC_XPLINK(SDValue Op,
4578 SelectionDAG &DAG) const {
4579 const TargetFrameLowering *TFI = Subtarget.getFrameLowering();
4581 bool RealignOpt = !MF.getFunction().hasFnAttribute("no-realign-stack");
4582 SDValue Chain = Op.getOperand(0);
4583 SDValue Size = Op.getOperand(1);
4584 SDValue Align = Op.getOperand(2);
4585 SDLoc DL(Op);
4586
4587 // If user has set the no alignment function attribute, ignore
4588 // alloca alignments.
4589 uint64_t AlignVal = (RealignOpt ? Align->getAsZExtVal() : 0);
4590
4591 uint64_t StackAlign = TFI->getStackAlignment();
4592 uint64_t RequiredAlign = std::max(AlignVal, StackAlign);
4593 uint64_t ExtraAlignSpace = RequiredAlign - StackAlign;
4594
4595 SDValue NeededSpace = Size;
4596
4597 // Add extra space for alignment if needed.
4598 EVT PtrVT = getPointerTy(MF.getDataLayout());
4599 if (ExtraAlignSpace)
4600 NeededSpace = DAG.getNode(ISD::ADD, DL, PtrVT, NeededSpace,
4601 DAG.getConstant(ExtraAlignSpace, DL, PtrVT));
4602
4603 bool IsSigned = false;
4604 bool DoesNotReturn = false;
4605 bool IsReturnValueUsed = false;
4606 EVT VT = Op.getValueType();
4607 SDValue AllocaCall =
4608 makeExternalCall(Chain, DAG, "@@ALCAXP", VT, ArrayRef(NeededSpace),
4609 CallingConv::C, IsSigned, DL, DoesNotReturn,
4610 IsReturnValueUsed)
4611 .first;
4612
4613 // Perform a CopyFromReg from %GPR4 (stack pointer register). Chain and Glue
4614 // to end of call in order to ensure it isn't broken up from the call
4615 // sequence.
4616 auto &Regs = Subtarget.getSpecialRegisters<SystemZXPLINK64Registers>();
4617 Register SPReg = Regs.getStackPointerRegister();
4618 Chain = AllocaCall.getValue(1);
4619 SDValue Glue = AllocaCall.getValue(2);
4620 SDValue NewSPRegNode = DAG.getCopyFromReg(Chain, DL, SPReg, PtrVT, Glue);
4621 Chain = NewSPRegNode.getValue(1);
4622
4623 MVT PtrMVT = getPointerMemTy(MF.getDataLayout());
4624 SDValue ArgAdjust = DAG.getNode(SystemZISD::ADJDYNALLOC, DL, PtrMVT);
4625 SDValue Result = DAG.getNode(ISD::ADD, DL, PtrMVT, NewSPRegNode, ArgAdjust);
4626
4627 // Dynamically realign if needed.
4628 if (ExtraAlignSpace) {
4629 Result = DAG.getNode(ISD::ADD, DL, PtrVT, Result,
4630 DAG.getConstant(ExtraAlignSpace, DL, PtrVT));
4631 Result = DAG.getNode(ISD::AND, DL, PtrVT, Result,
4632 DAG.getConstant(~(RequiredAlign - 1), DL, PtrVT));
4633 }
4634
4635 SDValue Ops[2] = {Result, Chain};
4636 return DAG.getMergeValues(Ops, DL);
4637}
4638
4639SDValue
4640SystemZTargetLowering::lowerDYNAMIC_STACKALLOC_ELF(SDValue Op,
4641 SelectionDAG &DAG) const {
4642 const TargetFrameLowering *TFI = Subtarget.getFrameLowering();
4644 bool RealignOpt = !MF.getFunction().hasFnAttribute("no-realign-stack");
4645 bool StoreBackchain = MF.getSubtarget<SystemZSubtarget>().hasBackChain();
4646
4647 SDValue Chain = Op.getOperand(0);
4648 SDValue Size = Op.getOperand(1);
4649 SDValue Align = Op.getOperand(2);
4650 SDLoc DL(Op);
4651
4652 // If user has set the no alignment function attribute, ignore
4653 // alloca alignments.
4654 uint64_t AlignVal = (RealignOpt ? Align->getAsZExtVal() : 0);
4655
4656 uint64_t StackAlign = TFI->getStackAlignment();
4657 uint64_t RequiredAlign = std::max(AlignVal, StackAlign);
4658 uint64_t ExtraAlignSpace = RequiredAlign - StackAlign;
4659
4661 SDValue NeededSpace = Size;
4662
4663 // Get a reference to the stack pointer.
4664 SDValue OldSP = DAG.getCopyFromReg(Chain, DL, SPReg, MVT::i64);
4665
4666 // If we need a backchain, save it now.
4667 SDValue Backchain;
4668 if (StoreBackchain)
4669 Backchain = DAG.getLoad(MVT::i64, DL, Chain, getBackchainAddress(OldSP, DAG),
4670 MachinePointerInfo());
4671
4672 // Add extra space for alignment if needed.
4673 if (ExtraAlignSpace)
4674 NeededSpace = DAG.getNode(ISD::ADD, DL, MVT::i64, NeededSpace,
4675 DAG.getConstant(ExtraAlignSpace, DL, MVT::i64));
4676
4677 // Get the new stack pointer value.
4678 SDValue NewSP;
4679 if (hasInlineStackProbe(MF)) {
4680 NewSP = DAG.getNode(SystemZISD::PROBED_ALLOCA, DL,
4681 DAG.getVTList(MVT::i64, MVT::Other), Chain, OldSP, NeededSpace);
4682 Chain = NewSP.getValue(1);
4683 }
4684 else {
4685 NewSP = DAG.getNode(ISD::SUB, DL, MVT::i64, OldSP, NeededSpace);
4686 // Copy the new stack pointer back.
4687 Chain = DAG.getCopyToReg(Chain, DL, SPReg, NewSP);
4688 }
4689
4690 // The allocated data lives above the 160 bytes allocated for the standard
4691 // frame, plus any outgoing stack arguments. We don't know how much that
4692 // amounts to yet, so emit a special ADJDYNALLOC placeholder.
4693 SDValue ArgAdjust = DAG.getNode(SystemZISD::ADJDYNALLOC, DL, MVT::i64);
4694 SDValue Result = DAG.getNode(ISD::ADD, DL, MVT::i64, NewSP, ArgAdjust);
4695
4696 // Dynamically realign if needed.
4697 if (RequiredAlign > StackAlign) {
4698 Result =
4699 DAG.getNode(ISD::ADD, DL, MVT::i64, Result,
4700 DAG.getConstant(ExtraAlignSpace, DL, MVT::i64));
4701 Result =
4702 DAG.getNode(ISD::AND, DL, MVT::i64, Result,
4703 DAG.getConstant(~(RequiredAlign - 1), DL, MVT::i64));
4704 }
4705
4706 if (StoreBackchain)
4707 Chain = DAG.getStore(Chain, DL, Backchain, getBackchainAddress(NewSP, DAG),
4708 MachinePointerInfo());
4709
4710 SDValue Ops[2] = { Result, Chain };
4711 return DAG.getMergeValues(Ops, DL);
4712}
4713
4714SDValue SystemZTargetLowering::lowerGET_DYNAMIC_AREA_OFFSET(
4715 SDValue Op, SelectionDAG &DAG) const {
4716 SDLoc DL(Op);
4717
4718 return DAG.getNode(SystemZISD::ADJDYNALLOC, DL, MVT::i64);
4719}
4720
4721SDValue SystemZTargetLowering::lowerMULH(SDValue Op,
4722 SelectionDAG &DAG,
4723 unsigned Opcode) const {
4724 EVT VT = Op.getValueType();
4725 SDLoc DL(Op);
4726 SDValue Even, Odd;
4727
4728 // This custom expander is only used on z17 and later for 64-bit types.
4729 assert(!is32Bit(VT));
4730 assert(Subtarget.hasMiscellaneousExtensions2());
4731
4732 // SystemZISD::xMUL_LOHI returns the low result in the odd register and
4733 // the high result in the even register. Return the latter.
4734 lowerGR128Binary(DAG, DL, VT, Opcode,
4735 Op.getOperand(0), Op.getOperand(1), Even, Odd);
4736 return Even;
4737}
4738
4739SDValue SystemZTargetLowering::lowerSMUL_LOHI(SDValue Op,
4740 SelectionDAG &DAG) const {
4741 EVT VT = Op.getValueType();
4742 SDLoc DL(Op);
4743 SDValue Ops[2];
4744 if (is32Bit(VT))
4745 // Just do a normal 64-bit multiplication and extract the results.
4746 // We define this so that it can be used for constant division.
4747 lowerMUL_LOHI32(DAG, DL, ISD::SIGN_EXTEND, Op.getOperand(0),
4748 Op.getOperand(1), Ops[1], Ops[0]);
4749 else if (Subtarget.hasMiscellaneousExtensions2())
4750 // SystemZISD::SMUL_LOHI returns the low result in the odd register and
4751 // the high result in the even register. ISD::SMUL_LOHI is defined to
4752 // return the low half first, so the results are in reverse order.
4753 lowerGR128Binary(DAG, DL, VT, SystemZISD::SMUL_LOHI,
4754 Op.getOperand(0), Op.getOperand(1), Ops[1], Ops[0]);
4755 else {
4756 // Do a full 128-bit multiplication based on SystemZISD::UMUL_LOHI:
4757 //
4758 // (ll * rl) + ((lh * rl) << 64) + ((ll * rh) << 64)
4759 //
4760 // but using the fact that the upper halves are either all zeros
4761 // or all ones:
4762 //
4763 // (ll * rl) - ((lh & rl) << 64) - ((ll & rh) << 64)
4764 //
4765 // and grouping the right terms together since they are quicker than the
4766 // multiplication:
4767 //
4768 // (ll * rl) - (((lh & rl) + (ll & rh)) << 64)
4769 SDValue C63 = DAG.getConstant(63, DL, MVT::i64);
4770 SDValue LL = Op.getOperand(0);
4771 SDValue RL = Op.getOperand(1);
4772 SDValue LH = DAG.getNode(ISD::SRA, DL, VT, LL, C63);
4773 SDValue RH = DAG.getNode(ISD::SRA, DL, VT, RL, C63);
4774 // SystemZISD::UMUL_LOHI returns the low result in the odd register and
4775 // the high result in the even register. ISD::SMUL_LOHI is defined to
4776 // return the low half first, so the results are in reverse order.
4777 lowerGR128Binary(DAG, DL, VT, SystemZISD::UMUL_LOHI,
4778 LL, RL, Ops[1], Ops[0]);
4779 SDValue NegLLTimesRH = DAG.getNode(ISD::AND, DL, VT, LL, RH);
4780 SDValue NegLHTimesRL = DAG.getNode(ISD::AND, DL, VT, LH, RL);
4781 SDValue NegSum = DAG.getNode(ISD::ADD, DL, VT, NegLLTimesRH, NegLHTimesRL);
4782 Ops[1] = DAG.getNode(ISD::SUB, DL, VT, Ops[1], NegSum);
4783 }
4784 return DAG.getMergeValues(Ops, DL);
4785}
4786
4787SDValue SystemZTargetLowering::lowerUMUL_LOHI(SDValue Op,
4788 SelectionDAG &DAG) const {
4789 EVT VT = Op.getValueType();
4790 SDLoc DL(Op);
4791 SDValue Ops[2];
4792 if (is32Bit(VT))
4793 // Just do a normal 64-bit multiplication and extract the results.
4794 // We define this so that it can be used for constant division.
4795 lowerMUL_LOHI32(DAG, DL, ISD::ZERO_EXTEND, Op.getOperand(0),
4796 Op.getOperand(1), Ops[1], Ops[0]);
4797 else
4798 // SystemZISD::UMUL_LOHI returns the low result in the odd register and
4799 // the high result in the even register. ISD::UMUL_LOHI is defined to
4800 // return the low half first, so the results are in reverse order.
4801 lowerGR128Binary(DAG, DL, VT, SystemZISD::UMUL_LOHI,
4802 Op.getOperand(0), Op.getOperand(1), Ops[1], Ops[0]);
4803 return DAG.getMergeValues(Ops, DL);
4804}
4805
4806SDValue SystemZTargetLowering::lowerSDIVREM(SDValue Op,
4807 SelectionDAG &DAG) const {
4808 SDValue Op0 = Op.getOperand(0);
4809 SDValue Op1 = Op.getOperand(1);
4810 EVT VT = Op.getValueType();
4811 SDLoc DL(Op);
4812
4813 // We use DSGF for 32-bit division. This means the first operand must
4814 // always be 64-bit, and the second operand should be 32-bit whenever
4815 // that is possible, to improve performance.
4816 if (is32Bit(VT))
4817 Op0 = DAG.getNode(ISD::SIGN_EXTEND, DL, MVT::i64, Op0);
4818 else if (DAG.ComputeNumSignBits(Op1) > 32)
4819 Op1 = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Op1);
4820
4821 // DSG(F) returns the remainder in the even register and the
4822 // quotient in the odd register.
4823 SDValue Ops[2];
4824 lowerGR128Binary(DAG, DL, VT, SystemZISD::SDIVREM, Op0, Op1, Ops[1], Ops[0]);
4825 return DAG.getMergeValues(Ops, DL);
4826}
4827
4828SDValue SystemZTargetLowering::lowerUDIVREM(SDValue Op,
4829 SelectionDAG &DAG) const {
4830 EVT VT = Op.getValueType();
4831 SDLoc DL(Op);
4832
4833 // DL(G) returns the remainder in the even register and the
4834 // quotient in the odd register.
4835 SDValue Ops[2];
4836 lowerGR128Binary(DAG, DL, VT, SystemZISD::UDIVREM,
4837 Op.getOperand(0), Op.getOperand(1), Ops[1], Ops[0]);
4838 return DAG.getMergeValues(Ops, DL);
4839}
4840
4841SDValue SystemZTargetLowering::lowerOR(SDValue Op, SelectionDAG &DAG) const {
4842 assert(Op.getValueType() == MVT::i64 && "Should be 64-bit operation");
4843
4844 // Get the known-zero masks for each operand.
4845 SDValue Ops[] = {Op.getOperand(0), Op.getOperand(1)};
4846 KnownBits Known[2] = {DAG.computeKnownBits(Ops[0]),
4847 DAG.computeKnownBits(Ops[1])};
4848
4849 // See if the upper 32 bits of one operand and the lower 32 bits of the
4850 // other are known zero. They are the low and high operands respectively.
4851 uint64_t Masks[] = { Known[0].Zero.getZExtValue(),
4852 Known[1].Zero.getZExtValue() };
4853 unsigned High, Low;
4854 if ((Masks[0] >> 32) == 0xffffffff && uint32_t(Masks[1]) == 0xffffffff)
4855 High = 1, Low = 0;
4856 else if ((Masks[1] >> 32) == 0xffffffff && uint32_t(Masks[0]) == 0xffffffff)
4857 High = 0, Low = 1;
4858 else
4859 return Op;
4860
4861 SDValue LowOp = Ops[Low];
4862 SDValue HighOp = Ops[High];
4863
4864 // If the high part is a constant, we're better off using IILH.
4865 if (HighOp.getOpcode() == ISD::Constant)
4866 return Op;
4867
4868 // If the low part is a constant that is outside the range of LHI,
4869 // then we're better off using IILF.
4870 if (LowOp.getOpcode() == ISD::Constant) {
4871 int64_t Value = int32_t(LowOp->getAsZExtVal());
4872 if (!isInt<16>(Value))
4873 return Op;
4874 }
4875
4876 // Check whether the high part is an AND that doesn't change the
4877 // high 32 bits and just masks out low bits. We can skip it if so.
4878 if (HighOp.getOpcode() == ISD::AND &&
4879 HighOp.getOperand(1).getOpcode() == ISD::Constant) {
4880 SDValue HighOp0 = HighOp.getOperand(0);
4882 if (DAG.MaskedValueIsZero(HighOp0, APInt(64, ~(Mask | 0xffffffff))))
4883 HighOp = HighOp0;
4884 }
4885
4886 // Take advantage of the fact that all GR32 operations only change the
4887 // low 32 bits by truncating Low to an i32 and inserting it directly
4888 // using a subreg. The interesting cases are those where the truncation
4889 // can be folded.
4890 SDLoc DL(Op);
4891 SDValue Low32 = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, LowOp);
4892 return DAG.getTargetInsertSubreg(SystemZ::subreg_l32, DL,
4893 MVT::i64, HighOp, Low32);
4894}
4895
4896// Lower SADDO/SSUBO/UADDO/USUBO nodes.
4897SDValue SystemZTargetLowering::lowerXALUO(SDValue Op,
4898 SelectionDAG &DAG) const {
4899 SDNode *N = Op.getNode();
4900 SDValue LHS = N->getOperand(0);
4901 SDValue RHS = N->getOperand(1);
4902 SDLoc DL(N);
4903
4904 if (N->getValueType(0) == MVT::i128) {
4905 unsigned BaseOp = 0;
4906 unsigned FlagOp = 0;
4907 bool IsBorrow = false;
4908 switch (Op.getOpcode()) {
4909 default: llvm_unreachable("Unknown instruction!");
4910 case ISD::UADDO:
4911 BaseOp = ISD::ADD;
4912 FlagOp = SystemZISD::VACC;
4913 break;
4914 case ISD::USUBO:
4915 BaseOp = ISD::SUB;
4916 FlagOp = SystemZISD::VSCBI;
4917 IsBorrow = true;
4918 break;
4919 }
4920 SDValue Result = DAG.getNode(BaseOp, DL, MVT::i128, LHS, RHS);
4921 SDValue Flag = DAG.getNode(FlagOp, DL, MVT::i128, LHS, RHS);
4922 Flag = DAG.getNode(ISD::AssertZext, DL, MVT::i128, Flag,
4923 DAG.getValueType(MVT::i1));
4924 Flag = DAG.getZExtOrTrunc(Flag, DL, N->getValueType(1));
4925 if (IsBorrow)
4926 Flag = DAG.getNode(ISD::XOR, DL, Flag.getValueType(),
4927 Flag, DAG.getConstant(1, DL, Flag.getValueType()));
4928 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, Flag);
4929 }
4930
4931 unsigned BaseOp = 0;
4932 unsigned CCValid = 0;
4933 unsigned CCMask = 0;
4934
4935 switch (Op.getOpcode()) {
4936 default: llvm_unreachable("Unknown instruction!");
4937 case ISD::SADDO:
4938 BaseOp = SystemZISD::SADDO;
4939 CCValid = SystemZ::CCMASK_ARITH;
4941 break;
4942 case ISD::SSUBO:
4943 BaseOp = SystemZISD::SSUBO;
4944 CCValid = SystemZ::CCMASK_ARITH;
4946 break;
4947 case ISD::UADDO:
4948 BaseOp = SystemZISD::UADDO;
4949 CCValid = SystemZ::CCMASK_LOGICAL;
4951 break;
4952 case ISD::USUBO:
4953 BaseOp = SystemZISD::USUBO;
4954 CCValid = SystemZ::CCMASK_LOGICAL;
4956 break;
4957 }
4958
4959 SDVTList VTs = DAG.getVTList(N->getValueType(0), MVT::i32);
4960 SDValue Result = DAG.getNode(BaseOp, DL, VTs, LHS, RHS);
4961
4962 SDValue SetCC = emitSETCC(DAG, DL, Result.getValue(1), CCValid, CCMask);
4963 if (N->getValueType(1) == MVT::i1)
4964 SetCC = DAG.getNode(ISD::TRUNCATE, DL, MVT::i1, SetCC);
4965
4966 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, SetCC);
4967}
4968
4969static bool isAddCarryChain(SDValue Carry) {
4970 while (Carry.getOpcode() == ISD::UADDO_CARRY &&
4971 Carry->getValueType(0) != MVT::i128)
4972 Carry = Carry.getOperand(2);
4973 return Carry.getOpcode() == ISD::UADDO &&
4974 Carry->getValueType(0) != MVT::i128;
4975}
4976
4977static bool isSubBorrowChain(SDValue Carry) {
4978 while (Carry.getOpcode() == ISD::USUBO_CARRY &&
4979 Carry->getValueType(0) != MVT::i128)
4980 Carry = Carry.getOperand(2);
4981 return Carry.getOpcode() == ISD::USUBO &&
4982 Carry->getValueType(0) != MVT::i128;
4983}
4984
4985// Lower UADDO_CARRY/USUBO_CARRY nodes.
4986SDValue SystemZTargetLowering::lowerUADDSUBO_CARRY(SDValue Op,
4987 SelectionDAG &DAG) const {
4988
4989 SDNode *N = Op.getNode();
4990 MVT VT = N->getSimpleValueType(0);
4991
4992 // Let legalize expand this if it isn't a legal type yet.
4993 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT))
4994 return SDValue();
4995
4996 SDValue LHS = N->getOperand(0);
4997 SDValue RHS = N->getOperand(1);
4998 SDValue Carry = Op.getOperand(2);
4999 SDLoc DL(N);
5000
5001 if (VT == MVT::i128) {
5002 unsigned BaseOp = 0;
5003 unsigned FlagOp = 0;
5004 bool IsBorrow = false;
5005 switch (Op.getOpcode()) {
5006 default: llvm_unreachable("Unknown instruction!");
5007 case ISD::UADDO_CARRY:
5008 BaseOp = SystemZISD::VAC;
5009 FlagOp = SystemZISD::VACCC;
5010 break;
5011 case ISD::USUBO_CARRY:
5012 BaseOp = SystemZISD::VSBI;
5013 FlagOp = SystemZISD::VSBCBI;
5014 IsBorrow = true;
5015 break;
5016 }
5017 if (IsBorrow)
5018 Carry = DAG.getNode(ISD::XOR, DL, Carry.getValueType(),
5019 Carry, DAG.getConstant(1, DL, Carry.getValueType()));
5020 Carry = DAG.getZExtOrTrunc(Carry, DL, MVT::i128);
5021 SDValue Result = DAG.getNode(BaseOp, DL, MVT::i128, LHS, RHS, Carry);
5022 SDValue Flag = DAG.getNode(FlagOp, DL, MVT::i128, LHS, RHS, Carry);
5023 Flag = DAG.getNode(ISD::AssertZext, DL, MVT::i128, Flag,
5024 DAG.getValueType(MVT::i1));
5025 Flag = DAG.getZExtOrTrunc(Flag, DL, N->getValueType(1));
5026 if (IsBorrow)
5027 Flag = DAG.getNode(ISD::XOR, DL, Flag.getValueType(),
5028 Flag, DAG.getConstant(1, DL, Flag.getValueType()));
5029 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, Flag);
5030 }
5031
5032 unsigned BaseOp = 0;
5033 unsigned CCValid = 0;
5034 unsigned CCMask = 0;
5035
5036 switch (Op.getOpcode()) {
5037 default: llvm_unreachable("Unknown instruction!");
5038 case ISD::UADDO_CARRY:
5039 if (!isAddCarryChain(Carry))
5040 return SDValue();
5041
5042 BaseOp = SystemZISD::ADDCARRY;
5043 CCValid = SystemZ::CCMASK_LOGICAL;
5045 break;
5046 case ISD::USUBO_CARRY:
5047 if (!isSubBorrowChain(Carry))
5048 return SDValue();
5049
5050 BaseOp = SystemZISD::SUBCARRY;
5051 CCValid = SystemZ::CCMASK_LOGICAL;
5053 break;
5054 }
5055
5056 // Set the condition code from the carry flag.
5057 Carry = DAG.getNode(SystemZISD::GET_CCMASK, DL, MVT::i32, Carry,
5058 DAG.getConstant(CCValid, DL, MVT::i32),
5059 DAG.getConstant(CCMask, DL, MVT::i32));
5060
5061 SDVTList VTs = DAG.getVTList(VT, MVT::i32);
5062 SDValue Result = DAG.getNode(BaseOp, DL, VTs, LHS, RHS, Carry);
5063
5064 SDValue SetCC = emitSETCC(DAG, DL, Result.getValue(1), CCValid, CCMask);
5065 if (N->getValueType(1) == MVT::i1)
5066 SetCC = DAG.getNode(ISD::TRUNCATE, DL, MVT::i1, SetCC);
5067
5068 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, SetCC);
5069}
5070
5071SDValue SystemZTargetLowering::lowerCTPOP(SDValue Op,
5072 SelectionDAG &DAG) const {
5073 EVT VT = Op.getValueType();
5074 SDLoc DL(Op);
5075 Op = Op.getOperand(0);
5076
5077 if (VT.getScalarSizeInBits() == 128) {
5078 Op = DAG.getNode(ISD::BITCAST, DL, MVT::v2i64, Op);
5079 Op = DAG.getNode(ISD::CTPOP, DL, MVT::v2i64, Op);
5080 SDValue Tmp = DAG.getSplatBuildVector(MVT::v2i64, DL,
5081 DAG.getConstant(0, DL, MVT::i64));
5082 Op = DAG.getNode(SystemZISD::VSUM, DL, VT, Op, Tmp);
5083 return Op;
5084 }
5085
5086 // Handle vector types via VPOPCT.
5087 if (VT.isVector()) {
5088 Op = DAG.getNode(ISD::BITCAST, DL, MVT::v16i8, Op);
5089 Op = DAG.getNode(SystemZISD::POPCNT, DL, MVT::v16i8, Op);
5090 switch (VT.getScalarSizeInBits()) {
5091 case 8:
5092 break;
5093 case 16: {
5094 Op = DAG.getNode(ISD::BITCAST, DL, VT, Op);
5095 SDValue Shift = DAG.getConstant(8, DL, MVT::i32);
5096 SDValue Tmp = DAG.getNode(SystemZISD::VSHL_BY_SCALAR, DL, VT, Op, Shift);
5097 Op = DAG.getNode(ISD::ADD, DL, VT, Op, Tmp);
5098 Op = DAG.getNode(SystemZISD::VSRL_BY_SCALAR, DL, VT, Op, Shift);
5099 break;
5100 }
5101 case 32: {
5102 SDValue Tmp = DAG.getSplatBuildVector(MVT::v16i8, DL,
5103 DAG.getConstant(0, DL, MVT::i32));
5104 Op = DAG.getNode(SystemZISD::VSUM, DL, VT, Op, Tmp);
5105 break;
5106 }
5107 case 64: {
5108 SDValue Tmp = DAG.getSplatBuildVector(MVT::v16i8, DL,
5109 DAG.getConstant(0, DL, MVT::i32));
5110 Op = DAG.getNode(SystemZISD::VSUM, DL, MVT::v4i32, Op, Tmp);
5111 Op = DAG.getNode(SystemZISD::VSUM, DL, VT, Op, Tmp);
5112 break;
5113 }
5114 default:
5115 llvm_unreachable("Unexpected type");
5116 }
5117 return Op;
5118 }
5119
5120 // Get the known-zero mask for the operand.
5121 KnownBits Known = DAG.computeKnownBits(Op);
5122 unsigned NumSignificantBits = Known.getMaxValue().getActiveBits();
5123 if (NumSignificantBits == 0)
5124 return DAG.getConstant(0, DL, VT);
5125
5126 // Skip known-zero high parts of the operand.
5127 int64_t OrigBitSize = VT.getSizeInBits();
5128 int64_t BitSize = llvm::bit_ceil(NumSignificantBits);
5129 BitSize = std::min(BitSize, OrigBitSize);
5130
5131 // The POPCNT instruction counts the number of bits in each byte.
5132 Op = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op);
5133 Op = DAG.getNode(SystemZISD::POPCNT, DL, MVT::i64, Op);
5134 Op = DAG.getNode(ISD::TRUNCATE, DL, VT, Op);
5135
5136 // Add up per-byte counts in a binary tree. All bits of Op at
5137 // position larger than BitSize remain zero throughout.
5138 for (int64_t I = BitSize / 2; I >= 8; I = I / 2) {
5139 SDValue Tmp = DAG.getNode(ISD::SHL, DL, VT, Op, DAG.getConstant(I, DL, VT));
5140 if (BitSize != OrigBitSize)
5141 Tmp = DAG.getNode(ISD::AND, DL, VT, Tmp,
5142 DAG.getConstant(((uint64_t)1 << BitSize) - 1, DL, VT));
5143 Op = DAG.getNode(ISD::ADD, DL, VT, Op, Tmp);
5144 }
5145
5146 // Extract overall result from high byte.
5147 if (BitSize > 8)
5148 Op = DAG.getNode(ISD::SRL, DL, VT, Op,
5149 DAG.getConstant(BitSize - 8, DL, VT));
5150
5151 return Op;
5152}
5153
5154SDValue SystemZTargetLowering::lowerATOMIC_FENCE(SDValue Op,
5155 SelectionDAG &DAG) const {
5156 SDLoc DL(Op);
5157 AtomicOrdering FenceOrdering =
5158 static_cast<AtomicOrdering>(Op.getConstantOperandVal(1));
5159 SyncScope::ID FenceSSID =
5160 static_cast<SyncScope::ID>(Op.getConstantOperandVal(2));
5161
5162 // The only fence that needs an instruction is a sequentially-consistent
5163 // cross-thread fence.
5164 if (FenceOrdering == AtomicOrdering::SequentiallyConsistent &&
5165 FenceSSID == SyncScope::System) {
5166 return SDValue(DAG.getMachineNode(SystemZ::Serialize, DL, MVT::Other,
5167 Op.getOperand(0)),
5168 0);
5169 }
5170
5171 // MEMBARRIER is a compiler barrier; it codegens to a no-op.
5172 return DAG.getNode(ISD::MEMBARRIER, DL, MVT::Other, Op.getOperand(0));
5173}
5174
5175SDValue SystemZTargetLowering::lowerATOMIC_LOAD(SDValue Op,
5176 SelectionDAG &DAG) const {
5177 EVT RegVT = Op.getValueType();
5178 if (RegVT.getSizeInBits() == 128)
5179 return lowerATOMIC_LDST_I128(Op, DAG);
5180 return lowerLoadF16(Op, DAG);
5181}
5182
5183SDValue SystemZTargetLowering::lowerATOMIC_STORE(SDValue Op,
5184 SelectionDAG &DAG) const {
5185 auto *Node = cast<AtomicSDNode>(Op.getNode());
5186 if (Node->getMemoryVT().getSizeInBits() == 128)
5187 return lowerATOMIC_LDST_I128(Op, DAG);
5188 return lowerStoreF16(Op, DAG);
5189}
5190
5191SDValue SystemZTargetLowering::lowerATOMIC_LDST_I128(SDValue Op,
5192 SelectionDAG &DAG) const {
5193 auto *Node = cast<AtomicSDNode>(Op.getNode());
5194 assert(
5195 (Node->getMemoryVT() == MVT::i128 || Node->getMemoryVT() == MVT::f128) &&
5196 "Only custom lowering i128 or f128.");
5197 // Use same code to handle both legal and non-legal i128 types.
5199 LowerOperationWrapper(Node, Results, DAG);
5200 return DAG.getMergeValues(Results, SDLoc(Op));
5201}
5202
5203// Prepare for a Compare And Swap for a subword operation. This needs to be
5204// done in memory with 4 bytes at natural alignment.
5206 SDValue &AlignedAddr, SDValue &BitShift,
5207 SDValue &NegBitShift) {
5208 EVT PtrVT = Addr.getValueType();
5209 EVT WideVT = MVT::i32;
5210
5211 // Get the address of the containing word.
5212 AlignedAddr = DAG.getNode(ISD::AND, DL, PtrVT, Addr,
5213 DAG.getSignedConstant(-4, DL, PtrVT));
5214
5215 // Get the number of bits that the word must be rotated left in order
5216 // to bring the field to the top bits of a GR32.
5217 BitShift = DAG.getNode(ISD::SHL, DL, PtrVT, Addr,
5218 DAG.getConstant(3, DL, PtrVT));
5219 BitShift = DAG.getNode(ISD::TRUNCATE, DL, WideVT, BitShift);
5220
5221 // Get the complementing shift amount, for rotating a field in the top
5222 // bits back to its proper position.
5223 NegBitShift = DAG.getNode(ISD::SUB, DL, WideVT,
5224 DAG.getConstant(0, DL, WideVT), BitShift);
5225
5226}
5227
5228// Op is an 8-, 16-bit or 32-bit ATOMIC_LOAD_* operation. Lower the first
5229// two into the fullword ATOMIC_LOADW_* operation given by Opcode.
5230SDValue SystemZTargetLowering::lowerATOMIC_LOAD_OP(SDValue Op,
5231 SelectionDAG &DAG,
5232 unsigned Opcode) const {
5233 auto *Node = cast<AtomicSDNode>(Op.getNode());
5234
5235 // 32-bit operations need no special handling.
5236 EVT NarrowVT = Node->getMemoryVT();
5237 EVT WideVT = MVT::i32;
5238 if (NarrowVT == WideVT)
5239 return Op;
5240
5241 int64_t BitSize = NarrowVT.getSizeInBits();
5242 SDValue ChainIn = Node->getChain();
5243 SDValue Addr = Node->getBasePtr();
5244 SDValue Src2 = Node->getVal();
5245 MachineMemOperand *MMO = Node->getMemOperand();
5246 SDLoc DL(Node);
5247
5248 // Convert atomic subtracts of constants into additions.
5249 if (Opcode == SystemZISD::ATOMIC_LOADW_SUB)
5250 if (auto *Const = dyn_cast<ConstantSDNode>(Src2)) {
5251 Opcode = SystemZISD::ATOMIC_LOADW_ADD;
5252 Src2 = DAG.getSignedConstant(-Const->getSExtValue(), DL,
5253 Src2.getValueType());
5254 }
5255
5256 SDValue AlignedAddr, BitShift, NegBitShift;
5257 getCSAddressAndShifts(Addr, DAG, DL, AlignedAddr, BitShift, NegBitShift);
5258
5259 // Extend the source operand to 32 bits and prepare it for the inner loop.
5260 // ATOMIC_SWAPW uses RISBG to rotate the field left, but all other
5261 // operations require the source to be shifted in advance. (This shift
5262 // can be folded if the source is constant.) For AND and NAND, the lower
5263 // bits must be set, while for other opcodes they should be left clear.
5264 if (Opcode != SystemZISD::ATOMIC_SWAPW)
5265 Src2 = DAG.getNode(ISD::SHL, DL, WideVT, Src2,
5266 DAG.getConstant(32 - BitSize, DL, WideVT));
5267 if (Opcode == SystemZISD::ATOMIC_LOADW_AND ||
5268 Opcode == SystemZISD::ATOMIC_LOADW_NAND)
5269 Src2 = DAG.getNode(ISD::OR, DL, WideVT, Src2,
5270 DAG.getConstant(uint32_t(-1) >> BitSize, DL, WideVT));
5271
5272 // Construct the ATOMIC_LOADW_* node.
5273 SDVTList VTList = DAG.getVTList(WideVT, MVT::Other);
5274 SDValue Ops[] = { ChainIn, AlignedAddr, Src2, BitShift, NegBitShift,
5275 DAG.getConstant(BitSize, DL, WideVT) };
5276 SDValue AtomicOp = DAG.getMemIntrinsicNode(Opcode, DL, VTList, Ops,
5277 NarrowVT, MMO);
5278
5279 // Rotate the result of the final CS so that the field is in the lower
5280 // bits of a GR32, then truncate it.
5281 SDValue ResultShift = DAG.getNode(ISD::ADD, DL, WideVT, BitShift,
5282 DAG.getConstant(BitSize, DL, WideVT));
5283 SDValue Result = DAG.getNode(ISD::ROTL, DL, WideVT, AtomicOp, ResultShift);
5284
5285 SDValue RetOps[2] = { Result, AtomicOp.getValue(1) };
5286 return DAG.getMergeValues(RetOps, DL);
5287}
5288
5289// Op is an ATOMIC_LOAD_SUB operation. Lower 8- and 16-bit operations into
5290// ATOMIC_LOADW_SUBs and convert 32- and 64-bit operations into additions.
5291SDValue SystemZTargetLowering::lowerATOMIC_LOAD_SUB(SDValue Op,
5292 SelectionDAG &DAG) const {
5293 auto *Node = cast<AtomicSDNode>(Op.getNode());
5294 EVT MemVT = Node->getMemoryVT();
5295 if (MemVT == MVT::i32 || MemVT == MVT::i64) {
5296 // A full-width operation: negate and use LAA(G).
5297 assert(Op.getValueType() == MemVT && "Mismatched VTs");
5298 assert(Subtarget.hasInterlockedAccess1() &&
5299 "Should have been expanded by AtomicExpand pass.");
5300 SDValue Src2 = Node->getVal();
5301 SDLoc DL(Src2);
5302 SDValue NegSrc2 =
5303 DAG.getNode(ISD::SUB, DL, MemVT, DAG.getConstant(0, DL, MemVT), Src2);
5304 return DAG.getAtomic(ISD::ATOMIC_LOAD_ADD, DL, MemVT,
5305 Node->getChain(), Node->getBasePtr(), NegSrc2,
5306 Node->getMemOperand());
5307 }
5308
5309 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_SUB);
5310}
5311
5312// Lower 8/16/32/64-bit ATOMIC_CMP_SWAP_WITH_SUCCESS node.
5313SDValue SystemZTargetLowering::lowerATOMIC_CMP_SWAP(SDValue Op,
5314 SelectionDAG &DAG) const {
5315 auto *Node = cast<AtomicSDNode>(Op.getNode());
5316 SDValue ChainIn = Node->getOperand(0);
5317 SDValue Addr = Node->getOperand(1);
5318 SDValue CmpVal = Node->getOperand(2);
5319 SDValue SwapVal = Node->getOperand(3);
5320 MachineMemOperand *MMO = Node->getMemOperand();
5321 SDLoc DL(Node);
5322
5323 if (Node->getMemoryVT() == MVT::i128) {
5324 // Use same code to handle both legal and non-legal i128 types.
5326 LowerOperationWrapper(Node, Results, DAG);
5327 return DAG.getMergeValues(Results, DL);
5328 }
5329
5330 // We have native support for 32-bit and 64-bit compare and swap, but we
5331 // still need to expand extracting the "success" result from the CC.
5332 EVT NarrowVT = Node->getMemoryVT();
5333 EVT WideVT = NarrowVT == MVT::i64 ? MVT::i64 : MVT::i32;
5334 if (NarrowVT == WideVT) {
5335 SDVTList Tys = DAG.getVTList(WideVT, MVT::i32, MVT::Other);
5336 SDValue Ops[] = { ChainIn, Addr, CmpVal, SwapVal };
5337 SDValue AtomicOp = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_CMP_SWAP,
5338 DL, Tys, Ops, NarrowVT, MMO);
5339 SDValue Success = emitSETCC(DAG, DL, AtomicOp.getValue(1),
5341
5342 DAG.ReplaceAllUsesOfValueWith(Op.getValue(0), AtomicOp.getValue(0));
5343 DAG.ReplaceAllUsesOfValueWith(Op.getValue(1), Success);
5344 DAG.ReplaceAllUsesOfValueWith(Op.getValue(2), AtomicOp.getValue(2));
5345 return SDValue();
5346 }
5347
5348 // Convert 8-bit and 16-bit compare and swap to a loop, implemented
5349 // via a fullword ATOMIC_CMP_SWAPW operation.
5350 int64_t BitSize = NarrowVT.getSizeInBits();
5351
5352 SDValue AlignedAddr, BitShift, NegBitShift;
5353 getCSAddressAndShifts(Addr, DAG, DL, AlignedAddr, BitShift, NegBitShift);
5354
5355 // Construct the ATOMIC_CMP_SWAPW node.
5356 SDVTList VTList = DAG.getVTList(WideVT, MVT::i32, MVT::Other);
5357 SDValue Ops[] = { ChainIn, AlignedAddr, CmpVal, SwapVal, BitShift,
5358 NegBitShift, DAG.getConstant(BitSize, DL, WideVT) };
5359 SDValue AtomicOp = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_CMP_SWAPW, DL,
5360 VTList, Ops, NarrowVT, MMO);
5361 SDValue Success = emitSETCC(DAG, DL, AtomicOp.getValue(1),
5363
5364 // emitAtomicCmpSwapW() will zero extend the result (original value).
5365 SDValue OrigVal = DAG.getNode(ISD::AssertZext, DL, WideVT, AtomicOp.getValue(0),
5366 DAG.getValueType(NarrowVT));
5367 DAG.ReplaceAllUsesOfValueWith(Op.getValue(0), OrigVal);
5368 DAG.ReplaceAllUsesOfValueWith(Op.getValue(1), Success);
5369 DAG.ReplaceAllUsesOfValueWith(Op.getValue(2), AtomicOp.getValue(2));
5370 return SDValue();
5371}
5372
5374SystemZTargetLowering::getTargetMMOFlags(const Instruction &I) const {
5375 // Because of how we convert atomic_load and atomic_store to normal loads and
5376 // stores in the DAG, we need to ensure that the MMOs are marked volatile
5377 // since DAGCombine hasn't been updated to account for atomic, but non
5378 // volatile loads. (See D57601)
5379 if (auto *SI = dyn_cast<StoreInst>(&I))
5380 if (SI->isAtomic())
5382 if (auto *LI = dyn_cast<LoadInst>(&I))
5383 if (LI->isAtomic())
5385 if (auto *AI = dyn_cast<AtomicRMWInst>(&I))
5386 if (AI->isAtomic())
5388 if (auto *AI = dyn_cast<AtomicCmpXchgInst>(&I))
5389 if (AI->isAtomic())
5392}
5393
5394SDValue SystemZTargetLowering::lowerSTACKSAVE(SDValue Op,
5395 SelectionDAG &DAG) const {
5397 auto *Regs = Subtarget.getSpecialRegisters();
5399 report_fatal_error("Variable-sized stack allocations are not supported "
5400 "in GHC calling convention");
5401 return DAG.getCopyFromReg(Op.getOperand(0), SDLoc(Op),
5402 Regs->getStackPointerRegister(), Op.getValueType());
5403}
5404
5405SDValue SystemZTargetLowering::lowerSTACKRESTORE(SDValue Op,
5406 SelectionDAG &DAG) const {
5408 auto *Regs = Subtarget.getSpecialRegisters();
5409 bool StoreBackchain = MF.getSubtarget<SystemZSubtarget>().hasBackChain();
5410
5412 report_fatal_error("Variable-sized stack allocations are not supported "
5413 "in GHC calling convention");
5414
5415 SDValue Chain = Op.getOperand(0);
5416 SDValue NewSP = Op.getOperand(1);
5417 SDValue Backchain;
5418 SDLoc DL(Op);
5419
5420 if (StoreBackchain) {
5421 SDValue OldSP = DAG.getCopyFromReg(
5422 Chain, DL, Regs->getStackPointerRegister(), MVT::i64);
5423 Backchain = DAG.getLoad(MVT::i64, DL, Chain, getBackchainAddress(OldSP, DAG),
5424 MachinePointerInfo());
5425 }
5426
5427 Chain = DAG.getCopyToReg(Chain, DL, Regs->getStackPointerRegister(), NewSP);
5428
5429 if (StoreBackchain)
5430 Chain = DAG.getStore(Chain, DL, Backchain, getBackchainAddress(NewSP, DAG),
5431 MachinePointerInfo());
5432
5433 return Chain;
5434}
5435
5436SDValue SystemZTargetLowering::lowerPREFETCH(SDValue Op,
5437 SelectionDAG &DAG) const {
5438 bool IsData = Op.getConstantOperandVal(4);
5439 if (!IsData)
5440 // Just preserve the chain.
5441 return Op.getOperand(0);
5442
5443 SDLoc DL(Op);
5444 bool IsWrite = Op.getConstantOperandVal(2);
5445 unsigned Code = IsWrite ? SystemZ::PFD_WRITE : SystemZ::PFD_READ;
5446 auto *Node = cast<MemIntrinsicSDNode>(Op.getNode());
5447 SDValue Ops[] = {Op.getOperand(0), DAG.getTargetConstant(Code, DL, MVT::i32),
5448 Op.getOperand(1)};
5449 return DAG.getMemIntrinsicNode(SystemZISD::PREFETCH, DL,
5450 Node->getVTList(), Ops,
5451 Node->getMemoryVT(), Node->getMemOperand());
5452}
5453
5454SDValue
5455SystemZTargetLowering::lowerINTRINSIC_W_CHAIN(SDValue Op,
5456 SelectionDAG &DAG) const {
5457 unsigned Opcode, CCValid;
5458 if (isIntrinsicWithCCAndChain(Op, Opcode, CCValid)) {
5459 assert(Op->getNumValues() == 2 && "Expected only CC result and chain");
5460 SDNode *Node = emitIntrinsicWithCCAndChain(DAG, Op, Opcode);
5461 SDValue CC = getCCResult(DAG, SDValue(Node, 0));
5462 DAG.ReplaceAllUsesOfValueWith(SDValue(Op.getNode(), 0), CC);
5463 return SDValue();
5464 }
5465
5466 return SDValue();
5467}
5468
5469SDValue
5470SystemZTargetLowering::lowerINTRINSIC_WO_CHAIN(SDValue Op,
5471 SelectionDAG &DAG) const {
5472 unsigned Opcode, CCValid;
5473 if (isIntrinsicWithCC(Op, Opcode, CCValid)) {
5474 SDNode *Node = emitIntrinsicWithCC(DAG, Op, Opcode);
5475 if (Op->getNumValues() == 1)
5476 return getCCResult(DAG, SDValue(Node, 0));
5477 assert(Op->getNumValues() == 2 && "Expected a CC and non-CC result");
5478 return DAG.getNode(ISD::MERGE_VALUES, SDLoc(Op), Op->getVTList(),
5479 SDValue(Node, 0), getCCResult(DAG, SDValue(Node, 1)));
5480 }
5481
5482 unsigned Id = Op.getConstantOperandVal(0);
5483 switch (Id) {
5484 case Intrinsic::thread_pointer:
5485 return lowerThreadPointer(SDLoc(Op), DAG);
5486
5487 case Intrinsic::s390_vpdi:
5488 return DAG.getNode(SystemZISD::PERMUTE_DWORDS, SDLoc(Op), Op.getValueType(),
5489 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5490
5491 case Intrinsic::s390_vperm:
5492 return DAG.getNode(SystemZISD::PERMUTE, SDLoc(Op), Op.getValueType(),
5493 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5494
5495 case Intrinsic::s390_vuphb:
5496 case Intrinsic::s390_vuphh:
5497 case Intrinsic::s390_vuphf:
5498 case Intrinsic::s390_vuphg:
5499 return DAG.getNode(SystemZISD::UNPACK_HIGH, SDLoc(Op), Op.getValueType(),
5500 Op.getOperand(1));
5501
5502 case Intrinsic::s390_vuplhb:
5503 case Intrinsic::s390_vuplhh:
5504 case Intrinsic::s390_vuplhf:
5505 case Intrinsic::s390_vuplhg:
5506 return DAG.getNode(SystemZISD::UNPACKL_HIGH, SDLoc(Op), Op.getValueType(),
5507 Op.getOperand(1));
5508
5509 case Intrinsic::s390_vuplb:
5510 case Intrinsic::s390_vuplhw:
5511 case Intrinsic::s390_vuplf:
5512 case Intrinsic::s390_vuplg:
5513 return DAG.getNode(SystemZISD::UNPACK_LOW, SDLoc(Op), Op.getValueType(),
5514 Op.getOperand(1));
5515
5516 case Intrinsic::s390_vupllb:
5517 case Intrinsic::s390_vupllh:
5518 case Intrinsic::s390_vupllf:
5519 case Intrinsic::s390_vupllg:
5520 return DAG.getNode(SystemZISD::UNPACKL_LOW, SDLoc(Op), Op.getValueType(),
5521 Op.getOperand(1));
5522
5523 case Intrinsic::s390_vsumb:
5524 case Intrinsic::s390_vsumh:
5525 case Intrinsic::s390_vsumgh:
5526 case Intrinsic::s390_vsumgf:
5527 case Intrinsic::s390_vsumqf:
5528 case Intrinsic::s390_vsumqg:
5529 return DAG.getNode(SystemZISD::VSUM, SDLoc(Op), Op.getValueType(),
5530 Op.getOperand(1), Op.getOperand(2));
5531
5532 case Intrinsic::s390_vaq:
5533 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5534 Op.getOperand(1), Op.getOperand(2));
5535 case Intrinsic::s390_vaccb:
5536 case Intrinsic::s390_vacch:
5537 case Intrinsic::s390_vaccf:
5538 case Intrinsic::s390_vaccg:
5539 case Intrinsic::s390_vaccq:
5540 return DAG.getNode(SystemZISD::VACC, SDLoc(Op), Op.getValueType(),
5541 Op.getOperand(1), Op.getOperand(2));
5542 case Intrinsic::s390_vacq:
5543 return DAG.getNode(SystemZISD::VAC, SDLoc(Op), Op.getValueType(),
5544 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5545 case Intrinsic::s390_vacccq:
5546 return DAG.getNode(SystemZISD::VACCC, SDLoc(Op), Op.getValueType(),
5547 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5548
5549 case Intrinsic::s390_vsq:
5550 return DAG.getNode(ISD::SUB, SDLoc(Op), Op.getValueType(),
5551 Op.getOperand(1), Op.getOperand(2));
5552 case Intrinsic::s390_vscbib:
5553 case Intrinsic::s390_vscbih:
5554 case Intrinsic::s390_vscbif:
5555 case Intrinsic::s390_vscbig:
5556 case Intrinsic::s390_vscbiq:
5557 return DAG.getNode(SystemZISD::VSCBI, SDLoc(Op), Op.getValueType(),
5558 Op.getOperand(1), Op.getOperand(2));
5559 case Intrinsic::s390_vsbiq:
5560 return DAG.getNode(SystemZISD::VSBI, SDLoc(Op), Op.getValueType(),
5561 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5562 case Intrinsic::s390_vsbcbiq:
5563 return DAG.getNode(SystemZISD::VSBCBI, SDLoc(Op), Op.getValueType(),
5564 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5565
5566 case Intrinsic::s390_vmhb:
5567 case Intrinsic::s390_vmhh:
5568 case Intrinsic::s390_vmhf:
5569 case Intrinsic::s390_vmhg:
5570 case Intrinsic::s390_vmhq:
5571 return DAG.getNode(ISD::MULHS, SDLoc(Op), Op.getValueType(),
5572 Op.getOperand(1), Op.getOperand(2));
5573 case Intrinsic::s390_vmlhb:
5574 case Intrinsic::s390_vmlhh:
5575 case Intrinsic::s390_vmlhf:
5576 case Intrinsic::s390_vmlhg:
5577 case Intrinsic::s390_vmlhq:
5578 return DAG.getNode(ISD::MULHU, SDLoc(Op), Op.getValueType(),
5579 Op.getOperand(1), Op.getOperand(2));
5580
5581 case Intrinsic::s390_vmahb:
5582 case Intrinsic::s390_vmahh:
5583 case Intrinsic::s390_vmahf:
5584 case Intrinsic::s390_vmahg:
5585 case Intrinsic::s390_vmahq:
5586 return DAG.getNode(SystemZISD::VMAH, SDLoc(Op), Op.getValueType(),
5587 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5588 case Intrinsic::s390_vmalhb:
5589 case Intrinsic::s390_vmalhh:
5590 case Intrinsic::s390_vmalhf:
5591 case Intrinsic::s390_vmalhg:
5592 case Intrinsic::s390_vmalhq:
5593 return DAG.getNode(SystemZISD::VMALH, SDLoc(Op), Op.getValueType(),
5594 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5595
5596 case Intrinsic::s390_vmeb:
5597 case Intrinsic::s390_vmeh:
5598 case Intrinsic::s390_vmef:
5599 case Intrinsic::s390_vmeg:
5600 return DAG.getNode(SystemZISD::VME, SDLoc(Op), Op.getValueType(),
5601 Op.getOperand(1), Op.getOperand(2));
5602 case Intrinsic::s390_vmleb:
5603 case Intrinsic::s390_vmleh:
5604 case Intrinsic::s390_vmlef:
5605 case Intrinsic::s390_vmleg:
5606 return DAG.getNode(SystemZISD::VMLE, SDLoc(Op), Op.getValueType(),
5607 Op.getOperand(1), Op.getOperand(2));
5608 case Intrinsic::s390_vmob:
5609 case Intrinsic::s390_vmoh:
5610 case Intrinsic::s390_vmof:
5611 case Intrinsic::s390_vmog:
5612 return DAG.getNode(SystemZISD::VMO, SDLoc(Op), Op.getValueType(),
5613 Op.getOperand(1), Op.getOperand(2));
5614 case Intrinsic::s390_vmlob:
5615 case Intrinsic::s390_vmloh:
5616 case Intrinsic::s390_vmlof:
5617 case Intrinsic::s390_vmlog:
5618 return DAG.getNode(SystemZISD::VMLO, SDLoc(Op), Op.getValueType(),
5619 Op.getOperand(1), Op.getOperand(2));
5620
5621 case Intrinsic::s390_vmaeb:
5622 case Intrinsic::s390_vmaeh:
5623 case Intrinsic::s390_vmaef:
5624 case Intrinsic::s390_vmaeg:
5625 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5626 DAG.getNode(SystemZISD::VME, SDLoc(Op), Op.getValueType(),
5627 Op.getOperand(1), Op.getOperand(2)),
5628 Op.getOperand(3));
5629 case Intrinsic::s390_vmaleb:
5630 case Intrinsic::s390_vmaleh:
5631 case Intrinsic::s390_vmalef:
5632 case Intrinsic::s390_vmaleg:
5633 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5634 DAG.getNode(SystemZISD::VMLE, SDLoc(Op), Op.getValueType(),
5635 Op.getOperand(1), Op.getOperand(2)),
5636 Op.getOperand(3));
5637 case Intrinsic::s390_vmaob:
5638 case Intrinsic::s390_vmaoh:
5639 case Intrinsic::s390_vmaof:
5640 case Intrinsic::s390_vmaog:
5641 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5642 DAG.getNode(SystemZISD::VMO, SDLoc(Op), Op.getValueType(),
5643 Op.getOperand(1), Op.getOperand(2)),
5644 Op.getOperand(3));
5645 case Intrinsic::s390_vmalob:
5646 case Intrinsic::s390_vmaloh:
5647 case Intrinsic::s390_vmalof:
5648 case Intrinsic::s390_vmalog:
5649 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5650 DAG.getNode(SystemZISD::VMLO, SDLoc(Op), Op.getValueType(),
5651 Op.getOperand(1), Op.getOperand(2)),
5652 Op.getOperand(3));
5653 }
5654
5655 return SDValue();
5656}
5657
5658namespace {
5659// Says that SystemZISD operation Opcode can be used to perform the equivalent
5660// of a VPERM with permute vector Bytes. If Opcode takes three operands,
5661// Operand is the constant third operand, otherwise it is the number of
5662// bytes in each element of the result.
5663struct Permute {
5664 unsigned Opcode;
5665 unsigned Operand;
5666 unsigned char Bytes[SystemZ::VectorBytes];
5667};
5668}
5669
5670static const Permute PermuteForms[] = {
5671 // VMRHG
5672 { SystemZISD::MERGE_HIGH, 8,
5673 { 0, 1, 2, 3, 4, 5, 6, 7, 16, 17, 18, 19, 20, 21, 22, 23 } },
5674 // VMRHF
5675 { SystemZISD::MERGE_HIGH, 4,
5676 { 0, 1, 2, 3, 16, 17, 18, 19, 4, 5, 6, 7, 20, 21, 22, 23 } },
5677 // VMRHH
5678 { SystemZISD::MERGE_HIGH, 2,
5679 { 0, 1, 16, 17, 2, 3, 18, 19, 4, 5, 20, 21, 6, 7, 22, 23 } },
5680 // VMRHB
5681 { SystemZISD::MERGE_HIGH, 1,
5682 { 0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23 } },
5683 // VMRLG
5684 { SystemZISD::MERGE_LOW, 8,
5685 { 8, 9, 10, 11, 12, 13, 14, 15, 24, 25, 26, 27, 28, 29, 30, 31 } },
5686 // VMRLF
5687 { SystemZISD::MERGE_LOW, 4,
5688 { 8, 9, 10, 11, 24, 25, 26, 27, 12, 13, 14, 15, 28, 29, 30, 31 } },
5689 // VMRLH
5690 { SystemZISD::MERGE_LOW, 2,
5691 { 8, 9, 24, 25, 10, 11, 26, 27, 12, 13, 28, 29, 14, 15, 30, 31 } },
5692 // VMRLB
5693 { SystemZISD::MERGE_LOW, 1,
5694 { 8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31 } },
5695 // VPKG
5696 { SystemZISD::PACK, 4,
5697 { 4, 5, 6, 7, 12, 13, 14, 15, 20, 21, 22, 23, 28, 29, 30, 31 } },
5698 // VPKF
5699 { SystemZISD::PACK, 2,
5700 { 2, 3, 6, 7, 10, 11, 14, 15, 18, 19, 22, 23, 26, 27, 30, 31 } },
5701 // VPKH
5702 { SystemZISD::PACK, 1,
5703 { 1, 3, 5, 7, 9, 11, 13, 15, 17, 19, 21, 23, 25, 27, 29, 31 } },
5704 // VPDI V1, V2, 4 (low half of V1, high half of V2)
5705 { SystemZISD::PERMUTE_DWORDS, 4,
5706 { 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23 } },
5707 // VPDI V1, V2, 1 (high half of V1, low half of V2)
5708 { SystemZISD::PERMUTE_DWORDS, 1,
5709 { 0, 1, 2, 3, 4, 5, 6, 7, 24, 25, 26, 27, 28, 29, 30, 31 } }
5710};
5711
5712// Called after matching a vector shuffle against a particular pattern.
5713// Both the original shuffle and the pattern have two vector operands.
5714// OpNos[0] is the operand of the original shuffle that should be used for
5715// operand 0 of the pattern, or -1 if operand 0 of the pattern can be anything.
5716// OpNos[1] is the same for operand 1 of the pattern. Resolve these -1s and
5717// set OpNo0 and OpNo1 to the shuffle operands that should actually be used
5718// for operands 0 and 1 of the pattern.
5719static bool chooseShuffleOpNos(int *OpNos, unsigned &OpNo0, unsigned &OpNo1) {
5720 if (OpNos[0] < 0) {
5721 if (OpNos[1] < 0)
5722 return false;
5723 OpNo0 = OpNo1 = OpNos[1];
5724 } else if (OpNos[1] < 0) {
5725 OpNo0 = OpNo1 = OpNos[0];
5726 } else {
5727 OpNo0 = OpNos[0];
5728 OpNo1 = OpNos[1];
5729 }
5730 return true;
5731}
5732
5733// Bytes is a VPERM-like permute vector, except that -1 is used for
5734// undefined bytes. Return true if the VPERM can be implemented using P.
5735// When returning true set OpNo0 to the VPERM operand that should be
5736// used for operand 0 of P and likewise OpNo1 for operand 1 of P.
5737//
5738// For example, if swapping the VPERM operands allows P to match, OpNo0
5739// will be 1 and OpNo1 will be 0. If instead Bytes only refers to one
5740// operand, but rewriting it to use two duplicated operands allows it to
5741// match P, then OpNo0 and OpNo1 will be the same.
5742static bool matchPermute(const SmallVectorImpl<int> &Bytes, const Permute &P,
5743 unsigned &OpNo0, unsigned &OpNo1) {
5744 int OpNos[] = { -1, -1 };
5745 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I) {
5746 int Elt = Bytes[I];
5747 if (Elt >= 0) {
5748 // Make sure that the two permute vectors use the same suboperand
5749 // byte number. Only the operand numbers (the high bits) are
5750 // allowed to differ.
5751 if ((Elt ^ P.Bytes[I]) & (SystemZ::VectorBytes - 1))
5752 return false;
5753 int ModelOpNo = P.Bytes[I] / SystemZ::VectorBytes;
5754 int RealOpNo = unsigned(Elt) / SystemZ::VectorBytes;
5755 // Make sure that the operand mappings are consistent with previous
5756 // elements.
5757 if (OpNos[ModelOpNo] == 1 - RealOpNo)
5758 return false;
5759 OpNos[ModelOpNo] = RealOpNo;
5760 }
5761 }
5762 return chooseShuffleOpNos(OpNos, OpNo0, OpNo1);
5763}
5764
5765// As above, but search for a matching permute.
5766static const Permute *matchPermute(const SmallVectorImpl<int> &Bytes,
5767 unsigned &OpNo0, unsigned &OpNo1) {
5768 for (auto &P : PermuteForms)
5769 if (matchPermute(Bytes, P, OpNo0, OpNo1))
5770 return &P;
5771 return nullptr;
5772}
5773
5774// Bytes is a VPERM-like permute vector, except that -1 is used for
5775// undefined bytes. This permute is an operand of an outer permute.
5776// See whether redistributing the -1 bytes gives a shuffle that can be
5777// implemented using P. If so, set Transform to a VPERM-like permute vector
5778// that, when applied to the result of P, gives the original permute in Bytes.
5780 const Permute &P,
5781 SmallVectorImpl<int> &Transform) {
5782 unsigned To = 0;
5783 for (unsigned From = 0; From < SystemZ::VectorBytes; ++From) {
5784 int Elt = Bytes[From];
5785 if (Elt < 0)
5786 // Byte number From of the result is undefined.
5787 Transform[From] = -1;
5788 else {
5789 while (P.Bytes[To] != Elt) {
5790 To += 1;
5791 if (To == SystemZ::VectorBytes)
5792 return false;
5793 }
5794 Transform[From] = To;
5795 }
5796 }
5797 return true;
5798}
5799
5800// As above, but search for a matching permute.
5801static const Permute *matchDoublePermute(const SmallVectorImpl<int> &Bytes,
5802 SmallVectorImpl<int> &Transform) {
5803 for (auto &P : PermuteForms)
5804 if (matchDoublePermute(Bytes, P, Transform))
5805 return &P;
5806 return nullptr;
5807}
5808
5809// Convert the mask of the given shuffle op into a byte-level mask,
5810// as if it had type vNi8.
5811static bool getVPermMask(SDValue ShuffleOp,
5812 SmallVectorImpl<int> &Bytes) {
5813 EVT VT = ShuffleOp.getValueType();
5814 unsigned NumElements = VT.getVectorNumElements();
5815 unsigned BytesPerElement = VT.getVectorElementType().getStoreSize();
5816
5817 if (auto *VSN = dyn_cast<ShuffleVectorSDNode>(ShuffleOp)) {
5818 Bytes.resize(NumElements * BytesPerElement, -1);
5819 for (unsigned I = 0; I < NumElements; ++I) {
5820 int Index = VSN->getMaskElt(I);
5821 if (Index >= 0)
5822 for (unsigned J = 0; J < BytesPerElement; ++J)
5823 Bytes[I * BytesPerElement + J] = Index * BytesPerElement + J;
5824 }
5825 return true;
5826 }
5827 if (SystemZISD::SPLAT == ShuffleOp.getOpcode() &&
5828 isa<ConstantSDNode>(ShuffleOp.getOperand(1))) {
5829 unsigned Index = ShuffleOp.getConstantOperandVal(1);
5830 Bytes.resize(NumElements * BytesPerElement, -1);
5831 for (unsigned I = 0; I < NumElements; ++I)
5832 for (unsigned J = 0; J < BytesPerElement; ++J)
5833 Bytes[I * BytesPerElement + J] = Index * BytesPerElement + J;
5834 return true;
5835 }
5836 return false;
5837}
5838
5839// Bytes is a VPERM-like permute vector, except that -1 is used for
5840// undefined bytes. See whether bytes [Start, Start + BytesPerElement) of
5841// the result come from a contiguous sequence of bytes from one input.
5842// Set Base to the selector for the first byte if so.
5843static bool getShuffleInput(const SmallVectorImpl<int> &Bytes, unsigned Start,
5844 unsigned BytesPerElement, int &Base) {
5845 Base = -1;
5846 for (unsigned I = 0; I < BytesPerElement; ++I) {
5847 if (Bytes[Start + I] >= 0) {
5848 unsigned Elem = Bytes[Start + I];
5849 if (Base < 0) {
5850 Base = Elem - I;
5851 // Make sure the bytes would come from one input operand.
5852 if (unsigned(Base) % Bytes.size() + BytesPerElement > Bytes.size())
5853 return false;
5854 } else if (unsigned(Base) != Elem - I)
5855 return false;
5856 }
5857 }
5858 return true;
5859}
5860
5861// Bytes is a VPERM-like permute vector, except that -1 is used for
5862// undefined bytes. Return true if it can be performed using VSLDB.
5863// When returning true, set StartIndex to the shift amount and OpNo0
5864// and OpNo1 to the VPERM operands that should be used as the first
5865// and second shift operand respectively.
5867 unsigned &StartIndex, unsigned &OpNo0,
5868 unsigned &OpNo1) {
5869 int OpNos[] = { -1, -1 };
5870 int Shift = -1;
5871 for (unsigned I = 0; I < 16; ++I) {
5872 int Index = Bytes[I];
5873 if (Index >= 0) {
5874 int ExpectedShift = (Index - I) % SystemZ::VectorBytes;
5875 int ModelOpNo = unsigned(ExpectedShift + I) / SystemZ::VectorBytes;
5876 int RealOpNo = unsigned(Index) / SystemZ::VectorBytes;
5877 if (Shift < 0)
5878 Shift = ExpectedShift;
5879 else if (Shift != ExpectedShift)
5880 return false;
5881 // Make sure that the operand mappings are consistent with previous
5882 // elements.
5883 if (OpNos[ModelOpNo] == 1 - RealOpNo)
5884 return false;
5885 OpNos[ModelOpNo] = RealOpNo;
5886 }
5887 }
5888 StartIndex = Shift;
5889 return chooseShuffleOpNos(OpNos, OpNo0, OpNo1);
5890}
5891
5892// Create a node that performs P on operands Op0 and Op1, casting the
5893// operands to the appropriate type. The type of the result is determined by P.
5895 const Permute &P, SDValue Op0, SDValue Op1) {
5896 // VPDI (PERMUTE_DWORDS) always operates on v2i64s. The input
5897 // elements of a PACK are twice as wide as the outputs.
5898 unsigned InBytes = (P.Opcode == SystemZISD::PERMUTE_DWORDS ? 8 :
5899 P.Opcode == SystemZISD::PACK ? P.Operand * 2 :
5900 P.Operand);
5901 // Cast both operands to the appropriate type.
5902 MVT InVT = MVT::getVectorVT(MVT::getIntegerVT(InBytes * 8),
5903 SystemZ::VectorBytes / InBytes);
5904 Op0 = DAG.getNode(ISD::BITCAST, DL, InVT, Op0);
5905 Op1 = DAG.getNode(ISD::BITCAST, DL, InVT, Op1);
5906 SDValue Op;
5907 if (P.Opcode == SystemZISD::PERMUTE_DWORDS) {
5908 SDValue Op2 = DAG.getTargetConstant(P.Operand, DL, MVT::i32);
5909 Op = DAG.getNode(SystemZISD::PERMUTE_DWORDS, DL, InVT, Op0, Op1, Op2);
5910 } else if (P.Opcode == SystemZISD::PACK) {
5911 MVT OutVT = MVT::getVectorVT(MVT::getIntegerVT(P.Operand * 8),
5912 SystemZ::VectorBytes / P.Operand);
5913 Op = DAG.getNode(SystemZISD::PACK, DL, OutVT, Op0, Op1);
5914 } else {
5915 Op = DAG.getNode(P.Opcode, DL, InVT, Op0, Op1);
5916 }
5917 return Op;
5918}
5919
5920static bool isZeroVector(SDValue N) {
5921 if (N->getOpcode() == ISD::BITCAST)
5922 N = N->getOperand(0);
5923 if (N->getOpcode() == ISD::SPLAT_VECTOR)
5924 if (auto *Op = dyn_cast<ConstantSDNode>(N->getOperand(0)))
5925 return Op->getZExtValue() == 0;
5926 return ISD::isBuildVectorAllZeros(N.getNode());
5927}
5928
5929// Return the index of the zero/undef vector, or UINT32_MAX if not found.
5930static uint32_t findZeroVectorIdx(SDValue *Ops, unsigned Num) {
5931 for (unsigned I = 0; I < Num ; I++)
5932 if (isZeroVector(Ops[I]))
5933 return I;
5934 return UINT32_MAX;
5935}
5936
5937// Bytes is a VPERM-like permute vector, except that -1 is used for
5938// undefined bytes. Implement it on operands Ops[0] and Ops[1] using
5939// VSLDB or VPERM.
5941 SDValue *Ops,
5942 const SmallVectorImpl<int> &Bytes) {
5943 for (unsigned I = 0; I < 2; ++I)
5944 Ops[I] = DAG.getNode(ISD::BITCAST, DL, MVT::v16i8, Ops[I]);
5945
5946 // First see whether VSLDB can be used.
5947 unsigned StartIndex, OpNo0, OpNo1;
5948 if (isShlDoublePermute(Bytes, StartIndex, OpNo0, OpNo1))
5949 return DAG.getNode(SystemZISD::SHL_DOUBLE, DL, MVT::v16i8, Ops[OpNo0],
5950 Ops[OpNo1],
5951 DAG.getTargetConstant(StartIndex, DL, MVT::i32));
5952
5953 // Fall back on VPERM. Construct an SDNode for the permute vector. Try to
5954 // eliminate a zero vector by reusing any zero index in the permute vector.
5955 unsigned ZeroVecIdx = findZeroVectorIdx(&Ops[0], 2);
5956 if (ZeroVecIdx != UINT32_MAX) {
5957 bool MaskFirst = true;
5958 int ZeroIdx = -1;
5959 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I) {
5960 unsigned OpNo = unsigned(Bytes[I]) / SystemZ::VectorBytes;
5961 unsigned Byte = unsigned(Bytes[I]) % SystemZ::VectorBytes;
5962 if (OpNo == ZeroVecIdx && I == 0) {
5963 // If the first byte is zero, use mask as first operand.
5964 ZeroIdx = 0;
5965 break;
5966 }
5967 if (OpNo != ZeroVecIdx && Byte == 0) {
5968 // If mask contains a zero, use it by placing that vector first.
5969 ZeroIdx = I + SystemZ::VectorBytes;
5970 MaskFirst = false;
5971 break;
5972 }
5973 }
5974 if (ZeroIdx != -1) {
5975 SDValue IndexNodes[SystemZ::VectorBytes];
5976 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I) {
5977 if (Bytes[I] >= 0) {
5978 unsigned OpNo = unsigned(Bytes[I]) / SystemZ::VectorBytes;
5979 unsigned Byte = unsigned(Bytes[I]) % SystemZ::VectorBytes;
5980 if (OpNo == ZeroVecIdx)
5981 IndexNodes[I] = DAG.getConstant(ZeroIdx, DL, MVT::i32);
5982 else {
5983 unsigned BIdx = MaskFirst ? Byte + SystemZ::VectorBytes : Byte;
5984 IndexNodes[I] = DAG.getConstant(BIdx, DL, MVT::i32);
5985 }
5986 } else
5987 IndexNodes[I] = DAG.getUNDEF(MVT::i32);
5988 }
5989 SDValue Mask = DAG.getBuildVector(MVT::v16i8, DL, IndexNodes);
5990 SDValue Src = ZeroVecIdx == 0 ? Ops[1] : Ops[0];
5991 if (MaskFirst)
5992 return DAG.getNode(SystemZISD::PERMUTE, DL, MVT::v16i8, Mask, Src,
5993 Mask);
5994 else
5995 return DAG.getNode(SystemZISD::PERMUTE, DL, MVT::v16i8, Src, Mask,
5996 Mask);
5997 }
5998 }
5999
6000 SDValue IndexNodes[SystemZ::VectorBytes];
6001 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I)
6002 if (Bytes[I] >= 0)
6003 IndexNodes[I] = DAG.getConstant(Bytes[I], DL, MVT::i32);
6004 else
6005 IndexNodes[I] = DAG.getUNDEF(MVT::i32);
6006 SDValue Op2 = DAG.getBuildVector(MVT::v16i8, DL, IndexNodes);
6007 return DAG.getNode(SystemZISD::PERMUTE, DL, MVT::v16i8, Ops[0],
6008 (!Ops[1].isUndef() ? Ops[1] : Ops[0]), Op2);
6009}
6010
6011namespace {
6012// Describes a general N-operand vector shuffle.
6013struct GeneralShuffle {
6014 GeneralShuffle(EVT vt)
6015 : VT(vt), UnpackFromEltSize(UINT_MAX), UnpackLow(false) {}
6016 void addUndef();
6017 bool add(SDValue, unsigned);
6018 SDValue getNode(SelectionDAG &, const SDLoc &);
6019 void tryPrepareForUnpack();
6020 bool unpackWasPrepared() { return UnpackFromEltSize <= 4; }
6021 SDValue insertUnpackIfPrepared(SelectionDAG &DAG, const SDLoc &DL, SDValue Op);
6022
6023 // The operands of the shuffle.
6025
6026 // Index I is -1 if byte I of the result is undefined. Otherwise the
6027 // result comes from byte Bytes[I] % SystemZ::VectorBytes of operand
6028 // Bytes[I] / SystemZ::VectorBytes.
6030
6031 // The type of the shuffle result.
6032 EVT VT;
6033
6034 // Holds a value of 1, 2 or 4 if a final unpack has been prepared for.
6035 unsigned UnpackFromEltSize;
6036 // True if the final unpack uses the low half.
6037 bool UnpackLow;
6038};
6039} // namespace
6040
6041// Add an extra undefined element to the shuffle.
6042void GeneralShuffle::addUndef() {
6043 unsigned BytesPerElement = VT.getVectorElementType().getStoreSize();
6044 for (unsigned I = 0; I < BytesPerElement; ++I)
6045 Bytes.push_back(-1);
6046}
6047
6048// Add an extra element to the shuffle, taking it from element Elem of Op.
6049// A null Op indicates a vector input whose value will be calculated later;
6050// there is at most one such input per shuffle and it always has the same
6051// type as the result. Aborts and returns false if the source vector elements
6052// of an EXTRACT_VECTOR_ELT are smaller than the destination elements. Per
6053// LLVM they become implicitly extended, but this is rare and not optimized.
6054bool GeneralShuffle::add(SDValue Op, unsigned Elem) {
6055 unsigned BytesPerElement = VT.getVectorElementType().getStoreSize();
6056
6057 // The source vector can have wider elements than the result,
6058 // either through an explicit TRUNCATE or because of type legalization.
6059 // We want the least significant part.
6060 EVT FromVT = Op.getNode() ? Op.getValueType() : VT;
6061 unsigned FromBytesPerElement = FromVT.getVectorElementType().getStoreSize();
6062
6063 // Return false if the source elements are smaller than their destination
6064 // elements.
6065 if (FromBytesPerElement < BytesPerElement)
6066 return false;
6067
6068 unsigned Byte = ((Elem * FromBytesPerElement) % SystemZ::VectorBytes +
6069 (FromBytesPerElement - BytesPerElement));
6070
6071 // Look through things like shuffles and bitcasts.
6072 while (Op.getNode()) {
6073 if (Op.getOpcode() == ISD::BITCAST)
6074 Op = Op.getOperand(0);
6075 else if (Op.getOpcode() == ISD::VECTOR_SHUFFLE && Op.hasOneUse()) {
6076 // See whether the bytes we need come from a contiguous part of one
6077 // operand.
6079 if (!getVPermMask(Op, OpBytes))
6080 break;
6081 int NewByte;
6082 if (!getShuffleInput(OpBytes, Byte, BytesPerElement, NewByte))
6083 break;
6084 if (NewByte < 0) {
6085 addUndef();
6086 return true;
6087 }
6088 Op = Op.getOperand(unsigned(NewByte) / SystemZ::VectorBytes);
6089 Byte = unsigned(NewByte) % SystemZ::VectorBytes;
6090 } else if (Op.isUndef()) {
6091 addUndef();
6092 return true;
6093 } else
6094 break;
6095 }
6096
6097 // Make sure that the source of the extraction is in Ops.
6098 unsigned OpNo = 0;
6099 for (; OpNo < Ops.size(); ++OpNo)
6100 if (Ops[OpNo] == Op)
6101 break;
6102 if (OpNo == Ops.size())
6103 Ops.push_back(Op);
6104
6105 // Add the element to Bytes.
6106 unsigned Base = OpNo * SystemZ::VectorBytes + Byte;
6107 for (unsigned I = 0; I < BytesPerElement; ++I)
6108 Bytes.push_back(Base + I);
6109
6110 return true;
6111}
6112
6113// Return SDNodes for the completed shuffle.
6114SDValue GeneralShuffle::getNode(SelectionDAG &DAG, const SDLoc &DL) {
6115 assert(Bytes.size() == SystemZ::VectorBytes && "Incomplete vector");
6116
6117 if (Ops.size() == 0)
6118 return DAG.getUNDEF(VT);
6119
6120 // Use a single unpack if possible as the last operation.
6121 tryPrepareForUnpack();
6122
6123 // Make sure that there are at least two shuffle operands.
6124 if (Ops.size() == 1)
6125 Ops.push_back(DAG.getUNDEF(MVT::v16i8));
6126
6127 // Create a tree of shuffles, deferring root node until after the loop.
6128 // Try to redistribute the undefined elements of non-root nodes so that
6129 // the non-root shuffles match something like a pack or merge, then adjust
6130 // the parent node's permute vector to compensate for the new order.
6131 // Among other things, this copes with vectors like <2 x i16> that were
6132 // padded with undefined elements during type legalization.
6133 //
6134 // In the best case this redistribution will lead to the whole tree
6135 // using packs and merges. It should rarely be a loss in other cases.
6136 unsigned Stride = 1;
6137 for (; Stride * 2 < Ops.size(); Stride *= 2) {
6138 for (unsigned I = 0; I < Ops.size() - Stride; I += Stride * 2) {
6139 SDValue SubOps[] = { Ops[I], Ops[I + Stride] };
6140
6141 // Create a mask for just these two operands.
6143 for (unsigned J = 0; J < SystemZ::VectorBytes; ++J) {
6144 unsigned OpNo = unsigned(Bytes[J]) / SystemZ::VectorBytes;
6145 unsigned Byte = unsigned(Bytes[J]) % SystemZ::VectorBytes;
6146 if (OpNo == I)
6147 NewBytes[J] = Byte;
6148 else if (OpNo == I + Stride)
6149 NewBytes[J] = SystemZ::VectorBytes + Byte;
6150 else
6151 NewBytes[J] = -1;
6152 }
6153 // See if it would be better to reorganize NewMask to avoid using VPERM.
6155 if (const Permute *P = matchDoublePermute(NewBytes, NewBytesMap)) {
6156 Ops[I] = getPermuteNode(DAG, DL, *P, SubOps[0], SubOps[1]);
6157 // Applying NewBytesMap to Ops[I] gets back to NewBytes.
6158 for (unsigned J = 0; J < SystemZ::VectorBytes; ++J) {
6159 if (NewBytes[J] >= 0) {
6160 assert(unsigned(NewBytesMap[J]) < SystemZ::VectorBytes &&
6161 "Invalid double permute");
6162 Bytes[J] = I * SystemZ::VectorBytes + NewBytesMap[J];
6163 } else
6164 assert(NewBytesMap[J] < 0 && "Invalid double permute");
6165 }
6166 } else {
6167 // Just use NewBytes on the operands.
6168 Ops[I] = getGeneralPermuteNode(DAG, DL, SubOps, NewBytes);
6169 for (unsigned J = 0; J < SystemZ::VectorBytes; ++J)
6170 if (NewBytes[J] >= 0)
6171 Bytes[J] = I * SystemZ::VectorBytes + J;
6172 }
6173 }
6174 }
6175
6176 // Now we just have 2 inputs. Put the second operand in Ops[1].
6177 if (Stride > 1) {
6178 Ops[1] = Ops[Stride];
6179 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I)
6180 if (Bytes[I] >= int(SystemZ::VectorBytes))
6181 Bytes[I] -= (Stride - 1) * SystemZ::VectorBytes;
6182 }
6183
6184 // Look for an instruction that can do the permute without resorting
6185 // to VPERM.
6186 unsigned OpNo0, OpNo1;
6187 SDValue Op;
6188 if (unpackWasPrepared() && Ops[1].isUndef())
6189 Op = Ops[0];
6190 else if (const Permute *P = matchPermute(Bytes, OpNo0, OpNo1))
6191 Op = getPermuteNode(DAG, DL, *P, Ops[OpNo0], Ops[OpNo1]);
6192 else
6193 Op = getGeneralPermuteNode(DAG, DL, &Ops[0], Bytes);
6194
6195 Op = insertUnpackIfPrepared(DAG, DL, Op);
6196
6197 return DAG.getNode(ISD::BITCAST, DL, VT, Op);
6198}
6199
6200#ifndef NDEBUG
6201static void dumpBytes(const SmallVectorImpl<int> &Bytes, std::string Msg) {
6202 dbgs() << Msg.c_str() << " { ";
6203 for (unsigned I = 0; I < Bytes.size(); I++)
6204 dbgs() << Bytes[I] << " ";
6205 dbgs() << "}\n";
6206}
6207#endif
6208
6209// If the Bytes vector matches an unpack operation, prepare to do the unpack
6210// after all else by removing the zero vector and the effect of the unpack on
6211// Bytes.
6212void GeneralShuffle::tryPrepareForUnpack() {
6213 uint32_t ZeroVecOpNo = findZeroVectorIdx(&Ops[0], Ops.size());
6214 if (ZeroVecOpNo == UINT32_MAX || Ops.size() == 1)
6215 return;
6216
6217 // Only do this if removing the zero vector reduces the depth, otherwise
6218 // the critical path will increase with the final unpack.
6219 if (Ops.size() > 2 &&
6220 Log2_32_Ceil(Ops.size()) == Log2_32_Ceil(Ops.size() - 1))
6221 return;
6222
6223 // Find an unpack that would allow removing the zero vector from Ops.
6224 UnpackFromEltSize = 1;
6225 for (; UnpackFromEltSize <= 4; UnpackFromEltSize *= 2) {
6226 bool MatchUnpack = true;
6228 for (unsigned Elt = 0; Elt < SystemZ::VectorBytes; Elt++) {
6229 unsigned ToEltSize = UnpackFromEltSize * 2;
6230 bool IsZextByte = (Elt % ToEltSize) < UnpackFromEltSize;
6231 if (!IsZextByte)
6232 SrcBytes.push_back(Bytes[Elt]);
6233 if (Bytes[Elt] != -1) {
6234 unsigned OpNo = unsigned(Bytes[Elt]) / SystemZ::VectorBytes;
6235 if (IsZextByte != (OpNo == ZeroVecOpNo)) {
6236 MatchUnpack = false;
6237 break;
6238 }
6239 }
6240 }
6241 if (MatchUnpack) {
6242 if (Ops.size() == 2) {
6243 // Don't use unpack if a single source operand needs rearrangement.
6244 bool CanUseUnpackLow = true, CanUseUnpackHigh = true;
6245 for (unsigned i = 0; i < SystemZ::VectorBytes / 2; i++) {
6246 if (SrcBytes[i] == -1)
6247 continue;
6248 if (SrcBytes[i] % 16 != int(i))
6249 CanUseUnpackHigh = false;
6250 if (SrcBytes[i] % 16 != int(i + SystemZ::VectorBytes / 2))
6251 CanUseUnpackLow = false;
6252 if (!CanUseUnpackLow && !CanUseUnpackHigh) {
6253 UnpackFromEltSize = UINT_MAX;
6254 return;
6255 }
6256 }
6257 if (!CanUseUnpackHigh)
6258 UnpackLow = true;
6259 }
6260 break;
6261 }
6262 }
6263 if (UnpackFromEltSize > 4)
6264 return;
6265
6266 LLVM_DEBUG(dbgs() << "Preparing for final unpack of element size "
6267 << UnpackFromEltSize << ". Zero vector is Op#" << ZeroVecOpNo
6268 << ".\n";
6269 dumpBytes(Bytes, "Original Bytes vector:"););
6270
6271 // Apply the unpack in reverse to the Bytes array.
6272 unsigned B = 0;
6273 if (UnpackLow) {
6274 while (B < SystemZ::VectorBytes / 2)
6275 Bytes[B++] = -1;
6276 }
6277 for (unsigned Elt = 0; Elt < SystemZ::VectorBytes;) {
6278 Elt += UnpackFromEltSize;
6279 for (unsigned i = 0; i < UnpackFromEltSize; i++, Elt++, B++)
6280 Bytes[B] = Bytes[Elt];
6281 }
6282 if (!UnpackLow) {
6283 while (B < SystemZ::VectorBytes)
6284 Bytes[B++] = -1;
6285 }
6286
6287 // Remove the zero vector from Ops
6288 Ops.erase(&Ops[ZeroVecOpNo]);
6289 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I)
6290 if (Bytes[I] >= 0) {
6291 unsigned OpNo = unsigned(Bytes[I]) / SystemZ::VectorBytes;
6292 if (OpNo > ZeroVecOpNo)
6293 Bytes[I] -= SystemZ::VectorBytes;
6294 }
6295
6296 LLVM_DEBUG(dumpBytes(Bytes, "Resulting Bytes vector, zero vector removed:");
6297 dbgs() << "\n";);
6298}
6299
6300SDValue GeneralShuffle::insertUnpackIfPrepared(SelectionDAG &DAG,
6301 const SDLoc &DL,
6302 SDValue Op) {
6303 if (!unpackWasPrepared())
6304 return Op;
6305 unsigned InBits = UnpackFromEltSize * 8;
6306 EVT InVT = MVT::getVectorVT(MVT::getIntegerVT(InBits),
6307 SystemZ::VectorBits / InBits);
6308 SDValue PackedOp = DAG.getNode(ISD::BITCAST, DL, InVT, Op);
6309 unsigned OutBits = InBits * 2;
6310 EVT OutVT = MVT::getVectorVT(MVT::getIntegerVT(OutBits),
6311 SystemZ::VectorBits / OutBits);
6312 return DAG.getNode(UnpackLow ? SystemZISD::UNPACKL_LOW
6313 : SystemZISD::UNPACKL_HIGH,
6314 DL, OutVT, PackedOp);
6315}
6316
6317// Return true if the given BUILD_VECTOR is a scalar-to-vector conversion.
6319 for (unsigned I = 1, E = Op.getNumOperands(); I != E; ++I)
6320 if (!Op.getOperand(I).isUndef())
6321 return false;
6322 return true;
6323}
6324
6325// Return a vector of type VT that contains Value in the first element.
6326// The other elements don't matter.
6328 SDValue Value) {
6329 // If we have a constant, replicate it to all elements and let the
6330 // BUILD_VECTOR lowering take care of it.
6331 if (Value.getOpcode() == ISD::Constant ||
6332 Value.getOpcode() == ISD::ConstantFP) {
6334 return DAG.getBuildVector(VT, DL, Ops);
6335 }
6336 if (Value.isUndef())
6337 return DAG.getUNDEF(VT);
6338 return DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VT, Value);
6339}
6340
6341// Return a vector of type VT in which Op0 is in element 0 and Op1 is in
6342// element 1. Used for cases in which replication is cheap.
6344 SDValue Op0, SDValue Op1) {
6345 if (Op0.isUndef()) {
6346 if (Op1.isUndef())
6347 return DAG.getUNDEF(VT);
6348 return DAG.getNode(SystemZISD::REPLICATE, DL, VT, Op1);
6349 }
6350 if (Op1.isUndef())
6351 return DAG.getNode(SystemZISD::REPLICATE, DL, VT, Op0);
6352 return DAG.getNode(SystemZISD::MERGE_HIGH, DL, VT,
6353 buildScalarToVector(DAG, DL, VT, Op0),
6354 buildScalarToVector(DAG, DL, VT, Op1));
6355}
6356
6357// Extend GPR scalars Op0 and Op1 to doublewords and return a v2i64
6358// vector for them.
6360 SDValue Op1) {
6361 if (Op0.isUndef() && Op1.isUndef())
6362 return DAG.getUNDEF(MVT::v2i64);
6363 // If one of the two inputs is undefined then replicate the other one,
6364 // in order to avoid using another register unnecessarily.
6365 if (Op0.isUndef())
6366 Op0 = Op1 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op1);
6367 else if (Op1.isUndef())
6368 Op0 = Op1 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op0);
6369 else {
6370 Op0 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op0);
6371 Op1 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op1);
6372 }
6373 return DAG.getNode(SystemZISD::JOIN_DWORDS, DL, MVT::v2i64, Op0, Op1);
6374}
6375
6376// If a BUILD_VECTOR contains some EXTRACT_VECTOR_ELTs, it's usually
6377// better to use VECTOR_SHUFFLEs on them, only using BUILD_VECTOR for
6378// the non-EXTRACT_VECTOR_ELT elements. See if the given BUILD_VECTOR
6379// would benefit from this representation and return it if so.
6381 BuildVectorSDNode *BVN) {
6382 EVT VT = BVN->getValueType(0);
6383 unsigned NumElements = VT.getVectorNumElements();
6384
6385 // Represent the BUILD_VECTOR as an N-operand VECTOR_SHUFFLE-like operation
6386 // on byte vectors. If there are non-EXTRACT_VECTOR_ELT elements that still
6387 // need a BUILD_VECTOR, add an additional placeholder operand for that
6388 // BUILD_VECTOR and store its operands in ResidueOps.
6389 GeneralShuffle GS(VT);
6391 bool FoundOne = false;
6392 for (unsigned I = 0; I < NumElements; ++I) {
6393 SDValue Op = BVN->getOperand(I);
6394 if (Op.getOpcode() == ISD::TRUNCATE)
6395 Op = Op.getOperand(0);
6396 if (Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
6397 Op.getOperand(1).getOpcode() == ISD::Constant) {
6398 unsigned Elem = Op.getConstantOperandVal(1);
6399 if (!GS.add(Op.getOperand(0), Elem))
6400 return SDValue();
6401 FoundOne = true;
6402 } else if (Op.isUndef()) {
6403 GS.addUndef();
6404 } else {
6405 if (!GS.add(SDValue(), ResidueOps.size()))
6406 return SDValue();
6407 ResidueOps.push_back(BVN->getOperand(I));
6408 }
6409 }
6410
6411 // Nothing to do if there are no EXTRACT_VECTOR_ELTs.
6412 if (!FoundOne)
6413 return SDValue();
6414
6415 // Create the BUILD_VECTOR for the remaining elements, if any.
6416 if (!ResidueOps.empty()) {
6417 while (ResidueOps.size() < NumElements)
6418 ResidueOps.push_back(DAG.getUNDEF(ResidueOps[0].getValueType()));
6419 for (auto &Op : GS.Ops) {
6420 if (!Op.getNode()) {
6421 Op = DAG.getBuildVector(VT, SDLoc(BVN), ResidueOps);
6422 break;
6423 }
6424 }
6425 }
6426 return GS.getNode(DAG, SDLoc(BVN));
6427}
6428
6429bool SystemZTargetLowering::isVectorElementLoad(SDValue Op) const {
6430 if (Op.getOpcode() == ISD::LOAD && cast<LoadSDNode>(Op)->isUnindexed())
6431 return true;
6432 if (auto *AL = dyn_cast<AtomicSDNode>(Op))
6433 if (AL->getOpcode() == ISD::ATOMIC_LOAD)
6434 return true;
6435 if (Subtarget.hasVectorEnhancements2() && Op.getOpcode() == SystemZISD::LRV)
6436 return true;
6437 return false;
6438}
6439
6441 unsigned MergedBits, EVT VT, SDValue Op0,
6442 SDValue Op1) {
6443 MVT IntVecVT = MVT::getVectorVT(MVT::getIntegerVT(MergedBits),
6444 SystemZ::VectorBits / MergedBits);
6445 assert(VT.getSizeInBits() == 128 && IntVecVT.getSizeInBits() == 128 &&
6446 "Handling full vectors only.");
6447 Op0 = DAG.getNode(ISD::BITCAST, DL, IntVecVT, Op0);
6448 Op1 = DAG.getNode(ISD::BITCAST, DL, IntVecVT, Op1);
6449 SDValue Op = DAG.getNode(SystemZISD::MERGE_HIGH, DL, IntVecVT, Op0, Op1);
6450 return DAG.getNode(ISD::BITCAST, DL, VT, Op);
6451}
6452
6454 EVT VT, SmallVectorImpl<SDValue> &Elems,
6455 unsigned Pos) {
6456 SDValue Op01 = buildMergeScalars(DAG, DL, VT, Elems[Pos + 0], Elems[Pos + 1]);
6457 SDValue Op23 = buildMergeScalars(DAG, DL, VT, Elems[Pos + 2], Elems[Pos + 3]);
6458 // Avoid unnecessary undefs by reusing the other operand.
6459 if (Op01.isUndef()) {
6460 if (Op23.isUndef())
6461 return Op01;
6462 Op01 = Op23;
6463 } else if (Op23.isUndef())
6464 Op23 = Op01;
6465 // Merging identical replications is a no-op.
6466 if (Op01.getOpcode() == SystemZISD::REPLICATE && Op01 == Op23)
6467 return Op01;
6468 unsigned MergedBits = VT.getSimpleVT().getScalarSizeInBits() * 2;
6469 return mergeHighParts(DAG, DL, MergedBits, VT, Op01, Op23);
6470}
6471
6472// Combine GPR scalar values Elems into a vector of type VT.
6473SDValue
6474SystemZTargetLowering::buildVector(SelectionDAG &DAG, const SDLoc &DL, EVT VT,
6475 SmallVectorImpl<SDValue> &Elems) const {
6476 // See whether there is a single replicated value.
6478 unsigned int NumElements = Elems.size();
6479 unsigned int Count = 0;
6480 for (auto Elem : Elems) {
6481 if (!Elem.isUndef()) {
6482 if (!Single.getNode())
6483 Single = Elem;
6484 else if (Elem != Single) {
6485 Single = SDValue();
6486 break;
6487 }
6488 Count += 1;
6489 }
6490 }
6491 // There are three cases here:
6492 //
6493 // - if the only defined element is a loaded one, the best sequence
6494 // is a replicating load.
6495 //
6496 // - otherwise, if the only defined element is an i64 value, we will
6497 // end up with the same VLVGP sequence regardless of whether we short-cut
6498 // for replication or fall through to the later code.
6499 //
6500 // - otherwise, if the only defined element is an i32 or smaller value,
6501 // we would need 2 instructions to replicate it: VLVGP followed by VREPx.
6502 // This is only a win if the single defined element is used more than once.
6503 // In other cases we're better off using a single VLVGx.
6504 if (Single.getNode() && (Count > 1 || isVectorElementLoad(Single)))
6505 return DAG.getNode(SystemZISD::REPLICATE, DL, VT, Single);
6506
6507 // If all elements are loads, use VLREP/VLEs (below).
6508 bool AllLoads = true;
6509 for (auto Elem : Elems)
6510 if (!isVectorElementLoad(Elem)) {
6511 AllLoads = false;
6512 break;
6513 }
6514
6515 // The best way of building a v2i64 from two i64s is to use VLVGP.
6516 if (VT == MVT::v2i64 && !AllLoads)
6517 return joinDwords(DAG, DL, Elems[0], Elems[1]);
6518
6519 // Use a 64-bit merge high to combine two doubles.
6520 if (VT == MVT::v2f64 && !AllLoads)
6521 return buildMergeScalars(DAG, DL, VT, Elems[0], Elems[1]);
6522
6523 // Build v4f32 values directly from the FPRs:
6524 //
6525 // <Axxx> <Bxxx> <Cxxxx> <Dxxx>
6526 // V V VMRHF
6527 // <ABxx> <CDxx>
6528 // V VMRHG
6529 // <ABCD>
6530 if (VT == MVT::v4f32 && !AllLoads)
6531 return buildFPVecFromScalars4(DAG, DL, VT, Elems, 0);
6532
6533 // Same for v8f16.
6534 if (VT == MVT::v8f16 && !AllLoads) {
6535 SDValue Op0123 = buildFPVecFromScalars4(DAG, DL, VT, Elems, 0);
6536 SDValue Op4567 = buildFPVecFromScalars4(DAG, DL, VT, Elems, 4);
6537 // Avoid unnecessary undefs by reusing the other operand.
6538 if (Op0123.isUndef())
6539 Op0123 = Op4567;
6540 else if (Op4567.isUndef())
6541 Op4567 = Op0123;
6542 // Merging identical replications is a no-op.
6543 if (Op0123.getOpcode() == SystemZISD::REPLICATE && Op0123 == Op4567)
6544 return Op0123;
6545 return mergeHighParts(DAG, DL, 64, VT, Op0123, Op4567);
6546 }
6547
6548 // Collect the constant terms.
6551
6552 unsigned NumConstants = 0;
6553 for (unsigned I = 0; I < NumElements; ++I) {
6554 SDValue Elem = Elems[I];
6555 if (Elem.getOpcode() == ISD::Constant ||
6556 Elem.getOpcode() == ISD::ConstantFP) {
6557 NumConstants += 1;
6558 Constants[I] = Elem;
6559 Done[I] = true;
6560 }
6561 }
6562 // If there was at least one constant, fill in the other elements of
6563 // Constants with undefs to get a full vector constant and use that
6564 // as the starting point.
6566 SDValue ReplicatedVal;
6567 if (NumConstants > 0) {
6568 for (unsigned I = 0; I < NumElements; ++I)
6569 if (!Constants[I].getNode())
6570 Constants[I] = DAG.getUNDEF(Elems[I].getValueType());
6571 Result = DAG.getBuildVector(VT, DL, Constants);
6572 } else {
6573 // Otherwise try to use VLREP or VLVGP to start the sequence in order to
6574 // avoid a false dependency on any previous contents of the vector
6575 // register.
6576
6577 // Use a VLREP if at least one element is a load. Make sure to replicate
6578 // the load with the most elements having its value.
6579 std::map<const SDNode*, unsigned> UseCounts;
6580 SDNode *LoadMaxUses = nullptr;
6581 for (unsigned I = 0; I < NumElements; ++I)
6582 if (isVectorElementLoad(Elems[I])) {
6583 SDNode *Ld = Elems[I].getNode();
6584 unsigned Count = ++UseCounts[Ld];
6585 if (LoadMaxUses == nullptr || UseCounts[LoadMaxUses] < Count)
6586 LoadMaxUses = Ld;
6587 }
6588 if (LoadMaxUses != nullptr) {
6589 ReplicatedVal = SDValue(LoadMaxUses, 0);
6590 Result = DAG.getNode(SystemZISD::REPLICATE, DL, VT, ReplicatedVal);
6591 } else {
6592 // Try to use VLVGP.
6593 unsigned I1 = NumElements / 2 - 1;
6594 unsigned I2 = NumElements - 1;
6595 bool Def1 = !Elems[I1].isUndef();
6596 bool Def2 = !Elems[I2].isUndef();
6597 if (Def1 || Def2) {
6598 SDValue Elem1 = Elems[Def1 ? I1 : I2];
6599 SDValue Elem2 = Elems[Def2 ? I2 : I1];
6600 Result = DAG.getNode(ISD::BITCAST, DL, VT,
6601 joinDwords(DAG, DL, Elem1, Elem2));
6602 Done[I1] = true;
6603 Done[I2] = true;
6604 } else
6605 Result = DAG.getUNDEF(VT);
6606 }
6607 }
6608
6609 // Use VLVGx to insert the other elements.
6610 for (unsigned I = 0; I < NumElements; ++I)
6611 if (!Done[I] && !Elems[I].isUndef() && Elems[I] != ReplicatedVal)
6612 Result = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, Result, Elems[I],
6613 DAG.getConstant(I, DL, MVT::i32));
6614 return Result;
6615}
6616
6617SDValue SystemZTargetLowering::lowerBUILD_VECTOR(SDValue Op,
6618 SelectionDAG &DAG) const {
6619 auto *BVN = cast<BuildVectorSDNode>(Op.getNode());
6620 SDLoc DL(Op);
6621 EVT VT = Op.getValueType();
6622
6623 if (BVN->isConstant()) {
6624 if (SystemZVectorConstantInfo(BVN).isVectorConstantLegal(Subtarget))
6625 return Op;
6626
6627 // Fall back to loading it from memory.
6628 return SDValue();
6629 }
6630
6631 // See if we should use shuffles to construct the vector from other vectors.
6632 if (SDValue Res = tryBuildVectorShuffle(DAG, BVN))
6633 return Res;
6634
6635 // Detect SCALAR_TO_VECTOR conversions.
6637 return buildScalarToVector(DAG, DL, VT, Op.getOperand(0));
6638
6639 // Otherwise use buildVector to build the vector up from GPRs.
6640 unsigned NumElements = Op.getNumOperands();
6642 for (unsigned I = 0; I < NumElements; ++I)
6643 Ops[I] = Op.getOperand(I);
6644 return buildVector(DAG, DL, VT, Ops);
6645}
6646
6647SDValue SystemZTargetLowering::lowerVECTOR_SHUFFLE(SDValue Op,
6648 SelectionDAG &DAG) const {
6649 auto *VSN = cast<ShuffleVectorSDNode>(Op.getNode());
6650 SDLoc DL(Op);
6651 EVT VT = Op.getValueType();
6652 unsigned NumElements = VT.getVectorNumElements();
6653
6654 if (VSN->isSplat()) {
6655 SDValue Op0 = Op.getOperand(0);
6656 unsigned Index = VSN->getSplatIndex();
6657 assert(Index < VT.getVectorNumElements() &&
6658 "Splat index should be defined and in first operand");
6659 // See whether the value we're splatting is directly available as a scalar.
6660 if ((Index == 0 && Op0.getOpcode() == ISD::SCALAR_TO_VECTOR) ||
6662 return DAG.getNode(SystemZISD::REPLICATE, DL, VT, Op0.getOperand(Index));
6663 // Otherwise keep it as a vector-to-vector operation.
6664 return DAG.getNode(SystemZISD::SPLAT, DL, VT, Op.getOperand(0),
6665 DAG.getTargetConstant(Index, DL, MVT::i32));
6666 }
6667
6668 GeneralShuffle GS(VT);
6669 for (unsigned I = 0; I < NumElements; ++I) {
6670 int Elt = VSN->getMaskElt(I);
6671 if (Elt < 0)
6672 GS.addUndef();
6673 else if (!GS.add(Op.getOperand(unsigned(Elt) / NumElements),
6674 unsigned(Elt) % NumElements))
6675 return SDValue();
6676 }
6677 return GS.getNode(DAG, SDLoc(VSN));
6678}
6679
6680SDValue SystemZTargetLowering::lowerSCALAR_TO_VECTOR(SDValue Op,
6681 SelectionDAG &DAG) const {
6682 SDLoc DL(Op);
6683 // Just insert the scalar into element 0 of an undefined vector.
6684 return DAG.getNode(ISD::INSERT_VECTOR_ELT, DL,
6685 Op.getValueType(), DAG.getUNDEF(Op.getValueType()),
6686 Op.getOperand(0), DAG.getConstant(0, DL, MVT::i32));
6687}
6688
6689// Shift the lower 2 bytes of Op to the left in order to insert into the
6690// upper 2 bytes of the FP register.
6692 assert(Op.getSimpleValueType() == MVT::i64 &&
6693 "Expexted to convert i64 to f16.");
6694 SDLoc DL(Op);
6695 SDValue Shft = DAG.getNode(ISD::SHL, DL, MVT::i64, Op,
6696 DAG.getConstant(48, DL, MVT::i64));
6697 SDValue BCast = DAG.getNode(ISD::BITCAST, DL, MVT::f64, Shft);
6698 SDValue F16Val =
6699 DAG.getTargetExtractSubreg(SystemZ::subreg_h16, DL, MVT::f16, BCast);
6700 return F16Val;
6701}
6702
6703// Extract Op into GPR and shift the 2 f16 bytes to the right.
6705 assert(Op.getSimpleValueType() == MVT::f16 &&
6706 "Expected to convert f16 to i64.");
6707 SDNode *U32 = DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::f64);
6708 SDValue In64 = DAG.getTargetInsertSubreg(SystemZ::subreg_h16, DL, MVT::f64,
6709 SDValue(U32, 0), Op);
6710 SDValue BCast = DAG.getNode(ISD::BITCAST, DL, MVT::i64, In64);
6711 SDValue Shft = DAG.getNode(ISD::SRL, DL, MVT::i64, BCast,
6712 DAG.getConstant(48, DL, MVT::i32));
6713 return Shft;
6714}
6715
6716SDValue SystemZTargetLowering::lowerINSERT_VECTOR_ELT(SDValue Op,
6717 SelectionDAG &DAG) const {
6718 // Handle insertions of floating-point values.
6719 SDLoc DL(Op);
6720 SDValue Op0 = Op.getOperand(0);
6721 SDValue Op1 = Op.getOperand(1);
6722 SDValue Op2 = Op.getOperand(2);
6723 EVT VT = Op.getValueType();
6724
6725 // Insertions into constant indices of a v2f64 can be done using VPDI.
6726 // However, if the inserted value is a bitcast or a constant then it's
6727 // better to use GPRs, as below.
6728 if (VT == MVT::v2f64 &&
6729 Op1.getOpcode() != ISD::BITCAST &&
6730 Op1.getOpcode() != ISD::ConstantFP &&
6731 Op2.getOpcode() == ISD::Constant) {
6732 uint64_t Index = Op2->getAsZExtVal();
6733 unsigned Mask = VT.getVectorNumElements() - 1;
6734 if (Index <= Mask)
6735 return Op;
6736 }
6737
6738 // Otherwise bitcast to the equivalent integer form and insert via a GPR.
6739 MVT IntVT = MVT::getIntegerVT(VT.getScalarSizeInBits());
6740 MVT IntVecVT = MVT::getVectorVT(IntVT, VT.getVectorNumElements());
6741 SDValue IntOp1 =
6742 VT == MVT::v8f16
6743 ? DAG.getZExtOrTrunc(convertFromF16(Op1, DL, DAG), DL, MVT::i32)
6744 : DAG.getNode(ISD::BITCAST, DL, IntVT, Op1);
6745 SDValue Res =
6746 DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, IntVecVT,
6747 DAG.getNode(ISD::BITCAST, DL, IntVecVT, Op0), IntOp1, Op2);
6748 return DAG.getNode(ISD::BITCAST, DL, VT, Res);
6749}
6750
6751SDValue
6752SystemZTargetLowering::lowerEXTRACT_VECTOR_ELT(SDValue Op,
6753 SelectionDAG &DAG) const {
6754 // Handle extractions of floating-point values.
6755 SDLoc DL(Op);
6756 SDValue Op0 = Op.getOperand(0);
6757 SDValue Op1 = Op.getOperand(1);
6758 EVT VT = Op.getValueType();
6759 EVT VecVT = Op0.getValueType();
6760
6761 // Extractions of constant indices can be done directly.
6762 if (auto *CIndexN = dyn_cast<ConstantSDNode>(Op1)) {
6763 uint64_t Index = CIndexN->getZExtValue();
6764 unsigned Mask = VecVT.getVectorNumElements() - 1;
6765 if (Index <= Mask)
6766 return Op;
6767 }
6768
6769 // Otherwise bitcast to the equivalent integer form and extract via a GPR.
6770 MVT IntVT = MVT::getIntegerVT(VT.getSizeInBits());
6771 MVT IntVecVT = MVT::getVectorVT(IntVT, VecVT.getVectorNumElements());
6772 MVT ExtrVT = IntVT == MVT::i16 ? MVT::i32 : IntVT;
6773 SDValue Extr = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ExtrVT,
6774 DAG.getNode(ISD::BITCAST, DL, IntVecVT, Op0), Op1);
6775 if (VT == MVT::f16)
6776 return convertToF16(DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Extr), DAG);
6777 return DAG.getNode(ISD::BITCAST, DL, VT, Extr);
6778}
6779
6780SDValue SystemZTargetLowering::
6781lowerSIGN_EXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const {
6782 SDValue PackedOp = Op.getOperand(0);
6783 EVT OutVT = Op.getValueType();
6784 EVT InVT = PackedOp.getValueType();
6785 unsigned ToBits = OutVT.getScalarSizeInBits();
6786 unsigned FromBits = InVT.getScalarSizeInBits();
6787 unsigned StartOffset = 0;
6788
6789 // If the input is a VECTOR_SHUFFLE, there are a number of important
6790 // cases where we can directly implement the sign-extension of the
6791 // original input lanes of the shuffle.
6792 if (PackedOp.getOpcode() == ISD::VECTOR_SHUFFLE) {
6793 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(PackedOp.getNode());
6794 ArrayRef<int> ShuffleMask = SVN->getMask();
6795 int OutNumElts = OutVT.getVectorNumElements();
6796
6797 // Recognize the special case where the sign-extension can be done
6798 // by the VSEG instruction. Handled via the default expander.
6799 if (ToBits == 64 && OutNumElts == 2) {
6800 int NumElem = ToBits / FromBits;
6801 if (ShuffleMask[0] == NumElem - 1 && ShuffleMask[1] == 2 * NumElem - 1)
6802 return SDValue();
6803 }
6804
6805 // Recognize the special case where we can fold the shuffle by
6806 // replacing some of the UNPACK_HIGH with UNPACK_LOW.
6807 int StartOffsetCandidate = -1;
6808 for (int Elt = 0; Elt < OutNumElts; Elt++) {
6809 if (ShuffleMask[Elt] == -1)
6810 continue;
6811 if (ShuffleMask[Elt] % OutNumElts == Elt) {
6812 if (StartOffsetCandidate == -1)
6813 StartOffsetCandidate = ShuffleMask[Elt] - Elt;
6814 if (StartOffsetCandidate == ShuffleMask[Elt] - Elt)
6815 continue;
6816 }
6817 StartOffsetCandidate = -1;
6818 break;
6819 }
6820 if (StartOffsetCandidate != -1) {
6821 StartOffset = StartOffsetCandidate;
6822 PackedOp = PackedOp.getOperand(0);
6823 }
6824 }
6825
6826 do {
6827 FromBits *= 2;
6828 unsigned OutNumElts = SystemZ::VectorBits / FromBits;
6829 EVT OutVT = MVT::getVectorVT(MVT::getIntegerVT(FromBits), OutNumElts);
6830 unsigned Opcode = SystemZISD::UNPACK_HIGH;
6831 if (StartOffset >= OutNumElts) {
6832 Opcode = SystemZISD::UNPACK_LOW;
6833 StartOffset -= OutNumElts;
6834 }
6835 PackedOp = DAG.getNode(Opcode, SDLoc(PackedOp), OutVT, PackedOp);
6836 } while (FromBits != ToBits);
6837 return PackedOp;
6838}
6839
6840// Lower a ZERO_EXTEND_VECTOR_INREG to a vector shuffle with a zero vector.
6841SDValue SystemZTargetLowering::
6842lowerZERO_EXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const {
6843 SDValue PackedOp = Op.getOperand(0);
6844 SDLoc DL(Op);
6845 EVT OutVT = Op.getValueType();
6846 EVT InVT = PackedOp.getValueType();
6847 unsigned InNumElts = InVT.getVectorNumElements();
6848 unsigned OutNumElts = OutVT.getVectorNumElements();
6849 unsigned NumInPerOut = InNumElts / OutNumElts;
6850
6851 SDValue ZeroVec =
6852 DAG.getSplatVector(InVT, DL, DAG.getConstant(0, DL, InVT.getScalarType()));
6853
6854 SmallVector<int, 16> Mask(InNumElts);
6855 unsigned ZeroVecElt = InNumElts;
6856 for (unsigned PackedElt = 0; PackedElt < OutNumElts; PackedElt++) {
6857 unsigned MaskElt = PackedElt * NumInPerOut;
6858 unsigned End = MaskElt + NumInPerOut - 1;
6859 for (; MaskElt < End; MaskElt++)
6860 Mask[MaskElt] = ZeroVecElt++;
6861 Mask[MaskElt] = PackedElt;
6862 }
6863 SDValue Shuf = DAG.getVectorShuffle(InVT, DL, PackedOp, ZeroVec, Mask);
6864 return DAG.getNode(ISD::BITCAST, DL, OutVT, Shuf);
6865}
6866
6867SDValue SystemZTargetLowering::lowerShift(SDValue Op, SelectionDAG &DAG,
6868 unsigned ByScalar) const {
6869 // Look for cases where a vector shift can use the *_BY_SCALAR form.
6870 SDValue Op0 = Op.getOperand(0);
6871 SDValue Op1 = Op.getOperand(1);
6872 SDLoc DL(Op);
6873 EVT VT = Op.getValueType();
6874 unsigned ElemBitSize = VT.getScalarSizeInBits();
6875
6876 // See whether the shift vector is a splat represented as BUILD_VECTOR.
6877 if (auto *BVN = dyn_cast<BuildVectorSDNode>(Op1)) {
6878 APInt SplatBits, SplatUndef;
6879 unsigned SplatBitSize;
6880 bool HasAnyUndefs;
6881 // Check for constant splats. Use ElemBitSize as the minimum element
6882 // width and reject splats that need wider elements.
6883 if (BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs,
6884 ElemBitSize, true) &&
6885 SplatBitSize == ElemBitSize) {
6886 SDValue Shift = DAG.getConstant(SplatBits.getZExtValue() & 0xfff,
6887 DL, MVT::i32);
6888 return DAG.getNode(ByScalar, DL, VT, Op0, Shift);
6889 }
6890 // Check for variable splats.
6891 BitVector UndefElements;
6892 SDValue Splat = BVN->getSplatValue(&UndefElements);
6893 if (Splat) {
6894 // Since i32 is the smallest legal type, we either need a no-op
6895 // or a truncation.
6896 SDValue Shift = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Splat);
6897 return DAG.getNode(ByScalar, DL, VT, Op0, Shift);
6898 }
6899 }
6900
6901 // See whether the shift vector is a splat represented as SHUFFLE_VECTOR,
6902 // and the shift amount is directly available in a GPR.
6903 if (auto *VSN = dyn_cast<ShuffleVectorSDNode>(Op1)) {
6904 if (VSN->isSplat()) {
6905 SDValue VSNOp0 = VSN->getOperand(0);
6906 unsigned Index = VSN->getSplatIndex();
6907 assert(Index < VT.getVectorNumElements() &&
6908 "Splat index should be defined and in first operand");
6909 if ((Index == 0 && VSNOp0.getOpcode() == ISD::SCALAR_TO_VECTOR) ||
6910 VSNOp0.getOpcode() == ISD::BUILD_VECTOR) {
6911 // Since i32 is the smallest legal type, we either need a no-op
6912 // or a truncation.
6913 SDValue Shift = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32,
6914 VSNOp0.getOperand(Index));
6915 return DAG.getNode(ByScalar, DL, VT, Op0, Shift);
6916 }
6917 }
6918 }
6919
6920 // Otherwise just treat the current form as legal.
6921 return Op;
6922}
6923
6924SDValue SystemZTargetLowering::lowerFSHL(SDValue Op, SelectionDAG &DAG) const {
6925 SDLoc DL(Op);
6926
6927 // i128 FSHL with a constant amount that is a multiple of 8 can be
6928 // implemented via VECTOR_SHUFFLE. If we have the vector-enhancements-2
6929 // facility, FSHL with a constant amount less than 8 can be implemented
6930 // via SHL_DOUBLE_BIT, and FSHL with other constant amounts by a
6931 // combination of the two.
6932 if (auto *ShiftAmtNode = dyn_cast<ConstantSDNode>(Op.getOperand(2))) {
6933 uint64_t ShiftAmt = ShiftAmtNode->getZExtValue() & 127;
6934 if ((ShiftAmt & 7) == 0 || Subtarget.hasVectorEnhancements2()) {
6935 SDValue Op0 = DAG.getBitcast(MVT::v16i8, Op.getOperand(0));
6936 SDValue Op1 = DAG.getBitcast(MVT::v16i8, Op.getOperand(1));
6937 if (ShiftAmt > 120) {
6938 // For N in 121..128, fshl N == fshr (128 - N), and for 1 <= N < 8
6939 // SHR_DOUBLE_BIT emits fewer instructions.
6940 SDValue Val =
6941 DAG.getNode(SystemZISD::SHR_DOUBLE_BIT, DL, MVT::v16i8, Op0, Op1,
6942 DAG.getTargetConstant(128 - ShiftAmt, DL, MVT::i32));
6943 return DAG.getBitcast(MVT::i128, Val);
6944 }
6945 SmallVector<int, 16> Mask(16);
6946 for (unsigned Elt = 0; Elt < 16; Elt++)
6947 Mask[Elt] = (ShiftAmt >> 3) + Elt;
6948 SDValue Shuf1 = DAG.getVectorShuffle(MVT::v16i8, DL, Op0, Op1, Mask);
6949 if ((ShiftAmt & 7) == 0)
6950 return DAG.getBitcast(MVT::i128, Shuf1);
6951 SDValue Shuf2 = DAG.getVectorShuffle(MVT::v16i8, DL, Op1, Op1, Mask);
6952 SDValue Val =
6953 DAG.getNode(SystemZISD::SHL_DOUBLE_BIT, DL, MVT::v16i8, Shuf1, Shuf2,
6954 DAG.getTargetConstant(ShiftAmt & 7, DL, MVT::i32));
6955 return DAG.getBitcast(MVT::i128, Val);
6956 }
6957 }
6958
6959 return SDValue();
6960}
6961
6962SDValue SystemZTargetLowering::lowerFSHR(SDValue Op, SelectionDAG &DAG) const {
6963 SDLoc DL(Op);
6964
6965 // i128 FSHR with a constant amount that is a multiple of 8 can be
6966 // implemented via VECTOR_SHUFFLE. If we have the vector-enhancements-2
6967 // facility, FSHR with a constant amount less than 8 can be implemented
6968 // via SHR_DOUBLE_BIT, and FSHR with other constant amounts by a
6969 // combination of the two.
6970 if (auto *ShiftAmtNode = dyn_cast<ConstantSDNode>(Op.getOperand(2))) {
6971 uint64_t ShiftAmt = ShiftAmtNode->getZExtValue() & 127;
6972 if ((ShiftAmt & 7) == 0 || Subtarget.hasVectorEnhancements2()) {
6973 SDValue Op0 = DAG.getBitcast(MVT::v16i8, Op.getOperand(0));
6974 SDValue Op1 = DAG.getBitcast(MVT::v16i8, Op.getOperand(1));
6975 if (ShiftAmt > 120) {
6976 // For N in 121..128, fshr N == fshl (128 - N), and for 1 <= N < 8
6977 // SHL_DOUBLE_BIT emits fewer instructions.
6978 SDValue Val =
6979 DAG.getNode(SystemZISD::SHL_DOUBLE_BIT, DL, MVT::v16i8, Op0, Op1,
6980 DAG.getTargetConstant(128 - ShiftAmt, DL, MVT::i32));
6981 return DAG.getBitcast(MVT::i128, Val);
6982 }
6983 SmallVector<int, 16> Mask(16);
6984 for (unsigned Elt = 0; Elt < 16; Elt++)
6985 Mask[Elt] = 16 - (ShiftAmt >> 3) + Elt;
6986 SDValue Shuf1 = DAG.getVectorShuffle(MVT::v16i8, DL, Op0, Op1, Mask);
6987 if ((ShiftAmt & 7) == 0)
6988 return DAG.getBitcast(MVT::i128, Shuf1);
6989 SDValue Shuf2 = DAG.getVectorShuffle(MVT::v16i8, DL, Op0, Op0, Mask);
6990 SDValue Val =
6991 DAG.getNode(SystemZISD::SHR_DOUBLE_BIT, DL, MVT::v16i8, Shuf2, Shuf1,
6992 DAG.getTargetConstant(ShiftAmt & 7, DL, MVT::i32));
6993 return DAG.getBitcast(MVT::i128, Val);
6994 }
6995 }
6996
6997 return SDValue();
6998}
6999
7001 SDLoc DL(Op);
7002 SDValue Src = Op.getOperand(0);
7003 MVT DstVT = Op.getSimpleValueType();
7004
7006 unsigned SrcAS = N->getSrcAddressSpace();
7007
7008 assert(SrcAS != N->getDestAddressSpace() &&
7009 "addrspacecast must be between different address spaces");
7010
7011 // addrspacecast [0 <- 1] : Assinging a ptr32 value to a 64-bit pointer.
7012 // addrspacecast [1 <- 0] : Assigining a 64-bit pointer to a ptr32 value.
7013 if (SrcAS == SYSTEMZAS::PTR32 && DstVT == MVT::i64) {
7014 Op = DAG.getNode(ISD::AND, DL, MVT::i32, Src,
7015 DAG.getConstant(0x7fffffff, DL, MVT::i32));
7016 Op = DAG.getNode(ISD::ZERO_EXTEND, DL, DstVT, Op);
7017 } else if (DstVT == MVT::i32) {
7018 Op = DAG.getNode(ISD::TRUNCATE, DL, DstVT, Src);
7019 Op = DAG.getNode(ISD::AND, DL, MVT::i32, Op,
7020 DAG.getConstant(0x7fffffff, DL, MVT::i32));
7021 Op = DAG.getNode(ISD::ZERO_EXTEND, DL, DstVT, Op);
7022 } else {
7023 report_fatal_error("Bad address space in addrspacecast");
7024 }
7025 return Op;
7026}
7027
7028SDValue SystemZTargetLowering::lowerFP_EXTEND(SDValue Op,
7029 SelectionDAG &DAG) const {
7030 SDValue In = Op.getOperand(Op->isStrictFPOpcode() ? 1 : 0);
7031 if (In.getSimpleValueType() != MVT::f16)
7032 return Op; // Legal
7033 return SDValue(); // Let legalizer emit the libcall.
7034}
7035
7037 MVT VT, SDValue Arg, SDLoc DL,
7038 SDValue Chain, bool IsStrict) const {
7039 assert(LC != RTLIB::UNKNOWN_LIBCALL && "Unexpected request for libcall!");
7040 MakeLibCallOptions CallOptions;
7041 SDValue Result;
7042 std::tie(Result, Chain) =
7043 makeLibCall(DAG, LC, VT, Arg, CallOptions, DL, Chain);
7044 return IsStrict ? DAG.getMergeValues({Result, Chain}, DL) : Result;
7045}
7046
7047SDValue SystemZTargetLowering::lower_FP_TO_INT(SDValue Op,
7048 SelectionDAG &DAG) const {
7049 bool IsSigned = (Op->getOpcode() == ISD::FP_TO_SINT ||
7050 Op->getOpcode() == ISD::STRICT_FP_TO_SINT);
7051 bool IsStrict = Op->isStrictFPOpcode();
7052 SDLoc DL(Op);
7053 MVT VT = Op.getSimpleValueType();
7054 SDValue InOp = Op.getOperand(IsStrict ? 1 : 0);
7055 SDValue Chain = IsStrict ? Op.getOperand(0) : DAG.getEntryNode();
7056 EVT InVT = InOp.getValueType();
7057
7058 // FP to unsigned is not directly supported on z10. Promoting an i32
7059 // result to (signed) i64 doesn't generate an inexact condition (fp
7060 // exception) for values that are outside the i32 range but in the i64
7061 // range, so use the default expansion.
7062 if (!Subtarget.hasFPExtension() && !IsSigned)
7063 // Expand i32/i64. F16 values will be recognized to fit and extended.
7064 return SDValue();
7065
7066 // Conversion from f16 is done via f32.
7067 if (InOp.getSimpleValueType() == MVT::f16) {
7069 LowerOperationWrapper(Op.getNode(), Results, DAG);
7070 return DAG.getMergeValues(Results, DL);
7071 }
7072
7073 if (VT == MVT::i128) {
7074 RTLIB::Libcall LC =
7075 IsSigned ? RTLIB::getFPTOSINT(InVT, VT) : RTLIB::getFPTOUINT(InVT, VT);
7076 return useLibCall(DAG, LC, VT, InOp, DL, Chain, IsStrict);
7077 }
7078
7079 return Op; // Legal
7080}
7081
7082SDValue SystemZTargetLowering::lower_INT_TO_FP(SDValue Op,
7083 SelectionDAG &DAG) const {
7084 bool IsSigned = (Op->getOpcode() == ISD::SINT_TO_FP ||
7085 Op->getOpcode() == ISD::STRICT_SINT_TO_FP);
7086 bool IsStrict = Op->isStrictFPOpcode();
7087 SDLoc DL(Op);
7088 MVT VT = Op.getSimpleValueType();
7089 SDValue InOp = Op.getOperand(IsStrict ? 1 : 0);
7090 SDValue Chain = IsStrict ? Op.getOperand(0) : DAG.getEntryNode();
7091 EVT InVT = InOp.getValueType();
7092
7093 // Conversion to f16 is done via f32.
7094 if (VT == MVT::f16) {
7096 LowerOperationWrapper(Op.getNode(), Results, DAG);
7097 return DAG.getMergeValues(Results, DL);
7098 }
7099
7100 // Unsigned to fp is not directly supported on z10.
7101 if (!Subtarget.hasFPExtension() && !IsSigned)
7102 return SDValue(); // Expand i64.
7103
7104 if (InVT == MVT::i128) {
7105 RTLIB::Libcall LC =
7106 IsSigned ? RTLIB::getSINTTOFP(InVT, VT) : RTLIB::getUINTTOFP(InVT, VT);
7107 return useLibCall(DAG, LC, VT, InOp, DL, Chain, IsStrict);
7108 }
7109
7110 return Op; // Legal
7111}
7112
7113// Lower an f16 LOAD in case of no vector support.
7114SDValue SystemZTargetLowering::lowerLoadF16(SDValue Op,
7115 SelectionDAG &DAG) const {
7116 EVT RegVT = Op.getValueType();
7117 assert(RegVT == MVT::f16 && "Expected to lower an f16 load.");
7118 (void)RegVT;
7119
7120 // Load as integer.
7121 SDLoc DL(Op);
7122 SDValue NewLd;
7123 if (auto *AtomicLd = dyn_cast<AtomicSDNode>(Op.getNode())) {
7124 assert(EVT(RegVT) == AtomicLd->getMemoryVT() && "Unhandled f16 load");
7125 NewLd = DAG.getAtomicLoad(ISD::EXTLOAD, DL, MVT::i16, MVT::i64,
7126 AtomicLd->getChain(), AtomicLd->getBasePtr(),
7127 AtomicLd->getMemOperand());
7128 } else {
7129 LoadSDNode *Ld = cast<LoadSDNode>(Op.getNode());
7130 assert(EVT(RegVT) == Ld->getMemoryVT() && "Unhandled f16 load");
7131 NewLd = DAG.getExtLoad(ISD::EXTLOAD, DL, MVT::i64, Ld->getChain(),
7132 Ld->getBasePtr(), Ld->getPointerInfo(), MVT::i16,
7133 Ld->getBaseAlign(), Ld->getMemOperand()->getFlags());
7134 }
7135 SDValue F16Val = convertToF16(NewLd, DAG);
7136 return DAG.getMergeValues({F16Val, NewLd.getValue(1)}, DL);
7137}
7138
7139// Lower an f16 STORE in case of no vector support.
7140SDValue SystemZTargetLowering::lowerStoreF16(SDValue Op,
7141 SelectionDAG &DAG) const {
7142 SDLoc DL(Op);
7143 SDValue Shft = convertFromF16(Op->getOperand(1), DL, DAG);
7144
7145 if (auto *AtomicSt = dyn_cast<AtomicSDNode>(Op.getNode()))
7146 return DAG.getAtomic(ISD::ATOMIC_STORE, DL, MVT::i16, AtomicSt->getChain(),
7147 Shft, AtomicSt->getBasePtr(),
7148 AtomicSt->getMemOperand());
7149
7150 StoreSDNode *St = cast<StoreSDNode>(Op.getNode());
7151 return DAG.getTruncStore(St->getChain(), DL, Shft, St->getBasePtr(), MVT::i16,
7152 St->getMemOperand());
7153}
7154
7155SDValue SystemZTargetLowering::lowerIS_FPCLASS(SDValue Op,
7156 SelectionDAG &DAG) const {
7157 SDLoc DL(Op);
7158 MVT ResultVT = Op.getSimpleValueType();
7159 SDValue Arg = Op.getOperand(0);
7160 unsigned Check = Op.getConstantOperandVal(1);
7161
7162 unsigned TDCMask = 0;
7163 if (Check & fcSNan)
7165 if (Check & fcQNan)
7167 if (Check & fcPosInf)
7169 if (Check & fcNegInf)
7171 if (Check & fcPosNormal)
7173 if (Check & fcNegNormal)
7175 if (Check & fcPosSubnormal)
7177 if (Check & fcNegSubnormal)
7179 if (Check & fcPosZero)
7180 TDCMask |= SystemZ::TDCMASK_ZERO_PLUS;
7181 if (Check & fcNegZero)
7182 TDCMask |= SystemZ::TDCMASK_ZERO_MINUS;
7183 SDValue TDCMaskV = DAG.getConstant(TDCMask, DL, MVT::i64);
7184
7185 SDValue Intr = DAG.getNode(SystemZISD::TDC, DL, ResultVT, Arg, TDCMaskV);
7186 return getCCResult(DAG, Intr);
7187}
7188
7189SDValue SystemZTargetLowering::lowerREADCYCLECOUNTER(SDValue Op,
7190 SelectionDAG &DAG) const {
7191 SDLoc DL(Op);
7192 SDValue Chain = Op.getOperand(0);
7193
7194 // STCKF only supports a memory operand, so we have to use a temporary.
7195 SDValue StackPtr = DAG.CreateStackTemporary(MVT::i64);
7196 int SPFI = cast<FrameIndexSDNode>(StackPtr.getNode())->getIndex();
7197 MachinePointerInfo MPI =
7199
7200 // Use STCFK to store the TOD clock into the temporary.
7201 SDValue StoreOps[] = {Chain, StackPtr};
7202 Chain = DAG.getMemIntrinsicNode(
7203 SystemZISD::STCKF, DL, DAG.getVTList(MVT::Other), StoreOps, MVT::i64,
7204 MPI, MaybeAlign(), MachineMemOperand::MOStore);
7205
7206 // And read it back from there.
7207 return DAG.getLoad(MVT::i64, DL, Chain, StackPtr, MPI);
7208}
7209
7211 SelectionDAG &DAG) const {
7212 switch (Op.getOpcode()) {
7213 case ISD::FRAMEADDR:
7214 return lowerFRAMEADDR(Op, DAG);
7215 case ISD::RETURNADDR:
7216 return lowerRETURNADDR(Op, DAG);
7217 case ISD::BR_CC:
7218 return lowerBR_CC(Op, DAG);
7219 case ISD::SELECT_CC:
7220 return lowerSELECT_CC(Op, DAG);
7221 case ISD::SETCC:
7222 return lowerSETCC(Op, DAG);
7223 case ISD::STRICT_FSETCC:
7224 return lowerSTRICT_FSETCC(Op, DAG, false);
7226 return lowerSTRICT_FSETCC(Op, DAG, true);
7227 case ISD::GlobalAddress:
7228 return lowerGlobalAddress(cast<GlobalAddressSDNode>(Op), DAG);
7230 return lowerGlobalTLSAddress(cast<GlobalAddressSDNode>(Op), DAG);
7231 case ISD::BlockAddress:
7232 return lowerBlockAddress(cast<BlockAddressSDNode>(Op), DAG);
7233 case ISD::JumpTable:
7234 return lowerJumpTable(cast<JumpTableSDNode>(Op), DAG);
7235 case ISD::ConstantPool:
7236 return lowerConstantPool(cast<ConstantPoolSDNode>(Op), DAG);
7237 case ISD::BITCAST:
7238 return lowerBITCAST(Op, DAG);
7239 case ISD::VASTART:
7240 return lowerVASTART(Op, DAG);
7241 case ISD::VACOPY:
7242 return lowerVACOPY(Op, DAG);
7244 return lowerDYNAMIC_STACKALLOC(Op, DAG);
7246 return lowerGET_DYNAMIC_AREA_OFFSET(Op, DAG);
7247 case ISD::MULHS:
7248 return lowerMULH(Op, DAG, SystemZISD::SMUL_LOHI);
7249 case ISD::MULHU:
7250 return lowerMULH(Op, DAG, SystemZISD::UMUL_LOHI);
7251 case ISD::SMUL_LOHI:
7252 return lowerSMUL_LOHI(Op, DAG);
7253 case ISD::UMUL_LOHI:
7254 return lowerUMUL_LOHI(Op, DAG);
7255 case ISD::SDIVREM:
7256 return lowerSDIVREM(Op, DAG);
7257 case ISD::UDIVREM:
7258 return lowerUDIVREM(Op, DAG);
7259 case ISD::SADDO:
7260 case ISD::SSUBO:
7261 case ISD::UADDO:
7262 case ISD::USUBO:
7263 return lowerXALUO(Op, DAG);
7264 case ISD::UADDO_CARRY:
7265 case ISD::USUBO_CARRY:
7266 return lowerUADDSUBO_CARRY(Op, DAG);
7267 case ISD::OR:
7268 return lowerOR(Op, DAG);
7269 case ISD::CTPOP:
7270 return lowerCTPOP(Op, DAG);
7271 case ISD::VECREDUCE_ADD:
7272 return lowerVECREDUCE_ADD(Op, DAG);
7273 case ISD::ATOMIC_FENCE:
7274 return lowerATOMIC_FENCE(Op, DAG);
7275 case ISD::ATOMIC_SWAP:
7276 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_SWAPW);
7277 case ISD::ATOMIC_STORE:
7278 return lowerATOMIC_STORE(Op, DAG);
7279 case ISD::ATOMIC_LOAD:
7280 return lowerATOMIC_LOAD(Op, DAG);
7282 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_ADD);
7284 return lowerATOMIC_LOAD_SUB(Op, DAG);
7286 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_AND);
7288 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_OR);
7290 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_XOR);
7292 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_NAND);
7294 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_MIN);
7296 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_MAX);
7298 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_UMIN);
7300 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_UMAX);
7302 return lowerATOMIC_CMP_SWAP(Op, DAG);
7303 case ISD::STACKSAVE:
7304 return lowerSTACKSAVE(Op, DAG);
7305 case ISD::STACKRESTORE:
7306 return lowerSTACKRESTORE(Op, DAG);
7307 case ISD::PREFETCH:
7308 return lowerPREFETCH(Op, DAG);
7310 return lowerINTRINSIC_W_CHAIN(Op, DAG);
7312 return lowerINTRINSIC_WO_CHAIN(Op, DAG);
7313 case ISD::BUILD_VECTOR:
7314 return lowerBUILD_VECTOR(Op, DAG);
7316 return lowerVECTOR_SHUFFLE(Op, DAG);
7318 return lowerSCALAR_TO_VECTOR(Op, DAG);
7320 return lowerINSERT_VECTOR_ELT(Op, DAG);
7322 return lowerEXTRACT_VECTOR_ELT(Op, DAG);
7324 return lowerSIGN_EXTEND_VECTOR_INREG(Op, DAG);
7326 return lowerZERO_EXTEND_VECTOR_INREG(Op, DAG);
7327 case ISD::SHL:
7328 return lowerShift(Op, DAG, SystemZISD::VSHL_BY_SCALAR);
7329 case ISD::SRL:
7330 return lowerShift(Op, DAG, SystemZISD::VSRL_BY_SCALAR);
7331 case ISD::SRA:
7332 return lowerShift(Op, DAG, SystemZISD::VSRA_BY_SCALAR);
7333 case ISD::ADDRSPACECAST:
7334 return lowerAddrSpaceCast(Op, DAG);
7335 case ISD::ROTL:
7336 return lowerShift(Op, DAG, SystemZISD::VROTL_BY_SCALAR);
7337 case ISD::FSHL:
7338 return lowerFSHL(Op, DAG);
7339 case ISD::FSHR:
7340 return lowerFSHR(Op, DAG);
7341 case ISD::FP_EXTEND:
7343 return lowerFP_EXTEND(Op, DAG);
7344 case ISD::FP_TO_UINT:
7345 case ISD::FP_TO_SINT:
7348 return lower_FP_TO_INT(Op, DAG);
7349 case ISD::UINT_TO_FP:
7350 case ISD::SINT_TO_FP:
7353 return lower_INT_TO_FP(Op, DAG);
7354 case ISD::LOAD:
7355 return lowerLoadF16(Op, DAG);
7356 case ISD::STORE:
7357 return lowerStoreF16(Op, DAG);
7358 case ISD::IS_FPCLASS:
7359 return lowerIS_FPCLASS(Op, DAG);
7360 case ISD::GET_ROUNDING:
7361 return lowerGET_ROUNDING(Op, DAG);
7363 return lowerREADCYCLECOUNTER(Op, DAG);
7366 // These operations are legal on our platform, but we cannot actually
7367 // set the operation action to Legal as common code would treat this
7368 // as equivalent to Expand. Instead, we keep the operation action to
7369 // Custom and just leave them unchanged here.
7370 return Op;
7371
7372 default:
7373 llvm_unreachable("Unexpected node to lower");
7374 }
7375}
7376
7378 const SDLoc &SL) {
7379 // If i128 is legal, just use a normal bitcast.
7380 if (DAG.getTargetLoweringInfo().isTypeLegal(MVT::i128))
7381 return DAG.getBitcast(MVT::f128, Src);
7382
7383 // Otherwise, f128 must live in FP128, so do a partwise move.
7385 &SystemZ::FP128BitRegClass);
7386
7387 SDValue Hi, Lo;
7388 std::tie(Lo, Hi) = DAG.SplitScalar(Src, SL, MVT::i64, MVT::i64);
7389
7390 Hi = DAG.getBitcast(MVT::f64, Hi);
7391 Lo = DAG.getBitcast(MVT::f64, Lo);
7392
7393 SDNode *Pair = DAG.getMachineNode(
7394 SystemZ::REG_SEQUENCE, SL, MVT::f128,
7395 {DAG.getTargetConstant(SystemZ::FP128BitRegClassID, SL, MVT::i32), Lo,
7396 DAG.getTargetConstant(SystemZ::subreg_l64, SL, MVT::i32), Hi,
7397 DAG.getTargetConstant(SystemZ::subreg_h64, SL, MVT::i32)});
7398 return SDValue(Pair, 0);
7399}
7400
7402 const SDLoc &SL) {
7403 // If i128 is legal, just use a normal bitcast.
7404 if (DAG.getTargetLoweringInfo().isTypeLegal(MVT::i128))
7405 return DAG.getBitcast(MVT::i128, Src);
7406
7407 // Otherwise, f128 must live in FP128, so do a partwise move.
7409 &SystemZ::FP128BitRegClass);
7410
7411 SDValue LoFP =
7412 DAG.getTargetExtractSubreg(SystemZ::subreg_l64, SL, MVT::f64, Src);
7413 SDValue HiFP =
7414 DAG.getTargetExtractSubreg(SystemZ::subreg_h64, SL, MVT::f64, Src);
7415 SDValue Lo = DAG.getNode(ISD::BITCAST, SL, MVT::i64, LoFP);
7416 SDValue Hi = DAG.getNode(ISD::BITCAST, SL, MVT::i64, HiFP);
7417
7418 return DAG.getNode(ISD::BUILD_PAIR, SL, MVT::i128, Lo, Hi);
7419}
7420
7421// Lower operations with invalid operand or result types.
7422void
7425 SelectionDAG &DAG) const {
7426 switch (N->getOpcode()) {
7427 case ISD::ATOMIC_LOAD: {
7428 SDLoc DL(N);
7429 SDVTList Tys = DAG.getVTList(MVT::Untyped, MVT::Other);
7430 SDValue Ops[] = { N->getOperand(0), N->getOperand(1) };
7431 MachineMemOperand *MMO = cast<AtomicSDNode>(N)->getMemOperand();
7432 SDValue Res = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_LOAD_128,
7433 DL, Tys, Ops, MVT::i128, MMO);
7434
7435 SDValue Lowered = lowerGR128ToI128(DAG, Res);
7436 if (N->getValueType(0) == MVT::f128)
7437 Lowered = expandBitCastI128ToF128(DAG, Lowered, DL);
7438 Results.push_back(Lowered);
7439 Results.push_back(Res.getValue(1));
7440 break;
7441 }
7442 case ISD::ATOMIC_STORE: {
7443 SDLoc DL(N);
7444 SDVTList Tys = DAG.getVTList(MVT::Other);
7445 SDValue Val = N->getOperand(1);
7446 if (Val.getValueType() == MVT::f128)
7447 Val = expandBitCastF128ToI128(DAG, Val, DL);
7448 Val = lowerI128ToGR128(DAG, Val);
7449
7450 SDValue Ops[] = {N->getOperand(0), Val, N->getOperand(2)};
7451 MachineMemOperand *MMO = cast<AtomicSDNode>(N)->getMemOperand();
7452 SDValue Res = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_STORE_128,
7453 DL, Tys, Ops, MVT::i128, MMO);
7454 // We have to enforce sequential consistency by performing a
7455 // serialization operation after the store.
7456 if (cast<AtomicSDNode>(N)->getSuccessOrdering() ==
7458 Res = SDValue(DAG.getMachineNode(SystemZ::Serialize, DL,
7459 MVT::Other, Res), 0);
7460 Results.push_back(Res);
7461 break;
7462 }
7464 SDLoc DL(N);
7465 SDVTList Tys = DAG.getVTList(MVT::Untyped, MVT::i32, MVT::Other);
7466 SDValue Ops[] = { N->getOperand(0), N->getOperand(1),
7467 lowerI128ToGR128(DAG, N->getOperand(2)),
7468 lowerI128ToGR128(DAG, N->getOperand(3)) };
7469 MachineMemOperand *MMO = cast<AtomicSDNode>(N)->getMemOperand();
7470 SDValue Res = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_CMP_SWAP_128,
7471 DL, Tys, Ops, MVT::i128, MMO);
7472 SDValue Success = emitSETCC(DAG, DL, Res.getValue(1),
7474 Success = DAG.getZExtOrTrunc(Success, DL, N->getValueType(1));
7475 Results.push_back(lowerGR128ToI128(DAG, Res));
7476 Results.push_back(Success);
7477 Results.push_back(Res.getValue(2));
7478 break;
7479 }
7480 case ISD::BITCAST: {
7481 if (useSoftFloat())
7482 return;
7483 SDLoc DL(N);
7484 SDValue Src = N->getOperand(0);
7485 EVT SrcVT = Src.getValueType();
7486 EVT ResVT = N->getValueType(0);
7487 if (ResVT == MVT::i128 && SrcVT == MVT::f128)
7488 Results.push_back(expandBitCastF128ToI128(DAG, Src, DL));
7489 else if (SrcVT == MVT::i16 && ResVT == MVT::f16) {
7490 if (Subtarget.hasVector()) {
7491 SDValue In32 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i32, Src);
7492 Results.push_back(SDValue(
7493 DAG.getMachineNode(SystemZ::LEFR_16, DL, MVT::f16, In32), 0));
7494 } else {
7495 SDValue In64 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Src);
7496 Results.push_back(convertToF16(In64, DAG));
7497 }
7498 } else if (SrcVT == MVT::f16 && ResVT == MVT::i16) {
7499 SDValue ExtractedVal =
7500 Subtarget.hasVector()
7501 ? SDValue(DAG.getMachineNode(SystemZ::LFER_16, DL, MVT::i32, Src),
7502 0)
7503 : convertFromF16(Src, DL, DAG);
7504 Results.push_back(DAG.getZExtOrTrunc(ExtractedVal, DL, ResVT));
7505 }
7506 break;
7507 }
7508 case ISD::UINT_TO_FP:
7509 case ISD::SINT_TO_FP:
7512 if (useSoftFloat())
7513 return;
7514 bool IsStrict = N->isStrictFPOpcode();
7515 SDLoc DL(N);
7516 SDValue InOp = N->getOperand(IsStrict ? 1 : 0);
7517 EVT ResVT = N->getValueType(0);
7518 SDValue Chain = IsStrict ? N->getOperand(0) : DAG.getEntryNode();
7519 if (ResVT == MVT::f16) {
7520 if (!IsStrict) {
7521 SDValue OpF32 = DAG.getNode(N->getOpcode(), DL, MVT::f32, InOp);
7522 Results.push_back(DAG.getFPExtendOrRound(OpF32, DL, MVT::f16));
7523 } else {
7524 SDValue OpF32 =
7525 DAG.getNode(N->getOpcode(), DL, DAG.getVTList(MVT::f32, MVT::Other),
7526 {Chain, InOp});
7527 SDValue F16Res;
7528 std::tie(F16Res, Chain) = DAG.getStrictFPExtendOrRound(
7529 OpF32, OpF32.getValue(1), DL, MVT::f16);
7530 Results.push_back(F16Res);
7531 Results.push_back(Chain);
7532 }
7533 }
7534 break;
7535 }
7536 case ISD::FP_TO_UINT:
7537 case ISD::FP_TO_SINT:
7540 if (useSoftFloat())
7541 return;
7542 bool IsStrict = N->isStrictFPOpcode();
7543 SDLoc DL(N);
7544 EVT ResVT = N->getValueType(0);
7545 SDValue InOp = N->getOperand(IsStrict ? 1 : 0);
7546 EVT InVT = InOp->getValueType(0);
7547 SDValue Chain = IsStrict ? N->getOperand(0) : DAG.getEntryNode();
7548 if (InVT == MVT::f16) {
7549 if (!IsStrict) {
7550 SDValue InF32 = DAG.getFPExtendOrRound(InOp, DL, MVT::f32);
7551 Results.push_back(DAG.getNode(N->getOpcode(), DL, ResVT, InF32));
7552 } else {
7553 SDValue InF32;
7554 std::tie(InF32, Chain) =
7555 DAG.getStrictFPExtendOrRound(InOp, Chain, DL, MVT::f32);
7556 SDValue OpF32 =
7557 DAG.getNode(N->getOpcode(), DL, DAG.getVTList(ResVT, MVT::Other),
7558 {Chain, InF32});
7559 Results.push_back(OpF32);
7560 Results.push_back(OpF32.getValue(1));
7561 }
7562 }
7563 break;
7564 }
7565 default:
7566 llvm_unreachable("Unexpected node to lower");
7567 }
7568}
7569
7570void
7576
7577// Return true if VT is a vector whose elements are a whole number of bytes
7578// in width. Also check for presence of vector support.
7579bool SystemZTargetLowering::canTreatAsByteVector(EVT VT) const {
7580 if (!Subtarget.hasVector())
7581 return false;
7582
7583 return VT.isVector() && VT.getScalarSizeInBits() % 8 == 0 && VT.isSimple();
7584}
7585
7586// Try to simplify an EXTRACT_VECTOR_ELT from a vector of type VecVT
7587// producing a result of type ResVT. Op is a possibly bitcast version
7588// of the input vector and Index is the index (based on type VecVT) that
7589// should be extracted. Return the new extraction if a simplification
7590// was possible or if Force is true.
7591SDValue SystemZTargetLowering::combineExtract(const SDLoc &DL, EVT ResVT,
7592 EVT VecVT, SDValue Op,
7593 unsigned Index,
7594 DAGCombinerInfo &DCI,
7595 bool Force) const {
7596 SelectionDAG &DAG = DCI.DAG;
7597
7598 // The number of bytes being extracted.
7599 unsigned BytesPerElement = VecVT.getVectorElementType().getStoreSize();
7600
7601 for (;;) {
7602 unsigned Opcode = Op.getOpcode();
7603 if (Opcode == ISD::BITCAST)
7604 // Look through bitcasts.
7605 Op = Op.getOperand(0);
7606 else if ((Opcode == ISD::VECTOR_SHUFFLE || Opcode == SystemZISD::SPLAT) &&
7607 canTreatAsByteVector(Op.getValueType())) {
7608 // Get a VPERM-like permute mask and see whether the bytes covered
7609 // by the extracted element are a contiguous sequence from one
7610 // source operand.
7612 if (!getVPermMask(Op, Bytes))
7613 break;
7614 int First;
7615 if (!getShuffleInput(Bytes, Index * BytesPerElement,
7616 BytesPerElement, First))
7617 break;
7618 if (First < 0)
7619 return DAG.getUNDEF(ResVT);
7620 // Make sure the contiguous sequence starts at a multiple of the
7621 // original element size.
7622 unsigned Byte = unsigned(First) % Bytes.size();
7623 if (Byte % BytesPerElement != 0)
7624 break;
7625 // We can get the extracted value directly from an input.
7626 Index = Byte / BytesPerElement;
7627 Op = Op.getOperand(unsigned(First) / Bytes.size());
7628 Force = true;
7629 } else if (Opcode == ISD::BUILD_VECTOR &&
7630 canTreatAsByteVector(Op.getValueType())) {
7631 // We can only optimize this case if the BUILD_VECTOR elements are
7632 // at least as wide as the extracted value.
7633 EVT OpVT = Op.getValueType();
7634 unsigned OpBytesPerElement = OpVT.getVectorElementType().getStoreSize();
7635 if (OpBytesPerElement < BytesPerElement)
7636 break;
7637 // Make sure that the least-significant bit of the extracted value
7638 // is the least significant bit of an input.
7639 unsigned End = (Index + 1) * BytesPerElement;
7640 if (End % OpBytesPerElement != 0)
7641 break;
7642 // We're extracting the low part of one operand of the BUILD_VECTOR.
7643 Op = Op.getOperand(End / OpBytesPerElement - 1);
7644 EVT ResIntVT = MVT::getIntegerVT(ResVT.getSizeInBits());
7645 if (!isTypeLegal(ResIntVT))
7646 break;
7647 if (!Op.getValueType().isInteger()) {
7648 EVT OpIntVT = MVT::getIntegerVT(Op.getValueSizeInBits());
7649 if (!isTypeLegal(OpIntVT))
7650 break;
7651 Op = DAG.getNode(ISD::BITCAST, DL, OpIntVT, Op);
7652 DCI.AddToWorklist(Op.getNode());
7653 }
7654 Op = DAG.getNode(ISD::TRUNCATE, DL, ResIntVT, Op);
7655 if (ResIntVT != ResVT) {
7656 DCI.AddToWorklist(Op.getNode());
7657 Op = DAG.getNode(ISD::BITCAST, DL, ResVT, Op);
7658 }
7659 return Op;
7660 } else if ((Opcode == ISD::SIGN_EXTEND_VECTOR_INREG ||
7662 Opcode == ISD::ANY_EXTEND_VECTOR_INREG) &&
7663 canTreatAsByteVector(Op.getValueType()) &&
7664 canTreatAsByteVector(Op.getOperand(0).getValueType())) {
7665 // Make sure that only the unextended bits are significant.
7666 EVT ExtVT = Op.getValueType();
7667 EVT OpVT = Op.getOperand(0).getValueType();
7668 unsigned ExtBytesPerElement = ExtVT.getVectorElementType().getStoreSize();
7669 unsigned OpBytesPerElement = OpVT.getVectorElementType().getStoreSize();
7670 unsigned Byte = Index * BytesPerElement;
7671 unsigned SubByte = Byte % ExtBytesPerElement;
7672 unsigned MinSubByte = ExtBytesPerElement - OpBytesPerElement;
7673 if (SubByte < MinSubByte ||
7674 SubByte + BytesPerElement > ExtBytesPerElement)
7675 break;
7676 // Get the byte offset of the unextended element
7677 Byte = Byte / ExtBytesPerElement * OpBytesPerElement;
7678 // ...then add the byte offset relative to that element.
7679 Byte += SubByte - MinSubByte;
7680 if (Byte % BytesPerElement != 0)
7681 break;
7682 Op = Op.getOperand(0);
7683 Index = Byte / BytesPerElement;
7684 Force = true;
7685 } else
7686 break;
7687 }
7688 if (Force) {
7689 if (Op.getValueType() != VecVT) {
7690 Op = DAG.getNode(ISD::BITCAST, DL, VecVT, Op);
7691 DCI.AddToWorklist(Op.getNode());
7692 }
7693 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ResVT, Op,
7694 DAG.getConstant(Index, DL, MVT::i32));
7695 }
7696 return SDValue();
7697}
7698
7699// Optimize vector operations in scalar value Op on the basis that Op
7700// is truncated to TruncVT.
7701SDValue SystemZTargetLowering::combineTruncateExtract(
7702 const SDLoc &DL, EVT TruncVT, SDValue Op, DAGCombinerInfo &DCI) const {
7703 // If we have (trunc (extract_vector_elt X, Y)), try to turn it into
7704 // (extract_vector_elt (bitcast X), Y'), where (bitcast X) has elements
7705 // of type TruncVT.
7706 if (Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
7707 TruncVT.getSizeInBits() % 8 == 0) {
7708 SDValue Vec = Op.getOperand(0);
7709 EVT VecVT = Vec.getValueType();
7710 if (canTreatAsByteVector(VecVT)) {
7711 if (auto *IndexN = dyn_cast<ConstantSDNode>(Op.getOperand(1))) {
7712 unsigned BytesPerElement = VecVT.getVectorElementType().getStoreSize();
7713 unsigned TruncBytes = TruncVT.getStoreSize();
7714 if (BytesPerElement % TruncBytes == 0) {
7715 // Calculate the value of Y' in the above description. We are
7716 // splitting the original elements into Scale equal-sized pieces
7717 // and for truncation purposes want the last (least-significant)
7718 // of these pieces for IndexN. This is easiest to do by calculating
7719 // the start index of the following element and then subtracting 1.
7720 unsigned Scale = BytesPerElement / TruncBytes;
7721 unsigned NewIndex = (IndexN->getZExtValue() + 1) * Scale - 1;
7722
7723 // Defer the creation of the bitcast from X to combineExtract,
7724 // which might be able to optimize the extraction.
7725 VecVT = EVT::getVectorVT(*DCI.DAG.getContext(),
7726 MVT::getIntegerVT(TruncBytes * 8),
7727 VecVT.getStoreSize() / TruncBytes);
7728 EVT ResVT = (TruncBytes < 4 ? MVT::i32 : TruncVT);
7729 return combineExtract(DL, ResVT, VecVT, Vec, NewIndex, DCI, true);
7730 }
7731 }
7732 }
7733 }
7734 return SDValue();
7735}
7736
7737SDValue SystemZTargetLowering::combineZERO_EXTEND(
7738 SDNode *N, DAGCombinerInfo &DCI) const {
7739 // Convert (zext (select_ccmask C1, C2)) into (select_ccmask C1', C2')
7740 SelectionDAG &DAG = DCI.DAG;
7741 SDValue N0 = N->getOperand(0);
7742 EVT VT = N->getValueType(0);
7743 if (N0.getOpcode() == SystemZISD::SELECT_CCMASK) {
7744 auto *TrueOp = dyn_cast<ConstantSDNode>(N0.getOperand(0));
7745 auto *FalseOp = dyn_cast<ConstantSDNode>(N0.getOperand(1));
7746 if (TrueOp && FalseOp) {
7747 SDLoc DL(N0);
7748 SDValue Ops[] = { DAG.getConstant(TrueOp->getZExtValue(), DL, VT),
7749 DAG.getConstant(FalseOp->getZExtValue(), DL, VT),
7750 N0.getOperand(2), N0.getOperand(3), N0.getOperand(4) };
7751 SDValue NewSelect = DAG.getNode(SystemZISD::SELECT_CCMASK, DL, VT, Ops);
7752 // If N0 has multiple uses, change other uses as well.
7753 if (!N0.hasOneUse()) {
7754 SDValue TruncSelect =
7755 DAG.getNode(ISD::TRUNCATE, DL, N0.getValueType(), NewSelect);
7756 DCI.CombineTo(N0.getNode(), TruncSelect);
7757 }
7758 return NewSelect;
7759 }
7760 }
7761 // Convert (zext (xor (trunc X), C)) into (xor (trunc X), C') if the size
7762 // of the result is smaller than the size of X and all the truncated bits
7763 // of X are already zero.
7764 if (N0.getOpcode() == ISD::XOR &&
7765 N0.hasOneUse() && N0.getOperand(0).hasOneUse() &&
7766 N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
7767 N0.getOperand(1).getOpcode() == ISD::Constant) {
7768 SDValue X = N0.getOperand(0).getOperand(0);
7769 if (VT.isScalarInteger() && VT.getSizeInBits() < X.getValueSizeInBits()) {
7770 KnownBits Known = DAG.computeKnownBits(X);
7771 APInt TruncatedBits = APInt::getBitsSet(X.getValueSizeInBits(),
7772 N0.getValueSizeInBits(),
7773 VT.getSizeInBits());
7774 if (TruncatedBits.isSubsetOf(Known.Zero)) {
7775 X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
7776 APInt Mask = N0.getConstantOperandAPInt(1).zext(VT.getSizeInBits());
7777 return DAG.getNode(ISD::XOR, SDLoc(N0), VT,
7778 X, DAG.getConstant(Mask, SDLoc(N0), VT));
7779 }
7780 }
7781 }
7782 // Recognize patterns for VECTOR SUBTRACT COMPUTE BORROW INDICATION
7783 // and VECTOR ADD COMPUTE CARRY for i128:
7784 // (zext (setcc_uge X Y)) --> (VSCBI X Y)
7785 // (zext (setcc_ule Y X)) --> (VSCBI X Y)
7786 // (zext (setcc_ult (add X Y) X/Y) -> (VACC X Y)
7787 // (zext (setcc_ugt X/Y (add X Y)) -> (VACC X Y)
7788 // For vector types, these patterns are recognized in the .td file.
7789 if (N0.getOpcode() == ISD::SETCC && isTypeLegal(VT) && VT == MVT::i128 &&
7790 N0.getOperand(0).getValueType() == VT) {
7791 SDValue Op0 = N0.getOperand(0);
7792 SDValue Op1 = N0.getOperand(1);
7793 const ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
7794 switch (CC) {
7795 case ISD::SETULE:
7796 std::swap(Op0, Op1);
7797 [[fallthrough]];
7798 case ISD::SETUGE:
7799 return DAG.getNode(SystemZISD::VSCBI, SDLoc(N0), VT, Op0, Op1);
7800 case ISD::SETUGT:
7801 std::swap(Op0, Op1);
7802 [[fallthrough]];
7803 case ISD::SETULT:
7804 if (Op0->hasOneUse() && Op0->getOpcode() == ISD::ADD &&
7805 (Op0->getOperand(0) == Op1 || Op0->getOperand(1) == Op1))
7806 return DAG.getNode(SystemZISD::VACC, SDLoc(N0), VT, Op0->getOperand(0),
7807 Op0->getOperand(1));
7808 break;
7809 default:
7810 break;
7811 }
7812 }
7813
7814 return SDValue();
7815}
7816
7817SDValue SystemZTargetLowering::combineSIGN_EXTEND_INREG(
7818 SDNode *N, DAGCombinerInfo &DCI) const {
7819 // Convert (sext_in_reg (setcc LHS, RHS, COND), i1)
7820 // and (sext_in_reg (any_extend (setcc LHS, RHS, COND)), i1)
7821 // into (select_cc LHS, RHS, -1, 0, COND)
7822 SelectionDAG &DAG = DCI.DAG;
7823 SDValue N0 = N->getOperand(0);
7824 EVT VT = N->getValueType(0);
7825 EVT EVT = cast<VTSDNode>(N->getOperand(1))->getVT();
7826 if (N0.hasOneUse() && N0.getOpcode() == ISD::ANY_EXTEND)
7827 N0 = N0.getOperand(0);
7828 if (EVT == MVT::i1 && N0.hasOneUse() && N0.getOpcode() == ISD::SETCC) {
7829 SDLoc DL(N0);
7830 SDValue Ops[] = { N0.getOperand(0), N0.getOperand(1),
7831 DAG.getAllOnesConstant(DL, VT),
7832 DAG.getConstant(0, DL, VT), N0.getOperand(2) };
7833 return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
7834 }
7835 return SDValue();
7836}
7837
7838SDValue SystemZTargetLowering::combineSIGN_EXTEND(
7839 SDNode *N, DAGCombinerInfo &DCI) const {
7840 // Convert (sext (ashr (shl X, C1), C2)) to
7841 // (ashr (shl (anyext X), C1'), C2')), since wider shifts are as
7842 // cheap as narrower ones.
7843 SelectionDAG &DAG = DCI.DAG;
7844 SDValue N0 = N->getOperand(0);
7845 EVT VT = N->getValueType(0);
7846 if (N0.hasOneUse() && N0.getOpcode() == ISD::SRA) {
7847 auto *SraAmt = dyn_cast<ConstantSDNode>(N0.getOperand(1));
7848 SDValue Inner = N0.getOperand(0);
7849 if (SraAmt && Inner.hasOneUse() && Inner.getOpcode() == ISD::SHL) {
7850 if (auto *ShlAmt = dyn_cast<ConstantSDNode>(Inner.getOperand(1))) {
7851 unsigned Extra = (VT.getSizeInBits() - N0.getValueSizeInBits());
7852 unsigned NewShlAmt = ShlAmt->getZExtValue() + Extra;
7853 unsigned NewSraAmt = SraAmt->getZExtValue() + Extra;
7854 EVT ShiftVT = N0.getOperand(1).getValueType();
7855 SDValue Ext = DAG.getNode(ISD::ANY_EXTEND, SDLoc(Inner), VT,
7856 Inner.getOperand(0));
7857 SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(Inner), VT, Ext,
7858 DAG.getConstant(NewShlAmt, SDLoc(Inner),
7859 ShiftVT));
7860 return DAG.getNode(ISD::SRA, SDLoc(N0), VT, Shl,
7861 DAG.getConstant(NewSraAmt, SDLoc(N0), ShiftVT));
7862 }
7863 }
7864 }
7865
7866 return SDValue();
7867}
7868
7869SDValue SystemZTargetLowering::combineMERGE(
7870 SDNode *N, DAGCombinerInfo &DCI) const {
7871 SelectionDAG &DAG = DCI.DAG;
7872 unsigned Opcode = N->getOpcode();
7873 SDValue Op0 = N->getOperand(0);
7874 SDValue Op1 = N->getOperand(1);
7875 if (Op0.getOpcode() == ISD::BITCAST)
7876 Op0 = Op0.getOperand(0);
7878 // (z_merge_* 0, 0) -> 0. This is mostly useful for using VLLEZF
7879 // for v4f32.
7880 if (Op1 == N->getOperand(0))
7881 return Op1;
7882 // (z_merge_? 0, X) -> (z_unpackl_? 0, X).
7883 EVT VT = Op1.getValueType();
7884 unsigned ElemBytes = VT.getVectorElementType().getStoreSize();
7885 if (ElemBytes <= 4) {
7886 Opcode = (Opcode == SystemZISD::MERGE_HIGH ?
7887 SystemZISD::UNPACKL_HIGH : SystemZISD::UNPACKL_LOW);
7888 EVT InVT = VT.changeVectorElementTypeToInteger();
7889 EVT OutVT = MVT::getVectorVT(MVT::getIntegerVT(ElemBytes * 16),
7890 SystemZ::VectorBytes / ElemBytes / 2);
7891 if (VT != InVT) {
7892 Op1 = DAG.getNode(ISD::BITCAST, SDLoc(N), InVT, Op1);
7893 DCI.AddToWorklist(Op1.getNode());
7894 }
7895 SDValue Op = DAG.getNode(Opcode, SDLoc(N), OutVT, Op1);
7896 DCI.AddToWorklist(Op.getNode());
7897 return DAG.getNode(ISD::BITCAST, SDLoc(N), VT, Op);
7898 }
7899 }
7900 return SDValue();
7901}
7902
7903static bool isI128MovedToParts(LoadSDNode *LD, SDNode *&LoPart,
7904 SDNode *&HiPart) {
7905 LoPart = HiPart = nullptr;
7906
7907 // Scan through all users.
7908 for (SDUse &Use : LD->uses()) {
7909 // Skip the uses of the chain.
7910 if (Use.getResNo() != 0)
7911 continue;
7912
7913 // Verify every user is a TRUNCATE to i64 of the low or high half.
7914 SDNode *User = Use.getUser();
7915 bool IsLoPart = true;
7916 if (User->getOpcode() == ISD::SRL &&
7917 User->getOperand(1).getOpcode() == ISD::Constant &&
7918 User->getConstantOperandVal(1) == 64 && User->hasOneUse()) {
7919 User = *User->user_begin();
7920 IsLoPart = false;
7921 }
7922 if (User->getOpcode() != ISD::TRUNCATE || User->getValueType(0) != MVT::i64)
7923 return false;
7924
7925 if (IsLoPart) {
7926 if (LoPart)
7927 return false;
7928 LoPart = User;
7929 } else {
7930 if (HiPart)
7931 return false;
7932 HiPart = User;
7933 }
7934 }
7935 return true;
7936}
7937
7938static bool isF128MovedToParts(LoadSDNode *LD, SDNode *&LoPart,
7939 SDNode *&HiPart) {
7940 LoPart = HiPart = nullptr;
7941
7942 // Scan through all users.
7943 for (SDUse &Use : LD->uses()) {
7944 // Skip the uses of the chain.
7945 if (Use.getResNo() != 0)
7946 continue;
7947
7948 // Verify every user is an EXTRACT_SUBREG of the low or high half.
7949 SDNode *User = Use.getUser();
7950 if (!User->hasOneUse() || !User->isMachineOpcode() ||
7951 User->getMachineOpcode() != TargetOpcode::EXTRACT_SUBREG)
7952 return false;
7953
7954 switch (User->getConstantOperandVal(1)) {
7955 case SystemZ::subreg_l64:
7956 if (LoPart)
7957 return false;
7958 LoPart = User;
7959 break;
7960 case SystemZ::subreg_h64:
7961 if (HiPart)
7962 return false;
7963 HiPart = User;
7964 break;
7965 default:
7966 return false;
7967 }
7968 }
7969 return true;
7970}
7971
7972SDValue SystemZTargetLowering::combineLOAD(
7973 SDNode *N, DAGCombinerInfo &DCI) const {
7974 SelectionDAG &DAG = DCI.DAG;
7975 EVT LdVT = N->getValueType(0);
7976 if (auto *LN = dyn_cast<LoadSDNode>(N)) {
7977 if (LN->getAddressSpace() == SYSTEMZAS::PTR32) {
7978 MVT PtrVT = getPointerTy(DAG.getDataLayout());
7979 MVT LoadNodeVT = LN->getBasePtr().getSimpleValueType();
7980 if (PtrVT != LoadNodeVT) {
7981 SDLoc DL(LN);
7982 SDValue AddrSpaceCast = DAG.getAddrSpaceCast(
7983 DL, PtrVT, LN->getBasePtr(), SYSTEMZAS::PTR32, 0);
7984 return DAG.getExtLoad(LN->getExtensionType(), DL, LN->getValueType(0),
7985 LN->getChain(), AddrSpaceCast, LN->getMemoryVT(),
7986 LN->getMemOperand());
7987 }
7988 }
7989 }
7990 SDLoc DL(N);
7991
7992 // Replace a 128-bit load that is used solely to move its value into GPRs
7993 // by separate loads of both halves.
7994 LoadSDNode *LD = cast<LoadSDNode>(N);
7995 if (LD->isSimple() && ISD::isNormalLoad(LD)) {
7996 SDNode *LoPart, *HiPart;
7997 if ((LdVT == MVT::i128 && isI128MovedToParts(LD, LoPart, HiPart)) ||
7998 (LdVT == MVT::f128 && isF128MovedToParts(LD, LoPart, HiPart))) {
7999 // Rewrite each extraction as an independent load.
8000 SmallVector<SDValue, 2> ArgChains;
8001 if (HiPart) {
8002 SDValue EltLoad = DAG.getLoad(
8003 HiPart->getValueType(0), DL, LD->getChain(), LD->getBasePtr(),
8004 LD->getPointerInfo(), LD->getBaseAlign(),
8005 LD->getMemOperand()->getFlags(), LD->getAAInfo());
8006
8007 DCI.CombineTo(HiPart, EltLoad, true);
8008 ArgChains.push_back(EltLoad.getValue(1));
8009 }
8010 if (LoPart) {
8011 SDValue EltLoad = DAG.getLoad(
8012 LoPart->getValueType(0), DL, LD->getChain(),
8013 DAG.getObjectPtrOffset(DL, LD->getBasePtr(), TypeSize::getFixed(8)),
8014 LD->getPointerInfo().getWithOffset(8), LD->getBaseAlign(),
8015 LD->getMemOperand()->getFlags(), LD->getAAInfo());
8016
8017 DCI.CombineTo(LoPart, EltLoad, true);
8018 ArgChains.push_back(EltLoad.getValue(1));
8019 }
8020
8021 // Collect all chains via TokenFactor.
8022 SDValue Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, ArgChains);
8023 DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
8024 DCI.AddToWorklist(Chain.getNode());
8025 return SDValue(N, 0);
8026 }
8027 }
8028
8029 if (LdVT.isVector() || LdVT.isInteger())
8030 return SDValue();
8031 // Transform a scalar load that is REPLICATEd as well as having other
8032 // use(s) to the form where the other use(s) use the first element of the
8033 // REPLICATE instead of the load. Otherwise instruction selection will not
8034 // produce a VLREP. Avoid extracting to a GPR, so only do this for floating
8035 // point loads.
8036
8037 SDValue Replicate;
8038 SmallVector<SDNode*, 8> OtherUses;
8039 for (SDUse &Use : N->uses()) {
8040 if (Use.getUser()->getOpcode() == SystemZISD::REPLICATE) {
8041 if (Replicate)
8042 return SDValue(); // Should never happen
8043 Replicate = SDValue(Use.getUser(), 0);
8044 } else if (Use.getResNo() == 0)
8045 OtherUses.push_back(Use.getUser());
8046 }
8047 if (!Replicate || OtherUses.empty())
8048 return SDValue();
8049
8050 SDValue Extract0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, LdVT,
8051 Replicate, DAG.getConstant(0, DL, MVT::i32));
8052 // Update uses of the loaded Value while preserving old chains.
8053 for (SDNode *U : OtherUses) {
8055 for (SDValue Op : U->ops())
8056 Ops.push_back((Op.getNode() == N && Op.getResNo() == 0) ? Extract0 : Op);
8057 DAG.UpdateNodeOperands(U, Ops);
8058 }
8059 return SDValue(N, 0);
8060}
8061
8062bool SystemZTargetLowering::canLoadStoreByteSwapped(EVT VT) const {
8063 if (VT == MVT::i16 || VT == MVT::i32 || VT == MVT::i64)
8064 return true;
8065 if (Subtarget.hasVectorEnhancements2())
8066 if (VT == MVT::v8i16 || VT == MVT::v4i32 || VT == MVT::v2i64 || VT == MVT::i128)
8067 return true;
8068 return false;
8069}
8070
8072 if (!VT.isVector() || !VT.isSimple() ||
8073 VT.getSizeInBits() != 128 ||
8074 VT.getScalarSizeInBits() % 8 != 0)
8075 return false;
8076
8077 unsigned NumElts = VT.getVectorNumElements();
8078 for (unsigned i = 0; i < NumElts; ++i) {
8079 if (M[i] < 0) continue; // ignore UNDEF indices
8080 if ((unsigned) M[i] != NumElts - 1 - i)
8081 return false;
8082 }
8083
8084 return true;
8085}
8086
8087static bool isOnlyUsedByStores(SDValue StoredVal, SelectionDAG &DAG) {
8088 for (auto *U : StoredVal->users()) {
8089 if (StoreSDNode *ST = dyn_cast<StoreSDNode>(U)) {
8090 EVT CurrMemVT = ST->getMemoryVT().getScalarType();
8091 if (CurrMemVT.isRound() && CurrMemVT.getStoreSize() <= 16)
8092 continue;
8093 } else if (isa<BuildVectorSDNode>(U)) {
8094 SDValue BuildVector = SDValue(U, 0);
8095 if (DAG.isSplatValue(BuildVector, true/*AllowUndefs*/) &&
8096 isOnlyUsedByStores(BuildVector, DAG))
8097 continue;
8098 }
8099 return false;
8100 }
8101 return true;
8102}
8103
8104static bool isI128MovedFromParts(SDValue Val, SDValue &LoPart,
8105 SDValue &HiPart) {
8106 if (Val.getOpcode() != ISD::OR || !Val.getNode()->hasOneUse())
8107 return false;
8108
8109 SDValue Op0 = Val.getOperand(0);
8110 SDValue Op1 = Val.getOperand(1);
8111
8112 if (Op0.getOpcode() == ISD::SHL)
8113 std::swap(Op0, Op1);
8114 if (Op1.getOpcode() != ISD::SHL || !Op1.getNode()->hasOneUse() ||
8115 Op1.getOperand(1).getOpcode() != ISD::Constant ||
8116 Op1.getConstantOperandVal(1) != 64)
8117 return false;
8118 Op1 = Op1.getOperand(0);
8119
8120 if (Op0.getOpcode() != ISD::ZERO_EXTEND || !Op0.getNode()->hasOneUse() ||
8121 Op0.getOperand(0).getValueType() != MVT::i64)
8122 return false;
8123 if (Op1.getOpcode() != ISD::ANY_EXTEND || !Op1.getNode()->hasOneUse() ||
8124 Op1.getOperand(0).getValueType() != MVT::i64)
8125 return false;
8126
8127 LoPart = Op0.getOperand(0);
8128 HiPart = Op1.getOperand(0);
8129 return true;
8130}
8131
8132static bool isF128MovedFromParts(SDValue Val, SDValue &LoPart,
8133 SDValue &HiPart) {
8134 if (!Val.getNode()->hasOneUse() || !Val.isMachineOpcode() ||
8135 Val.getMachineOpcode() != TargetOpcode::REG_SEQUENCE)
8136 return false;
8137
8138 if (Val->getNumOperands() != 5 ||
8139 Val->getOperand(0)->getAsZExtVal() != SystemZ::FP128BitRegClassID ||
8140 Val->getOperand(2)->getAsZExtVal() != SystemZ::subreg_l64 ||
8141 Val->getOperand(4)->getAsZExtVal() != SystemZ::subreg_h64)
8142 return false;
8143
8144 LoPart = Val->getOperand(1);
8145 HiPart = Val->getOperand(3);
8146 return true;
8147}
8148
8149SDValue SystemZTargetLowering::combineSTORE(
8150 SDNode *N, DAGCombinerInfo &DCI) const {
8151 SelectionDAG &DAG = DCI.DAG;
8152 auto *SN = cast<StoreSDNode>(N);
8153 auto &Op1 = N->getOperand(1);
8154 EVT MemVT = SN->getMemoryVT();
8155
8156 if (SN->getAddressSpace() == SYSTEMZAS::PTR32) {
8157 MVT PtrVT = getPointerTy(DAG.getDataLayout());
8158 MVT StoreNodeVT = SN->getBasePtr().getSimpleValueType();
8159 if (PtrVT != StoreNodeVT) {
8160 SDLoc DL(SN);
8161 SDValue AddrSpaceCast = DAG.getAddrSpaceCast(DL, PtrVT, SN->getBasePtr(),
8162 SYSTEMZAS::PTR32, 0);
8163 return DAG.getStore(SN->getChain(), DL, SN->getValue(), AddrSpaceCast,
8164 SN->getPointerInfo(), SN->getBaseAlign(),
8165 SN->getMemOperand()->getFlags(), SN->getAAInfo());
8166 }
8167 }
8168
8169 // If we have (truncstoreiN (extract_vector_elt X, Y), Z) then it is better
8170 // for the extraction to be done on a vMiN value, so that we can use VSTE.
8171 // If X has wider elements then convert it to:
8172 // (truncstoreiN (extract_vector_elt (bitcast X), Y2), Z).
8173 if (MemVT.isInteger() && SN->isTruncatingStore()) {
8174 if (SDValue Value =
8175 combineTruncateExtract(SDLoc(N), MemVT, SN->getValue(), DCI)) {
8176 DCI.AddToWorklist(Value.getNode());
8177
8178 // Rewrite the store with the new form of stored value.
8179 return DAG.getTruncStore(SN->getChain(), SDLoc(SN), Value,
8180 SN->getBasePtr(), SN->getMemoryVT(),
8181 SN->getMemOperand());
8182 }
8183 }
8184
8185 // combine STORE (LOAD_STACK_GUARD) into MOV_STACKGUARD_DAG
8186 if (Op1->isMachineOpcode() &&
8187 (Op1->getMachineOpcode() == SystemZ::LOAD_STACK_GUARD)) {
8188 // Obtain the frame index the store was targeting.
8189 int FI = cast<FrameIndexSDNode>(SN->getOperand(2))->getIndex();
8190 // Prepare operands of the MOV_STACKGUARD ISD Node - Chain and FrameIndex.
8191 SDValue Ops[] = {SN->getChain(), DAG.getTargetFrameIndex(FI, MVT::i64)};
8192 return DAG.getNode(SystemZISD::MOV_STACKGUARD, SDLoc(SN), MVT::Other, Ops);
8193 }
8194
8195 // Combine STORE (BSWAP) into STRVH/STRV/STRVG/VSTBR
8196 if (!SN->isTruncatingStore() &&
8197 Op1.getOpcode() == ISD::BSWAP &&
8198 Op1.getNode()->hasOneUse() &&
8199 canLoadStoreByteSwapped(Op1.getValueType())) {
8200
8201 SDValue BSwapOp = Op1.getOperand(0);
8202
8203 if (BSwapOp.getValueType() == MVT::i16)
8204 BSwapOp = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), MVT::i32, BSwapOp);
8205
8206 SDValue Ops[] = {
8207 N->getOperand(0), BSwapOp, N->getOperand(2)
8208 };
8209
8210 return
8211 DAG.getMemIntrinsicNode(SystemZISD::STRV, SDLoc(N), DAG.getVTList(MVT::Other),
8212 Ops, MemVT, SN->getMemOperand());
8213 }
8214 // Combine STORE (element-swap) into VSTER
8215 if (!SN->isTruncatingStore() &&
8216 Op1.getOpcode() == ISD::VECTOR_SHUFFLE &&
8217 Op1.getNode()->hasOneUse() &&
8218 Subtarget.hasVectorEnhancements2()) {
8219 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(Op1.getNode());
8220 ArrayRef<int> ShuffleMask = SVN->getMask();
8221 if (isVectorElementSwap(ShuffleMask, Op1.getValueType())) {
8222 SDValue Ops[] = {
8223 N->getOperand(0), Op1.getOperand(0), N->getOperand(2)
8224 };
8225
8226 return DAG.getMemIntrinsicNode(SystemZISD::VSTER, SDLoc(N),
8227 DAG.getVTList(MVT::Other),
8228 Ops, MemVT, SN->getMemOperand());
8229 }
8230 }
8231
8232 // Combine STORE (READCYCLECOUNTER) into STCKF.
8233 if (!SN->isTruncatingStore() &&
8235 Op1.hasOneUse() &&
8236 N->getOperand(0).reachesChainWithoutSideEffects(SDValue(Op1.getNode(), 1))) {
8237 SDValue Ops[] = { Op1.getOperand(0), N->getOperand(2) };
8238 return DAG.getMemIntrinsicNode(SystemZISD::STCKF, SDLoc(N),
8239 DAG.getVTList(MVT::Other),
8240 Ops, MemVT, SN->getMemOperand());
8241 }
8242
8243 // Transform a store of a 128-bit value moved from parts into two stores.
8244 if (SN->isSimple() && ISD::isNormalStore(SN)) {
8245 SDValue LoPart, HiPart;
8246 if ((MemVT == MVT::i128 && isI128MovedFromParts(Op1, LoPart, HiPart)) ||
8247 (MemVT == MVT::f128 && isF128MovedFromParts(Op1, LoPart, HiPart))) {
8248 SDLoc DL(SN);
8249 SDValue Chain0 = DAG.getStore(
8250 SN->getChain(), DL, HiPart, SN->getBasePtr(), SN->getPointerInfo(),
8251 SN->getBaseAlign(), SN->getMemOperand()->getFlags(), SN->getAAInfo());
8252 SDValue Chain1 = DAG.getStore(
8253 SN->getChain(), DL, LoPart,
8254 DAG.getObjectPtrOffset(DL, SN->getBasePtr(), TypeSize::getFixed(8)),
8255 SN->getPointerInfo().getWithOffset(8), SN->getBaseAlign(),
8256 SN->getMemOperand()->getFlags(), SN->getAAInfo());
8257
8258 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chain0, Chain1);
8259 }
8260 }
8261
8262 // Replicate a reg or immediate with VREP instead of scalar multiply or
8263 // immediate load. It seems best to do this during the first DAGCombine as
8264 // it is straight-forward to handle the zero-extend node in the initial
8265 // DAG, and also not worry about the keeping the new MemVT legal (e.g. when
8266 // extracting an i16 element from a v16i8 vector).
8267 if (Subtarget.hasVector() && DCI.Level == BeforeLegalizeTypes &&
8268 isOnlyUsedByStores(Op1, DAG)) {
8269 SDValue Word = SDValue();
8270 EVT WordVT;
8271
8272 // Find a replicated immediate and return it if found in Word and its
8273 // type in WordVT.
8274 auto FindReplicatedImm = [&](ConstantSDNode *C, unsigned TotBytes) {
8275 // Some constants are better handled with a scalar store.
8276 if (C->getAPIntValue().getBitWidth() > 64 || C->isAllOnes() ||
8277 isInt<16>(C->getSExtValue()) || MemVT.getStoreSize() <= 2)
8278 return;
8279
8280 APInt Val = C->getAPIntValue();
8281 // Truncate Val in case of a truncating store.
8282 if (!llvm::isUIntN(TotBytes * 8, Val.getZExtValue())) {
8283 assert(SN->isTruncatingStore() &&
8284 "Non-truncating store and immediate value does not fit?");
8285 Val = Val.trunc(TotBytes * 8);
8286 }
8287
8288 SystemZVectorConstantInfo VCI(APInt(TotBytes * 8, Val.getZExtValue()));
8289 if (VCI.isVectorConstantLegal(Subtarget) &&
8290 VCI.Opcode == SystemZISD::REPLICATE) {
8291 Word = DAG.getConstant(VCI.OpVals[0], SDLoc(SN), MVT::i32);
8292 WordVT = VCI.VecVT.getScalarType();
8293 }
8294 };
8295
8296 // Find a replicated register and return it if found in Word and its type
8297 // in WordVT.
8298 auto FindReplicatedReg = [&](SDValue MulOp) {
8299 EVT MulVT = MulOp.getValueType();
8300 if (MulOp->getOpcode() == ISD::MUL &&
8301 (MulVT == MVT::i16 || MulVT == MVT::i32 || MulVT == MVT::i64)) {
8302 // Find a zero extended value and its type.
8303 SDValue LHS = MulOp->getOperand(0);
8304 if (LHS->getOpcode() == ISD::ZERO_EXTEND)
8305 WordVT = LHS->getOperand(0).getValueType();
8306 else if (LHS->getOpcode() == ISD::AssertZext)
8307 WordVT = cast<VTSDNode>(LHS->getOperand(1))->getVT();
8308 else
8309 return;
8310 // Find a replicating constant, e.g. 0x00010001.
8311 if (auto *C = dyn_cast<ConstantSDNode>(MulOp->getOperand(1))) {
8312 SystemZVectorConstantInfo VCI(
8313 APInt(MulVT.getSizeInBits(), C->getZExtValue()));
8314 if (VCI.isVectorConstantLegal(Subtarget) &&
8315 VCI.Opcode == SystemZISD::REPLICATE && VCI.OpVals[0] == 1 &&
8316 WordVT == VCI.VecVT.getScalarType())
8317 Word = DAG.getZExtOrTrunc(LHS->getOperand(0), SDLoc(SN), WordVT);
8318 }
8319 }
8320 };
8321
8322 if (isa<BuildVectorSDNode>(Op1) &&
8323 DAG.isSplatValue(Op1, true/*AllowUndefs*/)) {
8324 SDValue SplatVal = Op1->getOperand(0);
8325 if (auto *C = dyn_cast<ConstantSDNode>(SplatVal))
8326 FindReplicatedImm(C, SplatVal.getValueType().getStoreSize());
8327 else
8328 FindReplicatedReg(SplatVal);
8329 } else {
8330 if (auto *C = dyn_cast<ConstantSDNode>(Op1))
8331 FindReplicatedImm(C, MemVT.getStoreSize());
8332 else
8333 FindReplicatedReg(Op1);
8334 }
8335
8336 if (Word != SDValue()) {
8337 assert(MemVT.getSizeInBits() % WordVT.getSizeInBits() == 0 &&
8338 "Bad type handling");
8339 unsigned NumElts = MemVT.getSizeInBits() / WordVT.getSizeInBits();
8340 EVT SplatVT = EVT::getVectorVT(*DAG.getContext(), WordVT, NumElts);
8341 SDValue SplatVal = DAG.getSplatVector(SplatVT, SDLoc(SN), Word);
8342 return DAG.getStore(SN->getChain(), SDLoc(SN), SplatVal,
8343 SN->getBasePtr(), SN->getMemOperand());
8344 }
8345 }
8346
8347 return SDValue();
8348}
8349
8350SDValue SystemZTargetLowering::combineVECTOR_SHUFFLE(
8351 SDNode *N, DAGCombinerInfo &DCI) const {
8352 SelectionDAG &DAG = DCI.DAG;
8353 // Combine element-swap (LOAD) into VLER
8354 if (ISD::isNON_EXTLoad(N->getOperand(0).getNode()) &&
8355 N->getOperand(0).hasOneUse() &&
8356 Subtarget.hasVectorEnhancements2()) {
8357 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
8358 ArrayRef<int> ShuffleMask = SVN->getMask();
8359 if (isVectorElementSwap(ShuffleMask, N->getValueType(0))) {
8360 SDValue Load = N->getOperand(0);
8361 LoadSDNode *LD = cast<LoadSDNode>(Load);
8362
8363 // Create the element-swapping load.
8364 SDValue Ops[] = {
8365 LD->getChain(), // Chain
8366 LD->getBasePtr() // Ptr
8367 };
8368 SDValue ESLoad =
8369 DAG.getMemIntrinsicNode(SystemZISD::VLER, SDLoc(N),
8370 DAG.getVTList(LD->getValueType(0), MVT::Other),
8371 Ops, LD->getMemoryVT(), LD->getMemOperand());
8372
8373 // First, combine the VECTOR_SHUFFLE away. This makes the value produced
8374 // by the load dead.
8375 DCI.CombineTo(N, ESLoad);
8376
8377 // Next, combine the load away, we give it a bogus result value but a real
8378 // chain result. The result value is dead because the shuffle is dead.
8379 DCI.CombineTo(Load.getNode(), ESLoad, ESLoad.getValue(1));
8380
8381 // Return N so it doesn't get rechecked!
8382 return SDValue(N, 0);
8383 }
8384 }
8385
8386 return SDValue();
8387}
8388
8389SDValue SystemZTargetLowering::combineEXTRACT_VECTOR_ELT(
8390 SDNode *N, DAGCombinerInfo &DCI) const {
8391 SelectionDAG &DAG = DCI.DAG;
8392
8393 if (!Subtarget.hasVector())
8394 return SDValue();
8395
8396 // Look through bitcasts that retain the number of vector elements.
8397 SDValue Op = N->getOperand(0);
8398 if (Op.getOpcode() == ISD::BITCAST &&
8399 Op.getValueType().isVector() &&
8400 Op.getOperand(0).getValueType().isVector() &&
8401 Op.getValueType().getVectorNumElements() ==
8402 Op.getOperand(0).getValueType().getVectorNumElements())
8403 Op = Op.getOperand(0);
8404
8405 // Pull BSWAP out of a vector extraction.
8406 if (Op.getOpcode() == ISD::BSWAP && Op.hasOneUse()) {
8407 EVT VecVT = Op.getValueType();
8408 EVT EltVT = VecVT.getVectorElementType();
8409 Op = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(N), EltVT,
8410 Op.getOperand(0), N->getOperand(1));
8411 DCI.AddToWorklist(Op.getNode());
8412 Op = DAG.getNode(ISD::BSWAP, SDLoc(N), EltVT, Op);
8413 if (EltVT != N->getValueType(0)) {
8414 DCI.AddToWorklist(Op.getNode());
8415 Op = DAG.getNode(ISD::BITCAST, SDLoc(N), N->getValueType(0), Op);
8416 }
8417 return Op;
8418 }
8419
8420 // Try to simplify a vector extraction.
8421 if (auto *IndexN = dyn_cast<ConstantSDNode>(N->getOperand(1))) {
8422 SDValue Op0 = N->getOperand(0);
8423 EVT VecVT = Op0.getValueType();
8424 if (canTreatAsByteVector(VecVT))
8425 return combineExtract(SDLoc(N), N->getValueType(0), VecVT, Op0,
8426 IndexN->getZExtValue(), DCI, false);
8427 }
8428 return SDValue();
8429}
8430
8431SDValue SystemZTargetLowering::combineJOIN_DWORDS(
8432 SDNode *N, DAGCombinerInfo &DCI) const {
8433 SelectionDAG &DAG = DCI.DAG;
8434 // (join_dwords X, X) == (replicate X)
8435 if (N->getOperand(0) == N->getOperand(1))
8436 return DAG.getNode(SystemZISD::REPLICATE, SDLoc(N), N->getValueType(0),
8437 N->getOperand(0));
8438 return SDValue();
8439}
8440
8442 SDValue Chain1 = N1->getOperand(0);
8443 SDValue Chain2 = N2->getOperand(0);
8444
8445 // Trivial case: both nodes take the same chain.
8446 if (Chain1 == Chain2)
8447 return Chain1;
8448
8449 // FIXME - we could handle more complex cases via TokenFactor,
8450 // assuming we can verify that this would not create a cycle.
8451 return SDValue();
8452}
8453
8454SDValue SystemZTargetLowering::combineFP_ROUND(
8455 SDNode *N, DAGCombinerInfo &DCI) const {
8456
8457 if (!Subtarget.hasVector())
8458 return SDValue();
8459
8460 // (fpround (extract_vector_elt X 0))
8461 // (fpround (extract_vector_elt X 1)) ->
8462 // (extract_vector_elt (VROUND X) 0)
8463 // (extract_vector_elt (VROUND X) 2)
8464 //
8465 // This is a special case since the target doesn't really support v2f32s.
8466 unsigned OpNo = N->isStrictFPOpcode() ? 1 : 0;
8467 SelectionDAG &DAG = DCI.DAG;
8468 SDValue Op0 = N->getOperand(OpNo);
8469 if (N->getValueType(0) == MVT::f32 && Op0.hasOneUse() &&
8471 Op0.getOperand(0).getValueType() == MVT::v2f64 &&
8472 Op0.getOperand(1).getOpcode() == ISD::Constant &&
8473 Op0.getConstantOperandVal(1) == 0) {
8474 SDValue Vec = Op0.getOperand(0);
8475 for (auto *U : Vec->users()) {
8476 if (U != Op0.getNode() && U->hasOneUse() &&
8477 U->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
8478 U->getOperand(0) == Vec &&
8479 U->getOperand(1).getOpcode() == ISD::Constant &&
8480 U->getConstantOperandVal(1) == 1) {
8481 SDValue OtherRound = SDValue(*U->user_begin(), 0);
8482 if (OtherRound.getOpcode() == N->getOpcode() &&
8483 OtherRound.getOperand(OpNo) == SDValue(U, 0) &&
8484 OtherRound.getValueType() == MVT::f32) {
8485 SDValue VRound, Chain;
8486 if (N->isStrictFPOpcode()) {
8487 Chain = MergeInputChains(N, OtherRound.getNode());
8488 if (!Chain)
8489 continue;
8490 VRound = DAG.getNode(SystemZISD::STRICT_VROUND, SDLoc(N),
8491 {MVT::v4f32, MVT::Other}, {Chain, Vec});
8492 Chain = VRound.getValue(1);
8493 } else
8494 VRound = DAG.getNode(SystemZISD::VROUND, SDLoc(N),
8495 MVT::v4f32, Vec);
8496 DCI.AddToWorklist(VRound.getNode());
8497 SDValue Extract1 =
8498 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(U), MVT::f32,
8499 VRound, DAG.getConstant(2, SDLoc(U), MVT::i32));
8500 DCI.AddToWorklist(Extract1.getNode());
8501 DAG.ReplaceAllUsesOfValueWith(OtherRound, Extract1);
8502 if (Chain)
8503 DAG.ReplaceAllUsesOfValueWith(OtherRound.getValue(1), Chain);
8504 SDValue Extract0 =
8505 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(Op0), MVT::f32,
8506 VRound, DAG.getConstant(0, SDLoc(Op0), MVT::i32));
8507 if (Chain)
8508 return DAG.getNode(ISD::MERGE_VALUES, SDLoc(Op0),
8509 N->getVTList(), Extract0, Chain);
8510 return Extract0;
8511 }
8512 }
8513 }
8514 }
8515 return SDValue();
8516}
8517
8518SDValue SystemZTargetLowering::combineFP_EXTEND(
8519 SDNode *N, DAGCombinerInfo &DCI) const {
8520
8521 if (!Subtarget.hasVector())
8522 return SDValue();
8523
8524 // (fpextend (extract_vector_elt X 0))
8525 // (fpextend (extract_vector_elt X 2)) ->
8526 // (extract_vector_elt (VEXTEND X) 0)
8527 // (extract_vector_elt (VEXTEND X) 1)
8528 //
8529 // This is a special case since the target doesn't really support v2f32s.
8530 unsigned OpNo = N->isStrictFPOpcode() ? 1 : 0;
8531 SelectionDAG &DAG = DCI.DAG;
8532 SDValue Op0 = N->getOperand(OpNo);
8533 if (N->getValueType(0) == MVT::f64 && Op0.hasOneUse() &&
8535 Op0.getOperand(0).getValueType() == MVT::v4f32 &&
8536 Op0.getOperand(1).getOpcode() == ISD::Constant &&
8537 Op0.getConstantOperandVal(1) == 0) {
8538 SDValue Vec = Op0.getOperand(0);
8539 for (auto *U : Vec->users()) {
8540 if (U != Op0.getNode() && U->hasOneUse() &&
8541 U->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
8542 U->getOperand(0) == Vec &&
8543 U->getOperand(1).getOpcode() == ISD::Constant &&
8544 U->getConstantOperandVal(1) == 2) {
8545 SDValue OtherExtend = SDValue(*U->user_begin(), 0);
8546 if (OtherExtend.getOpcode() == N->getOpcode() &&
8547 OtherExtend.getOperand(OpNo) == SDValue(U, 0) &&
8548 OtherExtend.getValueType() == MVT::f64) {
8549 SDValue VExtend, Chain;
8550 if (N->isStrictFPOpcode()) {
8551 Chain = MergeInputChains(N, OtherExtend.getNode());
8552 if (!Chain)
8553 continue;
8554 VExtend = DAG.getNode(SystemZISD::STRICT_VEXTEND, SDLoc(N),
8555 {MVT::v2f64, MVT::Other}, {Chain, Vec});
8556 Chain = VExtend.getValue(1);
8557 } else
8558 VExtend = DAG.getNode(SystemZISD::VEXTEND, SDLoc(N),
8559 MVT::v2f64, Vec);
8560 DCI.AddToWorklist(VExtend.getNode());
8561 SDValue Extract1 =
8562 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(U), MVT::f64,
8563 VExtend, DAG.getConstant(1, SDLoc(U), MVT::i32));
8564 DCI.AddToWorklist(Extract1.getNode());
8565 DAG.ReplaceAllUsesOfValueWith(OtherExtend, Extract1);
8566 if (Chain)
8567 DAG.ReplaceAllUsesOfValueWith(OtherExtend.getValue(1), Chain);
8568 SDValue Extract0 =
8569 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(Op0), MVT::f64,
8570 VExtend, DAG.getConstant(0, SDLoc(Op0), MVT::i32));
8571 if (Chain)
8572 return DAG.getNode(ISD::MERGE_VALUES, SDLoc(Op0),
8573 N->getVTList(), Extract0, Chain);
8574 return Extract0;
8575 }
8576 }
8577 }
8578 }
8579 return SDValue();
8580}
8581
8582SDValue SystemZTargetLowering::combineINT_TO_FP(
8583 SDNode *N, DAGCombinerInfo &DCI) const {
8584 if (DCI.Level != BeforeLegalizeTypes)
8585 return SDValue();
8586 SelectionDAG &DAG = DCI.DAG;
8587 LLVMContext &Ctx = *DAG.getContext();
8588 unsigned Opcode = N->getOpcode();
8589 EVT OutVT = N->getValueType(0);
8590 Type *OutLLVMTy = OutVT.getTypeForEVT(Ctx);
8591 SDValue Op = N->getOperand(0);
8592 unsigned OutScalarBits = OutLLVMTy->getScalarSizeInBits();
8593 unsigned InScalarBits = Op->getValueType(0).getScalarSizeInBits();
8594
8595 // Insert an extension before type-legalization to avoid scalarization, e.g.:
8596 // v2f64 = uint_to_fp v2i16
8597 // =>
8598 // v2f64 = uint_to_fp (v2i64 zero_extend v2i16)
8599 if (OutLLVMTy->isVectorTy() && OutScalarBits > InScalarBits &&
8600 OutScalarBits <= 64) {
8601 unsigned NumElts = cast<FixedVectorType>(OutLLVMTy)->getNumElements();
8602 EVT ExtVT = EVT::getVectorVT(
8603 Ctx, EVT::getIntegerVT(Ctx, OutLLVMTy->getScalarSizeInBits()), NumElts);
8604 unsigned ExtOpcode =
8606 SDValue ExtOp = DAG.getNode(ExtOpcode, SDLoc(N), ExtVT, Op);
8607 return DAG.getNode(Opcode, SDLoc(N), OutVT, ExtOp);
8608 }
8609 return SDValue();
8610}
8611
8612SDValue SystemZTargetLowering::combineFCOPYSIGN(
8613 SDNode *N, DAGCombinerInfo &DCI) const {
8614 SelectionDAG &DAG = DCI.DAG;
8615 EVT VT = N->getValueType(0);
8616 SDValue ValOp = N->getOperand(0);
8617 SDValue SignOp = N->getOperand(1);
8618
8619 // Remove the rounding which is not needed.
8620 if (SignOp.getOpcode() == ISD::FP_ROUND) {
8621 SDValue WideOp = SignOp.getOperand(0);
8622 return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, ValOp, WideOp);
8623 }
8624
8625 return SDValue();
8626}
8627
8628SDValue SystemZTargetLowering::combineBSWAP(
8629 SDNode *N, DAGCombinerInfo &DCI) const {
8630 SelectionDAG &DAG = DCI.DAG;
8631 // Combine BSWAP (LOAD) into LRVH/LRV/LRVG/VLBR
8632 if (ISD::isNON_EXTLoad(N->getOperand(0).getNode()) &&
8633 N->getOperand(0).hasOneUse() &&
8634 canLoadStoreByteSwapped(N->getValueType(0))) {
8635 SDValue Load = N->getOperand(0);
8636 LoadSDNode *LD = cast<LoadSDNode>(Load);
8637
8638 // Create the byte-swapping load.
8639 SDValue Ops[] = {
8640 LD->getChain(), // Chain
8641 LD->getBasePtr() // Ptr
8642 };
8643 EVT LoadVT = N->getValueType(0);
8644 if (LoadVT == MVT::i16)
8645 LoadVT = MVT::i32;
8646 SDValue BSLoad =
8647 DAG.getMemIntrinsicNode(SystemZISD::LRV, SDLoc(N),
8648 DAG.getVTList(LoadVT, MVT::Other),
8649 Ops, LD->getMemoryVT(), LD->getMemOperand());
8650
8651 // If this is an i16 load, insert the truncate.
8652 SDValue ResVal = BSLoad;
8653 if (N->getValueType(0) == MVT::i16)
8654 ResVal = DAG.getNode(ISD::TRUNCATE, SDLoc(N), MVT::i16, BSLoad);
8655
8656 // First, combine the bswap away. This makes the value produced by the
8657 // load dead.
8658 DCI.CombineTo(N, ResVal);
8659
8660 // Next, combine the load away, we give it a bogus result value but a real
8661 // chain result. The result value is dead because the bswap is dead.
8662 DCI.CombineTo(Load.getNode(), ResVal, BSLoad.getValue(1));
8663
8664 // Return N so it doesn't get rechecked!
8665 return SDValue(N, 0);
8666 }
8667
8668 // Look through bitcasts that retain the number of vector elements.
8669 SDValue Op = N->getOperand(0);
8670 if (Op.getOpcode() == ISD::BITCAST &&
8671 Op.getValueType().isVector() &&
8672 Op.getOperand(0).getValueType().isVector() &&
8673 Op.getValueType().getVectorNumElements() ==
8674 Op.getOperand(0).getValueType().getVectorNumElements())
8675 Op = Op.getOperand(0);
8676
8677 // Push BSWAP into a vector insertion if at least one side then simplifies.
8678 if (Op.getOpcode() == ISD::INSERT_VECTOR_ELT && Op.hasOneUse()) {
8679 SDValue Vec = Op.getOperand(0);
8680 SDValue Elt = Op.getOperand(1);
8681 SDValue Idx = Op.getOperand(2);
8682
8684 Vec.getOpcode() == ISD::BSWAP || Vec.isUndef() ||
8686 Elt.getOpcode() == ISD::BSWAP || Elt.isUndef() ||
8687 (canLoadStoreByteSwapped(N->getValueType(0)) &&
8688 ISD::isNON_EXTLoad(Elt.getNode()) && Elt.hasOneUse())) {
8689 EVT VecVT = N->getValueType(0);
8690 EVT EltVT = N->getValueType(0).getVectorElementType();
8691 if (VecVT != Vec.getValueType()) {
8692 Vec = DAG.getNode(ISD::BITCAST, SDLoc(N), VecVT, Vec);
8693 DCI.AddToWorklist(Vec.getNode());
8694 }
8695 if (EltVT != Elt.getValueType()) {
8696 Elt = DAG.getNode(ISD::BITCAST, SDLoc(N), EltVT, Elt);
8697 DCI.AddToWorklist(Elt.getNode());
8698 }
8699 Vec = DAG.getNode(ISD::BSWAP, SDLoc(N), VecVT, Vec);
8700 DCI.AddToWorklist(Vec.getNode());
8701 Elt = DAG.getNode(ISD::BSWAP, SDLoc(N), EltVT, Elt);
8702 DCI.AddToWorklist(Elt.getNode());
8703 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), VecVT,
8704 Vec, Elt, Idx);
8705 }
8706 }
8707
8708 // Push BSWAP into a vector shuffle if at least one side then simplifies.
8709 ShuffleVectorSDNode *SV = dyn_cast<ShuffleVectorSDNode>(Op);
8710 if (SV && Op.hasOneUse()) {
8711 SDValue Op0 = Op.getOperand(0);
8712 SDValue Op1 = Op.getOperand(1);
8713
8715 Op0.getOpcode() == ISD::BSWAP || Op0.isUndef() ||
8717 Op1.getOpcode() == ISD::BSWAP || Op1.isUndef()) {
8718 EVT VecVT = N->getValueType(0);
8719 if (VecVT != Op0.getValueType()) {
8720 Op0 = DAG.getNode(ISD::BITCAST, SDLoc(N), VecVT, Op0);
8721 DCI.AddToWorklist(Op0.getNode());
8722 }
8723 if (VecVT != Op1.getValueType()) {
8724 Op1 = DAG.getNode(ISD::BITCAST, SDLoc(N), VecVT, Op1);
8725 DCI.AddToWorklist(Op1.getNode());
8726 }
8727 Op0 = DAG.getNode(ISD::BSWAP, SDLoc(N), VecVT, Op0);
8728 DCI.AddToWorklist(Op0.getNode());
8729 Op1 = DAG.getNode(ISD::BSWAP, SDLoc(N), VecVT, Op1);
8730 DCI.AddToWorklist(Op1.getNode());
8731 return DAG.getVectorShuffle(VecVT, SDLoc(N), Op0, Op1, SV->getMask());
8732 }
8733 }
8734
8735 return SDValue();
8736}
8737
8738SDValue SystemZTargetLowering::combineSETCC(
8739 SDNode *N, DAGCombinerInfo &DCI) const {
8740 SelectionDAG &DAG = DCI.DAG;
8741 const ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
8742 const SDValue LHS = N->getOperand(0);
8743 const SDValue RHS = N->getOperand(1);
8744 bool CmpNull = isNullConstant(RHS);
8745 bool CmpAllOnes = isAllOnesConstant(RHS);
8746 EVT VT = N->getValueType(0);
8747 SDLoc DL(N);
8748
8749 // Match icmp_eq/ne(bitcast(icmp(X,Y)),0/-1) reduction patterns, and
8750 // change the outer compare to a i128 compare. This will normally
8751 // allow the reduction to be recognized in adjustICmp128, and even if
8752 // not, the i128 compare will still generate better code.
8753 if ((CC == ISD::SETNE || CC == ISD::SETEQ) && (CmpNull || CmpAllOnes)) {
8755 if (Src.getOpcode() == ISD::SETCC &&
8756 Src.getValueType().isFixedLengthVector() &&
8757 Src.getValueType().getScalarType() == MVT::i1) {
8758 EVT CmpVT = Src.getOperand(0).getValueType();
8759 if (CmpVT.getSizeInBits() == 128) {
8760 EVT IntVT = CmpVT.changeVectorElementTypeToInteger();
8761 SDValue LHS =
8762 DAG.getBitcast(MVT::i128, DAG.getSExtOrTrunc(Src, DL, IntVT));
8763 SDValue RHS = CmpNull ? DAG.getConstant(0, DL, MVT::i128)
8764 : DAG.getAllOnesConstant(DL, MVT::i128);
8765 return DAG.getNode(ISD::SETCC, DL, VT, LHS, RHS, N->getOperand(2),
8766 N->getFlags());
8767 }
8768 }
8769 }
8770
8771 return SDValue();
8772}
8773
8774static std::pair<SDValue, int> findCCUse(const SDValue &Val,
8775 unsigned Depth = 0) {
8776 // Limit depth of potentially exponential walk.
8777 if (Depth > 5)
8778 return std::make_pair(SDValue(), SystemZ::CCMASK_NONE);
8779
8780 switch (Val.getOpcode()) {
8781 default:
8782 return std::make_pair(SDValue(), SystemZ::CCMASK_NONE);
8783 case SystemZISD::IPM:
8784 if (Val.getOperand(0).getOpcode() == SystemZISD::CLC ||
8785 Val.getOperand(0).getOpcode() == SystemZISD::STRCMP)
8786 return std::make_pair(Val.getOperand(0), SystemZ::CCMASK_ICMP);
8787 return std::make_pair(Val.getOperand(0), SystemZ::CCMASK_ANY);
8788 case SystemZISD::SELECT_CCMASK: {
8789 SDValue Op4CCReg = Val.getOperand(4);
8790 if (Op4CCReg.getOpcode() == SystemZISD::ICMP ||
8791 Op4CCReg.getOpcode() == SystemZISD::TM) {
8792 auto [OpCC, OpCCValid] = findCCUse(Op4CCReg.getOperand(0), Depth + 1);
8793 if (OpCC != SDValue())
8794 return std::make_pair(OpCC, OpCCValid);
8795 }
8796 auto *CCValid = dyn_cast<ConstantSDNode>(Val.getOperand(2));
8797 if (!CCValid)
8798 return std::make_pair(SDValue(), SystemZ::CCMASK_NONE);
8799 int CCValidVal = CCValid->getZExtValue();
8800 return std::make_pair(Op4CCReg, CCValidVal);
8801 }
8802 case ISD::ADD:
8803 case ISD::AND:
8804 case ISD::OR:
8805 case ISD::XOR:
8806 case ISD::SHL:
8807 case ISD::SRA:
8808 case ISD::SRL:
8809 auto [Op0CC, Op0CCValid] = findCCUse(Val.getOperand(0), Depth + 1);
8810 if (Op0CC != SDValue())
8811 return std::make_pair(Op0CC, Op0CCValid);
8812 return findCCUse(Val.getOperand(1), Depth + 1);
8813 }
8814}
8815
8816static bool combineCCMask(SDValue &CCReg, int &CCValid, int &CCMask,
8817 SelectionDAG &DAG);
8818
8820 SelectionDAG &DAG) {
8821 SDLoc DL(Val);
8822 auto Opcode = Val.getOpcode();
8823 switch (Opcode) {
8824 default:
8825 return {};
8826 case ISD::Constant:
8827 return {Val, Val, Val, Val};
8828 case SystemZISD::IPM: {
8829 SDValue IPMOp0 = Val.getOperand(0);
8830 if (IPMOp0 != CC)
8831 return {};
8832 SmallVector<SDValue, 4> ShiftedCCVals;
8833 for (auto CC : {0, 1, 2, 3})
8834 ShiftedCCVals.emplace_back(
8835 DAG.getConstant((CC << SystemZ::IPM_CC), DL, MVT::i32));
8836 return ShiftedCCVals;
8837 }
8838 case SystemZISD::SELECT_CCMASK: {
8839 SDValue TrueVal = Val.getOperand(0), FalseVal = Val.getOperand(1);
8840 auto *CCValid = dyn_cast<ConstantSDNode>(Val.getOperand(2));
8841 auto *CCMask = dyn_cast<ConstantSDNode>(Val.getOperand(3));
8842 if (!CCValid || !CCMask)
8843 return {};
8844
8845 int CCValidVal = CCValid->getZExtValue();
8846 int CCMaskVal = CCMask->getZExtValue();
8847 // Pruning search tree early - Moving CC test and combineCCMask ahead of
8848 // recursive call to simplifyAssumingCCVal.
8849 SDValue Op4CCReg = Val.getOperand(4);
8850 if (Op4CCReg != CC)
8851 combineCCMask(Op4CCReg, CCValidVal, CCMaskVal, DAG);
8852 if (Op4CCReg != CC)
8853 return {};
8854 const auto &&TrueSDVals = simplifyAssumingCCVal(TrueVal, CC, DAG);
8855 const auto &&FalseSDVals = simplifyAssumingCCVal(FalseVal, CC, DAG);
8856 if (TrueSDVals.empty() || FalseSDVals.empty())
8857 return {};
8858 SmallVector<SDValue, 4> MergedSDVals;
8859 for (auto &CCVal : {0, 1, 2, 3})
8860 MergedSDVals.emplace_back(((CCMaskVal & (1 << (3 - CCVal))) != 0)
8861 ? TrueSDVals[CCVal]
8862 : FalseSDVals[CCVal]);
8863 return MergedSDVals;
8864 }
8865 case ISD::ADD:
8866 case ISD::AND:
8867 case ISD::OR:
8868 case ISD::XOR:
8869 case ISD::SRA:
8870 // Avoid introducing CC spills (because ADD/AND/OR/XOR/SRA
8871 // would clobber CC).
8872 if (!Val.hasOneUse())
8873 return {};
8874 [[fallthrough]];
8875 case ISD::SHL:
8876 case ISD::SRL:
8877 SDValue Op0 = Val.getOperand(0), Op1 = Val.getOperand(1);
8878 const auto &&Op0SDVals = simplifyAssumingCCVal(Op0, CC, DAG);
8879 const auto &&Op1SDVals = simplifyAssumingCCVal(Op1, CC, DAG);
8880 if (Op0SDVals.empty() || Op1SDVals.empty())
8881 return {};
8882 SmallVector<SDValue, 4> BinaryOpSDVals;
8883 for (auto CCVal : {0, 1, 2, 3})
8884 BinaryOpSDVals.emplace_back(DAG.getNode(
8885 Opcode, DL, Val.getValueType(), Op0SDVals[CCVal], Op1SDVals[CCVal]));
8886 return BinaryOpSDVals;
8887 }
8888}
8889
8890static bool combineCCMask(SDValue &CCReg, int &CCValid, int &CCMask,
8891 SelectionDAG &DAG) {
8892 // We have a SELECT_CCMASK or BR_CCMASK comparing the condition code
8893 // set by the CCReg instruction using the CCValid / CCMask masks,
8894 // If the CCReg instruction is itself a ICMP / TM testing the condition
8895 // code set by some other instruction, see whether we can directly
8896 // use that condition code.
8897 auto *CCNode = CCReg.getNode();
8898 if (!CCNode)
8899 return false;
8900
8901 if (CCNode->getOpcode() == SystemZISD::TM) {
8902 if (CCValid != SystemZ::CCMASK_TM)
8903 return false;
8904 auto emulateTMCCMask = [](const SDValue &Op0Val, const SDValue &Op1Val) {
8905 auto *Op0Node = dyn_cast<ConstantSDNode>(Op0Val.getNode());
8906 auto *Op1Node = dyn_cast<ConstantSDNode>(Op1Val.getNode());
8907 if (!Op0Node || !Op1Node)
8908 return -1;
8909 auto Op0APVal = Op0Node->getAPIntValue();
8910 auto Op1APVal = Op1Node->getAPIntValue();
8911 auto Result = Op0APVal & Op1APVal;
8912 bool AllOnes = Result == Op1APVal;
8913 bool AllZeros = Result == 0;
8914 bool IsLeftMostBitSet = Result[Op1APVal.getActiveBits() - 1] != 0;
8915 return AllZeros ? 0 : AllOnes ? 3 : IsLeftMostBitSet ? 2 : 1;
8916 };
8917 SDValue Op0 = CCNode->getOperand(0);
8918 SDValue Op1 = CCNode->getOperand(1);
8919 auto [Op0CC, Op0CCValid] = findCCUse(Op0);
8920 if (Op0CC == SDValue())
8921 return false;
8922 const auto &&Op0SDVals = simplifyAssumingCCVal(Op0, Op0CC, DAG);
8923 const auto &&Op1SDVals = simplifyAssumingCCVal(Op1, Op0CC, DAG);
8924 if (Op0SDVals.empty() || Op1SDVals.empty())
8925 return false;
8926 int NewCCMask = 0;
8927 for (auto CC : {0, 1, 2, 3}) {
8928 auto CCVal = emulateTMCCMask(Op0SDVals[CC], Op1SDVals[CC]);
8929 if (CCVal < 0)
8930 return false;
8931 NewCCMask <<= 1;
8932 NewCCMask |= (CCMask & (1 << (3 - CCVal))) != 0;
8933 }
8934 NewCCMask &= Op0CCValid;
8935 CCReg = Op0CC;
8936 CCMask = NewCCMask;
8937 CCValid = Op0CCValid;
8938 return true;
8939 }
8940 if (CCNode->getOpcode() != SystemZISD::ICMP ||
8941 CCValid != SystemZ::CCMASK_ICMP)
8942 return false;
8943
8944 SDValue CmpOp0 = CCNode->getOperand(0);
8945 SDValue CmpOp1 = CCNode->getOperand(1);
8946 SDValue CmpOp2 = CCNode->getOperand(2);
8947 auto [Op0CC, Op0CCValid] = findCCUse(CmpOp0);
8948 if (Op0CC != SDValue()) {
8949 const auto &&Op0SDVals = simplifyAssumingCCVal(CmpOp0, Op0CC, DAG);
8950 const auto &&Op1SDVals = simplifyAssumingCCVal(CmpOp1, Op0CC, DAG);
8951 if (Op0SDVals.empty() || Op1SDVals.empty())
8952 return false;
8953
8954 auto *CmpType = dyn_cast<ConstantSDNode>(CmpOp2);
8955 auto CmpTypeVal = CmpType->getZExtValue();
8956 const auto compareCCSigned = [&CmpTypeVal](const SDValue &Op0Val,
8957 const SDValue &Op1Val) {
8958 auto *Op0Node = dyn_cast<ConstantSDNode>(Op0Val.getNode());
8959 auto *Op1Node = dyn_cast<ConstantSDNode>(Op1Val.getNode());
8960 if (!Op0Node || !Op1Node)
8961 return -1;
8962 auto Op0APVal = Op0Node->getAPIntValue();
8963 auto Op1APVal = Op1Node->getAPIntValue();
8964 if (CmpTypeVal == SystemZICMP::SignedOnly)
8965 return Op0APVal == Op1APVal ? 0 : Op0APVal.slt(Op1APVal) ? 1 : 2;
8966 return Op0APVal == Op1APVal ? 0 : Op0APVal.ult(Op1APVal) ? 1 : 2;
8967 };
8968 int NewCCMask = 0;
8969 for (auto CC : {0, 1, 2, 3}) {
8970 auto CCVal = compareCCSigned(Op0SDVals[CC], Op1SDVals[CC]);
8971 if (CCVal < 0)
8972 return false;
8973 NewCCMask <<= 1;
8974 NewCCMask |= (CCMask & (1 << (3 - CCVal))) != 0;
8975 }
8976 NewCCMask &= Op0CCValid;
8977 CCMask = NewCCMask;
8978 CCReg = Op0CC;
8979 CCValid = Op0CCValid;
8980 return true;
8981 }
8982
8983 return false;
8984}
8985
8986// Merging versus split in multiple branches cost.
8989 const Value *Lhs,
8990 const Value *Rhs,
8991 const Function *) const {
8992 const auto isFlagOutOpCC = [](const Value *V) {
8993 using namespace llvm::PatternMatch;
8994 const Value *RHSVal;
8995 const APInt *RHSC;
8996 if (const auto *I = dyn_cast<Instruction>(V)) {
8997 // PatternMatch.h provides concise tree-based pattern match of llvm IR.
8998 if (match(I->getOperand(0), m_And(m_Value(RHSVal), m_APInt(RHSC))) ||
8999 match(I, m_Cmp(m_Value(RHSVal), m_APInt(RHSC)))) {
9000 if (const auto *CB = dyn_cast<CallBase>(RHSVal)) {
9001 if (CB->isInlineAsm()) {
9002 const InlineAsm *IA = cast<InlineAsm>(CB->getCalledOperand());
9003 return IA && IA->getConstraintString().contains("{@cc}");
9004 }
9005 }
9006 }
9007 }
9008 return false;
9009 };
9010 // Pattern (ICmp %asm) or (ICmp (And %asm)).
9011 // Cost of longest dependency chain (ICmp, And) is 2. CostThreshold or
9012 // BaseCost can be set >=2. If cost of instruction <= CostThreshold
9013 // conditionals will be merged or else conditionals will be split.
9014 if (isFlagOutOpCC(Lhs) && isFlagOutOpCC(Rhs))
9015 return {3, 0, -1};
9016 // Default.
9017 return {-1, -1, -1};
9018}
9019
9020SDValue SystemZTargetLowering::combineBR_CCMASK(SDNode *N,
9021 DAGCombinerInfo &DCI) const {
9022 SelectionDAG &DAG = DCI.DAG;
9023
9024 // Combine BR_CCMASK (ICMP (SELECT_CCMASK)) into a single BR_CCMASK.
9025 auto *CCValid = dyn_cast<ConstantSDNode>(N->getOperand(1));
9026 auto *CCMask = dyn_cast<ConstantSDNode>(N->getOperand(2));
9027 if (!CCValid || !CCMask)
9028 return SDValue();
9029
9030 int CCValidVal = CCValid->getZExtValue();
9031 int CCMaskVal = CCMask->getZExtValue();
9032 SDValue Chain = N->getOperand(0);
9033 SDValue CCReg = N->getOperand(4);
9034 // If combineCMask was able to merge or simplify ccvalid or ccmask, re-emit
9035 // the modified BR_CCMASK with the new values.
9036 // In order to avoid conditional branches with full or empty cc masks, do not
9037 // do this if ccmask is 0 or equal to ccvalid.
9038 if (combineCCMask(CCReg, CCValidVal, CCMaskVal, DAG) && CCMaskVal != 0 &&
9039 CCMaskVal != CCValidVal)
9040 return DAG.getNode(SystemZISD::BR_CCMASK, SDLoc(N), N->getValueType(0),
9041 Chain,
9042 DAG.getTargetConstant(CCValidVal, SDLoc(N), MVT::i32),
9043 DAG.getTargetConstant(CCMaskVal, SDLoc(N), MVT::i32),
9044 N->getOperand(3), CCReg);
9045 return SDValue();
9046}
9047
9048SDValue SystemZTargetLowering::combineSELECT_CCMASK(
9049 SDNode *N, DAGCombinerInfo &DCI) const {
9050 SelectionDAG &DAG = DCI.DAG;
9051
9052 // Combine SELECT_CCMASK (ICMP (SELECT_CCMASK)) into a single SELECT_CCMASK.
9053 auto *CCValid = dyn_cast<ConstantSDNode>(N->getOperand(2));
9054 auto *CCMask = dyn_cast<ConstantSDNode>(N->getOperand(3));
9055 if (!CCValid || !CCMask)
9056 return SDValue();
9057
9058 int CCValidVal = CCValid->getZExtValue();
9059 int CCMaskVal = CCMask->getZExtValue();
9060 SDValue CCReg = N->getOperand(4);
9061
9062 bool IsCombinedCCReg = combineCCMask(CCReg, CCValidVal, CCMaskVal, DAG);
9063
9064 // Populate SDVals vector for each condition code ccval for given Val, which
9065 // can again be another nested select_ccmask with the same CC.
9066 const auto constructCCSDValsFromSELECT = [&CCReg](SDValue &Val) {
9067 if (Val.getOpcode() == SystemZISD::SELECT_CCMASK) {
9069 if (Val.getOperand(4) != CCReg)
9070 return SmallVector<SDValue, 4>{};
9071 SDValue TrueVal = Val.getOperand(0), FalseVal = Val.getOperand(1);
9072 auto *CCMask = dyn_cast<ConstantSDNode>(Val.getOperand(3));
9073 if (!CCMask)
9074 return SmallVector<SDValue, 4>{};
9075
9076 int CCMaskVal = CCMask->getZExtValue();
9077 for (auto &CC : {0, 1, 2, 3})
9078 Res.emplace_back(((CCMaskVal & (1 << (3 - CC))) != 0) ? TrueVal
9079 : FalseVal);
9080 return Res;
9081 }
9082 return SmallVector<SDValue, 4>{Val, Val, Val, Val};
9083 };
9084 // Attempting to optimize TrueVal/FalseVal in outermost select_ccmask either
9085 // with CCReg found by combineCCMask or original CCReg.
9086 SDValue TrueVal = N->getOperand(0);
9087 SDValue FalseVal = N->getOperand(1);
9088 auto &&TrueSDVals = simplifyAssumingCCVal(TrueVal, CCReg, DAG);
9089 auto &&FalseSDVals = simplifyAssumingCCVal(FalseVal, CCReg, DAG);
9090 // TrueSDVals/FalseSDVals might be empty in case of non-constant
9091 // TrueVal/FalseVal for select_ccmask, which can not be optimized further.
9092 if (TrueSDVals.empty())
9093 TrueSDVals = constructCCSDValsFromSELECT(TrueVal);
9094 if (FalseSDVals.empty())
9095 FalseSDVals = constructCCSDValsFromSELECT(FalseVal);
9096 if (!TrueSDVals.empty() && !FalseSDVals.empty()) {
9097 SmallSet<SDValue, 4> MergedSDValsSet;
9098 // Ignoring CC values outside CCValiid.
9099 for (auto CC : {0, 1, 2, 3}) {
9100 if ((CCValidVal & ((1 << (3 - CC)))) != 0)
9101 MergedSDValsSet.insert(((CCMaskVal & (1 << (3 - CC))) != 0)
9102 ? TrueSDVals[CC]
9103 : FalseSDVals[CC]);
9104 }
9105 if (MergedSDValsSet.size() == 1)
9106 return *MergedSDValsSet.begin();
9107 if (MergedSDValsSet.size() == 2) {
9108 auto BeginIt = MergedSDValsSet.begin();
9109 SDValue NewTrueVal = *BeginIt, NewFalseVal = *next(BeginIt);
9110 if (NewTrueVal == FalseVal || NewFalseVal == TrueVal)
9111 std::swap(NewTrueVal, NewFalseVal);
9112 int NewCCMask = 0;
9113 for (auto CC : {0, 1, 2, 3}) {
9114 NewCCMask <<= 1;
9115 NewCCMask |= ((CCMaskVal & (1 << (3 - CC))) != 0)
9116 ? (TrueSDVals[CC] == NewTrueVal)
9117 : (FalseSDVals[CC] == NewTrueVal);
9118 }
9119 CCMaskVal = NewCCMask;
9120 CCMaskVal &= CCValidVal;
9121 TrueVal = NewTrueVal;
9122 FalseVal = NewFalseVal;
9123 IsCombinedCCReg = true;
9124 }
9125 }
9126 // If the condition is trivially false or trivially true after
9127 // combineCCMask, just collapse this SELECT_CCMASK to the indicated value
9128 // (possibly modified by constructCCSDValsFromSELECT).
9129 if (CCMaskVal == 0)
9130 return FalseVal;
9131 if (CCMaskVal == CCValidVal)
9132 return TrueVal;
9133
9134 if (IsCombinedCCReg)
9135 return DAG.getNode(
9136 SystemZISD::SELECT_CCMASK, SDLoc(N), N->getValueType(0), TrueVal,
9137 FalseVal, DAG.getTargetConstant(CCValidVal, SDLoc(N), MVT::i32),
9138 DAG.getTargetConstant(CCMaskVal, SDLoc(N), MVT::i32), CCReg);
9139
9140 return SDValue();
9141}
9142
9143SDValue SystemZTargetLowering::combineGET_CCMASK(
9144 SDNode *N, DAGCombinerInfo &DCI) const {
9145
9146 // Optimize away GET_CCMASK (SELECT_CCMASK) if the CC masks are compatible
9147 auto *CCValid = dyn_cast<ConstantSDNode>(N->getOperand(1));
9148 auto *CCMask = dyn_cast<ConstantSDNode>(N->getOperand(2));
9149 if (!CCValid || !CCMask)
9150 return SDValue();
9151 int CCValidVal = CCValid->getZExtValue();
9152 int CCMaskVal = CCMask->getZExtValue();
9153
9154 SDValue Select = N->getOperand(0);
9155 if (Select->getOpcode() == ISD::TRUNCATE)
9156 Select = Select->getOperand(0);
9157 if (Select->getOpcode() != SystemZISD::SELECT_CCMASK)
9158 return SDValue();
9159
9160 auto *SelectCCValid = dyn_cast<ConstantSDNode>(Select->getOperand(2));
9161 auto *SelectCCMask = dyn_cast<ConstantSDNode>(Select->getOperand(3));
9162 if (!SelectCCValid || !SelectCCMask)
9163 return SDValue();
9164 int SelectCCValidVal = SelectCCValid->getZExtValue();
9165 int SelectCCMaskVal = SelectCCMask->getZExtValue();
9166
9167 auto *TrueVal = dyn_cast<ConstantSDNode>(Select->getOperand(0));
9168 auto *FalseVal = dyn_cast<ConstantSDNode>(Select->getOperand(1));
9169 if (!TrueVal || !FalseVal)
9170 return SDValue();
9171 if (TrueVal->getZExtValue() == 1 && FalseVal->getZExtValue() == 0)
9172 ;
9173 else if (TrueVal->getZExtValue() == 0 && FalseVal->getZExtValue() == 1)
9174 SelectCCMaskVal ^= SelectCCValidVal;
9175 else
9176 return SDValue();
9177
9178 if (SelectCCValidVal & ~CCValidVal)
9179 return SDValue();
9180 if (SelectCCMaskVal != (CCMaskVal & SelectCCValidVal))
9181 return SDValue();
9182
9183 return Select->getOperand(4);
9184}
9185
9186SDValue SystemZTargetLowering::combineIntDIVREM(
9187 SDNode *N, DAGCombinerInfo &DCI) const {
9188 SelectionDAG &DAG = DCI.DAG;
9189 EVT VT = N->getValueType(0);
9190 // In the case where the divisor is a vector of constants a cheaper
9191 // sequence of instructions can replace the divide. BuildSDIV is called to
9192 // do this during DAG combining, but it only succeeds when it can build a
9193 // multiplication node. The only option for SystemZ is ISD::SMUL_LOHI, and
9194 // since it is not Legal but Custom it can only happen before
9195 // legalization. Therefore we must scalarize this early before Combine
9196 // 1. For widened vectors, this is already the result of type legalization.
9197 if (DCI.Level == BeforeLegalizeTypes && VT.isVector() && isTypeLegal(VT) &&
9198 DAG.isConstantIntBuildVectorOrConstantInt(N->getOperand(1)))
9199 return DAG.UnrollVectorOp(N);
9200 return SDValue();
9201}
9202
9203
9204// Transform a right shift of a multiply-and-add into a multiply-and-add-high.
9205// This is closely modeled after the common-code combineShiftToMULH.
9206SDValue SystemZTargetLowering::combineShiftToMulAddHigh(
9207 SDNode *N, DAGCombinerInfo &DCI) const {
9208 SelectionDAG &DAG = DCI.DAG;
9209 SDLoc DL(N);
9210
9211 assert((N->getOpcode() == ISD::SRL || N->getOpcode() == ISD::SRA) &&
9212 "SRL or SRA node is required here!");
9213
9214 if (!Subtarget.hasVector())
9215 return SDValue();
9216
9217 // Check the shift amount. Proceed with the transformation if the shift
9218 // amount is constant.
9219 ConstantSDNode *ShiftAmtSrc = isConstOrConstSplat(N->getOperand(1));
9220 if (!ShiftAmtSrc)
9221 return SDValue();
9222
9223 // The operation feeding into the shift must be an add.
9224 SDValue ShiftOperand = N->getOperand(0);
9225 if (ShiftOperand.getOpcode() != ISD::ADD)
9226 return SDValue();
9227
9228 // One operand of the add must be a multiply.
9229 SDValue MulOp = ShiftOperand.getOperand(0);
9230 SDValue AddOp = ShiftOperand.getOperand(1);
9231 if (MulOp.getOpcode() != ISD::MUL) {
9232 if (AddOp.getOpcode() != ISD::MUL)
9233 return SDValue();
9234 std::swap(MulOp, AddOp);
9235 }
9236
9237 // All operands must be equivalent extend nodes.
9238 SDValue LeftOp = MulOp.getOperand(0);
9239 SDValue RightOp = MulOp.getOperand(1);
9240
9241 bool IsSignExt = LeftOp.getOpcode() == ISD::SIGN_EXTEND;
9242 bool IsZeroExt = LeftOp.getOpcode() == ISD::ZERO_EXTEND;
9243
9244 if (!IsSignExt && !IsZeroExt)
9245 return SDValue();
9246
9247 EVT NarrowVT = LeftOp.getOperand(0).getValueType();
9248 unsigned NarrowVTSize = NarrowVT.getScalarSizeInBits();
9249
9250 SDValue MulhRightOp;
9251 if (ConstantSDNode *Constant = isConstOrConstSplat(RightOp)) {
9252 unsigned ActiveBits = IsSignExt
9253 ? Constant->getAPIntValue().getSignificantBits()
9254 : Constant->getAPIntValue().getActiveBits();
9255 if (ActiveBits > NarrowVTSize)
9256 return SDValue();
9257 MulhRightOp = DAG.getConstant(
9258 Constant->getAPIntValue().trunc(NarrowVT.getScalarSizeInBits()), DL,
9259 NarrowVT);
9260 } else {
9261 if (LeftOp.getOpcode() != RightOp.getOpcode())
9262 return SDValue();
9263 // Check that the two extend nodes are the same type.
9264 if (NarrowVT != RightOp.getOperand(0).getValueType())
9265 return SDValue();
9266 MulhRightOp = RightOp.getOperand(0);
9267 }
9268
9269 SDValue MulhAddOp;
9270 if (ConstantSDNode *Constant = isConstOrConstSplat(AddOp)) {
9271 unsigned ActiveBits = IsSignExt
9272 ? Constant->getAPIntValue().getSignificantBits()
9273 : Constant->getAPIntValue().getActiveBits();
9274 if (ActiveBits > NarrowVTSize)
9275 return SDValue();
9276 MulhAddOp = DAG.getConstant(
9277 Constant->getAPIntValue().trunc(NarrowVT.getScalarSizeInBits()), DL,
9278 NarrowVT);
9279 } else {
9280 if (LeftOp.getOpcode() != AddOp.getOpcode())
9281 return SDValue();
9282 // Check that the two extend nodes are the same type.
9283 if (NarrowVT != AddOp.getOperand(0).getValueType())
9284 return SDValue();
9285 MulhAddOp = AddOp.getOperand(0);
9286 }
9287
9288 EVT WideVT = LeftOp.getValueType();
9289 // Proceed with the transformation if the wide types match.
9290 assert((WideVT == RightOp.getValueType()) &&
9291 "Cannot have a multiply node with two different operand types.");
9292 assert((WideVT == AddOp.getValueType()) &&
9293 "Cannot have an add node with two different operand types.");
9294
9295 // Proceed with the transformation if the wide type is twice as large
9296 // as the narrow type.
9297 if (WideVT.getScalarSizeInBits() != 2 * NarrowVTSize)
9298 return SDValue();
9299
9300 // Check the shift amount with the narrow type size.
9301 // Proceed with the transformation if the shift amount is the width
9302 // of the narrow type.
9303 unsigned ShiftAmt = ShiftAmtSrc->getZExtValue();
9304 if (ShiftAmt != NarrowVTSize)
9305 return SDValue();
9306
9307 // Proceed if we support the multiply-and-add-high operation.
9308 if (!(NarrowVT == MVT::v16i8 || NarrowVT == MVT::v8i16 ||
9309 NarrowVT == MVT::v4i32 ||
9310 (Subtarget.hasVectorEnhancements3() &&
9311 (NarrowVT == MVT::v2i64 || NarrowVT == MVT::i128))))
9312 return SDValue();
9313
9314 // Emit the VMAH (signed) or VMALH (unsigned) operation.
9315 SDValue Result = DAG.getNode(IsSignExt ? SystemZISD::VMAH : SystemZISD::VMALH,
9316 DL, NarrowVT, LeftOp.getOperand(0),
9317 MulhRightOp, MulhAddOp);
9318 bool IsSigned = N->getOpcode() == ISD::SRA;
9319 return DAG.getExtOrTrunc(IsSigned, Result, DL, WideVT);
9320}
9321
9322// Op is an operand of a multiplication. Check whether this can be folded
9323// into an even/odd widening operation; if so, return the opcode to be used
9324// and update Op to the appropriate sub-operand. Note that the caller must
9325// verify that *both* operands of the multiplication support the operation.
9327 const SystemZSubtarget &Subtarget,
9328 SDValue &Op) {
9329 EVT VT = Op.getValueType();
9330
9331 // Check for (sign/zero_extend_vector_inreg (vector_shuffle)) corresponding
9332 // to selecting the even or odd vector elements.
9333 if (VT.isVector() && DAG.getTargetLoweringInfo().isTypeLegal(VT) &&
9334 (Op.getOpcode() == ISD::SIGN_EXTEND_VECTOR_INREG ||
9335 Op.getOpcode() == ISD::ZERO_EXTEND_VECTOR_INREG)) {
9336 bool IsSigned = Op.getOpcode() == ISD::SIGN_EXTEND_VECTOR_INREG;
9337 unsigned NumElts = VT.getVectorNumElements();
9338 Op = Op.getOperand(0);
9339 if (Op.getValueType().getVectorNumElements() == 2 * NumElts &&
9340 Op.getOpcode() == ISD::VECTOR_SHUFFLE) {
9342 ArrayRef<int> ShuffleMask = SVN->getMask();
9343 bool CanUseEven = true, CanUseOdd = true;
9344 for (unsigned Elt = 0; Elt < NumElts; Elt++) {
9345 if (ShuffleMask[Elt] == -1)
9346 continue;
9347 if (unsigned(ShuffleMask[Elt]) != 2 * Elt)
9348 CanUseEven = false;
9349 if (unsigned(ShuffleMask[Elt]) != 2 * Elt + 1)
9350 CanUseOdd = false;
9351 }
9352 Op = Op.getOperand(0);
9353 if (CanUseEven)
9354 return IsSigned ? SystemZISD::VME : SystemZISD::VMLE;
9355 if (CanUseOdd)
9356 return IsSigned ? SystemZISD::VMO : SystemZISD::VMLO;
9357 }
9358 }
9359
9360 // For z17, we can also support the v2i64->i128 case, which looks like
9361 // (sign/zero_extend (extract_vector_elt X 0/1))
9362 if (VT == MVT::i128 && Subtarget.hasVectorEnhancements3() &&
9363 (Op.getOpcode() == ISD::SIGN_EXTEND ||
9364 Op.getOpcode() == ISD::ZERO_EXTEND)) {
9365 bool IsSigned = Op.getOpcode() == ISD::SIGN_EXTEND;
9366 Op = Op.getOperand(0);
9367 if (Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
9368 Op.getOperand(0).getValueType() == MVT::v2i64 &&
9369 Op.getOperand(1).getOpcode() == ISD::Constant) {
9370 unsigned Elem = Op.getConstantOperandVal(1);
9371 Op = Op.getOperand(0);
9372 if (Elem == 0)
9373 return IsSigned ? SystemZISD::VME : SystemZISD::VMLE;
9374 if (Elem == 1)
9375 return IsSigned ? SystemZISD::VMO : SystemZISD::VMLO;
9376 }
9377 }
9378
9379 return 0;
9380}
9381
9382SDValue SystemZTargetLowering::combineMUL(
9383 SDNode *N, DAGCombinerInfo &DCI) const {
9384 SelectionDAG &DAG = DCI.DAG;
9385
9386 // Detect even/odd widening multiplication.
9387 SDValue Op0 = N->getOperand(0);
9388 SDValue Op1 = N->getOperand(1);
9389 unsigned OpcodeCand0 = detectEvenOddMultiplyOperand(DAG, Subtarget, Op0);
9390 unsigned OpcodeCand1 = detectEvenOddMultiplyOperand(DAG, Subtarget, Op1);
9391 if (OpcodeCand0 && OpcodeCand0 == OpcodeCand1)
9392 return DAG.getNode(OpcodeCand0, SDLoc(N), N->getValueType(0), Op0, Op1);
9393
9394 return SDValue();
9395}
9396
9397SDValue SystemZTargetLowering::combineINTRINSIC(
9398 SDNode *N, DAGCombinerInfo &DCI) const {
9399 SelectionDAG &DAG = DCI.DAG;
9400
9401 unsigned Id = N->getConstantOperandVal(1);
9402 switch (Id) {
9403 // VECTOR LOAD (RIGHTMOST) WITH LENGTH with a length operand of 15
9404 // or larger is simply a vector load.
9405 case Intrinsic::s390_vll:
9406 case Intrinsic::s390_vlrl:
9407 if (auto *C = dyn_cast<ConstantSDNode>(N->getOperand(2)))
9408 if (C->getZExtValue() >= 15)
9409 return DAG.getLoad(N->getValueType(0), SDLoc(N), N->getOperand(0),
9410 N->getOperand(3), MachinePointerInfo());
9411 break;
9412 // Likewise for VECTOR STORE (RIGHTMOST) WITH LENGTH.
9413 case Intrinsic::s390_vstl:
9414 case Intrinsic::s390_vstrl:
9415 if (auto *C = dyn_cast<ConstantSDNode>(N->getOperand(3)))
9416 if (C->getZExtValue() >= 15)
9417 return DAG.getStore(N->getOperand(0), SDLoc(N), N->getOperand(2),
9418 N->getOperand(4), MachinePointerInfo());
9419 break;
9420 }
9421
9422 return SDValue();
9423}
9424
9425SDValue SystemZTargetLowering::unwrapAddress(SDValue N) const {
9426 if (N->getOpcode() == SystemZISD::PCREL_WRAPPER)
9427 return N->getOperand(0);
9428 return N;
9429}
9430
9432 DAGCombinerInfo &DCI) const {
9433 switch(N->getOpcode()) {
9434 default: break;
9435 case ISD::ZERO_EXTEND: return combineZERO_EXTEND(N, DCI);
9436 case ISD::SIGN_EXTEND: return combineSIGN_EXTEND(N, DCI);
9437 case ISD::SIGN_EXTEND_INREG: return combineSIGN_EXTEND_INREG(N, DCI);
9438 case SystemZISD::MERGE_HIGH:
9439 case SystemZISD::MERGE_LOW: return combineMERGE(N, DCI);
9440 case ISD::LOAD: return combineLOAD(N, DCI);
9441 case ISD::STORE: return combineSTORE(N, DCI);
9442 case ISD::VECTOR_SHUFFLE: return combineVECTOR_SHUFFLE(N, DCI);
9443 case ISD::EXTRACT_VECTOR_ELT: return combineEXTRACT_VECTOR_ELT(N, DCI);
9444 case SystemZISD::JOIN_DWORDS: return combineJOIN_DWORDS(N, DCI);
9446 case ISD::FP_ROUND: return combineFP_ROUND(N, DCI);
9448 case ISD::FP_EXTEND: return combineFP_EXTEND(N, DCI);
9449 case ISD::SINT_TO_FP:
9450 case ISD::UINT_TO_FP: return combineINT_TO_FP(N, DCI);
9451 case ISD::FCOPYSIGN: return combineFCOPYSIGN(N, DCI);
9452 case ISD::BSWAP: return combineBSWAP(N, DCI);
9453 case ISD::SETCC: return combineSETCC(N, DCI);
9454 case SystemZISD::BR_CCMASK: return combineBR_CCMASK(N, DCI);
9455 case SystemZISD::SELECT_CCMASK: return combineSELECT_CCMASK(N, DCI);
9456 case SystemZISD::GET_CCMASK: return combineGET_CCMASK(N, DCI);
9457 case ISD::SRL:
9458 case ISD::SRA: return combineShiftToMulAddHigh(N, DCI);
9459 case ISD::MUL: return combineMUL(N, DCI);
9460 case ISD::SDIV:
9461 case ISD::UDIV:
9462 case ISD::SREM:
9463 case ISD::UREM: return combineIntDIVREM(N, DCI);
9465 case ISD::INTRINSIC_VOID: return combineINTRINSIC(N, DCI);
9466 }
9467
9468 return SDValue();
9469}
9470
9471// Return the demanded elements for the OpNo source operand of Op. DemandedElts
9472// are for Op.
9473static APInt getDemandedSrcElements(SDValue Op, const APInt &DemandedElts,
9474 unsigned OpNo) {
9475 EVT VT = Op.getValueType();
9476 unsigned NumElts = (VT.isVector() ? VT.getVectorNumElements() : 1);
9477 APInt SrcDemE;
9478 unsigned Opcode = Op.getOpcode();
9479 if (Opcode == ISD::INTRINSIC_WO_CHAIN) {
9480 unsigned Id = Op.getConstantOperandVal(0);
9481 switch (Id) {
9482 case Intrinsic::s390_vpksh: // PACKS
9483 case Intrinsic::s390_vpksf:
9484 case Intrinsic::s390_vpksg:
9485 case Intrinsic::s390_vpkshs: // PACKS_CC
9486 case Intrinsic::s390_vpksfs:
9487 case Intrinsic::s390_vpksgs:
9488 case Intrinsic::s390_vpklsh: // PACKLS
9489 case Intrinsic::s390_vpklsf:
9490 case Intrinsic::s390_vpklsg:
9491 case Intrinsic::s390_vpklshs: // PACKLS_CC
9492 case Intrinsic::s390_vpklsfs:
9493 case Intrinsic::s390_vpklsgs:
9494 // VECTOR PACK truncates the elements of two source vectors into one.
9495 SrcDemE = DemandedElts;
9496 if (OpNo == 2)
9497 SrcDemE.lshrInPlace(NumElts / 2);
9498 SrcDemE = SrcDemE.trunc(NumElts / 2);
9499 break;
9500 // VECTOR UNPACK extends half the elements of the source vector.
9501 case Intrinsic::s390_vuphb: // VECTOR UNPACK HIGH
9502 case Intrinsic::s390_vuphh:
9503 case Intrinsic::s390_vuphf:
9504 case Intrinsic::s390_vuplhb: // VECTOR UNPACK LOGICAL HIGH
9505 case Intrinsic::s390_vuplhh:
9506 case Intrinsic::s390_vuplhf:
9507 SrcDemE = APInt(NumElts * 2, 0);
9508 SrcDemE.insertBits(DemandedElts, 0);
9509 break;
9510 case Intrinsic::s390_vuplb: // VECTOR UNPACK LOW
9511 case Intrinsic::s390_vuplhw:
9512 case Intrinsic::s390_vuplf:
9513 case Intrinsic::s390_vupllb: // VECTOR UNPACK LOGICAL LOW
9514 case Intrinsic::s390_vupllh:
9515 case Intrinsic::s390_vupllf:
9516 SrcDemE = APInt(NumElts * 2, 0);
9517 SrcDemE.insertBits(DemandedElts, NumElts);
9518 break;
9519 case Intrinsic::s390_vpdi: {
9520 // VECTOR PERMUTE DWORD IMMEDIATE selects one element from each source.
9521 SrcDemE = APInt(NumElts, 0);
9522 if (!DemandedElts[OpNo - 1])
9523 break;
9524 unsigned Mask = Op.getConstantOperandVal(3);
9525 unsigned MaskBit = ((OpNo - 1) ? 1 : 4);
9526 // Demand input element 0 or 1, given by the mask bit value.
9527 SrcDemE.setBit((Mask & MaskBit)? 1 : 0);
9528 break;
9529 }
9530 case Intrinsic::s390_vsldb: {
9531 // VECTOR SHIFT LEFT DOUBLE BY BYTE
9532 assert(VT == MVT::v16i8 && "Unexpected type.");
9533 unsigned FirstIdx = Op.getConstantOperandVal(3);
9534 assert (FirstIdx > 0 && FirstIdx < 16 && "Unused operand.");
9535 unsigned NumSrc0Els = 16 - FirstIdx;
9536 SrcDemE = APInt(NumElts, 0);
9537 if (OpNo == 1) {
9538 APInt DemEls = DemandedElts.trunc(NumSrc0Els);
9539 SrcDemE.insertBits(DemEls, FirstIdx);
9540 } else {
9541 APInt DemEls = DemandedElts.lshr(NumSrc0Els);
9542 SrcDemE.insertBits(DemEls, 0);
9543 }
9544 break;
9545 }
9546 case Intrinsic::s390_vperm:
9547 SrcDemE = APInt::getAllOnes(NumElts);
9548 break;
9549 default:
9550 llvm_unreachable("Unhandled intrinsic.");
9551 break;
9552 }
9553 } else {
9554 switch (Opcode) {
9555 case SystemZISD::JOIN_DWORDS:
9556 // Scalar operand.
9557 SrcDemE = APInt(1, 1);
9558 break;
9559 case SystemZISD::SELECT_CCMASK:
9560 SrcDemE = DemandedElts;
9561 break;
9562 default:
9563 llvm_unreachable("Unhandled opcode.");
9564 break;
9565 }
9566 }
9567 return SrcDemE;
9568}
9569
9571 const APInt &DemandedElts,
9572 const SelectionDAG &DAG, unsigned Depth,
9573 unsigned OpNo) {
9574 APInt Src0DemE = getDemandedSrcElements(Op, DemandedElts, OpNo);
9575 APInt Src1DemE = getDemandedSrcElements(Op, DemandedElts, OpNo + 1);
9576 KnownBits LHSKnown =
9577 DAG.computeKnownBits(Op.getOperand(OpNo), Src0DemE, Depth + 1);
9578 KnownBits RHSKnown =
9579 DAG.computeKnownBits(Op.getOperand(OpNo + 1), Src1DemE, Depth + 1);
9580 Known = LHSKnown.intersectWith(RHSKnown);
9581}
9582
9583void
9586 const APInt &DemandedElts,
9587 const SelectionDAG &DAG,
9588 unsigned Depth) const {
9589 Known.resetAll();
9590
9591 // Intrinsic CC result is returned in the two low bits.
9592 unsigned Tmp0, Tmp1; // not used
9593 if (Op.getResNo() == 1 && isIntrinsicWithCC(Op, Tmp0, Tmp1)) {
9594 Known.Zero.setBitsFrom(2);
9595 return;
9596 }
9597 EVT VT = Op.getValueType();
9598 if (Op.getResNo() != 0 || VT == MVT::Untyped)
9599 return;
9600 assert (Known.getBitWidth() == VT.getScalarSizeInBits() &&
9601 "KnownBits does not match VT in bitwidth");
9602 assert ((!VT.isVector() ||
9603 (DemandedElts.getBitWidth() == VT.getVectorNumElements())) &&
9604 "DemandedElts does not match VT number of elements");
9605 unsigned BitWidth = Known.getBitWidth();
9606 unsigned Opcode = Op.getOpcode();
9607 if (Opcode == ISD::INTRINSIC_WO_CHAIN) {
9608 bool IsLogical = false;
9609 unsigned Id = Op.getConstantOperandVal(0);
9610 switch (Id) {
9611 case Intrinsic::s390_vpksh: // PACKS
9612 case Intrinsic::s390_vpksf:
9613 case Intrinsic::s390_vpksg:
9614 case Intrinsic::s390_vpkshs: // PACKS_CC
9615 case Intrinsic::s390_vpksfs:
9616 case Intrinsic::s390_vpksgs:
9617 case Intrinsic::s390_vpklsh: // PACKLS
9618 case Intrinsic::s390_vpklsf:
9619 case Intrinsic::s390_vpklsg:
9620 case Intrinsic::s390_vpklshs: // PACKLS_CC
9621 case Intrinsic::s390_vpklsfs:
9622 case Intrinsic::s390_vpklsgs:
9623 case Intrinsic::s390_vpdi:
9624 case Intrinsic::s390_vsldb:
9625 case Intrinsic::s390_vperm:
9626 computeKnownBitsBinOp(Op, Known, DemandedElts, DAG, Depth, 1);
9627 break;
9628 case Intrinsic::s390_vuplhb: // VECTOR UNPACK LOGICAL HIGH
9629 case Intrinsic::s390_vuplhh:
9630 case Intrinsic::s390_vuplhf:
9631 case Intrinsic::s390_vupllb: // VECTOR UNPACK LOGICAL LOW
9632 case Intrinsic::s390_vupllh:
9633 case Intrinsic::s390_vupllf:
9634 IsLogical = true;
9635 [[fallthrough]];
9636 case Intrinsic::s390_vuphb: // VECTOR UNPACK HIGH
9637 case Intrinsic::s390_vuphh:
9638 case Intrinsic::s390_vuphf:
9639 case Intrinsic::s390_vuplb: // VECTOR UNPACK LOW
9640 case Intrinsic::s390_vuplhw:
9641 case Intrinsic::s390_vuplf: {
9642 SDValue SrcOp = Op.getOperand(1);
9643 APInt SrcDemE = getDemandedSrcElements(Op, DemandedElts, 0);
9644 Known = DAG.computeKnownBits(SrcOp, SrcDemE, Depth + 1);
9645 if (IsLogical) {
9646 Known = Known.zext(BitWidth);
9647 } else
9648 Known = Known.sext(BitWidth);
9649 break;
9650 }
9651 default:
9652 break;
9653 }
9654 } else {
9655 switch (Opcode) {
9656 case SystemZISD::JOIN_DWORDS:
9657 case SystemZISD::SELECT_CCMASK:
9658 computeKnownBitsBinOp(Op, Known, DemandedElts, DAG, Depth, 0);
9659 break;
9660 case SystemZISD::REPLICATE: {
9661 SDValue SrcOp = Op.getOperand(0);
9662 Known = DAG.computeKnownBits(SrcOp, Depth + 1);
9663 if (Known.getBitWidth() < BitWidth && isa<ConstantSDNode>(SrcOp))
9664 Known = Known.sext(BitWidth); // VREPI sign extends the immedate.
9665 break;
9666 }
9667 default:
9668 break;
9669 }
9670 }
9671
9672 // Known has the width of the source operand(s). Adjust if needed to match
9673 // the passed bitwidth.
9674 if (Known.getBitWidth() != BitWidth)
9675 Known = Known.anyextOrTrunc(BitWidth);
9676}
9677
9678static unsigned computeNumSignBitsBinOp(SDValue Op, const APInt &DemandedElts,
9679 const SelectionDAG &DAG, unsigned Depth,
9680 unsigned OpNo) {
9681 APInt Src0DemE = getDemandedSrcElements(Op, DemandedElts, OpNo);
9682 unsigned LHS = DAG.ComputeNumSignBits(Op.getOperand(OpNo), Src0DemE, Depth + 1);
9683 if (LHS == 1) return 1; // Early out.
9684 APInt Src1DemE = getDemandedSrcElements(Op, DemandedElts, OpNo + 1);
9685 unsigned RHS = DAG.ComputeNumSignBits(Op.getOperand(OpNo + 1), Src1DemE, Depth + 1);
9686 if (RHS == 1) return 1; // Early out.
9687 unsigned Common = std::min(LHS, RHS);
9688 unsigned SrcBitWidth = Op.getOperand(OpNo).getScalarValueSizeInBits();
9689 EVT VT = Op.getValueType();
9690 unsigned VTBits = VT.getScalarSizeInBits();
9691 if (SrcBitWidth > VTBits) { // PACK
9692 unsigned SrcExtraBits = SrcBitWidth - VTBits;
9693 if (Common > SrcExtraBits)
9694 return (Common - SrcExtraBits);
9695 return 1;
9696 }
9697 assert (SrcBitWidth == VTBits && "Expected operands of same bitwidth.");
9698 return Common;
9699}
9700
9701unsigned
9703 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
9704 unsigned Depth) const {
9705 if (Op.getResNo() != 0)
9706 return 1;
9707 unsigned Opcode = Op.getOpcode();
9708 if (Opcode == ISD::INTRINSIC_WO_CHAIN) {
9709 unsigned Id = Op.getConstantOperandVal(0);
9710 switch (Id) {
9711 case Intrinsic::s390_vpksh: // PACKS
9712 case Intrinsic::s390_vpksf:
9713 case Intrinsic::s390_vpksg:
9714 case Intrinsic::s390_vpkshs: // PACKS_CC
9715 case Intrinsic::s390_vpksfs:
9716 case Intrinsic::s390_vpksgs:
9717 case Intrinsic::s390_vpklsh: // PACKLS
9718 case Intrinsic::s390_vpklsf:
9719 case Intrinsic::s390_vpklsg:
9720 case Intrinsic::s390_vpklshs: // PACKLS_CC
9721 case Intrinsic::s390_vpklsfs:
9722 case Intrinsic::s390_vpklsgs:
9723 case Intrinsic::s390_vpdi:
9724 case Intrinsic::s390_vsldb:
9725 case Intrinsic::s390_vperm:
9726 return computeNumSignBitsBinOp(Op, DemandedElts, DAG, Depth, 1);
9727 case Intrinsic::s390_vuphb: // VECTOR UNPACK HIGH
9728 case Intrinsic::s390_vuphh:
9729 case Intrinsic::s390_vuphf:
9730 case Intrinsic::s390_vuplb: // VECTOR UNPACK LOW
9731 case Intrinsic::s390_vuplhw:
9732 case Intrinsic::s390_vuplf: {
9733 SDValue PackedOp = Op.getOperand(1);
9734 APInt SrcDemE = getDemandedSrcElements(Op, DemandedElts, 1);
9735 unsigned Tmp = DAG.ComputeNumSignBits(PackedOp, SrcDemE, Depth + 1);
9736 EVT VT = Op.getValueType();
9737 unsigned VTBits = VT.getScalarSizeInBits();
9738 Tmp += VTBits - PackedOp.getScalarValueSizeInBits();
9739 return Tmp;
9740 }
9741 default:
9742 break;
9743 }
9744 } else {
9745 switch (Opcode) {
9746 case SystemZISD::SELECT_CCMASK:
9747 return computeNumSignBitsBinOp(Op, DemandedElts, DAG, Depth, 0);
9748 default:
9749 break;
9750 }
9751 }
9752
9753 return 1;
9754}
9755
9757 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
9758 UndefPoisonKind Kind, unsigned Depth) const {
9759 switch (Op->getOpcode()) {
9760 case SystemZISD::PCREL_WRAPPER:
9761 case SystemZISD::PCREL_OFFSET:
9762 return true;
9763 }
9764 return false;
9765}
9766
9767unsigned
9769 const TargetFrameLowering *TFI = Subtarget.getFrameLowering();
9770 unsigned StackAlign = TFI->getStackAlignment();
9771 assert(StackAlign >=1 && isPowerOf2_32(StackAlign) &&
9772 "Unexpected stack alignment");
9773 // The default stack probe size is 4096 if the function has no
9774 // stack-probe-size attribute.
9775 unsigned StackProbeSize =
9776 MF.getFunction().getFnAttributeAsParsedInteger("stack-probe-size", 4096);
9777 // Round down to the stack alignment.
9778 StackProbeSize &= ~(StackAlign - 1);
9779 return StackProbeSize ? StackProbeSize : StackAlign;
9780}
9781
9782//===----------------------------------------------------------------------===//
9783// Custom insertion
9784//===----------------------------------------------------------------------===//
9785
9786// Force base value Base into a register before MI. Return the register.
9788 const SystemZInstrInfo *TII) {
9789 MachineBasicBlock *MBB = MI.getParent();
9790 MachineFunction &MF = *MBB->getParent();
9791 MachineRegisterInfo &MRI = MF.getRegInfo();
9792
9793 if (Base.isReg()) {
9794 // Copy Base into a new virtual register to help register coalescing in
9795 // cases with multiple uses.
9796 Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
9797 BuildMI(*MBB, MI, MI.getDebugLoc(), TII->get(SystemZ::COPY), Reg)
9798 .add(Base);
9799 return Reg;
9800 }
9801
9802 Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
9803 BuildMI(*MBB, MI, MI.getDebugLoc(), TII->get(SystemZ::LA), Reg)
9804 .add(Base)
9805 .addImm(0)
9806 .addReg(0);
9807 return Reg;
9808}
9809
9810// The CC operand of MI might be missing a kill marker because there
9811// were multiple uses of CC, and ISel didn't know which to mark.
9812// Figure out whether MI should have had a kill marker.
9814 // Scan forward through BB for a use/def of CC.
9816 for (MachineBasicBlock::iterator miE = MBB->end(); miI != miE; ++miI) {
9817 const MachineInstr &MI = *miI;
9818 if (MI.readsRegister(SystemZ::CC, /*TRI=*/nullptr))
9819 return false;
9820 if (MI.definesRegister(SystemZ::CC, /*TRI=*/nullptr))
9821 break; // Should have kill-flag - update below.
9822 }
9823
9824 // If we hit the end of the block, check whether CC is live into a
9825 // successor.
9826 if (miI == MBB->end()) {
9827 for (const MachineBasicBlock *Succ : MBB->successors())
9828 if (Succ->isLiveIn(SystemZ::CC))
9829 return false;
9830 }
9831
9832 return true;
9833}
9834
9835// Return true if it is OK for this Select pseudo-opcode to be cascaded
9836// together with other Select pseudo-opcodes into a single basic-block with
9837// a conditional jump around it.
9839 switch (MI.getOpcode()) {
9840 case SystemZ::Select32:
9841 case SystemZ::Select64:
9842 case SystemZ::Select128:
9843 case SystemZ::SelectF32:
9844 case SystemZ::SelectF64:
9845 case SystemZ::SelectF128:
9846 case SystemZ::SelectVR32:
9847 case SystemZ::SelectVR64:
9848 case SystemZ::SelectVR128:
9849 return true;
9850
9851 default:
9852 return false;
9853 }
9854}
9855
9856// Helper function, which inserts PHI functions into SinkMBB:
9857// %Result(i) = phi [ %FalseValue(i), FalseMBB ], [ %TrueValue(i), TrueMBB ],
9858// where %FalseValue(i) and %TrueValue(i) are taken from Selects.
9860 MachineBasicBlock *TrueMBB,
9861 MachineBasicBlock *FalseMBB,
9862 MachineBasicBlock *SinkMBB) {
9863 MachineFunction *MF = TrueMBB->getParent();
9865
9866 MachineInstr *FirstMI = Selects.front();
9867 unsigned CCValid = FirstMI->getOperand(3).getImm();
9868 unsigned CCMask = FirstMI->getOperand(4).getImm();
9869
9870 MachineBasicBlock::iterator SinkInsertionPoint = SinkMBB->begin();
9871
9872 // As we are creating the PHIs, we have to be careful if there is more than
9873 // one. Later Selects may reference the results of earlier Selects, but later
9874 // PHIs have to reference the individual true/false inputs from earlier PHIs.
9875 // That also means that PHI construction must work forward from earlier to
9876 // later, and that the code must maintain a mapping from earlier PHI's
9877 // destination registers, and the registers that went into the PHI.
9879
9880 for (auto *MI : Selects) {
9881 Register DestReg = MI->getOperand(0).getReg();
9882 Register TrueReg = MI->getOperand(1).getReg();
9883 Register FalseReg = MI->getOperand(2).getReg();
9884
9885 // If this Select we are generating is the opposite condition from
9886 // the jump we generated, then we have to swap the operands for the
9887 // PHI that is going to be generated.
9888 if (MI->getOperand(4).getImm() == (CCValid ^ CCMask))
9889 std::swap(TrueReg, FalseReg);
9890
9891 if (auto It = RegRewriteTable.find(TrueReg); It != RegRewriteTable.end())
9892 TrueReg = It->second.first;
9893
9894 if (auto It = RegRewriteTable.find(FalseReg); It != RegRewriteTable.end())
9895 FalseReg = It->second.second;
9896
9897 DebugLoc DL = MI->getDebugLoc();
9898 BuildMI(*SinkMBB, SinkInsertionPoint, DL, TII->get(SystemZ::PHI), DestReg)
9899 .addReg(TrueReg).addMBB(TrueMBB)
9900 .addReg(FalseReg).addMBB(FalseMBB);
9901
9902 // Add this PHI to the rewrite table.
9903 RegRewriteTable[DestReg] = std::make_pair(TrueReg, FalseReg);
9904 }
9905
9906 MF->getProperties().resetNoPHIs();
9907}
9908
9910SystemZTargetLowering::emitAdjCallStack(MachineInstr &MI,
9911 MachineBasicBlock *BB) const {
9912 MachineFunction &MF = *BB->getParent();
9913 MachineFrameInfo &MFI = MF.getFrameInfo();
9914 auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
9915 assert(TFL->hasReservedCallFrame(MF) &&
9916 "ADJSTACKDOWN and ADJSTACKUP should be no-ops");
9917 (void)TFL;
9918 // Get the MaxCallFrameSize value and erase MI since it serves no further
9919 // purpose as the call frame is statically reserved in the prolog. Set
9920 // AdjustsStack as MI is *not* mapped as a frame instruction.
9921 uint32_t NumBytes = MI.getOperand(0).getImm();
9922 if (NumBytes > MFI.getMaxCallFrameSize())
9923 MFI.setMaxCallFrameSize(NumBytes);
9924 MFI.setAdjustsStack(true);
9925
9926 MI.eraseFromParent();
9927 return BB;
9928}
9929
9930// Implement EmitInstrWithCustomInserter for pseudo Select* instruction MI.
9932SystemZTargetLowering::emitSelect(MachineInstr &MI,
9933 MachineBasicBlock *MBB) const {
9934 assert(isSelectPseudo(MI) && "Bad call to emitSelect()");
9935 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
9936
9937 unsigned CCValid = MI.getOperand(3).getImm();
9938 unsigned CCMask = MI.getOperand(4).getImm();
9939
9940 // If we have a sequence of Select* pseudo instructions using the
9941 // same condition code value, we want to expand all of them into
9942 // a single pair of basic blocks using the same condition.
9943 SmallVector<MachineInstr*, 8> Selects;
9944 SmallVector<MachineInstr*, 8> DbgValues;
9945 Selects.push_back(&MI);
9946 unsigned Count = 0;
9947 for (MachineInstr &NextMI : llvm::make_range(
9948 std::next(MachineBasicBlock::iterator(MI)), MBB->end())) {
9949 if (isSelectPseudo(NextMI)) {
9950 assert(NextMI.getOperand(3).getImm() == CCValid &&
9951 "Bad CCValid operands since CC was not redefined.");
9952 if (NextMI.getOperand(4).getImm() == CCMask ||
9953 NextMI.getOperand(4).getImm() == (CCValid ^ CCMask)) {
9954 Selects.push_back(&NextMI);
9955 continue;
9956 }
9957 break;
9958 }
9959 if (NextMI.definesRegister(SystemZ::CC, /*TRI=*/nullptr) ||
9960 NextMI.usesCustomInsertionHook())
9961 break;
9962 bool User = false;
9963 for (auto *SelMI : Selects)
9964 if (NextMI.readsVirtualRegister(SelMI->getOperand(0).getReg())) {
9965 User = true;
9966 break;
9967 }
9968 if (NextMI.isDebugInstr()) {
9969 if (User) {
9970 assert(NextMI.isDebugValue() && "Unhandled debug opcode.");
9971 DbgValues.push_back(&NextMI);
9972 }
9973 } else if (User || ++Count > 20)
9974 break;
9975 }
9976
9977 MachineInstr *LastMI = Selects.back();
9978 bool CCKilled = (LastMI->killsRegister(SystemZ::CC, /*TRI=*/nullptr) ||
9979 checkCCKill(*LastMI, MBB));
9980 MachineBasicBlock *StartMBB = MBB;
9981 MachineBasicBlock *JoinMBB = SystemZ::splitBlockAfter(LastMI, MBB);
9982 MachineBasicBlock *FalseMBB = SystemZ::emitBlockAfter(StartMBB);
9983
9984 // Unless CC was killed in the last Select instruction, mark it as
9985 // live-in to both FalseMBB and JoinMBB.
9986 if (!CCKilled) {
9987 FalseMBB->addLiveIn(SystemZ::CC);
9988 JoinMBB->addLiveIn(SystemZ::CC);
9989 }
9990
9991 // StartMBB:
9992 // BRC CCMask, JoinMBB
9993 // # fallthrough to FalseMBB
9994 MBB = StartMBB;
9995 BuildMI(MBB, MI.getDebugLoc(), TII->get(SystemZ::BRC))
9996 .addImm(CCValid).addImm(CCMask).addMBB(JoinMBB);
9997 MBB->addSuccessor(JoinMBB);
9998 MBB->addSuccessor(FalseMBB);
9999
10000 // FalseMBB:
10001 // # fallthrough to JoinMBB
10002 MBB = FalseMBB;
10003 MBB->addSuccessor(JoinMBB);
10004
10005 // JoinMBB:
10006 // %Result = phi [ %FalseReg, FalseMBB ], [ %TrueReg, StartMBB ]
10007 // ...
10008 MBB = JoinMBB;
10009 createPHIsForSelects(Selects, StartMBB, FalseMBB, MBB);
10010 for (auto *SelMI : Selects)
10011 SelMI->eraseFromParent();
10012
10014 for (auto *DbgMI : DbgValues)
10015 MBB->splice(InsertPos, StartMBB, DbgMI);
10016
10017 return JoinMBB;
10018}
10019
10020// Implement EmitInstrWithCustomInserter for pseudo CondStore* instruction MI.
10021// StoreOpcode is the store to use and Invert says whether the store should
10022// happen when the condition is false rather than true. If a STORE ON
10023// CONDITION is available, STOCOpcode is its opcode, otherwise it is 0.
10024MachineBasicBlock *SystemZTargetLowering::emitCondStore(MachineInstr &MI,
10026 unsigned StoreOpcode,
10027 unsigned STOCOpcode,
10028 bool Invert) const {
10029 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10030
10031 Register SrcReg = MI.getOperand(0).getReg();
10032 MachineOperand Base = MI.getOperand(1);
10033 int64_t Disp = MI.getOperand(2).getImm();
10034 Register IndexReg = MI.getOperand(3).getReg();
10035 unsigned CCValid = MI.getOperand(4).getImm();
10036 unsigned CCMask = MI.getOperand(5).getImm();
10037 DebugLoc DL = MI.getDebugLoc();
10038
10039 StoreOpcode = TII->getOpcodeForOffset(StoreOpcode, Disp);
10040
10041 // ISel pattern matching also adds a load memory operand of the same
10042 // address, so take special care to find the storing memory operand.
10043 MachineMemOperand *MMO = nullptr;
10044 for (auto *I : MI.memoperands())
10045 if (I->isStore()) {
10046 MMO = I;
10047 break;
10048 }
10049
10050 // Use STOCOpcode if possible. We could use different store patterns in
10051 // order to avoid matching the index register, but the performance trade-offs
10052 // might be more complicated in that case.
10053 if (STOCOpcode && !IndexReg && Subtarget.hasLoadStoreOnCond()) {
10054 if (Invert)
10055 CCMask ^= CCValid;
10056
10057 BuildMI(*MBB, MI, DL, TII->get(STOCOpcode))
10058 .addReg(SrcReg)
10059 .add(Base)
10060 .addImm(Disp)
10061 .addImm(CCValid)
10062 .addImm(CCMask)
10063 .addMemOperand(MMO);
10064
10065 MI.eraseFromParent();
10066 return MBB;
10067 }
10068
10069 // Get the condition needed to branch around the store.
10070 if (!Invert)
10071 CCMask ^= CCValid;
10072
10073 MachineBasicBlock *StartMBB = MBB;
10074 MachineBasicBlock *JoinMBB = SystemZ::splitBlockBefore(MI, MBB);
10075 MachineBasicBlock *FalseMBB = SystemZ::emitBlockAfter(StartMBB);
10076
10077 // Unless CC was killed in the CondStore instruction, mark it as
10078 // live-in to both FalseMBB and JoinMBB.
10079 if (!MI.killsRegister(SystemZ::CC, /*TRI=*/nullptr) &&
10080 !checkCCKill(MI, JoinMBB)) {
10081 FalseMBB->addLiveIn(SystemZ::CC);
10082 JoinMBB->addLiveIn(SystemZ::CC);
10083 }
10084
10085 // StartMBB:
10086 // BRC CCMask, JoinMBB
10087 // # fallthrough to FalseMBB
10088 MBB = StartMBB;
10089 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10090 .addImm(CCValid).addImm(CCMask).addMBB(JoinMBB);
10091 MBB->addSuccessor(JoinMBB);
10092 MBB->addSuccessor(FalseMBB);
10093
10094 // FalseMBB:
10095 // store %SrcReg, %Disp(%Index,%Base)
10096 // # fallthrough to JoinMBB
10097 MBB = FalseMBB;
10098 BuildMI(MBB, DL, TII->get(StoreOpcode))
10099 .addReg(SrcReg)
10100 .add(Base)
10101 .addImm(Disp)
10102 .addReg(IndexReg)
10103 .addMemOperand(MMO);
10104 MBB->addSuccessor(JoinMBB);
10105
10106 MI.eraseFromParent();
10107 return JoinMBB;
10108}
10109
10110// Implement EmitInstrWithCustomInserter for pseudo [SU]Cmp128Hi instruction MI.
10112SystemZTargetLowering::emitICmp128Hi(MachineInstr &MI,
10114 bool Unsigned) const {
10115 MachineFunction &MF = *MBB->getParent();
10116 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10117 MachineRegisterInfo &MRI = MF.getRegInfo();
10118
10119 // Synthetic instruction to compare 128-bit values.
10120 // Sets CC 1 if Op0 > Op1, sets a different CC otherwise.
10121 Register Op0 = MI.getOperand(0).getReg();
10122 Register Op1 = MI.getOperand(1).getReg();
10123
10124 MachineBasicBlock *StartMBB = MBB;
10125 MachineBasicBlock *JoinMBB = SystemZ::splitBlockAfter(MI, MBB);
10126 MachineBasicBlock *HiEqMBB = SystemZ::emitBlockAfter(StartMBB);
10127
10128 // StartMBB:
10129 //
10130 // Use VECTOR ELEMENT COMPARE [LOGICAL] to compare the high parts.
10131 // Swap the inputs to get:
10132 // CC 1 if high(Op0) > high(Op1)
10133 // CC 2 if high(Op0) < high(Op1)
10134 // CC 0 if high(Op0) == high(Op1)
10135 //
10136 // If CC != 0, we'd done, so jump over the next instruction.
10137 //
10138 // VEC[L]G Op1, Op0
10139 // JNE JoinMBB
10140 // # fallthrough to HiEqMBB
10141 MBB = StartMBB;
10142 int HiOpcode = Unsigned? SystemZ::VECLG : SystemZ::VECG;
10143 BuildMI(MBB, MI.getDebugLoc(), TII->get(HiOpcode))
10144 .addReg(Op1).addReg(Op0);
10145 BuildMI(MBB, MI.getDebugLoc(), TII->get(SystemZ::BRC))
10147 MBB->addSuccessor(JoinMBB);
10148 MBB->addSuccessor(HiEqMBB);
10149
10150 // HiEqMBB:
10151 //
10152 // Otherwise, use VECTOR COMPARE HIGH LOGICAL.
10153 // Since we already know the high parts are equal, the CC
10154 // result will only depend on the low parts:
10155 // CC 1 if low(Op0) > low(Op1)
10156 // CC 3 if low(Op0) <= low(Op1)
10157 //
10158 // VCHLGS Tmp, Op0, Op1
10159 // # fallthrough to JoinMBB
10160 MBB = HiEqMBB;
10161 Register Temp = MRI.createVirtualRegister(&SystemZ::VR128BitRegClass);
10162 BuildMI(MBB, MI.getDebugLoc(), TII->get(SystemZ::VCHLGS), Temp)
10163 .addReg(Op0).addReg(Op1);
10164 MBB->addSuccessor(JoinMBB);
10165
10166 // Mark CC as live-in to JoinMBB.
10167 JoinMBB->addLiveIn(SystemZ::CC);
10168
10169 MI.eraseFromParent();
10170 return JoinMBB;
10171}
10172
10173// Implement EmitInstrWithCustomInserter for subword pseudo ATOMIC_LOADW_* or
10174// ATOMIC_SWAPW instruction MI. BinOpcode is the instruction that performs
10175// the binary operation elided by "*", or 0 for ATOMIC_SWAPW. Invert says
10176// whether the field should be inverted after performing BinOpcode (e.g. for
10177// NAND).
10178MachineBasicBlock *SystemZTargetLowering::emitAtomicLoadBinary(
10179 MachineInstr &MI, MachineBasicBlock *MBB, unsigned BinOpcode,
10180 bool Invert) const {
10181 MachineFunction &MF = *MBB->getParent();
10182 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10183 MachineRegisterInfo &MRI = MF.getRegInfo();
10184
10185 // Extract the operands. Base can be a register or a frame index.
10186 // Src2 can be a register or immediate.
10187 Register Dest = MI.getOperand(0).getReg();
10188 MachineOperand Base = earlyUseOperand(MI.getOperand(1));
10189 int64_t Disp = MI.getOperand(2).getImm();
10190 MachineOperand Src2 = earlyUseOperand(MI.getOperand(3));
10191 Register BitShift = MI.getOperand(4).getReg();
10192 Register NegBitShift = MI.getOperand(5).getReg();
10193 unsigned BitSize = MI.getOperand(6).getImm();
10194 DebugLoc DL = MI.getDebugLoc();
10195
10196 // Get the right opcodes for the displacement.
10197 unsigned LOpcode = TII->getOpcodeForOffset(SystemZ::L, Disp);
10198 unsigned CSOpcode = TII->getOpcodeForOffset(SystemZ::CS, Disp);
10199 assert(LOpcode && CSOpcode && "Displacement out of range");
10200
10201 // Create virtual registers for temporary results.
10202 Register OrigVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10203 Register OldVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10204 Register NewVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10205 Register RotatedOldVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10206 Register RotatedNewVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10207
10208 // Insert a basic block for the main loop.
10209 MachineBasicBlock *StartMBB = MBB;
10210 MachineBasicBlock *DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10211 MachineBasicBlock *LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10212
10213 // StartMBB:
10214 // ...
10215 // %OrigVal = L Disp(%Base)
10216 // # fall through to LoopMBB
10217 MBB = StartMBB;
10218 BuildMI(MBB, DL, TII->get(LOpcode), OrigVal).add(Base).addImm(Disp).addReg(0);
10219 MBB->addSuccessor(LoopMBB);
10220
10221 // LoopMBB:
10222 // %OldVal = phi [ %OrigVal, StartMBB ], [ %Dest, LoopMBB ]
10223 // %RotatedOldVal = RLL %OldVal, 0(%BitShift)
10224 // %RotatedNewVal = OP %RotatedOldVal, %Src2
10225 // %NewVal = RLL %RotatedNewVal, 0(%NegBitShift)
10226 // %Dest = CS %OldVal, %NewVal, Disp(%Base)
10227 // JNE LoopMBB
10228 // # fall through to DoneMBB
10229 MBB = LoopMBB;
10230 BuildMI(MBB, DL, TII->get(SystemZ::PHI), OldVal)
10231 .addReg(OrigVal).addMBB(StartMBB)
10232 .addReg(Dest).addMBB(LoopMBB);
10233 BuildMI(MBB, DL, TII->get(SystemZ::RLL), RotatedOldVal)
10234 .addReg(OldVal).addReg(BitShift).addImm(0);
10235 if (Invert) {
10236 // Perform the operation normally and then invert every bit of the field.
10237 Register Tmp = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10238 BuildMI(MBB, DL, TII->get(BinOpcode), Tmp).addReg(RotatedOldVal).add(Src2);
10239 // XILF with the upper BitSize bits set.
10240 BuildMI(MBB, DL, TII->get(SystemZ::XILF), RotatedNewVal)
10241 .addReg(Tmp).addImm(-1U << (32 - BitSize));
10242 } else if (BinOpcode)
10243 // A simply binary operation.
10244 BuildMI(MBB, DL, TII->get(BinOpcode), RotatedNewVal)
10245 .addReg(RotatedOldVal)
10246 .add(Src2);
10247 else
10248 // Use RISBG to rotate Src2 into position and use it to replace the
10249 // field in RotatedOldVal.
10250 BuildMI(MBB, DL, TII->get(SystemZ::RISBG32), RotatedNewVal)
10251 .addReg(RotatedOldVal).addReg(Src2.getReg())
10252 .addImm(32).addImm(31 + BitSize).addImm(32 - BitSize);
10253 BuildMI(MBB, DL, TII->get(SystemZ::RLL), NewVal)
10254 .addReg(RotatedNewVal).addReg(NegBitShift).addImm(0);
10255 BuildMI(MBB, DL, TII->get(CSOpcode), Dest)
10256 .addReg(OldVal)
10257 .addReg(NewVal)
10258 .add(Base)
10259 .addImm(Disp);
10260 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10262 MBB->addSuccessor(LoopMBB);
10263 MBB->addSuccessor(DoneMBB);
10264
10265 MI.eraseFromParent();
10266 return DoneMBB;
10267}
10268
10269// Implement EmitInstrWithCustomInserter for subword pseudo
10270// ATOMIC_LOADW_{,U}{MIN,MAX} instruction MI. CompareOpcode is the
10271// instruction that should be used to compare the current field with the
10272// minimum or maximum value. KeepOldMask is the BRC condition-code mask
10273// for when the current field should be kept.
10274MachineBasicBlock *SystemZTargetLowering::emitAtomicLoadMinMax(
10275 MachineInstr &MI, MachineBasicBlock *MBB, unsigned CompareOpcode,
10276 unsigned KeepOldMask) const {
10277 MachineFunction &MF = *MBB->getParent();
10278 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10279 MachineRegisterInfo &MRI = MF.getRegInfo();
10280
10281 // Extract the operands. Base can be a register or a frame index.
10282 Register Dest = MI.getOperand(0).getReg();
10283 MachineOperand Base = earlyUseOperand(MI.getOperand(1));
10284 int64_t Disp = MI.getOperand(2).getImm();
10285 Register Src2 = MI.getOperand(3).getReg();
10286 Register BitShift = MI.getOperand(4).getReg();
10287 Register NegBitShift = MI.getOperand(5).getReg();
10288 unsigned BitSize = MI.getOperand(6).getImm();
10289 DebugLoc DL = MI.getDebugLoc();
10290
10291 // Get the right opcodes for the displacement.
10292 unsigned LOpcode = TII->getOpcodeForOffset(SystemZ::L, Disp);
10293 unsigned CSOpcode = TII->getOpcodeForOffset(SystemZ::CS, Disp);
10294 assert(LOpcode && CSOpcode && "Displacement out of range");
10295
10296 // Create virtual registers for temporary results.
10297 Register OrigVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10298 Register OldVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10299 Register NewVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10300 Register RotatedOldVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10301 Register RotatedAltVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10302 Register RotatedNewVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10303
10304 // Insert 3 basic blocks for the loop.
10305 MachineBasicBlock *StartMBB = MBB;
10306 MachineBasicBlock *DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10307 MachineBasicBlock *LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10308 MachineBasicBlock *UseAltMBB = SystemZ::emitBlockAfter(LoopMBB);
10309 MachineBasicBlock *UpdateMBB = SystemZ::emitBlockAfter(UseAltMBB);
10310
10311 // StartMBB:
10312 // ...
10313 // %OrigVal = L Disp(%Base)
10314 // # fall through to LoopMBB
10315 MBB = StartMBB;
10316 BuildMI(MBB, DL, TII->get(LOpcode), OrigVal).add(Base).addImm(Disp).addReg(0);
10317 MBB->addSuccessor(LoopMBB);
10318
10319 // LoopMBB:
10320 // %OldVal = phi [ %OrigVal, StartMBB ], [ %Dest, UpdateMBB ]
10321 // %RotatedOldVal = RLL %OldVal, 0(%BitShift)
10322 // CompareOpcode %RotatedOldVal, %Src2
10323 // BRC KeepOldMask, UpdateMBB
10324 MBB = LoopMBB;
10325 BuildMI(MBB, DL, TII->get(SystemZ::PHI), OldVal)
10326 .addReg(OrigVal).addMBB(StartMBB)
10327 .addReg(Dest).addMBB(UpdateMBB);
10328 BuildMI(MBB, DL, TII->get(SystemZ::RLL), RotatedOldVal)
10329 .addReg(OldVal).addReg(BitShift).addImm(0);
10330 BuildMI(MBB, DL, TII->get(CompareOpcode))
10331 .addReg(RotatedOldVal).addReg(Src2);
10332 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10333 .addImm(SystemZ::CCMASK_ICMP).addImm(KeepOldMask).addMBB(UpdateMBB);
10334 MBB->addSuccessor(UpdateMBB);
10335 MBB->addSuccessor(UseAltMBB);
10336
10337 // UseAltMBB:
10338 // %RotatedAltVal = RISBG %RotatedOldVal, %Src2, 32, 31 + BitSize, 0
10339 // # fall through to UpdateMBB
10340 MBB = UseAltMBB;
10341 BuildMI(MBB, DL, TII->get(SystemZ::RISBG32), RotatedAltVal)
10342 .addReg(RotatedOldVal).addReg(Src2)
10343 .addImm(32).addImm(31 + BitSize).addImm(0);
10344 MBB->addSuccessor(UpdateMBB);
10345
10346 // UpdateMBB:
10347 // %RotatedNewVal = PHI [ %RotatedOldVal, LoopMBB ],
10348 // [ %RotatedAltVal, UseAltMBB ]
10349 // %NewVal = RLL %RotatedNewVal, 0(%NegBitShift)
10350 // %Dest = CS %OldVal, %NewVal, Disp(%Base)
10351 // JNE LoopMBB
10352 // # fall through to DoneMBB
10353 MBB = UpdateMBB;
10354 BuildMI(MBB, DL, TII->get(SystemZ::PHI), RotatedNewVal)
10355 .addReg(RotatedOldVal).addMBB(LoopMBB)
10356 .addReg(RotatedAltVal).addMBB(UseAltMBB);
10357 BuildMI(MBB, DL, TII->get(SystemZ::RLL), NewVal)
10358 .addReg(RotatedNewVal).addReg(NegBitShift).addImm(0);
10359 BuildMI(MBB, DL, TII->get(CSOpcode), Dest)
10360 .addReg(OldVal)
10361 .addReg(NewVal)
10362 .add(Base)
10363 .addImm(Disp);
10364 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10366 MBB->addSuccessor(LoopMBB);
10367 MBB->addSuccessor(DoneMBB);
10368
10369 MI.eraseFromParent();
10370 return DoneMBB;
10371}
10372
10373// Implement EmitInstrWithCustomInserter for subword pseudo ATOMIC_CMP_SWAPW
10374// instruction MI.
10376SystemZTargetLowering::emitAtomicCmpSwapW(MachineInstr &MI,
10377 MachineBasicBlock *MBB) const {
10378 MachineFunction &MF = *MBB->getParent();
10379 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10380 MachineRegisterInfo &MRI = MF.getRegInfo();
10381
10382 // Extract the operands. Base can be a register or a frame index.
10383 Register Dest = MI.getOperand(0).getReg();
10384 MachineOperand Base = earlyUseOperand(MI.getOperand(1));
10385 int64_t Disp = MI.getOperand(2).getImm();
10386 Register CmpVal = MI.getOperand(3).getReg();
10387 Register OrigSwapVal = MI.getOperand(4).getReg();
10388 Register BitShift = MI.getOperand(5).getReg();
10389 Register NegBitShift = MI.getOperand(6).getReg();
10390 int64_t BitSize = MI.getOperand(7).getImm();
10391 DebugLoc DL = MI.getDebugLoc();
10392
10393 const TargetRegisterClass *RC = &SystemZ::GR32BitRegClass;
10394
10395 // Get the right opcodes for the displacement and zero-extension.
10396 unsigned LOpcode = TII->getOpcodeForOffset(SystemZ::L, Disp);
10397 unsigned CSOpcode = TII->getOpcodeForOffset(SystemZ::CS, Disp);
10398 unsigned ZExtOpcode = BitSize == 8 ? SystemZ::LLCR : SystemZ::LLHR;
10399 assert(LOpcode && CSOpcode && "Displacement out of range");
10400
10401 // Create virtual registers for temporary results.
10402 Register OrigOldVal = MRI.createVirtualRegister(RC);
10403 Register OldVal = MRI.createVirtualRegister(RC);
10404 Register SwapVal = MRI.createVirtualRegister(RC);
10405 Register StoreVal = MRI.createVirtualRegister(RC);
10406 Register OldValRot = MRI.createVirtualRegister(RC);
10407 Register RetryOldVal = MRI.createVirtualRegister(RC);
10408 Register RetrySwapVal = MRI.createVirtualRegister(RC);
10409
10410 // Insert 2 basic blocks for the loop.
10411 MachineBasicBlock *StartMBB = MBB;
10412 MachineBasicBlock *DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10413 MachineBasicBlock *LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10414 MachineBasicBlock *SetMBB = SystemZ::emitBlockAfter(LoopMBB);
10415
10416 // StartMBB:
10417 // ...
10418 // %OrigOldVal = L Disp(%Base)
10419 // # fall through to LoopMBB
10420 MBB = StartMBB;
10421 BuildMI(MBB, DL, TII->get(LOpcode), OrigOldVal)
10422 .add(Base)
10423 .addImm(Disp)
10424 .addReg(0);
10425 MBB->addSuccessor(LoopMBB);
10426
10427 // LoopMBB:
10428 // %OldVal = phi [ %OrigOldVal, EntryBB ], [ %RetryOldVal, SetMBB ]
10429 // %SwapVal = phi [ %OrigSwapVal, EntryBB ], [ %RetrySwapVal, SetMBB ]
10430 // %OldValRot = RLL %OldVal, BitSize(%BitShift)
10431 // ^^ The low BitSize bits contain the field
10432 // of interest.
10433 // %RetrySwapVal = RISBG32 %SwapVal, %OldValRot, 32, 63-BitSize, 0
10434 // ^^ Replace the upper 32-BitSize bits of the
10435 // swap value with those that we loaded and rotated.
10436 // %Dest = LL[CH] %OldValRot
10437 // CR %Dest, %CmpVal
10438 // JNE DoneMBB
10439 // # Fall through to SetMBB
10440 MBB = LoopMBB;
10441 BuildMI(MBB, DL, TII->get(SystemZ::PHI), OldVal)
10442 .addReg(OrigOldVal).addMBB(StartMBB)
10443 .addReg(RetryOldVal).addMBB(SetMBB);
10444 BuildMI(MBB, DL, TII->get(SystemZ::PHI), SwapVal)
10445 .addReg(OrigSwapVal).addMBB(StartMBB)
10446 .addReg(RetrySwapVal).addMBB(SetMBB);
10447 BuildMI(MBB, DL, TII->get(SystemZ::RLL), OldValRot)
10448 .addReg(OldVal).addReg(BitShift).addImm(BitSize);
10449 BuildMI(MBB, DL, TII->get(SystemZ::RISBG32), RetrySwapVal)
10450 .addReg(SwapVal).addReg(OldValRot).addImm(32).addImm(63 - BitSize).addImm(0);
10451 BuildMI(MBB, DL, TII->get(ZExtOpcode), Dest)
10452 .addReg(OldValRot);
10453 BuildMI(MBB, DL, TII->get(SystemZ::CR))
10454 .addReg(Dest).addReg(CmpVal);
10455 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10458 MBB->addSuccessor(DoneMBB);
10459 MBB->addSuccessor(SetMBB);
10460
10461 // SetMBB:
10462 // %StoreVal = RLL %RetrySwapVal, -BitSize(%NegBitShift)
10463 // ^^ Rotate the new field to its proper position.
10464 // %RetryOldVal = CS %OldVal, %StoreVal, Disp(%Base)
10465 // JNE LoopMBB
10466 // # fall through to ExitMBB
10467 MBB = SetMBB;
10468 BuildMI(MBB, DL, TII->get(SystemZ::RLL), StoreVal)
10469 .addReg(RetrySwapVal).addReg(NegBitShift).addImm(-BitSize);
10470 BuildMI(MBB, DL, TII->get(CSOpcode), RetryOldVal)
10471 .addReg(OldVal)
10472 .addReg(StoreVal)
10473 .add(Base)
10474 .addImm(Disp);
10475 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10477 MBB->addSuccessor(LoopMBB);
10478 MBB->addSuccessor(DoneMBB);
10479
10480 // If the CC def wasn't dead in the ATOMIC_CMP_SWAPW, mark CC as live-in
10481 // to the block after the loop. At this point, CC may have been defined
10482 // either by the CR in LoopMBB or by the CS in SetMBB.
10483 if (!MI.registerDefIsDead(SystemZ::CC, /*TRI=*/nullptr))
10484 DoneMBB->addLiveIn(SystemZ::CC);
10485
10486 MI.eraseFromParent();
10487 return DoneMBB;
10488}
10489
10490// Emit a move from two GR64s to a GR128.
10492SystemZTargetLowering::emitPair128(MachineInstr &MI,
10493 MachineBasicBlock *MBB) const {
10494 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10495 const DebugLoc &DL = MI.getDebugLoc();
10496
10497 Register Dest = MI.getOperand(0).getReg();
10498 BuildMI(*MBB, MI, DL, TII->get(TargetOpcode::REG_SEQUENCE), Dest)
10499 .add(MI.getOperand(1))
10500 .addImm(SystemZ::subreg_h64)
10501 .add(MI.getOperand(2))
10502 .addImm(SystemZ::subreg_l64);
10503 MI.eraseFromParent();
10504 return MBB;
10505}
10506
10507// Emit an extension from a GR64 to a GR128. ClearEven is true
10508// if the high register of the GR128 value must be cleared or false if
10509// it's "don't care".
10510MachineBasicBlock *SystemZTargetLowering::emitExt128(MachineInstr &MI,
10512 bool ClearEven) const {
10513 MachineFunction &MF = *MBB->getParent();
10514 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10515 MachineRegisterInfo &MRI = MF.getRegInfo();
10516 DebugLoc DL = MI.getDebugLoc();
10517
10518 Register Dest = MI.getOperand(0).getReg();
10519 Register Src = MI.getOperand(1).getReg();
10520 Register In128 = MRI.createVirtualRegister(&SystemZ::GR128BitRegClass);
10521
10522 BuildMI(*MBB, MI, DL, TII->get(TargetOpcode::IMPLICIT_DEF), In128);
10523 if (ClearEven) {
10524 Register NewIn128 = MRI.createVirtualRegister(&SystemZ::GR128BitRegClass);
10525 Register Zero64 = MRI.createVirtualRegister(&SystemZ::GR64BitRegClass);
10526
10527 BuildMI(*MBB, MI, DL, TII->get(SystemZ::LLILL), Zero64)
10528 .addImm(0);
10529 BuildMI(*MBB, MI, DL, TII->get(TargetOpcode::INSERT_SUBREG), NewIn128)
10530 .addReg(In128).addReg(Zero64).addImm(SystemZ::subreg_h64);
10531 In128 = NewIn128;
10532 }
10533 BuildMI(*MBB, MI, DL, TII->get(TargetOpcode::INSERT_SUBREG), Dest)
10534 .addReg(In128).addReg(Src).addImm(SystemZ::subreg_l64);
10535
10536 MI.eraseFromParent();
10537 return MBB;
10538}
10539
10541SystemZTargetLowering::emitMemMemWrapper(MachineInstr &MI,
10543 unsigned Opcode, bool IsMemset) const {
10544 MachineFunction &MF = *MBB->getParent();
10545 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10546 MachineRegisterInfo &MRI = MF.getRegInfo();
10547 DebugLoc DL = MI.getDebugLoc();
10548
10549 MachineOperand DestBase = earlyUseOperand(MI.getOperand(0));
10550 uint64_t DestDisp = MI.getOperand(1).getImm();
10551 MachineOperand SrcBase = MachineOperand::CreateReg(0U, false);
10552 uint64_t SrcDisp;
10553
10554 // Fold the displacement Disp if it is out of range.
10555 auto foldDisplIfNeeded = [&](MachineOperand &Base, uint64_t &Disp) -> void {
10556 if (!isUInt<12>(Disp)) {
10557 Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
10558 unsigned Opcode = TII->getOpcodeForOffset(SystemZ::LA, Disp);
10559 BuildMI(*MI.getParent(), MI, MI.getDebugLoc(), TII->get(Opcode), Reg)
10560 .add(Base).addImm(Disp).addReg(0);
10562 Disp = 0;
10563 }
10564 };
10565
10566 if (!IsMemset) {
10567 SrcBase = earlyUseOperand(MI.getOperand(2));
10568 SrcDisp = MI.getOperand(3).getImm();
10569 } else {
10570 SrcBase = DestBase;
10571 SrcDisp = DestDisp++;
10572 foldDisplIfNeeded(DestBase, DestDisp);
10573 }
10574
10575 MachineOperand &LengthMO = MI.getOperand(IsMemset ? 2 : 4);
10576 bool IsImmForm = LengthMO.isImm();
10577 bool IsRegForm = !IsImmForm;
10578
10579 // Build and insert one Opcode of Length, with special treatment for memset.
10580 auto insertMemMemOp = [&](MachineBasicBlock *InsMBB,
10582 MachineOperand DBase, uint64_t DDisp,
10583 MachineOperand SBase, uint64_t SDisp,
10584 unsigned Length) -> void {
10585 assert(Length > 0 && Length <= 256 && "Building memory op with bad length.");
10586 if (IsMemset) {
10587 MachineOperand ByteMO = earlyUseOperand(MI.getOperand(3));
10588 if (ByteMO.isImm())
10589 BuildMI(*InsMBB, InsPos, DL, TII->get(SystemZ::MVI))
10590 .add(SBase).addImm(SDisp).add(ByteMO);
10591 else
10592 BuildMI(*InsMBB, InsPos, DL, TII->get(SystemZ::STC))
10593 .add(ByteMO).add(SBase).addImm(SDisp).addReg(0);
10594 if (--Length == 0)
10595 return;
10596 }
10597 BuildMI(*MBB, InsPos, DL, TII->get(Opcode))
10598 .add(DBase).addImm(DDisp).addImm(Length)
10599 .add(SBase).addImm(SDisp)
10600 .setMemRefs(MI.memoperands());
10601 };
10602
10603 bool NeedsLoop = false;
10604 uint64_t ImmLength = 0;
10605 Register LenAdjReg = SystemZ::NoRegister;
10606 if (IsImmForm) {
10607 ImmLength = LengthMO.getImm();
10608 ImmLength += IsMemset ? 2 : 1; // Add back the subtracted adjustment.
10609 if (ImmLength == 0) {
10610 MI.eraseFromParent();
10611 return MBB;
10612 }
10613 if (Opcode == SystemZ::CLC) {
10614 if (ImmLength > 3 * 256)
10615 // A two-CLC sequence is a clear win over a loop, not least because
10616 // it needs only one branch. A three-CLC sequence needs the same
10617 // number of branches as a loop (i.e. 2), but is shorter. That
10618 // brings us to lengths greater than 768 bytes. It seems relatively
10619 // likely that a difference will be found within the first 768 bytes,
10620 // so we just optimize for the smallest number of branch
10621 // instructions, in order to avoid polluting the prediction buffer
10622 // too much.
10623 NeedsLoop = true;
10624 } else if (ImmLength > 6 * 256)
10625 // The heuristic we use is to prefer loops for anything that would
10626 // require 7 or more MVCs. With these kinds of sizes there isn't much
10627 // to choose between straight-line code and looping code, since the
10628 // time will be dominated by the MVCs themselves.
10629 NeedsLoop = true;
10630 } else {
10631 NeedsLoop = true;
10632 LenAdjReg = LengthMO.getReg();
10633 }
10634
10635 // When generating more than one CLC, all but the last will need to
10636 // branch to the end when a difference is found.
10637 MachineBasicBlock *EndMBB =
10638 (Opcode == SystemZ::CLC && (ImmLength > 256 || NeedsLoop)
10640 : nullptr);
10641
10642 if (NeedsLoop) {
10643 Register StartCountReg =
10644 MRI.createVirtualRegister(&SystemZ::GR64BitRegClass);
10645 if (IsImmForm) {
10646 TII->loadImmediate(*MBB, MI, StartCountReg, ImmLength / 256);
10647 ImmLength &= 255;
10648 } else {
10649 BuildMI(*MBB, MI, DL, TII->get(SystemZ::SRLG), StartCountReg)
10650 .addReg(LenAdjReg)
10651 .addReg(0)
10652 .addImm(8);
10653 }
10654
10655 bool HaveSingleBase = DestBase.isIdenticalTo(SrcBase);
10656 auto loadZeroAddress = [&]() -> MachineOperand {
10657 Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
10658 BuildMI(*MBB, MI, DL, TII->get(SystemZ::LGHI), Reg).addImm(0);
10659 return MachineOperand::CreateReg(Reg, false);
10660 };
10661 if (DestBase.isReg() && DestBase.getReg() == SystemZ::NoRegister)
10662 DestBase = loadZeroAddress();
10663 if (SrcBase.isReg() && SrcBase.getReg() == SystemZ::NoRegister)
10664 SrcBase = HaveSingleBase ? DestBase : loadZeroAddress();
10665
10666 MachineBasicBlock *StartMBB = nullptr;
10667 MachineBasicBlock *LoopMBB = nullptr;
10668 MachineBasicBlock *NextMBB = nullptr;
10669 MachineBasicBlock *DoneMBB = nullptr;
10670 MachineBasicBlock *AllDoneMBB = nullptr;
10671
10672 Register StartSrcReg = forceReg(MI, SrcBase, TII);
10673 Register StartDestReg =
10674 (HaveSingleBase ? StartSrcReg : forceReg(MI, DestBase, TII));
10675
10676 const TargetRegisterClass *RC = &SystemZ::ADDR64BitRegClass;
10677 Register ThisSrcReg = MRI.createVirtualRegister(RC);
10678 Register ThisDestReg =
10679 (HaveSingleBase ? ThisSrcReg : MRI.createVirtualRegister(RC));
10680 Register NextSrcReg = MRI.createVirtualRegister(RC);
10681 Register NextDestReg =
10682 (HaveSingleBase ? NextSrcReg : MRI.createVirtualRegister(RC));
10683 RC = &SystemZ::GR64BitRegClass;
10684 Register ThisCountReg = MRI.createVirtualRegister(RC);
10685 Register NextCountReg = MRI.createVirtualRegister(RC);
10686
10687 if (IsRegForm) {
10688 AllDoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10689 StartMBB = SystemZ::emitBlockAfter(MBB);
10690 LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10691 NextMBB = (EndMBB ? SystemZ::emitBlockAfter(LoopMBB) : LoopMBB);
10692 DoneMBB = SystemZ::emitBlockAfter(NextMBB);
10693
10694 // MBB:
10695 // # Jump to AllDoneMBB if LenAdjReg means 0, or fall thru to StartMBB.
10696 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
10697 .addReg(LenAdjReg).addImm(IsMemset ? -2 : -1);
10698 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10700 .addMBB(AllDoneMBB);
10701 MBB->addSuccessor(AllDoneMBB);
10702 if (!IsMemset)
10703 MBB->addSuccessor(StartMBB);
10704 else {
10705 // MemsetOneCheckMBB:
10706 // # Jump to MemsetOneMBB for a memset of length 1, or
10707 // # fall thru to StartMBB.
10708 MachineBasicBlock *MemsetOneCheckMBB = SystemZ::emitBlockAfter(MBB);
10709 MachineBasicBlock *MemsetOneMBB = SystemZ::emitBlockAfter(&*MF.rbegin());
10710 MBB->addSuccessor(MemsetOneCheckMBB);
10711 MBB = MemsetOneCheckMBB;
10712 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
10713 .addReg(LenAdjReg).addImm(-1);
10714 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10716 .addMBB(MemsetOneMBB);
10717 MBB->addSuccessor(MemsetOneMBB, {10, 100});
10718 MBB->addSuccessor(StartMBB, {90, 100});
10719
10720 // MemsetOneMBB:
10721 // # Jump back to AllDoneMBB after a single MVI or STC.
10722 MBB = MemsetOneMBB;
10723 insertMemMemOp(MBB, MBB->end(),
10724 MachineOperand::CreateReg(StartDestReg, false), DestDisp,
10725 MachineOperand::CreateReg(StartSrcReg, false), SrcDisp,
10726 1);
10727 BuildMI(MBB, DL, TII->get(SystemZ::J)).addMBB(AllDoneMBB);
10728 MBB->addSuccessor(AllDoneMBB);
10729 }
10730
10731 // StartMBB:
10732 // # Jump to DoneMBB if %StartCountReg is zero, or fall through to LoopMBB.
10733 MBB = StartMBB;
10734 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
10735 .addReg(StartCountReg).addImm(0);
10736 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10738 .addMBB(DoneMBB);
10739 MBB->addSuccessor(DoneMBB);
10740 MBB->addSuccessor(LoopMBB);
10741 }
10742 else {
10743 StartMBB = MBB;
10744 DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10745 LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10746 NextMBB = (EndMBB ? SystemZ::emitBlockAfter(LoopMBB) : LoopMBB);
10747
10748 // StartMBB:
10749 // # fall through to LoopMBB
10750 MBB->addSuccessor(LoopMBB);
10751
10752 DestBase = MachineOperand::CreateReg(NextDestReg, false);
10753 SrcBase = MachineOperand::CreateReg(NextSrcReg, false);
10754 if (EndMBB && !ImmLength)
10755 // If the loop handled the whole CLC range, DoneMBB will be empty with
10756 // CC live-through into EndMBB, so add it as live-in.
10757 DoneMBB->addLiveIn(SystemZ::CC);
10758 }
10759
10760 // LoopMBB:
10761 // %ThisDestReg = phi [ %StartDestReg, StartMBB ],
10762 // [ %NextDestReg, NextMBB ]
10763 // %ThisSrcReg = phi [ %StartSrcReg, StartMBB ],
10764 // [ %NextSrcReg, NextMBB ]
10765 // %ThisCountReg = phi [ %StartCountReg, StartMBB ],
10766 // [ %NextCountReg, NextMBB ]
10767 // ( PFD 2, 768+DestDisp(%ThisDestReg) )
10768 // Opcode DestDisp(256,%ThisDestReg), SrcDisp(%ThisSrcReg)
10769 // ( JLH EndMBB )
10770 //
10771 // The prefetch is used only for MVC. The JLH is used only for CLC.
10772 MBB = LoopMBB;
10773 BuildMI(MBB, DL, TII->get(SystemZ::PHI), ThisDestReg)
10774 .addReg(StartDestReg).addMBB(StartMBB)
10775 .addReg(NextDestReg).addMBB(NextMBB);
10776 if (!HaveSingleBase)
10777 BuildMI(MBB, DL, TII->get(SystemZ::PHI), ThisSrcReg)
10778 .addReg(StartSrcReg).addMBB(StartMBB)
10779 .addReg(NextSrcReg).addMBB(NextMBB);
10780 BuildMI(MBB, DL, TII->get(SystemZ::PHI), ThisCountReg)
10781 .addReg(StartCountReg).addMBB(StartMBB)
10782 .addReg(NextCountReg).addMBB(NextMBB);
10783 if (Opcode == SystemZ::MVC)
10784 BuildMI(MBB, DL, TII->get(SystemZ::PFD))
10786 .addReg(ThisDestReg).addImm(DestDisp - IsMemset + 768).addReg(0);
10787 insertMemMemOp(MBB, MBB->end(),
10788 MachineOperand::CreateReg(ThisDestReg, false), DestDisp,
10789 MachineOperand::CreateReg(ThisSrcReg, false), SrcDisp, 256);
10790 if (EndMBB) {
10791 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10793 .addMBB(EndMBB);
10794 MBB->addSuccessor(EndMBB);
10795 MBB->addSuccessor(NextMBB);
10796 }
10797
10798 // NextMBB:
10799 // %NextDestReg = LA 256(%ThisDestReg)
10800 // %NextSrcReg = LA 256(%ThisSrcReg)
10801 // %NextCountReg = AGHI %ThisCountReg, -1
10802 // CGHI %NextCountReg, 0
10803 // JLH LoopMBB
10804 // # fall through to DoneMBB
10805 //
10806 // The AGHI, CGHI and JLH should be converted to BRCTG by later passes.
10807 MBB = NextMBB;
10808 BuildMI(MBB, DL, TII->get(SystemZ::LA), NextDestReg)
10809 .addReg(ThisDestReg).addImm(256).addReg(0);
10810 if (!HaveSingleBase)
10811 BuildMI(MBB, DL, TII->get(SystemZ::LA), NextSrcReg)
10812 .addReg(ThisSrcReg).addImm(256).addReg(0);
10813 BuildMI(MBB, DL, TII->get(SystemZ::AGHI), NextCountReg)
10814 .addReg(ThisCountReg).addImm(-1);
10815 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
10816 .addReg(NextCountReg).addImm(0);
10817 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10819 .addMBB(LoopMBB);
10820 MBB->addSuccessor(LoopMBB);
10821 MBB->addSuccessor(DoneMBB);
10822
10823 MBB = DoneMBB;
10824 if (IsRegForm) {
10825 // DoneMBB:
10826 // # Make PHIs for RemDestReg/RemSrcReg as the loop may or may not run.
10827 // # Use EXecute Relative Long for the remainder of the bytes. The target
10828 // instruction of the EXRL will have a length field of 1 since 0 is an
10829 // illegal value. The number of bytes processed becomes (%LenAdjReg &
10830 // 0xff) + 1.
10831 // # Fall through to AllDoneMBB.
10832 Register RemSrcReg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
10833 Register RemDestReg = HaveSingleBase ? RemSrcReg
10834 : MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
10835 BuildMI(MBB, DL, TII->get(SystemZ::PHI), RemDestReg)
10836 .addReg(StartDestReg).addMBB(StartMBB)
10837 .addReg(NextDestReg).addMBB(NextMBB);
10838 if (!HaveSingleBase)
10839 BuildMI(MBB, DL, TII->get(SystemZ::PHI), RemSrcReg)
10840 .addReg(StartSrcReg).addMBB(StartMBB)
10841 .addReg(NextSrcReg).addMBB(NextMBB);
10842 if (IsMemset)
10843 insertMemMemOp(MBB, MBB->end(),
10844 MachineOperand::CreateReg(RemDestReg, false), DestDisp,
10845 MachineOperand::CreateReg(RemSrcReg, false), SrcDisp, 1);
10846 MachineInstrBuilder EXRL_MIB =
10847 BuildMI(MBB, DL, TII->get(SystemZ::EXRL_Pseudo))
10848 .addImm(Opcode)
10849 .addReg(LenAdjReg)
10850 .addReg(RemDestReg).addImm(DestDisp)
10851 .addReg(RemSrcReg).addImm(SrcDisp);
10852 MBB->addSuccessor(AllDoneMBB);
10853 MBB = AllDoneMBB;
10854 if (Opcode != SystemZ::MVC) {
10855 EXRL_MIB.addReg(SystemZ::CC, RegState::ImplicitDefine);
10856 if (EndMBB)
10857 MBB->addLiveIn(SystemZ::CC);
10858 }
10859 }
10860 MF.getProperties().resetNoPHIs();
10861 }
10862
10863 // Handle any remaining bytes with straight-line code.
10864 while (ImmLength > 0) {
10865 uint64_t ThisLength = std::min(ImmLength, uint64_t(256));
10866 // The previous iteration might have created out-of-range displacements.
10867 // Apply them using LA/LAY if so.
10868 foldDisplIfNeeded(DestBase, DestDisp);
10869 foldDisplIfNeeded(SrcBase, SrcDisp);
10870 insertMemMemOp(MBB, MI, DestBase, DestDisp, SrcBase, SrcDisp, ThisLength);
10871 DestDisp += ThisLength;
10872 SrcDisp += ThisLength;
10873 ImmLength -= ThisLength;
10874 // If there's another CLC to go, branch to the end if a difference
10875 // was found.
10876 if (EndMBB && ImmLength > 0) {
10877 MachineBasicBlock *NextMBB = SystemZ::splitBlockBefore(MI, MBB);
10878 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10880 .addMBB(EndMBB);
10881 MBB->addSuccessor(EndMBB);
10882 MBB->addSuccessor(NextMBB);
10883 MBB = NextMBB;
10884 }
10885 }
10886 if (EndMBB) {
10887 MBB->addSuccessor(EndMBB);
10888 MBB = EndMBB;
10889 MBB->addLiveIn(SystemZ::CC);
10890 }
10891
10892 MI.eraseFromParent();
10893 return MBB;
10894}
10895
10897SystemZTargetLowering::emitMemmoveImm(MachineInstr &MI,
10898 MachineBasicBlock *MBB) const {
10899 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10900
10901 DebugLoc DL = MI.getDebugLoc();
10902 MachineOperand DstAddr = earlyUseOperand(MI.getOperand(0));
10903 MachineOperand SrcAddr = earlyUseOperand(MI.getOperand(1));
10904 uint64_t Len = MI.getOperand(2).getImm();
10905 assert(Len > 0 && Len <= 256 && "Memmove of of unsupported constant length.");
10906
10907 // Use MVC or MVCRL after comparing the addresses.
10908 MachineBasicBlock *DoneMBB = SystemZ::splitBlockAfter(MI, MBB);
10909 MachineBasicBlock *MvcMBB = SystemZ::emitBlockAfter(MBB);
10910 MachineBasicBlock *MvcrlMBB = SystemZ::emitBlockAfter(MvcMBB);
10911 MBB->addSuccessor(MvcMBB);
10912 MBB->addSuccessor(MvcrlMBB);
10913 MvcMBB->addSuccessor(DoneMBB);
10914 MvcrlMBB->addSuccessor(DoneMBB);
10915
10916 BuildMI(MBB, DL, TII->get(SystemZ::CLGR)).add(SrcAddr).add(DstAddr);
10917 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10919 .addMBB(MvcrlMBB);
10920
10921 BuildMI(MvcMBB, DL, TII->get(SystemZ::MVC))
10922 .add(DstAddr).addImm(0)
10923 .addImm(Len)
10924 .add(SrcAddr).addImm(0)
10925 .setMemRefs(MI.memoperands());
10926 BuildMI(MvcMBB, DL, TII->get(SystemZ::J)).addMBB(DoneMBB);
10927
10928 BuildMI(MvcrlMBB, DL, TII->get(SystemZ::LHI), SystemZ::R0L).addImm(Len - 1);
10929 BuildMI(MvcrlMBB, DL, TII->get(SystemZ::MVCRL))
10930 .add(DstAddr).addImm(0)
10931 .add(SrcAddr).addImm(0)
10932 .setMemRefs(MI.memoperands());
10933
10934 MI.eraseFromParent();
10935 return DoneMBB;
10936}
10937
10938// Decompose string pseudo-instruction MI into a loop that continually performs
10939// Opcode until CC != 3.
10940MachineBasicBlock *SystemZTargetLowering::emitStringWrapper(
10941 MachineInstr &MI, MachineBasicBlock *MBB, unsigned Opcode) const {
10942 MachineFunction &MF = *MBB->getParent();
10943 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10944 MachineRegisterInfo &MRI = MF.getRegInfo();
10945 DebugLoc DL = MI.getDebugLoc();
10946
10947 uint64_t End1Reg = MI.getOperand(0).getReg();
10948 uint64_t Start1Reg = MI.getOperand(1).getReg();
10949 uint64_t Start2Reg = MI.getOperand(2).getReg();
10950 uint64_t CharReg = MI.getOperand(3).getReg();
10951
10952 const TargetRegisterClass *RC = &SystemZ::GR64BitRegClass;
10953 uint64_t This1Reg = MRI.createVirtualRegister(RC);
10954 uint64_t This2Reg = MRI.createVirtualRegister(RC);
10955 uint64_t End2Reg = MRI.createVirtualRegister(RC);
10956
10957 MachineBasicBlock *StartMBB = MBB;
10958 MachineBasicBlock *DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10959 MachineBasicBlock *LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10960
10961 // StartMBB:
10962 // # fall through to LoopMBB
10963 MBB->addSuccessor(LoopMBB);
10964
10965 // LoopMBB:
10966 // %This1Reg = phi [ %Start1Reg, StartMBB ], [ %End1Reg, LoopMBB ]
10967 // %This2Reg = phi [ %Start2Reg, StartMBB ], [ %End2Reg, LoopMBB ]
10968 // R0L = %CharReg
10969 // %End1Reg, %End2Reg = CLST %This1Reg, %This2Reg -- uses R0L
10970 // JO LoopMBB
10971 // # fall through to DoneMBB
10972 //
10973 // The load of R0L can be hoisted by post-RA LICM.
10974 MBB = LoopMBB;
10975
10976 BuildMI(MBB, DL, TII->get(SystemZ::PHI), This1Reg)
10977 .addReg(Start1Reg).addMBB(StartMBB)
10978 .addReg(End1Reg).addMBB(LoopMBB);
10979 BuildMI(MBB, DL, TII->get(SystemZ::PHI), This2Reg)
10980 .addReg(Start2Reg).addMBB(StartMBB)
10981 .addReg(End2Reg).addMBB(LoopMBB);
10982 BuildMI(MBB, DL, TII->get(TargetOpcode::COPY), SystemZ::R0L).addReg(CharReg);
10983 BuildMI(MBB, DL, TII->get(Opcode))
10984 .addReg(End1Reg, RegState::Define).addReg(End2Reg, RegState::Define)
10985 .addReg(This1Reg).addReg(This2Reg);
10986 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10988 MBB->addSuccessor(LoopMBB);
10989 MBB->addSuccessor(DoneMBB);
10990
10991 DoneMBB->addLiveIn(SystemZ::CC);
10992
10993 MI.eraseFromParent();
10994 return DoneMBB;
10995}
10996
10997// Update TBEGIN instruction with final opcode and register clobbers.
10998MachineBasicBlock *SystemZTargetLowering::emitTransactionBegin(
10999 MachineInstr &MI, MachineBasicBlock *MBB, unsigned Opcode,
11000 bool NoFloat) const {
11001 MachineFunction &MF = *MBB->getParent();
11002 const TargetFrameLowering *TFI = Subtarget.getFrameLowering();
11003 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
11004
11005 // Update opcode.
11006 MI.setDesc(TII->get(Opcode));
11007
11008 // We cannot handle a TBEGIN that clobbers the stack or frame pointer.
11009 // Make sure to add the corresponding GRSM bits if they are missing.
11010 uint64_t Control = MI.getOperand(2).getImm();
11011 static const unsigned GPRControlBit[16] = {
11012 0x8000, 0x8000, 0x4000, 0x4000, 0x2000, 0x2000, 0x1000, 0x1000,
11013 0x0800, 0x0800, 0x0400, 0x0400, 0x0200, 0x0200, 0x0100, 0x0100
11014 };
11015 Control |= GPRControlBit[15];
11016 if (TFI->hasFP(MF))
11017 Control |= GPRControlBit[11];
11018 MI.getOperand(2).setImm(Control);
11019
11020 // Add GPR clobbers.
11021 for (int I = 0; I < 16; I++) {
11022 if ((Control & GPRControlBit[I]) == 0) {
11023 unsigned Reg = SystemZMC::GR64Regs[I];
11024 MI.addOperand(MachineOperand::CreateReg(Reg, true, true));
11025 }
11026 }
11027
11028 // Add FPR/VR clobbers.
11029 if (!NoFloat && (Control & 4) != 0) {
11030 if (Subtarget.hasVector()) {
11031 for (unsigned Reg : SystemZMC::VR128Regs) {
11032 MI.addOperand(MachineOperand::CreateReg(Reg, true, true));
11033 }
11034 } else {
11035 for (unsigned Reg : SystemZMC::FP64Regs) {
11036 MI.addOperand(MachineOperand::CreateReg(Reg, true, true));
11037 }
11038 }
11039 }
11040
11041 return MBB;
11042}
11043
11044MachineBasicBlock *SystemZTargetLowering::emitLoadAndTestCmp0(
11045 MachineInstr &MI, MachineBasicBlock *MBB, unsigned Opcode) const {
11046 MachineFunction &MF = *MBB->getParent();
11047 MachineRegisterInfo *MRI = &MF.getRegInfo();
11048 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
11049 DebugLoc DL = MI.getDebugLoc();
11050
11051 Register SrcReg = MI.getOperand(0).getReg();
11052
11053 // Create new virtual register of the same class as source.
11054 const TargetRegisterClass *RC = MRI->getRegClass(SrcReg);
11055 Register DstReg = MRI->createVirtualRegister(RC);
11056
11057 // Replace pseudo with a normal load-and-test that models the def as
11058 // well.
11059 BuildMI(*MBB, MI, DL, TII->get(Opcode), DstReg)
11060 .addReg(SrcReg)
11061 .setMIFlags(MI.getFlags());
11062 MI.eraseFromParent();
11063
11064 return MBB;
11065}
11066
11067MachineBasicBlock *SystemZTargetLowering::emitProbedAlloca(
11069 MachineFunction &MF = *MBB->getParent();
11070 MachineRegisterInfo *MRI = &MF.getRegInfo();
11071 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
11072 DebugLoc DL = MI.getDebugLoc();
11073 const unsigned ProbeSize = getStackProbeSize(MF);
11074 Register DstReg = MI.getOperand(0).getReg();
11075 Register SizeReg = MI.getOperand(2).getReg();
11076
11077 MachineBasicBlock *StartMBB = MBB;
11078 MachineBasicBlock *DoneMBB = SystemZ::splitBlockAfter(MI, MBB);
11079 MachineBasicBlock *LoopTestMBB = SystemZ::emitBlockAfter(StartMBB);
11080 MachineBasicBlock *LoopBodyMBB = SystemZ::emitBlockAfter(LoopTestMBB);
11081 MachineBasicBlock *TailTestMBB = SystemZ::emitBlockAfter(LoopBodyMBB);
11082 MachineBasicBlock *TailMBB = SystemZ::emitBlockAfter(TailTestMBB);
11083
11084 MachineMemOperand *VolLdMMO = MF.getMachineMemOperand(MachinePointerInfo(),
11086
11087 Register PHIReg = MRI->createVirtualRegister(&SystemZ::ADDR64BitRegClass);
11088 Register IncReg = MRI->createVirtualRegister(&SystemZ::ADDR64BitRegClass);
11089
11090 // LoopTestMBB
11091 // BRC TailTestMBB
11092 // # fallthrough to LoopBodyMBB
11093 StartMBB->addSuccessor(LoopTestMBB);
11094 MBB = LoopTestMBB;
11095 BuildMI(MBB, DL, TII->get(SystemZ::PHI), PHIReg)
11096 .addReg(SizeReg)
11097 .addMBB(StartMBB)
11098 .addReg(IncReg)
11099 .addMBB(LoopBodyMBB);
11100 BuildMI(MBB, DL, TII->get(SystemZ::CLGFI))
11101 .addReg(PHIReg)
11102 .addImm(ProbeSize);
11103 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
11105 .addMBB(TailTestMBB);
11106 MBB->addSuccessor(LoopBodyMBB);
11107 MBB->addSuccessor(TailTestMBB);
11108
11109 // LoopBodyMBB: Allocate and probe by means of a volatile compare.
11110 // J LoopTestMBB
11111 MBB = LoopBodyMBB;
11112 BuildMI(MBB, DL, TII->get(SystemZ::SLGFI), IncReg)
11113 .addReg(PHIReg)
11114 .addImm(ProbeSize);
11115 BuildMI(MBB, DL, TII->get(SystemZ::SLGFI), SystemZ::R15D)
11116 .addReg(SystemZ::R15D)
11117 .addImm(ProbeSize);
11118 BuildMI(MBB, DL, TII->get(SystemZ::CG)).addReg(SystemZ::R15D)
11119 .addReg(SystemZ::R15D).addImm(ProbeSize - 8).addReg(0)
11120 .setMemRefs(VolLdMMO);
11121 BuildMI(MBB, DL, TII->get(SystemZ::J)).addMBB(LoopTestMBB);
11122 MBB->addSuccessor(LoopTestMBB);
11123
11124 // TailTestMBB
11125 // BRC DoneMBB
11126 // # fallthrough to TailMBB
11127 MBB = TailTestMBB;
11128 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
11129 .addReg(PHIReg)
11130 .addImm(0);
11131 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
11133 .addMBB(DoneMBB);
11134 MBB->addSuccessor(TailMBB);
11135 MBB->addSuccessor(DoneMBB);
11136
11137 // TailMBB
11138 // # fallthrough to DoneMBB
11139 MBB = TailMBB;
11140 BuildMI(MBB, DL, TII->get(SystemZ::SLGR), SystemZ::R15D)
11141 .addReg(SystemZ::R15D)
11142 .addReg(PHIReg);
11143 BuildMI(MBB, DL, TII->get(SystemZ::CG)).addReg(SystemZ::R15D)
11144 .addReg(SystemZ::R15D).addImm(-8).addReg(PHIReg)
11145 .setMemRefs(VolLdMMO);
11146 MBB->addSuccessor(DoneMBB);
11147
11148 // DoneMBB
11149 MBB = DoneMBB;
11150 BuildMI(*MBB, MBB->begin(), DL, TII->get(TargetOpcode::COPY), DstReg)
11151 .addReg(SystemZ::R15D);
11152
11153 MI.eraseFromParent();
11154 return DoneMBB;
11155}
11156
11157SDValue SystemZTargetLowering::
11158getBackchainAddress(SDValue SP, SelectionDAG &DAG) const {
11160 auto *TFL = Subtarget.getFrameLowering<SystemZELFFrameLowering>();
11161 SDLoc DL(SP);
11162 return DAG.getNode(ISD::ADD, DL, MVT::i64, SP,
11163 DAG.getIntPtrConstant(TFL->getBackchainOffset(MF), DL));
11164}
11165
11166// Replace a _STACKGUARD_DAG pseudo with a _STACKGUARD pseudo, adding
11167// a dead early-clobber def reg that will be used as a scratch register
11168// when the pseudo is expanded.
11169MachineBasicBlock *SystemZTargetLowering::emitStackGuardPseudo(
11170 MachineInstr &MI, MachineBasicBlock *MBB, unsigned PseudoOp) const {
11171 MachineRegisterInfo *MRI = &MBB->getParent()->getRegInfo();
11172 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
11173 DebugLoc DL = MI.getDebugLoc();
11174 Register AddrReg = MRI->createVirtualRegister(&SystemZ::ADDR64BitRegClass);
11175 BuildMI(*MBB, MI, DL, TII->get(PseudoOp), AddrReg)
11176 .addFrameIndex(MI.getOperand(0).getIndex())
11177 .addImm(MI.getOperand(1).getImm());
11178 MI.eraseFromParent();
11179 return MBB;
11180}
11181
11184 switch (MI.getOpcode()) {
11185 case SystemZ::ADJCALLSTACKDOWN:
11186 case SystemZ::ADJCALLSTACKUP:
11187 return emitAdjCallStack(MI, MBB);
11188
11189 case SystemZ::Select32:
11190 case SystemZ::Select64:
11191 case SystemZ::Select128:
11192 case SystemZ::SelectF32:
11193 case SystemZ::SelectF64:
11194 case SystemZ::SelectF128:
11195 case SystemZ::SelectVR32:
11196 case SystemZ::SelectVR64:
11197 case SystemZ::SelectVR128:
11198 return emitSelect(MI, MBB);
11199
11200 case SystemZ::CondStore8Mux:
11201 return emitCondStore(MI, MBB, SystemZ::STCMux, 0, false);
11202 case SystemZ::CondStore8MuxInv:
11203 return emitCondStore(MI, MBB, SystemZ::STCMux, 0, true);
11204 case SystemZ::CondStore16Mux:
11205 return emitCondStore(MI, MBB, SystemZ::STHMux, 0, false);
11206 case SystemZ::CondStore16MuxInv:
11207 return emitCondStore(MI, MBB, SystemZ::STHMux, 0, true);
11208 case SystemZ::CondStore32Mux:
11209 return emitCondStore(MI, MBB, SystemZ::STMux, SystemZ::STOCMux, false);
11210 case SystemZ::CondStore32MuxInv:
11211 return emitCondStore(MI, MBB, SystemZ::STMux, SystemZ::STOCMux, true);
11212 case SystemZ::CondStore8:
11213 return emitCondStore(MI, MBB, SystemZ::STC, 0, false);
11214 case SystemZ::CondStore8Inv:
11215 return emitCondStore(MI, MBB, SystemZ::STC, 0, true);
11216 case SystemZ::CondStore16:
11217 return emitCondStore(MI, MBB, SystemZ::STH, 0, false);
11218 case SystemZ::CondStore16Inv:
11219 return emitCondStore(MI, MBB, SystemZ::STH, 0, true);
11220 case SystemZ::CondStore32:
11221 return emitCondStore(MI, MBB, SystemZ::ST, SystemZ::STOC, false);
11222 case SystemZ::CondStore32Inv:
11223 return emitCondStore(MI, MBB, SystemZ::ST, SystemZ::STOC, true);
11224 case SystemZ::CondStore64:
11225 return emitCondStore(MI, MBB, SystemZ::STG, SystemZ::STOCG, false);
11226 case SystemZ::CondStore64Inv:
11227 return emitCondStore(MI, MBB, SystemZ::STG, SystemZ::STOCG, true);
11228 case SystemZ::CondStoreF32:
11229 return emitCondStore(MI, MBB, SystemZ::STE, 0, false);
11230 case SystemZ::CondStoreF32Inv:
11231 return emitCondStore(MI, MBB, SystemZ::STE, 0, true);
11232 case SystemZ::CondStoreF64:
11233 return emitCondStore(MI, MBB, SystemZ::STD, 0, false);
11234 case SystemZ::CondStoreF64Inv:
11235 return emitCondStore(MI, MBB, SystemZ::STD, 0, true);
11236
11237 case SystemZ::SCmp128Hi:
11238 return emitICmp128Hi(MI, MBB, false);
11239 case SystemZ::UCmp128Hi:
11240 return emitICmp128Hi(MI, MBB, true);
11241
11242 case SystemZ::PAIR128:
11243 return emitPair128(MI, MBB);
11244 case SystemZ::AEXT128:
11245 return emitExt128(MI, MBB, false);
11246 case SystemZ::ZEXT128:
11247 return emitExt128(MI, MBB, true);
11248
11249 case SystemZ::ATOMIC_SWAPW:
11250 return emitAtomicLoadBinary(MI, MBB, 0);
11251
11252 case SystemZ::ATOMIC_LOADW_AR:
11253 return emitAtomicLoadBinary(MI, MBB, SystemZ::AR);
11254 case SystemZ::ATOMIC_LOADW_AFI:
11255 return emitAtomicLoadBinary(MI, MBB, SystemZ::AFI);
11256
11257 case SystemZ::ATOMIC_LOADW_SR:
11258 return emitAtomicLoadBinary(MI, MBB, SystemZ::SR);
11259
11260 case SystemZ::ATOMIC_LOADW_NR:
11261 return emitAtomicLoadBinary(MI, MBB, SystemZ::NR);
11262 case SystemZ::ATOMIC_LOADW_NILH:
11263 return emitAtomicLoadBinary(MI, MBB, SystemZ::NILH);
11264
11265 case SystemZ::ATOMIC_LOADW_OR:
11266 return emitAtomicLoadBinary(MI, MBB, SystemZ::OR);
11267 case SystemZ::ATOMIC_LOADW_OILH:
11268 return emitAtomicLoadBinary(MI, MBB, SystemZ::OILH);
11269
11270 case SystemZ::ATOMIC_LOADW_XR:
11271 return emitAtomicLoadBinary(MI, MBB, SystemZ::XR);
11272 case SystemZ::ATOMIC_LOADW_XILF:
11273 return emitAtomicLoadBinary(MI, MBB, SystemZ::XILF);
11274
11275 case SystemZ::ATOMIC_LOADW_NRi:
11276 return emitAtomicLoadBinary(MI, MBB, SystemZ::NR, true);
11277 case SystemZ::ATOMIC_LOADW_NILHi:
11278 return emitAtomicLoadBinary(MI, MBB, SystemZ::NILH, true);
11279
11280 case SystemZ::ATOMIC_LOADW_MIN:
11281 return emitAtomicLoadMinMax(MI, MBB, SystemZ::CR, SystemZ::CCMASK_CMP_LE);
11282 case SystemZ::ATOMIC_LOADW_MAX:
11283 return emitAtomicLoadMinMax(MI, MBB, SystemZ::CR, SystemZ::CCMASK_CMP_GE);
11284 case SystemZ::ATOMIC_LOADW_UMIN:
11285 return emitAtomicLoadMinMax(MI, MBB, SystemZ::CLR, SystemZ::CCMASK_CMP_LE);
11286 case SystemZ::ATOMIC_LOADW_UMAX:
11287 return emitAtomicLoadMinMax(MI, MBB, SystemZ::CLR, SystemZ::CCMASK_CMP_GE);
11288
11289 case SystemZ::ATOMIC_CMP_SWAPW:
11290 return emitAtomicCmpSwapW(MI, MBB);
11291 case SystemZ::MVCImm:
11292 case SystemZ::MVCReg:
11293 return emitMemMemWrapper(MI, MBB, SystemZ::MVC);
11294 case SystemZ::NCImm:
11295 return emitMemMemWrapper(MI, MBB, SystemZ::NC);
11296 case SystemZ::OCImm:
11297 return emitMemMemWrapper(MI, MBB, SystemZ::OC);
11298 case SystemZ::XCImm:
11299 case SystemZ::XCReg:
11300 return emitMemMemWrapper(MI, MBB, SystemZ::XC);
11301 case SystemZ::CLCImm:
11302 case SystemZ::CLCReg:
11303 return emitMemMemWrapper(MI, MBB, SystemZ::CLC);
11304 case SystemZ::MemsetImmImm:
11305 case SystemZ::MemsetImmReg:
11306 case SystemZ::MemsetRegImm:
11307 case SystemZ::MemsetRegReg:
11308 return emitMemMemWrapper(MI, MBB, SystemZ::MVC, true/*IsMemset*/);
11309 case SystemZ::MemmoveImm:
11310 return emitMemmoveImm(MI, MBB);
11311 case SystemZ::CLSTLoop:
11312 return emitStringWrapper(MI, MBB, SystemZ::CLST);
11313 case SystemZ::MVSTLoop:
11314 return emitStringWrapper(MI, MBB, SystemZ::MVST);
11315 case SystemZ::SRSTLoop:
11316 return emitStringWrapper(MI, MBB, SystemZ::SRST);
11317 case SystemZ::TBEGIN:
11318 return emitTransactionBegin(MI, MBB, SystemZ::TBEGIN, false);
11319 case SystemZ::TBEGIN_nofloat:
11320 return emitTransactionBegin(MI, MBB, SystemZ::TBEGIN, true);
11321 case SystemZ::TBEGINC:
11322 return emitTransactionBegin(MI, MBB, SystemZ::TBEGINC, true);
11323 case SystemZ::LTEBRCompare_Pseudo:
11324 return emitLoadAndTestCmp0(MI, MBB, SystemZ::LTEBR);
11325 case SystemZ::LTDBRCompare_Pseudo:
11326 return emitLoadAndTestCmp0(MI, MBB, SystemZ::LTDBR);
11327 case SystemZ::LTXBRCompare_Pseudo:
11328 return emitLoadAndTestCmp0(MI, MBB, SystemZ::LTXBR);
11329
11330 case SystemZ::PROBED_ALLOCA:
11331 return emitProbedAlloca(MI, MBB);
11332 case SystemZ::EH_SjLj_SetJmp:
11333 return emitEHSjLjSetJmp(MI, MBB);
11334 case SystemZ::EH_SjLj_LongJmp:
11335 return emitEHSjLjLongJmp(MI, MBB);
11336
11337 case TargetOpcode::STACKMAP:
11338 case TargetOpcode::PATCHPOINT:
11339 return emitPatchPoint(MI, MBB);
11340
11341 case SystemZ::MOV_STACKGUARD_DAG:
11342 return emitStackGuardPseudo(MI, MBB, SystemZ::MOV_STACKGUARD);
11343
11344 case SystemZ::CMP_STACKGUARD_DAG:
11345 return emitStackGuardPseudo(MI, MBB, SystemZ::CMP_STACKGUARD);
11346
11347 default:
11348 llvm_unreachable("Unexpected instr type to insert");
11349 }
11350}
11351
11352// This is only used by the isel schedulers, and is needed only to prevent
11353// compiler from crashing when list-ilp is used.
11354const TargetRegisterClass *
11355SystemZTargetLowering::getRepRegClassFor(MVT VT) const {
11356 if (VT == MVT::Untyped)
11357 return &SystemZ::ADDR128BitRegClass;
11359}
11360
11361SDValue SystemZTargetLowering::lowerGET_ROUNDING(SDValue Op,
11362 SelectionDAG &DAG) const {
11363 SDLoc dl(Op);
11364 /*
11365 The rounding method is in FPC Byte 3 bits 6-7, and has the following
11366 settings:
11367 00 Round to nearest
11368 01 Round to 0
11369 10 Round to +inf
11370 11 Round to -inf
11371
11372 FLT_ROUNDS, on the other hand, expects the following:
11373 -1 Undefined
11374 0 Round to 0
11375 1 Round to nearest
11376 2 Round to +inf
11377 3 Round to -inf
11378 */
11379
11380 // Save FPC to register.
11381 SDValue Chain = Op.getOperand(0);
11382 SDValue EFPC(
11383 DAG.getMachineNode(SystemZ::EFPC, dl, {MVT::i32, MVT::Other}, Chain), 0);
11384 Chain = EFPC.getValue(1);
11385
11386 // Transform as necessary
11387 SDValue CWD1 = DAG.getNode(ISD::AND, dl, MVT::i32, EFPC,
11388 DAG.getConstant(3, dl, MVT::i32));
11389 // RetVal = (CWD1 ^ (CWD1 >> 1)) ^ 1
11390 SDValue CWD2 = DAG.getNode(ISD::XOR, dl, MVT::i32, CWD1,
11391 DAG.getNode(ISD::SRL, dl, MVT::i32, CWD1,
11392 DAG.getConstant(1, dl, MVT::i32)));
11393
11394 SDValue RetVal = DAG.getNode(ISD::XOR, dl, MVT::i32, CWD2,
11395 DAG.getConstant(1, dl, MVT::i32));
11396 RetVal = DAG.getZExtOrTrunc(RetVal, dl, Op.getValueType());
11397
11398 return DAG.getMergeValues({RetVal, Chain}, dl);
11399}
11400
11401SDValue SystemZTargetLowering::lowerVECREDUCE_ADD(SDValue Op,
11402 SelectionDAG &DAG) const {
11403 EVT VT = Op.getValueType();
11404 Op = Op.getOperand(0);
11405 EVT OpVT = Op.getValueType();
11406
11407 assert(OpVT.isVector() && "Operand type for VECREDUCE_ADD is not a vector.");
11408
11409 SDLoc DL(Op);
11410
11411 // load a 0 vector for the third operand of VSUM.
11412 SDValue Zero = DAG.getSplatBuildVector(OpVT, DL, DAG.getConstant(0, DL, VT));
11413
11414 // execute VSUM.
11415 switch (OpVT.getScalarSizeInBits()) {
11416 case 8:
11417 case 16:
11418 Op = DAG.getNode(SystemZISD::VSUM, DL, MVT::v4i32, Op, Zero);
11419 [[fallthrough]];
11420 case 32:
11421 case 64:
11422 Op = DAG.getNode(SystemZISD::VSUM, DL, MVT::i128, Op,
11423 DAG.getBitcast(Op.getValueType(), Zero));
11424 break;
11425 case 128:
11426 break; // VSUM over v1i128 should not happen and would be a noop
11427 default:
11428 llvm_unreachable("Unexpected scalar size.");
11429 }
11430 // Cast to original vector type, retrieve last element.
11431 return DAG.getNode(
11432 ISD::EXTRACT_VECTOR_ELT, DL, VT, DAG.getBitcast(OpVT, Op),
11433 DAG.getConstant(OpVT.getVectorNumElements() - 1, DL, MVT::i32));
11434}
11435
11437 FunctionType *FT = F->getFunctionType();
11438 const AttributeList &Attrs = F->getAttributes();
11439 if (Attrs.hasRetAttrs())
11440 OS << Attrs.getAsString(AttributeList::ReturnIndex) << " ";
11441 OS << *F->getReturnType() << " @" << F->getName() << "(";
11442 for (unsigned I = 0, E = FT->getNumParams(); I != E; ++I) {
11443 if (I)
11444 OS << ", ";
11445 OS << *FT->getParamType(I);
11446 AttributeSet ArgAttrs = Attrs.getParamAttrs(I);
11447 for (auto A : {Attribute::SExt, Attribute::ZExt, Attribute::NoExt})
11448 if (ArgAttrs.hasAttribute(A))
11449 OS << " " << Attribute::getNameFromAttrKind(A);
11450 }
11451 OS << ")\n";
11452}
11453
11454bool SystemZTargetLowering::isInternal(const Function *Fn) const {
11455 std::map<const Function *, bool>::iterator Itr = IsInternalCache.find(Fn);
11456 if (Itr == IsInternalCache.end())
11457 Itr = IsInternalCache
11458 .insert(std::pair<const Function *, bool>(
11459 Fn, (Fn->hasLocalLinkage() && !Fn->hasAddressTaken())))
11460 .first;
11461 return Itr->second;
11462}
11463
11464void SystemZTargetLowering::
11465verifyNarrowIntegerArgs_Call(const SmallVectorImpl<ISD::OutputArg> &Outs,
11466 const Function *F, SDValue Callee) const {
11467 // Temporarily only do the check when explicitly requested, until it can be
11468 // enabled by default.
11470 return;
11471
11472 bool IsInternal = false;
11473 const Function *CalleeFn = nullptr;
11474 if (auto *G = dyn_cast<GlobalAddressSDNode>(Callee))
11475 if ((CalleeFn = dyn_cast<Function>(G->getGlobal())))
11476 IsInternal = isInternal(CalleeFn);
11477 if (!IsInternal && !verifyNarrowIntegerArgs(Outs)) {
11478 errs() << "ERROR: Missing extension attribute of passed "
11479 << "value in call to function:\n" << "Callee: ";
11480 if (CalleeFn != nullptr)
11481 printFunctionArgExts(CalleeFn, errs());
11482 else
11483 errs() << "-\n";
11484 errs() << "Caller: ";
11486 llvm_unreachable("");
11487 }
11488}
11489
11490void SystemZTargetLowering::
11491verifyNarrowIntegerArgs_Ret(const SmallVectorImpl<ISD::OutputArg> &Outs,
11492 const Function *F) const {
11493 // Temporarily only do the check when explicitly requested, until it can be
11494 // enabled by default.
11496 return;
11497
11498 if (!isInternal(F) && !verifyNarrowIntegerArgs(Outs)) {
11499 errs() << "ERROR: Missing extension attribute of returned "
11500 << "value from function:\n";
11502 llvm_unreachable("");
11503 }
11504}
11505
11506// Verify that narrow integer arguments are extended as required by the ABI.
11507// Return false if an error is found.
11508bool SystemZTargetLowering::verifyNarrowIntegerArgs(
11509 const SmallVectorImpl<ISD::OutputArg> &Outs) const {
11510 if (!Subtarget.isTargetELF())
11511 return true;
11512
11515 return true;
11516 } else if (!getTargetMachine().Options.VerifyArgABICompliance)
11517 return true;
11518
11519 for (unsigned i = 0; i < Outs.size(); ++i) {
11520 MVT VT = Outs[i].VT;
11521 ISD::ArgFlagsTy Flags = Outs[i].Flags;
11522 if (VT.isInteger()) {
11523 assert((VT == MVT::i32 || VT.getSizeInBits() >= 64) &&
11524 "Unexpected integer argument VT.");
11525 if (VT == MVT::i32 &&
11526 !Flags.isSExt() && !Flags.isZExt() && !Flags.isNoExt())
11527 return false;
11528 }
11529 }
11530
11531 return true;
11532}
11533
11535 Module &M, const LibcallLoweringInfo &Libcalls) const {
11536 StringRef GuardMode = M.getStackProtectorGuard();
11537
11538 // In the TLS case, no symbol needs to be inserted.
11539 if (GuardMode == "tls" || GuardMode.empty())
11540 return;
11541
11542 // Otherwise (in the global case), insert the appropriate global variable.
11544}
return SDValue()
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static msgpack::DocNode getNode(msgpack::DocNode DN, msgpack::Type Type, MCValue Val)
unsigned Imm
unsigned uint64_t
AMDGPU Register Bank Select
static bool isZeroVector(SDValue N)
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis false
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static SDValue convertValVTToLocVT(SelectionDAG &DAG, SDValue Val, const CCValAssign &VA, const SDLoc &DL)
static SDValue convertLocVTToValVT(SelectionDAG &DAG, SDValue Val, const CCValAssign &VA, const SDLoc &DL)
#define Check(C,...)
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Module.h This file contains the declarations for the Module class.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define RegName(no)
static LVOptions Options
Definition LVOptions.cpp:25
static bool isSelectPseudo(MachineInstr &MI)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
static bool isUndef(const MachineInstr &MI)
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
uint64_t High
uint64_t IntrinsicInst * II
#define P(N)
static constexpr MCPhysReg SPReg
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
const char * Msg
This file defines the SmallSet class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
static SDValue getI128Select(SelectionDAG &DAG, const SDLoc &DL, Comparison C, SDValue TrueOp, SDValue FalseOp)
static SmallVector< SDValue, 4 > simplifyAssumingCCVal(SDValue &Val, SDValue &CC, SelectionDAG &DAG)
static void adjustForTestUnderMask(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static void printFunctionArgExts(const Function *F, raw_fd_ostream &OS)
static void adjustForLTGFR(Comparison &C)
static void adjustSubwordCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static SDValue joinDwords(SelectionDAG &DAG, const SDLoc &DL, SDValue Op0, SDValue Op1)
#define CONV(X)
static cl::opt< bool > EnableIntArgExtCheck("argext-abi-check", cl::init(false), cl::desc("Verify that narrow int args are properly extended per the " "SystemZ ABI."))
static bool isOnlyUsedByStores(SDValue StoredVal, SelectionDAG &DAG)
static void lowerGR128Binary(SelectionDAG &DAG, const SDLoc &DL, EVT VT, unsigned Opcode, SDValue Op0, SDValue Op1, SDValue &Even, SDValue &Odd)
static void adjustForRedundantAnd(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static SDValue lowerAddrSpaceCast(SDValue Op, SelectionDAG &DAG)
static SDValue buildScalarToVector(SelectionDAG &DAG, const SDLoc &DL, EVT VT, SDValue Value)
static SDValue lowerI128ToGR128(SelectionDAG &DAG, SDValue In)
static bool isSimpleShift(SDValue N, unsigned &ShiftVal)
static SDValue mergeHighParts(SelectionDAG &DAG, const SDLoc &DL, unsigned MergedBits, EVT VT, SDValue Op0, SDValue Op1)
static bool isI128MovedToParts(LoadSDNode *LD, SDNode *&LoPart, SDNode *&HiPart)
static bool chooseShuffleOpNos(int *OpNos, unsigned &OpNo0, unsigned &OpNo1)
static uint32_t findZeroVectorIdx(SDValue *Ops, unsigned Num)
static bool isVectorElementSwap(ArrayRef< int > M, EVT VT)
static void getCSAddressAndShifts(SDValue Addr, SelectionDAG &DAG, SDLoc DL, SDValue &AlignedAddr, SDValue &BitShift, SDValue &NegBitShift)
static bool isShlDoublePermute(const SmallVectorImpl< int > &Bytes, unsigned &StartIndex, unsigned &OpNo0, unsigned &OpNo1)
static SDValue getPermuteNode(SelectionDAG &DAG, const SDLoc &DL, const Permute &P, SDValue Op0, SDValue Op1)
static SDNode * emitIntrinsicWithCCAndChain(SelectionDAG &DAG, SDValue Op, unsigned Opcode)
static SDValue getCCResult(SelectionDAG &DAG, SDValue CCReg)
static void adjustForStackGuardCompare(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static bool isIntrinsicWithCCAndChain(SDValue Op, unsigned &Opcode, unsigned &CCValid)
static void lowerMUL_LOHI32(SelectionDAG &DAG, const SDLoc &DL, unsigned Extend, SDValue Op0, SDValue Op1, SDValue &Hi, SDValue &Lo)
static bool isF128MovedToParts(LoadSDNode *LD, SDNode *&LoPart, SDNode *&HiPart)
static void createPHIsForSelects(SmallVector< MachineInstr *, 8 > &Selects, MachineBasicBlock *TrueMBB, MachineBasicBlock *FalseMBB, MachineBasicBlock *SinkMBB)
static SDValue getGeneralPermuteNode(SelectionDAG &DAG, const SDLoc &DL, SDValue *Ops, const SmallVectorImpl< int > &Bytes)
static unsigned getVectorComparisonOrInvert(ISD::CondCode CC, CmpMode Mode, bool &Invert)
static unsigned CCMaskForCondCode(ISD::CondCode CC)
static void adjustICmpTruncate(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static void adjustForFNeg(Comparison &C)
static bool isScalarToVector(SDValue Op)
static SDValue emitSETCC(SelectionDAG &DAG, const SDLoc &DL, SDValue CCReg, unsigned CCValid, unsigned CCMask)
static bool matchPermute(const SmallVectorImpl< int > &Bytes, const Permute &P, unsigned &OpNo0, unsigned &OpNo1)
static bool isAddCarryChain(SDValue Carry)
static SDValue emitCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static MachineOperand earlyUseOperand(MachineOperand Op)
static bool canUseSiblingCall(const CCState &ArgCCInfo, SmallVectorImpl< CCValAssign > &ArgLocs, SmallVectorImpl< ISD::OutputArg > &Outs)
static bool getzOSCalleeAndADA(SelectionDAG &DAG, SDValue &Callee, SDValue &ADA, SDLoc &DL, SDValue &Chain)
static SDValue convertToF16(SDValue Op, SelectionDAG &DAG)
static bool combineCCMask(SDValue &CCReg, int &CCValid, int &CCMask, SelectionDAG &DAG)
static bool shouldSwapCmpOperands(const Comparison &C)
static bool isNaturalMemoryOperand(SDValue Op, unsigned ICmpType)
static SDValue getADAEntry(SelectionDAG &DAG, SDValue Val, SDLoc DL, unsigned Offset, bool LoadAdr=false)
static SDNode * emitIntrinsicWithCC(SelectionDAG &DAG, SDValue Op, unsigned Opcode)
static void adjustForSubtraction(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static bool getVPermMask(SDValue ShuffleOp, SmallVectorImpl< int > &Bytes)
static const Permute PermuteForms[]
static bool isI128MovedFromParts(SDValue Val, SDValue &LoPart, SDValue &HiPart)
static std::pair< SDValue, int > findCCUse(const SDValue &Val, unsigned Depth=0)
static bool isSubBorrowChain(SDValue Carry)
static void adjustICmp128(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static bool analyzeArgSplit(const SmallVectorImpl< ArgTy > &Args, SmallVector< CCValAssign, 16 > &ArgLocs, unsigned I, MVT &PartVT, unsigned &NumParts)
static APInt getDemandedSrcElements(SDValue Op, const APInt &DemandedElts, unsigned OpNo)
static SDValue getAbsolute(SelectionDAG &DAG, const SDLoc &DL, SDValue Op, bool IsNegative)
static unsigned computeNumSignBitsBinOp(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth, unsigned OpNo)
static SDValue expandBitCastI128ToF128(SelectionDAG &DAG, SDValue Src, const SDLoc &SL)
static SDValue tryBuildVectorShuffle(SelectionDAG &DAG, BuildVectorSDNode *BVN)
static SDValue convertFromF16(SDValue Op, SDLoc DL, SelectionDAG &DAG)
static unsigned getVectorComparison(ISD::CondCode CC, CmpMode Mode)
static SDValue lowerGR128ToI128(SelectionDAG &DAG, SDValue In)
static SDValue MergeInputChains(SDNode *N1, SDNode *N2)
static SDValue expandBitCastF128ToI128(SelectionDAG &DAG, SDValue Src, const SDLoc &SL)
static unsigned getTestUnderMaskCond(unsigned BitSize, unsigned CCMask, uint64_t Mask, uint64_t CmpVal, unsigned ICmpType)
static bool isIntrinsicWithCC(SDValue Op, unsigned &Opcode, unsigned &CCValid)
static SDValue expandV4F32ToV2F64(SelectionDAG &DAG, int Start, const SDLoc &DL, SDValue Op, SDValue Chain)
static Comparison getCmp(SelectionDAG &DAG, SDValue CmpOp0, SDValue CmpOp1, ISD::CondCode Cond, const SDLoc &DL, SDValue Chain=SDValue(), bool IsSignaling=false)
static bool checkCCKill(MachineInstr &MI, MachineBasicBlock *MBB)
static Register forceReg(MachineInstr &MI, MachineOperand &Base, const SystemZInstrInfo *TII)
static bool is32Bit(EVT VT)
static std::pair< unsigned, const TargetRegisterClass * > parseRegisterNumber(StringRef Constraint, const TargetRegisterClass *RC, const unsigned *Map, unsigned Size)
static unsigned detectEvenOddMultiplyOperand(const SelectionDAG &DAG, const SystemZSubtarget &Subtarget, SDValue &Op)
static bool matchDoublePermute(const SmallVectorImpl< int > &Bytes, const Permute &P, SmallVectorImpl< int > &Transform)
static Comparison getIntrinsicCmp(SelectionDAG &DAG, unsigned Opcode, SDValue Call, unsigned CCValid, uint64_t CC, ISD::CondCode Cond)
static SDValue buildFPVecFromScalars4(SelectionDAG &DAG, const SDLoc &DL, EVT VT, SmallVectorImpl< SDValue > &Elems, unsigned Pos)
static bool isAbsolute(SDValue CmpOp, SDValue Pos, SDValue Neg)
static AddressingMode getLoadStoreAddrMode(bool HasVector, Type *Ty)
static SDValue buildMergeScalars(SelectionDAG &DAG, const SDLoc &DL, EVT VT, SDValue Op0, SDValue Op1)
static void computeKnownBitsBinOp(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth, unsigned OpNo)
static bool getShuffleInput(const SmallVectorImpl< int > &Bytes, unsigned Start, unsigned BytesPerElement, int &Base)
static AddressingMode supportedAddressingMode(Instruction *I, bool HasVector)
static bool isF128MovedFromParts(SDValue Val, SDValue &LoPart, SDValue &HiPart)
static void adjustZeroCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
BinaryOperator * Mul
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:231
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:226
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1561
unsigned getActiveBits() const
Compute the number of active bits in the value.
Definition APInt.h:1533
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1351
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:255
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1509
bool isSingleWord() const
Determine if this APInt just has one word to store value.
Definition APInt.h:319
LLVM_ABI void insertBits(const APInt &SubBits, unsigned bitPosition)
Insert the bits from a smaller APInt starting at bitPosition.
Definition APInt.cpp:393
bool isSubsetOf(const APInt &RHS) const
This operation checks that all bits set in this APInt are also set in RHS.
Definition APInt.h:1262
void lshrInPlace(unsigned ShiftAmt)
Logical right-shift this APInt by ShiftAmt in place.
Definition APInt.h:861
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:854
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
an instruction that atomically reads a memory location, combines it with another value,...
@ Add
*p = old + v
@ Sub
*p = old - v
@ And
*p = old & v
@ Xor
*p = old ^ v
BinOp getOperation() const
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:410
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Return true if the attribute exists in this set.
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
static LLVM_ABI StringRef getNameFromAttrKind(Attribute::AttrKind AttrKind)
LLVM Basic Block Representation.
Definition BasicBlock.h:62
A "pseudo-class" with methods for operating on BUILD_VECTORs.
LLVM_ABI bool isConstantSplat(APInt &SplatValue, APInt &SplatUndef, unsigned &SplatBitSize, bool &HasAnyUndefs, unsigned MinSplatBits=0, bool isBigEndian=false) const
Check if this is a constant splat, and if so, find the smallest element size that splats the vector.
LLVM_ABI bool isConstant() const
CCState - This class holds information needed while lowering arguments and return values.
LLVM_ABI void AnalyzeCallResult(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeCallResult - Analyze the return values of a call, incorporating info about the passed values i...
LLVM_ABI bool CheckReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
CheckReturn - Analyze the return values of a function, returning true if the return can be performed ...
LLVM_ABI void AnalyzeReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeReturn - Analyze the returned values of a return, incorporating info about the result values i...
LLVM_ABI void AnalyzeCallOperands(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeCallOperands - Analyze the outgoing arguments to a call, incorporating info about the passed v...
uint64_t getStackSize() const
Returns the size of the currently allocated portion of the stack.
LLVM_ABI void AnalyzeFormalArguments(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeFormalArguments - Analyze an array of argument values, incorporating info about the formals in...
CCValAssign - Represent assignment of one arg/retval to a location.
Register getLocReg() const
LocInfo getLocInfo() const
bool needsCustom() const
bool isExtInLoc() const
int64_t getLocMemOffset() const
This class represents a function call, abstracting a target machine's calling convention.
bool isTailCall() const
MachineConstantPoolValue * getMachineCPVal() const
const Constant * getConstVal() const
uint64_t getZExtValue() const
This is an important base class in LLVM.
Definition Constant.h:43
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:251
iterator end()
Definition DenseMap.h:169
bool hasAddressTaken(const User **=nullptr, bool IgnoreCallbackUses=false, bool IgnoreAssumeLikeCalls=true, bool IngoreLLVMUsed=false, bool IgnoreARCAttachedCall=false, bool IgnoreCastedDirectCall=false) const
hasAddressTaken - returns true if there are any uses of this function other than direct calls or invo...
Definition Function.cpp:940
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
Definition Function.cpp:765
uint64_t getFnAttributeAsParsedInteger(StringRef Kind, uint64_t Default=0) const
For a string attribute Kind, parse attribute as an integer.
Definition Function.cpp:777
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:730
LLVM_ABI const GlobalObject * getAliaseeObject() const
Definition Globals.cpp:730
bool hasLocalLinkage() const
bool hasPrivateLinkage() const
bool hasInternalLinkage() const
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Tracks which library functions to use for a particular subtarget or function.
An instruction for reading from memory.
This class is used to represent ISD::LOAD nodes.
Machine Value Type.
static auto integer_fixedlen_vector_valuetypes()
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
static auto integer_valuetypes()
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
static auto fixedlen_vector_valuetypes()
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
static MVT getVectorVT(MVT VT, unsigned NumElements)
static MVT getIntegerVT(unsigned BitWidth)
static auto fp_valuetypes()
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
void addLiveIn(MCRegister PhysReg, LaneBitmask LaneMask=LaneBitmask::getAll())
Adds the specified register as a live in.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
void setMachineBlockAddressTaken()
Set this block to indicate that its address is used as something other than the target of a terminato...
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
void setMaxCallFrameSize(uint64_t S)
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
void setFrameAddressIsTaken(bool T)
uint64_t getMaxCallFrameSize() const
Return the maximum size of a call frame that must be allocated for an outgoing function call.
void setReturnAddressIsTaken(bool s)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_back(MachineBasicBlock *MBB)
reverse_iterator rbegin()
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const MachineFunctionProperties & getProperties() const
Get the function properties.
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const MachineInstrBuilder & setMemRefs(ArrayRef< MachineMemOperand * > MMOs) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addRegMask(const uint32_t *Mask) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
bool killsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr kills the specified register.
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOLoad
The memory access reads data.
@ MOInvariant
The memory access always returns the same value (or traps).
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
Register getReg() const
getReg - Returns the register number.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
void addLiveIn(MCRegister Reg, Register vreg=Register())
addLiveIn - Add the specified register as a live-in.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
SDNodeFlags getFlags() const
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getNumValues() const
Return the number of values defined/returned by this operator.
unsigned getNumOperands() const
Return the number of values used by this operation.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
bool hasNUsesOfValue(unsigned NUses, unsigned Value) const
Return true if there are exactly NUSES uses of the indicated value.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
iterator_range< user_iterator > users()
void setFlags(SDNodeFlags NewFlags)
Represents a use of a SDNode.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
bool isMachineOpcode() const
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
const APInt & getConstantOperandAPInt(unsigned i) const
uint64_t getScalarValueSizeInBits() const
unsigned getResNo() const
get the index which selects a specific result in the SDNode
uint64_t getConstantOperandVal(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getMachineOpcode() const
unsigned getOpcode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
SDValue getExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT, unsigned Opcode)
Convert Op, which must be of integer type, to the integer type VT, by either any/sign/zero-extending ...
const TargetSubtargetInfo & getSubtarget() const
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDValue getAtomicLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT MemVT, EVT VT, SDValue Chain, SDValue Ptr, MachineMemOperand *MMO)
LLVM_ABI SDValue getConstantPool(const Constant *C, EVT VT, MaybeAlign Align=std::nullopt, int Offs=0, bool isT=false, unsigned TargetFlags=0)
LLVM_ABI bool isConstantIntBuildVectorOrConstantInt(SDValue N, bool AllowOpaques=true) const
Test whether the given value is a constant int or similar node.
LLVM_ABI SDValue UnrollVectorOp(SDNode *N, unsigned ResNE=0)
Utility function used by legalize and lowering to "unroll" a vector operation by splitting out the sc...
LLVM_ABI SDValue getAddrSpaceCast(const SDLoc &dl, EVT VT, SDValue Ptr, unsigned SrcAS, unsigned DestAS, const SDNodeFlags Flags=SDNodeFlags())
Return an AddrSpaceCastSDNode.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
SDValue getGLOBAL_OFFSET_TABLE(EVT VT)
Return a GLOBAL_OFFSET_TABLE node. This does not have a useful SDLoc.
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
LLVM_ABI SDValue getAtomic(unsigned Opcode, const SDLoc &dl, EVT MemVT, SDValue Chain, SDValue Ptr, SDValue Val, MachineMemOperand *MMO)
Gets a node for an atomic op, produces result (if relevant) and chain and takes 2 operands.
void addNoMergeSiteInfo(const SDNode *Node, bool NoMerge)
Set NoMergeSiteInfo to be associated with Node if NoMerge is true.
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
LLVM_ABI SDValue getMemcpy(SDValue Chain, const SDLoc &dl, SDValue Dst, SDValue Src, SDValue Size, Align DstAlign, Align SrcAlign, bool isVol, bool AlwaysInline, const CallInst *CI, std::optional< bool > OverrideTailCall, MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo, const AAMDNodes &AAInfo=AAMDNodes(), BatchAAResults *BatchAA=nullptr)
const TargetLowering & getTargetLoweringInfo() const
SDValue getTargetJumpTable(int JTI, EVT VT, unsigned TargetFlags=0)
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI bool isSplatValue(SDValue V, const APInt &DemandedElts, APInt &UndefElts, unsigned Depth=0) const
Test whether V has a splatted value for all the demanded elements.
LLVM_ABI SDValue getTruncStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, SDValue Offset, MachinePointerInfo PtrInfo, EVT SVT, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
const DataLayout & getDataLayout() const
SDValue getTargetFrameIndex(int FI, EVT VT)
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
SDValue getSignedTargetConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getSplatVector(EVT VT, const SDLoc &DL, SDValue Op)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
LLVM_ABI bool SignBitIsZero(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero.
LLVM_ABI SDValue getTargetExtractSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand)
A convenience function for creating TargetInstrInfo::EXTRACT_SUBREG nodes.
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI SDValue getExternalSymbol(const char *Sym, EVT VT)
const TargetMachine & getTarget() const
LLVM_ABI std::pair< SDValue, SDValue > getStrictFPExtendOrRound(SDValue Op, SDValue Chain, const SDLoc &DL, EVT VT)
Convert Op, which must be a STRICT operation of float type, to the float type VT, by either extending...
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI SDValue getFPExtendOrRound(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of float type, to the float type VT, by either extending or rounding (by tr...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
SDValue getTargetBlockAddress(const BlockAddress *BA, EVT VT, int64_t Offset=0, unsigned TargetFlags=0)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getSplatBuildVector(EVT VT, const SDLoc &DL, SDValue Op)
Return a splat ISD::BUILD_VECTOR node, consisting of Op splatted to all elements.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getRegisterMask(const uint32_t *RegMask)
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
SDValue getObjectPtrOffset(const SDLoc &SL, SDValue Ptr, TypeSize Offset)
Create an add instruction with appropriate flags when used for addressing some offset of an object.
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetExternalSymbol(const char *Sym, EVT VT, unsigned TargetFlags=0)
LLVM_ABI SDValue CreateStackTemporary(TypeSize Bytes, Align Alignment)
Create a stack temporary based on the size in bytes and the alignment.
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
SDValue getTargetConstantPool(const Constant *C, EVT VT, MaybeAlign Align=std::nullopt, int Offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getTargetInsertSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand, SDValue Subreg)
A convenience function for creating TargetInstrInfo::INSERT_SUBREG nodes.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
This SDNode is used to implement the code generator support for the llvm IR shufflevector instruction...
ArrayRef< int > getMask() const
const_iterator begin() const
Definition SmallSet.h:216
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
size_type size() const
Definition SmallSet.h:171
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void resize(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
This class is used to represent ISD::STORE nodes.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
Definition StringRef.h:258
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
StringRef slice(size_t Start, size_t End) const
Return a reference to the substring from [Start, End).
Definition StringRef.h:720
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
iterator end() const
Definition StringRef.h:116
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
A SystemZ-specific class detailing special use registers particular for calling conventions.
static SystemZConstantPoolValue * Create(const GlobalValue *GV, SystemZCP::SystemZCPModifier Modifier)
const SystemZInstrInfo * getInstrInfo() const override
SystemZCallingConventionRegisters * getSpecialRegisters() const
AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const override
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
EVT getOptimalMemOpType(LLVMContext &Context, const MemOp &Op, const AttributeList &FuncAttributes) const override
Returns the target specific optimal type for load and store operations as a result of memset,...
bool hasInlineStackProbe(const MachineFunction &MF) const override
Returns true if stack probing through inline assembly is requested.
MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *BB) const override
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
MachineBasicBlock * emitEHSjLjSetJmp(MachineInstr &MI, MachineBasicBlock *MBB) const
AtomicExpansionKind shouldCastAtomicLoadInIR(LoadInst *LI) const override
Returns how the given (atomic) load should be cast by the IR-level AtomicExpand pass.
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &, EVT) const override
Return the ValueType of the result of SETCC operations.
bool allowTruncateForTailCall(Type *, Type *) const override
Return true if a truncation from FromTy to ToTy is permitted when deciding whether a call is in tail ...
SDValue LowerAsmOutputForConstraint(SDValue &Chain, SDValue &Flag, const SDLoc &DL, const AsmOperandInfo &Constraint, SelectionDAG &DAG) const override
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
MachineBasicBlock * emitEHSjLjLongJmp(MachineInstr &MI, MachineBasicBlock *MBB) const
bool CanLowerReturn(CallingConv::ID CallConv, MachineFunction &MF, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, LLVMContext &Context, const Type *RetTy) const override
This hook should be implemented to check whether the return values described by the Outs array can fi...
std::pair< SDValue, SDValue > makeExternalCall(SDValue Chain, SelectionDAG &DAG, const char *CalleeName, EVT RetVT, ArrayRef< SDValue > Ops, CallingConv::ID CallConv, bool IsSigned, SDLoc DL, bool DoesNotReturn, bool IsReturnValueUsed) const
void insertSSPDeclarations(Module &M, const LibcallLoweringInfo &Libcalls) const override
Insert SSP declaration if global stack protector is used.
bool mayBeEmittedAsTailCall(const CallInst *CI) const override
Return true if the target may be able emit the call instruction as a tail call.
bool splitValueIntoRegisterParts(SelectionDAG &DAG, const SDLoc &DL, SDValue Val, SDValue *Parts, unsigned NumParts, MVT PartVT, std::optional< CallingConv::ID > CC) const override
Target-specific splitting of values into parts that fit a register storing a legal type.
bool isLegalAddressingMode(const DataLayout &DL, const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I=nullptr) const override
Return true if the addressing mode represented by AM is legal for this target, for a load/store of th...
unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain targets require unusual breakdowns of certain types.
bool isGuaranteedNotToBeUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, unsigned Depth) const override
Return true if this function can prove that Op is never poison and, Kind can be used to track poison ...
SystemZTargetLowering(const TargetMachine &TM, const SystemZSubtarget &STI)
bool isFMAFasterThanFMulAndFAdd(const MachineFunction &MF, EVT VT) const override
Return true if an FMA operation is faster than a pair of fmul and fadd instructions.
bool isLegalICmpImmediate(int64_t Imm) const override
Return true if the specified immediate is legal icmp immediate, that is the target has icmp instructi...
std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const override
Given a physical register constraint (e.g.
TargetLowering::ConstraintWeight getSingleConstraintMatchWeight(AsmOperandInfo &info, const char *constraint) const override
Examine constraint string and operand type and determine a weight value.
bool allowsMisalignedMemoryAccesses(EVT VT, unsigned AS, Align Alignment, MachineMemOperand::Flags Flags, unsigned *Fast) const override
Determine if the target supports unaligned memory accesses.
const MCPhysReg * getScratchRegisters(CallingConv::ID CC) const override
Returns a 0 terminated array of registers that can be safely used as scratch registers.
TargetLowering::ConstraintType getConstraintType(StringRef Constraint) const override
Given a constraint, return the type of constraint it is for this target.
Register getExceptionSelectorRegister(ExceptionHandling EH, const Constant *PersonalityFn) const override
If a physical register, this returns the register that receives the exception typeid on entry to a la...
bool isFPImmLegal(const APFloat &Imm, EVT VT, bool ForCodeSize) const override
Returns true if the target can instruction select the specified FP immediate natively.
SDValue joinRegisterPartsIntoValue(SelectionDAG &DAG, const SDLoc &DL, const SDValue *Parts, unsigned NumParts, MVT PartVT, EVT ValueVT, std::optional< CallingConv::ID > CC) const override
Target-specific combining of register parts into its original value.
bool isTruncateFree(Type *, Type *) const override
Return true if it's free to truncate a value of type FromTy to type ToTy.
SDValue useLibCall(SelectionDAG &DAG, RTLIB::Libcall LC, MVT VT, SDValue Arg, SDLoc DL, SDValue Chain, bool IsStrict) const
unsigned ComputeNumSignBitsForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth) const override
Determine the number of bits in the operation that are sign bits.
void LowerOperationWrapper(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked by the type legalizer to legalize nodes with an illegal operand type but leg...
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
SDValue LowerCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
bool isLegalAddImmediate(int64_t Imm) const override
Return true if the specified immediate is legal add immediate, that is the target has add instruction...
CondMergingParams getJumpConditionMergingParams(Instruction::BinaryOps Opc, const Value *Lhs, const Value *Rhs, const Function *F) const override
bool findOptimalMemOpLowering(LLVMContext &Context, std::vector< EVT > &MemOps, unsigned Limit, const MemOp &Op, unsigned DstAS, unsigned SrcAS, const AttributeList &FuncAttributes, EVT *LargestVT=nullptr) const override
Determines the optimal series of memory ops to replace the memset / memcpy.
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const override
Lower the specified operand into the Ops vector.
Register getExceptionPointerRegister(ExceptionHandling EH, const Constant *PersonalityFn) const override
If a physical register, this returns the register that receives the exception address on entry to an ...
unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const override
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
AtomicExpansionKind shouldCastAtomicStoreInIR(StoreInst *SI) const override
Returns how the given (atomic) store should be cast by the IR-level AtomicExpand pass into.
Register getRegisterByName(const char *RegName, LLT VT, const MachineFunction &MF) const override
Return the register ID of the name passed in.
bool hasAndNot(SDValue Y) const override
Return true if the target has a bitwise and-not operation: X = ~A & B This can be used to simplify se...
SDValue LowerFormalArguments(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower the incoming (formal) arguments, described by the Ins array,...
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
unsigned getStackProbeSize(const MachineFunction &MF) const
XPLINK64 calling convention specific use registers Particular to z/OS when in 64 bit mode.
Information about stack frame layout on the target.
unsigned getStackAlignment() const
getStackAlignment - This method returns the number of bytes to which the stack pointer must be aligne...
bool hasFP(const MachineFunction &MF) const
hasFP - Return true if the specified function should have a dedicated frame pointer register.
TargetInstrInfo - Interface to description of machine instruction set.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
unsigned MaxStoresPerMemcpyOptSize
Likewise for functions with the OptSize attribute.
MachineBasicBlock * emitPatchPoint(MachineInstr &MI, MachineBasicBlock *MBB) const
Replace/modify any TargetFrameIndex operands with a targte-dependent sequence of memory operands that...
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
virtual void insertSSPDeclarations(Module &M, const LibcallLoweringInfo &Libcalls) const
Inserts necessary declarations for SSP (stack protection) purpose.
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
void setAtomicLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Let target indicate that an extending atomic load of the specified type is legal.
virtual unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
Register getStackPointerRegisterToSaveRestore() const
If a physical register, this specifies the register that llvm.savestack/llvm.restorestack should save...
void setMinFunctionAlignment(Align Alignment)
Set the target's minimum function alignment.
unsigned MaxStoresPerMemsetOptSize
Likewise for functions with the OptSize attribute.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
unsigned MaxStoresPerMemmove
Specify maximum number of store instructions per memmove call.
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
unsigned MaxStoresPerMemmoveOptSize
Likewise for functions with the OptSize attribute.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setPrefFunctionAlignment(Align Alignment)
Set the target's preferred function alignment.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
unsigned MaxStoresPerMemset
Specify maximum number of store instructions per memset call.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
virtual const TargetRegisterClass * getRepRegClassFor(MVT VT) const
Return the 'representative' register class for the specified value type.
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
virtual bool shouldSignExtendTypeInLibCall(Type *Ty, bool IsSigned) const
Returns true if arguments should be sign-extended in lib calls.
std::vector< ArgListEntry > ArgListTy
unsigned MaxStoresPerMemcpy
Specify maximum number of store instructions per memcpy call.
virtual MVT getPointerMemTy(const DataLayout &DL, uint32_t AS=0) const
Return the in-memory pointer type for the given address space, defaults to the pointer type from the ...
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
LegalizeAction getOperationAction(unsigned Op, EVT VT) const
Return how this operation should be treated: either it is legal, needs to be promoted to a larger siz...
virtual ConstraintType getConstraintType(StringRef Constraint) const
Given a constraint, return the type of constraint it is for this target.
virtual bool findOptimalMemOpLowering(LLVMContext &Context, std::vector< EVT > &MemOps, unsigned Limit, const MemOp &Op, unsigned DstAS, unsigned SrcAS, const AttributeList &FuncAttributes, EVT *LargestVT=nullptr) const
Determines the optimal series of memory ops to replace the memset / memcpy.
virtual SDValue LowerToTLSEmulatedModel(const GlobalAddressSDNode *GA, SelectionDAG &DAG) const
Lower TLS global address SDNode for target independent emulated TLS model.
std::pair< SDValue, SDValue > LowerCallTo(CallLoweringInfo &CLI) const
This function lowers an abstract call to a function into an actual call.
virtual ConstraintWeight getSingleConstraintMatchWeight(AsmOperandInfo &info, const char *constraint) const
Examine constraint string and operand type and determine a weight value.
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
TargetLowering(const TargetLowering &)=delete
virtual void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const
Lower the specified operand into the Ops vector.
std::pair< SDValue, SDValue > makeLibCall(SelectionDAG &DAG, RTLIB::LibcallImpl LibcallImpl, EVT RetVT, ArrayRef< SDValue > Ops, MakeLibCallOptions CallOptions, const SDLoc &dl, SDValue Chain=SDValue()) const
Returns a pair of (return value, chain).
Primary interface to the complete machine description for the target machine.
TLSModel::Model getTLSModel(const GlobalValue *GV) const
Returns the TLS model which should be used for the given global variable.
bool useEmulatedTLS() const
Returns true if this target uses emulated TLS.
unsigned getPointerSize(unsigned AS) const
Get the pointer size for this target.
CodeModel::Model getCodeModel() const
Returns the code model.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
virtual const TargetInstrInfo * getInstrInfo() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
User * getUser() const
Returns the User that contains this Use.
Definition Use.h:61
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
user_iterator user_begin()
Definition Value.h:404
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
int getNumOccurrences() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
A raw_ostream that writes to a file descriptor.
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ GHC
Used by the Glasgow Haskell Compiler (GHC).
Definition CallingConv.h:50
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
bool isNON_EXTLoad(const SDNode *N)
Returns true if the specified node is a non-extending load.
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:829
@ MERGE_VALUES
MERGE_VALUES - This node takes multiple discrete operands and returns them all as its individual resu...
Definition ISDOpcodes.h:261
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ STRICT_FSETCC
STRICT_FSETCC/STRICT_FSETCCS - Constrained versions of SETCC, used for floating-point operands only.
Definition ISDOpcodes.h:513
@ POISON
POISON - A poison node.
Definition ISDOpcodes.h:236
@ EH_SJLJ_LONGJMP
OUTCHAIN = EH_SJLJ_LONGJMP(INCHAIN, buffer) This corresponds to the eh.sjlj.longjmp intrinsic.
Definition ISDOpcodes.h:168
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:275
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:789
@ VAEND
VAEND, VASTART - VAEND and VASTART have three operands: an input chain, pointer, and a SRCVALUE.
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:863
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:520
@ PSEUDO_FMIN
PSEUDO_FMIN is strictly equivalent to op0 olt op1 ?
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:220
@ GlobalAddress
Definition ISDOpcodes.h:88
@ ATOMIC_CMP_SWAP_WITH_SUCCESS
Val, Success, OUTCHAIN = ATOMIC_CMP_SWAP_WITH_SUCCESS(INCHAIN, ptr, cmp, swap) N.b.
@ STRICT_FMINIMUM
Definition ISDOpcodes.h:473
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:890
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:417
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:749
@ MEMBARRIER
MEMBARRIER - Compiler barrier only; generate a no-op.
@ ATOMIC_FENCE
OUTCHAIN = ATOMIC_FENCE(INCHAIN, ordering, scope) This corresponds to the fence instruction.
@ SIGN_EXTEND_VECTOR_INREG
SIGN_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register sign-extension of the low ...
Definition ISDOpcodes.h:920
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:280
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ STRICT_PSEUDO_FMAX
Definition ISDOpcodes.h:462
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:254
@ STRICT_FSQRT
Constrained versions of libm-equivalent floating point intrinsics.
Definition ISDOpcodes.h:438
@ BUILTIN_OP_END
BUILTIN_OP_END - This must be the last enum value in this list.
@ GlobalTLSAddress
Definition ISDOpcodes.h:89
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:798
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ STRICT_UINT_TO_FP
Definition ISDOpcodes.h:487
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:667
@ PREFETCH
PREFETCH - This corresponds to a prefetch intrinsic.
@ STRICT_PSEUDO_FMIN
Definition ISDOpcodes.h:461
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BR_CC
BR_CC - Conditional branch.
@ SSUBO
Same for subtraction.
Definition ISDOpcodes.h:352
@ BR_JT
BR_JT - Jumptable branch.
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
Definition ISDOpcodes.h:550
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:806
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:233
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:247
@ SPLAT_VECTOR
SPLAT_VECTOR(VAL) - Returns a vector with the scalar value VAL duplicated in all lanes.
Definition ISDOpcodes.h:674
@ VACOPY
VACOPY - VACOPY has 5 operands: an input chain, a destination pointer, a source pointer,...
@ SADDO
RESULT, BOOL = [SU]ADDO(LHS, RHS) - Overflow-aware nodes for addition.
Definition ISDOpcodes.h:348
@ VECREDUCE_ADD
Integer reductions may have a result type larger than the vector element type.
@ GET_ROUNDING
Returns current rounding mode: -1 Undefined 0 Round to 0 1 Round to nearest, ties to even 2 Round to ...
Definition ISDOpcodes.h:980
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:706
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:771
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:651
@ STRICT_FMAXIMUM
Definition ISDOpcodes.h:472
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:578
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:860
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:821
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ ANY_EXTEND_VECTOR_INREG
ANY_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register any-extension of the low la...
Definition ISDOpcodes.h:909
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:898
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:729
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:988
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:815
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:328
@ STRICT_SINT_TO_FP
STRICT_[US]INT_TO_FP - Convert a signed or unsigned integer to a floating point value.
Definition ISDOpcodes.h:486
@ STRICT_FROUNDEVEN
Definition ISDOpcodes.h:466
@ FRAMEADDR
FRAMEADDR, RETURNADDR - These nodes represent llvm.frameaddress and llvm.returnaddress on the DAG.
Definition ISDOpcodes.h:110
@ STRICT_FP_TO_UINT
Definition ISDOpcodes.h:480
@ STRICT_FP_ROUND
X = STRICT_FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision ...
Definition ISDOpcodes.h:502
@ STRICT_FP_TO_SINT
STRICT_FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:479
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:936
@ READCYCLECOUNTER
READCYCLECOUNTER - This corresponds to the readcyclecounter intrinsic.
@ STRICT_FP_EXTEND
X = STRICT_FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:507
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:741
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:205
@ STRICT_FADD
Constrained versions of the binary floating point operators.
Definition ISDOpcodes.h:427
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:567
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:53
@ ATOMIC_SWAP
Val, OUTCHAIN = ATOMIC_SWAP(INCHAIN, ptr, amt) Val, OUTCHAIN = ATOMIC_LOAD_[OpName](INCHAIN,...
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:797
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ ZERO_EXTEND_VECTOR_INREG
ZERO_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register zero-extension of the low ...
Definition ISDOpcodes.h:931
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ STRICT_FNEARBYINT
Definition ISDOpcodes.h:458
@ EH_SJLJ_SETJMP
RESULT, OUTCHAIN = EH_SJLJ_SETJMP(INCHAIN, buffer) This corresponds to the eh.sjlj....
Definition ISDOpcodes.h:162
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
@ BRCOND
BRCOND - Conditional branch.
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:843
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:62
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:536
@ GET_DYNAMIC_AREA_OFFSET
GET_DYNAMIC_AREA_OFFSET - get offset from native SP to the address of the most recent dynamic alloca.
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:213
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:558
bool isNormalStore(const SDNode *N)
Returns true if the specified node is a non-truncating and unindexed store.
LLVM_ABI bool isConstantSplatVectorAllZeros(const SDNode *N, bool BuildVectorOnly=false)
Return true if the specified node is a BUILD_VECTOR or SPLAT_VECTOR where all of the elements are 0 o...
LLVM_ABI CondCode getSetCCInverse(CondCode Operation, EVT Type)
Return the operation corresponding to !(X op Y), where 'op' is a valid SetCC operation.
LLVM_ABI CondCode getSetCCSwappedOperands(CondCode Operation)
Return the operation corresponding to (Y op X) when given the operation for (X op Y).
LLVM_ABI bool isBuildVectorAllZeros(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are 0 or undef.
LLVM_ABI bool isConstantSplatVector(const SDNode *N, APInt &SplatValue)
Node predicates.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
bool isNormalLoad(const SDNode *N)
Returns true if the specified node is a non-extending and unindexed load.
Flag
These should be considered private to the implementation of the MCInstrDesc class.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
LLVM_ABI Libcall getSINTTOFP(EVT OpVT, EVT RetVT)
getSINTTOFP - Return the SINTTOFP_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getUINTTOFP(EVT OpVT, EVT RetVT)
getUINTTOFP - Return the UINTTOFP_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getFPTOSINT(EVT OpVT, EVT RetVT)
getFPTOSINT - Return the FPTOSINT_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
const unsigned GR64Regs[16]
const unsigned VR128Regs[32]
const unsigned VR16Regs[32]
const unsigned GR128Regs[16]
const unsigned FP32Regs[16]
const unsigned FP16Regs[16]
const unsigned GR32Regs[16]
const unsigned FP64Regs[16]
const int64_t ELFCallFrameSize
const unsigned VR64Regs[32]
const unsigned FP128Regs[16]
const unsigned VR32Regs[32]
unsigned odd128(bool Is32bit)
const unsigned CCMASK_CMP_GE
Definition SystemZ.h:41
static bool isImmHH(uint64_t Val)
Definition SystemZ.h:177
const unsigned CCMASK_TEND
Definition SystemZ.h:98
const unsigned CCMASK_CS_EQ
Definition SystemZ.h:68
const unsigned CCMASK_TBEGIN
Definition SystemZ.h:93
const unsigned CCMASK_0
Definition SystemZ.h:28
const MCPhysReg ELFArgFPRs[ELFNumArgFPRs]
MachineBasicBlock * splitBlockBefore(MachineBasicBlock::iterator MI, MachineBasicBlock *MBB)
const unsigned CCMASK_TM_SOME_1
Definition SystemZ.h:83
const unsigned CCMASK_LOGICAL_CARRY
Definition SystemZ.h:61
const unsigned TDCMASK_NORMAL_MINUS
Definition SystemZ.h:123
const unsigned CCMASK_TDC
Definition SystemZ.h:110
const unsigned CCMASK_FCMP
Definition SystemZ.h:49
const unsigned CCMASK_TM_SOME_0
Definition SystemZ.h:82
static bool isImmHL(uint64_t Val)
Definition SystemZ.h:172
const unsigned TDCMASK_SUBNORMAL_MINUS
Definition SystemZ.h:125
const unsigned PFD_READ
Definition SystemZ.h:116
const unsigned CCMASK_1
Definition SystemZ.h:29
const unsigned TDCMASK_NORMAL_PLUS
Definition SystemZ.h:122
const unsigned PFD_WRITE
Definition SystemZ.h:117
const unsigned CCMASK_CMP_GT
Definition SystemZ.h:38
const unsigned TDCMASK_QNAN_MINUS
Definition SystemZ.h:129
const unsigned CCMASK_CS
Definition SystemZ.h:70
const unsigned CCMASK_ANY
Definition SystemZ.h:32
const unsigned CCMASK_ARITH
Definition SystemZ.h:56
const unsigned CCMASK_TM_MIXED_MSB_0
Definition SystemZ.h:79
const unsigned TDCMASK_SUBNORMAL_PLUS
Definition SystemZ.h:124
static bool isImmLL(uint64_t Val)
Definition SystemZ.h:162
const unsigned VectorBits
Definition SystemZ.h:155
static bool isImmLH(uint64_t Val)
Definition SystemZ.h:167
MachineBasicBlock * emitBlockAfter(MachineBasicBlock *MBB)
const unsigned TDCMASK_INFINITY_PLUS
Definition SystemZ.h:126
unsigned reverseCCMask(unsigned CCMask)
const unsigned CCMASK_TM_ALL_0
Definition SystemZ.h:78
const unsigned IPM_CC
Definition SystemZ.h:113
const unsigned CCMASK_CMP_LE
Definition SystemZ.h:40
const unsigned CCMASK_CMP_O
Definition SystemZ.h:45
const unsigned CCMASK_CMP_EQ
Definition SystemZ.h:36
const unsigned VectorBytes
Definition SystemZ.h:159
const unsigned TDCMASK_INFINITY_MINUS
Definition SystemZ.h:127
const unsigned CCMASK_ICMP
Definition SystemZ.h:48
const unsigned CCMASK_VCMP_ALL
Definition SystemZ.h:102
const unsigned CCMASK_VCMP_NONE
Definition SystemZ.h:104
MachineBasicBlock * splitBlockAfter(MachineBasicBlock::iterator MI, MachineBasicBlock *MBB)
const unsigned CCMASK_VCMP
Definition SystemZ.h:105
const unsigned CCMASK_TM_MIXED_MSB_1
Definition SystemZ.h:80
const unsigned CCMASK_TM_MSB_0
Definition SystemZ.h:84
const unsigned CCMASK_ARITH_OVERFLOW
Definition SystemZ.h:55
const unsigned CCMASK_CS_NE
Definition SystemZ.h:69
const unsigned TDCMASK_SNAN_PLUS
Definition SystemZ.h:130
const unsigned CCMASK_TM
Definition SystemZ.h:86
const unsigned CCMASK_3
Definition SystemZ.h:31
const unsigned CCMASK_NONE
Definition SystemZ.h:27
const unsigned CCMASK_CMP_LT
Definition SystemZ.h:37
const unsigned CCMASK_CMP_NE
Definition SystemZ.h:39
const unsigned TDCMASK_ZERO_PLUS
Definition SystemZ.h:120
const unsigned TDCMASK_QNAN_PLUS
Definition SystemZ.h:128
const unsigned TDCMASK_ZERO_MINUS
Definition SystemZ.h:121
unsigned even128(bool Is32bit)
const unsigned CCMASK_TM_ALL_1
Definition SystemZ.h:81
const unsigned CCMASK_LOGICAL_BORROW
Definition SystemZ.h:63
const unsigned ELFNumArgFPRs
const unsigned CCMASK_CMP_UO
Definition SystemZ.h:44
const unsigned CCMASK_LOGICAL
Definition SystemZ.h:65
const unsigned CCMASK_TM_MSB_1
Definition SystemZ.h:85
const unsigned TDCMASK_SNAN_MINUS
Definition SystemZ.h:131
initializer< Ty > init(const Ty &Val)
support::ulittle32_t Word
Definition IRSymtab.h:53
@ User
could "use" a pointer
NodeAddr< UseNode * > Use
Definition RDFGraph.h:385
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
NodeAddr< CodeNode * > Code
Definition RDFGraph.h:388
This is an optimization pass for GlobalISel generic memory operations.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
@ Length
Definition DWP.cpp:577
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
@ Define
Register definition.
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Done
Definition Threading.h:60
@ Load
The value being inserted comes from a load (InsertElement only).
testing::Matcher< const detail::ErrorHolder & > Failed()
Definition Error.h:198
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
constexpr T maskLeadingOnes(unsigned N)
Create a bitmask with the N left-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:89
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
Definition MathExtras.h:244
LLVM_ABI void dumpBytes(ArrayRef< uint8_t > Bytes, raw_ostream &OS)
Convert ‘Bytes’ to a hex string and output to ‘OS’.
T bit_ceil(T Value)
Returns the smallest integral power of two no smaller than Value if Value is nonzero.
Definition bit.h:362
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
LLVM_ABI bool isBitwiseNot(SDValue V, bool AllowUndefs=false)
Returns true if V is a bitwise not operation.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ Success
The lock was released successfully.
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
@ BeforeLegalizeTypes
Definition DAGCombine.h:16
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
@ Fast
Assign the register banks as fast as possible (default).
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
constexpr unsigned BitWidth
ExceptionHandling
Definition CodeGen.h:54
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
UndefPoisonKind
Enumeration to track whether we are interested in Undef, Poison, or both.
Definition UndefPoison.h:20
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
#define EQ(a, b)
Definition regexec.c:65
AddressingMode(bool LongDispl, bool IdxReg)
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
EVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
Definition ValueTypes.h:90
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
bool isRound() const
Return true if the size is a power-of-two number of bytes.
Definition ValueTypes.h:271
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isVectorOf(EVT EltVT) const
Return true if this is a vector with matching element type.
Definition ValueTypes.h:181
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
KnownBits intersectWith(const KnownBits &RHS) const
Returns KnownBits information that is known to be true for both this and RHS.
Definition KnownBits.h:325
APInt getMaxValue() const
Return the maximal unsigned value possible given these KnownBits.
Definition KnownBits.h:146
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getConstantPool(MachineFunction &MF)
Return a MachinePointerInfo record that refers to the constant pool.
MachinePointerInfo getWithOffset(int64_t O) const
static LLVM_ABI MachinePointerInfo getGOT(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a GOT entry.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
SmallVector< unsigned, 2 > OpVals
bool isVectorConstantLegal(const SystemZSubtarget &Subtarget)
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
This contains information for each constraint that we are lowering.
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
CallLoweringInfo & setDiscardResult(bool Value=true)
CallLoweringInfo & setZExtResult(bool Value=true)
CallLoweringInfo & setDebugLoc(const SDLoc &dl)
CallLoweringInfo & setSExtResult(bool Value=true)
CallLoweringInfo & setNoReturn(bool Value=true)
SmallVector< ISD::OutputArg, 32 > Outs
CallLoweringInfo & setChain(SDValue InChain)
CallLoweringInfo & setCallee(CallingConv::ID CC, Type *ResultType, SDValue Target, ArgListTy &&ArgsList, AttributeSet ResultAttrs={})
This structure is used to pass arguments to makeLibCall function.