LLVM 24.0.0git
X86FastISel.cpp
Go to the documentation of this file.
1//===-- X86FastISel.cpp - X86 FastISel implementation ---------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines the X86-specific support for the FastISel class. Much
10// of the target-specific code is generated by tablegen in the file
11// X86GenFastISel.inc, which is #included here.
12//
13//===----------------------------------------------------------------------===//
14
15#include "X86.h"
16#include "X86CallingConv.h"
17#include "X86InstrBuilder.h"
18#include "X86InstrInfo.h"
20#include "X86RegisterInfo.h"
21#include "X86Subtarget.h"
22#include "X86TargetMachine.h"
30#include "llvm/IR/CallingConv.h"
31#include "llvm/IR/DebugInfo.h"
37#include "llvm/IR/IntrinsicsX86.h"
38#include "llvm/IR/Module.h"
39#include "llvm/IR/Operator.h"
40#include "llvm/MC/MCAsmInfo.h"
41#include "llvm/MC/MCSymbol.h"
44using namespace llvm;
45
46namespace {
47
48class X86FastISel final : public FastISel {
49 /// Subtarget - Keep a pointer to the X86Subtarget around so that we can
50 /// make the right decision when generating code for different targets.
51 const X86Subtarget *Subtarget;
52
53public:
54 explicit X86FastISel(FunctionLoweringInfo &funcInfo,
55 const TargetLibraryInfo *libInfo,
56 const LibcallLoweringInfo *libcallLowering)
57 : FastISel(funcInfo, libInfo, libcallLowering) {
58 Subtarget = &funcInfo.MF->getSubtarget<X86Subtarget>();
59 }
60
61 bool fastSelectInstruction(const Instruction *I) override;
62
63 /// The specified machine instr operand is a vreg, and that
64 /// vreg is being provided by the specified load instruction. If possible,
65 /// try to fold the load as an operand to the instruction, returning true if
66 /// possible.
67 bool tryToFoldLoadIntoMI(MachineInstr *MI, unsigned OpNo,
68 const LoadInst *LI) override;
69
70 bool fastLowerArguments() override;
71 bool fastLowerCall(CallLoweringInfo &CLI) override;
72 bool fastLowerIntrinsicCall(const IntrinsicInst *II) override;
73
74#include "X86GenFastISel.inc"
75
76private:
77 bool X86FastEmitCompare(const Value *LHS, const Value *RHS, EVT VT,
78 const DebugLoc &DL);
79
80 bool X86FastEmitLoad(MVT VT, X86AddressMode &AM, MachineMemOperand *MMO,
81 Register &ResultReg, unsigned Alignment = 1);
82
83 bool X86FastEmitStore(EVT VT, const Value *Val, X86AddressMode &AM,
84 MachineMemOperand *MMO = nullptr, bool Aligned = false);
85 bool X86FastEmitStore(EVT VT, Register ValReg, X86AddressMode &AM,
86 MachineMemOperand *MMO = nullptr, bool Aligned = false);
87
88 bool X86FastEmitExtend(ISD::NodeType Opc, EVT DstVT, Register Src, EVT SrcVT,
89 Register &ResultReg);
90
91 /// Emit \p Opc with the register operands \p Op0 and \p Op1, leaving its
92 /// implicit EFLAGS def live for a following SETcc. EFLAGS must be the
93 /// instruction's only implicit def.
94 Register X86FastEmitLiveEFLAGS_rr(unsigned Opc, const TargetRegisterClass *RC,
95 Register Op0, Register Op1);
96
97 /// Emit \p Opc with the register operand \p Op0 and the immediate \p Imm,
98 /// leaving its implicit EFLAGS def live for a following SETcc. EFLAGS must
99 /// be the instruction's only implicit def.
100 Register X86FastEmitLiveEFLAGS_ri(unsigned Opc, const TargetRegisterClass *RC,
101 Register Op0, uint64_t Imm);
102
103 /// Emit a MUL or IMUL of \p LHSReg and \p RHSReg. \p AccReg is the
104 /// accumulator the instruction implicitly reads and implicitly defines with
105 /// the low half of the product. The high half is marked dead, as is EFLAGS
106 /// unless \p LiveEFLAGS is set for a following SETcc.
107 Register X86FastEmitMul(unsigned Opc, MVT VT, MCRegister AccReg,
108 Register LHSReg, Register RHSReg,
109 bool LiveEFLAGS = false);
110
111 /// Emit an add or sub of \p LHSReg and \p RHSReg, leaving the EFLAGS def
112 /// live for a following SETcc.
113 Register X86FastEmitAddSub_rr(unsigned BaseOpc, MVT VT, Register LHSReg,
114 Register RHSReg);
115
116 /// Emit an add or sub of \p LHSReg and \p CI, leaving the EFLAGS def live
117 /// for a following SETcc reading \p CondCode. Returns an invalid register
118 /// if the immediate form does not apply.
119 Register X86FastEmitAddSub_ri(unsigned BaseOpc, MVT VT, Register LHSReg,
120 const ConstantInt *CI, unsigned CondCode);
121
122 bool X86SelectAddress(const Value *V, X86AddressMode &AM);
123 bool X86SelectCallAddress(const Value *V, X86AddressMode &AM);
124
125 bool X86SelectLoad(const Instruction *I);
126
127 bool X86SelectStore(const Instruction *I);
128
129 bool X86SelectRet(const Instruction *I);
130
131 bool X86SelectCmp(const Instruction *I);
132
133 bool X86SelectZExt(const Instruction *I);
134
135 bool X86SelectSExt(const Instruction *I);
136
137 bool X86SelectBranch(const Instruction *I);
138
139 bool X86SelectShift(const Instruction *I);
140
141 bool X86SelectMul(const Instruction *I);
142
143 bool X86SelectDivRem(const Instruction *I);
144
145 bool X86FastEmitCMoveSelect(MVT RetVT, const Instruction *I);
146
147 bool X86FastEmitSSESelect(MVT RetVT, const Instruction *I);
148
149 bool X86FastEmitPseudoSelect(MVT RetVT, const Instruction *I);
150
151 bool X86SelectSelect(const Instruction *I);
152
153 bool X86SelectTrunc(const Instruction *I);
154
155 bool X86SelectFPExtOrFPTrunc(const Instruction *I, unsigned Opc,
156 const TargetRegisterClass *RC);
157
158 bool X86SelectFPExt(const Instruction *I);
159 bool X86SelectFPTrunc(const Instruction *I);
160 bool X86SelectSIToFP(const Instruction *I);
161 bool X86SelectUIToFP(const Instruction *I);
162 bool X86SelectIntToFP(const Instruction *I, bool IsSigned);
163 bool X86SelectBitCast(const Instruction *I);
164
165 const X86InstrInfo *getInstrInfo() const {
166 return Subtarget->getInstrInfo();
167 }
168 const X86TargetMachine *getTargetMachine() const {
169 return static_cast<const X86TargetMachine *>(&TM);
170 }
171
172 bool handleConstantAddresses(const Value *V, X86AddressMode &AM);
173
174 Register emitMOV32r0();
175
176 Register X86MaterializeInt(const ConstantInt *CI, MVT VT);
177 Register X86MaterializeFP(const ConstantFP *CFP, MVT VT);
178 Register X86MaterializeGV(const GlobalValue *GV, MVT VT);
179 Register fastMaterializeConstant(const Constant *C) override;
180
181 Register fastMaterializeAlloca(const AllocaInst *C) override;
182
183 Register fastMaterializeFloatZero(const ConstantFP *CF) override;
184
185 /// isScalarFPTypeInSSEReg - Return true if the specified scalar FP type is
186 /// computed in an SSE register, not on the X87 floating point stack.
187 bool isScalarFPTypeInSSEReg(EVT VT) const {
188 return (VT == MVT::f64 && Subtarget->hasSSE2()) ||
189 (VT == MVT::f32 && Subtarget->hasSSE1()) || VT == MVT::f16;
190 }
191
192 bool isTypeLegal(Type *Ty, MVT &VT, bool AllowI1 = false);
193
194 bool IsMemcpySmall(uint64_t Len);
195
196 bool TryEmitSmallMemcpy(X86AddressMode DestAM,
197 X86AddressMode SrcAM, uint64_t Len);
198
199 bool foldX86XALUIntrinsic(X86::CondCode &CC, const Instruction *I,
200 const Value *Cond);
201
202 const MachineInstrBuilder &addFullAddress(const MachineInstrBuilder &MIB,
203 X86AddressMode &AM);
204
205 Register fastEmitInst_rrrr(unsigned MachineInstOpcode,
206 const TargetRegisterClass *RC, Register Op0,
207 Register Op1, Register Op2, Register Op3);
208};
209
210} // end anonymous namespace.
211
212static std::pair<unsigned, bool>
214 unsigned CC;
215 bool NeedSwap = false;
216
217 // SSE Condition code mapping:
218 // 0 - EQ
219 // 1 - LT
220 // 2 - LE
221 // 3 - UNORD
222 // 4 - NEQ
223 // 5 - NLT
224 // 6 - NLE
225 // 7 - ORD
226 switch (Predicate) {
227 default: llvm_unreachable("Unexpected predicate");
228 case CmpInst::FCMP_OEQ: CC = 0; break;
229 case CmpInst::FCMP_OGT: NeedSwap = true; [[fallthrough]];
230 case CmpInst::FCMP_OLT: CC = 1; break;
231 case CmpInst::FCMP_OGE: NeedSwap = true; [[fallthrough]];
232 case CmpInst::FCMP_OLE: CC = 2; break;
233 case CmpInst::FCMP_UNO: CC = 3; break;
234 case CmpInst::FCMP_UNE: CC = 4; break;
235 case CmpInst::FCMP_ULE: NeedSwap = true; [[fallthrough]];
236 case CmpInst::FCMP_UGE: CC = 5; break;
237 case CmpInst::FCMP_ULT: NeedSwap = true; [[fallthrough]];
238 case CmpInst::FCMP_UGT: CC = 6; break;
239 case CmpInst::FCMP_ORD: CC = 7; break;
240 case CmpInst::FCMP_UEQ: CC = 8; break;
241 case CmpInst::FCMP_ONE: CC = 12; break;
242 }
243
244 return std::make_pair(CC, NeedSwap);
245}
246
247/// Adds a complex addressing mode to the given machine instr builder.
248/// Note, this will constrain the index register. If its not possible to
249/// constrain the given index register, then a new one will be created. The
250/// IndexReg field of the addressing mode will be updated to match in this case.
252X86FastISel::addFullAddress(const MachineInstrBuilder &MIB,
253 X86AddressMode &AM) {
254 // First constrain the index register. It needs to be a GR64_NOSP.
256 MIB->getNumOperands() +
258 return ::addFullAddress(MIB, AM);
259}
260
261/// Check if it is possible to fold the condition from the XALU intrinsic
262/// into the user. The condition code will only be updated on success.
263bool X86FastISel::foldX86XALUIntrinsic(X86::CondCode &CC, const Instruction *I,
264 const Value *Cond) {
266 return false;
267
268 const auto *EV = cast<ExtractValueInst>(Cond);
269 if (!isa<IntrinsicInst>(EV->getAggregateOperand()))
270 return false;
271
272 const auto *II = cast<IntrinsicInst>(EV->getAggregateOperand());
273 MVT RetVT;
274 const Function *Callee = II->getCalledFunction();
275 Type *RetTy =
276 cast<StructType>(Callee->getReturnType())->getTypeAtIndex(0U);
277 if (!isTypeLegal(RetTy, RetVT))
278 return false;
279
280 if (RetVT != MVT::i32 && RetVT != MVT::i64)
281 return false;
282
283 X86::CondCode TmpCC;
284 switch (II->getIntrinsicID()) {
285 default: return false;
286 case Intrinsic::sadd_with_overflow:
287 case Intrinsic::ssub_with_overflow: TmpCC = X86::COND_O; break;
288 case Intrinsic::smul_with_overflow:
289 case Intrinsic::umul_with_overflow:
290 case Intrinsic::uadd_with_overflow:
291 case Intrinsic::usub_with_overflow: TmpCC = X86::COND_B; break;
292 }
293
294 // Check if both instructions are in the same basic block.
295 if (II->getParent() != I->getParent())
296 return false;
297
298 // Make sure nothing is in the way
301 for (auto Itr = std::prev(Start); Itr != End; --Itr) {
302 // We only expect extractvalue instructions between the intrinsic and the
303 // instruction to be selected.
304 if (!isa<ExtractValueInst>(Itr))
305 return false;
306
307 // Check that the extractvalue operand comes from the intrinsic.
308 const auto *EVI = cast<ExtractValueInst>(Itr);
309 if (EVI->getAggregateOperand() != II)
310 return false;
311 }
312
313 // Make sure no potentially eflags clobbering phi moves can be inserted in
314 // between.
315 auto HasPhis = [](const BasicBlock *Succ) { return !Succ->phis().empty(); };
316 if (I->isTerminator() && llvm::any_of(successors(I), HasPhis))
317 return false;
318
319 // Make sure there are no potentially eflags clobbering constant
320 // materializations in between.
321 if (llvm::any_of(I->operands(), [](Value *V) { return isa<Constant>(V); }))
322 return false;
323
324 CC = TmpCC;
325 return true;
326}
327
328bool X86FastISel::isTypeLegal(Type *Ty, MVT &VT, bool AllowI1) {
329 EVT evt = TLI.getValueType(DL, Ty, /*AllowUnknown=*/true);
330 if (evt == MVT::Other || !evt.isSimple())
331 // Unhandled type. Halt "fast" selection and bail.
332 return false;
333
334 VT = evt.getSimpleVT();
335 // For now, require SSE/SSE2 for performing floating-point operations,
336 // since x87 requires additional work.
337 if (VT == MVT::f64 && !Subtarget->hasSSE2())
338 return false;
339 if (VT == MVT::f32 && !Subtarget->hasSSE1())
340 return false;
341 // Similarly, no f80 support yet.
342 if (VT == MVT::f80)
343 return false;
344 // We only handle legal types. For example, on x86-32 the instruction
345 // selector contains all of the 64-bit instructions from x86-64,
346 // under the assumption that i64 won't be used if the target doesn't
347 // support it.
348 return (AllowI1 && VT == MVT::i1) || TLI.isTypeLegal(VT);
349}
350
351/// X86FastEmitLoad - Emit a machine instruction to load a value of type VT.
352/// The address is either pre-computed, i.e. Ptr, or a GlobalAddress, i.e. GV.
353/// Return true and the result register by reference if it is possible.
354bool X86FastISel::X86FastEmitLoad(MVT VT, X86AddressMode &AM,
355 MachineMemOperand *MMO, Register &ResultReg,
356 unsigned Alignment) {
357 bool HasSSE1 = Subtarget->hasSSE1();
358 bool HasSSE2 = Subtarget->hasSSE2();
359 bool HasSSE41 = Subtarget->hasSSE41();
360 bool HasAVX = Subtarget->hasAVX();
361 bool HasAVX2 = Subtarget->hasAVX2();
362 bool HasAVX512 = Subtarget->hasAVX512();
363 bool HasVLX = Subtarget->hasVLX();
364 bool IsNonTemporal = MMO && MMO->isNonTemporal();
365
366 // Treat i1 loads the same as i8 loads. Masking will be done when storing.
367 if (VT == MVT::i1)
368 VT = MVT::i8;
369
370 // Get opcode and regclass of the output for the given load instruction.
371 unsigned Opc = 0;
372 switch (VT.SimpleTy) {
373 default: return false;
374 case MVT::i8:
375 Opc = X86::MOV8rm;
376 break;
377 case MVT::i16:
378 Opc = X86::MOV16rm;
379 break;
380 case MVT::i32:
381 Opc = X86::MOV32rm;
382 break;
383 case MVT::i64:
384 // Must be in x86-64 mode.
385 Opc = X86::MOV64rm;
386 break;
387 case MVT::f32:
388 Opc = HasAVX512 ? X86::VMOVSSZrm_alt
389 : HasAVX ? X86::VMOVSSrm_alt
390 : HasSSE1 ? X86::MOVSSrm_alt
391 : X86::LD_Fp32m;
392 break;
393 case MVT::f64:
394 Opc = HasAVX512 ? X86::VMOVSDZrm_alt
395 : HasAVX ? X86::VMOVSDrm_alt
396 : HasSSE2 ? X86::MOVSDrm_alt
397 : X86::LD_Fp64m;
398 break;
399 case MVT::f80:
400 // No f80 support yet.
401 return false;
402 case MVT::v4f32:
403 if (IsNonTemporal && Alignment >= 16 && HasSSE41)
404 Opc = HasVLX ? X86::VMOVNTDQAZ128rm :
405 HasAVX ? X86::VMOVNTDQArm : X86::MOVNTDQArm;
406 else if (Alignment >= 16)
407 Opc = HasVLX ? X86::VMOVAPSZ128rm :
408 HasAVX ? X86::VMOVAPSrm : X86::MOVAPSrm;
409 else
410 Opc = HasVLX ? X86::VMOVUPSZ128rm :
411 HasAVX ? X86::VMOVUPSrm : X86::MOVUPSrm;
412 break;
413 case MVT::v2f64:
414 if (IsNonTemporal && Alignment >= 16 && HasSSE41)
415 Opc = HasVLX ? X86::VMOVNTDQAZ128rm :
416 HasAVX ? X86::VMOVNTDQArm : X86::MOVNTDQArm;
417 else if (Alignment >= 16)
418 Opc = HasVLX ? X86::VMOVAPDZ128rm :
419 HasAVX ? X86::VMOVAPDrm : X86::MOVAPDrm;
420 else
421 Opc = HasVLX ? X86::VMOVUPDZ128rm :
422 HasAVX ? X86::VMOVUPDrm : X86::MOVUPDrm;
423 break;
424 case MVT::v4i32:
425 case MVT::v2i64:
426 case MVT::v8i16:
427 case MVT::v16i8:
428 if (IsNonTemporal && Alignment >= 16 && HasSSE41)
429 Opc = HasVLX ? X86::VMOVNTDQAZ128rm :
430 HasAVX ? X86::VMOVNTDQArm : X86::MOVNTDQArm;
431 else if (Alignment >= 16)
432 Opc = HasVLX ? X86::VMOVDQA64Z128rm :
433 HasAVX ? X86::VMOVDQArm : X86::MOVDQArm;
434 else
435 Opc = HasVLX ? X86::VMOVDQU64Z128rm :
436 HasAVX ? X86::VMOVDQUrm : X86::MOVDQUrm;
437 break;
438 case MVT::v8f32:
439 assert(HasAVX);
440 if (IsNonTemporal && Alignment >= 32 && HasAVX2)
441 Opc = HasVLX ? X86::VMOVNTDQAZ256rm : X86::VMOVNTDQAYrm;
442 else if (IsNonTemporal && Alignment >= 16)
443 return false; // Force split for X86::VMOVNTDQArm
444 else if (Alignment >= 32)
445 Opc = HasVLX ? X86::VMOVAPSZ256rm : X86::VMOVAPSYrm;
446 else
447 Opc = HasVLX ? X86::VMOVUPSZ256rm : X86::VMOVUPSYrm;
448 break;
449 case MVT::v4f64:
450 assert(HasAVX);
451 if (IsNonTemporal && Alignment >= 32 && HasAVX2)
452 Opc = HasVLX ? X86::VMOVNTDQAZ256rm : X86::VMOVNTDQAYrm;
453 else if (IsNonTemporal && Alignment >= 16)
454 return false; // Force split for X86::VMOVNTDQArm
455 else if (Alignment >= 32)
456 Opc = HasVLX ? X86::VMOVAPDZ256rm : X86::VMOVAPDYrm;
457 else
458 Opc = HasVLX ? X86::VMOVUPDZ256rm : X86::VMOVUPDYrm;
459 break;
460 case MVT::v8i32:
461 case MVT::v4i64:
462 case MVT::v16i16:
463 case MVT::v32i8:
464 assert(HasAVX);
465 if (IsNonTemporal && Alignment >= 32 && HasAVX2)
466 Opc = HasVLX ? X86::VMOVNTDQAZ256rm : X86::VMOVNTDQAYrm;
467 else if (IsNonTemporal && Alignment >= 16)
468 return false; // Force split for X86::VMOVNTDQArm
469 else if (Alignment >= 32)
470 Opc = HasVLX ? X86::VMOVDQA64Z256rm : X86::VMOVDQAYrm;
471 else
472 Opc = HasVLX ? X86::VMOVDQU64Z256rm : X86::VMOVDQUYrm;
473 break;
474 case MVT::v16f32:
475 assert(HasAVX512);
476 if (IsNonTemporal && Alignment >= 64)
477 Opc = X86::VMOVNTDQAZrm;
478 else
479 Opc = (Alignment >= 64) ? X86::VMOVAPSZrm : X86::VMOVUPSZrm;
480 break;
481 case MVT::v8f64:
482 assert(HasAVX512);
483 if (IsNonTemporal && Alignment >= 64)
484 Opc = X86::VMOVNTDQAZrm;
485 else
486 Opc = (Alignment >= 64) ? X86::VMOVAPDZrm : X86::VMOVUPDZrm;
487 break;
488 case MVT::v8i64:
489 case MVT::v16i32:
490 case MVT::v32i16:
491 case MVT::v64i8:
492 assert(HasAVX512);
493 // Note: There are a lot more choices based on type with AVX-512, but
494 // there's really no advantage when the load isn't masked.
495 if (IsNonTemporal && Alignment >= 64)
496 Opc = X86::VMOVNTDQAZrm;
497 else
498 Opc = (Alignment >= 64) ? X86::VMOVDQA64Zrm : X86::VMOVDQU64Zrm;
499 break;
500 }
501
502 const TargetRegisterClass *RC = TLI.getRegClassFor(VT);
503
504 ResultReg = createResultReg(RC);
505 MachineInstrBuilder MIB =
506 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Opc), ResultReg);
507 addFullAddress(MIB, AM);
508 if (MMO)
509 MIB->addMemOperand(*FuncInfo.MF, MMO);
510 return true;
511}
512
513/// X86FastEmitStore - Emit a machine instruction to store a value Val of
514/// type VT. The address is either pre-computed, consisted of a base ptr, Ptr
515/// and a displacement offset, or a GlobalAddress,
516/// i.e. V. Return true if it is possible.
517bool X86FastISel::X86FastEmitStore(EVT VT, Register ValReg, X86AddressMode &AM,
518 MachineMemOperand *MMO, bool Aligned) {
519 bool HasSSE1 = Subtarget->hasSSE1();
520 bool HasSSE2 = Subtarget->hasSSE2();
521 bool HasSSE4A = Subtarget->hasSSE4A();
522 bool HasAVX = Subtarget->hasAVX();
523 bool HasAVX512 = Subtarget->hasAVX512();
524 bool HasVLX = Subtarget->hasVLX();
525 bool IsNonTemporal = MMO && MMO->isNonTemporal();
526
527 // Get opcode and regclass of the output for the given store instruction.
528 unsigned Opc = 0;
529 switch (VT.getSimpleVT().SimpleTy) {
530 case MVT::f80: // No f80 support yet.
531 default: return false;
532 case MVT::i1: {
533 // Mask out all but lowest bit.
534 Register AndResult = createResultReg(&X86::GR8RegClass);
535 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::AND8ri),
536 AndResult)
537 .addReg(ValReg)
538 .addImm(1)
539 .setOperandDead(3); // implicit-def $eflags
540 ValReg = AndResult;
541 [[fallthrough]]; // handle i1 as i8.
542 }
543 case MVT::i8: Opc = X86::MOV8mr; break;
544 case MVT::i16: Opc = X86::MOV16mr; break;
545 case MVT::i32:
546 Opc = (IsNonTemporal && HasSSE2) ? X86::MOVNTImr : X86::MOV32mr;
547 break;
548 case MVT::i64:
549 // Must be in x86-64 mode.
550 Opc = (IsNonTemporal && HasSSE2) ? X86::MOVNTI_64mr : X86::MOV64mr;
551 break;
552 case MVT::f32:
553 if (HasSSE1) {
554 if (IsNonTemporal && HasSSE4A)
555 Opc = X86::MOVNTSS;
556 else
557 Opc = HasAVX512 ? X86::VMOVSSZmr :
558 HasAVX ? X86::VMOVSSmr : X86::MOVSSmr;
559 } else
560 Opc = X86::ST_Fp32m;
561 break;
562 case MVT::f64:
563 if (HasSSE2) {
564 if (IsNonTemporal && HasSSE4A)
565 Opc = X86::MOVNTSD;
566 else
567 Opc = HasAVX512 ? X86::VMOVSDZmr :
568 HasAVX ? X86::VMOVSDmr : X86::MOVSDmr;
569 } else
570 Opc = X86::ST_Fp64m;
571 break;
572 case MVT::x86mmx:
573 Opc = (IsNonTemporal && HasSSE1) ? X86::MMX_MOVNTQmr : X86::MMX_MOVQ64mr;
574 break;
575 case MVT::v4f32:
576 if (Aligned) {
577 if (IsNonTemporal)
578 Opc = HasVLX ? X86::VMOVNTPSZ128mr :
579 HasAVX ? X86::VMOVNTPSmr : X86::MOVNTPSmr;
580 else
581 Opc = HasVLX ? X86::VMOVAPSZ128mr :
582 HasAVX ? X86::VMOVAPSmr : X86::MOVAPSmr;
583 } else
584 Opc = HasVLX ? X86::VMOVUPSZ128mr :
585 HasAVX ? X86::VMOVUPSmr : X86::MOVUPSmr;
586 break;
587 case MVT::v2f64:
588 if (Aligned) {
589 if (IsNonTemporal)
590 Opc = HasVLX ? X86::VMOVNTPDZ128mr :
591 HasAVX ? X86::VMOVNTPDmr : X86::MOVNTPDmr;
592 else
593 Opc = HasVLX ? X86::VMOVAPDZ128mr :
594 HasAVX ? X86::VMOVAPDmr : X86::MOVAPDmr;
595 } else
596 Opc = HasVLX ? X86::VMOVUPDZ128mr :
597 HasAVX ? X86::VMOVUPDmr : X86::MOVUPDmr;
598 break;
599 case MVT::v4i32:
600 case MVT::v2i64:
601 case MVT::v8i16:
602 case MVT::v16i8:
603 if (Aligned) {
604 if (IsNonTemporal)
605 Opc = HasVLX ? X86::VMOVNTDQZ128mr :
606 HasAVX ? X86::VMOVNTDQmr : X86::MOVNTDQmr;
607 else
608 Opc = HasVLX ? X86::VMOVDQA64Z128mr :
609 HasAVX ? X86::VMOVDQAmr : X86::MOVDQAmr;
610 } else
611 Opc = HasVLX ? X86::VMOVDQU64Z128mr :
612 HasAVX ? X86::VMOVDQUmr : X86::MOVDQUmr;
613 break;
614 case MVT::v8f32:
615 assert(HasAVX);
616 if (Aligned) {
617 if (IsNonTemporal)
618 Opc = HasVLX ? X86::VMOVNTPSZ256mr : X86::VMOVNTPSYmr;
619 else
620 Opc = HasVLX ? X86::VMOVAPSZ256mr : X86::VMOVAPSYmr;
621 } else
622 Opc = HasVLX ? X86::VMOVUPSZ256mr : X86::VMOVUPSYmr;
623 break;
624 case MVT::v4f64:
625 assert(HasAVX);
626 if (Aligned) {
627 if (IsNonTemporal)
628 Opc = HasVLX ? X86::VMOVNTPDZ256mr : X86::VMOVNTPDYmr;
629 else
630 Opc = HasVLX ? X86::VMOVAPDZ256mr : X86::VMOVAPDYmr;
631 } else
632 Opc = HasVLX ? X86::VMOVUPDZ256mr : X86::VMOVUPDYmr;
633 break;
634 case MVT::v8i32:
635 case MVT::v4i64:
636 case MVT::v16i16:
637 case MVT::v32i8:
638 assert(HasAVX);
639 if (Aligned) {
640 if (IsNonTemporal)
641 Opc = HasVLX ? X86::VMOVNTDQZ256mr : X86::VMOVNTDQYmr;
642 else
643 Opc = HasVLX ? X86::VMOVDQA64Z256mr : X86::VMOVDQAYmr;
644 } else
645 Opc = HasVLX ? X86::VMOVDQU64Z256mr : X86::VMOVDQUYmr;
646 break;
647 case MVT::v16f32:
648 assert(HasAVX512);
649 if (Aligned)
650 Opc = IsNonTemporal ? X86::VMOVNTPSZmr : X86::VMOVAPSZmr;
651 else
652 Opc = X86::VMOVUPSZmr;
653 break;
654 case MVT::v8f64:
655 assert(HasAVX512);
656 if (Aligned) {
657 Opc = IsNonTemporal ? X86::VMOVNTPDZmr : X86::VMOVAPDZmr;
658 } else
659 Opc = X86::VMOVUPDZmr;
660 break;
661 case MVT::v8i64:
662 case MVT::v16i32:
663 case MVT::v32i16:
664 case MVT::v64i8:
665 assert(HasAVX512);
666 // Note: There are a lot more choices based on type with AVX-512, but
667 // there's really no advantage when the store isn't masked.
668 if (Aligned)
669 Opc = IsNonTemporal ? X86::VMOVNTDQZmr : X86::VMOVDQA64Zmr;
670 else
671 Opc = X86::VMOVDQU64Zmr;
672 break;
673 }
674
675 const MCInstrDesc &Desc = TII.get(Opc);
676 // Some of the instructions in the previous switch use FR128 instead
677 // of FR32 for ValReg. Make sure the register we feed the instruction
678 // matches its register class constraints.
679 // Note: This is fine to do a copy from FR32 to FR128, this is the
680 // same registers behind the scene and actually why it did not trigger
681 // any bugs before.
682 ValReg = constrainOperandRegClass(Desc, ValReg, Desc.getNumOperands() - 1);
683 MachineInstrBuilder MIB =
684 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, Desc);
685 addFullAddress(MIB, AM).addReg(ValReg);
686 if (MMO)
687 MIB->addMemOperand(*FuncInfo.MF, MMO);
688
689 return true;
690}
691
692bool X86FastISel::X86FastEmitStore(EVT VT, const Value *Val,
693 X86AddressMode &AM,
694 MachineMemOperand *MMO, bool Aligned) {
695 // Handle 'null' like i32/i64 0.
697 Val = Constant::getNullValue(DL.getIntPtrType(Val->getContext()));
698
699 // If this is a store of a simple constant, fold the constant into the store.
700 if (const ConstantInt *CI = dyn_cast<ConstantInt>(Val)) {
701 unsigned Opc = 0;
702 bool Signed = true;
703 switch (VT.getSimpleVT().SimpleTy) {
704 default: break;
705 case MVT::i1:
706 Signed = false;
707 [[fallthrough]]; // Handle as i8.
708 case MVT::i8: Opc = X86::MOV8mi; break;
709 case MVT::i16: Opc = X86::MOV16mi; break;
710 case MVT::i32: Opc = X86::MOV32mi; break;
711 case MVT::i64:
712 // Must be a 32-bit sign extended value.
713 if (isInt<32>(CI->getSExtValue()))
714 Opc = X86::MOV64mi32;
715 break;
716 }
717
718 if (Opc) {
719 MachineInstrBuilder MIB =
720 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Opc));
721 addFullAddress(MIB, AM).addImm(Signed ? (uint64_t) CI->getSExtValue()
722 : CI->getZExtValue());
723 if (MMO)
724 MIB->addMemOperand(*FuncInfo.MF, MMO);
725 return true;
726 }
727 }
728
729 Register ValReg = getRegForValue(Val);
730 if (!ValReg)
731 return false;
732
733 return X86FastEmitStore(VT, ValReg, AM, MMO, Aligned);
734}
735
736/// X86FastEmitExtend - Emit a machine instruction to extend a value Src of
737/// type SrcVT to type DstVT using the specified extension opcode Opc (e.g.
738/// ISD::SIGN_EXTEND).
739bool X86FastISel::X86FastEmitExtend(ISD::NodeType Opc, EVT DstVT, Register Src,
740 EVT SrcVT, Register &ResultReg) {
741 Register RR = fastEmit_r(SrcVT.getSimpleVT(), DstVT.getSimpleVT(), Opc, Src);
742 if (!RR)
743 return false;
744
745 ResultReg = RR;
746 return true;
747}
748
749Register X86FastISel::X86FastEmitLiveEFLAGS_rr(unsigned Opc,
750 const TargetRegisterClass *RC,
751 Register Op0, Register Op1) {
752 const MCInstrDesc &II = TII.get(Opc);
753 assert(II.getNumDefs() >= 1 && "instruction must define the result");
754 assert(II.implicit_defs().size() == 1 &&
755 II.implicit_defs()[0] == X86::EFLAGS && "unexpected implicit def");
756
757 Register ResultReg = createResultReg(RC);
758 Op0 = constrainOperandRegClass(II, Op0, II.getNumDefs());
759 Op1 = constrainOperandRegClass(II, Op1, II.getNumDefs() + 1);
760
761 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, II, ResultReg)
762 .addReg(Op0)
763 .addReg(Op1);
764 return ResultReg;
765}
766
767Register X86FastISel::X86FastEmitLiveEFLAGS_ri(unsigned Opc,
768 const TargetRegisterClass *RC,
769 Register Op0, uint64_t Imm) {
770 const MCInstrDesc &II = TII.get(Opc);
771 assert(II.getNumDefs() >= 1 && "instruction must define the result");
772 assert(II.implicit_defs().size() == 1 &&
773 II.implicit_defs()[0] == X86::EFLAGS && "unexpected implicit def");
774
775 Register ResultReg = createResultReg(RC);
776 Op0 = constrainOperandRegClass(II, Op0, II.getNumDefs());
777
778 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, II, ResultReg)
779 .addReg(Op0)
780 .addImm(Imm);
781 return ResultReg;
782}
783
784Register X86FastISel::X86FastEmitMul(unsigned Opc, MVT VT, MCRegister AccReg,
785 Register LHSReg, Register RHSReg,
786 bool LiveEFLAGS) {
787 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(TargetOpcode::COPY),
788 AccReg)
789 .addReg(LHSReg);
790
791 const MCInstrDesc &II = TII.get(Opc);
792 Register ResultReg = createResultReg(TLI.getRegClassFor(VT));
793 RHSReg = constrainOperandRegClass(II, RHSReg, II.getNumDefs());
794
795 MachineInstrBuilder MIB =
796 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, II).addReg(RHSReg);
797
798 // Operand 0 is the explicit source, followed by the implicit defs. MUL8r and
799 // IMUL8r define AL, EFLAGS and AX, where AX overlaps the AL result. The wider
800 // forms define the accumulator, the separate high half register and EFLAGS.
801 assert(MIB->getOperand(1).getReg() == AccReg && "unexpected operand order");
802 if (VT == MVT::i8) {
803 assert(MIB->getOperand(2).getReg() == X86::EFLAGS &&
804 "unexpected operand order");
805 if (!LiveEFLAGS)
806 MIB.setOperandDead(2);
807 } else {
808 assert(MIB->getOperand(3).getReg() == X86::EFLAGS &&
809 "unexpected operand order");
810 MIB.setOperandDead(2);
811 if (!LiveEFLAGS)
812 MIB.setOperandDead(3);
813 }
814 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(TargetOpcode::COPY),
815 ResultReg)
816 .addReg(AccReg);
817 return ResultReg;
818}
819
820bool X86FastISel::handleConstantAddresses(const Value *V, X86AddressMode &AM) {
821 // Handle constant address.
822 if (const GlobalValue *GV = dyn_cast<GlobalValue>(V)) {
823 // Can't handle alternate code models yet.
824 if (TM.getCodeModel() != CodeModel::Small &&
825 TM.getCodeModel() != CodeModel::Medium)
826 return false;
827
828 // Can't handle large objects yet.
829 if (TM.isLargeGlobalValue(GV))
830 return false;
831
832 // Can't handle TLS yet.
833 if (GV->isThreadLocal())
834 return false;
835
836 // Can't handle !absolute_symbol references yet.
837 if (GV->isAbsoluteSymbolRef())
838 return false;
839
840 // RIP-relative addresses can't have additional register operands, so if
841 // we've already folded stuff into the addressing mode, just force the
842 // global value into its own register, which we can use as the basereg.
843 if (!Subtarget->isPICStyleRIPRel() ||
844 (AM.Base.Reg == 0 && AM.IndexReg == 0)) {
845 // Okay, we've committed to selecting this global. Set up the address.
846 AM.GV = GV;
847
848 // Allow the subtarget to classify the global.
849 unsigned char GVFlags = Subtarget->classifyGlobalReference(GV);
850
851 // If this reference is relative to the pic base, set it now.
852 if (isGlobalRelativeToPICBase(GVFlags)) {
853 // FIXME: How do we know Base.Reg is free??
854 AM.Base.Reg = getInstrInfo()->getGlobalBaseReg(FuncInfo.MF);
855 }
856
857 // Unless the ABI requires an extra load, return a direct reference to
858 // the global.
859 if (!isGlobalStubReference(GVFlags)) {
860 if (Subtarget->isPICStyleRIPRel()) {
861 // Use rip-relative addressing if we can. Above we verified that the
862 // base and index registers are unused.
863 assert(AM.Base.Reg == 0 && AM.IndexReg == 0);
864 AM.Base.Reg = X86::RIP;
865 }
866 AM.GVOpFlags = GVFlags;
867 return true;
868 }
869
870 // Ok, we need to do a load from a stub. If we've already loaded from
871 // this stub, reuse the loaded pointer, otherwise emit the load now.
872 auto I = LocalValueMap.find(V);
873 Register LoadReg;
874 if (I != LocalValueMap.end() && I->second) {
875 LoadReg = I->second;
876 } else {
877 // Issue load from stub.
878 unsigned Opc = 0;
879 const TargetRegisterClass *RC = nullptr;
880 X86AddressMode StubAM;
881 StubAM.Base.Reg = AM.Base.Reg;
882 StubAM.GV = GV;
883 StubAM.GVOpFlags = GVFlags;
884
885 // Prepare for inserting code in the local-value area.
886 SavePoint SaveInsertPt = enterLocalValueArea();
887
888 if (TLI.getPointerTy(DL) == MVT::i64) {
889 Opc = X86::MOV64rm;
890 RC = &X86::GR64RegClass;
891 } else {
892 Opc = X86::MOV32rm;
893 RC = &X86::GR32RegClass;
894 }
895
896 if (Subtarget->isPICStyleRIPRel() || GVFlags == X86II::MO_GOTPCREL ||
898 StubAM.Base.Reg = X86::RIP;
899
900 LoadReg = createResultReg(RC);
901 MachineInstrBuilder LoadMI =
902 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Opc), LoadReg);
903 addFullAddress(LoadMI, StubAM);
904
905 // Ok, back to normal mode.
906 leaveLocalValueArea(SaveInsertPt);
907
908 // Prevent loading GV stub multiple times in same MBB.
909 LocalValueMap[V] = LoadReg;
910 }
911
912 // Now construct the final address. Note that the Disp, Scale,
913 // and Index values may already be set here.
914 AM.Base.Reg = LoadReg;
915 AM.GV = nullptr;
916 return true;
917 }
918 }
919
920 // If all else fails, try to materialize the value in a register.
921 if (!AM.GV || !Subtarget->isPICStyleRIPRel()) {
922 if (AM.Base.Reg == 0) {
923 AM.Base.Reg = getRegForValue(V);
924 return AM.Base.Reg != 0;
925 }
926 if (AM.IndexReg == 0) {
927 assert(AM.Scale == 1 && "Scale with no index!");
928 AM.IndexReg = getRegForValue(V);
929 return AM.IndexReg != 0;
930 }
931 }
932
933 return false;
934}
935
936/// X86SelectAddress - Attempt to fill in an address from the given value.
937///
938bool X86FastISel::X86SelectAddress(const Value *V, X86AddressMode &AM) {
940redo_gep:
941 const User *U = nullptr;
942 unsigned Opcode = Instruction::UserOp1;
943 if (const Instruction *I = dyn_cast<Instruction>(V)) {
944 // Don't walk into other basic blocks; it's possible we haven't
945 // visited them yet, so the instructions may not yet be assigned
946 // virtual registers.
947 if (FuncInfo.StaticAllocaMap.count(static_cast<const AllocaInst *>(V)) ||
948 FuncInfo.getMBB(I->getParent()) == FuncInfo.MBB) {
949 Opcode = I->getOpcode();
950 U = I;
951 }
952 } else if (const ConstantExpr *C = dyn_cast<ConstantExpr>(V)) {
953 Opcode = C->getOpcode();
954 U = C;
955 }
956
957 if (PointerType *Ty = dyn_cast<PointerType>(V->getType()))
958 if (Ty->getAddressSpace() > 255)
959 // Fast instruction selection doesn't support the special
960 // address spaces.
961 return false;
962
963 switch (Opcode) {
964 default: break;
965 case Instruction::BitCast:
966 // Look past bitcasts.
967 return X86SelectAddress(U->getOperand(0), AM);
968
969 case Instruction::IntToPtr:
970 // Look past no-op inttoptrs.
971 if (TLI.getValueType(DL, U->getOperand(0)->getType()) ==
972 TLI.getPointerTy(DL))
973 return X86SelectAddress(U->getOperand(0), AM);
974 break;
975
976 case Instruction::PtrToInt:
977 // Look past no-op ptrtoints.
978 if (TLI.getValueType(DL, U->getType()) == TLI.getPointerTy(DL))
979 return X86SelectAddress(U->getOperand(0), AM);
980 break;
981
982 case Instruction::Alloca: {
983 // Do static allocas.
984 const AllocaInst *A = cast<AllocaInst>(V);
985 auto SI = FuncInfo.StaticAllocaMap.find(A);
986 if (SI != FuncInfo.StaticAllocaMap.end()) {
988 AM.Base.FrameIndex = SI->second;
989 return true;
990 }
991 break;
992 }
993
994 case Instruction::Add: {
995 // Adds of constants are common and easy enough.
996 if (const ConstantInt *CI = dyn_cast<ConstantInt>(U->getOperand(1))) {
997 uint64_t Disp = (int32_t)AM.Disp + (uint64_t)CI->getSExtValue();
998 // They have to fit in the 32-bit signed displacement field though.
999 if (isInt<32>(Disp)) {
1000 AM.Disp = (uint32_t)Disp;
1001 return X86SelectAddress(U->getOperand(0), AM);
1002 }
1003 }
1004 break;
1005 }
1006
1007 case Instruction::GetElementPtr: {
1008 X86AddressMode SavedAM = AM;
1009
1010 // Pattern-match simple GEPs.
1011 uint64_t Disp = (int32_t)AM.Disp;
1012 Register IndexReg = AM.IndexReg;
1013 unsigned Scale = AM.Scale;
1014 MVT PtrVT = TLI.getValueType(DL, U->getType()).getSimpleVT();
1015
1017 // Iterate through the indices, folding what we can. Constants can be
1018 // folded, and one dynamic index can be handled, if the scale is supported.
1019 for (User::const_op_iterator i = U->op_begin() + 1, e = U->op_end();
1020 i != e; ++i, ++GTI) {
1021 const Value *Op = *i;
1022 if (StructType *STy = GTI.getStructTypeOrNull()) {
1023 const StructLayout *SL = DL.getStructLayout(STy);
1024 Disp += SL->getElementOffset(cast<ConstantInt>(Op)->getZExtValue());
1025 continue;
1026 }
1027
1028 // A array/variable index is always of the form i*S where S is the
1029 // constant scale size. See if we can push the scale into immediates.
1031 for (;;) {
1032 if (const ConstantInt *CI = dyn_cast<ConstantInt>(Op)) {
1033 // Constant-offset addressing. The index may be wider than 64 bits;
1034 // it is truncated to the pointer width like any other GEP index.
1035 Disp += CI->getValue().sextOrTrunc(64).getSExtValue() * S;
1036 break;
1037 }
1038 if (canFoldAddIntoGEP(U, Op)) {
1039 // A compatible add with a constant operand. Fold the constant.
1040 ConstantInt *CI =
1041 cast<ConstantInt>(cast<AddOperator>(Op)->getOperand(1));
1042 Disp += CI->getSExtValue() * S;
1043 // Iterate on the other operand.
1044 Op = cast<AddOperator>(Op)->getOperand(0);
1045 continue;
1046 }
1047 if (!IndexReg && (!AM.GV || !Subtarget->isPICStyleRIPRel()) &&
1048 (S == 1 || S == 2 || S == 4 || S == 8)) {
1049 // Scaled-index addressing.
1050 Scale = S;
1051 IndexReg = getRegForGEPIndex(PtrVT, Op);
1052 if (!IndexReg)
1053 return false;
1054 break;
1055 }
1056 // Unsupported.
1057 goto unsupported_gep;
1058 }
1059 }
1060
1061 // Check for displacement overflow.
1062 if (!isInt<32>(Disp))
1063 break;
1064
1065 AM.IndexReg = IndexReg;
1066 AM.Scale = Scale;
1067 AM.Disp = (uint32_t)Disp;
1068 GEPs.push_back(V);
1069
1070 if (const GetElementPtrInst *GEP =
1071 dyn_cast<GetElementPtrInst>(U->getOperand(0))) {
1072 // Ok, the GEP indices were covered by constant-offset and scaled-index
1073 // addressing. Update the address state and move on to examining the base.
1074 V = GEP;
1075 goto redo_gep;
1076 } else if (X86SelectAddress(U->getOperand(0), AM)) {
1077 return true;
1078 }
1079
1080 // If we couldn't merge the gep value into this addr mode, revert back to
1081 // our address and just match the value instead of completely failing.
1082 AM = SavedAM;
1083
1084 for (const Value *I : reverse(GEPs))
1085 if (handleConstantAddresses(I, AM))
1086 return true;
1087
1088 return false;
1089 unsupported_gep:
1090 // Ok, the GEP indices weren't all covered.
1091 break;
1092 }
1093 }
1094
1095 return handleConstantAddresses(V, AM);
1096}
1097
1098/// X86SelectCallAddress - Attempt to fill in an address from the given value.
1099///
1100bool X86FastISel::X86SelectCallAddress(const Value *V, X86AddressMode &AM) {
1101 const User *U = nullptr;
1102 unsigned Opcode = Instruction::UserOp1;
1104 // Record if the value is defined in the same basic block.
1105 //
1106 // This information is crucial to know whether or not folding an
1107 // operand is valid.
1108 // Indeed, FastISel generates or reuses a virtual register for all
1109 // operands of all instructions it selects. Obviously, the definition and
1110 // its uses must use the same virtual register otherwise the produced
1111 // code is incorrect.
1112 // Before instruction selection, FunctionLoweringInfo::set sets the virtual
1113 // registers for values that are alive across basic blocks. This ensures
1114 // that the values are consistently set between across basic block, even
1115 // if different instruction selection mechanisms are used (e.g., a mix of
1116 // SDISel and FastISel).
1117 // For values local to a basic block, the instruction selection process
1118 // generates these virtual registers with whatever method is appropriate
1119 // for its needs. In particular, FastISel and SDISel do not share the way
1120 // local virtual registers are set.
1121 // Therefore, this is impossible (or at least unsafe) to share values
1122 // between basic blocks unless they use the same instruction selection
1123 // method, which is not guarantee for X86.
1124 // Moreover, things like hasOneUse could not be used accurately, if we
1125 // allow to reference values across basic blocks whereas they are not
1126 // alive across basic blocks initially.
1127 bool InMBB = true;
1128 if (I) {
1129 Opcode = I->getOpcode();
1130 U = I;
1131 InMBB = I->getParent() == FuncInfo.MBB->getBasicBlock();
1132 } else if (const ConstantExpr *C = dyn_cast<ConstantExpr>(V)) {
1133 Opcode = C->getOpcode();
1134 U = C;
1135 }
1136
1137 switch (Opcode) {
1138 default: break;
1139 case Instruction::BitCast:
1140 // Look past bitcasts if its operand is in the same BB.
1141 if (InMBB)
1142 return X86SelectCallAddress(U->getOperand(0), AM);
1143 break;
1144
1145 case Instruction::IntToPtr:
1146 // Look past no-op inttoptrs if its operand is in the same BB.
1147 if (InMBB &&
1148 TLI.getValueType(DL, U->getOperand(0)->getType()) ==
1149 TLI.getPointerTy(DL))
1150 return X86SelectCallAddress(U->getOperand(0), AM);
1151 break;
1152
1153 case Instruction::PtrToInt:
1154 // Look past no-op ptrtoints if its operand is in the same BB.
1155 if (InMBB && TLI.getValueType(DL, U->getType()) == TLI.getPointerTy(DL))
1156 return X86SelectCallAddress(U->getOperand(0), AM);
1157 break;
1158 }
1159
1160 // Handle constant address.
1161 if (const GlobalValue *GV = dyn_cast<GlobalValue>(V)) {
1162 // Can't handle alternate code models yet.
1163 if (TM.getCodeModel() != CodeModel::Small &&
1164 TM.getCodeModel() != CodeModel::Medium)
1165 return false;
1166
1167 // RIP-relative addresses can't have additional register operands.
1168 if (Subtarget->isPICStyleRIPRel() &&
1169 (AM.Base.Reg != 0 || AM.IndexReg != 0))
1170 return false;
1171
1172 // Can't handle TLS.
1173 if (const GlobalVariable *GVar = dyn_cast<GlobalVariable>(GV))
1174 if (GVar->isThreadLocal())
1175 return false;
1176
1177 // Okay, we've committed to selecting this global. Set up the basic address.
1178 AM.GV = GV;
1179
1180 // Return a direct reference to the global. Fastisel can handle calls to
1181 // functions that require loads, such as dllimport and nonlazybind
1182 // functions.
1183 if (Subtarget->isPICStyleRIPRel()) {
1184 // Use rip-relative addressing if we can. Above we verified that the
1185 // base and index registers are unused.
1186 assert(AM.Base.Reg == 0 && AM.IndexReg == 0);
1187 AM.Base.Reg = X86::RIP;
1188 } else {
1189 AM.GVOpFlags = Subtarget->classifyLocalReference(nullptr);
1190 }
1191
1192 return true;
1193 }
1194
1195 // If all else fails, try to materialize the value in a register.
1196 if (!AM.GV || !Subtarget->isPICStyleRIPRel()) {
1197 auto GetCallRegForValue = [this](const Value *V) {
1198 Register Reg = getRegForValue(V);
1199
1200 // In 64-bit mode, we need a 64-bit register even if pointers are 32 bits.
1201 if (Reg && Subtarget->isTarget64BitILP32()) {
1202 Register CopyReg = createResultReg(&X86::GR32RegClass);
1203 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::MOV32rr),
1204 CopyReg)
1205 .addReg(Reg);
1206
1207 Register ExtReg = createResultReg(&X86::GR64RegClass);
1208 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
1209 TII.get(TargetOpcode::SUBREG_TO_REG), ExtReg)
1210 .addReg(CopyReg)
1211 .addImm(X86::sub_32bit);
1212 Reg = ExtReg;
1213 }
1214
1215 return Reg;
1216 };
1217
1218 if (AM.Base.Reg == 0) {
1219 AM.Base.Reg = GetCallRegForValue(V);
1220 return AM.Base.Reg != 0;
1221 }
1222 if (AM.IndexReg == 0) {
1223 assert(AM.Scale == 1 && "Scale with no index!");
1224 AM.IndexReg = GetCallRegForValue(V);
1225 return AM.IndexReg != 0;
1226 }
1227 }
1228
1229 return false;
1230}
1231
1232
1233/// X86SelectStore - Select and emit code to implement store instructions.
1234bool X86FastISel::X86SelectStore(const Instruction *I) {
1235 // Atomic stores need special handling.
1236 const StoreInst *S = cast<StoreInst>(I);
1237
1238 if (S->isAtomic())
1239 return false;
1240
1241 const Value *PtrV = I->getOperand(1);
1242 if (TLI.supportSwiftError()) {
1243 // Swifterror values can come from either a function parameter with
1244 // swifterror attribute or an alloca with swifterror attribute.
1245 if (const Argument *Arg = dyn_cast<Argument>(PtrV)) {
1246 if (Arg->hasSwiftErrorAttr())
1247 return false;
1248 }
1249
1250 if (const AllocaInst *Alloca = dyn_cast<AllocaInst>(PtrV)) {
1251 if (Alloca->isSwiftError())
1252 return false;
1253 }
1254 }
1255
1256 const Value *Val = S->getValueOperand();
1257 const Value *Ptr = S->getPointerOperand();
1258
1259 MVT VT;
1260 if (!isTypeLegal(Val->getType(), VT, /*AllowI1=*/true))
1261 return false;
1262
1263 Align Alignment = S->getAlign();
1264 Align ABIAlignment = DL.getABITypeAlign(Val->getType());
1265 bool Aligned = Alignment >= ABIAlignment;
1266
1267 X86AddressMode AM;
1268 if (!X86SelectAddress(Ptr, AM))
1269 return false;
1270
1271 return X86FastEmitStore(VT, Val, AM, createMachineMemOperandFor(I), Aligned);
1272}
1273
1274/// X86SelectRet - Select and emit code to implement ret instructions.
1275bool X86FastISel::X86SelectRet(const Instruction *I) {
1276 const ReturnInst *Ret = cast<ReturnInst>(I);
1277 const Function &F = *I->getParent()->getParent();
1278 const X86MachineFunctionInfo *X86MFInfo =
1279 FuncInfo.MF->getInfo<X86MachineFunctionInfo>();
1280
1281 if (!FuncInfo.CanLowerReturn)
1282 return false;
1283
1284 if (TLI.supportSwiftError() &&
1285 F.getAttributes().hasAttrSomewhere(Attribute::SwiftError))
1286 return false;
1287
1288 if (TLI.supportSplitCSR(FuncInfo.MF))
1289 return false;
1290
1291 CallingConv::ID CC = F.getCallingConv();
1292 if (CC != CallingConv::C &&
1293 CC != CallingConv::Fast &&
1294 CC != CallingConv::Tail &&
1295 CC != CallingConv::SwiftTail &&
1296 CC != CallingConv::X86_FastCall &&
1297 CC != CallingConv::X86_StdCall &&
1298 CC != CallingConv::X86_ThisCall &&
1299 CC != CallingConv::X86_64_SysV &&
1300 CC != CallingConv::Win64)
1301 return false;
1302
1303 // Don't handle popping bytes if they don't fit the ret's immediate.
1304 if (!isUInt<16>(X86MFInfo->getBytesToPopOnReturn()))
1305 return false;
1306
1307 // fastcc with -tailcallopt is intended to provide a guaranteed
1308 // tail call optimization. Fastisel doesn't know how to do that.
1309 if ((CC == CallingConv::Fast && TM.Options.GuaranteedTailCallOpt) ||
1310 CC == CallingConv::Tail || CC == CallingConv::SwiftTail)
1311 return false;
1312
1313 // Let SDISel handle vararg functions.
1314 if (F.isVarArg())
1315 return false;
1316
1317 // Build a list of return value registers.
1319
1320 if (Ret->getNumOperands() > 0) {
1322 GetReturnInfo(CC, F.getReturnType(), F.getAttributes(), Outs, TLI, DL);
1323
1324 // Analyze operands of the call, assigning locations to each operand.
1326 CCState CCInfo(CC, F.isVarArg(), *FuncInfo.MF, ValLocs, I->getContext());
1327 CCInfo.AnalyzeReturn(Outs, RetCC_X86);
1328
1329 const Value *RV = Ret->getOperand(0);
1330 Register Reg = getRegForValue(RV);
1331 if (!Reg)
1332 return false;
1333
1334 // Only handle a single return value for now.
1335 if (ValLocs.size() != 1)
1336 return false;
1337
1338 CCValAssign &VA = ValLocs[0];
1339
1340 // Don't bother handling odd stuff for now.
1341 if (VA.getLocInfo() != CCValAssign::Full)
1342 return false;
1343 // Only handle register returns for now.
1344 if (!VA.isRegLoc())
1345 return false;
1346
1347 // The calling-convention tables for x87 returns don't tell
1348 // the whole story.
1349 if (VA.getLocReg() == X86::FP0 || VA.getLocReg() == X86::FP1)
1350 return false;
1351
1352 Register SrcReg = Reg + VA.getValNo();
1353 EVT SrcVT = TLI.getValueType(DL, RV->getType());
1354 EVT DstVT = VA.getValVT();
1355 // Special handling for extended integers.
1356 if (SrcVT != DstVT) {
1357 if (SrcVT != MVT::i1 && SrcVT != MVT::i8 && SrcVT != MVT::i16)
1358 return false;
1359
1360 if (!Outs[0].Flags.isZExt() && !Outs[0].Flags.isSExt())
1361 return false;
1362
1363 if (SrcVT == MVT::i1) {
1364 if (Outs[0].Flags.isSExt())
1365 return false;
1366 SrcReg = fastEmitZExtFromI1(MVT::i8, SrcReg);
1367 SrcVT = MVT::i8;
1368 }
1369 if (SrcVT != DstVT) {
1370 unsigned Op =
1371 Outs[0].Flags.isZExt() ? ISD::ZERO_EXTEND : ISD::SIGN_EXTEND;
1372 SrcReg =
1373 fastEmit_r(SrcVT.getSimpleVT(), DstVT.getSimpleVT(), Op, SrcReg);
1374 }
1375 }
1376
1377 // Make the copy.
1378 Register DstReg = VA.getLocReg();
1379 const TargetRegisterClass *SrcRC = MRI.getRegClass(SrcReg);
1380 // Avoid a cross-class copy. This is very unlikely.
1381 if (!SrcRC->contains(DstReg))
1382 return false;
1383 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
1384 TII.get(TargetOpcode::COPY), DstReg).addReg(SrcReg);
1385
1386 // Add register to return instruction.
1387 RetRegs.push_back(VA.getLocReg());
1388 }
1389
1390 // Swift calling convention does not require we copy the sret argument
1391 // into %rax/%eax for the return, and SRetReturnReg is not set for Swift.
1392
1393 // All x86 ABIs require that for returning structs by value we copy
1394 // the sret argument into %rax/%eax (depending on ABI) for the return.
1395 // We saved the argument into a virtual register in the entry block,
1396 // so now we copy the value out and into %rax/%eax.
1397 if (F.hasStructRetAttr() && CC != CallingConv::Swift &&
1398 CC != CallingConv::SwiftTail) {
1399 Register Reg = X86MFInfo->getSRetReturnReg();
1400 assert(Reg &&
1401 "SRetReturnReg should have been set in LowerFormalArguments()!");
1402 Register RetReg = Subtarget->isTarget64BitLP64() ? X86::RAX : X86::EAX;
1403 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
1404 TII.get(TargetOpcode::COPY), RetReg).addReg(Reg);
1405 RetRegs.push_back(RetReg);
1406 }
1407
1408 // Now emit the RET.
1409 MachineInstrBuilder MIB;
1410 if (X86MFInfo->getBytesToPopOnReturn()) {
1411 MIB = BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
1412 TII.get(Subtarget->is64Bit() ? X86::RETI64 : X86::RETI32))
1413 .addImm(X86MFInfo->getBytesToPopOnReturn());
1414 } else {
1415 MIB = BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
1416 TII.get(Subtarget->is64Bit() ? X86::RET64 : X86::RET32));
1417 }
1418 for (Register Reg : RetRegs)
1419 MIB.addReg(Reg, RegState::Implicit);
1420 return true;
1421}
1422
1423/// X86SelectLoad - Select and emit code to implement load instructions.
1424///
1425bool X86FastISel::X86SelectLoad(const Instruction *I) {
1426 const LoadInst *LI = cast<LoadInst>(I);
1427
1428 // Atomic loads need special handling.
1429 if (LI->isAtomic())
1430 return false;
1431
1432 const Value *SV = I->getOperand(0);
1433 if (TLI.supportSwiftError()) {
1434 // Swifterror values can come from either a function parameter with
1435 // swifterror attribute or an alloca with swifterror attribute.
1436 if (const Argument *Arg = dyn_cast<Argument>(SV)) {
1437 if (Arg->hasSwiftErrorAttr())
1438 return false;
1439 }
1440
1441 if (const AllocaInst *Alloca = dyn_cast<AllocaInst>(SV)) {
1442 if (Alloca->isSwiftError())
1443 return false;
1444 }
1445 }
1446
1447 MVT VT;
1448 if (!isTypeLegal(LI->getType(), VT, /*AllowI1=*/true))
1449 return false;
1450
1451 const Value *Ptr = LI->getPointerOperand();
1452
1453 X86AddressMode AM;
1454 if (!X86SelectAddress(Ptr, AM))
1455 return false;
1456
1457 Register ResultReg;
1458 if (!X86FastEmitLoad(VT, AM, createMachineMemOperandFor(LI), ResultReg,
1459 LI->getAlign().value()))
1460 return false;
1461
1462 updateValueMap(I, ResultReg);
1463 return true;
1464}
1465
1466static unsigned X86ChooseCmpOpcode(EVT VT, const X86Subtarget *Subtarget) {
1467 bool HasAVX512 = Subtarget->hasAVX512();
1468 bool HasAVX = Subtarget->hasAVX();
1469 bool HasSSE1 = Subtarget->hasSSE1();
1470 bool HasSSE2 = Subtarget->hasSSE2();
1471
1472 switch (VT.getSimpleVT().SimpleTy) {
1473 default: return 0;
1474 case MVT::i8: return X86::CMP8rr;
1475 case MVT::i16: return X86::CMP16rr;
1476 case MVT::i32: return X86::CMP32rr;
1477 case MVT::i64: return X86::CMP64rr;
1478 case MVT::f32:
1479 return HasAVX512 ? X86::VUCOMISSZrr
1480 : HasAVX ? X86::VUCOMISSrr
1481 : HasSSE1 ? X86::UCOMISSrr
1482 : 0;
1483 case MVT::f64:
1484 return HasAVX512 ? X86::VUCOMISDZrr
1485 : HasAVX ? X86::VUCOMISDrr
1486 : HasSSE2 ? X86::UCOMISDrr
1487 : 0;
1488 }
1489}
1490
1491/// If we have a comparison with RHS as the RHS of the comparison, return an
1492/// opcode that works for the compare (e.g. CMP32ri) otherwise return 0.
1493static unsigned X86ChooseCmpImmediateOpcode(EVT VT, const ConstantInt *RHSC) {
1494 switch (VT.getSimpleVT().SimpleTy) {
1495 // Otherwise, we can't fold the immediate into this comparison.
1496 default:
1497 return 0;
1498 case MVT::i8:
1499 return X86::CMP8ri;
1500 case MVT::i16:
1501 return X86::CMP16ri;
1502 case MVT::i32:
1503 return X86::CMP32ri;
1504 case MVT::i64:
1505 // 64-bit comparisons are only valid if the immediate fits in a 32-bit sext
1506 // field.
1507 return isInt<32>(RHSC->getSExtValue()) ? X86::CMP64ri32 : 0;
1508 }
1509}
1510
1511bool X86FastISel::X86FastEmitCompare(const Value *Op0, const Value *Op1, EVT VT,
1512 const DebugLoc &CurMIMD) {
1513 Register Op0Reg = getRegForValue(Op0);
1514 if (!Op0Reg)
1515 return false;
1516
1517 // Handle 'null' like i32/i64 0.
1518 if (isa<ConstantPointerNull>(Op1))
1519 Op1 = Constant::getNullValue(DL.getIntPtrType(Op0->getContext()));
1520
1521 // We have two options: compare with register or immediate. If the RHS of
1522 // the compare is an immediate that we can fold into this compare, use
1523 // CMPri, otherwise use CMPrr.
1524 if (const ConstantInt *Op1C = dyn_cast<ConstantInt>(Op1)) {
1525 if (unsigned CompareImmOpc = X86ChooseCmpImmediateOpcode(VT, Op1C)) {
1526 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, CurMIMD, TII.get(CompareImmOpc))
1527 .addReg(Op0Reg)
1528 .addImm(Op1C->getSExtValue());
1529 return true;
1530 }
1531 }
1532
1533 unsigned CompareOpc = X86ChooseCmpOpcode(VT, Subtarget);
1534 if (CompareOpc == 0) return false;
1535
1536 Register Op1Reg = getRegForValue(Op1);
1537 if (!Op1Reg)
1538 return false;
1539 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, CurMIMD, TII.get(CompareOpc))
1540 .addReg(Op0Reg)
1541 .addReg(Op1Reg);
1542
1543 return true;
1544}
1545
1546#define GET_SETCC \
1547 ((!Subtarget->hasZU() || Subtarget->preferLegacySetCC()) ? X86::SETCCr \
1548 : X86::SETZUCCr)
1549
1550bool X86FastISel::X86SelectCmp(const Instruction *I) {
1551 const CmpInst *CI = cast<CmpInst>(I);
1552
1553 MVT VT;
1554 if (!isTypeLegal(I->getOperand(0)->getType(), VT))
1555 return false;
1556
1557 // Below code only works for scalars.
1558 if (VT.isVector())
1559 return false;
1560
1561 // Try to optimize or fold the cmp.
1562 CmpInst::Predicate Predicate = optimizeCmpPredicate(CI);
1563 Register ResultReg;
1564 switch (Predicate) {
1565 default: break;
1566 case CmpInst::FCMP_FALSE: {
1567 ResultReg = emitMOV32r0();
1568 ResultReg = fastEmitInst_extractsubreg(MVT::i8, ResultReg, X86::sub_8bit);
1569 if (!ResultReg)
1570 return false;
1571 break;
1572 }
1573 case CmpInst::FCMP_TRUE: {
1574 ResultReg = createResultReg(&X86::GR8RegClass);
1575 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::MOV8ri),
1576 ResultReg).addImm(1);
1577 break;
1578 }
1579 }
1580
1581 if (ResultReg) {
1582 updateValueMap(I, ResultReg);
1583 return true;
1584 }
1585
1586 const Value *LHS = CI->getOperand(0);
1587 const Value *RHS = CI->getOperand(1);
1588
1589 // The optimizer might have replaced fcmp oeq %x, %x with fcmp ord %x, 0.0.
1590 // We don't have to materialize a zero constant for this case and can just use
1591 // %x again on the RHS.
1593 const auto *RHSC = dyn_cast<ConstantFP>(RHS);
1594 if (RHSC && RHSC->isNullValue())
1595 RHS = LHS;
1596 }
1597
1598 // FCMP_OEQ and FCMP_UNE cannot be checked with a single instruction.
1599 static const uint16_t SETFOpcTable[2][3] = {
1600 { X86::COND_E, X86::COND_NP, X86::AND8rr },
1601 { X86::COND_NE, X86::COND_P, X86::OR8rr }
1602 };
1603 const uint16_t *SETFOpc = nullptr;
1604 switch (Predicate) {
1605 default: break;
1606 case CmpInst::FCMP_OEQ: SETFOpc = &SETFOpcTable[0][0]; break;
1607 case CmpInst::FCMP_UNE: SETFOpc = &SETFOpcTable[1][0]; break;
1608 }
1609
1610 ResultReg = createResultReg(&X86::GR8RegClass);
1611 if (SETFOpc) {
1612 if (!X86FastEmitCompare(LHS, RHS, VT, I->getDebugLoc()))
1613 return false;
1614
1615 Register FlagReg1 = createResultReg(&X86::GR8RegClass);
1616 Register FlagReg2 = createResultReg(&X86::GR8RegClass);
1617 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(GET_SETCC),
1618 FlagReg1)
1619 .addImm(SETFOpc[0]);
1620 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(GET_SETCC),
1621 FlagReg2)
1622 .addImm(SETFOpc[1]);
1623 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(SETFOpc[2]),
1624 ResultReg)
1625 .addReg(FlagReg1)
1626 .addReg(FlagReg2)
1627 .setOperandDead(3); // implicit-def $eflags
1628 updateValueMap(I, ResultReg);
1629 return true;
1630 }
1631
1632 X86::CondCode CC;
1633 bool SwapArgs;
1634 std::tie(CC, SwapArgs) = X86::getX86ConditionCode(Predicate);
1635 assert(CC <= X86::LAST_VALID_COND && "Unexpected condition code.");
1636
1637 if (SwapArgs)
1638 std::swap(LHS, RHS);
1639
1640 // Emit a compare of LHS/RHS.
1641 if (!X86FastEmitCompare(LHS, RHS, VT, I->getDebugLoc()))
1642 return false;
1643
1644 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(GET_SETCC), ResultReg)
1645 .addImm(CC);
1646 updateValueMap(I, ResultReg);
1647 return true;
1648}
1649
1650bool X86FastISel::X86SelectZExt(const Instruction *I) {
1651 EVT DstVT = TLI.getValueType(DL, I->getType());
1652 if (!TLI.isTypeLegal(DstVT))
1653 return false;
1654
1655 Register ResultReg = getRegForValue(I->getOperand(0));
1656 if (!ResultReg)
1657 return false;
1658
1659 // Handle zero-extension from i1 to i8, which is common.
1660 MVT SrcVT = TLI.getSimpleValueType(DL, I->getOperand(0)->getType());
1661 if (SrcVT == MVT::i1) {
1662 // Set the high bits to zero.
1663 ResultReg = fastEmitZExtFromI1(MVT::i8, ResultReg);
1664 SrcVT = MVT::i8;
1665
1666 if (!ResultReg)
1667 return false;
1668 }
1669
1670 if (DstVT == MVT::i64) {
1671 // Handle extension to 64-bits via sub-register shenanigans.
1672 unsigned MovInst;
1673
1674 switch (SrcVT.SimpleTy) {
1675 case MVT::i8: MovInst = X86::MOVZX32rr8; break;
1676 case MVT::i16: MovInst = X86::MOVZX32rr16; break;
1677 case MVT::i32: MovInst = X86::MOV32rr; break;
1678 default: llvm_unreachable("Unexpected zext to i64 source type");
1679 }
1680
1681 Register Result32 = createResultReg(&X86::GR32RegClass);
1682 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(MovInst), Result32)
1683 .addReg(ResultReg);
1684
1685 ResultReg = createResultReg(&X86::GR64RegClass);
1686 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
1687 TII.get(TargetOpcode::SUBREG_TO_REG), ResultReg)
1688 .addReg(Result32)
1689 .addImm(X86::sub_32bit);
1690 } else if (DstVT == MVT::i16) {
1691 // i8->i16 doesn't exist in the autogenerated isel table. Need to zero
1692 // extend to 32-bits and then extract down to 16-bits.
1693 Register Result32 = createResultReg(&X86::GR32RegClass);
1694 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::MOVZX32rr8),
1695 Result32).addReg(ResultReg);
1696
1697 ResultReg = fastEmitInst_extractsubreg(MVT::i16, Result32, X86::sub_16bit);
1698 } else if (DstVT != MVT::i8) {
1699 ResultReg = fastEmit_r(MVT::i8, DstVT.getSimpleVT(), ISD::ZERO_EXTEND,
1700 ResultReg);
1701 if (!ResultReg)
1702 return false;
1703 }
1704
1705 updateValueMap(I, ResultReg);
1706 return true;
1707}
1708
1709bool X86FastISel::X86SelectSExt(const Instruction *I) {
1710 EVT DstVT = TLI.getValueType(DL, I->getType());
1711 if (!TLI.isTypeLegal(DstVT))
1712 return false;
1713
1714 Register ResultReg = getRegForValue(I->getOperand(0));
1715 if (!ResultReg)
1716 return false;
1717
1718 // Handle sign-extension from i1 to i8.
1719 MVT SrcVT = TLI.getSimpleValueType(DL, I->getOperand(0)->getType());
1720 if (SrcVT == MVT::i1) {
1721 // Set the high bits to zero.
1722 Register ZExtReg = fastEmitZExtFromI1(MVT::i8, ResultReg);
1723 if (!ZExtReg)
1724 return false;
1725
1726 // Negate the result to make an 8-bit sign extended value.
1727 ResultReg = createResultReg(&X86::GR8RegClass);
1728 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::NEG8r),
1729 ResultReg)
1730 .addReg(ZExtReg)
1731 .setOperandDead(2);
1732
1733 SrcVT = MVT::i8;
1734 }
1735
1736 if (DstVT == MVT::i16) {
1737 // i8->i16 doesn't exist in the autogenerated isel table. Need to sign
1738 // extend to 32-bits and then extract down to 16-bits.
1739 Register Result32 = createResultReg(&X86::GR32RegClass);
1740 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::MOVSX32rr8),
1741 Result32).addReg(ResultReg);
1742
1743 ResultReg = fastEmitInst_extractsubreg(MVT::i16, Result32, X86::sub_16bit);
1744 } else if (DstVT != MVT::i8) {
1745 ResultReg = fastEmit_r(MVT::i8, DstVT.getSimpleVT(), ISD::SIGN_EXTEND,
1746 ResultReg);
1747 if (!ResultReg)
1748 return false;
1749 }
1750
1751 updateValueMap(I, ResultReg);
1752 return true;
1753}
1754
1755bool X86FastISel::X86SelectBranch(const Instruction *I) {
1756 // Unconditional branches are selected by tablegen-generated code.
1757 // Handle a conditional branch.
1758 const CondBrInst *BI = cast<CondBrInst>(I);
1759 MachineBasicBlock *TrueMBB = FuncInfo.getMBB(BI->getSuccessor(0));
1760 MachineBasicBlock *FalseMBB = FuncInfo.getMBB(BI->getSuccessor(1));
1761
1762 // Fold the common case of a conditional branch with a comparison
1763 // in the same block (values defined on other blocks may not have
1764 // initialized registers).
1765 X86::CondCode CC;
1766 if (const CmpInst *CI = dyn_cast<CmpInst>(BI->getCondition())) {
1767 if (CI->hasOneUse() && CI->getParent() == I->getParent()) {
1768 EVT VT = TLI.getValueType(DL, CI->getOperand(0)->getType());
1769
1770 // Try to optimize or fold the cmp.
1771 CmpInst::Predicate Predicate = optimizeCmpPredicate(CI);
1772 switch (Predicate) {
1773 default: break;
1774 case CmpInst::FCMP_FALSE: fastEmitBranch(FalseMBB, MIMD.getDL()); return true;
1775 case CmpInst::FCMP_TRUE: fastEmitBranch(TrueMBB, MIMD.getDL()); return true;
1776 }
1777
1778 const Value *CmpLHS = CI->getOperand(0);
1779 const Value *CmpRHS = CI->getOperand(1);
1780
1781 // The optimizer might have replaced fcmp oeq %x, %x with fcmp ord %x,
1782 // 0.0.
1783 // We don't have to materialize a zero constant for this case and can just
1784 // use %x again on the RHS.
1785 if (Predicate == CmpInst::FCMP_ORD || Predicate == CmpInst::FCMP_UNO) {
1786 const auto *CmpRHSC = dyn_cast<ConstantFP>(CmpRHS);
1787 if (CmpRHSC && CmpRHSC->isNullValue())
1788 CmpRHS = CmpLHS;
1789 }
1790
1791 // Try to take advantage of fallthrough opportunities.
1792 if (FuncInfo.MBB->isLayoutSuccessor(TrueMBB)) {
1793 std::swap(TrueMBB, FalseMBB);
1795 }
1796
1797 // FCMP_OEQ and FCMP_UNE cannot be expressed with a single flag/condition
1798 // code check. Instead two branch instructions are required to check all
1799 // the flags. First we change the predicate to a supported condition code,
1800 // which will be the first branch. Later one we will emit the second
1801 // branch.
1802 bool NeedExtraBranch = false;
1803 switch (Predicate) {
1804 default: break;
1805 case CmpInst::FCMP_OEQ:
1806 std::swap(TrueMBB, FalseMBB);
1807 [[fallthrough]];
1808 case CmpInst::FCMP_UNE:
1809 NeedExtraBranch = true;
1811 break;
1812 }
1813
1814 bool SwapArgs;
1815 std::tie(CC, SwapArgs) = X86::getX86ConditionCode(Predicate);
1816 assert(CC <= X86::LAST_VALID_COND && "Unexpected condition code.");
1817
1818 if (SwapArgs)
1819 std::swap(CmpLHS, CmpRHS);
1820
1821 // Emit a compare of the LHS and RHS, setting the flags.
1822 if (!X86FastEmitCompare(CmpLHS, CmpRHS, VT, CI->getDebugLoc()))
1823 return false;
1824
1825 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::JCC_1))
1826 .addMBB(TrueMBB).addImm(CC);
1827
1828 // X86 requires a second branch to handle UNE (and OEQ, which is mapped
1829 // to UNE above).
1830 if (NeedExtraBranch) {
1831 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::JCC_1))
1832 .addMBB(TrueMBB).addImm(X86::COND_P);
1833 }
1834
1835 finishCondBranch(BI->getParent(), TrueMBB, FalseMBB);
1836 return true;
1837 }
1838 } else if (TruncInst *TI = dyn_cast<TruncInst>(BI->getCondition())) {
1839 // Handle things like "%cond = trunc i32 %X to i1 / br i1 %cond", which
1840 // typically happen for _Bool and C++ bools.
1841 MVT SourceVT;
1842 if (TI->hasOneUse() && TI->getParent() == I->getParent() &&
1843 isTypeLegal(TI->getOperand(0)->getType(), SourceVT)) {
1844 unsigned TestOpc = 0;
1845 switch (SourceVT.SimpleTy) {
1846 default: break;
1847 case MVT::i8: TestOpc = X86::TEST8ri; break;
1848 case MVT::i16: TestOpc = X86::TEST16ri; break;
1849 case MVT::i32: TestOpc = X86::TEST32ri; break;
1850 case MVT::i64: TestOpc = X86::TEST64ri32; break;
1851 }
1852 if (TestOpc) {
1853 Register OpReg = getRegForValue(TI->getOperand(0));
1854 if (!OpReg)
1855 return false;
1856
1857 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(TestOpc))
1858 .addReg(OpReg).addImm(1);
1859
1860 unsigned JmpCond = X86::COND_NE;
1861 if (FuncInfo.MBB->isLayoutSuccessor(TrueMBB)) {
1862 std::swap(TrueMBB, FalseMBB);
1863 JmpCond = X86::COND_E;
1864 }
1865
1866 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::JCC_1))
1867 .addMBB(TrueMBB).addImm(JmpCond);
1868
1869 finishCondBranch(BI->getParent(), TrueMBB, FalseMBB);
1870 return true;
1871 }
1872 }
1873 } else if (foldX86XALUIntrinsic(CC, BI, BI->getCondition())) {
1874 // Fake request the condition, otherwise the intrinsic might be completely
1875 // optimized away.
1876 Register TmpReg = getRegForValue(BI->getCondition());
1877 if (!TmpReg)
1878 return false;
1879
1880 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::JCC_1))
1881 .addMBB(TrueMBB).addImm(CC);
1882 finishCondBranch(BI->getParent(), TrueMBB, FalseMBB);
1883 return true;
1884 }
1885
1886 // Otherwise do a clumsy setcc and re-test it.
1887 // Note that i1 essentially gets ANY_EXTEND'ed to i8 where it isn't used
1888 // in an explicit cast, so make sure to handle that correctly.
1889 Register OpReg = getRegForValue(BI->getCondition());
1890 if (!OpReg)
1891 return false;
1892
1893 // In case OpReg is a K register, COPY to a GPR
1894 if (MRI.getRegClass(OpReg) == &X86::VK1RegClass) {
1895 Register KOpReg = OpReg;
1896 OpReg = createResultReg(&X86::GR32RegClass);
1897 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
1898 TII.get(TargetOpcode::COPY), OpReg)
1899 .addReg(KOpReg);
1900 OpReg = fastEmitInst_extractsubreg(MVT::i8, OpReg, X86::sub_8bit);
1901 }
1902 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::TEST8ri))
1903 .addReg(OpReg)
1904 .addImm(1);
1905 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::JCC_1))
1906 .addMBB(TrueMBB).addImm(X86::COND_NE);
1907 finishCondBranch(BI->getParent(), TrueMBB, FalseMBB);
1908 return true;
1909}
1910
1911bool X86FastISel::X86SelectShift(const Instruction *I) {
1912 Register CReg;
1913 unsigned OpReg;
1914 const TargetRegisterClass *RC = nullptr;
1915 if (I->getType()->isIntegerTy(8)) {
1916 CReg = X86::CL;
1917 RC = &X86::GR8RegClass;
1918 switch (I->getOpcode()) {
1919 case Instruction::LShr: OpReg = X86::SHR8rCL; break;
1920 case Instruction::AShr: OpReg = X86::SAR8rCL; break;
1921 case Instruction::Shl: OpReg = X86::SHL8rCL; break;
1922 default: return false;
1923 }
1924 } else if (I->getType()->isIntegerTy(16)) {
1925 CReg = X86::CX;
1926 RC = &X86::GR16RegClass;
1927 switch (I->getOpcode()) {
1928 default: llvm_unreachable("Unexpected shift opcode");
1929 case Instruction::LShr: OpReg = X86::SHR16rCL; break;
1930 case Instruction::AShr: OpReg = X86::SAR16rCL; break;
1931 case Instruction::Shl: OpReg = X86::SHL16rCL; break;
1932 }
1933 } else if (I->getType()->isIntegerTy(32)) {
1934 CReg = X86::ECX;
1935 RC = &X86::GR32RegClass;
1936 switch (I->getOpcode()) {
1937 default: llvm_unreachable("Unexpected shift opcode");
1938 case Instruction::LShr: OpReg = X86::SHR32rCL; break;
1939 case Instruction::AShr: OpReg = X86::SAR32rCL; break;
1940 case Instruction::Shl: OpReg = X86::SHL32rCL; break;
1941 }
1942 } else if (I->getType()->isIntegerTy(64)) {
1943 CReg = X86::RCX;
1944 RC = &X86::GR64RegClass;
1945 switch (I->getOpcode()) {
1946 default: llvm_unreachable("Unexpected shift opcode");
1947 case Instruction::LShr: OpReg = X86::SHR64rCL; break;
1948 case Instruction::AShr: OpReg = X86::SAR64rCL; break;
1949 case Instruction::Shl: OpReg = X86::SHL64rCL; break;
1950 }
1951 } else {
1952 return false;
1953 }
1954
1955 MVT VT;
1956 if (!isTypeLegal(I->getType(), VT))
1957 return false;
1958
1959 Register Op0Reg = getRegForValue(I->getOperand(0));
1960 if (!Op0Reg)
1961 return false;
1962
1963 Register Op1Reg = getRegForValue(I->getOperand(1));
1964 if (!Op1Reg)
1965 return false;
1966 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(TargetOpcode::COPY),
1967 CReg).addReg(Op1Reg);
1968
1969 // The shift instruction uses X86::CL. If we defined a super-register
1970 // of X86::CL, emit a subreg KILL to precisely describe what we're doing here.
1971 if (CReg != X86::CL)
1972 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
1973 TII.get(TargetOpcode::KILL), X86::CL)
1974 .addReg(CReg, RegState::Kill);
1975
1976 Register ResultReg = createResultReg(RC);
1977 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(OpReg), ResultReg)
1978 .addReg(Op0Reg)
1979 .setOperandDead(2); // EFLAGS
1980 updateValueMap(I, ResultReg);
1981 return true;
1982}
1983
1984bool X86FastISel::X86SelectMul(const Instruction *I) {
1985 if (!I->getType()->isIntegerTy(8))
1986 return false;
1987
1988 Register LHSReg = getRegForValue(I->getOperand(0));
1989 if (!LHSReg)
1990 return false;
1991
1992 Register RHSReg = getRegForValue(I->getOperand(1));
1993 if (!RHSReg)
1994 return false;
1995
1996 Register ResultReg =
1997 X86FastEmitMul(X86::MUL8r, MVT::i8, X86::AL, LHSReg, RHSReg);
1998 updateValueMap(I, ResultReg);
1999 return true;
2000}
2001
2002bool X86FastISel::X86SelectDivRem(const Instruction *I) {
2003 const static unsigned NumTypes = 4; // i8, i16, i32, i64
2004 const static unsigned NumOps = 4; // SDiv, SRem, UDiv, URem
2005 const static bool S = true; // IsSigned
2006 const static bool U = false; // !IsSigned
2007 const static unsigned Copy = TargetOpcode::COPY;
2008 // For the X86 DIV/IDIV instruction, in most cases the dividend
2009 // (numerator) must be in a specific register pair highreg:lowreg,
2010 // producing the quotient in lowreg and the remainder in highreg.
2011 // For most data types, to set up the instruction, the dividend is
2012 // copied into lowreg, and lowreg is sign-extended or zero-extended
2013 // into highreg. The exception is i8, where the dividend is defined
2014 // as a single register rather than a register pair, and we
2015 // therefore directly sign-extend or zero-extend the dividend into
2016 // lowreg, instead of copying, and ignore the highreg.
2017 const static struct DivRemEntry {
2018 // The following portion depends only on the data type.
2019 const TargetRegisterClass *RC;
2020 unsigned LowInReg; // low part of the register pair
2021 unsigned HighInReg; // high part of the register pair
2022 // The following portion depends on both the data type and the operation.
2023 struct DivRemResult {
2024 unsigned OpDivRem; // The specific DIV/IDIV opcode to use.
2025 unsigned OpSignExtend; // Opcode for sign-extending lowreg into
2026 // highreg, or copying a zero into highreg.
2027 unsigned OpCopy; // Opcode for copying dividend into lowreg, or
2028 // zero/sign-extending into lowreg for i8.
2029 unsigned DivRemResultReg; // Register containing the desired result.
2030 bool IsOpSigned; // Whether to use signed or unsigned form.
2031 } ResultTable[NumOps];
2032 } OpTable[NumTypes] = {
2033 { &X86::GR8RegClass, X86::AX, 0, {
2034 { X86::IDIV8r, 0, X86::MOVSX16rr8, X86::AL, S }, // SDiv
2035 { X86::IDIV8r, 0, X86::MOVSX16rr8, X86::AH, S }, // SRem
2036 { X86::DIV8r, 0, X86::MOVZX16rr8, X86::AL, U }, // UDiv
2037 { X86::DIV8r, 0, X86::MOVZX16rr8, X86::AH, U }, // URem
2038 }
2039 }, // i8
2040 { &X86::GR16RegClass, X86::AX, X86::DX, {
2041 { X86::IDIV16r, X86::CWD, Copy, X86::AX, S }, // SDiv
2042 { X86::IDIV16r, X86::CWD, Copy, X86::DX, S }, // SRem
2043 { X86::DIV16r, X86::MOV32r0, Copy, X86::AX, U }, // UDiv
2044 { X86::DIV16r, X86::MOV32r0, Copy, X86::DX, U }, // URem
2045 }
2046 }, // i16
2047 { &X86::GR32RegClass, X86::EAX, X86::EDX, {
2048 { X86::IDIV32r, X86::CDQ, Copy, X86::EAX, S }, // SDiv
2049 { X86::IDIV32r, X86::CDQ, Copy, X86::EDX, S }, // SRem
2050 { X86::DIV32r, X86::MOV32r0, Copy, X86::EAX, U }, // UDiv
2051 { X86::DIV32r, X86::MOV32r0, Copy, X86::EDX, U }, // URem
2052 }
2053 }, // i32
2054 { &X86::GR64RegClass, X86::RAX, X86::RDX, {
2055 { X86::IDIV64r, X86::CQO, Copy, X86::RAX, S }, // SDiv
2056 { X86::IDIV64r, X86::CQO, Copy, X86::RDX, S }, // SRem
2057 { X86::DIV64r, X86::MOV32r0, Copy, X86::RAX, U }, // UDiv
2058 { X86::DIV64r, X86::MOV32r0, Copy, X86::RDX, U }, // URem
2059 }
2060 }, // i64
2061 };
2062
2063 MVT VT;
2064 if (!isTypeLegal(I->getType(), VT))
2065 return false;
2066
2067 unsigned TypeIndex, OpIndex;
2068 switch (VT.SimpleTy) {
2069 default: return false;
2070 case MVT::i8: TypeIndex = 0; break;
2071 case MVT::i16: TypeIndex = 1; break;
2072 case MVT::i32: TypeIndex = 2; break;
2073 case MVT::i64: TypeIndex = 3;
2074 if (!Subtarget->is64Bit())
2075 return false;
2076 break;
2077 }
2078
2079 switch (I->getOpcode()) {
2080 default: llvm_unreachable("Unexpected div/rem opcode");
2081 case Instruction::SDiv: OpIndex = 0; break;
2082 case Instruction::SRem: OpIndex = 1; break;
2083 case Instruction::UDiv: OpIndex = 2; break;
2084 case Instruction::URem: OpIndex = 3; break;
2085 }
2086
2087 const DivRemEntry &TypeEntry = OpTable[TypeIndex];
2088 const DivRemEntry::DivRemResult &OpEntry = TypeEntry.ResultTable[OpIndex];
2089 Register Op0Reg = getRegForValue(I->getOperand(0));
2090 if (!Op0Reg)
2091 return false;
2092 Register Op1Reg = getRegForValue(I->getOperand(1));
2093 if (!Op1Reg)
2094 return false;
2095
2096 // Move op0 into low-order input register.
2097 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2098 TII.get(OpEntry.OpCopy), TypeEntry.LowInReg).addReg(Op0Reg);
2099 // Zero-extend or sign-extend into high-order input register.
2100 if (OpEntry.OpSignExtend) {
2101 if (OpEntry.IsOpSigned)
2102 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2103 TII.get(OpEntry.OpSignExtend));
2104 else {
2105 Register Zero32 = emitMOV32r0();
2106
2107 // Copy the zero into the appropriate sub/super/identical physical
2108 // register. Unfortunately the operations needed are not uniform enough
2109 // to fit neatly into the table above.
2110 if (VT == MVT::i16) {
2111 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Copy),
2112 TypeEntry.HighInReg)
2113 .addReg(Zero32, {}, X86::sub_16bit);
2114 } else if (VT == MVT::i32) {
2115 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2116 TII.get(Copy), TypeEntry.HighInReg)
2117 .addReg(Zero32);
2118 } else if (VT == MVT::i64) {
2119 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2120 TII.get(TargetOpcode::SUBREG_TO_REG), TypeEntry.HighInReg)
2121 .addReg(Zero32)
2122 .addImm(X86::sub_32bit);
2123 }
2124 }
2125 }
2126 // For i8 remainder, we can't reference ah directly, as we'll end
2127 // up with bogus copies like %r9b = COPY %ah. Reference ax
2128 // instead to prevent ah references in a rex instruction.
2129 //
2130 // The current assumption of the fast register allocator is that isel
2131 // won't generate explicit references to the GR8_NOREX registers. If
2132 // the allocator and/or the backend get enhanced to be more robust in
2133 // that regard, this can be, and should be, removed.
2134 bool UseAXForRem = (I->getOpcode() == Instruction::SRem ||
2135 I->getOpcode() == Instruction::URem) &&
2136 OpEntry.DivRemResultReg == X86::AH && Subtarget->is64Bit();
2137
2138 // Generate the DIV/IDIV instruction.
2139 Register UsedReg =
2140 UseAXForRem ? Register(X86::AX) : Register(OpEntry.DivRemResultReg);
2141 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(OpEntry.OpDivRem))
2142 .addReg(Op1Reg)
2143 ->setPhysRegsDeadExcept(UsedReg, TRI);
2144
2145 Register ResultReg;
2146 if (UseAXForRem) {
2147 Register SourceSuperReg = createResultReg(&X86::GR16RegClass);
2148 Register ResultSuperReg = createResultReg(&X86::GR16RegClass);
2149 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2150 TII.get(Copy), SourceSuperReg).addReg(X86::AX);
2151
2152 // Shift AX right by 8 bits instead of using AH.
2153 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::SHR16ri),
2154 ResultSuperReg)
2155 .addReg(SourceSuperReg)
2156 .addImm(8)
2157 .setOperandDead(3);
2158
2159 // Now reference the 8-bit subreg of the result.
2160 ResultReg = fastEmitInst_extractsubreg(MVT::i8, ResultSuperReg,
2161 X86::sub_8bit);
2162 }
2163 // Copy the result out of the physreg if we haven't already.
2164 if (!ResultReg) {
2165 ResultReg = createResultReg(TypeEntry.RC);
2166 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Copy), ResultReg)
2167 .addReg(OpEntry.DivRemResultReg);
2168 }
2169 updateValueMap(I, ResultReg);
2170
2171 return true;
2172}
2173
2174/// Emit a conditional move instruction (if the are supported) to lower
2175/// the select.
2176bool X86FastISel::X86FastEmitCMoveSelect(MVT RetVT, const Instruction *I) {
2177 // Check if the subtarget supports these instructions.
2178 if (!Subtarget->canUseCMOV())
2179 return false;
2180
2181 // FIXME: Add support for i8.
2182 if (RetVT < MVT::i16 || RetVT > MVT::i64)
2183 return false;
2184
2185 const Value *Cond = I->getOperand(0);
2186 const TargetRegisterClass *RC = TLI.getRegClassFor(RetVT);
2187 bool NeedTest = true;
2189
2190 // Optimize conditions coming from a compare if both instructions are in the
2191 // same basic block (values defined in other basic blocks may not have
2192 // initialized registers).
2193 const auto *CI = dyn_cast<CmpInst>(Cond);
2194 if (CI && (CI->getParent() == I->getParent())) {
2195 CmpInst::Predicate Predicate = optimizeCmpPredicate(CI);
2196
2197 // FCMP_OEQ and FCMP_UNE cannot be checked with a single instruction.
2198 static const uint16_t SETFOpcTable[2][3] = {
2199 { X86::COND_NP, X86::COND_E, X86::TEST8rr },
2200 { X86::COND_P, X86::COND_NE, X86::OR8rr }
2201 };
2202 const uint16_t *SETFOpc = nullptr;
2203 switch (Predicate) {
2204 default: break;
2205 case CmpInst::FCMP_OEQ:
2206 SETFOpc = &SETFOpcTable[0][0];
2208 break;
2209 case CmpInst::FCMP_UNE:
2210 SETFOpc = &SETFOpcTable[1][0];
2212 break;
2213 }
2214
2215 bool NeedSwap;
2216 std::tie(CC, NeedSwap) = X86::getX86ConditionCode(Predicate);
2217 assert(CC <= X86::LAST_VALID_COND && "Unexpected condition code.");
2218
2219 const Value *CmpLHS = CI->getOperand(0);
2220 const Value *CmpRHS = CI->getOperand(1);
2221 if (NeedSwap)
2222 std::swap(CmpLHS, CmpRHS);
2223
2224 EVT CmpVT = TLI.getValueType(DL, CmpLHS->getType());
2225 // Emit a compare of the LHS and RHS, setting the flags.
2226 if (!X86FastEmitCompare(CmpLHS, CmpRHS, CmpVT, CI->getDebugLoc()))
2227 return false;
2228
2229 if (SETFOpc) {
2230 Register FlagReg1 = createResultReg(&X86::GR8RegClass);
2231 Register FlagReg2 = createResultReg(&X86::GR8RegClass);
2232 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(GET_SETCC),
2233 FlagReg1)
2234 .addImm(SETFOpc[0]);
2235 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(GET_SETCC),
2236 FlagReg2)
2237 .addImm(SETFOpc[1]);
2238 auto const &II = TII.get(SETFOpc[2]);
2239 if (II.getNumDefs()) {
2240 Register TmpReg = createResultReg(&X86::GR8RegClass);
2241 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, II, TmpReg)
2242 .addReg(FlagReg2).addReg(FlagReg1);
2243 } else {
2244 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, II)
2245 .addReg(FlagReg2).addReg(FlagReg1);
2246 }
2247 }
2248 NeedTest = false;
2249 } else if (foldX86XALUIntrinsic(CC, I, Cond)) {
2250 // Fake request the condition, otherwise the intrinsic might be completely
2251 // optimized away.
2252 Register TmpReg = getRegForValue(Cond);
2253 if (!TmpReg)
2254 return false;
2255
2256 NeedTest = false;
2257 }
2258
2259 if (NeedTest) {
2260 // Selects operate on i1, however, CondReg is 8 bits width and may contain
2261 // garbage. Indeed, only the less significant bit is supposed to be
2262 // accurate. If we read more than the lsb, we may see non-zero values
2263 // whereas lsb is zero. Therefore, we have to truncate Op0Reg to i1 for
2264 // the select. This is achieved by performing TEST against 1.
2265 Register CondReg = getRegForValue(Cond);
2266 if (!CondReg)
2267 return false;
2268
2269 // In case OpReg is a K register, COPY to a GPR
2270 if (MRI.getRegClass(CondReg) == &X86::VK1RegClass) {
2271 Register KCondReg = CondReg;
2272 CondReg = createResultReg(&X86::GR32RegClass);
2273 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2274 TII.get(TargetOpcode::COPY), CondReg)
2275 .addReg(KCondReg);
2276 CondReg = fastEmitInst_extractsubreg(MVT::i8, CondReg, X86::sub_8bit);
2277 }
2278 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::TEST8ri))
2279 .addReg(CondReg)
2280 .addImm(1);
2281 }
2282
2283 const Value *LHS = I->getOperand(1);
2284 const Value *RHS = I->getOperand(2);
2285
2286 Register RHSReg = getRegForValue(RHS);
2287 Register LHSReg = getRegForValue(LHS);
2288 if (!LHSReg || !RHSReg)
2289 return false;
2290
2291 const TargetRegisterInfo &TRI = *Subtarget->getRegisterInfo();
2292 unsigned Opc = X86::getCMovOpcode(TRI.getRegSizeInBits(*RC) / 8, false,
2293 Subtarget->hasNDD());
2294 Register ResultReg = fastEmitInst_rri(Opc, RC, RHSReg, LHSReg, CC);
2295 updateValueMap(I, ResultReg);
2296 return true;
2297}
2298
2299/// Emit SSE or AVX instructions to lower the select.
2300///
2301/// Try to use SSE1/SSE2 instructions to simulate a select without branches.
2302/// This lowers fp selects into a CMP/AND/ANDN/OR sequence when the necessary
2303/// SSE instructions are available. If AVX is available, try to use a VBLENDV.
2304bool X86FastISel::X86FastEmitSSESelect(MVT RetVT, const Instruction *I) {
2305 // Optimize conditions coming from a compare if both instructions are in the
2306 // same basic block (values defined in other basic blocks may not have
2307 // initialized registers).
2308 const auto *CI = dyn_cast<FCmpInst>(I->getOperand(0));
2309 if (!CI || (CI->getParent() != I->getParent()))
2310 return false;
2311
2312 if (I->getType() != CI->getOperand(0)->getType() ||
2313 !((Subtarget->hasSSE1() && RetVT == MVT::f32) ||
2314 (Subtarget->hasSSE2() && RetVT == MVT::f64)))
2315 return false;
2316
2317 const Value *CmpLHS = CI->getOperand(0);
2318 const Value *CmpRHS = CI->getOperand(1);
2319 CmpInst::Predicate Predicate = optimizeCmpPredicate(CI);
2320
2321 // The optimizer might have replaced fcmp oeq %x, %x with fcmp ord %x, 0.0.
2322 // We don't have to materialize a zero constant for this case and can just use
2323 // %x again on the RHS.
2324 if (Predicate == CmpInst::FCMP_ORD || Predicate == CmpInst::FCMP_UNO) {
2325 const auto *CmpRHSC = dyn_cast<ConstantFP>(CmpRHS);
2326 if (CmpRHSC && CmpRHSC->isNullValue())
2327 CmpRHS = CmpLHS;
2328 }
2329
2330 unsigned CC;
2331 bool NeedSwap;
2332 std::tie(CC, NeedSwap) = getX86SSEConditionCode(Predicate);
2333 if (CC > 7 && !Subtarget->hasAVX())
2334 return false;
2335
2336 if (NeedSwap)
2337 std::swap(CmpLHS, CmpRHS);
2338
2339 const Value *LHS = I->getOperand(1);
2340 const Value *RHS = I->getOperand(2);
2341
2342 Register LHSReg = getRegForValue(LHS);
2343 Register RHSReg = getRegForValue(RHS);
2344 Register CmpLHSReg = getRegForValue(CmpLHS);
2345 Register CmpRHSReg = getRegForValue(CmpRHS);
2346 if (!LHSReg || !RHSReg || !CmpLHSReg || !CmpRHSReg)
2347 return false;
2348
2349 const TargetRegisterClass *RC = TLI.getRegClassFor(RetVT);
2350 Register ResultReg;
2351
2352 if (Subtarget->hasAVX512()) {
2353 // If we have AVX512 we can use a mask compare and masked movss/sd.
2354 const TargetRegisterClass *VR128X = &X86::VR128XRegClass;
2355 const TargetRegisterClass *VK1 = &X86::VK1RegClass;
2356
2357 unsigned CmpOpcode =
2358 (RetVT == MVT::f32) ? X86::VCMPSSZrri : X86::VCMPSDZrri;
2359 Register CmpReg = fastEmitInst_rri(CmpOpcode, VK1, CmpLHSReg, CmpRHSReg,
2360 CC);
2361
2362 // Need an IMPLICIT_DEF for the input that is used to generate the upper
2363 // bits of the result register since its not based on any of the inputs.
2364 Register ImplicitDefReg = createResultReg(VR128X);
2365 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2366 TII.get(TargetOpcode::IMPLICIT_DEF), ImplicitDefReg);
2367
2368 // Place RHSReg is the passthru of the masked movss/sd operation and put
2369 // LHS in the input. The mask input comes from the compare.
2370 unsigned MovOpcode =
2371 (RetVT == MVT::f32) ? X86::VMOVSSZrrk : X86::VMOVSDZrrk;
2372 Register MovReg = fastEmitInst_rrrr(MovOpcode, VR128X, RHSReg, CmpReg,
2373 ImplicitDefReg, LHSReg);
2374
2375 ResultReg = createResultReg(RC);
2376 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2377 TII.get(TargetOpcode::COPY), ResultReg).addReg(MovReg);
2378
2379 } else if (Subtarget->hasAVX()) {
2380 const TargetRegisterClass *VR128 = &X86::VR128RegClass;
2381
2382 // If we have AVX, create 1 blendv instead of 3 logic instructions.
2383 // Blendv was introduced with SSE 4.1, but the 2 register form implicitly
2384 // uses XMM0 as the selection register. That may need just as many
2385 // instructions as the AND/ANDN/OR sequence due to register moves, so
2386 // don't bother.
2387 unsigned CmpOpcode =
2388 (RetVT == MVT::f32) ? X86::VCMPSSrri : X86::VCMPSDrri;
2389 unsigned BlendOpcode =
2390 (RetVT == MVT::f32) ? X86::VBLENDVPSrrr : X86::VBLENDVPDrrr;
2391
2392 Register CmpReg = fastEmitInst_rri(CmpOpcode, RC, CmpLHSReg, CmpRHSReg,
2393 CC);
2394 Register VBlendReg = fastEmitInst_rrr(BlendOpcode, VR128, RHSReg, LHSReg,
2395 CmpReg);
2396 ResultReg = createResultReg(RC);
2397 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2398 TII.get(TargetOpcode::COPY), ResultReg).addReg(VBlendReg);
2399 } else {
2400 // Choose the SSE instruction sequence based on data type (float or double).
2401 static const uint16_t OpcTable[2][4] = {
2402 { X86::CMPSSrri, X86::ANDPSrr, X86::ANDNPSrr, X86::ORPSrr },
2403 { X86::CMPSDrri, X86::ANDPDrr, X86::ANDNPDrr, X86::ORPDrr }
2404 };
2405
2406 const uint16_t *Opc = nullptr;
2407 switch (RetVT.SimpleTy) {
2408 default: return false;
2409 case MVT::f32: Opc = &OpcTable[0][0]; break;
2410 case MVT::f64: Opc = &OpcTable[1][0]; break;
2411 }
2412
2413 const TargetRegisterClass *VR128 = &X86::VR128RegClass;
2414 Register CmpReg = fastEmitInst_rri(Opc[0], RC, CmpLHSReg, CmpRHSReg, CC);
2415 Register AndReg = fastEmitInst_rr(Opc[1], VR128, CmpReg, LHSReg);
2416 Register AndNReg = fastEmitInst_rr(Opc[2], VR128, CmpReg, RHSReg);
2417 Register OrReg = fastEmitInst_rr(Opc[3], VR128, AndNReg, AndReg);
2418 ResultReg = createResultReg(RC);
2419 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2420 TII.get(TargetOpcode::COPY), ResultReg).addReg(OrReg);
2421 }
2422 updateValueMap(I, ResultReg);
2423 return true;
2424}
2425
2426bool X86FastISel::X86FastEmitPseudoSelect(MVT RetVT, const Instruction *I) {
2427 // These are pseudo CMOV instructions and will be later expanded into control-
2428 // flow.
2429 unsigned Opc;
2430 switch (RetVT.SimpleTy) {
2431 default: return false;
2432 case MVT::i8: Opc = X86::CMOV_GR8; break;
2433 case MVT::i16: Opc = X86::CMOV_GR16; break;
2434 case MVT::i32: Opc = X86::CMOV_GR32; break;
2435 case MVT::f16:
2436 Opc = Subtarget->hasAVX512() ? X86::CMOV_FR16X : X86::CMOV_FR16; break;
2437 case MVT::f32:
2438 Opc = Subtarget->hasAVX512() ? X86::CMOV_FR32X : X86::CMOV_FR32; break;
2439 case MVT::f64:
2440 Opc = Subtarget->hasAVX512() ? X86::CMOV_FR64X : X86::CMOV_FR64; break;
2441 }
2442
2443 const Value *Cond = I->getOperand(0);
2445
2446 // Optimize conditions coming from a compare if both instructions are in the
2447 // same basic block (values defined in other basic blocks may not have
2448 // initialized registers).
2449 const auto *CI = dyn_cast<CmpInst>(Cond);
2450 if (CI && (CI->getParent() == I->getParent())) {
2451 bool NeedSwap;
2452 std::tie(CC, NeedSwap) = X86::getX86ConditionCode(CI->getPredicate());
2453 if (CC > X86::LAST_VALID_COND)
2454 return false;
2455
2456 const Value *CmpLHS = CI->getOperand(0);
2457 const Value *CmpRHS = CI->getOperand(1);
2458
2459 if (NeedSwap)
2460 std::swap(CmpLHS, CmpRHS);
2461
2462 EVT CmpVT = TLI.getValueType(DL, CmpLHS->getType());
2463 if (!X86FastEmitCompare(CmpLHS, CmpRHS, CmpVT, CI->getDebugLoc()))
2464 return false;
2465 } else {
2466 Register CondReg = getRegForValue(Cond);
2467 if (!CondReg)
2468 return false;
2469
2470 // In case OpReg is a K register, COPY to a GPR
2471 if (MRI.getRegClass(CondReg) == &X86::VK1RegClass) {
2472 Register KCondReg = CondReg;
2473 CondReg = createResultReg(&X86::GR32RegClass);
2474 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2475 TII.get(TargetOpcode::COPY), CondReg)
2476 .addReg(KCondReg);
2477 CondReg = fastEmitInst_extractsubreg(MVT::i8, CondReg, X86::sub_8bit);
2478 }
2479 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::TEST8ri))
2480 .addReg(CondReg)
2481 .addImm(1);
2482 }
2483
2484 const Value *LHS = I->getOperand(1);
2485 const Value *RHS = I->getOperand(2);
2486
2487 Register LHSReg = getRegForValue(LHS);
2488 Register RHSReg = getRegForValue(RHS);
2489 if (!LHSReg || !RHSReg)
2490 return false;
2491
2492 const TargetRegisterClass *RC = TLI.getRegClassFor(RetVT);
2493
2494 Register ResultReg =
2495 fastEmitInst_rri(Opc, RC, RHSReg, LHSReg, CC);
2496 updateValueMap(I, ResultReg);
2497 return true;
2498}
2499
2500bool X86FastISel::X86SelectSelect(const Instruction *I) {
2501 MVT RetVT;
2502 if (!isTypeLegal(I->getType(), RetVT))
2503 return false;
2504
2505 // Check if we can fold the select.
2506 if (const auto *CI = dyn_cast<CmpInst>(I->getOperand(0))) {
2507 CmpInst::Predicate Predicate = optimizeCmpPredicate(CI);
2508 const Value *Opnd = nullptr;
2509 switch (Predicate) {
2510 default: break;
2511 case CmpInst::FCMP_FALSE: Opnd = I->getOperand(2); break;
2512 case CmpInst::FCMP_TRUE: Opnd = I->getOperand(1); break;
2513 }
2514 // No need for a select anymore - this is an unconditional move.
2515 if (Opnd) {
2516 Register OpReg = getRegForValue(Opnd);
2517 if (!OpReg)
2518 return false;
2519 const TargetRegisterClass *RC = TLI.getRegClassFor(RetVT);
2520 Register ResultReg = createResultReg(RC);
2521 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2522 TII.get(TargetOpcode::COPY), ResultReg)
2523 .addReg(OpReg);
2524 updateValueMap(I, ResultReg);
2525 return true;
2526 }
2527 }
2528
2529 // First try to use real conditional move instructions.
2530 if (X86FastEmitCMoveSelect(RetVT, I))
2531 return true;
2532
2533 // Try to use a sequence of SSE instructions to simulate a conditional move.
2534 if (X86FastEmitSSESelect(RetVT, I))
2535 return true;
2536
2537 // Fall-back to pseudo conditional move instructions, which will be later
2538 // converted to control-flow.
2539 if (X86FastEmitPseudoSelect(RetVT, I))
2540 return true;
2541
2542 return false;
2543}
2544
2545// Common code for X86SelectSIToFP and X86SelectUIToFP.
2546bool X86FastISel::X86SelectIntToFP(const Instruction *I, bool IsSigned) {
2547 // The target-independent selection algorithm in FastISel already knows how
2548 // to select a SINT_TO_FP if the target is SSE but not AVX.
2549 // Early exit if the subtarget doesn't have AVX.
2550 // Unsigned conversion requires avx512.
2551 bool HasAVX512 = Subtarget->hasAVX512();
2552 if (!Subtarget->hasAVX() || (!IsSigned && !HasAVX512))
2553 return false;
2554
2555 // TODO: We could sign extend narrower types.
2556 EVT SrcVT = TLI.getValueType(DL, I->getOperand(0)->getType());
2557 if (SrcVT != MVT::i32 && SrcVT != MVT::i64)
2558 return false;
2559
2560 // Select integer to float/double conversion.
2561 Register OpReg = getRegForValue(I->getOperand(0));
2562 if (!OpReg)
2563 return false;
2564
2565 unsigned Opcode;
2566
2567 static const uint16_t SCvtOpc[2][2][2] = {
2568 { { X86::VCVTSI2SSrr, X86::VCVTSI642SSrr },
2569 { X86::VCVTSI2SDrr, X86::VCVTSI642SDrr } },
2570 { { X86::VCVTSI2SSZrr, X86::VCVTSI642SSZrr },
2571 { X86::VCVTSI2SDZrr, X86::VCVTSI642SDZrr } },
2572 };
2573 static const uint16_t UCvtOpc[2][2] = {
2574 { X86::VCVTUSI2SSZrr, X86::VCVTUSI642SSZrr },
2575 { X86::VCVTUSI2SDZrr, X86::VCVTUSI642SDZrr },
2576 };
2577 bool Is64Bit = SrcVT == MVT::i64;
2578
2579 if (I->getType()->isDoubleTy()) {
2580 // s/uitofp int -> double
2581 Opcode = IsSigned ? SCvtOpc[HasAVX512][1][Is64Bit] : UCvtOpc[1][Is64Bit];
2582 } else if (I->getType()->isFloatTy()) {
2583 // s/uitofp int -> float
2584 Opcode = IsSigned ? SCvtOpc[HasAVX512][0][Is64Bit] : UCvtOpc[0][Is64Bit];
2585 } else
2586 return false;
2587
2588 MVT DstVT = TLI.getValueType(DL, I->getType()).getSimpleVT();
2589 const TargetRegisterClass *RC = TLI.getRegClassFor(DstVT);
2590 Register ImplicitDefReg = createResultReg(RC);
2591 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2592 TII.get(TargetOpcode::IMPLICIT_DEF), ImplicitDefReg);
2593 Register ResultReg = fastEmitInst_rr(Opcode, RC, ImplicitDefReg, OpReg);
2594 updateValueMap(I, ResultReg);
2595 return true;
2596}
2597
2598bool X86FastISel::X86SelectSIToFP(const Instruction *I) {
2599 return X86SelectIntToFP(I, /*IsSigned*/true);
2600}
2601
2602bool X86FastISel::X86SelectUIToFP(const Instruction *I) {
2603 return X86SelectIntToFP(I, /*IsSigned*/false);
2604}
2605
2606// Helper method used by X86SelectFPExt and X86SelectFPTrunc.
2607bool X86FastISel::X86SelectFPExtOrFPTrunc(const Instruction *I,
2608 unsigned TargetOpc,
2609 const TargetRegisterClass *RC) {
2610 assert((I->getOpcode() == Instruction::FPExt ||
2611 I->getOpcode() == Instruction::FPTrunc) &&
2612 "Instruction must be an FPExt or FPTrunc!");
2613 bool HasAVX = Subtarget->hasAVX();
2614
2615 Register OpReg = getRegForValue(I->getOperand(0));
2616 if (!OpReg)
2617 return false;
2618
2619 Register ImplicitDefReg;
2620 if (HasAVX) {
2621 ImplicitDefReg = createResultReg(RC);
2622 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2623 TII.get(TargetOpcode::IMPLICIT_DEF), ImplicitDefReg);
2624
2625 }
2626
2627 Register ResultReg = createResultReg(RC);
2628 MachineInstrBuilder MIB;
2629 MIB = BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(TargetOpc),
2630 ResultReg);
2631
2632 if (HasAVX)
2633 MIB.addReg(ImplicitDefReg);
2634
2635 MIB.addReg(OpReg);
2636 updateValueMap(I, ResultReg);
2637 return true;
2638}
2639
2640bool X86FastISel::X86SelectFPExt(const Instruction *I) {
2641 if (Subtarget->hasSSE2() && I->getType()->isDoubleTy() &&
2642 I->getOperand(0)->getType()->isFloatTy()) {
2643 bool HasAVX512 = Subtarget->hasAVX512();
2644 // fpext from float to double.
2645 unsigned Opc =
2646 HasAVX512 ? X86::VCVTSS2SDZrr
2647 : Subtarget->hasAVX() ? X86::VCVTSS2SDrr : X86::CVTSS2SDrr;
2648 return X86SelectFPExtOrFPTrunc(I, Opc, TLI.getRegClassFor(MVT::f64));
2649 }
2650
2651 return false;
2652}
2653
2654bool X86FastISel::X86SelectFPTrunc(const Instruction *I) {
2655 if (Subtarget->hasSSE2() && I->getType()->isFloatTy() &&
2656 I->getOperand(0)->getType()->isDoubleTy()) {
2657 bool HasAVX512 = Subtarget->hasAVX512();
2658 // fptrunc from double to float.
2659 unsigned Opc =
2660 HasAVX512 ? X86::VCVTSD2SSZrr
2661 : Subtarget->hasAVX() ? X86::VCVTSD2SSrr : X86::CVTSD2SSrr;
2662 return X86SelectFPExtOrFPTrunc(I, Opc, TLI.getRegClassFor(MVT::f32));
2663 }
2664
2665 return false;
2666}
2667
2668bool X86FastISel::X86SelectTrunc(const Instruction *I) {
2669 EVT SrcVT = TLI.getValueType(DL, I->getOperand(0)->getType());
2670 EVT DstVT = TLI.getValueType(DL, I->getType());
2671
2672 // This code only handles truncation to byte.
2673 if (DstVT != MVT::i8 && DstVT != MVT::i1)
2674 return false;
2675 if (!TLI.isTypeLegal(SrcVT))
2676 return false;
2677
2678 Register InputReg = getRegForValue(I->getOperand(0));
2679 if (!InputReg)
2680 // Unhandled operand. Halt "fast" selection and bail.
2681 return false;
2682
2683 if (SrcVT == MVT::i8) {
2684 // Truncate from i8 to i1; no code needed.
2685 updateValueMap(I, InputReg);
2686 return true;
2687 }
2688
2689 // Issue an extract_subreg.
2690 Register ResultReg = fastEmitInst_extractsubreg(MVT::i8, InputReg,
2691 X86::sub_8bit);
2692 if (!ResultReg)
2693 return false;
2694
2695 updateValueMap(I, ResultReg);
2696 return true;
2697}
2698
2699bool X86FastISel::X86SelectBitCast(const Instruction *I) {
2700 // Select SSE2/AVX bitcasts between 128/256/512 bit vector types.
2701 MVT SrcVT, DstVT;
2702 if (!Subtarget->hasSSE2() ||
2703 !isTypeLegal(I->getOperand(0)->getType(), SrcVT) ||
2704 !isTypeLegal(I->getType(), DstVT))
2705 return false;
2706
2707 // Only allow vectors that use xmm/ymm/zmm.
2708 if (!SrcVT.isVector() || !DstVT.isVector() ||
2709 SrcVT.getVectorElementType() == MVT::i1 ||
2710 DstVT.getVectorElementType() == MVT::i1)
2711 return false;
2712
2713 Register Reg = getRegForValue(I->getOperand(0));
2714 if (!Reg)
2715 return false;
2716
2717 // Emit a reg-reg copy so we don't propagate cached known bits information
2718 // with the wrong VT if we fall out of fast isel after selecting this.
2719 const TargetRegisterClass *DstClass = TLI.getRegClassFor(DstVT);
2720 Register ResultReg = createResultReg(DstClass);
2721 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(TargetOpcode::COPY),
2722 ResultReg)
2723 .addReg(Reg);
2724
2725 updateValueMap(I, ResultReg);
2726 return true;
2727}
2728
2729bool X86FastISel::IsMemcpySmall(uint64_t Len) {
2730 return Len <= (Subtarget->is64Bit() ? 32 : 16);
2731}
2732
2733bool X86FastISel::TryEmitSmallMemcpy(X86AddressMode DestAM,
2734 X86AddressMode SrcAM, uint64_t Len) {
2735
2736 // Make sure we don't bloat code by inlining very large memcpy's.
2737 if (!IsMemcpySmall(Len))
2738 return false;
2739
2740 bool i64Legal = Subtarget->is64Bit();
2741
2742 // We don't care about alignment here since we just emit integer accesses.
2743 while (Len) {
2744 MVT VT;
2745 if (Len >= 8 && i64Legal)
2746 VT = MVT::i64;
2747 else if (Len >= 4)
2748 VT = MVT::i32;
2749 else if (Len >= 2)
2750 VT = MVT::i16;
2751 else
2752 VT = MVT::i8;
2753
2754 Register Reg;
2755 bool RV = X86FastEmitLoad(VT, SrcAM, nullptr, Reg);
2756 RV &= X86FastEmitStore(VT, Reg, DestAM);
2757 assert(RV && "Failed to emit load or store??");
2758 (void)RV;
2759
2760 unsigned Size = VT.getSizeInBits()/8;
2761 Len -= Size;
2762 DestAM.Disp += Size;
2763 SrcAM.Disp += Size;
2764 }
2765
2766 return true;
2767}
2768
2769Register X86FastISel::X86FastEmitAddSub_rr(unsigned BaseOpc, MVT VT,
2770 Register LHSReg, Register RHSReg) {
2771 static const uint16_t Opc[2][2][4] = {
2772 {{X86::ADD8rr, X86::ADD16rr, X86::ADD32rr, X86::ADD64rr},
2773 {X86::ADD8rr_ND, X86::ADD16rr_ND, X86::ADD32rr_ND, X86::ADD64rr_ND}},
2774 {{X86::SUB8rr, X86::SUB16rr, X86::SUB32rr, X86::SUB64rr},
2775 {X86::SUB8rr_ND, X86::SUB16rr_ND, X86::SUB32rr_ND, X86::SUB64rr_ND}}};
2776
2777 bool IsSub = BaseOpc == ISD::SUB;
2778 unsigned TypeIdx = VT.SimpleTy - MVT::i8;
2779 return X86FastEmitLiveEFLAGS_rr(Opc[IsSub][Subtarget->hasNDD()][TypeIdx],
2780 TLI.getRegClassFor(VT), LHSReg, RHSReg);
2781}
2782
2783Register X86FastISel::X86FastEmitAddSub_ri(unsigned BaseOpc, MVT VT,
2784 Register LHSReg,
2785 const ConstantInt *CI,
2786 unsigned CondCode) {
2787 bool IsSub = BaseOpc == ISD::SUB;
2788 unsigned TypeIdx = VT.SimpleTy - MVT::i8;
2789
2790 if (CI->isOne() && CondCode == X86::COND_O) {
2791 // We can use INC/DEC.
2792 static const uint16_t IncDecOpc[2][4] = {
2793 {X86::INC8r, X86::INC16r, X86::INC32r, X86::INC64r},
2794 {X86::DEC8r, X86::DEC16r, X86::DEC32r, X86::DEC64r}};
2795
2796 Register ResultReg = createResultReg(TLI.getRegClassFor(VT));
2797 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2798 TII.get(IncDecOpc[IsSub][TypeIdx]), ResultReg)
2799 .addReg(LHSReg);
2800 return ResultReg;
2801 }
2802
2803 if (VT == MVT::i64 && !isInt<32>(CI->getSExtValue()))
2804 return Register();
2805
2806 static const uint16_t Opc[2][2][4] = {
2807 {{X86::ADD8ri, X86::ADD16ri, X86::ADD32ri, X86::ADD64ri32},
2808 {X86::ADD8ri_ND, X86::ADD16ri_ND, X86::ADD32ri_ND, X86::ADD64ri32_ND}},
2809 {{X86::SUB8ri, X86::SUB16ri, X86::SUB32ri, X86::SUB64ri32},
2810 {X86::SUB8ri_ND, X86::SUB16ri_ND, X86::SUB32ri_ND, X86::SUB64ri32_ND}}};
2811
2812 return X86FastEmitLiveEFLAGS_ri(Opc[IsSub][Subtarget->hasNDD()][TypeIdx],
2813 TLI.getRegClassFor(VT), LHSReg,
2814 CI->getZExtValue());
2815}
2816
2817bool X86FastISel::fastLowerIntrinsicCall(const IntrinsicInst *II) {
2818 // FIXME: Handle more intrinsics.
2819 switch (II->getIntrinsicID()) {
2820 default:
2821 return false;
2822 case Intrinsic::frameaddress: {
2823 MachineFunction *MF = FuncInfo.MF;
2825 return false;
2826
2827 Type *RetTy = II->getCalledFunction()->getReturnType();
2828
2829 MVT VT;
2830 if (!isTypeLegal(RetTy, VT))
2831 return false;
2832
2833 unsigned Opc;
2834 const TargetRegisterClass *RC = nullptr;
2835
2836 switch (VT.SimpleTy) {
2837 default: llvm_unreachable("Invalid result type for frameaddress.");
2838 case MVT::i32: Opc = X86::MOV32rm; RC = &X86::GR32RegClass; break;
2839 case MVT::i64: Opc = X86::MOV64rm; RC = &X86::GR64RegClass; break;
2840 }
2841
2842 // This needs to be set before we call getPtrSizedFrameRegister, otherwise
2843 // we get the wrong frame register.
2844 MachineFrameInfo &MFI = MF->getFrameInfo();
2845 MFI.setFrameAddressIsTaken(true);
2846
2847 const X86RegisterInfo *RegInfo = Subtarget->getRegisterInfo();
2848 Register FrameReg = RegInfo->getPtrSizedFrameRegister(*MF);
2849 assert(((FrameReg == X86::RBP && VT == MVT::i64) ||
2850 (FrameReg == X86::EBP && VT == MVT::i32)) &&
2851 "Invalid Frame Register!");
2852
2853 // Always make a copy of the frame register to a vreg first, so that we
2854 // never directly reference the frame register (the TwoAddressInstruction-
2855 // Pass doesn't like that).
2856 Register SrcReg = createResultReg(RC);
2857 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2858 TII.get(TargetOpcode::COPY), SrcReg).addReg(FrameReg);
2859
2860 // Now recursively load from the frame address.
2861 // movq (%rbp), %rax
2862 // movq (%rax), %rax
2863 // movq (%rax), %rax
2864 // ...
2865 unsigned Depth = cast<ConstantInt>(II->getOperand(0))->getZExtValue();
2866 while (Depth--) {
2867 Register DestReg = createResultReg(RC);
2868 addDirectMem(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2869 TII.get(Opc), DestReg), SrcReg);
2870 SrcReg = DestReg;
2871 }
2872
2873 updateValueMap(II, SrcReg);
2874 return true;
2875 }
2876 case Intrinsic::memcpy: {
2877 const MemCpyInst *MCI = cast<MemCpyInst>(II);
2878 // Don't handle volatile or variable length memcpys.
2879 if (MCI->isVolatile())
2880 return false;
2881
2882 if (isa<ConstantInt>(MCI->getLength())) {
2883 // Small memcpy's are common enough that we want to do them
2884 // without a call if possible.
2885 uint64_t Len = cast<ConstantInt>(MCI->getLength())->getZExtValue();
2886 if (IsMemcpySmall(Len)) {
2887 X86AddressMode DestAM, SrcAM;
2888 if (!X86SelectAddress(MCI->getRawDest(), DestAM) ||
2889 !X86SelectAddress(MCI->getRawSource(), SrcAM))
2890 return false;
2891 TryEmitSmallMemcpy(DestAM, SrcAM, Len);
2892 return true;
2893 }
2894 }
2895
2896 unsigned SizeWidth = Subtarget->is64Bit() ? 64 : 32;
2897 if (!MCI->getLength()->getType()->isIntegerTy(SizeWidth))
2898 return false;
2899
2900 if (MCI->getSourceAddressSpace() > 255 || MCI->getDestAddressSpace() > 255)
2901 return false;
2902
2903 return lowerCallTo(II, "memcpy", II->arg_size() - 1);
2904 }
2905 case Intrinsic::memset: {
2906 const MemSetInst *MSI = cast<MemSetInst>(II);
2907
2908 if (MSI->isVolatile())
2909 return false;
2910
2911 unsigned SizeWidth = Subtarget->is64Bit() ? 64 : 32;
2912 if (!MSI->getLength()->getType()->isIntegerTy(SizeWidth))
2913 return false;
2914
2915 if (MSI->getDestAddressSpace() > 255)
2916 return false;
2917
2918 return lowerCallTo(II, "memset", II->arg_size() - 1);
2919 }
2920 case Intrinsic::stackprotector: {
2921 // Emit code to store the stack guard onto the stack.
2922 EVT PtrTy = TLI.getPointerTy(DL);
2923
2924 const Value *Op1 = II->getArgOperand(0); // The guard's value.
2925 const AllocaInst *Slot = cast<AllocaInst>(II->getArgOperand(1));
2926
2927 MFI.setStackProtectorIndex(FuncInfo.StaticAllocaMap[Slot]);
2928
2929 // Grab the frame index.
2930 X86AddressMode AM;
2931 if (!X86SelectAddress(Slot, AM)) return false;
2932 if (!X86FastEmitStore(PtrTy, Op1, AM)) return false;
2933 return true;
2934 }
2935 case Intrinsic::dbg_declare: {
2936 const DbgDeclareInst *DI = cast<DbgDeclareInst>(II);
2937 X86AddressMode AM;
2938 assert(DI->getAddress() && "Null address should be checked earlier!");
2939 if (!X86SelectAddress(DI->getAddress(), AM))
2940 return false;
2941 const MCInstrDesc &II = TII.get(TargetOpcode::DBG_VALUE);
2942 assert(DI->getVariable()->isValidLocationForIntrinsic(MIMD.getDL()) &&
2943 "Expected inlined-at fields to agree");
2944 addFullAddress(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, II), AM)
2945 .addImm(0)
2946 .addMetadata(DI->getVariable())
2947 .addMetadata(DI->getExpression());
2948 return true;
2949 }
2950 case Intrinsic::trap: {
2951 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::TRAP));
2952 return true;
2953 }
2954 case Intrinsic::sqrt: {
2955 if (!Subtarget->hasSSE1())
2956 return false;
2957
2958 Type *RetTy = II->getCalledFunction()->getReturnType();
2959
2960 MVT VT;
2961 if (!isTypeLegal(RetTy, VT))
2962 return false;
2963
2964 // Unfortunately we can't use fastEmit_r, because the AVX version of FSQRT
2965 // is not generated by FastISel yet.
2966 // FIXME: Update this code once tablegen can handle it.
2967 static const uint16_t SqrtOpc[3][2] = {
2968 { X86::SQRTSSr, X86::SQRTSDr },
2969 { X86::VSQRTSSr, X86::VSQRTSDr },
2970 { X86::VSQRTSSZr, X86::VSQRTSDZr },
2971 };
2972 unsigned AVXLevel = Subtarget->hasAVX512() ? 2 :
2973 Subtarget->hasAVX() ? 1 :
2974 0;
2975 unsigned Opc;
2976 switch (VT.SimpleTy) {
2977 default: return false;
2978 case MVT::f32: Opc = SqrtOpc[AVXLevel][0]; break;
2979 case MVT::f64: Opc = SqrtOpc[AVXLevel][1]; break;
2980 }
2981
2982 const Value *SrcVal = II->getArgOperand(0);
2983 Register SrcReg = getRegForValue(SrcVal);
2984
2985 if (!SrcReg)
2986 return false;
2987
2988 const TargetRegisterClass *RC = TLI.getRegClassFor(VT);
2989 Register ImplicitDefReg;
2990 if (AVXLevel > 0) {
2991 ImplicitDefReg = createResultReg(RC);
2992 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
2993 TII.get(TargetOpcode::IMPLICIT_DEF), ImplicitDefReg);
2994 }
2995
2996 Register ResultReg = createResultReg(RC);
2997 MachineInstrBuilder MIB;
2998 MIB = BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Opc),
2999 ResultReg);
3000
3001 if (ImplicitDefReg)
3002 MIB.addReg(ImplicitDefReg);
3003
3004 MIB.addReg(SrcReg);
3005
3006 updateValueMap(II, ResultReg);
3007 return true;
3008 }
3009 case Intrinsic::sadd_with_overflow:
3010 case Intrinsic::uadd_with_overflow:
3011 case Intrinsic::ssub_with_overflow:
3012 case Intrinsic::usub_with_overflow:
3013 case Intrinsic::smul_with_overflow:
3014 case Intrinsic::umul_with_overflow: {
3015 // This implements the basic lowering of the xalu with overflow intrinsics
3016 // into add/sub/mul followed by either seto or setb.
3017 const Function *Callee = II->getCalledFunction();
3018 auto *Ty = cast<StructType>(Callee->getReturnType());
3019 Type *RetTy = Ty->getTypeAtIndex(0U);
3020 assert(Ty->getTypeAtIndex(1)->isIntegerTy() &&
3021 Ty->getTypeAtIndex(1)->getScalarSizeInBits() == 1 &&
3022 "Overflow value expected to be an i1");
3023
3024 MVT VT;
3025 if (!isTypeLegal(RetTy, VT))
3026 return false;
3027
3028 if (VT < MVT::i8 || VT > MVT::i64)
3029 return false;
3030
3031 const Value *LHS = II->getArgOperand(0);
3032 const Value *RHS = II->getArgOperand(1);
3033
3034 // Canonicalize immediate to the RHS.
3035 if (isa<ConstantInt>(LHS) && !isa<ConstantInt>(RHS) && II->isCommutative())
3036 std::swap(LHS, RHS);
3037
3038 unsigned BaseOpc, CondCode;
3039 switch (II->getIntrinsicID()) {
3040 default: llvm_unreachable("Unexpected intrinsic!");
3041 case Intrinsic::sadd_with_overflow:
3042 BaseOpc = ISD::ADD; CondCode = X86::COND_O; break;
3043 case Intrinsic::uadd_with_overflow:
3044 BaseOpc = ISD::ADD; CondCode = X86::COND_B; break;
3045 case Intrinsic::ssub_with_overflow:
3046 BaseOpc = ISD::SUB; CondCode = X86::COND_O; break;
3047 case Intrinsic::usub_with_overflow:
3048 BaseOpc = ISD::SUB; CondCode = X86::COND_B; break;
3049 case Intrinsic::smul_with_overflow:
3050 BaseOpc = X86ISD::SMUL; CondCode = X86::COND_B; break;
3051 case Intrinsic::umul_with_overflow:
3052 BaseOpc = X86ISD::UMUL; CondCode = X86::COND_B; break;
3053 }
3054
3055 Register LHSReg = getRegForValue(LHS);
3056 if (!LHSReg)
3057 return false;
3058
3059 bool IsAddSub = BaseOpc == ISD::ADD || BaseOpc == ISD::SUB;
3060
3061 Register ResultReg;
3062 // Check if we have an immediate version.
3063 if (const auto *CI = dyn_cast<ConstantInt>(RHS); CI && IsAddSub)
3064 ResultReg = X86FastEmitAddSub_ri(BaseOpc, VT, LHSReg, CI, CondCode);
3065
3066 Register RHSReg;
3067 if (!ResultReg) {
3068 RHSReg = getRegForValue(RHS);
3069 if (!RHSReg)
3070 return false;
3071
3072 if (IsAddSub)
3073 ResultReg = X86FastEmitAddSub_rr(BaseOpc, VT, LHSReg, RHSReg);
3074 }
3075
3076 if (BaseOpc == X86ISD::UMUL) {
3077 static const uint16_t MULOpc[] =
3078 { X86::MUL8r, X86::MUL16r, X86::MUL32r, X86::MUL64r };
3079 static const MCPhysReg Reg[] = { X86::AL, X86::AX, X86::EAX, X86::RAX };
3080 // The first operand goes in RAX, which is an implicit input to the
3081 // X86::MUL*r instruction.
3082 ResultReg = X86FastEmitMul(MULOpc[VT.SimpleTy - MVT::i8], VT,
3083 Reg[VT.SimpleTy - MVT::i8], LHSReg, RHSReg,
3084 /*LiveEFLAGS=*/true);
3085 } else if (BaseOpc == X86ISD::SMUL) {
3086 static const uint16_t MULOpc[] =
3087 { X86::IMUL8r, X86::IMUL16rr, X86::IMUL32rr, X86::IMUL64rr };
3088 if (VT == MVT::i8) {
3089 // The first operand goes in AL, which is an implicit input to the
3090 // X86::IMUL8r instruction.
3091 ResultReg = X86FastEmitMul(MULOpc[0], VT, X86::AL, LHSReg, RHSReg,
3092 /*LiveEFLAGS=*/true);
3093 } else {
3094 ResultReg =
3095 X86FastEmitLiveEFLAGS_rr(MULOpc[VT.SimpleTy - MVT::i8],
3096 TLI.getRegClassFor(VT), LHSReg, RHSReg);
3097 }
3098 }
3099
3100 if (!ResultReg)
3101 return false;
3102
3103 // Assign to a GPR since the overflow return value is lowered to a SETcc.
3104 Register ResultReg2 = createResultReg(&X86::GR8RegClass);
3105 assert((ResultReg+1) == ResultReg2 && "Nonconsecutive result registers.");
3106 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(GET_SETCC),
3107 ResultReg2)
3108 .addImm(CondCode);
3109
3110 updateValueMap(II, ResultReg, 2);
3111 return true;
3112 }
3113 case Intrinsic::x86_sse_cvttss2si:
3114 case Intrinsic::x86_sse_cvttss2si64:
3115 case Intrinsic::x86_sse2_cvttsd2si:
3116 case Intrinsic::x86_sse2_cvttsd2si64: {
3117 bool IsInputDouble;
3118 switch (II->getIntrinsicID()) {
3119 default: llvm_unreachable("Unexpected intrinsic.");
3120 case Intrinsic::x86_sse_cvttss2si:
3121 case Intrinsic::x86_sse_cvttss2si64:
3122 if (!Subtarget->hasSSE1())
3123 return false;
3124 IsInputDouble = false;
3125 break;
3126 case Intrinsic::x86_sse2_cvttsd2si:
3127 case Intrinsic::x86_sse2_cvttsd2si64:
3128 if (!Subtarget->hasSSE2())
3129 return false;
3130 IsInputDouble = true;
3131 break;
3132 }
3133
3134 Type *RetTy = II->getCalledFunction()->getReturnType();
3135 MVT VT;
3136 if (!isTypeLegal(RetTy, VT))
3137 return false;
3138
3139 static const uint16_t CvtOpc[3][2][2] = {
3140 { { X86::CVTTSS2SIrr, X86::CVTTSS2SI64rr },
3141 { X86::CVTTSD2SIrr, X86::CVTTSD2SI64rr } },
3142 { { X86::VCVTTSS2SIrr, X86::VCVTTSS2SI64rr },
3143 { X86::VCVTTSD2SIrr, X86::VCVTTSD2SI64rr } },
3144 { { X86::VCVTTSS2SIZrr, X86::VCVTTSS2SI64Zrr },
3145 { X86::VCVTTSD2SIZrr, X86::VCVTTSD2SI64Zrr } },
3146 };
3147 unsigned AVXLevel = Subtarget->hasAVX512() ? 2 :
3148 Subtarget->hasAVX() ? 1 :
3149 0;
3150 unsigned Opc;
3151 switch (VT.SimpleTy) {
3152 default: llvm_unreachable("Unexpected result type.");
3153 case MVT::i32: Opc = CvtOpc[AVXLevel][IsInputDouble][0]; break;
3154 case MVT::i64: Opc = CvtOpc[AVXLevel][IsInputDouble][1]; break;
3155 }
3156
3157 // Check if we can fold insertelement instructions into the convert.
3158 const Value *Op = II->getArgOperand(0);
3159 while (auto *IE = dyn_cast<InsertElementInst>(Op)) {
3160 const Value *Index = IE->getOperand(2);
3161 if (!isa<ConstantInt>(Index))
3162 break;
3163 unsigned Idx = cast<ConstantInt>(Index)->getZExtValue();
3164
3165 if (!Idx) {
3166 Op = IE->getOperand(1);
3167 break;
3168 }
3169 Op = IE->getOperand(0);
3170 }
3171
3172 Register Reg = getRegForValue(Op);
3173 if (!Reg)
3174 return false;
3175
3176 Register ResultReg = createResultReg(TLI.getRegClassFor(VT));
3177 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Opc), ResultReg)
3178 .addReg(Reg);
3179
3180 updateValueMap(II, ResultReg);
3181 return true;
3182 }
3183 case Intrinsic::x86_sse42_crc32_32_8:
3184 case Intrinsic::x86_sse42_crc32_32_16:
3185 case Intrinsic::x86_sse42_crc32_32_32:
3186 case Intrinsic::x86_sse42_crc32_64_64: {
3187 if (!Subtarget->hasCRC32())
3188 return false;
3189
3190 Type *RetTy = II->getCalledFunction()->getReturnType();
3191
3192 MVT VT;
3193 if (!isTypeLegal(RetTy, VT))
3194 return false;
3195
3196 unsigned Opc;
3197 const TargetRegisterClass *RC = nullptr;
3198
3199 switch (II->getIntrinsicID()) {
3200 default:
3201 llvm_unreachable("Unexpected intrinsic.");
3202#define GET_EGPR_IF_ENABLED(OPC) Subtarget->hasEGPR() ? OPC##_EVEX : OPC
3203 case Intrinsic::x86_sse42_crc32_32_8:
3204 Opc = GET_EGPR_IF_ENABLED(X86::CRC32r32r8);
3205 RC = &X86::GR32RegClass;
3206 break;
3207 case Intrinsic::x86_sse42_crc32_32_16:
3208 Opc = GET_EGPR_IF_ENABLED(X86::CRC32r32r16);
3209 RC = &X86::GR32RegClass;
3210 break;
3211 case Intrinsic::x86_sse42_crc32_32_32:
3212 Opc = GET_EGPR_IF_ENABLED(X86::CRC32r32r32);
3213 RC = &X86::GR32RegClass;
3214 break;
3215 case Intrinsic::x86_sse42_crc32_64_64:
3216 Opc = GET_EGPR_IF_ENABLED(X86::CRC32r64r64);
3217 RC = &X86::GR64RegClass;
3218 break;
3219#undef GET_EGPR_IF_ENABLED
3220 }
3221
3222 const Value *LHS = II->getArgOperand(0);
3223 const Value *RHS = II->getArgOperand(1);
3224
3225 Register LHSReg = getRegForValue(LHS);
3226 Register RHSReg = getRegForValue(RHS);
3227 if (!LHSReg || !RHSReg)
3228 return false;
3229
3230 Register ResultReg = fastEmitInst_rr(Opc, RC, LHSReg, RHSReg);
3231 if (!ResultReg)
3232 return false;
3233
3234 updateValueMap(II, ResultReg);
3235 return true;
3236 }
3237 }
3238}
3239
3240bool X86FastISel::fastLowerArguments() {
3241 if (!FuncInfo.CanLowerReturn)
3242 return false;
3243
3244 const Function *F = FuncInfo.Fn;
3245 if (F->isVarArg())
3246 return false;
3247
3248 CallingConv::ID CC = F->getCallingConv();
3249 if (CC != CallingConv::C)
3250 return false;
3251
3252 if (Subtarget->isCallingConvWin64(CC))
3253 return false;
3254
3255 if (!Subtarget->is64Bit())
3256 return false;
3257
3258 if (Subtarget->useSoftFloat())
3259 return false;
3260
3261 // Only handle simple cases. i.e. Up to 6 i32/i64 scalar arguments.
3262 unsigned GPRCnt = 0;
3263 unsigned FPRCnt = 0;
3264 for (auto const &Arg : F->args()) {
3265 if (Arg.hasAttribute(Attribute::ByVal) ||
3266 Arg.hasAttribute(Attribute::InReg) ||
3267 Arg.hasAttribute(Attribute::StructRet) ||
3268 Arg.hasAttribute(Attribute::SwiftSelf) ||
3269 Arg.hasAttribute(Attribute::SwiftAsync) ||
3270 Arg.hasAttribute(Attribute::SwiftError) ||
3271 Arg.hasAttribute(Attribute::Nest))
3272 return false;
3273
3274 Type *ArgTy = Arg.getType();
3275 if (ArgTy->isStructTy() || ArgTy->isArrayTy() || ArgTy->isVectorTy())
3276 return false;
3277
3278 EVT ArgVT = TLI.getValueType(DL, ArgTy);
3279 if (!ArgVT.isSimple()) return false;
3280 switch (ArgVT.getSimpleVT().SimpleTy) {
3281 default: return false;
3282 case MVT::i32:
3283 case MVT::i64:
3284 ++GPRCnt;
3285 break;
3286 case MVT::f32:
3287 case MVT::f64:
3288 if (!Subtarget->hasSSE1())
3289 return false;
3290 ++FPRCnt;
3291 break;
3292 }
3293
3294 if (GPRCnt > 6)
3295 return false;
3296
3297 if (FPRCnt > 8)
3298 return false;
3299 }
3300
3301 static const MCPhysReg GPR32ArgRegs[] = {
3302 X86::EDI, X86::ESI, X86::EDX, X86::ECX, X86::R8D, X86::R9D
3303 };
3304 static const MCPhysReg GPR64ArgRegs[] = {
3305 X86::RDI, X86::RSI, X86::RDX, X86::RCX, X86::R8 , X86::R9
3306 };
3307 static const MCPhysReg XMMArgRegs[] = {
3308 X86::XMM0, X86::XMM1, X86::XMM2, X86::XMM3,
3309 X86::XMM4, X86::XMM5, X86::XMM6, X86::XMM7
3310 };
3311
3312 unsigned GPRIdx = 0;
3313 unsigned FPRIdx = 0;
3314 for (auto const &Arg : F->args()) {
3315 MVT VT = TLI.getSimpleValueType(DL, Arg.getType());
3316 const TargetRegisterClass *RC = TLI.getRegClassFor(VT);
3317 MCRegister SrcReg;
3318 switch (VT.SimpleTy) {
3319 default: llvm_unreachable("Unexpected value type.");
3320 case MVT::i32: SrcReg = GPR32ArgRegs[GPRIdx++]; break;
3321 case MVT::i64: SrcReg = GPR64ArgRegs[GPRIdx++]; break;
3322 case MVT::f32: [[fallthrough]];
3323 case MVT::f64: SrcReg = XMMArgRegs[FPRIdx++]; break;
3324 }
3325 Register DstReg = FuncInfo.MF->addLiveIn(SrcReg, RC);
3326 // FIXME: Unfortunately it's necessary to emit a copy from the livein copy.
3327 // Without this, EmitLiveInCopies may eliminate the livein if its only
3328 // use is a bitcast (which isn't turned into an instruction).
3329 Register ResultReg = createResultReg(RC);
3330 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
3331 TII.get(TargetOpcode::COPY), ResultReg)
3332 .addReg(DstReg, getKillRegState(true));
3333 updateValueMap(&Arg, ResultReg);
3334 }
3335 return true;
3336}
3337
3338static unsigned computeBytesPoppedByCalleeForSRet(const X86Subtarget *Subtarget,
3339 CallingConv::ID CC,
3340 const CallBase *CB) {
3341 if (Subtarget->is64Bit())
3342 return 0;
3343 if (Subtarget->getTargetTriple().isOSMSVCRT())
3344 return 0;
3345 if (CC == CallingConv::Fast || CC == CallingConv::GHC ||
3346 CC == CallingConv::HiPE || CC == CallingConv::Tail ||
3348 return 0;
3349
3350 if (CB)
3351 if (CB->arg_empty() || !CB->paramHasAttr(0, Attribute::StructRet) ||
3352 CB->paramHasAttr(0, Attribute::InReg) || Subtarget->isTargetMCU())
3353 return 0;
3354
3355 return 4;
3356}
3357
3358bool X86FastISel::fastLowerCall(CallLoweringInfo &CLI) {
3359 auto &OutVals = CLI.OutVals;
3360 auto &OutFlags = CLI.OutFlags;
3361 auto &OutRegs = CLI.OutRegs;
3362 auto &Ins = CLI.Ins;
3363 auto &InRegs = CLI.InRegs;
3364 CallingConv::ID CC = CLI.CallConv;
3365 bool &IsTailCall = CLI.IsTailCall;
3366 bool IsVarArg = CLI.IsVarArg;
3367 const Value *Callee = CLI.Callee;
3368 MCSymbol *Symbol = CLI.Symbol;
3369 const auto *CB = CLI.CB;
3370
3371 bool Is64Bit = Subtarget->is64Bit();
3372 bool IsWin64 = Subtarget->isCallingConvWin64(CC);
3373
3374 // If the return type is illegal, check if the ABI requires a type conversion
3375 // that FastISel cannot handle. Fall back to DAG ISel in such cases.
3376 // For example, bfloat is returned as f16 in XMM0, however FastISel would
3377 // assign f32 register type and store it in FuncInfo.ValueMap. This would
3378 // cause DAG incorrectly perform type conversion from f32 to bfloat after get
3379 // the value from FuncInfo.ValueMap.
3380 // However, i1 is promoted to i8 and return i8 defined by ABI, so FastISel can
3381 // lower it without switching to DAGISel.
3382 SmallVector<Type *> RetTys;
3383 ComputeValueTypes(DL, CLI.RetTy, RetTys);
3384 for (Type *RetTy : RetTys) {
3385 MVT RetVT = MVT::Other;
3386 if (!isTypeLegal(RetTy, RetVT)) {
3387 if (RetVT == MVT::Other)
3388 return false; // Unknown type, let DAG ISel handle it.
3389
3390 // RetVT is not MVT::Other, it must be simple now. It is something rely on
3391 // the logic of isTypeLegal().
3392 MVT ABIVT = TLI.getRegisterTypeForCallingConv(CLI.RetTy->getContext(),
3393 CLI.CallConv, RetVT);
3394 MVT RegVT = TLI.getRegisterType(CLI.RetTy->getContext(), RetVT);
3395 if (ABIVT != RegVT)
3396 return false;
3397 }
3398 }
3399
3400 // Call / invoke instructions with NoCfCheck attribute require special
3401 // handling.
3402 if (CB && CB->doesNoCfCheck())
3403 return false;
3404
3405 // Functions with no_caller_saved_registers that need special handling.
3406 if ((CB && isa<CallInst>(CB) && CB->hasFnAttr("no_caller_saved_registers")))
3407 return false;
3408
3409 // Functions with no_callee_saved_registers that need special handling.
3410 if ((CB && CB->hasFnAttr("no_callee_saved_registers")))
3411 return false;
3412
3413 // Indirect calls with CFI checks need special handling.
3414 if (CB && CB->isIndirectCall() && CB->getOperandBundle(LLVMContext::OB_kcfi))
3415 return false;
3416
3417 // Functions using thunks for indirect calls need to use SDISel.
3418 if (Subtarget->useIndirectThunkCalls())
3419 return false;
3420
3421 // Handle only C and fastcc calling conventions for now.
3422 switch (CC) {
3423 default: return false;
3424 case CallingConv::C:
3425 case CallingConv::Fast:
3426 case CallingConv::Tail:
3427 case CallingConv::Swift:
3428 case CallingConv::SwiftTail:
3429 case CallingConv::X86_FastCall:
3430 case CallingConv::X86_StdCall:
3431 case CallingConv::X86_ThisCall:
3432 case CallingConv::Win64:
3433 case CallingConv::X86_64_SysV:
3434 case CallingConv::CFGuard_Check:
3435 break;
3436 }
3437
3438 // Allow SelectionDAG isel to handle tail calls.
3439 if (IsTailCall)
3440 return false;
3441
3442 // fastcc with -tailcallopt is intended to provide a guaranteed
3443 // tail call optimization. Fastisel doesn't know how to do that.
3444 if ((CC == CallingConv::Fast && TM.Options.GuaranteedTailCallOpt) ||
3445 CC == CallingConv::Tail || CC == CallingConv::SwiftTail)
3446 return false;
3447
3448 // Don't know how to handle Win64 varargs yet. Nothing special needed for
3449 // x86-32. Special handling for x86-64 is implemented.
3450 if (IsVarArg && IsWin64)
3451 return false;
3452
3453 // Don't know about inalloca yet.
3454 if (CLI.CB && CLI.CB->hasInAllocaArgument())
3455 return false;
3456
3457 for (auto Flag : CLI.OutFlags)
3458 if (Flag.isSwiftError() || Flag.isPreallocated())
3459 return false;
3460
3461 // Can't handle import call optimization.
3462 if (Is64Bit &&
3463 MF->getFunction().getParent()->getModuleFlag("import-call-optimization"))
3464 return false;
3465
3466 SmallVector<MVT, 16> OutVTs;
3468 SmallVector<Register, 16> ArgRegs;
3469
3470 // If this is a constant i1/i8/i16 argument, promote to i32 to avoid an extra
3471 // instruction. This is safe because it is common to all FastISel supported
3472 // calling conventions on x86.
3473 for (int i = 0, e = OutVals.size(); i != e; ++i) {
3474 Value *&Val = OutVals[i];
3475 ISD::ArgFlagsTy Flags = OutFlags[i];
3476 if (auto *CI = dyn_cast<ConstantInt>(Val)) {
3477 if (CI->getBitWidth() < 32) {
3478 if (Flags.isSExt())
3479 Val = ConstantInt::get(CI->getContext(), CI->getValue().sext(32));
3480 else
3481 Val = ConstantInt::get(CI->getContext(), CI->getValue().zext(32));
3482 }
3483 }
3484
3485 // Passing bools around ends up doing a trunc to i1 and passing it.
3486 // Codegen this as an argument + "and 1".
3487 MVT VT;
3488 auto *TI = dyn_cast<TruncInst>(Val);
3489 Register ResultReg;
3490 if (TI && TI->getType()->isIntegerTy(1) && CLI.CB &&
3491 (TI->getParent() == CLI.CB->getParent()) && TI->hasOneUse()) {
3492 Value *PrevVal = TI->getOperand(0);
3493 ResultReg = getRegForValue(PrevVal);
3494
3495 if (!ResultReg)
3496 return false;
3497
3498 if (!isTypeLegal(PrevVal->getType(), VT))
3499 return false;
3500
3501 ResultReg = fastEmit_ri(VT, VT, ISD::AND, ResultReg, 1);
3502 } else {
3503 if (!isTypeLegal(Val->getType(), VT) || VT.isVectorOf(MVT::i1))
3504 return false;
3505 ResultReg = getRegForValue(Val);
3506 }
3507
3508 if (!ResultReg)
3509 return false;
3510
3511 ArgRegs.push_back(ResultReg);
3512 OutVTs.push_back(VT);
3513 ArgTys.push_back(Val->getType());
3514 }
3515
3516 // Analyze operands of the call, assigning locations to each operand.
3518 CCState CCInfo(CC, IsVarArg, *FuncInfo.MF, ArgLocs, CLI.RetTy->getContext());
3519
3520 // Allocate shadow area for Win64
3521 if (IsWin64)
3522 CCInfo.AllocateStack(32, Align(8));
3523
3524 CCInfo.AnalyzeCallOperands(OutVTs, OutFlags, ArgTys, CC_X86);
3525
3526 // Get a count of how many bytes are to be pushed on the stack.
3527 unsigned NumBytes = CCInfo.getAlignedCallFrameSize();
3528
3529 // Issue CALLSEQ_START
3530 unsigned AdjStackDown = TII.getCallFrameSetupOpcode();
3531 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(AdjStackDown))
3532 .addImm(NumBytes)
3533 .addImm(0)
3534 .addImm(0)
3535 .setOperandDead(4); // eflags
3536
3537 // Walk the register/memloc assignments, inserting copies/loads.
3538 const X86RegisterInfo *RegInfo = Subtarget->getRegisterInfo();
3539 for (const CCValAssign &VA : ArgLocs) {
3540 const Value *ArgVal = OutVals[VA.getValNo()];
3541 MVT ArgVT = OutVTs[VA.getValNo()];
3542
3543 if (ArgVT == MVT::x86mmx)
3544 return false;
3545
3546 Register ArgReg = ArgRegs[VA.getValNo()];
3547
3548 // Promote the value if needed.
3549 switch (VA.getLocInfo()) {
3550 case CCValAssign::Full: break;
3551 case CCValAssign::SExt: {
3552 assert(VA.getLocVT().isInteger() && !VA.getLocVT().isVector() &&
3553 "Unexpected extend");
3554
3555 if (ArgVT == MVT::i1)
3556 return false;
3557
3558 bool Emitted = X86FastEmitExtend(ISD::SIGN_EXTEND, VA.getLocVT(), ArgReg,
3559 ArgVT, ArgReg);
3560 assert(Emitted && "Failed to emit a sext!"); (void)Emitted;
3561 ArgVT = VA.getLocVT();
3562 break;
3563 }
3564 case CCValAssign::ZExt: {
3565 assert(VA.getLocVT().isInteger() && !VA.getLocVT().isVector() &&
3566 "Unexpected extend");
3567
3568 // Handle zero-extension from i1 to i8, which is common.
3569 if (ArgVT == MVT::i1) {
3570 // Set the high bits to zero.
3571 ArgReg = fastEmitZExtFromI1(MVT::i8, ArgReg);
3572 ArgVT = MVT::i8;
3573
3574 if (!ArgReg)
3575 return false;
3576 }
3577
3578 bool Emitted = X86FastEmitExtend(ISD::ZERO_EXTEND, VA.getLocVT(), ArgReg,
3579 ArgVT, ArgReg);
3580 assert(Emitted && "Failed to emit a zext!"); (void)Emitted;
3581 ArgVT = VA.getLocVT();
3582 break;
3583 }
3584 case CCValAssign::AExt: {
3585 assert(VA.getLocVT().isInteger() && !VA.getLocVT().isVector() &&
3586 "Unexpected extend");
3587 bool Emitted = X86FastEmitExtend(ISD::ANY_EXTEND, VA.getLocVT(), ArgReg,
3588 ArgVT, ArgReg);
3589 if (!Emitted)
3590 Emitted = X86FastEmitExtend(ISD::ZERO_EXTEND, VA.getLocVT(), ArgReg,
3591 ArgVT, ArgReg);
3592 if (!Emitted)
3593 Emitted = X86FastEmitExtend(ISD::SIGN_EXTEND, VA.getLocVT(), ArgReg,
3594 ArgVT, ArgReg);
3595
3596 assert(Emitted && "Failed to emit a aext!"); (void)Emitted;
3597 ArgVT = VA.getLocVT();
3598 break;
3599 }
3600 case CCValAssign::BCvt: {
3601 ArgReg = fastEmit_r(ArgVT, VA.getLocVT(), ISD::BITCAST, ArgReg);
3602 assert(ArgReg && "Failed to emit a bitcast!");
3603 ArgVT = VA.getLocVT();
3604 break;
3605 }
3606 case CCValAssign::VExt:
3607 // VExt has not been implemented, so this should be impossible to reach
3608 // for now. However, fallback to Selection DAG isel once implemented.
3609 return false;
3613 case CCValAssign::FPExt:
3614 case CCValAssign::Trunc:
3615 llvm_unreachable("Unexpected loc info!");
3617 // FIXME: Indirect doesn't need extending, but fast-isel doesn't fully
3618 // support this.
3619 return false;
3620 }
3621
3622 if (VA.isRegLoc()) {
3623 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
3624 TII.get(TargetOpcode::COPY), VA.getLocReg()).addReg(ArgReg);
3625 OutRegs.push_back(VA.getLocReg());
3626 } else {
3627 assert(VA.isMemLoc() && "Unknown value location!");
3628
3629 // Don't emit stores for undef values.
3630 if (isa<UndefValue>(ArgVal))
3631 continue;
3632
3633 unsigned LocMemOffset = VA.getLocMemOffset();
3634 X86AddressMode AM;
3635 AM.Base.Reg = RegInfo->getStackRegister();
3636 AM.Disp = LocMemOffset;
3637 ISD::ArgFlagsTy Flags = OutFlags[VA.getValNo()];
3638 Align Alignment = DL.getABITypeAlign(ArgVal->getType());
3639 MachineMemOperand *MMO = FuncInfo.MF->getMachineMemOperand(
3640 MachinePointerInfo::getStack(*FuncInfo.MF, LocMemOffset),
3641 MachineMemOperand::MOStore, ArgVT.getStoreSize(), Alignment);
3642 if (Flags.isByVal()) {
3643 X86AddressMode SrcAM;
3644 SrcAM.Base.Reg = ArgReg;
3645 if (!TryEmitSmallMemcpy(AM, SrcAM, Flags.getByValSize()))
3646 return false;
3647 } else if (isa<ConstantInt>(ArgVal) || isa<ConstantPointerNull>(ArgVal)) {
3648 // If this is a really simple value, emit this with the Value* version
3649 // of X86FastEmitStore. If it isn't simple, we don't want to do this,
3650 // as it can cause us to reevaluate the argument.
3651 if (!X86FastEmitStore(ArgVT, ArgVal, AM, MMO))
3652 return false;
3653 } else {
3654 if (!X86FastEmitStore(ArgVT, ArgReg, AM, MMO))
3655 return false;
3656 }
3657 }
3658 }
3659
3660 // ELF / PIC requires GOT in the EBX register before function calls via PLT
3661 // GOT pointer.
3662 if (Subtarget->isPICStyleGOT()) {
3663 Register Base = getInstrInfo()->getGlobalBaseReg(FuncInfo.MF);
3664 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
3665 TII.get(TargetOpcode::COPY), X86::EBX).addReg(Base);
3666 }
3667
3668 if (Is64Bit && IsVarArg && !IsWin64) {
3669 // From AMD64 ABI document:
3670 // For calls that may call functions that use varargs or stdargs
3671 // (prototype-less calls or calls to functions containing ellipsis (...) in
3672 // the declaration) %al is used as hidden argument to specify the number
3673 // of SSE registers used. The contents of %al do not need to match exactly
3674 // the number of registers, but must be an ubound on the number of SSE
3675 // registers used and is in the range 0 - 8 inclusive.
3676
3677 // Count the number of XMM registers allocated.
3678 static const MCPhysReg XMMArgRegs[] = {
3679 X86::XMM0, X86::XMM1, X86::XMM2, X86::XMM3,
3680 X86::XMM4, X86::XMM5, X86::XMM6, X86::XMM7
3681 };
3682 unsigned NumXMMRegs = CCInfo.getFirstUnallocated(XMMArgRegs);
3683 assert((Subtarget->hasSSE1() || !NumXMMRegs)
3684 && "SSE registers cannot be used when SSE is disabled");
3685 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::MOV8ri),
3686 X86::AL).addImm(NumXMMRegs);
3687 }
3688
3689 // Materialize callee address in a register. FIXME: GV address can be
3690 // handled with a CALLpcrel32 instead.
3691 X86AddressMode CalleeAM;
3692 if (!X86SelectCallAddress(Callee, CalleeAM))
3693 return false;
3694
3695 Register CalleeOp;
3696 const GlobalValue *GV = nullptr;
3697 if (CalleeAM.GV != nullptr) {
3698 GV = CalleeAM.GV;
3699 } else if (CalleeAM.Base.Reg) {
3700 CalleeOp = CalleeAM.Base.Reg;
3701 } else
3702 return false;
3703
3704 // Issue the call.
3705 MachineInstrBuilder MIB;
3706 if (CalleeOp) {
3707 // Register-indirect call.
3708 unsigned CallOpc = Is64Bit ? X86::CALL64r : X86::CALL32r;
3709 MIB = BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(CallOpc))
3710 .addReg(CalleeOp);
3711 } else {
3712 // Direct call.
3713 assert(GV && "Not a direct call");
3714 // See if we need any target-specific flags on the GV operand.
3715 unsigned char OpFlags = Subtarget->classifyGlobalFunctionReference(GV);
3716 if (OpFlags == X86II::MO_PLT && !Is64Bit &&
3717 TM.getRelocationModel() == Reloc::Static && isa<Function>(GV) &&
3718 cast<Function>(GV)->isIntrinsic())
3719 OpFlags = X86II::MO_NO_FLAG;
3720
3721 // This will be a direct call, or an indirect call through memory for
3722 // NonLazyBind calls or dllimport calls.
3723 bool NeedLoad = OpFlags == X86II::MO_DLLIMPORT ||
3724 OpFlags == X86II::MO_GOTPCREL ||
3725 OpFlags == X86II::MO_GOTPCREL_NORELAX ||
3726 OpFlags == X86II::MO_COFFSTUB;
3727 unsigned CallOpc = NeedLoad
3728 ? (Is64Bit ? X86::CALL64m : X86::CALL32m)
3729 : (Is64Bit ? X86::CALL64pcrel32 : X86::CALLpcrel32);
3730
3731 MIB = BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(CallOpc));
3732 if (NeedLoad)
3733 MIB.addReg(Is64Bit ? X86::RIP : X86::NoRegister).addImm(1).addReg(0);
3734 if (Symbol)
3735 MIB.addSym(Symbol, OpFlags);
3736 else
3737 MIB.addGlobalAddress(GV, 0, OpFlags);
3738 if (NeedLoad)
3739 MIB.addReg(0);
3740 }
3741
3742 // Add a register mask operand representing the call-preserved registers.
3743 // Proper defs for return values will be added by setPhysRegsDeadExcept().
3744 MIB.addRegMask(TRI.getCallPreservedMask(*FuncInfo.MF, CC));
3745
3746 // Add an implicit use GOT pointer in EBX.
3747 if (Subtarget->isPICStyleGOT())
3748 MIB.addReg(X86::EBX, RegState::Implicit);
3749
3750 if (Is64Bit && IsVarArg && !IsWin64)
3751 MIB.addReg(X86::AL, RegState::Implicit);
3752
3753 // Add implicit physical register uses to the call.
3754 for (auto Reg : OutRegs)
3755 MIB.addReg(Reg, RegState::Implicit);
3756
3757 // Issue CALLSEQ_END
3758 unsigned NumBytesForCalleeToPop =
3759 X86::isCalleePop(CC, Subtarget->is64Bit(), IsVarArg,
3760 TM.Options.GuaranteedTailCallOpt)
3761 ? NumBytes // Callee pops everything.
3762 : computeBytesPoppedByCalleeForSRet(Subtarget, CC, CLI.CB);
3763 unsigned AdjStackUp = TII.getCallFrameDestroyOpcode();
3764 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(AdjStackUp))
3765 .addImm(NumBytes)
3766 .addImm(NumBytesForCalleeToPop)
3767 .setOperandDead(3); // eflags
3768
3769 // Now handle call return values.
3771 CCState CCRetInfo(CC, IsVarArg, *FuncInfo.MF, RVLocs,
3772 CLI.RetTy->getContext());
3773 CCRetInfo.AnalyzeCallResult(Ins, RetCC_X86);
3774
3775 // Copy all of the result registers out of their specified physreg.
3776 Register ResultReg = FuncInfo.CreateRegs(CLI.RetTy);
3777 for (unsigned i = 0; i != RVLocs.size(); ++i) {
3778 CCValAssign &VA = RVLocs[i];
3779 EVT CopyVT = VA.getValVT();
3780 Register CopyReg = ResultReg + i;
3781 Register SrcReg = VA.getLocReg();
3782
3783 // If this is x86-64, and we disabled SSE, we can't return FP values
3784 if ((CopyVT == MVT::f32 || CopyVT == MVT::f64) &&
3785 ((Is64Bit || Ins[i].Flags.isInReg()) && !Subtarget->hasSSE1())) {
3786 report_fatal_error("SSE register return with SSE disabled");
3787 }
3788
3789 // If we prefer to use the value in xmm registers, copy it out as f80 and
3790 // use a truncate to move it from fp stack reg to xmm reg.
3791 if ((SrcReg == X86::FP0 || SrcReg == X86::FP1) &&
3792 isScalarFPTypeInSSEReg(VA.getValVT())) {
3793 CopyVT = MVT::f80;
3794 CopyReg = createResultReg(&X86::RFP80RegClass);
3795 }
3796
3797 // Copy out the result.
3798 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
3799 TII.get(TargetOpcode::COPY), CopyReg).addReg(SrcReg);
3800 InRegs.push_back(VA.getLocReg());
3801
3802 // Round the f80 to the right size, which also moves it to the appropriate
3803 // xmm register. This is accomplished by storing the f80 value in memory
3804 // and then loading it back.
3805 if (CopyVT != VA.getValVT()) {
3806 EVT ResVT = VA.getValVT();
3807 unsigned Opc = ResVT == MVT::f32 ? X86::ST_Fp80m32 : X86::ST_Fp80m64;
3808 unsigned MemSize = ResVT.getSizeInBits()/8;
3809 int FI = MFI.CreateStackObject(MemSize, Align(MemSize), false);
3810 addFrameReference(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
3811 TII.get(Opc)), FI)
3812 .addReg(CopyReg);
3813 Opc = ResVT == MVT::f32 ? X86::MOVSSrm_alt : X86::MOVSDrm_alt;
3814 addFrameReference(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
3815 TII.get(Opc), ResultReg + i), FI);
3816 }
3817 }
3818
3819 CLI.ResultReg = ResultReg;
3820 CLI.NumResultRegs = RVLocs.size();
3821 CLI.Call = MIB;
3822
3823 // Add call site info for call graph section.
3824 if (TM.Options.EmitCallGraphSection && CB && CB->isIndirectCall()) {
3825 MachineFunction::CallSiteInfo CSInfo(*CB);
3826 MF->addCallSiteInfo(CLI.Call, std::move(CSInfo));
3827 }
3828
3829 return true;
3830}
3831
3832bool
3833X86FastISel::fastSelectInstruction(const Instruction *I) {
3834 switch (I->getOpcode()) {
3835 default: break;
3836 case Instruction::Load:
3837 return X86SelectLoad(I);
3838 case Instruction::Store:
3839 return X86SelectStore(I);
3840 case Instruction::Ret:
3841 return X86SelectRet(I);
3842 case Instruction::ICmp:
3843 case Instruction::FCmp:
3844 return X86SelectCmp(I);
3845 case Instruction::ZExt:
3846 return X86SelectZExt(I);
3847 case Instruction::SExt:
3848 return X86SelectSExt(I);
3849 case Instruction::CondBr:
3850 return X86SelectBranch(I);
3851 case Instruction::LShr:
3852 case Instruction::AShr:
3853 case Instruction::Shl:
3854 return X86SelectShift(I);
3855 case Instruction::Mul:
3856 return X86SelectMul(I);
3857 case Instruction::SDiv:
3858 case Instruction::UDiv:
3859 case Instruction::SRem:
3860 case Instruction::URem:
3861 return X86SelectDivRem(I);
3862 case Instruction::Select:
3863 return X86SelectSelect(I);
3864 case Instruction::Trunc:
3865 return X86SelectTrunc(I);
3866 case Instruction::FPExt:
3867 return X86SelectFPExt(I);
3868 case Instruction::FPTrunc:
3869 return X86SelectFPTrunc(I);
3870 case Instruction::SIToFP:
3871 return X86SelectSIToFP(I);
3872 case Instruction::UIToFP:
3873 return X86SelectUIToFP(I);
3874 case Instruction::IntToPtr: // Deliberate fall-through.
3875 case Instruction::PtrToInt: {
3876 EVT SrcVT = TLI.getValueType(DL, I->getOperand(0)->getType());
3877 EVT DstVT = TLI.getValueType(DL, I->getType());
3878 if (DstVT.bitsGT(SrcVT))
3879 return X86SelectZExt(I);
3880 if (DstVT.bitsLT(SrcVT))
3881 return X86SelectTrunc(I);
3882 Register Reg = getRegForValue(I->getOperand(0));
3883 if (!Reg)
3884 return false;
3885 updateValueMap(I, Reg);
3886 return true;
3887 }
3888 case Instruction::BitCast:
3889 return X86SelectBitCast(I);
3890 }
3891
3892 return false;
3893}
3894
3895Register X86FastISel::emitMOV32r0() {
3896 Register ResultReg = createResultReg(&X86::GR32RegClass);
3897 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::MOV32r0),
3898 ResultReg)
3899 .setOperandDead(1);
3900 return ResultReg;
3901}
3902
3903Register X86FastISel::X86MaterializeInt(const ConstantInt *CI, MVT VT) {
3904 if (VT > MVT::i64)
3905 return Register();
3906
3907 uint64_t Imm = CI->getZExtValue();
3908 if (Imm == 0) {
3909 Register SrcReg = emitMOV32r0();
3910 switch (VT.SimpleTy) {
3911 default: llvm_unreachable("Unexpected value type");
3912 case MVT::i1:
3913 case MVT::i8:
3914 return fastEmitInst_extractsubreg(MVT::i8, SrcReg, X86::sub_8bit);
3915 case MVT::i16:
3916 return fastEmitInst_extractsubreg(MVT::i16, SrcReg, X86::sub_16bit);
3917 case MVT::i32:
3918 return SrcReg;
3919 case MVT::i64: {
3920 Register ResultReg = createResultReg(&X86::GR64RegClass);
3921 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
3922 TII.get(TargetOpcode::SUBREG_TO_REG), ResultReg)
3923 .addReg(SrcReg)
3924 .addImm(X86::sub_32bit);
3925 return ResultReg;
3926 }
3927 }
3928 }
3929
3930 unsigned Opc = 0;
3931 switch (VT.SimpleTy) {
3932 default: llvm_unreachable("Unexpected value type");
3933 case MVT::i1:
3934 VT = MVT::i8;
3935 [[fallthrough]];
3936 case MVT::i8: Opc = X86::MOV8ri; break;
3937 case MVT::i16: Opc = X86::MOV16ri; break;
3938 case MVT::i32: Opc = X86::MOV32ri; break;
3939 case MVT::i64:
3940 Opc = X86::getMOVriOpcode(/*Use64BitReg=*/true, Imm);
3941 break;
3942 }
3943 return fastEmitInst_i(Opc, TLI.getRegClassFor(VT), Imm);
3944}
3945
3946Register X86FastISel::X86MaterializeFP(const ConstantFP *CFP, MVT VT) {
3947 if (CFP->isNullValue())
3948 return fastMaterializeFloatZero(CFP);
3949
3950 // Can't handle alternate code models yet.
3951 CodeModel::Model CM = TM.getCodeModel();
3952 if (CM != CodeModel::Small && CM != CodeModel::Medium &&
3953 CM != CodeModel::Large)
3954 return Register();
3955
3956 // Get opcode and regclass of the output for the given load instruction.
3957 unsigned Opc = 0;
3958 bool HasSSE1 = Subtarget->hasSSE1();
3959 bool HasSSE2 = Subtarget->hasSSE2();
3960 bool HasAVX = Subtarget->hasAVX();
3961 bool HasAVX512 = Subtarget->hasAVX512();
3962 switch (VT.SimpleTy) {
3963 default:
3964 return Register();
3965 case MVT::f32:
3966 Opc = HasAVX512 ? X86::VMOVSSZrm_alt
3967 : HasAVX ? X86::VMOVSSrm_alt
3968 : HasSSE1 ? X86::MOVSSrm_alt
3969 : X86::LD_Fp32m;
3970 break;
3971 case MVT::f64:
3972 Opc = HasAVX512 ? X86::VMOVSDZrm_alt
3973 : HasAVX ? X86::VMOVSDrm_alt
3974 : HasSSE2 ? X86::MOVSDrm_alt
3975 : X86::LD_Fp64m;
3976 break;
3977 case MVT::f80:
3978 // No f80 support yet.
3979 return Register();
3980 }
3981
3982 // MachineConstantPool wants an explicit alignment.
3983 Align Alignment = DL.getPrefTypeAlign(CFP->getType());
3984
3985 // x86-32 PIC requires a PIC base register for constant pools.
3986 Register PICBase;
3987 unsigned char OpFlag = Subtarget->classifyLocalReference(nullptr);
3988 if (OpFlag == X86II::MO_PIC_BASE_OFFSET)
3989 PICBase = getInstrInfo()->getGlobalBaseReg(FuncInfo.MF);
3990 else if (OpFlag == X86II::MO_GOTOFF)
3991 PICBase = getInstrInfo()->getGlobalBaseReg(FuncInfo.MF);
3992 else if (Subtarget->is64Bit() && TM.getCodeModel() != CodeModel::Large)
3993 PICBase = X86::RIP;
3994
3995 // Create the load from the constant pool.
3996 unsigned CPI = MCP.getConstantPoolIndex(CFP, Alignment);
3997 Register ResultReg = createResultReg(TLI.getRegClassFor(VT.SimpleTy));
3998
3999 // Large code model only applies to 64-bit mode.
4000 if (Subtarget->is64Bit() && CM == CodeModel::Large) {
4001 Register AddrReg = createResultReg(&X86::GR64RegClass);
4002 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::MOV64ri),
4003 AddrReg)
4004 .addConstantPoolIndex(CPI, 0, OpFlag);
4005 MachineInstrBuilder MIB = BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
4006 TII.get(Opc), ResultReg);
4007 addRegReg(MIB, AddrReg, false, X86::NoSubRegister, PICBase, false,
4008 X86::NoSubRegister);
4009 MachineMemOperand *MMO = FuncInfo.MF->getMachineMemOperand(
4011 MachineMemOperand::MOLoad, DL.getPointerSize(), Alignment);
4012 MIB->addMemOperand(*FuncInfo.MF, MMO);
4013 return ResultReg;
4014 }
4015
4016 addConstantPoolReference(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
4017 TII.get(Opc), ResultReg),
4018 CPI, PICBase, OpFlag);
4019 return ResultReg;
4020}
4021
4022Register X86FastISel::X86MaterializeGV(const GlobalValue *GV, MVT VT) {
4023 // Can't handle large GlobalValues yet.
4024 if (TM.getCodeModel() != CodeModel::Small &&
4025 TM.getCodeModel() != CodeModel::Medium)
4026 return Register();
4027 if (TM.isLargeGlobalValue(GV))
4028 return Register();
4029
4030 // Materialize addresses with LEA/MOV instructions.
4031 X86AddressMode AM;
4032 if (X86SelectAddress(GV, AM)) {
4033 // If the expression is just a basereg, then we're done, otherwise we need
4034 // to emit an LEA.
4036 AM.IndexReg == 0 && AM.Disp == 0 && AM.GV == nullptr)
4037 return AM.Base.Reg;
4038
4039 Register ResultReg = createResultReg(TLI.getRegClassFor(VT));
4040 if (TM.getRelocationModel() == Reloc::Static &&
4041 TLI.getPointerTy(DL) == MVT::i64) {
4042 // The displacement code could be more than 32 bits away so we need to use
4043 // an instruction with a 64 bit immediate
4044 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(X86::MOV64ri),
4045 ResultReg)
4046 .addGlobalAddress(GV);
4047 } else {
4048 unsigned Opc =
4049 TLI.getPointerTy(DL) == MVT::i32
4050 ? (Subtarget->isTarget64BitILP32() ? X86::LEA64_32r : X86::LEA32r)
4051 : X86::LEA64r;
4052 addFullAddress(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
4053 TII.get(Opc), ResultReg), AM);
4054 }
4055 return ResultReg;
4056 }
4057 return Register();
4058}
4059
4060Register X86FastISel::fastMaterializeConstant(const Constant *C) {
4061 EVT CEVT = TLI.getValueType(DL, C->getType(), true);
4062
4063 // Only handle simple types.
4064 if (!CEVT.isSimple())
4065 return Register();
4066 MVT VT = CEVT.getSimpleVT();
4067
4068 if (const auto *CI = dyn_cast<ConstantInt>(C))
4069 return X86MaterializeInt(CI, VT);
4070 if (const auto *CFP = dyn_cast<ConstantFP>(C))
4071 return X86MaterializeFP(CFP, VT);
4072 if (const auto *GV = dyn_cast<GlobalValue>(C))
4073 return X86MaterializeGV(GV, VT);
4074 if (isa<UndefValue>(C)) {
4075 unsigned Opc = 0;
4076 switch (VT.SimpleTy) {
4077 default:
4078 break;
4079 case MVT::f32:
4080 if (!Subtarget->hasSSE1())
4081 Opc = X86::LD_Fp032;
4082 break;
4083 case MVT::f64:
4084 if (!Subtarget->hasSSE2())
4085 Opc = X86::LD_Fp064;
4086 break;
4087 case MVT::f80:
4088 Opc = X86::LD_Fp080;
4089 break;
4090 }
4091
4092 if (Opc) {
4093 Register ResultReg = createResultReg(TLI.getRegClassFor(VT));
4094 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Opc),
4095 ResultReg);
4096 return ResultReg;
4097 }
4098 }
4099
4100 return Register();
4101}
4102
4103Register X86FastISel::fastMaterializeAlloca(const AllocaInst *C) {
4104 // Fail on dynamic allocas. At this point, getRegForValue has already
4105 // checked its CSE maps, so if we're here trying to handle a dynamic
4106 // alloca, we're not going to succeed. X86SelectAddress has a
4107 // check for dynamic allocas, because it's called directly from
4108 // various places, but targetMaterializeAlloca also needs a check
4109 // in order to avoid recursion between getRegForValue,
4110 // X86SelectAddrss, and targetMaterializeAlloca.
4111 if (!FuncInfo.StaticAllocaMap.count(C))
4112 return Register();
4113 assert(C->isStaticAlloca() && "dynamic alloca in the static alloca map?");
4114
4115 X86AddressMode AM;
4116 if (!X86SelectAddress(C, AM))
4117 return Register();
4118 unsigned Opc =
4119 TLI.getPointerTy(DL) == MVT::i32
4120 ? (Subtarget->isTarget64BitILP32() ? X86::LEA64_32r : X86::LEA32r)
4121 : X86::LEA64r;
4122 const TargetRegisterClass *RC = TLI.getRegClassFor(TLI.getPointerTy(DL));
4123 Register ResultReg = createResultReg(RC);
4124 addFullAddress(BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD,
4125 TII.get(Opc), ResultReg), AM);
4126 return ResultReg;
4127}
4128
4129Register X86FastISel::fastMaterializeFloatZero(const ConstantFP *CF) {
4130 MVT VT;
4131 if (!isTypeLegal(CF->getType(), VT))
4132 return Register();
4133
4134 // Get opcode and regclass for the given zero.
4135 bool HasSSE1 = Subtarget->hasSSE1();
4136 bool HasSSE2 = Subtarget->hasSSE2();
4137 bool HasAVX512 = Subtarget->hasAVX512();
4138 unsigned Opc = 0;
4139 switch (VT.SimpleTy) {
4140 default: return 0;
4141 case MVT::f16:
4142 Opc = HasAVX512 ? X86::AVX512_FsFLD0SH : X86::FsFLD0SH;
4143 break;
4144 case MVT::f32:
4145 Opc = HasAVX512 ? X86::AVX512_FsFLD0SS
4146 : HasSSE1 ? X86::FsFLD0SS
4147 : X86::LD_Fp032;
4148 break;
4149 case MVT::f64:
4150 Opc = HasAVX512 ? X86::AVX512_FsFLD0SD
4151 : HasSSE2 ? X86::FsFLD0SD
4152 : X86::LD_Fp064;
4153 break;
4154 case MVT::f80:
4155 // No f80 support yet.
4156 return Register();
4157 }
4158
4159 Register ResultReg = createResultReg(TLI.getRegClassFor(VT));
4160 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, TII.get(Opc), ResultReg);
4161 return ResultReg;
4162}
4163
4164bool X86FastISel::tryToFoldLoadIntoMI(MachineInstr *MI, unsigned OpNo,
4165 const LoadInst *LI) {
4166 const Value *Ptr = LI->getPointerOperand();
4167 X86AddressMode AM;
4168 if (!X86SelectAddress(Ptr, AM))
4169 return false;
4170
4171 const X86InstrInfo &XII = (const X86InstrInfo &)TII;
4172
4173 unsigned Size = DL.getTypeAllocSize(LI->getType());
4174
4176 AM.getFullAddress(AddrOps);
4177
4178 MachineInstr *CopyMI = nullptr;
4179 MachineInstr *Result = XII.foldMemoryOperandImpl(
4180 *FuncInfo.MF, *MI, OpNo, AddrOps, FuncInfo.InsertPt, Size, LI->getAlign(),
4181 /*AllowCommute=*/true, CopyMI);
4182 if (!Result)
4183 return false;
4184
4185 // The index register could be in the wrong register class. Unfortunately,
4186 // foldMemoryOperandImpl could have commuted the instruction so its not enough
4187 // to just look at OpNo + the offset to the index reg. We actually need to
4188 // scan the instruction to find the index reg and see if its the correct reg
4189 // class.
4190 unsigned OperandNo = 0;
4191 for (MachineInstr::mop_iterator I = Result->operands_begin(),
4192 E = Result->operands_end(); I != E; ++I, ++OperandNo) {
4193 MachineOperand &MO = *I;
4194 if (!MO.isReg() || MO.isDef() || MO.getReg() != AM.IndexReg)
4195 continue;
4196 // Found the index reg, now try to rewrite it.
4197 Register IndexReg = constrainOperandRegClass(Result->getDesc(),
4198 MO.getReg(), OperandNo);
4199 if (IndexReg == MO.getReg())
4200 continue;
4201 MO.setReg(IndexReg);
4202 }
4203
4204 if (MI->isCall())
4205 FuncInfo.MF->moveAdditionalCallInfo(MI, Result);
4206 Result->addMemOperand(*FuncInfo.MF, createMachineMemOperandFor(LI));
4207 Result->cloneInstrSymbols(*FuncInfo.MF, *MI);
4209 removeDeadCode(I, std::next(I));
4210 return true;
4211}
4212
4213Register X86FastISel::fastEmitInst_rrrr(unsigned MachineInstOpcode,
4214 const TargetRegisterClass *RC,
4215 Register Op0, Register Op1,
4216 Register Op2, Register Op3) {
4217 const MCInstrDesc &II = TII.get(MachineInstOpcode);
4218
4219 Register ResultReg = createResultReg(RC);
4220 Op0 = constrainOperandRegClass(II, Op0, II.getNumDefs());
4221 Op1 = constrainOperandRegClass(II, Op1, II.getNumDefs() + 1);
4222 Op2 = constrainOperandRegClass(II, Op2, II.getNumDefs() + 2);
4223 Op3 = constrainOperandRegClass(II, Op3, II.getNumDefs() + 3);
4224
4225 assert(II.getNumDefs() >= 1 && "instruction must define the result");
4226 BuildMI(*FuncInfo.MBB, FuncInfo.InsertPt, MIMD, II, ResultReg)
4227 .addReg(Op0)
4228 .addReg(Op1)
4229 .addReg(Op2)
4230 .addReg(Op3);
4231 return ResultReg;
4232}
4233
4234namespace llvm {
4236 const TargetLibraryInfo *libInfo,
4237 const LibcallLoweringInfo *libcallLowering) {
4238 return new X86FastISel(funcInfo, libInfo, libcallLowering);
4239}
4240}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
This file defines the FastISel class.
Hexagon Common GEP
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Module.h This file contains the declarations for the Module class.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
This file declares the MachineConstantPool class which is an abstract constant pool to keep track of ...
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
uint64_t IntrinsicInst * II
const SmallVectorImpl< MachineOperand > & Cond
#define GET_EGPR_IF_ENABLED(OPC)
static unsigned X86ChooseCmpImmediateOpcode(EVT VT, const ConstantInt *RHSC)
If we have a comparison with RHS as the RHS of the comparison, return an opcode that works for the co...
static std::pair< unsigned, bool > getX86SSEConditionCode(CmpInst::Predicate Predicate)
static unsigned computeBytesPoppedByCalleeForSRet(const X86Subtarget *Subtarget, CallingConv::ID CC, const CallBase *CB)
#define GET_SETCC
static unsigned X86ChooseCmpOpcode(EVT VT, const X86Subtarget *Subtarget)
static bool X86SelectAddress(MachineInstr &I, const X86TargetMachine &TM, const MachineRegisterInfo &MRI, const X86Subtarget &STI, X86AddressMode &AM)
Value * RHS
Value * LHS
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
Definition APInt.cpp:1030
InstListType::const_iterator const_iterator
Definition BasicBlock.h:171
Register getLocReg() const
LocInfo getLocInfo() const
int64_t getLocMemOffset() const
unsigned getValNo() const
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
bool arg_empty() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
This class is the base class for the comparison instructions.
Definition InstrTypes.h:728
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ FCMP_TRUE
1 1 1 1 Always true (always folded)
Definition InstrTypes.h:757
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_FALSE
0 0 0 0 Always false (always folded)
Definition InstrTypes.h:742
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Definition InstrTypes.h:852
Value * getCondition() const
BasicBlock * getSuccessor(unsigned i) const
This is the shared class of boolean and integer constants.
Definition Constants.h:87
bool isOne() const
This is just a convenience method to make client code smaller for a common case.
Definition Constants.h:225
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
unsigned getBitWidth() const
getBitWidth - Return the scalar bitwidth of this constant.
Definition Constants.h:162
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
Definition Constant.h:64
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
bool isValidLocationForIntrinsic(const DILocation *DL) const
Check that a location is valid for this variable.
Value * getAddress() const
DILocalVariable * getVariable() const
DIExpression * getExpression() const
This is a fast-path instruction selection class that generates poor code and doesn't support illegal ...
Definition FastISel.h:67
FunctionLoweringInfo - This contains information that is global to a function that is used when lower...
Module * getParent()
Get the module that this global value is contained inside of...
LLVM_ABI bool isAtomic() const LLVM_READONLY
Return true if this instruction has an AtomicOrdering of unordered or higher.
Tracks which library functions to use for a particular subtarget or function.
Value * getPointerOperand()
Align getAlign() const
Return the alignment of the access that is being performed.
bool usesWindowsCFI() const
Definition MCAsmInfo.h:675
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
Machine Value Type.
bool isVectorOf(MVT EltVT) const
Return true if this is a vector with matching element type.
SimpleValueType SimpleTy
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
MVT getVectorElementType() const
MachineInstrBundleIterator< MachineInstr > iterator
LLVM_ABI int CreateStackObject(uint64_t Size, Align Alignment, bool isSpillSlot, const AllocaInst *Alloca=nullptr, uint8_t ID=0)
Create a new statically sized stack object, returning a nonnegative identifier to represent it.
void setFrameAddressIsTaken(bool T)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
Function & getFunction()
Return the LLVM function that this machine code represents.
void addCallSiteInfo(const MachineInstr *CallI, CallSiteInfo &&CallInfo)
Start tracking the arguments passed to the call CallI.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & addMetadata(const MDNode *MD) const
const MachineInstrBuilder & addSym(MCSymbol *Sym, unsigned char TargetFlags=0) const
const MachineInstrBuilder & addConstantPoolIndex(unsigned Idx, int Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addRegMask(const uint32_t *Mask) const
const MachineInstrBuilder & addGlobalAddress(const GlobalValue *GV, int64_t Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
unsigned getNumOperands() const
Retuns the total number of operands.
const MCInstrDesc & getDesc() const
Returns the target instruction descriptor of this MachineInstr.
MachineOperand * mop_iterator
iterator/begin/end - Iterate over all operands of a machine instruction.
LLVM_ABI void setPhysRegsDeadExcept(ArrayRef< Register > UsedRegs, const TargetRegisterInfo &TRI)
Mark every physreg used by this instruction as dead except those in the UsedRegs list.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI void addMemOperand(MachineFunction &MF, MachineMemOperand *MO)
Add a MachineMemOperand to the machine instruction.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
Register getReg() const
getReg - Returns the register number.
Value * getLength() const
Value * getRawDest() const
unsigned getDestAddressSpace() const
bool isVolatile() const
Value * getRawSource() const
Return the arguments to the instruction.
unsigned getSourceAddressSpace() const
Metadata * getModuleFlag(StringRef Key) const
Return the corresponding value if Key appears in module flags, otherwise return null.
Definition Module.cpp:358
Wrapper class representing virtual and physical registers.
Definition Register.h:20
void push_back(const T &Elt)
Align getAlign() const
Value * getValueOperand()
Value * getPointerOperand()
TypeSize getElementOffset(unsigned Idx) const
Definition DataLayout.h:774
Provides information about what library functions are available for the current target.
const MCAsmInfo & getMCAsmInfo() const
Return target specific asm information.
bool isOSMSVCRT() const
Is this a "Windows" OS targeting a "MSVCRT.dll" environment.
Definition Triple.h:824
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
bool isArrayTy() const
True if this is an instance of ArrayType.
Definition Type.h:274
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:271
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
const Use * const_op_iterator
Definition User.h:255
Value * getOperand(unsigned i) const
Definition User.h:207
unsigned getNumOperands() const
Definition User.h:229
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
MachineInstr * foldMemoryOperandImpl(MachineFunction &MF, MachineInstr &MI, ArrayRef< unsigned > Ops, int FrameIndex, MachineInstr *&CopyMI, LiveIntervals *LIS=nullptr, VirtRegMap *VRM=nullptr) const override
Fold a load or store of the specified stack slot into the specified machine instruction for the speci...
Register getPtrSizedFrameRegister(const MachineFunction &MF) const
Register getStackRegister() const
bool hasSSE1() const
bool isTargetMCU() const
const Triple & getTargetTriple() const
bool hasAVX512() const
bool hasSSE2() const
bool hasAVX() const
TypeSize getSequentialElementStride(const DataLayout &DL) const
const ParentTy * getParent() const
Definition ilist_node.h:34
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ HiPE
Used by the High-Performance Erlang Compiler (HiPE).
Definition CallingConv.h:53
@ GHC
Used by the Glasgow Haskell Compiler (GHC).
Definition CallingConv.h:50
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ Tail
Attemps to make calls as fast as possible while guaranteeing that tail call optimization can always b...
Definition CallingConv.h:76
@ SwiftTail
This follows the Swift calling convention in how arguments are passed but guarantees tail calls will ...
Definition CallingConv.h:87
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:43
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
Flag
These should be considered private to the implementation of the MCInstrDesc class.
Predicate
Predicate - These are "(BI << 5) | BO" for various predicates.
@ X86
Windows x64, Windows Itanium (IA-64)
Definition MCAsmInfo.h:53
@ MO_GOTPCREL_NORELAX
MO_GOTPCREL_NORELAX - Same as MO_GOTPCREL except that R_X86_64_GOTPCREL relocations are guaranteed to...
@ MO_GOTOFF
MO_GOTOFF - On a symbol operand this indicates that the immediate is the offset to the location of th...
@ MO_COFFSTUB
MO_COFFSTUB - On a symbol operand "FOO", this indicates that the reference is actually to the "....
@ MO_PLT
MO_PLT - On a symbol operand this indicates that the immediate is offset to the PLT entry of symbol n...
@ MO_NO_FLAG
MO_NO_FLAG - No flag for the operand.
@ MO_DLLIMPORT
MO_DLLIMPORT - On a symbol operand "FOO", this indicates that the reference is actually to the "__imp...
@ MO_PIC_BASE_OFFSET
MO_PIC_BASE_OFFSET - On a symbol operand this indicates that the immediate should get the value of th...
@ MO_GOTPCREL
MO_GOTPCREL - On a symbol operand this indicates that the immediate is offset to the GOT entry for th...
@ LAST_VALID_COND
Definition X86BaseInfo.h:95
FastISel * createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo, const LibcallLoweringInfo *libcallLowering)
std::pair< CondCode, bool > getX86ConditionCode(CmpInst::Predicate Predicate)
Return a pair of condition code for the given predicate and whether the instruction operands should b...
bool isCalleePop(CallingConv::ID CallingConv, bool is64Bit, bool IsVarArg, bool GuaranteeTCO)
Determines whether the callee is required to pop its own arguments.
unsigned getMOVriOpcode(bool Use64BitReg, int64_t Imm)
Return a MOVri opcode for materializing Imm into a 32- or 64-bit GPR.
unsigned getCMovOpcode(unsigned RegBytes, bool HasMemoryOperand=false, bool HasNDD=false)
Return a cmov opcode for the given register size in bytes, and operand type.
StringMapEntry< std::atomic< TypeEntryBody * > > TypeEntry
Definition TypePool.h:28
@ User
could "use" a pointer
@ Emitted
Assigned address, still materializing.
Definition Core.h:551
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
static bool isGlobalStubReference(unsigned char TargetFlag)
isGlobalStubReference - Return true if the specified TargetFlag operand is a reference to a stub for ...
static bool isGlobalRelativeToPICBase(unsigned char TargetFlag)
isGlobalRelativeToPICBase - Return true if the specified global value reference is relative to a 32-b...
LLVM_ABI Register constrainOperandRegClass(const MachineFunction &MF, const TargetRegisterInfo &TRI, MachineRegisterInfo &MRI, const TargetInstrInfo &TII, const RegisterBankInfo &RBI, MachineInstr &InsertPt, const TargetRegisterClass &RegClass, MachineOperand &RegMO)
Constrain the Register operand OpIdx, so that it is now constrained to the TargetRegisterClass passed...
Definition Utils.cpp:60
LLVM_ABI void GetReturnInfo(CallingConv::ID CC, Type *ReturnType, AttributeList attr, SmallVectorImpl< ISD::OutputArg > &Outs, const TargetLowering &TLI, const DataLayout &DL)
Given an LLVM IR type and return type attributes, compute the return value EVTs and flags,...
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
auto successors(const MachineBasicBlock *BB)
static const MachineInstrBuilder & addConstantPoolReference(const MachineInstrBuilder &MIB, unsigned CPI, Register GlobalBaseReg, unsigned char OpFlags)
addConstantPoolReference - This function is used to add a reference to the base of a constant value s...
static const MachineInstrBuilder & addRegReg(const MachineInstrBuilder &MIB, Register Reg1, bool isKill1, unsigned SubReg1, Register Reg2, bool isKill2, unsigned SubReg2)
addRegReg - This function is used to add a memory reference of the form: [Reg + Reg].
static const MachineInstrBuilder & addFrameReference(const MachineInstrBuilder &MIB, int FI, int Offset=0, bool mem=true)
addFrameReference - This function is used to add a reference to the base of an abstract object on the...
static const MachineInstrBuilder & addFullAddress(const MachineInstrBuilder &MIB, const X86AddressMode &AM)
Op::Description Desc
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI void ComputeValueTypes(const DataLayout &DL, Type *Ty, SmallVectorImpl< Type * > &Types, SmallVectorImpl< TypeSize > *Offsets=nullptr, TypeSize StartingOffset=TypeSize::getZero())
Given an LLVM IR type, compute non-aggregate subtypes.
Definition Analysis.cpp:74
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
generic_gep_type_iterator<> gep_type_iterator
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
DWARFExpression::Operation Op
bool CC_X86(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
bool RetCC_X86(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
gep_type_iterator gep_type_begin(const User *GEP)
static const MachineInstrBuilder & addDirectMem(const MachineInstrBuilder &MIB, Register Reg)
addDirectMem - This function is used to add a direct memory reference to the current instruction – th...
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
Extended Value Type.
Definition ValueTypes.h:35
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
Definition ValueTypes.h:307
bool bitsLT(EVT VT) const
Return true if this has less bits than VT.
Definition ValueTypes.h:323
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
static LLVM_ABI MachinePointerInfo getConstantPool(MachineFunction &MF)
Return a MachinePointerInfo record that refers to the constant pool.
X86AddressMode - This struct holds a generalized full x86 address mode.
void getFullAddress(SmallVectorImpl< MachineOperand > &MO)
const GlobalValue * GV
union llvm::X86AddressMode::BaseUnion Base
enum llvm::X86AddressMode::@202116273335065351270200035056227005202106004277 BaseType