LLVM 24.0.0git
WebAssemblyISelLowering.cpp
Go to the documentation of this file.
1//=- WebAssemblyISelLowering.cpp - WebAssembly DAG Lowering Implementation -==//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file implements the WebAssemblyTargetLowering class.
11///
12//===----------------------------------------------------------------------===//
13
32#include "llvm/IR/Function.h"
34#include "llvm/IR/Intrinsics.h"
35#include "llvm/IR/IntrinsicsWebAssembly.h"
40using namespace llvm;
41
42#define DEBUG_TYPE "wasm-lower"
43
45 const TargetMachine &TM, const WebAssemblySubtarget &STI)
46 : TargetLowering(TM, STI), Subtarget(&STI) {
47 auto MVTPtr = Subtarget->hasAddr64() ? MVT::i64 : MVT::i32;
48
49 // Set the load count for memcmp expand optimization
52
53 // Booleans always contain 0 or 1.
55 // Except in SIMD vectors
57 // We don't know the microarchitecture here, so just reduce register pressure.
59 // Tell ISel that we have a stack pointer.
61 Subtarget->hasAddr64() ? WebAssembly::SP64 : WebAssembly::SP32);
62 // Set up the register classes.
63 addRegisterClass(MVT::i32, &WebAssembly::I32RegClass);
64 addRegisterClass(MVT::i64, &WebAssembly::I64RegClass);
65 addRegisterClass(MVT::f32, &WebAssembly::F32RegClass);
66 addRegisterClass(MVT::f64, &WebAssembly::F64RegClass);
67 if (Subtarget->hasSIMD128()) {
68 addRegisterClass(MVT::v16i8, &WebAssembly::V128RegClass);
69 addRegisterClass(MVT::v8i16, &WebAssembly::V128RegClass);
70 addRegisterClass(MVT::v4i32, &WebAssembly::V128RegClass);
71 addRegisterClass(MVT::v4f32, &WebAssembly::V128RegClass);
72 addRegisterClass(MVT::v2i64, &WebAssembly::V128RegClass);
73 addRegisterClass(MVT::v2f64, &WebAssembly::V128RegClass);
74 }
75 if (Subtarget->hasFP16()) {
76 addRegisterClass(MVT::v8f16, &WebAssembly::V128RegClass);
77 }
78 if (Subtarget->hasReferenceTypes()) {
79 addRegisterClass(MVT::externref, &WebAssembly::EXTERNREFRegClass);
80 addRegisterClass(MVT::funcref, &WebAssembly::FUNCREFRegClass);
81 if (Subtarget->hasExceptionHandling()) {
82 addRegisterClass(MVT::exnref, &WebAssembly::EXNREFRegClass);
83 }
84 }
85 // Compute derived properties from the register classes.
86 computeRegisterProperties(Subtarget->getRegisterInfo());
87
88 // Transform loads and stores to pointers in address space 1 to loads and
89 // stores to WebAssembly global variables, outside linear memory.
90 for (auto T : {MVT::i32, MVT::i64, MVT::f32, MVT::f64}) {
93 }
94 if (Subtarget->hasSIMD128()) {
95 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
96 MVT::v2f64}) {
99 }
100 }
101 if (Subtarget->hasFP16()) {
102 setOperationAction(ISD::LOAD, MVT::v8f16, Custom);
104 }
105 if (Subtarget->hasReferenceTypes()) {
106 // We need custom load and store lowering for both externref, funcref and
107 // Other. The MVT::Other here represents tables of reference types.
108 for (auto T : {MVT::externref, MVT::funcref, MVT::Other}) {
111 }
112 }
113
121
122 // Take the default expansion for va_arg, va_copy, and va_end. There is no
123 // default action for va_start, so we do that custom.
128
129 for (auto T : {MVT::f32, MVT::f64, MVT::v4f32, MVT::v2f64, MVT::v8f16}) {
130 if (!Subtarget->hasFP16() && T == MVT::v8f16) {
131 continue;
132 }
133 // Don't expand the floating-point types to constant pools.
135 // Expand floating-point comparisons.
136 for (auto CC : {ISD::SETO, ISD::SETUO, ISD::SETUEQ, ISD::SETONE,
139 // Expand floating-point library function operators.
142 // Expand vector FREM, but use a libcall rather than an expansion for scalar
143 if (MVT(T).isVector())
145 else
147 // Note supported floating-point library function operators that otherwise
148 // default to expand.
152 // Support minimum and maximum, which otherwise default to expand.
155 if (Subtarget->hasSIMD128() && MVT(T).isVector()) {
158 }
159 // When experimental v8f16 support is enabled these instructions don't need
160 // to be expanded.
161 if (T != MVT::v8f16) {
164 }
165 if (Subtarget->hasFP16() && T == MVT::f32) {
167 setTruncStoreAction(T, MVT::f16, Legal);
168 } else {
170 setTruncStoreAction(T, MVT::f16, Expand);
171 }
172 }
173
174 // Expand unavailable integer operations.
175 for (auto Op :
179 for (auto T : {MVT::i32, MVT::i64})
181 if (Subtarget->hasSIMD128())
182 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64})
184 }
185
186 if (Subtarget->hasWideArithmetic()) {
192 }
193
194 if (Subtarget->hasNontrappingFPToInt())
196 for (auto T : {MVT::i32, MVT::i64})
198
199 if (Subtarget->hasRelaxedSIMD()) {
202 {MVT::v4f32, MVT::v2f64}, Custom);
203 }
204
205 // Combine expands these operations, because wasi-libc and emscripten do not
206 // yet have the dedicated libcalls.
209
210 // SIMD-specific configuration
211 if (Subtarget->hasSIMD128()) {
212
214
215 // Combine wide-vector muls, with extend inputs, to extmul_half.
218
219 // Combine vector mask reductions into alltrue/anytrue
221
222 // Convert vector to integer bitcasts to bitmask
224
225 // Hoist bitcasts out of shuffles
227
228 // Combine extends of extract_subvectors into widening ops
230
231 // Combine int_to_fp or fp_extend of extract_vectors and vice versa into
232 // conversions ops
235
236 // Combine fp_to_{s,u}int_sat or fp_round of concat_vectors or vice versa
237 // into conversion ops
241
243
244 // Support saturating add/sub for i8x16 and i16x8
246 for (auto T : {MVT::v16i8, MVT::v8i16})
248
249 // Support integer abs
250 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64})
252
253 // Custom lower BUILD_VECTORs to minimize number of replace_lanes
254 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
255 MVT::v2f64})
257
258 if (Subtarget->hasFP16()) {
262 }
263
264 // We have custom shuffle lowering to expose the shuffle mask
265 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
266 MVT::v2f64})
268
269 if (Subtarget->hasFP16())
271
272 // Support splatting
273 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
274 MVT::v2f64})
276
277 setOperationAction(ISD::AVGCEILU, {MVT::v8i16, MVT::v16i8}, Legal);
278
279 // Custom lowering since wasm shifts must have a scalar shift amount
280 for (auto Op : {ISD::SHL, ISD::SRA, ISD::SRL})
281 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64})
283
284 // Custom lower lane accesses to expand out variable indices
286 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
287 MVT::v2f64})
289
290 // There is no i8x16.mul instruction
291 setOperationAction(ISD::MUL, MVT::v16i8, Expand);
292
293 // Expand integer operations supported for scalars but not SIMD
294 for (auto Op :
296 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64})
298
299 // But we do have integer min and max operations
300 for (auto Op : {ISD::SMIN, ISD::SMAX, ISD::UMIN, ISD::UMAX})
301 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32})
303
304 // And we have popcnt for i8x16. It can be used to expand ctlz/cttz.
305 setOperationAction(ISD::CTPOP, MVT::v16i8, Legal);
306 setOperationAction(ISD::CTLZ, MVT::v16i8, Expand);
307 setOperationAction(ISD::CTTZ, MVT::v16i8, Expand);
308
309 // Custom lower bit counting operations for other types to scalarize them.
310 for (auto Op : {ISD::CTLZ, ISD::CTTZ, ISD::CTPOP})
311 for (auto T : {MVT::v8i16, MVT::v4i32, MVT::v2i64})
313
314 // Expand float operations supported for scalars but not SIMD
317 for (auto T : {MVT::v4f32, MVT::v2f64})
319
320 // Unsigned comparison operations are unavailable for i64x2 vectors.
322 setCondCodeAction(CC, MVT::v2i64, Custom);
323
324 // 64x2 conversions are not in the spec
325 for (auto Op :
327 for (auto T : {MVT::v2i64, MVT::v2f64})
329
330 // But saturating fp_to_int converstions are
332 setOperationAction(Op, MVT::v4i32, Custom);
333 if (Subtarget->hasFP16()) {
334 setOperationAction(Op, MVT::v8i16, Custom);
335 }
336 }
337
338 // Support vector extending
343 }
344
345 if (Subtarget->hasFP16()) {
346 setOperationAction(ISD::FMA, MVT::v8f16, Legal);
347 }
348
349 if (Subtarget->hasRelaxedSIMD()) {
352 }
353
354 // Partial MLA reductions.
356 setPartialReduceMLAAction(Op, MVT::v4i32, MVT::v16i8, Legal);
357 setPartialReduceMLAAction(Op, MVT::v4i32, MVT::v8i16, Legal);
358 }
359 }
360
361 // As a special case, these operators use the type to mean the type to
362 // sign-extend from.
364 if (!Subtarget->hasSignExt()) {
365 // Sign extends are legal only when extending a vector extract
366 auto Action = Subtarget->hasSIMD128() ? Custom : Expand;
367 for (auto T : {MVT::i8, MVT::i16, MVT::i32})
369 }
372
373 // Dynamic stack allocation: use the default expansion.
377
381
382 // Expand these forms; we pattern-match the forms that we can handle in isel.
383 for (auto T : {MVT::i32, MVT::i64, MVT::f32, MVT::f64})
384 for (auto Op : {ISD::BR_CC, ISD::SELECT_CC})
386
387 if (Subtarget->hasReferenceTypes())
388 for (auto Op : {ISD::BR_CC, ISD::SELECT_CC})
389 for (auto T : {MVT::externref, MVT::funcref})
391
392 // There is no vector conditional select instruction
393 for (auto T :
394 {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64, MVT::v2f64})
396
397 // We have custom switch handling.
399
400 // WebAssembly doesn't have:
401 // - Floating-point extending loads.
402 // - Floating-point truncating stores.
403 // - i1 extending loads.
404 // - truncating SIMD stores and most extending loads
405 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f32, Expand);
406 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
407 for (auto T : MVT::integer_valuetypes())
408 for (auto Ext : {ISD::EXTLOAD, ISD::ZEXTLOAD, ISD::SEXTLOAD})
409 setLoadExtAction(Ext, T, MVT::i1, Promote);
410 if (Subtarget->hasSIMD128()) {
411 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64, MVT::v4f32,
412 MVT::v2f64}) {
413 for (auto MemT : MVT::fixedlen_vector_valuetypes()) {
414 if (MVT(T) != MemT) {
416 for (auto Ext : {ISD::EXTLOAD, ISD::ZEXTLOAD, ISD::SEXTLOAD})
417 setLoadExtAction(Ext, T, MemT, Expand);
418 }
419 }
420 }
421 // But some vector extending loads are legal
422 for (auto Ext : {ISD::EXTLOAD, ISD::SEXTLOAD, ISD::ZEXTLOAD}) {
423 setLoadExtAction(Ext, MVT::v8i16, MVT::v8i8, Legal);
424 setLoadExtAction(Ext, MVT::v4i32, MVT::v4i16, Legal);
425 setLoadExtAction(Ext, MVT::v2i64, MVT::v2i32, Legal);
426 }
427 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2f32, Legal);
428 }
429
430 // Don't do anything clever with build_pairs
432
433 // Trap lowers to wasm unreachable
434 setOperationAction(ISD::TRAP, MVT::Other, Legal);
436
437 // Exception handling intrinsics
441
443
444 // Always convert switches to br_tables unless there is only one case, which
445 // is equivalent to a simple branch. This reduces code size for wasm, and we
446 // defer possible jump table optimizations to the VM.
448}
449
451WebAssemblyTargetLowering::shouldExpandAtomicRMWInIR(
452 const AtomicRMWInst *AI) const {
453 // We have wasm instructions for these
454 switch (AI->getOperation()) {
462 default:
463 break;
464 }
466}
467
468bool WebAssemblyTargetLowering::shouldScalarizeBinop(SDValue VecOp) const {
469 // Implementation copied from X86TargetLowering.
470 unsigned Opc = VecOp.getOpcode();
471
472 // Assume target opcodes can't be scalarized.
473 // TODO - do we have any exceptions?
475 return false;
476
477 // If the vector op is not supported, try to convert to scalar.
478 EVT VecVT = VecOp.getValueType();
480 return true;
481
482 // If the vector op is supported, but the scalar op is not, the transform may
483 // not be worthwhile.
484 EVT ScalarVT = VecVT.getScalarType();
485 return isOperationLegalOrCustomOrPromote(Opc, ScalarVT);
486}
487
488FastISel *WebAssemblyTargetLowering::createFastISel(
489 FunctionLoweringInfo &FuncInfo, const TargetLibraryInfo *LibInfo,
490 const LibcallLoweringInfo *LibcallLowering) const {
491 return WebAssembly::createFastISel(FuncInfo, LibInfo, LibcallLowering);
492}
493
494MVT WebAssemblyTargetLowering::getScalarShiftAmountTy(const DataLayout & /*DL*/,
495 EVT VT) const {
496 unsigned BitWidth = NextPowerOf2(VT.getSizeInBits() - 1);
497 if (BitWidth > 1 && BitWidth < 8)
498 BitWidth = 8;
499
500 if (BitWidth > 64) {
501 // The shift will be lowered to a libcall, and compiler-rt libcalls expect
502 // the count to be an i32.
503 BitWidth = 32;
505 "32-bit shift counts ought to be enough for anyone");
506 }
507
510 "Unable to represent scalar shift amount type");
511 return Result;
512}
513
514// Lower an fp-to-int conversion operator from the LLVM opcode, which has an
515// undefined result on invalid/overflow, to the WebAssembly opcode, which
516// traps on invalid/overflow.
519 const TargetInstrInfo &TII,
520 bool IsUnsigned, bool Int64,
521 bool Float64, unsigned LoweredOpcode) {
523
524 Register OutReg = MI.getOperand(0).getReg();
525 Register InReg = MI.getOperand(1).getReg();
526
527 unsigned Abs = Float64 ? WebAssembly::ABS_F64 : WebAssembly::ABS_F32;
528 unsigned FConst = Float64 ? WebAssembly::CONST_F64 : WebAssembly::CONST_F32;
529 unsigned LT = Float64 ? WebAssembly::LT_F64 : WebAssembly::LT_F32;
530 unsigned GE = Float64 ? WebAssembly::GE_F64 : WebAssembly::GE_F32;
531 unsigned IConst = Int64 ? WebAssembly::CONST_I64 : WebAssembly::CONST_I32;
532 unsigned Eqz = WebAssembly::EQZ_I32;
533 unsigned And = WebAssembly::AND_I32;
534 int64_t Limit = Int64 ? INT64_MIN : INT32_MIN;
535 int64_t Substitute = IsUnsigned ? 0 : Limit;
536 double CmpVal = IsUnsigned ? -(double)Limit * 2.0 : -(double)Limit;
537 auto &Context = BB->getParent()->getFunction().getContext();
538 Type *Ty = Float64 ? Type::getDoubleTy(Context) : Type::getFloatTy(Context);
539
540 const BasicBlock *LLVMBB = BB->getBasicBlock();
541 MachineFunction *F = BB->getParent();
542 MachineBasicBlock *TrueMBB = F->CreateMachineBasicBlock(LLVMBB);
543 MachineBasicBlock *FalseMBB = F->CreateMachineBasicBlock(LLVMBB);
544 MachineBasicBlock *DoneMBB = F->CreateMachineBasicBlock(LLVMBB);
545
547 F->insert(It, FalseMBB);
548 F->insert(It, TrueMBB);
549 F->insert(It, DoneMBB);
550
551 // Transfer the remainder of BB and its successor edges to DoneMBB.
552 DoneMBB->splice(DoneMBB->begin(), BB, std::next(MI.getIterator()), BB->end());
554
555 BB->addSuccessor(TrueMBB);
556 BB->addSuccessor(FalseMBB);
557 TrueMBB->addSuccessor(DoneMBB);
558 FalseMBB->addSuccessor(DoneMBB);
559
560 unsigned Tmp0, Tmp1, CmpReg, EqzReg, FalseReg, TrueReg;
561 Tmp0 = MRI.createVirtualRegister(MRI.getRegClass(InReg));
562 Tmp1 = MRI.createVirtualRegister(MRI.getRegClass(InReg));
563 CmpReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
564 EqzReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
565 FalseReg = MRI.createVirtualRegister(MRI.getRegClass(OutReg));
566 TrueReg = MRI.createVirtualRegister(MRI.getRegClass(OutReg));
567
568 MI.eraseFromParent();
569 // For signed numbers, we can do a single comparison to determine whether
570 // fabs(x) is within range.
571 if (IsUnsigned) {
572 Tmp0 = InReg;
573 } else {
574 BuildMI(BB, DL, TII.get(Abs), Tmp0).addReg(InReg);
575 }
576 BuildMI(BB, DL, TII.get(FConst), Tmp1)
577 .addFPImm(cast<ConstantFP>(ConstantFP::get(Ty, CmpVal)));
578 BuildMI(BB, DL, TII.get(LT), CmpReg).addReg(Tmp0).addReg(Tmp1);
579
580 // For unsigned numbers, we have to do a separate comparison with zero.
581 if (IsUnsigned) {
582 Tmp1 = MRI.createVirtualRegister(MRI.getRegClass(InReg));
583 Register SecondCmpReg =
584 MRI.createVirtualRegister(&WebAssembly::I32RegClass);
585 Register AndReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
586 BuildMI(BB, DL, TII.get(FConst), Tmp1)
587 .addFPImm(cast<ConstantFP>(ConstantFP::get(Ty, 0.0)));
588 BuildMI(BB, DL, TII.get(GE), SecondCmpReg).addReg(Tmp0).addReg(Tmp1);
589 BuildMI(BB, DL, TII.get(And), AndReg).addReg(CmpReg).addReg(SecondCmpReg);
590 CmpReg = AndReg;
591 }
592
593 BuildMI(BB, DL, TII.get(Eqz), EqzReg).addReg(CmpReg);
594
595 // Create the CFG diamond to select between doing the conversion or using
596 // the substitute value.
597 BuildMI(BB, DL, TII.get(WebAssembly::BR_IF)).addMBB(TrueMBB).addReg(EqzReg);
598 BuildMI(FalseMBB, DL, TII.get(LoweredOpcode), FalseReg).addReg(InReg);
599 BuildMI(FalseMBB, DL, TII.get(WebAssembly::BR)).addMBB(DoneMBB);
600 BuildMI(TrueMBB, DL, TII.get(IConst), TrueReg).addImm(Substitute);
601 BuildMI(*DoneMBB, DoneMBB->begin(), DL, TII.get(TargetOpcode::PHI), OutReg)
602 .addReg(FalseReg)
603 .addMBB(FalseMBB)
604 .addReg(TrueReg)
605 .addMBB(TrueMBB);
606
607 return DoneMBB;
608}
609
610// Lower a `MEMCPY` instruction into a CFG triangle around a `MEMORY_COPY`
611// instuction to handle the zero-length case.
614 const TargetInstrInfo &TII, bool Int64) {
616
617 MachineOperand DstMem = MI.getOperand(0);
618 MachineOperand SrcMem = MI.getOperand(1);
619 MachineOperand Dst = MI.getOperand(2);
620 MachineOperand Src = MI.getOperand(3);
621 MachineOperand Len = MI.getOperand(4);
622
623 // If the length is a constant, we don't actually need the check.
624 if (MachineInstr *Def = MRI.getVRegDef(Len.getReg())) {
625 if (Def->getOpcode() == WebAssembly::CONST_I32 ||
626 Def->getOpcode() == WebAssembly::CONST_I64) {
627 if (Def->getOperand(1).getImm() == 0) {
628 // A zero-length memcpy is a no-op.
629 MI.eraseFromParent();
630 return BB;
631 }
632 // A non-zero-length memcpy doesn't need a zero check.
633 unsigned MemoryCopy =
634 Int64 ? WebAssembly::MEMORY_COPY_A64 : WebAssembly::MEMORY_COPY_A32;
635 BuildMI(*BB, MI, DL, TII.get(MemoryCopy))
636 .add(DstMem)
637 .add(SrcMem)
638 .add(Dst)
639 .add(Src)
640 .add(Len);
641 MI.eraseFromParent();
642 return BB;
643 }
644 }
645
646 // We're going to add an extra use to `Len` to test if it's zero; that
647 // use shouldn't be a kill, even if the original use is.
648 MachineOperand NoKillLen = Len;
649 NoKillLen.setIsKill(false);
650
651 // Decide on which `MachineInstr` opcode we're going to use.
652 unsigned Eqz = Int64 ? WebAssembly::EQZ_I64 : WebAssembly::EQZ_I32;
653 unsigned MemoryCopy =
654 Int64 ? WebAssembly::MEMORY_COPY_A64 : WebAssembly::MEMORY_COPY_A32;
655
656 // Create two new basic blocks; one for the new `memory.fill` that we can
657 // branch over, and one for the rest of the instructions after the original
658 // `memory.fill`.
659 const BasicBlock *LLVMBB = BB->getBasicBlock();
660 MachineFunction *F = BB->getParent();
661 MachineBasicBlock *TrueMBB = F->CreateMachineBasicBlock(LLVMBB);
662 MachineBasicBlock *DoneMBB = F->CreateMachineBasicBlock(LLVMBB);
663
665 F->insert(It, TrueMBB);
666 F->insert(It, DoneMBB);
667
668 // Transfer the remainder of BB and its successor edges to DoneMBB.
669 DoneMBB->splice(DoneMBB->begin(), BB, std::next(MI.getIterator()), BB->end());
671
672 // Connect the CFG edges.
673 BB->addSuccessor(TrueMBB);
674 BB->addSuccessor(DoneMBB);
675 TrueMBB->addSuccessor(DoneMBB);
676
677 // Create a virtual register for the `Eqz` result.
678 unsigned EqzReg;
679 EqzReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
680
681 // Erase the original `memory.copy`.
682 MI.eraseFromParent();
683
684 // Test if `Len` is zero.
685 BuildMI(BB, DL, TII.get(Eqz), EqzReg).add(NoKillLen);
686
687 // Insert a new `memory.copy`.
688 BuildMI(TrueMBB, DL, TII.get(MemoryCopy))
689 .add(DstMem)
690 .add(SrcMem)
691 .add(Dst)
692 .add(Src)
693 .add(Len);
694
695 // Create the CFG triangle.
696 BuildMI(BB, DL, TII.get(WebAssembly::BR_IF)).addMBB(DoneMBB).addReg(EqzReg);
697 BuildMI(TrueMBB, DL, TII.get(WebAssembly::BR)).addMBB(DoneMBB);
698
699 return DoneMBB;
700}
701
702// Lower a `MEMSET` instruction into a CFG triangle around a `MEMORY_FILL`
703// instuction to handle the zero-length case.
706 const TargetInstrInfo &TII, bool Int64) {
708
709 MachineOperand Mem = MI.getOperand(0);
710 MachineOperand Dst = MI.getOperand(1);
711 MachineOperand Val = MI.getOperand(2);
712 MachineOperand Len = MI.getOperand(3);
713
714 // If the length is a constant, we don't actually need the check.
715 if (MachineInstr *Def = MRI.getVRegDef(Len.getReg())) {
716 if (Def->getOpcode() == WebAssembly::CONST_I32 ||
717 Def->getOpcode() == WebAssembly::CONST_I64) {
718 if (Def->getOperand(1).getImm() == 0) {
719 // A zero-length memset is a no-op.
720 MI.eraseFromParent();
721 return BB;
722 }
723 // A non-zero-length memset doesn't need a zero check.
724 unsigned MemoryFill =
725 Int64 ? WebAssembly::MEMORY_FILL_A64 : WebAssembly::MEMORY_FILL_A32;
726 BuildMI(*BB, MI, DL, TII.get(MemoryFill))
727 .add(Mem)
728 .add(Dst)
729 .add(Val)
730 .add(Len);
731 MI.eraseFromParent();
732 return BB;
733 }
734 }
735
736 // We're going to add an extra use to `Len` to test if it's zero; that
737 // use shouldn't be a kill, even if the original use is.
738 MachineOperand NoKillLen = Len;
739 NoKillLen.setIsKill(false);
740
741 // Decide on which `MachineInstr` opcode we're going to use.
742 unsigned Eqz = Int64 ? WebAssembly::EQZ_I64 : WebAssembly::EQZ_I32;
743 unsigned MemoryFill =
744 Int64 ? WebAssembly::MEMORY_FILL_A64 : WebAssembly::MEMORY_FILL_A32;
745
746 // Create two new basic blocks; one for the new `memory.fill` that we can
747 // branch over, and one for the rest of the instructions after the original
748 // `memory.fill`.
749 const BasicBlock *LLVMBB = BB->getBasicBlock();
750 MachineFunction *F = BB->getParent();
751 MachineBasicBlock *TrueMBB = F->CreateMachineBasicBlock(LLVMBB);
752 MachineBasicBlock *DoneMBB = F->CreateMachineBasicBlock(LLVMBB);
753
755 F->insert(It, TrueMBB);
756 F->insert(It, DoneMBB);
757
758 // Transfer the remainder of BB and its successor edges to DoneMBB.
759 DoneMBB->splice(DoneMBB->begin(), BB, std::next(MI.getIterator()), BB->end());
761
762 // Connect the CFG edges.
763 BB->addSuccessor(TrueMBB);
764 BB->addSuccessor(DoneMBB);
765 TrueMBB->addSuccessor(DoneMBB);
766
767 // Create a virtual register for the `Eqz` result.
768 unsigned EqzReg;
769 EqzReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
770
771 // Erase the original `memory.fill`.
772 MI.eraseFromParent();
773
774 // Test if `Len` is zero.
775 BuildMI(BB, DL, TII.get(Eqz), EqzReg).add(NoKillLen);
776
777 // Insert a new `memory.copy`.
778 BuildMI(TrueMBB, DL, TII.get(MemoryFill)).add(Mem).add(Dst).add(Val).add(Len);
779
780 // Create the CFG triangle.
781 BuildMI(BB, DL, TII.get(WebAssembly::BR_IF)).addMBB(DoneMBB).addReg(EqzReg);
782 BuildMI(TrueMBB, DL, TII.get(WebAssembly::BR)).addMBB(DoneMBB);
783
784 return DoneMBB;
785}
786
787static MachineBasicBlock *
789 const WebAssemblySubtarget *Subtarget,
790 const TargetInstrInfo &TII) {
791 MachineInstr &CallParams = *CallResults.getPrevNode();
792 assert(CallParams.getOpcode() == WebAssembly::CALL_PARAMS);
793 assert(CallResults.getOpcode() == WebAssembly::CALL_RESULTS ||
794 CallResults.getOpcode() == WebAssembly::RET_CALL_RESULTS);
795
796 bool IsIndirect =
797 CallParams.getOperand(0).isReg() || CallParams.getOperand(0).isFI();
798 bool IsRetCall = CallResults.getOpcode() == WebAssembly::RET_CALL_RESULTS;
799
800 bool IsFuncrefCall = false;
801 if (IsIndirect && CallParams.getOperand(0).isReg()) {
802 Register Reg = CallParams.getOperand(0).getReg();
803 const MachineFunction *MF = BB->getParent();
804 const MachineRegisterInfo &MRI = MF->getRegInfo();
805 const TargetRegisterClass *TRC = MRI.getRegClass(Reg);
806 IsFuncrefCall = (TRC == &WebAssembly::FUNCREFRegClass);
807 assert(!IsFuncrefCall || Subtarget->hasReferenceTypes());
808 }
809
810 unsigned CallOp;
811 if (IsIndirect && IsRetCall) {
812 CallOp = WebAssembly::RET_CALL_INDIRECT;
813 } else if (IsIndirect) {
814 CallOp = WebAssembly::CALL_INDIRECT;
815 } else if (IsRetCall) {
816 CallOp = WebAssembly::RET_CALL;
817 } else {
818 CallOp = WebAssembly::CALL;
819 }
820
821 MachineFunction &MF = *BB->getParent();
822 const MCInstrDesc &MCID = TII.get(CallOp);
823 MachineInstrBuilder MIB(MF, MF.CreateMachineInstr(MCID, DL));
824
825 // Move the function pointer to the end of the arguments for indirect calls
826 if (IsIndirect) {
827 auto FnPtr = CallParams.getOperand(0);
828 CallParams.removeOperand(0);
829
830 // For funcrefs, call_indirect is done through __funcref_call_table and the
831 // funcref is always installed in slot 0 of the table, therefore instead of
832 // having the function pointer added at the end of the params list, a zero
833 // (the index in
834 // __funcref_call_table is added).
835 if (IsFuncrefCall) {
836 Register RegZero =
837 MF.getRegInfo().createVirtualRegister(&WebAssembly::I32RegClass);
838 MachineInstrBuilder MIBC0 =
839 BuildMI(MF, DL, TII.get(WebAssembly::CONST_I32), RegZero).addImm(0);
840
841 BB->insert(CallResults.getIterator(), MIBC0);
842 MachineInstrBuilder(MF, CallParams).addReg(RegZero);
843 } else
844 CallParams.addOperand(FnPtr);
845 }
846
847 for (auto Def : CallResults.defs())
848 MIB.add(Def);
849
850 if (IsIndirect) {
851 // Placeholder for the type index.
852 // This gets replaced with the correct value in WebAssemblyMCInstLower.cpp
853 MIB.addImm(0);
854 // The table into which this call_indirect indexes.
855 MCSymbolWasm *Table = IsFuncrefCall
857 MF.getContext(), Subtarget)
859 MF.getContext(), Subtarget);
860 if (Subtarget->hasCallIndirectOverlong()) {
861 MIB.addSym(Table);
862 } else {
863 // For the MVP there is at most one table whose number is 0, but we can't
864 // write a table symbol or issue relocations. Instead we just ensure the
865 // table is live and write a zero.
866 Table->setNoStrip();
867 MIB.addImm(0);
868 }
869 }
870
871 for (auto Use : CallParams.uses())
872 MIB.add(Use);
873
874 BB->insert(CallResults.getIterator(), MIB);
875 CallParams.eraseFromParent();
876 CallResults.eraseFromParent();
877
878 // If this is a funcref call, to avoid hidden GC roots, we need to clear the
879 // table slot with ref.null upon call_indirect return.
880 //
881 // This generates the following code, which comes right after a call_indirect
882 // of a funcref:
883 //
884 // i32.const 0
885 // ref.null func
886 // table.set __funcref_call_table
887 if (IsIndirect && IsFuncrefCall) {
889 MF.getContext(), Subtarget);
890 Register RegZero =
891 MF.getRegInfo().createVirtualRegister(&WebAssembly::I32RegClass);
892 MachineInstr *Const0 =
893 BuildMI(MF, DL, TII.get(WebAssembly::CONST_I32), RegZero).addImm(0);
894 BB->insertAfter(MIB.getInstr()->getIterator(), Const0);
895
896 Register RegFuncref =
897 MF.getRegInfo().createVirtualRegister(&WebAssembly::FUNCREFRegClass);
898 MachineInstr *RefNull =
899 BuildMI(MF, DL, TII.get(WebAssembly::REF_NULL_FUNCREF), RegFuncref);
900 BB->insertAfter(Const0->getIterator(), RefNull);
901
902 MachineInstr *TableSet =
903 BuildMI(MF, DL, TII.get(WebAssembly::TABLE_SET_FUNCREF))
904 .addSym(Table)
905 .addReg(RegZero)
906 .addReg(RegFuncref);
907 BB->insertAfter(RefNull->getIterator(), TableSet);
908 }
909
910 return BB;
911}
912
913MachineBasicBlock *WebAssemblyTargetLowering::EmitInstrWithCustomInserter(
914 MachineInstr &MI, MachineBasicBlock *BB) const {
915 const TargetInstrInfo &TII = *Subtarget->getInstrInfo();
916 DebugLoc DL = MI.getDebugLoc();
917
918 switch (MI.getOpcode()) {
919 default:
920 llvm_unreachable("Unexpected instr type to insert");
921 case WebAssembly::FP_TO_SINT_I32_F32:
922 return LowerFPToInt(MI, DL, BB, TII, false, false, false,
923 WebAssembly::I32_TRUNC_S_F32);
924 case WebAssembly::FP_TO_UINT_I32_F32:
925 return LowerFPToInt(MI, DL, BB, TII, true, false, false,
926 WebAssembly::I32_TRUNC_U_F32);
927 case WebAssembly::FP_TO_SINT_I64_F32:
928 return LowerFPToInt(MI, DL, BB, TII, false, true, false,
929 WebAssembly::I64_TRUNC_S_F32);
930 case WebAssembly::FP_TO_UINT_I64_F32:
931 return LowerFPToInt(MI, DL, BB, TII, true, true, false,
932 WebAssembly::I64_TRUNC_U_F32);
933 case WebAssembly::FP_TO_SINT_I32_F64:
934 return LowerFPToInt(MI, DL, BB, TII, false, false, true,
935 WebAssembly::I32_TRUNC_S_F64);
936 case WebAssembly::FP_TO_UINT_I32_F64:
937 return LowerFPToInt(MI, DL, BB, TII, true, false, true,
938 WebAssembly::I32_TRUNC_U_F64);
939 case WebAssembly::FP_TO_SINT_I64_F64:
940 return LowerFPToInt(MI, DL, BB, TII, false, true, true,
941 WebAssembly::I64_TRUNC_S_F64);
942 case WebAssembly::FP_TO_UINT_I64_F64:
943 return LowerFPToInt(MI, DL, BB, TII, true, true, true,
944 WebAssembly::I64_TRUNC_U_F64);
945 case WebAssembly::MEMCPY_A32:
946 return LowerMemcpy(MI, DL, BB, TII, false);
947 case WebAssembly::MEMCPY_A64:
948 return LowerMemcpy(MI, DL, BB, TII, true);
949 case WebAssembly::MEMSET_A32:
950 return LowerMemset(MI, DL, BB, TII, false);
951 case WebAssembly::MEMSET_A64:
952 return LowerMemset(MI, DL, BB, TII, true);
953 case WebAssembly::CALL_RESULTS:
954 case WebAssembly::RET_CALL_RESULTS:
955 return LowerCallResults(MI, DL, BB, Subtarget, TII);
956 }
957}
958
959std::pair<unsigned, const TargetRegisterClass *>
960WebAssemblyTargetLowering::getRegForInlineAsmConstraint(
961 const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const {
962 // First, see if this is a constraint that directly corresponds to a
963 // WebAssembly register class.
964 if (Constraint.size() == 1) {
965 switch (Constraint[0]) {
966 case 'r':
967 assert(VT != MVT::iPTR && "Pointer MVT not expected here");
968 if (Subtarget->hasSIMD128() && VT.isVector()) {
969 if (VT.getSizeInBits() == 128)
970 return std::make_pair(0U, &WebAssembly::V128RegClass);
971 }
972 if (VT.isInteger() && !VT.isVector()) {
973 if (VT.getSizeInBits() <= 32)
974 return std::make_pair(0U, &WebAssembly::I32RegClass);
975 if (VT.getSizeInBits() <= 64)
976 return std::make_pair(0U, &WebAssembly::I64RegClass);
977 }
978 if (VT.isFloatingPoint() && !VT.isVector()) {
979 switch (VT.getSizeInBits()) {
980 case 32:
981 return std::make_pair(0U, &WebAssembly::F32RegClass);
982 case 64:
983 return std::make_pair(0U, &WebAssembly::F64RegClass);
984 default:
985 break;
986 }
987 }
988 break;
989 default:
990 break;
991 }
992 }
993
995}
996
997bool WebAssemblyTargetLowering::isCheapToSpeculateCttz(Type *Ty) const {
998 // Assume ctz is a relatively cheap operation.
999 return true;
1000}
1001
1002bool WebAssemblyTargetLowering::isCheapToSpeculateCtlz(Type *Ty) const {
1003 // Assume clz is a relatively cheap operation.
1004 return true;
1005}
1006
1007bool WebAssemblyTargetLowering::isLegalAddressingMode(const DataLayout &DL,
1008 const AddrMode &AM,
1009 Type *Ty, unsigned AS,
1010 Instruction *I) const {
1011 // WebAssembly offsets are added as unsigned without wrapping. The
1012 // isLegalAddressingMode gives us no way to determine if wrapping could be
1013 // happening, so we approximate this by accepting only non-negative offsets.
1014 if (AM.BaseOffs < 0)
1015 return false;
1016
1017 // WebAssembly has no scale register operands.
1018 if (AM.Scale != 0)
1019 return false;
1020
1021 // Everything else is legal.
1022 return true;
1023}
1024
1025bool WebAssemblyTargetLowering::allowsMisalignedMemoryAccesses(
1026 EVT /*VT*/, unsigned /*AddrSpace*/, Align /*Align*/,
1027 MachineMemOperand::Flags /*Flags*/, unsigned *Fast) const {
1028 // WebAssembly supports unaligned accesses, though it should be declared
1029 // with the p2align attribute on loads and stores which do so, and there
1030 // may be a performance impact. We tell LLVM they're "fast" because
1031 // for the kinds of things that LLVM uses this for (merging adjacent stores
1032 // of constants, etc.), WebAssembly implementations will either want the
1033 // unaligned access or they'll split anyway.
1034 if (Fast)
1035 *Fast = 1;
1036 return true;
1037}
1038
1039bool WebAssemblyTargetLowering::isIntDivCheap(EVT VT,
1040 AttributeList Attr) const {
1041 // The current thinking is that wasm engines will perform this optimization,
1042 // so we can save on code size.
1043 return true;
1044}
1045
1046bool WebAssemblyTargetLowering::isVectorLoadExtDesirable(SDValue ExtVal) const {
1047 EVT ExtT = ExtVal.getValueType();
1048 SDValue N0 = peekThroughFreeze(ExtVal->getOperand(0));
1049 auto *Load = dyn_cast<LoadSDNode>(N0);
1050 if (!Load)
1051 return false;
1052 EVT MemT = Load->getValueType(0);
1053 return (ExtT == MVT::v8i16 && MemT == MVT::v8i8) ||
1054 (ExtT == MVT::v4i32 && MemT == MVT::v4i16) ||
1055 (ExtT == MVT::v2i64 && MemT == MVT::v2i32);
1056}
1057
1058bool WebAssemblyTargetLowering::isOffsetFoldingLegal(
1059 const GlobalAddressSDNode *GA) const {
1060 // Wasm doesn't support function addresses with offsets
1061 const GlobalValue *GV = GA->getGlobal();
1063}
1064
1065EVT WebAssemblyTargetLowering::getSetCCResultType(const DataLayout &DL,
1066 LLVMContext &C,
1067 EVT VT) const {
1068 if (VT.isVector()) {
1069 if (VT.getVectorElementType() == MVT::f16 && !Subtarget->hasFP16())
1070 return VT.changeElementType(C, MVT::i1);
1071
1073 }
1074
1075 // So far, all branch instructions in Wasm take an I32 condition.
1076 // The default TargetLowering::getSetCCResultType returns the pointer size,
1077 // which would be useful to reduce instruction counts when testing
1078 // against 64-bit pointers/values if at some point Wasm supports that.
1079 return EVT::getIntegerVT(C, 32);
1080}
1081
1082void WebAssemblyTargetLowering::getTgtMemIntrinsic(
1084 MachineFunction &MF, unsigned Intrinsic) const {
1086 switch (Intrinsic) {
1087 case Intrinsic::wasm_memory_atomic_notify:
1089 Info.memVT = MVT::i32;
1090 Info.ptrVal = I.getArgOperand(0);
1091 Info.offset = 0;
1092 Info.align = Align(4);
1093 // atomic.notify instruction does not really load the memory specified with
1094 // this argument, but MachineMemOperand should either be load or store, so
1095 // we set this to a load.
1096 // FIXME Volatile isn't really correct, but currently all LLVM atomic
1097 // instructions are treated as volatiles in the backend, so we should be
1098 // consistent. The same applies for wasm_atomic_wait intrinsics too.
1100 Infos.push_back(Info);
1101 return;
1102 case Intrinsic::wasm_memory_atomic_wait32:
1104 Info.memVT = MVT::i32;
1105 Info.ptrVal = I.getArgOperand(0);
1106 Info.offset = 0;
1107 Info.align = Align(4);
1109 Infos.push_back(Info);
1110 return;
1111 case Intrinsic::wasm_memory_atomic_wait64:
1113 Info.memVT = MVT::i64;
1114 Info.ptrVal = I.getArgOperand(0);
1115 Info.offset = 0;
1116 Info.align = Align(8);
1118 Infos.push_back(Info);
1119 return;
1120 case Intrinsic::wasm_loadf16_f32:
1122 Info.memVT = MVT::f16;
1123 Info.ptrVal = I.getArgOperand(0);
1124 Info.offset = 0;
1125 Info.align = Align(2);
1127 Infos.push_back(Info);
1128 return;
1129 case Intrinsic::wasm_storef16_f32:
1131 Info.memVT = MVT::f16;
1132 Info.ptrVal = I.getArgOperand(1);
1133 Info.offset = 0;
1134 Info.align = Align(2);
1136 Infos.push_back(Info);
1137 return;
1138 default:
1139 return;
1140 }
1141}
1142
1143void WebAssemblyTargetLowering::computeKnownBitsForTargetNode(
1144 const SDValue Op, KnownBits &Known, const APInt &DemandedElts,
1145 const SelectionDAG &DAG, unsigned Depth) const {
1146 switch (Op.getOpcode()) {
1147 default:
1148 break;
1150 unsigned IntNo = Op.getConstantOperandVal(0);
1151 switch (IntNo) {
1152 default:
1153 break;
1154 case Intrinsic::wasm_bitmask: {
1155 unsigned BitWidth = Known.getBitWidth();
1156 EVT VT = Op.getOperand(1).getSimpleValueType();
1157 unsigned PossibleBits = VT.getVectorNumElements();
1158 APInt ZeroMask = APInt::getHighBitsSet(BitWidth, BitWidth - PossibleBits);
1159 Known.Zero |= ZeroMask;
1160 break;
1161 }
1162 }
1163 break;
1164 }
1165 case WebAssemblyISD::EXTEND_LOW_U:
1166 case WebAssemblyISD::EXTEND_HIGH_U: {
1167 // We know the high half, of each destination vector element, will be zero.
1168 SDValue SrcOp = Op.getOperand(0);
1169 EVT VT = SrcOp.getSimpleValueType();
1170 unsigned BitWidth = Known.getBitWidth();
1171 if (VT == MVT::v8i8 || VT == MVT::v16i8) {
1172 assert(BitWidth >= 8 && "Unexpected width!");
1174 Known.Zero |= Mask;
1175 } else if (VT == MVT::v4i16 || VT == MVT::v8i16) {
1176 assert(BitWidth >= 16 && "Unexpected width!");
1178 Known.Zero |= Mask;
1179 } else if (VT == MVT::v2i32 || VT == MVT::v4i32) {
1180 assert(BitWidth >= 32 && "Unexpected width!");
1182 Known.Zero |= Mask;
1183 }
1184 break;
1185 }
1186 // For 128-bit addition if the upper bits are all zero then it's known that
1187 // the upper bits of the result will have all bits guaranteed zero except the
1188 // first.
1189 case WebAssemblyISD::I64_ADD128:
1190 if (Op.getResNo() == 1) {
1191 SDValue LHS_HI = Op.getOperand(1);
1192 SDValue RHS_HI = Op.getOperand(3);
1193 if (isNullConstant(LHS_HI) && isNullConstant(RHS_HI))
1194 Known.Zero.setBitsFrom(1);
1195 }
1196 break;
1197 }
1198}
1199
1201WebAssemblyTargetLowering::getPreferredVectorAction(MVT VT) const {
1202 if (VT.isFixedLengthVector()) {
1203 MVT EltVT = VT.getVectorElementType();
1204 // We have legal vector types with these lane types, so widening the
1205 // vector would let us use some of the lanes directly without having to
1206 // extend or truncate values.
1207 if (EltVT == MVT::i8 || EltVT == MVT::i16 || EltVT == MVT::i32 ||
1208 EltVT == MVT::i64 || EltVT == MVT::f32 || EltVT == MVT::f64)
1209 return TypeWidenVector;
1210 }
1211
1213}
1214
1215bool WebAssemblyTargetLowering::isFMAFasterThanFMulAndFAdd(
1216 const MachineFunction &MF, EVT VT) const {
1217 if (!Subtarget->hasFP16() || !VT.isVector())
1218 return false;
1219
1220 EVT ScalarVT = VT.getScalarType();
1221 if (!ScalarVT.isSimple())
1222 return false;
1223
1224 return ScalarVT.getSimpleVT().SimpleTy == MVT::f16;
1225}
1226
1227bool WebAssemblyTargetLowering::shouldSimplifyDemandedVectorElts(
1228 SDValue Op, const TargetLoweringOpt &TLO) const {
1229 // ISel process runs DAGCombiner after legalization; this step is called
1230 // SelectionDAG optimization phase. This post-legalization combining process
1231 // runs DAGCombiner on each node, and if there was a change to be made,
1232 // re-runs legalization again on it and its user nodes to make sure
1233 // everythiing is in a legalized state.
1234 //
1235 // The legalization calls lowering routines, and we do our custom lowering for
1236 // build_vectors (LowerBUILD_VECTOR), which converts undef vector elements
1237 // into zeros. But there is a set of routines in DAGCombiner that turns unused
1238 // (= not demanded) nodes into undef, among which SimplifyDemandedVectorElts
1239 // turns unused vector elements into undefs. But this routine does not work
1240 // with our custom LowerBUILD_VECTOR, which turns undefs into zeros. This
1241 // combination can result in a infinite loop, in which undefs are converted to
1242 // zeros in legalization and back to undefs in combining.
1243 //
1244 // So after DAG is legalized, we prevent SimplifyDemandedVectorElts from
1245 // running for build_vectors.
1246 if (Op.getOpcode() == ISD::BUILD_VECTOR && TLO.LegalOps && TLO.LegalTys)
1247 return false;
1248 return true;
1249}
1250
1251//===----------------------------------------------------------------------===//
1252// WebAssembly Lowering private implementation.
1253//===----------------------------------------------------------------------===//
1254
1255//===----------------------------------------------------------------------===//
1256// Lowering Code
1257//===----------------------------------------------------------------------===//
1258
1259static void fail(const SDLoc &DL, SelectionDAG &DAG, const char *Msg) {
1261 DAG.getContext()->diagnose(
1262 DiagnosticInfoUnsupported(MF.getFunction(), Msg, DL.getDebugLoc()));
1263}
1264
1265// Test whether the given calling convention is supported.
1267 // We currently support the language-independent target-independent
1268 // conventions. We don't yet have a way to annotate calls with properties like
1269 // "cold", and we don't have any call-clobbered registers, so these are mostly
1270 // all handled the same.
1271 return CallConv == CallingConv::C || CallConv == CallingConv::Fast ||
1272 CallConv == CallingConv::Cold ||
1273 CallConv == CallingConv::PreserveMost ||
1274 CallConv == CallingConv::PreserveAll ||
1275 CallConv == CallingConv::CXX_FAST_TLS ||
1277 CallConv == CallingConv::Swift || CallConv == CallingConv::SwiftTail;
1278}
1279
1280SDValue
1281WebAssemblyTargetLowering::LowerCall(CallLoweringInfo &CLI,
1282 SmallVectorImpl<SDValue> &InVals) const {
1283 SelectionDAG &DAG = CLI.DAG;
1284 SDLoc DL = CLI.DL;
1285 SDValue Chain = CLI.Chain;
1286 SDValue Callee = CLI.Callee;
1287 MachineFunction &MF = DAG.getMachineFunction();
1288 auto Layout = MF.getDataLayout();
1289
1290 // A call through a funcref is expressed in IR as a call through the pointer
1291 // produced by the llvm.wasm.funcref.to_ptr intrinsic. Detect this here and
1292 // recover the underlying funcref value so the call can be lowered to a
1293 // table.set + call_indirect through the dedicated __funcref_call_table.
1294 bool IsFuncrefCall = false;
1295 if (Callee.getOpcode() == ISD::INTRINSIC_WO_CHAIN &&
1296 Callee.getConstantOperandVal(0) == Intrinsic::wasm_funcref_to_ptr) {
1297 Callee = Callee.getOperand(1);
1298 IsFuncrefCall = true;
1299 }
1300
1301 CallingConv::ID CallConv = CLI.CallConv;
1302 if (!callingConvSupported(CallConv))
1303 fail(DL, DAG,
1304 "WebAssembly doesn't support language-specific or target-specific "
1305 "calling conventions yet");
1306 if (CLI.IsPatchPoint)
1307 fail(DL, DAG, "WebAssembly doesn't support patch point yet");
1308
1309 if (CLI.IsTailCall) {
1310 auto NoTail = [&](const char *Msg) {
1311 if (CLI.CB && CLI.CB->isMustTailCall())
1312 fail(DL, DAG, Msg);
1313 CLI.IsTailCall = false;
1314 };
1315
1316 if (!Subtarget->hasTailCall())
1317 NoTail("WebAssembly 'tail-call' feature not enabled");
1318
1319 // Varargs calls cannot be tail calls because the buffer is on the stack
1320 if (CLI.IsVarArg)
1321 NoTail("WebAssembly does not support varargs tail calls");
1322
1323 // Do not tail call unless caller and callee return types match
1324 const Function &F = MF.getFunction();
1325 const TargetMachine &TM = getTargetMachine();
1326 Type *RetTy = F.getReturnType();
1327 SmallVector<MVT, 4> CallerRetTys;
1328 SmallVector<MVT, 4> CalleeRetTys;
1329 computeLegalValueVTs(F, TM, RetTy, CallerRetTys);
1330 computeLegalValueVTs(F, TM, CLI.RetTy, CalleeRetTys);
1331 bool TypesMatch = CallerRetTys.size() == CalleeRetTys.size() &&
1332 std::equal(CallerRetTys.begin(), CallerRetTys.end(),
1333 CalleeRetTys.begin());
1334 if (!TypesMatch)
1335 NoTail("WebAssembly tail call requires caller and callee return types to "
1336 "match");
1337
1338 // If pointers to local stack values are passed, we cannot tail call
1339 if (CLI.CB) {
1340 for (auto &Arg : CLI.CB->args()) {
1341 Value *Val = Arg.get();
1342 // Trace the value back through pointer operations
1343 while (true) {
1344 Value *Src = Val->stripPointerCastsAndAliases();
1345 if (auto *GEP = dyn_cast<GetElementPtrInst>(Src))
1346 Src = GEP->getPointerOperand();
1347 if (Val == Src)
1348 break;
1349 Val = Src;
1350 }
1351 if (isa<AllocaInst>(Val)) {
1352 NoTail(
1353 "WebAssembly does not support tail calling with stack arguments");
1354 break;
1355 }
1356 }
1357 }
1358 }
1359
1360 SmallVectorImpl<ISD::InputArg> &Ins = CLI.Ins;
1361 SmallVectorImpl<ISD::OutputArg> &Outs = CLI.Outs;
1362 SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
1363
1364 // The generic code may have added an sret argument. If we're lowering an
1365 // invoke function, the ABI requires that the function pointer be the first
1366 // argument, so we may have to swap the arguments.
1367 if (CallConv == CallingConv::WASM_EmscriptenInvoke && Outs.size() >= 2 &&
1368 Outs[0].Flags.isSRet()) {
1369 std::swap(Outs[0], Outs[1]);
1370 std::swap(OutVals[0], OutVals[1]);
1371 }
1372
1373 bool HasSwiftSelfArg = false;
1374 bool HasSwiftErrorArg = false;
1375 bool HasSwiftAsyncArg = false;
1376 unsigned NumFixedArgs = 0;
1377 for (unsigned I = 0; I < Outs.size(); ++I) {
1378 const ISD::OutputArg &Out = Outs[I];
1379 SDValue &OutVal = OutVals[I];
1380 HasSwiftSelfArg |= Out.Flags.isSwiftSelf();
1381 HasSwiftErrorArg |= Out.Flags.isSwiftError();
1382 HasSwiftAsyncArg |= Out.Flags.isSwiftAsync();
1383 if (Out.Flags.isNest())
1384 fail(DL, DAG, "WebAssembly hasn't implemented nest arguments");
1385 if (Out.Flags.isInAlloca())
1386 fail(DL, DAG, "WebAssembly hasn't implemented inalloca arguments");
1387 if (Out.Flags.isInConsecutiveRegs())
1388 fail(DL, DAG, "WebAssembly hasn't implemented cons regs arguments");
1390 fail(DL, DAG, "WebAssembly hasn't implemented cons regs last arguments");
1391 if (Out.Flags.isByVal() && Out.Flags.getByValSize() != 0) {
1392 auto &MFI = MF.getFrameInfo();
1393 int FI = MFI.CreateStackObject(Out.Flags.getByValSize(),
1395 /*isSS=*/false);
1396 SDValue SizeNode =
1397 DAG.getConstant(Out.Flags.getByValSize(), DL, MVT::i32);
1398 SDValue FINode = DAG.getFrameIndex(FI, getPointerTy(Layout));
1399 Align Alignment = Out.Flags.getNonZeroByValAlign();
1400 Chain = DAG.getMemcpy(Chain, DL, FINode, OutVal, SizeNode, Alignment,
1401 Alignment,
1402 /*isVolatile*/ false, /*AlwaysInline=*/false,
1403 /*CI=*/nullptr, std::nullopt, MachinePointerInfo(),
1404 MachinePointerInfo());
1405 OutVal = FINode;
1406 }
1407 // Count the number of fixed args *after* legalization.
1408 NumFixedArgs += !Out.Flags.isVarArg();
1409 }
1410
1411 bool IsVarArg = CLI.IsVarArg;
1412 auto PtrVT = getPointerTy(Layout);
1413
1414 // For swiftcc and swifttailcc, emit additional swiftself, swifterror, and
1415 // (for swifttailcc) swiftasync arguments if there aren't. These additional
1416 // arguments are also added for callee signature. They are necessary to match
1417 // callee and caller signature for indirect call.
1418 if (CallConv == CallingConv::Swift || CallConv == CallingConv::SwiftTail) {
1419 Type *PtrTy = PointerType::getUnqual(*DAG.getContext());
1420 if (!HasSwiftSelfArg) {
1421 NumFixedArgs++;
1422 ISD::ArgFlagsTy Flags;
1423 Flags.setSwiftSelf();
1424 ISD::OutputArg Arg(Flags, PtrVT, EVT(PtrVT), PtrTy, 0, 0);
1425 CLI.Outs.push_back(Arg);
1426 SDValue ArgVal = DAG.getUNDEF(PtrVT);
1427 CLI.OutVals.push_back(ArgVal);
1428 }
1429 if (!HasSwiftErrorArg) {
1430 NumFixedArgs++;
1431 ISD::ArgFlagsTy Flags;
1432 Flags.setSwiftError();
1433 ISD::OutputArg Arg(Flags, PtrVT, EVT(PtrVT), PtrTy, 0, 0);
1434 CLI.Outs.push_back(Arg);
1435 SDValue ArgVal = DAG.getUNDEF(PtrVT);
1436 CLI.OutVals.push_back(ArgVal);
1437 }
1438 if (CallConv == CallingConv::SwiftTail && !HasSwiftAsyncArg) {
1439 NumFixedArgs++;
1440 ISD::ArgFlagsTy Flags;
1441 Flags.setSwiftAsync();
1442 ISD::OutputArg Arg(Flags, PtrVT, EVT(PtrVT), PtrTy, 0, 0);
1443 CLI.Outs.push_back(Arg);
1444 SDValue ArgVal = DAG.getUNDEF(PtrVT);
1445 CLI.OutVals.push_back(ArgVal);
1446 }
1447 }
1448
1449 // Analyze operands of the call, assigning locations to each operand.
1451 CCState CCInfo(CallConv, IsVarArg, MF, ArgLocs, *DAG.getContext());
1452
1453 if (IsVarArg) {
1454 // Outgoing non-fixed arguments are placed in a buffer. First
1455 // compute their offsets and the total amount of buffer space needed.
1456 for (unsigned I = NumFixedArgs; I < Outs.size(); ++I) {
1457 const ISD::OutputArg &Out = Outs[I];
1458 SDValue &Arg = OutVals[I];
1459 EVT VT = Arg.getValueType();
1460 assert(VT != MVT::iPTR && "Legalized args should be concrete");
1461 Type *Ty = VT.getTypeForEVT(*DAG.getContext());
1462 Align Alignment =
1463 std::max(Out.Flags.getNonZeroOrigAlign(), Layout.getABITypeAlign(Ty));
1464 unsigned Offset =
1465 CCInfo.AllocateStack(Layout.getTypeAllocSize(Ty), Alignment);
1466 CCInfo.addLoc(CCValAssign::getMem(ArgLocs.size(), VT.getSimpleVT(),
1467 Offset, VT.getSimpleVT(),
1469 }
1470 }
1471
1472 unsigned NumBytes = CCInfo.getAlignedCallFrameSize();
1473
1474 SDValue FINode;
1475 if (IsVarArg && NumBytes) {
1476 // For non-fixed arguments, next emit stores to store the argument values
1477 // to the stack buffer at the offsets computed above.
1478 MaybeAlign StackAlign = Layout.getStackAlignment();
1479 assert(StackAlign && "data layout string is missing stack alignment");
1480 int FI = MF.getFrameInfo().CreateStackObject(NumBytes, *StackAlign,
1481 /*isSS=*/false);
1482 unsigned ValNo = 0;
1484 for (SDValue Arg : drop_begin(OutVals, NumFixedArgs)) {
1485 assert(ArgLocs[ValNo].getValNo() == ValNo &&
1486 "ArgLocs should remain in order and only hold varargs args");
1487 unsigned Offset = ArgLocs[ValNo++].getLocMemOffset();
1488 FINode = DAG.getFrameIndex(FI, getPointerTy(Layout));
1489 SDValue Add = DAG.getNode(ISD::ADD, DL, PtrVT, FINode,
1490 DAG.getConstant(Offset, DL, PtrVT));
1491 Chains.push_back(
1492 DAG.getStore(Chain, DL, Arg, Add,
1494 }
1495 if (!Chains.empty())
1496 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
1497 } else if (IsVarArg) {
1498 FINode = DAG.getIntPtrConstant(0, DL);
1499 }
1500
1501 if (Callee->getOpcode() == ISD::GlobalAddress) {
1502 // If the callee is a GlobalAddress node (quite common, every direct call
1503 // is) turn it into a TargetGlobalAddress node so that LowerGlobalAddress
1504 // doesn't at MO_GOT which is not needed for direct calls.
1505 GlobalAddressSDNode *GA = cast<GlobalAddressSDNode>(Callee);
1508 GA->getOffset());
1509 Callee = DAG.getNode(WebAssemblyISD::Wrapper, DL,
1510 getPointerTy(DAG.getDataLayout()), Callee);
1511 }
1512
1513 // Compute the operands for the CALLn node.
1515 Ops.push_back(Chain);
1516 Ops.push_back(Callee);
1517
1518 // Add all fixed arguments. Note that for non-varargs calls, NumFixedArgs
1519 // isn't reliable.
1520 Ops.append(OutVals.begin(),
1521 IsVarArg ? OutVals.begin() + NumFixedArgs : OutVals.end());
1522 // Add a pointer to the vararg buffer.
1523 if (IsVarArg)
1524 Ops.push_back(FINode);
1525
1526 SmallVector<EVT, 8> InTys;
1527 for (const auto &In : Ins) {
1528 assert(!In.Flags.isByVal() && "byval is not valid for return values");
1529 assert(!In.Flags.isNest() && "nest is not valid for return values");
1530 if (In.Flags.isInAlloca())
1531 fail(DL, DAG, "WebAssembly hasn't implemented inalloca return values");
1532 if (In.Flags.isInConsecutiveRegs())
1533 fail(DL, DAG, "WebAssembly hasn't implemented cons regs return values");
1534 if (In.Flags.isInConsecutiveRegsLast())
1535 fail(DL, DAG,
1536 "WebAssembly hasn't implemented cons regs last return values");
1537 // Ignore In.getNonZeroOrigAlign() because all our arguments are passed in
1538 // registers.
1539 InTys.push_back(In.VT);
1540 }
1541
1542 // Lastly, if this is a call to a funcref we need to add an instruction
1543 // table.set to the chain and transform the call.
1544 if (IsFuncrefCall) {
1545 // In the absence of function references proposal where a funcref call is
1546 // lowered to call_ref, using reference types we generate a table.set to set
1547 // the funcref to a special table used solely for this purpose, followed by
1548 // a call_indirect. Here we just generate the table set, and return the
1549 // SDValue of the table.set so that LowerCall can finalize the lowering by
1550 // generating the call_indirect.
1551 SDValue Chain = Ops[0];
1552
1554 MF.getContext(), Subtarget);
1555 SDValue Sym = DAG.getMCSymbol(Table, PtrVT);
1556 SDValue TableSlot = DAG.getConstant(0, DL, MVT::i32);
1557 SDValue TableSetOps[] = {Chain, Sym, TableSlot, Callee};
1558 SDValue TableSet = DAG.getMemIntrinsicNode(
1559 WebAssemblyISD::TABLE_SET, DL, DAG.getVTList(MVT::Other), TableSetOps,
1560 MVT::funcref, MachinePointerInfo(), Align(1),
1562
1563 Ops[0] = TableSet; // The new chain is the TableSet itself
1564 }
1565
1566 if (CLI.IsTailCall) {
1567 // ret_calls do not return values to the current frame
1568 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
1569 return DAG.getNode(WebAssemblyISD::RET_CALL, DL, NodeTys, Ops);
1570 }
1571
1572 InTys.push_back(MVT::Other);
1573 SDVTList InTyList = DAG.getVTList(InTys);
1574 SDValue Res = DAG.getNode(WebAssemblyISD::CALL, DL, InTyList, Ops);
1575
1576 for (size_t I = 0; I < Ins.size(); ++I)
1577 InVals.push_back(Res.getValue(I));
1578
1579 // Return the chain
1580 return Res.getValue(Ins.size());
1581}
1582
1583bool WebAssemblyTargetLowering::CanLowerReturn(
1584 CallingConv::ID /*CallConv*/, MachineFunction & /*MF*/, bool /*IsVarArg*/,
1585 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext & /*Context*/,
1586 const Type *RetTy) const {
1587 // WebAssembly can only handle returning tuples with multivalue enabled
1588 return WebAssembly::canLowerReturn(Outs.size(), Subtarget);
1589}
1590
1591SDValue WebAssemblyTargetLowering::LowerReturn(
1592 SDValue Chain, CallingConv::ID CallConv, bool /*IsVarArg*/,
1594 const SmallVectorImpl<SDValue> &OutVals, const SDLoc &DL,
1595 SelectionDAG &DAG) const {
1596 assert(WebAssembly::canLowerReturn(Outs.size(), Subtarget) &&
1597 "MVP WebAssembly can only return up to one value");
1598 if (!callingConvSupported(CallConv))
1599 fail(DL, DAG, "WebAssembly doesn't support non-C calling conventions");
1600
1601 SmallVector<SDValue, 4> RetOps(1, Chain);
1602 RetOps.append(OutVals.begin(), OutVals.end());
1603 Chain = DAG.getNode(WebAssemblyISD::RETURN, DL, MVT::Other, RetOps);
1604
1605 // Record the number and types of the return values.
1606 for (const ISD::OutputArg &Out : Outs) {
1607 assert(!Out.Flags.isByVal() && "byval is not valid for return values");
1608 assert(!Out.Flags.isNest() && "nest is not valid for return values");
1609 assert(!Out.Flags.isVarArg() && "non-fixed return value is not valid");
1610 if (Out.Flags.isInAlloca())
1611 fail(DL, DAG, "WebAssembly hasn't implemented inalloca results");
1612 if (Out.Flags.isInConsecutiveRegs())
1613 fail(DL, DAG, "WebAssembly hasn't implemented cons regs results");
1615 fail(DL, DAG, "WebAssembly hasn't implemented cons regs last results");
1616 }
1617
1618 return Chain;
1619}
1620
1621SDValue WebAssemblyTargetLowering::LowerFormalArguments(
1622 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
1623 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
1624 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
1625 if (!callingConvSupported(CallConv))
1626 fail(DL, DAG, "WebAssembly doesn't support non-C calling conventions");
1627
1628 MachineFunction &MF = DAG.getMachineFunction();
1629 auto *MFI = MF.getInfo<WebAssemblyFunctionInfo>();
1630
1631 // Set up the incoming ARGUMENTS value, which serves to represent the liveness
1632 // of the incoming values before they're represented by virtual registers.
1633 MF.getRegInfo().addLiveIn(WebAssembly::ARGUMENTS);
1634
1635 bool HasSwiftErrorArg = false;
1636 bool HasSwiftSelfArg = false;
1637 bool HasSwiftAsyncArg = false;
1638 for (const ISD::InputArg &In : Ins) {
1639 HasSwiftSelfArg |= In.Flags.isSwiftSelf();
1640 HasSwiftErrorArg |= In.Flags.isSwiftError();
1641 HasSwiftAsyncArg |= In.Flags.isSwiftAsync();
1642 if (In.Flags.isInAlloca())
1643 fail(DL, DAG, "WebAssembly hasn't implemented inalloca arguments");
1644 if (In.Flags.isNest())
1645 fail(DL, DAG, "WebAssembly hasn't implemented nest arguments");
1646 if (In.Flags.isInConsecutiveRegs())
1647 fail(DL, DAG, "WebAssembly hasn't implemented cons regs arguments");
1648 if (In.Flags.isInConsecutiveRegsLast())
1649 fail(DL, DAG, "WebAssembly hasn't implemented cons regs last arguments");
1650 // Ignore In.getNonZeroOrigAlign() because all our arguments are passed in
1651 // registers.
1652 InVals.push_back(In.Used ? DAG.getNode(WebAssemblyISD::ARGUMENT, DL, In.VT,
1653 DAG.getTargetConstant(InVals.size(),
1654 DL, MVT::i32))
1655 : DAG.getUNDEF(In.VT));
1656
1657 // Record the number and types of arguments.
1658 MFI->addParam(In.VT);
1659 }
1660
1661 // For swiftcc and swifttailcc, emit additional swiftself, swifterror, and
1662 // (for swifttailcc) swiftasync arguments if there aren't. These additional
1663 // arguments are also added for callee signature. They are necessary to match
1664 // callee and caller signature for indirect call.
1665 auto PtrVT = getPointerTy(MF.getDataLayout());
1666 if (CallConv == CallingConv::Swift || CallConv == CallingConv::SwiftTail) {
1667 if (!HasSwiftSelfArg) {
1668 MFI->addParam(PtrVT);
1669 }
1670 if (!HasSwiftErrorArg) {
1671 MFI->addParam(PtrVT);
1672 }
1673 if (CallConv == CallingConv::SwiftTail && !HasSwiftAsyncArg) {
1674 MFI->addParam(PtrVT);
1675 }
1676 }
1677 // Varargs are copied into a buffer allocated by the caller, and a pointer to
1678 // the buffer is passed as an argument.
1679 if (IsVarArg) {
1680 MVT PtrVT = getPointerTy(MF.getDataLayout());
1681 Register VarargVreg =
1683 MFI->setVarargBufferVreg(VarargVreg);
1684 Chain = DAG.getCopyToReg(
1685 Chain, DL, VarargVreg,
1686 DAG.getNode(WebAssemblyISD::ARGUMENT, DL, PtrVT,
1687 DAG.getTargetConstant(Ins.size(), DL, MVT::i32)));
1688 MFI->addParam(PtrVT);
1689 }
1690
1691 // Record the number and types of arguments and results.
1692 SmallVector<MVT, 4> Params;
1695 MF.getFunction(), DAG.getTarget(), Params, Results);
1696 for (MVT VT : Results)
1697 MFI->addResult(VT);
1698 // TODO: Use signatures in WebAssemblyMachineFunctionInfo too and unify
1699 // the param logic here with ComputeSignatureVTs
1700 assert(MFI->getParams().size() == Params.size() &&
1701 std::equal(MFI->getParams().begin(), MFI->getParams().end(),
1702 Params.begin()));
1703
1704 return Chain;
1705}
1706
1707void WebAssemblyTargetLowering::ReplaceNodeResults(
1709 switch (N->getOpcode()) {
1711 // Do not add any results, signifying that N should not be custom lowered
1712 // after all. This happens because simd128 turns on custom lowering for
1713 // SIGN_EXTEND_INREG, but for non-vector sign extends the result might be an
1714 // illegal type.
1715 break;
1719 // Do not add any results, signifying that N should not be custom lowered.
1720 // EXTEND_VECTOR_INREG is implemented for some vectors, but not all.
1721 break;
1722 case ISD::FP_ROUND: {
1723 EVT VT = N->getValueType(0);
1724 SDValue Src = N->getOperand(0);
1725 if (VT == MVT::v4f16 && Src.getValueType() == MVT::v4f32) {
1726 Results.push_back(
1727 DAG.getNode(WebAssemblyISD::DEMOTE_ZERO, SDLoc(N), MVT::v8f16, Src));
1728 }
1729 break;
1730 }
1731 case ISD::ADD:
1732 case ISD::SUB:
1733 Results.push_back(Replace128Op(N, DAG));
1734 break;
1735 default:
1737 "ReplaceNodeResults not implemented for this op for WebAssembly!");
1738 }
1739}
1740
1741//===----------------------------------------------------------------------===//
1742// Custom lowering hooks.
1743//===----------------------------------------------------------------------===//
1744
1745SDValue WebAssemblyTargetLowering::LowerOperation(SDValue Op,
1746 SelectionDAG &DAG) const {
1747 SDLoc DL(Op);
1748 switch (Op.getOpcode()) {
1749 default:
1750 llvm_unreachable("unimplemented operation lowering");
1751 return SDValue();
1752 case ISD::FrameIndex:
1753 return LowerFrameIndex(Op, DAG);
1754 case ISD::GlobalAddress:
1755 return LowerGlobalAddress(Op, DAG);
1757 return LowerGlobalTLSAddress(Op, DAG);
1759 return LowerExternalSymbol(Op, DAG);
1760 case ISD::JumpTable:
1761 return LowerJumpTable(Op, DAG);
1762 case ISD::BR_JT:
1763 return LowerBR_JT(Op, DAG);
1764 case ISD::VASTART:
1765 return LowerVASTART(Op, DAG);
1766 case ISD::BlockAddress:
1767 case ISD::BRIND:
1768 fail(DL, DAG, "WebAssembly hasn't implemented computed gotos");
1769 return SDValue();
1770 case ISD::RETURNADDR:
1771 return LowerRETURNADDR(Op, DAG);
1772 case ISD::FRAMEADDR:
1773 return LowerFRAMEADDR(Op, DAG);
1774 case ISD::CopyToReg:
1775 return LowerCopyToReg(Op, DAG);
1778 return LowerAccessVectorElement(Op, DAG);
1782 return LowerIntrinsic(Op, DAG);
1784 return LowerSIGN_EXTEND_INREG(Op, DAG);
1788 return LowerEXTEND_VECTOR_INREG(Op, DAG);
1789 case ISD::BUILD_VECTOR:
1790 return LowerBUILD_VECTOR(Op, DAG);
1792 return LowerVECTOR_SHUFFLE(Op, DAG);
1793 case ISD::SETCC:
1794 return LowerSETCC(Op, DAG);
1795 case ISD::SHL:
1796 case ISD::SRA:
1797 case ISD::SRL:
1798 return LowerShift(Op, DAG);
1801 return LowerFP_TO_INT_SAT(Op, DAG);
1802 case ISD::FMINNUM:
1803 case ISD::FMINIMUMNUM:
1804 return LowerFMIN(Op, DAG);
1805 case ISD::FMAXNUM:
1806 case ISD::FMAXIMUMNUM:
1807 return LowerFMAX(Op, DAG);
1808 case ISD::LOAD:
1809 return LowerLoad(Op, DAG);
1810 case ISD::STORE:
1811 return LowerStore(Op, DAG);
1812 case ISD::CTPOP:
1813 case ISD::CTLZ:
1814 case ISD::CTTZ:
1815 return DAG.UnrollVectorOp(Op.getNode());
1816 case ISD::CLEAR_CACHE:
1817 report_fatal_error("llvm.clear_cache is not supported on wasm");
1818 case ISD::SMUL_LOHI:
1819 case ISD::UMUL_LOHI:
1820 return LowerMUL_LOHI(Op, DAG);
1821 case ISD::UADDO:
1822 return LowerUADDO(Op, DAG);
1823 }
1824}
1825
1829
1830 return false;
1831}
1832
1833static std::optional<unsigned> IsWebAssemblyLocal(SDValue Op,
1834 SelectionDAG &DAG) {
1836 if (!FI)
1837 return std::nullopt;
1838
1839 auto &MF = DAG.getMachineFunction();
1841}
1842
1843SDValue WebAssemblyTargetLowering::LowerStore(SDValue Op,
1844 SelectionDAG &DAG) const {
1845 SDLoc DL(Op);
1846 StoreSDNode *SN = cast<StoreSDNode>(Op.getNode());
1847 const SDValue &Value = SN->getValue();
1848 const SDValue &Base = SN->getBasePtr();
1849 const SDValue &Offset = SN->getOffset();
1850
1852 if (!Offset->isUndef())
1853 report_fatal_error("unexpected offset when storing to webassembly global",
1854 false);
1855
1856 SDVTList Tys = DAG.getVTList(MVT::Other);
1857 SDValue Ops[] = {SN->getChain(), Value, Base};
1858 return DAG.getMemIntrinsicNode(WebAssemblyISD::GLOBAL_SET, DL, Tys, Ops,
1859 SN->getMemoryVT(), SN->getMemOperand());
1860 }
1861
1862 if (std::optional<unsigned> Local = IsWebAssemblyLocal(Base, DAG)) {
1863 if (!Offset->isUndef())
1864 report_fatal_error("unexpected offset when storing to webassembly local",
1865 false);
1866
1867 SDValue Idx = DAG.getTargetConstant(*Local, Base, MVT::i32);
1868 SDVTList Tys = DAG.getVTList(MVT::Other); // The chain.
1869 SDValue Ops[] = {SN->getChain(), Idx, Value};
1870 return DAG.getNode(WebAssemblyISD::LOCAL_SET, DL, Tys, Ops);
1871 }
1872
1875 "Encountered an unlowerable store to the wasm_var address space",
1876 false);
1877
1878 return Op;
1879}
1880
1881SDValue WebAssemblyTargetLowering::LowerLoad(SDValue Op,
1882 SelectionDAG &DAG) const {
1883 SDLoc DL(Op);
1884 LoadSDNode *LN = cast<LoadSDNode>(Op.getNode());
1885 const SDValue &Base = LN->getBasePtr();
1886 const SDValue &Offset = LN->getOffset();
1887
1889 if (!Offset->isUndef())
1891 "unexpected offset when loading from webassembly global", false);
1892
1893 SDVTList Tys = DAG.getVTList(LN->getValueType(0), MVT::Other);
1894 SDValue Ops[] = {LN->getChain(), Base};
1895 return DAG.getMemIntrinsicNode(WebAssemblyISD::GLOBAL_GET, DL, Tys, Ops,
1896 LN->getMemoryVT(), LN->getMemOperand());
1897 }
1898
1899 if (std::optional<unsigned> Local = IsWebAssemblyLocal(Base, DAG)) {
1900 if (!Offset->isUndef())
1902 "unexpected offset when loading from webassembly local", false);
1903
1904 SDValue Idx = DAG.getTargetConstant(*Local, Base, MVT::i32);
1905 EVT LocalVT = LN->getValueType(0);
1906 return DAG.getNode(WebAssemblyISD::LOCAL_GET, DL, {LocalVT, MVT::Other},
1907 {LN->getChain(), Idx});
1908 }
1909
1912 "Encountered an unlowerable load from the wasm_var address space",
1913 false);
1914
1915 return Op;
1916}
1917
1918SDValue WebAssemblyTargetLowering::LowerMUL_LOHI(SDValue Op,
1919 SelectionDAG &DAG) const {
1920 assert(Subtarget->hasWideArithmetic());
1921 assert(Op.getValueType() == MVT::i64);
1922 SDLoc DL(Op);
1923 unsigned Opcode;
1924 switch (Op.getOpcode()) {
1925 case ISD::UMUL_LOHI:
1926 Opcode = WebAssemblyISD::I64_MUL_WIDE_U;
1927 break;
1928 case ISD::SMUL_LOHI:
1929 Opcode = WebAssemblyISD::I64_MUL_WIDE_S;
1930 break;
1931 default:
1932 llvm_unreachable("unexpected opcode");
1933 }
1934 SDValue LHS = Op.getOperand(0);
1935 SDValue RHS = Op.getOperand(1);
1936 SDValue Lo =
1937 DAG.getNode(Opcode, DL, DAG.getVTList(MVT::i64, MVT::i64), LHS, RHS);
1938 SDValue Hi(Lo.getNode(), 1);
1939 SDValue Ops[] = {Lo, Hi};
1940 return DAG.getMergeValues(Ops, DL);
1941}
1942
1943// Lowers `UADDO` intrinsics to an `i64.add128` instruction when it's enabled.
1944//
1945// This enables generating a single wasm instruction for this operation where
1946// the upper half of both operands are constant zeros. The upper half of the
1947// result is then whether the overflow happened.
1948SDValue WebAssemblyTargetLowering::LowerUADDO(SDValue Op,
1949 SelectionDAG &DAG) const {
1950 assert(Subtarget->hasWideArithmetic());
1951 assert(Op.getValueType() == MVT::i64);
1952 assert(Op.getOpcode() == ISD::UADDO);
1953 SDLoc DL(Op);
1954 SDValue LHS = Op.getOperand(0);
1955 SDValue RHS = Op.getOperand(1);
1956 SDValue Zero = DAG.getConstant(0, DL, MVT::i64);
1957 SDValue Result =
1958 DAG.getNode(WebAssemblyISD::I64_ADD128, DL,
1959 DAG.getVTList(MVT::i64, MVT::i64), LHS, Zero, RHS, Zero);
1960 SDValue CarryI64(Result.getNode(), 1);
1961 SDValue CarryI32 = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, CarryI64);
1962 SDValue Ops[] = {Result, CarryI32};
1963 return DAG.getMergeValues(Ops, DL);
1964}
1965
1966SDValue WebAssemblyTargetLowering::Replace128Op(SDNode *N,
1967 SelectionDAG &DAG) const {
1968 assert(Subtarget->hasWideArithmetic());
1969 assert(N->getValueType(0) == MVT::i128);
1970 SDLoc DL(N);
1971 unsigned Opcode;
1972 switch (N->getOpcode()) {
1973 case ISD::ADD:
1974 Opcode = WebAssemblyISD::I64_ADD128;
1975 break;
1976 case ISD::SUB:
1977 Opcode = WebAssemblyISD::I64_SUB128;
1978 break;
1979 default:
1980 llvm_unreachable("unexpected opcode");
1981 }
1982 SDValue LHS = N->getOperand(0);
1983 SDValue RHS = N->getOperand(1);
1984
1985 SDValue C0 = DAG.getConstant(0, DL, MVT::i64);
1986 SDValue C1 = DAG.getConstant(1, DL, MVT::i64);
1987 SDValue LHS_0 = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, LHS, C0);
1988 SDValue LHS_1 = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, LHS, C1);
1989 SDValue RHS_0 = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, RHS, C0);
1990 SDValue RHS_1 = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, RHS, C1);
1991 SDValue Result_LO = DAG.getNode(Opcode, DL, DAG.getVTList(MVT::i64, MVT::i64),
1992 LHS_0, LHS_1, RHS_0, RHS_1);
1993 SDValue Result_HI(Result_LO.getNode(), 1);
1994 return DAG.getNode(ISD::BUILD_PAIR, DL, N->getVTList(), Result_LO, Result_HI);
1995}
1996
1997SDValue WebAssemblyTargetLowering::LowerCopyToReg(SDValue Op,
1998 SelectionDAG &DAG) const {
1999 SDValue Src = Op.getOperand(2);
2000 if (isa<FrameIndexSDNode>(Src.getNode())) {
2001 // CopyToReg nodes don't support FrameIndex operands. Other targets select
2002 // the FI to some LEA-like instruction, but since we don't have that, we
2003 // need to insert some kind of instruction that can take an FI operand and
2004 // produces a value usable by CopyToReg (i.e. in a vreg). So insert a dummy
2005 // local.copy between Op and its FI operand.
2006 SDValue Chain = Op.getOperand(0);
2007 SDLoc DL(Op);
2008 Register Reg = cast<RegisterSDNode>(Op.getOperand(1))->getReg();
2009 EVT VT = Src.getValueType();
2010 SDValue Copy(DAG.getMachineNode(VT == MVT::i32 ? WebAssembly::COPY_I32
2011 : WebAssembly::COPY_I64,
2012 DL, VT, Src),
2013 0);
2014 return Op.getNode()->getNumValues() == 1
2015 ? DAG.getCopyToReg(Chain, DL, Reg, Copy)
2016 : DAG.getCopyToReg(Chain, DL, Reg, Copy,
2017 Op.getNumOperands() == 4 ? Op.getOperand(3)
2018 : SDValue());
2019 }
2020 return SDValue();
2021}
2022
2023SDValue WebAssemblyTargetLowering::LowerFrameIndex(SDValue Op,
2024 SelectionDAG &DAG) const {
2025 int FI = cast<FrameIndexSDNode>(Op)->getIndex();
2026 return DAG.getTargetFrameIndex(FI, Op.getValueType());
2027}
2028
2029SDValue WebAssemblyTargetLowering::LowerRETURNADDR(SDValue Op,
2030 SelectionDAG &DAG) const {
2031 SDLoc DL(Op);
2032
2033 if (!Subtarget->getTargetTriple().isOSEmscripten()) {
2034 fail(DL, DAG,
2035 "Non-Emscripten WebAssembly hasn't implemented "
2036 "__builtin_return_address");
2037 return SDValue();
2038 }
2039
2040 unsigned Depth = Op.getConstantOperandVal(0);
2041 MakeLibCallOptions CallOptions;
2042 return makeLibCall(DAG, RTLIB::RETURN_ADDRESS, Op.getValueType(),
2043 {DAG.getConstant(Depth, DL, MVT::i32)}, CallOptions, DL)
2044 .first;
2045}
2046
2047SDValue WebAssemblyTargetLowering::LowerFRAMEADDR(SDValue Op,
2048 SelectionDAG &DAG) const {
2049 // Non-zero depths are not supported by WebAssembly currently. Use the
2050 // legalizer's default expansion, which is to return 0 (what this function is
2051 // documented to do).
2052 if (Op.getConstantOperandVal(0) > 0)
2053 return SDValue();
2054
2056 EVT VT = Op.getValueType();
2057 Register FP =
2058 Subtarget->getRegisterInfo()->getFrameRegister(DAG.getMachineFunction());
2059 return DAG.getCopyFromReg(DAG.getEntryNode(), SDLoc(Op), FP, VT);
2060}
2061
2062SDValue
2063WebAssemblyTargetLowering::LowerGlobalTLSAddress(SDValue Op,
2064 SelectionDAG &DAG) const {
2065 SDLoc DL(Op);
2066 const auto *GA = cast<GlobalAddressSDNode>(Op);
2067
2068 MachineFunction &MF = DAG.getMachineFunction();
2069 if (!MF.getSubtarget<WebAssemblySubtarget>().hasBulkMemory())
2070 report_fatal_error("cannot use thread-local storage without bulk memory",
2071 false);
2072
2073 const GlobalValue *GV = GA->getGlobal();
2074
2075 // Currently only Emscripten supports dynamic linking with threads. Therefore,
2076 // on other targets, if we have thread-local storage, only the local-exec
2077 // model is possible.
2078 auto model = Subtarget->getTargetTriple().isOSEmscripten()
2079 ? GV->getThreadLocalMode()
2081
2082 // Unsupported TLS modes
2085
2086 if (model == GlobalValue::LocalExecTLSModel ||
2089 getTargetMachine().shouldAssumeDSOLocal(GV))) {
2090 // For DSO-local TLS variables we use offset from __tls_base, or
2091 // __wasm_get_tls_base() if using libcall thread context.
2092
2093 MVT PtrVT = getPointerTy(DAG.getDataLayout());
2094 SDValue BaseAddr(WebAssembly::getTLSBase(DAG, DL, Subtarget), 0);
2095
2096 SDValue TLSOffset = DAG.getTargetGlobalAddress(
2097 GV, DL, PtrVT, GA->getOffset(), WebAssemblyII::MO_TLS_BASE_REL);
2098 SDValue SymOffset =
2099 DAG.getNode(WebAssemblyISD::WrapperREL, DL, PtrVT, TLSOffset);
2100
2101 return DAG.getNode(ISD::ADD, DL, PtrVT, BaseAddr, SymOffset);
2102 }
2103
2105
2106 EVT VT = Op.getValueType();
2107 return DAG.getNode(WebAssemblyISD::Wrapper, DL, VT,
2108 DAG.getTargetGlobalAddress(GA->getGlobal(), DL, VT,
2109 GA->getOffset(),
2111}
2112
2113SDValue WebAssemblyTargetLowering::LowerGlobalAddress(SDValue Op,
2114 SelectionDAG &DAG) const {
2115 SDLoc DL(Op);
2116 const auto *GA = cast<GlobalAddressSDNode>(Op);
2117 EVT VT = Op.getValueType();
2118 assert(GA->getTargetFlags() == 0 &&
2119 "Unexpected target flags on generic GlobalAddressSDNode");
2121 fail(DL, DAG, "Invalid address space for WebAssembly target");
2122
2123 unsigned OperandFlags = 0;
2124 const GlobalValue *GV = GA->getGlobal();
2125 // Since WebAssembly tables cannot yet be shared accross modules, we don't
2126 // need special treatment for tables in PIC mode.
2127 if (isPositionIndependent() &&
2129 if (getTargetMachine().shouldAssumeDSOLocal(GV)) {
2130 MachineFunction &MF = DAG.getMachineFunction();
2131 MVT PtrVT = getPointerTy(MF.getDataLayout());
2132 const char *BaseName;
2133 if (GV->getValueType()->isFunctionTy()) {
2134 BaseName = MF.createExternalSymbolName("__table_base");
2136 } else {
2137 BaseName = MF.createExternalSymbolName("__memory_base");
2139 }
2141 DAG.getNode(WebAssemblyISD::Wrapper, DL, PtrVT,
2142 DAG.getTargetExternalSymbol(BaseName, PtrVT));
2143
2144 SDValue SymAddr = DAG.getNode(
2145 WebAssemblyISD::WrapperREL, DL, VT,
2146 DAG.getTargetGlobalAddress(GA->getGlobal(), DL, VT, GA->getOffset(),
2147 OperandFlags));
2148
2149 return DAG.getNode(ISD::ADD, DL, VT, BaseAddr, SymAddr);
2150 }
2152 }
2153
2154 return DAG.getNode(WebAssemblyISD::Wrapper, DL, VT,
2155 DAG.getTargetGlobalAddress(GA->getGlobal(), DL, VT,
2156 GA->getOffset(), OperandFlags));
2157}
2158
2159SDValue
2160WebAssemblyTargetLowering::LowerExternalSymbol(SDValue Op,
2161 SelectionDAG &DAG) const {
2162 SDLoc DL(Op);
2163 const auto *ES = cast<ExternalSymbolSDNode>(Op);
2164 EVT VT = Op.getValueType();
2165 assert(ES->getTargetFlags() == 0 &&
2166 "Unexpected target flags on generic ExternalSymbolSDNode");
2167 return DAG.getNode(WebAssemblyISD::Wrapper, DL, VT,
2168 DAG.getTargetExternalSymbol(ES->getSymbol(), VT));
2169}
2170
2171SDValue WebAssemblyTargetLowering::LowerJumpTable(SDValue Op,
2172 SelectionDAG &DAG) const {
2173 // There's no need for a Wrapper node because we always incorporate a jump
2174 // table operand into a BR_TABLE instruction, rather than ever
2175 // materializing it in a register.
2176 const JumpTableSDNode *JT = cast<JumpTableSDNode>(Op);
2177 return DAG.getTargetJumpTable(JT->getIndex(), Op.getValueType(),
2178 JT->getTargetFlags());
2179}
2180
2181SDValue WebAssemblyTargetLowering::LowerBR_JT(SDValue Op,
2182 SelectionDAG &DAG) const {
2183 SDLoc DL(Op);
2184 SDValue Chain = Op.getOperand(0);
2185 const auto *JT = cast<JumpTableSDNode>(Op.getOperand(1));
2186 SDValue Index = Op.getOperand(2);
2187 assert(JT->getTargetFlags() == 0 && "WebAssembly doesn't set target flags");
2188
2190 Ops.push_back(Chain);
2191 Ops.push_back(Index);
2192
2193 MachineJumpTableInfo *MJTI = DAG.getMachineFunction().getJumpTableInfo();
2194 const auto &MBBs = MJTI->getJumpTables()[JT->getIndex()].MBBs;
2195
2196 // Add an operand for each case.
2197 for (auto *MBB : MBBs)
2198 Ops.push_back(DAG.getBasicBlock(MBB));
2199
2200 // Add the first MBB as a dummy default target for now. This will be replaced
2201 // with the proper default target (and the preceding range check eliminated)
2202 // if possible by WebAssemblyFixBrTableDefaults.
2203 Ops.push_back(DAG.getBasicBlock(*MBBs.begin()));
2204 return DAG.getNode(WebAssemblyISD::BR_TABLE, DL, MVT::Other, Ops);
2205}
2206
2207SDValue WebAssemblyTargetLowering::LowerVASTART(SDValue Op,
2208 SelectionDAG &DAG) const {
2209 SDLoc DL(Op);
2210 EVT PtrVT = getPointerTy(DAG.getMachineFunction().getDataLayout());
2211
2212 auto *MFI = DAG.getMachineFunction().getInfo<WebAssemblyFunctionInfo>();
2213 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
2214
2215 SDValue ArgN = DAG.getCopyFromReg(DAG.getEntryNode(), DL,
2216 MFI->getVarargBufferVreg(), PtrVT);
2217 return DAG.getStore(Op.getOperand(0), DL, ArgN, Op.getOperand(1),
2218 MachinePointerInfo(SV));
2219}
2220
2221SDValue WebAssemblyTargetLowering::LowerIntrinsic(SDValue Op,
2222 SelectionDAG &DAG) const {
2223 MachineFunction &MF = DAG.getMachineFunction();
2224 unsigned IntNo;
2225 switch (Op.getOpcode()) {
2228 IntNo = Op.getConstantOperandVal(1);
2229 break;
2231 IntNo = Op.getConstantOperandVal(0);
2232 break;
2233 default:
2234 llvm_unreachable("Invalid intrinsic");
2235 }
2236 SDLoc DL(Op);
2237
2238 switch (IntNo) {
2239 default:
2240 return SDValue(); // Don't custom lower most intrinsics.
2241
2242 case Intrinsic::wasm_lsda: {
2243 auto PtrVT = getPointerTy(MF.getDataLayout());
2244 const char *SymName = MF.createExternalSymbolName(
2245 "GCC_except_table" + std::to_string(MF.getFunctionNumber()));
2246 if (isPositionIndependent()) {
2248 SymName, PtrVT, WebAssemblyII::MO_MEMORY_BASE_REL);
2249 const char *BaseName = MF.createExternalSymbolName("__memory_base");
2251 DAG.getNode(WebAssemblyISD::Wrapper, DL, PtrVT,
2252 DAG.getTargetExternalSymbol(BaseName, PtrVT));
2253 SDValue SymAddr =
2254 DAG.getNode(WebAssemblyISD::WrapperREL, DL, PtrVT, Node);
2255 return DAG.getNode(ISD::ADD, DL, PtrVT, BaseAddr, SymAddr);
2256 }
2257 SDValue Node = DAG.getTargetExternalSymbol(SymName, PtrVT);
2258 return DAG.getNode(WebAssemblyISD::Wrapper, DL, PtrVT, Node);
2259 }
2260
2261 case Intrinsic::wasm_shuffle: {
2262 // Drop in-chain and replace undefs, but otherwise pass through unchanged
2263 SDValue Ops[18];
2264 size_t OpIdx = 0;
2265 Ops[OpIdx++] = Op.getOperand(1);
2266 Ops[OpIdx++] = Op.getOperand(2);
2267 while (OpIdx < 18) {
2268 const SDValue &MaskIdx = Op.getOperand(OpIdx + 1);
2269 if (MaskIdx.isUndef() || MaskIdx.getNode()->getAsZExtVal() >= 32) {
2270 bool isTarget = MaskIdx.getNode()->getOpcode() == ISD::TargetConstant;
2271 Ops[OpIdx++] = DAG.getConstant(0, DL, MVT::i32, isTarget);
2272 } else {
2273 Ops[OpIdx++] = MaskIdx;
2274 }
2275 }
2276 return DAG.getNode(WebAssemblyISD::SHUFFLE, DL, Op.getValueType(), Ops);
2277 }
2278
2279 case Intrinsic::wasm_funcref_to_ptr: {
2280 // llvm.wasm.funcref.to_ptr only has a defined lowering when its result
2281 // feeds directly into an indirect call. Reaching here means the pointer
2282 // escapes a direct call. We haven't implemented conversion of a funcref
2283 // into a real function pointer so we crash if we get here.
2284 fail(DL, DAG,
2285 "a funcref can only be converted to a pointer to be directly called; "
2286 "the resulting pointer cannot otherwise be used");
2287 return DAG.getPOISON(Op.getValueType());
2288 }
2289
2290 case Intrinsic::thread_pointer: {
2291 return SDValue(WebAssembly::getTLSBase(DAG, DL, Subtarget), 0);
2292 }
2293 }
2294}
2295
2296SDValue
2297WebAssemblyTargetLowering::LowerSIGN_EXTEND_INREG(SDValue Op,
2298 SelectionDAG &DAG) const {
2299 SDLoc DL(Op);
2300 // If sign extension operations are disabled, allow sext_inreg only if operand
2301 // is a vector extract of an i8 or i16 lane. SIMD does not depend on sign
2302 // extension operations, but allowing sext_inreg in this context lets us have
2303 // simple patterns to select extract_lane_s instructions. Expanding sext_inreg
2304 // everywhere would be simpler in this file, but would necessitate large and
2305 // brittle patterns to undo the expansion and select extract_lane_s
2306 // instructions.
2307 assert(!Subtarget->hasSignExt() && Subtarget->hasSIMD128());
2308 if (Op.getOperand(0).getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2309 return SDValue();
2310
2311 const SDValue &Extract = Op.getOperand(0);
2312 MVT VecT = Extract.getOperand(0).getSimpleValueType();
2313 if (VecT.getVectorElementType().getSizeInBits() > 32)
2314 return SDValue();
2315 MVT ExtractedLaneT =
2316 cast<VTSDNode>(Op.getOperand(1).getNode())->getVT().getSimpleVT();
2317 MVT ExtractedVecT =
2318 MVT::getVectorVT(ExtractedLaneT, 128 / ExtractedLaneT.getSizeInBits());
2319 if (ExtractedVecT == VecT)
2320 return Op;
2321
2322 // Bitcast vector to appropriate type to ensure ISel pattern coverage
2323 const SDNode *Index = Extract.getOperand(1).getNode();
2324 if (!isa<ConstantSDNode>(Index))
2325 return SDValue();
2326 unsigned IndexVal = Index->getAsZExtVal();
2327 unsigned Scale =
2328 ExtractedVecT.getVectorNumElements() / VecT.getVectorNumElements();
2329 assert(Scale > 1);
2330 SDValue NewIndex =
2331 DAG.getConstant(IndexVal * Scale, DL, Index->getValueType(0));
2332 SDValue NewExtract = DAG.getNode(
2334 DAG.getBitcast(ExtractedVecT, Extract.getOperand(0)), NewIndex);
2335 return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, Op.getValueType(), NewExtract,
2336 Op.getOperand(1));
2337}
2338
2339static SDValue GetExtendHigh(SDValue Op, unsigned UserOpc, EVT VT,
2340 SelectionDAG &DAG) {
2341 SDValue Source = peekThroughBitcasts(Op);
2342 if (Source.getOpcode() != ISD::VECTOR_SHUFFLE)
2343 return SDValue();
2344
2345 assert((UserOpc == WebAssemblyISD::EXTEND_LOW_U ||
2346 UserOpc == WebAssemblyISD::EXTEND_LOW_S) &&
2347 "expected extend_low");
2348 auto *Shuffle = cast<ShuffleVectorSDNode>(Source.getNode());
2349
2350 ArrayRef<int> Mask = Shuffle->getMask();
2351 // Look for a shuffle which moves from the high half to the low half.
2352 size_t FirstIdx = Mask.size() / 2;
2353 for (size_t i = 0; i < Mask.size() / 2; ++i) {
2354 if (Mask[i] != static_cast<int>(FirstIdx + i)) {
2355 return SDValue();
2356 }
2357 }
2358
2359 SDLoc DL(Op);
2360 unsigned Opc = UserOpc == WebAssemblyISD::EXTEND_LOW_S
2361 ? WebAssemblyISD::EXTEND_HIGH_S
2362 : WebAssemblyISD::EXTEND_HIGH_U;
2363 SDValue ShuffleSrc = Shuffle->getOperand(0);
2364 if (Op.getOpcode() == ISD::BITCAST)
2365 ShuffleSrc = DAG.getBitcast(Op.getValueType(), ShuffleSrc);
2366
2367 return DAG.getNode(Opc, DL, VT, ShuffleSrc);
2368}
2369
2370SDValue
2371WebAssemblyTargetLowering::LowerEXTEND_VECTOR_INREG(SDValue Op,
2372 SelectionDAG &DAG) const {
2373 SDLoc DL(Op);
2374 EVT VT = Op.getValueType();
2375 SDValue Src = Op.getOperand(0);
2376 EVT SrcVT = Src.getValueType();
2377
2378 if (SrcVT.getVectorElementType() == MVT::i1 ||
2379 SrcVT.getVectorElementType() == MVT::i64)
2380 return SDValue();
2381
2382 assert(VT.getScalarSizeInBits() % SrcVT.getScalarSizeInBits() == 0 &&
2383 "Unexpected extension factor.");
2384 unsigned Scale = VT.getScalarSizeInBits() / SrcVT.getScalarSizeInBits();
2385
2386 if (Scale != 2 && Scale != 4 && Scale != 8)
2387 return SDValue();
2388
2389 unsigned Ext;
2390 switch (Op.getOpcode()) {
2391 default:
2392 llvm_unreachable("unexpected opcode");
2395 Ext = WebAssemblyISD::EXTEND_LOW_U;
2396 break;
2398 Ext = WebAssemblyISD::EXTEND_LOW_S;
2399 break;
2400 }
2401
2402 if (Scale == 2) {
2403 // See if we can use EXTEND_HIGH.
2404 if (auto ExtendHigh = GetExtendHigh(Op.getOperand(0), Ext, VT, DAG))
2405 return ExtendHigh;
2406 }
2407
2408 SDValue Ret = Src;
2409 while (Scale != 1) {
2410 Ret = DAG.getNode(Ext, DL,
2411 Ret.getValueType()
2414 Ret);
2415 Scale /= 2;
2416 }
2417 assert(Ret.getValueType() == VT);
2418 return Ret;
2419}
2420
2422 SDLoc DL(Op);
2423 if (Op.getValueType() != MVT::v2f64 && Op.getValueType() != MVT::v4f32)
2424 return SDValue();
2425
2426 auto GetConvertedLane = [](SDValue Op, unsigned &Opcode, SDValue &SrcVec,
2427 unsigned &Index) -> bool {
2428 switch (Op.getOpcode()) {
2429 case ISD::SINT_TO_FP:
2430 Opcode = WebAssemblyISD::CONVERT_LOW_S;
2431 break;
2432 case ISD::UINT_TO_FP:
2433 Opcode = WebAssemblyISD::CONVERT_LOW_U;
2434 break;
2435 case ISD::FP_EXTEND:
2436 case ISD::FP16_TO_FP:
2437 Opcode = WebAssemblyISD::PROMOTE_LOW;
2438 break;
2439 default:
2440 return false;
2441 }
2442
2443 auto ExtractVector = Op.getOperand(0);
2444 if (ExtractVector.getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2445 return false;
2446
2447 if (!isa<ConstantSDNode>(ExtractVector.getOperand(1).getNode()))
2448 return false;
2449
2450 SrcVec = ExtractVector.getOperand(0);
2451 Index = ExtractVector.getConstantOperandVal(1);
2452 return true;
2453 };
2454
2455 unsigned NumLanes = Op.getValueType() == MVT::v2f64 ? 2 : 4;
2456 unsigned FirstOpcode = 0, SecondOpcode = 0, ThirdOpcode = 0, FourthOpcode = 0;
2457 unsigned FirstIndex = 0, SecondIndex = 0, ThirdIndex = 0, FourthIndex = 0;
2458 SDValue FirstSrcVec, SecondSrcVec, ThirdSrcVec, FourthSrcVec;
2459
2460 if (!GetConvertedLane(Op.getOperand(0), FirstOpcode, FirstSrcVec,
2461 FirstIndex) ||
2462 !GetConvertedLane(Op.getOperand(1), SecondOpcode, SecondSrcVec,
2463 SecondIndex))
2464 return SDValue();
2465
2466 // If we're converting to v4f32, check the third and fourth lanes, too.
2467 if (NumLanes == 4 && (!GetConvertedLane(Op.getOperand(2), ThirdOpcode,
2468 ThirdSrcVec, ThirdIndex) ||
2469 !GetConvertedLane(Op.getOperand(3), FourthOpcode,
2470 FourthSrcVec, FourthIndex)))
2471 return SDValue();
2472
2473 if (FirstOpcode != SecondOpcode)
2474 return SDValue();
2475
2476 // TODO Add an optimization similar to the v2f64 below for shuffling the
2477 // vectors when the lanes are in the wrong order or come from different src
2478 // vectors.
2479 if (NumLanes == 4 &&
2480 (FirstOpcode != ThirdOpcode || FirstOpcode != FourthOpcode ||
2481 FirstSrcVec != SecondSrcVec || FirstSrcVec != ThirdSrcVec ||
2482 FirstSrcVec != FourthSrcVec || FirstIndex != 0 || SecondIndex != 1 ||
2483 ThirdIndex != 2 || FourthIndex != 3))
2484 return SDValue();
2485
2486 MVT ExpectedSrcVT;
2487 switch (FirstOpcode) {
2488 case WebAssemblyISD::CONVERT_LOW_S:
2489 case WebAssemblyISD::CONVERT_LOW_U:
2490 ExpectedSrcVT = MVT::v4i32;
2491 break;
2492 case WebAssemblyISD::PROMOTE_LOW:
2493 ExpectedSrcVT = NumLanes == 2 ? MVT::v4f32 : MVT::v8i16;
2494 break;
2495 }
2496 if (FirstSrcVec.getValueType() != ExpectedSrcVT)
2497 return SDValue();
2498
2499 auto Src = FirstSrcVec;
2500 if (NumLanes == 2 &&
2501 (FirstIndex != 0 || SecondIndex != 1 || FirstSrcVec != SecondSrcVec)) {
2502 // Shuffle the source vector so that the converted lanes are the low lanes.
2503 Src = DAG.getVectorShuffle(ExpectedSrcVT, DL, FirstSrcVec, SecondSrcVec,
2504 {static_cast<int>(FirstIndex),
2505 static_cast<int>(SecondIndex) + 4, -1, -1});
2506 }
2507 return DAG.getNode(FirstOpcode, DL, NumLanes == 2 ? MVT::v2f64 : MVT::v4f32,
2508 Src);
2509}
2510
2511SDValue WebAssemblyTargetLowering::LowerBUILD_VECTOR(SDValue Op,
2512 SelectionDAG &DAG) const {
2513 MVT VT = Op.getSimpleValueType();
2514 if (VT == MVT::v8f16) {
2515 // BUILD_VECTOR can't handle FP16 operands since Wasm doesn't have a scalar
2516 // FP16 type, so cast them to I16s.
2517 MVT IVT = VT.changeVectorElementType(MVT::i16);
2519 for (unsigned I = 0, E = Op.getNumOperands(); I < E; ++I)
2520 NewOps.push_back(DAG.getBitcast(MVT::i16, Op.getOperand(I)));
2521 SDValue Res = DAG.getNode(ISD::BUILD_VECTOR, SDLoc(), IVT, NewOps);
2522 return DAG.getBitcast(VT, Res);
2523 }
2524
2525 if (auto ConvertLow = LowerConvertLow(Op, DAG))
2526 return ConvertLow;
2527
2528 SDLoc DL(Op);
2529 const EVT VecT = Op.getValueType();
2530 const EVT LaneT = Op.getOperand(0).getValueType();
2531 const size_t Lanes = Op.getNumOperands();
2532 bool CanSwizzle = VecT == MVT::v16i8;
2533
2534 // BUILD_VECTORs are lowered to the instruction that initializes the highest
2535 // possible number of lanes at once followed by a sequence of replace_lane
2536 // instructions to individually initialize any remaining lanes.
2537
2538 // TODO: Tune this. For example, lanewise swizzling is very expensive, so
2539 // swizzled lanes should be given greater weight.
2540
2541 // TODO: Investigate looping rather than always extracting/replacing specific
2542 // lanes to fill gaps.
2543
2544 auto IsConstant = [](const SDValue &V) {
2545 return V.getOpcode() == ISD::Constant || V.getOpcode() == ISD::ConstantFP;
2546 };
2547
2548 // Returns the source vector and index vector pair if they exist. Checks for:
2549 // (extract_vector_elt
2550 // $src,
2551 // (sign_extend_inreg (extract_vector_elt $indices, $i))
2552 // )
2553 auto GetSwizzleSrcs = [](size_t I, const SDValue &Lane) {
2554 auto Bail = std::make_pair(SDValue(), SDValue());
2555 if (Lane->getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2556 return Bail;
2557 const SDValue &SwizzleSrc = Lane->getOperand(0);
2558 const SDValue &IndexExt = Lane->getOperand(1);
2559 if (IndexExt->getOpcode() != ISD::SIGN_EXTEND_INREG)
2560 return Bail;
2561 const SDValue &Index = IndexExt->getOperand(0);
2562 if (Index->getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2563 return Bail;
2564 const SDValue &SwizzleIndices = Index->getOperand(0);
2565 if (SwizzleSrc.getValueType() != MVT::v16i8 ||
2566 SwizzleIndices.getValueType() != MVT::v16i8 ||
2567 Index->getOperand(1)->getOpcode() != ISD::Constant ||
2568 Index->getConstantOperandVal(1) != I)
2569 return Bail;
2570 return std::make_pair(SwizzleSrc, SwizzleIndices);
2571 };
2572
2573 // If the lane is extracted from another vector at a constant index, return
2574 // that vector. The source vector must not have more lanes than the dest
2575 // because the shufflevector indices are in terms of the destination lanes and
2576 // would not be able to address the smaller individual source lanes.
2577 auto GetShuffleSrc = [&](const SDValue &Lane) {
2578 if (Lane->getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2579 return SDValue();
2580 if (!isa<ConstantSDNode>(Lane->getOperand(1).getNode()))
2581 return SDValue();
2582 if (Lane->getOperand(0).getValueType().getVectorNumElements() >
2583 VecT.getVectorNumElements())
2584 return SDValue();
2585 return Lane->getOperand(0);
2586 };
2587
2588 using ValueEntry = std::pair<SDValue, size_t>;
2589 SmallVector<ValueEntry, 16> SplatValueCounts;
2590
2591 using SwizzleEntry = std::pair<std::pair<SDValue, SDValue>, size_t>;
2592 SmallVector<SwizzleEntry, 16> SwizzleCounts;
2593
2594 using ShuffleEntry = std::pair<SDValue, size_t>;
2595 SmallVector<ShuffleEntry, 16> ShuffleCounts;
2596
2597 auto AddCount = [](auto &Counts, const auto &Val) {
2598 auto CountIt =
2599 llvm::find_if(Counts, [&Val](auto E) { return E.first == Val; });
2600 if (CountIt == Counts.end()) {
2601 Counts.emplace_back(Val, 1);
2602 } else {
2603 CountIt->second++;
2604 }
2605 };
2606
2607 auto GetMostCommon = [](auto &Counts) {
2608 auto CommonIt = llvm::max_element(Counts, llvm::less_second());
2609 assert(CommonIt != Counts.end() && "Unexpected all-undef build_vector");
2610 return *CommonIt;
2611 };
2612
2613 size_t NumConstantLanes = 0;
2614
2615 // Count eligible lanes for each type of vector creation op
2616 for (size_t I = 0; I < Lanes; ++I) {
2617 const SDValue &Lane = Op->getOperand(I);
2618 if (Lane.isUndef())
2619 continue;
2620
2621 AddCount(SplatValueCounts, Lane);
2622
2623 if (IsConstant(Lane))
2624 NumConstantLanes++;
2625 if (auto ShuffleSrc = GetShuffleSrc(Lane))
2626 AddCount(ShuffleCounts, ShuffleSrc);
2627 if (CanSwizzle) {
2628 auto SwizzleSrcs = GetSwizzleSrcs(I, Lane);
2629 if (SwizzleSrcs.first)
2630 AddCount(SwizzleCounts, SwizzleSrcs);
2631 }
2632 }
2633
2634 SDValue SplatValue;
2635 size_t NumSplatLanes;
2636 std::tie(SplatValue, NumSplatLanes) = GetMostCommon(SplatValueCounts);
2637
2638 SDValue SwizzleSrc;
2639 SDValue SwizzleIndices;
2640 size_t NumSwizzleLanes = 0;
2641 if (SwizzleCounts.size())
2642 std::forward_as_tuple(std::tie(SwizzleSrc, SwizzleIndices),
2643 NumSwizzleLanes) = GetMostCommon(SwizzleCounts);
2644
2645 // Shuffles can draw from up to two vectors, so find the two most common
2646 // sources.
2647 SDValue ShuffleSrc1, ShuffleSrc2;
2648 size_t NumShuffleLanes = 0;
2649 if (ShuffleCounts.size()) {
2650 std::tie(ShuffleSrc1, NumShuffleLanes) = GetMostCommon(ShuffleCounts);
2651 llvm::erase_if(ShuffleCounts,
2652 [&](const auto &Pair) { return Pair.first == ShuffleSrc1; });
2653 }
2654 if (ShuffleCounts.size()) {
2655 size_t AdditionalShuffleLanes;
2656 std::tie(ShuffleSrc2, AdditionalShuffleLanes) =
2657 GetMostCommon(ShuffleCounts);
2658 NumShuffleLanes += AdditionalShuffleLanes;
2659 }
2660
2661 // Predicate returning true if the lane is properly initialized by the
2662 // original instruction
2663 std::function<bool(size_t, const SDValue &)> IsLaneConstructed;
2665 // Prefer swizzles over shuffles over vector consts over splats
2666 if (NumSwizzleLanes >= NumShuffleLanes &&
2667 NumSwizzleLanes >= NumConstantLanes && NumSwizzleLanes >= NumSplatLanes) {
2668 Result = DAG.getNode(WebAssemblyISD::SWIZZLE, DL, VecT, SwizzleSrc,
2669 SwizzleIndices);
2670 auto Swizzled = std::make_pair(SwizzleSrc, SwizzleIndices);
2671 IsLaneConstructed = [&, Swizzled](size_t I, const SDValue &Lane) {
2672 return Swizzled == GetSwizzleSrcs(I, Lane);
2673 };
2674 } else if (NumShuffleLanes >= NumConstantLanes &&
2675 NumShuffleLanes >= NumSplatLanes) {
2676 size_t DestLaneSize = VecT.getVectorElementType().getFixedSizeInBits() / 8;
2677 size_t DestLaneCount = VecT.getVectorNumElements();
2678 size_t Scale1 = 1;
2679 size_t Scale2 = 1;
2680 SDValue Src1 = ShuffleSrc1;
2681 SDValue Src2 = ShuffleSrc2 ? ShuffleSrc2 : DAG.getUNDEF(VecT);
2682 if (Src1.getValueType() != VecT) {
2683 size_t LaneSize =
2685 assert(LaneSize > DestLaneSize);
2686 Scale1 = LaneSize / DestLaneSize;
2687 Src1 = DAG.getBitcast(VecT, Src1);
2688 }
2689 if (Src2.getValueType() != VecT) {
2690 size_t LaneSize =
2692 assert(LaneSize > DestLaneSize);
2693 Scale2 = LaneSize / DestLaneSize;
2694 Src2 = DAG.getBitcast(VecT, Src2);
2695 }
2696
2697 int Mask[16];
2698 assert(DestLaneCount <= 16);
2699 for (size_t I = 0; I < DestLaneCount; ++I) {
2700 const SDValue &Lane = Op->getOperand(I);
2701 SDValue Src = GetShuffleSrc(Lane);
2702 if (Src == ShuffleSrc1) {
2703 Mask[I] = Lane->getConstantOperandVal(1) * Scale1;
2704 } else if (Src && Src == ShuffleSrc2) {
2705 Mask[I] = DestLaneCount + Lane->getConstantOperandVal(1) * Scale2;
2706 } else {
2707 Mask[I] = -1;
2708 }
2709 }
2710 ArrayRef<int> MaskRef(Mask, DestLaneCount);
2711 Result = DAG.getVectorShuffle(VecT, DL, Src1, Src2, MaskRef);
2712 IsLaneConstructed = [&](size_t, const SDValue &Lane) {
2713 auto Src = GetShuffleSrc(Lane);
2714 return Src == ShuffleSrc1 || (Src && Src == ShuffleSrc2);
2715 };
2716 } else if (NumConstantLanes >= NumSplatLanes) {
2717 SmallVector<SDValue, 16> ConstLanes;
2718 for (const SDValue &Lane : Op->op_values()) {
2719 if (IsConstant(Lane)) {
2720 // Values may need to be fixed so that they will sign extend to be
2721 // within the expected range during ISel. Check whether the value is in
2722 // bounds based on the lane bit width and if it is out of bounds, lop
2723 // off the extra bits.
2724 uint64_t LaneBits = 128 / Lanes;
2725 if (auto *Const = dyn_cast<ConstantSDNode>(Lane.getNode())) {
2726 ConstLanes.push_back(DAG.getConstant(
2727 Const->getAPIntValue().trunc(LaneBits).getZExtValue(),
2728 SDLoc(Lane), LaneT));
2729 } else {
2730 ConstLanes.push_back(Lane);
2731 }
2732 } else if (LaneT.isFloatingPoint()) {
2733 ConstLanes.push_back(DAG.getConstantFP(0, DL, LaneT));
2734 } else {
2735 ConstLanes.push_back(DAG.getConstant(0, DL, LaneT));
2736 }
2737 }
2738 Result = DAG.getBuildVector(VecT, DL, ConstLanes);
2739 IsLaneConstructed = [&IsConstant](size_t _, const SDValue &Lane) {
2740 return IsConstant(Lane);
2741 };
2742 } else {
2743 size_t DestLaneSize = VecT.getVectorElementType().getFixedSizeInBits();
2744 if (NumSplatLanes == 1 && Op->getOperand(0) == SplatValue &&
2745 (DestLaneSize == 32 || DestLaneSize == 64)) {
2746 // Could be selected to load_zero.
2747 Result = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VecT, SplatValue);
2748 } else {
2749 // Use a splat (which might be selected as a load splat)
2750 Result = DAG.getSplatBuildVector(VecT, DL, SplatValue);
2751 }
2752 IsLaneConstructed = [&SplatValue](size_t _, const SDValue &Lane) {
2753 return Lane == SplatValue;
2754 };
2755 }
2756
2757 assert(Result);
2758 assert(IsLaneConstructed);
2759
2760 // Add replace_lane instructions for any unhandled values
2761 for (size_t I = 0; I < Lanes; ++I) {
2762 const SDValue &Lane = Op->getOperand(I);
2763 if (!Lane.isUndef() && !IsLaneConstructed(I, Lane))
2764 Result = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VecT, Result, Lane,
2765 DAG.getConstant(I, DL, MVT::i32));
2766 }
2767
2768 return Result;
2769}
2770
2771SDValue
2772WebAssemblyTargetLowering::LowerVECTOR_SHUFFLE(SDValue Op,
2773 SelectionDAG &DAG) const {
2774 SDLoc DL(Op);
2775 ArrayRef<int> Mask = cast<ShuffleVectorSDNode>(Op.getNode())->getMask();
2776 MVT VecType = Op.getOperand(0).getSimpleValueType();
2777 assert(VecType.is128BitVector() && "Unexpected shuffle vector type");
2778 size_t LaneBytes = VecType.getVectorElementType().getSizeInBits() / 8;
2779
2780 // Space for two vector args and sixteen mask indices
2781 SDValue Ops[18];
2782 size_t OpIdx = 0;
2783 Ops[OpIdx++] = Op.getOperand(0);
2784 Ops[OpIdx++] = Op.getOperand(1);
2785
2786 // Expand mask indices to byte indices and materialize them as operands
2787 for (int M : Mask) {
2788 for (size_t J = 0; J < LaneBytes; ++J) {
2789 // Lower undefs (represented by -1 in mask) to {0..J}, which use a
2790 // whole lane of vector input, to allow further reduction at VM. E.g.
2791 // match an 8x16 byte shuffle to an equivalent cheaper 32x4 shuffle.
2792 uint64_t ByteIndex = M == -1 ? J : (uint64_t)M * LaneBytes + J;
2793 Ops[OpIdx++] = DAG.getConstant(ByteIndex, DL, MVT::i32);
2794 }
2795 }
2796
2797 return DAG.getNode(WebAssemblyISD::SHUFFLE, DL, Op.getValueType(), Ops);
2798}
2799
2800SDValue WebAssemblyTargetLowering::LowerSETCC(SDValue Op,
2801 SelectionDAG &DAG) const {
2802 SDLoc DL(Op);
2803 // The legalizer does not know how to expand the unsupported comparison modes
2804 // of i64x2 vectors, so we manually unroll them here.
2805 assert(Op->getOperand(0)->getSimpleValueType(0) == MVT::v2i64);
2807 DAG.ExtractVectorElements(Op->getOperand(0), LHS);
2808 DAG.ExtractVectorElements(Op->getOperand(1), RHS);
2809 const SDValue &CC = Op->getOperand(2);
2810 auto MakeLane = [&](unsigned I) {
2811 return DAG.getNode(ISD::SELECT_CC, DL, MVT::i64, LHS[I], RHS[I],
2812 DAG.getConstant(uint64_t(-1), DL, MVT::i64),
2813 DAG.getConstant(uint64_t(0), DL, MVT::i64), CC);
2814 };
2815 return DAG.getBuildVector(Op->getValueType(0), DL,
2816 {MakeLane(0), MakeLane(1)});
2817}
2818
2819SDValue
2820WebAssemblyTargetLowering::LowerAccessVectorElement(SDValue Op,
2821 SelectionDAG &DAG) const {
2822 if (Op.getOpcode() == ISD::INSERT_VECTOR_ELT &&
2823 Op.getValueType() == MVT::v8f16) {
2824 // INSERT_VECTOR_ELT can't handle FP16 operands since Wasm doesn't have a
2825 // scalar FP16 type, so cast them to I16s.
2826 SDLoc DL(Op);
2827 SDValue IntVector = DAG.getBitcast(MVT::v8i16, Op.getOperand(0));
2828 SDValue IntElement = DAG.getBitcast(MVT::i16, Op.getOperand(1));
2830 IntVector, IntElement, Op.getOperand(2));
2831 return DAG.getBitcast(MVT::v8f16, Inserted);
2832 }
2833
2834 // Allow constant lane indices, expand variable lane indices
2835 SDNode *IdxNode = Op.getOperand(Op.getNumOperands() - 1).getNode();
2836 if (isa<ConstantSDNode>(IdxNode)) {
2837 // Ensure the index type is i32 to match the tablegen patterns
2838 uint64_t Idx = IdxNode->getAsZExtVal();
2839 SmallVector<SDValue, 3> Ops(Op.getNode()->ops());
2840 Ops[Op.getNumOperands() - 1] =
2841 DAG.getConstant(Idx, SDLoc(IdxNode), MVT::i32);
2842 return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(), Ops);
2843 }
2844 // Perform default expansion
2845 return SDValue();
2846}
2847
2849 EVT LaneT = Op.getSimpleValueType().getVectorElementType();
2850 // 32-bit and 64-bit unrolled shifts will have proper semantics
2851 if (LaneT.bitsGE(MVT::i32))
2852 return DAG.UnrollVectorOp(Op.getNode());
2853 // Otherwise mask the shift value to get proper semantics from 32-bit shift
2854 SDLoc DL(Op);
2855 size_t NumLanes = Op.getSimpleValueType().getVectorNumElements();
2856 SDValue Mask = DAG.getConstant(LaneT.getSizeInBits() - 1, DL, MVT::i32);
2857 unsigned ShiftOpcode = Op.getOpcode();
2858 SmallVector<SDValue, 16> ShiftedElements;
2859 DAG.ExtractVectorElements(Op.getOperand(0), ShiftedElements, 0, 0, MVT::i32);
2860 SmallVector<SDValue, 16> ShiftElements;
2861 DAG.ExtractVectorElements(Op.getOperand(1), ShiftElements, 0, 0, MVT::i32);
2862 SmallVector<SDValue, 16> UnrolledOps;
2863 for (size_t i = 0; i < NumLanes; ++i) {
2864 SDValue MaskedShiftValue =
2865 DAG.getNode(ISD::AND, DL, MVT::i32, ShiftElements[i], Mask);
2866 SDValue ShiftedValue = ShiftedElements[i];
2867 if (ShiftOpcode == ISD::SRA)
2868 ShiftedValue = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, MVT::i32,
2869 ShiftedValue, DAG.getValueType(LaneT));
2870 UnrolledOps.push_back(
2871 DAG.getNode(ShiftOpcode, DL, MVT::i32, ShiftedValue, MaskedShiftValue));
2872 }
2873 return DAG.getBuildVector(Op.getValueType(), DL, UnrolledOps);
2874}
2875
2876SDValue WebAssemblyTargetLowering::LowerShift(SDValue Op,
2877 SelectionDAG &DAG) const {
2878 SDLoc DL(Op);
2879 // Only manually lower vector shifts
2880 assert(Op.getSimpleValueType().isVector());
2881
2882 uint64_t LaneBits = Op.getValueType().getScalarSizeInBits();
2883 auto ShiftVal = Op.getOperand(1);
2884
2885 // Try to skip bitmask operation since it is implied inside shift instruction
2886 auto SkipImpliedMask = [](SDValue MaskOp, uint64_t MaskBits) {
2887 if (MaskOp.getOpcode() != ISD::AND)
2888 return MaskOp;
2889 SDValue LHS = MaskOp.getOperand(0);
2890 SDValue RHS = MaskOp.getOperand(1);
2891 if (MaskOp.getValueType().isVector()) {
2892 APInt MaskVal;
2893 if (!ISD::isConstantSplatVector(RHS.getNode(), MaskVal))
2894 std::swap(LHS, RHS);
2895
2896 if (ISD::isConstantSplatVector(RHS.getNode(), MaskVal) &&
2897 MaskVal == MaskBits)
2898 MaskOp = LHS;
2899 } else {
2900 if (!isa<ConstantSDNode>(RHS.getNode()))
2901 std::swap(LHS, RHS);
2902
2903 auto ConstantRHS = dyn_cast<ConstantSDNode>(RHS.getNode());
2904 if (ConstantRHS && ConstantRHS->getAPIntValue() == MaskBits)
2905 MaskOp = LHS;
2906 }
2907
2908 return MaskOp;
2909 };
2910
2911 // Skip vector and operation
2912 ShiftVal = SkipImpliedMask(ShiftVal, LaneBits - 1);
2913 ShiftVal = DAG.getSplatValue(ShiftVal);
2914 if (!ShiftVal)
2915 return unrollVectorShift(Op, DAG);
2916
2917 // Skip scalar and operation
2918 ShiftVal = SkipImpliedMask(ShiftVal, LaneBits - 1);
2919 // Use anyext because none of the high bits can affect the shift
2920 ShiftVal = DAG.getAnyExtOrTrunc(ShiftVal, DL, MVT::i32);
2921
2922 unsigned Opcode;
2923 switch (Op.getOpcode()) {
2924 case ISD::SHL:
2925 Opcode = WebAssemblyISD::VEC_SHL;
2926 break;
2927 case ISD::SRA:
2928 Opcode = WebAssemblyISD::VEC_SHR_S;
2929 break;
2930 case ISD::SRL:
2931 Opcode = WebAssemblyISD::VEC_SHR_U;
2932 break;
2933 default:
2934 llvm_unreachable("unexpected opcode");
2935 }
2936
2937 return DAG.getNode(Opcode, DL, Op.getValueType(), Op.getOperand(0), ShiftVal);
2938}
2939
2940SDValue WebAssemblyTargetLowering::LowerFP_TO_INT_SAT(SDValue Op,
2941 SelectionDAG &DAG) const {
2942 EVT ResT = Op.getValueType();
2943 EVT SatVT = cast<VTSDNode>(Op.getOperand(1))->getVT();
2944
2945 if ((ResT == MVT::i32 || ResT == MVT::i64) &&
2946 (SatVT == MVT::i32 || SatVT == MVT::i64))
2947 return Op;
2948
2949 if (ResT == MVT::v4i32 && SatVT == MVT::i32)
2950 return Op;
2951
2952 if (ResT == MVT::v8i16 && SatVT == MVT::i16)
2953 return Op;
2954
2955 return SDValue();
2956}
2957
2959 return (Op->getFlags().hasNoNaNs() ||
2960 (DAG.isKnownNeverNaN(Op->getOperand(0)) &&
2961 DAG.isKnownNeverNaN(Op->getOperand(1)))) &&
2962 (Op->getFlags().hasNoSignedZeros() ||
2963 DAG.isKnownNeverLogicalZero(Op->getOperand(0)) ||
2964 DAG.isKnownNeverLogicalZero(Op->getOperand(1)));
2965}
2966
2967SDValue WebAssemblyTargetLowering::LowerFMIN(SDValue Op,
2968 SelectionDAG &DAG) const {
2969 if (Subtarget->hasRelaxedSIMD() && HasNoSignedZerosOrNaNs(Op, DAG)) {
2970 return DAG.getNode(WebAssemblyISD::RELAXED_FMIN, SDLoc(Op),
2971 Op.getValueType(), Op.getOperand(0), Op.getOperand(1));
2972 }
2973 return SDValue();
2974}
2975
2976SDValue WebAssemblyTargetLowering::LowerFMAX(SDValue Op,
2977 SelectionDAG &DAG) const {
2978 if (Subtarget->hasRelaxedSIMD() && HasNoSignedZerosOrNaNs(Op, DAG)) {
2979 return DAG.getNode(WebAssemblyISD::RELAXED_FMAX, SDLoc(Op),
2980 Op.getValueType(), Op.getOperand(0), Op.getOperand(1));
2981 }
2982 return SDValue();
2983}
2984
2985//===----------------------------------------------------------------------===//
2986// Custom DAG combine hooks
2987//===----------------------------------------------------------------------===//
2988static SDValue
2990 auto &DAG = DCI.DAG;
2991 auto Shuffle = cast<ShuffleVectorSDNode>(N);
2992
2993 // Hoist vector bitcasts that don't change the number of lanes out of unary
2994 // shuffles, where they are less likely to get in the way of other combines.
2995 // (shuffle (vNxT1 (bitcast (vNxT0 x))), undef, mask) ->
2996 // (vNxT1 (bitcast (vNxT0 (shuffle x, undef, mask))))
2997 SDValue Bitcast = N->getOperand(0);
2998 if (Bitcast.getOpcode() != ISD::BITCAST)
2999 return SDValue();
3000 if (!N->getOperand(1).isUndef())
3001 return SDValue();
3002 SDValue CastOp = Bitcast.getOperand(0);
3003 EVT SrcType = CastOp.getValueType();
3004 EVT DstType = Bitcast.getValueType();
3005 if (!SrcType.is128BitVector() ||
3006 SrcType.getVectorNumElements() != DstType.getVectorNumElements())
3007 return SDValue();
3008 SDValue NewShuffle = DAG.getVectorShuffle(
3009 SrcType, SDLoc(N), CastOp, DAG.getUNDEF(SrcType), Shuffle->getMask());
3010 return DAG.getBitcast(DstType, NewShuffle);
3011}
3012
3013/// Convert ({u,s}itofp vec) --> ({u,s}itofp ({s,z}ext vec)) so it doesn't get
3014/// split up into scalar instructions during legalization, and the vector
3015/// extending instructions are selected in performVectorExtendCombine below.
3016static SDValue
3019 auto &DAG = DCI.DAG;
3020 assert(N->getOpcode() == ISD::UINT_TO_FP ||
3021 N->getOpcode() == ISD::SINT_TO_FP);
3022
3023 EVT InVT = N->getOperand(0)->getValueType(0);
3024 EVT ResVT = N->getValueType(0);
3025 MVT ExtVT;
3026 if (ResVT == MVT::v4f32 && (InVT == MVT::v4i16 || InVT == MVT::v4i8))
3027 ExtVT = MVT::v4i32;
3028 else if (ResVT == MVT::v2f64 && (InVT == MVT::v2i16 || InVT == MVT::v2i8))
3029 ExtVT = MVT::v2i32;
3030 else
3031 return SDValue();
3032
3033 unsigned Op =
3035 SDValue Conv = DAG.getNode(Op, SDLoc(N), ExtVT, N->getOperand(0));
3036 return DAG.getNode(N->getOpcode(), SDLoc(N), ResVT, Conv);
3037}
3038
3039static SDValue
3042 auto &DAG = DCI.DAG;
3043
3044 SDNodeFlags Flags = N->getFlags();
3045 SDValue Op0 = N->getOperand(0);
3046 EVT VT = N->getValueType(0);
3047
3048 // Optimize uitofp to sitofp when the sign bit is known to be zero.
3049 // Depending on the target (runtime) backend, this might be performance
3050 // neutral (e.g. AArch64) or a significant improvement (e.g. x86_64).
3051 if (VT.isVector() && (Flags.hasNonNeg() || DAG.SignBitIsZero(Op0))) {
3052 return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, Op0);
3053 }
3054
3055 return SDValue();
3056}
3057
3058static SDValue
3060 auto &DAG = DCI.DAG;
3061 assert(N->getOpcode() == ISD::SIGN_EXTEND ||
3062 N->getOpcode() == ISD::ZERO_EXTEND);
3063
3064 EVT ResVT = N->getValueType(0);
3065 bool IsSext = N->getOpcode() == ISD::SIGN_EXTEND;
3066 SDLoc DL(N);
3067
3068 if (ResVT == MVT::v16i32 && N->getOperand(0)->getValueType(0) == MVT::v16i8) {
3069 // Use a tree of extend low/high to split and extend the input in two
3070 // layers to avoid doing several shuffles and even more extends.
3071 unsigned LowOp =
3072 IsSext ? WebAssemblyISD::EXTEND_LOW_S : WebAssemblyISD::EXTEND_LOW_U;
3073 unsigned HighOp =
3074 IsSext ? WebAssemblyISD::EXTEND_HIGH_S : WebAssemblyISD::EXTEND_HIGH_U;
3075 SDValue Input = N->getOperand(0);
3076 SDValue LowHalf = DAG.getNode(LowOp, DL, MVT::v8i16, Input);
3077 SDValue HighHalf = DAG.getNode(HighOp, DL, MVT::v8i16, Input);
3078 SDValue Subvectors[] = {
3079 DAG.getNode(LowOp, DL, MVT::v4i32, LowHalf),
3080 DAG.getNode(HighOp, DL, MVT::v4i32, LowHalf),
3081 DAG.getNode(LowOp, DL, MVT::v4i32, HighHalf),
3082 DAG.getNode(HighOp, DL, MVT::v4i32, HighHalf),
3083 };
3084 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ResVT, Subvectors);
3085 }
3086
3087 // Combine ({s,z}ext (extract_subvector src, i)) into a widening operation if
3088 // possible before the extract_subvector can be expanded.
3089 auto Extract = N->getOperand(0);
3090 if (Extract.getOpcode() != ISD::EXTRACT_SUBVECTOR)
3091 return SDValue();
3092 auto Source = Extract.getOperand(0);
3093 auto *IndexNode = dyn_cast<ConstantSDNode>(Extract.getOperand(1));
3094 if (IndexNode == nullptr)
3095 return SDValue();
3096 auto Index = IndexNode->getZExtValue();
3097
3098 // Only v8i8, v4i16, and v2i32 extracts can be widened, and only if the
3099 // extracted subvector is the low or high half of its source.
3100 if (ResVT == MVT::v8i16) {
3101 if (Extract.getValueType() != MVT::v8i8 ||
3102 Source.getValueType() != MVT::v16i8 || (Index != 0 && Index != 8))
3103 return SDValue();
3104 } else if (ResVT == MVT::v4i32) {
3105 if (Extract.getValueType() != MVT::v4i16 ||
3106 Source.getValueType() != MVT::v8i16 || (Index != 0 && Index != 4))
3107 return SDValue();
3108 } else if (ResVT == MVT::v2i64) {
3109 if (Extract.getValueType() != MVT::v2i32 ||
3110 Source.getValueType() != MVT::v4i32 || (Index != 0 && Index != 2))
3111 return SDValue();
3112 } else {
3113 return SDValue();
3114 }
3115
3116 bool IsLow = Index == 0;
3117
3118 unsigned Op = IsSext ? (IsLow ? WebAssemblyISD::EXTEND_LOW_S
3119 : WebAssemblyISD::EXTEND_HIGH_S)
3120 : (IsLow ? WebAssemblyISD::EXTEND_LOW_U
3121 : WebAssemblyISD::EXTEND_HIGH_U);
3122
3123 return DAG.getNode(Op, DL, ResVT, Source);
3124}
3125
3126static SDValue
3128 auto &DAG = DCI.DAG;
3129
3130 auto GetWasmConversionOp = [](unsigned Op) {
3131 switch (Op) {
3133 return WebAssemblyISD::TRUNC_SAT_ZERO_S;
3135 return WebAssemblyISD::TRUNC_SAT_ZERO_U;
3136 case ISD::FP_ROUND:
3137 return WebAssemblyISD::DEMOTE_ZERO;
3138 }
3139 llvm_unreachable("unexpected op");
3140 };
3141
3142 auto IsZeroSplat = [](SDValue SplatVal) {
3143 auto *Splat = dyn_cast<BuildVectorSDNode>(SplatVal.getNode());
3144 APInt SplatValue, SplatUndef;
3145 unsigned SplatBitSize;
3146 bool HasAnyUndefs;
3147 // Endianness doesn't matter in this context because we are looking for
3148 // an all-zero value.
3149 return Splat &&
3150 Splat->isConstantSplat(SplatValue, SplatUndef, SplatBitSize,
3151 HasAnyUndefs) &&
3152 SplatValue == 0;
3153 };
3154
3155 if (N->getOpcode() == ISD::CONCAT_VECTORS) {
3156 // Combine this:
3157 //
3158 // (concat_vectors (v2i32 (fp_to_{s,u}int_sat $x, 32)), (v2i32 (splat 0)))
3159 //
3160 // into (i32x4.trunc_sat_f64x2_zero_{s,u} $x).
3161 //
3162 // Or this:
3163 //
3164 // (concat_vectors ({v2f32, v4f16} (fp_round ({v2f64, v4f32} $x))),
3165 // ({v2f32, v4f16} (splat 0)))
3166 //
3167 // into ({f32x4, f16x8}.demote_zero_{f64x2, f32x4} $x).
3168 EVT ResVT;
3169 EVT ExpectedConversionType;
3170 auto Conversion = N->getOperand(0);
3171 auto ConversionOp = Conversion.getOpcode();
3172 switch (ConversionOp) {
3175 ResVT = MVT::v4i32;
3176 ExpectedConversionType = MVT::v2i32;
3177 break;
3178 case ISD::FP_ROUND:
3179 if (Conversion.getValueType() == MVT::v2f32) {
3180 ResVT = MVT::v4f32;
3181 ExpectedConversionType = MVT::v2f32;
3182 } else if (Conversion.getValueType() == MVT::v4f16) {
3183 ResVT = MVT::v8f16;
3184 ExpectedConversionType = MVT::v4f16;
3185 } else {
3186 return SDValue();
3187 }
3188 break;
3189 default:
3190 return SDValue();
3191 }
3192
3193 if (N->getValueType(0) != ResVT)
3194 return SDValue();
3195
3196 if (Conversion.getValueType() != ExpectedConversionType)
3197 return SDValue();
3198
3199 auto Source = Conversion.getOperand(0);
3200 if (!((Source.getValueType() == MVT::v2f64 && ResVT == MVT::v4f32) ||
3201 (Source.getValueType() == MVT::v2f64 && ResVT == MVT::v4i32) ||
3202 (Source.getValueType() == MVT::v4f32 && ResVT == MVT::v8f16)))
3203 return SDValue();
3204
3205 if (!IsZeroSplat(N->getOperand(1)) ||
3206 N->getOperand(1).getValueType() != ExpectedConversionType)
3207 return SDValue();
3208
3209 unsigned Op = GetWasmConversionOp(ConversionOp);
3210 return DAG.getNode(Op, SDLoc(N), ResVT, Source);
3211 }
3212
3213 // Combine this:
3214 //
3215 // (fp_to_{s,u}int_sat (concat_vectors $x, (v2f64 (splat 0))), 32)
3216 //
3217 // into (i32x4.trunc_sat_f64x2_zero_{s,u} $x).
3218 //
3219 // Or this:
3220 //
3221 // ({v4f32, v8f16} (fp_round (concat_vectors $x,
3222 // ({v2f64, v4f32} (splat 0)))))
3223 //
3224 // into ({f32x4, f16x8}.demote_zero_{f64x2, f32x4} $x).
3225 EVT ResVT;
3226 auto ConversionOp = N->getOpcode();
3227 switch (ConversionOp) {
3230 ResVT = MVT::v4i32;
3231 break;
3232 case ISD::FP_ROUND:
3233 ResVT = N->getValueType(0);
3234 break;
3235 default:
3236 llvm_unreachable("unexpected op");
3237 }
3238
3239 if (N->getValueType(0) != ResVT)
3240 return SDValue();
3241
3242 auto Concat = N->getOperand(0);
3243 if (Concat.getOpcode() != ISD::CONCAT_VECTORS)
3244 return SDValue();
3245 EVT ConcatVT = Concat.getValueType();
3246 EVT SourceVT = Concat.getOperand(0).getValueType();
3247
3248 if (!IsZeroSplat(Concat.getOperand(1)))
3249 return SDValue();
3250
3251 if (ConversionOp == ISD::FP_ROUND) {
3252 bool IsF64ToF32 =
3253 ConcatVT == MVT::v4f64 && SourceVT == MVT::v2f64 && ResVT == MVT::v4f32;
3254 bool IsF32ToF16 =
3255 ConcatVT == MVT::v8f32 && SourceVT == MVT::v4f32 && ResVT == MVT::v8f16;
3256 if (!(IsF64ToF32 || IsF32ToF16))
3257 return SDValue();
3258 } else {
3259 if (ConcatVT != MVT::v4f64 || SourceVT != MVT::v2f64 || ResVT != MVT::v4i32)
3260 return SDValue();
3261 }
3262
3263 unsigned Op = GetWasmConversionOp(ConversionOp);
3264 return DAG.getNode(Op, SDLoc(N), ResVT, Concat.getOperand(0));
3265}
3266
3267// Helper to extract VectorWidth bits from Vec, starting from IdxVal.
3268static SDValue extractSubVector(SDValue Vec, unsigned IdxVal, SelectionDAG &DAG,
3269 const SDLoc &DL, unsigned VectorWidth) {
3270 EVT VT = Vec.getValueType();
3271 EVT ElVT = VT.getVectorElementType();
3272 unsigned Factor = VT.getSizeInBits() / VectorWidth;
3273 EVT ResultVT = EVT::getVectorVT(*DAG.getContext(), ElVT,
3274 VT.getVectorNumElements() / Factor);
3275
3276 // Extract the relevant VectorWidth bits. Generate an EXTRACT_SUBVECTOR
3277 unsigned ElemsPerChunk = VectorWidth / ElVT.getSizeInBits();
3278 assert(isPowerOf2_32(ElemsPerChunk) && "Elements per chunk not power of 2");
3279
3280 // This is the index of the first element of the VectorWidth-bit chunk
3281 // we want. Since ElemsPerChunk is a power of 2 just need to clear bits.
3282 IdxVal &= ~(ElemsPerChunk - 1);
3283
3284 // If the input is a buildvector just emit a smaller one.
3285 if (Vec.getOpcode() == ISD::BUILD_VECTOR)
3286 return DAG.getBuildVector(ResultVT, DL,
3287 Vec->ops().slice(IdxVal, ElemsPerChunk));
3288
3289 SDValue VecIdx = DAG.getIntPtrConstant(IdxVal, DL);
3290 return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, ResultVT, Vec, VecIdx);
3291}
3292
3293// Helper to recursively truncate vector elements in half with NARROW_U. DstVT
3294// is the expected destination value type after recursion. In is the initial
3295// input. Note that the input should have enough leading zero bits to prevent
3296// NARROW_U from saturating results.
3298 SelectionDAG &DAG) {
3299 EVT SrcVT = In.getValueType();
3300
3301 // No truncation required, we might get here due to recursive calls.
3302 if (SrcVT == DstVT)
3303 return In;
3304
3305 unsigned SrcSizeInBits = SrcVT.getSizeInBits();
3306 unsigned NumElems = SrcVT.getVectorNumElements();
3307 if (!isPowerOf2_32(NumElems))
3308 return SDValue();
3309 assert(DstVT.getVectorNumElements() == NumElems && "Illegal truncation");
3310 assert(SrcSizeInBits > DstVT.getSizeInBits() && "Illegal truncation");
3311
3312 LLVMContext &Ctx = *DAG.getContext();
3313 EVT PackedSVT = EVT::getIntegerVT(Ctx, SrcVT.getScalarSizeInBits() / 2);
3314
3315 // Narrow to the largest type possible:
3316 // vXi64/vXi32 -> i16x8.narrow_i32x4_u and vXi16 -> i8x16.narrow_i16x8_u.
3317 EVT InVT = MVT::i16, OutVT = MVT::i8;
3318 if (SrcVT.getScalarSizeInBits() > 16) {
3319 InVT = MVT::i32;
3320 OutVT = MVT::i16;
3321 }
3322 unsigned SubSizeInBits = SrcSizeInBits / 2;
3323 InVT = EVT::getVectorVT(Ctx, InVT, SubSizeInBits / InVT.getSizeInBits());
3324 OutVT = EVT::getVectorVT(Ctx, OutVT, SubSizeInBits / OutVT.getSizeInBits());
3325
3326 // Split lower/upper subvectors.
3327 SDValue Lo = extractSubVector(In, 0, DAG, DL, SubSizeInBits);
3328 SDValue Hi = extractSubVector(In, NumElems / 2, DAG, DL, SubSizeInBits);
3329
3330 // 256bit -> 128bit truncate - Narrow lower/upper 128-bit subvectors.
3331 if (SrcVT.is256BitVector() && DstVT.is128BitVector()) {
3332 Lo = DAG.getBitcast(InVT, Lo);
3333 Hi = DAG.getBitcast(InVT, Hi);
3334 SDValue Res = DAG.getNode(WebAssemblyISD::NARROW_U, DL, OutVT, Lo, Hi);
3335 return DAG.getBitcast(DstVT, Res);
3336 }
3337
3338 // Recursively narrow lower/upper subvectors, concat result and narrow again.
3339 EVT PackedVT = EVT::getVectorVT(Ctx, PackedSVT, NumElems / 2);
3340 Lo = truncateVectorWithNARROW(PackedVT, Lo, DL, DAG);
3341 Hi = truncateVectorWithNARROW(PackedVT, Hi, DL, DAG);
3342
3343 PackedVT = EVT::getVectorVT(Ctx, PackedSVT, NumElems);
3344 SDValue Res = DAG.getNode(ISD::CONCAT_VECTORS, DL, PackedVT, Lo, Hi);
3345 return truncateVectorWithNARROW(DstVT, Res, DL, DAG);
3346}
3347
3350 auto &DAG = DCI.DAG;
3351
3352 SDValue In = N->getOperand(0);
3353 EVT InVT = In.getValueType();
3354 if (!InVT.isSimple())
3355 return SDValue();
3356
3357 EVT OutVT = N->getValueType(0);
3358 if (!OutVT.isVector())
3359 return SDValue();
3360
3361 EVT OutSVT = OutVT.getVectorElementType();
3362 EVT InSVT = InVT.getVectorElementType();
3363 // Currently only cover truncate to v16i8 or v8i16.
3364 if (!((InSVT == MVT::i16 || InSVT == MVT::i32 || InSVT == MVT::i64) &&
3365 (OutSVT == MVT::i8 || OutSVT == MVT::i16) && OutVT.is128BitVector()))
3366 return SDValue();
3367
3368 SDLoc DL(N);
3370 OutVT.getScalarSizeInBits());
3371 In = DAG.getNode(ISD::AND, DL, InVT, In, DAG.getConstant(Mask, DL, InVT));
3372 return truncateVectorWithNARROW(OutVT, In, DL, DAG);
3373}
3374
3377 using namespace llvm::SDPatternMatch;
3378 auto &DAG = DCI.DAG;
3379 SDLoc DL(N);
3380 SDValue Src = N->getOperand(0);
3381 EVT VT = N->getValueType(0);
3382 EVT SrcVT = Src.getValueType();
3383
3384 if (!(DCI.isBeforeLegalize() && VT.isScalarInteger() &&
3385 SrcVT.isFixedLengthVectorOf(MVT::i1)))
3386 return SDValue();
3387
3388 unsigned NumElts = SrcVT.getVectorNumElements();
3389 EVT Width = MVT::getIntegerVT(128 / NumElts);
3390
3391 // bitcast <N x i1> to iN, where N = 2, 4, 8, 16 (legal)
3392 // ==> bitmask
3393 if (NumElts == 2 || NumElts == 4 || NumElts == 8 || NumElts == 16) {
3394 return DAG.getZExtOrTrunc(
3395 DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3396 {DAG.getConstant(Intrinsic::wasm_bitmask, DL, MVT::i32),
3397 DAG.getSExtOrTrunc(N->getOperand(0), DL,
3398 SrcVT.changeVectorElementType(
3399 *DAG.getContext(), Width))}),
3400 DL, VT);
3401 }
3402
3403 // bitcast <N x i1>(setcc ...) to concat iN, where N = 32 and 64 (illegal)
3404 if (NumElts == 32 || NumElts == 64) {
3405 SDValue Concat, SetCCVector;
3406 ISD::CondCode SetCond;
3407
3408 if (!sd_match(N, m_BitCast(m_c_SetCC(m_Value(Concat), m_Value(SetCCVector),
3409 m_CondCode(SetCond)))))
3410 return SDValue();
3411 if (Concat.getOpcode() != ISD::CONCAT_VECTORS)
3412 return SDValue();
3413
3414 // Reconstruct the wide bitmask from each CONCAT_VECTORS operand.
3415 // Derive the per-chunk mask/integer types from the actual operand type
3416 // instead of hardcoding v16i1 / i16 for every chunk.
3417 EVT ConcatOperandVT = Concat.getOperand(0).getValueType();
3418 unsigned ConcatOperandNumElts = ConcatOperandVT.getVectorNumElements();
3419
3420 EVT ConcatOperandMaskVT =
3421 EVT::getVectorVT(*DAG.getContext(), MVT::i1,
3422 ElementCount::getFixed(ConcatOperandNumElts));
3423 EVT ConcatOperandBitmaskVT =
3424 EVT::getIntegerVT(*DAG.getContext(), ConcatOperandNumElts);
3425 EVT ReturnVT = N->getValueType(0);
3426 SDValue ReconstructedBitmask = DAG.getConstant(0, DL, ReturnVT);
3427 // Example:
3428 // v32i16 = concat(v8i16, v8i16, v8i16, v8i16)
3429 // -> v8i1 + v8i1 + v8i1 + v8i1
3430 // -> i8 + i8 + i8 + i8
3431 // -> reconstructed i32 bitmask
3432 for (size_t I = 0; I < Concat->ops().size(); ++I) {
3433 SDValue ConcatOperand = Concat.getOperand(I);
3434 assert(ConcatOperand.getValueType() == ConcatOperandVT &&
3435 "concat_vectors operands must have the same type");
3436
3437 SDValue SetCCVectorOperand =
3438 extractSubVector(SetCCVector, I * ConcatOperandNumElts, DAG, DL, 128);
3439 if (!SetCCVectorOperand ||
3440 SetCCVectorOperand.getValueType() != ConcatOperandVT)
3441 return SDValue();
3442
3443 // Build the per-chunk mask using the correct chunk type:
3444 // v16i8 -> v16i1 -> i16
3445 // v8i16 -> v8i1 -> i8
3446 // v4i32 -> v4i1 -> i4
3447 // v2i64 -> v2i1 -> i2
3448 SDValue ConcatOperandMask = DAG.getSetCC(
3449 DL, ConcatOperandMaskVT, ConcatOperand, SetCCVectorOperand, SetCond);
3450 SDValue ConcatOperandBitmask =
3451 DAG.getBitcast(ConcatOperandBitmaskVT, ConcatOperandMask);
3452 SDValue ExtendedConcatOperandBitmask =
3453 DAG.getZExtOrTrunc(ConcatOperandBitmask, DL, ReturnVT);
3454
3455 // Shift the previously reconstructed bits to make room for this chunk.
3456 if (I != 0) {
3457 ReconstructedBitmask = DAG.getNode(
3458 ISD::SHL, DL, ReturnVT, ReconstructedBitmask,
3459 DAG.getShiftAmountConstant(ConcatOperandNumElts, ReturnVT, DL));
3460 }
3461
3462 // Merge disjoint partial bitmasks with OR.
3463 ReconstructedBitmask =
3464 DAG.getNode(ISD::OR, DL, ReturnVT, ReconstructedBitmask,
3465 ExtendedConcatOperandBitmask);
3466 }
3467
3468 return ReconstructedBitmask;
3469 }
3470
3471 return SDValue();
3472}
3473
3475 // bitmask (setcc <X>, 0, setlt) => bitmask X
3476 assert(N->getOpcode() == ISD::INTRINSIC_WO_CHAIN);
3477 using namespace llvm::SDPatternMatch;
3478
3479 if (N->getConstantOperandVal(0) != Intrinsic::wasm_bitmask)
3480 return SDValue();
3481
3482 SDValue LHS;
3483 if (!sd_match(N->getOperand(1), m_c_SetCC(m_Value(LHS), m_Zero(),
3485 return SDValue();
3486
3487 SDLoc DL(N);
3488 return DAG.getNode(
3489 ISD::INTRINSIC_WO_CHAIN, DL, N->getValueType(0),
3490 {DAG.getConstant(Intrinsic::wasm_bitmask, DL, MVT::i32), LHS});
3491}
3492
3494 // any_true (setcc <X>, 0, eq) => (not (all_true X))
3495 // all_true (setcc <X>, 0, eq) => (not (any_true X))
3496 // any_true (setcc <X>, 0, ne) => (any_true X)
3497 // all_true (setcc <X>, 0, ne) => (all_true X)
3498 assert(N->getOpcode() == ISD::INTRINSIC_WO_CHAIN);
3499 using namespace llvm::SDPatternMatch;
3500
3501 SDValue LHS;
3502 if (N->getNumOperands() < 2 ||
3503 !sd_match(N->getOperand(1),
3505 return SDValue();
3506 EVT LT = LHS.getValueType();
3507 if (LT.getScalarSizeInBits() > 128 / LT.getVectorNumElements())
3508 return SDValue();
3509
3510 auto CombineSetCC = [&N, &DAG](Intrinsic::WASMIntrinsics InPre,
3511 ISD::CondCode SetType,
3512 Intrinsic::WASMIntrinsics InPost) {
3513 if (N->getConstantOperandVal(0) != InPre)
3514 return SDValue();
3515
3516 SDValue LHS;
3517 if (!sd_match(N->getOperand(1), m_c_SetCC(m_Value(LHS), m_Zero(),
3518 m_SpecificCondCode(SetType))))
3519 return SDValue();
3520
3521 SDLoc DL(N);
3522 SDValue Ret = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3523 {DAG.getConstant(InPost, DL, MVT::i32), LHS});
3524 if (SetType == ISD::SETEQ)
3525 Ret = DAG.getNode(ISD::XOR, DL, MVT::i32, Ret,
3526 DAG.getConstant(1, DL, MVT::i32));
3527 return DAG.getZExtOrTrunc(Ret, DL, N->getValueType(0));
3528 };
3529
3530 if (SDValue AnyTrueEQ = CombineSetCC(Intrinsic::wasm_anytrue, ISD::SETEQ,
3531 Intrinsic::wasm_alltrue))
3532 return AnyTrueEQ;
3533 if (SDValue AllTrueEQ = CombineSetCC(Intrinsic::wasm_alltrue, ISD::SETEQ,
3534 Intrinsic::wasm_anytrue))
3535 return AllTrueEQ;
3536 if (SDValue AnyTrueNE = CombineSetCC(Intrinsic::wasm_anytrue, ISD::SETNE,
3537 Intrinsic::wasm_anytrue))
3538 return AnyTrueNE;
3539 if (SDValue AllTrueNE = CombineSetCC(Intrinsic::wasm_alltrue, ISD::SETNE,
3540 Intrinsic::wasm_alltrue))
3541 return AllTrueNE;
3542
3543 return SDValue();
3544}
3545
3551
3553 unsigned NumElts,
3554 const MaskReduceInfo &Info,
3555 SelectionDAG &DAG) {
3556 EVT VecVT = FromVT.changeVectorElementType(*DAG.getContext(),
3557 MVT::getIntegerVT(128 / NumElts));
3558 assert(VecVT.getSizeInBits() == 128 &&
3559 "mask reduction should be widened to a 128-bit vector");
3560
3561 SDLoc DL(N);
3562 SDValue Mask = N->getOperand(0)->getOperand(0);
3563 SDValue Ret = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3564 {DAG.getConstant(Info.IID, DL, MVT::i32),
3565 DAG.getSExtOrTrunc(Mask, DL, VecVT)});
3566 if (Info.Invert)
3567 Ret = DAG.getNode(ISD::XOR, DL, MVT::i32, Ret,
3568 DAG.getConstant(1, DL, MVT::i32));
3569 return DAG.getZExtOrTrunc(Ret, DL, N->getValueType(0));
3570}
3571
3573 unsigned NumElts,
3574 const MaskReduceInfo &Info,
3575 SelectionDAG &DAG) {
3576 assert((NumElts == 32 || NumElts == 64) &&
3577 "combineWideMaskReduction is only for wide masks");
3578 assert(MaskVT.isFixedLengthVector() &&
3579 MaskVT.getVectorElementType() == MVT::i1);
3580 SDLoc DL(N);
3581 unsigned ChunkElts = 16;
3582 EVT ChunkMaskVT = EVT::getVectorVT(*DAG.getContext(), MVT::i1,
3583 ElementCount::getFixed(ChunkElts));
3584 EVT LegalVecVT = ChunkMaskVT.changeVectorElementType(
3585 *DAG.getContext(), MVT::getIntegerVT(128 / ChunkElts));
3586
3587 SmallVector<SDValue, 4> ChunkResults;
3588 // Split the wide mask into v16i1 chunks and reduce each chunk separately.
3589 // For example:
3590 // v32i1: [0..15] [16..31]
3591 // | |
3592 // v v
3593 // chunk0 chunk1
3594 //
3595 // v64i1: [0..15] [16..31] [32..47] [48..63]
3596 // | | | |
3597 // v v v v
3598 // chunk0 chunk1 chunk2 chunk3
3599 //
3600 // each chunk:
3601 // v16i1 -> v16i8 -> wasm_anytrue/alltrue -> i32 0/1
3602 for (unsigned I = 0; I < NumElts; I += ChunkElts) {
3603 SDValue ChunkMask = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, ChunkMaskVT,
3604 Mask, DAG.getVectorIdxConstant(I, DL));
3605 SDValue LegalMask = DAG.getSExtOrTrunc(ChunkMask, DL, LegalVecVT);
3606 SDValue Reduced =
3607 DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3608 DAG.getConstant(Info.IID, DL, MVT::i32), LegalMask);
3609 ChunkResults.push_back(Reduced);
3610 }
3611
3612 SDValue Acc = ChunkResults[0];
3613 for (unsigned I = 1; I < ChunkResults.size(); ++I)
3614 Acc =
3615 DAG.getNode(Info.WideCombineOpcode, DL, MVT::i32, Acc, ChunkResults[I]);
3616
3617 if (Info.Invert)
3618 Acc = DAG.getNode(ISD::XOR, DL, MVT::i32, Acc,
3619 DAG.getConstant(1, DL, MVT::i32));
3620
3621 return DAG.getZExtOrTrunc(Acc, DL, N->getValueType(0));
3622}
3623
3624static std::optional<MaskReduceInfo> classifyMaskReduction(SDNode *N) {
3625 auto *C = dyn_cast<ConstantSDNode>(N->getOperand(1));
3626 if (!C)
3627 return std::nullopt;
3628
3629 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
3630
3631 // setcc (bitcast mask), 0, ne -> any_true(mask)
3632 if (C->isZero() && CC == ISD::SETNE)
3633 return MaskReduceInfo{Intrinsic::wasm_anytrue, ISD::OR, false};
3634
3635 // setcc (bitcast mask), 0, eq -> !any_true(mask)
3636 if (C->isZero() && CC == ISD::SETEQ)
3637 return MaskReduceInfo{Intrinsic::wasm_anytrue, ISD::OR, true};
3638
3639 // setcc (bitcast mask), -1, eq -> all_true(mask)
3640 if (C->isAllOnes() && CC == ISD::SETEQ)
3641 return MaskReduceInfo{Intrinsic::wasm_alltrue, ISD::AND, false};
3642
3643 // setcc (bitcast mask), -1, ne -> !all_true(mask)
3644 if (C->isAllOnes() && CC == ISD::SETNE)
3645 return MaskReduceInfo{Intrinsic::wasm_alltrue, ISD::AND, true};
3646
3647 return std::nullopt;
3648}
3649
3650/// Try to convert a i128 comparison to a v16i8 comparison before type
3651/// legalization splits it up into chunks
3652static SDValue
3654 const WebAssemblySubtarget *Subtarget) {
3655
3656 SDLoc DL(N);
3657 SDValue X = N->getOperand(0);
3658 SDValue Y = N->getOperand(1);
3659 EVT VT = N->getValueType(0);
3660 EVT OpVT = X.getValueType();
3661
3662 SelectionDAG &DAG = DCI.DAG;
3664 Attribute::NoImplicitFloat))
3665 return SDValue();
3666
3667 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
3668 // We're looking for an oversized integer equality comparison with SIMD
3669 if (!OpVT.isScalarInteger() || !OpVT.isByteSized() || OpVT != MVT::i128 ||
3670 !Subtarget->hasSIMD128() || !isIntEqualitySetCC(CC))
3671 return SDValue();
3672
3673 // Don't perform this combine if constructing the vector will be expensive.
3674 auto IsVectorBitCastCheap = [](SDValue X) {
3676 return isa<ConstantSDNode>(X) || X.getOpcode() == ISD::LOAD;
3677 };
3678
3679 if (!IsVectorBitCastCheap(X) || !IsVectorBitCastCheap(Y))
3680 return SDValue();
3681
3682 SDValue VecX = DAG.getBitcast(MVT::v16i8, X);
3683 SDValue VecY = DAG.getBitcast(MVT::v16i8, Y);
3684 SDValue Cmp = DAG.getSetCC(DL, MVT::v16i8, VecX, VecY, CC);
3685
3686 SDValue Intr =
3687 DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3688 {DAG.getConstant(CC == ISD::SETEQ ? Intrinsic::wasm_alltrue
3689 : Intrinsic::wasm_anytrue,
3690 DL, MVT::i32),
3691 Cmp});
3692
3693 return DAG.getSetCC(DL, VT, Intr, DAG.getConstant(0, DL, MVT::i32),
3694 ISD::SETNE);
3695}
3696
3699 const WebAssemblySubtarget *Subtarget) {
3700 if (!DCI.isBeforeLegalize())
3701 return SDValue();
3702
3703 EVT VT = N->getValueType(0);
3704 if (!VT.isScalarInteger())
3705 return SDValue();
3706
3707 if (SDValue V = combineVectorSizedSetCCEquality(N, DCI, Subtarget))
3708 return V;
3709
3710 SDValue LHS = N->getOperand(0);
3711 if (LHS->getOpcode() != ISD::BITCAST)
3712 return SDValue();
3713
3714 EVT FromVT = LHS->getOperand(0).getValueType();
3715 if (!FromVT.isFixedLengthVectorOf(MVT::i1))
3716 return SDValue();
3717
3718 unsigned NumElts = FromVT.getVectorNumElements();
3719 auto Info = classifyMaskReduction(N);
3720 if (!Info)
3721 return SDValue();
3722
3723 auto &DAG = DCI.DAG;
3724 if (NumElts == 2 || NumElts == 4 || NumElts == 8 || NumElts == 16)
3725 return combineSmallMaskReduction(N, FromVT, NumElts, *Info, DAG);
3726
3727 if (NumElts == 32 || NumElts == 64)
3728 return combineWideMaskReduction(N, LHS.getOperand(0), FromVT, NumElts,
3729 *Info, DAG);
3730
3731 return SDValue();
3732}
3733
3735 EVT VT = N->getValueType(0);
3736 if (VT != MVT::v8i32 && VT != MVT::v16i32)
3737 return SDValue();
3738
3739 // Mul with extending inputs.
3740 SDValue LHS = N->getOperand(0);
3741 SDValue RHS = N->getOperand(1);
3742 if (LHS.getOpcode() != RHS.getOpcode())
3743 return SDValue();
3744
3745 if (LHS.getOpcode() != ISD::SIGN_EXTEND &&
3746 LHS.getOpcode() != ISD::ZERO_EXTEND)
3747 return SDValue();
3748
3749 if (LHS->getOperand(0).getValueType() != RHS->getOperand(0).getValueType())
3750 return SDValue();
3751
3752 EVT FromVT = LHS->getOperand(0).getValueType();
3753 EVT EltTy = FromVT.getVectorElementType();
3754 if (EltTy != MVT::i8)
3755 return SDValue();
3756
3757 // For an input DAG that looks like this
3758 // %a = input_type
3759 // %b = input_type
3760 // %lhs = extend %a to output_type
3761 // %rhs = extend %b to output_type
3762 // %mul = mul %lhs, %rhs
3763
3764 // input_type | output_type | instructions
3765 // v16i8 | v16i32 | %low = i16x8.extmul_low_i8x16_ %a, %b
3766 // | | %high = i16x8.extmul_high_i8x16_, %a, %b
3767 // | | %low_low = i32x4.ext_low_i16x8_ %low
3768 // | | %low_high = i32x4.ext_high_i16x8_ %low
3769 // | | %high_low = i32x4.ext_low_i16x8_ %high
3770 // | | %high_high = i32x4.ext_high_i16x8_ %high
3771 // | | %res = concat_vector(...)
3772 // v8i8 | v8i32 | %low = i16x8.extmul_low_i8x16_ %a, %b
3773 // | | %low_low = i32x4.ext_low_i16x8_ %low
3774 // | | %low_high = i32x4.ext_high_i16x8_ %low
3775 // | | %res = concat_vector(%low_low, %low_high)
3776
3777 SDLoc DL(N);
3778 unsigned NumElts = VT.getVectorNumElements();
3779 SDValue ExtendInLHS = LHS->getOperand(0);
3780 SDValue ExtendInRHS = RHS->getOperand(0);
3781 bool IsSigned = LHS->getOpcode() == ISD::SIGN_EXTEND;
3782 unsigned ExtendLowOpc =
3783 IsSigned ? WebAssemblyISD::EXTEND_LOW_S : WebAssemblyISD::EXTEND_LOW_U;
3784 unsigned ExtendHighOpc =
3785 IsSigned ? WebAssemblyISD::EXTEND_HIGH_S : WebAssemblyISD::EXTEND_HIGH_U;
3786
3787 auto GetExtendLow = [&DAG, &DL, &ExtendLowOpc](EVT VT, SDValue Op) {
3788 return DAG.getNode(ExtendLowOpc, DL, VT, Op);
3789 };
3790 auto GetExtendHigh = [&DAG, &DL, &ExtendHighOpc](EVT VT, SDValue Op) {
3791 return DAG.getNode(ExtendHighOpc, DL, VT, Op);
3792 };
3793
3794 if (NumElts == 16) {
3795 SDValue LowLHS = GetExtendLow(MVT::v8i16, ExtendInLHS);
3796 SDValue LowRHS = GetExtendLow(MVT::v8i16, ExtendInRHS);
3797 SDValue MulLow = DAG.getNode(ISD::MUL, DL, MVT::v8i16, LowLHS, LowRHS);
3798 SDValue HighLHS = GetExtendHigh(MVT::v8i16, ExtendInLHS);
3799 SDValue HighRHS = GetExtendHigh(MVT::v8i16, ExtendInRHS);
3800 SDValue MulHigh = DAG.getNode(ISD::MUL, DL, MVT::v8i16, HighLHS, HighRHS);
3801 SDValue SubVectors[] = {
3802 GetExtendLow(MVT::v4i32, MulLow),
3803 GetExtendHigh(MVT::v4i32, MulLow),
3804 GetExtendLow(MVT::v4i32, MulHigh),
3805 GetExtendHigh(MVT::v4i32, MulHigh),
3806 };
3807 return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, SubVectors);
3808 } else {
3809 assert(NumElts == 8);
3810 SDValue LowLHS = DAG.getNode(LHS->getOpcode(), DL, MVT::v8i16, ExtendInLHS);
3811 SDValue LowRHS = DAG.getNode(RHS->getOpcode(), DL, MVT::v8i16, ExtendInRHS);
3812 SDValue MulLow = DAG.getNode(ISD::MUL, DL, MVT::v8i16, LowLHS, LowRHS);
3813 SDValue Lo = GetExtendLow(MVT::v4i32, MulLow);
3814 SDValue Hi = GetExtendHigh(MVT::v4i32, MulLow);
3815 return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
3816 }
3817 return SDValue();
3818}
3819
3822 assert(N->getOpcode() == ISD::MUL);
3823 EVT VT = N->getValueType(0);
3824 if (!VT.isVector())
3825 return SDValue();
3826
3827 if (auto Res = TryWideExtMulCombine(N, DCI.DAG))
3828 return Res;
3829
3830 // We don't natively support v16i8 or v8i8 mul, but we do support v8i16. So,
3831 // extend them to v8i16.
3832 if (VT != MVT::v8i8 && VT != MVT::v16i8)
3833 return SDValue();
3834
3835 SDLoc DL(N);
3836 SelectionDAG &DAG = DCI.DAG;
3837 SDValue LHS = N->getOperand(0);
3838 SDValue RHS = N->getOperand(1);
3839 EVT MulVT = MVT::v8i16;
3840
3841 if (VT == MVT::v8i8) {
3842 SDValue PromotedLHS = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v16i8, LHS,
3843 DAG.getUNDEF(MVT::v8i8));
3844 SDValue PromotedRHS = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v16i8, RHS,
3845 DAG.getUNDEF(MVT::v8i8));
3846 SDValue LowLHS =
3847 DAG.getNode(WebAssemblyISD::EXTEND_LOW_U, DL, MulVT, PromotedLHS);
3848 SDValue LowRHS =
3849 DAG.getNode(WebAssemblyISD::EXTEND_LOW_U, DL, MulVT, PromotedRHS);
3850 SDValue MulLow = DAG.getBitcast(
3851 MVT::v16i8, DAG.getNode(ISD::MUL, DL, MulVT, LowLHS, LowRHS));
3852 // Take the low byte of each lane.
3853 SDValue Shuffle = DAG.getVectorShuffle(
3854 MVT::v16i8, DL, MulLow, DAG.getUNDEF(MVT::v16i8),
3855 {0, 2, 4, 6, 8, 10, 12, 14, -1, -1, -1, -1, -1, -1, -1, -1});
3856 return extractSubVector(Shuffle, 0, DAG, DL, 64);
3857 } else {
3858 assert(VT == MVT::v16i8 && "Expected v16i8");
3859 SDValue LowLHS = DAG.getNode(WebAssemblyISD::EXTEND_LOW_U, DL, MulVT, LHS);
3860 SDValue LowRHS = DAG.getNode(WebAssemblyISD::EXTEND_LOW_U, DL, MulVT, RHS);
3861 SDValue HighLHS =
3862 DAG.getNode(WebAssemblyISD::EXTEND_HIGH_U, DL, MulVT, LHS);
3863 SDValue HighRHS =
3864 DAG.getNode(WebAssemblyISD::EXTEND_HIGH_U, DL, MulVT, RHS);
3865
3866 SDValue MulLow =
3867 DAG.getBitcast(VT, DAG.getNode(ISD::MUL, DL, MulVT, LowLHS, LowRHS));
3868 SDValue MulHigh =
3869 DAG.getBitcast(VT, DAG.getNode(ISD::MUL, DL, MulVT, HighLHS, HighRHS));
3870
3871 // Take the low byte of each lane.
3872 return DAG.getVectorShuffle(
3873 VT, DL, MulLow, MulHigh,
3874 {0, 2, 4, 6, 8, 10, 12, 14, 16, 18, 20, 22, 24, 26, 28, 30});
3875 }
3876}
3877
3878SDValue DoubleVectorWidth(SDValue In, unsigned RequiredNumElems,
3879 SelectionDAG &DAG) {
3880 SDLoc DL(In);
3881 LLVMContext &Ctx = *DAG.getContext();
3882 EVT InVT = In.getValueType();
3883 unsigned NumElems = InVT.getVectorNumElements() * 2;
3884 EVT OutVT = EVT::getVectorVT(Ctx, InVT.getVectorElementType(), NumElems);
3885 SDValue Concat =
3886 DAG.getNode(ISD::CONCAT_VECTORS, DL, OutVT, In, DAG.getPOISON(InVT));
3887 if (NumElems < RequiredNumElems) {
3888 return DoubleVectorWidth(Concat, RequiredNumElems, DAG);
3889 }
3890 return Concat;
3891}
3892
3894 EVT OutVT = N->getValueType(0);
3895 if (!OutVT.isVector())
3896 return SDValue();
3897
3898 EVT OutElTy = OutVT.getVectorElementType();
3899 if (OutElTy != MVT::i8 && OutElTy != MVT::i16)
3900 return SDValue();
3901
3902 unsigned NumElems = OutVT.getVectorNumElements();
3903 if (!isPowerOf2_32(NumElems))
3904 return SDValue();
3905
3906 EVT FPVT = N->getOperand(0)->getValueType(0);
3907 if (FPVT.getVectorElementType() != MVT::f32)
3908 return SDValue();
3909
3910 SDLoc DL(N);
3911
3912 // First, convert to i32.
3913 LLVMContext &Ctx = *DAG.getContext();
3914 EVT IntVT = EVT::getVectorVT(Ctx, MVT::i32, NumElems);
3915 SDValue ToInt = DAG.getNode(N->getOpcode(), DL, IntVT, N->getOperand(0));
3917 OutVT.getScalarSizeInBits());
3918 // Mask out the top MSBs.
3919 SDValue Masked =
3920 DAG.getNode(ISD::AND, DL, IntVT, ToInt, DAG.getConstant(Mask, DL, IntVT));
3921
3922 if (OutVT.getSizeInBits() < 128) {
3923 // Create a wide enough vector that we can use narrow.
3924 EVT NarrowedVT = OutElTy == MVT::i8 ? MVT::v16i8 : MVT::v8i16;
3925 unsigned NumRequiredElems = NarrowedVT.getVectorNumElements();
3926 SDValue WideVector = DoubleVectorWidth(Masked, NumRequiredElems, DAG);
3927 SDValue Trunc = truncateVectorWithNARROW(NarrowedVT, WideVector, DL, DAG);
3928 return DAG.getBitcast(
3929 OutVT, extractSubVector(Trunc, 0, DAG, DL, OutVT.getSizeInBits()));
3930 } else {
3931 return truncateVectorWithNARROW(OutVT, Masked, DL, DAG);
3932 }
3933 return SDValue();
3934}
3935
3936// Wide vector shift operations such as v8i32 with sign-extended
3937// operands cause Type Legalizer crashes because the target-specific
3938// extension nodes cannot be directly mapped to the 256-bit size.
3939//
3940// To resolve the crash and optimize performance, we intercept the
3941// illegal v8i32 shift in DAGCombine. We convert the shift amounts
3942// into multipliers and manually split the vector into two v4i32 halves.
3943//
3944// Before: t1: v8i32 = shl (sign_extend v8i16), const_vec
3945// After : t2: v4i32 = mul (ext_low_s v8i16), (ext_low_s narrow_vec)
3946// t3: v4i32 = mul (ext_high_s v8i16), (ext_high_s narrow_vec)
3947// t4: v8i32 = concat_vectors t2, t3
3950 SelectionDAG &DAG = DCI.DAG;
3951 assert(N->getOpcode() == ISD::SHL);
3952 EVT VT = N->getValueType(0);
3953 if (VT != MVT::v8i32)
3954 return SDValue();
3955
3956 SDValue LHS = N->getOperand(0);
3957 SDValue RHS = N->getOperand(1);
3958 unsigned ExtOpc = LHS.getOpcode();
3959 if (ExtOpc != ISD::SIGN_EXTEND && ExtOpc != ISD::ZERO_EXTEND)
3960 return SDValue();
3961
3962 if (RHS.getOpcode() != ISD::BUILD_VECTOR)
3963 return SDValue();
3964
3965 SDLoc DL(N);
3966 SDValue ExtendIn = LHS.getOperand(0);
3967 EVT FromVT = ExtendIn.getValueType();
3968 if (FromVT != MVT::v8i16)
3969 return SDValue();
3970
3971 unsigned NumElts = VT.getVectorNumElements();
3972 unsigned BitWidth = FromVT.getScalarSizeInBits();
3973 bool IsSigned = (ExtOpc == ISD::SIGN_EXTEND);
3974 unsigned MaxValidShift = IsSigned ? (BitWidth - 1) : BitWidth;
3975 SmallVector<SDValue, 16> MulConsts;
3976 for (unsigned I = 0; I < NumElts; ++I) {
3977 auto *C = dyn_cast<ConstantSDNode>(RHS.getOperand(I));
3978 if (!C)
3979 return SDValue();
3980
3981 const APInt &ShiftAmt = C->getAPIntValue();
3982 if (ShiftAmt.uge(MaxValidShift))
3983 return SDValue();
3984
3985 APInt MulAmt = APInt::getOneBitSet(BitWidth, ShiftAmt.getZExtValue());
3986 MulConsts.push_back(DAG.getConstant(MulAmt, DL, FromVT.getScalarType(),
3987 /*isTarget=*/false, /*isOpaque=*/true));
3988 }
3989
3990 SDValue NarrowConst = DAG.getBuildVector(FromVT, DL, MulConsts);
3991 unsigned ExtLowOpc =
3992 IsSigned ? WebAssemblyISD::EXTEND_LOW_S : WebAssemblyISD::EXTEND_LOW_U;
3993 unsigned ExtHighOpc =
3994 IsSigned ? WebAssemblyISD::EXTEND_HIGH_S : WebAssemblyISD::EXTEND_HIGH_U;
3995
3996 EVT HalfVT = MVT::v4i32;
3997 SDValue LHSLo = DAG.getNode(ExtLowOpc, DL, HalfVT, ExtendIn);
3998 SDValue LHSHi = DAG.getNode(ExtHighOpc, DL, HalfVT, ExtendIn);
3999 SDValue RHSLo = DAG.getNode(ExtLowOpc, DL, HalfVT, NarrowConst);
4000 SDValue RHSHi = DAG.getNode(ExtHighOpc, DL, HalfVT, NarrowConst);
4001 SDValue MulLo = DAG.getNode(ISD::MUL, DL, HalfVT, LHSLo, RHSLo);
4002 SDValue MulHi = DAG.getNode(ISD::MUL, DL, HalfVT, LHSHi, RHSHi);
4003 return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, MulLo, MulHi);
4004}
4005
4007 if (N->getValueType(0) != MVT::f128)
4008 return SDValue();
4009
4010 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
4011 switch (N->getOpcode()) {
4012 // wasi-libc and emscripten do not currently define fminimuml and fmaximuml.
4013 case ISD::FMINIMUM:
4014 case ISD::FMAXIMUM:
4015 return TLI.expandFMINIMUM_FMAXIMUM(N, DAG);
4016
4017 // wasi-libc and emscripten do not currently define fminimum_numl and
4018 // fmaximum_numl.
4019 case ISD::FMINIMUMNUM:
4020 case ISD::FMAXIMUMNUM:
4021 return TLI.expandFMINIMUMNUM_FMAXIMUMNUM(N, DAG);
4022
4023 default:
4024 return SDValue();
4025 }
4026}
4027
4028SDValue
4029WebAssemblyTargetLowering::PerformDAGCombine(SDNode *N,
4030 DAGCombinerInfo &DCI) const {
4031 switch (N->getOpcode()) {
4032 default:
4033 return SDValue();
4034 case ISD::BITCAST:
4035 return performBitcastCombine(N, DCI);
4036 case ISD::SETCC:
4037 return performSETCCCombine(N, DCI, Subtarget);
4039 return performVECTOR_SHUFFLECombine(N, DCI);
4040 case ISD::SIGN_EXTEND:
4041 case ISD::ZERO_EXTEND:
4042 return performVectorExtendCombine(N, DCI);
4043 case ISD::UINT_TO_FP:
4044 if (auto ExtCombine = performVectorExtendToFPCombine(N, DCI))
4045 return ExtCombine;
4046 return performVectorNonNegToFPCombine(N, DCI);
4047 case ISD::SINT_TO_FP:
4048 return performVectorExtendToFPCombine(N, DCI);
4051 case ISD::FP_ROUND:
4053 return performVectorTruncZeroCombine(N, DCI);
4054 case ISD::FP_TO_SINT:
4055 case ISD::FP_TO_UINT:
4056 return performConvertFPCombine(N, DCI.DAG);
4057 case ISD::TRUNCATE:
4058 return performTruncateCombine(N, DCI);
4060 if (SDValue V = performBitmaskCombine(N, DCI.DAG))
4061 return V;
4062 return performAnyAllCombine(N, DCI.DAG);
4063 }
4064 case ISD::MUL:
4065 return performMulCombine(N, DCI);
4066 case ISD::SHL:
4067 return performShiftCombine(N, DCI);
4068 case ISD::FMINIMUM:
4069 case ISD::FMAXIMUM:
4070 case ISD::FMINIMUMNUM:
4071 case ISD::FMAXIMUMNUM:
4072 return performMinMaxF128Combine(N, DCI.DAG);
4073 }
4074}
static SDValue performMulCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const AArch64Subtarget *Subtarget)
static SDValue performTruncateCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI)
return SDValue()
static SDValue performSETCCCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, SelectionDAG &DAG)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis false
Function Alias Analysis Results
static void fail(const SDLoc &DL, SelectionDAG &DAG, const Twine &Msg, SDValue Val={})
#define X(NUM, ENUM, NAME)
Definition ELF.h:856
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
Hexagon Common GEP
const HexagonInstrInfo * TII
#define _
IRTranslator LLVM IR MI
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define T
MachineInstr unsigned OpIdx
static SDValue performVECTOR_SHUFFLECombine(SDNode *N, SelectionDAG &DAG, const RISCVSubtarget &Subtarget, const RISCVTargetLowering &TLI)
static SDValue combineVectorSizedSetCCEquality(EVT VT, SDValue X, SDValue Y, ISD::CondCode CC, const SDLoc &DL, SelectionDAG &DAG, const RISCVSubtarget &Subtarget)
Try to map an integer comparison with size > XLEN to vector instructions before type legalization spl...
Contains matchers for matching SelectionDAG nodes and values.
const char * Msg
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static bool callingConvSupported(CallingConv::ID CallConv)
static MachineBasicBlock * LowerFPToInt(MachineInstr &MI, DebugLoc DL, MachineBasicBlock *BB, const TargetInstrInfo &TII, bool IsUnsigned, bool Int64, bool Float64, unsigned LoweredOpcode)
static SDValue TryWideExtMulCombine(SDNode *N, SelectionDAG &DAG)
static MachineBasicBlock * LowerMemcpy(MachineInstr &MI, DebugLoc DL, MachineBasicBlock *BB, const TargetInstrInfo &TII, bool Int64)
static std::optional< unsigned > IsWebAssemblyLocal(SDValue Op, SelectionDAG &DAG)
static SDValue performVectorExtendCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue performVectorNonNegToFPCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue unrollVectorShift(SDValue Op, SelectionDAG &DAG)
static SDValue performAnyAllCombine(SDNode *N, SelectionDAG &DAG)
static MachineBasicBlock * LowerCallResults(MachineInstr &CallResults, DebugLoc DL, MachineBasicBlock *BB, const WebAssemblySubtarget *Subtarget, const TargetInstrInfo &TII)
static std::optional< MaskReduceInfo > classifyMaskReduction(SDNode *N)
static SDValue GetExtendHigh(SDValue Op, unsigned UserOpc, EVT VT, SelectionDAG &DAG)
SDValue performConvertFPCombine(SDNode *N, SelectionDAG &DAG)
static SDValue performBitmaskCombine(SDNode *N, SelectionDAG &DAG)
static SDValue performVectorTruncZeroCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static bool IsWebAssemblyGlobal(SDValue Op)
static SDValue combineSmallMaskReduction(SDNode *N, EVT FromVT, unsigned NumElts, const MaskReduceInfo &Info, SelectionDAG &DAG)
static MachineBasicBlock * LowerMemset(MachineInstr &MI, DebugLoc DL, MachineBasicBlock *BB, const TargetInstrInfo &TII, bool Int64)
static bool HasNoSignedZerosOrNaNs(SDValue Op, SelectionDAG &DAG)
SDValue DoubleVectorWidth(SDValue In, unsigned RequiredNumElems, SelectionDAG &DAG)
static SDValue performVectorExtendToFPCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
Convert ({u,s}itofp vec) --> ({u,s}itofp ({s,z}ext vec)) so it doesn't get split up into scalar instr...
static SDValue performShiftCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue LowerConvertLow(SDValue Op, SelectionDAG &DAG)
static SDValue extractSubVector(SDValue Vec, unsigned IdxVal, SelectionDAG &DAG, const SDLoc &DL, unsigned VectorWidth)
static SDValue performBitcastCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue truncateVectorWithNARROW(EVT DstVT, SDValue In, const SDLoc &DL, SelectionDAG &DAG)
static SDValue performMinMaxF128Combine(SDNode *N, SelectionDAG &DAG)
static SDValue combineWideMaskReduction(SDNode *N, SDValue Mask, EVT MaskVT, unsigned NumElts, const MaskReduceInfo &Info, SelectionDAG &DAG)
This file defines the interfaces that WebAssembly uses to lower LLVM code into a selection DAG.
This file provides WebAssembly-specific target descriptions.
This file declares WebAssembly-specific per-machine-function information.
This file declares the WebAssembly-specific subclass of TargetSubtarget.
This file declares the WebAssembly-specific subclass of TargetMachine.
This file contains the declaration of the WebAssembly-specific type parsing utility functions.
This file contains the declaration of the WebAssembly-specific utility functions.
X86 cmov Conversion
static constexpr int Concat[]
Value * RHS
Value * LHS
The Input class is used to parse a yaml document into in-memory structs and vectors.
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1565
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:307
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:297
static APInt getOneBitSet(unsigned numBits, unsigned BitNo)
Return an APInt with exactly one bit set in the result.
Definition APInt.h:240
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Definition APInt.h:1230
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
an instruction that atomically reads a memory location, combines it with another value,...
@ Add
*p = old + v
@ Sub
*p = old - v
@ And
*p = old & v
@ Xor
*p = old ^ v
BinOp getOperation() const
LLVM Basic Block Representation.
Definition BasicBlock.h:62
static CCValAssign getMem(unsigned ValNo, MVT ValVT, int64_t Offset, MVT LocVT, LocInfo HTP, bool IsCustom=false)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
Diagnostic information for unsupported feature in backend.
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:309
This is a fast-path instruction selection class that generates poor code and doesn't support illegal ...
Definition FastISel.h:67
FunctionLoweringInfo - This contains information that is global to a function that is used when lower...
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:211
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:353
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:727
LLVM_ABI unsigned getAddressSpace() const
const GlobalValue * getGlobal() const
ThreadLocalMode getThreadLocalMode() const
Type * getValueType() const
unsigned getTargetFlags() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
Tracks which library functions to use for a particular subtarget.
const SDValue & getBasePtr() const
const SDValue & getOffset() const
Describe properties that are true of each instruction in the target description file.
Machine Value Type.
bool is128BitVector() const
Return true if this is a 128-bit vector type.
@ INVALID_SIMPLE_VALUE_TYPE
static auto integer_fixedlen_vector_valuetypes()
SimpleValueType SimpleTy
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
static auto integer_valuetypes()
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
static auto fixedlen_vector_valuetypes()
bool isFixedLengthVector() const
static MVT getVectorVT(MVT VT, unsigned NumElements)
MVT getVectorElementType() const
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
static MVT getIntegerVT(unsigned BitWidth)
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI instr_iterator insert(instr_iterator I, MachineInstr *M)
Insert MI into the instruction list before I, possibly inside a bundle.
const BasicBlock * getBasicBlock() const
Return the LLVM basic block that this instance corresponded to originally.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator insertAfter(iterator I, MachineInstr *MI)
Insert MI into the instruction list after I.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
LLVM_ABI int CreateStackObject(uint64_t Size, Align Alignment, bool isSpillSlot, const AllocaInst *Alloca=nullptr, uint8_t ID=0)
Create a new statically sized stack object, returning a nonnegative identifier to represent it.
void setFrameAddressIsTaken(bool T)
unsigned getFunctionNumber() const
getFunctionNumber - Return a unique ID for the current function.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
const char * createExternalSymbolName(StringRef Name)
Allocate a string and populate it with the given external symbol name.
MCContext & getContext() const
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const MachineJumpTableInfo * getJumpTableInfo() const
getJumpTableInfo - Return the jump table info object for the current function.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addSym(MCSymbol *Sym, unsigned char TargetFlags=0) const
const MachineInstrBuilder & addFPImm(const ConstantFP *Val) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
mop_range defs()
Returns all explicit operands that are register definitions.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
mop_range uses()
Returns all operands which may be register uses.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
const std::vector< MachineJumpTableEntry > & getJumpTables() const
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsKill(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
void addLiveIn(MCRegister Reg, Register vreg=Register())
addLiveIn - Add the specified register as a live-in.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const SDValue & getChain() const
EVT getMemoryVT() const
Return the type of the in-memory value.
static PointerType * getUnqual(LLVMContext &C)
This constructs an opaque pointer to an object in the default address space (address space zero).
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
ArrayRef< SDUse > ops() const
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
const SDValue & getOperand(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getOpcode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
LLVM_ABI bool isKnownNeverLogicalZero(SDValue Op, const APInt &DemandedElts, unsigned Depth=0) const
Test whether the given floating point SDValue (or all elements of it, if it is a vector) is known to ...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getShiftAmountConstant(uint64_t Val, EVT VT, const SDLoc &DL)
LLVM_ABI SDValue getSplatValue(SDValue V, bool LegalTypes=false)
If V is a splat vector, return its scalar source operand by extracting that element from the source v...
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI void ExtractVectorElements(SDValue Op, SmallVectorImpl< SDValue > &Args, unsigned Start=0, unsigned Count=0, EVT EltVT=EVT())
Append the extracted elements from Start to Count out of the vector Op in Args.
LLVM_ABI SDValue UnrollVectorOp(SDNode *N, unsigned ResNE=0)
Utility function used by legalize and lowering to "unroll" a vector operation by splitting out the sc...
LLVM_ABI SDValue getConstantFP(double Val, const SDLoc &DL, EVT VT, bool isTarget=false)
Create a ConstantFPSDNode wrapping a constant value.
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
SDValue getSetCC(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode Cond, SDValue Chain=SDValue(), bool IsSignaling=false, SDNodeFlags Flags={})
Helper function to make it easier to build SetCC's if you just have an ISD::CondCode instead of an SD...
LLVM_ABI SDValue getMemcpy(SDValue Chain, const SDLoc &dl, SDValue Dst, SDValue Src, SDValue Size, Align DstAlign, Align SrcAlign, bool isVol, bool AlwaysInline, const CallInst *CI, std::optional< bool > OverrideTailCall, MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo, const AAMDNodes &AAInfo=AAMDNodes(), BatchAAResults *BatchAA=nullptr)
const TargetLowering & getTargetLoweringInfo() const
SDValue getTargetJumpTable(int JTI, EVT VT, unsigned TargetFlags=0)
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
const DataLayout & getDataLayout() const
SDValue getTargetFrameIndex(int FI, EVT VT)
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const AAMDNodes &AAInfo=AAMDNodes())
Helper function to build ISD::STORE nodes.
LLVM_ABI bool SignBitIsZero(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero.
LLVM_ABI SDValue getBasicBlock(MachineBasicBlock *MBB)
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
const TargetMachine & getTarget() const
LLVM_ABI SDValue getAnyExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either any-extending or truncat...
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI bool isKnownNeverNaN(SDValue Op, const APInt &DemandedElts, bool SNaN=false, unsigned Depth=0) const
Test whether the given SDValue (or all elements of it, if it is a vector) is known to never be NaN in...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
SDValue getSplatBuildVector(EVT VT, const SDLoc &DL, SDValue Op)
Return a splat ISD::BUILD_VECTOR node, consisting of Op splatted to all elements.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetExternalSymbol(const char *Sym, EVT VT, unsigned TargetFlags=0)
LLVM_ABI SDValue getMCSymbol(MCSymbol *Sym, EVT VT)
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
const SDValue & getBasePtr() const
const SDValue & getOffset() const
const SDValue & getValue() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
TargetInstrInfo - Interface to description of machine instruction set.
Provides information about what library functions are available for the current target.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
const TargetMachine & getTargetMachine() const
unsigned MaxLoadsPerMemcmp
Specify maximum number of load instructions per memcmp call.
LegalizeTypeAction
This enum indicates whether a types are legal for a target, and if not, what action should be used to...
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
virtual TargetLoweringBase::LegalizeTypeAction getPreferredVectorAction(MVT VT) const
Return the preferred vector type legalization action.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setMinimumJumpTableEntries(unsigned Val)
Indicate the minimum number of blocks to generate jump tables.
void setPartialReduceMLAAction(unsigned Opc, MVT AccVT, MVT InputVT, LegalizeAction Action)
Indicate how a PARTIAL_REDUCE_U/SMLA node with Acc type AccVT and Input type InputVT should be treate...
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
unsigned MaxLoadsPerMemcmpOptSize
Likewise for functions with the OptSize attribute.
virtual bool isBinOp(unsigned Opcode) const
Return true if the node is a math/logic binary operator.
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setCondCodeAction(ArrayRef< ISD::CondCode > CCs, MVT VT, LegalizeAction Action)
Indicate that the specified condition code is or isn't supported on the target and indicate what to d...
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
bool isOperationLegalOrCustomOrPromote(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
SDValue expandFMINIMUMNUM_FMAXIMUMNUM(SDNode *N, SelectionDAG &DAG) const
Expand fminimumnum/fmaximumnum into multiple comparison with selects.
SDValue expandFMINIMUM_FMAXIMUM(SDNode *N, SelectionDAG &DAG) const
Expand fminimum/fmaximum into multiple comparison with selects.
bool isPositionIndependent() const
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
TargetLowering(const TargetLowering &)=delete
virtual bool isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const
Return true if folding a constant offset with the given GlobalAddress is legal.
std::pair< SDValue, SDValue > makeLibCall(SelectionDAG &DAG, RTLIB::LibcallImpl LibcallImpl, EVT RetVT, ArrayRef< SDValue > Ops, MakeLibCallOptions CallOptions, const SDLoc &dl, SDValue Chain=SDValue()) const
Returns a pair of (return value, chain).
Primary interface to the complete machine description for the target machine.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isFunctionTy() const
True if this is an instance of FunctionType.
Definition Type.h:273
static LLVM_ABI Type * getDoubleTy(LLVMContext &C)
Definition Type.cpp:287
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:286
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI const Value * stripPointerCastsAndAliases() const
Strip off pointer casts, all-zero GEPs, address space casts, and aliases.
Definition Value.cpp:717
static std::optional< unsigned > getLocalForStackObject(MachineFunction &MF, int FrameIndex)
WebAssemblyTargetLowering(const TargetMachine &TM, const WebAssemblySubtarget &STI)
self_iterator getIterator()
Definition ilist_node.h:123
#define INT64_MIN
Definition DataTypes.h:74
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ Swift
Calling convention for Swift.
Definition CallingConv.h:69
@ PreserveMost
Used for runtime calls that preserves most registers.
Definition CallingConv.h:63
@ CXX_FAST_TLS
Used for access functions.
Definition CallingConv.h:72
@ WASM_EmscriptenInvoke
For emscripten __invoke_* functions.
@ Cold
Attempts to make code in the caller as efficient as possible under the assumption that the call is no...
Definition CallingConv.h:47
@ PreserveAll
Used for runtime calls that preserves (almost) all registers.
Definition CallingConv.h:66
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ SwiftTail
This follows the Swift calling convention in how arguments are passed but guarantees tail calls will ...
Definition CallingConv.h:87
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:829
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ PARTIAL_REDUCE_SMLA
PARTIAL_REDUCE_[U|S]MLA(Accumulator, Input1, Input2) The partial reduction nodes sign or zero extend ...
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:275
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:789
@ VAEND
VAEND, VASTART - VAEND and VASTART have three operands: an input chain, pointer, and a SRCVALUE.
@ ADDC
Carry-setting nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:294
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:520
@ PSEUDO_FMIN
PSEUDO_FMIN is strictly equivalent to op0 olt op1 ?
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:220
@ GlobalAddress
Definition ISDOpcodes.h:88
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:890
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:586
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:749
@ SIGN_EXTEND_VECTOR_INREG
SIGN_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register sign-extension of the low ...
Definition ISDOpcodes.h:920
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:280
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ FMULADD
FMULADD - Performs a * b + c, with, or without, intermediate rounding.
Definition ISDOpcodes.h:530
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:254
@ BUILTIN_OP_END
BUILTIN_OP_END - This must be the last enum value in this list.
@ GlobalTLSAddress
Definition ISDOpcodes.h:89
@ PARTIAL_REDUCE_UMLA
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:667
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ BR_CC
BR_CC - Conditional branch.
@ BRIND
BRIND - Indirect branch.
@ BR_JT
BR_JT - Jumptable branch.
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
Definition ISDOpcodes.h:374
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:247
@ SPLAT_VECTOR
SPLAT_VECTOR(VAL) - Returns a vector with the scalar value VAL duplicated in all lanes.
Definition ISDOpcodes.h:674
@ VACOPY
VACOPY - VACOPY has 5 operands: an input chain, a destination pointer, a source pointer,...
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:706
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:771
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:651
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:616
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:578
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:224
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:860
@ DEBUGTRAP
DEBUGTRAP - Trap intended to get the attention of a debugger.
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:821
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ ANY_EXTEND_VECTOR_INREG
ANY_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register any-extension of the low la...
Definition ISDOpcodes.h:909
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:898
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:729
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:988
@ FRAMEADDR
FRAMEADDR, RETURNADDR - These nodes represent llvm.frameaddress and llvm.returnaddress on the DAG.
Definition ISDOpcodes.h:110
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:936
@ TargetConstant
TargetConstant* - Like Constant*, but the DAG does not do any folding, simplification,...
Definition ISDOpcodes.h:179
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:741
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:205
@ ADDE
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:304
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:567
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:53
@ ExternalSymbol
Definition ISDOpcodes.h:93
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ CLEAR_CACHE
llvm.clear_cache intrinsic Operands: Input Chain, Start Addres, End Address Outputs: Output Chain
@ ZERO_EXTEND_VECTOR_INREG
ZERO_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register zero-extension of the low ...
Definition ISDOpcodes.h:931
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:955
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
@ VAARG
VAARG - VAARG has four operands: an input chain, a pointer, a SRCVALUE, and the alignment.
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:843
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:536
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
Definition ISDOpcodes.h:365
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:213
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:558
LLVM_ABI bool isConstantSplatVector(const SDNode *N, APInt &SplatValue)
Node predicates.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
OperandFlags
These are flags set on operands, but should be considered private, all access should go through the M...
Definition MCInstrDesc.h:51
auto m_Value()
Match an arbitrary value and ignore it.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
bool sd_match(SDNode *N, const SelectionDAG *DAG, Pattern &&P)
CondCode_match m_SpecificCondCode(ISD::CondCode CC)
Match a conditional code SDNode with a specific ISD::CondCode.
CondCode_match m_CondCode()
Match any conditional code SDNode.
TernaryOpc_match< T0_P, T1_P, T2_P, true, false > m_c_SetCC(const T0_P &LHS, const T1_P &RHS, const T2_P &CC)
MCSymbolWasm * getOrCreateFunctionTableSymbol(MCContext &Ctx, const WebAssemblySubtarget *Subtarget)
Returns the __indirect_function_table, for use in call_indirect and in function bitcasts.
bool isWebAssemblyTableType(const Type *Ty)
Return true if the table represents a WebAssembly table type.
MCSymbolWasm * getOrCreateFuncrefCallTableSymbol(MCContext &Ctx, const WebAssemblySubtarget *Subtarget)
Returns the __funcref_call_table, for use in funcref calls when lowered to table.set + call_indirect.
bool isValidAddressSpace(unsigned AS)
FastISel * createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo, const LibcallLoweringInfo *libcallLowering)
bool canLowerReturn(size_t ResultSize, const WebAssemblySubtarget *Subtarget)
Returns true if the function's return value(s) can be lowered directly, i.e., not indirectly via a po...
MachineSDNode * getTLSBase(SelectionDAG &DAG, const SDLoc &DL, const WebAssemblySubtarget *Subtarget, const SDValue Chain=SDValue())
bool isWasmVarAddressSpace(unsigned AS)
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:315
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:345
@ Offset
Definition DWP.cpp:578
void computeSignatureVTs(const FunctionType *Ty, const Function *TargetFunc, const Function &ContextFunc, const TargetMachine &TM, SmallVectorImpl< MVT > &Params, SmallVectorImpl< MVT > &Results)
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1669
SDValue peekThroughFreeze(SDValue V)
Return the non-frozen source operand of V if it exists.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ Add
Sum of integers.
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
Definition STLExtras.h:2088
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1772
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2192
void computeLegalValueVTs(const WebAssemblyTargetLowering &TLI, LLVMContext &Ctx, const DataLayout &DL, Type *Ty, SmallVectorImpl< MVT > &ValueVTs)
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
Definition MathExtras.h:374
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
EVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
Definition ValueTypes.h:90
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
bool isByteSized() const
Return true if the bit size is a multiple of 8.
Definition ValueTypes.h:266
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
EVT changeVectorElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
Definition ValueTypes.h:98
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool is128BitVector() const
Return true if this is a 128-bit vector type.
Definition ValueTypes.h:230
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
EVT widenIntegerVectorElementType(LLVMContext &Context) const
Return a VT for an integer vector type with the size of the elements doubled.
Definition ValueTypes.h:475
bool isFixedLengthVector() const
Definition ValueTypes.h:199
bool isFixedLengthVectorOf(EVT EltVT) const
Return true if this is a fixed length vector with matching element type.
Definition ValueTypes.h:205
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool bitsGE(EVT VT) const
Return true if this has no less bits than VT.
Definition ValueTypes.h:315
bool is256BitVector() const
Return true if this is a 256-bit vector type.
Definition ValueTypes.h:235
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
EVT changeElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a type whose attributes match ourselves with the exception of the element type that i...
Definition ValueTypes.h:121
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
EVT getHalfNumVectorElementsVT(LLVMContext &Context) const
Definition ValueTypes.h:484
Align getNonZeroOrigAlign() const
unsigned getByValSize() const
bool isInConsecutiveRegsLast() const
Align getNonZeroByValAlign() const
Matching combinators.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
These are IR-level optimization flags that may be propagated to SDNodes.
This structure is used to pass arguments to makeLibCall function.