LLVM 24.0.0git
VPlanRecipes.cpp
Go to the documentation of this file.
1//===- VPlanRecipes.cpp - Implementations for VPlan recipes ---------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file contains implementations for different VPlan recipes.
11///
12//===----------------------------------------------------------------------===//
13
15#include "VPlan.h"
16#include "VPlanHelpers.h"
17#include "VPlanPatternMatch.h"
18#include "VPlanUtils.h"
19#include "llvm/ADT/APFloat.h"
20#include "llvm/ADT/STLExtras.h"
23#include "llvm/ADT/Twine.h"
29#include "llvm/IR/BasicBlock.h"
30#include "llvm/IR/IRBuilder.h"
31#include "llvm/IR/Instruction.h"
33#include "llvm/IR/Intrinsics.h"
35#include "llvm/IR/Type.h"
36#include "llvm/IR/Value.h"
39#include "llvm/Support/Debug.h"
43#include <cassert>
44
45using namespace llvm;
46using namespace llvm::VPlanPatternMatch;
47
48#define LV_NAME "loop-vectorize"
49#define DEBUG_TYPE LV_NAME
50
51namespace llvm {
53} // namespace llvm
54
56 switch (getVPRecipeID()) {
57 case VPExpressionSC:
58 return cast<VPExpressionRecipe>(this)->mayReadOrWriteMemory();
59 case VPInstructionSC: {
60 auto *VPI = cast<VPInstruction>(this);
61 // Loads read from memory but don't write to memory.
62 if (VPI->getOpcode() == Instruction::Load)
63 return false;
64 return VPI->opcodeMayReadOrWriteFromMemory();
65 }
66 case VPInterleaveEVLSC:
67 case VPInterleaveSC:
68 return cast<VPInterleaveBase>(this)->getNumStoreOperands() > 0;
69 case VPWidenStoreEVLSC:
70 case VPWidenStoreSC:
71 return true;
72 case VPReplicateSC:
73 return cast<Instruction>(getVPSingleValue()->getUnderlyingValue())
74 ->mayWriteToMemory();
75 case VPWidenCallSC:
76 return !cast<VPWidenCallRecipe>(this)
77 ->getCalledScalarFunction()
78 ->onlyReadsMemory();
79 case VPWidenMemIntrinsicSC:
80 case VPWidenIntrinsicSC:
81 return cast<VPWidenIntrinsicRecipe>(this)->mayWriteToMemory();
82 case VPActiveLaneMaskPHISC:
83 case VPCurrentIterationPHISC:
84 case VPBranchOnMaskSC:
85 case VPDerivedIVSC:
86 case VPFirstOrderRecurrencePHISC:
87 case VPReductionPHISC:
88 case VPScalarIVStepsSC:
89 case VPPredInstPHISC:
90 case VPExpandSCEVSC:
91 return false;
92 case VPBlendSC:
93 case VPReductionEVLSC:
94 case VPReductionSC:
95 case VPVectorPointerSC:
96 case VPWidenCanonicalIVSC:
97 case VPWidenCastSC:
98 case VPWidenGEPSC:
99 case VPWidenIntOrFpInductionSC:
100 case VPWidenLoadEVLSC:
101 case VPWidenLoadSC:
102 case VPWidenPHISC:
103 case VPWidenPointerInductionSC:
104 case VPWidenSC: {
105 const Instruction *I =
106 dyn_cast_or_null<Instruction>(getVPSingleValue()->getUnderlyingValue());
107 (void)I;
108 assert((!I || !I->mayWriteToMemory()) &&
109 "underlying instruction may write to memory");
110 return false;
111 }
112 default:
113 return true;
114 }
115}
116
118 switch (getVPRecipeID()) {
119 case VPExpressionSC:
120 return cast<VPExpressionRecipe>(this)->mayReadOrWriteMemory();
121 case VPInstructionSC:
122 return cast<VPInstruction>(this)->opcodeMayReadOrWriteFromMemory();
123 case VPWidenLoadEVLSC:
124 case VPWidenLoadSC:
125 return true;
126 case VPReplicateSC:
127 return cast<Instruction>(getVPSingleValue()->getUnderlyingValue())
128 ->mayReadFromMemory();
129 case VPWidenCallSC:
130 return !cast<VPWidenCallRecipe>(this)
131 ->getCalledScalarFunction()
132 ->onlyWritesMemory();
133 case VPWidenMemIntrinsicSC:
134 case VPWidenIntrinsicSC:
135 return cast<VPWidenIntrinsicRecipe>(this)->mayReadFromMemory();
136 case VPBranchOnMaskSC:
137 case VPDerivedIVSC:
138 case VPCurrentIterationPHISC:
139 case VPFirstOrderRecurrencePHISC:
140 case VPReductionPHISC:
141 case VPPredInstPHISC:
142 case VPScalarIVStepsSC:
143 case VPWidenStoreEVLSC:
144 case VPWidenStoreSC:
145 case VPExpandSCEVSC:
146 return false;
147 case VPBlendSC:
148 case VPReductionEVLSC:
149 case VPReductionSC:
150 case VPVectorPointerSC:
151 case VPWidenCanonicalIVSC:
152 case VPWidenCastSC:
153 case VPWidenGEPSC:
154 case VPWidenIntOrFpInductionSC:
155 case VPWidenPHISC:
156 case VPWidenPointerInductionSC:
157 case VPWidenSC: {
158 const Instruction *I =
159 dyn_cast_or_null<Instruction>(getVPSingleValue()->getUnderlyingValue());
160 (void)I;
161 assert((!I || !I->mayReadFromMemory()) &&
162 "underlying instruction may read from memory");
163 return false;
164 }
165 default:
166 // FIXME: Return false if the recipe represents an interleaved store.
167 return true;
168 }
169}
170
172 switch (getVPRecipeID()) {
173 case VPExpressionSC:
174 return cast<VPExpressionRecipe>(this)->mayHaveSideEffects();
175 case VPActiveLaneMaskPHISC:
176 case VPDerivedIVSC:
177 case VPCurrentIterationPHISC:
178 case VPFirstOrderRecurrencePHISC:
179 case VPReductionPHISC:
180 case VPPredInstPHISC:
181 case VPVectorEndPointerSC:
182 case VPExpandSCEVSC:
183 return false;
184 case VPInstructionSC: {
185 auto *VPI = cast<VPInstruction>(this);
186 return mayWriteToMemory() ||
187 VPI->getOpcode() == VPInstruction::BranchOnCount ||
188 VPI->getOpcode() == VPInstruction::BranchOnCond ||
189 VPI->getOpcode() == VPInstruction::BranchOnTwoConds;
190 }
191 case VPWidenCallSC: {
192 Function *Fn = cast<VPWidenCallRecipe>(this)->getCalledScalarFunction();
193 return mayWriteToMemory() || !Fn->doesNotThrow() || !Fn->willReturn();
194 }
195 case VPWidenMemIntrinsicSC:
196 case VPWidenIntrinsicSC:
197 return cast<VPWidenIntrinsicRecipe>(this)->mayHaveSideEffects();
198 case VPBlendSC:
199 case VPReductionEVLSC:
200 case VPReductionSC:
201 case VPScalarIVStepsSC:
202 case VPVectorPointerSC:
203 case VPWidenCanonicalIVSC:
204 case VPWidenCastSC:
205 case VPWidenGEPSC:
206 case VPWidenIntOrFpInductionSC:
207 case VPWidenPHISC:
208 case VPWidenPointerInductionSC:
209 case VPWidenSC: {
210 const Instruction *I =
211 dyn_cast_or_null<Instruction>(getVPSingleValue()->getUnderlyingValue());
212 (void)I;
213 assert((!I || !I->mayHaveSideEffects()) &&
214 "underlying instruction has side-effects");
215 return false;
216 }
217 case VPInterleaveEVLSC:
218 case VPInterleaveSC:
219 return mayWriteToMemory();
220 case VPWidenLoadEVLSC:
221 case VPWidenLoadSC:
222 case VPWidenStoreEVLSC:
223 case VPWidenStoreSC:
224 assert(
225 cast<VPWidenMemoryRecipe>(this)->getIngredient().mayHaveSideEffects() ==
227 "mayHaveSideffects result for ingredient differs from this "
228 "implementation");
229 return mayWriteToMemory();
230 case VPReplicateSC: {
231 auto *R = cast<VPReplicateRecipe>(this);
232 return R->getUnderlyingInstr()->mayHaveSideEffects();
233 }
234 default:
235 return true;
236 }
237}
238
240 switch (getVPRecipeID()) {
241 default:
242 return false;
243 case VPInstructionSC: {
244 unsigned Opcode = cast<VPInstruction>(this)->getOpcode();
245 if (Instruction::isCast(Opcode))
246 return true;
247
248 switch (Opcode) {
249 default:
250 return false;
251 case Instruction::Add:
252 case Instruction::Sub:
253 case Instruction::Mul:
254 case Instruction::GetElementPtr:
255 return true;
256 }
257 }
258 }
259}
260
262 assert(!Parent && "Recipe already in some VPBasicBlock");
263 assert(InsertPos->getParent() &&
264 "Insertion position not in any VPBasicBlock");
265 InsertPos->getParent()->insert(this, InsertPos->getIterator());
266}
267
268void VPRecipeBase::insertBefore(VPBasicBlock &BB,
270 assert(!Parent && "Recipe already in some VPBasicBlock");
271 assert(I == BB.end() || I->getParent() == &BB);
272 BB.insert(this, I);
273}
274
276 assert(!Parent && "Recipe already in some VPBasicBlock");
277 assert(InsertPos->getParent() &&
278 "Insertion position not in any VPBasicBlock");
279 InsertPos->getParent()->insert(this, std::next(InsertPos->getIterator()));
280}
281
283 assert(getParent() && "Recipe not in any VPBasicBlock");
285 Parent = nullptr;
286}
287
289 assert(getParent() && "Recipe not in any VPBasicBlock");
291}
292
295 insertAfter(InsertPos);
296}
297
303
305 // Get the underlying instruction for the recipe, if there is one. It is used
306 // to
307 // * decide if cost computation should be skipped for this recipe,
308 // * apply forced target instruction cost.
309 Instruction *UI = nullptr;
310 if (auto *S = dyn_cast<VPSingleDefRecipe>(this))
311 UI = dyn_cast_or_null<Instruction>(S->getUnderlyingValue());
312 else if (auto *IG = dyn_cast<VPInterleaveBase>(this))
313 UI = IG->getInsertPos();
314 else if (auto *WidenMem = dyn_cast<VPWidenMemoryRecipe>(this))
315 UI = &WidenMem->getIngredient();
316
317 InstructionCost RecipeCost;
318 if (UI && Ctx.skipCostComputation(UI, VF.isVector())) {
319 RecipeCost = 0;
320 } else {
321 RecipeCost = computeCost(VF, Ctx);
322 if (ForceTargetInstructionCost.getNumOccurrences() > 0 &&
323 RecipeCost.isValid()) {
324 // VPDerivedIVRecipe and VPScalarIVStepsRecipe never have underlying
325 // instructions.
328 else
329 RecipeCost = InstructionCost(0);
330 }
331 }
332
333 LLVM_DEBUG({
334 dbgs() << "Cost of " << RecipeCost << " for VF " << VF << ": ";
335 if (VPSlotTracker *SlotTracker = Ctx.getSlotTracker()) {
336 print(dbgs(), "", *SlotTracker);
337 dbgs() << "\n";
338 } else {
339 dump();
340 }
341 });
342 return RecipeCost;
343}
344
346 VPCostContext &Ctx) const {
347 llvm_unreachable("subclasses should implement computeCost");
348}
349
351 return (getVPRecipeID() >= VPFirstPHISC && getVPRecipeID() <= VPLastPHISC) ||
353}
354
356 assert(OpType == Other.OpType && "OpType must match");
357 switch (OpType) {
358 case OperationType::OverflowingBinOp:
359 WrapFlags.HasNUW &= Other.WrapFlags.HasNUW;
360 WrapFlags.HasNSW &= Other.WrapFlags.HasNSW;
361 break;
362 case OperationType::Trunc:
363 TruncFlags.HasNUW &= Other.TruncFlags.HasNUW;
364 TruncFlags.HasNSW &= Other.TruncFlags.HasNSW;
365 break;
366 case OperationType::DisjointOp:
367 DisjointFlags.IsDisjoint &= Other.DisjointFlags.IsDisjoint;
368 break;
369 case OperationType::PossiblyExactOp:
370 ExactFlags.IsExact &= Other.ExactFlags.IsExact;
371 break;
372 case OperationType::GEPOp:
373 GEPFlagsStorage &= Other.GEPFlagsStorage;
374 break;
375 case OperationType::FPMathOp:
376 case OperationType::FCmp:
377 assert((OpType != OperationType::FCmp ||
378 FCmpFlags.CmpPredStorage == Other.FCmpFlags.CmpPredStorage) &&
379 "Cannot drop CmpPredicate");
380 getFMFsRef() = getFastMathFlagsOrNone() & Other.getFastMathFlagsOrNone();
381 break;
382 case OperationType::NonNegOp:
383 NonNegFlags.NonNeg &= Other.NonNegFlags.NonNeg;
384 break;
385 case OperationType::Cmp:
386 assert(CmpPredStorage == Other.CmpPredStorage &&
387 "Cannot drop CmpPredicate");
388 break;
389 case OperationType::ReductionOp:
390 assert(ReductionFlags.Kind == Other.ReductionFlags.Kind &&
391 "Cannot change RecurKind");
392 assert(ReductionFlags.IsOrdered == Other.ReductionFlags.IsOrdered &&
393 "Cannot change IsOrdered");
394 assert(ReductionFlags.IsInLoop == Other.ReductionFlags.IsInLoop &&
395 "Cannot change IsInLoop");
396 getFMFsRef() = getFastMathFlagsOrNone() & Other.getFastMathFlagsOrNone();
397 break;
398 case OperationType::Other:
399 break;
400 }
401}
402
404 if (!hasFastMathFlags())
405 return {};
406 const FastMathFlagsTy &F = getFMFsRef();
407 FastMathFlags Res;
408 Res.setAllowReassoc(F.AllowReassoc);
409 Res.setNoNaNs(F.NoNaNs);
410 Res.setNoInfs(F.NoInfs);
411 Res.setNoSignedZeros(F.NoSignedZeros);
412 Res.setAllowReciprocal(F.AllowReciprocal);
413 Res.setAllowContract(F.AllowContract);
414 Res.setApproxFunc(F.ApproxFunc);
415 return Res;
416}
417
418#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
420
421void VPRecipeBase::print(raw_ostream &O, const Twine &Indent,
422 VPSlotTracker &SlotTracker) const {
423 printRecipe(O, Indent, SlotTracker);
424 if (auto DL = getDebugLoc()) {
425 O << ", !dbg ";
426 DL.print(O);
427 }
428
429 if (auto *Metadata = dyn_cast<VPIRMetadata>(this))
431}
432#endif
433
435 : VPSingleDefRecipe(VPRecipeBase::VPExpandSCEVSC, {}, Expr->getType()),
436 Expr(Expr) {}
437
438/// For call VPInstruction operands, return the operand index of the called
439/// function. The function is either the last operand (for unmasked calls) or
440/// the second-to-last operand (for masked calls).
442 unsigned NumOps = Operands.size();
443 auto *LastOp = dyn_cast<VPIRValue>(Operands[NumOps - 1]);
444 if (LastOp && isa<Function>(LastOp->getValue()))
445 return NumOps - 1;
447 "expected function operand");
448 return NumOps - 2;
449}
450
451/// For call VPInstruction operands, return the called function.
456
459 assert(!Operands.empty() &&
460 "zero-operand VPInstruction opcodes must pass explicit ResultTy");
461 // Assert operand \p Idx (if present and typed) has type \p ExpectedTy.
462 [[maybe_unused]] auto AssertOperandType = [&Operands](unsigned Idx,
463 Type *ExpectedTy) {
464 if (!ExpectedTy || Operands.size() <= Idx)
465 return;
466 [[maybe_unused]] Type *OpTy = Operands[Idx]->getScalarType();
467 assert((!OpTy || OpTy == ExpectedTy) &&
468 "different types inferred for different operands");
469 };
470
471 Type *Op0Ty = Operands[0]->getScalarType();
472 LLVMContext &Ctx = Op0Ty->getContext();
473 switch (Opcode) {
475 assert(Op0Ty->isIntegerTy(1) && "expected bool condition");
476 return Type::getVoidTy(Ctx);
478 assert(Op0Ty->isIntegerTy(1) && "expected bool condition");
479 AssertOperandType(1, IntegerType::get(Ctx, 1));
480 return Type::getVoidTy(Ctx);
482 assert(Op0Ty->isIntegerTy() && "expected integer operand");
483 AssertOperandType(1, Op0Ty);
484 return Type::getVoidTy(Ctx);
486 assert(Op0Ty->isIntegerTy() && "expected integer operand");
487 for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
488 AssertOperandType(Idx, Op0Ty);
489 return Op0Ty;
490 case Instruction::Switch:
491 for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
492 AssertOperandType(Idx, Op0Ty);
493 return Type::getVoidTy(Ctx);
494 case Instruction::Store:
495 return Type::getVoidTy(Ctx);
496 case Instruction::ICmp:
497 assert(Op0Ty->isIntOrPtrTy() && "expected integer or pointer operand");
498 AssertOperandType(1, Op0Ty);
499 return IntegerType::get(Ctx, 1);
500 case Instruction::FCmp:
501 assert(Op0Ty->isFloatingPointTy() && "expected floating-point operand");
502 AssertOperandType(1, Op0Ty);
503 return IntegerType::get(Ctx, 1);
506 assert(Op0Ty->isIntegerTy() && "expected integer operand");
507 AssertOperandType(1, Op0Ty);
508 return IntegerType::get(Ctx, 1);
510 assert(Op0Ty->isIntegerTy(1) && "expected bool operand");
511 return IntegerType::get(Ctx, 1);
514 assert(Op0Ty->isIntegerTy(1) && "expected bool operand");
515 AssertOperandType(1, Op0Ty);
516 return IntegerType::get(Ctx, 1);
518 assert(Op0Ty->isIntegerTy(1) && "expected bool operand");
519 for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
520 AssertOperandType(Idx, Op0Ty);
521 return IntegerType::get(Ctx, 1);
523 assert(Op0Ty->isIntegerTy() && "expected integer operand");
524 return IntegerType::get(Ctx, 32);
525 case Instruction::Select: {
526 assert((!Op0Ty || Op0Ty->isIntegerTy(1)) &&
527 "select condition must be bool");
528 Type *Op1Ty = Operands[1]->getScalarType();
529 AssertOperandType(2, Op1Ty);
530 return Op1Ty;
531 }
532 case Instruction::InsertElement:
533 // The inserted scalar (operand 1) must match the vector element type;
534 // operand 2 must be an integer.
535 AssertOperandType(1, Op0Ty);
536 assert(Operands[2]->getScalarType()->isIntegerTy() &&
537 "expected integer operand");
538 return Op0Ty;
540 // The start value and the identity value (operands 0 and 1) fill the same
541 // vector and must match in type; operand 2 is the scaling factor.
542 AssertOperandType(1, Op0Ty);
543 return Op0Ty;
545 assert(Operands.size() >= 2 && "ExtractLane requires a lane operand and "
546 "at least one source vector operand");
547 // Operand 0 is the lane index, used for integer arithmetic.
548 assert(Op0Ty->isIntegerTy() && "expected integer operand");
549 Type *Op1Ty = Operands[1]->getScalarType();
550 for (unsigned Idx = 2; Idx != Operands.size(); ++Idx)
551 AssertOperandType(Idx, Op1Ty);
552 return Op1Ty;
553 }
556 assert(Operands[0]->getScalarType()->isPointerTy() &&
557 "expected pointer operand");
558 assert(Operands[1]->getScalarType()->isIntegerTy() &&
559 "expected integer operand");
560 return Op0Ty;
561 case Instruction::ExtractValue: {
562 assert(Operands.size() == 2 && "expected single level extractvalue");
563 auto *StructTy = cast<StructType>(Op0Ty);
564 return StructTy->getTypeAtIndex(
565 cast<VPConstantInt>(Operands[1])->getZExtValue());
566 }
571 case Instruction::Load:
572 case Instruction::Alloca:
573 llvm_unreachable("type must be passed explicitly");
574 case Instruction::Call:
576 default:
577 if (Instruction::isCast(Opcode))
578 llvm_unreachable("type must be passed explicitly");
579 break;
580 }
581
582 // Opcodes that require all operands to share the same scalar type as the
583 // result.
584 bool AllOperandsSameType =
585 Instruction::isBinaryOp(Opcode) ||
589 Opcode);
590 if (AllOperandsSameType)
591 for (unsigned Idx = 1; Idx != Operands.size(); ++Idx)
592 AssertOperandType(Idx, Op0Ty);
593
594 return Op0Ty;
595}
596
599 unsigned Opcode = I->getOpcode();
600 if (Instruction::isCast(Opcode) ||
601 is_contained(ArrayRef<unsigned>({Instruction::ExtractValue,
602 Instruction::Load, Instruction::Alloca}),
603 Opcode))
604 return I->getType();
606}
607
609 const VPIRFlags &Flags, const VPIRMetadata &MD,
610 DebugLoc DL, const Twine &Name, Type *ResultTy)
612 VPRecipeBase::VPInstructionSC, Operands,
613 ResultTy ? ResultTy
615 Flags, DL),
616 VPIRMetadata(MD), Opcode(Opcode), Name(Name.str()) {
618 "Set flags not supported for the provided opcode");
620 "Opcode requires specific flags to be set");
624 "number of operands does not match opcode");
625}
626
628 if (Instruction::isUnaryOp(Opcode) || Instruction::isCast(Opcode))
629 return 1;
630
631 if (Instruction::isBinaryOp(Opcode))
632 return 2;
633
634 switch (Opcode) {
637 return 0;
638 case Instruction::Alloca:
639 case Instruction::ExtractValue:
640 case Instruction::Freeze:
641 case Instruction::Load:
654 return 1;
655 case Instruction::ICmp:
656 case Instruction::FCmp:
657 case Instruction::ExtractElement:
658 case Instruction::Store:
670 return 2;
671 case Instruction::InsertElement:
672 case Instruction::Select:
675 return 3;
676 case Instruction::Call:
677 return getCalledFnOperandIndex(operands()) + 1;
678 case Instruction::GetElementPtr:
679 case Instruction::PHI:
680 case Instruction::Switch:
681 case Instruction::AtomicRMW:
682 case Instruction::AtomicCmpXchg:
683 case Instruction::Fence:
694 // Cannot determine the number of operands from the opcode.
695 return -1u;
696 }
697 llvm_unreachable("all cases should be handled above");
698}
699
701 return Opcode == VPInstruction::Unpack ||
703}
704
705bool VPInstruction::doesGenerateSingleScalar() const {
707 return true;
708 switch (Opcode) {
709 case Instruction::Freeze:
710 case Instruction::ICmp:
711 case Instruction::PHI:
712 case Instruction::Select:
721 return vputils::onlyFirstLaneUsed(this);
722 default:
724 }
725}
726
728 if (Kind == RecurKind::Sub)
729 return Instruction::Add;
730 if (Kind == RecurKind::FSub)
731 return Instruction::FAdd;
732 llvm_unreachable("RecurKind should be Sub/FSub.");
733}
734
735Value *VPInstruction::generate(VPTransformState &State,
736 bool GenerateSingleScalar) {
737 IRBuilderBase &Builder = State.Builder;
738
740 Value *A = State.get(getOperand(0), GenerateSingleScalar);
741 Value *B = State.get(getOperand(1), GenerateSingleScalar);
742 auto *Res =
743 Builder.CreateBinOp((Instruction::BinaryOps)getOpcode(), A, B, Name);
744 if (auto *I = dyn_cast<Instruction>(Res))
745 applyFlags(*I);
746 return Res;
747 }
749 Value *Op = State.get(getOperand(0), VPLane(0));
751 getScalarType());
752 if (auto *CastOp = dyn_cast<Instruction>(Res)) {
753 applyFlags(*CastOp);
754 applyMetadata(*CastOp);
755 }
756 return Res;
757 }
758
759 switch (getOpcode()) {
760 case VPInstruction::Not: {
761 Value *A = State.get(getOperand(0), GenerateSingleScalar);
762 return Builder.CreateNot(A, Name);
763 }
765 // TODO: Use IsSingleScalar to produce a scalar value.
766 Value *A = State.get(getOperand(0));
767 Value *B = State.get(getOperand(1));
768 return Builder.CreateLogicalAnd(A, B, Name);
769 }
771 // TODO: Use IsSingleScalar to produce a scalar value.
772 Value *A = State.get(getOperand(0));
773 Value *B = State.get(getOperand(1));
774 return Builder.CreateLogicalOr(A, B, Name);
775 }
776 case Instruction::ExtractElement: {
777 assert(GenerateSingleScalar &&
778 "Can only generate first lane for ExtractElement");
779 assert(State.VF.isVector() && "Only extract elements from vectors");
780 if (auto *Idx = dyn_cast<VPConstantInt>(getOperand(1)))
781 return State.get(getOperand(0), VPLane(Idx->getZExtValue()));
782 Value *Vec = State.get(getOperand(0));
783 Value *Idx = State.get(getOperand(1), /*NeedsSingleScalar=*/true);
784 return Builder.CreateExtractElement(Vec, Idx, Name);
785 }
786 case Instruction::InsertElement: {
787 assert(!GenerateSingleScalar &&
788 "Cannot generate scalar value for InsertElement");
789 assert(State.VF.isVector() && "Can only insert elements into vectors");
790 Value *Vec = State.get(getOperand(0), /*NeedsSingleScalar=*/false);
791 Value *Elt = State.get(getOperand(1), /*NeedsSingleScalar=*/true);
792 Value *Idx = State.get(getOperand(2), /*NeedsSingleScalar=*/true);
793 return Builder.CreateInsertElement(Vec, Elt, Idx, Name);
794 }
795 case Instruction::Freeze: {
796 Value *Op = State.get(getOperand(0), GenerateSingleScalar);
797 return Builder.CreateFreeze(Op, Name);
798 }
799 case Instruction::FCmp:
800 case Instruction::ICmp: {
801 Value *A = State.get(getOperand(0), GenerateSingleScalar);
802 Value *B = State.get(getOperand(1), GenerateSingleScalar);
803 return Builder.CreateCmp(getPredicate(), A, B, Name);
804 }
805 case Instruction::PHI: {
806 llvm_unreachable("should be handled by VPPhi::execute");
807 }
808 case Instruction::Select: {
809 Value *Cond =
810 State.get(getOperand(0), GenerateSingleScalar ||
812 Value *Op1 = State.get(getOperand(1), GenerateSingleScalar);
813 Value *Op2 = State.get(getOperand(2), GenerateSingleScalar);
814 return Builder.CreateSelectFMF(Cond, Op1, Op2, getFastMathFlagsOrNone(),
815 Name);
816 }
819 // Can produce either a scalar value as a icmp of a phi, or a vector value,
820 // as a get.active.lane.mask intrinsic.
821 // Get first lane of vector induction variable.
822 Value *VIVElem0 = State.get(getOperand(0), VPLane(0));
823 // Get the original loop tripcount.
824 Value *ScalarTC = State.get(getOperand(1), VPLane(0));
825
826 uint64_t Multiplier =
828 ? cast<VPConstantInt>(getOperand(2))->getZExtValue()
829 : 1;
830
831 // If this part of the active lane mask is scalar, generate the CMP directly
832 // to avoid unnecessary extracts.
833 if (State.VF.isScalar() && Multiplier == 1)
834 return Builder.CreateCmp(CmpInst::Predicate::ICMP_ULT, VIVElem0, ScalarTC,
835 Name);
836
837 auto *PredTy = VectorType::get(Builder.getInt1Ty(), State.VF * Multiplier);
838 return Builder.CreateIntrinsic(Intrinsic::get_active_lane_mask,
839 {PredTy, ScalarTC->getType()},
840 {VIVElem0, ScalarTC}, nullptr, Name);
841 }
843 assert(GenerateSingleScalar &&
844 "Can only generate first lane for NumActiveLanes");
845 Value *Op = State.get(getOperand(0));
846 auto *VecTy = cast<VectorType>(Op->getType());
847 assert(VecTy->getScalarSizeInBits() == 1 &&
848 "NumActiveLanes only implemented for i1 vectors");
849
850 Type *Ty = getScalarType();
851 Value *ZExt = Builder.CreateCast(
852 Instruction::ZExt, Op, VectorType::get(Ty, VecTy->getElementCount()));
853 Value *NumActive =
854 Builder.CreateUnaryIntrinsic(Intrinsic::vector_reduce_add, ZExt);
855 return NumActive;
856 }
858 // Generate code to combine the previous and current values in vector v3.
859 //
860 // vector.ph:
861 // v_init = vector(..., ..., ..., a[-1])
862 // br vector.body
863 //
864 // vector.body
865 // i = phi [0, vector.ph], [i+4, vector.body]
866 // v1 = phi [v_init, vector.ph], [v2, vector.body]
867 // v2 = a[i, i+1, i+2, i+3];
868 // v3 = vector(v1(3), v2(0, 1, 2))
869
870 auto *V1 = State.get(getOperand(0));
871 if (!V1->getType()->isVectorTy())
872 return V1;
873 Value *V2 = State.get(getOperand(1));
874 return Builder.CreateVectorSpliceRight(V1, V2, 1, Name);
875 }
877 // TODO: Restructure this code with an explicit remainder loop, vsetvli can
878 // be outside of the main loop.
879 assert(GenerateSingleScalar &&
880 "Can only generate first lane for ExplicitVectorLength");
881 Value *AVL = State.get(getOperand(0), /*NeedsSingleScalar=*/true);
882 // Compute EVL
883 assert(AVL->getType()->isIntegerTy() &&
884 "Requested vector length should be an integer.");
885
886 assert(State.VF.isScalable() && "Expected scalable vector factor.");
887 Value *VFArg = Builder.getInt32(State.VF.getKnownMinValue());
888
889 Value *EVL = Builder.CreateIntrinsic(
890 Builder.getInt32Ty(), Intrinsic::experimental_get_vector_length,
891 {AVL, VFArg, Builder.getTrue()});
892 return EVL;
893 }
895 assert(GenerateSingleScalar &&
896 "Can only generate first lane for BranchOnCond");
897 Value *Cond = State.get(getOperand(0), VPLane(0));
898 // Replace the temporary unreachable terminator with a new conditional
899 // branch, hooking it up to backward destination for latch blocks now, and
900 // to forward destination(s) later when they are created.
901 // Second successor may be backwards - iff it is already in VPBB2IRBB.
902 VPBasicBlock *SecondVPSucc =
903 cast<VPBasicBlock>(getParent()->getSuccessors()[1]);
904 BasicBlock *SecondIRSucc = State.CFG.VPBB2IRBB.lookup(SecondVPSucc);
905 BasicBlock *IRBB = State.CFG.VPBB2IRBB[getParent()];
906 auto *Br = Builder.CreateCondBr(Cond, IRBB, SecondIRSucc);
907 // First successor is always forward, reset it to nullptr.
908 Br->setSuccessor(0, nullptr);
910 applyMetadata(*Br);
911 return Br;
912 }
914 assert(!GenerateSingleScalar &&
915 "Cannot generate scalar value for Broadcast");
916 return Builder.CreateVectorSplat(
917 State.VF, State.get(getOperand(0), /*NeedsSingleScalar=*/true),
918 "broadcast");
919 }
921 assert(!GenerateSingleScalar &&
922 "Cannot generate scalar value for BuildStructVector");
923 // For struct types, we need to build a new 'wide' struct type, where each
924 // element is widened, i.e., we create a struct of vectors.
925 auto *StructTy = cast<StructType>(getOperand(0)->getScalarType());
926 Value *Res = PoisonValue::get(toVectorizedTy(StructTy, State.VF));
927 for (const auto &[LaneIndex, Op] : enumerate(operands())) {
928 for (unsigned FieldIndex = 0; FieldIndex != StructTy->getNumElements();
929 FieldIndex++) {
930 Value *ScalarValue =
931 Builder.CreateExtractValue(State.get(Op, true), FieldIndex);
932 Value *VectorValue = Builder.CreateExtractValue(Res, FieldIndex);
933 VectorValue =
934 Builder.CreateInsertElement(VectorValue, ScalarValue, LaneIndex);
935 Res = Builder.CreateInsertValue(Res, VectorValue, FieldIndex);
936 }
937 }
938 return Res;
939 }
941 assert(!GenerateSingleScalar &&
942 "Cannot generate scalar value for BuildVector");
943 auto *ScalarTy = getOperand(0)->getScalarType();
944 auto NumOfElements = ElementCount::getFixed(getNumOperands());
945 Value *Res = PoisonValue::get(toVectorizedTy(ScalarTy, NumOfElements));
946 for (const auto &[Idx, Op] : enumerate(operands()))
947 Res = Builder.CreateInsertElement(Res, State.get(Op, true),
948 Builder.getInt64(Idx));
949 return Res;
950 }
952 if (State.VF.isScalar())
953 return State.get(getOperand(0), true);
954 IRBuilderBase::FastMathFlagGuard FMFG(Builder);
956 // If this start vector is scaled then it should produce a vector with fewer
957 // elements than the VF.
958 ElementCount VF = State.VF.divideCoefficientBy(
959 cast<VPConstantInt>(getOperand(2))->getZExtValue());
960 auto *Iden = Builder.CreateVectorSplat(VF, State.get(getOperand(1), true));
961 return Builder.CreateInsertElement(Iden, State.get(getOperand(0), true),
962 Builder.getInt64(0));
963 }
965 assert(GenerateSingleScalar &&
966 "Can only generate first lane for ComputeReductionResult");
967 RecurKind RK = getRecurKind();
968 bool IsOrdered = isReductionOrdered();
969 bool IsInLoop = isReductionInLoop();
971 "FindIV should use min/max reduction kinds");
972
973 // The recipe may have multiple operands to be reduced together.
974 unsigned NumOperandsToReduce = getNumOperands();
975 SmallVector<Value *, 2> RdxParts(NumOperandsToReduce);
976 for (unsigned Part = 0; Part < NumOperandsToReduce; ++Part)
977 RdxParts[Part] = State.get(getOperand(Part), IsInLoop);
978
979 IRBuilderBase::FastMathFlagGuard FMFG(Builder);
981
982 // Reduce multiple operands into one.
983 Value *ReducedPartRdx = RdxParts[0];
984 if (IsOrdered) {
985 ReducedPartRdx = RdxParts[NumOperandsToReduce - 1];
986 } else {
987 // Floating-point operations should have some FMF to enable the reduction.
988 for (unsigned Part = 1; Part < NumOperandsToReduce; ++Part) {
989 Value *RdxPart = RdxParts[Part];
991 ReducedPartRdx = createMinMaxOp(Builder, RK, ReducedPartRdx, RdxPart);
992 else {
993 // For sub-recurrences, each part's reduction variable is already
994 // negative, we need to do: reduce.add(-acc_uf0 + -acc_uf1)
998 : (Instruction::BinaryOps)RecurrenceDescriptor::getOpcode(RK);
999 ReducedPartRdx =
1000 Builder.CreateBinOp(Opcode, RdxPart, ReducedPartRdx, "bin.rdx");
1001 }
1002 }
1003 }
1004
1005 // Create the reduction after the loop. Note that inloop reductions create
1006 // the target reduction in the loop using a Reduction recipe.
1007 if (State.VF.isVector() && !IsInLoop) {
1008 // TODO: Support in-order reductions based on the recurrence descriptor.
1009 // All ops in the reduction inherit fast-math-flags from the recurrence
1010 // descriptor.
1011 ReducedPartRdx = createSimpleReduction(Builder, ReducedPartRdx, RK);
1012 }
1013
1014 return ReducedPartRdx;
1015 }
1018 assert(GenerateSingleScalar &&
1019 "Can only generate first lane for ExtractLane and "
1020 "ExtractPenultimateElement");
1021 unsigned Offset =
1023 Value *Res;
1024 if (State.VF.isVector()) {
1025 assert(Offset <= State.VF.getKnownMinValue() &&
1026 "invalid offset to extract from");
1027 // Extract lane VF - Offset from the operand.
1028 Res = State.get(getOperand(0), VPLane::getLaneFromEnd(State.VF, Offset));
1029 } else {
1030 // TODO: Remove ExtractLastLane for scalar VFs.
1031 assert(Offset <= 1 && "invalid offset to extract from");
1032 Res = State.get(getOperand(0));
1033 }
1034 if (isa<ExtractElementInst>(Res))
1035 Res->setName(Name);
1036 return Res;
1037 }
1038 case VPInstruction::PtrAdd: {
1039 assert(GenerateSingleScalar && "Can only generate first lane for PtrAdd");
1040 Value *Ptr = State.get(getOperand(0), VPLane(0));
1041 Value *Addend = State.get(getOperand(1), VPLane(0));
1042 return Builder.CreatePtrAdd(Ptr, Addend, Name, getGEPNoWrapFlags());
1043 }
1045 assert(!GenerateSingleScalar &&
1046 "Cannot generate scalar value for WidePtrAdd");
1047 Value *Ptr =
1049 Value *Addend = State.get(getOperand(1));
1050 return Builder.CreatePtrAdd(Ptr, Addend, Name, getGEPNoWrapFlags());
1051 }
1052 case VPInstruction::AnyOf: {
1053 assert(GenerateSingleScalar && "Can only generate first lane for AnyOf");
1054 Value *Res = State.get(getOperand(0));
1055 for (VPValue *Op : drop_begin(operands()))
1056 Res = Builder.CreateOr(Res, State.get(Op));
1057 return State.VF.isScalar() ? Res : Builder.CreateOrReduce(Res);
1058 }
1060 assert(GenerateSingleScalar &&
1061 "Can only generate first lane for ExtractLane");
1062 assert(getNumOperands() != 2 && "ExtractLane from single source should be "
1063 "simplified to ExtractElement.");
1064 Value *LaneToExtract = State.get(getOperand(0), true);
1065 Type *IdxTy = getOperand(0)->getScalarType();
1066 Value *Res = nullptr;
1067 Value *RuntimeVF = getRuntimeVF(Builder, IdxTy, State.VF);
1068
1069 for (unsigned Idx = 1; Idx != getNumOperands(); ++Idx) {
1070 Value *VectorStart =
1071 Builder.CreateMul(RuntimeVF, ConstantInt::get(IdxTy, Idx - 1));
1072 Value *VectorIdx = Idx == 1
1073 ? LaneToExtract
1074 : Builder.CreateSub(LaneToExtract, VectorStart);
1075 Value *Ext = State.VF.isScalar()
1076 ? State.get(getOperand(Idx))
1077 : Builder.CreateExtractElement(
1078 State.get(getOperand(Idx)), VectorIdx);
1079 if (Res) {
1080 Value *Cmp = Builder.CreateICmpUGE(LaneToExtract, VectorStart);
1081 Res = Builder.CreateSelect(Cmp, Ext, Res);
1082 } else {
1083 Res = Ext;
1084 }
1085 }
1086 return Res;
1087 }
1089 assert(GenerateSingleScalar &&
1090 "Can only generate first lane for FirstActiveLane");
1091 Type *Ty = this->getScalarType();
1092 if (getNumOperands() == 1) {
1093 Value *Mask = State.get(getOperand(0));
1094 return Builder.CreateCountTrailingZeroElems(Ty, Mask,
1095 /*ZeroIsPoison=*/false, Name);
1096 }
1097 // If there are multiple operands, create a chain of selects to pick the
1098 // first operand with an active lane and add the number of lanes of the
1099 // preceding operands.
1100 Value *RuntimeVF = getRuntimeVF(Builder, Ty, State.VF);
1101 unsigned LastOpIdx = getNumOperands() - 1;
1102 Value *Res = nullptr;
1103 for (int Idx = LastOpIdx; Idx >= 0; --Idx) {
1104 Value *TrailingZeros =
1105 State.VF.isScalar()
1106 ? Builder.CreateZExt(
1107 Builder.CreateICmpEQ(State.get(getOperand(Idx)),
1108 Builder.getFalse()),
1109 Ty)
1111 Ty, State.get(getOperand(Idx)),
1112 /*ZeroIsPoison=*/false, Name);
1113 Value *Current = Builder.CreateAdd(
1114 Builder.CreateMul(RuntimeVF, ConstantInt::get(Ty, Idx)),
1115 TrailingZeros);
1116 if (Res) {
1117 Value *Cmp = Builder.CreateICmpNE(TrailingZeros, RuntimeVF);
1118 Res = Builder.CreateSelect(Cmp, Current, Res);
1119 } else {
1120 Res = Current;
1121 }
1122 }
1123
1124 return Res;
1125 }
1127 assert(GenerateSingleScalar &&
1128 "Can only generate first lane for ResumeForEpilogue");
1129 return State.get(getOperand(0), true);
1131 assert(!GenerateSingleScalar && "Cannot generate scalar value for Reverse");
1132 return Builder.CreateVectorReverse(State.get(getOperand(0)), "reverse");
1134 assert(GenerateSingleScalar &&
1135 "Can only generate first lane for ExtractLastActive");
1136 Value *Result = State.get(getOperand(0), /*NeedsSingleScalar=*/true);
1137 for (unsigned Idx = 1; Idx < getNumOperands(); Idx += 2) {
1138 Value *Data = State.get(getOperand(Idx));
1139 Value *Mask = State.get(getOperand(Idx + 1));
1140 Type *VTy = Data->getType();
1141
1142 if (State.VF.isScalar())
1143 Result = Builder.CreateSelect(Mask, Data, Result);
1144 else
1145 Result = Builder.CreateIntrinsic(
1146 Intrinsic::experimental_vector_extract_last_active, {VTy},
1147 {Data, Mask, Result});
1148 }
1149
1150 return Result;
1151 }
1153 assert(!GenerateSingleScalar &&
1154 "Cannot generate scalar value for ExtractVectorForPart");
1155 Value *Src = State.get(getOperand(0));
1156 Type *DstTy = VectorType::get(getScalarType(), State.VF);
1157 uint64_t Part = cast<VPConstantInt>(getOperand(1))->getZExtValue();
1158
1159 if (Src->getType() == DstTy)
1160 return Src;
1161
1162 return Builder.CreateExtractVector(
1163 DstTy, Src, Builder.getInt64(State.VF.getKnownMinValue() * Part), Name);
1164 }
1166 assert(!GenerateSingleScalar &&
1167 "Cannot generate scalar value for StepVector");
1168 return State.Builder.CreateStepVector(
1169 VectorType::get(getScalarType(), State.VF));
1171 assert(GenerateSingleScalar &&
1172 "Can only generate first lane for Intrinsic");
1173 SmallVector<Value *, 2> Args;
1174 for (VPValue *Op : drop_end(operands()))
1175 Args.push_back(State.get(Op, /*NeedsSingleScalar=*/true));
1176 return State.Builder.CreateIntrinsic(getScalarType(),
1177 vputils::getIntrinsicID(this), Args,
1178 /*FMFSource=*/nullptr, getName());
1179 }
1180 default:
1181 llvm_unreachable("Unsupported opcode for instruction");
1182 }
1183}
1184
1186 unsigned Opcode, ElementCount VF, VPCostContext &Ctx) const {
1187 Type *ScalarTy = this->getScalarType();
1188 Type *ResultTy = VF.isVector() ? toVectorTy(ScalarTy, VF) : ScalarTy;
1189 switch (Opcode) {
1190 case Instruction::FNeg:
1191 return Ctx.TTI.getArithmeticInstrCost(Opcode, ResultTy, Ctx.CostKind);
1192 case Instruction::UDiv:
1193 case Instruction::SDiv:
1194 case Instruction::SRem:
1195 case Instruction::URem:
1196 case Instruction::Add:
1197 case Instruction::FAdd:
1198 case Instruction::Sub:
1199 case Instruction::FSub:
1200 case Instruction::Mul:
1201 case Instruction::FMul:
1202 case Instruction::FDiv:
1203 case Instruction::FRem:
1204 case Instruction::Shl:
1205 case Instruction::LShr:
1206 case Instruction::AShr:
1207 case Instruction::And:
1208 case Instruction::Or:
1209 case Instruction::Xor: {
1210 // Certain instructions can be cheaper if they have a constant second
1211 // operand. One example of this are shifts on x86.
1212 VPValue *RHS = getOperand(1);
1213 TargetTransformInfo::OperandValueInfo RHSInfo = Ctx.getOperandInfo(RHS);
1214
1215 if (RHSInfo.Kind == TargetTransformInfo::OK_AnyValue &&
1218
1221 if (CtxI)
1222 Operands.append(CtxI->value_op_begin(), CtxI->value_op_end());
1223 return Ctx.TTI.getArithmeticInstrCost(
1224 Opcode, ResultTy, Ctx.CostKind,
1225 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
1226 RHSInfo, Operands, CtxI, &Ctx.TLI);
1227 }
1228 case Instruction::Freeze:
1229 // NOTE: The only way to ask for the cost is via getInstructionCost, which
1230 // requires the actual vector instruction. Instead, both here and in the
1231 // LoopVectorizationCostModel::getInstructionCost the costs mirror the
1232 // current behaviour in llvm/Analysis/TargetTransformInfoImpl.h to keep
1233 // them in sync.
1234 return TTI::TCC_Free;
1235 case Instruction::ExtractValue:
1236 return Ctx.TTI.getInsertExtractValueCost(Instruction::ExtractValue,
1237 Ctx.CostKind);
1238 case Instruction::ICmp:
1239 case Instruction::FCmp: {
1240 Type *ScalarOpTy = getOperand(0)->getScalarType();
1241 Type *OpTy = VF.isVector() ? toVectorTy(ScalarOpTy, VF) : ScalarOpTy;
1243 return Ctx.TTI.getCmpSelInstrCost(
1245 Ctx.CostKind, {TTI::OK_AnyValue, TTI::OP_None},
1246 {TTI::OK_AnyValue, TTI::OP_None}, CtxI);
1247 }
1248 case Instruction::BitCast: {
1249 Type *ScalarTy = this->getScalarType();
1250 if (ScalarTy->isPointerTy())
1251 return 0;
1252 [[fallthrough]];
1253 }
1254 case Instruction::SExt:
1255 case Instruction::ZExt:
1256 case Instruction::FPToUI:
1257 case Instruction::FPToSI:
1258 case Instruction::FPExt:
1259 case Instruction::PtrToInt:
1260 case Instruction::PtrToAddr:
1261 case Instruction::IntToPtr:
1262 case Instruction::SIToFP:
1263 case Instruction::UIToFP:
1264 case Instruction::Trunc:
1265 case Instruction::FPTrunc:
1266 case Instruction::AddrSpaceCast: {
1267 // Computes the CastContextHint from a recipe that may access memory.
1268 auto ComputeCCH = [&](const VPRecipeBase *R) -> TTI::CastContextHint {
1269 if (isa<VPInterleaveBase>(R))
1271 if (const auto *ReplicateRecipe = dyn_cast<VPReplicateRecipe>(R)) {
1272 // Only compute CCH for memory operations, matching the legacy model
1273 // which only considers loads/stores for cast context hints.
1274 auto *UI = cast<Instruction>(ReplicateRecipe->getUnderlyingValue());
1275 if (!isa<LoadInst, StoreInst>(UI))
1277 return ReplicateRecipe->isPredicated() ? TTI::CastContextHint::Masked
1279 }
1280 const auto *WidenMemoryRecipe = dyn_cast<VPWidenMemoryRecipe>(R);
1281 if (WidenMemoryRecipe == nullptr)
1283 if (VF.isScalar())
1285 if (!WidenMemoryRecipe->isConsecutive())
1287 if (WidenMemoryRecipe->isMasked())
1290 };
1291
1292 VPValue *Operand = getOperand(0);
1294 bool IsReverse = false;
1295 // For Trunc/FPTrunc, get the context from the only user.
1296 if (Opcode == Instruction::Trunc || Opcode == Instruction::FPTrunc) {
1297 if (auto *Recipe = cast_or_null<VPRecipeBase>(getSingleUser())) {
1298 if (match(Recipe,
1302 IsReverse = true;
1304 Recipe->getVPSingleValue()->getSingleUser());
1305 }
1306 if (Recipe)
1307 CCH = ComputeCCH(Recipe);
1308 }
1309 }
1310 // For Z/Sext, get the context from the operand.
1311 else if (Opcode == Instruction::ZExt || Opcode == Instruction::SExt ||
1312 Opcode == Instruction::FPExt) {
1313 if (auto *Recipe = Operand->getDefiningRecipe()) {
1314 VPValue *ReverseOp;
1315 if (match(Recipe,
1316 m_CombineOr(m_Reverse(m_VPValue(ReverseOp)),
1318 m_VPValue(ReverseOp))))) {
1319 Recipe = ReverseOp->getDefiningRecipe();
1320 IsReverse = true;
1321 }
1322 if (Recipe)
1323 CCH = ComputeCCH(Recipe);
1324 }
1325 }
1326 if (IsReverse && CCH != TTI::CastContextHint::None)
1328
1329 auto *ScalarSrcTy = Operand->getScalarType();
1330 Type *SrcTy = VF.isVector() ? toVectorTy(ScalarSrcTy, VF) : ScalarSrcTy;
1331 // Arm TTI will use the underlying instruction to determine the cost.
1332 return Ctx.TTI.getCastInstrCost(
1333 Opcode, ResultTy, SrcTy, CCH, Ctx.CostKind,
1335 }
1336 case Instruction::Select: {
1338 bool IsScalarCond = getOperand(0)->isDefinedOutsideLoopRegions();
1339 Type *ScalarTy = this->getScalarType();
1340
1341 VPValue *Op0, *Op1;
1342 bool IsLogicalAnd =
1343 match(this, m_c_LogicalAnd(m_VPValue(Op0), m_VPValue(Op1)));
1344 bool IsLogicalOr =
1345 match(this, m_c_LogicalOr(m_VPValue(Op0), m_VPValue(Op1)));
1346 // Also match the inverted forms:
1347 // select x, false, y --> !x & y (still AND)
1348 // select x, y, true --> !x | y (still OR)
1349 IsLogicalAnd |=
1350 match(this, m_Select(m_VPValue(Op0), m_False(), m_VPValue(Op1)));
1351 IsLogicalOr |=
1352 match(this, m_Select(m_VPValue(Op0), m_VPValue(Op1), m_True()));
1353
1354 if (!IsScalarCond && ScalarTy->getScalarSizeInBits() == 1 &&
1355 (IsLogicalAnd || IsLogicalOr)) {
1356 // select x, y, false --> x & y
1357 // select x, true, y --> x | y
1358 const auto [Op1VK, Op1VP] = Ctx.getOperandInfo(Op0);
1359 const auto [Op2VK, Op2VP] = Ctx.getOperandInfo(Op1);
1360
1362 if (SI && all_of(operands(),
1363 [](VPValue *Op) { return Op->getUnderlyingValue(); }))
1364 append_range(Operands, SI->operands());
1365 return Ctx.TTI.getArithmeticInstrCost(
1366 IsLogicalOr ? Instruction::Or : Instruction::And, ResultTy,
1367 Ctx.CostKind, {Op1VK, Op1VP}, {Op2VK, Op2VP}, Operands, SI);
1368 }
1369
1370 Type *CondTy = getOperand(0)->getScalarType();
1371 if (!IsScalarCond && VF.isVector())
1372 CondTy = VectorType::get(CondTy, VF);
1373
1374 llvm::CmpPredicate Pred;
1375 if (!match(getOperand(0), m_Cmp(Pred, m_VPValue(), m_VPValue())))
1376 if (auto *CondIRV = dyn_cast<VPIRValue>(getOperand(0)))
1377 if (auto *Cmp = dyn_cast<CmpInst>(CondIRV->getValue()))
1378 Pred = Cmp->getPredicate();
1379 Type *VectorTy = toVectorTy(this->getScalarType(), VF);
1380 return Ctx.TTI.getCmpSelInstrCost(
1381 Instruction::Select, VectorTy, CondTy, Pred, Ctx.CostKind,
1382 {TTI::OK_AnyValue, TTI::OP_None}, {TTI::OK_AnyValue, TTI::OP_None}, SI);
1383 }
1384 }
1385 llvm_unreachable("called for unsupported opcode");
1386}
1387
1389 VPCostContext &Ctx) const {
1390 // NOTE: At the moment it seems only possible to expose this path for
1391 // the trunc, zext and sext opcodes.
1392 // TODO: Update VF arg to use onlyFirstLaneUsed once WidenCast is unified.
1394 // A scalar zext/trunc that only adjusts the width of an
1395 // ExplicitVectorLength to the canonical IV type is free: it feeds only
1396 // the IV increment and AVL decrement, which are modeled as free below.
1397 if (match(this, m_ZExtOrTrunc(m_EVL(m_VPValue()))))
1398 return 0;
1400 Ctx);
1401 }
1402
1404 if (!getUnderlyingValue() && getOpcode() != Instruction::FMul) {
1405 // TODO: Compute cost for VPInstructions without underlying values once
1406 // the legacy cost model has been retired.
1407 return 0;
1408 }
1409
1411 "Should only generate a vector value or single scalar, not scalars "
1412 "for all lanes.");
1414 getOpcode(),
1416 }
1417
1418 switch (getOpcode()) {
1419 case Instruction::Select: {
1421 match(getOperand(0), m_Cmp(Pred, m_VPValue(), m_VPValue()));
1422 auto *CondTy = getOperand(0)->getScalarType();
1423 auto *VecTy = getOperand(1)->getScalarType();
1424 if (!vputils::onlyFirstLaneUsed(this)) {
1425 CondTy = toVectorTy(CondTy, VF);
1426 VecTy = toVectorTy(VecTy, VF);
1427 }
1428 return Ctx.TTI.getCmpSelInstrCost(Instruction::Select, VecTy, CondTy, Pred,
1429 Ctx.CostKind);
1430 }
1431 case Instruction::ExtractElement:
1433 if (VF.isScalar()) {
1434 // ExtractLane with VF=1 takes care of handling extracting across multiple
1435 // parts.
1436 return 0;
1437 }
1438
1439 // Add on the cost of extracting the element.
1440 auto *VecTy = toVectorTy(getOperand(0)->getScalarType(), VF);
1441 return Ctx.TTI.getVectorInstrCost(Instruction::ExtractElement, VecTy,
1442 Ctx.CostKind);
1443 }
1444 case VPInstruction::AnyOf: {
1445 auto *VecTy = toVectorTy(this->getScalarType(), VF);
1446 return Ctx.TTI.getArithmeticReductionCost(
1447 Instruction::Or, cast<VectorType>(VecTy), std::nullopt, Ctx.CostKind);
1448 }
1450 Type *Ty = this->getScalarType();
1451 Type *ScalarTy = getOperand(0)->getScalarType();
1452 if (VF.isScalar())
1453 return Ctx.TTI.getCmpSelInstrCost(Instruction::ICmp, ScalarTy,
1455 CmpInst::ICMP_EQ, Ctx.CostKind);
1456 // Calculate the cost of determining the lane index.
1457 auto *PredTy = toVectorTy(ScalarTy, VF);
1458 IntrinsicCostAttributes Attrs(Intrinsic::experimental_cttz_elts, Ty,
1459 {PredTy, Type::getInt1Ty(Ctx.LLVMCtx)});
1460 return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1461 }
1463 Type *Ty = this->getScalarType();
1464 Type *ScalarTy = getOperand(0)->getScalarType();
1465 if (VF.isScalar())
1466 return Ctx.TTI.getCmpSelInstrCost(Instruction::ICmp, ScalarTy,
1468 CmpInst::ICMP_EQ, Ctx.CostKind);
1469 // Calculate the cost of determining the lane index: NOT + cttz_elts + SUB.
1470 auto *PredTy = toVectorTy(ScalarTy, VF);
1471 IntrinsicCostAttributes Attrs(Intrinsic::experimental_cttz_elts, Ty,
1472 {PredTy, Type::getInt1Ty(Ctx.LLVMCtx)});
1473 InstructionCost Cost = Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1474 // Add cost of NOT operation on the predicate.
1475 Cost += Ctx.TTI.getArithmeticInstrCost(
1476 Instruction::Xor, PredTy, Ctx.CostKind,
1477 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
1478 {TargetTransformInfo::OK_UniformConstantValue,
1479 TargetTransformInfo::OP_None});
1480 // Add cost of SUB operation on the index.
1481 Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Sub, Ty, Ctx.CostKind);
1482 return Cost;
1483 }
1485 Type *ScalarTy = this->getScalarType();
1486 Type *VecTy = toVectorTy(ScalarTy, VF);
1487 Type *MaskTy = toVectorTy(Type::getInt1Ty(Ctx.LLVMCtx), VF);
1489 Intrinsic::experimental_vector_extract_last_active, ScalarTy,
1490 {VecTy, MaskTy, ScalarTy});
1491 return Ctx.TTI.getIntrinsicInstrCost(ICA, Ctx.CostKind);
1492 }
1494 assert(VF.isVector() && "Scalar FirstOrderRecurrenceSplice?");
1495 Type *VectorTy = toVectorTy(this->getScalarType(), VF);
1496 return Ctx.TTI.getShuffleCost(
1498 cast<VectorType>(VectorTy), Ctx.CostKind, {}, -1);
1499 }
1502 Type *ArgTy = getOperand(0)->getScalarType();
1503 uint64_t Multiplier =
1505 ? cast<VPConstantInt>(getOperand(2))->getZExtValue()
1506 : 1;
1507 Type *RetTy = toVectorTy(Type::getInt1Ty(Ctx.LLVMCtx), VF * Multiplier);
1508 IntrinsicCostAttributes Attrs(Intrinsic::get_active_lane_mask, RetTy,
1509 {ArgTy, ArgTy});
1510 return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1511 }
1513 Type *Arg0Ty = getOperand(0)->getScalarType();
1514 Type *I32Ty = Type::getInt32Ty(Ctx.LLVMCtx);
1515 Type *I1Ty = Type::getInt1Ty(Ctx.LLVMCtx);
1516 IntrinsicCostAttributes Attrs(Intrinsic::experimental_get_vector_length,
1517 I32Ty, {Arg0Ty, I32Ty, I1Ty});
1518 return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1519 }
1521 assert(VF.isVector() && "Reverse operation must be vector type");
1522 Type *EltTy = this->getScalarType();
1523 // Skip the reverse operation cost for the mask.
1524 // FIXME: Remove this once redundant mask reverse operations can be
1525 // eliminated by VPlanTransforms::cse before cost computation.
1526 if (EltTy->isIntegerTy(1))
1527 return 0;
1528 auto *VectorTy = cast<VectorType>(toVectorTy(EltTy, VF));
1529 return Ctx.TTI.getShuffleCost(TargetTransformInfo::SK_Reverse, VectorTy,
1530 VectorTy, Ctx.CostKind, /*Mask=*/{},
1531 /*Index=*/0);
1532 }
1534 // Add on the cost of extracting the element.
1535 auto *VecTy = toVectorTy(getOperand(0)->getScalarType(), VF);
1536 return Ctx.TTI.getIndexedVectorInstrCostFromEnd(Instruction::ExtractElement,
1537 VecTy, Ctx.CostKind, 0);
1538 }
1539 case VPInstruction::Not: {
1540 Type *ValTy = this->getScalarType();
1541 // InstCombine will fold `xor` to the conditional branch.
1542 if (auto *U = const_cast<VPUser *>(getSingleUser()))
1543 if (match(U, m_BranchOnCond(m_VPValue())))
1544 return 0;
1545 if (!vputils::onlyFirstLaneUsed(this))
1546 ValTy = toVectorTy(ValTy, VF);
1547 return Ctx.TTI.getArithmeticInstrCost(Instruction::Xor, ValTy,
1548 Ctx.CostKind);
1549 }
1551 // If TC <= VF then this is just a branch.
1552 // FIXME: Removing the branch happens in simplifyBranchConditionForVFAndUF
1553 // where it checks TC <= VF * UF, but we don't know UF yet. This means in
1554 // some cases we get a cost that's too high due to counting a cmp that
1555 // later gets removed.
1556 // FIXME: The compare could also be removed if TC = M * vscale,
1557 // VF = N * vscale, and M <= N. Detecting that would require having the
1558 // trip count as a SCEV though.
1559 if (VPCostContext::executesAtMostOnce(*getParent()->getPlan(), VF))
1560 return 0;
1561 // Otherwise BranchOnCount generates ICmpEQ followed by a branch.
1562 Type *ValTy = getOperand(0)->getScalarType();
1563 return Ctx.TTI.getCmpSelInstrCost(Instruction::ICmp, ValTy,
1565 CmpInst::ICMP_EQ, Ctx.CostKind);
1566 }
1568 Type *Ty = getScalarType();
1570 for (const VPValue *Op : drop_end(operands()))
1571 ArgTys.push_back(Op->getScalarType());
1572 IntrinsicCostAttributes Attrs(vputils::getIntrinsicID(this), Ty, ArgTys);
1573 return Ctx.TTI.getIntrinsicInstrCost(Attrs, Ctx.CostKind);
1574 }
1576 // TODO: This isn't quite right since even if the step-vector is hoisted
1577 // out of the loop it has a non-zero cost in the middle block, etc.
1578 // Once the stepvector is correctly hoisted out of the vector loop by the
1579 // licm transform we can add the cost here so that it doesn't incorrectly
1580 // affect the choice of VF.
1581 return 0;
1583 // It isn't currently possible to expose cases where WideIVStep's cost is
1584 // queried.
1585 llvm_unreachable("Unhandled opcode");
1586 case Instruction::FCmp:
1587 case Instruction::ICmp:
1589 getOpcode(),
1592 if (VF == ElementCount::getScalable(1))
1594 [[fallthrough]];
1595 default:
1596 // TODO: Compute cost other VPInstructions once the legacy cost model has
1597 // been retired.
1599 "unexpected VPInstruction witht underlying value");
1600 return 0;
1601 }
1602}
1603
1616
1618 switch (getOpcode()) {
1619 case Instruction::Load:
1620 case Instruction::PHI:
1624 return true;
1625 default:
1627 }
1628}
1629
1631#ifndef NDEBUG
1632 Type *Ty = Op->getScalarType();
1633 switch (getOpcode()) {
1637 assert(Ty == getOperand(0)->getScalarType() &&
1638 "types of operand 0 and new operand must match");
1639 break;
1643 assert(Ty == getOperand(0)->getScalarType() &&
1644 "appended operand must match operand 0's scalar type");
1645 break;
1647 assert(Ty == getOperand(1)->getScalarType() &&
1648 "appended operand must match operand 1's scalar type");
1649 break;
1651 // The recipe is constructed with 3 operands (result, data, mask). Extra
1652 // operands beyond that are appended in (data, mask) pairs.
1653 constexpr unsigned NumInitialOperands = 3;
1654 assert(getNumOperands() >= NumInitialOperands &&
1655 "ExtractLastActive must have at least the initial 3 operands");
1656 bool IsMaskSlot = ((getNumOperands() - NumInitialOperands) & 1u) == 1u;
1657 assert((IsMaskSlot ? Ty->isIntegerTy(1)
1658 : Ty == getOperand(1)->getScalarType()) &&
1659 "ExtractLastActive expects alternating data/mask operands "
1660 "matching operand 1's type and i1, respectively");
1661 break;
1662 }
1663 default:
1664 llvm_unreachable("opcode does not support growing the operand list "
1665 "outside of construction");
1666 }
1667#endif
1669}
1670
1672 assert(!isMasked() && "cannot execute masked VPInstruction");
1673 IRBuilderBase::FastMathFlagGuard FMFGuard(State.Builder);
1675 "Set flags not supported for the provided opcode");
1677 "Opcode requires specific flags to be set");
1678 State.Builder.setFastMathFlags(getFastMathFlagsOrNone());
1679 bool GenerateSingleScalar = State.VF.isScalar() || doesGenerateSingleScalar();
1680 Value *GeneratedValue = generate(State, GenerateSingleScalar);
1681 if (!hasResult())
1682 return;
1683 assert(GeneratedValue && "generate must produce a value");
1684 assert(((GeneratedValue->getType()->isVectorTy() ||
1685 GeneratedValue->getType()->isStructTy()) == !GenerateSingleScalar) &&
1686 "scalar value but not only first lane defined");
1687 State.set(this, GeneratedValue, GenerateSingleScalar);
1689 getOpcode() == Instruction::Freeze) {
1690 // FIXME: This is a workaround to enable reliable updates of the scalar loop
1691 // resume phis, and to let epilogue vectorization recover the frozen
1692 // reduction start from the main plan. Must be removed once epilogue
1693 // vectorization explicitly connects VPlans.
1694 setUnderlyingValue(GeneratedValue);
1695 }
1696}
1697
1701 return false;
1702 switch (getOpcode()) {
1703 case Instruction::ExtractValue:
1704 case Instruction::InsertValue:
1705 case Instruction::GetElementPtr:
1706 case Instruction::ExtractElement:
1707 case Instruction::InsertElement:
1708 case Instruction::Freeze:
1709 case Instruction::FCmp:
1710 case Instruction::ICmp:
1711 case Instruction::Select:
1712 case Instruction::PHI:
1739 case VPInstruction::Not:
1747 return false;
1750 AttributeSet Attrs =
1752 return !Attrs.getMemoryEffects().doesNotAccessMemory();
1753 }
1754 case Instruction::Call:
1756 default:
1757 return true;
1758 }
1759}
1760
1762 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
1764 return vputils::onlyFirstLaneUsed(this);
1765
1766 switch (getOpcode()) {
1767 default:
1768 return false;
1769 case Instruction::ExtractElement:
1770 return Op == getOperand(1);
1771 case Instruction::InsertElement:
1772 return Op == getOperand(1) || Op == getOperand(2);
1774 return Op == getOperand(0);
1775 case Instruction::PHI:
1776 return true;
1777 case Instruction::FCmp:
1778 case Instruction::ICmp:
1779 case Instruction::Select:
1780 case Instruction::Or:
1781 case Instruction::Freeze:
1782 case VPInstruction::Not:
1783 // TODO: Cover additional opcodes.
1784 return vputils::onlyFirstLaneUsed(this);
1785 case Instruction::Load:
1797 return true;
1800 // Before replicating by VF, Build(Struct)Vector uses all lanes of the
1801 // operand, after replicating its operands only the first lane is used.
1802 // Before replicating, it will have only a single operand.
1803 return getNumOperands() > 1;
1805 return Op == getOperand(0) || vputils::onlyFirstLaneUsed(this);
1807 // WidePtrAdd supports scalar and vector base addresses.
1808 return false;
1811 return Op == getOperand(0);
1812 };
1813 llvm_unreachable("switch should return");
1814}
1815
1817 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
1819 return vputils::onlyFirstPartUsed(this);
1820
1821 switch (getOpcode()) {
1822 default:
1823 return false;
1824 case Instruction::FCmp:
1825 case Instruction::ICmp:
1826 case Instruction::Select:
1827 return vputils::onlyFirstPartUsed(this);
1832 return true;
1833 };
1834 llvm_unreachable("switch should return");
1835}
1836
1837#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1839 VPSlotTracker SlotTracker(getParent()->getPlan());
1841}
1842
1844 VPSlotTracker &SlotTracker) const {
1845 O << Indent << "EMIT" << (isSingleScalar() ? "-SCALAR" : "") << " ";
1846
1847 if (hasResult()) {
1849 O << " = ";
1850 }
1851
1852 switch (getOpcode()) {
1853 case VPInstruction::Not:
1854 O << "not";
1855 break;
1857 O << "active lane mask";
1858 break;
1860 O << "wide active lane mask";
1861 break;
1863 O << "incoming-alias-mask";
1864 break;
1866 O << "EXPLICIT-VECTOR-LENGTH";
1867 break;
1869 O << "first-order splice";
1870 break;
1872 O << "branch-on-cond";
1873 break;
1875 O << "branch-on-two-conds";
1876 break;
1878 O << "VF * Part +";
1879 break;
1881 O << "branch-on-count";
1882 break;
1884 O << "broadcast";
1885 break;
1887 O << "buildstructvector";
1888 break;
1890 O << "buildvector";
1891 break;
1893 O << "exiting-iv-value";
1894 break;
1896 O << "masked-cond";
1897 break;
1899 O << "extract-lane";
1900 break;
1902 O << "extract-last-lane";
1903 break;
1905 O << "extract-last-part";
1906 break;
1908 O << "extract-penultimate-element";
1909 break;
1911 O << "extract-vector-for-part";
1912 break;
1914 O << "compute-reduction-result";
1915 break;
1917 O << "logical-and";
1918 break;
1920 O << "logical-or";
1921 break;
1923 O << "ptradd";
1924 break;
1926 O << "wide-ptradd";
1927 break;
1929 O << "any-of";
1930 break;
1932 O << "first-active-lane";
1933 break;
1935 O << "last-active-lane";
1936 break;
1938 O << "reduction-start-vector";
1939 break;
1941 O << "resume-for-epilogue";
1942 break;
1944 O << "reverse";
1945 break;
1947 O << "unpack";
1948 break;
1950 O << "extract-last-active";
1951 break;
1953 O << "num-active-lanes";
1954 break;
1956 O << "wide-iv-step";
1957 break;
1959 O << "step-vector " << *getScalarType();
1960 break;
1962 O << "call " << *getScalarType() << " @"
1965 Op->printAsOperand(O, SlotTracker);
1966 });
1967 O << ")";
1968 return;
1969 }
1970 case Instruction::Load:
1971 O << "load";
1972 break;
1973 default:
1975 }
1976
1977 if (!operands_empty()) {
1978 printFlags(O);
1980 }
1982 O << " to " << *getScalarType();
1983}
1984#endif
1985
1986/// Shared execute logic for VPPhi and VPWidenPHIRecipe. Creates a PHI node,
1987/// adds incoming values, and stores the result in State. For header phis, only
1988/// the preheader incoming value is added; the backedge is fixed up later by
1989/// VPlan::execute().
1991 VPTransformState &State, bool IsScalar,
1992 const Twine &Name) {
1993 unsigned NumIncoming = VPBlockUtils::isHeader(R->getParent(), State.VPDT)
1994 ? 1
1995 : Phi.getNumIncoming();
1996 Value *FirstInc = State.get(Phi.getIncomingValue(0), IsScalar);
1997 PHINode *NewPhi = State.Builder.CreatePHI(FirstInc->getType(), 2, Name);
1998 NewPhi->addIncoming(FirstInc,
1999 State.CFG.VPBB2IRBB.at(Phi.getIncomingBlock(0)));
2000 for (unsigned Idx = 1; Idx != NumIncoming; ++Idx)
2001 NewPhi->addIncoming(State.get(Phi.getIncomingValue(Idx), IsScalar),
2002 State.CFG.VPBB2IRBB.at(Phi.getIncomingBlock(Idx)));
2003 State.set(R, NewPhi, IsScalar);
2004}
2005
2007 executePhiRecipe(this, *this, State, /*IsScalar=*/true, getName());
2008}
2009
2010#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2011void VPPhi::printRecipe(raw_ostream &O, const Twine &Indent,
2012 VPSlotTracker &SlotTracker) const {
2013 O << Indent << "EMIT" << (isSingleScalar() ? "-SCALAR" : "") << " ";
2015 O << " = phi";
2016 printFlags(O);
2018}
2019#endif
2020
2021VPIRInstruction *VPIRInstruction ::create(Instruction &I) {
2022 if (auto *Phi = dyn_cast<PHINode>(&I))
2023 return new VPIRPhi(*Phi);
2024 return new VPIRInstruction(I);
2025}
2026
2028 assert(!isa<VPIRPhi>(this) && getNumOperands() == 0 &&
2029 "PHINodes must be handled by VPIRPhi");
2030 // Advance the insert point after the wrapped IR instruction. This allows
2031 // interleaving VPIRInstructions and other recipes.
2032 State.Builder.SetInsertPoint(I.getParent(), std::next(I.getIterator()));
2033}
2034
2036 VPCostContext &Ctx) const {
2037 // The recipe wraps an existing IR instruction on the border of VPlan's scope,
2038 // hence it does not contribute to the cost-modeling for the VPlan.
2039 return 0;
2040}
2041
2042#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2044 VPSlotTracker &SlotTracker) const {
2045 O << Indent << "IR " << I;
2046}
2047#endif
2048
2050 PHINode *Phi = &getIRPhi();
2051 for (const auto &[Idx, Op] : enumerate(operands())) {
2052 VPValue *ExitValue = Op;
2053 auto Lane = vputils::isSingleScalar(ExitValue)
2055 : VPLane::getLastLaneForVF(State.VF);
2056 VPBlockBase *Pred = getParent()->getPredecessors()[Idx];
2057 auto *PredVPBB = Pred->getExitingBasicBlock();
2058 BasicBlock *PredBB = State.CFG.VPBB2IRBB[PredVPBB];
2059 // Set insertion point in PredBB in case an extract needs to be generated.
2060 // TODO: Model extracts explicitly.
2061 State.Builder.SetInsertPoint(PredBB->getTerminator());
2062 Value *V = State.get(ExitValue, VPLane(Lane));
2063 // If there is no existing block for PredBB in the phi, add a new incoming
2064 // value. Otherwise update the existing incoming value for PredBB.
2065 if (Phi->getBasicBlockIndex(PredBB) == -1)
2066 Phi->addIncoming(V, PredBB);
2067 else
2068 Phi->setIncomingValueForBlock(PredBB, V);
2069 }
2070
2071 // Advance the insert point after the wrapped IR instruction. This allows
2072 // interleaving VPIRInstructions and other recipes.
2073 State.Builder.SetInsertPoint(Phi->getParent(), std::next(Phi->getIterator()));
2074}
2075
2077 VPRecipeBase *R = const_cast<VPRecipeBase *>(getAsRecipe());
2078 assert(R->getNumOperands() == R->getParent()->getNumPredecessors() &&
2079 "Number of phi operands must match number of predecessors");
2080 unsigned Position = R->getParent()->getIndexForPredecessor(IncomingBlock);
2081 R->removeOperand(Position);
2082}
2083
2084VPValue *
2086 VPRecipeBase *R = const_cast<VPRecipeBase *>(getAsRecipe());
2087 return getIncomingValue(R->getParent()->getIndexForPredecessor(VPBB));
2088}
2089
2091 VPValue *V) const {
2092 VPRecipeBase *R = const_cast<VPRecipeBase *>(getAsRecipe());
2093 R->setOperand(R->getParent()->getIndexForPredecessor(VPBB), V);
2094}
2095
2096#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2098 VPSlotTracker &SlotTracker) const {
2100 O << "[ ";
2101 std::get<0>(Op)->printAsOperand(O, SlotTracker);
2102 O << ", ";
2103 std::get<1>(Op)->printAsOperand(O);
2104 O << " ]";
2105 });
2106}
2107#endif
2108
2109#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2111 VPSlotTracker &SlotTracker) const {
2113
2114 if (getNumOperands() != 0) {
2115 O << " (extra operand" << (getNumOperands() > 1 ? "s" : "") << ": ";
2117 [&O, &SlotTracker](auto Op) {
2118 std::get<0>(Op)->printAsOperand(O, SlotTracker);
2119 O << " from ";
2120 std::get<1>(Op)->printAsOperand(O);
2121 });
2122 O << ")";
2123 }
2124}
2125#endif
2126
2128 if (Metadata.empty())
2129 return;
2130 // Frequencies and estimated branch weights are VPlan-internal and must not
2131 // reach IR.
2132 unsigned ExecFreqKind = getMDKindID(ExecutionFrequencyMDName);
2133 unsigned EstProfKind = getMDKindID(EstimatedProfileMDName);
2134 for (const auto &[Kind, Node] : Metadata)
2135 if (Kind != ExecFreqKind && Kind != EstProfKind)
2136 I.setMetadata(Kind, Node);
2137}
2138
2139/// Returns the execution frequency recorded in \p Node.
2141 assert(Node->getNumOperands() <= 2 && "unexpected frequency node shape");
2142 uint64_t Freq =
2143 mdconst::extract<ConstantInt>(Node->getOperand(0))->getZExtValue();
2145 "frequency cannot exceed the one of an always executing block");
2146 return {BlockFrequency(Freq), Node->getNumOperands() == 2};
2147}
2148
2150 std::optional<VPExecutionFrequency> Freq, LLVMContext &Ctx) {
2151 // A recipe that always executes needs no annotation.
2152 if (!Freq || vputils::getExecutionProbability(Freq->Freq).isOne())
2153 return;
2155 ConstantInt::get(Type::getInt64Ty(Ctx), Freq->Freq.getFrequency()))};
2156 if (Freq->IsEstimated)
2158 setMetadata(Ctx.getMDKindID(ExecutionFrequencyMDName), MDNode::get(Ctx, Ops));
2159}
2160
2161std::optional<VPExecutionFrequency>
2163 if (MDNode *Node = getInternalMetadata(ExecutionFrequencyMDName))
2165 return std::nullopt;
2166}
2167
2169 if (Metadata.empty())
2170 return;
2171 unsigned ID = getMDKindID(ExecutionFrequencyMDName);
2172 erase_if(Metadata, [ID](const auto &P) { return P.first == ID; });
2173}
2174
2176 SmallVector<std::pair<unsigned, MDNode *>> MetadataIntersection;
2177 for (const auto &[KindA, MDA] : Metadata) {
2178 for (const auto &[KindB, MDB] : Other.Metadata) {
2179 if (KindA == KindB && MDA == MDB) {
2180 MetadataIntersection.emplace_back(KindA, MDA);
2181 break;
2182 }
2183 }
2184 }
2185 Metadata = std::move(MetadataIntersection);
2186}
2187
2188#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2190 const Module *M = SlotTracker.getModule();
2191 if (Metadata.empty() || !M)
2192 return;
2193
2194 ArrayRef<StringRef> MDNames = SlotTracker.getMDNames();
2195 O << " (";
2196 interleaveComma(Metadata, O, [&](const auto &KindNodePair) {
2197 auto [Kind, Node] = KindNodePair;
2198 assert(Kind < MDNames.size() && !MDNames[Kind].empty() &&
2199 "Unexpected unnamed metadata kind");
2200 O << "!" << MDNames[Kind] << " ";
2201 // Print the values of branch weights, which are more informative than the
2202 // ID of the metadata node holding them.
2203 SmallVector<uint32_t> Weights;
2204 bool IsEstimatedProfile = MDNames[Kind] == EstimatedProfileMDName;
2205 if ((Kind == LLVMContext::MD_prof || IsEstimatedProfile) &&
2206 extractBranchWeights(Node, Weights)) {
2207 if (IsEstimatedProfile)
2208 O << "estimated ";
2209 O << "{";
2210 interleaveComma(Weights, O);
2211 O << "}";
2212 } else if (MDNames[Kind] == ExecutionFrequencyMDName) {
2213 // Print the frequency together with the probability it corresponds to.
2214 auto [Freq, IsEstimated] = getExecutionFrequencyFromMD(Node);
2215 const fltSemantics &Sem = APFloat::IEEEdouble();
2216 uint64_t Full =
2218 APFloat Percent = APFloat(Sem, Freq.getFrequency()) * APFloat(Sem, 100) /
2219 APFloat(Sem, Full);
2220 SmallString<16> PercentStr;
2221 Percent.toString(PercentStr, /*FormatPrecision=*/4);
2222 O << Freq.getFrequency() << " (" << PercentStr << "%"
2223 << (IsEstimated ? ", estimated" : "") << ")";
2224 } else {
2225 SlotTracker.printMetadataAsOperand(O, Node);
2226 }
2227 });
2228 O << ")";
2229}
2230#endif
2231
2233 assert(State.VF.isVector() && "not widening");
2234 assert(Variant != nullptr && "Can't create vector function.");
2235
2236 FunctionType *VFTy = Variant->getFunctionType();
2237 // Add return type if intrinsic is overloaded on it.
2239 for (const auto &I : enumerate(args())) {
2240 Value *Arg;
2241 // Some vectorized function variants may also take a scalar argument,
2242 // e.g. linear parameters for pointers. This needs to be the scalar value
2243 // from the start of the respective part when interleaving.
2244 if (!VFTy->getParamType(I.index())->isVectorTy())
2245 Arg = State.get(I.value(), VPLane(0));
2246 else
2247 Arg = State.get(I.value(), usesFirstLaneOnly(I.value()));
2248 Args.push_back(Arg);
2249 }
2250
2253 if (CI)
2254 CI->getOperandBundlesAsDefs(OpBundles);
2255
2256 CallInst *V = State.Builder.CreateCall(Variant, Args, OpBundles);
2257 applyFlags(*V);
2258 applyMetadata(*V);
2259 V->setCallingConv(Variant->getCallingConv());
2260
2261 if (!V->getType()->isVoidTy())
2262 State.set(this, V);
2263}
2264
2266 VPCostContext &Ctx) const {
2267 assert(getVectorizedTypeVF(Variant->getReturnType()) == VF &&
2268 "Variant return type must match VF");
2269 return computeCallCost(Variant, Ctx);
2270}
2271
2273 VPCostContext &Ctx) {
2274 return Ctx.TTI.getCallInstrCost(nullptr, Variant->getReturnType(),
2275 Variant->getFunctionType()->params(),
2276 Ctx.CostKind);
2277}
2278
2280 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
2281 assert(Variant && "Variant not set");
2282 FunctionType *VFTy = Variant->getFunctionType();
2283 return all_of(enumerate(args()), [VFTy, &Op](const auto &Arg) {
2284 auto [Idx, V] = Arg;
2285 Type *ArgTy = VFTy->getParamType(Idx);
2286 return V != Op || ArgTy->isIntegerTy() || ArgTy->isFloatingPointTy() ||
2287 ArgTy->isPointerTy() || ArgTy->isByteTy();
2288 });
2289}
2290
2291#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2293 VPSlotTracker &SlotTracker) const {
2294 O << Indent << "WIDEN-CALL ";
2295
2296 Function *CalledFn = getCalledScalarFunction();
2297 if (CalledFn->getReturnType()->isVoidTy())
2298 O << "void ";
2299 else {
2301 O << " = ";
2302 }
2303
2304 O << "call";
2305 printFlags(O);
2306 O << "@" << CalledFn->getName() << "(";
2307 interleaveComma(args(), O, [&O, &SlotTracker](VPValue *Op) {
2308 Op->printAsOperand(O, SlotTracker);
2309 });
2310 O << ")";
2311
2312 O << " (using library function";
2313 if (Variant->hasName())
2314 O << ": " << Variant->getName();
2315 O << ")";
2316}
2317#endif
2318
2320 assert(State.VF.isVector() && "not widening");
2321
2322 SmallVector<Type *, 2> TysForDecl;
2323 // Add return type if intrinsic is overloaded on it.
2324 if (isVectorIntrinsicWithOverloadTypeAtArg(VectorIntrinsicID, -1,
2325 State.TTI)) {
2326 Type *RetTy = toVectorizedTy(getScalarType(), State.VF);
2327 ArrayRef<Type *> ContainedTys = getContainedTypes(RetTy);
2328 for (auto [Idx, Ty] : enumerate(ContainedTys)) {
2330 Idx, State.TTI))
2331 TysForDecl.push_back(Ty);
2332 }
2333 }
2335 for (const auto &I : enumerate(operands())) {
2336 // Some intrinsics have a scalar argument - don't replace it with a
2337 // vector.
2338 Value *Arg;
2339 if (isVectorIntrinsicWithScalarOpAtArg(VectorIntrinsicID, I.index(),
2340 State.TTI))
2341 Arg = State.get(I.value(), VPLane(0));
2342 else
2343 Arg = State.get(I.value(), usesFirstLaneOnly(I.value()));
2344 if (isVectorIntrinsicWithOverloadTypeAtArg(VectorIntrinsicID, I.index(),
2345 State.TTI))
2346 TysForDecl.push_back(Arg->getType());
2347 Args.push_back(Arg);
2348 }
2349
2350 // Use vector version of the intrinsic.
2351 Module *M = State.Builder.GetInsertBlock()->getModule();
2352 Function *VectorF =
2353 Intrinsic::getOrInsertDeclaration(M, VectorIntrinsicID, TysForDecl);
2354 assert(VectorF &&
2355 "Can't retrieve vector intrinsic or vector-predication intrinsics.");
2356
2359 if (CI)
2360 CI->getOperandBundlesAsDefs(OpBundles);
2361
2362 CallInst *V = State.Builder.CreateCall(VectorF, Args, OpBundles);
2363
2364 applyFlags(*V);
2365 applyMetadata(*V);
2366
2367 return V;
2368}
2369
2371 CallInst *V = createVectorCall(State);
2372 if (!V->getType()->isVoidTy())
2373 State.set(this, V);
2374}
2375
2378 const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx) {
2379 Type *ScalarRetTy = R.getScalarType();
2380 // Skip the reverse operation cost for the mask.
2381 // FIXME: Remove this once redundant mask reverse operations can be eliminated
2382 // by VPlanTransforms::cse before cost computation.
2383 if (ID == Intrinsic::experimental_vp_reverse && ScalarRetTy->isIntegerTy(1))
2384 return InstructionCost(0);
2385
2386 // Some backends analyze intrinsic arguments to determine cost. Use the
2387 // underlying value for the operand if it has one. Otherwise try to use the
2388 // operand of the underlying call instruction, if there is one. Otherwise
2389 // clear Arguments.
2390 // TODO: Rework TTI interface to be independent of concrete IR values.
2392 for (const auto &[Idx, Op] : enumerate(Operands)) {
2393 auto *V = Op->getUnderlyingValue();
2394 if (!V) {
2395 if (auto *UI = dyn_cast_or_null<CallBase>(R.getUnderlyingValue())) {
2396 Arguments.push_back(UI->getArgOperand(Idx));
2397 continue;
2398 }
2399 Arguments.clear();
2400 break;
2401 }
2402 Arguments.push_back(V);
2403 }
2404
2405 Type *RetTy = VF.isVector() ? toVectorizedTy(ScalarRetTy, VF) : ScalarRetTy;
2406 SmallVector<Type *> ParamTys =
2407 map_to_vector(Operands, [&](const VPValue *Op) {
2408 return toVectorTy(Op->getScalarType(), VF);
2409 });
2410
2412 for (const VPValue *Op : Operands)
2413 if (isa<VPWidenRecipe>(Op) &&
2416 break;
2417 }
2418
2419 // TODO: Rework TTI interface to avoid reliance on underlying IntrinsicInst.
2420 IntrinsicCostAttributes CostAttrs(
2421 ID, RetTy, Arguments, ParamTys, R.getFastMathFlagsOrNone(),
2422 dyn_cast_or_null<IntrinsicInst>(R.getUnderlyingValue()),
2424 return Ctx.TTI.getIntrinsicInstrCost(CostAttrs, Ctx.CostKind);
2425}
2426
2428 VPCostContext &Ctx) const {
2429 return computeCallCost(VectorIntrinsicID, operands(), *this, VF, Ctx);
2430}
2431
2433 return Intrinsic::getBaseName(VectorIntrinsicID);
2434}
2435
2437 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
2438 return all_of(enumerate(operands()), [this, &Op](const auto &X) {
2439 auto [Idx, V] = X;
2441 Idx, nullptr);
2442 });
2443}
2444
2445#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2447 VPSlotTracker &SlotTracker) const {
2448 O << Indent << "WIDEN-INTRINSIC ";
2449 if (getScalarType()->isVoidTy()) {
2450 O << "void ";
2451 } else {
2453 O << " = ";
2454 }
2455
2456 O << "call";
2457 printFlags(O);
2458 O << getIntrinsicName() << "(";
2460 O << ")";
2461}
2462#endif
2463
2465 CallInst *MemI = createVectorCall(State);
2467 assert(PtrPos && "Expected a memory intrinsic with a valid pointer position");
2468 MemI->addParamAttr(
2469 *PtrPos, Attribute::getWithAlignment(MemI->getContext(), Alignment));
2470 if (!MemI->getType()->isVoidTy())
2471 State.set(this, MemI);
2472}
2473
2475 Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment,
2476 VPCostContext &Ctx) {
2477 return Ctx.TTI.getMemIntrinsicInstrCost(
2478 MemIntrinsicCostAttributes(IID, Ty, /*Ptr=*/nullptr, IsMasked, Alignment),
2479 Ctx.CostKind);
2480}
2481
2484 VPCostContext &Ctx) const {
2485 Type *DataTy;
2487 DataTy = getOperand(*DataPos)->getScalarType();
2488 else
2489 DataTy = getScalarType();
2490 assert(!DataTy->isVoidTy() && "Expected a non-void data type");
2491 Type *Ty = toVectorTy(DataTy, VF);
2493 assert(MaskPos && "Expected a memory intrinsic with a valid mask position");
2495 !match(getOperand(*MaskPos), m_True()),
2496 Alignment, Ctx);
2497}
2498
2500 IRBuilderBase &Builder = State.Builder;
2501
2502 Value *Address = State.get(getOperand(0));
2503 Value *IncAmt = State.get(getOperand(1), /*NeedsSingleScalar=*/true);
2504 VectorType *VTy = cast<VectorType>(Address->getType());
2505
2506 // The histogram intrinsic requires a mask even if the recipe doesn't;
2507 // if the mask operand was omitted then all lanes should be executed and
2508 // we just need to synthesize an all-true mask.
2509 Value *Mask = nullptr;
2510 if (VPValue *VPMask = getMask())
2511 Mask = State.get(VPMask);
2512 else
2513 Mask =
2514 Builder.CreateVectorSplat(VTy->getElementCount(), Builder.getInt1(1));
2515
2516 // If this is a subtract, we want to invert the increment amount. We may
2517 // add a separate intrinsic in future, but for now we'll try this.
2518 if (Opcode == Instruction::Sub)
2519 IncAmt = Builder.CreateNeg(IncAmt);
2520 else
2521 assert(Opcode == Instruction::Add && "only add or sub supported for now");
2522
2523 Instruction *HistogramInst = State.Builder.CreateIntrinsicWithoutFolding(
2524 Intrinsic::experimental_vector_histogram_add, {VTy, IncAmt->getType()},
2525 {Address, IncAmt, Mask});
2526 applyMetadata(*HistogramInst);
2527}
2528
2530 VPCostContext &Ctx) const {
2531 // FIXME: Take the gather and scatter into account as well. For now we're
2532 // generating the same cost as the fallback path, but we'll likely
2533 // need to create a new TTI method for determining the cost, including
2534 // whether we can use base + vec-of-smaller-indices or just
2535 // vec-of-pointers.
2536 assert(VF.isVector() && "Invalid VF for histogram cost");
2537 Type *AddressTy = getOperand(0)->getScalarType();
2538 VPValue *IncAmt = getOperand(1);
2539 Type *IncTy = IncAmt->getScalarType();
2540 VectorType *VTy = VectorType::get(IncTy, VF);
2541
2542 // Assume that a non-constant update value (or a constant != 1) requires
2543 // a multiply, and add that into the cost.
2544 InstructionCost MulCost =
2545 Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, VTy, Ctx.CostKind);
2546 if (match(IncAmt, m_One()))
2547 MulCost = TTI::TCC_Free;
2548
2549 // Find the cost of the histogram operation itself.
2550 Type *PtrTy = VectorType::get(AddressTy, VF);
2551 Type *MaskTy = VectorType::get(Type::getInt1Ty(Ctx.LLVMCtx), VF);
2552 IntrinsicCostAttributes ICA(Intrinsic::experimental_vector_histogram_add,
2553 Type::getVoidTy(Ctx.LLVMCtx),
2554 {PtrTy, IncTy, MaskTy});
2555
2556 // Add the costs together with the add/sub operation.
2557 return Ctx.TTI.getIntrinsicInstrCost(ICA, Ctx.CostKind) + MulCost +
2558 Ctx.TTI.getArithmeticInstrCost(Opcode, VTy, Ctx.CostKind);
2559}
2560
2561#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2563 VPSlotTracker &SlotTracker) const {
2564 O << Indent << "WIDEN-HISTOGRAM buckets: ";
2566
2567 if (Opcode == Instruction::Sub)
2568 O << ", dec: ";
2569 else {
2570 assert(Opcode == Instruction::Add);
2571 O << ", inc: ";
2572 }
2574
2575 if (VPValue *Mask = getMask()) {
2576 O << ", mask: ";
2577 Mask->printAsOperand(O, SlotTracker);
2578 }
2579}
2580#endif
2581
2582VPIRFlags::FastMathFlagsTy::FastMathFlagsTy(const FastMathFlags &FMF) {
2583 AllowReassoc = FMF.allowReassoc();
2584 NoNaNs = FMF.noNaNs();
2585 NoInfs = FMF.noInfs();
2586 NoSignedZeros = FMF.noSignedZeros();
2587 AllowReciprocal = FMF.allowReciprocal();
2588 AllowContract = FMF.allowContract();
2589 ApproxFunc = FMF.approxFunc();
2590}
2591
2592VPIRFlags VPIRFlags::getDefaultFlags(unsigned Opcode, Type *ResultTy) {
2593 switch (Opcode) {
2594 case Instruction::Add:
2595 case Instruction::Sub:
2596 case Instruction::Mul:
2597 case Instruction::Shl:
2599 return WrapFlagsTy(false, false);
2600 case Instruction::Trunc:
2601 return TruncFlagsTy(false, false);
2602 case Instruction::Or:
2603 return DisjointFlagsTy(false);
2604 case Instruction::AShr:
2605 case Instruction::LShr:
2606 case Instruction::UDiv:
2607 case Instruction::SDiv:
2608 return ExactFlagsTy(false);
2609 case Instruction::GetElementPtr:
2612 return GEPNoWrapFlags::none();
2613 case Instruction::ZExt:
2614 case Instruction::UIToFP:
2615 return NonNegFlagsTy(false);
2616 case Instruction::FAdd:
2617 case Instruction::FSub:
2618 case Instruction::FMul:
2619 case Instruction::FDiv:
2620 case Instruction::FRem:
2621 case Instruction::FNeg:
2622 case Instruction::FPExt:
2623 case Instruction::FPTrunc:
2624 return FastMathFlags();
2625 case Instruction::Select:
2626 case Instruction::PHI:
2627 case Instruction::Call:
2628 // Selects, phis and calls only have fast-math flags if they have a
2629 // supported floating-point result type.
2631 return FastMathFlags();
2632 return VPIRFlags();
2633 case Instruction::ICmp:
2634 case Instruction::FCmp:
2636 llvm_unreachable("opcode requires explicit flags");
2637 default:
2638 return VPIRFlags();
2639 }
2640}
2641
2642#if !defined(NDEBUG)
2643bool VPIRFlags::flagsValidForOpcode(unsigned Opcode) const {
2644 switch (OpType) {
2645 case OperationType::OverflowingBinOp:
2646 return Opcode == Instruction::Add || Opcode == Instruction::Sub ||
2647 Opcode == Instruction::Mul || Opcode == Instruction::Shl ||
2648 Opcode == VPInstruction::VPInstruction::CanonicalIVIncrementForPart;
2649 case OperationType::Trunc:
2650 return Opcode == Instruction::Trunc;
2651 case OperationType::DisjointOp:
2652 return Opcode == Instruction::Or;
2653 case OperationType::PossiblyExactOp:
2654 return Opcode == Instruction::AShr || Opcode == Instruction::LShr ||
2655 Opcode == Instruction::UDiv || Opcode == Instruction::SDiv;
2656 case OperationType::GEPOp:
2657 return Opcode == Instruction::GetElementPtr ||
2658 Opcode == VPInstruction::PtrAdd ||
2659 Opcode == VPInstruction::WidePtrAdd;
2660 case OperationType::FPMathOp:
2661 return Opcode == Instruction::Call || Opcode == Instruction::FAdd ||
2662 Opcode == Instruction::FMul || Opcode == Instruction::FSub ||
2663 Opcode == Instruction::FNeg || Opcode == Instruction::FDiv ||
2664 Opcode == Instruction::FRem || Opcode == Instruction::FPExt ||
2665 Opcode == Instruction::FPTrunc || Opcode == Instruction::PHI ||
2666 Opcode == Instruction::Select || Opcode == Instruction::SIToFP ||
2667 Opcode == Instruction::UIToFP ||
2668 Opcode == VPInstruction::WideIVStep ||
2670 case OperationType::FCmp:
2671 return Opcode == Instruction::FCmp;
2672 case OperationType::NonNegOp:
2673 return Opcode == Instruction::ZExt || Opcode == Instruction::UIToFP;
2674 case OperationType::Cmp:
2675 return Opcode == Instruction::FCmp || Opcode == Instruction::ICmp;
2676 case OperationType::ReductionOp:
2678 case OperationType::Other:
2679 return true;
2680 }
2681 llvm_unreachable("Unknown OperationType enum");
2682}
2683
2685 Type *ResultTy) const {
2686 // Handle opcodes without default flags.
2687 if (Opcode == Instruction::ICmp)
2688 return OpType == OperationType::Cmp;
2689 if (Opcode == Instruction::FCmp)
2690 return OpType == OperationType::FCmp;
2692 return OpType == OperationType::ReductionOp;
2693
2694 OperationType Required = getDefaultFlags(Opcode, ResultTy).OpType;
2695 return Required == OperationType::Other || Required == OpType;
2696}
2697#endif
2698
2699#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2700static void printRecurrenceKind(raw_ostream &OS, const RecurKind &Kind) {
2701 switch (Kind) {
2702 case RecurKind::None:
2703 OS << "none";
2704 break;
2705 case RecurKind::Add:
2706 OS << "add";
2707 break;
2708 case RecurKind::Sub:
2709 OS << "sub";
2710 break;
2712 OS << "add-chain-with-subs";
2713 break;
2714 case RecurKind::Mul:
2715 OS << "mul";
2716 break;
2717 case RecurKind::Or:
2718 OS << "or";
2719 break;
2720 case RecurKind::And:
2721 OS << "and";
2722 break;
2723 case RecurKind::Xor:
2724 OS << "xor";
2725 break;
2726 case RecurKind::SMin:
2727 OS << "smin";
2728 break;
2729 case RecurKind::SMax:
2730 OS << "smax";
2731 break;
2732 case RecurKind::UMin:
2733 OS << "umin";
2734 break;
2735 case RecurKind::UMax:
2736 OS << "umax";
2737 break;
2738 case RecurKind::FAdd:
2739 OS << "fadd";
2740 break;
2742 OS << "fadd-chain-with-subs";
2743 break;
2744 case RecurKind::FSub:
2745 OS << "fsub";
2746 break;
2747 case RecurKind::FMul:
2748 OS << "fmul";
2749 break;
2750 case RecurKind::FMin:
2751 OS << "fmin";
2752 break;
2753 case RecurKind::FMax:
2754 OS << "fmax";
2755 break;
2756 case RecurKind::FMinNum:
2757 OS << "fminnum";
2758 break;
2759 case RecurKind::FMaxNum:
2760 OS << "fmaxnum";
2761 break;
2763 OS << "fminimum";
2764 break;
2766 OS << "fmaximum";
2767 break;
2769 OS << "fminimumnum";
2770 break;
2772 OS << "fmaximumnum";
2773 break;
2774 case RecurKind::FMulAdd:
2775 OS << "fmuladd";
2776 break;
2777 case RecurKind::AnyOf:
2778 OS << "any-of";
2779 break;
2780 case RecurKind::FindIV:
2781 OS << "find-iv";
2782 break;
2784 OS << "find-last";
2785 break;
2786 }
2787}
2788
2790 switch (OpType) {
2791 case OperationType::Cmp:
2793 break;
2794 case OperationType::FCmp:
2797 break;
2798 case OperationType::DisjointOp:
2799 if (DisjointFlags.IsDisjoint)
2800 O << " disjoint";
2801 break;
2802 case OperationType::PossiblyExactOp:
2803 if (ExactFlags.IsExact)
2804 O << " exact";
2805 break;
2806 case OperationType::OverflowingBinOp:
2807 if (WrapFlags.HasNUW)
2808 O << " nuw";
2809 if (WrapFlags.HasNSW)
2810 O << " nsw";
2811 break;
2812 case OperationType::Trunc:
2813 if (TruncFlags.HasNUW)
2814 O << " nuw";
2815 if (TruncFlags.HasNSW)
2816 O << " nsw";
2817 break;
2818 case OperationType::FPMathOp:
2820 break;
2821 case OperationType::GEPOp: {
2823 if (Flags.isInBounds())
2824 O << " inbounds";
2825 else if (Flags.hasNoUnsignedSignedWrap())
2826 O << " nusw";
2827 if (Flags.hasNoUnsignedWrap())
2828 O << " nuw";
2829 break;
2830 }
2831 case OperationType::NonNegOp:
2832 if (NonNegFlags.NonNeg)
2833 O << " nneg";
2834 break;
2835 case OperationType::ReductionOp: {
2836 O << " (";
2838 if (isReductionInLoop())
2839 O << ", in-loop";
2840 if (isReductionOrdered())
2841 O << ", ordered";
2842 O << ")";
2844 break;
2845 }
2846 case OperationType::Other:
2847 break;
2848 }
2849 O << " ";
2850}
2851#endif
2852
2854 auto &Builder = State.Builder;
2855 switch (Opcode) {
2856 case Instruction::Call:
2857 case Instruction::UncondBr:
2858 case Instruction::CondBr:
2859 case Instruction::PHI:
2860 case Instruction::GetElementPtr:
2861 llvm_unreachable("This instruction is handled by a different recipe.");
2862 case Instruction::UDiv:
2863 case Instruction::SDiv:
2864 case Instruction::SRem:
2865 case Instruction::URem:
2866 case Instruction::Add:
2867 case Instruction::FAdd:
2868 case Instruction::Sub:
2869 case Instruction::FSub:
2870 case Instruction::FNeg:
2871 case Instruction::Mul:
2872 case Instruction::FMul:
2873 case Instruction::FDiv:
2874 case Instruction::FRem:
2875 case Instruction::Shl:
2876 case Instruction::LShr:
2877 case Instruction::AShr:
2878 case Instruction::And:
2879 case Instruction::Or:
2880 case Instruction::Xor: {
2881 // Just widen unops and binops.
2883 for (VPValue *VPOp : operands())
2884 Ops.push_back(State.get(VPOp));
2885
2886 Value *V = Builder.CreateNAryOp(Opcode, Ops);
2887
2888 if (auto *VecOp = dyn_cast<Instruction>(V)) {
2889 applyFlags(*VecOp);
2890 applyMetadata(*VecOp);
2891 }
2892
2893 // Use this vector value for all users of the original instruction.
2894 State.set(this, V);
2895 break;
2896 }
2897 case Instruction::ExtractValue: {
2898 assert(getNumOperands() == 2 && "expected single level extractvalue");
2899 Value *Op = State.get(getOperand(0));
2900 Value *Extract = Builder.CreateExtractValue(
2901 Op, cast<VPConstantInt>(getOperand(1))->getZExtValue());
2902 State.set(this, Extract);
2903 break;
2904 }
2905 case Instruction::Freeze: {
2906 Value *Op = State.get(getOperand(0));
2907 Value *Freeze = Builder.CreateFreeze(Op);
2908 State.set(this, Freeze);
2909 break;
2910 }
2911 case Instruction::ICmp:
2912 case Instruction::FCmp: {
2913 // Widen compares. Generate vector compares.
2914 bool FCmp = Opcode == Instruction::FCmp;
2915 Value *A = State.get(getOperand(0));
2916 Value *B = State.get(getOperand(1));
2917 Value *C = nullptr;
2918 if (FCmp) {
2919 C = Builder.CreateFCmp(getPredicate(), A, B);
2920 } else {
2921 C = Builder.CreateICmp(getPredicate(), A, B);
2922 }
2923 if (auto *I = dyn_cast<Instruction>(C)) {
2924 applyFlags(*I);
2925 applyMetadata(*I);
2926 }
2927 State.set(this, C);
2928 break;
2929 }
2930 case Instruction::Select: {
2931 VPValue *CondOp = getOperand(0);
2932 Value *Cond = State.get(CondOp, vputils::isSingleScalar(CondOp));
2933 Value *Op0 = State.get(getOperand(1));
2934 Value *Op1 = State.get(getOperand(2));
2935 Value *Sel = State.Builder.CreateSelect(Cond, Op0, Op1);
2936 State.set(this, Sel);
2937 if (auto *I = dyn_cast<Instruction>(Sel)) {
2939 applyFlags(*I);
2940 applyMetadata(*I);
2941 }
2942 break;
2943 }
2944 default:
2945 // This instruction is not vectorized by simple widening.
2946 LLVM_DEBUG(dbgs() << "LV: Found an unhandled opcode : "
2947 << Instruction::getOpcodeName(Opcode));
2948 llvm_unreachable("Unhandled instruction!");
2949 } // end of switch.
2950
2951#if !defined(NDEBUG)
2952 // Verify that VPlan type inference results agree with the type of the
2953 // generated values.
2954 assert(VectorType::get(this->getScalarType(), State.VF) ==
2955 State.get(this)->getType() &&
2956 "inferred type and type from generated instructions do not match");
2957#endif
2958}
2959
2961 VPCostContext &Ctx) const {
2962 switch (Opcode) {
2963 case Instruction::UDiv:
2964 case Instruction::SDiv:
2965 case Instruction::SRem:
2966 case Instruction::URem:
2967 // If the div/rem operation isn't safe to speculate and requires
2968 // predication, then the only way we can even create a vplan is to insert
2969 // a select on the second input operand to ensure we use the value of 1
2970 // for the inactive lanes. The select will be costed separately.
2971 case Instruction::FNeg:
2972 case Instruction::Add:
2973 case Instruction::FAdd:
2974 case Instruction::Sub:
2975 case Instruction::FSub:
2976 case Instruction::Mul:
2977 case Instruction::FMul:
2978 case Instruction::FDiv:
2979 case Instruction::FRem:
2980 case Instruction::Shl:
2981 case Instruction::LShr:
2982 case Instruction::AShr:
2983 case Instruction::And:
2984 case Instruction::Or:
2985 case Instruction::Xor:
2986 case Instruction::Freeze:
2987 case Instruction::ExtractValue:
2988 case Instruction::ICmp:
2989 case Instruction::FCmp:
2990 case Instruction::Select:
2991 return getCostForRecipeWithOpcode(getOpcode(), VF, Ctx);
2992 default:
2993 llvm_unreachable("Unsupported opcode for instruction");
2994 }
2995}
2996
2997#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2999 VPSlotTracker &SlotTracker) const {
3000 O << Indent << "WIDEN ";
3002 O << " = " << Instruction::getOpcodeName(Opcode);
3003 printFlags(O);
3005}
3006#endif
3007
3009 auto &Builder = State.Builder;
3010 /// Vectorize casts.
3011 assert(State.VF.isVector() && "Not vectorizing?");
3012 Type *DestTy = VectorType::get(getScalarType(), State.VF);
3013 VPValue *Op = getOperand(0);
3014 Value *A = State.get(Op);
3015 Value *Cast = Builder.CreateCast(Instruction::CastOps(Opcode), A, DestTy);
3016 State.set(this, Cast);
3017 if (auto *CastOp = dyn_cast<Instruction>(Cast)) {
3018 applyFlags(*CastOp);
3019 applyMetadata(*CastOp);
3020 }
3021}
3022
3027
3028#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3030 VPSlotTracker &SlotTracker) const {
3031 O << Indent << "WIDEN-CAST ";
3033 O << " = " << Instruction::getOpcodeName(Opcode);
3034 printFlags(O);
3036 O << " to " << *getScalarType();
3037}
3038#endif
3039
3041 VPCostContext &Ctx) const {
3042 return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
3043}
3044
3045#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3047 raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
3048 O << Indent;
3050 O << " = WIDEN-INDUCTION";
3051 printFlags(O);
3053
3054 if (auto *TI = getTruncInst())
3055 O << " (truncated to " << *TI->getType() << ")";
3056}
3057#endif
3058
3060 // The step may be defined by a recipe in the preheader (e.g. if it requires
3061 // SCEV expansion), but for the canonical induction the step is required to be
3062 // 1, which is represented as live-in.
3063 return match(getStartValue(), m_ZeroInt()) &&
3064 match(getStepValue(), m_One()) &&
3065 getScalarType() == getRegion()->getCanonicalIVType();
3066}
3067
3070 VPCostContext &Ctx) const {
3071 // A widened induction generates a vector phi and increments it by the
3072 // splatted step each iteration.
3074 InstructionCost Cost = Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
3075 Type *StepTy = getScalarType();
3076 unsigned IncOpc = ID.getKind() == InductionDescriptor::IK_IntInduction
3077 ? Instruction::Add
3078 : ID.getInductionOpcode();
3079 assert(IncOpc != Instruction::BinaryOpsEnd &&
3080 "induction must have a valid increment opcode");
3081 return Cost + Ctx.TTI.getArithmeticInstrCost(IncOpc, toVectorTy(StepTy, VF),
3082 Ctx.CostKind);
3083}
3084
3085/// Returns the ConstantFP \p V wraps, or nullptr if it does not wrap one.
3086static const ConstantFP *getConstantFP(const VPValue *V) {
3087 auto *C = dyn_cast<VPConstant>(V);
3088 return C ? dyn_cast<ConstantFP>(C->getConstant()) : nullptr;
3089}
3090
3092 VPCostContext &Ctx) const {
3093 // The cost model for this is modelled on expandVPDerivedIV in
3094 // VPlanTransforms.cpp. In order to avoid overly pessimistic costs that can
3095 // negatively affect vectorization it takes into account any expected
3096 // simplifications that happen in simplifyRecipe.
3097 switch (getInductionKind()) {
3098 default:
3099 // TODO: Compute cost for remaining kinds.
3100 break;
3102 // There are currently no tests that expose a path where all lanes are
3103 // used, so it's better to bail out for now.
3104 if (!vputils::onlyFirstLaneUsed(this))
3105 break;
3106
3107 // Start off by assuming we need both mul and add, then refine this.
3108 bool NeedsMul = true, NeedsAdd = true, NeedsShl = false;
3109
3110 // If the start value is zero the add gets folded away.
3111 if (auto *StartC = dyn_cast<VPConstantInt>(getStartValue()))
3112 NeedsAdd = !StartC->isZero();
3113
3114 // For some values of step the arithmetic changes:
3115 // 1. A step of 1 requires no operation.
3116 // 2. A step of -1 requires a negate.
3117 // 3. A power-of-2 step will use a shl, instead of a mul.
3118 Type *StepTy = getStepValue()->getScalarType();
3120 if (auto *StepC = dyn_cast<VPConstantInt>(getStepValue())) {
3121 if (StepC->isOne())
3122 NeedsMul = false;
3123 else if (StepC->getAPInt().isAllOnes()) {
3124 // This will most likely end up as a negate in simplifyRecipe, and
3125 // the negate will be combined with the add to make a sub.
3126 // NOTE: This is perhaps an invalid assumption that the cost of an
3127 // 'add' is the same as a 'sub'.
3128 NeedsMul = false;
3129 NeedsAdd = true;
3130 } else if (StepC->getAPInt().isPowerOf2()) {
3131 // This will most likely end up as a shift-left in simplifyRecipe
3132 NeedsMul = false;
3133 NeedsShl = true;
3134 }
3135 }
3136
3137 // Add the cost of the conversion from index to step type if the index
3138 // will be used.
3139 Type *IndexTy = getIndex()->getScalarType();
3140 unsigned StepTySize = StepTy->getScalarSizeInBits();
3141 unsigned IndexTySize = IndexTy->getScalarSizeInBits();
3142 if ((NeedsAdd || NeedsMul || NeedsShl) && StepTySize != IndexTySize) {
3143 unsigned CastOpc =
3144 StepTySize < IndexTySize ? Instruction::Trunc : Instruction::ZExt;
3145 Cost += Ctx.TTI.getCastInstrCost(
3146 CastOpc, StepTy, IndexTy, TTI::CastContextHint::None, Ctx.CostKind);
3147 }
3148
3149 if (NeedsMul)
3150 Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Mul, StepTy,
3151 Ctx.CostKind);
3152 if (NeedsShl)
3153 Cost += Ctx.TTI.getArithmeticInstrCost(
3154 Instruction::Shl, StepTy, Ctx.CostKind,
3155 {TargetTransformInfo::OK_AnyValue, TargetTransformInfo::OP_None},
3156 {TargetTransformInfo::OK_UniformConstantValue,
3157 TargetTransformInfo::OP_None});
3158 if (NeedsAdd)
3159 Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::Add, StepTy,
3160 Ctx.CostKind);
3161 return Cost;
3162 }
3164 // There are currently no tests that expose a path where all lanes are
3165 // used, so it's better to bail out for now.
3166 if (!vputils::onlyFirstLaneUsed(this))
3167 break;
3168
3169 // Unlike the integer case, converting the index to the FP step type is
3170 // unavoidable: the index is always the integer canonical IV, so this
3171 // cast is never folded away.
3172 Type *StepTy = getStepValue()->getScalarType();
3173 Type *IndexTy = getIndex()->getScalarType();
3175 Ctx.TTI.getCastInstrCost(Instruction::SIToFP, StepTy, IndexTy,
3176 TTI::CastContextHint::None, Ctx.CostKind);
3177
3178 // If the step is 1.0, the multiply is an exact identity and gets folded
3179 // away, independent of fast-math flags.
3180 const ConstantFP *StepC = getConstantFP(getStepValue());
3181 bool NeedsMul = !StepC || !StepC->isOne();
3182
3183 // "fadd -0.0, X" folds to X unconditionally, but "fadd 0.0, X" only folds
3184 // to X without nsz if X can be proven to never be -0.0, which we cannot, as
3185 // Step may be -0.0.
3186 // TODO: Consider fast-math flags when they are available in
3187 // VPDerivedIVRecipe.
3188 const ConstantFP *StartC = getConstantFP(getStartValue());
3189 bool AddFolds = getFPBinOp()->getOpcode() == Instruction::FAdd && StartC &&
3190 StartC->isZero() && (StartC->isNegZero() || !NeedsMul);
3191
3192 if (NeedsMul)
3193 Cost += Ctx.TTI.getArithmeticInstrCost(Instruction::FMul, StepTy,
3194 Ctx.CostKind);
3195 if (!AddFolds)
3196 Cost += Ctx.TTI.getArithmeticInstrCost(getFPBinOp()->getOpcode(), StepTy,
3197 Ctx.CostKind);
3198 return Cost;
3199 }
3200 }
3201
3202 return 0;
3203}
3204
3205#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3207 VPSlotTracker &SlotTracker) const {
3208 O << Indent;
3210 O << " = DERIVED-IV";
3211 printFlags(O);
3212 getStartValue()->printAsOperand(O, SlotTracker);
3213 O << " + ";
3214 getOperand(1)->printAsOperand(O, SlotTracker);
3215 O << " * ";
3216 getStepValue()->printAsOperand(O, SlotTracker);
3217}
3218#endif
3219
3223
3225 VPCostContext &Ctx) const {
3226 Type *BaseIVTy = getOperand(0)->getScalarType();
3227 assert((BaseIVTy->isIntegerTy() || BaseIVTy->isFloatingPointTy()) &&
3228 "VPScalarIVStepsRecipe is only created for integer and FP inductions");
3229
3230 // If only the first lane is used, then there won't be any code that remains
3231 // in the loop for the first unrolled part.
3233 return 0;
3234
3235 // If the vector body executes at most once, the canonical IV is a constant
3236 // and every lane's step folds away with it.
3237 if (VPCostContext::executesAtMostOnce(*getParent()->getPlan(), VF))
3238 return 0;
3239
3240 // Typically the operations are:
3241 // 1. Add the start index to each lane value.
3242 // 2. Multiply the start index by the step.
3243 // 3. Add the scaled start index to base IV.
3244 // Any code generated for 1 and 2 should be loop invariant and therefore
3245 // hoisted out of the loop. We only need to add on the cost of 3.
3247 if (BaseIVTy->isFloatingPointTy()) {
3248 // Unlike the integer case, the users of an FP induction cannot be re-based
3249 // on a common value, so each lane needs its own FAdd/FSub.
3250 assert(!VF.isScalable() &&
3251 "FP scalar steps for all lanes are only created for fixed VFs");
3252 Cost = Ctx.TTI.getArithmeticInstrCost(InductionOpcode, BaseIVTy,
3253 Ctx.CostKind) *
3254 (VF.getFixedValue() - 1);
3255 } else {
3256 // Given the users of VPScalarIVStepsRecipe tend to be scalarized GEPs, i.e.
3257 // %add1 = add i32 %iv, 0
3258 // %add2 = add i32 %iv, 1
3259 // %gep1 = getelementptr i8, ptr %p, i32 %add1
3260 // %gep2 = getelementptr i8, ptr %p, i32 %add2
3261 // it's very likely that these GEPs will all be rewritten to have a common
3262 // base such that what's left is just
3263 // %base_gep = getelementptr i8, ptr %p, i32 %iv
3264 // %gep1 = getelementptr i8, ptr %base_gep, i32 0
3265 // %gep2 = getelementptr i8, ptr %base_gep, i32 1
3266 // Therefore, in reality the cost is somewhere betwen 1*AddCost and
3267 // (NumLanes - 1) * AddCost. For now, assume the cost of a single add.
3268 Cost = Ctx.TTI.getArithmeticInstrCost(Instruction::Add, BaseIVTy,
3269 Ctx.CostKind);
3270 }
3271
3272 // If the steps are generated inside a replicate region, scale by execution
3273 // probability.
3274 const VPRegionBlock *Region = getRegion();
3275 if (Region && Region->isReplicator())
3276 Cost /= Ctx.getCostDivisor(
3277 Region->getEntryBranchOnMask()->getExecutionFrequency());
3278 return Cost;
3279}
3280
3282 // Fast-math-flags propagate from the original induction instruction.
3283 IRBuilder<>::FastMathFlagGuard FMFG(State.Builder);
3284 State.Builder.setFastMathFlags(getFastMathFlagsOrNone());
3285
3286 /// Compute scalar induction steps. \p ScalarIV is the scalar induction
3287 /// variable on which to base the steps, \p Step is the size of the step.
3288
3289 Value *BaseIV = State.get(getOperand(0), VPLane(0));
3290 Value *Step = State.get(getStepValue(), VPLane(0));
3291 IRBuilderBase &Builder = State.Builder;
3292
3293 // Ensure step has the same type as that of scalar IV.
3294 Type *BaseIVTy = BaseIV->getType()->getScalarType();
3295 assert(BaseIVTy == Step->getType() && "Types of BaseIV and Step must match!");
3296
3297 // We build scalar steps for both integer and floating-point induction
3298 // variables. Here, we determine the kind of arithmetic we will perform.
3301 if (BaseIVTy->isIntegerTy()) {
3302 AddOp = Instruction::Add;
3303 MulOp = Instruction::Mul;
3304 } else {
3305 AddOp = InductionOpcode;
3306 MulOp = Instruction::FMul;
3307 }
3308
3309 // Lanes other than the first have been materialized as separate
3310 // single-scalar recipes by replicateByVF, each with its own start index.
3311 assert((vputils::onlyFirstLaneUsed(this) || State.VF.isScalar()) &&
3312 "must have been replicated by VF");
3313 Value *StartIdx = getStartIndex() ? State.get(getStartIndex(), true)
3314 : Constant::getNullValue(BaseIVTy);
3315 auto *Mul = Builder.CreateBinOp(MulOp, StartIdx, Step);
3316 auto *Add = Builder.CreateBinOp(AddOp, BaseIV, Mul);
3317 State.set(this, Add, VPLane(0));
3318}
3319
3320#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3322 VPSlotTracker &SlotTracker) const {
3323 O << Indent;
3325 O << " = SCALAR-STEPS ";
3327}
3328#endif
3329
3331 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
3333}
3334
3336 assert(State.VF.isVector() && "not widening");
3337 auto Ops = map_to_vector(operands(), [&](VPValue *Op) {
3338 return State.get(Op, vputils::isSingleScalar(Op));
3339 });
3340 auto *GEP =
3341 State.Builder.CreateGEP(getSourceElementType(), Ops.front(),
3342 drop_begin(Ops), "wide.gep", getGEPNoWrapFlags());
3343 State.set(this, GEP, vputils::isSingleScalar(this));
3344}
3345
3346#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3348 VPSlotTracker &SlotTracker) const {
3349 O << Indent << "WIDEN-GEP ";
3351 O << " = getelementptr";
3352 printFlags(O);
3354}
3355#endif
3356
3358 assert(!getOffset() && "Unexpected offset operand");
3359 VPBuilder Builder(this);
3360 VPlan &Plan = *getParent()->getPlan();
3361 VPValue *VFVal = getVFValue();
3362 const DataLayout &DL = Plan.getDataLayout();
3363 Type *IndexTy = DL.getIndexType(this->getScalarType());
3364 VPValue *Stride =
3365 Plan.getConstantInt(IndexTy, getStride(), /*IsSigned=*/true);
3366 VPValue *VF =
3367 Builder.createScalarZExtOrTrunc(VFVal, IndexTy, DebugLoc::getUnknown());
3368
3369 // Offset for Part0 = Offset0 = Stride * (VF - 1).
3370 VPInstruction *VFMinusOne =
3371 Builder.createSub(VF, Plan.getConstantInt(IndexTy, 1u),
3372 DebugLoc::getUnknown(), "", {true, true});
3373 VPInstruction *Offset0 =
3374 Builder.createOverflowingOp(Instruction::Mul, {VFMinusOne, Stride});
3375
3376 // Offset for PartN = Offset0 + Part * Stride * VF.
3377 VPValue *PartxStride =
3378 Plan.getConstantInt(IndexTy, Part * getStride(), /*IsSigned=*/true);
3379 VPValue *Offset = Builder.createAdd(
3380 Offset0,
3381 Builder.createOverflowingOp(Instruction::Mul, {PartxStride, VF}));
3383}
3384
3386 auto &Builder = State.Builder;
3387 assert(getOffset() && "Expected prior materialization of offset");
3388 Value *Ptr = State.get(getPointer(), true);
3389 Value *Offset = State.get(getOffset(), true);
3390 Value *ResultPtr = Builder.CreateGEP(getSourceElementType(), Ptr, Offset, "",
3392 State.set(this, ResultPtr, /*IsScalar*/ true);
3393}
3394
3395#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3397 VPSlotTracker &SlotTracker) const {
3398 O << Indent;
3400 O << " = vector-end-pointer";
3401 printFlags(O);
3402 getSourceElementType()->print(O);
3403 O << ", ";
3405}
3406#endif
3407
3409 assert(getVFxPart() &&
3410 "Expected prior simplification of recipe without VFxPart");
3411
3412 auto &Builder = State.Builder;
3413 Value *Ptr = State.get(getOperand(0), VPLane(0));
3414 Value *Offset = State.get(getVFxPart(), true);
3415 // TODO: Expand to VPInstruction to support constant folding.
3416 if (!match(getStride(), m_One())) {
3417 Value *Stride = Builder.CreateZExtOrTrunc(State.get(getStride(), true),
3418 Offset->getType());
3419 Offset = Builder.CreateMul(Offset, Stride);
3420 }
3421 Value *ResultPtr = Builder.CreateGEP(getSourceElementType(), Ptr, Offset, "",
3423 State.set(this, ResultPtr, /*IsScalar*/ true);
3424}
3425
3426#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3428 VPSlotTracker &SlotTracker) const {
3429 O << Indent;
3431 O << " = vector-pointer";
3432 printFlags(O);
3433 getSourceElementType()->print(O);
3434 O << ", ";
3436}
3437#endif
3438
3440 VPCostContext &Ctx) const {
3441 // A blend will be expanded to a select VPInstruction, which will generate a
3442 // scalar select if only the first lane is used.
3444 VF = ElementCount::getFixed(1);
3445
3446 Type *ResultTy = toVectorTy(this->getScalarType(), VF);
3447 Type *CmpTy = toVectorTy(Type::getInt1Ty(Ctx.LLVMCtx), VF);
3448
3450 for (unsigned I = 1, E = getNumIncomingValues(); I != E; ++I) {
3451 CmpPredicate Pred;
3452 if (!match(getMask(I), m_Cmp(Pred, m_VPValue(), m_VPValue())))
3453 Pred = getScalarType()->isFloatingPointTy() ? CmpInst::BAD_FCMP_PREDICATE
3455 Cost += Ctx.TTI.getCmpSelInstrCost(Instruction::Select, ResultTy, CmpTy,
3456 Pred, Ctx.CostKind);
3457 }
3458 return Cost;
3459}
3460
3461#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3463 VPSlotTracker &SlotTracker) const {
3464 O << Indent << "BLEND ";
3466 O << " =";
3467 printFlags(O);
3468 if (getNumIncomingValues() == 1) {
3469 // Not a User of any mask: not really blending, this is a
3470 // single-predecessor phi.
3471 getIncomingValue(0)->printAsOperand(O, SlotTracker);
3472 } else {
3473 for (unsigned I = 0, E = getNumIncomingValues(); I < E; ++I) {
3474 if (I != 0)
3475 O << " ";
3476 getIncomingValue(I)->printAsOperand(O, SlotTracker);
3477 if (I == 0 && isNormalized())
3478 continue;
3479 O << "/";
3480 getMask(I)->printAsOperand(O, SlotTracker);
3481 }
3482 }
3483}
3484#endif
3485
3489 "In-loop AnyOf reductions aren't currently supported");
3490 // Propagate the fast-math flags carried by the underlying instruction.
3491 IRBuilderBase::FastMathFlagGuard FMFGuard(State.Builder);
3492 State.Builder.setFastMathFlags(getFastMathFlagsOrNone());
3493 Value *NewVecOp = State.get(getVecOp());
3494 if (VPValue *Cond = getCondOp()) {
3495 Value *NewCond = State.get(Cond, State.VF.isScalar());
3496 VectorType *VecTy = dyn_cast<VectorType>(NewVecOp->getType());
3497 Type *ElementTy = VecTy ? VecTy->getElementType() : NewVecOp->getType();
3498
3499 Value *Start =
3501 if (State.VF.isVector())
3502 Start = State.Builder.CreateVectorSplat(VecTy->getElementCount(), Start);
3503
3504 Value *Select = State.Builder.CreateSelect(NewCond, NewVecOp, Start);
3505 NewVecOp = Select;
3506 }
3507 Value *NewRed;
3508 Value *NextInChain;
3509 if (isOrdered()) {
3510 Value *PrevInChain = State.get(getChainOp(), /*NeedsSingleScalar=*/true);
3511 if (State.VF.isVector())
3512 NewRed =
3513 createOrderedReduction(State.Builder, Kind, NewVecOp, PrevInChain);
3514 else
3515 NewRed = State.Builder.CreateBinOp(
3517 PrevInChain, NewVecOp);
3518 PrevInChain = NewRed;
3519 NextInChain = NewRed;
3520 } else if (isPartialReduction()) {
3521 assert((Kind == RecurKind::Add || Kind == RecurKind::FAdd) &&
3522 "Unexpected partial reduction kind");
3523 Value *PrevInChain = State.get(getChainOp(), /*NeedsSingleScalar=*/false);
3524 NewRed = State.Builder.CreateIntrinsic(
3525 PrevInChain->getType(),
3526 Kind == RecurKind::Add ? Intrinsic::vector_partial_reduce_add
3527 : Intrinsic::vector_partial_reduce_fadd,
3528 {PrevInChain, NewVecOp}, State.Builder.getFastMathFlags(),
3529 "partial.reduce");
3530 PrevInChain = NewRed;
3531 NextInChain = NewRed;
3532 } else {
3533 assert(isInLoop() &&
3534 "The reduction must either be ordered, partial or in-loop");
3535 Value *PrevInChain = State.get(getChainOp(), /*NeedsSingleScalar=*/true);
3536 NewRed = createSimpleReduction(State.Builder, NewVecOp, Kind);
3538 NextInChain = createMinMaxOp(State.Builder, Kind, NewRed, PrevInChain);
3539 else
3540 NextInChain = State.Builder.CreateBinOp(
3542 PrevInChain, NewRed);
3543 }
3544 State.set(this, NextInChain, /*IsScalar*/ !isPartialReduction());
3545}
3546
3548
3549 assert(State.VF.isVector() &&
3550 "Shouldn't generate VPReductionEVLRecipe with scalar VF");
3551 auto &Builder = State.Builder;
3552 // Propagate the fast-math flags carried by the underlying instruction.
3553 IRBuilderBase::FastMathFlagGuard FMFGuard(Builder);
3554 Builder.setFastMathFlags(getFastMathFlagsOrNone());
3555
3557 Value *Prev =
3558 State.get(getChainOp(), /*NeedsSingleScalar=*/!isPartialReduction());
3559 Value *VecOp = State.get(getVecOp());
3560 Value *EVL = State.get(getEVL(), VPLane(0));
3561
3562 Value *Mask;
3563 if (VPValue *CondOp = getCondOp())
3564 Mask = State.get(CondOp);
3565 else
3566 Mask = Builder.CreateVectorSplat(State.VF, Builder.getTrue());
3567
3568 Value *NewRed;
3569 if (isPartialReduction()) {
3570 // For partial reductions, we need to generate a predicated select
3571 // (vp.merge) since `@llvm.vector.partial.reduce()` doesn't have a vector
3572 // predicated version.
3573 VectorType *VecTy = cast<VectorType>(VecOp->getType());
3574 Value *Identity = getRecurrenceIdentity(Kind, VecTy->getElementType(),
3576 Identity =
3577 State.Builder.CreateVectorSplat(VecTy->getElementCount(), Identity);
3578
3579 // TODO: Calculate the predicate cost for the partial reduction.
3580 Value *NewVecOp = State.Builder.CreateIntrinsic(
3581 VecTy, Intrinsic::vp_merge, {Mask, VecOp, Identity, EVL});
3582 assert((Kind == RecurKind::Add || Kind == RecurKind::FAdd) &&
3583 "Unexpected partial reduction kind");
3584 NewRed = State.Builder.CreateIntrinsic(
3585 Prev->getType(),
3586 Kind == RecurKind::Add ? Intrinsic::vector_partial_reduce_add
3587 : Intrinsic::vector_partial_reduce_fadd,
3588 {Prev, NewVecOp}, State.Builder.getFastMathFlags(), "partial.reduce");
3589 } else if (isOrdered()) {
3590 NewRed = createOrderedReduction(Builder, Kind, VecOp, Prev, Mask, EVL);
3591 } else {
3592 NewRed = createSimpleReduction(Builder, VecOp, Kind, Mask, EVL);
3594 NewRed = createMinMaxOp(Builder, Kind, NewRed, Prev);
3595 else
3596 NewRed = Builder.CreateBinOp(
3598 Prev);
3599 }
3600 State.set(this, NewRed, !isPartialReduction());
3601}
3602
3604 VPCostContext &Ctx) const {
3605 RecurKind RdxKind = getRecurrenceKind();
3606 Type *ElementTy = this->getScalarType();
3607 auto *VectorTy = cast<VectorType>(toVectorTy(ElementTy, VF));
3608 unsigned Opcode = RecurrenceDescriptor::getOpcode(RdxKind);
3610 std::optional<FastMathFlags> OptionalFMF =
3611 ElementTy->isFloatingPointTy() ? std::make_optional(FMFs) : std::nullopt;
3612
3613 if (isPartialReduction()) {
3614 InstructionCost CondCost = 0;
3615 if (isConditional()) {
3617 auto *CondTy =
3619 CondCost = Ctx.TTI.getCmpSelInstrCost(Instruction::Select, VectorTy,
3620 CondTy, Pred, Ctx.CostKind);
3621 }
3622 return CondCost + Ctx.TTI.getPartialReductionCost(
3623 Opcode, ElementTy, nullptr, ElementTy, VF,
3624 TTI::PR_None, TTI::PR_None, {}, Ctx.CostKind,
3625 OptionalFMF);
3626 }
3627
3628 // TODO: Support any-of reductions.
3629 assert(
3631 ForceTargetInstructionCost.getNumOccurrences() > 0) &&
3632 "Any-of reduction not implemented in VPlan-based cost model currently.");
3633
3634 // Note that TTI should model the cost of moving result to the scalar register
3635 // and the BinOp cost in the getMinMaxReductionCost().
3638 return Ctx.TTI.getMinMaxReductionCost(Id, VectorTy, FMFs, Ctx.CostKind);
3639 }
3640
3641 // Note that TTI should model the cost of moving result to the scalar register
3642 // and the BinOp cost in the getArithmeticReductionCost().
3643 return Ctx.TTI.getArithmeticReductionCost(Opcode, VectorTy, OptionalFMF,
3644 Ctx.CostKind);
3645}
3646
3648 ExpressionTypes ExpressionType,
3649 ArrayRef<VPSingleDefRecipe *> ExpressionRecipes)
3650 : VPSingleDefRecipe(VPRecipeBase::VPExpressionSC, {},
3651 cast<VPReductionRecipe>(ExpressionRecipes.back())
3652 ->getChainOp()
3653 ->getScalarType()),
3654 ExpressionRecipes(ExpressionRecipes), ExpressionType(ExpressionType) {
3655 assert(!ExpressionRecipes.empty() && "Nothing to combine?");
3656 assert(
3657 none_of(ExpressionRecipes,
3658 [](VPSingleDefRecipe *R) { return R->mayHaveSideEffects(); }) &&
3659 "expression cannot contain recipes with side-effects");
3660
3661 // Maintain a copy of the expression recipes as a set of users.
3662 SmallPtrSet<VPUser *, 4> ExpressionRecipesAsSetOfUsers;
3663 for (auto *R : ExpressionRecipes)
3664 ExpressionRecipesAsSetOfUsers.insert(R);
3665
3666 // Recipes in the expression, except the last one, must only be used by
3667 // (other) recipes inside the expression. If there are other users, external
3668 // to the expression, use a clone of the recipe for external users.
3669 for (VPSingleDefRecipe *R : reverse(ExpressionRecipes)) {
3670 if (R != ExpressionRecipes.back() &&
3671 any_of(R->users(), [&ExpressionRecipesAsSetOfUsers](VPUser *U) {
3672 return !ExpressionRecipesAsSetOfUsers.contains(U);
3673 })) {
3674 // There are users outside of the expression. Clone the recipe and use the
3675 // clone those external users.
3676 VPSingleDefRecipe *CopyForExtUsers = R->clone();
3677 R->replaceUsesWithIf(CopyForExtUsers, [&ExpressionRecipesAsSetOfUsers](
3678 VPUser &U, unsigned) {
3679 return !ExpressionRecipesAsSetOfUsers.contains(&U);
3680 });
3681 CopyForExtUsers->insertBefore(R);
3682 }
3683 if (R->getParent())
3684 R->removeFromParent();
3685 }
3686
3687 // Internalize all external operands to the expression recipes. To do so,
3688 // create new temporary VPValues for all operands defined by a recipe outside
3689 // the expression. The original operands are added as operands of the
3690 // VPExpressionRecipe itself.
3691 for (auto *R : ExpressionRecipes) {
3692 for (const auto &[Idx, Op] : enumerate(R->operands())) {
3693 auto *Def = Op->getDefiningRecipe();
3694 if (Def && ExpressionRecipesAsSetOfUsers.contains(Def))
3695 continue;
3696 addOperand(Op);
3697 LiveInPlaceholders.push_back(new VPSymbolicValue(Op->getScalarType()));
3698 }
3699 }
3700
3701 // Replace each external operand with the first one created for it in
3702 // LiveInPlaceholders.
3703 for (auto *R : ExpressionRecipes)
3704 for (auto const &[LiveIn, Tmp] : zip(operands(), LiveInPlaceholders))
3705 R->replaceUsesOfWith(LiveIn, Tmp);
3706}
3707
3709 for (auto *R : ExpressionRecipes)
3710 // Since the list could contain duplicates, make sure the recipe hasn't
3711 // already been inserted.
3712 if (!R->getParent())
3713 R->insertBefore(this);
3714
3715 for (const auto &[Idx, Op] : enumerate(operands()))
3716 LiveInPlaceholders[Idx]->replaceAllUsesWith(Op);
3717
3718 replaceAllUsesWith(ExpressionRecipes.back());
3719 SmallVector<VPSingleDefRecipe *> DecomposedRecipes(ExpressionRecipes);
3720 ExpressionRecipes.clear();
3721 return DecomposedRecipes;
3722}
3723
3725 VPCostContext &Ctx) const {
3726 Type *RedTy = this->getScalarType();
3727 auto *SrcVecTy =
3729 unsigned Opcode = RecurrenceDescriptor::getOpcode(
3730 cast<VPReductionRecipe>(ExpressionRecipes.back())->getRecurrenceKind());
3731 switch (ExpressionType) {
3732 case ExpressionTypes::NegatedExtendedReduction:
3733 assert((Opcode == Instruction::Add || Opcode == Instruction::FAdd) &&
3734 "Unexpected opcode");
3735 Opcode = Opcode == Instruction::Add ? Instruction::Sub : Instruction::FSub;
3736 [[fallthrough]];
3737 case ExpressionTypes::ExtendedReduction: {
3738 auto *RedR = cast<VPReductionRecipe>(ExpressionRecipes.back());
3739 auto *ExtR = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3740
3741 if (RedR->isPartialReduction())
3742 return Ctx.TTI.getPartialReductionCost(
3743 Opcode, getOperand(0)->getScalarType(), nullptr, RedTy, VF,
3745 TargetTransformInfo::PR_None, std::nullopt, Ctx.CostKind,
3746 RedTy->isFloatingPointTy()
3747 ? std::optional{RedR->getFastMathFlagsOrNone()}
3748 : std::nullopt);
3749 else if (!RedTy->isFloatingPointTy())
3750 // TTI::getExtendedReductionCost only supports integer types.
3751 return Ctx.TTI.getExtendedReductionCost(
3752 Opcode, ExtR->getOpcode() == Instruction::ZExt, RedTy, SrcVecTy,
3753 std::nullopt, Ctx.CostKind);
3754 else
3756 }
3757 case ExpressionTypes::MulAccReduction:
3758 return Ctx.TTI.getMulAccReductionCost(false, Opcode, RedTy, SrcVecTy,
3759 Ctx.CostKind);
3760
3761 case ExpressionTypes::ExtNegatedMulAccReduction:
3762 switch (Opcode) {
3763 case Instruction::Add:
3764 Opcode = Instruction::Sub;
3765 break;
3766 case Instruction::FAdd:
3767 Opcode = Instruction::FSub;
3768 break;
3769 default:
3770 llvm_unreachable("Unsupported opcode for ExtNegatedMulAccReduction");
3771 }
3772 [[fallthrough]];
3773 case ExpressionTypes::ExtMulAccReduction: {
3774 auto *RedR = cast<VPReductionRecipe>(ExpressionRecipes.back());
3775 if (RedR->isPartialReduction()) {
3776 auto *Ext0R = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3777 auto *Ext1R = cast<VPWidenCastRecipe>(ExpressionRecipes[1]);
3778 auto *Mul = cast<VPWidenRecipe>(ExpressionRecipes[2]);
3779 return Ctx.TTI.getPartialReductionCost(
3780 Opcode, getOperand(0)->getScalarType(),
3781 getOperand(1)->getScalarType(), RedTy, VF,
3783 Ext0R->getOpcode()),
3785 Ext1R->getOpcode()),
3786 Mul->getOpcode(), Ctx.CostKind,
3787 RedTy->isFloatingPointTy()
3788 ? std::optional{RedR->getFastMathFlagsOrNone()}
3789 : std::nullopt);
3790 }
3791 assert(Opcode != Instruction::FSub && "Only integer types are supported");
3792 return Ctx.TTI.getMulAccReductionCost(
3793 cast<VPWidenCastRecipe>(ExpressionRecipes.front())->getOpcode() ==
3794 Instruction::ZExt,
3795 Opcode, RedTy, SrcVecTy, Ctx.CostKind);
3796 }
3797 }
3798 llvm_unreachable("Unknown VPExpressionRecipe::ExpressionTypes enum");
3799}
3800
3802 return any_of(ExpressionRecipes, [](VPSingleDefRecipe *R) {
3803 return R->mayReadFromMemory() || R->mayWriteToMemory();
3804 });
3805}
3806
3808 assert(
3809 none_of(ExpressionRecipes,
3810 [](VPSingleDefRecipe *R) { return R->mayHaveSideEffects(); }) &&
3811 "expression cannot contain recipes with side-effects");
3812 return false;
3813}
3814
3816 auto *RR = dyn_cast<VPReductionRecipe>(ExpressionRecipes.back());
3817 return RR && !RR->isPartialReduction();
3818}
3819
3820#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3821
3823 VPSlotTracker &SlotTracker) const {
3824 O << Indent << "EXPRESSION ";
3826 O << " = ";
3827 auto *Red = cast<VPReductionRecipe>(ExpressionRecipes.back());
3828 unsigned Opcode = RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind());
3829 VPValue *Mask = getOperand(getNumOperands() - 1);
3830 VPValue *EVL =
3832 ? getOperand(getNumOperands() - (Red->isConditional() ? 2 : 1))
3833 : nullptr;
3834 VPValue *RdxStart = getOperand(
3835 getNumOperands() - (Red->isConditional() ? 2 : 1) - (EVL ? 1 : 0));
3836 auto PrintEVLAndMask = [&]() {
3837 if (EVL) {
3838 O << ", ";
3839 EVL->printAsOperand(O, SlotTracker);
3840 }
3841 if (Red->isConditional()) {
3842 O << ", ";
3843 Mask->printAsOperand(O, SlotTracker);
3844 }
3845 };
3846
3847 switch (ExpressionType) {
3848 case ExpressionTypes::NegatedExtendedReduction:
3849 case ExpressionTypes::ExtendedReduction: {
3850 bool Negated = ExpressionType == ExpressionTypes::NegatedExtendedReduction;
3852 O << " + " << (Red->isPartialReduction() ? "partial." : "") << "reduce.";
3853 O << Instruction::getOpcodeName(Opcode) << " (";
3854 if (Negated)
3855 O << (Opcode == Instruction::Add ? "sub (0, " : "fneg(");
3857 if (Negated)
3858 O << ")";
3859 Red->printFlags(O);
3860
3861 auto *Ext0 = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3862 O << Instruction::getOpcodeName(Ext0->getOpcode()) << " to "
3863 << *Ext0->getScalarType();
3864 PrintEVLAndMask();
3865 O << ")";
3866 break;
3867 }
3868 case ExpressionTypes::ExtNegatedMulAccReduction: {
3869 RdxStart->printAsOperand(O, SlotTracker);
3870 O << " + " << (Red->isPartialReduction() ? "partial." : "") << "reduce.";
3872 RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind()))
3873 << " (sub (0, mul";
3874 auto *Mul = cast<VPWidenRecipe>(ExpressionRecipes[2]);
3875 Mul->printFlags(O);
3876 O << "(";
3878 auto *Ext0 = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3879 O << " " << Instruction::getOpcodeName(Ext0->getOpcode()) << " to "
3880 << *Ext0->getScalarType() << "), (";
3882 auto *Ext1 = cast<VPWidenCastRecipe>(ExpressionRecipes[1]);
3883 O << " " << Instruction::getOpcodeName(Ext1->getOpcode()) << " to "
3884 << *Ext1->getScalarType() << ")";
3885 PrintEVLAndMask();
3886 O << "))";
3887 break;
3888 }
3889 case ExpressionTypes::MulAccReduction:
3890 case ExpressionTypes::ExtMulAccReduction: {
3891 RdxStart->printAsOperand(O, SlotTracker);
3892 O << " + " << (Red->isPartialReduction() ? "partial." : "") << "reduce.";
3894 RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind()))
3895 << " (";
3896 O << "mul";
3897 bool IsExtended = ExpressionType == ExpressionTypes::ExtMulAccReduction;
3898 auto *Mul = cast<VPWidenRecipe>(IsExtended ? ExpressionRecipes[2]
3899 : ExpressionRecipes[0]);
3900 Mul->printFlags(O);
3901 if (IsExtended)
3902 O << "(";
3904 if (IsExtended) {
3905 auto *Ext0 = cast<VPWidenCastRecipe>(ExpressionRecipes[0]);
3906 O << " " << Instruction::getOpcodeName(Ext0->getOpcode()) << " to "
3907 << *Ext0->getScalarType() << "), (";
3908 } else {
3909 O << ", ";
3910 }
3912 if (IsExtended) {
3913 auto *Ext1 = cast<VPWidenCastRecipe>(ExpressionRecipes[1]);
3914 O << " " << Instruction::getOpcodeName(Ext1->getOpcode()) << " to "
3915 << *Ext1->getScalarType() << ")";
3916 }
3917 PrintEVLAndMask();
3918 O << ")";
3919 break;
3920 }
3921 }
3922}
3923
3925 VPSlotTracker &SlotTracker) const {
3926 if (isPartialReduction())
3927 O << Indent << "PARTIAL-REDUCE ";
3928 else
3929 O << Indent << "REDUCE ";
3931 O << " = ";
3933 O << " +";
3934 printFlags(O);
3935 O << " reduce.";
3937 O << " (";
3939 if (isConditional()) {
3940 O << ", ";
3942 }
3943 O << ")";
3944}
3945
3947 VPSlotTracker &SlotTracker) const {
3948 if (isPartialReduction())
3949 O << Indent << "PARTIAL-REDUCE ";
3950 else
3951 O << Indent << "REDUCE ";
3953 O << " = ";
3955 O << " +";
3956 printFlags(O);
3957 O << " vp.reduce."
3960 << " (";
3962 O << ", ";
3964 if (isConditional()) {
3965 O << ", ";
3967 }
3968 O << ")";
3969}
3970
3971#endif
3972
3974 assert(IsSingleScalar &&
3975 "VPReplicateRecipes must be unrolled before ::execute");
3976 auto *Instr = getUnderlyingInstr();
3977 Instruction *Cloned = Instr->clone();
3978 Type *ResultTy = getScalarType();
3979 if (!ResultTy->isVoidTy()) {
3980 Cloned->setName(Instr->getName() + ".cloned");
3981 // The operands of the replicate recipe may have been narrowed, resulting in
3982 // a narrower result type. Update the type of the cloned instruction to the
3983 // correct type.
3984 if (ResultTy != Cloned->getType())
3985 Cloned->mutateType(ResultTy);
3986 }
3987
3988 applyFlags(*Cloned);
3989 applyMetadata(*Cloned);
3990
3991 if (hasPredicate())
3992 cast<CmpInst>(Cloned)->setPredicate(getPredicate());
3993
3994 // Replace the operands of the cloned instructions with their scalar
3995 // equivalents in the new loop.
3996 for (const auto &[Idx, V] : enumerate(operands()))
3997 Cloned->setOperand(Idx, State.get(V, true));
3998
3999 // Place the cloned scalar in the new loop.
4000 State.Builder.Insert(Cloned);
4001
4002 State.set(this, Cloned, true);
4003
4004 // If we just cloned a new assumption, add it the assumption cache.
4005 if (auto *II = dyn_cast<AssumeInst>(Cloned))
4006 State.AC->registerAssumption(II);
4007}
4008
4009/// Returns a SCEV expression for \p Ptr if it is a pointer computation for
4010/// which the legacy cost model computes a SCEV expression when computing the
4011/// address cost. Computing SCEVs for VPValues is incomplete and returns
4012/// SCEVCouldNotCompute in cases the legacy cost model can compute SCEVs. In
4013/// those cases we fall back to the legacy cost model. Otherwise return nullptr.
4014static const SCEV *getAddressAccessSCEV(const VPValue *Ptr,
4016 const Loop *L) {
4017 const SCEV *Addr = vputils::getSCEVExprForVPValue(Ptr, PSE, L);
4018 if (isa<SCEVCouldNotCompute>(Addr))
4019 return Addr;
4020
4021 return vputils::isAddressSCEVForCost(Addr, *PSE.getSE(), L) ? Addr : nullptr;
4022}
4023
4025 VPCostContext &Ctx) const {
4027 // VPReplicateRecipe may be cloned as part of an existing VPlan-to-VPlan
4028 // transform, avoid computing their cost multiple times for now.
4029 Ctx.SkipCostComputation.insert(UI);
4030
4031 if (VF.isScalable() && !isSingleScalar())
4033
4034 switch (UI->getOpcode()) {
4035 case Instruction::Alloca:
4036 if (VF.isScalable())
4038 return Ctx.TTI.getArithmeticInstrCost(Instruction::Mul,
4039 this->getScalarType(), Ctx.CostKind);
4040 case Instruction::GetElementPtr:
4041 // We mark this instruction as zero-cost because the cost of GEPs in
4042 // vectorized code depends on whether the corresponding memory instruction
4043 // is scalarized or not. Therefore, we handle GEPs with the memory
4044 // instruction cost.
4045 return 0;
4046 case Instruction::Call: {
4047 auto *CalledFn =
4049 Type *ResultTy = this->getScalarType();
4050 return computeCallCost(CalledFn, ResultTy, drop_end(operands()),
4051 isSingleScalar(), VF, Ctx);
4052 }
4053 case Instruction::Add:
4054 case Instruction::Sub:
4055 case Instruction::FAdd:
4056 case Instruction::FSub:
4057 case Instruction::Mul:
4058 case Instruction::FMul:
4059 case Instruction::FDiv:
4060 case Instruction::FRem:
4061 case Instruction::Shl:
4062 case Instruction::LShr:
4063 case Instruction::AShr:
4064 case Instruction::And:
4065 case Instruction::Or:
4066 case Instruction::Xor:
4067 case Instruction::ICmp:
4068 case Instruction::FCmp:
4070 Ctx) *
4071 (isSingleScalar() ? 1 : VF.getFixedValue());
4072 case Instruction::SDiv:
4073 case Instruction::UDiv:
4074 case Instruction::SRem:
4075 case Instruction::URem: {
4076 InstructionCost ScalarCost =
4078 if (isSingleScalar())
4079 return ScalarCost;
4080
4081 // If any of the operands is from a different replicate region and has its
4082 // cost skipped, it may have been forced to scalar. Fall back to legacy cost
4083 // model to avoid cost mis-match.
4084 if (any_of(operands(), [&Ctx, VF](VPValue *Op) {
4085 auto *PredR = dyn_cast<VPPredInstPHIRecipe>(Op);
4086 if (!PredR)
4087 return false;
4088 return Ctx.skipCostComputation(
4090 PredR->getOperand(0)->getUnderlyingValue()),
4091 VF.isVector());
4092 }))
4093 break;
4094
4095 ScalarCost = ScalarCost * VF.getFixedValue() +
4096 Ctx.getScalarizationOverhead(this->getScalarType(),
4097 to_vector(operands()), VF);
4098 // If the recipe is not predicated (i.e. not in a replicate region), return
4099 // the scalar cost. Otherwise handle predicated cost.
4100 if (!getRegion()->isReplicator())
4101 return ScalarCost;
4102
4103 // Account for the phi nodes that we will create.
4104 ScalarCost += VF.getFixedValue() *
4105 Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
4106 // Scale the cost by the probability of executing the predicated blocks.
4107 // This assumes the predicated block for each vector lane is equally
4108 // likely.
4109 ScalarCost /= Ctx.getCostDivisor(
4110 getRegion()->getEntryBranchOnMask()->getExecutionFrequency());
4111 return ScalarCost;
4112 }
4113 case Instruction::Load:
4114 case Instruction::Store: {
4115 bool IsLoad = UI->getOpcode() == Instruction::Load;
4116 const VPValue *PtrOp = getOperand(!IsLoad);
4117 const SCEV *PtrSCEV = getAddressAccessSCEV(PtrOp, Ctx.PSE, Ctx.L);
4119 break;
4120
4121 Type *ValTy = (IsLoad ? this : getOperand(0))->getScalarType();
4122 Type *ScalarPtrTy = PtrOp->getScalarType();
4123 const Align Alignment = getLoadStoreAlignment(UI);
4124 unsigned AS = cast<PointerType>(ScalarPtrTy)->getAddressSpace();
4126 bool PreferVectorizedAddressing = Ctx.TTI.prefersVectorizedAddressing();
4127 bool UsedByLoadStoreAddress =
4128 !PreferVectorizedAddressing && vputils::isUsedByLoadStoreAddress(this);
4129 InstructionCost ScalarMemOpCost = Ctx.TTI.getMemoryOpCost(
4130 UI->getOpcode(), ValTy, Alignment, AS, Ctx.CostKind, OpInfo,
4131 UsedByLoadStoreAddress ? UI : nullptr);
4132
4133 Type *PtrTy = isSingleScalar() ? ScalarPtrTy : toVectorTy(ScalarPtrTy, VF);
4134 InstructionCost ScalarCost =
4135 ScalarMemOpCost +
4136 Ctx.TTI.getAddressComputationCost(
4137 PtrTy, UsedByLoadStoreAddress ? nullptr : Ctx.PSE.getSE(), PtrSCEV,
4138 Ctx.CostKind);
4139 if (isSingleScalar())
4140 return ScalarCost;
4141
4142 SmallVector<const VPValue *> OpsToScalarize;
4143 Type *ResultTy = Type::getVoidTy(PtrTy->getContext());
4144 // Set ResultTy and OpsToScalarize, if scalarization is needed. Currently we
4145 // don't assign scalarization overhead in general, if the target prefers
4146 // vectorized addressing or the loaded value is used as part of an address
4147 // of another load or store.
4148 if (!UsedByLoadStoreAddress) {
4149 bool EfficientVectorLoadStore =
4150 Ctx.TTI.supportsEfficientVectorElementLoadStore();
4151 if (!(IsLoad && !PreferVectorizedAddressing) &&
4152 !(!IsLoad && EfficientVectorLoadStore))
4153 append_range(OpsToScalarize, operands());
4154
4155 if (!EfficientVectorLoadStore)
4156 ResultTy = this->getScalarType();
4157 }
4158
4160 IsLoad ? TTI::VectorInstrContext::Load : TTI::VectorInstrContext::Store;
4162 (ScalarCost * VF.getFixedValue()) +
4163 Ctx.getScalarizationOverhead(ResultTy, OpsToScalarize, VF, VIC, true);
4164
4165 const VPRegionBlock *ParentRegion = getRegion();
4166 if (ParentRegion && ParentRegion->isReplicator()) {
4167 if (!PtrSCEV)
4168 break;
4169 Cost /= Ctx.getCostDivisor(
4170 ParentRegion->getEntryBranchOnMask()->getExecutionFrequency());
4171 Cost += Ctx.TTI.getCFInstrCost(Instruction::CondBr, Ctx.CostKind);
4172
4173 auto *VecI1Ty = VectorType::get(
4174 IntegerType::getInt1Ty(Ctx.L->getHeader()->getContext()), VF);
4175 Cost += Ctx.TTI.getScalarizationOverhead(
4176 VecI1Ty, APInt::getAllOnes(VF.getFixedValue()),
4177 /*Insert=*/false, /*Extract=*/true, Ctx.CostKind);
4178
4179 if (Ctx.useEmulatedMaskMemRefHack(this, VF)) {
4180 // Artificially setting to a high enough value to practically disable
4181 // vectorization with such operations.
4182 return 3000000;
4183 }
4184 }
4185 return Cost;
4186 }
4187 case Instruction::SExt:
4188 case Instruction::ZExt:
4189 case Instruction::FPToUI:
4190 case Instruction::FPToSI:
4191 case Instruction::FPExt:
4192 case Instruction::PtrToInt:
4193 case Instruction::PtrToAddr:
4194 case Instruction::IntToPtr:
4195 case Instruction::SIToFP:
4196 case Instruction::UIToFP:
4197 case Instruction::Trunc:
4198 case Instruction::FPTrunc:
4199 case Instruction::Select:
4200 case Instruction::AddrSpaceCast: {
4202 Ctx) *
4203 (isSingleScalar() ? 1 : VF.getFixedValue());
4204 }
4205 case Instruction::ExtractValue:
4206 case Instruction::InsertValue:
4207 return Ctx.TTI.getInsertExtractValueCost(getOpcode(), Ctx.CostKind);
4208 }
4209
4210 return Ctx.getLegacyCost(UI, VF);
4211}
4212
4214 Function *CalledFn, Type *ResultTy, ArrayRef<const VPValue *> ArgOps,
4215 bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx) {
4217 ArgOps, [&](const VPValue *Op) { return Op->getScalarType(); });
4218
4219 Intrinsic::ID IntrinID = CalledFn->getIntrinsicID();
4220 auto GetIntrinsicCost = [&] {
4221 if (!IntrinID)
4223 return Ctx.TTI.getIntrinsicInstrCost(
4224 IntrinsicCostAttributes(IntrinID, ResultTy, Tys), Ctx.CostKind);
4225 };
4226
4227 if (IntrinID && VPCostContext::isFreeScalarIntrinsic(IntrinID)) {
4228 assert(GetIntrinsicCost() == 0 && "scalarizing intrinsic should be free");
4229 return 0;
4230 }
4231
4232 InstructionCost ScalarCallCost =
4233 Ctx.TTI.getCallInstrCost(CalledFn, ResultTy, Tys, Ctx.CostKind);
4234 if (IsSingleScalar) {
4235 ScalarCallCost = std::min(ScalarCallCost, GetIntrinsicCost());
4236 return ScalarCallCost;
4237 }
4238
4239 // Scalarization overhead is undefined for scalable VFs.
4240 if (VF.isScalable())
4242
4243 return ScalarCallCost * VF.getFixedValue() +
4244 Ctx.getScalarizationOverhead(ResultTy, ArgOps, VF);
4245}
4246
4247#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4249 VPSlotTracker &SlotTracker) const {
4250 O << Indent << (IsSingleScalar ? "CLONE " : "REPLICATE ");
4251
4252 if (!getScalarType()->isVoidTy()) {
4254 O << " = ";
4255 }
4256 if (auto *CB = dyn_cast<CallBase>(getUnderlyingInstr())) {
4257 O << "call";
4258 printFlags(O);
4259 O << "@" << CB->getCalledFunction()->getName() << "(";
4261 Op->printAsOperand(O, SlotTracker);
4262 });
4263 O << ")";
4264 } else {
4266 printFlags(O);
4268 }
4269
4270 // Find if the recipe is used by a widened recipe via an intervening
4271 // VPPredInstPHIRecipe. In this case, also pack the scalar values in a vector.
4272 if (any_of(users(), [](const VPUser *U) {
4273 if (auto *PredR = dyn_cast<VPPredInstPHIRecipe>(U))
4274 return !vputils::onlyScalarValuesUsed(PredR);
4275 return false;
4276 }))
4277 O << " (S->V)";
4278}
4279#endif
4280
4282 llvm_unreachable("recipe must be removed when dissolving replicate region");
4283}
4284
4286 VPCostContext &Ctx) const {
4287 // The legacy cost model doesn't assign costs to branches for individual
4288 // replicate regions. Match the current behavior in the VPlan cost model for
4289 // now.
4290 return 0;
4291}
4292
4294 llvm_unreachable("recipe must be removed when dissolving replicate region");
4295}
4296
4297#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4299 VPSlotTracker &SlotTracker) const {
4300 O << Indent << "PHI-PREDICATED-INSTRUCTION ";
4302 O << " = ";
4304}
4305#endif
4306
4308const VPRecipeBase *VPWidenLoadRecipe::getAsRecipe() const { return this; }
4309
4312
4314const VPRecipeBase *VPWidenStoreRecipe::getAsRecipe() const { return this; }
4315
4318
4320 VPCostContext &Ctx) const {
4321 const VPRecipeBase *R = getAsRecipe();
4323 Type *ScalarTy = IsLoad ? cast<VPSingleDefRecipe>(R)->getScalarType()
4324 : R->getOperand(1)->getScalarType();
4325 Type *Ty = toVectorTy(ScalarTy, VF);
4326 unsigned AS =
4327 cast<PointerType>(getAddr()->getScalarType())->getAddressSpace();
4328 unsigned Opcode = IsLoad ? Instruction::Load : Instruction::Store;
4329
4330 if (!Consecutive) {
4331 // TODO: Using the original IR may not be accurate.
4332 // Currently, ARM will use the underlying IR to calculate gather/scatter
4333 // instruction cost.
4334 Type *PtrTy = getAddr()->getScalarType();
4335 const Value *Ptr = getAddr()->getUnderlyingValue();
4336
4337 // If the address value is uniform across all lanes, then the address can be
4338 // calculated with scalar type and broadcast.
4340 PtrTy = toVectorTy(PtrTy, VF);
4341
4342 unsigned IID = isa<VPWidenLoadRecipe>(R) ? Intrinsic::masked_gather
4343 : isa<VPWidenStoreRecipe>(R) ? Intrinsic::masked_scatter
4344 : isa<VPWidenLoadEVLRecipe>(R) ? Intrinsic::vp_gather
4345 : Intrinsic::vp_scatter;
4346 return Ctx.TTI.getAddressComputationCost(PtrTy, nullptr, nullptr,
4347 Ctx.CostKind) +
4348 Ctx.TTI.getMemIntrinsicInstrCost(
4350 &Ingredient),
4351 Ctx.CostKind);
4352 }
4353
4355 if (IsMasked) {
4356 unsigned IID = isa<VPWidenLoadRecipe>(R) ? Intrinsic::masked_load
4357 : Intrinsic::masked_store;
4358 Cost += Ctx.TTI.getMemIntrinsicInstrCost(
4359 MemIntrinsicCostAttributes(IID, Ty, Alignment, AS), Ctx.CostKind);
4360 } else {
4361 TTI::OperandValueInfo OpInfo = Ctx.getOperandInfo(
4363 : R->getOperand(1));
4364 Cost += Ctx.TTI.getMemoryOpCost(Opcode, Ty, Alignment, AS, Ctx.CostKind,
4365 OpInfo, &Ingredient);
4366 }
4367 return Cost;
4368}
4369
4371 Type *ScalarDataTy = getScalarType();
4372 auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
4373 bool CreateGather = !isConsecutive();
4374
4375 auto &Builder = State.Builder;
4376 Value *Mask = nullptr;
4377 if (auto *VPMask = getMask())
4378 Mask = State.get(VPMask);
4379
4380 Value *Addr = State.get(getAddr(), /*NeedsSingleScalar=*/!CreateGather);
4381 Value *NewLI;
4382 if (CreateGather) {
4383 NewLI = Builder.CreateMaskedGather(DataTy, Addr, Alignment, Mask, nullptr,
4384 "wide.masked.gather");
4385 } else if (Mask) {
4386 NewLI =
4387 Builder.CreateMaskedLoad(DataTy, Addr, Alignment, Mask,
4388 PoisonValue::get(DataTy), "wide.masked.load");
4389 } else {
4390 NewLI = Builder.CreateAlignedLoad(DataTy, Addr, Alignment, "wide.load");
4391 }
4393 State.set(this, NewLI);
4394}
4395
4396#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4398 VPSlotTracker &SlotTracker) const {
4399 O << Indent << "WIDEN ";
4401 O << " = load ";
4403}
4404#endif
4405
4407 Type *ScalarDataTy = getScalarType();
4408 auto *DataTy = VectorType::get(ScalarDataTy, State.VF);
4409 bool CreateGather = !isConsecutive();
4410
4411 auto &Builder = State.Builder;
4412 CallInst *NewLI;
4413 Value *EVL = State.get(getEVL(), VPLane(0));
4414 Value *Addr = State.get(getAddr(), !CreateGather);
4415 Value *Mask = nullptr;
4416 if (VPValue *VPMask = getMask())
4417 Mask = State.get(VPMask);
4418 else
4419 Mask = Builder.CreateVectorSplat(State.VF, Builder.getTrue());
4420
4421 if (CreateGather) {
4422 NewLI = Builder.CreateIntrinsicWithoutFolding(DataTy, Intrinsic::vp_gather,
4423 {Addr, Mask, EVL}, nullptr,
4424 "wide.masked.gather");
4425 } else {
4426 NewLI = Builder.CreateIntrinsicWithoutFolding(
4427 DataTy, Intrinsic::vp_load, {Addr, Mask, EVL}, nullptr, "vp.op.load");
4428 }
4429 NewLI->addParamAttr(
4431 applyMetadata(*NewLI);
4432 State.set(this, NewLI);
4433}
4434
4436 VPCostContext &Ctx) const {
4437 if (!Consecutive || IsMasked)
4438 return VPWidenMemoryRecipe::computeCost(VF, Ctx);
4439
4440 // We need to use the getMemIntrinsicInstrCost() instead of getMemoryOpCost()
4441 // here because the EVL recipes using EVL to replace the tail mask. But in the
4442 // legacy model, it will always calculate the cost of mask.
4443 // TODO: Using getMemoryOpCost() instead of getMemIntrinsicInstrCost when we
4444 // don't need to compare to the legacy cost model.
4445 Type *Ty = toVectorTy(getScalarType(), VF);
4446 unsigned AS =
4447 cast<PointerType>(getAddr()->getScalarType())->getAddressSpace();
4448 return Ctx.TTI.getMemIntrinsicInstrCost(
4449 MemIntrinsicCostAttributes(Intrinsic::vp_load, Ty, Alignment, AS),
4450 Ctx.CostKind);
4451}
4452
4453#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4455 VPSlotTracker &SlotTracker) const {
4456 O << Indent << "WIDEN ";
4458 O << " = vp.load ";
4460}
4461#endif
4462
4464 VPValue *StoredVPValue = getStoredValue();
4465 bool CreateScatter = !isConsecutive();
4466
4467 auto &Builder = State.Builder;
4468
4469 Value *Mask = nullptr;
4470 if (auto *VPMask = getMask())
4471 Mask = State.get(VPMask);
4472
4473 Value *StoredVal = State.get(StoredVPValue);
4474 Value *Addr = State.get(getAddr(), /*NeedsSingleScalar=*/!CreateScatter);
4475 Instruction *NewSI = nullptr;
4476 if (CreateScatter)
4477 NewSI = Builder.CreateMaskedScatter(StoredVal, Addr, Alignment, Mask);
4478 else if (Mask)
4479 NewSI = Builder.CreateMaskedStore(StoredVal, Addr, Alignment, Mask);
4480 else
4481 NewSI = Builder.CreateAlignedStore(StoredVal, Addr, Alignment);
4482 applyMetadata(*NewSI);
4483}
4484
4485#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4487 VPSlotTracker &SlotTracker) const {
4488 O << Indent << "WIDEN store ";
4490}
4491#endif
4492
4494 VPValue *StoredValue = getStoredValue();
4495 bool CreateScatter = !isConsecutive();
4496
4497 auto &Builder = State.Builder;
4498
4499 CallInst *NewSI = nullptr;
4500 Value *StoredVal = State.get(StoredValue);
4501 Value *EVL = State.get(getEVL(), VPLane(0));
4502 Value *Mask = nullptr;
4503 if (VPValue *VPMask = getMask())
4504 Mask = State.get(VPMask);
4505 else
4506 Mask = Builder.CreateVectorSplat(State.VF, Builder.getTrue());
4507
4508 Value *Addr = State.get(getAddr(), !CreateScatter);
4509 if (CreateScatter) {
4510 NewSI = Builder.CreateIntrinsicWithoutFolding(
4511 Type::getVoidTy(EVL->getContext()), Intrinsic::vp_scatter,
4512 {StoredVal, Addr, Mask, EVL});
4513 } else {
4514 NewSI = Builder.CreateIntrinsicWithoutFolding(
4515 Type::getVoidTy(EVL->getContext()), Intrinsic::vp_store,
4516 {StoredVal, Addr, Mask, EVL});
4517 }
4518 NewSI->addParamAttr(
4520 applyMetadata(*NewSI);
4521}
4522
4524 VPCostContext &Ctx) const {
4525 if (!Consecutive || IsMasked)
4526 return VPWidenMemoryRecipe::computeCost(VF, Ctx);
4527
4528 // We need to use the getMemIntrinsicInstrCost() instead of getMemoryOpCost()
4529 // here because the EVL recipes using EVL to replace the tail mask. But in the
4530 // legacy model, it will always calculate the cost of mask.
4531 // TODO: Using getMemoryOpCost() instead of getMemIntrinsicInstrCost when we
4532 // don't need to compare to the legacy cost model.
4533 Type *Ty = toVectorTy(getStoredValue()->getScalarType(), VF);
4534 unsigned AS =
4535 cast<PointerType>(getAddr()->getScalarType())->getAddressSpace();
4536 return Ctx.TTI.getMemIntrinsicInstrCost(
4537 MemIntrinsicCostAttributes(Intrinsic::vp_store, Ty, Alignment, AS),
4538 Ctx.CostKind);
4539}
4540
4541#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4543 VPSlotTracker &SlotTracker) const {
4544 O << Indent << "WIDEN vp.store ";
4546}
4547#endif
4548
4550 VectorType *DstVTy, const DataLayout &DL) {
4551 // Verify that V is a vector type with same number of elements as DstVTy.
4552 auto VF = DstVTy->getElementCount();
4553 auto *SrcVecTy = cast<VectorType>(V->getType());
4554 assert(VF == SrcVecTy->getElementCount() && "Vector dimensions do not match");
4555 Type *SrcElemTy = SrcVecTy->getElementType();
4556 Type *DstElemTy = DstVTy->getElementType();
4557 assert((DL.getTypeSizeInBits(SrcElemTy) == DL.getTypeSizeInBits(DstElemTy)) &&
4558 "Vector elements must have same size");
4559
4560 // Do a direct cast if element types are castable.
4561 if (CastInst::isBitOrNoopPointerCastable(SrcElemTy, DstElemTy, DL)) {
4562 return Builder.CreateBitOrPointerCast(V, DstVTy);
4563 }
4564 // V cannot be directly casted to desired vector type.
4565 // May happen when V is a floating point vector but DstVTy is a vector of
4566 // pointers or vice-versa. Handle this using a two-step bitcast using an
4567 // intermediate Integer type for the bitcast i.e. Ptr <-> Int <-> Float.
4568 assert((DstElemTy->isPointerTy() != SrcElemTy->isPointerTy()) &&
4569 "Only one type should be a pointer type");
4570 assert((DstElemTy->isFloatingPointTy() != SrcElemTy->isFloatingPointTy()) &&
4571 "Only one type should be a floating point type");
4572 Type *IntTy =
4573 IntegerType::getIntNTy(V->getContext(), DL.getTypeSizeInBits(SrcElemTy));
4574 auto *VecIntTy = VectorType::get(IntTy, VF);
4575 Value *CastVal = Builder.CreateBitOrPointerCast(V, VecIntTy);
4576 return Builder.CreateBitOrPointerCast(CastVal, DstVTy);
4577}
4578
4579/// Return a vector containing interleaved elements from multiple
4580/// smaller input vectors.
4582 const Twine &Name) {
4583 unsigned Factor = Vals.size();
4584 assert(Factor > 1 && "Tried to interleave invalid number of vectors");
4585
4586 VectorType *VecTy = cast<VectorType>(Vals[0]->getType());
4587#ifndef NDEBUG
4588 for (Value *Val : Vals)
4589 assert(Val->getType() == VecTy && "Tried to interleave mismatched types");
4590#endif
4591
4592 // Scalable vectors cannot use arbitrary shufflevectors (only splats), so
4593 // must use intrinsics to interleave.
4594 if (VecTy->isScalableTy()) {
4595 assert(Factor <= 8 && "Unsupported interleave factor for scalable vectors");
4596 return Builder.CreateVectorInterleave(Vals, Name);
4597 }
4598
4599 // Fixed length. Start by concatenating all vectors into a wide vector.
4600 Value *WideVec = concatenateVectors(Builder, Vals);
4601
4602 // Interleave the elements into the wide vector.
4603 const unsigned NumElts = VecTy->getElementCount().getFixedValue();
4604 return Builder.CreateShuffleVector(
4605 WideVec, createInterleaveMask(NumElts, Factor), Name);
4606}
4607
4608// Try to vectorize the interleave group that \p Instr belongs to.
4609//
4610// E.g. Translate following interleaved load group (factor = 3):
4611// for (i = 0; i < N; i+=3) {
4612// R = Pic[i]; // Member of index 0
4613// G = Pic[i+1]; // Member of index 1
4614// B = Pic[i+2]; // Member of index 2
4615// ... // do something to R, G, B
4616// }
4617// To:
4618// %wide.vec = load <12 x i32> ; Read 4 tuples of R,G,B
4619// %R.vec = shuffle %wide.vec, poison, <0, 3, 6, 9> ; R elements
4620// %G.vec = shuffle %wide.vec, poison, <1, 4, 7, 10> ; G elements
4621// %B.vec = shuffle %wide.vec, poison, <2, 5, 8, 11> ; B elements
4622//
4623// Or translate following interleaved store group (factor = 3):
4624// for (i = 0; i < N; i+=3) {
4625// ... do something to R, G, B
4626// Pic[i] = R; // Member of index 0
4627// Pic[i+1] = G; // Member of index 1
4628// Pic[i+2] = B; // Member of index 2
4629// }
4630// To:
4631// %R_G.vec = shuffle %R.vec, %G.vec, <0, 1, 2, ..., 7>
4632// %B_U.vec = shuffle %B.vec, poison, <0, 1, 2, 3, u, u, u, u>
4633// %interleaved.vec = shuffle %R_G.vec, %B_U.vec,
4634// <0, 4, 8, 1, 5, 9, 2, 6, 10, 3, 7, 11> ; Interleave R,G,B elements
4635// store <12 x i32> %interleaved.vec ; Write 4 tuples of R,G,B
4637 assert((!needsMaskForGaps() || !State.VF.isScalable()) &&
4638 "Masking gaps for scalable vectors is not yet supported.");
4640 Instruction *Instr = Group->getInsertPos();
4641
4642 // Prepare for the vector type of the interleaved load/store.
4643 Type *ScalarTy = getLoadStoreType(Instr);
4644 unsigned InterleaveFactor = Group->getFactor();
4645 auto *VecTy = VectorType::get(ScalarTy, State.VF * InterleaveFactor);
4646
4647 VPValue *BlockInMask = getMask();
4648 VPValue *Addr = getAddr();
4649 Value *ResAddr = State.get(Addr, VPLane(0));
4650
4651 auto CreateGroupMask = [&BlockInMask, &State,
4652 &InterleaveFactor](Value *MaskForGaps) -> Value * {
4653 if (State.VF.isScalable()) {
4654 assert(!MaskForGaps && "Interleaved groups with gaps are not supported.");
4655 assert(InterleaveFactor <= 8 &&
4656 "Unsupported deinterleave factor for scalable vectors");
4657 auto *ResBlockInMask = State.get(BlockInMask);
4658 SmallVector<Value *> Ops(InterleaveFactor, ResBlockInMask);
4659 return interleaveVectors(State.Builder, Ops, "interleaved.mask");
4660 }
4661
4662 if (!BlockInMask)
4663 return MaskForGaps;
4664
4665 Value *ResBlockInMask = State.get(BlockInMask);
4666 Value *ShuffledMask = State.Builder.CreateShuffleVector(
4667 ResBlockInMask,
4668 createReplicatedMask(InterleaveFactor, State.VF.getFixedValue()),
4669 "interleaved.mask");
4670 return MaskForGaps ? State.Builder.CreateBinOp(Instruction::And,
4671 ShuffledMask, MaskForGaps)
4672 : ShuffledMask;
4673 };
4674
4675 const DataLayout &DL = Instr->getDataLayout();
4676 // Vectorize the interleaved load group.
4677 if (isa<LoadInst>(Instr)) {
4678 Value *MaskForGaps = nullptr;
4679 if (needsMaskForGaps()) {
4680 MaskForGaps =
4681 createBitMaskForGaps(State.Builder, State.VF.getFixedValue(), *Group);
4682 assert(MaskForGaps && "Mask for Gaps is required but it is null");
4683 }
4684
4685 Instruction *NewLoad;
4686 if (BlockInMask || MaskForGaps) {
4687 Value *GroupMask = CreateGroupMask(MaskForGaps);
4688 Value *PoisonVec = PoisonValue::get(VecTy);
4689 NewLoad = State.Builder.CreateMaskedLoad(VecTy, ResAddr,
4690 Group->getAlign(), GroupMask,
4691 PoisonVec, "wide.masked.vec");
4692 } else
4693 NewLoad = State.Builder.CreateAlignedLoad(VecTy, ResAddr,
4694 Group->getAlign(), "wide.vec");
4695 applyMetadata(*NewLoad);
4696 // TODO: Also manage existing metadata using VPIRMetadata.
4697 Group->addMetadata(NewLoad);
4698
4700 if (VecTy->isScalableTy()) {
4701 // Scalable vectors cannot use arbitrary shufflevectors (only splats),
4702 // so must use intrinsics to deinterleave.
4703 assert(InterleaveFactor <= 8 &&
4704 "Unsupported deinterleave factor for scalable vectors");
4705 NewLoad = State.Builder.CreateIntrinsicWithoutFolding(
4706 Intrinsic::getDeinterleaveIntrinsicID(InterleaveFactor),
4707 NewLoad->getType(), NewLoad,
4708 /*FMFSource=*/nullptr, "strided.vec");
4709 }
4710
4711 auto CreateStridedVector = [&InterleaveFactor, &State,
4712 &NewLoad](unsigned Index) -> Value * {
4713 assert(Index < InterleaveFactor && "Illegal group index");
4714 if (State.VF.isScalable())
4715 return State.Builder.CreateExtractValue(NewLoad, Index);
4716
4717 // For fixed length VF, use shuffle to extract the sub-vectors from the
4718 // wide load.
4719 auto StrideMask =
4720 createStrideMask(Index, InterleaveFactor, State.VF.getFixedValue());
4721 return State.Builder.CreateShuffleVector(NewLoad, StrideMask,
4722 "strided.vec");
4723 };
4724
4725 for (unsigned I = 0, J = 0; I < InterleaveFactor; ++I) {
4726 Instruction *Member = Group->getMember(I);
4727
4728 // Skip the gaps in the group.
4729 if (!Member)
4730 continue;
4731
4732 Value *StridedVec = CreateStridedVector(I);
4733
4734 // If this member has different type, cast the result type.
4735 if (Member->getType() != ScalarTy) {
4736 VectorType *OtherVTy = VectorType::get(Member->getType(), State.VF);
4737 StridedVec =
4738 createBitOrPointerCast(State.Builder, StridedVec, OtherVTy, DL);
4739 }
4740
4741 if (Group->isReverse())
4742 StridedVec = State.Builder.CreateVectorReverse(StridedVec, "reverse");
4743
4744 State.set(VPDefs[J], StridedVec);
4745 ++J;
4746 }
4747 return;
4748 }
4749
4750 // The sub vector type for current instruction.
4751 auto *SubVT = VectorType::get(ScalarTy, State.VF);
4752
4753 // Vectorize the interleaved store group.
4754 Value *MaskForGaps =
4755 createBitMaskForGaps(State.Builder, State.VF.getKnownMinValue(), *Group);
4756 assert(((MaskForGaps != nullptr) == needsMaskForGaps()) &&
4757 "Mismatch between NeedsMaskForGaps and MaskForGaps");
4758 ArrayRef<VPValue *> StoredValues = getStoredValues();
4759 // Collect the stored vector from each member.
4760 SmallVector<Value *, 4> StoredVecs;
4761 unsigned StoredIdx = 0;
4762 for (unsigned i = 0; i < InterleaveFactor; i++) {
4763 assert((Group->getMember(i) || MaskForGaps) &&
4764 "Fail to get a member from an interleaved store group");
4765 Instruction *Member = Group->getMember(i);
4766
4767 // Skip the gaps in the group.
4768 if (!Member) {
4769 Value *Undef = PoisonValue::get(SubVT);
4770 StoredVecs.push_back(Undef);
4771 continue;
4772 }
4773
4774 Value *StoredVec = State.get(StoredValues[StoredIdx]);
4775 ++StoredIdx;
4776
4777 if (Group->isReverse())
4778 StoredVec = State.Builder.CreateVectorReverse(StoredVec, "reverse");
4779
4780 // If this member has different type, cast it to a unified type.
4781
4782 if (StoredVec->getType() != SubVT)
4783 StoredVec = createBitOrPointerCast(State.Builder, StoredVec, SubVT, DL);
4784
4785 StoredVecs.push_back(StoredVec);
4786 }
4787
4788 // Interleave all the smaller vectors into one wider vector.
4789 Value *IVec = interleaveVectors(State.Builder, StoredVecs, "interleaved.vec");
4790 Instruction *NewStoreInstr;
4791 if (BlockInMask || MaskForGaps) {
4792 Value *GroupMask = CreateGroupMask(MaskForGaps);
4793 NewStoreInstr = State.Builder.CreateMaskedStore(
4794 IVec, ResAddr, Group->getAlign(), GroupMask);
4795 } else
4796 NewStoreInstr =
4797 State.Builder.CreateAlignedStore(IVec, ResAddr, Group->getAlign());
4798
4799 applyMetadata(*NewStoreInstr);
4800 // TODO: Also manage existing metadata using VPIRMetadata.
4801 Group->addMetadata(NewStoreInstr);
4802}
4803
4804#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4806 VPSlotTracker &SlotTracker) const {
4808 O << Indent << "INTERLEAVE-GROUP with factor " << IG->getFactor() << ", ";
4810 VPValue *Mask = getMask();
4811 if (Mask) {
4812 O << ", ";
4813 Mask->printAsOperand(O, SlotTracker);
4814 }
4815
4816 unsigned OpIdx = 0;
4817 for (unsigned i = 0; i < IG->getFactor(); ++i) {
4818 if (!IG->getMember(i))
4819 continue;
4820 if (getNumStoreOperands() > 0) {
4821 O << "\n" << Indent << " store ";
4822 getOperand(1 + OpIdx)->printAsOperand(O, SlotTracker);
4823 O << " to index " << i;
4824 } else {
4825 O << "\n" << Indent << " ";
4827 O << " = load from index " << i;
4828 }
4829 ++OpIdx;
4830 }
4831}
4832#endif
4833
4835 assert(State.VF.isScalable() &&
4836 "Only support scalable VF for EVL tail-folding.");
4838 "Masking gaps for scalable vectors is not yet supported.");
4840 Instruction *Instr = Group->getInsertPos();
4841
4842 // Prepare for the vector type of the interleaved load/store.
4843 Type *ScalarTy = getLoadStoreType(Instr);
4844 unsigned InterleaveFactor = Group->getFactor();
4845 assert(InterleaveFactor <= 8 &&
4846 "Unsupported deinterleave/interleave factor for scalable vectors");
4847 ElementCount WideVF = State.VF * InterleaveFactor;
4848 auto *VecTy = VectorType::get(ScalarTy, WideVF);
4849
4850 VPValue *Addr = getAddr();
4851 Value *ResAddr = State.get(Addr, VPLane(0));
4852 Value *EVL = State.get(getEVL(), VPLane(0));
4853 Value *InterleaveEVL = State.Builder.CreateMul(
4854 EVL, ConstantInt::get(EVL->getType(), InterleaveFactor), "interleave.evl",
4855 /* NUW= */ true, /* NSW= */ true);
4856 LLVMContext &Ctx = State.Builder.getContext();
4857
4858 Value *GroupMask = nullptr;
4859 if (VPValue *BlockInMask = getMask()) {
4860 SmallVector<Value *> Ops(InterleaveFactor, State.get(BlockInMask));
4861 GroupMask = interleaveVectors(State.Builder, Ops, "interleaved.mask");
4862 } else {
4863 GroupMask =
4864 State.Builder.CreateVectorSplat(WideVF, State.Builder.getTrue());
4865 }
4866
4867 // Vectorize the interleaved load group.
4868 if (isa<LoadInst>(Instr)) {
4869 CallInst *NewLoad = State.Builder.CreateIntrinsicWithoutFolding(
4870 VecTy, Intrinsic::vp_load, {ResAddr, GroupMask, InterleaveEVL}, nullptr,
4871 "wide.vp.load");
4872 NewLoad->addParamAttr(0,
4873 Attribute::getWithAlignment(Ctx, Group->getAlign()));
4874
4875 applyMetadata(*NewLoad);
4876 // TODO: Also manage existing metadata using VPIRMetadata.
4877 Group->addMetadata(NewLoad);
4878
4879 // Scalable vectors cannot use arbitrary shufflevectors (only splats),
4880 // so must use intrinsics to deinterleave.
4881 NewLoad = State.Builder.CreateIntrinsicWithoutFolding(
4882 Intrinsic::getDeinterleaveIntrinsicID(InterleaveFactor),
4883 NewLoad->getType(), NewLoad,
4884 /*FMFSource=*/nullptr, "strided.vec");
4885
4886 const DataLayout &DL = Instr->getDataLayout();
4887 for (unsigned I = 0, J = 0; I < InterleaveFactor; ++I) {
4888 Instruction *Member = Group->getMember(I);
4889 // Skip the gaps in the group.
4890 if (!Member)
4891 continue;
4892
4893 Value *StridedVec = State.Builder.CreateExtractValue(NewLoad, I);
4894 // If this member has different type, cast the result type.
4895 if (Member->getType() != ScalarTy) {
4896 VectorType *OtherVTy = VectorType::get(Member->getType(), State.VF);
4897 StridedVec =
4898 createBitOrPointerCast(State.Builder, StridedVec, OtherVTy, DL);
4899 }
4900
4901 State.set(getVPValue(J), StridedVec);
4902 ++J;
4903 }
4904 return;
4905 } // End for interleaved load.
4906
4907 // The sub vector type for current instruction.
4908 auto *SubVT = VectorType::get(ScalarTy, State.VF);
4909 // Vectorize the interleaved store group.
4910 ArrayRef<VPValue *> StoredValues = getStoredValues();
4911 // Collect the stored vector from each member.
4912 SmallVector<Value *, 4> StoredVecs;
4913 const DataLayout &DL = Instr->getDataLayout();
4914 for (unsigned I = 0, StoredIdx = 0; I < InterleaveFactor; I++) {
4915 Instruction *Member = Group->getMember(I);
4916 // Skip the gaps in the group.
4917 if (!Member) {
4918 StoredVecs.push_back(PoisonValue::get(SubVT));
4919 continue;
4920 }
4921
4922 Value *StoredVec = State.get(StoredValues[StoredIdx]);
4923 // If this member has different type, cast it to a unified type.
4924 if (StoredVec->getType() != SubVT)
4925 StoredVec = createBitOrPointerCast(State.Builder, StoredVec, SubVT, DL);
4926
4927 StoredVecs.push_back(StoredVec);
4928 ++StoredIdx;
4929 }
4930
4931 // Interleave all the smaller vectors into one wider vector.
4932 Value *IVec = interleaveVectors(State.Builder, StoredVecs, "interleaved.vec");
4933 CallInst *NewStore = State.Builder.CreateIntrinsicWithoutFolding(
4934 Type::getVoidTy(Ctx), Intrinsic::vp_store,
4935 {IVec, ResAddr, GroupMask, InterleaveEVL});
4936
4937 NewStore->addParamAttr(1,
4938 Attribute::getWithAlignment(Ctx, Group->getAlign()));
4939
4940 applyMetadata(*NewStore);
4941 // TODO: Also manage existing metadata using VPIRMetadata.
4942 Group->addMetadata(NewStore);
4943}
4944
4945#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4947 VPSlotTracker &SlotTracker) const {
4949 O << Indent << "INTERLEAVE-GROUP with factor " << IG->getFactor() << ", ";
4951 O << ", ";
4953 if (VPValue *Mask = getMask()) {
4954 O << ", ";
4955 Mask->printAsOperand(O, SlotTracker);
4956 }
4957
4958 unsigned OpIdx = 0;
4959 for (unsigned i = 0; i < IG->getFactor(); ++i) {
4960 if (!IG->getMember(i))
4961 continue;
4962 if (getNumStoreOperands() > 0) {
4963 O << "\n" << Indent << " vp.store ";
4964 getOperand(2 + OpIdx)->printAsOperand(O, SlotTracker);
4965 O << " to index " << i;
4966 } else {
4967 O << "\n" << Indent << " ";
4969 O << " = vp.load from index " << i;
4970 }
4971 ++OpIdx;
4972 }
4973}
4974#endif
4975
4977 VPCostContext &Ctx) const {
4978 Instruction *InsertPos = getInsertPos();
4979 // Find the VPValue index of the interleave group. We need to skip gaps.
4980 unsigned InsertPosIdx = 0;
4981 for (unsigned Idx = 0; IG->getFactor(); ++Idx)
4982 if (auto *Member = IG->getMember(Idx)) {
4983 if (Member == InsertPos)
4984 break;
4985 InsertPosIdx++;
4986 }
4987 const VPValue *ValV = getNumDefinedValues() > 0
4988 ? getVPValue(InsertPosIdx)
4989 : getStoredValues()[InsertPosIdx];
4990 Type *ValTy = ValV->getScalarType();
4991 auto *VectorTy = cast<VectorType>(toVectorTy(ValTy, VF));
4992 unsigned AS =
4993 cast<PointerType>(getAddr()->getScalarType())->getAddressSpace();
4994
4995 unsigned InterleaveFactor = IG->getFactor();
4996 auto *WideVecTy = VectorType::get(ValTy, VF * InterleaveFactor);
4997
4998 // Holds the indices of existing members in the interleaved group.
5000 for (unsigned IF = 0; IF < InterleaveFactor; IF++)
5001 if (IG->getMember(IF))
5002 Indices.push_back(IF);
5003
5004 // Calculate the cost of the whole interleaved group.
5005 InstructionCost Cost = Ctx.TTI.getInterleavedMemoryOpCost(
5006 InsertPos->getOpcode(), WideVecTy, IG->getFactor(), Indices,
5007 IG->getAlign(), AS, Ctx.CostKind, getMask(), NeedsMaskForGaps);
5008
5009 if (!IG->isReverse())
5010 return Cost;
5011
5012 return Cost + IG->getNumMembers() *
5013 Ctx.TTI.getShuffleCost(TargetTransformInfo::SK_Reverse,
5014 VectorTy, VectorTy, Ctx.CostKind, {},
5015 0);
5016}
5017
5020 VPCostContext &Ctx) const {
5021 // The recipe creates a scalar phi, a GEP to increment the induction and
5022 // vector add to compute the vector of pointers.
5023 // TODO: Charge costs for induction increment and vector add as well.
5024 return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
5025}
5026
5028 return vputils::onlyScalarValuesUsed(this) &&
5029 (!IsScalable || vputils::onlyFirstLaneUsed(this));
5030}
5031
5032#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5034 raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
5035 assert((getNumOperands() == 3 || getNumOperands() == 5) &&
5036 "unexpected number of operands");
5037 O << Indent << "EMIT ";
5039 O << " = WIDEN-POINTER-INDUCTION ";
5041 O << ", ";
5043 O << ", ";
5045 if (getNumOperands() == 5) {
5046 O << ", ";
5048 O << ", ";
5050 }
5051}
5052
5054 VPSlotTracker &SlotTracker) const {
5055 O << Indent << "EMIT ";
5057 O << " = EXPAND SCEV " << *Expr;
5058}
5059#endif
5060
5061#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5063 VPSlotTracker &SlotTracker) const {
5064 O << Indent << "EMIT ";
5066 O << " = WIDEN-CANONICAL-INDUCTION";
5067 printFlags(O);
5069}
5070#endif
5071
5073 auto &Builder = State.Builder;
5074 // Create a vector from the initial value.
5075 auto *VectorInit = getStartValue()->getLiveInIRValue();
5076
5077 Type *VecTy = State.VF.isScalar()
5078 ? VectorInit->getType()
5079 : VectorType::get(VectorInit->getType(), State.VF);
5080
5081 BasicBlock *VectorPH =
5082 State.CFG.VPBB2IRBB.at(getParent()->getCFGPredecessor(0));
5083 if (State.VF.isVector()) {
5084 auto *IdxTy = Builder.getInt32Ty();
5085 auto *One = ConstantInt::get(IdxTy, 1);
5086 IRBuilder<>::InsertPointGuard Guard(Builder);
5087 Builder.SetInsertPoint(VectorPH->getTerminator());
5088 auto *RuntimeVF = getRuntimeVF(Builder, IdxTy, State.VF);
5089 auto *LastIdx = Builder.CreateSub(RuntimeVF, One);
5090 VectorInit = Builder.CreateInsertElement(
5091 PoisonValue::get(VecTy), VectorInit, LastIdx, "vector.recur.init");
5092 }
5093
5094 // Create a phi node for the new recurrence.
5095 PHINode *Phi = PHINode::Create(VecTy, 2, "vector.recur");
5096 Phi->insertBefore(State.CFG.PrevBB->getFirstInsertionPt());
5097 Phi->addIncoming(VectorInit, VectorPH);
5098 State.set(this, Phi);
5099}
5100
5103 VPCostContext &Ctx) const {
5104 if (VF.isScalar())
5105 return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
5106
5107 return 0;
5108}
5109
5110#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5112 raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
5113 O << Indent << "FIRST-ORDER-RECURRENCE-PHI ";
5115 O << " = phi ";
5117}
5118#endif
5119
5121 // Reductions do not have to start at zero. They can start with
5122 // any loop invariant values.
5123 VPValue *StartVPV = getStartValue();
5124
5125 // In order to support recurrences we need to be able to vectorize Phi nodes.
5126 // Phi nodes have cycles, so we need to vectorize them in two stages. This is
5127 // stage #1: We create a new vector PHI node with no incoming edges. We'll use
5128 // this value when we vectorize all of the instructions that use the PHI.
5129 BasicBlock *VectorPH =
5130 State.CFG.VPBB2IRBB.at(getParent()->getCFGPredecessor(0));
5131 bool ScalarPHI = State.VF.isScalar() || isInLoop();
5132 Value *StartV = State.get(StartVPV, ScalarPHI);
5133 Type *VecTy = StartV->getType();
5134
5135 BasicBlock *HeaderBB = State.CFG.PrevBB;
5136 assert(State.CurrentParentLoop->getHeader() == HeaderBB &&
5137 "recipe must be in the vector loop header");
5138 auto *Phi = PHINode::Create(VecTy, 2, "vec.phi");
5139 Phi->insertBefore(HeaderBB->getFirstInsertionPt());
5140 State.set(this, Phi, isInLoop());
5141
5142 Phi->addIncoming(StartV, VectorPH);
5143}
5144
5145#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5147 VPSlotTracker &SlotTracker) const {
5148 O << Indent << "WIDEN-REDUCTION-PHI ";
5149
5151 O << " = phi (";
5152 printRecurrenceKind(O, Kind);
5153 O << ")";
5154 printFlags(O);
5156 if (getVFScaleFactor() > 1)
5157 O << " (VF scaled by 1/" << getVFScaleFactor() << ")";
5158}
5159#endif
5160
5162 assert(is_contained(operands(), Op) && "Op must be an operand of the recipe");
5163 return vputils::onlyFirstLaneUsed(this);
5164}
5165
5167 executePhiRecipe(this, *this, State, /*IsScalar=*/false, Name);
5168}
5169
5171 VPCostContext &Ctx) const {
5172 return Ctx.TTI.getCFInstrCost(Instruction::PHI, Ctx.CostKind);
5173}
5174
5175#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5177 VPSlotTracker &SlotTracker) const {
5178 O << Indent << "WIDEN-PHI ";
5179
5181 O << " = phi ";
5183}
5184#endif
5185
5187 BasicBlock *VectorPH =
5188 State.CFG.VPBB2IRBB.at(getParent()->getCFGPredecessor(0));
5189 Value *StartMask = State.get(getOperand(0));
5190 PHINode *Phi =
5191 State.Builder.CreatePHI(StartMask->getType(), 2, "active.lane.mask");
5192 Phi->addIncoming(StartMask, VectorPH);
5193 State.set(this, Phi);
5194}
5195
5196#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5198 VPSlotTracker &SlotTracker) const {
5199 O << Indent << "ACTIVE-LANE-MASK-PHI ";
5200
5202 O << " = phi ";
5204}
5205#endif
5206
5207#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5209 raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const {
5210 O << Indent << "CURRENT-ITERATION-PHI ";
5211
5213 O << " = phi ";
5215}
5216#endif
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
static MCDisassembler::DecodeStatus addOperand(MCInst &Inst, const MCOperand &Opnd)
AMDGPU Lower Kernel Arguments
AMDGPU Register Bank Select
This file declares a class to represent arbitrary precision floating point values and provide a varie...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static const Function * getParent(const Value *V)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static void replaceAllUsesWith(Value *Old, Value *New, SmallPtrSet< BasicBlock *, 32 > &FreshBBs, bool IsHuge)
Replace all old uses with new ones, and push the updated BBs into FreshBBs.
Hexagon Common GEP
Value * getPointer(Value *Ptr)
iv users
Definition IVUsers.cpp:48
static constexpr Value * getValue(Ty &ValueOrUse)
static std::pair< Value *, APInt > getMask(Value *WideMask, unsigned Factor, ElementCount LeafValueEC)
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
This file provides a LoopVectorizationPlanner class.
static const SCEV * getAddressAccessSCEV(Value *Ptr, PredicatedScalarEvolution &PSE, const Loop *TheLoop)
Gets the address access SCEV for Ptr, if it should be used for cost modeling according to isAddressSC...
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
static bool isOrdered(const Instruction *I)
uint64_t IntrinsicInst * II
#define P(N)
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
SI Fold Operands
static SDValue getFPBinOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL, EVT VT, SDValue A, SDValue B, SDValue GlueChain, SDNodeFlags Flags)
This file contains some templates that are useful if you are working with the STL at all.
This file defines less commonly used SmallVector utilities.
This file defines the SmallVector class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
This file contains the declarations of different VPlan-related auxiliary helpers.
static Value * interleaveVectors(IRBuilderBase &Builder, ArrayRef< Value * > Vals, const Twine &Name)
Return a vector containing interleaved elements from multiple smaller input vectors.
static const ConstantFP * getConstantFP(const VPValue *V)
Returns the ConstantFP V wraps, or nullptr if it does not wrap one.
static void executePhiRecipe(VPSingleDefRecipe *R, VPPhiAccessors &Phi, VPTransformState &State, bool IsScalar, const Twine &Name)
Shared execute logic for VPPhi and VPWidenPHIRecipe.
static Value * createBitOrPointerCast(IRBuilderBase &Builder, Value *V, VectorType *DstVTy, const DataLayout &DL)
static Instruction::BinaryOps getSubRecurOpcode(RecurKind Kind)
static VPExecutionFrequency getExecutionFrequencyFromMD(const MDNode *Node)
Returns the execution frequency recorded in Node.
static void printRecurrenceKind(raw_ostream &OS, const RecurKind &Kind)
static unsigned getCalledFnOperandIndex(ArrayRef< VPValue * > Operands)
For call VPInstruction operands, return the operand index of the called function.
This file contains the declarations of the Vectorization Plan base classes:
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:410
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
LLVM_ABI const_iterator getFirstInsertionPt() const
Returns an iterator to the first instruction in this block that is suitable for inserting a non-PHI i...
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Definition BasicBlock.h:237
void addParamAttr(unsigned ArgNo, Attribute::AttrKind Kind)
Adds the attribute to the indicated argument.
This class represents a function call, abstracting a target machine's calling convention.
static LLVM_ABI bool isBitOrNoopPointerCastable(Type *SrcTy, Type *DestTy, const DataLayout &DL)
Check whether a bitcast, inttoptr, or ptrtoint cast between these types is valid and a no-op.
static Type * makeCmpResultType(Type *opnd_type)
Create a result type for fcmp/icmp.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
static LLVM_ABI StringRef getPredicateName(Predicate P)
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
void setSuccessor(unsigned idx, BasicBlock *NewSucc)
static ConstantAsMetadata * get(Constant *C)
Definition Metadata.h:548
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
bool isNegZero() const
Return true if the value is negative zero.
Definition Constants.h:473
bool isOne() const
Returns true if this value is exactly +1.0.
Definition Constants.h:485
bool isZero() const
Return true if the value is positive or negative zero.
Definition Constants.h:467
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
static DebugLoc getUnknown()
Definition DebugLoc.h:153
constexpr bool isVector() const
One or more elements.
Definition TypeSize.h:320
static constexpr ElementCount getScalable(ScalarTy MinVal)
Definition TypeSize.h:308
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:305
constexpr bool isScalar() const
Exactly one element.
Definition TypeSize.h:316
static bool isSupportedFloatingPointType(Type *Ty)
Returns true if Ty is a supported floating-point type for phi, select, or call FPMathOperators.
Definition Operator.h:302
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
LLVM_ABI void print(raw_ostream &O) const
Print fast-math flags to O.
Definition Operator.cpp:290
void setAllowContract(bool B=true)
Definition FMF.h:90
bool noSignedZeros() const
Definition FMF.h:67
bool noInfs() const
Definition FMF.h:66
void setAllowReciprocal(bool B=true)
Definition FMF.h:87
bool allowReciprocal() const
Definition FMF.h:68
void setNoSignedZeros(bool B=true)
Definition FMF.h:84
bool allowReassoc() const
Flag queries.
Definition FMF.h:64
bool approxFunc() const
Definition FMF.h:70
void setNoNaNs(bool B=true)
Definition FMF.h:78
void setAllowReassoc(bool B=true)
Flag setters.
Definition FMF.h:75
bool noNaNs() const
Definition FMF.h:65
void setApproxFunc(bool B=true)
Definition FMF.h:93
void setNoInfs(bool B=true)
Definition FMF.h:81
bool allowContract() const
Definition FMF.h:69
Class to represent function types.
Type * getParamType(unsigned i) const
Parameter type accessors.
bool willReturn() const
Determine if the function will return.
Definition Function.h:647
Intrinsic::ID getIntrinsicID() const LLVM_READONLY
getIntrinsicID - This method returns the ID number of the specified function, or Intrinsic::not_intri...
Definition Function.h:247
bool doesNotThrow() const
Determine if the function cannot unwind.
Definition Function.h:577
bool doesNotAccessMemory() const
Determine if the function does not access memory.
Definition Function.cpp:869
Type * getReturnType() const
Returns the type of the ret val.
Definition Function.h:217
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags none()
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Definition IRBuilder.h:2661
IntegerType * getInt1Ty()
Fetch the type representing a single bit.
Definition IRBuilder.h:498
Value * CreateInsertValue(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &Name="")
Definition IRBuilder.h:2715
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
Definition IRBuilder.h:2649
LLVM_ABI Value * CreateVectorSpliceRight(Value *V1, Value *V2, Value *Offset, const Twine &Name="")
Create a vector.splice.right intrinsic call, or a shufflevector that produces the same result if the ...
CondBrInst * CreateCondBr(Value *Cond, BasicBlock *True, BasicBlock *False, MDNode *BranchWeights=nullptr, MDNode *Unpredictable=nullptr)
Create a conditional 'br Cond, TrueDest, FalseDest' instruction.
Definition IRBuilder.h:1203
LLVM_ABI Value * CreateSelectFMF(Value *C, Value *True, Value *False, FMFSource FMFSource, const Twine &Name="", Instruction *MDFrom=nullptr)
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
Value * CreateExtractValue(Value *Agg, ArrayRef< unsigned > Idxs, const Twine &Name="")
Definition IRBuilder.h:2708
LLVM_ABI Value * CreateSelect(Value *C, Value *True, Value *False, const Twine &Name="", Instruction *MDFrom=nullptr)
Value * CreateFreeze(Value *V, const Twine &Name="")
Definition IRBuilder.h:2727
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Definition IRBuilder.h:513
Value * CreateExtractVector(Type *DstType, Value *SrcVec, Value *Idx, const Twine &Name="")
Create a call to the vector.extract intrinsic.
Definition IRBuilder.h:1099
Value * CreatePtrAdd(Value *Ptr, Value *Offset, const Twine &Name="", GEPNoWrapFlags NW=GEPNoWrapFlags::none())
Definition IRBuilder.h:2084
Value * CreateCast(Instruction::CastOps Op, Value *V, Type *DestTy, const Twine &Name="", MDNode *FPMathTag=nullptr, FMFSource FMFSource={})
Definition IRBuilder.h:2276
void setFastMathFlags(FastMathFlags NewFMF)
Set the fast-math flags to be used with generated fp-math operators.
Definition IRBuilder.h:280
LLVM_ABI Value * CreateVectorReverse(Value *V, const Twine &Name="")
Return a vector value that contains the vector V reversed.
Value * CreateICmpNE(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2378
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Definition IRBuilder.h:461
Value * CreateLogicalAnd(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Definition IRBuilder.h:1757
LLVM_ABI Value * CreateOrReduce(Value *Src)
Create a vector int OR reduction intrinsic of the source vector.
ConstantInt * getInt32(uint32_t C)
Get a constant 32-bit value.
Definition IRBuilder.h:456
Value * CreateCmp(CmpInst::Predicate Pred, Value *LHS, Value *RHS, const Twine &Name="", MDNode *FPMathTag=nullptr)
Definition IRBuilder.h:2508
Value * CreateNot(Value *V, const Twine &Name="")
Definition IRBuilder.h:1841
Value * CreateICmpEQ(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2374
Value * CreateCountTrailingZeroElems(Type *ResTy, Value *Mask, bool ZeroIsPoison=true, const Twine &Name="")
Create a call to llvm.experimental_cttz_elts.
Definition IRBuilder.h:1141
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1426
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2113
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1409
ConstantInt * getFalse()
Get the constant value for i1 false.
Definition IRBuilder.h:441
Value * CreateBinOp(Instruction::BinaryOps Opc, Value *LHS, Value *RHS, const Twine &Name="", MDNode *FPMathTag=nullptr)
Definition IRBuilder.h:1718
Value * CreateICmpUGE(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2386
Value * CreateLogicalOr(Value *Cond1, Value *Cond2, const Twine &Name="", Instruction *MDFrom=nullptr)
Definition IRBuilder.h:1765
Value * CreateOr(Value *LHS, Value *RHS, const Twine &Name="", bool IsDisjoint=false)
Definition IRBuilder.h:1579
LLVM_ABI Value * CreateStepVector(Type *DstType, const Twine &Name="")
Creates a vector of type DstType with the linear sequence <0, 1, ...>
Value * CreateMul(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1443
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
A struct for saving information about induction variables.
@ IK_FpInduction
Floating point induction variable.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
bool isCast() const
bool isBinaryOp() const
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
const char * getOpcodeName() const
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
bool isUnaryOp() const
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
The group of interleaved loads/stores sharing the same stride and close to each other.
uint32_t getFactor() const
InstTy * getMember(uint32_t Index) const
Get the member with the given index Index.
bool isReverse() const
InstTy * getInsertPos() const
void addMetadata(InstTy *NewInst) const
Add metadata (e.g.
Align getAlign() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Metadata node.
Definition Metadata.h:1081
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1579
Information for memory intrinsic cost model.
Root of the metadata hierarchy.
Definition Metadata.h:64
LLVM_ABI void print(raw_ostream &OS, const Module *M=nullptr, bool IsForDebug=false) const
Print.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static PHINode * Create(Type *Ty, unsigned NumReservedValues, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedValues is a hint for the number of incoming edges that this phi node will h...
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
static bool isAnyOfRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
static LLVM_ABI bool isSubRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is for a sub operation.
static bool isFindIVRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
static bool isMinMaxRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is any min/max kind.
This class represents an analyzed expression in the program.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
This class represents the LLVM 'select' instruction.
This class provides computation of slot numbers for LLVM Assembly writing.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
reference emplace_back(ArgTypes &&... Args)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static LLVM_ABI PartialReductionExtendKind getPartialReductionExtendKind(Instruction *I)
Get the kind of extension that an instruction represents.
static LLVM_ABI OperandValueInfo getOperandInfo(const Value *V)
Collect properties of V used in cost analysis, e.g. OP_PowerOf2.
llvm::VectorInstrContext VectorInstrContext
@ TCC_Free
Expected to fold away in lowering.
@ SK_Splice
Concatenates elements from the first input vector with elements of the second input vector.
@ SK_Reverse
Reverse the order of the vector.
CastContextHint
Represents a hint about the context in which a cast is used.
@ Reversed
The cast is used with a reversed load/store.
@ Masked
The cast is used with a masked load/store.
@ None
The cast is not used with a load/store of any kind.
@ Normal
The cast is used with a normal load/store.
@ Interleave
The cast is used with an interleaved load/store.
@ GatherScatter
The cast is used with a gather/scatter.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isByteTy() const
True if this is an instance of ByteType.
Definition Type.h:237
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
Definition Type.cpp:272
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:271
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
Definition Type.h:265
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
value_op_iterator value_op_end()
Definition User.h:288
void setOperand(unsigned i, Value *Val)
Definition User.h:212
Value * getOperand(unsigned i) const
Definition User.h:207
value_op_iterator value_op_begin()
Definition User.h:285
void execute(VPTransformState &State) override
Generate the active lane mask phi of the vector loop.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
Definition VPlan.h:4414
RecipeListTy & getRecipeList()
Returns a reference to the list of recipes.
Definition VPlan.h:4467
iterator end()
Definition VPlan.h:4451
void insert(VPRecipeBase *Recipe, iterator InsertPt)
Definition VPlan.h:4480
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenMemoryRecipe.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
Definition VPlan.h:3000
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
Definition VPlan.h:2995
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
Definition VPlan.h:2991
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
Definition VPlan.h:97
const VPBlocksTy & getPredecessors() const
Definition VPlan.h:230
static bool isHeader(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop header, based on regions or VPDT in their absence.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPBranchOnMaskRecipe.
void execute(VPTransformState &State) override
Generate the extraction of the appropriate bit from the block mask and the conditional branch.
LLVM_ABI_FOR_TEST void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
Definition VPlanValue.h:576
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
Definition VPlanValue.h:549
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
Definition VPlanValue.h:561
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Definition VPlanValue.h:571
InductionDescriptor::InductionKind getInductionKind() const
Definition VPlan.h:4232
VPValue * getIndex() const
Definition VPlan.h:4229
VPValue * getStepValue() const
Definition VPlan.h:4230
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPDerivedIVRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPValue * getStartValue() const
Definition VPlan.h:4228
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPExpandSCEVRecipe(const SCEV *Expr)
bool isVectorToScalar() const
Returns true if this VPExpressionRecipe produces a single scalar.
SmallVector< VPSingleDefRecipe * > decompose()
Return and insert the recipes of the expression back into the VPlan, directly before the current reci...
bool mayHaveSideEffects() const
Returns true if this expression contains recipes that may have side effects.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Compute the cost of this recipe either using a recipe's specialized implementation or using the legac...
bool mayReadOrWriteMemory() const
Returns true if this expression contains recipes that may read from or write to memory.
VPExpressionRecipe(ExpressionTypes ExpressionType, ArrayRef< VPSingleDefRecipe * > ExpressionRecipes)
Construct a new VPExpressionRecipe by internalizing recipes in ExpressionRecipes.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this header phi recipe.
VPValue * getStartValue()
Returns the start value of the phi, if one is set.
Definition VPlan.h:2484
void execute(VPTransformState &State) override
Produce a vectorized histogram operation.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPHistogramRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPValue * getMask() const
Return the mask operand if one was provided, or a null pointer if all lanes should be executed uncond...
Definition VPlan.h:2198
Class to record and manage LLVM IR flags.
Definition VPlan.h:696
FastMathFlagsTy FMFs
Definition VPlan.h:785
ReductionFlagsTy ReductionFlags
Definition VPlan.h:787
LLVM_ABI_FOR_TEST bool flagsValidForOpcode(unsigned Opcode) const
Returns true if the set flags are valid for Opcode.
WrapFlagsTy WrapFlags
Definition VPlan.h:779
void printFlags(raw_ostream &O) const
bool hasFastMathFlags() const
Returns true if the recipe has fast-math flags.
Definition VPlan.h:1002
static LLVM_ABI_FOR_TEST VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
bool isReductionOrdered() const
Definition VPlan.h:1057
TruncFlagsTy TruncFlags
Definition VPlan.h:780
CmpInst::Predicate getPredicate() const
Definition VPlan.h:974
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
ExactFlagsTy ExactFlags
Definition VPlan.h:782
void intersectFlags(const VPIRFlags &Other)
Only keep flags also present in Other.
uint8_t GEPFlagsStorage
Definition VPlan.h:783
GEPNoWrapFlags getGEPNoWrapFlags() const
Definition VPlan.h:992
bool hasPredicate() const
Returns true if the recipe has a comparison predicate.
Definition VPlan.h:997
LLVM_ABI_FOR_TEST bool hasRequiredFlagsForOpcode(unsigned Opcode, Type *ResultTy) const
Returns true if Opcode with scalar result type ResultTy has its required flags set.
DisjointFlagsTy DisjointFlags
Definition VPlan.h:781
FCmpFlagsTy FCmpFlags
Definition VPlan.h:786
NonNegFlagsTy NonNegFlags
Definition VPlan.h:784
bool isReductionInLoop() const
Definition VPlan.h:1063
void applyFlags(Instruction &I) const
Apply the IR flags to I.
Definition VPlan.h:931
uint8_t CmpPredStorage
Definition VPlan.h:778
RecurKind getRecurKind() const
Definition VPlan.h:1051
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
LLVM_ABI_FOR_TEST InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPIRInstruction.
VPIRInstruction(Instruction &I)
VPIRInstruction::create() should be used to create VPIRInstructions, as subclasses may need to be cre...
Definition VPlan.h:1728
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
std::optional< VPExecutionFrequency > getExecutionFrequency() const
Returns the frequency recorded by setExecutionFrequency, if any.
void intersect(const VPIRMetadata &MD)
Intersect this VPIRMetadata object with MD, keeping only metadata nodes that are common to both.
void clearExecutionFrequency()
Drop the frequency recorded by setExecutionFrequency, if any.
VPIRMetadata()=default
void print(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print metadata with node IDs.
void applyMetadata(Instruction &I) const
Add all metadata to I.
void setMetadata(unsigned Kind, MDNode *Node)
Set metadata with kind Kind to Node.
Definition VPlan.h:1230
void setExecutionFrequency(std::optional< VPExecutionFrequency > Freq, LLVMContext &Ctx)
Record that the recipe executes with frequency Freq, relative to the entry of the loop region.
This is a concrete Recipe that models a single VPlan-level instruction.
Definition VPlan.h:1300
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPInstruction.
VPInstruction(unsigned Opcode, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
bool doesGeneratePerAllLanes() const
Returns true if this recipe produces scalar values for all VF lanes.
@ ExtractLastActive
Extracts the last active lane from a set of vectors.
Definition VPlan.h:1409
@ Intrinsic
Calls a scalar intrinsic. The intrinsic ID is the last operand.
Definition VPlan.h:1421
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
Definition VPlan.h:1400
@ ExitingIVValue
Compute the exiting value of a wide induction after vectorization, that is the value of the last lane...
Definition VPlan.h:1413
@ WideIVStep
Scale the first operand (vector step) by the second operand (scalar-step).
Definition VPlan.h:1417
@ ResumeForEpilogue
Explicit user for the resume phi of the canonical induction in the main VPlan, used by the epilogue v...
Definition VPlan.h:1403
@ Unpack
Extracts all lanes from its (non-scalable) vector operand.
Definition VPlan.h:1351
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
Definition VPlan.h:1396
@ BuildVector
Creates a fixed-width vector containing all operands.
Definition VPlan.h:1346
@ BuildStructVector
Given operands of (the same) struct type, creates a struct of fixed- width vectors each containing a ...
Definition VPlan.h:1343
@ CanonicalIVIncrementForPart
Definition VPlan.h:1327
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
Definition VPlan.h:1354
bool hasResult() const
Definition VPlan.h:1506
bool opcodeMayReadOrWriteFromMemory() const
Returns true if the underlying opcode may read from or write to memory.
LLVM_DUMP_METHOD void dump() const
Print the VPInstruction to dbgs() (for debugging).
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the VPInstruction to O.
StringRef getName() const
Returns the symbolic name assigned to the VPInstruction.
Definition VPlan.h:1592
unsigned getOpcode() const
Definition VPlan.h:1485
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
void addOperand(VPValue *Op)
Add Op as operand of this VPInstruction.
bool isVectorToScalar() const
Returns true if this VPInstruction produces a scalar value from a vector, e.g.
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
unsigned getNumOperandsForOpcode() const
Return the number of operands determined by the opcode of the VPInstruction, excluding mask.
bool isMasked() const
Returns true if the VPInstruction has a mask operand.
Definition VPlan.h:1531
void execute(VPTransformState &State) override
Generate the instruction.
bool usesFirstPartOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first part of operand Op.
bool needsMaskForGaps() const
Return true if the access needs a mask because of the gaps.
Definition VPlan.h:3104
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this recipe.
Instruction * getInsertPos() const
Definition VPlan.h:3108
const InterleaveGroup< Instruction > * getInterleaveGroup() const
Definition VPlan.h:3106
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3098
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
Definition VPlan.h:3127
VPValue * getAddr() const
Return the address accessed by this recipe.
Definition VPlan.h:3092
VPValue * getEVL() const
The VPValue of the explicit vector length.
Definition VPlan.h:3201
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getNumStoreOperands() const override
Returns the number of stored operands of this interleave group.
Definition VPlan.h:3214
void execute(VPTransformState &State) override
Generate the wide load or store, and shuffles.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getNumStoreOperands() const override
Returns the number of stored operands of this interleave group.
Definition VPlan.h:3164
void execute(VPTransformState &State) override
Generate the wide load or store, and shuffles.
static LLVM_ABI std::optional< unsigned > getMaskParamPos(Intrinsic::ID IntrinsicID)
static LLVM_ABI std::optional< unsigned > getMemoryDataParamPos(Intrinsic::ID)
static LLVM_ABI std::optional< unsigned > getMemoryPointerParamPos(Intrinsic::ID)
In what follows, the term "input IR" refers to code that is fed into the vectorizer whereas the term ...
static VPLane getLastLaneForVF(const ElementCount &VF)
static VPLane getLaneFromEnd(const ElementCount &VF, unsigned Offset)
static VPLane getFirstLane()
Helper type to provide functions to access incoming values and blocks for phi-like recipes.
Definition VPlan.h:1607
virtual const VPRecipeBase * getAsRecipe() const =0
Return a VPRecipeBase* to the current object.
LLVM_ABI_FOR_TEST VPValue * getIncomingValueForBlock(const VPBasicBlock *VPBB) const
Returns the incoming value for VPBB. VPBB must be an incoming block.
void removeIncomingValueFor(VPBlockBase *IncomingBlock) const
Removes the incoming value for IncomingBlock, which must be a predecessor.
detail::zippy< llvm::detail::zip_first, VPUser::const_operand_range, const_incoming_blocks_range > incoming_values_and_blocks() const
Returns an iterator range over pairs of incoming values and corresponding incoming blocks.
Definition VPlan.h:1657
VPValue * getIncomingValue(unsigned Idx) const
Returns the incoming VPValue with index Idx.
Definition VPlan.h:1616
void printPhiOperands(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print the recipe.
void setIncomingValueForBlock(const VPBasicBlock *VPBB, VPValue *V) const
Sets the incoming value for VPBB to V.
void execute(VPTransformState &State) override
Generates phi nodes for live-outs (from a replicate region) as needed to retain SSA form.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
Definition VPlan.h:403
bool mayReadFromMemory() const
Returns true if the recipe may read from memory.
bool mayHaveSideEffects() const
Returns true if the recipe may have side-effects.
virtual void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const =0
Each concrete VPRecipe prints itself, without printing common information, like debug info or metadat...
VPRegionBlock * getRegion()
Definition VPlan.h:4813
void dump() const
Dump the recipe to stderr (for debugging).
Definition VPlan.cpp:115
bool isPhi() const
Returns true for PHI-like recipes.
bool mayWriteToMemory() const
Returns true if the recipe may write to memory.
VPRecipeTy getVPRecipeID() const
Definition VPlan.h:521
virtual InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const
Compute the cost of this recipe either using a recipe's specialized implementation or using the legac...
VPBasicBlock * getParent()
Definition VPlan.h:475
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
Definition VPlan.h:553
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
bool isSafeToSpeculativelyExecute() const
Return true if we can safely execute this recipe unconditionally even if it is masked originally.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
VPRecipeBase(VPRecipeTy SC, ArrayRef< VPValue * > Operands, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:465
InstructionCost cost(ElementCount VF, VPCostContext &Ctx)
Return the cost of this recipe, taking into account if the cost computation should be skipped and the...
void print(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const
Print the recipe, delegating to printRecipe().
void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
void moveAfter(VPRecipeBase *MovePos)
Unlink this recipe from its current VPBasicBlock and insert it into the VPBasicBlock that MovePos liv...
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
Definition VPlanValue.h:351
friend class VPValue
Definition VPlanValue.h:330
void execute(VPTransformState &State) override
Generate the reduction in the loop.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPValue * getEVL() const
The VPValue of the explicit vector length.
Definition VPlan.h:3375
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:2907
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
Definition VPlan.h:2926
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the phi/select nodes.
bool isConditional() const
Return true if the in-loop reduction is conditional.
Definition VPlan.h:3314
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of VPReductionRecipe.
VPValue * getVecOp() const
The VPValue of the vector value to be reduced.
Definition VPlan.h:3327
VPValue * getCondOp() const
The VPValue of the condition for the block.
Definition VPlan.h:3329
RecurKind getRecurrenceKind() const
Return the recurrence kind for the in-loop reduction.
Definition VPlan.h:3310
bool isPartialReduction() const
Returns true if the reduction outputs a vector with a scaled down VF.
Definition VPlan.h:3316
VPValue * getChainOp() const
The VPValue of the scalar Chain being accumulated.
Definition VPlan.h:3325
bool isInLoop() const
Returns true if the reduction is in-loop.
Definition VPlan.h:3320
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the reduction in the loop.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
Definition VPlan.h:4639
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
Definition VPlan.h:4715
const VPBranchOnMaskRecipe * getEntryBranchOnMask() const
Return the VPBranchOnMaskRecipe from the entry block of this replicating region.
Definition VPlan.cpp:718
void execute(VPTransformState &State) override
Generate replicas of the desired Ingredient.
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
Definition VPlan.h:3456
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPReplicateRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
static Type * computeScalarType(const Instruction *I, ArrayRef< VPValue * > Operands)
Compute the scalar result type for a VPReplicateRecipe wrapping I with Operands (excluding any predic...
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
unsigned getOpcode() const
Definition VPlan.h:3494
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPScalarIVStepsRecipe.
bool doesGeneratePerAllLanes() const
Returns true if this recipe produces scalar values for all VF lanes.
VPValue * getStepValue() const
Definition VPlan.h:4287
VPValue * getStartIndex() const
Return the StartIndex, or null if known to be zero, valid only after unrolling.
Definition VPlan.h:4295
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the scalarized versions of the phi node as needed by their users.
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Definition VPlan.h:611
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
Definition VPlan.h:681
LLVM_DUMP_METHOD void dump() const
Print this VPSingleDefRecipe to dbgs() (for debugging).
VPSingleDefRecipe(VPRecipeTy SC, ArrayRef< VPValue * > Operands, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:613
This class can be used to assign names to VPValues.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
Definition VPlanValue.h:214
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
Definition VPlanValue.h:398
void printOperands(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print the operands to O.
Definition VPlan.cpp:1510
operand_range operands()
Definition VPlanValue.h:471
unsigned getNumOperands() const
Definition VPlanValue.h:438
VPValue * getOperand(unsigned N) const
Definition VPlanValue.h:439
bool operands_empty() const
Definition VPlanValue.h:475
void addOperand(VPValue *Operand)
Definition VPlanValue.h:424
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Definition VPlanValue.h:50
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Definition VPlan.cpp:147
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
Definition VPlan.cpp:141
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
Definition VPlan.cpp:1461
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
Definition VPlan.cpp:128
void printAsOperand(raw_ostream &OS, VPSlotTracker &Tracker) const
Definition VPlan.cpp:1506
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
Definition VPlanValue.h:75
void setUnderlyingValue(Value *Val)
Definition VPlanValue.h:206
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
Definition VPlanValue.h:179
VPValue * getVFValue() const
Definition VPlan.h:2299
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Type * getSourceElementType() const
Definition VPlan.h:2296
int64_t getStride() const
Definition VPlan.h:2297
void materializeOffset(unsigned Part=0)
Adds the offset operand to the recipe.
VPValue * getStride() const
Definition VPlan.h:2373
Type * getSourceElementType() const
Definition VPlan.h:2388
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
VPValue * getVFxPart() const
Definition VPlan.h:2375
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
operand_range args()
Definition VPlan.h:2151
Function * getCalledScalarFunction() const
Definition VPlan.h:2147
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenCallRecipe.
void execute(VPTransformState &State) override
Produce a widened version of the call instruction.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Instruction::CastOps getOpcode() const
Definition VPlan.h:1921
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Produce widened copies of the cast.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenCastRecipe.
void execute(VPTransformState &State) override
Generate the gep nodes.
Type * getSourceElementType() const
Definition VPlan.h:2253
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
VPValue * getStepValue()
Returns the step value of the induction.
Definition VPlan.h:2568
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
Definition VPlan.h:2585
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenIntOrFpInductionRecipe.
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
Definition VPlan.h:2673
bool isCanonical() const
Returns true if the induction is canonical, i.e.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
CallInst * createVectorCall(VPTransformState &State)
Helper function to produce the widened intrinsic call.
Intrinsic::ID getVectorIntrinsicID() const
Return the ID of the intrinsic.
Definition VPlan.h:2036
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
StringRef getIntrinsicName() const
Return to name of the intrinsic as string.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first lane of operand Op.
void execute(VPTransformState &State) override
Produce a widened version of the vector intrinsic.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this vector intrinsic.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
void execute(VPTransformState &State) override
Produce a widened version of the vector memory intrinsic.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this vector memory intrinsic.
bool IsMasked
Whether the memory access is masked.
Definition VPlan.h:3761
bool isConsecutive() const
Return whether the loaded-from / stored-to addresses are consecutive.
Definition VPlan.h:3786
Instruction & Ingredient
Definition VPlan.h:3752
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const
Return the cost of this VPWidenMemoryRecipe.
bool Consecutive
Whether the accessed addresses are consecutive.
Definition VPlan.h:3758
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3796
Align Alignment
Alignment information for this memory access.
Definition VPlan.h:3755
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
VPValue * getAddr() const
Return the address accessed by this recipe.
Definition VPlan.h:3789
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenPHIRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the phi/select nodes.
bool onlyScalarsGenerated(bool IsScalable)
Returns true if only scalar values will be generated.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenPointerInductionRecipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
void execute(VPTransformState &State) override
Produce a widened instruction using the opcode and operands of the recipe, processing State....
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getOpcode() const
Definition VPlan.h:1863
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
Definition VPlan.h:4826
const DataLayout & getDataLayout() const
Definition VPlan.h:5040
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
Definition VPlan.h:5146
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
Definition Value.cpp:394
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
void mutateType(Type *Ty)
Mutate the type of this Value to be of the specified type.
Definition Value.h:809
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
self_iterator getIterator()
Definition ilist_node.h:123
iterator erase(iterator where)
Definition ilist.h:204
pointer remove(iterator &IT)
Definition ilist.h:188
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
LLVM_ABI Intrinsic::ID getDeinterleaveIntrinsicID(unsigned Factor)
Returns the corresponding llvm.vector.deinterleaveN intrinsic for factor N.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
LLVM_ABI AttributeSet getFnAttributes(LLVMContext &C, ID id)
Return the function attributes for an intrinsic.
LLVM_ABI StringRef getBaseName(ID id)
Return the LLVM name for an intrinsic, without encoded types for overloading, such as "llvm....
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
bool match(Val *V, const Pattern &P)
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::Or, true > m_c_LogicalOr(const LHS &L, const RHS &R)
Matches L || R with LHS and RHS in either order.
auto m_ZExtOrTrunc(const Op0_t &Op0)
int_pred_ty< is_zero_int, 1 > m_False()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExplicitVectorLength, Op0_t > m_EVL(const Op0_t &Op0)
int_pred_ty< is_one, 1 > m_True()
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:679
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
BranchProbability getExecutionProbability(BlockFrequency Freq)
Returns Freq as a BranchProbability, relative to the full mass.
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
bool isAddressSCEVForCost(const SCEV *Addr, ScalarEvolution &SE, const Loop *L)
Returns true if Addr is an address SCEV that can be passed to TTI::getAddressComputationCost,...
bool onlyFirstPartUsed(const VPValue *Def)
Returns true if only the first part of Def is used.
Intrinsic::ID getIntrinsicID(const Ty *R)
Return the intrinsic ID underlying a call.
Definition VPlanUtils.h:93
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
bool onlyScalarValuesUsed(const VPValue *Def)
Returns true if only scalar values of Def are used by all users.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
LLVM_ABI_FOR_TEST const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
LLVM_ABI Value * createSimpleReduction(IRBuilderBase &B, Value *Src, RecurKind RdxKind)
Create a reduction of the given vector.
@ Offset
Definition DWP.cpp:577
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
Definition STLExtras.h:846
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
LLVM_ABI Intrinsic::ID getMinMaxReductionIntrinsicOp(Intrinsic::ID RdxID)
Returns the min/max intrinsic used when expanding a min/max reduction.
InstructionCost Cost
@ Undef
Value of the register doesn't matter.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
VectorInstrContext
Represents a hint about the context in which a vector instruction or intrinsic is used.
@ None
The instruction is not folded.
@ BinaryOp
One of the operands is a binary op.
VPBuilderBase<> VPBuilder
Definition VPlan.h:67
auto map_to_vector(ContainerTy &&C, FuncTy &&F)
Map a range to a SmallVector with element types deduced from the mapping.
Value * getRuntimeVF(IRBuilderBase &B, Type *Ty, ElementCount VF)
Return the runtime value for VF.
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
void interleaveComma(const Container &c, StreamT &os, UnaryFunctor each_fn)
Definition STLExtras.h:2329
auto cast_or_null(const Y &Val)
Definition Casting.h:714
LLVM_ABI Value * concatenateVectors(IRBuilderBase &Builder, ArrayRef< Value * > Vecs)
Concatenate a list of vectors.
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
LLVM_ABI Value * createMinMaxOp(IRBuilderBase &Builder, RecurKind RK, Value *Left, Value *Right)
Returns a Min/Max operation corresponding to MinMaxRecurrenceKind.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
static Error getOffset(const SymbolRef &Sym, SectionRef Sec, uint64_t &Result)
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
LLVM_ABI Constant * createBitMaskForGaps(IRBuilderBase &Builder, unsigned VF, const InterleaveGroup< Instruction > &Group)
Create a mask that filters the members of an interleave group where there are gaps.
LLVM_ABI llvm::SmallVector< int, 16 > createStrideMask(unsigned Start, unsigned Stride, unsigned VF)
Create a stride shuffle mask.
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
cl::opt< unsigned > ForceTargetInstructionCost("force-target-instruction-cost", cl::init(0), cl::Hidden, cl::desc("A flag that overrides the target's expected cost for " "an instruction to a single constant value. Mostly " "useful for getting consistent testing."))
Definition VPlan.cpp:58
ElementCount getVectorizedTypeVF(Type *Ty)
Returns the number of vector elements for a vectorized type.
LLVM_ABI llvm::SmallVector< int, 16 > createReplicatedMask(unsigned ReplicationFactor, unsigned VF)
Create a mask with replicated elements.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool isPointerTy(const Type *T)
Definition SPIRVUtils.h:383
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
Type * toVectorizedTy(Type *Ty, ElementCount EC)
A helper for converting to vectorized types.
LLVM_ABI Type * computeScalarTypeForInstruction(unsigned Opcode, ArrayRef< VPValue * > Operands)
Compute the scalar result type for an IR Opcode given Operands.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
LLVM_ABI bool isVectorIntrinsicWithStructReturnOverloadAtField(Intrinsic::ID ID, int RetIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic that returns a struct is overloaded at the struct elem...
@ Other
Any other memory.
Definition ModRef.h:68
static const MachineInstrBuilder & addOffset(const MachineInstrBuilder &MIB, int Offset)
LLVM_ABI llvm::SmallVector< int, 16 > createInterleaveMask(unsigned VF, unsigned NumVecs)
Create an interleave shuffle mask.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FMinimumNum
FP min with llvm.minimumnum semantics.
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ FMinimum
FP min with llvm.minimum semantics.
@ FMaxNum
FP max with llvm.maxnum semantics including NaNs.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ FAddChainWithSubs
A chain of fadds and fsubs.
@ None
Not a recurrence.
@ AnyOf
AnyOf reduction with select(cmp(),x,y) where one of (x,y) is loop invariant, and both x and y are int...
@ Xor
Bitwise or logical XOR of integers.
@ FindLast
FindLast reduction with select(cmp(),x,y) where x and y.
@ FMax
FP max implemented in terms of select(cmp()).
@ FMaximum
FP max with llvm.maximum semantics.
@ FMulAdd
Sum of float products with llvm.fmuladd(a * b + sum).
@ FMul
Product of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ And
Bitwise or logical AND of integers.
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ FMin
FP min implemented in terms of select(cmp()).
@ FMinNum
FP min with llvm.minnum semantics including NaNs.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ FAdd
Sum of floats.
@ FMaximumNum
FP max with llvm.maximumnum semantics.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
DWARFExpression::Operation Op
LLVM_ABI bool extractBranchWeights(const MDNode *ProfileData, SmallVectorImpl< uint32_t > &Weights)
Extract branch weights from MD_prof metadata.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
LLVM_ABI Value * createOrderedReduction(IRBuilderBase &B, RecurKind RdxKind, Value *Src, Value *Start)
Create an ordered reduction intrinsic using the given recurrence kind RdxKind.
ArrayRef< Type * > getContainedTypes(Type *const &Ty)
Returns the types contained in Ty.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isVectorIntrinsicWithOverloadTypeAtArg(Intrinsic::ID ID, int OpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic is overloaded on the type of the operand at index OpdI...
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Struct to hold various analysis needed for cost computations.
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
Definition VPlan.cpp:1924
static bool executesAtMostOnce(const VPlan &Plan, ElementCount VF)
Returns true if the vector loop body of Plan is known to execute at most once at VF,...
TargetTransformInfo::TargetCostKind CostKind
The frequency with which a recipe executes, relative to the entry of the loop region.
Definition VPlan.h:1169
void execute(VPTransformState &State) override
Generate the phi nodes.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this first-order recurrence phi recipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
An overlay for VPIRInstructions wrapping PHI nodes enabling convenient use cast/dyn_cast/isa and exec...
Definition VPlan.h:1786
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
PHINode & getIRPhi() const
Definition VPlan.h:1799
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
void execute(VPTransformState &State) override
Generate the instruction.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPRecipeWithIRFlags(VPRecipeTy SC, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1114
InstructionCost getCostForRecipeWithOpcode(unsigned Opcode, ElementCount VF, VPCostContext &Ctx) const
Compute the cost for this recipe for VF, using Opcode and Ctx.
SmallDenseMap< const VPBasicBlock *, BasicBlock * > VPBB2IRBB
A mapping of each VPBasicBlock to the corresponding BasicBlock.
VPTransformState holds information passed down when "executing" a VPlan, needed for generating the ou...
struct llvm::VPTransformState::CFGState CFG
IRBuilderBase & Builder
Hold a reference to the IRBuilder used to generate output IR code.
Value * get(const VPValue *Def, bool NeedsSingleScalar=false)
Get the generated vector Value for a given VPValue Def if NeedsSingleScalar is false,...
Definition VPlan.cpp:282
ElementCount VF
The chosen Vectorization Factor of the loop being vectorized.
void execute(VPTransformState &State) override
Generate the wide load or gather.
VPRecipeBase * getAsRecipe() override
Return a VPRecipeBase* to the current object.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenLoadEVLRecipe.
VPValue * getEVL() const
Return the EVL operand.
Definition VPlan.h:3887
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate a wide load or gather.
VPRecipeBase * getAsRecipe() override
Return a VPRecipeBase* to the current object.
VPValue * getStoredValue() const
Return the address accessed by this recipe.
Definition VPlan.h:3989
void execute(VPTransformState &State) override
Generate the wide store or scatter.
VPRecipeBase * getAsRecipe() override
Return a VPRecipeBase* to the current object.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenStoreEVLRecipe.
VPValue * getEVL() const
Return the EVL operand.
Definition VPlan.h:3992
void execute(VPTransformState &State) override
Generate a wide store or scatter.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPRecipeBase * getAsRecipe() override
Return a VPRecipeBase* to the current object.
VPValue * getStoredValue() const
Return the value stored by this recipe.
Definition VPlan.h:3937