LLVM 24.0.0git
VPlan.h
Go to the documentation of this file.
1//===- VPlan.h - Represent A Vectorizer Plan --------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file contains the declarations of the Vectorization Plan base classes:
11/// 1. VPBasicBlock and VPRegionBlock that inherit from a common pure virtual
12/// VPBlockBase, together implementing a Hierarchical CFG;
13/// 2. Pure virtual VPRecipeBase serving as the base class for recipes contained
14/// within VPBasicBlocks;
15/// 3. Pure virtual VPSingleDefRecipe serving as a base class for recipes that
16/// also inherit from VPValue.
17/// 4. VPInstruction, a concrete Recipe and VPUser modeling a single planned
18/// instruction;
19/// 5. The VPlan class holding a candidate for vectorization;
20/// These are documented in docs/VectorizationPlan.rst.
21//
22//===----------------------------------------------------------------------===//
23
24#ifndef LLVM_TRANSFORMS_VECTORIZE_VPLAN_H
25#define LLVM_TRANSFORMS_VECTORIZE_VPLAN_H
26
27#include "VPlanValue.h"
28#include "llvm/ADT/Bitfields.h"
29#include "llvm/ADT/MapVector.h"
32#include "llvm/ADT/Twine.h"
33#include "llvm/ADT/ilist.h"
34#include "llvm/ADT/ilist_node.h"
38#include "llvm/IR/DebugLoc.h"
39#include "llvm/IR/FMF.h"
40#include "llvm/IR/Operator.h"
44#include <cassert>
45#include <cstddef>
46#include <functional>
47#include <optional>
48#include <string>
49#include <utility>
50#include <variant>
51
52namespace llvm {
53
54class BasicBlock;
55class DominatorTree;
57class IRBuilderBase;
58struct VPTransformState;
59class raw_ostream;
61class SCEV;
62class SCEVPredicate;
63class Type;
64class VPBasicBlock;
66template <typename InserterTy = VPBuilderDefaultInserter> class VPBuilderBase;
68class VPDominatorTree;
69class VPRegionBlock;
70class VPlan;
71class VPLane;
73class Value;
75
76struct VPCostContext;
77
78using VPlanPtr = std::unique_ptr<VPlan>;
79
80/// \enum UncountableExitStyle
81/// Different methods of handling early exits.
82///
84 /// No side effects to worry about, so we can process any uncountable exits
85 /// in the loop and branch either to the middle block if the trip count was
86 /// reached, or an early exitblock to determine which exit was taken.
88 /// All memory operations other than the load(s) required to determine whether
89 /// an uncountable exit occurre will be masked based on that condition. If an
90 /// uncountable exit is taken, then all lanes before the exiting lane will
91 /// complete, leaving just the final lane to execute in the scalar tail.
93};
94
95/// VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
96/// A VPBlockBase can be either a VPBasicBlock or a VPRegionBlock.
98 friend class VPBlockUtils;
99
100protected:
101 /// An enumeration for keeping track of the concrete subclass of VPBlockBase
102 /// that are actually instantiated. Values of this enumeration are kept in the
103 /// SubclassID field of the VPBlockBase objects. They are used for concrete
104 /// type identification.
105 using VPBlockTy = enum : unsigned char {
106 VPRegionBlockSC,
107 VPBasicBlockSC,
108 VPIRBasicBlockSC
109 };
110
111private:
112 /// An optional name for the block.
113 std::string Name;
114
115 /// The immediate VPRegionBlock which this VPBlockBase belongs to, or null if
116 /// it is a topmost VPBlockBase.
117 VPRegionBlock *Parent = nullptr;
118
119 /// List of predecessor blocks.
121
122 /// List of successor blocks.
124
125 /// VPlan containing the block. Set when the block is created via VPlan
126 /// helpers.
127 VPlan *Plan = nullptr;
128
129 /// Subclass identifier (for isa/dyn_cast).
130 const VPBlockTy SubclassID;
131
132 /// Unique number, used as node number in the dominator tree.
133 unsigned Number;
134
135 /// Add \p Successor as the last successor to this block.
136 void appendSuccessor(VPBlockBase *Successor) {
137 assert(Successor && "Cannot add nullptr successor!");
138 Successors.push_back(Successor);
139 }
140
141 /// Add \p Predecessor as the last predecessor to this block.
142 void appendPredecessor(VPBlockBase *Predecessor) {
143 assert(Predecessor && "Cannot add nullptr predecessor!");
144 Predecessors.push_back(Predecessor);
145 }
146
147 /// Remove \p Predecessor from the predecessors of this block.
148 void removePredecessor(VPBlockBase *Predecessor) {
149 auto Pos = find(Predecessors, Predecessor);
150 assert(Pos && "Predecessor does not exist");
151 Predecessors.erase(Pos);
152 }
153
154 /// Remove \p Successor from the successors of this block.
155 void removeSuccessor(VPBlockBase *Successor) {
156 auto Pos = find(Successors, Successor);
157 assert(Pos && "Successor does not exist");
158 Successors.erase(Pos);
159 }
160
161 /// This function replaces one predecessor with another, useful when
162 /// trying to replace an old block in the CFG with a new one.
163 void replacePredecessor(VPBlockBase *Old, VPBlockBase *New) {
164 auto I = find(Predecessors, Old);
165 assert(I != Predecessors.end());
166 assert(Old->getParent() == New->getParent() &&
167 "replaced predecessor must have the same parent");
168 *I = New;
169 }
170
171 /// This function replaces one successor with another, useful when
172 /// trying to replace an old block in the CFG with a new one.
173 void replaceSuccessor(VPBlockBase *Old, VPBlockBase *New) {
174 auto I = find(Successors, Old);
175 assert(I != Successors.end());
176 assert(Old->getParent() == New->getParent() &&
177 "replaced successor must have the same parent");
178 *I = New;
179 }
180
181public:
183
184 virtual ~VPBlockBase() = default;
185
186 const std::string &getName() const { return Name; }
187
188 void setName(const Twine &newName) { Name = newName.str(); }
189
190 /// \return an ID for the concrete type of this object.
191 /// This is used to implement the classof checks. This should not be used
192 /// for any other purpose, as the values may change as LLVM evolves.
193 unsigned getVPBlockID() const { return SubclassID; }
194
195 VPRegionBlock *getParent() { return Parent; }
196 const VPRegionBlock *getParent() const { return Parent; }
197
198 /// \return A pointer to the plan containing the current block.
199 VPlan *getPlan() { return Plan; }
200 const VPlan *getPlan() const { return Plan; }
201
202 /// Sets the pointer of the plan containing the block.
203 void setPlan(VPlan *ParentPlan) { Plan = ParentPlan; }
204
205 void setParent(VPRegionBlock *P) { Parent = P; }
206
207 /// \return the VPBasicBlock that is the entry of this VPBlockBase,
208 /// recursively, if the latter is a VPRegionBlock. Otherwise, if this
209 /// VPBlockBase is a VPBasicBlock, it is returned.
210 const VPBasicBlock *getEntryBasicBlock() const;
211 VPBasicBlock *getEntryBasicBlock();
212
213 /// \return the VPBasicBlock that is the exiting this VPBlockBase,
214 /// recursively, if the latter is a VPRegionBlock. Otherwise, if this
215 /// VPBlockBase is a VPBasicBlock, it is returned.
216 const VPBasicBlock *getExitingBasicBlock() const;
217 VPBasicBlock *getExitingBasicBlock();
218
219 const VPBlocksTy &getSuccessors() const { return Successors; }
220 VPBlocksTy &getSuccessors() { return Successors; }
221
222 /// Returns true if this block has any successors.
223 bool hasSuccessors() const { return !Successors.empty(); }
224 /// Returns true if this block has any predecessors.
225 bool hasPredecessors() const { return !Predecessors.empty(); }
226
229
230 const VPBlocksTy &getPredecessors() const { return Predecessors; }
231 VPBlocksTy &getPredecessors() { return Predecessors; }
232
233 /// \return the successor of this VPBlockBase if it has a single successor.
234 /// Otherwise return a null pointer.
236 return (Successors.size() == 1 ? *Successors.begin() : nullptr);
237 }
238
239 /// \return the predecessor of this VPBlockBase if it has a single
240 /// predecessor. Otherwise return a null pointer.
242 return (Predecessors.size() == 1 ? *Predecessors.begin() : nullptr);
243 }
244
245 size_t getNumSuccessors() const { return Successors.size(); }
246 size_t getNumPredecessors() const { return Predecessors.size(); }
247
248 /// An Enclosing Block of a block B is any block containing B, including B
249 /// itself. \return the closest enclosing block starting from "this", which
250 /// has successors. \return the root enclosing block if all enclosing blocks
251 /// have no successors.
252 VPBlockBase *getEnclosingBlockWithSuccessors();
253
254 /// \return the closest enclosing block starting from "this", which has
255 /// predecessors. \return the root enclosing block if all enclosing blocks
256 /// have no predecessors.
257 VPBlockBase *getEnclosingBlockWithPredecessors();
258
259 /// \return the successors either attached directly to this VPBlockBase or, if
260 /// this VPBlockBase is the exit block of a VPRegionBlock and has no
261 /// successors of its own, search recursively for the first enclosing
262 /// VPRegionBlock that has successors and return them. If no such
263 /// VPRegionBlock exists, return the (empty) successors of the topmost
264 /// VPBlockBase reached.
266 return getEnclosingBlockWithSuccessors()->getSuccessors();
267 }
268
269 /// \return the hierarchical predecessor of this VPBlockBase if it has a
270 /// single hierarchical predecessor. Otherwise return a null pointer.
274
275 /// Set a given VPBlockBase \p Successor as the single successor of this
276 /// VPBlockBase. This VPBlockBase is not added as predecessor of \p Successor.
277 /// This VPBlockBase must have no successors.
279 assert(Successors.empty() && "Setting one successor when others exist.");
280 assert(Successor->getParent() == getParent() &&
281 "connected blocks must have the same parent");
282 appendSuccessor(Successor);
283 }
284
285 /// Set two given VPBlockBases \p IfTrue and \p IfFalse to be the two
286 /// successors of this VPBlockBase. This VPBlockBase is not added as
287 /// predecessor of \p IfTrue or \p IfFalse. This VPBlockBase must have no
288 /// successors.
289 void setTwoSuccessors(VPBlockBase *IfTrue, VPBlockBase *IfFalse) {
290 assert(Successors.empty() && "Setting two successors when others exist.");
291 appendSuccessor(IfTrue);
292 appendSuccessor(IfFalse);
293 }
294
295 /// Set each VPBasicBlock in \p NewPreds as predecessor of this VPBlockBase.
296 /// This VPBlockBase must have no predecessors. This VPBlockBase is not added
297 /// as successor of any VPBasicBlock in \p NewPreds.
299 assert(Predecessors.empty() && "Block predecessors already set.");
300 for (auto *Pred : NewPreds)
301 appendPredecessor(Pred);
302 }
303
304 /// Set each VPBasicBlock in \p NewSuccss as successor of this VPBlockBase.
305 /// This VPBlockBase must have no successors. This VPBlockBase is not added
306 /// as predecessor of any VPBasicBlock in \p NewSuccs.
308 assert(Successors.empty() && "Block successors already set.");
309 for (auto *Succ : NewSuccs)
310 appendSuccessor(Succ);
311 }
312
313 /// Remove all the predecessor of this block.
314 void clearPredecessors() { Predecessors.clear(); }
315
316 /// Remove all the successors of this block.
317 void clearSuccessors() { Successors.clear(); }
318
319 /// Swap predecessors of the block. The block must have exactly 2
320 /// predecessors.
322 assert(Predecessors.size() == 2 && "must have 2 predecessors to swap");
323 std::swap(Predecessors[0], Predecessors[1]);
324 }
325
326 /// Swap successors of the block. The block must have exactly 2 successors.
327 // TODO: This should be part of introducing conditional branch recipes rather
328 // than being independent.
330 assert(Successors.size() == 2 && "must have 2 successors to swap");
331 std::swap(Successors[0], Successors[1]);
332 }
333
334 /// Returns the index for \p Pred in the blocks predecessors list.
335 unsigned getIndexForPredecessor(const VPBlockBase *Pred) const {
336 assert(count(Predecessors, Pred) == 1 &&
337 "must have Pred exactly once in Predecessors");
338 return std::distance(Predecessors.begin(), find(Predecessors, Pred));
339 }
340
341 /// Returns the index for \p Succ in the blocks successor list.
342 unsigned getIndexForSuccessor(const VPBlockBase *Succ) const {
343 assert(count(Successors, Succ) == 1 &&
344 "must have Succ exactly once in Successors");
345 return std::distance(Successors.begin(), find(Successors, Succ));
346 }
347
348 /// Return the unique number of the block.
349 unsigned getNumber() const { return Number; }
350
351 /// Set the unique number of the block, used for dominator tree.
352 void setNumber(unsigned N) { Number = N; }
353
354 /// The method which generates the output IR that correspond to this
355 /// VPBlockBase, thereby "executing" the VPlan.
356 virtual void execute(VPTransformState *State) = 0;
357
358 /// Return the cost of the block.
360
361 void printAsOperand(raw_ostream &OS, bool PrintType = false) const {
362 OS << getName();
363 }
364
365#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
366 /// Print plain-text dump of this VPBlockBase to \p O, prefixing all lines
367 /// with \p Indent. \p SlotTracker is used to print unnamed VPValue's using
368 /// consequtive numbers.
369 ///
370 /// Note that the numbering is applied to the whole VPlan, so printing
371 /// individual blocks is consistent with the whole VPlan printing.
372 virtual void print(raw_ostream &O, const Twine &Indent,
373 VPSlotTracker &SlotTracker) const = 0;
374
375 /// Print plain-text dump of this VPlan to \p O.
376 void print(raw_ostream &O) const;
377
378 /// Print the successors of this block to \p O, prefixing all lines with \p
379 /// Indent.
380 void printSuccessors(raw_ostream &O, const Twine &Indent) const;
381
382 /// Dump this VPBlockBase to dbgs().
383 LLVM_DUMP_METHOD void dump() const { print(dbgs()); }
384#endif
385
386 /// Clone the current block and it's recipes without updating the operands of
387 /// the cloned recipes, including all blocks in the single-entry single-exit
388 /// region for VPRegionBlocks.
389 virtual VPBlockBase *clone() = 0;
390
391protected:
392 VPBlockBase(VPBlockTy SC, const std::string &N) : Name(N), SubclassID(SC) {}
393};
394
395/// VPRecipeBase is a base class modeling a sequence of one or more output IR
396/// instructions. VPRecipeBase owns the VPValues it defines through VPDef
397/// and is responsible for deleting its defined values. Single-value
398/// recipes must inherit from VPSingleDef instead of inheriting from both
399/// VPRecipeBase and VPValue separately.
401 : public ilist_node_with_parent<VPRecipeBase, VPBasicBlock>,
402 public VPDef,
403 public VPUser {
404 friend VPBasicBlock;
405 friend class VPBlockUtils;
406
407 /// Each VPRecipe belongs to a single VPBasicBlock.
408 VPBasicBlock *Parent = nullptr;
409
410 /// The debug location for the recipe.
411 DebugLoc DL;
412
413public:
414 /// An enumeration for keeping track of the concrete subclass of VPRecipeBase
415 /// that is actually instantiated. Values of this enumeration are kept in the
416 /// SubclassID field of the VPRecipeBase objects. They are used for concrete
417 /// type identification.
418 using VPRecipeTy = enum : unsigned char {
419 VPBranchOnMaskSC,
420 VPDerivedIVSC,
421 VPExpandSCEVSC,
422 VPExpressionSC,
423 VPIRInstructionSC,
424 VPInstructionSC,
425 VPInterleaveEVLSC,
426 VPInterleaveSC,
427 VPReductionEVLSC,
428 VPReductionSC,
429 VPReplicateSC,
430 VPScalarIVStepsSC,
431 VPVectorPointerSC,
432 VPVectorEndPointerSC,
433 VPWidenCallSC,
434 VPWidenCanonicalIVSC,
435 VPWidenCastSC,
436 VPWidenGEPSC,
437 VPWidenIntrinsicSC,
438 VPWidenMemIntrinsicSC,
439 VPWidenLoadEVLSC,
440 VPWidenLoadSC,
441 VPWidenStoreEVLSC,
442 VPWidenStoreSC,
443 VPWidenSC,
444 VPBlendSC,
445 VPHistogramSC,
446 // START: Phi-like recipes. Need to be kept together.
447 VPWidenPHISC,
448 VPPredInstPHISC,
449 // START: SubclassID for recipes that inherit VPHeaderPHIRecipe.
450 // VPHeaderPHIRecipe need to be kept together.
451 VPCurrentIterationPHISC,
452 VPActiveLaneMaskPHISC,
453 VPFirstOrderRecurrencePHISC,
454 VPWidenIntOrFpInductionSC,
455 VPWidenPointerInductionSC,
456 VPReductionPHISC,
457 // END: SubclassID for recipes that inherit VPHeaderPHIRecipe
458 // END: Phi-like recipes
459 VPFirstPHISC = VPWidenPHISC,
460 VPFirstHeaderPHISC = VPCurrentIterationPHISC,
461 VPLastHeaderPHISC = VPReductionPHISC,
462 VPLastPHISC = VPReductionPHISC,
463 };
464
467 : VPDef(), VPUser(Operands), DL(DL), SubclassID(SC) {}
468
469 ~VPRecipeBase() override = default;
470
471 /// Clone the current recipe.
472 virtual VPRecipeBase *clone() = 0;
473
474 /// \return the VPBasicBlock which this VPRecipe belongs to.
475 VPBasicBlock *getParent() { return Parent; }
476 const VPBasicBlock *getParent() const { return Parent; }
477
478 /// \return the VPRegionBlock which the recipe belongs to.
479 VPRegionBlock *getRegion();
480 const VPRegionBlock *getRegion() const;
481
482 /// The method which generates the output IR instructions that correspond to
483 /// this VPRecipe, thereby "executing" the VPlan.
484 virtual void execute(VPTransformState &State) = 0;
485
486 /// Return the cost of this recipe, taking into account if the cost
487 /// computation should be skipped and the ForceTargetInstructionCost flag.
488 /// Also takes care of printing the cost for debugging.
490
491 /// Insert an unlinked recipe into a basic block immediately before
492 /// the specified recipe.
493 void insertBefore(VPRecipeBase *InsertPos);
494 /// Insert an unlinked recipe into \p BB immediately before the insertion
495 /// point \p IP;
496 void insertBefore(VPBasicBlock &BB, iplist<VPRecipeBase>::iterator IP);
497
498 /// Insert an unlinked Recipe into a basic block immediately after
499 /// the specified Recipe.
500 void insertAfter(VPRecipeBase *InsertPos);
501
502 /// Unlink this recipe from its current VPBasicBlock and insert it into
503 /// the VPBasicBlock that MovePos lives in, right after MovePos.
504 void moveAfter(VPRecipeBase *MovePos);
505
506 /// Unlink this recipe and insert into BB before I.
507 ///
508 /// \pre I is a valid iterator into BB.
509 void moveBefore(VPBasicBlock &BB, iplist<VPRecipeBase>::iterator I);
510
511 /// This method unlinks 'this' from the containing basic block, but does not
512 /// delete it.
513 void removeFromParent();
514
515 /// This method unlinks 'this' from the containing basic block and deletes it.
516 ///
517 /// \returns an iterator pointing to the element after the erased one
519
520 /// \return an ID for the concrete type of this object.
521 VPRecipeTy getVPRecipeID() const { return SubclassID; }
522
523 /// Method to support type inquiry through isa, cast, and dyn_cast.
524 static inline bool classof(const VPDef *D) {
525 // All VPDefs are also VPRecipeBases.
526 return true;
527 }
528
529 static inline bool classof(const VPUser *U) { return true; }
530
531 /// Returns true if the recipe may have side-effects.
532 bool mayHaveSideEffects() const;
533
534 /// Return true if we can safely execute this recipe unconditionally even if
535 /// it is masked originally.
536 bool isSafeToSpeculativelyExecute() const;
537
538 /// Returns true for PHI-like recipes.
539 bool isPhi() const;
540
541 /// Returns true if the recipe may read from memory.
542 bool mayReadFromMemory() const;
543
544 /// Returns true if the recipe may write to memory.
545 bool mayWriteToMemory() const;
546
547 /// Returns true if the recipe may read from or write to memory.
548 bool mayReadOrWriteMemory() const {
550 }
551
552 /// Returns the debug location of the recipe.
553 DebugLoc getDebugLoc() const { return DL; }
554
555 /// Set the recipe's debug location to \p NewDL.
556 void setDebugLoc(DebugLoc NewDL) { DL = NewDL; }
557
558#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
559 /// Dump the recipe to stderr (for debugging).
560 void dump() const;
561
562 /// Print the recipe, delegating to printRecipe().
563 void print(raw_ostream &O, const Twine &Indent,
565#endif
566
567private:
568 /// Subclass identifier (for isa/dyn_cast).
569 const VPRecipeTy SubclassID;
570
571protected:
572 /// Compute the cost of this recipe either using a recipe's specialized
573 /// implementation or using the legacy cost model and the underlying
574 /// instructions.
575 virtual InstructionCost computeCost(ElementCount VF,
576 VPCostContext &Ctx) const;
577
578#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
579 /// Each concrete VPRecipe prints itself, without printing common information,
580 /// like debug info or metadata.
581 virtual void printRecipe(raw_ostream &O, const Twine &Indent,
582 VPSlotTracker &SlotTracker) const = 0;
583#endif
584};
585
586// Helper macro to define common classof implementations for recipes.
587#define VP_CLASSOF_IMPL(VPRecipeID) \
588 static inline bool classof(const VPRecipeBase *R) { \
589 return R->getVPRecipeID() == VPRecipeID; \
590 } \
591 static inline bool classof(const VPValue *V) { \
592 auto *R = V->getDefiningRecipe(); \
593 return R && R->getVPRecipeID() == VPRecipeID; \
594 } \
595 static inline bool classof(const VPUser *U) { \
596 auto *R = dyn_cast<VPRecipeBase>(U); \
597 return R && R->getVPRecipeID() == VPRecipeID; \
598 } \
599 static inline bool classof(const VPSingleDefRecipe *R) { \
600 return R->getVPRecipeID() == VPRecipeID; \
601 }
602
603/// Compute the scalar result type for an IR \p Opcode given \p Operands.
604LLVM_ABI Type *computeScalarTypeForInstruction(unsigned Opcode,
606
607/// VPSingleDefRecipe is a base class for recipes that model a sequence of one
608/// or more output IR that define a single result VPValue. Note that
609/// VPSingleDefRecipe must inherit from VPRecipeBase before VPSingleDefValue.
611 public VPSingleDefValue {
612public:
616
619 : VPRecipeBase(SC, Operands, DL), VPSingleDefValue(this, UV) {}
620
622 Value *UV = nullptr, DebugLoc DL = DebugLoc::getUnknown())
623 : VPRecipeBase(SC, Operands, DL), VPSingleDefValue(this, UV, ResultTy) {}
624
625 static inline bool classof(const VPRecipeBase *R) {
626 switch (R->getVPRecipeID()) {
627 case VPRecipeBase::VPDerivedIVSC:
628 case VPRecipeBase::VPExpandSCEVSC:
629 case VPRecipeBase::VPExpressionSC:
630 case VPRecipeBase::VPInstructionSC:
631 case VPRecipeBase::VPReductionEVLSC:
632 case VPRecipeBase::VPReductionSC:
633 case VPRecipeBase::VPReplicateSC:
634 case VPRecipeBase::VPScalarIVStepsSC:
635 case VPRecipeBase::VPVectorPointerSC:
636 case VPRecipeBase::VPVectorEndPointerSC:
637 case VPRecipeBase::VPWidenCallSC:
638 case VPRecipeBase::VPWidenCanonicalIVSC:
639 case VPRecipeBase::VPWidenCastSC:
640 case VPRecipeBase::VPWidenGEPSC:
641 case VPRecipeBase::VPWidenIntrinsicSC:
642 case VPRecipeBase::VPWidenMemIntrinsicSC:
643 case VPRecipeBase::VPWidenSC:
644 case VPRecipeBase::VPBlendSC:
645 case VPRecipeBase::VPPredInstPHISC:
646 case VPRecipeBase::VPCurrentIterationPHISC:
647 case VPRecipeBase::VPActiveLaneMaskPHISC:
648 case VPRecipeBase::VPFirstOrderRecurrencePHISC:
649 case VPRecipeBase::VPWidenPHISC:
650 case VPRecipeBase::VPWidenIntOrFpInductionSC:
651 case VPRecipeBase::VPWidenPointerInductionSC:
652 case VPRecipeBase::VPReductionPHISC:
653 case VPRecipeBase::VPWidenLoadEVLSC:
654 case VPRecipeBase::VPWidenLoadSC:
655 return true;
656 case VPRecipeBase::VPBranchOnMaskSC:
657 case VPRecipeBase::VPInterleaveEVLSC:
658 case VPRecipeBase::VPInterleaveSC:
659 case VPRecipeBase::VPIRInstructionSC:
660 case VPRecipeBase::VPWidenStoreEVLSC:
661 case VPRecipeBase::VPWidenStoreSC:
662 case VPRecipeBase::VPHistogramSC:
663 return false;
664 }
665 llvm_unreachable("Unhandled VPRecipeID");
666 }
667
668 static inline bool classof(const VPValue *V) {
669 auto *R = V->getDefiningRecipe();
670 return R && classof(R);
671 }
672
673 static inline bool classof(const VPUser *U) {
674 auto *R = dyn_cast<VPRecipeBase>(U);
675 return R && classof(R);
676 }
677
678 VPSingleDefRecipe *clone() override = 0;
679
680 /// Returns the underlying instruction.
687
688#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
689 /// Print this VPSingleDefRecipe to dbgs() (for debugging).
690 LLVM_DUMP_METHOD void dump() const;
691#endif
692};
693
694/// Class to record and manage LLVM IR flags.
697 enum class OperationType : unsigned char {
698 Cmp,
699 FCmp,
700 OverflowingBinOp,
701 Trunc,
702 DisjointOp,
703 PossiblyExactOp,
704 GEPOp,
705 FPMathOp,
706 NonNegOp,
707 ReductionOp,
708 Other
709 };
710
711public:
712 struct WrapFlagsTy {
713 char HasNUW : 1;
714 char HasNSW : 1;
715
718 };
719
721 char HasNUW : 1;
722 char HasNSW : 1;
723
725 };
726
731
733 char NonNeg : 1;
734 NonNegFlagsTy(bool IsNonNeg) : NonNeg(IsNonNeg) {}
735 };
736
737private:
738 struct ExactFlagsTy {
739 char IsExact : 1;
740 ExactFlagsTy(bool Exact) : IsExact(Exact) {}
741 };
742 struct FastMathFlagsTy {
743 char AllowReassoc : 1;
744 char NoNaNs : 1;
745 char NoInfs : 1;
746 char NoSignedZeros : 1;
747 char AllowReciprocal : 1;
748 char AllowContract : 1;
749 char ApproxFunc : 1;
750
751 LLVM_ABI_FOR_TEST FastMathFlagsTy(const FastMathFlags &FMF);
752 };
753 /// Holds both the predicate and fast-math flags for floating-point
754 /// comparisons.
755 struct FCmpFlagsTy {
756 uint8_t CmpPredStorage;
757 FastMathFlagsTy FMFs;
758 };
759 /// Holds reduction-specific flags: RecurKind, IsOrdered, IsInLoop, and FMFs.
760 struct ReductionFlagsTy {
761 // RecurKind has ~26 values, needs 5 bits but uses 6 bits to account for
762 // additional kinds.
763 unsigned char Kind : 6;
764 // TODO: Derive order/in-loop from plan and remove here.
765 unsigned char IsOrdered : 1;
766 unsigned char IsInLoop : 1;
767 FastMathFlagsTy FMFs;
768
769 ReductionFlagsTy(RecurKind Kind, bool IsOrdered, bool IsInLoop,
770 FastMathFlags FMFs)
771 : Kind(static_cast<unsigned char>(Kind)), IsOrdered(IsOrdered),
772 IsInLoop(IsInLoop), FMFs(FMFs) {}
773 };
774
775 OperationType OpType;
776
777 union {
782 ExactFlagsTy ExactFlags;
785 FastMathFlagsTy FMFs;
786 FCmpFlagsTy FCmpFlags;
787 ReductionFlagsTy ReductionFlags;
789 };
790
791public:
792 VPIRFlags() : OpType(OperationType::Other), AllFlags() {}
793
795 if (auto *FCmp = dyn_cast<FCmpInst>(&I)) {
796 OpType = OperationType::FCmp;
798 FCmp->getPredicate());
799 assert(getPredicate() == FCmp->getPredicate() && "predicate truncated");
800 FCmpFlags.FMFs = FCmp->getFastMathFlags();
801 } else if (auto *Op = dyn_cast<CmpInst>(&I)) {
802 OpType = OperationType::Cmp;
804 Op->getPredicate());
805 assert(getPredicate() == Op->getPredicate() && "predicate truncated");
806 } else if (auto *Op = dyn_cast<PossiblyDisjointInst>(&I)) {
807 OpType = OperationType::DisjointOp;
808 DisjointFlags.IsDisjoint = Op->isDisjoint();
809 } else if (auto *Op = dyn_cast<OverflowingBinaryOperator>(&I)) {
810 OpType = OperationType::OverflowingBinOp;
811 WrapFlags = {Op->hasNoUnsignedWrap(), Op->hasNoSignedWrap()};
812 } else if (auto *Op = dyn_cast<TruncInst>(&I)) {
813 OpType = OperationType::Trunc;
814 TruncFlags = {Op->hasNoUnsignedWrap(), Op->hasNoSignedWrap()};
815 } else if (auto *Op = dyn_cast<PossiblyExactOperator>(&I)) {
816 OpType = OperationType::PossiblyExactOp;
817 ExactFlags.IsExact = Op->isExact();
818 } else if (auto *GEP = dyn_cast<GetElementPtrInst>(&I)) {
819 OpType = OperationType::GEPOp;
820 GEPFlagsStorage = GEP->getNoWrapFlags().getRaw();
821 assert(getGEPNoWrapFlags() == GEP->getNoWrapFlags() &&
822 "wrap flags truncated");
823 } else if (auto *PNNI = dyn_cast<PossiblyNonNegInst>(&I)) {
824 OpType = OperationType::NonNegOp;
825 NonNegFlags.NonNeg = PNNI->hasNonNeg();
826 } else if (auto *Op = dyn_cast<FPMathOperator>(&I)) {
827 OpType = OperationType::FPMathOp;
828 FMFs = Op->getFastMathFlags();
829 }
830 }
831
832 VPIRFlags(CmpInst::Predicate Pred) : OpType(OperationType::Cmp), AllFlags() {
834 assert(getPredicate() == Pred && "predicate truncated");
835 }
836
838 : OpType(OperationType::FCmp), AllFlags() {
840 assert(getPredicate() == Pred && "predicate truncated");
841 FCmpFlags.FMFs = FMFs;
842 }
843
845 : OpType(OperationType::OverflowingBinOp), AllFlags() {
846 this->WrapFlags = WrapFlags;
847 }
848
850 : OpType(OperationType::Trunc), AllFlags() {
851 this->TruncFlags = TruncFlags;
852 }
853
854 VPIRFlags(FastMathFlags FMFs) : OpType(OperationType::FPMathOp), AllFlags() {
855 this->FMFs = FMFs;
856 }
857
859 : OpType(OperationType::DisjointOp), AllFlags() {
860 this->DisjointFlags = DisjointFlags;
861 }
862
864 : OpType(OperationType::NonNegOp), AllFlags() {
865 this->NonNegFlags = NonNegFlags;
866 }
867
868 VPIRFlags(ExactFlagsTy ExactFlags)
869 : OpType(OperationType::PossiblyExactOp), AllFlags() {
870 this->ExactFlags = ExactFlags;
871 }
872
874 : OpType(OperationType::GEPOp), AllFlags() {
875 GEPFlagsStorage = GEPFlags.getRaw();
876 }
877
878 VPIRFlags(RecurKind Kind, bool IsOrdered, bool IsInLoop, FastMathFlags FMFs)
879 : OpType(OperationType::ReductionOp), AllFlags() {
880 ReductionFlags = ReductionFlagsTy(Kind, IsOrdered, IsInLoop, FMFs);
881 }
882
884 OpType = Other.OpType;
885 AllFlags[0] = Other.AllFlags[0];
886 AllFlags[1] = Other.AllFlags[1];
887 }
888
889 /// Only keep flags also present in \p Other. \p Other must have the same
890 /// OpType as the current object.
891 void intersectFlags(const VPIRFlags &Other);
892
893 /// Drop all poison-generating flags.
895 // NOTE: This needs to be kept in-sync with
896 // Instruction::dropPoisonGeneratingFlags.
897 switch (OpType) {
898 case OperationType::OverflowingBinOp:
899 WrapFlags.HasNUW = false;
900 WrapFlags.HasNSW = false;
901 break;
902 case OperationType::Trunc:
903 TruncFlags.HasNUW = false;
904 TruncFlags.HasNSW = false;
905 break;
906 case OperationType::DisjointOp:
907 DisjointFlags.IsDisjoint = false;
908 break;
909 case OperationType::PossiblyExactOp:
910 ExactFlags.IsExact = false;
911 break;
912 case OperationType::GEPOp:
913 GEPFlagsStorage = 0;
914 break;
915 case OperationType::FPMathOp:
916 case OperationType::FCmp:
917 case OperationType::ReductionOp:
918 getFMFsRef().NoNaNs = false;
919 getFMFsRef().NoInfs = false;
920 break;
921 case OperationType::NonNegOp:
922 NonNegFlags.NonNeg = false;
923 break;
924 case OperationType::Cmp:
925 case OperationType::Other:
926 break;
927 }
928 }
929
930 /// Apply the IR flags to \p I.
931 void applyFlags(Instruction &I) const {
932 switch (OpType) {
933 case OperationType::OverflowingBinOp:
934 I.setHasNoUnsignedWrap(WrapFlags.HasNUW);
935 I.setHasNoSignedWrap(WrapFlags.HasNSW);
936 break;
937 case OperationType::Trunc:
938 I.setHasNoUnsignedWrap(TruncFlags.HasNUW);
939 I.setHasNoSignedWrap(TruncFlags.HasNSW);
940 break;
941 case OperationType::DisjointOp:
942 cast<PossiblyDisjointInst>(&I)->setIsDisjoint(DisjointFlags.IsDisjoint);
943 break;
944 case OperationType::PossiblyExactOp:
945 I.setIsExact(ExactFlags.IsExact);
946 break;
947 case OperationType::GEPOp:
948 cast<GetElementPtrInst>(&I)->setNoWrapFlags(
950 break;
951 case OperationType::FPMathOp:
952 case OperationType::FCmp: {
953 const FastMathFlagsTy &F = getFMFsRef();
954 I.setHasAllowReassoc(F.AllowReassoc);
955 I.setHasNoNaNs(F.NoNaNs);
956 I.setHasNoInfs(F.NoInfs);
957 I.setHasNoSignedZeros(F.NoSignedZeros);
958 I.setHasAllowReciprocal(F.AllowReciprocal);
959 I.setHasAllowContract(F.AllowContract);
960 I.setHasApproxFunc(F.ApproxFunc);
961 break;
962 }
963 case OperationType::NonNegOp:
964 I.setNonNeg(NonNegFlags.NonNeg);
965 break;
966 case OperationType::ReductionOp:
967 llvm_unreachable("reduction ops should not use applyFlags");
968 case OperationType::Cmp:
969 case OperationType::Other:
970 break;
971 }
972 }
973
975 assert((OpType == OperationType::Cmp || OpType == OperationType::FCmp) &&
976 "recipe doesn't have a compare predicate");
977 uint8_t Storage = OpType == OperationType::FCmp ? FCmpFlags.CmpPredStorage
980 }
981
983 assert((OpType == OperationType::Cmp || OpType == OperationType::FCmp) &&
984 "recipe doesn't have a compare predicate");
985 if (OpType == OperationType::FCmp)
987 else
989 assert(getPredicate() == Pred && "predicate truncated");
990 }
991
995
996 /// Returns true if the recipe has a comparison predicate.
997 bool hasPredicate() const {
998 return OpType == OperationType::Cmp || OpType == OperationType::FCmp;
999 }
1000
1001 /// Returns true if the recipe has fast-math flags.
1002 bool hasFastMathFlags() const {
1003 return OpType == OperationType::FPMathOp || OpType == OperationType::FCmp ||
1004 OpType == OperationType::ReductionOp;
1005 }
1006
1008
1009 bool hasNoUnsignedWrap() const {
1010 switch (OpType) {
1011 case OperationType::OverflowingBinOp:
1012 return WrapFlags.HasNUW;
1013 case OperationType::Trunc:
1014 return TruncFlags.HasNUW;
1015 default:
1016 llvm_unreachable("recipe doesn't have a NUW flag");
1017 }
1018 }
1019
1020 bool hasNoSignedWrap() const {
1021 switch (OpType) {
1022 case OperationType::OverflowingBinOp:
1023 return WrapFlags.HasNSW;
1024 case OperationType::Trunc:
1025 return TruncFlags.HasNSW;
1026 default:
1027 llvm_unreachable("recipe doesn't have a NSW flag");
1028 }
1029 }
1030
1032 switch (OpType) {
1033 case OperationType::OverflowingBinOp:
1034 case OperationType::Trunc:
1035 return {hasNoUnsignedWrap(), hasNoSignedWrap()};
1036 default:
1037 return {};
1038 }
1039 }
1040
1042 return {hasNoUnsignedWrap(), hasNoSignedWrap()};
1043 }
1044
1045 bool isDisjoint() const {
1046 assert(OpType == OperationType::DisjointOp &&
1047 "recipe cannot have a disjoing flag");
1048 return DisjointFlags.IsDisjoint;
1049 }
1050
1052 assert(OpType == OperationType::ReductionOp &&
1053 "recipe doesn't have reduction flags");
1054 return static_cast<RecurKind>(ReductionFlags.Kind);
1055 }
1056
1057 bool isReductionOrdered() const {
1058 assert(OpType == OperationType::ReductionOp &&
1059 "recipe doesn't have reduction flags");
1060 return ReductionFlags.IsOrdered;
1061 }
1062
1063 bool isReductionInLoop() const {
1064 assert(OpType == OperationType::ReductionOp &&
1065 "recipe doesn't have reduction flags");
1066 return ReductionFlags.IsInLoop;
1067 }
1068
1069private:
1070 /// Get a reference to the fast-math flags for FPMathOp, FCmp or ReductionOp.
1071 FastMathFlagsTy &getFMFsRef() {
1072 if (OpType == OperationType::FCmp)
1073 return FCmpFlags.FMFs;
1074 if (OpType == OperationType::ReductionOp)
1075 return ReductionFlags.FMFs;
1076 return FMFs;
1077 }
1078 const FastMathFlagsTy &getFMFsRef() const {
1079 if (OpType == OperationType::FCmp)
1080 return FCmpFlags.FMFs;
1081 if (OpType == OperationType::ReductionOp)
1082 return ReductionFlags.FMFs;
1083 return FMFs;
1084 }
1085
1086public:
1087 /// Returns default flags for \p Opcode and scalar \p ResultTy for opcodes
1088 /// that support it, asserts otherwise. Opcodes not supporting default flags
1089 /// include compares and ComputeReductionResult.
1090 LLVM_ABI_FOR_TEST static VPIRFlags getDefaultFlags(unsigned Opcode,
1091 Type *ResultTy = nullptr);
1092
1093#if !defined(NDEBUG)
1094 /// Returns true if the set flags are valid for \p Opcode.
1095 LLVM_ABI_FOR_TEST bool flagsValidForOpcode(unsigned Opcode) const;
1096
1097 /// Returns true if \p Opcode with scalar result type \p ResultTy has its
1098 /// required flags set.
1099 LLVM_ABI_FOR_TEST bool hasRequiredFlagsForOpcode(unsigned Opcode,
1100 Type *ResultTy) const;
1101#endif
1102
1103#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1104 void printFlags(raw_ostream &O) const;
1105#endif
1106};
1108
1109static_assert(sizeof(VPIRFlags) <= 3, "VPIRFlags should not grow");
1110
1111/// A pure-virtual common base class for recipes defining a single VPValue and
1112/// using IR flags.
1115 const VPIRFlags &Flags,
1117 : VPSingleDefRecipe(SC, Operands, DL), VPIRFlags(Flags) {}
1118
1120 Type *ResultTy, const VPIRFlags &Flags,
1122 : VPSingleDefRecipe(SC, Operands, ResultTy, /*UV=*/nullptr, DL),
1123 VPIRFlags(Flags) {}
1124
1125 static inline bool classof(const VPRecipeBase *R) {
1126 return R->getVPRecipeID() == VPRecipeBase::VPBlendSC ||
1127 R->getVPRecipeID() == VPRecipeBase::VPInstructionSC ||
1128 R->getVPRecipeID() == VPRecipeBase::VPWidenSC ||
1129 R->getVPRecipeID() == VPRecipeBase::VPWidenGEPSC ||
1130 R->getVPRecipeID() == VPRecipeBase::VPWidenCallSC ||
1131 R->getVPRecipeID() == VPRecipeBase::VPWidenCastSC ||
1132 R->getVPRecipeID() == VPRecipeBase::VPWidenIntrinsicSC ||
1133 R->getVPRecipeID() == VPRecipeBase::VPWidenMemIntrinsicSC ||
1134 R->getVPRecipeID() == VPRecipeBase::VPReductionSC ||
1135 R->getVPRecipeID() == VPRecipeBase::VPReductionEVLSC ||
1136 R->getVPRecipeID() == VPRecipeBase::VPReplicateSC ||
1137 R->getVPRecipeID() == VPRecipeBase::VPVectorEndPointerSC ||
1138 R->getVPRecipeID() == VPRecipeBase::VPVectorPointerSC ||
1139 R->getVPRecipeID() == VPRecipeBase::VPWidenCanonicalIVSC ||
1140 R->getVPRecipeID() == VPRecipeBase::VPDerivedIVSC;
1141 }
1142
1143 static inline bool classof(const VPUser *U) {
1144 auto *R = dyn_cast<VPRecipeBase>(U);
1145 return R && classof(R);
1146 }
1147
1148 static inline bool classof(const VPValue *V) {
1149 auto *R = V->getDefiningRecipe();
1150 return R && classof(R);
1151 }
1152
1154
1155 static inline bool classof(const VPSingleDefRecipe *R) {
1156 return classof(static_cast<const VPRecipeBase *>(R));
1157 }
1158
1159 void execute(VPTransformState &State) override = 0;
1160
1161 /// Compute the cost for this recipe for \p VF, using \p Opcode and \p Ctx.
1163 VPCostContext &Ctx) const;
1164};
1165
1166/// The frequency with which a recipe executes, relative to the entry of the
1167/// loop region. IsEstimated is set if any branch weight it was composed from
1168/// was estimated from static heuristics.
1171 const bool IsEstimated;
1172
1175 assert(Freq > BlockFrequency() && "execution frequency must be non-zero");
1176 }
1177};
1178
1179/// Helper to manage IR metadata for recipes. It filters out metadata that
1180/// cannot be propagated.
1183
1184 /// Name of the VPlan-internal metadata kind holding the execution frequency.
1185 static constexpr StringLiteral ExecutionFrequencyMDName =
1186 "vplan.execution.frequency";
1187
1188 /// Name of the VPlan-internal metadata kind holding estimated branch weights.
1189 static constexpr StringLiteral EstimatedProfileMDName =
1190 "vplan.prof.estimated";
1191
1192 /// Returns the ID of the metadata kind named \p Kind, taking the context from
1193 /// any attached node; all belong to the context of the VPlan's function.
1194 unsigned getMDKindID(StringRef Kind) const {
1195 assert(!Metadata.empty() && "no node to take the context from");
1196 return Metadata.front().second->getContext().getMDKindID(Kind);
1197 }
1198
1199 /// Returns the node attached under the VPlan-internal metadata kind named
1200 /// \p Kind, or nullptr if there is none.
1201 MDNode *getInternalMetadata(StringRef Kind) const {
1202 return Metadata.empty() ? nullptr : getMetadata(getMDKindID(Kind));
1203 }
1204
1205public:
1206 VPIRMetadata() = default;
1207
1208 /// Adds metatadata that can be preserved from the original instruction
1209 /// \p I.
1211 getMetadataToPropagate(&I, Metadata);
1212 // Retain the branch weights of terminators. They are used to compute the
1213 // frequencies with which the blocks of the original loop execute. Also
1214 // retain !prof on selects.
1215 if (I.isTerminator() || isa<SelectInst>(&I))
1216 if (MDNode *BW = I.getMetadata(LLVMContext::MD_prof))
1217 Metadata.emplace_back(LLVMContext::MD_prof, BW);
1218 }
1219
1220 /// Copy constructor for cloning.
1222
1224
1225 /// Add all metadata to \p I.
1226 void applyMetadata(Instruction &I) const;
1227
1228 /// Set metadata with kind \p Kind to \p Node. If metadata with \p Kind
1229 /// already exists, it will be replaced. Otherwise, it will be added.
1230 void setMetadata(unsigned Kind, MDNode *Node) {
1231 auto It =
1232 llvm::find_if(Metadata, [Kind](const std::pair<unsigned, MDNode *> &P) {
1233 return P.first == Kind;
1234 });
1235 if (It != Metadata.end())
1236 It->second = Node;
1237 else
1238 Metadata.emplace_back(Kind, Node);
1239 }
1240
1241 /// Remove the metadata of kind \p Kind, if present.
1242 void eraseMetadata(unsigned Kind) {
1243 erase_if(Metadata, [Kind](const auto &P) { return P.first == Kind; });
1244 }
1245
1246 /// Intersect this VPIRMetadata object with \p MD, keeping only metadata
1247 /// nodes that are common to both.
1248 void intersect(const VPIRMetadata &MD);
1249
1250 /// Get metadata of kind \p Kind. Returns nullptr if not found.
1251 MDNode *getMetadata(unsigned Kind) const {
1252 auto It =
1253 find_if(Metadata, [Kind](const auto &P) { return P.first == Kind; });
1254 return It != Metadata.end() ? It->second : nullptr;
1255 }
1256
1257 /// Record that the recipe executes with frequency \p Freq, relative to the
1258 /// entry of the loop region.
1259 void setExecutionFrequency(std::optional<VPExecutionFrequency> Freq,
1260 LLVMContext &Ctx);
1261
1262 /// Returns the frequency recorded by setExecutionFrequency, if any.
1263 std::optional<VPExecutionFrequency> getExecutionFrequency() const;
1264
1265 /// Drop the frequency recorded by setExecutionFrequency, if any.
1266 void clearExecutionFrequency();
1267
1268 /// Returns the branch weights recorded for this terminator, preferring real
1269 /// profile data over an estimate, or nullptr if there are none.
1271 MDNode *Node = getMetadata(LLVMContext::MD_prof);
1272 return Node ? Node : getInternalMetadata(EstimatedProfileMDName);
1273 }
1274
1275 /// Returns true if the weights returned by getBranchWeights are estimated.
1277 return getInternalMetadata(EstimatedProfileMDName);
1278 }
1279
1280 /// Set estimated branch weights to \p Node.
1282 assert(!getMetadata(LLVMContext::MD_prof) &&
1283 "real profile data takes precedence over an estimate");
1284 setMetadata(Node->getContext().getMDKindID(EstimatedProfileMDName), Node);
1285 }
1286
1287#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1288 /// Print metadata with node IDs.
1289 void print(raw_ostream &O, VPSlotTracker &SlotTracker) const;
1290#endif
1291};
1292
1293/// This is a concrete Recipe that models a single VPlan-level instruction.
1294/// While as any Recipe it may generate a sequence of IR instructions when
1295/// executed, these instructions would always form a single-def expression as
1296/// the VPInstruction is also a single def-use vertex. Most VPInstruction
1297/// opcodes can take an optional mask. Masks may be assigned during
1298/// predication.
1300 public VPIRMetadata {
1301public:
1302 /// VPlan opcodes, extending LLVM IR with idiomatics instructions.
1303 enum {
1304 FirstOrderRecurrenceSplice = Instruction::OtherOpsEnd +
1305 1, // Combines the incoming and previous
1306 // values of a first-order recurrence.
1308 // Creates a mask where each lane is active (true) whilst the current
1309 // counter (first operand + index) is less than the second operand. i.e.
1310 // mask[i] = icmpt ult (op0 + i), op1
1311 // ActiveLaneMask is used for early-exit loops with stores, plus tail
1312 // folding for all styles except DataAndControlFlow. The size of the
1313 // mask returned is VF. When unrolled, ActiveLaneMask is duplicated.
1315 // As above, but takes an additional operand (Multiplier). The size of
1316 // the mask returned is VF * Multiplier (UF, op2).
1317 // WideActiveLaneMask is used for control flow and is unrolled by widening,
1318 // with one extract vector created per unroll part.
1320 // Extracts each unrolled part of a (VF * UF) widened vector/mask.
1323 // Represents the incoming loop-invariant alias-mask. All memory accesses
1324 // in the loop must stay within the active lanes.
1326 // Increment the canonical IV separately for each unrolled part.
1328 // Abstract instruction that compares two values and branches. This is
1329 // lowered to ICmp + BranchOnCond during VPlan to VPlan transformation.
1332 // Branch with 2 boolean condition operands and 3 successors. If condition
1333 // 0 is true, branches to successor 0; if condition 1 is true, branches to
1334 // successor 1; otherwise branches to successor 2. Expanded after region
1335 // dissolution into: (1) an OR of the two conditions branching to
1336 // middle.split or successor 2, and (2) middle.split branching to successor
1337 // 0 or successor 1 based on condition 0.
1340 /// Given operands of (the same) struct type, creates a struct of fixed-
1341 /// width vectors each containing a struct field of all operands. The
1342 /// number of operands matches the element count of every vector.
1344 /// Creates a fixed-width vector containing all operands. The number of
1345 /// operands matches the vector element count.
1347 /// Extracts all lanes from its (non-scalable) vector operand. This is an
1348 /// abstract VPInstruction whose single defined VPValue represents VF
1349 /// scalars extracted from a vector, to be replaced by VF ExtractElement
1350 /// VPInstructions.
1352 /// Reduce the operands to the final reduction result using the operation
1353 /// specified via the operation's VPIRFlags.
1355 // Extracts the last part of its operand. Removed during unrolling.
1357 // Extracts the last lane of its vector operand, per part.
1359 // Extracts the second-to-last lane from its operand or the second-to-last
1360 // part if it is scalar. In the latter case, the recipe will be removed
1361 // during unrolling.
1363 LogicalAnd, // Non-poison propagating logical And.
1364 LogicalOr, // Non-poison propagating logical Or.
1365 NumActiveLanes, // Counts the number of active lanes in a mask.
1366 // Add an offset in bytes (second operand) to a base pointer (first
1367 // operand). Only generates scalar values (either for the first lane only or
1368 // for all lanes, depending on its uses).
1370 // Add a vector offset in bytes (second operand) to a scalar base pointer
1371 // (first operand).
1373 // Returns a scalar boolean value, which is true if any lane of its
1374 // (boolean) vector operands is true. It produces the reduced value across
1375 // all unrolled iterations. Unrolling will add all copies of its original
1376 // operand as additional operands. Note does not block poison propagation.
1378 // Calculates the first active lane index of the vector predicate operands.
1379 // It produces the lane index across all unrolled iterations. Unrolling will
1380 // add all copies of its original operand as additional operands.
1381 // Implemented with @llvm.experimental.cttz.elts, but returns the expected
1382 // result even with operands that are all zeroes.
1384 // Calculates the last active lane index of the vector predicate operands.
1385 // The predicates must be prefix-masks (all 1s before all 0s). Used when
1386 // tail-folding to extract the correct live-out value from the last active
1387 // iteration. It produces the lane index across all unrolled iterations.
1388 // Unrolling will add all copies of its original operand as additional
1389 // operands.
1391 // Returns a reversed vector for the operand.
1393 /// Start vector for reductions with 3 operands: the original start value,
1394 /// the identity value for the reduction and an integer indicating the
1395 /// scaling factor.
1397 /// Extracts a single lane (first operand) from a set of vector operands.
1398 /// The lane specifies an index into a vector formed by combining all vector
1399 /// operands (all operands after the first one).
1401 /// Explicit user for the resume phi of the canonical induction in the main
1402 /// VPlan, used by the epilogue vector loop.
1404 /// Extracts the last active lane from a set of vectors. The first operand
1405 /// is the default value if no lanes in the masks are active. Conceptually,
1406 /// this concatenates all data vectors (odd operands), concatenates all
1407 /// masks (even operands -- ignoring the default value), and returns the
1408 /// last active value from the combined data vector using the combined mask.
1410 /// Compute the exiting value of a wide induction after vectorization, that
1411 /// is the value of the last lane of the induction increment (i.e. its
1412 /// backedge value). Has the wide induction recipe as operand.
1415 /// Scale the first operand (vector step) by the second operand
1416 /// (scalar-step). Casts both operands to the result type if needed.
1418 // Creates a step vector starting from 0 to VF with a step of 1.
1420 /// Calls a scalar intrinsic. The intrinsic ID is the last operand.
1422
1424 };
1425
1426 /// Returns true if this recipe produces scalar values for all VF lanes.
1427 bool doesGeneratePerAllLanes() const;
1428
1429 /// Return the number of operands determined by the opcode of the
1430 /// VPInstruction, excluding mask. Returns -1u if the number of operands
1431 /// cannot be determined directly by the opcode.
1432 unsigned getNumOperandsForOpcode() const;
1433
1434private:
1435 typedef unsigned char OpcodeTy;
1436 OpcodeTy Opcode;
1437
1438 /// An optional name that can be used for the generated IR instruction.
1439 std::string Name;
1440
1441 /// Returns true if we can generate a scalar for the first lane only if
1442 /// needed.
1443 bool doesGenerateSingleScalar() const;
1444
1445 /// Utility method serving execute: Generates either a single-scalar or vector
1446 /// value. \p GenerateSingleScalar determines whether to generate a
1447 /// single-scalar value.
1448 Value *generate(VPTransformState &State, bool GenerateSingleScalar);
1449
1450 /// Returns true if the VPInstruction does not need masking.
1451 bool alwaysUnmasked() const {
1452 if (Opcode == VPInstruction::MaskedCond)
1453 return false;
1454
1455 // For now only VPInstructions with underlying values use masks.
1456 // TODO: provide masks to VPInstructions w/o underlying values.
1457 if (!getUnderlyingValue())
1458 return true;
1459
1460 return Instruction::isCast(Opcode) || Opcode == Instruction::PHI ||
1461 Opcode == Instruction::GetElementPtr;
1462 }
1463
1464public:
1465 VPInstruction(unsigned Opcode, ArrayRef<VPValue *> Operands,
1466 const VPIRFlags &Flags = {}, const VPIRMetadata &MD = {},
1467 DebugLoc DL = DebugLoc::getUnknown(), const Twine &Name = "",
1468 Type *ResultTy = nullptr);
1469
1470 VP_CLASSOF_IMPL(VPRecipeBase::VPInstructionSC)
1471
1472 VPInstruction *clone() override {
1474 }
1475
1477 Type *ResultTy = nullptr) {
1478 auto *New = new VPInstruction(Opcode, NewOperands, *this, *this,
1479 getDebugLoc(), Name, ResultTy);
1480 if (getUnderlyingValue())
1481 New->setUnderlyingValue(getUnderlyingInstr());
1482 return New;
1483 }
1484
1485 unsigned getOpcode() const { return Opcode; }
1486
1487 /// Add \p Op as operand of this VPInstruction. Only supported for AnyOf,
1488 /// ComputeReductionResult, BuildVector, BuildStructVector, ExtractLane,
1489 /// ExtractLastActive, FirstActiveLane, LastActiveLane.
1490 void addOperand(VPValue *Op);
1491
1492 /// Generate the instruction.
1493 /// TODO: We currently execute only per-part unless a specific instance is
1494 /// provided.
1495 void execute(VPTransformState &State) override;
1496
1497 /// Return the cost of this VPInstruction.
1498 InstructionCost computeCost(ElementCount VF,
1499 VPCostContext &Ctx) const override;
1500
1501#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1502 /// Print the VPInstruction to dbgs() (for debugging).
1503 LLVM_DUMP_METHOD void dump() const;
1504#endif
1505
1506 bool hasResult() const {
1507 // CallInst may or may not have a result, depending on the called function.
1508 // Conservatively return calls have results for now.
1509 switch (getOpcode()) {
1510 case Instruction::Ret:
1511 case Instruction::UncondBr:
1512 case Instruction::CondBr:
1513 case Instruction::Store:
1514 case Instruction::Switch:
1515 case Instruction::IndirectBr:
1516 case Instruction::Resume:
1517 case Instruction::CatchRet:
1518 case Instruction::Unreachable:
1519 case Instruction::Fence:
1520 case Instruction::AtomicRMW:
1524 return false;
1525 default:
1526 return true;
1527 }
1528 }
1529
1530 /// Returns true if the VPInstruction has a mask operand.
1531 bool isMasked() const {
1532 unsigned NumOpsForOpcode = getNumOperandsForOpcode();
1533 // VPInstructions without a fixed number of operands cannot be masked.
1534 if (NumOpsForOpcode == -1u)
1535 return false;
1536 return NumOpsForOpcode + 1 == getNumOperands();
1537 }
1538
1539 /// Returns the number of operands, excluding the mask if the VPInstruction is
1540 /// masked.
1541 unsigned getNumOperandsWithoutMask() const {
1542 return getNumOperands() - isMasked();
1543 }
1544
1545 /// Add mask \p Mask to an unmasked VPInstruction, if it needs masking.
1546 void addMask(VPValue *Mask) {
1547 assert(!isMasked() && "recipe is already masked");
1548 if (alwaysUnmasked())
1549 return;
1550 assert(Mask->getScalarType()->isIntegerTy(1) &&
1551 "Mask must be an i1 (vector)");
1552 VPUser::addOperand(Mask);
1553 }
1554
1555 /// Returns the mask for the VPInstruction. Returns nullptr for unmasked
1556 /// VPInstructions.
1557 VPValue *getMask() const {
1558 return isMasked() ? getOperand(getNumOperands() - 1) : nullptr;
1559 }
1560
1561 /// Returns an iterator range over the operands excluding the mask operand
1562 /// if present.
1569
1570 /// Returns true if the underlying opcode may read from or write to memory.
1571 bool opcodeMayReadOrWriteFromMemory() const;
1572
1573 /// Returns true if the recipe only uses the first lane of operand \p Op.
1574 bool usesFirstLaneOnly(const VPValue *Op) const override;
1575
1576 /// Returns true if the recipe only uses scalars of operand \p Op.
1577 bool usesScalars(const VPValue *Op) const override {
1578 return isSingleScalar() || usesFirstLaneOnly(Op);
1579 }
1580
1581 /// Returns true if the recipe only uses the first part of operand \p Op.
1582 bool usesFirstPartOnly(const VPValue *Op) const override;
1583
1584 /// Returns true if this VPInstruction produces a scalar value from a vector,
1585 /// e.g. by performing a reduction or extracting a lane.
1586 bool isVectorToScalar() const;
1587
1588 /// Returns true if the recipe produces a single scalar value.
1589 bool isSingleScalar() const;
1590
1591 /// Returns the symbolic name assigned to the VPInstruction.
1592 StringRef getName() const { return Name; }
1593
1594 /// Set the symbolic name for the VPInstruction.
1595 void setName(StringRef NewName) { Name = NewName.str(); }
1596
1597protected:
1598#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1599 /// Print the VPInstruction to \p O.
1600 void printRecipe(raw_ostream &O, const Twine &Indent,
1601 VPSlotTracker &SlotTracker) const override;
1602#endif
1603};
1604
1605/// Helper type to provide functions to access incoming values and blocks for
1606/// phi-like recipes.
1608protected:
1609 /// Return a VPRecipeBase* to the current object.
1610 virtual const VPRecipeBase *getAsRecipe() const = 0;
1611
1612public:
1613 virtual ~VPPhiAccessors() = default;
1614
1615 /// Returns the incoming VPValue with index \p Idx.
1616 VPValue *getIncomingValue(unsigned Idx) const {
1617 return getAsRecipe()->getOperand(Idx);
1618 }
1619
1620 /// Returns the incoming block with index \p Idx.
1621 const VPBasicBlock *getIncomingBlock(unsigned Idx) const;
1622
1623 /// Returns the incoming value for \p VPBB. \p VPBB must be an incoming block.
1625 getIncomingValueForBlock(const VPBasicBlock *VPBB) const;
1626
1627 /// Sets the incoming value for \p VPBB to \p V. \p VPBB must be an incoming
1628 /// block.
1629 void setIncomingValueForBlock(const VPBasicBlock *VPBB, VPValue *V) const;
1630
1631 /// Returns the number of incoming values, also number of incoming blocks.
1632 virtual unsigned getNumIncoming() const {
1633 return getAsRecipe()->getNumOperands();
1634 }
1635
1636 /// Returns an interator range over the incoming values.
1638 return make_range(getAsRecipe()->op_begin(),
1639 getAsRecipe()->op_begin() + getNumIncoming());
1640 }
1641
1643 detail::index_iterator, std::function<const VPBasicBlock *(size_t)>>>;
1644
1645 /// Returns an iterator range over the incoming blocks.
1647 std::function<const VPBasicBlock *(size_t)> GetBlock = [this](size_t Idx) {
1648 return getIncomingBlock(Idx);
1649 };
1650 return map_range(index_range(0, getNumIncoming()), GetBlock);
1651 }
1652
1653 /// Returns an iterator range over pairs of incoming values and corresponding
1654 /// incoming blocks.
1660
1661 /// Removes the incoming value for \p IncomingBlock, which must be a
1662 /// predecessor.
1663 void removeIncomingValueFor(VPBlockBase *IncomingBlock) const;
1664
1665 /// Append \p IncomingV as an incoming value to the phi-like recipe.
1666 void addIncoming(VPValue *IncomingV) {
1667 auto *R = const_cast<VPRecipeBase *>(getAsRecipe());
1668 assert((R->getNumOperands() == 0 ||
1669 IncomingV->getScalarType() == R->getOperand(0)->getScalarType()) &&
1670 "all incoming values must have the same type");
1671 R->addOperand(IncomingV);
1672 }
1673
1674#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1675 /// Print the recipe.
1677#endif
1678};
1679
1682 const Twine &Name = "", Type *ResultTy = nullptr)
1683 : VPInstruction(Instruction::PHI, Operands, Flags, {}, DL, Name,
1684 ResultTy) {}
1685
1686 static inline bool classof(const VPUser *U) {
1687 auto *VPI = dyn_cast<VPInstruction>(U);
1688 return VPI && VPI->getOpcode() == Instruction::PHI;
1689 }
1690
1691 static inline bool classof(const VPValue *V) {
1692 auto *VPI = dyn_cast<VPInstruction>(V);
1693 return VPI && VPI->getOpcode() == Instruction::PHI;
1694 }
1695
1696 static inline bool classof(const VPSingleDefRecipe *SDR) {
1697 auto *VPI = dyn_cast<VPInstruction>(SDR);
1698 return VPI && VPI->getOpcode() == Instruction::PHI;
1699 }
1700
1701 VPPhi *clone() override {
1702 auto *PhiR = new VPPhi(operands(), *this, getDebugLoc(), getName());
1703 PhiR->setUnderlyingValue(getUnderlyingValue());
1704 return PhiR;
1705 }
1706
1707 void execute(VPTransformState &State) override;
1708
1709protected:
1710#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1711 /// Print the recipe.
1712 void printRecipe(raw_ostream &O, const Twine &Indent,
1713 VPSlotTracker &SlotTracker) const override;
1714#endif
1715
1716 const VPRecipeBase *getAsRecipe() const override { return this; }
1717};
1718
1719/// A recipe to wrap on original IR instruction not to be modified during
1720/// execution, except for PHIs. PHIs are modeled via the VPIRPhi subclass.
1721/// Expect PHIs, VPIRInstructions cannot have any operands.
1723 Instruction &I;
1724
1725protected:
1726 /// VPIRInstruction::create() should be used to create VPIRInstructions, as
1727 /// subclasses may need to be created, e.g. VPIRPhi.
1729 : VPRecipeBase(VPRecipeBase::VPIRInstructionSC, {}), I(I) {}
1730
1731public:
1732 ~VPIRInstruction() override = default;
1733
1734 /// Create a new VPIRPhi for \p \I, if it is a PHINode, otherwise create a
1735 /// VPIRInstruction.
1737
1738 VP_CLASSOF_IMPL(VPRecipeBase::VPIRInstructionSC)
1739
1741 auto *R = create(I);
1742 for (auto *Op : operands())
1743 R->addOperand(Op);
1744 return R;
1745 }
1746
1747 void execute(VPTransformState &State) override;
1748
1749 /// Return the cost of this VPIRInstruction.
1751 computeCost(ElementCount VF, VPCostContext &Ctx) const override;
1752
1753 Instruction &getInstruction() const { return I; }
1754
1755 bool usesScalars(const VPValue *Op) const override {
1757 "Op must be an operand of the recipe");
1758 return true;
1759 }
1760
1761 bool usesFirstPartOnly(const VPValue *Op) const override {
1763 "Op must be an operand of the recipe");
1764 return true;
1765 }
1766
1767 bool usesFirstLaneOnly(const VPValue *Op) const override {
1769 "Op must be an operand of the recipe");
1770 return true;
1771 }
1772
1773protected:
1774#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1775 /// Print the recipe.
1776 void printRecipe(raw_ostream &O, const Twine &Indent,
1777 VPSlotTracker &SlotTracker) const override;
1778#endif
1779};
1780
1781/// An overlay for VPIRInstructions wrapping PHI nodes enabling convenient use
1782/// cast/dyn_cast/isa and execute() implementation. A single VPValue operand is
1783/// allowed, and it is used to add a new incoming value for the single
1784/// predecessor VPBB.
1786 public VPPhiAccessors {
1788
1789 static inline bool classof(const VPRecipeBase *U) {
1790 auto *R = dyn_cast<VPIRInstruction>(U);
1791 return R && isa<PHINode>(R->getInstruction());
1792 }
1793
1794 static inline bool classof(const VPUser *U) {
1795 auto *R = dyn_cast<VPRecipeBase>(U);
1796 return R && classof(R);
1797 }
1798
1800
1801 void execute(VPTransformState &State) override;
1802
1803protected:
1804#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1805 /// Print the recipe.
1806 void printRecipe(raw_ostream &O, const Twine &Indent,
1807 VPSlotTracker &SlotTracker) const override;
1808#endif
1809
1810 const VPRecipeBase *getAsRecipe() const override { return this; }
1811};
1812
1813/// VPWidenRecipe is a recipe for producing a widened instruction using the
1814/// opcode and operands of the recipe. This recipe covers most of the
1815/// traditional vectorization cases where each recipe transforms into a
1816/// vectorized version of itself.
1818 public VPIRMetadata {
1819 unsigned Opcode;
1820
1821public:
1823 const VPIRFlags &Flags = {}, const VPIRMetadata &Metadata = {},
1824 DebugLoc DL = {})
1825 : VPWidenRecipe(I.getOpcode(), Operands, Flags, Metadata, DL) {
1826 setUnderlyingValue(&I);
1827 }
1828
1830 const VPIRFlags &Flags = {}, const VPIRMetadata &Metadata = {},
1831 DebugLoc DL = {})
1832 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenSC, Operands,
1834 Flags, DL),
1835 VPIRMetadata(Metadata), Opcode(Opcode) {
1836 assert(flagsValidForOpcode(Opcode) &&
1837 "Set flags not supported for the provided opcode");
1838 assert(hasRequiredFlagsForOpcode(Opcode, getScalarType()) &&
1839 "Opcode requires specific flags to be set");
1840 }
1841
1842 ~VPWidenRecipe() override = default;
1843
1845
1847 if (auto *UV = getUnderlyingValue())
1848 return new VPWidenRecipe(*cast<Instruction>(UV), NewOperands, *this,
1849 *this, getDebugLoc());
1850 return new VPWidenRecipe(Opcode, NewOperands, *this, *this, getDebugLoc());
1851 }
1852
1853 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenSC)
1854
1855 /// Produce a widened instruction using the opcode and operands of the recipe,
1856 /// processing State.VF elements.
1857 void execute(VPTransformState &State) override;
1858
1859 /// Return the cost of this VPWidenRecipe.
1860 InstructionCost computeCost(ElementCount VF,
1861 VPCostContext &Ctx) const override;
1862
1863 unsigned getOpcode() const { return Opcode; }
1864
1865protected:
1866#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1867 /// Print the recipe.
1868 void printRecipe(raw_ostream &O, const Twine &Indent,
1869 VPSlotTracker &SlotTracker) const override;
1870#endif
1871
1872 /// Returns true if the recipe only uses the first lane of operand \p Op.
1873 bool usesFirstLaneOnly(const VPValue *Op) const override {
1875 "Op must be an operand of the recipe");
1876 return Opcode == Instruction::Select && Op == getOperand(0) &&
1878 }
1879};
1880
1881/// VPWidenCastRecipe is a recipe to create vector cast instructions.
1882/// TODO: Merge with VPWidenRecipe now that type is associated to every
1883/// VPRecipeValue.
1885 public VPIRMetadata {
1886 /// Cast instruction opcode.
1887 Instruction::CastOps Opcode;
1888
1889public:
1891 CastInst *CI = nullptr, const VPIRFlags &Flags = {},
1892 const VPIRMetadata &Metadata = {},
1893 DebugLoc DL = DebugLoc::getUnknown())
1894 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenCastSC, Op, ResultTy, Flags,
1895 DL),
1896 VPIRMetadata(Metadata), Opcode(Opcode) {
1897 assert(flagsValidForOpcode(Opcode) &&
1898 "Set flags not supported for the provided opcode");
1899 assert(hasRequiredFlagsForOpcode(Opcode, ResultTy) &&
1900 "Opcode requires specific flags to be set");
1901 setUnderlyingValue(CI);
1902 }
1903
1904 ~VPWidenCastRecipe() override = default;
1905
1907 return new VPWidenCastRecipe(Opcode, getOperand(0), getScalarType(),
1909 *this, *this, getDebugLoc());
1910 }
1911
1912 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenCastSC)
1913
1914 /// Produce widened copies of the cast.
1915 void execute(VPTransformState &State) override;
1916
1917 /// Return the cost of this VPWidenCastRecipe.
1918 InstructionCost computeCost(ElementCount VF,
1919 VPCostContext &Ctx) const override;
1920
1921 Instruction::CastOps getOpcode() const { return Opcode; }
1922
1923protected:
1924#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
1925 /// Print the recipe.
1926 void printRecipe(raw_ostream &O, const Twine &Indent,
1927 VPSlotTracker &SlotTracker) const override;
1928#endif
1929};
1930
1931/// A recipe for widening vector intrinsics.
1933 public VPIRMetadata {
1934 /// ID of the vector intrinsic to widen.
1935 Intrinsic::ID VectorIntrinsicID;
1936
1937 /// True if the intrinsic may read from memory.
1938 bool MayReadFromMemory;
1939
1940 /// True if the intrinsic may read write to memory.
1941 bool MayWriteToMemory;
1942
1943 /// True if the intrinsic may have side-effects.
1944 bool MayHaveSideEffects;
1945
1946protected:
1948 ArrayRef<VPValue *> CallArguments, Type *Ty,
1949 const VPIRFlags &Flags = {},
1950 const VPIRMetadata &MD = {},
1951 DebugLoc DL = DebugLoc::getUnknown())
1952 : VPRecipeWithIRFlags(SC, CallArguments, Ty, Flags, DL), VPIRMetadata(MD),
1953 VectorIntrinsicID(VectorIntrinsicID) {
1954 LLVMContext &Ctx = Ty->getContext();
1955 AttributeSet Attrs = Intrinsic::getFnAttributes(Ctx, VectorIntrinsicID);
1956 MemoryEffects ME = Attrs.getMemoryEffects();
1957 MayReadFromMemory = !ME.onlyWritesMemory();
1958 MayWriteToMemory = !ME.onlyReadsMemory();
1959 MayHaveSideEffects = MayWriteToMemory ||
1960 !Attrs.hasAttribute(Attribute::NoUnwind) ||
1961 !Attrs.hasAttribute(Attribute::WillReturn);
1962 }
1963
1964 /// Helper function to produce the widened intrinsic call.
1965 CallInst *createVectorCall(VPTransformState &State);
1966
1967public:
1969 ArrayRef<VPValue *> CallArguments, Type *Ty,
1970 const VPIRFlags &Flags = {},
1971 const VPIRMetadata &MD = {},
1972 DebugLoc DL = DebugLoc::getUnknown())
1973 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenIntrinsicSC, CallArguments, Ty,
1974 Flags, DL),
1975 VPIRMetadata(MD), VectorIntrinsicID(VectorIntrinsicID),
1976 MayReadFromMemory(CI.mayReadFromMemory()),
1977 MayWriteToMemory(CI.mayWriteToMemory()),
1978 MayHaveSideEffects(CI.mayHaveSideEffects()) {
1979 setUnderlyingValue(&CI);
1980 }
1981
1983 ArrayRef<VPValue *> CallArguments, Type *Ty,
1984 const VPIRFlags &Flags = {},
1985 const VPIRMetadata &Metadata = {},
1986 DebugLoc DL = DebugLoc::getUnknown())
1987 : VPWidenIntrinsicRecipe(VPRecipeBase::VPWidenIntrinsicSC,
1988 VectorIntrinsicID, CallArguments, Ty, Flags,
1989 Metadata, DL) {}
1990
1991 ~VPWidenIntrinsicRecipe() override = default;
1992
1994 if (Value *CI = getUnderlyingValue())
1995 return new VPWidenIntrinsicRecipe(*cast<CallInst>(CI), VectorIntrinsicID,
1996 operands(), getScalarType(), *this,
1997 *this, getDebugLoc());
1998 return new VPWidenIntrinsicRecipe(VectorIntrinsicID, operands(),
1999 getScalarType(), *this, *this,
2000 getDebugLoc());
2001 }
2002
2003 static inline bool classof(const VPRecipeBase *R) {
2004 return R->getVPRecipeID() == VPRecipeBase::VPWidenIntrinsicSC ||
2005 R->getVPRecipeID() == VPRecipeBase::VPWidenMemIntrinsicSC;
2006 }
2007
2008 static inline bool classof(const VPUser *U) {
2009 auto *R = dyn_cast<VPRecipeBase>(U);
2010 return R && classof(R);
2011 }
2012
2013 static inline bool classof(const VPValue *V) {
2014 auto *R = V->getDefiningRecipe();
2015 return R && classof(R);
2016 }
2017
2018 static inline bool classof(const VPSingleDefRecipe *R) {
2019 return classof(static_cast<const VPRecipeBase *>(R));
2020 }
2021
2022 /// Produce a widened version of the vector intrinsic.
2023 void execute(VPTransformState &State) override;
2024
2025 /// Compute the cost of a vector intrinsic with \p ID and \p Operands.
2026 static InstructionCost computeCallCost(Intrinsic::ID ID,
2028 const VPRecipeWithIRFlags &R,
2029 ElementCount VF, VPCostContext &Ctx);
2030
2031 /// Return the cost of this vector intrinsic.
2032 InstructionCost computeCost(ElementCount VF,
2033 VPCostContext &Ctx) const override;
2034
2035 /// Return the ID of the intrinsic.
2036 Intrinsic::ID getVectorIntrinsicID() const { return VectorIntrinsicID; }
2037
2038 /// Return to name of the intrinsic as string.
2039 StringRef getIntrinsicName() const;
2040
2041 /// Returns true if the intrinsic may read from memory.
2042 bool mayReadFromMemory() const { return MayReadFromMemory; }
2043
2044 /// Returns true if the intrinsic may write to memory.
2045 bool mayWriteToMemory() const { return MayWriteToMemory; }
2046
2047 /// Returns true if the intrinsic may have side-effects.
2048 bool mayHaveSideEffects() const { return MayHaveSideEffects; }
2049
2050 bool usesFirstLaneOnly(const VPValue *Op) const override;
2051
2052protected:
2053#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2054 /// Print the recipe.
2055 void printRecipe(raw_ostream &O, const Twine &Indent,
2056 VPSlotTracker &SlotTracker) const override;
2057#endif
2058};
2059
2060/// A recipe for widening vector memory intrinsics.
2062 /// Alignment information for this memory access.
2063 Align Alignment;
2064
2065public:
2067 ArrayRef<VPValue *> CallArguments, Type *Ty,
2068 Align Alignment, const VPIRMetadata &MD = {},
2070 : VPWidenIntrinsicRecipe(VPRecipeBase::VPWidenMemIntrinsicSC,
2071 VectorIntrinsicID, CallArguments, Ty, {}, MD,
2072 DL),
2073 Alignment(Alignment) {
2074 assert((VectorIntrinsicID == Intrinsic::experimental_vp_strided_load ||
2075 VectorIntrinsicID == Intrinsic::experimental_vp_strided_store) &&
2076 "Unexpected intrinsic");
2077 }
2078
2079 ~VPWidenMemIntrinsicRecipe() override = default;
2080
2083 getScalarType(), Alignment, *this,
2084 getDebugLoc());
2085 }
2086
2087 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenMemIntrinsicSC)
2088
2089 /// Produce a widened version of the vector memory intrinsic.
2090 void execute(VPTransformState &State) override;
2091
2092 /// Helper function for computing the cost of vector memory intrinsic.
2094 bool IsMasked, Align Alignment,
2095 VPCostContext &Ctx);
2096
2097 /// Return the cost of this vector memory intrinsic.
2099 VPCostContext &Ctx) const override;
2100};
2101
2102/// A recipe for widening Call instructions using library calls.
2104 public VPIRMetadata {
2105 /// Variant stores a pointer to the chosen function. There is a 1:1 mapping
2106 /// between a given VF and the chosen vectorized variant, so there will be a
2107 /// different VPlan for each VF with a valid variant.
2108 Function *Variant;
2109
2110public:
2112 ArrayRef<VPValue *> CallArguments,
2113 const VPIRFlags &Flags = {},
2114 const VPIRMetadata &Metadata = {}, DebugLoc DL = {})
2115 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenCallSC, CallArguments,
2116 toScalarizedTy(Variant->getReturnType()), Flags,
2117 DL),
2118 VPIRMetadata(Metadata), Variant(Variant) {
2119 setUnderlyingValue(UV);
2120 assert(
2121 isa<Function>(getOperand(getNumOperands() - 1)->getLiveInIRValue()) &&
2122 "last operand must be the called function");
2123 assert(cast<Function>(CallArguments.back()->getLiveInIRValue())
2124 ->getReturnType() == getScalarType() &&
2125 "Scalar type must match return type of called scalar function");
2126 }
2127
2128 ~VPWidenCallRecipe() override = default;
2129
2131 return new VPWidenCallRecipe(getUnderlyingValue(), Variant, operands(),
2132 *this, *this, getDebugLoc());
2133 }
2134
2135 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenCallSC)
2136
2137 /// Produce a widened version of the call instruction.
2138 void execute(VPTransformState &State) override;
2139
2140 /// Return the cost of this VPWidenCallRecipe.
2141 InstructionCost computeCost(ElementCount VF,
2142 VPCostContext &Ctx) const override;
2143
2144 /// Return the cost of widening a call using the vector function \p Variant.
2145 static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx);
2146
2150
2153
2154 /// Returns true if the recipe only uses the first lane of operand \p Op.
2155 bool usesFirstLaneOnly(const VPValue *Op) const override;
2156
2157protected:
2158#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2159 /// Print the recipe.
2160 void printRecipe(raw_ostream &O, const Twine &Indent,
2161 VPSlotTracker &SlotTracker) const override;
2162#endif
2163};
2164
2165/// A recipe representing a sequence of load -> update -> store as part of
2166/// a histogram operation. This means there may be aliasing between vector
2167/// lanes, which is handled by the llvm.experimental.vector.histogram family
2168/// of intrinsics. The only update operations currently supported are
2169/// 'add' and 'sub' where the other term is loop-invariant.
2171 /// Opcode of the update operation, currently either add or sub.
2172 unsigned Opcode;
2173
2174public:
2175 VPHistogramRecipe(unsigned Opcode, ArrayRef<VPValue *> Operands,
2176 const VPIRMetadata &Metadata = {},
2178 : VPRecipeBase(VPRecipeBase::VPHistogramSC, Operands, DL),
2179 VPIRMetadata(Metadata), Opcode(Opcode) {}
2180
2181 ~VPHistogramRecipe() override = default;
2182
2184 return new VPHistogramRecipe(Opcode, operands(), *this, getDebugLoc());
2185 }
2186
2187 VP_CLASSOF_IMPL(VPRecipeBase::VPHistogramSC);
2188
2189 /// Produce a vectorized histogram operation.
2190 void execute(VPTransformState &State) override;
2191
2192 /// Return the cost of this VPHistogramRecipe.
2194 VPCostContext &Ctx) const override;
2195
2196 /// Return the mask operand if one was provided, or a null pointer if all
2197 /// lanes should be executed unconditionally.
2198 VPValue *getMask() const {
2199 return getNumOperands() == 3 ? getOperand(2) : nullptr;
2200 }
2201
2202 /// Returns true if the recipe only uses the first lane of operand \p Op.
2203 bool usesFirstLaneOnly(const VPValue *Op) const override {
2205 "Op must be an operand of the recipe");
2206 return Op == getOperand(1);
2207 }
2208
2209protected:
2210#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2211 /// Print the recipe
2212 void printRecipe(raw_ostream &O, const Twine &Indent,
2213 VPSlotTracker &SlotTracker) const override;
2214#endif
2215};
2216
2217/// A recipe for handling GEP instructions.
2219 Type *SourceElementTy;
2220
2221public:
2223 const VPIRFlags &Flags = {},
2225 GetElementPtrInst *UV = nullptr)
2226 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenGEPSC, Operands,
2227 Operands[0]->getScalarType(), Flags, DL),
2228 SourceElementTy(SourceElementTy) {
2229 if (UV) {
2230 setUnderlyingValue(UV);
2233 assert(Metadata.empty() && "unexpected metadata on GEP");
2234 }
2235 }
2236
2237 ~VPWidenGEPRecipe() override = default;
2238
2244
2245 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenGEPSC)
2246
2247 /// This recipe generates a GEP instruction.
2248 unsigned getOpcode() const { return Instruction::GetElementPtr; }
2249
2250 /// Generate the gep nodes.
2251 void execute(VPTransformState &State) override;
2252
2253 Type *getSourceElementType() const { return SourceElementTy; }
2254
2255 /// Return the cost of this VPWidenGEPRecipe.
2257 VPCostContext &Ctx) const override {
2258 // TODO: Compute accurate cost after retiring the legacy cost model.
2259 return 0;
2260 }
2261
2262 /// Returns true if the recipe only uses the first lane of operand \p Op.
2263 bool usesFirstLaneOnly(const VPValue *Op) const override;
2264
2265protected:
2266#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2267 /// Print the recipe.
2268 void printRecipe(raw_ostream &O, const Twine &Indent,
2269 VPSlotTracker &SlotTracker) const override;
2270#endif
2271};
2272
2273/// A recipe to compute a pointer to the last element of each part of a widened
2274/// memory access for widened memory accesses of SourceElementTy. Used for
2275/// VPWidenMemoryRecipes or VPInterleaveRecipes that are reversed. An extra
2276/// Offset operand is added by convertToConcreteRecipes when UF = 1, and by the
2277/// unroller otherwise.
2279 Type *SourceElementTy;
2280
2281 /// The constant stride of the pointer computed by this recipe, expressed in
2282 /// units of SourceElementTy.
2283 int64_t Stride;
2284
2285public:
2286 VPVectorEndPointerRecipe(VPValue *Ptr, VPValue *VF, Type *SourceElementTy,
2287 int64_t Stride, GEPNoWrapFlags GEPFlags, DebugLoc DL)
2288 : VPRecipeWithIRFlags(VPRecipeBase::VPVectorEndPointerSC, {Ptr, VF},
2289 Ptr->getScalarType(), GEPFlags, DL),
2290 SourceElementTy(SourceElementTy), Stride(Stride) {
2291 assert(Stride < 0 && "Stride must be negative");
2292 }
2293
2294 VP_CLASSOF_IMPL(VPRecipeBase::VPVectorEndPointerSC)
2295
2296 Type *getSourceElementType() const { return SourceElementTy; }
2297 int64_t getStride() const { return Stride; }
2298 VPValue *getPointer() const { return getOperand(0); }
2299 VPValue *getVFValue() const { return getOperand(1); }
2301 return getNumOperands() == 3 ? getOperand(2) : nullptr;
2302 }
2303
2304 /// Adds the offset operand to the recipe.
2305 /// Offset = Stride * (VF - 1) + Part * Stride * VF.
2306 void materializeOffset(unsigned Part = 0);
2307
2308 /// Append \p Offset as the offset operand. The offset is an integer index
2309 /// expressed in units of SourceElementTy.
2311 assert(Offset->getScalarType()->isIntegerTy() &&
2312 "offset must be an integer index");
2314 }
2315
2316 void execute(VPTransformState &State) override;
2317
2318 bool usesFirstLaneOnly(const VPValue *Op) const override {
2320 "Op must be an operand of the recipe");
2321 return true;
2322 }
2323
2324 /// Return the cost of this VPVectorPointerRecipe.
2326 VPCostContext &Ctx) const override {
2327 // TODO: Compute accurate cost after retiring the legacy cost model.
2328 return 0;
2329 }
2330
2331 /// Returns true if the recipe only uses the first part of operand \p Op.
2332 bool usesFirstPartOnly(const VPValue *Op) const override {
2334 "Op must be an operand of the recipe");
2335 assert(getNumOperands() <= 2 && "must have at most two operands");
2336 return true;
2337 }
2338
2340 auto *VEPR = new VPVectorEndPointerRecipe(
2343 if (auto *Offset = getOffset())
2344 VEPR->addOffset(Offset);
2345 return VEPR;
2346 }
2347
2348protected:
2349#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2350 /// Print the recipe.
2351 void printRecipe(raw_ostream &O, const Twine &Indent,
2352 VPSlotTracker &SlotTracker) const override;
2353#endif
2354};
2355
2356/// A recipe to compute the pointers for widened memory accesses of \p
2357/// SourceElementTy, with the \p Stride expressed in units of \p
2358/// SourceElementTy. Unrolling adds an extra \p VFxPart operand for unrolled
2359/// parts > 0 and it produces `GEP SourceElementTy Ptr, VFxPart * Stride`.
2361 Type *SourceElementTy;
2362
2363public:
2364 VPVectorPointerRecipe(VPValue *Ptr, Type *SourceElementTy, VPValue *Stride,
2365 GEPNoWrapFlags GEPFlags, DebugLoc DL)
2366 : VPRecipeWithIRFlags(VPRecipeBase::VPVectorPointerSC,
2367 ArrayRef<VPValue *>({Ptr, Stride}),
2368 Ptr->getScalarType(), GEPFlags, DL),
2369 SourceElementTy(SourceElementTy) {}
2370
2371 VP_CLASSOF_IMPL(VPRecipeBase::VPVectorPointerSC)
2372
2373 VPValue *getStride() const { return getOperand(1); }
2374
2376 return getNumOperands() > 2 ? getOperand(2) : nullptr;
2377 }
2378
2379 /// Add the per-part offset (VFxPart) used for unrolled parts > 0.
2380 void addPerPartOffset(VPValue *VFxPart) {
2381 assert(VFxPart->getScalarType()->isIntegerTy() &&
2382 "per-part offset must be an integer index");
2383 VPUser::addOperand(VFxPart);
2384 }
2385
2386 void execute(VPTransformState &State) override;
2387
2388 Type *getSourceElementType() const { return SourceElementTy; }
2389
2390 bool usesFirstLaneOnly(const VPValue *Op) const override {
2392 "Op must be an operand of the recipe");
2393 return true;
2394 }
2395
2396 /// Returns true if the recipe only uses the first part of operand \p Op.
2397 bool usesFirstPartOnly(const VPValue *Op) const override {
2399 "Op must be an operand of the recipe");
2400 assert(getNumOperands() <= 2 && "must have at most two operands");
2401 return true;
2402 }
2403
2405 auto *Clone =
2406 new VPVectorPointerRecipe(getOperand(0), SourceElementTy, getStride(),
2408 if (auto *VFxPart = getVFxPart())
2409 Clone->addPerPartOffset(VFxPart);
2410 return Clone;
2411 }
2412
2413 /// Return the cost of this VPHeaderPHIRecipe.
2415 VPCostContext &Ctx) const override {
2416 // TODO: Compute accurate cost after retiring the legacy cost model.
2417 return 0;
2418 }
2419
2420protected:
2421#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2422 /// Print the recipe.
2423 void printRecipe(raw_ostream &O, const Twine &Indent,
2424 VPSlotTracker &SlotTracker) const override;
2425#endif
2426};
2427
2428/// A pure virtual base class for all recipes modeling header phis, including
2429/// phis for first order recurrences, pointer inductions and reductions. The
2430/// start value is the first operand of the recipe and the incoming value from
2431/// the backedge is the second operand.
2432///
2433/// Inductions are modeled using the following sub-classes:
2434/// * VPWidenIntOrFpInductionRecipe: Generates vector values for integer and
2435/// floating point inductions with arbitrary start and step values. Produces
2436/// a vector PHI per-part.
2437/// * VPWidenPointerInductionRecipe: Generate vector and scalar values for a
2438/// pointer induction. Produces either a vector PHI per-part or scalar values
2439/// per-lane based on the canonical induction.
2440/// * VPFirstOrderRecurrencePHIRecipe
2441/// * VPReductionPHIRecipe
2442/// * VPActiveLaneMaskPHIRecipe
2443/// * VPEVLBasedIVPHIRecipe
2444///
2445/// Note that the canonical IV is modeled as a VPRegionValue associated with
2446/// its loop region.
2448 public VPPhiAccessors {
2449protected:
2450 VPHeaderPHIRecipe(VPRecipeTy VPRecipeID, Instruction *UnderlyingInstr,
2451 VPValue *Start, DebugLoc DL = DebugLoc::getUnknown())
2452 : VPHeaderPHIRecipe(VPRecipeID, UnderlyingInstr, Start,
2453 Start->getScalarType(), DL) {}
2454
2455 VPHeaderPHIRecipe(VPRecipeTy VPRecipeID, Instruction *UnderlyingInstr,
2456 VPValue *Start, Type *ResultTy, DebugLoc DL)
2457 : VPSingleDefRecipe(VPRecipeID, Start, ResultTy, UnderlyingInstr, DL) {}
2458
2459 const VPRecipeBase *getAsRecipe() const override { return this; }
2460
2461public:
2462 ~VPHeaderPHIRecipe() override = default;
2463
2464 /// Method to support type inquiry through isa, cast, and dyn_cast.
2465 static inline bool classof(const VPRecipeBase *R) {
2466 return R->getVPRecipeID() >= VPRecipeBase::VPFirstHeaderPHISC &&
2467 R->getVPRecipeID() <= VPRecipeBase::VPLastHeaderPHISC;
2468 }
2469 static inline bool classof(const VPValue *V) {
2470 return isa<VPHeaderPHIRecipe>(V->getDefiningRecipe());
2471 }
2472 static inline bool classof(const VPSingleDefRecipe *R) {
2473 return isa<VPHeaderPHIRecipe>(static_cast<const VPRecipeBase *>(R));
2474 }
2475
2476 /// Generate the phi nodes.
2477 void execute(VPTransformState &State) override = 0;
2478
2479 /// Return the cost of this header phi recipe.
2481 VPCostContext &Ctx) const override;
2482
2483 /// Returns the start value of the phi, if one is set.
2485 return getNumOperands() == 0 ? nullptr : getOperand(0);
2486 }
2488 return getNumOperands() == 0 ? nullptr : getOperand(0);
2489 }
2490
2491 /// Update the start value of the recipe.
2493
2494 /// Returns the incoming value from the loop backedge.
2495 virtual VPValue *getBackedgeValue() { return getOperand(1); }
2496
2497 /// Update the incoming value from the loop backedge.
2499
2500 /// Add \p V as the incoming value from the loop backedge.
2502 assert(getNumOperands() == 1 &&
2503 "backedge value must be appended right after construction");
2504 assert(V->getScalarType() == getScalarType() &&
2505 "backedge value must have the same type as the start value");
2507 }
2508
2509protected:
2510#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2511 /// Print the recipe.
2512 void printRecipe(raw_ostream &O, const Twine &Indent,
2513 VPSlotTracker &SlotTracker) const override = 0;
2514#endif
2515};
2516
2517/// Base class for widened induction (VPWidenIntOrFpInductionRecipe and
2518/// VPWidenPointerInductionRecipe), providing shared functionality, including
2519/// retrieving the step value, induction descriptor and original phi node.
2521 InductionDescriptor IndDesc;
2522
2523public:
2525 VPValue *Step, const InductionDescriptor &IndDesc,
2526 DebugLoc DL)
2527 : VPWidenInductionRecipe(Kind, IV, Start, Step, IndDesc,
2528 Start->getScalarType(), DL) {}
2529
2531 VPValue *Step, const InductionDescriptor &IndDesc,
2532 Type *ResultTy, DebugLoc DL)
2533 : VPHeaderPHIRecipe(Kind, IV, Start, ResultTy, DL), IndDesc(IndDesc) {
2534 addOperand(Step);
2535 }
2536
2537 /// After unrolling, append the splat-VF step (`VF * step`) and the value of
2538 /// the induction at the last unrolled part.
2539 void addUnrolledPartOperands(VPValue *SplatVFStep, VPValue *LastPart) {
2540 assert(LastPart->getScalarType() == getScalarType() &&
2541 "last-part value must match the induction recipe's scalar type");
2543 ? SplatVFStep->getScalarType()->isIntegerTy()
2544 : SplatVFStep->getScalarType() == getScalarType()) &&
2545 "splat-step must match the induction type for non-pointer "
2546 "inductions, or be an integer index for pointer inductions");
2547 VPUser::addOperand(SplatVFStep);
2548 VPUser::addOperand(LastPart);
2549 }
2550
2551 static inline bool classof(const VPRecipeBase *R) {
2552 return R->getVPRecipeID() == VPRecipeBase::VPWidenIntOrFpInductionSC ||
2553 R->getVPRecipeID() == VPRecipeBase::VPWidenPointerInductionSC;
2554 }
2555
2556 static inline bool classof(const VPValue *V) {
2557 auto *R = V->getDefiningRecipe();
2558 return R && classof(R);
2559 }
2560
2561 static inline bool classof(const VPSingleDefRecipe *R) {
2562 return classof(static_cast<const VPRecipeBase *>(R));
2563 }
2564
2565 void execute(VPTransformState &State) override = 0;
2566
2567 /// Returns the step value of the induction.
2569 const VPValue *getStepValue() const { return getOperand(1); }
2570
2572 const VPValue *getVFValue() const { return getOperand(2); }
2573
2574 /// Returns the number of incoming values, also number of incoming blocks.
2575 /// Note that at the moment, VPWidenPointerInductionRecipe only has a single
2576 /// incoming value, its start value.
2577 unsigned getNumIncoming() const override { return 1; }
2578
2579 /// Returns the underlying PHINode if one exists, or null otherwise.
2583
2584 /// Returns the induction descriptor for the recipe.
2585 const InductionDescriptor &getInductionDescriptor() const { return IndDesc; }
2586
2587 /// Returns the SCEV predicates associated with this induction.
2589 return IndDesc.getNoWrapPredicates();
2590 }
2591
2593 // TODO: All operands of base recipe must exist and be at same index in
2594 // derived recipe.
2596 "VPWidenIntOrFpInductionRecipe generates its own backedge value");
2597 }
2598
2599 /// Returns true if the recipe only uses the first lane of operand \p Op.
2600 bool usesFirstLaneOnly(const VPValue *Op) const override {
2602 "Op must be an operand of the recipe");
2603 // The recipe creates its own wide start value, so it only requests the
2604 // first lane of the operand.
2605 // TODO: Remove once creating the start value is modeled separately.
2606 return Op == getStartValue() || Op == getStepValue();
2607 }
2608};
2609
2610/// A recipe for handling phi nodes of integer and floating-point inductions,
2611/// producing their vector values. This is an abstract recipe and must be
2612/// converted to concrete recipes before executing.
2614 public VPIRFlags {
2615 TruncInst *Trunc;
2616
2617 // If this recipe is unrolled it will have 2 additional operands.
2618 bool isUnrolled() const { return getNumOperands() == 5; }
2619
2620public:
2622 VPValue *VF, const InductionDescriptor &IndDesc,
2623 const VPIRFlags &Flags, DebugLoc DL)
2624 : VPWidenInductionRecipe(VPRecipeBase::VPWidenIntOrFpInductionSC, IV,
2625 Start, Step, IndDesc, DL),
2626 VPIRFlags(Flags), Trunc(nullptr) {
2627 addOperand(VF);
2628 }
2629
2631 VPValue *VF, const InductionDescriptor &IndDesc,
2632 TruncInst *Trunc, const VPIRFlags &Flags,
2633 DebugLoc DL)
2635 VPRecipeBase::VPWidenIntOrFpInductionSC, IV, Start, Step, IndDesc,
2636 Trunc ? Trunc->getType() : Start->getScalarType(), DL),
2637 VPIRFlags(Flags), Trunc(Trunc) {
2638 addOperand(VF);
2640 if (Trunc)
2642 assert(Metadata.empty() && "unexpected metadata on Trunc");
2643 }
2644
2646
2652
2653 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenIntOrFpInductionSC)
2654
2655 void execute(VPTransformState &State) override {
2656 llvm_unreachable("cannot execute this recipe, should be expanded via "
2657 "expandVPWidenIntOrFpInductionRecipe");
2658 }
2659
2660 /// If the recipe has been unrolled, return the VPValue for the induction
2661 /// increment, otherwise return null.
2663 return isUnrolled() ? getOperand(getNumOperands() - 2) : nullptr;
2664 }
2665
2666 /// Returns the number of incoming values, also number of incoming blocks.
2667 /// Note that at the moment, VPWidenIntOrFpInductionRecipes only have a single
2668 /// incoming value, its start value.
2669 unsigned getNumIncoming() const override { return 1; }
2670
2671 /// Returns the first defined value as TruncInst, if it is one or nullptr
2672 /// otherwise.
2673 TruncInst *getTruncInst() { return Trunc; }
2674 const TruncInst *getTruncInst() const { return Trunc; }
2675
2676 /// Return the cost of this VPWidenIntOrFpInductionRecipe.
2678 VPCostContext &Ctx) const override;
2679
2680 /// Returns true if the induction is canonical, i.e. starting at 0 and
2681 /// incremented by UF * VF (= the original IV is incremented by 1) and has the
2682 /// same type as the canonical induction.
2683 bool isCanonical() const;
2684
2685 /// Returns the VPValue representing the value of this induction at
2686 /// the last unrolled part, if it exists. Returns itself if unrolling did not
2687 /// take place.
2689 return isUnrolled() ? getOperand(getNumOperands() - 1) : this;
2690 }
2691
2692protected:
2693#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2694 /// Print the recipe.
2695 void printRecipe(raw_ostream &O, const Twine &Indent,
2696 VPSlotTracker &SlotTracker) const override;
2697#endif
2698};
2699
2701public:
2702 /// Create a new VPWidenPointerInductionRecipe for \p Phi with start value \p
2703 /// Start and the number of elements unrolled \p NumUnrolledElems, typically
2704 /// VF*UF.
2706 VPValue *NumUnrolledElems,
2707 const InductionDescriptor &IndDesc, DebugLoc DL)
2708 : VPWidenInductionRecipe(VPRecipeBase::VPWidenPointerInductionSC, Phi,
2709 Start, Step, IndDesc, DL) {
2710 addOperand(NumUnrolledElems);
2711 }
2712
2714
2720
2721 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenPointerInductionSC)
2722
2723 /// Generate vector values for the pointer induction.
2724 void execute(VPTransformState &State) override {
2725 llvm_unreachable("cannot execute this recipe, should be expanded via "
2726 "expandVPWidenPointerInduction");
2727 };
2728
2729 /// Returns true if only scalar values will be generated.
2730 bool onlyScalarsGenerated(bool IsScalable);
2731
2732 /// Return the cost of this VPWidenPointerInductionRecipe.
2734 VPCostContext &Ctx) const override;
2735
2736protected:
2737#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2738 /// Print the recipe.
2739 void printRecipe(raw_ostream &O, const Twine &Indent,
2740 VPSlotTracker &SlotTracker) const override;
2741#endif
2742};
2743
2744/// A recipe for widened phis. Incoming values are operands of the recipe and
2745/// their operand index corresponds to the incoming predecessor block. If the
2746/// recipe is placed in an entry block to a (non-replicate) region, it must have
2747/// exactly 2 incoming values, the first from the predecessor of the region and
2748/// the second from the exiting block of the region.
2750 public VPPhiAccessors {
2751 /// Name to use for the generated IR instruction for the widened phi.
2752 std::string Name;
2753
2754public:
2755 /// Create a new VPWidenPHIRecipe with incoming values \p IncomingValues,
2756 /// debug location \p DL and \p Name.
2758 DebugLoc DL = DebugLoc::getUnknown(), const Twine &Name = "")
2759 : VPSingleDefRecipe(VPRecipeBase::VPWidenPHISC, IncomingValues,
2760 IncomingValues[0]->getScalarType(),
2761 /*UV=*/nullptr, DL),
2762 Name(Name.str()) {
2763 assert(all_of(IncomingValues,
2764 [this](VPValue *VPV) {
2765 return VPV->getScalarType() == getScalarType();
2766 }) &&
2767 "all incoming values must have the same type");
2768 }
2769
2771 return new VPWidenPHIRecipe(operands(), getDebugLoc(), Name);
2772 }
2773
2774 ~VPWidenPHIRecipe() override = default;
2775
2776 /// This recipe generates a PHI.
2777 unsigned getOpcode() const { return Instruction::PHI; }
2778
2779 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenPHISC)
2780
2781 /// Generate the phi/select nodes.
2782 void execute(VPTransformState &State) override;
2783
2784 /// Return the cost of this VPWidenPHIRecipe.
2785 InstructionCost computeCost(ElementCount VF,
2786 VPCostContext &Ctx) const override;
2787
2788protected:
2789#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2790 /// Print the recipe.
2791 void printRecipe(raw_ostream &O, const Twine &Indent,
2792 VPSlotTracker &SlotTracker) const override;
2793#endif
2794
2795 const VPRecipeBase *getAsRecipe() const override { return this; }
2796};
2797
2798/// A recipe for handling first-order recurrence phis. The start value is the
2799/// first operand of the recipe and the incoming value from the backedge is the
2800/// second operand.
2803 VPValue &BackedgeValue)
2804 : VPHeaderPHIRecipe(VPRecipeBase::VPFirstOrderRecurrencePHISC, Phi,
2805 &Start) {
2806 addOperand(&BackedgeValue);
2807 }
2808
2809 VP_CLASSOF_IMPL(VPRecipeBase::VPFirstOrderRecurrencePHISC)
2810
2815
2816 void execute(VPTransformState &State) override;
2817
2818 /// Return the cost of this first-order recurrence phi recipe.
2820 VPCostContext &Ctx) const override;
2821
2822 /// Returns true if the recipe only uses the first lane of operand \p Op.
2823 bool usesFirstLaneOnly(const VPValue *Op) const override {
2825 "Op must be an operand of the recipe");
2826 return Op == getStartValue();
2827 }
2828
2829protected:
2830#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2831 /// Print the recipe.
2832 void printRecipe(raw_ostream &O, const Twine &Indent,
2833 VPSlotTracker &SlotTracker) const override;
2834#endif
2835};
2836
2837/// Possible variants of a reduction.
2838
2839/// This reduction is ordered and in-loop.
2840struct RdxOrdered {};
2841/// This reduction is in-loop.
2842struct RdxInLoop {};
2843/// This reduction is unordered with the partial result scaled down by some
2844/// factor.
2847};
2848using ReductionStyle = std::variant<RdxOrdered, RdxInLoop, RdxUnordered>;
2849
2850inline ReductionStyle getReductionStyle(bool InLoop, bool Ordered,
2851 unsigned ScaleFactor) {
2852 assert((!Ordered || InLoop) && "Ordered implies in-loop");
2853 if (Ordered)
2854 return RdxOrdered{};
2855 if (InLoop)
2856 return RdxInLoop{};
2857 return RdxUnordered{/*VFScaleFactor=*/ScaleFactor};
2858}
2859
2860/// A recipe for handling reduction phis. The start value is the first operand
2861/// of the recipe and the incoming value from the backedge is the second
2862/// operand.
2864 /// The recurrence kind of the reduction.
2865 const RecurKind Kind;
2866
2867 ReductionStyle Style;
2868
2869 /// The phi is part of a multi-use reduction (e.g., used in FindIV
2870 /// patterns for argmin/argmax).
2871 /// TODO: Also support cases where the phi itself has a single use, but its
2872 /// compare has multiple uses.
2873 bool HasUsesOutsideReductionChain;
2874
2875public:
2876 /// Create a new VPReductionPHIRecipe for the reduction \p Phi.
2878 VPValue &BackedgeValue, ReductionStyle Style,
2879 const VPIRFlags &Flags,
2880 bool HasUsesOutsideReductionChain = false)
2881 : VPHeaderPHIRecipe(VPRecipeBase::VPReductionPHISC, Phi, &Start),
2882 VPIRFlags(Flags), Kind(Kind), Style(Style),
2883 HasUsesOutsideReductionChain(HasUsesOutsideReductionChain) {
2884 addOperand(&BackedgeValue);
2885 }
2886
2887 ~VPReductionPHIRecipe() override = default;
2888
2890 VPValue *BackedgeValue) {
2891 return new VPReductionPHIRecipe(
2893 *Start, *BackedgeValue, Style, *this, HasUsesOutsideReductionChain);
2894 }
2895
2899
2900 VP_CLASSOF_IMPL(VPRecipeBase::VPReductionPHISC)
2901
2902 /// Generate the phi/select nodes.
2903 void execute(VPTransformState &State) override;
2904
2905 /// Get the factor that the VF of this recipe's output should be scaled by, or
2906 /// 1 if it isn't scaled.
2907 unsigned getVFScaleFactor() const {
2908 auto *Partial = std::get_if<RdxUnordered>(&Style);
2909 return Partial ? Partial->VFScaleFactor : 1;
2910 }
2911
2912 /// Set the VFScaleFactor for this reduction phi. Can only be set to a factor
2913 /// > 1.
2914 void setVFScaleFactor(unsigned ScaleFactor) {
2915 assert(ScaleFactor > 1 && "must set to scale factor > 1");
2916 Style = RdxUnordered{ScaleFactor};
2917 }
2918
2919 /// Returns the recurrence kind of the reduction.
2920 RecurKind getRecurrenceKind() const { return Kind; }
2921
2922 /// Returns true, if the phi is part of an ordered reduction.
2923 bool isOrdered() const { return std::holds_alternative<RdxOrdered>(Style); }
2924
2925 /// Returns true if the phi is part of an in-loop reduction.
2926 bool isInLoop() const {
2927 return std::holds_alternative<RdxInLoop>(Style) ||
2928 std::holds_alternative<RdxOrdered>(Style);
2929 }
2930
2931 /// Returns true, if the phi is part of a multi-use reduction.
2933 return HasUsesOutsideReductionChain;
2934 }
2935
2936 /// Returns true if the recipe only uses the first lane of operand \p Op.
2937 bool usesFirstLaneOnly(const VPValue *Op) const override {
2939 "Op must be an operand of the recipe");
2940 return isOrdered() || isInLoop();
2941 }
2942
2943protected:
2944#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
2945 /// Print the recipe.
2946 void printRecipe(raw_ostream &O, const Twine &Indent,
2947 VPSlotTracker &SlotTracker) const override;
2948#endif
2949};
2950
2951/// A recipe for vectorizing a phi-node as a sequence of mask-based select
2952/// instructions.
2954public:
2955 /// The blend operation is a User of the incoming values and of their
2956 /// respective masks, ordered [I0, M0, I1, M1, I2, M2, ...]. Note that M0 can
2957 /// be omitted (implied by passing an odd number of operands) in which case
2958 /// all other incoming values are merged into it.
2960 const VPIRFlags &Flags, DebugLoc DL)
2962 Operands[0]->getScalarType(), Flags, DL) {
2963 assert(Operands.size() >= 2 && "Expected at least two operands!");
2965 [this](unsigned I) {
2966 return getIncomingValue(I)->getScalarType() ==
2967 getScalarType();
2968 }) &&
2969 "all incoming values must have the same type");
2971 [this](unsigned I) {
2972 return getMask(I)->getScalarType()->isIntegerTy(1);
2973 }) &&
2974 "masks must be a bool");
2975 assert(hasRequiredFlagsForOpcode(Instruction::PHI, getScalarType()) &&
2976 "blends require the flags of the phi they replace");
2977 setUnderlyingValue(Phi);
2978 }
2979
2981
2984 NewOperands, *this, getDebugLoc());
2985 }
2986
2987 VP_CLASSOF_IMPL(VPRecipeBase::VPBlendSC)
2988
2989 /// A normalized blend is one that has an odd number of operands, whereby the
2990 /// first operand does not have an associated mask.
2991 bool isNormalized() const { return getNumOperands() % 2; }
2992
2993 /// Return the number of incoming values, taking into account when normalized
2994 /// the first incoming value will have no mask.
2995 unsigned getNumIncomingValues() const {
2996 return (getNumOperands() + isNormalized()) / 2;
2997 }
2998
2999 /// Return incoming value number \p Idx.
3000 VPValue *getIncomingValue(unsigned Idx) const {
3001 return Idx == 0 ? getOperand(0) : getOperand(Idx * 2 - isNormalized());
3002 }
3003
3004 /// Return mask number \p Idx.
3005 VPValue *getMask(unsigned Idx) const {
3006 assert((Idx > 0 || !isNormalized()) && "First index has no mask!");
3007 return Idx == 0 ? getOperand(1) : getOperand(Idx * 2 + !isNormalized());
3008 }
3009
3010 /// Set mask number \p Idx to \p V.
3011 void setMask(unsigned Idx, VPValue *V) {
3012 assert((Idx > 0 || !isNormalized()) && "First index has no mask!");
3013 assert(V->getScalarType()->isIntegerTy(1) && "Mask must be an i1 (vector)");
3014 Idx == 0 ? setOperand(1, V) : setOperand(Idx * 2 + !isNormalized(), V);
3015 }
3016
3017 void execute(VPTransformState &State) override {
3018 llvm_unreachable("VPBlendRecipe should be expanded by simplifyBlends");
3019 }
3020
3021 /// Return the cost of this VPWidenMemoryRecipe.
3022 InstructionCost computeCost(ElementCount VF,
3023 VPCostContext &Ctx) const override;
3024
3025 /// Returns true if the recipe only uses the first lane of operand \p Op.
3026 bool usesFirstLaneOnly(const VPValue *Op) const override;
3027
3028protected:
3029#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3030 /// Print the recipe.
3031 void printRecipe(raw_ostream &O, const Twine &Indent,
3032 VPSlotTracker &SlotTracker) const override;
3033#endif
3034};
3035
3036/// A common base class for interleaved memory operations.
3037/// An Interleaved memory operation is a memory access method that combines
3038/// multiple strided loads/stores into a single wide load/store with shuffles.
3039/// The first operand is the start address. The optional operands are, in order,
3040/// the stored values and the mask.
3042 public VPIRMetadata {
3044
3045 /// Indicates if the interleave group is in a conditional block and requires a
3046 /// mask.
3047 bool HasMask = false;
3048
3049 /// Indicates if gaps between members of the group need to be masked out or if
3050 /// unusued gaps can be loaded speculatively.
3051 bool NeedsMaskForGaps = false;
3052
3053protected:
3055 ArrayRef<VPValue *> Operands,
3056 ArrayRef<VPValue *> StoredValues, VPValue *Mask,
3057 bool NeedsMaskForGaps, const VPIRMetadata &MD, DebugLoc DL)
3058 : VPRecipeBase(SC, Operands, DL), VPIRMetadata(MD), IG(IG),
3059 NeedsMaskForGaps(NeedsMaskForGaps) {
3060 // TODO: extend the masked interleaved-group support to reversed access.
3061 assert((!Mask || !IG->isReverse()) &&
3062 "Reversed masked interleave-group not supported.");
3063 if (StoredValues.empty()) {
3064 for (Instruction *Inst : IG->members()) {
3065 assert(!Inst->getType()->isVoidTy() && "must have result");
3066 new VPMultiDefValue(this, Inst, Inst->getType());
3067 }
3068 } else {
3069 for (auto *SV : StoredValues)
3070 addOperand(SV);
3071 }
3072 if (Mask) {
3073 HasMask = true;
3074 addOperand(Mask);
3075 }
3076 }
3077
3078public:
3079 VPInterleaveBase *clone() override = 0;
3080
3081 static inline bool classof(const VPRecipeBase *R) {
3082 return R->getVPRecipeID() == VPRecipeBase::VPInterleaveSC ||
3083 R->getVPRecipeID() == VPRecipeBase::VPInterleaveEVLSC;
3084 }
3085
3086 static inline bool classof(const VPUser *U) {
3087 auto *R = dyn_cast<VPRecipeBase>(U);
3088 return R && classof(R);
3089 }
3090
3091 /// Return the address accessed by this recipe.
3092 VPValue *getAddr() const {
3093 return getOperand(0); // Address is the 1st, mandatory operand.
3094 }
3095
3096 /// Return the mask used by this recipe. Note that a full mask is represented
3097 /// by a nullptr.
3098 VPValue *getMask() const {
3099 // Mask is optional and the last operand.
3100 return HasMask ? getOperand(getNumOperands() - 1) : nullptr;
3101 }
3102
3103 /// Return true if the access needs a mask because of the gaps.
3104 bool needsMaskForGaps() const { return NeedsMaskForGaps; }
3105
3107
3108 Instruction *getInsertPos() const { return IG->getInsertPos(); }
3109
3110 void execute(VPTransformState &State) override {
3111 llvm_unreachable("VPInterleaveBase should not be instantiated.");
3112 }
3113
3114 /// Return the cost of this recipe.
3115 InstructionCost computeCost(ElementCount VF,
3116 VPCostContext &Ctx) const override;
3117
3118 /// Returns true if the recipe only uses the first lane of operand \p Op.
3119 bool usesFirstLaneOnly(const VPValue *Op) const override = 0;
3120
3121 /// Returns the number of stored operands of this interleave group. Returns 0
3122 /// for load interleave groups.
3123 virtual unsigned getNumStoreOperands() const = 0;
3124
3125 /// Return the VPValues stored by this interleave group. If it is a load
3126 /// interleave group, return an empty ArrayRef.
3128 return {op_end() - (getNumStoreOperands() + (HasMask ? 1 : 0)),
3130 }
3131};
3132
3133/// VPInterleaveRecipe is a recipe for transforming an interleave group of load
3134/// or stores into one wide load/store and shuffles. The first operand of a
3135/// VPInterleave recipe is the address, followed by the stored values, followed
3136/// by an optional mask.
3138public:
3140 ArrayRef<VPValue *> StoredValues, VPValue *Mask,
3141 bool NeedsMaskForGaps, const VPIRMetadata &MD, DebugLoc DL)
3142 : VPInterleaveBase(VPRecipeBase::VPInterleaveSC, IG, Addr, StoredValues,
3143 Mask, NeedsMaskForGaps, MD, DL) {}
3144
3145 ~VPInterleaveRecipe() override = default;
3146
3150 needsMaskForGaps(), *this, getDebugLoc());
3151 }
3152
3153 VP_CLASSOF_IMPL(VPRecipeBase::VPInterleaveSC)
3154
3155 /// Generate the wide load or store, and shuffles.
3156 void execute(VPTransformState &State) override;
3157
3158 bool usesFirstLaneOnly(const VPValue *Op) const override {
3160 "Op must be an operand of the recipe");
3161 return Op == getAddr() && !llvm::is_contained(getStoredValues(), Op);
3162 }
3163
3164 unsigned getNumStoreOperands() const override {
3165 return getNumOperands() - (getMask() ? 2 : 1);
3166 }
3167
3168protected:
3169#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3170 /// Print the recipe.
3171 void printRecipe(raw_ostream &O, const Twine &Indent,
3172 VPSlotTracker &SlotTracker) const override;
3173#endif
3174};
3175
3176/// A recipe for interleaved memory operations with vector-predication
3177/// intrinsics. The first operand is the address, the second operand is the
3178/// explicit vector length. Stored values and mask are optional operands.
3180public:
3182 : VPInterleaveBase(VPRecipeBase::VPInterleaveEVLSC,
3183 R.getInterleaveGroup(), {R.getAddr(), &EVL},
3184 R.getStoredValues(), Mask, R.needsMaskForGaps(), R,
3185 R.getDebugLoc()) {
3186 assert(!getInterleaveGroup()->isReverse() &&
3187 "Reversed interleave-group with tail folding is not supported.");
3188 assert(!needsMaskForGaps() && "Interleaved access with gap mask is not "
3189 "supported for scalable vector.");
3190 }
3191
3192 ~VPInterleaveEVLRecipe() override = default;
3193
3195 llvm_unreachable("cloning not implemented yet");
3196 }
3197
3198 VP_CLASSOF_IMPL(VPRecipeBase::VPInterleaveEVLSC)
3199
3200 /// The VPValue of the explicit vector length.
3201 VPValue *getEVL() const { return getOperand(1); }
3202
3203 /// Generate the wide load or store, and shuffles.
3204 void execute(VPTransformState &State) override;
3205
3206 /// The recipe only uses the first lane of the address, and EVL operand.
3207 bool usesFirstLaneOnly(const VPValue *Op) const override {
3209 "Op must be an operand of the recipe");
3210 return (Op == getAddr() && !llvm::is_contained(getStoredValues(), Op)) ||
3211 Op == getEVL();
3212 }
3213
3214 unsigned getNumStoreOperands() const override {
3215 return getNumOperands() - (getMask() ? 3 : 2);
3216 }
3217
3218protected:
3219#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3220 /// Print the recipe.
3221 void printRecipe(raw_ostream &O, const Twine &Indent,
3222 VPSlotTracker &SlotTracker) const override;
3223#endif
3224};
3225
3226/// A recipe to represent inloop, ordered or partial reduction operations. It
3227/// performs a reduction on a vector operand into a scalar (vector in the case
3228/// of a partial reduction) value, and adds the result to a chain. The Operands
3229/// are {ChainOp, VecOp, [Condition]}.
3231
3232 /// The recurrence kind for the reduction in question.
3233 RecurKind RdxKind;
3234 /// Whether the reduction is conditional.
3235 bool IsConditional = false;
3236 ReductionStyle Style;
3237
3238protected:
3241 VPValue *CondOp, ReductionStyle Style, DebugLoc DL)
3243 DL),
3244 RdxKind(RdxKind), Style(Style) {
3246 [this](VPValue *VPV) {
3247 return VPV->getScalarType() == getScalarType() ||
3248 (isa<VPInstruction>(VPV) &&
3249 cast<VPInstruction>(VPV)->getOpcode() ==
3251 }) &&
3252 "all incoming values must have the same type");
3253 if (CondOp) {
3254 assert(CondOp->getScalarType()->isIntegerTy(1) &&
3255 "CondOp must be a bool");
3256 IsConditional = true;
3257 addOperand(CondOp);
3258 }
3260 }
3261
3262public:
3264 VPValue *ChainOp, VPValue *VecOp, VPValue *CondOp,
3266 : VPReductionRecipe(VPRecipeBase::VPReductionSC, RdxKind, FMFs, I,
3267 {ChainOp, VecOp}, CondOp, Style, DL) {}
3268
3270 VPValue *ChainOp, VPValue *VecOp, VPValue *CondOp,
3272 : VPReductionRecipe(VPRecipeBase::VPReductionSC, RdxKind, FMFs, nullptr,
3273 {ChainOp, VecOp}, CondOp, Style, DL) {}
3274
3275 ~VPReductionRecipe() override = default;
3276
3278 return new VPReductionRecipe(RdxKind, getFastMathFlagsOrNone(),
3280 getCondOp(), Style, getDebugLoc());
3281 }
3282
3283 static inline bool classof(const VPRecipeBase *R) {
3284 return R->getVPRecipeID() == VPRecipeBase::VPReductionSC ||
3285 R->getVPRecipeID() == VPRecipeBase::VPReductionEVLSC;
3286 }
3287
3288 static inline bool classof(const VPUser *U) {
3289 auto *R = dyn_cast<VPRecipeBase>(U);
3290 return R && classof(R);
3291 }
3292
3293 static inline bool classof(const VPValue *VPV) {
3294 const VPRecipeBase *R = VPV->getDefiningRecipe();
3295 return R && classof(R);
3296 }
3297
3298 static inline bool classof(const VPSingleDefRecipe *R) {
3299 return classof(static_cast<const VPRecipeBase *>(R));
3300 }
3301
3302 /// Generate the reduction in the loop.
3303 void execute(VPTransformState &State) override;
3304
3305 /// Return the cost of VPReductionRecipe.
3306 InstructionCost computeCost(ElementCount VF,
3307 VPCostContext &Ctx) const override;
3308
3309 /// Return the recurrence kind for the in-loop reduction.
3310 RecurKind getRecurrenceKind() const { return RdxKind; }
3311 /// Return true if the in-loop reduction is ordered.
3312 bool isOrdered() const { return std::holds_alternative<RdxOrdered>(Style); };
3313 /// Return true if the in-loop reduction is conditional.
3314 bool isConditional() const { return IsConditional; };
3315 /// Returns true if the reduction outputs a vector with a scaled down VF.
3316 bool isPartialReduction() const {
3317 return std::holds_alternative<RdxUnordered>(Style);
3318 }
3319 /// Returns true if the reduction is in-loop.
3320 bool isInLoop() const {
3321 return std::holds_alternative<RdxInLoop>(Style) ||
3322 std::holds_alternative<RdxOrdered>(Style);
3323 }
3324 /// The VPValue of the scalar Chain being accumulated.
3325 VPValue *getChainOp() const { return getOperand(0); }
3326 /// The VPValue of the vector value to be reduced.
3327 VPValue *getVecOp() const { return getOperand(1); }
3328 /// The VPValue of the condition for the block.
3330 return isConditional() ? getOperand(getNumOperands() - 1) : nullptr;
3331 }
3332 /// Get the factor that the VF of this recipe's output should be scaled by, or
3333 /// 1 if it isn't scaled.
3334 unsigned getVFScaleFactor() const {
3335 auto *Partial = std::get_if<RdxUnordered>(&Style);
3336 return Partial ? Partial->VFScaleFactor : 1;
3337 }
3338
3339protected:
3340#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3341 /// Print the recipe.
3342 void printRecipe(raw_ostream &O, const Twine &Indent,
3343 VPSlotTracker &SlotTracker) const override;
3344#endif
3345};
3346
3347/// A recipe to represent inloop reduction operations with vector-predication
3348/// intrinsics, performing a reduction on a vector operand with the explicit
3349/// vector length (EVL) into a scalar value, and adding the result to a chain.
3350/// The Operands are {ChainOp, VecOp, EVL, [Condition]}.
3352public:
3355 : VPReductionRecipe(VPRecipeBase::VPReductionEVLSC, R.getRecurrenceKind(),
3358 {R.getChainOp(), R.getVecOp(), &EVL}, CondOp,
3359 getReductionStyle(R.isInLoop(), R.isOrdered(),
3360 R.getVFScaleFactor()),
3361 DL) {}
3362
3363 ~VPReductionEVLRecipe() override = default;
3364
3366 llvm_unreachable("cloning not implemented yet");
3367 }
3368
3369 VP_CLASSOF_IMPL(VPRecipeBase::VPReductionEVLSC)
3370
3371 /// Generate the reduction in the loop
3372 void execute(VPTransformState &State) override;
3373
3374 /// The VPValue of the explicit vector length.
3375 VPValue *getEVL() const { return getOperand(2); }
3376
3377 /// Returns true if the recipe only uses the first lane of operand \p Op.
3378 bool usesFirstLaneOnly(const VPValue *Op) const override {
3380 "Op must be an operand of the recipe");
3381 return Op == getEVL();
3382 }
3383
3384protected:
3385#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3386 /// Print the recipe.
3387 void printRecipe(raw_ostream &O, const Twine &Indent,
3388 VPSlotTracker &SlotTracker) const override;
3389#endif
3390};
3391
3392/// VPReplicateRecipe replicates a given instruction producing multiple scalar
3393/// copies of the original scalar type, one per lane, instead of producing a
3394/// single copy of widened type for all lanes. If the instruction is known to be
3395/// a single scalar, only one copy will be generated.
3397 public VPIRMetadata {
3398 /// Indicator if only a single replica per lane is needed.
3399 bool IsSingleScalar;
3400
3401 /// Indicator if the replicas are also predicated.
3402 bool IsPredicated;
3403
3404public:
3406 bool IsSingleScalar, VPValue *Mask = nullptr,
3407 const VPIRFlags &Flags = {}, VPIRMetadata Metadata = {},
3408 DebugLoc DL = DebugLoc::getUnknown())
3409 : VPRecipeWithIRFlags(VPRecipeBase::VPReplicateSC, Operands,
3410 computeScalarType(I, Operands), Flags, DL),
3411 VPIRMetadata(Metadata), IsSingleScalar(IsSingleScalar),
3412 IsPredicated(Mask) {
3413 assert((!IsSingleScalar || !I->isCast()) &&
3414 "Single-scalar casts should use VPInstruction");
3415 setUnderlyingValue(I);
3416 if (Mask)
3417 addOperand(Mask);
3418 }
3419
3420 ~VPReplicateRecipe() override = default;
3421
3422 /// Compute the scalar result type for a VPReplicateRecipe wrapping \p I with
3423 /// \p Operands (excluding any predicate mask).
3424 static Type *computeScalarType(const Instruction *I,
3426
3428
3430 auto *Copy = new VPReplicateRecipe(
3431 getUnderlyingInstr(), NewOperands, IsSingleScalar,
3432 isPredicated() ? getMask() : nullptr, *this, *this, getDebugLoc());
3433 Copy->transferFlags(*this);
3434 return Copy;
3435 }
3436
3437 VP_CLASSOF_IMPL(VPRecipeBase::VPReplicateSC)
3438
3439 /// Generate replicas of the desired Ingredient. Replicas will be generated
3440 /// for all parts and lanes unless a specific part and lane are specified in
3441 /// the \p State.
3442 void execute(VPTransformState &State) override;
3443
3444 /// Return the cost of this VPReplicateRecipe.
3445 InstructionCost computeCost(ElementCount VF,
3446 VPCostContext &Ctx) const override;
3447
3448 /// Return the cost of scalarizing a call to \p CalledFn with argument
3449 /// operands \p ArgOps for a given \p VF.
3450 static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy,
3452 bool IsSingleScalar, ElementCount VF,
3453 VPCostContext &Ctx);
3454
3455 /// Returns true if the recipe produces a single scalar value.
3456 bool isSingleScalar() const { return IsSingleScalar; }
3457
3458 /// Returns true if the recipe produces scalar values for all VF lanes.
3459 bool doesGeneratePerAllLanes() const { return !IsSingleScalar; }
3460
3461 bool isPredicated() const { return IsPredicated; }
3462
3463 /// Returns true if the recipe only uses the first lane of operand \p Op.
3464 bool usesFirstLaneOnly(const VPValue *Op) const override {
3466 "Op must be an operand of the recipe");
3467 return isSingleScalar();
3468 }
3469
3470 /// Returns true if the recipe uses scalars of operand \p Op.
3471 bool usesScalars(const VPValue *Op) const override {
3473 "Op must be an operand of the recipe");
3474 return true;
3475 }
3476
3477 /// Return the mask of a predicated VPReplicateRecipe.
3479 assert(isPredicated() && "Trying to get the mask of a unpredicated recipe");
3480 return getOperand(getNumOperands() - 1);
3481 }
3482
3483 /// Return the recipe's operands, excluding the mask of a predicated recipe.
3487
3488 /// Returns the number of operands, excluding the mask if the recipe is
3489 /// predicated.
3490 unsigned getNumOperandsWithoutMask() const {
3491 return getNumOperands() - isPredicated();
3492 }
3493
3494 unsigned getOpcode() const { return getUnderlyingInstr()->getOpcode(); }
3495
3496protected:
3497#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3498 /// Print the recipe.
3499 void printRecipe(raw_ostream &O, const Twine &Indent,
3500 VPSlotTracker &SlotTracker) const override;
3501#endif
3502};
3503
3504/// A recipe for generating conditional branches on the bits of a mask.
3506 public VPIRMetadata {
3507public:
3509 const VPIRMetadata &Metadata = {})
3510 : VPRecipeBase(VPRecipeBase::VPBranchOnMaskSC, {BlockInMask}, DL),
3511 VPIRMetadata(Metadata) {}
3512
3514 return new VPBranchOnMaskRecipe(getOperand(0), getDebugLoc(), *this);
3515 }
3516
3517 VP_CLASSOF_IMPL(VPRecipeBase::VPBranchOnMaskSC)
3518
3519 /// Generate the extraction of the appropriate bit from the block mask and the
3520 /// conditional branch.
3521 void execute(VPTransformState &State) override;
3522
3523 /// Return the cost of this VPBranchOnMaskRecipe.
3524 InstructionCost computeCost(ElementCount VF,
3525 VPCostContext &Ctx) const override;
3526
3527#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3528 /// Print the recipe.
3529 void printRecipe(raw_ostream &O, const Twine &Indent,
3530 VPSlotTracker &SlotTracker) const override {
3531 O << Indent << "BRANCH-ON-MASK ";
3533 }
3534#endif
3535
3536 /// Returns true if the recipe uses scalars of operand \p Op.
3537 bool usesScalars(const VPValue *Op) const override {
3539 "Op must be an operand of the recipe");
3540 return true;
3541 }
3542};
3543
3544/// A recipe to combine multiple recipes into a single 'expression' recipe,
3545/// which should be considered a single entity for cost-modeling and transforms.
3546/// The recipe needs to be 'decomposed', i.e. replaced by its individual
3547/// expression recipes, before execute. The individual expression recipes are
3548/// completely disconnected from the def-use graph of other recipes not part of
3549/// the expression. Def-use edges between pairs of expression recipes remain
3550/// intact, whereas every edge between an expression recipe and a recipe outside
3551/// the expression is elevated to connect the non-expression recipe with the
3552/// VPExpressionRecipe itself.
3554 /// Recipes included in this VPExpressionRecipe. This could contain
3555 /// duplicates.
3556 SmallVector<VPSingleDefRecipe *> ExpressionRecipes;
3557
3558 /// Temporary VPValues used for external operands of the expression, i.e.
3559 /// operands not defined by recipes in the expression.
3560 SmallVector<VPValue *> LiveInPlaceholders;
3561
3562 enum class ExpressionTypes {
3563 /// Represents an inloop extended reduction operation, performing a
3564 /// reduction on an extended vector operand into a scalar value, and adding
3565 /// the result to a chain.
3566 ExtendedReduction,
3567 /// Represents an inloop extended reduction operation, which is negated,
3568 /// then reduced before adding the result to a chain.
3569 NegatedExtendedReduction,
3570 /// Represent an inloop multiply-accumulate reduction, multiplying the
3571 /// extended vector operands, performing a reduction.add on the result, and
3572 /// adding the scalar result to a chain.
3573 ExtMulAccReduction,
3574 /// Represent an inloop multiply-accumulate reduction, multiplying the
3575 /// vector operands, performing a reduction.add on the result, and adding
3576 /// the scalar result to a chain.
3577 MulAccReduction,
3578 /// Represent an inloop multiply-accumulate reduction, multiplying the
3579 /// extended vector operands, negating the multiplication, performing a
3580 /// reduction.add on the result, and adding the scalar result to a chain.
3581 ExtNegatedMulAccReduction,
3582 };
3583
3584 /// Type of the expression.
3585 ExpressionTypes ExpressionType;
3586
3587public:
3588 /// Construct a new VPExpressionRecipe by internalizing recipes in \p
3589 /// ExpressionRecipes. External operands (i.e. not defined by another recipe
3590 /// in the expression) are replaced by temporary VPValues and the original
3591 /// operands are transferred to the VPExpressionRecipe itself. Clone recipes
3592 /// as needed (excluding last) to ensure they are only used by other recipes
3593 /// in the expression.
3594 VPExpressionRecipe(ExpressionTypes ExpressionType,
3595 ArrayRef<VPSingleDefRecipe *> ExpressionRecipes);
3596
3598 : VPExpressionRecipe(ExpressionTypes::ExtendedReduction, {Ext, Red}) {}
3600 VPReductionRecipe *Red)
3601 : VPExpressionRecipe(ExpressionTypes::NegatedExtendedReduction,
3602 {Ext, Neg, Red}) {
3603 assert((Red->getRecurrenceKind() == RecurKind::Add ||
3604 Red->getRecurrenceKind() == RecurKind::FAdd ||
3605 Red->getRecurrenceKind() == RecurKind::AddChainWithSubs) &&
3606 "Expected an add or add-chain-with-subs reduction");
3607 if (Neg->getOpcode() == Instruction::Sub) {
3608 [[maybe_unused]] auto *SubConst = dyn_cast<VPConstantInt>(getOperand(1));
3609 assert(SubConst && SubConst->isZero() && "Expected a negating sub");
3610 } else
3611 assert(Neg->getOpcode() == Instruction::FNeg && "Unexpected opcode");
3612 }
3614 : VPExpressionRecipe(ExpressionTypes::MulAccReduction, {Mul, Red}) {}
3617 : VPExpressionRecipe(ExpressionTypes::ExtMulAccReduction,
3618 {Ext0, Ext1, Mul, Red}) {}
3621 VPReductionRecipe *Red)
3622 : VPExpressionRecipe(ExpressionTypes::ExtNegatedMulAccReduction,
3623 {Ext0, Ext1, Mul, Neg, Red}) {
3624 assert((Mul->getOpcode() == Instruction::Mul ||
3625 Mul->getOpcode() == Instruction::FMul) &&
3626 "Expected a mul");
3627 assert((Red->getRecurrenceKind() == RecurKind::Add ||
3628 Red->getRecurrenceKind() == RecurKind::FAdd ||
3629 Red->getRecurrenceKind() == RecurKind::AddChainWithSubs) &&
3630 "Expected an add or add-chain-with-subs reduction");
3631 assert(getNumOperands() >= 3 && "Expected at least three operands");
3632 if (Neg->getOpcode() == Instruction::Sub) {
3633 [[maybe_unused]] auto *SubConst = dyn_cast<VPConstantInt>(getOperand(2));
3634 assert(SubConst && SubConst->isZero() &&
3635 Neg->getOpcode() == Instruction::Sub && "Expected a negating sub");
3636 } else
3637 assert(Neg->getOpcode() == Instruction::FNeg && "Unexpected opcode");
3638 }
3639
3641 SmallPtrSet<VPSingleDefRecipe *, 4> ExpressionRecipesSeen;
3642 for (auto *R : reverse(ExpressionRecipes)) {
3643 if (ExpressionRecipesSeen.insert(R).second)
3644 delete R;
3645 }
3646 for (VPValue *T : LiveInPlaceholders)
3647 delete T;
3648 }
3649
3650 VP_CLASSOF_IMPL(VPRecipeBase::VPExpressionSC)
3651
3653 assert(!ExpressionRecipes.empty() && "empty expressions should be removed");
3654 SmallVector<VPSingleDefRecipe *> NewExpressiondRecipes;
3655 for (auto *R : ExpressionRecipes)
3656 NewExpressiondRecipes.push_back(R->clone());
3657 for (auto *New : NewExpressiondRecipes) {
3658 for (const auto &[Idx, Old] : enumerate(ExpressionRecipes))
3659 New->replaceUsesOfWith(Old, NewExpressiondRecipes[Idx]);
3660 // Update placeholder operands in the cloned recipe to use the external
3661 // operands, to be internalized when the cloned expression is constructed.
3662 for (const auto &[Placeholder, OutsideOp] :
3663 zip(LiveInPlaceholders, operands()))
3664 New->replaceUsesOfWith(Placeholder, OutsideOp);
3665 }
3666 return new VPExpressionRecipe(ExpressionType, NewExpressiondRecipes);
3667 }
3668
3669 /// Return and insert the recipes of the expression back into the VPlan,
3670 /// directly before the current recipe. Leaves the expression recipe empty,
3671 /// which must be removed before codegen.
3673
3674 /// Returns the expression type of this recipe.
3675 ExpressionTypes getExpressionType() const { return ExpressionType; }
3676
3677 unsigned getVFScaleFactor() const {
3678 auto *PR = dyn_cast<VPReductionRecipe>(ExpressionRecipes.back());
3679 return PR ? PR->getVFScaleFactor() : 1;
3680 }
3681
3682 /// Method for generating code, must not be called as this recipe is abstract.
3683 void execute(VPTransformState &State) override {
3684 llvm_unreachable("recipe must be removed before execute");
3685 }
3686
3688 VPCostContext &Ctx) const override;
3689
3690 /// Returns true if this expression contains recipes that may read from or
3691 /// write to memory.
3692 bool mayReadOrWriteMemory() const;
3693
3694 /// Returns true if this expression contains recipes that may have side
3695 /// effects.
3696 bool mayHaveSideEffects() const;
3697
3698 /// Returns true if this VPExpressionRecipe produces a single scalar.
3699 bool isVectorToScalar() const;
3700
3701protected:
3702#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3703 /// Print the recipe.
3704 void printRecipe(raw_ostream &O, const Twine &Indent,
3705 VPSlotTracker &SlotTracker) const override;
3706#endif
3707};
3708
3709/// VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when
3710/// control converges back from a Branch-on-Mask. The phi nodes are needed in
3711/// order to merge values that are set under such a branch and feed their uses.
3712/// The phi nodes can be scalar or vector depending on the users of the value.
3713/// This recipe works in concert with VPBranchOnMaskRecipe.
3715public:
3716 /// Construct a VPPredInstPHIRecipe given \p PredInst whose value needs a phi
3717 /// nodes after merging back from a Branch-on-Mask.
3719 : VPSingleDefRecipe(VPRecipeBase::VPPredInstPHISC, PredV,
3720 PredV->getScalarType(), /*UV=*/nullptr, DL) {}
3721 ~VPPredInstPHIRecipe() override = default;
3722
3724 return new VPPredInstPHIRecipe(getOperand(0), getDebugLoc());
3725 }
3726
3727 VP_CLASSOF_IMPL(VPRecipeBase::VPPredInstPHISC)
3728
3729 /// Generates phi nodes for live-outs (from a replicate region) as needed to
3730 /// retain SSA form.
3731 void execute(VPTransformState &State) override;
3732
3733 /// Return the cost of this VPPredInstPHIRecipe.
3735 VPCostContext &Ctx) const override {
3736 // TODO: Compute accurate cost after retiring the legacy cost model.
3737 return 0;
3738 }
3739
3740protected:
3741#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3742 /// Print the recipe.
3743 void printRecipe(raw_ostream &O, const Twine &Indent,
3744 VPSlotTracker &SlotTracker) const override;
3745#endif
3746};
3747
3748/// A common mixin class for widening memory operations. An optional mask can be
3749/// provided as the last operand.
3751protected:
3753
3754 /// Alignment information for this memory access.
3756
3757 /// Whether the accessed addresses are consecutive.
3759
3760 /// Whether the memory access is masked.
3761 bool IsMasked = false;
3762
3763 void setMask(VPValue *Mask) {
3764 assert(!IsMasked && "cannot re-set mask");
3765 if (!Mask)
3766 return;
3767 assert(Mask->getScalarType()->isIntegerTy(1) &&
3768 "Mask must be an i1 (vector)");
3769 getAsRecipe()->addOperand(Mask);
3770 IsMasked = true;
3771 }
3772
3777
3778public:
3779 virtual ~VPWidenMemoryRecipe() = default;
3780
3781 /// Return a VPRecipeBase* to the current object.
3783 virtual const VPRecipeBase *getAsRecipe() const = 0;
3784
3785 /// Return whether the loaded-from / stored-to addresses are consecutive.
3786 bool isConsecutive() const { return Consecutive; }
3787
3788 /// Return the address accessed by this recipe.
3789 VPValue *getAddr() const { return getAsRecipe()->getOperand(0); }
3790
3791 /// Returns true if the recipe is masked.
3792 bool isMasked() const { return IsMasked; }
3793
3794 /// Return the mask used by this recipe. Note that a full mask is represented
3795 /// by a nullptr.
3796 VPValue *getMask() const {
3797 // Mask is optional and therefore the last operand.
3798 const VPRecipeBase *R = getAsRecipe();
3799 return isMasked() ? R->getOperand(R->getNumOperands() - 1) : nullptr;
3800 }
3801
3802 /// Returns the alignment of the memory access.
3803 Align getAlign() const { return Alignment; }
3804
3805 /// Return the cost of this VPWidenMemoryRecipe.
3806 InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const;
3807
3809};
3810
3811/// A recipe for widening load operations, using the address to load from and an
3812/// optional mask.
3814 public VPWidenMemoryRecipe {
3816 bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
3817 : VPSingleDefRecipe(VPRecipeBase::VPWidenLoadSC, {Addr}, Load.getType(),
3818 &Load, DL),
3819 VPWidenMemoryRecipe(Load, Consecutive, Metadata) {
3820 setMask(Mask);
3821 }
3822
3825 getMask(), Consecutive, *this, getDebugLoc());
3826 }
3827
3828 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC);
3829
3830 /// Returns the opcode of the widened load.
3831 unsigned getOpcode() const { return Instruction::Load; }
3832
3833 /// Generate a wide load or gather.
3834 void execute(VPTransformState &State) override;
3835
3836 /// Return the cost of this VPWidenLoadRecipe.
3838 VPCostContext &Ctx) const override {
3839 return VPWidenMemoryRecipe::computeCost(VF, Ctx);
3840 }
3841
3842 /// Returns true if the recipe only uses the first lane of operand \p Op.
3843 bool usesFirstLaneOnly(const VPValue *Op) const override {
3845 "Op must be an operand of the recipe");
3846 // Widened, consecutive loads operations only demand the first lane of
3847 // their address.
3848 return Op == getAddr() && isConsecutive();
3849 }
3850
3851protected:
3852 VPRecipeBase *getAsRecipe() override;
3853 const VPRecipeBase *getAsRecipe() const override;
3854
3855#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3856 /// Print the recipe.
3857 void printRecipe(raw_ostream &O, const Twine &Indent,
3858 VPSlotTracker &SlotTracker) const override;
3859#endif
3860};
3861
3862/// A recipe for widening load operations with vector-predication intrinsics,
3863/// using the address to load from, the explicit vector length and an optional
3864/// mask.
3866 : public VPSingleDefRecipe,
3867 public VPWidenMemoryRecipe {
3869 VPValue *Mask)
3870 : VPSingleDefRecipe(VPRecipeBase::VPWidenLoadEVLSC, {Addr, &EVL},
3871 L.getIngredient().getType(), &L.getIngredient(),
3872 L.getDebugLoc()),
3873 VPWidenMemoryRecipe(L.getIngredient(), L.isConsecutive(), L) {
3874 setMask(Mask);
3875 }
3876
3878 llvm_unreachable("cloning not supported");
3879 }
3880
3881 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadEVLSC)
3882
3883 /// Returns the opcode of the widened load.
3884 unsigned getOpcode() const { return Instruction::Load; }
3885
3886 /// Return the EVL operand.
3887 VPValue *getEVL() const { return getOperand(1); }
3888
3889 /// Generate the wide load or gather.
3890 void execute(VPTransformState &State) override;
3891
3892 /// Return the cost of this VPWidenLoadEVLRecipe.
3893 InstructionCost computeCost(ElementCount VF,
3894 VPCostContext &Ctx) const override;
3895
3896 /// Returns true if the recipe only uses the first lane of operand \p Op.
3897 bool usesFirstLaneOnly(const VPValue *Op) const override {
3899 "Op must be an operand of the recipe");
3900 // Widened loads only demand the first lane of EVL and consecutive loads
3901 // only demand the first lane of their address.
3902 return Op == getEVL() || (Op == getAddr() && isConsecutive());
3903 }
3904
3905protected:
3906 VPRecipeBase *getAsRecipe() override;
3907 const VPRecipeBase *getAsRecipe() const override;
3908
3909#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3910 /// Print the recipe.
3911 void printRecipe(raw_ostream &O, const Twine &Indent,
3912 VPSlotTracker &SlotTracker) const override;
3913#endif
3914};
3915
3916/// A recipe for widening store operations, using the stored value, the address
3917/// to store to and an optional mask.
3919 public VPWidenMemoryRecipe {
3921 VPValue *Mask, bool Consecutive,
3922 const VPIRMetadata &Metadata, DebugLoc DL)
3923 : VPRecipeBase(VPRecipeBase::VPWidenStoreSC, {Addr, StoredVal}, DL),
3924 VPWidenMemoryRecipe(Store, Consecutive, Metadata) {
3925 setMask(Mask);
3926 }
3927
3931 *this, getDebugLoc());
3932 }
3933
3934 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreSC);
3935
3936 /// Return the value stored by this recipe.
3937 VPValue *getStoredValue() const { return getOperand(1); }
3938
3939 /// Generate a wide store or scatter.
3940 void execute(VPTransformState &State) override;
3941
3942 /// Return the cost of this VPWidenStoreRecipe.
3944 VPCostContext &Ctx) const override {
3945 return VPWidenMemoryRecipe::computeCost(VF, Ctx);
3946 }
3947
3948 /// Returns true if the recipe only uses the first lane of operand \p Op.
3949 bool usesFirstLaneOnly(const VPValue *Op) const override {
3951 "Op must be an operand of the recipe");
3952 // Widened, consecutive stores only demand the first lane of their address,
3953 // unless the same operand is also stored.
3954 return Op == getAddr() && isConsecutive() && Op != getStoredValue();
3955 }
3956
3957protected:
3958 VPRecipeBase *getAsRecipe() override;
3959 const VPRecipeBase *getAsRecipe() const override;
3960
3961#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
3962 /// Print the recipe.
3963 void printRecipe(raw_ostream &O, const Twine &Indent,
3964 VPSlotTracker &SlotTracker) const override;
3965#endif
3966};
3967
3968/// A recipe for widening store operations with vector-predication intrinsics,
3969/// using the value to store, the address to store to, the explicit vector
3970/// length and an optional mask.
3972 : public VPRecipeBase,
3973 public VPWidenMemoryRecipe {
3975 VPValue *StoredVal, VPValue &EVL, VPValue *Mask)
3976 : VPRecipeBase(VPRecipeBase::VPWidenStoreEVLSC, {Addr, StoredVal, &EVL},
3977 S.getDebugLoc()),
3978 VPWidenMemoryRecipe(S.getIngredient(), S.isConsecutive(), S) {
3979 setMask(Mask);
3980 }
3981
3983 llvm_unreachable("cloning not supported");
3984 }
3985
3986 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreEVLSC)
3987
3988 /// Return the address accessed by this recipe.
3989 VPValue *getStoredValue() const { return getOperand(1); }
3990
3991 /// Return the EVL operand.
3992 VPValue *getEVL() const { return getOperand(2); }
3993
3994 /// Generate the wide store or scatter.
3995 void execute(VPTransformState &State) override;
3996
3997 /// Return the cost of this VPWidenStoreEVLRecipe.
3998 InstructionCost computeCost(ElementCount VF,
3999 VPCostContext &Ctx) const override;
4000
4001 /// Returns true if the recipe only uses the first lane of operand \p Op.
4002 bool usesFirstLaneOnly(const VPValue *Op) const override {
4004 "Op must be an operand of the recipe");
4005 if (Op == getEVL()) {
4006 assert(getStoredValue() != Op && "unexpected store of EVL");
4007 return true;
4008 }
4009 // Widened, consecutive memory operations only demand the first lane of
4010 // their address, unless the same operand is also stored. That latter can
4011 // happen with opaque pointers.
4012 return Op == getAddr() && isConsecutive() && Op != getStoredValue();
4013 }
4014
4015protected:
4016 VPRecipeBase *getAsRecipe() override;
4017 const VPRecipeBase *getAsRecipe() const override;
4018
4019#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4020 /// Print the recipe.
4021 void printRecipe(raw_ostream &O, const Twine &Indent,
4022 VPSlotTracker &SlotTracker) const override;
4023#endif
4024};
4025
4026/// Recipe to expand a SCEV expression.
4028 const SCEV *Expr;
4029
4030public:
4031 VPExpandSCEVRecipe(const SCEV *Expr);
4032
4033 ~VPExpandSCEVRecipe() override = default;
4034
4035 VPExpandSCEVRecipe *clone() override { return new VPExpandSCEVRecipe(Expr); }
4036
4037 VP_CLASSOF_IMPL(VPRecipeBase::VPExpandSCEVSC)
4038
4039 void execute(VPTransformState &State) override {
4040 llvm_unreachable("SCEV expressions must be expanded before final execute");
4041 }
4042
4043 /// Return the cost of this VPExpandSCEVRecipe.
4045 VPCostContext &Ctx) const override {
4046 // TODO: Compute accurate cost after retiring the legacy cost model.
4047 return 0;
4048 }
4049
4050 const SCEV *getSCEV() const { return Expr; }
4051
4052protected:
4053#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4054 /// Print the recipe.
4055 void printRecipe(raw_ostream &O, const Twine &Indent,
4056 VPSlotTracker &SlotTracker) const override;
4057#endif
4058};
4059
4060/// A recipe for generating the active lane mask for the vector loop that is
4061/// used to predicate the vector operations.
4063public:
4065 : VPHeaderPHIRecipe(VPRecipeBase::VPActiveLaneMaskPHISC, nullptr,
4066 StartMask, DL) {}
4067
4068 ~VPActiveLaneMaskPHIRecipe() override = default;
4069
4072 if (getNumOperands() == 2)
4073 R->addBackedgeValue(getOperand(1));
4074 return R;
4075 }
4076
4077 VP_CLASSOF_IMPL(VPRecipeBase::VPActiveLaneMaskPHISC)
4078
4079 /// Generate the active lane mask phi of the vector loop.
4080 void execute(VPTransformState &State) override;
4081
4082protected:
4083#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4084 /// Print the recipe.
4085 void printRecipe(raw_ostream &O, const Twine &Indent,
4086 VPSlotTracker &SlotTracker) const override;
4087#endif
4088};
4089
4090/// A recipe for generating the phi node tracking the current scalar iteration
4091/// index. It starts at the start value of the canonical induction and gets
4092/// incremented by the number of scalar iterations processed by the vector loop
4093/// iteration. The increment does not have to be loop invariant.
4095public:
4097 : VPHeaderPHIRecipe(VPRecipeBase::VPCurrentIterationPHISC, nullptr,
4098 StartIV, DL) {}
4099
4100 ~VPCurrentIterationPHIRecipe() override = default;
4101
4103 llvm_unreachable("cloning not implemented yet");
4104 }
4105
4106 VP_CLASSOF_IMPL(VPRecipeBase::VPCurrentIterationPHISC)
4107
4108 void execute(VPTransformState &State) override {
4109 llvm_unreachable("cannot execute this recipe, should be replaced by a "
4110 "scalar phi recipe");
4111 }
4112
4113 /// Return the cost of this VPCurrentIterationPHIRecipe.
4115 VPCostContext &Ctx) const override {
4116 // For now, match the behavior of the legacy cost model.
4117 return 0;
4118 }
4119
4120 /// Returns true if the recipe only uses the first lane of operand \p Op.
4121 bool usesFirstLaneOnly(const VPValue *Op) const override {
4123 "Op must be an operand of the recipe");
4124 return true;
4125 }
4126
4127protected:
4128#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4129 /// Print the recipe.
4130 LLVM_ABI_FOR_TEST void printRecipe(raw_ostream &O, const Twine &Indent,
4131 VPSlotTracker &SlotTracker) const override;
4132#endif
4133};
4134
4135/// A Recipe for widening the canonical induction variable of the vector loop.
4136/// First operand is the canonical IV recipe, a second step operand (VF * Part)
4137/// is added during unrolling.
4139public:
4141 const VPIRFlags::WrapFlagsTy &Flags = {})
4142 : VPRecipeWithIRFlags(VPRecipeBase::VPWidenCanonicalIVSC, CanonicalIV,
4143 CanonicalIV->getType(), Flags) {}
4144
4145 ~VPWidenCanonicalIVRecipe() override = default;
4146
4148 auto *WideCanIV =
4150 if (VPValue *Step = getStepValue())
4151 WideCanIV->addPerPartStep(Step);
4152 return WideCanIV;
4153 }
4154
4155 VP_CLASSOF_IMPL(VPRecipeBase::VPWidenCanonicalIVSC)
4156
4157 void execute(VPTransformState &State) override {
4158 llvm_unreachable("Expected prior expansion of WidenCanonicalIV recipes");
4159 }
4160
4161 /// Return the cost of this VPWidenCanonicalIVPHIRecipe.
4163 VPCostContext &Ctx) const override {
4164 // TODO: Compute accurate cost after retiring the legacy cost model.
4165 return 0;
4166 }
4167
4168 /// Return the canonical IV being widened.
4172
4174 return getNumOperands() == 2 ? getOperand(1) : nullptr;
4175 }
4176
4177 /// Add the per-part step (VF * Part) used for unrolled parts.
4179 assert(Step->getScalarType() == getScalarType() &&
4180 "per-part step must have the same type as the canonical IV");
4181 VPUser::addOperand(Step);
4182 }
4183
4184protected:
4185#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4186 /// Print the recipe.
4187 void printRecipe(raw_ostream &O, const Twine &Indent,
4188 VPSlotTracker &SlotTracker) const override;
4189#endif
4190};
4191
4192/// A recipe for converting \p Current into \p Start + \p Current * \p Step.
4193/// FastMathFlags are derived from the \p FPBinOp in the case of FP inductions,
4194/// and the passed NoWrap \p Flags apply in the case of Ptr and Int inductions.
4196 /// Kind of the induction.
4198 /// If not nullptr, the floating point induction binary operator. Must be set
4199 /// for floating point inductions.
4200 const FPMathOperator *FPBinOp;
4201
4202public:
4204 const FPMathOperator *FPBinOp, VPValue *Start,
4205 VPValue *Current, VPValue *Step,
4206 const VPIRFlags::WrapFlagsTy &Flags = {})
4207 : VPRecipeWithIRFlags(VPRecipeBase::VPDerivedIVSC, {Start, Current, Step},
4208 Start->getScalarType(), Flags),
4209 Kind(Kind), FPBinOp(FPBinOp) {}
4210
4211 ~VPDerivedIVRecipe() override = default;
4212
4214 return new VPDerivedIVRecipe(Kind, FPBinOp, getStartValue(), getOperand(1),
4216 }
4217
4218 VP_CLASSOF_IMPL(VPRecipeBase::VPDerivedIVSC)
4219
4220 void execute(VPTransformState &State) override {
4221 llvm_unreachable("Expected prior expansion of this recipe");
4222 }
4223
4224 /// Return the cost of this VPDerivedIVRecipe.
4225 InstructionCost computeCost(ElementCount VF,
4226 VPCostContext &Ctx) const override;
4227
4228 VPValue *getStartValue() const { return getOperand(0); }
4229 VPValue *getIndex() const { return getOperand(1); }
4230 VPValue *getStepValue() const { return getOperand(2); }
4231 const FPMathOperator *getFPBinOp() const { return FPBinOp; }
4233
4234 /// Returns true if the recipe only uses the first lane of operand \p Op.
4235 bool usesFirstLaneOnly(const VPValue *Op) const override {
4237 "Op must be an operand of the recipe");
4238 return true;
4239 }
4240
4241protected:
4242#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4243 /// Print the recipe.
4244 void printRecipe(raw_ostream &O, const Twine &Indent,
4245 VPSlotTracker &SlotTracker) const override;
4246#endif
4247};
4248
4249/// A recipe for handling phi nodes of integer and floating-point inductions,
4250/// producing their scalar values. Before unrolling by UF the recipe represents
4251/// the VF*UF scalar values to be produced, or UF scalar values if only first
4252/// lane is used, and has 3 operands: IV, step and VF. Unrolling adds one extra
4253/// operand StartIndex to all unroll parts except part 0, as the recipe
4254/// represents the VF scalar values (this number of values is taken from
4255/// State.VF rather than from the VF operand) starting at IV + StartIndex.
4257 Instruction::BinaryOps InductionOpcode;
4258
4259public:
4263 : VPRecipeWithIRFlags(VPRecipeBase::VPScalarIVStepsSC, {IV, Step, VF},
4264 IV->getScalarType(), FMFs, DL),
4265 InductionOpcode(Opcode) {}
4266
4267 ~VPScalarIVStepsRecipe() override = default;
4268
4270 auto *NewR = new VPScalarIVStepsRecipe(
4271 getOperand(0), getOperand(1), getOperand(2), InductionOpcode,
4273 if (VPValue *StartIndex = getStartIndex())
4274 NewR->setStartIndex(StartIndex);
4275 return NewR;
4276 }
4277
4278 VP_CLASSOF_IMPL(VPRecipeBase::VPScalarIVStepsSC)
4279
4280 /// Generate the scalarized versions of the phi node as needed by their users.
4281 void execute(VPTransformState &State) override;
4282
4283 /// Return the cost of this VPScalarIVStepsRecipe.
4284 InstructionCost computeCost(ElementCount VF,
4285 VPCostContext &Ctx) const override;
4286
4287 VPValue *getStepValue() const { return getOperand(1); }
4288
4289 /// Return the number of scalars to produce per unroll part, used to compute
4290 /// StartIndex during unrolling.
4291 VPValue *getVFValue() const { return getOperand(2); }
4292
4293 /// Return the StartIndex, or null if known to be zero, valid only after
4294 /// unrolling.
4296 return getNumOperands() == 4 ? getOperand(3) : nullptr;
4297 }
4298
4299 /// Set or add the StartIndex operand.
4300 void setStartIndex(VPValue *StartIndex) {
4301 if (getNumOperands() == 4)
4302 setOperand(3, StartIndex);
4303 else
4304 addOperand(StartIndex);
4305 }
4306
4307 /// Returns true if this recipe produces scalar values for all VF lanes.
4308 bool doesGeneratePerAllLanes() const;
4309
4310 /// Returns true if the recipe only uses the first lane of operand \p Op.
4311 bool usesFirstLaneOnly(const VPValue *Op) const override {
4313 "Op must be an operand of the recipe");
4314 return true;
4315 }
4316
4317 Instruction::BinaryOps getInductionOpcode() const { return InductionOpcode; }
4318
4319protected:
4320#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4321 /// Print the recipe.
4322 void printRecipe(raw_ostream &O, const Twine &Indent,
4323 VPSlotTracker &SlotTracker) const override;
4324#endif
4325};
4326
4327/// CastInfo helper for casting from VPRecipeBase to a mixin class that is not
4328/// part of the VPRecipeBase class hierarchy (e.g. VPPhiAccessors,
4329/// VPIRMetadata).
4330namespace vpdetail {
4331template <typename VPMixin, typename... RecipeTys>
4333 : public DefaultDoCastIfPossible<VPMixin *, VPRecipeBase *,
4334 CastInfoMixinImpl<VPMixin, RecipeTys...>> {
4335 static_assert((std::is_base_of_v<VPMixin, RecipeTys> && ...),
4336 "Each type in RecipeTys must derive from VPMixin");
4337
4338 /// Used by isa.
4339 static bool isPossible(VPRecipeBase *R) { return isa<RecipeTys...>(R); }
4340
4341 /// Used by cast.
4342 static VPMixin *doCast(VPRecipeBase *R) {
4343 VPMixin *Out = nullptr;
4344 ((Out = dyn_cast<RecipeTys>(R)) || ...);
4345 assert(Out && "Illegal recipe for cast");
4346 return Out;
4347 }
4348 static VPMixin *castFailed() { return nullptr; }
4349};
4350} // namespace vpdetail
4351
4352/// Support casting from VPRecipeBase -> VPPhiAccessors.
4353template <>
4357
4358template <>
4363template <>
4365 : public ForwardToPointerCast<VPPhiAccessors, VPRecipeBase *,
4366 CastInfo<VPPhiAccessors, VPRecipeBase *>> {};
4367
4368/// Support casting from VPRecipeBase / VPUser -> VPWidenMemoryRecipe.
4369template <>
4374template <>
4379
4380/// Support casting from VPSingleDefRecipe -> VPWidenMemoryRecipe (loads only).
4381template <>
4385template <>
4390
4391/// Support casting from VPRecipeBase -> VPIRMetadata.
4392template <>
4399
4400template <>
4405template <>
4407 : public ForwardToPointerCast<VPIRMetadata, VPRecipeBase *,
4408 CastInfo<VPIRMetadata, VPRecipeBase *>> {};
4409
4410/// VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph. It
4411/// holds a sequence of zero or more VPRecipe's each representing a sequence of
4412/// output IR instructions. All PHI-like recipes must come before any non-PHI
4413/// recipes.
4414class LLVM_ABI_FOR_TEST VPBasicBlock : public VPBlockBase {
4415 friend class VPlan;
4416
4417 /// Use VPlan::createVPBasicBlock to create VPBasicBlocks.
4418 VPBasicBlock(const Twine &Name = "", VPRecipeBase *Recipe = nullptr)
4419 : VPBlockBase(VPBasicBlockSC, Name.str()) {
4420 if (Recipe)
4421 appendRecipe(Recipe);
4422 }
4423
4424public:
4426
4427protected:
4428 /// The VPRecipes held in the order of output instructions to generate.
4430
4431 VPBasicBlock(VPBlockTy BlockSC, const Twine &Name = "")
4432 : VPBlockBase(BlockSC, Name.str()) {}
4433
4434public:
4435 ~VPBasicBlock() override {
4436 while (!Recipes.empty())
4437 Recipes.pop_back();
4438 }
4439
4440 /// Instruction iterators...
4445
4446 //===--------------------------------------------------------------------===//
4447 /// Recipe iterator methods
4448 ///
4449 inline iterator begin() { return Recipes.begin(); }
4450 inline const_iterator begin() const { return Recipes.begin(); }
4451 inline iterator end() { return Recipes.end(); }
4452 inline const_iterator end() const { return Recipes.end(); }
4453
4454 inline reverse_iterator rbegin() { return Recipes.rbegin(); }
4455 inline const_reverse_iterator rbegin() const { return Recipes.rbegin(); }
4456 inline reverse_iterator rend() { return Recipes.rend(); }
4457 inline const_reverse_iterator rend() const { return Recipes.rend(); }
4458
4459 inline size_t size() const { return Recipes.size(); }
4460 inline bool empty() const { return Recipes.empty(); }
4461 inline const VPRecipeBase &front() const { return Recipes.front(); }
4462 inline VPRecipeBase &front() { return Recipes.front(); }
4463 inline const VPRecipeBase &back() const { return Recipes.back(); }
4464 inline VPRecipeBase &back() { return Recipes.back(); }
4465
4466 /// Returns a reference to the list of recipes.
4468
4469 /// Returns a pointer to a member of the recipe list.
4470 static RecipeListTy VPBasicBlock::*getSublistAccess(VPRecipeBase *) {
4471 return &VPBasicBlock::Recipes;
4472 }
4473
4474 /// Method to support type inquiry through isa, cast, and dyn_cast.
4475 static inline bool classof(const VPBlockBase *V) {
4476 return V->getVPBlockID() == VPBlockBase::VPBasicBlockSC ||
4477 V->getVPBlockID() == VPBlockBase::VPIRBasicBlockSC;
4478 }
4479
4480 void insert(VPRecipeBase *Recipe, iterator InsertPt) {
4481 assert(Recipe && "No recipe to append.");
4482 assert(!Recipe->Parent && "Recipe already in VPlan");
4483 Recipe->Parent = this;
4484 Recipes.insert(InsertPt, Recipe);
4485 }
4486
4487 /// Augment the existing recipes of a VPBasicBlock with an additional
4488 /// \p Recipe as the last recipe.
4489 void appendRecipe(VPRecipeBase *Recipe) { insert(Recipe, end()); }
4490
4491 /// The method which generates the output IR instructions that correspond to
4492 /// this VPBasicBlock, thereby "executing" the VPlan.
4493 void execute(VPTransformState *State) override;
4494
4495 /// Return the cost of this VPBasicBlock.
4496 InstructionCost cost(ElementCount VF, VPCostContext &Ctx) override;
4497
4498 /// Return the position of the first non-phi node recipe in the block.
4499 iterator getFirstNonPhi();
4500
4501 /// Returns an iterator range over the PHI-like recipes in the block.
4505
4506 /// Split current block at \p SplitAt by inserting a new block between the
4507 /// current block and its successors and moving all recipes starting at
4508 /// SplitAt to the new block. Returns the new block.
4509 VPBasicBlock *splitAt(iterator SplitAt);
4510
4511 VPRegionBlock *getEnclosingLoopRegion();
4512 const VPRegionBlock *getEnclosingLoopRegion() const;
4513
4514#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4515 /// Print this VPBsicBlock to \p O, prefixing all lines with \p Indent. \p
4516 /// SlotTracker is used to print unnamed VPValue's using consequtive numbers.
4517 ///
4518 /// Note that the numbering is applied to the whole VPlan, so printing
4519 /// individual blocks is consistent with the whole VPlan printing.
4520 void print(raw_ostream &O, const Twine &Indent,
4521 VPSlotTracker &SlotTracker) const override;
4522 using VPBlockBase::print; // Get the print(raw_stream &O) version.
4523#endif
4524
4525 /// If the block has multiple successors, return the branch recipe terminating
4526 /// the block. If there are no or only a single successor, return nullptr;
4527 VPRecipeBase *getTerminator();
4528 const VPRecipeBase *getTerminator() const;
4529
4530 /// Returns true if the block is exiting it's parent region.
4531 bool isExiting() const;
4532
4533 /// Clone the current block and it's recipes, without updating the operands of
4534 /// the cloned recipes.
4535 VPBasicBlock *clone() override;
4536
4537 /// Returns the predecessor block at index \p Idx with the predecessors as per
4538 /// the corresponding plain CFG. If the block is an entry block to a region,
4539 /// the first predecessor is the single predecessor of a region, and the
4540 /// second predecessor is the exiting block of the region.
4541 const VPBasicBlock *getCFGPredecessor(unsigned Idx) const;
4542
4543protected:
4544 /// Execute the recipes in the IR basic block \p BB.
4545 void executeRecipes(VPTransformState *State, BasicBlock *BB);
4546
4547 /// Connect the VPBBs predecessors' in the VPlan CFG to the IR basic block
4548 /// generated for this VPBB.
4549 void connectToPredecessors(VPTransformState &State);
4550
4551private:
4552 /// Create an IR BasicBlock to hold the output instructions generated by this
4553 /// VPBasicBlock, and return it. Update the CFGState accordingly.
4554 BasicBlock *createEmptyBasicBlock(VPTransformState &State);
4555};
4556
4557inline const VPBasicBlock *
4559 return getAsRecipe()->getParent()->getCFGPredecessor(Idx);
4560}
4561
4562/// A special type of VPBasicBlock that wraps an existing IR basic block.
4563/// Recipes of the block get added before the first non-phi instruction in the
4564/// wrapped block.
4565/// Note: At the moment, VPIRBasicBlock can only be used to wrap VPlan's
4566/// preheader block.
4567class VPIRBasicBlock : public VPBasicBlock {
4568 friend class VPlan;
4569
4570 BasicBlock *IRBB;
4571
4572 /// Use VPlan::createVPIRBasicBlock to create VPIRBasicBlocks.
4573 VPIRBasicBlock(BasicBlock *IRBB)
4574 : VPBasicBlock(VPIRBasicBlockSC,
4575 (Twine("ir-bb<") + IRBB->getName() + Twine(">")).str()),
4576 IRBB(IRBB) {}
4577
4578public:
4579 ~VPIRBasicBlock() override = default;
4580
4581 static inline bool classof(const VPBlockBase *V) {
4582 return V->getVPBlockID() == VPBlockBase::VPIRBasicBlockSC;
4583 }
4584
4585 /// The method which generates the output IR instructions that correspond to
4586 /// this VPBasicBlock, thereby "executing" the VPlan.
4587 void execute(VPTransformState *State) override;
4588
4589 VPIRBasicBlock *clone() override;
4590
4591 BasicBlock *getIRBasicBlock() const { return IRBB; }
4592};
4593
4594/// Track information about the canonical IV and header mask of a loop region.
4595/// TODO: Have it also track the canonical IV increment, subject of NUW flag.
4597 /// VPRegionValue for the canonical IV, whose allocation is managed by
4598 /// VPCanonicalIVInfo.
4599 std::unique_ptr<VPRegionValue> CanIV;
4600
4601 /// Optional VPRegionValue for the header mask, set when tail folding.
4602 std::unique_ptr<VPRegionValue> HeaderMask;
4603
4604 /// Whether the increment of the canonical IV may unsigned wrap or not.
4605 bool HasNUW = true;
4606
4607public:
4609 : CanIV(std::make_unique<VPRegionValue>(Ty, DL, Region)) {}
4610
4611 VPRegionValue *getRegionValue() { return CanIV.get(); }
4612 const VPRegionValue *getRegionValue() const { return CanIV.get(); }
4613
4614 VPRegionValue *getHeaderMask() const { return HeaderMask.get(); }
4615
4616 /// Create the header mask for the region and return it. Must only be called
4617 /// when no header mask exists yet.
4619 assert(!HeaderMask && "Header mask already created");
4620 HeaderMask = std::make_unique<VPRegionValue>(
4621 Type::getInt1Ty(CanIV->getType()->getContext()), DebugLoc::getUnknown(),
4622 CanIV->getDefiningRegion());
4623 return HeaderMask.get();
4624 }
4625
4626 bool hasNUW() const { return HasNUW; }
4627
4628 void clearNUW() { HasNUW = false; }
4629};
4630
4631/// VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks
4632/// which form a Single-Entry-Single-Exiting subgraph of the output IR CFG.
4633/// A VPRegionBlock may indicate that its contents are to be replicated several
4634/// times. This is designed to support predicated scalarization, in which a
4635/// scalar if-then code structure needs to be generated VF * UF times. Having
4636/// this replication indicator helps to keep a single model for multiple
4637/// candidate VF's. The actual replication takes place only once the desired VF
4638/// and UF have been determined.
4639class LLVM_ABI_FOR_TEST VPRegionBlock : public VPBlockBase {
4640 friend class VPlan;
4641
4642 /// Hold the Single Entry of the SESE region modelled by the VPRegionBlock.
4643 VPBlockBase *Entry;
4644
4645 /// Hold the Single Exiting block of the SESE region modelled by the
4646 /// VPRegionBlock.
4647 VPBlockBase *Exiting;
4648
4649 /// Holds the Canonical IV of the loop region along with additional
4650 /// information. If CanIVInfo is nullptr, the region is a replicating region.
4651 /// Loop regions retain their canonical IVs until they are dissolved, even if
4652 /// the canonical IV has no users.
4653 std::unique_ptr<VPCanonicalIVInfo> CanIVInfo;
4654
4655 /// Use VPlan::createLoopRegion() and VPlan::createReplicateRegion() to create
4656 /// VPRegionBlocks.
4657 VPRegionBlock(VPBlockBase *Entry, VPBlockBase *Exiting,
4658 const std::string &Name = "")
4659 : VPBlockBase(VPRegionBlockSC, Name), Entry(Entry), Exiting(Exiting) {
4660 if (Entry) {
4661 assert(!Entry->hasPredecessors() && "Entry block has predecessors.");
4662 assert(Exiting && "Must also pass Exiting if Entry is passed.");
4663 assert(!Exiting->hasSuccessors() && "Exit block has successors.");
4664 Entry->setParent(this);
4665 Exiting->setParent(this);
4666 }
4667 }
4668
4669 VPRegionBlock(Type *CanIVTy, DebugLoc DL, VPBlockBase *Entry,
4670 VPBlockBase *Exiting, const std::string &Name = "")
4671 : VPRegionBlock(Entry, Exiting, Name) {
4672 CanIVInfo = std::make_unique<VPCanonicalIVInfo>(CanIVTy, DL, this);
4673 }
4674
4675public:
4676 ~VPRegionBlock() override = default;
4677
4678 /// Method to support type inquiry through isa, cast, and dyn_cast.
4679 static inline bool classof(const VPBlockBase *V) {
4680 return V->getVPBlockID() == VPBlockBase::VPRegionBlockSC;
4681 }
4682
4683 const VPBlockBase *getEntry() const { return Entry; }
4684 VPBlockBase *getEntry() { return Entry; }
4685
4686 /// Set \p EntryBlock as the entry VPBlockBase of this VPRegionBlock. \p
4687 /// EntryBlock must have no predecessors.
4688 void setEntry(VPBlockBase *EntryBlock) {
4689 assert(!EntryBlock->hasPredecessors() &&
4690 "Entry block cannot have predecessors.");
4691 Entry = EntryBlock;
4692 EntryBlock->setParent(this);
4693 }
4694
4695 const VPBlockBase *getExiting() const { return Exiting; }
4696 VPBlockBase *getExiting() { return Exiting; }
4697
4698 /// Set \p ExitingBlock as the exiting VPBlockBase of this VPRegionBlock. \p
4699 /// ExitingBlock must have no successors.
4700 void setExiting(VPBlockBase *ExitingBlock) {
4701 assert(!ExitingBlock->hasSuccessors() &&
4702 "Exit block cannot have successors.");
4703 Exiting = ExitingBlock;
4704 ExitingBlock->setParent(this);
4705 }
4706
4707 /// Returns the pre-header VPBasicBlock of the loop region.
4709 assert(!isReplicator() && "should only get pre-header of loop regions");
4710 return getSinglePredecessor()->getExitingBasicBlock();
4711 }
4712
4713 /// An indicator whether this region is to generate multiple replicated
4714 /// instances of output IR corresponding to its VPBlockBases.
4715 bool isReplicator() const { return !CanIVInfo; }
4716
4717 /// Return the VPBranchOnMaskRecipe from the entry block of this replicating
4718 /// region.
4719 const VPBranchOnMaskRecipe *getEntryBranchOnMask() const;
4721 return const_cast<VPBranchOnMaskRecipe *>(
4722 static_cast<const VPRegionBlock *>(this)->getEntryBranchOnMask());
4723 }
4724
4725 /// The method which generates the output IR instructions that correspond to
4726 /// this VPRegionBlock, thereby "executing" the VPlan.
4727 void execute(VPTransformState *State) override;
4728
4729 // Return the cost of this region.
4730 InstructionCost cost(ElementCount VF, VPCostContext &Ctx) override;
4731
4732#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
4733 /// Print this VPRegionBlock to \p O (recursively), prefixing all lines with
4734 /// \p Indent. \p SlotTracker is used to print unnamed VPValue's using
4735 /// consequtive numbers.
4736 ///
4737 /// Note that the numbering is applied to the whole VPlan, so printing
4738 /// individual regions is consistent with the whole VPlan printing.
4739 void print(raw_ostream &O, const Twine &Indent,
4740 VPSlotTracker &SlotTracker) const override;
4741 using VPBlockBase::print; // Get the print(raw_stream &O) version.
4742#endif
4743
4744 /// Clone all blocks in the single-entry single-exit region of the block and
4745 /// their recipes without updating the operands of the cloned recipes.
4746 VPRegionBlock *clone() override;
4747
4748 /// Remove the current region from its VPlan, connecting its predecessor to
4749 /// its entry, and its exiting block to its successor.
4750 void dissolveToCFGLoop();
4751
4752 /// Get the canonical IV increment instruction if it exists. Otherwise, create
4753 /// a new increment before the terminator and return it. The canonical IV
4754 /// increment is subject to DCE if unused, unlike the canonical IV itself.
4755 VPInstruction *getOrCreateCanonicalIVIncrement();
4756
4757 /// Return the canonical induction variable of the region, null for
4758 /// replicating regions.
4760 return CanIVInfo ? CanIVInfo->getRegionValue() : nullptr;
4761 }
4763 return CanIVInfo ? CanIVInfo->getRegionValue() : nullptr;
4764 }
4765
4766 /// Return the type of the canonical IV for loop regions.
4768 return CanIVInfo->getRegionValue()->getType();
4769 }
4770
4771 /// Return the header mask of the region, or null if not set.
4773 return CanIVInfo ? CanIVInfo->getHeaderMask() : nullptr;
4774 }
4775
4776 /// Return the header mask if it exists and is used, or null otherwise. The
4777 /// mask is materialized into concrete recipes only after costing, so cost and
4778 /// codegen accounting sites use this to skip an unused mask.
4780 VPRegionValue *HeaderMask = getHeaderMask();
4781 return HeaderMask && HeaderMask->getNumUsers() > 0 ? HeaderMask : nullptr;
4782 }
4783
4784 /// Create the header mask for the region and return it. Must only be called
4785 /// on loop regions that don't already have a header mask.
4787 assert(CanIVInfo && "Can only create header mask for loop regions");
4788 return CanIVInfo->createHeaderMask();
4789 }
4790
4791 /// Return the region values of the loop region (canonical IV, header mask)
4792 /// or an empty vector for replicate regions.
4794 if (!CanIVInfo)
4795 return {};
4796 SmallVector<VPRegionValue *, 2> R = {CanIVInfo->getRegionValue()};
4797 if (auto *HM = CanIVInfo->getHeaderMask())
4798 R.push_back(HM);
4799 return R;
4800 }
4801
4802 /// Indicates if NUW is set for the canonical IV increment, for loop regions.
4803 bool hasCanonicalIVNUW() const { return CanIVInfo->hasNUW(); }
4804
4805 /// Unsets NUW for the canonical IV increment \p Increment, for loop regions.
4807 assert(Increment && "Must provide increment to clear");
4808 Increment->dropPoisonGeneratingFlags();
4809 CanIVInfo->clearNUW();
4810 }
4811};
4812
4814 return getParent()->getParent();
4815}
4816
4818 return getParent()->getParent();
4819}
4820
4821/// VPlan models a candidate for vectorization, encoding various decisions take
4822/// to produce efficient output IR, including which branches, basic-blocks and
4823/// output IR instructions to generate, and their cost. VPlan holds a
4824/// Hierarchical-CFG of VPBasicBlocks and VPRegionBlocks rooted at an Entry
4825/// VPBasicBlock.
4826class VPlan {
4827 friend class VPlanPrinter;
4828 friend class VPSlotTracker;
4829
4830 /// VPBasicBlock corresponding to the original preheader. Used to place
4831 /// VPExpandSCEV recipes for expressions used during skeleton creation and the
4832 /// rest of VPlan execution.
4833 /// When this VPlan is used for the epilogue vector loop, the entry will be
4834 /// replaced by a new entry block created during skeleton creation.
4835 VPBasicBlock *Entry;
4836
4837 /// VPIRBasicBlock wrapping the header of the original scalar loop.
4838 VPIRBasicBlock *ScalarHeader;
4839
4840 /// Immutable list of VPIRBasicBlocks wrapping the exit blocks of the original
4841 /// scalar loop. Note that some exit blocks may be unreachable at the moment,
4842 /// e.g. if the scalar epilogue always executes.
4844
4845 /// Holds the VFs applicable to this VPlan.
4847
4848 /// Holds the UFs applicable to this VPlan. If empty, the VPlan is valid for
4849 /// any UF.
4851
4852 /// Holds the name of the VPlan, for printing.
4853 std::string Name;
4854
4855 /// Represents the trip count of the original loop, for folding
4856 /// the tail.
4857 VPValue *TripCount = nullptr;
4858
4859 /// Represents the backedge taken count of the original loop, for folding
4860 /// the tail. It equals TripCount - 1.
4861 VPSymbolicValue *BackedgeTakenCount = nullptr;
4862
4863 /// Represents the vector trip count.
4864 VPSymbolicValue VectorTripCount;
4865
4866 /// Represents the vectorization factor of the loop.
4867 VPSymbolicValue VF;
4868
4869 /// Represents the unroll factor of the loop.
4870 VPSymbolicValue UF;
4871
4872 /// Represents the loop-invariant VF * UF of the vector loop region.
4873 VPSymbolicValue VFxUF;
4874
4875 /// Contains all the external definitions created for this VPlan, as a mapping
4876 /// from IR Values to VPIRValues.
4878
4879 /// Blocks allocated and owned by the VPlan. They will be deleted once the
4880 /// VPlan is destroyed.
4881 SmallVector<VPBlockBase *> CreatedBlocks;
4882
4883 /// Construct a VPlan with \p Entry to the plan and with \p ScalarHeader
4884 /// wrapping the original header of the scalar loop. The vector loop will have
4885 /// index type \p IdxTy.
4886 VPlan(VPBasicBlock *Entry, VPIRBasicBlock *ScalarHeader, Type *IdxTy)
4887 : Entry(Entry), ScalarHeader(ScalarHeader), VectorTripCount(IdxTy),
4888 VF(IdxTy), UF(IdxTy), VFxUF(IdxTy) {
4889 Entry->setPlan(this);
4890 assert(ScalarHeader->getNumSuccessors() == 0 &&
4891 "scalar header must be a leaf node");
4892 }
4893
4894public:
4895 /// Construct a VPlan for \p L. This will create VPIRBasicBlocks wrapping the
4896 /// original preheader and scalar header of \p L, to be used as entry and
4897 /// scalar header blocks of the new VPlan. The vector loop will have index
4898 /// type \p IdxTy.
4899 VPlan(Loop *L, Type *IdxTy);
4900
4901 /// Construct a VPlan with a new VPBasicBlock as entry, a VPIRBasicBlock
4902 /// wrapping \p ScalarHeaderBB and vector loop index of type \p IdxTy.
4903 VPlan(BasicBlock *ScalarHeaderBB, Type *IdxTy)
4904 : VectorTripCount(IdxTy), VF(IdxTy), UF(IdxTy), VFxUF(IdxTy) {
4905 setEntry(createVPBasicBlock("preheader"));
4906 ScalarHeader = createVPIRBasicBlock(ScalarHeaderBB);
4907 }
4908
4910
4912 Entry = VPBB;
4913 VPBB->setPlan(this);
4914 }
4915
4916 /// Generate the IR code for this VPlan.
4917 void execute(VPTransformState *State);
4918
4919 /// Return the cost of this plan.
4921
4922 VPBasicBlock *getEntry() { return Entry; }
4923 const VPBasicBlock *getEntry() const { return Entry; }
4924
4925 /// Returns the preheader of the vector loop region, if one exists, or null
4926 /// otherwise.
4928 const VPRegionBlock *VectorRegion = getVectorLoopRegion();
4929 return VectorRegion
4930 ? cast<VPBasicBlock>(VectorRegion->getSinglePredecessor())
4931 : nullptr;
4932 }
4933
4934 /// Returns the VPRegionBlock of the vector loop.
4937
4938 /// Returns true if this VPlan is for an outer loop, i.e., its vector
4939 /// loop region contains a nested loop region.
4940 LLVM_ABI_FOR_TEST bool isOuterLoop() const;
4941
4942 /// Returns true if the vector loop region is tail-folded.
4943 bool hasTailFolded() const {
4944 const VPRegionBlock *LoopRegion = getVectorLoopRegion();
4945 return LoopRegion && LoopRegion->getHeaderMask();
4946 }
4947
4948 /// Returns true if the plan requires a scalar epilogue after the vector
4949 /// loop. Must be called before removeBranchOnConst.
4951 const VPBasicBlock *MiddleVPBB = getMiddleBlock();
4952 return MiddleVPBB->getSingleSuccessor() == getScalarPreheader();
4953 }
4954
4955 /// Returns the 'middle' block of the plan, that is the block that selects
4956 /// whether to execute the scalar tail loop or the exit block from the loop
4957 /// latch. If there is an early exit from the vector loop, the middle block
4958 /// conceptully has the early exit block as third successor, split accross 2
4959 /// VPBBs. In that case, the second VPBB selects whether to execute the scalar
4960 /// tail loop or the exit block. If the scalar tail loop or exit block are
4961 /// known to always execute, the middle block may branch directly to that
4962 /// block. This function cannot be called once the vector loop region has been
4963 /// removed.
4965 VPRegionBlock *LoopRegion = getVectorLoopRegion();
4966 assert(
4967 LoopRegion &&
4968 "cannot call the function after vector loop region has been removed");
4969 // The middle block is always the last successor of the region.
4970 return cast<VPBasicBlock>(LoopRegion->getSuccessors().back());
4971 }
4972
4974 return const_cast<VPlan *>(this)->getMiddleBlock();
4975 }
4976
4977 /// Return the VPBasicBlock for the preheader of the scalar loop.
4980 getScalarHeader()->getSinglePredecessor());
4981 }
4982
4983 /// Return the VPIRBasicBlock wrapping the header of the scalar loop.
4984 VPIRBasicBlock *getScalarHeader() const { return ScalarHeader; }
4985
4986 /// Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of
4987 /// the original scalar loop.
4988 ArrayRef<VPIRBasicBlock *> getExitBlocks() const { return ExitBlocks; }
4989
4990 /// Returns true if \p VPBB is an exit block.
4991 bool isExitBlock(VPBlockBase *VPBB);
4992
4993 /// The trip count of the original loop.
4995 assert(TripCount && "trip count needs to be set before accessing it");
4996 return TripCount;
4997 }
4998
4999 /// Set the trip count assuming it is currently null; if it is not - use
5000 /// resetTripCount().
5001 void setTripCount(VPValue *NewTripCount) {
5002 assert(!TripCount && NewTripCount && "TripCount should not be set yet.");
5003 TripCount = NewTripCount;
5004 }
5005
5006 /// Resets the trip count for the VPlan. The caller must make sure all uses of
5007 /// the original trip count have been replaced.
5008 void resetTripCount(VPValue *NewTripCount) {
5009 assert(TripCount && NewTripCount && TripCount->user_empty() &&
5010 "TripCount must be set when resetting");
5011 TripCount = NewTripCount;
5012 }
5013
5014 /// The backedge taken count of the original loop.
5016 // BTC shares the canonical IV type with VectorTripCount.
5017 if (!BackedgeTakenCount)
5018 BackedgeTakenCount = new VPSymbolicValue(VectorTripCount.getType());
5019 return BackedgeTakenCount;
5020 }
5021 VPValue *getBackedgeTakenCount() const { return BackedgeTakenCount; }
5022
5023 /// The vector trip count.
5024 VPSymbolicValue &getVectorTripCount() { return VectorTripCount; }
5025
5026 /// Returns the VF of the vector loop region.
5027 VPSymbolicValue &getVF() { return VF; };
5028 const VPSymbolicValue &getVF() const { return VF; };
5029
5030 /// Returns the UF of the vector loop region.
5031 VPSymbolicValue &getUF() { return UF; };
5032
5033 /// Returns VF * UF of the vector loop region.
5034 VPSymbolicValue &getVFxUF() { return VFxUF; }
5035
5038 }
5039
5040 const DataLayout &getDataLayout() const {
5042 }
5043
5046 }
5047
5048 void addVF(ElementCount VF) { VFs.insert(VF); }
5049
5051 assert(hasVF(VF) && "Cannot set VF not already in plan");
5052 VFs.clear();
5053 VFs.insert(VF);
5054 }
5055
5056 /// Remove \p VF from the plan.
5058 assert(hasVF(VF) && "tried to remove VF not present in plan");
5059 VFs.remove(VF);
5060 }
5061
5062 bool hasVF(ElementCount VF) const { return VFs.count(VF); }
5063 bool hasScalableVF() const {
5064 return any_of(VFs, [](ElementCount VF) { return VF.isScalable(); });
5065 }
5066
5067 /// Returns an iterator range over all VFs of the plan.
5070 return VFs;
5071 }
5072
5073 /// Returns the single VF of the plan, asserting that the plan has exactly
5074 /// one VF.
5076 assert(VFs.size() == 1 && "expected plan with single VF");
5077 return VFs[0];
5078 }
5079
5080 bool hasScalarVFOnly() const {
5081 bool HasScalarVFOnly = VFs.size() == 1 && VFs[0].isScalar();
5082 assert(HasScalarVFOnly == hasVF(ElementCount::getFixed(1)) &&
5083 "Plan with scalar VF should only have a single VF");
5084 return HasScalarVFOnly;
5085 }
5086
5087 bool hasUF(unsigned UF) const { return UFs.empty() || UFs.contains(UF); }
5088
5089 /// Returns the concrete UF of the plan, after unrolling.
5090 unsigned getConcreteUF() const {
5091 assert(UFs.size() == 1 && "Expected a single UF");
5092 return UFs[0];
5093 }
5094
5095 void setUF(unsigned UF) {
5096 assert(hasUF(UF) && "Cannot set the UF not already in plan");
5097 UFs.clear();
5098 UFs.insert(UF);
5099 }
5100
5101 /// Returns true if the VPlan already has been unrolled, i.e. it has a single
5102 /// concrete UF.
5103 bool isUnrolled() const { return UFs.size() == 1; }
5104
5105 /// Return a string with the name of the plan and the applicable VFs and UFs.
5106 std::string getName() const;
5107
5108 void setName(const Twine &newName) { Name = newName.str(); }
5109
5110 /// Gets the live-in VPIRValue for \p V or adds a new live-in (if none exists
5111 /// yet) for \p V.
5113 assert(V && "Trying to get or add the VPIRValue of a null Value");
5114 auto [It, Inserted] = LiveIns.try_emplace(V);
5115 if (Inserted) {
5116 if (auto *CI = dyn_cast<ConstantInt>(V))
5117 It->second = new VPConstantInt(CI);
5118 else
5119 It->second = new VPIRValue(V);
5120 }
5121
5122 assert(isa<VPIRValue>(It->second) &&
5123 "Only VPIRValues should be in mapping");
5124 return It->second;
5125 }
5127 assert(V && "Trying to get or add the VPIRValue of a null VPIRValue");
5128 return getOrAddLiveIn(V->getValue());
5129 }
5130
5131 /// Return a VPIRValue wrapping i1 true.
5132 VPIRValue *getTrue() { return getConstantInt(1, 1); }
5133
5134 /// Return a VPIRValue wrapping i1 false.
5135 VPIRValue *getFalse() { return getConstantInt(1, 0); }
5136
5137 /// Return a VPIRValue wrapping the null value of type \p Ty.
5138 VPIRValue *getZero(Type *Ty) { return getConstantInt(Ty, 0); }
5139
5140 /// Return a VPIRValue wrapping the AllOnes value of type \p Ty.
5142 return getConstantInt(APInt::getAllOnes(Ty->getIntegerBitWidth()));
5143 }
5144
5145 /// Return a VPIRValue wrapping a ConstantInt with the given type and value.
5146 VPIRValue *getConstantInt(Type *Ty, uint64_t Val, bool IsSigned = false) {
5147 return getOrAddLiveIn(ConstantInt::get(Ty, Val, IsSigned));
5148 }
5149
5150 /// Return a VPIRValue wrapping a ConstantInt with the given bitwidth and
5151 /// value.
5153 bool IsSigned = false) {
5154 return getConstantInt(APInt(BitWidth, Val, IsSigned));
5155 }
5156
5157 /// Return a VPIRValue wrapping a ConstantInt with the given APInt value.
5159 return getOrAddLiveIn(ConstantInt::get(getContext(), Val));
5160 }
5161
5162 /// Return a VPIRValue wrapping a poison value of type \p Ty.
5164 return getOrAddLiveIn(PoisonValue::get(Ty));
5165 }
5166
5167 /// Return the live-in VPIRValue for \p V, if there is one or nullptr
5168 /// otherwise.
5169 VPIRValue *getLiveIn(Value *V) const { return LiveIns.lookup(V); }
5170
5171 /// Return the list of live-in VPValues available in the VPlan.
5172 auto getLiveIns() const { return LiveIns.values(); }
5173
5174#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5175 /// Print the live-ins of this VPlan to \p O.
5176 void printLiveIns(raw_ostream &O) const;
5177
5178 /// Print this VPlan to \p O.
5179 LLVM_ABI_FOR_TEST void print(raw_ostream &O) const;
5180
5181 /// Print this VPlan in DOT format to \p O.
5182 LLVM_ABI_FOR_TEST void printDOT(raw_ostream &O) const;
5183
5184 /// Dump the plan to stderr (for debugging).
5185 LLVM_DUMP_METHOD void dump() const;
5186#endif
5187
5188 /// Clone the current VPlan, update all VPValues of the new VPlan and cloned
5189 /// recipes to refer to the clones, and return it.
5191
5192 /// Create a new VPBasicBlock with \p Name and containing \p Recipe if
5193 /// present. The returned block is owned by the VPlan and deleted once the
5194 /// VPlan is destroyed.
5196 VPRecipeBase *Recipe = nullptr) {
5197 auto *VPB = new VPBasicBlock(Name, Recipe);
5198 VPB->setPlan(this);
5199 VPB->setNumber(CreatedBlocks.size());
5200 CreatedBlocks.push_back(VPB);
5201 return VPB;
5202 }
5203
5204 /// Create a new loop region with a canonical IV using \p CanIVTy and
5205 /// \p DL. Use \p Name as the region's name and set entry and exiting blocks
5206 /// to \p Entry and \p Exiting respectively, if provided. The returned block
5207 /// is owned by the VPlan and deleted once the VPlan is destroyed.
5209 const std::string &Name = "",
5210 VPBlockBase *Entry = nullptr,
5211 VPBlockBase *Exiting = nullptr) {
5212 auto *VPB = new VPRegionBlock(CanIVTy, DL, Entry, Exiting, Name);
5213 VPB->setPlan(this);
5214 VPB->setNumber(CreatedBlocks.size());
5215 CreatedBlocks.push_back(VPB);
5216 return VPB;
5217 }
5218
5219 /// Create a new replicate region with \p Entry, \p Exiting and \p Name. The
5220 /// returned block is owned by the VPlan and deleted once the VPlan is
5221 /// destroyed.
5223 const std::string &Name = "") {
5224 auto *VPB = new VPRegionBlock(Entry, Exiting, Name);
5225 VPB->setPlan(this);
5226 VPB->setNumber(CreatedBlocks.size());
5227 CreatedBlocks.push_back(VPB);
5228 return VPB;
5229 }
5230
5231 /// Create a VPIRBasicBlock wrapping \p IRBB, but do not create
5232 /// VPIRInstructions wrapping the instructions in t\p IRBB. The returned
5233 /// block is owned by the VPlan and deleted once the VPlan is destroyed.
5235
5236 /// Create a VPIRBasicBlock from \p IRBB containing VPIRInstructions for all
5237 /// instructions in \p IRBB, except its terminator which is managed by the
5238 /// successors of the block in VPlan. The returned block is owned by the VPlan
5239 /// and deleted once the VPlan is destroyed.
5241
5242 unsigned getMaxBlockNumber() const { return CreatedBlocks.size(); }
5243
5244 /// Returns true if the VPlan is based on a loop with an early exit.
5245 bool hasEarlyExit() const {
5246 unsigned NumExitPredecessors =
5247 sum_of(map_range(ExitBlocks, [](VPIRBasicBlock *EB) {
5248 return EB->getNumPredecessors();
5249 }));
5250
5251 // If the scalar preheader executes unconditionally, there's no branch from
5252 // middle block to any exit. If there is any edge to an exit block
5253 // remaining, it must be an early exit.
5254 VPBasicBlock *ScalarPH = getScalarPreheader();
5255 VPBlockBase *ScalarPHPred =
5256 ScalarPH ? ScalarPH->getSinglePredecessor() : nullptr;
5257 if (ScalarPHPred && ScalarPHPred->getNumSuccessors() == 1)
5258 return NumExitPredecessors >= 1;
5259
5260 // Otherwise there must be at least 2 edges to exit blocks (from the middle
5261 // block and the early exiting edge).
5262 return NumExitPredecessors > 1;
5263 }
5264
5265 /// Returns true if the scalar tail may execute after the vector loop, i.e.
5266 /// if the middle block is a predecessor of the scalar preheader. Note that
5267 /// this relies on unneeded branches to the scalar tail loop being removed.
5268 bool hasScalarTail() const {
5269 auto *ScalarPH = getScalarPreheader();
5270 return ScalarPH &&
5271 is_contained(ScalarPH->getPredecessors(), getMiddleBlock());
5272 }
5273
5274 /// The type of the canonical induction variable of the vector loop.
5275 Type *getIndexType() const { return VF.getType(); }
5276};
5277
5278#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
5279inline raw_ostream &operator<<(raw_ostream &OS, const VPlan &Plan) {
5280 Plan.print(OS);
5281 return OS;
5282}
5283#endif
5284
5285} // end namespace llvm
5286
5287#endif // LLVM_TRANSFORMS_VECTORIZE_VPLAN_H
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
aarch64 promote const
unsigned uint64_t
static MCDisassembler::DecodeStatus addOperand(MCInst &Inst, const MCOperand &Opnd)
Rewrite undef for PHI
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static void print(raw_ostream &Out, object::Archive::Kind Kind, T Val)
This file implements methods to test, set and extract typed bits from packed unsigned integers.
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
#define LLVM_ABI
Definition Compiler.h:215
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
Definition Compiler.h:686
#define LLVM_ABI_FOR_TEST
Definition Compiler.h:220
#define LLVM_PACKED_START
Definition Compiler.h:579
dxil translate DXIL Translate Metadata
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
Hexagon Common GEP
This file defines an InstructionCost class that is used when calculating the cost of an instruction,...
static std::pair< Value *, APInt > getMask(Value *WideMask, unsigned Factor, ElementCount LeafValueEC)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
This file implements a map that provides insertion order iteration.
static Interval intersect(const Interval &I1, const Interval &I2)
This file provides utility analysis objects describing memory locations.
#define T
#define P(N)
static StringRef getName(Value *V)
static bool mayHaveSideEffects(MachineInstr &MI)
SI Fold Operands
Func MI getDebugLoc()))
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
static const BasicSubtargetSubTypeKV * find(StringRef S, ArrayRef< BasicSubtargetSubTypeKV > A)
Find KV in array using binary search.
This file contains the declarations of the entities induced by Vectorization Plans,...
#define VP_CLASSOF_IMPL(VPRecipeID)
Definition VPlan.h:587
static const uint32_t IV[8]
Definition blake3_impl.h:83
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
LLVM Basic Block Representation.
Definition BasicBlock.h:62
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this basic block belongs to.
LLVM_ABI LLVMContext & getContext() const
Get the context in which this basic block lives.
This class represents a function call, abstracting a target machine's calling convention.
This is the base class for all instructions that perform data casts.
Definition InstrTypes.h:512
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
static DebugLoc getUnknown()
Definition DebugLoc.h:153
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:305
Utility class for floating point operations which can have information about relaxed accuracy require...
Definition Operator.h:202
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags fromRaw(unsigned Flags)
unsigned getRaw() const
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
InnerLoopVectorizer vectorizes loops which contain only one basic block to a specified vectorization ...
bool isCast() const
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
LoopVectorizationCostModel - estimates the expected speedups due to vectorization.
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
Metadata node.
Definition Metadata.h:1081
Root of the metadata hierarchy.
Definition Metadata.h:64
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
The RecurrenceDescriptor is used to identify recurrences variables in a loop.
This class represents an assumption made using SCEV expressions which can be checked at run-time.
This class represents an analyzed expression in the program.
This class provides computation of slot numbers for LLVM Assembly writing.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
A SetVector that performs no allocations if smaller than a certain size.
Definition SetVector.h:345
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator erase(const_iterator CI)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::string str() const
Get the contents as an std::string.
Definition StringRef.h:222
This class represents a truncation of integer types.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
Definition Twine.cpp:17
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
void execute(VPTransformState &State) override
Generate the active lane mask phi of the vector loop.
VPActiveLaneMaskPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4070
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPActiveLaneMaskPHIRecipe(VPValue *StartMask, DebugLoc DL)
Definition VPlan.h:4064
~VPActiveLaneMaskPHIRecipe() override=default
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
Definition VPlan.h:4414
RecipeListTy::const_iterator const_iterator
Definition VPlan.h:4442
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
Definition VPlan.h:4489
RecipeListTy::const_reverse_iterator const_reverse_iterator
Definition VPlan.h:4444
RecipeListTy::iterator iterator
Instruction iterators...
Definition VPlan.h:4441
RecipeListTy & getRecipeList()
Returns a reference to the list of recipes.
Definition VPlan.h:4467
iplist< VPRecipeBase > RecipeListTy
Definition VPlan.h:4425
iterator end()
Definition VPlan.h:4451
iterator begin()
Recipe iterator methods.
Definition VPlan.h:4449
RecipeListTy::reverse_iterator reverse_iterator
Definition VPlan.h:4443
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
Definition VPlan.h:4502
const VPBasicBlock * getCFGPredecessor(unsigned Idx) const
Returns the predecessor block at index Idx with the predecessors as per the corresponding plain CFG.
Definition VPlan.cpp:756
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
Definition VPlan.cpp:233
~VPBasicBlock() override
Definition VPlan.h:4435
const_reverse_iterator rbegin() const
Definition VPlan.h:4455
reverse_iterator rend()
Definition VPlan.h:4456
RecipeListTy Recipes
The VPRecipes held in the order of output instructions to generate.
Definition VPlan.h:4429
VPRecipeBase & back()
Definition VPlan.h:4464
const VPRecipeBase & front() const
Definition VPlan.h:4461
const_iterator begin() const
Definition VPlan.h:4450
VPRecipeBase & front()
Definition VPlan.h:4462
const VPRecipeBase & back() const
Definition VPlan.h:4463
void insert(VPRecipeBase *Recipe, iterator InsertPt)
Definition VPlan.h:4480
bool empty() const
Definition VPlan.h:4460
const_iterator end() const
Definition VPlan.h:4452
static bool classof(const VPBlockBase *V)
Method to support type inquiry through isa, cast, and dyn_cast.
Definition VPlan.h:4475
static RecipeListTy VPBasicBlock::* getSublistAccess(VPRecipeBase *)
Returns a pointer to a member of the recipe list.
Definition VPlan.h:4470
reverse_iterator rbegin()
Definition VPlan.h:4454
friend class VPlan
Definition VPlan.h:4415
size_t size() const
Definition VPlan.h:4459
const_reverse_iterator rend() const
Definition VPlan.h:4457
VPBasicBlock(VPBlockTy BlockSC, const Twine &Name="")
Definition VPlan.h:4431
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
Definition VPlan.h:3000
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
Definition VPlan.h:3005
VPBlendRecipe(PHINode *Phi, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags, DebugLoc DL)
The blend operation is a User of the incoming values and of their respective masks,...
Definition VPlan.h:2959
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
Definition VPlan.h:2995
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:3017
VPBlendRecipe * cloneWithOperands(ArrayRef< VPValue * > NewOperands)
Definition VPlan.h:2982
VPBlendRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2980
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
Definition VPlan.h:3011
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
Definition VPlan.h:2991
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
Definition VPlan.h:97
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
Definition VPlan.h:307
VPRegionBlock * getParent()
Definition VPlan.h:195
const VPlan * getPlan() const
Definition VPlan.h:200
void setPlan(VPlan *ParentPlan)
Sets the pointer of the plan containing the block.
Definition VPlan.h:203
VPBlocksTy & getPredecessors()
Definition VPlan.h:231
iterator_range< VPBlockBase ** > predecessors()
Definition VPlan.h:228
LLVM_DUMP_METHOD void dump() const
Dump this VPBlockBase to dbgs().
Definition VPlan.h:383
void setName(const Twine &newName)
Definition VPlan.h:188
size_t getNumSuccessors() const
Definition VPlan.h:245
iterator_range< VPBlockBase ** > successors()
Definition VPlan.h:227
virtual void print(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const =0
Print plain-text dump of this VPBlockBase to O, prefixing all lines with Indent.
bool hasPredecessors() const
Returns true if this block has any predecessors.
Definition VPlan.h:225
void swapSuccessors()
Swap successors of the block. The block must have exactly 2 successors.
Definition VPlan.h:329
void printSuccessors(raw_ostream &O, const Twine &Indent) const
Print the successors of this block to O, prefixing all lines with Indent.
Definition VPlan.cpp:641
SmallVectorImpl< VPBlockBase * > VPBlocksTy
Definition VPlan.h:182
virtual ~VPBlockBase()=default
unsigned getNumber() const
Return the unique number of the block.
Definition VPlan.h:349
void setNumber(unsigned N)
Set the unique number of the block, used for dominator tree.
Definition VPlan.h:352
unsigned getIndexForSuccessor(const VPBlockBase *Succ) const
Returns the index for Succ in the blocks successor list.
Definition VPlan.h:342
size_t getNumPredecessors() const
Definition VPlan.h:246
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
Definition VPlan.h:298
VPBlockBase * getEnclosingBlockWithPredecessors()
Definition VPlan.cpp:225
unsigned getIndexForPredecessor(const VPBlockBase *Pred) const
Returns the index for Pred in the blocks predecessors list.
Definition VPlan.h:335
enum :unsigned char { VPRegionBlockSC, VPBasicBlockSC, VPIRBasicBlockSC } VPBlockTy
An enumeration for keeping track of the concrete subclass of VPBlockBase that are actually instantiat...
Definition VPlan.h:105
bool hasSuccessors() const
Returns true if this block has any successors.
Definition VPlan.h:223
const VPBlocksTy & getPredecessors() const
Definition VPlan.h:230
virtual VPBlockBase * clone()=0
Clone the current block and it's recipes without updating the operands of the cloned recipes,...
virtual InstructionCost cost(ElementCount VF, VPCostContext &Ctx)=0
Return the cost of the block.
VPlan * getPlan()
Definition VPlan.h:199
const VPRegionBlock * getParent() const
Definition VPlan.h:196
const std::string & getName() const
Definition VPlan.h:186
void clearSuccessors()
Remove all the successors of this block.
Definition VPlan.h:317
void setTwoSuccessors(VPBlockBase *IfTrue, VPBlockBase *IfFalse)
Set two given VPBlockBases IfTrue and IfFalse to be the two successors of this VPBlockBase.
Definition VPlan.h:289
VPBlockBase * getSinglePredecessor() const
Definition VPlan.h:241
virtual void execute(VPTransformState *State)=0
The method which generates the output IR that correspond to this VPBlockBase, thereby "executing" the...
const VPBlocksTy & getHierarchicalSuccessors()
Definition VPlan.h:265
void clearPredecessors()
Remove all the predecessor of this block.
Definition VPlan.h:314
friend class VPBlockUtils
Definition VPlan.h:98
unsigned getVPBlockID() const
Definition VPlan.h:193
void printAsOperand(raw_ostream &OS, bool PrintType=false) const
Definition VPlan.h:361
void swapPredecessors()
Swap predecessors of the block.
Definition VPlan.h:321
VPBlocksTy & getSuccessors()
Definition VPlan.h:220
VPBlockBase * getEnclosingBlockWithSuccessors()
An Enclosing Block of a block B is any block containing B, including B itself.
Definition VPlan.cpp:217
void setOneSuccessor(VPBlockBase *Successor)
Set a given VPBlockBase Successor as the single successor of this VPBlockBase.
Definition VPlan.h:278
void setParent(VPRegionBlock *P)
Definition VPlan.h:205
VPBlockBase * getSingleHierarchicalPredecessor()
Definition VPlan.h:271
VPBlockBase * getSingleSuccessor() const
Definition VPlan.h:235
const VPBlocksTy & getSuccessors() const
Definition VPlan.h:219
VPBlockBase(VPBlockTy SC, const std::string &N)
Definition VPlan.h:392
A recipe for generating conditional branches on the bits of a mask.
Definition VPlan.h:3506
VPBranchOnMaskRecipe(VPValue *BlockInMask, DebugLoc DL, const VPIRMetadata &Metadata={})
Definition VPlan.h:3508
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Definition VPlan.h:3529
VPBranchOnMaskRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3513
bool usesScalars(const VPValue *Op) const override
Returns true if the recipe uses scalars of operand Op.
Definition VPlan.h:3537
VPlan-based builder utility similar to IRBuilder.
VPRegionValue * createHeaderMask()
Create the header mask for the region and return it.
Definition VPlan.h:4618
VPRegionValue * getHeaderMask() const
Definition VPlan.h:4614
VPRegionValue * getRegionValue()
Definition VPlan.h:4611
VPCanonicalIVInfo(Type *Ty, DebugLoc DL, VPRegionBlock *Region)
Definition VPlan.h:4608
const VPRegionValue * getRegionValue() const
Definition VPlan.h:4612
bool hasNUW() const
Definition VPlan.h:4626
VPCurrentIterationPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4102
VPCurrentIterationPHIRecipe(VPValue *StartIV, DebugLoc DL)
Definition VPlan.h:4096
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPCurrentIterationPHIRecipe.
Definition VPlan.h:4114
LLVM_ABI_FOR_TEST void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the phi nodes.
Definition VPlan.h:4108
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:4121
~VPCurrentIterationPHIRecipe() override=default
InductionDescriptor::InductionKind getInductionKind() const
Definition VPlan.h:4232
VPValue * getIndex() const
Definition VPlan.h:4229
const FPMathOperator * getFPBinOp() const
Definition VPlan.h:4231
VPDerivedIVRecipe(InductionDescriptor::InductionKind Kind, const FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Definition VPlan.h:4203
VPValue * getStepValue() const
Definition VPlan.h:4230
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:4220
VPDerivedIVRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4213
~VPDerivedIVRecipe() override=default
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:4235
VPValue * getStartValue() const
Definition VPlan.h:4228
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:4039
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPExpandSCEVRecipe.
Definition VPlan.h:4044
VPExpandSCEVRecipe(const SCEV *Expr)
const SCEV * getSCEV() const
Definition VPlan.h:4050
VPExpandSCEVRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4035
~VPExpandSCEVRecipe() override=default
void execute(VPTransformState &State) override
Method for generating code, must not be called as this recipe is abstract.
Definition VPlan.h:3683
bool isVectorToScalar() const
Returns true if this VPExpressionRecipe produces a single scalar.
VPExpressionRecipe(VPWidenCastRecipe *Ext, VPWidenRecipe *Neg, VPReductionRecipe *Red)
Definition VPlan.h:3599
VPExpressionRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3652
SmallVector< VPSingleDefRecipe * > decompose()
Return and insert the recipes of the expression back into the VPlan, directly before the current reci...
~VPExpressionRecipe() override
Definition VPlan.h:3640
ExpressionTypes getExpressionType() const
Returns the expression type of this recipe.
Definition VPlan.h:3675
VPExpressionRecipe(VPWidenCastRecipe *Ext, VPReductionRecipe *Red)
Definition VPlan.h:3597
bool mayHaveSideEffects() const
Returns true if this expression contains recipes that may have side effects.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Compute the cost of this recipe either using a recipe's specialized implementation or using the legac...
bool mayReadOrWriteMemory() const
Returns true if this expression contains recipes that may read from or write to memory.
VPExpressionRecipe(VPWidenCastRecipe *Ext0, VPWidenCastRecipe *Ext1, VPWidenRecipe *Mul, VPReductionRecipe *Red)
Definition VPlan.h:3615
VPExpressionRecipe(ExpressionTypes ExpressionType, ArrayRef< VPSingleDefRecipe * > ExpressionRecipes)
Construct a new VPExpressionRecipe by internalizing recipes in ExpressionRecipes.
VPExpressionRecipe(VPWidenCastRecipe *Ext0, VPWidenCastRecipe *Ext1, VPWidenRecipe *Mul, VPWidenRecipe *Neg, VPReductionRecipe *Red)
Definition VPlan.h:3619
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
unsigned getVFScaleFactor() const
Definition VPlan.h:3677
VPExpressionRecipe(VPWidenRecipe *Mul, VPReductionRecipe *Red)
Definition VPlan.h:3613
A pure virtual base class for all recipes modeling header phis, including phis for first order recurr...
Definition VPlan.h:2448
VPHeaderPHIRecipe(VPRecipeTy VPRecipeID, Instruction *UnderlyingInstr, VPValue *Start, Type *ResultTy, DebugLoc DL)
Definition VPlan.h:2455
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this header phi recipe.
VPHeaderPHIRecipe(VPRecipeTy VPRecipeID, Instruction *UnderlyingInstr, VPValue *Start, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:2450
const VPRecipeBase * getAsRecipe() const override
Return a VPRecipeBase* to the current object.
Definition VPlan.h:2459
void addBackedgeValue(VPValue *V)
Add V as the incoming value from the loop backedge.
Definition VPlan.h:2501
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:2472
static bool classof(const VPValue *V)
Definition VPlan.h:2469
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override=0
Print the recipe.
virtual VPValue * getBackedgeValue()
Returns the incoming value from the loop backedge.
Definition VPlan.h:2495
void setBackedgeValue(VPValue *V)
Update the incoming value from the loop backedge.
Definition VPlan.h:2498
VPValue * getStartValue()
Returns the start value of the phi, if one is set.
Definition VPlan.h:2484
void setStartValue(VPValue *V)
Update the start value of the recipe.
Definition VPlan.h:2492
static bool classof(const VPRecipeBase *R)
Method to support type inquiry through isa, cast, and dyn_cast.
Definition VPlan.h:2465
VPValue * getStartValue() const
Definition VPlan.h:2487
void execute(VPTransformState &State) override=0
Generate the phi nodes.
~VPHeaderPHIRecipe() override=default
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
Definition VPlan.h:2170
void execute(VPTransformState &State) override
Produce a vectorized histogram operation.
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:2203
VPHistogramRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2183
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPHistogramRecipe.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPValue * getMask() const
Return the mask operand if one was provided, or a null pointer if all lanes should be executed uncond...
Definition VPlan.h:2198
VP_CLASSOF_IMPL(VPRecipeBase::VPHistogramSC)
~VPHistogramRecipe() override=default
VPHistogramRecipe(unsigned Opcode, ArrayRef< VPValue * > Operands, const VPIRMetadata &Metadata={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:2175
A special type of VPBasicBlock that wraps an existing IR basic block.
Definition VPlan.h:4567
void execute(VPTransformState *State) override
The method which generates the output IR instructions that correspond to this VPBasicBlock,...
Definition VPlan.cpp:453
BasicBlock * getIRBasicBlock() const
Definition VPlan.h:4591
static bool classof(const VPBlockBase *V)
Definition VPlan.h:4581
~VPIRBasicBlock() override=default
friend class VPlan
Definition VPlan.h:4568
VPIRBasicBlock * clone() override
Clone the current block and it's recipes, without updating the operands of the cloned recipes.
Definition VPlan.cpp:478
Class to record and manage LLVM IR flags.
Definition VPlan.h:696
WrapFlagsTy getNoWrapFlagsOrNone() const
Definition VPlan.h:1031
FastMathFlagsTy FMFs
Definition VPlan.h:785
ReductionFlagsTy ReductionFlags
Definition VPlan.h:787
VPIRFlags(RecurKind Kind, bool IsOrdered, bool IsInLoop, FastMathFlags FMFs)
Definition VPlan.h:878
LLVM_ABI_FOR_TEST bool flagsValidForOpcode(unsigned Opcode) const
Returns true if the set flags are valid for Opcode.
VPIRFlags(DisjointFlagsTy DisjointFlags)
Definition VPlan.h:858
VPIRFlags(WrapFlagsTy WrapFlags)
Definition VPlan.h:844
WrapFlagsTy WrapFlags
Definition VPlan.h:779
void printFlags(raw_ostream &O) const
VPIRFlags(CmpInst::Predicate Pred, FastMathFlags FMFs)
Definition VPlan.h:837
bool hasFastMathFlags() const
Returns true if the recipe has fast-math flags.
Definition VPlan.h:1002
static LLVM_ABI_FOR_TEST VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
bool isReductionOrdered() const
Definition VPlan.h:1057
TruncFlagsTy TruncFlags
Definition VPlan.h:780
CmpInst::Predicate getPredicate() const
Definition VPlan.h:974
WrapFlagsTy getNoWrapFlags() const
Definition VPlan.h:1041
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
uint8_t AllFlags[2]
Definition VPlan.h:788
void transferFlags(VPIRFlags &Other)
Definition VPlan.h:883
ExactFlagsTy ExactFlags
Definition VPlan.h:782
bool hasNoSignedWrap() const
Definition VPlan.h:1020
void intersectFlags(const VPIRFlags &Other)
Only keep flags also present in Other.
bool isDisjoint() const
Definition VPlan.h:1045
VPIRFlags(TruncFlagsTy TruncFlags)
Definition VPlan.h:849
VPIRFlags(FastMathFlags FMFs)
Definition VPlan.h:854
VPIRFlags(NonNegFlagsTy NonNegFlags)
Definition VPlan.h:863
VPIRFlags(CmpInst::Predicate Pred)
Definition VPlan.h:832
uint8_t GEPFlagsStorage
Definition VPlan.h:783
VPIRFlags(ExactFlagsTy ExactFlags)
Definition VPlan.h:868
GEPNoWrapFlags getGEPNoWrapFlags() const
Definition VPlan.h:992
bool hasPredicate() const
Returns true if the recipe has a comparison predicate.
Definition VPlan.h:997
LLVM_ABI_FOR_TEST bool hasRequiredFlagsForOpcode(unsigned Opcode, Type *ResultTy) const
Returns true if Opcode with scalar result type ResultTy has its required flags set.
DisjointFlagsTy DisjointFlags
Definition VPlan.h:781
void setPredicate(CmpInst::Predicate Pred)
Definition VPlan.h:982
bool hasNoUnsignedWrap() const
Definition VPlan.h:1009
FCmpFlagsTy FCmpFlags
Definition VPlan.h:786
NonNegFlagsTy NonNegFlags
Definition VPlan.h:784
bool isReductionInLoop() const
Definition VPlan.h:1063
void dropPoisonGeneratingFlags()
Drop all poison-generating flags.
Definition VPlan.h:894
void applyFlags(Instruction &I) const
Apply the IR flags to I.
Definition VPlan.h:931
VPIRFlags(GEPNoWrapFlags GEPFlags)
Definition VPlan.h:873
uint8_t CmpPredStorage
Definition VPlan.h:778
RecurKind getRecurKind() const
Definition VPlan.h:1051
VPIRFlags(Instruction &I)
Definition VPlan.h:794
Instruction & getInstruction() const
Definition VPlan.h:1753
bool usesFirstPartOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first part of operand Op.
Definition VPlan.h:1761
~VPIRInstruction() override=default
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
VPIRInstruction * clone() override
Clone the current recipe.
Definition VPlan.h:1740
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first lane of operand Op.
Definition VPlan.h:1767
static LLVM_ABI_FOR_TEST VPIRInstruction * create(Instruction &I)
Create a new VPIRPhi for \I , if it is a PHINode, otherwise create a VPIRInstruction.
LLVM_ABI_FOR_TEST InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPIRInstruction.
bool usesScalars(const VPValue *Op) const override
Returns true if the VPUser uses scalars of operand Op.
Definition VPlan.h:1755
VPIRInstruction(Instruction &I)
VPIRInstruction::create() should be used to create VPIRInstructions, as subclasses may need to be cre...
Definition VPlan.h:1728
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Helper to manage IR metadata for recipes.
Definition VPlan.h:1181
MDNode * getBranchWeights() const
Returns the branch weights recorded for this terminator, preferring real profile data over an estimat...
Definition VPlan.h:1270
VPIRMetadata & operator=(const VPIRMetadata &Other)=default
MDNode * getMetadata(unsigned Kind) const
Get metadata of kind Kind. Returns nullptr if not found.
Definition VPlan.h:1251
VPIRMetadata(Instruction &I)
Adds metatadata that can be preserved from the original instruction I.
Definition VPlan.h:1210
VPIRMetadata(const VPIRMetadata &Other)=default
Copy constructor for cloning.
VPIRMetadata()=default
void setEstimatedBranchWeights(MDNode *Node)
Set estimated branch weights to Node.
Definition VPlan.h:1281
void applyMetadata(Instruction &I) const
Add all metadata to I.
void setMetadata(unsigned Kind, MDNode *Node)
Set metadata with kind Kind to Node.
Definition VPlan.h:1230
void eraseMetadata(unsigned Kind)
Remove the metadata of kind Kind, if present.
Definition VPlan.h:1242
bool hasEstimatedBranchWeights() const
Returns true if the weights returned by getBranchWeights are estimated.
Definition VPlan.h:1276
This is a concrete Recipe that models a single VPlan-level instruction.
Definition VPlan.h:1300
VPInstruction(unsigned Opcode, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
Definition VPlan.h:1541
iterator_range< operand_iterator > operandsWithoutMask()
Returns an iterator range over the operands excluding the mask operand if present.
Definition VPlan.h:1563
VPInstruction * clone() override
Clone the current recipe.
Definition VPlan.h:1472
@ ExtractLastActive
Extracts the last active lane from a set of vectors.
Definition VPlan.h:1409
@ Intrinsic
Calls a scalar intrinsic. The intrinsic ID is the last operand.
Definition VPlan.h:1421
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
Definition VPlan.h:1400
@ ExitingIVValue
Compute the exiting value of a wide induction after vectorization, that is the value of the last lane...
Definition VPlan.h:1413
@ WideIVStep
Scale the first operand (vector step) by the second operand (scalar-step).
Definition VPlan.h:1417
@ ResumeForEpilogue
Explicit user for the resume phi of the canonical induction in the main VPlan, used by the epilogue v...
Definition VPlan.h:1403
@ Unpack
Extracts all lanes from its (non-scalable) vector operand.
Definition VPlan.h:1351
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
Definition VPlan.h:1396
@ BuildVector
Creates a fixed-width vector containing all operands.
Definition VPlan.h:1346
@ BuildStructVector
Given operands of (the same) struct type, creates a struct of fixed- width vectors each containing a ...
Definition VPlan.h:1343
@ CanonicalIVIncrementForPart
Definition VPlan.h:1327
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
Definition VPlan.h:1354
bool hasResult() const
Definition VPlan.h:1506
iterator_range< const_operand_iterator > operandsWithoutMask() const
Definition VPlan.h:1566
void addMask(VPValue *Mask)
Add mask Mask to an unmasked VPInstruction, if it needs masking.
Definition VPlan.h:1546
StringRef getName() const
Returns the symbolic name assigned to the VPInstruction.
Definition VPlan.h:1592
unsigned getOpcode() const
Definition VPlan.h:1485
void setName(StringRef NewName)
Set the symbolic name for the VPInstruction.
Definition VPlan.h:1595
bool usesScalars(const VPValue *Op) const override
Returns true if the recipe only uses scalars of operand Op.
Definition VPlan.h:1577
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
VPValue * getMask() const
Returns the mask for the VPInstruction.
Definition VPlan.h:1557
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
VPInstruction * cloneWithOperands(ArrayRef< VPValue * > NewOperands, Type *ResultTy=nullptr)
Definition VPlan.h:1476
unsigned getNumOperandsForOpcode() const
Return the number of operands determined by the opcode of the VPInstruction, excluding mask.
bool isMasked() const
Returns true if the VPInstruction has a mask operand.
Definition VPlan.h:1531
A common base class for interleaved memory operations.
Definition VPlan.h:3042
virtual unsigned getNumStoreOperands() const =0
Returns the number of stored operands of this interleave group.
VPInterleaveBase(VPRecipeTy SC, const InterleaveGroup< Instruction > *IG, ArrayRef< VPValue * > Operands, ArrayRef< VPValue * > StoredValues, VPValue *Mask, bool NeedsMaskForGaps, const VPIRMetadata &MD, DebugLoc DL)
Definition VPlan.h:3054
bool usesFirstLaneOnly(const VPValue *Op) const override=0
Returns true if the recipe only uses the first lane of operand Op.
bool needsMaskForGaps() const
Return true if the access needs a mask because of the gaps.
Definition VPlan.h:3104
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:3110
static bool classof(const VPUser *U)
Definition VPlan.h:3086
Instruction * getInsertPos() const
Definition VPlan.h:3108
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:3081
const InterleaveGroup< Instruction > * getInterleaveGroup() const
Definition VPlan.h:3106
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3098
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
Definition VPlan.h:3127
VPInterleaveBase * clone() override=0
Clone the current recipe.
VPValue * getAddr() const
Return the address accessed by this recipe.
Definition VPlan.h:3092
bool usesFirstLaneOnly(const VPValue *Op) const override
The recipe only uses the first lane of the address, and EVL operand.
Definition VPlan.h:3207
VPValue * getEVL() const
The VPValue of the explicit vector length.
Definition VPlan.h:3201
~VPInterleaveEVLRecipe() override=default
unsigned getNumStoreOperands() const override
Returns the number of stored operands of this interleave group.
Definition VPlan.h:3214
VPInterleaveEVLRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3194
VPInterleaveEVLRecipe(VPInterleaveRecipe &R, VPValue &EVL, VPValue *Mask)
Definition VPlan.h:3181
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
Definition VPlan.h:3137
unsigned getNumStoreOperands() const override
Returns the number of stored operands of this interleave group.
Definition VPlan.h:3164
~VPInterleaveRecipe() override=default
VPInterleaveRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3147
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3158
VPInterleaveRecipe(const InterleaveGroup< Instruction > *IG, VPValue *Addr, ArrayRef< VPValue * > StoredValues, VPValue *Mask, bool NeedsMaskForGaps, const VPIRMetadata &MD, DebugLoc DL)
Definition VPlan.h:3139
In what follows, the term "input IR" refers to code that is fed into the vectorizer whereas the term ...
A VPRecipeValue defined by a multi-def recipe, stores a pointer to it.
Definition VPlanValue.h:378
Helper type to provide functions to access incoming values and blocks for phi-like recipes.
Definition VPlan.h:1607
virtual const VPRecipeBase * getAsRecipe() const =0
Return a VPRecipeBase* to the current object.
LLVM_ABI_FOR_TEST VPValue * getIncomingValueForBlock(const VPBasicBlock *VPBB) const
Returns the incoming value for VPBB. VPBB must be an incoming block.
VPUser::const_operand_range incoming_values() const
Returns an interator range over the incoming values.
Definition VPlan.h:1637
void addIncoming(VPValue *IncomingV)
Append IncomingV as an incoming value to the phi-like recipe.
Definition VPlan.h:1666
virtual unsigned getNumIncoming() const
Returns the number of incoming values, also number of incoming blocks.
Definition VPlan.h:1632
void removeIncomingValueFor(VPBlockBase *IncomingBlock) const
Removes the incoming value for IncomingBlock, which must be a predecessor.
const VPBasicBlock * getIncomingBlock(unsigned Idx) const
Returns the incoming block with index Idx.
Definition VPlan.h:4558
detail::zippy< llvm::detail::zip_first, VPUser::const_operand_range, const_incoming_blocks_range > incoming_values_and_blocks() const
Returns an iterator range over pairs of incoming values and corresponding incoming blocks.
Definition VPlan.h:1657
VPValue * getIncomingValue(unsigned Idx) const
Returns the incoming VPValue with index Idx.
Definition VPlan.h:1616
virtual ~VPPhiAccessors()=default
void printPhiOperands(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print the recipe.
void setIncomingValueForBlock(const VPBasicBlock *VPBB, VPValue *V) const
Sets the incoming value for VPBB to V.
iterator_range< mapped_iterator< detail::index_iterator, std::function< const VPBasicBlock *(size_t)> > > const_incoming_blocks_range
Definition VPlan.h:1642
const_incoming_blocks_range incoming_blocks() const
Returns an iterator range over the incoming blocks.
Definition VPlan.h:1646
~VPPredInstPHIRecipe() override=default
VPPredInstPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3723
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPPredInstPHIRecipe.
Definition VPlan.h:3734
VPPredInstPHIRecipe(VPValue *PredV, DebugLoc DL)
Construct a VPPredInstPHIRecipe given PredInst whose value needs a phi nodes after merging back from ...
Definition VPlan.h:3718
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
Definition VPlan.h:403
bool mayReadFromMemory() const
Returns true if the recipe may read from memory.
bool mayReadOrWriteMemory() const
Returns true if the recipe may read from or write to memory.
Definition VPlan.h:548
virtual void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const =0
Each concrete VPRecipe prints itself, without printing common information, like debug info or metadat...
VPRegionBlock * getRegion()
Definition VPlan.h:4813
void setDebugLoc(DebugLoc NewDL)
Set the recipe's debug location to NewDL.
Definition VPlan.h:556
bool mayWriteToMemory() const
Returns true if the recipe may write to memory.
VPRecipeTy getVPRecipeID() const
Definition VPlan.h:521
~VPRecipeBase() override=default
VPBasicBlock * getParent()
Definition VPlan.h:475
enum :unsigned char { VPBranchOnMaskSC, VPDerivedIVSC, VPExpandSCEVSC, VPExpressionSC, VPIRInstructionSC, VPInstructionSC, VPInterleaveEVLSC, VPInterleaveSC, VPReductionEVLSC, VPReductionSC, VPReplicateSC, VPScalarIVStepsSC, VPVectorPointerSC, VPVectorEndPointerSC, VPWidenCallSC, VPWidenCanonicalIVSC, VPWidenCastSC, VPWidenGEPSC, VPWidenIntrinsicSC, VPWidenMemIntrinsicSC, VPWidenLoadEVLSC, VPWidenLoadSC, VPWidenStoreEVLSC, VPWidenStoreSC, VPWidenSC, VPBlendSC, VPHistogramSC, VPWidenPHISC, VPPredInstPHISC, VPCurrentIterationPHISC, VPActiveLaneMaskPHISC, VPFirstOrderRecurrencePHISC, VPWidenIntOrFpInductionSC, VPWidenPointerInductionSC, VPReductionPHISC, VPFirstPHISC=VPWidenPHISC, VPFirstHeaderPHISC=VPCurrentIterationPHISC, VPLastHeaderPHISC=VPReductionPHISC, VPLastPHISC=VPReductionPHISC, } VPRecipeTy
An enumeration for keeping track of the concrete subclass of VPRecipeBase that is actually instantiat...
Definition VPlan.h:418
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
Definition VPlan.h:553
virtual void execute(VPTransformState &State)=0
The method which generates the output IR instructions that correspond to this VPRecipe,...
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
static bool classof(const VPDef *D)
Method to support type inquiry through isa, cast, and dyn_cast.
Definition VPlan.h:524
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
virtual VPRecipeBase * clone()=0
Clone the current recipe.
friend class VPBlockUtils
Definition VPlan.h:405
const VPBasicBlock * getParent() const
Definition VPlan.h:476
VPRecipeBase(VPRecipeTy SC, ArrayRef< VPValue * > Operands, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:465
InstructionCost cost(ElementCount VF, VPCostContext &Ctx)
Return the cost of this recipe, taking into account if the cost computation should be skipped and the...
static bool classof(const VPUser *U)
Definition VPlan.h:529
void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
void moveAfter(VPRecipeBase *MovePos)
Unlink this recipe from its current VPBasicBlock and insert it into the VPBasicBlock that MovePos liv...
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
Definition VPlanValue.h:351
VPValue * getEVL() const
The VPValue of the explicit vector length.
Definition VPlan.h:3375
VPReductionEVLRecipe(VPReductionRecipe &R, VPValue &EVL, VPValue *CondOp, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:3353
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3378
VPReductionEVLRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3365
~VPReductionEVLRecipe() override=default
bool isOrdered() const
Returns true, if the phi is part of an ordered reduction.
Definition VPlan.h:2923
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
Definition VPlan.h:2914
VPReductionPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2896
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:2907
~VPReductionPHIRecipe() override=default
bool hasUsesOutsideReductionChain() const
Returns true, if the phi is part of a multi-use reduction.
Definition VPlan.h:2932
VPReductionPHIRecipe(PHINode *Phi, RecurKind Kind, VPValue &Start, VPValue &BackedgeValue, ReductionStyle Style, const VPIRFlags &Flags, bool HasUsesOutsideReductionChain=false)
Create a new VPReductionPHIRecipe for the reduction Phi.
Definition VPlan.h:2877
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
Definition VPlan.h:2926
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:2937
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate the phi/select nodes.
VPReductionPHIRecipe * cloneWithOperands(VPValue *Start, VPValue *BackedgeValue)
Definition VPlan.h:2889
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
Definition VPlan.h:2920
A recipe to represent inloop, ordered or partial reduction operations.
Definition VPlan.h:3230
bool isConditional() const
Return true if the in-loop reduction is conditional.
Definition VPlan.h:3314
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:3283
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:3298
VPValue * getVecOp() const
The VPValue of the vector value to be reduced.
Definition VPlan.h:3327
VPValue * getCondOp() const
The VPValue of the condition for the block.
Definition VPlan.h:3329
RecurKind getRecurrenceKind() const
Return the recurrence kind for the in-loop reduction.
Definition VPlan.h:3310
VPReductionRecipe(RecurKind RdxKind, FastMathFlags FMFs, Instruction *I, VPValue *ChainOp, VPValue *VecOp, VPValue *CondOp, ReductionStyle Style, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:3263
bool isOrdered() const
Return true if the in-loop reduction is ordered.
Definition VPlan.h:3312
VPReductionRecipe(const RecurKind RdxKind, FastMathFlags FMFs, VPValue *ChainOp, VPValue *VecOp, VPValue *CondOp, ReductionStyle Style, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:3269
VPReductionRecipe(VPRecipeTy SC, RecurKind RdxKind, FastMathFlags FMFs, Instruction *I, ArrayRef< VPValue * > Operands, VPValue *CondOp, ReductionStyle Style, DebugLoc DL)
Definition VPlan.h:3239
bool isPartialReduction() const
Returns true if the reduction outputs a vector with a scaled down VF.
Definition VPlan.h:3316
~VPReductionRecipe() override=default
VPValue * getChainOp() const
The VPValue of the scalar Chain being accumulated.
Definition VPlan.h:3325
bool isInLoop() const
Returns true if the reduction is in-loop.
Definition VPlan.h:3320
VPReductionRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3277
static bool classof(const VPUser *U)
Definition VPlan.h:3288
static bool classof(const VPValue *VPV)
Definition VPlan.h:3293
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:3334
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
Definition VPlan.h:4639
const VPBlockBase * getEntry() const
Definition VPlan.h:4683
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
Definition VPlan.h:4715
~VPRegionBlock() override=default
VPRegionValue * createHeaderMask()
Create the header mask for the region and return it.
Definition VPlan.h:4786
VPRegionValue * getUsedHeaderMask() const
Return the header mask if it exists and is used, or null otherwise.
Definition VPlan.h:4779
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Definition VPlan.h:4700
VPBlockBase * getExiting()
Definition VPlan.h:4696
VPBranchOnMaskRecipe * getEntryBranchOnMask()
Definition VPlan.h:4720
const VPRegionValue * getCanonicalIV() const
Definition VPlan.h:4762
SmallVector< VPRegionValue *, 2 > getRegionValues() const
Return the region values of the loop region (canonical IV, header mask) or an empty vector for replic...
Definition VPlan.h:4793
void setEntry(VPBlockBase *EntryBlock)
Set EntryBlock as the entry VPBlockBase of this VPRegionBlock.
Definition VPlan.h:4688
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
Definition VPlan.h:4767
bool hasCanonicalIVNUW() const
Indicates if NUW is set for the canonical IV increment, for loop regions.
Definition VPlan.h:4803
void clearCanonicalIVNUW(VPInstruction *Increment)
Unsets NUW for the canonical IV increment Increment, for loop regions.
Definition VPlan.h:4806
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
Definition VPlan.h:4759
const VPBlockBase * getExiting() const
Definition VPlan.h:4695
VPBlockBase * getEntry()
Definition VPlan.h:4684
VPBasicBlock * getPreheaderVPBB()
Returns the pre-header VPBasicBlock of the loop region.
Definition VPlan.h:4708
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
Definition VPlan.h:4772
friend class VPlan
Definition VPlan.h:4640
static bool classof(const VPBlockBase *V)
Method to support type inquiry through isa, cast, and dyn_cast.
Definition VPlan.h:4679
VPValues are defined by a VPRegionBlock, like the canonical IV.
Definition VPlanValue.h:249
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
Definition VPlan.h:3397
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
Definition VPlan.h:3456
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the recipe is predicated.
Definition VPlan.h:3490
VPReplicateRecipe(Instruction *I, ArrayRef< VPValue * > Operands, bool IsSingleScalar, VPValue *Mask=nullptr, const VPIRFlags &Flags={}, VPIRMetadata Metadata={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:3405
~VPReplicateRecipe() override=default
static Type * computeScalarType(const Instruction *I, ArrayRef< VPValue * > Operands)
Compute the scalar result type for a VPReplicateRecipe wrapping I with Operands (excluding any predic...
VPReplicateRecipe * cloneWithOperands(ArrayRef< VPValue * > NewOperands)
Definition VPlan.h:3429
bool usesScalars(const VPValue *Op) const override
Returns true if the recipe uses scalars of operand Op.
Definition VPlan.h:3471
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
Definition VPlan.h:3484
bool isPredicated() const
Definition VPlan.h:3461
VPReplicateRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3427
bool doesGeneratePerAllLanes() const
Returns true if the recipe produces scalar values for all VF lanes.
Definition VPlan.h:3459
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3464
unsigned getOpcode() const
Definition VPlan.h:3494
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Definition VPlan.h:3478
Instruction::BinaryOps getInductionOpcode() const
Definition VPlan.h:4317
VPValue * getStepValue() const
Definition VPlan.h:4287
void setStartIndex(VPValue *StartIndex)
Set or add the StartIndex operand.
Definition VPlan.h:4300
VPScalarIVStepsRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4269
VPValue * getStartIndex() const
Return the StartIndex, or null if known to be zero, valid only after unrolling.
Definition VPlan.h:4295
VPValue * getVFValue() const
Return the number of scalars to produce per unroll part, used to compute StartIndex during unrolling.
Definition VPlan.h:4291
VPScalarIVStepsRecipe(VPValue *IV, VPValue *Step, VPValue *VF, Instruction::BinaryOps Opcode, FastMathFlags FMFs={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:4260
~VPScalarIVStepsRecipe() override=default
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:4311
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Definition VPlan.h:611
static bool classof(const VPValue *V)
Definition VPlan.h:668
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
Definition VPlan.h:681
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:625
VPSingleDefRecipe(VPRecipeTy SC, ArrayRef< VPValue * > Operands, Value *UV, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:617
const Instruction * getUnderlyingInstr() const
Definition VPlan.h:684
VPSingleDefRecipe(VPRecipeTy SC, ArrayRef< VPValue * > Operands, Type *ResultTy, Value *UV=nullptr, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:621
static bool classof(const VPUser *U)
Definition VPlan.h:673
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
VPSingleDefRecipe(VPRecipeTy SC, ArrayRef< VPValue * > Operands, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:613
VPSingleDefValue(VPSingleDefRecipe *Def, Value *UV=nullptr, Type *Ty=nullptr)
Construct a VPSingleDefValue. Must only be used by VPSingleDefRecipe.
Definition VPlan.cpp:167
This class can be used to assign names to VPValues.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
Definition VPlanValue.h:214
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
Definition VPlanValue.h:398
void printOperands(raw_ostream &O, VPSlotTracker &SlotTracker) const
Print the operands to O.
Definition VPlan.cpp:1510
operand_range operands()
Definition VPlanValue.h:471
void setOperand(unsigned I, VPValue *New)
Definition VPlanValue.h:444
unsigned getNumOperands() const
Definition VPlanValue.h:438
operand_iterator op_end()
Definition VPlanValue.h:469
operand_iterator op_begin()
Definition VPlanValue.h:467
VPValue * getOperand(unsigned N) const
Definition VPlanValue.h:439
VPUser(ArrayRef< VPValue * > Operands)
Definition VPlanValue.h:419
iterator_range< const_operand_iterator > const_operand_range
Definition VPlanValue.h:465
iterator_range< operand_iterator > operand_range
Definition VPlanValue.h:464
void addOperand(VPValue *Operand)
Definition VPlanValue.h:424
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Definition VPlanValue.h:50
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Definition VPlan.cpp:147
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
Definition VPlan.cpp:141
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
Definition VPlan.cpp:128
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
Definition VPlanValue.h:75
bool user_empty() const
Definition VPlanValue.h:161
void setUnderlyingValue(Value *Val)
Definition VPlanValue.h:206
unsigned getNumUsers() const
Definition VPlanValue.h:115
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first lane of operand Op.
Definition VPlan.h:2318
VPValue * getVFValue() const
Definition VPlan.h:2299
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
Type * getSourceElementType() const
Definition VPlan.h:2296
int64_t getStride() const
Definition VPlan.h:2297
VPVectorEndPointerRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2339
VPValue * getOffset() const
Definition VPlan.h:2300
bool usesFirstPartOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first part of operand Op.
Definition VPlan.h:2332
void addOffset(VPValue *Offset)
Append Offset as the offset operand.
Definition VPlan.h:2310
VPVectorEndPointerRecipe(VPValue *Ptr, VPValue *VF, Type *SourceElementTy, int64_t Stride, GEPNoWrapFlags GEPFlags, DebugLoc DL)
Definition VPlan.h:2286
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPVectorPointerRecipe.
Definition VPlan.h:2325
VPValue * getPointer() const
Definition VPlan.h:2298
void materializeOffset(unsigned Part=0)
Adds the offset operand to the recipe.
void addPerPartOffset(VPValue *VFxPart)
Add the per-part offset (VFxPart) used for unrolled parts > 0.
Definition VPlan.h:2380
VPValue * getStride() const
Definition VPlan.h:2373
Type * getSourceElementType() const
Definition VPlan.h:2388
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the VPUser only uses the first lane of operand Op.
Definition VPlan.h:2390
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
bool usesFirstPartOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first part of operand Op.
Definition VPlan.h:2397
VPVectorPointerRecipe(VPValue *Ptr, Type *SourceElementTy, VPValue *Stride, GEPNoWrapFlags GEPFlags, DebugLoc DL)
Definition VPlan.h:2364
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPHeaderPHIRecipe.
Definition VPlan.h:2414
VPVectorPointerRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2404
VPValue * getVFxPart() const
Definition VPlan.h:2375
A recipe for widening Call instructions using library calls.
Definition VPlan.h:2104
VPWidenCallRecipe(Value *UV, Function *Variant, ArrayRef< VPValue * > CallArguments, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL={})
Definition VPlan.h:2111
const_operand_range args() const
Definition VPlan.h:2152
VPWidenCallRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2130
operand_range args()
Definition VPlan.h:2151
Function * getCalledScalarFunction() const
Definition VPlan.h:2147
~VPWidenCallRecipe() override=default
~VPWidenCanonicalIVRecipe() override=default
VPValue * getStepValue() const
Definition VPlan.h:4173
void addPerPartStep(VPValue *Step)
Add the per-part step (VF * Part) used for unrolled parts.
Definition VPlan.h:4178
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenCanonicalIVPHIRecipe.
Definition VPlan.h:4162
VPRegionValue * getCanonicalIV() const
Return the canonical IV being widened.
Definition VPlan.h:4169
VPWidenCanonicalIVRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:4147
VPWidenCanonicalIVRecipe(VPRegionValue *CanonicalIV, const VPIRFlags::WrapFlagsTy &Flags={})
Definition VPlan.h:4140
void execute(VPTransformState &State) override
The method which generates the output IR instructions that correspond to this VPRecipe,...
Definition VPlan.h:4157
VPWidenCastRecipe is a recipe to create vector cast instructions.
Definition VPlan.h:1885
Instruction::CastOps getOpcode() const
Definition VPlan.h:1921
~VPWidenCastRecipe() override=default
VPWidenCastRecipe(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, CastInst *CI=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1890
VPWidenCastRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:1906
unsigned getOpcode() const
This recipe generates a GEP instruction.
Definition VPlan.h:2248
Type * getSourceElementType() const
Definition VPlan.h:2253
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenGEPRecipe.
Definition VPlan.h:2256
VPWidenGEPRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2239
~VPWidenGEPRecipe() override=default
VPWidenGEPRecipe(Type *SourceElementTy, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, DebugLoc DL=DebugLoc::getUnknown(), GetElementPtrInst *UV=nullptr)
Definition VPlan.h:2222
void execute(VPTransformState &State) override=0
Generate the phi nodes.
ArrayRef< const SCEVPredicate * > getNoWrapPredicates() const
Returns the SCEV predicates associated with this induction.
Definition VPlan.h:2588
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:2600
static bool classof(const VPValue *V)
Definition VPlan.h:2556
VPValue * getBackedgeValue() override
Returns the incoming value from the loop backedge.
Definition VPlan.h:2592
unsigned getNumIncoming() const override
Returns the number of incoming values, also number of incoming blocks.
Definition VPlan.h:2577
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
Definition VPlan.h:2580
VPValue * getStepValue()
Returns the step value of the induction.
Definition VPlan.h:2568
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
Definition VPlan.h:2585
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:2551
VPWidenInductionRecipe(VPRecipeTy Kind, PHINode *IV, VPValue *Start, VPValue *Step, const InductionDescriptor &IndDesc, Type *ResultTy, DebugLoc DL)
Definition VPlan.h:2530
const VPValue * getVFValue() const
Definition VPlan.h:2572
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:2561
const VPValue * getStepValue() const
Definition VPlan.h:2569
VPWidenInductionRecipe(VPRecipeTy Kind, PHINode *IV, VPValue *Start, VPValue *Step, const InductionDescriptor &IndDesc, DebugLoc DL)
Definition VPlan.h:2524
void addUnrolledPartOperands(VPValue *SplatVFStep, VPValue *LastPart)
After unrolling, append the splat-VF step (VF * step) and the value of the induction at the last unro...
Definition VPlan.h:2539
const TruncInst * getTruncInst() const
Definition VPlan.h:2674
void execute(VPTransformState &State) override
Generate the phi nodes.
Definition VPlan.h:2655
~VPWidenIntOrFpInductionRecipe() override=default
VPWidenIntOrFpInductionRecipe(PHINode *IV, VPValue *Start, VPValue *Step, VPValue *VF, const InductionDescriptor &IndDesc, TruncInst *Trunc, const VPIRFlags &Flags, DebugLoc DL)
Definition VPlan.h:2630
VPValue * getSplatVFValue() const
If the recipe has been unrolled, return the VPValue for the induction increment, otherwise return nul...
Definition VPlan.h:2662
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenIntOrFpInductionRecipe.
VPWidenIntOrFpInductionRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2647
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
Definition VPlan.h:2673
VPWidenIntOrFpInductionRecipe(PHINode *IV, VPValue *Start, VPValue *Step, VPValue *VF, const InductionDescriptor &IndDesc, const VPIRFlags &Flags, DebugLoc DL)
Definition VPlan.h:2621
VPValue * getLastUnrolledPartOperand()
Returns the VPValue representing the value of this induction at the last unrolled part,...
Definition VPlan.h:2688
unsigned getNumIncoming() const override
Returns the number of incoming values, also number of incoming blocks.
Definition VPlan.h:2669
bool isCanonical() const
Returns true if the induction is canonical, i.e.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
A recipe for widening vector intrinsics.
Definition VPlan.h:1933
VPWidenIntrinsicRecipe(VPRecipeTy SC, Intrinsic::ID VectorIntrinsicID, ArrayRef< VPValue * > CallArguments, Type *Ty, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1947
VPWidenIntrinsicRecipe(Intrinsic::ID VectorIntrinsicID, ArrayRef< VPValue * > CallArguments, Type *Ty, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1982
Intrinsic::ID getVectorIntrinsicID() const
Return the ID of the intrinsic.
Definition VPlan.h:2036
bool mayReadFromMemory() const
Returns true if the intrinsic may read from memory.
Definition VPlan.h:2042
VPWidenIntrinsicRecipe(CallInst &CI, Intrinsic::ID VectorIntrinsicID, ArrayRef< VPValue * > CallArguments, Type *Ty, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1968
bool mayHaveSideEffects() const
Returns true if the intrinsic may have side-effects.
Definition VPlan.h:2048
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:2018
static bool classof(const VPValue *V)
Definition VPlan.h:2013
VPWidenIntrinsicRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:1993
bool mayWriteToMemory() const
Returns true if the intrinsic may write to memory.
Definition VPlan.h:2045
~VPWidenIntrinsicRecipe() override=default
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:2003
static bool classof(const VPUser *U)
Definition VPlan.h:2008
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
void execute(VPTransformState &State) override
Produce a widened version of the vector memory intrinsic.
~VPWidenMemIntrinsicRecipe() override=default
VPWidenMemIntrinsicRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2081
VPWidenMemIntrinsicRecipe(Intrinsic::ID VectorIntrinsicID, ArrayRef< VPValue * > CallArguments, Type *Ty, Align Alignment, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:2066
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this vector memory intrinsic.
A common mixin class for widening memory operations.
Definition VPlan.h:3750
bool IsMasked
Whether the memory access is masked.
Definition VPlan.h:3761
bool isConsecutive() const
Return whether the loaded-from / stored-to addresses are consecutive.
Definition VPlan.h:3786
virtual ~VPWidenMemoryRecipe()=default
Instruction & Ingredient
Definition VPlan.h:3752
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const
Return the cost of this VPWidenMemoryRecipe.
Instruction & getIngredient() const
Definition VPlan.h:3808
bool Consecutive
Whether the accessed addresses are consecutive.
Definition VPlan.h:3758
virtual const VPRecipeBase * getAsRecipe() const =0
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3796
Align Alignment
Alignment information for this memory access.
Definition VPlan.h:3755
VPWidenMemoryRecipe(Instruction &I, bool Consecutive, const VPIRMetadata &Metadata)
Definition VPlan.h:3773
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
bool isMasked() const
Returns true if the recipe is masked.
Definition VPlan.h:3792
void setMask(VPValue *Mask)
Definition VPlan.h:3763
Align getAlign() const
Returns the alignment of the memory access.
Definition VPlan.h:3803
VPValue * getAddr() const
Return the address accessed by this recipe.
Definition VPlan.h:3789
A recipe for widened phis.
Definition VPlan.h:2750
const VPRecipeBase * getAsRecipe() const override
Return a VPRecipeBase* to the current object.
Definition VPlan.h:2795
unsigned getOpcode() const
This recipe generates a PHI.
Definition VPlan.h:2777
VPWidenPHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2770
~VPWidenPHIRecipe() override=default
VPWidenPHIRecipe(ArrayRef< VPValue * > IncomingValues, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new VPWidenPHIRecipe with incoming values IncomingValues, debug location DL and Name.
Definition VPlan.h:2757
VPWidenPointerInductionRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2715
~VPWidenPointerInductionRecipe() override=default
bool onlyScalarsGenerated(bool IsScalable)
Returns true if only scalar values will be generated.
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
void execute(VPTransformState &State) override
Generate vector values for the pointer induction.
Definition VPlan.h:2724
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenPointerInductionRecipe.
VPWidenPointerInductionRecipe(PHINode *Phi, VPValue *Start, VPValue *Step, VPValue *NumUnrolledElems, const InductionDescriptor &IndDesc, DebugLoc DL)
Create a new VPWidenPointerInductionRecipe for Phi with start value Start and the number of elements ...
Definition VPlan.h:2705
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
Definition VPlan.h:1818
VPWidenRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:1844
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:1873
VPWidenRecipe(Instruction &I, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL={})
Definition VPlan.h:1822
VPWidenRecipe(unsigned Opcode, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags={}, const VPIRMetadata &Metadata={}, DebugLoc DL={})
Definition VPlan.h:1829
~VPWidenRecipe() override=default
VPWidenRecipe * cloneWithOperands(ArrayRef< VPValue * > NewOperands)
Definition VPlan.h:1846
unsigned getOpcode() const
Definition VPlan.h:1863
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
Definition VPlan.h:4826
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
Definition VPlan.h:5169
LLVM_ABI_FOR_TEST void printDOT(raw_ostream &O) const
Print this VPlan in DOT format to O.
Definition VPlan.cpp:1160
friend class VPSlotTracker
Definition VPlan.h:4828
std::string getName() const
Return a string with the name of the plan and the applicable VFs and UFs.
Definition VPlan.cpp:1136
bool hasVF(ElementCount VF) const
Definition VPlan.h:5062
ElementCount getSingleVF() const
Returns the single VF of the plan, asserting that the plan has exactly one VF.
Definition VPlan.h:5075
const DataLayout & getDataLayout() const
Definition VPlan.h:5040
LLVMContext & getContext() const
Definition VPlan.h:5036
VPBasicBlock * getEntry()
Definition VPlan.h:4922
Type * getIndexType() const
The type of the canonical induction variable of the vector loop.
Definition VPlan.h:5275
void setName(const Twine &newName)
Definition VPlan.h:5108
bool hasScalableVF() const
Definition VPlan.h:5063
VPValue * getTripCount() const
The trip count of the original loop.
Definition VPlan.h:4994
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
Definition VPlan.h:5015
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
Definition VPlan.h:5069
LLVM_ABI_FOR_TEST ~VPlan()
Definition VPlan.cpp:888
VPIRValue * getOrAddLiveIn(VPIRValue *V)
Definition VPlan.h:5126
bool isExitBlock(VPBlockBase *VPBB)
Returns true if VPBB is an exit block.
Definition VPlan.cpp:907
const VPBasicBlock * getEntry() const
Definition VPlan.h:4923
friend class VPlanPrinter
Definition VPlan.h:4827
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
Definition VPlan.h:5135
VPIRValue * getConstantInt(const APInt &Val)
Return a VPIRValue wrapping a ConstantInt with the given APInt value.
Definition VPlan.h:5158
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
Definition VPlan.h:5034
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
Definition VPlan.h:5141
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
Definition VPlan.h:5222
VPIRBasicBlock * createEmptyVPIRBasicBlock(BasicBlock *IRBB)
Create a VPIRBasicBlock wrapping IRBB, but do not create VPIRInstructions wrapping the instructions i...
Definition VPlan.cpp:1300
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
Definition VPlan.h:5172
bool hasUF(unsigned UF) const
Definition VPlan.h:5087
VPIRValue * getPoison(Type *Ty)
Return a VPIRValue wrapping a poison value of type Ty.
Definition VPlan.h:5163
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
Definition VPlan.h:4988
VPlan(BasicBlock *ScalarHeaderBB, Type *IdxTy)
Construct a VPlan with a new VPBasicBlock as entry, a VPIRBasicBlock wrapping ScalarHeaderBB and vect...
Definition VPlan.h:4903
VPSymbolicValue & getVectorTripCount()
The vector trip count.
Definition VPlan.h:5024
VPValue * getBackedgeTakenCount() const
Definition VPlan.h:5021
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
Definition VPlan.h:5112
VPRegionBlock * createLoopRegion(Type *CanIVTy, DebugLoc DL, const std::string &Name="", VPBlockBase *Entry=nullptr, VPBlockBase *Exiting=nullptr)
Create a new loop region with a canonical IV using CanIVTy and DL.
Definition VPlan.h:5208
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
Definition VPlan.h:5138
void setVF(ElementCount VF)
Definition VPlan.h:5050
unsigned getMaxBlockNumber() const
Definition VPlan.h:5242
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
Definition VPlan.h:5103
Function * getIRFunction() const
Definition VPlan.h:5044
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
Definition VPlan.cpp:1042
bool hasEarlyExit() const
Returns true if the VPlan is based on a loop with an early exit.
Definition VPlan.h:5245
InstructionCost cost(ElementCount VF, VPCostContext &Ctx)
Return the cost of this plan.
Definition VPlan.cpp:1024
LLVM_ABI_FOR_TEST bool isOuterLoop() const
Returns true if this VPlan is for an outer loop, i.e., its vector loop region contains a nested loop ...
Definition VPlan.cpp:1066
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
Definition VPlan.h:5090
VPIRValue * getConstantInt(unsigned BitWidth, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given bitwidth and value.
Definition VPlan.h:5152
const VPBasicBlock * getMiddleBlock() const
Definition VPlan.h:4973
void setTripCount(VPValue *NewTripCount)
Set the trip count assuming it is currently null; if it is not - use resetTripCount().
Definition VPlan.h:5001
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
Definition VPlan.h:5008
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
Definition VPlan.h:4964
void setEntry(VPBasicBlock *VPBB)
Definition VPlan.h:4911
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
Definition VPlan.h:5195
LLVM_ABI_FOR_TEST VPIRBasicBlock * createVPIRBasicBlock(BasicBlock *IRBB)
Create a VPIRBasicBlock from IRBB containing VPIRInstructions for all instructions in IRBB,...
Definition VPlan.cpp:1308
void removeVF(ElementCount VF)
Remove VF from the plan.
Definition VPlan.h:5057
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
Definition VPlan.h:5132
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
Definition VPlan.h:4927
bool requiresScalarEpilogue() const
Returns true if the plan requires a scalar epilogue after the vector loop.
Definition VPlan.h:4950
LLVM_DUMP_METHOD void dump() const
Dump the plan to stderr (for debugging).
Definition VPlan.cpp:1166
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
Definition VPlan.h:5031
bool hasScalarVFOnly() const
Definition VPlan.h:5080
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
Definition VPlan.h:4978
void execute(VPTransformState *State)
Generate the IR code for this VPlan.
Definition VPlan.cpp:917
LLVM_ABI_FOR_TEST void print(raw_ostream &O) const
Print this VPlan to O.
Definition VPlan.cpp:1119
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
Definition VPlan.h:4943
void addVF(ElementCount VF)
Definition VPlan.h:5048
VPIRBasicBlock * getScalarHeader() const
Return the VPIRBasicBlock wrapping the header of the scalar loop.
Definition VPlan.h:4984
void printLiveIns(raw_ostream &O) const
Print the live-ins of this VPlan to O.
Definition VPlan.cpp:1075
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
Definition VPlan.h:5027
void setUF(unsigned UF)
Definition VPlan.h:5095
const VPSymbolicValue & getVF() const
Definition VPlan.h:5028
bool hasScalarTail() const
Returns true if the scalar tail may execute after the vector loop, i.e.
Definition VPlan.h:5268
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
Definition VPlan.cpp:1207
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
Definition VPlan.h:5146
LLVM Value Representation.
Definition Value.h:75
Increasing range of size_t indices.
Definition STLExtras.h:2523
typename base_list_type::const_reverse_iterator const_reverse_iterator
Definition ilist.h:124
typename base_list_type::reverse_iterator reverse_iterator
Definition ilist.h:123
typename base_list_type::const_iterator const_iterator
Definition ilist.h:122
An intrusive list with ownership and callbacks specified/controlled by ilist_traits,...
Definition ilist.h:328
A range adaptor for a pair of iterators.
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
This file defines classes to implement an intrusive doubly linked list class (i.e.
This file defines the ilist_node class template, which is a convenient base class for creating classe...
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
Definition DwarfDebug.h:190
CastInfo helper for casting from VPRecipeBase to a mixin class that is not part of the VPRecipeBase c...
Definition VPlan.h:4330
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
@ Offset
Definition DWP.cpp:577
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
Definition STLExtras.h:846
LLVM_PACKED_END
Definition VPlan.h:1109
auto cast_if_present(const Y &Val)
cast_if_present<X> - Functionally identical to cast, except that a null value is accepted.
Definition Casting.h:683
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1781
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
Definition STLExtras.h:856
ReductionStyle getReductionStyle(bool InLoop, bool Ordered, unsigned ScaleFactor)
Definition VPlan.h:2850
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
Type * toScalarizedTy(Type *Ty)
A helper for converting vectorized types to scalarized (non-vector) types.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
VPBuilderBase<> VPBuilder
Definition VPlan.h:67
LLVM_ABI void getMetadataToPropagate(Instruction *Inst, SmallVectorImpl< std::pair< unsigned, MDNode * > > &Metadata)
Add metadata from Inst to Metadata, if it can be preserved after vectorization.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
auto cast_or_null(const Y &Val)
Definition Casting.h:714
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
Definition STLExtras.h:366
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
UncountableExitStyle
Different methods of handling early exits.
Definition VPlan.h:83
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
Definition VPlan.h:92
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool isPointerTy(const Type *T)
Definition SPIRVUtils.h:383
LLVM_ABI Type * computeScalarTypeForInstruction(unsigned Opcode, ArrayRef< VPValue * > Operands)
Compute the scalar result type for an IR Opcode given Operands.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
@ Other
Any other memory.
Definition ModRef.h:68
RecurKind
These are the kinds of recurrences that we support.
@ Mul
Product of integers.
@ Add
Sum of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ FAdd
Sum of floats.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
Definition STLExtras.h:2028
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned BitWidth
auto sum_of(R &&Range, E Init=E{0})
Returns the sum of all values in Range with Init initial value.
Definition STLExtras.h:1733
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
Definition Sequence.h:341
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
std::variant< RdxOrdered, RdxInLoop, RdxUnordered > ReductionStyle
Definition VPlan.h:2848
@ Increment
Incrementally increasing token ID.
Definition AllocToken.h:26
std::unique_ptr< VPlan > VPlanPtr
Definition VPlan.h:78
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
static Bitfield::Type get(StorageType Packed)
Unpacks the field from the Packed value.
Definition Bitfields.h:207
static void set(StorageType &Packed, typename Bitfield::Type Value)
Sets the typed value in the provided Packed value.
Definition Bitfields.h:223
This struct provides a method for customizing the way a cast is performed.
Definition Casting.h:476
Provides a cast trait that strips const from types to make it easier to implement a const-version of ...
Definition Casting.h:388
This cast trait just provides the default implementation of doCastIfPossible to make CastInfo special...
Definition Casting.h:309
Provides a cast trait that uses a defined pointer to pointer cast as a base for reference-to-referenc...
Definition Casting.h:423
This reduction is in-loop.
Definition VPlan.h:2842
Possible variants of a reduction.
Definition VPlan.h:2840
This reduction is unordered with the partial result scaled down by some factor.
Definition VPlan.h:2845
unsigned VFScaleFactor
Definition VPlan.h:2846
A MapVector that performs no allocations if smaller than a certain size.
Definition MapVector.h:342
Default inserter for VPBuilderBase, inserting R at It in VPBB.
An overlay on VPConstant for VPValues that wrap a ConstantInt.
Definition VPlanValue.h:307
Struct to hold various analysis needed for cost computations.
const BlockFrequency Freq
Definition VPlan.h:1170
VPExecutionFrequency(BlockFrequency Freq, bool IsEstimated)
Definition VPlan.h:1173
void execute(VPTransformState &State) override
Generate the phi nodes.
VPFirstOrderRecurrencePHIRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:2811
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this first-order recurrence phi recipe.
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:2823
void printRecipe(raw_ostream &O, const Twine &Indent, VPSlotTracker &SlotTracker) const override
Print the recipe.
VPFirstOrderRecurrencePHIRecipe(PHINode *Phi, VPValue &Start, VPValue &BackedgeValue)
Definition VPlan.h:2802
DisjointFlagsTy(bool IsDisjoint)
Definition VPlan.h:729
NonNegFlagsTy(bool IsNonNeg)
Definition VPlan.h:734
TruncFlagsTy(bool HasNUW, bool HasNSW)
Definition VPlan.h:724
WrapFlagsTy(bool HasNUW, bool HasNSW)
Definition VPlan.h:716
An overlay for VPIRInstructions wrapping PHI nodes enabling convenient use cast/dyn_cast/isa and exec...
Definition VPlan.h:1786
VPIRPhi(PHINode &PN)
Definition VPlan.h:1787
static bool classof(const VPRecipeBase *U)
Definition VPlan.h:1789
static bool classof(const VPUser *U)
Definition VPlan.h:1794
PHINode & getIRPhi() const
Definition VPlan.h:1799
const VPRecipeBase * getAsRecipe() const override
Return a VPRecipeBase* to the current object.
Definition VPlan.h:1810
A VPValue representing a live-in from the input IR or a constant.
Definition VPlanValue.h:276
static bool classof(const VPUser *U)
Definition VPlan.h:1686
VPPhi * clone() override
Clone the current recipe.
Definition VPlan.h:1701
const VPRecipeBase * getAsRecipe() const override
Return a VPRecipeBase* to the current object.
Definition VPlan.h:1716
static bool classof(const VPSingleDefRecipe *SDR)
Definition VPlan.h:1696
static bool classof(const VPValue *V)
Definition VPlan.h:1691
VPPhi(ArrayRef< VPValue * > Operands, const VPIRFlags &Flags, DebugLoc DL, const Twine &Name="", Type *ResultTy=nullptr)
Definition VPlan.h:1681
A pure-virtual common base class for recipes defining a single VPValue and using IR flags.
Definition VPlan.h:1113
VPRecipeWithIRFlags(VPRecipeTy SC, ArrayRef< VPValue * > Operands, const VPIRFlags &Flags, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1114
static bool classof(const VPSingleDefRecipe *R)
Definition VPlan.h:1155
static bool classof(const VPRecipeBase *R)
Definition VPlan.h:1125
InstructionCost getCostForRecipeWithOpcode(unsigned Opcode, ElementCount VF, VPCostContext &Ctx) const
Compute the cost for this recipe for VF, using Opcode and Ctx.
static bool classof(const VPValue *V)
Definition VPlan.h:1148
VPRecipeWithIRFlags(VPRecipeTy SC, ArrayRef< VPValue * > Operands, Type *ResultTy, const VPIRFlags &Flags, DebugLoc DL=DebugLoc::getUnknown())
Definition VPlan.h:1119
void execute(VPTransformState &State) override=0
The method which generates the output IR instructions that correspond to this VPRecipe,...
VPRecipeWithIRFlags * clone() override=0
Clone the current recipe.
static bool classof(const VPUser *U)
Definition VPlan.h:1143
VPTransformState holds information passed down when "executing" a VPlan, needed for generating the ou...
A recipe for widening load operations with vector-predication intrinsics, using the address to load f...
Definition VPlan.h:3867
VPWidenLoadEVLRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3877
unsigned getOpcode() const
Returns the opcode of the widened load.
Definition VPlan.h:3884
VPValue * getEVL() const
Return the EVL operand.
Definition VPlan.h:3887
VPWidenLoadEVLRecipe(VPWidenLoadRecipe &L, VPValue *Addr, VPValue &EVL, VPValue *Mask)
Definition VPlan.h:3868
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3897
A recipe for widening load operations, using the address to load from and an optional mask.
Definition VPlan.h:3814
VPWidenLoadRecipe(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Definition VPlan.h:3815
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3843
unsigned getOpcode() const
Returns the opcode of the widened load.
Definition VPlan.h:3831
VPWidenLoadRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3823
VP_CLASSOF_IMPL(VPRecipeBase::VPWidenLoadSC)
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenLoadRecipe.
Definition VPlan.h:3837
A recipe for widening store operations with vector-predication intrinsics, using the value to store,...
Definition VPlan.h:3973
VPValue * getStoredValue() const
Return the address accessed by this recipe.
Definition VPlan.h:3989
VPWidenStoreEVLRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3982
VPWidenStoreEVLRecipe(VPWidenStoreRecipe &S, VPValue *Addr, VPValue *StoredVal, VPValue &EVL, VPValue *Mask)
Definition VPlan.h:3974
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:4002
VPValue * getEVL() const
Return the EVL operand.
Definition VPlan.h:3992
A recipe for widening store operations, using the stored value, the address to store to and an option...
Definition VPlan.h:3919
VPWidenStoreRecipe(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Definition VPlan.h:3920
VP_CLASSOF_IMPL(VPRecipeBase::VPWidenStoreSC)
VPValue * getStoredValue() const
Return the value stored by this recipe.
Definition VPlan.h:3937
VPWidenStoreRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:3928
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenStoreRecipe.
Definition VPlan.h:3943
bool usesFirstLaneOnly(const VPValue *Op) const override
Returns true if the recipe only uses the first lane of operand Op.
Definition VPlan.h:3949
static VPMixin * castFailed()
Definition VPlan.h:4348
static bool isPossible(VPRecipeBase *R)
Used by isa.
Definition VPlan.h:4339
static VPMixin * doCast(VPRecipeBase *R)
Used by cast.
Definition VPlan.h:4342