LLVM 24.0.0git
VPlanTransforms.cpp
Go to the documentation of this file.
1//===-- VPlanTransforms.cpp - Utility VPlan to VPlan transforms -----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file implements a set of utility VPlan to VPlan transformations.
11///
12//===----------------------------------------------------------------------===//
13
14#include "VPlanTransforms.h"
15#include "VPRecipeBuilder.h"
16#include "VPlan.h"
17#include "VPlanAnalysis.h"
18#include "VPlanCFG.h"
19#include "VPlanDominatorTree.h"
20#include "VPlanHelpers.h"
21#include "VPlanPatternMatch.h"
22#include "VPlanUtils.h"
23#include "llvm/ADT/APInt.h"
25#include "llvm/ADT/STLExtras.h"
26#include "llvm/ADT/SetVector.h"
28#include "llvm/ADT/TypeSwitch.h"
30#include "llvm/Analysis/Loads.h"
36#include "llvm/IR/Intrinsics.h"
37#include "llvm/IR/MDBuilder.h"
38#include "llvm/IR/Metadata.h"
44
45using namespace llvm;
46using namespace LoopVectorizationUtils;
47using namespace VPlanPatternMatch;
48using namespace SCEVPatternMatch;
49
50/// Returns the metadata attached to \p R, or an empty set for a recipe that
51/// does not carry any.
53 if (auto *MD = dyn_cast<VPIRMetadata>(R))
54 return *MD;
55 return {};
56}
57
58// TODO: Remove this once the partial reduction intrinsics are no worse than
59// normal vector operations.
61 "use-partial-reductions-by-default", cl::init(false), cl::Hidden,
62 cl::desc("Use partial reduction intrinsics for "
63 "all supported unordered reductions."));
64
67 Loop *OuterLoop) {
68
69 // Returns true if the access of \p AccessTy at \p Addr can be widened to a
70 // consecutive vector access.
71 auto IsConsecutiveAccess = [&](VPValue *Addr, Type *AccessTy) {
72 return !hasIrregularType(AccessTy, Plan.getDataLayout()) &&
73 vputils::getConstantStride(Addr, AccessTy, PSE, OuterLoop) == 1;
74 };
75
77 Plan.getVectorLoopRegion());
79 // Skip blocks outside region
80 if (!VPBB->getParent())
81 break;
82 VPRecipeBase *Term = VPBB->getTerminator();
83 auto EndIter = Term ? Term->getIterator() : VPBB->end();
84 // Introduce each ingredient into VPlan.
85 for (VPRecipeBase &Ingredient :
86 make_early_inc_range(make_range(VPBB->begin(), EndIter))) {
87
88 VPValue *VPV = Ingredient.getVPSingleValue();
89 if (!VPV->getUnderlyingValue())
90 continue;
91
93
94 VPRecipeBase *NewRecipe = nullptr;
95 if (auto *PhiR = dyn_cast<VPPhi>(&Ingredient)) {
96 auto *Phi = cast<PHINode>(PhiR->getUnderlyingValue());
97 NewRecipe = new VPWidenPHIRecipe(PhiR->operands(), PhiR->getDebugLoc(),
98 Phi->getName());
99 } else if (auto *VPI = dyn_cast<VPInstruction>(&Ingredient)) {
100 assert(!isa<PHINode>(Inst) && "phis should be handled above");
101 // Create VPWidenMemoryRecipe for loads and stores.
102 if (LoadInst *Load = dyn_cast<LoadInst>(Inst)) {
103 bool IsConsecutive =
104 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
105 NewRecipe = new VPWidenLoadRecipe(*Load, Ingredient.getOperand(0),
106 nullptr /*Mask*/, IsConsecutive,
107 *VPI, Ingredient.getDebugLoc());
108 } else if (StoreInst *Store = dyn_cast<StoreInst>(Inst)) {
109 bool IsConsecutive = IsConsecutiveAccess(
110 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
111 NewRecipe = new VPWidenStoreRecipe(
112 *Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
113 nullptr /*Mask*/, IsConsecutive, *VPI, Ingredient.getDebugLoc());
115 NewRecipe = new VPWidenGEPRecipe(GEP->getSourceElementType(),
116 Ingredient.operands(), *VPI,
117 Ingredient.getDebugLoc(), GEP);
118 } else if (CallInst *CI = dyn_cast<CallInst>(Inst)) {
119 Intrinsic::ID VectorID = getVectorIntrinsicIDForCall(CI, &TLI);
120 if (VectorID == Intrinsic::not_intrinsic)
121 return false;
122
123 // The noalias.scope.decl intrinsic declares a noalias scope that
124 // is valid for a single iteration. Emitting it as a single-scalar
125 // replicate would incorrectly extend the scope across multiple
126 // original iterations packed into one vector iteration.
127 // FIXME: If we want to vectorize this loop, then we have to drop
128 // all the associated !alias.scope and !noalias.
129 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
130 return false;
131
132 // These intrinsics are recognized by getVectorIntrinsicIDForCall
133 // but are not widenable. Emit them as replicate instead of widening.
134 if (VectorID == Intrinsic::assume ||
135 VectorID == Intrinsic::lifetime_end ||
136 VectorID == Intrinsic::lifetime_start ||
137 VectorID == Intrinsic::sideeffect ||
138 VectorID == Intrinsic::pseudoprobe) {
139 // If the operand of llvm.assume holds before vectorization, it will
140 // also hold per lane.
141 // llvm.pseudoprobe requires to be duplicated per lane for accurate
142 // sample count.
143 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
144 VectorID != Intrinsic::pseudoprobe;
145 NewRecipe = new VPReplicateRecipe(CI, Ingredient.operands(),
146 /*IsSingleScalar=*/IsSingleScalar,
147 /*Mask=*/nullptr, *VPI, *VPI,
148 Ingredient.getDebugLoc());
149 } else {
150 NewRecipe = new VPWidenIntrinsicRecipe(
151 *CI, VectorID, drop_end(Ingredient.operands()), CI->getType(),
152 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
153 }
154 } else if (auto *CI = dyn_cast<CastInst>(Inst)) {
155 NewRecipe = new VPWidenCastRecipe(
156 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
157 VPIRFlags(*CI), VPIRMetadata(*CI));
158 } else {
159 NewRecipe = new VPWidenRecipe(*Inst, Ingredient.operands(), *VPI,
160 *VPI, Ingredient.getDebugLoc());
161 }
162 } else {
164 "inductions must be created earlier");
165 continue;
166 }
167
168 NewRecipe->insertBefore(&Ingredient);
169 if (NewRecipe->getNumDefinedValues() == 1)
170 VPV->replaceAllUsesWith(NewRecipe->getVPSingleValue());
171 else
172 assert(NewRecipe->getNumDefinedValues() == 0 &&
173 "Only recpies with zero or one defined values expected");
174 Ingredient.eraseFromParent();
175 }
176 }
177 return true;
178}
179
180/// Helper for extra no-alias checks via known-safe recipe and SCEV.
183 VPReplicateRecipe &GroupLeader;
184 PredicatedScalarEvolution *PSE = nullptr;
185 const Loop *L = nullptr;
186
187 // Return true if \p A and \p B are known to not alias for all VFs in the
188 // plan, checked via the distance between the accesses
189 bool isNoAliasViaDistance(VPReplicateRecipe *A, VPReplicateRecipe *B) const {
190 if (A->getOpcode() != Instruction::Store ||
191 B->getOpcode() != Instruction::Store)
192 return false;
193
194 if (!PSE || !L)
195 return A == B;
196
197 VPValue *AddrA = A->getOperand(1);
198 const SCEV *SCEVA = vputils::getSCEVExprForVPValue(AddrA, *PSE, L);
199 VPValue *AddrB = B->getOperand(1);
200 const SCEV *SCEVB = vputils::getSCEVExprForVPValue(AddrB, *PSE, L);
202 return false;
203
204 const APInt *Distance;
205 ScalarEvolution &SE = *PSE->getSE();
206 if (!match(SE.getMinusSCEV(SCEVA, SCEVB), m_scev_APInt(Distance)))
207 return false;
208
209 const DataLayout &DL = SE.getDataLayout();
210 Type *TyA = A->getOperand(0)->getScalarType();
211 uint64_t SizeA = DL.getTypeStoreSize(TyA);
212 Type *TyB = B->getOperand(0)->getScalarType();
213 uint64_t SizeB = DL.getTypeStoreSize(TyB);
214
215 // Use the maximum store size to ensure no overlap from either direction.
216 // Currently only handles fixed sizes, as it is only used for
217 // replicating VPReplicateRecipes.
218 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
219
220 auto VFs = B->getParent()->getPlan()->vectorFactors();
222 if (MaxVF.isScalable())
223 return false;
224 return Distance->abs().uge(MaxVF.getFixedValue() * MaxStoreSize);
225 }
226
227public:
230 const Loop &L)
231 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
232 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
233
234 SinkStoreInfo(VPReplicateRecipe &GroupLeader) : GroupLeader(GroupLeader) {}
235
236 /// Return true if \p R should be skipped during alias checking, either
237 /// because it's in the exclude set or because no-alias can be proven via
238 /// SCEV.
239 bool shouldSkip(VPRecipeBase &R) const {
241 return ExcludeRecipes.contains(Store) ||
242 (Store && isNoAliasViaDistance(Store, &GroupLeader));
243 }
244};
245
246/// Check if a memory operation doesn't alias with memory operations using
247/// scoped noalias metadata, in blocks in the single-successor chain between \p
248/// FirstBB and \p LastBB. If \p SinkInfo is std::nullopt, only recipes that may
249/// write to memory are checked (for load hoisting). Otherwise recipes that both
250/// read and write memory are checked, and SCEV is used to prove no-alias
251/// between the group leader and other replicate recipes (for store sinking).
252static bool
254 VPBasicBlock *FirstBB, VPBasicBlock *LastBB,
255 std::optional<SinkStoreInfo> SinkInfo = {}) {
256 bool CheckReads = SinkInfo.has_value();
257 for (VPBasicBlock *VPBB :
259 for (VPRecipeBase &R : *VPBB) {
260 if (SinkInfo && SinkInfo->shouldSkip(R))
261 continue;
262
263 // Skip recipes that don't need checking.
264 if (!R.mayWriteToMemory() && !(CheckReads && R.mayReadFromMemory()))
265 continue;
266
268 if (!Loc)
269 // Conservatively assume aliasing for memory operations without
270 // location.
271 return false;
272
274 return false;
275 }
276 }
277 return true;
278}
279
280/// Get the value type of the replicate load or store. \p IsLoad indicates
281/// whether it is a load.
283 return (IsLoad ? R : R->getOperand(0))->getScalarType();
284}
285
286/// Collect either replicated Loads or Stores grouped by their address SCEV and
287/// their load-store type, in a deep-traversal of the vector loop region in \p
288/// Plan.
289template <unsigned Opcode>
292 VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L,
293 function_ref<bool(VPReplicateRecipe *)> FilterFn) {
294 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
295 "Only Load and Store opcodes supported");
296 constexpr bool IsLoad = (Opcode == Instruction::Load);
299 RecipesByAddressAndType;
303 if (RepR.getOpcode() != Opcode || !FilterFn(&RepR))
304 continue;
305
306 // For loads, operand 0 is address; for stores, operand 1 is address.
307 VPValue *Addr = RepR.getOperand(IsLoad ? 0 : 1);
308 const Type *LoadStoreTy = getLoadStoreValueType(&RepR, IsLoad);
309 const SCEV *AddrSCEV = vputils::getSCEVExprForVPValue(Addr, PSE, L);
310 if (!isa<SCEVCouldNotCompute>(AddrSCEV))
311 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(&RepR);
312 }
313 }
314 auto Groups = to_vector(RecipesByAddressAndType.values());
315 VPDominatorTree VPDT(Plan);
316 for (auto &Group : Groups) {
317 // Sort mem ops by dominance order, with earliest (most dominating) first.
319 return VPDT.properlyDominates(A, B);
320 });
321 }
322 return Groups;
323}
324
325static bool sinkScalarOperands(VPlan &Plan) {
326 auto Iter = vp_depth_first_deep(Plan.getEntry());
327 bool ScalarVFOnly = Plan.hasScalarVFOnly();
328 bool Changed = false;
329
331 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
332 VPBasicBlock *SinkTo, VPValue *Op) {
333 auto *Candidate = dyn_cast<VPSingleDefRecipe>(Op);
335 VPInstruction>(Candidate))
336 return;
337
338 if (Candidate->getParent() == SinkTo ||
339 all_of(Candidate->operands(),
340 [](VPValue *Op) { return Op->isDefinedOutsideLoopRegions(); }) ||
341 vputils::cannotHoistOrSinkRecipe(*Candidate, /*Sinking=*/true))
342 return;
343
344 if (!ScalarVFOnly && !vputils::doesGeneratePerAllLanes(Candidate))
345 return;
346
347 // Only single-scalar VPInstructions can be sunk.
348 if (auto *VPI = dyn_cast<VPInstruction>(Candidate))
349 if (!vputils::isSingleScalar(VPI))
350 return;
351
352 WorkList.insert({SinkTo, Candidate});
353 };
354
355 // First, collect the operands of all recipes in replicate blocks as seeds for
356 // sinking.
358 VPBasicBlock *EntryVPBB = VPR->getEntryBasicBlock();
359 if (!VPR->isReplicator() || EntryVPBB->getSuccessors().size() != 2)
360 continue;
361 VPBasicBlock *VPBB = cast<VPBasicBlock>(EntryVPBB->getSuccessors().front());
362 if (VPBB->getSingleSuccessor() != VPR->getExitingBasicBlock())
363 continue;
364 for (auto &Recipe : *VPBB)
365 for (VPValue *Op : Recipe.operands())
366 InsertIfValidSinkCandidate(VPBB, Op);
367 }
368
369 // Try to sink each replicate or scalar IV steps recipe in the worklist.
370 for (unsigned I = 0; I != WorkList.size(); ++I) {
371 VPBasicBlock *SinkTo;
372 VPSingleDefRecipe *SinkCandidate;
373 std::tie(SinkTo, SinkCandidate) = WorkList[I];
374
375 // All recipe users of SinkCandidate must be in the same block SinkTo or all
376 // users outside of SinkTo must only use the first lane of SinkCandidate. In
377 // the latter case, we need to duplicate SinkCandidate.
378 auto UsersOutsideSinkTo =
379 make_filter_range(SinkCandidate->users(), [SinkTo](VPUser *U) {
380 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
381 });
382 if (any_of(UsersOutsideSinkTo, [SinkCandidate](VPUser *U) {
383 return !U->usesFirstLaneOnly(SinkCandidate);
384 }))
385 continue;
386 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
387
388 if (NeedsDuplicating) {
389 if (ScalarVFOnly)
390 continue;
391 VPSingleDefRecipe *Clone;
392 if (auto *SinkCandidateRepR =
393 dyn_cast<VPReplicateRecipe>(SinkCandidate)) {
394 // TODO: Handle converting to uniform recipes as separate transform,
395 // then cloning should be sufficient here.
397 SinkCandidateRepR->getOpcode(), SinkCandidate->operands(),
398 /*Mask=*/nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
399 SinkCandidate->getDebugLoc(), SinkCandidate->getScalarType(),
400 SinkCandidate->getUnderlyingInstr());
401 // TODO: add ".cloned" suffix to name of Clone's VPValue.
402 } else {
403 Clone = SinkCandidate->clone();
404 }
405
406 Clone->insertBefore(SinkCandidate);
407 SinkCandidate->replaceUsesWithIf(Clone, [SinkTo](VPUser &U, unsigned) {
408 return cast<VPRecipeBase>(&U)->getParent() != SinkTo;
409 });
410 }
411 SinkCandidate->moveBefore(*SinkTo, SinkTo->getFirstNonPhi());
412 for (VPValue *Op : SinkCandidate->operands())
413 InsertIfValidSinkCandidate(SinkTo, Op);
414 Changed = true;
415 }
416 return Changed;
417}
418
419/// If \p R is a triangle region, return the 'then' block of the triangle.
421 auto *EntryBB = cast<VPBasicBlock>(R->getEntry());
422 if (EntryBB->getNumSuccessors() != 2)
423 return nullptr;
424
425 auto *Succ0 = dyn_cast<VPBasicBlock>(EntryBB->getSuccessors()[0]);
426 auto *Succ1 = dyn_cast<VPBasicBlock>(EntryBB->getSuccessors()[1]);
427 if (!Succ0 || !Succ1)
428 return nullptr;
429
430 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
431 return nullptr;
432 if (Succ0->getSingleSuccessor() == Succ1)
433 return Succ0;
434 if (Succ1->getSingleSuccessor() == Succ0)
435 return Succ1;
436 return nullptr;
437}
438
439// Merge replicate regions in their successor region, if a replicate region
440// is connected to a successor replicate region with the same predicate by a
441// single, empty VPBasicBlock.
443 SmallPtrSet<VPRegionBlock *, 4> TransformedRegions;
444
445 // Collect replicate regions followed by an empty block, followed by another
446 // replicate region with matching masks to process front. This is to avoid
447 // iterator invalidation issues while merging regions.
450 vp_depth_first_deep(Plan.getEntry()))) {
451 if (!Region1->isReplicator())
452 continue;
453 auto *MiddleBasicBlock =
454 dyn_cast_or_null<VPBasicBlock>(Region1->getSingleSuccessor());
455 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
456 continue;
457
458 auto *Region2 =
459 dyn_cast_or_null<VPRegionBlock>(MiddleBasicBlock->getSingleSuccessor());
460 if (!Region2 || !Region2->isReplicator())
461 continue;
462
463 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
464 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
465 if (!Mask1 || Mask1 != Mask2)
466 continue;
467
468 assert(Mask1 && Mask2 && "both region must have conditions");
469 WorkList.push_back(Region1);
470 }
471
472 // Move recipes from Region1 to its successor region, if both are triangles.
473 for (VPRegionBlock *Region1 : WorkList) {
474 if (TransformedRegions.contains(Region1))
475 continue;
476 auto *MiddleBasicBlock = cast<VPBasicBlock>(Region1->getSingleSuccessor());
477 auto *Region2 = cast<VPRegionBlock>(MiddleBasicBlock->getSingleSuccessor());
478
479 VPBasicBlock *Then1 = getPredicatedThenBlock(Region1);
480 VPBasicBlock *Then2 = getPredicatedThenBlock(Region2);
481 if (!Then1 || !Then2)
482 continue;
483
484 // The merged region is entered whenever either of the original regions was,
485 // so use the higher, i.e. more conservative, of their entry frequencies.
486 // If only one of the two is known, the higher one is unknown, so the
487 // result must be unknown too.
488 VPBranchOnMaskRecipe *Guard2 = Region2->getEntryBranchOnMask();
489 std::optional<VPExecutionFrequency> Freq1 =
490 Region1->getEntryBranchOnMask()->getExecutionFrequency();
491 std::optional<VPExecutionFrequency> Freq2 = Guard2->getExecutionFrequency();
492 if (Freq1 && Freq2) {
493 if (Freq2->Freq < Freq1->Freq) {
494 // Freq1's frequency is taken, but it is only as trustworthy as the
495 // less trustworthy of the two.
496 Freq1.emplace(Freq1->Freq, Freq1->IsEstimated || Freq2->IsEstimated);
497 Guard2->setExecutionFrequency(Freq1, Plan.getContext());
498 }
499 } else if (Freq2) {
500 Guard2->clearExecutionFrequency();
501 }
502
503 // Note: No fusion-preventing memory dependencies are expected in either
504 // region. Such dependencies should be rejected during earlier dependence
505 // checks, which guarantee accesses can be re-ordered for vectorization.
506 //
507 // Move recipes to the successor region.
508 for (VPRecipeBase &ToMove : make_early_inc_range(reverse(*Then1)))
509 ToMove.moveBefore(*Then2, Then2->getFirstNonPhi());
510
511 auto *Merge1 = cast<VPBasicBlock>(Then1->getSingleSuccessor());
512 auto *Merge2 = cast<VPBasicBlock>(Then2->getSingleSuccessor());
513
514 // Move VPPredInstPHIRecipes from the merge block to the successor region's
515 // merge block. Update all users inside the successor region to use the
516 // original values.
517 for (VPRecipeBase &Phi1ToMove : make_early_inc_range(reverse(*Merge1))) {
518 VPValue *PredInst1 =
519 cast<VPPredInstPHIRecipe>(&Phi1ToMove)->getOperand(0);
520 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
521 Phi1ToMoveV->replaceUsesWithIf(PredInst1, [Then2](VPUser &U, unsigned) {
522 return cast<VPRecipeBase>(&U)->getParent() == Then2;
523 });
524
525 // Remove phi recipes that are unused after merging the regions.
526 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
527 Phi1ToMove.eraseFromParent();
528 continue;
529 }
530 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
531 }
532
533 // Remove the dead recipes in Region1's entry block.
534 for (VPRecipeBase &R :
535 make_early_inc_range(reverse(*Region1->getEntryBasicBlock())))
536 R.eraseFromParent();
537
538 // Finally, remove the first region.
539 for (VPBlockBase *Pred : make_early_inc_range(Region1->getPredecessors())) {
540 VPBlockUtils::disconnectBlocks(Pred, Region1);
541 VPBlockUtils::connectBlocks(Pred, MiddleBasicBlock);
542 }
543 VPBlockUtils::disconnectBlocks(Region1, MiddleBasicBlock);
544 TransformedRegions.insert(Region1);
545 }
546
547 return !TransformedRegions.empty();
548}
549
551 VPRegionBlock *ParentRegion,
552 VPlan &Plan) {
553 Instruction *Instr = PredRecipe->getUnderlyingInstr();
554 // Build the triangular if-then region.
555 std::string RegionName = (Twine("pred.") + Instr->getOpcodeName()).str();
556 assert(Instr->getParent() && "Predicated instruction not in any basic block");
557 auto *BlockInMask = PredRecipe->getMask();
558 auto *MaskDef = BlockInMask->getDefiningRecipe();
559 auto *BOMRecipe = new VPBranchOnMaskRecipe(
560 BlockInMask, MaskDef ? MaskDef->getDebugLoc() : DebugLoc::getUnknown());
561 auto *Entry =
562 Plan.createVPBasicBlock(Twine(RegionName) + ".entry", BOMRecipe);
563
564 // Replace predicated replicate recipe with a replicate recipe without a
565 // mask but in the replicate region.
566 auto *RecipeWithoutMask = new VPReplicateRecipe(
567 PredRecipe->getUnderlyingInstr(), PredRecipe->operandsWithoutMask(),
568 PredRecipe->isSingleScalar(), nullptr /*Mask*/, *PredRecipe, *PredRecipe,
569 PredRecipe->getDebugLoc());
570 // The predicated recipe executes exactly when the guarding branch-on-mask is
571 // taken, so move its execution frequency there.
572 BOMRecipe->setExecutionFrequency(RecipeWithoutMask->getExecutionFrequency(),
573 Plan.getContext());
574 RecipeWithoutMask->clearExecutionFrequency();
575 auto *Pred =
576 Plan.createVPBasicBlock(Twine(RegionName) + ".if", RecipeWithoutMask);
577 auto *Exiting = Plan.createVPBasicBlock(Twine(RegionName) + ".continue");
579 Plan.createReplicateRegion(Entry, Exiting, RegionName);
580
581 // Note: first set Entry as region entry and then connect successors starting
582 // from it in order, to propagate the "parent" of each VPBasicBlock.
583 Region->setParent(ParentRegion);
584 VPBlockUtils::insertTwoBlocksAfter(Pred, Exiting, Entry);
585 VPBlockUtils::connectBlocks(Pred, Exiting);
586
587 if (!PredRecipe->user_empty()) {
588 auto *PHIRecipe = new VPPredInstPHIRecipe(RecipeWithoutMask,
589 RecipeWithoutMask->getDebugLoc());
590 Exiting->appendRecipe(PHIRecipe);
591 PredRecipe->replaceAllUsesWith(PHIRecipe);
592 }
593 PredRecipe->eraseFromParent();
594 return Region;
595}
596
597static void addReplicateRegions(VPlan &Plan) {
600 vp_depth_first_deep(Plan.getEntry()))) {
602 if (RepR.isPredicated())
603 WorkList.push_back(&RepR);
604 }
605
606 unsigned BBNum = 0;
607 for (VPReplicateRecipe *RepR : WorkList) {
608 VPBasicBlock *CurrentBlock = RepR->getParent();
609 VPBasicBlock *SplitBlock = CurrentBlock->splitAt(RepR->getIterator());
610
611 BasicBlock *OrigBB = RepR->getUnderlyingInstr()->getParent();
612 SplitBlock->setName(
613 OrigBB->hasName() ? OrigBB->getName() + "." + Twine(BBNum++) : "");
614 // Record predicated instructions for above packing optimizations.
616 createReplicateRegion(RepR, CurrentBlock->getParent(), Plan);
618
619 VPRegionBlock *ParentRegion = Region->getParent();
620 if (ParentRegion && ParentRegion->getExiting() == CurrentBlock)
621 ParentRegion->setExiting(SplitBlock);
622 }
623}
624
628 vp_depth_first_deep(Plan.getEntry()))) {
629 // Don't fold the blocks in the skeleton of the Plan into their single
630 // predecessors for now.
631 // TODO: Remove restriction once more of the skeleton is modeled in VPlan.
632 if (!VPBB->getParent())
633 continue;
634 auto *PredVPBB =
635 dyn_cast_or_null<VPBasicBlock>(VPBB->getSinglePredecessor());
636 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
637 isa<VPIRBasicBlock>(PredVPBB))
638 continue;
639 WorkList.push_back(VPBB);
640 }
641
642 for (VPBasicBlock *VPBB : WorkList) {
643 VPBasicBlock *PredVPBB = cast<VPBasicBlock>(VPBB->getSinglePredecessor());
644 for (VPRecipeBase &R : make_early_inc_range(*VPBB))
645 R.moveBefore(*PredVPBB, PredVPBB->end());
646 VPBlockUtils::disconnectBlocks(PredVPBB, VPBB);
647 auto *ParentRegion = VPBB->getParent();
648 if (ParentRegion && ParentRegion->getExiting() == VPBB)
649 ParentRegion->setExiting(PredVPBB);
650 VPBlockUtils::transferSuccessors(VPBB, PredVPBB);
651 // VPBB is now dead and will be cleaned up when the plan gets destroyed.
652 }
653 return !WorkList.empty();
654}
655
657 // Convert masked VPReplicateRecipes to if-then region blocks.
659
660 bool ShouldSimplify = true;
661 while (ShouldSimplify) {
662 ShouldSimplify = sinkScalarOperands(Plan);
663 ShouldSimplify |= mergeReplicateRegionsIntoSuccessors(Plan);
664 ShouldSimplify |= mergeBlocksIntoPredecessors(Plan);
665 }
666}
667
668/// Remove redundant casts of inductions.
669///
670/// Such redundant casts are casts of induction variables that can be ignored,
671/// because we already proved that the casted phi is equal to the uncasted phi
672/// in the vectorized loop. There is no need to vectorize the cast - the same
673/// value can be used for both the phi and casts in the vector loop.
678 if (IV.getTruncInst())
679 continue;
680
681 // A sequence of IR Casts has potentially been recorded for IV, which
682 // *must be bypassed* when the IV is vectorized, because the vectorized IV
683 // will produce the desired casted value. This sequence forms a def-use
684 // chain and is provided in reverse order, ending with the cast that uses
685 // the IV phi. Search for the recipe of the last cast in the chain and
686 // replace it with the original IV. Note that only the final cast is
687 // expected to have users outside the cast-chain and the dead casts left
688 // over will be cleaned up later.
689 ArrayRef<Instruction *> Casts = IV.getInductionDescriptor().getCastInsts();
690 VPValue *FindMyCast = &IV;
691 for (Instruction *IRCast : reverse(Casts)) {
692 VPSingleDefRecipe *FoundUserCast = nullptr;
693 for (auto *U : FindMyCast->users()) {
694 auto *UserCast = dyn_cast<VPSingleDefRecipe>(U);
695 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
696 FoundUserCast = UserCast;
697 break;
698 }
699 }
700 // A cast recipe in the chain may have been removed by earlier DCE.
701 if (!FoundUserCast)
702 break;
703 FindMyCast = FoundUserCast;
704 }
705 if (FindMyCast != &IV)
706 FindMyCast->replaceAllUsesWith(&IV);
707 }
708}
709
710/// If R is a phi-like recipe starting a dead cycle of recipes, erase all
711/// reachable recipes of the dead cycle and return true. Otherwise leave the
712/// plan unchanged and return false.
714 auto *PhiR = dyn_cast<VPSingleDefRecipe>(R);
715 if (!PhiR || !isa<VPPhi, VPReductionPHIRecipe>(R))
716 return false;
717
718 // The transitive users of PhiR are closed under users, so the cycle is dead
719 // if every one of them can be erased.
721 auto *R = cast<VPRecipeBase>(U);
722 // Bail out if a user must be retained, or if it is a phi-like recipe other
723 // than PhiR;
724 if (R->mayHaveSideEffects() || (R != PhiR && isa<VPPhiAccessors>(R)))
725 return false;
726 }
727
728 // Break the cycle by replacing PhiR with its first incoming value, which is
729 // defined outside the cycle. That leaves the rest of the cycle dead.
730 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
731 SmallVector<VPValue *> Incoming(PhiR->operands());
732 PhiR->eraseFromParent();
733 for (VPValue *Op : Incoming)
735 return true;
736}
737
740 Plan.getEntry());
742 // The recipes in the block are processed in reverse order, to catch chains
743 // of dead recipes.
744 for (VPRecipeBase &R : make_early_inc_range(reverse(*VPBB)))
746 R.eraseFromParent();
747
748 // Erase dead cycles starting at one of VPBB's phi-like recipes. Erasing a
749 // cycle may also erase other phi-like recipes of VPBB, so restart the scan
750 // of the phi section after each removal. This terminates, as each removal
751 // erases the cycle's phi.
752 bool Changed = true;
753 while (Changed) {
754 Changed = false;
755 for (VPRecipeBase &R : VPBB->phis()) {
756 if (tryToRemoveDeadCycle(&R)) {
757 Changed = true;
758 break;
759 }
760 }
761 }
762 }
763}
764
765/// Legalize VPWidenPointerInductionRecipe, by replacing it with a PtrAdd
766/// (IndStart, ScalarIVSteps (0, Step)) if only its scalar values are used, as
767/// VPWidenPointerInductionRecipe will generate vectors only. If some users
768/// require vectors while other require scalars, the scalar uses need to extract
769/// the scalars from the generated vectors (Note that this is different to how
770/// int/fp inductions are handled). Legalize extract-from-ends using uniform
771/// VPReplicateRecipe of wide inductions to use regular VPReplicateRecipe, so
772/// the correct end value is available. Also optimize
773/// VPWidenIntOrFpInductionRecipe, if any of its users needs scalar values, by
774/// providing them scalar steps built on the canonical scalar IV and update the
775/// original IV's users. This is an optional optimization to reduce the needs of
776/// vector extracts.
779 bool HasOnlyVectorVFs = !Plan.hasScalarVFOnly();
780
782 for (VPWidenInductionRecipe &PhiR :
784 WideIVs.push_back(&PhiR);
785
786 // Try to narrow wide and replicating recipes to uniform recipes, based on
787 // VPlan analysis.
788 // TODO: Apply to all recipes in the future, to replace legacy uniformity
789 // analysis.
790 for (VPWidenInductionRecipe *PhiR : WideIVs) {
792 for (VPUser *U : reverse(Users)) {
793 auto *Def = dyn_cast<VPRecipeWithIRFlags>(U);
794 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
795 // Skip recipes that shouldn't be narrowed.
796 if (!Def ||
798 Def->user_empty() || !Def->getUnderlyingValue() ||
799 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
800 continue;
801
802 // Skip recipes that may have other lanes than their first used.
804 continue;
805
806 // TODO: Support scalarizing ExtractValue.
807 if (match(Def,
809 continue;
810
812 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
813 /*Mask=*/nullptr, *Def, getMetadataOf(Def), DebugLoc::getUnknown(),
814 Def->getScalarType(), Def->getUnderlyingInstr());
815 Clone->insertAfter(Def);
816 Def->replaceAllUsesWith(Clone);
817 Def->eraseFromParent();
818 }
819 }
820
821 VPBuilder Builder(HeaderVPBB, HeaderVPBB->getFirstNonPhi());
822 for (VPWidenInductionRecipe *PhiR : WideIVs) {
823 // Replace wide pointer inductions which have only their scalars used by
824 // PtrAdd(IndStart, ScalarIVSteps (0, Step)).
825 if (auto *PtrIV = dyn_cast<VPWidenPointerInductionRecipe>(PhiR)) {
826 if (!Plan.hasScalarVFOnly() &&
827 !PtrIV->onlyScalarsGenerated(Plan.hasScalableVF()))
828 continue;
829
830 VPValue *PtrAdd =
831 vputils::scalarizeVPWidenPointerInduction(PtrIV, Plan, Builder);
832 PtrIV->replaceAllUsesWith(PtrAdd);
833 continue;
834 }
835
836 // Replace widened induction with scalar steps for users that only use
837 // scalars.
838 auto *WideIV = cast<VPWidenIntOrFpInductionRecipe>(PhiR);
839 if (HasOnlyVectorVFs && none_of(WideIV->users(), [WideIV](VPUser *U) {
840 return U->usesScalars(WideIV);
841 }))
842 continue;
843
844 const InductionDescriptor &ID = WideIV->getInductionDescriptor();
845 VPIRFlags::WrapFlagsTy WrapFlags;
846 // We can preserve nuw when the step is non-negative.
847 const APInt *Step;
848 if (match(WideIV->getStepValue(), m_APInt(Step)) && Step->isNonNegative())
849 WrapFlags = {static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
850 false};
852 Plan, ID.getKind(), ID.getInductionOpcode(),
853 dyn_cast_or_null<FPMathOperator>(ID.getInductionBinOp()),
854 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
855 WideIV->getDebugLoc(), Builder, WrapFlags);
856
857 // Update scalar users of IV to use Step instead.
858 if (!HasOnlyVectorVFs) {
859 assert(!Plan.hasScalableVF() &&
860 "plans containing a scalar VF cannot also include scalable VFs");
861 WideIV->replaceAllUsesWith(Steps);
862 } else {
863 bool HasScalableVF = Plan.hasScalableVF();
864 WideIV->replaceUsesWithIf(Steps,
865 [WideIV, HasScalableVF](VPUser &U, unsigned) {
866 if (HasScalableVF)
867 return U.usesFirstLaneOnly(WideIV);
868 return U.usesScalars(WideIV);
869 });
870 }
871 }
872}
873
874/// Check if \p VPV is an untruncated wide induction, either before or after the
875/// increment. If so return the header IV (before the increment), otherwise
876/// return null.
879 auto *WideIV = dyn_cast<VPWidenInductionRecipe>(VPV);
880 if (WideIV) {
881 // VPV itself is a wide induction, separately compute the end value for exit
882 // users if it is not a truncated IV.
883 auto *IntOrFpIV = dyn_cast<VPWidenIntOrFpInductionRecipe>(WideIV);
884 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
885 }
886
887 // Check if VPV is an optimizable induction increment.
888 VPRecipeBase *Def = VPV->getDefiningRecipe();
889 if (!Def || Def->getNumOperands() != 2)
890 return nullptr;
891 WideIV = dyn_cast<VPWidenInductionRecipe>(Def->getOperand(0));
892 if (!WideIV)
893 WideIV = dyn_cast<VPWidenInductionRecipe>(Def->getOperand(1));
894 if (!WideIV)
895 return nullptr;
896
897 auto IsWideIVInc = [&]() {
898 auto &ID = WideIV->getInductionDescriptor();
899
900 // Check if VPV increments the induction by the induction step.
901 VPValue *IVStep = WideIV->getStepValue();
902 switch (ID.getInductionOpcode()) {
903 case Instruction::Add:
904 return match(VPV, m_c_Add(m_Specific(WideIV), m_Specific(IVStep)));
905 case Instruction::FAdd:
906 return match(VPV, m_c_FAdd(m_Specific(WideIV), m_Specific(IVStep)));
907 case Instruction::FSub:
908 return match(VPV, m_Binary<Instruction::FSub>(m_Specific(WideIV),
909 m_Specific(IVStep)));
910 case Instruction::Sub: {
911 // IVStep will be the negated step of the subtraction. Check if Step == -1
912 // * IVStep.
913 VPValue *Step;
914 if (!match(VPV, m_Sub(m_VPValue(), m_VPValue(Step))))
915 return false;
916 const SCEV *IVStepSCEV = vputils::getSCEVExprForVPValue(IVStep, PSE);
917 const SCEV *StepSCEV = vputils::getSCEVExprForVPValue(Step, PSE);
918 ScalarEvolution &SE = *PSE.getSE();
919 return !isa<SCEVCouldNotCompute>(IVStepSCEV) &&
920 !isa<SCEVCouldNotCompute>(StepSCEV) &&
921 IVStepSCEV == SE.getNegativeSCEV(StepSCEV);
922 }
923 default:
924 return ID.getKind() == InductionDescriptor::IK_PtrInduction &&
925 match(VPV, m_GetElementPtr(m_Specific(WideIV),
926 m_Specific(WideIV->getStepValue())));
927 }
928 llvm_unreachable("should have been covered by switch above");
929 };
930 return IsWideIVInc() ? WideIV : nullptr;
931}
932
933/// Attempts to optimize the induction variable exit values for users in the
934/// early exit block.
937 VPValue *Incoming, *Mask;
939 m_VPValue(Incoming))))
940 return nullptr;
941
942 auto *WideIV = getOptimizableIVOf(Incoming, PSE);
943 if (!WideIV)
944 return nullptr;
945
946 // Calculate the final index.
947 VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
948 auto *CanonicalIV = LoopRegion->getCanonicalIV();
949 Type *CanonicalIVType = LoopRegion->getCanonicalIVType();
950 auto *ExtractR = cast<VPInstruction>(Op);
951 VPBuilder B(ExtractR);
952
953 DebugLoc DL = ExtractR->getDebugLoc();
954 VPValue *FirstActiveLane = B.createFirstActiveLane(Mask, DL);
955 FirstActiveLane =
956 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType, DL);
957 VPValue *EndValue = B.createAdd(CanonicalIV, FirstActiveLane, DL);
958
959 // `getOptimizableIVOf()` always returns the pre-incremented IV, so if it
960 // changed it means the exit is using the incremented value, so we need to
961 // add the step.
962 if (Incoming != WideIV) {
963 VPValue *One = Plan.getConstantInt(CanonicalIVType, 1);
964 EndValue = B.createAdd(EndValue, One, DL);
965 }
966
967 if (!match(WideIV, m_CanonicalWidenIV())) {
968 const InductionDescriptor &ID = WideIV->getInductionDescriptor();
969 VPValue *Start = WideIV->getStartValue();
970 VPValue *Step = WideIV->getStepValue();
971 EndValue = B.createDerivedIV(
972 ID.getKind(), dyn_cast_or_null<FPMathOperator>(ID.getInductionBinOp()),
973 Start, EndValue, Step);
974 }
975
976 return EndValue;
977}
978
979/// Compute the end value for \p WideIV, unless it is truncated. Creates a
980/// VPDerivedIVRecipe for non-canonical inductions.
982 VPBuilder &VectorPHBuilder,
983 VPValue *VectorTC) {
984 auto *WideIntOrFp = dyn_cast<VPWidenIntOrFpInductionRecipe>(WideIV);
985 // Truncated wide inductions resume from the last lane of their vector value
986 // in the last vector iteration which is handled elsewhere.
987 if (WideIntOrFp && WideIntOrFp->getTruncInst())
988 return nullptr;
989
990 VPValue *Start = WideIV->getStartValue();
991 VPValue *Step = WideIV->getStepValue();
992 const InductionDescriptor &ID = WideIV->getInductionDescriptor();
993 VPValue *EndValue = VectorTC;
994 if (!match(WideIV, m_CanonicalWidenIV())) {
995 EndValue = VectorPHBuilder.createDerivedIV(
996 ID.getKind(), dyn_cast_or_null<FPMathOperator>(ID.getInductionBinOp()),
997 Start, VectorTC, Step);
998 }
999
1000 // EndValue is derived from the vector trip count (which has the same type as
1001 // the widest induction) and thus may be wider than the induction here.
1002 Type *ScalarTypeOfWideIV = WideIV->getScalarType();
1003 if (ScalarTypeOfWideIV != EndValue->getScalarType()) {
1004 EndValue = VectorPHBuilder.createScalarCast(Instruction::Trunc, EndValue,
1005 ScalarTypeOfWideIV,
1006 WideIV->getDebugLoc());
1007 }
1008
1009 return EndValue;
1010}
1011
1012/// Attempts to optimize the induction variable exit values for users in the
1013/// exit block coming from the latch in the original scalar loop.
1014static VPValue *
1018 VPValue *Incoming;
1021 m_VPValue(Incoming)))))
1022 return nullptr;
1023
1024 VPWidenInductionRecipe *WideIV = getOptimizableIVOf(Incoming, PSE);
1025 if (!WideIV)
1026 return nullptr;
1027
1028 VPValue *EndValue = EndValues.lookup(WideIV);
1029 assert(EndValue && "Must have computed the end value up front");
1030
1031 // `getOptimizableIVOf()` always returns the pre-incremented IV, so if it
1032 // changed it means the exit is using the incremented value, so we don't
1033 // need to subtract the step.
1034 if (Incoming != WideIV)
1035 return EndValue;
1036
1037 // Otherwise, subtract the step from the EndValue.
1038 auto *ExtractR = cast<VPInstruction>(Op);
1039 VPBuilder B(ExtractR);
1040 VPValue *Step = WideIV->getStepValue();
1041 Type *ScalarTy = WideIV->getScalarType();
1042 if (ScalarTy->isIntegerTy())
1043 return B.createSub(EndValue, Step, DebugLoc::getUnknown(), "ind.escape");
1044 if (ScalarTy->isPointerTy()) {
1045 Type *StepTy = Step->getScalarType();
1046 auto *Zero = Plan.getZero(StepTy);
1047 return B.createPtrAdd(EndValue, B.createSub(Zero, Step),
1048 DebugLoc::getUnknown(), "ind.escape");
1049 }
1050 if (ScalarTy->isFloatingPointTy()) {
1051 const auto &ID = WideIV->getInductionDescriptor();
1052 return B.createNaryOp(
1053 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1054 ? Instruction::FSub
1055 : Instruction::FAdd,
1056 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1057 }
1058 llvm_unreachable("all possible induction types must be handled");
1059 return nullptr;
1060}
1061
1064 VPValue *ResumeTC,
1065 const Loop *L) {
1066 VPValue *Incoming;
1069 m_VPValue(Incoming)))))
1070 return nullptr;
1071
1072 const SCEV *IncomingSCEV = vputils::getSCEVExprForVPValue(Incoming, PSE, L);
1073 const SCEV *Start, *Step;
1074 if (!match(IncomingSCEV, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step),
1075 m_SpecificLoop(L))))
1076 return nullptr;
1077
1078 auto *ExtractR = cast<VPInstruction>(Op);
1079 DebugLoc DL = ExtractR->getDebugLoc();
1080 VPBuilder Builder(ExtractR);
1081 VPSCEVExpander Expander(Builder, *PSE.getSE(), DL);
1082 VPValue *StartVPV = Expander.expand(Start);
1083 VPValue *StepVPV = Expander.expand(Step);
1084
1085 Type *StartTy = StartVPV->getScalarType();
1086 assert(StartTy->isIntOrPtrTy() && "The type must be SCEVable");
1090 Type *TCTy = ResumeTC->getScalarType();
1091 VPValue *ExitCount = Builder.createOverflowingOp(
1092 Instruction::Sub, {ResumeTC, Plan.getConstantInt(TCTy, 1)},
1093 {/*HasNUW=*/true, /*HasNSW=*/false}, DebugLoc::getUnknown());
1094 return Builder.createDerivedIV(Kind, /*FPBinOp=*/nullptr, StartVPV, ExitCount,
1095 StepVPV);
1096}
1097
1099 VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L) {
1100 // Compute end values for all inductions.
1101 VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
1102 auto *VectorPH = cast<VPBasicBlock>(VectorRegion->getSinglePredecessor());
1103 VPBuilder VectorPHBuilder(VectorPH, VectorPH->getFirstNonPhi());
1105 VPValue *ResumeTC =
1106 Plan.hasTailFolded() ? Plan.getTripCount() : &Plan.getVectorTripCount();
1108 VectorRegion->getEntryBasicBlock()->phis())) {
1110 &WideIV, VectorPHBuilder, ResumeTC))
1111 EndValues[&WideIV] = EndValue;
1112 }
1113
1114 VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
1115 for (VPRecipeBase &R : make_early_inc_range(*MiddleVPBB)) {
1116 VPValue *Op;
1117 if (!match(&R, m_ExitingIVValue(m_VPValue(Op))))
1118 continue;
1119 auto *WideIV = cast<VPWidenInductionRecipe>(Op);
1120 if (VPValue *EndValue = EndValues.lookup(WideIV)) {
1121 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1122 R.eraseFromParent();
1123 }
1124 }
1125
1126 // Then, optimize exit block users.
1127 for (VPIRBasicBlock *ExitVPBB : Plan.getExitBlocks()) {
1128 for (VPRecipeBase &R : ExitVPBB->phis()) {
1129 auto *ExitIRI = cast<VPIRPhi>(&R);
1130
1131 for (auto [Idx, PredVPBB] : enumerate(ExitVPBB->getPredecessors())) {
1132 VPValue *Escape = nullptr;
1133 if (PredVPBB == MiddleVPBB) {
1135 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1136 if (!Escape)
1138 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1139 } else {
1141 Plan, ExitIRI->getOperand(Idx), PSE);
1142 }
1143 if (Escape)
1144 ExitIRI->setOperand(Idx, Escape);
1145 }
1146 }
1147 }
1148}
1149
1150/// Remove redundant ExpandSCEVRecipes in \p Plan's entry block by replacing
1151/// them with already existing recipes expanding the same SCEV expression.
1154
1155 for (VPExpandSCEVRecipe &ExpR :
1157 *Plan.getEntry()->getEntryBasicBlock()))) {
1158 const auto &[V, Inserted] = SCEV2VPV.try_emplace(ExpR.getSCEV(), &ExpR);
1159 if (Inserted)
1160 continue;
1161
1162 ExpR.replaceAllUsesWith(V->second);
1163 if (&ExpR == Plan.getTripCount())
1164 Plan.resetTripCount(V->second);
1165
1166 ExpR.eraseFromParent();
1167 }
1168}
1169
1170/// Try to simplify logical and bitwise recipes in \p Def.
1172 VPValue *X;
1173
1174 // X | AllOnes -> AllOnes
1175 if (match(Def, m_c_BinaryOr(m_VPValue(X), m_AllOnes())))
1176 return Plan.getAllOnesValue(Def->getScalarType());
1177
1178 // X | 0 -> X
1179 if (match(Def, m_c_BinaryOr(m_VPValue(X), m_ZeroInt())))
1180 return X;
1181
1182 // X | !X -> AllOnes
1184 return Plan.getAllOnesValue(Def->getScalarType());
1185
1186 // X & 0 -> 0
1187 if (match(Def, m_c_BinaryAnd(m_VPValue(X), m_ZeroInt())))
1188 return Plan.getZero(Def->getScalarType());
1189
1190 // X & AllOnes -> X
1191 if (match(Def, m_c_BinaryAnd(m_VPValue(X), m_AllOnes())))
1192 return X;
1193
1194 // X && false -> false
1195 if (match(Def, m_c_LogicalAnd(m_VPValue(X), m_False())))
1196 return Plan.getFalse();
1197
1198 // X && true -> X
1199 if (match(Def, m_c_LogicalAnd(m_VPValue(X), m_True())))
1200 return X;
1201
1202 // X && (X && Y) -> X && Y
1203 if (match(Def, m_LogicalAnd(m_VPValue(X),
1205 return Def->getOperand(1);
1206
1207 // X && !X -> 0
1209 return Plan.getFalse();
1210
1211 if (match(Def, m_Select(m_VPValue(), m_VPValue(X), m_Deferred(X))))
1212 return X;
1213
1214 return nullptr;
1215}
1216
1217/// Swap the branch weights recorded for \p R, a select whose two selected
1218/// operands are being swapped,
1220 auto *MD = dyn_cast<VPIRMetadata>(&R);
1221 if (!MD)
1222 return;
1224 if (!extractBranchWeights(MD->getMetadata(LLVMContext::MD_prof), Weights))
1225 return;
1226 assert(Weights.size() == 2 && "unexpected branch weights");
1227 MD->setMetadata(
1228 LLVMContext::MD_prof,
1229 MDBuilder(Plan.getContext()).createBranchWeights(Weights[1], Weights[0]));
1230}
1231
1232/// Return an existing value or a live in for VPSingleDefRecipe \p Def if
1233/// possible. This shouldn't create or modify recipes.
1235 // Simplification of live-in IR values for SingleDef recipes using
1236 // InstSimplifyFolder.
1237 const DataLayout &DL = Plan.getDataLayout();
1238 if (VPValue *V = vputils::tryToFoldLiveIns(*Def, Def->operands(), DL))
1239 return V;
1240
1241 // Fold PredPHI LiveIn -> LiveIn.
1242 if (auto *PredPHI = dyn_cast<VPPredInstPHIRecipe>(Def)) {
1243 VPValue *Op = PredPHI->getOperand(0);
1244 if (isa<VPIRValue>(Op))
1245 return Op;
1246 }
1247
1248 if (VPValue *V = simplifyLogicalRecipe(Plan, Def))
1249 return V;
1250
1251 VPValue *A, *B;
1252
1253 if (match(Def, m_c_Add(m_VPValue(A), m_ZeroInt())))
1254 return A;
1255
1256 if (match(Def, m_c_Mul(m_VPValue(A), m_One())))
1257 return A;
1258
1259 if (match(Def, m_c_Mul(m_VPValue(), m_ZeroInt())))
1260 return Plan.getZero(Def->getScalarType());
1261
1262 // A bitcast to the same type is a no-op.
1263 if (match(Def, m_BitCast(m_VPValue(A))) &&
1264 Def->getScalarType() == A->getScalarType())
1265 return A;
1266
1267 // Shifting by zero is a no-op.
1270 m_AShr(m_VPValue(A), m_ZeroInt())))))
1271 return A;
1272
1273 if (match(Def, m_Trunc(m_ZExtOrSExt(m_VPValue(A)))))
1274 if (Def->getScalarType() == A->getScalarType())
1275 return A;
1276
1277 if (match(Def, m_Not(m_Not(m_VPValue(A)))))
1278 return A;
1279
1280 // Remove redundant DerviedIVs, that is 0 + A * 1 -> A and 0 + 0 * x -> 0.
1281 if ((match(Def, m_DerivedIV(m_ZeroInt(), m_VPValue(A), m_One())) ||
1283 m_VPValue()))) &&
1284 A->getScalarType() == Def->getScalarType())
1285 return A;
1286
1287 // Simplify MaskedCond with no block mask to its single operand.
1289 !cast<VPInstruction>(Def)->isMasked())
1290 return Def->getOperand(0);
1291
1292 // Look through ExtractLastLane.
1293 if (match(Def, m_ExtractLastLane(m_VPValue(A)))) {
1294 if (match(A, m_BuildVector())) {
1295 auto *BuildVector = cast<VPInstruction>(A);
1296 return BuildVector->getOperand(BuildVector->getNumOperands() - 1);
1297 }
1298
1299 if (match(A, m_Broadcast(m_VPValue(B))))
1300 return B;
1301
1303 return A;
1304
1305 if (Plan.hasScalarVFOnly())
1306 return A;
1307 }
1308
1309 // Look through ExtractPenultimateElement (BuildVector ....).
1311 auto *BuildVector = cast<VPInstruction>(Def->getOperand(0));
1312 return BuildVector->getOperand(BuildVector->getNumOperands() - 2);
1313 }
1314
1315 uint64_t Idx;
1317 auto *BuildVector = cast<VPInstruction>(Def->getOperand(0));
1318 return BuildVector->getOperand(Idx);
1319 }
1320
1322 if (Def->getNumOperands() == 1) {
1323 return Def->getOperand(0);
1324 }
1325 if (auto *Phi = dyn_cast<VPFirstOrderRecurrencePHIRecipe>(Def)) {
1326 if (all_equal(Phi->incoming_values()))
1327 return Phi->getOperand(0);
1328 }
1329 return nullptr;
1330 }
1331
1332 VPIRValue *IRV;
1333 if (Def->getNumOperands() == 1 &&
1335 return IRV;
1336
1338 m_One())) &&
1339 A->getScalarType() == Def->getScalarType())
1340 return A;
1341
1342 // Some simplifications can only be applied after unrolling. Perform them
1343 // below.
1344 if (!Plan.isUnrolled())
1345 return nullptr;
1346
1347 // Simplify extracts of the same single-scalar.
1349 all_equal(drop_begin(Def->operands())) &&
1350 vputils::isSingleScalar(Def->getOperand(1)))
1351 return Def->getOperand(1);
1352
1353 // Replace extract-lane(0, canonical-WIDEN-INDUCTION) with the region's
1354 // scalar canonical IV.
1356 if (match(Def, m_ExtractLane(m_ZeroInt(), m_CanonicalWidenIV(WidenIV))))
1357 return WidenIV->getRegion()->getCanonicalIV();
1358
1359 // Simplify unrolled VectorPointer without offset, or with zero offset, to
1360 // just the pointer operand.
1361 if (auto *VPR = dyn_cast<VPVectorPointerRecipe>(Def))
1362 if (!VPR->getVFxPart() || match(VPR->getVFxPart(), m_ZeroInt()))
1363 return VPR->getOperand(0);
1364
1365 // VPScalarIVSteps after unrolling can be replaced by their start value, if
1366 // the start index is zero and only the first lane 0 is demanded.
1367 if (auto *Steps = dyn_cast<VPScalarIVStepsRecipe>(Def))
1368 if (!Steps->getStartIndex() && vputils::onlyFirstLaneUsed(Steps))
1369 return Steps->getOperand(0);
1370
1371 if (Plan.getConcreteUF() == 1 && match(Def, m_ExtractLastPart(m_VPValue(A))))
1372 return A;
1373
1374 return nullptr;
1375}
1376
1377/// Returns true if \p V is available at the end of \p VPBB, i.e. it either is a
1378/// live-in from the original IR or defined in \p VPBB.
1379static bool isAvailableAtEndOf(VPValue *V, const VPBasicBlock *VPBB) {
1380 VPRecipeBase *DefR = V->getDefiningRecipe();
1381 return DefR ? DefR->getParent() == VPBB : isa<VPIRValue>(V);
1382}
1383
1384namespace {
1385/// Inserter for VPBuilderBase which appends all created VPSingleDefRecipes to a
1386/// worklist, so they get combined as well.
1387struct VPCombineInserter {
1388 SmallVectorImpl<VPSingleDefRecipe *> &Worklist;
1389
1390 void insertHelper(VPRecipeBase *R, VPBasicBlock *VPBB,
1392 VPBB->insert(R, It);
1393 if (auto *Def = dyn_cast<VPSingleDefRecipe>(R))
1394 Worklist.push_back(Def);
1395 }
1396};
1397
1398using VPCombineBuilder = VPBuilderBase<VPCombineInserter>;
1399} // namespace
1400
1401/// Combine \p Def into a simpler recipe. May modify or create new recipes via
1402/// \p Builder.
1404 VPCombineBuilder &Builder) {
1405 if (auto *V = simplifyRecipe(Plan, Def)) {
1406 Def->replaceAllUsesWith(V);
1407 return Def;
1408 }
1409
1410 // Drop the mask of a predicated store masked by the header mask (which is
1411 // guaranteed to be true at least for the first lane) and both the stored
1412 // value and the address are uniform across VF and UF. The header mask is
1413 // still the abstract region value here.
1414 if (auto *RepR = dyn_cast<VPReplicateRecipe>(Def);
1415 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1416 all_of(RepR->operandsWithoutMask(), vputils::isUniformAcrossVFsAndUFs) &&
1417 match(RepR->getMask(), m_HeaderMask())) {
1418 auto *Unmasked = new VPReplicateRecipe(
1419 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1420 RepR->isSingleScalar(), /*Mask=*/nullptr, *RepR, *RepR,
1421 RepR->getDebugLoc());
1422 Builder.insert(Unmasked);
1423 return Unmasked;
1424 }
1425
1426 // Avoid replacing VPInstructions with underlying values with new
1427 // VPInstructions, as we would fail to create widen/replicate recpes from the
1428 // new VPInstructions without an underlying value, and miss out on some
1429 // transformations that only apply to widened/replicated recipes later, by
1430 // doing so.
1431 // TODO: We should also not replace non-VPInstructions like VPWidenRecipe with
1432 // VPInstructions without underlying values, as those will get skipped during
1433 // cost computation.
1434 bool CanCreateNewRecipe =
1435 !isa<VPInstruction>(Def) || !Def->getUnderlyingValue();
1436
1437 VPValue *X, *Y, *Z;
1438
1439 // X && (Y && X) -> X && Y
1440 if (CanCreateNewRecipe &&
1443 return Builder.createLogicalAnd(X, Y);
1444
1445 // (X && Y) | (X && Z) -> X && (Y | Z)
1446 if (CanCreateNewRecipe &&
1449 // Simplify only if one of the operands has one use to avoid creating an
1450 // extra recipe.
1451 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1452 !Def->getOperand(1)->hasMoreThanOneUniqueUser()))
1453 return Builder.createLogicalAnd(X, Builder.createOr(Y, Z));
1454
1455 // (X && Y) | !X -> !X || Y
1456 if (CanCreateNewRecipe &&
1457 match(Def,
1459 m_VPValue(Z, m_Not(m_Deferred(X))))))
1460 return Builder.createLogicalOr(Z, Y);
1461
1462 // select C, false, true -> not C
1463 VPValue *C;
1464 if (CanCreateNewRecipe &&
1465 match(Def, m_Select(m_VPValue(C), m_False(), m_True())))
1466 return Builder.createNot(C);
1467
1468 // select !C, X, Y -> select C, Y, X
1469 if (match(Def, m_Select(m_Not(m_VPValue(C)), m_VPValue(X), m_VPValue(Y)))) {
1470 Def->setOperand(0, C);
1471 Def->setOperand(1, Y);
1472 Def->setOperand(2, X);
1473 swapSelectBranchWeights(*Def, Plan);
1474 return Def;
1475 }
1476
1477 // select X, (i1 Y | Z), Y -> Y | (X && Z)
1478 if (CanCreateNewRecipe &&
1479 match(Def, m_Select(m_VPValue(X),
1481 m_Deferred(Y))) &&
1482 Y->getScalarType()->isIntegerTy(1))
1483 return Builder.createOr(Y, Builder.createLogicalAnd(X, Z));
1484
1485 // select M0, (select M1, X, Y), Y -> select (M0 && M1), X, Y
1486 VPValue *Mask0, *Mask1;
1487 if (CanCreateNewRecipe &&
1488 match(Def,
1489 m_SelectLike(m_VPValue(Mask0),
1491 m_VPValue(Y))),
1492 m_Deferred(Y))))
1493 return Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1), X, Y,
1494 Def->getDebugLoc());
1495
1496 if (match(Def, m_Trunc(m_VPValue(Y, m_ZExtOrSExt(m_VPValue(X)))))) {
1497 // Don't replace a non-widened cast recipe with a widened cast.
1498 if (!isa<VPWidenCastRecipe>(Def))
1499 return nullptr;
1500 Type *TruncTy = Def->getScalarType();
1501 Type *XTy = X->getScalarType();
1502 if (XTy->getScalarSizeInBits() < TruncTy->getScalarSizeInBits()) {
1503
1504 unsigned ExtOpcode =
1505 match(Y, m_SExt(m_VPValue())) ? Instruction::SExt : Instruction::ZExt;
1506 auto *Ext =
1507 Builder.createWidenCast(Instruction::CastOps(ExtOpcode), X, TruncTy);
1508 if (auto *UnderlyingExt = Y->getUnderlyingValue()) {
1509 // UnderlyingExt has distinct return type, used to retain legacy cost.
1510 Ext->setUnderlyingValue(UnderlyingExt);
1511 }
1512 return Ext;
1513 } else if (XTy->getScalarSizeInBits() > TruncTy->getScalarSizeInBits()) {
1514 auto *Trunc = Builder.createWidenCast(Instruction::Trunc, X, TruncTy);
1515 return Trunc;
1516 }
1517 }
1518
1519 if (CanCreateNewRecipe && match(Def, m_c_Mul(m_VPValue(X), m_AllOnes()))) {
1520 // Preserve nsw from the Mul on the new Sub.
1522 false, cast<VPRecipeWithIRFlags>(Def)->hasNoSignedWrap()};
1523 return Builder.createSub(Plan.getZero(X->getScalarType()), X,
1524 Def->getDebugLoc(), "", NW);
1525 }
1526
1527 if (CanCreateNewRecipe &&
1528 match(Def, m_c_Add(m_VPValue(X),
1529 m_VPValue(Z, m_Sub(m_ZeroInt(), m_VPValue(Y)))))) {
1530 // Preserve nsw from the Add and the Sub, if it's present on both, on the
1531 // new Sub.
1533 false, cast<VPRecipeWithIRFlags>(Def)->hasNoSignedWrap() &&
1534 cast<VPRecipeWithIRFlags>(Z)->hasNoSignedWrap()};
1535 return Builder.createSub(X, Y, Def->getDebugLoc(), "", NW);
1536 }
1537
1538 const APInt *APC;
1539 if (CanCreateNewRecipe && match(Def, m_URem(m_VPValue(X), m_APInt(APC))) &&
1540 APC->isPowerOf2())
1541 return Builder.createAnd(X, Plan.getConstantInt(*APC - 1),
1542 Def->getDebugLoc());
1543
1544 if (CanCreateNewRecipe && match(Def, m_c_Mul(m_VPValue(X), m_APInt(APC))) &&
1545 APC->isPowerOf2()) {
1546 auto *MulR = cast<VPRecipeWithIRFlags>(Def);
1547 unsigned ShiftAmt = APC->exactLogBase2();
1548 VPIRFlags::WrapFlagsTy NW(MulR->hasNoUnsignedWrap(),
1549 MulR->hasNoSignedWrap() &&
1550 ShiftAmt != APC->getBitWidth() - 1);
1551 return Builder.createNaryOp(
1552 Instruction::Shl,
1553 {X, Plan.getConstantInt(APC->getBitWidth(), ShiftAmt)}, NW,
1554 Def->getDebugLoc());
1555 }
1556
1557 if (CanCreateNewRecipe && match(Def, m_UDiv(m_VPValue(X), m_APInt(APC))) &&
1558 APC->isPowerOf2())
1559 return Builder.createNaryOp(
1560 Instruction::LShr,
1561 {X, Plan.getConstantInt(APC->getBitWidth(), APC->exactLogBase2())},
1562 *cast<VPRecipeWithIRFlags>(Def), Def->getDebugLoc());
1563
1564 // (X >> C) << C -> X & (-1 << C).
1565 if (CanCreateNewRecipe &&
1566 match(Def, m_Shl(m_LShr(m_VPValue(X), m_VPValue(Y, m_APInt(APC))),
1567 m_Deferred(Y))))
1568 return Builder.createAnd(
1569 X, Plan.getConstantInt(APInt::getAllOnes(APC->getBitWidth()) << *APC),
1570 Def->getDebugLoc());
1571
1572 if (match(Def, m_Not(m_VPValue(X)))) {
1573 // Try to fold Not into compares by adjusting the predicate in-place.
1574 CmpPredicate Pred;
1575 if (match(X, m_Cmp(Pred, m_VPValue(), m_VPValue()))) {
1576 auto *Cmp = cast<VPRecipeWithIRFlags>(X);
1577 // Only fold if every user is a Not of the cmp, or a select using the cmp
1578 // solely as its condition.
1579 if (all_of(Cmp->users(), [Cmp](VPUser *U) {
1580 return match(U, m_Not(m_Specific(Cmp))) ||
1581 (match(U, m_Select(m_Specific(Cmp), m_VPValue(),
1582 m_VPValue())) &&
1583 U->getOperand(1) != Cmp && U->getOperand(2) != Cmp);
1584 })) {
1585 Cmp->setPredicate(CmpInst::getInversePredicate(Pred));
1586 for (VPUser *U : to_vector(Cmp->users())) {
1587 auto *R = cast<VPSingleDefRecipe>(U);
1588 if (match(R, m_Select(m_Specific(Cmp), m_VPValue(X), m_VPValue(Y)))) {
1589 // select (cmp pred), X, Y -> select (cmp inv_pred), Y, X
1590 R->setOperand(1, Y);
1591 R->setOperand(2, X);
1592 swapSelectBranchWeights(*R, Plan);
1593 } else {
1594 // not (cmp pred) -> cmp inv_pred
1595 assert(match(R, m_Not(m_Specific(Cmp))) && "Unexpected user");
1596 R->replaceAllUsesWith(Cmp);
1597 }
1598 }
1599 // If Cmp doesn't have a debug location, use the one from the negation,
1600 // to preserve the location.
1601 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1602 Cmp->setDebugLoc(Def->getDebugLoc());
1603 return Def;
1604 }
1605 }
1606 }
1607
1608 // Fold any-of (fcmp uno A, A), (fcmp uno B, B), ... ->
1609 // any-of (fcmp uno A, B), ...
1610 if (match(Def, m_AnyOf())) {
1612 VPRecipeBase *UnpairedCmp = nullptr;
1613 for (VPValue *Op : Def->operands()) {
1614 VPValue *X;
1615 if (Op->getNumUsers() > 1 ||
1617 m_Deferred(X)))) {
1618 NewOps.push_back(Op);
1619 } else if (!UnpairedCmp) {
1620 UnpairedCmp = Op->getDefiningRecipe();
1621 } else {
1622 NewOps.push_back(Builder.createFCmp(CmpInst::FCMP_UNO,
1623 UnpairedCmp->getOperand(0), X));
1624 UnpairedCmp = nullptr;
1625 }
1626 }
1627
1628 if (UnpairedCmp)
1629 NewOps.push_back(UnpairedCmp->getVPSingleValue());
1630
1631 if (NewOps.size() < Def->getNumOperands())
1632 return Builder.createNaryOp(VPInstruction::AnyOf, NewOps);
1633 }
1634
1635 // Fold (fcmp uno X, X) | (fcmp uno Y, Y) -> fcmp uno X, Y
1636 // This is useful for fmax/fmin without fast-math flags, where we need to
1637 // check if any operand is NaN.
1638 if (CanCreateNewRecipe &&
1639 match(Def,
1640 m_BinaryOr(
1643 return Builder.createFCmp(CmpInst::FCMP_UNO, X, Y);
1644
1646 m_One())) &&
1647 X->getScalarType() != Def->getScalarType())
1648 return Builder.createWidenCast(Instruction::Trunc, X, Def->getScalarType());
1649
1650 // For i1 vp.merges produced by AnyOf reductions:
1651 // vp.merge true, (or X, Y), X, evl -> vp.merge Y, true, X, evl
1653 m_VPValue(X), m_VPValue())) &&
1655 Def->getScalarType()->isIntegerTy(1)) {
1656 Def->setOperand(1, Plan.getTrue());
1657 Def->setOperand(0, Y);
1658 return Def;
1659 }
1660
1661 if (match(Def, m_BuildVector()) && all_equal(Def->operands()))
1662 return Builder.createNaryOp(VPInstruction::Broadcast, Def->getOperand(0));
1663
1664 // Replace uses of a BuildVector by users that only use its first lane with
1665 // its first operand directly.
1666 if (match(Def, m_BuildVector())) {
1667 Def->replaceUsesWithIf(Def->getOperand(0), [Def](VPUser &U, unsigned) {
1668 return U.usesFirstLaneOnly(Def);
1669 });
1670 return Def;
1671 }
1672
1673 // Look through broadcast of single-scalar when used as select conditions; in
1674 // that case the scalar condition can be used directly.
1675 if (match(Def,
1678 "broadcast operand must be single-scalar");
1679 Def->setOperand(0, Z);
1680 return Def;
1681 }
1682
1683 if (match(Def, m_Broadcast(m_VPValue(X)))) {
1684 Def->replaceUsesWithIf(
1685 X, [Def](const VPUser &U, unsigned) { return U.usesScalars(Def); });
1686 return Def;
1687 }
1688
1689 // Some simplifications can only be applied after unrolling. Perform them
1690 // below.
1691 if (!Plan.isUnrolled())
1692 return nullptr;
1693
1694 // Simplify extract-lane with single source to extract-element.
1695 VPValue *LaneToExtract;
1696 if (match(Def, m_ExtractLane(m_VPValue(LaneToExtract), m_VPValue(X))))
1697 return Builder.createNaryOp(Instruction::ExtractElement, {X, LaneToExtract},
1698 Def->getDebugLoc());
1699
1700 // Look for cycles where Def is of the form:
1701 // X = phi(0, IVInc) ; used only by IVInc, or by IVInc and Inc = X + Y
1702 // IVInc = X + Step ; used by X and Def
1703 // Def = IVInc + Y
1704 // Fold the increment Y into the phi's start value, replace Def with IVInc,
1705 // and if Inc exists, replace it with X.
1706 VPValue *IVInc;
1707 if (match(Def, m_Add(m_VPValue(IVInc, m_Add(m_VPValue(X), m_VPValue())),
1708 m_VPValue(Y))) &&
1709 match(X, m_VPPhi(m_ZeroInt(), m_Specific(IVInc))) &&
1710 IVInc->getNumUsers() == 2) {
1711 auto *Phi = cast<VPPhi>(X);
1712 // If Phi has a second user (besides IVInc's defining recipe), it must be
1713 // Inc = Phi + Y for the fold to apply.
1715 findUserOf(Phi, m_Add(m_Specific(Phi), m_Specific(Y))));
1716 if ((Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) &&
1717 isAvailableAtEndOf(Y, Phi->getIncomingBlock(0))) {
1718 Def->replaceAllUsesWith(IVInc);
1719 if (Inc)
1720 Inc->replaceAllUsesWith(Phi);
1721 Phi->setOperand(0, Y);
1722 return Def;
1723 }
1724 }
1725
1726 // Simplify redundant ReductionStartVector recipes after unrolling.
1727 VPValue *StartV;
1729 m_VPValue(StartV), m_VPValue(), m_VPValue()))) {
1730 Def->replaceUsesWithIf(StartV, [](const VPUser &U, unsigned Idx) {
1731 auto *PhiR = dyn_cast<VPReductionPHIRecipe>(&U);
1732 return PhiR && PhiR->isInLoop();
1733 });
1734 return Def;
1735 }
1736
1737 return nullptr;
1738}
1739
1743 Plan.getEntry());
1745 for (VPSingleDefRecipe &Def :
1747 Worklist.push_back(&Def);
1748
1749 [[maybe_unused]] unsigned InitWorklistSize = Worklist.size();
1750
1751 VPCombineBuilder Builder({Worklist});
1752 while (!Worklist.empty()) {
1753 assert(Worklist.size() < InitWorklistSize * 2 &&
1754 "Worklist is growing large, possible cycle?");
1755 VPSingleDefRecipe *Def = Worklist.pop_back_val();
1756 Builder.setInsertPoint(Def);
1757 VPSingleDefRecipe *New = combineRecipe(Plan, Def, Builder);
1758 if (!New)
1759 continue;
1760 if (New != Def) {
1761 // Replace the recipe with a new one.
1762 Def->replaceAllUsesWith(New);
1763 Def->eraseFromParent();
1764 // TODO: Append users to the worklist (might need a setvector)
1765 } else if (vputils::isDeadRecipe(*Def)) {
1766 // Recipe was modified - it may be dead now.
1767 Def->eraseFromParent();
1768 }
1769 }
1770}
1771
1773 // Pull out reverses from any elementwise op.
1774 // binop(reverse(x), reverse(y)) -> reverse(binop(x,y))
1776 Plan, [](VPValue *&X) { return m_Reverse(m_VPValue(X)); },
1777 [](auto *X) { return new VPInstruction(VPInstruction::Reverse, X); });
1778
1779 // reverse(reverse(x)) -> x
1780 VPValue *X;
1783 for (VPRecipeBase &R : make_early_inc_range(*VPBB))
1784 if (match(&R, m_Reverse(m_Reverse(m_VPValue(X)))))
1785 R.getVPSingleValue()->replaceAllUsesWith(X);
1786}
1787
1788/// Reassociate (headermask && x) && y -> headermask && (x && y) to allow the
1789/// header mask to be simplified further when tail folding, e.g. in
1790/// optimizeEVLMasks.
1791static void reassociateHeaderMask(VPlan &Plan) {
1792 VPValue *HeaderMask = Plan.getVectorLoopRegion()->getHeaderMask();
1793 if (!HeaderMask)
1794 return;
1795
1796 SmallVector<VPUser *> Worklist;
1797 for (VPUser *U : HeaderMask->users())
1798 if (match(U, m_LogicalAnd(m_Specific(HeaderMask), m_VPValue())))
1800
1801 while (!Worklist.empty()) {
1802 auto *R = dyn_cast<VPSingleDefRecipe>(Worklist.pop_back_val());
1803 VPValue *X, *Y;
1804 if (!R || !match(R, m_LogicalAnd(
1805 m_LogicalAnd(m_Specific(HeaderMask), m_VPValue(X)),
1806 m_VPValue(Y))))
1807 continue;
1808 append_range(Worklist, R->users());
1809 VPBuilder Builder(R);
1810 R->replaceAllUsesWith(
1811 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(X, Y)));
1812 }
1813}
1814
1815static std::optional<Instruction::BinaryOps>
1817 switch (ID) {
1818 case Intrinsic::masked_udiv:
1819 return Instruction::UDiv;
1820 case Intrinsic::masked_sdiv:
1821 return Instruction::SDiv;
1822 case Intrinsic::masked_urem:
1823 return Instruction::URem;
1824 case Intrinsic::masked_srem:
1825 return Instruction::SRem;
1826 default:
1827 return {};
1828 }
1829}
1830
1832 if (Plan.hasScalarVFOnly())
1833 return;
1834
1836 vp_depth_first_deep(Plan.getEntry()))) {
1837 for (VPRecipeBase &R : make_early_inc_range(reverse(*VPBB))) {
1840 continue;
1841 auto *RepR = dyn_cast<VPReplicateRecipe>(&R);
1842 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1843 continue;
1844
1845 auto *RepOrWidenR = cast<VPRecipeWithIRFlags>(&R);
1846 if (RepR && RepR->getOpcode() == Instruction::Store &&
1847 vputils::isSingleScalar(RepR->getOperand(1))) {
1848 auto *Clone = new VPReplicateRecipe(
1849 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1850 true /*IsSingleScalar*/, nullptr /*Mask*/, *RepR /*Flags*/,
1851 *RepR /*Metadata*/, RepR->getDebugLoc());
1852 Clone->insertBefore(RepOrWidenR);
1853 VPBuilder Builder(Clone);
1854 VPValue *ExtractOp = Clone->getOperand(0);
1855 if (vputils::isUniformAcrossVFsAndUFs(RepR->getOperand(1)))
1856 ExtractOp =
1857 Builder.createNaryOp(VPInstruction::ExtractLastPart, ExtractOp);
1858 ExtractOp =
1859 Builder.createNaryOp(VPInstruction::ExtractLastLane, ExtractOp);
1860 Clone->setOperand(0, ExtractOp);
1861 RepR->eraseFromParent();
1862 continue;
1863 }
1864
1865 // Narrow llvm.masked.{u,s}{div,rem} intrinsics with a safe divisor.
1866 if (auto *IntrR = dyn_cast<VPWidenIntrinsicRecipe>(RepOrWidenR)) {
1867 if (!vputils::onlyFirstLaneUsed(IntrR))
1868 continue;
1869 auto Opc = getUnmaskedDivRemOpcode(IntrR->getVectorIntrinsicID());
1870 if (!Opc)
1871 continue;
1872 VPBuilder Builder(IntrR);
1873 VPValue *SafeDivisor = Builder.createSelect(
1874 IntrR->getOperand(2), IntrR->getOperand(1),
1875 Plan.getConstantInt(IntrR->getScalarType(), 1));
1876 VPValue *Clone = Builder.createNaryOp(
1877 *Opc, {IntrR->getOperand(0), SafeDivisor},
1878 VPIRFlags::getDefaultFlags(*Opc), IntrR->getDebugLoc());
1879 IntrR->replaceAllUsesWith(Clone);
1880 IntrR->eraseFromParent();
1881 continue;
1882 }
1883
1884 // Skip recipes that aren't single scalars.
1885 if (!vputils::isSingleScalar(RepOrWidenR))
1886 continue;
1887
1888 // Predicate to check if a user of Op introduces extra broadcasts.
1889 auto IntroducesBCastOf = [](const VPValue *Op) {
1890 return [Op](const VPUser *U) {
1891 if (auto *VPI = dyn_cast<VPInstruction>(U)) {
1895 VPI->getOpcode()))
1896 return false;
1897 }
1898 return !U->usesScalars(Op);
1899 };
1900 };
1901
1902 if (any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1903 none_of(RepOrWidenR->operands(), [&](VPValue *Op) {
1904 if (any_of(
1905 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1906 IntroducesBCastOf(Op)))
1907 return false;
1908 // Non-constant live-ins require broadcasts, while constants do not
1909 // need explicit broadcasts.
1910 bool LiveInNeedsBroadcast =
1911 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1912 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1913 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1914 }))
1915 continue;
1916
1917 auto *Clone = VPBuilder::createSingleScalarOp(
1918 vputils::getOpcode(RepOrWidenR), RepOrWidenR->operands(),
1919 /*Mask=*/nullptr, *RepOrWidenR, getMetadataOf(RepOrWidenR),
1920 DebugLoc::getUnknown(), RepOrWidenR->getScalarType(),
1921 RepOrWidenR->getUnderlyingInstr());
1922 Clone->insertBefore(RepOrWidenR);
1923 RepOrWidenR->replaceAllUsesWith(Clone);
1924 if (vputils::isDeadRecipe(*RepOrWidenR))
1925 RepOrWidenR->eraseFromParent();
1926 }
1927 }
1928}
1929
1930/// Try to see if all of \p Blend's masks share a common value logically and'ed
1931/// and remove it from the masks.
1933 if (Blend->isNormalized())
1934 return;
1935 VPValue *CommonEdgeMask;
1936 if (!match(Blend->getMask(0),
1937 m_LogicalAnd(m_VPValue(CommonEdgeMask), m_VPValue())))
1938 return;
1939 for (unsigned I = 0; I < Blend->getNumIncomingValues(); I++)
1940 if (!match(Blend->getMask(I),
1941 m_LogicalAnd(m_Specific(CommonEdgeMask), m_VPValue())))
1942 return;
1943 for (unsigned I = 0; I < Blend->getNumIncomingValues(); I++)
1944 Blend->setMask(I, Blend->getMask(I)->getDefiningRecipe()->getOperand(1));
1945}
1946
1947/// Normalize and simplify VPBlendRecipes. Should be run after combineRecipes
1948/// to make sure the masks are simplified.
1949static void simplifyBlends(VPlan &Plan) {
1952 for (VPBlendRecipe &Blend :
1954 removeCommonBlendMask(&Blend);
1955
1956 // Try to remove redundant blend recipes.
1957 SmallPtrSet<VPValue *, 4> UniqueValues;
1958 if (Blend.isNormalized() || !match(Blend.getMask(0), m_False()))
1959 UniqueValues.insert(Blend.getIncomingValue(0));
1960 for (unsigned I = 1; I != Blend.getNumIncomingValues(); ++I)
1961 if (!match(Blend.getMask(I), m_False()))
1962 UniqueValues.insert(Blend.getIncomingValue(I));
1963
1964 if (UniqueValues.size() == 1) {
1965 Blend.replaceAllUsesWith(*UniqueValues.begin());
1966 Blend.eraseFromParent();
1967 continue;
1968 }
1969
1970 if (Blend.isNormalized())
1971 continue;
1972
1973 // Normalize the blend so its first incoming value is used as the initial
1974 // value with the others blended into it.
1975
1976 unsigned StartIndex = 0;
1977 for (unsigned I = 0; I != Blend.getNumIncomingValues(); ++I) {
1978 // If a value's mask is used only by the blend then is can be deadcoded.
1979 // TODO: Find the most expensive mask that can be deadcoded, or a mask
1980 // that's used by multiple blends where it can be removed from them all.
1981 VPValue *Mask = Blend.getMask(I);
1982 if (Mask->hasOneUse() && !match(Mask, m_False())) {
1983 StartIndex = I;
1984 break;
1985 }
1986 }
1987
1988 SmallVector<VPValue *, 4> OperandsWithMask;
1989 OperandsWithMask.push_back(Blend.getIncomingValue(StartIndex));
1990
1991 for (unsigned I = 0; I != Blend.getNumIncomingValues(); ++I) {
1992 if (I == StartIndex)
1993 continue;
1994 OperandsWithMask.push_back(Blend.getIncomingValue(I));
1995 OperandsWithMask.push_back(Blend.getMask(I));
1996 }
1997
1998 auto *NewBlend =
1999 new VPBlendRecipe(cast_or_null<PHINode>(Blend.getUnderlyingValue()),
2000 OperandsWithMask, Blend, Blend.getDebugLoc());
2001 NewBlend->insertBefore(&Blend);
2002
2003 VPValue *DeadMask = Blend.getMask(StartIndex);
2004 Blend.replaceAllUsesWith(NewBlend);
2005 Blend.eraseFromParent();
2007
2008 /// Simplify BLEND %a, %b, Not(%mask) -> BLEND %b, %a, %mask.
2009 VPValue *NewMask;
2010 if (NewBlend->getNumOperands() == 3 &&
2011 match(NewBlend->getMask(1), m_Not(m_VPValue(NewMask)))) {
2012 VPValue *Inc0 = NewBlend->getOperand(0);
2013 VPValue *Inc1 = NewBlend->getOperand(1);
2014 VPValue *OldMask = NewBlend->getOperand(2);
2015 NewBlend->setOperand(0, Inc1);
2016 NewBlend->setOperand(1, Inc0);
2017 NewBlend->setOperand(2, NewMask);
2018 if (OldMask->user_empty())
2019 cast<VPInstruction>(OldMask)->eraseFromParent();
2020 }
2021 }
2022 }
2023}
2024
2025/// Optimize the width of vector induction variables in \p Plan based on a known
2026/// constant Trip Count, \p BestVF and \p BestUF.
2028 ElementCount BestVF,
2029 unsigned BestUF) {
2030 // Only proceed if we have not completely removed the vector region.
2031 if (!Plan.getVectorLoopRegion())
2032 return false;
2033
2034 const APInt *TC;
2035 if (!BestVF.isFixed() || !match(Plan.getTripCount(), m_APInt(TC)))
2036 return false;
2037
2038 // Calculate the minimum power-of-2 bit width that can fit the known TC, VF
2039 // and UF. Returns at least 8.
2040 auto ComputeBitWidth = [](APInt TC, uint64_t Align) {
2041 APInt AlignedTC =
2044 APInt MaxVal = AlignedTC - 1;
2045 return std::max<unsigned>(PowerOf2Ceil(MaxVal.getActiveBits()), 8);
2046 };
2047 unsigned NewBitWidth =
2048 ComputeBitWidth(*TC, BestVF.getKnownMinValue() * BestUF);
2049
2050 LLVMContext &Ctx = Plan.getContext();
2051 auto *NewIVTy = IntegerType::get(Ctx, NewBitWidth);
2052
2053 bool MadeChange = false;
2054
2055 VPBasicBlock *HeaderVPBB = Plan.getVectorLoopRegion()->getEntryBasicBlock();
2056 for (VPRecipeBase &Phi : HeaderVPBB->phis()) {
2057 // Currently only handle canonical IVs as it is trivial to replace the start
2058 // and stop values, and we currently only perform the optimization when the
2059 // IV has a single use.
2061 if (!match(&Phi, m_CanonicalWidenIV(WideIV)))
2062 continue;
2063 if (WideIV->hasMoreThanOneUniqueUser() ||
2064 NewIVTy == WideIV->getScalarType())
2065 continue;
2066
2067 // Currently only handle cases where the single user is a header-mask
2068 // comparison with the backedge-taken-count.
2069 VPUser *SingleUser = WideIV->getSingleUser();
2070 if (!SingleUser ||
2071 !match(SingleUser,
2072 m_ICmp(m_Specific(WideIV),
2074 continue;
2075
2076 // Update IV operands and comparison bound to use new narrower type.
2077 assert(!WideIV->getTruncInst() &&
2078 "canonical IV is not expected to have a truncation");
2079 auto *NewWideIV = new VPWidenIntOrFpInductionRecipe(
2080 WideIV->getPHINode(), Plan.getZero(NewIVTy),
2081 Plan.getConstantInt(NewIVTy, 1), WideIV->getVFValue(),
2082 WideIV->getInductionDescriptor(), *WideIV, WideIV->getDebugLoc());
2083 NewWideIV->insertBefore(WideIV);
2084
2085 auto *NewBTC = new VPWidenCastRecipe(
2086 Instruction::Trunc, Plan.getOrCreateBackedgeTakenCount(), NewIVTy,
2087 nullptr, VPIRFlags::getDefaultFlags(Instruction::Trunc));
2088 Plan.getVectorPreheader()->appendRecipe(NewBTC);
2089 auto *Cmp = cast<VPInstruction>(WideIV->getSingleUser());
2090 Cmp->replaceAllUsesWith(
2091 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2092
2093 MadeChange = true;
2094 }
2095
2096 return MadeChange;
2097}
2098
2099/// Return true if \p Cond is known to be true for given \p BestVF and \p
2100/// BestUF.
2102 ElementCount BestVF, unsigned BestUF,
2105 return any_of(Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2106 &PSE](VPValue *C) {
2107 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2108 });
2109
2110 auto *CanIV = Plan.getVectorLoopRegion()->getCanonicalIV();
2113 m_c_Add(m_Specific(CanIV), m_Specific(&Plan.getVFxUF())),
2114 m_Specific(&Plan.getVectorTripCount()))))
2115 return false;
2116
2117 // The compare checks CanIV + VFxUF == vector trip count. The vector trip
2118 // count is not conveniently available as SCEV so far, so we compare directly
2119 // against the original trip count. This is stricter than necessary, as we
2120 // will only return true if the trip count == vector trip count.
2121 const SCEV *VectorTripCount =
2123 if (isa<SCEVCouldNotCompute>(VectorTripCount))
2124 VectorTripCount = vputils::getSCEVExprForVPValue(Plan.getTripCount(), PSE);
2125 assert(!isa<SCEVCouldNotCompute>(VectorTripCount) &&
2126 "Trip count SCEV must be computable");
2127 ScalarEvolution &SE = *PSE.getSE();
2128 ElementCount NumElements = BestVF * BestUF;
2129 const SCEV *C = SE.getElementCount(VectorTripCount->getType(), NumElements);
2130 return SE.isKnownPredicate(CmpInst::ICMP_EQ, VectorTripCount, C);
2131}
2132
2133// Replaces ExtractVectorForPart instructions with ICMP when the VF is scalar
2134// and the source is a WideActiveLaneMask. The unused mask is removed later
2135// when removing dead recipes.
2137 ElementCount BestVF) {
2138 if (!BestVF.isScalar())
2139 return false;
2140
2141 bool MadeChange = false;
2142 VPBuilder Builder;
2143 VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
2144 VPBasicBlock *PreheaderVPBB = Plan.getVectorPreheader();
2145 VPBasicBlock *ExitingVPBB = VectorRegion->getExitingBasicBlock();
2146
2147 VPValue *Start, *TC;
2148 uint64_t Idx;
2149 for (VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
2150 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
2153 m_VPValue()),
2154 m_ConstantInt(Idx))))
2155 continue;
2156
2157 auto *Extract = cast<VPInstruction>(&R);
2158 Builder.setInsertPoint(Extract);
2159
2160 if (Idx > 0)
2161 Start = Builder.createAdd(
2162 Start, Plan.getConstantInt(Start->getScalarType(), Idx));
2163
2164 VPValue *ICmp = Builder.createICmp(CmpInst::ICMP_ULT, Start, TC);
2165 Extract->replaceAllUsesWith(ICmp);
2166 Extract->eraseFromParent();
2167 MadeChange = true;
2168 }
2169 }
2170
2171 return MadeChange;
2172}
2173
2174/// Try to simplify the branch condition of \p Plan. This may restrict the
2175/// resulting plan to \p BestVF and \p BestUF.
2177 unsigned BestUF,
2179 VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
2180 VPBasicBlock *ExitingVPBB = VectorRegion->getExitingBasicBlock();
2181 auto *Term = &ExitingVPBB->back();
2182 VPValue *Cond;
2183 VPValue *Offset = nullptr;
2184 auto m_CanIVInc = m_Add(m_VPValue(), m_Specific(&Plan.getVFxUF()));
2185 // Check if the branch condition compares the canonical IV increment (for main
2186 // loop), or the canonical IV increment plus an offset (for epilog loop).
2187 bool MatchedCanIVInc =
2188 match(Term,
2190 m_CombineOr(m_CanIVInc, m_c_Add(m_CanIVInc, m_VPValue(Offset))),
2191 m_VPValue())) &&
2192 (!Offset || Offset->isDefinedOutsideLoopRegions());
2193 if (MatchedCanIVInc ||
2194 match(Term,
2197 m_ZeroInt()))))) {
2198 // Try to simplify the branch condition if VectorTC <= VF * UF when the
2199 // latch terminator is BranchOnCount or
2200 // BranchOnCond(Not(ExtractVectorForPart(WideActiveLaneMask), 0))
2201 const SCEV *VectorTripCount =
2203 if (isa<SCEVCouldNotCompute>(VectorTripCount))
2204 VectorTripCount =
2206 assert(!isa<SCEVCouldNotCompute>(VectorTripCount) &&
2207 "Trip count SCEV must be computable");
2208 ScalarEvolution &SE = *PSE.getSE();
2209 ElementCount NumElements = BestVF * BestUF;
2210 const SCEV *C = SE.getElementCount(VectorTripCount->getType(), NumElements);
2211 if (!SE.isKnownPredicate(CmpInst::ICMP_ULE, VectorTripCount, C))
2212 return false;
2213 } else if (match(Term, m_BranchOnCond(m_VPValue(Cond))) ||
2215 // For BranchOnCond, check if we can prove the condition to be true using VF
2216 // and UF.
2217 if (!isConditionTrueViaVFAndUF(Cond, Plan, BestVF, BestUF, PSE))
2218 return false;
2219 } else {
2220 return false;
2221 }
2222
2223 // The vector loop region only executes once. Convert terminator of the
2224 // exiting block to exit in the first iteration.
2225 if (match(Term, m_BranchOnTwoConds())) {
2226 Term->setOperand(1, Plan.getTrue());
2227 return true;
2228 }
2229
2230 auto *BOC = new VPInstruction(VPInstruction::BranchOnCond, Plan.getTrue(), {},
2231 {}, Term->getDebugLoc());
2232 ExitingVPBB->appendRecipe(BOC);
2233 Term->eraseFromParent();
2234
2235 return true;
2236}
2237
2239 unsigned BestUF,
2241 assert(Plan.hasVF(BestVF) && "BestVF is not available in Plan");
2242 assert(Plan.hasUF(BestUF) && "BestUF is not available in Plan");
2243
2244 bool MadeChange =
2245 simplifyBranchConditionForVFAndUF(Plan, BestVF, BestUF, PSE);
2246 MadeChange |= replaceMaskWithCompareForScalarPlan(Plan, BestVF);
2247 MadeChange |= optimizeVectorInductionWidthForTCAndVFUF(Plan, BestVF, BestUF);
2248
2249 if (MadeChange) {
2250 Plan.setVF(BestVF);
2251 assert(Plan.getConcreteUF() == BestUF && "BestUF must match the Plan's UF");
2252 }
2253}
2254
2258 RecurKind RK = PhiR.getRecurrenceKind();
2259 if (RK != RecurKind::Add && RK != RecurKind::Mul && RK != RecurKind::Sub &&
2261 continue;
2262
2264 if (auto *RecWithFlags = dyn_cast<VPRecipeWithIRFlags>(U)) {
2265 RecWithFlags->dropPoisonGeneratingFlags();
2266 }
2267 }
2268}
2269
2270namespace {
2271struct VPCSEDenseMapInfo : public DenseMapInfo<VPSingleDefRecipe *> {
2272 /// If recipe \p R will lower to a GEP with a non-i8 source element type,
2273 /// return that source element type.
2274 static Type *getGEPSourceElementType(const VPSingleDefRecipe *R) {
2275 // All VPInstructions that lower to GEPs must have the i8 source element
2276 // type (as they are PtrAdds), so we omit it.
2278 .Case([](const VPReplicateRecipe *I) -> Type * {
2279 if (auto *GEP = dyn_cast<GetElementPtrInst>(I->getUnderlyingValue()))
2280 return GEP->getSourceElementType();
2281 return nullptr;
2282 })
2283 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2284 [](auto *I) { return I->getSourceElementType(); })
2285 .Default([](auto *) { return nullptr; });
2286 }
2287
2288 /// Returns true if recipe \p Def can be safely handed for CSE.
2289 static bool canHandle(const VPSingleDefRecipe *Def) {
2290 // We can extend the list of handled recipes in the future,
2291 // provided we account for the data embedded in them while checking for
2292 // equality or hashing.
2294
2295 // The issue with (Insert|Extract)Value is that the index of the
2296 // insert/extract is not a proper operand in LLVM IR, and hence also not in
2297 // VPlan.
2298 if (!C || (!C->first && (C->second == Instruction::InsertValue ||
2299 C->second == Instruction::ExtractValue)))
2300 return false;
2301
2302 // Widened loads (including the EVL variant) are handled, as cse() only
2303 // reuses them within a block with no intervening memory write. Any other
2304 // memory access is rejected.
2305 if (Def->mayWriteToMemory())
2306 return false;
2307 return !Def->mayReadFromMemory() ||
2309 }
2310
2311 /// Hash the underlying data of \p Def.
2312 static unsigned getHashValue(const VPSingleDefRecipe *Def) {
2313 hash_code Result = hash_combine(
2314 Def->getVPRecipeID(), vputils::getOpcodeOrIntrinsicID(Def),
2315 getGEPSourceElementType(Def), Def->getScalarType(),
2317 if (auto *RFlags = dyn_cast<VPRecipeWithIRFlags>(Def))
2318 if (RFlags->hasPredicate())
2319 return hash_combine(Result, RFlags->getPredicate());
2320 if (auto *SIVSteps = dyn_cast<VPScalarIVStepsRecipe>(Def))
2321 return hash_combine(Result, SIVSteps->getInductionOpcode());
2322 // Fold in the separately stored consecutive flag. Alignment is left out and
2323 // handled by cse.
2324 if (auto *Load = dyn_cast<VPWidenMemoryRecipe>(Def))
2325 return hash_combine(Result, Load->isConsecutive());
2326 return Result;
2327 }
2328
2329 /// Check equality of underlying data of \p L and \p R.
2330 static bool isEqual(const VPSingleDefRecipe *L, const VPSingleDefRecipe *R) {
2331 if (L->getVPRecipeID() != R->getVPRecipeID() ||
2334 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2336 !equal(L->operands(), R->operands()))
2337 return false;
2340 "must have valid opcode info for both recipes");
2341 if (auto *LFlags = dyn_cast<VPRecipeWithIRFlags>(L))
2342 if (LFlags->hasPredicate() &&
2343 LFlags->getPredicate() !=
2344 cast<VPRecipeWithIRFlags>(R)->getPredicate())
2345 return false;
2346 if (auto *LSIV = dyn_cast<VPScalarIVStepsRecipe>(L))
2347 if (LSIV->getInductionOpcode() !=
2348 cast<VPScalarIVStepsRecipe>(R)->getInductionOpcode())
2349 return false;
2350 // Compare the separately stored consecutive flag. Alignment is left out and
2351 // handled by cse.
2352 if (auto *LL = dyn_cast<VPWidenMemoryRecipe>(L))
2353 if (LL->isConsecutive() != cast<VPWidenMemoryRecipe>(R)->isConsecutive())
2354 return false;
2355 // Phi recipes can only be equal if they are in the same VPBB, as they
2356 // implicitly depend on their predecessors.
2357 if (isa<VPWidenPHIRecipe>(L) && L->getParent() != R->getParent())
2358 return false;
2359 // Recipes in replicate regions implicitly depend on predicate. If either
2360 // recipe is in a replicate region, only consider them equal if both have
2361 // the same parent.
2362 const VPRegionBlock *RegionL = L->getRegion();
2363 const VPRegionBlock *RegionR = R->getRegion();
2364 if (((RegionL && RegionL->isReplicator()) ||
2365 (RegionR && RegionR->isReplicator())) &&
2366 L->getParent() != R->getParent())
2367 return false;
2368 return L->getScalarType() == R->getScalarType();
2369 }
2370};
2371} // end anonymous namespace
2372
2373/// Perform a common-subexpression-elimination of VPSingleDefRecipes on the \p
2374/// Plan.
2376 VPDominatorTree VPDT(Plan);
2378 // CSE map for widened loads. Must be cleared on recipes that may write to
2379 // memory, and at the end of each VPBB.
2381 LoadCSEMap;
2382
2384 Plan.getEntry());
2386 for (VPRecipeBase &R : *VPBB) {
2387 if (R.mayWriteToMemory())
2388 LoadCSEMap.clear();
2389 auto *Def = dyn_cast<VPSingleDefRecipe>(&R);
2390 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2391 continue;
2393 auto [It, Inserted] =
2394 (IsLoad ? LoadCSEMap : CSEMap).try_emplace(Def, Def);
2395 if (Inserted)
2396 continue;
2397 VPSingleDefRecipe *V = It->second;
2398 // V must dominate Def for a valid replacement.
2399 if (!VPDT.dominates(V->getParent(), VPBB))
2400 continue;
2401 if (IsLoad) {
2402 auto *EarlierLoad = cast<VPWidenMemoryRecipe>(V);
2403 auto *Load = cast<VPWidenMemoryRecipe>(Def);
2404 if (EarlierLoad->getAlign() < Load->getAlign()) {
2405 // Record Load as the candidate for subsequent loads, as it may be
2406 // reusable where EarlierLoad is not.
2407 It->second = Def;
2408 continue;
2409 }
2410 // Keep only metadata common to both loads on the survivor.
2411 EarlierLoad->intersect(*Load);
2412 }
2413 // Only keep flags present on both V and Def.
2414 if (auto *RFlags = dyn_cast<VPRecipeWithIRFlags>(V))
2415 RFlags->intersectFlags(*cast<VPRecipeWithIRFlags>(Def));
2416 Def->replaceAllUsesWith(V);
2417 }
2418 LoadCSEMap.clear();
2419 }
2420}
2421
2422/// Return true if we do not know how to (mechanically) hoist or sink a
2423/// non-memory or memory recipe \p R out of a loop region. When sinking, passing
2424/// \p Sinking = true ensures that assumes aren't sunk.
2426 VPBasicBlock *LastBB,
2427 bool Sinking = false) {
2428 if (!isa<VPReplicateRecipe>(R) || !R.mayReadOrWriteMemory() ||
2430 return vputils::cannotHoistOrSinkRecipe(R, Sinking);
2431
2432 // Check that the memory operation doesn't alias between FirstBB and LastBB.
2433 auto MemLoc = vputils::getMemoryLocation(R);
2434
2435 // TODO: Could make use of SinkStoreInfo::isNoAliasViaDistance by collecting
2436 // stores upfront, and constructing a full SinkStoreInfo.
2437 auto SinkInfo =
2438 Sinking ? std::make_optional(SinkStoreInfo(cast<VPReplicateRecipe>(R)))
2439 : std::nullopt;
2440
2441 return !MemLoc ||
2442 !canHoistOrSinkWithNoAliasCheck(*MemLoc, FirstBB, LastBB, SinkInfo);
2443}
2444
2445/// Move loop-invariant recipes out of the vector loop region in \p Plan.
2446static void licm(VPlan &Plan) {
2447 VPBasicBlock *Preheader = Plan.getVectorPreheader();
2448
2449 // Hoist any loop invariant recipes from the vector loop region to the
2450 // preheader. Preform a shallow traversal of the vector loop region, to
2451 // exclude recipes in replicate regions. Since the top-level blocks in the
2452 // vector loop region are guaranteed to execute if the vector pre-header is,
2453 // we don't need to check speculation safety.
2454 VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
2455 assert(Preheader->getSingleSuccessor() == LoopRegion &&
2456 "Expected vector prehader's successor to be the vector loop region");
2458 vp_depth_first_shallow(LoopRegion->getEntry()))) {
2459 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
2460 if (any_of(R.operands(), [](VPValue *Op) {
2461 return !Op->isDefinedOutsideLoopRegions();
2462 }))
2463 continue;
2464 if (cannotHoistOrSinkRecipe(R, LoopRegion->getEntryBasicBlock(),
2465 LoopRegion->getExitingBasicBlock()))
2466 continue;
2467 R.moveBefore(*Preheader, Preheader->end());
2468 }
2469 }
2470
2471#ifndef NDEBUG
2472 VPDominatorTree VPDT(Plan);
2473#endif
2474 // Sink recipes with no users inside the vector loop region if all users are
2475 // in the same exit block of the region.
2476 // TODO: Extend to sink recipes from inner loops.
2478 LoopRegion->getEntry());
2480 for (VPRecipeBase &R : make_early_inc_range(reverse(*VPBB))) {
2481 // TODO: Use R.definedValues() instead of casting to VPSingleDefRecipe to
2482 // support recipes with multiple defined values (e.g., interleaved loads).
2483 auto *Def = dyn_cast<VPSingleDefRecipe>(&R);
2484 if (!Def)
2485 continue;
2486
2487 if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
2488 assert(!RepR->isPredicated() &&
2489 "Expected prior transformation of predicated replicates to "
2490 "replicate regions");
2491 // narrowToSingleScalarRecipes should have already maximally narrowed
2492 // replicates to single-scalar replicates.
2493 // TODO: When unrolling, replicateByVF doesn't handle sunk
2494 // non-single-scalar replicates correctly.
2495 if (!RepR->isSingleScalar())
2496 continue;
2497
2498 // The pointer operand of stores must be loop-invariant.
2499 if (RepR->getOpcode() == Instruction::Store &&
2500 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2501 continue;
2502 }
2503
2504 // Cannot sink the recipe if the user is defined in a loop region or a
2505 // non-successor of the vector loop region. Cannot sink if user is a phi
2506 // either.
2507 VPBasicBlock *SinkBB = nullptr;
2508 if (any_of(Def->users(), [&SinkBB, &LoopRegion](VPUser *U) {
2509 auto *UserR = cast<VPRecipeBase>(U);
2510 VPBasicBlock *Parent = UserR->getParent();
2511 // TODO: Support sinking when users are in multiple blocks.
2512 if (SinkBB && SinkBB != Parent)
2513 return true;
2514 SinkBB = Parent;
2515 // TODO: If the user is a PHI node, we should check the block of
2516 // incoming value. Support PHI node users if needed.
2517 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2518 Parent->getSinglePredecessor() != LoopRegion;
2519 }))
2520 continue;
2521
2522 if (cannotHoistOrSinkRecipe(R, LoopRegion->getEntryBasicBlock(),
2523 LoopRegion->getExitingBasicBlock(),
2524 /*Sinking=*/true))
2525 continue;
2526
2527 [[maybe_unused]] auto *RepR = dyn_cast<VPReplicateRecipe>(&R);
2528 assert((!R.mayWriteToMemory() ||
2529 (RepR && RepR->getOpcode() == Instruction::Store &&
2530 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2531 "The only recipes that may write to memory are expected to be "
2532 "stores with invariant pointer-operand");
2533
2534 if (!SinkBB)
2535 SinkBB = cast<VPBasicBlock>(LoopRegion->getSingleSuccessor());
2536
2537 // TODO: This will need to be a check instead of a assert after
2538 // conditional branches in vectorized loops are supported.
2539 assert(VPDT.properlyDominates(VPBB, SinkBB) &&
2540 "Defining block must dominate sink block");
2541 // TODO: Clone the recipe if users are on multiple exit paths, instead of
2542 // just moving.
2543 Def->moveBefore(*SinkBB, SinkBB->getFirstNonPhi());
2544 }
2545 }
2546}
2547
2549 VPlan &Plan, const MapVector<Instruction *, uint64_t> &MinBWs) {
2550 if (Plan.hasScalarVFOnly())
2551 return;
2552 // Keep track of created truncates, so they can be re-used. Note that we
2553 // cannot use RAUW after creating a new truncate, as this would could make
2554 // other uses have different types for their operands, making them invalidly
2555 // typed.
2557 VPBasicBlock *PH = Plan.getVectorPreheader();
2560 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
2563 continue;
2564
2565 VPValue *ResultVPV = R.getVPSingleValue();
2566 auto *UI = cast_or_null<Instruction>(ResultVPV->getUnderlyingValue());
2567 unsigned NewResSizeInBits = MinBWs.lookup(UI);
2568 if (!NewResSizeInBits)
2569 continue;
2570
2571 // If the value wasn't vectorized, we must maintain the original scalar
2572 // type. Skip those here, after incrementing NumProcessedRecipes. Also
2573 // skip casts which do not need to be handled explicitly here, as
2574 // redundant casts will be removed during recipe simplification.
2576 continue;
2577
2578 Type *OldResTy = ResultVPV->getScalarType();
2579 unsigned OldResSizeInBits = OldResTy->getScalarSizeInBits();
2580 assert(OldResTy->isIntegerTy() && "only integer types supported");
2581 (void)OldResSizeInBits;
2582
2583 auto *NewResTy = IntegerType::get(Plan.getContext(), NewResSizeInBits);
2584
2585 // Any wrapping introduced by shrinking this operation shouldn't be
2586 // considered undefined behavior. So, we can't unconditionally copy
2587 // arithmetic wrapping flags to VPW.
2588 if (auto *VPW = dyn_cast<VPRecipeWithIRFlags>(&R))
2589 VPW->dropPoisonGeneratingFlags();
2590
2591 assert((OldResSizeInBits != NewResSizeInBits ||
2592 match(&R, m_ICmp(m_VPValue(), m_VPValue()))) &&
2593 "Only ICmps should not need extending the result.");
2594 assert(!isa<VPWidenStoreRecipe>(&R) && "stores cannot be narrowed");
2595
2596 // Loads/intrinsics are not recreated; they keep producing their original
2597 // wide result and narrowed users will truncate it as needed below.
2599 continue;
2600
2601 // Shrink operands by introducing truncates as needed.
2602 unsigned StartIdx =
2603 match(&R, m_Select(m_VPValue(), m_VPValue(), m_VPValue())) ? 1 : 0;
2604 SmallVector<VPValue *> NewOperands(R.operands());
2605 for (VPValue *&Op : drop_begin(NewOperands, StartIdx)) {
2606 unsigned OpSizeInBits = Op->getScalarType()->getScalarSizeInBits();
2607 if (OpSizeInBits == NewResSizeInBits)
2608 continue;
2609 assert(OpSizeInBits > NewResSizeInBits && "nothing to truncate");
2610 auto [ProcessedIter, Inserted] = ProcessedTruncs.try_emplace(Op);
2611 if (Inserted) {
2612 VPBuilder Builder;
2613 if (isa<VPIRValue>(Op))
2614 Builder.setInsertPoint(PH);
2615 else
2616 Builder.setInsertPoint(&R);
2617 ProcessedIter->second =
2618 Builder.createWidenCast(Instruction::Trunc, Op, NewResTy);
2619 }
2620 Op = ProcessedIter->second;
2621 }
2622
2623 auto *NWR = cast<VPWidenRecipe>(&R)->cloneWithOperands(NewOperands);
2624 NWR->insertBefore(&R);
2625
2626 // Wrap NWR in a ZExt to preserve the original wide type for downstream
2627 // users. Not needed for ICmps, whose result type is i1 irrespective of
2628 // the narrowing of their operands.
2629 VPValue *Replacement = NWR->getVPSingleValue();
2630 if (Replacement->getScalarType() != OldResTy)
2631 Replacement =
2633 .createWidenCast(Instruction::ZExt, Replacement, OldResTy)
2634 ->getVPSingleValue();
2635 ResultVPV->replaceAllUsesWith(Replacement);
2636 R.eraseFromParent();
2637 }
2638 }
2639}
2640
2641bool VPlanTransforms::removeBranchOnConst(VPlan &Plan, bool OnlyLatches) {
2642 std::optional<VPDominatorTree> VPDT;
2643 if (OnlyLatches)
2644 VPDT.emplace(Plan);
2645
2646 // Collect all blocks before modifying the CFG so we can identify unreachable
2647 // ones after constant branch removal.
2649
2650 bool SimplifiedPhi = false;
2651 for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(AllBlocks)) {
2652 VPValue *Cond;
2653 // Skip blocks that are not terminated by BranchOnCond.
2654 if (VPBB->empty() || !match(&VPBB->back(), m_BranchOnCond(m_VPValue(Cond))))
2655 continue;
2656
2657 if (OnlyLatches && !VPBlockUtils::isLatch(VPBB, *VPDT))
2658 continue;
2659
2660 assert(VPBB->getNumSuccessors() == 2 &&
2661 "Two successors expected for BranchOnCond");
2662 unsigned RemovedIdx;
2663 if (match(Cond, m_True()))
2664 RemovedIdx = 1;
2665 else if (match(Cond, m_False()))
2666 RemovedIdx = 0;
2667 else
2668 continue;
2669
2670 VPBasicBlock *RemovedSucc =
2671 cast<VPBasicBlock>(VPBB->getSuccessors()[RemovedIdx]);
2672 assert(count(RemovedSucc->getPredecessors(), VPBB) == 1 &&
2673 "There must be a single edge between VPBB and its successor");
2674 // Values coming from VPBB into phi recipes of RemovedSucc are removed from
2675 // these recipes and single-entry header phis are removed.
2676 for (VPRecipeBase &R : make_early_inc_range(RemovedSucc->phis())) {
2677 cast<VPPhiAccessors>(&R)->removeIncomingValueFor(VPBB);
2678 SimplifiedPhi = true;
2679 // Remove now invalid header phis that are left single-entry after
2680 // removing their backedges.
2681 auto *PhiR = dyn_cast<VPHeaderPHIRecipe>(&R);
2682 if (!PhiR || PhiR->getNumIncoming() != 1)
2683 continue;
2684 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
2685 PhiR->eraseFromParent();
2686 }
2687
2688 // Disconnect blocks and remove the terminator.
2689 VPBlockUtils::disconnectBlocks(VPBB, RemovedSucc);
2690 VPBB->back().eraseFromParent();
2691 }
2692
2693 // Compute which blocks are still reachable from the entry after constant
2694 // branch removal.
2697
2698 // Detach all unreachable blocks from their successors, removing their recipes
2699 // and incoming values from phi recipes.
2700 VPSymbolicValue Tmp(nullptr);
2701 for (VPBlockBase *B : AllBlocks) {
2702 if (Reachable.contains(B))
2703 continue;
2704 for (VPBlockBase *Succ : to_vector(B->successors())) {
2705 if (auto *SuccBB = dyn_cast<VPBasicBlock>(Succ))
2706 for (VPRecipeBase &R : SuccBB->phis())
2707 cast<VPPhiAccessors>(&R)->removeIncomingValueFor(B);
2709 }
2710 for (VPBasicBlock *DeadBB :
2712 for (VPRecipeBase &R : make_early_inc_range(*DeadBB)) {
2713 for (VPValue *Def : R.definedValues())
2714 Def->replaceAllUsesWith(&Tmp);
2715 R.eraseFromParent();
2716 }
2717 }
2718 }
2719 return SimplifiedPhi;
2720}
2721
2742
2745 auto GetSimplifiedLiveInViaSCEV = [&](VPValue *VPV) -> VPValue * {
2746 const SCEV *Expr = vputils::getSCEVExprForVPValue(VPV, PSE);
2747 const APInt *C;
2748 if (match(Expr, m_scev_APInt(C)))
2749 return Plan.getConstantInt(*C);
2750 return nullptr;
2751 };
2752
2753 for (VPValue *LiveIn : to_vector(Plan.getLiveIns())) {
2754 if (VPValue *SimplifiedLiveIn = GetSimplifiedLiveInViaSCEV(LiveIn))
2755 LiveIn->replaceAllUsesWith(SimplifiedLiveIn);
2756 }
2757}
2758
2760 VPlan &Plan, PredicatedScalarEvolution &PSE,
2761 const SymbolicStrideMap &StridesMap, const VPDominatorTree &VPDT) {
2762 // Replace VPValues for known constant strides guaranteed by predicated scalar
2763 // evolution that are guaranteed to be guarded by the runtime checks; that is,
2764 // blocks dominated by the vector header.
2765 assert(!Plan.getVectorLoopRegion() &&
2766 "expected to run before loop regions are created");
2767 const auto &[Header, _] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
2768 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](VPUser &U,
2769 unsigned Idx) {
2770 auto *R = cast<VPRecipeBase>(&U);
2771 // Skip phis if the loop if loop is not yet guarded.
2772 if (isa<VPPhiAccessors>(R) &&
2773 Header == Plan.getEntry()->getSingleSuccessor())
2774 return false;
2775 return VPDT.dominates(Header, R->getParent());
2776 };
2777 ValueToSCEVMapTy RewriteMap;
2778 for (const SCEVUnknown *Stride : StridesMap.values()) {
2779 Value *StrideV = Stride->getValue();
2780 const APInt *StrideConst;
2781 const SCEV *StrideExpr = PSE.getSCEV(StrideV);
2782 if (!match(StrideExpr, m_scev_APInt(StrideConst)))
2783 // Only handle constant strides for now.
2784 continue;
2785 if (VPValue *StrideVPV = Plan.getLiveIn(StrideV))
2786 StrideVPV->replaceUsesWithIf(Plan.getConstantInt(*StrideConst),
2787 CanUseVersionedStride);
2788
2789 // The versioned value may not be used in the loop directly but through an
2790 // integral cast (sext/zext/trunc). Add new live-ins in those cases.
2791 for (Value *U : StrideV->users()) {
2793 continue;
2794 VPValue *StrideVPV = Plan.getLiveIn(U);
2795 if (!StrideVPV)
2796 continue;
2797 unsigned BW = U->getType()->getScalarSizeInBits();
2798 APInt C = isa<SExtInst>(U) ? StrideConst->sext(BW)
2799 : StrideConst->zextOrTrunc(BW);
2800 StrideVPV->replaceUsesWithIf(Plan.getConstantInt(C),
2801 CanUseVersionedStride);
2802 }
2803 RewriteMap[StrideV] = StrideExpr;
2804 }
2805
2806 for (VPExpandSCEVRecipe &ExpSCEV :
2808 const SCEV *ScevExpr = ExpSCEV.getSCEV();
2809 auto *NewSCEV =
2810 SCEVParameterRewriter::rewrite(ScevExpr, *PSE.getSE(), RewriteMap);
2811 if (NewSCEV != ScevExpr) {
2812 VPValue *NewExp = vputils::getOrCreateVPValueForSCEVExpr(Plan, NewSCEV);
2813 ExpSCEV.replaceAllUsesWith(NewExp);
2814 if (Plan.getTripCount() == &ExpSCEV)
2815 Plan.resetTripCount(NewExp);
2816 }
2817 }
2818}
2819
2821 // Collect recipes in the backward slice of `Root` that may generate a poison
2822 // value that is used after vectorization.
2824 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](VPRecipeBase *Root) {
2826 Worklist.push_back(Root);
2827
2828 // Traverse the backward slice of Root through its use-def chain.
2829 while (!Worklist.empty()) {
2830 VPRecipeBase *CurRec = Worklist.pop_back_val();
2831
2832 if (!Visited.insert(CurRec).second)
2833 continue;
2834
2835 // Prune search if we find another recipe generating a widen memory
2836 // instruction. Widen memory instructions involved in address computation
2837 // will lead to gather/scatter instructions, which don't need to be
2838 // handled.
2840 VPHeaderPHIRecipe>(CurRec))
2841 continue;
2842
2843 // This recipe contributes to the address computation of a widen
2844 // load/store. If the underlying instruction has poison-generating flags,
2845 // drop them directly.
2846 if (auto *RecWithFlags = dyn_cast<VPRecipeWithIRFlags>(CurRec)) {
2847 VPValue *A, *B;
2848 // Dropping disjoint from an OR may yield incorrect results, as some
2849 // analysis may have converted it to an Add implicitly (e.g. SCEV used
2850 // for dependence analysis). Instead, replace it with an equivalent Add.
2851 // This is possible as all users of the disjoint OR only access lanes
2852 // where the operands are disjoint or poison otherwise.
2853 if (match(RecWithFlags, m_BinaryOr(m_VPValue(A), m_VPValue(B))) &&
2854 RecWithFlags->isDisjoint()) {
2855 VPBuilder Builder(RecWithFlags);
2856 VPInstruction *New =
2857 Builder.createAdd(A, B, RecWithFlags->getDebugLoc());
2858 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2859 RecWithFlags->replaceAllUsesWith(New);
2860 RecWithFlags->eraseFromParent();
2861 CurRec = New;
2862 } else
2863 RecWithFlags->dropPoisonGeneratingFlags();
2864 } else {
2867 (void)Instr;
2868 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2869 "found instruction with poison generating flags not covered by "
2870 "VPRecipeWithIRFlags");
2871 }
2872
2873 // Add new definitions to the worklist.
2874 for (VPValue *Operand : CurRec->operands())
2875 if (VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2876 Worklist.push_back(OpDef);
2877 }
2878 });
2879
2880 // We want to exclude the tail folding case, as we don't need to drop flags
2881 // for operations computing the first lane in this case: the first lane of the
2882 // header mask must always be true. For reverse memory accesses, the mask is
2883 // wrapped in a Reverse, which is just a permutation of the header mask, so
2884 // peel it off before checking. The header mask is still the abstract region
2885 // value at this point (materialization happens later).
2886 auto m_UnlessHdrMask = m_Unless( // NOLINT
2888
2889 // Traverse all the recipes in the VPlan and collect the poison-generating
2890 // recipes in the backward slice starting at the address of a VPWidenRecipe or
2891 // VPInterleaveRecipe.
2892 auto Iter =
2895 for (VPRecipeBase &Recipe : *VPBB) {
2896 if (auto *WidenRec = dyn_cast<VPWidenMemoryRecipe>(&Recipe)) {
2897 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2898 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2899 match(WidenRec->getMask(), m_UnlessHdrMask))
2900 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2901 } else if (auto *InterleaveRec = dyn_cast<VPInterleaveRecipe>(&Recipe)) {
2902 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2903 if (AddrDef && InterleaveRec->getMask() &&
2904 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2905 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2906 }
2907 }
2908 }
2909}
2910
2912 VPlan &Plan,
2914 &InterleaveGroups,
2915 const bool &EpilogueAllowed) {
2916 if (InterleaveGroups.empty())
2917 return;
2918
2920 for (VPBasicBlock *VPBB :
2923 for (VPRecipeBase &R : make_filter_range(*VPBB, [](VPRecipeBase &R) {
2924 return isa<VPWidenMemoryRecipe>(&R);
2925 })) {
2926 auto *MemR = cast<VPWidenMemoryRecipe>(&R);
2927 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2928 }
2929
2930 // Interleave memory: for each Interleave Group we marked earlier as relevant
2931 // for this VPlan, replace the Recipes widening its memory instructions with a
2932 // single VPInterleaveRecipe at its insertion point.
2933 VPDominatorTree VPDT(Plan);
2934 for (const auto *IG : InterleaveGroups) {
2935 VPWidenMemoryRecipe *Start = nullptr;
2936 Instruction *StartMember = nullptr;
2937 for (auto *Member : IG->members())
2938 if (VPWidenMemoryRecipe *R = IRMemberToRecipe.lookup(Member)) {
2939 StartMember = Member;
2940 Start = R;
2941 break;
2942 }
2943 if (!StartMember) // All member recipes are dead, so the group is dead.
2944 continue;
2945 VPIRMetadata InterleaveMD(*Start);
2946 SmallVector<VPValue *, 4> StoredValues;
2947 for (unsigned I = 0; I < IG->getFactor(); ++I) {
2948 Instruction *MemberI = IG->getMember(I);
2949 if (!MemberI)
2950 continue;
2951 if (VPWidenMemoryRecipe *MemoryR = IRMemberToRecipe.lookup(MemberI)) {
2952 if (auto *StoreR = dyn_cast<VPWidenStoreRecipe>(MemoryR->getAsRecipe()))
2953 StoredValues.push_back(StoreR->getStoredValue());
2954 InterleaveMD.intersect(*MemoryR);
2955 } else {
2956 InterleaveMD.intersect(VPIRMetadata(*MemberI));
2957 }
2958 }
2959
2960 bool NeedsMaskForGaps =
2961 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2962 (!StoredValues.empty() && !IG->isFull());
2963
2964 Instruction *IRInsertPos = IG->getInsertPos();
2965 auto *InsertPos = IRMemberToRecipe.lookup(IRInsertPos);
2966 if (!InsertPos) {
2967 // InsertPos member is dead: find a new member that is alive.
2968 assert(isa<VPWidenLoadRecipe>(Start->getAsRecipe()) &&
2969 "Dead member in non-load group?");
2970 InsertPos = Start;
2971 for (Instruction *Member : IG->members())
2972 if (VPWidenMemoryRecipe *MemberR = IRMemberToRecipe.lookup(Member))
2973 if (VPDT.properlyDominates(MemberR->getAsRecipe(),
2974 InsertPos->getAsRecipe()))
2975 InsertPos = MemberR;
2976 IRInsertPos = &InsertPos->getIngredient();
2977 }
2978 VPRecipeBase *InsertPosR = InsertPos->getAsRecipe();
2979
2981 if (auto *Gep = dyn_cast<GetElementPtrInst>(
2982 getLoadStorePointerOperand(IRInsertPos)->stripPointerCasts()))
2983 NW = Gep->getNoWrapFlags().withoutNoUnsignedWrap();
2984
2985 // Get or create the start address for the interleave group.
2986 VPValue *Addr = Start->getAddr();
2987 VPRecipeBase *AddrDef = Addr->getDefiningRecipe();
2988 if (IG->getIndex(StartMember) != 0 ||
2989 (AddrDef && !VPDT.properlyDominates(AddrDef, InsertPosR))) {
2990 // Either member zero's recipe is dead, or we cannot re-use the address of
2991 // member zero because it does not dominate the insert position. Instead,
2992 // use the address of the insert position and create a PtrAdd adjusting it
2993 // to the address of member zero.
2994 // TODO: Hoist Addr's defining recipe (and any operands as needed) to
2995 // InsertPos or sink loads above zero members to join it.
2996 assert(IG->getIndex(IRInsertPos) != 0 &&
2997 "index of insert position shouldn't be zero");
2998 auto &DL = IRInsertPos->getDataLayout();
2999 APInt Offset(32,
3000 DL.getTypeAllocSize(getLoadStoreType(IRInsertPos)) *
3001 IG->getIndex(IRInsertPos),
3002 /*IsSigned=*/true);
3003 VPValue *OffsetVPV = Plan.getConstantInt(-Offset);
3004 VPBuilder B(InsertPosR);
3005 Addr = B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
3006 }
3007 // If the group is reverse, adjust the index to refer to the last vector
3008 // lane instead of the first. We adjust the index from the first vector
3009 // lane, rather than directly getting the pointer for lane VF - 1, because
3010 // the pointer operand of the interleaved access is supposed to be uniform.
3011 if (IG->isReverse()) {
3012 auto *ReversePtr = new VPVectorEndPointerRecipe(
3013 Addr, &Plan.getVF(), getLoadStoreType(IRInsertPos),
3014 -(int64_t)IG->getFactor(), NW, InsertPosR->getDebugLoc());
3015 ReversePtr->insertBefore(InsertPosR);
3016 Addr = ReversePtr;
3017 }
3018 auto *VPIG = new VPInterleaveRecipe(
3019 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
3020 InterleaveMD, InsertPosR->getDebugLoc());
3021 VPIG->insertBefore(InsertPosR);
3022
3023 unsigned J = 0;
3024 for (unsigned i = 0; i < IG->getFactor(); ++i)
3025 if (Instruction *Member = IG->getMember(i)) {
3026 VPWidenMemoryRecipe *MemberR = IRMemberToRecipe.lookup(Member);
3027 if (!Member->getType()->isVoidTy()) {
3028 if (MemberR) {
3029 VPValue *OriginalV = MemberR->getAsRecipe()->getVPSingleValue();
3030 OriginalV->replaceAllUsesWith(VPIG->getVPValue(J));
3031 }
3032 J++;
3033 }
3034 if (MemberR)
3035 MemberR->getAsRecipe()->eraseFromParent();
3036 }
3037 }
3038}
3039
3040/// Matches an exit condition formed by comparing a value loaded from memory
3041/// with a loop-invariant term. Binds the comparison for the condition.
3047
3048namespace {
3049struct CountableConditionMatch {
3050 VPValue *&Cmp;
3051 PredicatedScalarEvolution &PSE;
3052 Loop *L;
3053
3054 CountableConditionMatch(VPValue *&Cmp, PredicatedScalarEvolution &PSE,
3055 Loop *L)
3056 : Cmp(Cmp), PSE(PSE), L(L) {}
3057
3058 template <typename ITy> bool match(ITy *V) const {
3059 VPValue *Update;
3061 V, m_VPValue(Cmp, m_c_ICmp(m_VPValue(Update, m_Add(m_VPValue(),
3062 m_VPValue())),
3063 m_LiveIn()))))
3064 return false;
3065
3066 const SCEV *S = vputils::getSCEVExprForVPValue(Update, PSE, L);
3069 }
3070};
3071} // end anonymous namespace
3072
3073/// Matches an exit condition formed by comparing the current value of a
3074/// affine add recurrence in the given loop with a stride of 1 against a
3075/// loop-invariant term. Binds the comparison for the condition.
3077 Loop *L) {
3078 return CountableConditionMatch(Cmp, PSE, L);
3079}
3080
3083 Loop *L) {
3084 // Check for a single combined exit in the latch block.
3085 // TODO: Generalize to other blocks besides the latch.
3086 // If we don't find a combined condition in the latch, just return true
3087 // to proceed with vectorization.
3088 auto [_, LatchVPBB] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
3089
3090 // We're looking for a conditional branch...
3091 auto *Term = dyn_cast<VPInstruction>(LatchVPBB->getTerminator());
3092 if (!Term || Term->getOpcode() != VPInstruction::BranchOnCond)
3093 return true;
3094
3095 // ...where the condition is a combination of both a countable and an
3096 // uncountable comparison.
3097 VPValue *Uncountable = nullptr;
3098 VPValue *Countable = nullptr;
3099 VPValue *Cond = Term->getOperand(0);
3101 m_c_LogicalOr(m_Uncountable(Uncountable),
3102 m_Countable(Countable, PSE, L)),
3103 m_c_BinaryOr(m_Uncountable(Uncountable),
3104 m_Countable(Countable, PSE, L))))))
3105 return true;
3106
3107 // If the conditions are combined with a logical or (select), then we'll
3108 // need to freeze the individual terms when splitting.
3109 bool NeedsFreeze = match(Cond, m_LogicalOr(m_VPValue(), m_VPValue()));
3110
3111 // If we do have a combined exit condition, bail out if there's more than
3112 // one exit block.
3113 // TODO: Support additional exits.
3114 ArrayRef<VPIRBasicBlock *> ExitBlocks = Plan.getExitBlocks();
3115 if (ExitBlocks.size() != 1)
3116 return false;
3117
3118 // If there are any live-outs, bail out. The exit block is an existing IR
3119 // block, and if we split the exiting block then the incoming blocks and
3120 // values won't be correct.
3121 // TODO: Support live-outs with combined exits.
3122 if (!ExitBlocks.front()->phis().empty())
3123 return false;
3124
3125 // Split the latch block just before the terminator.
3126 VPBasicBlock *NewLatch = LatchVPBB->splitAt(Term->getIterator());
3127
3128 // Create new terminator for uncountable condition.
3129 VPBuilder EEBuilder(LatchVPBB);
3130 if (NeedsFreeze)
3131 Uncountable = EEBuilder.createFreeze(Uncountable);
3132 EEBuilder.createNaryOp(VPInstruction::BranchOnCond, {Uncountable});
3133
3134 // We need to connect the uncountable exit to the sole exit block. The
3135 // latch is expected to connect to the middle block instead.
3136 // In canonical form, the backedge is the last successor for the latch. So
3137 // the first successor (true path) should be the exit for both conditions.
3138 LatchVPBB->clearSuccessors();
3139 NewLatch->clearPredecessors();
3140 VPBlockUtils::connectBlocks(LatchVPBB, ExitBlocks.front());
3141 VPBlockUtils::connectBlocks(LatchVPBB, NewLatch);
3142
3143 // Set condition for latch block to countable condition.
3144 if (NeedsFreeze) {
3145 VPBuilder NewLatchBuilder(Term);
3146 Countable = NewLatchBuilder.createFreeze(Countable);
3147 }
3148 Term->setOperand(0, Countable);
3149
3150 // Remove the combining or.
3151 cast<VPInstruction>(Cond)->eraseFromParent();
3152
3153 return true;
3154}
3155
3156/// Returns the VPValue representing the uncountable exit comparison used by
3157/// AnyOf if the recipes it depends on can be traced back to live-ins and
3158/// the addresses (in GEP/PtrAdd form) of any (non-masked) load used in
3159/// generating the values for the comparison. The recipes are stored in
3160/// \p Recipes.
3161static VPValue *
3163 VPBasicBlock *LatchVPBB) {
3164 // Given a plain CFG VPlan loop with countable latch exiting block
3165 // \p LatchVPBB, we're looking to match the recipes contributing to the
3166 // uncountable exit condition comparison (here, vp<%4>) back to either
3167 // live-ins or the address nodes for the load used as part of the uncountable
3168 // exit comparison so that we can either move them within the loop, or copy
3169 // them to the preheader depending on the chosen method for dealing with
3170 // stores in uncountable exit loops.
3171 //
3172 // Currently, the address of the load is restricted to a GEP with 2 operands
3173 // and a live-in base address. This constraint may be relaxed later.
3174 //
3175 // VPlan ' for UF>=1' {
3176 // Live-in vp<%0> = VF * UF
3177 // Live-in vp<%1> = vector-trip-count
3178 // Live-in ir<20> = original trip-count
3179 //
3180 // ir-bb<entry>:
3181 // Successor(s): scalar.ph, vector.ph
3182 //
3183 // vector.ph:
3184 // Successor(s): for.body
3185 //
3186 // for.body:
3187 // EMIT vp<%2> = phi ir<0>, vp<%index.next>
3188 // EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, for.inc ]
3189 // EMIT ir<%uncountable.addr> = getelementptr inbounds nuw ir<%pred>,ir<%iv>
3190 // EMIT ir<%uncountable.val> = load ir<%uncountable.addr>
3191 // EMIT ir<%uncountable.cond> = icmp sgt ir<%uncountable.val>, ir<500>
3192 // EMIT vp<%3> = masked-cond ir<%uncountable.cond>
3193 // Successor(s): for.inc
3194 //
3195 // for.inc:
3196 // EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
3197 // EMIT ir<%countable.cond> = icmp eq ir<%iv.next>, ir<20>
3198 // EMIT vp<%index.next> = add nuw vp<%2>, vp<%0>
3199 // EMIT vp<%freeze> = freeze ir<%3>
3200 // EMIT vp<%4> = any-of ir<%freeze>
3201 // EMIT vp<%5> = icmp eq vp<%index.next>, vp<%1>
3202 // EMIT branch-on-two-conds vp<%4>, vp<%5>
3203 // Successor(s): middle.block, middle.block, for.body
3204 //
3205 // middle.block:
3206 // Successor(s): ir-bb<exit>, scalar.ph
3207 //
3208 // ir-bb<exit>:
3209 // No successors
3210 //
3211 // scalar.ph:
3212 // }
3213
3214 // Find the uncountable loop exit condition.
3215 VPValue *UncountableCondition = nullptr;
3216 if (!match(LatchVPBB->getTerminator(),
3217 m_BranchOnTwoConds(m_AnyOf(m_VPValue(UncountableCondition)),
3218 m_VPValue())))
3219 return nullptr;
3220
3222 Worklist.push_back(UncountableCondition);
3223 while (!Worklist.empty()) {
3224 VPValue *V = Worklist.pop_back_val();
3225
3226 // Any value defined outside the loop does not need to be copied.
3227 if (V->isDefinedOutsideLoopRegions())
3228 continue;
3229
3230 // FIXME: Remove the single user restriction; it's here because we're
3231 // starting with the simplest set of loops we can, and multiple
3232 // users means needing to add PHI nodes in the transform.
3233 if (V->getNumUsers() > 1)
3234 return nullptr;
3235
3236 VPValue *Op1, *Op2;
3237 // Walk back through recipes until we find at least one load from memory.
3238 if (match(V, m_ICmp(m_VPValue(Op1), m_VPValue(Op2)))) {
3239 Worklist.push_back(Op1);
3240 Worklist.push_back(Op2);
3241 Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
3242 } else if (match(V, m_VPInstruction<Instruction::Load>(m_VPValue(Op1)))) {
3243 VPRecipeBase *GepR = Op1->getDefiningRecipe();
3244 // Only matching base + single offset term for now.
3245 if (GepR->getNumOperands() != 2)
3246 return nullptr;
3247 // Matching a GEP with a loop-invariant base ptr.
3249 m_LiveIn(), m_VPValue())))
3250 return nullptr;
3251 Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
3252 Recipes.push_back(cast<VPInstruction>(GepR));
3253 } else if (match(V, m_Freeze(m_VPValue(Op1)))) {
3254 Worklist.push_back(Op1);
3255 Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
3257 m_VPValue(Op1)))) {
3258 Worklist.push_back(Op1);
3259 Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
3260 } else
3261 return nullptr;
3262 }
3263
3264 // If we couldn't match anything, don't return the condition. It may be
3265 // defined outside the loop.
3266 if (Recipes.empty() ||
3268 return nullptr;
3269
3270 return UncountableCondition;
3271}
3272
3278
3279/// Update \p Plan to mask memory operations in the loop based on whether the
3280/// early exit is taken or not.
3281///
3282/// We're currently expecting to find a loop with properties similar to the
3283/// following:
3284///
3285/// for.body:
3286/// ir<%indvars.iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<%0>
3287/// EMIT ir<%arrayidx> = getelementptr inbounds nuw ir<@c>, ir<%indvars.iv>
3288/// EMIT-SCALAR ir<%0> = load ir<%arrayidx>
3289/// EMIT ir<%cmp1> = icmp sgt ir<%0>, ir<5>
3290/// EMIT vp<%1> = masked-cond ir<%cmp1>
3291/// Successor(s): if.end
3292///
3293/// if.end:
3294/// EMIT ir<%arrayidx3> = getelementptr inbounds nuw ir<@src>, ir<%indvars.iv>
3295/// EMIT-SCALAR ir<%2> = load ir<%arrayidx3>
3296/// EMIT ir<%add> = add nsw ir<%2>, ir<42>
3297/// EMIT ir<%arrayidx5> = getelementptr inbounds nuw ir<@dst>, ir<%indvars.iv>
3298/// EMIT store ir<%add>, ir<%arrayidx5>
3299/// EMIT ir<%indvars.iv.next> = add nuw nsw ir<%indvars.iv>, ir<1>
3300/// EMIT vp<%freeze> = freeze ir<%1>
3301/// EMIT vp<%3> = any-of ir<%freeze>
3302/// EMIT ir<%exitcond.not> = icmp eq ir<%indvars.iv.next>, ir<10000>
3303/// EMIT branch-on-two-conds vp<%3>, ir<%exitcond.not>
3304/// Successor(s): middle.block, middle.block, for.body
3305///
3306/// We currently expect LoopVectorizationLegality to ensure that:
3307/// * There must also be a counted exit. We will need to support speculative
3308/// or first-faulting loads before we can remove this restriction.
3309/// * Any stores within the loop must not alias with the load used for the
3310/// uncountable exit. We can relax this a bit with runtime aliasing checks.
3311/// * Other memory operations in the loop can take place before or after the
3312/// uncountable exit, but must also be unconditional. We need to support
3313/// combining the conditions in VPlanPredicator.
3314/// * The loop must have a single unconditional load contributing to the
3315/// uncountable exit comparison, and the other term must be loop-invariant.
3316/// Improving upon this requires work in getRecipesForUncountableExit to
3317/// handle more complex recipe graphs.
3320 VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB, VPBasicBlock *MiddleVPBB,
3321 OptimizationRemarkEmitter *ORE, Loop *TheLoop,
3323
3324 // Disconnect early exiting blocks from successors, remove branches. We
3325 // currently don't support multiple uses for recipes involved in creating
3326 // the uncountable exit condition.
3327 for (auto &Exit : Exits) {
3328 if (Exit.EarlyExitingVPBB == LatchVPBB)
3329 continue;
3330
3331 for (VPRecipeBase &R : Exit.EarlyExitVPBB->phis())
3332 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3333 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3334 VPBlockUtils::disconnectBlocks(Exit.EarlyExitingVPBB, Exit.EarlyExitVPBB);
3335 }
3336
3337 VPDominatorTree VPDT(Plan);
3338
3339 // We can abandon a VPlan entirely if we return false here, so we shouldn't
3340 // crash if some earlier assumptions on scalar IR don't hold for the vplan
3341 // version of the loop.
3342 SmallVector<VPInstruction *, 8> ConditionRecipes;
3343
3344 VPValue *Cond = getRecipesForUncountableExit(ConditionRecipes, LatchVPBB);
3345 if (!Cond) {
3346 reportVectorizationFailure("Unable to determine early exit condition for "
3347 "loop with side effects",
3348 "EarlyExitSideEffectsCond", ORE, TheLoop);
3349 return false;
3350 }
3351
3352 // Find load contributing to condition.
3353 // At the moment LoopVectorizationLegality only supports a single
3354 // early-exit expression with a compare and a single load that must
3355 // be unconditional.
3356 // TODO: Support more than one load.
3357 auto *Load =
3358 find_singleton<VPInstruction>(ConditionRecipes, [](auto *I, bool _) {
3360 ? I
3361 : nullptr;
3362 });
3363 assert(Load && "Couldn't find exactly one load");
3364 // TODO: Support conditional loads for uncountable exits.
3365 assert(VPDT.dominates(Load->getParent(), LatchVPBB) &&
3366 "Uncountable exit condition load is conditional.");
3367 VPInstruction *Ptr = cast<VPInstruction>(Load->getOperand(0));
3368
3369 // Ensure that we are guaranteed to be able to dereference the memory used
3370 // for determining the uncountable exit for the maximum possible number of
3371 // scalar iterations of the loop.
3372 //
3373 // TODO: Support first-faulting loads in cases where we don't know whether
3374 // all possible addresses are dereferenceable.
3375 {
3377 const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, TheLoop);
3378 const DataLayout &DL = Plan.getDataLayout();
3379 APInt EltSize(DL.getIndexTypeSizeInBits(Ptr->getScalarType()),
3380 DL.getTypeStoreSize(Load->getScalarType()).getFixedValue());
3382 PtrSCEV, cast<LoadInst>(Load->getUnderlyingInstr())->getAlign(),
3383 PSE.getSE()->getConstant(EltSize), TheLoop, *PSE.getSE(), DT, AC,
3384 &Predicates)) {
3385 reportVectorizationFailure("Early exit loop with side effects contains "
3386 "load used by the exit condition that may "
3387 "fault",
3388 "EarlyExitSideEffectsFaultingLoad", ORE,
3389 TheLoop);
3390 return false;
3391 }
3392 }
3393
3394 // Check for a single GEP for the condition load to see if we can link it to
3395 // a widen IV recipe with a step of 1; we're only interested in contiguous
3396 // accesses for the condition load right now.
3397 auto *IV = cast<VPWidenInductionRecipe>(&HeaderVPBB->front());
3398 if (!match(IV->getStartValue(), m_SpecificInt(0)) ||
3399 !match(IV->getStepValue(), m_SpecificInt(1))) {
3400 reportVectorizationFailure("Early exit loop with side effects contains "
3401 "load used by the exit condition with an "
3402 "unsupported memory access pattern",
3403 "EarlyExitSideEffectsBadLoadAccessPattern", ORE,
3404 TheLoop);
3405 return false;
3406 }
3407
3409 m_LiveIn(), m_Specific(IV)))) {
3410 reportVectorizationFailure("Early exit loop with side effects contains "
3411 "load used by the exit condition with an "
3412 "unsupported memory access pattern",
3413 "EarlyExitSideEffectsBadLoadAccessPattern", ORE,
3414 TheLoop);
3415 return false;
3416 }
3417
3418 // We want to guarantee that the uncountable exit condition (and the mask
3419 // we will generate from it) are available for all operations in the loop
3420 // that need to be masked. If the condition recipes are not already the first
3421 // recipes in the header after the last phi, move them there.
3422 auto InsertIt = HeaderVPBB->getFirstNonPhi();
3423 while (InsertIt != HeaderVPBB->end() &&
3424 is_contained(ConditionRecipes, &*InsertIt)) {
3425 erase(ConditionRecipes, &*InsertIt);
3426 InsertIt++;
3427 }
3428 for (auto *Recipe : reverse(ConditionRecipes))
3429 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3430
3431 // Create a mask to represent all lanes that fully execute in the vector loop,
3432 // stopping short of any early exit.
3433 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3434 VPValue *FirstActive = MaskBuilder.createFirstActiveLane(Cond);
3435 Type *IVScalarTy = IV->getScalarType();
3436 VPValue *Zero = Plan.getZero(IVScalarTy);
3437 FirstActive =
3438 MaskBuilder.createScalarZExtOrTrunc(FirstActive, IVScalarTy, DebugLoc());
3440 {Zero, FirstActive}, DebugLoc(),
3441 "uncountable.exit.mask");
3442
3443 // Convert all other memory operations to use the mask.
3444 for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(HeaderVPBB))
3445 for (VPRecipeBase &R : *VPBB)
3446 if (R.mayReadOrWriteMemory() && &R != Load) {
3447 // TODO: Handle conditional memory operations in the loop.
3448 if (!VPDT.dominates(R.getParent(), LatchVPBB)) {
3450 "Early exit loop with side effects contains unsupported "
3451 "conditional memory operations",
3452 "EarlyExitSideEffectsUnsupportedConditionalMemOps", ORE, TheLoop);
3453 return false;
3454 }
3455 cast<VPInstruction>(&R)->addMask(Mask);
3456 }
3457
3458 // Update middle block branch to compare (IV + however many lanes were active)
3459 // against the full trip count, since we may be exiting the vector loop early.
3460 // If we didn't take an early exit, we should get the equivalent of VF from
3461 // the FirstActiveLane.
3462 assert(match(MiddleVPBB->getTerminator(), m_BranchOnCond()) &&
3463 "Expected BranchOnCond terminator for MiddleVPBB");
3464 VPBuilder MiddleBuilder(MiddleVPBB->getTerminator());
3465 VPValue *ScalarIV = MiddleBuilder.createNaryOp(VPInstruction::ExtractLane,
3466 {Zero, IV}, DebugLoc());
3467 VPValue *ExitIV = MiddleBuilder.createAdd(ScalarIV, FirstActive);
3468 VPValue *FullTC =
3469 MiddleBuilder.createICmp(CmpInst::ICMP_EQ, ExitIV, Plan.getTripCount());
3470 MiddleVPBB->getTerminator()->setOperand(0, FullTC);
3471
3472 // Update resume phi in scalar.ph.
3473 VPBasicBlock *ScalarPH = Plan.getScalarPreheader();
3474 auto Phis = ScalarPH->phis();
3475 // TODO: Handle more than one Phi; re-derive from IV.
3476 // TODO: Handle reductions.
3477 if (range_size(Phis) != 1) {
3479 "Early exit loop with side effects contains "
3480 "unsupported reductions, inductions or recurrences",
3481 "EarlyExitSideEffectsReductions", ORE, TheLoop);
3482 return false;
3483 }
3484 VPPhi *ContinueIV = cast<VPPhi>(Phis.begin());
3485 // Make sure we're referring to the same IV.
3486 assert(
3487 match(ContinueIV->getOperand(0),
3489 "Continuing from different IV");
3490 ContinueIV->setOperand(0, ExitIV);
3491 return true;
3492}
3493
3495 VPlan &Plan, OptimizationRemarkEmitter *ORE, Loop *TheLoop,
3497 UncountableExitStyle Style) {
3498#ifndef NDEBUG
3499 VPDominatorTree VPDT(Plan);
3500#endif
3501
3502 auto *MiddleVPBB = VPBlockUtils::getPlainCFGMiddleBlock(Plan);
3503 auto [HeaderVPBB, LatchVPBB] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
3504
3505 // Dereferenceability is checked separately for uncountable exit loops with
3506 // stores, as only the loads contributing to the exit condition need to
3507 // be checked.
3508 if (Style == UncountableExitStyle::ReadOnly &&
3509 !areAllLoadsDereferenceable(HeaderVPBB, TheLoop, PSE, DT, AC)) {
3511 "Auto-vectorization of early exit loops with potentially "
3512 "faulting loads is not supported",
3513 "EarlyExitFaultingLoads", ORE, TheLoop);
3514 return false;
3515 }
3516
3517 VPBuilder LatchBuilder(LatchVPBB->getTerminator());
3519 for (auto [EarlyExitingVPBB, ExitBlock] :
3520 vputils::getEarlyExits(Plan, MiddleVPBB)) {
3521 // Collect condition for this early exit.
3522 VPBlockBase *TrueSucc = EarlyExitingVPBB->getSuccessors()[0];
3523 VPValue *CondOfEarlyExitingVPBB;
3524 [[maybe_unused]] bool Matched =
3525 match(EarlyExitingVPBB->getTerminator(),
3526 m_BranchOnCond(m_VPValue(CondOfEarlyExitingVPBB)));
3527 assert(Matched && "Terminator must be BranchOnCond");
3528
3529 // Insert the MaskedCond in the EarlyExitingVPBB so the predicator adds
3530 // the correct block mask.
3531 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3532 auto *CondToEarlyExit = EarlyExitingBuilder.createNaryOp(
3534 TrueSucc == ExitBlock
3535 ? CondOfEarlyExitingVPBB
3536 : EarlyExitingBuilder.createNot(CondOfEarlyExitingVPBB));
3537 assert((isa<VPIRValue>(CondOfEarlyExitingVPBB) ||
3538 !VPDT.properlyDominates(EarlyExitingVPBB, LatchVPBB) ||
3539 VPDT.properlyDominates(
3540 CondOfEarlyExitingVPBB->getDefiningRecipe()->getParent(),
3541 LatchVPBB)) &&
3542 "exit condition must dominate the latch");
3543 Exits.push_back({
3544 EarlyExitingVPBB,
3545 ExitBlock,
3546 CondToEarlyExit,
3547 });
3548 }
3549
3550 assert(!Exits.empty() && "must have at least one early exit");
3551 // Sort exits by RPO order to get correct program order. RPO gives a
3552 // topological ordering of the CFG, ensuring upstream exits are checked
3553 // before downstream exits in the dispatch chain.
3555 HeaderVPBB);
3557 for (const auto &[Num, VPB] : enumerate(RPOT))
3558 RPOIdx[VPB] = Num;
3559 llvm::sort(Exits, [&RPOIdx](const EarlyExitInfo &A, const EarlyExitInfo &B) {
3560 return RPOIdx[A.EarlyExitingVPBB] < RPOIdx[B.EarlyExitingVPBB];
3561 });
3562#ifndef NDEBUG
3563 // After RPO sorting, verify that for any pair where one exit dominates
3564 // another, the dominating exit comes first. This is guaranteed by RPO
3565 // (topological order) and is required for the dispatch chain correctness.
3566 for (unsigned I = 0; I + 1 < Exits.size(); ++I)
3567 for (unsigned J = I + 1; J < Exits.size(); ++J)
3568 assert(!VPDT.properlyDominates(Exits[J].EarlyExitingVPBB,
3569 Exits[I].EarlyExitingVPBB) &&
3570 "RPO sort must place dominating exits before dominated ones");
3571#endif
3572
3573 // Build the AnyOf condition for the latch terminator using logical OR
3574 // to avoid poison propagation from later exit conditions when an earlier
3575 // exit is taken.
3576 VPValue *Combined = Exits[0].CondToExit;
3577 for (const EarlyExitInfo &Info : drop_begin(Exits))
3578 Combined = LatchBuilder.createLogicalOr(Combined, Info.CondToExit);
3579 Combined = LatchBuilder.createFreeze(Combined);
3580
3581 // Even though the logical or prevents posion propagation, we need to freeze
3582 // Combined to prevent poisoning the entire AnyOf result:
3583 //
3584 // Exits[0].CondToExit = [0,1,0,0]
3585 // Exits[1].CondToExit = [0,0,p,p]
3586 // Combined = [0,1,p,p]
3587 // freeze(Combined) = [0,1,?,?]
3588 // AnyOf = 1
3589 VPValue *IsAnyExitTaken =
3590 LatchBuilder.createNaryOp(VPInstruction::AnyOf, Combined);
3591
3592 // Create a comparison for the latch exit condition and replace the
3593 // BranchOnCond with a BranchOnTwoConds. The original BranchOnCond's condition
3594 // is used as the latch-exit condition; canonical IV recipes have not been
3595 // introduced yet, so there is no BranchOnCount to derive the condition from.
3596 auto *LatchExitingBranch = cast<VPInstruction>(LatchVPBB->getTerminator());
3597 assert(LatchExitingBranch->getOpcode() == VPInstruction::BranchOnCond &&
3598 "Unexpected terminator");
3599 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3600 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3601 LatchExitingBranch->eraseFromParent();
3602 LatchBuilder.setInsertPoint(LatchVPBB);
3604 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3605 LatchVPBB->clearSuccessors();
3606
3608 // If handling the exiting lane in the scalar loop, combine the exit
3609 // conditions into a single BranchOnCond.
3610 LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3611 MiddleVPBB->clearPredecessors();
3612 MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
3613 return handleUncountableExitsWithSideEffects(Plan, Exits, HeaderVPBB,
3614 LatchVPBB, MiddleVPBB, ORE,
3615 TheLoop, PSE, DT, AC);
3616 }
3617
3618 // Create the vector.early.exit blocks.
3619 SmallVector<VPBasicBlock *> VectorEarlyExitVPBBs(Exits.size());
3620 for (unsigned Idx = 0; Idx != Exits.size(); ++Idx) {
3621 Twine BlockSuffix = Exits.size() == 1 ? "" : Twine(".") + Twine(Idx);
3622 VPBasicBlock *VectorEarlyExitVPBB =
3623 Plan.createVPBasicBlock("vector.early.exit" + BlockSuffix);
3624 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3625 }
3626
3627 // Create the dispatch block (or reuse the single exit block if only one
3628 // exit). The dispatch block computes the first active lane of the combined
3629 // condition and, for multiple exits, chains through conditions to determine
3630 // which exit to take.
3631 VPBasicBlock *DispatchVPBB =
3632 Exits.size() == 1 ? VectorEarlyExitVPBBs[0]
3633 : Plan.createVPBasicBlock("vector.early.exit.check");
3634 DispatchVPBB->setPredecessors({LatchVPBB});
3635 LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3636 VPBuilder DispatchBuilder(DispatchVPBB, DispatchVPBB->begin());
3637 VPValue *FirstActiveLane = DispatchBuilder.createFirstActiveLane(
3638 {Combined}, DebugLoc::getUnknown(), "first.active.lane");
3639
3640 // For each early exit, disconnect the original exiting block
3641 // (early.exiting.I) from the exit block (ir-bb<exit.I>) and route through a
3642 // new vector.early.exit block. Update ir-bb<exit.I>'s phis to extract their
3643 // values at the first active lane:
3644 //
3645 // Input:
3646 // early.exiting.I:
3647 // ...
3648 // EMIT branch-on-cond vp<%cond.I>
3649 // Successor(s): in.loop.succ, ir-bb<exit.I>
3650 //
3651 // ir-bb<exit.I>:
3652 // IR %phi = phi [ vp<%incoming.I>, early.exiting.I ], ...
3653 //
3654 // Output:
3655 // early.exiting.I:
3656 // ...
3657 // Successor(s): in.loop.succ
3658 //
3659 // vector.early.exit.I:
3660 // EMIT vp<%exit.val> = extract-lane vp<%first.lane>, vp<%incoming.I>
3661 // Successor(s): ir-bb<exit.I>
3662 //
3663 // ir-bb<exit.I>:
3664 // IR %phi = phi ... (extra operand: vp<%exit.val> from
3665 // vector.early.exit.I)
3666 //
3667 for (auto [Exit, VectorEarlyExitVPBB] :
3668 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3669 auto &[EarlyExitingVPBB, EarlyExitVPBB, _] = Exit;
3670 // Adjust the phi nodes in EarlyExitVPBB.
3671 // 1. remove incoming values from EarlyExitingVPBB,
3672 // 2. extract the incoming value at FirstActiveLane
3673 // 3. add back the extracts as last operands for the phis
3674 // Then adjust the CFG, removing the edge between EarlyExitingVPBB and
3675 // EarlyExitVPBB and adding a new edge between VectorEarlyExitVPBB and
3676 // EarlyExitVPBB. The extracts at FirstActiveLane are now the incoming
3677 // values from VectorEarlyExitVPBB.
3678 for (VPRecipeBase &R : EarlyExitVPBB->phis()) {
3679 auto *ExitIRI = cast<VPIRPhi>(&R);
3680 VPValue *IncomingVal =
3681 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3682 VPValue *NewIncoming = IncomingVal;
3683 if (!isa<VPIRValue>(IncomingVal)) {
3684 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3685 NewIncoming = EarlyExitBuilder.createNaryOp(
3686 VPInstruction::ExtractLane, {FirstActiveLane, IncomingVal},
3687 DebugLoc::getUnknown(), "early.exit.value");
3688 }
3689 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3690 ExitIRI->addIncoming(NewIncoming);
3691 }
3692
3693 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3694 VPBlockUtils::disconnectBlocks(EarlyExitingVPBB, EarlyExitVPBB);
3695 VPBlockUtils::connectBlocks(VectorEarlyExitVPBB, EarlyExitVPBB);
3696 }
3697
3698 // Chain through exits: for each exit, check if its condition is true at
3699 // the first active lane. If so, take that exit; otherwise, try the next.
3700 // The last exit needs no check since it must be taken if all others fail.
3701 //
3702 // For 3 exits (cond.0, cond.1, cond.2), this creates:
3703 //
3704 // latch:
3705 // ...
3706 // EMIT vp<%combined> = logical-or vp<%cond.0>, vp<%cond.1>, vp<%cond.2>
3707 // EMIT vp<%combined.freeze> = freeze vp<%combined>
3708 // ...
3709 //
3710 // vector.early.exit.check:
3711 // EMIT vp<%first.lane> = first-active-lane vp<%combined.freeze>
3712 // EMIT vp<%at.cond.0> = extract-lane vp<%first.lane>, vp<%cond.0>
3713 // EMIT branch-on-cond vp<%at.cond.0>
3714 // Successor(s): vector.early.exit.0, vector.early.exit.check.0
3715 //
3716 // vector.early.exit.check.0:
3717 // EMIT vp<%at.cond.1> = extract-lane vp<%first.lane>, vp<%cond.1>
3718 // EMIT branch-on-cond vp<%at.cond.1>
3719 // Successor(s): vector.early.exit.1, vector.early.exit.2
3720 VPBasicBlock *CurrentBB = DispatchVPBB;
3721 for (auto [I, Exit] : enumerate(ArrayRef(Exits).drop_back())) {
3722 VPValue *LaneVal = DispatchBuilder.createNaryOp(
3723 VPInstruction::ExtractLane, {FirstActiveLane, Exit.CondToExit},
3724 DebugLoc::getUnknown(), "exit.cond.at.lane");
3725
3726 // For the last dispatch, branch directly to the last exit on false;
3727 // otherwise, create a new check block.
3728 bool IsLastDispatch = (I + 2 == Exits.size());
3729 VPBasicBlock *FalseBB =
3730 IsLastDispatch ? VectorEarlyExitVPBBs.back()
3731 : Plan.createVPBasicBlock(
3732 Twine("vector.early.exit.check.") + Twine(I));
3733
3734 DispatchBuilder.createNaryOp(VPInstruction::BranchOnCond, {LaneVal});
3735 CurrentBB->setSuccessors({VectorEarlyExitVPBBs[I], FalseBB});
3736 VectorEarlyExitVPBBs[I]->setPredecessors({CurrentBB});
3737 FalseBB->setPredecessors({CurrentBB});
3738
3739 CurrentBB = FalseBB;
3740 DispatchBuilder.setInsertPoint(CurrentBB);
3741 }
3742
3743 return true;
3744}
3745
3746/// This function tries convert extended in-loop reductions to
3747/// VPExpressionRecipe and clamp the \p Range if it is beneficial and
3748/// valid. The created recipe must be decomposed to its constituent
3749/// recipes before execution.
3750static VPExpressionRecipe *
3752 VFRange &Range) {
3753 Type *RedTy = Red->getScalarType();
3754 VPValue *VecOp = Red->getVecOp();
3755
3756 // We don't handle partial reductions here.
3757 if (Red->isPartialReduction())
3758 return nullptr;
3759
3760 // Clamp the range if using extended-reduction is profitable.
3761 auto IsExtendedRedValidAndClampRange =
3762 [&](unsigned Opcode, Instruction::CastOps ExtOpc, Type *SrcTy) -> bool {
3764 [&](ElementCount VF) {
3765 auto *SrcVecTy = cast<VectorType>(toVectorTy(SrcTy, VF));
3767
3769 InstructionCost ExtCost =
3770 cast<VPWidenCastRecipe>(VecOp)->computeCost(VF, Ctx);
3771 InstructionCost RedCost = Red->computeCost(VF, Ctx);
3772
3773 assert(!RedTy->isFloatingPointTy() &&
3774 "getExtendedReductionCost only supports integer types");
3775 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3776 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3777 Red->getFastMathFlagsOrNone(), CostKind);
3778 return ExtRedCost.isValid() && ExtRedCost < ExtCost + RedCost;
3779 },
3780 Range);
3781 };
3782
3783 VPValue *A;
3784 // Match reduce(ext)).
3786 IsExtendedRedValidAndClampRange(
3787 RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind()),
3788 cast<VPWidenCastRecipe>(VecOp)->getOpcode(), A->getScalarType()))
3789 return new VPExpressionRecipe(cast<VPWidenCastRecipe>(VecOp), Red);
3790
3791 return nullptr;
3792}
3793
3794/// This function tries convert extended in-loop reductions to
3795/// VPExpressionRecipe and clamp the \p Range if it is beneficial
3796/// and valid. The created VPExpressionRecipe must be decomposed to its
3797/// constituent recipes before execution. Patterns of the
3798/// VPExpressionRecipe:
3799/// reduce.add(mul(...)),
3800/// reduce.add(mul(ext(A), ext(B))),
3801/// reduce.add(ext(mul(ext(A), ext(B)))).
3802/// reduce.fadd(fmul(ext(A), ext(B)))
3803static VPExpressionRecipe *
3805 VPCostContext &Ctx, VFRange &Range) {
3806 unsigned Opcode = RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind());
3807 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3808 Opcode != Instruction::FAdd)
3809 return nullptr;
3810
3811 // We don't handle partial reductions here.
3812 if (Red->isPartialReduction())
3813 return nullptr;
3814
3815 Type *RedTy = Red->getScalarType();
3816
3817 // Clamp the range if using multiply-accumulate-reduction is profitable.
3818 auto IsMulAccValidAndClampRange =
3820 VPWidenCastRecipe *OuterExt) -> bool {
3822 [&](ElementCount VF) {
3824 Type *SrcTy = Ext0 ? Ext0->getOperand(0)->getScalarType() : RedTy;
3825 InstructionCost MulAccCost;
3826
3827 // getMulAccReductionCost for in-loop reductions does not support
3828 // mixed or floating-point extends.
3829 if (Ext0 && Ext1 &&
3830 (Ext0->getOpcode() != Ext1->getOpcode() ||
3831 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3832 return false;
3833
3834 bool IsZExt =
3835 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3836 auto *SrcVecTy = cast<VectorType>(toVectorTy(SrcTy, VF));
3837 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3838 SrcVecTy, CostKind);
3839
3840 InstructionCost MulCost = Mul->computeCost(VF, Ctx);
3841 InstructionCost RedCost = Red->computeCost(VF, Ctx);
3842 InstructionCost ExtCost = 0;
3843 if (Ext0)
3844 ExtCost += Ext0->computeCost(VF, Ctx);
3845 if (Ext1)
3846 ExtCost += Ext1->computeCost(VF, Ctx);
3847 if (OuterExt)
3848 ExtCost += OuterExt->computeCost(VF, Ctx);
3849
3850 return MulAccCost.isValid() &&
3851 MulAccCost < ExtCost + MulCost + RedCost;
3852 },
3853 Range);
3854 };
3855
3856 VPValue *VecOp = Red->getVecOp();
3857 VPRecipeBase *Sub = nullptr;
3858 VPValue *A, *B;
3859 VPValue *Tmp = nullptr;
3860
3861 if (RedTy->isFloatingPointTy())
3862 return nullptr;
3863
3864 // Sub reductions could have a sub between the add reduction and vec op.
3865 if (match(VecOp, m_Sub(m_ZeroInt(), m_VPValue(Tmp)))) {
3866 Sub = VecOp->getDefiningRecipe();
3867 VecOp = Tmp;
3868 }
3869
3870 // If ValB is a constant and can be safely extended, truncate it to the same
3871 // type as ExtA's operand, then extend it to the same type as ExtA. This
3872 // creates two uniform extends that can more easily be matched by the rest of
3873 // the bundling code. The ExtB reference, ValB and operand 1 of Mul are all
3874 // replaced with the new extend of the constant.
3875 auto ExtendAndReplaceConstantOp = [](VPWidenCastRecipe *ExtA,
3876 VPWidenCastRecipe *&ExtB, VPValue *&ValB,
3877 VPWidenRecipe *Mul) {
3878 if (!ExtA || ExtB || !isa<VPIRValue>(ValB))
3879 return;
3880 Type *NarrowTy = ExtA->getOperand(0)->getScalarType();
3881 Instruction::CastOps ExtOpc = ExtA->getOpcode();
3882 const APInt *Const;
3883 if (!match(ValB, m_APInt(Const)) ||
3885 Const, NarrowTy, TTI::getPartialReductionExtendKind(ExtOpc)))
3886 return;
3887 // The truncate ensures that the type of each extended operand is the
3888 // same, and it's been proven that the constant can be extended from
3889 // NarrowTy safely. Necessary since ExtA's extended operand would be
3890 // e.g. an i8, while the const will likely be an i32. This will be
3891 // elided by later optimisations.
3892 VPBuilder Builder(Mul);
3893 auto *Trunc =
3894 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3895 Type *WideTy = ExtA->getScalarType();
3896 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3897 Mul->setOperand(1, ExtB);
3898 };
3899
3900 // Try to match reduce.add(mul(...)).
3901 if (match(VecOp, m_Mul(m_VPValue(A), m_VPValue(B)))) {
3902 auto *RecipeA = dyn_cast<VPWidenCastRecipe>(A);
3903 auto *RecipeB = dyn_cast<VPWidenCastRecipe>(B);
3904 auto *Mul = cast<VPWidenRecipe>(VecOp);
3905
3906 // Convert reduce.add(mul(ext, const)) to reduce.add(mul(ext, ext(const)))
3907 ExtendAndReplaceConstantOp(RecipeA, RecipeB, B, Mul);
3908
3909 // Match reduce.add/sub(mul(ext, ext)).
3910 if (RecipeA && RecipeB && match(RecipeA, m_ZExtOrSExt(m_VPValue())) &&
3911 match(RecipeB, m_ZExtOrSExt(m_VPValue())) &&
3912 IsMulAccValidAndClampRange(Mul, RecipeA, RecipeB, nullptr)) {
3913 if (Sub)
3914 return new VPExpressionRecipe(RecipeA, RecipeB, Mul,
3915 cast<VPWidenRecipe>(Sub), Red);
3916 return new VPExpressionRecipe(RecipeA, RecipeB, Mul, Red);
3917 }
3918 // TODO: Add an expression type for this variant with a negated mul
3919 if (!Sub && IsMulAccValidAndClampRange(Mul, nullptr, nullptr, nullptr))
3920 return new VPExpressionRecipe(Mul, Red);
3921 }
3922 // TODO: Add an expression type for negated versions of other expression
3923 // variants.
3924 if (Sub)
3925 return nullptr;
3926
3927 // Match reduce.add(ext(mul(A, B))).
3928 if (match(VecOp, m_ZExtOrSExt(m_Mul(m_VPValue(A), m_VPValue(B))))) {
3929 auto *Ext = cast<VPWidenCastRecipe>(VecOp);
3930 auto *Mul = cast<VPWidenRecipe>(Ext->getOperand(0));
3931 auto *Ext0 = dyn_cast<VPWidenCastRecipe>(A);
3932 auto *Ext1 = dyn_cast<VPWidenCastRecipe>(B);
3933
3934 // reduce.add(ext(mul(ext, const)))
3935 // -> reduce.add(ext(mul(ext, ext(const))))
3936 ExtendAndReplaceConstantOp(Ext0, Ext1, B, Mul);
3937
3938 // reduce.add(ext(mul(ext(A), ext(B))))
3939 // -> reduce.add(mul(wider_ext(A), wider_ext(B)))
3940 // The inner extends must either have the same opcode as the outer extend or
3941 // be the same, in which case the multiply can never result in a negative
3942 // value and the outer extend can be folded away by doing wider
3943 // extends for the operands of the mul.
3944 if (Ext0 && Ext1 &&
3945 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3946 Ext0->getOpcode() == Ext1->getOpcode() &&
3947 IsMulAccValidAndClampRange(Mul, Ext0, Ext1, Ext) && Mul->hasOneUse()) {
3948 auto *NewExt0 = new VPWidenCastRecipe(
3949 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(), nullptr,
3950 *Ext0, *Ext0, Ext0->getDebugLoc());
3951 NewExt0->insertBefore(Ext0);
3952
3953 VPWidenCastRecipe *NewExt1 = NewExt0;
3954 if (Ext0 != Ext1) {
3955 NewExt1 = new VPWidenCastRecipe(Ext1->getOpcode(), Ext1->getOperand(0),
3956 Ext->getScalarType(), nullptr, *Ext1,
3957 *Ext1, Ext1->getDebugLoc());
3958 NewExt1->insertBefore(Ext1);
3959 }
3960 auto *NewMul = Mul->cloneWithOperands({NewExt0, NewExt1});
3961 NewMul->insertBefore(Mul);
3962 Ext->replaceAllUsesWith(NewMul);
3963 Ext->eraseFromParent();
3964 Mul->eraseFromParent();
3965 return new VPExpressionRecipe(NewExt0, NewExt1, NewMul, Red);
3966 }
3967 }
3968 return nullptr;
3969}
3970
3971/// This function tries to create abstract recipes from the reduction recipe for
3972/// following optimizations and cost estimation.
3974 VPCostContext &Ctx,
3975 VFRange &Range) {
3976 // Creation of VPExpressions for partial reductions is entirely handled in
3977 // transformToPartialReduction.
3978 if (Red->isPartialReduction())
3979 return;
3980
3981 VPExpressionRecipe *AbstractR = nullptr;
3982 auto IP = std::next(Red->getIterator());
3983 auto *VPBB = Red->getParent();
3984 if (auto *MulAcc = tryToMatchAndCreateMulAccumulateReduction(Red, Ctx, Range))
3985 AbstractR = MulAcc;
3986 else if (auto *ExtRed = tryToMatchAndCreateExtendedReduction(Red, Ctx, Range))
3987 AbstractR = ExtRed;
3988 // Cannot create abstract inloop reduction recipes.
3989 if (!AbstractR)
3990 return;
3991
3992 AbstractR->insertBefore(*VPBB, IP);
3993 Red->replaceAllUsesWith(AbstractR);
3994}
3995
4005
4006// Collect common metadata from a group of replicate recipes by intersecting
4007// metadata from all recipes in the group.
4009 VPIRMetadata CommonMetadata = *Recipes.front();
4010 for (VPReplicateRecipe *Recipe : drop_begin(Recipes))
4011 CommonMetadata.intersect(*Recipe);
4012 // The recipe using the common metadata is not predicated, so it does not
4013 // share the group's execution frequency.
4014 CommonMetadata.clearExecutionFrequency();
4015 return CommonMetadata;
4016}
4017
4018template <unsigned Opcode>
4022 const Loop *L) {
4023 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
4024 "Only Load and Store opcodes supported");
4025 [[maybe_unused]] constexpr bool IsLoad = (Opcode == Instruction::Load);
4026
4027 // For each address, collect operations with the same or complementary masks.
4030 Plan, PSE, L,
4031 [](VPReplicateRecipe *RepR) { return RepR->isPredicated(); });
4032 for (auto Recipes : Groups) {
4033 if (Recipes.size() < 2)
4034 continue;
4035
4037 map_range(Recipes, bind_back<getLoadStoreValueType>(IsLoad))) &&
4038 "Expected all recipes in group to have the same load-store type");
4039
4040 // Collect groups with the same or complementary masks.
4041 for (VPReplicateRecipe *&RecipeI : Recipes) {
4042 if (!RecipeI)
4043 continue;
4044
4045 VPValue *MaskI = RecipeI->getMask();
4047 Group.push_back(RecipeI);
4048 RecipeI = nullptr;
4049
4050 // Find all operations with the same or complementary masks.
4051 bool HasComplementaryMask = false;
4052 for (VPReplicateRecipe *&RecipeJ : Recipes) {
4053 if (!RecipeJ)
4054 continue;
4055
4056 VPValue *MaskJ = RecipeJ->getMask();
4057 // Check if any operation in the group has a complementary mask with
4058 // another, that is M1 == NOT(M2) or M2 == NOT(M1).
4059 HasComplementaryMask |= match(MaskI, m_Not(m_Specific(MaskJ))) ||
4060 match(MaskJ, m_Not(m_Specific(MaskI)));
4061 Group.push_back(RecipeJ);
4062 RecipeJ = nullptr;
4063 }
4064
4065 if (HasComplementaryMask) {
4066 assert(Group.size() >= 2 && "must have at least 2 entries");
4067 AllGroups.push_back(std::move(Group));
4068 }
4069 }
4070 }
4071
4072 return AllGroups;
4073}
4074
4075// Find the recipe with minimum alignment in the group.
4076template <typename InstType>
4077static VPReplicateRecipe *
4079 return *min_element(Group, [](VPReplicateRecipe *A, VPReplicateRecipe *B) {
4080 return cast<InstType>(A->getUnderlyingInstr())->getAlign() <
4081 cast<InstType>(B->getUnderlyingInstr())->getAlign();
4082 });
4083}
4084
4087 const Loop *L) {
4088 auto Groups =
4090 if (Groups.empty())
4091 return;
4092
4093 // Process each group of loads.
4094 for (auto &Group : Groups) {
4095 // Try to use the earliest (most dominating) load to replace all others.
4096 VPReplicateRecipe *EarliestLoad = Group[0];
4097 VPBasicBlock *FirstBB = EarliestLoad->getParent();
4098 VPBasicBlock *LastBB = Group.back()->getParent();
4099
4100 // Check that the load doesn't alias with stores between first and last.
4101 auto LoadLoc = vputils::getMemoryLocation(*EarliestLoad);
4102 if (!LoadLoc || !canHoistOrSinkWithNoAliasCheck(*LoadLoc, FirstBB, LastBB))
4103 continue;
4104
4105 // Collect common metadata from all loads in the group.
4106 VPIRMetadata CommonMetadata = getCommonMetadata(Group);
4107
4108 // Find the load with minimum alignment to use.
4109 auto *LoadWithMinAlign = findRecipeWithMinAlign<LoadInst>(Group);
4110
4111 bool IsSingleScalar = EarliestLoad->isSingleScalar();
4112 assert(all_of(Group,
4113 [IsSingleScalar](VPReplicateRecipe *R) {
4114 return R->isSingleScalar() == IsSingleScalar;
4115 }) &&
4116 "all members in group must agree on IsSingleScalar");
4117
4118 // Create an unpredicated version of the earliest load with common
4119 // metadata.
4120 auto *UnpredicatedLoad = new VPReplicateRecipe(
4121 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
4122 IsSingleScalar, /*Mask=*/nullptr, *EarliestLoad, CommonMetadata);
4123
4124 UnpredicatedLoad->insertBefore(EarliestLoad);
4125
4126 // Replace all loads in the group with the unpredicated load.
4127 for (VPReplicateRecipe *Load : Group) {
4128 Load->replaceAllUsesWith(UnpredicatedLoad);
4129 Load->eraseFromParent();
4130 }
4131 }
4132}
4133
4134static bool
4136 PredicatedScalarEvolution &PSE, const Loop &L) {
4137 auto StoreLoc = vputils::getMemoryLocation(*StoresToSink.front());
4138 if (!StoreLoc || !StoreLoc->AATags.Scope)
4139 return false;
4140
4141 // When sinking a group of stores, all members of the group alias each other.
4142 // Skip them during the alias checks.
4143 VPBasicBlock *FirstBB = StoresToSink.front()->getParent();
4144 VPBasicBlock *LastBB = StoresToSink.back()->getParent();
4145 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
4146 return canHoistOrSinkWithNoAliasCheck(*StoreLoc, FirstBB, LastBB, SinkInfo);
4147}
4148
4151 const Loop *L) {
4152 auto Groups =
4154 if (Groups.empty())
4155 return;
4156
4157 for (auto &Group : Groups) {
4158 if (!canSinkStoreWithNoAliasCheck(Group, PSE, *L))
4159 continue;
4160
4161 // Use the last (most dominated) store's location for the unconditional
4162 // store.
4163 VPReplicateRecipe *LastStore = Group.back();
4164 VPBasicBlock *InsertBB = LastStore->getParent();
4165
4166 // Collect common alias metadata from all stores in the group.
4167 VPIRMetadata CommonMetadata = getCommonMetadata(Group);
4168
4169 // Build select chain for stored values.
4170 VPValue *SelectedValue = Group[0]->getOperand(0);
4171 VPBuilder Builder(InsertBB, LastStore->getIterator());
4172
4173 bool IsSingleScalar = Group[0]->isSingleScalar();
4174 for (unsigned I = 1; I < Group.size(); ++I) {
4175 assert(IsSingleScalar == Group[I]->isSingleScalar() &&
4176 "all members in group must agree on IsSingleScalar");
4177 VPValue *Mask = Group[I]->getMask();
4178 VPValue *Value = Group[I]->getOperand(0);
4179 SelectedValue = Builder.createSelect(
4180 Mask, Value, SelectedValue, Group[I]->getDebugLoc(), "",
4181 VPIRFlags::getDefaultFlags(Instruction::Select,
4182 Value->getScalarType()));
4183 }
4184
4185 // Find the store with minimum alignment to use.
4186 auto *StoreWithMinAlign = findRecipeWithMinAlign<StoreInst>(Group);
4187
4188 // Create unconditional store with selected value and common metadata.
4189 auto *UnpredicatedStore = new VPReplicateRecipe(
4190 StoreWithMinAlign->getUnderlyingInstr(),
4191 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
4192 /*Mask=*/nullptr, *LastStore, CommonMetadata);
4193 UnpredicatedStore->insertBefore(*InsertBB, LastStore->getIterator());
4194
4195 // Remove all predicated stores from the group.
4196 for (VPReplicateRecipe *Store : Group)
4197 Store->eraseFromParent();
4198 }
4199}
4200
4201/// Returns true if \p V is VPWidenLoadRecipe or VPInterleaveRecipe that can be
4202/// converted to a narrower recipe. \p V is used by a wide recipe that feeds a
4203/// store interleave group at index \p Idx, \p WideMember0 is the recipe feeding
4204/// the same interleave group at index 0. A VPWidenLoadRecipe can be narrowed to
4205/// an index-independent load if it feeds all wide ops at all indices (\p OpV
4206/// must be the operand at index \p OpIdx for both the recipe at lane 0, \p
4207/// WideMember0). A VPInterleaveRecipe can be narrowed to a wide load, if \p V
4208/// is defined at \p Idx of a load interleave group.
4209/// A live-in or recipe defined outside the loop region can be converted, if it
4210/// is the same across all lanes, or we can create a BuildVector for it.
4211static bool canNarrowLoad(VPSingleDefRecipe *WideMember0, unsigned OpIdx,
4212 VPValue *OpV, unsigned Idx, bool IsScalable) {
4213 VPValue *Member0Op = WideMember0->getOperand(OpIdx);
4214 if (Member0Op->isDefinedOutsideLoopRegions()) {
4215 // Operand matches Member0, broadcast across all fields for both live-ins
4216 // and recipes.
4217 if (Member0Op == OpV)
4218 return true;
4219 // Otherwise distinct per-field VPValues are assembled into a BuildVector.
4220 return !IsScalable && OpV->isDefinedOutsideLoopRegions() &&
4221 OpV->getScalarType() == Member0Op->getScalarType();
4222 }
4223 VPRecipeBase *Member0OpR = Member0Op->getDefiningRecipe();
4224 if (auto *W = dyn_cast<VPWidenLoadRecipe>(Member0OpR))
4225 // For scalable VFs, the narrowed plan processes vscale iterations at once,
4226 // so a shared wide load cannot be narrowed to a uniform scalar; bail out.
4227 return !IsScalable && !W->getMask() && W->isConsecutive() &&
4228 Member0Op == OpV;
4229 if (auto *IR = dyn_cast<VPInterleaveRecipe>(Member0OpR))
4230 return IR->getInterleaveGroup()->isFull() && IR->getVPValue(Idx) == OpV;
4231 return false;
4232}
4233
4234static bool canNarrowOps(ArrayRef<VPValue *> Ops, bool IsScalable) {
4236 auto *WideMember0 = dyn_cast<VPRecipeWithIRFlags>(Ops[0]);
4237 if (!WideMember0)
4238 return false;
4239 for (VPValue *V : Ops) {
4241 return false;
4242 auto *R = cast<VPRecipeWithIRFlags>(V);
4243 if (vputils::getOpcode(R) != vputils::getOpcode(WideMember0))
4244 return false;
4245 if (R->getScalarType() != WideMember0->getScalarType())
4246 return false;
4247 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
4248 return false;
4249 }
4250
4251 for (unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
4253 for (VPValue *Op : Ops)
4254 OpsI.push_back(Op->getDefiningRecipe()->getOperand(Idx));
4255
4256 if (canNarrowOps(OpsI, IsScalable))
4257 continue;
4258
4259 if (any_of(enumerate(OpsI), [WideMember0, Idx, IsScalable](const auto &P) {
4260 const auto &[OpIdx, OpV] = P;
4261 return !canNarrowLoad(WideMember0, Idx, OpV, OpIdx, IsScalable);
4262 }))
4263 return false;
4264 }
4265
4266 return true;
4267}
4268
4269/// Returns VF from \p VFs if \p IR is a full interleave group with factor and
4270/// number of members both equal to VF. The interleave group must also access
4271/// the full vector width.
4272static std::optional<ElementCount>
4275 const TargetTransformInfo &TTI) {
4276 if (!InterleaveR || InterleaveR->getMask())
4277 return std::nullopt;
4278
4279 Type *GroupElementTy = nullptr;
4280 if (InterleaveR->getStoredValues().empty()) {
4281 GroupElementTy = InterleaveR->getVPValue(0)->getScalarType();
4282 if (!all_of(InterleaveR->definedValues(), [GroupElementTy](VPValue *Op) {
4283 return Op->getScalarType() == GroupElementTy;
4284 }))
4285 return std::nullopt;
4286 } else {
4287 GroupElementTy = InterleaveR->getStoredValues()[0]->getScalarType();
4288 if (!all_of(InterleaveR->getStoredValues(), [GroupElementTy](VPValue *Op) {
4289 return Op->getScalarType() == GroupElementTy;
4290 }))
4291 return std::nullopt;
4292 }
4293
4294 auto IG = InterleaveR->getInterleaveGroup();
4295 if (IG->getFactor() != IG->getNumMembers())
4296 return std::nullopt;
4297
4298 auto GetVectorBitWidthForVF = [&TTI](ElementCount VF) {
4299 TypeSize Size = TTI.getRegisterBitWidth(
4302 assert(Size.isScalable() == VF.isScalable() &&
4303 "if Size is scalable, VF must be scalable and vice versa");
4304 return Size.getKnownMinValue();
4305 };
4306
4307 for (ElementCount VF : VFs) {
4308 unsigned MinVal = VF.getKnownMinValue();
4309 unsigned GroupSize = GroupElementTy->getScalarSizeInBits() * MinVal;
4310 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
4311 return {VF};
4312 }
4313 return std::nullopt;
4314}
4315
4316/// Returns true if \p VPValue is a narrow VPValue.
4317static bool isAlreadyNarrow(VPValue *VPV) {
4318 if (isa<VPIRValue>(VPV))
4319 return true;
4320 auto *RepR = dyn_cast<VPReplicateRecipe>(VPV);
4321 return RepR && RepR->isSingleScalar();
4322}
4323
4324// Convert the wide recipes defining the VPValues in \p Members feeding an
4325// interleave group to a single narrow variant. The first member is reused as
4326// the narrowed recipe. BuildVectors for live-in operands are inserted into \p
4327// Preheader.
4329 SmallPtrSetImpl<VPValue *> &NarrowedOps,
4330 VPBasicBlock *Preheader) {
4331 VPValue *V = Members.front();
4332 if (NarrowedOps.contains(V))
4333 return V;
4334
4335 if (V->isDefinedOutsideLoopRegions()) {
4336 assert(all_of(Members,
4337 [V](VPValue *M) {
4338 return M->isDefinedOutsideLoopRegions() &&
4339 M->getScalarType() == V->getScalarType();
4340 }) &&
4341 "expected distinct loop-invariant values of matching scalar type");
4342 auto *BV = new VPInstruction(VPInstruction::BuildVector, Members);
4343 Preheader->appendRecipe(BV);
4344 NarrowedOps.insert(BV);
4345 return BV;
4346 }
4347
4348 if (isAlreadyNarrow(V))
4349 return V;
4350
4351 VPRecipeBase *R = V->getDefiningRecipe();
4353 auto *WideMember0 = cast<VPRecipeWithIRFlags>(R);
4354 for (VPValue *Member : Members.drop_front())
4355 WideMember0->intersectFlags(*cast<VPRecipeWithIRFlags>(Member));
4356 for (unsigned Idx = 0, E = WideMember0->getNumOperands(); Idx != E; ++Idx) {
4358 for (VPValue *Member : Members)
4359 OpsI.push_back(Member->getDefiningRecipe()->getOperand(Idx));
4360 WideMember0->setOperand(
4361 Idx, narrowInterleaveGroupOp(OpsI, NarrowedOps, Preheader));
4362 }
4363 return V;
4364 }
4365
4366 if (auto *LoadGroup = dyn_cast<VPInterleaveRecipe>(R)) {
4367 // Narrow interleave group to wide load, as transformed VPlan will only
4368 // process one original iteration.
4369 auto *LI = cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4370 auto *L = VPBuilder(LoadGroup).createWidenLoad(
4371 *LI, LoadGroup->getAddr(), LoadGroup->getMask(), /*Consecutive=*/true,
4372 *LoadGroup, LoadGroup->getDebugLoc());
4373 NarrowedOps.insert(L);
4374 return L;
4375 }
4376
4377 if (auto *RepR = dyn_cast<VPReplicateRecipe>(R)) {
4378 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4379 "must be a single scalar load");
4380 NarrowedOps.insert(RepR);
4381 return RepR;
4382 }
4383
4384 auto *WideLoad = cast<VPWidenLoadRecipe>(R);
4385 VPValue *PtrOp = WideLoad->getAddr();
4386 if (auto *VecPtr = dyn_cast<VPVectorPointerRecipe>(PtrOp))
4387 PtrOp = VecPtr->getOperand(0);
4388 // Narrow wide load to uniform scalar load, as transformed VPlan will only
4389 // process one original iteration.
4390 auto *N = new VPReplicateRecipe(&WideLoad->getIngredient(), {PtrOp},
4391 /*IsUniform*/ true,
4392 /*Mask*/ nullptr, {}, *WideLoad);
4393 N->insertBefore(WideLoad);
4394 NarrowedOps.insert(N);
4395 return N;
4396}
4397
4398std::unique_ptr<VPlan>
4400 const TargetTransformInfo &TTI) {
4401 VPRegionBlock *VectorLoop = Plan.getVectorLoopRegion();
4402
4403 if (!VectorLoop)
4404 return nullptr;
4405
4406 // Only handle single-block loops for now.
4407 if (VectorLoop->getEntryBasicBlock() != VectorLoop->getExitingBasicBlock())
4408 return nullptr;
4409
4410 // Skip plans when we may not be able to properly narrow.
4411 VPBasicBlock *Exiting = VectorLoop->getExitingBasicBlock();
4412 if (!match(&Exiting->back(), m_BranchOnCount()))
4413 return nullptr;
4414
4415 assert(match(&Exiting->back(),
4417 m_Specific(&Plan.getVectorTripCount()))) &&
4418 "unexpected branch-on-count");
4419
4421 std::optional<ElementCount> VFToOptimize;
4422 for (auto &R : *VectorLoop->getEntryBasicBlock()) {
4425 continue;
4426
4427 // Bail out on recipes not supported at the moment:
4428 // * phi recipes other than the canonical induction
4429 // * recipes writing to memory except interleave groups
4430 // Only support plans with a canonical induction phi.
4431 if (R.isPhi())
4432 return nullptr;
4433
4434 auto *InterleaveR = dyn_cast<VPInterleaveRecipe>(&R);
4435 if (R.mayWriteToMemory() && !InterleaveR)
4436 return nullptr;
4437
4438 // Bail out if any recipe defines a vector value used outside the
4439 // vector loop region.
4440 if (any_of(R.definedValues(), [&](VPValue *V) {
4441 return any_of(V->users(), [&](VPUser *U) {
4442 auto *UR = cast<VPRecipeBase>(U);
4443 return UR->getParent()->getParent() != VectorLoop;
4444 });
4445 }))
4446 return nullptr;
4447
4448 // All other ops are allowed, but we reject uses that cannot be converted
4449 // when checking all allowed consumers (store interleave groups) below.
4450 if (!InterleaveR)
4451 continue;
4452
4453 // Try to find a single VF, where all interleave groups are consecutive and
4454 // saturate the full vector width. If we already have a candidate VF, check
4455 // if it is applicable for the current InterleaveR, otherwise look for a
4456 // suitable VF across the Plan's VFs.
4458 VFToOptimize ? SmallVector<ElementCount>({*VFToOptimize})
4459 : to_vector(Plan.vectorFactors());
4460 std::optional<ElementCount> NarrowedVF =
4461 isConsecutiveInterleaveGroup(InterleaveR, VFs, TTI);
4462 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4463 return nullptr;
4464 VFToOptimize = NarrowedVF;
4465
4466 // Skip read interleave groups.
4467 if (InterleaveR->getStoredValues().empty())
4468 continue;
4469
4470 // Narrow interleave groups, if all operands are already matching narrow
4471 // ops.
4472 auto *Member0 = InterleaveR->getStoredValues()[0];
4473 if (isAlreadyNarrow(Member0) &&
4474 all_of(InterleaveR->getStoredValues(), equal_to(Member0))) {
4475 StoreGroups.push_back(InterleaveR);
4476 continue;
4477 }
4478
4479 // For now, we only support full interleave groups storing load interleave
4480 // groups.
4481 if (all_of(enumerate(InterleaveR->getStoredValues()), [](auto Op) {
4482 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4483 if (!DefR)
4484 return false;
4485 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4486 return IR && IR->getInterleaveGroup()->isFull() &&
4487 IR->getVPValue(Op.index()) == Op.value();
4488 })) {
4489 StoreGroups.push_back(InterleaveR);
4490 continue;
4491 }
4492
4493 // Check if all values feeding InterleaveR are matching wide recipes, which
4494 // operands that can be narrowed.
4495 if (!canNarrowOps(InterleaveR->getStoredValues(),
4496 VFToOptimize->isScalable()))
4497 return nullptr;
4498 StoreGroups.push_back(InterleaveR);
4499 }
4500
4501 if (StoreGroups.empty())
4502 return nullptr;
4503
4504 VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
4505 bool RequiresScalarEpilogue =
4506 MiddleVPBB->getNumSuccessors() == 1 &&
4507 MiddleVPBB->getSingleSuccessor() == Plan.getScalarPreheader();
4508 // Bail out for tail-folding (middle block with a single successor to exit).
4509 if (MiddleVPBB->getNumSuccessors() != 2 && !RequiresScalarEpilogue)
4510 return nullptr;
4511
4512 // All interleave groups in Plan can be narrowed for VFToOptimize. Split the
4513 // original Plan into 2: a) a new clone which contains all VFs of Plan, except
4514 // VFToOptimize, and b) the original Plan with VFToOptimize as single VF.
4515 // TODO: Handle cases where only some interleave groups can be narrowed.
4516 std::unique_ptr<VPlan> NewPlan;
4517 if (size(Plan.vectorFactors()) != 1) {
4518 NewPlan = std::unique_ptr<VPlan>(Plan.duplicate());
4519 Plan.setVF(*VFToOptimize);
4520 NewPlan->removeVF(*VFToOptimize);
4521 }
4522
4523 // Convert InterleaveGroup \p R to a single VPWidenLoadRecipe.
4524 SmallPtrSet<VPValue *, 4> NarrowedOps;
4525 VPBasicBlock *Preheader = Plan.getVectorPreheader();
4526 // Narrow operation tree rooted at store groups.
4527 for (auto *StoreGroup : StoreGroups) {
4528 VPValue *Res = narrowInterleaveGroupOp(StoreGroup->getStoredValues(),
4529 NarrowedOps, Preheader);
4530 auto *SI =
4531 cast<StoreInst>(StoreGroup->getInterleaveGroup()->getInsertPos());
4532 VPBuilder(StoreGroup)
4533 .createWidenStore(*SI, StoreGroup->getAddr(), Res, nullptr,
4534 /*Consecutive=*/true, *StoreGroup,
4535 StoreGroup->getDebugLoc());
4536 StoreGroup->eraseFromParent();
4537 }
4538
4539 // Adjust induction to reflect that the transformed plan only processes one
4540 // original iteration.
4542 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4543 VPBasicBlock *VectorPH = Plan.getVectorPreheader();
4544 VPBuilder PHBuilder(VectorPH, VectorPH->getFirstNonPhi());
4545
4546 VPValue *UF = &Plan.getUF();
4547 VPValue *Step;
4548 if (VFToOptimize->isScalable()) {
4549 VPValue *VScale =
4550 PHBuilder.createElementCount(CanIVTy, ElementCount::getScalable(1));
4551 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale, UF},
4552 {true, false});
4553 Plan.getVF().replaceAllUsesWith(VScale);
4554 } else {
4555 Step = UF;
4556 Plan.getVF().replaceAllUsesWith(Plan.getConstantInt(CanIVTy, 1));
4557 }
4558 // Materialize vector trip count with the narrowed step.
4559 materializeVectorTripCount(Plan, VectorPH, /*TailByMasking=*/false,
4560 RequiresScalarEpilogue, Step);
4561
4562 CanIVInc->setOperand(1, Step);
4563 Plan.getVFxUF().replaceAllUsesWith(Step);
4564
4565 removeDeadRecipes(Plan);
4566 assert(none_of(*VectorLoop->getEntryBasicBlock(),
4568 "All VPVectorPointerRecipes should have been removed");
4569 return NewPlan;
4570}
4571
4573 VFRange &Range) {
4574 VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
4575 auto *MiddleVPBB = Plan.getMiddleBlock();
4576 VPBuilder MiddleBuilder(MiddleVPBB, MiddleVPBB->getFirstNonPhi());
4577
4578 auto IsScalableOne = [](ElementCount VF) -> bool {
4579 return VF == ElementCount::getScalable(1);
4580 };
4581
4584 VectorRegion->getEntryBasicBlock()->phis())) {
4585 assert(VectorRegion->getSingleSuccessor() == Plan.getMiddleBlock() &&
4586 "Cannot handle loops with uncountable early exits");
4587
4588 // Find the existing splice for this FOR, created in
4589 // createHeaderPhiRecipes. All uses of FOR have already been replaced with
4590 // RecurSplice there; only RecurSplice itself still references FOR.
4591 auto *RecurSplice =
4593 assert(RecurSplice && "expected FirstOrderRecurrenceSplice");
4594
4595 // For VF vscale x 1, if vscale = 1, we are unable to extract the
4596 // penultimate value of the recurrence. Instead we rely on the existing
4597 // extract of the last element from the result of
4598 // VPInstruction::FirstOrderRecurrenceSplice.
4599 // TODO: Consider vscale_range info and UF.
4600 if (any_of(RecurSplice->users(),
4601 [](VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4603 Range))
4604 return;
4605
4606 // This is the second phase of vectorizing first-order recurrences, creating
4607 // extracts for users outside the loop. An overview of the transformation is
4608 // described below. Suppose we have the following loop with some use after
4609 // the loop of the last a[i-1],
4610 //
4611 // for (int i = 0; i < n; ++i) {
4612 // t = a[i - 1];
4613 // b[i] = a[i] - t;
4614 // }
4615 // use t;
4616 //
4617 // There is a first-order recurrence on "a". For this loop, the shorthand
4618 // scalar IR looks like:
4619 //
4620 // scalar.ph:
4621 // s.init = a[-1]
4622 // br scalar.body
4623 //
4624 // scalar.body:
4625 // i = phi [0, scalar.ph], [i+1, scalar.body]
4626 // s1 = phi [s.init, scalar.ph], [s2, scalar.body]
4627 // s2 = a[i]
4628 // b[i] = s2 - s1
4629 // br cond, scalar.body, exit.block
4630 //
4631 // exit.block:
4632 // use = lcssa.phi [s1, scalar.body]
4633 //
4634 // In this example, s1 is a recurrence because it's value depends on the
4635 // previous iteration. In the first phase of vectorization, we created a
4636 // VPFirstOrderRecurrencePHIRecipe v1 for s1. Now we create the extracts
4637 // for users in the scalar preheader and exit block.
4638 //
4639 // vector.ph:
4640 // v_init = vector(..., ..., ..., a[-1])
4641 // br vector.body
4642 //
4643 // vector.body
4644 // i = phi [0, vector.ph], [i+4, vector.body]
4645 // v1 = phi [v_init, vector.ph], [v2, vector.body]
4646 // v2 = a[i, i+1, i+2, i+3]
4647 // v1' = splice(v1(3), v2(0, 1, 2))
4648 // b[i, i+1, i+2, i+3] = v2 - v1'
4649 // br cond, vector.body, middle.block
4650 //
4651 // middle.block:
4652 // vector.recur.extract.for.phi = v2(2)
4653 // vector.recur.extract = v2(3)
4654 // br cond, scalar.ph, exit.block
4655 //
4656 // scalar.ph:
4657 // scalar.recur.init = phi [vector.recur.extract, middle.block],
4658 // [s.init, otherwise]
4659 // br scalar.body
4660 //
4661 // scalar.body:
4662 // i = phi [0, scalar.ph], [i+1, scalar.body]
4663 // s1 = phi [scalar.recur.init, scalar.ph], [s2, scalar.body]
4664 // s2 = a[i]
4665 // b[i] = s2 - s1
4666 // br cond, scalar.body, exit.block
4667 //
4668 // exit.block:
4669 // lo = lcssa.phi [s1, scalar.body],
4670 // [vector.recur.extract.for.phi, middle.block]
4671 //
4672 // Update extracts of the splice in the middle block: they extract the
4673 // penultimate element of the recurrence.
4675 make_range(MiddleVPBB->getFirstNonPhi(), MiddleVPBB->end()))) {
4676 if (!match(&R, m_ExtractLastLaneOfLastPart(m_Specific(RecurSplice))))
4677 continue;
4678
4679 auto *ExtractR = cast<VPInstruction>(&R);
4680 VPValue *PenultimateElement = MiddleBuilder.createNaryOp(
4681 VPInstruction::ExtractPenultimateElement, RecurSplice->getOperand(1),
4682 {}, "vector.recur.extract.for.phi");
4683 for (VPUser *ExitU : to_vector(ExtractR->users())) {
4684 if (auto *ExitPhi = dyn_cast<VPIRPhi>(ExitU))
4685 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4686 }
4687 }
4688 }
4689}
4690
4691/// Check if \p V is a binary expression of a widened IV and a loop-invariant
4692/// value. Returns the widened IV if found, nullptr otherwise.
4694 auto *BinOp = dyn_cast<VPWidenRecipe>(V);
4695 if (!BinOp || !Instruction::isBinaryOp(BinOp->getOpcode()) ||
4696 Instruction::isIntDivRem(BinOp->getOpcode()))
4697 return nullptr;
4698
4699 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4700 VPValue *InvariantCandidate = BinOp->getOperand(1);
4701 if (!isa<VPWidenIntOrFpInductionRecipe>(WidenIVCandidate))
4702 std::swap(WidenIVCandidate, InvariantCandidate);
4703
4704 if (!InvariantCandidate->isDefinedOutsideLoopRegions())
4705 return nullptr;
4706
4707 return dyn_cast<VPWidenIntOrFpInductionRecipe>(WidenIVCandidate);
4708}
4709
4710/// Create a scalar version of \p BinOp, with its \p WidenIV operand replaced
4711/// by \p ScalarIV, and place it after \p ScalarIV's defining recipe.
4715 BinOp->getNumOperands() == 2 && "BinOp must have 2 operands");
4716 auto *ClonedOp = BinOp->clone();
4717 if (ClonedOp->getOperand(0) == WidenIV) {
4718 ClonedOp->setOperand(0, ScalarIV);
4719 } else {
4720 assert(ClonedOp->getOperand(1) == WidenIV && "one operand must be WideIV");
4721 ClonedOp->setOperand(1, ScalarIV);
4722 }
4723 ClonedOp->insertAfter(ScalarIV->getDefiningRecipe());
4724 return ClonedOp;
4725}
4726
4727/// If \p S is an affine AddRec, returns true if its step is known to be
4728/// positive and false if it is known to be negative. Returns std::nullopt if
4729/// \p S is not an affine AddRec, or if the sign of its step cannot be
4730/// determined.
4731static std::optional<bool> getStepDirection(const SCEV *S,
4732 ScalarEvolution &SE) {
4733 const SCEV *Step;
4734 if (!match(S, m_scev_AffineAddRec(m_SCEV(), m_SCEV(Step))))
4735 return std::nullopt;
4736 if (SE.isKnownPositive(Step))
4737 return true;
4738 if (SE.isKnownNegative(Step))
4739 return false;
4740 return std::nullopt;
4741}
4742
4745 Loop &L) {
4746 ScalarEvolution &SE = *PSE.getSE();
4747 VPRegionBlock *VectorLoopRegion = Plan.getVectorLoopRegion();
4748
4749 // Helper lambda to check if the IV range excludes the sentinel value. Try
4750 // signed first, then unsigned. Return an excluded sentinel if found,
4751 // otherwise return std::nullopt.
4752 auto CheckSentinel = [&SE](const SCEV *IVSCEV,
4753 bool UseMax) -> std::optional<APSInt> {
4754 unsigned BW = IVSCEV->getType()->getScalarSizeInBits();
4755 for (bool Signed : {true, false}) {
4756 APSInt Sentinel = UseMax ? APSInt::getMinValue(BW, /*Unsigned=*/!Signed)
4757 : APSInt::getMaxValue(BW, /*Unsigned=*/!Signed);
4758
4759 ConstantRange IVRange =
4760 Signed ? SE.getSignedRange(IVSCEV) : SE.getUnsignedRange(IVSCEV);
4761 if (!IVRange.contains(Sentinel))
4762 return Sentinel;
4763 }
4764 return std::nullopt;
4765 };
4766
4767 VPValue *HeaderMask = VectorLoopRegion->getHeaderMask();
4768 for (VPRecipeBase &Phi :
4769 make_early_inc_range(VectorLoopRegion->getEntryBasicBlock()->phis())) {
4770 auto *PhiR = dyn_cast<VPReductionPHIRecipe>(&Phi);
4772 PhiR->getRecurrenceKind()))
4773 continue;
4774
4775 Type *PhiTy = PhiR->getScalarType();
4776 if (PhiTy->isPointerTy() || PhiTy->isFloatingPointTy())
4777 continue;
4778
4779 // If there's a header mask, the backedge select will not be the find-last
4780 // select.
4781 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4782 auto *FindLastSelect = cast<VPSingleDefRecipe>(BackedgeVal);
4783 if (HeaderMask &&
4784 !match(BackedgeVal,
4785 m_Select(m_Specific(HeaderMask),
4786 m_VPSingleDefRecipe(FindLastSelect), m_Specific(PhiR))))
4787 continue;
4788
4789 // Get the find-last expression from the find-last select of the reduction
4790 // phi. The find-last select should be a select between the phi and the
4791 // find-last expression.
4792 VPValue *Cond, *FindLastExpression;
4793 if (!match(FindLastSelect, m_SelectLike(m_VPValue(Cond), m_Specific(PhiR),
4794 m_VPValue(FindLastExpression))) &&
4795 !match(FindLastSelect,
4796 m_SelectLike(m_VPValue(Cond), m_VPValue(FindLastExpression),
4797 m_Specific(PhiR))))
4798 continue;
4799
4800 // Check if FindLastExpression is a simple expression of a widened IV. If
4801 // so, we can track the underlying IV instead and sink the expression.
4802 auto *IVOfExpressionToSink = getExpressionIV(FindLastExpression);
4803 const SCEV *IVSCEV = vputils::getSCEVExprForVPValue(
4804 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4805 &L);
4806 if (!match(IVSCEV, m_scev_AffineAddRec(m_SCEV(), m_SCEV()))) {
4807 assert(!match(vputils::getSCEVExprForVPValue(FindLastExpression, PSE, &L),
4809 "IVOfExpressionToSink not being an AddRec must imply "
4810 "FindLastExpression not being an AddRec.");
4811 continue;
4812 }
4813
4814 // Determine direction from the step of IVSCEV, if possible.
4815 std::optional<bool> StepDirection = getStepDirection(IVSCEV, SE);
4816 if (!StepDirection)
4817 continue;
4818
4819 bool UseMax = *StepDirection;
4820 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4821 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4822
4823 // Sinking an expression will disable epilogue vectorization. Only use it,
4824 // if FindLastExpression cannot be vectorized via a sentinel. Sinking may
4825 // also prevent vectorizing using a sentinel (e.g., if the expression is a
4826 // multiply or divide by large constant, respectively), which also makes
4827 // sinking undesirable.
4828 if (IVOfExpressionToSink) {
4829 const SCEV *FindLastExpressionSCEV =
4830 vputils::getSCEVExprForVPValue(FindLastExpression, PSE, &L);
4831 if (std::optional<bool> NewUseMax =
4832 getStepDirection(FindLastExpressionSCEV, SE)) {
4833 if (auto NewSentinel =
4834 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4835 // The original expression already has a sentinel, so prefer not
4836 // sinking to keep epilogue vectorization possible.
4837 SentinelVal = *NewSentinel;
4838 UseSigned = NewSentinel->isSigned();
4839 UseMax = *NewUseMax;
4840 IVSCEV = FindLastExpressionSCEV;
4841 IVOfExpressionToSink = nullptr;
4842 }
4843 }
4844 }
4845
4846 // If no sentinel was found, fall back to a boolean AnyOf reduction to track
4847 // if the condition was ever true. Requires the IV to not wrap, otherwise we
4848 // cannot use min/max.
4849 if (!SentinelVal) {
4850 auto *AR = cast<SCEVAddRecExpr>(IVSCEV);
4851 if (AR->hasNoSignedWrap())
4852 UseSigned = true;
4853 else if (AR->hasNoUnsignedWrap())
4854 UseSigned = false;
4855 else
4856 continue;
4857 }
4858
4860 BackedgeVal,
4862
4863 VPValue *NewFindLastSelect = BackedgeVal;
4864 VPValue *SelectCond = Cond;
4865 if (!SentinelVal || IVOfExpressionToSink) {
4866 // When we need to create a new select, normalize the condition so that
4867 // PhiR is the last operand and include the header mask if needed.
4868 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4869 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4870 if (match(FindLastSelect,
4872 SelectCond = LoopBuilder.createNot(SelectCond);
4873
4874 // When tail folding, mask the condition with the header mask to prevent
4875 // propagating poison from inactive lanes in the last vector iteration.
4876 if (HeaderMask)
4877 SelectCond = LoopBuilder.createLogicalAnd(HeaderMask, SelectCond);
4878
4879 if (SelectCond != Cond || IVOfExpressionToSink) {
4880 NewFindLastSelect = LoopBuilder.createSelect(
4881 SelectCond,
4882 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4883 PhiR, DL);
4884 }
4885 }
4886
4887 // Create the reduction result in the middle block using sentinel directly.
4888 RecurKind MinMaxKind =
4889 UseMax ? (UseSigned ? RecurKind::SMax : RecurKind::UMax)
4890 : (UseSigned ? RecurKind::SMin : RecurKind::UMin);
4891 VPIRFlags Flags(MinMaxKind, /*IsOrdered=*/false, /*IsInLoop=*/false,
4892 FastMathFlags());
4893 DebugLoc ExitDL = RdxResult->getDebugLoc();
4894 VPBuilder MiddleBuilder(RdxResult);
4895 VPValue *ReducedIV =
4897 NewFindLastSelect, Flags, ExitDL);
4898
4899 // If IVOfExpressionToSink is an expression to sink, sink it now.
4900 VPValue *VectorRegionExitingVal = ReducedIV;
4901 if (IVOfExpressionToSink)
4902 VectorRegionExitingVal =
4903 cloneBinOpForScalarIV(cast<VPWidenRecipe>(FindLastExpression),
4904 ReducedIV, IVOfExpressionToSink);
4905
4906 VPValue *NewRdxResult;
4907 VPValue *StartVPV = PhiR->getStartValue();
4908 if (SentinelVal) {
4909 // Sentinel-based approach: reduce IVs with min/max, compare against
4910 // sentinel to detect if condition was ever true, select accordingly.
4911 VPValue *Sentinel = Plan.getConstantInt(*SentinelVal);
4912 auto *Cmp = MiddleBuilder.createICmp(CmpInst::ICMP_NE, ReducedIV,
4913 Sentinel, ExitDL);
4914 NewRdxResult = MiddleBuilder.createSelect(Cmp, VectorRegionExitingVal,
4915 StartVPV, ExitDL);
4916 StartVPV = Sentinel;
4917 } else {
4918 // Introduce a boolean AnyOf reduction to track if the condition was ever
4919 // true in the loop. Use it to select the initial start value, if it was
4920 // never true.
4921 auto *AnyOfPhi = new VPReductionPHIRecipe(
4922 /*Phi=*/nullptr, RecurKind::Or, *Plan.getFalse(), *Plan.getFalse(),
4923 RdxUnordered{1}, {}, /*HasUsesOutsideReductionChain=*/false);
4924 AnyOfPhi->insertAfter(PhiR);
4925
4926 VPBuilder LoopBuilder(BackedgeVal->getDefiningRecipe());
4927 VPValue *OrVal = LoopBuilder.createOr(AnyOfPhi, SelectCond);
4928 AnyOfPhi->setOperand(1, OrVal);
4929
4930 NewRdxResult = MiddleBuilder.createAnyOfReduction(
4931 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4932
4933 // Initialize the IV reduction phi with the neutral element, not the
4934 // original start value, to ensure correct min/max reduction results.
4935 StartVPV = Plan.getOrAddLiveIn(
4936 getRecurrenceIdentity(MinMaxKind, IVSCEV->getType(), {}));
4937 }
4938 RdxResult->replaceAllUsesWith(NewRdxResult);
4939 RdxResult->eraseFromParent();
4940
4941 auto *NewPhiR = new VPReductionPHIRecipe(
4942 cast<PHINode>(PhiR->getUnderlyingInstr()), RecurKind::FindIV, *StartVPV,
4943 *NewFindLastSelect, RdxUnordered{1}, {},
4944 PhiR->hasUsesOutsideReductionChain());
4945 NewPhiR->insertBefore(PhiR);
4946 PhiR->replaceAllUsesWith(NewPhiR);
4947 PhiR->eraseFromParent();
4948 }
4949}
4950
4951namespace {
4952
4953using ExtendKind = TTI::PartialReductionExtendKind;
4954struct ReductionExtend {
4955 Type *SrcType = nullptr;
4956 ExtendKind Kind = ExtendKind::PR_None;
4957};
4958
4959/// Describes the extends used to compute the extended reduction operand.
4960/// ExtendB is optional. If ExtendB is present, ExtendsUser is a binary
4961/// operation.
4962struct ExtendedReductionOperand {
4963 /// The recipe that consumes the extends.
4964 VPWidenRecipe *ExtendsUser = nullptr;
4965 /// Extend descriptions (inputs to getPartialReductionCost).
4966 ReductionExtend ExtendA, ExtendB;
4967};
4968
4969/// A chain of recipes that form a partial reduction. Matches either
4970/// reduction_bin_op (extended op, accumulator), or
4971/// reduction_bin_op (accumulator, extended op).
4972/// The possible forms of the "extended op" are listed in
4973/// matchExtendedReductionOperand.
4974struct VPPartialReductionChain {
4975 /// The top-level binary operation that forms the reduction to a scalar
4976 /// after the loop body.
4977 VPWidenRecipe *ReductionBinOp = nullptr;
4978 /// The user of the extends that is then reduced.
4979 ExtendedReductionOperand ExtendedOp;
4980 /// The recurrence kind for the entire partial reduction chain.
4981 /// This allows distinguishing between Sub and AddWithSub recurrences,
4982 /// when the ReductionBinOp is a Instruction::Sub.
4983 RecurKind RK;
4984 /// The index of the accumulator operand of ReductionBinOp. The extended op
4985 /// is `1 - AccumulatorOpIdx`.
4986 unsigned AccumulatorOpIdx;
4987 unsigned ScaleFactor;
4988 /// Optional blend to represent predication for the block that updates the
4989 /// reduction.
4990 VPBlendRecipe *Blend = nullptr;
4991};
4992
4993// Return the incoming index of the single-use value in the blend, which is
4994// expected to be the predicated reduction update.
4995static std::optional<unsigned>
4996getBlendReductionUpdateValueIdx(VPBlendRecipe *Blend) {
4997 assert(Blend && !Blend->isNormalized() &&
4998 Blend->getNumIncomingValues() == 2 &&
4999 "Expected a non-normalized blend with two incoming values");
5000 bool FirstIncomingHasOneUse = Blend->getIncomingValue(0)->hasOneUse();
5001
5002 // Only the update value should have one use (the blend). The previous
5003 // value should always have at least two uses, the blend and the reduction.
5004 if (FirstIncomingHasOneUse == Blend->getIncomingValue(1)->hasOneUse())
5005 return std::nullopt;
5006 return FirstIncomingHasOneUse ? 0 : 1;
5007}
5008
5009static VPSingleDefRecipe *
5010optimizeExtendsForPartialReduction(VPSingleDefRecipe *Op) {
5011 // reduce.add(mul(ext(A), C))
5012 // -> reduce.add(mul(ext(A), ext(trunc(C))))
5013 const APInt *Const;
5014 if (match(Op, m_Mul(m_ZExtOrSExt(m_VPValue()), m_APInt(Const)))) {
5015 auto *ExtA = cast<VPWidenCastRecipe>(Op->getOperand(0));
5016 Instruction::CastOps ExtOpc = ExtA->getOpcode();
5017 Type *NarrowTy = ExtA->getOperand(0)->getScalarType();
5018 if (!Op->hasOneUse() ||
5020 Const, NarrowTy, TTI::getPartialReductionExtendKind(ExtOpc)))
5021 return Op;
5022
5023 VPBuilder Builder(Op);
5024 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
5025 Op->getOperand(1), NarrowTy);
5026 Type *WideTy = ExtA->getScalarType();
5027 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
5028 return Op;
5029 }
5030
5031 // reduce.add(abs(sub(ext(A), ext(B))))
5032 // -> reduce.add(ext(absolute-difference(A, B)))
5033 VPValue *X, *Y;
5036 auto *Sub = Op->getOperand(0)->getDefiningRecipe();
5037 auto *Ext = cast<VPWidenCastRecipe>(Sub->getOperand(0));
5038 assert(Ext->getOpcode() ==
5039 cast<VPWidenCastRecipe>(Sub->getOperand(1))->getOpcode() &&
5040 "Expected both the LHS and RHS extends to be the same");
5041 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
5042 VPBuilder Builder(Op);
5043 Type *SrcTy = X->getScalarType();
5044 auto *FreezeX = Builder.insert(new VPWidenRecipe(Instruction::Freeze, {X}));
5045 auto *FreezeY = Builder.insert(new VPWidenRecipe(Instruction::Freeze, {Y}));
5046 auto *Max = Builder.insert(
5047 new VPWidenIntrinsicRecipe(IsSigned ? Intrinsic::smax : Intrinsic::umax,
5048 {FreezeX, FreezeY}, SrcTy));
5049 auto *Min = Builder.insert(
5050 new VPWidenIntrinsicRecipe(IsSigned ? Intrinsic::smin : Intrinsic::umin,
5051 {FreezeX, FreezeY}, SrcTy));
5052 auto *AbsDiff = Builder.insert(
5053 new VPWidenRecipe(Instruction::Sub, {Max, Min},
5054 VPIRFlags::getDefaultFlags(Instruction::Sub)));
5055 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
5056 Op->getScalarType());
5057 }
5058
5059 // reduce.add(ext(mul(ext(A), ext(B))))
5060 // -> reduce.add(mul(wider_ext(A), wider_ext(B)))
5061 // TODO: Support this optimization for float types.
5063 m_ZExtOrSExt(m_VPValue()))))) {
5064 auto *Ext = cast<VPWidenCastRecipe>(Op);
5065 auto *Mul = cast<VPWidenRecipe>(Ext->getOperand(0));
5066 auto *MulLHS = cast<VPWidenCastRecipe>(Mul->getOperand(0));
5067 auto *MulRHS = cast<VPWidenCastRecipe>(Mul->getOperand(1));
5068 if (!Mul->hasOneUse() ||
5069 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
5070 MulLHS->getOpcode() != MulRHS->getOpcode())
5071 return Op;
5072 VPBuilder Builder(Mul);
5073 auto *NewLHS = Builder.createWidenCast(
5074 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
5075 auto *NewRHS = MulLHS == MulRHS
5076 ? NewLHS
5077 : Builder.createWidenCast(MulRHS->getOpcode(),
5078 MulRHS->getOperand(0),
5079 Ext->getScalarType());
5080 auto *NewMul = Mul->cloneWithOperands({NewLHS, NewRHS});
5081 Builder.insert(NewMul);
5082 Op->replaceAllUsesWith(NewMul);
5083 Op->eraseFromParent();
5084 Mul->eraseFromParent();
5085 return NewMul;
5086 }
5087
5088 return Op;
5089}
5090
5091static VPExpressionRecipe *
5092createPartialReductionExpression(VPReductionRecipe *Red) {
5093 VPValue *VecOp = Red->getVecOp();
5094
5095 // reduce.[f]add(ext(op))
5096 // -> VPExpressionRecipe(op, red)
5097 if (match(VecOp, m_WidenAnyExtend(m_VPValue())))
5098 return new VPExpressionRecipe(cast<VPWidenCastRecipe>(VecOp), Red);
5099
5100 // reduce.[f]add(neg(ext(op)))
5101 // -> VPExpressionRecipe(op, sub/neg, red)
5102 if (match(VecOp, m_AnyNeg(m_WidenAnyExtend(m_VPValue())))) {
5103 auto *Neg = cast<VPWidenRecipe>(VecOp);
5104 auto *Ext =
5105 cast<VPWidenCastRecipe>(Neg->getOperand(Neg->getNumOperands() - 1));
5106 return new VPExpressionRecipe(Ext, Neg, Red);
5107 }
5108
5109 // reduce.[f]add([f]mul(ext(a), ext(b)))
5110 // -> VPExpressionRecipe(a, b, mul, red)
5111 if (match(VecOp, m_FMul(m_FPExt(m_VPValue()), m_FPExt(m_VPValue()))) ||
5112 match(VecOp,
5114 auto *Mul = cast<VPWidenRecipe>(VecOp);
5115 auto *ExtA = cast<VPWidenCastRecipe>(Mul->getOperand(0));
5116 auto *ExtB = cast<VPWidenCastRecipe>(Mul->getOperand(1));
5117 return new VPExpressionRecipe(ExtA, ExtB, Mul, Red);
5118 }
5119
5120 // reduce.fadd(fneg(fmul(fpext(a), fpext(b))))
5121 // -> VPExpressionRecipe(a, b, fmul, fsub, red)
5122 if (match(VecOp,
5124 auto *FNeg = cast<VPWidenRecipe>(VecOp);
5125 auto *FMul = cast<VPWidenRecipe>(FNeg->getOperand(0));
5126 auto *ExtA = cast<VPWidenCastRecipe>(FMul->getOperand(0));
5127 auto *ExtB = cast<VPWidenCastRecipe>(FMul->getOperand(1));
5128 return new VPExpressionRecipe(ExtA, ExtB, FMul, FNeg, Red);
5129 }
5130
5131 // reduce.add(neg(mul(ext(a), ext(b))))
5132 // -> VPExpressionRecipe(a, b, mul, sub, red)
5134 m_ZExtOrSExt(m_VPValue()))))) {
5135 auto *Sub = cast<VPWidenRecipe>(VecOp);
5136 auto *Mul = cast<VPWidenRecipe>(Sub->getOperand(1));
5137 auto *ExtA = cast<VPWidenCastRecipe>(Mul->getOperand(0));
5138 auto *ExtB = cast<VPWidenCastRecipe>(Mul->getOperand(1));
5139 return new VPExpressionRecipe(ExtA, ExtB, Mul, Sub, Red);
5140 }
5141
5142 llvm_unreachable("Unsupported expression");
5143}
5144
5145// Helper to transform a partial reduction chain into a partial reduction
5146// recipe. Assumes profitability has been checked.
5147static void transformToPartialReduction(const VPPartialReductionChain &Chain,
5148 VPlan &Plan,
5149 VPReductionPHIRecipe *RdxPhi) {
5150 VPWidenRecipe *WidenRecipe = Chain.ReductionBinOp;
5151 assert(WidenRecipe->getNumOperands() == 2 && "Expected binary operation");
5152
5153 VPValue *Accumulator = WidenRecipe->getOperand(Chain.AccumulatorOpIdx);
5154 auto *ExtendedOp = cast<VPSingleDefRecipe>(
5155 WidenRecipe->getOperand(1 - Chain.AccumulatorOpIdx));
5156
5157 // FIXME: Do these transforms before invoking the cost-model.
5158 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
5159
5160 // Sub-reductions can be implemented in two ways:
5161 // (1) negate the operand in the vector loop (the default way).
5162 // (2) subtract the reduced value from the init value in the middle block.
5163 // Both ways keep the reduction itself as an 'add' reduction.
5164 //
5165 // The ISD nodes for partial reductions don't support folding the
5166 // sub/negation into its operands because the following is not a valid
5167 // transformation:
5168 // sub(0, mul(ext(a), ext(b)))
5169 // -> mul(ext(a), ext(sub(0, b)))
5170 //
5171 // It's therefore better to choose option (2) such that the partial
5172 // reduction is always positive (starting at '0') and to do a final
5173 // subtract in the middle block.
5174 if ((WidenRecipe->getOpcode() == Instruction::Sub &&
5175 Chain.RK != RecurKind::Sub) ||
5176 (WidenRecipe->getOpcode() == Instruction::FSub &&
5177 Chain.RK != RecurKind::FSub)) {
5178 VPBuilder Builder(WidenRecipe);
5179 Type *ElemTy = ExtendedOp->getScalarType();
5180 VPWidenRecipe *NegRecipe;
5181 if (WidenRecipe->getOpcode() == Instruction::FSub) {
5182 NegRecipe =
5183 new VPWidenRecipe(Instruction::FNeg, {ExtendedOp},
5184 VPIRFlags::getDefaultFlags(Instruction::FNeg),
5186 } else {
5187 auto *Zero = Plan.getZero(ElemTy);
5188 NegRecipe =
5189 new VPWidenRecipe(Instruction::Sub, {Zero, ExtendedOp},
5190 VPIRFlags::getDefaultFlags(Instruction::Sub),
5192 }
5193 Builder.insert(NegRecipe);
5194 ExtendedOp = NegRecipe;
5195 }
5196
5197 // Check if WidenRecipe is the final result of the reduction. If so, look
5198 // through the Select recipe introduced by tail-folding, otherwise look
5199 // through any Blend recipe introduced by predication for the block.
5200 VPValue *ExitSearch =
5201 Chain.Blend ? cast<VPValue>(Chain.Blend) : cast<VPValue>(WidenRecipe);
5202
5203 VPValue *Cond = nullptr;
5205 findUserOf(ExitSearch, m_Select(m_VPValue(Cond), m_Specific(ExitSearch),
5206 m_Specific(RdxPhi))));
5207
5208 if (Chain.Blend) {
5209 std::optional<unsigned> BlendReductionIdx =
5210 getBlendReductionUpdateValueIdx(Chain.Blend);
5211 assert(BlendReductionIdx &&
5212 Chain.Blend->getIncomingValue(*BlendReductionIdx) == WidenRecipe &&
5213 "Expected blend to contain the reduction update");
5214 VPValue *BlendCond = Chain.Blend->getMask(*BlendReductionIdx);
5215 Cond = ExitValue ? VPBuilder(WidenRecipe)
5216 .createLogicalAnd(Cond, BlendCond,
5217 WidenRecipe->getDebugLoc())
5218 : BlendCond;
5219 }
5220
5221 // When folding the tail, the inactive lanes of the reduction update are
5222 // computed from values that do not correspond to any scalar iteration
5223 // and must not be accumulated.
5224 if (!Cond)
5226
5227 bool IsLastInChain = RdxPhi->getBackedgeValue() == WidenRecipe ||
5228 RdxPhi->getBackedgeValue() == ExitValue ||
5229 RdxPhi->getBackedgeValue() == Chain.Blend;
5230 assert((!ExitValue || IsLastInChain) &&
5231 "if we found ExitValue, it must match RdxPhi's backedge value");
5232
5233 Type *PhiType = RdxPhi->getScalarType();
5234 RecurKind RdxKind =
5236 auto *PartialRed = new VPReductionRecipe(
5237 RdxKind,
5238 RdxKind == RecurKind::FAdd ? WidenRecipe->getFastMathFlagsOrNone()
5239 : FastMathFlags(),
5240 WidenRecipe->getUnderlyingInstr(), Accumulator, ExtendedOp, Cond,
5241 RdxUnordered{/*VFScaleFactor=*/Chain.ScaleFactor});
5242 PartialRed->insertBefore(WidenRecipe);
5243
5244 if (ExitValue)
5245 ExitValue->replaceAllUsesWith(PartialRed);
5246 if (Chain.Blend)
5247 Chain.Blend->replaceAllUsesWith(PartialRed);
5248 WidenRecipe->replaceAllUsesWith(PartialRed);
5249
5250 // For cost-model purposes, fold this into a VPExpression.
5251 VPExpressionRecipe *E = createPartialReductionExpression(PartialRed);
5252 E->insertBefore(WidenRecipe);
5253 PartialRed->replaceAllUsesWith(E);
5254
5255 // We only need to update the PHI node once, which is when we find the
5256 // last reduction in the chain.
5257 if (!IsLastInChain)
5258 return;
5259
5260 // Scale the PHI and ReductionStartVector by the VFScaleFactor
5261 assert(RdxPhi->getVFScaleFactor() == 1 && "scale factor must not be set");
5262 RdxPhi->setVFScaleFactor(Chain.ScaleFactor);
5263
5264 auto *StartInst = cast<VPInstruction>(RdxPhi->getStartValue());
5265 assert(StartInst->getOpcode() == VPInstruction::ReductionStartVector);
5266 auto *NewScaleFactor = Plan.getConstantInt(32, Chain.ScaleFactor);
5267 StartInst->setOperand(2, NewScaleFactor);
5268
5269 // If this is the last value in a sub-reduction chain, then update the PHI
5270 // node to start at `0` and update the reduction-result to subtract from
5271 // the PHI's start value.
5272 if (Chain.RK != RecurKind::Sub && Chain.RK != RecurKind::FSub)
5273 return;
5274
5275 VPValue *OldStartValue = StartInst->getOperand(0);
5276 StartInst->setOperand(0, StartInst->getOperand(1));
5277
5278 // Replace reduction_result by 'sub (startval, reductionresult)'.
5280 assert(RdxResult && "Could not find reduction result");
5281
5282 VPBuilder Builder = VPBuilder::getToInsertAfter(RdxResult);
5283 unsigned SubOpc = Chain.RK == RecurKind::FSub ? Instruction::BinaryOps::FSub
5284 : Instruction::BinaryOps::Sub;
5285 VPInstruction *NewResult = Builder.createNaryOp(
5286 SubOpc, {OldStartValue, RdxResult}, VPIRFlags::getDefaultFlags(SubOpc),
5287 RdxPhi->getDebugLoc());
5288 RdxResult->replaceUsesWithIf(
5289 NewResult,
5290 [&NewResult](VPUser &U, unsigned Idx) { return &U != NewResult; });
5291}
5292
5293/// Returns the cost of a link in a partial-reduction chain for a given VF.
5294static InstructionCost
5295getPartialReductionLinkCost(VPCostContext &CostCtx,
5296 const VPPartialReductionChain &Link,
5297 ElementCount VF) {
5298 Type *RdxType = Link.ReductionBinOp->getScalarType();
5299 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5300 std::optional<unsigned> BinOpc = std::nullopt;
5301 // If ExtendB is not none, then the "ExtendsUser" is the binary operation.
5302 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5303 BinOpc = ExtendedOp.ExtendsUser->getOpcode();
5304
5305 std::optional<llvm::FastMathFlags> Flags;
5306 if (RdxType->isFloatingPointTy())
5307 Flags = Link.ReductionBinOp->getFastMathFlagsOrNone();
5308
5309 auto GetLinkOpcode = [&Link]() -> unsigned {
5310 switch (Link.RK) {
5311 case RecurKind::Sub:
5312 return Instruction::Add;
5313 case RecurKind::FSub:
5314 return Instruction::FAdd;
5315 default:
5316 return Link.ReductionBinOp->getOpcode();
5317 }
5318 };
5319
5320 return CostCtx.TTI.getPartialReductionCost(
5321 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5322 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5323 CostCtx.CostKind, Flags);
5324}
5325
5326static ExtendKind getPartialReductionExtendKind(VPWidenCastRecipe *Cast) {
5328}
5329
5330/// Checks if \p Op (which is an operand of \p UpdateR) is an extended reduction
5331/// operand. This is an operand where the source of the value (e.g. a load) has
5332/// been extended (sext, zext, or fpext) before it is used in the reduction.
5333///
5334/// Possible forms matched by this function:
5335/// - UpdateR(PrevValue, ext(...))
5336/// - UpdateR(PrevValue, mul(ext(...), ext(...)))
5337/// - UpdateR(PrevValue, mul(ext(...), Constant))
5338/// - UpdateR(PrevValue, ext(mul(ext(...), ext(...))))
5339/// - UpdateR(PrevValue, ext(mul(ext(...), Constant)))
5340/// - UpdateR(PrevValue, abs(sub(ext(...), ext(...)))
5341///
5342/// Note: The second operand of UpdateR corresponds to \p Op in the examples.
5343static std::optional<ExtendedReductionOperand>
5344matchExtendedReductionOperand(VPWidenRecipe *UpdateR, VPValue *Op) {
5345 assert(is_contained(UpdateR->operands(), Op) &&
5346 "Op should be operand of UpdateR");
5347
5348 // Try matching an absolute difference operand of the form
5349 // `abs(sub(ext(A), ext(B)))`. This will be later transformed into
5350 // `ext(absolute-difference(A, B))`. This allows us to perform the absolute
5351 // difference on a wider type and get the extend for "free" from the partial
5352 // reduction.
5353 VPValue *X, *Y;
5354 if (Op->hasOneUse() &&
5358 auto *Abs = cast<VPWidenIntrinsicRecipe>(Op);
5359 auto *Sub = cast<VPWidenRecipe>(Abs->getOperand(0));
5360 auto *LHSExt = cast<VPWidenCastRecipe>(Sub->getOperand(0));
5361 auto *RHSExt = cast<VPWidenCastRecipe>(Sub->getOperand(1));
5362 Type *LHSInputType = X->getScalarType();
5363 Type *RHSInputType = Y->getScalarType();
5364 if (LHSInputType != RHSInputType ||
5365 LHSExt->getOpcode() != RHSExt->getOpcode())
5366 return std::nullopt;
5367 // Note: This is essentially the same as matching ext(...) as we will
5368 // rewrite this operand to ext(absolute-difference(A, B)).
5369 return ExtendedReductionOperand{
5370 Sub,
5371 /*ExtendA=*/{LHSInputType, getPartialReductionExtendKind(LHSExt)},
5372 /*ExtendB=*/{}};
5373 }
5374
5375 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5377 auto *CastRecipe = cast<VPWidenCastRecipe>(Op);
5378 VPValue *CastSource = CastRecipe->getOperand(0);
5379 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5380 if (match(CastSource, m_Mul(m_VPValue(), m_VPValue())) ||
5381 match(CastSource, m_FMul(m_VPValue(), m_VPValue()))) {
5382 // Match: ext(mul(...))
5383 // Record the outer extend kind and set `Op` to the mul. We can then match
5384 // this as a binary operation. Note: We can optimize out the outer extend
5385 // by widening the inner extends to match it. See
5386 // optimizeExtendsForPartialReduction.
5387 Op = CastSource;
5388 } else {
5389 return ExtendedReductionOperand{
5390 UpdateR,
5391 /*ExtendA=*/{CastSource->getScalarType(), *OuterExtKind},
5392 /*ExtendB=*/{}};
5393 }
5394 }
5395
5396 if (!Op->hasOneUse())
5397 return std::nullopt;
5398
5400 if (!MulOp ||
5401 !is_contained({Instruction::Mul, Instruction::FMul}, MulOp->getOpcode()))
5402 return std::nullopt;
5403
5404 // The rest of the matching assumes `Op` is a (possibly extended) mul
5405 // operation.
5406
5407 VPValue *LHS = MulOp->getOperand(0);
5408 VPValue *RHS = MulOp->getOperand(1);
5409
5410 // The LHS of the operation must always be an extend.
5412 return std::nullopt;
5413
5414 auto *LHSCast = cast<VPWidenCastRecipe>(LHS);
5415 Type *LHSInputType = LHSCast->getOperand(0)->getScalarType();
5416 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5417
5418 // The RHS of the operation can be an extend or a constant integer.
5419 const APInt *RHSConst = nullptr;
5420 VPWidenCastRecipe *RHSCast = nullptr;
5422 RHSCast = cast<VPWidenCastRecipe>(RHS);
5423 else if (!match(RHS, m_APInt(RHSConst)) ||
5424 !canConstantBeExtended(RHSConst, LHSInputType, LHSExtendKind))
5425 return std::nullopt;
5426
5427 // The outer extend kind must match the inner extends for folding.
5428 for (VPWidenCastRecipe *Cast : {LHSCast, RHSCast})
5429 if (Cast && OuterExtKind &&
5430 getPartialReductionExtendKind(Cast) != OuterExtKind)
5431 return std::nullopt;
5432
5433 Type *RHSInputType = LHSInputType;
5434 ExtendKind RHSExtendKind = LHSExtendKind;
5435 if (RHSCast) {
5436 RHSInputType = RHSCast->getOperand(0)->getScalarType();
5437 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5438 }
5439
5440 return ExtendedReductionOperand{
5441 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5442}
5443
5444/// Examines each operation in the reduction chain corresponding to \p RedPhiR,
5445/// and determines if the target can use a cheaper operation with a wider
5446/// per-iteration input VF and narrower PHI VF. If successful, returns the chain
5447/// of operations in the reduction.
5448static std::optional<SmallVector<VPPartialReductionChain>>
5449getScaledReductions(VPReductionPHIRecipe *RedPhiR) {
5450 // Get the backedge value from the reduction PHI and find the
5451 // ComputeReductionResult that uses it (directly or through a select for
5452 // predicated reductions).
5453 auto *RdxResult = vputils::findComputeReductionResult(RedPhiR);
5454 if (!RdxResult)
5455 return std::nullopt;
5456 VPValue *ExitValue = RdxResult->getOperand(0);
5457 match(ExitValue, m_Select(m_VPValue(), m_VPValue(ExitValue), m_VPValue()));
5458
5460 RecurKind RK = RedPhiR->getRecurrenceKind();
5461 Type *PhiType = RedPhiR->getScalarType();
5462 TypeSize PHISize = PhiType->getPrimitiveSizeInBits();
5463
5464 // Work backwards from the ExitValue examining each reduction operation.
5465 VPValue *CurrentValue = ExitValue;
5466 while (CurrentValue != RedPhiR) {
5467 VPBlendRecipe *Blend = dyn_cast<VPBlendRecipe>(CurrentValue);
5468 std::optional<unsigned> BlendReductionIdx;
5469 if (Blend) {
5470 assert(!Blend->isNormalized() && "Expect Blend not to be normalized.");
5471 if (Blend->getNumIncomingValues() != 2)
5472 return std::nullopt;
5473
5474 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5475 if (!BlendReductionIdx)
5476 return std::nullopt;
5477
5478 CurrentValue = Blend->getIncomingValue(*BlendReductionIdx);
5479 }
5480
5481 auto *UpdateR = dyn_cast<VPWidenRecipe>(CurrentValue);
5482 if (!UpdateR || !Instruction::isBinaryOp(UpdateR->getOpcode()))
5483 return std::nullopt;
5484
5485 VPValue *Op = UpdateR->getOperand(1);
5486 VPValue *PrevValue = UpdateR->getOperand(0);
5487
5488 // Find the extended operand. The other operand (PrevValue) is the next link
5489 // in the reduction chain.
5490 std::optional<ExtendedReductionOperand> ExtendedOp =
5491 matchExtendedReductionOperand(UpdateR, Op);
5492 if (!ExtendedOp) {
5493 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5494 if (!ExtendedOp)
5495 return std::nullopt;
5496 std::swap(Op, PrevValue);
5497 }
5498
5499 // Look for VPBlend(reduce(PrevValue, Op), PrevValue), where
5500 // reduce is equal to CurrentValue. This can be lowered as
5501 // a conditional reduction by hoisting the select to the inputs.
5502 if (Blend && Blend->getIncomingValue(1 - *BlendReductionIdx) != PrevValue)
5503 return std::nullopt;
5504
5505 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5506 TypeSize ExtSrcSize = ExtSrcType->getPrimitiveSizeInBits();
5507 if (!PHISize.hasKnownScalarFactor(ExtSrcSize))
5508 return std::nullopt;
5509
5510 VPPartialReductionChain Link(
5511 {UpdateR, *ExtendedOp, RK,
5512 PrevValue == UpdateR->getOperand(0) ? 0U : 1U,
5513 static_cast<unsigned>(PHISize.getKnownScalarFactor(ExtSrcSize)),
5514 Blend});
5515 Chain.push_back(Link);
5516 CurrentValue = PrevValue;
5517 }
5518
5519 // The chain links were collected by traversing backwards from the exit value.
5520 // Reverse the chains so they are in program order.
5521 std::reverse(Chain.begin(), Chain.end());
5522 return Chain;
5523}
5524} // namespace
5525
5527 VPCostContext &CostCtx,
5528 VFRange &Range) {
5529 // Find all possible valid partial reductions, grouping chains by their PHI.
5530 // This grouping allows invalidating the whole chain, if any link is not a
5531 // valid partial reduction.
5533 ChainsByPhi;
5534 VPBasicBlock *HeaderVPBB = Plan.getVectorLoopRegion()->getEntryBasicBlock();
5535 SmallVector<VPReductionPHIRecipe *, 4> UnorderedReductions;
5536 for (VPReductionPHIRecipe &RedPhiR :
5538 if (auto Chains = getScaledReductions(&RedPhiR))
5539 ChainsByPhi.try_emplace(&RedPhiR, std::move(*Chains));
5541 (RedPhiR.getRecurrenceKind() == RecurKind::Add ||
5542 (RedPhiR.getRecurrenceKind() == RecurKind::FAdd &&
5543 !RedPhiR.isOrdered() && !RedPhiR.isInLoop())))
5544 UnorderedReductions.push_back(&RedPhiR);
5545 }
5546
5547 // For general unordered reductions which aren't part of a candidate chain for
5548 // a scaled partial reduction, we can still use the intrinsic to allow for
5549 // more optimization later on.
5550 for (auto *Rdx : UnorderedReductions) {
5551 auto *Backedge = dyn_cast<VPWidenRecipe>(Rdx->getBackedgeValue());
5552 VPValue *OtherOp;
5553 if (!Backedge ||
5554 !match(Backedge,
5555 m_CombineOr(m_c_FAdd(m_Specific(Rdx), m_VPValue(OtherOp)),
5556 m_c_Add(m_Specific(Rdx), m_VPValue(OtherOp)))))
5557 continue;
5558
5559 // If the target indicates that the intrinsic is as cheap as (or cheaper
5560 // than) the add, then prefer the intrinsic.
5562 [&CostCtx, Rdx, Backedge](ElementCount VF) {
5563 InstructionCost CurrentCost = Backedge->computeCost(VF, CostCtx);
5564 Type *ScalarTy = Backedge->getScalarType();
5565 auto FMF = ScalarTy->isFloatingPointTy()
5566 ? std::make_optional(Rdx->getFastMathFlagsOrNone())
5567 : std::nullopt;
5568
5570 Backedge->getOpcode(), ScalarTy, /*InputTypeB=*/nullptr,
5571 ScalarTy, VF, TTI::PR_None, TTI::PR_None,
5572 /*BinOp=*/std::nullopt, CostCtx.CostKind, FMF);
5573 return PRCost <= CurrentCost;
5574 },
5575 Range))
5576 continue;
5577
5578 auto *Partial = new VPReductionRecipe(
5579 Rdx->getRecurrenceKind(), Rdx->getFastMathFlagsOrNone(),
5580 Backedge->getUnderlyingInstr(), Rdx, OtherOp, nullptr,
5581 getReductionStyle(/*InLoop=*/false, /*Ordered=*/false,
5582 /*ScaleFactor=*/1));
5583 Partial->insertBefore(Backedge);
5584 Backedge->replaceAllUsesWith(Partial);
5585 Backedge->eraseFromParent();
5586 }
5587
5588 if (ChainsByPhi.empty())
5589 return;
5590
5591 // Build set of partial reduction operations and blends for user validation
5592 // and a map of reduction bin ops to their scale factors for scale validation.
5593 SmallPtrSet<VPRecipeBase *, 4> PartialReductionOps;
5594 SmallPtrSet<VPBlendRecipe *, 4> PartialReductionBlends;
5595 DenseMap<VPSingleDefRecipe *, unsigned> ScaledReductionMap;
5596 for (const auto &[_, Chains] : ChainsByPhi)
5597 for (const VPPartialReductionChain &Chain : Chains) {
5598 PartialReductionOps.insert(Chain.ExtendedOp.ExtendsUser);
5599 if (Chain.Blend)
5600 PartialReductionBlends.insert(Chain.Blend);
5601 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5602 }
5603
5604 // A partial reduction is invalid if any of its extends are used by
5605 // something that isn't another partial reduction. This is because the
5606 // extends are intended to be lowered along with the reduction itself.
5607 auto ExtendUsersValid = [&](VPValue *Ext) {
5608 return !isa<VPWidenCastRecipe>(Ext) || all_of(Ext->users(), [&](VPUser *U) {
5609 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5610 });
5611 };
5612
5613 auto IsProfitablePartialReductionChainForVF =
5614 [&](ArrayRef<VPPartialReductionChain> Chain, ElementCount VF) -> bool {
5615 InstructionCost PartialCost = 0, RegularCost = 0;
5616
5617 // The chain is a profitable partial reduction chain if the cost of handling
5618 // the entire chain is cheaper when using partial reductions than when
5619 // handling the entire chain using regular reductions.
5620 for (const VPPartialReductionChain &Link : Chain) {
5621 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5622 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5623 if (!LinkCost.isValid())
5624 return false;
5625
5626 PartialCost += LinkCost;
5627 RegularCost += Link.ReductionBinOp->computeCost(VF, CostCtx);
5628 // If ExtendB is not none, then the "ExtendsUser" is the binary operation.
5629 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5630 RegularCost += ExtendedOp.ExtendsUser->computeCost(VF, CostCtx);
5631 for (VPValue *Op : ExtendedOp.ExtendsUser->operands())
5632 if (auto *Extend = dyn_cast<VPWidenCastRecipe>(Op))
5633 RegularCost += Extend->computeCost(VF, CostCtx);
5634 }
5635 return PartialCost.isValid() && PartialCost < RegularCost;
5636 };
5637
5638 // Validate chains: check that extends are only used by partial reductions,
5639 // and that reduction bin ops are only used by other partial reductions with
5640 // matching scale factors, are outside the loop region or the select
5641 // introduced by tail-folding. Otherwise we would create users of scaled
5642 // reductions where the types of the other operands don't match.
5643 for (auto &[RedPhiR, Chains] : ChainsByPhi) {
5644 for (const VPPartialReductionChain &Chain : Chains) {
5645 if (!all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5646 Chains.clear();
5647 break;
5648 }
5649 auto UseIsValid = [&, RedPhiR = RedPhiR](VPUser *U) {
5650 if (auto *PhiR = dyn_cast<VPReductionPHIRecipe>(U))
5651 return PhiR == RedPhiR;
5652 auto *R = cast<VPSingleDefRecipe>(U);
5653
5654 if (auto *Blend = dyn_cast<VPBlendRecipe>(R))
5655 return Blend == Chain.Blend || PartialReductionBlends.contains(Blend);
5656
5657 return Chain.ScaleFactor == ScaledReductionMap.lookup_or(R, 0) ||
5659 m_Specific(Chain.ReductionBinOp))) ||
5660 match(R, m_Select(m_VPValue(), m_Specific(Chain.ReductionBinOp),
5661 m_Specific(RedPhiR)));
5662 };
5663 if (!all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5664 Chains.clear();
5665 break;
5666 }
5667
5668 // Check if the compute-reduction-result is used by a sunk store.
5669 // TODO: Also form partial reductions in those cases.
5670 if (auto *RdxResult = vputils::findComputeReductionResult(RedPhiR)) {
5671 if (any_of(RdxResult->users(), [](VPUser *U) {
5672 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5673 return RepR && RepR->getOpcode() == Instruction::Store;
5674 })) {
5675 Chains.clear();
5676 break;
5677 }
5678 }
5679 }
5680
5681 // Clear the chain if it is not profitable.
5683 [&, &Chains = Chains](ElementCount VF) {
5684 return IsProfitablePartialReductionChainForVF(Chains, VF);
5685 },
5686 Range))
5687 Chains.clear();
5688 }
5689
5690 for (auto &[Phi, Chains] : ChainsByPhi)
5691 for (const VPPartialReductionChain &Chain : Chains)
5692 transformToPartialReduction(Chain, Plan, Phi);
5693}
5694
5696 VPRecipeBuilder &RecipeBuilder,
5697 VPCostContext &CostCtx) {
5698 // Collect all loads/stores first. We will start with ones having simpler
5699 // decisions followed by more complex ones that are potentially
5700 // guided/dependent on the simpler ones.
5702 for (VPBasicBlock *VPBB :
5705 for (VPInstruction &VPI : make_isa_range<VPInstruction>(*VPBB)) {
5706 if (VPI.getUnderlyingValue() &&
5707 is_contained({Instruction::Load, Instruction::Store},
5708 VPI.getOpcode()))
5709 MemOps.push_back(&VPI);
5710 }
5711 }
5712
5713 // Few helpers to process different kinds of memory operations.
5714
5715 // To be used as argument to `VPlanTransforms::runPass` which explicitly
5716 // specified pass name, hence `VPlan &` parameter.
5717 auto ProcessSubset = [&](VPlan &, auto ProcessVPInst) {
5718 SmallVector<VPInstruction *> RemainingMemOps;
5719 for (VPInstruction *VPI : MemOps) {
5720 if (!ProcessVPInst(VPI))
5721 RemainingMemOps.push_back(VPI);
5722 }
5723
5724 MemOps.clear();
5725 std::swap(MemOps, RemainingMemOps);
5726 };
5727
5728 auto ReplaceWith = [&](VPInstruction *VPI, VPRecipeBase *New) {
5729 assert(New->getParent() && "New recipe must have been inserted");
5730 if (VPI->getOpcode() == Instruction::Load)
5731 VPI->replaceAllUsesWith(New->getVPSingleValue());
5732 VPI->eraseFromParent();
5733
5734 // VPI has been processed.
5735 return true;
5736 };
5737
5738 auto Scalarize = [&](VPInstruction *VPI) {
5739 return ReplaceWith(VPI, VPBuilder(VPI).insert(
5740 RecipeBuilder.handleReplication(VPI, Range)));
5741 };
5742
5743 VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
5744 VPBuilder FinalRedStoresBuilder(MiddleVPBB, MiddleVPBB->getFirstNonPhi());
5746 "lowerMemoryIdioms", ProcessSubset, Plan, [&](VPInstruction *VPI) {
5747 if (RecipeBuilder.replaceWithFinalIfReductionStore(
5748 VPI, FinalRedStoresBuilder))
5749 return true;
5750
5751 // Filter out scalar VPlan for the remaining idioms.
5753 [](ElementCount VF) { return VF.isScalar(); }, Range))
5754 return false;
5755
5756 if (VPHistogramRecipe *Histogram = RecipeBuilder.widenIfHistogram(VPI))
5757 return ReplaceWith(VPI, VPBuilder(VPI).insert(Histogram));
5758
5759 return false;
5760 });
5761
5762 // Filter out scalar VPlan for the remaining memory operations.
5764 [](ElementCount VF) { return VF.isScalar(); }, Range))
5765 return;
5766
5767 // If the instruction's allocated size doesn't equal it's type size, it
5768 // requires padding and will be scalarized.
5770 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5771 [&](VPInstruction *VPI) {
5773 if (hasIrregularType(getLoadStoreType(I), I->getDataLayout()))
5774 return Scalarize(VPI);
5775
5776 return false;
5777 });
5778
5779 if (!RecipeBuilder.prefersVectorizedAddressing()) {
5781 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](VPInstruction *VPI) {
5783 bool IsLoad = VPI->getOpcode() == Instruction::Load;
5784 if (RecipeBuilder.isPredicatedInst(I) || !IsLoad ||
5786 return false;
5787
5788 // Scalarize loads used as addresses, matching the legacy CM. The load
5789 // is single-scalar if the pointer is loop-invariant, otherwise it is
5790 // replicated per-lane. No mask is needed as the load is not
5791 // predicated.
5792 VPValue *Ptr = VPI->getOperand(0);
5793 const SCEV *PtrSCEV =
5794 vputils::getSCEVExprForVPValue(Ptr, CostCtx.PSE, CostCtx.L);
5795 bool IsSingleScalarLoad =
5796 !isa<SCEVCouldNotCompute>(PtrSCEV) &&
5797 CostCtx.PSE.getSE()->isLoopInvariant(PtrSCEV, CostCtx.L);
5798
5799 ReplaceWith(VPI,
5800 VPBuilder(VPI).insert(new VPReplicateRecipe(
5801 I, Ptr, /*IsSingleScalar=*/IsSingleScalarLoad,
5802 /*Mask=*/nullptr, *VPI, *VPI, VPI->getDebugLoc())));
5803 return true;
5804 });
5805 }
5806
5807 // Widen unit-stride consecutive accesses, matching the legacy CM. Both
5808 // forward (stride +1) and reverse (stride -1) accesses are handled.
5810 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](VPInstruction *VPI) {
5812 bool IsLoad = VPI->getOpcode() == Instruction::Load;
5813 VPValue *Ptr = VPI->getOperand(!IsLoad);
5814 Type *ScalarTy =
5815 IsLoad ? VPI->getScalarType() : VPI->getOperand(0)->getScalarType();
5816 std::optional<int64_t> Stride =
5817 vputils::getConstantStride(Ptr, ScalarTy, CostCtx.PSE, CostCtx.L);
5818 if (Stride != 1 && Stride != -1)
5819 return false;
5820 bool Reverse = Stride == -1;
5821
5822 // A predicated access can only be widened (rather than scalarized) if
5823 // the target supports a masked load/store for it.
5824 // TODO: Determine if a load/store needs predication directly in VPlan.
5825 bool IsPredicated = RecipeBuilder.isPredicatedInst(I);
5826 if (IsPredicated && !CostCtx.Config.isLegalMaskedLoadOrStore(
5827 IsLoad, ScalarTy, getLoadStoreAlignment(I),
5829 return false;
5830
5831 VPBuilder Builder(VPI);
5832 VPSingleDefRecipe *VectorPtr = Builder.createConsecutiveVectorPointer(
5833 Ptr, ScalarTy, Reverse, VPI->getDebugLoc());
5834
5835 VPValue *Mask = IsPredicated ? VPI->getMask() : nullptr;
5836 // Reverse the mask so it matches the reversed access order.
5837 if (Reverse && Mask)
5838 Mask = Builder.createNaryOp(VPInstruction::Reverse, Mask,
5839 VPI->getDebugLoc());
5840
5841 if (IsLoad) {
5842 VPSingleDefRecipe *Load = Builder.createWidenLoad(
5843 *cast<LoadInst>(I), VectorPtr, Mask,
5844 /*Consecutive=*/true, *VPI, VPI->getDebugLoc());
5845 // Reverse the loaded values back into program order.
5846 if (Reverse)
5847 Load = Builder.createNaryOp(VPInstruction::Reverse, Load,
5848 VPI->getDebugLoc());
5849 return ReplaceWith(VPI, Load);
5850 }
5851
5852 VPValue *StoredVal = VPI->getOperand(0);
5853 if (Reverse)
5854 // Reverse the stored values so they are written in descending order.
5855 StoredVal = Builder.createNaryOp(VPInstruction::Reverse, StoredVal,
5856 VPI->getDebugLoc());
5857
5858 auto *StoreR = Builder.createWidenStore(
5859 *cast<StoreInst>(I), VectorPtr, StoredVal, Mask,
5860 /*Consecutive=*/true, *VPI, VPI->getDebugLoc());
5861 return ReplaceWith(VPI, StoreR);
5862 });
5863
5864 VPlanTransforms::runPass("delegateMemOpWideningToLegacyCM", ProcessSubset,
5865 Plan, [&](VPInstruction *VPI) {
5866 if (VPRecipeBase *Recipe =
5867 RecipeBuilder.tryToWidenMemory(VPI, Range))
5868 return ReplaceWith(VPI, Recipe);
5869
5870 return Scalarize(VPI);
5871 });
5872}
5873
5876 [&](ElementCount VF) { return VF.isScalar(); }, Range))
5877 return;
5878
5880 Plan.getEntry());
5882 for (VPInstruction &VPI :
5884 auto *I = cast_or_null<Instruction>(VPI.getUnderlyingValue());
5885 // Wouldn't be able to create a `VPReplicateRecipe` anyway.
5886 if (!I)
5887 continue;
5888
5889 // If executing other lanes produces side-effects we can't avoid them.
5890 if (VPI.mayHaveSideEffects())
5891 continue;
5892
5893 // We want to drop the mask operand, verify we can safely do that.
5894 if (VPI.isMasked() && !VPI.isSafeToSpeculativelyExecute())
5895 continue;
5896
5897 // Avoid rewriting IV increment as that interferes with
5898 // `removeRedundantCanonicalIVs`.
5899 if (VPI.getOpcode() == Instruction::Add &&
5901 continue;
5902
5903 // Other lanes are needed - can't drop them.
5904 if (!vputils::onlyFirstLaneUsed(&VPI))
5905 continue;
5906
5907 auto *Recipe = VPBuilder::createSingleScalarOp(
5908 VPI.getOpcode(), VPI.operandsWithoutMask(), /*Mask=*/nullptr, VPI,
5909 VPI, VPI.getDebugLoc(), VPI.getScalarType(), I);
5910 Recipe->insertBefore(&VPI);
5911 VPI.replaceAllUsesWith(Recipe);
5912 VPI.eraseFromParent();
5913 }
5914 }
5915}
5916
5917/// Returns true if \p Info's parameter kinds are compatible with \p Args.
5918static bool areVFParamsOk(const VFInfo &Info, ArrayRef<VPValue *> Args,
5919 PredicatedScalarEvolution &PSE, const Loop *L) {
5920 ScalarEvolution *SE = PSE.getSE();
5921 return all_of(Info.Shape.Parameters, [&](VFParameter Param) {
5922 switch (Param.ParamKind) {
5923 case VFParamKind::Vector:
5924 case VFParamKind::GlobalPredicate:
5925 return true;
5926 case VFParamKind::OMP_Uniform:
5927 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5928 SE->isLoopInvariant(
5929 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5930 L);
5931 case VFParamKind::OMP_Linear:
5932 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5933 m_scev_AffineAddRec(
5934 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5935 m_SpecificLoop(L)));
5936 default:
5937 return false;
5938 }
5939 });
5940}
5941
5942/// Find a vector variant of \p CI for \p VF, respecting \p MaskRequired.
5943/// Returns the variant function, or nullptr. Masked variants are assumed to
5944/// take the mask as a trailing parameter.
5946 ElementCount VF, bool MaskRequired,
5948 const Loop *L) {
5949 if (CI->isNoBuiltin())
5950 return nullptr;
5951 auto Mappings = VFDatabase::getMappings(*CI);
5952 const auto *It = find_if(Mappings, [&](const VFInfo &Info) {
5953 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5954 areVFParamsOk(Info, Args, PSE, L);
5955 });
5956 if (It == Mappings.end())
5957 return nullptr;
5958 return CI->getModule()->getFunction(It->VectorName);
5959}
5960
5961namespace {
5962/// The outcome of choosing how to widen a call at a given VF.
5963struct CallWideningDecision {
5964 enum class KindTy { Scalarize, Intrinsic, VectorVariant };
5965 CallWideningDecision(KindTy Kind, Function *Variant = nullptr)
5966 : Kind(Kind), Variant(Variant) {}
5967 KindTy Kind;
5968
5969 /// Set when Kind == VectorVariant.
5971
5972 bool operator==(const CallWideningDecision &Other) const {
5973 return Kind == Other.Kind && Variant == Other.Variant;
5974 }
5975};
5976} // namespace
5977
5978/// Pick the cheapest widening for the call \p VPI at \p VF among scalarization,
5979/// vector intrinsic, and vector library variant.
5980static CallWideningDecision decideCallWidening(VPInstruction &VPI,
5982 ElementCount VF,
5983 VPCostContext &CostCtx) {
5984 auto *CI = cast<CallInst>(VPI.getUnderlyingInstr());
5985
5986 // Scalar VFs and calls forced or known to scalarize always replicate.
5987 if (VF.isScalar() || CostCtx.willBeScalarized(CI, VF))
5988 return CallWideningDecision::KindTy::Scalarize;
5989
5990 auto *CalledFn = cast<Function>(
5992 Type *ResultTy = VPI.getScalarType();
5994 bool MaskRequired = CostCtx.isMaskRequired(CI);
5995
5996 // Pseudo intrinsics (assume, lifetime, ...) are always scalarized.
5998 return CallWideningDecision::KindTy::Scalarize;
5999
6000 InstructionCost ScalarCost =
6001 VPReplicateRecipe::computeCallCost(CalledFn, ResultTy, Ops,
6002 /*IsSingleScalar=*/false, VF, CostCtx);
6003
6004 Function *VecFunc =
6005 findVectorVariant(CI, Ops, VF, MaskRequired, CostCtx.PSE, CostCtx.L);
6007 if (VecFunc)
6008 VecCallCost = VPWidenCallRecipe::computeCallCost(VecFunc, CostCtx);
6009
6010 // Prefer the intrinsic if it is at least as cheap as scalarizing and any
6011 // available vector variant.
6012 if (ID) {
6014 VPWidenIntrinsicRecipe::computeCallCost(ID, Ops, VPI, VF, CostCtx);
6015 if (IntrinsicCost.isValid() && ScalarCost >= IntrinsicCost &&
6016 (!VecFunc || VecCallCost >= IntrinsicCost))
6017 return CallWideningDecision::KindTy::Intrinsic;
6018 }
6019
6020 // Otherwise, use a vector library variant when it beats scalarizing.
6021 if (VecFunc && ScalarCost >= VecCallCost)
6022 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
6023
6024 return CallWideningDecision::KindTy::Scalarize;
6025}
6026
6028 VPRecipeBuilder &RecipeBuilder,
6029 VPCostContext &CostCtx) {
6032 for (VPInstruction &VPI :
6034 if (!VPI.getUnderlyingValue() || VPI.getOpcode() != Instruction::Call)
6035 continue;
6036
6037 auto *CI = cast<CallInst>(VPI.getUnderlyingInstr());
6038 SmallVector<VPValue *, 4> Ops(VPI.op_begin(),
6039 VPI.op_begin() + CI->arg_size());
6040
6041 CallWideningDecision Decision =
6042 decideCallWidening(VPI, Ops, Range.Start, CostCtx);
6044 [&](ElementCount VF) {
6045 return Decision == decideCallWidening(VPI, Ops, VF, CostCtx);
6046 },
6047 Range);
6048
6049 VPSingleDefRecipe *Replacement = nullptr;
6050 switch (Decision.Kind) {
6051 case CallWideningDecision::KindTy::Intrinsic: {
6053 Type *ResultTy = VPI.getScalarType();
6054 Replacement = new VPWidenIntrinsicRecipe(*CI, ID, Ops, ResultTy, VPI,
6055 VPI, VPI.getDebugLoc());
6056 break;
6057 }
6058 case CallWideningDecision::KindTy::VectorVariant: {
6059 // Masked variants take the mask as a trailing parameter, so they have
6060 // one more parameter than the original call's arguments.
6061 if (Decision.Variant->arg_size() > Ops.size()) {
6062 VPValue *Mask = VPI.isMasked() ? VPI.getMask() : Plan.getTrue();
6063 Ops.push_back(Mask);
6064 }
6065 Ops.push_back(VPI.getOperand(VPI.getNumOperandsWithoutMask() - 1));
6066 Replacement = new VPWidenCallRecipe(CI, Decision.Variant, Ops, VPI, VPI,
6067 VPI.getDebugLoc());
6068 break;
6069 }
6070 case CallWideningDecision::KindTy::Scalarize:
6071 Replacement = RecipeBuilder.handleReplication(&VPI, Range);
6072 break;
6073 }
6074
6075 Replacement->insertBefore(&VPI);
6076 VPI.replaceAllUsesWith(Replacement);
6077 VPI.eraseFromParent();
6078 }
6079 }
6080}
6081
6083 const TargetTransformInfo &TTI,
6085 VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
6086 VPBasicBlock *HeaderVPBB = LoopRegion->getEntryBasicBlock();
6088 vp_depth_first_shallow(LoopRegion->getEntry()))) {
6089 for (VPInstruction &VPI :
6091 // Only truncates are handled, as sext/zext may wrap, FP conversions lose
6092 // precision and other casts depend on the pointer size.
6093 if (VPI.getOpcode() != Instruction::Trunc)
6094 continue;
6095
6096 // Underlying Trunc is necessary to create VPWidenIntOrFpInductionRecipe.
6097 auto *Trunc = cast_or_null<TruncInst>(VPI.getUnderlyingValue());
6098 if (!Trunc)
6099 continue;
6100
6101 // A truncate that is not widened is left to the scalarization decisions
6102 // made earlier.
6104 continue;
6105
6106 VPValue *Op = VPI.getOperand(0);
6107 auto *WideIV = getOptimizableIVOf(Op, PSE);
6108 if (!WideIV)
6109 continue;
6110
6111 // getOptimizableIVOf also matches an add of the IV and its step, which
6112 // is not handled here.
6113 // TODO: Also narrow truncates of the incremented IV.
6114 if (Op != WideIV)
6115 continue;
6116
6117 // Replacing a free truncate would add an induction update instruction to
6118 // each iteration of the loop. The canonical induction is exempt, as it
6119 // needs an update instruction regardless.
6120 auto IsNarrowingProfitable = [&](ElementCount VF) {
6121 return match(WideIV, m_CanonicalWidenIV()) ||
6122 !TTI.isTruncateFree(
6123 toVectorTy(VPI.getOperand(0)->getScalarType(), VF),
6124 toVectorTy(VPI.getScalarType(), VF));
6125 };
6127 IsNarrowingProfitable, Range))
6128 continue;
6129
6130 // Wrap flags of the original induction do not hold in the truncated
6131 // type, so do not propagate them.
6132 auto *NarrowIV = new VPWidenIntOrFpInductionRecipe(
6133 WideIV->getPHINode(), WideIV->getStartValue(), WideIV->getStepValue(),
6134 WideIV->getVFValue(), WideIV->getInductionDescriptor(), Trunc,
6135 VPIRFlags::WrapFlagsTy(false, false), VPI.getDebugLoc());
6136 NarrowIV->insertBefore(*HeaderVPBB, HeaderVPBB->getFirstNonPhi());
6137 VPI.replaceAllUsesWith(NarrowIV);
6138 VPI.eraseFromParent();
6139 }
6140 }
6141}
6142
6145 Loop &L, VPCostContext &Ctx,
6146 VFRange &Range) {
6147 if (Plan.hasScalarVFOnly())
6148 return;
6149
6150 VPRegionBlock *VectorLoop = Plan.getVectorLoopRegion();
6151 VPValue *I32VF = nullptr;
6153 vp_depth_first_shallow(VectorLoop->getEntry()))) {
6154 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
6155 auto *MemR = dyn_cast<VPWidenMemoryRecipe>(&R);
6156 // TODO: Transform reverse access into strided access with -1 stride.
6157 // TODO: Transform gather/scatter with uniform address into strided access
6158 // with 0 stride.
6159 // TODO: Transform interleave access into multiple strided accesses.
6160 if (!MemR || MemR->isConsecutive())
6161 continue;
6162
6163 VPValue *Ptr = MemR->getAddr();
6164 // Check if this is a strided access by analyzing the address SCEV for an
6165 // affine addRec.
6166 const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, &L);
6167 const SCEV *Start;
6168 const SCEVConstant *Step;
6169 // TODO: Support non-constant loop invariant stride.
6170 if (!match(PtrSCEV,
6172 m_SpecificLoop(&L))))
6173 continue;
6174
6175 VPValue *StoredValue = nullptr;
6176 Type *DataTy;
6177 Intrinsic::ID IntrinID;
6178 if (auto *StoreR = dyn_cast<VPWidenStoreRecipe>(&R)) {
6179 StoredValue = StoreR->getStoredValue();
6180 DataTy = StoredValue->getScalarType();
6181 IntrinID = Intrinsic::experimental_vp_strided_store;
6182 } else {
6183 auto *LoadR = cast<VPWidenLoadRecipe>(&R);
6184 DataTy = LoadR->getScalarType();
6185 IntrinID = Intrinsic::experimental_vp_strided_load;
6186 }
6187
6188 Align Alignment = MemR->getAlign();
6189 auto IsProfitable = [&](ElementCount VF) {
6190 Type *VectorTy = toVectorTy(DataTy, VF);
6191 if (!Ctx.TTI.isLegalStridedLoadStore(VectorTy, Alignment))
6192 return false;
6193 const InstructionCost CurrentCost = MemR->computeCost(VF, Ctx);
6194 const InstructionCost StridedLoadStoreCost =
6196 IntrinID, VectorTy, MemR->isMasked(), Alignment, Ctx);
6197 return StridedLoadStoreCost < CurrentCost;
6198 };
6199
6201 Range))
6202 continue;
6203
6204 // Invalidate the legacy widening decision so the cost of replaced load is
6205 // not counted during precomputeCosts.
6206 // TODO: Remove once the legacy exit cost computation is retired.
6207 for (ElementCount VF : Range)
6208 Ctx.invalidateWideningDecision(&MemR->getIngredient(), VF);
6209
6210 // Get VF as i32 for the vector length operand.
6211 if (!I32VF) {
6212 VPBuilder Builder(Plan.getVectorPreheader());
6213 I32VF = Builder.createScalarZExtOrTrunc(
6214 &Plan.getVF(), Type::getInt32Ty(Plan.getContext()),
6216 }
6217
6218 VPBuilder Builder(&R);
6219 // Create the base pointer of strided access.
6220 // TODO: reuse VPDerivedIVRecipe for base pointer computation when it
6221 // supports a general VPValue as the start value.
6222 VPValue *StartVPV =
6223 VPSCEVExpander(Builder, *PSE.getSE(), R.getDebugLoc()).expand(Start);
6224 VPValue *StrideInBytes = Plan.getOrAddLiveIn(Step->getValue());
6225 Type *IndexTy = Plan.getDataLayout().getIndexType(Ptr->getScalarType());
6226 assert(IndexTy == StrideInBytes->getScalarType() &&
6227 "Stride type from SCEV must match the index type");
6228 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
6229 VectorLoop->getCanonicalIV(), IndexTy, DebugLoc::getUnknown());
6230 auto *AddRecPtr = cast<SCEVAddRecExpr>(PtrSCEV);
6231 auto *Offset = Builder.createOverflowingOp(
6232 Instruction::Mul, {CanIV, StrideInBytes},
6233 {AddRecPtr->hasNoUnsignedWrap(), /*HasNSW=*/false});
6234 GEPNoWrapFlags NWFlags = AddRecPtr->hasNoUnsignedWrap()
6237 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV, Offset, NWFlags);
6238
6239 // Create a new vector pointer for strided access.
6240 VPValue *NewPtr = Builder.createVectorPointer(
6241 BasePtr, Type::getInt8Ty(Plan.getContext()), StrideInBytes, NWFlags,
6242 R.getDebugLoc());
6243
6244 VPValue *Mask = MemR->getMask();
6245 if (!Mask)
6246 Mask = Plan.getTrue();
6248 if (StoredValue)
6249 Ops.push_back(StoredValue);
6250 Ops.append({NewPtr, StrideInBytes, Mask, I32VF});
6251
6252 auto *StridedR = Builder.createWidenMemIntrinsic(
6253 IntrinID, Ops,
6254 StoredValue ? Type::getVoidTy(Plan.getContext()) : DataTy, Alignment,
6255 *MemR, R.getDebugLoc());
6256 if (!StoredValue)
6257 cast<VPWidenLoadRecipe>(&R)->replaceAllUsesWith(StridedR);
6258 R.eraseFromParent();
6259 }
6260 }
6261}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
@ Default
Hexagon Common GEP
#define _
iv Induction Variable Users
Definition IVUsers.cpp:48
iv users
Definition IVUsers.cpp:48
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
licm
Definition LICM.cpp:378
Legalize the Machine IR a function s Machine IR
Definition Legalizer.cpp:85
#define I(x, y, z)
Definition MD5.cpp:57
This file provides utility analysis objects describing memory locations.
This file contains the declarations for metadata subclasses.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
#define P(N)
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
Func MI getDebugLoc()))
This file contains some templates that are useful if you are working with the STL at all.
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
static SmallVector< SmallVector< VPReplicateRecipe *, 4 > > collectComplementaryPredicatedMemOps(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
static void removeCommonBlendMask(VPBlendRecipe *Blend)
Try to see if all of Blend's masks share a common value logically and'ed and remove it from the masks...
static void tryToCreateAbstractReductionRecipe(VPReductionRecipe *Red, VPCostContext &Ctx, VFRange &Range)
This function tries to create abstract recipes from the reduction recipe for following optimizations ...
static VPReplicateRecipe * findRecipeWithMinAlign(ArrayRef< VPReplicateRecipe * > Group)
static CallWideningDecision decideCallWidening(VPInstruction &VPI, ArrayRef< VPValue * > Ops, ElementCount VF, VPCostContext &CostCtx)
Pick the cheapest widening for the call VPI at VF among scalarization, vector intrinsic,...
static bool areVFParamsOk(const VFInfo &Info, ArrayRef< VPValue * > Args, PredicatedScalarEvolution &PSE, const Loop *L)
Returns true if Info's parameter kinds are compatible with Args.
static bool sinkScalarOperands(VPlan &Plan)
static bool simplifyBranchConditionForVFAndUF(VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
Try to simplify the branch condition of Plan.
static void swapSelectBranchWeights(VPRecipeBase &R, VPlan &Plan)
Swap the branch weights recorded for R, a select whose two selected operands are being swapped,...
static VPValue * simplifyLogicalRecipe(VPlan &Plan, VPSingleDefRecipe *Def)
Try to simplify logical and bitwise recipes in Def.
static auto m_Countable(VPValue *&Cmp, PredicatedScalarEvolution &PSE, Loop *L)
Matches an exit condition formed by comparing the current value of a affine add recurrence in the giv...
static VPValue * cloneBinOpForScalarIV(VPWidenRecipe *BinOp, VPValue *ScalarIV, VPWidenIntOrFpInductionRecipe *WidenIV)
Create a scalar version of BinOp, with its WidenIV operand replaced by ScalarIV, and place it after S...
static VPWidenIntOrFpInductionRecipe * getExpressionIV(VPValue *V)
Check if V is a binary expression of a widened IV and a loop-invariant value.
static void removeRedundantInductionCasts(VPlan &Plan)
Remove redundant casts of inductions.
static bool isConditionTrueViaVFAndUF(VPValue *Cond, VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
Return true if Cond is known to be true for given BestVF and BestUF.
static VPExpressionRecipe * tryToMatchAndCreateExtendedReduction(VPReductionRecipe *Red, VPCostContext &Ctx, VFRange &Range)
This function tries convert extended in-loop reductions to VPExpressionRecipe and clamp the Range if ...
static bool isAvailableAtEndOf(VPValue *V, const VPBasicBlock *VPBB)
Returns true if V is available at the end of VPBB, i.e.
static std::optional< ElementCount > isConsecutiveInterleaveGroup(VPInterleaveRecipe *InterleaveR, ArrayRef< ElementCount > VFs, const TargetTransformInfo &TTI)
Returns VF from VFs if IR is a full interleave group with factor and number of members both equal to ...
static Type * getLoadStoreValueType(VPReplicateRecipe *R, bool IsLoad)
Get the value type of the replicate load or store.
static VPIRMetadata getCommonMetadata(ArrayRef< VPReplicateRecipe * > Recipes)
static bool mergeReplicateRegionsIntoSuccessors(VPlan &Plan)
static Function * findVectorVariant(CallInst *CI, ArrayRef< VPValue * > Args, ElementCount VF, bool MaskRequired, PredicatedScalarEvolution &PSE, const Loop *L)
Find a vector variant of CI for VF, respecting MaskRequired.
static VPValue * getRecipesForUncountableExit(SmallVectorImpl< VPInstruction * > &Recipes, VPBasicBlock *LatchVPBB)
Returns the VPValue representing the uncountable exit comparison used by AnyOf if the recipes it depe...
static VPWidenInductionRecipe * getOptimizableIVOf(VPValue *VPV, PredicatedScalarEvolution &PSE)
Check if VPV is an untruncated wide induction, either before or after the increment.
static bool canNarrowLoad(VPSingleDefRecipe *WideMember0, unsigned OpIdx, VPValue *OpV, unsigned Idx, bool IsScalable)
Returns true if V is VPWidenLoadRecipe or VPInterleaveRecipe that can be converted to a narrower reci...
static bool handleUncountableExitsWithSideEffects(VPlan &Plan, SmallVectorImpl< EarlyExitInfo > &Exits, VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB, VPBasicBlock *MiddleVPBB, OptimizationRemarkEmitter *ORE, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC)
Update Plan to mask memory operations in the loop based on whether the early exit is taken or not.
static void legalizeAndOptimizeInductions(VPlan &Plan)
Legalize VPWidenPointerInductionRecipe, by replacing it with a PtrAdd (IndStart, ScalarIVSteps (0,...
static void addReplicateRegions(VPlan &Plan)
static VPValue * optimizeLatchExitIVUserViaSCEV(VPlan &Plan, VPValue *Op, PredicatedScalarEvolution &PSE, VPValue *ResumeTC, const Loop *L)
static cl::opt< bool > UsePartialReductionsByDefault("use-partial-reductions-by-default", cl::init(false), cl::Hidden, cl::desc("Use partial reduction intrinsics for " "all supported unordered reductions."))
static SmallVector< SmallVector< VPReplicateRecipe *, 4 > > collectGroupedReplicateMemOps(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L, function_ref< bool(VPReplicateRecipe *)> FilterFn)
Collect either replicated Loads or Stores grouped by their address SCEV and their load-store type,...
static VPValue * tryToComputeEndValueForInduction(VPWidenInductionRecipe *WideIV, VPBuilder &VectorPHBuilder, VPValue *VectorTC)
Compute the end value for WideIV, unless it is truncated.
static bool replaceMaskWithCompareForScalarPlan(VPlan &Plan, ElementCount BestVF)
static void removeRedundantExpandSCEVRecipes(VPlan &Plan)
Remove redundant ExpandSCEVRecipes in Plan's entry block by replacing them with already existing reci...
static VPValue * simplifyRecipe(VPlan &Plan, VPSingleDefRecipe *Def)
Return an existing value or a live in for VPSingleDefRecipe Def if possible.
static VPValue * optimizeEarlyExitInductionUser(VPlan &Plan, VPValue *Op, PredicatedScalarEvolution &PSE)
Attempts to optimize the induction variable exit values for users in the early exit block.
static VPValue * narrowInterleaveGroupOp(ArrayRef< VPValue * > Members, SmallPtrSetImpl< VPValue * > &NarrowedOps, VPBasicBlock *Preheader)
static VPValue * optimizeLatchExitInductionUser(VPlan &Plan, VPValue *Op, DenseMap< VPValue *, VPValue * > &EndValues, PredicatedScalarEvolution &PSE)
Attempts to optimize the induction variable exit values for users in the exit block coming from the l...
static void reassociateHeaderMask(VPlan &Plan)
Reassociate (headermask && x) && y -> headermask && (x && y) to allow the header mask to be simplifie...
static VPSingleDefRecipe * combineRecipe(VPlan &Plan, VPSingleDefRecipe *Def, VPCombineBuilder &Builder)
Combine Def into a simpler recipe.
static VPBasicBlock * getPredicatedThenBlock(VPRegionBlock *R)
If R is a triangle region, return the 'then' block of the triangle.
static bool tryToRemoveDeadCycle(VPRecipeBase *R)
If R is a phi-like recipe starting a dead cycle of recipes, erase all reachable recipes of the dead c...
static bool canHoistOrSinkWithNoAliasCheck(const MemoryLocation &MemLoc, VPBasicBlock *FirstBB, VPBasicBlock *LastBB, std::optional< SinkStoreInfo > SinkInfo={})
Check if a memory operation doesn't alias with memory operations using scoped noalias metadata,...
static VPRegionBlock * createReplicateRegion(VPReplicateRecipe *PredRecipe, VPRegionBlock *ParentRegion, VPlan &Plan)
static void simplifyBlends(VPlan &Plan)
Normalize and simplify VPBlendRecipes.
static bool cannotHoistOrSinkRecipe(VPRecipeBase &R, VPBasicBlock *FirstBB, VPBasicBlock *LastBB, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink a non-memory or memory recipe R out...
static auto m_Uncountable(VPValue *&Cond)
Matches an exit condition formed by comparing a value loaded from memory with a loop-invariant term.
static std::optional< Instruction::BinaryOps > getUnmaskedDivRemOpcode(Intrinsic::ID ID)
static bool isAlreadyNarrow(VPValue *VPV)
Returns true if VPValue is a narrow VPValue.
static bool canNarrowOps(ArrayRef< VPValue * > Ops, bool IsScalable)
static bool optimizeVectorInductionWidthForTCAndVFUF(VPlan &Plan, ElementCount BestVF, unsigned BestUF)
Optimize the width of vector induction variables in Plan based on a known constant Trip Count,...
static VPExpressionRecipe * tryToMatchAndCreateMulAccumulateReduction(VPReductionRecipe *Red, VPCostContext &Ctx, VFRange &Range)
This function tries convert extended in-loop reductions to VPExpressionRecipe and clamp the Range if ...
static bool canSinkStoreWithNoAliasCheck(ArrayRef< VPReplicateRecipe * > StoresToSink, PredicatedScalarEvolution &PSE, const Loop &L)
static std::optional< bool > getStepDirection(const SCEV *S, ScalarEvolution &SE)
If S is an affine AddRec, returns true if its step is known to be positive and false if it is known t...
static VPIRMetadata getMetadataOf(VPRecipeBase *R)
Returns the metadata attached to R, or an empty set for a recipe that does not carry any.
static void narrowToSingleScalarRecipes(VPlan &Plan)
This file provides utility VPlan to VPlan transformations.
#define RUN_VPLAN_PASS(PASS,...)
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
Value * RHS
Value * LHS
BinaryOperator * Mul
static const uint32_t IV[8]
Definition blake3_impl.h:83
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1078
unsigned getActiveBits() const
Compute the number of active bits in the value.
Definition APInt.h:1532
APInt abs() const
Get the absolute value.
Definition APInt.h:1815
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
int32_t exactLogBase2() const
Definition APInt.h:1803
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:330
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
Definition APInt.cpp:1030
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
Definition APInt.h:436
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Definition APInt.h:1225
An arbitrary precision integer that knows its signedness.
Definition APSInt.h:24
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
Definition APSInt.h:310
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
Definition APSInt.h:302
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
const T & front() const
Get the first element.
Definition ArrayRef.h:144
size_t size() const
Get the array size.
Definition ArrayRef.h:141
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Definition InstrTypes.h:852
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
A debug info location.
Definition DebugLoc.h:126
static DebugLoc getUnknown()
Definition DebugLoc.h:153
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
Definition DenseMap.h:819
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
Definition DenseMap.h:809
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:872
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
static constexpr ElementCount getScalable(ScalarTy MinVal)
Definition TypeSize.h:308
constexpr bool isScalar() const
Exactly one element.
Definition TypeSize.h:316
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
size_t arg_size() const
Definition Function.h:886
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
bool isBinaryOp() const
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
bool isIntDivRem() const
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Definition VPlan.cpp:1638
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
LLVM_ABI MDNode * createBranchWeights(uint32_t TrueWeight, uint32_t FalseWeight, bool IsExpected=false)
Return metadata containing two branch weights.
Definition MDBuilder.cpp:38
This class implements a map that also provides access to all stored values in a deterministic order.
Definition MapVector.h:38
ValueT lookup(const KeyT &Key) const
Definition MapVector.h:110
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Definition MapVector.h:118
bool empty() const
Definition MapVector.h:79
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Definition Module.cpp:235
The optimization diagnostic interface.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
Definition RegionInfo.h:362
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This means that we are dealing with an entirely unknown SCEV value, and only represent it as its LLVM...
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEVFlags Flags=SCEV::FlagNone)
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEVFlags Flags=SCEV::FlagNone, unsigned Depth=0)
Return LHS-RHS.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEVFlags Flags=SCEV::FlagNone)
Return the SCEV object corresponding to -V.
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
Definition SetVector.h:57
size_type size() const
Determine the number of elements in the SetVector.
Definition SetVector.h:103
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
size_type size() const
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
iterator begin() const
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
static LLVM_ABI PartialReductionExtendKind getPartialReductionExtendKind(Instruction *I)
Get the kind of extension that an instruction represents.
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
LLVM_ABI InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, PartialReductionExtendKind OpAExtend, PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
Definition TypeSwitch.h:89
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
Definition TypeSwitch.h:98
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
Definition Type.cpp:272
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
Definition Type.h:265
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
op_range operands()
Definition User.h:267
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
Definition VectorUtils.h:76
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
Definition VPlan.h:4414
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
Definition VPlan.h:4489
RecipeListTy::iterator iterator
Instruction iterators...
Definition VPlan.h:4441
iterator end()
Definition VPlan.h:4451
iterator begin()
Recipe iterator methods.
Definition VPlan.h:4449
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
Definition VPlan.h:4502
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
Definition VPlan.cpp:233
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
Definition VPlan.cpp:540
const VPRecipeBase & front() const
Definition VPlan.h:4461
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
Definition VPlan.cpp:619
const VPRecipeBase & back() const
Definition VPlan.h:4463
void insert(VPRecipeBase *Recipe, iterator InsertPt)
Definition VPlan.h:4480
bool empty() const
Definition VPlan.h:4460
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
Definition VPlan.h:2953
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
Definition VPlan.h:3000
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
Definition VPlan.h:3005
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
Definition VPlan.h:2995
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
Definition VPlan.h:3011
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
Definition VPlan.h:2991
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
Definition VPlan.h:97
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
Definition VPlan.h:307
VPRegionBlock * getParent()
Definition VPlan.h:195
const VPBasicBlock * getExitingBasicBlock() const
Definition VPlan.cpp:203
size_t getNumSuccessors() const
Definition VPlan.h:245
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
Definition VPlan.h:298
const VPBlocksTy & getPredecessors() const
Definition VPlan.h:230
VPBlockBase * getSinglePredecessor() const
Definition VPlan.h:241
void clearPredecessors()
Remove all the predecessor of this block.
Definition VPlan.h:314
const VPBasicBlock * getEntryBasicBlock() const
Definition VPlan.cpp:188
VPBlockBase * getSingleSuccessor() const
Definition VPlan.h:235
const VPBlocksTy & getSuccessors() const
Definition VPlan.h:219
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
Definition VPlanUtils.h:421
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
Definition VPlanUtils.h:440
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static VPBasicBlock * getPlainCFGMiddleBlock(const VPlan &Plan)
Returns the middle block of Plan in plain CFG form (before regions are formed).
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
Definition VPlanUtils.h:330
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
Definition VPlanUtils.h:348
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
Definition VPlanUtils.h:366
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
Definition VPlanUtils.h:414
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
Definition VPlanUtils.h:398
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
Definition VPlan.h:3506
VPlan-based builder utility similar to IRBuilder.
VPInstruction * createFreeze(VPValue *Op, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
static VPBuilderBase getToInsertAfter(VPRecipeBase *R)
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Type *ResultTy, Instruction *UV)
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, std::optional< VPIRFlags > Flags=std::nullopt, const VPIRMetadata &Metadata={})
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
Definition VPlanValue.h:576
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
Definition VPlanValue.h:549
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
Definition VPlanValue.h:561
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Definition VPlanValue.h:571
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
LLVM_ABI_FOR_TEST bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
Recipe to expand a SCEV expression.
Definition VPlan.h:4027
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
Definition VPlan.h:3553
A pure virtual base class for all recipes modeling header phis, including phis for first order recurr...
Definition VPlan.h:2448
virtual VPValue * getBackedgeValue()
Returns the incoming value from the loop backedge.
Definition VPlan.h:2495
VPValue * getStartValue()
Returns the start value of the phi, if one is set.
Definition VPlan.h:2484
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
Definition VPlan.h:2170
A special type of VPBasicBlock that wraps an existing IR basic block.
Definition VPlan.h:4567
Class to record and manage LLVM IR flags.
Definition VPlan.h:696
static LLVM_ABI_FOR_TEST VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
Helper to manage IR metadata for recipes.
Definition VPlan.h:1181
std::optional< VPExecutionFrequency > getExecutionFrequency() const
Returns the frequency recorded by setExecutionFrequency, if any.
void intersect(const VPIRMetadata &MD)
Intersect this VPIRMetadata object with MD, keeping only metadata nodes that are common to both.
void clearExecutionFrequency()
Drop the frequency recorded by setExecutionFrequency, if any.
void setExecutionFrequency(std::optional< VPExecutionFrequency > Freq, LLVMContext &Ctx)
Record that the recipe executes with frequency Freq, relative to the entry of the loop region.
This is a concrete Recipe that models a single VPlan-level instruction.
Definition VPlan.h:1300
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
Definition VPlan.h:1541
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
Definition VPlan.h:1400
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
Definition VPlan.h:1396
@ BuildVector
Creates a fixed-width vector containing all operands.
Definition VPlan.h:1346
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
Definition VPlan.h:1354
unsigned getOpcode() const
Definition VPlan.h:1485
VPValue * getMask() const
Returns the mask for the VPInstruction.
Definition VPlan.h:1557
const InterleaveGroup< Instruction > * getInterleaveGroup() const
Definition VPlan.h:3106
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3098
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
Definition VPlan.h:3127
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
Definition VPlan.h:3137
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
Definition VPlan.h:3714
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
Definition VPlan.h:403
VPRegionBlock * getRegion()
Definition VPlan.h:4813
VPBasicBlock * getParent()
Definition VPlan.h:475
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
Definition VPlan.h:553
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
Definition VPlanValue.h:351
A recipe for handling reduction phis.
Definition VPlan.h:2863
bool isOrdered() const
Returns true, if the phi is part of an ordered reduction.
Definition VPlan.h:2923
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
Definition VPlan.h:2914
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:2907
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
Definition VPlan.h:2926
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
Definition VPlan.h:2920
A recipe to represent inloop, ordered or partial reduction operations.
Definition VPlan.h:3230
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
Definition VPlan.h:4639
const VPBlockBase * getEntry() const
Definition VPlan.h:4683
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
Definition VPlan.h:4715
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Definition VPlan.h:4700
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
Definition VPlan.h:4767
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
Definition VPlan.h:4759
const VPBlockBase * getExiting() const
Definition VPlan.h:4695
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
Definition VPlan.h:4772
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
Definition VPlan.h:3397
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
Definition VPlan.h:3456
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
Definition VPlan.h:3484
bool isPredicated() const
Definition VPlan.h:3461
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Definition VPlan.h:3478
Lightweight SCEV-to-VPlan expander.
Definition VPlanUtils.h:268
VPValue * expand(const SCEV *S)
Expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
Definition VPlan.h:4256
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Definition VPlan.h:611
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
Definition VPlan.h:681
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
Definition VPlanValue.h:214
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
Definition VPlanValue.h:398
operand_range operands()
Definition VPlanValue.h:471
void setOperand(unsigned I, VPValue *New)
Definition VPlanValue.h:444
unsigned getNumOperands() const
Definition VPlanValue.h:438
VPValue * getOperand(unsigned N) const
Definition VPlanValue.h:439
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Definition VPlanValue.h:50
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Definition VPlan.cpp:147
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
Definition VPlan.cpp:141
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
Definition VPlan.cpp:1461
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
Definition VPlan.cpp:128
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Definition VPlanValue.h:164
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
Definition VPlanValue.h:75
bool user_empty() const
Definition VPlanValue.h:161
bool hasOneUse() const
Definition VPlanValue.h:175
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
Definition VPlanValue.h:179
void replaceAllUsesWith(VPValue *New)
Definition VPlan.cpp:1464
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
Definition VPlan.cpp:1470
user_range users()
Definition VPlanValue.h:157
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
Definition VPlan.h:2278
A recipe for widening Call instructions using library calls.
Definition VPlan.h:2104
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Definition VPlan.h:1885
Instruction::CastOps getOpcode() const
Definition VPlan.h:1921
A recipe for handling GEP instructions.
Definition VPlan.h:2218
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
Definition VPlan.h:2520
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
Definition VPlan.h:2580
VPValue * getStepValue()
Returns the step value of the induction.
Definition VPlan.h:2568
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
Definition VPlan.h:2585
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
Definition VPlan.h:2614
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
Definition VPlan.h:2673
A recipe for widening vector intrinsics.
Definition VPlan.h:1933
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
Definition VPlan.h:3750
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
Definition VPlan.h:2750
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
Definition VPlan.h:1818
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:1844
unsigned getOpcode() const
Definition VPlan.h:1863
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
Definition VPlan.h:4826
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
Definition VPlan.h:5169
bool hasVF(ElementCount VF) const
Definition VPlan.h:5062
const DataLayout & getDataLayout() const
Definition VPlan.h:5040
LLVMContext & getContext() const
Definition VPlan.h:5036
VPBasicBlock * getEntry()
Definition VPlan.h:4922
bool hasScalableVF() const
Definition VPlan.h:5063
VPValue * getTripCount() const
The trip count of the original loop.
Definition VPlan.h:4994
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
Definition VPlan.h:5015
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
Definition VPlan.h:5069
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
Definition VPlan.h:5135
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
Definition VPlan.h:5034
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
Definition VPlan.h:5141
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
Definition VPlan.h:5222
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
Definition VPlan.h:5172
bool hasUF(unsigned UF) const
Definition VPlan.h:5087
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
Definition VPlan.h:4988
VPSymbolicValue & getVectorTripCount()
The vector trip count.
Definition VPlan.h:5024
VPValue * getBackedgeTakenCount() const
Definition VPlan.h:5021
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
Definition VPlan.h:5112
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
Definition VPlan.h:5138
void setVF(ElementCount VF)
Definition VPlan.h:5050
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
Definition VPlan.h:5103
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
Definition VPlan.cpp:1042
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
Definition VPlan.h:5090
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
Definition VPlan.h:5008
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
Definition VPlan.h:4964
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
Definition VPlan.h:5195
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
Definition VPlan.h:5132
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
Definition VPlan.h:4927
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
Definition VPlan.h:5031
bool hasScalarVFOnly() const
Definition VPlan.h:5080
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
Definition VPlan.h:4978
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
Definition VPlan.h:4943
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
Definition VPlan.h:5027
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
Definition VPlan.cpp:1207
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
Definition VPlan.h:5146
LLVM Value Representation.
Definition Value.h:75
iterator_range< user_iterator > users()
Definition Value.h:428
bool hasName() const
Definition Value.h:263
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS*X will result in a value whose quantity matches our ...
Definition TypeSize.h:265
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS*X will result in a value whose quantity matches our own.
Definition TypeSize.h:273
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
Definition ilist_node.h:123
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
Definition APInt.cpp:2801
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
Definition DwarfDebug.h:190
void reportVectorizationFailure(const StringRef DebugMsg, const StringRef OREMsg, const StringRef ORETag, OptimizationRemarkEmitter *ORE, const Loop *TheLoop, Instruction *I=nullptr)
Reports a vectorization failure: print DebugMsg for debugging purposes along with the corresponding o...
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::AShr > m_AShr(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
OneOps_match< OpTy, Instruction::Freeze > m_Freeze(const OpTy &Op)
Matches FreezeInst.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
CmpClass_match< LHS, RHS, CmpInst, true > m_c_Cmp(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst, true > m_c_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
Matches an ICmp with a predicate over LHS and RHS in either order.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
LogicalOp_match< LHS, RHS, Instruction::Or, true > m_c_LogicalOr(const LHS &L, const RHS &R)
Matches L || R with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
cst_pred_ty< is_one > m_scev_One()
Match an integer 1.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
VPInstruction_match< VPInstruction::WideActiveLaneMask, Op0_t, Op1_t, Op2_t > m_WideActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
int_pred_ty< is_zero_int, 1 > m_False()
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractVectorForPart, Op0_t, Op1_t > m_ExtractVectorForPart(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
bool match(Val *V, const Pattern &P)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
int_pred_ty< is_one, 1 > m_True()
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
initializer< Ty > init(const Ty &Val)
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
std::optional< int64_t > getConstantStride(VPValue *Addr, Type *AccessTy, PredicatedScalarEvolution &PSE, const Loop *L)
If the pointer operand Addr of a memory access is an affine AddRec w.r.t.
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
Definition VPlanUtils.h:146
LLVM_ABI_FOR_TEST bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
LLVM_ABI_FOR_TEST const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
Definition VPlanUtils.h:254
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
Definition VPlanCFG.h:262
@ Offset
Definition DWP.cpp:577
void stable_sort(R &&Range)
Definition STLExtras.h:2132
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
Definition STLExtras.h:2094
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1685
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
Definition STLExtras.h:856
ReductionStyle getReductionStyle(bool InLoop, bool Ordered, unsigned ScaleFactor)
Definition VPlan.h:2850
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
VPBuilderBase<> VPBuilder
Definition VPlan.h:67
constexpr from_range_t from_range
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
auto cast_or_null(const Y &Val)
Definition Casting.h:714
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
Definition VPlanCFG.h:250
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
Definition VPlanCFG.h:285
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
Definition STLExtras.h:2189
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
Definition STLExtras.h:366
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
auto make_isa_range(RangeT &&Range)
Return a range over Range containing only elements for which isa<T> holds, casting each of them to T.
Definition STLExtras.h:567
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
Definition STLExtras.h:2216
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
auto reverse(ContainerTy &&C)
Definition STLExtras.h:408
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
Definition STLExtras.h:1710
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
Definition VPlan.h:83
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
Definition VPlan.h:87
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
Definition VPlan.h:92
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
Definition STLExtras.h:552
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
Definition VPlan.cpp:1820
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
Definition STLExtras.h:1853
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
@ Other
Any other memory.
Definition ModRef.h:68
TargetTransformInfo TTI
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ FMul
Product of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ FAdd
Sum of floats.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
Definition STLExtras.h:2028
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
Definition STLExtras.h:2104
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI bool extractBranchWeights(const MDNode *ProfileData, SmallVectorImpl< uint32_t > &Weights)
Extract branch weights from MD_prof metadata.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
Definition STLExtras.h:2182
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
Definition Hashing.h:307
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2162
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
Definition Loads.cpp:304
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
Definition Hashing.h:287
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Definition VPlan.h:2845
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
Definition VPlan.cpp:1924
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A recipe for handling first-order recurrence phis.
Definition VPlan.h:2801
A VPValue representing a live-in from the input IR or a constant.
Definition VPlanValue.h:276
Type * getType() const
Returns the type of the underlying IR value.
Definition VPlan.cpp:145
A recipe for widening load operations, using the address to load from and an optional mask.
Definition VPlan.h:3814
A recipe for widening store operations, using the stored value, the address to store to and an option...
Definition VPlan.h:3919
static void simplifyLiveInsWithSCEV(VPlan &Plan, PredicatedScalarEvolution &PSE)
Check Plan's live-ins and replace them with constants, if they can be simplified via SCEV.
static decltype(auto) runPass(StringRef PassName, PassTy &&Pass, VPlan &Plan, ArgsTy &&...Args)
Helper to run a VPlan pass Pass on VPlan, forwarding extra arguments to the pass.
static void createInterleaveGroups(VPlan &Plan, const SmallPtrSetImpl< const InterleaveGroup< Instruction > * > &InterleaveGroups, const bool &EpilogueAllowed)
static LLVM_ABI_FOR_TEST bool handleUncountableEarlyExits(VPlan &Plan, OptimizationRemarkEmitter *ORE, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style)
Update Plan to account for uncountable early exits by introducing appropriate branching logic in the ...
static LLVM_ABI_FOR_TEST bool tryToConvertVPInstructionsToVPRecipes(VPlan &Plan, const TargetLibraryInfo &TLI, PredicatedScalarEvolution &PSE, Loop *OuterLoop)
Replaces the VPInstructions in Plan with corresponding widen recipes.
static void createAndOptimizeReplicateRegions(VPlan &Plan)
Wrap predicated VPReplicateRecipes with a mask operand in an if-then region block and remove the mask...
static std::unique_ptr< VPlan > narrowInterleaveGroups(VPlan &Plan, const TargetTransformInfo &TTI)
Try to find a single VF among Plan's VFs for which all interleave groups (with known minimum VF eleme...
static void makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder, VPCostContext &CostCtx)
Convert load/store VPInstructions in Plan into widened or replicate recipes.
static void narrowInductionTruncates(VPlan &Plan, VFRange &Range, const TargetTransformInfo &TTI, PredicatedScalarEvolution &PSE)
Replace truncates of a wide induction, or of that induction's increment, by a VPWidenIntOrFpInduction...
static void hoistPredicatedLoads(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
Hoist predicated loads from the same address to the loop entry block, if they are guaranteed to execu...
static bool mergeBlocksIntoPredecessors(VPlan &Plan)
Remove redundant VPBasicBlocks by merging them into their single predecessor if the latter has a sing...
static void optimizeFindIVReductions(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &L)
Optimize FindLast reductions selecting IVs (or expressions of IVs) by converting them to FindIV reduc...
static void convertToAbstractRecipes(VPlan &Plan, VPCostContext &Ctx, VFRange &Range)
This function converts initial recipes to the abstract recipes and clamps Range based on cost model f...
static void makeScalarizationDecisions(VPlan &Plan, VFRange &Range)
Make VPlan-based scalarization decision prior to delegating to the ones made by the legacy CM.
static bool areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC)
Check if all loads in the loop are dereferenceable.
static void optimizeInductionLiveOutUsers(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
If there's a single exit block, optimize its phi recipes that use exiting IV values by feeding them p...
static void simplifyReverses(VPlan &Plan)
Cancel out redundant reverses in Plan, e.g. reverse(reverse(x)) -> x.
static void makeCallWideningDecisions(VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder, VPCostContext &CostCtx)
Convert call VPInstructions in Plan into widened call, vector intrinsic or replicate recipes based on...
static void adjustFirstOrderRecurrenceMiddleUsers(VPlan &Plan, VFRange &Range)
Adjust first-order recurrence users in the middle block: create penultimate element extracts for LCSS...
static void removeDeadRecipes(VPlan &Plan)
Remove dead recipes from Plan.
static void sinkPredicatedStores(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
Sink predicated stores to the same address with complementary predicates (P and NOT P) to an uncondit...
static bool removeBranchOnConst(VPlan &Plan, bool OnlyLatches=false)
Remove BranchOnCond recipes with true or false conditions together with removing dead edges to their ...
static void convertToStridedAccesses(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &L, VPCostContext &Ctx, VFRange &Range)
Transform widen memory recipes into strided access recipes when legal and profitable.
static void clearReductionWrapFlags(VPlan &Plan)
Clear NSW/NUW flags from reduction instructions if necessary.
static void createPartialReductions(VPlan &Plan, VPCostContext &CostCtx, VFRange &Range)
Detect and create partial reduction recipes for scaled or unordered reductions in Plan.
static void cse(VPlan &Plan)
Perform common-subexpression-elimination on Plan.
static void replaceSymbolicStrides(VPlan &Plan, PredicatedScalarEvolution &PSE, const SymbolicStrideMap &StridesMap, const VPDominatorTree &VPDT)
Replace symbolic strides from StridesMap in Plan with constants when possible.
static LLVM_ABI_FOR_TEST void optimize(VPlan &Plan)
Apply VPlan-to-VPlan optimizations to Plan, including induction recipe optimizations,...
static void truncateToMinimalBitwidths(VPlan &Plan, const MapVector< Instruction *, uint64_t > &MinBWs)
Insert truncates and extends for any truncated recipe.
static void dropPoisonGeneratingRecipes(VPlan &Plan)
Drop poison flags from recipes that may generate a poison value that is used after vectorization,...
static void optimizeForVFAndUF(VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
Optimize Plan based on BestVF and BestUF.
static bool splitCombinedExits(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop *TheLoop)
If a single exit has multiple conditions combined together, split them and create new exiting blocks.
static void combineRecipes(VPlan &Plan)
Perform instcombine-like simplifications on recipes in Plan.