LLVM 24.0.0git
MVEGatherScatterLowering.cpp
Go to the documentation of this file.
1//===- MVEGatherScatterLowering.cpp - Gather/Scatter lowering -------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// This pass custom lowers llvm.gather and llvm.scatter instructions to
10/// arm.mve.gather and arm.mve.scatter intrinsics, optimising the code to
11/// produce a better final result as we go.
12//
13//===----------------------------------------------------------------------===//
14
15#include "ARM.h"
16#include "ARMBaseInstrInfo.h"
17#include "ARMSubtarget.h"
24#include "llvm/IR/BasicBlock.h"
25#include "llvm/IR/Constant.h"
26#include "llvm/IR/Constants.h"
28#include "llvm/IR/Function.h"
29#include "llvm/IR/IRBuilder.h"
30#include "llvm/IR/InstrTypes.h"
31#include "llvm/IR/Instruction.h"
34#include "llvm/IR/Intrinsics.h"
35#include "llvm/IR/IntrinsicsARM.h"
37#include "llvm/IR/Type.h"
38#include "llvm/IR/Value.h"
40#include "llvm/Pass.h"
43#include <cassert>
44
45using namespace llvm;
46
47#define DEBUG_TYPE "arm-mve-gather-scatter-lowering"
48
50 "enable-arm-maskedgatscat", cl::Hidden, cl::init(true),
51 cl::desc("Enable the generation of masked gathers and scatters"));
52
53namespace {
54
55class MVEGatherScatterLowering : public FunctionPass {
56public:
57 static char ID; // Pass identification, replacement for typeid
58
59 explicit MVEGatherScatterLowering() : FunctionPass(ID) {
61 }
62
63 bool runOnFunction(Function &F) override;
64
65 StringRef getPassName() const override {
66 return "MVE gather/scatter lowering";
67 }
68
69 void getAnalysisUsage(AnalysisUsage &AU) const override {
70 AU.setPreservesCFG();
71 AU.addRequired<TargetPassConfig>();
72 AU.addRequired<LoopInfoWrapperPass>();
73 FunctionPass::getAnalysisUsage(AU);
74 }
75
76private:
77 LoopInfo *LI = nullptr;
78 const DataLayout *DL;
79
80 // Check this is a valid gather with correct alignment
81 bool isLegalTypeAndAlignment(unsigned NumElements, unsigned ElemSize,
82 Align Alignment);
83 // Check whether Ptr is hidden behind a bitcast and look through it
84 void lookThroughBitcast(Value *&Ptr);
85 // Decompose a ptr into Base and Offsets, potentially using a GEP to return a
86 // scalar base and vector offsets, or else fallback to using a base of 0 and
87 // offset of Ptr where possible.
88 Value *decomposePtr(Value *Ptr, Value *&Offsets, int &Scale,
89 FixedVectorType *Ty, Type *MemoryTy,
90 IRBuilder<> &Builder);
91 // Check for a getelementptr and deduce base and offsets from it, on success
92 // returning the base directly and the offsets indirectly using the Offsets
93 // argument
94 Value *decomposeGEP(Value *&Offsets, FixedVectorType *Ty,
95 GetElementPtrInst *GEP, IRBuilder<> &Builder);
96 // Compute the scale of this gather/scatter instruction
97 int computeScale(unsigned GEPElemSize, unsigned MemoryElemSize);
98 // If the value is a constant, or derived from constants via additions
99 // and multilications, return its numeric value
100 std::optional<int64_t> getIfConst(const Value *V);
101 // If Inst is an add instruction, check whether one summand is a
102 // constant. If so, scale this constant and return it together with
103 // the other summand.
104 std::pair<Value *, int64_t> getVarAndConst(Value *Inst, int TypeScale);
105
106 Instruction *lowerGather(IntrinsicInst *I);
107 // Create a gather from a base + vector of offsets
108 Instruction *tryCreateMaskedGatherOffset(IntrinsicInst *I, Value *Ptr,
109 Instruction *&Root,
110 IRBuilder<> &Builder);
111 // Create a gather from a vector of pointers
112 Instruction *tryCreateMaskedGatherBase(IntrinsicInst *I, Value *Ptr,
113 IRBuilder<> &Builder,
114 int64_t Increment = 0);
115 // Create an incrementing gather from a vector of pointers
116 Instruction *tryCreateMaskedGatherBaseWB(IntrinsicInst *I, Value *Ptr,
117 IRBuilder<> &Builder,
118 int64_t Increment = 0);
119
120 Instruction *lowerScatter(IntrinsicInst *I);
121 // Create a scatter to a base + vector of offsets
122 Instruction *tryCreateMaskedScatterOffset(IntrinsicInst *I, Value *Offsets,
123 IRBuilder<> &Builder);
124 // Create a scatter to a vector of pointers
125 Instruction *tryCreateMaskedScatterBase(IntrinsicInst *I, Value *Ptr,
126 IRBuilder<> &Builder,
127 int64_t Increment = 0);
128 // Create an incrementing scatter from a vector of pointers
129 Instruction *tryCreateMaskedScatterBaseWB(IntrinsicInst *I, Value *Ptr,
130 IRBuilder<> &Builder,
131 int64_t Increment = 0);
132
133 // QI gathers and scatters can increment their offsets on their own if
134 // the increment is a constant value (digit)
135 Instruction *tryCreateIncrementingGatScat(IntrinsicInst *I, Value *Ptr,
136 IRBuilder<> &Builder);
137 // QI gathers/scatters can increment their offsets on their own if the
138 // increment is a constant value (digit) - this creates a writeback QI
139 // gather/scatter
140 Instruction *tryCreateIncrementingWBGatScat(IntrinsicInst *I, Value *BasePtr,
141 Value *Ptr, unsigned TypeScale,
142 IRBuilder<> &Builder);
143
144 // Optimise the base and offsets of the given address
145 bool optimiseAddress(Value *Address, BasicBlock *BB, LoopInfo *LI);
146 // Try to fold consecutive geps together into one
147 Value *foldGEP(GetElementPtrInst *GEP, Value *&Offsets, unsigned &Scale,
148 IRBuilder<> &Builder);
149 // Check whether these offsets could be moved out of the loop they're in
150 bool optimiseOffsets(Value *Offsets, BasicBlock *BB, LoopInfo *LI);
151 // Pushes the given add out of the loop
152 void pushOutAdd(PHINode *&Phi, Value *OffsSecondOperand, unsigned StartIndex);
153 // Pushes the given mul or shl out of the loop
154 void pushOutMulShl(unsigned Opc, PHINode *&Phi, Value *IncrementPerRound,
155 Value *OffsSecondOperand, unsigned LoopIncrement,
156 IRBuilder<> &Builder);
157};
158
159} // end anonymous namespace
160
161char MVEGatherScatterLowering::ID = 0;
162
163INITIALIZE_PASS(MVEGatherScatterLowering, DEBUG_TYPE,
164 "MVE gather/scattering lowering pass", false, false)
165
167 return new MVEGatherScatterLowering();
168}
169
170bool MVEGatherScatterLowering::isLegalTypeAndAlignment(unsigned NumElements,
171 unsigned ElemSize,
172 Align Alignment) {
173 if (((NumElements == 4 &&
174 (ElemSize == 32 || ElemSize == 16 || ElemSize == 8)) ||
175 (NumElements == 8 && (ElemSize == 16 || ElemSize == 8)) ||
176 (NumElements == 16 && ElemSize == 8)) &&
177 Alignment >= ElemSize / 8)
178 return true;
179 LLVM_DEBUG(dbgs() << "masked gathers/scatters: instruction does not have "
180 << "valid alignment or vector type \n");
181 return false;
182}
183
184static bool checkOffsetSize(Value *Offsets, unsigned TargetElemCount) {
185 // Offsets that are not of type <N x i32> are sign extended by the
186 // getelementptr instruction, and MVE gathers/scatters treat the offset as
187 // unsigned. Thus, if the element size is smaller than 32, we can only allow
188 // positive offsets - i.e., the offsets are not allowed to be variables we
189 // can't look into.
190 // Additionally, <N x i32> offsets have to either originate from a zext of a
191 // vector with element types smaller or equal the type of the gather we're
192 // looking at, or consist of constants that we can check are small enough
193 // to fit into the gather type.
194 // Thus we check that 0 < value < 2^TargetElemSize.
195 unsigned TargetElemSize = 128 / TargetElemCount;
196 unsigned OffsetElemSize = cast<FixedVectorType>(Offsets->getType())
197 ->getElementType()
198 ->getScalarSizeInBits();
199 if (OffsetElemSize != TargetElemSize || OffsetElemSize != 32) {
200 Constant *ConstOff = dyn_cast<Constant>(Offsets);
201 if (!ConstOff)
202 return false;
203 int64_t TargetElemMaxSize = (1ULL << TargetElemSize);
204 auto CheckValueSize = [TargetElemMaxSize](Value *OffsetElem) {
205 ConstantInt *OConst = dyn_cast<ConstantInt>(OffsetElem);
206 if (!OConst)
207 return false;
208 int SExtValue = OConst->getSExtValue();
209 if (SExtValue >= TargetElemMaxSize || SExtValue < 0)
210 return false;
211 return true;
212 };
213 if (isa<FixedVectorType>(ConstOff->getType())) {
214 for (unsigned i = 0; i < TargetElemCount; i++) {
215 if (!CheckValueSize(ConstOff->getAggregateElement(i)))
216 return false;
217 }
218 } else {
219 if (!CheckValueSize(ConstOff))
220 return false;
221 }
222 }
223 return true;
224}
225
226Value *MVEGatherScatterLowering::decomposePtr(Value *Ptr, Value *&Offsets,
227 int &Scale, FixedVectorType *Ty,
228 Type *MemoryTy,
229 IRBuilder<> &Builder) {
230 if (auto *GEP = dyn_cast<GetElementPtrInst>(Ptr)) {
231 if (Value *V = decomposeGEP(Offsets, Ty, GEP, Builder)) {
232 Scale =
233 computeScale(DL->getTypeAllocSizeInBits(GEP->getSourceElementType()),
234 MemoryTy->getScalarSizeInBits());
235 return Scale == -1 ? nullptr : V;
236 }
237 }
238
239 // If we couldn't use the GEP (or it doesn't exist), attempt to use a
240 // BasePtr of 0 with Ptr as the Offsets, so long as there are only 4
241 // elements.
242 FixedVectorType *PtrTy = cast<FixedVectorType>(Ptr->getType());
243 if (PtrTy->getNumElements() != 4 || MemoryTy->getScalarSizeInBits() == 32)
244 return nullptr;
245 Value *Zero = ConstantInt::get(Builder.getInt32Ty(), 0);
246 Value *BasePtr = Builder.CreateIntToPtr(Zero, Builder.getPtrTy());
247 Offsets = Builder.CreatePtrToInt(
248 Ptr, FixedVectorType::get(Builder.getInt32Ty(), 4));
249 Scale = 0;
250 return BasePtr;
251}
252
253Value *MVEGatherScatterLowering::decomposeGEP(Value *&Offsets,
254 FixedVectorType *Ty,
255 GetElementPtrInst *GEP,
256 IRBuilder<> &Builder) {
257 if (!GEP) {
258 LLVM_DEBUG(dbgs() << "masked gathers/scatters: no getelementpointer "
259 << "found\n");
260 return nullptr;
261 }
262 LLVM_DEBUG(dbgs() << "masked gathers/scatters: getelementpointer found."
263 << " Looking at intrinsic for base + vector of offsets\n");
264 Value *GEPPtr = GEP->getPointerOperand();
265 Offsets = GEP->getOperand(1);
266 if (GEPPtr->getType()->isVectorTy() ||
267 !isa<FixedVectorType>(Offsets->getType()))
268 return nullptr;
269
270 if (GEP->getNumOperands() != 2) {
271 LLVM_DEBUG(dbgs() << "masked gathers/scatters: getelementptr with too many"
272 << " operands. Expanding.\n");
273 return nullptr;
274 }
275 Offsets = GEP->getOperand(1);
276 unsigned OffsetsElemCount =
277 cast<FixedVectorType>(Offsets->getType())->getNumElements();
278 // Paranoid check whether the number of parallel lanes is the same
279 assert(Ty->getNumElements() == OffsetsElemCount);
280
281 ZExtInst *ZextOffs = dyn_cast<ZExtInst>(Offsets);
282 if (ZextOffs)
283 Offsets = ZextOffs->getOperand(0);
284 FixedVectorType *OffsetType = cast<FixedVectorType>(Offsets->getType());
285
286 // If the offsets are already being zext-ed to <N x i32>, that relieves us of
287 // having to make sure that they won't overflow.
288 if (!ZextOffs || cast<FixedVectorType>(ZextOffs->getDestTy())
289 ->getElementType()
290 ->getScalarSizeInBits() != 32)
291 if (!checkOffsetSize(Offsets, OffsetsElemCount))
292 return nullptr;
293
294 // The offset sizes have been checked; if any truncating or zext-ing is
295 // required to fix them, do that now
296 if (Ty != Offsets->getType()) {
297 if ((Ty->getElementType()->getScalarSizeInBits() <
298 OffsetType->getElementType()->getScalarSizeInBits())) {
299 Offsets = Builder.CreateTrunc(Offsets, Ty);
300 } else {
301 Offsets = Builder.CreateZExt(Offsets, VectorType::getInteger(Ty));
302 }
303 }
304 // If none of the checks failed, return the gep's base pointer
305 LLVM_DEBUG(dbgs() << "masked gathers/scatters: found correct offsets\n");
306 return GEPPtr;
307}
308
309void MVEGatherScatterLowering::lookThroughBitcast(Value *&Ptr) {
310 // Look through bitcast instruction if #elements is the same
311 if (auto *BitCast = dyn_cast<BitCastInst>(Ptr)) {
312 auto *BCTy = cast<FixedVectorType>(BitCast->getType());
313 auto *BCSrcTy = cast<FixedVectorType>(BitCast->getOperand(0)->getType());
314 if (BCTy->getNumElements() == BCSrcTy->getNumElements()) {
315 LLVM_DEBUG(dbgs() << "masked gathers/scatters: looking through "
316 << "bitcast\n");
317 Ptr = BitCast->getOperand(0);
318 }
319 }
320}
321
322int MVEGatherScatterLowering::computeScale(unsigned GEPElemSize,
323 unsigned MemoryElemSize) {
324 // This can be a 32bit load/store scaled by 4, a 16bit load/store scaled by 2,
325 // or a 8bit, 16bit or 32bit load/store scaled by 1
326 if (GEPElemSize == 32 && MemoryElemSize == 32)
327 return 2;
328 else if (GEPElemSize == 16 && MemoryElemSize == 16)
329 return 1;
330 else if (GEPElemSize == 8)
331 return 0;
332 LLVM_DEBUG(dbgs() << "masked gathers/scatters: incorrect scale. Can't "
333 << "create intrinsic\n");
334 return -1;
335}
336
337std::optional<int64_t> MVEGatherScatterLowering::getIfConst(const Value *V) {
338 const Constant *C = dyn_cast<Constant>(V);
339 if (C && C->getSplatValue())
340 return std::optional<int64_t>{C->getUniqueInteger().getSExtValue()};
341 if (!isa<Instruction>(V))
342 return std::optional<int64_t>{};
343
344 const Instruction *I = cast<Instruction>(V);
345 if (I->getOpcode() == Instruction::Add || I->getOpcode() == Instruction::Or ||
346 I->getOpcode() == Instruction::Mul ||
347 I->getOpcode() == Instruction::Shl) {
348 std::optional<int64_t> Op0 = getIfConst(I->getOperand(0));
349 std::optional<int64_t> Op1 = getIfConst(I->getOperand(1));
350 if (!Op0 || !Op1)
351 return std::optional<int64_t>{};
352 if (I->getOpcode() == Instruction::Add)
353 return std::optional<int64_t>{*Op0 + *Op1};
354 if (I->getOpcode() == Instruction::Mul)
355 return std::optional<int64_t>{*Op0 * *Op1};
356 if (I->getOpcode() == Instruction::Shl)
357 return std::optional<int64_t>{*Op0 << *Op1};
358 if (I->getOpcode() == Instruction::Or)
359 return std::optional<int64_t>{*Op0 | *Op1};
360 }
361 return std::optional<int64_t>{};
362}
363
364// Return true if I is an Or instruction that is equivalent to an add, due to
365// the operands having no common bits set.
366static bool isAddLikeOr(Instruction *I, const DataLayout &DL) {
367 return I->getOpcode() == Instruction::Or &&
368 haveNoCommonBitsSet(I->getOperand(0), I->getOperand(1), DL);
369}
370
371std::pair<Value *, int64_t>
372MVEGatherScatterLowering::getVarAndConst(Value *Inst, int TypeScale) {
373 std::pair<Value *, int64_t> ReturnFalse =
374 std::pair<Value *, int64_t>(nullptr, 0);
375 // At this point, the instruction we're looking at must be an add or an
376 // add-like-or.
378 if (Add == nullptr ||
379 (Add->getOpcode() != Instruction::Add && !isAddLikeOr(Add, *DL)))
380 return ReturnFalse;
381
382 Value *Summand;
383 std::optional<int64_t> Const;
384 // Find out which operand the value that is increased is
385 if ((Const = getIfConst(Add->getOperand(0))))
386 Summand = Add->getOperand(1);
387 else if ((Const = getIfConst(Add->getOperand(1))))
388 Summand = Add->getOperand(0);
389 else
390 return ReturnFalse;
391
392 // Check that the constant is small enough for an incrementing gather
393 int64_t Immediate = *Const << TypeScale;
394 if (Immediate > 512 || Immediate < -512 || Immediate % 4 != 0)
395 return ReturnFalse;
396
397 return std::pair<Value *, int64_t>(Summand, Immediate);
398}
399
400Instruction *MVEGatherScatterLowering::lowerGather(IntrinsicInst *I) {
401 using namespace PatternMatch;
402 LLVM_DEBUG(dbgs() << "masked gathers: checking transform preconditions\n"
403 << *I << "\n");
404
405 // @llvm.masked.gather.*(Ptrs, alignment, Mask, Src0)
406 // Attempt to turn the masked gather in I into a MVE intrinsic
407 // Potentially optimising the addressing modes as we do so.
408 auto *Ty = cast<FixedVectorType>(I->getType());
409 Value *Ptr = I->getArgOperand(0);
410 Align Alignment = I->getParamAlign(0).valueOrOne();
411 Value *Mask = I->getArgOperand(1);
412 Value *PassThru = I->getArgOperand(2);
413
414 if (!isLegalTypeAndAlignment(Ty->getNumElements(), Ty->getScalarSizeInBits(),
415 Alignment))
416 return nullptr;
417 lookThroughBitcast(Ptr);
418 assert(Ptr->getType()->isVectorTy() && "Unexpected pointer type");
419
420 IRBuilder<> Builder(I);
421 Instruction *Root = I;
422
423 Instruction *Load = tryCreateIncrementingGatScat(I, Ptr, Builder);
424 if (!Load)
425 Load = tryCreateMaskedGatherOffset(I, Ptr, Root, Builder);
426 if (!Load)
427 Load = tryCreateMaskedGatherBase(I, Ptr, Builder);
428 if (!Load)
429 return nullptr;
430
431 if (!isa<UndefValue>(PassThru) && !match(PassThru, m_Zero())) {
432 LLVM_DEBUG(dbgs() << "masked gathers: found non-trivial passthru - "
433 << "creating select\n");
434 Load = SelectInst::Create(Mask, Load, PassThru);
435 Builder.Insert(Load);
436 }
437
439 Root->eraseFromParent();
440 if (Root != I)
441 // If this was an extending gather, we need to get rid of the sext/zext
442 // sext/zext as well as of the gather itself
443 I->eraseFromParent();
444
445 LLVM_DEBUG(dbgs() << "masked gathers: successfully built masked gather\n"
446 << *Load << "\n");
447 return Load;
448}
449
450Instruction *MVEGatherScatterLowering::tryCreateMaskedGatherBase(
451 IntrinsicInst *I, Value *Ptr, IRBuilder<> &Builder, int64_t Increment) {
452 using namespace PatternMatch;
453 auto *Ty = cast<FixedVectorType>(I->getType());
454 LLVM_DEBUG(dbgs() << "masked gathers: loading from vector of pointers\n");
455 if (Ty->getNumElements() != 4 || Ty->getScalarSizeInBits() != 32)
456 // Can't build an intrinsic for this
457 return nullptr;
458 Value *Mask = I->getArgOperand(1);
459 if (match(Mask, m_One()))
460 return Builder.CreateIntrinsicWithoutFolding(
461 Intrinsic::arm_mve_vldr_gather_base, {Ty, Ptr->getType()},
462 {Ptr, Builder.getInt32(Increment)});
463 return Builder.CreateIntrinsicWithoutFolding(
464 Intrinsic::arm_mve_vldr_gather_base_predicated,
465 {Ty, Ptr->getType(), Mask->getType()},
466 {Ptr, Builder.getInt32(Increment), Mask});
467}
468
469Instruction *MVEGatherScatterLowering::tryCreateMaskedGatherBaseWB(
470 IntrinsicInst *I, Value *Ptr, IRBuilder<> &Builder, int64_t Increment) {
471 using namespace PatternMatch;
472 auto *Ty = cast<FixedVectorType>(I->getType());
473 LLVM_DEBUG(dbgs() << "masked gathers: loading from vector of pointers with "
474 << "writeback\n");
475 if (Ty->getNumElements() != 4 || Ty->getScalarSizeInBits() != 32)
476 // Can't build an intrinsic for this
477 return nullptr;
478 Value *Mask = I->getArgOperand(1);
479 if (match(Mask, m_One()))
480 return Builder.CreateIntrinsicWithoutFolding(
481 Intrinsic::arm_mve_vldr_gather_base_wb, {Ty, Ptr->getType()},
482 {Ptr, Builder.getInt32(Increment)});
483 return Builder.CreateIntrinsicWithoutFolding(
484 Intrinsic::arm_mve_vldr_gather_base_wb_predicated,
485 {Ty, Ptr->getType(), Mask->getType()},
486 {Ptr, Builder.getInt32(Increment), Mask});
487}
488
489Instruction *MVEGatherScatterLowering::tryCreateMaskedGatherOffset(
490 IntrinsicInst *I, Value *Ptr, Instruction *&Root, IRBuilder<> &Builder) {
491 using namespace PatternMatch;
492
493 Type *MemoryTy = I->getType();
494 Type *ResultTy = MemoryTy;
495
496 unsigned Unsigned = 1;
497 // The size of the gather was already checked in isLegalTypeAndAlignment;
498 // if it was not a full vector width an appropriate extend should follow.
499 auto *Extend = Root;
500 bool TruncResult = false;
501 if (MemoryTy->getPrimitiveSizeInBits() < 128) {
502 if (I->hasOneUse()) {
503 // If the gather has a single extend of the correct type, use an extending
504 // gather and replace the ext. In which case the correct root to replace
505 // is not the CallInst itself, but the instruction which extends it.
506 Instruction* User = cast<Instruction>(*I->users().begin());
507 if (isa<SExtInst>(User) &&
508 User->getType()->getPrimitiveSizeInBits() == 128) {
509 LLVM_DEBUG(dbgs() << "masked gathers: Incorporating extend: "
510 << *User << "\n");
511 Extend = User;
512 ResultTy = User->getType();
513 Unsigned = 0;
514 } else if (isa<ZExtInst>(User) &&
515 User->getType()->getPrimitiveSizeInBits() == 128) {
516 LLVM_DEBUG(dbgs() << "masked gathers: Incorporating extend: "
517 << *ResultTy << "\n");
518 Extend = User;
519 ResultTy = User->getType();
520 }
521 }
522
523 // If an extend hasn't been found and the type is an integer, create an
524 // extending gather and truncate back to the original type.
525 if (ResultTy->getPrimitiveSizeInBits() < 128 &&
526 ResultTy->isIntOrIntVectorTy()) {
527 ResultTy = ResultTy->getWithNewBitWidth(
528 128 / cast<FixedVectorType>(ResultTy)->getNumElements());
529 TruncResult = true;
530 LLVM_DEBUG(dbgs() << "masked gathers: Small input type, truncing to: "
531 << *ResultTy << "\n");
532 }
533
534 // The final size of the gather must be a full vector width
535 if (ResultTy->getPrimitiveSizeInBits() != 128) {
536 LLVM_DEBUG(dbgs() << "masked gathers: Extend needed but not provided "
537 "from the correct type. Expanding\n");
538 return nullptr;
539 }
540 }
541
542 Value *Offsets;
543 int Scale;
544 Value *BasePtr = decomposePtr(
545 Ptr, Offsets, Scale, cast<FixedVectorType>(ResultTy), MemoryTy, Builder);
546 if (!BasePtr)
547 return nullptr;
548
549 Root = Extend;
550 Value *Mask = I->getArgOperand(1);
551 Instruction *Load = nullptr;
552 if (!match(Mask, m_One()))
554 Intrinsic::arm_mve_vldr_gather_offset_predicated,
555 {ResultTy, BasePtr->getType(), Offsets->getType(), Mask->getType()},
556 {BasePtr, Offsets, Builder.getInt32(MemoryTy->getScalarSizeInBits()),
557 Builder.getInt32(Scale), Builder.getInt32(Unsigned), Mask});
558 else
560 Intrinsic::arm_mve_vldr_gather_offset,
561 {ResultTy, BasePtr->getType(), Offsets->getType()},
562 {BasePtr, Offsets, Builder.getInt32(MemoryTy->getScalarSizeInBits()),
563 Builder.getInt32(Scale), Builder.getInt32(Unsigned)});
564
565 if (TruncResult) {
566 Load = TruncInst::Create(Instruction::Trunc, Load, MemoryTy);
567 Builder.Insert(Load);
568 }
569 return Load;
570}
571
572Instruction *MVEGatherScatterLowering::lowerScatter(IntrinsicInst *I) {
573 using namespace PatternMatch;
574 LLVM_DEBUG(dbgs() << "masked scatters: checking transform preconditions\n"
575 << *I << "\n");
576
577 // @llvm.masked.scatter.*(data, ptrs, alignment, mask)
578 // Attempt to turn the masked scatter in I into a MVE intrinsic
579 // Potentially optimising the addressing modes as we do so.
580 Value *Input = I->getArgOperand(0);
581 Value *Ptr = I->getArgOperand(1);
582 Align Alignment = I->getParamAlign(1).valueOrOne();
583 auto *Ty = cast<FixedVectorType>(Input->getType());
584
585 if (!isLegalTypeAndAlignment(Ty->getNumElements(), Ty->getScalarSizeInBits(),
586 Alignment))
587 return nullptr;
588
589 lookThroughBitcast(Ptr);
590 assert(Ptr->getType()->isVectorTy() && "Unexpected pointer type");
591
592 IRBuilder<> Builder(I);
593 Instruction *Store = tryCreateIncrementingGatScat(I, Ptr, Builder);
594 if (!Store)
595 Store = tryCreateMaskedScatterOffset(I, Ptr, Builder);
596 if (!Store)
597 Store = tryCreateMaskedScatterBase(I, Ptr, Builder);
598 if (!Store)
599 return nullptr;
600
601 LLVM_DEBUG(dbgs() << "masked scatters: successfully built masked scatter\n"
602 << *Store << "\n");
603 I->eraseFromParent();
604 return Store;
605}
606
607Instruction *MVEGatherScatterLowering::tryCreateMaskedScatterBase(
608 IntrinsicInst *I, Value *Ptr, IRBuilder<> &Builder, int64_t Increment) {
609 using namespace PatternMatch;
610 Value *Input = I->getArgOperand(0);
611 auto *Ty = cast<FixedVectorType>(Input->getType());
612 // Only QR variants allow truncating
613 if (!(Ty->getNumElements() == 4 && Ty->getScalarSizeInBits() == 32)) {
614 // Can't build an intrinsic for this
615 return nullptr;
616 }
617 Value *Mask = I->getArgOperand(2);
618 // int_arm_mve_vstr_scatter_base(_predicated) addr, offset, data(, mask)
619 LLVM_DEBUG(dbgs() << "masked scatters: storing to a vector of pointers\n");
620 if (match(Mask, m_One()))
621 return Builder.CreateIntrinsicWithoutFolding(
622 Intrinsic::arm_mve_vstr_scatter_base,
623 {Ptr->getType(), Input->getType()},
624 {Ptr, Builder.getInt32(Increment), Input});
625 return Builder.CreateIntrinsicWithoutFolding(
626 Intrinsic::arm_mve_vstr_scatter_base_predicated,
627 {Ptr->getType(), Input->getType(), Mask->getType()},
628 {Ptr, Builder.getInt32(Increment), Input, Mask});
629}
630
631Instruction *MVEGatherScatterLowering::tryCreateMaskedScatterBaseWB(
632 IntrinsicInst *I, Value *Ptr, IRBuilder<> &Builder, int64_t Increment) {
633 using namespace PatternMatch;
634 Value *Input = I->getArgOperand(0);
635 auto *Ty = cast<FixedVectorType>(Input->getType());
636 LLVM_DEBUG(dbgs() << "masked scatters: storing to a vector of pointers "
637 << "with writeback\n");
638 if (Ty->getNumElements() != 4 || Ty->getScalarSizeInBits() != 32)
639 // Can't build an intrinsic for this
640 return nullptr;
641 Value *Mask = I->getArgOperand(2);
642 if (match(Mask, m_One()))
643 return Builder.CreateIntrinsicWithoutFolding(
644 Intrinsic::arm_mve_vstr_scatter_base_wb,
645 {Ptr->getType(), Input->getType()},
646 {Ptr, Builder.getInt32(Increment), Input});
647 return Builder.CreateIntrinsicWithoutFolding(
648 Intrinsic::arm_mve_vstr_scatter_base_wb_predicated,
649 {Ptr->getType(), Input->getType(), Mask->getType()},
650 {Ptr, Builder.getInt32(Increment), Input, Mask});
651}
652
653Instruction *MVEGatherScatterLowering::tryCreateMaskedScatterOffset(
654 IntrinsicInst *I, Value *Ptr, IRBuilder<> &Builder) {
655 using namespace PatternMatch;
656 Value *Input = I->getArgOperand(0);
657 Value *Mask = I->getArgOperand(2);
658 Type *InputTy = Input->getType();
659 Type *MemoryTy = InputTy;
660
661 LLVM_DEBUG(dbgs() << "masked scatters: getelementpointer found. Storing"
662 << " to base + vector of offsets\n");
663 // If the input has been truncated, try to integrate that trunc into the
664 // scatter instruction (we don't care about alignment here)
665 if (TruncInst *Trunc = dyn_cast<TruncInst>(Input)) {
666 Value *PreTrunc = Trunc->getOperand(0);
667 Type *PreTruncTy = PreTrunc->getType();
668 if (PreTruncTy->getPrimitiveSizeInBits() == 128) {
669 Input = PreTrunc;
670 InputTy = PreTruncTy;
671 }
672 }
673 bool ExtendInput = false;
674 if (InputTy->getPrimitiveSizeInBits() < 128 &&
675 InputTy->isIntOrIntVectorTy()) {
676 // If we can't find a trunc to incorporate into the instruction, create an
677 // implicit one with a zext, so that we can still create a scatter. We know
678 // that the input type is 4x/8x/16x and of type i8/i16/i32, so any type
679 // smaller than 128 bits will divide evenly into a 128bit vector.
680 InputTy = InputTy->getWithNewBitWidth(
681 128 / cast<FixedVectorType>(InputTy)->getNumElements());
682 ExtendInput = true;
683 LLVM_DEBUG(dbgs() << "masked scatters: Small input type, will extend:\n"
684 << *Input << "\n");
685 }
686 if (InputTy->getPrimitiveSizeInBits() != 128) {
687 LLVM_DEBUG(dbgs() << "masked scatters: cannot create scatters for "
688 "non-standard input types. Expanding.\n");
689 return nullptr;
690 }
691
692 Value *Offsets;
693 int Scale;
694 Value *BasePtr = decomposePtr(
695 Ptr, Offsets, Scale, cast<FixedVectorType>(InputTy), MemoryTy, Builder);
696 if (!BasePtr)
697 return nullptr;
698
699 if (ExtendInput)
700 Input = Builder.CreateZExt(Input, InputTy);
701 if (!match(Mask, m_One()))
702 return Builder.CreateIntrinsicWithoutFolding(
703 Intrinsic::arm_mve_vstr_scatter_offset_predicated,
704 {BasePtr->getType(), Offsets->getType(), Input->getType(),
705 Mask->getType()},
706 {BasePtr, Offsets, Input,
707 Builder.getInt32(MemoryTy->getScalarSizeInBits()),
708 Builder.getInt32(Scale), Mask});
709 return Builder.CreateIntrinsicWithoutFolding(
710 Intrinsic::arm_mve_vstr_scatter_offset,
711 {BasePtr->getType(), Offsets->getType(), Input->getType()},
712 {BasePtr, Offsets, Input,
713 Builder.getInt32(MemoryTy->getScalarSizeInBits()),
714 Builder.getInt32(Scale)});
715}
716
717Instruction *MVEGatherScatterLowering::tryCreateIncrementingGatScat(
718 IntrinsicInst *I, Value *Ptr, IRBuilder<> &Builder) {
719 FixedVectorType *Ty;
720 if (I->getIntrinsicID() == Intrinsic::masked_gather)
721 Ty = cast<FixedVectorType>(I->getType());
722 else
723 Ty = cast<FixedVectorType>(I->getArgOperand(0)->getType());
724
725 // Incrementing gathers only exist for v4i32
726 if (Ty->getNumElements() != 4 || Ty->getScalarSizeInBits() != 32)
727 return nullptr;
728 // Incrementing gathers are not beneficial outside of a loop
729 Loop *L = LI->getLoopFor(I->getParent());
730 if (L == nullptr)
731 return nullptr;
732
733 // Decompose the GEP into Base and Offsets
734 GetElementPtrInst *GEP = dyn_cast<GetElementPtrInst>(Ptr);
735 Value *Offsets;
736 Value *BasePtr = decomposeGEP(Offsets, Ty, GEP, Builder);
737 if (!BasePtr)
738 return nullptr;
739
740 LLVM_DEBUG(dbgs() << "masked gathers/scatters: trying to build incrementing "
741 "wb gather/scatter\n");
742
743 // The gep was in charge of making sure the offsets are scaled correctly
744 // - calculate that factor so it can be applied by hand
745 int TypeScale =
746 computeScale(DL->getTypeAllocSizeInBits(GEP->getSourceElementType()),
747 DL->getTypeSizeInBits(GEP->getType()) /
748 cast<FixedVectorType>(GEP->getType())->getNumElements());
749 if (TypeScale == -1)
750 return nullptr;
751
752 if (GEP->hasOneUse()) {
753 // Only in this case do we want to build a wb gather, because the wb will
754 // change the phi which does affect other users of the gep (which will still
755 // be using the phi in the old way)
756 if (auto *Load = tryCreateIncrementingWBGatScat(I, BasePtr, Offsets,
757 TypeScale, Builder))
758 return Load;
759 }
760
761 LLVM_DEBUG(dbgs() << "masked gathers/scatters: trying to build incrementing "
762 "non-wb gather/scatter\n");
763
764 std::pair<Value *, int64_t> Add = getVarAndConst(Offsets, TypeScale);
765 if (Add.first == nullptr)
766 return nullptr;
767 Value *OffsetsIncoming = Add.first;
768 int64_t Immediate = Add.second;
769
770 // Make sure the offsets are scaled correctly
771 Instruction *ScaledOffsets = BinaryOperator::Create(
772 Instruction::Shl, OffsetsIncoming,
773 Builder.CreateVectorSplat(Ty->getNumElements(),
774 Builder.getInt32(TypeScale)),
775 "ScaledIndex", I->getIterator());
776 // Add the base to the offsets
777 OffsetsIncoming = BinaryOperator::Create(
778 Instruction::Add, ScaledOffsets,
779 Builder.CreateVectorSplat(
780 Ty->getNumElements(),
781 Builder.CreatePtrToInt(
782 BasePtr,
783 cast<VectorType>(ScaledOffsets->getType())->getElementType())),
784 "StartIndex", I->getIterator());
785
786 if (I->getIntrinsicID() == Intrinsic::masked_gather)
787 return tryCreateMaskedGatherBase(I, OffsetsIncoming, Builder, Immediate);
788 else
789 return tryCreateMaskedScatterBase(I, OffsetsIncoming, Builder, Immediate);
790}
791
792Instruction *MVEGatherScatterLowering::tryCreateIncrementingWBGatScat(
793 IntrinsicInst *I, Value *BasePtr, Value *Offsets, unsigned TypeScale,
794 IRBuilder<> &Builder) {
795 // Check whether this gather's offset is incremented by a constant - if so,
796 // and the load is of the right type, we can merge this into a QI gather
797 Loop *L = LI->getLoopFor(I->getParent());
798 // Offsets that are worth merging into this instruction will be incremented
799 // by a constant, thus we're looking for an add of a phi and a constant
800 PHINode *Phi = dyn_cast<PHINode>(Offsets);
801 if (Phi == nullptr || Phi->getNumIncomingValues() != 2 ||
802 Phi->getParent() != L->getHeader() || !Phi->hasNUses(2))
803 // No phi means no IV to write back to; if there is a phi, we expect it
804 // to have exactly two incoming values; the only phis we are interested in
805 // will be loop IV's and have exactly two uses, one in their increment and
806 // one in the gather's gep
807 return nullptr;
808
809 unsigned IncrementIndex =
810 Phi->getIncomingBlock(0) == L->getLoopLatch() ? 0 : 1;
811 // Look through the phi to the phi increment
812 Offsets = Phi->getIncomingValue(IncrementIndex);
813
814 std::pair<Value *, int64_t> Add = getVarAndConst(Offsets, TypeScale);
815 if (Add.first == nullptr)
816 return nullptr;
817 Value *OffsetsIncoming = Add.first;
818 int64_t Immediate = Add.second;
819 if (OffsetsIncoming != Phi)
820 // Then the increment we are looking at is not an increment of the
821 // induction variable, and we don't want to do a writeback
822 return nullptr;
823
824 Builder.SetInsertPoint(&Phi->getIncomingBlock(1 - IncrementIndex)->back());
825 unsigned NumElems =
826 cast<FixedVectorType>(OffsetsIncoming->getType())->getNumElements();
827
828 // Make sure the offsets are scaled correctly
829 Instruction *ScaledOffsets = BinaryOperator::Create(
830 Instruction::Shl, Phi->getIncomingValue(1 - IncrementIndex),
831 Builder.CreateVectorSplat(NumElems, Builder.getInt32(TypeScale)),
832 "ScaledIndex",
833 Phi->getIncomingBlock(1 - IncrementIndex)->back().getIterator());
834 // Add the base to the offsets
835 OffsetsIncoming = BinaryOperator::Create(
836 Instruction::Add, ScaledOffsets,
837 Builder.CreateVectorSplat(
838 NumElems,
839 Builder.CreatePtrToInt(
840 BasePtr,
841 cast<VectorType>(ScaledOffsets->getType())->getElementType())),
842 "StartIndex",
843 Phi->getIncomingBlock(1 - IncrementIndex)->back().getIterator());
844 // The gather is pre-incrementing
845 OffsetsIncoming = BinaryOperator::Create(
846 Instruction::Sub, OffsetsIncoming,
847 Builder.CreateVectorSplat(NumElems, Builder.getInt32(Immediate)),
848 "PreIncrementStartIndex",
849 Phi->getIncomingBlock(1 - IncrementIndex)->back().getIterator());
850 Phi->setIncomingValue(1 - IncrementIndex, OffsetsIncoming);
851
852 Builder.SetInsertPoint(I);
853
854 Instruction *EndResult;
855 Instruction *NewInduction;
856 if (I->getIntrinsicID() == Intrinsic::masked_gather) {
857 // Build the incrementing gather
858 Value *Load = tryCreateMaskedGatherBaseWB(I, Phi, Builder, Immediate);
859 // One value to be handed to whoever uses the gather, one is the loop
860 // increment
861 EndResult = ExtractValueInst::Create(Load, 0, "Gather");
862 NewInduction = ExtractValueInst::Create(Load, 1, "GatherIncrement");
863 Builder.Insert(EndResult);
864 Builder.Insert(NewInduction);
865 } else {
866 // Build the incrementing scatter
867 EndResult = NewInduction =
868 tryCreateMaskedScatterBaseWB(I, Phi, Builder, Immediate);
869 }
870 Instruction *AddInst = cast<Instruction>(Offsets);
871 AddInst->replaceAllUsesWith(NewInduction);
872 AddInst->eraseFromParent();
873 Phi->setIncomingValue(IncrementIndex, NewInduction);
874
875 return EndResult;
876}
877
878void MVEGatherScatterLowering::pushOutAdd(PHINode *&Phi,
879 Value *OffsSecondOperand,
880 unsigned StartIndex) {
881 LLVM_DEBUG(dbgs() << "masked gathers/scatters: optimising add instruction\n");
882 assert(Phi->getNumIncomingValues() == 2);
883 BasicBlock *NewIndexBlock = Phi->getIncomingBlock(StartIndex);
884 BasicBlock::iterator InsertionPoint = NewIndexBlock->back().getIterator();
885 // Initialize the phi with a vector that contains a sum of the constants
887 Instruction::Add, Phi->getIncomingValue(StartIndex), OffsSecondOperand,
888 "PushedOutAdd", InsertionPoint);
889 unsigned IncrementIndex = StartIndex == 0 ? 1 : 0;
890
891 // Order such that start index comes first (this reduces mov's)
892 Value *IncrementIndexValue = Phi->getIncomingValue(IncrementIndex);
893 BasicBlock *IncrementIndexBlock = Phi->getIncomingBlock(IncrementIndex);
894 Phi->setIncomingValue(0, NewIndex);
895 Phi->setIncomingBlock(0, NewIndexBlock);
896 Phi->setIncomingValue(1, IncrementIndexValue);
897 Phi->setIncomingBlock(1, IncrementIndexBlock);
898}
899
900void MVEGatherScatterLowering::pushOutMulShl(unsigned Opcode, PHINode *&Phi,
901 Value *IncrementPerRound,
902 Value *OffsSecondOperand,
903 unsigned LoopIncrement,
904 IRBuilder<> &Builder) {
905 LLVM_DEBUG(dbgs() << "masked gathers/scatters: optimising mul instruction\n");
906 assert(Phi->getNumIncomingValues() == 2);
907
908 // Create a new scalar add outside of the loop and transform it to a splat
909 // by which loop variable can be incremented
910 BasicBlock *StartIndexBlock =
911 Phi->getIncomingBlock(LoopIncrement == 1 ? 0 : 1);
912 BasicBlock::iterator InsertionPoint = StartIndexBlock->back().getIterator();
913
914 // Create a new index
915 Value *StartIndex =
917 Phi->getIncomingValue(LoopIncrement == 1 ? 0 : 1),
918 OffsSecondOperand, "PushedOutMul", InsertionPoint);
919
920 Instruction *Product =
921 BinaryOperator::Create((Instruction::BinaryOps)Opcode, IncrementPerRound,
922 OffsSecondOperand, "Product", InsertionPoint);
923 BasicBlock *NewIncrementBlock = Phi->getIncomingBlock(LoopIncrement);
924 BasicBlock::iterator NewIncrInsertPt =
925 NewIncrementBlock->back().getIterator();
926 NewIncrInsertPt = std::prev(NewIncrInsertPt);
927
928 // Increment NewIndex by Product instead of the multiplication
929 Instruction *NewIncrement = BinaryOperator::Create(
930 Instruction::Add, Phi, Product, "IncrementPushedOutMul", NewIncrInsertPt);
931
932 Phi->setIncomingValue(0, StartIndex);
933 Phi->setIncomingBlock(0, StartIndexBlock);
934 Phi->setIncomingValue(1, NewIncrement);
935 Phi->setIncomingBlock(1, NewIncrementBlock);
936}
937
938// Check whether all usages of this instruction are as offsets of
939// gathers/scatters or simple arithmetics only used by gathers/scatters
941 if (I->use_empty()) {
942 return false;
943 }
944 bool Gatscat = true;
945 for (User *U : I->users()) {
946 if (!isa<Instruction>(U))
947 return false;
948 if (isa<GetElementPtrInst>(U) ||
950 return Gatscat;
951 } else {
952 unsigned OpCode = cast<Instruction>(U)->getOpcode();
953 if ((OpCode == Instruction::Add || OpCode == Instruction::Mul ||
954 OpCode == Instruction::Shl ||
957 continue;
958 }
959 return false;
960 }
961 }
962 return Gatscat;
963}
964
965bool MVEGatherScatterLowering::optimiseOffsets(Value *Offsets, BasicBlock *BB,
966 LoopInfo *LI) {
967 LLVM_DEBUG(dbgs() << "masked gathers/scatters: trying to optimize: "
968 << *Offsets << "\n");
969 // Optimise the addresses of gathers/scatters by moving invariant
970 // calculations out of the loop
971 if (!isa<Instruction>(Offsets))
972 return false;
973 Instruction *Offs = cast<Instruction>(Offsets);
974 if (Offs->getOpcode() != Instruction::Add && !isAddLikeOr(Offs, *DL) &&
975 Offs->getOpcode() != Instruction::Mul &&
976 Offs->getOpcode() != Instruction::Shl)
977 return false;
978 Loop *L = LI->getLoopFor(BB);
979 if (L == nullptr)
980 return false;
981 if (!Offs->hasOneUse()) {
982 if (!hasAllGatScatUsers(Offs, *DL))
983 return false;
984 }
985
986 // Find out which, if any, operand of the instruction
987 // is a phi node
988 PHINode *Phi;
989 int OffsSecondOp;
990 if (isa<PHINode>(Offs->getOperand(0))) {
991 Phi = cast<PHINode>(Offs->getOperand(0));
992 OffsSecondOp = 1;
993 } else if (isa<PHINode>(Offs->getOperand(1))) {
994 Phi = cast<PHINode>(Offs->getOperand(1));
995 OffsSecondOp = 0;
996 } else {
997 bool Changed = false;
998 if (isa<Instruction>(Offs->getOperand(0)) &&
999 L->contains(cast<Instruction>(Offs->getOperand(0))))
1000 Changed |= optimiseOffsets(Offs->getOperand(0), BB, LI);
1001 if (isa<Instruction>(Offs->getOperand(1)) &&
1002 L->contains(cast<Instruction>(Offs->getOperand(1))))
1003 Changed |= optimiseOffsets(Offs->getOperand(1), BB, LI);
1004 if (!Changed)
1005 return false;
1006 if (isa<PHINode>(Offs->getOperand(0))) {
1007 Phi = cast<PHINode>(Offs->getOperand(0));
1008 OffsSecondOp = 1;
1009 } else if (isa<PHINode>(Offs->getOperand(1))) {
1010 Phi = cast<PHINode>(Offs->getOperand(1));
1011 OffsSecondOp = 0;
1012 } else {
1013 return false;
1014 }
1015 }
1016 // A phi node we want to perform this function on should be from the
1017 // loop header.
1018 if (Phi->getParent() != L->getHeader())
1019 return false;
1020
1021 // We're looking for a simple add recurrence.
1022 BinaryOperator *IncInstruction;
1023 Value *Start, *IncrementPerRound;
1024 if (!matchSimpleRecurrence(Phi, IncInstruction, Start, IncrementPerRound) ||
1025 IncInstruction->getOpcode() != Instruction::Add)
1026 return false;
1027
1028 int IncrementingBlock = Phi->getIncomingValue(0) == IncInstruction ? 0 : 1;
1029
1030 // Get the value that is added to/multiplied with the phi
1031 Value *OffsSecondOperand = Offs->getOperand(OffsSecondOp);
1032
1033 if (IncrementPerRound->getType() != OffsSecondOperand->getType() ||
1034 !L->isLoopInvariant(OffsSecondOperand))
1035 // Something has gone wrong, abort
1036 return false;
1037
1038 // Only proceed if the increment per round is a constant or an instruction
1039 // which does not originate from within the loop
1040 if (!isa<Constant>(IncrementPerRound) &&
1041 !(isa<Instruction>(IncrementPerRound) &&
1042 !L->contains(cast<Instruction>(IncrementPerRound))))
1043 return false;
1044
1045 // If the phi is not used by anything else, we can just adapt it when
1046 // replacing the instruction; if it is, we'll have to duplicate it
1047 PHINode *NewPhi;
1048 if (Phi->hasNUses(2)) {
1049 // No other users -> reuse existing phi (One user is the instruction
1050 // we're looking at, the other is the phi increment)
1051 if (!IncInstruction->hasOneUse()) {
1052 // If the incrementing instruction does have more users than
1053 // our phi, we need to copy it
1054 IncInstruction = BinaryOperator::Create(
1055 Instruction::BinaryOps(IncInstruction->getOpcode()), Phi,
1056 IncrementPerRound, "LoopIncrement", IncInstruction->getIterator());
1057 Phi->setIncomingValue(IncrementingBlock, IncInstruction);
1058 }
1059 NewPhi = Phi;
1060 } else {
1061 // There are other users -> create a new phi
1062 NewPhi = PHINode::Create(Phi->getType(), 2, "NewPhi", Phi->getIterator());
1063 // Copy the incoming values of the old phi
1064 NewPhi->addIncoming(Phi->getIncomingValue(IncrementingBlock == 1 ? 0 : 1),
1065 Phi->getIncomingBlock(IncrementingBlock == 1 ? 0 : 1));
1066 IncInstruction = BinaryOperator::Create(
1067 Instruction::BinaryOps(IncInstruction->getOpcode()), NewPhi,
1068 IncrementPerRound, "LoopIncrement", IncInstruction->getIterator());
1069 NewPhi->addIncoming(IncInstruction,
1070 Phi->getIncomingBlock(IncrementingBlock));
1071 IncrementingBlock = 1;
1072 }
1073
1074 IRBuilder<> Builder(Phi);
1075 Builder.SetCurrentDebugLocation(Offs->getDebugLoc());
1076
1077 switch (Offs->getOpcode()) {
1078 case Instruction::Add:
1079 case Instruction::Or:
1080 pushOutAdd(NewPhi, OffsSecondOperand, IncrementingBlock == 1 ? 0 : 1);
1081 break;
1082 case Instruction::Mul:
1083 case Instruction::Shl:
1084 pushOutMulShl(Offs->getOpcode(), NewPhi, IncrementPerRound,
1085 OffsSecondOperand, IncrementingBlock, Builder);
1086 break;
1087 default:
1088 return false;
1089 }
1090 LLVM_DEBUG(dbgs() << "masked gathers/scatters: simplified loop variable "
1091 << "add/mul\n");
1092
1093 // The instruction has now been "absorbed" into the phi value
1094 Offs->replaceAllUsesWith(NewPhi);
1095 Offs->eraseFromParent();
1096 // Clean up the old increment in case it's unused because we built a new
1097 // one
1098 if (IncInstruction->use_empty())
1099 IncInstruction->eraseFromParent();
1100
1101 return true;
1102}
1103
1104static Value *CheckAndCreateOffsetAdd(Value *X, unsigned ScaleX, Value *Y,
1105 unsigned ScaleY, IRBuilder<> &Builder) {
1106 // Splat the non-vector value to a vector of the given type - if the value is
1107 // a constant (and its value isn't too big), we can even use this opportunity
1108 // to scale it to the size of the vector elements
1109 auto FixSummands = [&Builder](FixedVectorType *&VT, Value *&NonVectorVal) {
1110 ConstantInt *Const;
1111 if ((Const = dyn_cast<ConstantInt>(NonVectorVal)) &&
1112 VT->getElementType() != NonVectorVal->getType()) {
1113 unsigned TargetElemSize = VT->getElementType()->getPrimitiveSizeInBits();
1114 uint64_t N = Const->getZExtValue();
1115 if (N < (unsigned)(1 << (TargetElemSize - 1))) {
1116 NonVectorVal = Builder.CreateVectorSplat(
1117 VT->getNumElements(), Builder.getIntN(TargetElemSize, N));
1118 return;
1119 }
1120 }
1121 NonVectorVal =
1122 Builder.CreateVectorSplat(VT->getNumElements(), NonVectorVal);
1123 };
1124
1125 FixedVectorType *XElType = dyn_cast<FixedVectorType>(X->getType());
1126 FixedVectorType *YElType = dyn_cast<FixedVectorType>(Y->getType());
1127 // If one of X, Y is not a vector, we have to splat it in order
1128 // to add the two of them.
1129 if (XElType && !YElType) {
1130 FixSummands(XElType, Y);
1131 YElType = cast<FixedVectorType>(Y->getType());
1132 } else if (YElType && !XElType) {
1133 FixSummands(YElType, X);
1134 XElType = cast<FixedVectorType>(X->getType());
1135 }
1136 assert(XElType && YElType && "Unknown vector types");
1137 // Check that the summands are of compatible types
1138 if (XElType != YElType) {
1139 LLVM_DEBUG(dbgs() << "masked gathers/scatters: incompatible gep offsets\n");
1140 return nullptr;
1141 }
1142
1143 if (XElType->getElementType()->getScalarSizeInBits() != 32) {
1144 // Check that by adding the vectors we do not accidentally
1145 // create an overflow
1146 Constant *ConstX = dyn_cast<Constant>(X);
1147 Constant *ConstY = dyn_cast<Constant>(Y);
1148 if (!ConstX || !ConstY)
1149 return nullptr;
1150 unsigned TargetElemSize = 128 / XElType->getNumElements();
1151 for (unsigned i = 0; i < XElType->getNumElements(); i++) {
1152 ConstantInt *ConstXEl =
1154 ConstantInt *ConstYEl =
1156 if (!ConstXEl || !ConstYEl ||
1157 ConstXEl->getZExtValue() * ScaleX +
1158 ConstYEl->getZExtValue() * ScaleY >=
1159 (unsigned)(1 << (TargetElemSize - 1)))
1160 return nullptr;
1161 }
1162 }
1163
1164 Value *XScale = Builder.CreateVectorSplat(
1165 XElType->getNumElements(),
1166 Builder.getIntN(XElType->getScalarSizeInBits(), ScaleX));
1167 Value *YScale = Builder.CreateVectorSplat(
1168 YElType->getNumElements(),
1169 Builder.getIntN(YElType->getScalarSizeInBits(), ScaleY));
1170 Value *Add = Builder.CreateAdd(Builder.CreateMul(X, XScale),
1171 Builder.CreateMul(Y, YScale));
1172
1173 if (checkOffsetSize(Add, XElType->getNumElements()))
1174 return Add;
1175 else
1176 return nullptr;
1177}
1178
1179Value *MVEGatherScatterLowering::foldGEP(GetElementPtrInst *GEP,
1180 Value *&Offsets, unsigned &Scale,
1181 IRBuilder<> &Builder) {
1182 Value *GEPPtr = GEP->getPointerOperand();
1183 Offsets = GEP->getOperand(1);
1184 Scale = DL->getTypeAllocSize(GEP->getSourceElementType());
1185 // We only merge geps with constant offsets, because only for those
1186 // we can make sure that we do not cause an overflow
1187 if (GEP->getNumIndices() != 1 || !isa<Constant>(Offsets))
1188 return nullptr;
1189 if (GetElementPtrInst *BaseGEP = dyn_cast<GetElementPtrInst>(GEPPtr)) {
1190 // Merge the two geps into one
1191 Value *BaseBasePtr = foldGEP(BaseGEP, Offsets, Scale, Builder);
1192 if (!BaseBasePtr)
1193 return nullptr;
1195 Offsets, Scale, GEP->getOperand(1),
1196 DL->getTypeAllocSize(GEP->getSourceElementType()), Builder);
1197 if (Offsets == nullptr)
1198 return nullptr;
1199 Scale = 1; // Scale is always an i8 at this point.
1200 return BaseBasePtr;
1201 }
1202 return GEPPtr;
1203}
1204
1205bool MVEGatherScatterLowering::optimiseAddress(Value *Address, BasicBlock *BB,
1206 LoopInfo *LI) {
1207 GetElementPtrInst *GEP = dyn_cast<GetElementPtrInst>(Address);
1208 if (!GEP)
1209 return false;
1210 bool Changed = false;
1211 if (GEP->hasOneUse() && isa<GetElementPtrInst>(GEP->getPointerOperand())) {
1212 IRBuilder<> Builder(GEP);
1213 Value *Offsets;
1214 unsigned Scale;
1215 Value *Base = foldGEP(GEP, Offsets, Scale, Builder);
1216 // We only want to merge the geps if there is a real chance that they can be
1217 // used by an MVE gather; thus the offset has to have the correct size
1218 // (always i32 if it is not of vector type) and the base has to be a
1219 // pointer.
1220 if (Offsets && Base && Base != GEP) {
1221 assert(Scale == 1 && "Expected to fold GEP to a scale of 1");
1222 Type *BaseTy = Builder.getPtrTy();
1223 if (auto *VecTy = dyn_cast<FixedVectorType>(Base->getType()))
1224 BaseTy = FixedVectorType::get(BaseTy, VecTy);
1225 GetElementPtrInst *NewAddress = GetElementPtrInst::Create(
1226 Builder.getInt8Ty(), Builder.CreateBitCast(Base, BaseTy), Offsets,
1227 "gep.merged", GEP->getIterator());
1228 LLVM_DEBUG(dbgs() << "Folded GEP: " << *GEP
1229 << "\n new : " << *NewAddress << "\n");
1230 GEP->replaceAllUsesWith(
1231 Builder.CreateBitCast(NewAddress, GEP->getType()));
1232 GEP = NewAddress;
1233 Changed = true;
1234 }
1235 }
1236 Changed |= optimiseOffsets(GEP->getOperand(1), GEP->getParent(), LI);
1237 return Changed;
1238}
1239
1240bool MVEGatherScatterLowering::runOnFunction(Function &F) {
1242 return false;
1243 auto &TPC = getAnalysis<TargetPassConfig>();
1244 auto &TM = TPC.getTM<TargetMachine>();
1245 auto *ST = &TM.getSubtarget<ARMSubtarget>(F);
1246 if (!ST->hasMVEIntegerOps())
1247 return false;
1248 LI = &getAnalysis<LoopInfoWrapperPass>().getLoopInfo();
1249 DL = &F.getDataLayout();
1252
1253 bool Changed = false;
1254
1255 for (BasicBlock &BB : F) {
1257
1258 for (Instruction &I : BB) {
1259 IntrinsicInst *II = dyn_cast<IntrinsicInst>(&I);
1260 if (II && II->getIntrinsicID() == Intrinsic::masked_gather &&
1261 isa<FixedVectorType>(II->getType())) {
1262 Gathers.push_back(II);
1263 Changed |= optimiseAddress(II->getArgOperand(0), II->getParent(), LI);
1264 } else if (II && II->getIntrinsicID() == Intrinsic::masked_scatter &&
1265 isa<FixedVectorType>(II->getArgOperand(0)->getType())) {
1266 Scatters.push_back(II);
1267 Changed |= optimiseAddress(II->getArgOperand(1), II->getParent(), LI);
1268 }
1269 }
1270 }
1271 for (IntrinsicInst *I : Gathers) {
1272 Instruction *L = lowerGather(I);
1273 if (L == nullptr)
1274 continue;
1275
1276 // Get rid of any now dead instructions
1277 SimplifyInstructionsInBlock(L->getParent());
1278 Changed = true;
1279 }
1280
1281 for (IntrinsicInst *I : Scatters) {
1282 Instruction *S = lowerScatter(I);
1283 if (S == nullptr)
1284 continue;
1285
1286 // Get rid of any now dead instructions
1288 Changed = true;
1289 }
1290 return Changed;
1291}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
cl::opt< bool > EnableMaskedGatherScatters
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static Decomposition decomposeGEP(GEPOperator &GEP, ConstraintInfo &Info, bool IsSigned, const DataLayout &DL)
static bool runOnFunction(Function &F, bool PostInlining)
#define DEBUG_TYPE
Hexagon Common GEP
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static bool isAddLikeOr(Instruction *I, const DataLayout &DL)
static bool hasAllGatScatUsers(Instruction *I, const DataLayout &DL)
static bool checkOffsetSize(Value *Offsets, unsigned TargetElemCount)
static Value * CheckAndCreateOffsetAdd(Value *X, unsigned ScaleX, Value *Y, unsigned ScaleY, IRBuilder<> &Builder)
cl::opt< bool > EnableMaskedGatherScatters("enable-arm-maskedgatscat", cl::Hidden, cl::init(true), cl::desc("Enable the generation of masked gathers and scatters"))
uint64_t IntrinsicInst * II
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
#define LLVM_DEBUG(...)
Definition Debug.h:119
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file describes how to lower LLVM code to machine code.
Target-Independent Code Generator Pass Configuration Options pass.
This pass exposes codegen information to IR-level passes.
AnalysisUsage & addRequired()
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Definition Pass.cpp:278
const Instruction & back() const
Definition BasicBlock.h:471
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
BinaryOps getOpcode() const
Definition InstrTypes.h:409
static LLVM_ABI BinaryOperator * Create(BinaryOps Op, Value *S1, Value *S2, const Twine &Name=Twine(), InsertPosition InsertBefore=nullptr)
Construct a binary instruction, given the opcode and the two operands.
Type * getDestTy() const
Return the destination type, as a convenience.
Definition InstrTypes.h:681
This is the shared class of boolean and integer constants.
Definition Constants.h:87
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
This is an important base class in LLVM.
Definition Constant.h:43
LLVM_ABI Constant * getAggregateElement(unsigned Elt) const
For aggregates (struct/array/vector) return the constant that corresponds to the specified element if...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
static ExtractValueInst * Create(Value *Agg, ArrayRef< unsigned > Idxs, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
static GetElementPtrInst * Create(Type *PointeeType, Value *Ptr, ArrayRef< Value * > IdxList, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
LLVM_ABI CallInst * CreateIntrinsicWithoutFolding(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={})
Create a call to intrinsic ID with Args, mangled using OverloadTypes.
LLVM_ABI Value * CreateVectorSplat(unsigned NumElts, Value *V, const Twine &Name="")
Return a vector value that contains.
Value * CreateIntToPtr(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2230
void SetCurrentDebugLocation(const DebugLoc &L)
Set location information used by debugging information.
Definition IRBuilder.h:220
IntegerType * getInt32Ty()
Fetch the type representing a 32-bit integer.
Definition IRBuilder.h:513
ConstantInt * getInt32(uint32_t C)
Get a constant 32-bit value.
Definition IRBuilder.h:456
InstTy * Insert(InstTy *I, const Twine &Name="") const
Insert and return the specified instruction.
Definition IRBuilder.h:146
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2235
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2113
Value * CreatePtrToInt(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2225
Value * CreateTrunc(Value *V, Type *DestTy, const Twine &Name="", bool IsNUW=false, bool IsNSW=false)
Definition IRBuilder.h:2099
PointerType * getPtrTy(unsigned AddrSpace=0)
Fetch the type representing a pointer.
Definition IRBuilder.h:556
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Definition IRBuilder.h:181
IntegerType * getInt8Ty()
Fetch the type representing an 8-bit integer.
Definition IRBuilder.h:503
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2901
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static PHINode * Create(Type *Ty, unsigned NumReservedValues, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Constructors - NumReservedValues is a hint for the number of incoming edges that this phi node will h...
static LLVM_ABI PassRegistry * getPassRegistry()
getPassRegistry - Access the global registry object, which is automatically initialized at applicatio...
Pass interface - Implemented by all 'passes'.
Definition Pass.h:99
static SelectInst * Create(Value *C, Value *S1, Value *S2, const Twine &NameStr="", InsertPosition InsertBefore=nullptr, const Instruction *MDFrom=nullptr)
void push_back(const T &Elt)
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:258
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
bool use_empty() const
Definition Value.h:348
Type * getElementType() const
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
Changed
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
bool match(Val *V, const Pattern &P)
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
Offsets
Offsets in bytes from the start of the input buffer.
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
NodeAddr< PhiNode * > Phi
Definition RDFGraph.h:390
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
unsigned getNumElements(Type *Ty)
Definition SLPUtils.cpp:87
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI bool haveNoCommonBitsSet(const WithCache< const Value * > &LHSCache, const WithCache< const Value * > &RHSCache, const SimplifyQuery &SQ)
Return true if LHS and RHS have no common bits set.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
Pass * createMVEGatherScatterLoweringPass()
LLVM_ABI bool SimplifyInstructionsInBlock(BasicBlock *BB, const TargetLibraryInfo *TLI=nullptr)
Scan the specified basic block and try to simplify any instructions in it and recursively delete dead...
Definition Local.cpp:715
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
LLVM_ABI bool matchSimpleRecurrence(const PHINode *P, BinaryOperator *&BO, Value *&Start, Value *&Step)
Attempt to match a simple first order recurrence cycle of the form: iv = phi Ty [Start,...
bool isGatherScatter(IntrinsicInst *IntInst)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
IRBuilder(LLVMContext &, FolderTy, InserterTy) -> IRBuilder< FolderTy, InserterTy >
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ Add
Sum of integers.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
void initializeMVEGatherScatterLoweringPass(PassRegistry &)
@ Increment
Incrementally increasing token ID.
Definition AllocToken.h:26
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39