70#define DEBUG_TYPE "openmp-ir-builder"
77 cl::desc(
"Use optimistic attributes describing "
78 "'as-if' properties of runtime calls."),
82 "openmp-ir-builder-unroll-threshold-factor",
cl::Hidden,
83 cl::desc(
"Factor for the unroll threshold to account for code "
84 "simplifications still taking place"),
88 "openmp-ir-builder-use-default-max-threads",
cl::Hidden,
99 if (!IP1.isSet() || !IP2.isSet())
101 return IP1.getBlock() == IP2.getBlock() && IP1.getPoint() == IP2.getPoint();
106 switch (SchedType & ~OMPScheduleType::MonotonicityMask) {
107 case OMPScheduleType::UnorderedStaticChunked:
108 case OMPScheduleType::UnorderedStatic:
109 case OMPScheduleType::UnorderedDynamicChunked:
110 case OMPScheduleType::UnorderedGuidedChunked:
111 case OMPScheduleType::UnorderedRuntime:
112 case OMPScheduleType::UnorderedAuto:
113 case OMPScheduleType::UnorderedTrapezoidal:
114 case OMPScheduleType::UnorderedGreedy:
115 case OMPScheduleType::UnorderedBalanced:
116 case OMPScheduleType::UnorderedGuidedIterativeChunked:
117 case OMPScheduleType::UnorderedGuidedAnalyticalChunked:
118 case OMPScheduleType::UnorderedSteal:
119 case OMPScheduleType::UnorderedStaticBalancedChunked:
120 case OMPScheduleType::UnorderedGuidedSimd:
121 case OMPScheduleType::UnorderedRuntimeSimd:
122 case OMPScheduleType::OrderedStaticChunked:
123 case OMPScheduleType::OrderedStatic:
124 case OMPScheduleType::OrderedDynamicChunked:
125 case OMPScheduleType::OrderedGuidedChunked:
126 case OMPScheduleType::OrderedRuntime:
127 case OMPScheduleType::OrderedAuto:
128 case OMPScheduleType::OrderdTrapezoidal:
129 case OMPScheduleType::NomergeUnorderedStaticChunked:
130 case OMPScheduleType::NomergeUnorderedStatic:
131 case OMPScheduleType::NomergeUnorderedDynamicChunked:
132 case OMPScheduleType::NomergeUnorderedGuidedChunked:
133 case OMPScheduleType::NomergeUnorderedRuntime:
134 case OMPScheduleType::NomergeUnorderedAuto:
135 case OMPScheduleType::NomergeUnorderedTrapezoidal:
136 case OMPScheduleType::NomergeUnorderedGreedy:
137 case OMPScheduleType::NomergeUnorderedBalanced:
138 case OMPScheduleType::NomergeUnorderedGuidedIterativeChunked:
139 case OMPScheduleType::NomergeUnorderedGuidedAnalyticalChunked:
140 case OMPScheduleType::NomergeUnorderedSteal:
141 case OMPScheduleType::NomergeOrderedStaticChunked:
142 case OMPScheduleType::NomergeOrderedStatic:
143 case OMPScheduleType::NomergeOrderedDynamicChunked:
144 case OMPScheduleType::NomergeOrderedGuidedChunked:
145 case OMPScheduleType::NomergeOrderedRuntime:
146 case OMPScheduleType::NomergeOrderedAuto:
147 case OMPScheduleType::NomergeOrderedTrapezoidal:
148 case OMPScheduleType::OrderedDistributeChunked:
149 case OMPScheduleType::OrderedDistribute:
157 SchedType & OMPScheduleType::MonotonicityMask;
158 if (MonotonicityFlags == OMPScheduleType::MonotonicityMask)
172 Builder.restoreIP(IP);
176 if (Builder.GetInsertPoint() != BB->
end())
186 unsigned Line = FSP->getScopeLine() ? FSP->getScopeLine() : FSP->getLine();
187 Builder.SetCurrentDebugLocation(
193 return T.isAMDGPU() ||
T.isNVPTX() ||
T.isSPIRV();
199 Kernel->getFnAttribute(
"target-features").getValueAsString();
200 if (Features.
count(
"+wavefrontsize64"))
215 bool HasSimdModifier,
bool HasDistScheduleChunks) {
217 switch (ClauseKind) {
218 case OMP_SCHEDULE_Default:
219 case OMP_SCHEDULE_Static:
220 return HasChunks ? OMPScheduleType::BaseStaticChunked
221 : OMPScheduleType::BaseStatic;
222 case OMP_SCHEDULE_Dynamic:
223 return OMPScheduleType::BaseDynamicChunked;
224 case OMP_SCHEDULE_Guided:
225 return HasSimdModifier ? OMPScheduleType::BaseGuidedSimd
226 : OMPScheduleType::BaseGuidedChunked;
227 case OMP_SCHEDULE_Auto:
229 case OMP_SCHEDULE_Runtime:
230 return HasSimdModifier ? OMPScheduleType::BaseRuntimeSimd
231 : OMPScheduleType::BaseRuntime;
232 case OMP_SCHEDULE_Distribute:
233 return HasDistScheduleChunks ? OMPScheduleType::BaseDistributeChunked
234 : OMPScheduleType::BaseDistribute;
242 bool HasOrderedClause) {
243 assert((BaseScheduleType & OMPScheduleType::ModifierMask) ==
244 OMPScheduleType::None &&
245 "Must not have ordering nor monotonicity flags already set");
248 ? OMPScheduleType::ModifierOrdered
249 : OMPScheduleType::ModifierUnordered;
250 OMPScheduleType OrderingScheduleType = BaseScheduleType | OrderingModifier;
253 if (OrderingScheduleType ==
254 (OMPScheduleType::BaseGuidedSimd | OMPScheduleType::ModifierOrdered))
255 return OMPScheduleType::OrderedGuidedChunked;
256 else if (OrderingScheduleType == (OMPScheduleType::BaseRuntimeSimd |
257 OMPScheduleType::ModifierOrdered))
258 return OMPScheduleType::OrderedRuntime;
260 return OrderingScheduleType;
266 bool HasSimdModifier,
bool HasMonotonic,
267 bool HasNonmonotonic,
bool HasOrderedClause) {
268 assert((ScheduleType & OMPScheduleType::MonotonicityMask) ==
269 OMPScheduleType::None &&
270 "Must not have monotonicity flags already set");
271 assert((!HasMonotonic || !HasNonmonotonic) &&
272 "Monotonic and Nonmonotonic are contradicting each other");
275 return ScheduleType | OMPScheduleType::ModifierMonotonic;
276 }
else if (HasNonmonotonic) {
277 return ScheduleType | OMPScheduleType::ModifierNonmonotonic;
287 if ((BaseScheduleType == OMPScheduleType::BaseStatic) ||
288 (BaseScheduleType == OMPScheduleType::BaseStaticChunked) ||
294 return ScheduleType | OMPScheduleType::ModifierNonmonotonic;
302 bool HasSimdModifier,
bool HasMonotonicModifier,
303 bool HasNonmonotonicModifier,
bool HasOrderedClause,
304 bool HasDistScheduleChunks) {
306 ClauseKind, HasChunks, HasSimdModifier, HasDistScheduleChunks);
310 OrderedSchedule, HasSimdModifier, HasMonotonicModifier,
311 HasNonmonotonicModifier, HasOrderedClause);
319static std::optional<omp::OMPTgtExecModeFlags>
324 if (
Call->getCalledFunction()->getName() ==
"__kmpc_target_init") {
325 TargetInitCall =
Call;
350 std::optional<omp::OMPTgtExecModeFlags> ExecMode =
362 if (
Instruction *Term = Source->getTerminatorOrNull()) {
371 NewBr->setDebugLoc(
DL);
376 assert(New->getFirstInsertionPt() == New->begin() &&
377 "Target BB must not have PHI nodes");
393 New->splice(New->begin(), Old, IP.
getPoint(), Old->
end());
397 NewBr->setDebugLoc(
DL);
409 Builder.SetInsertPoint(Old);
413 Builder.SetCurrentDebugLocation(
DebugLoc);
423 New->replaceSuccessorsPhiUsesWith(Old, New);
432 Builder.SetInsertPoint(Builder.GetInsertBlock()->getTerminator());
434 Builder.SetInsertPoint(Builder.GetInsertBlock());
437 Builder.SetCurrentDebugLocation(
DebugLoc);
446 Builder.SetInsertPoint(Builder.GetInsertBlock()->getTerminator());
448 Builder.SetInsertPoint(Builder.GetInsertBlock());
451 Builder.SetCurrentDebugLocation(
DebugLoc);
468 const Twine &Name =
"",
bool AsPtr =
true,
469 bool Is64Bit =
false) {
470 Builder.restoreIP(OuterAllocaIP);
474 Builder.CreateAlloca(IntTy,
nullptr, Name +
".addr");
478 FakeVal = FakeValAddr;
480 FakeVal = Builder.CreateLoad(IntTy, FakeValAddr, Name +
".val");
485 Builder.restoreIP(InnerAllocaIP);
488 UseFakeVal = Builder.CreateLoad(IntTy, FakeVal, Name +
".use");
491 FakeVal, Is64Bit ? Builder.getInt64(10) : Builder.getInt32(10)));
504enum OpenMPOffloadingRequiresDirFlags {
506 OMP_REQ_UNDEFINED = 0x000,
508 OMP_REQ_NONE = 0x001,
510 OMP_REQ_REVERSE_OFFLOAD = 0x002,
512 OMP_REQ_UNIFIED_ADDRESS = 0x004,
514 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008,
516 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010,
523 DominatorTree *DT =
nullptr,
bool AggregateArgs =
false,
524 BlockFrequencyInfo *BFI =
nullptr,
525 BranchProbabilityInfo *BPI =
nullptr,
526 AssumptionCache *AC =
nullptr,
bool AllowVarArgs =
false,
527 bool AllowAlloca =
false,
528 BasicBlock *AllocationBlock =
nullptr,
530 std::string Suffix =
"",
bool ArgsInZeroAddressSpace =
false)
531 : CodeExtractor(BBs, DT, AggregateArgs, BFI, BPI, AC, AllowVarArgs,
532 AllowAlloca, AllocationBlock, DeallocationBlocks, Suffix,
533 ArgsInZeroAddressSpace),
534 OMPBuilder(OMPBuilder) {}
536 virtual ~OMPCodeExtractor() =
default;
539 OpenMPIRBuilder &OMPBuilder;
542class DeviceSharedMemCodeExtractor :
public OMPCodeExtractor {
544 using OMPCodeExtractor::OMPCodeExtractor;
545 virtual ~DeviceSharedMemCodeExtractor() =
default;
549 allocateVar(IRBuilder<>::InsertPoint AllocaIP,
Type *VarType,
550 const Twine &Name = Twine(
""),
551 AddrSpaceCastInst **CastedAlloc =
nullptr)
override {
552 return OMPBuilder.createOMPAllocShared(AllocaIP, VarType, Name);
555 virtual Instruction *deallocateVar(IRBuilder<>::InsertPoint DeallocIP,
557 return OMPBuilder.createOMPFreeShared(DeallocIP, Var, VarType);
564 OpenMPIRBuilder &OMPBuilder;
566 DeviceSharedMemOutlineInfo(OpenMPIRBuilder &OMPBuilder)
567 : OMPBuilder(OMPBuilder) {}
568 virtual ~DeviceSharedMemOutlineInfo() =
default;
570 virtual std::unique_ptr<CodeExtractor>
572 bool ArgsInZeroAddressSpace,
573 Twine Suffix = Twine(
""))
override;
579 : RequiresFlags(OMP_REQ_UNDEFINED) {}
583 bool HasRequiresReverseOffload,
bool HasRequiresUnifiedAddress,
584 bool HasRequiresUnifiedSharedMemory,
bool HasRequiresDynamicAllocators)
587 RequiresFlags(OMP_REQ_UNDEFINED) {
588 if (HasRequiresReverseOffload)
589 RequiresFlags |= OMP_REQ_REVERSE_OFFLOAD;
590 if (HasRequiresUnifiedAddress)
591 RequiresFlags |= OMP_REQ_UNIFIED_ADDRESS;
592 if (HasRequiresUnifiedSharedMemory)
593 RequiresFlags |= OMP_REQ_UNIFIED_SHARED_MEMORY;
594 if (HasRequiresDynamicAllocators)
595 RequiresFlags |= OMP_REQ_DYNAMIC_ALLOCATORS;
599 return RequiresFlags & OMP_REQ_REVERSE_OFFLOAD;
603 return RequiresFlags & OMP_REQ_UNIFIED_ADDRESS;
607 return RequiresFlags & OMP_REQ_UNIFIED_SHARED_MEMORY;
611 return RequiresFlags & OMP_REQ_DYNAMIC_ALLOCATORS;
616 :
static_cast<int64_t
>(OMP_REQ_NONE);
621 RequiresFlags |= OMP_REQ_REVERSE_OFFLOAD;
623 RequiresFlags &= ~OMP_REQ_REVERSE_OFFLOAD;
628 RequiresFlags |= OMP_REQ_UNIFIED_ADDRESS;
630 RequiresFlags &= ~OMP_REQ_UNIFIED_ADDRESS;
635 RequiresFlags |= OMP_REQ_UNIFIED_SHARED_MEMORY;
637 RequiresFlags &= ~OMP_REQ_UNIFIED_SHARED_MEMORY;
642 RequiresFlags |= OMP_REQ_DYNAMIC_ALLOCATORS;
644 RequiresFlags &= ~OMP_REQ_DYNAMIC_ALLOCATORS;
657 constexpr size_t MaxDim = 3;
662 Value *DynCGroupMemFallbackFlag =
664 DynCGroupMemFallbackFlag =
Builder.CreateShl(DynCGroupMemFallbackFlag, 2);
667 StrictFlag =
Builder.CreateShl(StrictFlag, 6);
669 Value *Flags =
Builder.CreateOr(HasNoWaitFlag, DynCGroupMemFallbackFlag);
670 Flags =
Builder.CreateOr(Flags, StrictFlag);
676 Value *NumThreads3D =
707 auto FnAttrs = Attrs.getFnAttrs();
708 auto RetAttrs = Attrs.getRetAttrs();
710 for (
size_t ArgNo = 0; ArgNo < Fn.
arg_size(); ++ArgNo)
715 bool Param =
true) ->
void {
716 bool HasSignExt = AS.hasAttribute(Attribute::SExt);
717 bool HasZeroExt = AS.hasAttribute(Attribute::ZExt);
718 if (HasSignExt || HasZeroExt) {
719 assert(AS.getNumAttributes() == 1 &&
720 "Currently not handling extension attr combined with others.");
722 if (
auto AK = TargetLibraryInfo::getExtAttrForI32Param(
T, HasSignExt))
725 TargetLibraryInfo::getExtAttrForI32Return(
T, HasSignExt))
732#define OMP_ATTRS_SET(VarName, AttrSet) AttributeSet VarName = AttrSet;
733#include "llvm/Frontend/OpenMP/OMPKinds.def"
737#define OMP_RTL_ATTRS(Enum, FnAttrSet, RetAttrSet, ArgAttrSets) \
739 FnAttrs = FnAttrs.addAttributes(Ctx, FnAttrSet); \
740 addAttrSet(RetAttrs, RetAttrSet, false); \
741 for (size_t ArgNo = 0; ArgNo < ArgAttrSets.size(); ++ArgNo) \
742 addAttrSet(ArgAttrs[ArgNo], ArgAttrSets[ArgNo]); \
743 Fn.setAttributes(AttributeList::get(Ctx, FnAttrs, RetAttrs, ArgAttrs)); \
745#include "llvm/Frontend/OpenMP/OMPKinds.def"
759#define OMP_RTL(Enum, Str, IsVarArg, ReturnType, ...) \
761 FnTy = FunctionType::get(ReturnType, ArrayRef<Type *>{__VA_ARGS__}, \
763 Fn = M.getFunction(Str); \
765#include "llvm/Frontend/OpenMP/OMPKinds.def"
771#define OMP_RTL(Enum, Str, ...) \
773 Fn = Function::Create(FnTy, GlobalValue::ExternalLinkage, Str, M); \
775#include "llvm/Frontend/OpenMP/OMPKinds.def"
779 if (FnID == OMPRTL___kmpc_fork_call || FnID == OMPRTL___kmpc_fork_teams) {
789 LLVMContext::MD_callback,
791 2, {-1, -1},
true)}));
804 assert(Fn &&
"Failed to create OpenMP runtime function");
815 Builder.SetInsertPoint(FiniBB);
827 FiniBB = OtherFiniBB;
829 Builder.SetInsertPoint(FiniBB->getFirstNonPHIIt());
837 auto EndIt = FiniBB->end();
838 if (FiniBB->size() >= 1)
839 if (
auto Prev = std::prev(EndIt); Prev->isTerminator())
844 FiniBB->replaceAllUsesWith(OtherFiniBB);
845 FiniBB->eraseFromParent();
846 FiniBB = OtherFiniBB;
853 assert(Fn &&
"Failed to create OpenMP runtime function pointer");
876 for (
auto Inst =
Block->getReverseIterator()->begin();
877 Inst !=
Block->getReverseIterator()->end();) {
906 Block.getParent()->getEntryBlock().getTerminator()->getIterator();
927 DeferredOutlines.
push_back(std::move(OI));
931 ParallelRegionBlockSet.
clear();
933 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
943 bool ArgsInZeroAddressSpace =
Config.isTargetDevice();
944 std::unique_ptr<CodeExtractor> Extractor =
945 OI->createCodeExtractor(Blocks, ArgsInZeroAddressSpace,
".omp_par");
949 <<
" Exit: " << OI->ExitBB->getName() <<
"\n");
950 assert(Extractor->isEligible() &&
951 "Expected OpenMP outlining to be possible!");
953 for (
auto *V : OI->ExcludeArgsFromAggregate)
954 Extractor->excludeArgFromAggregate(V);
957 Extractor->extractCodeRegion(CEAC, OI->Inputs, OI->Outputs);
961 if (TargetCpuAttr.isStringAttribute())
964 auto TargetFeaturesAttr = OuterFn->
getFnAttribute(
"target-features");
965 if (TargetFeaturesAttr.isStringAttribute())
966 OutlinedFn->
addFnAttr(TargetFeaturesAttr);
969 LLVM_DEBUG(
dbgs() <<
" Outlined function: " << *OutlinedFn <<
"\n");
971 "OpenMP outlined functions should not return a value!");
976 M.getFunctionList().insertAfter(OuterFn->
getIterator(), OutlinedFn);
983 assert(OI->EntryBB->getUniquePredecessor() == &ArtificialEntry);
990 "Expected instructions to add in the outlined region entry");
992 End = ArtificialEntry.
rend();
997 if (
I.isTerminator()) {
999 if (
Instruction *TI = OI->EntryBB->getTerminatorOrNull())
1000 TI->adoptDbgRecords(&ArtificialEntry,
I.getIterator(),
false);
1004 I.moveBeforePreserving(*OI->EntryBB,
1005 OI->EntryBB->getFirstInsertionPt());
1008 OI->EntryBB->moveBefore(&ArtificialEntry);
1015 if (OI->PostOutlineCB)
1016 OI->PostOutlineCB(*OutlinedFn);
1018 if (OI->FixUpNonEntryAllocas)
1050 errs() <<
"Error of kind: " << Kind
1051 <<
" when emitting offload entries and metadata during "
1052 "OMPIRBuilder finalization \n";
1060 if (
Config.isTargetDevice())
1061 applyDeclareTargetGlobalReplacements();
1063 if (
Config.EmitLLVMUsedMetaInfo.value_or(
false)) {
1064 std::vector<WeakTrackingVH> LLVMCompilerUsed = {
1065 M.getGlobalVariable(
"__openmp_nvptx_data_transfer_temporary_storage")};
1066 emitUsed(
"llvm.compiler.used", LLVMCompilerUsed);
1076 assert(Original && Replacement &&
1077 "Null values provided to registerDeclareTargetGlobalReplacement");
1081void OpenMPIRBuilder::applyDeclareTargetGlobalReplacements() {
1087 "A null value was inserted into DeclareTargetGlobalReplacements");
1091 if (!OldGV || !NewGV)
1125 for (
unsigned I = 0, E =
PHI->getNumIncomingValues();
I < E; ++
I) {
1126 if (
PHI->getIncomingValue(
I) != OldGV)
1131 Builder.SetCurrentDebugLocation(
PHI->getDebugLoc());
1133 PHI->setIncomingValue(
I, EdgeLoad);
1139 Builder.SetCurrentDebugLocation(Insn->getDebugLoc());
1155 "Non-default address space declare target global");
1157 unsigned DestAS = ASC->getType()->getPointerAddressSpace();
1158 if (DestAS == 0 && NewGVAS != OldGVAS) {
1159 ASC->replaceAllUsesWith(
Load);
1160 ASC->eraseFromParent();
1165 Insn->replaceUsesOfWith(OldGV,
Load);
1181 ConstantInt::get(I32Ty,
Value), Name);
1194 for (
unsigned I = 0, E =
List.size();
I != E; ++
I)
1198 if (UsedArray.
empty())
1205 GV->setSection(
"llvm.metadata");
1211 auto *Int8Ty =
Builder.getInt8Ty();
1214 ConstantInt::get(Int8Ty, Mode),
Twine(KernelName,
"_exec_mode"));
1222 unsigned Reserve2Flags) {
1224 LocFlags |= OMP_IDENT_FLAG_KMPC;
1231 ConstantInt::get(Int32,
uint32_t(LocFlags)),
1232 ConstantInt::get(Int32, Reserve2Flags),
1233 ConstantInt::get(Int32, SrcLocStrSize), SrcLocStr};
1235 size_t SrcLocStrArgIdx = 4;
1236 if (OpenMPIRBuilder::Ident->getElementType(SrcLocStrArgIdx)
1240 SrcLocStr, OpenMPIRBuilder::Ident->getElementType(SrcLocStrArgIdx));
1247 if (
GV.getValueType() == OpenMPIRBuilder::Ident &&
GV.hasInitializer())
1248 if (
GV.getInitializer() == Initializer)
1253 M, OpenMPIRBuilder::Ident,
1256 M.getDataLayout().getDefaultGlobalsAddressSpace());
1268 SrcLocStrSize = LocStr.
size();
1277 if (
GV.isConstant() &&
GV.hasInitializer() &&
1278 GV.getInitializer() == Initializer)
1281 SrcLocStr =
Builder.CreateGlobalString(
1282 LocStr,
"",
M.getDataLayout().getDefaultGlobalsAddressSpace(),
1290 unsigned Line,
unsigned Column,
1296 Buffer.
append(FunctionName);
1298 Buffer.
append(std::to_string(Line));
1300 Buffer.
append(std::to_string(Column));
1308 StringRef UnknownLoc =
";unknown;unknown;0;0;;";
1319 !DIL->getFilename().empty() ? DIL->getFilename() :
M.getName();
1324 DIL->getColumn(), SrcLocStrSize);
1330 Loc.IP.getBlock()->getParent());
1336 "omp_global_thread_num");
1344 "expected one result pointer type per in_reduction item");
1347 if (OrigPtrs.
empty())
1348 return Builder.saveIP();
1367 for (
unsigned Idx = 0; Idx < OrigPtrs.
size(); ++Idx) {
1370 Value *OrigPtr = OrigPtrs[Idx];
1372 OrigPtrTy && OrigPtrTy->getAddressSpace() != 0)
1373 OrigPtr = Builder.CreateAddrSpaceCast(OrigPtr, PtrTy);
1375 Value *
Priv = Builder.CreateCall(GetThData, {Gtid, NullDesc, OrigPtr},
1381 ResPtrTy && ResPtrTy->getAddressSpace() != 0)
1382 Priv = Builder.CreateAddrSpaceCast(
Priv, ResultPtrTys[Idx]);
1384 MapPrivateCB(Idx,
Priv);
1391 bool ForceSimpleCall,
bool CheckCancelFlag) {
1401 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_FOR;
1404 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_SECTIONS;
1407 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_SINGLE;
1410 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_EXPL;
1413 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL;
1426 bool UseCancelBarrier =
1431 ? OMPRTL___kmpc_cancel_barrier
1432 : OMPRTL___kmpc_barrier),
1435 if (UseCancelBarrier && CheckCancelFlag)
1445 omp::Directive CanceledDirective) {
1450 auto *UI =
Builder.CreateUnreachable();
1458 Builder.SetInsertPoint(ElseTI);
1459 auto ElseIP =
Builder.saveIP();
1467 Builder.SetInsertPoint(ThenTI);
1469 Value *CancelKind =
nullptr;
1470 switch (CanceledDirective) {
1471#define OMP_CANCEL_KIND(Enum, Str, DirectiveEnum, Value) \
1472 case DirectiveEnum: \
1473 CancelKind = Builder.getInt32(Value); \
1475#include "llvm/Frontend/OpenMP/OMPKinds.def"
1492 Builder.SetInsertPoint(UI->getParent());
1493 UI->eraseFromParent();
1500 omp::Directive CanceledDirective) {
1505 auto *UI =
Builder.CreateUnreachable();
1508 Value *CancelKind =
nullptr;
1509 switch (CanceledDirective) {
1510#define OMP_CANCEL_KIND(Enum, Str, DirectiveEnum, Value) \
1511 case DirectiveEnum: \
1512 CancelKind = Builder.getInt32(Value); \
1514#include "llvm/Frontend/OpenMP/OMPKinds.def"
1531 Builder.SetInsertPoint(UI->getParent());
1532 UI->eraseFromParent();
1545 auto *KernelArgsPtr =
1546 Builder.CreateAlloca(OpenMPIRBuilder::KernelArgs,
nullptr,
"kernel_args");
1551 Builder.CreateStructGEP(OpenMPIRBuilder::KernelArgs, KernelArgsPtr,
I);
1554 M.getDataLayout().getPrefTypeAlign(KernelArgs[
I]->getType()));
1558 NumThreads, HostPtr, KernelArgsPtr};
1585 assert(OutlinedFnID &&
"Invalid outlined function ID!");
1589 Value *Return =
nullptr;
1609 Builder, AllocaIP, Return, RTLoc, DeviceID, Args.NumTeams.front(),
1610 Args.NumThreads.front(), OutlinedFnID, ArgsVector));
1617 Builder.CreateCondBr(
Failed, OffloadFailedBlock, OffloadContBlock);
1619 auto CurFn =
Builder.GetInsertBlock()->getParent();
1626 emitBlock(OffloadContBlock, CurFn,
true);
1631 Value *CancelFlag, omp::Directive CanceledDirective) {
1633 "Unexpected cancellation!");
1653 Builder.CreateCondBr(Cmp, NonCancellationBlock, CancellationBlock,
1662 Builder.SetInsertPoint(CancellationBlock);
1663 Builder.CreateBr(*FiniBBOrErr);
1666 Builder.SetInsertPoint(NonCancellationBlock, NonCancellationBlock->
begin());
1678 size_t NumArgs = OutlinedFn.
arg_size();
1679 assert((NumArgs == 2 || NumArgs == 3) &&
1680 "expected a 2-3 argument parallel outlined function");
1681 bool UseArgStruct = NumArgs == 3;
1686 {Builder.getInt16Ty(), Builder.getInt32Ty()},
1690 OutlinedFn.
getName() +
".wrapper", OMPIRBuilder->
M);
1692 WrapperFn->addParamAttr(0, Attribute::NoUndef);
1693 WrapperFn->addParamAttr(0, Attribute::ZExt);
1694 WrapperFn->addParamAttr(1, Attribute::NoUndef);
1698 Builder.SetInsertPoint(EntryBB);
1701 Value *AddrAlloca = Builder.CreateAlloca(Builder.getInt32Ty(),
1703 AddrAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1704 AddrAlloca, Builder.getPtrTy(0),
1705 AddrAlloca->
getName() +
".ascast");
1707 Value *ZeroAlloca = Builder.CreateAlloca(Builder.getInt32Ty(),
1709 ZeroAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1710 ZeroAlloca, Builder.getPtrTy(0),
1711 ZeroAlloca->
getName() +
".ascast");
1713 Value *ArgsAlloca =
nullptr;
1715 ArgsAlloca = Builder.CreateAlloca(Builder.getPtrTy(),
1716 nullptr,
"global_args");
1717 ArgsAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1718 ArgsAlloca, Builder.getPtrTy(0),
1719 ArgsAlloca->
getName() +
".ascast");
1723 Builder.CreateStore(WrapperFn->getArg(1), AddrAlloca);
1724 Builder.CreateStore(Builder.getInt32(0), ZeroAlloca);
1728 llvm::omp::RuntimeFunction::OMPRTL___kmpc_get_shared_variables),
1736 Value *StructArg = Builder.CreateLoad(Builder.getPtrTy(), ArgsAlloca);
1737 StructArg = Builder.CreateInBoundsGEP(Builder.getPtrTy(), StructArg,
1738 {Builder.getInt64(0)});
1739 StructArg = Builder.CreateLoad(Builder.getPtrTy(), StructArg,
"structArg");
1740 Args.push_back(StructArg);
1744 Builder.CreateCall(&OutlinedFn, Args);
1745 Builder.CreateRetVoid();
1760 "Expected at least tid and bounded tid as arguments");
1761 unsigned NumCapturedVars = OutlinedFn.
arg_size() - 2;
1769 OutlinedFn.
addFnAttr(Attribute::NoUnwind);
1772 assert(CI &&
"Expected call instruction to outlined function");
1773 CI->
getParent()->setName(
"omp_parallel");
1775 Builder.SetInsertPoint(CI);
1776 Type *PtrTy = OMPIRBuilder->VoidPtr;
1779 OpenMPIRBuilder ::InsertPointTy CurrentIP = Builder.saveIP();
1783 Value *Args = ArgsAlloca;
1787 Args = Builder.CreatePointerCast(ArgsAlloca, PtrTy);
1788 Builder.restoreIP(CurrentIP);
1791 for (
unsigned Idx = 0; Idx < NumCapturedVars; Idx++) {
1793 Value *StoreAddress = Builder.CreateConstInBoundsGEP2_64(
1795 Builder.CreateStore(V, StoreAddress);
1799 IfCondition ? Builder.CreateSExtOrTrunc(IfCondition, OMPIRBuilder->Int32)
1800 : Builder.getInt32(1);
1801 Value *NumThreadsArg =
1802 NumThreads ? Builder.CreateZExtOrTrunc(NumThreads, OMPIRBuilder->Int32)
1803 : Builder.getInt32(-1);
1813 Value *Parallel60CallArgs[] = {
1818 Builder.getInt32(-1),
1822 Builder.getInt64(NumCapturedVars),
1823 Builder.getInt32(0)};
1831 << *Builder.GetInsertBlock()->getParent() <<
"\n");
1834 Builder.SetInsertPoint(PrivTID);
1836 Builder.CreateStore(Builder.CreateLoad(OMPIRBuilder->Int32, OutlinedAI),
1843 I->eraseFromParent();
1866 if (!
F->hasMetadata(LLVMContext::MD_callback)) {
1874 F->addMetadata(LLVMContext::MD_callback,
1883 OutlinedFn.
addFnAttr(Attribute::NoUnwind);
1886 "Expected at least tid and bounded tid as arguments");
1887 unsigned NumCapturedVars = OutlinedFn.
arg_size() - 2;
1890 CI->
getParent()->setName(
"omp_parallel");
1891 Builder.SetInsertPoint(CI);
1894 Value *ForkCallArgs[] = {Ident, Builder.getInt32(NumCapturedVars),
1898 RealArgs.
append(std::begin(ForkCallArgs), std::end(ForkCallArgs));
1900 Value *
Cond = Builder.CreateSExtOrTrunc(IfCondition, OMPIRBuilder->Int32);
1907 auto PtrTy = OMPIRBuilder->VoidPtr;
1908 if (IfCondition && NumCapturedVars == 0) {
1916 << *Builder.GetInsertBlock()->getParent() <<
"\n");
1919 Builder.SetInsertPoint(PrivTID);
1921 Builder.CreateStore(Builder.CreateLoad(OMPIRBuilder->Int32, OutlinedAI),
1928 I->eraseFromParent();
1936 Value *NumThreads, omp::ProcBindKind ProcBind,
bool IsCancellable) {
1945 const bool NeedThreadID = NumThreads ||
Config.isTargetDevice() ||
1946 (ProcBind != OMP_PROC_BIND_default);
1953 bool ArgsInZeroAddressSpace =
Config.isTargetDevice();
1957 if (NumThreads && !
Config.isTargetDevice()) {
1960 Builder.CreateIntCast(NumThreads, Int32,
false)};
1965 if (ProcBind != OMP_PROC_BIND_default) {
1969 ConstantInt::get(Int32,
unsigned(ProcBind),
true)};
1991 Builder.CreateAlloca(Int32,
nullptr,
"zero.addr");
1994 if (ArgsInZeroAddressSpace &&
M.getDataLayout().getAllocaAddrSpace() != 0) {
1997 TIDAddrAlloca, PointerType ::get(
M.getContext(), 0),
"tid.addr.ascast");
2001 PointerType ::get(
M.getContext(), 0),
2002 "zero.addr.ascast");
2026 if (IP.getBlock()->end() == IP.getPoint()) {
2032 assert(IP.getBlock()->getTerminator()->getNumSuccessors() == 1 &&
2033 IP.getBlock()->getTerminator()->getSuccessor(0) == PRegExitBB &&
2034 "Unexpected insertion point for finalization call!");
2046 Builder.CreateAlloca(Int32,
nullptr,
"tid.addr.local");
2052 Builder.CreateLoad(Int32, ZeroAddr,
"zero.addr.use");
2070 LLVM_DEBUG(
dbgs() <<
"Before body codegen: " << *OuterFn <<
"\n");
2073 assert(BodyGenCB &&
"Expected body generation callback!");
2075 if (
Error Err = BodyGenCB(InnerAllocaIP, CodeGenIP, PRegExitBB))
2078 LLVM_DEBUG(
dbgs() <<
"After body codegen: " << *OuterFn <<
"\n");
2082 bool UsesDeviceSharedMemory =
2084 std::unique_ptr<OutlineInfo> OI =
2085 UsesDeviceSharedMemory
2086 ? std::make_unique<DeviceSharedMemOutlineInfo>(*
this)
2087 : std::make_unique<OutlineInfo>();
2089 if (
Config.isTargetDevice()) {
2091 OI->PostOutlineCB = [=, ToBeDeletedVec =
2092 std::move(ToBeDeleted)](
Function &OutlinedFn) {
2094 IfCondition, NumThreads, PrivTID, PrivTIDAddr,
2095 ThreadID, ToBeDeletedVec);
2099 OI->PostOutlineCB = [=, ToBeDeletedVec =
2100 std::move(ToBeDeleted)](
Function &OutlinedFn) {
2102 PrivTID, PrivTIDAddr, ToBeDeletedVec);
2106 OI->FixUpNonEntryAllocas =
true;
2107 OI->OuterAllocBB = OuterAllocaBlock;
2108 OI->EntryBB = PRegEntryBB;
2109 OI->ExitBB = PRegExitBB;
2110 OI->OuterDeallocBBs.reserve(OuterDeallocBlocks.
size());
2111 copy(OuterDeallocBlocks, OI->OuterDeallocBBs.
end());
2115 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
2127 ".omp_par", ArgsInZeroAddressSpace);
2132 Extractor.findAllocas(CEAC, SinkingCands, HoistingCands, CommonExit);
2134 Extractor.findInputsOutputs(Inputs, Outputs, SinkingCands,
2139 return GV->getValueType() == OpenMPIRBuilder::Ident;
2144 LLVM_DEBUG(
dbgs() <<
"Before privatization: " << *OuterFn <<
"\n");
2150 if (&V == TIDAddr || &V == ZeroAddr) {
2151 OI->ExcludeArgsFromAggregate.push_back(&V);
2156 for (
Use &U : V.uses())
2158 if (ParallelRegionBlockSet.
count(UserI->getParent()))
2168 if (!V.getType()->isPointerTy()) {
2172 Builder.restoreIP(OuterAllocIP);
2174 if (UsesDeviceSharedMemory) {
2177 V.getName() +
".reloaded");
2178 for (
BasicBlock *DeallocBlock : OuterDeallocBlocks)
2180 InsertPointTy(DeallocBlock, DeallocBlock->getFirstInsertionPt()),
2183 Ptr =
Builder.CreateAlloca(V.getType(),
nullptr,
2184 V.getName() +
".reloaded");
2189 Builder.SetInsertPoint(InsertBB,
2194 Builder.restoreIP(InnerAllocaIP);
2195 Inner =
Builder.CreateLoad(V.getType(), Ptr);
2198 Value *ReplacementValue =
nullptr;
2201 ReplacementValue = PrivTID;
2204 PrivCB(InnerAllocaIP,
Builder.saveIP(), V, *Inner, ReplacementValue);
2212 assert(ReplacementValue &&
2213 "Expected copy/create callback to set replacement value!");
2214 if (ReplacementValue == &V)
2219 UPtr->set(ReplacementValue);
2244 for (
Value *Output : Outputs)
2248 "OpenMP outlining should not produce live-out values!");
2250 LLVM_DEBUG(
dbgs() <<
"After privatization: " << *OuterFn <<
"\n");
2252 for (
auto *BB : Blocks)
2253 dbgs() <<
" PBR: " << BB->getName() <<
"\n";
2261 assert(FiniInfo.DK == OMPD_parallel &&
2262 "Unexpected finalization stack state!");
2273 Builder.CreateBr(*FiniBBOrErr);
2277 Term->eraseFromParent();
2283 InsertPointTy AfterIP(UI->getParent(), UI->getParent()->end());
2284 UI->eraseFromParent();
2316 Value *Severity = ConstantInt::get(Int32, IsFatal ? 2 : 1);
2318 Value *Args[] = {Ident, Severity, MessageArg};
2347 static_cast<unsigned int>(RTLDependInfoFields::BaseAddr));
2349 Builder.CreateStore(DepValPtr, Addr);
2352 DependInfo, Entry,
static_cast<unsigned int>(RTLDependInfoFields::Len));
2354 ConstantInt::get(SizeTy,
2359 DependInfo, Entry,
static_cast<unsigned int>(RTLDependInfoFields::Flags));
2361 static_cast<unsigned int>(Dep.
DepKind)),
2374 if (Dependencies.
empty())
2394 Type *DependInfo = OMPBuilder.DependInfo;
2396 Value *DepArray =
nullptr;
2398 Builder.SetInsertPoint(
2402 DepArray = Builder.CreateAlloca(DepArrayTy,
nullptr,
".dep.arr.addr");
2404 Builder.restoreIP(OldIP);
2406 for (
const auto &[DepIdx, Dep] :
enumerate(Dependencies)) {
2408 Builder.CreateConstInBoundsGEP2_64(DepArrayTy, DepArray, 0, DepIdx);
2432 Value *DepArray =
nullptr;
2433 Type *DepArrayTy =
nullptr;
2434 Value *NumDeps =
nullptr;
2437 NumDeps = Dependencies.
NumDeps;
2438 }
else if (!Dependencies.
Deps.empty()) {
2441 Builder.GetInsertBlock()->getParent()->getEntryBlock();
2445 DepArray =
Builder.CreateAlloca(DepArrayTy,
nullptr,
".dep.arr.addr");
2446 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
2449 for (
const auto &[DepIdx, Dep] :
enumerate(Dependencies.
Deps)) {
2451 Builder.CreateConstInBoundsGEP2_64(DepArrayTy, DepArray, 0, DepIdx);
2465 ConstantInt::get(
Builder.getInt32Ty(), 0),
2467 ConstantInt::get(
Builder.getInt32Ty(),
false)};
2470 omp::RuntimeFunction::OMPRTL___kmpc_omp_taskwait_deps_51),
2480 unsigned ProgramAddressSpace = M.getDataLayout().getProgramAddressSpace();
2492 auto *VoidPtrTy =
PointerType::get(Builder.getContext(), ProgramAddressSpace);
2495 Builder.getVoidTy(), {VoidPtrTy, VoidPtrTy, Builder.getInt32Ty()},
2499 "omp_taskloop_dup", M);
2502 Value *LastprivateFlagArg = DupFunction->
getArg(2);
2503 DestTaskArg->
setName(
"dest_task");
2504 SrcTaskArg->
setName(
"src_task");
2505 LastprivateFlagArg->
setName(
"lastprivate_flag");
2508 Builder.SetInsertPoint(
2511 auto GetTaskContextPtrFromArg = [&](
Value *Arg) ->
Value * {
2512 Type *TaskWithPrivatesTy =
2514 Value *TaskPrivates = Builder.CreateGEP(
2515 TaskWithPrivatesTy, Arg, {Builder.getInt32(0), Builder.getInt32(1)});
2516 Value *ContextPtr = Builder.CreateGEP(
2517 PrivatesTy, TaskPrivates,
2518 {Builder.getInt32(0), Builder.getInt32(PrivatesIndex)});
2522 Value *DestTaskContextPtr = GetTaskContextPtrFromArg(DestTaskArg);
2523 Value *SrcTaskContextPtr = GetTaskContextPtrFromArg(SrcTaskArg);
2525 DestTaskContextPtr->
setName(
"destPtr");
2526 SrcTaskContextPtr->
setName(
"srcPtr");
2531 Expected<IRBuilderBase::InsertPoint> AfterIPOrError =
2532 DupCB(AllocaIP, CodeGenIP, DestTaskContextPtr, SrcTaskContextPtr);
2533 if (!AfterIPOrError)
2535 Builder.restoreIP(*AfterIPOrError);
2545 llvm::function_ref<llvm::Expected<llvm::CanonicalLoopInfo *>()> LoopInfo,
2547 Value *GrainSize,
bool NoGroup,
int Sched,
Value *Final,
bool Mergeable,
2549 Value *TaskContextStructPtrVal) {
2554 uint32_t SrcLocStrSize;
2570 if (
Error Err = BodyGenCB(TaskloopAllocaIP, TaskloopBodyIP, TaskloopExitBB))
2573 llvm::Expected<llvm::CanonicalLoopInfo *> result = LoopInfo();
2578 llvm::CanonicalLoopInfo *CLI = result.
get();
2579 auto OI = std::make_unique<OutlineInfo>();
2580 OI->EntryBB = TaskloopAllocaBB;
2581 OI->OuterAllocBB = AllocaIP.getBlock();
2582 OI->ExitBB = TaskloopExitBB;
2583 OI->OuterDeallocBBs.reserve(DeallocBlocks.
size());
2584 copy(DeallocBlocks, OI->OuterDeallocBBs.end());
2587 SmallVector<Instruction *> ToBeDeleted;
2590 Builder, AllocaIP, ToBeDeleted, TaskloopAllocaIP,
"global.tid",
false));
2592 TaskloopAllocaIP,
"lb",
false,
true);
2594 TaskloopAllocaIP,
"ub",
false,
true);
2596 TaskloopAllocaIP,
"step",
false,
true);
2599 OI->Inputs.insert(FakeLB);
2600 OI->Inputs.insert(FakeUB);
2601 OI->Inputs.insert(FakeStep);
2602 if (TaskContextStructPtrVal)
2603 OI->Inputs.insert(TaskContextStructPtrVal);
2604 assert(((TaskContextStructPtrVal && DupCB) ||
2605 (!TaskContextStructPtrVal && !DupCB)) &&
2606 "Task context struct ptr and duplication callback must be both set "
2612 unsigned ProgramAddressSpace =
M.getDataLayout().getProgramAddressSpace();
2616 {FakeLB->getType(), FakeUB->getType(), FakeStep->getType(), PointerTy});
2617 Expected<Value *> TaskDupFnOrErr = createTaskDuplicationFunction(
2620 if (!TaskDupFnOrErr) {
2623 Value *TaskDupFn = *TaskDupFnOrErr;
2625 OI->PostOutlineCB = [
this, Ident, LBVal, UBVal, StepVal, Untied,
2626 TaskloopAllocaBB, CLI, TaskDupFn, ToBeDeleted, IfCond,
2627 GrainSize, NoGroup, Sched, FakeLB, FakeUB, FakeStep,
2628 FakeSharedsTy, Final, Mergeable, Priority,
2629 NumOfCollapseLoops](
Function &OutlinedFn)
mutable {
2631 assert(OutlinedFn.hasOneUse() &&
2632 "there must be a single user for the outlined function");
2639 Value *CastedLBVal =
2640 Builder.CreateIntCast(LBVal,
Builder.getInt64Ty(),
true,
"lb64");
2641 Value *CastedUBVal =
2642 Builder.CreateIntCast(UBVal,
Builder.getInt64Ty(),
true,
"ub64");
2643 Value *CastedStepVal =
2644 Builder.CreateIntCast(StepVal,
Builder.getInt64Ty(),
true,
"step64");
2646 Builder.SetInsertPoint(StaleCI);
2659 Builder.CreateCall(TaskgroupFn, {Ident, ThreadID});
2680 divideCeil(
M.getDataLayout().getTypeSizeInBits(Task), 8));
2682 AllocaInst *ArgStructAlloca =
2684 assert(ArgStructAlloca &&
2685 "Unable to find the alloca instruction corresponding to arguments "
2686 "for extracted function");
2687 std::optional<TypeSize> ArgAllocSize =
2690 "Unable to determine size of arguments for extracted function");
2691 Value *SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
2696 CallInst *TaskData =
Builder.CreateCall(
2697 TaskAllocFn, {Ident, ThreadID,
Flags,
2698 TaskSize, SharedsSize,
2703 Value *TaskShareds =
Builder.CreateLoad(VoidPtr, TaskData);
2704 Builder.CreateMemCpy(TaskShareds, Alignment, Shareds, Alignment,
2709 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(0)});
2712 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(1)});
2715 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(2)});
2721 IfCond ?
Builder.CreateIntCast(IfCond,
Builder.getInt32Ty(),
true)
2727 Value *GrainSizeVal =
2728 GrainSize ?
Builder.CreateIntCast(GrainSize,
Builder.getInt64Ty(),
true)
2730 Value *TaskDup = TaskDupFn;
2732 Value *
Args[] = {Ident, ThreadID, TaskData, IfCondVal, Lb, Ub,
2733 Loadstep, NoGroupVal, SchedVal, GrainSizeVal, TaskDup};
2738 Builder.CreateCall(TaskloopFn, Args);
2745 Builder.CreateCall(EndTaskgroupFn, {Ident, ThreadID});
2750 Builder.SetInsertPoint(TaskloopAllocaBB, TaskloopAllocaBB->begin());
2752 LoadInst *SharedsOutlined =
2753 Builder.CreateLoad(VoidPtr, OutlinedFn.getArg(1));
2754 OutlinedFn.getArg(1)->replaceUsesWithIf(
2756 [SharedsOutlined](Use &U) {
return U.getUser() != SharedsOutlined; });
2759 Type *IVTy =
IV->getType();
2765 Value *TaskLB =
nullptr;
2766 Value *TaskUB =
nullptr;
2767 Value *TaskStep =
nullptr;
2768 Value *LoadTaskLB =
nullptr;
2769 Value *LoadTaskUB =
nullptr;
2770 Value *LoadTaskStep =
nullptr;
2771 for (Instruction &
I : *TaskloopAllocaBB) {
2772 if (
I.getOpcode() == Instruction::GetElementPtr) {
2775 switch (CI->getZExtValue()) {
2787 }
else if (
I.getOpcode() == Instruction::Load) {
2789 if (
Load.getPointerOperand() == TaskLB) {
2790 assert(TaskLB !=
nullptr &&
"Expected value for TaskLB");
2792 }
else if (
Load.getPointerOperand() == TaskUB) {
2793 assert(TaskUB !=
nullptr &&
"Expected value for TaskUB");
2795 }
else if (
Load.getPointerOperand() == TaskStep) {
2796 assert(TaskStep !=
nullptr &&
"Expected value for TaskStep");
2802 Builder.SetInsertPoint(CLI->getPreheader()->getTerminator());
2804 assert(LoadTaskLB !=
nullptr &&
"Expected value for LoadTaskLB");
2805 assert(LoadTaskUB !=
nullptr &&
"Expected value for LoadTaskUB");
2806 assert(LoadTaskStep !=
nullptr &&
"Expected value for LoadTaskStep");
2808 Builder.CreateSub(LoadTaskUB, LoadTaskLB), LoadTaskStep);
2809 Value *TripCount =
Builder.CreateAdd(TripCountMinusOne, One,
"trip_cnt");
2810 Value *CastedTripCount =
Builder.CreateIntCast(TripCount, IVTy,
true);
2811 Value *CastedTaskLB =
Builder.CreateIntCast(LoadTaskLB, IVTy,
true);
2813 CLI->setTripCount(CastedTripCount);
2815 Builder.SetInsertPoint(CLI->getBody(),
2816 CLI->getBody()->getFirstInsertionPt());
2818 if (NumOfCollapseLoops > 1) {
2824 Builder.CreateSub(CastedTaskLB, ConstantInt::get(IVTy, 1)));
2827 for (
auto IVUse = CLI->getIndVar()->uses().begin();
2828 IVUse != CLI->getIndVar()->uses().end(); IVUse++) {
2829 User *IVUser = IVUse->getUser();
2831 if (
Op->getOpcode() == Instruction::URem ||
2832 Op->getOpcode() == Instruction::UDiv) {
2837 for (User *User : UsersToReplace) {
2838 User->replaceUsesOfWith(CLI->getIndVar(), IVPlusTaskLB);
2855 assert(CLI->getIndVar()->getNumUses() == 3 &&
2856 "Canonical loop should have exactly three uses of the ind var");
2857 for (User *IVUser : CLI->getIndVar()->users()) {
2859 if (
Mul->getOpcode() == Instruction::Mul) {
2860 for (User *MulUser :
Mul->users()) {
2862 if (
Add->getOpcode() == Instruction::Add) {
2863 Add->setOperand(1, CastedTaskLB);
2872 FakeLB->replaceAllUsesWith(CastedLBVal);
2873 FakeUB->replaceAllUsesWith(CastedUBVal);
2874 FakeStep->replaceAllUsesWith(CastedStepVal);
2876 I->eraseFromParent();
2881 Builder.SetInsertPoint(TaskloopExitBB, TaskloopExitBB->
begin());
2887 M.getContext(),
M.getDataLayout().getPointerSizeInBits());
2897 bool Mergeable,
Value *EventHandle,
Value *Priority) {
2929 if (
Error Err = BodyGenCB(TaskAllocaIP, TaskBodyIP, TaskExitBB))
2932 auto OI = std::make_unique<OutlineInfo>();
2933 OI->EntryBB = TaskAllocaBB;
2934 OI->OuterAllocBB = AllocaIP.
getBlock();
2935 OI->ExitBB = TaskExitBB;
2936 OI->OuterDeallocBBs.reserve(DeallocBlocks.
size());
2937 copy(DeallocBlocks, OI->OuterDeallocBBs.
end());
2942 Builder, AllocaIP, ToBeDeleted, TaskAllocaIP,
"global.tid",
false));
2944 OI->PostOutlineCB = [
this, Ident, Tied, Final, IfCondition, Dependencies,
2945 Affinities, Mergeable, Priority, EventHandle,
2947 ToBeDeleted](
Function &OutlinedFn)
mutable {
2949 assert(OutlinedFn.hasOneUse() &&
2950 "there must be a single user for the outlined function");
2955 bool HasShareds = StaleCI->
arg_size() > 1;
2956 Builder.SetInsertPoint(StaleCI);
2981 bool UseMergedIf0Path = ConstIfCondition && ConstIfCondition->isZero();
2985 Flags =
Builder.CreateOr(FinalFlag, Flags);
2988 if (Mergeable || UseMergedIf0Path)
3000 divideCeil(
M.getDataLayout().getTypeSizeInBits(Task), 8));
3009 assert(ArgStructAlloca &&
3010 "Unable to find the alloca instruction corresponding to arguments "
3011 "for extracted function");
3012 std::optional<TypeSize> ArgAllocSize =
3015 "Unable to determine size of arguments for extracted function");
3016 SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
3022 TaskAllocFn, {Ident, ThreadID, Flags,
3023 TaskSize, SharedsSize,
3026 if (Affinities.
Count && Affinities.
Info) {
3028 OMPRTL___kmpc_omp_reg_task_with_affinity);
3039 OMPRTL___kmpc_task_allow_completion_event);
3043 Builder.CreatePointerBitCastOrAddrSpaceCast(EventHandle,
3045 EventVal =
Builder.CreatePtrToInt(EventVal,
Builder.getInt64Ty());
3046 Builder.CreateStore(EventVal, EventHandleAddr);
3052 Value *TaskShareds =
Builder.CreateLoad(VoidPtr, TaskData);
3053 Builder.CreateMemCpy(TaskShareds, Alignment, Shareds, Alignment,
3067 Constant *Zero = ConstantInt::get(Int32Ty, 0);
3071 Builder.CreateInBoundsGEP(TaskPtr, TaskData, {Zero, Zero});
3074 VoidPtr, VoidPtr,
Builder.getInt32Ty(), VoidPtr, VoidPtr);
3076 TaskStructType, TaskGEP, {Zero, ConstantInt::get(Int32Ty, 4)});
3079 Value *CmplrData =
Builder.CreateInBoundsGEP(CmplrStructType,
3080 PriorityData, {Zero, Zero});
3081 Builder.CreateStore(Priority, CmplrData);
3084 Value *DepArray =
nullptr;
3085 Value *NumDeps =
nullptr;
3088 NumDeps = Dependencies.
NumDeps;
3089 }
else if (!Dependencies.
Deps.empty()) {
3091 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
3111 if (IfCondition && !UseMergedIf0Path) {
3116 Builder.GetInsertPoint()->getParent()->getTerminator();
3117 Instruction *ThenTI = IfTerminator, *ElseTI =
nullptr;
3118 Builder.SetInsertPoint(IfTerminator);
3121 Builder.SetInsertPoint(ElseTI);
3128 {Ident, ThreadID, NumDeps, DepArray,
3129 ConstantInt::get(
Builder.getInt32Ty(), 0),
3144 Builder.SetInsertPoint(ThenTI);
3152 {Ident, ThreadID, TaskData, NumDeps, DepArray,
3153 ConstantInt::get(
Builder.getInt32Ty(), 0),
3164 Builder.SetInsertPoint(TaskAllocaBB, TaskAllocaBB->
begin());
3166 LoadInst *Shareds =
Builder.CreateLoad(VoidPtr, OutlinedFn.getArg(1));
3167 OutlinedFn.getArg(1)->replaceUsesWithIf(
3168 Shareds, [Shareds](
Use &U) {
return U.getUser() != Shareds; });
3172 I->eraseFromParent();
3176 Builder.SetInsertPoint(TaskExitBB, TaskExitBB->
begin());
3198 if (
Error Err = BodyGenCB(AllocaIP,
Builder.saveIP(), DeallocBlocks))
3201 Builder.SetInsertPoint(TaskgroupExitBB);
3244 unsigned CaseNumber = 0;
3245 for (
auto SectionCB : SectionCBs) {
3247 M.getContext(),
"omp_section_loop.body.case", CurFn,
Continue);
3249 Builder.SetInsertPoint(CaseBB);
3264 Value *LB = ConstantInt::get(I32Ty, 0);
3265 Value *UB = ConstantInt::get(I32Ty, SectionCBs.
size());
3266 Value *ST = ConstantInt::get(I32Ty, 1);
3268 Loc, LoopBodyGenCB, LB, UB, ST,
true,
false, AllocaIP,
"section_loop");
3273 applyStaticWorkshareLoop(
Loc.DL, *
LoopInfo, AllocaIP,
3274 WorksharingLoopType::ForStaticLoop, !IsNowait);
3280 assert(LoopFini &&
"Bad structure of static workshare loop finalization");
3284 assert(FiniInfo.DK == OMPD_sections &&
3285 "Unexpected finalization stack state!");
3286 if (
Error Err = FiniInfo.mergeFiniBB(
Builder, LoopFini))
3300 if (IP.getBlock()->end() != IP.getPoint())
3311 auto *CaseBB =
Loc.IP.getBlock();
3312 auto *CondBB = CaseBB->getSinglePredecessor()->getSinglePredecessor();
3313 auto *ExitBB = CondBB->getTerminator()->getSuccessor(1);
3319 Directive OMPD = Directive::OMPD_sections;
3322 return EmitOMPInlinedRegion(OMPD,
nullptr,
nullptr, BodyGenCB, FiniCBWrapper,
3333Value *OpenMPIRBuilder::getGPUThreadID() {
3336 OMPRTL___kmpc_get_hardware_thread_id_in_block),
3340Value *OpenMPIRBuilder::getGPUWarpSize() {
3345Value *OpenMPIRBuilder::getNVPTXWarpID() {
3346 unsigned LaneIDBits =
Log2_32(
Config.getGridValue().GV_Warp_Size);
3347 return Builder.CreateAShr(getGPUThreadID(), LaneIDBits,
"nvptx_warp_id");
3350Value *OpenMPIRBuilder::getNVPTXLaneID() {
3351 unsigned LaneIDBits =
Log2_32(
Config.getGridValue().GV_Warp_Size);
3352 assert(LaneIDBits < 32 &&
"Invalid LaneIDBits size in NVPTX device.");
3353 unsigned LaneIDMask = ~0
u >> (32u - LaneIDBits);
3354 return Builder.CreateAnd(getGPUThreadID(),
Builder.getInt32(LaneIDMask),
3361 uint64_t FromSize =
M.getDataLayout().getTypeStoreSize(FromType);
3362 uint64_t ToSize =
M.getDataLayout().getTypeStoreSize(ToType);
3363 assert(FromSize > 0 &&
"From size must be greater than zero");
3364 assert(ToSize > 0 &&
"To size must be greater than zero");
3365 if (FromType == ToType)
3367 if (FromSize == ToSize)
3368 return Builder.CreateBitCast(From, ToType);
3370 return Builder.CreateIntCast(From, ToType,
true);
3376 Value *ValCastItem =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3377 CastItem,
Builder.getPtrTy(0));
3378 Builder.CreateStore(From, ValCastItem);
3379 return Builder.CreateLoad(ToType, CastItem);
3386 uint64_t
Size =
M.getDataLayout().getTypeStoreSize(ElementType);
3387 assert(
Size <= 8 &&
"Unsupported bitwidth in shuffle instruction");
3391 Value *ElemCast = castValueToType(AllocaIP, Element, CastTy);
3393 Builder.CreateIntCast(getGPUWarpSize(),
Builder.getInt16Ty(),
true);
3395 Size <= 4 ? RuntimeFunction::OMPRTL___kmpc_shuffle_int32
3396 : RuntimeFunction::OMPRTL___kmpc_shuffle_int64);
3397 Value *WarpSizeCast =
3399 Value *ShuffleCall =
3401 return castValueToType(AllocaIP, ShuffleCall, CastTy);
3408 uint64_t
Size =
M.getDataLayout().getTypeStoreSize(ElemType);
3420 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3421 Value *ElemPtr = DstAddr;
3422 Value *Ptr = SrcAddr;
3423 for (
unsigned IntSize = 8; IntSize >= 1; IntSize /= 2) {
3427 Ptr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3430 Builder.CreateGEP(ElemType, SrcAddr, {ConstantInt::get(IndexTy, 1)});
3431 ElemPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3435 if ((
Size / IntSize) > 1) {
3436 Value *PtrEnd =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3437 SrcAddrGEP,
Builder.getPtrTy());
3454 Builder.CreatePointerBitCastOrAddrSpaceCast(Ptr,
Builder.getPtrTy()));
3456 Builder.CreateICmpSGT(PtrDiff,
Builder.getInt64(IntSize - 1)), ThenBB,
3459 Value *Res = createRuntimeShuffleFunction(
3462 IntType, Ptr,
M.getDataLayout().getPrefTypeAlign(ElemType)),
3464 Builder.CreateAlignedStore(Res, ElemPtr,
3465 M.getDataLayout().getPrefTypeAlign(ElemType));
3467 Builder.CreateGEP(IntType, Ptr, {ConstantInt::get(IndexTy, 1)});
3468 Value *LocalElemPtr =
3469 Builder.CreateGEP(IntType, ElemPtr, {ConstantInt::get(IndexTy, 1)});
3475 Value *Res = createRuntimeShuffleFunction(
3476 AllocaIP,
Builder.CreateLoad(IntType, Ptr), IntType,
Offset);
3479 Res =
Builder.CreateTrunc(Res, ElemType);
3480 Builder.CreateStore(Res, ElemPtr);
3481 Ptr =
Builder.CreateGEP(IntType, Ptr, {ConstantInt::get(IndexTy, 1)});
3483 Builder.CreateGEP(IntType, ElemPtr, {ConstantInt::get(IndexTy, 1)});
3489Error OpenMPIRBuilder::emitReductionListCopy(
3494 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3495 Value *RemoteLaneOffset = CopyOptions.RemoteLaneOffset;
3499 for (
auto En :
enumerate(ReductionInfos)) {
3501 Value *SrcElementAddr =
nullptr;
3502 AllocaInst *DestAlloca =
nullptr;
3503 Value *DestElementAddr =
nullptr;
3504 Value *DestElementPtrAddr =
nullptr;
3506 bool ShuffleInElement =
false;
3509 bool UpdateDestListPtr =
false;
3513 ReductionArrayTy, SrcBase,
3514 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
3515 SrcElementAddr =
Builder.CreateLoad(
Builder.getPtrTy(), SrcElementPtrAddr);
3519 DestElementPtrAddr =
Builder.CreateInBoundsGEP(
3520 ReductionArrayTy, DestBase,
3521 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
3522 bool IsByRefElem = (!IsByRef.
empty() && IsByRef[En.index()]);
3528 Type *DestAllocaType =
3529 IsByRefElem ? RI.ByRefAllocatedType : RI.ElementType;
3530 DestAlloca =
Builder.CreateAlloca(DestAllocaType,
nullptr,
3531 ".omp.reduction.element");
3533 M.getDataLayout().getPrefTypeAlign(DestAllocaType));
3534 DestElementAddr = DestAlloca;
3537 DestElementAddr->
getName() +
".ascast");
3539 ShuffleInElement =
true;
3540 UpdateDestListPtr =
true;
3552 if (ShuffleInElement) {
3553 Type *ShuffleType = RI.ElementType;
3554 Value *ShuffleSrcAddr = SrcElementAddr;
3555 Value *ShuffleDestAddr = DestElementAddr;
3556 AllocaInst *LocalStorage =
nullptr;
3559 assert(RI.ByRefElementType &&
"Expected by-ref element type to be set");
3560 assert(RI.ByRefAllocatedType &&
3561 "Expected by-ref allocated type to be set");
3566 ShuffleType = RI.ByRefElementType;
3568 if (RI.DataPtrPtrGen) {
3571 Builder.saveIP(), ShuffleSrcAddr, ShuffleSrcAddr);
3574 return GenResult.takeError();
3583 LocalStorage =
Builder.CreateAlloca(ShuffleType);
3585 ShuffleDestAddr = LocalStorage;
3590 ShuffleDestAddr = DestElementAddr;
3594 shuffleAndStore(AllocaIP, ShuffleSrcAddr, ShuffleDestAddr, ShuffleType,
3595 RemoteLaneOffset, ReductionArrayTy, IsByRefElem);
3597 if (IsByRefElem && RI.DataPtrPtrGen) {
3599 Value *DestDescriptorAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3600 DestAlloca,
Builder.getPtrTy(),
".ascast");
3603 DestDescriptorAddr, LocalStorage, SrcElementAddr,
3604 RI.ByRefAllocatedType, RI.DataPtrPtrGen);
3607 return GenResult.takeError();
3610 switch (RI.EvaluationKind) {
3612 Value *Elem =
Builder.CreateLoad(RI.ElementType, SrcElementAddr);
3614 Builder.CreateStore(Elem, DestElementAddr);
3618 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
3619 RI.ElementType, SrcElementAddr, 0, 0,
".realp");
3621 RI.ElementType->getStructElementType(0), SrcRealPtr,
".real");
3623 RI.ElementType, SrcElementAddr, 0, 1,
".imagp");
3625 RI.ElementType->getStructElementType(1), SrcImgPtr,
".imag");
3627 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
3628 RI.ElementType, DestElementAddr, 0, 0,
".realp");
3629 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
3630 RI.ElementType, DestElementAddr, 0, 1,
".imagp");
3631 Builder.CreateStore(SrcReal, DestRealPtr);
3632 Builder.CreateStore(SrcImg, DestImgPtr);
3637 M.getDataLayout().getTypeStoreSize(RI.ElementType));
3639 DestElementAddr,
M.getDataLayout().getPrefTypeAlign(RI.ElementType),
3640 SrcElementAddr,
M.getDataLayout().getPrefTypeAlign(RI.ElementType),
3652 if (UpdateDestListPtr) {
3653 Value *CastDestAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3654 DestElementAddr,
Builder.getPtrTy(),
3655 DestElementAddr->
getName() +
".ascast");
3656 Builder.CreateStore(CastDestAddr, DestElementPtrAddr);
3663Expected<Function *> OpenMPIRBuilder::emitInterWarpCopyFunction(
3666 IRBuilder<>::InsertPointGuard IPG(
Builder);
3667 LLVMContext &Ctx =
M.getContext();
3669 Builder.getVoidTy(), {Builder.getPtrTy(), Builder.getInt32Ty()},
3673 "_omp_reduction_inter_warp_copy_func", &
M);
3679 Builder.SetInsertPoint(EntryBB);
3697 StringRef TransferMediumName =
3698 "__openmp_nvptx_data_transfer_temporary_storage";
3699 GlobalVariable *TransferMedium =
M.getGlobalVariable(TransferMediumName);
3700 unsigned WarpSize =
Config.getGridValue().GV_Warp_Size;
3702 if (!TransferMedium) {
3703 TransferMedium =
new GlobalVariable(
3711 Value *GPUThreadID = getGPUThreadID();
3713 Value *LaneID = getNVPTXLaneID();
3715 Value *WarpID = getNVPTXWarpID();
3719 Builder.GetInsertBlock()->getFirstInsertionPt());
3723 AllocaInst *ReduceListAlloca =
Builder.CreateAlloca(
3724 Arg0Type,
nullptr, ReduceListArg->
getName() +
".addr");
3725 AllocaInst *NumWarpsAlloca =
3726 Builder.CreateAlloca(Arg1Type,
nullptr, NumWarpsArg->
getName() +
".addr");
3727 Value *ReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3728 ReduceListAlloca, Arg0Type, ReduceListAlloca->
getName() +
".ascast");
3729 Value *NumWarpsAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3730 NumWarpsAlloca,
Builder.getPtrTy(0),
3731 NumWarpsAlloca->
getName() +
".ascast");
3732 Builder.CreateStore(ReduceListArg, ReduceListAddrCast);
3733 Builder.CreateStore(NumWarpsArg, NumWarpsAddrCast);
3742 for (
auto En :
enumerate(ReductionInfos)) {
3748 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
3749 unsigned RealTySize =
M.getDataLayout().getTypeAllocSize(
3750 IsByRefElem ? RI.ByRefElementType : RI.ElementType);
3751 for (
unsigned TySize = 4; TySize > 0 && RealTySize > 0; TySize /= 2) {
3754 unsigned NumIters = RealTySize / TySize;
3757 Value *Cnt =
nullptr;
3758 Value *CntAddr =
nullptr;
3765 Builder.CreateAlloca(
Builder.getInt32Ty(),
nullptr,
".cnt.addr");
3767 CntAddr =
Builder.CreateAddrSpaceCast(CntAddr,
Builder.getPtrTy(),
3768 CntAddr->
getName() +
".ascast");
3780 Cnt, ConstantInt::get(
Builder.getInt32Ty(), NumIters));
3781 Builder.CreateCondBr(Cmp, BodyBB, ExitBB);
3788 omp::Directive::OMPD_unknown,
3792 return BarrierIP1.takeError();
3798 Value *IsWarpMaster =
Builder.CreateIsNull(LaneID,
"warp_master");
3799 Builder.CreateCondBr(IsWarpMaster, ThenBB, ElseBB);
3803 auto *RedListArrayTy =
3806 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3808 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
3809 {ConstantInt::get(IndexTy, 0),
3810 ConstantInt::get(IndexTy, En.index())});
3814 if (IsByRefElem && RI.DataPtrPtrGen) {
3816 RI.DataPtrPtrGen(
Builder.saveIP(), ElemPtr, ElemPtr);
3819 return GenRes.takeError();
3830 ArrayTy, TransferMedium, {
Builder.getInt64(0), WarpID});
3835 Builder.CreateStore(Elem, MediumPtr,
3847 omp::Directive::OMPD_unknown,
3851 return BarrierIP2.takeError();
3858 Value *NumWarpsVal =
3861 Value *IsActiveThread =
3862 Builder.CreateICmpULT(GPUThreadID, NumWarpsVal,
"is_active_thread");
3863 Builder.CreateCondBr(IsActiveThread, W0ThenBB, W0ElseBB);
3870 ArrayTy, TransferMedium, {
Builder.getInt64(0), GPUThreadID});
3872 Value *TargetElemPtrPtr =
3873 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
3874 {ConstantInt::get(IndexTy, 0),
3875 ConstantInt::get(IndexTy, En.index())});
3876 Value *TargetElemPtrVal =
3878 Value *TargetElemPtr = TargetElemPtrVal;
3880 if (IsByRefElem && RI.DataPtrPtrGen) {
3882 RI.DataPtrPtrGen(
Builder.saveIP(), TargetElemPtr, TargetElemPtr);
3885 return GenRes.takeError();
3887 TargetElemPtr =
Builder.CreateLoad(
Builder.getPtrTy(), TargetElemPtr);
3895 Value *SrcMediumValue =
3896 Builder.CreateLoad(CType, SrcMediumPtrVal,
true);
3897 Builder.CreateStore(SrcMediumValue, TargetElemPtr);
3907 Cnt, ConstantInt::get(
Builder.getInt32Ty(), 1));
3908 Builder.CreateStore(Cnt, CntAddr,
false);
3910 auto *CurFn =
Builder.GetInsertBlock()->getParent();
3914 RealTySize %= TySize;
3923Expected<Function *> OpenMPIRBuilder::emitShuffleAndReduceFunction(
3926 LLVMContext &Ctx =
M.getContext();
3927 IRBuilder<>::InsertPointGuard IPG(
Builder);
3928 FunctionType *FuncTy =
3930 {Builder.getPtrTy(), Builder.getInt16Ty(),
3931 Builder.getInt16Ty(), Builder.getInt16Ty()},
3935 "_omp_reduction_shuffle_and_reduce_func", &
M);
3946 Builder.SetInsertPoint(EntryBB);
3958 Type *ReduceListArgType = ReduceListArg->
getType();
3962 ReduceListArgType,
nullptr, ReduceListArg->
getName() +
".addr");
3963 Value *LaneIdAlloca =
Builder.CreateAlloca(LaneIDArgType,
nullptr,
3964 LaneIDArg->
getName() +
".addr");
3966 LaneIDArgType,
nullptr, RemoteLaneOffsetArg->
getName() +
".addr");
3967 Value *AlgoVerAlloca =
Builder.CreateAlloca(LaneIDArgType,
nullptr,
3968 AlgoVerArg->
getName() +
".addr");
3975 RedListArrayTy,
nullptr,
".omp.reduction.remote_reduce_list");
3977 Value *ReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3978 ReduceListAlloca, ReduceListArgType,
3979 ReduceListAlloca->
getName() +
".ascast");
3980 Value *LaneIdAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3981 LaneIdAlloca, LaneIDArgPtrType, LaneIdAlloca->
getName() +
".ascast");
3982 Value *RemoteLaneOffsetAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3983 RemoteLaneOffsetAlloca, LaneIDArgPtrType,
3984 RemoteLaneOffsetAlloca->
getName() +
".ascast");
3985 Value *AlgoVerAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3986 AlgoVerAlloca, LaneIDArgPtrType, AlgoVerAlloca->
getName() +
".ascast");
3987 Value *RemoteListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3988 RemoteReductionListAlloca,
Builder.getPtrTy(),
3989 RemoteReductionListAlloca->
getName() +
".ascast");
3991 Builder.CreateStore(ReduceListArg, ReduceListAddrCast);
3992 Builder.CreateStore(LaneIDArg, LaneIdAddrCast);
3993 Builder.CreateStore(RemoteLaneOffsetArg, RemoteLaneOffsetAddrCast);
3994 Builder.CreateStore(AlgoVerArg, AlgoVerAddrCast);
3996 Value *ReduceList =
Builder.CreateLoad(ReduceListArgType, ReduceListAddrCast);
3997 Value *LaneId =
Builder.CreateLoad(LaneIDArgType, LaneIdAddrCast);
3998 Value *RemoteLaneOffset =
3999 Builder.CreateLoad(LaneIDArgType, RemoteLaneOffsetAddrCast);
4000 Value *AlgoVer =
Builder.CreateLoad(LaneIDArgType, AlgoVerAddrCast);
4007 Error EmitRedLsCpRes = emitReductionListCopy(
4009 ReduceList, RemoteListAddrCast, IsByRef,
4010 {RemoteLaneOffset,
nullptr,
nullptr});
4013 return EmitRedLsCpRes;
4038 Value *LaneComp =
Builder.CreateICmpULT(LaneId, RemoteLaneOffset);
4043 Value *Algo2AndLaneIdComp =
Builder.CreateAnd(Algo2, LaneIdComp);
4044 Value *RemoteOffsetComp =
4046 Value *CondAlgo2 =
Builder.CreateAnd(Algo2AndLaneIdComp, RemoteOffsetComp);
4047 Value *CA0OrCA1 =
Builder.CreateOr(CondAlgo0, CondAlgo1);
4048 Value *CondReduce =
Builder.CreateOr(CA0OrCA1, CondAlgo2);
4054 Builder.CreateCondBr(CondReduce, ThenBB, ElseBB);
4056 Value *LocalReduceListPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4057 ReduceList,
Builder.getPtrTy());
4058 Value *RemoteReduceListPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4059 RemoteListAddrCast,
Builder.getPtrTy());
4061 ->addFnAttr(Attribute::NoUnwind);
4072 Value *LaneIdGtOffset =
Builder.CreateICmpUGE(LaneId, RemoteLaneOffset);
4073 Value *CondCopy =
Builder.CreateAnd(Algo1, LaneIdGtOffset);
4078 Builder.CreateCondBr(CondCopy, CpyThenBB, CpyElseBB);
4082 EmitRedLsCpRes = emitReductionListCopy(
4084 RemoteListAddrCast, ReduceList, IsByRef);
4087 return EmitRedLsCpRes;
4102OpenMPIRBuilder::generateReductionDescriptor(
4104 Type *DescriptorType,
4110 Value *DescriptorSize =
4111 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(DescriptorType));
4113 DescriptorAddr,
M.getDataLayout().getPrefTypeAlign(DescriptorType),
4114 SrcDescriptorAddr,
M.getDataLayout().getPrefTypeAlign(DescriptorType),
4118 Value *DataPtrField;
4120 DataPtrPtrGen(
Builder.saveIP(), DescriptorAddr, DataPtrField);
4123 return GenResult.takeError();
4126 DataPtr,
Builder.getPtrTy(),
".ascast"),
4132Expected<Value *> OpenMPIRBuilder::createReductionDescriptorCopy(
4134 Value *SrcDescriptorAddr,
Type *DescriptorPtrTy,
const Twine &Name) {
4138 AllocaInst *DescriptorAlloca =
4139 Builder.CreateAlloca(RI.ByRefAllocatedType,
nullptr, Name);
4141 M.getDataLayout().getPrefTypeAlign(RI.ByRefAllocatedType));
4142 Value *DescriptorAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4143 DescriptorAlloca, DescriptorPtrTy,
4144 DescriptorAlloca->
getName() +
".ascast");
4149 generateReductionDescriptor(DescriptorAddr, DataPtr, SrcDescriptorAddr,
4150 RI.ByRefAllocatedType, RI.DataPtrPtrGen);
4152 return GenResult.takeError();
4154 return DescriptorAddr;
4157Expected<Function *> OpenMPIRBuilder::emitListToGlobalCopyFunction(
4160 IRBuilder<>::InsertPointGuard IPG(
Builder);
4161 LLVMContext &Ctx =
M.getContext();
4164 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4168 "_omp_reduction_list_to_global_copy_func", &
M);
4175 Builder.SetInsertPoint(EntryBlock);
4186 BufferArg->
getName() +
".addr");
4190 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4191 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4192 BufferArgAlloca,
Builder.getPtrTy(),
4193 BufferArgAlloca->
getName() +
".ascast");
4194 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4195 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4196 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4197 ReduceListArgAlloca,
Builder.getPtrTy(),
4198 ReduceListArgAlloca->
getName() +
".ascast");
4200 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4201 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4202 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4204 Value *LocalReduceList =
4206 Value *BufferArgVal =
4210 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4211 for (
auto En :
enumerate(ReductionInfos)) {
4213 auto *RedListArrayTy =
4217 RedListArrayTy, LocalReduceList,
4218 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4224 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferArgVal, Idxs);
4226 ReductionsBufferTy, BufferVD, 0, En.index());
4228 switch (RI.EvaluationKind) {
4230 Value *TargetElement;
4232 if (IsByRef.
empty() || !IsByRef[En.index()]) {
4233 TargetElement =
Builder.CreateLoad(RI.ElementType, ElemPtr);
4235 if (RI.DataPtrPtrGen) {
4237 RI.DataPtrPtrGen(
Builder.saveIP(), ElemPtr, ElemPtr);
4240 return GenResult.takeError();
4244 TargetElement =
Builder.CreateLoad(RI.ByRefElementType, ElemPtr);
4247 Builder.CreateStore(TargetElement, GlobVal);
4251 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4252 RI.ElementType, ElemPtr, 0, 0,
".realp");
4254 RI.ElementType->getStructElementType(0), SrcRealPtr,
".real");
4256 RI.ElementType, ElemPtr, 0, 1,
".imagp");
4258 RI.ElementType->getStructElementType(1), SrcImgPtr,
".imag");
4260 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4261 RI.ElementType, GlobVal, 0, 0,
".realp");
4262 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
4263 RI.ElementType, GlobVal, 0, 1,
".imagp");
4264 Builder.CreateStore(SrcReal, DestRealPtr);
4265 Builder.CreateStore(SrcImg, DestImgPtr);
4270 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(RI.ElementType));
4272 GlobVal,
M.getDataLayout().getPrefTypeAlign(RI.ElementType), ElemPtr,
4273 M.getDataLayout().getPrefTypeAlign(RI.ElementType), SizeVal,
false);
4283Expected<Function *> OpenMPIRBuilder::emitListToGlobalReduceFunction(
4286 IRBuilder<>::InsertPointGuard IPG(
Builder);
4287 LLVMContext &Ctx =
M.getContext();
4290 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4294 "_omp_reduction_list_to_global_reduce_func", &
M);
4301 Builder.SetInsertPoint(EntryBlock);
4312 BufferArg->
getName() +
".addr");
4316 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4317 auto *RedListArrayTy =
4322 Value *LocalReduceList =
4323 Builder.CreateAlloca(RedListArrayTy,
nullptr,
".omp.reduction.red_list");
4327 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4328 BufferArgAlloca,
Builder.getPtrTy(),
4329 BufferArgAlloca->
getName() +
".ascast");
4330 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4331 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4332 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4333 ReduceListArgAlloca,
Builder.getPtrTy(),
4334 ReduceListArgAlloca->
getName() +
".ascast");
4335 Value *LocalReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4336 LocalReduceList,
Builder.getPtrTy(),
4337 LocalReduceList->
getName() +
".ascast");
4339 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4340 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4341 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4346 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4347 for (
auto En :
enumerate(ReductionInfos)) {
4350 Value *TargetElementPtrPtr =
Builder.CreateInBoundsGEP(
4351 RedListArrayTy, LocalReduceListAddrCast,
4352 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4354 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4356 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4357 ReductionsBufferTy, BufferVD, 0, En.index());
4359 if (!IsByRef.
empty() && IsByRef[En.index()] && RI.DataPtrPtrGen) {
4363 Value *SrcElementPtrPtr =
4364 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
4365 {ConstantInt::get(IndexTy, 0),
4366 ConstantInt::get(IndexTy, En.index())});
4367 Value *SrcDescriptorAddr =
4371 Expected<Value *> ByRefAlloc = createReductionDescriptorCopy(
4372 AllocaIP, RI, GlobValPtr, SrcDescriptorAddr,
Builder.getPtrTy());
4376 Builder.CreateStore(*ByRefAlloc, TargetElementPtrPtr);
4378 Builder.CreateStore(GlobValPtr, TargetElementPtrPtr);
4386 ->addFnAttr(Attribute::NoUnwind);
4391Expected<Function *> OpenMPIRBuilder::emitGlobalToListCopyFunction(
4394 IRBuilder<>::InsertPointGuard IPG(
Builder);
4395 LLVMContext &Ctx =
M.getContext();
4398 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4402 "_omp_reduction_global_to_list_copy_func", &
M);
4409 Builder.SetInsertPoint(EntryBlock);
4420 BufferArg->
getName() +
".addr");
4424 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4425 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4426 BufferArgAlloca,
Builder.getPtrTy(),
4427 BufferArgAlloca->
getName() +
".ascast");
4428 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4429 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4430 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4431 ReduceListArgAlloca,
Builder.getPtrTy(),
4432 ReduceListArgAlloca->
getName() +
".ascast");
4433 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4434 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4435 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4437 Value *LocalReduceList =
4442 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4443 for (
auto En :
enumerate(ReductionInfos)) {
4444 const OpenMPIRBuilder::ReductionInfo &RI = En.value();
4445 auto *RedListArrayTy =
4449 RedListArrayTy, LocalReduceList,
4450 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4455 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4456 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4457 ReductionsBufferTy, BufferVD, 0, En.index());
4463 if (!IsByRef.
empty() && IsByRef[En.index()]) {
4470 return GenResult.takeError();
4476 Value *TargetElement =
Builder.CreateLoad(ElemType, GlobValPtr);
4477 Builder.CreateStore(TargetElement, ElemPtr);
4481 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4490 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4492 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
4494 Builder.CreateStore(SrcReal, DestRealPtr);
4495 Builder.CreateStore(SrcImg, DestImgPtr);
4502 ElemPtr,
M.getDataLayout().getPrefTypeAlign(RI.
ElementType),
4503 GlobValPtr,
M.getDataLayout().getPrefTypeAlign(RI.
ElementType),
4514Expected<Function *> OpenMPIRBuilder::emitGlobalToListReduceFunction(
4517 IRBuilder<>::InsertPointGuard IPG(
Builder);
4518 LLVMContext &Ctx =
M.getContext();
4521 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4525 "_omp_reduction_global_to_list_reduce_func", &
M);
4532 Builder.SetInsertPoint(EntryBlock);
4543 BufferArg->
getName() +
".addr");
4547 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4553 Value *LocalReduceList =
4554 Builder.CreateAlloca(RedListArrayTy,
nullptr,
".omp.reduction.red_list");
4558 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4559 BufferArgAlloca,
Builder.getPtrTy(),
4560 BufferArgAlloca->
getName() +
".ascast");
4561 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4562 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4563 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4564 ReduceListArgAlloca,
Builder.getPtrTy(),
4565 ReduceListArgAlloca->
getName() +
".ascast");
4566 Value *ReductionList =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4567 LocalReduceList,
Builder.getPtrTy(),
4568 LocalReduceList->
getName() +
".ascast");
4570 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4571 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4572 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4577 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4578 for (
auto En :
enumerate(ReductionInfos)) {
4581 Value *TargetElementPtrPtr =
Builder.CreateInBoundsGEP(
4582 RedListArrayTy, ReductionList,
4583 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4586 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4587 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4588 ReductionsBufferTy, BufferVD, 0, En.index());
4590 if (!IsByRef.
empty() && IsByRef[En.index()] && RI.DataPtrPtrGen) {
4592 Value *ReduceListVal =
4594 Value *SrcElementPtrPtr =
4595 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceListVal,
4596 {ConstantInt::get(IndexTy, 0),
4597 ConstantInt::get(IndexTy, En.index())});
4598 Value *SrcDescriptorAddr =
4602 Expected<Value *> ByRefAlloc = createReductionDescriptorCopy(
4603 AllocaIP, RI, GlobValPtr, SrcDescriptorAddr,
Builder.getPtrTy());
4607 Builder.CreateStore(*ByRefAlloc, TargetElementPtrPtr);
4609 Builder.CreateStore(GlobValPtr, TargetElementPtrPtr);
4617 ->addFnAttr(Attribute::NoUnwind);
4622std::string OpenMPIRBuilder::getReductionFuncName(StringRef Name)
const {
4623 std::string Suffix =
4625 return (Name + Suffix).str();
4628Expected<Function *> OpenMPIRBuilder::createReductionFunction(
4631 AttributeList FuncAttrs) {
4632 IRBuilder<>::InsertPointGuard IPG(
Builder);
4634 {Builder.getPtrTy(), Builder.getPtrTy()},
4636 std::string
Name = getReductionFuncName(ReducerName);
4645 Builder.SetInsertPoint(EntryBB);
4650 Value *LHSArrayPtr =
nullptr;
4651 Value *RHSArrayPtr =
nullptr;
4658 Builder.CreateAlloca(Arg0Type,
nullptr, Arg0->
getName() +
".addr");
4660 Builder.CreateAlloca(Arg1Type,
nullptr, Arg1->
getName() +
".addr");
4661 Value *LHSAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4662 LHSAlloca, Arg0Type, LHSAlloca->
getName() +
".ascast");
4663 Value *RHSAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4664 RHSAlloca, Arg1Type, RHSAlloca->
getName() +
".ascast");
4665 Builder.CreateStore(Arg0, LHSAddrCast);
4666 Builder.CreateStore(Arg1, RHSAddrCast);
4667 LHSArrayPtr =
Builder.CreateLoad(Arg0Type, LHSAddrCast);
4668 RHSArrayPtr =
Builder.CreateLoad(Arg1Type, RHSAddrCast);
4672 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4674 for (
auto En :
enumerate(ReductionInfos)) {
4677 RedArrayTy, RHSArrayPtr,
4678 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4680 Value *RHSPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4681 RHSI8Ptr, RI.PrivateVariable->getType(),
4682 RHSI8Ptr->
getName() +
".ascast");
4685 RedArrayTy, LHSArrayPtr,
4686 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4688 Value *LHSPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4689 LHSI8Ptr, RI.Variable->getType(), LHSI8Ptr->
getName() +
".ascast");
4698 if (!IsByRef.
empty() && !IsByRef[En.index()]) {
4699 LHS =
Builder.CreateLoad(RI.ElementType, LHSPtr);
4700 RHS =
Builder.CreateLoad(RI.ElementType, RHSPtr);
4707 return AfterIP.takeError();
4708 if (!
Builder.GetInsertBlock())
4709 return ReductionFunc;
4713 if (!IsByRef.
empty() && !IsByRef[En.index()])
4714 Builder.CreateStore(Reduced, LHSPtr);
4719 for (
auto En :
enumerate(ReductionInfos)) {
4720 unsigned Index = En.index();
4722 Value *LHSFixupPtr, *RHSFixupPtr;
4723 Builder.restoreIP(RI.ReductionGenClang(
4724 Builder.saveIP(), Index, &LHSFixupPtr, &RHSFixupPtr, ReductionFunc));
4729 LHSPtrs[Index], [ReductionFunc](
const Use &U) {
4734 RHSPtrs[Index], [ReductionFunc](
const Use &U) {
4748 return ReductionFunc;
4756 assert(RI.Variable &&
"expected non-null variable");
4757 assert(RI.PrivateVariable &&
"expected non-null private variable");
4758 assert((RI.ReductionGen || RI.ReductionGenClang) &&
4759 "expected non-null reduction generator callback");
4762 RI.Variable->getType() == RI.PrivateVariable->getType() &&
4763 "expected variables and their private equivalents to have the same "
4766 assert(RI.Variable->getType()->isPointerTy() &&
4767 "expected variables to be pointers");
4784 ArrayRef<bool> IsByRef,
bool IsNoWait,
bool IsTeamsReduction,
bool IsSPMD,
4786 Value *SrcLocInfo) {
4800 if (ReductionInfos.
size() == 0)
4810 Builder.SetInsertPoint(InsertBlock, InsertBlock->
end());
4814 AttributeList FuncAttrs;
4815 AttrBuilder AttrBldr(Ctx);
4817 AttrBldr.addAttribute(Attr);
4818 AttrBldr.removeAttribute(Attribute::OptimizeNone);
4819 FuncAttrs = FuncAttrs.addFnAttributes(Ctx, AttrBldr);
4823 Builder.GetInsertBlock()->getParent()->getName(), ReductionInfos, IsByRef,
4825 if (!ReductionResult)
4827 Function *ReductionFunc = *ReductionResult;
4831 if (GridValue.has_value())
4832 Config.setGridValue(GridValue.value());
4847 Builder.getPtrTy(
M.getDataLayout().getProgramAddressSpace());
4851 Value *ReductionListAlloca =
4852 Builder.CreateAlloca(RedArrayTy,
nullptr,
".omp.reduction.red_list");
4853 Value *ReductionList =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4854 ReductionListAlloca, PtrTy, ReductionListAlloca->
getName() +
".ascast");
4857 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4858 for (
auto En :
enumerate(ReductionInfos)) {
4861 RedArrayTy, ReductionList,
4862 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4865 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
4870 Builder.CreatePointerBitCastOrAddrSpaceCast(PrivateVar, PtrTy);
4871 Builder.CreateStore(CastElem, ElemPtr);
4875 ReductionInfos, ReductionFunc, FuncAttrs, IsByRef);
4881 emitInterWarpCopyFunction(
Loc, ReductionInfos, FuncAttrs, IsByRef);
4887 Value *RL =
Builder.CreatePointerBitCastOrAddrSpaceCast(ReductionList, PtrTy);
4896 unsigned MaxDataSize = 0;
4898 for (
auto En :
enumerate(ReductionInfos)) {
4902 Type *RedTypeArg = (!IsByRef.
empty() && IsByRef[En.index()])
4903 ? En.value().ByRefElementType
4904 : En.value().ElementType;
4905 auto Size =
M.getDataLayout().getTypeStoreSize(RedTypeArg);
4906 if (
Size > MaxDataSize)
4910 Value *ReductionDataSize =
4911 Builder.getInt64(MaxDataSize * ReductionInfos.
size());
4915 Function *CopyScratchToListFunc =
nullptr;
4917 Value *ScratchForCopyBack =
nullptr;
4920 Value *RLForCopyBack = RL;
4922 bool IsAtomicReduction =
4925 if (!IsTeamsReduction) {
4926 Value *SarFuncCast =
4927 Builder.CreatePointerBitCastOrAddrSpaceCast(*SarFunc, FuncPtrTy);
4929 Builder.CreatePointerBitCastOrAddrSpaceCast(WcFunc, FuncPtrTy);
4930 Value *Args[] = {SrcLocInfo, ReductionDataSize, RL, SarFuncCast,
4933 RuntimeFunction::OMPRTL___kmpc_nvptx_parallel_reduce_nowait_v2);
4935 }
else if (IsAtomicReduction) {
4939 RuntimeFunction::OMPRTL___kmpc_is_team_main_thread);
4944 Ctx, ReductionTypeArgs,
"struct._globalized_locals_ty");
4947 ReductionInfos, ReductionsBufferTy, FuncAttrs, IsByRef);
4952 ReductionInfos, ReductionsBufferTy, FuncAttrs, IsByRef);
4957 ReductionInfos, ReductionFunc, ReductionsBufferTy, FuncAttrs, IsByRef);
4980 Value *RuntimeRL = RL;
4987 ReductionsBufferTy,
nullptr,
".omp.reduction.scratch");
4988 Value *PerThreadScratch =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4989 PerThreadScratchAlloca, PtrTy,
4990 PerThreadScratchAlloca->
getName() +
".ascast");
4993 Value *PerThreadRedListAlloca =
4994 Builder.CreateAlloca(RedArrayTy,
nullptr,
4995 ".omp.reduction.per_thread_red_list");
4996 RuntimeRL =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4997 PerThreadRedListAlloca, PtrTy,
4998 PerThreadRedListAlloca->
getName() +
".ascast");
5003 for (
auto En :
enumerate(ReductionInfos)) {
5005 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
5008 ReductionsBufferTy, PerThreadScratch, 0, En.index());
5009 Value *Slot =
Builder.CreateConstInBoundsGEP2_32(RedArrayTy, RuntimeRL,
5012 Value *RuntimeListEntry = FieldPtr;
5014 Value *SrcDescriptor =
5017 AllocaIP, RI, FieldPtr, SrcDescriptor, PtrTy);
5020 RuntimeListEntry = *Descriptor;
5022 Builder.CreateStore(RuntimeListEntry, Slot);
5028 Type *CopyArg0Ty = (*LtGCFunc)->getFunctionType()->getParamType(0);
5029 Type *CopyArg2Ty = (*LtGCFunc)->getFunctionType()->getParamType(2);
5030 ScratchForCopyBack =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5031 PerThreadScratch, CopyArg0Ty);
5033 Builder.CreatePointerBitCastOrAddrSpaceCast(RL, CopyArg2Ty);
5041 *LtGCFunc, {ScratchForCopyBack,
Builder.getInt32(0), RLForCopyBack});
5042 CopyScratchToListFunc = *GtLCFunc;
5045 Value *Args3[] = {SrcLocInfo, RuntimeRL, *SarFunc, WcFunc,
5046 *LtGCFunc, *GtLCFunc, *GtLRFunc};
5049 RuntimeFunction::OMPRTL___kmpc_gpu_xteam_reduce_nowait);
5069 if (ScratchForCopyBack) {
5072 CopyScratchToListFunc,
5073 {ScratchForCopyBack,
Builder.getInt32(0), RLForCopyBack});
5077 for (
auto En :
enumerate(ReductionInfos)) {
5083 if (IsAtomicReduction) {
5099 Value *LHSPtr, *RHSPtr;
5101 &LHSPtr, &RHSPtr, CurFunc));
5107 RedValue =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5109 if (RHSPtr->
getType() != RHS->getType())
5111 Builder.CreatePointerBitCastOrAddrSpaceCast(RHS, RHSPtr->
getType());
5122 if (IsByRef.
empty() || !IsByRef[En.index()]) {
5124 "red.value." +
Twine(En.index()));
5135 if (!IsByRef.
empty() && !IsByRef[En.index()])
5140 if (ContinuationBlock) {
5141 Builder.CreateBr(ContinuationBlock);
5142 Builder.SetInsertPoint(ContinuationBlock);
5144 Config.setEmitLLVMUsed();
5155 ".omp.reduction.func", &M);
5166 Builder.SetInsertPoint(ReductionFuncBlock);
5168 Value *LHSArrayPtr =
nullptr;
5169 Value *RHSArrayPtr =
nullptr;
5180 Builder.CreateAlloca(Arg0Type,
nullptr, Arg0->
getName() +
".addr");
5182 Builder.CreateAlloca(Arg1Type,
nullptr, Arg1->
getName() +
".addr");
5183 Value *LHSAddrCast =
5184 Builder.CreatePointerBitCastOrAddrSpaceCast(LHSAlloca, Arg0Type);
5185 Value *RHSAddrCast =
5186 Builder.CreatePointerBitCastOrAddrSpaceCast(RHSAlloca, Arg1Type);
5187 Builder.CreateStore(Arg0, LHSAddrCast);
5188 Builder.CreateStore(Arg1, RHSAddrCast);
5189 LHSArrayPtr = Builder.CreateLoad(Arg0Type, LHSAddrCast);
5190 RHSArrayPtr = Builder.CreateLoad(Arg1Type, RHSAddrCast);
5192 LHSArrayPtr = ReductionFunc->
getArg(0);
5193 RHSArrayPtr = ReductionFunc->
getArg(1);
5196 unsigned NumReductions = ReductionInfos.
size();
5199 for (
auto En :
enumerate(ReductionInfos)) {
5201 Value *LHSI8PtrPtr = Builder.CreateConstInBoundsGEP2_64(
5202 RedArrayTy, LHSArrayPtr, 0, En.index());
5203 Value *LHSI8Ptr = Builder.CreateLoad(Builder.getPtrTy(), LHSI8PtrPtr);
5204 Value *LHSPtr = Builder.CreatePointerBitCastOrAddrSpaceCast(
5207 Value *RHSI8PtrPtr = Builder.CreateConstInBoundsGEP2_64(
5208 RedArrayTy, RHSArrayPtr, 0, En.index());
5209 Value *RHSI8Ptr = Builder.CreateLoad(Builder.getPtrTy(), RHSI8PtrPtr);
5210 Value *RHSPtr = Builder.CreatePointerBitCastOrAddrSpaceCast(
5219 Builder.restoreIP(*AfterIP);
5221 if (!Builder.GetInsertBlock())
5225 if (!IsByRef[En.index()])
5226 Builder.CreateStore(Reduced, LHSPtr);
5228 Builder.CreateRetVoid();
5235 bool IsNoWait,
bool IsTeamsReduction) {
5239 IsByRef, IsNoWait, IsTeamsReduction);
5246 if (ReductionInfos.
size() == 0)
5256 unsigned NumReductions = ReductionInfos.
size();
5259 Value *RedArray =
Builder.CreateAlloca(RedArrayTy,
nullptr,
"red.array");
5261 Builder.SetInsertPoint(InsertBlock, InsertBlock->
end());
5263 for (
auto En :
enumerate(ReductionInfos)) {
5264 unsigned Index = En.index();
5266 Value *RedArrayElemPtr =
Builder.CreateConstInBoundsGEP2_64(
5267 RedArrayTy, RedArray, 0, Index,
"red.array.elem." +
Twine(Index));
5274 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
5284 ? IdentFlag::OMP_IDENT_FLAG_ATOMIC_REDUCE
5289 unsigned RedArrayByteSize =
DL.getTypeStoreSize(RedArrayTy);
5290 Constant *RedArraySize = ConstantInt::get(IndexTy, RedArrayByteSize);
5292 Value *Lock = getOMPCriticalRegionLock(
".reduction");
5294 IsNoWait ? RuntimeFunction::OMPRTL___kmpc_reduce_nowait
5295 : RuntimeFunction::OMPRTL___kmpc_reduce);
5298 {Ident, ThreadId, NumVariables, RedArraySize,
5299 RedArray, ReductionFunc, Lock},
5310 Builder.CreateSwitch(ReduceCall, ContinuationBlock, 2);
5311 Switch->addCase(
Builder.getInt32(1), NonAtomicRedBlock);
5312 Switch->addCase(
Builder.getInt32(2), AtomicRedBlock);
5317 Builder.SetInsertPoint(NonAtomicRedBlock);
5318 for (
auto En :
enumerate(ReductionInfos)) {
5324 if (!IsByRef[En.index()]) {
5326 "red.value." +
Twine(En.index()));
5328 Value *PrivateRedValue =
5330 "red.private.value." +
Twine(En.index()));
5338 if (!
Builder.GetInsertBlock())
5341 if (!IsByRef[En.index()])
5345 IsNoWait ? RuntimeFunction::OMPRTL___kmpc_end_reduce_nowait
5346 : RuntimeFunction::OMPRTL___kmpc_end_reduce);
5348 Builder.CreateBr(ContinuationBlock);
5353 Builder.SetInsertPoint(AtomicRedBlock);
5354 if (CanGenerateAtomic &&
llvm::none_of(IsByRef, [](
bool P) {
return P; })) {
5361 if (!
Builder.GetInsertBlock())
5364 Builder.CreateBr(ContinuationBlock);
5377 if (!
Builder.GetInsertBlock())
5380 Builder.SetInsertPoint(ContinuationBlock);
5391 Directive OMPD = Directive::OMPD_master;
5396 Value *Args[] = {Ident, ThreadId};
5404 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
5415 Directive OMPD = Directive::OMPD_masked;
5421 Value *ArgsEnd[] = {Ident, ThreadId};
5429 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
5439 Call->setDoesNotThrow();
5454 bool IsInclusive,
ScanInfo *ScanRedInfo) {
5456 llvm::Error Err = emitScanBasedDirectiveDeclsIR(AllocaIP, ScanVars,
5457 ScanVarsType, ScanRedInfo);
5468 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5471 Type *DestTy = ScanVarsType[i];
5472 Value *Val =
Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5475 Builder.CreateStore(Src, Val);
5480 Builder.GetInsertBlock()->getParent());
5483 IV = ScanRedInfo->
IV;
5486 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5489 Type *DestTy = ScanVarsType[i];
5491 Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5493 Builder.CreateStore(Src, ScanVars[i]);
5507 Builder.GetInsertBlock()->getParent());
5512Error OpenMPIRBuilder::emitScanBasedDirectiveDeclsIR(
5516 Builder.restoreIP(AllocaIP);
5518 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5520 Builder.CreateAlloca(Builder.getPtrTy(),
nullptr,
"vla");
5527 Builder.restoreIP(CodeGenIP);
5529 Builder.CreateAdd(ScanRedInfo->
Span, Builder.getInt32(1));
5530 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5534 Value *Buff = Builder.CreateMalloc(
IntPtrTy, ScanVarsType[i], Allocsize,
5535 AllocSpan,
nullptr,
"arr");
5536 Builder.CreateStore(Buff, (*(ScanRedInfo->
ScanBuffPtrs))[ScanVars[i]]);
5554 Builder.SetInsertPoint(
Builder.GetInsertBlock()->getTerminator());
5563Error OpenMPIRBuilder::emitScanBasedDirectiveFinalsIR(
5569 Value *PrivateVar = RedInfo.PrivateVariable;
5570 Value *OrigVar = RedInfo.Variable;
5574 Type *SrcTy = RedInfo.ElementType;
5579 Builder.CreateStore(Src, OrigVar);
5602 Builder.SetInsertPoint(
Builder.GetInsertBlock()->getTerminator());
5627 Builder.GetInsertBlock()->getModule(),
5634 Builder.GetInsertBlock()->getModule(),
5640 llvm::ConstantInt::get(ScanRedInfo->
Span->
getType(), 1));
5641 Builder.SetInsertPoint(InputBB);
5644 Builder.SetInsertPoint(LoopBB);
5660 Builder.CreateCondBr(CmpI, InnerLoopBB, InnerExitBB);
5662 Builder.SetInsertPoint(InnerLoopBB);
5666 Value *ReductionVal = RedInfo.PrivateVariable;
5669 Type *DestTy = RedInfo.ElementType;
5672 Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5675 Builder.CreateInBoundsGEP(DestTy, Buff, OffsetIval,
"arrayOffset");
5680 RedInfo.ReductionGen(
Builder.saveIP(), LHS, RHS, Result);
5683 Builder.CreateStore(Result, LHSPtr);
5686 IVal, llvm::ConstantInt::get(
Builder.getInt32Ty(), 1));
5688 CmpI =
Builder.CreateICmpUGE(NextIVal, Pow2K);
5689 Builder.CreateCondBr(CmpI, InnerLoopBB, InnerExitBB);
5692 Counter, llvm::ConstantInt::get(Counter->
getType(), 1));
5698 Builder.CreateCondBr(Cmp, LoopBB, ExitBB);
5719 Error Err = emitScanBasedDirectiveFinalsIR(ReductionInfos, ScanRedInfo);
5726Error OpenMPIRBuilder::emitScanBasedDirectiveIR(
5738 Error Err = InputLoopGen();
5749 Error Err = ScanLoopGen(Builder.saveIP());
5756void OpenMPIRBuilder::createScanBBs(ScanInfo *ScanRedInfo) {
5793 Builder.SetInsertPoint(Preheader);
5796 Builder.SetInsertPoint(Header);
5797 PHINode *IndVarPHI =
Builder.CreatePHI(IndVarTy, 2,
"omp_" + Name +
".iv");
5798 IndVarPHI->
addIncoming(ConstantInt::get(IndVarTy, 0), Preheader);
5803 Builder.CreateICmpULT(IndVarPHI, TripCount,
"omp_" + Name +
".cmp");
5804 Builder.CreateCondBr(Cmp, Body, Exit);
5809 Builder.SetInsertPoint(Latch);
5811 "omp_" + Name +
".next",
true);
5822 CL->Header = Header;
5841 NextBB, NextBB, Name);
5873 Value *Start,
Value *Stop,
Value *Step,
bool IsSigned,
bool InclusiveStop,
5882 ComputeLoc, Start, Stop, Step, IsSigned, InclusiveStop, Name);
5883 ScanRedInfo->
Span = TripCount;
5889 ScanRedInfo->
IV =
IV;
5890 createScanBBs(ScanRedInfo);
5893 assert(Terminator->getNumSuccessors() == 1);
5894 BasicBlock *ContinueBlock = Terminator->getSuccessor(0);
5897 Builder.GetInsertBlock()->getParent());
5900 Builder.GetInsertBlock()->getParent());
5901 Builder.CreateBr(ContinueBlock);
5907 const auto &&InputLoopGen = [&]() ->
Error {
5909 Builder.saveIP(), BodyGen, Start, Stop, Step, IsSigned, InclusiveStop,
5910 ComputeIP, Name,
true, ScanRedInfo);
5914 Builder.restoreIP((*LoopInfo)->getAfterIP());
5920 InclusiveStop, ComputeIP, Name,
true, ScanRedInfo);
5924 Builder.restoreIP((*LoopInfo)->getAfterIP());
5928 Error Err = emitScanBasedDirectiveIR(InputLoopGen, ScanLoopGen, ScanRedInfo);
5936 bool IsSigned,
bool InclusiveStop,
const Twine &Name) {
5946 assert(IndVarTy == Stop->
getType() &&
"Stop type mismatch");
5947 assert(IndVarTy == Step->
getType() &&
"Step type mismatch");
5951 ConstantInt *Zero = ConstantInt::get(IndVarTy, 0);
5967 Incr =
Builder.CreateSelect(IsNeg,
Builder.CreateNeg(Step), Step);
5970 Span =
Builder.CreateSub(UB, LB,
"",
false,
true);
5974 Span =
Builder.CreateSub(Stop, Start,
"",
true);
5979 Value *CountIfLooping;
5980 if (InclusiveStop) {
5981 CountIfLooping =
Builder.CreateAdd(
Builder.CreateUDiv(Span, Incr), One);
5987 CountIfLooping =
Builder.CreateSelect(OneCmp, One, CountIfTwo);
5990 return Builder.CreateSelect(ZeroCmp, Zero, CountIfLooping,
5991 "omp_" + Name +
".tripcount");
5996 Value *Start,
Value *Stop,
Value *Step,
bool IsSigned,
bool InclusiveStop,
6003 ComputeLoc, Start, Stop, Step, IsSigned, InclusiveStop, Name);
6010 ScanRedInfo->
IV = IndVar;
6011 return BodyGenCB(
Builder.saveIP(), IndVar);
6017 Builder.getCurrentDebugLocation());
6028 unsigned Bitwidth = Ty->getIntegerBitWidth();
6031 M, omp::RuntimeFunction::OMPRTL___kmpc_dist_for_static_init_4u);
6034 M, omp::RuntimeFunction::OMPRTL___kmpc_dist_for_static_init_8u);
6044 unsigned Bitwidth = Ty->getIntegerBitWidth();
6047 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_init_4u);
6050 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_init_8u);
6058 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6060 "Require dedicated allocate IP");
6066 uint32_t SrcLocStrSize;
6070 case WorksharingLoopType::ForStaticLoop:
6071 Flag = OMP_IDENT_FLAG_WORK_LOOP;
6073 case WorksharingLoopType::DistributeStaticLoop:
6074 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6076 case WorksharingLoopType::DistributeForStaticLoop:
6077 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE | OMP_IDENT_FLAG_WORK_LOOP;
6084 Type *IVTy =
IV->getType();
6085 FunctionCallee StaticInit =
6086 LoopType == WorksharingLoopType::DistributeForStaticLoop
6089 FunctionCallee StaticFini =
6093 Builder.SetInsertPoint(AllocaIP.getBlock()->getFirstNonPHIOrDbgOrAlloca());
6096 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6097 Value *PLowerBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.lowerbound");
6098 Value *PUpperBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.upperbound");
6099 Value *PStride =
Builder.CreateAlloca(IVTy,
nullptr,
"p.stride");
6108 Constant *One = ConstantInt::get(IVTy, 1);
6109 Builder.CreateStore(Zero, PLowerBound);
6111 Builder.CreateStore(UpperBound, PUpperBound);
6112 Builder.CreateStore(One, PStride);
6118 (LoopType == WorksharingLoopType::DistributeStaticLoop)
6119 ? OMPScheduleType::OrderedDistribute
6122 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6126 auto BuildInitCall = [LoopType, SrcLoc, ThreadNum, PLastIter, PLowerBound,
6127 PUpperBound, IVTy, PStride, One,
Zero, StaticInit,
6130 PLowerBound, PUpperBound});
6131 if (LoopType == WorksharingLoopType::DistributeForStaticLoop) {
6132 Value *PDistUpperBound =
6133 Builder.CreateAlloca(IVTy,
nullptr,
"p.distupperbound");
6134 Args.push_back(PDistUpperBound);
6139 BuildInitCall(SchedulingType,
Builder);
6140 if (HasDistSchedule &&
6141 LoopType != WorksharingLoopType::DistributeStaticLoop) {
6142 Constant *DistScheduleSchedType = ConstantInt::get(
6147 BuildInitCall(DistScheduleSchedType,
Builder);
6149 Value *LowerBound =
Builder.CreateLoad(IVTy, PLowerBound);
6150 Value *InclusiveUpperBound =
Builder.CreateLoad(IVTy, PUpperBound);
6151 Value *TripCountMinusOne =
Builder.CreateSub(InclusiveUpperBound, LowerBound);
6152 Value *TripCount =
Builder.CreateAdd(TripCountMinusOne, One);
6153 CLI->setTripCount(TripCount);
6159 CLI->mapIndVar([&](Instruction *OldIV) ->
Value * {
6163 return Builder.CreateAdd(OldIV, LowerBound);
6175 omp::Directive::OMPD_for,
false,
6178 return BarrierIP.takeError();
6205 Reachable.insert(
Block);
6215 Ctx, {
MDString::get(Ctx,
"llvm.loop.parallel_accesses"), AccessGroup}));
6219OpenMPIRBuilder::applyStaticChunkedWorkshareLoop(
6223 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6224 assert((ChunkSize || DistScheduleChunkSize) &&
"Chunk size is required");
6229 Type *IVTy =
IV->getType();
6231 "Max supported tripcount bitwidth is 64 bits");
6233 :
Type::getInt64Ty(Ctx);
6236 Constant *One = ConstantInt::get(InternalIVTy, 1);
6241 SmallVector<Instruction *> UIs;
6242 for (BasicBlock &BB : *
F)
6243 if (!BB.hasTerminator())
6244 UIs.
push_back(
new UnreachableInst(
F->getContext(), &BB));
6249 LoopInfo &&LI = LIA.
run(*
F,
FAM);
6250 for (Instruction *
I : UIs)
6251 I->eraseFromParent();
6254 if (ChunkSize || DistScheduleChunkSize)
6259 FunctionCallee StaticInit =
6261 FunctionCallee StaticFini =
6267 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6268 Value *PLowerBound =
6269 Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.lowerbound");
6270 Value *PUpperBound =
6271 Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.upperbound");
6272 Value *PStride =
Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.stride");
6281 ChunkSize ? ChunkSize : Zero, InternalIVTy,
"chunksize");
6282 Value *CastedDistScheduleChunkSize =
Builder.CreateZExtOrTrunc(
6283 DistScheduleChunkSize ? DistScheduleChunkSize : Zero, InternalIVTy,
6284 "distschedulechunksize");
6285 Value *CastedTripCount =
6286 Builder.CreateZExt(OrigTripCount, InternalIVTy,
"tripcount");
6289 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6291 ConstantInt::get(I32Type,
static_cast<int>(DistScheduleSchedType));
6292 Builder.CreateStore(Zero, PLowerBound);
6293 Value *OrigUpperBound =
Builder.CreateSub(CastedTripCount, One);
6294 Value *IsTripCountZero =
Builder.CreateICmpEQ(CastedTripCount, Zero);
6296 Builder.CreateSelect(IsTripCountZero, Zero, OrigUpperBound);
6297 Builder.CreateStore(UpperBound, PUpperBound);
6298 Builder.CreateStore(One, PStride);
6302 uint32_t SrcLocStrSize;
6305 if (DistScheduleSchedType != OMPScheduleType::None) {
6306 Flag |= OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6311 auto BuildInitCall = [StaticInit, SrcLoc, ThreadNum, PLastIter, PLowerBound,
6312 PUpperBound, PStride, One,
6313 this](
Value *SchedulingType,
Value *ChunkSize,
6316 StaticInit, {SrcLoc, ThreadNum,
6317 SchedulingType, PLastIter,
6318 PLowerBound, PUpperBound,
6322 BuildInitCall(SchedulingType, CastedChunkSize,
Builder);
6323 if (DistScheduleSchedType != OMPScheduleType::None &&
6324 SchedType != OMPScheduleType::OrderedDistributeChunked &&
6325 SchedType != OMPScheduleType::OrderedDistribute) {
6329 BuildInitCall(DistSchedulingType, CastedDistScheduleChunkSize,
Builder);
6333 Value *FirstChunkStart =
6334 Builder.CreateLoad(InternalIVTy, PLowerBound,
"omp_firstchunk.lb");
6335 Value *FirstChunkStop =
6336 Builder.CreateLoad(InternalIVTy, PUpperBound,
"omp_firstchunk.ub");
6337 Value *FirstChunkEnd =
Builder.CreateAdd(FirstChunkStop, One);
6339 Builder.CreateSub(FirstChunkEnd, FirstChunkStart,
"omp_chunk.range");
6340 Value *NextChunkStride =
6341 Builder.CreateLoad(InternalIVTy, PStride,
"omp_dispatch.stride");
6345 Value *DispatchCounter;
6353 DispatchCounter = Counter;
6356 FirstChunkStart, CastedTripCount, NextChunkStride,
6379 Value *ChunkEnd =
Builder.CreateAdd(DispatchCounter, ChunkRange);
6380 Value *IsLastChunk =
6381 Builder.CreateICmpUGE(ChunkEnd, CastedTripCount,
"omp_chunk.is_last");
6382 Value *CountUntilOrigTripCount =
6383 Builder.CreateSub(CastedTripCount, DispatchCounter);
6385 IsLastChunk, CountUntilOrigTripCount, ChunkRange,
"omp_chunk.tripcount");
6386 Value *BackcastedChunkTC =
6387 Builder.CreateTrunc(ChunkTripCount, IVTy,
"omp_chunk.tripcount.trunc");
6388 CLI->setTripCount(BackcastedChunkTC);
6393 Value *BackcastedDispatchCounter =
6394 Builder.CreateTrunc(DispatchCounter, IVTy,
"omp_dispatch.iv.trunc");
6395 CLI->mapIndVar([&](Instruction *) ->
Value * {
6397 return Builder.CreateAdd(
IV, BackcastedDispatchCounter);
6410 return AfterIP.takeError();
6425static FunctionCallee
6428 unsigned Bitwidth = Ty->getIntegerBitWidth();
6431 case WorksharingLoopType::ForStaticLoop:
6434 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_loop_4u);
6437 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_loop_8u);
6439 case WorksharingLoopType::DistributeStaticLoop:
6442 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_static_loop_4u);
6445 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_static_loop_8u);
6447 case WorksharingLoopType::DistributeForStaticLoop:
6450 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_for_static_loop_4u);
6453 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_for_static_loop_8u);
6456 if (Bitwidth != 32 && Bitwidth != 64) {
6468 Function &LoopBodyFn,
bool NoLoop) {
6479 if (LoopType == WorksharingLoopType::DistributeStaticLoop) {
6480 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6481 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), 0));
6482 Builder.restoreIP({InsertBlock, std::prev(InsertBlock->
end())});
6487 M, omp::RuntimeFunction::OMPRTL_omp_get_num_threads);
6488 Builder.restoreIP({InsertBlock, std::prev(InsertBlock->
end())});
6492 Builder.CreateZExtOrTrunc(NumThreads, TripCountTy,
"num.threads.cast"));
6493 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6494 if (LoopType == WorksharingLoopType::DistributeForStaticLoop) {
6495 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6496 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), NoLoop));
6498 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), 0));
6522 Builder.restoreIP({Preheader, Preheader->
end()});
6525 Builder.CreateBr(CLI->
getExit());
6533 CleanUpInfo.
collectBlocks(RegionBlockSet, BlocksToBeRemoved);
6541 "Expected unique undroppable user of outlined function");
6543 assert(OutlinedFnCallInstruction &&
"Expected outlined function call");
6545 "Expected outlined function call to be located in loop preheader");
6547 if (OutlinedFnCallInstruction->
arg_size() > 1)
6554 LoopBodyArg, TripCount, OutlinedFn, NoLoop);
6556 for (
auto &ToBeDeletedItem : ToBeDeleted)
6557 ToBeDeletedItem->eraseFromParent();
6564 uint32_t SrcLocStrSize;
6568 case WorksharingLoopType::ForStaticLoop:
6569 Flag = OMP_IDENT_FLAG_WORK_LOOP;
6571 case WorksharingLoopType::DistributeStaticLoop:
6572 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6574 case WorksharingLoopType::DistributeForStaticLoop:
6575 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE | OMP_IDENT_FLAG_WORK_LOOP;
6580 auto OI = std::make_unique<OutlineInfo>();
6585 SmallVector<Instruction *, 4> ToBeDeleted;
6587 OI->OuterAllocBB = AllocaIP.getBlock();
6610 SmallPtrSet<BasicBlock *, 32> ParallelRegionBlockSet;
6612 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
6614 CodeExtractorAnalysisCache CEAC(*OuterFn);
6615 CodeExtractor Extractor(Blocks,
6629 SetVector<Value *> SinkingCands, HoistingCands;
6633 Extractor.findAllocas(CEAC, SinkingCands, HoistingCands, CommonExit);
6640 for (
auto Use :
Users) {
6642 if (ParallelRegionBlockSet.
count(Inst->getParent())) {
6643 Inst->replaceUsesOfWith(CLI->
getIndVar(), NewLoopCntLoad);
6649 OI->ExcludeArgsFromAggregate.push_back(NewLoopCntLoad);
6656 OI->PostOutlineCB = [=, ToBeDeletedVec =
6657 std::move(ToBeDeleted)](
Function &OutlinedFn) {
6667 bool NeedsBarrier, omp::ScheduleKind SchedKind,
Value *ChunkSize,
6668 bool HasSimdModifier,
bool HasMonotonicModifier,
6669 bool HasNonmonotonicModifier,
bool HasOrderedClause,
6671 Value *DistScheduleChunkSize) {
6672 if (
Config.isTargetDevice())
6673 return applyWorkshareLoopTarget(
DL, CLI, AllocaIP, LoopType, NoLoop);
6675 SchedKind, ChunkSize, HasSimdModifier, HasMonotonicModifier,
6676 HasNonmonotonicModifier, HasOrderedClause, DistScheduleChunkSize);
6678 bool IsOrdered = (EffectiveScheduleType & OMPScheduleType::ModifierOrdered) ==
6679 OMPScheduleType::ModifierOrdered;
6681 if (HasDistSchedule) {
6682 DistScheduleSchedType = DistScheduleChunkSize
6683 ? OMPScheduleType::OrderedDistributeChunked
6684 : OMPScheduleType::OrderedDistribute;
6686 switch (EffectiveScheduleType & ~OMPScheduleType::ModifierMask) {
6687 case OMPScheduleType::BaseStatic:
6688 case OMPScheduleType::BaseDistribute:
6689 assert((!ChunkSize || !DistScheduleChunkSize) &&
6690 "No chunk size with static-chunked schedule");
6691 if (IsOrdered && !HasDistSchedule)
6692 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6693 NeedsBarrier, ChunkSize);
6695 if (DistScheduleChunkSize)
6696 return applyStaticChunkedWorkshareLoop(
6697 DL, CLI, AllocaIP, NeedsBarrier, ChunkSize, EffectiveScheduleType,
6698 DistScheduleChunkSize, DistScheduleSchedType);
6699 return applyStaticWorkshareLoop(
DL, CLI, AllocaIP, LoopType, NeedsBarrier,
6702 case OMPScheduleType::BaseStaticChunked:
6703 case OMPScheduleType::BaseDistributeChunked:
6704 if (IsOrdered && !HasDistSchedule)
6705 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6706 NeedsBarrier, ChunkSize);
6708 return applyStaticChunkedWorkshareLoop(
6709 DL, CLI, AllocaIP, NeedsBarrier, ChunkSize, EffectiveScheduleType,
6710 DistScheduleChunkSize, DistScheduleSchedType);
6712 case OMPScheduleType::BaseRuntime:
6713 case OMPScheduleType::BaseAuto:
6714 case OMPScheduleType::BaseGreedy:
6715 case OMPScheduleType::BaseBalanced:
6716 case OMPScheduleType::BaseSteal:
6717 case OMPScheduleType::BaseRuntimeSimd:
6719 "schedule type does not support user-defined chunk sizes");
6721 case OMPScheduleType::BaseGuidedSimd:
6722 case OMPScheduleType::BaseDynamicChunked:
6723 case OMPScheduleType::BaseGuidedChunked:
6724 case OMPScheduleType::BaseGuidedIterativeChunked:
6725 case OMPScheduleType::BaseGuidedAnalyticalChunked:
6726 case OMPScheduleType::BaseStaticBalancedChunked:
6727 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6728 NeedsBarrier, ChunkSize);
6741 unsigned Bitwidth = Ty->getIntegerBitWidth();
6744 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_init_4u);
6747 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_init_8u);
6755static FunctionCallee
6757 unsigned Bitwidth = Ty->getIntegerBitWidth();
6760 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_next_4u);
6763 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_next_8u);
6770static FunctionCallee
6772 unsigned Bitwidth = Ty->getIntegerBitWidth();
6775 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_fini_4u);
6778 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_fini_8u);
6783OpenMPIRBuilder::applyDynamicWorkshareLoop(
DebugLoc DL, CanonicalLoopInfo *CLI,
6786 bool NeedsBarrier,
Value *Chunk) {
6787 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6789 "Require dedicated allocate IP");
6791 "Require valid schedule type");
6793 bool Ordered = (SchedType & OMPScheduleType::ModifierOrdered) ==
6794 OMPScheduleType::ModifierOrdered;
6799 uint32_t SrcLocStrSize;
6806 Type *IVTy =
IV->getType();
6811 Builder.SetInsertPoint(AllocaIP.getBlock()->getFirstNonPHIOrDbgOrAlloca());
6813 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6814 Value *PLowerBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.lowerbound");
6815 Value *PUpperBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.upperbound");
6816 Value *PStride =
Builder.CreateAlloca(IVTy,
nullptr,
"p.stride");
6825 Constant *One = ConstantInt::get(IVTy, 1);
6826 Builder.CreateStore(One, PLowerBound);
6828 Builder.CreateStore(UpperBound, PUpperBound);
6829 Builder.CreateStore(One, PStride);
6847 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6859 Builder.SetInsertPoint(OuterCond, OuterCond->getFirstInsertionPt());
6862 {SrcLoc, ThreadNum, PLastIter, PLowerBound, PUpperBound, PStride});
6863 Constant *Zero32 = ConstantInt::get(I32Type, 0);
6866 Builder.CreateSub(
Builder.CreateLoad(IVTy, PLowerBound), One,
"lb");
6867 Builder.CreateCondBr(MoreWork, Header, Exit);
6873 PI->setIncomingBlock(0, OuterCond);
6874 PI->setIncomingValue(0, LowerBound);
6879 Br->setSuccessor(OuterCond);
6885 UpperBound =
Builder.CreateLoad(IVTy, PUpperBound,
"ub");
6888 CI->setOperand(1, UpperBound);
6892 assert(BI->getSuccessor(1) == Exit);
6893 BI->setSuccessor(1, OuterCond);
6907 omp::Directive::OMPD_for,
false,
6910 return BarrierIP.takeError();
6962 assert(
Loops.size() >= 1 &&
"At least one loop required");
6963 size_t NumLoops =
Loops.size();
6967 return Loops.front();
6979 Loop->collectControlBlocks(OldControlBBs);
6983 if (ComputeIP.
isSet())
6990 Value *CollapsedTripCount =
nullptr;
6993 "All loops to collapse must be valid canonical loops");
6994 Value *OrigTripCount = L->getTripCount();
6995 if (!CollapsedTripCount) {
6996 CollapsedTripCount = OrigTripCount;
7001 CollapsedTripCount =
7002 Builder.CreateNUWMul(CollapsedTripCount, OrigTripCount);
7008 OrigPreheader->
getNextNode(), OrigAfter,
"collapsed");
7014 Builder.restoreIP(Result->getBodyIP());
7016 Value *Leftover = Result->getIndVar();
7018 NewIndVars.
resize(NumLoops);
7019 for (
int i = NumLoops - 1; i >= 1; --i) {
7020 Value *OrigTripCount =
Loops[i]->getTripCount();
7022 Value *NewIndVar =
Builder.CreateURem(Leftover, OrigTripCount);
7023 NewIndVars[i] = NewIndVar;
7025 Leftover =
Builder.CreateUDiv(Leftover, OrigTripCount);
7028 NewIndVars[0] = Leftover;
7037 BasicBlock *ContinueBlock = Result->getBody();
7039 auto ContinueWith = [&ContinueBlock, &ContinuePred,
DL](
BasicBlock *Dest,
7046 ContinueBlock =
nullptr;
7047 ContinuePred = NextSrc;
7054 for (
size_t i = 0; i < NumLoops - 1; ++i)
7055 ContinueWith(
Loops[i]->getBody(),
Loops[i + 1]->getHeader());
7061 for (
size_t i = NumLoops - 1; i > 0; --i)
7062 ContinueWith(
Loops[i]->getAfter(),
Loops[i - 1]->getLatch());
7065 ContinueWith(Result->getLatch(),
nullptr);
7072 for (
size_t i = 0; i < NumLoops; ++i)
7073 Loops[i]->getIndVar()->replaceAllUsesWith(NewIndVars[i]);
7087std::vector<CanonicalLoopInfo *>
7091 "Must pass as many tile sizes as there are loops");
7092 int NumLoops =
Loops.size();
7093 assert(NumLoops >= 1 &&
"At least one loop to tile required");
7105 Loop->collectControlBlocks(OldControlBBs);
7113 assert(L->isValid() &&
"All input loops must be valid canonical loops");
7114 OrigTripCounts.
push_back(L->getTripCount());
7125 for (
int i = 0; i < NumLoops - 1; ++i) {
7138 for (
int i = 0; i < NumLoops; ++i) {
7140 Value *OrigTripCount = OrigTripCounts[i];
7153 Value *FloorTripOverflow =
7154 Builder.CreateICmpNE(FloorTripRem, ConstantInt::get(IVType, 0));
7156 FloorTripOverflow =
Builder.CreateZExt(FloorTripOverflow, IVType);
7157 Value *FloorTripCount =
7158 Builder.CreateAdd(FloorCompleteTripCount, FloorTripOverflow,
7159 "omp_floor" +
Twine(i) +
".tripcount",
true);
7162 FloorCompleteCount.
push_back(FloorCompleteTripCount);
7168 std::vector<CanonicalLoopInfo *> Result;
7169 Result.reserve(NumLoops * 2);
7182 auto EmbeddNewLoop =
7183 [
this,
DL,
F, InnerEnter, &Enter, &
Continue, &OutroInsertBefore](
7186 DL, TripCount,
F, InnerEnter, OutroInsertBefore, Name);
7191 Enter = EmbeddedLoop->
getBody();
7193 OutroInsertBefore = EmbeddedLoop->
getLatch();
7194 return EmbeddedLoop;
7198 const Twine &NameBase) {
7201 EmbeddNewLoop(
P.value(), NameBase +
Twine(
P.index()));
7202 Result.push_back(EmbeddedLoop);
7206 EmbeddNewLoops(FloorCount,
"floor");
7212 for (
int i = 0; i < NumLoops; ++i) {
7216 Value *FloorIsEpilogue =
7218 Value *TileTripCount =
7225 EmbeddNewLoops(TileCounts,
"tile");
7230 for (std::pair<BasicBlock *, BasicBlock *>
P : InbetweenCode) {
7239 BodyEnter =
nullptr;
7240 BodyEntered = ExitBB;
7252 Builder.restoreIP(Result.back()->getBodyIP());
7253 for (
int i = 0; i < NumLoops; ++i) {
7256 Value *OrigIndVar = OrigIndVars[i];
7284 if (Properties.
empty())
7307 assert(
Loop->isValid() &&
"Expecting a valid CanonicalLoopInfo");
7311 assert(Latch &&
"A valid CanonicalLoopInfo must have a unique latch");
7319 if (
I.mayReadOrWriteMemory()) {
7323 I.setMetadata(LLVMContext::MD_access_group, AccessGroup);
7337 Loop->collectControlBlocks(oldControlBBs);
7342 assert(L->isValid() &&
"All input loops must be valid canonical loops");
7343 origTripCounts.
push_back(L->getTripCount());
7352 Builder.SetInsertPoint(TCBlock);
7353 Value *fusedTripCount =
nullptr;
7355 assert(L->isValid() &&
"All loops to fuse must be valid canonical loops");
7356 Value *origTripCount = L->getTripCount();
7357 if (!fusedTripCount) {
7358 fusedTripCount = origTripCount;
7361 Value *condTP =
Builder.CreateICmpSGT(fusedTripCount, origTripCount);
7362 fusedTripCount =
Builder.CreateSelect(condTP, fusedTripCount, origTripCount,
7376 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7377 Loops[i]->getPreheader()->moveBefore(TCBlock);
7378 Loops[i]->getAfter()->moveBefore(TCBlock);
7382 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7394 for (
size_t i = 0; i <
Loops.size(); ++i) {
7396 F->getContext(),
"omp.fused.inner.cond",
F,
Loops[i]->getBody());
7397 Builder.SetInsertPoint(condBlock);
7405 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7406 Builder.SetInsertPoint(condBBs[i]);
7407 Builder.CreateCondBr(condValues[i],
Loops[i]->getBody(), condBBs[i + 1]);
7423 "omp.fused.pre_latch");
7456 const Twine &NamePrefix) {
7485 C, NamePrefix +
".if.then",
Cond->getParent(),
Cond->getNextNode());
7487 C, NamePrefix +
".if.else",
Cond->getParent(), CanonicalLoop->
getExit());
7490 Builder.SetInsertPoint(SplitBeforeIt);
7492 Builder.CreateCondBr(IfCond, ThenBlock, ElseBlock);
7495 spliceBB(IP, ThenBlock,
false, Builder.getCurrentDebugLocation());
7498 Builder.SetInsertPoint(ElseBlock);
7504 ExistingBlocks.
reserve(L->getNumBlocks() + 1);
7506 ExistingBlocks.
append(L->block_begin(), L->block_end());
7512 assert(LoopCond && LoopHeader &&
"Invalid loop structure");
7514 if (
Block == L->getLoopPreheader() ||
Block == L->getLoopLatch() ||
7521 if (
Block == ThenBlock)
7522 NewBB->
setName(NamePrefix +
".if.else");
7525 VMap[
Block] = NewBB;
7533 L->getLoopLatch()->splitBasicBlockBefore(
L->getLoopLatch()->begin(),
7534 NamePrefix +
".pre_latch");
7538 L->addBasicBlockToLoop(ThenBlock, LI);
7544 if (TargetTriple.
isX86()) {
7545 if (Features.
lookup(
"avx512f"))
7547 else if (Features.
lookup(
"avx"))
7551 if (TargetTriple.
isPPC())
7553 if (TargetTriple.
isWasm())
7560 Value *IfCond, OrderKind Order,
7570 if (!BB.hasTerminator())
7586 I->eraseFromParent();
7589 if (AlignedVars.
size()) {
7591 for (
auto &AlignedItem : AlignedVars) {
7592 Value *AlignedPtr = AlignedItem.first;
7593 Value *Alignment = AlignedItem.second;
7596 Builder.CreateAlignmentAssumption(
F->getDataLayout(), AlignedPtr,
7604 createIfVersion(CanonicalLoop, IfCond, VMap, LIA, LI, L,
"simd");
7617 Reachable.insert(
Block);
7627 if ((Safelen ==
nullptr) || (Order == OrderKind::OMP_ORDER_concurrent))
7643 if (Simdlen || Safelen) {
7647 ConstantInt *VectorizeWidth = Simdlen ==
nullptr ? Safelen : Simdlen;
7673static std::unique_ptr<TargetMachine>
7677 StringRef CPU =
F->getFnAttribute(
"target-cpu").getValueAsString();
7678 StringRef Features =
F->getFnAttribute(
"target-features").getValueAsString();
7689 std::nullopt, OptLevel));
7707 if (!BB.hasTerminator())
7720 [&](
const Function &
F) {
return TM->getTargetTransformInfo(
F); });
7721 FAM.registerPass([&]() {
return TIRA; });
7735 I->eraseFromParent();
7738 assert(L &&
"Expecting CanonicalLoopInfo to be recognized as a loop");
7743 nullptr, ORE,
static_cast<int>(OptLevel),
7763 <<
" Threshold=" << UP.
Threshold <<
"\n"
7766 <<
" PartialOptSizeThreshold="
7786 Ptr =
Load->getPointerOperand();
7788 Ptr =
Store->getPointerOperand();
7795 if (Alloca->getParent() == &
F->getEntryBlock())
7815 int MaxTripCount = 0;
7816 bool MaxOrZero =
false;
7817 unsigned TripMultiple = 0;
7821 MaxTripCount, MaxOrZero, TripMultiple, UCE, UP, PP);
7822 LLVM_DEBUG(
dbgs() <<
"Suggesting unroll factor of " << Factor <<
"\n");
7833 assert(Factor >= 0 &&
"Unroll factor must not be negative");
7849 Ctx, {
MDString::get(Ctx,
"llvm.loop.unroll.count"), FactorConst}));
7862 *UnrolledCLI =
Loop;
7867 "unrolling only makes sense with a factor of 2 or larger");
7869 Type *IndVarTy =
Loop->getIndVarType();
7876 std::vector<CanonicalLoopInfo *>
LoopNest =
7891 Ctx, {
MDString::get(Ctx,
"llvm.loop.unroll.count"), FactorConst})});
7894 (*UnrolledCLI)->assertOK();
7912 Value *Args[] = {Ident, ThreadId, BufSize, CpyBuf, CpyFn, DidItLD};
7931 if (!CPVars.
empty()) {
7936 Directive OMPD = Directive::OMPD_single;
7941 Value *Args[] = {Ident, ThreadId};
7950 if (
Error Err = FiniCB(IP))
7971 EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCBWrapper,
7978 for (
size_t I = 0, E = CPVars.
size();
I < E; ++
I)
7981 ConstantInt::get(Int64, 0), CPVars[
I],
7984 }
else if (!IsNowait) {
7987 omp::Directive::OMPD_unknown,
false,
8005 Directive::OMPD_scope,
nullptr,
nullptr,
8006 BodyGenCB, FiniCB,
false,
true,
8014 omp::Directive::OMPD_unknown,
8030 Directive OMPD = Directive::OMPD_critical;
8035 Value *LockVar = getOMPCriticalRegionLock(CriticalName);
8036 Value *Args[] = {Ident, ThreadId, LockVar};
8053 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
8061 const Twine &Name,
bool IsDependSource) {
8065 "OpenMP runtime requires depend vec with i64 type");
8078 for (
unsigned I = 0;
I < NumLoops; ++
I) {
8092 Value *Args[] = {Ident, ThreadId, DependBaseAddrGEP};
8110 Directive OMPD = Directive::OMPD_ordered;
8119 Value *Args[] = {Ident, ThreadId};
8129 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
8136 bool HasFinalize,
bool IsCancellable) {
8143 BasicBlock *EntryBB = Builder.GetInsertBlock();
8152 emitCommonDirectiveEntry(OMPD, EntryCall, ExitBB, Conditional);
8164 "Unexpected control flow graph state!!");
8166 emitCommonDirectiveExit(OMPD, FinIP, ExitCall, HasFinalize);
8168 return AfterIP.takeError();
8173 "Unexpected Insertion point location!");
8176 auto InsertBB = merged ? ExitPredBB : ExitBB;
8179 Builder.SetInsertPoint(InsertBB);
8181 return Builder.saveIP();
8185 Directive OMPD,
Value *EntryCall, BasicBlock *ExitBB,
bool Conditional) {
8187 if (!Conditional || !EntryCall)
8193 auto *UI =
new UnreachableInst(
Builder.getContext(), ThenBB);
8203 Builder.CreateCondBr(CallBool, ThenBB, ExitBB);
8207 UI->eraseFromParent();
8215 omp::Directive OMPD,
InsertPointTy FinIP, Instruction *ExitCall,
8223 "Unexpected finalization stack state!");
8226 assert(Fi.DK == OMPD &&
"Unexpected Directive for Finalization call!");
8228 if (
Error Err = Fi.mergeFiniBB(
Builder, FinIP.getBlock()))
8229 return std::move(Err);
8233 Builder.SetInsertPoint(FinIP.getBlock()->getTerminator());
8243 return IRBuilder<>::InsertPoint(ExitCall->
getParent(),
8277 "copyin.not.master.end");
8284 Builder.SetInsertPoint(OMP_Entry);
8287 Value *cmp =
Builder.CreateICmpNE(MasterPtr, PrivatePtr);
8288 Builder.CreateCondBr(cmp, CopyBegin, CopyEnd);
8290 Builder.SetInsertPoint(CopyBegin);
8308 Value *Args[] = {ThreadId,
Size, Allocator};
8331 return Builder.CreateCall(Fn, Args, Name);
8345 Value *Args[] = {ThreadId, Addr, Allocator};
8352 const Twine &Name) {
8360 M.getContext(),
M.getDataLayout().getPrefTypeAlign(Int64)));
8366 const Twine &Name) {
8368 Loc,
Builder.getInt64(
M.getDataLayout().getTypeAllocSize(VarType)), Name);
8373 const Twine &Name) {
8379 return Builder.CreateCall(Fn, Args, Name);
8384 const Twine &Name) {
8386 Loc, Addr,
Builder.getInt64(
M.getDataLayout().getTypeAllocSize(VarType)),
8393 Value *DependenceAddress,
bool HaveNowaitClause) {
8401 if (Device ==
nullptr)
8403 else if (Device->getType() != Int32)
8404 Device =
Builder.CreateIntCast(Device, Int32,
true);
8405 Constant *InteropTypeVal = ConstantInt::get(Int32, (
int)InteropType);
8406 if (NumDependences ==
nullptr) {
8407 NumDependences = ConstantInt::get(Int32, 0);
8411 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8413 Ident, ThreadId, InteropVar, InteropTypeVal,
8414 Device, NumDependences, DependenceAddress, HaveNowaitClauseVal};
8423 Value *NumDependences,
Value *DependenceAddress,
bool HaveNowaitClause) {
8431 if (Device ==
nullptr)
8433 else if (Device->getType() != Int32)
8434 Device =
Builder.CreateIntCast(Device, Int32,
true);
8435 if (NumDependences ==
nullptr) {
8436 NumDependences = ConstantInt::get(Int32, 0);
8440 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8442 Ident, ThreadId, InteropVar, Device,
8443 NumDependences, DependenceAddress, HaveNowaitClauseVal};
8452 Value *NumDependences,
8453 Value *DependenceAddress,
8454 bool HaveNowaitClause) {
8461 if (Device ==
nullptr)
8463 else if (Device->getType() != Int32)
8464 Device =
Builder.CreateIntCast(Device, Int32,
true);
8465 if (NumDependences ==
nullptr) {
8466 NumDependences = ConstantInt::get(Int32, 0);
8470 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8472 Ident, ThreadId, InteropVar, Device,
8473 NumDependences, DependenceAddress, HaveNowaitClauseVal};
8503 assert(!Attrs.MaxThreads.empty() && !Attrs.MaxTeams.empty() &&
8504 "expected num_threads and num_teams to be specified");
8524 const std::string DebugPrefix =
"_debug__";
8525 if (KernelName.
ends_with(DebugPrefix)) {
8526 KernelName = KernelName.
drop_back(DebugPrefix.length());
8527 Kernel =
M.getFunction(KernelName);
8533 if (Attrs.MinTeams > 1 || Attrs.MaxTeams.front() > 0)
8538 int32_t MaxThreadsVal = Attrs.MaxThreads.front();
8545 MaxThreadsVal = Attrs.MinThreads;
8549 if (MaxThreadsVal > 0)
8560 omp::RuntimeFunction::OMPRTL___kmpc_target_init);
8563 Twine DynamicEnvironmentName = KernelName +
"_dynamic_environment";
8564 Constant *DynamicEnvironmentInitializer =
8568 DynamicEnvironmentInitializer, DynamicEnvironmentName,
8570 DL.getDefaultGlobalsAddressSpace());
8574 DynamicEnvironmentGV->
getType() == DynamicEnvironmentPtr
8575 ? DynamicEnvironmentGV
8577 DynamicEnvironmentPtr);
8580 ConfigurationEnvironment, {
8581 UseGenericStateMachineVal,
8582 MayUseNestedParallelismVal,
8591 KernelEnvironment, {
8592 ConfigurationEnvironmentInitializer,
8596 std::string KernelEnvironmentName =
8597 (KernelName +
"_kernel_environment").str();
8600 KernelEnvironmentInitializer, KernelEnvironmentName,
8602 DL.getDefaultGlobalsAddressSpace());
8606 KernelEnvironmentGV->
getType() == KernelEnvironmentPtr
8607 ? KernelEnvironmentGV
8609 KernelEnvironmentPtr);
8610 Value *KernelLaunchEnvironment =
8613 KernelLaunchEnvironment =
8614 KernelLaunchEnvironment->
getType() == KernelLaunchEnvParamTy
8615 ? KernelLaunchEnvironment
8616 :
Builder.CreateAddrSpaceCast(KernelLaunchEnvironment,
8617 KernelLaunchEnvParamTy);
8619 Fn, {KernelEnvironment, KernelLaunchEnvironment});
8631 auto *UI =
Builder.CreateUnreachable();
8637 Builder.SetInsertPoint(WorkerExitBB);
8641 Builder.SetInsertPoint(CheckBBTI);
8642 Builder.CreateCondBr(ExecUserCode, UI->getParent(), WorkerExitBB);
8644 CheckBBTI->eraseFromParent();
8645 UI->eraseFromParent();
8653 int32_t TeamsReductionDataSize) {
8658 omp::RuntimeFunction::OMPRTL___kmpc_target_deinit);
8662 if (!TeamsReductionDataSize)
8668 const std::string DebugPrefix =
"_debug__";
8670 KernelName = KernelName.
drop_back(DebugPrefix.length());
8671 auto *KernelEnvironmentGV =
8672 M.getNamedGlobal((KernelName +
"_kernel_environment").str());
8673 assert(KernelEnvironmentGV &&
"Expected kernel environment global\n");
8674 auto *KernelEnvironmentInitializer = KernelEnvironmentGV->getInitializer();
8676 KernelEnvironmentInitializer,
8677 ConstantInt::get(Int32, TeamsReductionDataSize), {0, 7});
8678 KernelEnvironmentGV->setInitializer(NewInitializer);
8683 if (
Kernel.hasFnAttribute(Name)) {
8684 int32_t OldLimit =
Kernel.getFnAttributeAsParsedInteger(Name);
8690std::pair<int32_t, int32_t>
8692 int32_t ThreadLimit =
8693 Kernel.getFnAttributeAsParsedInteger(
"omp_target_thread_limit");
8696 const auto &Attr =
Kernel.getFnAttribute(
"amdgpu-flat-work-group-size");
8697 if (!Attr.isValid() || !Attr.isStringAttribute())
8698 return {0, ThreadLimit};
8699 auto [LBStr, UBStr] = Attr.getValueAsString().split(
',');
8702 return {0, ThreadLimit};
8703 UB = ThreadLimit ? std::min(ThreadLimit, UB) : UB;
8711 return {0, ThreadLimit ? std::min(ThreadLimit, UB) : UB};
8713 return {0, ThreadLimit};
8719 Kernel.addFnAttr(
"omp_target_thread_limit", std::to_string(UB));
8722 Kernel.addFnAttr(
"amdgpu-flat-work-group-size",
8730std::pair<int32_t, int32_t>
8733 return {0,
Kernel.getFnAttributeAsParsedInteger(
"omp_target_num_teams")};
8737 int32_t LB, int32_t UB) {
8745 Kernel.addFnAttr(
"omp_target_num_teams", std::to_string(LB));
8748void OpenMPIRBuilder::setOutlinedTargetRegionFunctionAttributes(
8757 else if (
T.isNVPTX())
8759 else if (
T.isSPIRV())
8764Constant *OpenMPIRBuilder::createOutlinedFunctionID(Function *OutlinedFn,
8765 StringRef EntryFnIDName) {
8766 if (
Config.isTargetDevice()) {
8767 assert(OutlinedFn &&
"The outlined function must exist if embedded");
8771 return new GlobalVariable(
8776Constant *OpenMPIRBuilder::createTargetRegionEntryAddr(Function *OutlinedFn,
8777 StringRef EntryFnName) {
8781 assert(!
M.getGlobalVariable(EntryFnName,
true) &&
8782 "Named kernel already exists?");
8783 return new GlobalVariable(
8796 if (
Config.isTargetDevice() || !
Config.openMPOffloadMandatory()) {
8800 OutlinedFn = *CBResult;
8802 OutlinedFn =
nullptr;
8808 if (!IsOffloadEntry)
8811 std::string EntryFnIDName =
8813 ? std::string(EntryFnName)
8817 EntryFnName, EntryFnIDName);
8825 setOutlinedTargetRegionFunctionAttributes(OutlinedFn);
8826 auto OutlinedFnID = createOutlinedFunctionID(OutlinedFn, EntryFnIDName);
8827 auto EntryAddr = createTargetRegionEntryAddr(OutlinedFn, EntryFnName);
8829 EntryInfo, EntryAddr, OutlinedFnID,
8831 return OutlinedFnID;
8849 bool IsStandAlone = !BodyGenCB;
8856 MapInfo = &GenMapInfoCB(
Builder.saveIP());
8858 AllocaIP,
Builder.saveIP(), *MapInfo, Info, CustomMapperCB,
8859 true, DeviceAddrCB))
8866 Value *PointerNum =
Builder.getInt32(Info.NumberOfPtrs);
8876 SrcLocInfo, DeviceID,
8883 assert(MapperFunc &&
"MapperFunc missing for standalone target data");
8887 if (Info.HasNoWait) {
8897 if (Info.HasNoWait) {
8901 emitBlock(OffloadContBlock, CurFn,
true);
8907 bool RequiresOuterTargetTask = Info.HasNoWait;
8908 if (!RequiresOuterTargetTask)
8909 cantFail(TaskBodyCB(
nullptr,
nullptr,
8913 {}, RTArgs, Info.HasNoWait));
8916 omp::OMPRTL___tgt_target_data_begin_mapper);
8920 for (
auto DeviceMap : Info.DevicePtrInfoMap) {
8924 Builder.CreateStore(LI, DeviceMap.second.second);
8961 Value *PointerNum =
Builder.getInt32(Info.NumberOfPtrs);
8970 Value *OffloadingArgs[] = {SrcLocInfo, DeviceID,
8993 return emitIfClause(IfCond, BeginThenGen, BeginElseGen, AllocaIP);
8994 return BeginThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9009 return emitIfClause(IfCond, EndThenGen, EndElseGen, AllocaIP);
9010 return EndThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9013 return emitIfClause(IfCond, BeginThenGen, EndElseGen, AllocaIP);
9014 return BeginThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9025 bool IsGPUDistribute) {
9026 assert((IVSize == 32 || IVSize == 64) &&
9027 "IV size is not compatible with the omp runtime");
9029 if (IsGPUDistribute)
9031 ? (IVSigned ? omp::OMPRTL___kmpc_distribute_static_init_4
9032 : omp::OMPRTL___kmpc_distribute_static_init_4u)
9033 : (IVSigned ? omp::OMPRTL___kmpc_distribute_static_init_8
9034 : omp::OMPRTL___kmpc_distribute_static_init_8u);
9036 Name = IVSize == 32 ? (IVSigned ? omp::OMPRTL___kmpc_for_static_init_4
9037 : omp::OMPRTL___kmpc_for_static_init_4u)
9038 : (IVSigned ? omp::OMPRTL___kmpc_for_static_init_8
9039 : omp::OMPRTL___kmpc_for_static_init_8u);
9046 assert((IVSize == 32 || IVSize == 64) &&
9047 "IV size is not compatible with the omp runtime");
9049 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_init_4
9050 : omp::OMPRTL___kmpc_dispatch_init_4u)
9051 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_init_8
9052 : omp::OMPRTL___kmpc_dispatch_init_8u);
9059 assert((IVSize == 32 || IVSize == 64) &&
9060 "IV size is not compatible with the omp runtime");
9062 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_next_4
9063 : omp::OMPRTL___kmpc_dispatch_next_4u)
9064 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_next_8
9065 : omp::OMPRTL___kmpc_dispatch_next_8u);
9072 assert((IVSize == 32 || IVSize == 64) &&
9073 "IV size is not compatible with the omp runtime");
9075 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_fini_4
9076 : omp::OMPRTL___kmpc_dispatch_fini_4u)
9077 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_fini_8
9078 : omp::OMPRTL___kmpc_dispatch_fini_8u);
9089 DenseMap<
Value *, std::tuple<Value *, unsigned>> &ValueReplacementMap) {
9097 auto GetUpdatedDIVariable = [&](
DILocalVariable *OldVar,
unsigned arg) {
9101 if (NewVar && (arg == NewVar->
getArg()))
9111 auto UpdateDebugRecord = [&](
auto *DR) {
9114 for (
auto Loc : DR->location_ops()) {
9115 auto Iter = ValueReplacementMap.find(
Loc);
9116 if (Iter != ValueReplacementMap.end()) {
9117 DR->replaceVariableLocationOp(
Loc, std::get<0>(Iter->second));
9118 ArgNo = std::get<1>(Iter->second) + 1;
9122 DR->setVariable(GetUpdatedDIVariable(OldVar, ArgNo));
9127 if (DVR->getNumVariableLocationOps() != 1u) {
9128 DVR->setKillLocation();
9131 Value *
Loc = DVR->getVariableLocationOp(0u);
9138 RequiredBB = &DVR->getFunction()->getEntryBlock();
9140 if (RequiredBB && RequiredBB != CurBB) {
9152 "Unexpected debug intrinsic");
9154 UpdateDebugRecord(&DVR);
9155 MoveDebugRecordToCorrectBlock(&DVR);
9158 for (
auto *DVR : DVRsToDelete)
9159 DVR->getMarker()->MarkedInstr->dropOneDbgRecord(DVR);
9163 Module *M = Func->getParent();
9166 DB.createQualifiedType(dwarf::DW_TAG_pointer_type,
nullptr);
9167 unsigned ArgNo = Func->arg_size();
9169 NewSP,
"dyn_ptr", ArgNo, NewSP->
getFile(), 0, VoidPtrTy,
9170 false, DINode::DIFlags::FlagArtificial);
9172 Argument *LastArg = Func->getArg(Func->arg_size() - 1);
9173 DB.insertDeclare(LastArg, Var, DB.createExpression(),
Loc,
9194 for (
auto &Arg : Inputs)
9195 ParameterTypes.
push_back(Arg->getType()->isPointerTy()
9199 for (
auto &Arg : Inputs)
9200 ParameterTypes.
push_back(Arg->getType());
9208 auto BB = Builder.GetInsertBlock();
9209 auto M = BB->getModule();
9220 if (TargetCpuAttr.isStringAttribute())
9221 Func->addFnAttr(TargetCpuAttr);
9223 auto TargetFeaturesAttr = ParentFn->
getFnAttribute(
"target-features");
9224 if (TargetFeaturesAttr.isStringAttribute())
9225 Func->addFnAttr(TargetFeaturesAttr);
9230 OMPBuilder.
emitUsed(
"llvm.compiler.used", {ExecMode});
9241 Builder.SetInsertPoint(EntryBB);
9247 BasicBlock *UserCodeEntryBB = Builder.GetInsertBlock();
9257 splitBB(Builder,
true,
"outlined.body");
9264 Builder.SetInsertPoint(ExitBB);
9271 Builder.CreateRetVoid();
9275 auto AllocaIP = Builder.saveIP();
9280 const auto &ArgRange =
make_range(Func->arg_begin(), Func->arg_end() - 1);
9312 if (Instr->getFunction() == Func)
9313 Instr->replaceUsesOfWith(
Input, InputCopy);
9319 for (
auto InArg :
zip(Inputs, ArgRange)) {
9321 Argument &Arg = std::get<1>(InArg);
9322 Value *InputCopy =
nullptr;
9325 Arg,
Input, InputCopy, AllocaIP, Builder.saveIP(),
9329 Builder.restoreIP(*AfterIP);
9330 ValueReplacementMap[
Input] = std::make_tuple(InputCopy, Arg.
getArgNo());
9350 DeferredReplacement.push_back(std::make_pair(
Input, InputCopy));
9357 ReplaceValue(
Input, InputCopy, Func);
9361 for (
auto Deferred : DeferredReplacement)
9362 ReplaceValue(std::get<0>(Deferred), std::get<1>(Deferred), Func);
9365 ValueReplacementMap);
9373 Value *TaskWithPrivates,
9374 Type *TaskWithPrivatesTy) {
9376 Type *TaskTy = OMPIRBuilder.Task;
9379 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskWithPrivates, 0);
9380 Value *Shareds = TaskT;
9390 if (TaskWithPrivatesTy != TaskTy)
9391 Shareds = Builder.CreateStructGEP(TaskTy, TaskT, 0);
9408 const size_t NumOffloadingArrays,
const int SharedArgsOperandNo) {
9413 assert((!NumOffloadingArrays || PrivatesTy) &&
9414 "PrivatesTy cannot be nullptr when there are offloadingArrays"
9447 Type *TaskPtrTy = OMPBuilder.TaskPtr;
9448 [[maybe_unused]]
Type *TaskTy = OMPBuilder.Task;
9454 ".omp_target_task_proxy_func",
9455 Builder.GetInsertBlock()->getModule());
9456 Value *ThreadId = ProxyFn->getArg(0);
9457 Value *TaskWithPrivates = ProxyFn->getArg(1);
9458 ThreadId->
setName(
"thread.id");
9459 TaskWithPrivates->
setName(
"task");
9461 bool HasShareds = SharedArgsOperandNo > 0;
9462 bool HasOffloadingArrays = NumOffloadingArrays > 0;
9465 Builder.SetInsertPoint(EntryBB);
9471 if (HasOffloadingArrays) {
9472 assert(TaskTy != TaskWithPrivatesTy &&
9473 "If there are offloading arrays to pass to the target"
9474 "TaskTy cannot be the same as TaskWithPrivatesTy");
9477 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskWithPrivates, 1);
9478 for (
unsigned int i = 0; i < NumOffloadingArrays; ++i)
9480 Builder.CreateStructGEP(PrivatesTy, Privates, i));
9484 auto *ArgStructAlloca =
9486 assert(ArgStructAlloca &&
9487 "Unable to find the alloca instruction corresponding to arguments "
9488 "for extracted function");
9490 std::optional<TypeSize> ArgAllocSize =
9492 assert(ArgStructType && ArgAllocSize &&
9493 "Unable to determine size of arguments for extracted function");
9494 uint64_t StructSize = ArgAllocSize->getFixedValue();
9497 Builder.CreateAlloca(ArgStructType,
nullptr,
"structArg");
9499 Value *SharedsSize = Builder.getInt64(StructSize);
9502 OMPBuilder, Builder, TaskWithPrivates, TaskWithPrivatesTy);
9504 Builder.CreateMemCpy(
9505 NewArgStructAlloca, NewArgStructAlloca->
getAlign(), LoadShared,
9507 KernelLaunchArgs.
push_back(NewArgStructAlloca);
9510 Builder.CreateRetVoid();
9516 return GEP->getSourceElementType();
9518 return Alloca->getAllocatedType();
9541 if (OffloadingArraysToPrivatize.
empty())
9542 return OMPIRBuilder.Task;
9545 for (
Value *V : OffloadingArraysToPrivatize) {
9546 assert(V->getType()->isPointerTy() &&
9547 "Expected pointer to array to privatize. Got a non-pointer value "
9550 assert(ArrayTy &&
"ArrayType cannot be nullptr");
9556 "struct.task_with_privates");
9570 EntryFnName, Inputs, CBFunc,
9575 EntryInfo, GenerateOutlinedFunction, IsOffloadEntry, OutlinedFn,
9712 TargetTaskAllocaBB->
begin());
9715 auto OI = std::make_unique<OutlineInfo>();
9716 OI->EntryBB = TargetTaskAllocaBB;
9717 OI->OuterAllocBB = AllocaIP.
getBlock();
9722 Builder, AllocaIP, ToBeDeleted, TargetTaskAllocaIP,
"global.tid",
false));
9725 Builder.restoreIP(TargetTaskBodyIP);
9726 if (
Error Err = TaskBodyCB(DeviceID, RTLoc, TargetTaskAllocaIP))
9744 bool NeedsTargetTask = HasNoWait && DeviceID;
9745 if (NeedsTargetTask) {
9751 OffloadingArraysToPrivatize.
push_back(V);
9752 OI->ExcludeArgsFromAggregate.push_back(V);
9756 OI->PostOutlineCB = [
this, ToBeDeleted, Dependencies, NeedsTargetTask,
9757 DeviceID, OffloadingArraysToPrivatize](
9760 "there must be a single user for the outlined function");
9774 const unsigned int NumStaleCIArgs = StaleCI->
arg_size();
9775 bool HasShareds = NumStaleCIArgs > OffloadingArraysToPrivatize.
size() + 1;
9777 NumStaleCIArgs == (OffloadingArraysToPrivatize.
size() + 2)) &&
9778 "Wrong number of arguments for StaleCI when shareds are present");
9779 int SharedArgOperandNo =
9780 HasShareds ? OffloadingArraysToPrivatize.
size() + 1 : 0;
9786 if (!OffloadingArraysToPrivatize.
empty())
9791 *
this,
Builder, StaleCI, PrivatesTy, TaskWithPrivatesTy,
9792 OffloadingArraysToPrivatize.
size(), SharedArgOperandNo);
9794 LLVM_DEBUG(
dbgs() <<
"Proxy task entry function created: " << *ProxyFn
9797 Builder.SetInsertPoint(StaleCI);
9814 OMPRTL___kmpc_omp_target_task_alloc);
9826 M.getDataLayout().getTypeStoreSize(TaskWithPrivatesTy));
9833 auto *ArgStructAlloca =
9835 assert(ArgStructAlloca &&
9836 "Unable to find the alloca instruction corresponding to arguments "
9837 "for extracted function");
9838 std::optional<TypeSize> ArgAllocSize =
9841 "Unable to determine size of arguments for extracted function");
9842 SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
9861 TaskSize, SharedsSize,
9864 if (NeedsTargetTask) {
9865 assert(DeviceID &&
"Expected non-empty device ID.");
9875 *
this,
Builder, TaskData, TaskWithPrivatesTy);
9876 Builder.CreateMemCpy(TaskShareds, Alignment, Shareds, Alignment,
9879 if (!OffloadingArraysToPrivatize.
empty()) {
9881 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskData, 1);
9882 for (
unsigned int i = 0; i < OffloadingArraysToPrivatize.
size(); ++i) {
9883 Value *PtrToPrivatize = OffloadingArraysToPrivatize[i];
9890 "ElementType should match ArrayType");
9893 Value *Dst =
Builder.CreateStructGEP(PrivatesTy, Privates, i);
9895 Dst, Alignment, PtrToPrivatize, Alignment,
9896 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(ElementType)));
9900 Value *DepArray =
nullptr;
9901 Value *NumDeps =
nullptr;
9904 NumDeps = Dependencies.
NumDeps;
9905 }
else if (!Dependencies.
Deps.empty()) {
9907 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
9918 if (!NeedsTargetTask) {
9927 ConstantInt::get(
Builder.getInt32Ty(), 0),
9940 }
else if (DepArray) {
9948 {Ident, ThreadID, TaskData, NumDeps, DepArray,
9949 ConstantInt::get(
Builder.getInt32Ty(), 0),
9959 I->eraseFromParent();
9964 << *(
Builder.GetInsertBlock()) <<
"\n");
9966 << *(
Builder.GetInsertBlock()->getParent()->getParent())
9978 CustomMapperCB, IsNonContiguous, DeviceAddrCB))
10001 Builder.restoreIP(IP);
10007 return Builder.saveIP();
10010 bool HasDependencies = !Dependencies.
empty();
10011 bool RequiresOuterTargetTask = HasNoWait || HasDependencies;
10028 if (OutlinedFnID && DeviceID)
10030 EmitTargetCallFallbackCB, KArgs,
10031 DeviceID, RTLoc, TargetTaskAllocaIP);
10039 return EmitTargetCallFallbackCB(OMPBuilder.
Builder.
saveIP());
10046 auto &&EmitTargetCallElse =
10053 if (RequiresOuterTargetTask) {
10060 Dependencies, EmptyRTArgs, HasNoWait);
10062 return EmitTargetCallFallbackCB(Builder.saveIP());
10065 Builder.restoreIP(AfterIP);
10069 auto &&EmitTargetCallThen =
10073 Info.HasNoWait = HasNoWait;
10078 AllocaIP, Builder.saveIP(), Info, RTArgs, MapInfo, CustomMapperCB,
10084 for (
auto [DefaultVal, RuntimeVal] :
10086 NumTeamsC.
push_back(RuntimeVal ? RuntimeVal
10087 : Builder.getInt32(DefaultVal));
10091 auto InitMaxThreadsClause = [&Builder](
Value *
Clause) {
10093 Clause = Builder.CreateIntCast(
Clause, Builder.getInt32Ty(),
10097 auto CombineMaxThreadsClauses = [&Builder](
Value *
Clause,
Value *&Result) {
10100 Result ? Builder.CreateSelect(Builder.CreateICmpULT(Result,
Clause),
10108 Value *MaxThreadsClause =
10110 ? InitMaxThreadsClause(RuntimeAttrs.
MaxThreads)
10113 for (
auto [TeamsVal, TargetVal] :
zip_equal(
10115 Value *TeamsThreadLimitClause = InitMaxThreadsClause(TeamsVal);
10116 Value *NumThreads = InitMaxThreadsClause(TargetVal);
10118 CombineMaxThreadsClauses(TeamsThreadLimitClause, NumThreads);
10119 CombineMaxThreadsClauses(MaxThreadsClause, NumThreads);
10121 NumThreadsC.
push_back(NumThreads ? NumThreads : Builder.getInt32(0));
10124 unsigned NumTargetItems = Info.NumberOfPtrs;
10132 Builder.getInt64Ty(),
10134 : Builder.getInt64(0);
10138 DynCGroupMem = Builder.getInt32(0);
10141 NumTargetItems, RTArgs, TripCount, NumTeamsC, NumThreadsC, DynCGroupMem,
10142 HasNoWait,
false, DynCGroupMemFallback);
10149 if (RequiresOuterTargetTask)
10151 RTLoc, AllocaIP, Dependencies,
10152 KArgs.
RTArgs, Info.HasNoWait);
10155 Builder, OutlinedFnID, EmitTargetCallFallbackCB, KArgs,
10156 RuntimeAttrs.
DeviceID, RTLoc, AllocaIP);
10159 Builder.restoreIP(AfterIP);
10166 if (!OutlinedFnID) {
10167 cantFail(EmitTargetCallElse(AllocaIP, Builder.saveIP(), DeallocBlocks));
10173 cantFail(EmitTargetCallThen(AllocaIP, Builder.saveIP(), DeallocBlocks));
10178 EmitTargetCallElse, AllocaIP));
10191 bool HasNowait,
Value *DynCGroupMem,
10197 Builder.restoreIP(CodeGenIP);
10205 *
this,
Builder, IsOffloadEntry, EntryInfo, DefaultAttrs, OutlinedFn,
10206 OutlinedFnID, Inputs, CBFunc, ArgAccessorFuncCB))
10212 if (!
Config.isTargetDevice())
10214 RuntimeAttrs, IfCond, OutlinedFn, OutlinedFnID, Inputs,
10215 GenMapInfoCB, CustomMapperCB, Dependencies, HasNowait,
10216 DynCGroupMem, DynCGroupMemFallback);
10230 return OS.
str().str();
10235 return OpenMPIRBuilder::getNameWithSeparators(Parts,
Config.firstSeparator(),
10241 auto &Elem = *
InternalVars.try_emplace(Name,
nullptr).first;
10243 assert(Elem.second->getValueType() == Ty &&
10244 "OMP internal variable has different type than requested");
10257 :
M.getTargetTriple().isAMDGPU()
10259 :
DL.getDefaultGlobalsAddressSpace();
10268 const llvm::Align PtrAlign =
DL.getPointerABIAlignment(AddressSpaceVal);
10269 GV->setAlignment(std::max(TypeAlign, PtrAlign));
10273 return Elem.second;
10276Value *OpenMPIRBuilder::getOMPCriticalRegionLock(
StringRef CriticalName) {
10277 std::string Prefix =
Twine(
"gomp_critical_user_", CriticalName).
str();
10278 std::string Name = getNameWithSeparators({Prefix,
"var"},
".",
".");
10289 return SizePtrToInt;
10294 std::string VarName) {
10302 return MaptypesArrayGlobal;
10307 unsigned NumOperands,
10316 ArrI8PtrTy,
nullptr,
".offload_baseptrs");
10320 ArrI64Ty,
nullptr,
".offload_sizes");
10331 int64_t DeviceID,
unsigned NumOperands) {
10337 Value *ArgsBaseGEP =
10339 {Builder.getInt32(0), Builder.getInt32(0)});
10342 {Builder.getInt32(0), Builder.getInt32(0)});
10343 Value *ArgSizesGEP =
10345 {Builder.getInt32(0), Builder.getInt32(0)});
10349 Builder.getInt32(NumOperands),
10350 ArgsBaseGEP, ArgsGEP, ArgSizesGEP,
10351 MaptypesArg, MapnamesArg, NullPtr});
10358 assert((!ForEndCall || Info.separateBeginEndCalls()) &&
10359 "expected region end call to runtime only when end call is separate");
10361 auto VoidPtrTy = UnqualPtrTy;
10362 auto VoidPtrPtrTy = UnqualPtrTy;
10364 auto Int64PtrTy = UnqualPtrTy;
10366 if (!Info.NumberOfPtrs) {
10378 Info.RTArgs.BasePointersArray,
10381 ArrayType::get(VoidPtrTy, Info.NumberOfPtrs), Info.RTArgs.PointersArray,
10385 ArrayType::get(Int64Ty, Info.NumberOfPtrs), Info.RTArgs.SizesArray,
10389 ForEndCall && Info.RTArgs.MapTypesArrayEnd ? Info.RTArgs.MapTypesArrayEnd
10390 : Info.RTArgs.MapTypesArray,
10396 if (!Info.EmitDebug)
10400 ArrayType::get(VoidPtrTy, Info.NumberOfPtrs), Info.RTArgs.MapNamesArray,
10405 if (!Info.HasMapper)
10409 Builder.CreatePointerCast(Info.RTArgs.MappersArray, VoidPtrPtrTy);
10430 "struct.descriptor_dim");
10432 enum { OffsetFD = 0, CountFD, StrideFD };
10436 for (
unsigned I = 0, L = 0, E = NonContigInfo.
Dims.
size();
I < E; ++
I) {
10439 if (NonContigInfo.
Dims[
I] == 1)
10444 Builder.CreateAlloca(ArrayTy,
nullptr,
"dims");
10445 Builder.restoreIP(CodeGenIP);
10446 for (
unsigned II = 0, EE = NonContigInfo.
Dims[
I];
II < EE; ++
II) {
10447 unsigned RevIdx = EE -
II - 1;
10451 Value *OffsetLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, OffsetFD);
10453 NonContigInfo.
Offsets[L][RevIdx], OffsetLVal,
10454 M.getDataLayout().getPrefTypeAlign(OffsetLVal->
getType()));
10456 Value *CountLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, CountFD);
10458 NonContigInfo.
Counts[L][RevIdx], CountLVal,
10459 M.getDataLayout().getPrefTypeAlign(CountLVal->
getType()));
10461 Value *StrideLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, StrideFD);
10463 NonContigInfo.
Strides[L][RevIdx], StrideLVal,
10464 M.getDataLayout().getPrefTypeAlign(CountLVal->
getType()));
10467 Builder.restoreIP(CodeGenIP);
10468 Value *DAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
10469 DimsAddr,
Builder.getPtrTy());
10472 Info.RTArgs.PointersArray, 0,
I);
10474 DAddr,
P,
M.getDataLayout().getPrefTypeAlign(
Builder.getPtrTy()));
10479void OpenMPIRBuilder::emitUDMapperArrayInitOrDel(
10483 StringRef Prefix = IsInit ?
".init" :
".del";
10489 Builder.CreateICmpSGT(
Size, Builder.getInt64(1),
"omp.arrayinit.isarray");
10490 Value *DeleteBit = Builder.CreateAnd(
10493 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10494 OpenMPOffloadMappingFlags::OMP_MAP_DELETE)));
10499 Value *BaseIsBegin = Builder.CreateICmpNE(
Base, Begin);
10500 Cond = Builder.CreateOr(IsArray, BaseIsBegin);
10501 DeleteCond = Builder.CreateIsNull(
10506 DeleteCond =
Builder.CreateIsNotNull(
10522 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10523 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10524 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10525 MapTypeArg =
Builder.CreateOr(
10528 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10529 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT)));
10533 Value *OffloadingArgs[] = {MapperHandle,
Base, Begin,
10534 ArraySize, MapTypeArg, MapName};
10545 bool PreserveMemberOfFlags) {
10561 MapperFn->
addFnAttr(Attribute::NoInline);
10562 MapperFn->
addFnAttr(Attribute::NoUnwind);
10572 auto SavedIP =
Builder.saveIP();
10573 Builder.SetInsertPoint(EntryBB);
10585 TypeSize ElementSize =
M.getDataLayout().getTypeStoreSize(ElemTy);
10587 Value *PtrBegin = BeginIn;
10593 emitUDMapperArrayInitOrDel(MapperFn, MapperHandle, BaseIn, BeginIn,
Size,
10594 MapType, MapName, ElementSize, HeadBB,
10605 Builder.CreateICmpEQ(PtrBegin, PtrEnd,
"omp.arraymap.isempty");
10606 Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
10612 Builder.CreatePHI(PtrBegin->
getType(), 2,
"omp.arraymap.ptrcurrent");
10613 PtrPHI->addIncoming(PtrBegin, HeadBB);
10618 return Info.takeError();
10622 Value *OffloadingArgs[] = {MapperHandle};
10626 Value *ShiftedPreviousSize =
10630 for (
unsigned I = 0;
I < Info->BasePointers.size(); ++
I) {
10631 Value *CurBaseArg = Info->BasePointers[
I];
10632 Value *CurBeginArg = Info->Pointers[
I];
10633 Value *CurSizeArg = Info->Sizes[
I];
10634 Value *CurNameArg = Info->Names.size()
10639 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10642 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10645 static_cast<uint64_t>(OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF);
10647 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10648 OpenMPOffloadMappingFlags::OMP_MAP_ATTACH);
10706 Value *MemberMapType;
10707 if (PreserveMemberOfFlags || (RawType & AttachBit) ||
10708 Info->HasAttachPtr[
I]) {
10709 if (RawType & MemberOfMask)
10710 MemberMapType =
Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
10712 MemberMapType = OriMapType;
10714 MemberMapType =
Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
10732 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10733 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10734 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10744 Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB);
10750 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10751 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10752 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10758 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10759 OpenMPOffloadMappingFlags::OMP_MAP_TO)));
10760 Builder.CreateCondBr(IsTo, ToBB, ToElseBB);
10766 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10767 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10773 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10774 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10775 Builder.CreateCondBr(IsFrom, FromBB, EndBB);
10781 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10782 OpenMPOffloadMappingFlags::OMP_MAP_TO)));
10791 CurMapType->
addIncoming(MemberMapType, ToElseBB);
10810 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10811 OpenMPOffloadMappingFlags::OMP_MAP_ALWAYS |
10812 OpenMPOffloadMappingFlags::OMP_MAP_DELETE |
10813 OpenMPOffloadMappingFlags::OMP_MAP_CLOSE)));
10815 CurMapType, ImportedModifierBits,
"omp.maptype.with.modifiers");
10820 Value *FinalMapType =
10821 (RawType & AttachBit) ? CurMapType : CurMapTypeWithModifiers;
10823 Value *OffloadingArgs[] = {MapperHandle, CurBaseArg, CurBeginArg,
10824 CurSizeArg, FinalMapType, CurNameArg};
10826 auto ChildMapperFn = CustomMapperCB(
I);
10827 if (!ChildMapperFn)
10828 return ChildMapperFn.takeError();
10829 if (*ChildMapperFn) {
10844 Value *PtrNext =
Builder.CreateConstGEP1_32(ElemTy, PtrPHI, 1,
10845 "omp.arraymap.next");
10846 PtrPHI->addIncoming(PtrNext, LastBB);
10847 Value *IsDone =
Builder.CreateICmpEQ(PtrNext, PtrEnd,
"omp.arraymap.isdone");
10849 Builder.CreateCondBr(IsDone, ExitBB, BodyBB);
10854 emitUDMapperArrayInitOrDel(MapperFn, MapperHandle, BaseIn, BeginIn,
Size,
10855 MapType, MapName, ElementSize, DoneBB,
10869 bool IsNonContiguous,
10873 Info.clearArrayInfo();
10876 if (Info.NumberOfPtrs == 0)
10885 Info.RTArgs.BasePointersArray =
Builder.CreateAlloca(
10886 PointerArrayType,
nullptr,
".offload_baseptrs");
10888 Info.RTArgs.PointersArray =
Builder.CreateAlloca(
10889 PointerArrayType,
nullptr,
".offload_ptrs");
10891 PointerArrayType,
nullptr,
".offload_mappers");
10892 Info.RTArgs.MappersArray = MappersArray;
10899 ConstantInt::get(Int64Ty, 0));
10901 for (
unsigned I = 0, E = CombinedInfo.
Sizes.
size();
I < E; ++
I) {
10902 bool IsNonContigEntry =
10904 (
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10906 OpenMPOffloadMappingFlags::OMP_MAP_NON_CONTIG) != 0);
10909 if (IsNonContigEntry) {
10911 "Index must be in-bounds for NON_CONTIG Dims array");
10913 assert(DimCount > 0 &&
"NON_CONTIG DimCount must be > 0");
10914 ConstSizes[
I] = ConstantInt::get(Int64Ty, DimCount);
10919 ConstSizes[
I] = CI;
10923 RuntimeSizes.
set(
I);
10926 if (RuntimeSizes.
all()) {
10928 Info.RTArgs.SizesArray =
Builder.CreateAlloca(
10929 SizeArrayType,
nullptr,
".offload_sizes");
10935 auto *SizesArrayGbl =
10940 if (!RuntimeSizes.
any()) {
10941 Info.RTArgs.SizesArray = SizesArrayGbl;
10943 unsigned IndexSize =
M.getDataLayout().getIndexSizeInBits(0);
10944 Align OffloadSizeAlign =
M.getDataLayout().getABIIntegerTypeAlignment(64);
10947 SizeArrayType,
nullptr,
".offload_sizes");
10951 Buffer,
M.getDataLayout().getPrefTypeAlign(Buffer->
getType()),
10952 SizesArrayGbl, OffloadSizeAlign,
10957 Info.RTArgs.SizesArray = Buffer;
10965 for (
auto mapFlag : CombinedInfo.
Types)
10967 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10971 Info.RTArgs.MapTypesArray = MapTypesArrayGbl;
10977 Info.RTArgs.MapNamesArray = MapNamesArrayGbl;
10978 Info.EmitDebug =
true;
10980 Info.RTArgs.MapNamesArray =
10982 Info.EmitDebug =
false;
10987 if (Info.separateBeginEndCalls()) {
10988 bool EndMapTypesDiffer =
false;
10990 if (
Type &
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10991 OpenMPOffloadMappingFlags::OMP_MAP_PRESENT)) {
10992 Type &= ~static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
10993 OpenMPOffloadMappingFlags::OMP_MAP_PRESENT);
10994 EndMapTypesDiffer =
true;
10997 if (EndMapTypesDiffer) {
10999 Info.RTArgs.MapTypesArrayEnd = MapTypesArrayGbl;
11004 for (
unsigned I = 0;
I < Info.NumberOfPtrs; ++
I) {
11007 ArrayType::get(PtrTy, Info.NumberOfPtrs), Info.RTArgs.BasePointersArray,
11009 Builder.CreateAlignedStore(BPVal, BP,
11010 M.getDataLayout().getPrefTypeAlign(PtrTy));
11012 if (Info.requiresDevicePointerInfo()) {
11014 CodeGenIP =
Builder.saveIP();
11016 Info.DevicePtrInfoMap[BPVal] = {BP,
Builder.CreateAlloca(PtrTy)};
11019 DeviceAddrCB(
I, Info.DevicePtrInfoMap[BPVal].second);
11021 Info.DevicePtrInfoMap[BPVal] = {BP, BP};
11023 DeviceAddrCB(
I, BP);
11029 ArrayType::get(PtrTy, Info.NumberOfPtrs), Info.RTArgs.PointersArray, 0,
11032 Builder.CreateAlignedStore(PVal,
P,
11033 M.getDataLayout().getPrefTypeAlign(PtrTy));
11035 if (RuntimeSizes.
test(
I)) {
11037 ArrayType::get(Int64Ty, Info.NumberOfPtrs), Info.RTArgs.SizesArray,
11043 S,
M.getDataLayout().getPrefTypeAlign(PtrTy));
11046 unsigned IndexSize =
M.getDataLayout().getIndexSizeInBits(0);
11049 auto CustomMFunc = CustomMapperCB(
I);
11051 return CustomMFunc.takeError();
11053 MFunc =
Builder.CreatePointerCast(*CustomMFunc, PtrTy);
11056 PointerArrayType, MappersArray,
11059 MFunc, MAddr,
M.getDataLayout().getPrefTypeAlign(MAddr->
getType()));
11063 Info.NumberOfPtrs == 0)
11080 Builder.ClearInsertionPoint();
11111 auto CondConstant = CI->getSExtValue();
11113 return ThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
11115 return ElseGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
11125 Builder.CreateCondBr(
Cond, ThenBlock, ElseBlock);
11128 if (
Error Err = ThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks))
11134 if (
Error Err = ElseGen(AllocaIP,
Builder.saveIP(), DeallocBlocks))
11143bool OpenMPIRBuilder::checkAndEmitFlushAfterAtomic(
11147 "Unexpected Atomic Ordering.");
11149 bool Flush =
false;
11211 assert(
X.Var->getType()->isPointerTy() &&
11212 "OMP Atomic expects a pointer to target memory");
11213 Type *XElemTy =
X.ElemTy;
11216 "OMP atomic read expected a scalar type");
11218 Value *XRead =
nullptr;
11222 Builder.CreateLoad(XElemTy,
X.Var,
X.IsVolatile,
"omp.atomic.read");
11231 unsigned LoadSize =
DL.getTypeStoreSize(XElemTy);
11234 OldVal->
getAlign(),
true , AllocaIP,
X.Var);
11236 XRead = AtomicLoadRes.first;
11243 Builder.CreateLoad(IntCastTy,
X.Var,
X.IsVolatile,
"omp.atomic.load");
11246 XRead =
Builder.CreateBitCast(XLoad, XElemTy,
"atomic.flt.cast");
11248 XRead =
Builder.CreateIntToPtr(XLoad, XElemTy,
"atomic.ptr.cast");
11251 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Read);
11252 Builder.CreateStore(XRead, V.Var, V.IsVolatile);
11263 assert(
X.Var->getType()->isPointerTy() &&
11264 "OMP Atomic expects a pointer to target memory");
11265 Type *XElemTy =
X.ElemTy;
11268 "OMP atomic write expected a scalar type");
11276 unsigned LoadSize =
DL.getTypeStoreSize(XElemTy);
11279 OldVal->
getAlign(),
true , AllocaIP,
X.Var);
11287 Builder.CreateBitCast(Expr, IntCastTy,
"atomic.src.int.cast");
11292 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Write);
11299 AtomicUpdateCallbackTy &UpdateOp,
bool IsXBinopExpr,
11300 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11306 Type *XTy =
X.Var->getType();
11308 "OMP Atomic expects a pointer to target memory");
11309 Type *XElemTy =
X.ElemTy;
11312 "OMP atomic update expected a scalar or struct type");
11315 "OpenMP atomic does not support LT or GT operations");
11319 AllocaIP,
X.Var,
X.ElemTy, Expr, AO, RMWOp, UpdateOp,
X.IsVolatile,
11320 IsXBinopExpr, IsIgnoreDenormalMode, IsFineGrainedMemory, IsRemoteMemory);
11322 return AtomicResult.takeError();
11323 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Update);
11328Value *OpenMPIRBuilder::emitRMWOpAsInstruction(
Value *Src1,
Value *Src2,
11332 return Builder.CreateAdd(Src1, Src2);
11334 return Builder.CreateSub(Src1, Src2);
11336 return Builder.CreateAnd(Src1, Src2);
11338 return Builder.CreateNeg(Builder.CreateAnd(Src1, Src2));
11340 return Builder.CreateOr(Src1, Src2);
11342 return Builder.CreateXor(Src1, Src2);
11381Expected<std::pair<Value *, Value *>> OpenMPIRBuilder::emitAtomicUpdate(
11384 AtomicUpdateCallbackTy &UpdateOp,
bool VolatileX,
bool IsXBinopExpr,
11385 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11387 bool emitRMWOp =
false;
11395 emitRMWOp = XElemTy;
11398 emitRMWOp = (IsXBinopExpr && XElemTy);
11405 std::pair<Value *, Value *> Res;
11407 AtomicRMWInst *RMWInst =
11408 Builder.CreateAtomicRMW(RMWOp,
X, Expr, llvm::MaybeAlign(), AO);
11409 if (
T.isAMDGPU()) {
11410 if (IsIgnoreDenormalMode)
11411 RMWInst->
setMetadata(
"amdgpu.ignore.denormal.mode",
11413 if (!IsFineGrainedMemory)
11414 RMWInst->
setMetadata(
"amdgpu.no.fine.grained.memory",
11416 if (!IsRemoteMemory)
11420 Res.first = RMWInst;
11425 Res.second = Res.first;
11427 Res.second = emitRMWOpAsInstruction(Res.first, Expr, RMWOp);
11430 Builder.CreateLoad(XElemTy,
X,
X->getName() +
".atomic.load");
11436 OpenMPIRBuilder::AtomicInfo atomicInfo(
11438 OldVal->
getAlign(),
true , AllocaIP,
X);
11439 auto AtomicLoadRes = atomicInfo.EmitAtomicLoadLibcall(AO);
11442 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11449 AllocaInst *NewAtomicAddr =
Builder.CreateAlloca(XElemTy);
11450 NewAtomicAddr->
setName(
X->getName() +
"x.new.val");
11451 Builder.SetInsertPoint(ContBB);
11453 PHI->addIncoming(AtomicLoadRes.first, CurBB);
11455 Expected<Value *> CBResult = UpdateOp(OldExprVal,
Builder);
11458 Value *Upd = *CBResult;
11459 Builder.CreateStore(Upd, NewAtomicAddr);
11462 auto Result = atomicInfo.EmitAtomicCompareExchangeLibcall(
11463 AtomicLoadRes.second, NewAtomicAddr, AO, Failure);
11464 LoadInst *PHILoad =
Builder.CreateLoad(XElemTy,
Result.first);
11465 PHI->addIncoming(PHILoad,
Builder.GetInsertBlock());
11468 Res.first = OldExprVal;
11471 if (UnreachableInst *ExitTI =
11474 Builder.SetInsertPoint(ExitBB);
11476 Builder.SetInsertPoint(ExitTI);
11479 IntegerType *IntCastTy =
11482 Builder.CreateLoad(IntCastTy,
X,
X->getName() +
".atomic.load");
11492 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11499 AllocaInst *NewAtomicAddr =
Builder.CreateAlloca(XElemTy);
11500 NewAtomicAddr->
setName(
X->getName() +
"x.new.val");
11501 Builder.SetInsertPoint(ContBB);
11503 PHI->addIncoming(OldVal, CurBB);
11508 OldExprVal =
Builder.CreateBitCast(
PHI, XElemTy,
11509 X->getName() +
".atomic.fltCast");
11511 OldExprVal =
Builder.CreateIntToPtr(
PHI, XElemTy,
11512 X->getName() +
".atomic.ptrCast");
11516 Expected<Value *> CBResult = UpdateOp(OldExprVal,
Builder);
11519 Value *Upd = *CBResult;
11520 Builder.CreateStore(Upd, NewAtomicAddr);
11521 LoadInst *DesiredVal =
Builder.CreateLoad(IntCastTy, NewAtomicAddr);
11525 X,
PHI, DesiredVal, llvm::MaybeAlign(), AO, Failure);
11526 Result->setVolatile(VolatileX);
11527 Value *PreviousVal =
Builder.CreateExtractValue(Result, 0);
11528 Value *SuccessFailureVal =
Builder.CreateExtractValue(Result, 1);
11529 PHI->addIncoming(PreviousVal,
Builder.GetInsertBlock());
11530 Builder.CreateCondBr(SuccessFailureVal, ExitBB, ContBB);
11532 Res.first = OldExprVal;
11536 if (UnreachableInst *ExitTI =
11539 Builder.SetInsertPoint(ExitBB);
11541 Builder.SetInsertPoint(ExitTI);
11552 bool UpdateExpr,
bool IsPostfixUpdate,
bool IsXBinopExpr,
11553 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11558 Type *XTy =
X.Var->getType();
11560 "OMP Atomic expects a pointer to target memory");
11561 Type *XElemTy =
X.ElemTy;
11564 "OMP atomic capture expected a scalar or struct type");
11566 "OpenMP atomic does not support LT or GT operations");
11573 AllocaIP,
X.Var,
X.ElemTy, Expr, AO, AtomicOp, UpdateOp,
X.IsVolatile,
11574 IsXBinopExpr, IsIgnoreDenormalMode, IsFineGrainedMemory, IsRemoteMemory);
11577 Value *CapturedVal =
11578 (IsPostfixUpdate ? AtomicResult->first : AtomicResult->second);
11579 Builder.CreateStore(CapturedVal, V.Var, V.IsVolatile);
11581 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Capture);
11589 bool IsFailOnly,
bool IsWeak) {
11593 IsPostfixUpdate, IsFailOnly, Failure, IsWeak);
11605 assert(
X.Var->getType()->isPointerTy() &&
11606 "OMP atomic expects a pointer to target memory");
11609 assert(V.Var->getType()->isPointerTy() &&
"v.var must be of pointer type");
11610 assert(V.ElemTy ==
X.ElemTy &&
"x and v must be of same type");
11613 bool IsInteger = E->getType()->isIntegerTy();
11615 if (
Op == OMPAtomicCompareOp::EQ) {
11618 Value *OldValue =
nullptr;
11619 Value *SuccessOrFail =
nullptr;
11657 X.Var->getName() +
".atomic.load");
11663 Value *EIsNaN =
Builder.CreateFCmpUNO(E, E,
"atomic.e.isnan");
11664 Value *XIsNaN =
Builder.CreateFCmpUNO(XFP, XFP,
"atomic.x.isnan");
11665 Value *EitherNaN =
Builder.CreateOr(EIsNaN, XIsNaN,
"atomic.either.nan");
11670 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11674 M.getContext(),
X.Var->getName() +
".atomic.nan",
F, ExitBB);
11676 M.getContext(),
X.Var->getName() +
".atomic.notnan",
F, ExitBB);
11678 M.getContext(),
X.Var->getName() +
".atomic.zero",
F, ExitBB);
11680 M.getContext(),
X.Var->getName() +
".atomic.normal",
F, ExitBB);
11684 Builder.SetInsertPoint(CurBB);
11685 Builder.CreateCondBr(EitherNaN, NaNBB, NotNaNBB);
11688 Builder.SetInsertPoint(NaNBB);
11692 Builder.SetInsertPoint(NotNaNBB);
11695 X.Var->getName() +
".atomic.xiszero");
11697 "atomic.e.iszero");
11698 Value *BothZero =
Builder.CreateAnd(XIsZero, EIsZero,
"atomic.both.zero");
11699 Builder.CreateCondBr(BothZero, ZeroBB, NormalBB);
11702 Builder.SetInsertPoint(ZeroBB);
11704 X.Var, XCurr, DBCast,
MaybeAlign(), AO, Failure);
11706 Value *OldZero =
Builder.CreateExtractValue(ResZero, 0);
11707 Value *OkZero =
Builder.CreateExtractValue(ResZero, 1);
11711 Builder.SetInsertPoint(NormalBB);
11713 X.Var, EBCast, DBCast,
MaybeAlign(), AO, Failure);
11715 Value *OldNormal =
Builder.CreateExtractValue(ResNormal, 0);
11716 Value *OkNormal =
Builder.CreateExtractValue(ResNormal, 1);
11722 Builder.CreatePHI(IntCastTy, 3,
X.Var->getName() +
".atomic.old");
11727 X.Var->getName() +
".atomic.ok");
11734 Builder.SetInsertPoint(ExitBB);
11739 OldValue =
Builder.CreateBitCast(OldIntPHI,
X.ElemTy,
11740 X.Var->getName() +
".atomic.old.fp");
11741 SuccessOrFail = SuccessPHI;
11749 Result =
Builder.CreateAtomicCmpXchg(
X.Var, EBCast, DBCast,
11755 Result->setWeak(IsWeak);
11758 OldValue =
Builder.CreateExtractValue(Result, 0);
11760 OldValue =
Builder.CreateBitCast(OldValue,
X.ElemTy);
11762 "OldValue and V must be of same type");
11763 if (IsPostfixUpdate) {
11764 Builder.CreateStore(OldValue, V.Var, V.IsVolatile);
11766 SuccessOrFail =
Builder.CreateExtractValue(Result, 1);
11770 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11772 CurBBTI,
X.Var->getName() +
".atomic.exit");
11778 Builder.CreateCondBr(SuccessOrFail, ExitBB, ContBB);
11780 Builder.SetInsertPoint(ContBB);
11781 Builder.CreateStore(OldValue, V.Var);
11787 Builder.SetInsertPoint(ExitBB);
11789 Builder.SetInsertPoint(ExitTI);
11792 Value *CapturedValue =
11793 Builder.CreateSelect(SuccessOrFail, E, OldValue);
11794 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
11800 assert(R.Var->getType()->isPointerTy() &&
11801 "r.var must be of pointer type");
11802 assert(R.ElemTy->isIntegerTy() &&
"r must be of integral type");
11804 Value *SuccessFailureVal =
11805 Builder.CreateExtractValue(Result, 1);
11806 Value *ResultCast =
11807 R.IsSigned ?
Builder.CreateSExt(SuccessFailureVal, R.ElemTy)
11808 :
Builder.CreateZExt(SuccessFailureVal, R.ElemTy);
11809 Builder.CreateStore(ResultCast, R.Var, R.IsVolatile);
11818 "OldValue and V must be of same type");
11819 if (IsPostfixUpdate) {
11820 Builder.CreateStore(OldValue, V.Var, V.IsVolatile);
11825 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11827 CurBBTI,
X.Var->getName() +
".atomic.exit");
11833 Builder.CreateCondBr(SuccessOrFail, ExitBB, ContBB);
11835 Builder.SetInsertPoint(ContBB);
11836 Builder.CreateStore(OldValue, V.Var);
11842 Builder.SetInsertPoint(ExitBB);
11844 Builder.SetInsertPoint(ExitTI);
11847 Value *CapturedValue =
11848 Builder.CreateSelect(SuccessOrFail, E, OldValue);
11849 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
11855 assert(R.Var->getType()->isPointerTy() &&
11856 "r.var must be of pointer type");
11857 assert(R.ElemTy->isIntegerTy() &&
"r must be of integral type");
11859 Value *ResultCast = R.IsSigned
11860 ?
Builder.CreateSExt(SuccessOrFail, R.ElemTy)
11861 :
Builder.CreateZExt(SuccessOrFail, R.ElemTy);
11862 Builder.CreateStore(ResultCast, R.Var, R.IsVolatile);
11866 assert((
Op == OMPAtomicCompareOp::MAX ||
Op == OMPAtomicCompareOp::MIN) &&
11867 "Op should be either max or min at this point");
11868 assert(!IsFailOnly &&
"IsFailOnly is only valid when the comparison is ==");
11879 if (IsXBinopExpr) {
11908 Value *CapturedValue =
nullptr;
11909 if (IsPostfixUpdate) {
11910 CapturedValue = OldValue;
11935 Value *NonAtomicCmp =
Builder.CreateCmp(Pred, OldValue, E);
11936 CapturedValue =
Builder.CreateSelect(NonAtomicCmp, E, OldValue);
11938 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
11942 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Compare);
11962 if (&OuterAllocaBB ==
Builder.GetInsertBlock()) {
11989 bool SubClausesPresent =
11990 (NumTeamsLower || NumTeamsUpper || ThreadLimit || IfExpr);
11992 if (!
Config.isTargetDevice() && SubClausesPresent) {
11993 assert((NumTeamsLower ==
nullptr || NumTeamsUpper !=
nullptr) &&
11994 "if lowerbound is non-null, then upperbound must also be non-null "
11995 "for bounds on num_teams");
11997 if (NumTeamsUpper ==
nullptr)
11998 NumTeamsUpper =
Builder.getInt32(0);
12000 if (NumTeamsLower ==
nullptr)
12001 NumTeamsLower = NumTeamsUpper;
12005 "argument to if clause must be an integer value");
12009 IfExpr =
Builder.CreateICmpNE(IfExpr,
12010 ConstantInt::get(IfExpr->
getType(), 0));
12011 NumTeamsUpper =
Builder.CreateSelect(
12012 IfExpr, NumTeamsUpper,
Builder.getInt32(1),
"numTeamsUpper");
12015 NumTeamsLower =
Builder.CreateSelect(
12016 IfExpr, NumTeamsLower,
Builder.getInt32(1),
"numTeamsLower");
12019 if (ThreadLimit ==
nullptr)
12020 ThreadLimit =
Builder.getInt32(0);
12024 Value *NumTeamsLowerInt32 =
12026 Value *NumTeamsUpperInt32 =
12028 Value *ThreadLimitInt32 =
12035 {Ident, ThreadNum, NumTeamsLowerInt32, NumTeamsUpperInt32,
12036 ThreadLimitInt32});
12041 if (
Error Err = BodyGenCB(AllocaIP, CodeGenIP, ExitBB))
12044 auto OI = std::make_unique<OutlineInfo>();
12045 OI->EntryBB = AllocaBB;
12046 OI->ExitBB = ExitBB;
12047 OI->OuterAllocBB = &OuterAllocaBB;
12053 Builder, OuterAllocaIP, ToBeDeleted, AllocaIP,
"gid",
true));
12055 Builder, OuterAllocaIP, ToBeDeleted, AllocaIP,
"tid",
true));
12057 auto HostPostOutlineCB = [
this, Ident,
12058 ToBeDeleted](
Function &OutlinedFn)
mutable {
12063 "there must be a single user for the outlined function");
12068 "Outlined function must have two or three arguments only");
12070 bool HasShared = OutlinedFn.
arg_size() == 3;
12078 assert(StaleCI &&
"Error while outlining - no CallInst user found for the "
12079 "outlined function.");
12080 Builder.SetInsertPoint(StaleCI);
12087 omp::RuntimeFunction::OMPRTL___kmpc_fork_teams),
12091 I->eraseFromParent();
12094 if (!
Config.isTargetDevice())
12095 OI->PostOutlineCB = HostPostOutlineCB;
12099 Builder.SetInsertPoint(ExitBB);
12112 if (OuterAllocaBB ==
Builder.GetInsertBlock()) {
12127 if (
Error Err = BodyGenCB(AllocaIP, CodeGenIP, ExitBB))
12132 if (
Config.isTargetDevice()) {
12133 auto OI = std::make_unique<OutlineInfo>();
12134 OI->OuterAllocBB = OuterAllocIP.
getBlock();
12135 OI->EntryBB = AllocaBB;
12136 OI->ExitBB = ExitBB;
12137 OI->OuterDeallocBBs.reserve(OuterDeallocBlocks.
size());
12138 copy(OuterDeallocBlocks, OI->OuterDeallocBBs.
end());
12142 Builder.SetInsertPoint(ExitBB);
12149 std::string VarName) {
12158 return MapNamesArrayGlobal;
12163void OpenMPIRBuilder::initializeTypes(
Module &M) {
12167 unsigned ProgramAS = M.getDataLayout().getProgramAddressSpace();
12168#define OMP_TYPE(VarName, InitValue) VarName = InitValue;
12169#define OMP_ARRAY_TYPE(VarName, ElemTy, ArraySize) \
12170 VarName##Ty = ArrayType::get(ElemTy, ArraySize); \
12171 VarName##PtrTy = PointerType::get(Ctx, DefaultTargetAS);
12172#define OMP_FUNCTION_TYPE(VarName, IsVarArg, ReturnType, ...) \
12173 VarName = FunctionType::get(ReturnType, {__VA_ARGS__}, IsVarArg); \
12174 VarName##Ptr = PointerType::get(Ctx, ProgramAS);
12175#define OMP_STRUCT_TYPE(VarName, StructName, Packed, ...) \
12176 T = StructType::getTypeByName(Ctx, StructName); \
12178 T = StructType::create(Ctx, {__VA_ARGS__}, StructName, Packed); \
12180 VarName##Ptr = PointerType::get(Ctx, DefaultTargetAS);
12181#include "llvm/Frontend/OpenMP/OMPKinds.def"
12192 while (!Worklist.
empty()) {
12196 if (
BlockSet.insert(SuccBB).second)
12201std::unique_ptr<CodeExtractor>
12203 bool ArgsInZeroAddressSpace,
12205 return std::make_unique<CodeExtractor>(
12215 Suffix.
str(), ArgsInZeroAddressSpace);
12218std::unique_ptr<CodeExtractor> DeviceSharedMemOutlineInfo::createCodeExtractor(
12220 return std::make_unique<DeviceSharedMemCodeExtractor>(
12221 OMPBuilder, Blocks,
nullptr,
12229 OuterDeallocBBs.empty()
12232 Suffix.
str(), ArgsInZeroAddressSpace);
12242 Name.empty() ? Addr->
getName() : Name,
Size, Flags, 0);
12254 Fn->
addFnAttr(
"uniform-work-group-size");
12255 Fn->
addFnAttr(Attribute::MustProgress);
12273 auto &&GetMDInt = [
this](
unsigned V) {
12280 NamedMDNode *MD =
M.getOrInsertNamedMetadata(
"omp_offload.info");
12281 auto &&TargetRegionMetadataEmitter =
12282 [&
C, MD, &OrderedEntries, &GetMDInt, &GetMDString](
12297 GetMDInt(E.getKind()), GetMDInt(EntryInfo.DeviceID),
12298 GetMDInt(EntryInfo.FileID), GetMDString(EntryInfo.ParentName),
12299 GetMDInt(EntryInfo.Line), GetMDInt(EntryInfo.Count),
12300 GetMDInt(E.getOrder())};
12303 OrderedEntries[E.getOrder()] = std::make_pair(&E, EntryInfo);
12312 auto &&DeviceGlobalVarMetadataEmitter =
12313 [&
C, &OrderedEntries, &GetMDInt, &GetMDString, MD](
12323 Metadata *
Ops[] = {GetMDInt(E.getKind()), GetMDString(MangledName),
12324 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())};
12328 OrderedEntries[E.getOrder()] = std::make_pair(&E, varInfo);
12335 DeviceGlobalVarMetadataEmitter);
12337 for (
const auto &E : OrderedEntries) {
12338 assert(E.first &&
"All ordered entries must exist!");
12339 if (
const auto *CE =
12342 if (!CE->getID() || !CE->getAddress()) {
12346 if (!
M.getNamedValue(FnName))
12354 }
else if (
const auto *CE =
dyn_cast<
12363 if (
Config.isTargetDevice() &&
Config.hasRequiresUnifiedSharedMemory())
12365 if (!CE->getAddress()) {
12370 if (CE->getVarSize() == 0)
12374 assert(((
Config.isTargetDevice() && !CE->getAddress()) ||
12375 (!
Config.isTargetDevice() && CE->getAddress())) &&
12376 "Declaret target link address is set.");
12377 if (
Config.isTargetDevice())
12379 if (!CE->getAddress()) {
12386 if (!CE->getAddress()) {
12399 if ((
GV->hasLocalLinkage() ||
GV->hasHiddenVisibility()) &&
12403 OMPTargetGlobalVarEntryIndirectVTable))
12412 Flags, CE->getLinkage(), CE->getVarName());
12415 Flags, CE->getLinkage());
12426 if (
Config.hasRequiresFlags() && !
Config.isTargetDevice())
12432 Config.getRequiresFlags());
12442 OS <<
"_" <<
Count;
12447 unsigned NewCount = getTargetRegionEntryInfoCount(EntryInfo);
12450 EntryInfo.
Line, NewCount);
12458 auto FileIDInfo = CallBack();
12461 ID =
Status->getUniqueID();
12462 FileID =
Status->getUniqueID().getFile();
12466 FileID =
hash_value(std::get<0>(FileIDInfo));
12470 std::get<1>(FileIDInfo));
12476 static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12478 !(Remain & 1); Remain = Remain >> 1)
12496 if (
static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12498 static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12505 if (
static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12511 Flags &=
~omp::OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF;
12512 Flags |= MemberOfFlag;
12518 bool IsDeclaration,
bool IsExternallyVisible,
12520 std::vector<GlobalVariable *> &GeneratedRefs,
bool OpenMPSIMD,
12521 std::vector<Triple> TargetTriple,
Type *LlvmPtrTy,
12522 std::function<
Constant *()> GlobalInitializer,
12533 Config.hasRequiresUnifiedSharedMemory())) {
12538 if (!IsExternallyVisible)
12540 OS <<
"_decl_tgt_ref_ptr";
12543 Value *Ptr =
M.getNamedValue(PtrName);
12552 if (!
Config.isTargetDevice()) {
12553 if (GlobalInitializer)
12554 GV->setInitializer(GlobalInitializer());
12560 CaptureClause, DeviceClause, IsDeclaration, IsExternallyVisible,
12561 EntryInfo, MangledName, GeneratedRefs, OpenMPSIMD, TargetTriple,
12562 GlobalInitializer, VariableLinkage, LlvmPtrTy,
cast<Constant>(Ptr));
12574 bool IsDeclaration,
bool IsExternallyVisible,
12576 std::vector<GlobalVariable *> &GeneratedRefs,
bool OpenMPSIMD,
12577 std::vector<Triple> TargetTriple,
12578 std::function<
Constant *()> GlobalInitializer,
12582 (TargetTriple.empty() && !
Config.isTargetDevice()))
12593 !
Config.hasRequiresUnifiedSharedMemory()) {
12595 VarName = MangledName;
12598 if (!IsDeclaration)
12600 M.getDataLayout().getTypeSizeInBits(LlvmVal->
getValueType()), 8);
12603 Linkage = (VariableLinkage) ? VariableLinkage() : LlvmVal->
getLinkage();
12607 if (
Config.isTargetDevice() &&
12616 if (!
M.getNamedValue(RefName)) {
12620 GvAddrRef->setConstant(
true);
12622 GvAddrRef->setInitializer(Addr);
12623 GeneratedRefs.push_back(GvAddrRef);
12632 if (
Config.isTargetDevice()) {
12633 VarName = (Addr) ? Addr->
getName() :
"";
12637 CaptureClause, DeviceClause, IsDeclaration, IsExternallyVisible,
12638 EntryInfo, MangledName, GeneratedRefs, OpenMPSIMD, TargetTriple,
12639 LlvmPtrTy, GlobalInitializer, VariableLinkage);
12640 VarName = (Addr) ? Addr->
getName() :
"";
12642 VarSize =
M.getDataLayout().getPointerSize();
12661 auto &&GetMDInt = [MN](
unsigned Idx) {
12666 auto &&GetMDString = [MN](
unsigned Idx) {
12668 return V->getString();
12671 switch (GetMDInt(0)) {
12675 case OffloadEntriesInfoManager::OffloadEntryInfo::
12676 OffloadingEntryInfoTargetRegion: {
12686 case OffloadEntriesInfoManager::OffloadEntryInfo::
12687 OffloadingEntryInfoDeviceGlobalVar:
12700 if (HostFilePath.
empty())
12704 if (std::error_code Err = Buf.getError()) {
12706 "OpenMPIRBuilder: " +
12714 if (std::error_code Err =
M.getError()) {
12716 (
"error parsing host file inside of OpenMPIRBuilder: " + Err.message())
12730 "expected a valid insertion block for creating an iterator loop");
12740 Builder.getCurrentDebugLocation(),
"omp.it.cont");
12752 T->eraseFromParent();
12761 if (!BodyBr || BodyBr->getSuccessor() != CLI->
getLatch()) {
12763 "iterator bodygen must terminate the canonical body with an "
12764 "unconditional branch to the loop latch",
12788 for (
const auto &
ParamAttr : ParamAttrs) {
12831 return std::string(Out.
str());
12839 unsigned VecRegSize;
12841 ISADataTy ISAData[] = {
12860 for (
char Mask :
Masked) {
12861 for (
const ISADataTy &
Data : ISAData) {
12864 Out <<
"_ZGV" <<
Data.ISA << Mask;
12866 assert(NumElts &&
"Non-zero simdlen/cdtsize expected");
12880template <
typename T>
12883 StringRef MangledName,
bool OutputBecomesInput,
12887 Out << Prefix << ISA << LMask << VLEN;
12888 if (OutputBecomesInput)
12890 Out << ParSeq <<
'_' << MangledName;
12899 bool OutputBecomesInput,
12904 OutputBecomesInput, Fn);
12906 OutputBecomesInput, Fn);
12910 OutputBecomesInput, Fn);
12912 OutputBecomesInput, Fn);
12916 OutputBecomesInput, Fn);
12918 OutputBecomesInput, Fn);
12923 OutputBecomesInput, Fn);
12934 char ISA,
unsigned NarrowestDataSize,
bool OutputBecomesInput) {
12935 assert((ISA ==
'n' || ISA ==
's') &&
"Expected ISA either 's' or 'n'.");
12947 OutputBecomesInput, Fn);
12954 OutputBecomesInput, Fn);
12956 OutputBecomesInput, Fn);
12960 OutputBecomesInput, Fn);
12964 OutputBecomesInput, Fn);
12973 OutputBecomesInput, Fn);
12980 MangledName, OutputBecomesInput, Fn);
12982 MangledName, OutputBecomesInput, Fn);
12986 MangledName, OutputBecomesInput, Fn);
12990 MangledName, OutputBecomesInput, Fn);
13000 return OffloadEntriesTargetRegion.empty() &&
13001 OffloadEntriesDeviceGlobalVar.empty();
13004unsigned OffloadEntriesInfoManager::getTargetRegionEntryInfoCount(
13006 auto It = OffloadEntriesTargetRegionCount.find(
13007 getTargetRegionEntryCountKey(EntryInfo));
13008 if (It == OffloadEntriesTargetRegionCount.end())
13013void OffloadEntriesInfoManager::incrementTargetRegionEntryInfoCount(
13015 OffloadEntriesTargetRegionCount[getTargetRegionEntryCountKey(EntryInfo)] =
13016 EntryInfo.
Count + 1;
13022 OffloadEntriesTargetRegion[EntryInfo] =
13025 ++OffloadingEntriesNum;
13031 assert(EntryInfo.
Count == 0 &&
"expected default EntryInfo");
13034 EntryInfo.
Count = getTargetRegionEntryInfoCount(EntryInfo);
13038 if (OMPBuilder->Config.isTargetDevice()) {
13043 auto &Entry = OffloadEntriesTargetRegion[EntryInfo];
13044 Entry.setAddress(Addr);
13046 Entry.setFlags(Flags);
13052 "Target region entry already registered!");
13054 OffloadEntriesTargetRegion[EntryInfo] = Entry;
13055 ++OffloadingEntriesNum;
13057 incrementTargetRegionEntryInfoCount(EntryInfo);
13064 EntryInfo.
Count = getTargetRegionEntryInfoCount(EntryInfo);
13066 auto It = OffloadEntriesTargetRegion.find(EntryInfo);
13067 if (It == OffloadEntriesTargetRegion.end()) {
13071 if (!IgnoreAddressId && (It->second.getAddress() || It->second.getID()))
13079 for (
const auto &It : OffloadEntriesTargetRegion) {
13080 Action(It.first, It.second);
13086 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags);
13087 ++OffloadingEntriesNum;
13093 if (OMPBuilder->Config.isTargetDevice()) {
13097 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
13099 if (Entry.getVarSize() == 0) {
13100 Entry.setVarSize(VarSize);
13101 Entry.setLinkage(Linkage);
13105 Entry.setVarSize(VarSize);
13106 Entry.setLinkage(Linkage);
13107 Entry.setAddress(Addr);
13110 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
13111 assert(Entry.isValid() && Entry.getFlags() == Flags &&
13112 "Entry not initialized!");
13113 if (Entry.getVarSize() == 0) {
13114 Entry.setVarSize(VarSize);
13115 Entry.setLinkage(Linkage);
13122 OffloadEntriesDeviceGlobalVar.try_emplace(VarName, OffloadingEntriesNum,
13123 Addr, VarSize, Flags, Linkage,
13126 OffloadEntriesDeviceGlobalVar.try_emplace(
13127 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage,
"");
13128 ++OffloadingEntriesNum;
13135 for (
const auto &E : OffloadEntriesDeviceGlobalVar)
13136 Action(E.getKey(), E.getValue());
13143void CanonicalLoopInfo::collectControlBlocks(
13150 BBs.
append({getPreheader(), Header,
Cond, Latch, Exit, getAfter()});
13162void CanonicalLoopInfo::setTripCount(
Value *TripCount) {
13174void CanonicalLoopInfo::mapIndVar(
13184 for (
Use &U : OldIV->
uses()) {
13188 if (
User->getParent() == getCond())
13190 if (
User->getParent() == getLatch())
13196 Value *NewIV = Updater(OldIV);
13199 for (Use *U : ReplacableUses)
13220 "Preheader must terminate with unconditional branch");
13222 "Preheader must jump to header");
13226 "Header must terminate with unconditional branch");
13227 assert(Header->getSingleSuccessor() == Cond &&
13228 "Header must jump to exiting block");
13231 assert(Cond->getSinglePredecessor() == Header &&
13232 "Exiting block only reachable from header");
13235 "Exiting block must terminate with conditional branch");
13237 "Exiting block's first successor jump to the body");
13239 "Exiting block's second successor must exit the loop");
13243 "Body only reachable from exiting block");
13248 "Latch must terminate with unconditional branch");
13249 assert(Latch->getSingleSuccessor() == Header &&
"Latch must jump to header");
13252 assert(Latch->getSinglePredecessor() !=
nullptr);
13257 "Exit block must terminate with unconditional branch");
13258 assert(Exit->getSingleSuccessor() == After &&
13259 "Exit block must jump to after block");
13263 "After block only reachable from exit block");
13267 assert(IndVar &&
"Canonical induction variable not found?");
13269 "Induction variable must be an integer");
13271 "Induction variable must be a PHI in the loop header");
13277 auto *NextIndVar =
cast<PHINode>(IndVar)->getIncomingValue(1);
13285 assert(TripCount &&
"Loop trip count not found?");
13287 "Trip count and induction variable must have the same type");
13291 "Exit condition must be a signed less-than comparison");
13293 "Exit condition must compare the induction variable");
13295 "Exit condition must compare with the trip count");
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Expand Atomic instructions
This file contains the simple types necessary to represent the attributes associated with functions a...
static const Function * getParent(const Value *V)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
This file provides various utilities for inspecting and working with the control flow graph in LLVM I...
This header defines various interfaces for pass management in LLVM.
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static bool isZero(Value *V, const DataLayout &DL, DominatorTree *DT, AssumptionCache *AC)
static cl::opt< unsigned > TileSize("fuse-matrix-tile-size", cl::init(4), cl::Hidden, cl::desc("Tile size for matrix instruction fusion using square-shaped tiles."))
uint64_t IntrinsicInst * II
#define OMP_KERNEL_ARG_VERSION
Provides definitions for Target specific Grid Values.
static Value * removeASCastIfPresent(Value *V)
static void createTargetLoopWorkshareCall(OpenMPIRBuilder *OMPBuilder, WorksharingLoopType LoopType, BasicBlock *InsertBlock, Value *Ident, Value *LoopBodyArg, Value *TripCount, Function &LoopBodyFn, bool NoLoop)
Value * createFakeIntVal(IRBuilderBase &Builder, OpenMPIRBuilder::InsertPointTy OuterAllocaIP, llvm::SmallVectorImpl< Instruction * > &ToBeDeleted, OpenMPIRBuilder::InsertPointTy InnerAllocaIP, const Twine &Name="", bool AsPtr=true, bool Is64Bit=false)
static Function * createTargetParallelWrapper(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn)
Create wrapper function used to gather the outlined function's argument structure from a shared buffe...
static void redirectTo(BasicBlock *Source, BasicBlock *Target, DebugLoc DL)
Make Source branch to Target.
static FunctionCallee getKmpcDistForStaticInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
static void applyParallelAccessesMetadata(CanonicalLoopInfo *CLI, LLVMContext &Ctx, Loop *Loop, LoopInfo &LoopInfo, SmallVector< Metadata * > &LoopMDList)
static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, char ISA, StringRef ParSeq, StringRef MangledName, bool OutputBecomesInput, llvm::Function *Fn)
static FunctionCallee getKmpcForDynamicFiniForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for finalizing the dynamic loop using depending on type.
static Expected< Function * > createOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, StringRef FuncName, SmallVectorImpl< Value * > &Inputs, OpenMPIRBuilder::TargetBodyGenCallbackTy &CBFunc, OpenMPIRBuilder::TargetGenArgAccessorsCallbackTy &ArgAccessorFuncCB)
static void FixupDebugInfoForOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, Function *Func, DenseMap< Value *, std::tuple< Value *, unsigned > > &ValueReplacementMap)
static OMPScheduleType getOpenMPOrderingScheduleType(OMPScheduleType BaseScheduleType, bool HasOrderedClause)
Adds ordering modifier flags to schedule type.
static OMPScheduleType getOpenMPMonotonicityScheduleType(OMPScheduleType ScheduleType, bool HasSimdModifier, bool HasMonotonic, bool HasNonmonotonic, bool HasOrderedClause)
Adds monotonicity modifier flags to schedule type.
static std::string mangleVectorParameters(ArrayRef< llvm::OpenMPIRBuilder::DeclareSimdAttrTy > ParamAttrs)
Mangle the parameter part of the vector function name according to their OpenMP classification.
static bool isGenericKernel(Function &Fn)
static void workshareLoopTargetCallback(OpenMPIRBuilder *OMPIRBuilder, CanonicalLoopInfo *CLI, Value *Ident, Function &OutlinedFn, const SmallVector< Instruction *, 4 > &ToBeDeleted, WorksharingLoopType LoopType, bool NoLoop)
static bool isValidWorkshareLoopScheduleType(OMPScheduleType SchedType)
static bool isAtomicableReductionSet(ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos)
static llvm::CallInst * emitNoUnwindRuntimeCall(IRBuilder<> &Builder, llvm::FunctionCallee Callee, ArrayRef< llvm::Value * > Args, const llvm::Twine &Name)
static Error populateReductionFunction(Function *ReductionFunc, ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos, IRBuilder<> &Builder, ArrayRef< bool > IsByRef, bool IsGPU)
static Function * getFreshReductionFunc(Module &M)
static void raiseUserConstantDataAllocasToEntryBlock(IRBuilderBase &Builder, Function *Function)
static FunctionCallee getKmpcForDynamicNextForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for updating the next loop using OpenMP dynamic scheduling depending...
static bool isConflictIP(IRBuilder<>::InsertPoint IP1, IRBuilder<>::InsertPoint IP2)
Return whether IP1 and IP2 are ambiguous, i.e.
static void checkReductionInfos(ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos, bool IsGPU)
static Type * getOffloadingArrayType(Value *V)
static OMPScheduleType getOpenMPBaseScheduleType(llvm::omp::ScheduleKind ClauseKind, bool HasChunks, bool HasSimdModifier, bool HasDistScheduleChunks)
Determine which scheduling algorithm to use, determined from schedule clause arguments.
static OMPScheduleType computeOpenMPScheduleType(ScheduleKind ClauseKind, bool HasChunks, bool HasSimdModifier, bool HasMonotonicModifier, bool HasNonmonotonicModifier, bool HasOrderedClause, bool HasDistScheduleChunks)
Determine the schedule type using schedule and ordering clause arguments.
static FunctionCallee getKmpcForDynamicInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for initializing loop bounds using OpenMP dynamic scheduling dependi...
static std::optional< omp::OMPTgtExecModeFlags > getTargetKernelExecMode(Function &Kernel)
Given a function, if it represents the entry point of a target kernel, this returns the execution mod...
static StructType * createTaskWithPrivatesTy(OpenMPIRBuilder &OMPIRBuilder, ArrayRef< Value * > OffloadingArraysToPrivatize)
static cl::opt< double > UnrollThresholdFactor("openmp-ir-builder-unroll-threshold-factor", cl::Hidden, cl::desc("Factor for the unroll threshold to account for code " "simplifications still taking place"), cl::init(1.5))
static cl::opt< bool > UseDefaultMaxThreads("openmp-ir-builder-use-default-max-threads", cl::Hidden, cl::desc("Use a default max threads if none is provided."), cl::init(true))
static int32_t computeHeuristicUnrollFactor(CanonicalLoopInfo *CLI)
Heuristically determine the best-performant unroll factor for CLI.
static void emitTargetCall(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, OpenMPIRBuilder::InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, OpenMPIRBuilder::TargetDataInfo &Info, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, const OpenMPIRBuilder::TargetKernelRuntimeAttrs &RuntimeAttrs, Value *IfCond, Function *OutlinedFn, Constant *OutlinedFnID, SmallVectorImpl< Value * > &Args, OpenMPIRBuilder::GenMapInfoCallbackTy GenMapInfoCB, OpenMPIRBuilder::CustomMapperCallbackTy CustomMapperCB, const OpenMPIRBuilder::DependenciesInfo &Dependencies, bool HasNoWait, Value *DynCGroupMem, OMPDynGroupprivateFallbackType DynCGroupMemFallback)
static Value * emitTaskDependencies(OpenMPIRBuilder &OMPBuilder, const SmallVectorImpl< OpenMPIRBuilder::DependData > &Dependencies)
static Error emitTargetOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, bool IsOffloadEntry, TargetRegionEntryInfo &EntryInfo, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, Function *&OutlinedFn, Constant *&OutlinedFnID, SmallVectorImpl< Value * > &Inputs, OpenMPIRBuilder::TargetBodyGenCallbackTy &CBFunc, OpenMPIRBuilder::TargetGenArgAccessorsCallbackTy &ArgAccessorFuncCB)
static void updateNVPTXAttr(Function &Kernel, StringRef Name, int32_t Value, bool Min)
static OpenMPIRBuilder::InsertPointTy getInsertPointAfterInstr(Instruction *I)
static void redirectAllPredecessorsTo(BasicBlock *OldTarget, BasicBlock *NewTarget, DebugLoc DL)
Redirect all edges that branch to OldTarget to NewTarget.
static void hoistNonEntryAllocasToEntryBlock(llvm::BasicBlock &Block)
static std::unique_ptr< TargetMachine > createTargetMachine(Function *F, CodeGenOptLevel OptLevel)
Create the TargetMachine object to query the backend for optimization preferences.
static FunctionCallee getKmpcForStaticInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
static void addAccessGroupMetadata(BasicBlock *Block, MDNode *AccessGroup, LoopInfo &LI)
Attach llvm.access.group metadata to the memref instructions of Block.
static void addBasicBlockMetadata(BasicBlock *BB, ArrayRef< Metadata * > Properties)
Attach metadata Properties to the basic block described by BB.
static void restoreIPandDebugLoc(llvm::IRBuilderBase &Builder, llvm::IRBuilderBase::InsertPoint IP)
This is a wrapper over IRBuilderBase::restoreIP that also restores a current debug location when the ...
static LoadInst * loadSharedDataFromTaskDescriptor(OpenMPIRBuilder &OMPIRBuilder, IRBuilderBase &Builder, Value *TaskWithPrivates, Type *TaskWithPrivatesTy)
Given a task descriptor, TaskWithPrivates, return the pointer to the block of pointers containing sha...
static cl::opt< bool > OptimisticAttributes("openmp-ir-builder-optimistic-attributes", cl::Hidden, cl::desc("Use optimistic attributes describing " "'as-if' properties of runtime calls."), cl::init(false))
static bool hasGridValue(const Triple &T)
static FunctionCallee getKmpcForStaticLoopForType(Type *Ty, OpenMPIRBuilder *OMPBuilder, WorksharingLoopType LoopType)
static const omp::GV & getGridValue(const Triple &T, Function *Kernel)
static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, StringRef Prefix, char ISA, StringRef ParSeq, StringRef MangledName, bool OutputBecomesInput, llvm::Function *Fn)
static Function * emitTargetTaskProxyFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, CallInst *StaleCI, StructType *PrivatesTy, StructType *TaskWithPrivatesTy, const size_t NumOffloadingArrays, const int SharedArgsOperandNo)
Create an entry point for a target task with the following.
static void addLoopMetadata(CanonicalLoopInfo *Loop, ArrayRef< Metadata * > Properties)
Attach loop metadata Properties to the loop described by Loop.
static AtomicOrdering TransformReleaseAcquireRelease(AtomicOrdering AO)
static void removeUnusedBlocksFromParent(ArrayRef< BasicBlock * > BBs)
static void targetParallelCallback(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn, Function *OuterFn, BasicBlock *OuterAllocaBB, Value *Ident, Value *IfCondition, Value *NumThreads, Instruction *PrivTID, AllocaInst *PrivTIDAddr, Value *ThreadID, const SmallVector< Instruction *, 4 > &ToBeDeleted)
static void hostParallelCallback(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn, Function *OuterFn, Value *Ident, Value *IfCondition, Instruction *PrivTID, AllocaInst *PrivTIDAddr, const SmallVector< Instruction *, 4 > &ToBeDeleted)
FunctionAnalysisManager FAM
This file defines the Pass Instrumentation classes that provide instrumentation points into the pass ...
const SmallVectorImpl< MachineOperand > & Cond
Remove Loads Into Fake Uses
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
SmallPtrSet< BasicBlock *, 0 > BlockSet
This file implements the SmallBitVector class.
This file defines the SmallSet class.
static SymbolRef::Type getType(const Symbol *Sym)
Defines the virtual file system interface vfs::FileSystem.
static cl::opt< unsigned > MaxThreads("xcore-max-threads", cl::Optional, cl::desc("Maximum number of threads (for emulation thread-local storage)"), cl::Hidden, cl::value_desc("number"), cl::init(8))
static const uint32_t IV[8]
Class for arbitrary precision integers.
An arbitrary precision integer that knows its signedness.
static APSInt getUnsigned(uint64_t X)
This class represents a conversion between pointers from one address space to another.
an instruction to allocate memory on the stack
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
PointerType * getType() const
Overload to return most specific pointer type.
Type * getAllocatedType() const
Return the type that is being allocated by the instruction.
unsigned getAddressSpace() const
Return the address space for the allocation.
LLVM_ABI std::optional< TypeSize > getAllocationSize(const DataLayout &DL) const
Get allocation size in bytes.
LLVM_ABI bool isArrayAllocation() const
Return true if there is an allocation size parameter to the allocation instruction that is not 1.
void setAlignment(Align Align)
const Value * getArraySize() const
Get the number of elements allocated.
bool registerPass(PassBuilderT &&PassBuilder)
Register an analysis pass with the manager.
This class represents an incoming formal argument to a Function.
unsigned getArgNo() const
Return the index of this formal argument in its containing function.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
Class to represent array types.
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
A function analysis which provides an AssumptionCache.
LLVM_ABI AssumptionCache run(Function &F, FunctionAnalysisManager &)
A cache of @llvm.assume calls within a function.
An instruction that atomically checks whether a specified value is in a memory location,...
void setWeak(bool IsWeak)
static AtomicOrdering getStrongestFailureOrdering(AtomicOrdering SuccessOrdering)
Returns the strongest permitted ordering on failure, given the desired ordering on success.
LLVM_ABI std::pair< LoadInst *, AllocaInst * > EmitAtomicLoadLibcall(AtomicOrdering AO)
LLVM_ABI void EmitAtomicStoreLibcall(AtomicOrdering AO, Value *Source)
an instruction that atomically reads a memory location, combines it with another value,...
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ USubCond
Subtract only if no unsigned overflow.
@ FMinimum
*p = minimum(old, v) minimum matches the behavior of llvm.minimum.
@ Min
*p = old <signed v ? old : v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FMaximum
*p = maximum(old, v) maximum matches the behavior of llvm.maximum.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMaximumNum
*p = maximumnum(old, v) maximumnum matches the behavior of llvm.maximumnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ FMinimumNum
*p = minimumnum(old, v) minimumnum matches the behavior of llvm.minimumnum.
This class holds the attributes for a particular argument, parameter, function, or return value.
LLVM_ABI AttributeSet addAttributes(LLVMContext &C, AttributeSet AS) const
Add attributes to the attribute set.
LLVM_ABI AttributeSet addAttribute(LLVMContext &C, Attribute::AttrKind Kind) const
Add an argument attribute.
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
LLVM_ABI void replaceSuccessorsPhiUsesWith(BasicBlock *Old, BasicBlock *New)
Update all phi nodes in this basic block's successors to refer to basic block New instead of basic bl...
iterator begin()
Instruction iterator methods.
LLVM_ABI const_iterator getFirstInsertionPt() const
Returns an iterator to the first instruction in this block that is suitable for inserting a non-PHI i...
LLVM_ABI BasicBlock * splitBasicBlock(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction.
const Function * getParent() const
Return the enclosing method, or null if none.
reverse_iterator rbegin()
bool hasTerminator() const LLVM_READONLY
Returns whether the block has a terminator.
const Instruction & back() const
LLVM_ABI BasicBlock * splitBasicBlockBefore(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction and insert the new basic blo...
LLVM_ABI InstListType::const_iterator getFirstNonPHIIt() const
Returns an iterator to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI void insertDbgRecordBefore(DbgRecord *DR, InstListType::iterator Here)
Insert a DbgRecord into a block at the position given by Here.
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
LLVM_ABI InstListType::const_iterator getFirstNonPHIOrDbg(bool SkipPseudoOp=true) const
Returns a pointer to the first instruction in this block that is not a PHINode or a debug intrinsic,...
LLVM_ABI const BasicBlock * getUniqueSuccessor() const
Return the successor of this block if it has a unique successor.
LLVM_ABI const BasicBlock * getSinglePredecessor() const
Return the predecessor of this block if it has a single predecessor block.
const Instruction & front() const
InstListType::reverse_iterator reverse_iterator
LLVM_ABI const BasicBlock * getUniquePredecessor() const
Return the predecessor of this block if it has a unique predecessor block.
const Instruction * getTerminatorOrNull() const LLVM_READONLY
Returns the terminator instruction if the block is well formed or null if the block is not well forme...
LLVM_ABI const BasicBlock * getSingleSuccessor() const
Return the successor of this block if it has a single successor.
LLVM_ABI SymbolTableList< BasicBlock >::iterator eraseFromParent()
Unlink 'this' from the containing function and delete it.
InstListType::iterator iterator
Instruction iterators...
LLVM_ABI LLVMContext & getContext() const
Get the context in which this basic block lives.
void moveBefore(BasicBlock *MovePos)
Unlink this basic block from its current function and insert it into the function that MovePos lives ...
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
void splice(BasicBlock::iterator ToIt, BasicBlock *FromBB)
Transfer all instructions from FromBB to this basic block at ToIt.
LLVM_ABI void removePredecessor(BasicBlock *Pred, bool KeepOneInputPHIs=false)
Update PHI nodes in this BasicBlock before removal of predecessor Pred.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
Value * getArgOperand(unsigned i) const
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Class to represented the control flow structure of an OpenMP canonical loop.
Value * getTripCount() const
Returns the llvm::Value containing the number of loop iterations.
BasicBlock * getHeader() const
The header is the entry for each iteration.
LLVM_ABI void assertOK() const
Consistency self-check.
Type * getIndVarType() const
Return the type of the induction variable (and the trip count).
BasicBlock * getBody() const
The body block is the single entry for a loop iteration and not controlled by CanonicalLoopInfo.
bool isValid() const
Returns whether this object currently represents the IR of a loop.
void setLastIter(Value *IterVar)
Sets the last iteration variable for this loop.
OpenMPIRBuilder::InsertPointTy getAfterIP() const
Return the insertion point for user code after the loop.
OpenMPIRBuilder::InsertPointTy getBodyIP() const
Return the insertion point for user code in the body.
BasicBlock * getAfter() const
The after block is intended for clean-up code such as lifetime end markers.
Function * getFunction() const
LLVM_ABI void invalidate()
Invalidate this loop.
BasicBlock * getLatch() const
Reaching the latch indicates the end of the loop body code.
OpenMPIRBuilder::InsertPointTy getPreheaderIP() const
Return the insertion point for user code before the loop.
BasicBlock * getCond() const
The condition block computes whether there is another loop iteration.
BasicBlock * getExit() const
Reaching the exit indicates no more iterations are being executed.
LLVM_ABI BasicBlock * getPreheader() const
The preheader ensures that there is only a single edge entering the loop.
Instruction * getIndVar() const
Returns the instruction representing the current logical induction variable.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
static LLVM_ABI Constant * get(ArrayType *T, ArrayRef< Constant * > V)
static Constant * get(LLVMContext &Context, ArrayRef< ElementTy > Elts)
get() constructor - Return a constant with array type with an element count and element type matching...
static LLVM_ABI Constant * getString(LLVMContext &Context, StringRef Initializer, bool AddNull=true, bool ByteString=false)
This method constructs a CDS and initializes it with a text string.
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getTruncOrBitCast(Constant *C, Type *Ty)
static LLVM_ABI Constant * getPointerBitCastOrAddrSpaceCast(Constant *C, Type *Ty)
Create a BitCast or AddrSpaceCast for a pointer type depending on the address space.
static LLVM_ABI Constant * getSizeOf(Type *Ty)
getSizeOf constant expr - computes the (alloc) size of a type (in address-units, not bits) in a targe...
static LLVM_ABI Constant * getAddrSpaceCast(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
This is an important base class in LLVM.
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
DILocalScope * getScope() const
Get the local scope for this variable.
DINodeArray getAnnotations() const
Subprogram description. Uses SubclassData1.
uint32_t getAlignInBits() const
StringRef getName() const
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeStoreSize(Type *Ty) const
Returns the maximum number of bytes that may be overwritten by storing the specified type.
Record of a variable value-assignment, aka a non instruction representation of the dbg....
Analysis pass which computes a DominatorTree.
LLVM_ABI DominatorTree run(Function &F, FunctionAnalysisManager &)
Run the analysis pass over a function and produce a dominator tree.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Represents either an error or a value T.
Lightweight error class with error context and mandatory checking.
static ErrorSuccess success()
Create a success value.
Tagged union holding either a T or a Error.
Error takeError()
Take ownership of the stored error.
reference get()
Returns a reference to the stored T value.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
Class to represent function types.
Type * getParamType(unsigned i) const
Parameter type accessors.
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
void addFnAttr(Attribute::AttrKind Kind)
Add function attributes to this function.
static Function * Create(FunctionType *Ty, LinkageTypes Linkage, unsigned AddrSpace, const Twine &N="", Module *M=nullptr)
const BasicBlock & getEntryBlock() const
FunctionType * getFunctionType() const
Returns the FunctionType for me.
void removeFromParent()
removeFromParent - This method unlinks 'this' from the containing module, but does not delete it.
const DataLayout & getDataLayout() const
Get the data layout of the module this function belongs to.
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
DISubprogram * getSubprogram() const
Get the attached subprogram.
AttributeList getAttributes() const
Return the attribute list for this Function.
const Function & getFunction() const
void setAttributes(AttributeList Attrs)
Set the attribute list for this Function.
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
void addParamAttr(unsigned ArgNo, Attribute::AttrKind Kind)
adds the attribute to the list of attributes for the given arg.
Function::iterator insert(Function::iterator Position, BasicBlock *BB)
Insert BB in the basic block list at Position.
Type * getReturnType() const
Returns the type of the ret val.
void setCallingConv(CallingConv::ID CC)
Argument * getArg(unsigned i) const
bool hasMetadata() const
Return true if this GlobalObject has any metadata attached to it.
LLVM_ABI void addMetadata(unsigned KindID, MDNode &MD)
Add a metadata attachment.
LinkageTypes getLinkage() const
void setLinkage(LinkageTypes LT)
Module * getParent()
Get the module that this global value is contained inside of...
void setDSOLocal(bool Local)
PointerType * getType() const
Global values are always pointers.
@ HiddenVisibility
The GV is hidden.
@ ProtectedVisibility
The GV is protected.
void setVisibility(VisibilityTypes V)
LinkageTypes
An enumeration for the kinds of linkage for global values.
@ PrivateLinkage
Like Internal, but omit from symbol table.
@ CommonLinkage
Tentative definitions.
@ InternalLinkage
Rename collisions when linking (static functions).
@ WeakODRLinkage
Same, but only replaced by something equivalent.
@ WeakAnyLinkage
Keep one copy of named function when linking (weak)
@ AppendingLinkage
Special purpose, only applies to global arrays.
@ LinkOnceODRLinkage
Same, but only replaced by something equivalent.
Type * getValueType() const
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
InsertPoint - A saved insertion point.
BasicBlock * getBlock() const
bool isSet() const
Returns true if this insert point is set.
BasicBlock::iterator getPoint() const
Common base class shared among various IRBuilders.
InsertPoint saveIP() const
Returns the current insert point.
void restoreIP(InsertPoint IP)
Sets the current insert point to a previously-saved location.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
LLVM_ABI const DebugLoc & getStableDebugLoc() const
Fetch the debug location for this node, unless this is a debug intrinsic, in which case fetch the deb...
LLVM_ABI void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
LLVM_ABI unsigned getNumSuccessors() const LLVM_READONLY
Return the number of successors that this instruction has.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
LLVM_ABI BasicBlock * getSuccessor(unsigned Idx) const LLVM_READONLY
Return the specified successor. This instruction must be a terminator.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI void moveBeforePreserving(InstListType::iterator MovePos)
Perform a moveBefore operation, while signalling that the caller intends to preserve the original ord...
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void insertAfter(Instruction *InsertPos)
Insert an unlinked instruction into a basic block immediately after the specified instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this load instruction.
Align getAlign() const
Return the alignment of the access that is being performed.
Analysis pass that exposes the LoopInfo for a function.
LLVM_ABI LoopInfo run(Function &F, FunctionAnalysisManager &AM)
ArrayRef< BlockT * > getBlocks() const
Get a list of the basic blocks which make up this loop.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
This class represents a loop nest and can be used to query its properties.
Represents a single loop in the control flow graph.
LLVM_ABI MDNode * createCallbackEncoding(unsigned CalleeArgNo, ArrayRef< int > Arguments, bool VarArgsArePassed)
Return metadata describing a callback (see llvm::AbstractCallSite).
LLVM_ABI void replaceOperandWith(unsigned I, Metadata *New)
Replace a specific operand.
static MDTuple * getDistinct(LLVMContext &Context, ArrayRef< Metadata * > MDs)
ArrayRef< MDOperand > operands() const
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
This class implements a map that also provides access to all stored values in a deterministic order.
A Module instance is used to store all the information related to an LLVM module.
LLVMContext & getContext() const
Get the global data context.
const DataLayout & getDataLayout() const
Get the data layout for the module's target platform.
iterator_range< op_iterator > operands()
LLVM_ABI void addOperand(MDNode *M)
Device global variable entries info.
Target region entries info.
Base class of the entries info.
Class that manages information about offload code regions and data.
function_ref< void(StringRef, const OffloadEntryInfoDeviceGlobalVar &)> OffloadDeviceGlobalVarEntryInfoActTy
Applies action Action on all registered entries.
OMPTargetDeviceClauseKind
Kind of device clause for declare target variables and functions NOTE: Currently not used as a part o...
@ OMPTargetDeviceClauseAny
The target is marked for all devices.
LLVM_ABI void registerDeviceGlobalVarEntryInfo(StringRef VarName, Constant *Addr, int64_t VarSize, OMPTargetGlobalVarEntryKind Flags, GlobalValue::LinkageTypes Linkage)
Register device global variable entry.
LLVM_ABI void initializeDeviceGlobalVarEntryInfo(StringRef Name, OMPTargetGlobalVarEntryKind Flags, unsigned Order)
Initialize device global variable entry.
LLVM_ABI void actOnDeviceGlobalVarEntriesInfo(const OffloadDeviceGlobalVarEntryInfoActTy &Action)
OMPTargetRegionEntryKind
Kind of the target registry entry.
@ OMPTargetRegionEntryTargetRegion
Mark the entry as target region.
LLVM_ABI void getTargetRegionEntryFnName(SmallVectorImpl< char > &Name, const TargetRegionEntryInfo &EntryInfo)
LLVM_ABI bool hasTargetRegionEntryInfo(TargetRegionEntryInfo EntryInfo, bool IgnoreAddressId=false) const
Return true if a target region entry with the provided information exists.
LLVM_ABI void registerTargetRegionEntryInfo(TargetRegionEntryInfo EntryInfo, Constant *Addr, Constant *ID, OMPTargetRegionEntryKind Flags)
Register target region entry.
LLVM_ABI void actOnTargetRegionEntriesInfo(const OffloadTargetRegionEntryInfoActTy &Action)
LLVM_ABI void initializeTargetRegionEntryInfo(const TargetRegionEntryInfo &EntryInfo, unsigned Order)
Initialize target region entry.
OMPTargetGlobalVarEntryKind
Kind of the global variable entry..
@ OMPTargetGlobalVarEntryEnter
Mark the entry as a declare target enter.
@ OMPTargetGlobalRegisterRequires
Mark the entry as a register requires global.
@ OMPTargetGlobalVarEntryIndirect
Mark the entry as a declare target indirect global.
@ OMPTargetGlobalVarEntryLink
Mark the entry as a to declare target link.
@ OMPTargetGlobalVarEntryTo
Mark the entry as a to declare target.
@ OMPTargetGlobalVarEntryIndirectVTable
Mark the entry as a declare target indirect vtable.
function_ref< void(const TargetRegionEntryInfo &EntryInfo, const OffloadEntryInfoTargetRegion &)> OffloadTargetRegionEntryInfoActTy
brief Applies action Action on all registered entries.
bool hasDeviceGlobalVarEntryInfo(StringRef VarName) const
Checks if the variable with the given name has been registered already.
LLVM_ABI bool empty() const
Return true if a there are no entries defined.
std::optional< bool > IsTargetDevice
Flag to define whether to generate code for the role of the OpenMP host (if set to false) or device (...
std::optional< bool > IsGPU
Flag for specifying if the compilation is done for an accelerator.
LLVM_ABI int64_t getRequiresFlags() const
Returns requires directive clauses as flags compatible with those expected by libomptarget.
std::optional< bool > OpenMPOffloadMandatory
Flag for specifying if offloading is mandatory.
LLVM_ABI void setHasRequiresReverseOffload(bool Value)
LLVM_ABI OpenMPIRBuilderConfig()
LLVM_ABI bool hasRequiresUnifiedSharedMemory() const
LLVM_ABI void setHasRequiresUnifiedSharedMemory(bool Value)
unsigned getDefaultTargetAS() const
LLVM_ABI bool hasRequiresDynamicAllocators() const
LLVM_ABI void setHasRequiresUnifiedAddress(bool Value)
bool isTargetDevice() const
LLVM_ABI void setHasRequiresDynamicAllocators(bool Value)
LLVM_ABI bool hasRequiresReverseOffload() const
bool hasRequiresFlags() const
LLVM_ABI bool hasRequiresUnifiedAddress() const
Struct that keeps the information that should be kept throughout a 'target data' region.
An interface to create LLVM-IR for OpenMP directives.
LLVM_ABI InsertPointOrErrorTy createOrderedThreadsSimd(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsThreads)
Generator for 'omp ordered [threads | simd]'.
LLVM_ABI void emitAArch64DeclareSimdFunction(llvm::Function *Fn, unsigned VLENVal, llvm::ArrayRef< DeclareSimdAttrTy > ParamAttrs, DeclareSimdBranch Branch, char ISA, unsigned NarrowestDataSize, bool OutputBecomesInput)
Emit AArch64 vector-function ABI attributes for a declare simd function.
LLVM_ABI Constant * getOrCreateIdent(Constant *SrcLocStr, uint32_t SrcLocStrSize, omp::IdentFlag Flags=omp::IdentFlag(0), unsigned Reserve2Flags=0)
Return an ident_t* encoding the source location SrcLocStr and Flags.
LLVM_ABI void registerDeclareTargetGlobalReplacement(GlobalValue *Original, GlobalValue *Replacement)
Register a module-scope replacement of a declare target global variable.
LLVM_ABI FunctionCallee getOrCreateRuntimeFunction(Module &M, omp::RuntimeFunction FnID)
Return the function declaration for the runtime function with FnID.
LLVM_ABI InsertPointOrErrorTy createCancel(const LocationDescription &Loc, Value *IfCondition, omp::Directive CanceledDirective)
Generator for 'omp cancel'.
std::function< Expected< Function * >(StringRef FunctionName)> FunctionGenCallback
Functions used to generate a function with the given name.
LLVM_ABI CallInst * createOMPAllocShared(const LocationDescription &Loc, Value *Size, const Twine &Name=Twine(""))
Create a runtime call for kmpc_alloc_shared.
ReductionGenCBKind
Enum class for the RedctionGen CallBack type to be used.
LLVM_ABI CanonicalLoopInfo * collapseLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops, InsertPointTy ComputeIP)
Collapse a loop nest into a single loop.
LLVM_ABI void createTaskyield(const LocationDescription &Loc)
Generator for 'omp taskyield'.
std::function< Error(InsertPointTy CodeGenIP)> FinalizeCallbackTy
Callback type for variable finalization (think destructors).
LLVM_ABI void emitBranch(BasicBlock *Target)
LLVM_ABI Error emitCancelationCheckImpl(Value *CancelFlag, omp::Directive CanceledDirective)
Generate control flow and cleanup for cancellation.
static LLVM_ABI void writeThreadBoundsForKernel(const Triple &T, Function &Kernel, int32_t LB, int32_t UB)
LLVM_ABI void emitTaskwaitImpl(const LocationDescription &Loc)
Generate a taskwait runtime call.
LLVM_ABI Constant * registerTargetRegionFunction(TargetRegionEntryInfo &EntryInfo, Function *OutlinedFunction, StringRef EntryFnName, StringRef EntryFnIDName)
Registers the given function and sets up the attribtues of the function Returns the FunctionID.
LLVM_ABI GlobalVariable * emitKernelExecutionMode(StringRef KernelName, omp::OMPTgtExecModeFlags Mode)
Emit the kernel execution mode.
LLVM_ABI void initialize()
Initialize the internal state, this will put structures types and potentially other helpers into the ...
LLVM_ABI InsertPointTy createAtomicCompare(const LocationDescription &Loc, AtomicOpValue &X, AtomicOpValue &V, AtomicOpValue &R, Value *E, Value *D, AtomicOrdering AO, omp::OMPAtomicCompareOp Op, bool IsXBinopExpr, bool IsPostfixUpdate, bool IsFailOnly, bool IsWeak=false)
LLVM_ABI InsertPointTy createAtomicWrite(const LocationDescription &Loc, AtomicOpValue &X, Value *Expr, AtomicOrdering AO, InsertPointTy AllocaIP)
Emit atomic write for : X = Expr — Only Scalar data types.
LLVM_ABI void loadOffloadInfoMetadata(Module &M)
Loads all the offload entries information from the host IR metadata.
function_ref< MapInfosTy &(InsertPointTy CodeGenIP)> GenMapInfoCallbackTy
Callback type for creating the map infos for the kernel parameters.
LLVM_ABI Error emitOffloadingArrays(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, MapInfosTy &CombinedInfo, TargetDataInfo &Info, CustomMapperCallbackTy CustomMapperCB, bool IsNonContiguous=false, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr)
Emit the arrays used to pass the captures and map information to the offloading runtime library.
LLVM_ABI void unrollLoopFull(DebugLoc DL, CanonicalLoopInfo *Loop)
Fully unroll a loop.
function_ref< Error(InsertPointTy CodeGenIP, Value *IndVar)> LoopBodyGenCallbackTy
Callback type for loop body code generation.
LLVM_ABI InsertPointOrErrorTy emitScanReduction(const LocationDescription &Loc, ArrayRef< llvm::OpenMPIRBuilder::ReductionInfo > ReductionInfos, ScanInfo *ScanRedInfo)
This function performs the scan reduction of the values updated in the input phase.
LLVM_ABI void emitFlush(const LocationDescription &Loc)
Generate a flush runtime call.
LLVM_ABI InsertPointOrErrorTy createScope(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsNowait)
Generator for 'omp scope'.
static LLVM_ABI std::pair< int32_t, int32_t > readThreadBoundsForKernel(const Triple &T, Function &Kernel)
}
OpenMPIRBuilderConfig Config
The OpenMPIRBuilder Configuration.
LLVM_ABI CallInst * createOMPInteropDestroy(const LocationDescription &Loc, Value *InteropVar, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_destroy.
LLVM_ABI void emitUsed(StringRef Name, ArrayRef< llvm::WeakTrackingVH > List)
Emit the llvm.used metadata.
LLVM_ABI InsertPointOrErrorTy createSingle(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsNowait, ArrayRef< llvm::Value * > CPVars={}, ArrayRef< llvm::Function * > CPFuncs={})
Generator for 'omp single'.
LLVM_ABI InsertPointOrErrorTy createTeams(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, Value *NumTeamsLower=nullptr, Value *NumTeamsUpper=nullptr, Value *ThreadLimit=nullptr, Value *IfExpr=nullptr)
Generator for #omp teams
std::forward_list< CanonicalLoopInfo > LoopInfos
Collection of owned canonical loop objects that eventually need to be free'd.
LLVM_ABI llvm::StructType * getKmpTaskAffinityInfoTy()
Return the LLVM struct type matching runtime kmp_task_affinity_info_t.
LLVM_ABI CanonicalLoopInfo * createLoopSkeleton(DebugLoc DL, Value *TripCount, Function *F, BasicBlock *PreInsertBefore, BasicBlock *PostInsertBefore, const Twine &Name={})
Create the control flow structure of a canonical OpenMP loop.
LLVM_ABI std::string createPlatformSpecificName(ArrayRef< StringRef > Parts) const
Get the create a name using the platform specific separators.
LLVM_ABI FunctionCallee createDispatchNextFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_next_* runtime function for the specified size IVSize and sign IVSigned.
static LLVM_ABI void getKernelArgsVector(TargetKernelArgs &KernelArgs, IRBuilderBase &Builder, SmallVector< Value * > &ArgsVector)
Create the kernel args vector used by emitTargetKernel.
LLVM_ABI InsertPointOrErrorTy createTarget(const LocationDescription &Loc, bool IsOffloadEntry, OpenMPIRBuilder::InsertPointTy AllocaIP, OpenMPIRBuilder::InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks, TargetDataInfo &Info, TargetRegionEntryInfo &EntryInfo, const TargetKernelDefaultAttrs &DefaultAttrs, const TargetKernelRuntimeAttrs &RuntimeAttrs, Value *IfCond, SmallVectorImpl< Value * > &Inputs, GenMapInfoCallbackTy GenMapInfoCB, TargetBodyGenCallbackTy BodyGenCB, TargetGenArgAccessorsCallbackTy ArgAccessorFuncCB, CustomMapperCallbackTy CustomMapperCB, const DependenciesInfo &Dependencies={}, bool HasNowait=false, Value *DynCGroupMem=nullptr, omp::OMPDynGroupprivateFallbackType DynCGroupMemFallback=omp::OMPDynGroupprivateFallbackType::Abort)
Generator for 'omp target'.
LLVM_ABI void unrollLoopHeuristic(DebugLoc DL, CanonicalLoopInfo *Loop)
Fully or partially unroll a loop.
LLVM_ABI omp::OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position)
Get OMP_MAP_MEMBER_OF flag with extra bits reserved based on the position given.
LLVM_ABI void addAttributes(omp::RuntimeFunction FnID, Function &Fn)
Add attributes known for FnID to Fn.
Module & M
The underlying LLVM-IR module.
StringMap< Constant * > SrcLocStrMap
Map to remember source location strings.
LLVM_ABI void createMapperAllocas(const LocationDescription &Loc, InsertPointTy AllocaIP, unsigned NumOperands, struct MapperAllocas &MapperAllocas)
Create the allocas instruction used in call to mapper functions.
SmallVector< DeclareTargetGlobalReplacement, 8 > DeclareTargetGlobalReplacements
Collection of declare target globals to rewrite uses of during device module finalizaiton.
LLVM_ABI Constant * getOrCreateSrcLocStr(StringRef LocStr, uint32_t &SrcLocStrSize)
Return the (LLVM-IR) string describing the source location LocStr.
LLVM_ABI Error emitTargetRegionFunction(TargetRegionEntryInfo &EntryInfo, FunctionGenCallback &GenerateFunctionCallback, bool IsOffloadEntry, Function *&OutlinedFn, Constant *&OutlinedFnID)
Create a unique name for the entry function using the source location information of the current targ...
LLVM_ABI InsertPointOrErrorTy createIteratorLoop(LocationDescription Loc, llvm::Value *TripCount, IteratorBodyGenTy BodyGen, llvm::StringRef Name="iterator")
Create a canonical iterator loop at the current insertion point.
LLVM_ABI Expected< SmallVector< llvm::CanonicalLoopInfo * > > createCanonicalScanLoops(const LocationDescription &Loc, LoopBodyGenCallbackTy BodyGenCB, Value *Start, Value *Stop, Value *Step, bool IsSigned, bool InclusiveStop, InsertPointTy ComputeIP, const Twine &Name, ScanInfo *ScanRedInfo)
Generator for the control flow structure of an OpenMP canonical loops if the parent directive has an ...
LLVM_ABI FunctionCallee createDispatchFiniFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_fini_* runtime function for the specified size IVSize and sign IVSigned.
function_ref< InsertPointOrErrorTy( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks)> TargetBodyGenCallbackTy
LLVM_ABI void unrollLoopPartial(DebugLoc DL, CanonicalLoopInfo *Loop, int32_t Factor, CanonicalLoopInfo **UnrolledCLI)
Partially unroll a loop.
function_ref< Error(Value *DeviceID, Value *RTLoc, IRBuilderBase::InsertPoint TargetTaskAllocaIP)> TargetTaskBodyCallbackTy
Callback type for generating the bodies of device directives that require outer target tasks (e....
Expected< MapInfosTy & > MapInfosOrErrorTy
bool HandleFPNegZero
Emit atomic compare for constructs: — Only scalar data types cond-expr-stmt: x = x ordop expr ?
LLVM_ABI void emitTaskyieldImpl(const LocationDescription &Loc)
Generate a taskyield runtime call.
LLVM_ABI void emitMapperCall(const LocationDescription &Loc, Function *MapperFunc, Value *SrcLocInfo, Value *MaptypesArg, Value *MapnamesArg, struct MapperAllocas &MapperAllocas, int64_t DeviceID, unsigned NumOperands)
Create the call for the target mapper function.
LLVM_ABI InsertPointOrErrorTy createDistribute(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB)
Generator for #omp distribute
LLVM_ABI Expected< Function * > emitUserDefinedMapper(function_ref< MapInfosOrErrorTy(InsertPointTy CodeGenIP, llvm::Value *PtrPHI, llvm::Value *BeginArg)> PrivAndGenMapInfoCB, llvm::Type *ElemTy, StringRef FuncName, CustomMapperCallbackTy CustomMapperCB, bool PreserveMemberOfFlags=false)
Emit the user-defined mapper function.
LLVM_ABI InsertPointOrErrorTy createTask(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB, bool Tied=true, Value *Final=nullptr, Value *IfCondition=nullptr, const DependenciesInfo &Dependencies={}, const AffinityData &Affinities={}, bool Mergeable=false, Value *EventHandle=nullptr, Value *Priority=nullptr)
Generator for #omp taskloop
function_ref< Expected< Function * >(unsigned int)> CustomMapperCallbackTy
LLVM_ABI InsertPointTy createOrderedDepend(const LocationDescription &Loc, InsertPointTy AllocaIP, unsigned NumLoops, ArrayRef< llvm::Value * > StoreValues, const Twine &Name, bool IsDependSource)
Generator for 'omp ordered depend (source | sink)'.
LLVM_ABI InsertPointTy createCopyinClauseBlocks(InsertPointTy IP, Value *MasterAddr, Value *PrivateAddr, llvm::IntegerType *IntPtrTy, bool BranchtoEnd=true)
Generate conditional branch and relevant BasicBlocks through which private threads copy the 'copyin' ...
function_ref< InsertPointOrErrorTy( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, Value &Original, Value &Inner, Value *&ReplVal)> PrivatizeCallbackTy
Callback type for variable privatization (think copy & default constructor).
LLVM_ABI bool isFinalized()
Check whether the finalize function has already run.
SmallVector< FinalizationInfo, 8 > FinalizationStack
The finalization stack made up of finalize callbacks currently in-flight, wrapped into FinalizationIn...
LLVM_ABI std::vector< CanonicalLoopInfo * > tileLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops, ArrayRef< Value * > TileSizes)
Tile a loop nest.
LLVM_ABI CallInst * createOMPInteropInit(const LocationDescription &Loc, Value *InteropVar, omp::OMPInteropType InteropType, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_init.
LLVM_ABI Error emitIfClause(Value *Cond, BodyGenCallbackTy ThenGen, BodyGenCallbackTy ElseGen, InsertPointTy AllocaIP={}, ArrayRef< BasicBlock * > DeallocBlocks={})
Emits code for OpenMP 'if' clause using specified BodyGenCallbackTy Here is the logic: if (Cond) { Th...
LLVM_ABI void finalize(Function *Fn=nullptr)
Finalize the underlying module, e.g., by outlining regions.
LLVM_ABI Function * getOrCreateRuntimeFunctionPtr(omp::RuntimeFunction FnID)
void addOutlineInfo(std::unique_ptr< OutlineInfo > &&OI)
Add a new region that will be outlined later.
LLVM_ABI InsertPointTy createTargetInit(const LocationDescription &Loc, const llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs &Attrs)
The omp target interface.
LLVM_ABI InsertPointOrErrorTy createReductions(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< ReductionInfo > ReductionInfos, ArrayRef< bool > IsByRef, bool IsNoWait=false, bool IsTeamsReduction=false)
Generator for 'omp reduction'.
const Triple T
The target triple of the underlying module.
DenseMap< std::pair< Constant *, uint64_t >, Constant * > IdentMap
Map to remember existing ident_t*.
LLVM_ABI CallInst * createOMPFree(const LocationDescription &Loc, Value *Addr, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_free.
LLVM_ABI InsertPointOrErrorTy createReductionsGPU(const LocationDescription &Loc, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< ReductionInfo > ReductionInfos, ArrayRef< bool > IsByRef, bool IsNoWait=false, bool IsTeamsReduction=false, bool IsSPMD=false, ReductionGenCBKind ReductionGenCBKind=ReductionGenCBKind::MLIR, std::optional< omp::GV > GridValue={}, Value *SrcLocInfo=nullptr)
Design of OpenMP reductions on the GPU.
LLVM_ABI FunctionCallee createForStaticInitFunction(unsigned IVSize, bool IVSigned, bool IsGPUDistribute)
Returns __kmpc_for_static_init_* runtime function for the specified size IVSize and sign IVSigned.
LLVM_ABI CallInst * createOMPAlloc(const LocationDescription &Loc, Value *Size, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_alloc.
LLVM_ABI void emitNonContiguousDescriptor(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, MapInfosTy &CombinedInfo, TargetDataInfo &Info)
Emit an array of struct descriptors to be assigned to the offload args.
LLVM_ABI InsertPointOrErrorTy createSection(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB)
Generator for 'omp section'.
LLVM_ABI InsertPointOrErrorTy createTaskgroup(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB)
Generator for the taskgroup construct.
LLVM_ABI InsertPointOrErrorTy createParallel(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB, PrivatizeCallbackTy PrivCB, FinalizeCallbackTy FiniCB, Value *IfCondition, Value *NumThreads, omp::ProcBindKind ProcBind, bool IsCancellable)
Generator for 'omp parallel'.
function_ref< InsertPointOrErrorTy(InsertPointTy)> EmitFallbackCallbackTy
Callback function type for functions emitting the host fallback code that is executed when the kernel...
static LLVM_ABI TargetRegionEntryInfo getTargetEntryUniqueInfo(FileIdentifierInfoCallbackTy CallBack, vfs::FileSystem &VFS, StringRef ParentName="")
Creates a unique info for a target entry when provided a filename and line number from.
LLVM_ABI void emitTaskDependency(IRBuilderBase &Builder, Value *Entry, const DependData &Dep)
Store one kmp_depend_info entry at the given Entry pointer.
LLVM_ABI void emitBlock(BasicBlock *BB, Function *CurFn, bool IsFinished=false)
LLVM_ABI Value * getOrCreateThreadID(Value *Ident)
Return the current thread ID.
LLVM_ABI InsertPointOrErrorTy createMaster(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB)
Generator for 'omp master'.
LLVM_ABI InsertPointOrErrorTy createTargetData(const LocationDescription &Loc, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks, Value *DeviceID, Value *IfCond, TargetDataInfo &Info, GenMapInfoCallbackTy GenMapInfoCB, CustomMapperCallbackTy CustomMapperCB, omp::RuntimeFunction *MapperFunc=nullptr, function_ref< InsertPointOrErrorTy(InsertPointTy CodeGenIP, BodyGenTy BodyGenType)> BodyGenCB=nullptr, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr, Value *SrcLocInfo=nullptr)
Generator for 'omp target data'.
LLVM_ABI CallInst * createRuntimeFunctionCall(FunctionCallee Callee, ArrayRef< Value * > Args, StringRef Name="")
LLVM_ABI InsertPointOrErrorTy emitKernelLaunch(const LocationDescription &Loc, Value *OutlinedFnID, EmitFallbackCallbackTy EmitTargetCallFallbackCB, TargetKernelArgs &Args, Value *DeviceID, Value *RTLoc, InsertPointTy AllocaIP)
Generate a target region entry call and host fallback call.
StringMap< GlobalVariable *, BumpPtrAllocator > InternalVars
An ordered map of auto-generated variables to their unique names.
LLVM_ABI InsertPointOrErrorTy createCancellationPoint(const LocationDescription &Loc, omp::Directive CanceledDirective)
Generator for 'omp cancellation point'.
LLVM_ABI CallInst * createOMPAlignedAlloc(const LocationDescription &Loc, Value *Align, Value *Size, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_align_alloc.
LLVM_ABI FunctionCallee createDispatchInitFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_init_* runtime function for the specified size IVSize and sign IVSigned.
LLVM_ABI InsertPointOrErrorTy createScan(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< llvm::Value * > ScanVars, ArrayRef< llvm::Type * > ScanVarsType, bool IsInclusive, ScanInfo *ScanRedInfo)
This directive split and directs the control flow to input phase blocks or scan phase blocks based on...
LLVM_ABI CallInst * createOMPFreeShared(const LocationDescription &Loc, Value *Addr, Value *Size, const Twine &Name=Twine(""))
Create a runtime call for kmpc_free_shared.
LLVM_ABI CallInst * createOMPInteropUse(const LocationDescription &Loc, Value *InteropVar, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_use.
IRBuilder<>::InsertPoint InsertPointTy
Type used throughout for insertion points.
LLVM_ABI GlobalVariable * getOrCreateInternalVariable(Type *Ty, const StringRef &Name, std::optional< unsigned > AddressSpace={})
Gets (if variable with the given name already exist) or creates internal global variable with the spe...
LLVM_ABI GlobalVariable * createOffloadMapnames(SmallVectorImpl< llvm::Constant * > &Names, std::string VarName)
Create the global variable holding the offload names information.
std::forward_list< ScanInfo > ScanInfos
Collection of owned ScanInfo objects that eventually need to be free'd.
static LLVM_ABI void writeTeamsForKernel(const Triple &T, Function &Kernel, int32_t LB, int32_t UB)
LLVM_ABI Value * calculateCanonicalLoopTripCount(const LocationDescription &Loc, Value *Start, Value *Stop, Value *Step, bool IsSigned, bool InclusiveStop, const Twine &Name="loop")
Calculate the trip count of a canonical loop.
LLVM_ABI InsertPointOrErrorTy createBarrier(const LocationDescription &Loc, omp::Directive Kind, bool ForceSimpleCall=false, bool CheckCancelFlag=true)
Emitter methods for OpenMP directives.
LLVM_ABI void setCorrectMemberOfFlag(omp::OpenMPOffloadMappingFlags &Flags, omp::OpenMPOffloadMappingFlags MemberOfFlag)
Given an initial flag set, this function modifies it to contain the passed in MemberOfFlag generated ...
LLVM_ABI Error emitOffloadingArraysAndArgs(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, TargetDataInfo &Info, TargetDataRTArgs &RTArgs, MapInfosTy &CombinedInfo, CustomMapperCallbackTy CustomMapperCB, bool IsNonContiguous=false, bool ForEndCall=false, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr)
Allocates memory for and populates the arrays required for offloading (offload_{baseptrs|ptrs|mappers...
LLVM_ABI Constant * getOrCreateDefaultSrcLocStr(uint32_t &SrcLocStrSize)
Return the (LLVM-IR) string describing the default source location.
LLVM_ABI InsertPointOrErrorTy createCritical(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, StringRef CriticalName, Value *HintInst)
Generator for 'omp critical'.
LLVM_ABI void createError(const LocationDescription &Loc, bool IsFatal, Value *Message)
Generate a call to the runtime to emit the diagnostic of an OpenMP error directive with at(execution)...
LLVM_ABI void createOffloadEntry(Constant *ID, Constant *Addr, uint64_t Size, int32_t Flags, GlobalValue::LinkageTypes, StringRef Name="")
Creates offloading entry for the provided entry ID ID, address Addr, size Size, and flags Flags.
static LLVM_ABI unsigned getOpenMPDefaultSimdAlign(const Triple &TargetTriple, const StringMap< bool > &Features)
Get the default alignment value for given target.
LLVM_ABI unsigned getFlagMemberOffset()
Get the offset of the OMP_MAP_MEMBER_OF field.
LLVM_ABI InsertPointOrErrorTy applyWorkshareLoop(DebugLoc DL, CanonicalLoopInfo *CLI, InsertPointTy AllocaIP, bool NeedsBarrier, llvm::omp::ScheduleKind SchedKind=llvm::omp::OMP_SCHEDULE_Default, Value *ChunkSize=nullptr, bool HasSimdModifier=false, bool HasMonotonicModifier=false, bool HasNonmonotonicModifier=false, bool HasOrderedClause=false, omp::WorksharingLoopType LoopType=omp::WorksharingLoopType::ForStaticLoop, bool NoLoop=false, bool HasDistSchedule=false, Value *DistScheduleChunkSize=nullptr)
Modifies the canonical loop to be a workshare loop.
LLVM_ABI InsertPointOrErrorTy createAtomicCapture(const LocationDescription &Loc, InsertPointTy AllocaIP, AtomicOpValue &X, AtomicOpValue &V, Value *Expr, AtomicOrdering AO, AtomicRMWInst::BinOp RMWOp, AtomicUpdateCallbackTy &UpdateOp, bool UpdateExpr, bool IsPostfixUpdate, bool IsXBinopExpr, bool IsIgnoreDenormalMode=false, bool IsFineGrainedMemory=false, bool IsRemoteMemory=false)
Emit atomic update for constructs: — Only Scalar data types V = X; X = X BinOp Expr ,...
LLVM_ABI void createOffloadEntriesAndInfoMetadata(EmitMetadataErrorReportFunctionTy &ErrorReportFunction)
LLVM_ABI void applySimd(CanonicalLoopInfo *Loop, MapVector< Value *, Value * > AlignedVars, Value *IfCond, omp::OrderKind Order, ConstantInt *Simdlen, ConstantInt *Safelen)
Add metadata to simd-ize a loop.
SmallVector< std::unique_ptr< OutlineInfo >, 16 > OutlineInfos
Collection of regions that need to be outlined during finalization.
LLVM_ABI InsertPointOrErrorTy createAtomicUpdate(const LocationDescription &Loc, InsertPointTy AllocaIP, AtomicOpValue &X, Value *Expr, AtomicOrdering AO, AtomicRMWInst::BinOp RMWOp, AtomicUpdateCallbackTy &UpdateOp, bool IsXBinopExpr, bool IsIgnoreDenormalMode=false, bool IsFineGrainedMemory=false, bool IsRemoteMemory=false)
Emit atomic update for constructs: X = X BinOp Expr ,or X = Expr BinOp X For complex Operations: X = ...
std::function< std::tuple< std::string, uint64_t >()> FileIdentifierInfoCallbackTy
bool isLastFinalizationInfoCancellable(omp::Directive DK)
Return true if the last entry in the finalization stack is of kind DK and cancellable.
LLVM_ABI InsertPointTy emitTargetKernel(const LocationDescription &Loc, InsertPointTy AllocaIP, Value *&Return, Value *Ident, Value *DeviceID, Value *NumTeams, Value *NumThreads, Value *HostPtr, ArrayRef< Value * > KernelArgs)
Generate a target region entry call.
LLVM_ABI GlobalVariable * createOffloadMaptypes(SmallVectorImpl< uint64_t > &Mappings, std::string VarName)
Create the global variable holding the offload mappings information.
LLVM_ABI ~OpenMPIRBuilder()
LLVM_ABI CallInst * createCachedThreadPrivate(const LocationDescription &Loc, llvm::Value *Pointer, llvm::ConstantInt *Size, const llvm::Twine &Name=Twine(""))
Create a runtime call for kmpc_threadprivate_cached.
IRBuilder Builder
The LLVM-IR Builder used to create IR.
LLVM_ABI GlobalValue * createGlobalFlag(unsigned Value, StringRef Name)
Create a hidden global flag Name in the module with initial value Value.
LLVM_ABI void emitOffloadingArraysArgument(IRBuilderBase &Builder, OpenMPIRBuilder::TargetDataRTArgs &RTArgs, OpenMPIRBuilder::TargetDataInfo &Info, bool ForEndCall=false)
Emit the arguments to be passed to the runtime library based on the arrays of base pointers,...
LLVM_ABI InsertPointOrErrorTy createMasked(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, Value *Filter)
Generator for 'omp masked'.
LLVM_ABI Expected< CanonicalLoopInfo * > createCanonicalLoop(const LocationDescription &Loc, LoopBodyGenCallbackTy BodyGenCB, Value *TripCount, const Twine &Name="loop")
Generator for the control flow structure of an OpenMP canonical loop.
function_ref< Expected< InsertPointTy >( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, Value *DestPtr, Value *SrcPtr)> TaskDupCallbackTy
Callback type for task duplication function code generation.
LLVM_ABI Value * getSizeInBytes(Value *BasePtr)
Computes the size of type in bytes.
llvm::function_ref< llvm::Error( InsertPointTy BodyIP, llvm::Value *LinearIV)> IteratorBodyGenTy
LLVM_ABI FunctionCallee createDispatchDeinitFunction()
Returns __kmpc_dispatch_deinit runtime function.
LLVM_ABI void registerTargetGlobalVariable(OffloadEntriesInfoManager::OMPTargetGlobalVarEntryKind CaptureClause, OffloadEntriesInfoManager::OMPTargetDeviceClauseKind DeviceClause, bool IsDeclaration, bool IsExternallyVisible, TargetRegionEntryInfo EntryInfo, StringRef MangledName, std::vector< GlobalVariable * > &GeneratedRefs, bool OpenMPSIMD, std::vector< Triple > TargetTriple, std::function< Constant *()> GlobalInitializer, std::function< GlobalValue::LinkageTypes()> VariableLinkage, Type *LlvmPtrTy, Constant *Addr)
Registers a target variable for device or host.
LLVM_ABI void createTargetDeinit(const LocationDescription &Loc, int32_t TeamsReductionDataSize=0)
Create a runtime call for kmpc_target_deinit.
BodyGenTy
Type of BodyGen to use for region codegen.
LLVM_ABI CanonicalLoopInfo * fuseLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops)
Fuse a sequence of loops.
LLVM_ABI void emitX86DeclareSimdFunction(llvm::Function *Fn, unsigned NumElements, const llvm::APSInt &VLENVal, llvm::ArrayRef< DeclareSimdAttrTy > ParamAttrs, DeclareSimdBranch Branch)
Emit x86 vector-function ABI attributes for a declare simd function.
SmallVector< llvm::Function *, 16 > ConstantAllocaRaiseCandidates
A collection of candidate target functions that's constant allocas will attempt to be raised on a cal...
OffloadEntriesInfoManager OffloadInfoManager
Info manager to keep track of target regions.
static LLVM_ABI std::pair< int32_t, int32_t > readTeamBoundsForKernel(const Triple &T, Function &Kernel)
Read/write a bounds on teams for Kernel.
const std::string ompOffloadInfoName
OMP Offload Info Metadata name string.
Expected< InsertPointTy > InsertPointOrErrorTy
Type used to represent an insertion point or an error value.
LLVM_ABI InsertPointTy createCopyPrivate(const LocationDescription &Loc, llvm::Value *BufSize, llvm::Value *CpyBuf, llvm::Value *CpyFn, llvm::Value *DidIt)
Generator for __kmpc_copyprivate.
LLVM_ABI InsertPointOrErrorTy createSections(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< StorableBodyGenCallbackTy > SectionCBs, PrivatizeCallbackTy PrivCB, FinalizeCallbackTy FiniCB, bool IsCancellable, bool IsNowait)
Generator for 'omp sections'.
std::function< void(EmitMetadataErrorKind, TargetRegionEntryInfo)> EmitMetadataErrorReportFunctionTy
Callback function type.
function_ref< InsertPointOrErrorTy( Argument &Arg, Value *Input, Value *&RetVal, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< InsertPointTy > DeallocIPs)> TargetGenArgAccessorsCallbackTy
LLVM_ABI Expected< ScanInfo * > scanInfoInitialize()
Creates a ScanInfo object, allocates and returns the pointer.
LLVM_ABI InsertPointOrErrorTy emitTargetTask(TargetTaskBodyCallbackTy TaskBodyCB, Value *DeviceID, Value *RTLoc, OpenMPIRBuilder::InsertPointTy AllocaIP, const DependenciesInfo &Dependencies, const TargetDataRTArgs &RTArgs, bool HasNoWait)
Generate a target-task for the target construct.
LLVM_ABI InsertPointTy createAtomicRead(const LocationDescription &Loc, AtomicOpValue &X, AtomicOpValue &V, AtomicOrdering AO, InsertPointTy AllocaIP)
Emit atomic Read for : V = X — Only Scalar data types.
function_ref< Error(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks)> BodyGenCallbackTy
Callback type for body (=inner region) code generation.
bool updateToLocation(const LocationDescription &Loc)
Update the internal location to Loc.
LLVM_ABI void createFlush(const LocationDescription &Loc)
Generator for 'omp flush'.
LLVM_ABI void createTaskwait(const LocationDescription &Loc, DependenciesInfo Dependencies={})
Generator for 'omp taskwait'.
LLVM_ABI Constant * getAddrOfDeclareTargetVar(OffloadEntriesInfoManager::OMPTargetGlobalVarEntryKind CaptureClause, OffloadEntriesInfoManager::OMPTargetDeviceClauseKind DeviceClause, bool IsDeclaration, bool IsExternallyVisible, TargetRegionEntryInfo EntryInfo, StringRef MangledName, std::vector< GlobalVariable * > &GeneratedRefs, bool OpenMPSIMD, std::vector< Triple > TargetTriple, Type *LlvmPtrTy, std::function< Constant *()> GlobalInitializer, std::function< GlobalValue::LinkageTypes()> VariableLinkage)
Retrieve (or create if non-existent) the address of a declare target variable, used in conjunction wi...
origPtr *with the address space normalization required by the runtime entry point *The NULL descriptor makes the runtime walk the enclosing taskgroups to *find the matching task_reduction registration for the item The lookups *are emitted at p Loc
EmitMetadataErrorKind
The kind of errors that can occur when emitting the offload entries and metadata.
@ EMIT_MD_DECLARE_TARGET_ERROR
@ EMIT_MD_GLOBAL_VAR_INDIRECT_ERROR
@ EMIT_MD_GLOBAL_VAR_LINK_ERROR
@ EMIT_MD_TARGET_REGION_ERROR
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
Pseudo-analysis pass that exposes the PassInstrumentation to pass managers.
Class to represent pointers.
static PointerType * getUnqual(LLVMContext &C)
This constructs an opaque pointer to an object in the default address space (address space zero).
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
PostDominatorTree Class - Concrete subclass of DominatorTree that is used to compute the post-dominat...
Analysis pass that exposes the ScalarEvolution for a function.
LLVM_ABI ScalarEvolution run(Function &F, FunctionAnalysisManager &AM)
The main scalar evolution driver.
ScanInfo holds the information to assist in lowering of Scan reduction.
llvm::SmallDenseMap< llvm::Value *, llvm::Value * > * ScanBuffPtrs
Maps the private reduction variable to the pointer of the temporary buffer.
llvm::BasicBlock * OMPScanLoopExit
Exit block of loop body.
llvm::Value * IV
Keeps track of value of iteration variable for input/scan loop to be used for Scan directive lowering...
llvm::BasicBlock * OMPAfterScanBlock
Dominates the body of the loop before scan directive.
llvm::BasicBlock * OMPScanInit
Block before loop body where scan initializations are done.
llvm::BasicBlock * OMPBeforeScanBlock
Dominates the body of the loop before scan directive.
llvm::BasicBlock * OMPScanFinish
Block after loop body where scan finalizations are done.
llvm::Value * Span
Stores the span of canonical loop being lowered to be used for temporary buffer allocation or Finaliz...
bool OMPFirstScanLoop
If true, it indicates Input phase is lowered; else it indicates ScanPhase is lowered.
llvm::BasicBlock * OMPScanDispatch
Controls the flow to before or after scan blocks.
A vector that has set insertion semantics.
bool remove_if(UnaryPredicate P)
Remove items from the set vector based on a predicate function.
bool empty() const
Determine if the SetVector is empty or not.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool all() const
Returns true if all bits are set.
bool any() const
Returns true if any bit is set.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
void append(StringRef RHS)
Append from a StringRef.
StringRef str() const
Explicit conversion to StringRef.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setAlignment(Align Align)
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this store instruction.
StringMap - This is an unconventional map that is specialized for handling keys that are "strings",...
ValueTy lookup(StringRef Key) const
lookup - Return the entry for the specified key, or a default constructed value if no such entry exis...
Represent a constant reference to a string, i.e.
std::string str() const
Get the contents as an std::string.
constexpr bool empty() const
Check if the string is empty.
constexpr size_t size() const
Get the string size.
size_t count(char C) const
Return the number of occurrences of C in the string.
bool ends_with(StringRef Suffix) const
Check if this string ends with the given Suffix.
StringRef drop_back(size_t N=1) const
Return a StringRef equal to 'this' but with the last N elements dropped.
Class to represent struct types.
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
static LLVM_ABI StructType * create(LLVMContext &Context, StringRef Name)
This creates an identified struct.
Type * getElementType(unsigned N) const
LLVM_ABI void addCase(ConstantInt *OnVal, BasicBlock *Dest)
Add an entry to the switch instruction.
Analysis pass providing the TargetTransformInfo.
LLVM_ABI Result run(const Function &F, FunctionAnalysisManager &)
TargetTransformInfo Result
Analysis pass providing the TargetLibraryInfo.
Target - Wrapper for Target specific information.
TargetMachine * createTargetMachine(const Triple &TT, StringRef CPU, StringRef Features, const TargetOptions &Options, std::optional< Reloc::Model > RM, std::optional< CodeModel::Model > CM=std::nullopt, CodeGenOptLevel OL=CodeGenOptLevel::Default, bool JIT=false) const
createTargetMachine - Create a target specific machine implementation for the specified Triple.
Triple - Helper class for working with autoconf configuration names.
bool isPPC() const
Tests whether the target is PowerPC (32- or 64-bit LE or BE).
bool isX86() const
Tests whether the target is x86 (32- or 64-bit).
bool isWasm() const
Tests whether the target is wasm (32- and 64-bit).
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
LLVM_ABI unsigned getIntegerBitWidth() const
LLVM_ABI Type * getStructElementType(unsigned N) const
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
bool isStructTy() const
True if this is an instance of StructType.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
bool isVoidTy() const
Return true if this is 'void'.
Unconditional Branch instruction.
static UncondBrInst * Create(BasicBlock *Target, InsertPosition InsertBefore=nullptr)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
This function has undefined behavior.
Produce an estimate of the unrolled cost of the specified loop.
LLVM_ABI bool canUnroll(OptimizationRemarkEmitter *ORE=nullptr, const Loop *L=nullptr) const
Whether it is legal to unroll this loop.
uint64_t getRolledLoopSize() const
A Use represents the edge between a Value definition and its users.
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
user_iterator user_begin()
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
iterator_range< user_iterator > users()
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI bool hasNUses(unsigned N) const
Return true if this Value has exactly N uses.
LLVM_ABI User * getUniqueUndroppableUser()
Return true if there is exactly one unique user of this value that cannot be dropped (that user can h...
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
LLVM_ABI bool replaceUsesWithIf(Value *New, llvm::function_ref< bool(Use &U)> ShouldReplace)
Go through the uses list for this definition and make each use point to "V" if the callback ShouldRep...
iterator_range< use_iterator > uses()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
An efficient, type-erasing, non-owning reference to a callable.
const ParentTy * getParent() const
self_iterator getIterator()
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
A raw_ostream that writes to an SmallVector or SmallString.
StringRef str() const
Return a StringRef for the vector contents.
The virtual file system interface.
llvm::ErrorOr< std::unique_ptr< llvm::MemoryBuffer > > getBufferForFile(const Twine &Name, int64_t FileSize=-1, bool RequiresNullTerminator=true, bool IsVolatile=false, bool IsText=true)
This is a convenience method that opens a file, gets its content and then closes the file.
virtual llvm::ErrorOr< Status > status(const Twine &Path)=0
Get the status of the entry at Path, if one exists.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ PTX_Kernel
Call to a PTX kernel. Passes all arguments in parameter space.
@ BasicBlock
Various leaf nodes.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
Flag
These should be considered private to the implementation of the MCInstrDesc class.
constexpr StringLiteral MaxNTID("nvvm.maxntid")
constexpr StringLiteral MaxClusterRank("nvvm.maxclusterrank")
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
LLVM_ABI GlobalVariable * emitOffloadingEntry(Module &M, object::OffloadKind Kind, Constant *Addr, StringRef Name, uint64_t Size, uint32_t Flags, uint64_t Data, Constant *AuxAddr=nullptr)
OpenMPOffloadMappingFlags
Values for bit flags used to specify the mapping type for offloading.
@ OMP_MAP_PTR_AND_OBJ
The element being mapped is a pointer-pointee pair; both the pointer and the pointee should be mapped...
@ OMP_MAP_MEMBER_OF
The 16 MSBs of the flags indicate whether the entry is member of some struct/class.
IdentFlag
IDs for all omp runtime library ident_t flag encodings (see their defintion in openmp/runtime/src/kmp...
RuntimeFunction
IDs for all omp runtime library (RTL) functions.
constexpr const GV & getAMDGPUGridValues()
static constexpr GV SPIRVGridValues
For generic SPIR-V GPUs.
OMPDynGroupprivateFallbackType
The fallback types for the dyn_groupprivate clause.
static constexpr GV NVPTXGridValues
For Nvidia GPUs.
@ OMP_TGT_EXEC_MODE_SPMD_NO_LOOP
@ OMP_TGT_EXEC_MODE_GENERIC
Function * Kernel
Summary of a kernel (=entry point for target offloading).
WorksharingLoopType
A type of worksharing loop construct.
OMPAtomicCompareOp
Atomic compare operations. Currently OpenMP only supports ==, >, and <.
NodeAddr< PhiNode * > Phi
friend class Instruction
Iterator for Instructions in a `BasicBlock.
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
LLVM_ABI BasicBlock * splitBBWithSuffix(IRBuilderBase &Builder, bool CreateBranch, llvm::Twine Suffix=".split")
Like splitBB, but reuses the current block's name for the new name.
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
LLVM_ABI unsigned computeUnrollCount(Loop *L, const TargetTransformInfo &TTI, DominatorTree &DT, LoopInfo *LI, AssumptionCache *AC, ScalarEvolution &SE, const SmallPtrSetImpl< const Value * > &EphValues, OptimizationRemarkEmitter *ORE, unsigned TripCount, unsigned MaxTripCount, bool MaxOrZero, unsigned TripMultiple, const UnrollCostEstimator &UCE, TargetTransformInfo::UnrollingPreferences &UP, TargetTransformInfo::PeelingPreferences &PP)
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
hash_code hash_value(const FixedPointSemantics &Val)
LLVM_ABI Expected< std::unique_ptr< Module > > parseBitcodeFile(MemoryBufferRef Buffer, LLVMContext &Context, ParserCallbacks Callbacks={})
Read the specified bitcode file, returning the module.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
@ LLVM_MARK_AS_BITMASK_ENUM
LLVM_ABI BasicBlock * CloneBasicBlock(const BasicBlock *BB, ValueToValueMapTy &VMap, const Twine &NameSuffix="", Function *F=nullptr, ClonedCodeInfo *CodeInfo=nullptr, bool MapAtoms=true)
Return a copy of the specified basic block, but without embedding the block into a particular functio...
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
unsigned getPointerAddressSpace(const Type *T)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
auto successors(const MachineBasicBlock *BB)
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
LLVM_ABI std::error_code inconvertibleErrorCode()
The value returned by this function can be returned from convertToErrorCode for Error values where no...
testing::Matcher< const detail::ErrorHolder & > Failed()
constexpr from_range_t from_range
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE()
LLVM_ABI BasicBlock * splitBB(IRBuilderBase::InsertPoint IP, bool CreateBranch, DebugLoc DL, llvm::Twine Name={})
Split a BasicBlock at an InsertPoint, even if the block is degenerate (missing the terminator).
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
LLVM_ABI TargetTransformInfo::UnrollingPreferences gatherUnrollingPreferences(Loop *L, ScalarEvolution &SE, const TargetTransformInfo &TTI, BlockFrequencyInfo *BFI, ProfileSummaryInfo *PSI, llvm::OptimizationRemarkEmitter &ORE, int OptLevel, std::optional< unsigned > UserThreshold, std::optional< bool > UserAllowPartial, std::optional< bool > UserRuntime, std::optional< bool > UserUpperBound, std::optional< unsigned > UserFullUnrollMaxCount)
Gather the various unrolling parameters based on the defaults, compiler flags, TTI overrides and user...
std::string utostr(uint64_t X, bool isNeg=false)
ErrorOr< T > expectedToErrorOrAndEmitErrors(LLVMContext &Ctx, Expected< T > Val)
bool isa_and_nonnull(const Y &Val)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
auto dyn_cast_or_null(const Y &Val)
LLVM_ABI bool convertUsersOfConstantsToInstructions(ArrayRef< Constant * > Consts, Function *RestrictToFunc=nullptr, bool RemoveDeadConstants=true, bool IncludeSelf=false)
Replace constant expressions users of the given constants with instructions.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
auto reverse(ContainerTy &&C)
LLVM_ABI TargetTransformInfo::PeelingPreferences gatherPeelingPreferences(Loop *L, ScalarEvolution &SE, const TargetTransformInfo &TTI, std::optional< bool > UserAllowPeeling, std::optional< bool > UserAllowProfileBasedPeeling, bool UnrollingSpecficValues=false)
LLVM_ABI void SplitBlockAndInsertIfThenElse(Value *Cond, BasicBlock::iterator SplitBefore, Instruction **ThenTerm, Instruction **ElseTerm, MDNode *BranchWeights=nullptr, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr)
SplitBlockAndInsertIfThenElse is similar to SplitBlockAndInsertIfThen, but also creates the ElseBlock...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
CodeGenOptLevel
Code generation optimization level.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
format_object< Ts... > format(const char *Fmt, const Ts &... Vals)
These are helper functions used to produce formatted output.
Error make_error(ArgTs &&... Args)
Make a Error instance representing failure using the given error info type.
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
void cantFail(Error Err, const char *Msg=nullptr)
Report a fatal error if Err is a failure value.
LLVM_ABI bool MergeBlockIntoPredecessor(BasicBlock *BB, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, MemoryDependenceResults *MemDep=nullptr, bool PredecessorWithTwoSuccessors=false, DominatorTree *DT=nullptr)
Attempts to merge a block into its predecessor, if possible.
@ Mul
Product of integers.
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
DWARFExpression::Operation Op
LLVM_ABI void remapInstructionsInBlocks(ArrayRef< BasicBlock * > Blocks, ValueToValueMapTy &VMap)
Remaps instructions in Blocks using the mapping in VMap.
ArrayRef(const T &OneElt) -> ArrayRef< T >
OutputIt copy(R &&Range, OutputIt Out)
ValueMap< const Value *, WeakTrackingVH > ValueToValueMapTy
LLVM_ABI void spliceBB(IRBuilderBase::InsertPoint IP, BasicBlock *New, bool CreateBranch, DebugLoc DL)
Move the instruction after an InsertPoint to the beginning of another BasicBlock.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
auto predecessors(const MachineBasicBlock *BB)
auto filter_to_vector(ContainerTy &&C, PredicateFn &&Pred)
Filter a range to a SmallVector with the element types deduced.
PointerUnion< const Value *, const PseudoSourceValue * > ValueType
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
LLVM_ABI Constant * ConstantFoldInsertValueInstruction(Constant *Agg, Constant *Val, ArrayRef< unsigned > Idxs)
Attempt to constant fold an insertvalue instruction with the specified operands and indices.
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI void DeleteDeadBlocks(ArrayRef< BasicBlock * > BBs, DomTreeUpdater *DTU=nullptr, bool KeepOneInputPHIs=false)
Delete the specified blocks from BB.
bool to_integer(StringRef S, N &Num, unsigned Base=0)
Convert the string S to an integer of the specified type using the radix Base. If Base is 0,...
static auto filterDbgVars(iterator_range< simple_ilist< DbgRecord >::iterator > R)
Filter the DbgRecord range to DbgVariableRecord types only and downcast.
This struct is a compact representation of a valid (non-zero power of two) alignment.
static LLVM_ABI void collectEphemeralValues(const Loop *L, AssumptionCache *AC, SmallPtrSetImpl< const Value * > &EphValues)
Collect a loop's ephemeral values (those used only by an assume or similar intrinsics in the loop).
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
A struct to pack the relevant information for an OpenMP affinity clause.
a struct to pack relevant information while generating atomic Ops
A struct to pack the relevant information for an OpenMP depend clause.
omp::RTLDependenceKindTy DepKind
A struct to pack static and dynamic dependency information for a task.
SmallVector< DependData > Deps
LLVM_ABI Error mergeFiniBB(IRBuilderBase &Builder, BasicBlock *ExistingFiniBB)
For cases where there is an unavoidable existing finalization block (e.g.
LLVM_ABI Expected< BasicBlock * > getFiniBB(IRBuilderBase &Builder)
The basic block to which control should be transferred to implement the FiniCB.
Description of a LLVM-IR insertion point (IP) and a debug/source location (filename,...
MapNonContiguousArrayTy Offsets
MapNonContiguousArrayTy Counts
MapNonContiguousArrayTy Strides
This structure contains combined information generated for mappable clauses, including base pointers,...
MapDeviceInfoArrayTy DevicePointers
MapValuesArrayTy BasePointers
MapValuesArrayTy Pointers
StructNonContiguousInfo NonContigInfo
Helper that contains information about regions we need to outline during finalization.
void collectBlocks(SmallPtrSetImpl< BasicBlock * > &BlockSet, SmallVectorImpl< BasicBlock * > &BlockVector)
Collect all blocks in between EntryBB and ExitBB in both the given vector and set.
BasicBlock * OuterAllocBB
virtual std::unique_ptr< CodeExtractor > createCodeExtractor(ArrayRef< BasicBlock * > Blocks, bool ArgsInZeroAddressSpace, Twine Suffix=Twine(""))
Create a CodeExtractor instance based on the information stored in this structure,...
Information about an OpenMP reduction.
EvalKind EvaluationKind
Reduction evaluation kind - scalar, complex or aggregate.
ReductionGenAtomicCBTy AtomicReductionGen
Callback for generating the atomic reduction body, may be null.
ReductionGenCBTy ReductionGen
Callback for generating the reduction body.
Value * Variable
Reduction variable of pointer type.
Value * PrivateVariable
Thread-private partial reduction variable.
ReductionGenClangCBTy ReductionGenClang
Clang callback for generating the reduction body.
Type * ElementType
Reduction element type, must match pointee type of variable.
ReductionGenDataPtrPtrCBTy DataPtrPtrGen
Container for the arguments used to pass data to the runtime library.
Value * SizesArray
The array of sizes passed to the runtime library.
Value * PointersArray
The array of section pointers passed to the runtime library.
Value * MappersArray
The array of user-defined mappers passed to the runtime library.
Value * MapTypesArrayEnd
The array of map types passed to the runtime library for the end of the region, or nullptr if there a...
Value * BasePointersArray
The array of base pointer passed to the runtime library.
Value * MapTypesArray
The array of map types passed to the runtime library for the beginning of the region or for the entir...
Value * MapNamesArray
The array of original declaration names of mapped pointers sent to the runtime library for debugging.
Data structure that contains the needed information to construct the kernel args vector.
ArrayRef< Value * > NumThreads
The number of threads.
TargetDataRTArgs RTArgs
Arguments passed to the runtime library.
Value * NumIterations
The number of iterations.
Value * DynCGroupMem
The size of the dynamic shared memory.
unsigned NumTargetItems
Number of arguments passed to the runtime library.
bool StrictBlocksAndThreads
True if the kernel strictly requires the number of blocks and threads above to run.
bool HasNoWait
True if the kernel has 'no wait' clause.
ArrayRef< Value * > NumTeams
The number of teams.
omp::OMPDynGroupprivateFallbackType DynCGroupMemFallback
The fallback mechanism for the shared memory.
Container to pass the default attributes with which a kernel must be launched, used to set kernel att...
omp::OMPTgtExecModeFlags ExecFlags
SmallVector< int32_t, 3 > MaxTeams
Container to pass LLVM IR runtime values or constants related to the number of teams and threads with...
Value * DeviceID
Device ID value used in the kernel launch.
SmallVector< Value *, 3 > MaxTeams
Value * MaxThreads
'parallel' construct 'num_threads' clause value, if present and it is an SPMD kernel.
Value * LoopTripCount
Total number of iterations of the SPMD or Generic-SPMD kernel or null if it is a generic kernel.
SmallVector< Value *, 3 > TargetThreadLimit
SmallVector< Value *, 3 > TeamsThreadLimit
Data structure to contain the information needed to uniquely identify a target entry.
static LLVM_ABI void getTargetRegionEntryFnName(SmallVectorImpl< char > &Name, StringRef ParentName, unsigned DeviceID, unsigned FileID, unsigned Line, unsigned Count)
static constexpr const char * KernelNamePrefix
The prefix used for kernel names.
static LLVM_ABI const Target * lookupTarget(const Triple &TheTriple, std::string &Error)
lookupTarget - Lookup a target based on a target triple.
Defines various target-specific GPU grid values that must be consistent between host RTL (plugin),...