69#define DEBUG_TYPE "openmp-ir-builder"
76 cl::desc(
"Use optimistic attributes describing "
77 "'as-if' properties of runtime calls."),
81 "openmp-ir-builder-unroll-threshold-factor",
cl::Hidden,
82 cl::desc(
"Factor for the unroll threshold to account for code "
83 "simplifications still taking place"),
87 "openmp-ir-builder-use-default-max-threads",
cl::Hidden,
98 if (!IP1.isValid() || !IP2.isValid())
105 switch (SchedType & ~OMPScheduleType::MonotonicityMask) {
106 case OMPScheduleType::UnorderedStaticChunked:
107 case OMPScheduleType::UnorderedStatic:
108 case OMPScheduleType::UnorderedDynamicChunked:
109 case OMPScheduleType::UnorderedGuidedChunked:
110 case OMPScheduleType::UnorderedRuntime:
111 case OMPScheduleType::UnorderedAuto:
112 case OMPScheduleType::UnorderedTrapezoidal:
113 case OMPScheduleType::UnorderedGreedy:
114 case OMPScheduleType::UnorderedBalanced:
115 case OMPScheduleType::UnorderedGuidedIterativeChunked:
116 case OMPScheduleType::UnorderedGuidedAnalyticalChunked:
117 case OMPScheduleType::UnorderedSteal:
118 case OMPScheduleType::UnorderedStaticBalancedChunked:
119 case OMPScheduleType::UnorderedGuidedSimd:
120 case OMPScheduleType::UnorderedRuntimeSimd:
121 case OMPScheduleType::OrderedStaticChunked:
122 case OMPScheduleType::OrderedStatic:
123 case OMPScheduleType::OrderedDynamicChunked:
124 case OMPScheduleType::OrderedGuidedChunked:
125 case OMPScheduleType::OrderedRuntime:
126 case OMPScheduleType::OrderedAuto:
127 case OMPScheduleType::OrderdTrapezoidal:
128 case OMPScheduleType::NomergeUnorderedStaticChunked:
129 case OMPScheduleType::NomergeUnorderedStatic:
130 case OMPScheduleType::NomergeUnorderedDynamicChunked:
131 case OMPScheduleType::NomergeUnorderedGuidedChunked:
132 case OMPScheduleType::NomergeUnorderedRuntime:
133 case OMPScheduleType::NomergeUnorderedAuto:
134 case OMPScheduleType::NomergeUnorderedTrapezoidal:
135 case OMPScheduleType::NomergeUnorderedGreedy:
136 case OMPScheduleType::NomergeUnorderedBalanced:
137 case OMPScheduleType::NomergeUnorderedGuidedIterativeChunked:
138 case OMPScheduleType::NomergeUnorderedGuidedAnalyticalChunked:
139 case OMPScheduleType::NomergeUnorderedSteal:
140 case OMPScheduleType::NomergeOrderedStaticChunked:
141 case OMPScheduleType::NomergeOrderedStatic:
142 case OMPScheduleType::NomergeOrderedDynamicChunked:
143 case OMPScheduleType::NomergeOrderedGuidedChunked:
144 case OMPScheduleType::NomergeOrderedRuntime:
145 case OMPScheduleType::NomergeOrderedAuto:
146 case OMPScheduleType::NomergeOrderedTrapezoidal:
147 case OMPScheduleType::OrderedDistributeChunked:
148 case OMPScheduleType::OrderedDistribute:
156 SchedType & OMPScheduleType::MonotonicityMask;
157 if (MonotonicityFlags == OMPScheduleType::MonotonicityMask)
171 Builder.restoreIP(IP);
175 if (Builder.GetInsertPoint() != BB->
end())
185 unsigned Line = FSP->getScopeLine() ? FSP->getScopeLine() : FSP->getLine();
186 Builder.SetCurrentDebugLocation(
192 return T.isAMDGPU() ||
T.isNVPTX() ||
T.isSPIRV();
198 Kernel->getFnAttribute(
"target-features").getValueAsString();
199 if (Features.
count(
"+wavefrontsize64"))
214 bool HasSimdModifier,
bool HasDistScheduleChunks) {
216 switch (ClauseKind) {
217 case OMP_SCHEDULE_Default:
218 case OMP_SCHEDULE_Static:
219 return HasChunks ? OMPScheduleType::BaseStaticChunked
220 : OMPScheduleType::BaseStatic;
221 case OMP_SCHEDULE_Dynamic:
222 return OMPScheduleType::BaseDynamicChunked;
223 case OMP_SCHEDULE_Guided:
224 return HasSimdModifier ? OMPScheduleType::BaseGuidedSimd
225 : OMPScheduleType::BaseGuidedChunked;
226 case OMP_SCHEDULE_Auto:
228 case OMP_SCHEDULE_Runtime:
229 return HasSimdModifier ? OMPScheduleType::BaseRuntimeSimd
230 : OMPScheduleType::BaseRuntime;
231 case OMP_SCHEDULE_Distribute:
232 return HasDistScheduleChunks ? OMPScheduleType::BaseDistributeChunked
233 : OMPScheduleType::BaseDistribute;
241 bool HasOrderedClause) {
242 assert((BaseScheduleType & OMPScheduleType::ModifierMask) ==
243 OMPScheduleType::None &&
244 "Must not have ordering nor monotonicity flags already set");
247 ? OMPScheduleType::ModifierOrdered
248 : OMPScheduleType::ModifierUnordered;
252 if (OrderingScheduleType ==
253 (OMPScheduleType::BaseGuidedSimd | OMPScheduleType::ModifierOrdered))
254 return OMPScheduleType::OrderedGuidedChunked;
255 else if (OrderingScheduleType == (OMPScheduleType::BaseRuntimeSimd |
256 OMPScheduleType::ModifierOrdered))
257 return OMPScheduleType::OrderedRuntime;
259 return OrderingScheduleType;
265 bool HasSimdModifier,
bool HasMonotonic,
266 bool HasNonmonotonic,
bool HasOrderedClause) {
267 assert((ScheduleType & OMPScheduleType::MonotonicityMask) ==
268 OMPScheduleType::None &&
269 "Must not have monotonicity flags already set");
270 assert((!HasMonotonic || !HasNonmonotonic) &&
271 "Monotonic and Nonmonotonic are contradicting each other");
274 return ScheduleType | OMPScheduleType::ModifierMonotonic;
275 }
else if (HasNonmonotonic) {
276 return ScheduleType | OMPScheduleType::ModifierNonmonotonic;
286 if ((BaseScheduleType == OMPScheduleType::BaseStatic) ||
287 (BaseScheduleType == OMPScheduleType::BaseStaticChunked) ||
293 return ScheduleType | OMPScheduleType::ModifierNonmonotonic;
301 bool HasSimdModifier,
bool HasMonotonicModifier,
302 bool HasNonmonotonicModifier,
bool HasOrderedClause,
303 bool HasDistScheduleChunks) {
305 ClauseKind, HasChunks, HasSimdModifier, HasDistScheduleChunks);
309 OrderedSchedule, HasSimdModifier, HasMonotonicModifier,
310 HasNonmonotonicModifier, HasOrderedClause);
318static std::optional<omp::OMPTgtExecModeFlags>
323 if (
Call->getCalledFunction()->getName() ==
"__kmpc_target_init") {
324 TargetInitCall =
Call;
349 std::optional<omp::OMPTgtExecModeFlags> ExecMode =
361 if (
Instruction *Term = Source->getTerminatorOrNull()) {
370 NewBr->setDebugLoc(
DL);
375 assert(New->getFirstInsertionPt() == New->begin() &&
376 "Target BB must not have PHI nodes");
392 New->splice(New->begin(), Old, IP, Old->
end());
396 NewBr->setDebugLoc(
DL);
408 Builder.SetInsertPoint(Old);
412 Builder.SetCurrentDebugLocation(
DebugLoc);
422 New->replaceSuccessorsPhiUsesWith(Old, New);
431 Builder.SetInsertPoint(Builder.GetInsertBlock()->getTerminator());
433 Builder.SetInsertPoint(Builder.GetInsertBlock());
436 Builder.SetCurrentDebugLocation(
DebugLoc);
445 Builder.SetInsertPoint(Builder.GetInsertBlock()->getTerminator());
447 Builder.SetInsertPoint(Builder.GetInsertBlock());
450 Builder.SetCurrentDebugLocation(
DebugLoc);
467 const Twine &Name =
"",
bool AsPtr =
true,
468 bool Is64Bit =
false) {
469 Builder.restoreIP(OuterAllocaIP);
473 Builder.CreateAlloca(IntTy,
nullptr, Name +
".addr");
477 FakeVal = FakeValAddr;
482 FakeValAddr, Builder.getPtrTy(), Name +
".ascast"));
486 FakeVal = Builder.CreateLoad(IntTy, FakeValAddr, Name +
".val");
491 Builder.restoreIP(InnerAllocaIP);
494 UseFakeVal = Builder.CreateLoad(IntTy, FakeVal, Name +
".use");
497 FakeVal, Is64Bit ? Builder.getInt64(10) : Builder.getInt32(10)));
510enum OpenMPOffloadingRequiresDirFlags {
512 OMP_REQ_UNDEFINED = 0x000,
514 OMP_REQ_NONE = 0x001,
516 OMP_REQ_REVERSE_OFFLOAD = 0x002,
518 OMP_REQ_UNIFIED_ADDRESS = 0x004,
520 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008,
522 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010,
529 DominatorTree *DT =
nullptr,
bool AggregateArgs =
false,
530 BlockFrequencyInfo *BFI =
nullptr,
531 BranchProbabilityInfo *BPI =
nullptr,
532 AssumptionCache *AC =
nullptr,
bool AllowVarArgs =
false,
533 bool AllowAlloca =
false,
534 BasicBlock *AllocationBlock =
nullptr,
536 std::string Suffix =
"",
bool ArgsInZeroAddressSpace =
false)
537 : CodeExtractor(BBs, DT, AggregateArgs, BFI, BPI, AC, AllowVarArgs,
538 AllowAlloca, AllocationBlock, DeallocationBlocks, Suffix,
539 ArgsInZeroAddressSpace),
540 OMPBuilder(OMPBuilder) {}
542 virtual ~OMPCodeExtractor() =
default;
545 OpenMPIRBuilder &OMPBuilder;
548class DeviceSharedMemCodeExtractor :
public OMPCodeExtractor {
550 using OMPCodeExtractor::OMPCodeExtractor;
551 virtual ~DeviceSharedMemCodeExtractor() =
default;
555 allocateVar(IRBuilder<>::InsertPoint AllocaIP,
DebugLoc DL,
Type *VarType,
556 const Twine &Name = Twine(
""),
557 AddrSpaceCastInst **CastedAlloc =
nullptr)
override {
558 return OMPBuilder.createOMPAllocShared({AllocaIP,
DL}, VarType,
Name);
561 virtual Instruction *deallocateVar(IRBuilder<>::InsertPoint DeallocIP,
563 Type *VarType)
override {
564 return OMPBuilder.createOMPFreeShared({DeallocIP,
DL}, Var, VarType);
571 OpenMPIRBuilder &OMPBuilder;
573 DeviceSharedMemOutlineInfo(OpenMPIRBuilder &OMPBuilder)
574 : OMPBuilder(OMPBuilder) {}
575 virtual ~DeviceSharedMemOutlineInfo() =
default;
577 virtual std::unique_ptr<CodeExtractor>
579 bool ArgsInZeroAddressSpace,
580 Twine Suffix = Twine(
""))
override;
586 : RequiresFlags(OMP_REQ_UNDEFINED) {}
590 bool HasRequiresReverseOffload,
bool HasRequiresUnifiedAddress,
591 bool HasRequiresUnifiedSharedMemory,
bool HasRequiresDynamicAllocators)
594 RequiresFlags(OMP_REQ_UNDEFINED) {
595 if (HasRequiresReverseOffload)
596 RequiresFlags |= OMP_REQ_REVERSE_OFFLOAD;
597 if (HasRequiresUnifiedAddress)
598 RequiresFlags |= OMP_REQ_UNIFIED_ADDRESS;
599 if (HasRequiresUnifiedSharedMemory)
600 RequiresFlags |= OMP_REQ_UNIFIED_SHARED_MEMORY;
601 if (HasRequiresDynamicAllocators)
602 RequiresFlags |= OMP_REQ_DYNAMIC_ALLOCATORS;
606 return RequiresFlags & OMP_REQ_REVERSE_OFFLOAD;
610 return RequiresFlags & OMP_REQ_UNIFIED_ADDRESS;
614 return RequiresFlags & OMP_REQ_UNIFIED_SHARED_MEMORY;
618 return RequiresFlags & OMP_REQ_DYNAMIC_ALLOCATORS;
623 :
static_cast<int64_t
>(OMP_REQ_NONE);
628 RequiresFlags |= OMP_REQ_REVERSE_OFFLOAD;
630 RequiresFlags &= ~OMP_REQ_REVERSE_OFFLOAD;
635 RequiresFlags |= OMP_REQ_UNIFIED_ADDRESS;
637 RequiresFlags &= ~OMP_REQ_UNIFIED_ADDRESS;
642 RequiresFlags |= OMP_REQ_UNIFIED_SHARED_MEMORY;
644 RequiresFlags &= ~OMP_REQ_UNIFIED_SHARED_MEMORY;
649 RequiresFlags |= OMP_REQ_DYNAMIC_ALLOCATORS;
651 RequiresFlags &= ~OMP_REQ_DYNAMIC_ALLOCATORS;
664 constexpr size_t MaxDim = 3;
669 Value *DynCGroupMemFallbackFlag =
671 DynCGroupMemFallbackFlag =
Builder.CreateShl(DynCGroupMemFallbackFlag, 2);
676 StrictBlocksFlag =
Builder.CreateShl(StrictBlocksFlag, 6);
677 StrictThreadsFlag =
Builder.CreateShl(StrictThreadsFlag, 7);
679 Value *Flags =
Builder.CreateOr(HasNoWaitFlag, DynCGroupMemFallbackFlag);
680 Flags =
Builder.CreateOr(Flags, StrictBlocksFlag);
681 Flags =
Builder.CreateOr(Flags, StrictThreadsFlag);
687 Value *NumThreads3D =
718 auto FnAttrs = Attrs.getFnAttrs();
719 auto RetAttrs = Attrs.getRetAttrs();
721 for (
size_t ArgNo = 0; ArgNo < Fn.
arg_size(); ++ArgNo)
726 bool Param =
true) ->
void {
727 bool HasSignExt = AS.hasAttribute(Attribute::SExt);
728 bool HasZeroExt = AS.hasAttribute(Attribute::ZExt);
729 if (HasSignExt || HasZeroExt) {
730 assert(AS.getNumAttributes() == 1 &&
731 "Currently not handling extension attr combined with others.");
733 if (
auto AK = TargetLibraryInfo::getExtAttrForI32Param(
T, HasSignExt))
736 TargetLibraryInfo::getExtAttrForI32Return(
T, HasSignExt))
743#define OMP_ATTRS_SET(VarName, AttrSet) AttributeSet VarName = AttrSet;
744#include "llvm/Frontend/OpenMP/OMPKinds.def"
748#define OMP_RTL_ATTRS(Enum, FnAttrSet, RetAttrSet, ArgAttrSets) \
750 FnAttrs = FnAttrs.addAttributes(Ctx, FnAttrSet); \
751 addAttrSet(RetAttrs, RetAttrSet, false); \
752 for (size_t ArgNo = 0; ArgNo < ArgAttrSets.size(); ++ArgNo) \
753 addAttrSet(ArgAttrs[ArgNo], ArgAttrSets[ArgNo]); \
754 Fn.setAttributes(AttributeList::get(Ctx, FnAttrs, RetAttrs, ArgAttrs)); \
756#include "llvm/Frontend/OpenMP/OMPKinds.def"
770#define OMP_RTL(Enum, Str, IsVarArg, ReturnType, ...) \
772 FnTy = FunctionType::get(ReturnType, ArrayRef<Type *>{__VA_ARGS__}, \
774 Fn = M.getFunction(Str); \
776#include "llvm/Frontend/OpenMP/OMPKinds.def"
782#define OMP_RTL(Enum, Str, ...) \
784 Fn = Function::Create(FnTy, GlobalValue::ExternalLinkage, Str, M); \
786#include "llvm/Frontend/OpenMP/OMPKinds.def"
790 if (FnID == OMPRTL___kmpc_fork_call || FnID == OMPRTL___kmpc_fork_teams) {
800 LLVMContext::MD_callback,
802 2, {-1, -1},
true)}));
815 assert(Fn &&
"Failed to create OpenMP runtime function");
826 Builder.SetInsertPoint(FiniBB);
838 FiniBB = OtherFiniBB;
840 Builder.SetInsertPoint(FiniBB->getFirstNonPHIIt());
848 auto EndIt = FiniBB->end();
849 if (FiniBB->size() >= 1)
850 if (
auto Prev = std::prev(EndIt); Prev->isTerminator())
855 FiniBB->replaceAllUsesWith(OtherFiniBB);
856 FiniBB->eraseFromParent();
857 FiniBB = OtherFiniBB;
864 assert(Fn &&
"Failed to create OpenMP runtime function pointer");
887 for (
auto Inst =
Block->getReverseIterator()->begin();
888 Inst !=
Block->getReverseIterator()->end();) {
917 Block.getParent()->getEntryBlock().getTerminator()->getIterator();
938 DeferredOutlines.
push_back(std::move(OI));
942 ParallelRegionBlockSet.
clear();
944 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
954 bool ArgsInZeroAddressSpace =
Config.isTargetDevice();
955 std::unique_ptr<CodeExtractor> Extractor =
956 OI->createCodeExtractor(Blocks, ArgsInZeroAddressSpace,
".omp_par");
960 <<
" Exit: " << OI->ExitBB->getName() <<
"\n");
961 assert(Extractor->isEligible() &&
962 "Expected OpenMP outlining to be possible!");
964 for (
auto *V : OI->ExcludeArgsFromAggregate)
965 Extractor->excludeArgFromAggregate(V);
968 Extractor->extractCodeRegion(CEAC, OI->Inputs, OI->Outputs);
972 if (TargetCpuAttr.isStringAttribute())
975 auto TargetFeaturesAttr = OuterFn->
getFnAttribute(
"target-features");
976 if (TargetFeaturesAttr.isStringAttribute())
977 OutlinedFn->
addFnAttr(TargetFeaturesAttr);
980 LLVM_DEBUG(
dbgs() <<
" Outlined function: " << *OutlinedFn <<
"\n");
982 "OpenMP outlined functions should not return a value!");
987 M.getFunctionList().insertAfter(OuterFn->
getIterator(), OutlinedFn);
994 assert(OI->EntryBB->getUniquePredecessor() == &ArtificialEntry);
1001 "Expected instructions to add in the outlined region entry");
1003 End = ArtificialEntry.
rend();
1008 if (
I.isTerminator()) {
1010 if (
Instruction *TI = OI->EntryBB->getTerminatorOrNull())
1011 TI->adoptDbgRecords(&ArtificialEntry,
I.getIterator(),
false);
1015 I.moveBeforePreserving(*OI->EntryBB,
1016 OI->EntryBB->getFirstInsertionPt());
1019 OI->EntryBB->moveBefore(&ArtificialEntry);
1026 if (OI->PostOutlineCB)
1027 OI->PostOutlineCB(*OutlinedFn);
1029 if (OI->FixUpNonEntryAllocas)
1061 errs() <<
"Error of kind: " << Kind
1062 <<
" when emitting offload entries and metadata during "
1063 "OMPIRBuilder finalization \n";
1071 if (
Config.isTargetDevice())
1072 applyDeclareTargetGlobalReplacements();
1074 if (
Config.EmitLLVMUsedMetaInfo.value_or(
false)) {
1075 std::vector<WeakTrackingVH> LLVMCompilerUsed = {
1076 M.getGlobalVariable(
"__openmp_nvptx_data_transfer_temporary_storage")};
1077 emitUsed(
"llvm.compiler.used", LLVMCompilerUsed);
1087 assert(Original && Replacement &&
1088 "Null values provided to registerDeclareTargetGlobalReplacement");
1092void OpenMPIRBuilder::applyDeclareTargetGlobalReplacements() {
1098 "A null value was inserted into DeclareTargetGlobalReplacements");
1102 if (!OldGV || !NewGV)
1136 for (
unsigned I = 0, E =
PHI->getNumIncomingValues();
I < E; ++
I) {
1137 if (
PHI->getIncomingValue(
I) != OldGV)
1142 Builder.SetCurrentDebugLocation(
PHI->getDebugLoc());
1144 PHI->setIncomingValue(
I, EdgeLoad);
1150 Builder.SetCurrentDebugLocation(Insn->getDebugLoc());
1166 "Non-default address space declare target global");
1168 unsigned DestAS = ASC->getType()->getPointerAddressSpace();
1169 if (DestAS == 0 && NewGVAS != OldGVAS) {
1170 ASC->replaceAllUsesWith(
Load);
1171 ASC->eraseFromParent();
1176 Insn->replaceUsesOfWith(OldGV,
Load);
1192 ConstantInt::get(I32Ty,
Value), Name);
1205 for (
unsigned I = 0, E =
List.size();
I != E; ++
I)
1209 if (UsedArray.
empty())
1216 GV->setSection(
"llvm.metadata");
1222 auto *Int8Ty =
Builder.getInt8Ty();
1225 ConstantInt::get(Int8Ty, Mode),
Twine(KernelName,
"_exec_mode"));
1233 unsigned Reserve2Flags) {
1235 LocFlags |= OMP_IDENT_FLAG_KMPC;
1242 ConstantInt::get(Int32,
uint32_t(LocFlags)),
1243 ConstantInt::get(Int32, Reserve2Flags),
1244 ConstantInt::get(Int32, SrcLocStrSize), SrcLocStr};
1246 size_t SrcLocStrArgIdx = 4;
1247 if (OpenMPIRBuilder::Ident->getElementType(SrcLocStrArgIdx)
1251 SrcLocStr, OpenMPIRBuilder::Ident->getElementType(SrcLocStrArgIdx));
1258 if (
GV.getValueType() == OpenMPIRBuilder::Ident &&
GV.hasInitializer())
1259 if (
GV.getInitializer() == Initializer)
1264 M, OpenMPIRBuilder::Ident,
1267 M.getDataLayout().getDefaultGlobalsAddressSpace());
1279 SrcLocStrSize = LocStr.
size();
1288 if (
GV.isConstant() &&
GV.hasInitializer() &&
1289 GV.getInitializer() == Initializer)
1292 SrcLocStr =
Builder.CreateGlobalString(
1293 LocStr,
"",
M.getDataLayout().getDefaultGlobalsAddressSpace(),
1301 unsigned Line,
unsigned Column,
1307 Buffer.
append(FunctionName);
1309 Buffer.
append(std::to_string(Line));
1311 Buffer.
append(std::to_string(Column));
1319 StringRef UnknownLoc =
";unknown;unknown;0;0;;";
1330 !DIL->getFilename().empty() ? DIL->getFilename() :
M.getName();
1335 DIL->getColumn(), SrcLocStrSize);
1341 Loc.IP.getNodeParent()->getParent());
1347 "omp_global_thread_num");
1355 "expected one result pointer type per in_reduction item");
1358 if (OrigPtrs.
empty())
1359 return Builder.saveIP();
1378 for (
unsigned Idx = 0; Idx < OrigPtrs.
size(); ++Idx) {
1381 Value *OrigPtr = OrigPtrs[Idx];
1383 OrigPtrTy && OrigPtrTy->getAddressSpace() != 0)
1384 OrigPtr = Builder.CreateAddrSpaceCast(OrigPtr, PtrTy);
1386 Value *
Priv = Builder.CreateCall(GetThData, {Gtid, NullDesc, OrigPtr},
1392 ResPtrTy && ResPtrTy->getAddressSpace() != 0)
1393 Priv = Builder.CreateAddrSpaceCast(
Priv, ResultPtrTys[Idx]);
1395 MapPrivateCB(Idx,
Priv);
1402 bool ForceSimpleCall,
bool CheckCancelFlag) {
1412 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_FOR;
1415 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_SECTIONS;
1418 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_SINGLE;
1421 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_EXPL;
1424 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL;
1437 bool UseCancelBarrier =
1442 ? OMPRTL___kmpc_cancel_barrier
1443 : OMPRTL___kmpc_barrier),
1446 if (UseCancelBarrier && CheckCancelFlag)
1456 omp::Directive CanceledDirective) {
1461 auto *UI =
Builder.CreateUnreachable();
1469 Builder.SetInsertPoint(ElseTI);
1470 auto ElseIP =
Builder.saveIP();
1478 Builder.SetInsertPoint(ThenTI);
1480 Value *CancelKind =
nullptr;
1481 switch (CanceledDirective) {
1482#define OMP_CANCEL_KIND(Enum, Str, DirectiveEnum, Value) \
1483 case DirectiveEnum: \
1484 CancelKind = Builder.getInt32(Value); \
1486#include "llvm/Frontend/OpenMP/OMPKinds.def"
1503 Builder.SetInsertPoint(UI->getParent());
1504 UI->eraseFromParent();
1511 omp::Directive CanceledDirective) {
1516 auto *UI =
Builder.CreateUnreachable();
1519 Value *CancelKind =
nullptr;
1520 switch (CanceledDirective) {
1521#define OMP_CANCEL_KIND(Enum, Str, DirectiveEnum, Value) \
1522 case DirectiveEnum: \
1523 CancelKind = Builder.getInt32(Value); \
1525#include "llvm/Frontend/OpenMP/OMPKinds.def"
1542 Builder.SetInsertPoint(UI->getParent());
1543 UI->eraseFromParent();
1556 auto *KernelArgsPtr =
1557 Builder.CreateAlloca(OpenMPIRBuilder::KernelArgs,
nullptr,
"kernel_args");
1562 Builder.CreateStructGEP(OpenMPIRBuilder::KernelArgs, KernelArgsPtr,
I);
1565 M.getDataLayout().getPrefTypeAlign(KernelArgs[
I]->getType()));
1569 NumThreads, HostPtr, KernelArgsPtr};
1596 assert(OutlinedFnID &&
"Invalid outlined function ID!");
1600 Value *Return =
nullptr;
1620 Builder, AllocaIP, Return, RTLoc, DeviceID, Args.NumTeams.front(),
1621 Args.NumThreads.front(), OutlinedFnID, ArgsVector));
1628 Builder.CreateCondBr(
Failed, OffloadFailedBlock, OffloadContBlock);
1630 auto CurFn =
Builder.GetInsertBlock()->getParent();
1637 emitBlock(OffloadContBlock, CurFn,
true);
1642 Value *CancelFlag, omp::Directive CanceledDirective) {
1644 "Unexpected cancellation!");
1664 Builder.CreateCondBr(Cmp, NonCancellationBlock, CancellationBlock,
1673 Builder.SetInsertPoint(CancellationBlock);
1674 Builder.CreateBr(*FiniBBOrErr);
1677 Builder.SetInsertPoint(NonCancellationBlock, NonCancellationBlock->
begin());
1689 size_t NumArgs = OutlinedFn.
arg_size();
1690 assert((NumArgs == 2 || NumArgs == 3) &&
1691 "expected a 2-3 argument parallel outlined function");
1692 bool UseArgStruct = NumArgs == 3;
1697 {Builder.getInt16Ty(), Builder.getInt32Ty()},
1701 OutlinedFn.
getName() +
".wrapper", OMPIRBuilder->
M);
1703 WrapperFn->addParamAttr(0, Attribute::NoUndef);
1704 WrapperFn->addParamAttr(0, Attribute::ZExt);
1705 WrapperFn->addParamAttr(1, Attribute::NoUndef);
1709 Builder.SetInsertPoint(EntryBB);
1712 Value *AddrAlloca = Builder.CreateAlloca(Builder.getInt32Ty(),
1714 AddrAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1715 AddrAlloca, Builder.getPtrTy(0),
1716 AddrAlloca->
getName() +
".ascast");
1718 Value *ZeroAlloca = Builder.CreateAlloca(Builder.getInt32Ty(),
1720 ZeroAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1721 ZeroAlloca, Builder.getPtrTy(0),
1722 ZeroAlloca->
getName() +
".ascast");
1724 Value *ArgsAlloca =
nullptr;
1726 ArgsAlloca = Builder.CreateAlloca(Builder.getPtrTy(),
1727 nullptr,
"global_args");
1728 ArgsAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1729 ArgsAlloca, Builder.getPtrTy(0),
1730 ArgsAlloca->
getName() +
".ascast");
1734 Builder.CreateStore(WrapperFn->getArg(1), AddrAlloca);
1735 Builder.CreateStore(Builder.getInt32(0), ZeroAlloca);
1739 llvm::omp::RuntimeFunction::OMPRTL___kmpc_get_shared_variables),
1747 Value *StructArg = Builder.CreateLoad(Builder.getPtrTy(), ArgsAlloca);
1748 StructArg = Builder.CreateInBoundsGEP(Builder.getPtrTy(), StructArg,
1749 {Builder.getInt64(0)});
1750 StructArg = Builder.CreateLoad(Builder.getPtrTy(), StructArg,
"structArg");
1751 Args.push_back(StructArg);
1755 Builder.CreateCall(&OutlinedFn, Args);
1756 Builder.CreateRetVoid();
1771 "Expected at least tid and bounded tid as arguments");
1772 unsigned NumCapturedVars = OutlinedFn.
arg_size() - 2;
1780 OutlinedFn.
addFnAttr(Attribute::NoUnwind);
1783 assert(CI &&
"Expected call instruction to outlined function");
1784 CI->
getParent()->setName(
"omp_parallel");
1786 Builder.SetInsertPoint(CI);
1787 Type *PtrTy = OMPIRBuilder->VoidPtr;
1790 OpenMPIRBuilder ::InsertPointTy CurrentIP = Builder.saveIP();
1794 Value *Args = ArgsAlloca;
1798 Args = Builder.CreatePointerCast(ArgsAlloca, PtrTy);
1799 Builder.restoreIP(CurrentIP);
1802 for (
unsigned Idx = 0; Idx < NumCapturedVars; Idx++) {
1804 Value *StoreAddress = Builder.CreateConstInBoundsGEP2_64(
1806 Builder.CreateStore(V, StoreAddress);
1810 IfCondition ? Builder.CreateSExtOrTrunc(IfCondition, OMPIRBuilder->Int32)
1811 : Builder.getInt32(1);
1812 Value *NumThreadsArg =
1813 NumThreads ? Builder.CreateZExtOrTrunc(NumThreads, OMPIRBuilder->Int32)
1814 : Builder.getInt32(-1);
1824 Value *Parallel60CallArgs[] = {
1829 Builder.getInt32(-1),
1833 Builder.getInt64(NumCapturedVars),
1834 Builder.getInt32(0)};
1842 << *Builder.GetInsertBlock()->getParent() <<
"\n");
1845 Builder.SetInsertPoint(PrivTID);
1847 Builder.CreateStore(Builder.CreateLoad(OMPIRBuilder->Int32, OutlinedAI),
1854 I->eraseFromParent();
1877 if (!
F->hasMetadata(LLVMContext::MD_callback)) {
1885 F->addMetadata(LLVMContext::MD_callback,
1894 OutlinedFn.
addFnAttr(Attribute::NoUnwind);
1897 "Expected at least tid and bounded tid as arguments");
1898 unsigned NumCapturedVars = OutlinedFn.
arg_size() - 2;
1901 CI->
getParent()->setName(
"omp_parallel");
1902 Builder.SetInsertPoint(CI);
1905 Value *ForkCallArgs[] = {Ident, Builder.getInt32(NumCapturedVars),
1909 RealArgs.
append(std::begin(ForkCallArgs), std::end(ForkCallArgs));
1911 Value *
Cond = Builder.CreateSExtOrTrunc(IfCondition, OMPIRBuilder->Int32);
1918 auto PtrTy = OMPIRBuilder->VoidPtr;
1919 if (IfCondition && NumCapturedVars == 0) {
1927 << *Builder.GetInsertBlock()->getParent() <<
"\n");
1930 Builder.SetInsertPoint(PrivTID);
1932 Builder.CreateStore(Builder.CreateLoad(OMPIRBuilder->Int32, OutlinedAI),
1939 I->eraseFromParent();
1947 Value *NumThreads, omp::ProcBindKind ProcBind,
bool IsCancellable) {
1956 const bool NeedThreadID = NumThreads ||
Config.isTargetDevice() ||
1957 (ProcBind != OMP_PROC_BIND_default);
1964 bool ArgsInZeroAddressSpace =
Config.isTargetDevice();
1968 if (NumThreads && !
Config.isTargetDevice()) {
1971 Builder.CreateIntCast(NumThreads, Int32,
false)};
1976 if (ProcBind != OMP_PROC_BIND_default) {
1980 ConstantInt::get(Int32,
unsigned(ProcBind),
true)};
1990 BasicBlock *OuterAllocaBlock = OuterAllocIP.getNodeParent();
2002 Builder.CreateAlloca(Int32,
nullptr,
"zero.addr");
2005 if (ArgsInZeroAddressSpace &&
M.getDataLayout().getAllocaAddrSpace() != 0) {
2008 TIDAddrAlloca, PointerType ::get(
M.getContext(), 0),
"tid.addr.ascast");
2012 PointerType ::get(
M.getContext(), 0),
2013 "zero.addr.ascast");
2037 if (IP == IP.getNodeParent()->end()) {
2041 IP =
I->getIterator();
2043 assert(IP.getNodeParent()->getTerminator()->getNumSuccessors() == 1 &&
2044 IP.getNodeParent()->getTerminator()->getSuccessor(0) == PRegExitBB &&
2045 "Unexpected insertion point for finalization call!");
2057 Builder.CreateAlloca(Int32,
nullptr,
"tid.addr.local");
2063 Builder.CreateLoad(Int32, ZeroAddr,
"zero.addr.use");
2081 LLVM_DEBUG(
dbgs() <<
"Before body codegen: " << *OuterFn <<
"\n");
2084 assert(BodyGenCB &&
"Expected body generation callback!");
2086 if (
Error Err = BodyGenCB(InnerAllocaIP, CodeGenIP, PRegExitBB))
2089 LLVM_DEBUG(
dbgs() <<
"After body codegen: " << *OuterFn <<
"\n");
2093 bool UsesDeviceSharedMemory =
2095 std::unique_ptr<OutlineInfo> OI =
2096 UsesDeviceSharedMemory
2097 ? std::make_unique<DeviceSharedMemOutlineInfo>(*
this)
2098 : std::make_unique<OutlineInfo>();
2100 if (
Config.isTargetDevice()) {
2102 OI->PostOutlineCB = [=, ToBeDeletedVec =
2103 std::move(ToBeDeleted)](
Function &OutlinedFn) {
2105 IfCondition, NumThreads, PrivTID, PrivTIDAddr,
2106 ThreadID, ToBeDeletedVec);
2110 OI->PostOutlineCB = [=, ToBeDeletedVec =
2111 std::move(ToBeDeleted)](
Function &OutlinedFn) {
2113 PrivTID, PrivTIDAddr, ToBeDeletedVec);
2117 OI->FixUpNonEntryAllocas =
true;
2118 OI->OuterAllocBB = OuterAllocaBlock;
2119 OI->EntryBB = PRegEntryBB;
2120 OI->ExitBB = PRegExitBB;
2121 OI->OuterDeallocBBs.reserve(OuterDeallocBlocks.
size());
2122 copy(OuterDeallocBlocks, OI->OuterDeallocBBs.
end());
2126 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
2138 ".omp_par", ArgsInZeroAddressSpace);
2143 Extractor.findAllocas(CEAC, SinkingCands, HoistingCands, CommonExit);
2145 Extractor.findInputsOutputs(Inputs, Outputs, SinkingCands,
2150 return GV->getValueType() == OpenMPIRBuilder::Ident;
2155 LLVM_DEBUG(
dbgs() <<
"Before privatization: " << *OuterFn <<
"\n");
2161 if (&V == TIDAddr || &V == ZeroAddr) {
2162 OI->ExcludeArgsFromAggregate.push_back(&V);
2167 for (
Use &U : V.uses())
2169 if (ParallelRegionBlockSet.
count(UserI->getParent()))
2179 if (!V.getType()->isPointerTy()) {
2183 Builder.restoreIP(OuterAllocIP);
2185 if (UsesDeviceSharedMemory) {
2188 V.getName() +
".reloaded");
2189 for (
BasicBlock *DeallocBlock : OuterDeallocBlocks) {
2190 assert(DeallocBlock->getParent() ==
2191 OuterAllocIP.getNodeParent()->getParent() &&
2192 "Dealloc block must be in the allocation's function to reuse "
2193 "its debug location");
2195 Builder.getCurrentDebugLocation()},
2199 Ptr =
Builder.CreateAlloca(V.getType(),
nullptr,
2200 V.getName() +
".reloaded");
2205 Builder.SetInsertPoint(InsertBB,
2210 Builder.restoreIP(InnerAllocaIP);
2211 Inner =
Builder.CreateLoad(V.getType(), Ptr);
2214 Value *ReplacementValue =
nullptr;
2217 ReplacementValue = PrivTID;
2220 PrivCB(InnerAllocaIP,
Builder.saveIP(), V, *Inner, ReplacementValue);
2225 InnerAllocaIP.getNodeParent()->getTerminator()->getIterator();
2227 assert(ReplacementValue &&
2228 "Expected copy/create callback to set replacement value!");
2229 if (ReplacementValue == &V)
2234 UPtr->set(ReplacementValue);
2257 for (
Value *Output : Outputs)
2261 "OpenMP outlining should not produce live-out values!");
2263 LLVM_DEBUG(
dbgs() <<
"After privatization: " << *OuterFn <<
"\n");
2265 for (
auto *BB : Blocks)
2266 dbgs() <<
" PBR: " << BB->getName() <<
"\n";
2274 assert(FiniInfo.DK == OMPD_parallel &&
2275 "Unexpected finalization stack state!");
2286 Builder.CreateBr(*FiniBBOrErr);
2290 Term->eraseFromParent();
2297 UI->eraseFromParent();
2329 Value *Severity = ConstantInt::get(Int32, IsFatal ? 2 : 1);
2331 Value *Args[] = {Ident, Severity, MessageArg};
2360 static_cast<unsigned int>(RTLDependInfoFields::BaseAddr));
2362 Builder.CreateStore(DepValPtr, Addr);
2365 DependInfo, Entry,
static_cast<unsigned int>(RTLDependInfoFields::Len));
2367 ConstantInt::get(SizeTy,
2372 DependInfo, Entry,
static_cast<unsigned int>(RTLDependInfoFields::Flags));
2374 static_cast<unsigned int>(Dep.
DepKind)),
2387 if (Dependencies.
empty())
2407 Type *DependInfo = OMPBuilder.DependInfo;
2409 Value *DepArray =
nullptr;
2415 Builder.SetInsertPoint(
2416 Builder.GetInsertBlock()->getParent()->getEntryBlock().getTerminator());
2417 DepArray = Builder.CreateAlloca(DepArrayTy,
nullptr,
".dep.arr.addr");
2420 for (
const auto &[DepIdx, Dep] :
enumerate(Dependencies)) {
2422 Builder.CreateConstInBoundsGEP2_64(DepArrayTy, DepArray, 0, DepIdx);
2447 Value *DepArray =
nullptr;
2448 Type *DepArrayTy =
nullptr;
2449 Value *NumDeps =
nullptr;
2452 NumDeps = Dependencies.
NumDeps;
2453 }
else if (!Dependencies.
Deps.empty()) {
2455 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
2459 Builder.GetInsertBlock()->getParent()->getEntryBlock();
2461 DepArray =
Builder.CreateAlloca(DepArrayTy,
nullptr,
".dep.arr.addr");
2464 for (
const auto &[DepIdx, Dep] :
enumerate(Dependencies.
Deps)) {
2466 Builder.CreateConstInBoundsGEP2_64(DepArrayTy, DepArray, 0, DepIdx);
2480 ConstantInt::get(
Builder.getInt32Ty(), 0),
2482 ConstantInt::get(
Builder.getInt32Ty(), IsNowait)};
2485 omp::RuntimeFunction::OMPRTL___kmpc_omp_taskwait_deps_51),
2495 unsigned ProgramAddressSpace = M.getDataLayout().getProgramAddressSpace();
2507 auto *VoidPtrTy =
PointerType::get(Builder.getContext(), ProgramAddressSpace);
2510 Builder.getVoidTy(), {VoidPtrTy, VoidPtrTy, Builder.getInt32Ty()},
2514 "omp_taskloop_dup", M);
2517 Value *LastprivateFlagArg = DupFunction->
getArg(2);
2518 DestTaskArg->
setName(
"dest_task");
2519 SrcTaskArg->
setName(
"src_task");
2520 LastprivateFlagArg->
setName(
"lastprivate_flag");
2523 Builder.SetInsertPoint(
2526 auto GetTaskContextPtrFromArg = [&](
Value *Arg) ->
Value * {
2527 Type *TaskWithPrivatesTy =
2529 Value *TaskPrivates = Builder.CreateGEP(
2530 TaskWithPrivatesTy, Arg, {Builder.getInt32(0), Builder.getInt32(1)});
2531 Value *ContextPtr = Builder.CreateGEP(
2532 PrivatesTy, TaskPrivates,
2533 {Builder.getInt32(0), Builder.getInt32(PrivatesIndex)});
2537 Value *DestTaskContextPtr = GetTaskContextPtrFromArg(DestTaskArg);
2538 Value *SrcTaskContextPtr = GetTaskContextPtrFromArg(SrcTaskArg);
2540 DestTaskContextPtr->
setName(
"destPtr");
2541 SrcTaskContextPtr->
setName(
"srcPtr");
2545 Expected<IRBuilderBase::InsertPoint> AfterIPOrError =
2546 DupCB(AllocaIP, CodeGenIP, DestTaskContextPtr, SrcTaskContextPtr);
2547 if (!AfterIPOrError)
2549 Builder.restoreIP(*AfterIPOrError);
2559 llvm::function_ref<llvm::Expected<llvm::CanonicalLoopInfo *>()> LoopInfo,
2561 Value *GrainSize,
bool NoGroup,
int Sched,
Value *Final,
bool Mergeable,
2563 Value *TaskContextStructPtrVal,
bool FreeAgent) {
2568 uint32_t SrcLocStrSize;
2582 if (
Error Err = BodyGenCB(TaskloopAllocaIP, TaskloopBodyIP, TaskloopExitBB))
2585 llvm::Expected<llvm::CanonicalLoopInfo *> result = LoopInfo();
2590 llvm::CanonicalLoopInfo *CLI = result.
get();
2591 auto OI = std::make_unique<OutlineInfo>();
2592 OI->EntryBB = TaskloopAllocaBB;
2593 OI->OuterAllocBB = AllocaIP.getNodeParent();
2594 OI->ExitBB = TaskloopExitBB;
2595 OI->OuterDeallocBBs.reserve(DeallocBlocks.
size());
2596 copy(DeallocBlocks, OI->OuterDeallocBBs.end());
2599 SmallVector<Instruction *> ToBeDeleted;
2602 Builder, AllocaIP, ToBeDeleted, TaskloopAllocaIP,
"global.tid",
false));
2604 TaskloopAllocaIP,
"lb",
false,
true);
2606 TaskloopAllocaIP,
"ub",
false,
true);
2608 TaskloopAllocaIP,
"step",
false,
true);
2611 OI->Inputs.insert(FakeLB);
2612 OI->Inputs.insert(FakeUB);
2613 OI->Inputs.insert(FakeStep);
2614 if (TaskContextStructPtrVal)
2615 OI->Inputs.insert(TaskContextStructPtrVal);
2616 assert(((TaskContextStructPtrVal && DupCB) ||
2617 (!TaskContextStructPtrVal && !DupCB)) &&
2618 "Task context struct ptr and duplication callback must be both set "
2624 unsigned ProgramAddressSpace =
M.getDataLayout().getProgramAddressSpace();
2628 {FakeLB->getType(), FakeUB->getType(), FakeStep->getType(), PointerTy});
2629 Expected<Value *> TaskDupFnOrErr = createTaskDuplicationFunction(
2632 if (!TaskDupFnOrErr) {
2635 Value *TaskDupFn = *TaskDupFnOrErr;
2637 OI->PostOutlineCB = [
this, Ident, LBVal, UBVal, StepVal, Untied,
2638 TaskloopAllocaBB, CLI, TaskDupFn, ToBeDeleted, IfCond,
2639 GrainSize, NoGroup, Sched, FakeLB, FakeUB, FakeStep,
2640 FakeSharedsTy, Final, Mergeable, Priority,
2642 FreeAgent](
Function &OutlinedFn)
mutable {
2644 assert(OutlinedFn.hasOneUse() &&
2645 "there must be a single user for the outlined function");
2652 Value *CastedLBVal =
2653 Builder.CreateIntCast(LBVal,
Builder.getInt64Ty(),
true,
"lb64");
2654 Value *CastedUBVal =
2655 Builder.CreateIntCast(UBVal,
Builder.getInt64Ty(),
true,
"ub64");
2656 Value *CastedStepVal =
2657 Builder.CreateIntCast(StepVal,
Builder.getInt64Ty(),
true,
"step64");
2659 Builder.SetInsertPoint(StaleCI);
2672 Builder.CreateCall(TaskgroupFn, {Ident, ThreadID});
2697 divideCeil(
M.getDataLayout().getTypeSizeInBits(Task), 8));
2699 AllocaInst *ArgStructAlloca =
2701 assert(ArgStructAlloca &&
2702 "Unable to find the alloca instruction corresponding to arguments "
2703 "for extracted function");
2704 std::optional<TypeSize> ArgAllocSize =
2707 "Unable to determine size of arguments for extracted function");
2708 Value *SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
2713 CallInst *TaskData =
Builder.CreateCall(
2714 TaskAllocFn, {Ident, ThreadID,
Flags,
2715 TaskSize, SharedsSize,
2720 Value *TaskShareds =
Builder.CreateLoad(VoidPtr, TaskData);
2726 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(0)});
2729 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(1)});
2732 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(2)});
2738 IfCond ?
Builder.CreateIntCast(IfCond,
Builder.getInt32Ty(),
true)
2744 Value *GrainSizeVal =
2745 GrainSize ?
Builder.CreateIntCast(GrainSize,
Builder.getInt64Ty(),
true)
2747 Value *TaskDup = TaskDupFn;
2749 Value *
Args[] = {Ident, ThreadID, TaskData, IfCondVal, Lb, Ub,
2750 Loadstep, NoGroupVal, SchedVal, GrainSizeVal, TaskDup};
2755 Builder.CreateCall(TaskloopFn, Args);
2762 Builder.CreateCall(EndTaskgroupFn, {Ident, ThreadID});
2767 Builder.SetInsertPoint(TaskloopAllocaBB, TaskloopAllocaBB->begin());
2769 LoadInst *SharedsOutlined =
2770 Builder.CreateLoad(VoidPtr, OutlinedFn.getArg(1));
2771 OutlinedFn.getArg(1)->replaceUsesWithIf(
2773 [SharedsOutlined](Use &U) {
return U.getUser() != SharedsOutlined; });
2776 Type *IVTy =
IV->getType();
2782 Value *TaskLB =
nullptr;
2783 Value *TaskUB =
nullptr;
2784 Value *TaskStep =
nullptr;
2785 Value *LoadTaskLB =
nullptr;
2786 Value *LoadTaskUB =
nullptr;
2787 Value *LoadTaskStep =
nullptr;
2788 for (Instruction &
I : *TaskloopAllocaBB) {
2789 if (
I.getOpcode() == Instruction::GetElementPtr) {
2792 switch (CI->getZExtValue()) {
2804 }
else if (
I.getOpcode() == Instruction::Load) {
2806 if (
Load.getPointerOperand() == TaskLB) {
2807 assert(TaskLB !=
nullptr &&
"Expected value for TaskLB");
2809 }
else if (
Load.getPointerOperand() == TaskUB) {
2810 assert(TaskUB !=
nullptr &&
"Expected value for TaskUB");
2812 }
else if (
Load.getPointerOperand() == TaskStep) {
2813 assert(TaskStep !=
nullptr &&
"Expected value for TaskStep");
2819 Builder.SetInsertPoint(CLI->getPreheader()->getTerminator());
2821 assert(LoadTaskLB !=
nullptr &&
"Expected value for LoadTaskLB");
2822 assert(LoadTaskUB !=
nullptr &&
"Expected value for LoadTaskUB");
2823 assert(LoadTaskStep !=
nullptr &&
"Expected value for LoadTaskStep");
2825 Builder.CreateSub(LoadTaskUB, LoadTaskLB), LoadTaskStep);
2826 Value *TripCount =
Builder.CreateAdd(TripCountMinusOne, One,
"trip_cnt");
2827 Value *CastedTripCount =
Builder.CreateIntCast(TripCount, IVTy,
true);
2828 Value *CastedTaskLB =
Builder.CreateIntCast(LoadTaskLB, IVTy,
true);
2830 CLI->setTripCount(CastedTripCount);
2832 Builder.SetInsertPoint(CLI->getBody(),
2833 CLI->getBody()->getFirstInsertionPt());
2835 if (NumOfCollapseLoops > 1) {
2841 Builder.CreateSub(CastedTaskLB, ConstantInt::get(IVTy, 1)));
2844 for (
auto IVUse = CLI->getIndVar()->uses().begin();
2845 IVUse != CLI->getIndVar()->uses().end(); IVUse++) {
2846 User *IVUser = IVUse->getUser();
2848 if (
Op->getOpcode() == Instruction::URem ||
2849 Op->getOpcode() == Instruction::UDiv) {
2854 for (User *User : UsersToReplace) {
2855 User->replaceUsesOfWith(CLI->getIndVar(), IVPlusTaskLB);
2872 assert(CLI->getIndVar()->getNumUses() == 3 &&
2873 "Canonical loop should have exactly three uses of the ind var");
2874 for (User *IVUser : CLI->getIndVar()->users()) {
2876 if (
Mul->getOpcode() == Instruction::Mul) {
2877 for (User *MulUser :
Mul->users()) {
2879 if (
Add->getOpcode() == Instruction::Add) {
2880 Add->setOperand(1, CastedTaskLB);
2889 FakeLB->replaceAllUsesWith(CastedLBVal);
2890 FakeUB->replaceAllUsesWith(CastedUBVal);
2891 FakeStep->replaceAllUsesWith(CastedStepVal);
2893 I->eraseFromParent();
2898 Builder.SetInsertPoint(TaskloopExitBB, TaskloopExitBB->
begin());
2904 M.getContext(),
M.getDataLayout().getPointerSizeInBits());
2914 bool Mergeable,
Value *EventHandle,
Value *Priority,
bool FreeAgent) {
2945 if (
Error Err = BodyGenCB(TaskAllocaIP, TaskBodyIP, TaskExitBB))
2948 auto OI = std::make_unique<OutlineInfo>();
2949 OI->EntryBB = TaskAllocaBB;
2950 OI->OuterAllocBB = AllocaIP.getNodeParent();
2951 OI->ExitBB = TaskExitBB;
2952 OI->OuterDeallocBBs.reserve(DeallocBlocks.
size());
2953 copy(DeallocBlocks, OI->OuterDeallocBBs.
end());
2958 Builder, AllocaIP, ToBeDeleted, TaskAllocaIP,
"global.tid",
false));
2960 OI->PostOutlineCB = [
this, Ident, Tied, Final, IfCondition, Dependencies,
2961 Affinities, Mergeable, Priority, EventHandle, FreeAgent,
2963 ToBeDeleted](
Function &OutlinedFn)
mutable {
2965 assert(OutlinedFn.hasOneUse() &&
2966 "there must be a single user for the outlined function");
2971 bool HasShareds = StaleCI->
arg_size() > 1;
2972 Builder.SetInsertPoint(StaleCI);
2999 bool UseMergedIf0Path = ConstIfCondition && ConstIfCondition->isZero();
3003 Flags =
Builder.CreateOr(FinalFlag, Flags);
3006 if (Mergeable || UseMergedIf0Path)
3020 divideCeil(
M.getDataLayout().getTypeSizeInBits(Task), 8));
3029 assert(ArgStructAlloca &&
3030 "Unable to find the alloca instruction corresponding to arguments "
3031 "for extracted function");
3032 std::optional<TypeSize> ArgAllocSize =
3035 "Unable to determine size of arguments for extracted function");
3036 SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
3042 TaskAllocFn, {Ident, ThreadID, Flags,
3043 TaskSize, SharedsSize,
3046 if (Affinities.
Count && Affinities.
Info) {
3048 OMPRTL___kmpc_omp_reg_task_with_affinity);
3059 OMPRTL___kmpc_task_allow_completion_event);
3063 Builder.CreatePointerBitCastOrAddrSpaceCast(EventHandle,
3065 EventVal =
Builder.CreatePtrToInt(EventVal,
Builder.getInt64Ty());
3066 Builder.CreateStore(EventVal, EventHandleAddr);
3072 Value *TaskShareds =
Builder.CreateLoad(VoidPtr, TaskData);
3087 Constant *Zero = ConstantInt::get(Int32Ty, 0);
3091 Builder.CreateInBoundsGEP(TaskPtr, TaskData, {Zero, Zero});
3094 VoidPtr, VoidPtr,
Builder.getInt32Ty(), VoidPtr, VoidPtr);
3096 TaskStructType, TaskGEP, {Zero, ConstantInt::get(Int32Ty, 4)});
3099 Value *CmplrData =
Builder.CreateInBoundsGEP(CmplrStructType,
3100 PriorityData, {Zero, Zero});
3101 Builder.CreateStore(Priority, CmplrData);
3104 Value *DepArray =
nullptr;
3105 Value *NumDeps =
nullptr;
3108 NumDeps = Dependencies.
NumDeps;
3109 }
else if (!Dependencies.
Deps.empty()) {
3111 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
3131 if (IfCondition && !UseMergedIf0Path) {
3136 Builder.GetInsertPoint()->getParent()->getTerminator();
3137 Instruction *ThenTI = IfTerminator, *ElseTI =
nullptr;
3138 Builder.SetInsertPoint(IfTerminator);
3141 Builder.SetInsertPoint(ElseTI);
3148 {Ident, ThreadID, NumDeps, DepArray,
3149 ConstantInt::get(
Builder.getInt32Ty(), 0),
3164 Builder.SetInsertPoint(ThenTI);
3172 {Ident, ThreadID, TaskData, NumDeps, DepArray,
3173 ConstantInt::get(
Builder.getInt32Ty(), 0),
3184 Builder.SetInsertPoint(TaskAllocaBB, TaskAllocaBB->
begin());
3186 LoadInst *Shareds =
Builder.CreateLoad(VoidPtr, OutlinedFn.getArg(1));
3187 OutlinedFn.getArg(1)->replaceUsesWithIf(
3188 Shareds, [Shareds](
Use &U) {
return U.getUser() != Shareds; });
3194 Builder.ClearInsertionPoint();
3196 I->eraseFromParent();
3200 Builder.SetInsertPoint(TaskExitBB, TaskExitBB->
begin());
3222 if (
Error Err = BodyGenCB(AllocaIP,
Builder.saveIP(), DeallocBlocks))
3225 Builder.SetInsertPoint(TaskgroupExitBB);
3268 unsigned CaseNumber = 0;
3269 for (
auto SectionCB : SectionCBs) {
3271 M.getContext(),
"omp_section_loop.body.case", CurFn,
Continue);
3273 Builder.SetInsertPoint(CaseBB);
3286 Value *LB = ConstantInt::get(I32Ty, 0);
3287 Value *UB = ConstantInt::get(I32Ty, SectionCBs.
size());
3288 Value *ST = ConstantInt::get(I32Ty, 1);
3290 Loc, LoopBodyGenCB, LB, UB, ST,
true,
false, AllocaIP,
"section_loop");
3295 applyStaticWorkshareLoop(
Loc.DL, *
LoopInfo, AllocaIP,
3296 WorksharingLoopType::ForStaticLoop, !IsNowait);
3302 assert(LoopFini &&
"Bad structure of static workshare loop finalization");
3306 assert(FiniInfo.DK == OMPD_sections &&
3307 "Unexpected finalization stack state!");
3308 if (
Error Err = FiniInfo.mergeFiniBB(
Builder, LoopFini))
3322 if (IP != IP.getNodeParent()->end())
3333 auto *CaseBB =
Loc.IP.getNodeParent();
3334 auto *CondBB = CaseBB->getSinglePredecessor()->getSinglePredecessor();
3335 auto *ExitBB = CondBB->getTerminator()->getSuccessor(1);
3337 IP =
I->getIterator();
3341 Directive OMPD = Directive::OMPD_sections;
3344 return EmitOMPInlinedRegion(OMPD,
nullptr,
nullptr, BodyGenCB, FiniCBWrapper,
3355Value *OpenMPIRBuilder::getGPUThreadID() {
3358 OMPRTL___kmpc_get_hardware_thread_id_in_block),
3362Value *OpenMPIRBuilder::getGPUWarpSize() {
3367Value *OpenMPIRBuilder::getNVPTXWarpID() {
3368 unsigned LaneIDBits =
Log2_32(
Config.getGridValue().GV_Warp_Size);
3369 return Builder.CreateAShr(getGPUThreadID(), LaneIDBits,
"nvptx_warp_id");
3372Value *OpenMPIRBuilder::getNVPTXLaneID() {
3373 unsigned LaneIDBits =
Log2_32(
Config.getGridValue().GV_Warp_Size);
3374 assert(LaneIDBits < 32 &&
"Invalid LaneIDBits size in NVPTX device.");
3375 unsigned LaneIDMask = ~0
u >> (32u - LaneIDBits);
3376 return Builder.CreateAnd(getGPUThreadID(),
Builder.getInt32(LaneIDMask),
3383 uint64_t FromSize =
M.getDataLayout().getTypeStoreSize(FromType);
3384 uint64_t ToSize =
M.getDataLayout().getTypeStoreSize(ToType);
3385 assert(FromSize > 0 &&
"From size must be greater than zero");
3386 assert(ToSize > 0 &&
"To size must be greater than zero");
3387 if (FromType == ToType)
3389 if (FromSize == ToSize)
3390 return Builder.CreateBitCast(From, ToType);
3392 return Builder.CreateIntCast(From, ToType,
true);
3398 Value *ValCastItem =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3399 CastItem,
Builder.getPtrTy(0));
3400 Builder.CreateStore(From, ValCastItem);
3401 return Builder.CreateLoad(ToType, CastItem);
3408 uint64_t Size =
M.getDataLayout().getTypeStoreSize(ElementType);
3409 assert(
Size <= 8 &&
"Unsupported bitwidth in shuffle instruction");
3413 Value *ElemCast = castValueToType(AllocaIP, Element, CastTy);
3415 Builder.CreateIntCast(getGPUWarpSize(),
Builder.getInt16Ty(),
true);
3417 Size <= 4 ? RuntimeFunction::OMPRTL___kmpc_shuffle_int32
3418 : RuntimeFunction::OMPRTL___kmpc_shuffle_int64);
3419 Value *WarpSizeCast =
3421 Value *ShuffleCall =
3426 return castValueToType(AllocaIP, ShuffleCall, ElementType);
3433 uint64_t Size =
M.getDataLayout().getTypeStoreSize(ElemType);
3445 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3446 Value *ElemPtr = DstAddr;
3447 Value *Ptr = SrcAddr;
3448 for (
unsigned IntSize = 8; IntSize >= 1; IntSize /= 2) {
3452 Ptr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3455 Builder.CreateGEP(ElemType, SrcAddr, {ConstantInt::get(IndexTy, 1)});
3456 ElemPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3460 if ((
Size / IntSize) > 1) {
3461 Value *PtrEnd =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3462 SrcAddrGEP,
Builder.getPtrTy());
3479 Builder.CreatePointerBitCastOrAddrSpaceCast(Ptr,
Builder.getPtrTy()));
3481 Builder.CreateICmpSGT(PtrDiff,
Builder.getInt64(IntSize - 1)), ThenBB,
3484 Value *Res = createRuntimeShuffleFunction(
3487 IntType, Ptr,
M.getDataLayout().getPrefTypeAlign(ElemType)),
3489 Builder.CreateAlignedStore(Res, ElemPtr,
3490 M.getDataLayout().getPrefTypeAlign(ElemType));
3492 Builder.CreateGEP(IntType, Ptr, {ConstantInt::get(IndexTy, 1)});
3493 Value *LocalElemPtr =
3494 Builder.CreateGEP(IntType, ElemPtr, {ConstantInt::get(IndexTy, 1)});
3502 Value *Res = createRuntimeShuffleFunction(
3503 AllocaIP,
Builder.CreateLoad(IntType, Ptr), IntType,
Offset);
3504 Builder.CreateStore(Res, ElemPtr);
3505 Ptr =
Builder.CreateGEP(IntType, Ptr, {ConstantInt::get(IndexTy, 1)});
3507 Builder.CreateGEP(IntType, ElemPtr, {ConstantInt::get(IndexTy, 1)});
3513Error OpenMPIRBuilder::emitReductionListCopy(
3518 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3519 Value *RemoteLaneOffset = CopyOptions.RemoteLaneOffset;
3523 for (
auto En :
enumerate(ReductionInfos)) {
3525 Value *SrcElementAddr =
nullptr;
3526 AllocaInst *DestAlloca =
nullptr;
3527 Value *DestElementAddr =
nullptr;
3528 Value *DestElementPtrAddr =
nullptr;
3530 bool ShuffleInElement =
false;
3533 bool UpdateDestListPtr =
false;
3537 ReductionArrayTy, SrcBase,
3538 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
3539 SrcElementAddr =
Builder.CreateLoad(
Builder.getPtrTy(), SrcElementPtrAddr);
3543 DestElementPtrAddr =
Builder.CreateInBoundsGEP(
3544 ReductionArrayTy, DestBase,
3545 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
3546 bool IsByRefElem = (!IsByRef.
empty() && IsByRef[En.index()]);
3552 Type *DestAllocaType =
3553 IsByRefElem ? RI.ByRefAllocatedType : RI.ElementType;
3554 DestAlloca =
Builder.CreateAlloca(DestAllocaType,
nullptr,
3555 ".omp.reduction.element");
3557 M.getDataLayout().getPrefTypeAlign(DestAllocaType));
3558 DestElementAddr = DestAlloca;
3561 DestElementAddr->
getName() +
".ascast");
3563 ShuffleInElement =
true;
3564 UpdateDestListPtr =
true;
3576 if (ShuffleInElement) {
3577 Type *ShuffleType = RI.ElementType;
3578 Value *ShuffleSrcAddr = SrcElementAddr;
3579 Value *ShuffleDestAddr = DestElementAddr;
3580 AllocaInst *LocalStorage =
nullptr;
3583 assert(RI.ByRefElementType &&
"Expected by-ref element type to be set");
3584 assert(RI.ByRefAllocatedType &&
3585 "Expected by-ref allocated type to be set");
3590 ShuffleType = RI.ByRefElementType;
3592 if (RI.DataPtrPtrGen) {
3595 Builder.saveIP(), ShuffleSrcAddr, ShuffleSrcAddr);
3598 return GenResult.takeError();
3607 LocalStorage =
Builder.CreateAlloca(ShuffleType);
3609 ShuffleDestAddr = LocalStorage;
3614 ShuffleDestAddr = DestElementAddr;
3618 shuffleAndStore(AllocaIP, ShuffleSrcAddr, ShuffleDestAddr, ShuffleType,
3619 RemoteLaneOffset, ReductionArrayTy, IsByRefElem);
3621 if (IsByRefElem && RI.DataPtrPtrGen) {
3623 Value *DestDescriptorAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3624 DestAlloca,
Builder.getPtrTy(),
".ascast");
3627 DestDescriptorAddr, LocalStorage, SrcElementAddr,
3628 RI.ByRefAllocatedType, RI.DataPtrPtrGen);
3631 return GenResult.takeError();
3634 switch (RI.EvaluationKind) {
3636 Value *Elem =
Builder.CreateLoad(RI.ElementType, SrcElementAddr);
3638 Builder.CreateStore(Elem, DestElementAddr);
3642 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
3643 RI.ElementType, SrcElementAddr, 0, 0,
".realp");
3645 RI.ElementType->getStructElementType(0), SrcRealPtr,
".real");
3647 RI.ElementType, SrcElementAddr, 0, 1,
".imagp");
3649 RI.ElementType->getStructElementType(1), SrcImgPtr,
".imag");
3651 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
3652 RI.ElementType, DestElementAddr, 0, 0,
".realp");
3653 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
3654 RI.ElementType, DestElementAddr, 0, 1,
".imagp");
3655 Builder.CreateStore(SrcReal, DestRealPtr);
3656 Builder.CreateStore(SrcImg, DestImgPtr);
3661 M.getDataLayout().getTypeStoreSize(RI.ElementType));
3663 DestElementAddr,
M.getDataLayout().getPrefTypeAlign(RI.ElementType),
3664 SrcElementAddr,
M.getDataLayout().getPrefTypeAlign(RI.ElementType),
3676 if (UpdateDestListPtr) {
3677 Value *CastDestAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3678 DestElementAddr,
Builder.getPtrTy(),
3679 DestElementAddr->
getName() +
".ascast");
3680 Builder.CreateStore(CastDestAddr, DestElementPtrAddr);
3687Expected<Function *> OpenMPIRBuilder::emitInterWarpCopyFunction(
3690 IRBuilder<>::InsertPointGuard IPG(
Builder);
3691 LLVMContext &Ctx =
M.getContext();
3693 Builder.getVoidTy(), {Builder.getPtrTy(), Builder.getInt32Ty()},
3697 "_omp_reduction_inter_warp_copy_func", &
M);
3703 Builder.SetInsertPoint(EntryBB);
3721 StringRef TransferMediumName =
3722 "__openmp_nvptx_data_transfer_temporary_storage";
3723 GlobalVariable *TransferMedium =
M.getGlobalVariable(TransferMediumName);
3724 unsigned WarpSize =
Config.getGridValue().GV_Warp_Size;
3726 if (!TransferMedium) {
3727 TransferMedium =
new GlobalVariable(
3735 Value *GPUThreadID = getGPUThreadID();
3737 Value *LaneID = getNVPTXLaneID();
3739 Value *WarpID = getNVPTXWarpID();
3745 AllocaInst *ReduceListAlloca =
Builder.CreateAlloca(
3746 Arg0Type,
nullptr, ReduceListArg->
getName() +
".addr");
3747 AllocaInst *NumWarpsAlloca =
3748 Builder.CreateAlloca(Arg1Type,
nullptr, NumWarpsArg->
getName() +
".addr");
3749 Value *ReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3750 ReduceListAlloca, Arg0Type, ReduceListAlloca->
getName() +
".ascast");
3751 Value *NumWarpsAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3752 NumWarpsAlloca,
Builder.getPtrTy(0),
3753 NumWarpsAlloca->
getName() +
".ascast");
3754 Builder.CreateStore(ReduceListArg, ReduceListAddrCast);
3755 Builder.CreateStore(NumWarpsArg, NumWarpsAddrCast);
3764 for (
auto En :
enumerate(ReductionInfos)) {
3770 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
3771 unsigned RealTySize =
M.getDataLayout().getTypeAllocSize(
3772 IsByRefElem ? RI.ByRefElementType : RI.ElementType);
3773 for (
unsigned TySize = 4; TySize > 0 && RealTySize > 0; TySize /= 2) {
3776 unsigned NumIters = RealTySize / TySize;
3779 Value *Cnt =
nullptr;
3780 Value *CntAddr =
nullptr;
3787 Builder.CreateAlloca(
Builder.getInt32Ty(),
nullptr,
".cnt.addr");
3789 CntAddr =
Builder.CreateAddrSpaceCast(CntAddr,
Builder.getPtrTy(),
3790 CntAddr->
getName() +
".ascast");
3802 Cnt, ConstantInt::get(
Builder.getInt32Ty(), NumIters));
3803 Builder.CreateCondBr(Cmp, BodyBB, ExitBB);
3810 omp::Directive::OMPD_unknown,
3814 return BarrierIP1.takeError();
3820 Value *IsWarpMaster =
Builder.CreateIsNull(LaneID,
"warp_master");
3821 Builder.CreateCondBr(IsWarpMaster, ThenBB, ElseBB);
3825 auto *RedListArrayTy =
3828 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3830 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
3831 {ConstantInt::get(IndexTy, 0),
3832 ConstantInt::get(IndexTy, En.index())});
3836 if (IsByRefElem && RI.DataPtrPtrGen) {
3838 RI.DataPtrPtrGen(
Builder.saveIP(), ElemPtr, ElemPtr);
3841 return GenRes.takeError();
3852 ArrayTy, TransferMedium, {
Builder.getInt64(0), WarpID});
3857 Builder.CreateStore(Elem, MediumPtr,
3869 omp::Directive::OMPD_unknown,
3873 return BarrierIP2.takeError();
3880 Value *NumWarpsVal =
3883 Value *IsActiveThread =
3884 Builder.CreateICmpULT(GPUThreadID, NumWarpsVal,
"is_active_thread");
3885 Builder.CreateCondBr(IsActiveThread, W0ThenBB, W0ElseBB);
3892 ArrayTy, TransferMedium, {
Builder.getInt64(0), GPUThreadID});
3894 Value *TargetElemPtrPtr =
3895 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
3896 {ConstantInt::get(IndexTy, 0),
3897 ConstantInt::get(IndexTy, En.index())});
3898 Value *TargetElemPtrVal =
3900 Value *TargetElemPtr = TargetElemPtrVal;
3902 if (IsByRefElem && RI.DataPtrPtrGen) {
3904 RI.DataPtrPtrGen(
Builder.saveIP(), TargetElemPtr, TargetElemPtr);
3907 return GenRes.takeError();
3909 TargetElemPtr =
Builder.CreateLoad(
Builder.getPtrTy(), TargetElemPtr);
3917 Value *SrcMediumValue =
3918 Builder.CreateLoad(CType, SrcMediumPtrVal,
true);
3919 Builder.CreateStore(SrcMediumValue, TargetElemPtr);
3929 Cnt, ConstantInt::get(
Builder.getInt32Ty(), 1));
3930 Builder.CreateStore(Cnt, CntAddr,
false);
3932 auto *CurFn =
Builder.GetInsertBlock()->getParent();
3936 RealTySize %= TySize;
3945Expected<Function *> OpenMPIRBuilder::emitShuffleAndReduceFunction(
3948 LLVMContext &Ctx =
M.getContext();
3949 IRBuilder<>::InsertPointGuard IPG(
Builder);
3950 FunctionType *FuncTy =
3952 {Builder.getPtrTy(), Builder.getInt16Ty(),
3953 Builder.getInt16Ty(), Builder.getInt16Ty()},
3957 "_omp_reduction_shuffle_and_reduce_func", &
M);
3968 Builder.SetInsertPoint(EntryBB);
3980 Type *ReduceListArgType = ReduceListArg->
getType();
3984 ReduceListArgType,
nullptr, ReduceListArg->
getName() +
".addr");
3985 Value *LaneIdAlloca =
Builder.CreateAlloca(LaneIDArgType,
nullptr,
3986 LaneIDArg->
getName() +
".addr");
3988 LaneIDArgType,
nullptr, RemoteLaneOffsetArg->
getName() +
".addr");
3989 Value *AlgoVerAlloca =
Builder.CreateAlloca(LaneIDArgType,
nullptr,
3990 AlgoVerArg->
getName() +
".addr");
3997 RedListArrayTy,
nullptr,
".omp.reduction.remote_reduce_list");
3999 Value *ReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4000 ReduceListAlloca, ReduceListArgType,
4001 ReduceListAlloca->
getName() +
".ascast");
4002 Value *LaneIdAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4003 LaneIdAlloca, LaneIDArgPtrType, LaneIdAlloca->
getName() +
".ascast");
4004 Value *RemoteLaneOffsetAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4005 RemoteLaneOffsetAlloca, LaneIDArgPtrType,
4006 RemoteLaneOffsetAlloca->
getName() +
".ascast");
4007 Value *AlgoVerAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4008 AlgoVerAlloca, LaneIDArgPtrType, AlgoVerAlloca->
getName() +
".ascast");
4009 Value *RemoteListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4010 RemoteReductionListAlloca,
Builder.getPtrTy(),
4011 RemoteReductionListAlloca->
getName() +
".ascast");
4013 Builder.CreateStore(ReduceListArg, ReduceListAddrCast);
4014 Builder.CreateStore(LaneIDArg, LaneIdAddrCast);
4015 Builder.CreateStore(RemoteLaneOffsetArg, RemoteLaneOffsetAddrCast);
4016 Builder.CreateStore(AlgoVerArg, AlgoVerAddrCast);
4018 Value *ReduceList =
Builder.CreateLoad(ReduceListArgType, ReduceListAddrCast);
4019 Value *LaneId =
Builder.CreateLoad(LaneIDArgType, LaneIdAddrCast);
4020 Value *RemoteLaneOffset =
4021 Builder.CreateLoad(LaneIDArgType, RemoteLaneOffsetAddrCast);
4022 Value *AlgoVer =
Builder.CreateLoad(LaneIDArgType, AlgoVerAddrCast);
4029 Error EmitRedLsCpRes = emitReductionListCopy(
4031 ReduceList, RemoteListAddrCast, IsByRef,
4032 {RemoteLaneOffset,
nullptr,
nullptr});
4035 return EmitRedLsCpRes;
4060 Value *LaneComp =
Builder.CreateICmpULT(LaneId, RemoteLaneOffset);
4065 Value *Algo2AndLaneIdComp =
Builder.CreateAnd(Algo2, LaneIdComp);
4066 Value *RemoteOffsetComp =
4068 Value *CondAlgo2 =
Builder.CreateAnd(Algo2AndLaneIdComp, RemoteOffsetComp);
4069 Value *CA0OrCA1 =
Builder.CreateOr(CondAlgo0, CondAlgo1);
4070 Value *CondReduce =
Builder.CreateOr(CA0OrCA1, CondAlgo2);
4076 Builder.CreateCondBr(CondReduce, ThenBB, ElseBB);
4078 Value *LocalReduceListPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4079 ReduceList,
Builder.getPtrTy());
4080 Value *RemoteReduceListPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4081 RemoteListAddrCast,
Builder.getPtrTy());
4083 ->addFnAttr(Attribute::NoUnwind);
4094 Value *LaneIdGtOffset =
Builder.CreateICmpUGE(LaneId, RemoteLaneOffset);
4095 Value *CondCopy =
Builder.CreateAnd(Algo1, LaneIdGtOffset);
4100 Builder.CreateCondBr(CondCopy, CpyThenBB, CpyElseBB);
4104 EmitRedLsCpRes = emitReductionListCopy(
4106 RemoteListAddrCast, ReduceList, IsByRef);
4109 return EmitRedLsCpRes;
4124OpenMPIRBuilder::generateReductionDescriptor(
4126 Type *DescriptorType,
4132 Value *DescriptorSize =
4133 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(DescriptorType));
4135 DescriptorAddr,
M.getDataLayout().getPrefTypeAlign(DescriptorType),
4136 SrcDescriptorAddr,
M.getDataLayout().getPrefTypeAlign(DescriptorType),
4140 Value *DataPtrField;
4142 DataPtrPtrGen(
Builder.saveIP(), DescriptorAddr, DataPtrField);
4145 return GenResult.takeError();
4148 DataPtr,
Builder.getPtrTy(),
".ascast"),
4154Expected<Value *> OpenMPIRBuilder::createReductionDescriptorCopy(
4156 Value *SrcDescriptorAddr,
Type *DescriptorPtrTy,
const Twine &Name) {
4160 AllocaInst *DescriptorAlloca =
4161 Builder.CreateAlloca(RI.ByRefAllocatedType,
nullptr, Name);
4163 M.getDataLayout().getPrefTypeAlign(RI.ByRefAllocatedType));
4164 Value *DescriptorAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4165 DescriptorAlloca, DescriptorPtrTy,
4166 DescriptorAlloca->
getName() +
".ascast");
4171 generateReductionDescriptor(DescriptorAddr, DataPtr, SrcDescriptorAddr,
4172 RI.ByRefAllocatedType, RI.DataPtrPtrGen);
4174 return GenResult.takeError();
4176 return DescriptorAddr;
4179Expected<Function *> OpenMPIRBuilder::emitListToGlobalCopyFunction(
4182 IRBuilder<>::InsertPointGuard IPG(
Builder);
4183 LLVMContext &Ctx =
M.getContext();
4186 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4190 "_omp_reduction_list_to_global_copy_func", &
M);
4197 Builder.SetInsertPoint(EntryBlock);
4208 BufferArg->
getName() +
".addr");
4212 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4213 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4214 BufferArgAlloca,
Builder.getPtrTy(),
4215 BufferArgAlloca->
getName() +
".ascast");
4216 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4217 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4218 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4219 ReduceListArgAlloca,
Builder.getPtrTy(),
4220 ReduceListArgAlloca->
getName() +
".ascast");
4222 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4223 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4224 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4226 Value *LocalReduceList =
4228 Value *BufferArgVal =
4232 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4233 for (
auto En :
enumerate(ReductionInfos)) {
4235 auto *RedListArrayTy =
4239 RedListArrayTy, LocalReduceList,
4240 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4246 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferArgVal, Idxs);
4248 ReductionsBufferTy, BufferVD, 0, En.index());
4250 switch (RI.EvaluationKind) {
4252 Value *TargetElement;
4254 if (IsByRef.
empty() || !IsByRef[En.index()]) {
4255 TargetElement =
Builder.CreateLoad(RI.ElementType, ElemPtr);
4257 if (RI.DataPtrPtrGen) {
4259 RI.DataPtrPtrGen(
Builder.saveIP(), ElemPtr, ElemPtr);
4262 return GenResult.takeError();
4266 TargetElement =
Builder.CreateLoad(RI.ByRefElementType, ElemPtr);
4269 Builder.CreateStore(TargetElement, GlobVal);
4273 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4274 RI.ElementType, ElemPtr, 0, 0,
".realp");
4276 RI.ElementType->getStructElementType(0), SrcRealPtr,
".real");
4278 RI.ElementType, ElemPtr, 0, 1,
".imagp");
4280 RI.ElementType->getStructElementType(1), SrcImgPtr,
".imag");
4282 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4283 RI.ElementType, GlobVal, 0, 0,
".realp");
4284 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
4285 RI.ElementType, GlobVal, 0, 1,
".imagp");
4286 Builder.CreateStore(SrcReal, DestRealPtr);
4287 Builder.CreateStore(SrcImg, DestImgPtr);
4292 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(RI.ElementType));
4294 GlobVal,
M.getDataLayout().getPrefTypeAlign(RI.ElementType), ElemPtr,
4295 M.getDataLayout().getPrefTypeAlign(RI.ElementType), SizeVal,
false);
4305Expected<Function *> OpenMPIRBuilder::emitListToGlobalReduceFunction(
4308 IRBuilder<>::InsertPointGuard IPG(
Builder);
4309 LLVMContext &Ctx =
M.getContext();
4312 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4316 "_omp_reduction_list_to_global_reduce_func", &
M);
4323 Builder.SetInsertPoint(EntryBlock);
4334 BufferArg->
getName() +
".addr");
4338 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4339 auto *RedListArrayTy =
4344 Value *LocalReduceList =
4345 Builder.CreateAlloca(RedListArrayTy,
nullptr,
".omp.reduction.red_list");
4349 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4350 BufferArgAlloca,
Builder.getPtrTy(),
4351 BufferArgAlloca->
getName() +
".ascast");
4352 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4353 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4354 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4355 ReduceListArgAlloca,
Builder.getPtrTy(),
4356 ReduceListArgAlloca->
getName() +
".ascast");
4357 Value *LocalReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4358 LocalReduceList,
Builder.getPtrTy(),
4359 LocalReduceList->
getName() +
".ascast");
4361 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4362 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4363 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4368 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4369 for (
auto En :
enumerate(ReductionInfos)) {
4372 Value *TargetElementPtrPtr =
Builder.CreateInBoundsGEP(
4373 RedListArrayTy, LocalReduceListAddrCast,
4374 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4376 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4378 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4379 ReductionsBufferTy, BufferVD, 0, En.index());
4381 if (!IsByRef.
empty() && IsByRef[En.index()] && RI.DataPtrPtrGen) {
4385 Value *SrcElementPtrPtr =
4386 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
4387 {ConstantInt::get(IndexTy, 0),
4388 ConstantInt::get(IndexTy, En.index())});
4389 Value *SrcDescriptorAddr =
4393 Expected<Value *> ByRefAlloc = createReductionDescriptorCopy(
4394 AllocaIP, RI, GlobValPtr, SrcDescriptorAddr,
Builder.getPtrTy());
4398 Builder.CreateStore(*ByRefAlloc, TargetElementPtrPtr);
4400 Builder.CreateStore(GlobValPtr, TargetElementPtrPtr);
4408 ->addFnAttr(Attribute::NoUnwind);
4413Expected<Function *> OpenMPIRBuilder::emitGlobalToListCopyFunction(
4416 IRBuilder<>::InsertPointGuard IPG(
Builder);
4417 LLVMContext &Ctx =
M.getContext();
4420 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4424 "_omp_reduction_global_to_list_copy_func", &
M);
4431 Builder.SetInsertPoint(EntryBlock);
4442 BufferArg->
getName() +
".addr");
4446 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4447 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4448 BufferArgAlloca,
Builder.getPtrTy(),
4449 BufferArgAlloca->
getName() +
".ascast");
4450 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4451 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4452 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4453 ReduceListArgAlloca,
Builder.getPtrTy(),
4454 ReduceListArgAlloca->
getName() +
".ascast");
4455 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4456 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4457 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4459 Value *LocalReduceList =
4464 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4465 for (
auto En :
enumerate(ReductionInfos)) {
4466 const OpenMPIRBuilder::ReductionInfo &RI = En.value();
4467 auto *RedListArrayTy =
4471 RedListArrayTy, LocalReduceList,
4472 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4477 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4478 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4479 ReductionsBufferTy, BufferVD, 0, En.index());
4485 if (!IsByRef.
empty() && IsByRef[En.index()]) {
4492 return GenResult.takeError();
4498 Value *TargetElement =
Builder.CreateLoad(ElemType, GlobValPtr);
4499 Builder.CreateStore(TargetElement, ElemPtr);
4503 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4512 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4514 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
4516 Builder.CreateStore(SrcReal, DestRealPtr);
4517 Builder.CreateStore(SrcImg, DestImgPtr);
4524 ElemPtr,
M.getDataLayout().getPrefTypeAlign(RI.
ElementType),
4525 GlobValPtr,
M.getDataLayout().getPrefTypeAlign(RI.
ElementType),
4536Expected<Function *> OpenMPIRBuilder::emitGlobalToListReduceFunction(
4539 IRBuilder<>::InsertPointGuard IPG(
Builder);
4540 LLVMContext &Ctx =
M.getContext();
4543 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4547 "_omp_reduction_global_to_list_reduce_func", &
M);
4554 Builder.SetInsertPoint(EntryBlock);
4565 BufferArg->
getName() +
".addr");
4569 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4575 Value *LocalReduceList =
4576 Builder.CreateAlloca(RedListArrayTy,
nullptr,
".omp.reduction.red_list");
4580 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4581 BufferArgAlloca,
Builder.getPtrTy(),
4582 BufferArgAlloca->
getName() +
".ascast");
4583 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4584 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4585 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4586 ReduceListArgAlloca,
Builder.getPtrTy(),
4587 ReduceListArgAlloca->
getName() +
".ascast");
4588 Value *ReductionList =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4589 LocalReduceList,
Builder.getPtrTy(),
4590 LocalReduceList->
getName() +
".ascast");
4592 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4593 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4594 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4599 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4600 for (
auto En :
enumerate(ReductionInfos)) {
4603 Value *TargetElementPtrPtr =
Builder.CreateInBoundsGEP(
4604 RedListArrayTy, ReductionList,
4605 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4608 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4609 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4610 ReductionsBufferTy, BufferVD, 0, En.index());
4612 if (!IsByRef.
empty() && IsByRef[En.index()] && RI.DataPtrPtrGen) {
4614 Value *ReduceListVal =
4616 Value *SrcElementPtrPtr =
4617 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceListVal,
4618 {ConstantInt::get(IndexTy, 0),
4619 ConstantInt::get(IndexTy, En.index())});
4620 Value *SrcDescriptorAddr =
4624 Expected<Value *> ByRefAlloc = createReductionDescriptorCopy(
4625 AllocaIP, RI, GlobValPtr, SrcDescriptorAddr,
Builder.getPtrTy());
4629 Builder.CreateStore(*ByRefAlloc, TargetElementPtrPtr);
4631 Builder.CreateStore(GlobValPtr, TargetElementPtrPtr);
4639 ->addFnAttr(Attribute::NoUnwind);
4644std::string OpenMPIRBuilder::getReductionFuncName(StringRef Name)
const {
4645 std::string Suffix =
4647 return (Name + Suffix).str();
4650Expected<Function *> OpenMPIRBuilder::createReductionFunction(
4653 AttributeList FuncAttrs) {
4654 IRBuilder<>::InsertPointGuard IPG(
Builder);
4656 {Builder.getPtrTy(), Builder.getPtrTy()},
4658 std::string
Name = getReductionFuncName(ReducerName);
4667 Builder.SetInsertPoint(EntryBB);
4672 Value *LHSArrayPtr =
nullptr;
4673 Value *RHSArrayPtr =
nullptr;
4680 Builder.CreateAlloca(Arg0Type,
nullptr, Arg0->
getName() +
".addr");
4682 Builder.CreateAlloca(Arg1Type,
nullptr, Arg1->
getName() +
".addr");
4683 Value *LHSAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4684 LHSAlloca, Arg0Type, LHSAlloca->
getName() +
".ascast");
4685 Value *RHSAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4686 RHSAlloca, Arg1Type, RHSAlloca->
getName() +
".ascast");
4687 Builder.CreateStore(Arg0, LHSAddrCast);
4688 Builder.CreateStore(Arg1, RHSAddrCast);
4689 LHSArrayPtr =
Builder.CreateLoad(Arg0Type, LHSAddrCast);
4690 RHSArrayPtr =
Builder.CreateLoad(Arg1Type, RHSAddrCast);
4694 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4696 for (
auto En :
enumerate(ReductionInfos)) {
4699 RedArrayTy, RHSArrayPtr,
4700 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4702 Value *RHSPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4703 RHSI8Ptr, RI.PrivateVariable->getType(),
4704 RHSI8Ptr->
getName() +
".ascast");
4707 RedArrayTy, LHSArrayPtr,
4708 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4710 Value *LHSPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4711 LHSI8Ptr, RI.Variable->getType(), LHSI8Ptr->
getName() +
".ascast");
4720 if (!IsByRef.
empty() && !IsByRef[En.index()]) {
4721 LHS =
Builder.CreateLoad(RI.ElementType, LHSPtr);
4722 RHS =
Builder.CreateLoad(RI.ElementType, RHSPtr);
4729 return AfterIP.takeError();
4730 if (!
Builder.GetInsertBlock())
4731 return ReductionFunc;
4735 if (!IsByRef.
empty() && !IsByRef[En.index()])
4736 Builder.CreateStore(Reduced, LHSPtr);
4741 for (
auto En :
enumerate(ReductionInfos)) {
4742 unsigned Index = En.index();
4744 Value *LHSFixupPtr, *RHSFixupPtr;
4745 Builder.restoreIP(RI.ReductionGenClang(
4746 Builder.saveIP(), Index, &LHSFixupPtr, &RHSFixupPtr, ReductionFunc));
4751 LHSPtrs[Index], [ReductionFunc](
const Use &U) {
4756 RHSPtrs[Index], [ReductionFunc](
const Use &U) {
4770 return ReductionFunc;
4778 assert(RI.Variable &&
"expected non-null variable");
4779 assert(RI.PrivateVariable &&
"expected non-null private variable");
4780 assert((RI.ReductionGen || RI.ReductionGenClang) &&
4781 "expected non-null reduction generator callback");
4784 RI.Variable->getType() == RI.PrivateVariable->getType() &&
4785 "expected variables and their private equivalents to have the same "
4788 assert(RI.Variable->getType()->isPointerTy() &&
4789 "expected variables to be pointers");
4806 ArrayRef<bool> IsByRef,
bool IsNoWait,
bool IsTeamsReduction,
bool IsSPMD,
4808 Value *SrcLocInfo) {
4822 if (ReductionInfos.
size() == 0)
4831 Builder.SetInsertPoint(InsertBlock, InsertBlock->
end());
4836 AttrBuilder AttrBldr(Ctx);
4838 AttrBldr.addAttribute(Attr);
4839 AttrBldr.removeAttribute(Attribute::OptimizeNone);
4840 FuncAttrs = FuncAttrs.addFnAttributes(Ctx, AttrBldr);
4844 Builder.GetInsertBlock()->getParent()->getName(), ReductionInfos, IsByRef,
4846 if (!ReductionResult)
4848 Function *ReductionFunc = *ReductionResult;
4852 if (GridValue.has_value())
4853 Config.setGridValue(GridValue.value());
4868 Builder.getPtrTy(
M.getDataLayout().getProgramAddressSpace());
4872 Value *ReductionListAlloca =
4873 Builder.CreateAlloca(RedArrayTy,
nullptr,
".omp.reduction.red_list");
4874 Value *ReductionList =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4875 ReductionListAlloca, PtrTy, ReductionListAlloca->
getName() +
".ascast");
4878 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4879 for (
auto En :
enumerate(ReductionInfos)) {
4882 RedArrayTy, ReductionList,
4883 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4886 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
4891 Builder.CreatePointerBitCastOrAddrSpaceCast(PrivateVar, PtrTy);
4892 Builder.CreateStore(CastElem, ElemPtr);
4896 ReductionInfos, ReductionFunc, FuncAttrs, IsByRef);
4902 emitInterWarpCopyFunction(
Loc, ReductionInfos, FuncAttrs, IsByRef);
4908 Value *RL =
Builder.CreatePointerBitCastOrAddrSpaceCast(ReductionList, PtrTy);
4917 unsigned MaxDataSize = 0;
4919 for (
auto En :
enumerate(ReductionInfos)) {
4923 Type *RedTypeArg = (!IsByRef.
empty() && IsByRef[En.index()])
4924 ? En.value().ByRefElementType
4925 : En.value().ElementType;
4926 auto Size =
M.getDataLayout().getTypeStoreSize(RedTypeArg);
4927 if (
Size > MaxDataSize)
4931 Value *ReductionDataSize =
4932 Builder.getInt64(MaxDataSize * ReductionInfos.
size());
4936 Function *CopyScratchToListFunc =
nullptr;
4938 Value *ScratchForCopyBack =
nullptr;
4941 Value *RLForCopyBack = RL;
4943 bool IsAtomicReduction =
4946 if (!IsTeamsReduction) {
4947 Value *SarFuncCast =
4948 Builder.CreatePointerBitCastOrAddrSpaceCast(*SarFunc, FuncPtrTy);
4950 Builder.CreatePointerBitCastOrAddrSpaceCast(WcFunc, FuncPtrTy);
4951 Value *Args[] = {SrcLocInfo, ReductionDataSize, RL, SarFuncCast,
4954 RuntimeFunction::OMPRTL___kmpc_nvptx_parallel_reduce_nowait_v2);
4956 }
else if (IsAtomicReduction) {
4960 RuntimeFunction::OMPRTL___kmpc_is_team_main_thread);
4965 Ctx, ReductionTypeArgs,
"struct._globalized_locals_ty");
4968 ReductionInfos, ReductionsBufferTy, FuncAttrs, IsByRef);
4973 ReductionInfos, ReductionsBufferTy, FuncAttrs, IsByRef);
4978 ReductionInfos, ReductionFunc, ReductionsBufferTy, FuncAttrs, IsByRef);
5001 Value *RuntimeRL = RL;
5008 ReductionsBufferTy,
nullptr,
".omp.reduction.scratch");
5009 Value *PerThreadScratch =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5010 PerThreadScratchAlloca, PtrTy,
5011 PerThreadScratchAlloca->
getName() +
".ascast");
5014 Value *PerThreadRedListAlloca =
5015 Builder.CreateAlloca(RedArrayTy,
nullptr,
5016 ".omp.reduction.per_thread_red_list");
5017 RuntimeRL =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5018 PerThreadRedListAlloca, PtrTy,
5019 PerThreadRedListAlloca->
getName() +
".ascast");
5024 for (
auto En :
enumerate(ReductionInfos)) {
5026 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
5029 ReductionsBufferTy, PerThreadScratch, 0, En.index());
5030 Value *Slot =
Builder.CreateConstInBoundsGEP2_32(RedArrayTy, RuntimeRL,
5033 Value *RuntimeListEntry = FieldPtr;
5035 Value *SrcDescriptor =
5038 AllocaIP, RI, FieldPtr, SrcDescriptor, PtrTy);
5041 RuntimeListEntry = *Descriptor;
5043 Builder.CreateStore(RuntimeListEntry, Slot);
5049 Type *CopyArg0Ty = (*LtGCFunc)->getFunctionType()->getParamType(0);
5050 Type *CopyArg2Ty = (*LtGCFunc)->getFunctionType()->getParamType(2);
5051 ScratchForCopyBack =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5052 PerThreadScratch, CopyArg0Ty);
5054 Builder.CreatePointerBitCastOrAddrSpaceCast(RL, CopyArg2Ty);
5062 *LtGCFunc, {ScratchForCopyBack,
Builder.getInt32(0), RLForCopyBack});
5063 CopyScratchToListFunc = *GtLCFunc;
5066 Value *Args3[] = {SrcLocInfo, RuntimeRL, *SarFunc, WcFunc,
5067 *LtGCFunc, *GtLCFunc, *GtLRFunc};
5070 RuntimeFunction::OMPRTL___kmpc_gpu_xteam_reduce_nowait);
5090 if (ScratchForCopyBack) {
5093 CopyScratchToListFunc,
5094 {ScratchForCopyBack,
Builder.getInt32(0), RLForCopyBack});
5098 for (
auto En :
enumerate(ReductionInfos)) {
5104 if (IsAtomicReduction) {
5120 Value *LHSPtr, *RHSPtr;
5122 &LHSPtr, &RHSPtr, CurFunc));
5128 RedValue =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5130 if (RHSPtr->
getType() != RHS->getType())
5132 Builder.CreatePointerBitCastOrAddrSpaceCast(RHS, RHSPtr->
getType());
5143 if (IsByRef.
empty() || !IsByRef[En.index()]) {
5145 "red.value." +
Twine(En.index()));
5156 if (!IsByRef.
empty() && !IsByRef[En.index()])
5161 if (ContinuationBlock) {
5162 Builder.CreateBr(ContinuationBlock);
5163 Builder.SetInsertPoint(ContinuationBlock);
5165 Config.setEmitLLVMUsed();
5176 ".omp.reduction.func", &M);
5187 Builder.SetInsertPoint(ReductionFuncBlock);
5189 Value *LHSArrayPtr =
nullptr;
5190 Value *RHSArrayPtr =
nullptr;
5201 Builder.CreateAlloca(Arg0Type,
nullptr, Arg0->
getName() +
".addr");
5203 Builder.CreateAlloca(Arg1Type,
nullptr, Arg1->
getName() +
".addr");
5204 Value *LHSAddrCast =
5205 Builder.CreatePointerBitCastOrAddrSpaceCast(LHSAlloca, Arg0Type);
5206 Value *RHSAddrCast =
5207 Builder.CreatePointerBitCastOrAddrSpaceCast(RHSAlloca, Arg1Type);
5208 Builder.CreateStore(Arg0, LHSAddrCast);
5209 Builder.CreateStore(Arg1, RHSAddrCast);
5210 LHSArrayPtr = Builder.CreateLoad(Arg0Type, LHSAddrCast);
5211 RHSArrayPtr = Builder.CreateLoad(Arg1Type, RHSAddrCast);
5213 LHSArrayPtr = ReductionFunc->
getArg(0);
5214 RHSArrayPtr = ReductionFunc->
getArg(1);
5217 unsigned NumReductions = ReductionInfos.
size();
5220 for (
auto En :
enumerate(ReductionInfos)) {
5222 Value *LHSI8PtrPtr = Builder.CreateConstInBoundsGEP2_64(
5223 RedArrayTy, LHSArrayPtr, 0, En.index());
5224 Value *LHSI8Ptr = Builder.CreateLoad(Builder.getPtrTy(), LHSI8PtrPtr);
5225 Value *LHSPtr = Builder.CreatePointerBitCastOrAddrSpaceCast(
5228 Value *RHSI8PtrPtr = Builder.CreateConstInBoundsGEP2_64(
5229 RedArrayTy, RHSArrayPtr, 0, En.index());
5230 Value *RHSI8Ptr = Builder.CreateLoad(Builder.getPtrTy(), RHSI8PtrPtr);
5231 Value *RHSPtr = Builder.CreatePointerBitCastOrAddrSpaceCast(
5240 Builder.restoreIP(*AfterIP);
5242 if (!Builder.GetInsertBlock())
5246 if (!IsByRef[En.index()])
5247 Builder.CreateStore(Reduced, LHSPtr);
5249 Builder.CreateRetVoid();
5256 bool IsNoWait,
bool IsTeamsReduction) {
5260 IsByRef, IsNoWait, IsTeamsReduction);
5267 if (ReductionInfos.
size() == 0)
5277 unsigned NumReductions = ReductionInfos.
size();
5279 Builder.SetInsertPoint(AllocaIP.getNodeParent()->getTerminator());
5280 Value *RedArray =
Builder.CreateAlloca(RedArrayTy,
nullptr,
"red.array");
5282 Builder.SetInsertPoint(InsertBlock, InsertBlock->
end());
5287 for (
auto En :
enumerate(ReductionInfos)) {
5288 unsigned Index = En.index();
5290 Value *RedArrayElemPtr =
Builder.CreateConstInBoundsGEP2_64(
5291 RedArrayTy, RedArray, 0, Index,
"red.array.elem." +
Twine(Index));
5298 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
5308 ? IdentFlag::OMP_IDENT_FLAG_ATOMIC_REDUCE
5313 unsigned RedArrayByteSize =
DL.getTypeStoreSize(RedArrayTy);
5314 Constant *RedArraySize = ConstantInt::get(IndexTy, RedArrayByteSize);
5316 Value *Lock = getOMPCriticalRegionLock(
".reduction");
5318 IsNoWait ? RuntimeFunction::OMPRTL___kmpc_reduce_nowait
5319 : RuntimeFunction::OMPRTL___kmpc_reduce);
5322 {Ident, ThreadId, NumVariables, RedArraySize,
5323 RedArray, ReductionFunc, Lock},
5334 Builder.CreateSwitch(ReduceCall, ContinuationBlock, 2);
5335 Switch->addCase(
Builder.getInt32(1), NonAtomicRedBlock);
5336 Switch->addCase(
Builder.getInt32(2), AtomicRedBlock);
5341 Builder.SetInsertPoint(NonAtomicRedBlock);
5342 for (
auto En :
enumerate(ReductionInfos)) {
5348 if (!IsByRef[En.index()]) {
5350 "red.value." +
Twine(En.index()));
5352 Value *PrivateRedValue =
5354 "red.private.value." +
Twine(En.index()));
5362 if (!
Builder.GetInsertBlock())
5365 if (!IsByRef[En.index()])
5369 IsNoWait ? RuntimeFunction::OMPRTL___kmpc_end_reduce_nowait
5370 : RuntimeFunction::OMPRTL___kmpc_end_reduce);
5372 Builder.CreateBr(ContinuationBlock);
5377 Builder.SetInsertPoint(AtomicRedBlock);
5378 if (CanGenerateAtomic &&
llvm::none_of(IsByRef, [](
bool P) {
return P; })) {
5385 if (!
Builder.GetInsertBlock())
5388 Builder.CreateBr(ContinuationBlock);
5401 if (!
Builder.GetInsertBlock())
5404 Builder.SetInsertPoint(ContinuationBlock);
5415 Directive OMPD = Directive::OMPD_master;
5420 Value *Args[] = {Ident, ThreadId};
5428 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
5440 Directive OMPD = Directive::OMPD_masked;
5446 Value *ArgsEnd[] = {Ident, ThreadId};
5454 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
5464 Call->setDoesNotThrow();
5479 bool IsInclusive,
ScanInfo *ScanRedInfo) {
5481 llvm::Error Err = emitScanBasedDirectiveDeclsIR(AllocaIP, ScanVars,
5482 ScanVarsType, ScanRedInfo);
5493 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5496 Type *DestTy = ScanVarsType[i];
5497 Value *Val =
Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5500 Builder.CreateStore(Src, Val);
5505 Builder.GetInsertBlock()->getParent());
5508 IV = ScanRedInfo->
IV;
5511 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5514 Type *DestTy = ScanVarsType[i];
5516 Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5518 Builder.CreateStore(Src, ScanVars[i]);
5532 Builder.GetInsertBlock()->getParent());
5537Error OpenMPIRBuilder::emitScanBasedDirectiveDeclsIR(
5541 Builder.restoreIP(AllocaIP);
5543 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5545 Builder.CreateAlloca(Builder.getPtrTy(),
nullptr,
"vla");
5552 Builder.restoreIP(CodeGenIP);
5554 Builder.CreateAdd(ScanRedInfo->
Span, Builder.getInt32(1));
5555 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5557 Value *Allocsize = Builder.CreateTypeSize(
5558 IntPtrTy, M.getDataLayout().getTypeAllocSize(ScanVarsType[i]));
5560 Builder.CreateMalloc(
IntPtrTy, Allocsize, AllocSpan,
nullptr,
"arr");
5561 Builder.CreateStore(Buff, (*(ScanRedInfo->
ScanBuffPtrs))[ScanVars[i]]);
5588Error OpenMPIRBuilder::emitScanBasedDirectiveFinalsIR(
5594 Value *PrivateVar = RedInfo.PrivateVariable;
5595 Value *OrigVar = RedInfo.Variable;
5599 Type *SrcTy = RedInfo.ElementType;
5604 Builder.CreateStore(Src, OrigVar);
5652 Builder.GetInsertBlock()->getModule(),
5659 Builder.GetInsertBlock()->getModule(),
5665 llvm::ConstantInt::get(ScanRedInfo->
Span->
getType(), 1));
5666 Builder.SetInsertPoint(InputBB);
5669 Builder.SetInsertPoint(LoopBB);
5685 Builder.CreateCondBr(CmpI, InnerLoopBB, InnerExitBB);
5687 Builder.SetInsertPoint(InnerLoopBB);
5691 Value *ReductionVal = RedInfo.PrivateVariable;
5694 Type *DestTy = RedInfo.ElementType;
5697 Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5700 Builder.CreateInBoundsGEP(DestTy, Buff, OffsetIval,
"arrayOffset");
5705 RedInfo.ReductionGen(
Builder.saveIP(), LHS, RHS, Result);
5708 Builder.CreateStore(Result, LHSPtr);
5711 IVal, llvm::ConstantInt::get(
Builder.getInt32Ty(), 1));
5713 CmpI =
Builder.CreateICmpUGE(NextIVal, Pow2K);
5714 Builder.CreateCondBr(CmpI, InnerLoopBB, InnerExitBB);
5717 Counter, llvm::ConstantInt::get(Counter->
getType(), 1));
5723 Builder.CreateCondBr(Cmp, LoopBB, ExitBB);
5744 Error Err = emitScanBasedDirectiveFinalsIR(ReductionInfos, ScanRedInfo);
5751Error OpenMPIRBuilder::emitScanBasedDirectiveIR(
5763 Error Err = InputLoopGen();
5774 Error Err = ScanLoopGen(Builder);
5781void OpenMPIRBuilder::createScanBBs(ScanInfo *ScanRedInfo) {
5818 Builder.SetInsertPoint(Preheader);
5821 Builder.SetInsertPoint(Header);
5822 PHINode *IndVarPHI =
Builder.CreatePHI(IndVarTy, 2,
"omp_" + Name +
".iv");
5823 IndVarPHI->
addIncoming(ConstantInt::get(IndVarTy, 0), Preheader);
5828 Builder.CreateICmpULT(IndVarPHI, TripCount,
"omp_" + Name +
".cmp");
5829 Builder.CreateCondBr(Cmp, Body, Exit);
5834 Builder.SetInsertPoint(Latch);
5844 bool HasNSW =
Config.hasNoSignedWrap();
5847 unsigned BitWidth = CI->getType()->getIntegerBitWidth();
5849 if (CI->getValue().ugt(SignedMax))
5851 }
else if (IsCollapsed) {
5856 Builder.CreateAdd(IndVarPHI, ConstantInt::get(IndVarTy, 1),
5857 "omp_" + Name +
".next",
true, HasNSW);
5868 CL->Header = Header;
5887 NextBB, NextBB, Name);
5919 Value *Start,
Value *Stop,
Value *Step,
bool IsSigned,
bool InclusiveStop,
5928 ComputeLoc, Start, Stop, Step, IsSigned, InclusiveStop, Name);
5929 ScanRedInfo->
Span = TripCount;
5935 ScanRedInfo->
IV =
IV;
5936 createScanBBs(ScanRedInfo);
5939 assert(Terminator->getNumSuccessors() == 1);
5940 BasicBlock *ContinueBlock = Terminator->getSuccessor(0);
5943 Builder.GetInsertBlock()->getParent());
5946 Builder.GetInsertBlock()->getParent());
5947 Builder.CreateBr(ContinueBlock);
5953 const auto &&InputLoopGen = [&]() ->
Error {
5956 InclusiveStop, ComputeIP, Name,
true, ScanRedInfo);
5960 Builder.restoreIP((*LoopInfo)->getAfterIP());
5966 InclusiveStop, ComputeIP, Name,
true, ScanRedInfo);
5970 Builder.restoreIP((*LoopInfo)->getAfterIP());
5974 Error Err = emitScanBasedDirectiveIR(InputLoopGen, ScanLoopGen, ScanRedInfo);
5982 bool IsSigned,
bool InclusiveStop,
const Twine &Name) {
5992 assert(IndVarTy == Stop->
getType() &&
"Stop type mismatch");
5993 assert(IndVarTy == Step->
getType() &&
"Step type mismatch");
5997 ConstantInt *Zero = ConstantInt::get(IndVarTy, 0);
6013 Incr =
Builder.CreateSelect(IsNeg,
Builder.CreateNeg(Step), Step);
6016 Span =
Builder.CreateSub(UB, LB,
"",
false,
true);
6020 Span =
Builder.CreateSub(Stop, Start,
"",
true);
6025 Value *CountIfLooping;
6026 if (InclusiveStop) {
6027 CountIfLooping =
Builder.CreateAdd(
Builder.CreateUDiv(Span, Incr), One);
6033 CountIfLooping =
Builder.CreateSelect(OneCmp, One, CountIfTwo);
6036 return Builder.CreateSelect(ZeroCmp, Zero, CountIfLooping,
6037 "omp_" + Name +
".tripcount");
6042 Value *Start,
Value *Stop,
Value *Step,
bool IsSigned,
bool InclusiveStop,
6049 ComputeLoc, Start, Stop, Step, IsSigned, InclusiveStop, Name);
6054 Config.hasNoSignedWrap());
6055 Value *IndVar =
Builder.CreateAdd(Span, Start,
"",
false,
6056 Config.hasNoSignedWrap());
6058 ScanRedInfo->
IV = IndVar;
6059 return BodyGenCB(
Builder.saveIP(), IndVar);
6065 Builder.getCurrentDebugLocation());
6076 unsigned Bitwidth = Ty->getIntegerBitWidth();
6079 M, omp::RuntimeFunction::OMPRTL___kmpc_dist_for_static_init_4u);
6082 M, omp::RuntimeFunction::OMPRTL___kmpc_dist_for_static_init_8u);
6092 unsigned Bitwidth = Ty->getIntegerBitWidth();
6095 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_init_4u);
6098 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_init_8u);
6106 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6108 "Require dedicated allocate IP");
6114 uint32_t SrcLocStrSize;
6118 case WorksharingLoopType::ForStaticLoop:
6119 Flag = OMP_IDENT_FLAG_WORK_LOOP;
6121 case WorksharingLoopType::DistributeStaticLoop:
6122 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6124 case WorksharingLoopType::DistributeForStaticLoop:
6125 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE | OMP_IDENT_FLAG_WORK_LOOP;
6132 Type *IVTy =
IV->getType();
6133 FunctionCallee StaticInit =
6134 LoopType == WorksharingLoopType::DistributeForStaticLoop
6137 FunctionCallee StaticFini =
6142 AllocaIP.getNodeParent()->getFirstNonPHIOrDbgOrAlloca());
6145 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6146 Value *PLowerBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.lowerbound");
6147 Value *PUpperBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.upperbound");
6148 Value *PStride =
Builder.CreateAlloca(IVTy,
nullptr,
"p.stride");
6157 Constant *One = ConstantInt::get(IVTy, 1);
6158 Builder.CreateStore(Zero, PLowerBound);
6160 Builder.CreateStore(UpperBound, PUpperBound);
6161 Builder.CreateStore(One, PStride);
6167 (LoopType == WorksharingLoopType::DistributeStaticLoop)
6168 ? OMPScheduleType::OrderedDistribute
6171 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6175 auto BuildInitCall = [LoopType, SrcLoc, ThreadNum, PLastIter, PLowerBound,
6176 PUpperBound, IVTy, PStride, One,
Zero, StaticInit,
6179 PLowerBound, PUpperBound});
6180 if (LoopType == WorksharingLoopType::DistributeForStaticLoop) {
6181 Value *PDistUpperBound =
6182 Builder.CreateAlloca(IVTy,
nullptr,
"p.distupperbound");
6183 Args.push_back(PDistUpperBound);
6188 BuildInitCall(SchedulingType,
Builder);
6189 if (HasDistSchedule &&
6190 LoopType != WorksharingLoopType::DistributeStaticLoop) {
6191 Constant *DistScheduleSchedType = ConstantInt::get(
6196 BuildInitCall(DistScheduleSchedType,
Builder);
6199 Value *InclusiveUpperBound =
Builder.CreateLoad(IVTy, PUpperBound);
6201 Value *TripCount =
Builder.CreateAdd(TripCountMinusOne, One);
6202 CLI->setTripCount(TripCount);
6208 CLI->mapIndVar([&](Instruction *OldIV) ->
Value * {
6213 Config.hasNoSignedWrap());
6225 omp::Directive::OMPD_for,
false,
6228 return BarrierIP.takeError();
6255 Reachable.insert(
Block);
6269OpenMPIRBuilder::applyStaticChunkedWorkshareLoop(
6273 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6274 assert((ChunkSize || DistScheduleChunkSize) &&
"Chunk size is required");
6279 Type *IVTy =
IV->getType();
6281 "Max supported tripcount bitwidth is 64 bits");
6283 :
Type::getInt64Ty(Ctx);
6286 Constant *One = ConstantInt::get(InternalIVTy, 1);
6291 SmallVector<Instruction *> UIs;
6292 for (BasicBlock &BB : *
F)
6293 if (!BB.hasTerminator())
6294 UIs.
push_back(
new UnreachableInst(
F->getContext(), &BB));
6299 LoopInfo &&LI = LIA.
run(*
F,
FAM);
6300 for (Instruction *
I : UIs)
6301 I->eraseFromParent();
6304 if (ChunkSize || DistScheduleChunkSize)
6309 FunctionCallee StaticInit =
6311 FunctionCallee StaticFini =
6317 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6318 Value *PLowerBound =
6319 Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.lowerbound");
6320 Value *PUpperBound =
6321 Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.upperbound");
6322 Value *PStride =
Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.stride");
6331 ChunkSize ? ChunkSize : Zero, InternalIVTy,
"chunksize");
6332 Value *CastedDistScheduleChunkSize =
Builder.CreateZExtOrTrunc(
6333 DistScheduleChunkSize ? DistScheduleChunkSize : Zero, InternalIVTy,
6334 "distschedulechunksize");
6335 Value *CastedTripCount =
6336 Builder.CreateZExt(OrigTripCount, InternalIVTy,
"tripcount");
6339 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6341 ConstantInt::get(I32Type,
static_cast<int>(DistScheduleSchedType));
6342 Builder.CreateStore(Zero, PLowerBound);
6343 Value *OrigUpperBound =
Builder.CreateSub(CastedTripCount, One);
6344 Value *IsTripCountZero =
Builder.CreateICmpEQ(CastedTripCount, Zero);
6346 Builder.CreateSelect(IsTripCountZero, Zero, OrigUpperBound);
6347 Builder.CreateStore(UpperBound, PUpperBound);
6348 Builder.CreateStore(One, PStride);
6352 uint32_t SrcLocStrSize;
6355 if (DistScheduleSchedType != OMPScheduleType::None) {
6356 Flag |= OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6361 auto BuildInitCall = [StaticInit, SrcLoc, ThreadNum, PLastIter, PLowerBound,
6362 PUpperBound, PStride, One,
6363 this](
Value *SchedulingType,
Value *ChunkSize,
6366 StaticInit, {SrcLoc, ThreadNum,
6367 SchedulingType, PLastIter,
6368 PLowerBound, PUpperBound,
6372 BuildInitCall(SchedulingType, CastedChunkSize,
Builder);
6373 if (DistScheduleSchedType != OMPScheduleType::None &&
6374 SchedType != OMPScheduleType::OrderedDistributeChunked &&
6375 SchedType != OMPScheduleType::OrderedDistribute) {
6379 BuildInitCall(DistSchedulingType, CastedDistScheduleChunkSize,
Builder);
6383 Value *FirstChunkStart =
6384 Builder.CreateLoad(InternalIVTy, PLowerBound,
"omp_firstchunk.lb");
6385 Value *FirstChunkStop =
6386 Builder.CreateLoad(InternalIVTy, PUpperBound,
"omp_firstchunk.ub");
6387 Value *FirstChunkEnd =
Builder.CreateAdd(FirstChunkStop, One);
6389 Builder.CreateSub(FirstChunkEnd, FirstChunkStart,
"omp_chunk.range");
6390 Value *NextChunkStride =
6391 Builder.CreateLoad(InternalIVTy, PStride,
"omp_dispatch.stride");
6395 Value *DispatchCounter;
6403 DispatchCounter = Counter;
6406 FirstChunkStart, CastedTripCount, NextChunkStride,
6429 Value *ChunkEnd =
Builder.CreateAdd(DispatchCounter, ChunkRange);
6430 Value *IsLastChunk =
6431 Builder.CreateICmpUGE(ChunkEnd, CastedTripCount,
"omp_chunk.is_last");
6432 Value *CountUntilOrigTripCount =
6433 Builder.CreateSub(CastedTripCount, DispatchCounter);
6435 IsLastChunk, CountUntilOrigTripCount, ChunkRange,
"omp_chunk.tripcount");
6436 Value *BackcastedChunkTC =
6437 Builder.CreateTrunc(ChunkTripCount, IVTy,
"omp_chunk.tripcount.trunc");
6438 CLI->setTripCount(BackcastedChunkTC);
6443 Value *BackcastedDispatchCounter =
6444 Builder.CreateTrunc(DispatchCounter, IVTy,
"omp_dispatch.iv.trunc");
6445 CLI->mapIndVar([&](Instruction *) ->
Value * {
6447 return Builder.CreateAdd(
IV, BackcastedDispatchCounter);
6460 return AfterIP.takeError();
6475static FunctionCallee
6478 unsigned Bitwidth = Ty->getIntegerBitWidth();
6481 case WorksharingLoopType::ForStaticLoop:
6484 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_loop_4u);
6487 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_loop_8u);
6489 case WorksharingLoopType::DistributeStaticLoop:
6492 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_static_loop_4u);
6495 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_static_loop_8u);
6497 case WorksharingLoopType::DistributeForStaticLoop:
6500 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_for_static_loop_4u);
6503 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_for_static_loop_8u);
6506 if (Bitwidth != 32 && Bitwidth != 64) {
6518 Function &LoopBodyFn,
bool NoLoop) {
6529 if (LoopType == WorksharingLoopType::DistributeStaticLoop) {
6530 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6531 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), 0));
6532 Builder.restoreIP(std::prev(InsertBlock->
end()));
6537 M, omp::RuntimeFunction::OMPRTL_omp_get_num_threads);
6538 Builder.restoreIP(std::prev(InsertBlock->
end()));
6542 Builder.CreateZExtOrTrunc(NumThreads, TripCountTy,
"num.threads.cast"));
6543 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6544 if (LoopType == WorksharingLoopType::DistributeForStaticLoop) {
6545 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6546 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), NoLoop));
6548 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), 0));
6572 Builder.restoreIP(Preheader->
end());
6575 Builder.CreateBr(CLI->
getExit());
6583 CleanUpInfo.
collectBlocks(RegionBlockSet, BlocksToBeRemoved);
6591 "Expected unique undroppable user of outlined function");
6593 assert(OutlinedFnCallInstruction &&
"Expected outlined function call");
6595 "Expected outlined function call to be located in loop preheader");
6597 if (OutlinedFnCallInstruction->
arg_size() > 1)
6604 LoopBodyArg, TripCount, OutlinedFn, NoLoop);
6606 for (
auto &ToBeDeletedItem : ToBeDeleted)
6607 ToBeDeletedItem->eraseFromParent();
6614 uint32_t SrcLocStrSize;
6618 case WorksharingLoopType::ForStaticLoop:
6619 Flag = OMP_IDENT_FLAG_WORK_LOOP;
6621 case WorksharingLoopType::DistributeStaticLoop:
6622 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6624 case WorksharingLoopType::DistributeForStaticLoop:
6625 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE | OMP_IDENT_FLAG_WORK_LOOP;
6630 auto OI = std::make_unique<OutlineInfo>();
6635 SmallVector<Instruction *, 4> ToBeDeleted;
6637 OI->OuterAllocBB = AllocaIP.getNodeParent();
6660 SmallPtrSet<BasicBlock *, 32> ParallelRegionBlockSet;
6662 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
6664 CodeExtractorAnalysisCache CEAC(*OuterFn);
6665 CodeExtractor Extractor(Blocks,
6679 SetVector<Value *> SinkingCands, HoistingCands;
6683 Extractor.findAllocas(CEAC, SinkingCands, HoistingCands, CommonExit);
6690 for (
auto Use :
Users) {
6692 if (ParallelRegionBlockSet.
count(Inst->getParent())) {
6693 Inst->replaceUsesOfWith(CLI->
getIndVar(), NewLoopCntLoad);
6699 OI->ExcludeArgsFromAggregate.push_back(NewLoopCntLoad);
6706 OI->PostOutlineCB = [=, ToBeDeletedVec =
6707 std::move(ToBeDeleted)](
Function &OutlinedFn) {
6724 return BarrierIP.takeError();
6731 bool NeedsBarrier, omp::ScheduleKind SchedKind,
Value *ChunkSize,
6732 bool HasSimdModifier,
bool HasMonotonicModifier,
6733 bool HasNonmonotonicModifier,
bool HasOrderedClause,
6735 Value *DistScheduleChunkSize) {
6736 if (
Config.isTargetDevice())
6737 return applyWorkshareLoopTarget(
DL, CLI, AllocaIP, LoopType, NeedsBarrier,
6740 SchedKind, ChunkSize, HasSimdModifier, HasMonotonicModifier,
6741 HasNonmonotonicModifier, HasOrderedClause, DistScheduleChunkSize);
6743 bool IsOrdered = (EffectiveScheduleType & OMPScheduleType::ModifierOrdered) ==
6744 OMPScheduleType::ModifierOrdered;
6746 if (HasDistSchedule) {
6747 DistScheduleSchedType = DistScheduleChunkSize
6748 ? OMPScheduleType::OrderedDistributeChunked
6749 : OMPScheduleType::OrderedDistribute;
6751 switch (EffectiveScheduleType & ~OMPScheduleType::ModifierMask) {
6752 case OMPScheduleType::BaseStatic:
6753 case OMPScheduleType::BaseDistribute:
6754 assert((!ChunkSize || !DistScheduleChunkSize) &&
6755 "No chunk size with static-chunked schedule");
6756 if (IsOrdered && !HasDistSchedule)
6757 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6758 NeedsBarrier, ChunkSize);
6760 if (DistScheduleChunkSize)
6761 return applyStaticChunkedWorkshareLoop(
6762 DL, CLI, AllocaIP, NeedsBarrier, ChunkSize, EffectiveScheduleType,
6763 DistScheduleChunkSize, DistScheduleSchedType);
6764 return applyStaticWorkshareLoop(
DL, CLI, AllocaIP, LoopType, NeedsBarrier,
6767 case OMPScheduleType::BaseStaticChunked:
6768 case OMPScheduleType::BaseDistributeChunked:
6769 if (IsOrdered && !HasDistSchedule)
6770 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6771 NeedsBarrier, ChunkSize);
6773 return applyStaticChunkedWorkshareLoop(
6774 DL, CLI, AllocaIP, NeedsBarrier, ChunkSize, EffectiveScheduleType,
6775 DistScheduleChunkSize, DistScheduleSchedType);
6777 case OMPScheduleType::BaseRuntime:
6778 case OMPScheduleType::BaseAuto:
6779 case OMPScheduleType::BaseGreedy:
6780 case OMPScheduleType::BaseBalanced:
6781 case OMPScheduleType::BaseSteal:
6782 case OMPScheduleType::BaseRuntimeSimd:
6784 "schedule type does not support user-defined chunk sizes");
6786 case OMPScheduleType::BaseGuidedSimd:
6787 case OMPScheduleType::BaseDynamicChunked:
6788 case OMPScheduleType::BaseGuidedChunked:
6789 case OMPScheduleType::BaseGuidedIterativeChunked:
6790 case OMPScheduleType::BaseGuidedAnalyticalChunked:
6791 case OMPScheduleType::BaseStaticBalancedChunked:
6792 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6793 NeedsBarrier, ChunkSize);
6806 unsigned Bitwidth = Ty->getIntegerBitWidth();
6809 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_init_4u);
6812 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_init_8u);
6820static FunctionCallee
6822 unsigned Bitwidth = Ty->getIntegerBitWidth();
6825 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_next_4u);
6828 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_next_8u);
6835static FunctionCallee
6837 unsigned Bitwidth = Ty->getIntegerBitWidth();
6840 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_fini_4u);
6843 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_fini_8u);
6848OpenMPIRBuilder::applyDynamicWorkshareLoop(
DebugLoc DL, CanonicalLoopInfo *CLI,
6851 bool NeedsBarrier,
Value *Chunk) {
6852 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6854 "Require dedicated allocate IP");
6856 "Require valid schedule type");
6858 bool Ordered = (SchedType & OMPScheduleType::ModifierOrdered) ==
6859 OMPScheduleType::ModifierOrdered;
6864 uint32_t SrcLocStrSize;
6871 Type *IVTy =
IV->getType();
6877 AllocaIP.getNodeParent()->getFirstNonPHIOrDbgOrAlloca());
6879 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6880 Value *PLowerBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.lowerbound");
6881 Value *PUpperBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.upperbound");
6882 Value *PStride =
Builder.CreateAlloca(IVTy,
nullptr,
"p.stride");
6891 Constant *One = ConstantInt::get(IVTy, 1);
6892 Builder.CreateStore(One, PLowerBound);
6894 Builder.CreateStore(UpperBound, PUpperBound);
6895 Builder.CreateStore(One, PStride);
6913 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6925 Builder.SetInsertPoint(OuterCond, OuterCond->getFirstInsertionPt());
6928 {SrcLoc, ThreadNum, PLastIter, PLowerBound, PUpperBound, PStride});
6929 Constant *Zero32 = ConstantInt::get(I32Type, 0);
6932 Builder.CreateSub(
Builder.CreateLoad(IVTy, PLowerBound), One,
"lb");
6933 Builder.CreateCondBr(MoreWork, Header, Exit);
6939 PI->setIncomingBlock(0, OuterCond);
6945 Br->setSuccessor(OuterCond);
6951 UpperBound =
Builder.CreateLoad(IVTy, PUpperBound,
"ub");
6954 CI->setOperand(1, UpperBound);
6958 assert(BI->getSuccessor(1) == Exit);
6959 BI->setSuccessor(1, OuterCond);
6973 omp::Directive::OMPD_for,
false,
6976 return BarrierIP.takeError();
7028 assert(
Loops.size() >= 1 &&
"At least one loop required");
7029 size_t NumLoops =
Loops.size();
7033 return Loops.front();
7045 Loop->collectControlBlocks(OldControlBBs);
7049 if (ComputeIP.isValid())
7056 Value *CollapsedTripCount =
nullptr;
7059 "All loops to collapse must be valid canonical loops");
7060 Value *OrigTripCount = L->getTripCount();
7061 if (!CollapsedTripCount) {
7062 CollapsedTripCount = OrigTripCount;
7067 CollapsedTripCount =
7068 Builder.CreateNUWMul(CollapsedTripCount, OrigTripCount);
7074 OrigPreheader->
getNextNode(), OrigAfter,
"collapsed",
7081 Builder.restoreIP(Result->getBodyIP());
7083 Value *Leftover = Result->getIndVar();
7085 NewIndVars.
resize(NumLoops);
7086 for (
int i = NumLoops - 1; i >= 1; --i) {
7087 Value *OrigTripCount =
Loops[i]->getTripCount();
7089 Value *NewIndVar =
Builder.CreateURem(Leftover, OrigTripCount);
7090 NewIndVars[i] = NewIndVar;
7092 Leftover =
Builder.CreateUDiv(Leftover, OrigTripCount);
7095 NewIndVars[0] = Leftover;
7104 BasicBlock *ContinueBlock = Result->getBody();
7106 auto ContinueWith = [&ContinueBlock, &ContinuePred,
DL](
BasicBlock *Dest,
7113 ContinueBlock =
nullptr;
7114 ContinuePred = NextSrc;
7121 for (
size_t i = 0; i < NumLoops - 1; ++i)
7122 ContinueWith(
Loops[i]->getBody(),
Loops[i + 1]->getHeader());
7128 for (
size_t i = NumLoops - 1; i > 0; --i)
7129 ContinueWith(
Loops[i]->getAfter(),
Loops[i - 1]->getLatch());
7132 ContinueWith(Result->getLatch(),
nullptr);
7139 for (
size_t i = 0; i < NumLoops; ++i)
7140 Loops[i]->getIndVar()->replaceAllUsesWith(NewIndVars[i]);
7154std::vector<CanonicalLoopInfo *>
7158 "Must pass as many tile sizes as there are loops");
7159 int NumLoops =
Loops.size();
7160 assert(NumLoops >= 1 &&
"At least one loop to tile required");
7172 Loop->collectControlBlocks(OldControlBBs);
7180 assert(L->isValid() &&
"All input loops must be valid canonical loops");
7181 OrigTripCounts.
push_back(L->getTripCount());
7192 for (
int i = 0; i < NumLoops - 1; ++i) {
7205 for (
int i = 0; i < NumLoops; ++i) {
7207 Value *OrigTripCount = OrigTripCounts[i];
7220 Value *FloorTripOverflow =
7221 Builder.CreateICmpNE(FloorTripRem, ConstantInt::get(IVType, 0));
7223 FloorTripOverflow =
Builder.CreateZExt(FloorTripOverflow, IVType);
7224 Value *FloorTripCount =
7225 Builder.CreateAdd(FloorCompleteTripCount, FloorTripOverflow,
7226 "omp_floor" +
Twine(i) +
".tripcount",
true);
7229 FloorCompleteCount.
push_back(FloorCompleteTripCount);
7235 std::vector<CanonicalLoopInfo *> Result;
7236 Result.reserve(NumLoops * 2);
7249 auto EmbeddNewLoop =
7250 [
this,
DL,
F, InnerEnter, &Enter, &
Continue, &OutroInsertBefore](
7253 DL, TripCount,
F, InnerEnter, OutroInsertBefore, Name);
7258 Enter = EmbeddedLoop->
getBody();
7260 OutroInsertBefore = EmbeddedLoop->
getLatch();
7261 return EmbeddedLoop;
7265 const Twine &NameBase) {
7268 EmbeddNewLoop(
P.value(), NameBase +
Twine(
P.index()));
7269 Result.push_back(EmbeddedLoop);
7273 EmbeddNewLoops(FloorCount,
"floor");
7279 for (
int i = 0; i < NumLoops; ++i) {
7283 Value *FloorIsEpilogue =
7285 Value *TileTripCount =
7292 EmbeddNewLoops(TileCounts,
"tile");
7297 for (std::pair<BasicBlock *, BasicBlock *>
P : InbetweenCode) {
7306 BodyEnter =
nullptr;
7307 BodyEntered = ExitBB;
7319 Builder.restoreIP(Result.back()->getBodyIP());
7320 for (
int i = 0; i < NumLoops; ++i) {
7323 Value *OrigIndVar = OrigIndVars[i];
7374 assert(
Loop->isValid() &&
"Expecting a valid CanonicalLoopInfo");
7378 assert(Latch &&
"A valid CanonicalLoopInfo must have a unique latch");
7386 if (
I.mayReadOrWriteMemory()) {
7390 I.setMetadata(LLVMContext::MD_access_group,
AccessGroup);
7404 Loop->collectControlBlocks(oldControlBBs);
7409 assert(L->isValid() &&
"All input loops must be valid canonical loops");
7410 origTripCounts.
push_back(L->getTripCount());
7419 Builder.SetInsertPoint(TCBlock);
7420 Value *fusedTripCount =
nullptr;
7422 assert(L->isValid() &&
"All loops to fuse must be valid canonical loops");
7423 Value *origTripCount = L->getTripCount();
7424 if (!fusedTripCount) {
7425 fusedTripCount = origTripCount;
7428 Value *condTP =
Builder.CreateICmpSGT(fusedTripCount, origTripCount);
7429 fusedTripCount =
Builder.CreateSelect(condTP, fusedTripCount, origTripCount,
7443 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7444 Loops[i]->getPreheader()->moveBefore(TCBlock);
7445 Loops[i]->getAfter()->moveBefore(TCBlock);
7449 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7461 for (
size_t i = 0; i <
Loops.size(); ++i) {
7463 F->getContext(),
"omp.fused.inner.cond",
F,
Loops[i]->getBody());
7464 Builder.SetInsertPoint(condBlock);
7472 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7473 Builder.SetInsertPoint(condBBs[i]);
7474 Builder.CreateCondBr(condValues[i],
Loops[i]->getBody(), condBBs[i + 1]);
7490 "omp.fused.pre_latch");
7523 const Twine &NamePrefix) {
7552 C, NamePrefix +
".if.then",
Cond->getParent(),
Cond->getNextNode());
7554 C, NamePrefix +
".if.else",
Cond->getParent(), CanonicalLoop->
getExit());
7557 Builder.SetInsertPoint(SplitBeforeIt);
7559 Builder.CreateCondBr(IfCond, ThenBlock, ElseBlock);
7562 spliceBB(IP, ThenBlock,
false, Builder.getCurrentDebugLocation());
7565 Builder.SetInsertPoint(ElseBlock);
7571 ExistingBlocks.
reserve(L->getNumBlocks() + 1);
7573 ExistingBlocks.
append(L->block_begin(), L->block_end());
7579 assert(LoopCond && LoopHeader &&
"Invalid loop structure");
7581 if (
Block == L->getLoopPreheader() ||
Block == L->getLoopLatch() ||
7588 if (
Block == ThenBlock)
7589 NewBB->
setName(NamePrefix +
".if.else");
7592 VMap[
Block] = NewBB;
7600 L->getLoopLatch()->splitBasicBlockBefore(
L->getLoopLatch()->begin(),
7601 NamePrefix +
".pre_latch");
7605 L->addBasicBlockToLoop(ThenBlock, LI);
7611 if (TargetTriple.
isX86()) {
7612 if (Features.
lookup(
"avx512f"))
7614 else if (Features.
lookup(
"avx"))
7618 if (TargetTriple.
isPPC())
7620 if (TargetTriple.
isWasm())
7629 Value *IfCond, OrderKind Order,
7639 if (!BB.hasTerminator())
7655 I->eraseFromParent();
7658 if (AlignedVars.
size()) {
7660 for (
auto &AlignedItem : AlignedVars) {
7661 Value *AlignedPtr = AlignedItem.first;
7665 Builder.CreateAlignmentAssumption(
F->getDataLayout(), AlignedPtr,
7673 createIfVersion(CanonicalLoop, IfCond, VMap, LIA, LI, L,
"simd");
7686 Reachable.insert(
Block);
7696 if ((Safelen ==
nullptr) || (Order == OrderKind::OMP_ORDER_concurrent))
7712 if (Simdlen || Safelen) {
7716 ConstantInt *VectorizeWidth = Simdlen ==
nullptr ? Safelen : Simdlen;
7742static std::unique_ptr<TargetMachine>
7746 StringRef CPU =
F->getFnAttribute(
"target-cpu").getValueAsString();
7747 StringRef Features =
F->getFnAttribute(
"target-features").getValueAsString();
7758 std::nullopt, OptLevel));
7776 if (!BB.hasTerminator())
7789 [&](
const Function &
F) {
return TM->getTargetTransformInfo(
F); });
7790 FAM.registerPass([&]() {
return TIRA; });
7804 I->eraseFromParent();
7807 assert(L &&
"Expecting CanonicalLoopInfo to be recognized as a loop");
7812 nullptr, ORE,
static_cast<int>(OptLevel),
7832 <<
" Threshold=" << UP.
Threshold <<
"\n"
7835 <<
" PartialOptSizeThreshold="
7855 Ptr =
Load->getPointerOperand();
7857 Ptr =
Store->getPointerOperand();
7864 if (Alloca->getParent() == &
F->getEntryBlock())
7884 int MaxTripCount = 0;
7885 bool MaxOrZero =
false;
7886 unsigned TripMultiple = 0;
7890 MaxTripCount, MaxOrZero, TripMultiple, UCE, UP, PP);
7891 LLVM_DEBUG(
dbgs() <<
"Suggesting unroll factor of " << Factor <<
"\n");
7902 assert(Factor >= 0 &&
"Unroll factor must not be negative");
7918 Ctx, {
MDString::get(Ctx,
"llvm.loop.unroll.count"), FactorConst}));
7931 *UnrolledCLI =
Loop;
7936 "unrolling only makes sense with a factor of 2 or larger");
7938 Type *IndVarTy =
Loop->getIndVarType();
7945 std::vector<CanonicalLoopInfo *>
LoopNest =
7960 Ctx, {
MDString::get(Ctx,
"llvm.loop.unroll.count"), FactorConst})});
7963 (*UnrolledCLI)->assertOK();
7981 Value *Args[] = {Ident, ThreadId, BufSize, CpyBuf, CpyFn, DidItLD};
8000 if (!CPVars.
empty()) {
8005 Directive OMPD = Directive::OMPD_single;
8010 Value *Args[] = {Ident, ThreadId};
8019 if (
Error Err = FiniCB(IP))
8040 EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCBWrapper,
8047 for (
size_t I = 0, E = CPVars.
size();
I < E; ++
I)
8050 ConstantInt::get(Int64, 0), CPVars[
I],
8053 }
else if (!IsNowait) {
8056 omp::Directive::OMPD_unknown,
false,
8074 Directive::OMPD_scope,
nullptr,
nullptr,
8075 BodyGenCB, FiniCB,
false,
true,
8083 omp::Directive::OMPD_unknown,
8099 Directive OMPD = Directive::OMPD_critical;
8104 Value *LockVar = getOMPCriticalRegionLock(CriticalName);
8105 Value *Args[] = {Ident, ThreadId, LockVar};
8122 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
8130 const Twine &Name,
bool IsDependSource) {
8133 [](
Value *SV) {
return SV->getType()->isIntegerTy(64); }) &&
8134 "OpenMP runtime requires depend vec with i64 type");
8147 for (
unsigned I = 0;
I < NumLoops; ++
I) {
8161 Value *Args[] = {Ident, ThreadId, DependBaseAddrGEP};
8179 Directive OMPD = Directive::OMPD_ordered_blockassoc;
8188 Value *Args[] = {Ident, ThreadId};
8198 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
8205 bool HasFinalize,
bool IsCancellable) {
8212 BasicBlock *EntryBB = Builder.GetInsertBlock();
8221 emitCommonDirectiveEntry(OMPD, EntryCall, ExitBB, Conditional);
8233 "Unexpected control flow graph state!!");
8235 emitCommonDirectiveExit(OMPD, FinIP, ExitCall, HasFinalize);
8237 return AfterIP.takeError();
8242 "Unexpected Insertion point location!");
8245 auto InsertBB = merged ? ExitPredBB : ExitBB;
8248 Builder.SetInsertPoint(InsertBB);
8250 return Builder.saveIP();
8254 Directive OMPD,
Value *EntryCall, BasicBlock *ExitBB,
bool Conditional) {
8256 if (!Conditional || !EntryCall)
8262 auto *UI =
new UnreachableInst(
Builder.getContext(), ThenBB);
8272 Builder.CreateCondBr(CallBool, ThenBB, ExitBB);
8276 UI->eraseFromParent();
8284 omp::Directive OMPD,
InsertPointTy FinIP, Instruction *ExitCall,
8292 "Unexpected finalization stack state!");
8295 assert(Fi.DK == OMPD &&
"Unexpected Directive for Finalization call!");
8299 return std::move(Err);
8346 "copyin.not.master.end");
8353 Builder.SetInsertPoint(OMP_Entry);
8356 Value *cmp =
Builder.CreateICmpNE(MasterPtr, PrivatePtr);
8357 Builder.CreateCondBr(cmp, CopyBegin, CopyEnd);
8359 Builder.SetInsertPoint(CopyBegin);
8377 Value *Args[] = {ThreadId,
Size, Allocator};
8400 return Builder.CreateCall(Fn, Args, Name);
8414 Value *Args[] = {ThreadId, Addr, Allocator};
8421 const Twine &Name) {
8429 M.getContext(),
M.getDataLayout().getPrefTypeAlign(Int64)));
8435 const Twine &Name) {
8437 Loc,
Builder.getInt64(
M.getDataLayout().getTypeAllocSize(VarType)), Name);
8442 const Twine &Name) {
8448 return Builder.CreateCall(Fn, Args, Name);
8453 const Twine &Name) {
8455 Loc, Addr,
Builder.getInt64(
M.getDataLayout().getTypeAllocSize(VarType)),
8462 Value *DependenceAddress,
bool HaveNowaitClause) {
8472 else if (
Device->getType() != Int32)
8475 if (NumDependences ==
nullptr) {
8476 NumDependences = ConstantInt::get(Int32, 0);
8480 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8482 Ident, ThreadId, InteropVar, InteropTypeVal,
8483 Device, NumDependences, DependenceAddress, HaveNowaitClauseVal};
8492 Value *NumDependences,
Value *DependenceAddress,
bool HaveNowaitClause) {
8502 else if (
Device->getType() != Int32)
8504 if (NumDependences ==
nullptr) {
8505 NumDependences = ConstantInt::get(Int32, 0);
8509 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8511 Ident, ThreadId, InteropVar,
Device,
8512 NumDependences, DependenceAddress, HaveNowaitClauseVal};
8521 Value *NumDependences,
8522 Value *DependenceAddress,
8523 bool HaveNowaitClause) {
8532 else if (
Device->getType() != Int32)
8534 if (NumDependences ==
nullptr) {
8535 NumDependences = ConstantInt::get(Int32, 0);
8539 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8541 Ident, ThreadId, InteropVar,
Device,
8542 NumDependences, DependenceAddress, HaveNowaitClauseVal};
8572 assert(!Attrs.MaxThreads.empty() && !Attrs.MaxTeams.empty() &&
8573 "expected num_threads and num_teams to be specified");
8593 const std::string DebugPrefix =
"_debug__";
8594 if (KernelName.
ends_with(DebugPrefix)) {
8595 KernelName = KernelName.
drop_back(DebugPrefix.length());
8596 Kernel =
M.getFunction(KernelName);
8602 if (Attrs.MinTeams.front() > 1 || Attrs.MaxTeams.front() > 0)
8604 Attrs.MaxTeams.front());
8607 int32_t MaxThreadsVal = Attrs.MaxThreads.front();
8617 Attrs.MinThreads.front());
8622 if (MaxThreadsVal > 0 &&
8625 MaxThreadsVal = int32_t(
8626 std::min<int64_t>(int64_t(MaxThreadsVal) + 64,
8629 if (MaxThreadsVal > 0)
8644 Twine DynamicEnvironmentName = KernelName +
"_dynamic_environment";
8645 Constant *DynamicEnvironmentInitializer =
8649 DynamicEnvironmentInitializer, DynamicEnvironmentName,
8651 DL.getDefaultGlobalsAddressSpace());
8655 DynamicEnvironmentGV->
getType() == DynamicEnvironmentPtr
8656 ? DynamicEnvironmentGV
8658 DynamicEnvironmentPtr);
8661 ConfigurationEnvironment, {
8662 UseGenericStateMachineVal,
8663 MayUseNestedParallelismVal,
8672 KernelEnvironment, {
8673 ConfigurationEnvironmentInitializer,
8677 std::string KernelEnvironmentName =
8678 (KernelName +
"_kernel_environment").str();
8681 KernelEnvironmentInitializer, KernelEnvironmentName,
8683 DL.getDefaultGlobalsAddressSpace());
8686 return KernelEnvironmentGV->
getType() == KernelEnvironmentPtr
8687 ? KernelEnvironmentGV
8689 KernelEnvironmentPtr);
8696 if (!KernelEnvironment)
8704 omp::RuntimeFunction::OMPRTL___kmpc_target_init);
8706 Value *KernelLaunchEnvironment =
8709 KernelLaunchEnvironment =
8710 KernelLaunchEnvironment->
getType() == KernelLaunchEnvParamTy
8711 ? KernelLaunchEnvironment
8712 :
Builder.CreateAddrSpaceCast(KernelLaunchEnvironment,
8713 KernelLaunchEnvParamTy);
8715 Fn, {KernelEnvironment, KernelLaunchEnvironment});
8727 auto *UI =
Builder.CreateUnreachable();
8733 Builder.SetInsertPoint(WorkerExitBB);
8737 Builder.SetInsertPoint(CheckBBTI);
8738 Builder.CreateCondBr(ExecUserCode, UI->getParent(), WorkerExitBB);
8740 CheckBBTI->eraseFromParent();
8741 UI->eraseFromParent();
8749 int32_t TeamsReductionDataSize) {
8754 omp::RuntimeFunction::OMPRTL___kmpc_target_deinit);
8758 if (!TeamsReductionDataSize)
8764 const std::string DebugPrefix =
"_debug__";
8766 KernelName = KernelName.
drop_back(DebugPrefix.length());
8767 auto *KernelEnvironmentGV =
8768 M.getNamedGlobal((KernelName +
"_kernel_environment").str());
8769 assert(KernelEnvironmentGV &&
"Expected kernel environment global\n");
8770 auto *KernelEnvironmentInitializer = KernelEnvironmentGV->getInitializer();
8772 KernelEnvironmentInitializer,
8773 ConstantInt::get(Int32, TeamsReductionDataSize), {0, 7});
8774 KernelEnvironmentGV->setInitializer(NewInitializer);
8779 if (
Kernel.hasFnAttribute(Name)) {
8780 int32_t OldLimit =
Kernel.getFnAttributeAsParsedInteger(Name);
8786std::pair<int32_t, int32_t>
8788 int32_t ThreadLimit =
8789 Kernel.getFnAttributeAsParsedInteger(
"omp_target_thread_limit");
8792 const auto &Attr =
Kernel.getFnAttribute(
"amdgpu-flat-work-group-size");
8793 if (!Attr.isValid() || !Attr.isStringAttribute())
8794 return {0, ThreadLimit};
8795 auto [LBStr, UBStr] = Attr.getValueAsString().split(
',');
8798 return {0, ThreadLimit};
8799 UB = ThreadLimit ? std::min(ThreadLimit, UB) : UB;
8807 return {0, ThreadLimit ? std::min(ThreadLimit, UB) : UB};
8809 return {0, ThreadLimit};
8815 Kernel.addFnAttr(
"omp_target_thread_limit", std::to_string(UB));
8818 Kernel.addFnAttr(
"amdgpu-flat-work-group-size",
8826std::pair<int32_t, int32_t>
8829 return {0,
Kernel.getFnAttributeAsParsedInteger(
"omp_target_num_teams")};
8833 int32_t LB, int32_t UB) {
8841 Kernel.addFnAttr(
"omp_target_num_teams", std::to_string(LB));
8844void OpenMPIRBuilder::setOutlinedTargetRegionFunctionAttributes(
8853 else if (
T.isNVPTX())
8855 else if (
T.isSPIRV())
8861 StringRef EntryFnIDName) {
8862 if (
Config.isTargetDevice()) {
8863 assert(OutlinedFn &&
"The outlined function must exist if embedded");
8867 return new GlobalVariable(
8872Constant *OpenMPIRBuilder::createTargetRegionEntryAddr(
Function *OutlinedFn,
8873 StringRef EntryFnName) {
8877 assert(!
M.getGlobalVariable(EntryFnName,
true) &&
8878 "Named kernel already exists?");
8879 return new GlobalVariable(
8892 if (
Config.isTargetDevice() || !
Config.openMPOffloadMandatory()) {
8896 OutlinedFn = *CBResult;
8898 OutlinedFn =
nullptr;
8904 if (!IsOffloadEntry)
8907 std::string EntryFnIDName =
8909 ? std::string(EntryFnName)
8913 EntryFnName, EntryFnIDName);
8921 setOutlinedTargetRegionFunctionAttributes(OutlinedFn);
8922 auto OutlinedFnID = createOutlinedFunctionID(OutlinedFn, EntryFnIDName);
8923 auto EntryAddr = createTargetRegionEntryAddr(OutlinedFn, EntryFnName);
8925 EntryInfo, EntryAddr, OutlinedFnID,
8927 return OutlinedFnID;
8945 bool IsStandAlone = !BodyGenCB;
8952 MapInfo = &GenMapInfoCB(
Builder.saveIP());
8954 AllocaIP,
Builder.saveIP(), *MapInfo, Info, CustomMapperCB,
8955 true, DeviceAddrCB))
8962 Value *PointerNum =
Builder.getInt32(Info.NumberOfPtrs);
8972 SrcLocInfo, DeviceID,
8979 assert(MapperFunc &&
"MapperFunc missing for standalone target data");
8983 if (Info.HasNoWait) {
8993 if (Info.HasNoWait) {
8997 emitBlock(OffloadContBlock, CurFn,
true);
9003 bool RequiresOuterTargetTask = Info.HasNoWait;
9004 if (!RequiresOuterTargetTask)
9005 cantFail(TaskBodyCB(
nullptr,
nullptr,
9009 {}, RTArgs, Info.HasNoWait));
9012 omp::OMPRTL___tgt_target_data_begin_mapper);
9016 for (
auto DeviceMap : Info.DevicePtrInfoMap) {
9020 Builder.CreateStore(LI, DeviceMap.second.second);
9057 Value *PointerNum =
Builder.getInt32(Info.NumberOfPtrs);
9066 Value *OffloadingArgs[] = {SrcLocInfo, DeviceID,
9089 return emitIfClause(IfCond, BeginThenGen, BeginElseGen, AllocaIP);
9090 return BeginThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9105 return emitIfClause(IfCond, EndThenGen, EndElseGen, AllocaIP);
9106 return EndThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9109 return emitIfClause(IfCond, BeginThenGen, EndElseGen, AllocaIP);
9110 return BeginThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9121 bool IsGPUDistribute) {
9122 assert((IVSize == 32 || IVSize == 64) &&
9123 "IV size is not compatible with the omp runtime");
9125 if (IsGPUDistribute)
9127 ? (IVSigned ? omp::OMPRTL___kmpc_distribute_static_init_4
9128 : omp::OMPRTL___kmpc_distribute_static_init_4u)
9129 : (IVSigned ? omp::OMPRTL___kmpc_distribute_static_init_8
9130 : omp::OMPRTL___kmpc_distribute_static_init_8u);
9132 Name = IVSize == 32 ? (IVSigned ? omp::OMPRTL___kmpc_for_static_init_4
9133 : omp::OMPRTL___kmpc_for_static_init_4u)
9134 : (IVSigned ? omp::OMPRTL___kmpc_for_static_init_8
9135 : omp::OMPRTL___kmpc_for_static_init_8u);
9142 assert((IVSize == 32 || IVSize == 64) &&
9143 "IV size is not compatible with the omp runtime");
9145 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_init_4
9146 : omp::OMPRTL___kmpc_dispatch_init_4u)
9147 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_init_8
9148 : omp::OMPRTL___kmpc_dispatch_init_8u);
9155 assert((IVSize == 32 || IVSize == 64) &&
9156 "IV size is not compatible with the omp runtime");
9158 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_next_4
9159 : omp::OMPRTL___kmpc_dispatch_next_4u)
9160 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_next_8
9161 : omp::OMPRTL___kmpc_dispatch_next_8u);
9168 assert((IVSize == 32 || IVSize == 64) &&
9169 "IV size is not compatible with the omp runtime");
9171 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_fini_4
9172 : omp::OMPRTL___kmpc_dispatch_fini_4u)
9173 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_fini_8
9174 : omp::OMPRTL___kmpc_dispatch_fini_8u);
9185 DenseMap<
Value *, std::tuple<Value *, unsigned>> &ValueReplacementMap) {
9193 auto GetUpdatedDIVariable = [&](
DILocalVariable *OldVar,
unsigned arg) {
9197 if (NewVar && (arg == NewVar->
getArg()))
9207 auto UpdateDebugRecord = [&](
auto *DR) {
9210 for (
auto Loc : DR->location_ops()) {
9211 auto Iter = ValueReplacementMap.find(
Loc);
9212 if (Iter != ValueReplacementMap.end()) {
9213 DR->replaceVariableLocationOp(
Loc, std::get<0>(Iter->second));
9214 ArgNo = std::get<1>(Iter->second) + 1;
9218 DR->setVariable(GetUpdatedDIVariable(OldVar, ArgNo));
9223 if (DVR->getNumVariableLocationOps() != 1u) {
9224 DVR->setKillLocation();
9227 Value *
Loc = DVR->getVariableLocationOp(0u);
9234 RequiredBB = &DVR->getFunction()->getEntryBlock();
9236 if (RequiredBB && RequiredBB != CurBB) {
9248 "Unexpected debug intrinsic");
9250 UpdateDebugRecord(&DVR);
9251 MoveDebugRecordToCorrectBlock(&DVR);
9254 for (
auto *DVR : DVRsToDelete)
9255 DVR->getMarker()->MarkedInstr->dropOneDbgRecord(DVR);
9259 Module *M = Func->getParent();
9262 DB.createQualifiedType(dwarf::DW_TAG_pointer_type,
nullptr);
9263 unsigned ArgNo = Func->arg_size();
9265 NewSP,
"dyn_ptr", ArgNo, NewSP->
getFile(), 0, VoidPtrTy,
9266 false, DINode::DIFlags::FlagArtificial);
9268 Argument *LastArg = Func->getArg(Func->arg_size() - 1);
9269 DB.insertDeclare(LastArg, Var, DB.createExpression(),
Loc,
9291 for (
auto &Arg : Inputs)
9292 ParameterTypes.
push_back(Arg->getType()->isPointerTy()
9296 for (
auto &Arg : Inputs)
9297 ParameterTypes.
push_back(Arg->getType());
9305 auto BB = Builder.GetInsertBlock();
9306 auto M = BB->getModule();
9317 if (TargetCpuAttr.isStringAttribute())
9318 Func->addFnAttr(TargetCpuAttr);
9320 auto TargetFeaturesAttr = ParentFn->
getFnAttribute(
"target-features");
9321 if (TargetFeaturesAttr.isStringAttribute())
9322 Func->addFnAttr(TargetFeaturesAttr);
9327 OMPBuilder.
emitUsed(
"llvm.compiler.used", {ExecMode});
9337 Builder.SetCurrentDebugLocation(OutlinedFnLoc);
9341 Builder.SetInsertPoint(EntryBB);
9352 BasicBlock *UserCodeEntryBB = Builder.GetInsertBlock();
9362 splitBB(Builder,
true,
"outlined.body");
9364 CBFunc(Builder.saveIP(), OutlinedBodyBB->
begin(), ExitBB);
9367 Builder.SetInsertPoint(ExitBB);
9375 Builder.SetCurrentDebugLocation(OutlinedFnLoc);
9382 Builder.CreateRetVoid();
9386 auto AllocaIP = Builder.saveIP();
9391 const auto &ArgRange =
make_range(Func->arg_begin(), Func->arg_end() - 1);
9423 if (Instr->getFunction() == Func)
9424 Instr->replaceUsesOfWith(
Input, InputCopy);
9430 for (
auto InArg :
zip(Inputs, ArgRange)) {
9432 Argument &Arg = std::get<1>(InArg);
9433 Value *InputCopy =
nullptr;
9436 Arg,
Input, InputCopy, AllocaIP, Builder.saveIP(), ExitBB->
begin());
9439 Builder.restoreIP(*AfterIP);
9440 ValueReplacementMap[
Input] = std::make_tuple(InputCopy, Arg.
getArgNo());
9460 DeferredReplacement.push_back(std::make_pair(
Input, InputCopy));
9467 ReplaceValue(
Input, InputCopy, Func);
9471 for (
auto Deferred : DeferredReplacement)
9472 ReplaceValue(std::get<0>(Deferred), std::get<1>(Deferred), Func);
9475 ValueReplacementMap);
9483 Value *TaskWithPrivates,
9484 Type *TaskWithPrivatesTy) {
9486 Type *TaskTy = OMPIRBuilder.Task;
9489 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskWithPrivates, 0);
9490 Value *Shareds = TaskT;
9500 if (TaskWithPrivatesTy != TaskTy)
9501 Shareds = Builder.CreateStructGEP(TaskTy, TaskT, 0);
9518 const size_t NumOffloadingArrays,
const int SharedArgsOperandNo) {
9523 assert((!NumOffloadingArrays || PrivatesTy) &&
9524 "PrivatesTy cannot be nullptr when there are offloadingArrays"
9554 Type *TaskPtrTy = OMPBuilder.TaskPtr;
9555 [[maybe_unused]]
Type *TaskTy = OMPBuilder.Task;
9561 ".omp_target_task_proxy_func", M);
9562 Value *ThreadId = ProxyFn->getArg(0);
9563 Value *TaskWithPrivates = ProxyFn->getArg(1);
9564 ThreadId->
setName(
"thread.id");
9565 TaskWithPrivates->
setName(
"task");
9567 bool HasShareds = SharedArgsOperandNo > 0;
9568 bool HasOffloadingArrays = NumOffloadingArrays > 0;
9572 Builder.SetInsertPoint(EntryBB);
9579 if (HasOffloadingArrays) {
9580 assert(TaskTy != TaskWithPrivatesTy &&
9581 "If there are offloading arrays to pass to the target"
9582 "TaskTy cannot be the same as TaskWithPrivatesTy");
9585 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskWithPrivates, 1);
9586 for (
unsigned int i = 0; i < NumOffloadingArrays; ++i)
9588 Builder.CreateStructGEP(PrivatesTy, Privates, i));
9592 auto *ArgStructAlloca =
9594 assert(ArgStructAlloca &&
9595 "Unable to find the alloca instruction corresponding to arguments "
9596 "for extracted function");
9598 std::optional<TypeSize> ArgAllocSize =
9600 assert(ArgStructType && ArgAllocSize &&
9601 "Unable to determine size of arguments for extracted function");
9602 uint64_t StructSize = ArgAllocSize->getFixedValue();
9605 Builder.CreateAlloca(ArgStructType,
nullptr,
"structArg");
9607 Value *SharedsSize = Builder.getInt64(StructSize);
9610 OMPBuilder, Builder, TaskWithPrivates, TaskWithPrivatesTy);
9612 Builder.CreateMemCpy(
9613 NewArgStructAlloca, NewArgStructAlloca->
getAlign(), LoadShared,
9615 KernelLaunchArgs.
push_back(NewArgStructAlloca);
9618 Builder.CreateRetVoid();
9624 return GEP->getSourceElementType();
9626 return Alloca->getAllocatedType();
9649 if (OffloadingArraysToPrivatize.
empty())
9650 return OMPIRBuilder.Task;
9653 for (
Value *V : OffloadingArraysToPrivatize) {
9654 assert(V->getType()->isPointerTy() &&
9655 "Expected pointer to array to privatize. Got a non-pointer value "
9658 assert(ArrayTy &&
"ArrayType cannot be nullptr");
9664 "struct.task_with_privates");
9679 EntryFnName, Inputs, CBFunc,
9680 ArgAccessorFuncCB, OutlinedFnLoc);
9684 EntryInfo, GenerateOutlinedFunction, IsOffloadEntry, OutlinedFn,
9823 auto OI = std::make_unique<OutlineInfo>();
9824 OI->EntryBB = TargetTaskAllocaBB;
9825 OI->OuterAllocBB = AllocaIP.getNodeParent();
9830 Builder, AllocaIP, ToBeDeleted, TargetTaskAllocaIP,
"global.tid",
false));
9833 Builder.restoreIP(TargetTaskBodyIP);
9834 if (
Error Err = TaskBodyCB(DeviceID, RTLoc, TargetTaskAllocaIP))
9852 bool NeedsTargetTask = HasNoWait && DeviceID;
9853 if (NeedsTargetTask) {
9859 OffloadingArraysToPrivatize.
push_back(V);
9860 OI->ExcludeArgsFromAggregate.push_back(V);
9864 OI->PostOutlineCB = [
this, ToBeDeleted, Dependencies, NeedsTargetTask,
9865 DeviceID, OffloadingArraysToPrivatize](
9868 "there must be a single user for the outlined function");
9882 const unsigned int NumStaleCIArgs = StaleCI->
arg_size();
9883 bool HasShareds = NumStaleCIArgs > OffloadingArraysToPrivatize.
size() + 1;
9885 NumStaleCIArgs == (OffloadingArraysToPrivatize.
size() + 2)) &&
9886 "Wrong number of arguments for StaleCI when shareds are present");
9887 int SharedArgOperandNo =
9888 HasShareds ? OffloadingArraysToPrivatize.
size() + 1 : 0;
9894 if (!OffloadingArraysToPrivatize.
empty())
9899 *
this,
Builder, StaleCI, PrivatesTy, TaskWithPrivatesTy,
9900 OffloadingArraysToPrivatize.
size(), SharedArgOperandNo);
9902 LLVM_DEBUG(
dbgs() <<
"Proxy task entry function created: " << *ProxyFn
9905 Builder.SetInsertPoint(StaleCI);
9922 OMPRTL___kmpc_omp_target_task_alloc);
9934 M.getDataLayout().getTypeStoreSize(TaskWithPrivatesTy));
9941 auto *ArgStructAlloca =
9943 assert(ArgStructAlloca &&
9944 "Unable to find the alloca instruction corresponding to arguments "
9945 "for extracted function");
9946 std::optional<TypeSize> ArgAllocSize =
9949 "Unable to determine size of arguments for extracted function");
9950 SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
9969 TaskSize, SharedsSize,
9972 if (NeedsTargetTask) {
9973 assert(DeviceID &&
"Expected non-empty device ID.");
9983 *
this,
Builder, TaskData, TaskWithPrivatesTy);
9987 if (!OffloadingArraysToPrivatize.
empty()) {
9989 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskData, 1);
9990 for (
unsigned int i = 0; i < OffloadingArraysToPrivatize.
size(); ++i) {
9991 Value *PtrToPrivatize = OffloadingArraysToPrivatize[i];
9998 "ElementType should match ArrayType");
10001 Value *Dst =
Builder.CreateStructGEP(PrivatesTy, Privates, i);
10004 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(ElementType)));
10008 Value *DepArray =
nullptr;
10009 Value *NumDeps =
nullptr;
10012 NumDeps = Dependencies.
NumDeps;
10013 }
else if (!Dependencies.
Deps.empty()) {
10015 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
10026 if (!NeedsTargetTask) {
10035 ConstantInt::get(
Builder.getInt32Ty(), 0),
10048 }
else if (DepArray) {
10056 {Ident, ThreadID, TaskData, NumDeps, DepArray,
10057 ConstantInt::get(
Builder.getInt32Ty(), 0),
10065 Builder.ClearInsertionPoint();
10068 I->eraseFromParent();
10073 << *(
Builder.GetInsertBlock()) <<
"\n");
10075 << *(
Builder.GetInsertBlock()->getParent()->getParent())
10087 CustomMapperCB, IsNonContiguous, DeviceAddrCB))
10110 Builder.restoreIP(IP);
10116 return Builder.saveIP();
10119 bool HasDependencies = !Dependencies.
empty();
10120 bool RequiresOuterTargetTask = HasNoWait || HasDependencies;
10137 if (OutlinedFnID && DeviceID)
10139 EmitTargetCallFallbackCB, KArgs,
10140 DeviceID, RTLoc, TargetTaskAllocaIP);
10148 return EmitTargetCallFallbackCB(OMPBuilder.
Builder.
saveIP());
10155 auto &&EmitTargetCallElse =
10162 if (RequiresOuterTargetTask) {
10169 Dependencies, EmptyRTArgs, HasNoWait);
10171 return EmitTargetCallFallbackCB(Builder.saveIP());
10174 Builder.restoreIP(AfterIP);
10178 auto &&EmitTargetCallThen =
10182 Info.HasNoWait = HasNoWait;
10187 AllocaIP, Builder.saveIP(), Info, RTArgs, MapInfo, CustomMapperCB,
10193 for (
auto [DefaultVal, RuntimeVal] :
10195 NumTeamsC.
push_back(RuntimeVal ? RuntimeVal
10196 : Builder.getInt32(DefaultVal));
10200 auto InitMaxThreadsClause = [&Builder](
Value *
Clause) {
10202 Clause = Builder.CreateIntCast(
Clause, Builder.getInt32Ty(),
10206 auto CombineMaxThreadsClauses = [&Builder](
Value *
Clause,
Value *&Result) {
10209 Result ? Builder.CreateSelect(Builder.CreateICmpULT(Result,
Clause),
10217 Value *MaxThreadsClause =
10219 ? InitMaxThreadsClause(RuntimeAttrs.
MaxThreads.front())
10222 for (
auto [TeamsVal, TargetVal] :
zip_equal(
10224 Value *TeamsThreadLimitClause = InitMaxThreadsClause(TeamsVal);
10225 Value *NumThreads = InitMaxThreadsClause(TargetVal);
10227 CombineMaxThreadsClauses(TeamsThreadLimitClause, NumThreads);
10228 CombineMaxThreadsClauses(MaxThreadsClause, NumThreads);
10230 NumThreadsC.
push_back(NumThreads ? NumThreads : Builder.getInt32(0));
10233 unsigned NumTargetItems = Info.NumberOfPtrs;
10234 Value *RTLoc = RTLocOverride;
10245 Builder.getInt64Ty(),
10247 : Builder.getInt64(0);
10251 DynCGroupMem = Builder.getInt32(0);
10254 NumTargetItems, RTArgs, TripCount, NumTeamsC, NumThreadsC, DynCGroupMem,
10255 HasNoWait,
false,
false,
10256 DynCGroupMemFallback);
10263 if (RequiresOuterTargetTask)
10265 RTLoc, AllocaIP, Dependencies,
10266 KArgs.
RTArgs, Info.HasNoWait);
10269 Builder, OutlinedFnID, EmitTargetCallFallbackCB, KArgs,
10270 RuntimeAttrs.
DeviceID, RTLoc, AllocaIP);
10273 Builder.restoreIP(AfterIP);
10280 if (!OutlinedFnID) {
10281 cantFail(EmitTargetCallElse(AllocaIP, Builder.saveIP(), DeallocBlocks));
10287 cantFail(EmitTargetCallThen(AllocaIP, Builder.saveIP(), DeallocBlocks));
10292 EmitTargetCallElse, AllocaIP));
10305 bool HasNowait,
Value *DynCGroupMem,
10307 Value *RTLocOverride) {
10312 Builder.restoreIP(CodeGenIP);
10320 *
this,
Builder, IsOffloadEntry, EntryInfo, DefaultAttrs, OutlinedFn,
10321 OutlinedFnID, Inputs, CBFunc, ArgAccessorFuncCB, OutlinedFnLoc))
10327 if (!
Config.isTargetDevice())
10329 DefaultAttrs, RuntimeAttrs, IfCond, OutlinedFn, OutlinedFnID,
10330 Inputs, GenMapInfoCB, CustomMapperCB, Dependencies,
10331 HasNowait, DynCGroupMem, DynCGroupMemFallback);
10345 return OS.
str().str();
10350 return OpenMPIRBuilder::getNameWithSeparators(Parts,
Config.firstSeparator(),
10356 auto &Elem = *
InternalVars.try_emplace(Name,
nullptr).first;
10358 assert(Elem.second->getValueType() == Ty &&
10359 "OMP internal variable has different type than requested");
10372 :
M.getTargetTriple().isAMDGPU()
10374 :
DL.getDefaultGlobalsAddressSpace();
10375 auto Linkage = this->
M.getTargetTriple().isWasm()
10383 const llvm::Align PtrAlign =
DL.getPointerABIAlignment(AddressSpaceVal);
10384 GV->setAlignment(std::max(TypeAlign, PtrAlign));
10388 return Elem.second;
10391Value *OpenMPIRBuilder::getOMPCriticalRegionLock(
StringRef CriticalName) {
10392 std::string Prefix =
Twine(
"gomp_critical_user_", CriticalName).
str();
10393 std::string Name = getNameWithSeparators({Prefix,
"var"},
".",
".");
10404 return SizePtrToInt;
10409 std::string VarName) {
10417 return MaptypesArrayGlobal;
10422 unsigned NumOperands,
10431 ArrI8PtrTy,
nullptr,
".offload_baseptrs");
10435 ArrI64Ty,
nullptr,
".offload_sizes");
10446 int64_t DeviceID,
unsigned NumOperands) {
10452 Value *ArgsBaseGEP =
10454 {Builder.getInt32(0), Builder.getInt32(0)});
10457 {Builder.getInt32(0), Builder.getInt32(0)});
10458 Value *ArgSizesGEP =
10460 {Builder.getInt32(0), Builder.getInt32(0)});
10464 Builder.getInt32(NumOperands),
10465 ArgsBaseGEP, ArgsGEP, ArgSizesGEP,
10466 MaptypesArg, MapnamesArg, NullPtr});
10473 assert((!ForEndCall || Info.separateBeginEndCalls()) &&
10474 "expected region end call to runtime only when end call is separate");
10476 auto VoidPtrTy = UnqualPtrTy;
10477 auto VoidPtrPtrTy = UnqualPtrTy;
10479 auto Int64PtrTy = UnqualPtrTy;
10481 if (!Info.NumberOfPtrs) {
10493 Info.RTArgs.BasePointersArray,
10496 ArrayType::get(VoidPtrTy, Info.NumberOfPtrs), Info.RTArgs.PointersArray,
10500 ArrayType::get(Int64Ty, Info.NumberOfPtrs), Info.RTArgs.SizesArray,
10504 ForEndCall && Info.RTArgs.MapTypesArrayEnd ? Info.RTArgs.MapTypesArrayEnd
10505 : Info.RTArgs.MapTypesArray,
10511 if (!Info.EmitDebug)
10515 ArrayType::get(VoidPtrTy, Info.NumberOfPtrs), Info.RTArgs.MapNamesArray,
10520 if (!Info.HasMapper)
10524 Builder.CreatePointerCast(Info.RTArgs.MappersArray, VoidPtrPtrTy);
10545 "struct.descriptor_dim");
10547 enum { OffsetFD = 0, CountFD, StrideFD };
10551 for (
unsigned I = 0, L = 0, E = NonContigInfo.
Dims.
size();
I < E; ++
I) {
10554 if (NonContigInfo.
Dims[
I] == 1)
10559 Builder.CreateAlloca(ArrayTy,
nullptr,
"dims");
10560 Builder.restoreIP(CodeGenIP);
10561 for (
unsigned II = 0, EE = NonContigInfo.
Dims[
I];
II < EE; ++
II) {
10562 unsigned RevIdx = EE -
II - 1;
10566 Value *OffsetLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, OffsetFD);
10568 NonContigInfo.
Offsets[L][RevIdx], OffsetLVal,
10569 M.getDataLayout().getPrefTypeAlign(OffsetLVal->
getType()));
10571 Value *CountLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, CountFD);
10573 NonContigInfo.
Counts[L][RevIdx], CountLVal,
10574 M.getDataLayout().getPrefTypeAlign(CountLVal->
getType()));
10576 Value *StrideLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, StrideFD);
10578 NonContigInfo.
Strides[L][RevIdx], StrideLVal,
10579 M.getDataLayout().getPrefTypeAlign(CountLVal->
getType()));
10582 Builder.restoreIP(CodeGenIP);
10583 Value *DAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
10584 DimsAddr,
Builder.getPtrTy());
10587 Info.RTArgs.PointersArray, 0,
I);
10589 DAddr,
P,
M.getDataLayout().getPrefTypeAlign(
Builder.getPtrTy()));
10594void OpenMPIRBuilder::emitUDMapperArrayInitOrDel(
10598 StringRef Prefix = IsInit ?
".init" :
".del";
10604 Builder.CreateICmpSGT(
Size, Builder.getInt64(1),
"omp.arrayinit.isarray");
10605 Value *DeleteBit = Builder.CreateAnd(
10608 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10609 OpenMPOffloadMappingFlags::OMP_MAP_DELETE)));
10614 Value *BaseIsBegin = Builder.CreateICmpNE(
Base, Begin);
10615 Cond = Builder.CreateOr(IsArray, BaseIsBegin);
10616 DeleteCond = Builder.CreateIsNull(
10621 DeleteCond =
Builder.CreateIsNotNull(
10637 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10638 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10639 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10640 MapTypeArg =
Builder.CreateOr(
10643 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10644 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT)));
10648 Value *OffloadingArgs[] = {MapperHandle,
Base, Begin,
10649 ArraySize, MapTypeArg, MapName};
10660 bool PreserveMemberOfFlags,
bool PropagatePresentToPointee) {
10676 MapperFn->
addFnAttr(Attribute::NoInline);
10677 MapperFn->
addFnAttr(Attribute::NoUnwind);
10688 Builder.SetInsertPoint(EntryBB);
10703 Value *PtrBegin = BeginIn;
10709 emitUDMapperArrayInitOrDel(MapperFn, MapperHandle, BaseIn, BeginIn,
Size,
10710 MapType, MapName, ElementSize, HeadBB,
10721 Builder.CreateICmpEQ(PtrBegin, PtrEnd,
"omp.arraymap.isempty");
10722 Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
10728 Builder.CreatePHI(PtrBegin->
getType(), 2,
"omp.arraymap.ptrcurrent");
10729 PtrPHI->addIncoming(PtrBegin, HeadBB);
10734 return Info.takeError();
10738 Value *OffloadingArgs[] = {MapperHandle};
10742 Value *ShiftedPreviousSize =
10746 for (
unsigned I = 0;
I < Info->BasePointers.size(); ++
I) {
10747 Value *CurBaseArg = Info->BasePointers[
I];
10748 Value *CurBeginArg = Info->Pointers[
I];
10749 Value *CurSizeArg = Info->Sizes[
I];
10750 Value *CurNameArg = Info->Names.size()
10755 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10758 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10760 constexpr uint64_t MemberOfMask =
10761 static_cast<uint64_t
>(OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF);
10762 constexpr uint64_t AttachBit =
10763 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10764 OpenMPOffloadMappingFlags::OMP_MAP_ATTACH);
10822 Value *MemberMapType;
10823 if (PreserveMemberOfFlags || (RawType & AttachBit) ||
10824 Info->HasAttachPtr[
I]) {
10825 if (RawType & MemberOfMask)
10826 MemberMapType =
Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
10828 MemberMapType = OriMapType;
10830 MemberMapType =
Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
10848 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10849 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10850 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10860 Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB);
10866 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10867 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10868 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10874 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10875 OpenMPOffloadMappingFlags::OMP_MAP_TO)));
10876 Builder.CreateCondBr(IsTo, ToBB, ToElseBB);
10882 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10883 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10889 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10890 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10891 Builder.CreateCondBr(IsFrom, FromBB, EndBB);
10897 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10898 OpenMPOffloadMappingFlags::OMP_MAP_TO)));
10907 CurMapType->
addIncoming(MemberMapType, ToElseBB);
10944 uint64_t ModifierBits =
10945 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10946 OpenMPOffloadMappingFlags::OMP_MAP_ALWAYS |
10947 OpenMPOffloadMappingFlags::OMP_MAP_DELETE |
10948 OpenMPOffloadMappingFlags::OMP_MAP_CLOSE);
10949 if (PropagatePresentToPointee && Info->HasAttachPtr[
I])
10951 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10952 OpenMPOffloadMappingFlags::OMP_MAP_PRESENT);
10953 Value *ImportedModifierBits =
10956 CurMapType, ImportedModifierBits,
"omp.maptype.with.modifiers");
10961 Value *FinalMapType =
10962 (RawType & AttachBit) ? CurMapType : CurMapTypeWithModifiers;
10964 Value *OffloadingArgs[] = {MapperHandle, CurBaseArg, CurBeginArg,
10965 CurSizeArg, FinalMapType, CurNameArg};
10967 auto ChildMapperFn = CustomMapperCB(
I);
10968 if (!ChildMapperFn)
10969 return ChildMapperFn.takeError();
10970 if (*ChildMapperFn) {
10986 "omp.arraymap.next");
10987 PtrPHI->addIncoming(PtrNext, LastBB);
10988 Value *IsDone =
Builder.CreateICmpEQ(PtrNext, PtrEnd,
"omp.arraymap.isdone");
10990 Builder.CreateCondBr(IsDone, ExitBB, BodyBB);
10995 emitUDMapperArrayInitOrDel(MapperFn, MapperHandle, BaseIn, BeginIn,
Size,
10996 MapType, MapName, ElementSize, DoneBB,
11009 bool IsNonContiguous,
11013 Info.clearArrayInfo();
11016 if (Info.NumberOfPtrs == 0)
11025 Info.RTArgs.BasePointersArray =
Builder.CreateAlloca(
11026 PointerArrayType,
nullptr,
".offload_baseptrs");
11028 Info.RTArgs.PointersArray =
Builder.CreateAlloca(
11029 PointerArrayType,
nullptr,
".offload_ptrs");
11031 PointerArrayType,
nullptr,
".offload_mappers");
11032 Info.RTArgs.MappersArray = MappersArray;
11039 ConstantInt::get(Int64Ty, 0));
11041 for (
unsigned I = 0, E = CombinedInfo.
Sizes.
size();
I < E; ++
I) {
11042 bool IsNonContigEntry =
11044 (
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
11046 OpenMPOffloadMappingFlags::OMP_MAP_NON_CONTIG) != 0);
11049 if (IsNonContigEntry) {
11051 "Index must be in-bounds for NON_CONTIG Dims array");
11053 assert(DimCount > 0 &&
"NON_CONTIG DimCount must be > 0");
11054 ConstSizes[
I] = ConstantInt::get(Int64Ty, DimCount);
11059 ConstSizes[
I] = CI;
11063 RuntimeSizes.
set(
I);
11066 if (RuntimeSizes.
all()) {
11068 Info.RTArgs.SizesArray =
Builder.CreateAlloca(
11069 SizeArrayType,
nullptr,
".offload_sizes");
11075 auto *SizesArrayGbl =
11080 if (!RuntimeSizes.
any()) {
11081 Info.RTArgs.SizesArray = SizesArrayGbl;
11083 unsigned IndexSize =
M.getDataLayout().getIndexSizeInBits(0);
11084 Align OffloadSizeAlign =
M.getDataLayout().getABIIntegerTypeAlignment(64);
11087 SizeArrayType,
nullptr,
".offload_sizes");
11091 Buffer,
M.getDataLayout().getPrefTypeAlign(Buffer->
getType()),
11092 SizesArrayGbl, OffloadSizeAlign,
11097 Info.RTArgs.SizesArray = Buffer;
11105 for (
auto mapFlag : CombinedInfo.
Types)
11107 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
11111 Info.RTArgs.MapTypesArray = MapTypesArrayGbl;
11117 Info.RTArgs.MapNamesArray = MapNamesArrayGbl;
11118 Info.EmitDebug =
true;
11120 Info.RTArgs.MapNamesArray =
11122 Info.EmitDebug =
false;
11127 if (Info.separateBeginEndCalls()) {
11128 bool EndMapTypesDiffer =
false;
11129 for (uint64_t &
Type : Mapping) {
11130 if (
Type &
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
11131 OpenMPOffloadMappingFlags::OMP_MAP_PRESENT)) {
11132 Type &= ~static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
11133 OpenMPOffloadMappingFlags::OMP_MAP_PRESENT);
11134 EndMapTypesDiffer =
true;
11137 if (EndMapTypesDiffer) {
11139 Info.RTArgs.MapTypesArrayEnd = MapTypesArrayGbl;
11144 for (
unsigned I = 0;
I < Info.NumberOfPtrs; ++
I) {
11147 ArrayType::get(PtrTy, Info.NumberOfPtrs), Info.RTArgs.BasePointersArray,
11149 Builder.CreateAlignedStore(BPVal, BP,
11150 M.getDataLayout().getPrefTypeAlign(PtrTy));
11152 if (Info.requiresDevicePointerInfo()) {
11154 CodeGenIP =
Builder.saveIP();
11156 Info.DevicePtrInfoMap[BPVal] = {BP,
Builder.CreateAlloca(PtrTy)};
11159 DeviceAddrCB(
I, Info.DevicePtrInfoMap[BPVal].second);
11161 Info.DevicePtrInfoMap[BPVal] = {BP, BP};
11163 DeviceAddrCB(
I, BP);
11169 ArrayType::get(PtrTy, Info.NumberOfPtrs), Info.RTArgs.PointersArray, 0,
11172 Builder.CreateAlignedStore(PVal,
P,
11173 M.getDataLayout().getPrefTypeAlign(PtrTy));
11175 if (RuntimeSizes.
test(
I)) {
11177 ArrayType::get(Int64Ty, Info.NumberOfPtrs), Info.RTArgs.SizesArray,
11183 S,
M.getDataLayout().getPrefTypeAlign(PtrTy));
11186 unsigned IndexSize =
M.getDataLayout().getIndexSizeInBits(0);
11189 auto CustomMFunc = CustomMapperCB(
I);
11191 return CustomMFunc.takeError();
11193 MFunc =
Builder.CreatePointerCast(*CustomMFunc, PtrTy);
11196 PointerArrayType, MappersArray,
11199 MFunc, MAddr,
M.getDataLayout().getPrefTypeAlign(MAddr->
getType()));
11203 Info.NumberOfPtrs == 0)
11220 Builder.ClearInsertionPoint();
11251 auto CondConstant = CI->getSExtValue();
11253 return ThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
11255 return ElseGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
11265 Builder.CreateCondBr(
Cond, ThenBlock, ElseBlock);
11268 if (
Error Err = ThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks))
11274 if (
Error Err = ElseGen(AllocaIP,
Builder.saveIP(), DeallocBlocks))
11283bool OpenMPIRBuilder::checkAndEmitFlushAfterAtomic(
11287 "Unexpected Atomic Ordering.");
11289 bool Flush =
false;
11351 assert(
X.Var->getType()->isPointerTy() &&
11352 "OMP Atomic expects a pointer to target memory");
11353 Type *XElemTy =
X.ElemTy;
11356 "OMP atomic read expected a scalar type");
11358 Value *XRead =
nullptr;
11362 Builder.CreateLoad(XElemTy,
X.Var,
X.IsVolatile,
"omp.atomic.read");
11371 unsigned LoadSize =
DL.getTypeStoreSize(XElemTy);
11374 OldVal->
getAlign(),
true , AllocaIP,
X.Var);
11376 XRead = AtomicLoadRes.first;
11383 Builder.CreateLoad(IntCastTy,
X.Var,
X.IsVolatile,
"omp.atomic.load");
11386 XRead =
Builder.CreateBitCast(XLoad, XElemTy,
"atomic.flt.cast");
11388 XRead =
Builder.CreateIntToPtr(XLoad, XElemTy,
"atomic.ptr.cast");
11391 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Read);
11392 Builder.CreateStore(XRead, V.Var, V.IsVolatile);
11403 assert(
X.Var->getType()->isPointerTy() &&
11404 "OMP Atomic expects a pointer to target memory");
11405 Type *XElemTy =
X.ElemTy;
11408 "OMP atomic write expected a scalar type");
11416 unsigned LoadSize =
DL.getTypeStoreSize(XElemTy);
11419 OldVal->
getAlign(),
true , AllocaIP,
X.Var);
11427 Builder.CreateBitCast(Expr, IntCastTy,
"atomic.src.int.cast");
11432 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Write);
11439 AtomicUpdateCallbackTy &UpdateOp,
bool IsXBinopExpr,
11440 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11446 Type *XTy =
X.Var->getType();
11448 "OMP Atomic expects a pointer to target memory");
11449 Type *XElemTy =
X.ElemTy;
11452 "OMP atomic update expected a scalar or struct type");
11455 "OpenMP atomic does not support LT or GT operations");
11459 AllocaIP,
X.Var,
X.ElemTy, Expr, AO, RMWOp, UpdateOp,
X.IsVolatile,
11460 IsXBinopExpr, IsIgnoreDenormalMode, IsFineGrainedMemory, IsRemoteMemory);
11462 return AtomicResult.takeError();
11463 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Update);
11468Value *OpenMPIRBuilder::emitRMWOpAsInstruction(
Value *Src1,
Value *Src2,
11472 return Builder.CreateAdd(Src1, Src2);
11474 return Builder.CreateSub(Src1, Src2);
11476 return Builder.CreateAnd(Src1, Src2);
11478 return Builder.CreateNeg(Builder.CreateAnd(Src1, Src2));
11480 return Builder.CreateOr(Src1, Src2);
11482 return Builder.CreateXor(Src1, Src2);
11521Expected<std::pair<Value *, Value *>> OpenMPIRBuilder::emitAtomicUpdate(
11524 AtomicUpdateCallbackTy &UpdateOp,
bool VolatileX,
bool IsXBinopExpr,
11525 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11527 bool emitRMWOp =
false;
11535 emitRMWOp = XElemTy;
11538 emitRMWOp = (IsXBinopExpr && XElemTy);
11545 std::pair<Value *, Value *> Res;
11547 AtomicRMWInst *RMWInst =
11548 Builder.CreateAtomicRMW(RMWOp,
X, Expr, llvm::MaybeAlign(), AO);
11549 if (IsIgnoreDenormalMode)
11550 RMWInst->
setMetadata(llvm::LLVMContext::MD_atomic_ignore_denormal_mode,
11552 if (
T.isAMDGPU()) {
11553 if (!IsFineGrainedMemory)
11554 RMWInst->
setMetadata(
"amdgpu.no.fine.grained.memory",
11556 if (!IsRemoteMemory)
11560 Res.first = RMWInst;
11565 Res.second = Res.first;
11567 Res.second = emitRMWOpAsInstruction(Res.first, Expr, RMWOp);
11570 Builder.CreateLoad(XElemTy,
X,
X->getName() +
".atomic.load");
11576 OpenMPIRBuilder::AtomicInfo atomicInfo(
11578 OldVal->
getAlign(),
true , AllocaIP,
X);
11579 auto AtomicLoadRes = atomicInfo.EmitAtomicLoadLibcall(AO);
11582 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11589 AllocaInst *NewAtomicAddr =
Builder.CreateAlloca(XElemTy);
11590 NewAtomicAddr->
setName(
X->getName() +
"x.new.val");
11591 Builder.SetInsertPoint(ContBB);
11593 PHI->addIncoming(AtomicLoadRes.first, CurBB);
11595 Expected<Value *> CBResult = UpdateOp(OldExprVal,
Builder);
11598 Value *Upd = *CBResult;
11599 Builder.CreateStore(Upd, NewAtomicAddr);
11602 auto Result = atomicInfo.EmitAtomicCompareExchangeLibcall(
11603 AtomicLoadRes.second, NewAtomicAddr, AO, Failure);
11604 LoadInst *PHILoad =
Builder.CreateLoad(XElemTy,
Result.first);
11605 PHI->addIncoming(PHILoad,
Builder.GetInsertBlock());
11608 Res.first = OldExprVal;
11611 if (UnreachableInst *ExitTI =
11614 Builder.SetInsertPoint(ExitBB);
11616 Builder.SetInsertPoint(ExitTI);
11619 IntegerType *IntCastTy =
11622 Builder.CreateLoad(IntCastTy,
X,
X->getName() +
".atomic.load");
11632 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11639 AllocaInst *NewAtomicAddr =
Builder.CreateAlloca(XElemTy);
11640 NewAtomicAddr->
setName(
X->getName() +
"x.new.val");
11641 Builder.SetInsertPoint(ContBB);
11643 PHI->addIncoming(OldVal, CurBB);
11648 OldExprVal =
Builder.CreateBitCast(
PHI, XElemTy,
11649 X->getName() +
".atomic.fltCast");
11651 OldExprVal =
Builder.CreateIntToPtr(
PHI, XElemTy,
11652 X->getName() +
".atomic.ptrCast");
11656 Expected<Value *> CBResult = UpdateOp(OldExprVal,
Builder);
11659 Value *Upd = *CBResult;
11660 Builder.CreateStore(Upd, NewAtomicAddr);
11661 LoadInst *DesiredVal =
Builder.CreateLoad(IntCastTy, NewAtomicAddr);
11665 X,
PHI, DesiredVal, llvm::MaybeAlign(), AO, Failure);
11666 Result->setVolatile(VolatileX);
11667 Value *PreviousVal =
Builder.CreateExtractValue(Result, 0);
11668 Value *SuccessFailureVal =
Builder.CreateExtractValue(Result, 1);
11669 PHI->addIncoming(PreviousVal,
Builder.GetInsertBlock());
11670 Builder.CreateCondBr(SuccessFailureVal, ExitBB, ContBB);
11672 Res.first = OldExprVal;
11676 if (UnreachableInst *ExitTI =
11679 Builder.SetInsertPoint(ExitBB);
11681 Builder.SetInsertPoint(ExitTI);
11692 bool UpdateExpr,
bool IsPostfixUpdate,
bool IsXBinopExpr,
11693 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11698 Type *XTy =
X.Var->getType();
11700 "OMP Atomic expects a pointer to target memory");
11701 Type *XElemTy =
X.ElemTy;
11704 "OMP atomic capture expected a scalar or struct type");
11706 "OpenMP atomic does not support LT or GT operations");
11713 AllocaIP,
X.Var,
X.ElemTy, Expr, AO, AtomicOp, UpdateOp,
X.IsVolatile,
11714 IsXBinopExpr, IsIgnoreDenormalMode, IsFineGrainedMemory, IsRemoteMemory);
11717 Value *CapturedVal =
11718 (IsPostfixUpdate ? AtomicResult->first : AtomicResult->second);
11719 Builder.CreateStore(CapturedVal, V.Var, V.IsVolatile);
11721 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Capture);
11729 bool IsFailOnly,
bool IsWeak) {
11733 IsPostfixUpdate, IsFailOnly, Failure, IsWeak);
11745 assert(
X.Var->getType()->isPointerTy() &&
11746 "OMP atomic expects a pointer to target memory");
11749 assert(V.Var->getType()->isPointerTy() &&
"v.var must be of pointer type");
11750 assert(V.ElemTy ==
X.ElemTy &&
"x and v must be of same type");
11753 bool IsInteger = E->getType()->isIntegerTy();
11755 if (
Op == OMPAtomicCompareOp::EQ) {
11758 Value *OldValue =
nullptr;
11759 Value *SuccessOrFail =
nullptr;
11797 X.Var->getName() +
".atomic.load");
11803 Value *EIsNaN =
Builder.CreateFCmpUNO(E, E,
"atomic.e.isnan");
11804 Value *XIsNaN =
Builder.CreateFCmpUNO(XFP, XFP,
"atomic.x.isnan");
11805 Value *EitherNaN =
Builder.CreateOr(EIsNaN, XIsNaN,
"atomic.either.nan");
11810 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11814 M.getContext(),
X.Var->getName() +
".atomic.nan",
F, ExitBB);
11816 M.getContext(),
X.Var->getName() +
".atomic.notnan",
F, ExitBB);
11818 M.getContext(),
X.Var->getName() +
".atomic.zero",
F, ExitBB);
11820 M.getContext(),
X.Var->getName() +
".atomic.normal",
F, ExitBB);
11824 Builder.SetInsertPoint(CurBB);
11825 Builder.CreateCondBr(EitherNaN, NaNBB, NotNaNBB);
11828 Builder.SetInsertPoint(NaNBB);
11832 Builder.SetInsertPoint(NotNaNBB);
11835 X.Var->getName() +
".atomic.xiszero");
11837 "atomic.e.iszero");
11838 Value *BothZero =
Builder.CreateAnd(XIsZero, EIsZero,
"atomic.both.zero");
11839 Builder.CreateCondBr(BothZero, ZeroBB, NormalBB);
11842 Builder.SetInsertPoint(ZeroBB);
11844 X.Var, XCurr, DBCast,
MaybeAlign(), AO, Failure);
11846 Value *OldZero =
Builder.CreateExtractValue(ResZero, 0);
11847 Value *OkZero =
Builder.CreateExtractValue(ResZero, 1);
11851 Builder.SetInsertPoint(NormalBB);
11853 X.Var, EBCast, DBCast,
MaybeAlign(), AO, Failure);
11855 Value *OldNormal =
Builder.CreateExtractValue(ResNormal, 0);
11856 Value *OkNormal =
Builder.CreateExtractValue(ResNormal, 1);
11862 Builder.CreatePHI(IntCastTy, 3,
X.Var->getName() +
".atomic.old");
11867 X.Var->getName() +
".atomic.ok");
11874 Builder.SetInsertPoint(ExitBB);
11879 OldValue =
Builder.CreateBitCast(OldIntPHI,
X.ElemTy,
11880 X.Var->getName() +
".atomic.old.fp");
11881 SuccessOrFail = SuccessPHI;
11889 Result =
Builder.CreateAtomicCmpXchg(
X.Var, EBCast, DBCast,
11895 Result->setWeak(IsWeak);
11898 OldValue =
Builder.CreateExtractValue(Result, 0);
11900 OldValue =
Builder.CreateBitCast(OldValue,
X.ElemTy);
11902 "OldValue and V must be of same type");
11903 if (IsPostfixUpdate) {
11904 Builder.CreateStore(OldValue, V.Var, V.IsVolatile);
11906 SuccessOrFail =
Builder.CreateExtractValue(Result, 1);
11910 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11912 CurBBTI,
X.Var->getName() +
".atomic.exit");
11918 Builder.CreateCondBr(SuccessOrFail, ExitBB, ContBB);
11920 Builder.SetInsertPoint(ContBB);
11921 Builder.CreateStore(OldValue, V.Var);
11927 Builder.SetInsertPoint(ExitBB);
11929 Builder.SetInsertPoint(ExitTI);
11932 Value *CapturedValue =
11933 Builder.CreateSelect(SuccessOrFail, E, OldValue);
11934 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
11940 assert(R.Var->getType()->isPointerTy() &&
11941 "r.var must be of pointer type");
11942 assert(R.ElemTy->isIntegerTy() &&
"r must be of integral type");
11944 Value *SuccessFailureVal =
11945 Builder.CreateExtractValue(Result, 1);
11946 Value *ResultCast =
11947 R.IsSigned ?
Builder.CreateSExt(SuccessFailureVal, R.ElemTy)
11948 :
Builder.CreateZExt(SuccessFailureVal, R.ElemTy);
11949 Builder.CreateStore(ResultCast, R.Var, R.IsVolatile);
11958 "OldValue and V must be of same type");
11959 if (IsPostfixUpdate) {
11960 Builder.CreateStore(OldValue, V.Var, V.IsVolatile);
11965 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11967 CurBBTI,
X.Var->getName() +
".atomic.exit");
11973 Builder.CreateCondBr(SuccessOrFail, ExitBB, ContBB);
11975 Builder.SetInsertPoint(ContBB);
11976 Builder.CreateStore(OldValue, V.Var);
11982 Builder.SetInsertPoint(ExitBB);
11984 Builder.SetInsertPoint(ExitTI);
11987 Value *CapturedValue =
11988 Builder.CreateSelect(SuccessOrFail, E, OldValue);
11989 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
11995 assert(R.Var->getType()->isPointerTy() &&
11996 "r.var must be of pointer type");
11997 assert(R.ElemTy->isIntegerTy() &&
"r must be of integral type");
11999 Value *ResultCast = R.IsSigned
12000 ?
Builder.CreateSExt(SuccessOrFail, R.ElemTy)
12001 :
Builder.CreateZExt(SuccessOrFail, R.ElemTy);
12002 Builder.CreateStore(ResultCast, R.Var, R.IsVolatile);
12006 assert((
Op == OMPAtomicCompareOp::MAX ||
Op == OMPAtomicCompareOp::MIN) &&
12007 "Op should be either max or min at this point");
12008 assert(!IsFailOnly &&
"IsFailOnly is only valid when the comparison is ==");
12019 if (IsXBinopExpr) {
12048 Value *CapturedValue =
nullptr;
12049 if (IsPostfixUpdate) {
12050 CapturedValue = OldValue;
12075 Value *NonAtomicCmp =
Builder.CreateCmp(Pred, OldValue, E);
12076 CapturedValue =
Builder.CreateSelect(NonAtomicCmp, E, OldValue);
12078 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
12082 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Compare);
12102 if (&OuterAllocaBB ==
Builder.GetInsertBlock()) {
12129 bool SubClausesPresent =
12130 (NumTeamsLower || NumTeamsUpper || ThreadLimit || IfExpr);
12132 if (!
Config.isTargetDevice() && SubClausesPresent) {
12133 assert((NumTeamsLower ==
nullptr || NumTeamsUpper !=
nullptr) &&
12134 "if lowerbound is non-null, then upperbound must also be non-null "
12135 "for bounds on num_teams");
12137 if (NumTeamsUpper ==
nullptr)
12138 NumTeamsUpper =
Builder.getInt32(0);
12140 if (NumTeamsLower ==
nullptr)
12141 NumTeamsLower = NumTeamsUpper;
12145 "argument to if clause must be an integer value");
12149 IfExpr =
Builder.CreateICmpNE(IfExpr,
12150 ConstantInt::get(IfExpr->
getType(), 0));
12151 NumTeamsUpper =
Builder.CreateSelect(
12152 IfExpr, NumTeamsUpper,
Builder.getInt32(1),
"numTeamsUpper");
12155 NumTeamsLower =
Builder.CreateSelect(
12156 IfExpr, NumTeamsLower,
Builder.getInt32(1),
"numTeamsLower");
12159 if (ThreadLimit ==
nullptr)
12160 ThreadLimit =
Builder.getInt32(0);
12164 Value *NumTeamsLowerInt32 =
12166 Value *NumTeamsUpperInt32 =
12168 Value *ThreadLimitInt32 =
12175 {Ident, ThreadNum, NumTeamsLowerInt32, NumTeamsUpperInt32,
12176 ThreadLimitInt32});
12181 if (
Error Err = BodyGenCB(AllocaIP, CodeGenIP, ExitBB))
12184 auto OI = std::make_unique<OutlineInfo>();
12185 OI->EntryBB = AllocaBB;
12186 OI->ExitBB = ExitBB;
12187 OI->OuterAllocBB = &OuterAllocaBB;
12193 Builder, OuterAllocaIP, ToBeDeleted, AllocaIP,
"gid",
true));
12195 Builder, OuterAllocaIP, ToBeDeleted, AllocaIP,
"tid",
true));
12197 auto HostPostOutlineCB = [
this, Ident,
12198 ToBeDeleted](
Function &OutlinedFn)
mutable {
12203 "there must be a single user for the outlined function");
12208 "Outlined function must have two or three arguments only");
12210 bool HasShared = OutlinedFn.
arg_size() == 3;
12218 assert(StaleCI &&
"Error while outlining - no CallInst user found for the "
12219 "outlined function.");
12220 Builder.SetInsertPoint(StaleCI);
12227 omp::RuntimeFunction::OMPRTL___kmpc_fork_teams),
12230 Builder.ClearInsertionPoint();
12232 I->eraseFromParent();
12235 if (!
Config.isTargetDevice())
12236 OI->PostOutlineCB = HostPostOutlineCB;
12240 Builder.SetInsertPoint(ExitBB);
12251 BasicBlock *OuterAllocaBB = OuterAllocIP.getNodeParent();
12253 if (OuterAllocaBB ==
Builder.GetInsertBlock()) {
12268 if (
Error Err = BodyGenCB(AllocaIP, CodeGenIP, ExitBB))
12273 if (
Config.isTargetDevice()) {
12274 auto OI = std::make_unique<OutlineInfo>();
12275 OI->OuterAllocBB = OuterAllocIP.getNodeParent();
12276 OI->EntryBB = AllocaBB;
12277 OI->ExitBB = ExitBB;
12278 OI->OuterDeallocBBs.reserve(OuterDeallocBlocks.
size());
12279 copy(OuterDeallocBlocks, OI->OuterDeallocBBs.
end());
12283 Builder.SetInsertPoint(ExitBB);
12290 std::string VarName) {
12299 return MapNamesArrayGlobal;
12304void OpenMPIRBuilder::initializeTypes(
Module &M) {
12308 unsigned ProgramAS = M.getDataLayout().getProgramAddressSpace();
12309#define OMP_TYPE(VarName, InitValue) VarName = InitValue;
12310#define OMP_ARRAY_TYPE(VarName, ElemTy, ArraySize) \
12311 VarName##Ty = ArrayType::get(ElemTy, ArraySize); \
12312 VarName##PtrTy = PointerType::get(Ctx, DefaultTargetAS);
12313#define OMP_FUNCTION_TYPE(VarName, IsVarArg, ReturnType, ...) \
12314 VarName = FunctionType::get(ReturnType, {__VA_ARGS__}, IsVarArg); \
12315 VarName##Ptr = PointerType::get(Ctx, ProgramAS);
12316#define OMP_STRUCT_TYPE(VarName, StructName, Packed, ...) \
12317 T = StructType::getTypeByName(Ctx, StructName); \
12319 T = StructType::create(Ctx, {__VA_ARGS__}, StructName, Packed); \
12321 VarName##Ptr = PointerType::get(Ctx, DefaultTargetAS);
12322#include "llvm/Frontend/OpenMP/OMPKinds.def"
12333 while (!Worklist.
empty()) {
12337 if (
BlockSet.insert(SuccBB).second)
12342std::unique_ptr<CodeExtractor>
12344 bool ArgsInZeroAddressSpace,
12346 return std::make_unique<CodeExtractor>(
12356 Suffix.
str(), ArgsInZeroAddressSpace);
12359std::unique_ptr<CodeExtractor> DeviceSharedMemOutlineInfo::createCodeExtractor(
12361 return std::make_unique<DeviceSharedMemCodeExtractor>(
12362 OMPBuilder, Blocks,
nullptr,
12370 OuterDeallocBBs.empty()
12373 Suffix.
str(), ArgsInZeroAddressSpace);
12377 uint64_t
Size, int32_t Flags,
12383 Name.empty() ? Addr->
getName() : Name,
Size, Flags, 0);
12395 Fn->
addFnAttr(
"uniform-work-group-size");
12396 Fn->
addFnAttr(Attribute::MustProgress);
12414 auto &&GetMDInt = [
this](
unsigned V) {
12421 NamedMDNode *MD =
M.getOrInsertNamedMetadata(
"omp_offload.info");
12422 auto &&TargetRegionMetadataEmitter =
12423 [&
C, MD, &OrderedEntries, &GetMDInt, &GetMDString](
12438 GetMDInt(E.getKind()), GetMDInt(EntryInfo.DeviceID),
12439 GetMDInt(EntryInfo.FileID), GetMDString(EntryInfo.ParentName),
12440 GetMDInt(EntryInfo.Line), GetMDInt(EntryInfo.Count),
12441 GetMDInt(E.getOrder())};
12444 OrderedEntries[E.getOrder()] = std::make_pair(&E, EntryInfo);
12453 auto &&DeviceGlobalVarMetadataEmitter =
12454 [&
C, &OrderedEntries, &GetMDInt, &GetMDString, MD](
12464 Metadata *
Ops[] = {GetMDInt(E.getKind()), GetMDString(MangledName),
12465 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())};
12469 OrderedEntries[E.getOrder()] = std::make_pair(&E, varInfo);
12476 DeviceGlobalVarMetadataEmitter);
12478 for (
const auto &E : OrderedEntries) {
12479 assert(E.first &&
"All ordered entries must exist!");
12480 if (
const auto *CE =
12483 if (!CE->getID() || !CE->getAddress()) {
12487 if (!
M.getNamedValue(FnName))
12495 }
else if (
const auto *CE =
dyn_cast<
12504 if (
Config.isTargetDevice() &&
Config.hasRequiresUnifiedSharedMemory())
12506 if (!CE->getAddress()) {
12511 if (CE->getVarSize() == 0)
12515 assert(((
Config.isTargetDevice() && !CE->getAddress()) ||
12516 (!
Config.isTargetDevice() && CE->getAddress())) &&
12517 "Declaret target link address is set.");
12518 if (
Config.isTargetDevice())
12520 if (!CE->getAddress()) {
12527 if (!CE->getAddress()) {
12540 if ((
GV->hasLocalLinkage() ||
GV->hasHiddenVisibility()) &&
12544 OMPTargetGlobalVarEntryIndirectVTable))
12553 Flags, CE->getLinkage(), CE->getVarName());
12556 Flags, CE->getLinkage());
12567 if (
Config.hasRequiresFlags() && !
Config.isTargetDevice())
12573 Config.getRequiresFlags());
12583 OS <<
"_" <<
Count;
12588 unsigned NewCount = getTargetRegionEntryInfoCount(EntryInfo);
12591 EntryInfo.
Line, NewCount);
12599 auto FileIDInfo = CallBack();
12600 uint64_t FileID = 0;
12602 ID =
Status->getUniqueID();
12603 FileID =
Status->getUniqueID().getFile();
12607 FileID =
hash_value(std::get<0>(FileIDInfo));
12611 std::get<1>(FileIDInfo));
12616 for (uint64_t Remain =
12617 static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12619 !(Remain & 1); Remain = Remain >> 1)
12637 if (
static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12639 static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12646 if (
static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12652 Flags &=
~omp::OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF;
12653 Flags |= MemberOfFlag;
12659 bool IsDeclaration,
bool IsExternallyVisible,
12661 std::vector<GlobalVariable *> &GeneratedRefs,
bool OpenMPSIMD,
12662 std::vector<Triple> TargetTriple,
Type *LlvmPtrTy,
12663 std::function<
Constant *()> GlobalInitializer,
12674 Config.hasRequiresUnifiedSharedMemory())) {
12679 if (!IsExternallyVisible)
12681 OS <<
"_decl_tgt_ref_ptr";
12684 Value *Ptr =
M.getNamedValue(PtrName);
12693 if (!
Config.isTargetDevice()) {
12694 if (GlobalInitializer)
12695 GV->setInitializer(GlobalInitializer());
12701 CaptureClause, DeviceClause, IsDeclaration, IsExternallyVisible,
12702 EntryInfo, MangledName, GeneratedRefs, OpenMPSIMD, TargetTriple,
12703 GlobalInitializer, VariableLinkage, LlvmPtrTy,
cast<Constant>(Ptr));
12715 bool IsDeclaration,
bool IsExternallyVisible,
12717 std::vector<GlobalVariable *> &GeneratedRefs,
bool OpenMPSIMD,
12718 std::vector<Triple> TargetTriple,
12719 std::function<
Constant *()> GlobalInitializer,
12723 (TargetTriple.empty() && !
Config.isTargetDevice()))
12734 !
Config.hasRequiresUnifiedSharedMemory()) {
12736 VarName = MangledName;
12739 if (!IsDeclaration)
12741 M.getDataLayout().getTypeSizeInBits(LlvmVal->
getValueType()), 8);
12744 Linkage = (VariableLinkage) ? VariableLinkage() : LlvmVal->
getLinkage();
12748 if (
Config.isTargetDevice() &&
12757 if (!
M.getNamedValue(RefName)) {
12761 GvAddrRef->setConstant(
true);
12763 GvAddrRef->setInitializer(Addr);
12764 GeneratedRefs.push_back(GvAddrRef);
12773 if (
Config.isTargetDevice()) {
12774 VarName = (Addr) ? Addr->
getName() :
"";
12778 CaptureClause, DeviceClause, IsDeclaration, IsExternallyVisible,
12779 EntryInfo, MangledName, GeneratedRefs, OpenMPSIMD, TargetTriple,
12780 LlvmPtrTy, GlobalInitializer, VariableLinkage);
12781 VarName = (Addr) ? Addr->
getName() :
"";
12783 VarSize =
M.getDataLayout().getPointerSize();
12802 auto &&GetMDInt = [MN](
unsigned Idx) {
12807 auto &&GetMDString = [MN](
unsigned Idx) {
12809 return V->getString();
12812 switch (GetMDInt(0)) {
12816 case OffloadEntriesInfoManager::OffloadEntryInfo::
12817 OffloadingEntryInfoTargetRegion: {
12827 case OffloadEntriesInfoManager::OffloadEntryInfo::
12828 OffloadingEntryInfoDeviceGlobalVar:
12841 if (HostFilePath.
empty())
12845 if (std::error_code Err = Buf.getError()) {
12847 "OpenMPIRBuilder: " +
12855 if (std::error_code Err =
M.getError()) {
12857 (
"error parsing host file inside of OpenMPIRBuilder: " + Err.message())
12871 "expected a valid insertion block for creating an iterator loop");
12875 if (SplitIP == CurBB->
end())
12877 SplitIP = Terminator->getIterator();
12881 Builder.getCurrentDebugLocation(),
"omp.it.cont");
12893 T->eraseFromParent();
12902 if (!BodyBr || BodyBr->getSuccessor() != CLI->
getLatch()) {
12904 "iterator bodygen must terminate the canonical body with an "
12905 "unconditional branch to the loop latch",
12919 return ContBB->
begin();
12929 for (
const auto &
ParamAttr : ParamAttrs) {
12972 return std::string(Out.str());
12980 unsigned VecRegSize;
12982 ISADataTy ISAData[] = {
13001 for (
char Mask :
Masked) {
13002 for (
const ISADataTy &
Data : ISAData) {
13005 Out <<
"_ZGV" <<
Data.ISA << Mask;
13007 assert(NumElts &&
"Non-zero simdlen/cdtsize expected");
13021template <
typename T>
13024 StringRef MangledName,
bool OutputBecomesInput,
13028 Out << Prefix << ISA << LMask << VLEN;
13029 if (OutputBecomesInput)
13031 Out << ParSeq <<
'_' << MangledName;
13040 bool OutputBecomesInput,
13045 OutputBecomesInput, Fn);
13047 OutputBecomesInput, Fn);
13051 OutputBecomesInput, Fn);
13053 OutputBecomesInput, Fn);
13057 OutputBecomesInput, Fn);
13059 OutputBecomesInput, Fn);
13064 OutputBecomesInput, Fn);
13075 char ISA,
unsigned NarrowestDataSize,
bool OutputBecomesInput) {
13076 assert((ISA ==
'n' || ISA ==
's') &&
"Expected ISA either 's' or 'n'.");
13088 OutputBecomesInput, Fn);
13095 OutputBecomesInput, Fn);
13097 OutputBecomesInput, Fn);
13101 OutputBecomesInput, Fn);
13105 OutputBecomesInput, Fn);
13114 OutputBecomesInput, Fn);
13121 MangledName, OutputBecomesInput, Fn);
13123 MangledName, OutputBecomesInput, Fn);
13127 MangledName, OutputBecomesInput, Fn);
13131 MangledName, OutputBecomesInput, Fn);
13141 return OffloadEntriesTargetRegion.empty() &&
13142 OffloadEntriesDeviceGlobalVar.empty();
13145unsigned OffloadEntriesInfoManager::getTargetRegionEntryInfoCount(
13147 auto It = OffloadEntriesTargetRegionCount.find(
13148 getTargetRegionEntryCountKey(EntryInfo));
13149 if (It == OffloadEntriesTargetRegionCount.end())
13154void OffloadEntriesInfoManager::incrementTargetRegionEntryInfoCount(
13156 OffloadEntriesTargetRegionCount[getTargetRegionEntryCountKey(EntryInfo)] =
13157 EntryInfo.
Count + 1;
13163 OffloadEntriesTargetRegion[EntryInfo] =
13166 ++OffloadingEntriesNum;
13172 assert(EntryInfo.
Count == 0 &&
"expected default EntryInfo");
13175 EntryInfo.
Count = getTargetRegionEntryInfoCount(EntryInfo);
13179 if (OMPBuilder->Config.isTargetDevice()) {
13184 auto &Entry = OffloadEntriesTargetRegion[EntryInfo];
13185 Entry.setAddress(Addr);
13187 Entry.setFlags(Flags);
13193 "Target region entry already registered!");
13195 OffloadEntriesTargetRegion[EntryInfo] = Entry;
13196 ++OffloadingEntriesNum;
13198 incrementTargetRegionEntryInfoCount(EntryInfo);
13205 EntryInfo.
Count = getTargetRegionEntryInfoCount(EntryInfo);
13207 auto It = OffloadEntriesTargetRegion.find(EntryInfo);
13208 if (It == OffloadEntriesTargetRegion.end()) {
13212 if (!IgnoreAddressId && (It->second.getAddress() || It->second.getID()))
13220 for (
const auto &It : OffloadEntriesTargetRegion) {
13221 Action(It.first, It.second);
13227 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags);
13228 ++OffloadingEntriesNum;
13234 if (OMPBuilder->Config.isTargetDevice()) {
13238 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
13240 if (Entry.getVarSize() == 0) {
13241 Entry.setVarSize(VarSize);
13242 Entry.setLinkage(Linkage);
13246 Entry.setVarSize(VarSize);
13247 Entry.setLinkage(Linkage);
13248 Entry.setAddress(Addr);
13251 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
13252 assert(Entry.isValid() && Entry.getFlags() == Flags &&
13253 "Entry not initialized!");
13254 if (Entry.getVarSize() == 0) {
13255 Entry.setVarSize(VarSize);
13256 Entry.setLinkage(Linkage);
13263 OffloadEntriesDeviceGlobalVar.try_emplace(VarName, OffloadingEntriesNum,
13264 Addr, VarSize, Flags, Linkage,
13267 OffloadEntriesDeviceGlobalVar.try_emplace(
13268 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage,
"");
13269 ++OffloadingEntriesNum;
13276 for (
const auto &E : OffloadEntriesDeviceGlobalVar)
13277 Action(E.getKey(), E.getValue());
13284void CanonicalLoopInfo::collectControlBlocks(
13291 BBs.
append({getPreheader(), Header,
Cond, Latch, Exit, getAfter()});
13303void CanonicalLoopInfo::setTripCount(
Value *TripCount) {
13315void CanonicalLoopInfo::mapIndVar(
13325 for (
Use &U : OldIV->
uses()) {
13329 if (
User->getParent() == getCond())
13331 if (
User->getParent() == getLatch())
13337 Value *NewIV = Updater(OldIV);
13340 for (Use *U : ReplacableUses)
13361 "Preheader must terminate with unconditional branch");
13363 "Preheader must jump to header");
13367 "Header must terminate with unconditional branch");
13368 assert(Header->getSingleSuccessor() == Cond &&
13369 "Header must jump to exiting block");
13372 assert(Cond->getSinglePredecessor() == Header &&
13373 "Exiting block only reachable from header");
13376 "Exiting block must terminate with conditional branch");
13378 "Exiting block's first successor jump to the body");
13380 "Exiting block's second successor must exit the loop");
13384 "Body only reachable from exiting block");
13389 "Latch must terminate with unconditional branch");
13390 assert(Latch->getSingleSuccessor() == Header &&
"Latch must jump to header");
13393 assert(Latch->getSinglePredecessor() !=
nullptr);
13398 "Exit block must terminate with unconditional branch");
13399 assert(Exit->getSingleSuccessor() == After &&
13400 "Exit block must jump to after block");
13404 "After block only reachable from exit block");
13408 assert(IndVar &&
"Canonical induction variable not found?");
13410 "Induction variable must be an integer");
13412 "Induction variable must be a PHI in the loop header");
13418 auto *NextIndVar =
cast<PHINode>(IndVar)->getIncomingValue(1);
13426 assert(TripCount &&
"Loop trip count not found?");
13428 "Trip count and induction variable must have the same type");
13432 "Exit condition must be a signed less-than comparison");
13434 "Exit condition must compare the induction variable");
13436 "Exit condition must compare with the trip count");
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Expand Atomic instructions
This file contains the simple types necessary to represent the attributes associated with functions a...
static const Function * getParent(const Value *V)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
This file provides various utilities for inspecting and working with the control flow graph in LLVM I...
This header defines various interfaces for pass management in LLVM.
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static bool isZero(Value *V, const DataLayout &DL, DominatorTree *DT, AssumptionCache *AC)
static cl::opt< unsigned > TileSize("fuse-matrix-tile-size", cl::init(4), cl::Hidden, cl::desc("Tile size for matrix instruction fusion using square-shaped tiles."))
uint64_t IntrinsicInst * II
#define OMP_KERNEL_ARG_VERSION
Provides definitions for Target specific Grid Values.
static void emitTargetCall(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, Value *RTLocOverride, OpenMPIRBuilder::InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, OpenMPIRBuilder::TargetDataInfo &Info, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, const OpenMPIRBuilder::TargetKernelRuntimeAttrs &RuntimeAttrs, Value *IfCond, Function *OutlinedFn, Constant *OutlinedFnID, SmallVectorImpl< Value * > &Args, OpenMPIRBuilder::GenMapInfoCallbackTy GenMapInfoCB, OpenMPIRBuilder::CustomMapperCallbackTy CustomMapperCB, const OpenMPIRBuilder::DependenciesInfo &Dependencies, bool HasNoWait, Value *DynCGroupMem, OMPDynGroupprivateFallbackType DynCGroupMemFallback)
static Value * removeASCastIfPresent(Value *V)
static void createTargetLoopWorkshareCall(OpenMPIRBuilder *OMPBuilder, WorksharingLoopType LoopType, BasicBlock *InsertBlock, Value *Ident, Value *LoopBodyArg, Value *TripCount, Function &LoopBodyFn, bool NoLoop)
Value * createFakeIntVal(IRBuilderBase &Builder, OpenMPIRBuilder::InsertPointTy OuterAllocaIP, llvm::SmallVectorImpl< Instruction * > &ToBeDeleted, OpenMPIRBuilder::InsertPointTy InnerAllocaIP, const Twine &Name="", bool AsPtr=true, bool Is64Bit=false)
static Function * createTargetParallelWrapper(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn)
Create wrapper function used to gather the outlined function's argument structure from a shared buffe...
static void redirectTo(BasicBlock *Source, BasicBlock *Target, DebugLoc DL)
Make Source branch to Target.
static FunctionCallee getKmpcDistForStaticInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
static void applyParallelAccessesMetadata(CanonicalLoopInfo *CLI, LLVMContext &Ctx, Loop *Loop, LoopInfo &LoopInfo, SmallVector< Metadata * > &LoopMDList)
static Expected< Function * > createOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, StringRef FuncName, SmallVectorImpl< Value * > &Inputs, OpenMPIRBuilder::TargetBodyGenCallbackTy &CBFunc, OpenMPIRBuilder::TargetGenArgAccessorsCallbackTy &ArgAccessorFuncCB, DebugLoc OutlinedFnLoc)
static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, char ISA, StringRef ParSeq, StringRef MangledName, bool OutputBecomesInput, llvm::Function *Fn)
static FunctionCallee getKmpcForDynamicFiniForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for finalizing the dynamic loop using depending on type.
static void FixupDebugInfoForOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, Function *Func, DenseMap< Value *, std::tuple< Value *, unsigned > > &ValueReplacementMap)
static OMPScheduleType getOpenMPOrderingScheduleType(OMPScheduleType BaseScheduleType, bool HasOrderedClause)
Adds ordering modifier flags to schedule type.
static OMPScheduleType getOpenMPMonotonicityScheduleType(OMPScheduleType ScheduleType, bool HasSimdModifier, bool HasMonotonic, bool HasNonmonotonic, bool HasOrderedClause)
Adds monotonicity modifier flags to schedule type.
static std::string mangleVectorParameters(ArrayRef< llvm::OpenMPIRBuilder::DeclareSimdAttrTy > ParamAttrs)
Mangle the parameter part of the vector function name according to their OpenMP classification.
static bool isGenericKernel(Function &Fn)
static void workshareLoopTargetCallback(OpenMPIRBuilder *OMPIRBuilder, CanonicalLoopInfo *CLI, Value *Ident, Function &OutlinedFn, const SmallVector< Instruction *, 4 > &ToBeDeleted, WorksharingLoopType LoopType, bool NoLoop)
static bool isValidWorkshareLoopScheduleType(OMPScheduleType SchedType)
static bool isAtomicableReductionSet(ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos)
static llvm::CallInst * emitNoUnwindRuntimeCall(IRBuilder<> &Builder, llvm::FunctionCallee Callee, ArrayRef< llvm::Value * > Args, const llvm::Twine &Name)
static Error populateReductionFunction(Function *ReductionFunc, ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos, IRBuilder<> &Builder, ArrayRef< bool > IsByRef, bool IsGPU)
static Function * getFreshReductionFunc(Module &M)
static void raiseUserConstantDataAllocasToEntryBlock(IRBuilderBase &Builder, Function *Function)
static FunctionCallee getKmpcForDynamicNextForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for updating the next loop using OpenMP dynamic scheduling depending...
static bool isConflictIP(IRBuilder<>::InsertPoint IP1, IRBuilder<>::InsertPoint IP2)
Return whether IP1 and IP2 are ambiguous, i.e.
static void checkReductionInfos(ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos, bool IsGPU)
static Type * getOffloadingArrayType(Value *V)
static OMPScheduleType getOpenMPBaseScheduleType(llvm::omp::ScheduleKind ClauseKind, bool HasChunks, bool HasSimdModifier, bool HasDistScheduleChunks)
Determine which scheduling algorithm to use, determined from schedule clause arguments.
static OMPScheduleType computeOpenMPScheduleType(ScheduleKind ClauseKind, bool HasChunks, bool HasSimdModifier, bool HasMonotonicModifier, bool HasNonmonotonicModifier, bool HasOrderedClause, bool HasDistScheduleChunks)
Determine the schedule type using schedule and ordering clause arguments.
static FunctionCallee getKmpcForDynamicInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for initializing loop bounds using OpenMP dynamic scheduling dependi...
static std::optional< omp::OMPTgtExecModeFlags > getTargetKernelExecMode(Function &Kernel)
Given a function, if it represents the entry point of a target kernel, this returns the execution mod...
static StructType * createTaskWithPrivatesTy(OpenMPIRBuilder &OMPIRBuilder, ArrayRef< Value * > OffloadingArraysToPrivatize)
static cl::opt< double > UnrollThresholdFactor("openmp-ir-builder-unroll-threshold-factor", cl::Hidden, cl::desc("Factor for the unroll threshold to account for code " "simplifications still taking place"), cl::init(1.5))
static cl::opt< bool > UseDefaultMaxThreads("openmp-ir-builder-use-default-max-threads", cl::Hidden, cl::desc("Use a default max threads if none is provided."), cl::init(true))
static int32_t computeHeuristicUnrollFactor(CanonicalLoopInfo *CLI)
Heuristically determine the best-performant unroll factor for CLI.
static Error emitTargetOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, bool IsOffloadEntry, TargetRegionEntryInfo &EntryInfo, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, Function *&OutlinedFn, Constant *&OutlinedFnID, SmallVectorImpl< Value * > &Inputs, OpenMPIRBuilder::TargetBodyGenCallbackTy &CBFunc, OpenMPIRBuilder::TargetGenArgAccessorsCallbackTy &ArgAccessorFuncCB, DebugLoc OutlinedFnLoc)
static Value * emitTaskDependencies(OpenMPIRBuilder &OMPBuilder, const SmallVectorImpl< OpenMPIRBuilder::DependData > &Dependencies)
static void updateNVPTXAttr(Function &Kernel, StringRef Name, int32_t Value, bool Min)
static OpenMPIRBuilder::InsertPointTy getInsertPointAfterInstr(Instruction *I)
static void redirectAllPredecessorsTo(BasicBlock *OldTarget, BasicBlock *NewTarget, DebugLoc DL)
Redirect all edges that branch to OldTarget to NewTarget.
static void hoistNonEntryAllocasToEntryBlock(llvm::BasicBlock &Block)
static std::unique_ptr< TargetMachine > createTargetMachine(Function *F, CodeGenOptLevel OptLevel)
Create the TargetMachine object to query the backend for optimization preferences.
static FunctionCallee getKmpcForStaticInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
static void addAccessGroupMetadata(BasicBlock *Block, MDNode *AccessGroup, LoopInfo &LI)
Attach llvm.access.group metadata to the memref instructions of Block.
static void addBasicBlockMetadata(BasicBlock *BB, ArrayRef< Metadata * > Properties)
Attach metadata Properties to the basic block described by BB.
static void restoreIPandDebugLoc(llvm::IRBuilderBase &Builder, llvm::IRBuilderBase::InsertPoint IP)
This is a wrapper over IRBuilderBase::restoreIP that also restores a current debug location when the ...
static LoadInst * loadSharedDataFromTaskDescriptor(OpenMPIRBuilder &OMPIRBuilder, IRBuilderBase &Builder, Value *TaskWithPrivates, Type *TaskWithPrivatesTy)
Given a task descriptor, TaskWithPrivates, return the pointer to the block of pointers containing sha...
static cl::opt< bool > OptimisticAttributes("openmp-ir-builder-optimistic-attributes", cl::Hidden, cl::desc("Use optimistic attributes describing " "'as-if' properties of runtime calls."), cl::init(false))
static bool hasGridValue(const Triple &T)
static FunctionCallee getKmpcForStaticLoopForType(Type *Ty, OpenMPIRBuilder *OMPBuilder, WorksharingLoopType LoopType)
static const omp::GV & getGridValue(const Triple &T, Function *Kernel)
static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, StringRef Prefix, char ISA, StringRef ParSeq, StringRef MangledName, bool OutputBecomesInput, llvm::Function *Fn)
static Function * emitTargetTaskProxyFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, CallInst *StaleCI, StructType *PrivatesTy, StructType *TaskWithPrivatesTy, const size_t NumOffloadingArrays, const int SharedArgsOperandNo)
Create an entry point for a target task with the following.
static void addLoopMetadata(CanonicalLoopInfo *Loop, ArrayRef< Metadata * > Properties)
Attach loop metadata Properties to the loop described by Loop.
static AtomicOrdering TransformReleaseAcquireRelease(AtomicOrdering AO)
static void removeUnusedBlocksFromParent(ArrayRef< BasicBlock * > BBs)
static void targetParallelCallback(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn, Function *OuterFn, BasicBlock *OuterAllocaBB, Value *Ident, Value *IfCondition, Value *NumThreads, Instruction *PrivTID, AllocaInst *PrivTIDAddr, Value *ThreadID, const SmallVector< Instruction *, 4 > &ToBeDeleted)
static void hostParallelCallback(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn, Function *OuterFn, Value *Ident, Value *IfCondition, Instruction *PrivTID, AllocaInst *PrivTIDAddr, const SmallVector< Instruction *, 4 > &ToBeDeleted)
FunctionAnalysisManager FAM
This file defines the Pass Instrumentation classes that provide instrumentation points into the pass ...
const SmallVectorImpl< MachineOperand > & Cond
Remove Loads Into Fake Uses
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
SmallPtrSet< BasicBlock *, 0 > BlockSet
This file implements the SmallBitVector class.
This file defines the SmallSet class.
static SymbolRef::Type getType(const Symbol *Sym)
Defines the virtual file system interface vfs::FileSystem.
static cl::opt< unsigned > MaxThreads("xcore-max-threads", cl::desc("Maximum number of threads (for emulation thread-local storage)"), cl::Hidden, cl::value_desc("number"), cl::init(8))
static const uint32_t IV[8]
Class for arbitrary precision integers.
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
An arbitrary precision integer that knows its signedness.
static APSInt getUnsigned(uint64_t X)
This class represents a conversion between pointers from one address space to another.
an instruction to allocate memory on the stack
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
PointerType * getType() const
Overload to return most specific pointer type.
Type * getAllocatedType() const
Return the type that is being allocated by the instruction.
unsigned getAddressSpace() const
Return the address space for the allocation.
LLVM_ABI std::optional< TypeSize > getAllocationSize(const DataLayout &DL) const
Get allocation size in bytes.
LLVM_ABI bool isArrayAllocation() const
Return true if there is an allocation size parameter to the allocation instruction that is not 1.
void setAlignment(Align Align)
const Value * getArraySize() const
Get the number of elements allocated.
bool registerPass(PassBuilderT &&PassBuilder)
Register an analysis pass with the manager.
This class represents an incoming formal argument to a Function.
unsigned getArgNo() const
Return the index of this formal argument in its containing function.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
Class to represent array types.
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
A function analysis which provides an AssumptionCache.
LLVM_ABI AssumptionCache run(Function &F, FunctionAnalysisManager &)
A cache of @llvm.assume calls within a function.
An instruction that atomically checks whether a specified value is in a memory location,...
void setWeak(bool IsWeak)
static AtomicOrdering getStrongestFailureOrdering(AtomicOrdering SuccessOrdering)
Returns the strongest permitted ordering on failure, given the desired ordering on success.
LLVM_ABI std::pair< LoadInst *, AllocaInst * > EmitAtomicLoadLibcall(AtomicOrdering AO)
LLVM_ABI void EmitAtomicStoreLibcall(AtomicOrdering AO, Value *Source)
an instruction that atomically reads a memory location, combines it with another value,...
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ USubCond
Subtract only if no unsigned overflow.
@ FMinimum
*p = minimum(old, v) minimum matches the behavior of llvm.minimum.
@ Min
*p = old <signed v ? old : v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FMaximum
*p = maximum(old, v) maximum matches the behavior of llvm.maximum.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMaximumNum
*p = maximumnum(old, v) maximumnum matches the behavior of llvm.maximumnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ FMinimumNum
*p = minimumnum(old, v) minimumnum matches the behavior of llvm.minimumnum.
This class holds the attributes for a particular argument, parameter, function, or return value.
LLVM_ABI AttributeSet addAttributes(LLVMContext &C, AttributeSet AS) const
Add attributes to the attribute set.
LLVM_ABI AttributeSet addAttribute(LLVMContext &C, Attribute::AttrKind Kind) const
Add an argument attribute.
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
LLVM_ABI void replaceSuccessorsPhiUsesWith(BasicBlock *Old, BasicBlock *New)
Update all phi nodes in this basic block's successors to refer to basic block New instead of basic bl...
iterator begin()
Instruction iterator methods.
LLVM_ABI const_iterator getFirstInsertionPt() const
Returns an iterator to the first instruction in this block that is suitable for inserting a non-PHI i...
LLVM_ABI BasicBlock * splitBasicBlock(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction.
const Function * getParent() const
Return the enclosing method, or null if none.
reverse_iterator rbegin()
bool hasTerminator() const LLVM_READONLY
Returns whether the block has a terminator.
const Instruction & back() const
LLVM_ABI BasicBlock * splitBasicBlockBefore(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction and insert the new basic blo...
LLVM_ABI InstListType::const_iterator getFirstNonPHIIt() const
Returns an iterator to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI void insertDbgRecordBefore(DbgRecord *DR, InstListType::iterator Here)
Insert a DbgRecord into a block at the position given by Here.
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
LLVM_ABI InstListType::const_iterator getFirstNonPHIOrDbg(bool SkipPseudoOp=true) const
Returns a pointer to the first instruction in this block that is not a PHINode or a debug intrinsic,...
LLVM_ABI const BasicBlock * getUniqueSuccessor() const
Return the successor of this block if it has a unique successor.
LLVM_ABI const BasicBlock * getSinglePredecessor() const
Return the predecessor of this block if it has a single predecessor block.
const Instruction & front() const
InstListType::reverse_iterator reverse_iterator
LLVM_ABI const BasicBlock * getUniquePredecessor() const
Return the predecessor of this block if it has a unique predecessor block.
const Instruction * getTerminatorOrNull() const LLVM_READONLY
Returns the terminator instruction if the block is well formed or null if the block is not well forme...
LLVM_ABI const BasicBlock * getSingleSuccessor() const
Return the successor of this block if it has a single successor.
LLVM_ABI SymbolTableList< BasicBlock >::iterator eraseFromParent()
Unlink 'this' from the containing function and delete it.
InstListType::iterator iterator
Instruction iterators...
LLVM_ABI LLVMContext & getContext() const
Get the context in which this basic block lives.
void moveBefore(BasicBlock *MovePos)
Unlink this basic block from its current function and insert it into the function that MovePos lives ...
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
void splice(BasicBlock::iterator ToIt, BasicBlock *FromBB)
Transfer all instructions from FromBB to this basic block at ToIt.
LLVM_ABI void removePredecessor(BasicBlock *Pred, bool KeepOneInputPHIs=false)
Update PHI nodes in this BasicBlock before removal of predecessor Pred.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
Value * getArgOperand(unsigned i) const
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Class to represented the control flow structure of an OpenMP canonical loop.
Value * getTripCount() const
Returns the llvm::Value containing the number of loop iterations.
BasicBlock * getHeader() const
The header is the entry for each iteration.
LLVM_ABI void assertOK() const
Consistency self-check.
Type * getIndVarType() const
Return the type of the induction variable (and the trip count).
BasicBlock * getBody() const
The body block is the single entry for a loop iteration and not controlled by CanonicalLoopInfo.
bool isValid() const
Returns whether this object currently represents the IR of a loop.
void setLastIter(Value *IterVar)
Sets the last iteration variable for this loop.
OpenMPIRBuilder::InsertPointTy getAfterIP() const
Return the insertion point for user code after the loop.
OpenMPIRBuilder::InsertPointTy getBodyIP() const
Return the insertion point for user code in the body.
BasicBlock * getAfter() const
The after block is intended for clean-up code such as lifetime end markers.
Function * getFunction() const
LLVM_ABI void invalidate()
Invalidate this loop.
BasicBlock * getLatch() const
Reaching the latch indicates the end of the loop body code.
OpenMPIRBuilder::InsertPointTy getPreheaderIP() const
Return the insertion point for user code before the loop.
BasicBlock * getCond() const
The condition block computes whether there is another loop iteration.
BasicBlock * getExit() const
Reaching the exit indicates no more iterations are being executed.
LLVM_ABI BasicBlock * getPreheader() const
The preheader ensures that there is only a single edge entering the loop.
Instruction * getIndVar() const
Returns the instruction representing the current logical induction variable.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
static LLVM_ABI Constant * get(ArrayType *T, ArrayRef< Constant * > V)
static Constant * get(LLVMContext &Context, ArrayRef< ElementTy > Elts)
get() constructor - Return a constant with array type with an element count and element type matching...
static LLVM_ABI Constant * getString(LLVMContext &Context, StringRef Initializer, bool AddNull=true, bool ByteString=false)
This method constructs a CDS and initializes it with a text string.
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getPointerBitCastOrAddrSpaceCast(Constant *C, Type *Ty)
Create a BitCast or AddrSpaceCast for a pointer type depending on the address space.
static LLVM_ABI Constant * getAddrSpaceCast(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
This is an important base class in LLVM.
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
DILocalScope * getScope() const
Get the local scope for this variable.
DINodeArray getAnnotations() const
Subprogram description. Uses SubclassData1.
uint32_t getAlignInBits() const
StringRef getName() const
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeStoreSize(Type *Ty) const
Returns the maximum number of bytes that may be overwritten by storing the specified type.
Record of a variable value-assignment, aka a non instruction representation of the dbg....
Analysis pass which computes a DominatorTree.
LLVM_ABI DominatorTree run(Function &F, FunctionAnalysisManager &)
Run the analysis pass over a function and produce a dominator tree.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Represents either an error or a value T.
Lightweight error class with error context and mandatory checking.
static ErrorSuccess success()
Create a success value.
Tagged union holding either a T or a Error.
Error takeError()
Take ownership of the stored error.
reference get()
Returns a reference to the stored T value.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
Class to represent function types.
Type * getParamType(unsigned i) const
Parameter type accessors.
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
void addFnAttr(Attribute::AttrKind Kind)
Add function attributes to this function.
static Function * Create(FunctionType *Ty, LinkageTypes Linkage, unsigned AddrSpace, const Twine &N="", Module *M=nullptr)
const BasicBlock & getEntryBlock() const
FunctionType * getFunctionType() const
Returns the FunctionType for me.
void removeFromParent()
removeFromParent - This method unlinks 'this' from the containing module, but does not delete it.
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
DISubprogram * getSubprogram() const
Get the attached subprogram.
AttributeList getAttributes() const
Return the attribute list for this Function.
const Function & getFunction() const
void setAttributes(AttributeList Attrs)
Set the attribute list for this Function.
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
void addParamAttr(unsigned ArgNo, Attribute::AttrKind Kind)
adds the attribute to the list of attributes for the given arg.
Function::iterator insert(Function::iterator Position, BasicBlock *BB)
Insert BB in the basic block list at Position.
Type * getReturnType() const
Returns the type of the ret val.
void setCallingConv(CallingConv::ID CC)
Argument * getArg(unsigned i) const
bool hasMetadata() const
Return true if this GlobalObject has any metadata attached to it.
LLVM_ABI void addMetadata(unsigned KindID, MDNode &MD)
Add a metadata attachment.
LinkageTypes getLinkage() const
void setLinkage(LinkageTypes LT)
Module * getParent()
Get the module that this global value is contained inside of...
void setDSOLocal(bool Local)
PointerType * getType() const
Global values are always pointers.
@ HiddenVisibility
The GV is hidden.
@ ProtectedVisibility
The GV is protected.
void setVisibility(VisibilityTypes V)
LinkageTypes
An enumeration for the kinds of linkage for global values.
@ PrivateLinkage
Like Internal, but omit from symbol table.
@ CommonLinkage
Tentative definitions.
@ InternalLinkage
Rename collisions when linking (static functions).
@ WeakODRLinkage
Same, but only replaced by something equivalent.
@ WeakAnyLinkage
Keep one copy of named function when linking (weak)
@ AppendingLinkage
Special purpose, only applies to global arrays.
@ LinkOnceODRLinkage
Same, but only replaced by something equivalent.
Type * getValueType() const
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
Common base class shared among various IRBuilders.
InsertPoint saveIP() const
Returns the current insert point.
BasicBlock::iterator InsertPoint
InsertPoint - A saved insertion point.
void restoreIP(InsertPoint IP)
Sets the current insert point to a previously-saved location.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
LLVM_ABI const DebugLoc & getStableDebugLoc() const
Fetch the debug location for this node, unless this is a debug intrinsic, in which case fetch the deb...
LLVM_ABI void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
LLVM_ABI unsigned getNumSuccessors() const LLVM_READONLY
Return the number of successors that this instruction has.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
LLVM_ABI BasicBlock * getSuccessor(unsigned Idx) const LLVM_READONLY
Return the specified successor. This instruction must be a terminator.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
user_iterator user_begin()
LLVM_ABI void moveBeforePreserving(InstListType::iterator MovePos)
Perform a moveBefore operation, while signalling that the caller intends to preserve the original ord...
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void insertAfter(Instruction *InsertPos)
Insert an unlinked instruction into a basic block immediately after the specified instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this load instruction.
Align getAlign() const
Return the alignment of the access that is being performed.
Analysis pass that exposes the LoopInfo for a function.
LLVM_ABI LoopInfo run(Function &F, FunctionAnalysisManager &AM)
ArrayRef< BlockT * > getBlocks() const
Get a list of the basic blocks which make up this loop.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
This class represents a loop nest and can be used to query its properties.
Represents a single loop in the control flow graph.
LLVM_ABI MDNode * createCallbackEncoding(unsigned CalleeArgNo, ArrayRef< int > Arguments, bool VarArgsArePassed)
Return metadata describing a callback (see llvm::AbstractCallSite).
LLVM_ABI void replaceOperandWith(unsigned I, Metadata *New)
Replace a specific operand.
static MDTuple * getDistinct(LLVMContext &Context, ArrayRef< Metadata * > MDs)
ArrayRef< MDOperand > operands() const
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
This class implements a map that also provides access to all stored values in a deterministic order.
A Module instance is used to store all the information related to an LLVM module.
LLVMContext & getContext() const
Get the global data context.
const DataLayout & getDataLayout() const
Get the data layout for the module's target platform.
iterator_range< op_iterator > operands()
LLVM_ABI void addOperand(MDNode *M)
Device global variable entries info.
Target region entries info.
Base class of the entries info.
Class that manages information about offload code regions and data.
function_ref< void(StringRef, const OffloadEntryInfoDeviceGlobalVar &)> OffloadDeviceGlobalVarEntryInfoActTy
Applies action Action on all registered entries.
OMPTargetDeviceClauseKind
Kind of device clause for declare target variables and functions NOTE: Currently not used as a part o...
@ OMPTargetDeviceClauseAny
The target is marked for all devices.
LLVM_ABI void registerDeviceGlobalVarEntryInfo(StringRef VarName, Constant *Addr, int64_t VarSize, OMPTargetGlobalVarEntryKind Flags, GlobalValue::LinkageTypes Linkage)
Register device global variable entry.
LLVM_ABI void initializeDeviceGlobalVarEntryInfo(StringRef Name, OMPTargetGlobalVarEntryKind Flags, unsigned Order)
Initialize device global variable entry.
LLVM_ABI void actOnDeviceGlobalVarEntriesInfo(const OffloadDeviceGlobalVarEntryInfoActTy &Action)
OMPTargetRegionEntryKind
Kind of the target registry entry.
@ OMPTargetRegionEntryTargetRegion
Mark the entry as target region.
LLVM_ABI void getTargetRegionEntryFnName(SmallVectorImpl< char > &Name, const TargetRegionEntryInfo &EntryInfo)
LLVM_ABI bool hasTargetRegionEntryInfo(TargetRegionEntryInfo EntryInfo, bool IgnoreAddressId=false) const
Return true if a target region entry with the provided information exists.
LLVM_ABI void registerTargetRegionEntryInfo(TargetRegionEntryInfo EntryInfo, Constant *Addr, Constant *ID, OMPTargetRegionEntryKind Flags)
Register target region entry.
LLVM_ABI void actOnTargetRegionEntriesInfo(const OffloadTargetRegionEntryInfoActTy &Action)
LLVM_ABI void initializeTargetRegionEntryInfo(const TargetRegionEntryInfo &EntryInfo, unsigned Order)
Initialize target region entry.
OMPTargetGlobalVarEntryKind
Kind of the global variable entry..
@ OMPTargetGlobalVarEntryEnter
Mark the entry as a declare target enter.
@ OMPTargetGlobalRegisterRequires
Mark the entry as a register requires global.
@ OMPTargetGlobalVarEntryIndirect
Mark the entry as a declare target indirect global.
@ OMPTargetGlobalVarEntryLink
Mark the entry as a to declare target link.
@ OMPTargetGlobalVarEntryTo
Mark the entry as a to declare target.
@ OMPTargetGlobalVarEntryIndirectVTable
Mark the entry as a declare target indirect vtable.
function_ref< void(const TargetRegionEntryInfo &EntryInfo, const OffloadEntryInfoTargetRegion &)> OffloadTargetRegionEntryInfoActTy
brief Applies action Action on all registered entries.
bool hasDeviceGlobalVarEntryInfo(StringRef VarName) const
Checks if the variable with the given name has been registered already.
LLVM_ABI bool empty() const
Return true if a there are no entries defined.
std::optional< bool > IsTargetDevice
Flag to define whether to generate code for the role of the OpenMP host (if set to false) or device (...
std::optional< bool > IsGPU
Flag for specifying if the compilation is done for an accelerator.
LLVM_ABI int64_t getRequiresFlags() const
Returns requires directive clauses as flags compatible with those expected by libomptarget.
std::optional< bool > OpenMPOffloadMandatory
Flag for specifying if offloading is mandatory.
LLVM_ABI void setHasRequiresReverseOffload(bool Value)
LLVM_ABI OpenMPIRBuilderConfig()
LLVM_ABI bool hasRequiresUnifiedSharedMemory() const
LLVM_ABI void setHasRequiresUnifiedSharedMemory(bool Value)
unsigned getDefaultTargetAS() const
LLVM_ABI bool hasRequiresDynamicAllocators() const
LLVM_ABI void setHasRequiresUnifiedAddress(bool Value)
bool isTargetDevice() const
LLVM_ABI void setHasRequiresDynamicAllocators(bool Value)
LLVM_ABI bool hasRequiresReverseOffload() const
bool hasRequiresFlags() const
LLVM_ABI bool hasRequiresUnifiedAddress() const
Struct that keeps the information that should be kept throughout a 'target data' region.
An interface to create LLVM-IR for OpenMP directives.
LLVM_ABI InsertPointOrErrorTy createOrderedThreadsSimd(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsThreads)
Generator for 'omp ordered [threads | simd]'.
LLVM_ABI void emitAArch64DeclareSimdFunction(llvm::Function *Fn, unsigned VLENVal, llvm::ArrayRef< DeclareSimdAttrTy > ParamAttrs, DeclareSimdBranch Branch, char ISA, unsigned NarrowestDataSize, bool OutputBecomesInput)
Emit AArch64 vector-function ABI attributes for a declare simd function.
LLVM_ABI Constant * getOrCreateIdent(Constant *SrcLocStr, uint32_t SrcLocStrSize, omp::IdentFlag Flags=omp::IdentFlag(0), unsigned Reserve2Flags=0)
Return an ident_t* encoding the source location SrcLocStr and Flags.
LLVM_ABI void registerDeclareTargetGlobalReplacement(GlobalValue *Original, GlobalValue *Replacement)
Register a module-scope replacement of a declare target global variable.
LLVM_ABI FunctionCallee getOrCreateRuntimeFunction(Module &M, omp::RuntimeFunction FnID)
Return the function declaration for the runtime function with FnID.
LLVM_ABI InsertPointOrErrorTy createCancel(const LocationDescription &Loc, Value *IfCondition, omp::Directive CanceledDirective)
Generator for 'omp cancel'.
std::function< Expected< Function * >(StringRef FunctionName)> FunctionGenCallback
Functions used to generate a function with the given name.
LLVM_ABI CallInst * createOMPAllocShared(const LocationDescription &Loc, Value *Size, const Twine &Name=Twine(""))
Create a runtime call for kmpc_alloc_shared.
ReductionGenCBKind
Enum class for the RedctionGen CallBack type to be used.
LLVM_ABI CanonicalLoopInfo * collapseLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops, InsertPointTy ComputeIP)
Collapse a loop nest into a single loop.
LLVM_ABI void createTaskyield(const LocationDescription &Loc)
Generator for 'omp taskyield'.
std::function< Error(InsertPointTy CodeGenIP)> FinalizeCallbackTy
Callback type for variable finalization (think destructors).
LLVM_ABI void emitBranch(BasicBlock *Target)
LLVM_ABI Error emitCancelationCheckImpl(Value *CancelFlag, omp::Directive CanceledDirective)
Generate control flow and cleanup for cancellation.
static LLVM_ABI void writeThreadBoundsForKernel(const Triple &T, Function &Kernel, int32_t LB, int32_t UB)
LLVM_ABI void emitTaskwaitImpl(const LocationDescription &Loc)
Generate a taskwait runtime call.
LLVM_ABI Constant * registerTargetRegionFunction(TargetRegionEntryInfo &EntryInfo, Function *OutlinedFunction, StringRef EntryFnName, StringRef EntryFnIDName)
Registers the given function and sets up the attribtues of the function Returns the FunctionID.
LLVM_ABI GlobalVariable * emitKernelExecutionMode(StringRef KernelName, omp::OMPTgtExecModeFlags Mode)
Emit the kernel execution mode.
LLVM_ABI void initialize()
Initialize the internal state, this will put structures types and potentially other helpers into the ...
LLVM_ABI InsertPointTy createAtomicCompare(const LocationDescription &Loc, AtomicOpValue &X, AtomicOpValue &V, AtomicOpValue &R, Value *E, Value *D, AtomicOrdering AO, omp::OMPAtomicCompareOp Op, bool IsXBinopExpr, bool IsPostfixUpdate, bool IsFailOnly, bool IsWeak=false)
LLVM_ABI InsertPointTy createAtomicWrite(const LocationDescription &Loc, AtomicOpValue &X, Value *Expr, AtomicOrdering AO, InsertPointTy AllocaIP)
Emit atomic write for : X = Expr — Only Scalar data types.
LLVM_ABI void loadOffloadInfoMetadata(Module &M)
Loads all the offload entries information from the host IR metadata.
function_ref< MapInfosTy &(InsertPointTy CodeGenIP)> GenMapInfoCallbackTy
Callback type for creating the map infos for the kernel parameters.
LLVM_ABI Error emitOffloadingArrays(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, MapInfosTy &CombinedInfo, TargetDataInfo &Info, CustomMapperCallbackTy CustomMapperCB, bool IsNonContiguous=false, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr)
Emit the arrays used to pass the captures and map information to the offloading runtime library.
LLVM_ABI void unrollLoopFull(DebugLoc DL, CanonicalLoopInfo *Loop)
Fully unroll a loop.
function_ref< Error(InsertPointTy CodeGenIP, Value *IndVar)> LoopBodyGenCallbackTy
Callback type for loop body code generation.
LLVM_ABI InsertPointOrErrorTy emitScanReduction(const LocationDescription &Loc, ArrayRef< llvm::OpenMPIRBuilder::ReductionInfo > ReductionInfos, ScanInfo *ScanRedInfo)
This function performs the scan reduction of the values updated in the input phase.
LLVM_ABI void emitFlush(const LocationDescription &Loc)
Generate a flush runtime call.
LLVM_ABI InsertPointOrErrorTy createScope(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsNowait)
Generator for 'omp scope'.
static LLVM_ABI std::pair< int32_t, int32_t > readThreadBoundsForKernel(const Triple &T, Function &Kernel)
}
OpenMPIRBuilderConfig Config
The OpenMPIRBuilder Configuration.
LLVM_ABI CallInst * createOMPInteropDestroy(const LocationDescription &Loc, Value *InteropVar, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_destroy.
LLVM_ABI void emitUsed(StringRef Name, ArrayRef< llvm::WeakTrackingVH > List)
Emit the llvm.used metadata.
LLVM_ABI InsertPointOrErrorTy createSingle(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsNowait, ArrayRef< llvm::Value * > CPVars={}, ArrayRef< llvm::Function * > CPFuncs={})
Generator for 'omp single'.
LLVM_ABI InsertPointOrErrorTy createTarget(const LocationDescription &Loc, bool IsOffloadEntry, OpenMPIRBuilder::InsertPointTy AllocaIP, OpenMPIRBuilder::InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks, TargetDataInfo &Info, TargetRegionEntryInfo &EntryInfo, const TargetKernelDefaultAttrs &DefaultAttrs, const TargetKernelRuntimeAttrs &RuntimeAttrs, Value *IfCond, SmallVectorImpl< Value * > &Inputs, GenMapInfoCallbackTy GenMapInfoCB, TargetBodyGenCallbackTy BodyGenCB, TargetGenArgAccessorsCallbackTy ArgAccessorFuncCB, CustomMapperCallbackTy CustomMapperCB, const DependenciesInfo &Dependencies={}, bool HasNowait=false, Value *DynCGroupMem=nullptr, omp::OMPDynGroupprivateFallbackType DynCGroupMemFallback=omp::OMPDynGroupprivateFallbackType::Abort, DebugLoc OutlinedFnLoc={}, Value *RTLocOverride=nullptr)
Generator for 'omp target'.
LLVM_ABI InsertPointOrErrorTy createTeams(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, Value *NumTeamsLower=nullptr, Value *NumTeamsUpper=nullptr, Value *ThreadLimit=nullptr, Value *IfExpr=nullptr)
Generator for #omp teams
std::forward_list< CanonicalLoopInfo > LoopInfos
Collection of owned canonical loop objects that eventually need to be free'd.
LLVM_ABI llvm::StructType * getKmpTaskAffinityInfoTy()
Return the LLVM struct type matching runtime kmp_task_affinity_info_t.
LLVM_ABI Constant * emitKernelEnvironment(const LocationDescription &Loc, const llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs &Attrs)
The omp target interface.
LLVM_ABI std::string createPlatformSpecificName(ArrayRef< StringRef > Parts) const
Get the create a name using the platform specific separators.
LLVM_ABI FunctionCallee createDispatchNextFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_next_* runtime function for the specified size IVSize and sign IVSigned.
static LLVM_ABI void getKernelArgsVector(TargetKernelArgs &KernelArgs, IRBuilderBase &Builder, SmallVector< Value * > &ArgsVector)
Create the kernel args vector used by emitTargetKernel.
LLVM_ABI void unrollLoopHeuristic(DebugLoc DL, CanonicalLoopInfo *Loop)
Fully or partially unroll a loop.
LLVM_ABI omp::OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position)
Get OMP_MAP_MEMBER_OF flag with extra bits reserved based on the position given.
LLVM_ABI void addAttributes(omp::RuntimeFunction FnID, Function &Fn)
Add attributes known for FnID to Fn.
Module & M
The underlying LLVM-IR module.
StringMap< Constant * > SrcLocStrMap
Map to remember source location strings.
LLVM_ABI void createMapperAllocas(const LocationDescription &Loc, InsertPointTy AllocaIP, unsigned NumOperands, struct MapperAllocas &MapperAllocas)
Create the allocas instruction used in call to mapper functions.
SmallVector< DeclareTargetGlobalReplacement, 8 > DeclareTargetGlobalReplacements
Collection of declare target globals to rewrite uses of during device module finalizaiton.
LLVM_ABI Constant * getOrCreateSrcLocStr(StringRef LocStr, uint32_t &SrcLocStrSize)
Return the (LLVM-IR) string describing the source location LocStr.
LLVM_ABI Error emitTargetRegionFunction(TargetRegionEntryInfo &EntryInfo, FunctionGenCallback &GenerateFunctionCallback, bool IsOffloadEntry, Function *&OutlinedFn, Constant *&OutlinedFnID)
Create a unique name for the entry function using the source location information of the current targ...
LLVM_ABI InsertPointOrErrorTy createIteratorLoop(LocationDescription Loc, llvm::Value *TripCount, IteratorBodyGenTy BodyGen, llvm::StringRef Name="iterator")
Create a canonical iterator loop at the current insertion point.
LLVM_ABI Expected< SmallVector< llvm::CanonicalLoopInfo * > > createCanonicalScanLoops(const LocationDescription &Loc, LoopBodyGenCallbackTy BodyGenCB, Value *Start, Value *Stop, Value *Step, bool IsSigned, bool InclusiveStop, InsertPointTy ComputeIP, const Twine &Name, ScanInfo *ScanRedInfo)
Generator for the control flow structure of an OpenMP canonical loops if the parent directive has an ...
LLVM_ABI FunctionCallee createDispatchFiniFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_fini_* runtime function for the specified size IVSize and sign IVSigned.
function_ref< InsertPointOrErrorTy( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks)> TargetBodyGenCallbackTy
LLVM_ABI void unrollLoopPartial(DebugLoc DL, CanonicalLoopInfo *Loop, int32_t Factor, CanonicalLoopInfo **UnrolledCLI)
Partially unroll a loop.
function_ref< Error(Value *DeviceID, Value *RTLoc, IRBuilderBase::InsertPoint TargetTaskAllocaIP)> TargetTaskBodyCallbackTy
Callback type for generating the bodies of device directives that require outer target tasks (e....
Expected< MapInfosTy & > MapInfosOrErrorTy
bool HandleFPNegZero
Emit atomic compare for constructs: — Only scalar data types cond-expr-stmt: x = x ordop expr ?
LLVM_ABI void emitTaskyieldImpl(const LocationDescription &Loc)
Generate a taskyield runtime call.
LLVM_ABI void emitMapperCall(const LocationDescription &Loc, Function *MapperFunc, Value *SrcLocInfo, Value *MaptypesArg, Value *MapnamesArg, struct MapperAllocas &MapperAllocas, int64_t DeviceID, unsigned NumOperands)
Create the call for the target mapper function.
LLVM_ABI InsertPointOrErrorTy createDistribute(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB)
Generator for #omp distribute
function_ref< Expected< Function * >(unsigned int)> CustomMapperCallbackTy
LLVM_ABI InsertPointTy createOrderedDepend(const LocationDescription &Loc, InsertPointTy AllocaIP, unsigned NumLoops, ArrayRef< llvm::Value * > StoreValues, const Twine &Name, bool IsDependSource)
Generator for 'omp ordered depend (source | sink)'.
LLVM_ABI InsertPointTy createCopyinClauseBlocks(InsertPointTy IP, Value *MasterAddr, Value *PrivateAddr, llvm::IntegerType *IntPtrTy, bool BranchtoEnd=true)
Generate conditional branch and relevant BasicBlocks through which private threads copy the 'copyin' ...
function_ref< InsertPointOrErrorTy( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, Value &Original, Value &Inner, Value *&ReplVal)> PrivatizeCallbackTy
Callback type for variable privatization (think copy & default constructor).
LLVM_ABI bool isFinalized()
Check whether the finalize function has already run.
SmallVector< FinalizationInfo, 8 > FinalizationStack
The finalization stack made up of finalize callbacks currently in-flight, wrapped into FinalizationIn...
LLVM_ABI std::vector< CanonicalLoopInfo * > tileLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops, ArrayRef< Value * > TileSizes)
Tile a loop nest.
LLVM_ABI CallInst * createOMPInteropInit(const LocationDescription &Loc, Value *InteropVar, omp::OMPInteropType InteropType, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_init.
LLVM_ABI Error emitIfClause(Value *Cond, BodyGenCallbackTy ThenGen, BodyGenCallbackTy ElseGen, InsertPointTy AllocaIP={}, ArrayRef< BasicBlock * > DeallocBlocks={})
Emits code for OpenMP 'if' clause using specified BodyGenCallbackTy Here is the logic: if (Cond) { Th...
LLVM_ABI void finalize(Function *Fn=nullptr)
Finalize the underlying module, e.g., by outlining regions.
LLVM_ABI Function * getOrCreateRuntimeFunctionPtr(omp::RuntimeFunction FnID)
void addOutlineInfo(std::unique_ptr< OutlineInfo > &&OI)
Add a new region that will be outlined later.
LLVM_ABI InsertPointTy createTargetInit(const LocationDescription &Loc, const llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs &Attrs)
Create a runtime call for kmpc_target_init.
LLVM_ABI InsertPointOrErrorTy createReductions(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< ReductionInfo > ReductionInfos, ArrayRef< bool > IsByRef, bool IsNoWait=false, bool IsTeamsReduction=false)
Generator for 'omp reduction'.
const Triple T
The target triple of the underlying module.
DenseMap< std::pair< Constant *, uint64_t >, Constant * > IdentMap
Map to remember existing ident_t*.
LLVM_ABI CallInst * createOMPFree(const LocationDescription &Loc, Value *Addr, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_free.
LLVM_ABI InsertPointOrErrorTy createReductionsGPU(const LocationDescription &Loc, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< ReductionInfo > ReductionInfos, ArrayRef< bool > IsByRef, bool IsNoWait=false, bool IsTeamsReduction=false, bool IsSPMD=false, ReductionGenCBKind ReductionGenCBKind=ReductionGenCBKind::MLIR, std::optional< omp::GV > GridValue={}, Value *SrcLocInfo=nullptr)
Design of OpenMP reductions on the GPU.
LLVM_ABI FunctionCallee createForStaticInitFunction(unsigned IVSize, bool IVSigned, bool IsGPUDistribute)
Returns __kmpc_for_static_init_* runtime function for the specified size IVSize and sign IVSigned.
LLVM_ABI CallInst * createOMPAlloc(const LocationDescription &Loc, Value *Size, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_alloc.
LLVM_ABI void emitNonContiguousDescriptor(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, MapInfosTy &CombinedInfo, TargetDataInfo &Info)
Emit an array of struct descriptors to be assigned to the offload args.
LLVM_ABI InsertPointOrErrorTy createSection(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB)
Generator for 'omp section'.
LLVM_ABI InsertPointOrErrorTy createTaskgroup(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB)
Generator for the taskgroup construct.
LLVM_ABI InsertPointOrErrorTy createParallel(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB, PrivatizeCallbackTy PrivCB, FinalizeCallbackTy FiniCB, Value *IfCondition, Value *NumThreads, omp::ProcBindKind ProcBind, bool IsCancellable)
Generator for 'omp parallel'.
function_ref< InsertPointOrErrorTy(InsertPointTy)> EmitFallbackCallbackTy
Callback function type for functions emitting the host fallback code that is executed when the kernel...
static LLVM_ABI TargetRegionEntryInfo getTargetEntryUniqueInfo(FileIdentifierInfoCallbackTy CallBack, vfs::FileSystem &VFS, StringRef ParentName="")
Creates a unique info for a target entry when provided a filename and line number from.
LLVM_ABI void emitTaskDependency(IRBuilderBase &Builder, Value *Entry, const DependData &Dep)
Store one kmp_depend_info entry at the given Entry pointer.
LLVM_ABI void emitBlock(BasicBlock *BB, Function *CurFn, bool IsFinished=false)
LLVM_ABI Value * getOrCreateThreadID(Value *Ident)
Return the current thread ID.
LLVM_ABI InsertPointOrErrorTy createMaster(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB)
Generator for 'omp master'.
LLVM_ABI InsertPointOrErrorTy createTargetData(const LocationDescription &Loc, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks, Value *DeviceID, Value *IfCond, TargetDataInfo &Info, GenMapInfoCallbackTy GenMapInfoCB, CustomMapperCallbackTy CustomMapperCB, omp::RuntimeFunction *MapperFunc=nullptr, function_ref< InsertPointOrErrorTy(InsertPointTy CodeGenIP, BodyGenTy BodyGenType)> BodyGenCB=nullptr, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr, Value *SrcLocInfo=nullptr)
Generator for 'omp target data'.
LLVM_ABI CallInst * createRuntimeFunctionCall(FunctionCallee Callee, ArrayRef< Value * > Args, StringRef Name="")
LLVM_ABI InsertPointOrErrorTy emitKernelLaunch(const LocationDescription &Loc, Value *OutlinedFnID, EmitFallbackCallbackTy EmitTargetCallFallbackCB, TargetKernelArgs &Args, Value *DeviceID, Value *RTLoc, InsertPointTy AllocaIP)
Generate a target region entry call and host fallback call.
StringMap< GlobalVariable *, BumpPtrAllocator > InternalVars
An ordered map of auto-generated variables to their unique names.
LLVM_ABI InsertPointOrErrorTy createCancellationPoint(const LocationDescription &Loc, omp::Directive CanceledDirective)
Generator for 'omp cancellation point'.
LLVM_ABI CallInst * createOMPAlignedAlloc(const LocationDescription &Loc, Value *Align, Value *Size, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_align_alloc.
LLVM_ABI FunctionCallee createDispatchInitFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_init_* runtime function for the specified size IVSize and sign IVSigned.
LLVM_ABI InsertPointOrErrorTy createScan(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< llvm::Value * > ScanVars, ArrayRef< llvm::Type * > ScanVarsType, bool IsInclusive, ScanInfo *ScanRedInfo)
This directive split and directs the control flow to input phase blocks or scan phase blocks based on...
LLVM_ABI CallInst * createOMPFreeShared(const LocationDescription &Loc, Value *Addr, Value *Size, const Twine &Name=Twine(""))
Create a runtime call for kmpc_free_shared.
LLVM_ABI CallInst * createOMPInteropUse(const LocationDescription &Loc, Value *InteropVar, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_use.
IRBuilder<>::InsertPoint InsertPointTy
Type used throughout for insertion points.
LLVM_ABI GlobalVariable * getOrCreateInternalVariable(Type *Ty, const StringRef &Name, std::optional< unsigned > AddressSpace={})
Gets (if variable with the given name already exist) or creates internal global variable with the spe...
LLVM_ABI GlobalVariable * createOffloadMapnames(SmallVectorImpl< llvm::Constant * > &Names, std::string VarName)
Create the global variable holding the offload names information.
LLVM_ABI InsertPointOrErrorTy createTask(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB, bool Tied=true, Value *Final=nullptr, Value *IfCondition=nullptr, const DependenciesInfo &Dependencies={}, const AffinityData &Affinities={}, bool Mergeable=false, Value *EventHandle=nullptr, Value *Priority=nullptr, bool FreeAgent=false)
Generator for #omp taskloop
std::forward_list< ScanInfo > ScanInfos
Collection of owned ScanInfo objects that eventually need to be free'd.
static LLVM_ABI void writeTeamsForKernel(const Triple &T, Function &Kernel, int32_t LB, int32_t UB)
LLVM_ABI Value * calculateCanonicalLoopTripCount(const LocationDescription &Loc, Value *Start, Value *Stop, Value *Step, bool IsSigned, bool InclusiveStop, const Twine &Name="loop")
Calculate the trip count of a canonical loop.
LLVM_ABI InsertPointOrErrorTy createBarrier(const LocationDescription &Loc, omp::Directive Kind, bool ForceSimpleCall=false, bool CheckCancelFlag=true)
Emitter methods for OpenMP directives.
LLVM_ABI void setCorrectMemberOfFlag(omp::OpenMPOffloadMappingFlags &Flags, omp::OpenMPOffloadMappingFlags MemberOfFlag)
Given an initial flag set, this function modifies it to contain the passed in MemberOfFlag generated ...
LLVM_ABI Error emitOffloadingArraysAndArgs(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, TargetDataInfo &Info, TargetDataRTArgs &RTArgs, MapInfosTy &CombinedInfo, CustomMapperCallbackTy CustomMapperCB, bool IsNonContiguous=false, bool ForEndCall=false, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr)
Allocates memory for and populates the arrays required for offloading (offload_{baseptrs|ptrs|mappers...
LLVM_ABI Constant * getOrCreateDefaultSrcLocStr(uint32_t &SrcLocStrSize)
Return the (LLVM-IR) string describing the default source location.
LLVM_ABI InsertPointOrErrorTy createCritical(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, StringRef CriticalName, Value *HintInst)
Generator for 'omp critical'.
LLVM_ABI void createError(const LocationDescription &Loc, bool IsFatal, Value *Message)
Generate a call to the runtime to emit the diagnostic of an OpenMP error directive with at(execution)...
LLVM_ABI void createOffloadEntry(Constant *ID, Constant *Addr, uint64_t Size, int32_t Flags, GlobalValue::LinkageTypes, StringRef Name="")
Creates offloading entry for the provided entry ID ID, address Addr, size Size, and flags Flags.
static LLVM_ABI unsigned getOpenMPDefaultSimdAlign(const Triple &TargetTriple, const StringMap< bool > &Features)
Get the default alignment value for given target.
LLVM_ABI unsigned getFlagMemberOffset()
Get the offset of the OMP_MAP_MEMBER_OF field.
LLVM_ABI InsertPointOrErrorTy applyWorkshareLoop(DebugLoc DL, CanonicalLoopInfo *CLI, InsertPointTy AllocaIP, bool NeedsBarrier, llvm::omp::ScheduleKind SchedKind=llvm::omp::OMP_SCHEDULE_Default, Value *ChunkSize=nullptr, bool HasSimdModifier=false, bool HasMonotonicModifier=false, bool HasNonmonotonicModifier=false, bool HasOrderedClause=false, omp::WorksharingLoopType LoopType=omp::WorksharingLoopType::ForStaticLoop, bool NoLoop=false, bool HasDistSchedule=false, Value *DistScheduleChunkSize=nullptr)
Modifies the canonical loop to be a workshare loop.
LLVM_ABI InsertPointOrErrorTy createAtomicCapture(const LocationDescription &Loc, InsertPointTy AllocaIP, AtomicOpValue &X, AtomicOpValue &V, Value *Expr, AtomicOrdering AO, AtomicRMWInst::BinOp RMWOp, AtomicUpdateCallbackTy &UpdateOp, bool UpdateExpr, bool IsPostfixUpdate, bool IsXBinopExpr, bool IsIgnoreDenormalMode=false, bool IsFineGrainedMemory=false, bool IsRemoteMemory=false)
Emit atomic update for constructs: — Only Scalar data types V = X; X = X BinOp Expr ,...
LLVM_ABI CanonicalLoopInfo * createLoopSkeleton(DebugLoc DL, Value *TripCount, Function *F, BasicBlock *PreInsertBefore, BasicBlock *PostInsertBefore, const Twine &Name={}, bool IsCollapsed=false)
Create the control flow structure of a canonical OpenMP loop.
LLVM_ABI void createOffloadEntriesAndInfoMetadata(EmitMetadataErrorReportFunctionTy &ErrorReportFunction)
LLVM_ABI void applySimd(CanonicalLoopInfo *Loop, MapVector< Value *, Value * > AlignedVars, Value *IfCond, omp::OrderKind Order, ConstantInt *Simdlen, ConstantInt *Safelen)
Add metadata to simd-ize a loop.
SmallVector< std::unique_ptr< OutlineInfo >, 16 > OutlineInfos
Collection of regions that need to be outlined during finalization.
LLVM_ABI InsertPointOrErrorTy createAtomicUpdate(const LocationDescription &Loc, InsertPointTy AllocaIP, AtomicOpValue &X, Value *Expr, AtomicOrdering AO, AtomicRMWInst::BinOp RMWOp, AtomicUpdateCallbackTy &UpdateOp, bool IsXBinopExpr, bool IsIgnoreDenormalMode=false, bool IsFineGrainedMemory=false, bool IsRemoteMemory=false)
Emit atomic update for constructs: X = X BinOp Expr ,or X = Expr BinOp X For complex Operations: X = ...
std::function< std::tuple< std::string, uint64_t >()> FileIdentifierInfoCallbackTy
bool isLastFinalizationInfoCancellable(omp::Directive DK)
Return true if the last entry in the finalization stack is of kind DK and cancellable.
LLVM_ABI InsertPointTy emitTargetKernel(const LocationDescription &Loc, InsertPointTy AllocaIP, Value *&Return, Value *Ident, Value *DeviceID, Value *NumTeams, Value *NumThreads, Value *HostPtr, ArrayRef< Value * > KernelArgs)
Generate a target region entry call.
LLVM_ABI GlobalVariable * createOffloadMaptypes(SmallVectorImpl< uint64_t > &Mappings, std::string VarName)
Create the global variable holding the offload mappings information.
LLVM_ABI ~OpenMPIRBuilder()
LLVM_ABI Expected< Function * > emitUserDefinedMapper(function_ref< MapInfosOrErrorTy(InsertPointTy CodeGenIP, llvm::Value *PtrPHI, llvm::Value *BeginArg)> PrivAndGenMapInfoCB, llvm::Type *ElemTy, StringRef FuncName, CustomMapperCallbackTy CustomMapperCB, bool PreserveMemberOfFlags=false, bool PropagatePresentToPointee=false)
Emit the user-defined mapper function.
LLVM_ABI CallInst * createCachedThreadPrivate(const LocationDescription &Loc, llvm::Value *Pointer, llvm::ConstantInt *Size, const llvm::Twine &Name=Twine(""))
Create a runtime call for kmpc_threadprivate_cached.
IRBuilder Builder
The LLVM-IR Builder used to create IR.
LLVM_ABI GlobalValue * createGlobalFlag(unsigned Value, StringRef Name)
Create a hidden global flag Name in the module with initial value Value.
LLVM_ABI void emitOffloadingArraysArgument(IRBuilderBase &Builder, OpenMPIRBuilder::TargetDataRTArgs &RTArgs, OpenMPIRBuilder::TargetDataInfo &Info, bool ForEndCall=false)
Emit the arguments to be passed to the runtime library based on the arrays of base pointers,...
LLVM_ABI InsertPointOrErrorTy createMasked(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, Value *Filter)
Generator for 'omp masked'.
LLVM_ABI Expected< CanonicalLoopInfo * > createCanonicalLoop(const LocationDescription &Loc, LoopBodyGenCallbackTy BodyGenCB, Value *TripCount, const Twine &Name="loop")
Generator for the control flow structure of an OpenMP canonical loop.
function_ref< Expected< InsertPointTy >( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, Value *DestPtr, Value *SrcPtr)> TaskDupCallbackTy
Callback type for task duplication function code generation.
LLVM_ABI Value * getSizeInBytes(Value *BasePtr)
Computes the size of type in bytes.
llvm::function_ref< llvm::Error( InsertPointTy BodyIP, llvm::Value *LinearIV)> IteratorBodyGenTy
LLVM_ABI FunctionCallee createDispatchDeinitFunction()
Returns __kmpc_dispatch_deinit runtime function.
LLVM_ABI void registerTargetGlobalVariable(OffloadEntriesInfoManager::OMPTargetGlobalVarEntryKind CaptureClause, OffloadEntriesInfoManager::OMPTargetDeviceClauseKind DeviceClause, bool IsDeclaration, bool IsExternallyVisible, TargetRegionEntryInfo EntryInfo, StringRef MangledName, std::vector< GlobalVariable * > &GeneratedRefs, bool OpenMPSIMD, std::vector< Triple > TargetTriple, std::function< Constant *()> GlobalInitializer, std::function< GlobalValue::LinkageTypes()> VariableLinkage, Type *LlvmPtrTy, Constant *Addr)
Registers a target variable for device or host.
LLVM_ABI void createTargetDeinit(const LocationDescription &Loc, int32_t TeamsReductionDataSize=0)
Create a runtime call for kmpc_target_deinit.
BodyGenTy
Type of BodyGen to use for region codegen.
LLVM_ABI CanonicalLoopInfo * fuseLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops)
Fuse a sequence of loops.
LLVM_ABI void emitX86DeclareSimdFunction(llvm::Function *Fn, unsigned NumElements, const llvm::APSInt &VLENVal, llvm::ArrayRef< DeclareSimdAttrTy > ParamAttrs, DeclareSimdBranch Branch)
Emit x86 vector-function ABI attributes for a declare simd function.
SmallVector< llvm::Function *, 16 > ConstantAllocaRaiseCandidates
A collection of candidate target functions that's constant allocas will attempt to be raised on a cal...
OffloadEntriesInfoManager OffloadInfoManager
Info manager to keep track of target regions.
static LLVM_ABI std::pair< int32_t, int32_t > readTeamBoundsForKernel(const Triple &T, Function &Kernel)
Read/write a bounds on teams for Kernel.
const std::string ompOffloadInfoName
OMP Offload Info Metadata name string.
Expected< InsertPointTy > InsertPointOrErrorTy
Type used to represent an insertion point or an error value.
LLVM_ABI InsertPointTy createCopyPrivate(const LocationDescription &Loc, llvm::Value *BufSize, llvm::Value *CpyBuf, llvm::Value *CpyFn, llvm::Value *DidIt)
Generator for __kmpc_copyprivate.
LLVM_ABI InsertPointOrErrorTy createSections(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< StorableBodyGenCallbackTy > SectionCBs, PrivatizeCallbackTy PrivCB, FinalizeCallbackTy FiniCB, bool IsCancellable, bool IsNowait)
Generator for 'omp sections'.
std::function< void(EmitMetadataErrorKind, TargetRegionEntryInfo)> EmitMetadataErrorReportFunctionTy
Callback function type.
function_ref< InsertPointOrErrorTy( Argument &Arg, Value *Input, Value *&RetVal, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< InsertPointTy > DeallocIPs)> TargetGenArgAccessorsCallbackTy
LLVM_ABI Expected< ScanInfo * > scanInfoInitialize()
Creates a ScanInfo object, allocates and returns the pointer.
LLVM_ABI InsertPointOrErrorTy emitTargetTask(TargetTaskBodyCallbackTy TaskBodyCB, Value *DeviceID, Value *RTLoc, OpenMPIRBuilder::InsertPointTy AllocaIP, const DependenciesInfo &Dependencies, const TargetDataRTArgs &RTArgs, bool HasNoWait)
Generate a target-task for the target construct.
LLVM_ABI InsertPointTy createAtomicRead(const LocationDescription &Loc, AtomicOpValue &X, AtomicOpValue &V, AtomicOrdering AO, InsertPointTy AllocaIP)
Emit atomic Read for : V = X — Only Scalar data types.
function_ref< Error(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks)> BodyGenCallbackTy
Callback type for body (=inner region) code generation.
bool updateToLocation(const LocationDescription &Loc)
Update the internal location to Loc.
LLVM_ABI void createFlush(const LocationDescription &Loc)
Generator for 'omp flush'.
LLVM_ABI Constant * getAddrOfDeclareTargetVar(OffloadEntriesInfoManager::OMPTargetGlobalVarEntryKind CaptureClause, OffloadEntriesInfoManager::OMPTargetDeviceClauseKind DeviceClause, bool IsDeclaration, bool IsExternallyVisible, TargetRegionEntryInfo EntryInfo, StringRef MangledName, std::vector< GlobalVariable * > &GeneratedRefs, bool OpenMPSIMD, std::vector< Triple > TargetTriple, Type *LlvmPtrTy, std::function< Constant *()> GlobalInitializer, std::function< GlobalValue::LinkageTypes()> VariableLinkage)
Retrieve (or create if non-existent) the address of a declare target variable, used in conjunction wi...
LLVM_ABI void createTaskwait(const LocationDescription &Loc, DependenciesInfo Dependencies={}, bool IsNowait=false)
Generator for 'omp taskwait'.
origPtr *with the address space normalization required by the runtime entry point *The NULL descriptor makes the runtime walk the enclosing taskgroups to *find the matching task_reduction registration for the item The lookups *are emitted at p Loc
EmitMetadataErrorKind
The kind of errors that can occur when emitting the offload entries and metadata.
@ EMIT_MD_DECLARE_TARGET_ERROR
@ EMIT_MD_GLOBAL_VAR_INDIRECT_ERROR
@ EMIT_MD_GLOBAL_VAR_LINK_ERROR
@ EMIT_MD_TARGET_REGION_ERROR
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
Pseudo-analysis pass that exposes the PassInstrumentation to pass managers.
Class to represent pointers.
static PointerType * getUnqual(LLVMContext &C)
This constructs an opaque pointer to an object in the default address space (address space zero).
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
PostDominatorTree Class - Concrete subclass of DominatorTree that is used to compute the post-dominat...
Analysis pass that exposes the ScalarEvolution for a function.
LLVM_ABI ScalarEvolution run(Function &F, FunctionAnalysisManager &AM)
The main scalar evolution driver.
ScanInfo holds the information to assist in lowering of Scan reduction.
llvm::SmallDenseMap< llvm::Value *, llvm::Value * > * ScanBuffPtrs
Maps the private reduction variable to the pointer of the temporary buffer.
llvm::BasicBlock * OMPScanLoopExit
Exit block of loop body.
llvm::Value * IV
Keeps track of value of iteration variable for input/scan loop to be used for Scan directive lowering...
llvm::BasicBlock * OMPAfterScanBlock
Dominates the body of the loop before scan directive.
llvm::BasicBlock * OMPScanInit
Block before loop body where scan initializations are done.
llvm::BasicBlock * OMPBeforeScanBlock
Dominates the body of the loop before scan directive.
llvm::BasicBlock * OMPScanFinish
Block after loop body where scan finalizations are done.
llvm::Value * Span
Stores the span of canonical loop being lowered to be used for temporary buffer allocation or Finaliz...
bool OMPFirstScanLoop
If true, it indicates Input phase is lowered; else it indicates ScanPhase is lowered.
llvm::BasicBlock * OMPScanDispatch
Controls the flow to before or after scan blocks.
A vector that has set insertion semantics.
bool remove_if(UnaryPredicate P)
Remove items from the set vector based on a predicate function.
bool empty() const
Determine if the SetVector is empty or not.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool all() const
Returns true if all bits are set.
bool any() const
Returns true if any bit is set.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
void append(StringRef RHS)
Append from a StringRef.
StringRef str() const
Explicit conversion to StringRef.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setAlignment(Align Align)
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this store instruction.
StringMap - This is an unconventional map that is specialized for handling keys that are "strings",...
ValueTy lookup(StringRef Key) const
lookup - Return the entry for the specified key, or a default constructed value if no such entry exis...
Represent a constant reference to a string, i.e.
std::string str() const
Get the contents as an std::string.
constexpr bool empty() const
Check if the string is empty.
constexpr size_t size() const
Get the string size.
size_t count(char C) const
Return the number of occurrences of C in the string.
bool ends_with(StringRef Suffix) const
Check if this string ends with the given Suffix.
StringRef drop_back(size_t N=1) const
Return a StringRef equal to 'this' but with the last N elements dropped.
Class to represent struct types.
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
static LLVM_ABI StructType * create(LLVMContext &Context, StringRef Name)
This creates an identified struct.
Type * getElementType(unsigned N) const
LLVM_ABI void addCase(ConstantInt *OnVal, BasicBlock *Dest)
Add an entry to the switch instruction.
Analysis pass providing the TargetTransformInfo.
LLVM_ABI Result run(const Function &F, FunctionAnalysisManager &)
TargetTransformInfo Result
Analysis pass providing the TargetLibraryInfo.
Target - Wrapper for Target specific information.
TargetMachine * createTargetMachine(const Triple &TT, StringRef CPU, StringRef Features, const TargetOptions &Options, std::optional< Reloc::Model > RM, std::optional< CodeModel::Model > CM=std::nullopt, CodeGenOptLevel OL=CodeGenOptLevel::Default, bool JIT=false) const
createTargetMachine - Create a target specific machine implementation for the specified Triple.
Triple - Helper class for working with autoconf configuration names.
bool isPPC() const
Tests whether the target is PowerPC (32- or 64-bit LE or BE).
bool isX86() const
Tests whether the target is x86 (32- or 64-bit).
bool isWasm() const
Tests whether the target is wasm (32- and 64-bit).
bool isSystemZ() const
Tests whether the target is SystemZ.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
LLVM_ABI unsigned getIntegerBitWidth() const
LLVM_ABI Type * getStructElementType(unsigned N) const
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
bool isStructTy() const
True if this is an instance of StructType.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
bool isVoidTy() const
Return true if this is 'void'.
Unconditional Branch instruction.
static UncondBrInst * Create(BasicBlock *Target, InsertPosition InsertBefore=nullptr)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
This function has undefined behavior.
Produce an estimate of the unrolled cost of the specified loop.
LLVM_ABI bool canUnroll(OptimizationRemarkEmitter *ORE=nullptr, const Loop *L=nullptr) const
Whether it is legal to unroll this loop.
uint64_t getRolledLoopSize() const
A Use represents the edge between a Value definition and its users.
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
iterator_range< user_iterator > users()
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI bool hasNUses(unsigned N) const
Return true if this Value has exactly N uses.
LLVM_ABI User * getUniqueUndroppableUser()
Return true if there is exactly one unique user of this value that cannot be dropped (that user can h...
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
LLVM_ABI bool replaceUsesWithIf(Value *New, llvm::function_ref< bool(Use &U)> ShouldReplace)
Go through the uses list for this definition and make each use point to "V" if the callback ShouldRep...
iterator_range< use_iterator > uses()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
An efficient, type-erasing, non-owning reference to a callable.
const ParentTy * getParent() const
self_iterator getIterator()
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
A raw_ostream that writes to an SmallVector or SmallString.
The virtual file system interface.
llvm::ErrorOr< std::unique_ptr< llvm::MemoryBuffer > > getBufferForFile(const Twine &Name, int64_t FileSize=-1, bool RequiresNullTerminator=true, bool IsVolatile=false, bool IsText=true)
This is a convenience method that opens a file, gets its content and then closes the file.
virtual llvm::ErrorOr< Status > status(const Twine &Path)=0
Get the status of the entry at Path, if one exists.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ PTX_Kernel
Call to a PTX kernel. Passes all arguments in parameter space.
@ BasicBlock
Various leaf nodes.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
Flag
These should be considered private to the implementation of the MCInstrDesc class.
constexpr StringLiteral MaxNTID("nvvm.maxntid")
constexpr StringLiteral MaxClusterRank("nvvm.maxclusterrank")
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
LLVM_ABI GlobalVariable * emitOffloadingEntry(Module &M, object::OffloadKind Kind, Constant *Addr, StringRef Name, uint64_t Size, uint32_t Flags, uint64_t Data, Constant *AuxAddr=nullptr)
OpenMPOffloadMappingFlags
Values for bit flags used to specify the mapping type for offloading.
@ OMP_MAP_PTR_AND_OBJ
The element being mapped is a pointer-pointee pair; both the pointer and the pointee should be mapped...
@ OMP_MAP_MEMBER_OF
The 16 MSBs of the flags indicate whether the entry is member of some struct/class.
IdentFlag
IDs for all omp runtime library ident_t flag encodings (see their defintion in openmp/runtime/src/kmp...
RuntimeFunction
IDs for all omp runtime library (RTL) functions.
constexpr const GV & getAMDGPUGridValues()
static constexpr GV SPIRVGridValues
For generic SPIR-V GPUs.
OMPDynGroupprivateFallbackType
The fallback types for the dyn_groupprivate clause.
static constexpr GV NVPTXGridValues
For Nvidia GPUs.
@ OMP_TGT_EXEC_MODE_SPMD_NO_LOOP
@ OMP_TGT_EXEC_MODE_GENERIC
Function * Kernel
Summary of a kernel (=entry point for target offloading).
WorksharingLoopType
A type of worksharing loop construct.
OMPAtomicCompareOp
Atomic compare operations. Currently OpenMP only supports ==, >, and <.
EnumSet< Property > Properties
NodeAddr< PhiNode * > Phi
friend class Instruction
Iterator for Instructions in a `BasicBlock.
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
LLVM_ABI BasicBlock * splitBBWithSuffix(IRBuilderBase &Builder, bool CreateBranch, llvm::Twine Suffix=".split")
Like splitBB, but reuses the current block's name for the new name.
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
LLVM_ABI unsigned computeUnrollCount(Loop *L, const TargetTransformInfo &TTI, DominatorTree &DT, LoopInfo *LI, AssumptionCache *AC, ScalarEvolution &SE, const SmallPtrSetImpl< const Value * > &EphValues, OptimizationRemarkEmitter *ORE, unsigned TripCount, unsigned MaxTripCount, bool MaxOrZero, unsigned TripMultiple, const UnrollCostEstimator &UCE, TargetTransformInfo::UnrollingPreferences &UP, TargetTransformInfo::PeelingPreferences &PP)
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
hash_code hash_value(const FixedPointSemantics &Val)
LLVM_ABI Expected< std::unique_ptr< Module > > parseBitcodeFile(MemoryBufferRef Buffer, LLVMContext &Context, ParserCallbacks Callbacks={})
Read the specified bitcode file, returning the module.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
@ LLVM_MARK_AS_BITMASK_ENUM
LLVM_ABI BasicBlock * CloneBasicBlock(const BasicBlock *BB, ValueToValueMapTy &VMap, const Twine &NameSuffix="", Function *F=nullptr, ClonedCodeInfo *CodeInfo=nullptr, bool MapAtoms=true)
Return a copy of the specified basic block, but without embedding the block into a particular functio...
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
unsigned getPointerAddressSpace(const Type *T)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
auto successors(const MachineBasicBlock *BB)
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
LLVM_ABI std::error_code inconvertibleErrorCode()
The value returned by this function can be returned from convertToErrorCode for Error values where no...
testing::Matcher< const detail::ErrorHolder & > Failed()
constexpr from_range_t from_range
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE()
LLVM_ABI BasicBlock * splitBB(IRBuilderBase::InsertPoint IP, bool CreateBranch, DebugLoc DL, llvm::Twine Name={})
Split a BasicBlock at an InsertPoint, even if the block is degenerate (missing the terminator).
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
LLVM_ABI TargetTransformInfo::UnrollingPreferences gatherUnrollingPreferences(Loop *L, ScalarEvolution &SE, const TargetTransformInfo &TTI, BlockFrequencyInfo *BFI, ProfileSummaryInfo *PSI, llvm::OptimizationRemarkEmitter &ORE, int OptLevel, std::optional< unsigned > UserThreshold, std::optional< bool > UserAllowPartial, std::optional< bool > UserRuntime, std::optional< bool > UserUpperBound, std::optional< unsigned > UserFullUnrollMaxCount)
Gather the various unrolling parameters based on the defaults, compiler flags, TTI overrides and user...
std::string utostr(uint64_t X, bool isNeg=false)
ErrorOr< T > expectedToErrorOrAndEmitErrors(LLVMContext &Ctx, Expected< T > Val)
bool isa_and_nonnull(const Y &Val)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
auto dyn_cast_or_null(const Y &Val)
LLVM_ABI bool convertUsersOfConstantsToInstructions(ArrayRef< Constant * > Consts, Function *RestrictToFunc=nullptr, bool RemoveDeadConstants=true, bool IncludeSelf=false)
Replace constant expressions users of the given constants with instructions.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
auto reverse(ContainerTy &&C)
LLVM_ABI TargetTransformInfo::PeelingPreferences gatherPeelingPreferences(Loop *L, ScalarEvolution &SE, const TargetTransformInfo &TTI, std::optional< bool > UserAllowPeeling, std::optional< bool > UserAllowProfileBasedPeeling, bool UnrollingSpecficValues=false)
LLVM_ABI void SplitBlockAndInsertIfThenElse(Value *Cond, BasicBlock::iterator SplitBefore, Instruction **ThenTerm, Instruction **ElseTerm, MDNode *BranchWeights=nullptr, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr)
SplitBlockAndInsertIfThenElse is similar to SplitBlockAndInsertIfThen, but also creates the ElseBlock...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
CodeGenOptLevel
Code generation optimization level.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
format_object< Ts... > format(const char *Fmt, const Ts &... Vals)
These are helper functions used to produce formatted output.
Error make_error(ArgTs &&... Args)
Make a Error instance representing failure using the given error info type.
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
void cantFail(Error Err, const char *Msg=nullptr)
Report a fatal error if Err is a failure value.
LLVM_ABI bool MergeBlockIntoPredecessor(BasicBlock *BB, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, MemoryDependenceResults *MemDep=nullptr, bool PredecessorWithTwoSuccessors=false, DominatorTree *DT=nullptr)
Attempts to merge a block into its predecessor, if possible.
@ Mul
Product of integers.
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
DWARFExpression::Operation Op
LLVM_ABI void remapInstructionsInBlocks(ArrayRef< BasicBlock * > Blocks, ValueToValueMapTy &VMap)
Remaps instructions in Blocks using the mapping in VMap.
ArrayRef(const T &OneElt) -> ArrayRef< T >
OutputIt copy(R &&Range, OutputIt Out)
constexpr unsigned BitWidth
ValueMap< const Value *, WeakTrackingVH > ValueToValueMapTy
LLVM_ABI void spliceBB(IRBuilderBase::InsertPoint IP, BasicBlock *New, bool CreateBranch, DebugLoc DL)
Move the instruction after an InsertPoint to the beginning of another BasicBlock.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
auto predecessors(const MachineBasicBlock *BB)
auto filter_to_vector(ContainerTy &&C, PredicateFn &&Pred)
Filter a range to a SmallVector with the element types deduced.
PointerUnion< const Value *, const PseudoSourceValue * > ValueType
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
LLVM_ABI Constant * ConstantFoldInsertValueInstruction(Constant *Agg, Constant *Val, ArrayRef< unsigned > Idxs)
Attempt to constant fold an insertvalue instruction with the specified operands and indices.
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI void DeleteDeadBlocks(ArrayRef< BasicBlock * > BBs, DomTreeUpdater *DTU=nullptr, bool KeepOneInputPHIs=false)
Delete the specified blocks from BB.
bool to_integer(StringRef S, N &Num, unsigned Base=0)
Convert the string S to an integer of the specified type using the radix Base. If Base is 0,...
static auto filterDbgVars(iterator_range< simple_ilist< DbgRecord >::iterator > R)
Filter the DbgRecord range to DbgVariableRecord types only and downcast.
This struct is a compact representation of a valid (non-zero power of two) alignment.
static LLVM_ABI void collectEphemeralValues(const Loop *L, AssumptionCache *AC, SmallPtrSetImpl< const Value * > &EphValues)
Collect a loop's ephemeral values (those used only by an assume or similar intrinsics in the loop).
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
A struct to pack the relevant information for an OpenMP affinity clause.
a struct to pack relevant information while generating atomic Ops
A struct to pack the relevant information for an OpenMP depend clause.
omp::RTLDependenceKindTy DepKind
A struct to pack static and dynamic dependency information for a task.
SmallVector< DependData > Deps
LLVM_ABI Error mergeFiniBB(IRBuilderBase &Builder, BasicBlock *ExistingFiniBB)
For cases where there is an unavoidable existing finalization block (e.g.
LLVM_ABI Expected< BasicBlock * > getFiniBB(IRBuilderBase &Builder)
The basic block to which control should be transferred to implement the FiniCB.
Description of a LLVM-IR insertion point (IP) and a debug/source location (filename,...
MapNonContiguousArrayTy Offsets
MapNonContiguousArrayTy Counts
MapNonContiguousArrayTy Strides
This structure contains combined information generated for mappable clauses, including base pointers,...
MapDeviceInfoArrayTy DevicePointers
MapValuesArrayTy BasePointers
MapValuesArrayTy Pointers
StructNonContiguousInfo NonContigInfo
Helper that contains information about regions we need to outline during finalization.
void collectBlocks(SmallPtrSetImpl< BasicBlock * > &BlockSet, SmallVectorImpl< BasicBlock * > &BlockVector)
Collect all blocks in between EntryBB and ExitBB in both the given vector and set.
BasicBlock * OuterAllocBB
virtual std::unique_ptr< CodeExtractor > createCodeExtractor(ArrayRef< BasicBlock * > Blocks, bool ArgsInZeroAddressSpace, Twine Suffix=Twine(""))
Create a CodeExtractor instance based on the information stored in this structure,...
Information about an OpenMP reduction.
EvalKind EvaluationKind
Reduction evaluation kind - scalar, complex or aggregate.
ReductionGenAtomicCBTy AtomicReductionGen
Callback for generating the atomic reduction body, may be null.
ReductionGenCBTy ReductionGen
Callback for generating the reduction body.
Value * Variable
Reduction variable of pointer type.
Value * PrivateVariable
Thread-private partial reduction variable.
ReductionGenClangCBTy ReductionGenClang
Clang callback for generating the reduction body.
Type * ElementType
Reduction element type, must match pointee type of variable.
ReductionGenDataPtrPtrCBTy DataPtrPtrGen
Container for the arguments used to pass data to the runtime library.
Value * SizesArray
The array of sizes passed to the runtime library.
Value * PointersArray
The array of section pointers passed to the runtime library.
Value * MappersArray
The array of user-defined mappers passed to the runtime library.
Value * MapTypesArrayEnd
The array of map types passed to the runtime library for the end of the region, or nullptr if there a...
Value * BasePointersArray
The array of base pointer passed to the runtime library.
Value * MapTypesArray
The array of map types passed to the runtime library for the beginning of the region or for the entir...
Value * MapNamesArray
The array of original declaration names of mapped pointers sent to the runtime library for debugging.
Data structure that contains the needed information to construct the kernel args vector.
bool StrictBlocks
True if the kernel strictly requires the number of blocks and threads above to run.
ArrayRef< Value * > NumThreads
The number of threads.
TargetDataRTArgs RTArgs
Arguments passed to the runtime library.
Value * NumIterations
The number of iterations.
Value * DynCGroupMem
The size of the dynamic shared memory.
unsigned NumTargetItems
Number of arguments passed to the runtime library.
bool HasNoWait
True if the kernel has 'no wait' clause.
ArrayRef< Value * > NumTeams
The number of teams.
omp::OMPDynGroupprivateFallbackType DynCGroupMemFallback
The fallback mechanism for the shared memory.
Container to pass the default attributes with which a kernel must be launched, used to set kernel att...
omp::OMPTgtExecModeFlags ExecFlags
SmallVector< int32_t, 3 > MaxTeams
Container to pass LLVM IR runtime values or constants related to the number of teams and threads with...
Value * DeviceID
Device ID value used in the kernel launch.
SmallVector< Value *, 3 > MaxTeams
Value * LoopTripCount
Total number of iterations of the SPMD or Generic-SPMD kernel or null if it is a generic kernel.
SmallVector< Value *, 3 > TargetThreadLimit
SmallVector< Value *, 3 > TeamsThreadLimit
SmallVector< Value * > MaxThreads
'parallel' construct 'num_threads' clause value, if present and it is an SPMD kernel.
Data structure to contain the information needed to uniquely identify a target entry.
static LLVM_ABI void getTargetRegionEntryFnName(SmallVectorImpl< char > &Name, StringRef ParentName, unsigned DeviceID, unsigned FileID, unsigned Line, unsigned Count)
static constexpr const char * KernelNamePrefix
The prefix used for kernel names.
static LLVM_ABI const Target * lookupTarget(const Triple &TheTriple, std::string &Error)
lookupTarget - Lookup a target based on a target triple.
Defines various target-specific GPU grid values that must be consistent between host RTL (plugin),...